From 039411d839bcb0f42af2bd84857c6796746a81e1 Mon Sep 17 00:00:00 2001 From: Mikhail Koviazin Date: Wed, 26 Aug 2026 12:54:20 +0200 Subject: [PATCH 01/15] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/2159 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS [experimental] CAS (Content-addressed storage) over shared object storage for antalya-26.6 # Conflicts: # .github/workflows/fast_builds.yml # .github/workflows/master.yml # .github/workflows/pull_request.yml # .github/workflows/pull_request_community.yml # ci/jobs/functional_tests.py # ci/jobs/scripts/clickhouse_proc.py # ci/jobs/scripts/clickhouse_service.py # ci/workflows/pull_request.py # ci/workflows/pull_request_community.py # ci/workflows/release_branches.py # docs/concepts/features/configuration/server-config/storing-data.mdx # programs/disks/DisksApp.cpp # programs/disks/ICommand.h # programs/server/Server.cpp # src/Common/FailPoint.cpp # src/Common/ThreadStatus.h # src/Core/ServerSettings.cpp # src/Disks/DiskObjectStorage/DiskObjectStorage.cpp # src/Disks/DiskObjectStorage/DiskObjectStorage.h # src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.cpp # src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h # src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp # src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h # src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp # src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp # src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp # src/Disks/IDiskTransaction.h # src/IO/ReadPipeline.cpp # src/IO/ReadPipeline.h # src/IO/S3/copyS3File.cpp # src/IO/S3/copyS3File.h # src/IO/S3Common.cpp # src/IO/WriteBufferFromS3.cpp # src/IO/tests/gtest_writebuffer_s3.cpp # src/Interpreters/InterpreterSystemQuery.cpp # src/Interpreters/ServerAsynchronousMetrics.cpp # src/Parsers/ASTSystemQuery.h # src/Parsers/tests/gtest_Parser.cpp # src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp # src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp # src/Storages/MergeTree/DataPartsExchange.cpp # src/Storages/MergeTree/IMergeTreeDataPart.cpp # src/Storages/MergeTree/MergeTask.cpp # src/Storages/MergeTree/MergeTreeData.cpp # tests/integration/test_replicated_database/test.py # tests/queries/0_stateless/02253_empty_part_checksums.sh # tests/queries/0_stateless/02254_projection_broken_part.sh # tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh # tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh # tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh # tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh # tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh # tests/queries/0_stateless/04327_reader_executor_metrics.sql # tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql --- .github/workflows/fast_builds.yml | 160 + .github/workflows/master.yml | 524 ++ .github/workflows/pull_request.yml | 1765 +++++- .github/workflows/pull_request_community.yml | 552 ++ ci/defs/altinity_jobs.py | 50 + ci/defs/job_configs.py | 2 + ci/jobs/functional_tests.py | 37 +- ci/jobs/scripts/clickhouse_proc.py | 107 + ci/jobs/scripts/clickhouse_service.py | 62 + ci/workflows/backport_branches.py | 7 +- ci/workflows/fast_builds.py | 7 +- ci/workflows/master.py | 7 +- ci/workflows/pull_request.py | 18 +- ci/workflows/pull_request_community.py | 8 + ci/workflows/release_branches.py | 12 +- .../server-config/storing-data.mdx | 108 + docs/en/antalya/cas/architecture/backend.md | 138 + .../antalya/cas/architecture/blob-protocol.md | 244 + .../antalya/cas/architecture/correctness.md | 92 + .../cas/architecture/design-history.md | 63 + .../cas/architecture/garbage-collection.md | 263 + docs/en/antalya/cas/architecture/index.md | 114 + .../cas/architecture/manifests-and-refs.md | 292 + .../cas/architecture/mounts-and-leases.md | 252 + .../en/antalya/cas/architecture/namespaces.md | 172 + .../cas/architecture/part-lifecycle.md | 153 + docs/en/antalya/cas/architecture/read-path.md | 84 + .../antalya/cas/architecture/replication.md | 119 + .../cas/architecture/storage-layout.md | 161 + docs/en/antalya/cas/bucket-requirements.md | 73 + docs/en/antalya/cas/configuration.md | 159 + docs/en/antalya/cas/index.md | 89 + docs/en/antalya/cas/operations/debugging.md | 268 + docs/en/antalya/cas/operations/migration.md | 209 + docs/en/antalya/cas/operations/monitoring.md | 150 + .../antalya/cas/operations/troubleshooting.md | 66 + docs/en/antalya/cas/quick-start.md | 144 + docs/en/antalya/cas/roadmap.md | 116 + .../en/operations/system-tables/cas_gc_log.md | 156 + docs/en/operations/system-tables/cas_log.md | 65 + .../en/operations/system-tables/cas_mounts.md | 81 + docs/reference/statements/system.mdx | 145 + programs/disks/CMakeLists.txt | 5 + programs/disks/CommandCaDropMember.cpp | 75 + programs/disks/CommandCaGcDryRun.cpp | 56 + programs/disks/CommandCaGcRebuild.cpp | 85 + programs/disks/CommandCaInspect.cpp | 77 + programs/disks/CommandFsck.cpp | 188 + programs/disks/DisksApp.cpp | 38 + programs/disks/DisksApp.h | 4 + programs/disks/ICommand.h | 8 + programs/local/LocalServer.cpp | 12 + programs/server/Server.cpp | 9 + programs/server/config.xml | 29 + src/Access/Common/AccessType.h | 7 + src/CMakeLists.txt | 11 + src/Common/CurrentMetrics.cpp | 8 + src/Common/FailPoint.cpp | 6 + src/Common/ProfileEvents.cpp | 165 + src/Common/SystemLogBase.cpp | 2 + src/Common/SystemLogBase.h | 2 + src/Common/ThreadStatus.h | 12 + src/Common/setThreadName.h | 6 + src/Core/ServerSettings.cpp | 10 + .../DiskObjectStorage/DiskObjectStorage.cpp | 57 + .../DiskObjectStorage/DiskObjectStorage.h | 14 + .../DiskObjectStorageCache.cpp | 14 +- .../DiskObjectStorageTransaction.cpp | 154 +- .../DiskObjectStorageTransaction.h | 20 + .../MetadataStorageFromCacheObjectStorage.cpp | 7 + .../MetadataStorageFromCacheObjectStorage.h | 1 + .../ContentAddressed/Backend/CasBackend.h | 427 ++ .../Backend/CasInMemoryBackend.cpp | 382 ++ .../Backend/CasInMemoryBackend.h | 157 + .../Backend/CasInstrumentedBackend.cpp | 143 + .../Backend/CasInstrumentedBackend.h | 180 + .../Backend/CasObjectStorageBackend.cpp | 1179 ++++ .../Backend/CasObjectStorageBackend.h | 282 + .../ContentAddressed/Backend/CasProbe.cpp | 261 + .../ContentAddressed/Backend/CasProbe.h | 33 + .../Backend/CasRequestControl.cpp | 912 +++ .../Backend/CasRequestControl.h | 635 ++ .../Backend/CasSentinelProbe.cpp | 110 + .../Backend/CasSentinelProbe.h | 55 + .../ContentAddressedExchange.cpp | 170 + .../ContentAddressedExchange.h | 260 + .../ContentAddressedMetadataStorage.cpp | 2374 ++++++++ .../ContentAddressedMetadataStorage.h | 798 +++ .../ContentAddressedSettings.cpp | 267 + .../ContentAddressedSettings.h | 94 + .../ContentAddressedTransaction.cpp | 2002 ++++++ .../ContentAddressedTransaction.h | 420 ++ .../Formats/CasBlobEnvelopeFormat.cpp | 253 + .../Formats/CasBlobEnvelopeFormat.h | 110 + .../Formats/CasBlobMetaFormat.cpp | 94 + .../Formats/CasBlobMetaFormat.h | 46 + .../ContentAddressed/Formats/CasByteBudget.h | 57 + .../Formats/CasFoldSealFormat.cpp | 546 ++ .../Formats/CasFoldSealFormat.h | 225 + .../ContentAddressed/Formats/CasFormat.cpp | 205 + .../ContentAddressed/Formats/CasFormat.h | 210 + .../Formats/CasGcMaintenanceStateFormat.cpp | 67 + .../Formats/CasGcMaintenanceStateFormat.h | 21 + .../Formats/CasGcOutcomesFormat.cpp | 129 + .../Formats/CasGcOutcomesFormat.h | 58 + .../Formats/CasGcStateFormat.cpp | 118 + .../Formats/CasGcStateFormat.h | 71 + .../ContentAddressed/Formats/CasLayout.cpp | 345 ++ .../ContentAddressed/Formats/CasLayout.h | 480 ++ .../Formats/CasPartManifestFormat.cpp | 361 ++ .../Formats/CasPartManifestFormat.h | 122 + .../Formats/CasPoolMetaFormat.cpp | 183 + .../Formats/CasPoolMetaFormat.h | 91 + .../Formats/CasRecordStreamFormat.cpp | 325 + .../Formats/CasRecordStreamFormat.h | 164 + .../Formats/CasRefCatalogFormat.cpp | 397 ++ .../Formats/CasRefCatalogFormat.h | 168 + .../Formats/CasRefCkptFormat.cpp | 179 + .../Formats/CasRefCkptFormat.h | 124 + .../Formats/CasRefLogFormat.cpp | 443 ++ .../Formats/CasRefLogFormat.h | 177 + .../Formats/CasRefSnapshotFormat.cpp | 305 + .../Formats/CasRefSnapshotFormat.h | 95 + .../Formats/CasRefWireVocab.cpp | 47 + .../Formats/CasRefWireVocab.h | 65 + .../Formats/CasServerRootFormats.cpp | 185 + .../Formats/CasServerRootFormats.h | 97 + .../Formats/CasTextFormat.cpp | 414 ++ .../ContentAddressed/Formats/CasTextFormat.h | 241 + .../ContentAddressed/Formats/CasWireVocab.cpp | 103 + .../ContentAddressed/Formats/CasWireVocab.h | 62 + .../ContentAddressed/Formats/README.md | 79 + .../ContentAddressed/Gc/CasBlobInDegree.cpp | 724 +++ .../ContentAddressed/Gc/CasBlobInDegree.h | 413 ++ .../ContentAddressed/Gc/CasGc.cpp | 4527 ++++++++++++++ .../ContentAddressed/Gc/CasGc.h | 978 +++ .../Gc/CasGcMaintenanceState.cpp | 40 + .../Gc/CasGcMaintenanceState.h | 29 + .../ContentAddressed/Gc/CasGcMetaWriter.cpp | 223 + .../ContentAddressed/Gc/CasGcMetaWriter.h | 87 + .../ContentAddressed/Gc/CasGcPhaseTimer.h | 86 + .../ContentAddressed/Gc/CasGcScheduler.cpp | 419 ++ .../ContentAddressed/Gc/CasGcScheduler.h | 227 + .../ContentAddressed/Gc/CasGcShardPlan.cpp | 65 + .../ContentAddressed/Gc/CasGcShardPlan.h | 138 + .../Gc/CasNamespaceJanitor.cpp | 143 + .../ContentAddressed/Gc/CasNamespaceJanitor.h | 34 + .../Gc/CasOrphanManifestSweep.cpp | 934 +++ .../Gc/CasOrphanManifestSweep.h | 209 + .../Gc/CatalogLifecycleReconciler.cpp | 121 + .../Gc/CatalogLifecycleReconciler.h | 64 + .../Parts/PartFolderAccess.cpp | 762 +++ .../ContentAddressed/Parts/PartFolderAccess.h | 411 ++ .../ContentAddressed/Parts/PartPathParser.cpp | 400 ++ .../ContentAddressed/Parts/PartPathParser.h | 146 + .../ContentAddressed/Pool/CasBlobMeta.cpp | 46 + .../ContentAddressed/Pool/CasBlobMeta.h | 65 + .../Pool/CasBlobUploadPool.cpp | 75 + .../ContentAddressed/Pool/CasBlobUploadPool.h | 43 + .../ContentAddressed/Pool/CasDetachedWork.cpp | 52 + .../ContentAddressed/Pool/CasDetachedWork.h | 65 + .../Pool/CasEventDispatcher.cpp | 56 + .../Pool/CasEventDispatcher.h | 64 + .../Pool/CasManifestReader.cpp | 170 + .../ContentAddressed/Pool/CasManifestReader.h | 101 + .../ContentAddressed/Pool/CasMountRuntime.cpp | 1154 ++++ .../ContentAddressed/Pool/CasMountRuntime.h | 513 ++ .../ContentAddressed/Pool/CasPartWriteTxn.cpp | 1169 ++++ .../ContentAddressed/Pool/CasPartWriteTxn.h | 390 ++ .../ContentAddressed/Pool/CasPlainObjects.cpp | 154 + .../ContentAddressed/Pool/CasPlainObjects.h | 111 + .../ContentAddressed/Pool/CasPool.cpp | 2064 +++++++ .../ContentAddressed/Pool/CasPool.h | 1182 ++++ .../ContentAddressed/Pool/CasPoolMeta.cpp | 168 + .../ContentAddressed/Pool/CasRefCatalog.cpp | 654 ++ .../ContentAddressed/Pool/CasRefCatalog.h | 351 ++ .../ContentAddressed/Pool/CasRefCkpt.cpp | 374 ++ .../ContentAddressed/Pool/CasRefCkpt.h | 175 + .../Pool/CasRefCowManifestSet.cpp | 119 + .../Pool/CasRefCowManifestSet.h | 127 + .../ContentAddressed/Pool/CasRefCowMap.cpp | 245 + .../ContentAddressed/Pool/CasRefCowMap.h | 209 + .../ContentAddressed/Pool/CasRefLedger.cpp | 5232 ++++++++++++++++ .../ContentAddressed/Pool/CasRefLedger.h | 1269 ++++ .../ContentAddressed/Pool/CasRefProtocol.cpp | 1115 ++++ .../ContentAddressed/Pool/CasRefProtocol.h | 797 +++ .../ContentAddressed/Pool/CasServerRoot.cpp | 1906 ++++++ .../ContentAddressed/Pool/CasServerRoot.h | 567 ++ .../Primitives/CasBlobDigest.cpp | 47 + .../Primitives/CasBlobDigest.h | 247 + .../Primitives/CasBlobHashingWriteBuffer.cpp | 260 + .../Primitives/CasBlobHashingWriteBuffer.h | 50 + .../Primitives/CasCodecUtil.h | 118 + .../ContentAddressed/Primitives/CasEvent.cpp | 95 + .../ContentAddressed/Primitives/CasEvent.h | 126 + .../Primitives/CasNamespaceLifeId.h | 116 + .../ContentAddressed/Primitives/CasTypes.h | 334 + .../Primitives/CasXxh3Streamer.h | 83 + .../ContentAddressed/README.md | 205 + .../Tools/CasDecommission.cpp | 476 ++ .../ContentAddressed/Tools/CasDecommission.h | 55 + .../ContentAddressed/Tools/CasFsck.cpp | 1180 ++++ .../ContentAddressed/Tools/CasFsck.h | 285 + .../ContentAddressed/Tools/CasInspect.cpp | 639 ++ .../ContentAddressed/Tools/CasInspect.h | 30 + .../benchmarks/CMakeLists.txt | 4 + .../benchmarks/benchmark_cas_ref_protocol.cpp | 553 ++ .../MetadataStorages/IMetadataStorage.h | 53 + .../MetadataStorageFactory.cpp | 42 + .../ObjectStorages/IObjectStorage.h | 61 + .../Local/LocalObjectStorage.cpp | 23 + .../ObjectStorages/S3/S3ObjectStorage.cpp | 218 +- .../ObjectStorages/S3/S3ObjectStorage.h | 44 + .../ObjectStorages/S3/diskSettings.cpp | 2 + .../RegisterDiskObjectStorage.cpp | 25 + src/Disks/DiskType.cpp | 2 + src/Disks/DiskType.h | 1 + src/Disks/IDisk.h | 7 + src/Disks/IDiskTransaction.h | 19 + src/Disks/ReadOnlyDiskWrapper.h | 5 + src/Disks/tests/cas_format_test_battery.h | 111 + src/Disks/tests/cas_sweep_test_support.h | 36 + src/Disks/tests/cas_test_helpers.h | 2328 +++++++ src/Disks/tests/gtest_ca_transaction.cpp | 754 +++ src/Disks/tests/gtest_ca_wiring.cpp | 3001 +++++++++ src/Disks/tests/gtest_cas_b140_dangle.cpp | 127 + src/Disks/tests/gtest_cas_backend.cpp | 1679 +++++ .../tests/gtest_cas_backend_contract.cpp | 160 + .../tests/gtest_cas_backend_generation.cpp | 746 +++ src/Disks/tests/gtest_cas_backend_listing.cpp | 43 + src/Disks/tests/gtest_cas_blob_digest.cpp | 266 + .../tests/gtest_cas_blob_envelope_format.cpp | 174 + src/Disks/tests/gtest_cas_blob_hasher.cpp | 182 + src/Disks/tests/gtest_cas_blob_indegree.cpp | 1000 +++ src/Disks/tests/gtest_cas_blob_meta.cpp | 186 + .../tests/gtest_cas_blob_meta_format.cpp | 49 + src/Disks/tests/gtest_cas_blob_ref.cpp | 34 + .../tests/gtest_cas_blob_upload_pool.cpp | 144 + .../tests/gtest_cas_blob_upload_pool_env.cpp | 48 + .../tests/gtest_cas_bootstrap_ordering.cpp | 442 ++ .../tests/gtest_cas_confirm_exact_ref.cpp | 1015 +++ src/Disks/tests/gtest_cas_decommission.cpp | 1365 +++++ .../gtest_cas_decommission_catalog_duties.cpp | 330 + src/Disks/tests/gtest_cas_detached_work.cpp | 825 +++ src/Disks/tests/gtest_cas_empty_proof.cpp | 281 + src/Disks/tests/gtest_cas_encoding_pins.cpp | 110 + src/Disks/tests/gtest_cas_envelope.cpp | 55 + .../tests/gtest_cas_event_dispatcher.cpp | 194 + src/Disks/tests/gtest_cas_event_log.cpp | 672 ++ .../tests/gtest_cas_fence_generation.cpp | 447 ++ src/Disks/tests/gtest_cas_fold_seal_codec.cpp | 41 + .../tests/gtest_cas_fold_seal_format.cpp | 327 + src/Disks/tests/gtest_cas_forget.cpp | 599 ++ src/Disks/tests/gtest_cas_format.cpp | 126 + src/Disks/tests/gtest_cas_format_battery.cpp | 23 + src/Disks/tests/gtest_cas_fsck.cpp | 1584 +++++ src/Disks/tests/gtest_cas_gc_ack_floor.cpp | 1289 ++++ .../tests/gtest_cas_gc_arithmetic_intake.cpp | 595 ++ src/Disks/tests/gtest_cas_gc_attempt.cpp | 189 + src/Disks/tests/gtest_cas_gc_bounded_walk.cpp | 570 ++ src/Disks/tests/gtest_cas_gc_fold.cpp | 624 ++ .../tests/gtest_cas_gc_frontier_gate.cpp | 3488 +++++++++++ src/Disks/tests/gtest_cas_gc_hold_grammar.cpp | 1590 +++++ src/Disks/tests/gtest_cas_gc_leak.cpp | 525 ++ src/Disks/tests/gtest_cas_gc_log.cpp | 489 ++ .../gtest_cas_gc_maintenance_state_format.cpp | 202 + src/Disks/tests/gtest_cas_gc_meta_writer.cpp | 188 + .../tests/gtest_cas_gc_outcomes_format.cpp | 101 + src/Disks/tests/gtest_cas_gc_rebuild.cpp | 681 +++ src/Disks/tests/gtest_cas_gc_resume.cpp | 184 + src/Disks/tests/gtest_cas_gc_round.cpp | 1999 ++++++ src/Disks/tests/gtest_cas_gc_round_defer.cpp | 640 ++ .../tests/gtest_cas_gc_shard_incarnation.cpp | 534 ++ src/Disks/tests/gtest_cas_gc_shard_plan.cpp | 637 ++ src/Disks/tests/gtest_cas_gc_source_edge.cpp | 84 + src/Disks/tests/gtest_cas_gc_state_format.cpp | 179 + src/Disks/tests/gtest_cas_gc_stop_start.cpp | 492 ++ .../tests/gtest_cas_gc_undercount_repro.cpp | 416 ++ src/Disks/tests/gtest_cas_heartbeat.cpp | 940 +++ .../tests/gtest_cas_holey_list_detector.cpp | 306 + src/Disks/tests/gtest_cas_ids.cpp | 41 + .../tests/gtest_cas_inline_placement.cpp | 28 + src/Disks/tests/gtest_cas_inspect.cpp | 222 + src/Disks/tests/gtest_cas_json_writer.cpp | 214 + src/Disks/tests/gtest_cas_layout.cpp | 382 ++ .../tests/gtest_cas_lifecycle_condition.cpp | 258 + .../tests/gtest_cas_lifecycle_snapshot.cpp | 236 + .../tests/gtest_cas_list_liar_end_to_end.cpp | 624 ++ src/Disks/tests/gtest_cas_manifest_id.cpp | 86 + src/Disks/tests/gtest_cas_mount.cpp | 1947 ++++++ .../tests/gtest_cas_mount_claim_conflicts.cpp | 186 + ...est_cas_namespace_file_request_profile.cpp | 558 ++ .../tests/gtest_cas_namespace_janitor.cpp | 607 ++ .../tests/gtest_cas_namespace_life_id.cpp | 429 ++ .../tests/gtest_cas_ns_creation_lifecycle.cpp | 530 ++ .../tests/gtest_cas_ns_file_incarnation.cpp | 279 + .../tests/gtest_cas_ns_file_read_contract.cpp | 247 + src/Disks/tests/gtest_cas_observability.cpp | 492 ++ src/Disks/tests/gtest_cas_operation_gate.cpp | 436 ++ .../tests/gtest_cas_orphan_manifest_sweep.cpp | 678 +++ .../tests/gtest_cas_orphan_nomination.cpp | 327 + src/Disks/tests/gtest_cas_parallel_commit.cpp | 308 + .../tests/gtest_cas_part_folder_access.cpp | 1285 ++++ .../tests/gtest_cas_part_folder_view.cpp | 105 + .../tests/gtest_cas_part_manifest_format.cpp | 523 ++ src/Disks/tests/gtest_cas_part_write.cpp | 2604 ++++++++ .../gtest_cas_part_write_root_dangle.cpp | 256 + src/Disks/tests/gtest_cas_pluggable_hash.cpp | 944 +++ src/Disks/tests/gtest_cas_pool.cpp | 3646 +++++++++++ src/Disks/tests/gtest_cas_probe.cpp | 360 ++ .../tests/gtest_cas_promote_republish.cpp | 402 ++ .../tests/gtest_cas_protocol_scenarios.cpp | 640 ++ .../gtest_cas_rebuild_condemn_nothing.cpp | 636 ++ .../tests/gtest_cas_record_stream_format.cpp | 255 + .../tests/gtest_cas_recovery_grounding.cpp | 701 +++ .../tests/gtest_cas_recovery_streaming.cpp | 647 ++ src/Disks/tests/gtest_cas_ref_carve.cpp | 366 ++ src/Disks/tests/gtest_cas_ref_catalog.cpp | 1718 ++++++ .../gtest_cas_ref_catalog_birth_wiring.cpp | 512 ++ .../tests/gtest_cas_ref_chunk_preparation.cpp | 276 + .../tests/gtest_cas_ref_chunked_flush.cpp | 919 +++ src/Disks/tests/gtest_cas_ref_ckpt.cpp | 1313 ++++ src/Disks/tests/gtest_cas_ref_ckpt_join.cpp | 549 ++ .../tests/gtest_cas_ref_contiguous_alloc.cpp | 641 ++ .../tests/gtest_cas_ref_cow_manifest_set.cpp | 392 ++ src/Disks/tests/gtest_cas_ref_cow_map.cpp | 516 ++ .../tests/gtest_cas_ref_decode_bounds.cpp | 136 + .../tests/gtest_cas_ref_epoch_seal_format.cpp | 491 ++ src/Disks/tests/gtest_cas_ref_gc.cpp | 987 +++ .../tests/gtest_cas_ref_install_safety.cpp | 959 +++ src/Disks/tests/gtest_cas_ref_intake.cpp | 260 + .../gtest_cas_ref_lane_exception_safety.cpp | 217 + src/Disks/tests/gtest_cas_ref_log_format.cpp | 790 +++ .../tests/gtest_cas_ref_read_contract.cpp | 256 + .../tests/gtest_cas_ref_recovery_cas_walk.cpp | 2010 ++++++ .../tests/gtest_cas_ref_snapshot_format.cpp | 436 ++ ...test_cas_ref_snapshot_publish_ordering.cpp | 561 ++ .../tests/gtest_cas_ref_statemachine.cpp | 1445 +++++ .../gtest_cas_ref_wedge_every_attempt.cpp | 1476 +++++ src/Disks/tests/gtest_cas_ref_writer.cpp | 5421 +++++++++++++++++ src/Disks/tests/gtest_cas_repoint.cpp | 107 + src/Disks/tests/gtest_cas_request_control.cpp | 1494 +++++ .../tests/gtest_cas_retirement_sweep.cpp | 420 ++ src/Disks/tests/gtest_cas_s3_staging.cpp | 1196 ++++ src/Disks/tests/gtest_cas_sentinel_probe.cpp | 263 + .../tests/gtest_cas_server_root_format.cpp | 152 + src/Disks/tests/gtest_cas_settings.cpp | 529 ++ .../tests/gtest_cas_shutdown_context.cpp | 205 + src/Disks/tests/gtest_cas_slot_occupy.cpp | 358 ++ .../gtest_cas_sweep_deletion_premise.cpp | 504 ++ src/Disks/tests/gtest_cas_text_format.cpp | 292 + .../tests/gtest_cas_truncate_reclaim.cpp | 282 + .../tests/gtest_cas_txn_apply_ledger.cpp | 159 + src/Disks/tests/gtest_cas_upload_detached.cpp | 732 +++ src/Disks/tests/gtest_cas_upload_fanout.cpp | 984 +++ src/Disks/tests/gtest_cas_wire_vocab.cpp | 53 + src/Disks/tests/gtest_cas_writer_duties.cpp | 545 ++ src/IO/ReadBufferFromFileView.cpp | 63 +- src/IO/ReadBufferFromMemory.cpp | 5 +- src/IO/ReadBufferFromS3.cpp | 7 + src/IO/ReadPipeline.cpp | 43 +- src/IO/ReadPipeline.h | 26 +- src/IO/S3/Client.cpp | 56 +- src/IO/S3/Client.h | 26 +- src/IO/S3/GCSConditionalDialect.cpp | 256 + src/IO/S3/GCSConditionalDialect.h | 55 + src/IO/S3/GOOG4Signer.cpp | 149 + src/IO/S3/GOOG4Signer.h | 30 + src/IO/S3/PocoHTTPClient.cpp | 105 +- src/IO/S3/PocoHTTPClient.h | 33 + src/IO/S3/PocoHTTPClientFactory.cpp | 12 +- src/IO/S3/PocoHTTPClientFactory.h | 22 + src/IO/S3/Requests.h | 27 +- src/IO/S3/copyS3File.cpp | 80 +- src/IO/S3/copyS3File.h | 10 + src/IO/S3/getObjectInfo.cpp | 15 +- src/IO/S3/getObjectInfo.h | 6 +- src/IO/S3/tests/gtest_aws_s3_client.cpp | 922 +++ .../tests/gtest_gcs_conditional_dialect.cpp | 438 ++ src/IO/S3/tests/gtest_goog4_signer.cpp | 151 + src/IO/S3AuthSettings.cpp | 2 + src/IO/S3Common.cpp | 46 + src/IO/S3Common.h | 53 +- src/IO/S3Defines.h | 8 + src/IO/WriteBufferFromFileBase.h | 7 + src/IO/WriteBufferFromFileDecorator.h | 9 + src/IO/WriteBufferFromS3.cpp | 42 +- src/IO/WriteBufferFromS3.h | 8 + src/IO/WriteSettings.h | 60 + .../gtest_read_buffer_from_file_view.cpp | 280 + .../tests/gtest_read_buffer_from_memory.cpp | 19 + src/IO/tests/gtest_s3_auth_settings.cpp | 40 + src/IO/tests/gtest_writebuffer_s3.cpp | 588 +- .../ContentAddressedGarbageCollectionLog.cpp | 114 + .../ContentAddressedGarbageCollectionLog.h | 63 + src/Interpreters/ContentAddressedLog.cpp | 73 + src/Interpreters/ContentAddressedLog.h | 47 + src/Interpreters/Context.cpp | 24 + src/Interpreters/Context.h | 4 + src/Interpreters/InterpreterSystemQuery.cpp | 537 +- src/Interpreters/InterpreterSystemQuery.h | 8 + .../VersionMetadataOnDisk.cpp | 12 + .../ServerAsynchronousMetrics.cpp | 37 + src/Interpreters/SystemLog.cpp | 2 + src/Interpreters/SystemLog.h | 2 + src/Interpreters/ThreadStatusExt.cpp | 4 + src/Parsers/ASTSystemQuery.cpp | 40 + src/Parsers/ASTSystemQuery.h | 13 + src/Parsers/ParserSystemQuery.cpp | 64 + src/Parsers/tests/gtest_Parser.cpp | 46 + .../MergeTree/DataPartStorageOnDiskBase.cpp | 262 +- .../MergeTree/DataPartStorageOnDiskBase.h | 2 + .../MergeTree/DataPartStorageOnDiskFull.cpp | 213 +- src/Storages/MergeTree/DataPartsExchange.cpp | 662 +- src/Storages/MergeTree/DataPartsExchange.h | 77 +- src/Storages/MergeTree/IDataPartStorage.h | 9 + src/Storages/MergeTree/IMergeTreeDataPart.cpp | 10 + .../MergeTree/MergeProjectionPartsTask.cpp | 4 + src/Storages/MergeTree/MergeTask.cpp | 19 + src/Storages/MergeTree/MergeTask.h | 4 + src/Storages/MergeTree/MergeTreeData.cpp | 124 + src/Storages/MergeTree/MergeTreeData.h | 14 +- .../MergeTree/MergeTreeDataWriter.cpp | 2 + .../MergeTree/MergeTreeDeduplicationLog.cpp | 29 +- src/Storages/MergeTree/MutateTask.cpp | 3 + .../gtest_deduplication_log_null_writer.cpp | 139 + .../gtest_projection_borrowed_transaction.cpp | 86 + src/Storages/StorageMergeTree.cpp | 15 +- src/Storages/StorageProxy.h | 9 + src/Storages/StorageReplicatedMergeTree.cpp | 12 + src/Storages/StorageTableProxy.h | 8 + .../StorageSystemContentAddressedMounts.cpp | 273 + .../StorageSystemContentAddressedMounts.h | 38 + src/Storages/System/attachSystemTables.cpp | 4 + tests/clickhouse-test | 42 +- ...orage_policy_for_merge_tree_by_default.xml | 66 + ...orage_policy_for_merge_tree_by_default.xml | 35 + tests/config/install.sh | 16 + .../compose/docker_compose_rustfs.yml | 17 + tests/integration/helpers/cluster.py | 85 + .../test_cas_drop_pool_member/__init__.py | 0 .../configs/server_root_id_node1.xml | 16 + .../configs/server_root_id_node2.xml | 16 + .../configs/storage_conf.xml | 46 + .../test_cas_drop_pool_member/test.py | 323 + .../test_cas_file_cache/__init__.py | 0 .../configs/storage_conf.xml | 34 + tests/integration/test_cas_file_cache/test.py | 117 + tests/integration/test_cas_gc_s3/__init__.py | 0 .../test_cas_gc_s3/configs/storage_conf.xml | 33 + tests/integration/test_cas_gc_s3/test.py | 189 + .../test_cas_gc_sharded/__init__.py | 0 .../configs/server_root_id_node1.xml | 12 + .../configs/server_root_id_node2.xml | 12 + .../configs/storage_conf.xml | 38 + tests/integration/test_cas_gc_sharded/test.py | 362 ++ tests/integration/test_cas_gcs/__init__.py | 0 .../test_cas_gcs/configs/config.xml | 54 + .../test_cas_gcs/gcs_mocks/auth.py | 90 + .../test_cas_gcs/gcs_mocks/server.py | 932 +++ tests/integration/test_cas_gcs/test.py | 1327 ++++ .../__init__.py | 0 .../configs/server_root_id_node1.xml | 10 + .../configs/server_root_id_node2.xml | 10 + .../configs/storage_conf.xml | 27 + .../test_cas_insert_fault_recovery/test.py | 163 + .../test_cas_lazy_load_recovery/__init__.py | 0 .../configs/server_root_id_node1.xml | 10 + .../configs/storage_conf.xml | 27 + .../test_cas_lazy_load_recovery/test.py | 88 + .../test_cas_mount_renewal_retry/__init__.py | 1 + .../configs/storage_conf.xml | 34 + .../docker_compose_proxy.yml | 30 + .../s3_fault_proxy.py | 397 ++ .../test_cas_mount_renewal_retry/test.py | 333 + .../test_cas_ref_snaplog/__init__.py | 0 .../configs/storage_conf.xml | 43 + .../integration/test_cas_ref_snaplog/test.py | 170 + .../test_cas_replicated_relink/__init__.py | 0 .../configs/server_root_id_node1.xml | 12 + .../configs/server_root_id_node2.xml | 12 + .../configs/storage_conf.xml | 36 + .../configs/storage_conf_other_pool.xml | 31 + .../test_cas_replicated_relink/test.py | 905 +++ tests/integration/test_cas_s3/__init__.py | 0 .../test_cas_s3/configs/storage_conf.xml | 36 + tests/integration/test_cas_s3/test.py | 223 + .../test_cas_shared_pool/__init__.py | 0 .../configs/server_root_id_node1.xml | 12 + .../configs/server_root_id_node2.xml | 12 + .../configs/storage_conf.xml | 36 + .../integration/test_cas_shared_pool/test.py | 347 ++ tests/integration/test_disks_app_func/test.py | 10 +- tests/integration/test_gcs_live/.gitignore | 3 + tests/integration/test_gcs_live/__init__.py | 0 tests/integration/test_gcs_live/test.py | 1532 +++++ .../test_replicated_database/test.py | 15 +- .../configs/filesystem_caches.xml | 8 + .../configs/named_collections.xml | 11 + .../configs/page_cache.xml | 4 + .../test_storage_gcp_auth/gcs_mocks/echo.py | 285 +- .../integration/test_storage_gcp_auth/test.py | 305 +- .../01271_show_privileges.reference | 7 + .../0_stateless/02253_empty_part_checksums.sh | 40 + .../02254_projection_broken_part.sh | 45 + .../02255_broken_parts_chain_on_start.sh | 44 + .../02369_lost_part_intersecting_merges.sh | 52 + .../02370_lost_part_intersecting_merges.sh | 58 + ...2444_async_broken_outdated_part_loading.sh | 36 + .../02486_truncate_and_unexpected_parts.sql | 1 - .../02980_s3_plain_DROP_TABLE_MergeTree.sh | 5 +- ...s3_plain_DROP_TABLE_ReplicatedMergeTree.sh | 3 +- ...lter_table_fetch_partition_thread_pool.sql | 1 + .../03352_allow_suspicious_ttl.sql | 2 +- .../0_stateless/03541_rename_column_start.sql | 2 +- .../03572_export_merge_tree_part_basic.sh | 2 +- ...ge_tree_part_limits_and_table_functions.sh | 2 +- ..._export_merge_tree_part_special_columns.sh | 2 +- ...export_merge_tree_part_filename_pattern.sh | 2 +- ...03829_insert_deduplication_info_memory.sql | 10 +- ...eplicated_missing_covered_part_on_start.sh | 59 + .../0_stateless/04278_cas_disk.reference | 9 + tests/queries/0_stateless/04278_cas_disk.sql | 49 + .../0_stateless/04279_cas_gc.reference | 6 + tests/queries/0_stateless/04279_cas_gc.sql | 66 + .../04280_cas_clone_partition_works.reference | 8 + .../04280_cas_clone_partition_works.sql | 54 + .../04282_cas_mutable_state.reference | 7 + .../0_stateless/04282_cas_mutable_state.sql | 57 + .../04283_cas_replicated_rejected.reference | 5 + .../04283_cas_replicated_rejected.sql | 47 + ...04284_cas_backup_pointer_holding.reference | 5 + .../04284_cas_backup_pointer_holding.sh | 52 + ...deduplication_window_inline_disk.reference | 5 + ...5_cas_deduplication_window_inline_disk.sql | 45 + .../04286_cas_remote_data_paths.reference | 2 + .../04286_cas_remote_data_paths.sql | 41 + ...287_cas_detach_partition_listing.reference | 4 + .../04287_cas_detach_partition_listing.sql | 38 + ..._detached_part_modification_time.reference | 4 + ...88_cas_detached_part_modification_time.sql | 40 + .../04289_cas_multi_detach_drop.reference | 7 + .../04289_cas_multi_detach_drop.sql | 43 + .../04290_cas_no_leftovers.reference | 5 + .../0_stateless/04290_cas_no_leftovers.sh | 132 + .../0_stateless/04292_cas_mutations.reference | 7 + .../0_stateless/04292_cas_mutations.sh | 114 + .../04293_cas_lightweight_delete.reference | 6 + .../04293_cas_lightweight_delete.sh | 101 + .../04294_cas_patch_parts.reference | 6 + .../0_stateless/04294_cas_patch_parts.sh | 88 + .../04295_cas_mutation_no_leftovers.reference | 5 + .../04295_cas_mutation_no_leftovers.sh | 131 + ...04299_cas_projection_inline_disk.reference | 60 + .../04299_cas_projection_inline_disk.sql | 125 + .../04300_cas_projection_multiblock.reference | 14 + .../04300_cas_projection_multiblock.sql | 53 + .../04316_reader_executor_basic.sql | 7 +- .../04327_reader_executor_metrics.sql | 10 + ...04328_reader_executor_kpi_async_metric.sql | 9 + ...000_cas_projection_carry_forward.reference | 10 + .../05000_cas_projection_carry_forward.sql | 45 + ..._cas_attach_partition_projection.reference | 11 + .../05001_cas_attach_partition_projection.sql | 46 + .../05002_cas_fetch_partition.reference | 8 + .../0_stateless/05002_cas_fetch_partition.sql | 58 + .../0_stateless/05003_cas_freeze.reference | 7 + tests/queries/0_stateless/05003_cas_freeze.sh | 62 + .../05004_cas_transactions.reference | 7 + .../0_stateless/05004_cas_transactions.sh | 62 + .../05005_cas_backup_restore.reference | 6 + .../0_stateless/05005_cas_backup_restore.sh | 50 + ...06_cas_deduplication_blob_insert.reference | 1 + .../05006_cas_deduplication_blob_insert.sql | 18 + .../05007_cas_gc_introspection.reference | 6 + .../0_stateless/05007_cas_gc_introspection.sh | 109 + .../05008_cas_gc_snapshot_prune.reference | 1 + .../05008_cas_gc_snapshot_prune.sh | 77 + .../0_stateless/05009_cas_event_log.reference | 4 + .../0_stateless/05009_cas_event_log.sql | 44 + .../05010_cas_mounts_gc_health.reference | 3 + .../0_stateless/05010_cas_mounts_gc_health.sh | 52 + .../05011_cas_gc_rebuild_access.reference | 0 .../05011_cas_gc_rebuild_access.sh | 66 + .../05012_cas_mounts_typed_columns.reference | 3 + .../05012_cas_mounts_typed_columns.sql | 8 + ...5013_system_cas_drop_pool_member.reference | 0 .../05013_system_cas_drop_pool_member.sql | 2 + ...sert_dedup_disk_commit_failpoint.reference | 1 + ...014_insert_dedup_disk_commit_failpoint.sql | 28 + ...5015_cas_reject_fake_transaction.reference | 2 + .../05015_cas_reject_fake_transaction.sh | 28 + ...5016_cas_drop_pool_member_access.reference | 0 .../05016_cas_drop_pool_member_access.sh | 37 + ...17_lazy_load_tables_sync_replica.reference | 2 + .../05017_lazy_load_tables_sync_replica.sh | 40 + .../05019_cas_fsck_access.reference | 0 .../0_stateless/05019_cas_fsck_access.sh | 44 + .../0_stateless/05020_cas_fsck.reference | 7 + tests/queries/0_stateless/05020_cas_fsck.sh | 85 + ...05021_lazy_load_tables_mutations.reference | 2 + .../05021_lazy_load_tables_mutations.sh | 41 + .../05022_cas_verb_access.reference | 0 .../0_stateless/05022_cas_verb_access.sh | 69 + ...5023_cas_dropns_leaked_namespace.reference | 14 + .../05023_cas_dropns_leaked_namespace.sh | 196 + .../05024_cas_freeze_two_roots.reference | 9 + .../0_stateless/05024_cas_freeze_two_roots.sh | 124 + ..._cas_attach_partition_cross_disk.reference | 6 + .../05025_cas_attach_partition_cross_disk.sh | 142 + .../05026_cas_manifest_path_newline.reference | 11 + .../05026_cas_manifest_path_newline.sh | 54 + 612 files changed, 172920 insertions(+), 308 deletions(-) create mode 100644 docs/en/antalya/cas/architecture/backend.md create mode 100644 docs/en/antalya/cas/architecture/blob-protocol.md create mode 100644 docs/en/antalya/cas/architecture/correctness.md create mode 100644 docs/en/antalya/cas/architecture/design-history.md create mode 100644 docs/en/antalya/cas/architecture/garbage-collection.md create mode 100644 docs/en/antalya/cas/architecture/index.md create mode 100644 docs/en/antalya/cas/architecture/manifests-and-refs.md create mode 100644 docs/en/antalya/cas/architecture/mounts-and-leases.md create mode 100644 docs/en/antalya/cas/architecture/namespaces.md create mode 100644 docs/en/antalya/cas/architecture/part-lifecycle.md create mode 100644 docs/en/antalya/cas/architecture/read-path.md create mode 100644 docs/en/antalya/cas/architecture/replication.md create mode 100644 docs/en/antalya/cas/architecture/storage-layout.md create mode 100644 docs/en/antalya/cas/bucket-requirements.md create mode 100644 docs/en/antalya/cas/configuration.md create mode 100644 docs/en/antalya/cas/index.md create mode 100644 docs/en/antalya/cas/operations/debugging.md create mode 100644 docs/en/antalya/cas/operations/migration.md create mode 100644 docs/en/antalya/cas/operations/monitoring.md create mode 100644 docs/en/antalya/cas/operations/troubleshooting.md create mode 100644 docs/en/antalya/cas/quick-start.md create mode 100644 docs/en/antalya/cas/roadmap.md create mode 100644 docs/en/operations/system-tables/cas_gc_log.md create mode 100644 docs/en/operations/system-tables/cas_log.md create mode 100644 docs/en/operations/system-tables/cas_mounts.md create mode 100644 programs/disks/CommandCaDropMember.cpp create mode 100644 programs/disks/CommandCaGcDryRun.cpp create mode 100644 programs/disks/CommandCaGcRebuild.cpp create mode 100644 programs/disks/CommandCaInspect.cpp create mode 100644 programs/disks/CommandFsck.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedExchange.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedExchange.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasByteBudget.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/README.md create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcPhaseTimer.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartPathParser.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartPathParser.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobUploadPool.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobUploadPool.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasEventDispatcher.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasEventDispatcher.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPoolMeta.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCowManifestSet.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCowManifestSet.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCowMap.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCowMap.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobHashingWriteBuffer.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobHashingWriteBuffer.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasCodecUtil.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasNamespaceLifeId.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasTypes.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasXxh3Streamer.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/README.md create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/CMakeLists.txt create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp create mode 100644 src/Disks/tests/cas_format_test_battery.h create mode 100644 src/Disks/tests/cas_sweep_test_support.h create mode 100644 src/Disks/tests/cas_test_helpers.h create mode 100644 src/Disks/tests/gtest_ca_transaction.cpp create mode 100644 src/Disks/tests/gtest_ca_wiring.cpp create mode 100644 src/Disks/tests/gtest_cas_b140_dangle.cpp create mode 100644 src/Disks/tests/gtest_cas_backend.cpp create mode 100644 src/Disks/tests/gtest_cas_backend_contract.cpp create mode 100644 src/Disks/tests/gtest_cas_backend_generation.cpp create mode 100644 src/Disks/tests/gtest_cas_backend_listing.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_digest.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_envelope_format.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_hasher.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_indegree.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_meta.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_meta_format.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_ref.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_upload_pool.cpp create mode 100644 src/Disks/tests/gtest_cas_blob_upload_pool_env.cpp create mode 100644 src/Disks/tests/gtest_cas_bootstrap_ordering.cpp create mode 100644 src/Disks/tests/gtest_cas_confirm_exact_ref.cpp create mode 100644 src/Disks/tests/gtest_cas_decommission.cpp create mode 100644 src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp create mode 100644 src/Disks/tests/gtest_cas_detached_work.cpp create mode 100644 src/Disks/tests/gtest_cas_empty_proof.cpp create mode 100644 src/Disks/tests/gtest_cas_encoding_pins.cpp create mode 100644 src/Disks/tests/gtest_cas_envelope.cpp create mode 100644 src/Disks/tests/gtest_cas_event_dispatcher.cpp create mode 100644 src/Disks/tests/gtest_cas_event_log.cpp create mode 100644 src/Disks/tests/gtest_cas_fence_generation.cpp create mode 100644 src/Disks/tests/gtest_cas_fold_seal_codec.cpp create mode 100644 src/Disks/tests/gtest_cas_fold_seal_format.cpp create mode 100644 src/Disks/tests/gtest_cas_forget.cpp create mode 100644 src/Disks/tests/gtest_cas_format.cpp create mode 100644 src/Disks/tests/gtest_cas_format_battery.cpp create mode 100644 src/Disks/tests/gtest_cas_fsck.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_ack_floor.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_attempt.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_bounded_walk.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_fold.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_frontier_gate.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_hold_grammar.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_leak.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_log.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_meta_writer.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_outcomes_format.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_rebuild.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_resume.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_round.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_round_defer.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_shard_plan.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_source_edge.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_state_format.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_stop_start.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_undercount_repro.cpp create mode 100644 src/Disks/tests/gtest_cas_heartbeat.cpp create mode 100644 src/Disks/tests/gtest_cas_holey_list_detector.cpp create mode 100644 src/Disks/tests/gtest_cas_ids.cpp create mode 100644 src/Disks/tests/gtest_cas_inline_placement.cpp create mode 100644 src/Disks/tests/gtest_cas_inspect.cpp create mode 100644 src/Disks/tests/gtest_cas_json_writer.cpp create mode 100644 src/Disks/tests/gtest_cas_layout.cpp create mode 100644 src/Disks/tests/gtest_cas_lifecycle_condition.cpp create mode 100644 src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp create mode 100644 src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp create mode 100644 src/Disks/tests/gtest_cas_manifest_id.cpp create mode 100644 src/Disks/tests/gtest_cas_mount.cpp create mode 100644 src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp create mode 100644 src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp create mode 100644 src/Disks/tests/gtest_cas_namespace_janitor.cpp create mode 100644 src/Disks/tests/gtest_cas_namespace_life_id.cpp create mode 100644 src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp create mode 100644 src/Disks/tests/gtest_cas_ns_file_incarnation.cpp create mode 100644 src/Disks/tests/gtest_cas_ns_file_read_contract.cpp create mode 100644 src/Disks/tests/gtest_cas_observability.cpp create mode 100644 src/Disks/tests/gtest_cas_operation_gate.cpp create mode 100644 src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp create mode 100644 src/Disks/tests/gtest_cas_orphan_nomination.cpp create mode 100644 src/Disks/tests/gtest_cas_parallel_commit.cpp create mode 100644 src/Disks/tests/gtest_cas_part_folder_access.cpp create mode 100644 src/Disks/tests/gtest_cas_part_folder_view.cpp create mode 100644 src/Disks/tests/gtest_cas_part_manifest_format.cpp create mode 100644 src/Disks/tests/gtest_cas_part_write.cpp create mode 100644 src/Disks/tests/gtest_cas_part_write_root_dangle.cpp create mode 100644 src/Disks/tests/gtest_cas_pluggable_hash.cpp create mode 100644 src/Disks/tests/gtest_cas_pool.cpp create mode 100644 src/Disks/tests/gtest_cas_probe.cpp create mode 100644 src/Disks/tests/gtest_cas_promote_republish.cpp create mode 100644 src/Disks/tests/gtest_cas_protocol_scenarios.cpp create mode 100644 src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp create mode 100644 src/Disks/tests/gtest_cas_record_stream_format.cpp create mode 100644 src/Disks/tests/gtest_cas_recovery_grounding.cpp create mode 100644 src/Disks/tests/gtest_cas_recovery_streaming.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_carve.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_catalog.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_chunk_preparation.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_chunked_flush.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_ckpt.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_ckpt_join.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_cow_manifest_set.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_cow_map.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_decode_bounds.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_gc.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_install_safety.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_intake.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_lane_exception_safety.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_log_format.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_read_contract.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_snapshot_format.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_statemachine.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_writer.cpp create mode 100644 src/Disks/tests/gtest_cas_repoint.cpp create mode 100644 src/Disks/tests/gtest_cas_request_control.cpp create mode 100644 src/Disks/tests/gtest_cas_retirement_sweep.cpp create mode 100644 src/Disks/tests/gtest_cas_s3_staging.cpp create mode 100644 src/Disks/tests/gtest_cas_sentinel_probe.cpp create mode 100644 src/Disks/tests/gtest_cas_server_root_format.cpp create mode 100644 src/Disks/tests/gtest_cas_settings.cpp create mode 100644 src/Disks/tests/gtest_cas_shutdown_context.cpp create mode 100644 src/Disks/tests/gtest_cas_slot_occupy.cpp create mode 100644 src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp create mode 100644 src/Disks/tests/gtest_cas_text_format.cpp create mode 100644 src/Disks/tests/gtest_cas_truncate_reclaim.cpp create mode 100644 src/Disks/tests/gtest_cas_txn_apply_ledger.cpp create mode 100644 src/Disks/tests/gtest_cas_upload_detached.cpp create mode 100644 src/Disks/tests/gtest_cas_upload_fanout.cpp create mode 100644 src/Disks/tests/gtest_cas_wire_vocab.cpp create mode 100644 src/Disks/tests/gtest_cas_writer_duties.cpp create mode 100644 src/IO/S3/GCSConditionalDialect.cpp create mode 100644 src/IO/S3/GCSConditionalDialect.h create mode 100644 src/IO/S3/GOOG4Signer.cpp create mode 100644 src/IO/S3/GOOG4Signer.h create mode 100644 src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp create mode 100644 src/IO/S3/tests/gtest_goog4_signer.cpp create mode 100644 src/IO/tests/gtest_read_buffer_from_file_view.cpp create mode 100644 src/IO/tests/gtest_read_buffer_from_memory.cpp create mode 100644 src/IO/tests/gtest_s3_auth_settings.cpp create mode 100644 src/Interpreters/ContentAddressedGarbageCollectionLog.cpp create mode 100644 src/Interpreters/ContentAddressedGarbageCollectionLog.h create mode 100644 src/Interpreters/ContentAddressedLog.cpp create mode 100644 src/Interpreters/ContentAddressedLog.h create mode 100644 src/Storages/MergeTree/tests/gtest_deduplication_log_null_writer.cpp create mode 100644 src/Storages/MergeTree/tests/gtest_projection_borrowed_transaction.cpp create mode 100644 src/Storages/System/StorageSystemContentAddressedMounts.cpp create mode 100644 src/Storages/System/StorageSystemContentAddressedMounts.h create mode 100644 tests/config/config.d/cas_s3_storage_policy_for_merge_tree_by_default.xml create mode 100644 tests/config/config.d/cas_storage_policy_for_merge_tree_by_default.xml create mode 100644 tests/integration/compose/docker_compose_rustfs.yml create mode 100644 tests/integration/test_cas_drop_pool_member/__init__.py create mode 100644 tests/integration/test_cas_drop_pool_member/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_drop_pool_member/configs/server_root_id_node2.xml create mode 100644 tests/integration/test_cas_drop_pool_member/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_drop_pool_member/test.py create mode 100644 tests/integration/test_cas_file_cache/__init__.py create mode 100644 tests/integration/test_cas_file_cache/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_file_cache/test.py create mode 100644 tests/integration/test_cas_gc_s3/__init__.py create mode 100644 tests/integration/test_cas_gc_s3/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_gc_s3/test.py create mode 100644 tests/integration/test_cas_gc_sharded/__init__.py create mode 100644 tests/integration/test_cas_gc_sharded/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_gc_sharded/configs/server_root_id_node2.xml create mode 100644 tests/integration/test_cas_gc_sharded/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_gc_sharded/test.py create mode 100644 tests/integration/test_cas_gcs/__init__.py create mode 100644 tests/integration/test_cas_gcs/configs/config.xml create mode 100644 tests/integration/test_cas_gcs/gcs_mocks/auth.py create mode 100644 tests/integration/test_cas_gcs/gcs_mocks/server.py create mode 100644 tests/integration/test_cas_gcs/test.py create mode 100644 tests/integration/test_cas_insert_fault_recovery/__init__.py create mode 100644 tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node2.xml create mode 100644 tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_insert_fault_recovery/test.py create mode 100644 tests/integration/test_cas_lazy_load_recovery/__init__.py create mode 100644 tests/integration/test_cas_lazy_load_recovery/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_lazy_load_recovery/test.py create mode 100644 tests/integration/test_cas_mount_renewal_retry/__init__.py create mode 100644 tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_mount_renewal_retry/docker_compose_proxy.yml create mode 100644 tests/integration/test_cas_mount_renewal_retry/s3_fault_proxy.py create mode 100644 tests/integration/test_cas_mount_renewal_retry/test.py create mode 100644 tests/integration/test_cas_ref_snaplog/__init__.py create mode 100644 tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_ref_snaplog/test.py create mode 100644 tests/integration/test_cas_replicated_relink/__init__.py create mode 100644 tests/integration/test_cas_replicated_relink/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_replicated_relink/configs/server_root_id_node2.xml create mode 100644 tests/integration/test_cas_replicated_relink/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_replicated_relink/configs/storage_conf_other_pool.xml create mode 100644 tests/integration/test_cas_replicated_relink/test.py create mode 100644 tests/integration/test_cas_s3/__init__.py create mode 100644 tests/integration/test_cas_s3/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_s3/test.py create mode 100644 tests/integration/test_cas_shared_pool/__init__.py create mode 100644 tests/integration/test_cas_shared_pool/configs/server_root_id_node1.xml create mode 100644 tests/integration/test_cas_shared_pool/configs/server_root_id_node2.xml create mode 100644 tests/integration/test_cas_shared_pool/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_shared_pool/test.py create mode 100644 tests/integration/test_gcs_live/.gitignore create mode 100644 tests/integration/test_gcs_live/__init__.py create mode 100644 tests/integration/test_gcs_live/test.py create mode 100644 tests/integration/test_storage_gcp_auth/configs/filesystem_caches.xml create mode 100644 tests/integration/test_storage_gcp_auth/configs/page_cache.xml create mode 100755 tests/queries/0_stateless/02253_empty_part_checksums.sh create mode 100755 tests/queries/0_stateless/02254_projection_broken_part.sh create mode 100755 tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh create mode 100755 tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh create mode 100755 tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh create mode 100755 tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh create mode 100755 tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh create mode 100644 tests/queries/0_stateless/04278_cas_disk.reference create mode 100644 tests/queries/0_stateless/04278_cas_disk.sql create mode 100644 tests/queries/0_stateless/04279_cas_gc.reference create mode 100644 tests/queries/0_stateless/04279_cas_gc.sql create mode 100644 tests/queries/0_stateless/04280_cas_clone_partition_works.reference create mode 100644 tests/queries/0_stateless/04280_cas_clone_partition_works.sql create mode 100644 tests/queries/0_stateless/04282_cas_mutable_state.reference create mode 100644 tests/queries/0_stateless/04282_cas_mutable_state.sql create mode 100644 tests/queries/0_stateless/04283_cas_replicated_rejected.reference create mode 100644 tests/queries/0_stateless/04283_cas_replicated_rejected.sql create mode 100644 tests/queries/0_stateless/04284_cas_backup_pointer_holding.reference create mode 100755 tests/queries/0_stateless/04284_cas_backup_pointer_holding.sh create mode 100644 tests/queries/0_stateless/04285_cas_deduplication_window_inline_disk.reference create mode 100644 tests/queries/0_stateless/04285_cas_deduplication_window_inline_disk.sql create mode 100644 tests/queries/0_stateless/04286_cas_remote_data_paths.reference create mode 100644 tests/queries/0_stateless/04286_cas_remote_data_paths.sql create mode 100644 tests/queries/0_stateless/04287_cas_detach_partition_listing.reference create mode 100644 tests/queries/0_stateless/04287_cas_detach_partition_listing.sql create mode 100644 tests/queries/0_stateless/04288_cas_detached_part_modification_time.reference create mode 100644 tests/queries/0_stateless/04288_cas_detached_part_modification_time.sql create mode 100644 tests/queries/0_stateless/04289_cas_multi_detach_drop.reference create mode 100644 tests/queries/0_stateless/04289_cas_multi_detach_drop.sql create mode 100644 tests/queries/0_stateless/04290_cas_no_leftovers.reference create mode 100755 tests/queries/0_stateless/04290_cas_no_leftovers.sh create mode 100644 tests/queries/0_stateless/04292_cas_mutations.reference create mode 100755 tests/queries/0_stateless/04292_cas_mutations.sh create mode 100644 tests/queries/0_stateless/04293_cas_lightweight_delete.reference create mode 100755 tests/queries/0_stateless/04293_cas_lightweight_delete.sh create mode 100644 tests/queries/0_stateless/04294_cas_patch_parts.reference create mode 100755 tests/queries/0_stateless/04294_cas_patch_parts.sh create mode 100644 tests/queries/0_stateless/04295_cas_mutation_no_leftovers.reference create mode 100755 tests/queries/0_stateless/04295_cas_mutation_no_leftovers.sh create mode 100644 tests/queries/0_stateless/04299_cas_projection_inline_disk.reference create mode 100644 tests/queries/0_stateless/04299_cas_projection_inline_disk.sql create mode 100644 tests/queries/0_stateless/04300_cas_projection_multiblock.reference create mode 100644 tests/queries/0_stateless/04300_cas_projection_multiblock.sql create mode 100644 tests/queries/0_stateless/05000_cas_projection_carry_forward.reference create mode 100644 tests/queries/0_stateless/05000_cas_projection_carry_forward.sql create mode 100644 tests/queries/0_stateless/05001_cas_attach_partition_projection.reference create mode 100644 tests/queries/0_stateless/05001_cas_attach_partition_projection.sql create mode 100644 tests/queries/0_stateless/05002_cas_fetch_partition.reference create mode 100644 tests/queries/0_stateless/05002_cas_fetch_partition.sql create mode 100644 tests/queries/0_stateless/05003_cas_freeze.reference create mode 100755 tests/queries/0_stateless/05003_cas_freeze.sh create mode 100644 tests/queries/0_stateless/05004_cas_transactions.reference create mode 100755 tests/queries/0_stateless/05004_cas_transactions.sh create mode 100644 tests/queries/0_stateless/05005_cas_backup_restore.reference create mode 100755 tests/queries/0_stateless/05005_cas_backup_restore.sh create mode 100644 tests/queries/0_stateless/05006_cas_deduplication_blob_insert.reference create mode 100644 tests/queries/0_stateless/05006_cas_deduplication_blob_insert.sql create mode 100644 tests/queries/0_stateless/05007_cas_gc_introspection.reference create mode 100755 tests/queries/0_stateless/05007_cas_gc_introspection.sh create mode 100644 tests/queries/0_stateless/05008_cas_gc_snapshot_prune.reference create mode 100755 tests/queries/0_stateless/05008_cas_gc_snapshot_prune.sh create mode 100644 tests/queries/0_stateless/05009_cas_event_log.reference create mode 100644 tests/queries/0_stateless/05009_cas_event_log.sql create mode 100644 tests/queries/0_stateless/05010_cas_mounts_gc_health.reference create mode 100755 tests/queries/0_stateless/05010_cas_mounts_gc_health.sh create mode 100644 tests/queries/0_stateless/05011_cas_gc_rebuild_access.reference create mode 100755 tests/queries/0_stateless/05011_cas_gc_rebuild_access.sh create mode 100644 tests/queries/0_stateless/05012_cas_mounts_typed_columns.reference create mode 100644 tests/queries/0_stateless/05012_cas_mounts_typed_columns.sql create mode 100644 tests/queries/0_stateless/05013_system_cas_drop_pool_member.reference create mode 100644 tests/queries/0_stateless/05013_system_cas_drop_pool_member.sql create mode 100644 tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.reference create mode 100644 tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.sql create mode 100644 tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference create mode 100755 tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh create mode 100644 tests/queries/0_stateless/05016_cas_drop_pool_member_access.reference create mode 100755 tests/queries/0_stateless/05016_cas_drop_pool_member_access.sh create mode 100644 tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.reference create mode 100755 tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.sh create mode 100644 tests/queries/0_stateless/05019_cas_fsck_access.reference create mode 100755 tests/queries/0_stateless/05019_cas_fsck_access.sh create mode 100644 tests/queries/0_stateless/05020_cas_fsck.reference create mode 100755 tests/queries/0_stateless/05020_cas_fsck.sh create mode 100644 tests/queries/0_stateless/05021_lazy_load_tables_mutations.reference create mode 100755 tests/queries/0_stateless/05021_lazy_load_tables_mutations.sh create mode 100644 tests/queries/0_stateless/05022_cas_verb_access.reference create mode 100755 tests/queries/0_stateless/05022_cas_verb_access.sh create mode 100644 tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.reference create mode 100755 tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.sh create mode 100644 tests/queries/0_stateless/05024_cas_freeze_two_roots.reference create mode 100755 tests/queries/0_stateless/05024_cas_freeze_two_roots.sh create mode 100644 tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.reference create mode 100755 tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.sh create mode 100644 tests/queries/0_stateless/05026_cas_manifest_path_newline.reference create mode 100755 tests/queries/0_stateless/05026_cas_manifest_path_newline.sh diff --git a/.github/workflows/fast_builds.yml b/.github/workflows/fast_builds.yml index 9ea30c0766a8..99429e92916e 100644 --- a/.github/workflows/fast_builds.yml +++ b/.github/workflows/fast_builds.yml @@ -1297,9 +1297,166 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, sequential)' --workflow "Fast Builds" --ci --timestamp + stateless_tests_amd_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_BINARY_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "Release Builds" --ci --timestamp + + stateless_tests_arm_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_ARM_BIN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_ARM_BIN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "Release Builds" --ci --timestamp + + stateless_tests_amd_binary_cas_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_binary, cas storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_BINARY_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "Release Builds" --ci --timestamp + finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD:.github/workflows/fast_builds.yml needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, sign_release_amd_release, sign_release_arm_release, source_upload, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] +======= + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, sign_release_amd_release, sign_release_arm_release, source_upload, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS):.github/workflows/release_builds.yml if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: @@ -1395,6 +1552,9 @@ jobs: - source_upload - stateless_tests_arm_binary_parallel - stateless_tests_arm_binary_sequential + - stateless_tests_amd_binary_cas_s3_storage_parallel + - stateless_tests_arm_binary_cas_s3_storage_parallel + - stateless_tests_amd_binary_cas_storage_parallel - finish_workflow - GrypeScanServer - GrypeScanKeeper diff --git a/.github/workflows/master.yml b/.github/workflows/master.yml index 6db56e583ee4..50fe3da00779 100644 --- a/.github/workflows/master.yml +++ b/.github/workflows/master.yml @@ -3291,6 +3291,516 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, sequential)' --workflow "MasterCI" --ci --timestamp + stateless_tests_amd_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_BINARY_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_ASAN_UBSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_ASAN_UBSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_arm_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_ARM_BIN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_ARM_BIN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "MasterCI" --ci --timestamp + + stateless_tests_amd_binary_cas_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_binary, cas storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + cat > ./ci/tmp/workflow_inputs.json << 'EOF' + ${{ toJson(github.event.inputs) }} + EOF + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_BINARY_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "MasterCI" --ci --timestamp + stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_llvm_coverage_per_test, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest] @@ -6753,7 +7263,11 @@ jobs: finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_llvm_coverage_per_test, build_amd_msan, build_amd_release, build_amd_release_pr_cache_warmup, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_release_pr_cache_warmup, build_arm_tsan, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, sign_release_amd_release, sign_release_arm_release, source_upload, sqllogic_test, sqlstorm_test, sqltest, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_3_3, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_4_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_5_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_6_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_7_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_8_8, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_parallel_1_4, stateless_tests_amd_tsan_parallel_2_4, stateless_tests_amd_tsan_parallel_3_4, stateless_tests_amd_tsan_parallel_4_4, stateless_tests_amd_tsan_s3_storage_parallel_1_3, stateless_tests_amd_tsan_s3_storage_parallel_2_3, stateless_tests_amd_tsan_s3_storage_parallel_3_3, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_azure_amd_msan, stress_test_azure_amd_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_llvm_coverage_per_test, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, clickbench_amd_release, clickbench_arm_release, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_6, integration_tests_amd_tsan_2_6, integration_tests_amd_tsan_3_6, integration_tests_amd_tsan_4_6, integration_tests_amd_tsan_5_6, integration_tests_amd_tsan_6_6, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, sign_release_amd_release, sign_release_arm_release, source_upload, sqllogic_test, sqlstorm_test, sqltest, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_4_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_5_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_6_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_7_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_8_8, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_4, stateless_tests_amd_msan_wasmedge_parallel_2_4, stateless_tests_amd_msan_wasmedge_parallel_3_4, stateless_tests_amd_msan_wasmedge_parallel_4_4, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_amd_tsan_s3_storage_parallel_1_2, stateless_tests_amd_tsan_s3_storage_parallel_2_2, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_arm_ubsan, stress_test_azure_amd_msan, stress_test_azure_amd_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: @@ -6915,6 +7429,16 @@ jobs: - stateless_tests_amd_tsan_s3_storage_sequential_2_2 - stateless_tests_arm_binary_parallel - stateless_tests_arm_binary_sequential + - stateless_tests_amd_binary_cas_s3_storage_parallel + - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2 + - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2 + - stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2 + - stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2 + - stateless_tests_amd_msan_cas_s3_storage_parallel_1_3 + - stateless_tests_amd_msan_cas_s3_storage_parallel_2_3 + - stateless_tests_amd_msan_cas_s3_storage_parallel_3_3 + - stateless_tests_arm_binary_cas_s3_storage_parallel + - stateless_tests_amd_binary_cas_storage_parallel - stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8 - stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8 - stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8 diff --git a/.github/workflows/pull_request.yml b/.github/workflows/pull_request.yml index 734e8779091b..f91d1f0e9f73 100644 --- a/.github/workflows/pull_request.yml +++ b/.github/workflows/pull_request.yml @@ -576,6 +576,63 @@ jobs: if: steps.upload_CH_ARM_ASAN_UBSAN_GH.outcome == 'failure' run: echo "::warning title=GH artifact upload failed::Failed to upload [CH_ARM_ASAN_UBSAN_GH] to GitHub artifacts (e.g. quota/rate limit). Downstream consumers will fall back to S3." +<<<<<<< HEAD +======= + build_arm_ubsan: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFybV91YnNhbik=') }} + name: "Build (arm_ubsan)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Build (arm_ubsan)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Build (arm_ubsan)' --workflow "PR" --ci --timestamp + + - name: Upload artifact CH_ARM_UBSAN_GH + id: upload_CH_ARM_UBSAN_GH + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: CH_ARM_UBSAN_GH + path: ci/tmp/build/programs/self-extracting/clickhouse + retention-days: 1 + - name: Warn on failed upload of CH_ARM_UBSAN_GH + if: steps.upload_CH_ARM_UBSAN_GH.outcome == 'failure' + run: echo "::warning title=GH artifact upload failed::Failed to upload [CH_ARM_UBSAN_GH] to GitHub artifacts (e.g. quota/rate limit). Downstream consumers will fall back to S3." + +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) build_arm_binary: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] @@ -848,7 +905,11 @@ jobs: build_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-builder] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFtZF9yZWxlYXNlKQ==') }} name: "Build (amd_release)" outputs: @@ -890,7 +951,11 @@ jobs: build_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-builder] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFybV9yZWxlYXNlKQ==') }} name: "Build (arm_release)" outputs: @@ -1217,11 +1282,1095 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'AST fuzzer (amd_debug, targeted, old_compatibility)' --workflow "PR" --ci --timestamp - bugfix_validation_unit_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] +<<<<<<< HEAD + bugfix_validation_unit_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVnZml4IHZhbGlkYXRpb24gKHVuaXQgdGVzdHMp') }} + name: "Bugfix validation (unit tests)" +======= + stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIDEvMik=') }} + name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 1/2)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Bugfix validation (unit tests)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh +<<<<<<< HEAD + PYTHONUNBUFFERED=1 python3 -m praktika run 'Bugfix validation (unit tests)' --workflow "PR" --ci --timestamp +======= + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, 1/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIDIvMik=') }} + name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_ASAN_UBSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgMS8yKQ==') }} + name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_ASAN_UBSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgMi8yKQ==') }} + name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_ASAN_UBSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)' --workflow "PR" --ci --timestamp +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + + stateless_tests_amd_debug_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHBhcmFsbGVsKQ==') }} + name: "Stateless tests (amd_debug, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_debug, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_DEBUG_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_DEBUG_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, parallel)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_debug_sequential: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHNlcXVlbnRpYWwp') }} + name: "Stateless tests (amd_debug, sequential)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_debug, sequential)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_DEBUG_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_DEBUG_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, sequential)' --workflow "PR" --ci --timestamp + +<<<<<<< HEAD + stateless_tests_amd_msan_wasmedge_parallel_1_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" +======= + stateless_tests_amd_tsan_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] + needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIDEvMik=') }} + name: "Stateless tests (amd_tsan, parallel, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, parallel, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, 1/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_tsan_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] + needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIDIvMik=') }} + name: "Stateless tests (amd_tsan, parallel, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, parallel, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, 2/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_tsan_sequential_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgMS8yKQ==') }} + name: "Stateless tests (amd_tsan, sequential, 1/2)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: +<<<<<<< HEAD + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" +======= + test_name: "Stateless tests (amd_tsan, sequential, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, 1/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_tsan_sequential_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgMi8yKQ==') }} + name: "Stateless tests (amd_tsan, sequential, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, sequential, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, 2/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_1_4: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzQp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/4)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/4)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 1/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_2_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzQp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/4)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 2/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_3_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzQp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/4)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 3/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_4_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0LzQp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/4)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 4/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_5_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA1Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 5/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_6_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA2Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 6/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_7_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA3Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 7/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_parallel_8_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA4Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 8/8)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_sequential_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDEvMik=') }} + name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 1/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_msan_wasmedge_sequential_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDIvMik=') }} + name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 2/2)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_debug_distributed_plan_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHBhcmFsbGVsKQ==') }} + name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_DEBUG_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_DEBUG_GH + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, parallel)' --workflow "PR" --ci --timestamp + + stateless_tests_amd_debug_distributed_plan_s3_storage_sequential: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVnZml4IHZhbGlkYXRpb24gKHVuaXQgdGVzdHMp') }} - name: "Bugfix validation (unit tests)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHNlcXVlbnRpYWwp') }} + name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1236,7 +2385,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Bugfix validation (unit tests)" + test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" - name: Prepare env script run: | @@ -1253,17 +2402,26 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_DEBUG_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_DEBUG_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Bugfix validation (unit tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, sequential)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_parallel: +<<<<<<< HEAD +======= + stateless_tests_amd_tsan_s3_storage_parallel_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHBhcmFsbGVsKQ==') }} - name: "Stateless tests (amd_debug, parallel)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIDEvMik=') }} + name: "Stateless tests (amd_tsan, s3 storage, parallel, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1278,7 +2436,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, parallel)" + test_name: "Stateless tests (amd_tsan, s3 storage, parallel, 1/2)" - name: Prepare env script run: | @@ -1295,24 +2453,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_sequential: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHNlcXVlbnRpYWwp') }} - name: "Stateless tests (amd_debug, sequential)" + stateless_tests_amd_tsan_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIDIvMik=') }} + name: "Stateless tests (amd_tsan, s3 storage, parallel, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1327,7 +2485,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, sequential)" + test_name: "Stateless tests (amd_tsan, s3 storage, parallel, 2/2)" - name: Prepare env script run: | @@ -1344,24 +2502,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, sequential)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_1_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" + stateless_tests_amd_tsan_s3_storage_sequential_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgMS8yKQ==') }} + name: "Stateless tests (amd_tsan, s3 storage, sequential, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1376,7 +2534,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" + test_name: "Stateless tests (amd_tsan, s3 storage, sequential, 1/2)" - name: Prepare env script run: | @@ -1393,24 +2551,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 1/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_2_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" + stateless_tests_amd_tsan_s3_storage_sequential_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgMi8yKQ==') }} + name: "Stateless tests (amd_tsan, s3 storage, sequential, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1425,7 +2583,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" + test_name: "Stateless tests (amd_tsan, s3 storage, sequential, 2/2)" - name: Prepare env script run: | @@ -1442,24 +2600,25 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 2/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_3_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + stateless_tests_arm_binary_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBwYXJhbGxlbCk=') }} + name: "Stateless tests (arm_binary, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1474,7 +2633,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" + test_name: "Stateless tests (arm_binary, parallel)" - name: Prepare env script run: | @@ -1491,24 +2650,28 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_ARM_BIN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_ARM_BIN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 3/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_4_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" + stateless_tests_arm_binary_sequential: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBzZXF1ZW50aWFsKQ==') }} + name: "Stateless tests (arm_binary, sequential)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1523,7 +2686,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" + test_name: "Stateless tests (arm_binary, sequential)" - name: Prepare env script run: | @@ -1540,24 +2703,31 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_ARM_BIN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_ARM_BIN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 4/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, sequential)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_5_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA1Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" +<<<<<<< HEAD + stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} + name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)" +======= + stateless_tests_amd_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1572,7 +2742,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" + test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" - name: Prepare env script run: | @@ -1589,24 +2759,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_BINARY_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_BINARY_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 5/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_6_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA2Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1621,7 +2791,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" - name: Prepare env script run: | @@ -1638,24 +2808,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_ASAN_UBSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_ASAN_UBSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 6/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_7_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA3Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1670,7 +2840,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" - name: Prepare env script run: | @@ -1687,24 +2857,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_ASAN_UBSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_ASAN_UBSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 7/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_8_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA4Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" + stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1719,7 +2889,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" - name: Prepare env script run: | @@ -1736,24 +2906,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 8/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_sequential_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDEvMik=') }} - name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" + stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1768,7 +2938,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" - name: Prepare env script run: | @@ -1785,24 +2955,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_sequential_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDIvMik=') }} - name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" + stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1817,7 +2987,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" - name: Prepare env script run: | @@ -1845,13 +3015,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_distributed_plan_s3_storage_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHBhcmFsbGVsKQ==') }} - name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" + stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1866,7 +3036,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" - name: Prepare env script run: | @@ -1883,24 +3053,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_distributed_plan_s3_storage_sequential: + stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHNlcXVlbnRpYWwp') }} - name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1915,7 +3085,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" - name: Prepare env script run: | @@ -1932,24 +3102,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, sequential)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)' --workflow "PR" --ci --timestamp - stateless_tests_arm_binary_parallel: + stateless_tests_arm_binary_cas_s3_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] - needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBwYXJhbGxlbCk=') }} - name: "Stateless tests (arm_binary, parallel)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (arm_binary, cas s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1964,7 +3134,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (arm_binary, parallel)" + test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" - name: Prepare env script run: | @@ -1992,13 +3162,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_arm_binary_sequential: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] - needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBzZXF1ZW50aWFsKQ==') }} - name: "Stateless tests (arm_binary, sequential)" + stateless_tests_amd_binary_cas_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2013,7 +3183,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (arm_binary, sequential)" + test_name: "Stateless tests (amd_binary, cas storage, parallel)" - name: Prepare env script run: | @@ -2030,24 +3200,25 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_ARM_BIN_GH + - name: Download artifact CH_AMD_BINARY_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_ARM_BIN_GH + name: CH_AMD_BINARY_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, sequential)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] - needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} - name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)" + stateless_tests_arm_asan_ubsan_azure_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHBhcmFsbGVsKQ==') }} + name: "Stateless tests (arm_asan_ubsan, azure, parallel)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2689,7 +3860,11 @@ jobs: stateless_tests_arm_asan_ubsan_azure_sequential_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 32g] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHNlcXVlbnRpYWwsIDEvMik=') }} name: "Stateless tests (arm_asan_ubsan, azure, sequential, 1/2)" outputs: @@ -2738,7 +3913,11 @@ jobs: stateless_tests_arm_asan_ubsan_azure_sequential_2_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 32g] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHNlcXVlbnRpYWwsIDIvMik=') }} name: "Stateless tests (arm_asan_ubsan, azure, sequential, 2/2)" outputs: @@ -2787,7 +3966,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDEvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 1/8)" outputs: @@ -2836,7 +4019,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDIvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 2/8)" outputs: @@ -2885,7 +4072,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDMvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 3/8)" outputs: @@ -2934,7 +4125,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDQvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 4/8)" outputs: @@ -2983,7 +4178,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDUvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 5/8)" outputs: @@ -3032,7 +4231,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDYvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 6/8)" outputs: @@ -3081,7 +4284,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDcvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 7/8)" outputs: @@ -3130,7 +4337,11 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDgvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 8/8)" outputs: @@ -3179,7 +4390,11 @@ jobs: integration_tests_arm_binary_distributed_plan_1_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDEvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 1/4)" outputs: @@ -3228,7 +4443,11 @@ jobs: integration_tests_arm_binary_distributed_plan_2_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDIvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 2/4)" outputs: @@ -3277,7 +4496,11 @@ jobs: integration_tests_arm_binary_distributed_plan_3_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDMvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 3/4)" outputs: @@ -3326,7 +4549,11 @@ jobs: integration_tests_arm_binary_distributed_plan_4_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDQvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 4/4)" outputs: @@ -3375,9 +4602,15 @@ jobs: integration_tests_amd_tsan_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAxLzgp') }} name: "Integration tests (amd_tsan, 1/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAxLzYp') }} + name: "Integration tests (amd_tsan, 1/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3424,9 +4657,15 @@ jobs: integration_tests_amd_tsan_2_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAyLzgp') }} name: "Integration tests (amd_tsan, 2/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAyLzYp') }} + name: "Integration tests (amd_tsan, 2/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3473,9 +4712,15 @@ jobs: integration_tests_amd_tsan_3_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAzLzgp') }} name: "Integration tests (amd_tsan, 3/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAzLzYp') }} + name: "Integration tests (amd_tsan, 3/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3522,9 +4767,15 @@ jobs: integration_tests_amd_tsan_4_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA0Lzgp') }} name: "Integration tests (amd_tsan, 4/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA0LzYp') }} + name: "Integration tests (amd_tsan, 4/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3571,9 +4822,15 @@ jobs: integration_tests_amd_tsan_5_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA1Lzgp') }} name: "Integration tests (amd_tsan, 5/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA1LzYp') }} + name: "Integration tests (amd_tsan, 5/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3620,9 +4877,15 @@ jobs: integration_tests_amd_tsan_6_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA2Lzgp') }} name: "Integration tests (amd_tsan, 6/8)" +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA2LzYp') }} + name: "Integration tests (amd_tsan, 6/6)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3767,7 +5030,11 @@ jobs: integration_tests_amd_msan_1_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAxLzEwKQ==') }} name: "Integration tests (amd_msan, 1/10)" outputs: @@ -3816,7 +5083,11 @@ jobs: integration_tests_amd_msan_2_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAyLzEwKQ==') }} name: "Integration tests (amd_msan, 2/10)" outputs: @@ -3865,7 +5136,11 @@ jobs: integration_tests_amd_msan_3_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAzLzEwKQ==') }} name: "Integration tests (amd_msan, 3/10)" outputs: @@ -3914,7 +5189,11 @@ jobs: integration_tests_amd_msan_4_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA0LzEwKQ==') }} name: "Integration tests (amd_msan, 4/10)" outputs: @@ -3963,7 +5242,11 @@ jobs: integration_tests_amd_msan_5_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA1LzEwKQ==') }} name: "Integration tests (amd_msan, 5/10)" outputs: @@ -4012,7 +5295,11 @@ jobs: integration_tests_amd_msan_6_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA2LzEwKQ==') }} name: "Integration tests (amd_msan, 6/10)" outputs: @@ -4061,7 +5348,11 @@ jobs: integration_tests_amd_msan_7_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA3LzEwKQ==') }} name: "Integration tests (amd_msan, 7/10)" outputs: @@ -4110,7 +5401,11 @@ jobs: integration_tests_amd_msan_8_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA4LzEwKQ==') }} name: "Integration tests (amd_msan, 8/10)" outputs: @@ -4159,7 +5454,11 @@ jobs: integration_tests_amd_msan_9_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA5LzEwKQ==') }} name: "Integration tests (amd_msan, 9/10)" outputs: @@ -4208,7 +5507,11 @@ jobs: integration_tests_amd_msan_10_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAxMC8xMCk=') }} name: "Integration tests (amd_msan, 10/10)" outputs: @@ -4509,7 +5812,11 @@ jobs: docker_server_image: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_release, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'RG9ja2VyIHNlcnZlciBpbWFnZQ==') }} name: "Docker server image" outputs: @@ -4551,7 +5858,11 @@ jobs: docker_keeper_image: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +======= + needs: [build_amd_release, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'RG9ja2VyIGtlZXBlciBpbWFnZQ==') }} name: "Docker keeper image" outputs: @@ -4593,7 +5904,11 @@ jobs: install_packages_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_release, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW5zdGFsbCBwYWNrYWdlcyAoYW1kX3JlbGVhc2Up') }} name: "Install packages (amd_release)" outputs: @@ -4635,7 +5950,11 @@ jobs: install_packages_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW5zdGFsbCBwYWNrYWdlcyAoYXJtX3JlbGVhc2Up') }} name: "Install packages (arm_release)" outputs: @@ -4677,7 +5996,11 @@ jobs: compatibility_check_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_release, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'Q29tcGF0aWJpbGl0eSBjaGVjayAoYW1kX3JlbGVhc2Up') }} name: "Compatibility check (amd_release)" outputs: @@ -4719,7 +6042,11 @@ jobs: compatibility_check_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'Q29tcGF0aWJpbGl0eSBjaGVjayAoYXJtX3JlbGVhc2Up') }} name: "Compatibility check (arm_release)" outputs: @@ -4761,7 +6088,11 @@ jobs: stress_test_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9kZWJ1Zyk=') }} name: "Stress test (amd_debug)" outputs: @@ -4803,7 +6134,11 @@ jobs: stress_test_amd_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9hc2FuX3Vic2FuKQ==') }} name: "Stress test (amd_asan_ubsan)" outputs: @@ -4845,7 +6180,11 @@ jobs: stress_test_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF90c2FuKQ==') }} name: "Stress test (amd_tsan)" outputs: @@ -4887,7 +6226,11 @@ jobs: stress_test_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9tc2FuKQ==') }} name: "Stress test (amd_msan)" outputs: @@ -4929,7 +6272,11 @@ jobs: stress_test_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9yZWxlYXNlKQ==') }} name: "Stress test (arm_release)" outputs: @@ -4971,7 +6318,11 @@ jobs: stress_test_arm_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9kZWJ1Zyk=') }} name: "Stress test (arm_debug)" outputs: @@ -5013,7 +6364,11 @@ jobs: stress_test_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9hc2FuX3Vic2FuKQ==') }} name: "Stress test (arm_asan_ubsan)" outputs: @@ -5055,7 +6410,11 @@ jobs: stress_test_arm_asan_ubsan_s3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9hc2FuX3Vic2FuLCBzMyk=') }} name: "Stress test (arm_asan_ubsan, s3)" outputs: @@ -5097,7 +6456,11 @@ jobs: stress_test_arm_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV90c2FuKQ==') }} name: "Stress test (arm_tsan)" outputs: @@ -5139,7 +6502,11 @@ jobs: stress_test_arm_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9tc2FuKQ==') }} name: "Stress test (arm_msan)" outputs: @@ -5179,9 +6546,57 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'Stress test (arm_msan)' --workflow "PR" --ci --timestamp +<<<<<<< HEAD ast_fuzzer_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + stress_test_arm_ubsan: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV91YnNhbik=') }} + name: "Stress test (arm_ubsan)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stress test (arm_ubsan)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stress test (arm_ubsan)' --workflow "PR" --ci --timestamp + + ast_fuzzer_amd_debug: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX2RlYnVnKQ==') }} name: "AST fuzzer (amd_debug)" outputs: @@ -5230,7 +6645,11 @@ jobs: ast_fuzzer_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYXJtX2FzYW5fdWJzYW4p') }} name: "AST fuzzer (arm_asan_ubsan)" outputs: @@ -5279,7 +6698,11 @@ jobs: ast_fuzzer_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX3RzYW4p') }} name: "AST fuzzer (amd_tsan)" outputs: @@ -5328,7 +6751,11 @@ jobs: ast_fuzzer_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX21zYW4p') }} name: "AST fuzzer (amd_msan)" outputs: @@ -5377,7 +6804,11 @@ jobs: buzzhouse_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfZGVidWcp') }} name: "BuzzHouse (amd_debug)" outputs: @@ -5426,7 +6857,11 @@ jobs: buzzhouse_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhcm1fYXNhbl91YnNhbik=') }} name: "BuzzHouse (arm_asan_ubsan)" outputs: @@ -5475,7 +6910,11 @@ jobs: buzzhouse_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfdHNhbik=') }} name: "BuzzHouse (amd_tsan)" outputs: @@ -5524,7 +6963,11 @@ jobs: buzzhouse_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfbXNhbik=') }} name: "BuzzHouse (amd_msan)" outputs: @@ -5657,7 +7100,11 @@ jobs: sqllogic_test: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U1FMTG9naWMgdGVzdA==') }} name: "SQLLogic test" outputs: @@ -5699,7 +7146,11 @@ jobs: sqlstorm_test: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] +<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U1FMU3Rvcm0gdGVzdA==') }} name: "SQLStorm test" outputs: @@ -5918,7 +7369,11 @@ jobs: finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] +<<<<<<< HEAD needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_debug_targeted, ast_fuzzer_amd_debug_targeted_old_compatibility, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, bugfix_validation_unit_tests, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_toolchain_pgo_bolt_aarch64, build_toolchain_pgo_bolt_amd64, build_wasm_parser, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, ci_tests, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_asan_ubsan_targeted, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, keeper_stress_tests_pr, parser_memory_check, promql_compliance, quick_functional_tests, source_upload, sqllogic_test, sqlstorm_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_sequential_selected_tests, stateless_tests_amd_tsan_sequential_selected_tests, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_asan_ubsan_targeted, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +======= + needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_debug_targeted, ast_fuzzer_amd_debug_targeted_old_compatibility, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, build_toolchain_pgo_bolt_aarch64, build_toolchain_pgo_bolt_amd64, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, ci_tests, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_asan_ubsan_targeted, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_6, integration_tests_amd_tsan_2_6, integration_tests_amd_tsan_3_6, integration_tests_amd_tsan_4_6, integration_tests_amd_tsan_5_6, integration_tests_amd_tsan_6_6, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, keeper_stress_tests_pr, quick_functional_tests, source_upload, sqllogic_test, sqlstorm_test, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_4, stateless_tests_amd_msan_wasmedge_parallel_2_4, stateless_tests_amd_msan_wasmedge_parallel_3_4, stateless_tests_amd_msan_wasmedge_parallel_4_4, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_amd_tsan_s3_storage_parallel_1_2, stateless_tests_amd_tsan_s3_storage_parallel_2_2, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_asan_ubsan_targeted, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_arm_ubsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: @@ -6053,6 +7508,7 @@ jobs: - stateless_tests_amd_debug_distributed_plan_s3_storage_sequential - stateless_tests_arm_binary_parallel - stateless_tests_arm_binary_sequential +<<<<<<< HEAD - stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests - stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests - stateless_tests_amd_tsan_parallel_selected_tests @@ -6067,6 +7523,19 @@ jobs: - stateless_tests_arm_asan_ubsan_azure_parallel_6_8 - stateless_tests_arm_asan_ubsan_azure_parallel_7_8 - stateless_tests_arm_asan_ubsan_azure_parallel_8_8 +======= + - stateless_tests_amd_binary_cas_s3_storage_parallel + - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2 + - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2 + - stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2 + - stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2 + - stateless_tests_amd_msan_cas_s3_storage_parallel_1_3 + - stateless_tests_amd_msan_cas_s3_storage_parallel_2_3 + - stateless_tests_amd_msan_cas_s3_storage_parallel_3_3 + - stateless_tests_arm_binary_cas_s3_storage_parallel + - stateless_tests_amd_binary_cas_storage_parallel + - stateless_tests_arm_asan_ubsan_azure_parallel +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - stateless_tests_arm_asan_ubsan_azure_sequential_1_2 - stateless_tests_arm_asan_ubsan_azure_sequential_2_2 - integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8 diff --git a/.github/workflows/pull_request_community.yml b/.github/workflows/pull_request_community.yml index 9968fd9cc635..25854d5fba10 100644 --- a/.github/workflows/pull_request_community.yml +++ b/.github/workflows/pull_request_community.yml @@ -2088,6 +2088,7 @@ jobs: if-no-files-found: ignore retention-days: 14 +<<<<<<< HEAD stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests: runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] needs: [build_amd_asan_ubsan, config_workflow, fast_test] @@ -2349,6 +2350,13 @@ jobs: needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" +======= + stateless_tests_amd_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas s3 storage, parallel)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2363,7 +2371,203 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: +<<<<<<< HEAD test_name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" +======= + test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY + uses: actions/download-artifact@v8 + with: + name: CH_AMD_BINARY + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_binary_cas_s3_storage_parallel + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_ASAN_UBSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_ASAN_UBSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_ASAN_UBSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Prepare env script run: | @@ -2390,14 +2594,22 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh +<<<<<<< HEAD PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, selected tests)' --workflow "Community PR" --ci --timestamp +======= + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "Community PR" --ci --timestamp +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Upload failure report artifact if: failure() uses: actions/upload-artifact@v7 continue-on-error: true with: +<<<<<<< HEAD name: failure-stateless_tests_amd_tsan_s3_storage_parallel_selected_tests +======= + name: failure-stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2 +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) path: | ci/tmp/result_*.json ci/tmp/test_result.txt @@ -2408,11 +2620,19 @@ jobs: if-no-files-found: ignore retention-days: 14 +<<<<<<< HEAD stateless_tests_amd_tsan_s3_storage_sequential_selected_tests: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" +======= + stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2427,7 +2647,11 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: +<<<<<<< HEAD test_name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" +======= + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Prepare env script run: | @@ -2454,14 +2678,342 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh +<<<<<<< HEAD PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, selected tests)' --workflow "Community PR" --ci --timestamp +======= + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "Community PR" --ci --timestamp +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Upload failure report artifact if: failure() uses: actions/upload-artifact@v7 continue-on-error: true with: +<<<<<<< HEAD name: failure-stateless_tests_amd_tsan_s3_storage_sequential_selected_tests +======= + name: failure-stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_MSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_msan_cas_s3_storage_parallel_1_3 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_MSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_msan_cas_s3_storage_parallel_2_3 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_MSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_MSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_msan_cas_s3_storage_parallel_3_3 + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_arm_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_ARM_BIN + uses: actions/download-artifact@v8 + with: + name: CH_ARM_BIN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_arm_binary_cas_s3_storage_parallel + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_binary_cas_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_binary, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas storage, parallel)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_binary, cas storage, parallel)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_BINARY + uses: actions/download-artifact@v8 + with: + name: CH_AMD_BINARY + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_binary_cas_storage_parallel +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) path: | ci/tmp/result_*.json ci/tmp/test_result.txt diff --git a/ci/defs/altinity_jobs.py b/ci/defs/altinity_jobs.py index 5c8f832bdb03..b10dd792a307 100644 --- a/ci/defs/altinity_jobs.py +++ b/ci/defs/altinity_jobs.py @@ -121,3 +121,53 @@ class AltinityJobConfigs: requires=[ArtifactNames.CH_AMD_BINARY_GH], ), ) + # Stateless tests with a content-addressed disk as the default MergeTree storage. + cas_functional_tests_jobs = common_ft_job_config.parametrize( + # CAS over S3: RustFS, not MinIO OSS, because the incarnation pool needs + # enforced conditional deletes. + Job.ParamSet( + parameter="amd_binary, cas s3 storage, parallel", + runs_on=RunnerLabels.AMD_MEDIUM_CPU, + requires=[ArtifactNames.CH_AMD_BINARY_GH], + ), + # The sanitizer lanes are sharded because an unsharded one exceeds the 6h + # GitHub job timeout and is killed before it uploads any results. + *[ + Job.ParamSet( + parameter=f"amd_asan_ubsan, cas s3 storage, parallel, {batch}/{total_batches}", + runs_on=RunnerLabels.AMD_MEDIUM_CPU, + requires=[ArtifactNames.CH_AMD_ASAN_UBSAN_GH], + ) + for total_batches in (2,) + for batch in range(1, total_batches + 1) + ], + *[ + Job.ParamSet( + parameter=f"amd_tsan, cas s3 storage, parallel, {batch}/{total_batches}", + runs_on=RunnerLabels.AMD_MEDIUM, + requires=[ArtifactNames.CH_AMD_TSAN_GH], + ) + for total_batches in (2,) + for batch in range(1, total_batches + 1) + ], + *[ + Job.ParamSet( + parameter=f"amd_msan, cas s3 storage, parallel, {batch}/{total_batches}", + runs_on=RunnerLabels.FUNC_TESTER_AMD, + requires=[ArtifactNames.CH_AMD_MSAN_GH], + ) + for total_batches in (3,) + for batch in range(1, total_batches + 1) + ], + Job.ParamSet( + parameter="arm_binary, cas s3 storage, parallel", + runs_on=RunnerLabels.ARM_MEDIUM_CPU, + requires=[ArtifactNames.CH_ARM_BINARY_GH], + ), + # CAS over local object storage. + Job.ParamSet( + parameter="amd_binary, cas storage, parallel", + runs_on=RunnerLabels.AMD_MEDIUM_CPU, + requires=[ArtifactNames.CH_AMD_BINARY_GH], + ), + ) diff --git a/ci/defs/job_configs.py b/ci/defs/job_configs.py index 0d96a727b19f..7420f75d9fec 100644 --- a/ci/defs/job_configs.py +++ b/ci/defs/job_configs.py @@ -233,6 +233,8 @@ class JobConfigs: runs_on=RunnerLabels.FUNC_TESTER_AMD, command="python3 ./ci/jobs/ci_tests_job.py", timeout=1200, + # NOTE (strtgbb): temp non-blocking — CH server exits 137 during CI Tests setup; dig deeper separately + allow_failure=True, run_in_docker=f"altinityinfra/integration-tests-runner+root+--privileged+--dns-search='.'+--security-opt seccomp=unconfined+--cap-add=SYS_PTRACE+{docker_sock_mount}+--volume=clickhouse_integration_tests_volume:/var/lib/docker+--cgroupns=host", digest_config=Job.CacheDigestConfig( include_paths=[ diff --git a/ci/jobs/functional_tests.py b/ci/jobs/functional_tests.py index 82df551bdb2e..28e471327708 100644 --- a/ci/jobs/functional_tests.py +++ b/ci/jobs/functional_tests.py @@ -158,7 +158,13 @@ def run_tests( "old analyzer": "--analyzer", "WasmEdge": "--wasm-engine wasmedge", "s3 storage": "--s3-storage", +<<<<<<< HEAD "DBReplicated": "--db-replicated", +======= + "cas storage": "--cas-storage", + "cas s3 storage": "--cas-s3-storage", + "DatabaseReplicated": "--db-replicated", +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) "DatabaseOrdinary": "--db-ordinary", "wide parts enabled": "--wide-parts", "ParallelReplicas": "--parallel-rep", @@ -171,6 +177,8 @@ def run_tests( OPTIONS_TO_TEST_RUNNER_ARGUMENTS = { "s3 storage": "--s3-storage --no-stateful", + "cas storage": "--cas-storage", + "cas s3 storage": "--cas-s3-storage", "ParallelReplicas": "--no-zookeeper --no-shard --no-parallel-replicas", "AsyncInsert": " --no-async-insert", "DBReplicated": " --no-stateful --replicated-database", @@ -407,6 +415,7 @@ def main(): is_selected_tests_run = False is_bugfix_validation = False is_s3_storage = False + is_cas_s3 = False is_azure_storage = False is_database_replicated = False is_shared_catalog = False @@ -464,8 +473,13 @@ def main(): is_excluded_from_llvm = True if "per_test_coverage" in to: is_per_test_coverage = True - if "s3 storage" in to: + if "s3 storage" in to and "cas" not in to: + # The CAS-over-s3 variant ("cas s3 storage") installs + # only its own default policy and must not pull in the s3 stateful-data / encrypted + # storage machinery, so it is deliberately excluded from is_s3_storage. is_s3_storage = True + if "cas s3 storage" in to: + is_cas_s3 = True if "azure" in to: is_azure_storage = True if "DBReplicated" in to: @@ -981,6 +995,7 @@ def main(): setup_notes = [] def start(): +<<<<<<< HEAD # `from_commands_run` captures this closure's stdout into the step # Result.info (hence CIDB test_context_raw) only when it returns a # failing value. Print a concise "SETUP FAILURE: " marker @@ -998,6 +1013,26 @@ def start(): # timeout; the marker just names the sub-step for triage. print("SETUP FAILURE: clickhouse-server not ready (wait_ready)") return False +======= + res = CH.start_minio(test_type="stateless") and CH.start_azurite() + if res and is_cas_s3: + # The CA-over-S3 pool lives on RustFS (M-W D-W8): the incarnation pool + # needs ENFORCED conditional deletes, which MinIO OSS lacks (the + # fail-closed capability probe rejects it). start_rustfs wipes its data + # dir per run, so no pool state bleeds between runs (the local-CA + # analogue is the per-run server-store wipe). MinIO keeps the non-CA + # s3 disks. + res = CH.start_rustfs() + res = res and CH.start() + res = res and CH.wait_ready() + if res: + if not CH.start_kafka(): + info.add_workflow_warning("Failed to start Kafka") + print("Failed to start Kafka") + # Fail fast on infra setup errors so we don't burn time + # triaging Kafka/Avro test failures caused by a broken setup. + return False +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if not CH.start_kafka(): info.add_workflow_warning("Failed to start Kafka") diff --git a/ci/jobs/scripts/clickhouse_proc.py b/ci/jobs/scripts/clickhouse_proc.py index 384cfa1800a9..2023daef88f5 100644 --- a/ci/jobs/scripts/clickhouse_proc.py +++ b/ci/jobs/scripts/clickhouse_proc.py @@ -71,6 +71,7 @@ class ClickHouseProc: SEAWEEDFS_LOG = f"{temp_dir}/seaweedfs.log" AZURITE_LOG = f"{temp_dir}/azurite.log" KAFKA_LOG = f"{temp_dir}/kafka.log" + RUSTFS_LOG = f"{temp_dir}/rustfs.log" LOGS_SAVER_CLIENT_OPTIONS = "--max_memory_usage 10G --max_threads 1 --max_rows_to_read=0 --max_result_rows 0 --max_result_bytes 0 --max_bytes_to_read 0 --max_execution_time 0 --max_execution_time_leaf 0 --max_estimated_execution_time 0" DMESG_LOG = f"{temp_dir}/dmesg.log" # TODO: run servers in dedicated wds to keep trash localised @@ -236,6 +237,77 @@ def start_seaweedfs(self, test_type): return False return True + RUSTFS_VERSION = "1.0.0-rc.3" + + def download_rustfs(self, rustfs_bin): + machine = platform.machine() + if machine not in ("x86_64", "aarch64", "arm64"): + print(f"unsupported architecture for rustfs [{machine}]") + return False + arch = "aarch64" if machine in ("aarch64", "arm64") else "x86_64" + url = ( + f"https://github.com/rustfs/rustfs/releases/download/{self.RUSTFS_VERSION}" + f"/rustfs-linux-{arch}-musl-v{self.RUSTFS_VERSION}.zip" + ) + zip_path = f"{temp_dir}/rustfs.zip" + if not Shell.check( + f"curl -sSfL --retry 3 --retry-delay 5 -o {zip_path} {url}", verbose=True + ): + print(f"failed to download rustfs from {url}") + return False + # The release zip contains the single `rustfs` binary at its root. + with zipfile.ZipFile(zip_path) as archive: + archive.extract("rustfs", temp_dir) + os.remove(zip_path) + os.chmod(rustfs_bin, 0o755) + return True + + def start_rustfs(self): + # RustFS backs the CAS-over-S3 pool because the incarnation pool needs enforced + # conditional operations (a wrong-token DELETE must fail with 412) that MinIO OSS lacks; + # MinIO keeps serving the non-CAS s3 disks on its own port. Binary and data dir live + # under ci/tmp, which CI wipes per run, so no pool state bleeds between runs. + rustfs_bin = f"{temp_dir}/rustfs" + if not Path(rustfs_bin).is_file() and not self.download_rustfs(rustfs_bin): + print(f"rustfs binary not found at {rustfs_bin} and download failed") + return False + data_dir = f"{temp_dir}/rustfs_data" + Shell.check(f"rm -rf {data_dir} && mkdir -p {data_dir}", verbose=True) + # The background data-scanner and auto-heal manager do no useful work on a single-disk + # ephemeral pool, but their namespace locks produced multi-minute bursts of 503 + # ServiceUnavailable that stalled client I/O. Client GET/PUT/LIST/DELETE do not depend on + # either. The RUSTFS_ENABLE_* spellings are deprecated since 1.0.0-beta.8. + # Raise the open-files limit for the same reason start_azurite does: under parallel load + # the server holds thousands of S3 connections, and at the default soft limit (1024) + # rustfs runs out of fds and refuses new TCP connections in bursts. + command = ( + "(ulimit -n 1048576 2>/dev/null || ulimit -n $(ulimit -Hn)) && " + f"RUSTFS_SCANNER_ENABLED=false RUSTFS_HEAL_ENABLED=false " + f"{rustfs_bin} server --address 0.0.0.0:11121 " + f"--access-key clickhouse --secret-key clickhouse {data_dir}" + ) + with open(self.RUSTFS_LOG, "w") as log_file: + self.rustfs_proc = subprocess.Popen( + command, stdout=log_file, stderr=subprocess.STDOUT, shell=True + ) + print(f"Started rustfs asynchronously with PID {self.rustfs_proc.pid}") + + if not Shell.check( + "curl -s -o /dev/null -w '%{http_code}' http://127.0.0.1:11121/ | grep -qE '403|200'", + verbose=False, + retries=6, + ): + print("Failed to start rustfs") + return False + # The `test` bucket the storage policy expects. + res = Shell.check( + "/mc alias set carustfs http://localhost:11121 clickhouse clickhouse && /mc mb --ignore-existing carustfs/test", + verbose=True, + ) + if not res: + print("Failed to create rustfs test bucket") + return res + def start_azurite(self): # Raise the open files limit before launching azurite-rs. # Each concurrent test query opens a TCP connection plus an in-memory @@ -883,6 +955,8 @@ def prepare_logs(self, info, all=False): res.append(self.AZURITE_LOG) if Path(self.KAFKA_LOG).exists(): res.append(self.KAFKA_LOG) + if Path(self.RUSTFS_LOG).exists(): + res.append(self.RUSTFS_LOG) if Path(self.DMESG_LOG).exists(): res.append(self.DMESG_LOG) if Path(self.CH_LOCAL_ERR_LOG).exists(): @@ -1276,6 +1350,29 @@ def dump_system_tables(self): Shell.check( f"sed -i 's|.*|{self.CH_LOCAL_ERR_LOG}|' /etc/clickhouse-server/config.xml" ) + # Open any CAS disk read-only: a writable open claims server-root ownership and fails + # closed against the real server's persisted owner uuid, while a read-only open skips the + # claim and is all a dump needs. Keyed on the `cas` marker + # rather than on disk names, so it covers every CAS disk however this job names it. + # `grep -R` and `sed --follow-symlinks` are required: `tests/config/install.sh` symlinks + # these configs into `config.d`, and `-r`/plain `sed` would silently match nothing. + Shell.check( + "grep -Rl 'cas' /etc/clickhouse-server/ 2>/dev/null " + "| xargs -r sed -i --follow-symlinks 's|cas|castrue|g'" + ) + # Report loudly if the substitution stops matching: a declared but not read-only CAS disk + # means this scrape is about to die on ownership. Reports; does not abort the dump. + if Shell.check( + "grep -Rlq 'cas' /etc/clickhouse-server/", + verbose=False, + ) and not Shell.check( + "grep -Rlq 'castrue' /etc/clickhouse-server/", + verbose=False, + ): + print( + "WARNING: a CAS disk is declared but the read-only marker was not inserted " + "-- `clickhouse local` will claim server-root ownership and this scrape will fail" + ) # FIXME: Hack for s3_with_keeper (note, that we don't need the disk, # the problem is that whenever we need disks all disks will be # initialized [1]) @@ -1299,11 +1396,21 @@ def dump_system_tables(self): self.restore_system_metadata_files_from_remote_database_disk() +<<<<<<< HEAD # Caches created via the disk() function live one level deeper, under # disks//status. cache_status_files = glob.glob( f"{self.ch_var_lib_dir}/filesystem_caches/*/status" ) + glob.glob(f"{self.ch_var_lib_dir}/filesystem_caches/disks/*/status") +======= + # `**`, not `*`: dynamic cache disks created by tests nest their path, e.g. + # `filesystem_caches/disks/cache_03517/status` — a one-level glob missed exactly that file, + # and the scrape died on its flock (`StatusFile.cpp` "Another server instance ... is already + # running") when the server had not released it. + cache_status_files = glob.glob( + f"{self.ch_var_lib_dir}/filesystem_caches/**/status", recursive=True + ) +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if cache_status_files: print( f"WARNING: Server died? Removing cache status files: {cache_status_files}" diff --git a/ci/jobs/scripts/clickhouse_service.py b/ci/jobs/scripts/clickhouse_service.py index 448c7b7d4168..6dc7ddaaa86f 100644 --- a/ci/jobs/scripts/clickhouse_service.py +++ b/ci/jobs/scripts/clickhouse_service.py @@ -61,6 +61,68 @@ def install_base(config_dir, var_lib_dir) -> None: ) def __enter__(self): +<<<<<<< HEAD +======= + Utils.add_to_PATH(temp_dir) + + # Download binary if absent + clickhouse_bin = Path(temp_dir) / "clickhouse" + if not clickhouse_bin.exists(): + self._download_binary() + + # Create symlinks if absent + for link_name in ("clickhouse-server", "clickhouse-client", "clickhouse-local"): + link_path = Path(temp_dir) / link_name + if not link_path.exists(): + Utils.link(clickhouse_bin, link_path) + + # Copy server config files if absent + config_dir = Path(self.ch_config_dir) + if not (config_dir / "config.xml").exists(): + config_dir.mkdir(parents=True, exist_ok=True) + src_dir = Path("./programs/server") + for name in ("config.xml", "users.xml"): + shutil.copy(src_dir / name, config_dir / name) + shutil.copytree( + src_dir / "config.d", + config_dir / "config.d", + symlinks=False, + dirs_exist_ok=True, + ) + + # Recreate data directory so it is owned by the current process user. + # If the directory was created on the host by a different UID (e.g. 501 + # on macOS) and the server runs as root inside Docker, ClickHouse raises + # MISMATCHING_USERS_FOR_PROCESS_AND_DATA and refuses to start. + if Path(self.run_path).exists(): + shutil.rmtree(self.run_path) + Path(self.run_path).mkdir(parents=True, exist_ok=True) + Path(self.log_dir).mkdir(parents=True, exist_ok=True) + Path(self.pid_file).unlink(missing_ok=True) + + argv = [ + str(Path(temp_dir) / "clickhouse-server"), + "--config-file", self.config_file, + "--pid-file", self.pid_file, + "--", + "--path", self.run_path, + "--user_files_path", self.user_files_path, + "--top_level_domains_path", f"{self.ch_config_dir}/top_level_domains", + "--logger.stderr", f"{self.log_dir}/stderr.log", + # NOTE (strtgbb): master binary rejects unknown cas_log keys (ErrorCodes 137) + "--skip_check_for_incorrect_settings", "1", + ] + print(f"Starting ClickHouse server: {shlex.join(argv)}") + with open(f"{self.log_dir}/clickhouse-server.log", "w") as log_fd: + self._proc = subprocess.Popen( + argv, + stderr=subprocess.STDOUT, + stdout=log_fd, + start_new_session=True, + cwd=self.run_path, + ) + +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) try: Utils.add_to_PATH(temp_dir) diff --git a/ci/workflows/backport_branches.py b/ci/workflows/backport_branches.py index 088aa861e6c7..61652cda23d5 100644 --- a/ci/workflows/backport_branches.py +++ b/ci/workflows/backport_branches.py @@ -5,6 +5,11 @@ from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] + workflow = Workflow.Config( name="BackportPR", event=Workflow.Event.PULL_REQUEST, @@ -25,7 +30,7 @@ JobConfigs.docker_keeper, *JobConfigs.install_check_jobs, *JobConfigs.compatibility_test_jobs, - *[job for job in JobConfigs.functional_tests_jobs if "amd_asan_ubsan" in job.name], + *[job for job in FUNCTIONAL_TESTS_JOBS if "amd_asan_ubsan" in job.name], *[ job for job in JobConfigs.unittest_jobs diff --git a/ci/workflows/fast_builds.py b/ci/workflows/fast_builds.py index 2b0607653a0f..0e10c801cfd7 100644 --- a/ci/workflows/fast_builds.py +++ b/ci/workflows/fast_builds.py @@ -5,6 +5,11 @@ from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] + # Add long retention tags to subset of artifacts clickhouse_binaries_with_tags = [] for artifact in ArtifactConfigs.clickhouse_binaries + ArtifactConfigs.clickhouse_stripped_binaries: @@ -45,7 +50,7 @@ AltinityJobConfigs.source_upload_job, *[ job - for job in JobConfigs.functional_tests_jobs + for job in FUNCTIONAL_TESTS_JOBS if any(t in job.name for t in ("release", "binary")) ], ], diff --git a/ci/workflows/master.py b/ci/workflows/master.py index 4fbdcee628ea..fbe4d824b149 100644 --- a/ci/workflows/master.py +++ b/ci/workflows/master.py @@ -14,6 +14,11 @@ from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job from ci.workflows.pull_request import REGULAR_BUILD_NAMES +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] + # Add long retention tags to subset of artifacts clickhouse_binaries_with_tags = [] for artifact in ArtifactConfigs.clickhouse_binaries + ArtifactConfigs.clickhouse_stripped_binaries: @@ -58,7 +63,7 @@ *JobConfigs.compatibility_test_jobs, *[ j - for j in JobConfigs.functional_tests_jobs + for j in FUNCTIONAL_TESTS_JOBS if "coverage" not in j.name ], # *JobConfigs.functional_test_llvm_coverage_jobs, diff --git a/ci/workflows/pull_request.py b/ci/workflows/pull_request.py index ab1c92dc9763..217bef7a4a56 100644 --- a/ci/workflows/pull_request.py +++ b/ci/workflows/pull_request.py @@ -14,6 +14,7 @@ from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job from ci.jobs.scripts.workflow_hooks.trusted import can_be_tested +<<<<<<< HEAD # Functional tests with sanitizers are trimmed down in pull requests: instead of # the full suite, their `selected tests` counterparts run only the tests selected # for the change. The full suite still runs here in the debug and plain binary @@ -35,6 +36,16 @@ ALL_FUNCTIONAL_TESTS = [job.name for job in FUNCTIONAL_TESTS_JOBS] CORE_BLOCKING_JOB_NAMES = [ +======= +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] + +ALL_FUNCTIONAL_TESTS = [job.name for job in FUNCTIONAL_TESTS_JOBS] + +FUNCTIONAL_TESTS_PARALLEL_BLOCKING_JOB_NAMES = [ +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) job.name for job in FUNCTIONAL_TESTS_JOBS if any( @@ -60,6 +71,11 @@ STYLE_AND_FAST_TESTS = [ # JobNames.STYLE_CHECK, JobNames.FAST_TEST, +<<<<<<< HEAD +======= + # NOTE (strtgbb): CI_TESTS temporarily not gating builds (allow_failure + 137 OOM during setup) + # JobNames.CI_TESTS, +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) # *[j.name for j in JobConfigs.tidy_build_arm_jobs], ] @@ -71,7 +87,7 @@ REGULAR_BUILD_NAMES = [job.name for job in JobConfigs.build_jobs] PLAIN_FUNCTIONAL_TEST_JOB = [ - j for j in JobConfigs.functional_tests_jobs if "amd_debug, parallel" in j.name + j for j in FUNCTIONAL_TESTS_JOBS if "amd_debug, parallel" in j.name ][0] workflow = Workflow.Config( diff --git a/ci/workflows/pull_request_community.py b/ci/workflows/pull_request_community.py index 7c45f1e8d748..552eb43fcf5d 100644 --- a/ci/workflows/pull_request_community.py +++ b/ci/workflows/pull_request_community.py @@ -2,9 +2,11 @@ from praktika import Workflow, Artifact from ci.defs.defs import BASE_BRANCH, DOCKERS, ArtifactConfigs, JobNames +from ci.defs.altinity_jobs import AltinityJobConfigs from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job +<<<<<<< HEAD # Functional tests with sanitizers are trimmed down in pull requests: instead of # the full suite, their `selected tests` counterparts run only the tests selected # for the change. Keep this aligned with `ci/workflows/pull_request.py`. @@ -20,6 +22,12 @@ # established full-suite MSan/WasmEdge lanes until that coverage exists. or "amd_msan, WasmEdge" in job.name ] + JobConfigs.stateless_tests_selected_pr_jobs +======= +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) FUNCTIONAL_TESTS_PARALLEL_BLOCKING_JOB_NAMES = [ job.name diff --git a/ci/workflows/release_branches.py b/ci/workflows/release_branches.py index 6ca35107561d..fb030a1eb0b4 100644 --- a/ci/workflows/release_branches.py +++ b/ci/workflows/release_branches.py @@ -1,5 +1,6 @@ from praktika import Workflow +<<<<<<< HEAD from ci.defs.defs import ( BINARIES_WITH_LONG_RETENTION, DOCKERS, @@ -7,9 +8,18 @@ SECRETS, ArtifactConfigs, ) +======= +from ci.defs.defs import BINARIES_WITH_LONG_RETENTION, DOCKERS, SECRETS, ArtifactConfigs +from ci.defs.altinity_jobs import AltinityJobConfigs +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job +FUNCTIONAL_TESTS_JOBS = [ + *JobConfigs.functional_tests_jobs, + *AltinityJobConfigs.cas_functional_tests_jobs, +] + builds_for_release_branch = [ job for job in JobConfigs.build_jobs @@ -38,7 +48,7 @@ JobConfigs.docker_server, JobConfigs.docker_keeper, *JobConfigs.install_check_master_jobs, - *[job for job in JobConfigs.functional_tests_jobs if "asan" in job.name], + *[job for job in FUNCTIONAL_TESTS_JOBS if "asan" in job.name], *[job for job in JobConfigs.unittest_jobs if "fuzzer" not in job.name], *[ job diff --git a/docs/concepts/features/configuration/server-config/storing-data.mdx b/docs/concepts/features/configuration/server-config/storing-data.mdx index 284d11ea04a9..080525b89666 100644 --- a/docs/concepts/features/configuration/server-config/storing-data.mdx +++ b/docs/concepts/features/configuration/server-config/storing-data.mdx @@ -49,8 +49,13 @@ It requires specifying:
+<<<<<<< HEAD:docs/concepts/features/configuration/server-config/storing-data.mdx Optionally, `metadata_type` can be specified (it is equal to `local` by default), but it can also be set to `plain`, `web` and, starting from `24.4`, `plain_rewritable`. Usage of `plain` metadata type is described in [plain storage section](/concepts/features/configuration/server-config/storing-data#plain-storage), `web` metadata type can be used only with `web` object storage type, `local` metadata type stores metadata files locally (each metadata files contains mapping to files in object storage and some additional meta information about them). +======= +Optionally, `metadata_type` can be specified (it is equal to `local` by default), but it can also be set to `plain`, `web`, `plain_rewritable` (starting from `24.4`) and `cas`. +Usage of `plain` metadata type is described in [plain storage section](/operations/storing-data#plain-storage), `web` metadata type can be used only with `web` object storage type, `local` metadata type stores metadata files locally (each metadata files contains mapping to files in object storage and some additional meta information about them). +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS):docs/en/operations/storing-data.md For example: @@ -449,6 +454,109 @@ is equal to Starting from `24.5` it is possible to configure any object storage disk (`s3`, `azure`, `local`) using the `plain_rewritable` metadata type. +### Using Content-Addressed Storage {#content-addressed-storage} + +Setting `metadata_type` to `cas` turns a disk into a content-addressed (CAS) disk: every +object is addressed by the hash of its content rather than by a randomly generated blob name, so +identical content written by different parts (or different tables) is stored once and shared. A +background garbage collector reclaims objects once no part references them anymore; see +[`SYSTEM CAS GC RUN`](/sql-reference/statements/system#system-cas-gc-run), +[`SYSTEM CAS GC REBUILD`](/sql-reference/statements/system#system-cas-gc-rebuild), +[`SYSTEM CAS DROP POOL MEMBER`](/sql-reference/statements/system#system-cas-drop-pool-member), +and the [`system.cas_gc_log`](/operations/system-tables/cas_gc_log), +[`system.cas_mounts`](/operations/system-tables/cas_mounts), and +[`system.cas_log`](/operations/system-tables/cas_log) system tables. See the +[content-addressed storage documentation](/antalya/cas) for the architecture, operations +runbooks, and a live-validated quick start. + +Configuration: + +```xml + + object_storage + s3 + cas + https://s3.eu-west-1.amazonaws.com/clickhouse-eu-west-1.clickhouse.com/data/ + 1 + + server-{replica} + disks/s3_cas/cas_scratch/ + local + cityhash128 + true + 60 + 1 + 67108864 + always + +``` + +The `CAS` settings are written directly inside the disk element, alongside +`object_storage` / `` and the connection settings. Several +components read this shared disk element, so `CAS` settings use the `cas_` prefix; every other key +belongs to the object-storage or generic disk layer. + +#### Required parameters {#required-parameters-content-addressed} + +- `cas_server_root_id` — the subtree of the shared pool that this server owns. When several replicas + mount the same pool (same `endpoint`), each one must own a distinct subtree, so this is normally + written with a macro, e.g. `server-{replica}`. Missing this key is + a startup error. + +#### Optional parameters {#optional-parameters-content-addressed} + +These are the commonly used settings; see [Configuration](/antalya/cas/configuration) for the full +disk-level and server-level settings surface. + +- `cas_scratch_path` — a real, server-local filesystem directory used to spill the write buffer before it + is committed to the pool (never the object-storage key prefix). Defaults to + `/disks//cas_scratch/`. A relative override is anchored to the server + data path, not the process's current working directory. +- `cas_staging_backend` — `local` (default) or `s3`. Selects where in-flight part data is staged before + being committed into the pool; `local` is byte-for-byte the original write path, `s3` enables + S3-native staging. A writable disk configured with `s3` must support native same-store copy and + fails closed instead of falling back to client-side copy. +- `cas_blob_hash` — `cityhash128` (default), `xxh3-128`, or `sha256`. Selects the pool's blob + content-hash function. The choice is fixed at pool creation; a reopen whose `cas_blob_hash` disagrees + with the pool's recorded algorithm fails closed. See + [choosing `cas_blob_hash`](/antalya/cas/configuration#choosing-blob-hash) for the trade-offs between + the three. +- `cas_blob_hash_allow_new` — `false` by default. Admits a new hash algorithm into an existing pool's set + of recorded algorithms; without it, a `cas_blob_hash` that disagrees with what the pool already recorded + fails closed instead of silently turning the pool mixed-algorithm. +- `cas_gc_enabled` — `true` by default. Enables the background garbage collector for this disk. +- `cas_gc_interval_sec` — `60` by default; must be `>= 1`. Interval between background GC rounds. +- `cas_gc_shards` — `1` by default; must be `>= 1`. Number of blob-hash-prefix shards the GC reducer + splits work across. This is a creation-time-only setting: on reopen the pool's persisted GC state is + authoritative. +- Every physical blob materialization starts with `HEAD`. A present non-condemned blob is adopted; + an absent or condemned blob is published unconditionally and its freshness metadata is reconciled + to `Clean`. A genuine fresh miss issues no metadata GET before publication. +- `cas_gc_snapshot_generations_to_keep` — `3` by default. Number of past GC snapshot generations retained. +- `gcs_max_conditional_put_bytes` — `1` GiB by default. Bounds every conditional non-blob `PUT` on + generation-token backends, where the precondition must survive one request. This includes + create-if-absent metadata/control artifacts and conditional replacements. Blob bodies do not + consume a write-response token: their unconditional publication can use ordinary multipart and + is not subject to this cap. Irrelevant on `ETag`-based backends such as AWS S3. +- `cas_part_folder_cache_bytes` — `64` MiB by default. Size of the part-folder view cache. `0` disables + retention; this is a supported permanent operational configuration, not only a debug aid. +- `cas_part_folder_cache_max_entries` — `10000` by default. Maximum number of entries in the part-folder + view cache. +- `cas_part_folder_cache_max_entry_bytes` — `16` MiB by default. Maximum size of a single cached + part-folder view entry. +- `cas_part_folder_validate` — `always` (default), `never`, or `age `. Controls how often a + `ForceFresh` read re-proves a cached manifest body via a `HEAD` request: `always` re-proves every + time (the original, pre-optimization behavior), `never` trusts the cache without re-proving, and + `age ` re-proves only once the cached entry is older than the given number of seconds. +- `cas_manifest_decode_cache_bytes` — `128` MiB by default. Byte bound for the decoded-manifest cache. + `0` disables decode caching entirely (a diagnostic mode). +- `cas_gc_meta_pool_size` — `16` by default. Bounded thread-pool size for the GC's per-hash freshness-meta + writes (condemn/spare/delete), so a mass `DROP` condemning millions of blobs does not run fully + sequentially. +- `skip_access_check` — `false` by default. Skips the disk's `CAS` capability probe ("start now, + fix later"). The server-level `skip_access_check` flag skips the generic disk access check; + this disk key governs the `CAS` capability probe. + ### Using Azure Blob Storage {#azure-blob-storage} `MergeTree` family table engines can store data to [Azure Blob Storage](https://azure.microsoft.com/en-us/services/storage/blobs/) diff --git a/docs/en/antalya/cas/architecture/backend.md b/docs/en/antalya/cas/architecture/backend.md new file mode 100644 index 000000000000..b1d88843471e --- /dev/null +++ b/docs/en/antalya/cas/architecture/backend.md @@ -0,0 +1,138 @@ +--- +description: 'The Cas::Backend storage seam, its token contract, the per-provider conditional-write dialects, and the mount-time capability probe.' +sidebar_label: 'Backend abstraction' +sidebar_position: 11 +slug: /antalya/cas/architecture/backend +title: 'CAS Architecture — Backend Abstraction' +doc_type: 'reference' +--- + +# Backend abstraction {#backend-abstraction} + +Every protocol described elsewhere in this set — blobs, manifests, refs, mounts, GC — is written +against one interface, `Cas::Backend` (`Backend/CasBackend.h`). It is a token-aware storage seam: +every present key has exactly one current incarnation identified by an opaque `Token`, and +`putOverwrite`/`casPut` succeed only against the expected current token (or expected absence). + +## The interface {#interface} + +| Method | Contract | +|---|---| +| `get` / `getStream` | Read bytes (or a forward-only stream, for write-once objects) plus the token of the incarnation read | +| `head` | Existence, size, token, and metadata without reading the body | +| `putIfAbsent` | Create a write-once metadata/control object only when absent; `PreconditionFailed` is a returned outcome, never an exception | +| `publishBlob` | Publish a complete blob unconditionally by streaming rewrite or native same-store copy; it makes no lifecycle decision and returns no token | +| `putOverwrite` | Replace the current object only when its token equals `expected`; a mismatch is a returned outcome | +| `casPut` | `expected == nullopt` ⇒ create-if-absent CAS (used for the first write of a root object); a set `expected` conditionally replaces that exact incarnation | +| `deleteExact` | Delete only the incarnation named by `token`; a token mismatch (`TokenMismatch`) leaves the object untouched and is distinguished from `NotFound` | +| `list` | One page of keys under a prefix, resumed by the backend's own cursor | +| `supportsListTokens` | Whether `list` can surface a per-key incarnation token, letting GC discovery skip an unchanged root shard without a `GET` | + +`deleteExact`, `putIfAbsent`, and `putOverwrite`/`casPut` are safety-critical for exact deletion, +write-once metadata/control objects, and mutual exclusion. Blob-body publication deliberately has +different semantics: `PartWriteTxn::ensureBlobPresent` owns `HEAD`, freshness metadata, and proof; +`publishBlob` only moves the selected bytes. + +Writer readiness is represented by `BlobDependencyProof`, not by token presence. `Materialized` +means the writer observed a present non-condemned body or completed publication and metadata +reconciliation. `TrustedManifest` means a durable source manifest proves the blob and requires no +blob I/O. Pending state and writer tokens are not stored in the dependency record. + +**`TOKEN ⟹ CONTENT`** is the one contract item the capability probe cannot check: a token must +uniquely identify the byte content of the incarnation it labels, so that a repeated token never +means different bytes. The read-path decode cache skips a re-read on a token match, so a backend +that recycled tokens across different content would serve stale manifests — a wrong-result bug, not +merely inefficiency. `S3` `ETag`s are content-derived; the in-memory and emulated backends mint a +strictly monotonic sequence that is never reused. This remains a standing requirement of every +backend implementation, not a property the probe verifies. + +## Provider dialects {#dialects} + +`ObjectStorageBackend` (`Backend/CasObjectStorageBackend.cpp`) wraps one `IObjectStorage` and picks +its token dialect from `IObjectStorage::conditionalOpsUseGenerationTokens`: + +| Dialect | Token type | How a conditional write is expressed | +|---|---|---| +| `AWS` (default) | `ETag` | `If-None-Match: *` / `If-Match: ` sent as-is | +| `GCS` | `Generation` | The backend rewrites conditional headers before the request goes out: `If-None-Match: *` becomes `x-goog-if-generation-match: 0`, and `If-Match: ` becomes `x-goog-if-generation-match: ` (`applyGcsConditionalDialectToRequest`, `IO/S3/GCSConditionalDialect.cpp`) | + +The GCS dialect is opted into by client configuration (`http_client = gcs_hmac` or `gcp_oauth`), not +auto-detected from the endpoint host. Within such a disk it applies only to `CAS`'s own requests; +ordinary reads, writes, copies, and all blob-body publications through the same disk keep standard +`ETag` semantics. Every conditional non-blob write rejects conditional `CompleteMultipartUpload` +rather than silently dropping the precondition, so create-if-absent artifacts (`putIfAbsent` and +`casPut` with no expected token) and conditional replacements (`putOverwrite` and `casPut` with an +expected token) take the single-`PUT` path on a generation-dialect backend. Unconditional +`publishBlob` uses Default request mode and ordinary multipart policy, including above the +conditional cap. + +With `http_client = gcs_hmac`, requests are signed with Google's native `GOOG4-HMAC-SHA256` scheme, +and the request is deliberately normalised to `x-goog-` prefixes before signing. Every `x-amz-*` +header must therefore have a known GCS counterpart before signing, and one that does not is refused +with an error naming it. Two configurations reach that refusal: server-side encryption, whether +KMS-based or with a customer-supplied key, because GCS expresses encryption through a different +contract than `x-amz-server-side-encryption*`; and any custom `x-amz-*` header set on the disk with +`
`. Both fail with a clear error rather than being sent under a guessed `x-goog-` name GCS +would not honour. Configure such a disk against an AWS-compatible endpoint instead. + +`gcs_max_conditional_put_bytes` bounds every conditional non-blob `PUT` on a generation dialect, +including create-if-absent metadata/control artifacts and conditional replacements. It does not +apply to blob publication because the writer neither consumes nor records the write-response +generation. + +The two authentication paths clean up differently, and neither renames headers wholesale. On +`gcs_hmac` every request the client sends — marked or not — goes through +`prepareGcsRequestForGoog4Authentication`, which drops the stale AWS signing artifacts and then +resolves each remaining `x-amz-*` header against an explicit per-header rule table, raising an error +naming any header for which there is no rule. On `gcp_oauth` only a marked request is touched at all, +by `prepareGcsRequestForOAuthAuthentication`: it removes the AWS signing artifacts so the Bearer +token is the sole credential and passes every other `x-amz-*` header through unchanged. + +Azure Blob Storage's REST API documents equivalent conditional headers (`If-None-Match`, +`If-Match`), but no third dialect exists in this backend yet — `IObjectStorage`'s Azure +implementation does not currently wire up a `CAS` conditional path, so Azure is untested by the +capability probe below, not merely a slower-verified third case. + +## Exact-token delete, per provider {#exact-token-delete} + +`deleteExact(key, token)` is realized as a conditional `DELETE` naming the token as a precondition: +an `If-Match`-style delete on `AWS` (`ETag`), a generation-match delete on `GCS`. A precondition +failure — `S3::isPreconditionFailedError` — is reported as `DeleteOutcome::TokenMismatch`, never as +an exception; the object is left untouched. `DeleteOutcome::created_delete_marker` reports whether +the backend created a delete marker instead of actually removing the object, which the capability +probe rejects: a bucket with versioning enabled would let `CAS` "delete" a blob without freeing any +storage, and GC would silently stop reclaiming. + +## The capability probe {#capability-probe} + +`runCapabilityProbe` (`Backend/CasProbe.cpp`) runs a throwaway-key battery against every writable +mount, described in full on the [bucket requirements](/antalya/cas/bucket-requirements) page. It is +fail-closed: any check that does not pass throws `NOT_IMPLEMENTED` naming the specific failure, and +the mount refuses to become writable. Two further gates run as the battery's opening steps, and one +sits genuinely alongside it. The distinction matters: because the versioning check runs *inside* the +battery, skipping the battery used to skip it too, which is exactly why the third gate exists. + +- `checkPoolPreconditions` — inside the battery. On the `GCS`-dialect combination only, requires bucket versioning to be + *verifiably* off. A confirmed `Enabled` and an inconclusive probe both throw: `CAS` cannot assume + the safe answer here, because what it would do on a versioned bucket is delete objects it believes + it reclaimed. A probe is inconclusive when the credential may not read the bucket's versioning + configuration, or when the backend cannot answer at all. +- `checkSkipAccessCheckSupport` — alongside the battery, in the skip branch of `Pool::open`, since it + is the gate that decides whether the battery may be skipped at all. It asks whether the backend may serve a writable mount that skips the + battery at all. The `GCS`-dialect combination refuses, so `skip_access_check = true` cannot reach a + writable generation-token mount; every other backend still skips only its permitted access-check + I/O. This is what makes the exact-token delete check below unskippable on an *ordinary* writable + mount. Decommissioning a pool member is the deliberate exception — it opens writable with the + battery skipped, because the fail-closed tradeoff inverts there: refusing an ordinary mount costs + availability and protects data, whereas refusing a decommission strands a pool with a dead replica + in it and leaves the operator no way forward. +- `checkConditionalWriteSingleAttemptSupport` — inside the battery. Refuses to mount writable unless the underlying + object storage supports a single-HTTP-attempt retry profile for conditional writes. A hidden SDK + retry can outlive the writer's mount lease and obscure whether a conditional operation actually + committed, so retries on the conditional path must be explicit CAS state-machine transitions, not + transparent client behavior. + +A third staging check requires `supportsCopyMode(ObjectStorageCopyMode::NativeOnly)` when +`cas_staging_backend = s3`. Writable mount fails closed when native same-store copy is unavailable; it +does not silently fall back to local staging. Ordinary non-CAS `copyObject` fallback behavior is +unchanged. diff --git a/docs/en/antalya/cas/architecture/blob-protocol.md b/docs/en/antalya/cas/architecture/blob-protocol.md new file mode 100644 index 000000000000..3454baa393ff --- /dev/null +++ b/docs/en/antalya/cas/architecture/blob-protocol.md @@ -0,0 +1,244 @@ +--- +description: 'How CAS observes, publishes, deduplicates, and reclaims a blob: mandatory HEAD, unconditional publication, and the writer-versus-GC race.' +sidebar_label: 'Blob protocol' +sidebar_position: 3 +slug: /antalya/cas/architecture/blob-protocol +title: 'CAS Architecture — Blob Protocol' +doc_type: 'reference' +--- + +# CAS architecture — blob protocol {#blob-protocol} + +A blob is the unit of content-addressed storage: one part file's bytes, keyed by a hash of its +own content. This page covers how a blob is observed or published, how a duplicate write is +turned into a no-op, and how a writer and a `GC` round racing over the same blob are kept safe +without ever comparing multi-gigabyte bodies. Object layout and the four durable object kinds +are covered on the [overview page](/antalya/cas/architecture/); `GC`'s fold and round structure +is covered on the GC page. + +## HEAD-then-publication sequence {#conditional-write-sequence} + +Every blob body lives at a key derived purely from its content hash +(`blobs///`, `CasLayout::blobKey`), with a sidecar `.meta` object at the +same key plus `.meta`. Because the key already encodes the digest, concurrent writers may safely +replace one physical incarnation with another carrying the same logical payload. Blob publication +therefore needs no create-if-absent condition and returns no incarnation token. Conditional writes +remain necessary for mutable metadata and control objects; `GC` still uses exact-token deletion. + +```mermaid +sequenceDiagram + autonumber + participant Writer + participant S3 as Object store + + Writer->>Writer: hash source, derive key from digest + Writer->>S3: HEAD blobs/algo/hex + alt body present + S3-->>Writer: present, size, backend token t1 + Writer->>S3: GET .meta (point read, body never streamed) + alt meta Clean or absent + Writer->>Writer: record token-free BlobDependencyProof::Materialized (never adopt t1) + else meta Condemned + Writer->>S3: unconditional publish with fresh envelope + Writer->>S3: reconcile .meta to Clean + end + else body absent + S3-->>Writer: absent + Writer->>S3: unconditional publish + Writer->>S3: create or reconcile .meta to Clean + end + Writer->>Writer: record token-free BlobDependencyProof::Materialized (never retain a body token) +``` + +Ordered steps in `PartWriteTxn::ensureBlobPresent`: + +1. `requireAlive()` — the build is not abandoned, the namespace not dropped, the writer epoch + still live. +2. **Mandatory observation.** Every physical materialization begins with one blob `HEAD`, regardless + of size, provider, staging backend, or whether another writer probably uploaded the same hash. +3. On a hit, the writer reads `.meta`. `Clean` or absent metadata permits adoption; the writer + records a `Materialized` dependency proof without retaining the observed token. `Condemned` + requires a new publication. +4. On a miss, the writer does not read `.meta` before publication. It publishes its own payload + unconditionally, then creates or reconciles `.meta` to `Clean`. +5. Streaming publication mints a fresh `incarnation_tag` and can use ordinary multipart. The first + publication of an S3-staged source may use native same-store copy only after a miss; a condemned + or subsequent publication retags and streams the staged payload. +6. `BlobSource::beginPublication` consumes the shared, monotonic `publication_attempted` state + before backend I/O. A lost response cannot re-enable verbatim copy on a later attempt. +7. Retryable or ambiguous failures restart at `HEAD`. No dependency proof is recorded until a + present non-condemned body was observed, or publication and metadata reconciliation completed. + +**Two writers uploading identical content** may both observe absence and both publish. The last +physical incarnation wins, but the key proves that both payloads have the same logical identity and +durable references name that identity, not an ETag or generation. Each writer records only a +`Materialized` proof. Its durable precommit edge protects the logical blob while the physical race +settles (see [the writer-versus-GC race](#writer-gc-race)). + +### Request budget and release evidence {#request-budget-and-release-evidence} + +The protocol deliberately pays one blob `HEAD` per materialization task. A genuine fresh miss then +publishes one body and attempts one `Clean` metadata create, with no pre-publication metadata GET. A +duplicate pays the metadata read and avoids the body publication. All blob tasks remain in the +bounded `cas_blob_upload_pool_size` fan-out rather than serializing the part. + +The [performance report](/superpowers/cas/unconditional-blob-publication-performance) confirms that +request shape on three target-only runs, but it has no matched same-environment pre-change binary. +Its control-adjusted sequence ratios are not a code-version delta; performance acceptance remains +blocked pending a matched before/after pair and explicit human acceptance. The +[real-GCS result](/superpowers/cas/unconditional-blob-publication-live-results) likewise records +deterministic coverage but no credentialed Google run. Release readiness remains blocked until the +OAuth and HMAC groups pass against real GCS; ordinary `test_storage_s3` is also externally blocked +by the unavailable `clickhouse/clickhouse-server:23.3.19.33.altinitystable` image. + +## Dedup and the identity primitive {#dedup-identity} + +Two blobs are the same object if and only if they hash to the same digest under the pool's +configured algorithm. Nothing else — not size, not `LIST` order, not a cheap prefix compare — +is allowed to stand in for that check. This follows the same rule everywhere in CAS: identity +is *proven* by hash equality, never *inferred* by a cheap signal, and re-hashing on read is the +identity primitive wherever the correctness of a decision depends on it. + +The blob content hash is pluggable per pool, fixed at pool creation: `cas_blob_hash` selects +`cityhash128` (default), `xxh3-128`, or `sha256` (`parseBlobHashAlgo`, +`Primitives/CasBlobDigest.h`). A blob is identified by the pair `BlobRef = (BlobHashAlgo, digest)`, +never by a bare digest — a bare digest is ambiguous once more than one algorithm can appear in a +pool. `cas_blob_hash_allow_new` gates admitting a second algorithm into an already-populated pool's +`algos_used` set; it defaults to off. + +`cityhash128` is not cryptographically collision-resistant. A pool shared across mutually +untrusted writers should run `sha256` — CAS enforces no policy choice here; the operator picks +the threat model via `cas_blob_hash`. This is why the materialization gate is a `HEAD` (occupancy) +rather than a body compare: it tells the writer *something* already claims this key, and the +digest is the only claim CAS trusts. + +## The writer-versus-GC race {#writer-gc-race} + +This is the interleaving that gets the most reviewer attention, because a writer and a `GC` round +can legitimately disagree about whether a blob is still needed. + +```mermaid +sequenceDiagram + autonumber + participant W as Writer + participant S3 as Object store + participant GC as GC leader + + Note over GC: round n -- fold finds in-degree 0 + GC->>S3: HEAD blob -- capture exact token t1 + GC->>S3: write .meta = Condemned round n + + rect rgba(120,160,255,0.12) + Note over W,S3: a writer arrives wanting this content + W->>S3: append ref-log PRECOMMIT (durable +1 edge) + W->>S3: HEAD blob (present, token t1) + W->>S3: GET .meta + alt meta is Clean + W->>W: adopt t1 as dependency + Note over GC: next fold sees in-degree >= 1 -- spared + else meta is Condemned + W->>S3: PUT blob unconditional re-upload of writer own source, fresh incarnation tag -- token t2 not t1 + W->>S3: CAS .meta back to Clean + end + end + + Note over GC: round n+1 -- graduation, only if still zero + GC->>S3: re-verify in-degree, requires confirmed durable Condemned evidence for hash+token t1 + Note over GC: publishes delete_pending + + Note over GC: round n+2 -- the single content-delete site + GC->>S3: deleteExact(blob, t1) + alt writer republished + S3-->>GC: TokenMismatch -- nothing deleted, blob is live at t2 + else genuinely dead + S3-->>GC: Deleted -- then drop the .meta + end +``` + +The invariant that makes every interleaving safe: **revival is re-publication only — never `GET` a +condemned object to revive it.** A writer that finds `Condemned` metadata does not reuse the +existing body; it re-uploads its own source bytes under a fresh `incarnation_tag`, producing a +new token that no prior `deleteExact` call can name. `GC` never streams a body it might delete, +and a writer never trusts a body it did not itself just write. + +Why this closes the race in both directions: + +- A writer that **adopts** a present body must have read a non-`Condemned` marker, and its precommit + edge was durable *before* that read. The next fold therefore sees in-degree ≥ 1 and spares the + blob. +- A writer that **replaces a condemned incarnation** changes the token. A stale `deleteExact(t1)` then returns + `TokenMismatch` and reclaims nothing — the delete names an exact incarnation, never "the object + at this key". +- The delete lags condemnation by at least two full rounds, and publishing the one edge that + authorizes an irreversible delete requires confirmed durable `Condemned` evidence for that + exact `(hash, token)` pair. Without it `GC` never throws — it carries the entry and retries the + marker write on the next round. + +Both directions degrade to a spurious re-upload or a no-op delete. Neither can lose data or leave +a dangling manifest entry. + +**One asymmetry worth flagging:** on a local emulated disk `publishBlob` materializes the full +`[header][payload]` in memory. These publications are serialized, so at most one body is held whole +in RAM at a time. Native object storage streams and can use multipart. + +### The `.meta` sidecar {#meta-sidecar} + +`.meta` has exactly two states: `Clean` (body present, may be referenced) and `Condemned` +(`GC` observed zero in-degree; the body is still present and a writer may replace it). An +*absent* `.meta` reads exactly like `Clean` — there is no third "unaccounted" state in the +stored format; `unaccounted` is an `ca-fsck` classification, not something `GC` ever writes. + +The record carries `state`, `condemn_round`, and `size`, and deliberately carries **no token**: +it is a per-hash hint, not a per-incarnation fact. All safety comes from the body's in-envelope +`incarnation_tag` plus exact-token deletes; a stale marker costs at worst one spurious re-upload, +never a lost delete or a false revival. + +## Deterministic artifacts and the adoption pin {#deterministic-artifacts} + +Some CAS objects are a pure function of their inputs: the `GC` source-edge run files (`cas_run`) +and fold seals (`cas_fold_seal`). For these, `putDeterministicArtifact` +(`Gc/CasBlobInDegree.cpp:341-352`) is the write-once helper: + +```cpp +if (backend.putIfAbsent(key, bytes).outcome == PutOutcome::PreconditionFailed) +{ + const auto existing = backend.get(key); + if (!existing || existing->bytes != bytes) + throw Exception(ErrorCodes::CORRUPTED_DATA, ...); + /// byte-equal => our own deterministic replay; adopt (no-op). +} +``` + +The idempotency argument: identical inputs produce byte-identical output, so a replayed round — +leader deposed mid-round, round `CAS` aborted, crash-restart — re-derives exactly the same bytes. +A 412 therefore means "already occupied by our own replay", verified by comparing the fetched +bytes, not inferred from occupancy alone as blob uploads do. Divergent bytes are impossible under +correct operation and fail closed as `CORRUPTED_DATA`. + +This is the format-evolution **adoption pin**, documented in the persisted-format registry +(`Formats/README.md`): on a `putDeterministicArtifact` conflict, the writer re-encodes at the `v` +of the *existing* object rather than at its own current build's version, so two writers on +different builds replaying the same deterministic round still land on byte-identical output. + +The helper is explicitly **not** for observation-bearing artifacts — `GC` outcome logs carry +`HEAD`-observed tokens on which two observers may legitimately disagree, so those use +first-durable-write-wins byte-adopt semantics instead. And a blob body can never use this path: +the fresh-tag rule means two attempts at the same logical create are allowed to legitimately +differ, which is exactly what `putDeterministicArtifact`'s divergence check would reject. + +## Settings {#settings} + +The disk configuration element is shared by several consumers. The `CAS` settings below carry the +`cas_` prefix; the deliberately bare `gcs_max_conditional_put_bytes` is an S3 client setting. + +| Setting | Controls | Default | +|---|---|---| +| `cas_blob_hash` | Pool blob content-hash function (`cityhash128` \| `xxh3-128` \| `sha256`); fixed at pool creation | `cityhash128` | +| `cas_blob_hash_allow_new` | Explicit opt-in to admit a new hash algorithm into an existing pool's `algos_used` | `false` | +| `cas_staging_backend` | Blob staging backend (`local` \| `s3`); `s3` is opt-in | `local` | +| `cas_scratch_path` | Server-local scratch directory for the local-staging write-buffer spill; a relative value is anchored to the server data path | `/disks//cas_scratch/` | +| `gcs_max_conditional_put_bytes` | Largest conditional non-blob `PUT` on a generation-token store, covering create-if-absent artifacts and conditional replacements; unconditional blob publication is not subject to this cap | 1 GiB | + +`GC`-round budgets that gate condemnation and reclaim of these same blobs (graduation, redelete, +sweep budgets) live on the GC architecture page, not here — they govern the `GC` side of the race +in [Writer-versus-GC race](#writer-gc-race), not the write path. diff --git a/docs/en/antalya/cas/architecture/correctness.md b/docs/en/antalya/cas/architecture/correctness.md new file mode 100644 index 000000000000..35576b562c74 --- /dev/null +++ b/docs/en/antalya/cas/architecture/correctness.md @@ -0,0 +1,92 @@ +--- +description: 'What the TLA+ model corpus proves about CAS safety, the counterexamples that shaped the design, and how model-checking and soak/chaos testing complement each other.' +sidebar_label: 'Correctness' +sidebar_position: 12 +slug: /antalya/cas/architecture/correctness +title: 'CAS Architecture — Correctness' +doc_type: 'reference' +--- + +# CAS architecture — correctness {#correctness} + +`CAS` treats formal modelling as a pre-implementation gate, not after-the-fact documentation. No +task that changes safety-relevant behavior starts until the relevant `TLA+` model is green, and +"green" means every safety and liveness stage holds **and** every deliberately sabotaged variant +(`sab_*`) violates the specific rule it targets. A sabotage that fails to reproduce its named +counterexample is treated as seriously as a real violation — it means the model was not actually +covering the case it claimed to cover. This is why every safety rule below ships with the +counterexample that appears when you remove it. + +The full model index (source `.tla` files and proof-run records) lives at +`docs/superpowers/models/`; this page is the reader-facing summary. + +## Model → invariant → counterexample {#model-invariant-counterexample} + +| Model (`docs/superpowers/models/`) | Invariant it proves | Counterexample it caught | +|---|---|---| +| `CaBlobPublishCore.tla` | Split `HEAD` from unconditional publication while preserving logical content, fresh incarnation identity, monotonic `publication_attempted`, fence safety, and readiness only after metadata reconciliation | Eleven sabotage configurations cover condemned adoption, stale exact delete, ambiguous copy/PUT landing, envelope reuse after a later miss, missing precommit/meta reconciliation, fence loss, and wrong-content publication; three witnesses prove the safe paths are reachable | +| `CaIncarnationCore.tla` | `INV_NO_DANGLE`, `INV_NO_LOSS`, `INV_NO_RETURN` — the safety spine for the whole GC core | `sab_unconddelete`: replacing the exact-token delete with an unconditional one lets a stale delete kill a replacement incarnation | +| `CaBuildRootPrecommit.tla` | `INV_NO_DANGLE_COMMITTED` — a committed manifest never references an absent blob | Reproduces the dangling-manifest hazard exactly: `WriteBlob → AdoptBlob → BuildDie → GcDelete → Commit` with no presence re-check publishes a manifest over a deleted blob | +| `CaGcLeaseCore.tla` | `NoFalseSteal` — no leader steals leadership from a live, mid-round incumbent | Without the advisory heartbeat, a frozen `seq` during a round looks identical to a dead leader, and a second leader steals from the alive one | +| `CaCasMountCore.tla` | Reclaim exclusivity for an expired mount | `sab_wallclockreclaim`: trusting the foreign mount body's wall-clock timestamp (instead of observing a stable token on the reclaimer's own monotonic clock) breaks exclusivity | +| `CaB140DangleMerge.tla` | `INV_NO_LOSS` across a GC lease handoff | Trim-before-durable: a fold cursor trimmed from in-memory state (not the durable snapshot) skips a live edge across a lease handoff, and the referenced blob is deleted while still live | +| `CaGcRootLocalPartManifestCore.tla` | `INV_NO_DANGLE` over the root-local part-manifest fold | `sab_lazyfenceunsafe`: reusing a stale parent fence position instead of a fresh all-shard fence dangles a live object | +| `CaGcShardIncarnationCore.tla` | `INV_NO_DANGLING` — safety of registry-free namespace discovery | `sab_pathkeyedcursor`: dropping the per-shard incarnation from the fold cursor reintroduces an ABA hazard on delete-then-recreate at the same path | +| `CaGcAckFloorZombie.tla` | `INV_NO_DANGLE` under two fully-interleaved GC leaders | `sab_eagerdelete`: a leader deleting its own fresh (not-yet-pending) graduations — the pre-amendment single-phase behavior — dangles when a deposed leader's pass overlaps a live one | +| `CaGcRoundDeferCore.tla` | `NoOverDelete` — a deferred round may skip a rebuild only when nothing destructive is pending | `sab_graduate_on_stale`: dropping the "an unfolded delta covers this blob" guard lets a deferred round delete a blob its own unread history still protects | +| `CaEdgeBeforeObserve.tla` | The writer/`GC` publish order is safe to simplify | `sab_late_edge`: allowing adoption before the precommit closure is durable (the pre-fix order) dangles | +| `CaGcCondemnMarkerGate.tla` | Graduation requires confirmed durable `Condemned` evidence | Swallowing a failed asynchronous condemn-marker write let a writer adopt a token a later graduation was about to delete | +| `CaRefTableSnapshotLogCore.tla` | Dense, per-life ref ids with an in-band `_ckpt` recovery frontier | `sab_scanistruth`: trusting a listing as the source of truth for "acked" reproduces the real production incident where a `LIST` omitted two already-durable, already-acknowledged ref entries | + +## Soak and chaos: the empirical oracle {#soak-and-chaos} + +Model-checking and the soak/chaos harness (`utils/ca-soak/`) catch different classes of error, and +the design leans on both rather than either alone. `TLA+` proves a protocol's constraints before a +line of `C++` exists — the two-coordinate namespace-incarnation proof and the build-root necessity +proof were both design decisions made this way. The soak, running two `ReplicatedMergeTree` +replicas against one shared pool under a seeded workload and a seeded fault injector, finds what an +idealized model necessarily abstracts away: the dangling-manifest hazard and the condemned-body replacement orphan were +both first observed live in `system.cas_log` during soak runs, before either got a focused model. Each +quiesced soak checkpoint cross-checks `SQL` results against a model oracle and runs +`clickhouse-disks ca-fsck` plus `ca-gc-dryrun`, asserting `dangling=0`. + +The relationship runs in both directions: the historical resurrect-reupload orphan (`utils/ca-soak` scenario +S30, root-caused via `system.cas_log`) got a focused `TLA+` reproduction that proved the fix and +was then retired once a deterministic `gtest` (`CASGCLeak.ResurrectReplacedIncarnationReclaimed`) +covered the same scenario for less ongoing cost — the model did its job as a pre-implementation +gate and the regression coverage moved to the cheaper, faster tool. A model's proven-safe shape +also becomes the thing a later soak scenario is written to stress. The `0x1430c` +incident — a `LIST` that omitted two already-durable ref entries, caught live by an instrumented +probe rather than reproduced by brute-force enumeration — is the clearest example: it is what made +`sab_scanistruth` a permanent, named counterexample rather than a one-off incident report. + +## What this buys a reader {#what-this-buys} + +None of this proves the shipped `C++` is bug-free — a model proves its own abstraction, and several +entries in the index are annotated `MIXED` or `DRIFTED` where the concrete mechanism has moved on +from what a model checks, with the audit trail kept precisely so that gap is visible rather than +implied. What it does buy: every safety rule in the GC core has an explicit counterexample on +record for the world where that rule is missing, and the corpus is itself periodically re-audited +for faithfulness to the code — a model whose guarantee the code no longer needs is deleted rather +than kept as false comfort. + +## Test coverage {#test-coverage} + +The implementation was built test-first (TDD), and the coverage is correspondingly dense: + +| Layer | Volume | +|---|---| +| Unit tests (`gtest`, `CAS*` suites) | ~1,900 test cases across ~130 files, covering formats, the write and read paths, the ref machinery, `GC`, recovery, and the backend contract | +| Integration tests | 10 dedicated `test_cas_*` suites (shared pools, `GC` on S3, sharded `GC`, relink replication, fault-injected `INSERT` recovery, member decommission, and more) | +| Stateless tests | dozens of dedicated `CAS` tests (pool integrity, leftovers, fsck, GC), in addition to the whole standard suite running on a `CAS`-default server (below) | + +## The whole test suite, on CAS by default {#stateless-suite-on-cas} + +Beyond the model corpus and the soak harness, the standard ClickHouse **stateless test suite runs +green with `CAS` as the default `MergeTree` storage**: dedicated CI lanes +(the `cas storage` and `cas s3 storage` job families — the latter covering ASan/TSan/MSan/UBSan +and ARM against a real S3-compatible store) run every stateless test against a server whose default disk is +a `CAS` pool. A small set of tests carries the `no-cas-storage` tag and is skipped in those lanes — +tests that exercise a mechanism a content-addressed disk deliberately does not have (for example, +`s3_plain` layouts or deliberately corrupted on-disk part chains). Everything else — the thousands +of tests that define what `MergeTree` is supposed to do — passes unchanged on top of `CAS`. diff --git a/docs/en/antalya/cas/architecture/design-history.md b/docs/en/antalya/cas/architecture/design-history.md new file mode 100644 index 000000000000..6b5c371dd86f --- /dev/null +++ b/docs/en/antalya/cas/architecture/design-history.md @@ -0,0 +1,63 @@ +--- +description: 'A condensed record of the paths CAS explored and rejected, and the major design pivots that produced the current architecture.' +sidebar_label: 'Design history' +sidebar_position: 13 +slug: /antalya/cas/architecture/design-history +title: 'CAS Architecture — Design History' +doc_type: 'reference' +--- + +# CAS architecture — design history {#design-history} + +This page is a condensed record of the roads not taken: what was tried, why it was abandoned, and +the sequence of pivots that produced the architecture described elsewhere in this section. + +## Rejected paths {#rejected-paths} + +| What it was | Why it was abandoned | What replaced it | +|---|---|---| +| **Generation-in-the-key** (Epoch-Based Reclamation core; blob keys carried a generation, `blobs//`) | Required `O(files)` persistent `Keeper` writes per commit and colliding intent keys across writers building identical content; a stuck writer stalled reclamation pool-wide | The incarnation-token design: identity moved into the object body and delete precision into the backend token, removing the generation from every key | +| **Merkle tree layer** (a `Tree` object kind, `trees/` prefix, `child_gen` carried inside a tree's own identity) | Depended on the generation-in-the-key core: a reclaim at any child propagated a new generation up the entire tree chain, and the tree layer was itself an extra surface for the same class of bug | Removed entirely; trees became manifest-internal, and `Blob` is the sole durable object kind besides the manifest and the ref | +| **Integer in-degree refcount** (a mutable counter, incremented per reference, decremented per release) | The decide-to-reference-then-not-yet-durable window let the fold observe in-degree 0 for a still-live blob; a mutable counter also costs a `CAS` round-trip proportional to write volume | A derived count: `GC` folds a multiset of `+`/`-` source-edge deltas, so losing or duplicating a record can only delay reclamation, never accelerate one | +| **Extending zero-copy replication instead of a new mechanism** | Zero-copy's structural costs (a commit spanning local disk, S3, and `Keeper`; a mutable refcount) are inherent to its design, not a bug to patch | `CAS` is an alternative to zero-copy, not a replacement: both remain available, `metadata_type = cas` is opt-in per disk, and no existing deployment needs to migrate | +| **Per-incarnation body keys** (`blobs/xx/.`, an alternative to the in-body incarnation tag) | A resurrect reusing the condemned incarnation instead of minting a fresh one reintroduced the shared-key race; structurally this was generation-in-the-key again | The in-body `incarnation_tag` plus exact-token body delete, which keeps the generation out of every object key | +| **Meta as the lifecycle linearizer** (a per-hash `.meta` object whose presence/absence *was* the authority for a blob's lifetime) | The marker is a point-read hint only, never consulted by reads; treating it as the linearizer would assert a guarantee the design does not make | The meta stays advisory: the in-body incarnation tag and exact-token delete are the real authority, and an absent meta reads identically to `Clean` | +| **Raw immutable bodies with a three-state tombstone meta** | A resurrect displacing the body forced a terminal-tombstone handshake — a writer↔`GC` liveness coupling that could re-enable data loss | The settled one-key-per-hash design with an in-body incarnation tag | +| **Conditional blob creation and conditional staged copy** | Coupled immutable payload publication to provider-specific ETag/generation responses, forced generation-token GCS blobs into a single-`PUT` size cliff, and split the writer into provider/source-specific branches | One provider-neutral state machine: durable precommit, mandatory blob `HEAD`, unconditional publication for absent or condemned bodies, freshness-metadata reconciliation, and explicit `Materialized`/`TrustedManifest` proof | +| **A persistent, append-only namespace registry for `GC` discovery** | Never deregistered on drop, so it grew monotonically forever; its fence cost scaled with namespaces ever created, not namespaces live | Discovery from the ref data itself, made safe by two independent coordinates: a durable per-shard incarnation plus a pool-global round | +| **A separate all-shard fence-and-recheck phase per `GC` round** | Both phases cost `O(pool size)` GET+CAS every round regardless of churn — roughly 2.4 million requests at 100k tables | A causal ack-floor: one streaming merge per round, with no separate fence or recheck phase, cutting the request count by roughly three orders of magnitude | +| **A pool-wide sparse ref-id allocator with completeness certificates bolted on** | Successive additive fixes kept growing without closing the root cause: absence is undecidable in a sparse id space | The invariants were changed instead of patched: dense per-life ids derived from applied state, an in-band epoch seal, and a `_ckpt` head object carrying the exact acknowledged frontier | + +## Turns at a glance {#turns-at-a-glance} + +| Date | Turn | +|---|---| +| 2026-06-01 | Starting point: "content-addressed storage for `MergeTree`" thesis and a working proof of concept | +| 2026-06-07 – 10 | The generation-in-the-key core is abandoned; the incarnation-token design replaces it | +| 2026-06-11 | The incarnation model passes exhaustive model checking with zero violations | +| 2026-06-18 | A dangling manifest reference — a committed manifest naming an already-deleted blob — leads to replacing per-blob protection hints with structural build-root reachability | +| 2026-06-24 – 26 | Formats begin converging on a single self-describing envelope (completed in July as the all-text, JSON-based codec set) | +| 2026-06-26 | Root-local full-tree manifests collapse a forest of small `GC` objects into one hot/cold split | +| 2026-07-01 | The namespace registry is deleted; discovery moves to the two-coordinate incarnation-and-round scheme | +| 2026-07-02 | Fence-and-recheck `GC` rounds are replaced by the one-pass, causal ack-floor round | +| 2026-07-06 – 10 | Writer/`GC` simplification: promote-time revalidation of tokened dependencies is proved redundant | +| 2026-07-13 | Mount-lease handover becomes boundary-exclusive, closing the cross-epoch grace window without a timeout | +| 2026-07-15 | All part files become content-addressed: the mutable file set drops to empty, and disk-transaction dispatch collapses to one precommit contract | +| 2026-07-17 | An acknowledged `INSERT` that could be lost is traced to a removed durability guard and fixed | +| 2026-07-26 | A `LIST` omitting two already-durable ref entries is caught during a soak run — the incident that settles the trust model for listings | +| 2026-07-27 – 29 | The sparse-id certificate stack is abandoned; the invariants change instead — dense ids, an in-band epoch seal, and a `_ckpt` recovery frontier | +| 2026-08-01 – 03 | Recovery stops reading listings entirely and works from authoritative objects; the listing trust model is finalized | +| 2026-08-21 – 23 | Blob creation stops using conditional PUT/copy; the focused `CaBlobPublishCore` model gates mandatory `HEAD` followed by unconditional, multipart-capable publication | + +## The pattern underneath {#the-pattern} + +A few reflexes recur across these pivots and still apply to new design work: + +- **Re-derive the invariant, don't patch the mechanism.** Every durable fix came from asking what + property must hold, not from patching the specific failure observed. +- **Delay is acceptable, authorization is not.** A stale-but-honest observation can only ever + postpone a decision; the design consistently rejects any mechanism that could *accelerate* a + destructive action past its safety gates — a delay is a latency cost, a wrongful authorization is + data loss. +- **A model that no longer matches the code is worse than no model.** Superseded models are + removed rather than kept: an unfaithful proof is false comfort, not documentation. diff --git a/docs/en/antalya/cas/architecture/garbage-collection.md b/docs/en/antalya/cas/architecture/garbage-collection.md new file mode 100644 index 000000000000..c918a0a35f14 --- /dev/null +++ b/docs/en/antalya/cas/architecture/garbage-collection.md @@ -0,0 +1,263 @@ +--- +description: 'The CAS garbage collector: leadership as work de-duplication, the 18-phase round pipeline, condemnation and exact-token deletion, sharding, and observability.' +sidebar_label: 'Garbage collection' +sidebar_position: 8 +slug: /antalya/cas/architecture/garbage-collection +title: 'CAS Architecture — Garbage Collection' +doc_type: 'reference' +--- + +# CAS architecture — garbage collection {#garbage-collection} + +`GC` is the only place in `CAS` that ever deletes a blob body or a manifest body. It runs as a +background, lease-paced loop per mount (`Gc::runRegularRound`, `Gc/CasGc.cpp`), folding ref-log +history into blob in-degree, condemning what reaches zero, and deleting only after that +condemnation has survived a full extra round. This page covers leadership, the round's 18 phases, +condemnation and deletion, sharding, pruning, round cost, and observability. Manifest and ref +mechanics that `GC` folds are covered on the +[manifests-and-refs page](/antalya/cas/architecture/manifests-and-refs); the writer-versus-`GC` +race over one blob is covered on the +[blob-protocol page](/antalya/cas/architecture/blob-protocol#writer-gc-race). + +## Leadership {#leadership} + +There is **no separate `GC` lease object**. The lease lives inside `gc/state` itself as +`{owner, seq}`. + +```mermaid +stateDiagram-v2 + [*] --> Reading: GET gc/state + Reading --> Creating: object absent, never observed before + Creating --> Leader: casPut create-if-absent, cas_gc_shards fixed here, once + Reading --> Renewing: lease owner is me + Renewing --> Leader: casPut seq+1, guarded by the observed token + Reading --> Evaluating: foreign owner + Evaluating --> NotLeader: incumbent lease moved, or heartbeat moved, or steal not allowed + Evaluating --> Stealing: both frozen across a full observation window + Stealing --> Leader: casPut owner=me seq+1, on the observed token + Stealing --> NotLeader: lost the CAS, re-read and re-arm + Leader --> [*]: run the round +``` + +Two independent liveness signals are consulted before a steal: whether `(owner, seq)` moved since +the last tick, and whether the separate `gc/hb` heartbeat moved. The heartbeat is compared only +under the same remembered heartbeat owner, deliberately not against `lease.owner` — a deposed +leader's heartbeat thread keeps pulsing, and that must not cause a live new leader's lease to be +stolen. The paced background loop may steal; a manual `SYSTEM CAS GC RUN` may not, because the +safety argument needs two observations separated by real wall time. Because every renew or steal +bumps `seq`, `seq` doubles as the round's attempt id. + +**A deposed leader that keeps running cannot corrupt anything**, and the argument does not rely on +exclusivity at all: + +1. `gc/state` is published by exactly **one** `CAS` per round; a deposed leader's `CAS` fails and + its entire round evaporates. +2. Every fold artifact is written under that leader's own attempt number, invisible to every + reader, and reclaimed later by wholesale generation pruning. +3. Destructive pre-`CAS` actions are justified only by previously published durable state, so they + are replay-idempotent. +4. Deletes are exact-token, so a stale leader can never delete a newer incarnation. + +The lease is therefore **work de-duplication, not mutual exclusion**. + +## The round {#the-round} + +A round is one pass of 18 named phases ending in exactly one `gc/state` `CAS` +(`Gc::runRegularRound`, `Gc/CasGc.cpp`). + +| # | Phase (`GcPhaseTimer` name) | What it does | +|---|---|---| +| 1 | `lease` | Acquire, renew or steal the lease inside `gc/state`. The only phase a not-a-leader round emits | +| 2 | `pre_fold_ref_drain` | Resolve catalog `Removing` rows whose cleanup evidence the adopted parent already sealed; exact-CAS-delete the completed ones before anything else can act | +| 3 | `heartbeat_floor` | One `LIST` of `gc/server-roots/`, one `GET` per mount slot, fence-out `PUT` for any mount whose write-token has held stable past the threshold | +| 4 | `defer_decision` | One full `LIST` of `cas/ns/stream/`, build the catalog-keyed ref walk plan; decide `DEFER` (nothing changed, no graduation due) or continue to a full fold. A `DEFER` verdict still runs one namespace-janitor page — the same work phase 16 does on a folding round — with its deletes suppressed | +| 5 | `parent_seal_read` | Capture the parent fold seal's run references before the fold mutates the in-memory generation/attempt, to detect a ref that moved off an already-pruned generation | +| 6 | `fold_ref_group` | Regroup the one `LIST` from phase 4 into per-table listings — no I/O, the keys are already in hand | +| 7 | `fold_seal_read` | `GET` and decode the adopted fold seal that anchors this fold's coverage | +| 8 | `fold_ref_intake` | `GET` every new ref-log record and every referenced manifest, extracting blob source edges | +| 9 | `fold_reduce` | The three-cursor merge over prior edges, new deltas and the parent's condemned rows: spare, condemn, graduate or redelete each candidate | +| 10 | `fold_seal_write` | Write the new fold seal once, write-once deterministic, adopting a byte-identical replay instead of rewriting it | +| 11 | `pending_deletes` | The single content-delete site: exact-token `deleteExact` of every entry the *previous* round marked `delete_pending`, plus the forensic outcome-log writes | +| 12 | `meta_pool_wait` | Drain the bounded pool of async `.meta` condemn-marker writes queued during the fold | +| 13 | `round_commit` | Retention-prune old generations, then publish the single `gc/state` `CAS` that adopts the whole round | +| 14 | `handoff_reclaim` | Post-`CAS`: reclaim any generation that a ref moved off during this very round, before the ordinary wholesale prune would reach it | +| 15 | `manifest_deletes` | Delete manifest bodies whose owner-removal minus-one edge the `CAS` in phase 13 just adopted | +| 16 | `namespace_cleanup` | One bounded page of the perpetual namespace janitor, reclaiming dead-life debris | +| 17 | `ref_object_cleanup` | Prune ref logs and snapshots once both fold coverage and a live snapshot make them safe to delete | +| 18 | `orphan_sweep` | Post-`CAS` exact-token deletion for the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep), after phase 13 adopted each candidate's exact blob-source retirements and the cursor. Planning retains, counts, logs, and advances past undecodable bodies without wedging later candidates; a decoded identity mismatch during planning or token ABA during deletion still fails the round with `CORRUPTED_DATA` | + +Phases 5 through 18 run only when phase 4 decides to fold. A `DEFER` verdict is not a bare no-op: +it still runs one bounded namespace-janitor page with `suppress_destructive = true` — cursor +progress and diagnostics only, no deletes — and then returns, publishing no fold artifact and no +`gc/state` `CAS` at all: + +```mermaid +flowchart LR + D4{"4 defer_decision"} -->|"nothing changed, no graduation due"| DEF["DEFER: one suppressed
namespace-janitor page, then return"] + D4 -->|"changed shards, or graduation due"| FOLD["phases 5 through 18: full fold and round commit"] +``` + +Orderings that are load-bearing: + +- **2 before 4** — a row proved complete by the adopted parent is resolved before `DEFER` or any + successor plan can publish. +- **15 after 13** — manifest bodies are deleted only after the `CAS` adopted their decrements. +- **13's prune before the `CAS`** — a pre-`CAS` destructive action may rely only on already- + published state. + +**Clamp suppression.** `suppress_destructive = !anomalies.empty() || !carried_holds.empty() || +!frontier_complete` is computed once and threaded into the merge, current-life ref cleanup and the +perpetual namespace janitor, so they cannot desynchronize. Under suppression there is no +graduation, no redelete, and no ref or namespace deletion; condemnation and sparing continue, +because both are non-destructive. + +**Fail-closed aborts.** A throw before the `CAS` means nothing is adopted: unapplied transactions, +a cursor/apply mismatch, a missing adopted seal, a table with a snapshot but no surviving log and +no cursor, a non-total condemned summary, and an observed delete marker (bucket versioning is on). + +## The one-pass commit {#gc-state} + +`gc/state` is the durable safety and round-adoption state: `round`, `gc_shards`, +`snap_generation`, `snap_pruned_through`, `snap_attempt`, `manifest_sweep_cursor`, and the lease. +Exactly one `CAS` per round publishes it; the fold itself performs no `CAS` of its own. + +**The fold seal *is* the coverage record**: generation, parent generation, one `ref_lives` row per +catalog-admitted opaque life (coverage plus optional cleanup evidence), references to the +source-edge run segments, and a per-shard condemned summary. It is encoded deterministically, so a +replayed round produces byte-identical bytes and adopts its own output through the +`putDeterministicArtifact` adoption pin (see the [blob-protocol page](/antalya/cas/architecture/blob-protocol#deterministic-artifacts)). +There is **no separate retired-list object** — condemned entries ride the source-edge run as +sentinel rows at `source_id = 0` — and **no run-file list outside the seal**; runs are resolved +*through* the seal's references, never by key construction. + +## Finding orphans {#finding-orphans} + +In-degree is a set of source edges, not a refcount. A blob becomes a candidate when its edge set +becomes empty and it was touched this pass: one `HEAD` captures the exact incarnation token and +size that a future delete will name. A blob merely carried from the parent run pays no `HEAD`. + +**The grace period is measured in rounds, not acks:** an entry graduates once it has survived one +full round (`condemn_round < current_round`). The heartbeat floor is liveness only and **never** +gates graduation. + +**The 404 rule.** A body that is present but invalid is `CORRUPTED_DATA`, hard. A body that is +missing is **never** a throw — the fold records and continues, and the caller decides by position: +a precommit activation clamps as a barrier; a committed or removal fold clamps only that table. +Prunes are likewise fail-open on 404. + +## Condemnation and deletion {#condemn-delete} + +```mermaid +flowchart LR + A["round n: in-degree hits zero
HEAD -- exact token t"] --> B["write .meta = Condemned round n
async, bounded pool, drained pre-CAS"] + B --> C["retired with condemn_round = n+1"] + C --> D{"round n+1: re-verify"} + D -->|"in-degree recovered"| S["SPARED -- recovery wins, even past the floor"] + D -->|"still zero, confirmed durable Condemned evidence for hash and t"| G["GRADUATED -- delete_pending"] + D -->|"still zero, evidence unconfirmed"| C2["carried unchanged, retry the marker, never throw"] + D -->|"current token not equal to t"| SUP["SUPERSEDED -- a writer resurrected, re-condemn the CURRENT token"] + G --> E["round n+2, pre-CAS: deleteExact blob, t"] + E -->|"Deleted or Absent"| F["then drop the .meta"] + E -->|TokenMismatch| H["nothing deleted -- live at a newer token, leave the .meta alone"] +``` + +The `.meta` sidecar carries **no token** — it is a per-hash hint. The exact incarnation token lives +in the condemned sentinel row inside the run, together with the condemn round and two flags, +`delete_pending` and `marker_confirmed`. `GC`'s marker is add-only: `Clean → Condemned` yes, the +reverse never, not even when sparing — only a writer that has already displaced the body may clear +it. Minimum two full rounds separate condemnation from deletion, and `delete_pending` is terminal — +an entry is never un-pended. + +## Sharding {#sharding} + +`cas_gc_shards` is fixed at first lease acquire and immutable; decoders reject `0`. A blob routes by +the **high** 64 bits of its digest, read big-endian. + +The role split is worth internalizing: the **coordinator** — the lease holder — owns discovery, +round visibility, the single global fence, and the generation advance, because a publish into +*one* namespace can protect a blob owned by *any* shard, so these span the whole universe and must +not be sharded. **Reducers** own only their disjoint shard; their run-key namespaces never +collide, so two servers could reduce different shards concurrently and reducer work needs no +lease. + +A shard with an empty delta bucket and no condemned entries in the parent summary copies the +parent's run references verbatim — zero run I/O, a "pure carry". A missing parent summary entry on +a non-fresh pool is `CORRUPTED_DATA`, never silently treated as zero. + +## Pruning old objects {#pruning} + +- **Current-life ref logs and snapshots** (phase 17) — a log is deletable only when covered by + both durable fold coverage and a durable live snapshot; snapshots strictly older than the newest + observed one are deletable. There is no batch delete; it is `HEAD` plus `deleteExact` per key. +- **Generations** (phase 13) — keep the last `cas_gc_snapshot_generations_to_keep` (default 3; `0` + means keep everything, for forensics). Pruning is wholesale: `LIST` the generation prefix and + delete everything under it, including deposed-leader debris and attempt-scoped outcome sets. A + generation still referenced by the live seal is skipped, but the cursor still advances past it — + leak-freedom then rests on the post-`CAS` hand-off reclaim in phase 14. +- **Manifests** — owner-removed bodies delete in phase 15; never-precommitted bodies go through the + [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep) in phase 18. + +## What a round costs {#round-cost} + +Per **folding** round, with `N` live mounts, `S` ref tables and `S_changed` tables carrying new +logs: + +| Operation | Count | +|---|---| +| `LIST cas/ns/stream/` | 1 full enumeration | +| `LIST gc/server-roots/` | 1, plus 1 `GET` per mount | +| `GET` the adopted fold seal | 5, explicitly instrumented | +| `GET` ref logs | 1 per new log | +| `GET` manifests | 1 per emitted edge — no manifest-body cache within a round | +| `PUT` run segments | 1 per non-pure-carry shard, plus 1 fold seal | +| `HEAD` blobs | 1 per newly condemned | +| `DELETE` | 1 per graduate | +| `CAS gc/state` | 1 | + +The measured `GET` formula is exact: total `GET`s equal ref-log body `GET`s plus manifest body +`GET`s, i.e. `1 + edges_per_log`. An idle round is one `LIST` sweep, `N` heartbeat `GET`s, and one +`CAS`. A deferred round is cheaper still: one `LIST`, three seal `GET`s, the lease `GET`/`PUT` and +the heartbeat floor — no `gc/state` `CAS` at all. + +The round's work is internally self-regulated: anything a pass cannot finish is carried and retried +by the next round's cursors, never dropped. The internal pacing knobs are deliberately not part of +the user-facing configuration surface. + +| Setting | Default | Bounds | +|---|---|---| +| `cas_gc_meta_pool_size` | 16 | bounded pool for condemn-marker writes | + +## Observability {#observability} + +`system.cas_gc_log` emits `Start`, `Finish` and per-`Phase` rows, correlated by `round_id` — not +`round`, which is `0` on `Start` and does not exist at all on a not-a-leader round. Phase rows +carry no verb columns by design: per-phase operation counts ride the row's own `ProfileEvents` +delta, so grouping by phase over an S3 event attributes the LIST/GET/PUT/DELETE budget without +inventing schema. `phase_metrics` carries the semantic counts no counter can supply (clamped +tables, dead precommits skipped, pure-carry shards, generations visited). `Deferred` is kept +distinct from `Success` precisely so "folded and found nothing" is distinguishable from "never +folded". Every `GC`-related `ProfileEvent` carries the uppercase `CAS`/`CASGC` prefix — for example +`CASGCRetiredCondemned`, `CASGCRetiredGraduated`, `CASGCRetiredRedeleted`, +`CASGCClampSuppressedPasses`, `CASGCHeartbeatFenceOuts`. + +Alongside it, `system.cas_log` carries the audit trail: the condemn chain, fence-outs, anomalies +(capped per round, each carrying the true total), and manifest deletes. + +`ca-fsck` distinguishes two classes that are easy to conflate: `dangling` — referenced but missing, +data loss — versus `unreachable`/`awaiting-gc` — present, unreferenced, and +simply waiting for graduation. + +## Operational surface {#operational-surface} + +| Command | Effect | +|---|---| +| `SYSTEM CAS GC RUN ''` | One synchronous round on the contacted node; only the lease holder makes progress | +| `SYSTEM CAS GC STOP` / `SYSTEM CAS GC START` | Stop or resume future rounds on the same scheduler, preserving its identity | +| `SYSTEM CAS GC REBUILD` (`clickhouse-disks ca-gc-rebuild`) | Fail-closed disaster-recovery path that every "GC refuses to run" error points at; deliberately over-protects — it prefers bounded leaks over risking an under-count. It cannot delete live data directly: deletions it produces still flow through the normal round's condemn, graduate, exact-token path | +| `clickhouse-disks ca-gc-dryrun` | Opens the disk read-only, constructs a non-leader `GC`, and prints what would be deleted with a reason per entry. Write-free, resolves runs through the seal's references. Documented caveat: it does not fold new owner events, so away from quiescence it can **over-report** — the subset guarantee holds only at quiescence, and its output must never feed a real delete | + +`SYSTEM CAS DROP POOL MEMBER '' FROM DISK ''` — permanent removal of a dead +replica, distinct from ordinary `GC` — is covered on the +[mounts-and-leases page](/antalya/cas/architecture/mounts-and-leases#mount-lifecycle). `SYSTEM CAS +FSCK` and its `dangling`/`unreachable` vocabulary are a read-only diagnostic pass, not part of the +`GC` protocol itself. diff --git a/docs/en/antalya/cas/architecture/index.md b/docs/en/antalya/cas/architecture/index.md new file mode 100644 index 000000000000..15a9516c9ae7 --- /dev/null +++ b/docs/en/antalya/cas/architecture/index.md @@ -0,0 +1,114 @@ +--- +description: 'What CAS is, the Git-analogy mental model, the object model, and the safety invariants a reviewer should hold every CAS protocol against.' +sidebar_label: 'Architecture overview' +sidebar_position: 1 +slug: /antalya/cas/architecture/ +title: 'CAS Architecture — Overview' +doc_type: 'reference' +--- + +# CAS architecture — overview {#overview} + +`CAS` ("content-addressed storage") is a `MetadataStorage` back-end for object-storage disks +(`metadata_type = cas`) that stores every `MergeTree` part file once, addressed by the hash of +its content. Many servers share one object-storage pool with no byte duplication, no zero-copy +bookkeeping in `Keeper`, no per-replica local-disk reference state that grows with data volume, +and no mutable per-blob refcount. + +It is still experimental — that is deliberate, not a caveat to apologize for. Pre-release means +the format can still change cheaply, with zero compatibility scaffolding, and the design can +still be iterated on invariants rather than migrations. The bet: all you need underneath is a +good S3 bucket. No external coordinator, no metadata service, no Keeper state proportional to +data — the pool is self-describing, and everything CAS needs to agree on (refs, leases, GC +leadership, fencing tokens) is an object in the bucket. + +This page is the entry point of a 4-page set: it gives the mental model. Deeper detail on +storage layout, the write/read protocols, and GC lives in the other three pages. + +## The Git analogy {#git-analogy} + +The fastest way to load the model is Git, which most readers already carry: + +| Git | CAS | +|---|---| +| blob (file content by hash) | **blob** — one part file's bytes, keyed by content hash | +| tree (directory listing) | **part manifest** — the immutable file list of one part | +| ref (`refs/heads/main`) | **ref** — `part name → manifest id`, the only mutable state | +| `gc` / reachability | **GC round** — an in-degree fold over refs → manifests → blobs | + +Where the analogy breaks: Git's objects are locally addressed and GC runs against a single +repository with no concurrent writers; CAS objects are addressed inside a shared, multi-writer +object-storage pool, and its GC round has to reason about ambiguity (crashed writers, +in-flight precommits, eventually-consistent `LIST`) that a local Git repository never faces. +Git also has no equivalent of a CAS ref's precommit state — a CAS ref transition is durable +before the blob it names is guaranteed reachable, never the other way round. + +## The object model {#object-model} + +Four durable object kinds exist in a pool: one mutable (the ref), three immutable +(part manifest, blob, and a blob's condemnation-marker sidecar). + +```mermaid +graph TD + R["Ref: part name maps to manifest id"] + M["Part manifest: file list of one part"] + B["Blob: one part file's bytes, keyed by content hash"] + BM["Blob meta: condemnation marker sidecar"] + + R -->|names| M + M -->|entry references| B + B -.->|sidecar| BM +``` + +**The reachability rule, stated once:** a blob is live if and only if some live manifest names +it, and a manifest is live if and only if some ref — committed or precommitted — names it. `GC` +computes exactly this and nothing else. + +## Safety invariants {#safety-invariants} + +The full numbered list lives in the CAS agent guide; this is the reader-facing summary of the +substance: + +| Invariant | What it means | +|---|---| +| No silent data loss | No path may delete an object a committed reference still names | +| Revival is re-upload only | A condemned blob is never revived by copying it — only by re-uploading the original bytes under a fresh identity | +| Exact-token deletes | Every delete names the exact object incarnation it removes, never "the object at this key" | +| `TOKEN ⟹ CONTENT` | A repeated write token implies unchanged bytes — the backend must never let a token be reused over different content | +| Fail closed on ambiguity | An operation that may have landed is never treated as one that did not | +| One content-delete site | Exactly one place in the whole codebase ever deletes a blob body, gated on a previously published `GC` round | +| `GC` never invents history | Cleaning up an abandoned write is the writer's job, not `GC`'s | +| Over-count only | A lost or duplicated `GC` fold can only delay a reclaim, never bring one forward | +| No dangle / no loss / no return | A live ref always resolves through present objects; a delete requires proven unreachability at an exact token; a retired object identity is never valid again (though the same logical key can return under a new token) | + +## Positioning: shared-nothing, not shared-state {#positioning} + +Each server owns the catalog rows under its own identity and writes only its own state objects +— that part is shared-nothing, same as `ReplicatedMergeTree` today. What CAS adds is a single +**shared** resource: the blob content space, addressed purely by content hash, which is +write-once and conflict-free by construction — two servers writing the same content write the +same key with the same bytes, so there is nothing to reconcile. The only mutual exclusion CAS +needs anywhere is a conditional write (create-if-absent, or compare-and-swap on a token) against +a single object. + +That is deliberately not a coordinator or a serializable metadata service: there is no external +coordinator, and no `ZooKeeper`/`Keeper` usage inside the pool protocol itself. `Keeper` stays +exactly where `ReplicatedMergeTree` already used it — replication log and part-set consensus — +and its load does not grow with pool size, because the pool's own bookkeeping never touches it. + +## The subsystem pages {#subsystem-pages} + +| Page | Covers | +|---|---| +| [Storage layout](/antalya/cas/architecture/storage-layout) | Every S3 key shape, the object envelope, codecs, a worked example tree | +| [Namespaces](/antalya/cas/architecture/namespaces) | Namespaces, `life_id`, the catalog, and their lifetime | +| [Blob protocol](/antalya/cas/architecture/blob-protocol) | Conditional writes, deduplication, the writer-vs-GC race | +| [Part lifecycle](/antalya/cas/architecture/part-lifecycle) | Build, precommit, upload, promote; crash points and their cleaners | +| [Manifests and refs](/antalya/cas/architecture/manifests-and-refs) | Part manifests and the ref machinery: publish, fold, recovery | +| [Mounts and leases](/antalya/cas/architecture/mounts-and-leases) | Server identity, the owner claim, the mount lease, fencing | +| [Replication](/antalya/cas/architecture/replication) | Fetch-by-relink between replicas sharing one pool | +| [Read path](/antalya/cas/architecture/read-path) | Ref resolution, manifest reads, ranged blob reads, the caches | +| [Garbage collection](/antalya/cas/architecture/garbage-collection) | Leadership, the round, sharding, cost, observability | +| [Backend abstraction](/antalya/cas/architecture/backend) | Provider dialects for conditional writes, the capability probe | +| [Correctness](/antalya/cas/architecture/correctness) | TLA+ models, counterexamples, soak methodology, test coverage | +| [Design history](/antalya/cas/architecture/design-history) | The rejected designs and the major pivots | diff --git a/docs/en/antalya/cas/architecture/manifests-and-refs.md b/docs/en/antalya/cas/architecture/manifests-and-refs.md new file mode 100644 index 000000000000..df86d82e575d --- /dev/null +++ b/docs/en/antalya/cas/architecture/manifests-and-refs.md @@ -0,0 +1,292 @@ +--- +description: 'Part manifest structure and lifecycle, the ref table as the only mutable state in a CAS pool, the publish protocol, and the orphan-manifest sweep.' +sidebar_label: 'Manifests and refs' +sidebar_position: 5 +slug: /antalya/cas/architecture/manifests-and-refs +title: 'CAS Architecture — Manifests and Refs' +doc_type: 'reference' +--- + +# CAS architecture — manifests and refs {#manifests-and-refs} + +A part manifest is the immutable file list of one `MergeTree` part; a ref is the mutable pointer +from a part name to the manifest that currently backs it. Together they are the two object kinds +that make a CAS pool's state machine: manifests never change, refs are the only place anything +moves. This page covers what a manifest contains, how a manifest becomes reachable or becomes an +orphan, how a ref mutation is published durably, and how a mounted server recovers a ref table +after a crash or a fresh mount. The write/promote sequence that drives these primitives is on the +[part-lifecycle page](/antalya/cas/architecture/part-lifecycle); how `GC` folds ref history into +blob liveness is on the [garbage-collection page](/antalya/cas/architecture/garbage-collection). + +## Part manifests {#part-manifests} + +A manifest (`cas_part_manifest`, `Formats/CasPartManifestFormat.h`) has four top-level fields: +`ref` (its own id, repeated in the body for fail-closed validation), `root_namespace_id` (the +owning namespace, likewise repeated), `payload_digest` (integrity/debug only — never a key, never +a dedup input, never a `GC` edge), and `entries` — strictly ascending by path after decode. Each +entry is `{path, placement, BlobRef, blob_size, inline_bytes}`; the hash algorithm travels **per +entry**, so one manifest may legitimately mix algorithms if the pool has more than one enabled. + +A manifest deliberately holds **no** offsets, no packed-file support, no projections field, no +codec info, no parent-manifest link, no source edges, and no incarnation token. One blob is one +file's bytes; a read window is `{blobKey, blob_header_len, blob_size}`. A projection is an +ordinary entry whose path has a `.proj` component. The incarnation token is the backend `ETag` +observed by a `HEAD`, never stored in the manifest. + +**The manifest id is neither a content hash nor random.** It is +`ManifestRef = {writer_epoch, build_sequence, manifest_ordinal}` — durable writer epoch times +monotone per-incarnation build sequence times monotone per-build ordinal — which gives "no +manifest id reuse" by construction with no randomness needed. The `GC`-level identity is the pair +`ManifestId = (RootNamespace, ManifestRef)`; two namespaces may legally carry the same +`ManifestRef`. + +Backpressure caps are enforced before the body is written (`Pool/CasPartWriteTxn.cpp`): + +| Cap | Limit | +|---|---| +| Entries per manifest | 1 048 576 | +| Encoded manifest text | 256 MiB | +| Total inline bytes | 16 MiB | +| Largest single inline entry | 1 MiB | + +A manifest is written once with a conditional create (`putIfAbsent`) and **never rewritten**. +A different object at that key would be an id collision and is `CORRUPTED_DATA`, fail-closed, +before any owner transition names it. Rewriting a part therefore writes a **new** manifest over +the **same** blobs and moves the ref in one ref-log record — a repoint, covered in full on the +[part-lifecycle page](/antalya/cas/architecture/part-lifecycle#repoint). + +## Manifest lifecycle and the orphan sweep {#manifest-lifecycle} + +```mermaid +stateDiagram-v2 + [*] --> Staged: stageManifest, body PUT write-once + Staged --> PrecommitOwned: precommitAdd, ref-log OwnerTransition, plus-one edges on fold + PrecommitOwned --> Committed: promote, Precommit to Committed, no edge, net zero + Committed --> OwnerRemoved: drop or repoint or namespace removal, minus-one edges + OwnerRemoved --> [*]: GC deletes the body after the decrements are sealed + + Staged --> OrphanA: writer died before precommitAdd + OrphanA --> [*]: writer best-effort delete, else the orphan sweep + + PrecommitOwned --> DanglingPrecommit: writer died before promote + DanglingPrecommit --> OwnerRemoved: binding removed by abandon or a successor stale-precommit sweep +``` + +Two disjoint failure classes matter here: + +- **Pre-precommit orphan.** The body exists but no ref-log record ever named it. It contributes no + edges and nobody protects it — this is exactly what the orphan sweep below reclaims. +- **Dangling precommit.** The transaction died between `precommitAdd` and `promote`. Nothing wakes + it up on its own: a `PartWriteTxn` is never persisted. The binding must be removed by a ref-log + transaction — either the live writer's own `abandon`, or a fenced successor's stale-precommit + sweep, which removes precommits whose `manifest_ref.writer_epoch < live_epoch` + (`Pool/CasRefLedger.cpp`). Only after that minus-one folds does `GC` delete the body, on the + ordinary owner-removal path. + +The writer's own best-effort cleanup deliberately **skips** the precommit target once a precommit +was even attempted — including an uncertain outcome — because deleting a body that turns out to be +a live precommit would clamp `GC`'s fold barrier forever. + +### The orphan-manifest sweep {#orphan-sweep} + +The cursor-paced, budgeted sweep has two stages (`Gc/CasOrphanManifestSweep.cpp`). During fold +planning, it freezes candidates with exact `GET`s, opens and decodes their bodies, and derives the +state changes needed to make deletion safe. Only after the round `CAS` adopts those changes does +phase 18 perform physical deletion. Eligibility comes **exclusively** from the durable watermark +in the mount lease — there is no age threshold and no time-based grace period anywhere in this +protocol. No mount lease for the `server_root_id` means no deletion authority means nothing is +swept for that root. + +```mermaid +flowchart TD + A["LIST one page of cas/manifests/
freeze candidates with exact GET"] --> B{"build-prefix eligible?
durable watermark fact only"} + B -->|"epoch less than lease epoch"| ELIG["eligible, old-epoch debris"] + B -->|"same epoch, min_active clears build_seq"| ELIG + B -->|"no lease, or epoch ahead, or build may be live"| SKIP["skip"] + ELIG --> C["protection view: committed manifests
plus live precommits
plus manifests with an unfolded minus-one"] + C -->|"key protected"| SKIP2["skip"] + C -->|"not protected"| D{"open and decode
frozen body"} + D -->|"cannot open or decode"| U["retain; skipped++ and undecodable++
log exact key; advance decision cursor
continue to later candidates"] + D -->|decoded| I{"body ref and namespace
match key?"} + I -->|no| BAD["CORRUPTED_DATA
fail-closed round error"] + I -->|yes| R["derive exact blob-source
retirement records"] + R --> F["round CAS adopts retirements
and advanced cursor"] + F --> X["phase 18: deleteExact
key and frozen token"] + X -->|Deleted| E["emit ManifestDelete audit event"] + X -->|NotFound| NF["spared"] + X -->|"token ABA"| ABA["retain replacement;
CORRUPTED_DATA round error"] +``` + +The protection view is built from the **same complete replay** that writer recovery uses, and a +namespace whose view fails to build is added to an errored set with **all** of its deletions +skipped — an empty owner set is never substituted for a failed one. A body that cannot be opened +or decoded is likewise retained: it increments both `skipped` and `undecodable`, advances the page +decision cursor, logs the exact key, and does not prevent later candidates from being examined. It +is not repaired or deleted, and remains visible to `ca-fsck` as an unreachable object. A decoded +body whose ref or namespace does not match its key instead fails the round with `CORRUPTED_DATA`. + +For every legal nomination, the sweep derives exact source-retirement records for the body's blob +entries. The fold places those retirements in the new generation, and the round `CAS` adopts both +that generation and the advanced sweep cursor before any manifest body is deleted. This is the +same adopt-before-delete safety ordering as owner removal; here the retirements come directly from +the frozen body rather than from a ref-log minus-one. Phase 18 then calls `deleteExact` with the +frozen token. Every outcome emits a `ManifestDelete` audit event: `Deleted` records a physical +deletion, `NotFound` is spared, and a token ABA retains the replacement and fails the round with +`CORRUPTED_DATA`. + +Operators can find rounds that retained undecodable bodies through +`system.cas_gc_log` and the `phase_metrics['undecodable']` count on `Phase` rows for +`orphan_sweep`: + +```sql +SELECT event_time, round_id, + phase_metrics['skipped'] AS skipped, + phase_metrics['undecodable'] AS undecodable +FROM system.cas_gc_log +WHERE event_type = 'Phase' + AND phase = 'orphan_sweep' + AND phase_metrics['undecodable'] > 0 +ORDER BY event_time DESC; +``` + +## Source edges: how a manifest makes blobs live {#source-edges} + +Blob liveness is a **set of source edges**, not a counter (`Gc/CasBlobInDegree.h`) — which is what +makes `GC`'s fold idempotent. An edge id is `sourceEdgeId(ManifestId, path)`, a deterministic hash +over the namespace, epoch, build sequence, ordinal and path — an edge *identity*, deliberately not +a content hash and not reconstructable. + +Edges are never written at manifest-write time. They materialize only when `GC` folds a ref-log +transaction that changes ownership: add-precommit means `+1` per blob entry; either removal means +`-1`; **promote means no edge at all**, because the manifest never loses an owner, so it is net +zero. Inline entries produce no edges — they have no separate object to reclaim. + +## The ref table {#ref-table} + +A ref is the only mutable state in the whole system, so this is where the concurrency design is +concentrated. + +- **Name** — a canonical clean relative path, in practice the part directory name with an optional + `detached/` or `moving/` prefix. +- **Value** — `{ref_name, ManifestRef, published_at_ms}`. There is **no** token/`ETag` in a ref + row; the cross-server "confirm token" is the text form `epoch:build:ordinal`. +- **Scope** — one ref table per `RootNamespace`, i.e. per table per server root. +- **Ownership slots** — a `ManifestRef` has at most one owner across the table, in one of two + slots: `Committed` or `Precommit`. Precommits are keyed by the pair `(ref_name, manifest_ref)`, + so several in-flight builds may legitimately contend for one ref name. + +In memory, `RefTableState` holds a copy-on-write map of committed rows, a set of precommits, an +ownership index enforcing the one-owner rule, a lifecycle (`Live`/`Removed`), the greatest applied +transaction id, and byte-size counters used for admission. Copying a state is a refcount bump, so +a flush's trial and candidate copies cost proportional to touched rows, not the whole table. +Network I/O is never performed while holding the state lock, so a reader sees either a whole +transaction or none of it. + +Two immutable object kinds carry the durable form under `cas/ns/stream//` (see the +[storage-layout key table](/antalya/cas/architecture/storage-layout#key-table)): a log object +holds exactly one transaction, `{namespace, txn_id, ops[]}`; a snapshot object holds one live table image +— sorted committed rows plus precommits. Mutable/path-addressed state lives separately under +`cas/ns/state//`: the per-life `_ckpt` checkpoint and any namespace-owned `_files/`. + +`RefTxnId = {writer_epoch, ref_sequence}` renders as two fixed-width hex fields, so lexical key +order equals tuple order. Ids are per-namespace and contiguous: within one `(namespace, +writer_epoch)` they run `1, 2, 3, …` with no holes, and a new mount epoch restarts the sequence at +`1`. A hole is therefore corruption, not an allocation artifact, and a non-successor id is rejected +as `CORRUPTED_DATA`. + +The op vocabulary is deliberately tiny: `NamespaceBirth`, `OwnerTransition{old?, new?}`, +`SetPublishedAt`, `RemoveNamespace`. There are exactly four legal `OwnerTransition` shapes — add +precommit, remove precommit, remove committed, and promote — enumerated identically by the state +machine and by `GC`'s edge extractor, so the two readers of the format cannot drift. + +Logs are pure conditional creates on write-once keys. There is no append-to-object and no +`CAS`-swapped mutable pointer anywhere in the ref lane. The writer never deletes ref objects; only +`GC` does, once coverage and a live snapshot both make a log safe to remove. + +Snapshots publish in the background, best-effort, one in flight per table, when the tail exceeds a +log-count or log-byte threshold. + +## Publishing a ref mutation {#publish-protocol} + +All mutations funnel through one flat-combining lane, `CasRefLedger::appendRefOps`. A single flush +carves a batch out of the queue and commits it as one or more transactions. + +```mermaid +flowchart TD + Q["appendRefOps enqueues ops"] --> REC["ensure the table is recovered"] + REC --> FEN{"mount fence still live?"} + FEN -->|no| FAIL0["fail the whole carved queue, retry error"] + FEN -->|yes| W{"outstanding wedge?"} + W -->|yes| WR["resolve the wedge by its exact key first"] + WR -->|resolved durable| INST0["install candidate, clear wedge"] + WR -->|still unresolved| FAIL1["fail the queue, stay wedged, never allocate a new id"] + W -->|no| CARVE["two-phase carve: plan may throw, publish never throws"] + CARVE --> VAL["per-item validation: caps, shape, byte budget
a failing item fails alone"] + VAL --> PREP["build candidate state and the complete wedge before the PUT"] + PREP --> PUT["putIfAbsent the ref-log key"] + PUT -->|Committed| OK["allocation-free install: swap state, bump counters, complete waiters"] + PUT -->|DefiniteFailure| GAP["fail survivors, id not consumed"] + PUT -->|"Unresolved, provably nothing sent"| NOSEND["do not wedge"] + PUT -->|"Unresolved, otherwise"| WEDGE["install the prepared wedge, survivors fail Uncertain"] + OK --> SNAP["maybe schedule a snapshot publish"] +``` + +The **wedge** is the mechanism that makes fail-closed ambiguity concrete: at most one per table, +recording the single conditional `PUT` whose outcome is unknown, complete with the key and the +sealed bytes. The next flush must resolve *that exact key* before it may allocate a new transaction +id — an unresolved write can never silently become a gap, and the ledger never double-publishes. + +Crash points: between the `PUT` and the install, the object is durable and unapplied — the next +mount's recovery replays it. Between a precommit and its promote, a dangling precommit is reclaimed +by the successor's stale-precommit sweep, described above. + +## Recovery {#recovery} + +Recovery is lazy per table, on first touch (`Pool/CasRefLedger.cpp`), and reads only named, +authoritative objects — there is no `LIST` anywhere in this path: + +1. **Exact `GET` of `_ckpt`.** The durable checkpoint is the sole source of the recovery grounding: + `chooseRecoveryGrounding` derives the base (a snapshot id, or genesis if there is none) and the + exact transaction to walk from purely from the checkpoint's own fields + (`committed_through`/`checkpoint_snapshot_id`/`life_epoch`) — recovery never enumerates its own + stream to find them. +2. If the grounding names a snapshot, `GET` and decode it as the replay base. +3. Walk forward by exact key from there, one transaction resident at a time: `GET` + `cas/ns/stream//-`, decode, apply, discard, advance to the next + arithmetic id. Every key this walk touches is a dense, deterministic successor of the last — + never a listed or guessed one. +4. **Absence is a decision point, not an error.** Finding a slot empty is either the live epoch's + stream legitimately ending there, or — for a dead predecessor epoch — the exact slot where its + closing `EpochSeal` must be written before the table may be trusted; the two cases are + distinguished by whether the epoch being walked is still live, not by retrying a listing. +5. Recovery may itself advance `_ckpt` as it replays, each time via a conditional write against the + checkpoint it last read; the write is re-verified with a fresh exact `GET` afterward, and a + concurrent winner's farther frontier is honored by restarting from that newer checkpoint rather + than trusting the write blindly. +6. Transient network errors retry the whole attempt with capped backoff; corruption and logic + errors fail fast. + +For a mounted writer the recovered in-memory table is authoritative for reads of its own +namespaces — there is no other writer of that namespace. S3 is authoritative for durability: +in-memory state advances only after a durable `PUT`, and a caller's `appendRefOps` returns only +after the durable install. In-flight precommits are visible only through the precommit set, never +through an ordinary ref resolve. + +Two cross-process readers see a different, colder view, but only at the discovery boundary: `GC` +and `ca-fsck` `LIST` once to discover which namespaces exist, staleness-bounded by whatever was +durable at `LIST` time, so a namespace born after that `LIST` is invisible to this pass. Within +each discovered namespace, the replay itself is not `LIST`-driven — it is the same exact-`GET`, +`_ckpt`-grounded arithmetic walk described above, just called from a caller-supplied catalog entry +instead of a live mount. The relink-confirm handshake (see the +[replication page](/antalya/cas/architecture/replication#relink-gates)) does zero object-store I/O +and answers `Yes` only against the resident, warm, fence-live in-memory table — `No` is not proof +of the negative, only `Yes` is fence-gated. + +## Namespace removal {#namespace-removal} + +Namespace removal has no physical-empty handshake. The writer changes the catalog row from `Live` +to `Removing`, appends the exact removals plus `RemoveNamespace`, and deletes nothing itself. The +`GC` fold attaches cleanup evidence to that life row; a later invocation's pre-fold drain exact-CAS +-deletes the matching `Removing` catalog row before any successor plan publishes. A perpetual +namespace janitor and the orphan-manifest sweep reclaim physical debris independently — a same-name +birth waits only for the catalog row to disappear, never for physical emptiness. diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md new file mode 100644 index 000000000000..d778756ce323 --- /dev/null +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -0,0 +1,252 @@ +--- +description: 'How a CAS server establishes identity, claims its mount slot, and holds a renewable lease that fences stale writers out of the pool.' +sidebar_label: 'Mounts and leases' +sidebar_position: 4 +slug: /antalya/cas/architecture/mounts-and-leases +title: 'CAS Architecture — Mounts and Leases' +doc_type: 'reference' +--- + +Page 4 of 4 in the CAS architecture set. Covers server identity, the mount lease that fences +writers, and the server-scoped control-plane objects. No external coordinator is involved: there +is no ZooKeeper/Keeper client anywhere in this protocol — `MountLeaseKeeper` is a local lease +*renewer*, not a Keeper client. + +## `cas_server_root_id` — the identity {#server-root-id} + +Every content-addressed disk must be configured with an explicit `cas_server_root_id`. It is +validated and immutable, and deliberately **not** derived from `ServerUUID` — two replicas can +otherwise regenerate the same `ServerUUID` from a wiped local state directory, which must not +silently steal an existing identity. + +Validation (`validateServerRootId`, `Pool/CasServerRoot.h`) is fail-closed `BAD_ARGUMENTS`, no +sanitizing fallback: non-empty, at most 255 bytes, no empty/`.`/`..` path segment, no `_files` or +`_manifests` segment. + +It roots four subtrees and owns catalog names at or below ``: + +| Subtree | Contents | +|---|---| +| `gc/server-roots//` | owner, epoch, mount — the three control-plane objects below | +| `roots//` | loose mountpoint objects, no namespace/catalog association | +| `cas/manifests//` | part manifests | +| `staging//` | S3-staging debris, outside every GC `LIST`, reclaimed only by this server's next mount | + +`blobs/` is **not** under the `server_root_id` — content is pool-global, which is what makes cross-server +dedup work. Ref/namespace keys are also deliberately opaque and do not embed the `server_root_id`. + +Each replica sharing a backend endpoint must use a distinct `cas_server_root_id`; omitting the setting +is a startup error. + +## The owner claim {#owner-claim} + +`claimOwnerOrThrow` binds `server_root_id` ↔ `server_uuid` **permanently**. The owner object is never deleted +and never reassigned — decommission only tombstones it in place. + +| Observed at `gc/server-roots//owner` | Action | +|---|---| +| present, same `server_uuid`, not tombstoned | proceed | +| present, `retired_at_ms` set | `CORRUPTED_DATA` — explicitly decommissioned, refuses to resume | +| present, different `server_uuid` | `CORRUPTED_DATA` — names the regenerated-uuid-file cause | +| absent, subtree provably empty | `putIfAbsent` the owner (claim) | +| absent, subtree non-empty | `CORRUPTED_DATA` — identity lost over existing data | +| lost the `putIfAbsent` race | re-read; equal uuid proceeds, else `CORRUPTED_DATA` | + +"Provably empty" requires both an authoritative decoded catalog naming no life owned by `server_root_id` and +a 1-key `LIST` probe finding nothing under `cas/manifests//` or `roots//`. + +Two failure modes this closes: + +- A **second server with a different `server_uuid`** is refused at this gate and can never take + over, regardless of lease expiry. +- A **same-uuid live twin** (two processes sharing one uuid file and `server_root_id`) is caught separately, by + the mount claim's token-stability observation, and aborts with an operator-facing message rather + than corrupting the pool. + +## The mount lease {#mount-lease} + +One object, `gc/server-roots//mount`, carries **both** the liveness lease and the build +watermark — there is no separate watermark object. `MountLease` fields: `server_uuid`, +`writer_epoch`, `write_attempt_id`, `hostname`, `pid`, `started_at_ms`, renewal `seq`, +`expires_at_ms`, `min_active` (the build-watermark floor), and `gc_fenced`. + +- **Logical renewal identity.** Each holder-originated body has a fresh nonzero + `write_attempt_id`. One logical renewal fixes one immutable `(key, bytes, expected token, + write_attempt_id)` tuple before I/O. Every physical retry repeats it byte-for-byte; a later GC + fence preserves the observed ID, while reclaim and successor bodies mint new IDs. +- **Resolve before retry.** A transient or ambiguous conditional `PUT` is followed by one exact + `GET`. The keeper adopts the result only when the complete body, including `write_attempt_id`, + equals its immutable request. If the predecessor token is still current, another identical `PUT` + may follow bounded backoff. A same-pair twin, GC-fenced body, successor, foreign holder, or absent + body is never treated as this renewal. +- **Absolute deadline.** Renewal uses `CLOCK_BOOTTIME`, not `CLOCK_MONOTONIC`, so a VM resumed from + suspend correctly observes itself expired. Its absolute deadline is the minimum of the existing + request-operation budget and the last confirmed lease deadline minus the safety margin. The + controller checks that one configured attempt still fits before each backend `PUT` or resolving + `GET`, after each interruptible backoff, and before accepting success. A retry, `GET`, response + timestamp, or wall-clock step never extends authority. +- **Cadence.** The runtime normally starts a logical renewal every `mount_renew_period` (default + 10 s), with TTL `mount_lease_ttl_ms` (default 30 s, TTL/3 renewal ratio). The next beat is anchored + at the committed body's pre-I/O BOOTTIME start. A slow recovery therefore causes an immediate + catch-up beat when the nominal cadence has elapsed; it does not wait a fresh full period after the + response. +- **Per-write recheck.** Every durable write or delete captures the fence generation at admission + and rechecks it immediately before the object-store call and on every conditional retry. Reads + are not gated. +- **Request-budget admission.** `refAppendFenceOk` refuses to *start* a ref-log attempt unless + `attempt_timeout + safety_margin` fits inside the remaining lease, rejecting with + `BAD_ARGUMENTS` at request-admission time rather than mid-flight. + +**Losing the lease is neither read-only mode nor a process abort.** `MountLeaseKeeper` is a +synchronous durable-slot state machine. A committed result advances its token, sequence, confirmed +BOOTTIME deadline, and cadence anchor. Any admitted deterministic failure, confirmed conflict, or +ambiguity left at the deadline/attempt limit moves it to `RenewalTerminal`; it cannot mint another +body or publish a clean farewell. Owner cancellation before any request is the only +`NotAttempted` result and leaves clean release possible. Cancellation after a request was sent is +terminal because that request may still land. + +After the keeper call returns, `CasMountRuntime` consumes the result. A terminal result trips the +local fence (latches `lost`, bumps the fence generation, moves the in-process runtime to +`TransientNotLive`) and latches one self-remount generation. A confirmed foreign/successor or +same-pair conflict remains a typed fail-closed error; it is never adopted. A real fence still costs +only an epoch: recovery reclaims with a fresh one, bounded at three whole-chain attempts. This is the +general CAS posture: doubt about the source fails closed, while transport ambiguity may retry only +inside authority already proved by the last confirmed lease. + +GC's own view of a dead server is symmetric and clock-skew-immune: a slot becomes fence-eligible +only after the leader observes the *same* renewal token hold stable, on its own monotonic clock, +for `TTL + TTL/20 + cadence` — the identical formula a re-mounting server uses to wait out a +predecessor. The stamped `expires_at_ms` never participates in that decision; wall-clock `now` is +audit-only. + +## The two monotone counters {#counters} + +| Counter | Storage | Scope | Protects against | +|---|---|---|---| +| `writer_epoch` | durable, `gc/server-roots//epoch` (`ServerEpoch::next_writer_epoch`, CAS-bumped by `allocateWriterEpoch`) | across crashes and restarts | a same-`(uuid, epoch)` twin: a present mount under a normal claim attempt is `CORRUPTED_DATA` | +| `build_seq` | in-memory only, `CasMountRuntime::next_build_seq`, reset to 1 on every process start | one process incarnation | orders builds *within* an epoch; combined with `writer_epoch` it gives GC a total order | + +The absent-epoch branch of `allocateWriterEpoch` is deliberately paranoid: absent with a +non-empty subtree is `CORRUPTED_DATA` (reset hazard); absent with an empty subtree decides by an +authoritative probe, never by plain-`get` absence, because a transport fault must not be flattened +into "not found". + +Global build ordering is the **pair** `(writer_epoch, build_seq)` compared lexicographically — the +exact comparison GC uses for eligibility. The durable authority for both is the mount object +itself: no mount means no deletion authority means nothing is swept. `min_active`, the oldest +in-flight `build_seq`, rides in the same mount object as the watermark floor; `UINT64_MAX` in +`min_active` is the farewell/retired sentinel, not a real build. + +## Mount claim outcomes {#claim-outcomes} + +The implementation does not expose a single named durable-slot enum; `claimMount` instead returns +a `MountClaimResult::Kind` together with a `MountPriorState` describing which certificate of death +(if any) justified a reclaim: + +| `Kind` | Meaning | +|---|---| +| `Claimed` | fresh claim (absent slot), same-`(uuid, epoch)` refresh, or a certified reclaim | +| `LiveDoubleStart` | same `server_uuid`, different `writer_epoch`, and no certificate of death yet — a live twin, wait it out | +| `ForeignOwner` | different `server_uuid` — refused unconditionally | +| `FencedSelf` | same `(uuid, epoch)`, but `gc_fenced` — terminal for *this* epoch; the caller must mint a fresh one | + +| `MountPriorState` | Certificate that justified the reclaim | +|---|---| +| `None` | no reclaim needed (fresh claim or same-epoch refresh) | +| `Clean` | the predecessor's own graceful farewell (`min_active == UINT64_MAX`) | +| `Fenced` | GC's own threshold-gated fence-out (`gc_fenced`) | +| `UncleanObserved` | this claimant's own token-stability observation held for the full `TTL + drift` window | + +## Behavioral mount-slot model {#mount-state-machines} + +Two coupled state pictures. Neither is a literal source enum — the durable slot is derived from +the claim outcomes above and is shown here as behavior, not as a type in the code: + +```mermaid +stateDiagram-v2 + [*] --> Absent + Absent --> Live: claimMount putIfAbsent, seq=1 + Live --> Live: keeper beat, putOverwrite seq+1 + Live --> Fenced: GC observes a stable token past threshold, gc_fenced=1, body preserved + Live --> Terminated: certified drain, terminal farewell (expires_at=now, min_active=MAX) + Fenced --> Live: same-uuid claim with a fresh writer_epoch, instant reclaim + Terminated --> Live: same-uuid claim with a fresh writer_epoch, instant reclaim + Live --> Live: same-uuid claim, proven-dead token via UncleanObserved + Fenced --> Fenced: same uuid and epoch claim, FencedSelf, no write + Live --> Absent: decommission tail, mount then epoch then owner tombstone + Terminated --> [*] +``` + +The in-process `PoolLifecycle` runtime, by contrast, is a literal enum (`CasMountRuntime.h`): + +```mermaid +stateDiagram-v2 + [*] --> Live: Pool constructed, fence unarmed + Live --> Live: mountWritable arms the fence + Live --> TransientNotLive: renewal failure, tripMountLost, lost=true + TransientNotLive --> Live: self-remount succeeds with a fresh epoch + TransientNotLive --> TransientNotLive: probe inconclusive, retry with backoff + TransientNotLive --> IdentityLost: pool meta and owner both authoritatively absent + TransientNotLive --> VanishedReplaced: foreign pool_id observed + Live --> VanishedForgotten: SYSTEM CAS FORGET + IdentityLost --> [*] + VanishedReplaced --> [*] + VanishedForgotten --> [*] +``` + +`IdentityLost`, `VanishedReplaced` and `VanishedForgotten` are terminal and absorbing: the remount +and GC threads self-exit, and there is deliberately no auto-revive — an identity disappearing +under a live mount is an operator-level event. + +## Mount, unmount, crash {#mount-lifecycle} + +**Writable open** runs in a strict order: bootstrap-residual proof, capability probe under a +random per-mount prefix, pool-meta create-or-validate, `validateServerRootId`, owner claim, +`allocateWriterEpoch`, mount claim and synchronous keeper start, materialization grace if the +predecessor was unclean (default 30 s), arm the fence, then create and release the runtime-owned +renewal and remount workers before the writable pool becomes externally visible. If the grace period +consumed the TTL, one fresh synchronous renewal re-anchors the deadline before the fence is armed. +Failure to construct either worker joins the partial pair, closes the fence, and fails the writable +open. No incident path constructs a thread. + +The renewal and remount workers are separate and long-lived under one stable `CasMountRuntime`. +`scheduleRemount` increments a requested-generation latch and wakes the persistent remount worker, +including while an older generation is active. Before keeper replacement, remount requests +`ParkRequested` and waits for the renewal driver to report `Parked`, which proves that no keeper call +is in flight. A successful remount handles only its snapshotted generation; a newer request is +processed before renewal resumes. + +**Clean unmount:** request stop and join both persistent workers, drain the ref lanes, and only if +the drain *certified* quiescence call `MountLeaseKeeper::release` on an `Active` keeper to write the +terminal farewell (`expires_at_ms` already expired, `min_active = UINT64_MAX`). That sentinel is what +lets a successor reclaim instantly. A `RenewalTerminal` keeper, an unresolved ref write, or a sent +renewal ambiguity writes no farewell — an unearned farewell would let a successor start mutating +while a stale conditional request from the predecessor is still in flight. + +**Crash:** no farewell; the renewal token freezes. Recovery is either the same server restarting +and waiting out the token-stability observation, or the GC leader fencing the slot first, after +which any reclaim is instant. + +**Permanent removal** of a dead replica (`Cas::decommissionPoolMember`, driven by +`SYSTEM CAS DROP POOL MEMBER '' FROM DISK ''`) claims the victim's mount slot +as an administrative writer with a no-wait policy (refuses immediately if the member is alive), +drops every ref-bearing namespace, sweeps manifest debris before the slot (deleting the mount +removes the watermark authority), drains staging and roots, then — only with zero warnings — +retires in order: mount, epoch, a final liveness re-check, owner tombstone. + +## `system.cas_mounts` {#mounts-table} + +A read-only view of the same heartbeat-floor computation GC uses: one `LIST` of +`gc/server-roots/` plus one `GET` per slot, zero writes, per-row fail-open (an undecodable body +becomes `state = 'corrupt'`, never an exception). Shows every `server_root_id` in the pool, including peers. + +| Column | Notes | +|---|---| +| `disk`, `server_root_id`, `server_uuid`, `hostname`, `process_id` | identity | +| `writer_epoch`, `renewal_sequence`, `started_at`, `expires_at`, `min_active_build_sequence`, `gc_fenced` | lease state (`DateTime64(3)` columns; the millisecond-integer field names live only in the internal `MountLease` struct and the on-disk body) | +| `state` | one of `live`, `expired`, `terminated`, `fenced`, `corrupt` | +| `is_leader`, `pending_reclaim`, `last_success_age_seconds`, `wedged_namespace_count` | GC health, process-local; **`NULL` on every peer row** — a process-local fact must never be stamped onto another server's row | +| `lifecycle`, `lifecycle_reason`, `lifecycle_detail`, `lifecycle_since` | the SQL surface for the in-process `PoolLifecycle` runtime above: `lifecycle` is one of `live`, `not_live`, `identity_lost`, `vanished`, `constructing`, `shutdown`; `lifecycle_reason` distinguishes `replaced` from `forgotten` for a `vanished` disk; `lifecycle_detail` carries the full diagnosis text; `lifecycle_since` is when the current non-live state began (`NULL` while live) | + +The lifecycle snapshot is I/O-free and ungated, so a not-live, never-started, or vanished disk +still produces a row instead of silently disappearing from the table. diff --git a/docs/en/antalya/cas/architecture/namespaces.md b/docs/en/antalya/cas/architecture/namespaces.md new file mode 100644 index 000000000000..c38b65b0ce17 --- /dev/null +++ b/docs/en/antalya/cas/architecture/namespaces.md @@ -0,0 +1,172 @@ +--- +description: 'What a namespace is, the opaque life_id that qualifies every object it owns, the pool-wide namespace catalog, and a namespace lifetime end to end from first write to catalog-row deletion.' +sidebar_label: 'Namespaces' +sidebar_position: 10 +slug: /antalya/cas/architecture/namespaces +title: 'CAS Architecture — Namespaces' +doc_type: 'reference' +--- + +# CAS architecture — namespaces {#namespaces} + +A namespace (`Cas::RootNamespace`) is the opaque, per-table, per-server-root string under which one +table's part manifests and one ref table live — in practice something the wiring layer composes, +such as `srv1/` for an ordinary table or `srv1/shadow//` for a +`FREEZE` shadow. `CAS` never interprets its contents beyond a shape check (non-empty, no empty or +reserved path segment, at most 512 bytes). The [manifests-and-refs page](/antalya/cas/architecture/manifests-and-refs#ref-table) +covers the ref table one namespace owns; this page covers the namespace itself — its physical +identity, the catalog that is the sole authority for whether it exists, and its full lifetime from +first write to the catalog row's deletion. + +## `life_id`: the physical identity {#life-id} + +A namespace **name** can be reused — a table dropped and recreated keeps the same name. What must +never be reused is the **physical identity** any durable object under that name is keyed by, so +that a stale reader of the old incarnation can never be handed bytes belonging to the new one. That +identity is `life_id`: an opaque, pool-wide, randomly minted 128-bit value (two `thread_local_rng` +draws; retried on the astronomically unlikely zero draw, since `0` is reserved as "never a valid +life"). Internally it is the catalog's `incarnation` field, aliased as `NamespaceLifePhysicalId`; +paired with the namespace name it forms `NamespaceLifeId{ns, incarnation}` +(`Primitives/CasNamespaceLifeId.h`). + +`NamespaceLifeId` deliberately has no default construction and no conversion from a bare namespace +name: code holding only the name cannot address a ref object or a namespace file at all, so +forgetting the life qualifier is a compile error, not a runtime aliasing bug. The only legitimate +source of a `NamespaceLifeId` is `fromCatalogEntry` — reading it off one immutable catalog cut — +which is what makes "this life belongs to this name" a catalog fact rather than something a caller +could reconstruct incorrectly. + +`life_id` renders as 32 fixed-width lowercase hex digits and appears in exactly the two subtrees +that are life-owned (see the [storage-layout key table](/antalya/cas/architecture/storage-layout#key-table)): + +| Subtree | Contents | +|---|---| +| `cas/ns/stream//` | The immutable `_log`/`_snap` ref-transaction history | +| `cas/ns/state//` | The mutable `_ckpt` checkpoint and any namespace-owned `_files/` | + +Part manifests deliberately do **not** carry `life_id` — a manifest already has its own globally +unique identity (`{writer_epoch, build_sequence, manifest_ordinal}` under the server root, see the +[manifests-and-refs page](/antalya/cas/architecture/manifests-and-refs#part-manifests)) and needs no +further qualification. Loose mountpoint objects under `roots/` are outside namespace ownership +altogether and carry no `life_id` either. + +## The namespace catalog {#catalog} + +One pool-wide object, `cas/ref_catalog` (`Layout::refCatalogKey`), is the sole authority for which +namespaces exist. It is read on every fold round and every ref-table recovery, and mutated by one +token-`CAS` write per lifecycle transition. Its entries are canonically ordered by namespace bytes, +strictly ascending, with no duplicate name — both the encoder and the decoder enforce this, so an +out-of-order or duplicate-keyed catalog can never become durable. + +Each row (`CatalogEntry`) carries: + +| Field | Meaning | +|---|---| +| `ns` | The namespace name | +| `state` | `Creating`, `Live`, or `Removing` — see below | +| `incarnation` | The `life_id` for this row, nonzero, never reused | +| `creator` | The mounted writer's fence identity (server root, writer epoch, admission fence generation) that is creating this row — **required** iff `state == Creating`, **forbidden** otherwise | +| `removal_started_round` | The `GC` round observed when removal began — **required** iff `state == Removing`, absent otherwise | + +`NsState`'s three wire values (`Creating = 1`, `Live = 2`, `Removing = 3`) are append-only, exactly +like every other persisted enum in `CAS`: a catalog object written by one build is read by another, +so a value is never renumbered or repurposed. + +```mermaid +stateDiagram-v2 + [*] --> Creating: casAdmitEntry -- fresh random life_id, creator fence stamped + Creating --> Live: completeCreation -- publish genesis _ckpt, then flip, clear creator + Creating --> Creating: a live foreign creator fence -- retry later, no steal + Creating --> Live: reconcileStaleCreator finds the creator fence provably dead,
a fresh opener steals and completes it + Live --> Removing: beginRemoving -- table drop, stamps removal_started_round + Removing --> [*]: GC drains the row once a fold sealed positive cleanup evidence + [*] --> Creating: a fresh createNamespace call, only once the old row is fully absent -- brand new life_id +``` + +A row's own state machine is linear per row (`Creating → Live → Removing → gone`); what makes the +catalog non-linear as a whole is that a stalled `Creating` row can resolve two different ways +depending on whether its creator fence is still alive, and that a name only becomes creatable again +once its prior row is completely gone — both shown above. + +## Lifetime end to end {#lifetime} + +### Creation, on first write {#creation} + +There is no explicit "create namespace" statement; a namespace is born the first time anything +resolves its ref table (`CasRefLedger::resolveNamespaceLife`, bounded at 32 loop attempts). If the +catalog has no row for the name at all, the resolving mount admits a `Creating` entry stamped with +its own creator fence and a freshly minted `life_id` +(`CasRefCatalog::createNamespace` → `casAdmitEntry`). Two more steps make it usable: + +1. **Publish the genesis checkpoint.** The first `_ckpt` ever written for this `life_id` carries + `life_epoch = creator.writer_epoch` — the only writer that will ever know this namespace's + genesis epoch. +2. **Flip to `Live`.** One token-`CAS` moves the row from `Creating` to `Live` and clears `creator`. + +Both steps re-check the resolving mount's own fence before writing, so a mount that lost its lease +mid-creation reports `FencedOut` rather than silently completing. Several openers racing the same +brand-new name all observe "no entry", but only one wins the admit; the rest see `Superseded` and +simply re-read the catalog, landing on the winner's `Creating` row. + +A `Creating` row under a **different** mount's creator fence is not this opener's problem to force: +if that fence is still provably alive, the opener retries later; only once the fence is provably +dead (the same mount-lease terminality check `GC`'s heartbeat floor uses) does +`reconcileStaleCreator` let a fresh opener steal the row onto its own fence and finish the two steps +above itself. + +### Removal {#removal} + +Dropping a table (`DROP TABLE`, and every operation that reduces to it) calls +`CasRefLedger::dropNamespace`. It closes the namespace's local positive-mutation lane first — new +positive writers are refused while the in-flight ones drain — then transitions the catalog row from +`Live` to `Removing` in one token-`CAS` (`beginRemoving`, stamping `removal_started_round` from the +currently observed `GC` round), then appends **one** ref-log transaction that removes every current +committed and precommit binding and ends with a terminal `RemoveNamespace` op. Removal is never +refused by an admission check — Constraint 13 in the catalog's own spec — it always succeeds once +the fence holds. + +Nothing is deleted by the writer at this point. No blob, no manifest, no ref-log object physically +disappears here — only pointers move, exactly like an ordinary [`DROP TABLE`](/antalya/cas/architecture/part-lifecycle#operation-mapping) +on any other ref. + +### What `GC` does with a `Removing` namespace {#gc-and-removal} + +The terminal `RemoveNamespace` transaction is folded like any other ref-log record, during the +[round's fold phases](/antalya/cas/architecture/garbage-collection#the-round). Folding it stamps +positive **cleanup evidence** directly onto that `life_id`'s row in the new fold seal — there is no +physical listing and no `Pending`/`Completed` handshake; the evidence is a pure fact about which +ref-log transaction folded. + +The **next** round's `pre_fold_ref_drain` phase is what actually removes the catalog row: it reads +the just-adopted parent fold seal, and for every `Removing` row whose life carries durable cleanup +evidence, it exact-`CAS`-deletes the catalog entry before that round does anything else. This +two-round shape — evidence sealed in round *n*, catalog row deleted in round *n+1* — is why removal +needs no separate physical-emptiness proof: by the time the row is deleted, a fold has already +proven its ref history is fully drained. + +### What disappears, and when {#what-disappears} + +| Object class | Reclaimed by | When | +|---|---|---| +| Catalog row (`cas/ref_catalog` entry) | `GC` phase 2, `pre_fold_ref_drain` | The round after the fold that sealed cleanup evidence for this life | +| Part manifest bodies | Ordinary owner-removal ([phase 15](/antalya/cas/architecture/garbage-collection#the-round)) for anything that had a committed or precommit binding, the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep) for anything that never got that far | As each owning ref is dropped by the removal transaction itself, independent of the catalog row | +| Blob bodies | The ordinary condemn/graduate/delete pipeline | Whenever the manifests that named them stop being live, same as any other blob | +| Ref stream/state objects (`_log`, `_snap`, `_ckpt`, `_files`) under the dead `life_id` | The perpetual namespace janitor ([phase 16](/antalya/cas/architecture/garbage-collection#the-round)) | Best-effort, one bounded `LIST` page at a time, whenever it next lists a key whose `life_id` a fresh catalog cut no longer names — independent of, and not gated on, catalog-row deletion | + +The janitor is leak-only: it never fails a round, never blocks progress on an unreadable key, and a +crash mid-page simply leaves debris for its next page. + +### Recreate while removing {#recreate-while-removing} + +A fresh `createNamespace` call for a name whose catalog row is still `Live` or `Removing` is +refused outright — internally this is a misuse `LOGICAL_ERROR`, because the higher-level open loop +(`resolveNamespaceLife`) filters that case out first and reports a typed retry-later error instead: +"creation waits for its terminal fold and catalog removal to complete". A caller that keeps +resolving the same name simply keeps retrying until the row is gone. + +Once `pre_fold_ref_drain` has deleted the row, the name is free again, and the very next opener mints +a **brand new**, independently random `life_id` — never the retired one. That is the whole answer to +"what happens on recreate": the old physical identity is never revived, so every key ever written +under it — its `_log`, its `_snap`, its `_ckpt`, its `_files` — stays permanently addressed by a +value nothing will ever mint again, and a reader still holding the old `NamespaceLifeId` observes +only stale-or-absent data, never a byte that belongs to the new incarnation. diff --git a/docs/en/antalya/cas/architecture/part-lifecycle.md b/docs/en/antalya/cas/architecture/part-lifecycle.md new file mode 100644 index 000000000000..aa431b200880 --- /dev/null +++ b/docs/en/antalya/cas/architecture/part-lifecycle.md @@ -0,0 +1,153 @@ +--- +description: 'The part-add protocol from local build through blob upload to promote, its nine crash points and their cleaners, and how each MergeTree operation maps onto it.' +sidebar_label: 'Part lifecycle' +sidebar_position: 6 +slug: /antalya/cas/architecture/part-lifecycle +title: 'CAS Architecture — Part Lifecycle' +doc_type: 'reference' +--- + +# CAS architecture — part lifecycle {#part-lifecycle} + +Publishing a `MergeTree` part on a `CAS` disk is one durable protocol, +`stageManifest → precommitAdd → putBlob → promote`, driven by `Cas::PartWriteTxn` +(`Pool/CasPartWriteTxn.cpp`). This page walks that protocol end to end: local build, the durable +order and why each step is where it is, every crash window and who cleans it up, and how each +`MergeTree`-level operation (insert, merge, mutation, detach, …) maps onto it. Manifest structure +and the ref table it writes into are covered on the +[manifests-and-refs page](/antalya/cas/architecture/manifests-and-refs); the fetch-side protocol +for replicated parts is on the [replication page](/antalya/cas/architecture/replication). + +## The protocol {#protocol} + +```mermaid +sequenceDiagram + autonumber + participant MT as MergeTree + participant TX as CA transaction overlay + participant PW as PartWriteTxn + participant S3 as Object store + + rect rgba(140,190,140,0.12) + Note over MT,S3: Phase A -- local build, nothing durable, nothing visible + MT->>TX: writeFile data.bin + TX->>TX: classify: blob class spills and hashes to scratch or S3 staging + MT->>TX: writeFile count.txt, columns.txt, ... + TX->>TX: buffer small files in memory as inline candidates + MT->>TX: moveDirectory tmp_insert to final name + Note over TX: pure overlay re-key, not a publish + end + + rect rgba(120,160,255,0.12) + Note over MT,S3: Phase B -- publish, per part, serially + MT->>TX: commit + TX->>PW: stageManifest entries + PW->>S3: PUT manifest, write-once, no preliminary HEAD + PW->>S3: append ref-log PRECOMMIT, plus NamespaceBirth if needed + Note over PW: precommit durable, the observe gate opens + TX->>PW: fan out blob uploads, one task per unique BlobRef + par blob 1 + PW->>S3: HEAD, then adopt or unconditional publish + and blob 2 + PW->>S3: ... + end + PW->>PW: merge upload results on the owning thread, one no-throw swap + TX->>PW: promote + PW->>S3: GET and validate the precommit manifest body + PW->>S3: append ref-log txn: retire old committed, Precommit to Committed, SetPublishedAt + Note over PW: commit durable, then retire the build sequence + end +``` + +**Phase A — staging.** The transaction is an eager overlay, not a queue: `writeFile` immediately +classifies the path and either spills bytes to a hashing buffer or holds them in memory as an +inline candidate. Blob-class files stage to local scratch by default. Explicit +`cas_staging_backend = s3` requires native same-store copy at writable mount and stages a complete +`[header][payload]` object. Its first publication after a destination miss may copy that object +verbatim; a condemned or subsequent publication opens the staged payload, retags it, and streams. +The `tmp_ → final` rename is a pure overlay re-key; durable publication happens only in `commit`. + +**Step 1 — `stageManifest`.** Caps (see the +[manifests-and-refs page](/antalya/cas/architecture/manifests-and-refs#part-manifests)) are +checked before the write; the id is minted as `{epoch, build_seq, ordinal++}`; the body goes out +with a conditional create and no preliminary `HEAD`. Blob publication is different; manifests +remain small write-once metadata objects. Both a definite failure and an unresolved +outcome throw retry-later. + +**Step 2 — `precommitAdd`.** The intent — target namespace, final ref name, manifest — is recorded +before the append, because an unresolved append may have landed anyway. One ref-log transaction +adds the precommit binding. A same-name birth is refused with retry-later while the catalog still +says `Removing`; once the predecessor row is absent, creation receives a new opaque life id and +starts its own stream. On return the precommit is durable, and only now may the writer adopt +existing blobs. + +**Step 3 — blob materialization fan-out.** One task per unique `BlobRef`, deterministic dispatch order, one +pre-sized result slot per ref (see the write-path sequence on the +[blob-protocol page](/antalya/cas/architecture/blob-protocol#conditional-write-sequence)). The +calling thread only submits and joins, never occupies a pool slot, so a pool of size one degenerates +to a correct serial run and can never deadlock. The contract is merge-nothing: if any task threw, +nothing is merged and the first error in dispatch order is rethrown. Results are folded into the +dependency set on the owning thread, into a copy, committed by one no-throw swap. Every physical +task starts with blob `HEAD`; a present non-condemned body or a completed publication yields an +explicit `Materialized` proof. A trusted source manifest instead yields `TrustedManifest` without +blob I/O. Pool size is the +server setting `cas_blob_upload_pool_size` (default 16). + +**Step 4 — `promote`.** Reads and revalidates the precommit manifest body once; sets the commit +state to Uncertain before the append — past that point, failure is no longer proof of the negative +— then checks that the precommit is still the live owner and validates explicit dependency proofs. +`Materialized` leaves are edge-protected; `TrustedManifest` leaves are trusted through the durable +source-manifest edge with no per-file `HEAD`; anything else is a `LOGICAL_ERROR`. The +whole thing lands as one ref-log record: optional retirement of the old committed binding, the pure +Precommit-to-Committed owner move, and `SetPublishedAt`. Promotion emits no blob deltas — the +manifest never loses an owner, so it is net zero. + +## Crash points and their cleaners {#crash-points} + +This table is the single best summary of the design's crash-safety story: every row leaks +something recoverable; no row loses data or leaves a dangling reference. + +| # | Crash window | Left behind | Who cleans it | +|---|---|---|---| +| C1 | During staging | Local temp files, or S3 staging objects | Local: unconditional cleanup plus buffer destructor. S3: the mount's own staging sweep at next mount — never deleted on abort | +| C2 | After `stageManifest`, before `precommitAdd` | An unreferenced manifest body | Writer's best-effort exact-token delete; durable backstop is the orphan-manifest sweep | +| C3 | `precommitAdd` returned Unresolved | A possibly-live precommit binding | Intent recorded pre-append; `abandon` appends the exact removal, tolerating absence. The body is never writer-deleted | +| C4 | Between `precommitAdd` and `promote` | A live precommit plus uploaded blobs | No resume path exists. Removed by `abandon`, else by a fenced successor's stale-precommit sweep | +| C5 | Mid blob fan-out | Already-uploaded blobs | Nothing merged; blobs become `GC`-reclaimable debris; the part is not published | +| C6 | `promote` append Unresolved | The ref may or may not be committed | Commit state Uncertain — the relink layer maps this to "retry the whole fetch", never to a byte fetch | +| C7 | A later part throws after earlier parts published | A partial multi-part commit | Precise rollback: drop only the refs this call created, matching the exact manifest — never clobbers a concurrent writer's repoint | +| C8 | Transaction destroyed uncommitted | Open builds | Destructor abandons every build | +| C9 | Namespace dropped mid-build | — | One atomic flag; every further op fails closed at the alive check | + +## The repoint {#repoint} + +Writing into an already-committed part — an `ALTER`-style metadata rewrite, or any standalone +write against a committed source — never mutates the existing manifest. It writes a **new** +manifest over the (possibly partly reused) blob set and moves the ref to it in one ref-log record. +Unchanged columns are adopted by hash through a tokenless evidence dependency with no `HEAD` and no +`GET`; changed columns are fresh uploads. A repoint therefore costs zero bytes moved for the +carry-forward portion of the file set — only the changed content re-uploads. + +## How each MergeTree operation maps {#operation-mapping} + +| Operation | CAS mechanics | +|---|---| +| `INSERT` | The canonical path above. Projections ride the parent part's transaction | +| Merge | Identical for the output part. `.tmp_proj → .proj` is an entry-prefix re-key inside the staged manifest, not a rename | +| Mutation | `createHardLink` per unchanged file: a source staged in *this* transaction copies the entry and its pending-blob record; a **committed** source records a tokenless evidence dependency with no `HEAD` and no `GET`. A mutation is a manifest rewrite where zero bytes move for the carry-forward | +| `ALTER` / metadata rewrites | Standalone writes into a committed part, i.e. a repoint | +| `DROP PART` | `removeDirectory` drops the ref and clears any per-file removal marks — one ref-drop, zero repoints | +| `DROP TABLE` / `DETACHED` / `UNFREEZE` | A namespace or prefixed-ref drop. Blobs are never deleted here — removal is pointer-unlink plus deferred `GC` | +| `RENAME TABLE` | Republishes every ref and verbatim file into the new namespace, then drops the old one. Not atomic across namespaces, but idempotent and re-drivable — true atomicity would need a move journal and is out of scope | +| `FREEZE` / `BACKUP` / `RESTORE` / cross-disk `MOVE` | Each wraps the whole clone in one disk transaction, because a CAS part is one atomic unit | + +`FREEZE` is the one operation that materializes real bytes into a genuinely separate shadow +namespace rather than reusing a table's own ref names — that shadow namespace is a `GC` +reachability root, and `UNFREEZE` releases its refs. + +## Reads while a part is in flight {#in-flight-reads} + +Read-your-writes for a part still inside an open transaction is served by an explicit overlay +rather than by any durable object — `tryGetInFlightStorageObjects`, `tryReadFileInFlight`, +`listInFlightDirectory`. One deliberate subtlety: the bare part directory reports as absent in the +overlay, so cleanup of a deduplication-rejected temporary part does not mistake it for a real part. diff --git a/docs/en/antalya/cas/architecture/read-path.md b/docs/en/antalya/cas/architecture/read-path.md new file mode 100644 index 000000000000..beb2d87d99fb --- /dev/null +++ b/docs/en/antalya/cas/architecture/read-path.md @@ -0,0 +1,84 @@ +--- +description: 'How a CAS read resolves a ref to a manifest and then to ranged blob reads, and the two caches — manifest decode and part-folder view — that sit on that path.' +sidebar_label: 'Read path' +sidebar_position: 9 +slug: /antalya/cas/architecture/read-path +title: 'CAS Architecture — Read Path' +doc_type: 'reference' +--- + +# CAS architecture — read path {#read-path} + +A `CAS` read never touches a classical local-metadata path: there is no local directory listing to +consult, only a ref resolve followed by object-store reads. This page covers the three ways a file +access is served, the full chain for the common case, the two caches that sit on that chain, and +how a part still open inside a write transaction serves its own reads. + +## How a file access is served {#access-kinds} + +| Access kind | How it is served | S3 cost | +|---|---|---| +| Inline entry — small files such as `count.txt`, `columns.txt` | Decoded straight out of the manifest body | Zero additional operations | +| Blob-backed file — `.bin`, marks, large `primary.idx` | Ranged `GET` bounded by `[header_len, header_len + blob_size)` | One `GET` per column file per part open | +| Verbatim file — `roots/…` objects | Plain object read, no `CAS` indirection | One `GET` | + +The full chain for a blob-backed file is: resolve the ref, read the manifest, look up the path, +build a blob view plan, ranged `GET`, then `ReadBufferFromFileView`. Because the payload always +starts at a pool-constant offset (the manifest's `blob_header_len`), no header parse is needed to +locate content — see the [envelope format](/antalya/cas/architecture/storage-layout#envelope-format) +on the storage-layout page. + +Part manifests themselves are read whole after opening the object: there is no on-disk random +access, `seek`, or streaming requirement for their entry records — a manifest is small enough that +decoding the whole body is cheaper than any partial-read machinery would be. + +## The two caches {#caches} + +| Cache | Keyed by | Setting | Default | What still hits the network | +|---|---|---|---|---| +| Manifest decode cache | `(ManifestId, Token)` | `cas_manifest_decode_cache_bytes` | 128 MiB | A mandatory `HEAD` on **every** access, cache hit or miss | +| Part-folder view cache (`Cas::CachedPartFolderAccess`, `Parts/PartFolderAccess.h`) | Part ref key | `cas_part_folder_cache_bytes`, `cas_part_folder_cache_max_entries`, `cas_part_folder_cache_max_entry_bytes` | 64 MiB / 10 000 entries / 16 MiB | Its `ForceFresh` policy re-proves the manifest body via that same mandatory `HEAD`, paced by `cas_part_folder_validate` (`always` \| `never` \| `age `) | + +**The `HEAD` is mandatory even on a cache hit** — the page's most counter-intuitive fact, because it +means a cache hit still costs one object-store round trip: + +```mermaid +flowchart TD + A["readManifestShared(ManifestId)"] --> B["HEAD the manifest key"] + B -->|"absent"| C["throw FILE_DOESNT_EXIST --
a live ref must never name a missing object"] + B -->|"present, token t"| D{"cache lookup (ManifestId, t)"} + D -->|hit| E["return the cached decode -- no GET"] + D -->|miss| F["GET the body"] + F --> G{"body's own ref and namespace
match the key?"} + G -->|no| H["throw CORRUPTED_DATA"] + G -->|yes| I["decode, insert into cache keyed by (ManifestId, t), return"] +``` + +The `HEAD` is what proves the live ref still names an existing object — the no-dangle invariant — +and it supplies the token that keys the cache; only then is the decode cache consulted. On a miss, +the `GET` is followed by the two identity checks in the diagram, each `CORRUPTED_DATA` on failure. +Only a fully validated decode enters the cache. Setting either cache's byte budget to `0` disables +retention while leaving the `HEAD`-and-validate sequence intact — a cache is purely an +optimization, never a trust boundary. + +The part-folder view cache is invalidated on every promote and repoint, and is single-flight on a +cold build: concurrent readers of the same not-yet-cached view coalesce into one build rather than +racing independent `GET`s. + +## Reads while a part is still being written {#in-flight-reads} + +An in-flight part inside an open write transaction is not yet visible through the ordinary ref +resolve — reading it goes through the same explicit overlay used for read-your-writes, covered on +the [part-lifecycle page](/antalya/cas/architecture/part-lifecycle#in-flight-reads). The bare part +directory itself reports as absent in that overlay, precisely so that cleanup of a rejected +temporary part is never mistaken for a real, resolvable part. + +## Diagnostic and read-only access {#read-only-access} + +A read-only or diagnostic opener of a `CAS` disk (`ca-fsck`, `ca-gc-dryrun`, and similar tools) +must not claim mount ownership, schedule `GC`, or mint writer state — read-only enforcement sits +below the ordinary facade checks, at the backend layer itself. A mounted `Pool` caches its ref +table and does not re-recover it on every read; a diagnostic tool that deliberately performs a +fresh cold recovery on each pass can therefore observe a **less** stale ref table than a live +mounted read, which is intentional for tools whose entire purpose is catching drift a live mount +would not notice. diff --git a/docs/en/antalya/cas/architecture/replication.md b/docs/en/antalya/cas/architecture/replication.md new file mode 100644 index 000000000000..371ea74b39f7 --- /dev/null +++ b/docs/en/antalya/cas/architecture/replication.md @@ -0,0 +1,119 @@ +--- +description: 'Fetch by relink between two replicas sharing a pool: the gates in order, what actually seals commit-before-release, and detach/attach/drop.' +sidebar_label: 'Replication' +sidebar_position: 7 +slug: /antalya/cas/architecture/replication +title: 'CAS Architecture — Replication' +doc_type: 'reference' +--- + +# CAS architecture — replication {#replication} + +When two `ReplicatedMergeTree` replicas share a `CAS` pool, a fetch should move **no bytes** — the +receiver already has access to the same blobs the sender does. The mechanism is a three-phase +handshake, fetch by relink, layered directly on the ordinary interserver part-fetch protocol. This +page covers the handshake, the gates that decide whether it fires, what actually makes it safe +against a concurrent `GC` round, and how detach/attach/drop reduce to the same primitives. The +writer owns table semantics and part publication (see the +[part-lifecycle page](/antalya/cas/architecture/part-lifecycle)); `GC` owns ref-log folding and +physical cleanup (see the [garbage-collection page](/antalya/cas/architecture/garbage-collection)) +— ordinary replication traffic never reads `gc/state` or waits on a `GC` round. + +## The handshake {#handshake} + +Only two of the three phases are round trips to the sender — the offer and the confirm. The +publish and the promote are the receiver's own writes to the pool. + +```mermaid +sequenceDiagram + autonumber + participant R as Receiver + participant Snd as Sender + participant S3 as Shared pool + + R->>Snd: GET part, cas_pool_uuid = R's pool uuid, client_protocol_version = 11 + Note over R: advertising 11 is a promise to confirm before promoting + Snd->>Snd: same disk pool uuid? identity, never endpoint plus prefix + Snd->>S3: resolve the offer once -- manifest bytes and confirm token from the SAME view + Snd-->>R: cookie cas_relink = part_manifest_v2, cookie cas_source_token = ..., body = manifest bytes + Note over Snd: sender is fire-and-forget -- it releases the part here + + rect rgba(120,160,255,0.12) + Note over R,S3: T1 -- publish, the plus-one lands first + R->>S3: adopt entries by evidence, no HEAD, no bytes, stageManifest fresh receiver-local id, precommitAdd + Note over R: the sender's ManifestRef, namespace and digest are ignored -- only entries are used + end + + rect rgba(255,190,120,0.15) + Note over R,Snd: T2 -- confirm + R->>Snd: POST cas_confirm = token + Snd->>Snd: confirmExactRef, zero object-store I/O, never throws + Snd-->>R: cookie cas_confirm_answer = yes or unproven + end + + alt answer is yes + R->>S3: T3 -- promote, ref published + else anything else -- unproven, missing cookie, timeout, transport error + R->>R: throw a locally generated NETWORK_ERROR, retry later + Note over R: never a byte re-request -- that would go back to the very source whose state is in doubt + end +``` + +## The gates, in order {#relink-gates} + +| # | Gate | What it enforces | +|---|---|---| +| 1 | Pool identity | The receiver advertises `cas_pool_uuid`; the sender offers relink only if its own disk's pool uuid is **equal**. Matching by endpoint and prefix was tried and rejected — a minted pool uuid is the identity | +| 2 | Protocol version 11 | On the receiver side, advertising it is a promise to run the confirm round trip before promoting | +| 3 | One resolution for two outputs | The manifest bytes and the confirm token come from the **same** view. Two separate calls would allow a repoint in between and hand the receiver a token naming a manifest whose entries it never adopted | +| 4 | The receiver trusts nothing from the wire but the entry list | The sender's manifest id, namespace and payload digest are ignored; the target namespace and ref come from the receiver's own router, and manifest path hygiene is validated at decode | +| 5 | The confirm is I/O-free and fail-closed | A cold, evicted, unfenced or terminal mount answers `Unknown`. `No` and `Unknown` both go on the wire as `unproven`, because the fence check is evaluated last, so a `No` cannot be distinguished from "cannot prove it right now" | +| 6 | Only the literal `yes` authorizes promotion | Everything else — including a timeout — is one outcome: throw and retry later | +| 7 | Promote outcomes are three-way | `Committed` proceeds; a **proven** not-committed state (body-absent precommit, precommit no longer live owner, ref conflict) falls back to a byte fetch; `Unresolved` **throws**, because returning "fall back" there would publish the part twice | + +The byte-fetch fallback is bounded: it re-invokes the fetch with relink disabled, which stops the +receiver advertising its pool uuid, which stops the sender offering relink — so the relink path +cannot be entered twice for one fetch. Byte-fetched files content-address and dedup on arrival +anyway, so falling back never loses the dedup property, only the zero-byte-move property for that +one fetch. + +## What actually seals "commit before release" {#relink-seal} + +The receiver's `+1` — its precommit binding — is durable **before** the sender is asked anything, +and any removal of the sender's own binding is appended strictly after that `+1` is in the ref +log. That ordering, steps T1 then T2 then T3, is the whole seal. + +This does **not** establish that every subsequent `GC` fold *sees* that `+1` under every listing +behavior: a configuration with one incomplete listing page can, in principle, let a fold miss a +freshly published edge. A confirmed relink therefore proves only "the source still holds exactly +this manifest right now", not "no future fold can ever miss this edge" — `ca-fsck`'s +reachable-but-absent scan is the backstop for that gap, not the relink protocol itself. Relink +also races `GC` in the ordinary sense any writer does: between the sender encoding its offer and +the receiver's promote, `GC` on the shared pool may condemn a blob that was live only through the +sender's own ref. The [writer-versus-GC race](/antalya/cas/architecture/blob-protocol#writer-gc-race) +on the blob-protocol page is what makes that interleaving safe — revival is re-upload only, and the +receiver's evidence-adopt is protected by its own durable precommit edge exactly like any other +writer's adopt. + +A fetch whose source part is still a live, held `DataPartPtr` on the sender's own replica — the +common case for a local, same-process relink — keeps the source pinned through the destination's +commit by ordinary part-lifetime rules, independent of the ref-log seal above. + +## Detach, attach, drop {#detach-attach-drop} + +A detached part is **not** a separate namespace — it is a ref in the table's own namespace with a +`detached/` prefix (the same is true of `moving/`). Only `FREEZE` uses a separate shadow namespace, +but it remains under the root that created it: `/shadow//…`. The ownership +check therefore attributes it to exactly that server root, under the same strict prefix rule as live +content, and that root can confirm its exact refs. + +`DETACH`, `ATTACH`, `delete_tmp_` cleanup, and merge-result renames all reduce to the same two +moves: re-key any *staged* source into the destination, then `republishRef(src → dst)` for any +*committed* source. `republishRef` re-reads the source manifest freshly, publishes an +equivalent-entry manifest under the destination ref — a **new** manifest id, with blobs untouched +and adopted by evidence — then drops the source ref. A destination that already exists with +identical entries just drops the source, an idempotent re-drive; one with different entries +throws. + +Manifests are therefore per-ref and never moved: a detach creates a new manifest for +`detached/` and retires the old one, and the blobs' net in-degree is unchanged. diff --git a/docs/en/antalya/cas/architecture/storage-layout.md b/docs/en/antalya/cas/architecture/storage-layout.md new file mode 100644 index 000000000000..e4d20725836b --- /dev/null +++ b/docs/en/antalya/cas/architecture/storage-layout.md @@ -0,0 +1,161 @@ +--- +description: 'S3 key layout and on-disk text-object formats used by the content-addressed storage (CAS) MergeTree disk backend.' +sidebar_label: 'Storage layout' +sidebar_position: 2 +slug: /antalya/cas/architecture/storage-layout +title: 'CAS Architecture — Storage Layout' +doc_type: 'reference' +--- + +# CAS architecture — storage layout {#storage-layout} + +Every key in a pool is built by one class, `Cas::Layout` (`Formats/CasLayout.h`), which owns +exactly the pool prefix. Every persisted object opens with a one-line JSON envelope header, and +control-plane bodies are JSON Lines — one JSON object per line, sorted where the object is a log +or a set of entries (`Formats/README.md`; see [Envelope format](#envelope-format) below for which +parts are a single JSON object versus JSON Lines versus raw payload bytes). The format is +deliberately this plain: any object can be fetched and read with ordinary line-oriented tools +while debugging, and a new field is additive — a tolerant reader skips it — so the format evolves +without a migration. + +## Key table {#key-table} + +All key patterns are shown under the pool prefix. A **namespace** is the opaque per-table string +under which one `MergeTree` table's part manifests and ref history live: for a live table it is +the table's canonical disk path (`store//`, `@cas@`-marked) prefixed by the owning +server's `server_root_id`, and a `FREEZE` backup gets its own +`/shadow//…` namespace under that same root; `Cas::Layout` only validates a +namespace's shape and never interprets its contents. + +| Key pattern | Object | Codec | Writer | +|---|---|---|---| +| `_pool_meta` | pool identity + floors | `cas_pool_meta` | pool create/admit | +| `blobs///` | blob envelope + payload | `cas_blob` | uploads | +| `blobs///.meta` | blob freshness sidecar | `cas_blob_meta` | dedup/GC | +| `cas/ns/stream//_log/-.zst` | ref transaction log | `cas_ref_log` | writer commit path | +| `cas/ns/stream//_snap/-.zst` | complete ref table snapshot | `cas_ref_snap` | writer/GC fold | +| `cas/ns/state//_ckpt` | mutable per-life checkpoint | `cas_ref_ckpt` | writer/GC fold | +| `cas/ns/state//_files/` | namespace-owned verbatim file | — (raw passthrough) | upper layers | +| `cas/manifests//-/.zst` | part manifest | `cas_part_manifest` | part build | +| `gc/state` | GC state (incl. GC lease) | `cas_gc_state` | GC | +| `gc/hb` | GC leader heartbeat | `cas_gc_hb` | GC | +| `gc/maintenance_state` | leak-only namespace-janitor cursor | `cas_gc_maintenance_state` | future janitor | +| `gc/gen//attempt//fold_seal` | fold seal (deterministic) | `cas_fold_seal` | GC | +| `gc/gen//attempt//blob_target//` | GC source-edge run segment | `cas_run` | GC | +| `gc/gen//attempt//outcomes//.zst` | GC outcome log | `cas_gc_outcomes` | GC | +| `gc/server-roots//owner` | server-root owner singleton | `cas_owner` | mount | +| `gc/server-roots//epoch` | server-root epoch singleton | `cas_epoch` | mount | +| `gc/server-roots//mount` | mount lease (incl. `min_active` watermark) | `cas_mount_lease` | mount | +| `roots/` | loose mountpoint object, verbatim | — (never interpreted) | upper layers | +| `staging//…` | S3-native upload staging scratch | — | writer, own mount only | + +`` is `ch128`, `xxh3`, or `sha256` — the hash algorithm is a path segment because one pool may +legally hold blobs under several algorithms at once. `` is a flat two-character S3 key +shard for request-fan-out, unrelated to the separate `cas_gc_shards` GC-internal reduction fan-out +(which appears only inside `gc/gen/…` keys and routes by the digest's high 64 bits, read +big-endian). Discovery LISTs use fixed prefixes: `cas/ns/stream/`, `cas/ns/`, `cas/manifests/`, +`blobs/` (deliberately without the algorithm segment, so one recursive LIST covers every +algorithm), `roots/`, `gc/server-roots/`. `staging/` is a top-level sibling that no GC LIST ever +touches — it is reclaimed only by its own server's next mount. + +## Envelope format {#envelope-format} + +Every persisted CAS metadata object is text: a header line, a body, and an optional trailer. + +``` +{"type":"cas_","v":N} <- header line, always present + <- one JSON object, sorted NDJSON records, + or a descriptor + raw payload zone +{"n":…} <- optional trailer (record/entry count) +``` + +`v` is the only version field; a reader rejects `v` above what the build supports with +`UNKNOWN_FORMAT_VERSION`, checked before the body. A `.zst` key suffix means, exactly, that the +object kind's compression policy is `Always`: the object is stored as one zstd frame with the +checksum flag on, and its declared content size is checked against a per-kind cap before +allocation. Always-small and deterministic kinds (`cas_ref_ckpt`, `cas_blob_meta`, `cas_fold_seal`, +`cas_run`, …) are stored raw, with no `.zst` suffix. + +The blob envelope is a special case of the header/body shape: a JSON descriptor padded with ASCII +spaces to a pool-constant `blob_header_len` (256 bytes, a `cas_pool_meta` field), terminated by +`\n`, so the raw payload always starts at that fixed offset with no header parse needed to locate +it. The part manifest is the other `PayloadHybrid` kind: text header, descriptor, sorted NDJSON +entry records, `{"n":…}` trailer, then a banner-framed raw payload zone for small inline file +bytes. + +## Codec table {#codec-table} + +Condensed from the authoritative traits table in `CasFormat.cpp` (`TRAITS`, asserted complete by +`gtest_cas_text_format.cpp`). + +| Type string | Family | Key strictness | Compression | +|---|---|---|---| +| `cas_blob` | `PayloadHybrid` | tolerant | never (raw, fixed offset) | +| `cas_blob_meta` | `Control` | tolerant | never | +| `cas_pool_meta` | `Control` | tolerant | never | +| `cas_ref_log` | `Control` | tolerant | always (`.zst`) | +| `cas_ref_snap` | `Control` | tolerant | always (`.zst`) | +| `cas_ref_ckpt` | `Control` | strict | never | +| `cas_ref_catalog` | `Control` | strict | never | +| `cas_part_manifest` | `PayloadHybrid` | tolerant | always (`.zst`) | +| `cas_run` | `RecordStream` | strict | pinned raw | +| `cas_fold_seal` | `Control` | strict | pinned raw | +| `cas_gc_state` | `Control` | tolerant | never | +| `cas_gc_hb` | `Control` | tolerant | never | +| `cas_gc_outcomes` | `Control` | tolerant | always (`.zst`) | +| `cas_gc_maintenance_state` | `Control` | strict | never | +| `cas_owner` | `Control` | tolerant | never | +| `cas_epoch` | `Control` | tolerant | never | +| `cas_mount_lease` | `Control` | tolerant | never | + +"Strict" means unknown keys are rejected rather than skipped, used for objects where every field +decides a durability or cleanup decision (`cas_ref_ckpt`, `cas_ref_catalog`, `cas_fold_seal`, +`cas_run`, `cas_gc_maintenance_state`); a `!`-prefixed key is always critical regardless of the +kind's strictness. "Pinned raw" objects (`cas_run`, `cas_fold_seal`) need stable bytes across +re-encodes for deterministic-artifact adoption, so their bytes are never recompressed once +written. `cas_blob` and `cas_part_manifest` are the `PayloadHybrid` family: a text descriptor +followed by a raw payload zone, rather than a single JSON body. + +## Worked example tree {#worked-example} + +Pool prefix `ca-pool`, server root `srv1`, one `Atomic` table, one part `all_1_1_0` with one blob +column file, written at `writer_epoch = 1, sequence = 3`: + +``` +ca-pool/_pool_meta + +ca-pool/cas/ns/stream/0123456789abcdef0123456789abcdef/_log/0000000000000001-0000000000000003.zst +ca-pool/cas/ns/stream/0123456789abcdef0123456789abcdef/_snap/0000000000000001-0000000000000003.zst +ca-pool/cas/ns/state/0123456789abcdef0123456789abcdef/_ckpt + +ca-pool/cas/manifests/srv1/store/3f2/3f2a1b7c-…-abcdefabcdef@cas@/0000000000000001-0000000000000003/000001.zst + +ca-pool/blobs/xxh3/a1/a1b2c3d4e5f60708b1c2d3e4f5061728 +ca-pool/blobs/xxh3/a1/a1b2c3d4e5f60708b1c2d3e4f5061728.meta + +ca-pool/roots/srv1/clickhouse_access_check_8f3a1c2d + +ca-pool/gc/state +ca-pool/gc/hb +ca-pool/gc/server-roots/srv1/{owner,epoch,mount} +ca-pool/gc/gen/7/attempt/1/fold_seal +ca-pool/gc/gen/7/attempt/1/blob_target/0/1 +ca-pool/gc/gen/7/attempt/1/outcomes/1/0.zst + +ca-pool/staging/srv1/ +``` + +`0123456789abcdef0123456789abcdef` is the opaque physical `life_id` the catalog maps the table's +namespace to; the ref log and snapshot keys reuse the same `RefTxnId` rendering +(`0000000000000001-0000000000000003`) as the manifest's build-scoped directory, but they are +different counters with different semantics, not the same identifier. The `data.bin` entry inside +the part manifest names the blob by `{XXH3_128, a1b2…1728}`, which is what resolves to the +`blobs/xxh3/a1/…` key above. A small file such as `count.txt` has no object of its own — it is +inline inside the manifest's raw payload zone, not a separate key. + +## Notes {#notes} + +- `cas/ns/state//_ckpt` carries **no** `.zst` suffix: `cas_ref_ckpt`'s compression policy + is `never`, while its `_log`/`_snap` siblings in the same `cas/ns/` tree compress `always`. +- The namespace-stream tree is `cas/ns/stream/` (immutable `_log`/`_snap` objects) and + `cas/ns/state/` (mutable `_ckpt`, verbatim `_files/`). diff --git a/docs/en/antalya/cas/bucket-requirements.md b/docs/en/antalya/cas/bucket-requirements.md new file mode 100644 index 000000000000..67a3795ab682 --- /dev/null +++ b/docs/en/antalya/cas/bucket-requirements.md @@ -0,0 +1,73 @@ +--- +description: 'The object-store contract a bucket must satisfy to host content-addressed storage, and which providers qualify.' +sidebar_label: 'Bucket requirements' +sidebar_position: 4 +slug: /antalya/cas/bucket-requirements +title: 'CAS Bucket Requirements' +doc_type: 'reference' +--- + +# Bucket requirements {#bucket-requirements} + +`CAS` is built on a small object-store contract (`Backend/CasBackend.h`), checked by a capability +probe that runs at every writable mount and fails closed: an object store that does not enforce +these conditions is refused rather than trusted. + +## The capability table {#capability-table} + +| Requirement | Interface method | Why it is needed | +|---|---|---| +| Read-after-write on a fresh key | `Backend::get` / `Backend::head` | Recovery listings and point reads must see what was just written | +| Conditional create (`If-None-Match: *`) | `Backend::putIfAbsent`, `Backend::casPut` with expected absence | Write-once creation of manifests and control/log objects; blob bodies use unconditional publication after `HEAD` | +| Conditional overwrite (`If-Match: `) | `Backend::putOverwrite`, `Backend::casPut` | The one mutual-exclusion primitive: mount leases, `gc/state` | +| Unconditional complete-object publication | `Backend::publishBlob` | An absent or condemned content-addressed body is replaced atomically; native stores may use multipart | +| Native same-store copy when `cas_staging_backend = s3` | `IObjectStorage::copyObject` with `ObjectStorageCopyMode::NativeOnly` | The first absent staged publication may copy its complete object without a client-side fallback | +| Exact-token delete | `Backend::deleteExact` | GC must delete only the incarnation it condemned, never a replacement | +| Ranged `GET` | `Backend::get` / `Backend::getStream` with a `Range` | Opening one column file of a part costs one bounded read, not a whole-object fetch | +| `LIST` with a resumable cursor | `Backend::list` | GC discovery and the orphan-manifest sweep page through the pool without a separate index | +| No versioning / no delete markers | probed by `runCapabilityProbe`; `created_delete_marker` on `DeleteOutcome` | A delete marker over a live key would break exact-token semantics — GC would archive instead of reclaim | +| `TOKEN ⟹ CONTENT` (a repeated token implies unchanged bytes) | standing requirement on every `Backend` implementation | Not probed — it cannot be tested cheaply. A backend that recycled tokens would serve stale manifests, i.e. wrong query results, not merely an inefficiency | + +Bucket **versioning is not required** — in fact it must be **disabled** on the generation-token +dialect (see below), because a token-exact delete on a versioned bucket archives a noncurrent +generation instead of reclaiming storage, silently stopping GC reclamation. + +On the generation-token dialect that requirement is checked, and checked strictly: a writable mount +proceeds only when the probe *confirms* versioning is disabled. A bucket reported as versioned and a +probe that could not answer — the credential may not read the bucket's versioning configuration, or +the backend cannot report it — both refuse the mount. `CAS` does not assume the safe answer, because +the failure it would be assuming away is `GC` deleting objects it believes it reclaimed. + +Because that check is part of the mount battery, `skip_access_check = true` is refused on a writable +generation-token disk. Mount the disk read-only if you need to start before the access check can +pass. + +## Soft delete is an operator precondition {#soft-delete} + +Object **soft delete must be disabled** on a `CAS` bucket, and unlike versioning this one is *not* +verified at mount. Google Cloud Storage exposes the soft-delete policy through its JSON API, while +this backend and both of its authentication modes speak the XML API, so the storage path `CAS` uses +cannot inspect it. Disabling it is therefore your responsibility, not something a successful mount +attests to. + +Soft delete does not leave the deleted generation live, so it does not break exact-token semantics +the way versioning does. What it does is delay physical reclamation until the retention period +expires: `GC` reports space as reclaimed while the bill still reflects it. + +## Platform support {#platform-support} + +The deterministic request-construction coverage is green, but the +[real-GCS release gate](/superpowers/cas/unconditional-blob-publication-live-results) remains blocked +until its credentialed OAuth and HMAC groups run against Google Cloud Storage. A fake service cannot +establish acceptance of Google's multipart, native-copy, and exact-delete wire behavior. + +| Platform | Status | Notes | +|---|---|---| +| AWS S3 | ✓ | Native `ETag`-based conditional dialect for mutable objects and exact deletion; blob publication is unconditional | +| Google Cloud Storage | implementation complete; release gate pending | Generation-token dialect for mutable objects/native-token `HEAD`/exact deletion, opted into via `http_client = gcs_hmac` or `gcp_oauth`; blob publication uses ordinary copy/multipart. Real credentialed GCS groups have not run yet | +| Azure Blob Storage | probably | Azure's REST API documents the equivalent conditional headers, but ClickHouse's Azure object-storage backend does not yet wire up a `CAS` conditional dialect the way the S3 and GCS paths do — untested, not validated by the capability probe | +| Other S3-compatible stores | only with enforced conditional operations | The capability probe is the actual gate: a store that silently ignores `If-None-Match`/`If-Match` (accepting and applying the write regardless) fails the probe and is refused. `RustFS` passes the full battery and is used as the project's test backend; `Garage` was evaluated and rejected because it silently ignores conditional operations | + +The full mechanics of the two dialects — how the backend detects which one a given endpoint speaks, +what the capability probe actually checks, and how exact-token deletes map onto each provider's +primitives — are in [the Backend architecture page](/antalya/cas/architecture/backend). diff --git a/docs/en/antalya/cas/configuration.md b/docs/en/antalya/cas/configuration.md new file mode 100644 index 000000000000..d7ee12130993 --- /dev/null +++ b/docs/en/antalya/cas/configuration.md @@ -0,0 +1,159 @@ +--- +description: 'Every disk-level and server-level setting content-addressed storage exposes, generated from ContentAddressedSettings and ServerSettings at HEAD.' +sidebar_label: 'Configuration' +sidebar_position: 3 +slug: /antalya/cas/configuration +title: 'CAS Configuration Reference' +doc_type: 'reference' +--- + +# Configuration reference {#configuration-reference} + +## The disk config block {#disk-config} + +A `CAS` disk is an `object_storage` disk with `metadata_type` set to `cas` and an explicit +`cas_server_root_id`. The recommended shape layers a `type=cache` disk in front of it — the local +filesystem cache absorbs repeated reads of the same blob, while the `CAS` disk underneath stays the +single source of truth the pool's other members and GC also read from. The storage policy references +the **cached** disk, not the raw `CAS` disk directly: + +```xml + + + + + object_storage + s3 + cas + {replica} + https://bucket.s3.amazonaws.com/cas/ + ... + ... + + + cache + cas + /var/lib/clickhouse/cas_cache/ + 10Gi + + + + + +
+ cas_cache +
+
+
+
+
+
+``` + +`path` and `max_size` are ordinary `type=cache` disk settings (see +[external disk cache](/operations/storing-data#using-local-cache)), not `CAS`-specific — size the +cache to the working set of blobs a node reads repeatedly, not to the pool's total size. `type`, +`object_storage_type`, `metadata_type`, `endpoint`, `access_key_id`, `secret_access_key`, and the +other generic object-storage/disk keys (`path`, `name`, `region`, `use_environment_credentials`, +`readonly`, `use_fake_transaction`, and a handful more) belong to the shared disk layer, not to +`CAS` — they are accepted inside the `cas` disk's own block but are not `CAS` settings. `CAS` +validates its `cas_` namespace and leaves every other key, apart from the temporary unprefixed +aliases described below, to its relevant consumer. + +The bare, uncached form — a storage policy pointing directly at the `CAS` disk, as used by +[quick start](/antalya/cas/quick-start) — remains valid and is the minimal way to try `CAS` out: + +```xml + + + +
+ cas +
+
+
+
+``` + +## Disk-level settings {#disk-settings} + +The disk element is read by several components at once. `CAS` settings carry the `cas_` prefix; +every other key belongs to the object-storage or generic disk layer. + +`CAS` is experimental: any setting below may change semantics, change its default, or disappear +entirely before release. Treat this table as a snapshot of the current build, not a stable contract. + +| Setting | Default | Description | +|---|---|---| +| `cas_server_root_id` | — (required) | Explicit layout subtree identity; macros expand as in the `s3` `endpoint`. Anchored in the pool by a write-once owner claim — a colliding identity is refused at mount | +| `cas_scratch_path` | `/disks//cas_scratch/` | Server-local scratch dir for the write-buffer spill; a relative value is anchored to the server data path | +| `cas_gc_enabled` | `true` | Run the background GC scheduler on this disk. `false` is a debugging aid, not an operating mode: garbage then accumulates indefinitely and silently — watch `system.cas_gc_log` for round activity if you ever toggle it | +| `cas_gc_interval_sec` | `60` | Seconds between background GC rounds (≥ 1) | +| `cas_blob_hash` | `cityhash128` | Pool blob content-hash function (`cityhash128` \| `xxh3-128` \| `sha256`). Recorded in the pool at creation; a mismatching config is refused at mount | +| `cas_blob_hash_allow_new` | `false` | Explicit opt-in to admit a new hash algorithm into an existing pool. One-way: once admitted, the pool carries both algorithms permanently | +| `skip_access_check` | `false` | Skip the boot-time capability probe (start now, fix later). Only the preflight probe is skipped — the conditional-write correctness check still runs on every writable mount. **Not available on a writable generation-token (GCS) disk**, which refuses to mount with it: there, the probe battery is the only proof that a token-exact delete carries its generation precondition. Mount such a disk read-only if you need to defer the check | +| `cas_gc_snapshot_generations_to_keep` | `3` | GC snapshot generations retained | +| `cas_gc_shards` | `1` | Blob-hash-prefix reducer shards (≥ 1). Recorded in the pool at creation; a mismatching config is refused at mount | +| `gcs_max_conditional_put_bytes` | 1 GiB | Largest conditional non-blob `PUT` on a generation-token store, including create-if-absent metadata/control artifacts and conditional replacements. Blob publication is unconditional, uses ordinary multipart, and is not subject to this cap | +| `cas_part_folder_cache_bytes` | 64 MiB | Part-folder view cache byte budget (`0` disables retention) | +| `cas_part_folder_cache_max_entries` | `10000` | Part-folder view cache entry cap | +| `cas_part_folder_cache_max_entry_bytes` | 16 MiB | Oversized part-folder views bypass retention above this size | +| `cas_part_folder_validate` | `always` | Cache body re-proof policy (`always` \| `never` \| `age `). **Leave at `always`**: the other modes trade the fail-closed body-existence check for an optimization — this is a trust decision about unverified data, not a performance knob | +| `cas_manifest_decode_cache_bytes` | 128 MiB | Manifest decode cache byte budget (`0` disables) | +| `cas_gc_meta_pool_size` | `16` | Bounded pool size for GC per-hash freshness-meta writes | +| `cas_staging_backend` | `local` | Blob staging backend (`local` \| `s3`); `s3` is opt-in and requires native same-store copy on writable mount | + +## Advanced GC pacing settings {#advanced-gc-pacing-settings} + +These settings bound individual phases of a `GC` round. The first two accept any `UInt64` value; +for the remaining caps, `0` means unbounded. + +| Setting | Default | Bounds | Description | +|---|---|---|---| +| `cas_manifest_sweep_list_budget_keys` | `1000` | `UInt64` | Orphan-manifest sweep `LIST` budget per round | +| `cas_manifest_sweep_delete_budget_keys` | `100` | `UInt64` | Orphan-manifest sweep `DELETE` budget per round | +| `cas_gc_round_graduation_budget` | `5000` | `0` = unbounded | Blob-graduation (`condemned` → `delete_pending`) cohort cap per round | +| `cas_gc_round_redelete_budget` | `5000` | `0` = unbounded | Exact-token re-delete cohort cap for prior `delete_pending` rows per round | +| `cas_gc_round_sweep_namespace_budget` | `20` | `0` = unbounded | Distinct namespaces per orphan-manifest sweep page whose protection view may be built | +| `cas_gc_round_sweep_recovery_op_budget` | `5000` | `0` = unbounded | Committed-tail ref-log `GET`/decode operations the orphan-manifest recovery walk may spend per round | +| `cas_gc_round_ref_cleanup_budget` | `5000` | `0` = unbounded | Ref-object cleanup cap for covered log and snapshot deletes per round | +| `cas_gc_round_prefix_wholesale_budget` | `20000` | `0` = unbounded | Generation-prefix wholesale-delete object cap during pruning per round | +| `cas_gc_round_handoff_prefix_wholesale_budget` | `5000` | `0` = unbounded | Post-`CAS` hand-off generation-prefix reclaim cap per round, reserved separately so pruning cannot starve the one-shot hand-off | +| `cas_gc_round_outcome_entry_budget` | `5000` | `0` = unbounded | `GcOutcomes` entry cap across the re-delete/spared audit log per round | + +## Migration from unprefixed keys {#migration-from-unprefixed-keys} + +The unprefixed spelling of a `CAS` setting is accepted for now and reported at server startup. It +will stop being accepted; update configurations to the `cas_` names in the table above. + +Two keys deliberately remain unprefixed: `skip_access_check`, shared with the generic disk layer, +and `gcs_max_conditional_put_bytes`, an S3 client setting. The server-level +`skip_access_check` flag skips the generic disk access check, while the `CAS` capability probe is +governed by the disk's own `skip_access_check` key. + +### Choosing `cas_blob_hash` {#choosing-blob-hash} + +`cas_blob_hash` is fixed at pool creation, so pick it deliberately. `cas_blob_hash_allow_new` is the +escape hatch — it admits a second algorithm into an existing pool's `algos_used` rather than +requiring a fresh pool. + +| Algorithm | Pick it for | Trade-off | +|---|---|---| +| `sha256` | Maximum safety | No known collision classes; slightly slower than the other two | +| `xxh3-128` | Maximum speed | Fastest, 128-bit, no known collision classes | +| `cityhash128` (default) | ClickHouse-ecosystem compatibility, and a possible future hash-reuse mode that avoids recomputation | Fast, but has a known class of collisions that occurs far more often than an ideal hash function would predict | + +## Server-level settings {#server-settings} + +Source: `ServerSettings.cpp`. This setting is process-wide rather than scoped to one disk block. + +| Setting | Default | Description | +|---|---|---| +| `cas_blob_upload_pool_size` | `16` | Size of the dedicated server-wide thread pool used to upload blobs in parallel when committing a `CAS` part. Zero is rejected: the pool must have at least one thread | + +## `SYSTEM CAS` commands {#system-commands} + +`SYSTEM CAS GC RUN`, `SYSTEM CAS GC STOP`, `SYSTEM CAS GC START`, `SYSTEM CAS GC REBUILD`, +`SYSTEM CAS FSCK`, `SYSTEM CAS FORGET`, and `SYSTEM CAS DROP POOL MEMBER '' FROM +DISK ''` operate on a mounted `CAS` disk. Introspection lives in `system.cas_log`, +`system.cas_gc_log`, and `system.cas_mounts`. diff --git a/docs/en/antalya/cas/index.md b/docs/en/antalya/cas/index.md new file mode 100644 index 000000000000..2bc71046494b --- /dev/null +++ b/docs/en/antalya/cas/index.md @@ -0,0 +1,89 @@ +--- +description: 'What content-addressed storage is, the problem it solves, its current status, and where to go next.' +sidebar_label: 'Overview' +sidebar_position: 1 +slug: /antalya/cas +title: 'Content-Addressed Storage' +doc_type: 'guide' +--- + +# Content-addressed storage {#content-addressed-storage} + +`ReplicatedMergeTree` on object storage has two unattractive options today. Plain replication +stores a byte-identical copy of every part on every replica, so storage cost multiplies with the +replication factor. Zero-copy replication shares the bytes, but at a structural price: every +replica keeps local metadata referencing each shared S3 object, and that state grows with the +data; a commit spans three independent systems — local disk, S3, and `Keeper` — whose interleaving +is easy to get subtly wrong, and a failure in any one of the three hurts availability; sharing is +tracked by a numeric refcount, so a lost or duplicated retry can corrupt the count; and the +special cases supporting all of this are scattered widely through the `MergeTree` code. + +Content-addressed storage (`CAS`) is a `MetadataStorage` back-end for object-storage disks +(`metadata_type = cas`) that takes the same sharing goal and collapses it onto one system: every +`MergeTree` part file is stored once, keyed by the hash of its content, in the object-storage pool +itself. There is no `CAS` state in `Keeper` at all — a commit is one conditional write against a +single object in the pool — and the reachability accounting is a derived in-degree edge set folded +from append-only deltas, not a mutable refcount a lost message can corrupt. + +```mermaid +graph LR + subgraph today["Today: zero-copy replication"] + R1["Replica 1
local disk: object refs
(grows with data)"] -->|"in-flight ops only"| K["Keeper"] + R2["Replica 2
local disk: object refs
(grows with data)"] -->|"in-flight ops only"| K + R1 -.->|"shares bytes"| S1["S3"] + R2 -.->|"shares bytes"| S1 + end + subgraph cas["CAS: content-addressed pool"] + C1["Replica 1"] -->|"publish a ref"| P["S3 pool
(refs, leases, GC — all in-bucket)"] + C2["Replica 2"] -->|"publish a ref"| P + end +``` + +Every CAS bookkeeping object — refs, mount leases, GC leadership, fencing tokens — lives in the +bucket. There is no external coordinator, and no `Keeper` usage inside the pool protocol; `Keeper` +stays exactly where `ReplicatedMergeTree` already used it, for replication log and part-set +consensus, and its load does not grow with pool size. + +## Deployment guidance {#deployment-guidance} + +`GC` throughput is proportional to how much changes in the pool: a pool holding a very large +number of parts from many servers, or data that churns very quickly, means longer `GC` rounds. +Two consequences for planning: + +- **The preferred deployment is a second tier for cold data**: hot, fast-churning parts stay on + the local (or plain S3) tier, and `CAS` holds the large, slow-moving cold tail — where + deduplication pays the most and `GC` traffic is minimal. +- **At large scale, shard the pool by key prefix.** With tens of servers, or thousands of tables + and millions of parts, split the deployment into several independent pools by giving each shard + its own prefix — the shards can share one bucket: + + ```xml + https://bucket.s3.amazonaws.com/cas/{shard} + ``` + + Each prefix is a fully independent pool (its own refs, leases, and `GC`), so rounds stay short + regardless of the total fleet size. + +## Status {#status} + +`CAS` is **experimental**. It ships in Altinity Antalya builds. Experimental means the on-disk +format and the SQL surface can still change between releases — that is deliberate, not a caveat to +apologize for. Pre-release means the format can change cheaply, with zero compatibility +scaffolding, and the design can keep being iterated on invariants rather than migrations. The bet +underneath it: all you need is a good S3 bucket. See [bucket requirements](/antalya/cas/bucket-requirements) +for exactly what "good" means. + +`CAS` coexists with zero-copy replication; it does not replace it. `metadata_type = cas` is opt-in +per disk, so adopting it never requires migrating an existing deployment. + +## Where to go next {#nav} + +| Page | Covers | +|---|---| +| [Quick start](/antalya/cas/quick-start) | A minimal disk config and the first `CREATE TABLE` / `INSERT` / `SELECT` | +| [Configuration](/antalya/cas/configuration) | Every disk-level and server-level setting | +| [Bucket requirements](/antalya/cas/bucket-requirements) | What an object store must support, and which providers qualify | +| [Architecture overview](/antalya/cas/architecture/) | The object model, the Git analogy, and the safety invariants | +| [Correctness](/antalya/cas/architecture/correctness) | How the design was verified: TLA+ models, counterexamples, soak methodology | +| [Design history](/antalya/cas/architecture/design-history) | What earlier designs were tried and rejected, and why | +| [Roadmap](/antalya/cas/roadmap) | What is shipped, planned, and deliberately not pursued | diff --git a/docs/en/antalya/cas/operations/debugging.md b/docs/en/antalya/cas/operations/debugging.md new file mode 100644 index 000000000000..cbc8b0c693f0 --- /dev/null +++ b/docs/en/antalya/cas/operations/debugging.md @@ -0,0 +1,268 @@ +--- +description: 'SQL-first CAS debugging: live investigation queries against cas_log/cas_gc_log/cas_mounts/blob_storage_log, SYSTEM CAS FSCK/GC RUN/GC STOP-START/FORGET, and the offline clickhouse-disks tools for when the server cannot answer.' +sidebar_label: 'Debugging' +sidebar_position: 4 +slug: /antalya/cas/operations/debugging +title: 'CAS Operations — Debugging' +doc_type: 'guide' +--- + +# Operations — debugging {#debugging} + +Debugging a content-addressed (`CAS`) incident starts on a **live server**, with SQL: the three +system tables plus `SYSTEM CAS` commands cover reachability checks, forced GC rounds, and +per-object/per-round forensics without ever touching the bucket directly. The offline +`clickhouse-disks` tools at the [end of this page](#offline-tools) are the fallback for when SQL +cannot reach the pool at all — the server is down, or the access is deliberately read-only forensic. + +## Investigating on a live server {#live-investigation} + +See [monitoring](/antalya/cas/operations/monitoring#system-tables) for the three system tables' +grain and general health queries; this section is investigation queries for a specific incident, +not a health dashboard. + +### What happened to this part or blob {#part-blob-history} + +`system.cas_log` carries one row per writer/GC decision, keyed by `ref_name` (a part name) or +`object_hash` (a blob's content hash): + +```sql +SELECT event_time_microseconds, event_type, outcome, reason, object_kind, object_hash, token, round, detail +FROM system.cas_log +WHERE disk_name = 'cas' AND ref_name = '' +ORDER BY event_time_microseconds; +``` + +```sql +SELECT event_time_microseconds, event_type, outcome, reason, ref_name, round, detail +FROM system.cas_log +WHERE disk_name = 'cas' AND object_kind = 'blob' AND object_hash = '' +ORDER BY event_time_microseconds; +``` + +`outcome` (`ok`, `adopt`, `resurrect`, `deleted`, `replaced`, `spared`, `absent`, `zeroed`, +`skipped`) and `reason` are the two columns to read first; `detail` is a +`Map(LowCardinality(String), String)` of decision-specific facts (`condemn_round`, +`superseded_token`, `code`, `site`) worth `arrayJoin(detail)` when the summary columns alone do not +explain the decision. See [`system.cas_log`](/operations/system-tables/cas_log) for the full column +reference. + +### Why GC is not reclaiming {#why-not-reclaiming} + +Two questions, in order: is this node's scheduler leading, and did its recent rounds actually fold? + +```sql +SELECT server_root_id, is_leader, state, last_success_age_seconds, pending_reclaim +FROM system.cas_mounts WHERE disk = 'cas'; + +SELECT event_time, outcome, candidates_marked, entries_condemned, entries_graduated, + entries_redeleted, anomalies +FROM system.cas_gc_log +WHERE event_type = 'Finish' AND disk_name = 'cas' +ORDER BY event_time DESC LIMIT 10; +``` + +A `0`/`false` `is_leader` means this node never reclaims for this disk — check the peer that holds +leadership instead. A steady `entries_condemned` with `entries_graduated` stuck at `0` means objects +are being found but never crossing the safety floor (recall the grace period is measured in full +rounds, not acks — see [condemnation and deletion](/antalya/cas/architecture/garbage-collection#condemn-delete)). +A specific blob's own story — was it ever condemned, spared, or is it not being seen at all — is the +per-object query in the previous section, filtered to `object_kind = 'blob'`. + +### What one GC round did {#gc-round-detail} + +Every round writes a `Start` and a `Finish` row to +[`system.cas_gc_log`](/operations/system-tables/cas_gc_log), correlated by `round_id` (not `round`, +which is `0` on `Start` and absent on a round that never led). One `Phase` row per phase reached +carries that phase's own `phase_duration_microseconds`, `ProfileEvents` delta, and `phase_metrics` — +group by `round_id` to reconstruct one round in order: + +```sql +SELECT event_type, outcome, phase, phase_duration_microseconds, duration_ms +FROM system.cas_gc_log +WHERE round_id = '' +ORDER BY event_time_microseconds; +``` + +### Who holds the mount {#who-holds-mount} + +```sql +SELECT server_root_id, hostname, process_id, state, writer_epoch, renewal_sequence, + expires_at, is_leader +FROM system.cas_mounts +WHERE disk = 'cas' +ORDER BY is_leader DESC; +``` + +Every `server_root_id` sharing the pool shows up here, not just this node's own — a `state` other +than `live` (`expired`, `terminated`, `fenced`, `corrupt`) on a member that should be up is the first +thing to check before assuming a lease problem is this node's own. `is_leader` and the other +process-local columns are `NULL` on every peer's row; run the query on that peer to see its own view. + +### Trace a renewal through remount {#trace-renewal-remount} + +Nontrivial mount recovery is represented by aggregate `watermark_renew` and `mount_remount` rows, +not by one warning per physical request. Query both event types in one timeline: + +```sql +SELECT event_time_microseconds, event_type, outcome, reason, + detail['server_root_id'] AS server_root_id, + detail['writer_epoch'] AS writer_epoch, + detail['seq'] AS renewal_sequence, + detail['write_attempt_id'] AS write_attempt_id, + detail['attempts_sent'] AS attempts_sent, + detail['classification'] AS classification, + detail['deadline_source'] AS deadline_source, + detail['stop_cause'] AS stop_cause, + detail['attempt_no'] AS remount_attempt, + detail['step'] AS remount_step, + detail['error'] AS error +FROM system.cas_log +WHERE disk_name = 'cas' + AND event_type IN ('watermark_renew', 'mount_remount') +ORDER BY event_time_microseconds; +``` + +Interpret the sequence as follows: + +- `retrying -> recovered` with the same `write_attempt_id` means an in-budget blip recovered in the + existing epoch; `classification = 'committed_by_get'` means exact `GET` proved a landed request, + while `committed_after_retry` means a later identical physical `PUT` completed. +- A `failed` renewal carries the decisive `unresolved_reason`, `deadline_source`, `stop_cause`, and + `classification`. `external_lease_deadline`, `cancelled`, `conflict`, + `fence_or_lifecycle_lost`, and `attempts_exhausted` are different operator diagnoses; do not + collapse them into a generic timeout. +- A following `mount_remount` row names the whole-chain `attempt_no` and final `step`. An `ok` row + restored `Live` under the reported fresh `writer_epoch`; a `failed` row's `step` and optional + `error` identify where that whole-chain attempt stopped. + +Use deltas of the mount counters from +[monitoring](/antalya/cas/operations/monitoring#mount-renewal-remount-counters) to check completeness: +a recovered blip increments renewal work/recovery but not `CASMountLeaseLost` or remount counters; +a terminal operational loss increments `CASMountLeaseLost` once, then each whole-chain attempt +increments exactly one of `CASRemountSucceeded` or `CASRemountFailed`. + +## SQL commands for live diagnosis {#sql-commands} + +### SYSTEM CAS FSCK {#sql-fsck} + +The online consistency check — unlike the offline tools below, this runs against a disk the server +already has **mounted and serving traffic**; the scan re-validates every finding against a fresh +authoritative read, so it needs no quiesce: + +```sql +SYSTEM CAS FSCK cas; +``` + +Returns one row: `disk`, `reachable`, `dangling`, `unreachable`, `pending_gc`, `awaiting_gc`, +`unaccounted`, `stale_edge`, `corrupted_runs`, `chain_broken`, `unchecked`, `lifeless_keys`, +`namespace_janitor_pending` (+`_bytes`/`_lives`), `ref_records_walked`, `physical_bytes`, +`referenced_logical_bytes`, `distinct_blobs`, `total_blob_refs`. `dangling` is the one column that +means data loss — `unreachable`, `pending_gc`, and `awaiting_gc` are objects still +moving through the normal condemn/graduate/delete pipeline, not a problem on their own. +`chain_broken` and `corrupted_runs` are the other two hard findings: a hole in a ref-log stream and a +GC source-edge run that failed its checksum, respectively. This summary-only form has no +per-object `--detail` equivalent yet — for that, the offline `cas-fsck --detail` below is still +needed. + +### SYSTEM CAS GC RUN {#sql-gc-run} + +Runs one round synchronously and returns exactly the shape of a `cas_gc_log` `Finish` row — driving +a round on demand while watching its outcome interactively is one of the most direct diagnostics +available: + +```sql +SYSTEM CAS GC RUN cas; +``` + +One row per disk it ran on: `disk`, `acquired_lease`, `deferred`, `round`, `candidates_marked`, +`objects_deleted`, `objects_absent`, `objects_replaced`, `objects_spared`, `manifests_deleted`, +`entries_condemned`, `entries_graduated`, `entries_redeleted`, `fence_outs`, `anomalies`, +`pending_candidates`, `pending_condemned`, `pending_retired`. Omitting +the disk name runs one round on every content-addressed disk on the node. A manual run executes +regardless of `SYSTEM CAS GC STOP` — `STOP` pauses only the background scheduler. + +### SYSTEM CAS GC STOP / START {#sql-gc-stop-start} + +Pause the background scheduler on one disk while investigating a suspect object, so it cannot be +condemned or deleted mid-investigation, then resume it: + +```sql +SYSTEM CAS GC STOP cas; +-- investigate, e.g. cas-inspect a specific blob's raw key +SYSTEM CAS GC START cas; +``` + +`STOP` is idempotent and stops-in-place (the same scheduler instance resumes on `START`, keeping its +`gc_id` and lease-observation history); it works even on a not-live disk. It does not stop a manual +`SYSTEM CAS GC RUN`. See the [operational surface](/antalya/cas/architecture/garbage-collection#operational-surface) +table for the full command list. + +### SYSTEM CAS FORGET {#sql-forget} + +Node-local operator assertion that a disk is permanently gone — the "fire marshal" verb for a stuck +disk (a transient/`IdentityLost` pool, an operator-asserted decommission): + +```sql +SYSTEM CAS FORGET cas; +``` + +It is an assertion, not a proof of erasure: the disk stays registered and answers further store-class +access with a typed error, and a server restart re-registers the name. This is different from +[`SYSTEM CAS DROP POOL MEMBER`](/antalya/cas/operations/migration#decommission), which permanently +retires one pool *member*'s identity across the whole shared pool — `FORGET` only affects this node's +own local view of one disk. + +## Offline tools {#offline-tools} + +When the server cannot answer — it is down, or the access needs to be read-only forensic against the +bucket directly, disaster recovery of `gc/state`, or a raw object decode — `clickhouse-disks` runs +these against the pool's backend without a live server. All five require the disk to be opened with +`true` in the `clickhouse-disks` config; they must never claim a live server's +mount. + +| Command | Use it for | +|---|---| +| `cas-fsck [--detail] [--timeout N] [--namespace PREFIX] [--partial]` | The same reachability scan as `SYSTEM CAS FSCK`, offline. `--detail` adds a per-object `\t\t` listing (`reachable`, `dangling`, `unreachable`, `pending-gc`, `awaiting-gc`, `unaccounted`, `stale-edge`, `corrupted-run`, `chain-broken`, `unchecked`, `lifeless-key`, `janitor-pending`) — the only way to get per-object, not just per-pool, findings. `--timeout`/`--partial` bound a scan on a large pool | +| `cas-gc-dryrun` | Previews the next round's deletes, read-only, no lease. Over-reports away from quiescence (does not fold new owner events) — a diagnostic only, never a delete source | +| `cas-inspect ''` | Decodes one raw object-storage key (as printed by `cas-fsck`/`cas-gc-dryrun`) straight to JSON | +| `cas-gc-rebuild [--force]` | Disaster recovery: rebuilds a `gc/state` baseline from raw owner state after the GC guard has refused every regular round. `--force` bypasses only the healthy-state refusal, never a competing leader or a failed `CAS`. See [`SYSTEM CAS GC REBUILD`](/sql-reference/statements/system#system-cas-gc-rebuild) for the destructive-tool caveats | + +```bash +clickhouse-disks -C config.xml --disk cas cas-fsck --detail +clickhouse-disks -C config.xml --disk cas cas-gc-dryrun +clickhouse-disks -C config.xml --disk cas cas-inspect '' +clickhouse-disks -C config.xml --disk cas cas-gc-rebuild --force +``` + +`cas-drop-member` — the offline twin of `SYSTEM CAS DROP POOL MEMBER` — is covered on the +[migration page](/antalya/cas/operations/migration#decommission) alongside the SQL form, since +decommissioning a pool member is a migration/scale-down operation, not an incident-time tool. + +## The CLICKHOUSE_USER_FILES gotcha when reproducing a test manually {#user-files-gotcha} + +Running a `CAS` stateless test directly with `tests/clickhouse-test` against a manually started +`clickhouse-server` (outside a configured praktika lane) requires exporting `CLICKHOUSE_USER_FILES` +to match the server's actual data path. The harness's default, +`/var/lib/clickhouse/user_files`, will not match a custom data path, which makes the pool directory +invisible to the server — the symptom is an `Unknown disk` error together with a diagnostic that +reads like an empty pool (e.g. `baseline=0 after_insert=0`) even though the server is otherwise +healthy. + +## What to collect before filing a bug {#filing-a-bug} + +- `SYSTEM CAS FSCK ''` output (or `clickhouse-disks cas-fsck --detail`, if the server cannot + answer or a per-object listing is needed) — the authoritative reachability snapshot at the time of + the incident. +- The `system.cas_gc_log` rows for the relevant `round_id`(s): `Start`, every `Phase`, and `Finish`. +- The `system.cas_log` rows for the specific ref name, blob hash, or object key involved, filtered by + `event_time` around the incident. +- `system.cas_mounts` output from every node sharing the pool, to capture lease/epoch state at + incident time — it is a live view and will not reflect a state that has since changed. +- For a suspected object-store issue, `system.blob_storage_log` rows for the affected `disk_name` + with a nonzero `error_code`, and the relevant `CAS*` `ProfileEvents` (`system.query_log`'s + `ProfileEvents` map for one query, or `system.metric_log`'s `ProfileEvent_*` columns for a window — + see [monitoring](/antalya/cas/operations/monitoring#key-metrics) for which counters matter and the + restart-resets-`system.events` caveat). +- The server version and, if the incident is reproducible, the exact `CREATE TABLE` / `INSERT` / + `ALTER` sequence that triggers it. diff --git a/docs/en/antalya/cas/operations/migration.md b/docs/en/antalya/cas/operations/migration.md new file mode 100644 index 000000000000..df51e79144a3 --- /dev/null +++ b/docs/en/antalya/cas/operations/migration.md @@ -0,0 +1,209 @@ +--- +description: 'Adding a content-addressed disk to an existing deployment, moving a partition onto it with ALTER TABLE MOVE PARTITION, rolling back, and permanently decommissioning a pool member.' +sidebar_label: 'Migration' +sidebar_position: 1 +slug: /antalya/cas/operations/migration +title: 'CAS Operations — Migration' +doc_type: 'guide' +--- + +# Operations — migration {#migration} + +This page walks through moving `MergeTree` data onto a content-addressed (`CAS`) disk from an +existing disk, and the reverse. `metadata_type = cas` is opt-in per disk (see the +[overview](/antalya/cas)), so this is an additive change to a running deployment: the existing +disk and its data are untouched until a partition is explicitly moved. + +## Add a CAS disk alongside an existing one {#add-disk} + +A storage policy can carry both an ordinary disk and a `CAS` disk as separate volumes. `ALTER TABLE +... MOVE PARTITION ... TO DISK` then moves data between them without an `INSERT`/`DROP` cycle. As on +the [configuration](/antalya/cas/configuration#disk-config) page, the recommended shape layers a +`type=cache` disk over the `CAS` disk, and the policy's volume references the **cached** disk name: + +```xml + + + + + local + /var/lib/clickhouse/local_disk/ + + + object_storage + s3 + cas + {replica} + https://bucket.s3.amazonaws.com/cas/ + ... + ... + + + cache + cas + /var/lib/clickhouse/cas_cache/ + 10Gi + + + + + + + local_disk + + + cas_cache + + + + + + +``` + +See [configuration](/antalya/cas/configuration) for the full disk-level settings surface and +[bucket requirements](/antalya/cas/bucket-requirements) for what the target bucket needs to +support. A table does not need to be created for the first time on `CAS` to use it — an existing +table just needs its storage policy widened to include a volume backed by a `CAS` disk, which is a +metadata-only change (`ALTER TABLE ... MODIFY SETTING storage_policy = ...`, subject to the usual +constraint that the new policy must still contain every volume and disk of the old one — a storage +policy can only grow, never lose a disk it once had). + +## Move a partition onto CAS {#move-partition} + +`ALTER TABLE ... MOVE PARTITION ... TO DISK` moves every part of one partition to the named disk in +place — the ordinary `MergeTree` partition-move mechanism, unchanged by `CAS`: + +```sql +CREATE TABLE events (event_date Date, event_id UInt64, payload String) +ENGINE = MergeTree ORDER BY event_id PARTITION BY event_date +SETTINGS storage_policy = 'tiered'; + +INSERT INTO events VALUES ('2026-08-04', 1, 'hello'), ('2026-08-04', 2, 'world'); + +SELECT name, partition, disk_name FROM system.parts WHERE table = 'events' AND active; +``` + +```text +Row 1: +────── +name: 20260804_1_1_0 +partition: 2026-08-04 +disk_name: local_disk +``` + +The partition starts on `local_disk`, the first volume in the policy. Moving it onto `CAS` uploads +each part's files as content-addressed blobs, writes a part manifest, and publishes a ref — the same +write path an `INSERT` directly onto `CAS` takes (see +[what just happened](/antalya/cas/quick-start#what-happened) in the quick start). `TO DISK` names the +disk actually listed in the policy's volume — with a cache layered in front, that is the **cache** +disk's name (`cas_cache`), not the raw `CAS` disk's name (`cas`) underneath it; naming the raw disk +is refused, because it is not a member of the table's storage policy: + +```sql +ALTER TABLE events MOVE PARTITION '2026-08-04' TO DISK 'cas_cache'; + +SELECT name, partition, disk_name FROM system.parts WHERE table = 'events' AND active; +``` + +```text +Row 1: +────── +name: 20260804_1_1_0 +partition: 2026-08-04 +disk_name: cas_cache +``` + +```sql +SELECT * FROM events ORDER BY event_id; +``` + +```text +2026-08-04 1 hello +2026-08-04 2 world +``` + +`system.parts.disk_name` reports the cache disk's name, not the underlying `CAS` disk's — this is +the ordinary `type=cache` disk behavior (the same happens layering a cache over any other disk type) +and is not `CAS`-specific. `system.filesystem_cache` shows the part's files populated into the +`cas_cache` cache on this read-through. + +## Roll back {#rollback} + +The move is symmetric: `MOVE PARTITION ... TO DISK` back onto the original disk name returns the +partition to its previous location, with the data intact throughout: + +```sql +ALTER TABLE events MOVE PARTITION '2026-08-04' TO DISK 'local_disk'; + +SELECT name, partition, disk_name FROM system.parts WHERE table = 'events' AND active; +``` + +```text +Row 1: +────── +name: 20260804_1_1_0 +partition: 2026-08-04 +disk_name: local_disk +``` + +Moving a partition off `CAS` does not itself delete the blobs it stops referencing — dropping the +old ref makes them eligible for reclamation by the next +[GC round](/antalya/cas/architecture/garbage-collection), the same as dropping a part. + +This exact three-disk, cache-over-`CAS` configuration and the forward/rollback `ALTER TABLE ... MOVE +PARTITION` sequence above were run against a live server before publication, using the `local` +object-storage backend for the `cas` disk: `CREATE TABLE`, `INSERT`, both `MOVE PARTITION` +directions, the `system.parts` checks, the `system.filesystem_cache` check, and the `SELECT` all +completed with zero errors and the shown output. A prior attempt to move onto `TO DISK 'cas'` +directly (the raw disk, not the cache) was refused with `All parts of partition '20260804' are +already on disk 'cas_cache'. (UNKNOWN_DISK)` — a real error message from the run, kept here because +it is exactly what an operator sees after guessing the wrong disk name. + +## Permanently removing a pool member {#decommission} + +A `CAS` pool can be shared by several servers (see [`server_root_id`](/antalya/cas/architecture/mounts-and-leases#server-root-id)). +Scaling down — permanently removing a server that will never rejoin the pool — is a distinct, +irreversible operation from an ordinary restart or a temporary outage: it fences the member's +`server_root_id` and reclaims the storage attributable only to it. + +`SYSTEM CAS DROP POOL MEMBER` claims the victim's mount slot as an administrative writer (refusing +immediately if the member is still alive), drops every table namespace the member owned, sweeps +manifest debris, drains its staging and mountpoint objects, and — only once every drain is +confirmed — retires the mount slot itself. It emits ordinary ref-edge deltas rather than a GC +transition: it does not synchronously reclaim shared blob content, it only makes the now-unreferenced +blobs eligible for an ordinary GC round to reclaim later. + +```sql +SYSTEM CAS DROP POOL MEMBER 'server_root_id' FROM DISK 'disk_name' [ON CLUSTER cluster_name] +``` + +Both `server_root_id` and `disk_name` are required string literals. The offline CLI twin, +`clickhouse-disks cas-drop-member `, does the same work against a disk opened +read-only — the pool-admin claim happens internally, so the disk it runs against must not be the +live server's own mount: + +```bash +clickhouse-disks -C config.xml --disk cas cas-drop-member 'replica-2' +``` + +The command returns one row (or, offline, one line per field) with `namespaces_removed`, +`namespaces_already_removed`, `committed_refs_removed`, `precommits_removed`, +`manifest_debris_removed`, `staging_objects_removed`, `mountpoint_objects_removed`, and +`slot_removed`. It is resumable: a rerun skips namespaces already marked removed and reports them +under `namespaces_already_removed` rather than redoing the work. A per-object drain failure is +recorded as a `warning` rather than raised as an exception, leaving the slot terminated but not +fully drained so a later invocation can resume; a non-empty `warnings` means exactly that, and the +mount slot stays in place as a resume anchor rather than being fully retired. + +**Preconditions.** Confirm the member is actually and permanently dead before running this: the +operation fences that `server_root_id` out even if the server comes back online, and it deletes +namespace and drain state that cannot be recovered. Check `system.cas_mounts` for the member's +`state` and `last_success_age_seconds` first — a `live` row, or one with a recent lease renewal, +means the member is not a decommission candidate yet. + +**Verification.** After the command reports `slot_removed = true` with no warnings, the member's +`server_root_id` no longer appears as a row in `system.cas_mounts` on any peer, and a subsequent +`SYSTEM CAS GC RUN` on the pool will no longer wait on or fence its heartbeat. See +[mount, unmount, crash](/antalya/cas/architecture/mounts-and-leases#mount-lifecycle) for how the +claim, drain, and retirement steps fit into the mount-slot lifecycle. diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md new file mode 100644 index 000000000000..58600e8e058c --- /dev/null +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -0,0 +1,150 @@ +--- +description: 'The three content-addressed system tables, a key-metrics table with healthy ranges, and queries for reading GC health from cas_gc_log.' +sidebar_label: 'Monitoring' +sidebar_position: 2 +slug: /antalya/cas/operations/monitoring +title: 'CAS Operations — Monitoring' +doc_type: 'guide' +--- + +# Operations — monitoring {#monitoring} + +Content-addressed (`CAS`) storage exposes three system tables and a family of `CAS`-prefixed +`ProfileEvents`. This page is the entry point for day-to-day health checks; see +[debugging](/antalya/cas/operations/debugging) for incident-time tooling and +[troubleshooting](/antalya/cas/operations/troubleshooting) for symptom-driven diagnosis. + +## The three system tables {#system-tables} + +| Table | Grain | Use it for | +|---|---|---| +| [`system.cas_mounts`](/operations/system-tables/cas_mounts) | One row per mount slot in the pool, read live from the backend on every query | Who is in the pool right now, lease/epoch state, which node holds GC leadership | +| [`system.cas_gc_log`](/operations/system-tables/cas_gc_log) | One `Start`/`Finish` row per GC round, plus one `Phase` row per phase reached | GC round outcomes, duration, and where a round's `LIST`/`GET`/`PUT`/`DELETE` budget went | +| [`system.cas_log`](/operations/system-tables/cas_log) | One row per writer/GC decision (blob puts, dedup adoptions, retire decisions, dangling-access findings) | Fine-grained forensics for one part, one blob hash, or one round | + +`system.cas_mounts` is the only one of the three with no persisted backing log — it is a live view, +so a transient backend error on one disk is skipped rather than blinding the whole query. The other +two are ordinary `system.*_log` tables and follow the usual flush/retention settings. + +## Key metrics {#key-metrics} + +Every `CAS`-related `ProfileEvent` carries the uppercase `CAS`/`CASGC` prefix. This is a curated +subset for a first health pass; the full list groups by object class (`CASBlob*`, `CASManifest*`, +`CASRoot*`, `CASGC*`, `CASServer*`, `CASOther*`, `CASRef*`, `CASMeta*`) and is enumerated in +`src/Common/ProfileEvents.cpp`. + +| Metric | Healthy range | A spike or nonzero means | +|---|---|---| +| `CASBlobCompareSwapConflict` | Near zero relative to `CASBlobCompareSwap` | Concurrent-update contention on blob metadata | +| `CASBlobHead` / `CASBlobHeadMiss` | Aggregate present/missing outcomes for every successful backend `HEAD` under `/blobs/`, across writer, `GC`, validation, and other callers | Global totals do not by themselves measure the one-`HEAD` writer budget or diagnose retries; attribute by query and path before drawing either conclusion | +| `CASBlobBodyPutAvoided` | Safe writer observations increment it when a physical body publication is avoided | Compare with query-attributed writer materialization tasks; the aggregate HEAD counters include unrelated callers | +| `CASRefAppendWedged` | Zero | A ref-log append lane exhausted its retries after an uncertain `PUT`; ref-log progress on that namespace may be stalled | +| `CASRefNeedsRecovery` | Zero | A ref-append lane could not install a known-durable transaction and now refuses writes, snapshots, and confirmation until durable replay completes | +| `CASRefAppendSealRejected` | Occasional (a deposed writer losing a race is the protocol working); sustained growth is not | A writer keeps retrying after losing its mount and does not yet know it | +| `CASGCHeartbeatFenceOuts` | Zero on a healthy pool | GC fenced an expired mount; check `system.cas_mounts` for a member that should have cleanly unmounted | +| `CASGCUnmatchedRemoveDeltas` | Occasional (benign per-key no-op by design) | A persistent nonzero rate means removal deltas are reaching the reducer without their matching activation — a correctness signal worth a look, not an automatic false deletion | +| `CASGCCondemnMarkerUnconfirmedCarry` | Zero | A durable condemn marker could not be confirmed; deletion is safely postponed but investigate marker write/read failures | +| `CASGCMetaWriteAnomaly` | Zero | The bounded GC metadata pool failed an operation; backend or pool pressure may delay metadata convergence | +| `CASRefRollbackBestEffortDropFailed` | Zero | A rollback cleanup drop hit a backend failure; refs may remain live and GC may be delayed on that namespace | + +### Mount renewal and remount counters {#mount-renewal-remount-counters} + +These counters separate physical renewal work, logical renewal outcomes, and whole-chain remount +attempts. They are process-global counters, not tagged metrics; compare deltas over the incident +window and correlate them with the `server_root_id` in `system.cas_log`. + +| Metric | Counting dimension | Interpretation | +|---|---|---| +| `CASMountRenewalAttempts` | One per physical conditional renewal `PUT` sent | Physical object-store load; one logical renewal can contribute several | +| `CASMountRenewalRetries` | One per physical renewal `PUT` after the first in the same logical renewal | Positive growth shows in-period retry, not a later cadence beat | +| `CASMountRenewalResolved` | One per logical renewal proved committed by an exact resolving `GET` | A response was ambiguous, but exact bytes and `write_attempt_id` proved the write | +| `CASMountRenewalRecovered` | One per logical renewal committed after a retry or exact resolving `GET` | Recovered object-store blips that retained the existing mount incarnation | +| `CASMountRenewalDeadlineExceeded` | One per logical renewal stopped by the external lease-safety deadline | The last confirmed lease no longer left enough safe time; this is narrower than request-budget exhaustion | +| `CASRemountAttempts` | One per invocation of the existing whole-chain remount attempt | Includes both successful and failed attempts | +| `CASRemountSucceeded` | One per whole-chain attempt that restored `Live` under a fresh writer epoch | Must be a subset of `CASRemountAttempts` | +| `CASRemountFailed` | One per whole-chain attempt that returned without restoring `Live` | Includes a named step exception or a step that returned transiently | + +`CASMountLeaseLost` complements those eight counters. It increments exactly once per operational +`Live -> TransientNotLive` recovery generation: either the initiating external loss or the first +ordinary terminal renewal consumer owns it. A parked terminal result and shutdown do not duplicate +the count. + +To inspect the current cumulative values, including counters that have never incremented: + +```sql +SELECT event, value +FROM system.events +WHERE event IN ( + 'CASMountRenewalAttempts', 'CASMountRenewalRetries', 'CASMountRenewalResolved', + 'CASMountRenewalRecovered', 'CASMountRenewalDeadlineExceeded', 'CASMountLeaseLost', + 'CASRemountAttempts', 'CASRemountSucceeded', 'CASRemountFailed') +SETTINGS system_events_show_zero_values = 1; +``` + +`system.cas_log` records only nontrivial logical renewals. A `watermark_renew` row has outcome +`retrying`, `recovered`, or `failed`, with detail keys `server_root_id`, `writer_epoch`, `seq`, a +shortened `write_attempt_id`, `attempts_sent`, `elapsed_ms`, `remaining_confirmed_budget_ms`, +`unresolved_reason`, `deadline_source`, `stop_cause`, and `classification`. Ordinary first-attempt +success produces no row. Every `mount_remount` attempt produces one final row with outcome `ok` or +`failed` and details `attempt_no`, `step`, `server_root_id`, optional `writer_epoch`, and optional +`error`. + +Default-level text logging is bounded per logical operation: the first ambiguous transition may +emit one retry `WARNING`, followed by one recovery `INFO` or final fence `WARNING`; individual +physical retries remain `DEBUG`. Each whole-chain remount attempt emits one final default-level line +with its attempt number and last/current step. Use the structured rows for correlation instead of +counting backend-attempt log lines. + +Two counter-reading caveats that apply to `system.events`-backed metrics generally, not only `CAS` +ones: a counter that has never incremented can be **absent** from `system.events` rather than +present at zero — query with `system_events_show_zero_values = 1` to tell "never happened" from "not +shown". A server restart resets `system.events` to zero, so a cumulative `CAS` total across a +restart has to be computed from summed per-second deltas in `system.metric_log`, not read directly +off `system.events`. + +## Reading GC health from cas_gc_log {#gc-health} + +Round outcomes over the last day, per disk: + +```sql +SELECT disk_name, outcome, count() AS rounds, avg(duration_ms) AS avg_ms +FROM system.cas_gc_log +WHERE event_type = 'Finish' AND event_time > now() - INTERVAL 1 DAY +GROUP BY disk_name, outcome +ORDER BY disk_name, rounds DESC; +``` + +A steady stream of `Success` and `Deferred` rows is healthy; `Deferred` means the round found no +changed shard needing a fold and no graduation was due — a cheap round, not a stuck one (see +[the round](/antalya/cas/architecture/garbage-collection#the-round)). Recurring `Error` rows, or +`NotALeader` outcomes for the disk's own scheduler, warrant investigation. `anomalies` in the +`Finish` row is worth a steady watch: it is fold clamps surfaced and survived, so a non-zero value +that persists across rounds is more interesting than an isolated one. + +Which phase dominates round duration or the `LIST` budget — reproduced from the +[per-phase rows](/operations/system-tables/cas_gc_log#per-phase-rows) reference: + +```sql +SELECT phase, + count() AS rounds, + quantile(0.99)(phase_duration_microseconds) AS p99_microseconds, + sum(ProfileEvents['S3ListObjects']) AS lists +FROM system.cas_gc_log +WHERE event_type = 'Phase' AND disk_name = 'cas' +GROUP BY phase +ORDER BY p99_microseconds DESC; +``` + +Pending-reclaim backlog and time since a disk's GC last led, from the live mount view: + +```sql +SELECT disk, server_root_id, is_leader, pending_reclaim, last_success_age_seconds, wedged_namespace_count +FROM system.cas_mounts +WHERE is_leader IS NOT NULL +ORDER BY disk, server_root_id; +``` + +`is_leader`, `pending_reclaim`, `last_success_age_seconds`, and `wedged_namespace_count` are +process-local — `NULL` on every row describing a peer's mount — so this query is only informative +run against the node whose GC leadership you are checking; run it on each node to see the whole +pool's view of itself. diff --git a/docs/en/antalya/cas/operations/troubleshooting.md b/docs/en/antalya/cas/operations/troubleshooting.md new file mode 100644 index 000000000000..be541add39f8 --- /dev/null +++ b/docs/en/antalya/cas/operations/troubleshooting.md @@ -0,0 +1,66 @@ +--- +description: 'Symptom-to-action table for common content-addressed storage incidents: mount lease loss, stalled GC, startup failures, fsck timeouts, and read-only pools.' +sidebar_label: 'Troubleshooting' +sidebar_position: 3 +slug: /antalya/cas/operations/troubleshooting +title: 'CAS Operations — Troubleshooting' +doc_type: 'guide' +--- + +# Operations — troubleshooting {#troubleshooting} + +Start from the symptom, not the mechanism. Each row below names a concrete diagnostic query or +command and the action it points to; see [monitoring](/antalya/cas/operations/monitoring) for the +system tables referenced and [debugging](/antalya/cas/operations/debugging) for the underlying +tools. + +| Symptom | Diagnosis | Action | +|---|---|---| +| A server keeps losing its mount lease and self-remounting | Check `system.cas_mounts` for the server's own `state`/`expires_at`, then correlate `watermark_renew` and `mount_remount` in `system.cas_log`; losing the lease trips a local fence and latches a remount generation | Read `classification`, `deadline_source`, and `stop_cause` before changing anything. Look for object-store latency consuming the confirmed lease or BOOTTIME advancement; see [the decision flow](#mount-renewal-remount-flow) and [the mount lease](/antalya/cas/architecture/mounts-and-leases#mount-lease) | +| Writes slow down or stall under load, with no exception reaching the client | S3 `SlowDown`/`ServiceUnavailable`/`RequestTimeout`/`InternalError` (5xx) responses are not on `CasRequestController`'s definite-failure whitelist (only malformed-request, entity-too-large, and access-denied are), so they classify as `Unresolved` and are retried automatically. Confirm with `sum(ProfileEvents['CASConditionalWriteUnresolved'])` rising alongside `sum(ProfileEvents['CASConditionalWriteAttempts'])` over `system.query_log` for the affected window (or `ProfileEvent_CASConditionalWriteUnresolved` in `system.metric_log` for a cumulative view across queries), and check `system.blob_storage_log` for `disk_name = ''` rows with a nonzero `error_code` around the same window | Nothing to configure per-request: the controller retries the same `(key, bytes)` with capped-exponential backoff (200ms initial, capped at 5s) for up to 16 attempts inside a 90-second operation deadline, and the mount-lease renewer keeps extending the fence across the disruption — this is the "blips, throttling, partial outages" case the write path is built to survive. Confirm the mount lease itself is still renewing (`system.cas_mounts.expires_at` moving forward, `last_success_age_seconds` not climbing) — if it is, this is expected and self-resolving. If `SlowDown` responses are sustained rather than transient, check the bucket's request-rate limits against the pool's actual PUT/GET rate (see [bucket requirements](/antalya/cas/bucket-requirements)) and consider lowering `cas_blob_upload_pool_size` to reduce concurrent upload traffic; a write only surfaces a client-visible `NETWORK_ERROR` if the 90-second deadline is exhausted before the store recovers, and that error is retried by the ordinary merge/insert backoff, not silently dropped | +| `GC` never seems to reclaim space after tables are dropped | `SELECT * FROM system.cas_gc_log WHERE event_type='Finish' ORDER BY event_time DESC LIMIT 5` — check `outcome`; also `SELECT is_leader FROM system.cas_mounts` on this node | If `outcome != 'Success'`/`'Deferred'`, see [reading GC health](/antalya/cas/operations/monitoring#gc-health); if this node is not the leader (`is_leader = 0`), it never reclaims for this disk — check the peer holding leadership. Reclamation also needs at least two full rounds past condemnation by design (the grace period is rounds, not acks) — a single manual `SYSTEM CAS GC RUN` will not finish it | +| A dangling-access exception or `CORRUPTED_DATA` on read | Run `clickhouse-disks cas-fsck --detail` and check `dangling` specifically — it is the one class that means data loss, distinct from `unreachable`/`awaiting-gc`, which are just waiting for graduation | A nonzero `dangling` count is a real incident: collect the `--detail` output (see [what to collect before filing a bug](/antalya/cas/operations/debugging#filing-a-bug)) before taking any destructive action | +| `SYSTEM CAS FSCK` or `clickhouse-disks cas-fsck` times out on a large pool | The scan is bounded by `--timeout` (default 600s / the `SYSTEM` form has no override); a large `roots/` prefix can make the scan slow | Retry with `--partial` to see the counts accumulated so far instead of aborting empty-handed, or `--namespace ` to scope the scan to a subset of namespaces | +| `SYSTEM CAS DROP POOL MEMBER` returns a non-empty `warnings` column | A per-object drain step could not confirm emptiness; the mount slot is left terminated but not fully drained, as a resume anchor | Rerun the same command — it is resumable and skips namespaces already marked removed, reporting them under `namespaces_already_removed` | +| Writes or `ALTER`s on a `CAS` disk fail with a `READONLY`-class error | The disk's metadata storage rejects every mutating entry point; this is deliberate for a disk opened with `true`, used by every offline `clickhouse-disks` tool | Confirm whether the disk was intentionally configured read-only (offline inspection, `cas-fsck`, `cas-gc-dryrun`, `cas-gc-rebuild`, `cas-drop-member` all require it); a production disk serving writes must not carry `true` | +| A table stays unavailable after a transient network error during startup | `AsyncLoader` has no retry/requeue path for a failed table load job: a transient S3 `NETWORK_ERROR` during `CAS` ref-table startup recovery can leave the job permanently `FAILED` | Restart the server, or issue a fresh load for the table; this is a one-shot job design, not a `CAS`-specific bug | +| A mounted pool directory was removed or renamed out of band | Renewal observes an absent, foreign, successor, or otherwise conflicting mount body and terminates the keeper with a typed fail-closed exception; the runtime closes the local write fence and requests remount rather than adopting the body | Never remove or rename a live pool's storage path. To retire a member permanently use [`SYSTEM CAS DROP POOL MEMBER`](/antalya/cas/operations/migration#decommission) instead of raw filesystem operations; collect the `watermark_renew` classification and subsequent `mount_remount` step | +| Stale-looking part metadata after an out-of-band change to the pool | The part-folder view cache may be serving a retained (not re-validated) view | Set the disk-level `cas_part_folder_cache_bytes = 0` as a diagnostic kill switch to disable retention, and run `fsck`/integrity checks with `cas_part_folder_validate = always` so every read re-proves the body | +| A wide merge (many thousands of columns) fails with a port-exhaustion error from the network layer | Each column in a wide part can cost a separate object-store operation in one merge, and a very wide part can issue on the order of the column count in requests, exhausting local ephemeral TCP ports under load | Reduce concurrent merge parallelism on that table, or increase the host's ephemeral port range; this is a general high-fan-out-merge limit, not specific to content addressing | + +## Mount renewal and remount decision flow {#mount-renewal-remount-flow} + +Start with the `watermark_renew` timeline described in +[debugging](/antalya/cas/operations/debugging#trace-renewal-remount), then follow the matching case: + +1. **Recovered blip.** `retrying` is followed by `recovered` for the same shortened + `write_attempt_id`; `CASMountRenewalRecovered` rises while `CASMountLeaseLost` and all remount + counters stay flat. No intervention is needed unless the rate is sustained; investigate backend + throttling/latency before the blips consume the lease budget. +2. **External lease-safety exhaustion.** The failed row has + `classification = 'external_lease_deadline'` and + `deadline_source = 'external_lease_safety'`; `CASMountRenewalDeadlineExceeded` and + `CASMountLeaseLost` rise. The runtime correctly refused to manufacture authority beyond the last + confirmed lease. Check object-store latency and BOOTTIME/suspend history, then follow the ensuing + remount. +3. **Cancellation.** `stop_cause = 'cancelled'` after a sent request is terminal and suppresses a + clean farewell because the request may still land. Cancellation before any request is + `NotAttempted`, remains `Active`, and emits no failed aggregate row; during graceful shutdown that + is the expected clean-release path. +4. **Confirmed conflict.** `classification = 'conflict'` means exact resolution found another body; + inspect `server_root_id`, `writer_epoch`, `seq`, and `write_attempt_id`. Same-pair twins, GC-fenced + bodies, successor epochs, and foreign holders all remain fail closed. Do not delete or rewrite the + mount key by hand. +5. **Fence or lifecycle loss.** `stop_cause = 'fence_or_lifecycle_lost'` means another local loss, + remount park request, or terminal lifecycle closed admission while the operation was active. A + parked result reuses the already-requested recovery generation and must not double-count + `CASMountLeaseLost`. +6. **Whole-chain remount failure.** Read the following `mount_remount` row. Its `attempt_no`, `step`, + and optional `error` identify the failed owner/catalog/epoch/claim/install/quiescence/fence step. + The current protocol retries the whole chain with bounded backoff; it does not preserve per-step + progress. Repeated failure at the same step is the actionable signal. + +The default-level log policy is intentionally bounded: one warning on the first transition to retry, +then one recovery info or terminal fence warning, plus one final line per whole-chain remount attempt. +Use `system.cas_log` and counter deltas to reconstruct the incident; `DEBUG` contains individual +physical retries when that extra transport detail is necessary. diff --git a/docs/en/antalya/cas/quick-start.md b/docs/en/antalya/cas/quick-start.md new file mode 100644 index 000000000000..a54d9345fbb0 --- /dev/null +++ b/docs/en/antalya/cas/quick-start.md @@ -0,0 +1,144 @@ +--- +description: 'A minimal content-addressed storage disk config and the first CREATE TABLE, INSERT, and SELECT against it, executed live before publication.' +sidebar_label: 'Quick start' +sidebar_position: 2 +slug: /antalya/cas/quick-start +title: 'CAS Quick Start' +doc_type: 'guide' +--- + +# Quick start {#quick-start} + +## The disk config {#disk-config} + +A `CAS` disk is an `object_storage` disk with `metadata_type` set to `cas` and an explicit, +per-server `cas_server_root_id`. This example uses the `local` object-storage backend so it needs +nothing beyond a `ClickHouse` binary — no bucket, no credentials: + +```xml + + + + + object_storage + local + cas + quickstart-demo + cas_pool/ + + + cache + cas + cas_cache/ + 10Gi + + + + + +
+ cas_cache +
+
+
+
+
+
+``` + +The `cas_cache` disk layers a local filesystem cache over `cas`: it absorbs repeated reads of the +same blob while `cas` stays the source of truth, and the policy's volume points at the cached disk +— see [configuration](/antalya/cas/configuration#disk-config) for the sizing note. + +`cas_server_root_id` must be unique per server sharing a pool. On a single, non-replicated server a +literal string, as above, is enough; on a replicated cluster where every replica shares one config, +`{replica}` expands through the same macro substitution an `s3` +disk's `endpoint` already uses, giving each replica a distinct subtree from one template. + +**S3 endpoint variant.** Swap `object_storage_type` to `s3` and add the usual object-storage +connection keys; nothing else in this config changes: + +```xml + + object_storage + s3 + cas + quickstart-demo + https://bucket.s3.amazonaws.com/cas/ + ... + ... + +``` + +`cas_cache` is unaffected by this swap — it wraps `disk cas` regardless of which object-storage +backend `cas` itself uses. See [bucket requirements](/antalya/cas/bucket-requirements) for what the +target bucket needs to support, and [configuration](/antalya/cas/configuration) for the full +settings surface. + +## First table {#first-table} + +```sql +CREATE TABLE events (event_date Date, event_id UInt64, payload String) +ENGINE = MergeTree ORDER BY event_id +SETTINGS storage_policy = 'cas'; + +INSERT INTO events VALUES ('2026-08-04', 1, 'hello'), ('2026-08-04', 2, 'world'); + +SELECT * FROM events ORDER BY event_id; +``` + +```text + ┌─event_date─┬─event_id─┬─payload─┐ +1. │ 2026-08-04 │ 1 │ hello │ +2. │ 2026-08-04 │ 2 │ world │ + └────────────┴──────────┴─────────┘ +``` + +An ordinary `MergeTree` table on a `CAS` disk. `INSERT`, `SELECT`, merges, and mutations all work +exactly as on any other `MergeTree` — the content-addressing is invisible at the SQL surface. + +## Checking the mount {#checking-the-mount} + +```sql +SELECT disk, server_root_id, state, is_leader FROM system.cas_mounts; +``` + +```text +Row 1: +────── +disk: cas +server_root_id: quickstart-demo +state: live +is_leader: 0 + +Row 2: +────── +disk: cas_cache +server_root_id: quickstart-demo +state: live +is_leader: 0 +``` + +`system.cas_mounts` shows every server currently sharing this pool, not just the local one. With a +cache layered in front, the same mount shows up **twice** — once under each configured disk name +(`cas` and `cas_cache`), both reporting the one underlying `server_root_id` — because the table +lists a row per configured disk, not per mount; this is the one visible change the cache layer adds +to this page's output. `is_leader` is `0` on both rows because `GC` leader election is asynchronous +and had not yet run at query time on this freshly mounted disk — see +[mounts and leases](/antalya/cas/architecture/mounts-and-leases) for the full column reference and +[garbage collection](/antalya/cas/architecture/garbage-collection) for leadership. + +## What just happened {#what-happened} + +The `INSERT` wrote two part files as content-addressed blobs, a part manifest listing them, and a +ref pointing the part name at that manifest — the only mutable object the write touched. On a +second replica sharing this same pool, inserting or fetching the identical content publishes a ref +without re-uploading a single byte; see +[garbage collection](/antalya/cas/architecture/garbage-collection) for how a dropped part's blobs +get reclaimed once nothing references them anymore. + +This exact cache-layered configuration and SQL were run against a live server before publication: +`CREATE TABLE`, `INSERT`, `SELECT`, and the `system.cas_mounts` query above all completed with zero +errors, with the two-row `system.cas_mounts` output shown above captured from that run. The +`INSERT`/`SELECT` output is unaffected by the cache — the one visible difference the cache layer +adds anywhere on this page is that second `system.cas_mounts` row. diff --git a/docs/en/antalya/cas/roadmap.md b/docs/en/antalya/cas/roadmap.md new file mode 100644 index 000000000000..d9675cf9e3e5 --- /dev/null +++ b/docs/en/antalya/cas/roadmap.md @@ -0,0 +1,116 @@ +--- +description: 'What CAS ships today, what is still planned, known platform limitations, and design directions deliberately not taken.' +sidebar_label: 'Roadmap' +sidebar_position: 5 +slug: /antalya/cas/roadmap +title: 'CAS Roadmap' +doc_type: 'guide' +--- + +# CAS roadmap {#cas-roadmap} + +CAS is experimental (see [status](/antalya/cas/)): the format and SQL surface can still change. +This page tracks what already works, what is still ahead, and — since a project this deep in +adversarial verification collects real dead ends — what was tried and deliberately not shipped. + +## Shipped {#shipped} + +**Storage and object model.** Content-addressed blobs deduplicated across every replica sharing +a pool; immutable part manifests; a pluggable blob-hash algorithm (`cityhash128` default, +`xxh3-128`, or `sha256`) fixed per pool at creation; a JSON-text object format end to end (no +binary framing, no protobuf) so any object can be read with ordinary line-oriented tools. + +**Write path.** Conditional writes for mutable metadata/control objects; mandatory blob `HEAD` +followed by adoption or unconditional, multipart-capable publication; a bounded thread pool +fanning out multi-blob part materialization in parallel; carry-forward on mutation for `Wide` parts (an +untouched column is re-referenced, not re-hashed). + +**Read path.** Ref resolution to manifest to ranged blob reads, with a manifest-decode cache and +a part-folder view cache sitting on that path. + +**Replication.** Fetch by relink between replicas sharing a pool — a replicated fetch publishes a +ref pointing at blobs the pool already has, at zero bytes on the wire — with a publish-then-confirm +protocol that closes the sender-crash and stale-cache races a naive relink would be exposed to. + +**Garbage collection.** An 18-phase round built on a causal ack-floor (no separate fence-and-recheck +phase); sharded folding (`cas_gc_shards`); condemn/spare bookkeeping; generation pruning with a +configurable retention window; a dry-run mode and a rebuild path for recovery. + +**Mounts and identity.** Explicit `cas_server_root_id` per disk; a renewable mount lease with +observation-based reclaim of an expired predecessor (never trusting a foreign body's wall-clock +timestamp); clean decommission of a permanently departed pool member +(`SYSTEM CAS DROP POOL MEMBER`). + +**Backends.** AWS S3 (`ETag`-based conditional dialect) is validated. Google Cloud Storage's +generation-token implementation and deterministic tests are complete, but its credentialed +[real-GCS release gate](/superpowers/cas/unconditional-blob-publication-live-results) has not run. A +capability probe runs at every writable mount and refuses a backend that does not enforce the +conditions CAS depends on. + +**Operability.** `system.cas_log`, `system.cas_gc_log`, and `system.cas_mounts` for introspection; +`clickhouse-disks` commands `ca-fsck`, `ca-inspect`, `ca-gc-dryrun`, and `ca-gc-rebuild`; the +`SYSTEM CAS` SQL control surface (`GC RUN`/`STOP`/`START`/`REBUILD`, `FSCK`, `FORGET`, `DROP POOL +MEMBER`). + +**Coexistence.** `metadata_type = cas` is opt-in per disk; zero-copy replication keeps working +unmodified on disks that do not opt in — see [why CAS exists](/antalya/cas/) for the fuller +positioning. + +## In progress / planned {#in-progress} + +- **GCS and Azure real-store validation.** The new GCS publication groups still require credentials; + Azure has no wired CAS conditional dialect. See [known limitations](#known-limitations) below. +- **WORM deployments.** A read-only disk mode exists today; a fuller write-once story — a pool + served immutably, with pinned snapshots for read-only replicas — has a draft design and is not + yet implemented. +- **Backup and restore.** See [Backups](#backups) below — this is further along as a design than as + an implementation. +- **First-class local-disk pools.** Today a pool over local paths runs a minimal best-effort + emulation of the token-conditional mutable-object dialect. Blob publication materializes a whole + object and is serialized to retain a one-body memory bound. Making the + local mode efficient in its own right is under consideration: a local CAS tier is a natural target + for backups, pinned snapshots, and moving data between CAS tiers. + +## Known limitations {#known-limitations} + +- **Azure Blob Storage's REST API documents the equivalent conditional headers CAS needs, but no + CAS conditional-write dialect is wired up for it yet** — untested, not validated by the capability + probe. See [bucket requirements](/antalya/cas/bucket-requirements) and + [the backend page](/antalya/cas/architecture/backend) for the AWS/GCS dialects that are wired. +- **Other S3-compatible object stores qualify only if they pass the capability probe** — a store + that silently ignores conditional writes is refused at mount time rather than trusted. Bucket + versioning must be off; it is not required to be on. +- **An `encrypted` disk wrapping a `CAS` disk is not supported yet** — `CREATE TABLE` on such a + disk succeeds, but the first `INSERT` fails (`Autocommit writes are not supported for content part + files on a content-addressed disk`). Layering a `cache` disk in front of `CAS` is the supported + wrapper shape (see [configuration](/antalya/cas/configuration)); encryption at rest currently has + to come from the object store side. +- **The format and settings surface can still change.** CAS is pre-release: there is no persisted + production data to keep compatible, so a format change costs a version bump, not a migration. + Treat every detail on these pages as subject to change until the format is declared stable. + +## Backups {#backups} + +A `snapshot` / `mirror` / `fetch` / `restore` design is **approved but not implemented**. The +model is deliberately git-shaped: `snapshot` is instant and free (like `git tag` — it references +existing manifests, copies nothing); `mirror` is a continuous pull from a production pool into a +backup pool (like `git push --mirror`); `fetch` is a selective pull from a backup pool into a +fresh pool (a partial clone); `restore` is an in-pool relink (like `git checkout`, instant). One +closure-walk-and-hash-verification primitive is meant to serve all three pool-to-pool movements. +None of this is wired into the `BACKUP`/`RESTORE` SQL surface yet. + +## Deliberately rejected directions {#rejected} + +A short pointer list; the reasons and the counterexamples that drove each decision are in +[design history](/antalya/cas/architecture/design-history). + +- A Merkle tree layer as a distinct object kind. +- Epoch-based reclamation as the GC core. +- An integer in-degree refcount instead of a folded edge set. +- A persistent, append-only namespace registry for GC discovery. +- Per-incarnation body keys as an alternative to an in-body incarnation tag. +- Using a blob's freshness metadata as the authority for its lifecycle instead of an advisory hint. +- A separate all-shard fence-and-recheck phase per GC round. +- A sparse ref-id allocator with a certificate stack bolted on to prove completeness. +- Extending zero-copy replication instead of building a new mechanism — CAS is an alternative to + zero-copy, not a replacement; both remain available. diff --git a/docs/en/operations/system-tables/cas_gc_log.md b/docs/en/operations/system-tables/cas_gc_log.md new file mode 100644 index 000000000000..5fd04b4fb11d --- /dev/null +++ b/docs/en/operations/system-tables/cas_gc_log.md @@ -0,0 +1,156 @@ +--- +description: 'System table containing per-round records of the content-addressed (CAS) MergeTree garbage collector.' +sidebar_label: 'cas_gc_log' +sidebar_position: 30 +slug: /operations/system-tables/cas_gc_log +title: 'system.cas_gc_log' +doc_type: 'reference' +--- + +## Description {#description} + +The `system.cas_gc_log` table contains per-round records of the +content-addressed (CAS) MergeTree garbage collector. For every garbage-collection round it stores a +`Start` row and a `Finish` row (like `system.part_log` stores events per data part), with the counts +of objects marked and deleted, the round duration, the outcome, and a per-round `ProfileEvents` +delta. + +Between them it also stores one `Phase` row per GC phase the round reached, each carrying that +phase's own duration, its `ProfileEvents` delta, and its phase-specific counts. All rows of one round +share a `round_id`. See [Per-phase rows](#per-phase-rows). + +Rounds are emitted both by the background GC scheduler (`trigger = 'Scheduled'`) and by the +synchronous [`SYSTEM CAS GC RUN`](/sql-reference/statements/system#system-cas-gc-run) +command (`trigger = 'Manual'`). + +The table is created only if the `cas_gc_log` server setting is +specified (it is enabled by default in the shipped `config.xml`). + +## Columns {#columns} + +- `hostname` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Host name of the server executing the round. +- `event_date` ([Date](/sql-reference/data-types/date)) — Event date. +- `event_time` ([DateTime](/sql-reference/data-types/datetime)) — Event time. +- `event_time_microseconds` ([DateTime64(6)](/sql-reference/data-types/datetime64)) — Event time with microseconds precision. +- `event_type` ([Enum8](/sql-reference/data-types/enum)) — `Start` or `Finish` of a GC round, or one `Phase` of it. +- `disk_name` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The content-addressed disk the round ran on. +- `server_root_id` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Identifies the mount whose GC scheduler ran this round. Distinguishes concurrent mounters of the same shared pool; join on this column when correlating rounds against [`system.cas_mounts`](/operations/system-tables/cas_mounts). +- `gc_id` ([String](/sql-reference/data-types/string)) — The GC scheduler instance id (which mounter ran the round). +- `trigger` ([Enum8](/sql-reference/data-types/enum)) — `Scheduled` (background tick) or `Manual` (`SYSTEM` command). +- `round` ([UInt64](/sql-reference/data-types/int-uint)) — The GC round number (`0` on a `Start` row). +- `outcome` ([Enum8](/sql-reference/data-types/enum)) — `Unknown` (on a `Start` row), `Success` (led, folded, and completed), `NotALeader` (another replica holds the GC lease), `Deferred` (led but took the skip-unchanged fast path — no fold ran, because no changed shard reached the fold threshold and no graduation was due), or `Error` (the round threw). +- `candidates_marked` ([UInt64](/sql-reference/data-types/int-uint)) — Objects retired (marked) this round. +- `objects_deleted` ([UInt64](/sql-reference/data-types/int-uint)) — Objects physically deleted this round. +- `objects_absent` ([UInt64](/sql-reference/data-types/int-uint)) — Retire candidates found already absent. +- `objects_replaced` ([UInt64](/sql-reference/data-types/int-uint)) — `412`-saves (a resurrection won the race against the delete). +- `objects_spared` ([UInt64](/sql-reference/data-types/int-uint)) — Candidates spared because their in-degree was greater than zero at recheck. +- `manifests_deleted` ([UInt64](/sql-reference/data-types/int-uint)) — Owner-removed manifest bodies physically deleted this round, counted separately from blob deletes. +- `entries_condemned` ([UInt64](/sql-reference/data-types/int-uint)) — Retired entries newly condemned this round (retired-cursor pipeline stage 1). +- `entries_graduated` ([UInt64](/sql-reference/data-types/int-uint)) — Retired entries newly floor-passed and republished `delete_pending` this round (pipeline stage 2; deleted the next round). +- `entries_redeleted` ([UInt64](/sql-reference/data-types/int-uint)) — Pending exact-token blob deletes executed this round (pipeline stage 3). +- `fence_outs` ([UInt64](/sql-reference/data-types/int-uint)) — Expired mounts fenced out by this round's heartbeat floor. +- `anomalies` ([UInt64](/sql-reference/data-types/int-uint)) — Fold clamps surfaced (and survived) this round. A steady non-zero value warrants a look at the round log details. +- `duration_ms` ([UInt64](/sql-reference/data-types/int-uint)) — The round wall-clock duration (on a `Finish` row). +- `error` ([String](/sql-reference/data-types/string)) — The exception text when `outcome = 'Error'`. +- `ProfileEvents` ([Map(LowCardinality(String), UInt64)](/sql-reference/data-types/map)) — On a `Start`/`Finish` row, the per-round `ProfileEvents` delta (the `CAS*` counters and S3/disk events for this round). On a `Phase` row, **that phase's** delta, so `GROUP BY phase` over `ProfileEvents['S3ListObjects']` attributes the round's `LIST` budget to the phase that spent it. +- `round_id` ([String](/sql-reference/data-types/string)) — The correlator for every row of one round attempt: its `Start`, each of its `Phase` rows, and its `Finish`. Minted per attempt, so unlike `round` it exists even for a round that never committed and for a round that never led. Group by this column to reconstruct one round. +- `phase` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The GC phase this row describes; empty on `Start`/`Finish`. See [Per-phase rows](#per-phase-rows) for the phase list. +- `phase_duration_microseconds` ([UInt64](/sql-reference/data-types/int-uint)) — The wall-clock duration of this phase, in microseconds (`Phase` rows only). Microseconds rather than milliseconds because several phases are routinely sub-millisecond and the point of the row is to see when they are not. +- `phase_metrics` ([Map(LowCardinality(String), UInt64)](/sql-reference/data-types/map)) — Phase-specific semantic counts (`Phase` rows only) that a phase computes for itself and no `ProfileEvents` counter can supply. The verb counts ride the `ProfileEvents` column of the same row. + +## Per-phase rows {#per-phase-rows} + +Besides the `Start` and `Finish` row of each round, the collector emits one `Phase` row per GC phase. +Every row of one round attempt — `Start`, each `Phase`, and `Finish` — shares a `round_id`. A round +that defers, or that never acquires the GC lease, emits only the phases it actually reached; a round +that throws still emits the row of the phase it died in. + +The phases, in execution order: + +| `phase` | What it covers | Dominant I/O | +|---|---|---| +| `lease` | Acquire, renew, or observe the GC lease. The only phase a `NotALeader` round emits. | `gc/state` `GET` + compare-and-swap | +| `pre_fold_ref_drain` | Resolve catalog rows whose terminal fold evidence is already adopted before this invocation publishes or defers. | catalog `GET` + exact compare-and-swap | +| `heartbeat_floor` | Classify every mount slot and fence out the dead ones. | `LIST` of the mount prefix, one `GET` per mount, one `PUT` per fence | +| `defer_decision` | The skip-unchanged decision: graduation check plus the round's one enumeration of the ref prefix. | one full ref-prefix `LIST`, two fold-seal `GET`s | +| `parent_seal_read` | Capture the pre-fold seal's run refs for the hand-off reclaim. | one fold-seal `GET` | +| `fold_ref_group` | Regroup the round's enumeration into per-table listings — what this round will fold. | none | +| `fold_seal_read` | The adopted fold seal, read twice at the same generation and attempt. | two fold-seal `GET`s | +| `fold_ref_intake` | Read and fold every new ref log and the manifest bodies its edges name. | one `GET` per new log, one `GET` per manifest edge | +| `fold_reduce` | The per-shard in-degree merge: condemn, spare, graduate. | prior-run streaming `GET`s, one `HEAD` per zero-transition candidate, run `PUT`s | +| `fold_seal_write` | Publish the new fold seal. | one `PUT` | +| `pending_deletes` | The single content-delete site: exact-token deletes of previously published `delete_pending` entries, plus the outcome logs. | one `DELETE` per entry, one outcome-log `PUT` per shard | +| `meta_pool_wait` | Drain the round's per-hash freshness-meta writes. | none on this thread — see the caveat below | +| `round_commit` | The generation-retention prune and the round's single `gc/state` compare-and-swap. | prune `LIST`s and deletes, one compare-and-swap | +| `handoff_reclaim` | Wholesale-reclaim generations a moved run ref stranded below the retention cursor. | prefix `LIST`s and deletes | +| `manifest_deletes` | Exact-token deletes of owner-removed manifest bodies, after their decrements were adopted. | one `DELETE` per body | +| `namespace_cleanup` | Run one bounded `cas/ns/` page across the stream and state subtrees for the perpetual dead-life janitor. This phase is physical reclamation, not a lifecycle gate. | one namespace-root page `LIST`, catalog cut, exact-token deletes | +| `ref_object_cleanup` | Delete ref logs covered by both the durable fold cursor and a durable snapshot, plus superseded snapshots. | one `HEAD` + one `DELETE` per deletable object | +| `orphan_sweep` | The budgeted, cursor-paced orphan part-manifest backstop. | budgeted `LIST` and deletes | + +Which phase dominates a round: + +```sql +SELECT phase, + count() AS rounds, + quantile(0.5)(phase_duration_microseconds) AS p50_microseconds, + quantile(0.99)(phase_duration_microseconds) AS p99_microseconds, + sum(phase_duration_microseconds) AS total_microseconds +FROM system.cas_gc_log +WHERE event_type = 'Phase' AND disk_name = 'ca' +GROUP BY phase +ORDER BY total_microseconds DESC; +``` + +Which phase spends the `LIST` budget: + +```sql +SELECT phase, sum(ProfileEvents['S3ListObjects']) AS lists +FROM system.cas_gc_log +WHERE event_type = 'Phase' AND disk_name = 'ca' +GROUP BY phase +ORDER BY lists DESC; +``` + +One round, in order — including a round that failed, which is why the correlator is `round_id` and +not `round`: + +```sql +SELECT phase, phase_duration_microseconds, phase_metrics, ProfileEvents['S3ListObjects'] AS lists +FROM system.cas_gc_log +WHERE round_id = '...' AND event_type = 'Phase' +ORDER BY event_time_microseconds; +``` + +Two caveats when reading these rows: + +- Work scheduled onto the GC meta pool runs on other threads, so the `meta_pool_wait` row's + `ProfileEvents` delta is **empty by construction**. Read its `phase_metrics` `jobs_scheduled` / + `jobs_completed` next to its duration instead: they distinguish a deep queue from a slow endpoint. +- Phase durations do not sum to the round's `duration_ms`. The round also performs untimed + bookkeeping between phases, and the `Finish` row's `duration_ms` remains the authority on total + round time. + +## Example {#example} + +```sql +SELECT + event_type, + disk_name, + trigger, + outcome, + candidates_marked, + objects_deleted, + duration_ms +FROM system.cas_gc_log +ORDER BY event_time_microseconds DESC +LIMIT 2 +FORMAT Vertical; +``` + +## See Also {#see-also} + +- [`SYSTEM CAS GC RUN`](/sql-reference/statements/system#system-cas-gc-run) — run one GC round synchronously. +- [`system.cas_mounts`](/operations/system-tables/cas_mounts) — live per-`server_root_id` mount and GC-health state. +- [`system.cas_log`](/operations/system-tables/cas_log) — per-decision event log for the CAS garbage collector and writer. +- [`system.part_log`](/operations/system-tables/part_log) — the analogous per-part event log. diff --git a/docs/en/operations/system-tables/cas_log.md b/docs/en/operations/system-tables/cas_log.md new file mode 100644 index 000000000000..17616cd9b329 --- /dev/null +++ b/docs/en/operations/system-tables/cas_log.md @@ -0,0 +1,65 @@ +--- +description: 'System table containing a per-decision event log for the content-addressed (CAS) MergeTree writer and garbage collector.' +sidebar_label: 'cas_log' +sidebar_position: 32 +slug: /operations/system-tables/cas_log +title: 'system.cas_log' +doc_type: 'reference' +--- + +## Description {#description} + +The `system.cas_log` table contains a per-decision event log for the content-addressed +(CAS) MergeTree storage engine: blob puts and dedup adoptions, root/ref transitions, in-degree changes, +garbage-collector retire decisions and recheck verdicts, blob deletes, and dangling-access/corruption +findings. It is a much finer-grained, per-event complement to +[`system.cas_gc_log`](/operations/system-tables/cas_gc_log), +which only records one `Start`/`Finish` row per GC round. + +The table is created only if the `cas_log` server setting is +specified (it is enabled by default in the shipped `config.xml`). + +## Columns {#columns} + +- `hostname` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Host name of the server that emitted the event. +- `event_date` ([Date](/sql-reference/data-types/date)) — Event date. +- `event_time` ([DateTime](/sql-reference/data-types/datetime)) — Event time. +- `event_time_microseconds` ([DateTime64(6)](/sql-reference/data-types/datetime64)) — Event time with microseconds precision. +- `event_type` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The CAS decision/event, e.g. `blob_put`, `blob_reuse_adopt`, `root_remove`, `indegree_zero`, `gc_retire_decision`, `gc_recheck_verdict`, `blob_delete`, `dangling_access`, `corrupt_dangle`. +- `disk_name` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The content-addressed disk / pool the event belongs to. +- `namespace` ([String](/sql-reference/data-types/string)) — `roots/` (server/table); empty if not applicable. +- `ref_name` ([String](/sql-reference/data-types/string)) — Part name / ref the event concerns; empty if not applicable. +- `object_kind` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — One of `none`, `blob`, `manifest`, `root`, `snapshot`. +- `object_hash` ([String](/sql-reference/data-types/string)) — Content hash (lowercase hex) of the object; empty if not applicable. +- `token` ([String](/sql-reference/data-types/string)) — Incarnation token (`ETag`) involved; empty if not applicable. +- `round` ([UInt64](/sql-reference/data-types/int-uint)) — GC round (`0` if not applicable). +- `generation` ([UInt64](/sql-reference/data-types/int-uint)) — GC snapshot generation (`0` if not applicable). +- `at_version` ([UInt64](/sql-reference/data-types/int-uint)) — Manifest `shard_version` of the driving journal record (`0` if not applicable). +- `outcome` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Decision outcome, e.g. `ok`, `adopt`, `resurrect`, `deleted`, `replaced`, `spared`, `absent`, `zeroed`, `skipped`. +- `reason` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Human-readable rationale for the decision. Templated across rows, so it is `LowCardinality`. +- `thread_id` ([UInt64](/sql-reference/data-types/int-uint)) — OS thread that emitted the event. +- `query_id` ([String](/sql-reference/data-types/string)) — Query id for correlation with [`system.query_log`](/operations/system-tables/query_log); empty if not applicable. +- `detail` ([Map(LowCardinality(String), String)](/sql-reference/data-types/map)) — Structured event-specific facts, e.g. `condemn_round`, `superseded_token`, `code`, `site`. + +## Example {#example} + +```sql +SELECT + event_time_microseconds, + event_type, + disk_name, + ref_name, + object_kind, + outcome, + reason +FROM system.cas_log +ORDER BY event_time_microseconds DESC +LIMIT 10 +FORMAT Vertical; +``` + +## See Also {#see-also} + +- [`system.cas_gc_log`](/operations/system-tables/cas_gc_log) — per-round GC event log. +- [`system.cas_mounts`](/operations/system-tables/cas_mounts) — live per-`server_root_id` mount and GC-health state. +- [`system.query_log`](/operations/system-tables/query_log) — correlate via `query_id`. diff --git a/docs/en/operations/system-tables/cas_mounts.md b/docs/en/operations/system-tables/cas_mounts.md new file mode 100644 index 000000000000..5a9465db04af --- /dev/null +++ b/docs/en/operations/system-tables/cas_mounts.md @@ -0,0 +1,81 @@ +--- +description: 'System table containing the live mount and GC-health state of every server mounted onto a content-addressed (CAS) disk pool.' +sidebar_label: 'cas_mounts' +sidebar_position: 31 +slug: /operations/system-tables/cas_mounts +title: 'system.cas_mounts' +doc_type: 'reference' +--- + +## Description {#description} + +The `system.cas_mounts` table contains one row per mount slot discovered on every +content-addressed (CAS) disk configured on the node. A pool may be shared by several servers (or +several `server_root_id` mounts on the same server), and this table lists every mount visible in +the pool's backend at query time, not only the querying server's own mount — it exists for +incident-time diagnosis of leases, epochs, and GC leadership across a shared pool. + +The table is read directly from the CAS disk's backend on every query (there is no persisted log +behind it); a transient backend error on one disk is skipped and does not blind the rest of the +rows. + +## Columns {#columns} + +- `disk` ([String](/sql-reference/data-types/string)) — Name of the content-addressed disk. +- `server_root_id` ([String](/sql-reference/data-types/string)) — Server root id owning the mount slot. +- `server_uuid` ([UUID](/sql-reference/data-types/uuid)) — UUID of the server incarnation holding the lease. +- `hostname` ([String](/sql-reference/data-types/string)) — Hostname recorded in the lease body. +- `process_id` ([UInt64](/sql-reference/data-types/int-uint)) — Process id recorded in the lease body. +- `writer_epoch` ([UInt64](/sql-reference/data-types/int-uint)) — Fenced writer epoch of the incarnation. +- `renewal_sequence` ([UInt64](/sql-reference/data-types/int-uint)) — Lease renewal sequence number. +- `started_at` ([DateTime64(3)](/sql-reference/data-types/datetime64)) — Time when the lease started. +- `expires_at` ([DateTime64(3)](/sql-reference/data-types/datetime64)) — Time when the lease expires. +- `min_active_build_sequence` ([UInt64](/sql-reference/data-types/int-uint)) — Oldest in-flight build sequence (`UINT64_MAX` means the mount said farewell). +- `gc_fenced` ([UInt8](/sql-reference/data-types/int-uint)) — `1` if GC fenced this slot out (terminal). +- `state` ([String](/sql-reference/data-types/string)) — One of `live`, `expired`, `terminated`, `fenced`, `corrupt`. +- `is_leader` ([Nullable(UInt8)](/sql-reference/data-types/nullable)) — `1` if this server's GC scheduler currently holds this disk's leadership lease. +- `pending_reclaim` ([Nullable(Int64)](/sql-reference/data-types/nullable)) — Cumulative two-phase deletion backlog observed by this process's GC on this disk (condemned entries minus executed exact-token deletes). +- `last_success_age_seconds` ([Nullable(UInt64)](/sql-reference/data-types/nullable)) — Seconds since this disk's GC last led a round (`0` if it has never led or GC is not running here). +- `wedged_namespace_count` ([Nullable(UInt64)](/sql-reference/data-types/nullable)) — Ref-append lanes currently wedged on this disk (an uncertain `PUT` exhausted its retry budget). +- `lifecycle` ([String](/sql-reference/data-types/string)) — This server's content-addressed pool lifecycle for the disk (a non-gated snapshot, always populated so a not-live disk stays visible): one of `live`, `not_live`, `identity_lost`, `vanished`, `constructing` (never started), or `shutdown` (torn down). +- `lifecycle_reason` ([String](/sql-reference/data-types/string)) — The enum-clean sub-state word for a `vanished` disk: `replaced` or `forgotten`. Empty for every other lifecycle, so `lifecycle || '(' || lifecycle_reason || ')'` reads e.g. `vanished(forgotten)`. +- `lifecycle_detail` ([String](/sql-reference/data-types/string)) — The full typed reason text naming the actual cause when not live: the vanish diagnosis (a data root replaced by a foreign pool, or decommissioned by `SYSTEM CAS FORGET` at a given time) or the identity-loss message. Empty when live. +- `lifecycle_since` ([Nullable(DateTime)](/sql-reference/data-types/nullable)) — When this server entered the current non-live lifecycle state. `NULL` when live, or when the state has no backing pool to date from. + +`lifecycle`/`lifecycle_reason`/`lifecycle_detail`/`lifecycle_since` are the SQL surface for +diagnosing an identity-lost or forgotten disk without reading server logs — see the +`IdentityLost`/`VanishedReplaced`/`VanishedForgotten` states on the +[mount-slot behavioral model](/antalya/cas/architecture/mounts-and-leases#mount-state-machines) for +what each lifecycle value means, and [`SYSTEM CAS FORGET`](/antalya/cas/operations/debugging#sql-forget) +for the command that produces `vanished(forgotten)`. + +:::note Local-only GC-health columns +`is_leader`, `pending_reclaim`, `last_success_age_seconds`, and `wedged_namespace_count` are process-local +facts about *this* server's own GC scheduler. They are populated **only** on the row whose `server_root_id` matches +this server's own mount, and are `NULL` on every row describing another server's mount — stamping a local +health fact onto a peer's row would misread as "the peer is the GC leader" during an incident. To see the +peer's own view of these columns, query `system.cas_mounts` on that server. +::: + +## Example {#example} + +```sql +SELECT + disk, + server_root_id, + state, + writer_epoch, + is_leader, + pending_reclaim, + last_success_age_seconds +FROM system.cas_mounts +ORDER BY disk, server_root_id +FORMAT Vertical; +``` + +## See Also {#see-also} + +- [`system.cas_gc_log`](/operations/system-tables/cas_gc_log) — per-round GC event log. +- [`system.cas_log`](/operations/system-tables/cas_log) — per-decision event log for the CAS garbage collector and writer. +- [`SYSTEM CAS GC RUN`](/sql-reference/statements/system#system-cas-gc-run) — run one GC round synchronously. +- [`SYSTEM CAS DROP POOL MEMBER`](/sql-reference/statements/system#system-cas-drop-pool-member) — permanently decommission a dead pool member's `server_root_id`. diff --git a/docs/reference/statements/system.mdx b/docs/reference/statements/system.mdx index 9e083a92e529..3a40be72ed06 100644 --- a/docs/reference/statements/system.mdx +++ b/docs/reference/statements/system.mdx @@ -575,6 +575,151 @@ Wait until all asynchronously loading data parts of a table (outdated data parts SYSTEM WAIT LOADING PARTS [ON CLUSTER cluster_name] [db.]merge_tree_family_table_name ``` +### SYSTEM CAS GC RUN {#system-cas-gc-run} + +Runs one garbage-collection round of the content-addressed (CAS) MergeTree garbage collector synchronously and node-local: it reclaims content-addressed objects that are no longer referenced by any part. This is the on-demand counterpart of the background GC scheduler; it is useful for tests and diagnostics. + +```sql +SYSTEM CAS GC RUN [ON CLUSTER cluster_name] [disk_name] +``` + +When `disk_name` is given, the round runs on that content-addressed disk only; targeting a non-content-addressed disk raises an exception. When `disk_name` is omitted, one round runs on every content-addressed disk configured on the node; if none are configured, the command raises an exception. + +Each round is recorded in [`system.cas_gc_log`](/operations/system-tables/cas_gc_log) as a `Start` and a `Finish` row (with `trigger = 'Manual'`). + +The command returns one row per disk it ran on (multiple rows when `disk_name` is omitted), with columns `disk`, `acquired_lease`, `deferred`, `round`, `candidates_marked`, `objects_deleted`, `objects_absent`, `objects_replaced`, `objects_spared`, `manifests_deleted`, `entries_condemned`, `entries_graduated`, `entries_redeleted`, `fence_outs`, `anomalies`, `pending_candidates`, `pending_condemned`, and `pending_retired`, describing the outcome of that round. The `pending_*` columns are the retire pipeline's remaining backlog sizes read from the `gc/state` this round's own commit just published (not this round's own delta, unlike the columns before them) — `0` on a non-authoritative row (`acquired_lease = 0` or `deferred = 1`), same as every other counter. + +A manual run always executes, regardless of [`SYSTEM CAS GC STOP`](#system-cas-gc-stop-start): `STOP` pauses only the background scheduler on that disk. + +### SYSTEM CAS GC REBUILD {#system-cas-gc-rebuild} + +Disaster-recovery command for the content-addressed (CAS) MergeTree garbage collector. It rebuilds a +CAS disk's `gc/state` baseline from scratch, by re-discovering the whole ref universe and re-folding +manifest edges into a fresh generation. It writes only the GC plane (`gc/state` and the `gc/gen/*` +artifacts) and never touches ref shards, manifests, or blobs, and it never deletes anything itself — +but the rebuilt baseline drives every subsequent GC round's retire decisions, so this is a +**destructive disaster-recovery tool**, not something to run routinely: a rebuild performed against +a state that was not actually corrupted discards live bookkeeping, and an incorrect rebuild can make +a later round delete objects that are still referenced. + +```sql +SYSTEM CAS GC REBUILD [FORCE] [ON CLUSTER cluster_name] disk_name +``` + +Unlike `SYSTEM CAS GC RUN`, `disk_name` is **required**: the destructive +baseline rebuild must never fan out across every content-addressed disk on the node from a bare +command; targeting a non-content-addressed disk raises an exception. + +By default the command refuses to run when the disk's existing `gc/state` and every artifact it +references decode successfully and are present — a rebuild would needlessly discard healthy live +bookkeeping. Add `FORCE` to rebuild deliberately even though the existing state looks healthy. The +command also refuses (regardless of `FORCE`) when another GC leader currently holds the disk's +lease. In both refusal cases it raises an exception instead of returning a row. + +On success it returns one row with columns `disk`, `performed`, `round`, `generation`, `namespaces`, +`shards`, `committed_refs`, `live_precommits`, `unowned_alive_manifests`, `edges`, +`clamped_shards`, `virgin_by_enumeration`, and `adopted_seal_generation`, describing the freshly +rebuilt baseline. `virgin_by_enumeration = 1` means the rebuild found no fold seal at all and +carried no durable hold forward, concluding from enumeration alone that the pool never sealed a +baseline — on a pool that has ever completed a GC round this means the object listing lied. +`adopted_seal_generation` names which generation's fold seal the rebuild carried holds from; `0` +when it carried none. + +### SYSTEM CAS GC STOP / SYSTEM CAS GC START {#system-cas-gc-stop-start} + +Pause or resume the background GC scheduler on one content-addressed disk, without affecting reads +or writes on that disk. This is granular operator control of GC alone — for example to pause +reclamation during an incident — not a lifecycle transition; the disk stays fully usable throughout. + +```sql +SYSTEM CAS GC STOP [ON CLUSTER cluster_name] disk_name +SYSTEM CAS GC START [ON CLUSTER cluster_name] disk_name +``` + +`disk_name` is **required** for both — unlike `SYSTEM CAS GC RUN`, there is no fan-out form, since +each command targets exactly one disk's scheduler. + +`GC STOP` stops in place: the scheduler object is retained, so a later `GC START` resumes the *same* +instance, preserving its `gc_id` and lease-observation history. It is idempotent, and works even on +a disk that is not currently live (stopping GC on a sick disk is a legitimate operation). It does +not stop a manual [`SYSTEM CAS GC RUN`](#system-cas-gc-run) on the same disk. + +`GC START` re-enters that same scheduler instance rather than creating a new one; leadership is +**not** automatically restored — the scheduler re-acquires the durable `gc/state` lease through the +next round's normal acquisition, the same as any other contender. It is idempotent (a no-op on an +already-running scheduler), and refuses with a typed error on a decommissioned or uncertain pool, +since restarting GC there would only spin failing rounds. + +Neither command returns a result set. + +### SYSTEM CAS FSCK {#system-cas-fsck} + +Independently verifies content-addressed pool reachability against a **running, mounted** disk — the +scan re-validates every finding against a fresh authoritative read, so unlike the offline +`clickhouse-disks cas-fsck` tool it needs no quiesce and no read-only mount. + +```sql +SYSTEM CAS FSCK [ON CLUSTER cluster_name] disk_name +``` + +`disk_name` is **required**. The command returns one row with columns `disk`, `reachable`, +`dangling`, `unreachable`, `pending_gc`, `awaiting_gc`, `unaccounted`, `stale_edge`, +`corrupted_runs`, `chain_broken`, `unchecked`, `lifeless_keys`, `namespace_janitor_pending`, +`namespace_janitor_pending_bytes`, `namespace_janitor_pending_lives`, `ref_records_walked`, +`physical_bytes`, `referenced_logical_bytes`, `distinct_blobs`, and `total_blob_refs`. `dangling` is +the one column that means data loss; `unreachable`, `pending_gc`, and `awaiting_gc` are objects +still moving through the normal condemn/graduate/delete pipeline, not a problem on their own. This +is a summary-only scan; per-object detail requires the offline `clickhouse-disks cas-fsck --detail`. + +### SYSTEM CAS FORGET {#system-cas-forget} + +Node-local operator assertion that a content-addressed disk is permanently gone — the "fire marshal" +verb for a stuck disk (a transient or identity-lost pool, or an operator-asserted decommission). +Unlike the other `SYSTEM CAS` commands, it deliberately works on a disk that is **not** live, since +that is its whole purpose. + +```sql +SYSTEM CAS FORGET [ON CLUSTER cluster_name] disk_name +``` + +`disk_name` is **required**. It is an assertion, not a proof of erasure: the disk stays registered +and answers further store-class access with a typed error, and a server restart re-registers the +name. Returns no result set. This is different from +[`SYSTEM CAS DROP POOL MEMBER`](#system-cas-drop-pool-member), which permanently retires one pool +*member's* identity across the whole shared pool — `FORGET` only affects this node's own local view +of one disk. + +### SYSTEM CAS DROP POOL MEMBER {#system-cas-drop-pool-member} + +Permanently decommissions a dead member (`server_root_id`) of a content-addressed disk +pool. It claims the member's mount slot as an administrative writer — fencing that `server_root_id` from ever +writing again — then drops every table namespace the member owned, drains eligible manifest debris, +staging objects, and mountpoint objects belonging to it, and retires the mount slot once all drains +are confirmed. This is a **destructive, irreversible** operation: only run it once the `server_root_id` is +confirmed permanently dead, since it fences the member out even if it later comes back online, and +it deletes namespace and drain state that cannot be recovered. + +It is a writer operation, not GC: it emits ordinary ref-edge deltas rather than inventing GC +transitions, and it does not synchronously reclaim shared blob content — removing the ref edges only +makes the now-unreferenced blobs eligible for an ordinary GC round to reclaim later. + +```sql +SYSTEM CAS DROP POOL MEMBER 'server_root_id' FROM DISK 'disk_name' [ON CLUSTER cluster_name] +``` + +Both `server_root_id` and `disk_name` are required string literals (a `server_root_id` is an opaque server-root path +that may contain `/`, not a plain identifier, so it cannot be written unquoted). + +The operation is resumable: a rerun skips namespaces already marked `Removed` and reports them +separately from namespaces newly removed by this invocation. Per-object drain failures are recorded +as warnings and leave the slot in a terminated-but-not-fully-drained state that a later invocation +can resume from, rather than raising an exception. + +The command returns one row with columns `server_root_id`, `namespaces_removed`, `namespaces_already_removed`, +`committed_refs_removed`, `precommits_removed`, `manifest_debris_removed`, `staging_objects_removed`, +`mountpoint_objects_removed`, `slot_removed`, and `warnings`. A non-empty `warnings` means some +drain was not confirmed and the mount slot was left in place as a resume anchor. + ## Managing ReplicatedMergeTree Tables {#managing-replicatedmergetree-tables} ClickHouse can manage background replication related processes in [ReplicatedMergeTree](/reference/engines/table-engines/mergetree-family/replication) tables. diff --git a/programs/disks/CMakeLists.txt b/programs/disks/CMakeLists.txt index 079bf2d9c2dd..c5f570e27fc6 100644 --- a/programs/disks/CMakeLists.txt +++ b/programs/disks/CMakeLists.txt @@ -18,6 +18,11 @@ set (CLICKHOUSE_DISKS_SOURCES CommandSed.cpp CommandHelp.cpp CommandTouch.cpp + CommandFsck.cpp + CommandCaGcDryRun.cpp + CommandCaGcRebuild.cpp + CommandCaInspect.cpp + CommandCaDropMember.cpp CommandGetCurrentDiskAndPath.cpp CommandPackedIO.cpp CommandDiskUsage.cpp diff --git a/programs/disks/CommandCaDropMember.cpp b/programs/disks/CommandCaDropMember.cpp new file mode 100644 index 000000000000..f95693c2a02f --- /dev/null +++ b/programs/disks/CommandCaDropMember.cpp @@ -0,0 +1,75 @@ +#include +#include +#include +#include +#include + +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +/// Operator-driven decommission of a DEAD pool member's namespaces, debris, staging, roots objects +/// and mount slot (`Cas::decommissionPoolMember`, design 2026-07-13-cas-pool-member-decommission +/// §core). Refuses a live member internally; this command only opens the CA disk read-only and +/// forwards the pool handle -- the admin claim itself happens inside `decommissionPoolMember`. +class CommandCaDropMember final : public ICommand +{ +public: + CommandCaDropMember() : ICommand("CommandCaDropMember") + { + command_name = "cas-drop-member"; + description = "Decommission a DEAD pool member: erase its namespaces, debris, staging, roots " + "objects and mount slot. Refuses a live member. Open the CA disk read-only " + "(the admin claim is made internally)."; + options_description.add_options()("member", po::value(), "server_root_id of the dead member"); + positional_options_description.add("member", 1); + } + + void executeImpl(const CommandLineOptions & options, DisksClient & client) override + { + const String srid = getValueFromCommandLineOptionsThrow(options, "member"); + auto disk = client.getCurrentDiskWithPath().getDisk(); + + auto * dos = dynamic_cast(disk.get()); + if (!dos) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-drop-member: '{}' is not an object-storage disk", disk->getName()); + + auto * ca = dynamic_cast(dos->getMetadataStorage().get()); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-drop-member: disk '{}' is not content-addressed", disk->getName()); + + if (!ca->isReadOnly()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "cas-drop-member: open the CA disk read-only (a writable open would claim this tool's " + "own server_root_id; the decommission claim happens internally)"); + + const auto host_store = ca->store(); + const auto report = Cas::decommissionPoolMember( + host_store->poolBackendPtr(), host_store->poolConfig(), srid); + + std::cout << "server_root_id=" << report.srid << "\n" + << "namespaces_removed=" << report.namespaces_removed << "\n" + << "namespaces_already_removed=" << report.namespaces_already_removed << "\n" + << "committed_refs_removed=" << report.committed_refs_removed << "\n" + << "precommits_removed=" << report.precommits_removed << "\n" + << "manifest_debris_removed=" << report.manifest_debris_removed << "\n" + << "staging_objects_removed=" << report.staging_objects_removed << "\n" + << "mountpoint_objects_removed=" << report.mountpoint_objects_removed << "\n" + << "slot_removed=" << (report.slot_removed ? "true" : "false") << "\n"; + for (const auto & w : report.warnings) + std::cout << "warning=" << w << "\n"; + } +}; + +CommandPtr makeCommandCaDropMember() +{ + return std::make_shared(); +} + +} diff --git a/programs/disks/CommandCaGcDryRun.cpp b/programs/disks/CommandCaGcDryRun.cpp new file mode 100644 index 000000000000..2e4ba6a7849e --- /dev/null +++ b/programs/disks/CommandCaGcDryRun.cpp @@ -0,0 +1,56 @@ +#include +#include +#include +#include +#include + +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +class CommandCaGcDryRun final : public ICommand +{ +public: + CommandCaGcDryRun() : ICommand("CommandCaGcDryRun") + { + command_name = "cas-gc-dryrun"; + description = "Preview the next GC round's deletes for a content-addressed pool (read-only, no deletes)."; + } + + void executeImpl(const CommandLineOptions &, DisksClient & client) override + { + auto disk = client.getCurrentDiskWithPath().getDisk(); + + auto * dos = dynamic_cast(disk.get()); + if (!dos) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-dryrun: '{}' is not an object-storage disk", disk->getName()); + + auto * ca = dynamic_cast(dos->getMetadataStorage().get()); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-dryrun: disk '{}' is not content-addressed", disk->getName()); + + if (!ca->isReadOnly()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-dryrun: open the CA disk read-only"); + + /// A non-leader, read-only Gc handle: previewDeletes never acquires the lease or writes. + Cas::Gc gc(ca->store(), UInt128(1)); + const auto preview = gc.previewDeletes(); + + std::cout << "preview_deletes=" << preview.size() << "\n"; + for (const auto & p : preview) + std::cout << p.reason << "\t" << p.key << "\t" << p.size << "\n"; + } +}; + +CommandPtr makeCommandCaGcDryRun() +{ + return std::make_shared(); +} + +} diff --git a/programs/disks/CommandCaGcRebuild.cpp b/programs/disks/CommandCaGcRebuild.cpp new file mode 100644 index 000000000000..5e488296ac4e --- /dev/null +++ b/programs/disks/CommandCaGcRebuild.cpp @@ -0,0 +1,85 @@ +#include +#include +#include +#include +#include +#include + +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +/// The gc/state disaster-recovery command (spec 2026-07-03): recomputes the in-degree baseline from +/// raw owner state and CASes a fresh gc/state when the guard has refused every regular round (a lost +/// gc/state over trimmed journal history — see docs/superpowers/cas/04-gc-protocol.md#gc-rebuild). +/// +/// REQUIRES a read-only-opened disk, same as fsck/cas-gc-dryrun: this tool must never claim the live +/// server's mount (a second live mounter racing the real GC's lease/writes is exactly the split-brain +/// class the protocol is designed to prevent). Unlike fsck/cas-gc-dryrun, rebuildBaseline DOES write +/// (a single gc/state CAS) — that write is a deliberate, explicit, operator-invoked exception to +/// "read-only means no writes", gated on the SAME `isReadOnly()` check so it can only run against a +/// disk configured with true (i.e. never against the disk a live server has +/// mounted for read-write traffic). +class CommandCaGcRebuild final : public ICommand +{ +public: + CommandCaGcRebuild() : ICommand("CommandCaGcRebuild") + { + command_name = "cas-gc-rebuild"; + description = "Disaster recovery: rebuild a content-addressed pool's gc/state baseline from raw owner " + "state after the GC guard has refused every round (see CORRUPTED_DATA in the gc log). " + "Requires a read-only-opened disk; never run against a disk a live server has mounted."; + options_description.add_options()("force", "bypass the \"healthy state\" refusal (rebuild even though gc/state and every referenced artifact look fine)"); + } + + void executeImpl(const CommandLineOptions & options, DisksClient & client) override + { + const bool force = options.contains("force"); + auto disk = client.getCurrentDiskWithPath().getDisk(); + + auto * dos = dynamic_cast(disk.get()); + if (!dos) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-rebuild: '{}' is not an object-storage disk", disk->getName()); + + auto * ca = dynamic_cast(dos->getMetadataStorage().get()); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-rebuild: disk '{}' is not content-addressed", disk->getName()); + + if (!ca->isReadOnly()) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "cas-gc-rebuild: open the CA disk read-only (true) — this tool must never " + "claim the live server's mount"); + + /// gc_id uniqueness across instances is the Gc caller obligation (a random u128 per invocation); + /// this is a one-shot command, so a fresh mint per run is exactly right (no stable-instance + /// requirement here — rebuildBaseline does its own lease acquire/steal check internally). + const UInt128 gc_id = (static_cast(thread_local_rng()) << 64) | thread_local_rng(); + Cas::Gc gc(ca->store(), gc_id); + const Cas::RebuildReport rep = gc.rebuildBaseline(force); + + std::cout << "performed=" << (rep.performed ? 1 : 0) << " round=" << rep.round << " generation=" << rep.generation + << " namespaces=" << rep.namespaces << " shards=" << rep.shards << " committed_refs=" << rep.committed_refs + << " live_precommits=" << rep.live_precommits << " unowned_alive_manifests=" << rep.unowned_alive_manifests + << " edges=" << rep.edges << " clamped_shards=" << rep.clamped_shards << "\n"; + + if (!rep.performed) + { + std::cout << "refusal=" << rep.refusal << "\n"; + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-gc-rebuild: refused: {}", rep.refusal); + } + } +}; + +CommandPtr makeCommandCaGcRebuild() +{ + return std::make_shared(); +} + +} diff --git a/programs/disks/CommandCaInspect.cpp b/programs/disks/CommandCaInspect.cpp new file mode 100644 index 000000000000..b4a3a24dbb89 --- /dev/null +++ b/programs/disks/CommandCaInspect.cpp @@ -0,0 +1,77 @@ +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +/// Read-only "decode any object" command: takes the RAW object-storage key (e.g. as printed by +/// `cas-gc-dryrun` or `fsck`) rather than a ClickHouse-relative path, GETs its bytes straight from +/// the pool's backend, and dispatches to `Cas::caInspectToJson` (the same free function the unit +/// tests exercise directly against encoder output). Never writes; safe to run against a live pool. +class CommandCaInspect final : public ICommand +{ +public: + CommandCaInspect() : ICommand("CommandCaInspect") + { + command_name = "cas-inspect"; + description = "Decode a content-addressed pool object (by its raw object-storage key) to JSON (read-only)."; + options_description.add_options()("key", po::value(), "the raw object-storage key to decode (mandatory, positional)"); + positional_options_description.add("key", 1); + } + + void executeImpl(const CommandLineOptions & options, DisksClient & client) override + { + const String key = getValueFromCommandLineOptionsThrow(options, "key"); + + auto disk = client.getCurrentDiskWithPath().getDisk(); + + auto * dos = dynamic_cast(disk.get()); + if (!dos) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: '{}' is not an object-storage disk", disk->getName()); + + auto * ca = dynamic_cast(dos->getMetadataStorage().get()); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: disk '{}' is not content-addressed", disk->getName()); + + if (!ca->isReadOnly()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: open the CA disk read-only"); + + const auto got = ca->store()->backend().get(key); + if (!got) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: key '{}' does not exist", key); + + const Cas::Layout & layout = ca->store()->layout(); + std::optional resolved_life; + std::optional life_id; + if (const auto parsed = layout.parseRefObjectKey(key)) + life_id = parsed->life_id; + else if (const auto parsed_ckpt = layout.parseRefCkptKey(key)) + life_id = *parsed_ckpt; + if (life_id) + { + const Cas::CasRefCatalog::Snapshot cut = Cas::CasRefCatalog::read(ca->store()->backend(), layout); + resolved_life = cut.life_index.resolve(*life_id); + } + + std::cout << Cas::caInspectToJson(layout, key, got->bytes, resolved_life) << "\n"; + } +}; + +CommandPtr makeCommandCaInspect() +{ + return std::make_shared(); +} + +} diff --git a/programs/disks/CommandFsck.cpp b/programs/disks/CommandFsck.cpp new file mode 100644 index 000000000000..5727f7d69c11 --- /dev/null +++ b/programs/disks/CommandFsck.cpp @@ -0,0 +1,188 @@ +#include +#include +#include +#include +#include + +#include +#include +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +class CommandFsck final : public ICommand +{ +public: + CommandFsck() : ICommand("CommandFsck") + { + command_name = "cas-fsck"; + description = "Independently verify content-addressed pool reachability (read-only). " + "Exits nonzero if any reachable object is missing (dangling)."; + options_description.add_options()("detail", "list per-object rows (class, key, size, reachable_from)")( + "timeout", po::value(), "abort the scan after N seconds with a clear error instead of hanging (default 600; 0 = unbounded)")( + "namespace", po::value(), "scope the scan to namespaces with this prefix (skips the pool-wide " + "physical/pipeline classification; still reports the scoped namespaces' " + "dangling refs and orphan-manifest debris as unreachable)")( + "partial", "on --timeout, print the counts accumulated so far flagged partial=1 instead of aborting empty-handed"); + } + + void executeImpl(const CommandLineOptions & options, DisksClient & client) override + { + const bool detail = options.contains("detail"); + const UInt64 timeout_sec = getValueFromCommandLineOptionsWithDefault(options, "timeout", 600); + const String namespace_prefix = options.contains("namespace") ? options["namespace"].as() : ""; + const bool partial = options.contains("partial"); + auto disk = client.getCurrentDiskWithPath().getDisk(); + + auto * dos = dynamic_cast(disk.get()); + if (!dos) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-fsck: '{}' is not an object-storage disk", disk->getName()); + + auto * ca = dynamic_cast(dos->getMetadataStorage().get()); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-fsck: disk '{}' is not content-addressed", disk->getName()); + + if (!ca->isReadOnly()) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "cas-fsck: open the CA disk read-only (true) so inspection never probes/schedules a live pool"); + + /// Progress to stderr so a long scan is visibly working (the reachable=… summary stays on + /// stdout, machine-parseable). The deadline bounds a slow-but-progressing scan with a clear + /// error; for a single LIST page stuck in S3-client retries, lower the disk's S3 retry budget. + Cas::FsckProgress on_progress = [](std::string_view phase, uint64_t objects, uint64_t pages) + { + std::cerr << "cas-fsck: " << phase << " — " << objects << " objects, " << pages << " pages\n"; + }; + std::optional deadline; + if (timeout_sec > 0) + deadline = std::chrono::steady_clock::now() + std::chrono::seconds(timeout_sec); + + const Cas::FsckReport report = Cas::runFsck(*ca->store(), detail, on_progress, deadline, partial, namespace_prefix); + + /// Built by `Cas::formatFsckSummary` rather than here, so the line is reachable from a unit test. + /// It was assembled inline until 2026-07-26, and in that time `corrupted_runs` was added to the + /// report and to `clean()` without ever being rendered — a hard finding no run could report. + std::cout << Cas::formatFsckSummary(report) << "\n"; + + /// De-alarm the pipeline classes for humans: on an active pool a nonzero pending/awaiting + /// count is the ack-floor deletion pipeline working as designed, not a leak. `stale_edge` is + /// deliberately NOT part of this sentence: those blobs look exactly like an `AwaitingGc` + /// backlog but will never drain, and being swept into "expected, no action needed" is what + /// hid them. + if (report.pending_gc + report.awaiting_gc > 0) + std::cout << "note: " << report.pending_gc + report.awaiting_gc + << " unreferenced object(s) are inside the normal GC deletion pipeline " + "(condemn -> graduate -> exact-token delete takes ~2-3 rounds) — expected, no action needed\n"; + if (report.stale_edge > 0) + std::cout << "note: " << report.stale_edge + << " unreferenced object(s) carry ONLY source edges naming manifests that no longer " + "exist: their in-degree can never reach zero, so the incremental GC will never " + "reclaim them — NOT expected, investigate (a rebuild of the in-degree state is the " + "only way to clear them)\n"; + /// `unchecked` is not a finding and does not exit nonzero — it says the audit could not PROVE + /// those namespaces either way, which is a statement about coverage. Saying so out loud is the + /// whole point: a silent verdict of "no complaints" would read as a clean bill of health. + if (report.unchecked > 0) + std::cout << "note: " << report.unchecked + << " namespace(s) could NOT be proved either way (an unprovable epoch crossing, an " + "unreadable record, or a namespace the scan could not examine) — this run says " + "nothing about them; the per-namespace reason is listed as an `unchecked` row " + "under --detail\n"; + if (report.unaccounted > 0) + std::cout << "note: " << report.unaccounted + << " object(s) are outside the current GC view — normal only as a transient " + "(created+dropped between GC rounds); re-run cas-fsck after the next round and " + "investigate any that persist\n"; + /// Not a finding: a canonical namespace-life key whose life is absent from the catalog is the + /// protocol-produced interval between a fenced GC exact-deleting a `Removing` row and the + /// perpetual namespace janitor reaching it on a later bounded page (its own deletes are + /// suppressed for the whole of Stage A). Persistent non-convergence is a leak/liveness question + /// for `CASGCNamespaceCleanupLeaks` and the `namespace_cleanup` GC-log phase, not this scan. + if (report.namespace_janitor_pending > 0) + std::cout << "note: " << report.namespace_janitor_pending + << " namespace-life object(s) (" << report.namespace_janitor_pending_bytes + << " byte(s) across " << report.namespace_janitor_pending_lives + << " life/lives) are janitor-pending — their catalog row is already gone, and the " + "perpetual namespace janitor is the sole intended reclaimer, but its deletes can " + "be deferred (e.g. a destructive-round suppression policy) — not corruption; " + "investigate only if the same objects persist across many completed janitor " + "cycles (listed as `janitor-pending` rows under --detail)\n"; + + if (detail) + { + for (const auto & o : report.objects) + { + const char * c = "unreachable"; // NOLINT(clang-analyzer-deadcode.DeadStores) - defensive fallback if the enum grows + switch (o.cls) + { + case Cas::FsckClass::Reachable: c = "reachable"; break; + case Cas::FsckClass::Dangling: c = "dangling"; break; + case Cas::FsckClass::Unreachable: c = "unreachable"; break; + case Cas::FsckClass::PendingGc: c = "pending-gc"; break; + case Cas::FsckClass::AwaitingGc: c = "awaiting-gc"; break; + case Cas::FsckClass::Unaccounted: c = "unaccounted"; break; + case Cas::FsckClass::StaleEdge: c = "stale-edge"; break; + case Cas::FsckClass::CorruptedRun: c = "corrupted-run"; break; + case Cas::FsckClass::ChainBroken: c = "chain-broken"; break; + case Cas::FsckClass::Unchecked: c = "unchecked"; break; + case Cas::FsckClass::LifelessKey: c = "lifeless-key"; break; + case Cas::FsckClass::JanitorPending: c = "janitor-pending"; break; + } + std::cout << c << "\t" << o.key << "\t" << o.size; + for (const auto & r : o.reachable_from) + std::cout << "\t" << r; + std::cout << "\n"; + } + } + + if (report.dangling > 0) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, "cas-fsck: {} reachable object(s) MISSING (INV-NO-LOSS violation)", report.dangling); + /// A hole in a ref stream is loss of a different kind: the records above it are unreachable, so + /// the table's own history is truncated wherever recovery next reads it. Fatal in the summary AND + /// in the exit code (spec §7) — a verdict only a `--detail` reader would notice is a verdict no + /// automation acts on. + if (report.chain_broken > 0) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "cas-fsck: {} namespace(s) have a HOLE in their ref-log stream — an id is absent below a " + "durable id of the same epoch, which contiguity (INV-1) makes impossible without a lost " + "record; every transaction above the hole is unreachable (positions are listed as " + "`chain-broken` rows under --detail)", report.chain_broken); + /// A term of `clean()`, and until 2026-07-26 the only one that neither printed nor exited + /// nonzero — so a corrupt run was invisible twice over. A seal-checksum mismatch is not debris: + /// `fold`/`zeroInDegree`/`previewDeletes` all fail closed on the same run, so GC cannot make + /// progress past it, and the audit deliberately continues only so ONE pass enumerates them all. + if (report.corrupted_runs > 0) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "cas-fsck: {} GC source-edge run(s) failed their whole-file seal checksum — the deletion-" + "deriving consumers fail closed on these, so GC cannot advance past them (run keys are " + "listed as `corrupted-run` rows under --detail)", report.corrupted_runs); + /// A key the `Layout` parsers refuse (no current writer can produce it), or a catalog + /// incarnation that is ambiguous or unreadable, is corruption nothing clears on its own: the + /// namespace enumeration now reports it instead of aborting, which is what makes an exit code + /// the only signal automation can act on. A COMPLETE, canonical namespace-life key whose life is + /// simply absent from the catalog is NOT counted here — see the `janitor-pending` note above. + if (report.lifeless_keys > 0) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "cas-fsck: {} key(s) under this pool name are malformed or unresolvable — no current " + "writer could have produced them, or their catalog incarnation is ambiguous/unreadable " + "(the keys are listed as `lifeless-key` rows under --detail)", report.lifeless_keys); + } +}; + +CommandPtr makeCommandFsck() +{ + return std::make_shared(); +} + +} diff --git a/programs/disks/DisksApp.cpp b/programs/disks/DisksApp.cpp index 15aeed60015d..66e1e4dbef6b 100644 --- a/programs/disks/DisksApp.cpp +++ b/programs/disks/DisksApp.cpp @@ -25,6 +25,7 @@ #include #include +#include #include #include #include "config.h" @@ -34,6 +35,7 @@ #include #include #include +#include #include @@ -44,6 +46,7 @@ namespace ErrorCodes { extern const int BAD_ARGUMENTS; extern const int LOGICAL_ERROR; + extern const int STD_EXCEPTION; }; LineReader::Patterns DisksApp::query_extenders = {"\\"}; @@ -212,6 +215,8 @@ bool DisksApp::processQueryText(const String & text) return false; CommandPtr command; + last_command_exit_code = 0; + auto subqueries = splitOnUnquotedSemicolons(text); for (const auto & subquery : subqueries) { @@ -230,6 +235,7 @@ bool DisksApp::processQueryText(const String & text) { int code = err.code(); error_string = getExceptionMessageForLogging(err, true, false); + last_command_exit_code = code; if (code == ErrorCodes::BAD_ARGUMENTS) { if (command.get()) @@ -246,10 +252,12 @@ bool DisksApp::processQueryText(const String & text) catch (std::exception & err) { error_string = err.what(); + last_command_exit_code = ErrorCodes::STD_EXCEPTION; } catch (...) // Ok: report unknown exception { error_string = "Unknown exception"; + last_command_exit_code = ErrorCodes::STD_EXCEPTION; } if (error_string.has_value()) { @@ -334,8 +342,16 @@ void DisksApp::registerCommands() command_descriptions.emplace("switch-disk", makeCommandSwitchDisk()); command_descriptions.emplace("current_disk_with_path", makeCommandGetCurrentDiskAndPath()); command_descriptions.emplace("touch", makeCommandTouch()); +<<<<<<< HEAD command_descriptions.emplace("du", makeCommandDiskUsage()); command_descriptions.emplace("wc", makeCommandWordCount()); +======= + command_descriptions.emplace("cas-fsck", makeCommandFsck()); + command_descriptions.emplace("cas-gc-dryrun", makeCommandCaGcDryRun()); + command_descriptions.emplace("cas-gc-rebuild", makeCommandCaGcRebuild()); + command_descriptions.emplace("cas-inspect", makeCommandCaInspect()); + command_descriptions.emplace("cas-drop-member", makeCommandCaDropMember()); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) command_descriptions.emplace("read-checksums", makeCommandReadChecksums()); command_descriptions.emplace("help", makeCommandHelp(*this)); command_descriptions.emplace("packed-io", makeCommandPackedIO()); @@ -539,6 +555,13 @@ int DisksApp::main(const std::vector & /*args*/) /*max_io_thread_pool_free_size*/ 0, /*io_thread_pool_queue_size*/ 10000); + /// `clickhouse-disks` loads no `ServerSettings`, so this can't read + /// `cas_blob_upload_pool_size`; 16 mirrors that setting's default + /// (`src/Core/ServerSettings.cpp`). A `write` command that commits through a + /// `cas` disk reaches `uploadPendingBlobs`, which calls this pool + /// unconditionally (see the analogous init in `Server.cpp`/`LocalServer.cpp`). + DB::Cas::initializeBlobUploadPool(16); + registerCommands(); registerDisks(/* global_skip_access_check= */ true); @@ -575,6 +598,16 @@ int DisksApp::main(const std::vector & /*args*/) global_context->setPath(path); + /// Load the server UUID so that live CA namespaces resolve correctly. + /// Only load when the uuid file already exists — clickhouse-disks inspects existing + /// pools and must NOT create or mutate the uuid file (the disk may be read-only). + /// If the file is absent, ServerUUID stays Nil and shadow/non-live navigation works. + { + fs::path uuid_file = fs::path(path) / "uuid"; + if (fs::exists(uuid_file)) + ServerUUID::load(uuid_file, &logger()); + } + client = std::make_unique(config(), global_context); suggest.setCompletionsCallback([&](const String & prefix, size_t /* prefix_length */) { return getCompletions(prefix); }); @@ -591,6 +624,10 @@ int DisksApp::main(const std::vector & /*args*/) if (log_file) log_file->close(); + /// Non-interactive runs surface a failing command as a nonzero process exit (CI/cron gating, + /// e.g. `cas-fsck` reporting dangling objects). Interactive sessions are unaffected. + if (query.has_value() && last_command_exit_code != 0) + return last_command_exit_code; return Application::EXIT_OK; } @@ -642,6 +679,7 @@ int mainEntryClickHouseDisks(int argc, char ** argv) /// That way, accesses happen-before destruction. SCOPE_EXIT_SAFE({ DB::StaticThreadPool::shutdownAll(); + DB::Cas::shutdownBlobUploadPool(); GlobalThreadPool::shutdown(); }); diff --git a/programs/disks/DisksApp.h b/programs/disks/DisksApp.h index fbe0639e00f3..27d7e03a6720 100644 --- a/programs/disks/DisksApp.h +++ b/programs/disks/DisksApp.h @@ -90,6 +90,10 @@ class DisksApp : public Poco::Util::Application std::optional query; + /// Set when a command threw during processQueryText; used to make non-interactive (--query) + /// runs exit nonzero (e.g. `fsck` reporting dangling objects). Reset per processQueryText call. + int last_command_exit_code = 0; + const std::unordered_map aliases = { {"cp", "copy"}, {"mv", "move"}, diff --git a/programs/disks/ICommand.h b/programs/disks/ICommand.h index db543a9be407..0bfb8a3d2ae9 100644 --- a/programs/disks/ICommand.h +++ b/programs/disks/ICommand.h @@ -133,8 +133,16 @@ DB::CommandPtr makeCommandSwitchDisk(); DB::CommandPtr makeCommandGetCurrentDiskAndPath(); DB::CommandPtr makeCommandHelp(const DisksApp & disks_app); DB::CommandPtr makeCommandTouch(); +<<<<<<< HEAD DB::CommandPtr makeCommandDiskUsage(); DB::CommandPtr makeCommandWordCount(); +======= +DB::CommandPtr makeCommandFsck(); +DB::CommandPtr makeCommandCaGcDryRun(); +DB::CommandPtr makeCommandCaGcRebuild(); +DB::CommandPtr makeCommandCaInspect(); +DB::CommandPtr makeCommandCaDropMember(); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) DB::CommandPtr makeCommandReadChecksums(); DB::CommandPtr makeCommandPackedIO(); } diff --git a/programs/local/LocalServer.cpp b/programs/local/LocalServer.cpp index 822087044f05..4a6e7b836c0d 100644 --- a/programs/local/LocalServer.cpp +++ b/programs/local/LocalServer.cpp @@ -68,6 +68,7 @@ #include #include #include +#include #include #include #include @@ -216,6 +217,7 @@ namespace ServerSetting extern const ServerSettingsUInt64 max_format_parsing_thread_pool_size; extern const ServerSettingsUInt64 max_format_parsing_thread_pool_free_size; extern const ServerSettingsUInt64 format_parsing_thread_pool_queue_size; + extern const ServerSettingsUInt64 cas_blob_upload_pool_size; extern const ServerSettingsUInt64 page_cache_history_window_ms; extern const ServerSettingsString page_cache_policy; extern const ServerSettingsDouble page_cache_size_ratio; @@ -466,6 +468,11 @@ void LocalServer::initialize(Poco::Util::Application & self) server_settings[ServerSetting::max_format_parsing_thread_pool_size], server_settings[ServerSetting::max_format_parsing_thread_pool_free_size], server_settings[ServerSetting::format_parsing_thread_pool_queue_size]); + + /// See the explanation near the same line in Server.cpp: `uploadPendingBlobs` reaches this + /// pool unconditionally once a `cas` disk commits a part, so every entry point + /// that can run a CA INSERT must initialize it, not only `clickhouse-server`. + DB::Cas::initializeBlobUploadPool(server_settings[ServerSetting::cas_blob_upload_pool_size]); } @@ -964,6 +971,11 @@ void LocalServer::cleanup() client_context.reset(); + /// Joins any outstanding blob-upload fan-out tasks before the context they reference + /// is torn down. Idempotent and noexcept, so safe even if never initialized (e.g. no + /// `cas` disk was ever used). + DB::Cas::shutdownBlobUploadPool(); + if (global_context) { global_context->shutdown(); diff --git a/programs/server/Server.cpp b/programs/server/Server.cpp index c8fcfd0a50e8..792313e8e354 100644 --- a/programs/server/Server.cpp +++ b/programs/server/Server.cpp @@ -113,6 +113,11 @@ #include #include #include +<<<<<<< HEAD +======= +#include +#include +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include #include @@ -430,6 +435,7 @@ namespace ServerSetting extern const ServerSettingsUInt64 max_format_parsing_thread_pool_size; extern const ServerSettingsUInt64 max_format_parsing_thread_pool_free_size; extern const ServerSettingsUInt64 format_parsing_thread_pool_queue_size; + extern const ServerSettingsUInt64 cas_blob_upload_pool_size; extern const ServerSettingsUInt64 page_cache_history_window_ms; extern const ServerSettingsString page_cache_policy; extern const ServerSettingsDouble page_cache_size_ratio; @@ -1710,6 +1716,7 @@ try Stopwatch watch; LOG_INFO(log, "Waiting for background threads"); DB::StaticThreadPool::shutdownAll(); + DB::Cas::shutdownBlobUploadPool(); GlobalThreadPool::instance().shutdown(); LOG_INFO(log, "Background threads finished in {} ms", watch.elapsedMilliseconds()); }); @@ -2000,6 +2007,8 @@ try server_settings[ServerSetting::max_format_parsing_thread_pool_free_size], server_settings[ServerSetting::format_parsing_thread_pool_queue_size]); + DB::Cas::initializeBlobUploadPool(server_settings[ServerSetting::cas_blob_upload_pool_size]); + std::string path_str = getCanonicalPath(String(server_settings[ServerSetting::path]), original_working_directory); fs::path path = path_str; diff --git a/programs/server/config.xml b/programs/server/config.xml index e99da390eb15..5fb082be6d09 100644 --- a/programs/server/config.xml +++ b/programs/server/config.xml @@ -1206,6 +1206,23 @@ --> + + + system + cas_log
+ toYYYYMM(event_date) + 7500 + 1048576 + 8192 + 524288 + false + +
+ + + system + cas_gc_log
+ toYYYYMM(event_date) + 7500 + 1048576 + 8192 + 524288 + false +
+ + cas + + replica-1 + cas_pool/ + + cas_scratch/ + 1 + 60 +
+ +``` + +`1` opens the disk in observe-only mode: no mount-slot +claim, no capability probe, no writes — the mode `clickhouse-disks` tools and +post-mortem inspection use. The full knob set (staging backend, cache sizes, +GC sharding, hash algorithm, request budgets) is parsed in +`ContentAddressedSettings.cpp`; each knob is documented at its declaration site. Blob publication +has no presence-cache setting: `HEAD` is mandatory. `gcs_max_conditional_put_bytes` applies to all +conditional non-blob writes, including create-if-absent artifacts and conditional replacements, but +not to multipart-capable blob publication. + +## Operations and observability + +- `clickhouse-disks` verbs (all require the disk opened read-only): `fsck` + (independent reachability audit of refs → manifests → blobs), `cas-inspect` + (decode one pool object by its raw key to JSON), `cas-gc-dryrun` (preview the + next GC round's deletes), `cas-gc-rebuild` (disaster-recovery rebuild of the + `gc/state` baseline), `cas-drop-member` (decommission a dead pool member). +- `system.cas_log` — one row per CAS protocol event + (uploads, adopts, promotes, condemns, deletes, mount-slot writes, ...); + the primary audit trail when investigating pool state. +- The GC and writer paths also emit `ProfileEvents` counters (grep + `ProfileEvents.cpp` for `Cas`). + +## Testing + +- **Unit tests** (`unit_tests_dbms`): every CAS suite name starts with `Cas`, so + `--gtest_filter='Cas*'` runs the whole set — including parameterized suites, + whose instantiation prefixes are `Cas`-prefixed too so the `/` + spelling still matches. `utils/cas-gate/generate_cas_suites.sh` fails loud on a + CAS suite that does not match, so a new suite cannot silently sit outside the + filter; `utils/cas-gate/run_cas_gate_per_suite.sh` runs them one process per + suite, so an abort cannot hide the suites after it. +- **Stateless lanes**: the functional-test jobs "`cas storage`" + (local object storage) and "`cas s3 storage`" run the whole + stateless suite with `MergeTree` defaulting to a CAS disk. Tests that + legitimately cannot run there carry the `no-cas-storage` tag. +- **Soak / chaos**: `utils/ca-soak/` — multi-replica docker-compose + harnesses (fault proxies, GC sharding variants, AWS S3/GCS backends) and + adversarial scenarios. + +## Reading order + +To understand a request end to end, read in this order: + +1. `ContentAddressedMetadataStorage` — the facade / routing. +2. `Parts/PartFolderAccess` (`PartRefKey` → the folder view / cache). +3. `Pool/CasPool` — the pool composition root and `open` protocol. +4. `Pool/CasPartWriteTxn` — one-part write transaction. +5. `Gc/CasGc` — the GC round engine. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp new file mode 100644 index 000000000000..416f5c327663 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp @@ -0,0 +1,476 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int CORRUPTED_DATA; +} +} + +namespace DB::Cas +{ + +namespace +{ + +uint64_t nowMs() +{ + return static_cast(std::chrono::duration_cast( + std::chrono::system_clock::now().time_since_epoch()).count()); +} + +/// Delete every object listed under `prefix` by its listed (or, absent a list-token backend, HEAD'd) +/// token. This backs the staging and roots drain phases below: the victim's writers are fenced by the +/// decommission claim (`Pool::openForDecommission`), so nothing should be racing these deletes, and a +/// plain exact-token delete of every listed object is race-free. +/// +/// A per-object failure — a backend exception, a `TokenMismatch` or `NotFound` outcome, or an object +/// disappearing between `LIST` and `HEAD` — is recorded as a warning and does not prevent the remaining +/// objects from being attempted. The caller keeps the pool slot whenever warnings are present, so the +/// terminated slot remains available as a resume anchor instead of being deleted after an unconfirmed +/// drain. Returns only the objects whose exact-token delete was reported as `Deleted`. +uint64_t deleteListedPrefix(Backend & backend, const String & prefix, std::vector & warnings) +{ + uint64_t deleted = 0; + forEachListedKey(backend, prefix, [&](const ListedKey & listed) + { + try + { + Token token; + if (listed.token) + token = *listed.token; + else + { + const HeadResult head = backend.head(listed.key); + if (!head.exists) + { + warnings.push_back("decommission drain: " + listed.key + " vanished before delete"); + return; + } + token = head.token; + } + + const DeleteOutcome outcome = backend.deleteExact(listed.key, token); + const DeleteClass outcome_class = classifyDeleteOutcome(outcome); + if (outcome_class == DeleteClass::Deleted) + ++deleted; + else + warnings.push_back("decommission drain: " + listed.key + " delete outcome " + + String(deleteClassName(outcome_class))); + } + catch (...) + { + warnings.push_back("decommission drain: " + listed.key + " delete failed: " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + }); + return deleted; +} + +/// Delete one slot control object by a token captured at the protocol-defined fence point. Slot +/// retirement is fail-closed: unlike the debris drains above, any non-`Deleted` outcome or exception +/// stops the tail before it can touch the next control object. +bool deleteSlotObject(Backend & backend, const String & key, const Token & token, std::vector & warnings) +{ + try + { + const DeleteOutcome outcome = backend.deleteExact(key, token); + const DeleteClass outcome_class = classifyDeleteOutcome(outcome); + if (outcome_class == DeleteClass::Deleted) + return true; + + warnings.push_back("slot delete failed: " + key + ": delete outcome " + + String(deleteClassName(outcome_class))); + } + catch (...) + { + warnings.push_back("slot delete failed: " + key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + return false; +} + +} + +DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, + const String & victim_srid, const CasEventSink & sink, + const std::function & request_gc_round) +{ + DecommissionReport report; + report.srid = victim_srid; + bool gc_round_needed = false; + /// A namespace may have reached `Removing` before a later namespace fails closed. Preserve the + /// already-earned liveness signal on every exit: the callback only wakes the existing serialized + /// GC worker and cannot perform catalog work itself. + SCOPE_EXIT({ + if (gc_round_needed && request_gc_round) + request_gc_round(); + }); + + /// Validate one required immutable ownership cut before impersonating the victim. The admin open + /// performs its own fresh catalog observation for mount safety, but namespace selection below + /// must reuse this exact pre-mutation decision rather than read a later authority set. + const Layout catalog_layout(config.pool_prefix); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, catalog_layout); + catalog_cut.life_index.throwIfAmbiguous("CAS decommission"); + + config.event_sink = sink; + PoolPtr admin = Pool::openForDecommission(std::move(backend), std::move(config), victim_srid); + + EventEmitter{*admin}.emit([&](CasEvent & e) + { + e.type = CasEventType::MemberDecommission; + e.outcome = "begin"; + e.reason = "operator decommission of pool member"; + e.detail = {{"server_root_id", victim_srid}}; + }); + + /// The pre-impersonation catalog cut is the complete ownership universe. Physical life keys carry + /// no logical path, and raw string prefixes such as `victim` must not select the distinct owner + /// `victim2`; the slash makes `victim` one canonical path component. + const String victim_namespace_prefix = victim_srid + "/"; + std::vector> owned_lives; + for (const CatalogEntry & entry : catalog_cut.catalog.entries) + { + if (entry.ns.string() != victim_srid && !entry.ns.string().starts_with(victim_namespace_prefix)) + continue; + const auto life = catalog_cut.life_index.resolve(entry.incarnation); + if (!life) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "ca-decommission: catalog entry '{}' has no physical life resolution", entry.ns.string()); + owned_lives.emplace_back(entry, *life); + } + + for (const auto & [selected_entry, life] : owned_lives) + { + const RootNamespace & ns = life.ns; + const String & ns_str = ns.string(); + + /// Refuse a same-name lifecycle move that landed after the immutable selection cut. The + /// exact-life overloads below also pin recovery to `life`, closing the race after this check: + /// a later replacement can never redirect a removal to its new incarnation. + const CasRefCatalog::Snapshot current_catalog = CasRefCatalog::read(admin->backend(), admin->layout()); + const auto current_entry = std::find_if( + current_catalog.catalog.entries.begin(), current_catalog.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns.string() == ns_str; }); + if (current_entry == current_catalog.catalog.entries.end() || *current_entry != selected_entry) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "ca-decommission: namespace '{}' changed incarnation after the validated catalog cut; " + "refusing destructive work", + ns_str); + + if (selected_entry.state == NsState::Removing) + { + if (!admin->backend().head(admin->layout().refCkptKey(life)).exists) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "ca-decommission: namespace '{}' is Removing but its exact checkpoint is absent; " + "the catalog row remains owned and the victim slot cannot be retired", + ns_str); + + /// `dropNamespace` is the sole terminal writer. On an already-complete removal this is an + /// idempotent observation; on a pre-terminal `Removing` life it resumes the exact terminal + /// append under the administrative writer fence. Catalog deletion remains GC's job. + (void)admin->dropNamespace(life); + ++report.namespaces_already_removed; + gc_round_needed = true; + continue; + } + + const auto stats = admin->dropNamespace(life); + ++report.namespaces_removed; + report.committed_refs_removed += stats.committed_refs; + report.precommits_removed += stats.precommits; + report.edge_deltas_emitted += stats.committed_refs + stats.precommits; + if (selected_entry.state != NsState::Creating) + gc_round_needed = true; + + EventEmitter{*admin}.emit([&](CasEvent & e) + { + e.type = CasEventType::MemberDecommission; + e.outcome = "namespace_removed"; + e.reason = "decommission dropped a victim namespace"; + e.detail = {{"server_root_id", victim_srid}, {"namespace", ns_str}, + {"committed", std::to_string(stats.committed_refs)}, + {"precommits", std::to_string(stats.precommits)}}; + }); + } + + /// Manifest debris must be removed before the mount slot: deleting the mount body removes the + /// watermark authority, after which `floorForNamespace` returns no value and the ordinary orphan + /// sweep cannot prove that old-epoch debris is eligible. The decommission claim has advanced the + /// writer epoch, so every build prefix with `prefix.writer_epoch < w.writer_epoch` is eligible here. + /// Group the listed keys by namespace and build prefix so each group can use the exact-token orphan + /// sweep while the mount body still supplies its authority. + { + const String debris_prefix = admin->layout().casManifestsServerPrefix(victim_srid); + std::set> groups; /// (namespace, writer epoch, build sequence) + forEachListedKey(admin->backend(), debris_prefix, [&](const ListedKey & listed) + { + if (const auto parsed = admin->layout().parseManifestKey(listed.key)) + groups.emplace(parsed->root_namespace.string(), parsed->ref.writer_epoch, parsed->ref.build_sequence); + }); + for (const auto & [ns_str, writer_epoch, build_sequence] : groups) + report.manifest_debris_removed += sweepNamespace( + *admin, RootNamespace(ns_str), BuildPrefix{writer_epoch, build_sequence}, &report.warnings); + } + + /// Drain the victim's own `/staging//` area. The live-mount staging helper uses + /// an `IObjectStorage`, while this command intentionally works at the `Backend` layer, so the same + /// prefix is listed and deleted directly. The claim fences the victim's writers during this sweep. + report.staging_objects_removed += deleteListedPrefix( + admin->backend(), admin->poolConfig().pool_prefix + "/staging/" + victim_srid + "/", report.warnings); + + /// Drain the victim's mountpoint objects. These are loose, non-content-addressed files under + /// `Layout::serverRootDataPrefix`; they have no writer epoch of their own, so the claim is what + /// prevents a returning victim from racing this deletion. + report.mountpoint_objects_removed += deleteListedPrefix( + admin->backend(), admin->layout().serverRootDataPrefix(victim_srid), report.warnings); + + /// The catalog, not physical debris, owns the slot-retirement decision. A terminal append only + /// moves a row to `Removing`; GC must fold/prune/delete it before the member's ownership anchor can + /// disappear. Capture one exact whole-catalog cut after every drain, then revalidate its token and + /// canonical value immediately before entering the retirement tail. The administrative claim fences + /// the victim writer between those observations. + std::optional retirement_catalog_cut; + if (report.warnings.empty()) + { + retirement_catalog_cut = CasRefCatalog::read(admin->backend(), admin->layout()); + const uint64_t victim_owned_count = std::count_if( + retirement_catalog_cut->catalog.entries.begin(), retirement_catalog_cut->catalog.entries.end(), + [&](const CatalogEntry & entry) + { + return entry.ns.string() == victim_srid + || entry.ns.string().starts_with(victim_namespace_prefix); + }); + if (victim_owned_count > 0) + report.warnings.push_back( + "pool member decommission underway: " + std::to_string(victim_owned_count) + + " namespace(s) are still owned by this member; upcoming GC rounds perform the final " + "cleanup — re-run this command afterwards to retire the slot"); + } + + /// Retire the slot strictly last and only after a clean drain. Copy the layout and shared backend + /// before `admin.reset()`: graceful close destroys the `Pool`, while the backend must remain alive to + /// retire the slot objects afterwards. + const Layout layout = admin->layout(); + const BackendPtr pool_backend = admin->poolBackendPtr(); + if (report.warnings.empty()) + { + const CasRefCatalog::Snapshot fresh_retirement_catalog + = CasRefCatalog::read(admin->backend(), admin->layout()); + if (!retirement_catalog_cut + || fresh_retirement_catalog.token != retirement_catalog_cut->token + || fresh_retirement_catalog.catalog != retirement_catalog_cut->catalog) + { + report.warnings.push_back( + "catalog changed after the victim ownership check; refusing slot retirement against a stale cut"); + } + } + if (report.warnings.empty()) + { + const String mount_key = layout.mountKey(victim_srid); + const String epoch_key = layout.epochKey(victim_srid); + const String owner_key = layout.ownerKey(victim_srid); + + /// Capture both the epoch value and its exact token while the decommission claim still fences + /// the victim. A successor can only bump this object after the farewell below releases the + /// claim, so this token is the epoch-side successor fence for the retirement tail. + std::optional claimed_epoch; + try + { + claimed_epoch = pool_backend->get(epoch_key); + if (!claimed_epoch) + report.warnings.push_back("slot capture failed: " + epoch_key + " is absent under the admin claim"); + } + catch (...) + { + report.warnings.push_back("slot capture failed: " + epoch_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + + /// Graceful close stamps an already-expired lease and the watermark farewell + /// (`min_active = UINT64_MAX`), making the slot `terminated` before its mutable control objects + /// are removed and its owner anchor is tombstoned. + admin.reset(); + + /// Read the farewell immediately after `finishTeardown` wrote it. Its exact token is the + /// mount-side fence: deleting by this token can remove only THIS decommission's farewell, not + /// a successor reclaim. Validate the body against the epoch value captured under the claim so + /// a successor that completed before this GET is also recognized and left untouched. + std::optional farewell_mount; + try + { + farewell_mount = pool_backend->get(mount_key); + if (!farewell_mount) + report.warnings.push_back("slot capture failed: " + mount_key + " farewell is absent"); + } + catch (...) + { + report.warnings.push_back("slot capture failed: " + mount_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + + bool captures_match = claimed_epoch && farewell_mount; + if (captures_match) + { + try + { + const ServerEpoch epoch_value = decodeServerEpoch(claimed_epoch->bytes); + const MountLease mount_value = decodeMountLease(farewell_mount->bytes); + captures_match = epoch_value.next_writer_epoch != 0 + && mount_value.writer_epoch == epoch_value.next_writer_epoch - 1 + && mount_value.min_active == std::numeric_limits::max() + && !mount_value.gc_fenced; + if (!captures_match) + { + report.warnings.push_back( + "slot capture failed: " + mount_key + + " is not this decommission's farewell for the epoch captured under the admin claim"); + } + } + catch (...) + { + report.warnings.push_back("slot capture failed while validating " + mount_key + " and " + epoch_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + captures_match = false; + } + } + + /// Mount first: if a successor reclaimed it after the farewell capture, the stale farewell + /// token yields `TokenMismatch` and the tail stops before touching epoch or owner. Epoch second: + /// its under-claim token similarly detects a successor allocation. Before touching owner, re-read + /// both mutable objects: a same-UUID successor can recreate them after both deletes without + /// rewriting the owner identity anchor. Mere presence proves that the slot is live again. Every + /// delete must be explicitly confirmed as `Deleted`, and the final owner tombstone rewrite must + /// succeed against the exact token read immediately before it. + /// + /// ACCEPTED RESIDUAL WINDOW (final review, not closed by this recheck): a same-UUID successor + /// can still recreate epoch/mount in the narrow gap strictly AFTER this liveness recheck but + /// BEFORE the owner CAS below reads its own token -- the successor's owner anchor (same + /// server_uuid, not yet retired) then gets tombstoned by this decommission run. The successor's + /// live process is not deleted (only its owner anchor is marked retired), but a LATER restart of + /// that same identity would refuse to reclaim it (claimOwnerOrThrow's tombstone guard). This is + /// a narrow, low-probability window, deliberately not closed here: T5's owner-tombstone design + /// (finding #9) intentionally stopped short of making concurrent decommission-vs-recreate + /// airtight to the microsecond, since that was explicitly not the priority for this fix. + report.slot_removed = false; + if (captures_match && deleteSlotObject(*pool_backend, mount_key, farewell_mount->token, report.warnings) + && deleteSlotObject(*pool_backend, epoch_key, claimed_epoch->token, report.warnings)) + { + std::optional current_mount; + std::optional current_epoch; + bool liveness_recheck_succeeded = true; + try + { + current_mount = pool_backend->get(mount_key); + } + catch (...) + { + report.warnings.push_back("slot liveness recheck failed: " + mount_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + liveness_recheck_succeeded = false; + } + try + { + current_epoch = pool_backend->get(epoch_key); + } + catch (...) + { + report.warnings.push_back("slot liveness recheck failed: " + epoch_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + liveness_recheck_succeeded = false; + } + + if (liveness_recheck_succeeded && (current_mount || current_epoch)) + { + report.warnings.push_back( + "slot delete aborted: successor reappeared after mutable control-object deletion; owner kept"); + } + else if (liveness_recheck_succeeded) + { + try + { + if (const auto owner = pool_backend->get(owner_key)) + { + OwnerObject tombstoned = decodeOwner(owner->bytes); + tombstoned.retired_at_ms = nowMs(); + /// Controlled, not a bare putOverwrite: a transient transport error here (or + /// one whose response was simply lost) must not be reported as a hard failure + /// when the write actually landed. A standalone controller (decommission is an + /// administrative, non-hot-path operation; no mount-lease fence applies to it + /// -- the exact-token CAS itself is the safety mechanism, same as the mount/ + /// epoch deletes above) resolves an ambiguous attempt with one GET: unchanged + /// token means the write never applied (legitimately retryable within budget); + /// matching bytes means this exact tombstone already landed (Committed, not a + /// failure); anything else is a genuine successor reclaim (Conflict). + CasRequestController controller(pool_backend, CasRequestBudget{}); + const CasOverwriteResult result = controller.putOverwriteControlled( + owner_key, encodeOwner(tombstoned), owner->token, [] { return true; }); + if (result.outcome == CasOverwriteOutcome::Committed) + report.slot_removed = true; + else if (result.outcome == CasOverwriteOutcome::Conflict) + report.warnings.push_back( + "slot tombstone failed: " + owner_key + + ": successor reclaimed the owner anchor before this decommission's tombstone write"); + else + report.warnings.push_back( + "slot tombstone failed: " + owner_key + + ": tombstone write outcome could not be resolved (retry budget exhausted " + "or the resolve GET itself failed) -- rerun the command to retry"); + } + else + report.warnings.push_back( + "slot tombstone failed: " + owner_key + ": object absent before tombstone write"); + } + catch (...) + { + report.warnings.push_back("slot tombstone failed: " + owner_key + ": " + + getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + } + } + } + else + { + report.slot_removed = false; + LOG_WARNING(getLogger("CasDecommission"), + "CAS decommission '{}': drain incomplete ({} warnings) — mount slot kept (terminated); " + "re-run the command to finish", victim_srid, report.warnings.size()); + admin.reset(); /// Graceful close still stamps the farewell, leaving the slot `terminated`. + } + + /// The `end` event is emitted via `sink` directly, not `EventEmitter{*admin}`: `admin` is gone by + /// now. This also means its `warnings` count reflects the FINAL total, including a slot-retirement + /// failure appended just above -- `EventEmitter`'s own zero-cost-when-absent guard is reproduced by + /// the `if (sink)` below. + if (sink) + { + CasEvent e; + e.type = CasEventType::MemberDecommission; + e.outcome = "end"; + e.reason = "decommission finished"; + e.detail = {{"server_root_id", victim_srid}, + {"namespaces_removed", std::to_string(report.namespaces_removed)}, + {"warnings", std::to_string(report.warnings.size())}, + {"slot_removed", report.slot_removed ? "1" : "0"}}; + sink(std::move(e)); + } + return report; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h new file mode 100644 index 000000000000..a86edf9286c4 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h @@ -0,0 +1,55 @@ +#pragma once + +#include +#include + +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// Counts the work performed by `decommissionPoolMember` for one pool member. The namespace counters +/// describe metadata and ref-log transitions; the object counters describe physical objects deleted by +/// the manifest, staging, and mountpoint drains. Blob bytes are intentionally not reported: removing +/// ref edges makes them eligible for ordinary GC, but this operation does not synchronously reclaim +/// shared content. +/// +/// A decommission is resumable. A previous run may already have moved namespaces to `Removing`, and a +/// warning means that the corresponding drain was not confirmed. In either case the report lets the +/// caller distinguish work done by this invocation from work observed from an earlier invocation. +struct DecommissionReport +{ + String srid; /// The decommissioned member's `server_root_id`. + uint64_t namespaces_removed = 0; /// Namespaces erased by this invocation. + uint64_t namespaces_already_removed = 0; /// Namespaces already `Removing` on entry. + uint64_t committed_refs_removed = 0; /// Committed ref records removed by namespace drops. + uint64_t precommits_removed = 0; /// Precommit records removed by namespace drops. + uint64_t edge_deltas_emitted = 0; /// The sum of `committed_refs_removed` and `precommits_removed`. + uint64_t manifest_debris_removed = 0; /// Eligible manifest objects deleted from old build prefixes. + uint64_t staging_objects_removed = 0; /// Objects deleted from the member's staging prefix. + uint64_t mountpoint_objects_removed = 0; /// Objects deleted from the member's roots/mountpoint prefix. + bool slot_removed = false; /// Whether mount and epoch were deleted and the owner was tombstoned. + std::vector warnings; /// Drain or slot-retirement failures; a non-empty list keeps the slot. +}; + +/// Erases all content owned by a permanently dead pool member. The operation first claims the member's +/// slot as an administrative writer; a live lease is refused, and the claim fences the dead member from +/// writing while cleanup runs. It then drops each table namespace through `Pool::dropNamespace`, drains +/// eligible manifest debris, staging objects, and mountpoint objects, and retires the slot only after all +/// drains are confirmed. Namespace drops are idempotent: a rerun resumes any missing terminal append +/// and leaves exact catalog-row deletion to GC. The member slot remains while any catalog entry still +/// belongs to the victim. +/// +/// This is a writer operation, not GC: it emits the normal ref-edge deltas and does not invent ref +/// transitions. Per-object drain failures are recorded in `DecommissionReport::warnings` and leave the +/// terminated slot as a resume anchor; other failures, including refusal to claim the member, propagate +/// as exceptions. When set, `sink` receives `MemberDecommission` audit events for the run's begin, +/// per-namespace, and end milestones. +DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, + const String & victim_srid, const CasEventSink & sink = {}, + const std::function & request_gc_round = {}); + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp new file mode 100644 index 000000000000..7848ae591f27 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp @@ -0,0 +1,1180 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int TIMEOUT_EXCEEDED; +} +} + +namespace DB::Cas +{ + +namespace +{ +constexpr uint64_t PROGRESS_PAGES = 16; + +using Deadline = std::optional; + +/// Enforce the optional overall scan deadline between backend operations. A timeout is propagated as +/// `TIMEOUT_EXCEEDED`; the public `runFsck` wrapper may convert that exception into a partial report when +/// explicitly requested. +void checkDeadline(const Deadline & deadline, std::string_view phase) +{ + if (deadline && std::chrono::steady_clock::now() > *deadline) + throw Exception(ErrorCodes::TIMEOUT_EXCEEDED, + "fsck: exceeded the deadline during '{}' — run against a QUIESCED pool or raise --timeout.", phase); +} + +void listAll(Backend & backend, const String & prefix, std::unordered_map & out, + const FsckProgress & on_progress, const Deadline & deadline, std::string_view phase) +{ + static constexpr size_t kPageLimit = 1000; + uint64_t pages = 0; + size_t count_in_page = 0; + forEachListedKey(backend, prefix, [&](const ListedKey & k) + { + out[k.key] = k.size; + if (++count_in_page == kPageLimit) + { + count_in_page = 0; + ++pages; + checkDeadline(deadline, phase); + if (on_progress && pages % PROGRESS_PAGES == 0) + on_progress(phase, out.size(), pages); + } + }, kPageLimit); + /// The walk's `backend.list` lands at least once even for an empty/undersized final page -- + /// check it here, mirroring the original per-page loop (deadline checked after every physical page). + if (count_in_page > 0 || pages == 0) + { + ++pages; + checkDeadline(deadline, phase); + } + if (on_progress) + on_progress(phase, out.size(), pages); +} + +/// Parse (writer_epoch, build_sequence) from a manifest object key. Delegates to the one shared +/// `Layout::parseManifestKey` instead of hand-rolling a second parser; returns false on a +/// malformed or foreign key. +bool parseBuildPrefix(const Layout & layout, const String & key, BuildPrefix & out) +{ + const auto parsed = layout.parseManifestKey(key); + if (!parsed) + return false; + out.writer_epoch = parsed->ref.writer_epoch; + out.build_sequence = parsed->ref.build_sequence; + return true; +} + +/// The ref-walk (which builds `reachable_blobs`/`blob_labels`) and the HEAD-confirm below run minutes +/// apart with no snapshot between them. A ref that gets republished (now names a +/// different manifest) or DROPPED in that window, combined with a legitimate GC delete of the OLD +/// blob, makes the stale walk look like a genuine dangle (a "phantom dangling") — this made the fsck +/// oracle dishonest and falsely report a dangle during long-running validation. +/// +/// Before counting a HEAD-absent blob as `Dangling`, re-resolve every `"ns/ref"` label under the same +/// immutable catalog row, using a fresh exact `_ckpt` from that original physical life. This admits a +/// same-life repoint/drop while refusing a competing rebirth. `label` is split on the LAST '/' — +/// mirroring exactly how the walk built it (`ns_str + "/" + ref_name`): `ref_name` never contains '/', +/// but `ns_str` may, so the join separator is always the rightmost one. +/// +/// Fails CLOSED on any ambiguity (a malformed label, a recovery error, a corrupt manifest): treated as +/// "still referenced", i.e. the original conservative verdict. +/// The fix can only SHRINK false positives — it must never hide a real one. +struct FsckRecoveryAuthority +{ + NamespaceLifeId life; + CatalogEntry catalog_entry; + std::optional checkpoint; +}; + +using FsckRecoveryAuthorities = std::unordered_map; +using RecordRecoveryUnchecked = std::function; + +/// Recheck one ref table against a newer `_ckpt` from the SAME physical life selected by fsck's +/// original catalog cut. The catalog row and life id never move; only the monotone checkpoint may +/// advance, which is how a same-life drop/repoint that completed during a long scan becomes visible +/// without admitting a competing rebirth. A missing or unreadable checkpoint cannot prove that an +/// old owner went away, so the caller records lost coverage and keeps the conservative verdict. +std::optional recoverLateRefTable( + Backend & backend, const Layout & layout, const FsckRecoveryAuthority & authority, + const RecordRecoveryUnchecked & record_unchecked) +{ + try + { + const std::optional sampled = readCkpt(backend, layout, authority.life); + if (!sampled) + { + record_unchecked(authority.life.ns, layout.refCkptKey(authority.life), + "late ref recheck: the original life checkpoint is absent"); + return std::nullopt; + } + return recoverRefTableDetailedFromAuthority( + backend, layout, authority.catalog_entry, sampled->ckpt).state; + } + catch (const Exception & e) + { + record_unchecked(authority.life.ns, layout.refCkptKey(authority.life), + "late ref recheck: the original life checkpoint or replay is unreadable: " + e.message()); + return std::nullopt; + } + catch (...) + { + record_unchecked(authority.life.ns, layout.refCkptKey(authority.life), + "late ref recheck: the original life checkpoint or replay could not be read"); + return std::nullopt; + } +} + +bool blobStillReferenced(Pool & store, const Layout & layout, + const FsckRecoveryAuthorities & authorities, const String & bkey, + const std::vector & labels, const Deadline & deadline, + const RecordRecoveryUnchecked & record_unchecked) +{ + if (labels.empty()) + return true; + for (const String & label : labels) + { + checkDeadline(deadline, "re-resolving refs at HEAD-absent"); + const size_t slash = label.rfind('/'); + if (slash == String::npos) + return true; /// malformed label — cannot re-resolve, fail closed + const String ns_part = label.substr(0, slash); + const String ref_name = label.substr(slash + 1); + try + { + /// Never read a second catalog cut here. A later rebirth may name the same logical namespace + /// but it is not the life whose original row made this blob reachable in this fsck pass. + const auto authority_it = authorities.find(ns_part); + if (authority_it == authorities.end()) + { + record_unchecked(RootNamespace{ns_part}, layout.refCatalogKey(), + "late blob recheck: no original Live/Removing authority was retained"); + return true; /// no original Live/Removing authority -- fail closed + } + const RootNamespace rns{ns_part}; + const std::optional table = recoverLateRefTable( + store.backend(), layout, authority_it->second, record_unchecked); + if (!table) + return true; + const auto rit = table->getCommitted().find(ref_name); + if (rit == table->getCommitted().end()) + continue; /// the ref was DROPPED since the walk — this label no longer applies + const PartManifest body = store.readManifest(ManifestId{rns, rit->second.manifest_ref}); + for (const ManifestEntry & e : body.entries) + { + if (e.placement != EntryPlacement::Blob) + continue; + if (layout.blobKey(e.ref) == bkey) + return true; /// an original-life ref still names this exact blob — a real dangle + } + } + catch (...) + { + return true; /// cannot confirm the ref moved away — keep the conservative verdict + } + } + return false; /// no original-life label names this blob — the stale-walk artifact is gone +} + +/// The manifest sibling of the `blobStillReferenced` recheck above. The ref-walk captures each committed +/// `(ref_name -> manifest_ref)` from a FRESH per-namespace recovery, but the `backend.get(mkey)` that +/// confirms the manifest body runs LATER in the same (possibly long) namespace loop. A ref republished to +/// a DIFFERENT manifest — or DROPPED — in that window, combined with a legitimate GC delete of the OLD +/// manifest body, makes the stale captured row look like a committed ref over a missing manifest (a +/// "phantom dangling manifest"), the same dishonest-oracle failure `blobStillReferenced` kills for blobs. +/// +/// Before counting a missing manifest body as `Dangling`, re-resolve the EXACT ref from the SAME frozen +/// catalog row with a fresh exact `_ckpt` from that original physical life, then check whether the +/// committed row still names THIS exact manifest key. A later catalog cut must not replace that row, +/// but a same-life checkpoint advance must be visible. Fails CLOSED on any ambiguity (a throw, a corrupt +/// table): treated as "still referenced", the original conservative verdict — the fix can only SHRINK +/// false positives, never hide a real loss. +bool manifestStillReferenced(Backend & backend, const Layout & layout, const RootNamespace & ns, + const FsckRecoveryAuthorities & authorities, const String & ref_name, + const String & mkey, const Deadline & deadline, + const RecordRecoveryUnchecked & record_unchecked) +{ + checkDeadline(deadline, "re-resolving ref at missing-manifest"); + try + { + const auto authority_it = authorities.find(ns.string()); + if (authority_it == authorities.end()) + { + record_unchecked(ns, layout.refCatalogKey(), + "late manifest recheck: no original Live/Removing authority was retained"); + return true; /// no original Live/Removing authority -- fail closed + } + const std::optional table = recoverLateRefTable( + backend, layout, authority_it->second, record_unchecked); + if (!table) + return true; + const auto rit = table->getCommitted().find(ref_name); + if (rit == table->getCommitted().end()) + return false; /// the ref was DROPPED since the walk — no longer a committed owner + /// A republish moved the ref to a different manifest key: this old key is no longer owned. + return layout.manifestKey(ManifestId{ns, rit->second.manifest_ref}) == mkey; + } + catch (...) + { + return true; /// cannot confirm the ref moved away — keep the conservative verdict + } +} + +String renderId(const RefTxnId & id) +{ + return std::to_string(id.writer_epoch) + "-" + std::to_string(id.ref_sequence); +} + +/// Per-NAMESPACE verdicts of the stream audit. Both counters count namespaces, not rows: a namespace +/// has exactly one answer about its stream even when several checks reach it. +/// +/// A namespace PROVEN broken is never also counted `unchecked`. "Proved broken" and "could not prove" +/// are different answers, and letting the second overwrite or accompany the first would turn a fatal +/// into an ambiguity — the recovery path throws on a holed stream, so a chain-broken namespace reliably +/// produces a downstream failure too, and that failure must not dilute the verdict that explains it. +struct NsVerdicts +{ + std::set chain_broken; + std::set unchecked; + + void recordChainBroken(FsckReport & report, const RootNamespace & ns, const String & key, String note) + { + chain_broken.insert(ns.string()); + unchecked.erase(ns.string()); + push(report, key, FsckClass::ChainBroken, std::move(note)); + } + + void recordUnchecked(FsckReport & report, const RootNamespace & ns, const String & key, String note) + { + if (chain_broken.contains(ns.string())) + return; + unchecked.insert(ns.string()); + push(report, key, FsckClass::Unchecked, std::move(note)); + } + + /// Both classes are emitted in EVERY mode, not just `detail`: they are namespace verdicts, bounded + /// by the namespace count, and a summary run that hid them would report a number nobody could act on. + void push(FsckReport & report, const String & key, FsckClass cls, String note) const + { + FsckObject o; + o.key = key; + o.kind = ObjectKind::Blob; /// ref objects have no ObjectKind; reuse Blob as the generic kind + o.size = 0; + o.cls = cls; + o.reachable_from = {std::move(note)}; + report.objects.push_back(std::move(o)); + } + + void publish(FsckReport & report) const + { + report.chain_broken = chain_broken.size(); + report.unchecked = unchecked.size(); + } +}; + +/// THE ARITHMETIC STREAM WALK (spec §7). Read-only, one namespace. +/// +/// The frozen catalog row and exact `_ckpt` define the complete finite walk. LIST supplies no genesis, +/// witness, frontier or stop condition, and the walker never probes the position after +/// `_ckpt.committed_through`. Every required id is point-read from the checkpoint base's successor (or +/// `{life_epoch, 1}`) through that inclusive frontier. A missing required id is therefore a proven hole; +/// no above-hole listing witness is needed. An epoch seal advances directly to the next epoch's first id, +/// exactly as authoritative read-only recovery does. +void checkRefStream(Backend & backend, const Layout & layout, const NamespaceLifeId & life, + const CatalogEntry & catalog_entry, const std::optional & checkpoint_sample, + const Deadline & deadline, FsckReport & report, NsVerdicts & verdicts) +{ + checkDeadline(deadline, "ref stream"); + const RootNamespace & ns = life.ns; + const std::optional checkpoint + = checkpoint_sample ? std::optional{checkpoint_sample->ckpt} : std::nullopt; + const RecoveryGrounding grounding = chooseRecoveryGrounding(catalog_entry, checkpoint); + if (grounding.base) + { + try + { + /// Even when the base IS the frontier and there is no replay tail, a checkpoint may not + /// turn an `EpochSeal` into a state snapshot by naming a same-id `_snap`. + (void)readCheckpointSnapshotBase(backend, layout, life, *checkpoint); + } + catch (const Exception & e) + { + const String key = layout.refSnapshotKey(life, *grounding.base); + const String note = "ref stream: checkpoint snapshot base " + renderId(*grounding.base) + + " is invalid: " + e.message(); + if (e.code() != ErrorCodes::CORRUPTED_DATA) + { + verdicts.recordUnchecked(report, ns, key, note); + return; + } + + /// A concurrent checkpoint advance may retire the sampled base between these exact reads. + /// Only the SAME checkpoint incarnation turns a missing/invalid member of its required + /// triple into durable corruption. A changed, absent, or unreadable authority proves no + /// such thing and remains the honest `Unchecked` answer. + checkDeadline(deadline, "checkpoint-base authority revalidation"); + try + { + const std::optional current = readCkpt(backend, layout, life); + if (!current || !checkpoint_sample || current->token != checkpoint_sample->token) + { + verdicts.recordUnchecked(report, ns, key, + note + "; checkpoint authority changed while validating its snapshot base"); + return; + } + } + catch (const Exception & revalidation_error) + { + verdicts.recordUnchecked(report, ns, key, + note + "; checkpoint authority could not be revalidated: " + revalidation_error.message()); + return; + } + catch (...) + { + verdicts.recordUnchecked(report, ns, key, + note + "; checkpoint authority could not be revalidated"); + return; + } + + verdicts.recordChainBroken(report, ns, key, note); + return; + } + } + if (!grounding.walk_from || !grounding.committed_through) + return; + + RefTxnId expected = *grounding.walk_from; + while (expected <= *grounding.committed_through) + { + checkDeadline(deadline, "ref stream"); + const auto got = backend.get(layout.refLogKey(life, expected)); + if (!got) + { + verdicts.recordChainBroken(report, ns, layout.refLogKey(life, expected), + "ref stream: checkpoint requires id " + renderId(expected) + " at or below inclusive frontier " + + renderId(*grounding.committed_through) + ", but its exact key is absent"); + return; + } + + bool is_seal = false; + try + { + is_seal = refLogTxnIsEpochSeal( + decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), expected)); + } + catch (const Exception & e) + { + verdicts.recordUnchecked(report, ns, layout.refLogKey(life, expected), + "ref stream: the checkpoint-required record at " + renderId(expected) + + " could not be decoded: " + e.message()); + return; + } + ++report.ref_records_walked; + + try + { + if (const std::optional next = nextRefLogIdWithinCommittedFrontier( + expected, is_seal, *grounding.committed_through)) + expected = *next; + else + break; + } + catch (const Exception & e) + { + verdicts.recordChainBroken(report, ns, layout.refLogKey(life, expected), + "ref stream: " + e.message()); + return; + } + } +} + +/// Perform the scan and accumulate into `report`. This helper owns the read-only traversal: it first +/// recovers authoritative refs, then checks physical objects and GC labels, while preserving the +/// distinction between a missing live object and expected in-flight cleanup. Deadline exceptions are +/// intentionally left to `runFsck`, which decides whether partial results were requested. +void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, const Deadline & deadline, + const String & namespace_prefix, FsckReport & report) +{ + const Layout & layout = store.layout(); + Backend & backend = store.backend(); + /// Path-derived per-object algorithm parsing: every listed blob-tree key -- across every + /// admitted algo, not just the pool's node-local write algo -- is classified via + /// `Layout::parseBlobKey`, which derives the `BlobRef` from the key's OWN `` path segment + /// (and its `.meta` sibling). A foreign/malformed key (unknown algo segment, wrong-width hex, a + /// non-`.meta`/non-blob shape) parses to `std::nullopt` and is classified as debris, never an + /// exception. + + /// Reachability is recomputed from the authoritative refs (never from GC state): + /// for each namespace, each committed ref resolves to a ManifestId; read its body; a committed ref + /// naming a MISSING body is an ERROR (Dangling); a present body whose blobs are missing is an ERROR. + std::set reachable_blobs; /// blob object keys named by a live owner + std::set owned_manifest_keys; /// manifest object keys named by a committed owner + /// blob key -> "ns/ref" labels of the refs that named it. Always populated (not just under + /// `detail`) — the HEAD-absent re-resolve below needs it in every mode. + std::unordered_map> blob_labels; + + uint64_t refs_walked = 0; + NsVerdicts verdicts; + SCOPE_EXIT({ verdicts.publish(report); }); + const RecordRecoveryUnchecked record_recovery_unchecked = + [&](const RootNamespace & ns, const String & key, const String & detail_text) + { + verdicts.recordUnchecked(report, ns, key, detail_text); + }; + + /// RECORD AND CONTINUE for a key that belongs to no namespace at all. fsck is the forensic tool an + /// operator reaches for once something is already wrong, so a key it cannot attribute must become a + /// FINDING and not an abort: an audit that died on the first bad key would report nothing about the + /// healthy namespaces it never reached, which is the wrong failure order for a read-only diagnostic. + /// + /// `seen` is what makes the count a count of DEFECTS: each sweep below enumerates namespaces again + /// and sees the same offending key, and only the first sighting is recorded. + std::set lifeless_seen; + auto recordLifelessKeys = [&](const NamespaceListing & listing) + { + for (const UnattributableNamespaceKey & bad : listing.skipped) + { + if (!lifeless_seen.insert(bad.key).second) + continue; + ++report.lifeless_keys; + FsckObject o; + o.key = bad.key; + o.kind = ObjectKind::Blob; /// a lifeless key has no ObjectKind; reuse Blob as the generic kind + o.cls = FsckClass::LifelessKey; + o.size = 0; + o.reachable_from = {bad.reason}; + report.objects.push_back(std::move(o)); + } + }; + + /// One immutable cut owns every physical-id join in this walk. `Creating` participates in that + /// attribution (its physical keys may exist) but is never recovered: only Live/Removing rows have a + /// durable publication frontier. A diagnostic records duplicate ids and keeps walking unrelated + /// unique lives. + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + struct FsckWalkLife + { + NamespaceLifeId life; + CatalogEntry catalog_entry; + }; + std::vector walk_lives; + walk_lives.reserve(catalog_cut.catalog.entries.size()); + for (const CatalogEntry & entry : catalog_cut.catalog.entries) + { + if (!entry.ns.string().starts_with(namespace_prefix)) + continue; + if (entry.state == NsState::Creating) + continue; + try + { + if (const auto life = catalog_cut.life_index.resolve(entry.incarnation)) + walk_lives.push_back(FsckWalkLife{.life = *life, .catalog_entry = entry}); + } + catch (const Exception & e) + { + if (e.code() != ErrorCodes::CORRUPTED_DATA) + throw; + recordLifelessKeys(NamespaceListing{{}, {{ + layout.refCatalogKey() + "#" + renderIncarnation(entry.incarnation), e.message()}}}); + } + } + + /// Physical life-owned keys carry no logical name. Classify each COMPLETE, canonical key against a + /// catalog cut taken AFTER this physical listing finishes (observe-then-cut), not the earlier + /// `catalog_cut` above: `NamespaceJanitor::runOnePage` (the only real deleter of this debris) uses + /// the identical ordering, and it is what makes "life absent from a LATER cut" sound -- creation + /// always admits a `Creating` catalog row before writing any life-owned object (spec §2), so a life + /// that is absent from a cut taken after the listing cannot be a concurrent birth this listing raced. + /// A malformed shape (the parser refuses, or the reserved segment names no clean relative file) is + /// classified immediately as it cannot become residue no matter which cut resolves it. + if (namespace_prefix.empty()) + { + struct CanonicalNamespaceKey + { + String key; + uint64_t size; + NamespaceLifePhysicalId life_id; + }; + std::vector canonical_candidates; + + forEachListedKey(backend, layout.namespaceRootPrefix(), [&](const ListedKey & listed) + { + std::optional physical_id; + try + { + if (const auto ref_object = layout.parseRefObjectKey(listed.key)) + physical_id = ref_object->life_id; + else if (const auto checkpoint = layout.parseRefCkptKey(listed.key)) + physical_id = *checkpoint; + else if (const auto namespace_file = layout.parseNamespaceFileKey(listed.key)) + physical_id = namespace_file->life_id; + else + { + recordLifelessKeys(NamespaceListing{{}, {{listed.key, "unrecognized key under the namespace ownership tree"}}}); + return; + } + } + catch (const Exception & e) + { + if (e.code() != ErrorCodes::CORRUPTED_DATA) + throw; + recordLifelessKeys(NamespaceListing{{}, {{listed.key, e.message()}}}); + return; + } + canonical_candidates.push_back(CanonicalNamespaceKey{listed.key, listed.size, *physical_id}); + }); + + /// The post-observation cut. All three catalog states -- `Creating`, `Live`, `Removing` -- + /// protect a life for this purpose; only a life absent from every one of them is residue. + const CasRefCatalog::Snapshot post_listing_cut = CasRefCatalog::read(backend, layout); + std::unordered_set pending_lives; + for (const CanonicalNamespaceKey & candidate : canonical_candidates) + { + try + { + if (post_listing_cut.life_index.resolve(candidate.life_id)) + continue; /// protected by some catalog state as of the later cut -- not residue + } + catch (const Exception & e) + { + /// The reverse life index throws `CORRUPTED_DATA` when the post-listing cut carries a + /// duplicated life id: a catalog defect, not evidence about THIS key. Record and keep + /// walking, same as every other catalog-authority failure in this scan -- an audit that + /// aborted on the first bad key would report nothing about the healthy candidates + /// still queued behind it. + if (e.code() != ErrorCodes::CORRUPTED_DATA) + throw; + recordLifelessKeys(NamespaceListing{{}, {{candidate.key, e.message()}}}); + continue; + } + ++report.namespace_janitor_pending; + report.namespace_janitor_pending_bytes += candidate.size; + pending_lives.insert(candidate.life_id); + FsckObject o; + o.key = candidate.key; + o.kind = ObjectKind::Blob; /// no ObjectKind names namespace-life debris; reuse Blob as the generic kind + o.cls = FsckClass::JanitorPending; + o.size = candidate.size; + o.reachable_from = {"physical life id is absent from a catalog cut taken after this listing; " + "janitor-pending, not corruption"}; + report.objects.push_back(std::move(o)); + } + report.namespace_janitor_pending_lives = pending_lives.size(); + } + + /// Every replay and late recheck below reuses the same exact catalog row and physical life. The + /// primary walk also retains its checkpoint sample; a late recheck exact-reads `_ckpt` again at that + /// SAME life so a concurrent same-life drop/repoint is visible without ever accepting a rebirth. + FsckRecoveryAuthorities recovery_authorities; + recovery_authorities.reserve(walk_lives.size()); + + for (const FsckWalkLife & walk_life : walk_lives) + { + const NamespaceLifeId & life = walk_life.life; + const RootNamespace & ns = life.ns; + const String & ns_str = ns.string(); + /// RECORD AND CONTINUE, NEVER WEDGE. Everything below is per-namespace, and every one of these + /// steps can raise `CORRUPTED_DATA` on a namespace whose stream is damaged — the replay refuses a + /// non-contiguous tail, the codecs refuse an invalid body. For RECOVERY that throw is the correct + /// fail-close; for a read-only diagnostic it is a bug, because the audit then reports NOTHING + /// about the namespaces it never reached, including the healthy ones. So one namespace's failure + /// becomes that namespace's verdict and the sweep goes on. + /// + /// `TIMEOUT_EXCEEDED` is deliberately NOT caught: the deadline is a property of the whole scan, + /// and `runFsck`'s `partial` handling owns it. + try + { + /// One materialized `_ckpt` body is part of this namespace's frozen audit authority. The + /// recovery API receives exactly these bytes; `checkRefStream` receives the same decoded + /// value, so the two legs cannot quietly choose different frontiers after a concurrent CAS. + const std::optional checkpoint_sample = readCkpt(backend, layout, life); + const std::optional checkpoint + = checkpoint_sample ? std::optional{checkpoint_sample->ckpt} : std::nullopt; + const auto [authority_it, inserted] = recovery_authorities.emplace( + ns.string(), FsckRecoveryAuthority{ + .life = life, .catalog_entry = walk_life.catalog_entry, .checkpoint = checkpoint}); + chassert(inserted); + + /// The arithmetic stream audit runs FIRST, so a holed stream gets the verdict that EXPLAINS + /// it (`chain-broken`) rather than the downstream `CORRUPTED_DATA` the replay below would + /// raise about the same hole. + checkRefStream( + backend, layout, life, walk_life.catalog_entry, checkpoint_sample, deadline, report, verdicts); + + /// This recovery's finite range comes from the original catalog row and exact `_ckpt`, never + /// from a stream listing, a self-resolved name, or an F+1 probe. + const RefTableState table = recoverRefTableDetailedFromAuthority( + backend, layout, authority_it->second.catalog_entry, authority_it->second.checkpoint).state; + for (const auto [ref_name, row] : table.getCommitted()) + { + const ManifestId id{ns, row.manifest_ref}; + const String mkey = layout.manifestKey(id); + owned_manifest_keys.insert(mkey); + const String label = ns_str + "/" + ref_name; + + const auto got = backend.get(mkey); + if (!got) + { + /// A committed ref naming a missing manifest body would be an INV-NO-DANGLE violation — + /// but the per-ref GET runs later than the namespace's ref recovery, so a stale captured + /// row plus a legitimate GC delete of a since-superseded manifest can masquerade as one, + /// and a bare GET can lag a present object. Revalidate exactly like the blob `Dangling` + /// recheck below: HEAD the exact object AND re-resolve under the original catalog row + /// plus a fresh checkpoint from its physical life. Count the dangle ONLY when the exact + /// object is HEAD-absent AND that life still names THIS exact manifest — otherwise it is + /// LIST/GET lag or a phantom stale-row, never a loss. + if (!backend.head(mkey).exists + && manifestStillReferenced(backend, layout, ns, recovery_authorities, ref_name, mkey, + deadline, record_recovery_unchecked)) + { + ++report.dangling; + FsckObject o; + o.key = mkey; + o.kind = ObjectKind::Blob; /// manifests have no ObjectKind; reuse Blob as the generic kind + o.size = 0; + o.cls = FsckClass::Dangling; + o.reachable_from = {label}; + report.objects.push_back(std::move(o)); + } + /// A present object is GET lag. A row not named by the original-life authority is a + /// stale-walk artifact, not a dangle; its original owner cannot contribute blobs. + ++refs_walked; + continue; + } + + PartManifest body = decodePartManifest(openObject(FormatId::PartManifest, got->bytes)); + if (!refMatchesBody(id.ref, body) || !manifestNamespaceMatches(id.root_namespace, body)) + { + ++report.dangling; + FsckObject o; + o.key = mkey; + o.kind = ObjectKind::Blob; + o.size = got->bytes.size(); + o.cls = FsckClass::Dangling; + o.reachable_from = {label}; + report.objects.push_back(std::move(o)); + ++refs_walked; + continue; + } + + for (const ManifestEntry & e : body.entries) + { + if (e.placement != EntryPlacement::Blob) + continue; + const String bkey = layout.blobKey(e.ref); + reachable_blobs.insert(bkey); + ++report.total_blob_refs; + report.referenced_logical_bytes += e.blob_size; + blob_labels[bkey].push_back(label); + } + + ++refs_walked; + checkDeadline(deadline, "walking refs"); + if (on_progress && refs_walked % 64 == 0) + on_progress("walking refs", reachable_blobs.size(), refs_walked); + } + } + catch (const Exception & e) + { + if (e.code() == ErrorCodes::TIMEOUT_EXCEEDED) + throw; + verdicts.recordUnchecked(report, ns, + layout.namespaceStreamPrefix(life), + "fsck could not examine this namespace: " + e.message()); + } + } + report.distinct_blobs = reachable_blobs.size(); + + /// Scoped mode skips the GLOBAL physical classification below: it is meaningless under a + /// filter (blobs owned by other namespaces would read as unreachable) and would cost a + /// pool-wide LIST for what should be O(scoped refs). + if (namespace_prefix.empty()) + { + /// Physical listing: blobs + manifest bodies. The per-hash `.meta` descriptor sibling + /// (`blobMetaKey(id) == blobKey(id) + ".meta"`) lives under the SAME + /// `blobsPrefix()` as the body, so partition the raw LIST into bodies vs `.meta` objects up + /// front — a `.meta` key must never be classified as a content body (it would otherwise be + /// misread as an unreferenced blob and fall into the dangling/pending/unaccounted pipeline + /// below), and a body must never be misread as a `.meta`. + std::unordered_map present_all; + listAll(backend, layout.blobsPrefix(), present_all, on_progress, deadline, "listing blobs"); + std::unordered_map present_blobs; + std::unordered_set present_meta_hashes; + present_blobs.reserve(present_all.size()); + for (const auto & [key, sz] : present_all) + { + if (key.ends_with(".meta")) + { + if (const std::optional ref = layout.parseBlobKey(key)) + present_meta_hashes.insert(*ref); + /// else: foreign key shape under blobs/ — not ours to pair + } + else + present_blobs.emplace(key, sz); + } + for (const auto & [_, sz] : present_blobs) + report.physical_bytes += sz; + + /// Reachable blobs must be present (HEAD-confirm against LIST lag before declaring loss). + for (const String & bkey : reachable_blobs) + { + auto it = present_blobs.find(bkey); + bool exists = it != present_blobs.end(); + uint64_t size = exists ? it->second : 0; + if (!exists) + { + const HeadResult h = backend.head(bkey); + if (h.exists) + { + exists = true; + size = h.size; + report.physical_bytes += h.size; + } + } + + const auto lit = blob_labels.find(bkey); + if (!exists) + { + /// Before declaring a loss, re-resolve the referencing refs from the original audit + /// authority. A later rebirth must not replace the old owner while this verdict is being + /// decided. + const bool still_referenced = blobStillReferenced(store, layout, recovery_authorities, bkey, + lit != blob_labels.end() ? lit->second : std::vector{}, deadline, + record_recovery_unchecked); + if (!still_referenced) + continue; /// stale-walk artifact: neither reachable nor dangling — skip entirely + } + + if (exists) + ++report.reachable; + else + ++report.dangling; + if (detail || !exists) + { + FsckObject o; + o.key = bkey; + o.kind = ObjectKind::Blob; + o.size = size; + o.cls = exists ? FsckClass::Reachable : FsckClass::Dangling; + if (detail && lit != blob_labels.end()) + o.reachable_from = lit->second; + report.objects.push_back(std::move(o)); + } + } + + /// Present-but-unreferenced blobs: classify through the GC pipeline view instead of one + /// suspicious "unreachable" lump (the multi-stage graduation keeps a nonzero churning + /// set here on ANY active pool, and beta testers read "unreachable" as a leak). The GC state is + /// read for LABELING ONLY — reachability above never consults it. + std::unordered_map retired_by_hash; + std::unordered_set unref_hashes; + std::unordered_set in_run_hashes; + /// The NON-SENTINEL source edges the snapshot still holds on each unreferenced blob, collected in + /// `detail` mode only. `in_run_hashes` alone answers "does GC still see this blob at all"; the + /// stale-edge cross-check below needs the edge IDENTITIES so it can ask whether their source + /// manifests still exist. Sentinel rows (`source_id == 0` — `kZeroMarker`/`kCondemned`) are not + /// edges and are excluded. + std::unordered_map, BlobRefHash> unref_edge_sources; + bool have_gc_state = false; + + for (const auto & [bkey, sz] : present_blobs) + if (!reachable_blobs.contains(bkey)) + { + if (const std::optional ref = layout.parseBlobKey(bkey)) + unref_hashes.insert(*ref); + } + + if (!unref_hashes.empty()) + { + if (const auto state_got = backend.get(layout.gcStateKey())) + { + have_gc_state = true; + const GcState gc_state = decodeGcState(state_got->bytes); + /// The adopted fold seal names the snapshot runs; resolution is by ref, never by key + /// construction. Every row whose hash is in our candidate set marks "known to GC" — + /// edges still counted (drop unfolded), an explicit zero-marker mid-pipeline, or a + /// `kCondemned` sentinel row that carries the condemned state (retired-in-snapshot): + /// the `kCondemned` rows feed `retired_by_hash` (the `PendingGc` classification) in the + /// SAME pass, replacing the removed `retired_refs`/`decodeRetiredSet` loop. + /// + /// These sets are keyed by the full `BlobRef`, not a narrowed digest. The run's own + /// algorithm-prefixed key is parsed by `SourceEdgeKeyCodec` and compared directly with + /// the full identity parsed from the listed blob key. This is required for mixed-algorithm + /// pools: a 64-hex digest must not be truncated or compared as though it used the pool's + /// local write algorithm, or its true GC state could be hidden as `Unaccounted`. + if (const auto seal_got = backend.get(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt))) + { + uint64_t rows = 0; + for (const RunRef & run : decodeFoldSeal(seal_got->bytes, gc_state.snap_generation).blob_target_runs) + { + checkDeadline(deadline, "reading gc snapshot runs"); + /// Typed open: the source-edge run reader goes through openSourceEdgeRun (the NDJSON + /// header gates type == cas_run + kind == source_edge). Fsck keys off the row's hash + /// (the record's own algo-prefixed key, never from pool meta). + SourceEdgeRunView reader = openSourceEdgeRun(backend, run.key); + String key; + String payload; + while (reader.next(key, payload)) + { + BlobRef ref; + UInt128 source_id; + SourceEdgeKeyCodec::parse(key, ref, source_id); // throws CORRUPTED_DATA on malformed (fail-closed) + if (unref_hashes.contains(ref)) + { + in_run_hashes.insert(ref); + if (detail && source_id != UInt128{0}) + unref_edge_sources[ref].push_back(source_id); + if (!payload.empty() && payload[0] == kCondemned) + { + const CondemnedRow row = decodeCondemnedRow(payload); + RetiredEntry e; + e.kind = ObjectKind::Blob; + e.ref = ref; + e.token = row.token; + e.size = row.size; + e.condemn_round = row.condemn_round; + e.delete_pending = row.delete_pending; + retired_by_hash.emplace(ref, std::move(e)); + } + } + if (on_progress && ++rows % 65536 == 0) + on_progress("reading gc snapshot runs", in_run_hashes.size(), rows); + } + /// Whole-file seal checksum: compare the drained run's accumulated + /// checksum to the seal's `RunRef::checksum`. Fsck is a read-only auditor — instead of + /// throwing (which would abort the whole scan on the first corrupt run), catalogue the + /// mismatch as a `CorruptedRun` finding (with the run key) and continue so the audit + /// enumerates every problem in one pass. The deletion-deriving consumers + /// (`fold`/`zeroInDegree`/`previewDeletes`) still fail closed on the same mismatch. + if (reader.accumulatedChecksum() != run.checksum) + { + ++report.corrupted_runs; + if (detail) + report.objects.push_back(FsckObject{.key = run.key, .cls = FsckClass::CorruptedRun, .reachable_from = {}}); + } + } + } + } + } + + /// STALE-EDGE cross-check. A residual `+1` whose matching `-1` never folded pins its blob at + /// in-degree 1 forever: every GC round recomputes the same nonzero in-degree and never nominates + /// the blob, so the `AwaitingGc` "expected, no action needed" label is a lie — nothing will ever + /// reclaim it. The edge names its source, so the check is to ask whether that source still exists: + /// build the set of source ids that every manifest body PRESENT in the pool would contribute, and + /// treat an edge outside that set as one whose source manifest is gone. + /// + /// COST: one LIST per namespace plus one GET per manifest body. It is therefore gated on `detail` + /// — the cheap summary path (the ca-soak fixpoint poll calls it in a loop) must not gain a single + /// extra request — and additionally on some unreferenced blob actually carrying a real edge, so a + /// pool with nothing to cross-check pays nothing. + /// + /// `stale_edge_check_available` is the fail-closed switch: a manifest body we cannot decode would + /// silently withhold its edges from the live set and turn every blob it owns into a false hard + /// finding, so one undecodable body disables the whole cross-check for this scan rather than + /// manufacture an error. The check may only ever SHRINK to silence, never invent a finding. + std::unordered_set live_source_ids; + bool stale_edge_check_available = detail && !unref_edge_sources.empty(); + if (stale_edge_check_available) + { + const NamespaceListing stale_edge_listing = store.listNamespaces(namespace_prefix); + recordLifelessKeys(stale_edge_listing); + for (const String & ns_str : stale_edge_listing.namespaces) + { + const RootNamespace ns{ns_str}; + std::unordered_map manifest_bodies; + listAll(backend, layout.manifestNamespacePrefix(ns), manifest_bodies, on_progress, deadline, + "listing manifests for the stale-edge check"); + for (const auto & [mkey, _] : manifest_bodies) + { + checkDeadline(deadline, "reading manifests for the stale-edge check"); + const std::optional id = layout.parseManifestKey(mkey); + if (!id) + continue; /// foreign/malformed key under `manifests/` — contributes no source edge + const auto got = backend.get(mkey); + if (!got) + continue; /// gone between the LIST and the GET — genuinely not a live source + try + { + const PartManifest body = decodePartManifest(openObject(FormatId::PartManifest, got->bytes)); + for (const ManifestEntry & e : body.entries) + if (e.placement == EntryPlacement::Blob) + live_source_ids.insert(sourceEdgeId(*id, e.path)); + } + catch (...) + { + stale_edge_check_available = false; /// incomplete live set — do not accuse anyone + break; + } + } + if (!stale_edge_check_available) + break; + } + } + + for (const auto & [bkey, sz] : present_blobs) + { + if (reachable_blobs.contains(bkey)) + continue; + ++report.unreachable; + + /// A foreign/malformed key (`parseBlobKey` -> `nullopt`) falls back to the default `BlobRef{}`, + /// which cannot match a real `retired_by_hash`/`in_run_hashes` entry — it lands in the generic + /// `Unaccounted` bucket below, exactly the "debris, not ours" classification `parseBlobKey` + /// documents: foreign algorithm segments are debris, not pool objects. + const BlobRef hash = layout.parseBlobKey(bkey).value_or(BlobRef{}); + + FsckClass cls = FsckClass::Unaccounted; + String note; + if (const auto rit = retired_by_hash.find(hash); rit != retired_by_hash.end() + && backend.head(bkey).token == rit->second.token) + { + /// The PRESENT incarnation is the condemned one — deletion is scheduled. A token + /// mismatch means the listed entry belongs to a displaced older incarnation and says + /// nothing about this object; fall through to the snapshot check. + cls = FsckClass::PendingGc; + note = rit->second.delete_pending + ? "delete_pending: exact-token delete executes next GC round" + : "condemned at round " + std::to_string(rit->second.condemn_round) + + "; graduates once every writer acks past it (expected)"; + } + else if (in_run_hashes.contains(hash)) + { + /// `in_run_hashes` only says the GC snapshot still holds SOMETHING for this blob. Split on + /// whether any of it is still actionable. One edge whose source manifest is PRESENT keeps + /// the ordinary `AwaitingGc` verdict — that manifest's removal still folds its `-1`, and an + /// unowned-but-present manifest is reclaimed by the orphan sweep, so the blob is genuinely + /// mid-pipeline. When EVERY edge names a manifest that no longer exists, no `-1` is left to + /// fold: the in-degree is pinned above zero for good and only a rebuild can clear it. + uint64_t stale_edges = 0; + bool all_edges_stale = false; + if (const auto eit = unref_edge_sources.find(hash); + stale_edge_check_available && eit != unref_edge_sources.end() && !eit->second.empty()) + { + for (const UInt128 & source_id : eit->second) + if (!live_source_ids.contains(source_id)) + ++stale_edges; + all_edges_stale = stale_edges == eit->second.size(); + } + + if (all_edges_stale) + { + cls = FsckClass::StaleEdge; + note = "all " + std::to_string(stale_edges) + " source edges name manifests that no longer " + "exist — unreclaimable by the incremental GC (needs `cas-gc-rebuild`); NOT expected, investigate"; + } + else + { + cls = FsckClass::AwaitingGc; + note = "edges still in the GC snapshot; the drop has not folded yet (expected)"; + } + } + else if (!have_gc_state) + { + cls = FsckClass::AwaitingGc; + note = "GC has not run on this pool yet"; + } + else + { + note = "not in the current GC view — transient for a fast create+drop between rounds; " + "PERSISTENT occurrences violate INV-2 (reachability-before-content), investigate"; + } + + switch (cls) + { + case FsckClass::PendingGc: ++report.pending_gc; break; + case FsckClass::AwaitingGc: ++report.awaiting_gc; break; + case FsckClass::StaleEdge: ++report.stale_edge; break; + default: ++report.unaccounted; break; + } + if (detail) + { + FsckObject o; + o.key = bkey; + o.kind = ObjectKind::Blob; + o.size = sz; + o.cls = cls; + o.reachable_from = {std::move(note)}; + report.objects.push_back(std::move(o)); + } + } + + /// Meta <-> body pairing: a `.meta` object with no + /// body is an INV-META-BODY violation (the fixed meta/body lifecycle never leaves a meta + /// orphaned of its body) — a real ERROR, distinct from `dangling` (which is reachability-driven). + /// A body with no `.meta` is a benign not-yet-adopted (or interrupted-birth) artifact, NOT a dangle + /// — it still classifies through the ordinary present-but-unreferenced pipeline above. + std::unordered_set present_body_hashes; + present_body_hashes.reserve(present_blobs.size()); + for (const auto & [bkey, _] : present_blobs) + if (const std::optional ref = layout.parseBlobKey(bkey)) + present_body_hashes.insert(*ref); + /// else: foreign key shape under blobs/ — not ours to pair + for (const BlobRef & hash : present_meta_hashes) + if (!present_body_hashes.contains(hash)) + ++report.meta_without_body; + for (const BlobRef & hash : present_body_hashes) + if (!present_meta_hashes.contains(hash)) + ++report.body_without_meta; + } + else + { + /// Scoped mode: dangling-only for the selected namespaces. Each blob named by a scoped ref + /// is HEAD-verified (O(scoped refs), no pool-wide LIST); the unreachable/pending pipeline + /// classification needs the whole pool and is intentionally skipped. + for (const String & bkey : reachable_blobs) + { + checkDeadline(deadline, "head-checking scoped blobs"); + const HeadResult h = backend.head(bkey); + const auto lit = blob_labels.find(bkey); + bool exists = h.exists; + if (!exists) + { + /// Use the same HEAD-absent re-resolve as the global-mode loop above. + const bool still_referenced = blobStillReferenced(store, layout, recovery_authorities, bkey, + lit != blob_labels.end() ? lit->second : std::vector{}, deadline, + record_recovery_unchecked); + if (!still_referenced) + continue; /// stale-walk artifact — neither reachable nor dangling + } + if (exists) + { + ++report.reachable; + report.physical_bytes += h.size; + } + else + ++report.dangling; + if (detail || !exists) + { + FsckObject o; + o.key = bkey; + o.kind = ObjectKind::Blob; + o.size = exists ? h.size : 0; + o.cls = exists ? FsckClass::Reachable : FsckClass::Dangling; + if (detail && lit != blob_labels.end()) + o.reachable_from = lit->second; + report.objects.push_back(std::move(o)); + } + } + } + + /// Pre-precommit manifest debris: a `cas/manifests/` body with no committed owner. An ELIGIBLE prefix's + /// orphan is reclaimable debris => INFO (Unreachable); a non-eligible (in-flight) one is also info, + /// never an error. The owner-visible missing-body case is the error above. + const NamespaceListing manifest_debris_listing = store.listNamespaces(namespace_prefix); + recordLifelessKeys(manifest_debris_listing); + for (const String & ns_str : manifest_debris_listing.namespaces) + { + const RootNamespace ns{ns_str}; + const String manifests_prefix = layout.manifestNamespacePrefix(ns); + std::unordered_map manifest_bodies; + listAll(backend, manifests_prefix, manifest_bodies, on_progress, deadline, "listing manifests"); + for (const auto & [mkey, sz] : manifest_bodies) + { + if (owned_manifest_keys.contains(mkey)) + continue; /// owned by a committed ref — accounted above + ++report.unreachable; + if (detail) + { + BuildPrefix prefix; + const bool parsed = parseBuildPrefix(layout, mkey, prefix); + FsckObject o; + o.key = mkey; + o.kind = ObjectKind::Blob; + o.size = sz; + o.cls = FsckClass::Unreachable; + if (parsed && prefixEligible(store, ns, prefix)) + o.reachable_from = {"reclaimable-pre-precommit"}; + else + o.reachable_from = {"in-flight-pre-precommit"}; + report.objects.push_back(std::move(o)); + } + } + } + +} + +} + +FsckReport runFsck(Pool & store, bool detail, FsckProgress on_progress, + std::optional deadline, + bool partial_on_deadline, const String & namespace_prefix) +{ + FsckReport report; + try + { + runFsckImpl(store, detail, on_progress, deadline, namespace_prefix, report); + } + catch (const Exception & e) + { + if (!partial_on_deadline || e.code() != ErrorCodes::TIMEOUT_EXCEEDED) + throw; + report.partial = true; + report.partial_reason = e.message(); + } + return report; +} + +String formatFsckSummary(const FsckReport & report) +{ + /// Field order is load-bearing for humans only; every consumer parses `key=value` tokens. `partial` + /// and its free-text reason go LAST because the reason can contain spaces and quotes, so a parser + /// splitting on whitespace has to trim from the tail (see the harness's `parse_fsck_summary`). + /// `std::ostringstream`, not a ClickHouse write buffer: this reproduces the exact `std::cout` + /// formatting the line has always had, `dedup_ratio`'s default double precision included, so + /// extracting the line from the command changes nothing a parser can observe. + std::ostringstream out; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + out << "reachable=" << report.reachable + << " dangling=" << report.dangling + << " unreachable=" << report.unreachable + << " pending_gc=" << report.pending_gc + << " awaiting_gc=" << report.awaiting_gc + << " unaccounted=" << report.unaccounted + << " stale_edge=" << report.stale_edge + << " corrupted_runs=" << report.corrupted_runs + << " chain_broken=" << report.chain_broken + << " lifeless_keys=" << report.lifeless_keys + << " janitor_pending=" << report.namespace_janitor_pending + << " janitor_pending_bytes=" << report.namespace_janitor_pending_bytes + << " janitor_pending_lives=" << report.namespace_janitor_pending_lives + << " unchecked=" << report.unchecked + << " ref_records_walked=" << report.ref_records_walked + << " physical_bytes=" << report.physical_bytes + << " referenced_logical_bytes=" << report.referenced_logical_bytes + << " distinct_blobs=" << report.distinct_blobs + << " total_blob_refs=" << report.total_blob_refs + << " dedup_ratio=" << report.dedupRatio(); + if (report.partial) + out << " partial=1 reason='" << report.partial_reason << "'"; + return out.str(); +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.h new file mode 100644 index 000000000000..620904f4c37f --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.h @@ -0,0 +1,285 @@ +#pragma once + +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// Optional progress sink for `runFsck`: called periodically during the listing and reachability +/// walk so a long scan over a large/slow pool is visibly progressing (not hung). `phase` names the +/// current step; `objects`/`pages` are running counts. Default {} = no progress (existing callers). +using FsckProgress = std::function; + +/// Classification assigned to each object examined by `runFsck`. +/// +/// The reachability classes are derived only from authoritative refs and the physical object listing. +/// The GC-related classes are an additional explanation for present-but-unreferenced blobs; GC state +/// is used for labeling only and can never make a referenced object appear safe. Integrity classes are +/// hard findings: the report remains unclean when any of them is present. +enum class FsckClass : uint8_t +{ + Reachable, /// reachable from a live ref AND present in the object store + Dangling, /// reachable from a live ref but the object is MISSING — INV-NO-LOSS violation + Unreachable, /// pre-precommit manifest debris (labeled reclaimable / in-flight) + /// The GC pipeline deletes present-but-unreferenced blobs in explicit stages, so these classes + /// distinguish expected in-flight work from an object outside the GC view. They are labels only, + /// never inputs to reachability. + PendingGc, /// listed in the retired set (condemned / delete_pending) — deletion is scheduled; EXPECTED + AwaitingGc, /// edges still in the GC snapshot (drop/reclaim not folded yet) or GC never ran — EXPECTED + Unaccounted, /// absent from the whole GC view — transient for a fast create+drop between rounds; + /// PERSISTENT occurrences should be impossible (INV-2 reachability-before-content) + StaleEdge, /// every source edge the GC snapshot still holds on this blob names a manifest that no + /// longer exists anywhere in the pool, so the matching `-1` can never fold: the blob's + /// in-degree can never reach zero and the incremental GC can never reclaim it. Only a + /// full rebuild of the in-degree state can. ERROR — never an `AwaitingGc` "expected" + /// backlog, which is exactly the label that used to hide it. + CorruptedRun, /// a GC source-edge run's whole-file seal checksum (`RunRef::checksum`) disagrees with + /// the stored bytes — cataloged so the read-only audit enumerates every finding in one + /// pass; deletion-deriving consumers (`fold`, `zeroInDegree`, `previewDeletes`) still + /// fail closed on the same mismatch. ERROR + /// The two verdicts of the arithmetic ref-stream walk (spec §7). They are about a NAMESPACE, not an + /// object; the row's `key` identifies the exact log where the walk stopped or the checkpoint-named + /// snapshot base whose required triple could not be validated. + ChainBroken, /// the exact checkpoint authority is durably inconsistent: its required snapshot-base + /// triple is corrupt, or a ref-log id is absent below its confirmed frontier. Ids are + /// dense `1..T` within `(namespace, epoch)` (INV-1), so neither is a stream end. ERROR + Unchecked, /// the walk could not prove this namespace's stream EITHER WAY (an unprovable epoch + /// crossing, an undecodable body, or unstable authority/transport). Not a finding and + /// not a clean bill of health: the honest third answer, reported so nobody reads a + /// silence as a proof. + LifelessKey, /// a namespace-tree key the `Layout` parsers refuse (a malformed/non-canonical shape, + /// including the un-incarnated Stage A layout), OR a catalog incarnation that is + /// ambiguous or otherwise unreadable. Neither a current writer nor the catalog's own + /// reverse life index can produce this key's meaning, so it belongs to no namespace + /// and no per-namespace verdict can carry it. ERROR + JanitorPending,/// a COMPLETE, canonical namespace-life key (parses via the exact writer grammar, + /// nonzero 32-hex life id) whose life is simply absent from a catalog cut taken AFTER + /// the physical listing. This is the protocol-produced interval between a fenced GC + /// exact-deleting a `Removing` catalog row and the perpetual `NamespaceJanitor` + /// reaching this key on a later bounded page -- inert debris, not damage. Reported + /// as a soft finding: NOT in `kFsckHardFindings`, does not fail the report. +}; + +/// One object or integrity finding emitted in detailed mode, or emitted for every missing reachable +/// object even in summary mode. `key` identifies the physical or logical object; `size` is its listed +/// size and is zero for a missing object. `reachable_from` contains `"namespace/ref"` owners for +/// reachable and dangling objects, or a diagnostic note for other classifications. +struct FsckObject +{ + String key; + ObjectKind kind = ObjectKind::Blob; + uint64_t size = 0; /// on-disk object size (0 when dangling) + FsckClass cls = FsckClass::Reachable; + std::vector reachable_from; /// "ns/ref" labels (populated for reachable/dangling when detail) +}; + +/// Aggregate result of a read-only `runFsck` scan. +/// +/// Reachability and byte counters describe the scan's authoritative-ref view. `unreachable` is the +/// total of all present-but-unreferenced objects, including the GC pipeline classes and manifest debris, +/// and is intentionally retained as one monotone number for residual-settling monitoring. The detailed +/// `objects` list is populated according to the scan's `detail` mode. In partial mode all counters are +/// lower bounds over the portion walked before the deadline; `clean` must not be used as a claim about +/// the unvisited part of the pool. +struct FsckReport +{ + uint64_t reachable = 0; + uint64_t dangling = 0; + /// TOTAL of everything present-but-unreferenced (blob pipeline classes below + manifest debris). + /// Kept as the sum so residual-settling loops (soak) keep one monotone number to watch. + uint64_t unreachable = 0; + uint64_t pending_gc = 0; /// blobs in the retired set — deletion scheduled (expected) + uint64_t awaiting_gc = 0; /// blobs whose drop is not folded yet / GC never ran (expected) + uint64_t unaccounted = 0; /// blobs outside the GC view (transient or anomaly) + /// Blobs whose every remaining source edge names a manifest that no longer exists — permanently + /// stuck at a nonzero in-degree, unreclaimable by the incremental GC. A hard ERROR (see + /// `FsckClass::StaleEdge`). Populated only in `detail` mode: naming the live sources costs one GET + /// per manifest body, and the cheap summary path must stay request-for-request unchanged. + uint64_t stale_edge = 0; + + /// The per-hash `.meta` descriptor sibling of a blob body: + /// pairing check between the `blobs/` physical listing's `.meta` keys and its body keys. + /// ADVISORY, not a hard finding: GC deletes the body FIRST and then drops the `.meta` on a bounded, + /// error-suppressed advisory pool that runs strictly after (and may drop the op — see `CasGc`), so a + /// single raw LIST legitimately observes a body-less `.meta` mid-graduation and NO finite grace makes + /// a persistent one hard evidence. Counted and reported; excluded from `clean()`. + uint64_t meta_without_body = 0; /// a `.meta` object with no body — INV-META-BODY advisory + uint64_t body_without_meta = 0; /// a body with no `.meta` — a not-yet-adopted or interrupted-birth + /// artifact; benign, NOT a dangle + + /// GC source-edge runs whose whole-file seal checksum did not match the stored bytes. Cataloged + /// with the run key in `objects`; the audit CONTINUES — a read-only auditor + /// enumerates all problems in one pass rather than aborting on the first corrupt run. + uint64_t corrupted_runs = 0; + + /// The arithmetic ref-stream walk (spec §7). fsck reads each namespace's stream by EXACT KEY from + /// `_ckpt.checkpoint`'s successor upward — never from a listing, which may omit durable records — + /// and reports one verdict per namespace. + /// + /// `chain_broken` counts namespaces with a proven hole (see `FsckClass::ChainBroken`) and is a HARD + /// ERROR: part of `clean`, and the command exits nonzero on it. `unchecked` counts namespaces the + /// walk could not prove either way; it is COVERAGE, not a finding, so it + /// is reported and printed but does not make a report unclean — exactly like `partial`. A pool with + /// nothing wrong reads `chain_broken=0 unchecked=0`, so `unchecked` is never a resting state. + /// + /// `ref_records_walked` is how many ref-log records the walk actually read and proved, summed over + /// namespaces. It is what makes "the tail above the checkpoint was walked" observable rather than + /// inferred from the absence of a complaint. + uint64_t chain_broken = 0; + uint64_t unchecked = 0; + uint64_t ref_records_walked = 0; + + /// Keys the namespace enumeration could not attribute to any namespace, OR a catalog incarnation + /// that is ambiguous or unreadable (see `FsckClass::LifelessKey` and `Cas::NamespaceListing`). Does + /// NOT include a complete, canonical namespace-life key whose life is simply absent from the catalog + /// -- that is `namespace_janitor_pending`, counted separately and not a hard finding. Counted + /// DISTINCT by key: the scan enumerates namespaces several times and every sweep sees the same + /// offending key, so a per-sweep count would multiply one defect. A hard finding: no current writer + /// can produce this key's meaning, and an audit is where an operator finds out about it. + uint64_t lifeless_keys = 0; + + /// Canonical namespace-life keys whose life is absent from a catalog cut taken AFTER the physical + /// listing (see `FsckClass::JanitorPending`). SOFT: never in `kFsckHardFindings`, never fails the + /// report. Persistent non-convergence across authorized janitor cycles is an operational leak + /// question (`CASGCNamespaceCleanupLeaks`, the `namespace_cleanup` GC-log phase), not an integrity + /// finding this counter can answer on its own -- one snapshot cannot prove an unbounded leak. + uint64_t namespace_janitor_pending = 0; + uint64_t namespace_janitor_pending_bytes = 0; + uint64_t namespace_janitor_pending_lives = 0; /// distinct life ids counted above + + uint64_t physical_bytes = 0; + uint64_t referenced_logical_bytes = 0; + uint64_t total_blob_refs = 0; + uint64_t distinct_blobs = 0; + + /// Set when the scan hit its deadline in partial mode: counts cover only what was walked + /// before the deadline — a lower bound, not the pool truth. + bool partial = false; + String partial_reason; + + std::vector objects; + + /// Return logical blob references per distinct reachable blob, or zero when no distinct blob was seen. + double dedupRatio() const { return distinct_blobs ? double(total_blob_refs) / double(distinct_blobs) : 0.0; } + + /// Return whether the scan found no missing reachable object or hard integrity violation. Expected + /// GC backlog classes do not make a report unclean, and `meta_without_body` is advisory (see its + /// field: GC's body-then-meta delete ordering makes a body-less `.meta` a legitimate transient with + /// no finite hard horizon); a partial report only covers the visited subset. `stale_edge` is a hard + /// finding, but it is only ever nonzero in `detail` mode — a clean summary report says nothing about + /// stale edges, exactly as a partial report says nothing about the unvisited part of the pool. + /// `chain_broken` is a hard finding in every mode. `unchecked` deliberately is NOT one: it says the + /// walk proved nothing about those namespaces, which is a statement about COVERAGE, and folding it + /// in here would make "cannot prove" indistinguishable from "found broken". + /// Defined out-of-line below, over `kFsckHardFindings`, so that "a term of `clean`" and "a row of + /// that list" are the same thing rather than two lists that can drift. + bool clean() const; +}; + +/// ONE hard finding: the name every surface renders it under, and the counter it reads. +struct FsckHardFinding +{ + std::string_view name; + uint64_t FsckReport::* value; +}; + +/// THE HARD FINDINGS, and the single authority on what they are. `FsckReport::clean` is computed from +/// this list, so adding a term means adding a row here. +/// +/// The name is the one the text summary line and the SQL result column both use, which is what lets a +/// test check a rendering surface by iterating this list instead of restating its contents. +/// The SIZE IS DEDUCED, deliberately. A fixed `std::array` rejects an added row with +/// an "excess elements in ..." diagnostic -- which stops the build, but its text carries none of the +/// guidance the assert below does, so the author learns only that they miscounted. (Which noun that +/// diagnostic uses depends on the brace form, so it is not quoted here.) Deduced, an added row compiles +/// and the assert is what speaks. +inline constexpr std::array kFsckHardFindings{ + FsckHardFinding{"dangling", &FsckReport::dangling}, + FsckHardFinding{"corrupted_runs", &FsckReport::corrupted_runs}, + FsckHardFinding{"stale_edge", &FsckReport::stale_edge}, + FsckHardFinding{"chain_broken", &FsckReport::chain_broken}, + FsckHardFinding{"lifeless_keys", &FsckReport::lifeless_keys}, +}; + +/// TRIPWIRE. A hard finding has to reach three CODE surfaces, and each has been forgotten at least once: +/// the text summary line (`formatFsckSummary`), `CommandFsck::executeImpl`'s nonzero-exit set, and the +/// SQL result row (`contentAddressedFsckColumns` + `appendContentAddressedFsckRow`). It has happened +/// repeatedly, on more than one occasion and to more than one term, each time with the rule written down +/// in prose and each time the prose not holding. (No count is given: the records that document those +/// episodes do not support one number, and a tally nobody can reconstruct is the same defect as the rest.) +/// +/// ONE ROW OF THIS LIST IS DELIBERATELY NOT IN THE EXIT SET, so the "three surfaces" rule has a named +/// exception rather than a silent violation: `stale_edge` is nonzero only under `--detail`, and +/// `CommandFsck::executeImpl` prints it as a `note:` and never throws. What licenses that is the pair -- +/// a documented reason AND a compensating gate elsewhere (`stale_edge_verdict` in +/// `utils/ca-soak/soak/fsck.py`, asserted by the soak checkpoint in `soak/run.py`, which fails closed +/// when the key is absent). A new finding may take the same exception only WITH both halves; without +/// them it belongs in the exit set. +/// +/// WHAT THIS ASSERT CHECKS, precisely: that the number of hard findings still equals the number written +/// here. Nothing more. It does NOT check that any surface renders them -- it cannot see the renderers, +/// which is the whole reason it lives with the struct: this header is included by the summary formatter, +/// by `programs/disks/CommandFsck.cpp`, and by `src/Interpreters/InterpreterSystemQuery.cpp`, so changing +/// the list breaks the build in every TU that owes an update, including the two no unit test can reach. +/// +/// The summary line is checked for real, by a test that iterates the list +/// (`CasFsckSummary.EveryHardFindingAppearsOnTheSummaryLine`). The exit set and the SQL row are NOT -- +/// for those, this assert plus the list below it is the whole of the mechanism, so bumping the number +/// without visiting them defeats it. Bump it only after all three are done. +/// +/// AND IT REACHES NO PROSE. The rule is also restated in `docs/superpowers/cas/AGENTS.md` +/// and in the soak harness's comments and messages; those restatements have gone stale before -- more +/// than once, about the exit set -- and nothing here can break a build over them. (No count is given, +/// for the same reason the paragraph above gives none: nobody keeping a tally of restatements can +/// promise its own count will not go stale next.) They are a fourth surface, unfenced by construction. +static_assert(kFsckHardFindings.size() == 5, + "A hard finding was added to or removed from `kFsckHardFindings`, which is `FsckReport::clean`. " + "Before updating this count, render it in ALL THREE code surfaces: `formatFsckSummary`'s line, " + "`CommandFsck::executeImpl`'s nonzero-exit set, and `contentAddressedFsckColumns` + " + "`appendContentAddressedFsckRow`. Two of the three have no test that can fail for you -- the " + "comment above this assert says which. A finding may be left out of the exit set only the way " + "`stale_edge` is: with a documented reason AND a compensating soak assert."); + +inline bool FsckReport::clean() const +{ + for (const FsckHardFinding & finding : kFsckHardFindings) + if (this->*finding.value != 0) + return false; + return true; +} + +/// Independently recompute reachability from authoritative refs (never from GC state or snapshots) and +/// diff it against a raw object listing. The operation is read-only; `detail` populates per-object rows. +/// `deadline`, if set, bounds the WHOLE scan: it is checked between list pages and reachability +/// refs, throwing `TIMEOUT_EXCEEDED` if exceeded (a slow-but-progressing scan surfaces a clear +/// error instead of an opaque hang) — unless `partial_on_deadline` is set, in which case the +/// accumulated lower-bound counts are returned instead, flagged via `FsckReport::partial`. A single +/// LIST page stuck in S3-client retries is bounded separately by the disk's S3 retry/timeout +/// settings, not here. `namespace_prefix`, if non-empty, scopes the scan to namespaces with this +/// prefix and skips the pool-wide unreachable classification (dangling-only mode). +FsckReport runFsck(Pool & store, bool detail, FsckProgress on_progress = {}, + std::optional deadline = {}, + bool partial_on_deadline = false, const String & namespace_prefix = {}); + +/// Render the single machine-parseable summary line (no trailing newline). This is the ONLY view of a +/// report most consumers ever get -- the soak harness parses it, CI greps it, an operator reads it -- so +/// it lives here, next to the report and under test, rather than inline in the command where nothing +/// could reach it. Every term of `FsckReport::clean` MUST appear: a hard finding the line omits is a +/// finding no run will ever report, which is how `corrupted_runs` stayed invisible from the day it was +/// first counted. That requirement is CHECKED for this surface, not merely stated: +/// `CasFsckSummary.EveryHardFindingAppearsOnTheSummaryLine` iterates `kFsckHardFindings` and looks for +/// each name in the line, so a term added to the list and not rendered here fails that test. Zeros are +/// printed, never omitted: "absent" and "zero" are different facts, and consumers (e.g. the harness's +/// `stale_edge_verdict`) fail closed on absence by design. +String formatFsckSummary(const FsckReport & report); + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp new file mode 100644 index 000000000000..1428be282f30 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp @@ -0,0 +1,639 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int CORRUPTED_DATA; +} +} + +namespace DB::Cas +{ + +namespace +{ + +/// Escapes `s` as a JSON string LITERAL (including the surrounding quotes). Handles the standard +/// two-char escapes plus a `\uXXXX` fallback for any other control byte; everything else (including +/// raw multi-byte UTF-8) passes through unchanged. This is a debug/inspection rendering, not a wire +/// format, so it deliberately does not attempt full Unicode validation. +String jsonEscape(std::string_view s) +{ + String out; + out.reserve(s.size() + 2); + out += '"'; + for (unsigned char c : s) + { + switch (c) + { + case '"': out += "\\\""; break; + case '\\': out += "\\\\"; break; + case '\b': out += "\\b"; break; + case '\f': out += "\\f"; break; + case '\n': out += "\\n"; break; + case '\r': out += "\\r"; break; + case '\t': out += "\\t"; break; + default: + if (c < 0x20) + out += fmt::format("\\u{:04x}", c); + else + out += static_cast(c); + } + } + out += '"'; + return out; +} + +/// u128 fields (hashes, ids, tokens-as-u128) render as a lowercase-hex JSON string, matching +/// `u128ToHex` — never as a nested {high,low} object or a decimal number. +String jsonHex(const UInt128 & v) { return jsonEscape(u128ToHex(v)); } +String jsonUInt(uint64_t v) { return std::to_string(v); } +String jsonBool(bool b) { return b ? "true" : "false"; } + +/// A minimal JSON object builder: each `add` takes a key and an already-rendered JSON fragment +/// (a quoted string, a number, `true`/`false`/`null`, or a nested `{...}`/`[...]`) and joins them +/// with commas. No pretty-printing — this is a debug/inspection tool, not a wire format. +class JsonObj +{ +public: + JsonObj & add(std::string_view key, const String & raw_value) + { + if (!first) + out += ","; + first = false; + out += jsonEscape(key); + out += ":"; + out += raw_value; + return *this; + } + + String str() const { return "{" + out + "}"; } + +private: + String out; + bool first = true; +}; + +String jsonArray(const std::vector & items) +{ + String out = "["; + for (size_t i = 0; i < items.size(); ++i) + { + if (i) + out += ","; + out += items[i]; + } + out += "]"; + return out; +} + +String renderManifestRef(const ManifestRef & r) +{ + return JsonObj() + .add("writer_epoch", jsonUInt(r.writer_epoch)) + .add("build_sequence", jsonUInt(r.build_sequence)) + .add("manifest_ordinal", jsonUInt(r.manifest_ordinal)) + .str(); +} + +/// Snapshot and log ref objects use `RefTxnId` values with `writer_epoch` and `ref_sequence` fields. +/// `renderRefTxnIdObj` renders those raw numeric fields rather than the canonical hex form, which +/// rejects a zero field, so inspection can dump any object, including a malformed one, without +/// failing while rendering its identifiers. +String renderRefTxnIdObj(const RefTxnId & id) +{ + return JsonObj() + .add("writer_epoch", jsonUInt(id.writer_epoch)) + .add("ref_sequence", jsonUInt(id.ref_sequence)) + .str(); +} + +String refOwnerKindName(RefOwnerKind k) +{ + switch (k) + { + case RefOwnerKind::Committed: return "Committed"; + case RefOwnerKind::Precommit: return "Precommit"; + } + return "Unknown"; +} + +String renderRefOwnerBinding(const RefOwnerBinding & b) +{ + return JsonObj() + .add("kind", jsonEscape(refOwnerKindName(b.kind))) + .add("ref_name", jsonEscape(b.ref_name)) + .add("manifest_ref", renderManifestRef(b.manifest_ref)) + .str(); +} + +String renderRefCommittedRow(const RefCommittedRow & r) +{ + return JsonObj() + .add("ref_name", jsonEscape(r.ref_name)) + .add("manifest_ref", renderManifestRef(r.manifest_ref)) + .add("published_at_ms", jsonUInt(r.published_at_ms)) + .str(); +} + +String renderRefTableSnapshot(const RefTableSnapshot & s) +{ + std::vector committed; + committed.reserve(s.committed.size()); + for (const auto & row : s.committed) + committed.push_back(renderRefCommittedRow(row)); + + std::vector precommits; + precommits.reserve(s.precommits.size()); + for (const auto & b : s.precommits) + precommits.push_back(renderRefOwnerBinding(b)); + + return JsonObj() + .add("object", jsonEscape("ref_snapshot")) + .add("namespace", jsonEscape(s.ns)) + .add("snapshot_id", renderRefTxnIdObj(s.snapshot_id)) + .add("committed", jsonArray(committed)) + .add("precommits", jsonArray(precommits)) + .str(); +} + +/// The namespace's checkpoint (spec INV-4). Every field is optional and each absence means something +/// different an operator needs to see: no `life_epoch` means no writer that knew this namespace's +/// genesis epoch has written here yet, no `committed_through` means the life has no committed +/// transaction, no `checkpoint_snapshot_id` means recovery has no snapshot base, and no +/// `last_epoch_seal` means no epoch of this namespace has been closed. They are rendered as explicit +/// `null`s rather than omitted keys so all four cases are visible. +/// `ns` comes from the KEY -- unlike the log and snapshot objects, a `_ckpt` body does not name its +/// namespace, so there is no key-to-body binding to cross-check here. +String renderRefCkpt(const RootNamespace & ns, const RefCkpt & c) +{ + return JsonObj() + .add("object", jsonEscape("ref_ckpt")) + .add("namespace", jsonEscape(ns.string())) + .add("life_epoch", c.life_epoch ? jsonUInt(*c.life_epoch) : "null") + .add("committed_through", c.committed_through ? renderRefTxnIdObj(*c.committed_through) : "null") + .add("checkpoint_snapshot_id", + c.checkpoint_snapshot_id ? renderRefTxnIdObj(*c.checkpoint_snapshot_id) : "null") + .add("last_epoch_seal", c.last_epoch_seal ? renderRefTxnIdObj(*c.last_epoch_seal) : "null") + .str(); +} + +String refOpKindName(RefOpKind k) +{ + switch (k) + { + case RefOpKind::NamespaceBirth: return "NamespaceBirth"; + case RefOpKind::OwnerTransition: return "OwnerTransition"; + case RefOpKind::SetPublishedAt: return "SetPublishedAt"; + case RefOpKind::RemoveNamespace: return "RemoveNamespace"; + case RefOpKind::EpochSeal: return "EpochSeal"; + } + return "Unknown"; +} + +String renderRefOp(const RefOp & op) +{ + return JsonObj() + .add("kind", jsonEscape(refOpKindName(op.kind))) + .add("old_binding", op.old_binding ? renderRefOwnerBinding(*op.old_binding) : "null") + .add("new_binding", op.new_binding ? renderRefOwnerBinding(*op.new_binding) : "null") + .add("ref_name", jsonEscape(op.ref_name)) + .add("expected_manifest_ref", renderManifestRef(op.expected_manifest_ref)) + .add("published_at_ms", jsonUInt(op.published_at_ms)) + .str(); +} + +String renderRefLogTxn(const RefLogTxn & t) +{ + std::vector ops; + ops.reserve(t.ops.size()); + for (const auto & op : t.ops) + ops.push_back(renderRefOp(op)); + + return JsonObj() + .add("object", jsonEscape("ref_log")) + .add("namespace", jsonEscape(t.ns)) + .add("txn_id", renderRefTxnIdObj(t.txn_id)) + .add("ops", jsonArray(ops)) + .add("prev_epoch_seal", t.prev_epoch_seal ? renderRefTxnIdObj(*t.prev_epoch_seal) : "null") + .str(); +} + +String placementName(EntryPlacement p) +{ + switch (p) + { + case EntryPlacement::Inline: return "Inline"; + case EntryPlacement::Blob: return "Blob"; + } + return "Unknown"; +} + +/// `inline_bytes` renders as its LENGTH only, not its content — an inline file's bytes are payload +/// data, not part-manifest identity, and may be arbitrarily large / non-UTF8. +String renderManifestEntry(const ManifestEntry & e) +{ + /// Render `blobIdOf(e.ref)` (":"). The algorithm must remain part of the + /// rendered identity: a bare digest is ambiguous in a pool containing algorithms with different + /// digest widths, and each entry's own `ref.algo` determines its width. + return JsonObj() + .add("path", jsonEscape(e.path)) + .add("placement", jsonEscape(placementName(e.placement))) + .add("blob", jsonEscape(blobIdOf(e.ref))) + .add("blob_size", jsonUInt(e.blob_size)) + .add("inline_bytes_size", jsonUInt(e.inline_bytes.size())) + .str(); +} + +String renderPartManifest(const PartManifest & m) +{ + std::vector entries; + entries.reserve(m.entries.size()); + for (const auto & e : m.entries) + entries.push_back(renderManifestEntry(e)); + + return JsonObj() + .add("ref", renderManifestRef(m.ref)) + .add("root_namespace_id", jsonEscape(m.root_namespace_id.string())) + .add("payload_digest", jsonHex(m.payload_digest)) + .add("entries", jsonArray(entries)) + .str(); +} + +String renderMountLease(const MountLease & m) +{ + return JsonObj() + .add("server_uuid", jsonHex(m.server_uuid)) + .add("writer_epoch", jsonUInt(m.writer_epoch)) + .add("hostname", jsonEscape(m.hostname)) + .add("pid", jsonUInt(m.pid)) + .add("started_at_ms", jsonUInt(m.started_at_ms)) + .add("seq", jsonUInt(m.seq)) + .add("expires_at_ms", jsonUInt(m.expires_at_ms)) + .add("min_active", jsonUInt(m.min_active)) + .add("gc_fenced", jsonBool(m.gc_fenced)) + .add("write_attempt_id", jsonHex(m.write_attempt_id)) + .str(); +} + +String renderGcLease(const GcLease & l) +{ + return JsonObj() + .add("owner", jsonHex(l.owner)) + .add("seq", jsonUInt(l.seq)) + .str(); +} + +String renderGcState(const GcState & s) +{ + return JsonObj() + .add("round", jsonUInt(s.round)) + .add("gc_shards", jsonUInt(s.gc_shards)) + .add("snap_generation", jsonUInt(s.snap_generation)) + .add("snap_pruned_through", jsonUInt(s.snap_pruned_through)) + .add("snap_attempt", jsonUInt(s.snap_attempt)) + .add("manifest_sweep_cursor", jsonEscape(s.manifest_sweep_cursor)) + .add("lease", renderGcLease(s.lease)) + .str(); +} + +String tokenTypeName(TokenType t) +{ + switch (t) + { + case TokenType::ETag: return "ETag"; + case TokenType::Generation: return "Generation"; + case TokenType::Emulated: return "Emulated"; + } + return "Unknown"; +} + +/// `Token::value` is an opaque backend-native string (e.g. an S3 ETag) — NOT a 128-bit hash — so it +/// renders verbatim (escaped), not hex-converted; `type` names which backend family minted it. +String renderToken(const Token & t) +{ + return JsonObj() + .add("value", jsonEscape(t.value)) + .add("type", jsonEscape(tokenTypeName(t.type))) + .str(); +} + +String objectKindName(ObjectKind k) +{ + switch (k) + { + case ObjectKind::Blob: return "Blob"; + } + return "Unknown"; +} + +String renderRunRef(const RunRef & r) +{ + return JsonObj() + .add("key", jsonEscape(r.key)) + .add("checksum", jsonHex(r.checksum)) + .add("shard", jsonUInt(r.shard)) + .add("generation", jsonUInt(r.generation)) + .str(); +} + +String renderRefCoverage(const RefCoverage & c) +{ + return JsonObj() + .add("classification", jsonUInt(c.classification)) + .add("last_folded_ref_id", renderRefTxnIdObj(c.last_folded_ref_id)) + .str(); +} + +String renderFoldSeal(const CasFoldSeal & seal) +{ + JsonObj ref_lives; + for (const auto & [life_id, state] : seal.ref_lives) + ref_lives.add(renderIncarnation(life_id), JsonObj() + .add("coverage", renderRefCoverage(state.coverage)) + .add("cleanup_evidence", state.cleanup_evidence + ? JsonObj().add("remove_txn_id", renderRefTxnIdObj(state.cleanup_evidence->remove_txn_id)).str() + : "null") + .str()); + + std::vector blob_target_runs; + blob_target_runs.reserve(seal.blob_target_runs.size()); + for (const auto & r : seal.blob_target_runs) + blob_target_runs.push_back(renderRunRef(r)); + + /// A fold seal carries per-GC-shard totals for `kCondemned` rows in its source runs. Render the + /// summary from the seal itself; the older separate retired-reference object is no longer part + /// of the current layout. + JsonObj condemned_summary; + for (const auto & [shard, cs] : seal.condemned_summary) + condemned_summary.add(std::to_string(shard), JsonObj() + .add("condemned_total", jsonUInt(cs.condemned_total)) + .add("pending_total", jsonUInt(cs.pending_total)) + .add("oldest_nonpending_condemn_round", jsonUInt(cs.oldest_nonpending_condemn_round)) + .str()); + + return JsonObj() + .add("generation", jsonUInt(seal.generation)) + .add("parent_generation", jsonUInt(seal.parent_generation)) + .add("ref_lives", ref_lives.str()) + .add("blob_target_runs", jsonArray(blob_target_runs)) + .add("condemned_summary", condemned_summary.str()) + .str(); +} + +String provenanceOpName(ProvenanceOp op) +{ + switch (op) + { + case ProvenanceOp::Other: return "Other"; + case ProvenanceOp::Insert: return "Insert"; + case ProvenanceOp::Merge: return "Merge"; + case ProvenanceOp::Mutation: return "Mutation"; + case ProvenanceOp::Attach: return "Attach"; + case ProvenanceOp::Repack: return "Repack"; + } + return "Unknown"; +} + +String renderProvenance(const Provenance & p) +{ + return JsonObj() + .add("created_at_ms", jsonUInt(p.created_at_ms)) + .add("creator_server_id", jsonHex(p.creator_server_id)) + .add("ch_version", jsonUInt(p.ch_version)) + .add("op", jsonEscape(provenanceOpName(p.op))) + .str(); +} + +String metaStateName(MetaState s) +{ + switch (s) + { + case MetaState::Clean: return "clean"; + case MetaState::Condemned: return "condemned"; + } + return "unknown"; +} + +/// The per-hash `.meta` descriptor is the blob body's sibling and records its freshness state +/// (`Clean` or `Condemned`), not its payload. It is rendered separately from `renderEnvelopeHeader`: +/// the body remains an enveloped object, while the descriptor has its own format. +String renderBlobMeta(const BlobMeta & m) +{ + return JsonObj() + .add("object", jsonEscape("blob_meta")) + .add("version", jsonUInt(m.version)) + .add("state", jsonEscape(metaStateName(m.state))) + .add("condemn_round", jsonUInt(m.condemn_round)) + .add("size", jsonUInt(m.size)) + .str(); +} + +String renderEnvelopeHeader(const EnvelopeHeader & h) +{ + return JsonObj() + .add("kind", jsonEscape(objectKindName(h.kind))) + /// The blob identity is carried by the object key, so the envelope keeps only the provenance + /// fields needed for forensics (`ch` and `bld`) together with its compatibility version. + .add("compatibility_version", jsonUInt(h.compatibility_version)) + .add("incarnation_tag", jsonHex(h.incarnation_tag)) + .add("build_id", jsonHex(h.build_id)) + .add("header_len", jsonUInt(h.header_len)) + .add("provenance", h.provenance ? renderProvenance(*h.provenance) : "null") + .add("intended_ref", h.intended_ref ? jsonEscape(*h.intended_ref) : "null") + .str(); +} + +/// The word vocabulary a row's marker byte renders as, matching the `cas_run` NDJSON's own `"m"` field +/// words (`CasRecordStreamFormat.cpp`'s private `markerToWord`) so cas-inspect speaks the same vocabulary +/// as the on-disk format rather than inventing a second one. +String sourceEdgeRowKindName(char marker) +{ + switch (marker) + { + case kEdgeActive: return "edge"; + case kZeroMarker: return "zero"; + case kCondemned: return "condemned"; + default: return "unknown"; + } +} + +String renderCondemnedRow(const CondemnedRow & r) +{ + return JsonObj() + .add("delete_pending", jsonBool(r.delete_pending)) + .add("token", renderToken(r.token)) + .add("size", jsonUInt(r.size)) + .add("condemn_round", jsonUInt(r.condemn_round)) + .add("marker_confirmed", jsonBool(r.marker_confirmed)) + .str(); +} + +/// Renders one blob-target source-edge run segment (`Layout::blobTargetRunKey`): every row (edge, +/// zero-marker, or condemned sentinel), plus a summary. `parsed` carries the run's own coordinates +/// recovered from the key; `bytes` is decoded with the same typed `SourceEdgeRunView` reader the fold / +/// `zeroInDegree` / `fsck` consumers use (the memory overload, since `caInspectToJson` is a pure +/// function of (key, bytes) with no backend access here). A malformed key or payload propagates the +/// codec's own `CORRUPTED_DATA` (`SourceEdgeKeyCodec::parse`, `decodeCondemnedRow`) -- rows are never +/// silently skipped. +String renderBlobTargetRun(const ParsedBlobTargetRunKey & parsed, std::string_view bytes) +{ + SourceEdgeRunView reader = openSourceEdgeRun(bytes); + + std::vector rows; + std::set distinct_blobs; + uint64_t edge_count = 0; + uint64_t condemned_count = 0; + uint64_t zero_marker_count = 0; + + String key; + String payload; + while (reader.next(key, payload)) + { + BlobRef ref; + UInt128 source_id; + SourceEdgeKeyCodec::parse(key, ref, source_id); // throws CORRUPTED_DATA on a malformed key (fail-closed) + if (payload.empty()) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "cas-inspect: source-edge run row for blob {} has an empty payload", blobIdOf(ref)); + const char marker = payload[0]; + + distinct_blobs.insert(ref); + JsonObj row; + row.add("blob", jsonEscape(blobIdOf(ref))) + /// `source_id` is a `CityHash128` of (namespace, writer_epoch, build_sequence, + /// manifest_ordinal, path) -- not invertible here, so it renders as plain hex, exactly like + /// every other opaque u128 identifier in this file. + .add("source_id", jsonHex(source_id)) + .add("kind", jsonEscape(sourceEdgeRowKindName(marker))); + + switch (marker) + { + case kEdgeActive: + ++edge_count; + break; + case kZeroMarker: + ++zero_marker_count; + break; + case kCondemned: + ++condemned_count; + row.add("condemned", renderCondemnedRow(decodeCondemnedRow(payload))); // CORRUPTED_DATA on malformed (fail-closed) + break; + default: + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "cas-inspect: source-edge run row for blob {} has an unknown marker 0x{:02x}", + blobIdOf(ref), static_cast(marker)); + } + rows.push_back(row.str()); + } + + return JsonObj() + .add("object", jsonEscape("blob_target_run")) + .add("generation", jsonUInt(parsed.generation)) + .add("attempt", jsonUInt(parsed.attempt)) + .add("shard", jsonUInt(parsed.shard)) + .add("seq", jsonUInt(parsed.seq)) + .add("rows", jsonArray(rows)) + .add("summary", JsonObj() + .add("rows", jsonUInt(rows.size())) + .add("distinct_blobs", jsonUInt(distinct_blobs.size())) + .add("edges", jsonUInt(edge_count)) + .add("condemned", jsonUInt(condemned_count)) + .add("zero_markers", jsonUInt(zero_marker_count)) + .str()) + .str(); +} + +} + +String caInspectToJson(const Layout & layout, const String & key, std::string_view bytes, + const std::optional & resolved_life) +{ + /// Most-specific first: `cas/manifests/.../NNNNNN.zst` before the pool-wide `cas/ns/stream/` + /// prefix, the `/mount` and `/fold_seal` suffixes before the pool-wide `gc/state` exact match, + /// and the `.meta` sibling suffix before the bare `blobs/` prefix it also matches. + if (key.starts_with(layout.casManifestsPrefix()) && key.ends_with(storedSuffix(FormatId::PartManifest))) + return renderPartManifest(decodePartManifest(openObject(FormatId::PartManifest, bytes))); + + const auto requireResolvedLife = [&](NamespaceLifePhysicalId life_id) -> const NamespaceLifeId & + { + if (!resolved_life || resolved_life->incarnation != life_id) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "cas-inspect: life_id {} has no unique resolution in the supplied catalog cut", + renderIncarnation(life_id)); + return *resolved_life; + }; + + if (key.starts_with(layout.namespaceStateRootPrefix())) + { + if (const auto life_id = layout.parseRefCkptKey(key)) + return renderRefCkpt(requireResolvedLife(*life_id).ns, decodeRefCkpt(bytes)); + } + + if (key.starts_with(layout.casRefsPrefix())) + { + + const auto parsed = layout.parseRefObjectKey(key); + if (!parsed) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "cas-inspect: key under cas/ns/stream is not a recognized ref-object key '{}'", key); + const NamespaceLifeId & life = requireResolvedLife(parsed->life_id); + if (parsed->kind == RefObjectKind::Snap) + return renderRefTableSnapshot(decodeRefTableSnapshot( + openObject(FormatId::RefSnapshot, bytes), life.ns.string(), parsed->txn_id)); + if (parsed->kind == RefObjectKind::Log) + return renderRefLogTxn(decodeRefLogTxn( + openObject(FormatId::RefLog, bytes), life.ns.string(), parsed->txn_id)); + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "cas-inspect: unhandled ref-object kind for key '{}'", key); + } + + if (key == layout.gcStateKey()) + return renderGcState(decodeGcState(bytes)); + + if (key.ends_with("/mount")) + return renderMountLease(decodeMountLease(bytes)); + + if (key.ends_with("/fold_seal")) + return renderFoldSeal(decodeFoldSeal(bytes)); + + /// Blob-target source-edge run segments (`Layout::blobTargetRunKey`) are the ground truth for + /// every in-degree question, so they get a typed decode too, not just the fold seal that names + /// them. Checked before the pool-wide `blobs/` prefix below (disjoint anyway -- these keys live + /// under `gc/gen/`, never `blobs/` -- but most-specific-first stays the dispatch's rule). + if (const auto parsed = layout.parseBlobTargetRunKey(key)) + return renderBlobTargetRun(*parsed, bytes); + + /// `blobMetaKey(id) == blobKey(id) + ".meta"`, so a meta descriptor also matches + /// `blobsPrefix()` below. Check it first or it would be decoded incorrectly as an envelope. A + /// non-`.meta` blob body still carries its envelope. + if (key.starts_with(layout.blobsPrefix()) && key.ends_with(".meta")) + return renderBlobMeta(decodeBlobMeta(bytes)); + + if (key.starts_with(layout.blobsPrefix())) + return renderEnvelopeHeader(decodeEnvelopeHeader(bytes, bytes.size(), ObjectKind::Blob)); + + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "cas-inspect: unrecognized key layout '{}' (recognized: cas/ns/stream, cas/ns/state, cas/manifests, " + "gc/server-roots/*/mount, gc/state, gc/gen/*/fold_seal, gc/gen/*/attempt/*/blob_target/*/*, " + "retired, blobs, blobs/*.meta)", key); +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h new file mode 100644 index 000000000000..0c6bfa3e0cee --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h @@ -0,0 +1,30 @@ +#pragma once +#include +#include +#include + +namespace DB::Cas +{ + +/// Read-only decode-to-JSON dispatch for `clickhouse-disks cas-inspect` (and its unit tests): given +/// any key that could live in a content-addressed pool plus the raw bytes stored at it, decode with +/// the matching codec and render the struct's fields as human-readable JSON. `layout` supplies the +/// pool's key shapes (there is no live pool/backend access here — pure function of (key, bytes)), so +/// it can be exercised directly against encoder output in unit tests, with no disk / object storage +/// involved. +/// +/// Dispatch is by KEY SHAPE, most-specific first (`cas/manifests/.../NNNNNN.zst` before the +/// `cas/ns/stream/` and `cas/ns/state/` roots, `/mount` and `/fold_seal` suffixes, the +/// `gc/gen/*/attempt/*/blob_target/*/*` source-edge run segments, then the pool-wide `gc/state` +/// and `blobs/` prefix). u128 and hash fields render as lowercase hex strings (matching +/// `u128ToHex`), while backend-native `Token` values render as escaped strings. Neither is exposed +/// as an array of bytes or a raw struct dump. +/// +/// Throws `ErrorCodes::BAD_ARGUMENTS` when `key` matches none of the recognized CA layouts. Any +/// decode failure of a matched key (invalid header, corrupted bytes, future format version, ...) +/// propagates as-is from the underlying `decode*` function (typically `CORRUPTED_DATA` or +/// `UNKNOWN_FORMAT_VERSION`) — this function performs no fallback decode and swallows nothing. +String caInspectToJson(const Layout & layout, const String & key, std::string_view bytes, + const std::optional & resolved_life = std::nullopt); + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/CMakeLists.txt b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/CMakeLists.txt new file mode 100644 index 000000000000..0f792624cea1 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/CMakeLists.txt @@ -0,0 +1,4 @@ +clickhouse_add_executable(benchmark_cas_ref_protocol benchmark_cas_ref_protocol.cpp) +target_link_libraries (benchmark_cas_ref_protocol PRIVATE + ch_contrib::gbenchmark_all + dbms) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp new file mode 100644 index 000000000000..f23f4dc06bae --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp @@ -0,0 +1,553 @@ +#include + +#include +#include + +#include +#include +#include +#include +#include +#include + +/// Pure measurement, no pass/fail assertions -- see the cas-gc-rebuild BACKLOG.md entries +/// "OPTIMIZATION OPPORTUNITY -- ref-ledger JSON encoding writes byte-by-byte" and the (now +/// RESOLVED) "admits() re-encodes the WHOLE ref table once per state-growing op" entry for the +/// investigation these benchmarks measure. Build with `-DENABLE_BENCHMARKS=ON` and run the +/// resulting `benchmark_cas_ref_protocol` binary directly; never wired into `ninja test` +/// or CI. +/// +/// BM_Admits history (synthetic RefTableState, time/call, this binary): +/// Before incremental admits() (2026-07-19) -- full O(N) rebuild+encode per call: +/// N=100: 48.8 us N=1,000: 476 us N=10,000: 5,018 us N=100,000: 55,976 us +/// Google Benchmark complexity fit: O(N log N), RMS 2%. +/// After incremental admits() (2026-07-20) -- O(1) via incremental body-byte counters on +/// RefTableState: +/// N=100: 1842 ns N=1,000: 1875 ns N=10,000: 1864 ns N=100,000: 1919 ns +/// Google Benchmark complexity fit: O(1), RMS 1-2%. +/// +/// BM_EncodeRefLogTxn history (this binary; acceptance gate for the CasJsonWriter migration): +/// Before CasJsonWriter, field-by-field WriteBuffer calls (baseline): 753 ns. +/// After CasJsonWriter bulk-append migration (2026-07-20): 333 ns -- this is the shipped code. +/// BM_MemcpyTxnBytes floor (same bytes, plain String appends of 16-byte fragments): 30.7 ns. +/// Ratio EncodeRefLogTxn / MemcpyTxnBytes = 333 / 30.7 ~= 10.8x -- above the 3x acceptance gate. +/// A `keyLiteral` "rung-1" contingency variant (merging separator+key text into one literal +/// append for the fixed unprefixed keys in writeOp/writeCommittedRow) was also measured: 325 ns +/// ~= 10.8x -- a negligible ~2.5% move, not worth a third key-rendering path. It was NOT shipped; +/// writeOp/writeCommittedRow keep the single `writeKey` path for clarity. Per the contingency +/// ladder, rung 2 was NOT attempted either (it trades readability and needs a human decision); +/// reported as DONE_WITH_CONCERNS. CasEncodingPins.* stayed byte-identical (green) throughout. +/// +/// Phase B baselines, 2026-07-21, pre-encapsulation (this binary; `--benchmark_repetitions=3 +/// --benchmark_report_aggregates_only=true`; medians reported). Recorded ahead of the +/// `RefTableState` encapsulation refactor so later phases can re-run this exact suite unchanged and +/// diff against these numbers. +/// BM_Admits (promote op; stays O(1) via the incremental budget counters, untouched by this round): +/// N=100: 963 ns N=1,000: 979 ns N=10,000: 988 ns N=100,000: 1,029 ns +/// Complexity fit: O(1), RMS 2%. +/// BM_AdmitsAddPrecommit (add op -- THE production hotspot shape: `manifestAlreadyOwned`'s linear +/// value scan AT THIS BASELINE; O(1) via the owned-manifest index since E2 -- see the Final block +/// below): +/// N=100: 995 ns N=1,000: 4,266 ns N=10,000: 38,771 ns N=100,000: 400,222 ns +/// Complexity fit: O(N), ~4.0 ns/row, RMS 2%. +/// BM_ApplyRefLogTxn (scratch copy + validate + apply + install of one promote): +/// N=100: 724 ns N=1,000: 738 ns N=10,000: 784 ns N=100,000: 788 ns +/// Complexity fit: O(1), RMS 4%. +/// BM_ReplayHistory (fold/recovery profile: snapshot of size N, 256 tail txns, 2 ops each): +/// N=100: 6.15 ms N=1,000: 46.1 ms N=10,000: 454.0 ms N=100,000: 4.93 s +/// Complexity fit: O(N), ~48,859 ns/row, RMS 3%. +/// BM_ScratchCopy (one full RefTableState copy off a materialized state -- the isolation floor): +/// N=100: 45.7 ns N=1,000: 46.0 ns N=10,000: 46.7 ns N=100,000: 46.8 ns +/// Complexity fit: O(1), RMS 1%. +/// BM_SnapshotEncode (encodeRefTableSnapshot(snapshotOf(state))): +/// N=100: 14,955 ns N=1,000: 150,061 ns N=10,000: 1,508,586 ns N=100,000: 15,885,841 ns +/// Complexity fit: O(N), ~159 ns/row, RMS 1%. +/// BM_MergedIteration (full base + 10%-overlay merged iteration, post-copy pre-materialize shape): +/// N=100: 759 ns N=1,000: 7,719 ns N=10,000: 81,073 ns N=100,000: 864,552 ns +/// Complexity fit: O(N), ~8.6 ns/row, RMS 4%. +/// BM_Materialize (RefCowMap::materialize after one overlay insert on an N-row base): +/// N=100: 12,069 ns N=1,000: 126,687 ns N=10,000: 1,296,326 ns N=100,000: 18,145,559 ns +/// Complexity fit: O(N log N), RMS 2%. +/// +/// Final, 2026-07-21, shipped tree (post E1+E2+E3; E4 tried and REVERTED -- full per-phase tables in +/// `bench_t5_e3.log`): +/// BM_AdmitsAddPrecommit: ~692-714 ns FLAT across N=100..100,000 -- O(1), RMS 1% +/// (the owned-manifest index replaced the linear scan; ~571x at N=100k). +/// BM_ReplayHistory: 1,725.58 ns/row (was 48,859) -- in-place `TrustedReplay` apply, -96.5%. +/// BM_ApplyRefLogTxn: ~778-822 ns O(1). BM_Admits (promote): ~996-1,056 ns O(1). +/// BM_ScratchCopy: ~58 ns O(1) (+~11 ns vs baseline: one more shared_ptr copy for the index). +/// BM_SnapshotEncode / BM_MergedIteration / BM_Materialize: unchanged from baseline (E4 reverted). +/// +/// Implementation note for later phases: `makeSyntheticState` calls `RefCowMap::materialize()` +/// after `replay` (which never does -- it is the pure state-machine equation, and +/// `stateFromSnapshot` loads every row through `emplace`, which only ever touches the overlay). +/// Skipping that call makes every `RefTableState` copy in this suite (including `admits`'s and +/// `applyRefLogTxn`'s own internal scratch copies) an O(N) deep-copy of an un-materialized overlay +/// map instead of an O(1) shared-base copy -- this was caught during this round because it made +/// BM_Admits regress from the documented O(1) to visibly O(N log N), contradicting its own history +/// above. Production's RETAINED states are all materialized before reuse (the live table materializes +/// once per flush; post-consult the recovery-install site in CasRefLedger.cpp materializes the +/// replayed state before retaining it -- it previously did not, which is the recovery-latency cliff +/// BM_FlushInstall now measures against), so the fix was to materialize in the helper, not to accept +/// the contaminated numbers. (replay's own internal per-txn states are never materialized mid-fold; +/// BM_ReplayHistory models that path on purpose.) + +using namespace DB::Cas; + +namespace +{ + +/// A ref-ledger key shape as actually written on the wire: table_uuid + database + table + part_name. +constexpr std::string_view kSafeKeyLikeString + = "eeeb74a2-606a-4ee9-840a-1aac7b5ac25b_ca_stress_default_part_20260719_0_89811_538"; + +RefLogTxn makeSamplePromoteTxn() +{ + RefLogTxn txn; + txn.ns = "roots/ca_soak_ch1"; + txn.txn_id = RefTxnId{1, 12345}; + + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "20260719_0_89811_538_89818", ManifestRef{1, 1, 999999}}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "20260719_0_89811_538_89818", ManifestRef{1, 1, 999999}}; + txn.ops.push_back(op); + return txn; +} + +/// A synthetic snapshot of `n` committed rows plus one pending precommit ready to promote. +/// Built as a RefTableSnapshot and materialized via the public `replay` entry point, so this +/// helper keeps compiling unchanged when RefTableState's fields become private (Phase A). +RefTableSnapshot makeSyntheticSnapshot(size_t n) +{ + RefTableSnapshot snapshot; + snapshot.ns = "roots/bench"; + snapshot.snapshot_id = RefTxnId{1, 1}; + for (size_t i = 0; i < n; ++i) + { + RefCommittedRow row; + row.ref_name = "part_" + std::to_string(i) + "_20260719_0_1000_1"; + row.manifest_ref = ManifestRef{1, 1, static_cast(i + 1)}; + snapshot.committed.push_back(row); + } + std::sort(snapshot.committed.begin(), snapshot.committed.end(), + [](const auto & a, const auto & b) { return a.ref_name < b.ref_name; }); + snapshot.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "new_part_x", ManifestRef{1, 1, 999999}}); + return snapshot; +} + +/// A synthetic committed-ref table of `n` rows, plus one pending precommit ready to promote -- +/// exactly the shape `admits()` previews on every state-growing ref op. Rebuilt through `replay` +/// (the public state-machine entry point) rather than by poking `RefTableState` fields directly, +/// so this helper survives Phase A's encapsulation of `RefTableState`. +/// +/// `replay` (the pure state-machine equation) never materializes: `stateFromSnapshot` loads every +/// committed row through `RefCowMap::emplace`, which only ever touches the overlay. Left alone, +/// every subsequent `RefTableState` copy here (`admits`'s and `applyRefLogTxn`'s own internal +/// scratch copies, and every benchmark's own scratch copy below) would deep-copy an N-row overlay +/// map instead of sharing an immutable base pointer -- silently turning "the cost of the operation +/// under test" into "the cost of copying an un-materialized map" and swamping the O(1) `admits` +/// result the header history documents. The RETAINED long-lived states production keeps are all +/// materialized: the writer's live table materializes once per flush, and -- post-consult -- the +/// recovery-install site in `CasRefLedger.cpp` now calls `materializeCommitted()` on the replayed +/// state before retaining it (it previously did NOT, so the first flush copied an N-row overlay -- +/// exactly the cliff this fix removed and the reason `BM_FlushInstall` below measures the fully +/// materialized flush cost). So this helper materializes too, matching what every real caller does +/// immediately after building or replaying a state it will keep. (Note that `replay`'s own INTERNAL +/// per-transaction states are never materialized mid-fold -- `BM_ReplayHistory` deliberately models +/// that, feeding `replay(snapshot, tail)` an un-materialized base on purpose.) +RefTableState makeSyntheticState(size_t n) +{ + RefTableState state = replay(makeSyntheticSnapshot(n), {}); + state.materializeCommitted(); + return state; +} + +} + +/// Floor comparison: writeJSONString's per-character escaping loop (WriteHelpers.h) on a string +/// that needs no escaping at all (a real ref-ledger key shape) vs a raw bulk write of the same +/// bytes. See BM_RawBulkWriteSafe below for the delta. +static void BM_WriteJSONStringSafe(benchmark::State & state) +{ + DB::FormatSettings settings; + DB::PODArray buf; + for (auto _ : state) + { + buf.clear(); + DB::WriteBufferFromVector> out(buf); + DB::writeJSONString(kSafeKeyLikeString, out, settings); + benchmark::DoNotOptimize(buf.data()); + } +} +BENCHMARK(BM_WriteJSONStringSafe); + +static void BM_RawBulkWriteSafe(benchmark::State & state) +{ + DB::PODArray buf; + for (auto _ : state) + { + buf.clear(); + DB::WriteBufferFromVector> out(buf); + DB::writeChar('"', out); + out.write(kSafeKeyLikeString.data(), kSafeKeyLikeString.size()); + DB::writeChar('"', out); + benchmark::DoNotOptimize(buf.data()); + } +} +BENCHMARK(BM_RawBulkWriteSafe); + +/// Absolute cost of encoding one ref-log transaction (a single promote op) with +/// `encodeRefLogTxn`'s migrated `CasJsonWriter` bulk-append implementation (see the history +/// comment at the top of this file and the BACKLOG resolution). `BM_MemcpyTxnBytes` right below +/// is the floor to diff this against. +static void BM_EncodeRefLogTxn(benchmark::State & state) +{ + const RefLogTxn txn = makeSamplePromoteTxn(); + for (auto _ : state) + benchmark::DoNotOptimize(encodeRefLogTxn(txn)); +} +BENCHMARK(BM_EncodeRefLogTxn); + +/// The "near-memcpy" floor for BM_EncodeRefLogTxn: the SAME encoded bytes assembled from +/// precomputed 16-byte fragments by plain String appends -- approximating the writer's append +/// granularity with zero formatting/escaping work. Originally an acceptance gate for the +/// CasJsonWriter migration; measurement showed the <=3x-of-floor target is physically unreachable for a validating, +/// JSON-escaping encoder (BM_EncodeRefLogTxn lands at ~10.8x this floor even after the 2.26x +/// CasJsonWriter speedup -- see the BACKLOG resolution for the profiled breakdown). Kept as a +/// documented reference floor, not a pass/fail gate. +static void BM_MemcpyTxnBytes(benchmark::State & state) +{ + const RefLogTxn txn = makeSamplePromoteTxn(); + const String encoded = encodeRefLogTxn(txn); + std::vector fragments; + constexpr size_t kFragment = 16; + for (size_t off = 0; off < encoded.size(); off += kFragment) + fragments.push_back(std::string_view(encoded).substr(off, kFragment)); + + String buf; + buf.reserve(encoded.size()); + for (auto _ : state) + { + buf.clear(); + for (const auto f : fragments) + buf.append(f.data(), f.size()); + benchmark::DoNotOptimize(buf.data()); + } +} +BENCHMARK(BM_MemcpyTxnBytes); + +/// admits() used to re-derive and re-encode the WHOLE committed-ref snapshot on every call +/// (CasRefProtocol.cpp), showing O(N log N) growth with table size; it now maintains +/// incremental body-byte counters on RefTableState instead, so this should show flat (O(1)) +/// time/call across the range. ->Complexity() has Google Benchmark fit and print the +/// empirical big-O across the range. +static void BM_Admits(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableState table = makeSyntheticState(n); + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "new_part_x", ManifestRef{1, 1, 999999}}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "new_part_x", ManifestRef{1, 1, 999999}}; + + for (auto _ : state) + benchmark::DoNotOptimize(admits(table, op, 1ull << 40, 1ull << 40)); + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_Admits)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// THE production hotspot shape: add-precommit runs `manifestAlreadyOwned` (a linear value scan +/// today). Expected O(N) before the experiments, O(1) after the winning combination. Unlike +/// BM_Admits (a promote, which never calls `manifestAlreadyOwned`), this previews a pure add -- +/// the op every part publication starts with -- so it is the shape production traces show as +/// linear even after the incremental-budget fix landed for BM_Admits' promote shape. +static void BM_AdmitsAddPrecommit(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableState table = makeSyntheticState(n); + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "brand_new_part", ManifestRef{2, 1, 1}}; + + for (auto _ : state) + benchmark::DoNotOptimize(admits(table, op, 1ull << 40, 1ull << 40)); + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_AdmitsAddPrecommit)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// One transaction end-to-end: scratch copy + validate + apply + install (a promote of the +/// staged precommit). The copy is part of the measured cost on purpose -- it is what E3 attacks. +static void BM_ApplyRefLogTxn(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableState table = makeSyntheticState(n); + + RefLogTxn txn; + txn.ns = "roots/bench"; + txn.txn_id = RefTxnId{1, 2}; + RefOp promote; + promote.kind = RefOpKind::OwnerTransition; + promote.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "new_part_x", ManifestRef{1, 1, 999999}}; + promote.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "new_part_x", ManifestRef{1, 1, 999999}}; + txn.ops.push_back(promote); + + for (auto _ : state) + { + RefTableState scratch = table; + applyRefLogTxn(scratch, txn); + benchmark::DoNotOptimize(&scratch); + } + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_ApplyRefLogTxn)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// End-to-end FLUSH-INSTALL cost: apply one state-growing transaction (add a fresh precommit, then +/// promote it -- touching BOTH the committed map AND the owned-manifest index) and then +/// `materializeCommitted()`, which folds BOTH COW overlays into fresh shared bases. THIS is the O(N) +/// critical section production holds `state_mutex` for, once per ref-log flush -- the number the +/// "writer path is flat" claim (drawn from `BM_ApplyRefLogTxn`, which stops before materialize) must be +/// weighed against. `BM_ApplyRefLogTxn` measures apply-without-install; the shipped-report +/// `BM_Materialize` measures only `RefCowMap`'s half; this measures the whole install including the +/// second (`owned_manifests`) container the index added, over the same N range. +static void BM_FlushInstall(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableState table = makeSyntheticState(n); // materialized, as a live table is at a flush boundary + + /// add + promote of a fresh ref: the add inserts into `owned_manifests`, the promote grows + /// `committed` -- so materialize below folds a nonempty overlay in BOTH containers. Manifest {4,1,1} + /// and ref name are unique against the synthetic snapshot's {1,1,*} rows and "new_part_x" precommit. + RefLogTxn txn; + txn.ns = "roots/bench"; + txn.txn_id = RefTxnId{1, 2}; + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "flush_install_new_part", ManifestRef{4, 1, 1}}; + txn.ops.push_back(add); + RefOp promote; + promote.kind = RefOpKind::OwnerTransition; + promote.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "flush_install_new_part", ManifestRef{4, 1, 1}}; + promote.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "flush_install_new_part", ManifestRef{4, 1, 1}}; + txn.ops.push_back(promote); + + for (auto _ : state) + { + RefTableState working = table; // O(1): shared base + applyRefLogTxn(working, txn); // O(ops): bounded overlay + working.materializeCommitted(); // O(N): the critical-section fold this benchmark exists to measure + benchmark::DoNotOptimize(&working); + } + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_FlushInstall)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// Same flush-install as `BM_FlushInstall`, but exercising the E5 uniquely-owned-base fast path that +/// production actually hits. `BM_FlushInstall` copies a shared fixture (`working = table`), so at +/// `materializeCommitted()` the base still has `use_count() == 2` and the fold must build a fresh +/// base -- O(N). Production's live table has NO outstanding scratch copy at the install point: +/// `CasRefLedger::flushRefBatch` EXPLICITLY releases its trial-validation copy (`working = RefTableState{}`) +/// before allocating the id and doing the post-PUT install, so at `materializeCommitted()` the live +/// base is uniquely owned and the fold happens in place -- O(overlay). This variant models that by +/// rebuilding a private, +/// materialized state each iteration (its base `use_count()` is 1), timing only the apply + in-place +/// materialize. The per-iteration rebuild AND the prior iteration's O(N) teardown are excluded from +/// the measurement by hoisting `working` out of the loop and rebuilding it via move-assignment under +/// Pause/ResumeTiming (the reassignment both destroys the previous grown state and installs a fresh +/// materialized one, all untimed). The residual per-iteration Pause/Resume overhead is a constant +/// floor, so the signal to read is FLATNESS across N (O(overlay)), not the absolute small-N number. +static void BM_FlushInstallUniqueOwner(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + + RefLogTxn txn; + txn.ns = "roots/bench"; + txn.txn_id = RefTxnId{1, 2}; + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "flush_install_new_part", ManifestRef{4, 1, 1}}; + txn.ops.push_back(add); + RefOp promote; + promote.kind = RefOpKind::OwnerTransition; + promote.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "flush_install_new_part", ManifestRef{4, 1, 1}}; + promote.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "flush_install_new_part", ManifestRef{4, 1, 1}}; + txn.ops.push_back(promote); + + /// Hoisted out of the loop so the O(N) teardown of the previous iteration's grown state is folded + /// into the untimed move-assignment below, not charged to the timed apply + materialize region. + RefTableState working; + for (auto _ : state) + { + state.PauseTiming(); + working = makeSyntheticState(n); // private, materialized: base use_count() == 1 + state.ResumeTiming(); + + applyRefLogTxn(working, txn); // O(ops): bounded overlay + working.materializeCommitted(); // O(overlay): uniquely-owned base folded IN PLACE (the E5 win) + benchmark::DoNotOptimize(&working); + } + + state.SetComplexityN(static_cast(n)); +} +/// Fixed iteration count: the E5 fast path makes the timed apply + in-place-materialize region tiny +/// and N-independent, so google-benchmark's default min-time targeting would demand millions of +/// iterations at every N -- each paying an untimed O(N) `makeSyntheticState` rebuild, which explodes +/// at large N. A fixed, modest count keeps every point cheap while still averaging enough samples to +/// read the flatness across N (the whole point of this variant). +BENCHMARK(BM_FlushInstallUniqueOwner)->RangeMultiplier(10)->Range(100, 100000)->Iterations(500)->Complexity(); + +/// The fold/recovery profile: K transactions replayed over a size-N snapshot. Each txn creates +/// and promotes one new ref (two ops), so each add pays today's `manifestAlreadyOwned` scan. +/// K fixed at 256; complexity fit is over N. +static void BM_ReplayHistory(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableSnapshot snapshot = makeSyntheticSnapshot(n); + + constexpr size_t kTailTxns = 256; + std::vector tail; + tail.reserve(kTailTxns); + for (size_t k = 0; k < kTailTxns; ++k) + { + RefLogTxn txn; + txn.ns = "roots/bench"; + txn.txn_id = RefTxnId{1, 2 + k}; + + /// Refs unique per k, and namespaced under writer_epoch 3 so they collide with nothing in + /// the snapshot's own {1,1,i} committed series or its {1,1,999999} precommit. + const String ref_name = "replay_part_" + std::to_string(k); + const ManifestRef manifest_ref{3, 1, static_cast(k + 1)}; + + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref_name, manifest_ref}; + txn.ops.push_back(add); + + RefOp promote; + promote.kind = RefOpKind::OwnerTransition; + promote.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref_name, manifest_ref}; + promote.new_binding = RefOwnerBinding{RefOwnerKind::Committed, ref_name, manifest_ref}; + txn.ops.push_back(promote); + + tail.push_back(std::move(txn)); + } + + for (auto _ : state) + benchmark::DoNotOptimize(replay(snapshot, tail)); + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_ReplayHistory)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// The isolation primitive on its own: one full state copy (COW committed + std::set precommits +/// + counters). Overlay is empty (state fresh from replay+materialize), so this is the floor. +static void BM_ScratchCopy(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + RefTableState table = makeSyntheticState(n); + table.materializeCommitted(); /// makeSyntheticState already materializes; repeated here + /// defensively (a no-op on an empty overlay) so this benchmark's + /// floor claim does not silently depend on that helper's internals. + + for (auto _ : state) + { + RefTableState copy = table; + benchmark::DoNotOptimize(©); + } + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_ScratchCopy)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// Canonical snapshot encoding for size N (per-flush cost, expected O(N) -- the question is the +/// constant, which E4's contiguous scan attacks). +static void BM_SnapshotEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableState table = makeSyntheticState(n); + + for (auto _ : state) + benchmark::DoNotOptimize(encodeRefTableSnapshot(snapshotOf(table, "roots/bench"))); + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_SnapshotEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// Full merged iteration with a 10% overlay (post-copy, pre-materialize shape): an N-row +/// materialized base, then a fresh overlay of N/10 rows layered on top with `materialize()` +/// deliberately not called again -- so iteration must merge base and overlay in sorted order the +/// way the cold full-scan paths (snapshotOf, listRefs, dropNamespace) do against an in-flight batch. +/// Benchmarks `RefCowMap` directly (like `BM_Materialize` below) rather than through +/// `RefTableState::getCommitted()`: this isolates the merge-iteration primitive itself, and building +/// the overlay via `RefTableState`'s promote/precommit transactions would additionally measure the +/// state machine's own per-op bookkeeping, which is not what this benchmark is about. +static void BM_MergedIteration(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + + RefCowMap map; + for (size_t i = 0; i < n; ++i) + { + RefCommittedRow row; + row.ref_name = "part_" + std::to_string(i) + "_20260719_0_1000_1"; + row.manifest_ref = ManifestRef{1, 1, static_cast(i + 1)}; + map.emplace(row.ref_name, row); + } + map.materialize(); + + const size_t overlay_n = std::max(1, n / 10); + for (size_t i = 0; i < overlay_n; ++i) + { + RefCommittedRow row; + row.ref_name = "overlay_part_" + std::to_string(i) + "_20260719_0_1000_1"; + row.manifest_ref = ManifestRef{2, 1, static_cast(i + 1)}; + map.insert_or_assign(row.ref_name, row); + } + + for (auto _ : state) + { + size_t total = 0; + for (const auto [ref_name, row] : map) + total += row.ref_name.size(); + benchmark::DoNotOptimize(total); + } + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_MergedIteration)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// RefCowMap::materialize after one overlay insert on an N-row base (per-flush install cost). +/// Benchmarks RefCowMap directly -- it is a public class. +static void BM_Materialize(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + RefCowMap base_map; + for (size_t i = 0; i < n; ++i) + { + RefCommittedRow row; + row.ref_name = "part_" + std::to_string(i) + "_20260719_0_1000_1"; + row.manifest_ref = ManifestRef{1, 1, static_cast(i + 1)}; + base_map.emplace(row.ref_name, row); + } + base_map.materialize(); + + for (auto _ : state) + { + RefCowMap copy = base_map; + RefCommittedRow new_row; + new_row.ref_name = "brand_new_part_20260719_0_1000_1"; + new_row.manifest_ref = ManifestRef{2, 1, 1}; + copy.insert_or_assign(new_row.ref_name, new_row); + copy.materialize(); + benchmark::DoNotOptimize(©); + } + + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_Materialize)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +BENCHMARK_MAIN(); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h b/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h index 98ea21201c70..d22a0ed0f0f7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h @@ -27,6 +27,8 @@ namespace ErrorCodes extern const int NOT_IMPLEMENTED; } +struct IDiskTransaction; + /// Tries to provide some "transactions" interface, which allow /// to execute (commit) operations simultaneously. We don't provide /// any snapshot isolation here, so no read operations in transactions @@ -115,6 +117,16 @@ class IMetadataTransaction : private boost::noncopyable throwNotImplemented(); } + /// [TXN-ONE-PIPELINE] Optional per-metadata write buffer. Returns a ready-to-use buffer when the + /// metadata implementation owns its write mechanism (e.g. a content-addressed hash-on-write buffer + /// whose blob key is known only after the last byte). `owner` is the disk transaction that must be + /// kept alive for the returned buffer's lifetime and, when `autocommit`, committed from the finalize + /// callback. Default nullptr: the caller uses the generic streaming write path unchanged. + virtual std::unique_ptr tryCreateWriteBuffer( + const std::shared_ptr & /*owner*/, + const std::string & /*path*/, size_t /*buf_size*/, WriteMode /*mode*/, + const WriteSettings & /*settings*/, bool /*autocommit*/) { return nullptr; } + /// Metadata related methods /// Generate blob name for passed absolute local path. @@ -141,6 +153,7 @@ class IMetadataTransaction : private boost::noncopyable throwNotImplemented(); } +<<<<<<< HEAD /// Increment the reference count of a data blob shared between metadata files. virtual void incrementBlobRefCount(const std::string & /* blob */) { @@ -158,6 +171,23 @@ class IMetadataTransaction : private boost::noncopyable { throwNotImplemented(); } +======= + /// In-flight read-your-writes for a part being assembled by THIS transaction (B59). A CA part-build + /// transaction stages blobs (uploaded) + mutable bytes before the single commit; these let a reader + /// that holds the transaction resolve those staged files before they are committed. Default: no + /// in-flight visibility (the committed metadata path is authoritative). + virtual std::optional tryGetInFlightStorageObjects(const std::string & /*path*/) const { return {}; } + virtual std::unique_ptr tryReadFileInFlight( + const std::string & /*path*/, const ReadSettings & /*settings*/, std::optional /*read_hint*/) const { return nullptr; } + virtual std::optional tryGetInFlightFileSize(const std::string & /*path*/) const { return {}; } + /// Directory-granularity counterpart of the file trio: true iff this transaction has STAGED at least one + /// file under `path` for `path`'s part. Used so a carried-forward projection dir is visible to + /// loadProjections during finalize. Default: no in-flight directory visibility. + virtual bool hasInFlightDirectory(const std::string & /*path*/) const { return false; } + /// Immediate-child names staged directly under `path` (one level). Used so loadProjections' + /// withPartFormatFromDisk can iterate a staged projection dir to find its mark file. Default: empty. + virtual std::vector listInFlightDirectory(const std::string & /*path*/) const { return {}; } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) virtual ~IMetadataTransaction() = default; @@ -323,6 +353,23 @@ class IMetadataStorage : private boost::noncopyable return false; } + /// Returns true if the metadata storage is content-addressed, i.e. blob keys are derived + /// from content hashes and are only known after all bytes have been written. Such a storage + /// cannot use the up-front-key streaming write path of `DiskObjectStorageTransaction`; the + /// disk transaction delegates writes to the metadata transaction's content-addressed buffer. + virtual bool isContentAddressed() const { return false; } + + /// [TXN-ONE-PIPELINE] True when a transaction from this storage stages every mutation into a + /// transaction-private overlay at call time (eager) rather than queuing effects for FIFO replay in + /// commit. When true, DiskObjectStorageTransaction routes every mutating method straight to the + /// metadata transaction and keeps its own operations_to_execute queue empty. Default false + /// (ordinary object storage). + virtual bool transactionIsStagingOverlay() const { return false; } + + /// True when a file write through this metadata storage publishes atomically, i.e. no partial + /// content is ever observable under the file's final name (see `IDataPartStorage::supportsAtomicFileWrites`). + virtual bool supportsAtomicFileWrites() const { return false; } + using BlobsToRemove = std::unordered_map; virtual BlobsToRemove getBlobsToRemove(const ClusterConfigurationPtr & /*cluster*/, int64_t /*max_count*/) { return {}; } virtual int64_t recordAsRemoved(const StoredObjects & /*blobs*/) { return 0; } @@ -361,6 +408,12 @@ class IMetadataStorage : private boost::noncopyable /// True if write with Append mode supported. virtual bool supportWritingWithAppend() const { return false; } + /// True iff this metadata storage can persist the per-part mutable transaction file (txn_version.txt) + /// under MVCC. Distinct from supportWritingWithAppend: transactions rewrite txn_version.txt (tmp + + /// replaceFile), they never WriteMode::Append, so append-capability is the wrong proxy. A + /// content-addressed disk supports the mutable txn file via its per-ref sidecar. + virtual bool supportsTransactionalMutableFiles() const { return false; } + protected: [[noreturn]] static void throwNotImplemented() { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp index 1b16157562b1..0dbe2048e81e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp @@ -7,9 +7,15 @@ #endif #include #include +<<<<<<< HEAD #include +======= +#include +#include +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include +#include #include @@ -22,6 +28,12 @@ namespace ErrorCodes extern const int UNKNOWN_ELEMENT_IN_CONFIG; extern const int INVALID_CONFIG_PARAMETER; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; +} + +namespace ContentAddressedSetting +{ + extern const ContentAddressedSettingsString scratch_path; } namespace @@ -206,6 +218,35 @@ static void registerPlainRewritableMetadataStorage(MetadataStorageFactory & fact }); } +static void registerContentAddressedMetadataStorage(MetadataStorageFactory & factory) +{ + factory.registerMetadataStorageType("cas", []( + const std::string & name, + const Poco::Util::AbstractConfiguration & config, + const std::string & config_prefix, + const ClusterConfigurationPtr & cluster, + const ObjectStorageRouterPtr & object_storages) -> MetadataStoragePtr + { + checkSingleLocation(cluster); + + const auto local_object_storage = object_storages->takePointingTo(cluster->getLocalLocation()); + std::string key_compatibility_prefix = getObjectKeyCompatiblePrefix(local_object_storage, config, config_prefix); + + auto global_context = Context::getGlobalContextInstance(); + ContentAddressedSettings settings; + settings.loadFromConfig( + config, config_prefix, + /*scratch_path_anchor_if_relative=*/ global_context->getPath(), + /*default_scratch_path=*/ fs::path(global_context->getPath()) / "disks" / name / "cas_scratch" / "", + [&](const std::string & s) { return global_context->getMacros()->expand(s); }); + fs::create_directories(settings[ContentAddressedSetting::scratch_path].value); + + return std::make_shared( + local_object_storage, key_compatibility_prefix, toString(ServerUUID::get()), + name, global_context, settings); + }); +} + static void registerMetadataStorageFromStaticFilesWebServer(MetadataStorageFactory & factory) { factory.registerMetadataStorageType("web", []( @@ -248,6 +289,7 @@ void registerMetadataStorages() registerMetadataStorageFromDisk(factory); registerPlainMetadataStorage(factory); registerPlainRewritableMetadataStorage(factory); + registerContentAddressedMetadataStorage(factory); registerMetadataStorageFromStaticFilesWebServer(factory); registerMetadataStorageFromIndexPages(factory); #if CLICKHOUSE_CLOUD diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index b156519c30b6..42b02114fe14 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -289,6 +289,14 @@ using ObjectKeysWithMetadata = std::vector; class IObjectStorageIterator; using ObjectStorageIteratorPtr = std::shared_ptr; +/// Outcome of a token-conditional single-object removal (content-addressed disks). +enum class ConditionalRemoveOutcome : uint8_t { Removed, TokenMismatch, NotFound }; +struct ConditionalRemoveResult +{ + ConditionalRemoveOutcome outcome = ConditionalRemoveOutcome::NotFound; + bool created_delete_marker = false; /// backend reported a versioning delete marker +}; + /// Base class for all object storages which implement some subset of ordinary filesystem operations. /// /// Examples of object storages are S3, Azure Blob Storage, HDFS. @@ -346,6 +354,14 @@ class IObjectStorage return tryGetObjectMetadata(object.getPath(), with_tags); } + /// Same as tryGetObjectMetadata(), but lets a backend that speaks a native conditional-request + /// dialect (GCS generation tokens) read one while consulting this metadata. Object storages with + /// no such dialect fall back to the ordinary read. + virtual std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const + { + return tryGetObjectMetadata(path, with_tags); + } + /// Read single object virtual std::unique_ptr readObject( /// NOLINT const StoredObject & object, @@ -403,6 +419,15 @@ class IObjectStorage /// Remove objects on path if exists virtual void removeObjectsIfExist(const StoredObjects & object) = 0; + /// Remove `object` ONLY if its current entity tag equals `etag`. Backends without enforced + /// conditional removal MUST NOT override this: the content-addressed capability probe relies on the + /// default to fail closed. Supported: S3 (DeleteObject If-Match, GA 2025-09). + virtual ConditionalRemoveResult removeObjectIfTokenMatches(const StoredObject & /*object*/, const std::string & /*etag*/) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Conditional (token-exact) object removal is not implemented for {} object storage", getName()); + } + /// Copy object with different attributes if required virtual void copyObject( /// NOLINT const StoredObject & object_from, @@ -464,6 +489,7 @@ class IObjectStorage virtual bool supportParallelWrite() const { return false; } +<<<<<<< HEAD /// Whether a fetched `ObjectMetadata` is guaranteed to carry at least one comparable generation /// token — a non-empty `etag`, a known size, or a known modification time — so that two fetches /// of the same path can prove the object was not overwritten in between. Web origins may @@ -471,6 +497,41 @@ class IObjectStorage /// that must reread the same generation of an object (e.g. lazy materialization) have to skip /// such storages instead of failing close at read time. virtual bool supportsObjectGenerationComparison() const { return true; } +======= + /// True when the incarnation tokens this storage returns from writes/HEADs are GCS generation + /// numbers riding the ETag plumbing (http_client = gcs_hmac or gcp_oauth). + /// Consumers (the CAS backend) stamp TokenType::Generation and route conditional writes + /// through the single-PUT path (GCS enforces no preconditions on CompleteMultipartUpload). + virtual bool conditionalOpsUseGenerationTokens() const { return false; } + + /// Declare that this storage's answer to `conditionalOpsUseGenerationTokens` must not change for + /// the rest of its life, and what that answer is expected to be. A caller that has already derived + /// persistent state from the dialect pins it here; a later `applyNewSettings` that would flip it + /// must then be refused rather than silently swapping the client underneath that state. + /// + /// The check has to live at this layer because the effective value is only known here: it is merged + /// from the storage's current settings, any endpoint-level block and the disk's own section, and no + /// caller holding configuration text alone can reproduce that resolution. + /// + /// Unpinned by default, so an ordinary storage's settings and reload behaviour are unchanged. + virtual void pinConditionalOpsGenerationDialect(bool /*expect_generation_tokens*/) {} + + /// Whether the underlying bucket has object versioning enabled; nullopt when unknown or not + /// applicable. Used by the CAS capability probe to fail closed on GCS: on a versioned bucket + /// a token-exact DELETE archives a noncurrent generation instead of reclaiming storage. + virtual std::optional isBucketVersioningEnabled() const { return std::nullopt; } + + /// True when this object storage can execute writes under the given retry profile. + /// A caller that sets a non-Default profile on WriteSettings MUST check this first and + /// fail closed if unsupported (the profile is advisory only to backends that opt in). + virtual bool supportsRetryProfile(ObjectStorageRetryProfile profile) const { return profile == ObjectStorageRetryProfile::Default; } + + /// True when this object storage can execute copies under the given transport requirement. + virtual bool supportsCopyMode(ObjectStorageCopyMode mode) const + { + return mode == ObjectStorageCopyMode::Default; + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) virtual ReadSettings patchSettings(const ReadSettings & read_settings) const; diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp index 8d29f6a9a85d..afc0ae4c65f1 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp @@ -788,7 +788,30 @@ ObjectMetadata LocalObjectStorage::getObjectMetadata(const std::string & path, b throw fs::filesystem_error( "Got unexpected error while getting file metadata", resolved_path, std::error_code(errno, std::generic_category())); +<<<<<<< HEAD return makeObjectMetadata(file_stat); +======= + object_metadata.size_bytes = fs::file_size(resolved_path); + object_metadata.etag = std::to_string(std::chrono::duration_cast(time.time_since_epoch()).count()); + object_metadata.last_modified = Poco::Timestamp::fromEpochTime( + std::chrono::duration_cast(time.time_since_epoch()).count()); + return object_metadata; +} + +std::optional LocalObjectStorage::tryGetObjectMetadata(const std::string & path, bool) const +{ + auto resolved_path = resolvePathRelativelyToKeyPrefix(path); + LOG_TEST(log, "Getting metadata for path: {}", resolved_path); + + /// A directory is not an object: fs::file_size would throw "Is a directory". Treat it as a + /// missing object (nullopt) so callers probing whether a path is a readable object do not get + /// a raw filesystem error (B38: system.remote_data_paths traversal on a CAS pool). + std::error_code error; + if (fs::is_directory(resolved_path, error)) + return {}; + + return tryStatResolvedPath(resolved_path); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } void LocalObjectStorage::listObjects(const std::string & path, RelativePathsWithMetadata & children, size_t/* max_keys */) const diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index 5fdd13cef073..65fb7753ca69 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -17,6 +17,8 @@ #include #include #include +#include +#include #include #include #include @@ -33,6 +35,7 @@ #include #include #include +#include #include #include #include @@ -70,8 +73,21 @@ namespace Setting namespace S3RequestSetting { + extern const S3RequestSettingsBool allow_native_copy; + extern const S3RequestSettingsBool check_objects_after_upload; extern const S3RequestSettingsUInt64 list_object_keys_size; extern const S3RequestSettingsUInt64 objects_chunk_size_to_delete; + extern const S3RequestSettingsUInt64 max_single_part_upload_size; + extern const S3RequestSettingsUInt64 min_upload_part_size; + extern const S3RequestSettingsUInt64 max_unexpected_write_error_retries; + extern const S3RequestSettingsUInt64 max_single_operation_copy_size; +} + + +namespace S3AuthSetting +{ + extern const S3AuthSettingsString http_client; + extern const S3AuthSettingsUInt64 gcs_max_conditional_put_bytes; } @@ -79,6 +95,8 @@ namespace ErrorCodes { extern const int BAD_ARGUMENTS; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; + extern const int S3_ERROR; } namespace @@ -239,7 +257,8 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync bool S3ObjectStorage::exists(const StoredObject & object) const { auto settings_ptr = s3_settings.get(); - return S3::objectExists(*client.get(), uri.bucket, object.remote_path, {}); + const bool e = S3::objectExists(*client.get(), uri.bucket, object.remote_path, {}); + return e; } std::unique_ptr S3ObjectStorage::readObject( /// NOLINT @@ -329,6 +348,28 @@ std::unique_ptr S3ObjectStorage::writeObject( /// NOLIN request_settings.updateFromSettings(settings, /* if_changed */ true, settings[Setting::s3_validate_request_settings]); } + if (write_settings.s3_check_objects_after_upload_override) + request_settings[S3RequestSetting::check_objects_after_upload] = *write_settings.s3_check_objects_after_upload_override; + + if (write_settings.s3_force_single_part_upload) + { + /// A conditional write on a generation-token store must stay in ONE buffered part, so the + /// single-PUT path remains available up to the configured ceiling. + const UInt64 cap = s3_settings.get()->auth_settings[S3AuthSetting::gcs_max_conditional_put_bytes]; + request_settings[S3RequestSetting::max_single_part_upload_size] = cap; + request_settings[S3RequestSetting::min_upload_part_size] = cap; + } + + if (write_settings.s3_max_unexpected_write_error_retries_override) + { + /// WriteBufferFromS3's OWN retry loop (makeSinglepartUpload/completeMultipartUpload) reissues + /// the identical request — WITH its If-None-Match/If-Match condition — on a NO_SUCH_KEY + /// response; this sits ABOVE the S3 client, so a client-level profile override does not bound + /// it. See WriteSettings. + request_settings[S3RequestSetting::max_unexpected_write_error_retries] + = write_settings.s3_max_unexpected_write_error_retries_override; + } + ThreadPoolCallbackRunnerUnsafe scheduler; if (write_settings.s3_allow_parallel_part_upload) scheduler = threadPoolCallbackRunnerUnsafe(getThreadPoolWriter(), ThreadName::REMOTE_FS_WRITE_THREAD_POOL); @@ -337,8 +378,18 @@ std::unique_ptr S3ObjectStorage::writeObject( /// NOLIN if (blob_storage_log) blob_storage_log->local_path = object.local_path; + /// The SingleAttempt profile (e.g. CAS conditional writes, RFC cas-s3-timeout-retry-control) rides + /// on WriteSettings instead of changing this disk's shared client — every other write keeps using + /// client.get() and its normal retry policy unchanged. getSingleAttemptClient() is only invoked + /// when actually selected, so a plain write never pays for building/locking the clone. + std::shared_ptr used_client; + if (write_settings.object_storage_retry_profile == ObjectStorageRetryProfile::SingleAttempt) + used_client = getSingleAttemptClient(); + else + used_client = client.get(); + return std::make_unique( - client.get(), + used_client, uri.bucket, object.remote_path, write_settings.use_adaptive_write_buffer ? write_settings.adaptive_write_buffer_initial_size : buf_size, @@ -471,6 +522,80 @@ void S3ObjectStorage::removeObjectsIfExist(const StoredObjects & objects) removeObjectsImpl(objects, true); } +ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatches(const StoredObject & object, const std::string & etag) +{ + S3::DeleteObjectRequest request; + request.SetBucket(uri.bucket); + request.SetKey(object.remote_path); + request.SetIfMatch(etag); + /// This is a content-addressed exact-token DELETE: mark it eligible for the typed NativeConditional + /// mode, so a GCS-native client can send the generation token this etag actually encodes. + request.setNativeConditional(); + + ProfileEvents::increment(ProfileEvents::DiskS3DeleteObjects); + + auto outcome = client.get()->DeleteObject(request); + + /// Mirror removeObjectImpl (deleteFileFromS3): every conditional delete lands in + /// system.blob_storage_log too — GC reclaim was invisible there otherwise. TokenMismatch + /// and NotFound are routine protocol outcomes, recorded with the S3 error for filtering. + if (auto blob_storage_log = BlobStorageLogWriter::create(disk_name)) + blob_storage_log->addEvent(BlobStorageLogElement::EventType::Delete, + uri.bucket, object.remote_path, + object.local_path, object.bytes_size, + /* elapsed_microseconds */ 0, + outcome.IsSuccess() ? 0 : static_cast(outcome.GetError().GetErrorType()), + outcome.IsSuccess() ? "" : outcome.GetError().GetMessage()); + + if (outcome.IsSuccess()) + return {ConditionalRemoveOutcome::Removed, outcome.GetResult().GetDeleteMarker()}; + + const auto & err = outcome.GetError(); + + /// The token did not match the current incarnation: the conditional delete is rejected with a 412 + /// (see `S3::isPreconditionFailedError` for the one policy). Callers treat 'mismatch' and 'gone' + /// alike (re-validate); a genuine absence is disambiguated downstream by a HEAD re-check. + if (S3::isPreconditionFailedError(err)) + return {ConditionalRemoveOutcome::TokenMismatch, false}; + + /// The object no longer exists (404). Protocol callers treat 'mismatch' and 'gone' alike (re-validate). + if (S3::isNotFoundError(err.GetErrorType())) + return {ConditionalRemoveOutcome::NotFound, false}; + + throw S3Exception(err.GetErrorType(), + "{} (Code: {}, S3 exception: '{}') while conditionally removing object with path {} from S3", + err.GetMessage(), static_cast(err.GetErrorType()), err.GetExceptionName(), object.remote_path); +} + +bool S3ObjectStorage::conditionalOpsUseGenerationTokens() const +{ + return client.get()->supportsGcsNativeConditionalRequests(); +} + +bool S3ObjectStorage::supportsCopyMode(ObjectStorageCopyMode mode) const +{ + return mode == ObjectStorageCopyMode::Default + || (mode == ObjectStorageCopyMode::NativeOnly + && s3_settings.get()->request_settings[S3RequestSetting::allow_native_copy]); +} + +void S3ObjectStorage::pinConditionalOpsGenerationDialect(bool expect_generation_tokens) +{ + pinned_generation_dialect.store(expect_generation_tokens ? 1 : 0); +} + +std::optional S3ObjectStorage::isBucketVersioningEnabled() const +{ + S3::GetBucketVersioningRequest request; + request.SetBucket(uri.bucket); + + auto outcome = client.get()->GetBucketVersioning(request); + if (!outcome.IsSuccess()) + return std::nullopt; + + return outcome.GetResult().GetStatus() == Aws::S3::Model::BucketVersioningStatus::Enabled; +} + static void putObjectsTagOnS3( const std::shared_ptr & s3_client, const String & bucket, @@ -543,9 +668,20 @@ void S3ObjectStorage::tagObjects(const StoredObjects & objects, const std::strin } std::optional S3ObjectStorage::tryGetObjectMetadata(const std::string & path, bool with_tags) const +{ + return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::Default); +} + +std::optional S3ObjectStorage::tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const +{ + return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::NativeConditional); +} + +std::optional S3ObjectStorage::tryGetObjectMetadataImpl(const std::string & path, bool with_tags, ObjectStorageRequestMode request_mode) const { auto settings_ptr = s3_settings.get(); - auto object_info = S3::getObjectInfoIfExists(*client.get(), uri.bucket, path, {}, /* with_metadata= */ true, with_tags); + auto object_info = S3::getObjectInfoIfExists( + *client.get(), uri.bucket, path, {}, /* with_metadata= */ true, with_tags, request_mode); if (object_info.size == 0 && object_info.last_modification_time == 0 && object_info.metadata.empty()) return {}; @@ -632,13 +768,16 @@ void S3ObjectStorage::copyObjectToAnotherObjectStorage( // NOLINT BlobStorageLogWriter::create(disk_name), scheduler, [&, this]{ return readObject(object_from, read_settings_to_use);}, - object_to_attributes); + object_to_attributes, + write_settings.object_storage_copy_mode); return; } catch (S3Exception & exc) { - /// If authentication/permissions error occurs then fallthrough to copy with buffer. - if (exc.getS3ErrorCode() != Aws::S3::S3Errors::ACCESS_DENIED) + /// Default mode may fall through to a buffered copy after an authentication/permissions error; + /// NativeOnly must preserve the native-copy failure. + if (write_settings.object_storage_copy_mode == ObjectStorageCopyMode::NativeOnly + || exc.getS3ErrorCode() != Aws::S3::S3Errors::ACCESS_DENIED) throw; else { @@ -661,6 +800,11 @@ void S3ObjectStorage::copyObjectToAnotherObjectStorage( // NOLINT } } + if (write_settings.object_storage_copy_mode == ObjectStorageCopyMode::NativeOnly) + throw Exception( + ErrorCodes::NOT_IMPLEMENTED, + "Native-only object copy requires both object storages to use the native S3 copy path"); + IObjectStorage::copyObjectToAnotherObjectStorage(object_from, object_to, read_settings, write_settings, object_storage_to, object_to_attributes); } @@ -668,9 +812,16 @@ void S3ObjectStorage::copyObject( // NOLINT const StoredObject & object_from, const StoredObject & object_to, const ReadSettings & read_settings, - const WriteSettings &, + const WriteSettings & write_settings, std::optional object_to_attributes) { + if (!supportsCopyMode(write_settings.object_storage_copy_mode)) + throw Exception( + ErrorCodes::NOT_IMPLEMENTED, + "Native-only object copy requires the native S3 copy path, which is disabled " + "(allow_native_copy=false) for object storage {}", + getName()); + auto current_client = client.get(); auto settings_ptr = s3_settings.get(); auto size = S3::getObjectSize(*current_client, uri.bucket, object_from.remote_path, {}); @@ -690,7 +841,8 @@ void S3ObjectStorage::copyObject( // NOLINT BlobStorageLogWriter::create(disk_name), scheduler, [&, this]{ return readObject(object_from, read_settings_to_use);}, - object_to_attributes); + object_to_attributes, + write_settings.object_storage_copy_mode); } void S3ObjectStorage::shutdown() @@ -754,6 +906,7 @@ void S3ObjectStorage::applyNewSettings( modified_settings->request_settings.proxy_resolver = DB::ProxyConfigurationResolverProvider::getFromOldSettingsFormat( ProxyConfiguration::protocolFromString(uri.uri.getScheme()), config_prefix, config); +<<<<<<< HEAD /// The effective credentials of a non-disk S3 storage depend on the accessing session's restriction mode /// (`s3_allow_server_credentials_in_user_queries`), not only on the stored settings. Rebuild the client when /// that mode differs from the one the current client was built under, so an opt-in session cannot leave a @@ -762,6 +915,32 @@ void S3ObjectStorage::applyNewSettings( /// constant and this adds no rebuilds. const bool restricts_now = !for_disk_s3 && context->shouldRestrictUserQueryS3Credentials(); const bool restriction_mode_changed = client_restricts_server_credentials != restricts_now; +======= + /// A caller that derived persistent state from the conditional-ops dialect pinned it (see + /// `IObjectStorage::pinConditionalOpsGenerationDialect`). Refuse before the client is replaced, so a + /// rejected reload leaves the working client and its dialect in place. + /// + /// This is the only point where the question can be answered: `modified_settings` above is the merge + /// of the current settings, any endpoint-level block and the disk's own section, and `http_client` + /// may be set by any of them. Checking a single config section instead would miss an endpoint-level + /// flip entirely, and would refuse a reload that changes nothing whenever the effective value comes + /// from somewhere other than that section. + if (const int8_t pinned = pinned_generation_dialect.load(); pinned >= 0) + { + const bool would_be_generation + = S3::httpClientImpliesGcsGenerationDialect(modified_settings->auth_settings[S3AuthSetting::http_client]); + if (would_be_generation != (pinned == 1)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Object storage {} cannot change its conditional-operation dialect on reload: it is in " + "use by a mount that has already recorded {} incarnation tokens, and the new settings " + "resolve `http_client` to '{}', which would mint {} ones. Persisted tokens would no " + "longer be comparable. Keep the previous `http_client`, or recreate the mount.", + getName(), + pinned == 1 ? "generation" : "ETag", + modified_settings->auth_settings[S3AuthSetting::http_client].value, + would_be_generation ? "generation" : "ETag"); + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) auto current_settings = s3_settings.get(); /// A change in the accessing session's restriction mode forces a client rebuild even for an otherwise static @@ -799,6 +978,29 @@ std::shared_ptr S3ObjectStorage::tryGetS3StorageClient() return client.get(); } +std::shared_ptr S3ObjectStorage::getSingleAttemptClient() const +{ + auto base = client.get(); + std::lock_guard lock(single_attempt_client_mutex); + if (single_attempt_client && single_attempt_client_base == base) + return single_attempt_client; + + auto cfg = base->getClientConfiguration(); + cfg.retry_strategy.max_retries = 0; + cfg.retryStrategy = std::make_shared(); + + /// A server can reject an If-Match/If-None-Match request before accepting its body; waiting for + /// the 100-continue response avoids uploading a large body that cannot commit. Respect the + /// disk's configured expect_continue_min_bytes; if unset, use the established 1 MiB floor. + static constexpr uint64_t fallback_expect_continue_min_bytes = 1024 * 1024; + if (cfg.expect_continue_min_bytes == 0) + cfg.expect_continue_min_bytes = fallback_expect_continue_min_bytes; + + single_attempt_client = base->cloneWithConfigurationOverride(cfg); + single_attempt_client_base = base; + return single_attempt_client; +} + bool S3ObjectStorage::tryRefreshCredentialsViaCallback() { fiu_do_on(FailPoints::object_storage_force_refresh_callback_success, { return true; }); diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h index 52ba5691bf91..ad19801ed744 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -112,12 +113,19 @@ class S3ObjectStorage : public IObjectStorage /// `DeleteObjectsRequest` does not exist on GCS, see https://issuetracker.google.com/issues/162653700 . void removeObjectsIfExist(const StoredObjects & objects) override; + /// Uses `DeleteObjectRequest` with `If-Match` (token-exact removal for content-addressed disks). + ConditionalRemoveResult removeObjectIfTokenMatches(const StoredObject & object, const std::string & etag) override; + void tagObjects(const StoredObjects & objects, const std::string & tag_key, const std::string & tag_value) override; ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override; std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override; + /// Marks the HEAD request eligible for the typed NativeConditional mode, so the CAS backend's + /// `nativeHead` can read a GCS generation token where the client's HTTP layer supports one. + std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const override; + void copyObject( /// NOLINT const StoredObject & object_from, const StoredObject & object_to, @@ -153,6 +161,16 @@ class S3ObjectStorage : public IObjectStorage bool isReadOnly() const override { return s3_settings.get()->request_settings[S3RequestSetting::read_only]; } + bool conditionalOpsUseGenerationTokens() const override; + + void pinConditionalOpsGenerationDialect(bool expect_generation_tokens) override; + + std::optional isBucketVersioningEnabled() const override; + + bool supportsRetryProfile(ObjectStorageRetryProfile) const override { return true; } + + bool supportsCopyMode(ObjectStorageCopyMode mode) const override; + std::shared_ptr getS3StorageClient() override; std::shared_ptr tryGetS3StorageClient() override; @@ -160,10 +178,20 @@ class S3ObjectStorage : public IObjectStorage S3::URI getURI() const { return uri; } S3Settings getS3Settings() const { return *s3_settings.get(); } + + /// Lazily-built clone of the current disk client with the single-attempt retry profile + /// (SingleAttemptRetryStrategy, max_retries=0, Expect:100-continue floor). Rebuilt whenever the + /// disk client rotates (applyNewSettings/credentials refresh) — the cached clone is keyed by the + /// base client's identity, so a stale clone can never outlive a rotation. + std::shared_ptr getSingleAttemptClient() const; private: void removeObjectImpl(const StoredObject & object, bool if_exists); void removeObjectsImpl(const StoredObjects & objects, bool if_exists); + /// Shared by tryGetObjectMetadata/tryGetObjectMetadataWithNativeToken: the only difference between + /// the two public overrides is which ObjectStorageRequestMode the HEAD wrapper carries. + std::optional tryGetObjectMetadataImpl(const std::string & path, bool with_tags, ObjectStorageRequestMode request_mode) const; + const S3::URI uri; std::string disk_name; @@ -185,6 +213,22 @@ class S3ObjectStorage : public IObjectStorage const bool for_disk_s3; S3CredentialsRefreshCallback credentials_refresh_callback; + + /// Set once by a caller that has derived persistent state from the conditional-ops dialect (see + /// `pinConditionalOpsGenerationDialect`). Once set, `applyNewSettings` refuses a reload whose + /// effective `http_client` would flip the dialect, and keeps the working client. + std::atomic pinned_generation_dialect{-1}; /// -1 unpinned, 0 pinned ETag, 1 pinned generation + + mutable std::mutex single_attempt_client_mutex; + mutable std::shared_ptr single_attempt_client; + /// The base client the cached clone above was built from. Deliberately held as a shared_ptr (not + /// a raw pointer): a raw pointer would be compared for identity AFTER the object it once pointed + /// to could have been freed and a new client reallocated at the same address by an unrelated + /// rotation (ABA), which would false-match and serve a stale clone (e.g. built from retired + /// credentials) indefinitely. Holding the shared_ptr pins at most one retired client version — + /// released as soon as the next rotation is observed and the clone is rebuilt — which is what + /// makes the identity comparison in getSingleAttemptClient sound. + mutable std::shared_ptr single_attempt_client_base; }; } diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/diskSettings.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/diskSettings.cpp index 8f5fa5494f91..ca2676978080 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/diskSettings.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/diskSettings.cpp @@ -53,6 +53,7 @@ namespace S3AuthSetting extern const S3AuthSettingsString access_key_id; extern const S3AuthSettingsUInt64 connect_timeout_ms; extern const S3AuthSettingsBool disable_checksum; + extern const S3AuthSettingsUInt64 expect_continue_min_bytes; extern const S3AuthSettingsUInt64 expiration_window_seconds; extern const S3AuthSettingsBool gcs_issue_compose_request; extern const S3AuthSettingsUInt64 http_keep_alive_max_requests; @@ -182,6 +183,7 @@ getClient(const S3::URI & url, const S3Settings & settings, ContextPtr context, client_configuration.endpointOverride = url.endpoint; client_configuration.s3_use_adaptive_timeouts = auth_settings[S3AuthSetting::use_adaptive_timeouts]; + client_configuration.expect_continue_min_bytes = auth_settings[S3AuthSetting::expect_continue_min_bytes]; if (request_settings.proxy_resolver) { diff --git a/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp b/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp index c0fc0689065b..f40cb2fc2eaf 100644 --- a/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp @@ -6,12 +6,19 @@ #include #include #include +#include +#include #include namespace DB { +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + void registerObjectStorages(); void registerMetadataStorages(); void registerDiskObjectStorage(DiskFactory & factory, bool global_skip_access_check); @@ -77,6 +84,24 @@ void registerDiskObjectStorage(DiskFactory & factory, bool global_skip_access_ch LOG_DEBUG(getLogger("registerDiskObjectStorage"), "Metadata type hint: {}", compatibility_metadata_type_hint); auto metadata_storage = MetadataStorageFactory::instance().create(name, config, config_prefix, cluster, object_storages, compatibility_metadata_type_hint); +<<<<<<< HEAD +======= + /// Content-addressed metadata (like Keeper) requires real, deferred disk transactions: a part's + /// file->blob mappings are accumulated across the whole part write and the manifest + ref are + /// published atomically when the transaction commits. A fake (per-file autocommit) transaction + /// would write each file independently with no commit point for the manifest/ref publish. + const auto metadata_type = metadata_storage->getType(); + const bool needs_real_transaction = metadata_type == MetadataStorageType::Keeper + || metadata_type == MetadataStorageType::CAS; + /// An explicit `use_fake_transaction=true` on a metadata type that requires deferred + /// transactions would silently break the atomic manifest/ref publish (per-file autocommit, + /// no commit point). Reject it instead of honoring it. + if (needs_real_transaction && config.getBool(config_prefix + ".use_fake_transaction", false)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Disk '{}': `use_fake_transaction` cannot be enabled for metadata type '{}'", + name, magic_enum::enum_name(metadata_type)); + bool use_fake_transaction = config.getBool(config_prefix + ".use_fake_transaction", !needs_real_transaction); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) DiskPtr disk = std::make_shared( name, std::move(cluster), diff --git a/src/Disks/DiskType.cpp b/src/Disks/DiskType.cpp index 186e169ba483..09dc4e79e6ef 100644 --- a/src/Disks/DiskType.cpp +++ b/src/Disks/DiskType.cpp @@ -19,6 +19,8 @@ MetadataStorageType metadataTypeFromString(const std::string & type) return MetadataStorageType::Plain; if (check_type == "plain_rewritable") return MetadataStorageType::PlainRewritable; + if (check_type == "cas") + return MetadataStorageType::CAS; if (check_type == "web") return MetadataStorageType::StaticWeb; if (check_type == "web_index") diff --git a/src/Disks/DiskType.h b/src/Disks/DiskType.h index f1b9aebef261..76da5ec98bac 100644 --- a/src/Disks/DiskType.h +++ b/src/Disks/DiskType.h @@ -32,6 +32,7 @@ enum class MetadataStorageType : uint8_t Keeper, Plain, PlainRewritable, + CAS, StaticWeb, WebIndex, Memory, diff --git a/src/Disks/IDisk.h b/src/Disks/IDisk.h index 478795f523f1..b79a3171bde7 100644 --- a/src/Disks/IDisk.h +++ b/src/Disks/IDisk.h @@ -472,6 +472,13 @@ class IDisk : public Space /// If the disk is plain object storage. virtual bool isPlain() const { return false; } + /// If the disk is a content-addressed object-storage pool (`metadata_type = cas`). + /// A clean predicate so callers do not have to reach through `getDataSourceDescription`. + virtual bool isContentAddressed() const { return false; } + + /// True when a file write on this disk publishes atomically (see `IDataPartStorage::supportsAtomicFileWrites`). + virtual bool supportsAtomicFileWrites() const { return false; } + virtual bool isWriteOnce() const { return false; } virtual bool supportsHardLinks() const { return true; } diff --git a/src/Disks/IDiskTransaction.h b/src/Disks/IDiskTransaction.h index decf12ce31f6..3c57c9f63fe4 100644 --- a/src/Disks/IDiskTransaction.h +++ b/src/Disks/IDiskTransaction.h @@ -139,6 +139,7 @@ struct IDiskTransaction : private boost::noncopyable /// Truncate file to the target size. virtual void truncateFile(const std::string & src_path, size_t size) = 0; +<<<<<<< HEAD /// Increment the reference count of a data blob shared between metadata files. virtual void incrementBlobRefCount(const std::string & /* blob */) { @@ -150,6 +151,24 @@ struct IDiskTransaction : private boost::noncopyable { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Blob reference counting is not implemented for this disk transaction"); } +======= + /// In-flight read-your-writes for a part being assembled by THIS transaction (B59). Forwarded to the + /// metadata transaction by object-storage disk transactions; default (e.g. local disk) is no in-flight + /// visibility, so a reader falls through to the committed path. + virtual std::optional tryGetInFlightStorageObjects(const std::string & /*path*/) const { return {}; } + virtual std::unique_ptr tryReadFileInFlight( + const std::string & /*path*/, const ReadSettings & /*settings*/, std::optional /*read_hint*/) const { return nullptr; } + virtual std::optional tryGetInFlightFileSize(const std::string & /*path*/) const { return {}; } + /// In-flight read-your-writes at DIRECTORY granularity: true iff this transaction has STAGED at least one + /// file under `path` for `path`'s part (mirrors the file trio above). Forwarded to the metadata + /// transaction by object-storage disk transactions; default (e.g. local disk) is no in-flight directory + /// visibility, so a reader falls through to the committed path. + virtual bool hasInFlightDirectory(const std::string & /*path*/) const { return false; } + /// In-flight read-your-writes directory ENUMERATION: the immediate-child names this transaction has + /// STAGED directly under `path` (one level, the directory prefix stripped). Forwarded to the metadata + /// transaction; default (e.g. local disk) is empty. + virtual std::vector listInFlightDirectory(const std::string & /*path*/) const { return {}; } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) }; using DiskTransactionPtr = std::shared_ptr; diff --git a/src/Disks/ReadOnlyDiskWrapper.h b/src/Disks/ReadOnlyDiskWrapper.h index 784d84b0655c..9a38e85cde77 100644 --- a/src/Disks/ReadOnlyDiskWrapper.h +++ b/src/Disks/ReadOnlyDiskWrapper.h @@ -89,6 +89,11 @@ class ReadOnlyDiskWrapper : public IDisk NameSet getCacheLayersNames() const override { return delegate->getCacheLayersNames(); } MetadataStoragePtr getMetadataStorage() override { return delegate->getMetadataStorage(); } + /// Forwarded alongside getMetadataStorage: callers that gate on this predicate before reaching + /// for the metadata storage (ContentAddressedMetadataStorage::tryFromDisk and friends) must see + /// the delegate's answer through the wrapper, or a wrapped content-addressed disk silently + /// drops out of the CAS introspection paths. + bool isContentAddressed() const override { return delegate->isContentAddressed(); } std::unordered_map getSerializedMetadata(const std::vector & file_paths) const override { return delegate->getSerializedMetadata(file_paths); } diff --git a/src/Disks/tests/cas_format_test_battery.h b/src/Disks/tests/cas_format_test_battery.h new file mode 100644 index 000000000000..173bfcc5c4df --- /dev/null +++ b/src/Disks/tests/cas_format_test_battery.h @@ -0,0 +1,111 @@ +#pragma once +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int UNKNOWN_FORMAT_VERSION; +} + +/// The shape-level failure-mode battery every v3 format registers with (spec §testing): one call +/// exercises decode-of-encode, golden text, truncation at line boundaries and inside line 1, +/// the v+1 gate, wrong-type, and leading garbage. Key-level rules (tolerant/strict/critical/ +/// duplicate) are unit-tested once on JsonObjectReader — the battery stays format-agnostic. + +struct FormatBatteryCase +{ + DB::Cas::FormatId id; + std::function encode; + std::function decode; + String golden; + /// Optional format-specific construction for the unsupported-version sample. Fixed-size formats + /// use this to preserve their physical envelope while growing a version field across a digit + /// boundary; ordinary line-oriented formats use the default textual replacement below. + std::function make_future_version = {}; +}; + +/// Canonical object headers track the current compatibility generation. The type remains an +/// explicit test literal at every call site, so a registry/type mismatch cannot be hidden by a +/// self-derived expectation. +inline String currentFormatHeader(std::string_view type) +{ + return fmt::format("{{\"type\":\"{}\",\"v\":{}}}\n", type, DB::Cas::currentCompatibilityVersion()); +} + +namespace cas_battery_detail +{ +template +void expectCode(int code, F && f, const String & context) +{ + try + { + f(); + FAIL() << context << ": expected exception " << code; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), code) << context << ": " << e.message(); + } +} +} + +inline void runFormatBattery(const FormatBatteryCase & c) +{ + using namespace DB::Cas; + namespace ec = DB::ErrorCodes; + const FormatTraits & t = traitsFor(c.id); + + const String stored = c.encode(); + c.decode(stored); /// round-trip: must not throw + + /// Work on the canonical text (identical to `stored` for raw formats). + const String text = openObject(c.id, stored); + ASSERT_TRUE(text.starts_with("{\"type\":\"")) << t.type; + + if (!c.golden.empty()) + EXPECT_EQ(text, c.golden) << "golden text drifted for " << t.type; + if (looksZstd(stored) && !c.golden.empty()) + EXPECT_EQ(stored, sealObject(c.id, c.golden)) << "pinned compressed arm drifted for " << t.type; + + /// Truncation at every line boundary (drop the terminator too) fails closed. + for (size_t i = 0; i < text.size(); ++i) + if (text[i] == '\n') + cas_battery_detail::expectCode(ec::CORRUPTED_DATA, + [&] { c.decode(text.substr(0, i)); }, fmt::format("{}: cut at line boundary {}", t.type, i)); + + /// Truncation inside line 1. + const size_t line1 = text.find('\n'); + ASSERT_NE(line1, String::npos); + for (size_t i = 1; i < line1; i += 3) + cas_battery_detail::expectCode(ec::CORRUPTED_DATA, + [&] { c.decode(text.substr(0, i)); }, fmt::format("{}: cut inside header at {}", t.type, i)); + + /// v+1 gate. + const String v_now = fmt::format("\"v\":{}", currentCompatibilityVersion()); + const String v_next = fmt::format("\"v\":{}", currentCompatibilityVersion() + 1); + String future; + if (c.make_future_version) + future = c.make_future_version(text); + else + { + future = text; + future.replace(future.find(v_now), v_now.size(), v_next); + } + cas_battery_detail::expectCode(ec::UNKNOWN_FORMAT_VERSION, [&] { c.decode(future); }, + fmt::format("{}: v+1", t.type)); + + /// Wrong type: another VALID registered type in the header. + const std::string_view other = (t.id == FormatId::PoolMeta) ? "cas_owner" : "cas_pool_meta"; + String mistyped = text; + mistyped.replace(mistyped.find(t.type), t.type.size(), String(other)); + cas_battery_detail::expectCode(ec::CORRUPTED_DATA, [&] { c.decode(mistyped); }, + fmt::format("{}: wrong type", t.type)); + + /// Leading garbage. + cas_battery_detail::expectCode(ec::CORRUPTED_DATA, [&] { c.decode("X" + text); }, + fmt::format("{}: garbage byte", t.type)); +} diff --git a/src/Disks/tests/cas_sweep_test_support.h b/src/Disks/tests/cas_sweep_test_support.h new file mode 100644 index 000000000000..c1b5466a1cb1 --- /dev/null +++ b/src/Disks/tests/cas_sweep_test_support.h @@ -0,0 +1,36 @@ +#pragma once +#include +#include +#include +#include + +namespace DB::Cas::tests +{ + +/// TEST-ONLY variant of the cursor page: plans a page via the production `planManifestCursorPage` and +/// then exact-token-deletes every nomination immediately, with no source-edge retirement and no +/// `gc/state` adoption of the retirement. Production deletion always goes through `Gc::fold`'s +/// orphan_sweep phase instead, which adopts the retirements in the same round CAS before deleting — +/// this shortcut recreates the accounting hole that path exists to close, so it must never be reached +/// from a production translation unit. +inline ManifestSweepResult sweepManifestCursorPageForTest( + Pool & store, + const String & cursor, + uint64_t list_budget, + uint64_t delete_budget, + GcRoundWorkBudget * work_budget = nullptr) +{ + ManifestSweepResult result = planManifestCursorPage( + store, cursor, list_budget, delete_budget, /*catalog_recovery_authoritative=*/true, work_budget); + for (const ManifestSweepResult::Nomination & nomination : result.nominations) + { + const DeleteOutcome outcome = store.backend().deleteExact(nomination.key, nomination.token); + if (classifyDeleteOutcome(outcome) == DeleteClass::Deleted) + ++result.deleted; + else + ++result.skipped; + } + return result; +} + +} diff --git a/src/Disks/tests/cas_test_helpers.h b/src/Disks/tests/cas_test_helpers.h new file mode 100644 index 000000000000..e7b5f6221c18 --- /dev/null +++ b/src/Disks/tests/cas_test_helpers.h @@ -0,0 +1,2328 @@ +#pragma once + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#include +#include +#include +/// For `ChunkFaultBackend`'s `DefiniteFailure` mode, which needs a real S3-classified error, and for +/// the ambiguity it raises otherwise. +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Per-TU extern declarations for the `ContentAddressedSetting` entries this header's helpers use -- +/// the established pattern for `BaseSettings`-derived classes in this codebase (see e.g. +/// `RegisterDiskCache.cpp`'s `namespace FileCacheSetting` block): the entries are DEFINED once in +/// `ContentAddressedSettings.cpp`, and each consumer TU declares only the ones it references. +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsString server_root_id; + extern const ContentAddressedSettingsString scratch_path; +} + +/// Same per-TU pattern for the error codes this header's fault backends raise (`ChunkFaultBackend`'s +/// non-S3 build of the `Definite` mode). +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +namespace DB::Cas::tests +{ + +/// Deterministic two-phase barrier for worker-lifecycle tests. The worker calls `arriveAndWait` at +/// the exact operation boundary under test; the test waits for that arrival and later calls +/// `release`. The bounded waits are only hang protection -- correctness never depends on elapsed +/// time or a polling sleep. +class ManualBarrier +{ +public: + void arriveAndWait() + { + std::unique_lock lock(mutex); + arrived = true; + cv.notify_all(); + if (!cv.wait_for(lock, std::chrono::seconds(20), [this] { return released; })) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "CAS test barrier timed out waiting for release"); + } + + void waitUntilArrived() + { + std::unique_lock lock(mutex); + if (!cv.wait_for(lock, std::chrono::seconds(20), [this] { return arrived; })) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "CAS test barrier timed out waiting for arrival"); + } + + void release() + { + std::lock_guard lock(mutex); + released = true; + cv.notify_all(); + } + +private: + std::mutex mutex; + std::condition_variable cv; + bool arrived = false; + bool released = false; +}; + +/// Bring up the server-wide blob upload pool (stage-1 §1) if it is not already up, so any test that +/// drives a `ContentAddressedTransaction` commit -- whose `uploadPendingBlobs` fans out on this pool -- +/// finds it initialized. ROBUST (init-if-not-initialized, NOT `call_once`): the raw-lifecycle suite in +/// `gtest_cas_blob_upload_pool.cpp` deliberately shuts the pool down, so a `call_once` helper would fail +/// to bring it back for a later test. A global test-event listener (`gtest_cas_blob_upload_pool_env.cpp`) +/// calls this before every test, which is what makes it robust to test ordering. +inline void ensureBlobUploadPoolForTest(size_t size = 8) +{ + if (!DB::Cas::blobUploadPoolInitializedForTest()) + DB::Cas::initializeBlobUploadPool(size); +} + + +/// Minimal `ContentAddressedSettings` for a direct-construction gtest fixture: sets only +/// `server_root_id` and `scratch_path` (the two values every positional-ctor call site used to pass +/// explicitly) and validates, so the cached enum-valued accessors (`stagingBackend`, `blobHashAlgo`, +/// `partFolderValidate`) are populated from their (default) string settings exactly as the disk-factory +/// path would populate them. Callers that need a non-default setting (e.g. `staging_backend=s3`) apply +/// the override via `settings[ContentAddressedSetting::x] = value;` and re-run `settings.validate()` +/// themselves before constructing. +inline DB::ContentAddressedSettings makeSettingsForTest(const std::string & server_root_id, const std::filesystem::path & scratch_path) +{ + DB::ContentAddressedSettings settings; + settings[DB::ContentAddressedSetting::server_root_id] = server_root_id; + settings[DB::ContentAddressedSetting::scratch_path] = scratch_path.string(); + settings.validate(); + return settings; +} + +/// Run `fn`, expect a DB::Exception with EXACTLY `expected_code` (CORRUPTED_DATA-vs-NOT_IMPLEMENTED +/// is part of the fail-closed contract: an unknown future format must be NOT_IMPLEMENTED, never +/// misreported as corruption). +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} + +/// Build a `LocalObjectStorage` rooted at a fresh, unique temporary directory (one per call). +/// +/// Used by the unit tests that exercise the `Cas::Backend` seam against a real on-disk object storage +/// (the `EmulatedSingleProcess` adapter mode and the capability probe). For `LocalObjectStorage` the +/// object key IS the local path verbatim, so the unique root keeps every test instance isolated even +/// under the parallel gtest runner. +inline DB::ObjectStoragePtr makeLocalObjectStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_unit_" + unique)).string(); + + /// A silently missing root would surface later as a bewildering storage-layer error (e.g. a + /// failed bootstrap LIST), so a setup failure must be loud and named. + std::error_code ec; + std::filesystem::remove_all(root, ec); + if (ec) + throw std::runtime_error("cannot clear test object storage root " + root + ": " + ec.message()); + std::filesystem::create_directories(root, ec); + if (ec) + throw std::runtime_error("cannot create test object storage root " + root + ": " + ec.message()); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +/// Anchor a key under an object storage's own root, for the `Mode::Native` tests. +/// +/// `Mode::Native` uses a key VERBATIM as the physical `LocalObjectStorage` path — no root-prefix +/// mapping the way `EmulatedSingleProcess`'s `emuPath` does (`LocalObjectStorage::writeObject`/ +/// `readObject` pass `object.remote_path` straight through). A bare relative key like `"some/key"` +/// would therefore resolve relative to the TEST PROCESS's working directory rather than the +/// backend's own unique temp root, leaking a real file on disk that outlives the run and that a +/// later run then observes as pre-existing state. Worse for an assertion of ABSENCE: it answers +/// "absent" for a reason that has nothing to do with the property under test. +inline String nativeKeyUnder(const DB::ObjectStoragePtr & storage, const String & suffix) +{ + String root = storage->getCommonKeyPrefix(); + while (!root.empty() && root.back() == '/') + root.pop_back(); + return root + "/" + suffix; +} + +/// ---- on-storage write fixtures (shared by the Pool read/lifecycle/build tests, Tasks 9-13) ---- +/// +/// These produce objects through the SAME codecs the Pool reads — the documented on-storage +/// interface, not white-box pokes — so a test asserts a real round trip across the format boundary. + +/// CityHash128 of bytes, composed into the canonical lowercase-hex id. +inline String hexOf(const String & bytes) +{ + return getHexUIntLowercase(CityHash_v1_0_2::CityHash128(bytes.data(), bytes.size())); +} + +/// The POOL-WIDE streaming content hash (the production `HashingWriteBuffer` convention: chunked +/// CityHash128, block = DBMS_DEFAULT_HASHING_BLOCK_SIZE). Tests that exercise the copy-forward +/// VERIFICATION path must mint blob ids with THIS — the plain `idOf`/`u128Of` below are a +/// test-local convention (fine everywhere hashes are opaque; refused by the verifier). +inline String streamingHexOf(const String & payload) +{ + DB::ReadBufferFromMemory in(payload.data(), payload.size()); + DB::HashingReadBuffer hashing(in); + hashing.ignoreAll(); + return getHexUIntLowercase(hashing.getHash()); +} + +/// The content id of `bytes` as a UInt128 — definitionally consistent with `idOf` (parses the same hex). +inline DB::UInt128 u128Of(const String & bytes) +{ + return DB::Cas::hexToU128(hexOf(bytes)); +} + +/// The content id of `bytes` as a `BlobRef` (CityHash128 — every test pool's default write algo). +inline DB::Cas::BlobRef idOf(const String & bytes) +{ + return DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(bytes))}; +} + +/// Write a Blob object: a fixed-length (blob_header_len) envelope followed by the raw payload, keyed +/// by content. Mirrors what PartWriteTxn::putBlob will emit (Task 11). +inline DB::Cas::BlobRef writeBlobRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const String & payload, + uint64_t blob_header_len, [[maybe_unused]] const DB::UInt128 & domain_id) +{ + const DB::Cas::BlobRef id = idOf(payload); + + /// v3 envelope: domain_id/hash_algo dropped (identity is the content key); the `domain_id` param + /// is kept for call-site compatibility but no longer stamped. + DB::Cas::EnvelopeHeader header; + header.kind = DB::Cas::ObjectKind::Blob; + header.incarnation_tag = DB::UInt128(0x1234); + header.build_id = DB::UInt128(0x5678); + + const String head = DB::Cas::encodeEnvelopeHeader(header, static_cast(blob_header_len)); + backend.putIfAbsent(layout.blobKey(id), head + payload); + return id; +} + +/// Forward declaration: `appendOwnerEvent` (below) calls `registerNamespaceRaw`, which after Task 4 +/// is a no-op (LIST-based discovery needs no explicit registration) defined further down. +inline void registerNamespaceRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns); + +/// Write a part-manifest body object directly via the manifest codec, exactly as PartWriteTxn::stageManifest +/// emits it. Returns the ManifestId. Used by GC fold/retire/fsck tests to stage owner targets. +inline DB::Cas::ManifestId writeManifestRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::Cas::RootNamespace & ns, const DB::Cas::ManifestRef & ref, + const std::vector & entries) +{ + const DB::Cas::ManifestId id{ns, ref}; + DB::Cas::PartManifest body; + body.ref = ref; + body.root_namespace_id = ns; + body.entries = entries; + body.payload_digest = DB::Cas::computePayloadDigest(body); + backend.putIfAbsent(layout.manifestKey(id), + DB::Cas::sealObject(DB::Cas::FormatId::PartManifest, DB::Cas::encodePartManifest(body))); + return id; +} + +/// A blob ManifestEntry referencing `hash` at `path` (size 1, the GC fold counts edges, not bytes). +inline DB::Cas::ManifestEntry blobEntryFor(const String & path, const DB::UInt128 & hash, uint64_t size = 1) +{ + DB::Cas::ManifestEntry e; + e.path = path; + e.placement = DB::Cas::EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; + e.blob_size = size; + return e; +} + +/// Forward declarations of the ref snapshot+log raw fixtures defined further down (they emit the +/// snapshot+log objects GC and recovery actually read); the seeding wrappers below emit through them. +inline void writeRefLogTxnRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RefLogTxn & txn); +namespace fixture +{ + inline DB::Cas::NamespaceLifeId fixtureLife(const DB::Cas::RootNamespace & ns); +} +inline void publishRecoverableCkptForSemanticWrapper( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & txn_id); +inline DB::Cas::RefOp namespaceBirthOp(); +inline std::vector publishCommittedOps( + const String & ref_name, const DB::Cas::ManifestRef & manifest_ref); + +/// One `owner_transition` op built from an optional old/new `RefOwnerBinding` (removal = old set / new +/// unset; add-precommit = new set / old unset; promote = both set naming the SAME manifest). +inline DB::Cas::RefOp ownerTransitionOp( + std::optional old_binding, std::optional new_binding) +{ + DB::Cas::RefOp op; + op.kind = DB::Cas::RefOpKind::OwnerTransition; + op.old_binding = std::move(old_binding); + op.new_binding = std::move(new_binding); + return op; +} + +/// Seed ONE ref-log transaction directly into a table's `_log/` stream -- the snapshot+log replacement +/// for the removed mutable-shard `appendOwnerEvent`. LIST the table's ref prefix, find the greatest +/// existing log/snapshot `ref_sequence` (and whether ANY log or snapshot exists at all), prepend a +/// `namespace_birth` op iff the table has none yet, allocate `txn_id = {writer_epoch=1, greatest+1}`, +/// and write `RefLogTxn{ns, txn_id, ops}` (no `prev_epoch_seal` -- this fixture never crosses an +/// epoch transition) via `writeRefLogTxnRaw`. Returns the allocated `ref_sequence`. +/// `ops` must form a REPLAY-VALID transaction: `fsck`/recovery replay them through the same state +/// machine the writer uses, and the GC edge extractor reads their manifest edges. The bytes are real +/// wire-format (the same codec `Pool`'s recovery reads) -- never hand-rolled. +inline uint64_t appendRefLogSeed( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::Cas::RootNamespace & ns, std::vector ops) +{ + /// Stage B (Task 4-C): resolve to whichever life is ALREADY on record (real production birth or the + /// sentinel), exactly as `writeRefLogTxnRaw` below now does -- otherwise this scan can miss a REAL + /// incarnation's existing log/snap objects, wrongly conclude the table has none, and prepend a second + /// `namespaceBirthOp` on top of a namespace that already has one. + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + const String prefix = layout.namespaceStreamPrefix(life); + uint64_t greatest_seq = 0; + bool any_log_or_snap = false; + String cursor; + while (true) + { + const DB::Cas::ListPage page = backend.list(prefix, cursor, /*limit=*/1000); + for (const DB::Cas::ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (!parsed) + continue; + if (parsed->kind == DB::Cas::RefObjectKind::Log || parsed->kind == DB::Cas::RefObjectKind::Snap) + { + any_log_or_snap = true; + greatest_seq = std::max(greatest_seq, parsed->txn_id.ref_sequence); + } + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + + if (!any_log_or_snap) + ops.insert(ops.begin(), namespaceBirthOp()); + + DB::Cas::RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = DB::Cas::RefTxnId{/*writer_epoch=*/1, /*ref_sequence=*/greatest_seq + 1}; + txn.ops = std::move(ops); + writeRefLogTxnRaw(backend, layout, txn); + return txn.txn_id.ref_sequence; +} + +/// Append ONE `owner_transition` op as a standalone ref-log transaction. `shard` is ignored (the +/// immutable ref model has no per-shard journal); it stays in the signature so existing shard-passing +/// callers compile unchanged. Returns the allocated `ref_sequence`. +inline uint64_t appendOwnerEvent( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::Cas::RootNamespace & ns, uint64_t /*shard*/, + std::optional old_binding, + std::optional new_binding) +{ + return appendRefLogSeed(backend, layout, ns, {ownerTransitionOp(std::move(old_binding), std::move(new_binding))}); +} + +/// Publish a committed ref over `ref_name` (no old unless `old_ref` set). Emits a REPLAY-VALID +/// transaction: an optional owner-removal of the old committed binding, then add-precommit + promote of +/// the new manifest (spec §State Transitions has no direct "add committed" shape). Edges: -1(old)+1(new) +/// or +1(new). Returns the allocated `ref_sequence`. +inline uint64_t publishCommittedTransition( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const String & ref_name, std::optional old_ref, const DB::Cas::ManifestRef & new_ref, + uint64_t /*shard*/ = 0) +{ + std::vector ops; + if (old_ref) + ops.push_back(ownerTransitionOp( + DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Committed, ref_name, *old_ref}, std::nullopt)); + const std::vector commit_ops = publishCommittedOps(ref_name, new_ref); + ops.insert(ops.end(), commit_ops.begin(), commit_ops.end()); + const uint64_t sequence = appendRefLogSeed(backend, layout, ns, std::move(ops)); + publishRecoverableCkptForSemanticWrapper(backend, layout, ns, RefTxnId{1, sequence}); + return sequence; +} + +/// Drop a committed ref (old committed / new none). Edge -1. Returns the allocated `ref_sequence`. +inline uint64_t dropRefTransition( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const String & ref_name, const DB::Cas::ManifestRef & old_ref, uint64_t /*shard*/ = 0) +{ + const uint64_t sequence = appendRefLogSeed(backend, layout, ns, + {ownerTransitionOp(DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Committed, ref_name, old_ref}, std::nullopt)}); + publishRecoverableCkptForSemanticWrapper(backend, layout, ns, RefTxnId{1, sequence}); + return sequence; +} + +/// Add a precommit binding (optional owner-removal of a stale committed manifest, then add-precommit of +/// the new manifest). Edge -1(old)+1(new) or +1(new). `build_id` is dropped (RefLog bindings carry no +/// build_id; build identity lives in `manifest_ref`). Returns the allocated `ref_sequence`. +inline uint64_t addPrecommitTransition( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::UInt128 & /*build_id*/, const String & final_ref_name, std::optional old_ref, + const DB::Cas::ManifestRef & new_ref, uint64_t /*shard*/ = 0) +{ + std::vector ops; + if (old_ref) + ops.push_back(ownerTransitionOp( + DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Committed, final_ref_name, *old_ref}, std::nullopt)); + ops.push_back(ownerTransitionOp( + std::nullopt, DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Precommit, final_ref_name, new_ref})); + const uint64_t sequence = appendRefLogSeed(backend, layout, ns, std::move(ops)); + publishRecoverableCkptForSemanticWrapper(backend, layout, ns, RefTxnId{1, sequence}); + return sequence; +} + +/// Promote a precommit to committed at the SAME manifest_ref (old=Precommit, new=Committed). No edge +/// (net-zero owner move). `build_id` is dropped. Returns the allocated `ref_sequence`. +inline uint64_t promoteTransition( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::UInt128 & /*build_id*/, const String & final_ref_name, const DB::Cas::ManifestRef & ref, + uint64_t /*shard*/ = 0) +{ + const uint64_t sequence = appendRefLogSeed(backend, layout, ns, + {ownerTransitionOp( + DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Precommit, final_ref_name, ref}, + DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Committed, final_ref_name, ref})}); + publishRecoverableCkptForSemanticWrapper(backend, layout, ns, RefTxnId{1, sequence}); + return sequence; +} + +/// Exact-token delete of a manifest body (HEAD then deleteExact). No-op when absent. +inline void deleteManifestBody( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::ManifestId & id) +{ + const String key = layout.manifestKey(id); + const DB::Cas::HeadResult h = backend.head(key); + if (h.exists) + backend.deleteExact(key, h.token); +} + +/// Formerly wrote the namespace into `gc/registry`. Real write helpers now admit the authoritative +/// catalog row themselves, so this legacy fixture hook has no independent registration work. +inline void registerNamespaceRaw( + DB::Cas::Backend & /*backend*/, const DB::Cas::Layout & /*layout*/, const DB::Cas::RootNamespace & /*ns*/) +{ + /// No-op: Task 4 deleted the registry; `cas/ref_catalog` is now the discovery authority. +} + +/// Encode a CAGS document carrying only {round} — everything else defaulted. Callers that only care +/// about this field (e.g. `injectRetire`) use this shorthand. +inline String encodeMinimalGcState(uint64_t round) +{ + DB::Cas::GcState state; + state.round = round; + return DB::Cas::encodeGcState(state); +} + +/// Inject condemned bookkeeping + gc/state directly (bypassing a real GC round) so a test can seed the +/// GC ledger's condemned state at an arbitrary round. Retired-in-snapshot: the condemned entries are +/// seeded the way a real round leaves them — as `kCondemned` sentinel rows inside an adopted fold seal's +/// shard run (there is no separate retired-list object). A synthetic +edge/-edge pair nets each blob to +/// in-degree 0 and a `seed_head` replays the captured token/size so the fold mints the `kCondemned` row. +/// Also sets {round} on gc/state. Entries carry a `condemn_round` (default 0 → uses `round`); callers +/// pass fresh (non-pending) condemns. An empty `entries` set just advances {round}. +inline void injectRetire( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + uint64_t round, uint64_t shard, std::vector entries) +{ + DB::Cas::GcState gc_state; + const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); + if (head.exists) + gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + gc_state.round = round; + + if (!entries.empty()) + { + const uint64_t generation = 1; + const uint64_t attempt = 1; + uint64_t condemn_round = round; + std::unordered_map seeded; + std::vector synth; + synth.reserve(entries.size() * 2); + for (const DB::Cas::RetiredEntry & e : entries) + { + if (e.condemn_round) + condemn_round = e.condemn_round; + seeded.emplace(e.ref, DB::Cas::HeadResult{.exists = true, .size = e.size, .token = e.token, .attributes = {}}); + synth.push_back(DB::Cas::BlobDelta{.ref = e.ref, .source_id = DB::UInt128{1}, .remove = false}); + synth.push_back(DB::Cas::BlobDelta{.ref = e.ref, .source_id = DB::UInt128{1}, .remove = true}); + } + const auto seed_head = [&seeded](const DB::Cas::BlobRef & h) -> std::optional + { + const auto it = seeded.find(h); + return it == seeded.end() ? std::nullopt : std::optional(it->second); + }; + std::vector out; + DB::Cas::foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, generation, attempt, + shard, std::move(synth), out, /*current_round*/0, condemn_round, seed_head, + /*peek_head*/{}, /*confirm_condemned_marker*/{}, + /*out_retired*/nullptr, /*suppress_destructive*/false); + + DB::Cas::CasFoldSeal seal; + seal.generation = generation; + for (DB::Cas::RunRef & r : out) + seal.blob_target_runs.push_back(std::move(r)); + /// Totality over gc_shards so a later real round's graduation/carry reads it zero-I/O. + const uint64_t gc_shards = gc_state.gc_shards ? gc_state.gc_shards : 1; + for (uint64_t s = 0; s < gc_shards; ++s) + seal.condemned_summary[s] = DB::Cas::CondemnedSummary{}; + DB::Cas::CondemnedSummary cs; + cs.condemned_total = entries.size(); + cs.oldest_nonpending_condemn_round = condemn_round; + seal.condemned_summary[shard] = cs; + backend.putIfAbsent(layout.foldSealKey(generation, attempt), DB::Cas::encodeFoldSeal(seal)); + + gc_state.snap_generation = generation; + gc_state.snap_attempt = attempt; + } + + const String state = DB::Cas::encodeGcState(gc_state); + if (!head.exists) + backend.putIfAbsent(layout.gcStateKey(), state); + else + backend.putOverwrite(layout.gcStateKey(), state, head.token); +} + +/// Adopt a fold seal carrying a given per-gc-shard `condemned_summary` (retired-in-snapshot T4) and point +/// gc/state at it (snap_generation / snap_attempt / gc_shards), bypassing a real GC round. If a seal +/// already exists at (generation, attempt) it is overwritten with the new summary (its other fields are +/// preserved); otherwise a fresh minimal seal is created. Read-modify-CAS on gc/state preserves the lease. +/// Used by graduationDue tests to drive the zero-I/O signal directly off a controlled seal. +inline void injectCondemnedSummarySeal( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + uint64_t generation, uint64_t attempt, uint64_t gc_shards, + const std::map & summary) +{ + const String seal_key = layout.foldSealKey(generation, attempt); + DB::Cas::CasFoldSeal seal; + const auto existing = backend.get(seal_key); + if (existing) + seal = DB::Cas::decodeFoldSeal(existing->bytes); + else + seal.parent_generation = generation ? generation - 1 : 0; + seal.generation = generation; + seal.condemned_summary = summary; + const String seal_bytes = DB::Cas::encodeFoldSeal(seal); + if (existing) + backend.putOverwrite(seal_key, seal_bytes, existing->token); + else + backend.putIfAbsent(seal_key, seal_bytes); + + DB::Cas::GcState gc_state; + const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); + if (head.exists) + gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + gc_state.gc_shards = gc_shards; + gc_state.snap_generation = generation; + gc_state.snap_attempt = attempt; + const String state = DB::Cas::encodeGcState(gc_state); + if (!head.exists) + backend.putIfAbsent(layout.gcStateKey(), state); + else + backend.putOverwrite(layout.gcStateKey(), state, head.token); +} + +/// Whether blob `hash` is absent from the backend (its exact-token content object is gone). +inline bool blobAbsent(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash) +{ + return !backend.head(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)})).exists; +} + +/// ONE round that is allowed to RECLAIM -- the name is the point, so that grepping for the tests whose +/// subject is reclamation finds exactly them. +/// +/// The policy is spelled out rather than defaulted so that grepping this name finds every test whose +/// subject is reclamation, and so that a future change to the default cannot silently change what those +/// tests mean. It says the same thing the production default says (`UniversePolicy`); a test whose +/// subject is a SUPPRESSOR passes `StageA_Suppressed` explicitly instead. +inline DB::Cas::RoundReport runRegularRoundReclaiming(DB::Cas::Gc & gc) +{ + return gc.runRegularRound({}, /*allow_steal*/true, DB::Cas::UniversePolicy::Authoritative); +} + +/// Reclaim loop (the canonical retired-cursor pipeline driver): run regular rounds, renewing the store's +/// own heartbeat after each round (`renewWatermarkOnce` — keeps the lease + build-watermark floor +/// current; unrelated to graduation, which paces on GC rounds alone). A blob condemned at round K is +/// deleted by round K+2 (condemn at K -> graduate to delete_pending at K+1, unconditionally -> physical +/// delete at K+2). Returns true as soon as the blob became absent. Reclamation is the whole point of the +/// loop, so every round it drives is an authoritative one. +inline bool runRoundsUntilAbsent( + const DB::Cas::PoolPtr & store, DB::Cas::Gc & gc, DB::Cas::Backend & backend, + const DB::Cas::Layout & layout, const DB::UInt128 & hash, int max_rounds = 8) +{ + for (int i = 0; i < max_rounds; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + if (blobAbsent(backend, layout, hash)) + return true; + } + return blobAbsent(backend, layout, hash); +} + +/// The CURRENT condemned entries for `shard`, read from the adopted fold seal's `blob_target_runs` +/// (retired-in-snapshot T4): the round no longer writes a separate retired-list object — condemned +/// entries RIDE the source-edge run as `kCondemned` sentinel rows at the zero-sentinel key. This reads +/// the seal at (snap_generation, snap_attempt), opens every run for `shard`, and reconstructs the +/// `RetiredEntry` shape (hash from the run key, the rest from the decoded `CondemnedRow`). Empty when +/// gc/state / the seal / the runs are absent. Used by ack-floor tests to assert pending/condemn state. +inline std::vector currentRetiredSet( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t shard) +{ + const auto st = backend.get(layout.gcStateKey()); + if (!st) + return {}; + const DB::Cas::GcState gc_state = DB::Cas::decodeGcState(st->bytes); + if (gc_state.snap_generation == 0) + return {}; + const auto seal_bytes = backend.get(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt)); + if (!seal_bytes) + return {}; + const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(seal_bytes->bytes); + + std::vector out; + for (const DB::Cas::RunRef & run : seal.blob_target_runs) + { + if (run.shard != shard) + continue; + auto r = DB::Cas::openSourceEdgeRun(backend, run.key); + String k; + String p; + while (r.next(k, p)) + { + if (p.empty() || p[0] != DB::Cas::kCondemned) + continue; + DB::Cas::BlobRef ref; + DB::UInt128 source_id{}; + DB::Cas::SourceEdgeKeyCodec::parse(k, ref, source_id); // throws CORRUPTED_DATA on malformed (fail-closed) + const DB::Cas::CondemnedRow row = DB::Cas::decodeCondemnedRow(p); + out.push_back(DB::Cas::RetiredEntry{ + .kind = DB::Cas::ObjectKind::Blob, + .ref = ref, + .token = row.token, + .size = row.size, + .condemn_round = row.condemn_round, + .delete_pending = row.delete_pending, + .marker_confirmed = row.marker_confirmed}); + } + } + return out; +} + +/// True iff ANY gc-shard's adopted-seal run still holds a `kCondemned` row — the ack-floor deletion +/// pipeline is in flight while this is true (retired-in-snapshot T4 replacement for the old +/// "iterate gc/state.retired_refs" probe). `gc_shards` is read from gc/state when 0 is passed. +inline bool anyCondemnedInSeal( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t gc_shards = 0) +{ + const auto st = backend.get(layout.gcStateKey()); + if (!st) + return false; + const DB::Cas::GcState gc_state = DB::Cas::decodeGcState(st->bytes); + const uint64_t shards = gc_shards ? gc_shards : gc_state.gc_shards; + for (uint64_t shard = 0; shard < shards; ++shard) + if (!currentRetiredSet(backend, layout, shard).empty()) + return true; + return false; +} + +/// Displace a blob's incarnation out-of-band (as a racing writer would): GET it, mint a fresh +/// incarnation_tag in its envelope header (preserving header_len + payload), putOverwrite against the +/// current token, and return the NEW token. Used to drive the W-REVALIDATE adopt branch (current token +/// differs from the writer's stale observation). +inline DB::Cas::Token displaceObjectToken( + DB::Cas::Backend & backend, const String & key, DB::Cas::ObjectKind kind) +{ + const auto got = backend.get(key); + if (!got) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "displaceObjectToken: object {} absent", key); + + DB::Cas::EnvelopeHeader header = + DB::Cas::decodeEnvelopeHeader(got->bytes, got->bytes.size(), kind); + /// A fresh, distinct incarnation_tag forces a distinct body so the displaced token differs. + header.incarnation_tag = header.incarnation_tag + DB::UInt128(1); + /// Re-encode at the SAME header length the object was decoded with (the v3 pad target). + const String new_head = DB::Cas::encodeEnvelopeHeader(header, header.header_len); + const String body = new_head + got->bytes.substr(header.header_len); + + return backend.putOverwrite(key, body, got->token).token; +} + +inline DB::Cas::Token displaceBlobToken( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::BlobRef & id) +{ + return displaceObjectToken(backend, layout.blobKey(id), DB::Cas::ObjectKind::Blob); +} + +/// ---- GC-core (Phase 1d) test helpers over the part-manifest model ---- + +/// Open a Pool over `backend`. +/// +/// `gc_fold_max_defer_rounds` defaults to the PoolConfig default (8) -- unchanged behaviour for every +/// existing caller. A test that drives MANY consecutive genuinely-idle `runRegularRound` calls and +/// asserts each one performs a full fold (round/generation advance, trim/sweep/retention) -- exactly +/// what Phase-4 Lever A (spec 2026-07-06-cas-gc-round-skip-unchanged) is designed to skip -- passes 0 +/// here to force fold-every-round (shouldDeferRound's liveness bound: rounds_since_last_fold(0) >= 0 +/// is always true). +inline DB::Cas::PoolPtr openPoolForTest( + std::shared_ptr backend, uint64_t gc_fold_max_defer_rounds = 8) +{ + return DB::Cas::Pool::open(std::move(backend), + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = gc_fold_max_defer_rounds}); +} + +/// Seed the mandatory control objects for an already-existing pool so a subsequent `Pool::open` +/// VALIDATES a restart instead of bootstrapping a fresh one. Recovery/replay tests seed ref-log, +/// snapshot, manifest, or gc-state residue directly into a bare backend; in production such residue +/// only ever exists inside a pool whose FIRST open already minted `_pool_meta` and explicitly +/// initialized `cas/ref_catalog`. Task 7's zero-write bootstrap check (spec §2 [C4][D2], +/// `probePoolBootstrapResidual`) REFUSES to bootstrap over residual data, so a raw restart fixture must +/// establish both mandatory objects itself rather than rely on a production fallback. +/// +/// Idempotent: `createOrValidate` validates an existing `_pool_meta`. The catalog initializer is +/// deliberately narrower: it accepts only a canonical EMPTY conflict, so raw recovery fixtures that +/// have already populated their catalog must mandatory-read and validate it instead of re-running a +/// new-pool initializer. The default `blob_header_len`/`blob_hash_algo` match `PoolConfig`'s defaults, +/// so a later `Pool::open` with a default config validates cleanly. +inline void seedPoolMetaForRestart( + DB::Cas::Backend & backend, const String & pool_prefix = "p", uint64_t gc_shards = 1) +{ + const DB::Cas::Layout layout(pool_prefix); + DB::Cas::PoolMeta::createOrValidate( + backend, layout, /*blob_header_len=*/256, gc_shards, + DB::Cas::BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/true); + if (!backend.get(layout.refCatalogKey())) + DB::Cas::CasRefCatalog::initializeEmptyForNewPool(backend, layout); + else + (void)DB::Cas::CasRefCatalog::read(backend, layout); +} + +/// Write a blob object (envelope + payload) addressed by `hash`, so a HEAD returns a token. The bytes +/// are arbitrary (GC never reads them); the hash is what the manifest entry references. +inline void writeBlobBody( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash, + uint64_t blob_header_len = 256) +{ + DB::Cas::EnvelopeHeader header; + header.kind = DB::Cas::ObjectKind::Blob; + header.incarnation_tag = DB::UInt128(0x1234); + header.build_id = DB::UInt128(0x5678); + const String head = DB::Cas::encodeEnvelopeHeader(header, static_cast(blob_header_len)); + backend.putIfAbsent(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), head + String("x")); +} + +/// Write a raw blob body (payload written verbatim, no envelope) — the raw-body-refinement shape +/// (Phase B): the meta descriptor (via the ops layer below) carries all state, the body carries none. +inline void writeRawBlobBody(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::UInt128 & hash, const String & payload) +{ + backend.casPut(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), payload, std::nullopt); +} + +/// These `UInt128`-hash meta-op wrappers are the pre-mixed-algo 128-bit-only test convenience surface: +/// every existing caller operates on a 128-bit (`cityHash128`) test pool, so the ref is built at +/// `CityHash128` here. The shared `.meta` API (Phase 3 T3) is `BlobRef`-keyed directly and derives its +/// own codec internally — no codec is threaded from here anymore. +inline DB::Cas::BlobRef legacyMetaTestRef(const DB::UInt128 & hash) +{ + return DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; +} + +/// Create a Clean meta descriptor for `hash` directly in a test backend. This setup helper deliberately +/// stays usable before a `Pool` is open; production writes use `putMetaIfAbsent` through the controller. +inline void writeMetaClean(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::UInt128 & hash, uint64_t size) +{ + const DB::Cas::BlobRef ref = legacyMetaTestRef(hash); + backend.putIfAbsent(layout.blobMetaKey(ref), DB::Cas::encodeBlobMeta( + DB::Cas::BlobMeta{.state = DB::Cas::MetaState::Clean, .condemn_round = 0, .size = size})); +} + +/// Transition an existing meta descriptor to Condemned at `condemn_round`, via a read-modify-CAS on +/// its current token (asserts the meta exists — a direct test setup helper, not production code). +inline void condemnMeta(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const DB::UInt128 & hash, uint64_t condemn_round) +{ + const DB::Cas::BlobRef ref = legacyMetaTestRef(hash); + const auto lm = DB::Cas::loadMeta(backend, layout, ref); + ASSERT_TRUE(lm.has_value()); + DB::Cas::BlobMeta c = lm->meta; + c.state = DB::Cas::MetaState::Condemned; + c.condemn_round = condemn_round; + backend.putOverwrite(layout.blobMetaKey(ref), DB::Cas::encodeBlobMeta(c), lm->etag); +} + +/// Load the meta descriptor for `hash` via the shared ops layer (nullopt = absent). +inline std::optional loadMetaForTest(DB::Cas::Backend & backend, + const DB::Cas::Layout & layout, const DB::UInt128 & hash) +{ + return DB::Cas::loadMeta(backend, layout, legacyMetaTestRef(hash)); +} + +/// The latest GC generation (snap_generation pointer in gc/state), or 0 when absent. +inline uint64_t currentGenerationOf(DB::Cas::Backend & backend, const DB::Cas::Layout & layout) +{ + const auto got = backend.get(layout.gcStateKey()); + if (!got) + return 0; + return DB::Cas::decodeGcState(got->bytes).snap_generation; +} + +/// The adopted attempt (snap_attempt pointer in gc/state), or 0 when absent. +inline uint64_t currentAttemptOf(DB::Cas::Backend & backend, const DB::Cas::Layout & layout) +{ + const auto got = backend.get(layout.gcStateKey()); + if (!got) + return 0; + return DB::Cas::decodeGcState(got->bytes).snap_attempt; +} + +/// The current seal's `blob_target_runs` filtered to `shard` (2026-07-02 T0: consumers resolve runs +/// through seal refs, not by key construction). Scans downward from the current generation for the most +/// recent existing fold seal (mirrors `foldCursorOf`'s reasoning); absent => empty. +inline std::vector runsForShard( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t shard) +{ + const uint64_t gen = currentGenerationOf(backend, layout); + const uint64_t attempt = currentAttemptOf(backend, layout); + for (uint64_t g = gen; ; --g) + { + if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + { + const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(got->bytes); + std::vector out; + for (const DB::Cas::RunRef & r : seal.blob_target_runs) + if (r.shard == shard) + out.push_back(r); + return out; + } + if (g == 0) + return {}; + } +} + +/// Stream the sealed in-degree run segments `runs` and count the active source edges (`kEdgeActive` +/// rows) for `ref`. Test-side replacement for the deleted per-blob point query `inDegreeInGeneration` +/// (codecs-v3 phase 5: a `cas_run` is a sequential NDJSON stream with no random access, so a blob's +/// in-degree is recomputed by a full stream-and-count rather than a seek). A condemned / zero-marker +/// row is not an active edge, so it contributes 0 — matching the old point query's semantics. +inline int64_t inDegreeInRuns( + DB::Cas::Backend & backend, const std::vector & runs, const DB::Cas::BlobRef & ref) +{ + int64_t degree = 0; + for (const DB::Cas::RunRef & run : runs) + { + auto r = DB::Cas::openSourceEdgeRun(backend, run.key); + String k; + String p; + while (r.next(k, p)) + { + if (p.empty() || p[0] != DB::Cas::kEdgeActive) + continue; + DB::Cas::BlobRef row_ref; + DB::UInt128 source_id{}; + DB::Cas::SourceEdgeKeyCodec::parse(k, row_ref, source_id); // throws CORRUPTED_DATA on malformed (fail-closed) + if (row_ref == ref) + ++degree; + } + } + return degree; +} + +/// The in-degree of a blob in the current GC generation's sealed run (0 when absent/zeroed). +inline int64_t inDegreeOf(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash) +{ + return inDegreeInRuns(backend, runsForShard(backend, layout, /*shard*/0), legacyMetaTestRef(hash)); +} + +/// The single named entry point for the nonproduction CA shapes this test tree constructs directly, +/// rather than through the production birth/write paths. Every raw fixture below is one of three +/// deliberate divergences from what production can ever produce, gathered here under one name so a +/// future change to any of them has exactly one place to change, not every call site that needs it: +/// 1. `fixtureLife` returns a DETERMINISTIC namespace-derived life identity, never a fresh random +/// mint the way a real birth (`CasRefCatalog::createNamespace`) would -- opaque and catalog-born +/// in production, but every raw fixture below needs to derive the SAME identity a namespace's +/// catalog entry will carry before that entry exists, so its writes and a later read agree on +/// where to look. +/// 2. `admitLive` reaches `Live` with NO `_ckpt` at all. Production only ever reaches `Live` through +/// `completeCreation`, which publishes `_ckpt` FIRST; this shape is kept deliberately, because +/// recovery and failure tests need to exercise a `Live` or `Removing` row missing that authority. +/// 3. `writeRefLogRaw` writes ref-log bytes directly at the resolved fixture identity, bypassing the +/// writer's own birth/append lane entirely -- exercising the on-storage object shape a real writer +/// would emit without driving a real writer to produce it. +namespace fixture +{ + /// The deterministic identity a raw fixture uses for a namespace before any catalog entry exists: + /// a stable hash of the namespace name, so two fixture writes against the same namespace (and a + /// later read) always agree on where to look, without needing a catalog entry to agree through. + /// Production incarnations are always catalog-minted (`CasRefCatalog::createNamespace`); this is + /// deliberately not that, and every raw fixture below depends on it staying stable byte-for-byte. + inline DB::Cas::NamespaceLifeId fixtureLife(const DB::Cas::RootNamespace & ns) + { + UInt128 fixture_incarnation = sipHash128(ns.string().data(), ns.string().size()); + if (fixture_incarnation == 0) + fixture_incarnation = 1; + return DB::Cas::NamespaceLifeId::fromCatalogEntry(ns, fixture_incarnation); + } +} + +/// Resolve the opaque life id that keys this namespace's single fold-coverage row. +inline UInt128 catalogLifeIdForTest( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns) +{ + const std::optional life = + DB::Cas::CasRefCatalog::lifeIfCataloged(backend, layout, ns); + chassert(life.has_value()); + return life->incarnation; +} + +/// Seed the ADOPTED fold seal's catalog-life coverage row for `ns` and point `gc/state` at it, bypassing a +/// real round. This is the durable fact the sweep's §6 deletion premise reads +/// (`CasOrphanManifestSweep.cpp`): `cursor` is the namespace's `last_folded_ref_id`, and a manifest of +/// an epoch-`E` build is deletable only once that cursor sits in an epoch STRICTLY above `E`. +/// `hold`, when set, makes the row classification 4 — the strict grammar `encodeFoldSeal` enforces in +/// both directions, so a hold and a non-4 classification cannot be seeded together. +/// +/// SHARP EDGE, HANDLED HERE SO NO CALLER HAS TO KNOW IT: a fold seal must carry a `condemned_summary` +/// entry for EVERY shard in `0..gc_shards-1`. A later real round adopts this object as its PARENT and +/// throws `CORRUPTED_DATA` — "parent fold seal (generation G, attempt A) lacks a condemned_summary +/// entry for gc-shard N — the seal is not total over gc_shards" — on a seal that is missing one. The +/// symptom is nowhere near the cause: the round fails at fold time, or (if it fails before taking the +/// lease) merely reports `acquired_lease == false`, so a test that seeds a partial seal looks like a +/// leadership problem. This helper fills the map from `gc/state`'s own `gc_shards`, so seeding a +/// coverage row is safe to combine with real rounds. +/// +/// One thing it does NOT do: create `gc/state` in a state a first-ever `Gc` round can take the lease +/// over. `acquireOrRenewLease` only creates-and-owns when `gc/state` is ABSENT, so seeding before the +/// first round makes that round back off. Seed AFTER the first round (passing that round's +/// `currentGenerationOf`/`currentAttemptOf`) when a test drives real rounds. +inline void seedFoldCursorForTest( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + DB::Cas::RefTxnId cursor, std::optional hold = std::nullopt, + uint64_t generation = 1, uint64_t attempt = 1) +{ + DB::Cas::NamespaceLifeId life = fixture::fixtureLife(ns); + const DB::Cas::CasRefCatalog::Snapshot catalog_cut = DB::Cas::CasRefCatalog::read(backend, layout); + const auto catalog_it = std::find_if( + catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), + [&](const DB::Cas::CatalogEntry & entry) { return entry.ns.string() == ns.string(); }); + if (catalog_it == catalog_cut.catalog.entries.end()) + { + DB::Cas::CatalogEntry entry; + entry.ns = ns; + entry.state = DB::Cas::NsState::Live; + entry.incarnation = fixture::fixtureLife(ns).incarnation; + DB::Cas::CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + life = DB::Cas::NamespaceLifeId::fromCatalogEntry(ns, entry.incarnation); + } + else + { + life = DB::Cas::NamespaceLifeId::fromCatalogEntry(ns, catalog_it->incarnation); + } + + const String seal_key = layout.foldSealKey(generation, attempt); + DB::Cas::CasFoldSeal seal; + const auto existing = backend.get(seal_key); + if (existing) + seal = DB::Cas::decodeFoldSeal(existing->bytes); + seal.generation = generation; + + DB::Cas::RefCoverage cov; + cov.classification = hold ? 4 : 2; + cov.last_folded_ref_id = cursor; + cov.hold = hold; + seal.ref_lives[life.incarnation].coverage = cov; + + DB::Cas::GcState gc_state; + const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); + if (head.exists) + gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + + /// Totality over `gc_shards` — see the doc comment's SHARP EDGE note for what throws without it. + const uint64_t gc_shards = gc_state.gc_shards ? gc_state.gc_shards : 1; + for (uint64_t s = 0; s < gc_shards; ++s) + seal.condemned_summary.emplace(s, DB::Cas::CondemnedSummary{}); + + const String seal_bytes = DB::Cas::encodeFoldSeal(seal); + if (existing) + backend.putOverwrite(seal_key, seal_bytes, existing->token); + else + backend.putIfAbsent(seal_key, seal_bytes); + + gc_state.snap_generation = generation; + gc_state.snap_attempt = attempt; + const String state = DB::Cas::encodeGcState(gc_state); + if (!head.exists) + backend.putIfAbsent(layout.gcStateKey(), state); + else + backend.putOverwrite(layout.gcStateKey(), state, head.token); +} + +/// The folded cursor sealed for (ns, shard) by the latest fold seal, or 0 when absent. After a COMPLETE +/// round the gc/state generation pointer is the recheck's COMPLETION generation (G+2 for a round started +/// at G), but the fold seal is written at the FOLD generation (G+1) — recheck writes a completion seal, +/// not a fold seal. So scan downward from the current generation for the most recent existing fold seal. +inline uint64_t foldCursorOf( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, uint64_t shard) +{ + chassert(shard == 0); + const std::optional life = + DB::Cas::CasRefCatalog::lifeIfCataloged(backend, layout, ns); + if (!life) + return 0; + const uint64_t gen = currentGenerationOf(backend, layout); + const uint64_t attempt = currentAttemptOf(backend, layout); + for (uint64_t g = gen; ; --g) + { + if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + { + const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(got->bytes); + const auto it = seal.ref_lives.find(life->incarnation); + /// Snapshot+log ref model: the per-table durable cursor is `last_folded_ref_id` (a RefTxnId). + /// Seeds allocate `writer_epoch = 1`, so the `ref_sequence` is the monotone cursor the seeding + /// wrappers return and tests compare against. + return it != seal.ref_lives.end() ? it->second.coverage.last_folded_ref_id.ref_sequence : 0; + } + if (g == 0) + return 0; + } +} + +/// Set a server root's durable floor (so orphan-sweep eligibility can be driven). After the ack-floor +/// merge the floor rides the mount lease body (`mountKey`), so this seeds a MountLease carrying +/// `{writer_epoch, min_active}` — exactly what `prefixEligible` reads. +inline void setWatermarkMinActive( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const String & server_root_id, + uint64_t writer_epoch, uint64_t min_active) +{ + DB::Cas::MountLease m; + m.server_uuid = DB::UInt128(0); + m.writer_epoch = writer_epoch; + m.min_active = min_active; + m.seq = 1; + m.write_attempt_id = DB::UInt128{1}; + const String key = layout.mountKey(server_root_id); + const DB::Cas::HeadResult h = backend.head(key); + if (h.exists) + backend.putOverwrite(key, DB::Cas::encodeMountLease(m), h.token); + else + backend.putIfAbsent(key, DB::Cas::encodeMountLease(m)); +} + +/// ---- Task 10 ref snapshot+log raw fixtures ---- +/// Mirror the pre-Task-10 `appendOwnerEvent`/`publishRaw` helpers above, but for the new snapshot+log +/// object layout: write a ref-object body directly via the SAME codecs `Pool`'s recovery reads, +/// bypassing the writer's own append lane entirely. Used to seed pre-existing table state before a +/// fresh `Pool` ever touches the namespace (recovery tests), and to control exact keys/bytes +/// (restart-on-vanish tests). + +/// Writes `snapshot` at `_snap/.proto` (create-if-absent). Keys at whichever life the +/// namespace's catalog entry ALREADY names (a prior real birth's random incarnation, or the sentinel +/// if none exists yet -- see `writeRefLogTxnRaw`'s identical note); does NOT itself admit an entry, so +/// a namespace this helper is the ONLY writer for stays exactly as invisible to the catalog as it was +/// before Task 4-C (unchanged from this helper's own pre-existing scope). +inline void writeRefSnapshotRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RefTableSnapshot & snapshot) +{ + const DB::Cas::RootNamespace ns{snapshot.ns}; + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + const String key = layout.refSnapshotKey(life, snapshot.snapshot_id); + backend.putIfAbsent(key, DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(snapshot))); +} + +/// Admits `ns` into the catalog as a `Live` entry, IDEMPOTENTLY (a no-op once `ns` already carries +/// any entry, of any state -- a test that drove one there itself through the real catalog API is left +/// alone). Pinned to the deterministic `fixture::fixtureLife` incarnation, NOT +/// `CasRefCatalog::createNamespace`'s fresh-random mint: every raw fixture below keys its ref-log/ +/// snapshot objects at that SAME derived id, so a randomly minted incarnation would not match them and +/// the fold's own R10 incarnation filter (`{#r10-groupref-alias}`) would drop every one of their keys +/// as belonging to a dead life. +/// +/// All ten raw-write helpers place ref-log bytes at states production's real birth path structurally +/// cannot produce (INV-1 holes, out-of-order ids, a table with no `_ckpt` -- see the Task 4-B map), so +/// they can never route through `createNamespace` and mint a real incarnation of their own. +/// +/// TWO DIVERGENCES from what `createNamespace`/`completeCreation` would produce, both deliberate and +/// both left as-is rather than "fixed": +/// 1. the incarnation is a deterministic namespace-derived fixture id, not a fresh random mint; +/// 2. this entry reaches `Live` with NO `_ckpt` at all, whereas production only ever reaches `Live` +/// through `completeCreation`, which publishes `_ckpt` FIRST (INV-4). Several fixtures exist +/// SPECIFICALLY to build a table with no `_ckpt`, but they must exercise that corruption directly: +/// lifecycle-authoritative recovery correctly rejects a `Live` or `Removing` row without a +/// readable `life_epoch`. Ordinary fixtures use `casAdmitRecoverableEntry` below instead. +inline void casAdmitEntry(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns) +{ + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + for (const CatalogEntry & entry : snap.catalog.entries) + if (entry.ns.string() == ns.string()) + return; /// already admitted -- by an earlier raw write to the same namespace, or by the + /// test itself + CatalogEntry entry; + entry.ns = ns; + entry.state = NsState::Live; + entry.incarnation = fixture::fixtureLife(ns).incarnation; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); +} + +namespace fixture +{ + /// The admit-Live-without-`_ckpt` pattern (divergence 2 above), reachable through the seam. + inline void admitLive(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns) + { + casAdmitEntry(backend, layout, ns); + } +} + +/// Write the checkpoint frontier that makes a raw `Live` fixture a normal recoverable life. Raw logs +/// intentionally do not synthesize `_ckpt`: many tests need the missing-checkpoint corruption shape. +/// A test that invokes lifecycle-authoritative recovery therefore has to state its exact frontier here. +inline void writeRecoverableCkptForRawFixture( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefCkpt & ckpt) +{ + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const auto it = std::find_if( + catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), + [&] (const CatalogEntry & entry) { return entry.ns == ns; }); + if (it == catalog_cut.catalog.entries.end()) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); + const PutResult put = backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(ckpt)); + if (put.outcome != PutOutcome::Done) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' could not publish its checkpoint", ns.string()); +} + +/// Advance an existing recoverable raw fixture's exact checkpoint frontier. This intentionally never +/// creates a missing `_ckpt` or repairs an invalid one: those are distinct raw corruption fixtures. +inline void advanceRecoverableCkptForRawFixture( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & through) +{ + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const auto it = std::find_if( + catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), + [&] (const CatalogEntry & entry) { return entry.ns == ns; }); + if (it == catalog_cut.catalog.entries.end()) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); + const std::optional sample = readCkpt(backend, layout, life); + if (!sample) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' has no checkpoint to advance", ns.string()); + + chooseRecoveryGrounding(*it, sample->ckpt); + if (!sample->ckpt.committed_through || through <= *sample->ckpt.committed_through) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' cannot advance its checkpoint monotonically", ns.string()); + + RefCkpt advanced = sample->ckpt; + advanced.committed_through = through; + if (backend.casPut(layout.refCkptKey(life), encodeRefCkpt(advanced), sample->token).outcome != CasOutcome::Committed) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' could not advance its checkpoint", ns.string()); +} + +/// Replace an existing recoverable raw fixture checkpoint with the caller's complete next state. +/// Unlike `advanceRecoverableCkptForRawFixture`, this does not preserve any field implicitly: callers +/// that model a snapshot or epoch-seal change must name the entire authoritative checkpoint. Missing or +/// invalid current checkpoints stay corruption fixtures and are never repaired here. +inline void replaceRecoverableCkptForRawFixture( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefCkpt & next) +{ + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const auto it = std::find_if( + catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), + [&] (const CatalogEntry & entry) { return entry.ns == ns; }); + if (it == catalog_cut.catalog.entries.end()) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); + const std::optional existing = readCkpt(backend, layout, life); + if (!existing) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' has no checkpoint to replace", ns.string()); + + chooseRecoveryGrounding(*it, existing->ckpt); + chooseRecoveryGrounding(*it, next); + if (next.life_epoch != existing->ckpt.life_epoch) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' cannot replace its checkpoint with a different life epoch", ns.string()); + if (existing->ckpt.committed_through + && (!next.committed_through || *next.committed_through < *existing->ckpt.committed_through)) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' cannot regress its checkpoint frontier", ns.string()); + + if (backend.casPut(layout.refCkptKey(life), encodeRefCkpt(next), existing->token).outcome != CasOutcome::Committed) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "raw recovery fixture for namespace '{}' could not replace its checkpoint", ns.string()); +} + +/// Publish the checkpoint authority a semantic fixture wrapper owes immediately after its durable raw +/// log transaction. Raw writers deliberately do not call this: missing, stale, and malformed `_ckpt` +/// fixtures are meaningful corruption inputs. A semantic wrapper creates the first valid authority or +/// advances the existing exact checkpoint without discarding its snapshot or epoch-seal fields. +inline void publishRecoverableCkptForSemanticWrapper( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & txn_id) +{ + const std::optional life = CasRefCatalog::lifeIfCataloged(backend, layout, ns); + if (!life) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "semantic ref fixture for namespace '{}' was not admitted", ns.string()); + + if (!readCkpt(backend, layout, *life)) + { + const PutResult put = backend.putIfAbsent(layout.refCkptKey(*life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = txn_id, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + })); + if (put.outcome == PutOutcome::Done) + return; + } + + advanceRecoverableCkptForRawFixture(backend, layout, ns, txn_id); +} + +/// Admit an otherwise empty `Live` fixture together with the immutable checkpoint authority that a +/// production-created life already has. This is deliberately a SEPARATE helper from `casAdmitEntry`: +/// raw fixtures that exercise a missing or corrupt `_ckpt` must keep constructing that invalid shape +/// explicitly. The empty frontier is valid because no raw log has been published yet; a fixture that +/// seeds logs instead has to name its own exact `committed_through` through +/// `writeRecoverableCkptForRawFixture`. +inline void casAdmitRecoverableEntry( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + uint64_t life_epoch = 1) +{ + casAdmitEntry(backend, layout, ns); + + const std::optional life = CasRefCatalog::lifeIfCataloged(backend, layout, ns); + if (!life) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "recoverable raw fixture for namespace '{}' was not admitted", ns.string()); + + if (backend.head(layout.refCkptKey(*life)).exists) + return; + + writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = life_epoch, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); +} + +/// Recover from the caller's catalog cut, reading `_ckpt` exactly once for the row in that same cut. +/// Keeping the cut an argument forces raw-fixture consumers to make the immutable authority visible; +/// this helper never resolves the namespace or re-reads the catalog on their behalf. +inline DB::Cas::RecoveredRefTable recoverRefTableDetailedAtCatalogCutForTest( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const CasRefCatalog::Snapshot & catalog_cut, + const DB::Cas::RootNamespace & ns) +{ + std::optional catalog_entry; + const auto it = std::find_if( + catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), + [&] (const CatalogEntry & entry) { return entry.ns == ns; }); + if (it != catalog_cut.catalog.entries.end()) + catalog_entry = *it; + + std::optional ckpt; + if (catalog_entry) + { + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(catalog_entry->ns, catalog_entry->incarnation); + if (const std::optional sample = readCkpt(backend, layout, life)) + ckpt = sample->ckpt; + } + + return recoverRefTableDetailedFromAuthority(backend, layout, catalog_entry, ckpt); +} + +/// Writes `txn` at `_log/` (create-if-absent). Admits `txn.ns` into the catalog first +/// (`casAdmitEntry`, above) -- the fold's universe is catalog-authoritative (Task 4-C), so a raw +/// fixture that skipped this would be invisible to GC/rebuild/fsck no matter what it wrote to `_log`. +/// +/// KEYS AT THE NAMESPACE'S CURRENT CATALOG LIFE, NOT UNCONDITIONALLY AT THE SENTINEL: a test that +/// mixes a REAL birth (`beginPartWrite`/`precommitAdd`, which mints a real random incarnation via +/// `CasRefLedger::resolveNamespaceLife`) with a raw follow-up write to the SAME namespace (a +/// repoint/removal simulation, say) needs this write to land where the real content already lives, not +/// at an unrelated sentinel prefix the fold never reads for that namespace. `casAdmitEntry` above is a +/// no-op once any entry exists, so resolving the catalog life here yields whichever life is ALREADY on +/// record -- the real one if a real birth landed first, the fixture identity if this call is what +/// admitted it (via `casAdmitEntry`, moments ago, in this same function). +inline void writeRefLogTxnRaw( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RefLogTxn & txn) +{ + const DB::Cas::RootNamespace ns{txn.ns}; + casAdmitEntry(backend, layout, ns); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + const String key = layout.refLogKey(life, txn.txn_id); + backend.putIfAbsent(key, DB::Cas::sealObject(DB::Cas::FormatId::RefLog, DB::Cas::encodeRefLogTxn(txn))); +} + +namespace fixture +{ + /// The raw ref-log write pattern (divergence 3 above), reachable through the seam. + inline void writeRefLogRaw(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RefLogTxn & txn) + { + writeRefLogTxnRaw(backend, layout, txn); + } +} + +/// A `Live` snapshot naming exactly `committed` (already-sorted-by-ref_name input expected) with no +/// precommits — the common recovery-fixture shape. +inline DB::Cas::RefTableSnapshot minimalLiveSnapshot( + const String & ns, DB::Cas::RefTxnId snapshot_id, std::vector committed = {}) +{ + DB::Cas::RefTableSnapshot s; + s.ns = ns; + s.snapshot_id = snapshot_id; + s.committed = std::move(committed); + return s; +} + +/// One committed row naming `ref_name` -> `manifest_ref` with `published_at_ms` left at its default +/// (0, unset) — for tests that don't care about the publish stamp. +inline DB::Cas::RefCommittedRow committedRow(const String & ref_name, const DB::Cas::ManifestRef & manifest_ref) +{ + DB::Cas::RefCommittedRow row; + row.ref_name = ref_name; + row.manifest_ref = manifest_ref; + return row; +} + +/// A `namespace_birth` op — the first op any never-born table's first transaction needs. +inline DB::Cas::RefOp namespaceBirthOp() +{ + DB::Cas::RefOp op; + op.kind = DB::Cas::RefOpKind::NamespaceBirth; + return op; +} + +/// An `epoch_seal` op — the record that CLOSES an epoch (INV-2). A seal transaction carries exactly +/// this op and nothing else (grammar enforced by the codec in both directions). The next epoch's first +/// transaction names the seal it consumed in `prev_epoch_seal`, and that back-chain is what lets a fold +/// cross epochs without trusting a listing. +inline DB::Cas::RefOp epochSealOp() +{ + DB::Cas::RefOp op; + op.kind = DB::Cas::RefOpKind::EpochSeal; + return op; +} + +/// Write ONE ref-log transaction at an EXACT id. `appendRefLogSeed` and the wrappers above ALLOCATE +/// ids arithmetically inside writer epoch 1, so anything that needs a chosen id — a gap, an +/// out-of-order arrival, or an epoch CROSSING — writes through here instead. The bytes go through the +/// real codec, so every grammar rule the fold's decoder enforces is enforced here too. +inline void writeTxnAt( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & id, std::vector ops, + std::optional prev_epoch_seal = std::nullopt) +{ + DB::Cas::RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = id; + txn.ops = std::move(ops); + txn.prev_epoch_seal = prev_epoch_seal; + writeRefLogTxnRaw(backend, layout, txn); +} + +/// Close an epoch at exactly `id`. +inline void writeSealAt( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & id, std::optional prev_epoch_seal = std::nullopt) +{ + writeTxnAt(backend, layout, ns, id, {epochSealOp()}, prev_epoch_seal); +} + +/// Publish `ref_name` -> a fresh manifest pinning `blob`, as ONE transaction at exactly `id` +/// (add-precommit + promote, the only shape that reaches a committed owner). `birth` prepends the +/// `namespace_birth` op the table's first transaction owes; `prev_epoch_seal` is required on sequence 1 +/// of every epoch above the namespace's genesis. The manifest's prefix is +/// `{id.writer_epoch, build_sequence}`, so a caller controlling `build_sequence` also controls whether +/// the orphan sweep's watermark considers that manifest eligible. +inline void publishAt( + DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, + const DB::Cas::RefTxnId & id, const String & ref_name, uint64_t build_sequence, const DB::UInt128 & blob, + bool birth = false, std::optional prev_epoch_seal = std::nullopt) +{ + const DB::Cas::ManifestRef mref{.writer_epoch = id.writer_epoch, .build_sequence = build_sequence, + .manifest_ordinal = 1}; + writeBlobBody(backend, layout, blob); + writeManifestRaw(backend, layout, ns, mref, {blobEntryFor("data.bin", blob)}); + + std::vector ops; + if (birth) + ops.push_back(namespaceBirthOp()); + for (const DB::Cas::RefOp & op : publishCommittedOps(ref_name, mref)) + ops.push_back(op); + writeTxnAt(backend, layout, ns, id, std::move(ops), prev_epoch_seal); +} + +/// The two ops a fixture transaction needs to go straight from nothing to a committed ref (spec +/// §State Transitions has no direct "add committed" shape — only precommit -> promote): an +/// `owner_transition` add-precommit followed by an `owner_transition` promote of the SAME +/// (ref_name, manifest_ref). Legal as the tail of one transaction whose earlier ops (if any) left the +/// table `Live` (prepend `namespaceBirthOp()` for a never-born table). +inline std::vector publishCommittedOps(const String & ref_name, const DB::Cas::ManifestRef & manifest_ref) +{ + DB::Cas::RefOp add; + add.kind = DB::Cas::RefOpKind::OwnerTransition; + add.new_binding = DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Precommit, ref_name, manifest_ref}; + + DB::Cas::RefOp promote; + promote.kind = DB::Cas::RefOpKind::OwnerTransition; + promote.old_binding = DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Precommit, ref_name, manifest_ref}; + promote.new_binding = DB::Cas::RefOwnerBinding{DB::Cas::RefOwnerKind::Committed, ref_name, manifest_ref}; + + return {add, promote}; +} + +/// Counts head/get/putIfAbsent per key for op-count assertions (Pillar B / A1 tests). +class CountingBackend : public DB::Cas::InMemoryBackend +{ +public: + /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the + /// overrides below would otherwise shadow them for callers holding a concrete backend type. + using DB::Cas::Backend::get; + using DB::Cas::Backend::getStream; + using DB::Cas::Backend::putIfAbsent; + using DB::Cas::Backend::putOverwrite; + using DB::Cas::Backend::casPut; + + DB::Cas::HeadResult head(const String & key) override + { + { + std::lock_guard lock(count_mutex); + ++head_counts[key]; + ++head_total; + } + return InMemoryBackend::head(key); + } + + std::optional get(const String & key, DB::Cas::Range range) override + { + { + std::lock_guard lock(count_mutex); + ++get_counts[key]; + ++get_total; + /// Record the request-size shape per key so streaming-memory gates (Task 3/4) can assert + /// the resident-memory bound at the seam: a whole-object read (range.whole()) is a + /// violation for a run object; a ranged read tracks its MAX window length per key. + if (range.whole()) + ++whole_get_counts[key]; + else + { + const uint64_t len = range.length.has_value() ? *range.length : 0; + uint64_t & mx = max_ranged_get_len[key]; + mx = std::max(mx, len); + } + } + return InMemoryBackend::get(key, range); + } + + std::optional getStream(const String & key, DB::Cas::Range range) override + { + { + std::lock_guard lock(count_mutex); + ++get_stream_counts[key]; + ++get_stream_total; + } + return InMemoryBackend::getStream(key, range); + } + + DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + { + std::lock_guard lock(count_mutex); + ++list_counts[prefix]; + ++list_total; + } + return InMemoryBackend::list(prefix, cursor, limit); + } + + DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + { + std::lock_guard lock(count_mutex); + ++put_counts[key]; + ++put_total; + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + + /// Counted separately from `putIfAbsent` and `casPut`, for the same reason those two are separate: a + /// replacement conditioned on an expected token is its own op with its own cost. The namespace-file + /// request-profile goldens tell the create path from the replace path on exactly this counter. + DB::Cas::PutResult putOverwrite(const String & key, const String & bytes, const DB::Cas::Token & expected, + const DB::Cas::ObjectMeta & meta) override + { + { + std::lock_guard lock(count_mutex); + ++put_overwrite_counts[key]; + ++put_overwrite_total; + } + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + + /// Counted separately from `putIfAbsent`: a token-CAS is a DIFFERENT op with a different cost, and + /// the `_ckpt` no-op contract ("identical merged body issues no write") is asserted on exactly this + /// counter -- a create-if-absent count would not see the replace path at all. + DB::Cas::CasResult casPut(const String & key, const String & bytes, + const std::optional & expected, const DB::Cas::ObjectMeta & meta) override + { + { + std::lock_guard lock(count_mutex); + ++cas_put_counts[key]; + ++cas_put_total; + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + /// Every ATTEMPTED delete is counted, whatever the backend answers. The destructive gate's tests + /// assert that a suppressed round issues NONE, and an attempt that came back `NotFound` is still an + /// attempt -- counting only successful ones would let a gate that leaks deletes over already-absent + /// keys read as green. + DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + { + { + std::lock_guard lock(count_mutex); + ++delete_counts[key]; + ++delete_total; + } + return InMemoryBackend::deleteExact(key, token); + + } + + uint64_t headCount(const String & key) const { return lookup(head_counts, key); } + uint64_t casPutCount(const String & key) const { return lookup(cas_put_counts, key); } + uint64_t putOverwriteCount(const String & key) const { return lookup(put_overwrite_counts, key); } + uint64_t getCount(const String & key) const { return lookup(get_counts, key); } + uint64_t putCount(const String & key) const { return lookup(put_counts, key); } + uint64_t deleteCount(const String & key) const { return lookup(delete_counts, key); } + uint64_t deleteTotal() const { std::lock_guard lock(count_mutex); return delete_total; } + /// Attempted deletes against any key whose path CONTAINS `substr` — the per-site assertion the + /// destructive-gate tests make ("the generation prune deleted nothing", "the sweep deleted nothing"). + uint64_t deleteCountForKeysContaining(const String & substr) const + { + std::lock_guard lock(count_mutex); + uint64_t total = 0; + for (const auto & [key, n] : delete_counts) + if (key.find(substr) != String::npos) + total += n; + return total; + } + /// Every key this backend was ever asked to delete, in sorted order — so a failing zero-delete + /// assertion names the sites that leaked instead of just reporting a count. + std::vector deletedKeys() const + { + std::lock_guard lock(count_mutex); + std::vector keys; + keys.reserve(delete_counts.size()); + for (const auto & [key, n] : delete_counts) + keys.push_back(key); + return keys; + } + uint64_t getStreamCount(const String & key) const { return lookup(get_stream_counts, key); } + uint64_t listCount(const String & prefix) const { return lookup(list_counts, prefix); } + /// The max ranged-get window length observed for `key` (0 if only whole-object gets, or none). + uint64_t maxRangedGetLen(const String & key) const { return lookup(max_ranged_get_len, key); } + /// How many whole-object gets (range.whole()) hit `key` — nonzero flags a resident-memory + /// violation for a run/seal object that a streaming caller must never read whole. + uint64_t wholeGetCount(const String & key) const { return lookup(whole_get_counts, key); } + /// Every key any counted operation was issued against, plus every LIST prefix, sorted and + /// de-duplicated. A request-profile gate asserts the SET, not only the totals, so a new request the + /// profile does not allow names its own key in the failure instead of moving an anonymous counter. + std::vector touchedKeys() const + { + std::lock_guard lock(count_mutex); + std::vector keys; + for (const std::map * m : + {&head_counts, &get_counts, &put_counts, &put_overwrite_counts, &cas_put_counts, + &get_stream_counts, &list_counts, &delete_counts}) + for (const auto & [key, n] : *m) + keys.push_back(key); + std::sort(keys.begin(), keys.end()); + keys.erase(std::unique(keys.begin(), keys.end()), keys.end()); + return keys; + } + + uint64_t headTotal() const { std::lock_guard lock(count_mutex); return head_total; } + uint64_t getTotal() const { std::lock_guard lock(count_mutex); return get_total; } + uint64_t putTotal() const { std::lock_guard lock(count_mutex); return put_total; } + uint64_t putOverwriteTotal() const { std::lock_guard lock(count_mutex); return put_overwrite_total; } + uint64_t casPutTotal() const { std::lock_guard lock(count_mutex); return cas_put_total; } + uint64_t getStreamTotal() const { std::lock_guard lock(count_mutex); return get_stream_total; } + uint64_t listTotal() const { std::lock_guard lock(count_mutex); return list_total; } + + /// The total number of get + getStream + putIfAbsent operations against any key whose path + /// CONTAINS `substr` (T0 idle-round gate: zero run I/O touches every `.../blob_target/...` key). + uint64_t ioCountForKeysContaining(const String & substr) const + { + std::lock_guard lock(count_mutex); + uint64_t total = 0; + for (const auto & [key, n] : get_counts) + if (key.find(substr) != String::npos) total += n; + for (const auto & [key, n] : get_stream_counts) + if (key.find(substr) != String::npos) total += n; + for (const auto & [key, n] : put_counts) + if (key.find(substr) != String::npos) total += n; + return total; + } + + void resetCounts() + { + std::lock_guard lock(count_mutex); + head_counts.clear(); + get_counts.clear(); + put_counts.clear(); + put_overwrite_counts.clear(); + cas_put_counts.clear(); + get_stream_counts.clear(); + list_counts.clear(); + delete_counts.clear(); + max_ranged_get_len.clear(); + whole_get_counts.clear(); + head_total = get_total = put_total = cas_put_total = get_stream_total = list_total = delete_total = 0; + put_overwrite_total = 0; + + } + +private: + uint64_t lookup(const std::map & m, const String & key) const + { + std::lock_guard lock(count_mutex); + const auto it = m.find(key); + return it == m.end() ? 0 : it->second; + } + + mutable std::mutex count_mutex; + std::map head_counts; + std::map get_counts; + std::map put_counts; + std::map put_overwrite_counts; + std::map cas_put_counts; + std::map get_stream_counts; + std::map list_counts; + std::map delete_counts; + std::map max_ranged_get_len; + std::map whole_get_counts; + uint64_t head_total = 0; + uint64_t get_total = 0; + uint64_t put_total = 0; + uint64_t put_overwrite_total = 0; + uint64_t cas_put_total = 0; + uint64_t get_stream_total = 0; + uint64_t list_total = 0; + uint64_t delete_total = 0; +}; + +/// Records the ORDER of body-PUT / `_ckpt`-CAS operations (so a test can compare indices) and lets a +/// test inject a persistent `Conflict` on one chosen `_ckpt` key -- the same technique +/// `gtest_cas_ref_writer.cpp`'s `RefWriterTestBackend::ckpt_conflict_key`/`ckpt_conflict_count` uses to +/// drive the ledger into `NeedsRecovery`, reproduced here so this suite has no dependency on that file's +/// internal (non-exported) test type. Delegates every operation to `CountingBackend` unchanged, so the +/// per-key counters (`putCount`/`casPutCount`) remain available as the positive control. +class OrderedFaultBackend : public CountingBackend +{ +public: + using CountingBackend::casPut; + using CountingBackend::get; + using CountingBackend::putIfAbsent; + + enum class Op : uint8_t { Put, Cas }; + struct Entry + { + Op op; + String key; + }; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + record(Op::Put, key); + if (fail_put_count > 0 && !fail_put_substr.empty() && key.find(fail_put_substr) != String::npos) + { + --fail_put_count; + throw Poco::TimeoutException("OrderedFaultBackend: simulated PUT response lost, nothing landed"); + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + record(Op::Cas, key); + if (key == fail_cas_key && fail_cas_count > 0) + { + --fail_cas_count; + /// A `Conflict` (not a thrown/ambiguous response): the caller's own re-read-and-merge loop + /// (`publishCkpt`) treats this exactly like a concurrent writer that landed first, and + /// exhausts `MAX_CKPT_CAS_ATTEMPTS` (100) without ever committing -- deterministically, with + /// no wall-clock wait, since the loop is attempt-bounded rather than only deadline-bounded. + return {CasOutcome::Conflict, {}}; + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + + /// Arms a persistent CAS conflict at `key` for the next `count` attempts. + void armCasConflict(const String & key, size_t count) + { + fail_cas_key = key; + fail_cas_count = count; + } + + /// Arms a persistent, never-committed PUT failure for the next `count` `putIfAbsent` calls whose key + /// contains `substr`: the object is never actually written (unlike a real ambiguous response, which + /// may or may not have landed), so the resolve-by-exact-GET a controlled `CasRequestBudget` with + /// `max_attempts = 1` performs always finds the key absent and classifies the attempt a definite, + /// non-`Committed` failure -- deterministically, with no internal retry and no wall-clock wait. + void armPutFailure(const String & substr, int count) + { + fail_put_substr = substr; + fail_put_count = count; + } + + /// The current length of the journal -- a caller's baseline for `indicesFrom` below, so a query can + /// be scoped to "since I last looked" rather than "since the pool opened" (whose earlier entries + /// belong to unrelated setup writes, e.g. the birth transaction's own checkpoint CAS). + size_t journalSize() const + { + std::lock_guard lock(mutex); + return journal.size(); + } + + /// Every index at or after `from` where `op`/`key` matches, in order. + std::vector indicesFrom(Op op, const String & key, size_t from) const + { + std::lock_guard lock(mutex); + std::vector result; + for (size_t i = from; i < journal.size(); ++i) + if (journal[i].op == op && journal[i].key == key) + result.push_back(i); + return result; + } + + /// The first index at or after `from` where `op`/`key` matches, if any. + std::optional firstIndexFrom(Op op, const String & key, size_t from) const + { + const auto indices = indicesFrom(op, key, from); + return indices.empty() ? std::nullopt : std::make_optional(indices.front()); + } + +private: + void record(Op op, const String & key) + { + std::lock_guard lock(mutex); + journal.push_back({op, key}); + } + + mutable std::mutex mutex; + std::vector journal; + String fail_cas_key; + size_t fail_cas_count = 0; + String fail_put_substr; + int fail_put_count = 0; +}; + +/// A backend whose LIST permanently omits every key under a chosen prefix while those keys stay fully +/// readable by exact key -- the lying-store shape observed in production (`0x1430c`/`0x1430d`), and the +/// premise of every arithmetic-walk test: a record a listing never mentions is still THERE, so a walk +/// that computes the id finds it and a walk that enumerates does not. +/// +/// PERMANENT (not nth-call) omission is deliberate: a lying store need not ever recover the key, and the +/// arithmetic walk that finds it anyway is the property under test -- these fixtures are about the walk, +/// not about any one `list` call. +/// +/// Erasing keys from a page cannot disturb pagination: `ListPage::next_cursor` is computed by the base +/// backend before the erase, so the next page still resumes strictly after the last key it returned. +/// +/// Templated on the base so a suite that also needs request COUNTS composes it over `CountingBackend` +/// without a second copy of the hiding rule (which is a rule about what the store may legally do, and +/// must therefore read the same everywhere it is modelled). +template +class HintHoleBackendOn : public Base +{ +public: + /// Hide every key under `prefix` from LIST -- a whole namespace, including objects a later publish + /// adds. + void hidePrefix(const String & prefix) + { + std::lock_guard lock(hide_mutex); + hidden_prefixes.push_back(prefix); + } + + /// Hide exactly one key. Call AFTER seeding: a fixture that allocates ids by listing would + /// otherwise allocate over a hidden record. + void hide(const String & key) + { + std::lock_guard lock(hide_mutex); + hidden_keys.insert(key); + } + + /// Make the store's enumeration omit EXACTLY `keys` and nothing else -- the whole omission set in + /// one call, replacing whatever was hidden before. + /// + /// This is the RustFS defect reproduced as an interface: every one of these keys stays durable and + /// honestly served by `get` / `head` / `putIfAbsent` / `casPut` / `deleteExact`, and only + /// enumeration pretends they are not there. Stating the omission as a SET is what lets a test say + /// the thing the defect report says -- "ids 3 and 4 are invisible while the LATER id 5 is visible" + /// -- in one line, instead of assembling it from repeated single-key calls whose combined effect a + /// reader has to reconstruct. + /// + /// A setter rather than an adder: the omission set is the store's declared behaviour for the rest + /// of the test, so a second call REPLACES it (pass `{}` to stop lying, same as `revealAll`). + void setListOmissions(std::vector keys) + { + std::lock_guard lock(hide_mutex); + hidden_keys.clear(); + hidden_prefixes.clear(); + hidden_keys.insert(keys.begin(), keys.end()); + } + + /// How many LIST pages actually had a key erased. Every test that hides a key asserts this, so a + /// mistyped key cannot let the test pass vacuously -- the hole has to have been SERVED. + size_t holesServed() const + { + std::lock_guard lock(hide_mutex); + return served; + } + + /// The store stops lying: everything hidden is listed again. + void revealAll() + { + std::lock_guard lock(hide_mutex); + hidden_keys.clear(); + hidden_prefixes.clear(); + } + + DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + DB::Cas::ListPage page = Base::list(prefix, cursor, limit); + std::lock_guard lock(hide_mutex); + if (hidden_keys.empty() && hidden_prefixes.empty()) + return page; + const size_t before = page.keys.size(); + std::erase_if(page.keys, [&](const DB::Cas::ListedKey & k) + { + if (hidden_keys.contains(k.key)) + return true; + for (const String & hidden : hidden_prefixes) + if (k.key.starts_with(hidden)) + return true; + return false; + }); + if (page.keys.size() != before) + ++served; + return page; + } + +private: + mutable std::mutex hide_mutex; + std::set hidden_keys; + std::vector hidden_prefixes; + size_t served = 0; +}; + +/// The plain form, over a bare `InMemoryBackend`. +using HintHoleBackend = HintHoleBackendOn; + +/// Stand in for the self-remount that `Pool::reportImpossibleInterference` schedules. That reaction +/// trips the local write fence closed AND schedules a remount; a unit-test Pool runs no background +/// remount (`background_watermark` is off by design there), so without this the fence stays closed and +/// every later mutation is refused at the gate -- which is a test-harness artifact, not the production +/// behaviour. Re-arming directly is the smallest faithful stand-in: it restores writability without the +/// claim machinery and without discarding the cached ref runtimes, so a test can observe what happens +/// AFTER the reaction. It bumps the fence GENERATION, exactly as a real re-arm does. +inline void rearmMountFenceAfterAnomalyForTest(const DB::Cas::PoolPtr & store) +{ + store->armMountFence(DB::UInt128{0, 1}, store->liveWriterEpoch(), store->bootMsNow() + 600000); +} + +/// Delegates the FIRST matching `putIfAbsent` to `CountingBackend` -- so the write actually LANDS -- +/// and only THEN throws an ambiguous exception, modelling "our own PUT committed but its response was +/// lost". Every later call behaves normally, so a caller that retries the SAME (key, bytes) meets its +/// OWN earlier write as the occupant: the exact input the every-attempt rule's adoption arm adjudicates +/// (`slotOccupy` reports `Occupied` with bytes equal to the attempt's own). +/// +/// `key_substr` empty means "the first putIfAbsent of any key"; set it to scope the fault to one key +/// family when the caller drives a whole Pool (whose bootstrap PUTs would otherwise consume the fault). +/// +/// Shared rather than TU-local because two suites need exactly this shape: `gtest_cas_slot_occupy.cpp` +/// pins the primitive's same-call resolve, and `gtest_cas_ref_wedge_every_attempt.cpp` drives the +/// writer's wedge adoption through it. +class LandedButAckLostOnceBackend : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + using CountingBackend::get; + String key_substr; + bool fired = false; + /// Also lose the caller's IMMEDIATE resolve read of the same key, once. Needed only by a caller + /// whose conditional-write layer resolves before reissuing (`putIfAbsentControlled`): without it + /// that resolve proves the object durable inside the very same attempt and reports `Committed`, so + /// no wedge over a DURABLE object can ever form. `slotOccupy` needs no such thing -- it has no + /// retry loop -- which is why this defaults off and this file's original caller is unaffected. + bool lose_resolve_read = false; + + DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + if (!fired && (key_substr.empty() || key.find(key_substr) != String::npos)) + { + fired = true; + CountingBackend::putIfAbsent(key, bytes, meta); /// the write LANDS + if (lose_resolve_read) + fail_get_once_key = key; + throw Poco::TimeoutException("LandedButAckLostOnceBackend: simulated lost PUT response"); + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } + + std::optional get(const String & key, DB::Cas::Range range) override + { + if (!fail_get_once_key.empty() && key == fail_get_once_key) + { + fail_get_once_key.clear(); + throw Poco::TimeoutException( + "LandedButAckLostOnceBackend: simulated lost GET (read response never arrived)"); + } + return CountingBackend::get(key, range); + } + +private: + String fail_get_once_key; +}; + +/// A `CountingBackend` that can fault selected PUTs by key substring (skip the first `fault_skip` +/// matches, then fault the next `fault_count`), and can latch a matching PUT mid-flight. Same class of +/// seam as the wedge tests in `gtest_cas_ref_writer.cpp` use +/// (`fault_key_substr`/`corrupt_key_substr`/`armPutBlock`), narrowed to what the ref-lane tests need. +/// Shared (rather than TU-local) because the chunk-boundary tests and the post-durable install-safety +/// tests need exactly the same seam. +class ChunkFaultBackend : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + using CountingBackend::get; + + /// Unresolved -> a lost-response ambiguity, NOTHING landed; with a single-attempt budget this + /// wedges the lane and a later resolve proves the key ABSENT. + /// LandedThenLost -> our OWN exact bytes land and only the acknowledgement is lost, AND the + /// controller's immediate resolve-before-reissue GET is lost too. Both legs are + /// required to wedge over a DURABLE object: the resolve happens inside the same + /// attempt, so a readable key would prove `Committed` there and no wedge would + /// ever form. Real-world shape: the write succeeded server-side, the connection + /// dropped, and the verification read hit the same transient outage. With a + /// single-attempt budget the lane then wedges over an object that IS durable, so + /// the NEXT flush's `resolveByExactGet` reports `Committed` and drives the + /// wedge-RESOLUTION install (spec §A1 site 2) -- the only mode that reaches it. + /// Definite -> an S3-classified malformed request -> `CasWriteOutcome::DefiniteFailure`. + /// ForeignConflict -> a DIFFERENT object lands at the key, then the response is lost -> the + /// controller's resolve-before-reissue GET observes foreign bytes and throws + /// CORRUPTED_DATA straight out of the PUT (a proven conflict). + enum class Mode { None, Unresolved, LandedThenLost, Definite, ForeignConflict }; + + /// Fault matching is single-threaded during a flush (one leader per table PUTs `_log/`), so these + /// need no lock; set them before driving the flush. + String fault_substr; + Mode mode = Mode::None; + int fault_skip = 0; + int fault_count = 0; + /// One-shot: the next `get` of exactly this key throws, then it is cleared. Armed by + /// `Mode::LandedThenLost` (see above); settable directly for a bare lost-read fault. + String fail_get_once_key; + + std::optional get(const String & key, DB::Cas::Range range) override + { + if (!fail_get_once_key.empty() && key == fail_get_once_key) + { + fail_get_once_key.clear(); + throw Poco::TimeoutException("ChunkFaultBackend: simulated lost GET (read response never arrived)"); + } + return CountingBackend::get(key, range); + } + + DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + if (mode != Mode::None && !fault_substr.empty() && key.find(fault_substr) != String::npos) + { + if (fault_skip > 0) + { + --fault_skip; + } + else if (fault_count > 0) + { + --fault_count; + switch (mode) + { + case Mode::Unresolved: + throw Poco::TimeoutException("ChunkFaultBackend: simulated ambiguous _log PUT (response lost)"); + case Mode::LandedThenLost: + /// The write SUCCEEDS -- byte-for-byte what the caller asked for, through the + /// counting path so the object is indistinguishable from a normal PUT -- and only + /// the acknowledgement is lost. The controller's resolve-before-reissue GET is + /// armed to fail ONCE for this key as well, or it would prove the object durable + /// inside this very attempt and the lane would never wedge; the wedge-resolution + /// GET a flush later then reads it normally. + CountingBackend::putIfAbsent(key, bytes, meta); + fail_get_once_key = key; + throw Poco::TimeoutException("ChunkFaultBackend: object landed; response lost"); + case Mode::Definite: +#if USE_AWS_S3 + throw DB::S3Exception("ChunkFaultBackend: simulated malformed request", + Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); +#else + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "ChunkFaultBackend: DefiniteFailure requires S3 error classification (USE_AWS_S3 off)"); +#endif + case Mode::ForeignConflict: + /// A foreign writer lands DIFFERENT bytes at this exact key; then our response is + /// lost, so resolve-before-reissue GETs foreign bytes -> CORRUPTED_DATA. + CountingBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT")); + throw Poco::TimeoutException("ChunkFaultBackend: foreign different object landed; response lost"); + case Mode::None: + break; + } + } + } + { + std::unique_lock lk(block_mutex); + if (block_armed && !block_substr.empty() && key.find(block_substr) != String::npos) + { + block_entered = true; + block_cv.notify_all(); + /// Bounded (20s) so a wiring bug bounds the wait rather than hanging the whole suite. + block_cv.wait_for(lk, std::chrono::seconds(20), [&] { return !block_armed; }); + } + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } + + void armBlock(const String & substr) + { + std::lock_guard lk(block_mutex); + block_substr = substr; + block_armed = true; + block_entered = false; + } + void awaitBlockEntered() + { + std::unique_lock lk(block_mutex); + /// Bounded (20s): if the latched publisher never reaches its PUT, fail LOUDLY rather than hang. + /// The assertion is load-bearing -- without it a wiring regression that never parks the publisher + /// would let `SnapshotPublisherLatchedAcrossChunks` pass VACUOUSLY (its final re-fire assertion + /// can still hold via a direct, non-coalesced dispatch). + block_cv.wait_for(lk, std::chrono::seconds(20), [&] { return block_entered; }); + ASSERT_TRUE(block_entered) << "latched publisher never entered its blocked PUT within 20s -- " + "coalescing was not exercised"; + } + void releaseBlock() + { + { + std::lock_guard lk(block_mutex); + block_armed = false; + } + block_cv.notify_all(); + } + +private: + std::mutex block_mutex; + std::condition_variable block_cv; + String block_substr; + bool block_armed = false; + bool block_entered = false; +}; + +/// Fault decorator for the condemn-marker gate tests (codex-review triage 2026-07-17 §3.4): while +/// armed, every conditional-write attempt against a blob `.meta` key throws. The request controller +/// exhausts its budget and reports `Unresolved`, so `writeCondemnedMeta` returns false while the round +/// still commits the unconfirmed retired entry. Every other write passes through. Armed by default; +/// disarm (`fail_meta_writes = false`) to model the backend healing. +class MetaWriteFaultBackend : public DB::Cas::InMemoryBackend +{ +public: + /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the + /// overrides below would otherwise shadow them for callers holding a concrete backend type. + using DB::Cas::Backend::get; + using DB::Cas::Backend::getStream; + using DB::Cas::Backend::putIfAbsent; + using DB::Cas::Backend::putOverwrite; + using DB::Cas::Backend::casPut; + + DB::Cas::PutResult putIfAbsent( + const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + if (fail_meta_writes.load() && key.ends_with(".meta")) + throw std::runtime_error("injected fault: blob meta write lost"); + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + DB::Cas::PutResult putOverwrite( + const String & key, const String & bytes, const DB::Cas::Token & expected, + const DB::Cas::ObjectMeta & meta) override + { + if (fail_meta_writes.load() && key.ends_with(".meta")) + throw std::runtime_error("injected fault: blob meta write lost"); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + + DB::Cas::CasResult casPut(const String & key, const String & bytes, + const std::optional & expected, + const DB::Cas::ObjectMeta & meta) override + { + if (fail_meta_writes.load() && key.ends_with(".meta")) + throw std::runtime_error("injected fault: blob meta write lost"); + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + std::atomic fail_meta_writes{true}; +}; + +/// Blocks INSIDE a blob-meta mutation until `release` is called, so a test can hold a real meta job in +/// flight and observe that it got there. `entered` is set before blocking. +/// +/// STARTS DISARMED, and that is load-bearing: the write path itself writes Clean blob meta +/// (`Pool/CasPartWriteTxn.cpp:314`), so a latch that blocked from construction would block the test's +/// own fixture instead of the job under test. Call `arm` only once the fixture is built. +class MetaWriteLatchBackend : public DB::Cas::InMemoryBackend +{ +public: + using DB::Cas::Backend::get; + using DB::Cas::Backend::getStream; + using DB::Cas::Backend::putIfAbsent; + using DB::Cas::Backend::putOverwrite; + using DB::Cas::Backend::casPut; + + std::atomic entered{false}; + + void arm() + { + armed.store(true); + } + + void release() + { + std::lock_guard lock(latch_mutex); + released = true; + latch_cv.notify_all(); + } + + DB::Cas::PutResult putIfAbsent( + const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + waitIfMeta(key); + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + DB::Cas::PutResult putOverwrite( + const String & key, const String & bytes, const DB::Cas::Token & expected, + const DB::Cas::ObjectMeta & meta) override + { + waitIfMeta(key); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + + DB::Cas::CasResult casPut(const String & key, const String & bytes, + const std::optional & expected, + const DB::Cas::ObjectMeta & meta) override + { + waitIfMeta(key); + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + { + waitIfMeta(key); + return InMemoryBackend::deleteExact(key, token); + } + +private: + void waitIfMeta(const String & key) + { + if (!armed.load() || !key.ends_with(".meta")) + return; + entered.store(true); + std::unique_lock lock(latch_mutex); + latch_cv.wait(lock, [this] { return released; }); + } + + std::atomic armed{false}; + std::mutex latch_mutex; + std::condition_variable latch_cv; + bool released = false; +}; + +/// Makes a GC round throw at its outcome-log write -- after the round has scheduled its confirmed-meta +/// delete (`Gc/CasGc.cpp`) and before the round's meta-pool wait. Inherits the `.meta` latch so that +/// job can be held in flight across the throw. Both the fault and the latch start off. +class OutcomeLogFaultBackend : public MetaWriteLatchBackend +{ +public: + using DB::Cas::Backend::get; + using DB::Cas::Backend::putIfAbsent; + + std::atomic fail_outcome_logs{false}; + + DB::Cas::PutResult putIfAbsent( + const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + if (fail_outcome_logs.load() && key.contains("outcomes/")) + return DB::Cas::PutResult{.outcome = DB::Cas::PutOutcome::PreconditionFailed, .token = {}}; + return MetaWriteLatchBackend::putIfAbsent(key, bytes, meta); + } + + std::optional get(const String & key, DB::Cas::Range range) override + { + if (fail_outcome_logs.load() && key.contains("outcomes/")) + return std::nullopt; + return DB::Cas::InMemoryBackend::get(key, range); + } +}; + +/// Wait until a latched job has provably reached the backend. A bounded wait that FAILS rather than +/// hangs: a job that never arrives is a broken fixture, and a test that hangs on it reports nothing. +inline void awaitLatchEntered(MetaWriteLatchBackend & backend) +{ + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (!backend.entered.load()) + { + ASSERT_LT(std::chrono::steady_clock::now(), deadline) + << "no meta job reached the backend latch -- the fixture never scheduled one"; + std::this_thread::yield(); + } +} + +/// Runs a caller-supplied action ONCE, immediately before the named backend call, so a test can make +/// the mount slot change inside a window `MountLeaseKeeper::claim` holds open. Each hook clears +/// itself after firing. +class MountSlotRaceBackend : public DB::Cas::InMemoryBackend +{ +public: + using DB::Cas::Backend::get; + using DB::Cas::Backend::getStream; + using DB::Cas::Backend::putIfAbsent; + using DB::Cas::Backend::putOverwrite; + using DB::Cas::Backend::casPut; + + std::function before_put_if_absent; + std::function before_get; + std::function before_put_overwrite; + + DB::Cas::PutResult putIfAbsent( + const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + fire(before_put_if_absent); + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + std::optional get(const String & key, DB::Cas::Range range) override + { + fire(before_get); + return InMemoryBackend::get(key, range); + } + + DB::Cas::PutResult putOverwrite( + const String & key, const String & bytes, const DB::Cas::Token & expected, + const DB::Cas::ObjectMeta & meta) override + { + fire(before_put_overwrite); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + +private: + static void fire(std::function & hook) + { + if (!hook) + return; + auto once = std::move(hook); + hook = nullptr; + once(); + } +}; + +/// Expect a DB::Exception with EXACTLY `expected_code` AND a message containing `expected_substring`. +/// Needed wherever several distinct branches share one code: the code alone does not identify which +/// one ran, so a test that silently takes the wrong branch would still pass. +template +void expectThrowsCodeWithMessage(int expected_code, const String & expected_substring, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + EXPECT_NE(e.message().find(expected_substring), String::npos) + << "wrong branch: " << e.message(); + } +} + +} diff --git a/src/Disks/tests/gtest_ca_transaction.cpp b/src/Disks/tests/gtest_ca_transaction.cpp new file mode 100644 index 000000000000..324e4ad72bd0 --- /dev/null +++ b/src/Disks/tests/gtest_ca_transaction.cpp @@ -0,0 +1,754 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event CASRefRepoint; +extern const Event CASManifestHead; +} + +namespace DB::ErrorCodes +{ +extern const int NOT_IMPLEMENTED; +extern const int INVALID_STATE; +} + +/// [TXN-ONE-PIPELINE] CA publish-at-commit lock-scope tests. +/// Proves that a freshly-written part's FINAL manifest ref is published only by commit(); the +/// tmp->final rename (moveDirectory) is a pure re-key of the transaction-private overlay and +/// publishes nothing. This inverts the former B151 publish-at-rename behavior. + +namespace +{ + +/// Constructs the storage but deliberately does NOT call `startup()` -- used by tests that need to +/// control when/how startup runs (e.g. injecting a late fault before the atomic publish step). +std::shared_ptr makeUnstartedTxStorage() +{ + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_tx_lockscope_scratch"); + return std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); +} + +std::shared_ptr openTxStorage() +{ + auto storage = makeUnstartedTxStorage(); + storage->startup(); + return storage; +} + +void writeFileTx(DB::IMetadataTransaction & tx, const std::string & path, const std::string & bytes) +{ + auto & ca_tx = dynamic_cast(tx); + auto buf = ca_tx.writeFile(path, 65536, DB::WriteMode::Rewrite, {}); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); +} + +/// Match a manifest entry by its basename (the canonical `path` is the full part-relative path). +const DB::Cas::ManifestEntry * findByName(const std::vector & entries, const std::string & name) +{ + for (const auto & e : entries) + { + const auto slash = e.path.find_last_of('/'); + const std::string base = slash == std::string::npos ? e.path : e.path.substr(slash + 1); + if (base == name) + return &e; + } + return nullptr; +} + +} + +/// Regression for STID 0883 on the CAS write path: an extreme `max_compress_block_size` (the exact +/// 2^63-1 the `04070_no_crash_extreme_compress_block_size` stateless test sets) flows into +/// `writeFile`'s `buf_size` and, unclamped, reaches `Memory::alloc` -- where the allocator's +/// `checkSize` (>= 0x8000000000000000) fires a `LOGICAL_ERROR` and aborts the server. The ordinary +/// MergeTree writers clamp compress-block sizes to 256 MiB; the CAS write buffer must do the same at +/// its own allocation site. Building the buffer with the extreme size must NOT throw/abort, and the +/// resulting buffer must be clamped -- never allocated at the extreme size. +TEST(CASContentWriteBuffer, ExtremeBufferSizeIsClampedNotPassedToAllocator) +{ + const auto scratch = std::filesystem::temp_directory_path() / "ca_extreme_bufsize_scratch"; + + constexpr size_t extreme = 0x7FFFFFFFFFFFFFFFULL; /// 2^63 - 1; unclamped this crashes the allocator + constexpr size_t max_clamped = 256ULL * 1024 * 1024; + + std::unique_ptr buf; + ASSERT_NO_THROW( + buf = std::make_unique( + scratch.string(), + DB::Cas::BlobHashAlgo::CityHash128, + /*buf_size=*/extreme, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/extreme, + [](const std::string &, size_t, const std::string &) {})); + ASSERT_TRUE(buf); + EXPECT_LE(buf->internalBuffer().size(), max_clamped) + << "the CAS write buffer must be clamped, never allocated at the extreme compress-block size"; + + buf.reset(); + std::filesystem::remove_all(scratch); +} + +/// [TXN-ONE-PIPELINE] A freshly-written part is published by commit(), NOT at the tmp->final rename. +/// moveDirectory only re-keys the transaction overlay; the durable ref appears at commit(). +TEST(CASTransactionLockScope, PublishHappensAtCommitNotRename) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0/data.bin", "content-A"); + + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + + /// Re-key only: the final ref is NOT durable yet. + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + + tx->commit(DB::NoCommitOptions{}); + + /// Published by commit(). + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 9u); +} + +TEST(CASTransactionOps, TruncateFileIsNotSupported) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, + [&] { ca_tx.truncateFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", 0); }); +} + +/// [TXN-ONE-PIPELINE] An abandoned transaction (destructed without commit) never published, so the +/// final ref is simply absent — no early-published ref to drop. +TEST(CASTransactionLockScope, AbandonedPartLeavesNoRef) +{ + auto storage = openTxStorage(); + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_3_3_0/data.bin", "abandoned"); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_3_3_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_3_3_0"); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_3_3_0")); /// not published at the rename + /// tx goes out of scope WITHOUT commit(). + } + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_3_3_0")); +} + +/// [TXN-ONE-PIPELINE] commit() publishes the re-keyed part. +TEST(CASTransactionLockScope, RefPublishedByCommit) +{ + auto storage = openTxStorage(); + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_4_4_0/data.bin", "kept"); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_4_4_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_4_4_0"); + tx->commit(DB::NoCommitOptions{}); + } + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_4_4_0")); +} + +/// A committed-ref rename (no staged source) must NOT spuriously publish — it goes via republishRef. +TEST(CASTransactionLockScope, CommittedRefMoveDoesNotSpuriouslyPublish) +{ + auto storage = openTxStorage(); + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_2_2_0/data.bin", "payload"); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_2_2_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0")); + { + auto tx = storage->createTransaction(); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0", "a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_2_2_0"); + tx->commit(DB::NoCommitOptions{}); + } + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_2_2_0")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0")); +} + +/// [TXN-ONE-PIPELINE] B183 migration gate: a scratch ref durably published at the part's own (tmp) +/// BUILD path by a nested sub-storage must be dropped on the staged-source tmp->final finalize, and +/// commit() must publish the AUTHORITATIVE staged manifest (not the scratch content). This mirrors +/// `createTemporaryTextIndexStorage`, which publishes scratch under the new_data_part's STILL-TMP +/// relative path (`MergeTask.cpp` uses `getDataPartStorage().getRelativePath()`) — i.e. the SOURCE of +/// the tmp->final rename, which is exactly what `moveDirectory`'s `dropRefIfPresent(src->refKey())` +/// drops. (The plan's destination-path scenario would not reproduce this: `publishStaging`'s +/// repoint-merge would carry the scratch file forward.) +TEST(CASTransactionLockScope, StagedFinalizeDropsForeignScratchRef) +{ + auto storage = openTxStorage(); + + /// A SEPARATE transaction (the nested text-index sub-storage) durably publishes a committed ref at + /// the tmp BUILD path holding only a scratch file under `text_index_tmp/`. + { + auto scratch_tx = storage->createTransaction(); + writeFileTx(*scratch_tx, "a77/a77a77a7-7777-4777-8777-777777777777/tmp_merge_all_1_1_0/text_index_tmp/scratch.bin", "scratch"); + scratch_tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsDirectory("a77/a77a77a7-7777-4777-8777-777777777777/tmp_merge_all_1_1_0")); + + /// The real part build: stage the authoritative data.bin under the SAME tmp path, then finalize + /// tmp->final. The staged-source finalize drops the foreign scratch ref at the tmp path. + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a77/a77a77a7-7777-4777-8777-777777777777/tmp_merge_all_1_1_0/data.bin", std::string(50000, 'D')); + tx->moveDirectory("a77/a77a77a7-7777-4777-8777-777777777777/tmp_merge_all_1_1_0", "a77/a77a77a7-7777-4777-8777-777777777777/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + /// The published manifest is the authoritative one (has data.bin), not the scratch ref. + const auto ns = storage->liveNamespace("a77a77a7-7777-4777-8777-777777777777"); + const auto resolved = storage->store()->resolveRef(ns, "all_1_1_0"); + ASSERT_TRUE(resolved.has_value()); + const auto manifest = storage->store()->readManifest(resolved->manifest_id); + EXPECT_TRUE(findByName(manifest.entries, "data.bin")); + EXPECT_FALSE(findByName(manifest.entries, "scratch.bin")); + + /// The foreign scratch ref at the tmp build path is gone (dropped, not carried forward). + EXPECT_FALSE(storage->existsDirectory("a77/a77a77a7-7777-4777-8777-777777777777/tmp_merge_all_1_1_0")); +} + +/// [TXN-ONE-PIPELINE] After a tmp->final re-key, a read THROUGH the open transaction resolves the +/// staged content under the FINAL path (read-your-writes), before commit(); the inner-directory +/// overlay is likewise re-keyed and answers under the final path. The staged file lives under an +/// inner projection dir because the directory overlay tracks INNER dirs only — the part dir itself +/// answers `hasInFlightDirectory`=false by contract (removeIfNeeded clean early-return; see +/// `CASWiringInFlight`), so asserting the bare part dir would contradict that invariant. +TEST(CASTransactionLockScope, ReadYourWritesAfterReKey) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + + writeFileTx(*tx, "a33/a33a33a3-3333-4333-8333-333333333333/tmp_insert_all_1_1_0/p.proj/checksums.txt", "the-checksums"); + tx->moveDirectory("a33/a33a33a3-3333-4333-8333-333333333333/tmp_insert_all_1_1_0", "a33/a33a33a3-3333-4333-8333-333333333333/all_1_1_0"); + + /// The overlay answers the final path before commit (read-your-writes). + auto buf = ca_tx.tryReadFileInFlight("a33/a33a33a3-3333-4333-8333-333333333333/all_1_1_0/p.proj/checksums.txt", DB::ReadSettings{}, std::nullopt); + ASSERT_NE(buf, nullptr); + std::string got; + DB::readStringUntilEOF(got, *buf); + EXPECT_EQ(got, "the-checksums"); + /// The inner-directory overlay is re-keyed too and resolves under the final path. + EXPECT_TRUE(ca_tx.hasInFlightDirectory("a33/a33a33a3-3333-4333-8333-333333333333/all_1_1_0/p.proj")); +} + +/// [TXN-ONE-PIPELINE] Program order in the overlay: create -> delete -> create leaves the file PRESENT +/// (no delayed delete fires after the later create); delete of a staged file makes it absent to reads. +TEST(CASTransactionLockScope, OverlayProgramOrder) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + + writeFileTx(*tx, "a55/a55a55a5-5555-4555-8555-555555555555/tmp_insert_all_1_1_0/a.txt", "v1"); + ca_tx.unlinkFile("a55/a55a55a5-5555-4555-8555-555555555555/tmp_insert_all_1_1_0/a.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + EXPECT_EQ(ca_tx.tryReadFileInFlight("a55/a55a55a5-5555-4555-8555-555555555555/tmp_insert_all_1_1_0/a.txt", DB::ReadSettings{}, std::nullopt), nullptr); + + writeFileTx(*tx, "a55/a55a55a5-5555-4555-8555-555555555555/tmp_insert_all_1_1_0/a.txt", "v2"); + tx->moveDirectory("a55/a55a55a5-5555-4555-8555-555555555555/tmp_insert_all_1_1_0", "a55/a55a55a5-5555-4555-8555-555555555555/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + ASSERT_TRUE(storage->existsFile("a55/a55a55a5-5555-4555-8555-555555555555/all_1_1_0/a.txt")); + EXPECT_EQ(storage->getFileSize("a55/a55a55a5-5555-4555-8555-555555555555/all_1_1_0/a.txt"), 2u); +} + +/// [02941 root-cause] A carried-forward projection sidecar (createHardLink from a COMMITTED source part +/// into a mutated tmp part) must be readable through the transaction's in-flight read path BOTH at the +/// tmp build path (loadProjections runs here during MutateTask finalize) AND after the tmp->final re-key. +/// This is the exact sequence MATERIALIZE PROJECTION drives on a part that already has the projection. +/// If the in-flight read returns empty, the mutated part's in-memory projection sub-part loads with 0 +/// marks (the 02941 "Empty marks file: 0, must be: 144" corruption on a same-session projection SELECT). +TEST(CASTransactionLockScope, InFlightReadCarriedForwardProjectionSidecar) +{ + auto storage = openTxStorage(); + + /// 1. Commit a source part with a small INLINE projection sidecar (marks-like) + a blob. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b01/b01b01b0-0101-4101-8101-010101010101/tmp_insert_all_1_1_0/data.bin", "the-main-data-bytes"); + writeFileTx(*tx, "b01/b01b01b0-0101-4101-8101-010101010101/tmp_insert_all_1_1_0/aaaa.proj/data.cmrk4", "PROJMARKS9"); + tx->moveDirectory("b01/b01b01b0-0101-4101-8101-010101010101/tmp_insert_all_1_1_0", "b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsFile("b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0/aaaa.proj/data.cmrk4")); + + /// 2. Mutation: build a new tmp part + carry the projection sidecar forward via createHardLink. + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + writeFileTx(*tx, "b01/b01b01b0-0101-4101-8101-010101010101/tmp_mut_all_1_1_0_2/data.bin", "mutated-main-data"); + ca_tx.createHardLink("b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0/aaaa.proj/data.cmrk4", + "b01/b01b01b0-0101-4101-8101-010101010101/tmp_mut_all_1_1_0_2/aaaa.proj/data.cmrk4"); + + /// 2a. loadProjections timing: read the carried sidecar in-flight at the TMP build path (pre-re-key). + { + auto buf = ca_tx.tryReadFileInFlight("b01/b01b01b0-0101-4101-8101-010101010101/tmp_mut_all_1_1_0_2/aaaa.proj/data.cmrk4", DB::ReadSettings{}, std::nullopt); + ASSERT_NE(buf, nullptr) << "carried-forward projection sidecar not readable in-flight at the tmp path"; + std::string got; DB::readStringUntilEOF(got, *buf); + EXPECT_EQ(got, "PROJMARKS9"); + EXPECT_EQ(ca_tx.tryGetInFlightFileSize("b01/b01b01b0-0101-4101-8101-010101010101/tmp_mut_all_1_1_0_2/aaaa.proj/data.cmrk4"), + std::optional(10)); + } + + /// 2b. After the tmp->final re-key (Phase 1), the sidecar must still resolve at the final path. + tx->moveDirectory("b01/b01b01b0-0101-4101-8101-010101010101/tmp_mut_all_1_1_0_2", "b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0_2"); + { + auto buf = ca_tx.tryReadFileInFlight("b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0_2/aaaa.proj/data.cmrk4", DB::ReadSettings{}, std::nullopt); + ASSERT_NE(buf, nullptr) << "carried-forward projection sidecar not readable in-flight at the final path after re-key"; + std::string got; DB::readStringUntilEOF(got, *buf); + EXPECT_EQ(got, "PROJMARKS9"); + } + + /// 3. And after commit it is durable + correct. + tx->commit(DB::NoCommitOptions{}); + EXPECT_EQ(storage->getFileSize("b01/b01b01b0-0101-4101-8101-010101010101/all_1_1_0_2/aaaa.proj/data.cmrk4"), 10u); +} + +/// [TXN-ONE-PIPELINE] Audit 5: on a commit, only refs this commit CREATED are eligible for rollback; +/// a repoint of an already-existing ref is NEVER dropped as compensation. `publishStaging` writes a +/// `CommitOutcome` with `created=false` for the repoint path (a committed ref exists), so `commit`'s +/// rollback loop skips this slot and the pre-existing part survives with its content carried forward. +TEST(CASTransactionLockScope, CommitRollbackSparesPreexistingRef) +{ + auto storage = openTxStorage(); + + /// Pre-existing committed part. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a88/a88a88a8-8888-4888-8888-888888888888/tmp_insert_all_1_1_0/data.bin", "orig"); + tx->moveDirectory("a88/a88a88a8-8888-4888-8888-888888888888/tmp_insert_all_1_1_0", "a88/a88a88a8-8888-4888-8888-888888888888/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsDirectory("a88/a88a88a8-8888-4888-8888-888888888888/all_1_1_0")); + + /// A standalone write on the committed part repoints the EXISTING ref. Even if a later part in the + /// same commit were to fail, the existing ref must survive: `publishStaging` writes `created=false` + /// for this slot, so `commit`'s rollback loop skips it and it is never dropped on the error path. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a88/a88a88a8-8888-4888-8888-888888888888/all_1_1_0/metadata_version.txt", "1"); + tx->commit(DB::NoCommitOptions{}); + } + EXPECT_TRUE(storage->existsDirectory("a88/a88a88a8-8888-4888-8888-888888888888/all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a88/a88a88a8-8888-4888-8888-888888888888/all_1_1_0/data.bin")); /// original content carried forward +} + +/// Plan 2d: a small eager metadata file (checksums.txt) is staged INLINE — it rides the single tree +/// object (one-GET part open) — while per-column data (data.bin) stays a standalone Blob (preserving +/// column-read selectivity). The inlined file is still readable through the normal read path. +TEST(CASTransactionInlining, EagerFileInlinedDataBinBlobbed) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + writeFileTx(*tx, "a99/a99a99a9-9999-4999-8999-999999999999/tmp_insert_all_1_1_0/checksums.txt", "the-checksums"); + writeFileTx(*tx, "a99/a99a99a9-9999-4999-8999-999999999999/tmp_insert_all_1_1_0/data.bin", std::string(50000, 'D')); + tx->moveDirectory("a99/a99a99a9-9999-4999-8999-999999999999/tmp_insert_all_1_1_0", "a99/a99a99a9-9999-4999-8999-999999999999/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + /// Resolve the published part to its manifest and inspect placements (the Pool read API, as in + /// gtest_cas_pool.cpp: resolveRef -> readManifest). + const auto ns = storage->liveNamespace("a99a99a9-9999-4999-8999-999999999999"); + const auto resolved = storage->store()->resolveRef(ns, "all_1_1_0"); + ASSERT_TRUE(resolved.has_value()); + const DB::Cas::PartManifest manifest = storage->store()->readManifest(resolved->manifest_id); + const auto & entries = manifest.entries; + + const auto * checksums = findByName(entries, "checksums.txt"); + const auto * databin = findByName(entries, "data.bin"); + ASSERT_TRUE(checksums && databin); + EXPECT_EQ(checksums->placement, DB::Cas::EntryPlacement::Inline); + EXPECT_EQ(checksums->inline_bytes, "the-checksums"); + EXPECT_EQ(databin->placement, DB::Cas::EntryPlacement::Blob); + + /// And the inlined file is still readable through the normal read path. + EXPECT_EQ(storage->getFileSize("a99/a99a99a9-9999-4999-8999-999999999999/all_1_1_0/checksums.txt"), 13u); +} + +/// all-tree-part-files Task 4: a standalone +/// write of ONE file onto an ALREADY-COMMITTED part must carry every other file of that part forward +/// (a repoint, Task 3), never replace the manifest with just the touched file. +TEST(CASTransactionRepoint, StandaloneWriteOnCommittedPartRepoints) +{ + auto storage = openTxStorage(); + + /// 1. Write a part (checksums.txt inline + data.bin blob) through a normal transaction; commit. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b02/b02b02b0-0202-4202-8202-020202020202/tmp_insert_all_1_1_0/checksums.txt", "old-checksums"); + writeFileTx(*tx, "b02/b02b02b0-0202-4202-8202-020202020202/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + tx->moveDirectory("b02/b02b02b0-0202-4202-8202-020202020202/tmp_insert_all_1_1_0", "b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsFile("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/checksums.txt")); + ASSERT_TRUE(storage->existsFile("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/data.bin")); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + + /// 2. New transaction: standalone write of checksums.txt onto the ALREADY-COMMITTED part. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/checksums.txt", "new-checksums-longer"); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. The new content is served, the untouched file is carried forward unchanged, exactly one + /// repoint fired, and an independent fsck reachability walk finds nothing dangling. + EXPECT_EQ(storage->getFileSize("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/checksums.txt"), 20u); + EXPECT_EQ(storage->getFileSize("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/data.bin"), 14u) + << "carry-forward: the untouched file must survive a standalone write on the same part"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + + const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); + EXPECT_EQ(rep.dangling, 0u); +} + +/// Task 9 coverage gap (closed here, folded in from the T8 review): ONE uncommitted transaction that +/// BOTH writes a file and unlinks a DIFFERENT file of the SAME already-committed part must resolve to +/// exactly ONE repoint carrying the write, the removal, AND every untouched file forward together -- +/// not two independent repoints, and not a lost update from one staged change clobbering the other. +/// `publishStaging`'s Task 4/8 merge already handles `st.entries` and `st.content_removed` together +/// (both conditions can be true on the same staging); this pins that the combined shape actually works +/// end to end through the real transaction, not just through each half in isolation. +TEST(CASTransactionRepoint, CombinedWriteAndUnlinkSameTxnRepointsOnce) +{ + auto storage = openTxStorage(); + + /// 1. Commit a part with three files. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b03/b03b03b0-0303-4303-8303-030303030303/tmp_insert_all_1_1_0/checksums.txt", "old-checksums"); + writeFileTx(*tx, "b03/b03b03b0-0303-4303-8303-030303030303/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + writeFileTx(*tx, "b03/b03b03b0-0303-4303-8303-030303030303/tmp_insert_all_1_1_0/txn_version.txt", + "creation_tid: (1,1,00000000-0000-0000-0000-000000000000)"); + tx->moveDirectory("b03/b03b03b0-0303-4303-8303-030303030303/tmp_insert_all_1_1_0", "b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsFile("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/txn_version.txt")); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + + /// 2. ONE transaction: write checksums.txt (new bytes) AND unlink txn_version.txt (a DIFFERENT + /// file of the same part) -- must resolve to exactly one repoint carrying both changes plus the + /// untouched data.bin. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/checksums.txt", "new-checksums-longer"); + tx->unlinkFile("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. The written file is updated, the unlinked file is honestly gone, the untouched file survives + /// (carry-forward), exactly ONE repoint fired (not two, not zero), and fsck finds nothing dangling. + EXPECT_EQ(storage->getFileSize("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/checksums.txt"), 20u); + EXPECT_FALSE(storage->existsFile("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/txn_version.txt")); + EXPECT_EQ(storage->getFileSize("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/data.bin"), 14u) + << "carry-forward: the untouched file must survive a combined write+unlink on the same part"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1) + << "one uncommitted transaction combining a write and an unlink must resolve to exactly one repoint"; + + const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); + EXPECT_EQ(rep.dangling, 0u); +} + +/// all-tree-part-files Task 6: the mutable- +/// per-part-file branch is deleted from `writeFile` -- uuid.txt/metadata_version.txt/txn_version.txt +/// now flow down the ordinary content path, landing in the manifest like any other file. +TEST(CASTransactionAllTree, BuildTimeSidecarsLandInManifest) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b04/b04b04b0-0404-4404-8404-040404040404/tmp_insert_all_1_1_0/uuid.txt", "part-uuid-bytes"); + writeFileTx(*tx, "b04/b04b04b0-0404-4404-8404-040404040404/tmp_insert_all_1_1_0/metadata_version.txt", "3"); + writeFileTx(*tx, "b04/b04b04b0-0404-4404-8404-040404040404/tmp_insert_all_1_1_0/txn_version.txt", "creation_tid: (1,1,00000000-0000-0000-0000-000000000000)"); + writeFileTx(*tx, "b04/b04b04b0-0404-4404-8404-040404040404/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + tx->moveDirectory("b04/b04b04b0-0404-4404-8404-040404040404/tmp_insert_all_1_1_0", "b04/b04b04b0-0404-4404-8404-040404040404/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + const auto ns = storage->liveNamespace("b04b04b0-0404-4404-8404-040404040404"); + const auto resolved = storage->store()->resolveRef(ns, "all_1_1_0"); + ASSERT_TRUE(resolved.has_value()); + + const DB::Cas::PartManifest manifest = storage->store()->readManifest(resolved->manifest_id); + const auto & entries = manifest.entries; + const auto * uuid_entry = findByName(entries, "uuid.txt"); + const auto * meta_version_entry = findByName(entries, "metadata_version.txt"); + const auto * txn_version_entry = findByName(entries, "txn_version.txt"); + ASSERT_TRUE(uuid_entry && meta_version_entry && txn_version_entry) + << "all three sidecar files must land in the manifest as ordinary tree entries"; + EXPECT_EQ(uuid_entry->placement, DB::Cas::EntryPlacement::Inline); + EXPECT_EQ(meta_version_entry->placement, DB::Cas::EntryPlacement::Inline); + EXPECT_EQ(txn_version_entry->placement, DB::Cas::EntryPlacement::Inline); + EXPECT_EQ(meta_version_entry->inline_bytes, "3"); + + /// And they are readable through the normal read path — Task 9 deleted the ForceFresh special + /// case these reads used to go through; they now resolve purely via the manifest view like any + /// other entry (existsFile / getFileSize / tryGetInManifestBytes). + EXPECT_TRUE(storage->existsFile("b04/b04b04b0-0404-4404-8404-040404040404/all_1_1_0/metadata_version.txt")); + EXPECT_EQ(storage->getFileSize("b04/b04b04b0-0404-4404-8404-040404040404/all_1_1_0/metadata_version.txt"), 1u); +} + +/// A standalone one-shot write of txn_version.txt onto an ALREADY-COMMITTED part (the MVCC creation- +/// CSN fill-in / removal-TID rewrite shape) must repoint (Task 4), never orphan the rest of the part. +TEST(CASTransactionAllTree, CommittedTxnVersionStoreRepoints) +{ + auto storage = openTxStorage(); + + /// 1. Commit a part WITHOUT txn_version.txt. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b05/b05b05b0-0505-4505-8505-050505050505/tmp_insert_all_1_1_0/checksums.txt", "cs-bytes"); + writeFileTx(*tx, "b05/b05b05b0-0505-4505-8505-050505050505/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + tx->moveDirectory("b05/b05b05b0-0505-4505-8505-050505050505/tmp_insert_all_1_1_0", "b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_FALSE(storage->existsFile("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt")); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + + /// 2. A single-op transaction writes ONLY txn_version.txt onto the already-committed part (mirrors + /// the MVCC one-shot autocommit shape: no other file touched in this transaction). + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt", "creation_tid: (2,2,00000000-0000-0000-0000-000000000000)"); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. Exactly one repoint; the new file is served; the original files are intact (carry-forward). + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + EXPECT_TRUE(storage->existsFile("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt")); + EXPECT_EQ(storage->getFileSize("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt"), 56u); + EXPECT_EQ(storage->getFileSize("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/checksums.txt"), 8u); + EXPECT_EQ(storage->getFileSize("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/data.bin"), 14u); + + const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); + EXPECT_EQ(rep.dangling, 0u); +} + +/// all-tree-part-files Task 8 (B123 evolution): +/// a lone surgical unlink of ONE committed content file (not followed by a whole-part removal in the +/// same transaction — the ATTACH `removeVersionMetadata` shape) must actually delete the file via a +/// repoint-remove, closing the pre-Task-8 fail-open (unlinkFile of a committed content file used to be +/// an unconditional no-op). +TEST(CASTransactionRemove, SurgicalUnlinkRepoints) +{ + auto storage = openTxStorage(); + + /// 1. Commit a part with txn_version.txt among its files. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b06/b06b06b0-0606-4606-8606-060606060606/tmp_insert_all_1_1_0/checksums.txt", "cs-bytes"); + writeFileTx(*tx, "b06/b06b06b0-0606-4606-8606-060606060606/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + writeFileTx(*tx, "b06/b06b06b0-0606-4606-8606-060606060606/tmp_insert_all_1_1_0/txn_version.txt", + "creation_tid: (1,1,00000000-0000-0000-0000-000000000000)"); + tx->moveDirectory("b06/b06b06b0-0606-4606-8606-060606060606/tmp_insert_all_1_1_0", "b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsFile("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/txn_version.txt")); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + + /// 2. A single-op transaction unlinks ONLY txn_version.txt on the already-committed part (mirrors + /// ATTACH's removeVersionMetadata: no dir-drop in the same transaction). + { + auto tx = storage->createTransaction(); + tx->unlinkFile("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. The file is honestly gone, the untouched files survive (carry-forward), exactly one repoint + /// fired, and an independent fsck reachability walk finds nothing dangling. + EXPECT_FALSE(storage->existsFile("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/txn_version.txt")); + EXPECT_EQ(storage->getFileSize("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/checksums.txt"), 8u); + EXPECT_EQ(storage->getFileSize("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/data.bin"), 14u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + + const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); + EXPECT_EQ(rep.dangling, 0u); +} + +/// all-tree-part-files Task 8 (B123 evolution, spec §6): the DOMINANT CA removal path — the MergeTree +/// fast-removal shape that unlinks every part file one by one and THEN calls removeDirectory — must +/// stay exactly one ref-drop and pay ZERO repoints. The per-file removal marks staged by the unlink +/// storm are superseded by the ref-drop, not individually repointed. +TEST(CASTransactionRemove, UnlinkStormThenDirDropIsOneRefDrop) +{ + auto storage = openTxStorage(); + + /// 1. Commit a part with three files. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b07/b07b07b0-0707-4707-8707-070707070707/tmp_insert_all_1_1_0/checksums.txt", "cs-bytes"); + writeFileTx(*tx, "b07/b07b07b0-0707-4707-8707-070707070707/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + writeFileTx(*tx, "b07/b07b07b0-0707-4707-8707-070707070707/tmp_insert_all_1_1_0/txn_version.txt", + "creation_tid: (1,1,00000000-0000-0000-0000-000000000000)"); + tx->moveDirectory("b07/b07b07b0-0707-4707-8707-070707070707/tmp_insert_all_1_1_0", "b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsDirectory("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0")); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + + /// 2. The MergeTree fast-removal shape (IMergeTreeDataPart::remove, B123): unlink every file + /// one-by-one, THEN removeDirectory the part — all in one transaction. + { + auto tx = storage->createTransaction(); + tx->unlinkFile("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0/checksums.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->unlinkFile("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0/data.bin", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->unlinkFile("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->removeDirectory("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. The whole part is gone via the single ref-drop; the storm of marks never repointed anything. + EXPECT_FALSE(storage->existsDirectory("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0")); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before) + << "unlink-storm-then-dir-drop must supersede the marks, not repoint per file"; + + const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); + EXPECT_EQ(rep.dangling, 0u); +} + +/// Task 22 (URF plan phase 7): the MergeTree fast-removal shape's per-file ForceFresh proof is +/// memoized per (transaction, ref) in `unlinkFile` — the first unlink's `ForceFresh` `getView` re-proves +/// the manifest body with one HEAD; the rest of the burst reuses that proof via `CachedForLoad`. This +/// pins the HEAD-count side of `UnlinkStormThenDirDropIsOneRefDrop` above (which already pins the +/// repoint count): before this memoization, N unlinks of the same part paid N manifest-body HEADs; now +/// the whole storm-then-drop transaction pays exactly one. +TEST(CASTransactionRemove, UnlinkStormMemoizesOneForceFreshHead) +{ + auto storage = openTxStorage(); + + /// 1. Commit a part with three files. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_all_1_1_0/checksums.txt", "cs-bytes"); + writeFileTx(*tx, "b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_all_1_1_0/data.bin", "the-data-bytes"); + writeFileTx(*tx, "b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_all_1_1_0/txn_version.txt", + "creation_tid: (1,1,00000000-0000-0000-0000-000000000000)"); + tx->moveDirectory("b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_all_1_1_0", "b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->existsDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0")); + + const uint64_t heads_before = ProfileEvents::global_counters[ProfileEvents::CASManifestHead].load(); + + /// 2. The MergeTree fast-removal shape: unlink every file one-by-one, THEN removeDirectory — all + /// in ONE transaction (mirrors UnlinkStormThenDirDropIsOneRefDrop above). + { + auto tx = storage->createTransaction(); + tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/checksums.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/data.bin", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + tx->removeDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + + /// 3. The whole part is gone, and the three-file unlink storm paid exactly ONE manifest-body HEAD + /// (the first unlink's ForceFresh proof) — not three. removeDirectory clears the staged removal + /// marks, so publishStaging's own (unmemoized) ForceFresh getView never fires for this ref either + /// (see UnlinkStormThenDirDropIsOneRefDrop's zero-repoints assertion above). + EXPECT_FALSE(storage->existsDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0")); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASManifestHead].load(), heads_before + 1) + << "unlink-storm-then-dir-drop must pay exactly one ForceFresh manifest-body HEAD, not one per file"; +} + +/// A create-then-remove of a new part in one transaction must discard both the manifest entries +/// and the in-flight build, so commit() leaves no ref and no live precommit behind. +TEST(CASTransactionRemove, CreateThenDirDropDoesNotPublish) +{ + auto storage = openTxStorage(); + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b08/b08b08b0-0808-4808-8808-080808080808/all_1_1_0/data.bin", "created-then-removed"); + tx->removeDirectory("b08/b08b08b0-0808-4808-8808-080808080808/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + EXPECT_FALSE(storage->existsDirectory("b08/b08b08b0-0808-4808-8808-080808080808/all_1_1_0")); +} + +/// [Task 3] `startup()` publishes `cas_store`/`part_access`/`gc_scheduler` (and sets +/// `pool_uuid`) atomically as its LAST action. Everything +/// before that point -- opening the pool, building the part-folder facade, the capability probe, +/// starting the GC scheduler -- happens into locals first, so a throw anywhere along the way (here +/// simulated via `startup_fault_injection_for_test`, injected right before the publish step) must +/// leave nothing published: `store()` still refuses (null pool -- the Constructing lifecycle, +/// `INVALID_STATE` "not started") even though `Pool::open` and everything else already succeeded. +/// Clearing the hook and retrying `startup()` must then succeed cleanly. +TEST(CASTransactionLifecycle, StartupFailureLatePublishesNothing) +{ + auto storage = makeUnstartedTxStorage(); + + storage->startup_fault_injection_for_test = [] { throw std::runtime_error("injected late-startup failure"); }; + EXPECT_ANY_THROW(storage->startup()); + /// Nothing was published by the failed attempt: store() must still refuse (null pool, not started). + EXPECT_ANY_THROW(storage->store()); + + storage->startup_fault_injection_for_test = {}; + EXPECT_NO_THROW(storage->startup()); + EXPECT_NO_THROW(storage->store()); +} + +/// [Task 4] The storage-lifecycle gate: `store()` (and every other caller of `poolAccess()`) must +/// refuse with `INVALID_STATE` -- an operational condition, not a programming invariant -- whenever +/// no pool is published (the null-pool ShutDown lifecycle after `shutdown`). This is a deliberate +/// behavior change from the previous `LOGICAL_ERROR "accessed before startup"`, which would abort a +/// debug/sanitizer build on a mis-sequenced access instead of surfacing a catchable, operator-actionable +/// error. +TEST(CASTransactionWiring, OperationsRefuseWithoutPublishedPool) +{ + auto storage = openTxStorage(); + storage->shutdown(); /// terminal path resets cas_store to null (the ShutDown storage lifecycle) + + try + { + storage->store(); + FAIL() << "store() must refuse once the disk has no published pool"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::INVALID_STATE) << e.message(); + } +} + +/// (rev.8, Task 15) The Dormant/UNMOUNT lifecycle rollback flips the transitional benign-absent probe +/// behavior to fail-loud on a NULL pool. A storage with no published pool (here: after `shutdown()`, the +/// null-pool ShutDown storage lifecycle) refuses the ENTIRE surface -- including the read-only +/// existence/enumeration probes that the old (now-deleted) `DormantDiskAnswersExistenceProbesAsAbsent` +/// asserted answered benign-absent. This is spec §1's null-pool fail-loud contract: every op class, +/// `Probe` included, throws `INVALID_STATE` ("not started"); a genuinely `Vanished` POOL is the only +/// state that answers truth-absent, and a null pool is not that. (Behavior change documented in the +/// Task 15 report: generic all-disk existence sweeps during server shutdown now see a throw here, not a +/// benign absent; the shutdown window is the deliberate cost of never lying about a not-started disk.) +TEST(CASLifecycle, ShutdownDiskProbesFailLoud) +{ + auto storage = openTxStorage(); + storage->shutdown(); /// null pool -- the ShutDown storage lifecycle + + const std::string file = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"; + const std::string part_dir = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"; + + /// Every read-only probe now THROWS (not started), where the transitional Dormant path answered benign. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { storage->existsDirectory("store"); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { storage->existsFile(file); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { storage->existsFileOrDirectory(part_dir); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { storage->isDirectoryEmpty("store"); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { (void)storage->listDirectory("store"); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { (void)storage->iterateDirectory("store"); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { (void)storage->getStorageObjectsIfExist(file); }); + + /// The content/size surface stays fail-close too (unchanged from the transitional behavior). + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { (void)storage->getFileSize(file); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { (void)storage->getStorageObjects(file); }); +} diff --git a/src/Disks/tests/gtest_ca_wiring.cpp b/src/Disks/tests/gtest_ca_wiring.cpp new file mode 100644 index 000000000000..9d12f9ccb759 --- /dev/null +++ b/src/Disks/tests/gtest_ca_wiring.cpp @@ -0,0 +1,3001 @@ +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +/// M-W wiring tier (design 2026-06-11 section 7 tier 3): the ClickHouse-facing translation layer +/// tested through its own seams. Task 1: PartPathParser — the path-classification rows plus the +/// shadow/detached/mutable rows the later tasks route on. + +using namespace DB::Cas; + +TEST(CASPartPathParser, ParsePartFilePathAtomic) +{ + auto file = parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/columns.txt"); + ASSERT_TRUE(file.has_value()); + EXPECT_EQ(file->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(file->part_name, "all_1_1_0"); + EXPECT_EQ(file->file, "columns.txt"); + EXPECT_TRUE(file->backup_name.empty()); + EXPECT_TRUE(file->shadow_table_dir.empty()); + + auto part_dir = parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/"); // trailing slash, no file + ASSERT_TRUE(part_dir.has_value()); + EXPECT_EQ(part_dir->part_name, "all_1_1_0"); + EXPECT_TRUE(part_dir->file.empty()); + + EXPECT_FALSE(parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111").has_value()); // table dir, not a part + EXPECT_FALSE(parsePartFilePath("123").has_value()); // shallower + + // The real-server shape carries a leading store/; the uuid-pair anchor makes it equivalent. + auto atomic = parsePartFilePath("store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + ASSERT_TRUE(atomic.has_value()); + EXPECT_EQ(atomic->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(atomic->part_name, "all_1_1_0"); + EXPECT_EQ(atomic->file, "data.bin"); +} + +TEST(CASPartPathParser, ThreeCharDatabaseSharingTablePrefixDoesNotFalseAnchorAsAtomic) +{ + // T12: a non-Atomic 3-char database directory whose table directory happens to start with the + // SAME 3 characters (db "abc", table "abcxyz") used to satisfy the old loose Atomic-anchor shape + // check (`prefix.size() == 3 && uuid.compare(0, 3, prefix) == 0`), false-anchoring "abc" as a + // UUID hash-prefix and "abcxyz" as the table UUID -- even though neither looks anything like a + // real UUID. The anchor now additionally requires the prefix to be lowercase-hex and the + // candidate to have the exact 36-char dashed UUID shape, so this path falls through to the + // non-Atomic fallback split instead (folding the whole leading path into table_uuid, exactly like + // ParsePartFilePathNonAtomic's "data/memory_01069/mt" case). + auto d = parsePartFilePath("data/abc/abcxyz/1_1_1_0/x.bin"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "data/abc/abcxyz"); + EXPECT_EQ(d->part_name, "1_1_1_0"); + EXPECT_EQ(d->file, "x.bin"); +} + +TEST(CASPartPathParser, RealHexPrefixUuidPairStillAnchorsAsAtomic) +{ + // Positive control for the tightened anchor: a REAL Atomic on-disk shape -- + // store// with the UUID correctly 36-char dashed and genuinely sharing its first + // 3 characters with the prefix -- still anchors exactly as before. + auto a = parsePartFilePath("store/abc/abc12345-1234-5678-9abc-def012345678/all_1_1_0/x.bin"); + ASSERT_TRUE(a.has_value()); + EXPECT_EQ(a->table_uuid, "abc12345-1234-5678-9abc-def012345678"); + EXPECT_EQ(a->part_name, "all_1_1_0"); + EXPECT_EQ(a->file, "x.bin"); +} + +TEST(CASPartPathParser, ParsePartFilePathProjectionSubPath) +{ + // A projection file keeps its FULL in-part relative path as the file (the tree entry name). + auto proj = parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj/data.bin"); + ASSERT_TRUE(proj.has_value()); + EXPECT_EQ(proj->part_name, "all_1_1_0"); + EXPECT_EQ(proj->file, "p.proj/data.bin"); +} + +TEST(CASPartPathParser, ParsePartFilePathNonAtomic) +{ + // Non-Atomic (Ordinary/Memory/Lazy) layout: data//// — no uuid anchor; + // the part dir is recognized by its block-range suffix (B40). + auto file = parsePartFilePath("data/memory_01069/mt/all_1_1_0/data.cmrk4"); + ASSERT_TRUE(file.has_value()); + EXPECT_EQ(file->table_uuid, "data/memory_01069/mt"); + EXPECT_EQ(file->part_name, "all_1_1_0"); + EXPECT_EQ(file->file, "data.cmrk4"); + + // Temporary/operation prefixes keep the suffix and stay part dirs. + auto tmp = parsePartFilePath("data/memory_01069/mt/tmp_insert_all_1_1_0/data.cmrk4"); + ASSERT_TRUE(tmp.has_value()); + EXPECT_EQ(tmp->part_name, "tmp_insert_all_1_1_0"); + + // Mutation-level form ____. + auto mut = parsePartFilePath("data/db/tbl/20200101_1_1_0_5/data.bin"); + ASSERT_TRUE(mut.has_value()); + EXPECT_EQ(mut->part_name, "20200101_1_1_0_5"); + + // A non-Atomic table-level file is NOT a part file. + EXPECT_FALSE(isPartFilePath("data/memory_01069/mt/format_version.txt")); + auto tf = parseTableFilePath("data/memory_01069/mt/format_version.txt"); + ASSERT_TRUE(tf.has_value()); + EXPECT_EQ(tf->table_uuid, "data/memory_01069/mt"); + EXPECT_EQ(tf->tail, "format_version.txt"); + + EXPECT_EQ(parseTableUuid("data/memory_01069/mt"), std::optional("data/memory_01069/mt")); + + // Generic disk-root files classify as nothing (verbatim passthrough). + EXPECT_FALSE(isPartFilePath("clickhouse_access_check_xyz")); + EXPECT_FALSE(parseTableFilePath("clickhouse_access_check_xyz").has_value()); + EXPECT_FALSE(parseTableUuid("clickhouse_access_check_xyz").has_value()); +} + +TEST(CASPartPathParser, ParseTableUuid) +{ + EXPECT_EQ(parseTableUuid("a11/a11a11a1-1111-4111-8111-111111111111/"), std::optional("a11a11a1-1111-4111-8111-111111111111")); + EXPECT_EQ(parseTableUuid("a11/a11a11a1-1111-4111-8111-111111111111"), std::optional("a11a11a1-1111-4111-8111-111111111111")); + EXPECT_FALSE(parseTableUuid("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").has_value()); // part dir, not table dir + + EXPECT_TRUE(endsWithTableUuidPair("store/a11/a11a11a1-1111-4111-8111-111111111111")); + EXPECT_FALSE(endsWithTableUuidPair("store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(endsWithTableUuidPair("shadow/bk1/store")); +} + +TEST(CASPartPathParser, ParseTableFilePathNested) +{ + // The reserved deduplication_logs/ subdir is a table-level namespace, never a part dir. + EXPECT_FALSE(isPartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs/deduplication_log_1.txt")); + auto tf = parseTableFilePath("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs/deduplication_log_1.txt"); + ASSERT_TRUE(tf.has_value()); + EXPECT_EQ(tf->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(tf->tail, "deduplication_logs/deduplication_log_1.txt"); + + auto flat = parseTableFilePath("a11/a11a11a1-1111-4111-8111-111111111111/format_version.txt"); + ASSERT_TRUE(flat.has_value()); + EXPECT_EQ(flat->tail, "format_version.txt"); + + EXPECT_FALSE(parseTableFilePath("a11/a11a11a1-1111-4111-8111-111111111111").has_value()); + EXPECT_FALSE(parseTableFilePath("a11/a11a11a1-1111-4111-8111-111111111111/").has_value()); + + EXPECT_TRUE(isPartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); +} + +TEST(CASPartPathParser, ShadowFreezePaths) +{ + EXPECT_TRUE(isShadowPath("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_TRUE(isShadowPath("/shadow/bk1")); + EXPECT_FALSE(isShadowPath("store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_FALSE(isShadowPath("shadowy/bk1")); + + auto s = parsePartFilePath("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + ASSERT_TRUE(s.has_value()); + EXPECT_EQ(s->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(s->part_name, "all_1_1_0"); + EXPECT_EQ(s->file, "data.bin"); + EXPECT_EQ(s->backup_name, "bk1"); + EXPECT_EQ(s->shadow_table_dir, "shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111"); +} + +TEST(CASPartPathParser, DetachedPathsReportTheSharedDetachedComponent) +{ + // The PoC contract (B36): "detached" parses as the part_name; the real detached part dir is + // the first component of `file`. The transaction/read routing re-splits on this shape. + auto d = parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/detached/attaching_all_0_0_0/metadata_version.txt"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(d->part_name, std::string(kDetachedDirName)); + EXPECT_EQ(d->file, "attaching_all_0_0_0/metadata_version.txt"); +} + +TEST(CASPartPathParser, MovingPathsReportTheSharedMovingComponent) +{ + // Atomic layout: "moving" lands on part_idx for free (it is the component right after the + // table , same mechanism as "detached" -- no parser change needed here, only route()). + auto d = parsePartFilePath("a11/a11a11a1-1111-4111-8111-111111111111/moving/all_1_1_0/data.bin"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(d->part_name, std::string(kMovingDirName)); + EXPECT_EQ(d->file, "all_1_1_0/data.bin"); +} + +TEST(CASPartPathParser, MovingPathsNonAtomicFoldIntoTheTableNamespace) +{ + // Mirrors DetachedPathsNonAtomicFoldIntoTheTableNamespace (U#6): without an explicit anchor + // the right-to-left part-dir scan would anchor on the INNER real part dir and fold "moving" + // into a spurious table_uuid ("data//
/moving"), diverging from the table's real + // namespace -- the identical bug class the detached anchor was added to prevent. + auto d = parsePartFilePath("data/db/tbl/moving/all_1_1_0/data.bin"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "data/db/tbl"); + EXPECT_EQ(d->part_name, std::string(kMovingDirName)); + EXPECT_EQ(d->file, "all_1_1_0/data.bin"); + + // The bare non-Atomic moving CONTAINER dir folds to part_name == "moving" with an empty + // file, exactly like the Atomic container. + auto c = parsePartFilePath("data/db/tbl/moving"); + ASSERT_TRUE(c.has_value()); + EXPECT_EQ(c->table_uuid, "data/db/tbl"); + EXPECT_EQ(c->part_name, std::string(kMovingDirName)); + EXPECT_TRUE(c->file.empty()); +} + +TEST(CASPartPathParser, DetachedPathsNonAtomicFoldIntoTheTableNamespace) +{ + // U#6: the Ordinary/non-Atomic detached form data//
/detached// must fold + // into the table's OWN namespace with part_name == "detached" (mirroring the Atomic form), so + // route() keys the detached/ ref off it. The right-to-left part-dir scan would otherwise + // anchor on the INNER part dir and fold `detached` into a spurious table_uuid + // ("data//
/detached") that DROP TABLE never cleans — a permanently orphaned live ref. + auto d = parsePartFilePath("data/db/tbl/detached/attaching_all_0_0_0/metadata_version.txt"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "data/db/tbl"); + EXPECT_EQ(d->part_name, std::string(kDetachedDirName)); + EXPECT_EQ(d->file, "attaching_all_0_0_0/metadata_version.txt"); + + // The bare non-Atomic detached CONTAINER dir folds to part_name == "detached" with an empty file, + // exactly like the Atomic container, so route()'s empty-ref branch is reached for both layouts. + auto c = parsePartFilePath("data/db/tbl/detached"); + ASSERT_TRUE(c.has_value()); + EXPECT_EQ(c->table_uuid, "data/db/tbl"); + EXPECT_EQ(c->part_name, std::string(kDetachedDirName)); + EXPECT_TRUE(c->file.empty()); +} + +TEST(CASPartPathParser, DetachedNamedTableIsKnownAmbiguityFoldedAsReservedDir) +{ + // ACCEPTED LIMITATION (see the anchor-site comment in findPartDirComponent): a non-Atomic + // database or TABLE literally named "detached" is structurally indistinguishable, from the path + // string alone, from the reserved detached subdir of a table one level up — so it gets folded + // as the reserved dir, not as a table name. This test PINS that known, deliberately-accepted + // behavior (backlogged by the stabilization campaign) so any future change to it is a conscious + // one, not an accidental regression. + auto d = parsePartFilePath("data/db/detached/all_1_1_0/data.bin"); + ASSERT_TRUE(d.has_value()); + EXPECT_EQ(d->table_uuid, "data/db"); + EXPECT_EQ(d->part_name, std::string(kDetachedDirName)); + EXPECT_EQ(d->file, "all_1_1_0/data.bin"); + + // Consequently the table dir itself is unrecognized: it looks like a detached container instead. + EXPECT_FALSE(parseTableUuid("data/db/detached").has_value()); +} + +TEST(CASPartPathParser, RawPathSplitMemoizedAcrossClassifiers) +{ + // The CA read path runs isPartFilePath then parsePartFilePath on the SAME raw path several times + // per logical file-open (existsFile -> getFileSize -> getStorageObjects). The split is a pure + // function of the path, so all of those must split the path exactly ONCE (B1). + resetSplitCacheForTest(); + const std::string path = "store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/columns.txt"; + EXPECT_TRUE(isPartFilePath(path)); + ASSERT_TRUE(parsePartFilePath(path).has_value()); + ASSERT_TRUE(parsePartFilePath(path).has_value()); + EXPECT_EQ(splitCacheMissesForTest(), 1u) << "the same raw path must be split only once"; + + // A distinct raw path is a fresh split (miss #2); repeats of it reuse the memo. + const std::string other = "store/a22/a22a22a2-2222-4222-8222-222222222222/all_1_1_0/data.bin"; + EXPECT_TRUE(isPartFilePath(other)); + EXPECT_TRUE(isPartFilePath(other)); + EXPECT_EQ(splitCacheMissesForTest(), 2u); + + // Correctness is unchanged: the memoized parse yields the same fields the direct parse would. + const auto parsed = parsePartFilePath(path); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(parsed->part_name, "all_1_1_0"); + EXPECT_EQ(parsed->file, "columns.txt"); +} + +TEST(CASPartPathParser, SplitCacheEvictionStaysCorrect) +{ + // The split cache is a small fixed-capacity FIFO ring, NOT an LRU/MRU: a hit never promotes its + // slot, so a path seen recently can still be evicted by unrelated churn through the same thread. + // That is only ever a cache-EFFECTIVENESS tradeoff, never a correctness one: pin that once enough + // distinct paths evict the first path's cached split, re-parsing it still yields the exact right + // result (a forced re-split / cache miss on the re-parse is expected and fine here — the + // assertion is correctness under eviction, not hit rate). + resetSplitCacheForTest(); + const std::string first = "store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/columns.txt"; + ASSERT_TRUE(parsePartFilePath(first).has_value()); + + // 8 more distinct paths churn through the ring (capacity 8), evicting `first`'s slot. + const std::vector table_dirs = { + "", + "a11/a11a11a1-1111-4111-8111-111111111111", + "a22/a22a22a2-2222-4222-8222-222222222222", + "a33/a33a33a3-3333-4333-8333-333333333333", + "a44/a44a44a4-4444-4444-8444-444444444444", + "a55/a55a55a5-5555-4555-8555-555555555555", + "a66/a66a66a6-6666-4666-8666-666666666666", + "a77/a77a77a7-7777-4777-8777-777777777777", + "a88/a88a88a8-8888-4888-8888-888888888888", + "a99/a99a99a9-9999-4999-8999-999999999999", + }; + for (int i = 2; i <= 9; ++i) + { + const std::string path = "store/" + table_dirs[i] + "/all_1_1_0/columns.txt"; + ASSERT_TRUE(parsePartFilePath(path).has_value()); + } + + const size_t misses_before_reparse = splitCacheMissesForTest(); + const auto reparsed = parsePartFilePath(first); + ASSERT_TRUE(reparsed.has_value()); + EXPECT_EQ(reparsed->table_uuid, "a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(reparsed->part_name, "all_1_1_0"); + EXPECT_EQ(reparsed->file, "columns.txt"); + // Confirms the re-parse really was a forced re-split (the slot was evicted), not a lucky hit. + EXPECT_EQ(splitCacheMissesForTest(), misses_before_reparse + 1); +} + +/// ==== M-W Task 2: the read side over Cas::Pool ==== +/// Fixture: publish parts through the CORE API, then read through the IMetadataStorage surface of +/// the rewritten ContentAddressedMetadataStorage (real ctor over a Local object storage; the +/// backend self-selects EmulatedSingleProcess token semantics). + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace DB::ErrorCodes +{ + extern const int FILE_DOESNT_EXIST; + extern const int BAD_ARGUMENTS; +} + +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsString staging_backend; +} + +namespace +{ + +DB::Cas::ManifestEntry wiringBlobEntry(const String & path, const String & payload) +{ + DB::Cas::ManifestEntry e; + e.path = path; + e.placement = DB::Cas::EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + return e; +} + +/// All-tree-part-files Task 6/9: the small per-part files (uuid.txt, metadata_version.txt, ...) are +/// ordinary Inline-placement manifest entries now — this is the low-level PartWriteTxn-API equivalent of +/// what `ContentAddressedTransaction::writeFile`'s inline candidate path stages in production. +DB::Cas::ManifestEntry wiringInlineEntry(const String & path, const String & bytes) +{ + DB::Cas::ManifestEntry e; + e.path = path; + e.placement = DB::Cas::EntryPlacement::Inline; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(bytes))}; + + e.blob_size = bytes.size(); + e.inline_bytes = bytes; + return e; +} + +/// The one table identity these tests use, as a namespace LIFE: namespace files are life-keyed +/// (directive §2), resolved from the CATALOG exactly as the disk's own write path resolves it. Naming +/// the Stage-A sentinel here instead would put the fixture's files under a prefix the disk no longer +/// reads (Task 4b), so `existsFile`/`listDirectory` below would report them absent -- the fixture and +/// the code under test must agree on the life, and the only way to guarantee that is to ask the same +/// resolver. +DB::Cas::NamespaceLifeId wiringLife(DB::ContentAddressedMetadataStorage & storage) +{ + return storage.store()->namespaceLife( + storage.liveNamespace("a11a11a1-1111-4111-8111-111111111111")); +} + +std::shared_ptr openWiringStorage() +{ + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_wiring_scratch"); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// One part with a content blob, a projection file, and the small per-part files (uuid.txt, +/// metadata_version.txt — ordinary Inline entries now, all-tree-part-files Task 6/9), published +/// through the real PartWriteTxn into `ns` under `ref`. +void publishWiredPart( + DB::ContentAddressedMetadataStorage & storage, const DB::Cas::RootNamespace & ns, const String & ref) +{ + /// Port off the removed PartWriteTxn::putTree/publish API onto the part-manifest write flow + /// (beginPartWrite → stageManifest → precommitAdd → putBlob → promote). The wiring sets the owning + /// namespace EXPLICITLY (intended_namespace) — faithful to ContentAddressedTransaction — so a + /// `detached/` ref (which itself contains '/') is staged in the TABLE namespace, not in a + /// spurious `/detached` namespace. intended_ref stays as "ns/ref" diagnostic forensics. + DB::Cas::PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + info.intended_namespace = ns; + auto build = storage.store()->beginPartWrite(info); + + /// Strictly ascending canonical path order (PartFolderView's binary-search precondition): + /// data.bin < metadata_version.txt < p.proj/data.bin < uuid.txt. + const auto id = build->stageManifest( + {wiringBlobEntry("data.bin", "payload-A"), wiringInlineEntry("metadata_version.txt", "5"), + wiringBlobEntry("p.proj/data.bin", "payload-B"), wiringInlineEntry("uuid.txt", "u-123")}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf("payload-A"), DB::Cas::BlobSource::fromString("payload-A")); + build->putBlob(idOf("payload-B"), DB::Cas::BlobSource::fromString("payload-B")); + build->promote(ns, ref, build->buildId(), id); + + /// promote stamps published_at_ms with nowMs(); the read assertions want a FIXED stamp, so pin it + /// through the set_published_at path (no journal record for anything but the stamp itself). + storage.store()->updateRefPublishedAt(ns, ref, + [](DB::Cas::RefPublishedAtUpdate & r) { r.published_at_ms = 1700000000ULL * 1000; }); /// epoch ms; getLastModified /1000 +} + +} + +/// `supportsAtomicFileWrites` (all-tree task 5): the CA metadata storage publishes a file write in +/// one shot, so `VersionMetadataOnDisk::storeInfoToDataPartStorage` can skip the tmp+replace dance. +/// A plain (non-content-addressed) metadata storage keeps the base-class default of `false`. +TEST(CASWiringCapability, SupportsAtomicFileWrites) +{ + auto ca_storage = openWiringStorage(); + EXPECT_TRUE(ca_storage->supportsAtomicFileWrites()); + + auto plain_storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "", /*object_metadata_cache_size=*/0); + EXPECT_FALSE(plain_storage->supportsAtomicFileWrites()); +} +TEST(CASWiringRead, ResolvesPublishedPart) +{ + auto storage = openWiringStorage(); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/missing.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 9u); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111")); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9")); + EXPECT_FALSE(storage->existsDirectory("a22/a22a22a2-2222-4222-8222-222222222222")); + + /// Part dir listing: nested keys collapse to their first component; the publish stamp + /// (published_at_ms typed field) never surfaces as a dir entry — every staged file does. + auto names = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"data.bin", "metadata_version.txt", "p.proj", "uuid.txt"})); + + auto parts = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111"); + EXPECT_EQ(parts, (std::vector{"all_1_1_0"})); + + /// The part dir reports EMPTY (virtual files; B45) so removeDirectory goes straight to the + /// ref-unlink; the table dir keeps listing-based emptiness. + EXPECT_TRUE(storage->isDirectoryEmpty("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->isDirectoryEmpty("a11/a11a11a1-1111-4111-8111-111111111111")); + + /// Blob-backed file: a real key, PAYLOAD-sized (the envelope header is a read-path concern). + auto objects = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + ASSERT_EQ(objects.size(), 1u); + EXPECT_FALSE(objects[0].remote_path.empty()); + EXPECT_EQ(objects[0].bytes_size, 9u); + + /// Small Inline entry: bytes live in the shard manifest, not as their own object. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt"), 5u); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt"), std::optional("u-123")); + auto mobj = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt"); + ASSERT_EQ(mobj.size(), 1u); + EXPECT_TRUE(mobj[0].remote_path.empty()); /// sized placeholder; bytes ride prepareInManifestRead + + /// The typed publish stamp (published_at_ms epoch ms) backs getLastModified for the part dir and its files. + EXPECT_EQ(storage->getLastModified("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").epochTime(), 1700000000); + EXPECT_EQ(storage->getLastModified("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin").epochTime(), 1700000000); +} + +TEST(CASWiringRead, BlobViewPlanRidesTheStandardPipeline) +{ + /// The committed read path (B116): an in-manifest file is served from memory via + /// prepareInManifestRead; a blob-backed file translates to its physical blob object + + /// payload window (getBlobViewPlan) and rides the STANDARD object-storage pipeline, + /// bounded by the FileView stage — composed here the way DiskObjectStorage::prepareRead + /// composes it. + auto object_storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_wiring_scratch"); + auto storage = std::make_shared( + object_storage, "pool", "srv1", "", nullptr, settings); + storage->startup(); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + /// In-manifest file: memory source, no blob plan. + DB::ReadPipeline manifest_pipeline; + ASSERT_TRUE(storage->prepareInManifestRead("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt", DB::ReadSettings{}, manifest_pipeline)); + String manifest_bytes; + { + auto buf = manifest_pipeline.build(); + DB::readStringUntilEOF(manifest_bytes, *buf); + } + EXPECT_EQ(manifest_bytes, "u-123"); + /// Not a `getBlobViewPlan` call on the in-manifest path here (all-tree Task 6/9: uuid.txt is now + /// a real Inline manifest entry): `getBlobViewPlan`'s only production caller + /// (`DiskObjectStorage::prepareRead`) never reaches it once `prepareInManifestRead` returns true + /// above — `getBlobViewPlan`'s precondition is "confirmed not in-manifest-servable," which calling + /// it directly on an Inline path violates. Pre-Task-9 this assertion passed only by coincidence + /// (uuid.txt was not a manifest entry at all, so `findFile` returned not-found, not because + /// `getBlobViewPlan` gracefully handles an Inline entry it does find). + + /// Blob-backed file: a real physical key and a payload-sized window whose extent equals + /// the object's readable size (a right-bounded read never overshoots the window). + const std::string path = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"; + auto plan = storage->getBlobViewPlan(path); + ASSERT_TRUE(plan.has_value()); + EXPECT_FALSE(plan->object.remote_path.empty()); + EXPECT_EQ(plan->object.local_path, path); + EXPECT_EQ(plan->payload_end - plan->payload_offset, 9u); + EXPECT_EQ(plan->object.bytes_size, plan->payload_end); + EXPECT_FALSE(storage->prepareInManifestRead(path, DB::ReadSettings{}, manifest_pipeline = {})); + + auto make_pipeline = [&] + { + DB::ReadPipeline pipeline; + pipeline.setSource(object_storage, {plan->object}, DB::ReadSettings{}); + pipeline.needGather(); + pipeline.needFileView(path, plan->payload_offset, plan->payload_end); + return pipeline; + }; + EXPECT_EQ(make_pipeline().describe(), "Source(ObjectStorage) -> Gather -> FileView"); + + { + auto buf = make_pipeline().build(); + EXPECT_EQ(buf->getFileName(), path); + EXPECT_EQ(buf->tryGetFileSize(), std::optional(9)); + String bytes; + DB::readStringUntilEOF(bytes, *buf); + EXPECT_EQ(bytes, "payload-A"); + } + + /// Right-bounded read through the view (the MergeTreeReaderStream::adjustRightMark shape): + /// the bound is window-relative and forwarded down the chain. + { + auto buf = make_pipeline().build(); + buf->setReadUntilPosition(7); + String head(7, '\0'); + buf->readStrict(head.data(), 7); + EXPECT_EQ(head, "payload"); + EXPECT_TRUE(buf->eof()); + buf->setReadUntilEnd(); + String tail; + DB::readStringUntilEOF(tail, *buf); + EXPECT_EQ(tail, "-A"); + } + + /// Seek inside the window. + { + auto buf = make_pipeline().build(); + buf->seek(8, SEEK_SET); + String last; + DB::readStringUntilEOF(last, *buf); + EXPECT_EQ(last, "A"); + } +} + +TEST(CASWiringRead, ProjectionDirectory) +{ + auto storage = openWiringStorage(); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/q.proj")); + EXPECT_EQ(storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj"), (std::vector{"data.bin"})); + EXPECT_TRUE(storage->isDirectoryEmpty("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj")); /// B60 + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj/data.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj/data.bin"), 9u); +} + +TEST(CASWiringRead, DetachedFoldedIntoTableNamespace) +{ + auto storage = openWiringStorage(); + /// B181: a detached part is a `detached/`-prefixed ref INSIDE the table's own archive namespace, + /// not a separate sibling namespace. Publish it that way through the core, and ALSO a live part + /// that shares the same base name to prove the live↔detached collision is impossible (the ref + /// names `all_1_1_0` and `detached/all_1_1_0` differ — one namespace, no re-split needed). + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "detached/broken_all_1_1_0"); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "broken_all_1_1_0"); + + /// The TABLE dir collapses the `detached/` refs to the single `detached` subdir entry + /// alongside the live part name. + auto top = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111"); + std::sort(top.begin(), top.end()); + EXPECT_EQ(top, (std::vector{"broken_all_1_1_0", "detached"})); + + /// The detached CONTAINER lists the detached part DIRECTORY names (B36's intent), prefix-stripped + /// — and NOT the live part of the same base name. + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached")); + EXPECT_EQ(storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached"), (std::vector{"broken_all_1_1_0"})); + /// A single detached part dir + its files (the detached part is its own `detached/`-prefixed ref). + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0")); + auto names = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"data.bin", "metadata_version.txt", "p.proj", "uuid.txt"})); + /// The B62 shape: a detached part's mutable file resolves through the `detached/`-prefixed ref. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0/metadata_version.txt")); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0/metadata_version.txt"), + std::optional("5")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0/data.bin")); +} + +TEST(CASWiringRoute, DetachedFoldsIntoTableNamespaceWithPrefixedRef) +{ + /// B181: a detached part file routes to the table's OWN archive namespace under a + /// `detached/`-prefixed ref — NOT a separate sibling namespace. + auto storage = openWiringStorage(); + auto p = parsePartFilePath("store/a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0/data.bin"); + ASSERT_TRUE(p.has_value()); + auto r = storage->route(*p); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->ns.string(), storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string()); + EXPECT_EQ(r->ref, "detached/broken_all_1_1_0"); + EXPECT_EQ(r->file, "data.bin"); + + /// The detached CONTAINER dir routes to the table ns with an empty ref (filtered listing). + auto pc = parsePartFilePath("store/a11/a11a11a1-1111-4111-8111-111111111111/detached/broken_all_1_1_0"); + ASSERT_TRUE(pc.has_value()); + auto rc = storage->route(*pc); + ASSERT_TRUE(rc.has_value()); + EXPECT_EQ(rc->ns.string(), storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string()); + EXPECT_EQ(rc->ref, "detached/broken_all_1_1_0"); + EXPECT_TRUE(rc->file.empty()); +} + +TEST(CASWiringRoute, MovingFoldsOntoAPrefixedStagingRef) +{ + /// L1 (MOVE-to-CA fix): the mover clones a part under TABLE/moving// before the + /// atomic rename into place. Mirroring `detached`, a moved part resolves onto a + /// `moving/`-PREFIXED staging ref -- NOT the part's final live ref directly. Publishing under + /// the final ref before the mover's swap would break move crash-atomicity (a crash between the + /// clone commit and the swap would leave a committed live ref that never went through the + /// swap). The staging ref keeps the pre-swap clone un-live; the mover's rename does a real ref + /// repoint moving/ -> . + auto storage = openWiringStorage(); + auto p = parsePartFilePath("store/a11/a11a11a1-1111-4111-8111-111111111111/moving/all_1_1_0/data.bin"); + ASSERT_TRUE(p.has_value()); + EXPECT_EQ(p->part_name, std::string(kMovingDirName)); + EXPECT_EQ(p->file, "all_1_1_0/data.bin"); + + auto r = storage->route(*p); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->ns.string(), storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string()); + EXPECT_EQ(r->ref, "moving/all_1_1_0"); + EXPECT_EQ(r->file, "data.bin"); + + /// The bare moving CONTAINER dir TABLE/moving routes to the table ns with an empty ref. + auto pc = parsePartFilePath("store/a11/a11a11a1-1111-4111-8111-111111111111/moving"); + ASSERT_TRUE(pc.has_value()); + auto rc = storage->route(*pc); + ASSERT_TRUE(rc.has_value()); + EXPECT_EQ(rc->ns.string(), storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string()); + EXPECT_TRUE(rc->ref.empty()); + EXPECT_TRUE(rc->file.empty()); +} + +TEST(CASWiringRead, ShadowFreezeTree) +{ + auto storage = openWiringStorage(); + publishWiredPart(*storage, storage->shadowNamespace("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + EXPECT_EQ(storage->shadowNamespace("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111").string(), + storage->serverRootId() + "/shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111"); + + /// Intermediate dirs derive from the registered shadow namespaces. + EXPECT_TRUE(storage->existsDirectory("shadow/bk1")); + EXPECT_TRUE(storage->existsDirectory("shadow/bk1/store")); + EXPECT_FALSE(storage->existsDirectory("shadow/bk2")); + EXPECT_EQ(storage->listDirectory("shadow"), (std::vector{"bk1"})); + EXPECT_EQ(storage->listDirectory("shadow/bk1"), (std::vector{"store"})); + EXPECT_EQ(storage->listDirectory("shadow/bk1/store"), (std::vector{"a11"})); + EXPECT_EQ(storage->listDirectory("shadow/bk1/store/a11"), (std::vector{"a11a11a1-1111-4111-8111-111111111111"})); + /// Shadow TABLE dir (strict uuid-pair anchor) and PART dir. + EXPECT_TRUE(storage->existsDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111")); + EXPECT_EQ(storage->listDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111"), (std::vector{"all_1_1_0"})); + EXPECT_TRUE(storage->existsDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + auto names = storage->listDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"data.bin", "metadata_version.txt", "p.proj", "uuid.txt"})); + EXPECT_TRUE(storage->existsFile("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(storage->getFileSize("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 9u); +} + +TEST(CASWiringRead, VerbatimNamespaceFiles) +{ + auto storage = openWiringStorage(); + EXPECT_TRUE(storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string().starts_with("test/")) + << storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string(); + EXPECT_NE(storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string().find("/store/a11/a11a11a1-1111-4111-8111-111111111111@cas@"), std::string::npos) + << storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string(); + + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + storage->store()->putNamespaceFile(wiringLife(*storage), "format_version.txt", "1\n"); + storage->store()->putNamespaceFile( + wiringLife(*storage), "deduplication_logs/deduplication_log_1.txt", "log-bytes"); + /// Loose disk-root files are plain mountpoint objects (design §5.2), not namespace files. + storage->store()->putMountpointObject(storage->serverRootId() + "/" + "clickhouse_access_check_xyz", "ok"); + + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/format_version.txt")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/format_version.txt"), 2u); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/format_version.txt"), std::optional("1\n")); + + /// Table dir listing merges part names + verbatim file first components. + auto names = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"all_1_1_0", "deduplication_logs", "format_version.txt"})); + + /// The reserved table-level subdir. + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs")); + EXPECT_EQ(storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs"), + (std::vector{"deduplication_log_1.txt"})); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs/deduplication_log_1.txt")); + + /// Loose disk-root files are plain objects — existsFile checks the mountpoint object, not a namespace file. + EXPECT_TRUE(storage->existsFile("clickhouse_access_check_xyz")); + /// Loose files are real objects — tryGetInManifestBytes returns nullopt (not in-manifest bytes). + EXPECT_EQ(storage->tryGetInManifestBytes("clickhouse_access_check_xyz"), std::nullopt); + EXPECT_EQ(storage->getFileSize("clickhouse_access_check_xyz"), 2u); + EXPECT_FALSE(storage->existsFile("clickhouse_access_check_other")); +} + +/// `DirShape::TableDir`'s `existsDirectory` used to answer "has at least one committed part", so an +/// Atomic table that only ever wrote its namespace-level `format_version.txt` (no part published yet) +/// reported its own root as absent. `existsDirectory` is the precheck `MergeTreeData::dropAllData` +/// uses to decide whether `removeRecursive`/`dropNamespace` needs to run at all -- a false negative +/// here means `DROP TABLE` on such a table never admits removal, leaking a `Live` catalog row forever. +TEST(CASWiringRead, TableRootExistsWithNamespaceFilesButNoCommittedRef) +{ + auto storage = openWiringStorage(); + storage->store()->putNamespaceFile(wiringLife(*storage), "format_version.txt", "1\n"); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111")); +} + +/// Same defect, non-Atomic fallback shape (`parseTableUuid` folds the whole leading path into the +/// "uuid"): a files-only table under `data//
` must be present too. +TEST(CASWiringRead, TableRootExistsWithNamespaceFilesButNoCommittedRefNonAtomic) +{ + auto storage = openWiringStorage(); + const auto ns = storage->liveNamespace("data/memory_01069/mt"); + storage->store()->putNamespaceFile(storage->store()->namespaceLife(ns), "format_version.txt", "1\n"); + EXPECT_TRUE(storage->existsDirectory("data/memory_01069/mt")); +} + +/// A cataloged `Live` life with ZERO refs and ZERO namespace files -- not just zero refs -- must still +/// report present. `namespaceLife` is the write-side resolution that mints a `Live` catalog row on +/// first touch; calling it alone (no ref, no namespace file written afterward) is the minimal way to +/// reach this state, and it prevents a future regression from "catalog OR files" back to "files only". +TEST(CASWiringRead, EmptyCatalogedLiveTableRootExists) +{ + auto storage = openWiringStorage(); + (void)storage->store()->namespaceLife(storage->liveNamespace("a55a55a5-5555-4555-8555-555555555555")); + EXPECT_TRUE(storage->existsDirectory("a55/a55a55a5-5555-4555-8555-555555555555")); +} + +/// C4: the fixed dispatch order is the invariant. Pins the two ambiguous early guards that make the +/// order load-bearing: store/ (AtomicShard, ambiguous with the non-Atomic table fallback) and a +/// shadow table dir (which also satisfies parseTableUuid). existsDirectory/listDirectory must agree. +TEST(CASWiringRoute, DirShapeDispatchOrderIsStable) +{ + auto storage = openWiringStorage(); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + publishWiredPart(*storage, storage->shadowNamespace("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + using DS = DB::ContentAddressedMetadataStorage::DirShape; + EXPECT_EQ(storage->classifyDirectoryForTest("store/uui").shape, DS::AtomicShard); + EXPECT_EQ(storage->classifyDirectoryForTest("a11/a11a11a1-1111-4111-8111-111111111111").shape, DS::TableDir); + EXPECT_EQ(storage->classifyDirectoryForTest("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").shape, DS::PartDir); + EXPECT_EQ(storage->classifyDirectoryForTest("a11/a11a11a1-1111-4111-8111-111111111111/detached").shape, DS::DetachedContainer); + EXPECT_EQ(storage->classifyDirectoryForTest("a11/a11a11a1-1111-4111-8111-111111111111/moving").shape, DS::MovingContainer); + EXPECT_EQ(storage->classifyDirectoryForTest("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111").shape, DS::ShadowTable); + EXPECT_EQ(storage->classifyDirectoryForTest("shadow/bk1").shape, DS::ShadowIntermediate); + EXPECT_EQ(storage->classifyDirectoryForTest("a11/a11a11a1-1111-4111-8111-111111111111/deduplication_logs").shape, DS::TableSubdir); + EXPECT_EQ(storage->classifyDirectoryForTest("store").shape, DS::GenericIntermediate); +} + +/// ==== M-W Task 3: the write path through IMetadataTransaction ==== + + +namespace +{ + +void writeThroughTransaction(DB::IMetadataTransaction & tx, const String & path, const String & bytes) +{ + auto & ca_tx = dynamic_cast(tx); + auto buf = ca_tx.writeFile(path, 65536, DB::WriteMode::Rewrite, {}); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); +} + +} + +TEST(CASWiring, LocalStagingRemainsDefault) +{ + auto object_storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_wiring_local_staging_default"); + auto storage = std::make_shared( + object_storage, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + EXPECT_EQ(storage->stagingBackend(), DB::Cas::StagingBackend::Local); + + auto tx = storage->createTransaction(); + writeThroughTransaction( + *tx, + "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", + "local-staging"); + + DB::RelativePathsWithMetadata staged; + object_storage->listObjects(storage->stagingKeyPrefix(), staged, /*max_keys=*/0); + EXPECT_TRUE(staged.empty()); +} + +TEST(CASWiringWrite, ContentRoundTripThroughTransaction) +{ + auto storage = openWiringStorage(); + + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "content-A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/checksums.txt", "sums"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt", "u-42"); + /// Nothing visible before commit. + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + tx->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 9u); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt"), std::optional("u-42")); + auto names = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"checksums.txt", "data.bin", "uuid.txt"})); + /// The publish stamp was added automatically and is filtered from listings. + EXPECT_GT(storage->getLastModified("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").epochTime(), 1700000000); +} + +TEST(CASWiringWrite, InlineOnlyPartPublishesWithoutBuildCrash) +{ + /// Regression (CRASH-CA-S3 "staged entries without a PartWriteTxn"): a part whose files are ALL inline + /// — no `partFileMustStayBlob` file (`.bin`/`.mrk*`/`primary.idx`), e.g. an EMPTY merge output that + /// writes only `checksums.txt`/`count.txt` and no `data.bin` — staged manifest entries via the + /// inline write path, which did NOT establish a PartWriteTxn (only the blob path did, via `buildFor`). So + /// `publishStaging` reached its `st.build != nullptr` invariant with entries but no PartWriteTxn and threw + /// LOGICAL_ERROR — a SERVER CRASH under `abort_on_logical_error`. Writing only inline metadata files + /// to a fresh part and committing must SUCCEED and publish the part. (Bug pre-existed the inline-files + /// feature; fix: the inline path now calls `buildFor` like the blob path.) + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/checksums.txt", "sums"); // inline (no blob) + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/count.txt", "0"); // inline (no blob) + EXPECT_NO_THROW(tx->commit(DB::NoCommitOptions{})); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/checksums.txt"), std::optional("sums")); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/count.txt"), std::optional("0")); + auto names = storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + std::sort(names.begin(), names.end()); + EXPECT_EQ(names, (std::vector{"checksums.txt", "count.txt"})); +} + +TEST(CASWiringWrite, IdenticalContentDedupsToOneBlob) +{ + auto storage = openWiringStorage(); + + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "same-bytes"); + tx->commit(DB::NoCommitOptions{}); + auto tx2 = storage->createTransaction(); + writeThroughTransaction(*tx2, "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin", "same-bytes"); + tx2->commit(DB::NoCommitOptions{}); + + /// Identical content => the SAME blob object (the key is the content hash). + auto a = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + auto b = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin"); + ASSERT_EQ(a.size(), 1u); + ASSERT_EQ(b.size(), 1u); + EXPECT_EQ(a[0].remote_path, b[0].remote_path); +} + +TEST(CASWiringWrite, UncommittedTransactionPublishesNothing) +{ + auto storage = openWiringStorage(); + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "doomed"); + /// destroyed without commit => PartWriteTxn abandoned (uploads are heartbeat-gated debris) + } + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); +} + +TEST(CASWiringWrite, MutableOnlyUpdateOnCommittedPart) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "content-A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/txn_version.txt", "v1"); + tx->commit(DB::NoCommitOptions{}); + + /// The MVCC autocommit one-shot shape: a fresh transaction rewriting ONLY a mutable file of a + /// COMMITTED part goes through updateRefPublishedAt (no tree rebuild, no journal record). + auto tx2 = storage->createTransaction(); + writeThroughTransaction(*tx2, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/txn_version.txt", "v2"); + tx2->commit(DB::NoCommitOptions{}); + + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/txn_version.txt"), std::optional("v2")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); /// the tree is untouched +} + +TEST(CASWiringWrite, VerbatimFilesDurableOnFinalizeAndAppendable) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + /// Verbatim files are durable on FINALIZE, with no commit (the disk layer's autocommit + /// contract for table-level files). + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt", "commands\n"); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt")); + + /// Append = read-modify-rewrite (the MVCC mutation-entry CSN append). + { + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt", 65536, DB::WriteMode::Append, {}); + buf->write("csn 42\n", 7); + buf->finalize(); + } + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt"), + std::optional("commands\ncsn 42\n")); +} + +/// ==== M-W Tasks 5-7: carry-forward, renames, removals, detached/ATTACH/FREEZE ==== + +TEST(CASWiringOps, HardLinkCarriesForwardWithoutReupload) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "shared-payload"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt", "u-1"); + tx->commit(DB::NoCommitOptions{}); + + /// A mutation/merge carries unchanged files into the new part by hardlink. + auto tx2 = storage->createTransaction(); + tx2->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_5/data.bin"); + tx2->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/uuid.txt", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_5/uuid.txt"); + tx2->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_5/data.bin")); + EXPECT_EQ(storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")[0].remote_path, + storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_5/data.bin")[0].remote_path); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_5/uuid.txt"), std::optional("u-1")); +} + +TEST(CASWiringOps, TmpToFinalRenamePublishesUnderFinalName) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0/data.bin", "fresh"); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_insert_all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); +} + +TEST(CASWiringOps, CommittedPartRenameMovesTheRef) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "bytes"); + tx->commit(DB::NoCommitOptions{}); + + /// MergeTree renames a part to delete_tmp_ before removing it. + auto tx2 = storage->createTransaction(); + tx2->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0"); + tx2->commit(DB::NoCommitOptions{}); + + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0/data.bin")); +} + +TEST(CASWiringOps, ProjectionTmpRenameRekeysStagedEntries) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "main"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p_1.tmp_proj/data.bin", "proj"); + auto & ca_tx = dynamic_cast(*tx); + ca_tx.moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p_1.tmp_proj", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj"); + tx->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj/data.bin")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p_1.tmp_proj/data.bin")); + EXPECT_EQ(storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/p.proj"), (std::vector{"data.bin"})); +} + +TEST(CASWiringOps, DetachAttachRoundTrip) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "detachable"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/metadata_version.txt", "3"); + tx->commit(DB::NoCommitOptions{}); + + /// DETACH: a committed part moves into the detached namespace - pure ref ops. + auto tx2 = storage->createTransaction(); + tx2->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/detached/all_1_1_0"); + tx2->commit(DB::NoCommitOptions{}); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/all_1_1_0")); + EXPECT_EQ(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/detached/all_1_1_0/metadata_version.txt"), + std::optional("3")); + + /// ATTACH: stage-rename within detached, then publish back into the live namespace. + auto tx3 = storage->createTransaction(); + tx3->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/detached/attaching_all_1_1_0"); + tx3->commit(DB::NoCommitOptions{}); + auto tx4 = storage->createTransaction(); + tx4->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/attaching_all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0"); + tx4->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached/attaching_all_1_1_0")); + EXPECT_EQ(storage->listDirectory("a11/a11a11a1-1111-4111-8111-111111111111/detached"), (std::vector{})); +} + +TEST(CASWiringOps, RemovalsDropRefsAndNamespaces) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "gone-soon"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin", "stays"); + tx->commit(DB::NoCommitOptions{}); + storage->store()->putNamespaceFile(wiringLife(*storage), "format_version.txt", "1\n"); + + /// The fast-removal path (all-tree Task 8, B123 evolution): per-file unlinks stage removal marks + /// (`content_removed`) but nothing durable changes until commit; removeDirectory() drops the + /// ref and supersedes any marks staged for it in the SAME transaction — still exactly one ref-drop, + /// zero repoints. `existsFile` below stays true because this whole sequence is one uncommitted + /// transaction (`tx2`), not because the unlink was a no-op. + auto tx2 = storage->createTransaction(); + tx2->unlinkFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", false, false); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); /// still committed + tx2->removeDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0")); + + /// DROP TABLE: removeRecursive on the table dir drops the live + detached namespaces. + tx2->removeRecursive("a11/a11a11a1-1111-4111-8111-111111111111", {}); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/format_version.txt")); +} + +/// The negative test that forbids hooking last-part removal as table-drop admission: removing a +/// table's ONLY part is indistinguishable, from that call alone, from a merge, a TTL cleanup, or a +/// `TRUNCATE` that leaves the table usable. The root must stay present, and a fresh part must still be +/// publishable into it. +TEST(CASWiringOps, LastRefRemovalIsNotNamespaceRemoval) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a66/a66a66a6-6666-4666-8666-666666666666/all_1_1_0/data.bin", "only-part"); + tx->commit(DB::NoCommitOptions{}); + storage->store()->putNamespaceFile( + storage->store()->namespaceLife(storage->liveNamespace("a66a66a6-6666-4666-8666-666666666666")), + "format_version.txt", "1\n"); + + auto tx2 = storage->createTransaction(); + tx2->removeDirectory("a66/a66a66a6-6666-4666-8666-666666666666/all_1_1_0"); + EXPECT_FALSE(storage->existsDirectory("a66/a66a66a6-6666-4666-8666-666666666666/all_1_1_0")); + EXPECT_TRUE(storage->existsDirectory("a66/a66a66a6-6666-4666-8666-666666666666")) + << "removing the table's last part must not be treated as DROP TABLE admission"; + + auto tx3 = storage->createTransaction(); + writeThroughTransaction(*tx3, "a66/a66a66a6-6666-4666-8666-666666666666/all_2_2_0/data.bin", "new-part"); + tx3->commit(DB::NoCommitOptions{}); + EXPECT_TRUE(storage->existsDirectory("a66/a66a66a6-6666-4666-8666-666666666666/all_2_2_0")) + << "the namespace never transitioned to Removing, so a fresh part publishes normally"; +} + +/// A files-only table root (no part ever published) becomes logically absent IMMEDIATELY once +/// `removeRecursive` durably completes the removal -- no GC round required. This is the same-call +/// synchronous half of the fix: `DROP TABLE ... SYNC` must not depend on GC latency to observe removal. +TEST(CASWiringOps, FilesOnlyTableRootRemovalIsImmediatelyAbsentWithoutGc) +{ + auto storage = openWiringStorage(); + const auto ns = storage->liveNamespace("a77a77a7-7777-4777-8777-777777777777"); + storage->store()->putNamespaceFile(storage->store()->namespaceLife(ns), "format_version.txt", "1\n"); + EXPECT_TRUE(storage->existsDirectory("a77/a77a77a7-7777-4777-8777-777777777777")); + + auto tx = storage->createTransaction(); + tx->removeRecursive("a77/a77a77a7-7777-4777-8777-777777777777", {}); + EXPECT_FALSE(storage->existsDirectory("a77/a77a77a7-7777-4777-8777-777777777777")) + << "the terminal remove_namespace transaction is durable synchronously"; + EXPECT_FALSE(storage->existsFile("a77/a77a77a7-7777-4777-8777-777777777777/format_version.txt")); +} + +/// REMOVED (all-tree-part-files Task 6): +/// `MutableTmpMoveOnCommittedPart` exercised `VersionMetadataOnDisk`'s OLD atomic-write dance — +/// autocommit `txn_version.txt.tmp`, then a standalone one-shot `moveFile(.tmp -> txn_version.txt)` +/// — via `ContentAddressedTransaction::moveFile` directly. That dance no longer exists in production: +/// Task 5's `supportsAtomicFileWrites` short-circuit makes `VersionMetadataOnDisk::storeInfoToData- +/// PartStorage` write `txn_version.txt` directly in one shot on a CA disk, with no `.tmp` file and no +/// rename ever produced. Task 9 completed the cleanup this comment used to defer: `moveFile`'s legacy +/// "rename FROM a committed mutable-per-part-file, source not staged in this transaction" branch is +/// now DELETED (it had been provably unreachable since Task 5, and rebuilding it against `entries` +/// would only add unused surface for a dead path). Coverage that remains valid: Task 5's own +/// capability test proves no `.tmp` file is ever created; `CASTransactionAllTree.CommittedTxnVersion- +/// StoreRepoints` (`gtest_ca_transaction.cpp`) proves the real, live path — a standalone write of +/// `txn_version.txt` directly onto an already-committed part — repoints correctly. + +TEST(CASWiringOps, VerbatimMoveAndUnlink) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_5.txt", "cmds"); + auto & ca_tx = dynamic_cast(*tx); + ca_tx.moveFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_5.txt", "a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt"); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_5.txt")); + ca_tx.unlinkFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt", false, false); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_5.txt")); +} + +TEST(CASWiringOps, UnlinkHonorsIfExistsForPartFiles) +{ + auto storage = openWiringStorage(); + const String path = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"; + + auto missing_tx = storage->createTransaction(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, + [&] { missing_tx->unlinkFile(path, /*if_exists=*/false, /*should_remove_objects=*/true); }); + + auto ignored_tx = storage->createTransaction(); + EXPECT_NO_THROW(ignored_tx->unlinkFile(path, /*if_exists=*/true, /*should_remove_objects=*/true)); + EXPECT_NO_THROW(ignored_tx->commit(DB::NoCommitOptions{})); + EXPECT_FALSE(storage->existsFile(path)); + + auto create_tx = storage->createTransaction(); + writeThroughTransaction(*create_tx, path, "payload"); + create_tx->commit(DB::NoCommitOptions{}); + ASSERT_TRUE(storage->existsFile(path)); + + auto existing_tx = storage->createTransaction(); + EXPECT_NO_THROW(existing_tx->unlinkFile(path, /*if_exists=*/false, /*should_remove_objects=*/true)); + EXPECT_NO_THROW(existing_tx->commit(DB::NoCommitOptions{})); + EXPECT_FALSE(storage->existsFile(path)); +} + +TEST(CASWiringOps, TableRenameMovesRefsFilesAndDetached) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "live"); + tx->commit(DB::NoCommitOptions{}); + auto tx2 = storage->createTransaction(); + tx2->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/detached/all_1_1_0"); /// one detached part + tx2->commit(DB::NoCommitOptions{}); + auto tx3 = storage->createTransaction(); + writeThroughTransaction(*tx3, "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin", "live2"); + tx3->commit(DB::NoCommitOptions{}); + storage->store()->putNamespaceFile(wiringLife(*storage), "format_version.txt", "1\n"); + + auto tx4 = storage->createTransaction(); + tx4->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111", "a22/a22a22a2-2222-4222-8222-222222222222"); + tx4->commit(DB::NoCommitOptions{}); + + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111")); + EXPECT_TRUE(storage->existsDirectory("a22/a22a22a2-2222-4222-8222-222222222222")); + EXPECT_TRUE(storage->existsFile("a22/a22a22a2-2222-4222-8222-222222222222/all_2_2_0/data.bin")); + EXPECT_TRUE(storage->existsFile("a22/a22a22a2-2222-4222-8222-222222222222/format_version.txt")); + EXPECT_TRUE(storage->existsDirectory("a22/a22a22a2-2222-4222-8222-222222222222/detached/all_1_1_0")); +} + +/// B126: RENAME TABLE move_namespace is idempotent — re-driving the SAME rename after it completed is a +/// clean no-op (the source namespace is already gone), so a partial-failure re-drive is safe. +TEST(CASWiringOps, TableRenameIsIdempotentOnRedrive) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "live"); + tx->commit(DB::NoCommitOptions{}); + storage->store()->putNamespaceFile(wiringLife(*storage), "format_version.txt", "1\n"); + + auto tx2 = storage->createTransaction(); + tx2->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111", "a22/a22a22a2-2222-4222-8222-222222222222"); + tx2->commit(DB::NoCommitOptions{}); + + /// Re-drive the identical rename: a11a11a1-1111-4111-8111-111111111111 is empty/gone, so every step no-ops; must not throw. + auto tx3 = storage->createTransaction(); + EXPECT_NO_THROW(tx3->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111", "a22/a22a22a2-2222-4222-8222-222222222222")); + tx3->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsFile("a22/a22a22a2-2222-4222-8222-222222222222/all_1_1_0/data.bin")); + EXPECT_TRUE(storage->existsFile("a22/a22a22a2-2222-4222-8222-222222222222/format_version.txt")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111")); +} + +/// B123: a verbatim-file move (get->put->remove, no native rename) is idempotent on re-drive — once the +/// source is gone but the destination is present, a re-driven move is a no-op, not a FILE_DOESNT_EXIST. +TEST(CASWiringOps, VerbatimMoveIsIdempotentOnRedrive) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_7.txt", "cmds"); + auto & ca_tx = dynamic_cast(*tx); + ca_tx.moveFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_7.txt", "a11/a11a11a1-1111-4111-8111-111111111111/mutation_7.txt"); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_7.txt")); + /// Re-drive: source gone, destination present → no-op (no throw). + EXPECT_NO_THROW(ca_tx.moveFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_7.txt", "a11/a11a11a1-1111-4111-8111-111111111111/mutation_7.txt")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/mutation_7.txt")); + /// Both source and destination absent → genuine missing source still throws. + EXPECT_ANY_THROW(ca_tx.moveFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mutation_8.txt", "a11/a11a11a1-1111-4111-8111-111111111111/mutation_8.txt")); +} + +/// B124: moveDirectory's staged-merge is source-wins, and a genuine collision (the same mutable file +/// staged under BOTH the source and destination part keys with DIFFERING bytes) fails loud instead of +/// silently dropping a just-written file. Identical bytes are a benign idempotent re-key. +/// +/// The "fails loud" collision throws LOGICAL_ERROR, which aborts the whole process in debug/sanitizer +/// builds (Exception.cpp's handle_error_code) instead of behaving like a catchable exception -- so +/// EXPECT_ANY_THROW only makes sense in a plain release build. CASWiringOpsDeathTest below proves the +/// SAME collision positively aborts under debug/sanitizer builds instead (same pattern as the existing +/// CASBlobDigestDeathTest precedent). +TEST(CASWiringOps, MoveDirectoryMutableCollisionPolicy) +{ +#ifndef DEBUG_OR_SANITIZER_BUILD + /// Differing bytes → fail loud. + { + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_x/txn_version.txt", "A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9/txn_version.txt", "B"); + auto & ca_tx = dynamic_cast(*tx); + EXPECT_ANY_THROW(ca_tx.moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_x", "a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9")); + } +#endif + /// Identical bytes → benign, no throw (source-wins, idempotent). Both parts carry real content so + /// the eager publish-at-rename builds a proper ref (a mutable-only staging would instead hit + /// updateRefPublishedAt on a not-yet-committed ref — unrelated to the collision policy under test). + /// data.bin must ALSO match now: all-tree Task 9 generalized the differing-bytes collision check + /// from the legacy mutable-file names to every entry, so a differing data.bin would (correctly) + /// throw too and defeat this block's "benign" premise. + { + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_y/data.bin", "d1"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_y/txn_version.txt", "SAME"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_8_8_8/data.bin", "d1"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_8_8_8/txn_version.txt", "SAME"); + auto & ca_tx = dynamic_cast(*tx); + EXPECT_NO_THROW(ca_tx.moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_y", "a11/a11a11a1-1111-4111-8111-111111111111/all_8_8_8")); + } +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +/// Debug/sanitizer-build counterpart to MoveDirectoryMutableCollisionPolicy's "differing bytes → fail +/// loud" case: LOGICAL_ERROR aborts the process here instead of throwing a catchable exception, so the +/// check must be a death test (same pattern as CASBlobDigestDeathTest in gtest_cas_blob_digest.cpp). +TEST(CASWiringOpsDeathTest, MoveDirectoryMutableCollisionPolicyAborts) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_x/txn_version.txt", "A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9/txn_version.txt", "B"); + auto & ca_tx = dynamic_cast(*tx); + EXPECT_DEATH({ ca_tx.moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_x", "a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9"); }, ""); +} +#endif + +/// D3 review pin: moveDirectory's staged-merge collision code has four (src build?, dst build?) +/// combinations. This one — destination already holds a staged PartWriteTxn, source has none — proved +/// confusable when the plan's author sketched a fix: a naive rewrite of the four-way branch can fall +/// through to `src_st.build->abandon()` on a null build. The merge must be a pure no-op on the +/// destination's build in this combination — no abandon, no adopt — while everything else (any +/// removal marks carried from the source) still merges in and the destination's own content +/// publishes exactly as staged. +/// +/// T9-review fix (all-tree-part-files): the ORIGINAL construction staged the source via a +/// `txn_version.txt` WRITE, relying on the pre-Task-6 "the mutable-file write path never calls +/// buildFor" fact to keep `src_st.build` null. Since Task 6/9, `writeFile`'s inline-candidate path +/// (which `txn_version.txt` now takes — it is an ordinary tree entry, not a mutable sidecar file) +/// unconditionally calls `buildFor` for ANY inline entry, so the source silently acquired a REAL +/// PartWriteTxn and this test drifted onto the *other* merge branch (`else if (src_st.build)`) without +/// failing — both branches produce the same externally-visible result (assertions passed either +/// way), so the drift was invisible. Fixed by staging the source via `unlinkFile` instead of a +/// write: Task 8's removal-mark staging (`content_removed`) is the one remaining staging shape that +/// genuinely never calls `buildFor` (`publishStaging`'s own `!st.build && ...` guard depends on +/// this), so `parts[src_key]` exists but `src_st.build` stays null again, restoring the test's +/// documented precondition. +/// +/// Made RED-able (the review's ask): `PartWriteTxn::abandon()` unconditionally emits a `BuildAbort` +/// `CasEvent` (`CasPartWriteTxn.cpp`) — this only happens if the buggy `else if (src_st.build)` branch +/// runs `src_st.build->abandon()`. Registering an event sink (`Cas::Pool::setEventSink`, the same +/// public test hook `gtest_cas_event_log.cpp` uses) and asserting no `BuildAbort` event fires is a +/// genuine behavioral discriminator between the two merge branches — not just "assertions pass +/// either way" — so a future regression that gives the source a PartWriteTxn again fails this test loudly. +TEST(CASWiringOps, MoveDirectoryOntoExistingDestinationBuildSurvives) +{ + std::vector events; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto storage = openWiringStorage(); + storage->store()->setEventSink([&](const DB::Cas::CasEvent & e) { events.push_back(e); }); + + /// unlinkFile now honors if_exists=false (triage #24, 8fc0c964a5b): the target must be real. Commit + /// it in its own transaction first so the removal below targets a genuinely-committed file, not a + /// never-existed path. + auto setup_tx = storage->createTransaction(); + writeThroughTransaction(*setup_tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_z/txn_version.txt", "creation_tid: (7,7,00000000-0000-0000-0000-000000000000)"); + setup_tx->commit(DB::NoCommitOptions{}); + + auto tx = storage->createTransaction(); + /// Destination already has a real blob upload staged -> a live PartWriteTxn. + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_7_7_7/data.bin", "dst-content"); + /// Source is staged with ONLY a removal mark (Task 8's content_removed staging) -> parts[src_key] + /// exists, but src_st.build stays null (unlinkFile never calls buildFor). + tx->unlinkFile("a11/a11a11a1-1111-4111-8111-111111111111/tmp_z/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + + auto & ca_tx = dynamic_cast(*tx); + EXPECT_NO_THROW(ca_tx.moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_z", "a11/a11a11a1-1111-4111-8111-111111111111/all_7_7_7")); + + /// The discriminator: no BuildAbort event means src_st.build->abandon() was never called during + /// the re-key merge, i.e. the intended neither-branch no-op merge ran, not the two-builds + /// merge-and-abandon branch. Checked right after the re-key so it stays scoped to the merge. + EXPECT_FALSE(std::any_of(events.begin(), events.end(), + [](const DB::Cas::CasEvent & e) { return e.type == DB::Cas::CasEventType::BuildAbort; })) + << "src_st.build->abandon() fired — the source unexpectedly has a real PartWriteTxn again"; + + /// [TXN-ONE-PIPELINE] the re-key does not publish; the destination's build is materialized only at + /// commit(). The destination's own build then publishes its own content untouched; the source's + /// removal mark names a path never committed anywhere, so it is a harmless no-op once merged into + /// the destination's (first-time-published) staging. + tx->commit(DB::NoCommitOptions{}); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_7_7_7")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_7_7_7/data.bin"), 11u); /// "dst-content" + EXPECT_FALSE(storage->tryGetInManifestBytes("a11/a11a11a1-1111-4111-8111-111111111111/all_7_7_7/txn_version.txt").has_value()); + + storage->store()->setEventSink(nullptr); +} + +TEST(CASWiringOps, FreezeViaHardLinksIntoShadow) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "frozen-bytes"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/metadata_version.txt", "7"); + tx->commit(DB::NoCommitOptions{}); + + /// `FREEZE` clones a committed part file-by-file into the shadow tree via hardlinks; the staged + /// shadow part publishes at commit under this server root. + auto tx2 = storage->createTransaction(); + tx2->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + tx2->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/metadata_version.txt", + "shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/metadata_version.txt"); + tx2->commit(DB::NoCommitOptions{}); + + EXPECT_TRUE(storage->existsDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsFile("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(storage->tryGetInManifestBytes("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/metadata_version.txt"), + std::optional("7")); + + /// UNFREEZE: removeRecursive of the backup root drops every shadow namespace under it. + auto tx3 = storage->createTransaction(); + tx3->removeRecursive("shadow/bk1", {}); + EXPECT_FALSE(storage->existsDirectory("shadow/bk1/store/a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->existsDirectory("shadow/bk1")); +} + +/// ==== M-W Task 8: in-flight read-your-writes (B59) ==== + +TEST(CASWiringInFlight, StagedFilesVisibleBeforeCommit) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj/data.bin", "proj-bytes"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/uuid.txt", "u-9"); + + /// B188 precommit-first: content blobs are PENDING (staged locally, not yet uploaded). So + /// tryGetInFlightStorageObjects returns {} — the pool object does not exist yet. The caller + /// (DataPartStorageOnDiskFull::prepareRead) falls back to tryGetInFlightFileSize to get the size + /// and then serves the content via tryReadFileInFlight (local temp file). File sizes and directory + /// overlay still work because they are driven by the staged tree entry, not the pool. + auto objects = tx->tryGetInFlightStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj/data.bin"); + EXPECT_FALSE(objects.has_value()); + EXPECT_EQ(tx->tryGetInFlightFileSize("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj/data.bin"), std::optional(10)); + EXPECT_EQ(tx->tryGetInFlightFileSize("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/uuid.txt"), std::optional(3)); + EXPECT_FALSE(tx->tryGetInFlightFileSize("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/missing.bin").has_value()); + + /// Bytes read back: a pending blob from the local temp file (B188); staged mutable bytes from memory. + { + auto buf = tx->tryReadFileInFlight("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj/data.bin", {}, std::nullopt); + ASSERT_TRUE(buf); + String read; + readStringUntilEOF(read, *buf); + EXPECT_EQ(read, "proj-bytes"); /// B188: served from local temp file (pending upload) + } + { + auto buf = tx->tryReadFileInFlight("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/uuid.txt", {}, std::nullopt); + ASSERT_TRUE(buf); + String read; + readStringUntilEOF(read, *buf); + EXPECT_EQ(read, "u-9"); + } + + /// The directory overlay answers for INNER dirs only (the PoC contract): the part dir itself + /// is FALSE so a rejected temporary part's removeIfNeeded takes the clean early-return path. + EXPECT_FALSE(tx->hasInFlightDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0")); + EXPECT_TRUE(tx->hasInFlightDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj")); + EXPECT_FALSE(tx->hasInFlightDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/q.proj")); + auto top = tx->listInFlightDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0"); + EXPECT_EQ(top, (std::vector{"p.proj", "uuid.txt"})); + EXPECT_EQ(tx->listInFlightDirectory("a11/a11a11a1-1111-4111-8111-111111111111/tmp_mut_all_1_1_0/p.proj"), + (std::vector{"data.bin"})); +} + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +/// ==== M-W Task 10: the GC scheduler end-to-end through the wiring ==== + +TEST(CASWiringGc, DroppedPartIsReclaimedByRounds) +{ + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "reclaim-me"); + tx->commit(DB::NoCommitOptions{}); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + const auto blob_key = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")[0].remote_path; + + auto tx2 = storage->createTransaction(); + tx2->removeDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); /// dropRef - the part is unreachable now + + /// Round 1 folds the drop and retires+deletes the part MANIFEST; the freed blob is retired+deleted + /// by a FOLLOWING round (next-round reclamation, M-C3). The steal needs one extra observation + /// window between rounds (the pacing scheduler is stable across these calls - each call after the + /// first re-acquires via renewal). + storage->runOneGcRoundForTest(); + storage->runOneGcRoundForTest(); + storage->runOneGcRoundForTest(); + + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + /// The relink offer (B7 part_manifest_v2): a reclaimed part is no longer a committed CA part here, + /// so getRelinkOffer offers NOTHING and the sender streams bytes — the documented fallback. + EXPECT_FALSE(exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").has_value()); + + /// A fresh identical write re-CREATES the content at the same key and reads back fine. + auto tx3 = storage->createTransaction(); + writeThroughTransaction(*tx3, "a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_0/data.bin", "reclaim-me"); + tx3->commit(DB::NoCommitOptions{}); + EXPECT_EQ(storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_0/data.bin")[0].remote_path, blob_key); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_0/data.bin")); +} + +/// B199 (real-path displacement reclamation, ported off the tree model to part manifests): re-writing +/// the SAME part path with DISTINCT content publishes a NEW part ManifestId over the ref (a true-removal +/// of the old owner manifest + an activation of the new one in the single ordered journal — no shared +/// content-addressed identity between the two parts). GC must reclaim the displaced (manifestA) unique +/// blobs while never losing the live (manifestB) closure. +/// +/// NOTE (port): the original repro pre-deleted treeA's TREE OBJECT before the fold to exercise the +/// tree-era inline-closure 404 path (the precommit `Add` carried treeA's closure INLINE so the fold +/// recorded edges without a `readTree`). That mechanism is gone: a part manifest carries its OWN blob +/// edges and the fold reads the ONE removal-target body to release them (a missing removal body clamps +/// + records an anomaly, never guesses). So this port drives the genuine manifest displacement WITHOUT +/// the out-of-band pre-delete twist — the reclamation contract (no leak / no loss) is what survives. +/// +/// PORT (rev. 15 displacement shape): a part is a single-owner ManifestId and `promote` is a PURE OWNER +/// MOVE (precommit→committed). Re-publishing over a LIVE committed ref does NOT emit a removal of the +/// displaced owner (the displaced manifest is not named in any event), so its blobs would never get a +/// -1 — there is no in-place "republish-over-committed". The genuine displacement that DOES journal a +/// true-removal is the real MergeTree pattern: DROP the old part (dropRef appends old→none, leaving the +/// old body present for the fold to read the -1 edges), THEN publish the new part. GC folds manifestA's +/// removal, retires its now-zero-in-degree blobs, and the recheck cleanup deletes the owner-removed +/// body. We do NOT pre-delete manifestA's body — only GC deletes an owner-removed body, after sealing +/// its decrements. +TEST(CASWiringGc, DisplacedTreeBlobsReclaimedThroughRealPath) +{ + auto storage = openWiringStorage(); + + /// Commit manifestA with unique content (data-A / mark-A), through the real precommit-first transaction. + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin", "data-A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.cmrk3", "mark-A"); + tx->commit(DB::NoCommitOptions{}); + } + const auto resolved_a = storage->store()->resolveRef(storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_0_0_0"); + ASSERT_TRUE(resolved_a.has_value()); + const DB::Cas::ManifestId manifest_a = resolved_a->manifest_id; + + /// DISPLACE (true-removal repoint): drop the old part so dropRef journals manifestA's removal + /// (old=committed(manifestA)→new=none) — this leaves manifestA's body PRESENT for the fold to read + /// its -1 edges. Then re-write the SAME part path with DISTINCT content (data-B / mark-B), which + /// publishes a NEW part ManifestId over the (now free) ref. Confirm the displacement is real: the + /// ref resolves to a DIFFERENT manifest. + { + auto tx = storage->createTransaction(); + tx->removeDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0"); + tx->commit(DB::NoCommitOptions{}); + } + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin", "data-B"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.cmrk3", "mark-B"); + tx->commit(DB::NoCommitOptions{}); + } + const auto resolved_b = storage->store()->resolveRef(storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_0_0_0"); + ASSERT_TRUE(resolved_b.has_value()); + ASSERT_FALSE(manifest_a == resolved_b->manifest_id) + << "the second write must displace the ref to a distinct part manifest (last-op-wins)"; + + /// Drive GC to a fixpoint. Displacement reclamation needs the next-round cascade (manifestA's + /// removal folds, its blobs hit zero in-degree, a following round retires+deletes them); give a + /// generous bound so the displaced closure fully drains. + for (int i = 0; i < 8; ++i) + storage->runOneGcRoundForTest(); + + const DB::Cas::FsckReport after = DB::Cas::runFsck(*storage->store(), /*detail=*/false); + EXPECT_EQ(after.dangling, 0u) << "displacement must never lose a reachable object (manifestB stays live)"; + EXPECT_GT(after.reachable, 0u) << "the live ref points at manifestB; manifestB's closure is reachable"; + /// The REAL path: `runOneGcRoundForTest` drives the production scheduler, so the displaced closure + /// is not merely recognized as unreachable but actually reclaimed -- recognition alone would leave + /// every displacement leaking one part's unique blobs forever. + EXPECT_EQ(after.unreachable, 0u) + << "manifestA's unique blobs (data-A / mark-A) must be RECLAIMED once the displacement folds, " + << "not just recognized as unreachable; unreachable=" << after.unreachable; +} + +/// ==== M-W Task 11 / B7: the DataPartsExchange facade (manifest relink, part_manifest_v2) ==== + +/// Publish-then-confirm (Task 14) split the receiver's adoption into `prepare` + `promote`, with the +/// interserver confirm interposed between them. The confirm belongs to `Fetcher`, not to the storage, so +/// the tests below that only care about the ADOPTION drive both halves back to back through this helper +/// -- which is exactly what `publishEntries` does for the atomic callers. `false` is the +/// `MechanismFallbackAllowed` outcome of either half: nothing published, the caller byte-fetches. +namespace +{ + +/// The RECEIVER's disk-relative staging path for every relink test below: the tmp-fetch dir of the +/// receiving table (a22...), which is a DIFFERENT table from the sender's (a11...) -- that is what makes +/// the "the sender's namespace id is ignored" assertions meaningful. `prepareAdoptFromManifest` is +/// addressed by path, exactly like `getRelinkOffer`, so the ref name is the router's business. +constexpr auto kReceiverTmpFetchPath = "a22/a22a22a2-2222-4222-8222-222222222222/tmp-fetch_all_1_1_0"; + +bool adoptPartFromManifestAndPromote(DB::IContentAddressedExchange & exchange, const String & part_path, + const String & manifest_bytes) +{ + std::unique_ptr prepared; + if (exchange.prepareAdoptFromManifest(part_path, manifest_bytes, prepared) + == DB::CaRelinkPrepare::MechanismFallbackAllowed) + return false; + EXPECT_NE(prepared, nullptr) << "a Prepared outcome must carry the handle that owes the terminal operation"; + return prepared->promote() == DB::CaRelinkPromote::Committed; +} + +} + +/// B7 sender side: getRelinkOffer returns the COMMITTED part's encoded PartManifest body — the +/// opaque payload the receiver decodes. The bytes must decode to the same entries the part was +/// published with; an absent part offers nothing (the sender streams bytes — the documented fallback). +/// Task 13 adds the second half of the offer: the confirm token, which must name the SAME manifest the +/// body carries. That equality is the offer's whole safety property — a token naming anything else +/// would have the receiver confirm a manifest whose entries it never adopted. +TEST(CASWiringExchange, GetRelinkOfferReturnsBodyAndTokenForCommittedPart) +{ + auto storage = openWiringStorage(); + /// Publish a real committed part (data.bin + a projection blob + mutable per-part files). + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + EXPECT_FALSE(exchange->getPoolUUID().empty()); + + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + EXPECT_FALSE(offer->manifest_bytes.empty()); + + /// The transferred body decodes to the SAME entries the part names — the blob entries AND the + /// per-part files (uuid.txt/metadata_version.txt are ordinary tree entries now, all-tree Task 6/9). + /// The sender's ManifestRef/namespace/digest are present but non-authoritative downstream. + const DB::Cas::PartManifest decoded = DB::Cas::decodePartManifest(offer->manifest_bytes); + ASSERT_EQ(decoded.entries.size(), 4u); + EXPECT_EQ(decoded.entries[0].path, "data.bin"); + EXPECT_EQ(decoded.entries[0].ref.digest.toU128(), u128Of("payload-A")); + EXPECT_EQ(decoded.entries[2].path, "p.proj/data.bin"); + EXPECT_EQ(decoded.entries[2].ref.digest.toU128(), u128Of("payload-B")); + + /// The token: it decodes, it names this mount and this pool, and it names the manifest that the + /// body just decoded to. + const auto token = DB::decodeCasRelinkSourceToken(offer->confirm_token); + ASSERT_TRUE(token.has_value()) << "the sender minted a token its own decoder rejects: " << offer->confirm_token; + EXPECT_EQ(token->pool_uuid, exchange->getPoolUUID()); + EXPECT_EQ(token->root_namespace, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111").string()); + EXPECT_EQ(token->ref_name, "all_1_1_0"); + EXPECT_EQ(token->part_name, "all_1_1_0"); + EXPECT_EQ(token->manifest_ref_text, DB::Cas::manifestRefDebugString(decoded.ref)); + EXPECT_TRUE(exchange->ownsNamespace(token->server_root_id, token->root_namespace)) + << "the minted token must route back to the mount that minted it"; + + /// An absent part is not a committed CA part here -> no offer. + EXPECT_FALSE(exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_9_9_9").has_value()); +} + +/// B7 receiver side (the core): take a COMMITTED part's transferred manifest bytes and adopt them into +/// a DIFFERENT table namespace WITHOUT moving any blob body (blobs are shared by hash in the pool). +/// The receiver stages its OWN fresh local manifest, precommitAdd + promote it, and reports success. +/// Asserts: success; the adopted ref is live + loadable; the receiver's ManifestId differs from the +/// sender's (no shared identity); the ref lives in the RECEIVER namespace (no cross-namespace adoption); +/// and NO blob body was uploaded by the receiver (the put-counter stays flat across adopt). +TEST(CASWiringExchange, AdoptPartFromManifestPublishesFreshLocalManifest) +{ + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + publishWiredPart(*storage, sender_ns, "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + const String & bytes = offer->manifest_bytes; + const DB::Cas::ManifestId sender_id = + storage->store()->resolveRef(sender_ns, "all_1_1_0")->manifest_id; + + /// Count blob PUTs over the adopt: a manifest relink must NOT upload any blob body (the blobs are + /// already in the shared pool, adopted by hash). We assert via the blob keys' presence/incarnation: + /// the receiver never overwrites or re-creates the blobs — their head tokens are unchanged. + const auto data_key = storage->store()->layout().blobKey(idOf("payload-A")); + const auto proj_key = storage->store()->layout().blobKey(idOf("payload-B")); + const auto data_tok_before = storage->store()->backend().head(data_key).token; + const auto proj_tok_before = storage->store()->backend().head(proj_key).token; + + /// Adopt into a DIFFERENT table (a22a22a2-2222-4222-8222-222222222222). The transferred body's root_namespace_id is the sender's + /// (a11a11a1-1111-4111-8111-111111111111) — the receiver must IGNORE it and use a22a22a2-2222-4222-8222-222222222222. + const bool ok = adoptPartFromManifestAndPromote(*exchange, kReceiverTmpFetchPath, bytes); + EXPECT_TRUE(ok); + + /// The adopted ref is live in the RECEIVER namespace and loadable. + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + auto receiver_resolved = storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0"); + ASSERT_TRUE(receiver_resolved.has_value()); + const DB::Cas::PartManifest receiver_manifest = + storage->store()->readManifest(receiver_resolved->manifest_id); + ASSERT_EQ(receiver_manifest.entries.size(), 4u); + EXPECT_EQ(receiver_manifest.entries[0].ref.digest.toU128(), u128Of("payload-A")); + + /// FRESH receiver-local identity: a DIFFERENT ManifestId from the sender's, in the RECEIVER namespace. + EXPECT_FALSE(sender_id == receiver_resolved->manifest_id) + << "the receiver must mint its OWN manifest id, not share the sender's"; + EXPECT_EQ(receiver_resolved->manifest_id.root_namespace.string(), receiver_ns.string()) + << "the adopted manifest must live in the receiver namespace (derived from table_uuid), not the sender's"; + EXPECT_FALSE(receiver_ns.string() == sender_ns.string()); + + /// NO blob body was uploaded: the shared blobs' incarnations are untouched by the adopt. + EXPECT_EQ(storage->store()->backend().head(data_key).token, data_tok_before) + << "adopt-from-manifest must not re-upload a blob already in the shared pool"; + EXPECT_EQ(storage->store()->backend().head(proj_key).token, proj_tok_before); +} + +/// B7 fail-closed: if a referenced blob is absent/condemned in the pool, adoptPartFromManifest must +/// promote-abort and return FALSE (NOT throw) so the caller byte-fetches — exactly where the old pin +/// protocol fell back. Nothing is published (no dangling ref). +TEST(CASWiringExchange, AdoptFailsClosedAndFallsBackOnCondemnedBlob) +{ + /// §4 manifest-trust (test name is legacy — adopt no longer fails closed on a raced pool blob): + /// adoptPartFromManifest runs the receiver's local promote, which TRUSTS the committed-source adopted + /// leaves via the durable manifest edge — no per-file HEAD/loadMeta probe on the pool blobs. So even if + /// a pool blob raced to absent, adopt SUCCEEDS and publishes the receiver ref. This matches ordinary + /// ReplicatedMergeTree interserver trust: the sender served the manifest from a LIVE part whose refs pin + /// the blobs at in-degree >= 1, so this scenario cannot arise on the real fetch path; a genuinely-absent + /// adopted blob is an fsck finding, not an adopt-time abort. + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + publishWiredPart(*storage, sender_ns, "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + const String & bytes = offer->manifest_bytes; + + /// Artificially delete a referenced pool blob — the live-sender invariant excludes this on the real + /// path; §4 promote does not re-probe it, so adopt trusts the manifest edge and publishes. + const auto data_key = storage->store()->layout().blobKey(idOf("payload-A")); + const auto h = storage->store()->backend().head(data_key); + ASSERT_TRUE(h.exists); + ASSERT_EQ(storage->store()->backend().deleteExact(data_key, h.token).kind, + DB::Cas::DeleteOutcome::Kind::Deleted); + + /// §4: promote trusts the adopted leaves — no re-probe — so adopt SUCCEEDS (returns true) and publishes. + const bool ok = adoptPartFromManifestAndPromote(*exchange, kReceiverTmpFetchPath, bytes); + EXPECT_TRUE(ok) << "§4: adopt trusts the manifest edge; a raced pool blob is not re-probed at promote"; + + /// The receiver ref publishes (the D4 trade-off), and the deleted pool blob surfaces via fsck's + /// reachable-but-absent scan (the backstop — INV-NO-DANGLE-via-fsck). + EXPECT_TRUE(storage->store()->resolveRef(storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"), "tmp-fetch_all_1_1_0").has_value()); + const DB::Cas::FsckReport rep = DB::Cas::runFsck(*storage->store(), /*detail=*/true); + EXPECT_GE(rep.dangling, 1u) << "§4 D4 backstop: the deleted pool blob must surface as an fsck dangling " + "finding (dangling=" << rep.dangling << ")"; +} + +/// All-tree task 7/9: relink self-containment. Task 6 routes uuid.txt/metadata_version.txt through +/// the content path, so a committed part's manifest ENTRIES already carry these files — the receiver +/// no longer needs a mutable_files sidecar to reconstruct them. Task 9 completed the cleanup: +/// `adoptPartFromManifest` no longer even HAS a sidecar parameter (Fetcher::relinkPartToDisk's call +/// site simply dropped the argument). This publishes a part whose per-part files are ordinary +/// manifest entries and adopts it, mirroring the post-task-9 call site exactly. +TEST(CASWiringExchange, AdoptPartFromManifestSelfContainedWithoutMutableFilesSidecar) +{ + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + + DB::Cas::PartWriteInfo info; + info.intended_ref = sender_ns.string() + "/all_1_1_0"; + info.intended_namespace = sender_ns; + auto build = storage->store()->beginPartWrite(info); + const auto id = build->stageManifest( + {wiringBlobEntry("data.bin", "payload-A"), + wiringBlobEntry("uuid.txt", "payload-uuid"), + wiringBlobEntry("metadata_version.txt", "payload-mv")}); + build->precommitAdd(sender_ns, "all_1_1_0", id); + build->putBlob(idOf("payload-A"), DB::Cas::BlobSource::fromString("payload-A")); + build->putBlob(idOf("payload-uuid"), DB::Cas::BlobSource::fromString("payload-uuid")); + build->putBlob(idOf("payload-mv"), DB::Cas::BlobSource::fromString("payload-mv")); + build->promote(sender_ns, "all_1_1_0", build->buildId(), id); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + const String & bytes = offer->manifest_bytes; + const DB::Cas::PartManifest decoded = DB::Cas::decodePartManifest(bytes); + ASSERT_EQ(decoded.entries.size(), 3u) << "uuid.txt/metadata_version.txt travel as ordinary entries"; + + /// No sidecar parameter to pass anymore — exactly what Fetcher::relinkPartToDisk's call looks like + /// now that the manifest is self-contained (no reconstruction from a wire-transferred header). + const bool ok = adoptPartFromManifestAndPromote(*exchange, kReceiverTmpFetchPath, bytes); + EXPECT_TRUE(ok); + + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + auto resolved = storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0"); + ASSERT_TRUE(resolved.has_value()); + + const DB::Cas::PartManifest receiver_manifest = storage->store()->readManifest(resolved->manifest_id); + ASSERT_EQ(receiver_manifest.entries.size(), 3u); + bool has_uuid_entry = false; + bool has_metadata_version_entry = false; + for (const auto & entry : receiver_manifest.entries) + { + if (entry.path == "uuid.txt") + has_uuid_entry = true; + if (entry.path == "metadata_version.txt") + has_metadata_version_entry = true; + } + EXPECT_TRUE(has_uuid_entry) << "uuid.txt must read back as an ordinary content entry, not mutable_files"; + EXPECT_TRUE(has_metadata_version_entry) + << "metadata_version.txt must read back as an ordinary content entry, not mutable_files"; +} + +/// Publish-then-confirm, receiver half (Task 14): `prepare` must make the receiver's `+1` DURABLE while +/// publishing NOTHING. That combination is the protocol -- the durable `+1` is what a later `yes` is +/// worth anything against, and the absent committed ref is what makes an unproven source cost nothing. +TEST(CASWiringExchange, PrepareAdoptIsDurableButPublishesNothingUntilPromote) +{ + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + publishWiredPart(*storage, sender_ns, "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + std::unique_ptr prepared; + ASSERT_EQ(exchange->prepareAdoptFromManifest(kReceiverTmpFetchPath, offer->manifest_bytes, prepared), + DB::CaRelinkPrepare::Prepared); + ASSERT_NE(prepared, nullptr); + + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0").has_value()) + << "prepare must not commit the ref -- the source has not been asked anything yet"; + EXPECT_EQ(storage->store()->livePrecommitsForTest(receiver_ns).size(), 1u) + << "prepare must leave the receiver's +1 durable, or a later confirm proves nothing"; + + EXPECT_EQ(prepared->promote(), DB::CaRelinkPromote::Committed); + EXPECT_TRUE(storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0").has_value()); + EXPECT_TRUE(storage->store()->livePrecommitsForTest(receiver_ns).empty()) + << "promote moves the binding out of the precommit view"; +} + +/// The unproven-source branch of the taxonomy (row 3), at the storage seam: `abort` releases the durable +/// `+1` and publishes nothing. A leaked same-epoch precommit is reclaimed by nothing -- not the +/// prior-epoch stale sweep, not GC -- so this removal is the ONLY thing standing between an unproven +/// confirm and permanently retained blobs. +TEST(CASWiringExchange, AbortedPrepareReleasesThePrecommitAndPublishesNothing) +{ + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + publishWiredPart(*storage, sender_ns, "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + std::unique_ptr prepared; + ASSERT_EQ(exchange->prepareAdoptFromManifest(kReceiverTmpFetchPath, offer->manifest_bytes, prepared), + DB::CaRelinkPrepare::Prepared); + ASSERT_NE(prepared, nullptr); + ASSERT_EQ(storage->store()->livePrecommitsForTest(receiver_ns).size(), 1u); + + prepared->abort(); + EXPECT_TRUE(storage->store()->livePrecommitsForTest(receiver_ns).empty()) + << "abort must append the exact precommit removal, not merely drop the transaction"; + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0").has_value()) + << "an aborted relink must leave no committed ref behind"; + + /// A second `abort` -- what the scope guard does after an explicit one -- must be a silent no-op + /// rather than an error, and destruction of an aborted handle must not re-drive anything. + prepared->abort(); + prepared.reset(); + EXPECT_TRUE(storage->store()->livePrecommitsForTest(receiver_ns).empty()); +} + +/// The `MechanismFallbackAllowed` branch (taxonomy row 2): an undecodable manifest is a mechanism +/// failure, not a source failure -- the sender still has the part, so the receiver byte-fetches. Nothing +/// may be staged, because there is no handle to abort it with. +TEST(CASWiringExchange, PrepareAdoptOfAnUndecodableManifestAllowsTheByteFallback) +{ + auto storage = openWiringStorage(); + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + std::unique_ptr prepared; + EXPECT_EQ(exchange->prepareAdoptFromManifest(kReceiverTmpFetchPath, "not a manifest at all", prepared), + DB::CaRelinkPrepare::MechanismFallbackAllowed); + EXPECT_EQ(prepared, nullptr) << "no handle may be returned when nothing was staged"; + EXPECT_TRUE(storage->store()->livePrecommitsForTest(receiver_ns).empty()); + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0").has_value()); +} + +/// B66b: a relink whose TARGET is a DETACHED part dir -- what `FETCH PARTITION ... TO detached` now +/// does instead of streaming bytes. Nothing about the detached case is special-cased on the receiver: +/// `Fetcher::relinkPartToDisk` hands over the staging path under the `detached/` parent and the router +/// folds it onto a `detached/`-prefixed ref in the table's OWN namespace, exactly as every other read +/// and write of a detached part is routed. +/// +/// The load-bearing assertion is the NEGATIVE one. A detached fetch must publish a detached ref and +/// nothing else: a live ref of the same name would make an un-attached part visible to the table, which +/// is the one way a detached target could differ from the active one in a way that matters. +TEST(CASWiringExchange, AdoptIntoADetachedTargetPublishesADetachedRefAndNoLiveRef) +{ + auto storage = openWiringStorage(); + const auto sender_ns = storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"); + publishWiredPart(*storage, sender_ns, "all_1_1_0"); + + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + auto offer = exchange->getRelinkOffer("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0"); + ASSERT_TRUE(offer.has_value()); + + /// The receiver's staging path under the detached parent -- the path `relinkPartToDisk` composes + /// with `to_detached`, and the same one `downloadPartToDisk` would have written bytes into. + const String detached_tmp_path + = "a22/a22a22a2-2222-4222-8222-222222222222/detached/tmp-fetch_all_1_1_0"; + EXPECT_TRUE(adoptPartFromManifestAndPromote(*exchange, detached_tmp_path, offer->manifest_bytes)); + + const auto receiver_ns = storage->liveNamespace("a22a22a2-2222-4222-8222-222222222222"); + EXPECT_TRUE(storage->store()->resolveRef(receiver_ns, "detached/tmp-fetch_all_1_1_0").has_value()) + << "the detached target must publish the `detached/`-prefixed ref in the table's own namespace"; + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "tmp-fetch_all_1_1_0").has_value()) + << "a detached fetch must NOT publish a live ref of the same name"; + + /// The adopted part reads back through the ordinary path surface, blobs and per-part files alike -- + /// no bytes were transferred for any of them. + EXPECT_TRUE(storage->existsFile(detached_tmp_path + "/data.bin")); + EXPECT_TRUE(storage->existsFile(detached_tmp_path + "/p.proj/data.bin")); + EXPECT_TRUE(storage->existsFile(detached_tmp_path + "/uuid.txt")); + + /// Finalization, unchanged by this task: `IMergeTreeDataPart::renameTo(detached/)` is a + /// moveDirectory of the staged dir to its final detached name, which on a content-addressed disk is + /// a ref repoint WITHIN the same namespace -- the same shape the active path's + /// `renameTempPartAndReplace` uses, and the reason the relinked detached part needs no new + /// finalization of its own. + { + auto tx = storage->createTransaction(); + tx->moveDirectory(detached_tmp_path, "a22/a22a22a2-2222-4222-8222-222222222222/detached/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + EXPECT_TRUE(storage->existsFile( + "a22/a22a22a2-2222-4222-8222-222222222222/detached/all_1_1_0/data.bin")); + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "detached/tmp-fetch_all_1_1_0").has_value()); + EXPECT_EQ(storage->detachedRefNames(receiver_ns), (std::vector{"detached/all_1_1_0"})); + EXPECT_FALSE(storage->store()->resolveRef(receiver_ns, "all_1_1_0").has_value()) + << "the detached finalization must stay inside the `detached/` ref space"; +} + +/// A relink target that is not a part DIRECTORY is a caller bug, and it must be loud rather than +/// answered with `MechanismFallbackAllowed`: the byte fetch that a fallback invites would write to the +/// same wrong place. The table dir stands in for the whole class (a file inside a part, a FREEZE shadow +/// path, a bare `detached` container) -- all of them route to something that is not a part ref. +/// +/// The refusal throws LOGICAL_ERROR, which aborts the whole process in debug/sanitizer builds +/// (Exception.cpp's handle_error_code) instead of behaving like a catchable exception -- so the +/// EXPECT_THROW form only makes sense in a plain release build, and CASWiringExchangeDeathTest below +/// proves the SAME refusals positively abort under debug/sanitizer builds instead (same pattern as +/// CASWiringOpsDeathTest above). +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASWiringExchange, PrepareAdoptRefusesATargetThatIsNotAPartDirectory) +{ + auto storage = openWiringStorage(); + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + + std::unique_ptr prepared; + EXPECT_THROW(exchange->prepareAdoptFromManifest( + "a22/a22a22a2-2222-4222-8222-222222222222", std::string{}, prepared), + DB::Exception); + EXPECT_THROW(exchange->prepareAdoptFromManifest( + "a22/a22a22a2-2222-4222-8222-222222222222/tmp-fetch_all_1_1_0/data.bin", + std::string{}, prepared), + DB::Exception); + EXPECT_THROW(exchange->prepareAdoptFromManifest( + "shadow/bk1/store/a22/a22a22a2-2222-4222-8222-222222222222/all_1_1_0", + std::string{}, prepared), + DB::Exception); + EXPECT_EQ(prepared, nullptr); +} +#else +TEST(CASWiringExchangeDeathTest, PrepareAdoptRefusesATargetThatIsNotAPartDirectoryAborts) +{ + auto storage = openWiringStorage(); + auto * exchange = dynamic_cast(storage.get()); + ASSERT_NE(exchange, nullptr); + + std::unique_ptr prepared; + EXPECT_DEATH(exchange->prepareAdoptFromManifest( + "a22/a22a22a2-2222-4222-8222-222222222222", std::string{}, prepared), + "does not address a content-addressed part directory"); + EXPECT_DEATH(exchange->prepareAdoptFromManifest( + "a22/a22a22a2-2222-4222-8222-222222222222/tmp-fetch_all_1_1_0/data.bin", + std::string{}, prepared), + "does not address a content-addressed part directory"); + EXPECT_DEATH(exchange->prepareAdoptFromManifest( + "shadow/bk1/store/a22/a22a22a2-2222-4222-8222-222222222222/all_1_1_0", + std::string{}, prepared), + "does not address a content-addressed part directory"); + EXPECT_EQ(prepared, nullptr); +} +#endif + +/// ==== Commit atomicity (B122): a publish failing mid-loop must not leave a PARTIAL commit ==== + +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; + extern const int CORRUPTED_DATA; + extern const int READONLY; +} + +namespace +{ + +/// A LocalObjectStorage whose writeObject can be armed to throw — the single seam needed to drive a +/// backend write failure at a chosen point. The hook runs BEFORE the write is created; throwing from +/// it fails the put exactly as a real backend error would. Everything else delegates to the base. +class FaultyLocalObjectStorage : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::function on_write; + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + if (on_write) + on_write(object.remote_path); + return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } +}; + +/// True for a per-part manifest BODY object (<...>/cas/manifests///.zst) +/// — the FIRST durable object `publishStaging` writes for a part (via `PartWriteTxn::stageManifest`). Since Task B +/// (chaos-tolerance-report) that write rides the CAS request controller: a transient fault is retried +/// (budgeted attempts + resolve-before-reissue), so an injected fault must be PERSISTENT to fail the +/// publish — the controller exhausts its budget and `stageManifest` throws ABORTED out of `publishStaging`. +/// Exactly one body per part (retries re-PUT the same per-part key), so counting FIRST attempts isolates +/// part publishes one-for-one. Ref-log txns (`cas/ns/stream/.../_log/...`), tree blobs (`blobs/`), GC state +/// (`gc/`) and verbatim files are excluded. +/// +/// The suffix is taken from `storedSuffix(FormatId::PartManifest)` (the registered v3 stored suffix, now +/// `.zst`) rather than hard-coded: codecs-v3 phase-3 made the part manifest an Always-compressed text +/// object, changing the body key from the pre-v3 `.proto` to `.zst`. The old hard-coded +/// `.ends_with(".proto")` stopped matching after that cutover, so the fault never fired and this +/// (test-local) predicate silently no-op'd — the same failure mode this comment already recorded for the +/// earlier `RootShardManifest` removal (commit `318291fe5e5`, whose all-digits key stopped matching). +/// Sourcing the suffix from the format registry keeps the predicate correct across future +/// compression-policy changes. +bool isPartManifestBodyPath(const std::string & path) +{ + return path.find("/cas/manifests/") != std::string::npos + && path.ends_with(DB::Cas::storedSuffix(DB::Cas::FormatId::PartManifest)); +} + +std::shared_ptr makeFaultyStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("ca_b122_" + unique)).string(); + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + return std::make_shared(DB::LocalObjectStorageSettings("test", root, /*read_only_=*/false)); +} + +} + +TEST(CASWiringWrite, PartialCommitRollsBackPublishedParts) +{ + auto faulty = makeFaultyStorageForTest(); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b122_scratch"); + auto storage = std::make_shared( + faulty, "pool", "srv1", "", nullptr, settings); + storage->startup(); + /// The manifest-body PUT rides the CAS request controller, whose inter-attempt backoff would + /// otherwise serve the REAL capped-exponential sleeps (~56s at the default budget) while the + /// persistent injected fault exhausts the whole attempt budget. Neutralize only the sleeps — the + /// retry/exhaustion/rollback semantics under test are unchanged. + storage->store()->setCasRetrySleepForTest([](uint64_t) {}); + + /// Two parts in ONE transaction, published sequentially at commit (the staging map orders all_1_1_0 + /// before all_2_2_0). writeThroughTransaction only STAGES to local temp files here — the pool writes + /// (manifest bodies, blob uploads, ref-log promotes) all happen later, inside commit's publishStaging. + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "content-A"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/data.bin", "content-B"); + + /// Fail the SECOND part's manifest-body write (all_2_2_0's stageManifest) — by then all_1_1_0 has + /// fully published (its manifest body + blob + promoted ref). A pre-B122 commit() would leave + /// all_1_1_0 durably visible: a partial commit. PERSISTENT (`>= 2`, not one-shot): the manifest + /// body PUT rides the CAS request controller (Task B), which absorbs a transient fault by design — + /// only a fault that outlasts the whole attempt budget fails the publish (as ABORTED). + /// CORRUPTED_DATA (not LOGICAL_ERROR): `handle_error_code` (Exception.cpp) aborts the whole + /// process for LOGICAL_ERROR under debug/sanitizer builds, since that code means "an internal + /// invariant broke" there -- but this is a simulated BACKEND write failure, not an invariant + /// violation, so it must stay a catchable exception. CORRUPTED_DATA keeps the exact same + /// `isDeterministicLocalFailure` classification LOGICAL_ERROR had (CasRequestControl.cpp), so the + /// controller's retry/exhaustion behavior under test is unchanged. + int manifest_writes = 0; + faulty->on_write = [&](const std::string & path) + { + if (isPartManifestBodyPath(path) && ++manifest_writes >= 2) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected publish failure (B122)"); + }; + + EXPECT_THROW(tx->commit(DB::NoCommitOptions{}), DB::Exception); + + /// All-or-nothing: the part that DID publish must have been rolled back (commit's compensating + /// `dropRefIfMatches`, keyed on the exact `CommitOutcome` `all_1_1_0`'s own publish produced). + /// Disarm first so the read-back assertions run clean — the rollback itself only writes ref-log + /// ops, never a manifest body, so it does not re-trip the count-2 fault. + faulty->on_write = nullptr; + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0")); +} + +TEST(CASWiringReadOnly, ObserveOnlyOpenReadsButRejectsWrites) +{ + /// 1. Writable storage publishes a part into a fixed root. + const auto root = (std::filesystem::temp_directory_path() + / ("ca_ro_" + std::to_string(::getpid()))).string(); + std::error_code ec; std::filesystem::remove_all(root, ec); std::filesystem::create_directories(root, ec); + auto writable_os = std::make_shared( + DB::LocalObjectStorageSettings("test", root, /*read_only_=*/false)); + { + auto w_settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_ro_scratch"); + auto w = std::make_shared( + writable_os, "pool", "srv1", "", nullptr, w_settings); + w->startup(); + auto tx = w->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "ro-bytes"); + tx->commit(DB::NoCommitOptions{}); + } + + /// 2. Read-only object storage over the SAME root => observe-only metadata storage. + auto ro_os = std::make_shared( + DB::LocalObjectStorageSettings("test", root, /*read_only_=*/true)); + /// Same `server_root_id` as the writer: live namespaces are rooted by configured layout identity, so an + /// observe-only mount reads the same server-root's data — the WORM scenario. + auto ro_settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_ro_scratch2"); + /// An explicit S3-staging selection still opens read-only without native-copy support because + /// this mount cannot enter staged publication. + ro_settings[DB::ContentAddressedSetting::staging_backend] = "s3"; + ro_settings.validate(); + auto ro = std::make_shared( + ro_os, "pool", "srv1", "", nullptr, ro_settings); + ro->startup(); /// must NOT throw: read-only mounts cannot enter staged publication + + EXPECT_TRUE(ro->isReadOnly()); + /// Reads work: + EXPECT_TRUE(ro->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(ro->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 8u); + /// Writes fail closed: + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::READONLY, + [&] { ro->createTransaction(); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::READONLY, + [&] + { + std::unique_ptr prepared; + ro->prepareAdoptFromManifest("a11/a11a11a1-1111-4111-8111-111111111111/tmp-fetch", std::string{}, prepared); + }); +} + +TEST(CASWiringRead, UnsetPublishedAtMsReturnsEpoch) +{ + /// A ref published without a stamp (published_at_ms == 0, the default) must return the epoch + /// (Poco::Timestamp(0)) rather than throwing: stamps only feed cleanup TTLs and system tables, + /// so a missing stamp is harmless. + auto storage = openWiringStorage(); + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "x"); + tx->commit(DB::NoCommitOptions{}); + + /// Ensure published_at_ms is unset (the default is 0). + storage->store()->updateRefPublishedAt(storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0", + [](DB::Cas::RefPublishedAtUpdate & r) { r.published_at_ms = 0; }); + + EXPECT_EQ(storage->getLastModified("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0").epochTime(), 0); +} + +/// ==== B188 precommit-first order invariant (Task 6) ==== +/// +/// A RecordingLocalObjectStorage records the four IObjectStorage methods the CA emulated-mode backend +/// uses on the commit path — writeObject (PUT), exists + getObjectMetadata (the HEAD), and readObject +/// (the GET) — as (op_name, logical_key). "Logical" means the bare pool key (without the emu_root +/// prefix) — the same string the Layout functions produce, so the `/blobs/`, `/trees/`, and opaque +/// ref-stream (`/cas/ns/stream//`) substring tests are unambiguous. +/// +/// After commit the test asserts: the FIRST write that appends the create-precommit owner event (the +/// first durable CAS to the target ROOT SHARD's key — owner_kind == Precommit; the converged rev. 15 +/// model has NO `_precommits` namespace, the precommit binding lives in the target shard's journal) +/// happened before ALL ops (read OR write) on keys containing "/blobs/" or "/trees/". The precommit +/// owner record is what pins the in-flight build-root closure so GC cannot reclaim the not-yet-uploaded +/// content objects; therefore every pool op touching a content blob or the manifest tree must be AFTER +/// the precommit owner record is durably written. The READ gating is the heart of the B188 fix: the +/// original bug was an EAGER HEAD on a content blob during staging, before any precommit protection +/// existed — a write-only assertion would not catch its reintroduction. + +namespace +{ + +/// Records the four IObjectStorage methods the CA emulated-mode backend uses on the commit path +/// (writeObject/exists/getObjectMetadata/readObject). listObjects/copyObject are deliberately NOT +/// overridden — they are not on the commit path the order invariant gates. +class RecordingLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + struct Record + { + std::string op; /// "writeObject" | "exists" | "getObjectMetadata" | "readObject" + std::string key; /// logical (emu_root stripped) + }; + + /// Append-only; mutable so the const read methods (exists/readObject/tryGetObjectMetadata) can + /// record. No mutex — these tests are single-threaded. + mutable std::vector ops; + + /// Strip the common-key-prefix (emu_root) to recover the logical key. The emu_root is returned by + /// getCommonKeyPrefix() and always ends with a path separator in LocalObjectStorage. + std::string toLogical(const std::string & physical) const + { + const std::string root = getCommonKeyPrefix(); + std::string logical; + if (!root.empty() && physical.starts_with(root)) + logical = physical.substr(root.size()); + else + logical = physical; + /// Strip any leading slash left after prefix removal. + if (!logical.empty() && logical.front() == '/') + logical = logical.substr(1); + return logical; + } + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + ops.push_back({"writeObject", toLogical(object.remote_path)}); + return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } + + /// Backs the CA backend's `head` (emuExists) and gates its `get` (emuExists before emuRead). + bool exists(const DB::StoredObject & object) const override + { + ops.push_back({"exists", toLogical(object.remote_path)}); + return DB::LocalObjectStorage::exists(object); + } + + /// Backs the CA backend's `head` size/attributes lookup (emuPath stat). + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + ops.push_back({"getObjectMetadata", toLogical(path)}); + return DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + } + + /// Backs the CA backend's `get` body read (readObjectRanged). + std::unique_ptr readObject( + const DB::StoredObject & object, + const DB::ReadSettings & read_settings, + std::optional read_hint, + bool use_external_buffer, + bool restrict_seek) const override + { + ops.push_back({"readObject", toLogical(object.remote_path)}); + return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); + } +}; + +std::shared_ptr makeRecordingStorageForTest(const std::string & tag) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("ca_b188_" + tag + "_" + unique)).string(); + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + return std::make_shared( + DB::LocalObjectStorageSettings("test", root, /*read_only_=*/false)); +} + +/// True for a durable ref-object write key under `/cas/ns/stream/`. In the snapshot+log ref model the +/// writer's first durable ref write on the precommit path is an immutable transaction-log object +/// (`<...>/cas/ns/stream//_log/.zst`); a published table snapshot is +/// `<...>/_snap/.zst`. The predicate anchors on whichever durable ref write comes first. It +/// excludes blobs (`/blobs/`), part-manifests (`/cas/manifests/...`), GC state (`/gc/`), and verbatim +/// files (`/_files/...`). +bool isRefWriteKey(const std::string & key) +{ + if (key.find("/cas/ns/stream/") == std::string::npos) + return false; + return key.find("/_log/") != std::string::npos || key.find("/_snap/") != std::string::npos; +} + +/// Index of the first writeObject that durably appends the create-precommit ref transaction — i.e. the +/// first durable write (writeObject) of a ref-object key (a `_log/` object in the snapshot+log +/// model). Anchors on the WRITE, not on any op: recovery READS the ref prefix before the durable write, +/// so an any-op scan would anchor on that READ rather than the durable write. Returns -1 if no ref write +/// was recorded. +int firstPrecommitWriteIdx(const std::vector & log) +{ + for (int i = 0; i < static_cast(log.size()); ++i) + if (log[i].op == "writeObject" && isRefWriteKey(log[i].key)) + return i; + return -1; +} + +} + +/// B188: every pool op (read OR write) on /blobs/ or /trees/ must come AFTER the first write that +/// appends the create-precommit owner event (the first root-shard CAS) — including HEAD +/// (exists/getObjectMetadata) and GET (readObject), since the +/// exact bug was an eager HEAD on a content blob during staging. The transaction writes a fresh +/// content file (pending blob) AND adopts an existing committed blob via hardlink — both paths must +/// satisfy the invariant. +TEST(CASWiringPrecommitOrder, NoContentPoolOpBeforePrecommit) +{ + auto recording = makeRecordingStorageForTest("order"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b188_order_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + /// Phase 1: publish a committed source part — this gives us a committed blob to adopt in Phase 2. + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin", "source-blob"); + tx->commit(DB::NoCommitOptions{}); + } + + /// Phase 2: a new transaction that BOTH writes a fresh content blob (all_1_1_0/data.bin, pending) + /// AND carries forward that PENDING blob via hardlink into a second fresh part (all_2_2_0/extra.bin, + /// the cross-part pending-source adopt path). We clear the op log after Phase 1 so only Phase 2's + /// ops are analysed. + recording->ops.clear(); + + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "fresh-content"); + /// Adopt by hardlinking a PENDING blob (the file just written above) into a SECOND fresh part + /// (all_2_2_0). This is the B188-relevant adopt: the cross-part pending-source branch copies the + /// PendingBlob into the dst build (NO eager pool op — the blob is not durable yet, so a HEAD/GET on + /// it before precommit would be the exact bug). We deliberately do NOT adopt from the committed + /// source part here: adoptFromTree(committed source) legitimately READS that source's + /// already-durable, ref-pinned tree during staging — a foreign-tree read that is NOT a B188 + /// violation (the invariant is about THIS build's own not-yet-uploaded content, never a committed + /// object owned by a live part). Gating it would be a false positive; see the committed-source + /// adopt coverage in CASWiringOps.HardLinkCarriesForwardWithoutReupload. + tx->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/extra.bin"); + tx->commit(DB::NoCommitOptions{}); + + const auto & log = recording->ops; + + /// The content objects THIS transaction publishes are exactly the BLOB keys it WRITES under + /// /blobs/ (the fresh/pending content blobs). The B188 invariant is that the build must not touch + /// ITS OWN not-yet-protected content before precommit. NOTE (rev. 15 manifest model): the staged + /// part-manifest body (`/_manifests/...`) is the precommit's EVIDENCE and is therefore written + /// BEFORE precommitAdd by design (stageManifest → precommitAdd → putBlob → promote) — it is NOT a + /// gated content object. Only the content BLOBS must wait for the precommit. Reads of foreign + /// committed objects (another part's blob) are legitimate and must not be gated — so we restrict + /// the gate to the set of /blobs/ keys this transaction itself wrote. + std::set own_content_keys; + for (const auto & r : log) + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos) + own_content_keys.insert(r.key); + + /// Anchor on the first precommit WRITE (the durable casPut), not on any precommit-key op. + const int first_precommit_idx = firstPrecommitWriteIdx(log); + ASSERT_GE(first_precommit_idx, 0) + << "No create-precommit owner write (root-shard CAS) was recorded — precommit step did not fire"; + + /// Every op (read OR write) on one of THIS build's own content blobs must have an index AFTER + /// first_precommit_idx. This gates HEAD (exists/getObjectMetadata) and GET (readObject), not just + /// PUT (writeObject) — an eager HEAD/GET on the build's own pending blob before precommit is the + /// exact B188 regression this guards against. + for (int i = 0; i < static_cast(log.size()); ++i) + { + if (!own_content_keys.contains(log[i].key)) + continue; + EXPECT_GT(i, first_precommit_idx) + << "Own-content pool op '" << log[i].op << "' on '" << log[i].key << "' at index " << i + << " came BEFORE the first precommit write at index " << first_precommit_idx + << " — violates B188 precommit-first invariant (no HEAD/GET/PUT on this build's content before precommit)"; + } + + /// Sanity: both parts are readable after commit, with the SAME underlying blob (content identity). + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/extra.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 13u); /// "fresh-content" + EXPECT_EQ(storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")[0].remote_path, + storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_2_2_0/extra.bin")[0].remote_path); + + /// Confirm at least one blob WRITE and one staged-manifest WRITE were recorded (both the upload + /// path and the manifest-evidence path were exercised), so the gate above actually had content + /// keys to check and the precommit anchored on a real build. + const bool has_blob_write = std::any_of(log.begin(), log.end(), + [](const RecordingLocalObjectStorage::Record & r) + { return r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos; }); + const bool has_tree_write = std::any_of(log.begin(), log.end(), + [](const RecordingLocalObjectStorage::Record & r) + { return r.op == "writeObject" && r.key.find("/cas/manifests/") != std::string::npos; }); + EXPECT_TRUE(has_blob_write) << "No /blobs/ write recorded — fresh blob path not exercised"; + EXPECT_TRUE(has_tree_write) << "No /cas/manifests/ write recorded — manifest staging path not exercised"; + EXPECT_FALSE(own_content_keys.empty()) << "No own content keys collected — gate would be vacuous"; +} + +/// B188 committed-source adopt (the LITERAL bug path): when createHardLink carries forward a blob +/// from a COMMITTED source part (the source is NOT staged in this transaction), it takes the +/// adoptFromTree -> adoptEvidence branch — a TOKENLESS W-EVIDENCE dep with NO eager HEAD on the +/// adopted blob. The regression this guards is reverting adoptEvidence to a reuseBlob(false) (or any +/// `ensureBlobPresent`) that HEADs a materialized blob during staging, before any precommit protection +/// exists. The own-content gate in NoContentPoolOpBeforePrecommit CANNOT catch this: the adopted blob +/// is FOREIGN (owned by the live source part, never written by this transaction), so it is absent from +/// own_content_keys. This test asserts a TARGETED invariant on that exact foreign blob key: no +/// exists/getObjectMetadata/readObject/writeObject on it before first_precommit_idx. +/// +/// adoptFromTree legitimately READS the source TREE during staging (to find the entry) — that is fine +/// and is NOT asserted here; the assertion is scoped to the adopted BLOB key alone. +TEST(CASWiringPrecommitOrder, CommittedSourceAdoptNoHeadBeforePrecommit) +{ + auto recording = makeRecordingStorageForTest("committed_adopt"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b188_committed_adopt_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + /// Phase 1: commit a source part with a content blob. Capture the source blob's logical key from + /// the recorded /blobs/ write (the SAME key derivation the recorder uses, so the substring/index + /// comparisons in Phase 2 line up exactly). + recording->ops.clear(); + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin", "committed-source-blob"); + tx->commit(DB::NoCommitOptions{}); + } + std::string source_blob_key; + for (const auto & r : recording->ops) + { + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos) + { + source_blob_key = r.key; + break; + } + } + ASSERT_FALSE(source_blob_key.empty()) + << "Phase 1 recorded no /blobs/ write — could not capture the committed-source blob key"; + + /// Phase 2: a FRESH transaction that hardlinks the COMMITTED source blob into a NEW part. The + /// source part (all_0_0_0) is not staged here, so createHardLink takes the committed-source branch + /// (adoptFromTree -> adoptEvidence). Clear the log so only Phase 2's ops are analysed. + recording->ops.clear(); + { + auto tx = storage->createTransaction(); + tx->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_5_5_0/data.bin"); + tx->commit(DB::NoCommitOptions{}); + } + + const auto & log = recording->ops; + + /// Anchor on the first precommit WRITE (the durable casPut), not on any precommit-key op. + const int first_precommit_idx = firstPrecommitWriteIdx(log); + ASSERT_GE(first_precommit_idx, 0) + << "No create-precommit owner write (root-shard CAS) was recorded — precommit step did not fire"; + + /// TARGETED assertion: the adopted (foreign, committed) blob key must NOT be touched by ANY op + /// (HEAD via exists/getObjectMetadata, GET via readObject, or PUT via writeObject) before the + /// precommit write. With the bug reintroduced, `adoptEvidence` would route through physical + /// materialization and `ensureBlobPresent`, which would + /// HEAD this exact key during staging at an index < first_precommit_idx, failing here. + bool adopted_blob_touched_before_precommit = false; + for (int i = 0; i < first_precommit_idx; ++i) + { + if (log[i].key == source_blob_key) + { + adopted_blob_touched_before_precommit = true; + ADD_FAILURE() + << "Adopted committed-source blob op '" << log[i].op << "' on '" << log[i].key + << "' at index " << i << " came BEFORE the first precommit write at index " + << first_precommit_idx << " — violates B188 (committed-source adopt must not HEAD/GET/" + << "PUT the adopted blob before precommit; expected TrustedManifest evidence)"; + } + } + EXPECT_FALSE(adopted_blob_touched_before_precommit); + + /// The committed-source adopt also must NOT re-upload the blob at all (content carried forward by + /// reference): no writeObject on the source blob key in Phase 2. + const bool reuploaded = std::any_of(log.begin(), log.end(), + [&](const RecordingLocalObjectStorage::Record & r) + { return r.op == "writeObject" && r.key == source_blob_key; }); + EXPECT_FALSE(reuploaded) << "Committed-source adopt re-uploaded the blob — should carry by reference"; + + /// Sanity: the new part reads back and shares the source blob object. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_5_5_0/data.bin")); + EXPECT_EQ(storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_0_0_0/data.bin")[0].remote_path, + storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_5_5_0/data.bin")[0].remote_path); +} + +/// B188 pending-blob hardlink (Task 6 Test 2): within a SINGLE transaction, write a content file +/// into part X (pending blob, not yet uploaded), then createHardLink that SAME file into part Y +/// (the cross-part pending-source branch: `&dst_st != src_st`, copies the PendingBlob record so +/// publishStaging uploads it for the dst part too). After commit both parts must read back the +/// identical content. +TEST(CASWiringPending, HardlinkOfPendingBlobCommitsAndReadsBack) +{ + auto storage = openWiringStorage(); + + auto tx = storage->createTransaction(); + + /// Write fresh content into part X — the blob is PENDING (not uploaded yet, temp-file only). + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin", "pending-payload"); + + /// Before commit, hardlink part X's file into part Y. At this point: + /// - src_st = staging for all_10_10_0 (exists: contains the pending blob) + /// - dst_st = staging for all_11_11_0 (created fresh here) + /// - &dst_st != src_st => PendingBlob is COPIED into dst_st.pending_blobs + /// - Neither build has a dependency proof until its post-precommit upload succeeds + tx->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_11_11_0/data.bin"); + + /// Nothing visible yet (B188: no uploads before precommit). + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin")); + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_11_11_0/data.bin")); + + tx->commit(DB::NoCommitOptions{}); + + /// Both parts must be visible and carry the same content. + ASSERT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin")); + ASSERT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_11_11_0/data.bin")); + + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin"), 15u); /// "pending-payload" + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_11_11_0/data.bin"), 15u); + + /// Both parts must point to the SAME underlying blob object (content-addressed identity). + auto objs_x = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_10_10_0/data.bin"); + auto objs_y = storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_11_11_0/data.bin"); + ASSERT_EQ(objs_x.size(), 1u); + ASSERT_EQ(objs_y.size(), 1u); + EXPECT_EQ(objs_x[0].remote_path, objs_y[0].remote_path) + << "Hardlinked pending blob must map to the SAME pool object in both parts"; +} + +/// ==== B190 Task 4: precommit-first for republishRef and committed-source createHardLink ==== +/// +/// B190-A: republishRef (called by moveDirectory for a COMMITTED part rename — RENAME TABLE, DETACH, +/// ATTACH, delete_tmp_ rename) must carry the source part's BLOBS forward by TOKENLESS W-EVIDENCE +/// (adoptEvidence), NOT by HEAD/GET/PUT on the source blob before precommit. In the rev. 15 manifest +/// model republishRef legitimately READS the FOREIGN source MANIFEST body (to copy its entries into a +/// fresh dst manifest) during staging — that is the manifest-era analog of the old adoptFromTree +/// source-tree read and is NOT a violation (see CommittedSourceAdoptNoHeadBeforePrecommit). The +/// invariant that survives: the source BLOB key must not be touched before the first precommit write. +TEST(CASWiringPrecommitOrder, RepublishRefNoTreeHeadBeforePrecommit) +{ + auto recording = makeRecordingStorageForTest("republish"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b190_republish_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + /// Phase 1: commit a source part. Capture its BLOB key from the /blobs/ write (republishRef must + /// carry this blob by reference, never touching it before precommit). + recording->ops.clear(); + { + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "republish-source"); + tx->commit(DB::NoCommitOptions{}); + } + std::string source_blob_key; + for (const auto & r : recording->ops) + { + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos) + { + source_blob_key = r.key; + break; + } + } + ASSERT_FALSE(source_blob_key.empty()) + << "Phase 1 recorded no /blobs/ write — could not capture the source blob key"; + + /// Phase 2: a COMMITTED rename (delete_tmp_ pattern) that triggers republishRef. Clear the log + /// so only Phase 2's ops are analysed. + recording->ops.clear(); + { + auto tx = storage->createTransaction(); + tx->moveDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0", "a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); + } + + const auto & log = recording->ops; + + const int first_precommit_idx = firstPrecommitWriteIdx(log); + ASSERT_GE(first_precommit_idx, 0) + << "No create-precommit owner write (root-shard CAS) was recorded — precommit step did not fire"; + + /// The source BLOB key must NOT be accessed (HEAD via exists/getObjectMetadata, GET via readObject, + /// or PUT via writeObject) before the precommit write. With an eager adopt-by-HEAD on the source + /// blob (the regression), `ensureBlobPresent` HEADs the blob key at an index < first_precommit_idx, + /// failing here. `TrustedManifest` evidence touches nothing. + bool blob_touched_before_precommit = false; + for (int i = 0; i < first_precommit_idx; ++i) + { + if (log[i].key == source_blob_key) + { + blob_touched_before_precommit = true; + ADD_FAILURE() + << "republishRef blob op '" << log[i].op << "' on '" << log[i].key + << "' at index " << i << " came BEFORE the first precommit write at index " + << first_precommit_idx << " — violates B190 precommit-first: republishRef must not " + << "HEAD/GET/PUT the source blob before precommit (use TrustedManifest evidence)"; + } + } + EXPECT_FALSE(blob_touched_before_precommit); + + /// Sanity: the renamed part is visible under the new name and NOT under the old name. + EXPECT_FALSE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0")); + EXPECT_TRUE(storage->existsDirectory("a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/delete_tmp_all_1_1_0/data.bin")); +} + +/// B190-B: the adoptStagedBlob helper unifies the 6 inline pending/uploaded adopt blocks from +/// createHardLink / moveFile / moveDirectory. The observable invariant: after refactoring, ALL +/// six sites still produce the same result as before — pending blobs are copied (hardlink) or +/// moved (moveFile/moveDirectory), and uploaded blobs are adopted with `TrustedManifest`. This test +/// exercises the non-trivial CROSS-PART pending path (createHardLink copies; moveFile moves) and +/// verifies both a copy and a move of the SAME pending source produce the correct committed state. +TEST(CASWiringPrecommitOrder, AdoptStagedBlobHelperUnifiesSixSites) +{ + /// Use a recording storage so we can verify no pre-precommit pool ops on own content. + auto recording = makeRecordingStorageForTest("adopt_helper"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b190_adopt_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + recording->ops.clear(); + + /// One transaction: write a pending blob into part A, hardlink (COPY pending) into part B, + /// and moveFile (MOVE pending) of a DIFFERENT pending blob from part A into part C. + auto tx = storage->createTransaction(); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/data.bin", "blob-for-copy"); + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/extra.bin", "blob-for-move"); + + /// createHardLink = COPY semantics: both src and dst should see the blob after commit. + tx->createHardLink("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/data.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_B_B_0/data.bin"); + + /// moveFile cross-part = MOVE semantics: src loses the blob, dst gains it. + { + auto & ca_tx = dynamic_cast(*tx); + ca_tx.moveFile("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/extra.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_C_C_0/extra.bin"); + } + + tx->commit(DB::NoCommitOptions{}); + + const auto & log = recording->ops; + const int first_precommit_idx = firstPrecommitWriteIdx(log); + ASSERT_GE(first_precommit_idx, 0) + << "No create-precommit owner write (root-shard CAS) was recorded — precommit step did not fire"; + + /// Collect own content keys (the content BLOBS this transaction wrote). The staged part-manifest + /// body (`/_manifests/...`) is the precommit's evidence and is written before precommit by design, + /// so it is NOT gated content — only /blobs/ are. + std::set own_content_keys; + for (const auto & r : log) + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos) + own_content_keys.insert(r.key); + + /// No own-content pool op before precommit (B188 invariant extends to all adopt sites). + for (int i = 0; i < static_cast(log.size()); ++i) + { + if (!own_content_keys.contains(log[i].key)) + continue; + EXPECT_GT(i, first_precommit_idx) + << "Own-content op '" << log[i].op << "' on '" << log[i].key << "' at index " << i + << " before precommit at " << first_precommit_idx; + } + + /// COPY semantics: both A and B see the copied blob. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/data.bin")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_B_B_0/data.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/data.bin"), 13u); /// "blob-for-copy" (13 bytes) + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_B_B_0/data.bin"), 13u); + EXPECT_EQ(storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/data.bin")[0].remote_path, + storage->getStorageObjects("a11/a11a11a1-1111-4111-8111-111111111111/all_B_B_0/data.bin")[0].remote_path) + << "COPY (hardlink): both parts must share the same blob object"; + + /// MOVE semantics: A loses extra.bin, C gains it. + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_A_A_0/extra.bin")); + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_C_C_0/extra.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_C_C_0/extra.bin"), 13u); /// "blob-for-move" (13 bytes) +} + +/// ==== B189: orphaned pending blob must NOT be uploaded after unlinkFile / replaceFile ==== +/// +/// When a file is written (pending blob X) and then unlinked (or replaced) within the same +/// transaction, X's tree entry is removed — so X is NOT referenced by the staged tree. Before the +/// B189 fix, publishStaging iterated pending_blobs unconditionally and uploaded X anyway (a wasted +/// PUT of an unreferenced blob). After the fix, publishStaging builds the set of blob hashes +/// referenced by the staged tree entries and uploads ONLY those — orphaned blobs are skipped. +/// +/// The test uses RecordingLocalObjectStorage to capture every writeObject call. After commit it +/// checks that the orphaned blob's pool key received NO writeObject, while a kept blob (written and +/// NOT removed in the same transaction) IS uploaded. +TEST(CASWiringOps, OrphanedPendingBlobNotUploadedAfterUnlink) +{ + auto recording = makeRecordingStorageForTest("b189_unlink"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b189_unlink_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + recording->ops.clear(); + + auto tx = storage->createTransaction(); + + /// Write blob X — this will be unlinked (orphaned) before commit. + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin", "orphan-bytes"); + + /// Write blob Y — this is kept (its tree entry survives to the staged tree). + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/kept.bin", "kept-bytes"); + + /// Unlink blob X — removes its tree entry; the pending_blobs record remains but is now orphaned. + tx->unlinkFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin", false, false); + + /// Sanity: the unlinked file is no longer staged (in-flight should not report it). + EXPECT_FALSE(tx->tryGetInFlightFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin").has_value()); + EXPECT_EQ(tx->tryGetInFlightFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/kept.bin"), std::optional(10)); + + tx->commit(DB::NoCommitOptions{}); + + const auto & log = recording->ops; + + /// Collect blob BODY keys written by this transaction (only /blobs/ writeObjects). Exclude the + /// per-hash `.meta` freshness descriptor sibling (`blobMetaKey` = body key + `.meta`, spec + /// §meta-protocols v3): it lives under the same /blobs/ prefix but is NOT a blob upload, so it must + /// not inflate the body-upload count. `putBlob` writes exactly one such `.meta` per body. + std::vector blob_writes; + for (const auto & r : log) + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos && !r.key.ends_with(".meta")) + blob_writes.push_back(r.key); + + /// Exactly ONE blob must have been uploaded (the kept one). The orphaned blob's pool key must + /// NOT appear in any writeObject — B189: orphan is filtered out of the publish upload. + EXPECT_EQ(blob_writes.size(), 1u) + << "Expected exactly 1 blob upload (the kept blob); got " << blob_writes.size() + << ". If 2, the orphaned pending blob was uploaded — B189 regression."; + + /// The kept file is visible after commit; the orphaned file is not. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/kept.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/kept.bin"), 10u); /// "kept-bytes" + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin")); +} + +/// B189 companion: the same orphan-filter applies when the tree entry is removed by replaceFile +/// (the destination entry erased before the move). Write blob X to dst, then replaceFile src->dst +/// (erases X's entry, moves src's entry to dst). The orphaned X must not be uploaded. +TEST(CASWiringOps, OrphanedPendingBlobNotUploadedAfterReplace) +{ + auto recording = makeRecordingStorageForTest("b189_replace"); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_b189_replace_scratch"); + auto storage = std::make_shared( + recording, "pool", "srv1", "", nullptr, settings); + storage->startup(); + + recording->ops.clear(); + + auto tx = storage->createTransaction(); + + /// Write blob X into the destination slot — it will be erased by replaceFile. + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", "original-bytes"); + + /// Write blob Y into the source slot — it will replace the destination. + writeThroughTransaction(*tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/new.bin", "replacement-bytes"); + + /// replaceFile: erases the dst entry (X orphaned), then moves src->dst. + { + auto & ca_tx = dynamic_cast(*tx); + ca_tx.replaceFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/new.bin", "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"); + } + + tx->commit(DB::NoCommitOptions{}); + + const auto & log = recording->ops; + + /// Exactly ONE blob must have been uploaded (the replacement blob Y). Exclude the per-hash `.meta` + /// freshness descriptor sibling (see the AfterUnlink test) — it is not a blob body upload. + std::vector blob_writes; + for (const auto & r : log) + if (r.op == "writeObject" && r.key.find("/blobs/") != std::string::npos && !r.key.ends_with(".meta")) + blob_writes.push_back(r.key); + + EXPECT_EQ(blob_writes.size(), 1u) + << "Expected exactly 1 blob upload (the replacement blob); got " << blob_writes.size() + << ". If 2, the orphaned original blob was uploaded — B189 regression."; + + /// After commit the destination slot carries the replacement content. + EXPECT_TRUE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin")); + EXPECT_EQ(storage->getFileSize("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"), 17u); /// "replacement-bytes" + EXPECT_FALSE(storage->existsFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/new.bin")); +} + +/// ==== Promote materialized-leaf edge protection (spec 2026-07-09-cas-writer-gc-simplification, Phase A) ==== +/// +/// A fast GC can PREMATURELY condemn a blob a writer just putBlob'd, in the tiny putBlob->promote window +/// (the precommit->blob edge is not yet folded, so GC reads in-degree 0). Under EDGE-BEFORE-OBSERVE the +/// precommit closure named the blob BEFORE putBlob observed it, so the condemnation cannot graduate to a +/// delete (the next fold sees the edge, d >= 1, spared) — it is doomed, not the blob. promote therefore +/// does not revalidate or republish a `Materialized` leaf; it commits with the blob's token unchanged. The only +/// blob-side abort promote still performs is the owner-liveness check (a reclaimed precommit) — which runs +/// BEFORE any blob work and touches nothing. +/// +/// These tests drive the REAL writer sequence (stageManifest -> precommitAdd -> putBlob -> promote) against +/// a raw in-memory Pool (no background GC → deterministic), and condemn the blob's CURRENT token by seeding +/// gc/state + the per-hash freshness meta the way a real GC condemn does (see `seedCondemnBlobToken` below). + +namespace DB::ErrorCodes +{ + extern const int ABORTED; + extern const int NETWORK_ERROR; +} + +namespace +{ + +DB::Cas::PoolPtr openResurrectStore(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return DB::Cas::Pool::open( + out_backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// Condemn (kind=Blob, hash, token) by seeding gc/state + a per-shard retired set (the durable GC ledger +/// shape — RetiredEntry, exact-token delete, unchanged by this task) AND condemning the per-hash freshness +/// meta, which is what the writer's condemned decision ACTUALLY point-reads (spec §meta-protocols v3). +/// Bumps the round so the retirement is a fresh one; leaves the object itself in place (condemn, NOT delete). +void seedCondemnBlobToken(DB::Cas::Pool & store, const DB::UInt128 & hash, + [[maybe_unused]] const DB::Cas::Token & token, [[maybe_unused]] uint64_t size) +{ + using namespace DB::Cas; + Backend & b = store.backend(); + const Layout & layout = store.layout(); + + GcState state; + const HeadResult head = b.head(layout.gcStateKey()); + if (head.exists) + { + const auto got = b.get(layout.gcStateKey()); + state = decodeGcState(got->bytes); + } + state.round += 1; + + /// Retired-in-snapshot: there is no separate retired-list object to seed — condemned state rides the + /// GC snapshot runs, which this writer-side edge-protection test does not exercise. The writer's + /// condemned decision point-reads the per-hash freshness meta (condemned below), so bumping the round + /// and condemning the meta is enough. + if (head.exists) + b.putOverwrite(layout.gcStateKey(), encodeGcState(state), head.token); + else + b.putIfAbsent(layout.gcStateKey(), encodeGcState(state)); + + /// The writer's fresh upload (putBlob) already wrote a Clean meta for `hash` (Task 3), so this is a + /// plain Clean -> Condemned CAS — exactly what GC's real condemn path does. + DB::Cas::tests::condemnMeta(b, layout, hash, state.round); +} + +} + +/// A blob condemned in the putBlob->promote window is EDGE-PROTECTED (spec +/// 2026-07-09-cas-writer-gc-simplification, Phase A): the precommit closure naming the blob was durable +/// BEFORE putBlob observed it, so a condemnation in this window cannot graduate to a delete (the next fold +/// sees the edge, d >= 1, spared). promote therefore does not re-check or republish a `Materialized` leaf — it +/// commits leaving the blob's token unchanged (no replacement PUT). The premature condemn is doomed on its own. +TEST(CASWiringResurrect, PromoteIgnoresCondemnedMaterializedBlobEdgeProtected) +{ + using namespace DB::Cas; + std::shared_ptr backend; + auto store = openResurrectStore(backend); + const RootNamespace ns{"test/tbl"}; + const String ref = "all_1_1_0"; + const String P = "republish-me"; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + + const ManifestId id = build->stageManifest({wiringBlobEntry("data.bin", P)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(P), BlobSource::fromString(P)); + + /// Condemn the freshly-uploaded blob's CURRENT token (GC condemning the not-yet-folded fresh incarnation). + const String blob_key = store->layout().blobKey(idOf(P)); + const HeadResult h1 = store->backend().head(blob_key); + ASSERT_TRUE(h1.exists); + const Token t0 = h1.token; + seedCondemnBlobToken(*store, u128Of(P), t0, h1.size); + { + const auto lm = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned) + << "precondition: the putBlob'd token must be condemned before promote"; + } + + /// Promote must not abort or touch the materialized leaf — it is edge-protected. + EXPECT_NO_THROW(build->promote(ns, ref, build->buildId(), id)); + + /// The ref is committed and the blob's token is unchanged — no replacement PUT ran (`Materialized` leaves are + /// not re-validated: EDGE-BEFORE-OBSERVE guarantees the condemnation is doomed, not the blob). + EXPECT_TRUE(store->resolveRef(ns, ref).has_value()) << "the ref must resolve after promote"; + const HeadResult h2 = store->backend().head(blob_key); + ASSERT_TRUE(h2.exists); + EXPECT_EQ(h2.token, t0) + << "materialized leaf is edge-protected: promote must not re-upload it (token unchanged)"; +} + +/// promote is a PURE owner MOVE (Δ=0 blob delta) — sound ONLY while this build's precommit is STILL the +/// live owner of the ref (`WPromote owner==bld` / INV_NO_DANGLE): a Δ=0 move over a ref with no live +/// precommit edge would republish a committed manifest onto to-be-deleted blobs. So when the precommit +/// binding is absent from the ref-table state, promote MUST fail closed with ABORTED — at the owner-liveness +/// check in the append closure, which runs before any blob revalidation, so no consequential blob publication +/// happens (a condemned leaf is left untouched, exactly as on the success path). +/// +/// This drives that guard the DETERMINISTIC way: a promote whose precommit was NEVER added (so the binding +/// is simply absent). The original "precommit added, then REMOVED out from under a still-live build" shape is +/// NOT reachable by any deterministic single-threaded in-runtime actor: `PartWriteTxn::abandon` marks the build +/// not-alive (`requireAlive` → LOGICAL_ERROR) and `Pool::dropNamespace` cancels the build (`requireAlive` → +/// ABORTED) — BOTH trip `requireAlive` at promote's first line, before this closure ever runs. Only a narrow +/// promote-vs-dropNamespace RACE (dropNamespace clears the binding in the window between promote's +/// `requireAlive` and its append closure) reaches the closure guard, which is therefore a defensive backstop +/// (a candidate for a later dead-code review — out of scope here). The previous version of this test faked +/// the removal with an out-of-band `appendOwnerEvent` the single-leader runtime never observes — an +/// unreachable state that surfaced as a CORRUPTED_DATA ref-log collision, not the intended ABORTED. +TEST(CASWiringResurrect, PromoteWithoutLivePrecommitAbortsWithoutResurrect) +{ + using namespace DB::Cas; + std::shared_ptr backend; + auto store = openResurrectStore(backend); + const RootNamespace ns{"test/tbl"}; + const String ref = "all_2_2_0"; + const String P = "abandoned-me"; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + + /// Seed a materialized leaf independently, then stage the manifest but DO NOT call `precommitAdd`. + /// Physical publication through `putBlob` requires that durable edge, while this test deliberately + /// needs the owner binding absent when `promote` runs. + DB::Cas::tests::writeBlobRaw( + store->backend(), store->layout(), P, store->poolMeta().blob_header_len, store->poolMeta().pool_id); + DB::Cas::tests::writeMetaClean(store->backend(), store->layout(), u128Of(P), P.size()); + const ManifestId id = build->stageManifest({wiringBlobEntry("data.bin", P)}); + + const String blob_key = store->layout().blobKey(idOf(P)); + const HeadResult h1 = store->backend().head(blob_key); + ASSERT_TRUE(h1.exists); + /// Condemn the leaf so that, were the blob gate reached, promote would republish it — proving the abort + /// happens strictly BEFORE any blob work. + seedCondemnBlobToken(*store, u128Of(P), h1.token, h1.size); + { + const auto lm = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned); + } + + /// promote aborts at the owner-liveness check (NETWORK_ERROR, fix #37 phase 2), before the blob gate. + try + { + build->promote(ns, ref, build->buildId(), id); + FAIL() << "expected promote to abort: the precommit is not the live owner of the ref"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + } + + /// No blob work ran before the abort: the leaf's token is UNCHANGED (still the condemned one) and its + /// metadata is still Condemned — the owner check aborts before any blob publication. + const HeadResult h2 = store->backend().head(blob_key); + ASSERT_TRUE(h2.exists); + EXPECT_EQ(h2.token, h1.token) + << "the aborting path must perform no PUT — the materialized leaf is untouched"; + const auto lm_after = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + EXPECT_TRUE(lm_after.has_value() && lm_after->meta.state == MetaState::Condemned) + << "no republication before the owner check — the token is still the condemned one"; +} + +/// tryFromDisk must be exception-free for a plain local disk: it runs on every +/// asynchronous-metrics tick for every configured disk, and probing via +/// `getMetadataStorage`'s NOT_IMPLEMENTED throw pollutes `system.errors` (the Exception +/// constructor counts the error even when the throw is caught) — a steady +N/s stream on a +/// pure-local server, caught as a stray-error failure by strict-error tests +/// (`test_cancel_backup`'s NoTrashChecker, Altinity PR#2073). +TEST(CASWiring, TryFromDiskOnLocalDiskIsExceptionFreeAndCountsNoError) +{ + auto tmp = std::filesystem::temp_directory_path() / "ca_wiring_tryfromdisk_test"; + std::filesystem::create_directories(tmp); + const DB::DiskPtr local = std::make_shared("tryfromdisk_local", tmp.string()); + + const auto before = DB::ErrorCodes::values[DB::ErrorCodes::NOT_IMPLEMENTED].get().local.count; + auto * ca = DB::ContentAddressedMetadataStorage::tryFromDisk(local); + const auto after = DB::ErrorCodes::values[DB::ErrorCodes::NOT_IMPLEMENTED].get().local.count; + + EXPECT_EQ(ca, nullptr); + EXPECT_EQ(after, before) + << "tryFromDisk on a non-content-addressed disk must not construct (and thereby count) " + "a NOT_IMPLEMENTED exception — it runs per disk on every asynchronous-metrics tick"; + std::filesystem::remove_all(tmp); +} diff --git a/src/Disks/tests/gtest_cas_b140_dangle.cpp b/src/Disks/tests/gtest_cas_b140_dangle.cpp new file mode 100644 index 000000000000..05af3e47e1cf --- /dev/null +++ b/src/Disks/tests/gtest_cas_b140_dangle.cpp @@ -0,0 +1,127 @@ +#include +#include +#include +#include +#include +#include +#include + +#include + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// The dangle is about the SINGLE snap shard's in-degree, and one cursor_key covers both refs. +PoolPtr openTestPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +size_t runGcToFixpoint(Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + RoundReport rep; + try + { + rep = gc.runRegularRound(); + } + catch (const DB::Exception &) + { + /// The fail-closed coherence guard refused this round (CORRUPTED_DATA): no delete + /// happened, the live blob is safe. Stop — re-running would just throw again. + break; + } + if (!rep.acquired_lease) + continue; + if (rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0) + break; + } + return rounds; +} + +} + +/// B140-DANGLE — the soak's INV-NO-LOSS finding, ported to the root-local part-manifest model. +/// +/// THE PROPERTY (unchanged across the redesign): a content-shared / deduplicated blob `B` referenced +/// by TWO live parts must NEVER be deleted when only ONE of those refs is dropped. In the old tree +/// model the loss arose from a `GcSnap` cursor-skip under-count (the committed `folded_cursor` ran +/// ahead of the snap's edges, so the second live part's edge was never folded). That white-box +/// failure mode is structurally IMPOSSIBLE in the manifest model: there is no separate snap; per-blob +/// in-degree is derived by folding the ONE ordered `RootOwnerEvent` journal, the fold cursor lives in +/// the `CasFoldSeal` (one durable unit with the sealed deltas, never diverging), and each part's blob +/// edges come from reading its OWN manifest body at fold time. So this is now a black-box no-loss +/// oracle: two live refs share `B`, drop one, GC to a fixpoint, assert `B` survives (`dangling == 0`) +/// because the surviving ref's manifest still contributes its +1 edge — `B`'s in-degree never reaches 0. +TEST(CASGCDangle, SharedBlobSurvivesDropOfOneOfTwoLiveRefs) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// rb_live -> manifest { data.bin: B }. B is uploaded here. + { + PartWriteInfo info; + info.intended_ref = ns.string() + "/rb_live"; + auto build = s->beginPartWrite(info); + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("B"))}; + + e.blob_size = std::string("B").size(); + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(ns, "rb_live", id); + build->putBlob(idOf("B"), BlobSource::fromString("B")); + build->promote(ns, "rb_live", build->buildId(), id); + s->renewWatermarkOnce(); + } + + /// rb_cur -> a DISTINCT manifest { other.bin: B } that REUSES the same shared blob B (tokenless + /// adopt — the soak's cross-node `adopt`). Still live. + { + PartWriteInfo info; + info.intended_ref = ns.string() + "/rb_cur"; + auto build = s->beginPartWrite(info); + ManifestEntry e; + e.path = "other.bin"; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("B"))}; + + e.blob_size = std::string("B").size(); + build->adoptEvidence(e); /// tokenless dep (no HEAD) — the cross-node adopt + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(ns, "rb_cur", id); + build->promote(ns, "rb_cur", build->buildId(), id); + s->renewWatermarkOnce(); + } + + /// Drop rb_live: its manifest's -1 on B lands, but rb_cur's manifest still contributes +1, so B's + /// in-degree stays >= 1 and B is never a zero-in-degree candidate. + s->dropRef(ns, "rb_live"); + s->renewWatermarkOnce(); + + Gc gc(s, hexToU128("00000000000000000000000000000001")); + const size_t rounds = runGcToFixpoint(gc); + + /// rb_cur is still LIVE and still resolves through a present manifest — its blob B must survive. + ASSERT_TRUE(s->resolveRef(ns, "rb_cur").has_value()); + + const FsckReport rep = runFsck(*s, /*detail=*/true); + + /// THE DANGLE ASSERTION: GC must NEVER delete a blob a live ref references. + EXPECT_EQ(rep.dangling, 0u) + << "B140-dangle: GC deleted shared blob B still referenced by the live ref rb_cur " + << "after " << rounds << " rounds (dangling=" << rep.dangling << ", reachable=" << rep.reachable + << ", B_present=" << b->head(s->layout().blobKey(idOf("B"))).exists << ")."; + EXPECT_TRUE(b->head(s->layout().blobKey(idOf("B"))).exists) + << "shared blob B must remain present while rb_cur references it"; +} diff --git a/src/Disks/tests/gtest_cas_backend.cpp b/src/Disks/tests/gtest_cas_backend.cpp new file mode 100644 index 000000000000..6a01821d9472 --- /dev/null +++ b/src/Disks/tests/gtest_cas_backend.cpp @@ -0,0 +1,1679 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +#if USE_AWS_S3 +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#endif + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int NOT_IMPLEMENTED; +} + +namespace +{ + +BlobPublishRequest streamingPublication( + String destination_key, String fresh_envelope, String payload, uint64_t payload_size) +{ + return BlobPublishRequest{ + .destination_key = std::move(destination_key), + .publication = StreamingBlobPublication{ + .payload_size = payload_size, + .fresh_envelope = std::move(fresh_envelope), + .open_payload = [stored_payload = std::move(payload)] + { + return std::make_unique(stored_payload); + }}}; +} + +struct CountingSourceState +{ + size_t bytes_exposed = 0; +}; + +class OneByteAtATimeReadBuffer final : public DB::ReadBuffer +{ +public: + OneByteAtATimeReadBuffer(size_t total_bytes_, std::shared_ptr state_) + : DB::ReadBuffer(nullptr, 0) + , total_bytes(total_bytes_) + , state(std::move(state_)) + { + } + +private: + bool nextImpl() override + { + if (state->bytes_exposed == total_bytes) + return false; + + ++state->bytes_exposed; + working_buffer = Buffer(&byte, &byte + 1); + return true; + } + + const size_t total_bytes; + const std::shared_ptr state; + char byte = 'x'; +}; + +BlobPublishRequest countedLongPublication( + String destination_key, + String fresh_envelope, + uint64_t payload_size, + size_t source_size, + const std::shared_ptr & state) +{ + return BlobPublishRequest{ + .destination_key = std::move(destination_key), + .publication = StreamingBlobPublication{ + .payload_size = payload_size, + .fresh_envelope = std::move(fresh_envelope), + .open_payload = [source_size, state] + { + return std::make_unique(source_size, state); + }}}; +} + +class PublishCountingInMemoryBackend final : public InMemoryBackend +{ +public: + void publishBlob(const BlobPublishRequest & request) override + { + ++publish_calls; + InMemoryBackend::publishBlob(request); + } + + size_t publish_calls = 0; +}; + +} + +/// Minimal concrete implementation that overrides every pure virtual with trivial defaults. +/// Purpose: verify the interface compiles, is overridable, and result-type defaults are sane. +struct NullBackend final : Backend +{ + std::optional get(const String & /*key*/, Range /*range*/) override + { + return std::nullopt; + } + + std::optional getStream(const String & /*key*/, Range /*range*/) override + { + return std::nullopt; + } + + HeadResult head(const String & /*key*/) override + { + return HeadResult{}; + } + + PutResult putIfAbsent(const String & /*key*/, const String & /*bytes*/, const ObjectMeta & /*meta*/) override + { + return {PutOutcome::Done, {}}; + } + + void publishBlob(const BlobPublishRequest & /*request*/) override + { + } + + PutResult putOverwrite(const String & /*key*/, const String & /*bytes*/, const Token & /*expected*/, const ObjectMeta & /*meta*/) override + { + return {PutOutcome::PreconditionFailed, {}}; + } + + CasResult casPut(const String & /*key*/, const String & /*bytes*/, const std::optional & /*expected*/, const ObjectMeta & /*meta*/) override + { + return {CasOutcome::Conflict, {}}; + } + + DeleteOutcome deleteExact(const String & /*key*/, const Token & /*token*/) override + { + return DeleteOutcome{}; + } + + ListPage list(const String & /*prefix*/, const String & /*cursor*/, size_t /*limit*/) override + { + return ListPage{}; + } + + bool supportsListTokens() const override { return false; } +}; + +TEST(CASBackend, PublishBlobReturnsNoIncarnationToken) +{ + static_assert(std::is_same_v< + decltype(std::declval().publishBlob(std::declval())), + void>); +} + +TEST(CASBackend, NullBackendShapeAndDefaults) +{ + NullBackend b; + // Use the base-class reference so virtual dispatch uses base-class default args. + Backend & ref = b; + + // get returns absent + EXPECT_FALSE(ref.get("k").has_value()); + + // head returns non-existent + HeadResult h = b.head("k"); + EXPECT_FALSE(h.exists); + EXPECT_EQ(h.size, 0u); + EXPECT_TRUE(h.token.empty()); + + // putIfAbsent returns Done + EXPECT_EQ(ref.putIfAbsent("k", "v").outcome, PutOutcome::Done); + + // putOverwrite returns PreconditionFailed + EXPECT_EQ(ref.putOverwrite("k", "v", Token{}).outcome, PutOutcome::PreconditionFailed); + + // casPut returns Conflict + EXPECT_EQ(ref.casPut("k", "v", std::nullopt).outcome, CasOutcome::Conflict); + + // deleteExact default kind is NotFound + DeleteOutcome d = b.deleteExact("k", Token{}); + EXPECT_EQ(d.kind, DeleteOutcome::Kind::NotFound); + EXPECT_FALSE(d.created_delete_marker); + + // list returns empty page + ListPage page = b.list("p/", "", 10); + EXPECT_TRUE(page.keys.empty()); + EXPECT_TRUE(page.next_cursor.empty()); + + // Range::whole() helper + EXPECT_TRUE(Range{}.whole()); + Range r1; r1.offset = 1; + EXPECT_FALSE(r1.whole()); + Range r2; r2.length = 5u; + EXPECT_FALSE(r2.whole()); +} + +// ===================================================================== +// Task 3: CasInMemoryBackend — enforcing token semantics +// ===================================================================== + +TEST(CASInMemory, PutIfAbsentAndGet) +{ + InMemoryBackend b; + const auto put = b.putIfAbsent("k", "v1"); + const Token t1 = put.token; + EXPECT_EQ(put.outcome, PutOutcome::Done); + EXPECT_FALSE(t1.empty()); + EXPECT_EQ(b.putIfAbsent("k", "clobber").outcome, PutOutcome::PreconditionFailed); + auto g = b.get("k"); + ASSERT_TRUE(g.has_value()); + EXPECT_EQ(g->bytes, "v1"); + EXPECT_EQ(g->token, t1); + EXPECT_FALSE(b.get("absent").has_value()); +} + +TEST(CASInMemory, OverwriteIsTokenExactAndMintsFreshToken) +{ + InMemoryBackend b; + const Token t1 = b.putIfAbsent("k", "v1").token; + EXPECT_EQ(b.putOverwrite("k", "v2", Token{"wrong", TokenType::Emulated}).outcome, PutOutcome::PreconditionFailed); + EXPECT_EQ(b.get("k")->bytes, "v1"); // untouched on mismatch + const auto overwrite = b.putOverwrite("k", "v2", t1); + EXPECT_EQ(overwrite.outcome, PutOutcome::Done); + EXPECT_NE(overwrite.token, t1); // tokens never repeat + EXPECT_EQ(b.get("k")->bytes, "v2"); +} + +TEST(CASInMemory, CasPutCreateAndSwap) +{ + InMemoryBackend b; + const auto create = b.casPut("m", "s1", std::nullopt); + const Token t1 = create.token; + EXPECT_EQ(create.outcome, CasOutcome::Committed); // create-if-absent + EXPECT_EQ(b.casPut("m", "s1x", std::nullopt).outcome, CasOutcome::Conflict); // exists now + EXPECT_EQ(b.casPut("m", "s2", Token{"stale", TokenType::Emulated}).outcome, CasOutcome::Conflict); + EXPECT_EQ(b.get("m")->bytes, "s1"); + EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Committed); + EXPECT_EQ(b.get("m")->bytes, "s2"); +} + +TEST(CASInMemory, DeleteExactEnforced) +{ + InMemoryBackend b; + const Token t1 = b.putIfAbsent("k", "v1").token; + auto d1 = b.deleteExact("k", Token{"wrong", TokenType::Emulated}); + EXPECT_EQ(d1.kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_TRUE(b.get("k").has_value()); // SURVIVES wrong-token delete + auto d2 = b.deleteExact("k", t1); + EXPECT_EQ(d2.kind, DeleteOutcome::Kind::Deleted); + EXPECT_FALSE(d2.created_delete_marker); + EXPECT_FALSE(b.get("k").has_value()); + EXPECT_EQ(b.deleteExact("k", t1).kind, DeleteOutcome::Kind::NotFound); +} + +TEST(CASInMemory, RangeGetAndHeadAndList) +{ + InMemoryBackend b; + b.putIfAbsent("p/a", "0123456789"); + b.putIfAbsent("p/b", "xy"); + b.putIfAbsent("q/c", "z"); + EXPECT_EQ(b.get("p/a", Range{.offset = 2, .length = 3})->bytes, "234"); + auto h = b.head("p/a"); + EXPECT_TRUE(h.exists); + EXPECT_EQ(h.size, 10u); + auto page = b.list("p/", "", 10); + ASSERT_EQ(page.keys.size(), 2u); // sorted, prefix-scoped + EXPECT_EQ(page.keys[0].key, "p/a"); + EXPECT_EQ(page.keys[1].key, "p/b"); + EXPECT_TRUE(page.next_cursor.empty()); + auto page1 = b.list("p/", "", 1); // pagination + EXPECT_EQ(page1.keys.size(), 1u); + EXPECT_EQ(page1.keys[0].key, "p/a"); + EXPECT_EQ(page1.next_cursor, "p/a"); + EXPECT_FALSE(page1.next_cursor.empty()); + auto page2 = b.list("p/", page1.next_cursor, 1); + EXPECT_EQ(page2.keys[0].key, "p/b"); +} + +TEST(CASInMemory, PublishBlobStreamingWritesFreshEnvelopeAndExactPayload) +{ + InMemoryBackend backend; + const auto request = streamingPublication("blob", "fresh-envelope", "payload", 7); + + backend.publishBlob(request); + + const auto result = backend.get("blob"); + ASSERT_TRUE(result.has_value()); + EXPECT_EQ(result->bytes, "fresh-envelopepayload"); +} + +TEST(CASInMemory, PublishBlobRejectsShortAndLongStreamingSourcesWithoutVisibility) +{ + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("short", "old-short").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent("long", "old-long").outcome, PutOutcome::Done); + + for (const auto & [key, payload, declared_size] : std::vector>{ + {"short", "abc", 4}, + {"long", "abcd", 3}}) + { + const auto request = streamingPublication(key, "fresh", payload, declared_size); + try + { + backend.publishBlob(request); + FAIL() << "expected a source-size mismatch for " << key; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + } + + EXPECT_EQ(backend.get("short")->bytes, "old-short"); + EXPECT_EQ(backend.get("long")->bytes, "old-long"); +} + +TEST(CASInMemory, PublishBlobLongSourceReadsOnlyDeclaredPayloadAndOneProbeByte) +{ + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("long", "old-complete-body").outcome, PutOutcome::Done); + auto state = std::make_shared(); + + try + { + backend.publishBlob(countedLongPublication("long", "fresh", 3, 1024, state)); + FAIL() << "expected a long-source mismatch"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + + EXPECT_EQ(state->bytes_exposed, 4u); + ASSERT_TRUE(backend.get("long").has_value()); + EXPECT_EQ(backend.get("long")->bytes, "old-complete-body"); +} + +TEST(CASInMemory, PublishBlobKeepsThePreviousIncarnationVisibleUntilTheCompleteBodyIsReady) +{ + using namespace std::chrono_literals; + + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("blob", "old-complete-body").outcome, PutOutcome::Done); + + std::promise source_opened; + std::promise release_source; + const std::shared_future release = release_source.get_future().share(); + const BlobPublishRequest request{ + .destination_key = "blob", + .publication = StreamingBlobPublication{ + .payload_size = 7, + .fresh_envelope = "fresh-envelope", + .open_payload = [&source_opened, release] + { + source_opened.set_value(); + release.wait(); + return std::make_unique(String("payload")); + }}}; + + auto publication = std::async(std::launch::async, [&] { backend.publishBlob(request); }); + source_opened.get_future().wait(); + + auto observation = std::async(std::launch::async, [&] { return backend.get("blob"); }); + const auto observation_status = observation.wait_for(2s); + EXPECT_EQ(observation_status, std::future_status::ready) + << "publication must not hold the visibility lock while draining its source"; + if (observation_status == std::future_status::ready) + { + const auto visible = observation.get(); + ASSERT_TRUE(visible.has_value()); + EXPECT_EQ(visible->bytes, "old-complete-body"); + } + + release_source.set_value(); + EXPECT_NO_THROW(publication.get()); + ASSERT_TRUE(backend.get("blob").has_value()); + EXPECT_EQ(backend.get("blob")->bytes, "fresh-envelopepayload"); +} + +TEST(CASInMemory, PublishBlobCopiesStagedObjectBytesVerbatim) +{ + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("stage", "staged-envelopepayload").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent("blob", "old-body").outcome, PutOutcome::Done); + + backend.publishBlob(BlobPublishRequest{ + .destination_key = "blob", + .publication = VerbatimStagedBlobPublication{ + .object_key = "stage", + .object_size = 22}}); + + ASSERT_TRUE(backend.get("blob").has_value()); + EXPECT_EQ(backend.get("blob")->bytes, "staged-envelopepayload"); +} + +// ===================================================================== +// Task 4: CasInMemoryBackend — fault injection and probe-test modes +// ===================================================================== + +TEST(CASInMemoryFaults, HeldDeleteLandsLater) +{ + InMemoryBackend b; + const Token t1 = b.putIfAbsent("k", "v1").token; + b.setHoldDeletes(true); + auto d = b.deleteExact("k", t1); // message "sent", not landed + EXPECT_EQ(d.kind, DeleteOutcome::Kind::Deleted); // caller sees the send accepted + EXPECT_TRUE(b.get("k").has_value()); // ... but nothing landed yet + ASSERT_EQ(b.pendingDeletes(), 1u); + // the object is recreated before the zombie lands: + b.putOverwrite("k", "v1'", t1); + auto landed = b.landPendingDelete(0); // the zombie lands NOW + EXPECT_EQ(landed.kind, DeleteOutcome::Kind::TokenMismatch); // 412 — INV-NO-RETURN in miniature + EXPECT_EQ(b.get("k")->bytes, "v1'"); +} + +TEST(CASInMemoryFaults, InjectedCasConflictFiresOnce) +{ + InMemoryBackend b; + const Token t1 = b.casPut("m", "s1", std::nullopt).token; + b.failNextCasPut("m"); + EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Conflict); // injected + EXPECT_EQ(b.get("m")->bytes, "s1"); + EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Committed); // next attempt is real +} + +TEST(CASInMemoryFaults, NonEnforcingModeMimicsBadBackend) +{ + InMemoryBackend b; + b.setEnforceTokens(false); // MinIO-OSS-shaped backend + b.putIfAbsent("k", "v1"); + auto d = b.deleteExact("k", Token{"totally-wrong", TokenType::Emulated}); + EXPECT_EQ(d.kind, DeleteOutcome::Kind::Deleted); // silently deletes anyway — the dangerous behavior + EXPECT_FALSE(b.get("k").has_value()); +} + +TEST(CASInMemoryFaults, VersioningMarkerMode) +{ + InMemoryBackend b; + b.setSimulateDeleteMarkers(true); + const Token t1 = b.putIfAbsent("k", "v1").token; + EXPECT_TRUE(b.deleteExact("k", t1).created_delete_marker); // probe must reject this pool +} + +TEST(CASInMemoryBackend, RoundTripsUserMetadata) +{ + DB::Cas::InMemoryBackend backend; + const DB::Cas::ObjectMeta meta{{"cas_owner", "ab:7:42"}}; + ASSERT_EQ(backend.putIfAbsent("k/key", "body", meta).outcome, DB::Cas::PutOutcome::Done); + + const auto hr = backend.head("k/key"); + ASSERT_TRUE(hr.exists); + ASSERT_EQ(hr.attributes.at("cas_owner"), "ab:7:42"); + + const auto gr = backend.get("k/key"); + ASSERT_TRUE(gr.has_value()); + ASSERT_EQ(gr->attributes.at("cas_owner"), "ab:7:42"); +} + +// ===================================================================== +// getStream seam (forward-only reads of write-once objects) +// ===================================================================== + +TEST(CASBackendStream, StreamsBodyWindow) +{ + auto backend = std::make_shared(); + backend->putIfAbsent("k", "0123456789"); + auto got = backend->getStream("k", DB::Cas::Range{.offset = 2, .length = 5}); + ASSERT_TRUE(got.has_value()); + String out; + DB::readStringUntilEOF(out, *got->stream); + EXPECT_EQ(out, "23456"); + EXPECT_FALSE(got->token.empty()); + EXPECT_FALSE(backend->getStream("absent").has_value()); +} + +// ===================================================================== +// B168 P0: InstrumentedBackend per-namespace/op ProfileEvents +// ===================================================================== + +namespace ProfileEvents +{ +extern const Event CASBlobPut; +extern const Event CASBlobPutDeduplicated; +extern const Event CASBlobHead; +extern const Event CASBlobHeadMiss; +extern const Event CASGCCompareSwap; +} + +TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) +{ + /// Namespace classification by substring. + EXPECT_EQ(classifyCasNs("pool/blobs/ab/abcdef"), CasNs::Blob); + EXPECT_EQ(classifyCasNs("pool/gc/registry"), CasNs::Gc); /// gc/ prefix covers GC state (state, retired sets, etc.) + EXPECT_EQ(classifyCasNs("pool/roots/default/_files/x"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/gc/state"), CasNs::Gc); + /// D3: the old per-server-control key shapes (`_watermark`, `_precommits/`) have no producer + /// anymore -- control state now lives under `/gc/server-roots/...` (classifies as Gc). A key of + /// this legacy shape, if it ever showed up, would fall through to the generic /roots/ rule. + EXPECT_EQ(classifyCasNs("pool/roots/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa/_watermark"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/roots/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa/_precommits/3"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/_pool_meta"), CasNs::Other); + /// Final opaque-life layout: both immutable streams and point/path-addressed state remain Root + /// instrumentation, while part manifests remain Manifest. None may fall into Other (the + /// 2026-07-03 operator-stand CREATE storm misread as CASOtherHeadMiss=102 because of this). + EXPECT_EQ(classifyCasNs("pool/cas/ns/stream/00000000000000000000000000000017/_log/1-1.zst"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/cas/ns/state/00000000000000000000000000000017/_ckpt.zst"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/cas/ns/state/00000000000000000000000000000017/_files/format_version.txt"), CasNs::Root); + EXPECT_EQ(classifyCasNs("pool/cas/manifests/0/srv/store/d18/uuid@cas@/24/1/000001.proto"), CasNs::Manifest); + + auto inner = std::make_shared(); + InstrumentedBackend b(inner); + + using ProfileEvents::global_counters; + const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); + const auto blob_dedup_before = global_counters[ProfileEvents::CASBlobPutDeduplicated].load(); + const auto blob_head_before = global_counters[ProfileEvents::CASBlobHead].load(); + const auto blob_miss_before = global_counters[ProfileEvents::CASBlobHeadMiss].load(); + const auto gc_cas_before = global_counters[ProfileEvents::CASGCCompareSwap].load(); + + const String blob_key = "pool/blobs/ab/abcdef0123456789"; + + /// First put of a blob ⇒ Put. + EXPECT_EQ(b.putIfAbsent(blob_key, "payload").outcome, PutOutcome::Done); + /// Second put of the same key ⇒ PutDeduplicated (content already exists). + EXPECT_EQ(b.putIfAbsent(blob_key, "payload").outcome, PutOutcome::PreconditionFailed); + /// head of an absent blob key ⇒ HeadMiss (the 404 signal). + EXPECT_FALSE(b.head("pool/blobs/zz/absent").exists); + /// head of the present blob key ⇒ Head. + EXPECT_TRUE(b.head(blob_key).exists); + /// casPut create on a gc key ⇒ Gc Cas. + EXPECT_EQ(b.casPut("pool/gc/state", "g1", std::nullopt).outcome, CasOutcome::Committed); + /// Under coverage builds ProfileEvents propagate into a thread-local subtree that does not reach + /// `global_counters`; deltas read 0 there only (see gtest_unique_key_index_cache). +#if !WITH_COVERAGE + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut].load() - blob_put_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPutDeduplicated].load() - blob_dedup_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobHead].load() - blob_head_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobHeadMiss].load() - blob_miss_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASGCCompareSwap].load() - gc_cas_before, 1u); +#else + (void)blob_put_before; (void)blob_dedup_before; (void)blob_head_before; + (void)blob_miss_before; (void)gc_cas_before; +#endif +} + +TEST(CASInstrumentedBackend, PublishBlobDelegatesOnceAndRecordsOnePhysicalBlobWrite) +{ + auto inner = std::make_shared(); + InstrumentedBackend backend(inner); + + using ProfileEvents::global_counters; + const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); + + const auto request = streamingPublication("pool/blobs/ab/published", "fresh", "payload", 7); + backend.publishBlob(request); + + EXPECT_EQ(inner->publish_calls, 1u); + ASSERT_TRUE(inner->get("pool/blobs/ab/published").has_value()); + EXPECT_EQ(inner->get("pool/blobs/ab/published")->bytes, "freshpayload"); +#if !WITH_COVERAGE + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut].load() - blob_put_before, 1u); +#else + (void)blob_put_before; +#endif +} + +// ===================================================================== +// M-C2 Task 2: typed S3 precondition signal +// ===================================================================== + +#if USE_AWS_S3 + +namespace +{ + +class PublicationRecordingWriteBuffer final : public DB::WriteBufferFromFileBase +{ +public: + PublicationRecordingWriteBuffer(size_t & cancel_calls_, size_t & finalize_calls_, size_t & bytes_at_cancel_) + : DB::WriteBufferFromFileBase(DB::DBMS_DEFAULT_BUFFER_SIZE, nullptr, 0) + , cancel_calls(cancel_calls_) + , finalize_calls(finalize_calls_) + , bytes_at_cancel(bytes_at_cancel_) + { + } + + void sync() override + { + next(); + } + + std::string getFileName() const override + { + return "publication-recording-write-buffer"; + } + +private: + void nextImpl() override + { + } + + void finalizeImpl() override + { + next(); + ++finalize_calls; + } + + void cancelImpl() noexcept override + { + bytes_at_cancel = count(); + ++cancel_calls; + } + + size_t & cancel_calls; + size_t & finalize_calls; + size_t & bytes_at_cancel; +}; + +struct PublicationWriteBarrier +{ + std::promise opened; + std::promise release; + std::shared_future release_future = release.get_future().share(); +}; + +class PublicationRecordingLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + ++write_calls; + last_opened_key = object.remote_path; + last_write_mode = mode; + last_write_settings = write_settings; + if (record_cancellation_only) + return std::make_unique(cancel_calls, finalize_calls, bytes_at_cancel); + + auto out = DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + if (throw_after_open) + throw std::runtime_error("injected write failure after opening local object"); + if (write_barrier) + { + write_barrier->opened.set_value(); + write_barrier->release_future.wait(); + } + return out; + } + + void copyObject( + const DB::StoredObject & object_from, + const DB::StoredObject & object_to, + const DB::ReadSettings & read_settings, + const DB::WriteSettings & write_settings, + std::optional object_to_attributes) override + { + ++copy_calls; + last_copy_settings = write_settings; + DB::LocalObjectStorage::copyObject( + object_from, object_to, read_settings, write_settings, object_to_attributes); + } + + bool supportsCopyMode(DB::ObjectStorageCopyMode copy_mode) const override + { + return copy_mode == DB::ObjectStorageCopyMode::Default + || (copy_mode == DB::ObjectStorageCopyMode::NativeOnly && native_copy_supported); + } + + std::optional tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags) const override + { + ++native_metadata_calls; + return DB::LocalObjectStorage::tryGetObjectMetadataWithNativeToken(path, with_tags); + } + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + ++metadata_calls; + return DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + } + + void resetRecording() + { + write_calls = 0; + copy_calls = 0; + metadata_calls = 0; + native_metadata_calls = 0; + cancel_calls = 0; + finalize_calls = 0; + bytes_at_cancel = 0; + last_write_mode.reset(); + last_write_settings.reset(); + last_copy_settings.reset(); + } + + bool native_copy_supported = true; + bool record_cancellation_only = false; + bool throw_after_open = false; + std::shared_ptr write_barrier; + size_t write_calls = 0; + size_t copy_calls = 0; + mutable size_t metadata_calls = 0; + mutable size_t native_metadata_calls = 0; + size_t cancel_calls = 0; + size_t finalize_calls = 0; + size_t bytes_at_cancel = 0; + String last_opened_key; + std::optional last_write_mode; + std::optional last_write_settings; + std::optional last_copy_settings; +}; + +std::shared_ptr makePublicationRecordingStorage() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_publish_blob_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +String readStorageObject(const DB::ObjectStoragePtr & storage, const String & key) +{ + auto in = storage->readObject(DB::StoredObject(key), DB::ReadSettings{}); + String bytes; + DB::readStringUntilEOF(bytes, *in); + return bytes; +} + +} + +TEST(CASObjectStorageBackend, PublishBlobStreamingUsesOrdinaryDefaultWriteTransport) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + backend.setNativeTokenTypeForTest(TokenType::Generation); + const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/streaming"); + + const auto request = streamingPublication(destination, "fresh-envelope", "payload", 7); + backend.publishBlob(request); + + ASSERT_EQ(storage->write_calls, 1u); + ASSERT_TRUE(storage->last_write_mode.has_value()); + EXPECT_EQ(*storage->last_write_mode, DB::WriteMode::Rewrite); + ASSERT_TRUE(storage->last_write_settings.has_value()); + EXPECT_EQ(storage->last_write_settings->object_storage_request_mode, DB::ObjectStorageRequestMode::Default); + EXPECT_EQ(storage->last_write_settings->object_storage_retry_profile, DB::ObjectStorageRetryProfile::Default); + EXPECT_EQ(storage->last_write_settings->s3_max_unexpected_write_error_retries_override, 0u); + EXPECT_FALSE(storage->last_write_settings->s3_force_single_part_upload); + EXPECT_TRUE(storage->last_write_settings->object_storage_write_if_none_match.empty()); + EXPECT_TRUE(storage->last_write_settings->object_storage_write_if_match.empty()); + EXPECT_EQ(storage->native_metadata_calls, 0u) + << "tokenless publication must not issue a response-token HEAD"; + EXPECT_EQ(readStorageObject(storage, destination), "fresh-envelopepayload"); +} + +TEST(CASObjectStorageBackend, PublishBlobEmulatedKeepsDestinationCompleteUntilAtomicReplacement) +{ + using namespace std::chrono_literals; + + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + const String key = "publish/emulated-atomic"; + const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); + + { + auto out = storage->writeObject( + DB::StoredObject(physical_key), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("old-complete-body"), *out); + out->finalize(); + } + storage->resetRecording(); + + auto barrier = std::make_shared(); + storage->write_barrier = barrier; + auto opened = barrier->opened.get_future(); + auto publication = std::async(std::launch::async, [&] + { + backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)); + }); + + const auto opened_status = opened.wait_for(2s); + EXPECT_EQ(opened_status, std::future_status::ready); + if (opened_status == std::future_status::ready) + EXPECT_EQ(readStorageObject(storage, physical_key), "old-complete-body"); + + barrier->release.set_value(); + EXPECT_NO_THROW(publication.get()); + EXPECT_EQ(storage->metadata_calls, 0u); + EXPECT_EQ(readStorageObject(storage, physical_key), "fresh-envelopepayload"); +} + +TEST(CASObjectStorageBackend, PublishBlobEmulatedWriteFailurePreservesDestinationAndCleansTemporary) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + const String key = "publish/emulated-failure"; + const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); + + { + auto out = storage->writeObject( + DB::StoredObject(physical_key), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("old-complete-body"), *out); + out->finalize(); + } + const Token old_token = backend.head(key).token; + + storage->throw_after_open = true; + EXPECT_THROW( + backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)), + std::runtime_error); + storage->throw_after_open = false; + + EXPECT_NE(storage->last_opened_key, physical_key); + EXPECT_FALSE(storage->exists(DB::StoredObject(storage->last_opened_key))); + EXPECT_EQ(readStorageObject(storage, physical_key), "old-complete-body"); + EXPECT_EQ(backend.head(key).token, old_token); +} + +TEST(CASObjectStorageBackend, PublishBlobCancelsShortAndLongStreamingSourcesBeforeVisibility) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/mismatch"); + + { + auto out = storage->writeObject( + DB::StoredObject(destination), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("old-complete-body"), *out); + out->finalize(); + } + storage->resetRecording(); + storage->record_cancellation_only = true; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + backend.publishBlob(streamingPublication(destination, "fresh", "abc", 4)); + }); + EXPECT_EQ(storage->cancel_calls, 1u); + EXPECT_EQ(storage->finalize_calls, 0u); + EXPECT_EQ(storage->bytes_at_cancel, 8u); + EXPECT_EQ(readStorageObject(storage, destination), "old-complete-body"); + + auto state = std::make_shared(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + backend.publishBlob(countedLongPublication(destination, "fresh", 3, 1024, state)); + }); + EXPECT_EQ(state->bytes_exposed, 4u); + EXPECT_EQ(storage->cancel_calls, 2u); + EXPECT_EQ(storage->finalize_calls, 0u); + EXPECT_EQ(storage->bytes_at_cancel, 8u); + EXPECT_EQ(readStorageObject(storage, destination), "old-complete-body"); +} + +TEST(CASObjectStorageBackend, PublishBlobEmulatedLongSourceReadsOnlyDeclaredPayloadAndOneProbeByte) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + const String key = "publish/emulated-long"; + const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); + + { + auto out = storage->writeObject( + DB::StoredObject(physical_key), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("old-complete-body"), *out); + out->finalize(); + } + + auto state = std::make_shared(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + backend.publishBlob(countedLongPublication(key, "fresh", 3, 1024, state)); + }); + + EXPECT_EQ(state->bytes_exposed, 4u); + EXPECT_EQ(readStorageObject(storage, physical_key), "old-complete-body"); +} + +TEST(CASObjectStorageBackend, PublishBlobCopiesStagedBytesWithNativeOnlyDefaultRequestMode) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + const String staging = DB::Cas::tests::nativeKeyUnder(storage, "publish/staging"); + const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/copied"); + + { + auto out = storage->writeObject( + DB::StoredObject(staging), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("staged-envelopepayload"), *out); + out->finalize(); + } + storage->resetRecording(); + + backend.publishBlob(BlobPublishRequest{ + .destination_key = destination, + .publication = VerbatimStagedBlobPublication{ + .object_key = staging, + .object_size = 22}}); + + ASSERT_EQ(storage->copy_calls, 1u); + ASSERT_TRUE(storage->last_copy_settings.has_value()); + EXPECT_EQ(storage->last_copy_settings->object_storage_copy_mode, DB::ObjectStorageCopyMode::NativeOnly); + EXPECT_EQ(storage->last_copy_settings->object_storage_request_mode, DB::ObjectStorageRequestMode::Default); + EXPECT_EQ(storage->last_copy_settings->object_storage_retry_profile, DB::ObjectStorageRetryProfile::Default); + EXPECT_TRUE(storage->last_copy_settings->object_storage_write_if_none_match.empty()); + EXPECT_TRUE(storage->last_copy_settings->object_storage_write_if_match.empty()); + EXPECT_EQ(readStorageObject(storage, destination), "staged-envelopepayload"); +} + +TEST(CASObjectStorageBackend, PublishBlobRefusesVerbatimCopyWithoutNativeTransport) +{ + auto storage = makePublicationRecordingStorage(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + const String staging = DB::Cas::tests::nativeKeyUnder(storage, "publish/unsupported-staging"); + const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/unsupported-copy"); + + { + auto out = storage->writeObject( + DB::StoredObject(staging), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("complete-staged-object"), *out); + out->finalize(); + } + storage->resetRecording(); + storage->native_copy_supported = false; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, [&] + { + backend.publishBlob(BlobPublishRequest{ + .destination_key = destination, + .publication = VerbatimStagedBlobPublication{ + .object_key = staging, + .object_size = 22}}); + }); + + EXPECT_EQ(storage->copy_calls, 0u); + EXPECT_FALSE(storage->exists(DB::StoredObject(destination))); +} + +/// The Native conditional-PUT path discriminates a lost precondition by the canonical S3 error code +/// string ("PreconditionFailed", "NoSuchKey", ...) that `S3Exception` carries from the response XML +/// `` — a 412 is UNMODELED for the AWS SDK (the enum value is UNKNOWN), so the name is the only +/// machine-readable signal. +TEST(CASS3Signal, S3ExceptionCarriesCanonicalErrorName) +{ + DB::S3Exception e("412 from backend", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); + EXPECT_EQ(e.getExceptionName(), "PreconditionFailed"); + DB::S3Exception bare("no name attached", Aws::S3::S3Errors::UNKNOWN); + EXPECT_TRUE(bare.getExceptionName().empty()); +} + +namespace +{ + +/// WriteBuffer stub whose finalize throws a configured S3Exception — drives the classifier directly. +class ThrowOnFinalizeBuffer final : public DB::WriteBuffer +{ +public: + ThrowOnFinalizeBuffer() : DB::WriteBuffer(nullptr, 0) {} + + explicit ThrowOnFinalizeBuffer(DB::S3Exception e) : DB::WriteBuffer(nullptr, 0), to_throw(std::move(e)) {} + +private: + void nextImpl() override {} + + void finalizeImpl() override + { + if (to_throw) + throw *to_throw; /// NOLINT(cert-err09-cpp,cert-err60-cpp,cert-err61-cpp,misc-throw-by-value-catch-by-reference) -- the mock stores the configured exception to throw later, so it cannot be an anonymous temporary + } + + std::optional to_throw; +}; + +} + +/// detail::finalizeConditionalWrite maps a lost precondition to an OUTCOME by exact-matching the +/// canonical S3 error name (plus the modeled NO_SUCH_KEY enum, which WriteBufferFromS3 surfaces +/// nameless on retry exhaustion) and rethrows anything else. +TEST(CASS3Signal, FinalizeClassifierMapsPreconditionLossExactly) +{ + using DB::Cas::detail::finalizeConditionalWrite; + + auto classify = [](DB::S3Exception e) + { + ThrowOnFinalizeBuffer buf(std::move(e)); + return finalizeConditionalWrite(buf); + }; + + EXPECT_EQ(classify(DB::S3Exception("412", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed")), + PutOutcome::PreconditionFailed); + EXPECT_EQ(classify(DB::S3Exception("404 gone under If-Match", Aws::S3::S3Errors::UNKNOWN, "NoSuchKey")), + PutOutcome::PreconditionFailed); + EXPECT_EQ(classify(DB::S3Exception("retries exhausted, no name attached", Aws::S3::S3Errors::NO_SUCH_KEY)), + PutOutcome::PreconditionFailed); + + ThrowOnFinalizeBuffer unrelated(DB::S3Exception("503", Aws::S3::S3Errors::UNKNOWN, "SlowDown")); + EXPECT_THROW(finalizeConditionalWrite(unrelated), DB::S3Exception); + + ThrowOnFinalizeBuffer clean; + EXPECT_EQ(finalizeConditionalWrite(clean), PutOutcome::Done); +} + +namespace +{ + +/// A `LocalObjectStorage` that round-trips user metadata in-process. The production +/// `LocalObjectStorage` deliberately drops the `attributes` argument of `writeObject` and never +/// populates `ObjectMetadata::attributes` (local files carry no `x-amz-meta-*`), so it cannot stand +/// in for S3/RustFS when verifying the metadata threading. This test-only subclass records the +/// attributes passed on write, keyed by physical path, and injects them back on metadata reads — +/// exactly what a real object store does for `x-amz-meta-*`. It exercises the `EmulatedSingleProcess` +/// `ObjectStorageBackend` threading (`putIfAbsent` → `writeObject` attributes → `head` attributes) +/// without a live S3 backend; the real S3/RustFS round trip is verified empirically out-of-band. +class AttributePreservingLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + if (attributes.has_value()) + { + std::lock_guard lock(mutex); + saved_attributes[object.remote_path] = *attributes; + } + return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + if (metadata) + inject(path, *metadata); + return metadata; + } + + DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::getObjectMetadata(path, with_tags); + inject(path, metadata); + return metadata; + } + +private: + void inject(const std::string & path, DB::ObjectMetadata & metadata) const + { + std::lock_guard lock(mutex); + if (auto it = saved_attributes.find(path); it != saved_attributes.end()) + metadata.attributes = it->second; + } + + mutable std::mutex mutex; + mutable std::map saved_attributes; +}; + +DB::ObjectStoragePtr makeAttributePreservingStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_meta_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +} + +/// The `EmulatedSingleProcess` `ObjectStorageBackend` must thread user metadata through to the +/// underlying object storage's `writeObject` attributes on `putIfAbsent` and read it back into +/// `HeadResult::attributes` on `head`. Verified here over an attribute-preserving object storage +/// (the production `LocalObjectStorage` drops attributes); the live S3/RustFS round trip is verified +/// empirically out-of-band. +TEST(CASObjectStorageBackend, EmulatedRoundTripsUserMetadata) +{ + ObjectStorageBackend backend(makeAttributePreservingStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + const DB::Cas::ObjectMeta meta{{"cas_owner", "ab:7:42"}}; + ASSERT_EQ(backend.putIfAbsent("k/key", "body", meta).outcome, DB::Cas::PutOutcome::Done); + + const auto hr = backend.head("k/key"); + ASSERT_TRUE(hr.exists); + ASSERT_EQ(hr.attributes.at("cas_owner"), "ab:7:42"); +} + +namespace +{ + +/// A `LocalObjectStorage` whose `readObject` throws `S3Exception(NO_SUCH_KEY)` for a configured +/// physical key, while `tryGetObjectMetadata` still reports that key as PRESENT. +/// This simulates the HEAD→GET race window: the HEAD succeeds, then the object is deleted before +/// the GET arrives. +class NativeReadThrowsNoSuchKeyObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + void setThrowOnRead(const std::string & path) + { + throw_on_read_path = path; + } + + std::unique_ptr readObject( + const DB::StoredObject & object, + const DB::ReadSettings & read_settings, + std::optional read_hint, + bool use_external_buffer, + bool restrict_seek) const override + { + if (object.remote_path == throw_on_read_path) + throw DB::S3Exception( + "NoSuchKey: The specified key does not exist.", + Aws::S3::S3Errors::NO_SUCH_KEY); + + return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); + } + +private: + std::string throw_on_read_path; +}; + +struct ThrowOnReadFixture +{ + DB::ObjectStoragePtr storage; + /// Anchored under `storage`'s own root, because `Mode::Native` hands the key to the object storage + /// verbatim and this one is a real filesystem. + std::string key; +}; + +ThrowOnReadFixture makeThrowOnReadStorageForTest(const std::string & key_suffix) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_midget_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + auto storage = std::make_shared(std::move(settings)); + const std::string key = DB::Cas::tests::nativeKeyUnder(storage, key_suffix); + + /// Write the object so tryGetObjectMetadata reports it present (HEAD succeeds). + { + auto buf = storage->writeObject(DB::StoredObject(key), DB::WriteMode::Rewrite, std::nullopt); + buf->write("content", 7); + buf->finalize(); + } + + /// Now configure: future readObject calls for this key will throw NO_SUCH_KEY. + storage->setThrowOnRead(key); + return {std::move(storage), key}; +} + +} + +/// `ObjectStorageBackend::get` in `Native` mode: when `tryGetObjectMetadata` (`nativeHead`) reports the +/// key PRESENT but `readObject` throws `S3Exception(NO_SUCH_KEY)` — simulating a deletion in the +/// HEAD→GET window — `get` MUST return `std::nullopt` rather than letting the raw exception escape. +TEST(CASObjectStorageBackend, NativeModeGetReturnsNulloptOnMidGetNoSuchKey) +{ + /// The Native mode backend uses the key verbatim as the physical path (no emu_root prefix), so the + /// logical key IS the physical one the fixture wrote and armed. + const auto fixture = makeThrowOnReadStorageForTest("pool/blobs/ab/abcdef0123456789abcdef0123456789"); + + ObjectStorageBackend backend(fixture.storage, ObjectStorageBackend::Mode::Native); + + /// `get` HEADs before it reads and answers nullopt for an absent key, so without this the nullopt + /// below would be satisfied by an object the fixture failed to place — the mid-GET race would go + /// untested and the case would still pass. + Backend & iface = backend; + ASSERT_TRUE(iface.head(fixture.key).exists); + + /// HEAD reports the key present; readObject then throws NO_SUCH_KEY. + /// Contract: get must return std::nullopt, not propagate the S3Exception. + /// Call through the base-class interface so the default `Range{}` arg is available. + const auto result = iface.get(fixture.key); + EXPECT_FALSE(result.has_value()); +} + +/// A ranged `get` over a real `LocalObjectStorage` returns exactly the requested window, with the +/// same clamping the old read-whole-then-substr path had: a window whose offset is at or past EOF +/// yields an empty result. The only-the-window I/O property (no whole-object read) is enforced by +/// the `readObjectRanged` rewrite and cross-checked by the request-size gate in a later task. +TEST(CASObjectStorageBackend, RangedGetReadsOnlyTheWindow) +{ + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + const String payload = String(300000, 'a') + String(300000, 'b') + String(300000, 'c'); + backend->putIfAbsent("p/obj", payload); + + const auto mid = backend->get("p/obj", DB::Cas::Range{.offset = 300000, .length = 300000}); + ASSERT_TRUE(mid.has_value()); + EXPECT_EQ(mid->bytes, String(300000, 'b')); + + const auto tail = backend->get("p/obj", DB::Cas::Range{.offset = 600000, .length = std::nullopt}); + ASSERT_TRUE(tail.has_value()); + EXPECT_EQ(tail->bytes, String(300000, 'c')); + + const auto past = backend->get("p/obj", DB::Cas::Range{.offset = 1000000, .length = 10}); + ASSERT_TRUE(past.has_value()); + EXPECT_TRUE(past->bytes.empty()); +} + +/// codex-review-triage §3.18, finding 19c: the `EmulatedSingleProcess` adapter used to mint tokens +/// from a plain in-process counter (`emu_seq`), NOT actually seeded from the underlying object's etag +/// despite the class comment's claim. After a process restart (modeled here as a fresh +/// `ObjectStorageBackend` instance over the SAME storage) the counter restarts at 0 and can re-mint a +/// value that TEXTUALLY collides with a token persisted before the restart (e.g. a GC condemned-delete +/// token queued for replay), even though the two values name completely different incarnations of the +/// key. `deleteExact` must never let a stale, pre-restart token match a freshly recreated object. +TEST(CASObjectStorageBackend, EmuTokenSurvivesProcessRestartAcrossRecreate) +{ + auto storage = tests::makeLocalObjectStorageForTest(); + + auto backend1 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + /// A throwaway prior mutation on a DIFFERENT key: with the old counter this advances backend1's + /// process-wide op counter to 1, so "k/restart"'s own mint below lands on 2 — chosen so it collides + /// with backend2's post-restart recreate mint further down (also its SECOND op; see there). + ASSERT_EQ(backend1->putIfAbsent("k/other", "junk").outcome, PutOutcome::Done); + ASSERT_EQ(backend1->putIfAbsent("k/restart", "v1").outcome, PutOutcome::Done); + const Token stale_token = backend1->head("k/restart").token; + + /// Simulate a process restart: a brand-new `ObjectStorageBackend` instance (fresh emu state) over + /// the SAME underlying storage — exactly what happens when the CAS process restarts. + auto backend2 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + + /// Delete and recreate the key through the NEW instance — a fresh incarnation with a fresh mtime. + /// This is backend2's first-ever op (op 1) then a delete (no mint) then the recreate (op 2) — the + /// same op-index as `stale_token` above under the old counter, so the two textually collide there. + const Token current = backend2->head("k/restart").token; + ASSERT_EQ(backend2->deleteExact("k/restart", current).kind, DeleteOutcome::Kind::Deleted); + ASSERT_EQ(backend2->putIfAbsent("k/restart", "v2-after-restart").outcome, PutOutcome::Done); + + /// The pre-restart token must NEVER match the post-restart incarnation, however coincidentally a + /// process-local counter would have re-minted the identical textual value. + const auto stale_delete = backend2->deleteExact("k/restart", stale_token); + EXPECT_EQ(stale_delete.kind, DeleteOutcome::Kind::TokenMismatch); + + /// The live (post-restart) incarnation must be untouched by the rejected stale delete. + EXPECT_TRUE(backend2->head("k/restart").exists); +} + +/// codex-review-triage §3.18, finding №18: `list`'s `EmulatedSingleProcess` branch minted its per-key +/// token via `tokenForList`, which always stamps `native_token_type` (ETag) REGARDLESS of `mode` -- +/// while `head`/`get` mint `TokenType::Emulated`. `Token::operator==` compares type AND value, so a +/// list-derived token could never satisfy an emulated `deleteExact`/`putOverwrite` expectation: a +/// fail-safe leak (never a wrong delete), but every consumer of listed tokens (GC namespace cleanup, +/// `deletePrefixWholesale`, orphan sweep, decommission drain) always saw `TokenMismatch` against a +/// LOCAL pool. `list` must surface the SAME (type, value) as `head` for the same key. +TEST(CASObjectStorageBackend, EmulatedListTokenMatchesHeadToken) +{ + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + ASSERT_EQ(backend->putIfAbsent("k/listed", "body").outcome, PutOutcome::Done); + + const Token head_token = backend->head("k/listed").token; + ASSERT_EQ(head_token.type, TokenType::Emulated); + + const ListPage page = backend->list("k/", "", /*limit=*/10); + ASSERT_EQ(page.keys.size(), 1u); + ASSERT_TRUE(page.keys.front().token.has_value()); + EXPECT_EQ(*page.keys.front().token, head_token); +} + +namespace +{ + +/// A `LocalObjectStorage` whose reported etag never changes -- simulating a filesystem/clock whose +/// mtime resolution is too coarse to separate two writes issued back-to-back (the "same mtime +/// quantum" hazard flagged for the etag-seeded emu token: two DIFFERENT incarnations must still mint +/// DIFFERENT tokens even when the storage's own etag does not advance between them). +class FixedEtagLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + if (metadata) + metadata->etag = "same-quantum"; + return metadata; + } +}; + +DB::ObjectStoragePtr makeFixedEtagStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_fixed_etag_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +} + +/// The mtime-resolution guard (codex-review-triage §3.18, 19c step 4): two writes to the same key +/// whose underlying etag does not advance between them (stubbed here to model a coarse clock) must +/// still mint DISTINCT emulated tokens, and a stale token from the first incarnation must not match +/// the second. +TEST(CASObjectStorageBackend, EmuTokenDisambiguatesSameEtagRewrite) +{ + ObjectStorageBackend backend(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + const auto put1 = backend.putIfAbsent("k/tick", "v1"); + ASSERT_EQ(put1.outcome, PutOutcome::Done); + const auto put2 = backend.putOverwrite("k/tick", "v2", put1.token); + ASSERT_EQ(put2.outcome, PutOutcome::Done); + + EXPECT_NE(put1.token.value, put2.token.value); + EXPECT_EQ(put1.token.type, TokenType::Emulated); + EXPECT_EQ(put2.token.type, TokenType::Emulated); + + /// A stale delete using the FIRST incarnation's token must not match the live (second) one. + EXPECT_EQ(backend.deleteExact("k/tick", put1.token).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_TRUE(backend.head("k/tick").exists); +} + +TEST(CASObjectStorageBackend, PublishBlobEmulatedDisambiguatesSameEtagFromStaleDelete) +{ + ObjectStorageBackend backend(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + const String key = "k/publish-tick"; + + ASSERT_EQ(backend.putIfAbsent(key, "old-complete-body").outcome, PutOutcome::Done); + const Token stale_token = backend.head(key).token; + + backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)); + + const HeadResult published = backend.head(key); + ASSERT_TRUE(published.exists); + EXPECT_NE(published.token, stale_token); + EXPECT_EQ(published.token.type, TokenType::Emulated); + EXPECT_EQ(backend.deleteExact(key, stale_token).kind, DeleteOutcome::Kind::TokenMismatch); + + const auto live = backend.get(key); + ASSERT_TRUE(live.has_value()); + EXPECT_EQ(live->bytes, "fresh-envelopepayload"); + EXPECT_EQ(live->token, published.token); +} + +namespace +{ + +/// A `LocalObjectStorage` that always reports a caller-supplied, fixed NUMERIC etag string — lets a +/// test pin `emuMintToken`'s etag input to a precise, controlled nanosecond value (an old timestamp +/// vs. one close to "now") regardless of the real filesystem clock. Used to test the +/// `emu_token_state` erase-on-delete bound (codex-review-triage §3.18, Important #1): the entry +/// must be erased only when the deleted incarnation's own etag is comfortably in the past. +class FixedNumericEtagLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + FixedNumericEtagLocalObjectStorage(DB::LocalObjectStorageSettings settings, String etag_) + : DB::LocalObjectStorage(std::move(settings)), etag(std::move(etag_)) + { + } + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + if (metadata) + metadata->etag = etag; + return metadata; + } + +private: + String etag; +}; + +DB::ObjectStoragePtr makeFixedNumericEtagStorageForTest(const String & etag) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_fixed_numeric_etag_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings), etag); +} + +class ClockEtagLocalObjectStorage final : public DB::LocalObjectStorage +{ +public: + ClockEtagLocalObjectStorage(DB::LocalObjectStorageSettings settings, std::shared_ptr> now_ns_) + : DB::LocalObjectStorage(std::move(settings)), now_ns(std::move(now_ns_)) + { + } + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + if (metadata) + metadata->etag = std::to_string(now_ns->load()); + return metadata; + } + +private: + std::shared_ptr> now_ns; +}; + +DB::ObjectStoragePtr makeClockEtagStorageForTest(const std::shared_ptr> & now_ns) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_clock_etag_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings), now_ns); +} + +} + +/// codex-review-triage §3.18, Important #1: `emu_token_state` must be BOUNDED, not grow for the +/// lifetime of the backend instance. `deleteExact` erases a key's entry only when its last-minted +/// etag is comfortably (>= 2s) in the past — recent enough to still collide with an immediate +/// same-process recreate must be RETAINED (the mtime-quantum guard stays intact). +TEST(CASObjectStorageBackend, DeleteExactErasesEmuTokenStateOnlyWhenEtagIsComfortablyOld) +{ + /// An etag far in the past (nanoseconds since epoch, ~2001): delete must erase the entry, so an + /// immediate recreate reporting the SAME fixed etag is treated as a brand-new incarnation (bare + /// etag, no disambiguator) rather than a same-quantum tie with the just-consumed delete token. + { + const String old_etag = "1000000000000000000"; + ObjectStorageBackend backend(makeFixedNumericEtagStorageForTest(old_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + const auto put1 = backend.putIfAbsent("k/old", "v1"); + ASSERT_EQ(put1.outcome, PutOutcome::Done); + ASSERT_EQ(put1.token.value, old_etag); + ASSERT_EQ(backend.deleteExact("k/old", put1.token).kind, DeleteOutcome::Kind::Deleted); + + const auto put2 = backend.putIfAbsent("k/old", "v2"); + ASSERT_EQ(put2.outcome, PutOutcome::Done); + EXPECT_EQ(put2.token.value, old_etag) << "entry should have been erased on delete (etag comfortably old), " + "so the recreate mints the bare etag, not a disambiguated one"; + } + + /// An etag within the safety margin of "now": delete must RETAIN the entry, so the same + /// immediate-recreate scenario still gets disambiguated -- the guard this bound must not break. + { + const auto now_ns = std::chrono::duration_cast( + std::chrono::system_clock::now().time_since_epoch()).count(); + const String recent_etag = std::to_string(now_ns); + ObjectStorageBackend backend(makeFixedNumericEtagStorageForTest(recent_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + const auto put1 = backend.putIfAbsent("k/fresh", "v1"); + ASSERT_EQ(put1.outcome, PutOutcome::Done); + ASSERT_EQ(put1.token.value, recent_etag); + ASSERT_EQ(backend.deleteExact("k/fresh", put1.token).kind, DeleteOutcome::Kind::Deleted); + + const auto put2 = backend.putIfAbsent("k/fresh", "v2"); + ASSERT_EQ(put2.outcome, PutOutcome::Done); + EXPECT_EQ(put2.token.value, recent_etag + "#1") << "entry should have been RETAINED on delete (etag recent), " + "so the recreate is disambiguated against it"; + } +} + +TEST(CASObjectStorageBackend, EmuTokenStateEventuallyPrunesDistinctShortLivedKeys) +{ + constexpr uint64_t start_ns = 1'700'000'000'000'000'000ULL; + constexpr uint64_t step_ns = 100'000'000ULL; + constexpr size_t key_count = 128; + constexpr size_t expected_recent_key_bound = 24; + + auto now_ns = std::make_shared>(start_ns); + ObjectStorageBackend backend(makeClockEtagStorageForTest(now_ns), ObjectStorageBackend::Mode::EmulatedSingleProcess); + + for (size_t i = 0; i < key_count; ++i) + { + const uint64_t current_ns = start_ns + i * step_ns; + now_ns->store(current_ns); + backend.setEmuNowNsForTest(current_ns); + + const String key = "k/short-lived-" + std::to_string(i); + const auto put = backend.putIfAbsent(key, "body"); + ASSERT_EQ(put.outcome, PutOutcome::Done); + ASSERT_EQ(backend.deleteExact(key, put.token).kind, DeleteOutcome::Kind::Deleted); + } + + const uint64_t sweep_ns = start_ns + key_count * step_ns + 2'000'000'000ULL; + now_ns->store(sweep_ns); + backend.setEmuNowNsForTest(sweep_ns); + const auto trigger = backend.putIfAbsent("k/sweep-trigger", "body"); + ASSERT_EQ(trigger.outcome, PutOutcome::Done); + ASSERT_EQ(backend.deleteExact("k/sweep-trigger", trigger.token).kind, DeleteOutcome::Kind::Deleted); + + EXPECT_LE(backend.emuTokenStateSizeForTest(), expected_recent_key_bound) + << "token state should track only the bounded recent-key window, not all " << key_count << " deleted keys"; +} + +namespace +{ + +/// A `LocalObjectStorage` that counts `writeObject`/`removeObjectIfTokenMatches` calls -- used to +/// prove that a wrong-dialect expected token is rejected LOCALLY, before anything reaches the wire. +class CallCountingObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + ++write_calls; + return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } + + DB::ConditionalRemoveResult removeObjectIfTokenMatches(const DB::StoredObject & object, const std::string & etag) override + { + ++remove_if_matches_calls; + return DB::LocalObjectStorage::removeObjectIfTokenMatches(object, etag); + } + + std::atomic write_calls{0}; + std::atomic remove_if_matches_calls{0}; +}; + +DB::ObjectStoragePtr makeCallCountingStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_call_counting_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +} + +/// codex-review-triage §3.18, finding №19: Native-mode conditional mutations forward only +/// `Token::value` to the wire (`object_storage_write_if_match` / `removeObjectIfTokenMatches`), +/// blind to `Token::type`. A wrong-dialect token whose VALUE happens to equal the live incarnation's +/// must be rejected LOCALLY -- before any wire call is made -- never merely rely on the remote +/// backend to reject a foreign-dialect value it was never designed to compare. +TEST(CASObjectStorageBackend, NativeRejectsWrongDialectTokenBeforeTouchingTheWire) +{ + auto storage = std::static_pointer_cast(makeCallCountingStorageForTest()); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + ASSERT_EQ(backend.putIfAbsent("k/dialect", "v1").outcome, PutOutcome::Done); + const Token live = backend.head("k/dialect").token; + ASSERT_EQ(live.type, TokenType::ETag); + + storage->write_calls = 0; + storage->remove_if_matches_calls = 0; + + /// Same wire VALUE, wrong dialect TYPE (Emulated instead of this backend's native ETag dialect). + const Token wrong_type_token{live.value, TokenType::Emulated}; + + EXPECT_EQ(backend.putOverwrite("k/dialect", "v2", wrong_type_token).outcome, PutOutcome::PreconditionFailed); + EXPECT_EQ(backend.casPut("k/dialect", "v2", wrong_type_token).outcome, CasOutcome::Conflict); + EXPECT_EQ(backend.deleteExact("k/dialect", wrong_type_token).kind, DeleteOutcome::Kind::TokenMismatch); + + EXPECT_EQ(storage->write_calls.load(), 0); + EXPECT_EQ(storage->remove_if_matches_calls.load(), 0); + + /// The live incarnation must be untouched by all three rejected attempts. + EXPECT_EQ(backend.head("k/dialect").token, live); +} + +/// §1 (opt round-B): the fold/point GETs read tiny bodies but a default `ReadBufferFromS3` preallocates +/// ~1 MiB. `casSizedReadSettings` shrinks the buffer to the known body size + slack, capped at the +/// caller's default — never larger than before, regardless of the reported size. +TEST(CASSizedReadSettings, CapsToKnownSizePlusSlackButNeverAboveBase) +{ + DB::ReadSettings base; + base.remote_fs_settings.buffer_size = 1ULL << 20; /// 1 MiB default + base.local_fs_settings.buffer_size = 1ULL << 20; + + /// A ~3.7 KB fold body: buffer shrinks to size + slack, far below the 1 MiB default. + const auto small = DB::Cas::casSizedReadSettings(base, 3700); + EXPECT_EQ(small.remote_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); + EXPECT_EQ(small.local_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); + + /// A body larger than the default is capped AT the default (never grown). + const auto big = DB::Cas::casSizedReadSettings(base, 8ULL << 20); + EXPECT_EQ(big.remote_fs_settings.buffer_size, 1ULL << 20); + + /// Unknown size (0) = leave the base untouched (the metadata-fetch fallback path). + const auto unknown = DB::Cas::casSizedReadSettings(base, 0); + EXPECT_EQ(unknown.remote_fs_settings.buffer_size, 1ULL << 20); +} + +/// The CountingBackend request-shape recorders that the streaming-memory gates (Task 3/4) consume: +/// per-key/total getStream counts, the max ranged-get window per key, and the whole-object get flag. +TEST(CASCountingBackendShape, RecordsGetStreamAndRangeShape) +{ + DB::Cas::tests::CountingBackend backend; + backend.putIfAbsent("k", String(1000, 'x')); + + /// A whole-object get flags the resident-memory violation; a ranged get tracks the max window. + backend.get("k"); + backend.get("k", DB::Cas::Range{.offset = 0, .length = 100}); + backend.get("k", DB::Cas::Range{.offset = 10, .length = 400}); + EXPECT_EQ(backend.wholeGetCount("k"), 1u); + EXPECT_EQ(backend.maxRangedGetLen("k"), 400u); + + /// getStream counters (per-key and total). + backend.getStream("k", DB::Cas::Range{.offset = 2, .length = 5}); + backend.getStream("k"); + backend.getStream("absent"); + EXPECT_EQ(backend.getStreamCount("k"), 2u); + EXPECT_EQ(backend.getStreamTotal(), 3u); + + backend.resetCounts(); + EXPECT_EQ(backend.wholeGetCount("k"), 0u); + EXPECT_EQ(backend.maxRangedGetLen("k"), 0u); + EXPECT_EQ(backend.getStreamTotal(), 0u); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_backend_contract.cpp b/src/Disks/tests/gtest_cas_backend_contract.cpp new file mode 100644 index 000000000000..bf5c2652ca2a --- /dev/null +++ b/src/Disks/tests/gtest_cas_backend_contract.cpp @@ -0,0 +1,160 @@ +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +/// Parameterized contract suite: every case creates a fresh backend from the factory, +/// then exercises the Backend seam generically (no InMemoryBackend-specific calls). +/// Fault-injection-only features are excluded — those are InMemory-specific tests. +class CASBackendContract : public ::testing::TestWithParam> +{ +}; + +TEST_P(CASBackendContract, PutIfAbsentAndGet) +{ + auto b = GetParam()(); + const auto put = b->putIfAbsent("k", "v1"); + const Token t1 = put.token; + EXPECT_EQ(put.outcome, PutOutcome::Done); + EXPECT_FALSE(t1.empty()); + EXPECT_EQ(b->putIfAbsent("k", "clobber").outcome, PutOutcome::PreconditionFailed); + auto g = b->get("k"); + ASSERT_TRUE(g.has_value()); + EXPECT_EQ(g->bytes, "v1"); + EXPECT_EQ(g->token, t1); + EXPECT_FALSE(b->get("absent").has_value()); +} + +TEST_P(CASBackendContract, OverwriteIsTokenExactAndMintsFreshToken) +{ + auto b = GetParam()(); + const Token t1 = b->putIfAbsent("k", "v1").token; + EXPECT_EQ(b->putOverwrite("k", "v2", Token{"wrong", TokenType::Emulated}).outcome, PutOutcome::PreconditionFailed); + EXPECT_EQ(b->get("k")->bytes, "v1"); // untouched on mismatch + const auto overwrite = b->putOverwrite("k", "v2", t1); + EXPECT_EQ(overwrite.outcome, PutOutcome::Done); + EXPECT_NE(overwrite.token, t1); // tokens never repeat + EXPECT_EQ(b->get("k")->bytes, "v2"); +} + +TEST_P(CASBackendContract, CasPutCreateAndSwap) +{ + auto b = GetParam()(); + const auto create = b->casPut("m", "s1", std::nullopt); + const Token t1 = create.token; + EXPECT_EQ(create.outcome, CasOutcome::Committed); // create-if-absent + EXPECT_EQ(b->casPut("m", "s1x", std::nullopt).outcome, CasOutcome::Conflict); // exists now + EXPECT_EQ(b->casPut("m", "s2", Token{"stale", TokenType::Emulated}).outcome, CasOutcome::Conflict); + EXPECT_EQ(b->get("m")->bytes, "s1"); + EXPECT_EQ(b->casPut("m", "s2", t1).outcome, CasOutcome::Committed); + EXPECT_EQ(b->get("m")->bytes, "s2"); +} + +TEST_P(CASBackendContract, DeleteExactnessAndSurvival) +{ + auto b = GetParam()(); + const Token t1 = b->putIfAbsent("k", "v1").token; + auto d1 = b->deleteExact("k", Token{"wrong", TokenType::Emulated}); + EXPECT_EQ(d1.kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_TRUE(b->get("k").has_value()); // SURVIVES wrong-token delete + auto d2 = b->deleteExact("k", t1); + EXPECT_EQ(d2.kind, DeleteOutcome::Kind::Deleted); + EXPECT_FALSE(d2.created_delete_marker); + EXPECT_FALSE(b->get("k").has_value()); +} + +TEST_P(CASBackendContract, DeleteNotFound) +{ + auto b = GetParam()(); + const Token t1 = b->putIfAbsent("k", "v1").token; + b->deleteExact("k", t1); + EXPECT_EQ(b->deleteExact("k", t1).kind, DeleteOutcome::Kind::NotFound); +} + +TEST_P(CASBackendContract, RangeGet) +{ + auto b = GetParam()(); + b->putIfAbsent("k", "0123456789"); + Range r; + r.offset = 2; + r.length = 3u; + EXPECT_EQ(b->get("k", r)->bytes, "234"); +} + +TEST_P(CASBackendContract, Head) +{ + auto b = GetParam()(); + b->putIfAbsent("k", "hello"); + auto h = b->head("k"); + EXPECT_TRUE(h.exists); + EXPECT_EQ(h.size, 5u); + EXPECT_FALSE(h.token.empty()); + auto h2 = b->head("missing"); + EXPECT_FALSE(h2.exists); +} + +TEST_P(CASBackendContract, ListPagination) +{ + auto b = GetParam()(); + b->putIfAbsent("p/a", "0123456789"); + b->putIfAbsent("p/b", "xy"); + b->putIfAbsent("q/c", "z"); + auto page = b->list("p/", "", 10); + ASSERT_EQ(page.keys.size(), 2u); // sorted, prefix-scoped + EXPECT_EQ(page.keys[0].key, "p/a"); + EXPECT_EQ(page.keys[1].key, "p/b"); + EXPECT_TRUE(page.next_cursor.empty()); + auto page1 = b->list("p/", "", 1); // pagination + EXPECT_EQ(page1.keys.size(), 1u); + EXPECT_EQ(page1.keys[0].key, "p/a"); + EXPECT_EQ(page1.next_cursor, "p/a"); + EXPECT_FALSE(page1.next_cursor.empty()); + auto page2 = b->list("p/", page1.next_cursor, 1); + EXPECT_EQ(page2.keys[0].key, "p/b"); +} + +TEST_P(CASBackendContract, ReadAfterWrite) +{ + auto b = GetParam()(); + const Token t1 = b->putIfAbsent("rw", "payload").token; + auto g = b->get("rw"); + ASSERT_TRUE(g.has_value()); + EXPECT_EQ(g->bytes, "payload"); + EXPECT_EQ(g->token, t1); + auto h = b->head("rw"); + EXPECT_TRUE(h.exists); + EXPECT_EQ(h.token, t1); +} + +/// After an object is created then deleted (key absent again), BOTH conditional updates against a stale +/// token must be rejected with the object still absent — a token-conditional update can never recreate a +/// missing key. For the Native S3 adapter this pins the 404-on-If-Match -> PreconditionFailed/Conflict +/// mapping; for every backend it pins that absence is not a write opportunity for a stale token. +TEST_P(CASBackendContract, OverwriteAndCasOnMissingKey) +{ + auto b = GetParam()(); + const Token t1 = b->putIfAbsent("k", "v1").token; + EXPECT_EQ(b->deleteExact("k", t1).kind, DeleteOutcome::Kind::Deleted); + ASSERT_FALSE(b->get("k").has_value()); // key is absent + + EXPECT_EQ(b->putOverwrite("k", "v2", t1).outcome, PutOutcome::PreconditionFailed); + EXPECT_FALSE(b->get("k").has_value()); // still absent + + EXPECT_EQ(b->casPut("k", "v2", t1).outcome, CasOutcome::Conflict); + EXPECT_FALSE(b->get("k").has_value()); // still absent +} + +INSTANTIATE_TEST_SUITE_P(CASInMemory, CASBackendContract, + ::testing::Values(+[]() -> BackendPtr { return std::make_shared(); })); + +INSTANTIATE_TEST_SUITE_P(CASLocal, CASBackendContract, + ::testing::Values(+[]() -> BackendPtr + { + return std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + })); diff --git a/src/Disks/tests/gtest_cas_backend_generation.cpp b/src/Disks/tests/gtest_cas_backend_generation.cpp new file mode 100644 index 000000000000..b6f9502c45fa --- /dev/null +++ b/src/Disks/tests/gtest_cas_backend_generation.cpp @@ -0,0 +1,746 @@ +#include +#include +#include +#include + +#include "config.h" + +#if USE_AWS_S3 +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +#include +#include +#endif + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +#if USE_AWS_S3 +namespace DB::S3RequestSetting +{ +extern const S3RequestSettingsUInt64 max_single_part_upload_size; +extern const S3RequestSettingsUInt64 min_upload_part_size; +} + +namespace DB::S3AuthSetting +{ +extern const S3AuthSettingsUInt64 gcs_max_conditional_put_bytes; +} +#endif + +namespace +{ +/// A `LocalObjectStorage` that records which of the two metadata-read virtuals a caller reached, so a +/// test can prove `nativeHead` calls `tryGetObjectMetadataWithNativeToken` specifically -- reverting +/// that one line back to `tryGetObjectMetadata` makes `NativeHeadUsesNativeTokenMetadataApi` fail. +class RecordingObjectStorage : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + mutable int ordinary_calls = 0; + mutable int native_calls = 0; + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + ++ordinary_calls; + return DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + } + + std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const override + { + ++native_calls; + return DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + } +}; + +/// Same unique-temp-root convention as `DB::Cas::tests::makeLocalObjectStorageForTest`, but returning +/// the concrete recording type so the test can read its call counters. +std::shared_ptr makeRecordingObjectStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_unit_native_head_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +/// A `LocalObjectStorage` that answers the bucket-versioning probe with a value the test chooses, so +/// the three outcomes `checkPoolPreconditions` distinguishes — verified disabled, verified enabled, +/// and unverifiable — can each be driven exactly. The base `IObjectStorage` default answers only the +/// third. +class VersioningObjectStorage : public DB::LocalObjectStorage +{ +public: + VersioningObjectStorage(DB::LocalObjectStorageSettings settings_, std::optional versioned_) + : DB::LocalObjectStorage(std::move(settings_)), versioned(versioned_) + { + } + + std::optional isBucketVersioningEnabled() const override { return versioned; } + +private: + const std::optional versioned; +}; + +std::shared_ptr makeVersioningObjectStorageForTest(std::optional versioned) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_unit_versioning_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings), versioned); +} + +/// Every refusal reached from these mount gates is `NOT_IMPLEMENTED`, so the code alone cannot tell +/// which one fired. Match a phrase unique to the intended message as well, or a test asserting the +/// unverifiable-versioning refusal would pass on the enabled-bucket refusal and vice versa. +template +void expectThrowsNotImplementedSaying(const std::string & needle, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception saying '" << needle << "'"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + EXPECT_NE(e.message().find(needle), std::string::npos) << "actual message: " << e.message(); + } +} +} + +/// `ObjectStorageBackend::nativeHead` must route through `tryGetObjectMetadataWithNativeToken` (the +/// hook that lets a GCS-native client read a generation token), not the ordinary `tryGetObjectMetadata`. +TEST(CASBackendGeneration, NativeHeadUsesNativeTokenMetadataApi) +{ + auto storage = makeRecordingObjectStorageForTest(); + auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + + ASSERT_EQ(b->putIfAbsent("p/native-head/key", "v1").outcome, PutOutcome::Done); + + /// putIfAbsent's own HEAD-fallback stamping path calls the ordinary API (untouched by this task); + /// reset the counters so only nativeHead's call, below, is observed. + storage->ordinary_calls = 0; + storage->native_calls = 0; + + const auto hr = b->head("p/native-head/key"); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(storage->native_calls, 1); + EXPECT_EQ(storage->ordinary_calls, 0); +} + +/// Every Token{...} the backend mints must carry native_token_type instead of a hardcoded +/// TokenType::ETag (Task 5). Mode::Native over a LocalObjectStorage has no write-time ETag, so +/// putIfAbsent's PutResult falls back to a HEAD internally — that HEAD is also a stamping site, +/// so the assertion below exercises both the direct-etag and the HEAD-fallback mint paths. +TEST(CASBackendGeneration, StampedTokenTypeFollowsNativeKind) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + + const auto put = b->putIfAbsent("p/gen/tok", "v1"); + EXPECT_EQ(put.token.type, TokenType::Generation); + + const auto hr = b->head("p/gen/tok"); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(hr.token.type, TokenType::Generation); +} + +/// A generation-dialect (GCS) mount needs bucket versioning to be VERIFIABLY off: a token-exact +/// DELETE against a versioned bucket archives a noncurrent generation, so GC would delete objects it +/// believes it reclaimed. A probe that cannot answer therefore refuses the mount rather than +/// assuming the safe answer. +TEST(CASBackendGeneration, CheckPoolPreconditionsFailsClosedOnUnverifiableVersioning) +{ + auto b = std::make_shared( + makeVersioningObjectStorageForTest(std::nullopt), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + + expectThrowsNotImplementedSaying("could not VERIFY", [&] { b->checkPoolPreconditions(); }); +} + +TEST(CASBackendGeneration, CheckPoolPreconditionsRejectsEnabledVersioning) +{ + auto b = std::make_shared( + makeVersioningObjectStorageForTest(true), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + + expectThrowsNotImplementedSaying("VERSIONING enabled", [&] { b->checkPoolPreconditions(); }); +} + +/// The one accepting case: a probe that answered, and answered "disabled". +TEST(CASBackendGeneration, CheckPoolPreconditionsAcceptsVerifiedDisabledVersioning) +{ + auto b = std::make_shared( + makeVersioningObjectStorageForTest(false), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + + EXPECT_NO_THROW(b->checkPoolPreconditions()); +} + +/// The ETag-dialect (AWS-compatible) backend never consults bucket versioning at all — the check is +/// a silent no-op for any backend that is not Native + TokenType::Generation. Driven over a storage +/// whose probe is unverifiable, which is what a generation-dialect backend now refuses: dropping the +/// dialect guard from checkPoolPreconditions would fail this test. +TEST(CASBackendGeneration, CheckPoolPreconditionsNoOpOnEtagDialect) +{ + auto b = std::make_shared( + makeVersioningObjectStorageForTest(std::nullopt), ObjectStorageBackend::Mode::Native); + ASSERT_EQ(b->nativeTokenType(), TokenType::ETag); + + EXPECT_NO_THROW(b->checkPoolPreconditions()); +} + +/// A writable generation-dialect (GCS) mount may not skip the mutating capability battery: that +/// battery is the only proof that a token-exact DELETE actually carries its generation precondition. +TEST(CASBackendGeneration, CheckSkipAccessCheckSupportRejectsGenerationDialect) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + + expectThrowsNotImplementedSaying("skip_access_check=true is not supported", [&] { b->checkSkipAccessCheckSupport(); }); +} + +/// Scoped to the generation dialect: an ETag-dialect Native backend and the emulated backend keep the +/// pre-existing skip_access_check behaviour, so widening the refusal would fail this test. +TEST(CASBackendGeneration, CheckSkipAccessCheckSupportAllowsEtagAndEmulatedBackends) +{ + auto etag = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + ASSERT_EQ(etag->nativeTokenType(), TokenType::ETag); + EXPECT_NO_THROW(etag->checkSkipAccessCheckSupport()); + + auto emulated = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + EXPECT_NO_THROW(emulated->checkSkipAccessCheckSupport()); +} + +/// GCS enforces NO preconditions on CompleteMultipartUpload (measured 2026-07-03), so a conditional +/// write on a generation-token store must never take the multipart path. conditionalWriteSettings +/// must force the single-PUT path when the backend's native token kind is Generation, and stay a +/// no-op otherwise (ETag dialect). +TEST(CASBackendGeneration, ListTokensDisabledOnGenerationStores) +{ + /// XML LIST bodies carry MD5-style ETags that the dialect cannot rewrite to generations; a + /// list-derived token on a generation store is a poisoned If-Match (live GC on GCS died there). + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + EXPECT_TRUE(b->supportsListTokens()); + b->setNativeTokenTypeForTest(TokenType::Generation); + EXPECT_FALSE(b->supportsListTokens()); + b->setNativeTokenTypeForTest(TokenType::ETag); + EXPECT_TRUE(b->supportsListTokens()); +} + +TEST(CASBackendGeneration, ConditionalWriteSettingsForceSinglePutOnGenerationStores) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(TokenType::Generation); + const auto ws = b->conditionalWriteSettingsForTest(); + EXPECT_EQ(ws.object_storage_request_mode, DB::ObjectStorageRequestMode::NativeConditional); + EXPECT_TRUE(ws.s3_force_single_part_upload); + EXPECT_EQ(ws.object_storage_retry_profile, DB::ObjectStorageRetryProfile::SingleAttempt); + EXPECT_EQ(ws.s3_max_unexpected_write_error_retries_override, 1u); + ASSERT_TRUE(ws.s3_check_objects_after_upload_override.has_value()); + EXPECT_FALSE(*ws.s3_check_objects_after_upload_override); + + b->setNativeTokenTypeForTest(TokenType::ETag); + const auto ws2 = b->conditionalWriteSettingsForTest(); + EXPECT_EQ(ws2.object_storage_request_mode, DB::ObjectStorageRequestMode::NativeConditional); + EXPECT_FALSE(ws2.s3_force_single_part_upload); + EXPECT_EQ(ws2.object_storage_retry_profile, DB::ObjectStorageRetryProfile::SingleAttempt); + EXPECT_EQ(ws2.s3_max_unexpected_write_error_retries_override, 1u); + ASSERT_TRUE(ws2.s3_check_objects_after_upload_override.has_value()); + EXPECT_FALSE(*ws2.s3_check_objects_after_upload_override); +} + +/// C1: the three token-policy helpers are the single source of truth for how a Native-mode backend +/// mints a HEAD/PUT token, gates a LIST token, and compares tokens. Characterizes the behavior the +/// scattered call sites have today so the consolidation stays byte-for-byte behavior-preserving. +TEST(CASBackendGeneration, TokenPolicyHelpersAreConsistentWithDialect) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + + /// ETag dialect: head/put tokens carry ETag; list surfaces the same-typed token for a non-empty etag. + ASSERT_EQ(b->nativeTokenType(), TokenType::ETag); + EXPECT_EQ(b->tokenForHead("abc").type, TokenType::ETag); + EXPECT_EQ(b->tokenForHead("abc"), (Token{"abc", TokenType::ETag})); + ASSERT_TRUE(b->tokenForList("abc").has_value()); + EXPECT_EQ(*b->tokenForList("abc"), b->tokenForHead("abc")); /// list token == head token (same etag) + EXPECT_FALSE(b->tokenForList("").has_value()); /// empty etag => no list token + + /// Generation dialect (GCS): head token flips to Generation; list tokens are disabled wholesale + /// (poisoned If-Match), so tokenForList is always nullopt regardless of the etag. + b->setNativeTokenTypeForTest(TokenType::Generation); + EXPECT_EQ(b->tokenForHead("g1").type, TokenType::Generation); + EXPECT_FALSE(b->tokenForList("g1").has_value()); + + /// tokenMatches is exact identity (value AND type) — a same-value/different-type token never matches. + EXPECT_TRUE(ObjectStorageBackend::tokenMatches(Token{"x", TokenType::ETag}, Token{"x", TokenType::ETag})); + EXPECT_FALSE(ObjectStorageBackend::tokenMatches(Token{"x", TokenType::ETag}, Token{"x", TokenType::Emulated})); +} + +#if USE_AWS_S3 + +namespace +{ + +/// Minimal S3 double for the CasObjectStorageBackend generation-token write battery: just enough of +/// `DB::S3::Client` to drive a real `WriteBufferFromS3` end to end (`PutObject`, multipart upload, and +/// `HeadObject`). `GetObject` is not overridden: reading a written body back verifies against +/// `objects` directly (see the tests below), rather than through the considerably more involved +/// `ReadBufferFromS3` read path (range/retry/prefetch machinery), which this fake does not attempt to +/// support. +class FakeGenerationS3Client : public DB::S3::Client +{ +private: + struct State + { + std::string next_put_etag = "1000"; + bool put_returns_no_etag = false; + std::string next_head_etag; + + size_t put_object_calls = 0; + size_t head_object_calls = 0; + size_t create_multipart_calls = 0; + size_t upload_part_calls = 0; + size_t complete_multipart_calls = 0; + size_t abort_multipart_calls = 0; + + std::map objects; + std::map multipart_parts; + std::mutex mutex; + }; + + const std::shared_ptr state; + +public: + + FakeGenerationS3Client() + : FakeGenerationS3Client(std::make_shared(), GetClientConfiguration()) + { + } + + static DB::S3::PocoHTTPClientConfiguration GetClientConfiguration() + { + DB::RemoteHostFilter remote_host_filter; + return DB::S3::ClientFactory::instance().createClientConfiguration( + "some-region", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ true, + /* s3_slow_all_threads_after_retryable_error = */ true, + /* enable_s3_requests_logging = */ true, + /* for_disk_s3 = */ false, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + } + + /// The response ETag/generation the NEXT successful PutObject returns; empty means the response + /// carries no ETag at all (SetETag never called) -- the "broken/lying remote" case Step 7 guards. + std::string & next_put_etag; + bool & put_returns_no_etag; + std::string & next_head_etag; + + size_t & put_object_calls; + size_t & head_object_calls; + size_t & create_multipart_calls; + size_t & upload_part_calls; + size_t & complete_multipart_calls; + size_t & abort_multipart_calls; + + std::map & objects; + std::map & multipart_parts; + std::mutex & mutex; + + std::unique_ptr cloneWithConfigurationOverride( + const DB::S3::PocoHTTPClientConfiguration & client_configuration_override) const override + { + return std::unique_ptr(new FakeGenerationS3Client(state, client_configuration_override)); + } + + Aws::S3::Model::PutObjectOutcome PutObject(const Aws::S3::Model::PutObjectRequest & request) const override + { + std::lock_guard lock(mutex); + ++put_object_calls; + std::stringstream data; + data << request.GetBody()->rdbuf(); + objects[request.GetKey()] = data.str(); + + Aws::S3::Model::PutObjectResult result; + if (!put_returns_no_etag) + result.SetETag(next_put_etag); + return result; + } + + Aws::S3::Model::HeadObjectOutcome HeadObject(const Aws::S3::Model::HeadObjectRequest & request) const override + { + std::lock_guard lock(mutex); + ++head_object_calls; + Aws::S3::Model::HeadObjectOutcome outcome; + Aws::S3::Model::HeadObjectResult result(outcome.GetResultWithOwnership()); + auto it = objects.find(request.GetKey()); + result.SetContentLength(it == objects.end() ? 0 : it->second.size()); + if (!next_head_etag.empty()) + result.SetETag(next_head_etag); + return result; + } + + Aws::S3::Model::CreateMultipartUploadOutcome CreateMultipartUpload( + const Aws::S3::Model::CreateMultipartUploadRequest & /*request*/) const override + { + std::lock_guard lock(mutex); + ++create_multipart_calls; + multipart_parts.clear(); + Aws::S3::Model::CreateMultipartUploadResult result; + result.SetUploadId("publish-upload"); + return result; + } + + Aws::S3::Model::UploadPartOutcome UploadPart(const Aws::S3::Model::UploadPartRequest & request) const override + { + std::lock_guard lock(mutex); + ++upload_part_calls; + std::stringstream data; + data << request.GetBody()->rdbuf(); + multipart_parts[request.GetPartNumber()] = data.str(); + + Aws::S3::Model::UploadPartResult result; + result.SetETag("part-" + std::to_string(request.GetPartNumber())); + return result; + } + + Aws::S3::Model::CompleteMultipartUploadOutcome CompleteMultipartUpload( + const Aws::S3::Model::CompleteMultipartUploadRequest & request) const override + { + std::lock_guard lock(mutex); + ++complete_multipart_calls; + String body; + for (const auto & [part_number, part] : multipart_parts) + { + (void)part_number; + body += part; + } + objects[request.GetKey()] = std::move(body); + + Aws::S3::Model::CompleteMultipartUploadResult result; + if (!put_returns_no_etag) + result.SetETag(next_put_etag); + return result; + } + + Aws::S3::Model::AbortMultipartUploadOutcome AbortMultipartUpload( + const Aws::S3::Model::AbortMultipartUploadRequest & /*request*/) const override + { + std::lock_guard lock(mutex); + ++abort_multipart_calls; + multipart_parts.clear(); + return Aws::S3::Model::AbortMultipartUploadResult{}; + } + +private: + FakeGenerationS3Client( + std::shared_ptr state_, + const DB::S3::PocoHTTPClientConfiguration & client_configuration) + : DB::S3::Client( + 100, + DB::S3::ServerSideEncryptionKMSConfig(), + std::make_shared("", ""), + client_configuration, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + DB::S3::ClientSettings{ + .use_virtual_addressing = true, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }) + , state(std::move(state_)) + , next_put_etag(state->next_put_etag) + , put_returns_no_etag(state->put_returns_no_etag) + , next_head_etag(state->next_head_etag) + , put_object_calls(state->put_object_calls) + , head_object_calls(state->head_object_calls) + , create_multipart_calls(state->create_multipart_calls) + , upload_part_calls(state->upload_part_calls) + , complete_multipart_calls(state->complete_multipart_calls) + , abort_multipart_calls(state->abort_multipart_calls) + , objects(state->objects) + , multipart_parts(state->multipart_parts) + , mutex(state->mutex) + { + } + +}; + +std::shared_ptr makeGenerationS3ObjectStorageForTest( + FakeGenerationS3Client *& out_client, + bool force_multipart = false, + std::optional conditional_put_cap = {}) +{ + auto owned_client = std::make_unique(); + out_client = owned_client.get(); + + DB::S3::URI uri; + uri.bucket = "cas-generation-bucket"; + DB::S3Capabilities capabilities; + DB::ObjectStorageKeyGeneratorPtr key_generator; + + auto settings = std::make_unique(); + if (force_multipart) + { + settings->request_settings[DB::S3RequestSetting::max_single_part_upload_size] = 0; + settings->request_settings[DB::S3RequestSetting::min_upload_part_size] = 64; + } + + if (conditional_put_cap) + settings->auth_settings[DB::S3AuthSetting::gcs_max_conditional_put_bytes] = *conditional_put_cap; + + return std::make_shared( + std::move(owned_client), std::move(settings), std::move(uri), capabilities, key_generator, "cas-generation-disk"); +} + +} + +/// The "generation-token write kind" battery (Task 3, Step 2): a real WriteBufferFromS3 over a fake +/// S3 client, so the single-PUT cap enforcement and exact-token attribution are exercised for real, +/// not merely characterized through settings. Suite name deliberately starts with "CASBackendGeneration" +/// and every test name below contains "SinglePut", matching this plan's gtest filter. +class CASBackendGenerationS3 : public ::testing::Test +{ +protected: + FakeGenerationS3Client * client = nullptr; + std::shared_ptr backend; + + void SetUp() override + { + (void)getContext(); /// see S3ObjectStorageConditionalOpsTest::SetUp in gtest_writebuffer_s3.cpp + } + + /// A fresh backend, native token type forced to Generation unless overridden (the ETag dialect + /// is needed to prove the generation-only quote handling does not touch it). + std::shared_ptr makeBackend(TokenType token_type = TokenType::Generation) + { + auto storage = makeGenerationS3ObjectStorageForTest(client); + auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(token_type); + return b; + } +}; + +TEST(CASBackendGeneration, PublishBlobAboveFormerGenerationCapUsesOrdinaryMultipart) +{ + (void)getContext(); + FakeGenerationS3Client * client = nullptr; + auto storage = makeGenerationS3ObjectStorageForTest( + client, /*force_multipart=*/true, /*conditional_put_cap=*/16); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + backend.setNativeTokenTypeForTest(TokenType::Generation); + + const String payload(1024, 'x'); + backend.publishBlob(BlobPublishRequest{ + .destination_key = "p/gen/publish-multipart", + .publication = StreamingBlobPublication{ + .payload_size = payload.size(), + .fresh_envelope = "fresh", + .open_payload = [payload] + { + return std::make_unique(payload); + }}}); + + EXPECT_EQ(client->put_object_calls, 0u); + EXPECT_EQ(client->create_multipart_calls, 1u); + EXPECT_GT(client->upload_part_calls, 0u); + EXPECT_EQ(client->complete_multipart_calls, 1u); + EXPECT_EQ(client->abort_multipart_calls, 0u); + EXPECT_EQ(client->head_object_calls, 0u); + EXPECT_EQ(client->objects.at("p/gen/publish-multipart"), "fresh" + payload); +} + +TEST(CASBackendGeneration, PublishBlobSucceedsWithoutResponseGeneration) +{ + (void)getContext(); + FakeGenerationS3Client * client = nullptr; + auto storage = makeGenerationS3ObjectStorageForTest( + client, /*force_multipart=*/false, /*conditional_put_cap=*/1); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + backend.setNativeTokenTypeForTest(TokenType::Generation); + client->put_returns_no_etag = true; + + const String payload = "payload"; + EXPECT_NO_THROW(backend.publishBlob(BlobPublishRequest{ + .destination_key = "p/gen/publish-no-generation", + .publication = StreamingBlobPublication{ + .payload_size = payload.size(), + .fresh_envelope = "fresh", + .open_payload = [payload] + { + return std::make_unique(payload); + }}})); + + EXPECT_EQ(client->put_object_calls, 1u); + EXPECT_EQ(client->head_object_calls, 0u); + EXPECT_EQ(client->objects.at("p/gen/publish-no-generation"), "freshpayload"); +} + +/// The moved cap, end to end: a conditional write on a generation store stays in ONE PUT up to the +/// cap the OBJECT STORAGE carries, and refuses rather than silently taking the multipart path above +/// it -- GCS enforces no precondition on CompleteMultipartUpload. +TEST(CASBackendGeneration, ConditionalWriteHonoursTheObjectStorageConditionalPutCap) +{ + (void)getContext(); + FakeGenerationS3Client * client = nullptr; + auto storage = makeGenerationS3ObjectStorageForTest( + client, /*force_multipart=*/false, /*conditional_put_cap=*/64); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + backend.setNativeTokenTypeForTest(TokenType::Generation); + + const String small(32, 'a'); + EXPECT_NO_THROW(backend.casPut("p/gen/under-cap", small, std::nullopt, ObjectMeta{})); + EXPECT_EQ(client->put_object_calls, 1u); + EXPECT_EQ(client->create_multipart_calls, 0u); + const auto single_attempt_client = storage->getSingleAttemptClient(); + EXPECT_NE(dynamic_cast(single_attempt_client.get()), nullptr); + EXPECT_NE( + dynamic_cast( + single_attempt_client->getClientConfiguration().retryStrategy.get()), + nullptr); + + const String large(4096, 'b'); + try + { + backend.casPut("p/gen/over-cap", large, std::nullopt, ObjectMeta{}); + FAIL() << "a conditional write above the cap must refuse, not go multipart"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + } + EXPECT_EQ(client->create_multipart_calls, 0u); +} + +/// ---- The transport-quoting seam ---- +/// +/// A GCS generation reaches this layer through the SDK's ETag field, and the HTTP boundary fills that +/// field with an ETag-shaped, QUOTED value. Every test above this point feeds the write path an +/// UNQUOTED generation (`next_put_etag = "778899"`), and the HTTP-layer tests assert the field is +/// quoted -- each half self-consistent, neither crossing the seam between them. Nothing checked what +/// the CAS layer receives in the shape production actually produces, which is why a mount that could +/// never succeed passed every unit test. These three tests are that crossing. + +TEST_F(CASBackendGenerationS3, WriteEmptyGenerationThrows) +{ + backend = makeBackend(); + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::CORRUPTED_DATA, + [&] { backend->tokenFromWriteResult("p/gen/no-etag", String{}); }); +} + +TEST_F(CASBackendGenerationS3, WriteNonNumericGenerationThrows) +{ + backend = makeBackend(); + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::CORRUPTED_DATA, + [&] { backend->tokenFromWriteResult("p/gen/bad-etag", "\"d41d8cd98f00b204e9800998ecf8427e\""); }); +} + +/// A mutable conditional write whose response generation arrives quoted -- exactly what +/// `applyGcsConditionalDialectToResponse` produces -- must yield an UNQUOTED, all-digits token. +/// Before the fix this threw CORRUPTED_DATA, so every GCS CAS write failed and no pool could mount. +TEST_F(CASBackendGenerationS3, WriteGenerationTokenStripsTransportQuoting) +{ + backend = makeBackend(); + const Token tok = backend->tokenFromWriteResult("p/gen/quoted-write", "\"1783078552147137\""); + EXPECT_EQ(tok, (Token{"1783078552147137", TokenType::Generation})); +} + +/// The same crossing on the read side: a marked HEAD whose ETag field carries a quoted generation +/// must mint the same unquoted token, so a token observed by HEAD compares equal to one returned by +/// the write that created it. +TEST_F(CASBackendGenerationS3, HeadGenerationTokenStripsTransportQuoting) +{ + backend = makeBackend(); + client->objects["p/gen/quoted-head"] = "body"; + client->next_head_etag = "\"1783078552147137\""; + + const auto hr = backend->head("p/gen/quoted-head"); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(hr.token, (Token{"1783078552147137", TokenType::Generation})); +} + +/// The bound on that stripping. An ETag-dialect token IS the quoted ETag, and the quotes are required +/// syntax when it goes back out as `If-Match`, so the AWS-compatible path must keep them verbatim. +/// This is the test that fails if the quote handling is ever made unconditional. +TEST_F(CASBackendGenerationS3, EtagDialectKeepsTransportQuotingVerbatim) +{ + backend = makeBackend(TokenType::ETag); + client->objects["p/etag/quoted-head"] = "body"; + client->next_head_etag = "\"d41d8cd98f00b204e9800998ecf8427e\""; + + const auto hr = backend->head("p/etag/quoted-head"); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(hr.token, (Token{"\"d41d8cd98f00b204e9800998ecf8427e\"", TokenType::ETag})); +} + +/// A successful HEAD on a generation-dialect backend whose response carries no ETag/generation at all must not mint a token +/// from it -- there is no follow-up HEAD to patch this over, so nativeHead must refuse it directly. +TEST_F(CASBackendGenerationS3, HeadMissingGenerationThrows) +{ + backend = makeBackend(); + client->objects["p/gen/no-generation-head"] = "body"; + /// next_head_etag stays empty: SetETag is never called, so the response carries no ETag field. + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { backend->head("p/gen/no-generation-head"); }); +} + +/// An ordinary AWS-style ETag reaching a generation-dialect backend through a successful HEAD (a proxy dropping +/// x-goog-generation, a service regression) must not be minted as a generation token either. +TEST_F(CASBackendGenerationS3, HeadNonNumericGenerationThrows) +{ + backend = makeBackend(); + client->objects["p/gen/bad-etag-head"] = "body"; + client->next_head_etag = "\"d41d8cd98f00b204e9800998ecf8427e\""; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { backend->head("p/gen/bad-etag-head"); }); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_backend_listing.cpp b/src/Disks/tests/gtest_cas_backend_listing.cpp new file mode 100644 index 000000000000..591dfdba65db --- /dev/null +++ b/src/Disks/tests/gtest_cas_backend_listing.cpp @@ -0,0 +1,43 @@ +#include + +#include +#include + +#include +#include + +using namespace DB::Cas; + +TEST(CASBackendListing, ForEachWalksEveryPageOnce) +{ + InMemoryBackend b; + for (int i = 0; i < 2500; ++i) + b.putIfAbsent("p/" + std::to_string(1000000 + i), "v"); + b.putIfAbsent("q/other", "v"); /// out of prefix — must not be visited + + std::vector seen; + forEachListedKey(b, "p/", [&](const ListedKey & k) { seen.push_back(k.key); }, /*page_limit=*/1000); + EXPECT_EQ(seen.size(), 2500u); /// paged (3 pages), no key dropped/duplicated + EXPECT_TRUE(std::is_sorted(seen.begin(), seen.end())); +} + +TEST(CASBackendListing, ForEachEmptyPrefixVisitsNothing) +{ + InMemoryBackend b; + b.putIfAbsent("q/other", "v"); + + size_t visits = 0; + forEachListedKey(b, "p/", [&](const ListedKey &) { ++visits; }); + EXPECT_EQ(visits, 0u); +} + +TEST(CASBackendListing, ClassifyMapsEveryDeleteKind) +{ + EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::Deleted, false}), DeleteClass::Deleted); + EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::NotFound, false}), DeleteClass::Absent); + EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::TokenMismatch, false}), DeleteClass::Replaced); + + EXPECT_EQ(deleteClassName(DeleteClass::Deleted), "deleted"); + EXPECT_EQ(deleteClassName(DeleteClass::Absent), "absent"); + EXPECT_EQ(deleteClassName(DeleteClass::Replaced), "replaced"); +} diff --git a/src/Disks/tests/gtest_cas_blob_digest.cpp b/src/Disks/tests/gtest_cas_blob_digest.cpp new file mode 100644 index 000000000000..9a87f08ccda2 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_digest.cpp @@ -0,0 +1,266 @@ +#include + +/// CAS pluggable-blob-hash Phase 2, Task 1: `BlobDigest` (the pool-scoped variable-length content digest, ADDITIVE-ONLY -- no +/// existing `UInt128 blob_hash` field is migrated in this task) + the ONE `PoolMeta`-scoped +/// `DigestCodec` all digest<->hex/bytes conversion must route through. +/// +/// THE KEY GATE (`ShardOfBitIdenticalToOldHighBitsOver200RandomValues` below): `DigestCodec`'s +/// `shardOf` (an explicit big-endian read of the first 8 digest bytes) must be bit-identical to +/// today's `static_cast(blob_hash >> 64)` (`CasGcShardPlan.h`'s `blobShard`) for every +/// 128-bit digest -- otherwise an existing cityHash128/xxh3-128 pool would silently reshard on +/// upgrade. This is load-bearing: it is what makes Phase 2 safe to land under running pools. + +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include + +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +UInt128 randomU128(std::mt19937_64 & rng) +{ + const UInt128 hi = rng(); + const UInt128 lo = rng(); + return (hi << 64) | lo; +} + +} + +/// ---- THE KEY GATE ---- + +TEST(CASBlobDigest, ShardOfBitIdenticalToOldHighBitsOver200RandomValues) +{ + std::mt19937_64 rng(0xC0FFEE); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + const DigestCodec codec16(/*blob_hash_len*/ 16); + + for (int i = 0; i < 200; ++i) + { + const UInt128 v = randomU128(rng); + const uint64_t old_high64 = static_cast(v >> 64); + const uint64_t got = codec16.shardOf(BlobDigest::fromU128(v)); + EXPECT_EQ(got, old_high64) << "mismatch for random UInt128 iteration " << i; + } + + /// Edge cases: all-zero and all-one high halves. + EXPECT_EQ(codec16.shardOf(BlobDigest::fromU128(UInt128(0))), 0u); + const UInt128 all_ones = ~UInt128(0); + EXPECT_EQ(codec16.shardOf(BlobDigest::fromU128(all_ones)), static_cast(all_ones >> 64)); +} + +/// The same gate, but via `Cas::codecFor` (`CasBlobRef.h`), the ONE way production code obtains a +/// codec (Phase 3 T4 deleted the pool-scoped `DigestCodec(PoolMeta)` constructor -- a mixed-algo +/// pool has no single width; the codec is selected per-algo, never per-pool). +TEST(CASBlobDigest, ShardOfViaPoolMetaConstructedCodecMatchesOldBlobShard) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + ASSERT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); + const DigestCodec codec = codecFor(BlobHashAlgo::CityHash128); + + std::mt19937_64 rng(12345); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int i = 0; i < 200; ++i) + { + const UInt128 v = randomU128(rng); + EXPECT_EQ(codec.shardOf(BlobDigest::fromU128(v)), static_cast(v >> 64)); + /// `blobShard` (`CasGcShardPlan.h`) additionally takes `% gc_shards`; at `gc_shards == 1` + /// every hash routes to shard 0, so this only pins the trivial single-shard case -- the + /// bit-identical pre-mod value is already pinned by the assertion above. + EXPECT_EQ(blobShard(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(v)}, /*gc_shards*/ 1), 0u); + } +} + +/// ---- round-trip ---- + +TEST(CASBlobDigest, HexRoundTripLen16) +{ + const DigestCodec codec(16); + std::mt19937_64 rng(1); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int i = 0; i < 50; ++i) + { + const BlobDigest d = BlobDigest::fromU128(randomU128(rng)); + const String hex = codec.toHex(d); + EXPECT_EQ(hex.size(), 32u); + EXPECT_EQ(codec.fromHex(hex), d); + } +} + +TEST(CASBlobDigest, HexRoundTripLen32) +{ + const DigestCodec codec(32); + BlobDigest d; + for (size_t i = 0; i < d.bytes.size(); ++i) + d.bytes[i] = static_cast(i * 7 + 1); + + const String hex = codec.toHex(d); + EXPECT_EQ(hex.size(), 64u); + EXPECT_EQ(codec.fromHex(hex), d); +} + +TEST(CASBlobDigest, BytesBERoundTripLen16) +{ + const DigestCodec codec(16); + std::mt19937_64 rng(2); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int i = 0; i < 50; ++i) + { + const BlobDigest d = BlobDigest::fromU128(randomU128(rng)); + const String bytes = codec.toBytesBE(d); + EXPECT_EQ(bytes.size(), 16u); + EXPECT_EQ(codec.fromBytesBE(bytes), d); + } +} + +TEST(CASBlobDigest, BytesBERoundTripLen32) +{ + const DigestCodec codec(32); + BlobDigest d; + for (size_t i = 0; i < d.bytes.size(); ++i) + d.bytes[i] = static_cast(255 - i); + + const String bytes = codec.toBytesBE(d); + EXPECT_EQ(bytes.size(), 32u); + EXPECT_EQ(codec.fromBytesBE(bytes), d); +} + +/// `toBytesBE` at len16 must produce exactly the `u128ToBytesBE` bytes for the 16-byte prefix -- +/// same byte order, so a 128-bit pool's on-wire bytes stay unchanged when a later task migrates a +/// field from `UInt128` to `BlobDigest`. +TEST(CASBlobDigest, BytesBEAgreesWithU128ToBytesBEAtLen16) +{ + const DigestCodec codec(16); + std::mt19937_64 rng(3); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int i = 0; i < 20; ++i) + { + const UInt128 v = randomU128(rng); + EXPECT_EQ(codec.toBytesBE(BlobDigest::fromU128(v)), u128ToBytesBE(v)); + } +} + +/// ---- width rejection ---- + +TEST(CASBlobDigest, FromHexRejectsWrongWidth) +{ + const DigestCodec codec16(16); + const DigestCodec codec32(32); + + /// A 16-byte codec must reject a 64-hex (32-byte) string. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec16.fromHex(std::string(64, 'a')); }); + /// A 32-byte codec must reject a 32-hex (16-byte) string. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec32.fromHex(std::string(32, 'a')); }); + /// Any non-hex character is rejected too. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec16.fromHex(std::string(31, 'a') + "z"); }); +} + +TEST(CASBlobDigest, FromBytesBERejectsWrongWidth) +{ + const DigestCodec codec16(16); + const DigestCodec codec32(32); + + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec16.fromBytesBE(std::string(32, '\0')); }); + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec32.fromBytesBE(std::string(16, '\0')); }); + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { codec16.fromBytesBE(std::string(15, '\0')); }); +} + +/// ---- UInt128 conversion ---- + +TEST(CASBlobDigest, U128RoundTrip) +{ + std::mt19937_64 rng(4); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int i = 0; i < 200; ++i) + { + const UInt128 v = randomU128(rng); + EXPECT_EQ(BlobDigest::fromU128(v).toU128(), v); + } + EXPECT_EQ(BlobDigest::fromU128(UInt128(0)).toU128(), UInt128(0)); +} + +TEST(CASBlobDigest, FromU128LeavesTailZero) +{ + std::mt19937_64 rng(5); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + const UInt128 v = randomU128(rng); + const BlobDigest d = BlobDigest::fromU128(v); + for (size_t i = 16; i < d.bytes.size(); ++i) + EXPECT_EQ(d.bytes[i], 0u) << "tail byte " << i << " must be zero for a 128-bit-pool digest"; +} + +/// ---- hasher / container use ---- + +TEST(CASBlobDigest, UsableAsUnorderedMapKey) +{ + std::mt19937_64 rng(6); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + std::unordered_map m; + std::vector digests; + for (int i = 0; i < 20; ++i) + { + const BlobDigest d = BlobDigest::fromU128(randomU128(rng)); + digests.push_back(d); + m[d] = i; + } + for (int i = 0; i < 20; ++i) + EXPECT_EQ(m.at(digests[static_cast(i)]), i); +} + +/// ---- PoolMeta::algos_used records the creating algo (Phase 3 T4 -- the width itself is no longer +/// pool state at all: `blobHashLenFor(algo)`/`codecFor(algo)` derive it per-algo, never per-pool) ---- + +TEST(CASBlobDigest, PoolMetaRecordsCreatingAlgoAndWidthDerivesFromIt) +{ + { + auto backend = std::make_shared(); + const Layout layout("p1"); + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); + EXPECT_EQ(blobHashLenFor(BlobHashAlgo::CityHash128), 16u); + } + { + auto backend = std::make_shared(); + const Layout layout("p2"); + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); + EXPECT_EQ(blobHashLenFor(BlobHashAlgo::XXH3_128), 16u); + } + { + auto backend = std::make_shared(); + const Layout layout("p3"); + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::Sha256)})); + EXPECT_EQ(blobHashLenFor(BlobHashAlgo::Sha256), 32u); + + /// Reopen (decode path) must re-derive the same recorded algo. + const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256); + EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::Sha256)})); + } +} + +/// ---- zero-tail len-drift guard (debug/sanitizer builds only: chassert aborts the process) ---- + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASBlobDigestDeathTest, ZeroTailChassertFiresOnNonZeroTailAtLen16) +{ + const DigestCodec codec16(16); + BlobDigest d = BlobDigest::fromU128(UInt128(1)); + d.bytes[16] = 0x42; /// corrupt a tail byte beyond the pool's 16-byte width + + EXPECT_DEATH({ (void)codec16.toHex(d); }, ""); + EXPECT_DEATH({ (void)codec16.toBytesBE(d); }, ""); +} +#endif diff --git a/src/Disks/tests/gtest_cas_blob_envelope_format.cpp b/src/Disks/tests/gtest_cas_blob_envelope_format.cpp new file mode 100644 index 000000000000..1119a5be0a64 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_envelope_format.cpp @@ -0,0 +1,174 @@ +#include "cas_format_test_battery.h" +#include +#include + +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int UNKNOWN_FORMAT_VERSION; } + +namespace +{ +EnvelopeHeader sampleHeader(const String & ref) +{ + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = hexToU128("0102030405060708090a0b0c0d0e0f10"); + h.build_id = hexToU128("1112131415161718191a1b1c1d1e1f20"); + h.provenance = Provenance{1752537600123ULL, hexToU128("2122232425262728292a2b2c2d2e2f30"), 26006001u, ProvenanceOp::Merge}; + h.intended_ref = ref; + return h; +} +constexpr uint32_t L = 256; + +/// The envelope has a fixed physical length. At generation 9 there is no unsupported one-digit +/// version, so replacing `9` with `10` must consume one byte from the space pad rather than silently +/// turning the 256-byte fixture into a different wire shape. +String blobEnvelopeWithFutureVersion(std::string_view text) +{ + const String v_now = fmt::format("\"v\":{}", currentCompatibilityVersion()); + const String v_next = fmt::format("\"v\":{}", currentCompatibilityVersion() + 1); + String future(text); + const size_t version_pos = future.find(v_now); + if (version_pos == String::npos || v_next.size() < v_now.size()) + throw std::logic_error("blob-envelope future-version fixture cannot locate the current version"); + + future.replace(version_pos, v_now.size(), v_next); + const size_t growth = v_next.size() - v_now.size(); + const size_t newline_pos = future.find('\n'); + if (newline_pos == String::npos || newline_pos < growth + || future.substr(newline_pos - growth, growth) != String(growth, ' ')) + throw std::logic_error("blob-envelope future-version fixture has insufficient padding"); + future.erase(newline_pos - growth, growth); + return future; +} +} + +TEST(CASBlobEnvelopeFormat, FixedLengthAndPadZone) +{ + EnvelopeHeader h = sampleHeader("t-abc/all_1_2_0"); + const String head = encodeEnvelopeHeader(h, L); + ASSERT_EQ(head.size(), L); /// exactly blob_header_len + EXPECT_EQ(head[L - 1], '\n'); /// terminator at byte 255 + const String json = fmt::format(R"({{"type":"cas_blob","v":{},)", currentCompatibilityVersion()) + + "\"tag\":\"0102030405060708090a0b0c0d0e0f10\"," + "\"bld\":\"1112131415161718191a1b1c1d1e1f20\",\"ts\":1752537600123," + "\"by\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"ch\":26006001," + "\"ref\":\"t-abc/all_1_2_0\"}"; + ASSERT_LT(json.size(), L); + EXPECT_EQ(head.substr(0, json.size()), json); /// '/' UNescaped (local escaper) + EXPECT_EQ(head.substr(json.size(), (L - 1) - json.size()), String((L - 1) - json.size(), ' ')); /// pad = spaces + /// round-trip + const EnvelopeHeader back = decodeEnvelopeHeader(head, head.size(), ObjectKind::Blob); + EXPECT_EQ(back.incarnation_tag, h.incarnation_tag); + EXPECT_EQ(back.build_id, h.build_id); + ASSERT_TRUE(back.provenance.has_value()); + EXPECT_EQ(back.provenance->created_at_ms, 1752537600123ULL); + EXPECT_EQ(back.provenance->ch_version, 26006001u); + EXPECT_EQ(back.provenance->op, ProvenanceOp::Merge); + ASSERT_TRUE(back.intended_ref.has_value()); + EXPECT_EQ(*back.intended_ref, "t-abc/all_1_2_0"); + EXPECT_EQ(back.header_len, L); + EXPECT_EQ(payloadOffset(back), L); +} + +TEST(CASBlobEnvelopeFormat, RefTruncatedToExactBudget) +{ + /// A 200-char ref cannot fit; it is truncated so the header is EXACTLY 256 bytes and the pad holds. + EnvelopeHeader h = sampleHeader(String(200, 'a')); + const String head = encodeEnvelopeHeader(h, L); + ASSERT_EQ(head.size(), L); + EXPECT_EQ(head[L - 1], '\n'); + const EnvelopeHeader back = decodeEnvelopeHeader(head, head.size(), ObjectKind::Blob); + ASSERT_TRUE(back.intended_ref.has_value()); + /// Budget is deterministic. Compute json_len for the SAME header with an empty ref; each extra 'a' + /// is one escaped byte, so the truncated 'a' count is exactly (L-1) - json_len(empty ref). + EnvelopeHeader probe = sampleHeader(""); + const String empty_ref_head = encodeEnvelopeHeader(probe, L); + const size_t json_len_empty = empty_ref_head.find_last_not_of(' ', (L - 1) - 1) + 1; + const size_t budget = (L - 1) - json_len_empty; + EXPECT_EQ(back.intended_ref->size(), budget) << "ref truncated to the exact byte budget"; + for (char c : *back.intended_ref) + EXPECT_EQ(c, 'a'); +} + +TEST(CASBlobEnvelopeFormat, PadZoneSmugglingFailsClosed) +{ + EnvelopeHeader h = sampleHeader("r"); + const String head = encodeEnvelopeHeader(h, L); + const size_t json_len = head.find_last_not_of(' ', (L - 1) - 1) + 1; /// first pad byte index = json_len + ASSERT_LT(json_len, L - 1); + /// A non-space byte smuggled into the pad zone -> CORRUPTED_DATA. + String smuggled = head; + smuggled[json_len + 1] = 'x'; + EXPECT_THROW(decodeEnvelopeHeader(smuggled, smuggled.size(), ObjectKind::Blob), DB::Exception); + /// Byte 255 not '\n' -> CORRUPTED_DATA. + String no_nl = head; + no_nl[L - 1] = ' '; + EXPECT_THROW(decodeEnvelopeHeader(no_nl, no_nl.size(), ObjectKind::Blob), DB::Exception); +} + +TEST(CASBlobEnvelopeFormat, GatesAndCriticalKey) +{ + /// wrong type -> CORRUPTED_DATA; future v -> UNKNOWN_FORMAT_VERSION. + EnvelopeHeader h = sampleHeader("r"); + const String head = encodeEnvelopeHeader(h, L); + String wrong_type = head; + wrong_type.replace(wrong_type.find("cas_blob"), 8, "cas_xxxx"); + EXPECT_THROW(decodeEnvelopeHeader(wrong_type, wrong_type.size(), ObjectKind::Blob), DB::Exception); + const String current_version = fmt::format("\"v\":{}", currentCompatibilityVersion()); + String future = blobEnvelopeWithFutureVersion(head); + cas_battery_detail::expectCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, + [&] { decodeEnvelopeHeader(future, future.size(), ObjectKind::Blob); }, "future blob-envelope version"); + + String out_of_range = head; + const size_t out_of_range_version_at = out_of_range.find(current_version); + ASSERT_NE(out_of_range_version_at, String::npos); + out_of_range.replace(out_of_range_version_at, current_version.size(), "\"v\":4294967299"); + try + { + decodeEnvelopeHeader(out_of_range, out_of_range.size(), ObjectKind::Blob); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + /// an unknown `!`-critical key fails closed. + EnvelopeHeader hc = sampleHeader("r"); + hc.emit_unknown_critical_key = true; + const String crit = encodeEnvelopeHeader(hc, L); + EXPECT_THROW(decodeEnvelopeHeader(crit, crit.size(), ObjectKind::Blob), DB::Exception); +} + +TEST(CASBlobEnvelopeFormat, RefEscaperAlphabetPinned) +{ + /// Pins the LOCAL escaper's alphabet (§ref-escaper): " and \ escape, control chars -> \uXXXX, + /// '/' passes VERBATIM. Goes RED if anyone "unifies" this with writeStringValue/FormatSettings — + /// the 256-byte budget arithmetic depends on this alphabet being codec-owned and frozen. + EnvelopeHeader h = sampleHeader(String("a/b\"c\\d") + '\x01' + "e"); + const String head = encodeEnvelopeHeader(h, L); + const String expected_ref_json = R"("a/b\"c\\d\u0001e")"; + EXPECT_NE(head.find("\"ref\":" + expected_ref_json), String::npos) + << "escaper alphabet drifted: '/' must be verbatim, quote/backslash escaped, control -> \\uXXXX"; +} + +TEST(CASFormatBattery, BlobEnvelope) +{ + /// The golden is CONSTRUCTED from the hand-pinned json literal (same one FixedLengthAndPadZone + /// asserts) + the derived pad — NOT self-computed via encodeEnvelopeHeader, which would compare + /// the encoder to itself and pin nothing. + const String json = fmt::format(R"({{"type":"cas_blob","v":{},)", currentCompatibilityVersion()) + + "\"tag\":\"0102030405060708090a0b0c0d0e0f10\"," + "\"bld\":\"1112131415161718191a1b1c1d1e1f20\",\"ts\":1752537600123," + "\"by\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"ch\":26006001," + "\"ref\":\"t-abc/all_1_2_0\"}"; + const String golden = json + String((L - 1) - json.size(), ' ') + '\n'; + runFormatBattery(FormatBatteryCase{ + .id = FormatId::Blob, + .encode = [&] { EnvelopeHeader e = sampleHeader("t-abc/all_1_2_0"); return sealObject(FormatId::Blob, encodeEnvelopeHeader(e, L)); }, + .decode = [](std::string_view s) { decodeEnvelopeHeader(String(openObject(FormatId::Blob, s)), s.size(), ObjectKind::Blob); }, + .golden = golden, + .make_future_version = blobEnvelopeWithFutureVersion}); +} diff --git a/src/Disks/tests/gtest_cas_blob_hasher.cpp b/src/Disks/tests/gtest_cas_blob_hasher.cpp new file mode 100644 index 000000000000..0aa1b3ff35a4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_hasher.cpp @@ -0,0 +1,182 @@ +#include + +#include +/// `CasXxh3Streamer.h` is the isolated xxHash wrapper (a system header): it gives us `Cas::xxh3_128_oneshot` +/// as an independent one-shot reference without pulling raw xxHash symbols (or their warnings) into +/// this test — see the header's own comment for the lz4-shadowing / `-Werror` reasons. +#include +#include +#include +#include +#include + +#include +#include + +using namespace DB; +using namespace DB::Cas; + +namespace +{ + +/// A deterministic, non-repeating-byte payload (not all-zero / all-same, so a byte-order or +/// endianness bug in either hash path would not accidentally cancel out). +std::string makePayload(size_t size) +{ + std::string s; + s.reserve(size); + for (size_t i = 0; i < size; ++i) + s.push_back(static_cast('a' + (i % 23))); + return s; +} + +} + +TEST(CASBlobHasher, Xxh3StreamingMatchesOneShotAndBlobHashHexOneShot) +{ + const std::string payload = makePayload(10000); + + std::string sink_data; + std::string streaming_hex; + { + WriteBufferFromString sink(sink_data); + auto hashing = makeBlobHashingWriteBuffer(BlobHashAlgo::XXH3_128, sink); + + /// Feed the payload through several `write()` chunks to exercise the streaming state across + /// multiple `nextImpl` flushes, not just a single call. + size_t offset = 0; + constexpr size_t chunk = 777; + while (offset < payload.size()) + { + const size_t n = std::min(chunk, payload.size() - offset); + hashing->write(payload.data() + offset, n); + offset += n; + } + + streaming_hex = hashing->getHashHex(); + hashing->finalize(); + sink.finalize(); + } + + /// The passthrough forwarded every byte unchanged. + EXPECT_EQ(sink_data, payload); + EXPECT_EQ(streaming_hex.size(), 32u); + + /// xxh3 streaming == xxh3 one-shot (unlike cityHash128, xxh3's streaming digest is defined to + /// agree with the one-shot digest -- see `ImplXXH3_128` in `Functions/FunctionsHashing.h`). + UInt64 os_low = 0; + UInt64 os_high = 0; + Cas::xxh3_128_oneshot(payload.data(), payload.size(), os_low, os_high); + const std::string one_shot_hex = getHexUIntLowercase(UInt128{os_low, os_high}); + EXPECT_EQ(streaming_hex, one_shot_hex); + + /// The one-shot re-hash helper must agree with both. + EXPECT_EQ(blobHashHexOneShot(BlobHashAlgo::XXH3_128, payload), one_shot_hex); +} + +TEST(CASBlobHasher, CityHash128ByteIdenticalToHashingWriteBuffer) +{ + /// Cover payloads both under and over one `DBMS_DEFAULT_HASHING_BLOCK_SIZE` (2048 B) hash block, + /// plus exactly at the boundary, since the chunked convention only matters once a payload spans + /// more than one block. + for (const size_t size : {size_t(100), size_t(2000), size_t(2048), size_t(5000)}) + { + SCOPED_TRACE(size); + const std::string payload = makePayload(size); + + /// Reference: today's convention, `HashingWriteBuffer` used directly. + std::string ref_sink_data; + std::string ref_hex; + { + WriteBufferFromString ref_sink(ref_sink_data); + HashingWriteBuffer ref_hashing(ref_sink); + ref_hashing.write(payload.data(), payload.size()); + ref_hex = getHexUIntLowercase(ref_hashing.getHash()); + ref_hashing.finalize(); + ref_sink.finalize(); + } + + /// The selectable factory, defaulted to CityHash128 -- must be byte-identical. + std::string sink_data; + std::string hex; + { + WriteBufferFromString sink(sink_data); + auto hashing = makeBlobHashingWriteBuffer(BlobHashAlgo::CityHash128, sink); + hashing->write(payload.data(), payload.size()); + hex = hashing->getHashHex(); + hashing->finalize(); + sink.finalize(); + } + + EXPECT_EQ(hex, ref_hex); + EXPECT_EQ(hex.size(), 32u); + EXPECT_EQ(sink_data, ref_sink_data); + EXPECT_EQ(sink_data, payload); + + /// The one-shot re-hash helper must agree too. + EXPECT_EQ(blobHashHexOneShot(BlobHashAlgo::CityHash128, payload), ref_hex); + } +} + +TEST(CASBlobHasher, AlgoNameAndParseRoundTrip) +{ + EXPECT_EQ(blobHashAlgoName(BlobHashAlgo::CityHash128), "ch128"); + EXPECT_EQ(blobHashAlgoName(BlobHashAlgo::XXH3_128), "xxh3"); + EXPECT_EQ(blobHashAlgoName(BlobHashAlgo::Sha256), "sha256"); + + EXPECT_EQ(parseBlobHashAlgo("cityhash128"), BlobHashAlgo::CityHash128); + EXPECT_EQ(parseBlobHashAlgo("xxh3-128"), BlobHashAlgo::XXH3_128); + /// Parses even though it is rejected downstream (config-layer rejection is a later task). + EXPECT_EQ(parseBlobHashAlgo("sha256"), BlobHashAlgo::Sha256); + + EXPECT_THROW(parseBlobHashAlgo("bogus"), DB::Exception); + EXPECT_THROW(parseBlobHashAlgo("cityHash128"), DB::Exception); // case-sensitive + EXPECT_THROW(parseBlobHashAlgo(""), DB::Exception); +} + +TEST(CASBlobHasher, Sha256OneShotGoldenVectors) +{ + /// NIST/FIPS 180-2 test vectors, the standard SHA-256 sanity check. + EXPECT_EQ(blobHashHexOneShot(BlobHashAlgo::Sha256, "abc"), + "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"); + EXPECT_EQ(blobHashHexOneShot(BlobHashAlgo::Sha256, ""), + "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"); +} + +TEST(CASBlobHasher, Sha256StreamingMatchesOneShotAndIsPassthrough) +{ + /// Bigger than `DBMS_DEFAULT_HASHING_BLOCK_SIZE` (2048 B) so the payload chunks through several + /// `nextImpl` flushes, not just a single call. + const std::string payload = makePayload(200 * 1024); + + std::string sink_data; + std::string streaming_hex; + { + WriteBufferFromString sink(sink_data); + auto hashing = makeBlobHashingWriteBuffer(BlobHashAlgo::Sha256, sink); + + /// Feed the payload through several `write()` chunks to exercise the streaming EVP digest + /// across multiple `nextImpl` flushes. + size_t offset = 0; + constexpr size_t chunk = 4096; + while (offset < payload.size()) + { + const size_t n = std::min(chunk, payload.size() - offset); + hashing->write(payload.data() + offset, n); + offset += n; + } + + streaming_hex = hashing->getHashHex(); + hashing->finalize(); + sink.finalize(); + } + + /// The passthrough forwarded every byte unchanged. + EXPECT_EQ(sink_data, payload); + EXPECT_EQ(streaming_hex.size(), 64u); + + /// SHA-256 streaming == SHA-256 one-shot (like xxh3, unlike cityHash128 -- SHA-256 has no + /// chunked convention to preserve). + const std::string one_shot_hex = blobHashHexOneShot(BlobHashAlgo::Sha256, payload); + EXPECT_EQ(streaming_hex, one_shot_hex); +} diff --git a/src/Disks/tests/gtest_cas_blob_indegree.cpp b/src/Disks/tests/gtest_cas_blob_indegree.cpp new file mode 100644 index 000000000000..e47b7351f063 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_indegree.cpp @@ -0,0 +1,1000 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int NOT_IMPLEMENTED; } + +using namespace DB::Cas; + +namespace +{ +UInt128 b(uint64_t n) { return UInt128(n); } +UInt128 s(uint64_t n) { return UInt128(n); } // source-edge id +/// A `BlobRef` (CityHash128) for the same literal `n` — every existing test's `BlobDelta.ref` / +/// `BlobCandidate.ref` / `inDegreeInRuns` argument is a `BlobRef` as of Phase 3 T3. +BlobRef bh(uint64_t n) { return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(n))}; } + +/// Scale thresholds for the "the run genuinely spans several blocks" sanity assertions below. These are +/// NOT format constants — the SourceEdge run is a plain NDJSON stream (`CasRecordStreamFormat`) with no +/// block framing of its own — they only pin the same byte-size scale the (now-deleted, codecs-v3 phase 6) +/// `CasRunFile` block codec used, so the multi-block-sized fixtures below stay meaningfully large. +/// (Previously read straight off `CasRunFile.h`'s own `kRunTargetBlockSize`/`kRunHardCapBlockSize`; this +/// file's `#include` of that header looked removable when `CasRunFile` was deleted in the phase-6 cutover, +/// but these two thresholds turned out to be the only remaining users — hence the local, explicitly-legacy +/// copies here instead of a dangling include. Values unchanged.) +constexpr uint32_t kLegacyBlockSize = 256u * 1024u; +constexpr uint32_t kLegacyHardCapBlockSize = 1024u * 1024u; +} + +TEST(CASBlobInDegree, FoldStartsFromEmptyPriorGeneration) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Generation 1 from empty prior: two distinct edges on b1 and one on b2. + /// Edge (b1,s1), (b1,s2), (b2,s1) => indeg(b1)=2, indeg(b2)=1. + std::vector deltas{ + {bh(1), s(1), false}, + {bh(1), s(2), false}, + {bh(2), s(1), false}, + }; + std::vector runs; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new*/1, /*attempt*/0, /*shard*/0, deltas, runs); + ASSERT_FALSE(runs.empty()); + + const auto zero = zeroInDegree(backend, runs); + EXPECT_TRUE(zero.empty()); /// nothing at zero yet +} + +TEST(CASBlobInDegree, PlusMinusCancelToZeroDetectsCandidate) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Gen 1: activate edge (b1,s1) and (b2,s1). + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + {{bh(1), s(1), false}, {bh(2), s(1), false}}, runs1); + + /// Generation 2 merges prior gen-1 run (resolved via runs1 refs) with removal of (b1,s1): indeg(b1)=0, indeg(b2)=1. + std::vector runs2; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new*/2, /*attempt*/0, 0, + {{bh(1), s(1), true}}, runs2); + + const auto zero = zeroInDegree(backend, runs2); + ASSERT_EQ(zero.size(), 1u); + EXPECT_EQ(zero[0].ref, bh(1)); +} + +TEST(CASBlobInDegree, RunsAreByteDeterministic) +{ + InMemoryBackend a; + InMemoryBackend b2; + Layout layout{"pool"}; + std::vector ra; + std::vector rb; + /// Same deltas in a DIFFERENT input order must produce the same sealed run bytes (sorted by key). + foldDeltasIntoGeneration(a, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + {{bh(3), s(1), false}, {bh(1), s(1), false}, {bh(2), s(1), false}}, ra); + foldDeltasIntoGeneration(b2, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(3), s(1), false}}, rb); + const auto ga = a.get(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0)); + const auto gb = b2.get(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0)); + ASSERT_TRUE(ga.has_value()); + ASSERT_TRUE(gb.has_value()); + EXPECT_EQ(ga->bytes, gb->bytes); + ASSERT_EQ(ra.size(), 1u); + ASSERT_EQ(rb.size(), 1u); + EXPECT_EQ(ra[0].checksum, rb[0].checksum); +} + +TEST(CASBlobInDegree, SameEdgeActivatedTwiceCountsOnce) +{ + /// Idempotency: activating the same (blob_hash, source_id) twice must not double-count. + /// The source-edge set is a SET, not a counter — re-adding the same edge is a no-op. + /// indeg(b1) must be 1 after both activations, not 2. + InMemoryBackend backend; + Layout layout{"pool"}; + std::vector deltas{ + {bh(1), s(1), false}, // activate (b1,s1) + {bh(1), s(1), false}, // same edge again — must deduplicate + }; + std::vector runs; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, deltas, runs); + ASSERT_FALSE(runs.empty()); + + const int64_t deg = DB::Cas::tests::inDegreeInRuns(backend, runs, bh(1)); + EXPECT_EQ(deg, 1); /// deduplicated, not 2 + + const auto zero = zeroInDegree(backend, runs); + EXPECT_TRUE(zero.empty()); /// b1 still has an active edge +} + +TEST(CASBlobInDegree, FoldDeltaByteEqualReplayAdopts) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + std::vector deltas{{bh(1), s(1), false}}; + std::vector runs1; + std::vector runs2; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs1); + /// Same inputs, same attempt => byte-identical run already present => adopt, no throw. + EXPECT_NO_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs2)); + EXPECT_EQ(runs1, runs2); +} + +TEST(CASBlobInDegree, FoldDeltaDivergentBytesThrowsCorrupted) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + /// Pre-occupy the run key (attempt 7) with junk, then fold => divergent => CORRUPTED_DATA. + backend.putIfAbsent(layout.blobTargetRunKey(1, /*attempt*/7, /*shard*/0, /*seq*/0), "not-a-valid-run"); + std::vector deltas{{bh(1), s(1), false}}; + std::vector runs; + EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs), + DB::Exception); +} + +/// ==== two-cursor settlement merge (retired-in-snapshot T3, spec §2.1/§3) ==== +/// +/// The retired input is no longer a separate `prior_retired` vector — the prior generation's `kCondemned` +/// rows RIDE the source-edge run at the zero-sentinel key. These helpers build such a prior run directly +/// (via the sorted-NDJSON `SourceEdgeRunWriter`, codecs-v3 phase 5) and decode a run for assertions. + +namespace +{ + +/// A `kCondemned` sentinel record for `h` at the zero source_id, carrying the condemned incarnation. +SourceEdgeRecord condemnedRec(UInt128 h, const CondemnedRow & row) +{ + return SourceEdgeRecord{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}, + .source_id = UInt128{0}, .marker = kCondemned, + .delete_pending = row.delete_pending, .token = row.token, + .size = row.size, .condemn_round = row.condemn_round}; +} + +/// An active-edge record (`kEdgeActive`) for `h` at source `sid`. +SourceEdgeRecord edgeRec(UInt128 h, UInt128 sid) +{ + return SourceEdgeRecord{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}, + .source_id = sid, .marker = kEdgeActive}; +} + +/// head_blob / peek_head stub: present with a fixed token/size. +std::function(const BlobRef &)> headPresent(const String & tok, uint64_t size) +{ + return [tok, size](const BlobRef &) -> std::optional + { + HeadResult hr; + hr.exists = true; + hr.size = size; + hr.token = Token{.value = tok, .type = TokenType::Emulated}; + return hr; + }; +} + +/// A `CondemnedRow` mirroring the old `entry(hash, condemn_round)` fixture (token "t", size 1). +CondemnedRow condemnedRowFor(uint64_t condemn_round, const String & tok = "t", + bool delete_pending = false, uint64_t size = 1) +{ + return CondemnedRow{.delete_pending = delete_pending, + .token = Token{.value = tok, .type = TokenType::Emulated}, + .size = size, .condemn_round = condemn_round}; +} + +/// Build a source-edge run (`kSourceEdgeKeySchema128`) carrying the given `kCondemned` sentinel rows +/// and surviving edges, write it under `blobTargetRunKey(gen, attempt, shard, 0)`, and return its +/// `RunRef`. Rows are emitted in (blob_hash, source_id) order (sentinels at source_id 0 sort first +/// per blob). +RunRef writeSourceEdgeRun(InMemoryBackend & backend, const Layout & layout, + uint64_t gen, uint64_t attempt, uint64_t shard, + const std::vector> & condemned, + const std::vector> & edges = {}) +{ + std::vector recs; + for (const auto & [h, row] : condemned) + recs.push_back(condemnedRec(h, row)); + for (const auto & [h, sid] : edges) + recs.push_back(edgeRec(h, sid)); + /// The writer requires non-decreasing (ref, source_id) order (sentinels at source_id 0 sort first + /// per blob, exactly reproducing the old raw-key order). + std::stable_sort(recs.begin(), recs.end(), [](const SourceEdgeRecord & a, const SourceEdgeRecord & bb) + { + if (a.ref < bb.ref) + return true; + if (bb.ref < a.ref) + return false; + return a.source_id < bb.source_id; + }); + + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + for (const auto & rec : recs) + writer.append(rec); + writer.finish(); + out.finalize(); + + const String bytes = out.str(); + const String key = layout.blobTargetRunKey(gen, attempt, shard, 0); + backend.putIfAbsent(key, bytes); + return RunRef{.key = key, .checksum = sourceEdgeRunChecksum(bytes), .shard = shard, .generation = gen}; +} + +struct DecodedRun +{ + std::vector> condemned; /// (blob_hash, row) + std::vector zero_markers; /// blob hashes with a zero-transition marker + std::vector> edges; /// (blob_hash, source_id) +}; + +DecodedRun decodeRun(InMemoryBackend & backend, const RunRef & run) +{ + DecodedRun d; + auto r = openSourceEdgeRun(backend, run.key); + /// Every run this test helper decodes is CityHash128 (16-byte), so `.toU128()` is a + /// provably-exact round trip. + String k; + String p; + while (r.next(k, p)) + { + BlobRef bh_ref; + UInt128 sid; + SourceEdgeKeyCodec::parse(k, bh_ref, sid); // throws CORRUPTED_DATA on a malformed key (fail-closed) + const UInt128 bh = bh_ref.digest.toU128(); + EXPECT_FALSE(p.empty()); + if (p.empty()) + continue; + if (p[0] == kCondemned) + d.condemned.emplace_back(bh, decodeCondemnedRow(p)); + else if (p[0] == kZeroMarker) + d.zero_markers.push_back(bh); + else if (p[0] == kEdgeActive) + d.edges.emplace_back(bh, sid); + else + ADD_FAILURE() << "unknown run row type"; + } + return d; +} + +} + +/// Per-consumer whole-file seal-checksum RED tests (codecs-v3 phase 5, Task 6): a run whose ROWS are +/// well-formed (so `cursor.advance()` never aborts first) but whose `RunRef.checksum` disagrees with the +/// stored bytes must fail closed at each deletion-deriving consumer BEFORE any decision is produced. The +/// stored bytes are the valid run; only the seal checksum handed to the consumer is wrong. +TEST(CASBlobInDegree, FoldSealChecksumMismatchFailsClosed) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + const RunRef good = writeSourceEdgeRun(backend, layout, /*gen*/1, /*attempt*/0, /*shard*/0, + /*condemned*/{}, /*edges*/{{b(1), s(1)}}); + RunRef bad = good; + bad.checksum = good.checksum + 1; /// rows still parse; only the seal disagrees + std::vector prior{bad}; + std::vector out; + /// A delta on a DIFFERENT blob forces the two-cursor merge to stream the prior run to completion, so + /// the end-of-segment verifyAgainst fires (not a row-invariant abort). + EXPECT_THROW( + foldDeltasIntoGeneration(backend, layout, prior, /*new*/2, /*attempt*/0, /*shard*/0, + std::vector{{bh(2), s(1), false}}, out), + DB::Exception); +} + +TEST(CASBlobInDegree, ZeroInDegreeSealChecksumMismatchFailsClosed) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + const RunRef good = writeSourceEdgeRun(backend, layout, /*gen*/1, /*attempt*/0, /*shard*/0, + /*condemned*/{}, /*edges*/{{b(1), s(1)}}); + RunRef bad = good; + bad.checksum = good.checksum + 1; + std::vector runs{bad}; + EXPECT_THROW(zeroInDegree(backend, runs), DB::Exception); +} + +TEST(CASThreeCursorMerge, FloorBoundary) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Gen 1's run holds one unrelated surviving edge (b9) plus the carried kCondemned rows for A=b1 + /// (condemned round 2) and B=b2 (round 3); neither A nor B has any edge (in-degree 0 by definition). + /// current_round = 3: strictly-below graduates, at-the-current-round stays. + const RunRef gen1 = writeSourceEdgeRun(backend, layout, /*gen*/1, 0, 0, + {{b(1), condemnedRowFor(2)}, {b(2), condemnedRowFor(3)}}, {{b(9), s(1)}}); + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + /*current_round*/3, /*condemn_round*/4, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + /// Two-phase graduation: the floor-passed entry is REPUBLISHED pending (still in the list); + /// its physical delete belongs to the NEXT pass. + ASSERT_EQ(rmr.graduated.size(), 1u); + EXPECT_EQ(rmr.graduated[0].ref, bh(1)); + EXPECT_TRUE(rmr.graduated[0].delete_pending); + ASSERT_EQ(rmr.still_retired.size(), 2u); + EXPECT_EQ(rmr.still_retired[0].ref, bh(1)); + EXPECT_TRUE(rmr.still_retired[0].delete_pending); + EXPECT_EQ(rmr.still_retired[1].ref, bh(2)); + EXPECT_FALSE(rmr.still_retired[1].delete_pending); + EXPECT_EQ(rmr.still_retired[1].condemn_round, 3u); /// carried unchanged, not re-stamped + EXPECT_TRUE(rmr.spared.empty()); + EXPECT_TRUE(rmr.redelete.empty()); + + /// still_retired mirrors exactly the kCondemned rows written into the output run, in order. + const DecodedRun out = decodeRun(backend, runs2[0]); + ASSERT_EQ(out.condemned.size(), 2u); + EXPECT_EQ(out.condemned[0].first, b(1)); + EXPECT_TRUE(out.condemned[0].second.delete_pending); + EXPECT_EQ(out.condemned[1].first, b(2)); + EXPECT_FALSE(out.condemned[1].second.delete_pending); + EXPECT_TRUE(out.zero_markers.empty()); +} + +TEST(CASThreeCursorMerge, PendingRedeletesAndDrops) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// A row the PRIOR pass published as delete_pending (carried on gen 1's run): this pass hands it to + /// `redelete` (executed pre-CAS by the caller) and drops it from the output run. + const RunRef gen1 = writeSourceEdgeRun(backend, layout, /*gen*/1, 0, 0, + {{b(1), condemnedRowFor(1, "t", /*delete_pending*/true)}}); + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + /*current_round*/9, /*condemn_round*/9, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + ASSERT_EQ(rmr.redelete.size(), 1u); + EXPECT_EQ(rmr.redelete[0].ref, bh(1)); + EXPECT_TRUE(rmr.still_retired.empty()); + EXPECT_TRUE(rmr.graduated.empty()); + EXPECT_TRUE(rmr.spared.empty()); + + /// The redeleted blob leaves the run entirely (no sentinel carried, no zero marker — untouched). + const DecodedRun out = decodeRun(backend, runs2[0]); + EXPECT_TRUE(out.condemned.empty()); + EXPECT_TRUE(out.zero_markers.empty()); +} + +TEST(CASThreeCursorMerge, RecoverySpares) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// A (=b1) is retired at round 1 and would long since have graduated (current_round = 5) — but this + /// pass's delta adds an edge to it: recovery WINS over graduation, the entry is dropped as spared. + const RunRef gen1 = writeSourceEdgeRun(backend, layout, /*gen*/1, 0, 0, {{b(1), condemnedRowFor(1)}}); + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {{bh(1), s(1), false}}, runs2, + /*current_round*/5, /*condemn_round*/6, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + ASSERT_EQ(rmr.spared.size(), 1u); + EXPECT_EQ(rmr.spared[0].ref, bh(1)); + EXPECT_TRUE(rmr.graduated.empty()); + EXPECT_TRUE(rmr.still_retired.empty()); + + /// b1 recovered its edge: the output run carries the surviving edge and no sentinel for it. + const DecodedRun out = decodeRun(backend, runs2[0]); + EXPECT_TRUE(out.condemned.empty()); + ASSERT_EQ(out.edges.size(), 1u); + EXPECT_EQ(out.edges[0].first, b(1)); +} + +TEST(CASThreeCursorMerge, NewCandidateCondemned) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Gen 1: C (=b3) has one edge. Gen 2 removes it => transition to zero, not retired => + /// condemned with the head-captured token at THIS pass's condemn_round. + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, + /*current_round*/0, /*condemn_round*/7, headPresent("t9", 42), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + ASSERT_EQ(rmr.still_retired.size(), 1u); + EXPECT_EQ(rmr.still_retired[0].ref, bh(3)); + EXPECT_EQ(rmr.still_retired[0].token.value, "t9"); + EXPECT_EQ(rmr.still_retired[0].size, 42u); + EXPECT_EQ(rmr.still_retired[0].condemn_round, 7u); + EXPECT_TRUE(rmr.graduated.empty()); + EXPECT_TRUE(rmr.spared.empty()); + + /// The fresh condemn is emitted as a kCondemned row (not a zero marker) into the output run. + const DecodedRun out = decodeRun(backend, runs2[0]); + ASSERT_EQ(out.condemned.size(), 1u); + EXPECT_EQ(out.condemned[0].first, b(3)); + EXPECT_EQ(out.condemned[0].second.token.value, "t9"); + EXPECT_TRUE(out.zero_markers.empty()); +} + +TEST(CASThreeCursorMerge, AbsentBlobNotCondemned) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Same transition-to-zero as above, but the blob object is already gone at condemn time: + /// nothing to delete later, so no entry is minted — a plain zero marker is emitted instead. + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, + /*current_round*/0, /*condemn_round*/7, + [](const BlobRef &) -> std::optional { return std::nullopt; }, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + EXPECT_TRUE(rmr.still_retired.empty()); + EXPECT_TRUE(rmr.graduated.empty()); + EXPECT_TRUE(rmr.spared.empty()); + + const DecodedRun out = decodeRun(backend, runs2[0]); + EXPECT_TRUE(out.condemned.empty()); + ASSERT_EQ(out.zero_markers.size(), 1u); + EXPECT_EQ(out.zero_markers[0], b(3)); +} + +TEST(CASThreeCursorMerge, SnapshotEdgesUnperturbedByRetired) +{ + /// Retired-in-snapshot changes the byte-invariant: the retired machinery now WRITES kCondemned + /// sentinel rows into the run, so a retired-engaged run is no longer byte-identical to a plain one. + /// The preserved invariant (spec §2.1) is narrower: the retired machinery touches ONLY the sentinel + /// namespace — the surviving EDGE rows are byte-identical to a plain fold of the same deltas. + InMemoryBackend plain; + InMemoryBackend engaged; + Layout layout{"pool"}; + + std::vector r1; + foldDeltasIntoGeneration(plain, layout, /*prior_runs*/{}, 1, 0, 0, + {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(2), s(2), true}}, r1); + + /// Engaged: the SAME deltas, but the prior run carries retired rows for b1 (which the delta re-edges + /// => spared) and b5 (no edge => graduates past the floor). + const RunRef prior = writeSourceEdgeRun(engaged, layout, /*gen*/1, 0, 0, + {{b(1), condemnedRowFor(1)}, {b(5), condemnedRowFor(2)}}); + std::vector r2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(engaged, layout, /*prior_runs*/{prior}, 2, 0, 0, + {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(2), s(2), true}}, r2, + /*current_round*/9, /*condemn_round*/3, headPresent("t", 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + const DecodedRun plain_run = decodeRun(plain, r1[0]); + const DecodedRun engaged_run = decodeRun(engaged, r2[0]); + EXPECT_EQ(plain_run.edges, engaged_run.edges); /// edge rows byte-identical + EXPECT_TRUE(plain_run.condemned.empty()); + /// The engaged run carries only the retired sentinel(s) on top: b1 spared (no row), b5 graduated. + ASSERT_EQ(engaged_run.condemned.size(), 1u); + EXPECT_EQ(engaged_run.condemned[0].first, b(5)); + EXPECT_TRUE(engaged_run.condemned[0].second.delete_pending); +} + +TEST(CASTwoCursorMerge, CarriedSentinelIsNotATouch) +{ + /// Gen 1 condemns b (a real +edge/-edge net-to-zero with head_blob present) -> a kCondemned row. Gen 2 + /// has NO deltas at all: the carried row must (a) survive byte-identically, (b) emit no zero marker, + /// (c) never call peek_head (a carried sentinel is not a touch). + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Gen 1: (b,s1) added then removed => net-to-zero => fresh condemn at round 5 (token "tok", size 7). + std::vector runs1; + RetiredMergeResult rmr1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, + {{bh(2), s(1), false}, {bh(2), s(1), true}}, runs1, + /*current_round*/0, /*condemn_round*/5, headPresent("tok", 7), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr1); + ASSERT_EQ(rmr1.still_retired.size(), 1u); + { + const DecodedRun g1 = decodeRun(backend, runs1[0]); + ASSERT_EQ(g1.condemned.size(), 1u); + EXPECT_EQ(g1.condemned[0].first, b(2)); + EXPECT_TRUE(g1.zero_markers.empty()); /// a condemned blob emits kCondemned, never a zero marker + } + + /// Gen 2: empty deltas, current_round 1 (< 5 => b carries, does not graduate). peek_head must NOT fire. + size_t peek_calls = 0; + auto peek = [&](const BlobRef &) -> std::optional { ++peek_calls; return {}; }; + std::vector runs2; + RetiredMergeResult rmr2; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {}, runs2, + /*current_round*/1, /*condemn_round*/6, /*head_blob*/{}, peek, /*confirm_condemned_marker*/{}, &rmr2); + + EXPECT_EQ(peek_calls, 0u); + ASSERT_EQ(rmr2.still_retired.size(), 1u); + EXPECT_EQ(rmr2.still_retired[0].ref, bh(2)); + EXPECT_EQ(rmr2.still_retired[0].condemn_round, 5u); /// carried unchanged + EXPECT_TRUE(rmr2.graduated.empty()); + + const DecodedRun g2 = decodeRun(backend, runs2[0]); + ASSERT_EQ(g2.condemned.size(), 1u); + EXPECT_EQ(g2.condemned[0].first, b(2)); + EXPECT_EQ(g2.condemned[0].second.token.value, "tok"); + EXPECT_EQ(g2.condemned[0].second.size, 7u); + EXPECT_TRUE(g2.zero_markers.empty()); +} + +TEST(CASTwoCursorMerge, MalformedRunFailsClosed) +{ + Layout layout{"pool"}; + + /// (1) An active edge at the reserved sentinel source_id 0 -> the merge cursor fails closed. + { + InMemoryBackend backend; + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(edgeRec(1, UInt128{0})); // edge at sentinel key + writer.finish(); + out.finalize(); + const String bytes = out.str(); + const RunRef bad{.key = layout.blobTargetRunKey(1, 0, 0, 0), + .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .generation = 1}; + backend.putIfAbsent(bad.key, bytes); + + std::vector runs2; + EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), + DB::Exception); + } + + /// (2) Two sentinel rows for one blob -> duplicate sentinel -> the merge cursor fails closed. + { + InMemoryBackend backend; + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + /// Same (b,0) key twice (equal keys are allowed by the writer) — two condemned sentinels for b1. + writer.append(condemnedRec(1, condemnedRowFor(1))); + writer.append(condemnedRec(1, condemnedRowFor(2))); + writer.finish(); + out.finalize(); + const String bytes = out.str(); + const RunRef bad{.key = layout.blobTargetRunKey(1, 0, 0, 0), + .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .generation = 1}; + backend.putIfAbsent(bad.key, bytes); + + std::vector runs2; + EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), + DB::Exception); + } +} + +/// A prior run spanning several blocks folds correctly with the streaming prior cursor AND the backend +/// sees only block-bounded ranged/stream requests for it — never a whole-object get of the prior run +/// key. Byte-reproducibility of the merged output is the load-bearing canary (the merge logic is +/// unchanged; only the prior cursor's byte source moved from materialize-whole to stream). +TEST(CASBlobInDegree, FoldStreamsPriorRunBlockBounded) +{ + using DB::Cas::tests::CountingBackend; + CountingBackend backend; + /// InMemory oracle: the SAME two folds against a plain backend must yield byte-identical runs — + /// the streaming cursor changes I/O shape, not bytes. + InMemoryBackend oracle; + Layout layout{"pool"}; + + /// Gen 1 from empty prior: enough edges that the SourceEdge run spills across many 256KB blocks. + /// Each record is 4 + 32(key) + 4 + 1(payload) = 41 bytes, so ~20000 edges is ~820KB => several + /// blocks under the default block_size, exercising the multi-block streaming path in the fold. + std::vector gen1; + gen1.reserve(20000); + for (uint64_t i = 0; i < 20000; ++i) + gen1.push_back({bh(i), s(1), false}); + + std::vector runs1_c; + std::vector runs1_o; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); + foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); + + const String gen1_run_key = layout.blobTargetRunKey(1, 0, 0, 0); + const auto gen1_run = backend.get(gen1_run_key); + ASSERT_TRUE(gen1_run.has_value()); + const String gen1_run_bytes = gen1_run->bytes; + /// Sanity: the prior run really spans several blocks (else the block-bounded assertions are + /// vacuous). Blocks seal at kLegacyBlockSize (256KB); ~820KB is 3-4 blocks. + ASSERT_GT(gen1_run_bytes.size(), static_cast(kLegacyBlockSize) * 3); + + /// Reset counters and fold gen 2 with a small delta: remove one edge and add another. The prior + /// gen-1 run must be consumed via the streaming cursor (head + tail get + body getStream + per-seq + /// head probe), NEVER a whole-object get. + backend.resetCounts(); + std::vector gen2{{bh(0), s(1), true}, {bh(19999), s(2), false}}; + std::vector runs2_c; + std::vector runs2_o; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); + foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); + + /// Byte-reproducibility canary: streaming and materialized folds produce identical output bytes. + const String gen2_run_key = layout.blobTargetRunKey(2, 0, 0, 0); + const auto gen2_c = backend.get(gen2_run_key); + const auto gen2_o = oracle.get(gen2_run_key); + ASSERT_TRUE(gen2_c.has_value()); + ASSERT_TRUE(gen2_o.has_value()); + EXPECT_EQ(gen2_c->bytes, gen2_o->bytes); + ASSERT_EQ(runs2_c.size(), 1u); + ASSERT_EQ(runs2_o.size(), 1u); + EXPECT_EQ(runs2_c[0].checksum, runs2_o[0].checksum); + + /// The core assertion: no whole-object get of the prior run key — every read carried a Range or a + /// stream (the resident-memory proof at the seam). + EXPECT_EQ(backend.wholeGetCount(gen1_run_key), 0u); + /// The cursor opened the prior run's segment via the streaming reader (head + tail get + getStream). + EXPECT_GE(backend.getStreamCount(gen1_run_key), 1u); + /// Every ranged-get window on the prior run stays within one block + the footer allowance. This + /// bound is strict here because the prior run's footer fits inside the fixed tail probe (only very + /// large runs — ~13k blocks — spill the footer past the probe and add one exact-footer get; a note + /// for that regime lives in the streaming reader's open comment). + EXPECT_LE(backend.maxRangedGetLen(gen1_run_key), + static_cast(kLegacyHardCapBlockSize) + 64u * 1024u); + /// Streaming open touches the prior run's tail probe (and at most one exact-footer get); it is never + /// re-materialized whole. + EXPECT_LE(backend.getCount(gen1_run_key), 2u); +} + +/// The preview consumer `zeroInDegree` streams a multi-block run instead of materializing it whole: the +/// backend sees only block-bounded ranged/stream requests for the run key (never a whole-object get), and +/// the candidate set equals the pre-change (borrowed-mode) result. Byte-parity against an InMemory oracle +/// is the load-bearing canary — the scan logic is unchanged; only the byte source moved to the stream. +TEST(CASBlobInDegree, ZeroInDegreeStreamsBlockBounded) +{ + using DB::Cas::tests::CountingBackend; + CountingBackend backend; + InMemoryBackend oracle; + Layout layout{"pool"}; + + /// Gen 1 from empty prior: ~20000 active edges spill the SourceEdge run across several 256KB blocks. + std::vector gen1; + gen1.reserve(20000); + for (uint64_t i = 0; i < 20000; ++i) + gen1.push_back({bh(i), s(1), false}); + + std::vector runs1_c; + std::vector runs1_o; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); + foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); + + /// Gen 2 removes every edge on two of the blobs => two zero-transition markers in the gen-2 run, + /// which is itself multi-block (the surviving-edge rows still span blocks). + std::vector gen2{{bh(0), s(1), true}, {bh(19999), s(1), true}}; + std::vector runs2_c; + std::vector runs2_o; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); + foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); + + const String gen2_run_key = layout.blobTargetRunKey(2, 0, 0, 0); + const auto gen2_run = backend.get(gen2_run_key); + ASSERT_TRUE(gen2_run.has_value()); + /// Sanity: the run genuinely spans several blocks (else the block-bounded assertions are vacuous). + ASSERT_GT(gen2_run->bytes.size(), static_cast(kLegacyBlockSize) * 3); + + backend.resetCounts(); + const auto zero_c = zeroInDegree(backend, runs2_c); + const auto zero_o = zeroInDegree(oracle, runs2_o); + + /// Equivalence with the borrowed-mode (InMemory oracle) result: same candidates, in the same order. + ASSERT_EQ(zero_c.size(), zero_o.size()); + ASSERT_EQ(zero_c.size(), 2u); + for (size_t i = 0; i < zero_c.size(); ++i) + EXPECT_EQ(zero_c[i].ref, zero_o[i].ref); + + /// The core assertion: no whole-object get of the run key — every read carried a Range or a stream. + EXPECT_EQ(backend.wholeGetCount(gen2_run_key), 0u); + /// The scan opened the run via the streaming reader (head + tail get + getStream). + EXPECT_GE(backend.getStreamCount(gen2_run_key), 1u); + /// Every ranged-get window stays within one block + the footer allowance (the seam memory bound). + EXPECT_LE(backend.maxRangedGetLen(gen2_run_key), + static_cast(kLegacyHardCapBlockSize) + 64u * 1024u); + /// Streaming open touches the tail probe (and at most one exact-footer get); never re-materialized whole. + EXPECT_LE(backend.getCount(gen2_run_key), 2u); +} + +/// ==== kCondemned row codec + typed source-edge open (retired-in-snapshot T2, spec §2.1) ==== + +TEST(CASCondemnedRow, RoundTripAllTokenTypes) +{ + for (auto type : {DB::Cas::TokenType::ETag, DB::Cas::TokenType::Generation, DB::Cas::TokenType::Emulated}) + { + DB::Cas::CondemnedRow row; + row.delete_pending = (type == DB::Cas::TokenType::Generation); + row.marker_confirmed = (type == DB::Cas::TokenType::Emulated); + row.token = DB::Cas::Token{.value = "etag-abc-123", .type = type}; + row.size = 4096; + row.condemn_round = 7; + const auto bytes = DB::Cas::encodeCondemnedRow(row); + ASSERT_EQ(bytes[0], DB::Cas::kCondemned); + EXPECT_EQ(DB::Cas::decodeCondemnedRow(bytes), row); + } +} + +TEST(CASCondemnedRow, UnknownFlagBitsFailClosed) +{ + DB::Cas::CondemnedRow row; + row.token = DB::Cas::Token{.value = "t", .type = DB::Cas::TokenType::ETag}; + auto bytes = DB::Cas::encodeCondemnedRow(row); + bytes[1] = 4; // flags byte: only bits 0 (delete_pending) and 1 (marker_confirmed) are defined + EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); +} + +TEST(CASCondemnedRow, UnknownTokenTypeFailsClosed) +{ + DB::Cas::CondemnedRow row; + row.token = DB::Cas::Token{.value = "t", .type = DB::Cas::TokenType::ETag}; + auto bytes = DB::Cas::encodeCondemnedRow(row); + bytes[2] = 99; // token_type byte (offset: [0]=0x02 [1]=flags [2]=token_type) + EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); +} + +TEST(CASCondemnedRow, TruncatedPayloadFailsClosed) +{ + DB::Cas::CondemnedRow row; + row.token = DB::Cas::Token{.value = "0123456789", .type = DB::Cas::TokenType::ETag}; + auto bytes = DB::Cas::encodeCondemnedRow(row); + bytes.resize(bytes.size() - 3); // token bytes shorter than declared token_len + EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); +} + +TEST(CASSourceEdgeRun, SourceEdgeIdZeroIsReserved) +{ + /// The zero source_id is the sentinel namespace; producers fail closed on a zero hash + /// (probability 2^-128 — the check documents the reservation). + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + DB::Cas::assertValidSourceEdgeId(UInt128{0}); + }, + "source_id 0 is the reserved sentinel key"); + EXPECT_NO_THROW(DB::Cas::assertValidSourceEdgeId(UInt128{1})); +} + +/// ==== schema 3 key codec (Phase 3 T3, mixed-algo pools) ==== + +TEST(CASSourceEdgeKeySchema3, MixedWidthKeysOrderAlgoFirst) +{ + const BlobDigest d16 = BlobDigest::fromU128((UInt128(0xFFFFFFFFFFFFFFFFULL) << 64) | 0xFFULL); + BlobDigest d32{}; /// sha256 digest starting 0x00,0x01 — small bytes + d32.bytes[1] = 0x01; + const BlobRef ch{BlobHashAlgo::CityHash128, d16}; /// algo=1, digest all-FF prefix + const BlobRef sh{BlobHashAlgo::Sha256, d32}; /// algo=3, tiny digest + const String k_ch = SourceEdgeKeyCodec::key(ch, UInt128(7)); /// 33 bytes + const String k_sh = SourceEdgeKeyCodec::key(sh, UInt128(7)); /// 49 bytes + EXPECT_EQ(k_ch.size(), 33u); + EXPECT_EQ(k_sh.size(), 49u); + /// algo byte decides BEFORE any digest byte can: ch128(1) < sha256(3) even though the ch128 + /// digest bytes are all 0xFF and the sha256 digest bytes are almost all zero. + EXPECT_LT(k_ch, k_sh); + /// sentinel-first inside one blob group: + EXPECT_LT(SourceEdgeKeyCodec::key(ch, UInt128(0)), k_ch); +} + +TEST(CASSourceEdgeKeySchema3, ParseFailsClosed) +{ + BlobRef r; UInt128 sid; + String k = SourceEdgeKeyCodec::key(BlobRef{BlobHashAlgo::XXH3_128, BlobDigest::fromU128(UInt128(5))}, UInt128(9)); + SourceEdgeKeyCodec::parse(k, r, sid); + EXPECT_EQ(r.algo, BlobHashAlgo::XXH3_128); + EXPECT_EQ(r.digest.toU128(), UInt128(5)); + EXPECT_EQ(sid, UInt128(9)); + k[0] = static_cast(99); /// unknown algo byte + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, [&]{ SourceEdgeKeyCodec::parse(k, r, sid); }); + k[0] = static_cast(1); /// known algo, wrong length (33 expected, this is 33 — truncate) + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&]{ SourceEdgeKeyCodec::parse(std::string_view(k).substr(0, 20), r, sid); }); +} + +TEST(CASBlobInDegree, TwoAlgoFoldSettlesBothInOneShardRun) +{ + /// Step 3 (Phase 3 T3): extend the fold with deltas for ch128:X and sha256:Y in ONE shard run — + /// both settle (edges present, condemn on removal works per ref), mixed rows in one run, no + /// algo loop. + InMemoryBackend backend; + Layout layout{"pool"}; + + const BlobRef ch_x{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(11))}; + BlobDigest sha_y{}; + sha_y.bytes[0] = 0xAB; + const BlobRef sha_y_ref{BlobHashAlgo::Sha256, sha_y}; + + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + {{ch_x, s(1), false}, {sha_y_ref, s(1), false}}, runs1); + ASSERT_FALSE(runs1.empty()); + + EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs1, ch_x), 1); + EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs1, sha_y_ref), 1); + EXPECT_TRUE(zeroInDegree(backend, runs1).empty()); + + /// Remove both edges in gen 2: each transitions to zero independently, condemned per its own ref. + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, /*attempt*/0, 0, + {{ch_x, s(1), true}, {sha_y_ref, s(1), true}}, runs2, + /*current_round*/0, /*condemn_round*/1, headPresent("t", 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + ASSERT_EQ(rmr.still_retired.size(), 2u); + std::vector condemned_refs{rmr.still_retired[0].ref, rmr.still_retired[1].ref}; + EXPECT_NE(std::find(condemned_refs.begin(), condemned_refs.end(), ch_x), condemned_refs.end()); + EXPECT_NE(std::find(condemned_refs.begin(), condemned_refs.end(), sha_y_ref), condemned_refs.end()); + EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs2, ch_x), 0); + EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs2, sha_y_ref), 0); +} + +/// [UNMATCHED-MINUS-ONE] pin. In-degree is a SET of source edges applied last-wins per +/// (ref, ManifestId, path) key -- NOT a counter. A removal delta whose matching activation was +/// never folded (reachable today via a false-404 at the activation fold plus a dead-build skip) +/// must therefore be a per-key NO-OP: it marks an already-absent edge absent and cannot strip a +/// sibling manifest's edge for the SAME blob. The whole "that interleaving is harmless" argument in +/// the publish-confirm design rests on this; if the model ever regresses to counter arithmetic this +/// test goes red and premature deletion becomes reachable again. +TEST(CASBlobInDegree, UnmatchedRemovalIsAPerKeyNoOpAndSparesSiblingEdges) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Generation 1: blob b1 is referenced by TWO distinct sources (two manifests). + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, + {{bh(1), s(1), false}, {bh(1), s(2), false}}, runs1); + + /// Generation 2: fold a removal for a THIRD source that never had an activation folded. + std::vector runs2; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, + {{bh(1), s(99), true}}, runs2); + + /// Both original edges survive: the unmatched removal touched only its own (absent) key. + const DecodedRun out = decodeRun(backend, runs2[0]); + ASSERT_EQ(out.edges.size(), 2u) << "an unmatched removal must not strip sibling edges"; + /// And the blob is NOT a deletion candidate. + const auto zero = zeroInDegree(backend, runs2); + EXPECT_TRUE(zero.empty()) << "b1 still has two live source edges"; +} + +/// The silence in `UnmatchedRemovalIsAPerKeyNoOpAndSparesSiblingEdges` above is exactly what let a whole +/// class of GC defects survive months of soak runs undetected — the fold's per-key no-op left no trace. +/// This test pins the COUNTING surface added on top: `RetiredMergeResult::unmatched_removes` / +/// `unmatched_remove_example` must report the unmatched remove precisely (one hit, naming the right blob +/// and source id), while the byte-level no-op behaviour (asserted above) is unchanged. +TEST(CASBlobInDegree, UnmatchedRemovalIsCountedWithAnExample) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + /// Generation 1: blob b1 is referenced by TWO distinct sources (two manifests), same fixture as the + /// no-op test above. + std::vector runs1; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, + {{bh(1), s(1), false}, {bh(1), s(2), false}}, runs1); + + /// Generation 2: fold a removal for a THIRD source that never had an activation folded. + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, + {{bh(1), s(99), true}}, runs2, + /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + + /// The run is byte-identical to the no-op test's outcome for the blob's OTHER edges: both survive. + const DecodedRun out = decodeRun(backend, runs2[0]); + ASSERT_EQ(out.edges.size(), 2u) << "the counting surface must not perturb the no-op fold outcome"; + EXPECT_EQ(out.edges[0].first, b(1)); + EXPECT_EQ(out.edges[1].first, b(1)); + std::vector surviving_sources{out.edges[0].second, out.edges[1].second}; + EXPECT_NE(std::find(surviving_sources.begin(), surviving_sources.end(), s(1)), surviving_sources.end()); + EXPECT_NE(std::find(surviving_sources.begin(), surviving_sources.end(), s(2)), surviving_sources.end()); + + /// The counting surface reports exactly the one unmatched remove, naming the right blob and source id. + EXPECT_EQ(rmr.unmatched_removes, 1u); + ASSERT_TRUE(rmr.unmatched_remove_example.has_value()); + EXPECT_EQ(rmr.unmatched_remove_example->ref, bh(1)); + EXPECT_EQ(rmr.unmatched_remove_example->source_id, s(99)); +} + +namespace +{ +/// N distinct condemned rows for blobs b(1)..b(n), same shape `condemnedRowFor` produces, varying +/// only the token so distinct rows are trivially distinguishable in a failure message. +std::vector> condemnedCohort(uint64_t n, uint64_t condemn_round, bool delete_pending) +{ + std::vector> rows; + for (uint64_t i = 1; i <= n; ++i) + rows.push_back({b(i), condemnedRowFor(condemn_round, "t" + std::to_string(i), delete_pending)}); + return rows; +} +} + +/// The redelete cohort is capped at `GcRoundWorkBudget::max_redeletes` per call. Excess +/// entries stay in `still_retired`, still `delete_pending`, to be redeleted by a later round — the +/// durable pipeline never loses one to the cap. +TEST(CASThreeCursorMerge, RedeleteBudgetCapsCohortAndCarriesExcess) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + const RunRef gen1 = writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, 1, /*delete_pending*/true)); + + GcRoundWorkBudget budget; + budget.max_redeletes = 3; + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + /*current_round*/9, /*condemn_round*/9, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, + &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, + /*source_retirements*/{}, &budget); + + EXPECT_EQ(rmr.redelete.size(), 3u); + EXPECT_EQ(budget.redeletes_used, 3u); + EXPECT_TRUE(rmr.graduated.empty()); + EXPECT_TRUE(rmr.spared.empty()); + ASSERT_EQ(rmr.still_retired.size(), 7u); + for (const RetiredEntry & e : rmr.still_retired) + EXPECT_TRUE(e.delete_pending) << "carried entries stay delete_pending, unexecuted this round"; +} + +/// Mirror test for the graduation cap: entries past `max_graduations` carry unchanged (still +/// condemned, NOT yet delete_pending) rather than being force-graduated; the floor re-evaluates them +/// next round. +TEST(CASThreeCursorMerge, GraduationBudgetCapsCohortAndCarriesExcess) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + const RunRef gen1 = writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, /*condemn_round*/1, /*delete_pending*/false)); + + GcRoundWorkBudget budget; + budget.max_graduations = 3; + + std::vector runs2; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + /*current_round*/5, /*condemn_round*/6, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, + &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, + /*source_retirements*/{}, &budget); + + EXPECT_EQ(rmr.graduated.size(), 3u); + EXPECT_EQ(budget.graduations_used, 3u); + ASSERT_EQ(rmr.still_retired.size(), 10u); + size_t pending_count = 0; + size_t carried_count = 0; + for (const RetiredEntry & e : rmr.still_retired) + e.delete_pending ? ++pending_count : ++carried_count; + EXPECT_EQ(pending_count, 3u) << "only the graduated 3 are republished delete_pending"; + EXPECT_EQ(carried_count, 7u) << "the rest carry unchanged, still eligible next round"; +} + +/// The mandatory convergence proof: a cohort well past the per-round cap fully drains over +/// ceil(N / cap) rounds, feeding each round's output run back as the next round's prior — the exact +/// shape a real GC round repeats every pass. +TEST(CASThreeCursorMerge, RedeleteBudgetDrainsCohortToFixpointOverRounds) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + std::vector priors{writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, 1, /*delete_pending*/true))}; + + uint64_t total_redeleted = 0; + uint64_t rounds = 0; + while (rounds < 10) + { + GcRoundWorkBudget budget; + budget.max_redeletes = 3; + std::vector out_runs; + RetiredMergeResult rmr; + foldDeltasIntoGeneration(backend, layout, priors, 2 + rounds, 0, 0, {}, out_runs, + /*current_round*/100, /*condemn_round*/100, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, + &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, + /*source_retirements*/{}, &budget); + total_redeleted += rmr.redelete.size(); + ++rounds; + if (rmr.still_retired.empty()) + break; + ASSERT_FALSE(out_runs.empty()); + priors = out_runs; + } + EXPECT_EQ(total_redeleted, 10u) << "no entry lost to the cap across the whole drain"; + EXPECT_EQ(rounds, 4u) << "ceil(10 / 3) rounds to fully drain"; +} diff --git a/src/Disks/tests/gtest_cas_blob_meta.cpp b/src/Disks/tests/gtest_cas_blob_meta.cpp new file mode 100644 index 000000000000..a0a29bf2e8f4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_meta.cpp @@ -0,0 +1,186 @@ +#include + +#include +#include +#include +#include "cas_test_helpers.h" + +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +/// Codec tests (round-trip both states, fail-closed decode) moved to gtest_cas_blob_meta_format.cpp +/// with the v3 text cutover; the lifecycle + inspect tests below stay — they exercise the Core ops +/// and CasInspect against the stable encode/decode signatures and must pass unchanged. + +TEST(CASBlobMeta, PutIfAbsentThenCasTransitions) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-a"))}; + const BlobMeta clean{.state = MetaState::Clean, .size = 10}; + + const CasOverwriteResult created = putMetaIfAbsent(*store, ref, clean); + EXPECT_EQ(created.outcome, CasOverwriteOutcome::Committed); + + const CasOverwriteResult dup = putMetaIfAbsent(*store, ref, clean); + EXPECT_EQ(dup.outcome, CasOverwriteOutcome::Committed); /// exact-byte resolution adopts the existing marker + + const auto lm = loadMeta(*backend, store->layout(), ref); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean); + + const CasOverwriteResult condemned = casMeta(*store, ref, lm->etag, + BlobMeta{.state = MetaState::Condemned, .condemn_round = 5, .size = 10}); + EXPECT_EQ(condemned.outcome, CasOverwriteOutcome::Committed); + + const CasOverwriteResult stale = casMeta(*store, ref, lm->etag, /// stale token loses + BlobMeta{.state = MetaState::Clean}); + EXPECT_EQ(stale.outcome, CasOverwriteOutcome::Conflict); +} + +TEST(CASBlobMeta, DeleteMetaExactMatchesEtag) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-b"))}; + putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Condemned}); + const auto lm = loadMeta(*backend, store->layout(), ref); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(deleteMetaExact(*backend, store->layout(), ref, lm->etag).kind, DeleteOutcome::Kind::Deleted); + EXPECT_FALSE(loadMeta(*backend, store->layout(), ref).has_value()); +} + +/// Phase 3 T3 (mixed-algo pools, was CAS pluggable-blob-hash Phase 2 Task 5 crux Test 2): the `.meta` +/// API round-trips a 32-byte (`sha256`-width) `BlobRef` key — the meta object lands under a 64-hex +/// key, exercising the SAME `putMetaIfAbsent`/`loadMeta`/`casMeta`/`deleteMetaExact` surface PartWriteTxn/Gc +/// use, just at a wider algo. Writes use the `Pool`'s controller; reads and exact deletion retain their +/// direct `Backend`/`Layout` surface. +TEST(CASBlobMeta, PutLoadCasDeleteRoundTripAtWidth32) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + /// A distinguishable 32-byte digest (not merely a 16-byte value zero-tailed): every byte set. + BlobDigest h; + for (size_t i = 0; i < h.bytes.size(); ++i) + h.bytes[i] = static_cast(i + 1); + const BlobRef ref{BlobHashAlgo::Sha256, h}; + const String hex = codecFor(BlobHashAlgo::Sha256).toHex(h); + EXPECT_EQ(hex.size(), 64u) << "a 32-byte digest renders 64 hex chars"; + + const CasOverwriteResult created = putMetaIfAbsent(*store, ref, + BlobMeta{.state = MetaState::Clean, .size = 555}); + ASSERT_EQ(created.outcome, CasOverwriteOutcome::Committed); + EXPECT_TRUE(backend->head(layout.blobMetaKey(ref)).exists) + << "the meta object must land under the 64-hex key, not a truncated 32-hex one"; + + const auto lm = loadMeta(*backend, layout, ref); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean); + EXPECT_EQ(lm->meta.size, 555u); + + const CasOverwriteResult condemned = casMeta(*store, ref, lm->etag, + BlobMeta{.state = MetaState::Condemned, .condemn_round = 7, .size = 555}); + ASSERT_EQ(condemned.outcome, CasOverwriteOutcome::Committed); + const auto lm2 = loadMeta(*backend, layout, ref); + ASSERT_TRUE(lm2.has_value()); + EXPECT_EQ(lm2->meta.state, MetaState::Condemned); + + EXPECT_EQ(deleteMetaExact(*backend, layout, ref, lm2->etag).kind, DeleteOutcome::Kind::Deleted); + EXPECT_FALSE(loadMeta(*backend, layout, ref).has_value()); +} + +namespace +{ + +class ControlledMetaWriteFaultBackend : public InMemoryBackend +{ +public: + bool throw_next_create = false; + bool throw_next_overwrite = false; + uint64_t create_attempts = 0; + uint64_t overwrite_attempts = 0; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + ++create_attempts; + if (throw_next_create) + { + throw_next_create = false; + throw Poco::TimeoutException("scripted meta create ambiguity"); + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + ++overwrite_attempts; + if (throw_next_overwrite) + { + throw_next_overwrite = false; + throw Poco::TimeoutException("scripted meta overwrite ambiguity"); + } + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } +}; + +} + +TEST(CASBlobMeta, WritesUsePoolRequestController) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + store->setCasRetrySleepForTest([](uint64_t) {}); + backend->create_attempts = 0; + backend->overwrite_attempts = 0; + + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-controlled"))}; + backend->throw_next_create = true; + EXPECT_EQ( + putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Clean, .size = 10}).outcome, + CasOverwriteOutcome::Committed); + EXPECT_EQ(backend->create_attempts, 2u); + + const auto clean = loadMeta(*backend, store->layout(), ref); + ASSERT_TRUE(clean.has_value()); + backend->throw_next_overwrite = true; + EXPECT_EQ( + casMeta(*store, ref, clean->etag, BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 10}).outcome, + CasOverwriteOutcome::Committed); + EXPECT_EQ(backend->overwrite_attempts, 2u); +} + +/// `cas-inspect` dispatch (CasInspect.cpp): a `.meta` key must decode as a BlobMeta, NOT fall through +/// to the `blobs/` envelope branch (the `.meta` key shares the `blobsPrefix()` prefix with a body key). +TEST(CASBlobMeta, InspectRendersCondemnedMeta) +{ + const Layout layout("p"); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-inspect"))}; + const String key = layout.blobMetaKey(ref); + const BlobMeta m{.version = 1, .state = MetaState::Condemned, .condemn_round = 9, .size = 123}; + + const String json = caInspectToJson(layout, key, encodeBlobMeta(m)); + EXPECT_NE(json.find("\"object\":\"blob_meta\""), String::npos); + EXPECT_NE(json.find("\"condemned\""), String::npos); + EXPECT_NE(json.find("\"condemn_round\":9"), String::npos); + EXPECT_NE(json.find("\"size\":123"), String::npos); +} + +TEST(CASBlobMeta, InspectRendersCleanMeta) +{ + const Layout layout("p"); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-inspect-clean"))}; + const String key = layout.blobMetaKey(ref); + const BlobMeta m{.version = 1, .state = MetaState::Clean, .condemn_round = 0, .size = 7}; + + const String json = caInspectToJson(layout, key, encodeBlobMeta(m)); + EXPECT_NE(json.find("\"clean\""), String::npos); +} diff --git a/src/Disks/tests/gtest_cas_blob_meta_format.cpp b/src/Disks/tests/gtest_cas_blob_meta_format.cpp new file mode 100644 index 000000000000..bb055d2cef71 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_meta_format.cpp @@ -0,0 +1,49 @@ +#include "cas_format_test_battery.h" +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; } + +TEST(CASFormatBattery, BlobMeta) +{ + BlobMeta m; + m.state = MetaState::Clean; + m.condemn_round = 0; + m.size = 12345; + runFormatBattery(FormatBatteryCase{ + .id = FormatId::BlobMeta, + .encode = [&] { return sealObject(FormatId::BlobMeta, encodeBlobMeta(m)); }, + .decode = [](std::string_view s) { decodeBlobMeta(std::string(openObject(FormatId::BlobMeta, s))); }, + .golden = "{\"type\":\"cas_blob_meta\",\"v\":10}\n" + "{\"st\":\"clean\",\"cr\":\"0\",\"sz\":\"12345\"}\n"}); +} + +TEST(CASBlobMetaFormat, CondemnedRoundTripAllFields) +{ + BlobMeta m; + m.state = MetaState::Condemned; + m.condemn_round = 7; + m.size = 4096; + const BlobMeta back = decodeBlobMeta(encodeBlobMeta(m)); + EXPECT_EQ(back.state, MetaState::Condemned); + EXPECT_EQ(back.condemn_round, 7u); + EXPECT_EQ(back.size, 4096u); + EXPECT_EQ(encodeBlobMeta(m), + "{\"type\":\"cas_blob_meta\",\"v\":10}\n{\"st\":\"condemned\",\"cr\":\"7\",\"sz\":\"4096\"}\n"); +} + +TEST(CASBlobMetaFormat, FailsClosedOnUnknownStateAndTruncation) +{ + /// Unknown state word -> CORRUPTED_DATA (mirrors the old `state > Condemned` reject). + /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes + /// the header gate, which is the point — the BODY is what has to fail here. + const String bad_state = "{\"type\":\"cas_blob_meta\",\"v\":3}\n{\"st\":\"zombie\",\"cr\":\"0\",\"sz\":\"0\"}\n"; + EXPECT_THROW(decodeBlobMeta(bad_state), DB::Exception); + /// Missing state key -> CORRUPTED_DATA. + const String no_state = "{\"type\":\"cas_blob_meta\",\"v\":3}\n{\"cr\":\"0\",\"sz\":\"0\"}\n"; + EXPECT_THROW(decodeBlobMeta(no_state), DB::Exception); + /// Truncated (header only) -> CORRUPTED_DATA. + EXPECT_THROW(decodeBlobMeta("{\"type\":\"cas_blob_meta\",\"v\":3}\n"), DB::Exception); +} diff --git a/src/Disks/tests/gtest_cas_blob_ref.cpp b/src/Disks/tests/gtest_cas_blob_ref.cpp new file mode 100644 index 000000000000..a75b5b968502 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_ref.cpp @@ -0,0 +1,34 @@ +#include +#include +#include + +using namespace DB::Cas; + +TEST(CASBlobRef, SameDigestDifferentAlgoAreDistinct) +{ + const BlobDigest d = BlobDigest::fromU128(UInt128(0xDEADBEEF)); + const BlobRef a{BlobHashAlgo::CityHash128, d}; + const BlobRef b{BlobHashAlgo::XXH3_128, d}; + EXPECT_NE(a, b); + EXPECT_LT(a, b); /// algo=1 < algo=2 + std::unordered_set s{a, b}; + EXPECT_EQ(s.size(), 2u); +} + +TEST(CASBlobRef, OrderIsAlgoThenDigest) +{ + const BlobRef small_algo_big_digest{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(0) - 1)}; + const BlobRef big_algo_small_digest{BlobHashAlgo::Sha256, BlobDigest::fromU128(UInt128(1))}; + EXPECT_LT(small_algo_big_digest, big_algo_small_digest); /// algo decides first +} + +TEST(CASBlobRef, HexAndIdRenderAtAlgoWidth) +{ + BlobRef r16{BlobHashAlgo::XXH3_128, BlobDigest::fromU128(UInt128(0xAB))}; + EXPECT_EQ(blobHexOf(r16).size(), 32u); + EXPECT_EQ(blobIdOf(r16).substr(0, 5), "xxh3:"); + BlobRef r32{BlobHashAlgo::Sha256, {}}; + for (size_t i = 0; i < 32; ++i) r32.digest.bytes[i] = static_cast(i); + EXPECT_EQ(blobHexOf(r32).size(), 64u); + EXPECT_EQ(blobIdOf(r32).substr(0, 7), "sha256:"); +} diff --git a/src/Disks/tests/gtest_cas_blob_upload_pool.cpp b/src/Disks/tests/gtest_cas_blob_upload_pool.cpp new file mode 100644 index 000000000000..3e451810f744 --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_upload_pool.cpp @@ -0,0 +1,144 @@ +#include +#include +#include +#include + +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int LOGICAL_ERROR; +} + +namespace +{ + +/// Mirrors `gtest_cas_part_manifest_format.cpp`'s inlined assertion helper rather than pulling in +/// `Disks/tests/cas_test_helpers.h` for one tiny check. +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} + +/// The pattern stage-1 T5's fan-out fixtures reuse: lazily bring the server-wide pool up on first +/// need. Deliberately NOT torn down between calls -- `initializeBlobUploadPool` is once-only for +/// the lifetime of the binary via this helper, matching how the real server wires it once at +/// startup. Tests that need to exercise the raw init/shutdown lifecycle contract itself (this file) +/// call `initializeBlobUploadPool`/`shutdownBlobUploadPool` directly instead of through this helper. +void ensureBlobUploadPoolForTest(size_t size) +{ + static std::once_flag once; + std::call_once(once, [size] { initializeBlobUploadPool(size); }); +} + +} + +/// `blobUploadPool()` on an uninitialized pool throws `LOGICAL_ERROR`, which aborts the whole +/// process in debug/sanitizer builds instead of behaving like a catchable exception (see +/// `handle_error_code` in `Common/Exception.cpp`) -- `CASBlobUploadPoolDeathTest` below proves the +/// abort positively in those builds instead, following `gtest_cas_gc_state_format.cpp`'s pattern. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASBlobUploadPool, GetterThrowsBeforeInit) +{ + shutdownBlobUploadPool(); + EXPECT_FALSE(blobUploadPoolInitializedForTest()); + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [] { blobUploadPool(); }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASBlobUploadPoolDeathTest, GetterAbortsBeforeInit) +{ + shutdownBlobUploadPool(); + ASSERT_FALSE(blobUploadPoolInitializedForTest()); + EXPECT_DEATH({ (void)blobUploadPool(); }, ""); +} +#endif + +TEST(CASBlobUploadPool, InitZeroRejected) +{ + shutdownBlobUploadPool(); + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [] { initializeBlobUploadPool(0); }); + /// A rejected init must not leave the pool half-initialized. + EXPECT_FALSE(blobUploadPoolInitializedForTest()); + shutdownBlobUploadPool(); +} + +TEST(CASBlobUploadPool, InitThenGetWorks) +{ + shutdownBlobUploadPool(); + initializeBlobUploadPool(4); + EXPECT_TRUE(blobUploadPoolInitializedForTest()); + + std::atomic ran{0}; + blobUploadPool().scheduleOrThrowOnError([&ran] { ++ran; }); + blobUploadPool().wait(); + EXPECT_EQ(ran.load(), 1); + + shutdownBlobUploadPool(); +} + +/// Same debug/sanitizer-abort caveat as `GetterThrowsBeforeInit` above: the second +/// `initializeBlobUploadPool` call throws `LOGICAL_ERROR`, which aborts under +/// `DEBUG_OR_SANITIZER_BUILD`. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASBlobUploadPool, DoubleInitThrows) +{ + shutdownBlobUploadPool(); + initializeBlobUploadPool(2); + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [] { initializeBlobUploadPool(2); }); + shutdownBlobUploadPool(); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASBlobUploadPoolDeathTest, DoubleInitAborts) +{ + shutdownBlobUploadPool(); + initializeBlobUploadPool(2); + EXPECT_DEATH({ (void)initializeBlobUploadPool(2); }, ""); + shutdownBlobUploadPool(); +} +#endif + +TEST(CASBlobUploadPool, ShutdownIdempotent) +{ + /// Idempotent even when never initialized. + shutdownBlobUploadPool(); + shutdownBlobUploadPool(); + EXPECT_FALSE(blobUploadPoolInitializedForTest()); + + initializeBlobUploadPool(3); + shutdownBlobUploadPool(); + /// Idempotent after a real init + shutdown too. + shutdownBlobUploadPool(); + EXPECT_FALSE(blobUploadPoolInitializedForTest()); +} + +TEST(CASBlobUploadPool, EnsureForTestHelperLazilyInitializes) +{ + shutdownBlobUploadPool(); + EXPECT_FALSE(blobUploadPoolInitializedForTest()); + + ensureBlobUploadPoolForTest(4); + EXPECT_TRUE(blobUploadPoolInitializedForTest()); + + /// Idempotent: a pool already up must not throw on a repeated call. + ensureBlobUploadPoolForTest(4); + EXPECT_TRUE(blobUploadPoolInitializedForTest()); + + shutdownBlobUploadPool(); +} diff --git a/src/Disks/tests/gtest_cas_blob_upload_pool_env.cpp b/src/Disks/tests/gtest_cas_blob_upload_pool_env.cpp new file mode 100644 index 000000000000..6f1ed8c0171e --- /dev/null +++ b/src/Disks/tests/gtest_cas_blob_upload_pool_env.cpp @@ -0,0 +1,48 @@ +#include +#include + +/// Stage-1 §1: `ContentAddressedTransaction::uploadPendingBlobs` fans out on the server-wide blob +/// upload pool, whose getter is fail-loud (throws `LOGICAL_ERROR` -- an ABORT under a sanitizer build -- +/// if the pool was never initialized). Any CA test that commits a transaction with a pending blob would +/// therefore abort the whole `unit_tests_dbms` process if the pool happened to be down. +/// +/// This listener brings the pool up before EVERY test, so the pool is always initialized at the start of +/// a test body regardless of link/run order. It is deliberately a before-each hook (not a one-shot +/// `Environment::SetUp`): the raw-lifecycle suite in `gtest_cas_blob_upload_pool.cpp` shuts the pool +/// down inside its own bodies, and those tests explicitly re-establish whatever pool state they assert +/// on as their FIRST action, so re-ensuring the pool before them is harmless. +/// +/// It ALSO shuts the pool down once, in `OnTestProgramEnd` (the last gtest event, fired from inside +/// `RUN_ALL_TESTS` before `gtest_main`'s exit `SCOPE_EXIT`). This is mandatory, not cosmetic: the blob +/// upload pool is a `ThreadFromGlobalPool`-backed `ThreadPool`, so its idle workers occupy GlobalThreadPool +/// std::threads. `gtest_main` shuts the GlobalThreadPool down at process exit by JOINING those std::threads +/// -- but a lingering blob-pool worker never returns until the blob pool itself is destroyed, so leaving +/// the pool up at exit deadlocks the whole binary (main joins a std::thread that is running a blob-pool +/// worker that waits for the blob pool to shut down). Draining it here, before `RUN_ALL_TESTS` returns, +/// releases those std::threads first. +namespace +{ + +class BlobUploadPoolEnsuringListener : public ::testing::EmptyTestEventListener +{ +public: + void OnTestStart(const ::testing::TestInfo &) override + { + DB::Cas::tests::ensureBlobUploadPoolForTest(); + } + + void OnTestProgramEnd(const ::testing::UnitTest &) override + { + /// Release the pool's GlobalThreadPool-backed workers BEFORE `gtest_main` joins the GlobalThreadPool + /// at exit (see the class comment) -- otherwise the binary deadlocks at exit. Idempotent. + DB::Cas::shutdownBlobUploadPool(); + } +}; + +const bool registered_blob_upload_pool_listener = [] +{ + ::testing::UnitTest::GetInstance()->listeners().Append(new BlobUploadPoolEnsuringListener); + return true; +}(); + +} diff --git a/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp b/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp new file mode 100644 index 000000000000..cfb150401cf2 --- /dev/null +++ b/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp @@ -0,0 +1,442 @@ +#include + +#include +#include +#include +#include +#include "cas_test_helpers.h" +#include + +#include +#include +#include +#include + +/// Task 7 (spec §2 "Startup [C4], ordered vs the capability probe [D2]"): the writable `Pool::open` +/// bootstrap sequence is (0) a ZERO-WRITE residual check FIRST — before any probe write — that ignores +/// structurally-valid `_probe/` debris; (1) only then the mutating `_probe/` capability battery; (2) then +/// `PoolMeta::createOrValidate`, which may mint a missing `_pool_meta` only over a genuinely empty prefix. +/// A missing `_pool_meta` over residual (non-`_probe`) data fails startup loud with ZERO writes — closing +/// the "restart poisons a partially-erased pool" hole. These are black-box tests over `Pool::open`, +/// asserting behavior AND ordering via an op-recording backend (they fail on the pre-Task-7 open, which +/// bootstraps a fresh identity unconditionally and performs no residual LIST before the battery). + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +} + +using namespace DB::Cas; + +namespace +{ + +const String kPrefix = "p"; +const String kSrid = "test"; +const String kPoolMetaKey = "p/_pool_meta"; +/// A well-formed per-mount probe uid: exactly 32 lowercase hex chars (`u128ToHex`'s shape). +const String kProbeUid = "0123456789abcdef0123456789abcdef"; +const String kProbeUid2 = "fedcba9876543210fedcba9876543210"; + +/// Records the ORDER of backend operations so a test can assert that the residual LIST precedes the first +/// write, and that a fail path performs zero writes. Delegates every operation to `InMemoryBackend` +/// unchanged; `Pool::open` wraps this in its `InstrumentedBackend`, which forwards every op here. +class RecordingBackend final : public InMemoryBackend +{ +public: + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + enum class Op : uint8_t { List, PutIfAbsent, PutOverwrite, CasPut, Delete }; + struct Entry + { + Op op; + String key; /// the LIST prefix, or the written key + }; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + record(Op::List, prefix); + return InMemoryBackend::list(prefix, cursor, limit); + } + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + record(Op::PutIfAbsent, key); + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + record(Op::PutOverwrite, key); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override + { + record(Op::CasPut, key); + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + record(Op::Delete, key); + return InMemoryBackend::deleteExact(key, token); + } + /// The bootstrap path (battery + createOrValidate + mount protocol) issues only whole-String writes, + /// never a streaming create, so recording the four write ops above captures every write `open` can do. + + static bool isWrite(Op op) + { + return op == Op::PutIfAbsent || op == Op::PutOverwrite || op == Op::CasPut || op == Op::Delete; + } + + void clearLog() + { + std::lock_guard l(mutex_); + log_.clear(); + } + std::vector snapshot() const + { + std::lock_guard l(mutex_); + return log_; + } + size_t writeCount() const + { + std::lock_guard l(mutex_); + size_t n = 0; + for (const auto & e : log_) + if (isWrite(e.op)) + ++n; + return n; + } + +private: + void record(Op op, const String & key) + { + std::lock_guard l(mutex_); + log_.push_back({op, key}); + } + mutable std::mutex mutex_; + std::vector log_; +}; + +/// Models a stale LIST result for `cas/ref_catalog`: the object was listed, then disappeared before +/// the exact validation GET. The bootstrap must treat this as residual, never as a new-pool proof. +class CatalogMissingAfterListBackend final : public InMemoryBackend +{ +public: + using Backend::get; + + std::optional get(const String & key, Range range) override + { + if (key == Layout{kPrefix}.refCatalogKey()) + return std::nullopt; + return InMemoryBackend::get(key, range); + } +}; + +PoolConfig makeConfig() +{ + PoolConfig cfg; + cfg.pool_prefix = kPrefix; + cfg.server_root_id = kSrid; + cfg.wait_sleep_fn = [](uint64_t) {}; /// never block a synchronous test on an open/teardown wait + return cfg; +} + +template +void expectThrowsCodeContaining(int expected_code, const String & needle, F && fn); + +void expectCatalogResidueRefusesWithoutPoolMeta(const String & bytes, const String & extra_key = {}) +{ + auto backend = std::make_shared(); + const Layout layout{kPrefix}; + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), bytes).outcome, PutOutcome::Done); + if (!extra_key.empty()) + ASSERT_EQ(backend->putIfAbsent(extra_key, "residual").outcome, PutOutcome::Done); + backend->clearLog(); + + try + { + Pool::open(backend, makeConfig()); + FAIL() << "expected residual catalog bootstrap refusal"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::INVALID_STATE); + } + EXPECT_EQ(backend->writeCount(), 0u); + EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); +} + +/// Index of the first op matching `pred`, if any. +template +std::optional firstIndex(const std::vector & log, Pred && pred) +{ + for (size_t i = 0; i < log.size(); ++i) + if (pred(log[i])) + return i; + return std::nullopt; +} + +/// Assert `fn` throws a DB::Exception with `expected_code` AND a message containing `needle`. +template +void expectThrowsCodeContaining(int expected_code, const String & needle, F && fn) +{ + try + { + fn(); + FAIL() << "expected a DB::Exception, none thrown"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + EXPECT_NE(e.message().find(needle), String::npos) + << "message did not contain '" << needle << "': " << e.message(); + } +} + +} + +/// (a) Empty prefix → open succeeds, `_pool_meta` is created, AND the op-log proves the residual LIST of +/// the pool prefix happened BEFORE any write (the ordering [D2] mandates: no probe write may precede the +/// emptiness proof). +TEST(CASBootstrapOrdering, EmptyPrefixOpensAndListsBeforeAnyWrite) +{ + auto backend = std::make_shared(); + backend->clearLog(); + + PoolPtr store = Pool::open(backend, makeConfig()); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + EXPECT_TRUE(backend->get(kPoolMetaKey).has_value()) << "_pool_meta must be created on a fresh empty prefix"; + + const auto log = backend->snapshot(); + const auto residual_list = firstIndex(log, [](const RecordingBackend::Entry & e) + { return e.op == RecordingBackend::Op::List && e.key == kPrefix + "/"; }); + const auto first_write = firstIndex(log, [](const RecordingBackend::Entry & e) + { return RecordingBackend::isWrite(e.op); }); + + ASSERT_TRUE(residual_list.has_value()) << "the zero-write residual LIST of '" << kPrefix << "/' must run"; + ASSERT_TRUE(first_write.has_value()) << "a fresh open must eventually write (battery/meta/mount)"; + EXPECT_LT(*residual_list, *first_write) << "the residual LIST must precede every write"; +} + +/// The residue an incomplete erase would have left behind: a real ref-log object key, built through +/// `Layout` so it carries the life segment every ref key has. The residual check is LIST-based and +/// never parses it, but seeding a shape this build cannot write would make the comment below a lie. +namespace +{ +String residualRefLogKey() +{ + return Layout{"p"}.refLogKey(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"test%2Fabcd"}), RefTxnId{1, 1}); +} +} + +/// (b) A prefix holding `cas/ns/stream/…` residue but NO `_pool_meta` → open fails typed (INVALID_STATE), +/// and ZERO writes hit the backend (the mutating battery must NOT have run — the residual check throws +/// first). +TEST(CASBootstrapOrdering, ResidualWithoutMetaFailsTypedWithZeroWrites) +{ + auto backend = std::make_shared(); + /// Seed residue an incomplete erase would have left behind (a ref-log object), with no `_pool_meta`. + ASSERT_EQ(backend->putIfAbsent(residualRefLogKey(), "x").outcome, + PutOutcome::Done); + backend->clearLog(); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", + [&] { Pool::open(backend, makeConfig()); }); + + EXPECT_EQ(backend->writeCount(), 0u) << "the fail path must perform zero writes (battery never ran)"; + EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()) << "a fresh _pool_meta must NOT have been minted"; +} + +/// (c) A prefix containing ONLY stale, structurally-valid `_probe//…` debris (a crash-mid-battery +/// leftover) → treated as empty → open succeeds and bootstraps a fresh pool. The debris-skip is what makes +/// a normal restart-after-crash recover instead of wedging. +TEST(CASBootstrapOrdering, StaleProbeDebrisOnlyIsTreatedAsEmpty) +{ + auto backend = std::make_shared(); + ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/cas", "cas-s1").outcome, PutOutcome::Done); + backend->clearLog(); + + PoolPtr store; + ASSERT_NO_THROW(store = Pool::open(backend, makeConfig())); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); + EXPECT_TRUE(backend->get(kPoolMetaKey).has_value()) << "_pool_meta must be created over a probe-only prefix"; +} + +TEST(CASBootstrapOrdering, CanonicalEmptyCatalogOnlyIsTheSoleRetryablePreMetaResidue) +{ + auto backend = std::make_shared(); + const Layout layout{kPrefix}; + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(kPrefix + "/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); + backend->clearLog(); + + PoolPtr store; + ASSERT_NO_THROW(store = Pool::open(backend, makeConfig())); + EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); +} + +TEST(CASBootstrapOrdering, MalformedCatalogOnlyResidueRefusesWithoutPoolMeta) +{ + expectCatalogResidueRefusesWithoutPoolMeta("not a catalog"); +} + +TEST(CASBootstrapOrdering, NoncanonicalCatalogOnlyResidueRefusesWithoutPoolMeta) +{ + String noncanonical = encodeRefCatalog(RefCatalog{}); + noncanonical.insert(noncanonical.find('\n') - 1, ",\"noncanonical\":0"); + ASSERT_TRUE(decodeRefCatalog(noncanonical).entries.empty()) << "fixture must be decodable but noncanonical"; + expectCatalogResidueRefusesWithoutPoolMeta(noncanonical); +} + +TEST(CASBootstrapOrdering, NonemptyCatalogOnlyResidueRefusesWithoutPoolMeta) +{ + const RefCatalog nonempty{.entries = {CatalogEntry{ + .ns = RootNamespace{"test/nonempty"}, .state = NsState::Live, .incarnation = UInt128{1}, .creator = std::nullopt}}}; + expectCatalogResidueRefusesWithoutPoolMeta(encodeRefCatalog(nonempty)); +} + +TEST(CASBootstrapOrdering, CatalogWithAnyOtherCasResidueRefusesWithoutPoolMeta) +{ + const String canonical_empty = encodeRefCatalog(RefCatalog{}); + const Layout layout{kPrefix}; + const std::vector residuals{ + layout.ownerKey("test"), layout.epochKey("test"), layout.mountKey("test"), + layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"test/ns"}), RefTxnId{1, 1}), + layout.manifestKey(ManifestId{RootNamespace{"test/ns"}, ManifestRef{1, 1, 1}}), + layout.serverRootDataPrefix("test") + "residual", kPrefix + "/unknown"}; + for (const String & residual : residuals) + expectCatalogResidueRefusesWithoutPoolMeta(canonical_empty, residual); +} + +TEST(CASBootstrapOrdering, ListedCatalogMissingAtExactGetRefusesWithoutPoolMeta) +{ + auto backend = std::make_shared(); + const Layout layout{kPrefix}; + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, PutOutcome::Done); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", + [&] { Pool::open(backend, makeConfig()); }); + EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); +} + +/// (d) An existing healthy pool (meta present + data) → reopen is unchanged: the pool identity is +/// PRESERVED (the residual check sees `_pool_meta` present → the normal validate path; `_pool_meta` is +/// never re-minted). +TEST(CASBootstrapOrdering, HealthyPoolReopenPreservesIdentity) +{ + auto backend = std::make_shared(); + + UInt128 pool_id_first; + { + PoolPtr store = Pool::open(backend, makeConfig()); + pool_id_first = store->poolMeta().pool_id; + } /// clean teardown: drained farewell, so the reopen reclaims immediately + + PoolPtr store2 = Pool::open(backend, makeConfig()); + EXPECT_EQ(store2->lifecycle(), PoolLifecycle::Live); + EXPECT_EQ(store2->poolMeta().pool_id, pool_id_first) + << "a healthy reopen must NOT re-mint _pool_meta — the pool identity must be preserved"; +} + +/// (e) [D2] concurrent-opener case: debris from a SECOND concurrent fresh opener's in-flight battery (a +/// distinct probe uid) is skipped by the SAME structural rule as (c). Two openers racing over one shared +/// pool prefix must not make each other's zero-write residual check fail. +TEST(CASBootstrapOrdering, ConcurrentOpenerProbeDebrisIsAlsoSkipped) +{ + auto backend = std::make_shared(); + /// This mount's own crashed battery AND a concurrent opener's in-flight battery. + ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid2 + "/token", "probe-v1").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid2 + "/cas", "cas-s1").outcome, PutOutcome::Done); + backend->clearLog(); + + PoolPtr store; + ASSERT_NO_THROW(store = Pool::open(backend, makeConfig())); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); +} + +/// (f) The reserved subtree boundary: only objects strictly under `/_probe/` are ignorable +/// debris. A SIBLING look-alike that merely starts with `_probe` but is NOT under the `_probe/` subtree +/// (here `_probelike/…`) is genuine residual — the trailing `/` in the reserved prefix keeps it out — so +/// bootstrap fails closed over it. (Any object literally under `_probe/`, whatever its leaf shape, is +/// ephemeral capability-probe scratch a content-addressed pool never uses for durable state.) +TEST(CASBootstrapOrdering, ProbeSiblingLookalikeIsResidualNotDebris) +{ + auto backend = std::make_shared(); + ASSERT_EQ(backend->putIfAbsent("p/_probelike/token", "x").outcome, PutOutcome::Done); + backend->clearLog(); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", + [&] { Pool::open(backend, makeConfig()); }); + EXPECT_EQ(backend->writeCount(), 0u); + EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); +} + +/// (g) An OBSERVE / read-only open over a partially-erased pool (residual data, `_pool_meta` deleted) +/// must NOT mint a fresh `_pool_meta` — there is no truly-read-only backend, so a mint here is a real +/// write that would poison the next writable mount's residual check. It fails closed (typed INVALID_STATE) +/// with ZERO writes. The read-only path skips the residual check, so the fail-closed gate lives in +/// `createOrValidate` (`allow_mint=false`). +TEST(CASBootstrapOrdering, ReadOnlyOverResidualWithoutMetaFailsClosedNoMint) +{ + auto backend = std::make_shared(); + ASSERT_EQ(backend->putIfAbsent(residualRefLogKey(), "x").outcome, + PutOutcome::Done); + backend->clearLog(); + + PoolConfig cfg = makeConfig(); + cfg.read_only = true; + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to mint outside the verified bootstrap path", + [&] { Pool::open(backend, cfg); }); + + EXPECT_EQ(backend->writeCount(), 0u) << "an observe open must never write (least of all mint _pool_meta)"; + EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); +} + +/// (h) An observe / read-only open over a HEALTHY pool (meta present) is unchanged: it validates the +/// existing `_pool_meta` and succeeds, preserving the pool identity. `allow_mint=false` is never consulted +/// on the validate path. +TEST(CASBootstrapOrdering, ReadOnlyOverHealthyPoolSucceedsUnchanged) +{ + auto backend = std::make_shared(); + UInt128 pool_id_first; + { + PoolPtr store = Pool::open(backend, makeConfig()); /// writable: creates _pool_meta + pool_id_first = store->poolMeta().pool_id; + } + + PoolConfig cfg = makeConfig(); + cfg.read_only = true; + PoolPtr ro; + ASSERT_NO_THROW(ro = Pool::open(backend, cfg)); + ASSERT_TRUE(ro); + EXPECT_EQ(ro->poolMeta().pool_id, pool_id_first) << "an observe open over a healthy pool must not re-mint"; +} + +/// (i) `openForDecommission` over a pool whose `_pool_meta` is absent but whose owner anchor survives (a +/// partial erase) must NOT bootstrap a fresh identity — it fails closed (typed INVALID_STATE) with no +/// mint. Decommission operates on an existing member; a missing meta is a broken state, not a bootstrap. +TEST(CASBootstrapOrdering, DecommissionWithAbsentMetaFailsClosedNoMint) +{ + auto backend = std::make_shared(); + { + PoolPtr store = Pool::open(backend, makeConfig()); /// establishes owner anchor + _pool_meta + } + /// Delete only `_pool_meta`, leaving the owner anchor (and other control objects) behind. + { + const auto h = backend->head(kPoolMetaKey); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->deleteExact(kPoolMetaKey, h.token).kind, DeleteOutcome::Kind::Deleted); + } + backend->clearLog(); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to mint outside the verified bootstrap path", + [&] { Pool::openForDecommission(backend, makeConfig(), kSrid); }); + + EXPECT_EQ(backend->writeCount(), 0u) << "decommission must not mint a fresh _pool_meta"; + EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); +} diff --git a/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp new file mode 100644 index 000000000000..4b19f9ae967e --- /dev/null +++ b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp @@ -0,0 +1,1015 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Task 10 (spec §confirm-primitive, "Gate 1 -- exact-token identity under a lane snapshot"): the +/// ledger-side half of the publish-then-confirm relink handoff. +/// +/// `confirmExactRef` is a GATE, and the only direction in which it may fail is `Unknown`. A `Yes` +/// authorizes a remote receiver to promote a manifest it staged from this writer's blobs, so a `Yes` +/// produced from a stale, lagging or partially-recovered view is a live-blob-deletion bug, not a +/// missed optimization. Every test below therefore pins one of the six snapshot rules by constructing +/// the exact state in which a naive "look the row up and compare" implementation would answer `Yes` +/// (or `No`) and asserting `Unknown` instead. +/// +/// Two properties are contract, not detail, and are asserted as such: +/// - ZERO object-store I/O. A cold, evicted or recovering table answers `Unknown`; it must not +/// recover from storage to answer, and it must not even MATERIALIZE a runtime -- a read-only +/// interserver query must never be able to make this writer do work. +/// - The snapshot spans BOTH lane mutexes, so an append admitted concurrently is ordered strictly +/// after it: there is no window in which the confirm says `Yes` while a removal of that ref is +/// already admitted. +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int MEMORY_LIMIT_EXCEEDED; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +/// A `CountingBackend` with two recovery-side seams: a one-shot NON-transient exact GET failure (which +/// leaves the namespace runtime resident but UNRECOVERED, because recovery fails closed), and a +/// blocking exact GET (which parks a caller inside `ensureRefTableRecovered` with +/// `recovery_in_progress` set). The failure is deliberately `CORRUPTED_DATA`: +/// `isTransientRecoveryError` does not list it, so recovery fails fast instead of burning its retry +/// budget. +class RecoveryLatchBackend : public CountingBackend +{ +public: + using CountingBackend::get; + using CountingBackend::getStream; + using CountingBackend::putIfAbsent; + using CountingBackend::putOverwrite; + using CountingBackend::casPut; + + /// Set before the driving call; consumed by the first matching recovery GET. + String fail_get_once_key; + + std::optional get(const String & key, Range range) override + { + if (!fail_get_once_key.empty() && key == fail_get_once_key) + { + fail_get_once_key.clear(); + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "RecoveryLatchBackend: simulated non-transient exact GET failure"); + } + { + std::unique_lock lk(m); + if (!block_key.empty() && key == block_key) + { + entered = true; + cv.notify_all(); + /// Bounded (20s) so a wiring bug bounds the wait instead of hanging the suite. + cv.wait_for(lk, std::chrono::seconds(20), [&] { return block_key.empty(); }); + } + } + return CountingBackend::get(key, range); + } + + void armBlockedGet(const String & key) + { + std::lock_guard lk(m); + block_key = key; + entered = false; + } + void awaitBlockedGet() + { + std::unique_lock lk(m); + cv.wait_for(lk, std::chrono::seconds(20), [&] { return entered; }); + ASSERT_TRUE(entered) << "the recovery GET never parked -- the in-progress window was not exercised"; + } + void releaseBlockedGet() + { + { + std::lock_guard lk(m); + block_key.clear(); + } + cv.notify_all(); + } + +private: + std::mutex m; + std::condition_variable cv; + String block_key; + bool entered = false; +}; + +/// Parks the append lane's leader in the pre-carve window -- BEFORE it takes `ref_queue_mutex`, before +/// any `PUT`, and therefore with the table's apply-state still `Clean`. That is what isolates the +/// quiescence rule: the only thing wrong with the table while parked is that an append is in flight. +struct LeaderLatch +{ + std::mutex m; + std::condition_variable cv; + bool entered = false; + bool released = false; + + void arm(const PoolPtr & store) + { + store->setRefPreCarveHookForTest([this] + { + std::unique_lock lk(m); + if (entered) + return; /// only the FIRST carve parks; retries proceed straight through + entered = true; + cv.notify_all(); + /// Bounded (20s): a staging bug must bound the wait, not block the whole suite. + cv.wait_for(lk, std::chrono::seconds(20), [this] { return released; }); + }); + } + void awaitEntered() + { + std::unique_lock lk(m); + cv.wait_for(lk, std::chrono::seconds(20), [this] { return entered; }); + ASSERT_TRUE(entered) << "the append lane's leader never reached the pre-carve window"; + } + void release() + { + { + std::lock_guard lk(m); + released = true; + } + cv.notify_all(); + } +}; + +/// Rendezvous for the co-batching pre-carve hook of the chunked-flush case (same shape as +/// `gtest_cas_ref_chunked_flush.cpp`'s `CaseSync`). +struct CaseSync +{ + std::mutex m; + std::condition_variable cv; + bool entered = false; +}; + +/// `num_pairs` add-then-remove precommit op pairs for distinct refs, each naming a distinct manifest. +/// Every pair is undone immediately, so the LIVE state stays ~empty and validating thousands of ops +/// stays linear -- it is the OP COUNT, not the resident state, that drives the chunk split under test. +std::vector precommitAddRemovePairs(const String & prefix, size_t num_pairs, uint64_t manifest_epoch) +{ + std::vector ops; + ops.reserve(num_pairs * 2); + for (size_t i = 0; i < num_pairs; ++i) + { + const String ref = prefix + std::to_string(i); + const ManifestRef manifest{manifest_epoch, i + 1, 1}; + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref, manifest}; + ops.push_back(std::move(add)); + RefOp remove; + remove.kind = RefOpKind::OwnerTransition; + remove.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref, manifest}; + ops.push_back(std::move(remove)); + } + return ops; +} + +PoolPtr openPool(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +PoolPtr openPoolWithConfig(const BackendPtr & backend, PoolConfig config) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, std::move(config)); +} + +/// A legal blob-free part: an empty-entry manifest is enough to drive a real precommit+promote pair +/// through the append lane and leave a committed ref behind. Returns the committed `ManifestId`. +ManifestId publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref, + bool allow_repoint = false) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id, allow_repoint); + return id; +} + +/// Every request class `CountingBackend` observes, summed. The zero-I/O contract is asserted against +/// this total, so a confirm that quietly grew a HEAD or a GET fails the test rather than the review. +uint64_t backendRequests(const CountingBackend & b) +{ + return b.headTotal() + b.getTotal() + b.getStreamTotal() + b.putTotal() + b.listTotal(); +} + +/// One-shot throwing probe in the post-durable install region -- the only way to reach `NeedsRecovery` +/// transition now that §A1 made every install region allocation-free. Copied in shape from +/// `gtest_cas_ref_install_safety.cpp`: the exception is built OUTSIDE the region (constructing one +/// inside would allocate and trip `DENY_ALLOCATIONS_IN_SCOPE`), and it is `MEMORY_LIMIT_EXCEEDED` +/// rather than `LOGICAL_ERROR`, which aborts at construction in debug builds. +void armOneShotInstallFailure(const PoolPtr & store) +{ + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); +} + +} + + +/// Rule 5, the affirmative case: a warm, quiescent, `Ready`, fenced table whose committed row for +/// the ref names EXACTLY the asked-about manifest answers `Yes`. Its two negatives share the test +/// because they are the same rule read the other way: a different `ManifestRef` under the right name, +/// and a name that has no committed row at all, are both `No` -- a PROOF of the negative, not an +/// ambiguity. +TEST(CASConfirmExactRef, QuiescentExactMatchIsYes) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_yes"}; + + const ManifestId id = publishEmptyPart(store, ns, "x"); + + EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Yes); + + /// Same name, a manifest this ref never named. + ManifestRef other = id.ref; + ++other.manifest_ordinal; + EXPECT_EQ(store->confirmExactRef(ns, "x", other), ConfirmAnswer::No); + + /// A name with no committed row at all -- on a warm table that is knowledge, not ambiguity. + EXPECT_EQ(store->confirmExactRef(ns, "no_such_ref", id.ref), ConfirmAnswer::No); +} + + +/// Rule 5, the repoint case (spec §testing "repointed live part"): the part is still live and the ref +/// name still resolves, but it now names a DIFFERENT manifest. The old token must be `No` -- this is +/// the case gate 0 cannot see at all, because the part object is `Active` throughout. +TEST(CASConfirmExactRef, RepointedRefIsNo) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_repoint"}; + + const ManifestId first = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", first.ref), ConfirmAnswer::Yes); + + const ManifestId second = publishEmptyPart(store, ns, "x", /*allow_repoint=*/true); + ASSERT_NE(first.ref, second.ref) << "a repoint must mint a fresh ManifestRef (the ABA barrier)"; + + EXPECT_EQ(store->confirmExactRef(ns, "x", first.ref), ConfirmAnswer::No); + EXPECT_EQ(store->confirmExactRef(ns, "x", second.ref), ConfirmAnswer::Yes); +} + + +/// Rule 5, the drop-and-recreate case: the ref name is removed and then published again. The name +/// resolves again, so only EXACT `ManifestRef` equality separates the new binding from the old one -- +/// mint-tightening (spec §A3) is what guarantees the two can never collide. +TEST(CASConfirmExactRef, DroppedAndRecreatedRefIsNo) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_recreate"}; + + const ManifestId first = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", first.ref), ConfirmAnswer::Yes); + + store->dropRef(ns, "x"); + EXPECT_EQ(store->confirmExactRef(ns, "x", first.ref), ConfirmAnswer::No) + << "a dropped ref cannot authorize anything"; + + const ManifestId second = publishEmptyPart(store, ns, "x"); + ASSERT_NE(first.ref, second.ref); + EXPECT_EQ(store->confirmExactRef(ns, "x", first.ref), ConfirmAnswer::No); + EXPECT_EQ(store->confirmExactRef(ns, "x", second.ref), ConfirmAnswer::Yes); +} + + +/// Rule 2, the cold case: a namespace this mount has never touched has no resident runtime, so the +/// answer is `Unknown` -- and producing it must cost ZERO object-store requests AND must not create a +/// runtime. Materializing one here would let a remote caller populate this writer's table cache with +/// unrecovered entries by asking about namespaces that do not exist. +TEST(CASConfirmExactRef, ColdTableIsUnknownWithZeroBackendRequests) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace warm{"srv1/confirm_cold_warm"}; + const RootNamespace cold{"srv1/confirm_cold_never_touched"}; + + const ManifestId id = publishEmptyPart(store, warm, "x"); + + const size_t cached_before = store->refTablesCachedCountForTest(); + backend->resetCounts(); + + EXPECT_EQ(store->confirmExactRef(cold, "x", id.ref), ConfirmAnswer::Unknown); + + EXPECT_EQ(backendRequests(*backend), 0u) + << "a cold table must answer Unknown without recovering from storage"; + EXPECT_EQ(store->refTablesCachedCountForTest(), cached_before) + << "confirmExactRef must find the runtime, never create one"; +} + + +/// Rule 2, the evicted case: a table that WAS warm and was dropped by the whole-table cache budget is +/// indistinguishable from a cold one here -- the runtime is gone, so the committed view is gone with +/// it, and re-reading it would be object-store I/O. +TEST(CASConfirmExactRef, EvictedTableIsUnknownWithZeroBackendRequests) +{ + auto backend = std::make_shared(); + auto store = openPoolWithConfig(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .ref_table_cache_bytes = 1}); + const RootNamespace ns_a{"srv1/confirm_evict_a"}; + const RootNamespace ns_b{"srv1/confirm_evict_b"}; + + const ManifestId id_a = publishEmptyPart(store, ns_a, "x"); + ASSERT_EQ(store->confirmExactRef(ns_a, "x", id_a.ref), ConfirmAnswer::Yes); + + /// A 1-byte budget is below one table's weight, so touching another table evicts the idle one. + publishEmptyPart(store, ns_b, "y"); + ASSERT_FALSE(store->refTableCachedForTest(ns_a)) << "ns_a must have been evicted"; + + backend->resetCounts(); + EXPECT_EQ(store->confirmExactRef(ns_a, "x", id_a.ref), ConfirmAnswer::Unknown); + EXPECT_EQ(backendRequests(*backend), 0u) + << "an evicted table must answer Unknown without re-recovering"; + EXPECT_FALSE(store->refTableCachedForTest(ns_a)) + << "the confirm must not have re-recovered the evicted table as a side effect"; +} + + +/// Rule 2, the resident-but-unrecovered case: recovery failed closed, so a runtime EXISTS in the cache +/// with an empty, meaningless `state`. A naive lookup reads that empty state and answers `No`; the +/// correct answer is `Unknown`, because nothing about the durable table is known here. +TEST(CASConfirmExactRef, UnrecoveredResidentTableIsUnknownWithZeroBackendRequests) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_unrecovered"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns); + + const size_t cached_before = store->refTablesCachedCountForTest(); + backend->fail_get_once_key = store->layout().refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_THROW(store->resolveRef(ns, "x"), DB::Exception); + + ASSERT_EQ(store->refTablesCachedCountForTest(), cached_before + 1u) + << "the failed recovery must still leave a runtime resident"; + ASSERT_FALSE(store->refTableCachedForTest(ns)) << "that runtime must be UNRECOVERED"; + + backend->resetCounts(); + EXPECT_EQ(store->confirmExactRef(ns, "x", ManifestRef{1, 1, 1}), ConfirmAnswer::Unknown); + EXPECT_EQ(backendRequests(*backend), 0u) + << "an unrecovered table must answer Unknown without driving recovery"; + EXPECT_FALSE(store->refTableCachedForTest(ns)) + << "the confirm must not have recovered the table as a side effect"; +} + + +/// Rule 2, the recovering case: another caller is INSIDE `ensureRefTableRecovered`, parked on its +/// exact `_ckpt` GET. The runtime is resident, `recovery_in_progress` is set, and the state is still +/// empty. Waiting for that recovery would be exactly the "recover from storage to answer" this +/// primitive refuses. +TEST(CASConfirmExactRef, RecoveryInProgressIsUnknownWithZeroBackendRequests) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_recovering"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns); + backend->armBlockedGet(store->layout().refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns))); + std::exception_ptr recovery_error; + std::thread recoverer([&] + { + try + { + store->resolveRef(ns, "x"); + } + catch (...) + { + recovery_error = std::current_exception(); + } + }); + backend->awaitBlockedGet(); + + backend->resetCounts(); + /// The parked exact GET is outside `state_mutex` (up to the 20s block bound), so this + /// call must NOT be a blocking acquire of that mutex: waiting would make the confirm pay for + /// somebody else's recovery while holding pool-wide append admission. The elapsed bound is what + /// pins that -- it is an order of magnitude below the park, so it cannot pass by luck. + const auto started = std::chrono::steady_clock::now(); + const ConfirmAnswer answer = store->confirmExactRef(ns, "x", ManifestRef{1, 1, 1}); + const auto elapsed = std::chrono::steady_clock::now() - started; + const uint64_t requests = backendRequests(*backend); + + backend->releaseBlockedGet(); + recoverer.join(); + + if (recovery_error) + { + try + { + std::rethrow_exception(recovery_error); + } + catch (const std::exception & e) + { + FAIL() << "the driving recovery unexpectedly failed: " << e.what(); + } + catch (...) + { + FAIL() << "the driving recovery unexpectedly failed with a non-standard exception"; + } + } + + EXPECT_EQ(answer, ConfirmAnswer::Unknown); + EXPECT_EQ(requests, 0u) + << "a recovering table must answer Unknown without issuing (or waiting on) any request"; + EXPECT_LT(elapsed, std::chrono::seconds(5)) + << "the confirm waited for the in-progress recovery instead of answering Unknown"; +} + + +/// Rule 3, the in-flight case: an append is admitted and its leader is parked in the pre-carve window. +/// Nothing is durable yet and the committed row still matches EXACTLY -- which is precisely why a +/// naive implementation answers `Yes` here, and precisely why that is the TOCTOU this design closes. +/// The apply-state is still `Clean` at this point, so rule 3 is what produces the `Unknown`, not +/// rule 4. +TEST(CASConfirmExactRef, InFlightAppendIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_inflight"}; + + const ManifestId id = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Yes); + + LeaderLatch latch; + latch.arm(store); + std::thread dropper([&] { store->dropRef(ns, "x"); }); + latch.awaitEntered(); + + /// Sampled while parked, asserted after the join: a failed assertion here must not skip the + /// release, or the still-joinable `dropper` would terminate the whole suite instead of failing one + /// test. + const bool leader_active = store->refLeaderActiveForTest(ns); + const RefLaneState apply_state = store->laneStateForTest(ns); + const ConfirmAnswer while_in_flight = store->confirmExactRef(ns, "x", id.ref); + + latch.release(); + dropper.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_TRUE(leader_active); + EXPECT_EQ(apply_state, RefLaneState::Ready) + << "the pre-carve window is before any PUT, so rule 4 must not be what answers here"; + EXPECT_EQ(while_in_flight, ConfirmAnswer::Unknown) + << "an admitted append makes the whole table's committed view provisional"; + + EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::No); +} + + +/// Rule 3, mid-tenure (spec §testing "mid-tenure chunked flush", `CarvePhaseForTest::ChunkReseed`): +/// one leader tenure commits MULTIPLE durable transactions, so at a chunk boundary the table is +/// PARTIALLY durable. `leader_active` covers the whole tenure, so this is already `Unknown` -- a wider +/// unknown window under load, never a hole. The confirm is issued on the leader's own thread, which is +/// safe because the boundary holds neither lane mutex. +TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_mid_tenure"}; + + const ManifestId id = publishEmptyPart(store, ns, "seed"); + ASSERT_EQ(store->confirmExactRef(ns, "seed", id.ref), ConfirmAnswer::Yes); + + std::atomic boundaries{0}; + std::atomic unknown_at_boundary{0}; + std::atomic requests_at_boundary{0}; + store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::ChunkReseed) + return; + boundaries.fetch_add(1); + const uint64_t before = backendRequests(*backend); + if (store->confirmExactRef(ns, "seed", id.ref) == ConfirmAnswer::Unknown) + unknown_at_boundary.fetch_add(1); + requests_at_boundary.fetch_add(static_cast(backendRequests(*backend) - before)); + }); + + /// Two co-batched items of 3000 ops each (1500 precommit add/remove pairs). One item may not + /// exceed the 5000-op `ref_txn_max_ops` cap on its own -- that fails the item outright -- so the + /// chunk boundary has to come from a BATCH: 6000 ops carved into one tenure split into two + /// transactions, firing exactly one boundary. The pre-carve hook parks the first caller until the + /// second is queued, which is what makes the co-batching deterministic. + auto sync = std::make_shared(); + store->setRefPreCarveHookForTest([sync, store, ns] + { + std::unique_lock lk(sync->m); + if (sync->entered) + return; + sync->entered = true; + sync->cv.notify_all(); + sync->cv.wait_for(lk, std::chrono::seconds(20), + [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + + auto append = [&store, &ns](const String & prefix, uint64_t manifest_epoch) + { + std::vector item_ops = precommitAddRemovePairs(prefix, 1500, manifest_epoch); + store->appendRefOps(ns, MutationScope::ref(prefix), + [ops = std::move(item_ops)](const RefTableState &) { return ops; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }; + std::thread a([&] { append("aaa_", 900000001); }); + { + std::unique_lock lk(sync->m); + sync->cv.wait_for(lk, std::chrono::seconds(20), [&] { return sync->entered; }); + } + std::thread b([&] { append("bbb_", 900000002); }); + /// The parked leader re-evaluates its predicate only when notified, so the queue depth is polled + /// here and the leader released explicitly once both items are admitted. + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(20); + while (store->refQueuePendingForTest(ns) < 2 && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); + sync->cv.notify_all(); + a.join(); + b.join(); + store->setRefPreCarveHookForTest(nullptr); + store->setCarveHookForTest(nullptr); + + ASSERT_GE(boundaries.load(), 1) << "the flush did not chunk -- the mid-tenure window was not exercised"; + EXPECT_EQ(unknown_at_boundary.load(), boundaries.load()) + << "a mid-tenure, partially-durable table must never confirm"; + EXPECT_EQ(requests_at_boundary.load(), 0) << "the mid-tenure confirm must still be I/O-free"; + + /// The tenure is over: the seed ref is untouched by it and confirms again. + EXPECT_EQ(store->confirmExactRef(ns, "seed", id.ref), ConfirmAnswer::Yes); +} + + +/// Rule 3, the wedge case: the lane holds one conditional `PUT` whose outcome is unknown, so the table +/// may be MISSING a durable transaction -- possibly the very removal being asked about. The committed +/// row still matches exactly, so only the wedge can produce the refusal. +TEST(CASConfirmExactRef, WedgedLaneIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_wedge"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns); + + const ManifestId id = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Yes); + + store->forceWedgeForTest(ns, /*writer_epoch=*/1, /*ref_sequence=*/9999, + store->layout().refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 9999}), "synthetic"); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + backend->resetCounts(); + EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Unknown); + EXPECT_EQ(backendRequests(*backend), 0u) + << "a wedged lane must answer Unknown without trying to resolve the wedge"; +} + + +/// `NeedsRecovery` is table-scoped, so confirmation refuses even a row that still looks perfect. +TEST(CASConfirmExactRef, NeedsRecoveryIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_poison"}; + + const ManifestId keep = publishEmptyPart(store, ns, "keep"); + ASSERT_EQ(store->confirmExactRef(ns, "keep", keep.ref), ConfirmAnswer::Yes); + + armOneShotInstallFailure(store); + EXPECT_THROW(publishEmptyPart(store, ns, "other"), DB::Exception); + store->setInstallRegionProbeForTest(nullptr); + + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + ASSERT_FALSE(store->refLaneWedgedForTest(ns)); + ASSERT_FALSE(store->refLeaderActiveForTest(ns)); + + EXPECT_EQ(store->confirmExactRef(ns, "keep", keep.ref), ConfirmAnswer::Unknown) + << "a table that may be missing a durable transaction cannot confirm ANY of its rows"; +} + + +/// Rule 6, checked LAST: the committed row matches exactly, the lane is quiescent and clean -- but this +/// node no longer holds the mount incarnation, so it is no longer the namespace's single writer and +/// cannot speak for the durable table at all. Another writer may already have repointed the ref. +TEST(CASConfirmExactRef, LostMountFenceIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_fence"}; + + const ManifestId id = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Yes); + + store->tripMountLost(); + + ASSERT_TRUE(store->refTableCachedForTest(ns)) + << "the table must still be resident, so it is the FENCE that refuses, not residency"; + backend->resetCounts(); + EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Unknown); + EXPECT_EQ(backendRequests(*backend), 0u); + + /// The fence is checked LAST, so it gates only the `Yes`: a token that does not match the committed + /// row is still reported as `No` under a lost fence. That is deliberate and harmless -- `No` and + /// `Unknown` are the same outcome for the caller (both `SourceProofFailed`) -- and pinning it here + /// keeps a future reordering of the rules from changing the answer silently. + ManifestRef other = id.ref; + ++other.manifest_ordinal; + EXPECT_EQ(store->confirmExactRef(ns, "x", other), ConfirmAnswer::No); +} + + +/// The two-mutex snapshot race (spec §testing, "an append admitted concurrently is ordered strictly +/// after the snapshot"). The confirm holds `ref_queue_mutex` across the whole snapshot, and admission +/// (`pending.push_back`) takes that same mutex, so every append is either entirely before the snapshot +/// (and visible as a pending item -> `Unknown`) or entirely after it. What must NOT exist is a window +/// in which the removal is admitted and the confirm still says `Yes`. +/// +/// The three phases are driven deterministically rather than hammered: before admission -> `Yes`; +/// from admission until the transaction is durable -> `Unknown`; after -> `No`. `Yes` is never +/// observable once the removal has been admitted. +TEST(CASConfirmExactRef, ConcurrentAppendIsOrderedAfterTheSnapshot) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_race"}; + + const ManifestId id = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Yes) << "phase 1: before admission"; + + LeaderLatch latch; + latch.arm(store); + std::thread dropper([&] { store->dropRef(ns, "x"); }); + latch.awaitEntered(); + + /// Phase 2: admitted, nothing durable. Sampled repeatedly so a single lucky interleaving cannot + /// pass for the invariant, and TALLIED rather than asserted -- an assertion here would skip the + /// release below and terminate the suite on the still-joinable `dropper`. + int not_unknown = 0; + int saw_yes = 0; + for (int i = 0; i < 64; ++i) + { + const ConfirmAnswer a = store->confirmExactRef(ns, "x", id.ref); + if (a != ConfirmAnswer::Unknown) + ++not_unknown; + if (a == ConfirmAnswer::Yes) + ++saw_yes; + } + + latch.release(); + dropper.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_EQ(saw_yes, 0) << "phase 2: an admitted removal must never leave a Yes visible"; + EXPECT_EQ(not_unknown, 0) << "phase 2: an admitted append makes the committed view provisional"; + + /// Phase 3: durable and applied. + EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::No) << "phase 3: after the removal"; +} + + +/// =========================================================================================== +/// Task 11: the EXCHANGE-level confirm -- `IContentAddressedExchange::ownsNamespace` (routing) and +/// `::confirmExactRef` (the storage forward of gate 1, plus the token text and the disk lifecycle). +/// +/// Gate 0 is not exercised here on purpose: it reads a `StorageReplicatedMergeTree` parts set, so +/// `Deleting`, absent, other-disk and the `MOVE ... TO DISK` same-name case are integration-level and +/// belong to the Task 16 pytest battery. +/// =========================================================================================== + +namespace +{ + +/// A storage adapter over its own private local object storage. `startup` is the caller's business: +/// several tests below assert behavior BEFORE it and AFTER `shutdown`. +std::shared_ptr makeExchangeStorage(const std::string & server_root_id) +{ + auto settings = tests::makeSettingsForTest( + server_root_id, std::filesystem::temp_directory_path() / "ca_confirm_exchange_scratch"); + return std::make_shared( + tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); +} + +const std::string kExchangeTableDir = "e11/e11e11e1-0808-4808-8808-080808080808"; +const std::string kExchangePartName = "all_1_1_0"; +const std::string kExchangePartDir = kExchangeTableDir + "/" + kExchangePartName; + +/// Commit one real part through the ordinary transaction path, so the committed binding under test is +/// produced exactly the way an INSERT produces it. +void commitExchangePart(DB::ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kExchangeTableDir + "/tmp_insert_" + kExchangePartName + "/data.bin", + 65536, DB::WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kExchangeTableDir + "/tmp_insert_" + kExchangePartName, kExchangePartDir); + tx->commit(DB::NoCommitOptions{}); +} + +/// The token the sender mints for the committed part, decoded back into its fields. Read through the +/// real offer path rather than reconstructed, so the tests below exercise exactly what goes on the wire. +DB::CasRelinkSourceToken readSourceToken(const DB::ContentAddressedMetadataStorage & storage) +{ + const auto offer = storage.getRelinkOffer(kExchangePartDir); + EXPECT_TRUE(offer.has_value()) << "the committed part must offer a manifest to relink"; + if (!offer) + return {}; + const auto token = DB::decodeCasRelinkSourceToken(offer->confirm_token); + EXPECT_TRUE(token.has_value()) << "the sender minted a token its own decoder rejects"; + return token.value_or(DB::CasRelinkSourceToken{}); +} + +} + + +/// The canonical text form of a `ManifestRef` becomes wire input with Task 11: it is what the confirm +/// token carries and what `confirmExactRef` compares. It is therefore parsed as untrusted input -- +/// exactly three decimal fields, nothing consumed partially, no sign, no padding -- and the parser is +/// pinned as the exact inverse of the renderer, because a mismatch between the two would silently turn +/// every confirm into an `Unknown` (or, far worse, make two different manifests compare equal). +TEST(CASConfirmExactRef, ManifestRefTextRoundTripsAndRejectsMalformedTokens) +{ + for (const ManifestRef & ref : {ManifestRef{1, 1, 1}, ManifestRef{7, 42, 999999}, + ManifestRef{18446744073709551615ULL, 18446744073709551615ULL, 123}}) + { + const String text = manifestRefDebugString(ref); + const auto parsed = tryParseManifestRef(text); + ASSERT_TRUE(parsed.has_value()) << "the renderer produced text its own parser rejects: " << text; + EXPECT_EQ(*parsed, ref) << text; + } + + EXPECT_EQ(manifestRefDebugString(ManifestRef{7, 42, 3}), "7:42:3") << "the canonical form is epoch:build:ordinal"; + + for (const std::string_view malformed : { + "", "1", "1:2", "1:2:3:4", ":2:3", "1::3", "1:2:", "1:2:x", "x:2:3", " 1:2:3", "1:2:3 ", + "1: 2:3", "+1:2:3", "-1:2:3", "1:2:3\n", "0x1:2:3", + /// `0` is the reserved invalid ordinal and is never emitted; `1000000` is past the six-digit + /// filename range, so neither can name a real manifest. + "1:2:0", "1:2:1000000", + /// One past `uint64` / `uint32` -- `from_chars` reports overflow rather than truncating. + "18446744073709551616:1:1", "1:18446744073709551616:1", "1:1:4294967296"}) + { + EXPECT_FALSE(tryParseManifestRef(malformed).has_value()) + << "accepted a malformed manifest reference: '" << malformed << "'"; + } +} + + +/// Task 13, the confirm token's wire codec (spec §wire-protocol). The token is minted by the sender, +/// stored nowhere, and handed back by an untrusted peer, so the only property that matters is that +/// decode is the exact inverse of encode: the fields the sender meant are the fields that route and +/// compare. A codec that merged two fields, or that let a separator through unescaped, would let a +/// peer aim a confirm at a namespace the sender never named. +TEST(CASConfirmExactRef, SourceTokenRoundTripsThroughItsWireForm) +{ + const auto round_trip = [](const DB::CasRelinkSourceToken & token, const char * what) + { + const auto text = DB::encodeCasRelinkSourceToken(token); + ASSERT_TRUE(text.has_value()) << what; + /// The wire form is cookie-safe and URL-safe by construction: only the RFC 3986 unreserved set, + /// the escape character, and the field separator ever appear in it. + for (const char ch : *text) + EXPECT_TRUE(std::isalnum(static_cast(ch)) + || std::string_view("-._~%|").find(ch) != std::string_view::npos) + << "the wire form leaked an unsafe character '" << ch << "' from " << what << ": " << *text; + + const auto decoded = DB::decodeCasRelinkSourceToken(*text); + ASSERT_TRUE(decoded.has_value()) << what << ": " << *text; + EXPECT_EQ(decoded->pool_uuid, token.pool_uuid) << what; + EXPECT_EQ(decoded->server_root_id, token.server_root_id) << what; + EXPECT_EQ(decoded->root_namespace, token.root_namespace) << what; + EXPECT_EQ(decoded->ref_name, token.ref_name) << what; + EXPECT_EQ(decoded->part_name, token.part_name) << what; + EXPECT_EQ(decoded->manifest_ref_text, token.manifest_ref_text) << what; + }; + + round_trip({"abcdef0123456789", "srv1", "srv1/store/abc/abcdef@cas@", "all_1_1_0", "all_1_1_0", "1:1:1"}, + "the ordinary shape"); + /// The characters that make a naive codec wrong: the separator itself, the escape character, the + /// cookie-forbidden set, and the `/`+`@` a namespace and a detached ref carry as a matter of course. + round_trip({"p|o%o=l", "srv 1;x,y", "srv 1;x,y/store/abc/abcdef@cas@", "detached/broken_all_1_1_0", + "all_1_1_0", "18446744073709551615:18446744073709551615:999999"}, + "the hostile shape"); + /// A field at the cap must survive; one past it must not (asserted below). + round_trip({String(256, 'a'), "srv1", "srv1/store/abc/abcdef@cas@", "all_1_1_0", "all_1_1_0", "1:1:1"}, + "a field at the length cap"); +} + +/// Everything a peer can hand back that is not a token this sender minted. None of these may decode: +/// a decoded-but-wrong token routes a confirm somewhere, and "somewhere" is exactly what routing exists +/// to prevent. Refusing costs a byte fetch and nothing else. +TEST(CASConfirmExactRef, SourceTokenRejectsMalformedAndOverlongInput) +{ + const DB::CasRelinkSourceToken good{"pool", "srv1", "srv1/store/abc/abcdef@cas@", "all_1_1_0", "all_1_1_0", "1:1:1"}; + const String text = DB::encodeCasRelinkSourceToken(good).value(); + + /// Every literal below is a SEVEN-segment token (version + six fields) unless it is testing the + /// segment count itself, so each case fails for the reason it names and not because it is short. + for (const std::string_view malformed : { + /// Empty, no version, the wrong version, and versions that merely start or end right. + "", "|a|b|c|d|e|f", "car0|a|b|c|d|e|f", "car|a|b|c|d|e|f", "car11|a|b|c|d|e|f", + /// Too few and too many fields -- a shape that is one field off must not shift the rest. + "car1|a|b|c|d|e", "car1|a|b|c|d|e|f|g", "car1", "car1|", + /// An empty field: a token with a hole in it routes somewhere it was not meant to. + "car1||b|c|d|e|f", "car1|a|b|c|d|e|", + /// Malformed escapes: truncated, non-hex, and a lone escape character. + "car1|%|b|c|d|e|f", "car1|%4|b|c|d|e|f", "car1|%zz|b|c|d|e|f", "car1|a%|b|c|d|e|f", + /// Unescaped bytes outside the unreserved set: the decoder is the encoder's inverse, so + /// anything the encoder would have escaped is not a token, however readable it looks. + "car1|a/b|c|d|e|f|g", "car1|a b|c|d|e|f|g", "car1|a@b|c|d|e|f|g", "car1|1:1:1|b|c|d|e|f", + /// A control character smuggled in as an escape -- the classic forged-log-line vector. + "car1|a%00b|c|d|e|f|g", "car1|a%0Ab|c|d|e|f|g"}) + { + EXPECT_FALSE(DB::decodeCasRelinkSourceToken(malformed).has_value()) + << "accepted a malformed source token: '" << malformed << "'"; + } + + /// Over-long: refused in BOTH directions, so an over-long field can neither be minted nor accepted. + DB::CasRelinkSourceToken too_long = good; + too_long.ref_name = String(257, 'a'); + EXPECT_FALSE(DB::encodeCasRelinkSourceToken(too_long).has_value()); + EXPECT_FALSE(DB::decodeCasRelinkSourceToken( + "car1|pool|srv1|ns|" + String(257, 'a') + "|all_1_1_0|1%3A1%3A1").has_value()); + /// A field whose ENCODED form blows the whole-token cap (every byte escapes to three). + DB::CasRelinkSourceToken all_escaped = good; + all_escaped.root_namespace = String(200, ' '); + all_escaped.ref_name = String(200, ' '); + EXPECT_FALSE(DB::encodeCasRelinkSourceToken(all_escaped).has_value()); + EXPECT_FALSE(DB::decodeCasRelinkSourceToken(String(2000, 'a')).has_value()); + + /// An empty field is refused on the way out too, not only on the way in. + DB::CasRelinkSourceToken empty_field = good; + empty_field.part_name.clear(); + EXPECT_FALSE(DB::encodeCasRelinkSourceToken(empty_field).has_value()); + + /// The control-character refusal is symmetric: the sender cannot mint one either. + DB::CasRelinkSourceToken control = good; + control.server_root_id = "srv\n1"; + EXPECT_FALSE(DB::encodeCasRelinkSourceToken(control).has_value()); + + /// Sanity: the good token itself decodes, so the rejections above are about the input and not about + /// a codec that refuses everything. + EXPECT_TRUE(DB::decodeCasRelinkSourceToken(text).has_value()); +} + + +/// Routing (spec §wire-protocol). A pool UUID is shared by every server root writing into the pool, so +/// it cannot select the mount entitled to answer for a namespace; `ownsNamespace` is what does. It is a +/// pure string question about the mount's own identity: no pool, no I/O, no lifecycle -- asserted here +/// by answering the same before `startup`, while live, and after `shutdown`. A routing predicate that +/// could throw would turn a misrouted question into an error instead of an unproven answer. +TEST(CASConfirmExactRef, OwnsNamespaceSelectsTheMountByServerRootInEveryLifecycleState) +{ + auto storage = makeExchangeStorage("srv1"); + + const auto assert_routing = [&](const char * phase) + { + /// Live and detached namespaces are `/` (`liveNamespace`). + EXPECT_TRUE(storage->ownsNamespace("srv1", "srv1/store/abc/abcdef@cas@")) << phase; + EXPECT_TRUE(storage->ownsNamespace("srv1", storage->liveNamespace("abcdef").string())) << phase; + + /// A different server root's namespace, and this namespace asked about under a different server + /// root: the same pool, a different owner. Both must miss, or a confirm could be answered by a + /// mount that never wrote the ref. + EXPECT_FALSE(storage->ownsNamespace("srv2", "srv1/store/abc/abcdef@cas@")) << phase; + EXPECT_FALSE(storage->ownsNamespace("srv1", "srv2/store/abc/abcdef@cas@")) << phase; + + /// The prefix trap: a bare `starts_with(server_root_id)` would let `srv1` claim `srv10`. + EXPECT_FALSE(storage->ownsNamespace("srv1", "srv10/store/abc/abcdef@cas@")) << phase; + EXPECT_FALSE(storage->ownsNamespace("srv10", "srv10/store/abc/abcdef@cas@")) << phase; + + /// The server root itself is not a namespace, and an unprefixed shadow path belongs to no mount. + EXPECT_FALSE(storage->ownsNamespace("srv1", "srv1")) << phase; + EXPECT_FALSE(storage->ownsNamespace("srv1", "shadow/backup/store/abc/abcdef")) << phase; + + /// A prefixed shadow namespace is ordinary owned content: the freeze belongs to the root + /// that made it, so relink routing treats it exactly like a live namespace. + EXPECT_TRUE(storage->ownsNamespace("srv1", "srv1/shadow/backup/store/abc/abcdef")) << phase; + EXPECT_FALSE(storage->ownsNamespace("srv10", "srv1/shadow/backup/store/abc/abcdef")) << phase; + + /// Empty fields are never a match -- an absent token field must not route anywhere. + EXPECT_FALSE(storage->ownsNamespace("", "")) << phase; + EXPECT_FALSE(storage->ownsNamespace("", "srv1/store/abc/abcdef@cas@")) << phase; + EXPECT_FALSE(storage->ownsNamespace("srv1", "")) << phase; + }; + + assert_routing("before startup"); + storage->startup(); + assert_routing("while live"); + storage->shutdown(); + assert_routing("after shutdown"); +} + + +/// The storage forward of gate 1, driven end to end: a part committed through the ordinary transaction +/// path, and a token read out of the very manifest body the sender puts on the wire. The three +/// non-`Yes` cases pin the two halves this layer adds on top of the ledger -- the token text is decoded +/// here, and a namespace this mount holds no resident runtime for is an ambiguity, not a `No`. +TEST(CASConfirmExactRef, StorageConfirmAnswersForTheCommittedBinding) +{ + auto storage = makeExchangeStorage("test"); + storage->startup(); + commitExchangePart(*storage); + + const DB::CasRelinkSourceToken token = readSourceToken(*storage); + ASSERT_TRUE(storage->ownsNamespace("test", token.root_namespace)) + << "the sender must route its own committed namespace to itself: " << token.root_namespace; + + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, token.manifest_ref_text), + DB::CasConfirmAnswer::Yes); + + /// A manifest this ref never named, and a ref name that was never committed: both are knowledge on + /// a warm table, and both are `No` -- which the caller must still treat as "not proven". + const auto other_ref = tryParseManifestRef(token.manifest_ref_text); + ASSERT_TRUE(other_ref.has_value()); + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, + manifestRefDebugString(ManifestRef{other_ref->writer_epoch, + other_ref->build_sequence, + other_ref->manifest_ordinal + 1})), + DB::CasConfirmAnswer::No); + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, "all_9_9_9", token.manifest_ref_text), + DB::CasConfirmAnswer::No); + + /// A namespace with no resident runtime: the ledger will not recover one to answer, so this is an + /// ambiguity. It must not read as "the ref does not exist". + EXPECT_EQ(storage->confirmExactRef("test/store/zzz/zzzzzz@cas@", kExchangePartName, token.manifest_ref_text), + DB::CasConfirmAnswer::Unknown); + + /// An unparsable token: the question cannot be understood, so it cannot be answered `No`. + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, "not-a-manifest-ref"), + DB::CasConfirmAnswer::Unknown); + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, ""), + DB::CasConfirmAnswer::Unknown); + + storage->shutdown(); +} + + +/// The disk's own lifecycle is an answer, not an exception. The confirm is served on an interserver +/// request, and the caller has a durable precommit waiting on it: a thrown `INVALID_STATE` would have +/// to be classified by the HTTP layer, whereas `Unknown` is already the taxonomy's "not proven". Only +/// `Yes` authorizes, and no lifecycle state can produce one. +TEST(CASConfirmExactRef, StorageConfirmIsUnknownWhenTheDiskCannotSpeakForItsView) +{ + auto storage = makeExchangeStorage("test"); + + /// Never started: no pool has ever been published, so there is no committed view at all. + EXPECT_EQ(storage->confirmExactRef("test/store/abc/abcdef@cas@", kExchangePartName, "1:1:1"), + DB::CasConfirmAnswer::Unknown); + + storage->startup(); + commitExchangePart(*storage); + const DB::CasRelinkSourceToken token = readSourceToken(*storage); + ASSERT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, token.manifest_ref_text), + DB::CasConfirmAnswer::Yes); + + /// Shut down: the same question, the same token, and the same table -- but this process no longer + /// speaks for the namespace. + storage->shutdown(); + EXPECT_EQ(storage->confirmExactRef(token.root_namespace, kExchangePartName, token.manifest_ref_text), + DB::CasConfirmAnswer::Unknown); +} diff --git a/src/Disks/tests/gtest_cas_decommission.cpp b/src/Disks/tests/gtest_cas_decommission.cpp new file mode 100644 index 000000000000..620dea710a3f --- /dev/null +++ b/src/Disks/tests/gtest_cas_decommission.cpp @@ -0,0 +1,1365 @@ +#include "cas_test_helpers.h" +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +using namespace DB; +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +/// Open a store for the VICTIM srid over `backend` (the pool's future dead member). +PoolPtr openVictim(std::shared_ptr backend) +{ + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); +} + +void drainCompletedNamespaceRemovals(const std::shared_ptr & backend) +{ + PoolConfig config{ + .pool_prefix = "p", + .server_root_id = "gc", + .gc_fold_threshold = 1, + .gc_fold_max_defer_rounds = 0}; + auto store = Pool::open(backend, config); + Gc gc(store, UInt128{991}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); +} + +/// Fails `deleteExact` for one or two designated keys -- either by throwing (a transient backend +/// hiccup) or by returning a synthetic `TokenMismatch` (a "listed but raced" outcome) -- delegating +/// every other key to the base `InMemoryBackend` untouched. Drives the drain phases' per-object +/// fail-close path (`deleteListedPrefix`/`sweepNamespace`, `CasDecommission.cpp`/ +/// `CasOrphanManifestSweep.cpp`): a failure on one listed object must record a warning and let the rest +/// of the sweep proceed, never abort the whole phase. +/// +class FailingDeleteBackend : public InMemoryBackend +{ +public: + void failWithThrow(const String & key) { throw_key = key; } + void failWithTokenMismatch(const String & key) { mismatch_key = key; } + /// Clears every injected failure -- the resume half of a fail-then-retry test (Task 4). + void disarm() { throw_key.clear(); mismatch_key.clear(); } + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (key == throw_key) + throw std::runtime_error("injected transient delete failure for " + key); + if (key == mismatch_key) + return DeleteOutcome{.kind = DeleteOutcome::Kind::TokenMismatch}; + return InMemoryBackend::deleteExact(key, token); + } + +private: + String throw_key; + String mismatch_key; +}; + +/// Replaces the durable catalog immediately after returning the first armed catalog read. This +/// distinguishes the immutable cut validated before decommission impersonation from a later mount +/// safety observation without assuming those two decisions share one GET. +class CatalogChangesAfterFirstReadBackend : public InMemoryBackend +{ +public: + using Backend::get; + + void armCatalogReplacement( + const String & key, RefCatalog replacement_, size_t completed_reads_before_replacement = 0) + { + catalog_key = key; + replacement = std::move(replacement_); + reads_to_skip = completed_reads_before_replacement; + armed = true; + } + + bool fired() const { return replacement_fired; } + + std::optional get(const String & key, Range range) override + { + auto got = InMemoryBackend::get(key, range); + if (!armed || replacement_fired || key != catalog_key) + return got; + if (reads_to_skip > 0) + { + --reads_to_skip; + return got; + } + if (!got) + throw std::runtime_error("catalog replacement fixture: catalog is absent"); + + replacement_fired = true; + const PutResult put = InMemoryBackend::putOverwrite( + key, encodeRefCatalog(replacement), got->token, {}); + if (put.outcome != PutOutcome::Done) + throw std::runtime_error("catalog replacement fixture: rewrite conflicted"); + return got; + } + +private: + String catalog_key; + RefCatalog replacement; + size_t reads_to_skip = 0; + bool armed = false; + bool replacement_fired = false; +}; + +std::vector> snapshotPrefixObjects( + InMemoryBackend & backend, const String & prefix) +{ + std::vector> objects; + String cursor; + while (true) + { + const ListPage page = backend.list(prefix, cursor, 1000); + for (const ListedKey & listed : page.keys) + { + const auto got = backend.get(listed.key); + if (!got) + throw std::runtime_error("prefix snapshot fixture: listed object disappeared"); + objects.emplace_back(listed.key, got->bytes, got->token); + } + if (page.next_cursor.empty()) + return objects; + cursor = page.next_cursor; + } +} + +/// Installs a same-UUID successor deterministically in the retirement tail's read/delete window. +/// Once armed, the backend recognizes the admin's clean farewell `putOverwrite`. On the next read of +/// either mutable control object it first captures the value that read observed, then bumps `epoch` +/// and reclaims `mount` with fresh tokens before returning the captured result. Thus the caller holds +/// exactly the stale token it would have obtained immediately before a concurrent restart reclaimed +/// the slot, without threads or sleeps. +class SuccessorReclaimAfterFarewellBackend : public InMemoryBackend +{ +public: + using Backend::get; + using Backend::putOverwrite; + + void armForSuccessorReclaim() { armed = true; } + + std::optional get(const String & key, Range range) override + { + std::optional result = InMemoryBackend::get(key, range); + if (farewell_seen && !successor_injected && (key == mount_key || key == epoch_key)) + injectSuccessor(); + return result; + } + + PutResult putOverwrite( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + const PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + if (armed && key == mount_key && result.outcome == PutOutcome::Done) + { + const MountLease mount = decodeMountLease(bytes); + if (mount.min_active == std::numeric_limits::max()) + farewell_seen = true; + } + return result; + } + + bool successorInjected() const { return successor_injected; } + const Token & successorMountToken() const { return successor_mount_token; } + const Token & successorEpochToken() const { return successor_epoch_token; } + const String & successorMountBytes() const { return successor_mount_bytes; } + const String & successorEpochBytes() const { return successor_epoch_bytes; } + +private: + void injectSuccessor() + { + const auto epoch = InMemoryBackend::get(epoch_key, {}); + const auto mount = InMemoryBackend::get(mount_key, {}); + if (!epoch || !mount) + throw std::runtime_error("successor-reclaim fixture: control object disappeared before reclaim"); + + ServerEpoch epoch_value = decodeServerEpoch(epoch->bytes); + const uint64_t successor_writer_epoch = epoch_value.next_writer_epoch; + ++epoch_value.next_writer_epoch; + successor_epoch_bytes = encodeServerEpoch(epoch_value); + const CasResult epoch_put = InMemoryBackend::casPut( + epoch_key, successor_epoch_bytes, std::optional{epoch->token}, {}); + if (epoch_put.outcome != CasOutcome::Committed) + throw std::runtime_error("successor-reclaim fixture: epoch bump conflicted"); + successor_epoch_token = epoch_put.token; + + MountLease mount_value = decodeMountLease(mount->bytes); + mount_value.writer_epoch = successor_writer_epoch; + ++mount_value.seq; + ++mount_value.started_at_ms; + mount_value.expires_at_ms = mount_value.started_at_ms + 30'000; + mount_value.min_active = 0; + mount_value.gc_fenced = false; + successor_mount_bytes = encodeMountLease(mount_value); + const PutResult mount_put = InMemoryBackend::putOverwrite( + mount_key, successor_mount_bytes, mount->token, {}); + if (mount_put.outcome != PutOutcome::Done) + throw std::runtime_error("successor-reclaim fixture: mount reclaim conflicted"); + successor_mount_token = mount_put.token; + successor_injected = true; + } + + inline static const String mount_key = "p/gc/server-roots/victim/mount"; + inline static const String epoch_key = "p/gc/server-roots/victim/epoch"; + bool armed = false; + bool farewell_seen = false; + bool successor_injected = false; + Token successor_mount_token; + Token successor_epoch_token; + String successor_mount_bytes; + String successor_epoch_bytes; +}; + +/// Recreates the mutable slot objects immediately after decommission successfully deletes `epoch`. +/// This models a same-UUID successor starting in the final retirement window: `owner` remains the +/// unchanged identity anchor, while the successor legitimately creates a fresh `epoch` and `mount`. +class SuccessorReclaimAfterEpochDeleteBackend : public InMemoryBackend +{ +public: + using Backend::get; + using Backend::putOverwrite; + + void armForSuccessorReclaim() { armed = true; } + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + const DeleteOutcome result = InMemoryBackend::deleteExact(key, token); + if (armed && !successor_injected && key == epoch_key + && classifyDeleteOutcome(result) == DeleteClass::Deleted) + { + injectSuccessor(); + } + return result; + } + + bool successorInjected() const { return successor_injected; } + uint64_t ownerRewriteAttempts() const { return owner_rewrite_attempts; } + const Token & successorMountToken() const { return successor_mount_token; } + const Token & successorEpochToken() const { return successor_epoch_token; } + const String & successorMountBytes() const { return successor_mount_bytes; } + const String & successorEpochBytes() const { return successor_epoch_bytes; } + +private: + PutResult putOverwrite( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + if (key == owner_key) + ++owner_rewrite_attempts; + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + + void injectSuccessor() + { + successor_epoch_bytes = encodeServerEpoch(ServerEpoch{.next_writer_epoch = 102}); + const PutResult epoch_put = InMemoryBackend::putIfAbsent(epoch_key, successor_epoch_bytes, {}); + if (epoch_put.outcome != PutOutcome::Done) + throw std::runtime_error("late-successor fixture: epoch recreation conflicted"); + successor_epoch_token = epoch_put.token; + + successor_mount_bytes = encodeMountLease(MountLease{ + .server_uuid = UInt128(0x1234), + .writer_epoch = 101, + .hostname = "successor", + .pid = 42, + .started_at_ms = 1'000, + .seq = 1, + .expires_at_ms = 31'000, + .min_active = 0, + }); + const PutResult mount_put = InMemoryBackend::putIfAbsent(mount_key, successor_mount_bytes, {}); + if (mount_put.outcome != PutOutcome::Done) + throw std::runtime_error("late-successor fixture: mount recreation conflicted"); + successor_mount_token = mount_put.token; + successor_injected = true; + } + + inline static const String mount_key = "p/gc/server-roots/victim/mount"; + inline static const String epoch_key = "p/gc/server-roots/victim/epoch"; + inline static const String owner_key = "p/gc/server-roots/victim/owner"; + bool armed = false; + bool successor_injected = false; + uint64_t owner_rewrite_attempts = 0; + Token successor_mount_token; + Token successor_epoch_token; + String successor_mount_bytes; + String successor_epoch_bytes; +}; + +/// Rewrites the owner anchor after decommission reads it but before its conditional tombstone write. +/// Returning the captured result gives decommission a stale owner token, deterministically modeling +/// the successor race without threads or sleeps. +class SuccessorOwnerRewriteBeforeTombstoneBackend : public InMemoryBackend +{ +public: + using Backend::get; + + void armForSuccessorRewrite() { armed = true; } + + std::optional get(const String & key, Range range) override + { + std::optional result = InMemoryBackend::get(key, range); + if (armed && epoch_deleted && !successor_injected && key == owner_key && result) + { + successor_owner_bytes = encodeOwner(OwnerObject{ + .server_uuid = decodeOwner(result->bytes).server_uuid, + .retired_at_ms = std::nullopt, + }); + const PutResult put = InMemoryBackend::putOverwrite( + owner_key, successor_owner_bytes, result->token, {}); + if (put.outcome != PutOutcome::Done) + throw std::runtime_error("owner-successor fixture: owner rewrite conflicted"); + successor_owner_token = put.token; + successor_injected = true; + } + return result; + } + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + const DeleteOutcome result = InMemoryBackend::deleteExact(key, token); + if (armed && key == epoch_key && classifyDeleteOutcome(result) == DeleteClass::Deleted) + epoch_deleted = true; + return result; + } + + bool successorInjected() const { return successor_injected; } + const Token & successorOwnerToken() const { return successor_owner_token; } + const String & successorOwnerBytes() const { return successor_owner_bytes; } + +private: + inline static const String epoch_key = "p/gc/server-roots/victim/epoch"; + inline static const String owner_key = "p/gc/server-roots/victim/owner"; + bool armed = false; + bool epoch_deleted = false; + bool successor_injected = false; + Token successor_owner_token; + String successor_owner_bytes; +}; + +/// Models an "ambiguous success" on the final owner tombstone write: the conditional overwrite +/// actually lands (InMemoryBackend applies it), but the response is then lost (a transient +/// exception is thrown on the SAME call, exactly as a real SDK timeout after a landed write would +/// look). Before the fix, decommission caught any exception here and reported failure +/// unconditionally; the controlled overwrite must resolve this via a GET (the current bytes match +/// what was intended) and report Committed instead. +class AmbiguousOwnerTombstoneBackend : public InMemoryBackend +{ +public: + using Backend::putOverwrite; + + void armForAmbiguousTombstone() { armed = true; } + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + const PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + if (armed && !fired && key == owner_key && result.outcome == PutOutcome::Done) + { + fired = true; + throw std::runtime_error("ambiguous-tombstone fixture: response lost after the write landed"); + } + return result; + } + +private: + inline static const String owner_key = "p/gc/server-roots/victim/owner"; + bool armed = false; + bool fired = false; +}; + +/// Seed one victim table with `committed` committed refs and `precommits` dangling precommit bindings, +/// via the raw ref-log seeding helpers (fixture idiom of e.g. `gtest_cas_gc_fold.cpp`: `writeManifestRaw` +/// + `publishCommittedTransition`/`addPrecommitTransition` against `victim`'s own backend/layout) -- this +/// fixture only needs the ref-table SHAPE `dropNamespace` erases, not a real build. Precommit bindings +/// are seeded at an artificially high `writer_epoch` so the writer's own stale-precommit sweep (armed +/// unconditionally by this table's recovery, unrelated to decommission -- spec §Clean Up Old Precommits) +/// never reclaims them, in its OWN separate transaction, ahead of `dropNamespace`'s removal. +void makeTableWithRefs(Pool & victim, const String & ns_str, uint64_t committed, uint64_t precommits) +{ + const RootNamespace ns(ns_str); + Backend & backend = victim.backend(); + const Layout & layout = victim.layout(); + + /// Final physical ids are pool-wide. The generic raw-write helper intentionally uses one shared + /// transition sentinel, so this multi-namespace fixture admits a distinct deterministic test life + /// before invoking it; the helper then resolves and preserves that existing catalog identity. + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + const auto existing = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns.string() == ns.string(); }); + if (existing == catalog.catalog.entries.end()) + { + static std::atomic next_test_life{1000}; + CatalogEntry entry; + entry.ns = ns; + entry.state = NsState::Live; + entry.incarnation = UInt128{next_test_life.fetch_add(1)}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + } + + uint64_t last_ref_sequence = 0; + for (uint64_t i = 0; i < committed; ++i) + { + const ManifestRef ref{.writer_epoch = 1, .build_sequence = i + 1, .manifest_ordinal = 1}; + writeManifestRaw(backend, layout, ns, ref, {}); + last_ref_sequence = publishCommittedTransition(backend, layout, ns, "committed_" + std::to_string(i), std::nullopt, ref); + } + for (uint64_t i = 0; i < precommits; ++i) + { + const ManifestRef ref{.writer_epoch = 999999, .build_sequence = i + 1, .manifest_ordinal = 1}; + writeManifestRaw(backend, layout, ns, ref, {}); + last_ref_sequence = addPrecommitTransition(backend, layout, ns, UInt128(1), "precommit_" + std::to_string(i), std::nullopt, ref); + } + + /// Semantic transition helpers already publish `_ckpt`; replace their final checkpoint through + /// the exact token-CAS fixture helper to make this fixture's complete intended state explicit. + replaceRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = last_ref_sequence ? std::optional{RefTxnId{1, last_ref_sequence}} : std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + /// Self-checking: `listRefs` must observe exactly `committed` committed refs before returning. + ASSERT_EQ(victim.listRefs(ns).size(), committed); +} + +/// Pre-precommit manifest debris: a staged manifest body under `ns_str`, at the store's own +/// `writer_epoch`, named by NO owner event -- a build the writer staged and never finished (fixture +/// idiom of `gtest_cas_orphan_manifest_sweep.cpp`'s `EligibleAndUnownedIsDeleted`). `build_sequence = 99` +/// is picked well clear of `makeTableWithRefs`'s own committed/precommit build sequences so it can never +/// collide with a real owned manifest key. Returns the seeded body's `ManifestId` so a caller can target +/// it (e.g. its exact object key) for further fixture setup. +ManifestId seedOrphanManifestBody(Pool & victim, const String & ns_str) +{ + const RootNamespace ns(ns_str); + const ManifestRef ref{.writer_epoch = victim.writerEpoch(), .build_sequence = 99, .manifest_ordinal = 1}; + const ManifestId id = writeManifestRaw(victim.backend(), victim.layout(), ns, ref, {}); + /// EXPECT, not ASSERT: this function returns a value now, and ASSERT_* expands to a bare `return;` + /// -- invalid in a non-void function. + EXPECT_TRUE(victim.backend().head(victim.layout().manifestKey(id)).exists); + return id; +} + +/// THE MANIFEST-DEBRIS DRAIN NO LONGER DELETES, AND THE FIXTURES BELOW SAY SO RATHER THAN WORKING +/// AROUND IT. The drain goes through `sweepNamespace`, which is subject to the §6 deletion premise: a +/// manifest of an epoch-`E` build is deletable only once the namespace's sealed fold cursor sits in an +/// epoch STRICTLY above `E`. Every object in these fixtures -- the table's ref stream, the debris, and +/// the removal transaction decommission itself appends -- lives in ONE writer epoch, and a single-epoch +/// pool cannot satisfy that: any cursor high enough to clear the debris's epoch also sits above the +/// removal record, which would strip the tail-removal protection off the table's real manifests. The +/// two facts are mutually exclusive here, so there is no honest seeding that restores the deletions; +/// the tests assert retention, and the drain's reclaim path returns with registers R2/R3 (Stage B). + +} + +TEST(CASDecommission, RefusesLiveMember) +{ + auto backend = std::make_shared(); + auto victim = openVictim(backend); /// keeps its mount lease unexpired — the member is alive + + expectThrowsCode(ErrorCodes::ABORTED, [&] + { + Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + }); +} + +TEST(CASDecommission, ClaimsDeadMemberAndBumpsEpoch) +{ + auto backend = std::make_shared(); + uint64_t victim_epoch = 0; + { + auto victim = openVictim(backend); + victim_epoch = victim->writerEpoch(); + } /// graceful close: lease stamped already-expired + farewell — the slot is claimable + + auto admin = Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + ASSERT_TRUE(admin != nullptr); + EXPECT_GT(admin->writerEpoch(), victim_epoch); + /// The admin store IS the victim server root now (impersonation). + EXPECT_EQ(admin->poolConfig().server_root_id, "victim"); +} + +TEST(CASDecommission, AlwaysRenewsAdminClaimEvenWhenHostDiskIsObserveOnly) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + } /// graceful close: lease stamped already-expired + farewell — the slot is claimable + + /// The calling (host) disk may be observe-only, i.e. its own PoolConfig carries + /// background_watermark = false. The decommission admin claim must renew its lease + /// regardless -- a long drain must not expire midway just because the host mount doesn't + /// run a background renewer for its OWN mount. + auto admin = Pool::openForDecommission( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin", .background_watermark = false}, "victim"); + ASSERT_TRUE(admin != nullptr); + EXPECT_TRUE(admin->poolConfig().background_watermark); +} + +TEST(CASDecommission, RefusesUnknownMember) +{ + auto backend = std::make_shared(); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, [&] + { + Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "never_existed"); + }); +} + +TEST(CASDecommission, SecondConcurrentDecommissionRefused) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + + auto first = Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + expectThrowsCode(ErrorCodes::ABORTED, [&] + { + Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin2"}, "victim"); + }); +} + +TEST(CASDecommission, DuplicateLifeIdRefusesBeforeAnyNamespaceOrSlotMutation) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + const Layout layout("p"); + RefCatalog catalog; + catalog.entries = { + CatalogEntry{.ns = RootNamespace{"victim/a"}, .state = NsState::Live, .incarnation = UInt128{77}}, + CatalogEntry{ + .ns = RootNamespace{"victim/b"}, + .state = NsState::Removing, + .incarnation = UInt128{77}, + .removal_started_round = 1}, + }; + const auto empty_catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(empty_catalog); + ASSERT_EQ(backend->putOverwrite( + layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->token).outcome, PutOutcome::Done); + const auto owner_before = backend->get(layout.ownerKey("victim")); + const auto epoch_before = backend->get(layout.epochKey("victim")); + const auto mount_before = backend->get(layout.mountKey("victim")); + ASSERT_TRUE(owner_before); + ASSERT_TRUE(epoch_before); + ASSERT_TRUE(mount_before); + + EXPECT_THROW(decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"), DB::Exception); + const auto owner_after = backend->get(layout.ownerKey("victim")); + const auto epoch_after = backend->get(layout.epochKey("victim")); + const auto mount_after = backend->get(layout.mountKey("victim")); + ASSERT_TRUE(owner_after); + ASSERT_TRUE(epoch_after); + ASSERT_TRUE(mount_after); + EXPECT_EQ(owner_after->bytes, owner_before->bytes); + EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); + EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(mount_after->bytes, mount_before->bytes); + EXPECT_EQ(mount_after->token, mount_before->token); +} + +TEST(CASDecommission, CatalogCutIsValidatedBeforeImpersonationAndReusedForSelection) +{ + auto backend = std::make_shared(); + { auto victim = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); } + const Layout layout("p"); + + RefCatalog ambiguous; + ambiguous.entries = { + CatalogEntry{.ns = RootNamespace{"other/a"}, .state = NsState::Live, .incarnation = UInt128{77}}, + CatalogEntry{ + .ns = RootNamespace{"other/b"}, + .state = NsState::Removing, + .incarnation = UInt128{77}, + .removal_started_round = 1}, + }; + + const auto owner_before = backend->get(layout.ownerKey("victim")); + const auto epoch_before = backend->get(layout.epochKey("victim")); + const auto mount_before = backend->get(layout.mountKey("victim")); + ASSERT_TRUE(owner_before); + ASSERT_TRUE(epoch_before); + ASSERT_TRUE(mount_before); + backend->armCatalogReplacement(layout.refCatalogKey(), std::move(ambiguous)); + + EXPECT_THROW(decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"), DB::Exception); + ASSERT_TRUE(backend->fired()); + + const auto owner_after = backend->get(layout.ownerKey("victim")); + const auto epoch_after = backend->get(layout.epochKey("victim")); + const auto mount_after = backend->get(layout.mountKey("victim")); + ASSERT_TRUE(owner_after); + ASSERT_TRUE(epoch_after); + ASSERT_TRUE(mount_after); + EXPECT_EQ(owner_after->bytes, owner_before->bytes); + EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); + EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(mount_after->bytes, mount_before->bytes); + EXPECT_EQ(mount_after->token, mount_before->token); +} + +TEST(CASDecommission, NamespaceSelectionUsesThePreImpersonationCut) +{ + auto backend = std::make_shared(); + { auto victim = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); } + const Layout layout("p"); + + RefCatalog later; + later.entries = { + CatalogEntry{.ns = RootNamespace{"victim/late"}, .state = NsState::Live, .incarnation = UInt128{88}}, + }; + backend->armCatalogReplacement(layout.refCatalogKey(), std::move(later)); + + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + ASSERT_TRUE(backend->fired()); + EXPECT_EQ(report.namespaces_removed, 0u) + << "a namespace visible only to mount safety's later observation is outside the validated cut"; + EXPECT_EQ(report.namespaces_already_removed, 0u); +} + +TEST(CASDecommission, SameNameRebirthAfterTheCutIsRefusedWithoutTouchingTheNewLife) +{ + auto backend = std::make_shared(); + { auto victim = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); } + const Layout layout("p"); + const RootNamespace ns{"victim/same"}; + const NamespaceLifeId old_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128{70}); + const NamespaceLifeId new_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128{71}); + + RefCatalog old_catalog; + old_catalog.entries = { + CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = old_life.incarnation}, + }; + const auto empty_catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(empty_catalog); + ASSERT_EQ(backend->putOverwrite( + layout.refCatalogKey(), encodeRefCatalog(old_catalog), empty_catalog->token).outcome, + PutOutcome::Done); + + RefLogTxn new_birth; + new_birth.ns = ns.string(); + new_birth.txn_id = RefTxnId{1, 1}; + new_birth.ops = {namespaceBirthOp()}; + ASSERT_EQ(backend->putIfAbsent( + layout.refLogKey(new_life, new_birth.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(new_birth))).outcome, + PutOutcome::Done); + RefLogTxn new_seal; + new_seal.ns = ns.string(); + new_seal.txn_id = RefTxnId{1, 2}; + new_seal.ops = {epochSealOp()}; + ASSERT_EQ(backend->putIfAbsent( + layout.refLogKey(new_life, new_seal.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(new_seal))).outcome, + PutOutcome::Done); + const auto new_life_before = snapshotPrefixObjects(*backend, layout.namespaceStreamPrefix(new_life)); + + RefCatalog replacement; + replacement.entries = { + CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = new_life.incarnation}, + }; + /// Read 1 captures the immutable selection cut. Read 2 is mount safety; replace immediately + /// after returning that old observation, so the name-only call is the first consumer of the + /// same-name new incarnation. + backend->armCatalogReplacement( + layout.refCatalogKey(), std::move(replacement), /*completed_reads_before_replacement=*/1); + + String refusal; + try + { + (void)decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + } + catch (const DB::Exception & e) + { + refusal = e.message(); + } + EXPECT_NE(refusal.find("changed incarnation after the validated catalog cut"), String::npos) + << refusal; + ASSERT_TRUE(backend->fired()); + EXPECT_EQ(snapshotPrefixObjects(*backend, layout.namespaceStreamPrefix(new_life)), new_life_before) + << "decommission must not append a removal transaction to the post-cut incarnation"; +} + +TEST(CASDecommission, VictimNameMatchesOneCanonicalPathComponent) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + + const RootNamespace neighbor_ns{"victim2/db/t1"}; + { + auto neighbor = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim2"}); + makeTableWithRefs(*neighbor, neighbor_ns.string(), /*committed=*/1, /*precommits=*/0); + } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + EXPECT_EQ(report.namespaces_removed, 0u); + + auto neighbor = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim2"}); + EXPECT_EQ(neighbor->listRefs(neighbor_ns).size(), 1u) + << "decommissioning victim must not select victim2 by raw string prefix"; +} + +TEST(CASDecommission, ErasesAllVictimNamespaces) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + /// Two tables: ns "victim/db/t1" with 2 committed refs, ns "victim/db/t2" with 1 committed + /// ref + 1 stale precommit (fixture idiom of gtest_cas_ref_writer.cpp). + makeTableWithRefs(*victim, "victim/db/t1", /*committed=*/2, /*precommits=*/0); + makeTableWithRefs(*victim, "victim/db/t2", /*committed=*/1, /*precommits=*/1); + } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_EQ(report.srid, "victim"); + EXPECT_EQ(report.namespaces_removed, 2u); + EXPECT_EQ(report.namespaces_already_removed, 0u); + EXPECT_EQ(report.committed_refs_removed, 3u); + EXPECT_EQ(report.precommits_removed, 1u); + EXPECT_EQ(report.edge_deltas_emitted, 4u); + + /// Terminal publication is writer work; exact catalog-row deletion remains GC work. The first + /// command therefore keeps the slot as an ownership anchor, and a retry may retire it only after + /// GC's next invocation drains the completed `Removing` rows. + EXPECT_FALSE(report.warnings.empty()); + EXPECT_FALSE(report.slot_removed); + drainCompletedNamespaceRemovals(backend); + const auto retired = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin2"}, "victim"); + EXPECT_TRUE(retired.warnings.empty()); + EXPECT_TRUE(retired.slot_removed); +} + +/// Task 2 review finding 1: `makeTableWithRefs`'s precommit seed uses an artificially high +/// `writer_epoch` (999999) specifically to dodge the writer's OWN stale-precommit sweep -- which +/// means it never exercised the path a REAL victim precommit takes. A genuine writer stamps +/// `manifest_ref.writer_epoch` from its OWN `liveWriterEpoch()` at precommit time +/// (`PartWriteTxn::precommitAdd`, CasPool.cpp:2087), i.e. the victim's era -- always LOWER than the admin +/// mount's freshly-minted epoch (`openForDecommission` always bumps strictly higher). `appendRefOps` +/// hoists `maybeSweepStalePrecommits` at its top (CasPool.cpp:1716), so without the +/// `skip_stale_precommit_sweep` fix that sweep would reclaim this realistic-epoch precommit in its +/// OWN transaction before `dropNamespace`'s removal transaction ever counts it, leaving +/// `precommits_removed` at 0 for exactly the case that matters. +TEST(CASDecommission, CountsRealisticEpochPrecommit) +{ + auto backend = std::make_shared(); + uint64_t victim_epoch = 0; + { + auto victim = openVictim(backend); + victim_epoch = victim->writerEpoch(); + makeTableWithRefs(*victim, "victim/db/t1", /*committed=*/1, /*precommits=*/0); + + const RootNamespace ns("victim/db/t1"); + /// `build_sequence = 2`: distinct from `makeTableWithRefs`'s committed ref (`build_sequence = 1`) + /// -- a REAL build's `ManifestRef` is unique per build, and a colliding one would trip the ref + /// state machine's "manifest already has a conflicting owner" guard. + const ManifestRef ref{.writer_epoch = victim_epoch, .build_sequence = 2, .manifest_ordinal = 1}; + writeManifestRaw(victim->backend(), victim->layout(), ns, ref, {}); + addPrecommitTransition(victim->backend(), victim->layout(), ns, UInt128(1), "precommit_0", std::nullopt, ref); + } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_EQ(report.namespaces_removed, 1u); + EXPECT_EQ(report.committed_refs_removed, 1u); + EXPECT_EQ(report.precommits_removed, 1u); + EXPECT_EQ(report.edge_deltas_emitted, 2u); +} + +/// Task 2 review finding 2: the `member_decommission` begin/namespace_removed/end events +/// (CasDecommission.cpp) had no assertion at all. Wire a capturing sink (the `gtest_cas_event_log.cpp` +/// idiom) into `decommissionPoolMember` and check the emitted sequence and its per-namespace detail. +TEST(CASDecommission, EmitsMemberDecommissionEvents) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", /*committed=*/1, /*precommits=*/0); + } + + std::vector seen; + (void)decommissionPoolMember(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", + [&](const CasEvent & e) { seen.push_back(e); }); + + std::vector member_events; + for (const auto & e : seen) + if (e.type == CasEventType::MemberDecommission) + member_events.push_back(e); + + ASSERT_EQ(member_events.size(), 3u); + EXPECT_EQ(member_events[0].outcome, "begin"); + EXPECT_EQ(member_events[1].outcome, "namespace_removed"); + EXPECT_EQ(member_events[1].detail.at("namespace"), "victim/db/t1"); + EXPECT_EQ(member_events[1].detail.at("committed"), "1"); + EXPECT_EQ(member_events[1].detail.at("precommits"), "0"); + EXPECT_EQ(member_events[2].outcome, "end"); + EXPECT_EQ(member_events[2].detail.at("namespaces_removed"), "1"); +} + +/// Task 3: the manifest-debris / staging / roots drain phases fill their three `DecommissionReport` +/// counters and leave nothing of the victim behind under `staging/` or `roots/`. +TEST(CASDecommission, DrainsDebrisStagingAndRoots) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + seedOrphanManifestBody(*victim, "victim/db/t1"); + } + /// Foreign staging + mountpoint objects, written raw (no writer machinery needed): the victim's + /// writers are fenced by the claim before decommission ever gets here, so these are ordinary debris, + /// not a live in-flight write. + backend->putIfAbsent("p/staging/victim/upload1.tmp", "x"); + backend->putIfAbsent("p/staging/victim/upload2.tmp", "x"); + backend->putIfAbsent("p/roots/victim/clickhouse_access_check_abc", "x"); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + /// The staging and mountpoint phases are unchanged and still drain completely. The manifest-debris + /// phase retains under the §6 premise (see the note on the helpers above) and reports why, which is + /// what keeps the slot; `RetainsDebrisWhoseEpochSealIsUnconsumed` is that outcome's own test. + EXPECT_EQ(report.manifest_debris_removed, 0u); + EXPECT_EQ(report.staging_objects_removed, 2u); + EXPECT_EQ(report.mountpoint_objects_removed, 1u); + EXPECT_FALSE(report.warnings.empty()) + << "the retained debris is reported, so the incomplete drain is visible"; + + /// Nothing of the victim remains under staging/ or roots/ (scoped LISTs are empty). Those two phases + /// run to completion even though the debris phase retained -- the drain is per-phase, not all-or-nothing. + EXPECT_TRUE(backend->list("p/staging/victim/", "", 10).keys.empty()); + EXPECT_TRUE(backend->list("p/roots/victim/", "", 10).keys.empty()); +} + +/// The §6 deletion premise applies to the decommission drain too, and this pins what that COSTS. With no +/// sealed fold cursor for the victim's namespace — the state of a pool whose GC has not folded past the +/// victim's closed epoch — the drain cannot show the debris is unreferenced, so it RETAINS it, says why +/// in `warnings`, and therefore keeps the slot for a later re-run. Delay, not damage: the objects are +/// untouched and a re-run after GC catches up drains them (`DrainsDebrisStagingAndRoots`). +/// +/// This is the visible edge of a real Stage-A limitation, not a test-only artifact: debris under a +/// namespace GC never folds — the pure pre-precommit orphan, whose whole point is that no ref record was +/// ever appended for it — has no cursor to consume any seal, so the premise retains it indefinitely. +/// Reclaiming it needs the sweep's own rework (registers R2/R3, Stage B), which is why the premise ships +/// as the safety floor and not as the reclaim policy. +TEST(CASDecommission, RetainsDebrisWhoseEpochSealIsUnconsumed) +{ + auto backend = std::make_shared(); + String debris_key; + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + const ManifestId debris_id = seedOrphanManifestBody(*victim, "victim/db/t1"); + debris_key = victim->layout().manifestKey(debris_id); + /// Deliberately NO `seedFoldedPastVictimEpoch` here — that absence is the subject. + } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_EQ(report.manifest_debris_removed, 0u); + EXPECT_TRUE(backend->head(debris_key).exists) + << "the body is retained untouched, not deleted and not corrupted"; + ASSERT_FALSE(report.warnings.empty()) + << "a retained manifest is a visible decision -- the operator must be able to see why the drain " + "did not complete"; + bool named = false; + for (const String & w : report.warnings) + if (w.find(debris_key) != String::npos && w.find("seal") != String::npos) + named = true; + EXPECT_TRUE(named) << "the warning names the object and the premise that retained it"; + EXPECT_FALSE(report.slot_removed) + << "an incomplete drain keeps the slot as the resume anchor, exactly as a per-key failure does"; +} + +/// Task 3 fail-close nuance (spec §core "Fail-close"): a per-object failure in the staging/roots drain +/// -- a thrown exception (a transient hiccup) or a `TokenMismatch` outcome (a "listed but raced" miss) +/// -- must record a warning and let the rest of the sweep proceed, never abort the whole phase or the +/// whole command. One staging object throws, the roots object comes back `TokenMismatch`; the OTHER +/// staging object must still be deleted and counted. +TEST(CASDecommission, PerObjectFailureWarnsAndContinuesDrain) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + } + backend->putIfAbsent("p/staging/victim/upload_ok.tmp", "x"); + backend->putIfAbsent("p/staging/victim/upload_throws.tmp", "x"); + backend->putIfAbsent("p/roots/victim/clickhouse_access_check_abc", "x"); + backend->failWithThrow("p/staging/victim/upload_throws.tmp"); + backend->failWithTokenMismatch("p/roots/victim/clickhouse_access_check_abc"); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_EQ(report.staging_objects_removed, 1u) + << "the OTHER staging object must still be deleted despite the injected failure on its sibling"; + EXPECT_EQ(report.mountpoint_objects_removed, 0u); + EXPECT_EQ(report.warnings.size(), 2u) + << "one warning for the thrown exception, one for the TokenMismatch outcome"; + + EXPECT_FALSE(backend->head("p/staging/victim/upload_ok.tmp").exists) + << "the healthy staging object was actually deleted, not merely skipped"; + EXPECT_TRUE(backend->head("p/staging/victim/upload_throws.tmp").exists) + << "the failing object is left behind (untouched) so a re-run can retry it"; + EXPECT_TRUE(backend->head("p/roots/victim/clickhouse_access_check_abc").exists) + << "TokenMismatch means nothing was actually deleted -- the object survives"; +} + +/// Opaque physical debris carries no logical owner and therefore cannot widen or redirect +/// decommission's catalog-derived victim set. Task 5's ownership-tree janitor owns that debris. +TEST(CASDecommission, LifelessPhysicalKeyCannotRedirectCatalogOwnedDecommission) +{ + auto backend = std::make_shared(); + String lifeless; + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + /// Hand-built: no helper can mint the un-incarnated shape any more. + lifeless = victim->layout().casRefsPrefix() + String("victim/db/t1/_log/") + + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + ASSERT_EQ(backend->putIfAbsent(lifeless, "garbage").outcome, PutOutcome::Done); + } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + EXPECT_EQ(report.namespaces_removed, 1u); + EXPECT_TRUE(backend->head(lifeless).exists) + << "decommission must neither adopt nor delete an unowned physical life key"; +} + +/// The manifest-debris drain honors the same tolerate-and-continue contract as +/// `deleteListedPrefix`: a per-key `deleteExact` failure becomes a warning, while the namespace +/// erasure and subsequent staging drain continue. Protection reads now use opaque physical life +/// prefixes, so a logical-name substring can no longer target an otherwise unlisted namespace. +TEST(CASDecommission, ManifestDebrisDeleteFailureWarnsAndContinues) +{ + auto backend = std::make_shared(); + String debris_key; + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + const ManifestId debris_id = seedOrphanManifestBody(*victim, "victim/db/t1"); + debris_key = victim->layout().manifestKey(debris_id); + } + backend->failWithThrow(debris_key); + backend->putIfAbsent("p/staging/victim/upload_ok.tmp", "x"); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_EQ(report.namespaces_removed, 1u) + << "victim/db/t1's namespace erasure (Task 2) is untouched by either injected failure"; + EXPECT_EQ(report.manifest_debris_removed, 0u); + EXPECT_EQ(report.warnings.size(), 1u) + << "the thrown per-key delete must keep the retirement tail fail-closed"; + EXPECT_EQ(report.staging_objects_removed, 1u) + << "the staging phase still ran to completion after the manifest-debris phase's failures -- " + "the whole command did not abort"; + + EXPECT_TRUE(backend->head(debris_key).exists) + << "the failing object is left behind (untouched) so a re-run can retry it"; +} + +/// GC owns the completed catalog-row deletion. Once it drains the row, a clean decommission retry +/// removes the mutable slot objects and tombstones the owner anchor. +TEST(CASDecommission, RemovesMutableSlotAndRefusesTombstonedRerun) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + } + const auto pending = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + EXPECT_FALSE(pending.slot_removed); + EXPECT_FALSE(pending.warnings.empty()); + drainCompletedNamespaceRemovals(backend); + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin2"}, "victim"); + EXPECT_TRUE(report.slot_removed); + EXPECT_TRUE(report.warnings.empty()); + EXPECT_FALSE(backend->get("p/gc/server-roots/victim/mount").has_value()); + const auto owner = backend->get("p/gc/server-roots/victim/owner"); + ASSERT_TRUE(owner.has_value()); + EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); + EXPECT_FALSE(backend->get("p/gc/server-roots/victim/epoch").has_value()); + + expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] + { + decommissionPoolMember(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a2"}, "victim"); + }); +} + +/// Triage #9: a successor may reclaim the same UUID immediately after the decommission admin writes +/// its farewell. The retirement tail must use the farewell/claimed-epoch tokens captured around that +/// release, delete `mount` first, and stop on its `TokenMismatch`; re-reading current tokens would +/// delete the live successor's control objects and falsely report the slot removed. +TEST(CASDecommission, SuccessorReclaimFencesSlotRetirementTail) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + backend->armForSuccessorReclaim(); + + std::vector seen; + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", + [&](const CasEvent & event) { seen.push_back(event); }); + + ASSERT_TRUE(backend->successorInjected()); + EXPECT_FALSE(report.slot_removed); + ASSERT_EQ(report.warnings.size(), 1u); + EXPECT_NE(report.warnings.front().find("p/gc/server-roots/victim/mount"), String::npos); + EXPECT_NE(report.warnings.front().find("replaced"), String::npos); + + const auto mount = backend->get("p/gc/server-roots/victim/mount"); + ASSERT_TRUE(mount.has_value()); + EXPECT_EQ(mount->token, backend->successorMountToken()); + EXPECT_EQ(mount->bytes, backend->successorMountBytes()); + + const auto epoch = backend->get("p/gc/server-roots/victim/epoch"); + ASSERT_TRUE(epoch.has_value()); + EXPECT_EQ(epoch->token, backend->successorEpochToken()); + EXPECT_EQ(epoch->bytes, backend->successorEpochBytes()); + EXPECT_TRUE(backend->get("p/gc/server-roots/victim/owner").has_value()); + + ASSERT_FALSE(seen.empty()); + EXPECT_EQ(seen.back().outcome, "end"); + EXPECT_EQ(seen.back().detail.at("slot_removed"), "0"); +} + +/// A successor can also restart after both stale mutable objects were deleted but before `owner` is +/// retired. Mere presence of either freshly recreated mutable object must stop owner retirement. +TEST(CASDecommission, SuccessorReclaimAfterEpochDeleteKeepsOwnerAnchor) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + + const String owner_key = "p/gc/server-roots/victim/owner"; + const auto original_owner = backend->get(owner_key); + ASSERT_TRUE(original_owner.has_value()); + backend->armForSuccessorReclaim(); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + ASSERT_TRUE(backend->successorInjected()); + EXPECT_FALSE(report.slot_removed); + EXPECT_FALSE(report.warnings.empty()); + EXPECT_EQ(backend->ownerRewriteAttempts(), 0u); + + const auto owner = backend->get(owner_key); + ASSERT_TRUE(owner.has_value()); + EXPECT_EQ(owner->token, original_owner->token); + EXPECT_EQ(owner->bytes, original_owner->bytes); + + const auto mount = backend->get("p/gc/server-roots/victim/mount"); + ASSERT_TRUE(mount.has_value()); + EXPECT_EQ(mount->token, backend->successorMountToken()); + EXPECT_EQ(mount->bytes, backend->successorMountBytes()); + + const auto epoch = backend->get("p/gc/server-roots/victim/epoch"); + ASSERT_TRUE(epoch.has_value()); + EXPECT_EQ(epoch->token, backend->successorEpochToken()); + EXPECT_EQ(epoch->bytes, backend->successorEpochBytes()); +} + +/// Triage #9 control: absent a successor interleaving, the fenced tail removes both mutable control +/// objects, tombstones the owner anchor, and preserves the existing successful `slot_removed=1` result. +TEST(CASDecommission, FencedSlotRetirementTailRetiresUncontendedSlot) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + EXPECT_FALSE(backend->get("p/gc/server-roots/victim/mount").has_value()); + EXPECT_FALSE(backend->get("p/gc/server-roots/victim/epoch").has_value()); + const auto owner = backend->get("p/gc/server-roots/victim/owner"); + ASSERT_TRUE(owner.has_value()); + EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); +} + +TEST(CASDecommission, SuccessfulDecommissionLeavesTombstonedOwnerAnchor) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + + const String owner_key = "p/gc/server-roots/victim/owner"; + const auto before = backend->get(owner_key); + ASSERT_TRUE(before.has_value()); + EXPECT_FALSE(decodeOwner(before->bytes).retired_at_ms.has_value()); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + const auto after = backend->get(owner_key); + ASSERT_TRUE(after.has_value()); + EXPECT_NE(after->token, before->token); + EXPECT_EQ(decodeOwner(after->bytes).server_uuid, decodeOwner(before->bytes).server_uuid); + EXPECT_TRUE(decodeOwner(after->bytes).retired_at_ms.has_value()); +} + +TEST(CASDecommission, SuccessorOwnerRewriteWinsBeforeTombstone) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + backend->armForSuccessorRewrite(); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + ASSERT_TRUE(backend->successorInjected()); + EXPECT_FALSE(report.slot_removed); + ASSERT_EQ(report.warnings.size(), 1u); + EXPECT_NE(report.warnings.front().find("successor reclaimed"), String::npos); + + const auto owner = backend->get("p/gc/server-roots/victim/owner"); + ASSERT_TRUE(owner.has_value()); + EXPECT_EQ(owner->token, backend->successorOwnerToken()); + EXPECT_EQ(owner->bytes, backend->successorOwnerBytes()); + EXPECT_FALSE(decodeOwner(owner->bytes).retired_at_ms.has_value()); +} + +/// Final whole-branch review finding (Important): +/// a transient exception on the owner tombstone write must not be reported as a hard failure when the +/// write actually landed -- the controlled overwrite resolves this via GET (current bytes already +/// match the intended tombstone) instead of the old bare putOverwrite's "any exception = failure". +TEST(CASDecommission, OwnerTombstoneAmbiguousSuccessResolvesToCommitted) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + backend->armForAmbiguousTombstone(); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(report.slot_removed) << "the ambiguous write actually landed and must resolve to Committed"; + EXPECT_TRUE(report.warnings.empty()); + + const auto owner = backend->get("p/gc/server-roots/victim/owner"); + ASSERT_TRUE(owner.has_value()); + EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); +} + +/// Delegates every op to `inner`, except `deleteExact`: while `armed`, any key starting with +/// `fail_prefix` throws an injected transient failure instead of deleting -- models a real backend +/// transiently failing to delete under one whole prefix. `disarm()` clears the failure (the resume +/// half of `FailedDrainKeepsSlotThenResumes`). Forwards every pure-virtual `Backend` member (the +/// `CasBackend.h` list) to `inner` untouched. +class FailDeletesUnderPrefixBackend : public Backend +{ +public: + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + FailDeletesUnderPrefixBackend(std::shared_ptr inner_, String fail_prefix_) + : inner(std::move(inner_)), fail_prefix(std::move(fail_prefix_)) + { + } + + void disarm() { armed = false; } + + std::optional get(const String & key, Range range) override { return inner->get(key, range); } + std::optional getStream(const String & key, Range range) override { return inner->getStream(key, range); } + HeadResult head(const String & key) override { return inner->head(key); } + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + return inner->putIfAbsent(key, bytes, meta); + } + void publishBlob(const BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + return inner->putOverwrite(key, bytes, expected, meta); + } + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override + { + return inner->casPut(key, bytes, expected, meta); + } + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (armed && key.starts_with(fail_prefix)) + throw Exception(ErrorCodes::S3_ERROR, "injected transient delete failure for {}", key); + return inner->deleteExact(key, token); + } + ListPage list(const String & prefix, const String & cursor, size_t limit) override { return inner->list(prefix, cursor, limit); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + std::shared_ptr inner; + String fail_prefix; + bool armed = true; +}; + +/// Task 4 fail-close: a drain failure under the roots prefix keeps the slot terminated-but-present +/// (`report.slot_removed == false`, the mount object survives as the resume anchor). Once the fault is +/// cleared, a re-run finishes the job: the already-erased namespace is counted as +/// `namespaces_already_removed`, the leftover roots object is finally swept, and the slot is removed. +TEST(CASDecommission, FailedDrainKeepsSlotThenResumes) +{ + auto inner = std::make_shared(); + { + auto victim = Pool::open(inner, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + } + inner->putIfAbsent("p/roots/victim/loose_file", "x"); + + auto failing = std::make_shared(inner, "p/roots/victim/"); + const auto first = decommissionPoolMember( + failing, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim"); + EXPECT_FALSE(first.warnings.empty()); + EXPECT_FALSE(first.slot_removed); + EXPECT_TRUE(inner->get("p/gc/server-roots/victim/mount").has_value()) + << "slot kept -- resume anchor"; + + failing->disarm(); + const auto second = decommissionPoolMember( + failing, PoolConfig{.pool_prefix = "p", .server_root_id = "a2"}, "victim"); + EXPECT_FALSE(second.warnings.empty()); + EXPECT_FALSE(second.slot_removed); + EXPECT_EQ(second.namespaces_already_removed, 1u); + EXPECT_EQ(second.mountpoint_objects_removed, 1u); + + drainCompletedNamespaceRemovals(inner); + const auto third = decommissionPoolMember( + failing, PoolConfig{.pool_prefix = "p", .server_root_id = "a3"}, "victim"); + EXPECT_TRUE(third.warnings.empty()); + EXPECT_TRUE(third.slot_removed); +} + +/// Task 4 fail-close, manifest-debris variant (review follow-up: the plan's own example only exercises +/// a roots-phase failure). A per-key `deleteExact` throw inside the manifest-debris drain must ALSO +/// keep the slot: `report.slot_removed == false`, the mount object survives, and once the injected +/// failure is cleared a re-run drains the leftover debris and removes the slot. +TEST(CASDecommission, ManifestDebrisFailureKeepsSlotThenResumes) +{ + auto backend = std::make_shared(); + String debris_key; + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + const ManifestId debris_id = seedOrphanManifestBody(*victim, "victim/db/t1"); + debris_key = victim->layout().manifestKey(debris_id); + } + backend->failWithThrow(debris_key); + + const auto first = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim"); + EXPECT_FALSE(first.warnings.empty()); + EXPECT_FALSE(first.slot_removed); + EXPECT_EQ(first.manifest_debris_removed, 0u); + EXPECT_TRUE(backend->get("p/gc/server-roots/victim/mount").has_value()) + << "slot kept -- resume anchor"; + EXPECT_TRUE(backend->head(debris_key).exists) + << "the failing object is left behind (untouched) so a re-run can retry it"; + + /// COVERAGE LOST HERE, DELIBERATELY NAMED. Before the §6 premise, clearing the injected failure let + /// a re-run drain the debris and retire the slot, which is what proved the per-key fail-close path + /// RESUMES rather than merely refuses. Under the premise the sweep never reaches `deleteExact` for + /// this body at all (single-epoch pool -- see the note on the helpers above), so disarming changes + /// nothing and the resume half of this test is no longer expressible. What survives is the half that + /// still has a mechanism: the slot stays kept across the re-run, and the object stays untouched. + /// The resume assertion comes back with the drain's reclaim path (registers R2/R3, Stage B). + backend->disarm(); + const auto second = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a2"}, "victim"); + EXPECT_EQ(second.namespaces_already_removed, 1u); + EXPECT_EQ(second.manifest_debris_removed, 0u); + EXPECT_FALSE(second.slot_removed); + EXPECT_TRUE(backend->head(debris_key).exists); + EXPECT_TRUE(backend->get("p/gc/server-roots/victim/mount").has_value()) + << "the slot is still the resume anchor -- nothing was retired against unreclaimed debris"; +} + +/// Task 5 (Task-1 carry-forward, escalated by review): preserve recovery from the legacy partial +/// hand-cleanup shape where owner and epoch are absent but the mount lease remains. Triage #9 changed +/// new retirements to delete `mountKey`/`epochKey` and tombstone `ownerKey`, so the current tail no +/// longer creates this shape, but `openForDecommission`'s owner-anchor-absent + +/// mount-lease-present fallback ("partial hand-cleanup: adopt from the lease", `CasPool.cpp`) remains +/// compatibility-critical for slots left by older binaries or manual repair. +/// +/// `claimOwnerOrThrow` (`CasServerRoot.cpp`) gates the owner-absent path a SECOND, stricter way: the +/// same catalog cut must name no `Creating`, `Live` or `Removing` namespace under this canonical root, +/// and the name-bearing `cas/manifests//` and `roots//` families must be empty. Opaque +/// stream/state debris cannot be attributed to a server root and is deliberately inert. This test +/// therefore uses a victim with NO namespaces at all: identity persisted +/// (mount/owner/epoch exist from a real graceful close), data subtree genuinely empty -- the exact +/// precondition the fallback is designed for. Simulate the crash directly: claim the slot once (exactly +/// `decommissionPoolMember`'s own first step), let it close gracefully (the mount-lease keeper's +/// farewell stamp, same as a real `admin.reset()`), then manually strike `epochKey`+`ownerKey`, leaving +/// `mountKey`. A `decommissionPoolMember` re-run must resolve identity via the mount-lease fallback and +/// finish retiring the slot; a further re-run then sees the tombstone and refuses to resume it. +TEST(CASDecommission, MidRetirementCrashResumesViaMountLeaseFallback) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } /// identity only -- no namespace, so the subtree stays empty + + const Layout layout("p"); + /// Claim the slot once, exactly as `decommissionPoolMember`'s own first step would -- this (re)writes + /// fresh epoch/owner/mount control objects. Closing gracefully (scope exit) stamps the mount lease's + /// farewell, matching what a real slot retirement's `admin.reset()` does right before its delete loop. + { + auto admin = Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "chk"}, "victim"); + } + + /// Manually strike epoch + owner, leaving the mount -- the legacy partial hand-cleanup shape. + for (const String & key : {layout.epochKey("victim"), layout.ownerKey("victim")}) + { + const auto head = backend->head(key); + ASSERT_TRUE(head.exists); + backend->deleteExact(key, head.token); + } + ASSERT_FALSE(backend->get(layout.epochKey("victim")).has_value()); + ASSERT_FALSE(backend->get(layout.ownerKey("victim")).has_value()); + ASSERT_TRUE(backend->get(layout.mountKey("victim")).has_value()) + << "the mount lease must survive -- it is the resume anchor the fallback reads"; + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a2"}, "victim"); + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_EQ(report.namespaces_removed, 0u); + EXPECT_TRUE(report.slot_removed); + EXPECT_FALSE(backend->get(layout.epochKey("victim")).has_value()); + const auto owner = backend->get(layout.ownerKey("victim")); + ASSERT_TRUE(owner.has_value()); + EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); + EXPECT_FALSE(backend->get(layout.mountKey("victim")).has_value()); + + expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] + { + decommissionPoolMember(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a3"}, "victim"); + }); +} diff --git a/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp b/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp new file mode 100644 index 000000000000..a674f13667b6 --- /dev/null +++ b/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp @@ -0,0 +1,330 @@ +#include "cas_test_helpers.h" + +#include +#include + +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +using namespace DB; +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +PoolPtr openVictim(const std::shared_ptr & backend) +{ + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); +} + +CatalogEntry catalogEntry(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; }); + if (it == catalog.entries.end()) + throw Exception(ErrorCodes::CORRUPTED_DATA, "fixture catalog entry '{}' is absent", ns.string()); + return *it; +} + +void makeRemoving(Backend & backend, const Layout & layout, const CatalogEntry & live) +{ + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find(next.entries.begin(), next.entries.end(), live); + if (it == next.entries.end()) + throw Exception(ErrorCodes::CORRUPTED_DATA, "fixture exact Live row changed"); + it->state = NsState::Removing; + it->removal_started_round = 0; + return next; + }); +} + +bool slotObjectExists(Backend & backend, const String & leaf) +{ + return backend.head("p/gc/server-roots/victim/" + leaf).exists; +} + +class AddVictimEntryDuringRootDrainBackend final : public InMemoryBackend +{ +public: + void arm() { armed = true; } + bool fired() const { return added; } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + if (armed && !added && prefix == "p/roots/victim/" && cursor.empty()) + { + added = true; + CasRefCatalog::casAdmitEntry( + *this, Layout("p"), 1, + CatalogEntry{ + .ns = RootNamespace("victim/db/late"), + .state = NsState::Live, + .incarnation = UInt128{707}}); + } + return page; + } + +private: + bool armed = false; + bool added = false; +}; + +/// Admits the late catalog entry between the retirement tail's two exact catalog reads +/// (`retirement_catalog_cut`, then `fresh_retirement_catalog`), never before. The mountpoint drain's +/// `list("p/roots/victim/", ...)` is the last LIST call in `decommissionPoolMember` before either +/// read, so it orders the two `get("p/cas/ref_catalog")` calls that follow it: the first is +/// `retirement_catalog_cut`, the second is `fresh_retirement_catalog`. Mutating on the second call +/// makes that read observe a catalog the first read did not. +class MutateCatalogBetweenRetirementReadsBackend final : public InMemoryBackend +{ +public: + void arm() { armed = true; } + bool fired() const { return added; } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + if (armed && !past_mountpoint_drain && prefix == "p/roots/victim/" && cursor.empty()) + past_mountpoint_drain = true; + return page; + } + + std::optional get(const String & key, Range range) override + { + if (armed && past_mountpoint_drain && !added && key == "p/cas/ref_catalog") + { + if (!seen_retirement_catalog_cut) + seen_retirement_catalog_cut = true; + else + { + added = true; + CasRefCatalog::casAdmitEntry( + *this, Layout("p"), 1, + CatalogEntry{ + .ns = RootNamespace("victim/db/late"), + .state = NsState::Live, + .incarnation = UInt128{707}}); + } + } + return InMemoryBackend::get(key, range); + } + +private: + bool armed = false; + bool past_mountpoint_drain = false; + bool seen_retirement_catalog_cut = false; + bool added = false; +}; + +TEST(CASDecommissionCatalogDuties, RemovingWithoutCheckpointIsCorruptionAndKeepsSlot) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + const CatalogEntry live{ + .ns = RootNamespace("victim/db/missing_ckpt"), + .state = NsState::Live, + .incarnation = UInt128{701}}; + CasRefCatalog::casAdmitEntry( + *backend, victim->layout(), victim->poolConfig().gc_shards, + live); + makeRemoving(*backend, victim->layout(), live); + } + + expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] + { + (void)decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + }); + + EXPECT_TRUE(slotObjectExists(*backend, "owner")); + EXPECT_TRUE(slotObjectExists(*backend, "epoch")); + EXPECT_TRUE(slotObjectExists(*backend, "mount")); + EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/missing_ckpt")).state, + NsState::Removing); +} + +TEST(CASDecommissionCatalogDuties, RemovingWithCheckpointResumesTerminalAndKeepsSlotForGc) +{ + auto backend = std::make_shared(); + const RootNamespace ns("victim/db/pending_terminal"); + std::optional life; + { + auto victim = openVictim(backend); + life = victim->namespaceLife(ns); + const CatalogEntry live = catalogEntry(*backend, victim->layout(), ns); + makeRemoving(*backend, victim->layout(), live); + ASSERT_TRUE(backend->head(victim->layout().refCkptKey(*life)).exists); + ASSERT_TRUE(backend->list(victim->layout().namespaceStreamPrefix(*life), "", 100).keys.empty()); + } + + std::atomic wake_requests{0}; + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", {}, + [&] { wake_requests.fetch_add(1); }); + + EXPECT_EQ(report.namespaces_already_removed, 1u); + EXPECT_EQ(wake_requests.load(), 1u); + EXPECT_FALSE(report.slot_removed); + EXPECT_FALSE(report.warnings.empty()); + EXPECT_TRUE(slotObjectExists(*backend, "owner")); + + const ListPage stream = backend->list(Layout("p").namespaceStreamPrefix(*life), "", 100); + ASSERT_EQ(stream.keys.size(), 1u); + const auto parsed = Layout("p").parseRefObjectKey(stream.keys.front().key); + ASSERT_TRUE(parsed); + const auto body = backend->get(stream.keys.front().key); + ASSERT_TRUE(body); + const RefLogTxn terminal = decodeRefLogTxn( + openObject(FormatId::RefLog, body->bytes), ns.string(), parsed->txn_id); + ASSERT_EQ(terminal.ops.size(), 2u); + EXPECT_EQ(terminal.ops.front().kind, RefOpKind::NamespaceBirth); + EXPECT_EQ(terminal.ops.back().kind, RefOpKind::RemoveNamespace); +} + +TEST(CASDecommissionCatalogDuties, PartialRemovalProgressStillWakesGcWhenLaterNamespaceFails) +{ + auto backend = std::make_shared(); + const RootNamespace progressed_ns("victim/db/a_progressed"); + const RootNamespace broken_ns("victim/db/z_missing_ckpt"); + std::optional progressed_life; + { + auto victim = openVictim(backend); + progressed_life = victim->namespaceLife(progressed_ns); + const CatalogEntry progressed_live = catalogEntry(*backend, victim->layout(), progressed_ns); + makeRemoving(*backend, victim->layout(), progressed_live); + + const CatalogEntry broken_live{ + .ns = broken_ns, + .state = NsState::Live, + .incarnation = UInt128{713}}; + CasRefCatalog::casAdmitEntry( + *backend, victim->layout(), victim->poolConfig().gc_shards, broken_live); + makeRemoving(*backend, victim->layout(), broken_live); + ASSERT_FALSE(backend->head(victim->layout().refCkptKey( + NamespaceLifeId::fromCatalogEntry(broken_ns, broken_live.incarnation))).exists); + } + + std::atomic wake_requests{0}; + expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] + { + (void)decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", {}, + [&] { wake_requests.fetch_add(1); }); + }); + + EXPECT_EQ(wake_requests.load(), 1u) + << "progress already made for an earlier life must wake GC even when a later life fails closed"; + EXPECT_TRUE(slotObjectExists(*backend, "owner")); + const ListPage progressed_stream + = backend->list(Layout("p").namespaceStreamPrefix(*progressed_life), "", 100); + ASSERT_EQ(progressed_stream.keys.size(), 1u); +} + +TEST(CASDecommissionCatalogDuties, VictimEntryAppearingBeforeTheOwnershipCutKeepsSlot) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + backend->arm(); + + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(backend->fired()); + EXPECT_FALSE(report.slot_removed); + ASSERT_FALSE(report.warnings.empty()); + EXPECT_NE(report.warnings.front().find("pool member decommission underway: 1 namespace(s)"), String::npos) + << report.warnings.front(); + EXPECT_TRUE(slotObjectExists(*backend, "owner")); + EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); +} + +TEST(CASDecommissionCatalogDuties, CatalogTokenMovedBetweenOwnershipCutAndRetirementKeepsSlot) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + backend->arm(); + + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(backend->fired()); + EXPECT_FALSE(report.slot_removed); + ASSERT_FALSE(report.warnings.empty()); + EXPECT_NE(report.warnings.front().find("catalog changed after the victim ownership check"), String::npos) + << report.warnings.front(); + EXPECT_TRUE(slotObjectExists(*backend, "owner")); + EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); +} + +TEST(CASDecommissionCatalogDuties, FoldedTerminalRemainsGcOwnedAndOnlyRequestsAnotherRound) +{ + auto backend = std::make_shared(); + const RootNamespace ns("victim/db/folded_terminal"); + std::optional life; + std::vector stream_before; + { + PoolConfig config{ + .pool_prefix = "p", + .server_root_id = "victim", + .gc_fold_threshold = 1, + .gc_fold_max_defer_rounds = 0}; + auto victim = Pool::open(backend, config); + life = victim->namespaceLife(ns); + victim->putNamespaceFile(*life, "format_version.txt", "1\n"); + victim->dropNamespace(ns); + + Gc gc(victim, UInt128{811}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + ASSERT_EQ(catalogEntry(*backend, victim->layout(), ns).state, NsState::Removing); + for (const ListedKey & key : backend->list(victim->layout().namespaceStreamPrefix(*life), "", 100).keys) + stream_before.push_back(key.key); + ASSERT_FALSE(stream_before.empty()); + } + + std::atomic wake_requests{0}; + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", {}, + [&] { wake_requests.fetch_add(1); }); + + EXPECT_EQ(wake_requests.load(), 1u); + EXPECT_EQ(report.namespaces_already_removed, 1u); + EXPECT_FALSE(report.slot_removed); + EXPECT_EQ(catalogEntry(*backend, Layout("p"), ns).state, NsState::Removing); + std::vector stream_after; + for (const ListedKey & key : backend->list(Layout("p").namespaceStreamPrefix(*life), "", 100).keys) + stream_after.push_back(key.key); + EXPECT_EQ(stream_after, stream_before) + << "decommission must not append a second terminal or become a catalog deletion driver"; +} + +TEST(CASDecommissionCatalogDuties, OpaqueLifeDebrisWithoutCatalogOwnershipDoesNotBlockRetirement) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } + const Layout layout("p"); + const NamespaceLifeId dead_life + = NamespaceLifeId::fromCatalogEntry(RootNamespace("historical/name"), UInt128{709}); + const String debris_key = layout.refCkptKey(dead_life); + ASSERT_EQ(backend->putIfAbsent(debris_key, "debris").outcome, PutOutcome::Done); + + const DecommissionReport report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + EXPECT_TRUE(backend->head(debris_key).exists); +} + +} diff --git a/src/Disks/tests/gtest_cas_detached_work.cpp b/src/Disks/tests/gtest_cas_detached_work.cpp new file mode 100644 index 000000000000..a35e3a5adf31 --- /dev/null +++ b/src/Disks/tests/gtest_cas_detached_work.cpp @@ -0,0 +1,825 @@ +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace ProfileEvents +{ +extern const Event CASDetachedWorkDrainTimeouts; +} + +namespace +{ + +/// A gate a test opens explicitly, so a task can be held in flight without a sleep. +struct Gate +{ + void wait() + { + std::unique_lock lock(m); + cv.wait(lock, [this] { return open_; }); + } + void open() + { + std::lock_guard lock(m); + open_ = true; + cv.notify_all(); + } + std::mutex m; + std::condition_variable cv; + bool open_ = false; +}; + +/// Completes the watched first `GET`, then withholds its return so teardown can latch before the +/// helper is able to issue its next raw request. +class BetweenRecoveryGetsBackend : public DB::Cas::tests::OrderedFaultBackend +{ +public: + using DB::Cas::tests::OrderedFaultBackend::get; + + void armBetweenGets(String first_key_, std::shared_ptr first_completed_, std::shared_ptr release_first_) + { + first_key = std::move(first_key_); + first_completed = std::move(first_completed_); + release_first = std::move(release_first_); + armed.store(true); + } + + std::optional get(const String & key, Range range) override + { + auto result = DB::Cas::tests::OrderedFaultBackend::get(key, range); + if (key == first_key && armed.exchange(false)) + { + first_completed->open(); + release_first->wait(); + } + return result; + } + +private: + String first_key; + std::shared_ptr first_completed; + std::shared_ptr release_first; + std::atomic armed{false}; +}; + +/// Identifies recovery's final authority read without changing the recovery implementation: after the +/// recovered-frontier CAS, its first checkpoint `GET` verifies that contribution and its second is the +/// final authority read immediately preceding materialization. +class FinalAuthorityBackend : public DB::Cas::tests::OrderedFaultBackend +{ +public: + using DB::Cas::tests::OrderedFaultBackend::casPut; + using DB::Cas::tests::OrderedFaultBackend::get; + + void armFinalAuthorityRead(String checkpoint_key_) + { + checkpoint_key = std::move(checkpoint_key_); + checkpoint_cas_committed.store(false); + gets_after_checkpoint_cas.store(0); + final_authority_returned.store(false); + armed.store(true); + } + + std::optional get(const String & key, Range range) override + { + auto result = DB::Cas::tests::OrderedFaultBackend::get(key, range); + if (armed.load() && checkpoint_cas_committed.load() && key == checkpoint_key + && gets_after_checkpoint_cas.fetch_add(1) + 1 == 2) + final_authority_returned.store(true); + return result; + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + CasResult result = DB::Cas::tests::OrderedFaultBackend::casPut(key, bytes, expected, meta); + if (armed.load() && key == checkpoint_key && result.outcome == CasOutcome::Committed) + checkpoint_cas_committed.store(true); + return result; + } + + bool finalAuthorityReturned() const { return final_authority_returned.load(); } + +private: + String checkpoint_key; + std::atomic armed{false}; + std::atomic checkpoint_cas_committed{false}; + std::atomic gets_after_checkpoint_cas{0}; + std::atomic final_authority_returned{false}; +}; + +CasRequestBudget oneAttemptBudget() +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 100; + return budget; +} + +/// A ledger-level fixture keeps the real detached publisher but injects its already-public mount-fence +/// callback. That callback is the existing deterministic boundary after recovery materialized its +/// result and before `installRecoveryResult`. +class ManualDetachedLedger +{ +public: + ManualDetachedLedger() + : backend(std::make_shared()) + , ledger( + backend, + layout, + RefLedgerConfig{ + .server_root_id = "test", + .gc_shards = 1, + .snapshot_log_count_threshold = 0, + .snapshot_log_bytes_threshold = 1ULL << 40, + .snapshot_publish_backoff_initial_ms = 0, + .snapshot_publish_backoff_max_ms = 0}, + event_sink, + oneAttemptBudget(), + "test", + [] { return uint64_t{0}; }, + [] { return uint64_t{1}; }, + [] { return true; }, + [] { return uint64_t{1}; }, + [this](uint64_t) + { + if (backend->finalAuthorityReturned() && !final_install_gate_claimed.exchange(true)) + { + final_install_reached.open(); + release_final_install.wait(); + } + }, + [] { return uint64_t{0}; }, + [] { return true; }, + [](const String &, const String &, const std::optional &) {}, + [this](std::function task) + { + std::lock_guard lock(tasks_mutex); + tasks.push_back(std::move(task)); + return true; + }, + {}, + [](const RootNamespace &) {}) + { + CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + } + + std::function takeDetachedTask() + { + std::lock_guard lock(tasks_mutex); + if (tasks.empty()) + throw std::logic_error("ManualDetachedLedger: no queued detached task"); + auto task = std::move(tasks.front()); + tasks.pop_front(); + return task; + } + + void latchStop() + { + std::lock_guard lock(registry->mutex); + registry->stopping = true; + registry->cv.notify_all(); + } + + std::shared_ptr backend; + Layout layout{"p"}; + CasEventSink event_sink; + std::shared_ptr registry = std::make_shared(); + Gate final_install_reached; + Gate release_final_install; + CasRefLedger ledger; + +private: + std::mutex tasks_mutex; + std::deque> tasks; + std::atomic final_install_gate_claimed{false}; +}; + +/// Spin until the drain has latched `stopping`. Bounded so a broken implementation fails the test +/// instead of hanging it. +void awaitStopLatched(const PoolPtr & store) +{ + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (!store->detachedWorkStoppingForTest()) + { + ASSERT_LT(std::chrono::steady_clock::now(), deadline) << "the drain never latched `stopping`"; + std::this_thread::yield(); + } +} + +PoolPtr openPlainPool(const std::shared_ptr & backend, PoolConfig config = {}) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + return Pool::open(backend, config); +} + +std::shared_ptr openTestStorage(bool tiny_budget = false) +{ + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "cas_detached_work_scratch"); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + if (tiny_budget) + { + storage->poolForTest()->setDetachedDrainDeadlineBudgetForTest( + /*attempt_timeout_ms=*/10, /*lease_safety_margin_ms=*/10); + } + return storage; +} + +/// A pool where any nonempty tail is over-threshold, so every mutation auto-dispatches a publish. +PoolPtr openPublishingPool(const std::shared_ptr & backend, + PoolConfig config = {}) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.snapshot_log_count_threshold = 0; + config.snapshot_log_bytes_threshold = 1ULL << 40; + /// One attempt, so a faulted PUT resolves to a definite non-committed outcome with no internal + /// retry loop and no wall-clock wait -- the same budget the snapshot-ordering suite uses. + config.cas_request_budget.max_attempts = 1; + config.cas_request_budget.attempt_timeout_ms = 100; + config.cas_request_budget.operation_deadline_ms = 5000; + config.cas_request_budget.lease_safety_margin_ms = 100; + return Pool::open(backend, config); +} + +/// The same one-transaction publish every other ref suite drives, so a namespace reaches `Live` through +/// the REAL append lane (which is also what creates its `_ckpt`). +RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const String & ref, uint64_t ordinal) +{ + return store->appendRefOps(ns, MutationScope::ref(ref), + [&ref, ordinal](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(DB::Cas::tests::namespaceBirthOp()); + for (const RefOp & op : DB::Cas::tests::publishCommittedOps(ref, ManifestRef{1, ordinal, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +RefTxnId publishRef(CasRefLedger & ledger, const RootNamespace & ns, const String & ref, uint64_t ordinal) +{ + return ledger.appendRefOps(ns, MutationScope::ref(ref), + [&ref, ordinal](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(DB::Cas::tests::namespaceBirthOp()); + for (const RefOp & op : DB::Cas::tests::publishCommittedOps(ref, ManifestRef{1, ordinal, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +/// Leaves the runtime in `NeedsRecovery` while its first background publisher is held after capture. +/// Releasing that publisher consumes the armed snapshot failure; zero backoff then makes settlement +/// redispatch the real token-carrying publisher, whose first action is recovery of this exact runtime. +void preparePendingRecoveryPublisher( + const PoolPtr & store, + const std::shared_ptr & backend, + const RootNamespace & ns, + const std::shared_ptr & first_publisher_captured, + const std::shared_ptr & release_first_publisher, + String & ckpt_key) +{ + auto capture_calls = std::make_shared>(0); + store->setSnapshotAfterCaptureHookForTest( + [capture_calls, first_publisher_captured, release_first_publisher] + { + if (capture_calls->fetch_add(1) != 0) + return; + first_publisher_captured->open(); + release_first_publisher->wait(); + }); + + backend->armPutFailure("_snap/", 1); + ASSERT_NO_THROW(publishRef(store, ns, "ref_1", 1)); + first_publisher_captured->wait(); + + const auto life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + ASSERT_TRUE(life); + ckpt_key = store->layout().refCkptKey(*life); + + /// The log lands before the checkpoint conflict, leaving a real unfrontiered durable transaction. + backend->armCasConflict(ckpt_key, 100); + EXPECT_ANY_THROW(store->dropRef(ns, "ref_1")); + backend->armCasConflict(ckpt_key, 0); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); +} + +} + +/// A zero in-flight count must mean no tracked task still holds the pool. The hook fires at the exact +/// boundary between releasing the lease's pool reference and decrementing the count, so an +/// implementation that decrements first is caught HERE rather than by a racy post-hoc check. +TEST(CASDetachedWork, LeaseReleasesPoolBeforeDecrementing) +{ + auto backend = std::make_shared(); + + std::weak_ptr weak; + std::atomic use_count_at_boundary{-1}; + + PoolConfig config; + config.detached_lease_release_hook_for_test + = [&use_count_at_boundary, &weak] { use_count_at_boundary.store(weak.use_count()); }; + + auto store = openPlainPool(backend, config); + weak = store; + + ASSERT_TRUE(store->tryDispatchDetached([](DetachedStopToken) {})); + ASSERT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/10000)); + + EXPECT_EQ(store->detachedWorkInFlightForTest(), 0u); + EXPECT_EQ(use_count_at_boundary.load(), 1L) + << "at the release boundary the task still held a pool reference: the lease decremented " + "before releasing it, so a zero count does not imply the pool is free"; + EXPECT_EQ(weak.use_count(), 1L); +} + +/// The drain must not RETURN while a task is still running. Asserted as "the drain is still blocked +/// while the task is held" -- the only formulation that does not race the task's completion. +TEST(CASDetachedWork, DrainDoesNotReturnWhileWorkIsInFlight) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + ASSERT_TRUE(store->tryDispatchDetached([entered, release](DetachedStopToken) + { + entered->open(); + release->wait(); + })); + entered->wait(); + + auto drain = std::async(std::launch::async, + [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/60000); }); + + awaitStopLatched(store); + EXPECT_EQ(drain.wait_for(std::chrono::seconds(2)), std::future_status::timeout) + << "the drain returned while a tracked task was still in flight"; + + release->open(); + EXPECT_TRUE(drain.get()); + EXPECT_EQ(store->detachedWorkInFlightForTest(), 0u); +} + +/// `shutdown` must not RETURN while tracked detached work is in flight. +TEST(CASDetachedWork, ShutdownDoesNotReturnWhileWorkIsInFlight) +{ + auto storage = openTestStorage(); + auto pool = storage->poolForTest(); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + ASSERT_TRUE(pool->tryDispatchDetached([entered, release](DetachedStopToken) + { + entered->open(); + release->wait(); + })); + entered->wait(); + + auto done = std::async(std::launch::async, [&storage] { storage->shutdown(); }); + awaitStopLatched(pool); + EXPECT_EQ(done.wait_for(std::chrono::seconds(2)), std::future_status::timeout) + << "shutdown returned while tracked detached work was still in flight"; + + release->open(); + done.get(); +} + +/// A storage destroyed WITHOUT `shutdown` must still drain: nothing prevents that path today. +TEST(CASDetachedWork, ImplicitDestructionDrains) +{ + auto storage = openTestStorage(); + auto pool = storage->poolForTest(); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + std::atomic finished{false}; + ASSERT_TRUE(pool->tryDispatchDetached([entered, release, &finished](DetachedStopToken) + { + entered->open(); + release->wait(); + finished.store(true); + })); + entered->wait(); + + auto destroyed = std::async(std::launch::async, [&storage] { storage.reset(); }); + awaitStopLatched(pool); + release->open(); + destroyed.get(); + + EXPECT_TRUE(finished.load()); + EXPECT_EQ(pool->detachedWorkInFlightForTest(), 0u); +} + +/// A storage destroyed AFTER `shutdown` must find nothing to do rather than waiting a second deadline. +TEST(CASDetachedWork, TeardownHelperIsIdempotent) +{ + auto storage = openTestStorage(); + storage->shutdown(); + + const auto started = std::chrono::steady_clock::now(); + storage.reset(); + EXPECT_LT(std::chrono::steady_clock::now() - started, std::chrono::seconds(1)) + << "the second teardown repeated the wait instead of finding nothing to do"; +} + +/// The timeout path must be OBSERVABLE. Without this the increment could be missing entirely and every +/// other test here would stay green, because they all exercise successful drains. +TEST(CASDetachedWork, ExpiredDrainIncrementsTheTimeoutCounter) +{ + auto storage = openTestStorage(/*tiny_budget=*/true); + auto pool = storage->poolForTest(); + + /// A task that deliberately IGNORES its token, standing in for work that cannot be interrupted. + auto release = std::make_shared(); + auto entered = std::make_shared(); + ASSERT_TRUE(pool->tryDispatchDetached([entered, release](DetachedStopToken) + { + entered->open(); + release->wait(); + })); + entered->wait(); + + const auto before = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts] + .load(std::memory_order_relaxed); + storage->shutdown(); + const auto after = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts] + .load(std::memory_order_relaxed); + EXPECT_EQ(after - before, 1u); + + release->open(); +} + +/// A task must see the stop that is already latched. The handshake matters: releasing the task before +/// the drain latches would make a CORRECT implementation record `false`. +TEST(CASDetachedWork, TaskObservesStopTokenOnceLatched) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + std::atomic saw_stop{false}; + + ASSERT_TRUE(store->tryDispatchDetached([entered, release, &saw_stop](DetachedStopToken token) + { + entered->open(); + release->wait(); + saw_stop.store(token.stopping()); + })); + entered->wait(); + + auto drain = std::async(std::launch::async, + [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/60000); }); + awaitStopLatched(store); + release->open(); + + EXPECT_TRUE(drain.get()); + EXPECT_TRUE(saw_stop.load()); +} + +/// After stopping, no new detached work may be created. +TEST(CASDetachedWork, DispatchIsRefusedAfterStop) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + + ASSERT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/10000)); + EXPECT_FALSE(store->tryDispatchDetached([](DetachedStopToken) {})); + EXPECT_EQ(store->detachedWorkInFlightForTest(), 0u); +} + +/// A dispatch that cannot allocate must leave the count untouched, not stranded above zero. +TEST(CASDetachedWork, FailedAdmissionLeavesNoStrandedCount) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.detached_dispatch_fault_for_test = DetachedDispatchFault::ThrowBeforeLaunch; + auto store = openPlainPool(backend, config); + + EXPECT_ANY_THROW(store->tryDispatchDetached([](DetachedStopToken) {})); + EXPECT_EQ(store->detachedWorkInFlightForTest(), 0u); + EXPECT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/1000)) + << "a stranded count makes every later drain run to its full deadline"; +} + +/// A launch that fails after admission must roll the count back, and must not throw at its caller. +TEST(CASDetachedWork, FailedLaunchRollsBackAndDoesNotThrow) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.detached_dispatch_fault_for_test = DetachedDispatchFault::RefuseLaunch; + auto store = openPlainPool(backend, config); + + EXPECT_NO_THROW(EXPECT_FALSE(store->tryDispatchDetached([](DetachedStopToken) {}))); + EXPECT_EQ(store->detachedWorkInFlightForTest(), 0u); + EXPECT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/1000)); +} + +/// A dispatch that fails must not fail the mutation that triggered it, and must not strand the +/// publisher's single-flight reservation. +TEST(CASDetachedWork, FailedPublisherDispatchKeepsMutationAndClearsReservation) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.detached_dispatch_fault_for_test = DetachedDispatchFault::ThrowBeforeLaunch; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/dispatch_fail"}; + + EXPECT_NO_THROW(publishRef(store, ns, "ref_1", 1)) + << "a best-effort maintenance dispatch must never fail an otherwise-successful mutation"; + EXPECT_EQ(store->pendingSnapshotPublishesForTest(ns), 0) + << "the reservation was stranded: quiescence and dropNamespace would wait on it forever"; +} + +/// Settlement must survive a throwing error handler. Today it is a bare call after the handler, so a +/// handler that throws skips it and strands the reservation for the life of the process. +/// +/// The publish is FAULTED deliberately: with a healthy backend it would succeed, the handler would +/// never run, and this test would pass while exercising nothing. +TEST(CASDetachedWork, SettlementSurvivesAThrowingErrorHandler) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.publish_error_hook_for_test + = [] { throw std::runtime_error("injected: the error handler itself throws"); }; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/handler_throws"}; + + /// Arm the fault so the publisher's own PUT fails and its `catch` is entered. Use the same arming + /// call the snapshot-ordering suite uses against this backend. + backend->armPutFailure("_snap/", 1); + + ASSERT_NO_THROW(publishRef(store, ns, "ref_1", 1)); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(store->pendingSnapshotPublishesForTest(ns), 0); +} + +/// A publisher asleep in recovery backoff must be woken by the stop, not waited out. The injected +/// sleep stands in for a long backoff without spending wall-clock time. +TEST(CASDetachedWork, StopWakesRecoveryBackoffSleep) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.snapshot_publish_backoff_initial_ms = 0; + config.snapshot_publish_backoff_max_ms = 0; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/backoff"}; + + auto first_publisher_captured = std::make_shared(); + auto release_first_publisher = std::make_shared(); + String ckpt_key; + preparePendingRecoveryPublisher( + store, backend, ns, first_publisher_captured, release_first_publisher, ckpt_key); + + auto sleeping = std::make_shared(); + auto release_sleep = std::make_shared>(false); + store->setRefRecoveryRetrySleepForTest( + [sleeping, release_sleep](uint64_t, const std::optional & token) + { + sleeping->open(); + while (!(token && token->stopping()) && !release_sleep->load()) + std::this_thread::yield(); + }); + + /// Exhaust one checkpoint publication inside recovery so the outer retry loop enters backoff. + backend->armCasConflict(ckpt_key, 100); + release_first_publisher->open(); + sleeping->wait(); + + const bool drained = store->stopAndDrainDetachedWork(/*deadline_ms=*/5000); + release_sleep->store(true); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_TRUE(drained); +} + +/// A publisher parked behind another runtime's in-flight recovery is below every I/O checkpoint. +TEST(CASDetachedWork, StopWakesAConcurrentRecoveryWaiter) +{ + auto backend = std::make_shared(); + auto recovery_entered = std::make_shared(); + auto release_recovery = std::make_shared(); + auto recovery_hook_armed = std::make_shared>(false); + + PoolConfig config; + config.snapshot_publish_backoff_initial_ms = 0; + config.snapshot_publish_backoff_max_ms = 0; + config.recovery_pre_first_request_hook_for_test = [recovery_entered, release_recovery, recovery_hook_armed] + { + if (!recovery_hook_armed->load()) + return; + recovery_entered->open(); + release_recovery->wait(); + }; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/concurrent_recovery"}; + + auto first_publisher_captured = std::make_shared(); + auto release_first_publisher = std::make_shared(); + String ckpt_key; + preparePendingRecoveryPublisher( + store, backend, ns, first_publisher_captured, release_first_publisher, ckpt_key); + + /// A synchronous caller owns the first recovery. It is deliberately not detached work, so the + /// drain below waits only for the publisher parked behind it. + recovery_hook_armed->store(true); + auto first_recovery = std::async(std::launch::async, [&store, &ns] { store->listRefs(ns); }); + recovery_entered->wait(); + + release_first_publisher->open(); + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->refRecoveryWaitersForTest(ns) == 0) + { + ASSERT_LT(std::chrono::steady_clock::now(), deadline) << "no second caller reached the wait"; + std::this_thread::yield(); + } + + const bool drained = store->stopAndDrainDetachedWork(/*deadline_ms=*/5000); + release_recovery->open(); + EXPECT_NO_THROW(first_recovery.get()); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_TRUE(drained); +} + +/// The token is checked BEFORE the walk's first backend request, so a stop latched while the +/// publisher is at that boundary means the request is never issued. +/// +/// A request already in flight is not interrupted. That case is a bounded drain timeout by design, so +/// the publisher is parked at the pre-request hook -- not inside a stalled `GET` -- and the drain is +/// latched before the hook is released. +TEST(CASDetachedWork, StopIsObservedBeforeTheFirstRecoveryRequest) +{ + auto backend = std::make_shared(); + auto at_boundary = std::make_shared(); + auto release = std::make_shared(); + auto recovery_hook_armed = std::make_shared>(false); + + PoolConfig config; + config.snapshot_publish_backoff_initial_ms = 0; + config.snapshot_publish_backoff_max_ms = 0; + config.recovery_pre_first_request_hook_for_test = [at_boundary, release, recovery_hook_armed] + { + if (!recovery_hook_armed->load()) + return; + at_boundary->open(); + release->wait(); + }; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/pre_first_request"}; + + auto first_publisher_captured = std::make_shared(); + auto release_first_publisher = std::make_shared(); + String ckpt_key; + preparePendingRecoveryPublisher( + store, backend, ns, first_publisher_captured, release_first_publisher, ckpt_key); + + recovery_hook_armed->store(true); + release_first_publisher->open(); + at_boundary->wait(); + const uint64_t gets_before = backend->getTotal(); + + auto drain = std::async(std::launch::async, + [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/5000); }); + awaitStopLatched(store); + release->open(); + + const bool drained = drain.get(); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_TRUE(drained); + EXPECT_EQ(backend->getTotal(), gets_before) + << "the walk issued its first request after the stop was already latched"; +} + +/// `readCheckpointSnapshotBase` is one recovery boundary but performs several raw requests. Once its +/// base-log `GET` has completed, a latched stop must prevent the later predecessor-seal `GET` rather +/// than waiting until the whole helper returns. +TEST(CASDetachedWork, StopBetweenSnapshotBaseRequestsPreventsThePredecessorGet) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.snapshot_publish_backoff_initial_ms = 0; + config.snapshot_publish_backoff_max_ms = 0; + auto store = openPublishingPool(backend, config); + store->setLiveWriterEpochForTest(2); + const RootNamespace ns{"srv1/between_snapshot_base_requests"}; + const RefTxnId birth_id{1, 1}; + const RefTxnId predecessor_seal_id{1, 2}; + const RefTxnId base_id{2, 1}; + + std::vector birth_ops{DB::Cas::tests::namespaceBirthOp()}; + for (const RefOp & op : DB::Cas::tests::publishCommittedOps("seed_1", ManifestRef{1, 101, 1})) + birth_ops.push_back(op); + DB::Cas::tests::writeTxnAt(*backend, store->layout(), ns, birth_id, std::move(birth_ops)); + DB::Cas::tests::writeSealAt(*backend, store->layout(), ns, predecessor_seal_id); + DB::Cas::tests::writeTxnAt( + *backend, + store->layout(), + ns, + base_id, + DB::Cas::tests::publishCommittedOps("seed_2", ManifestRef{2, 101, 1}), + predecessor_seal_id); + DB::Cas::tests::writeRefSnapshotRaw( + *backend, + store->layout(), + DB::Cas::tests::minimalLiveSnapshot( + ns.string(), + base_id, + {DB::Cas::tests::committedRow("seed_1", ManifestRef{1, 101, 1}), + DB::Cas::tests::committedRow("seed_2", ManifestRef{2, 101, 1})})); + DB::Cas::tests::writeRecoverableCkptForRawFixture( + *backend, + store->layout(), + ns, + RefCkpt{ + .life_epoch = 1, + .committed_through = base_id, + .checkpoint_snapshot_id = base_id, + .last_epoch_seal = predecessor_seal_id}); + ASSERT_NO_THROW(store->listRefs(ns)); + + auto first_publisher_captured = std::make_shared(); + auto release_first_publisher = std::make_shared(); + String ckpt_key; + preparePendingRecoveryPublisher( + store, backend, ns, first_publisher_captured, release_first_publisher, ckpt_key); + + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const String base_log_key = store->layout().refLogKey(life, base_id); + const String predecessor_key = store->layout().refLogKey(life, predecessor_seal_id); + auto first_get_completed = std::make_shared(); + auto release_first_get = std::make_shared(); + backend->armBetweenGets(base_log_key, first_get_completed, release_first_get); + release_first_publisher->open(); + first_get_completed->wait(); + const uint64_t predecessor_gets_before = backend->getCount(predecessor_key); + + auto drain = std::async(std::launch::async, + [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/5000); }); + awaitStopLatched(store); + release_first_get->open(); + + EXPECT_TRUE(drain.get()); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(backend->getCount(predecessor_key), predecessor_gets_before) + << "recovery started the predecessor-seal GET after detached stop was latched"; +} + +/// The final fence callback runs only after the final authority read, `finish`, and O(N) +/// materialization. A stop latched there must leave the still-unrecovered runtime uninstalled. +TEST(CASDetachedWork, StopAfterRecoveryMaterializationPreventsFinalInstall) +{ + ManualDetachedLedger fixture; + const RootNamespace ns{"srv1/stop_before_recovery_install"}; + ASSERT_NO_THROW(publishRef(fixture.ledger, ns, "ref_1", 1)); + + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*fixture.backend, fixture.layout, ns).value(); + const String ckpt_key = fixture.layout.refCkptKey(life); + fixture.backend->armCasConflict(ckpt_key, 100); + EXPECT_ANY_THROW(fixture.ledger.dropRef(ns, "ref_1")); + fixture.backend->armCasConflict(ckpt_key, 0); + ASSERT_EQ(fixture.ledger.laneStateForTest(ns), RefLaneState::NeedsRecovery); + + auto detached_publisher = fixture.takeDetachedTask(); + fixture.backend->armFinalAuthorityRead(ckpt_key); + const uint64_t installs_before = fixture.ledger.recoveryInstallCountForTest(); + auto running = std::async(std::launch::async, + [&fixture, task = std::move(detached_publisher)]() mutable + { + task(DetachedStopToken(fixture.registry)); + }); + + fixture.final_install_reached.wait(); + fixture.latchStop(); + fixture.release_final_install.open(); + EXPECT_NO_THROW(running.get()); + + EXPECT_EQ(fixture.ledger.recoveryInstallCountForTest(), installs_before) + << "recovery installed a materialized result after detached stop was latched"; + EXPECT_EQ(fixture.ledger.laneStateForTest(ns), RefLaneState::NeedsRecovery); +} diff --git a/src/Disks/tests/gtest_cas_empty_proof.cpp b/src/Disks/tests/gtest_cas_empty_proof.cpp new file mode 100644 index 000000000000..4754786e7624 --- /dev/null +++ b/src/Disks/tests/gtest_cas_empty_proof.cpp @@ -0,0 +1,281 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +/// Task 9 (rev.7 spec §1 "empty-proof rule" [B3]): the last silent-empty-load killer. On a PRE-TERMINAL +/// (Live) or READ-ONLY pool, an enumeration about to answer EMPTY at a table root must first CONFIRM the +/// pool identity object (`_pool_meta`) exists with an AUTHORITATIVE, UNCACHED probe -- because "empty" at a +/// table root is exactly what a silently-erased backing looks like, and a read-only pool has no +/// keeper/lease/observer to catch that erasure any other way. These tests build a real +/// `ContentAddressedMetadataStorage` over a Local object storage (the gtest_cas_operation_gate.cpp harness) +/// and exercise the rule across the six cells the brief enumerates. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int INVALID_STATE; +extern const int NETWORK_ERROR; +} + +using namespace DB; +using DB::Cas::PoolLifecycle; +using DB::Cas::ProbeOutcome; +using DB::Cas::SentinelProbeResult; + +namespace +{ + +/// A committed (non-empty) table dir + part reused across the tests (the exact shape +/// gtest_ca_transaction.cpp / gtest_cas_operation_gate.cpp use). +const std::string kTableDir = "g80/g80g80g8-0808-4808-8808-080808080808"; +const std::string kPartDir = kTableDir + "/all_1_1_0"; +/// A DIFFERENT, never-committed-to table dir: genuinely empty for every test, distinct uuid so a +/// commit to kTableDir can never make it non-empty. +const std::string kEmptyTableDir = "g99/g99g99g9-0909-4909-8909-090909090909"; + +std::shared_ptr openStorage() +{ + auto settings = Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_empty_proof_scratch"); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// Commit one real part into `kTableDir`, leaving that table dir non-empty (tmp -> final rename -> commit). +void commitOnePart(ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kTableDir + "/tmp_insert_all_1_1_0/data.bin", 65536, WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kTableDir + "/tmp_insert_all_1_1_0", kPartDir); + tx->commit(NoCommitOptions{}); +} + +/// A read-only mount over a backing a writable mount already bootstrapped (`_pool_meta` present). The +/// writable mount minted the pool identity then shut down; the read-only mount validates `_pool_meta`, +/// takes NO lease and runs NO erasure observer (it stays `Live` forever) -- exactly the state in which +/// enumeration is the ONLY line of defense against a later erasure. Returns {ro storage, backing root}. +struct ReadOnlyMount +{ + std::shared_ptr ro; + std::string root; +}; + +/// Delete ONLY the physical `_pool_meta` object under `root`, leaving the container directory and every +/// other object intact — so a subsequent authoritative `probeSentinel` verdicts `KeyAbsent` (the identity +/// key is gone while the container is alive), NOT `ContainerAbsent` (which a whole-root `remove_all` yields). +/// This models the realistic "someone rm'd just the identity object" / partial-erase shape. Returns whether +/// exactly one `_pool_meta` file was found and removed, so the test can guard against a vacuous pass. +bool deleteOnlyPoolMetaUnder(const std::string & root) +{ + size_t removed = 0; + for (const auto & entry : std::filesystem::recursive_directory_iterator(root)) + { + if (entry.is_regular_file() && entry.path().filename() == "_pool_meta") + { + std::filesystem::remove(entry.path()); + ++removed; + } + } + return removed == 1; +} + +/// The message thrown by `fn`, or a failure if it did not throw a `DB::Exception`. +std::string messageOf(const std::function & fn) +{ + try + { + fn(); + } + catch (const Exception & e) + { + return std::string(e.message()); + } + ADD_FAILURE() << "expected a DB::Exception"; + return {}; +} + +ReadOnlyMount openReadOnlyOverBootstrappedBacking() +{ + auto settings = Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_empty_proof_ro_scratch"); + + /// (1) A writable mount bootstraps `_pool_meta` over a fresh backing, then shuts down. + auto rw_os = Cas::tests::makeLocalObjectStorageForTest(); + const std::string root = rw_os->getCommonKeyPrefix(); + { + auto w = std::make_shared( + rw_os, "pool", "srv1", "", nullptr, settings); + w->startup(); + w->shutdown(); + } + + /// (2) A read-only mount over the SAME backing validates `_pool_meta` and mounts `Live` (no lease, + /// no watermark, no observer -- read-only opens never enter the lifecycle machinery). + DB::LocalObjectStorageSettings ro_settings("test", root, /*read_only_=*/true); + auto ro_os = std::make_shared(std::move(ro_settings)); + auto ro = std::make_shared( + ro_os, "pool", "srv1", "", nullptr, settings); + ro->startup(); + return {std::move(ro), root}; +} + +} + +/// (a) THE RO-ATTACH silent-empty killer: a read-only pool whose whole backing was erased must throw, +/// never answer empty. Both mandatory authorities disappear; table enumeration observes the missing +/// `cas/ref_catalog` first, so `CORRUPTED_DATA` takes precedence over the later `_pool_meta` empty-proof +/// check. The pool-meta-only companion below keeps the typed 668 contract pinned separately. +TEST(CASEmptyProof, ReadOnlyOverErasedBackingThrowsInsteadOfEmpty) +{ + auto mount = openReadOnlyOverBootstrappedBacking(); + + /// Baseline while the backing is intact: the empty table root answers empty truthfully (`_pool_meta` + /// present authorizes it), issuing exactly one confirming probe. + mount.ro->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(mount.ro->listDirectory(kEmptyTableDir).empty()); + EXPECT_EQ(mount.ro->emptyProofProbeCountForTest(), 1u); + + /// Erase the backing out from under the (still Live) read-only mount: `_pool_meta` and everything. + std::filesystem::remove_all(mount.root); + + /// Now the SAME empty listing must refuse on the first missing mandatory control it observes. + Cas::tests::expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { mount.ro->listDirectory(kEmptyTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { mount.ro->iterateDirectory(kEmptyTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { mount.ro->isDirectoryEmpty(kEmptyTableDir); }); +} + +/// (a2, acceptance matrix — T9 review's KeyAbsent-specific real-backend follow-up) Test (a) erases the +/// WHOLE backing (`remove_all(root)`), so its probe verdicts `ContainerAbsent`. This test deletes ONLY the +/// `_pool_meta` object against the REAL Local backend — the container directory and every other object stay +/// intact — so the authoritative probe verdicts `KeyAbsent` instead. Both flavours must reach the SAME +/// "backing may be erased" refusal (distinct from the transient "transport or permission fault" one), so a +/// targeted deletion of just the identity object (a partial erase) is caught exactly like a whole-root wipe. +TEST(CASEmptyProof, ReadOnlyWithOnlyPoolMetaDeletedThrowsErasedFlavoredOnKeyAbsent) +{ + auto mount = openReadOnlyOverBootstrappedBacking(); + + /// Baseline while the backing is intact: the empty table root answers empty truthfully with one probe. + mount.ro->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(mount.ro->listDirectory(kEmptyTableDir).empty()); + EXPECT_EQ(mount.ro->emptyProofProbeCountForTest(), 1u); + + /// Delete ONLY `_pool_meta` (container + every sibling object intact) → the probe verdicts KeyAbsent. + ASSERT_TRUE(deleteOnlyPoolMetaUnder(mount.root)) + << "expected exactly one _pool_meta object to remove; otherwise this test is vacuous"; + + /// The KeyAbsent miss reaches the erased-flavored typed 668, NOT the transient one, and never answers empty. + const std::string msg = messageOf([&] { mount.ro->listDirectory(kEmptyTableDir); }); + EXPECT_NE(msg.find("pool identity object absent"), std::string::npos) << msg; + EXPECT_NE(msg.find("the backing may be erased"), std::string::npos) << msg; + EXPECT_EQ(msg.find("transport or permission fault"), std::string::npos) + << "a KeyAbsent miss must give the erased message, not the transient/retry one: " << msg; + + /// The other enumeration entry points refuse identically. + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { mount.ro->iterateDirectory(kEmptyTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { mount.ro->isDirectoryEmpty(kEmptyTableDir); }); +} + +/// (b) A Live pool over a genuinely-empty table dir with `_pool_meta` present answers empty AND issues +/// EXACTLY ONE uncached sentinel probe -- and it happens on the empty (`isDirectoryEmpty` == true) path. +TEST(CASEmptyProof, LiveEmptyTableDirAnswersEmptyWithExactlyOneProbe) +{ + auto storage = openStorage(); + + storage->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(storage->isDirectoryEmpty(kEmptyTableDir)); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 1u) + << "the empty table-root answer must confirm the pool identity with exactly one probe"; + + /// listDirectory / iterateDirectory each independently issue exactly one confirming probe too. + storage->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(storage->listDirectory(kEmptyTableDir).empty()); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 1u); +} + +/// (c) The zero-cost hot path: a NON-empty table dir issues NO probe at all. +TEST(CASEmptyProof, LiveNonEmptyTableDirIssuesNoProbe) +{ + auto storage = openStorage(); + commitOnePart(*storage); + + storage->resetEmptyProofProbeCountForTest(); + EXPECT_FALSE(storage->listDirectory(kTableDir).empty()); + EXPECT_FALSE(storage->isDirectoryEmpty(kTableDir)); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 0u) + << "the non-empty hot path must never touch the empty-proof probe"; +} + +/// (d) A Vanished pool answers truth-empty WITHOUT any probe: `checkOpAdmitted`'s Probe -> TruthAbsent +/// short-circuit answers before classification, so the terminal path never pays the empty-proof. +TEST(CASEmptyProof, VanishedPoolAnswersTruthEmptyWithoutProbe) +{ + auto storage = openStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live + pool->setLifecycleForTest(PoolLifecycle::VanishedForgotten); + + storage->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(storage->listDirectory(kTableDir).empty()); + EXPECT_TRUE(storage->isDirectoryEmpty(kTableDir)); + EXPECT_FALSE(storage->iterateDirectory(kTableDir)->isValid()); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 0u) + << "a Vanished pool answers truth-empty directly -- the gate short-circuits before the empty-proof"; +} + +/// (e) Scope discipline: a deeper (non-root) part-dir enumeration that answers empty is NOT gated. +TEST(CASEmptyProof, DeeperPartDirEmptyAnswerIsNotGated) +{ + auto storage = openStorage(); + + /// A never-committed part dir under a table root: classifies as PartDir, answers empty, no probe. + const std::string absent_part_dir = kEmptyTableDir + "/all_9_9_0"; + storage->resetEmptyProofProbeCountForTest(); + EXPECT_TRUE(storage->listDirectory(absent_part_dir).empty()); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 0u) + << "only the TableDir/DetachedContainer roots are gated -- deeper part-dirs are not"; +} + +/// (f) A probe that cannot establish absence (transport/permission fault) throws the typed TRANSIENT +/// refusal, never an empty answer. Unproven absence is unavailability, so the refusal carries the +/// upstream-retryable class -- unlike the `KeyAbsent`/`ContainerAbsent` arm, where absence IS proven and +/// the 668 stands. The fault is injected through the empty-proof override seam. +TEST(CASEmptyProof, IndeterminateProbeThrowsTransientNeverEmpty) +{ + auto storage = openStorage(); + storage->setEmptyProofProbeOverrideForTest( + [] { return SentinelProbeResult{ProbeOutcome::Indeterminate, std::nullopt}; }); + + storage->resetEmptyProofProbeCountForTest(); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->listDirectory(kEmptyTableDir); }); + EXPECT_EQ(storage->emptyProofProbeCountForTest(), 1u); + + /// The transient message names the fault (a retryable condition), distinct from the erased message. + std::string msg; + try + { + storage->listDirectory(kEmptyTableDir); + } + catch (const Exception & e) + { + msg = std::string(e.message()); + } + EXPECT_NE(msg.find("transport or permission fault"), std::string::npos) << msg; + EXPECT_NE(msg.find("TRANSIENT"), std::string::npos) << msg; +} diff --git a/src/Disks/tests/gtest_cas_encoding_pins.cpp b/src/Disks/tests/gtest_cas_encoding_pins.cpp new file mode 100644 index 000000000000..f4c3a913fda7 --- /dev/null +++ b/src/Disks/tests/gtest_cas_encoding_pins.cpp @@ -0,0 +1,110 @@ +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; +using namespace DB::Cas; + +/// These literals pin the CANONICAL BYTES of the CAS text encoders as of the commit that +/// introduced this file. The CasJsonWriter migration (2026-07-20 spec) must keep every one of +/// them green UNMODIFIED: canonical text is byte-compared on retries and deterministic adoption, +/// and the incremental ref budget counters assume these exact sizes. Never edit an expected +/// string here to make a test pass — that means the encoder's bytes drifted, which is the bug. + +TEST(CASEncodingPins, RefLogTxnAllOpKinds) +{ + RefLogTxn txn; + txn.ns = "roots/pin"; + txn.txn_id = RefTxnId{7, 9}; + + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(birth); + + RefOp transition; + transition.kind = RefOpKind::OwnerTransition; + transition.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "20260101_0_1_1_1", ManifestRef{1, 2, 3}}; + transition.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "20260101_0_1_1_1", ManifestRef{1, 2, 3}}; + txn.ops.push_back(transition); + + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + /// NOTE the split literals: "\x01" "e" (else the hex escape would swallow the 'e') and + /// "\xA8" "f" (else it would swallow the 'f'). `checkCanonicalRefName` forbids '\\' and NUL but + /// not quote/newline/control bytes/U+2028, so `ref_name` -- the only free-form string `RefOp` + /// still carries now that `payload` is gone -- exercises quote, newline, a bare control byte, + /// and the three-byte U+2028 sequence. Backslash escaping is pinned separately, over an + /// unrestricted string, by `gtest_cas_json_writer.cpp`'s `CASJsonWriterEscaping` suite. + set_published_at.ref_name = String("20260101_0_1_1_1\"c\nd") + "\x01" "e" + "\xE2\x80\xA8" "f"; + set_published_at.expected_manifest_ref = ManifestRef{1, 2, 3}; + set_published_at.published_at_ms = 1234; + txn.ops.push_back(set_published_at); + + RefOp removal; + removal.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(removal); + + const String expected = fmt::format("{{\"type\":\"cas_ref_log\",\"v\":{}}}\n", currentCompatibilityVersion()) + + "{\"ns\":\"roots/pin\",\"we\":\"7\",\"rs\":\"9\"}\n" + "{\"op\":\"namespace_birth\"}\n" + "{\"op\":\"owner_transition\",\"obk\":\"precommit\",\"orn\":\"20260101_0_1_1_1\"," + "\"ome\":\"1\",\"omb\":\"2\",\"omo\":3,\"nbk\":\"committed\",\"nrn\":\"20260101_0_1_1_1\"," + "\"nme\":\"1\",\"nmb\":\"2\",\"nmo\":3}\n" + "{\"op\":\"set_published_at\",\"rn\":\"20260101_0_1_1_1\\\"c\\nd\\u0001e\\u2028f\"," + "\"me\":\"1\",\"mb\":\"2\",\"mo\":3,\"ts\":1234}\n" + "{\"op\":\"remove_namespace\"}\n" + "{\"n\":4}\n"; + EXPECT_EQ(encodeRefLogTxn(txn), expected); +} + +TEST(CASEncodingPins, RefSnapshotLive) +{ + RefTableSnapshot snap; + snap.ns = "roots/pin"; + snap.snapshot_id = RefTxnId{7, 9}; + + RefCommittedRow row; + row.ref_name = "20260101_0_1_1_1"; + row.manifest_ref = ManifestRef{1, 2, 3}; + row.published_at_ms = 5; + snap.committed.push_back(row); + + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "20260102_0_2_2_2", ManifestRef{4, 5, 6}}); + + const String expected = fmt::format("{{\"type\":\"cas_ref_snap\",\"v\":{}}}\n", currentCompatibilityVersion()) + + "{\"ns\":\"roots/pin\",\"we\":\"7\",\"rs\":\"9\",\"lc\":\"live\"}\n" + "{\"k\":\"c\",\"rn\":\"20260101_0_1_1_1\",\"me\":\"1\",\"mb\":\"2\",\"mo\":3,\"ts\":5}\n" + "{\"k\":\"p\",\"rn\":\"20260102_0_2_2_2\",\"me\":\"4\",\"mb\":\"5\",\"mo\":6}\n" + "{\"n\":2}\n"; + EXPECT_EQ(encodeRefTableSnapshot(snap), expected); +} + +TEST(CASEncodingPins, SourceEdgeRunLines) +{ + WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + + SourceEdgeRecord active; + active.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(2))}; + active.source_id = UInt128(5); + active.marker = kEdgeActive; + writer.append(active); + + writer.finish(); + out.finalize(); + + /// The exact "b" rendering (algo byte + digest hex) is pinned as a whole line; the point is + /// that Task 8's line-scratch rewrite must reproduce it byte-for-byte. + const String text = out.str(); + const String header = fmt::format("{{\"type\":\"cas_run\",\"v\":{},\"kind\":\"source_edge\"}}\n", currentCompatibilityVersion()); + const String expected_record = + "{\"b\":\"0100000000000000000000000000000002\",\"s\":\"00000000000000000000000000000005\",\"m\":\"edge\"}\n"; + const String trailer = "{\"n\":1}\n"; + /// There is exactly one record, so the whole buffer must be byte-identical to header + record + trailer. + const String expected_full = header + expected_record + trailer; + EXPECT_EQ(text, expected_full) << text; +} diff --git a/src/Disks/tests/gtest_cas_envelope.cpp b/src/Disks/tests/gtest_cas_envelope.cpp new file mode 100644 index 000000000000..7dac7ddf6eed --- /dev/null +++ b/src/Disks/tests/gtest_cas_envelope.cpp @@ -0,0 +1,55 @@ +#include +#include +#include +#include + +using namespace DB; +using namespace DB::Cas; + +/// The v3 blob-envelope shape (256-byte JSON header + payload). Full round-trip / gate / pad-zone / +/// budget / critical-key coverage lives in gtest_cas_blob_envelope_format.cpp; these two keep the +/// cases that file does not exercise: a header with NO provenance/ref, and the incarnation-zone +/// independence of the payload. + +TEST(CASEnvelope, BlobRoundTripNoExtensions) +{ + const std::string payload = "hello payload"; + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = 0x22; + h.build_id = 0x33; + const std::string obj = encodeEnvelopeHeader(h, 256) + payload; + + const EnvelopeHeader d = decodeEnvelopeHeader(obj, obj.size(), ObjectKind::Blob); + EXPECT_EQ(d.kind, ObjectKind::Blob); + EXPECT_EQ(d.compatibility_version, G_BUILD); + EXPECT_FALSE(d.provenance.has_value()); /// none set -> the ts/by/op/ch keys are absent + EXPECT_FALSE(d.intended_ref.has_value()); /// none set -> the ref key is omitted + EXPECT_EQ(d.header_len, 256u); + /// payload starts right after the fixed-length header. + EXPECT_EQ(obj.substr(payloadOffset(d)), payload); +} + +TEST(CASEnvelope, IncarnationZoneDoesNotAffectPayload) +{ + /// Two objects with the SAME payload but DIFFERENT incarnation_tag/build_id encode to different + /// header bytes, yet both carry the same payload at the same fixed offset — the incarnation zone + /// never affects the payload. Identity is the content key, not any header field. + const std::string payload = "same content"; + EnvelopeHeader a; + a.kind = ObjectKind::Blob; + a.incarnation_tag = 0xAAAA; + a.build_id = 0xBBBB; + EnvelopeHeader b = a; + b.incarnation_tag = 0xCCCC; + b.build_id = 0xDDDD; + + const std::string ha = encodeEnvelopeHeader(a, 256); + const std::string hb = encodeEnvelopeHeader(b, 256); + EXPECT_NE(ha, hb); /// headers differ (incarnation zone) + + const EnvelopeHeader da = decodeEnvelopeHeader(ha + payload, ha.size() + payload.size(), ObjectKind::Blob); + const EnvelopeHeader db = decodeEnvelopeHeader(hb + payload, hb.size() + payload.size(), ObjectKind::Blob); + EXPECT_EQ((ha + payload).substr(payloadOffset(da)), payload); + EXPECT_EQ((hb + payload).substr(payloadOffset(db)), payload); +} diff --git a/src/Disks/tests/gtest_cas_event_dispatcher.cpp b/src/Disks/tests/gtest_cas_event_dispatcher.cpp new file mode 100644 index 000000000000..325c70de59f4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_event_dispatcher.cpp @@ -0,0 +1,194 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// A single-blob part: upload one blob, stage a one-entry manifest naming it, precommit + promote the +/// ref. Mirrors `publishOneBlobPart` in `gtest_cas_event_log.cpp` so a committed ref exists for +/// `resolveRef` to resolve and emit against. +void publishOneBlobPart(const PoolPtr & s, const String & ns, const String & ref, const String & payload) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}; + e.blob_size = payload.size(); + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(nsr, ref, build->buildId(), id); +} + +} + +/// Concurrent emitters must be serialized: N threads emit M events each into a sink that appends to a +/// DELIBERATELY UNGUARDED vector. If the dispatcher did not serialize delivery the concurrent +/// `push_back`s would tear the vector (and TSan on that lane would flag the data race); serialized +/// delivery makes the unguarded append correct. The count/uniqueness assertions catch dropped or +/// duplicated events on any lane. +TEST(CASEventDispatcher, SerializesConcurrentEmitters) +{ + EventDispatcher disp; + std::vector seen; /// unguarded on purpose -- the dispatcher is the only serialization + disp.setSink([&](CasEvent e) { seen.push_back(std::move(e)); }); + + constexpr int N = 8; /// emitter threads + constexpr int M = 250; /// emits per thread + std::latch start{N}; /// release all emitters together to maximize contention on the dispatcher + std::vector threads; + threads.reserve(N); + for (int t = 0; t < N; ++t) + threads.emplace_back([&, t] + { + start.arrive_and_wait(); + for (int m = 0; m < M; ++m) + { + CasEvent e; + e.type = CasEventType::BlobPut; + e.object_hash = std::to_string(t * M + m); + disp.emit(std::move(e)); + } + }); + for (auto & th : threads) + th.join(); + + ASSERT_EQ(seen.size(), static_cast(N * M)); + std::set ids; + for (const auto & e : seen) + ids.insert(e.object_hash); + EXPECT_EQ(ids.size(), static_cast(N * M)) << "every emitted event delivered exactly once"; +} + +/// A sink that emits again from inside its own delivery must not deadlock. The drain-loop design +/// never holds the dispatcher mutex across the sink call, so the reentrant `emit` acquires the mutex, +/// finds a drain already running, enqueues, and returns; the running loop delivers it after the +/// current sink returns. Delivery is synchronous on the emitting thread, so no timed wait is needed: +/// `emit` returns only after the whole queue (including the reentrant event) has drained. +TEST(CASEventDispatcher, ReentrantSinkDoesNotDeadlock) +{ + EventDispatcher disp; + std::vector delivered; + std::atomic reentered_once{false}; + disp.setSink([&](CasEvent e) + { + delivered.push_back(e.type); + if (e.type == CasEventType::BlobPut && !reentered_once.exchange(true)) + { + CasEvent second; + second.type = CasEventType::BlobDelete; + disp.emit(std::move(second)); + } + }); + + CasEvent first; + first.type = CasEventType::BlobPut; + disp.emit(std::move(first)); + + ASSERT_EQ(delivered.size(), 2u) << "both the original and the reentrant event must be delivered"; + EXPECT_EQ(delivered[0], CasEventType::BlobPut); + EXPECT_EQ(delivered[1], CasEventType::BlobDelete) + << "the reentrant event is drained AFTER the current sink returns, not recursively"; +} + +/// Test 17: a ledger emission must fire OUTSIDE the ledger lock. Install a sink that, on delivery of +/// a `RefResolve` event, re-enters a ledger read (`resolveRef`) that itself takes `state_mutex`; then +/// drive a real emitting `resolveRef` on a worker thread while a second thread emits upload-style +/// events concurrently. If `resolveRef` emitted while holding `state_mutex` (the pre-fix defect), the +/// worker would re-lock `state_mutex` on the same thread from inside the sink and self-deadlock. The +/// restructured emit (after the lock scope) lets the reentrant read take the lock freshly, and the +/// dispatcher serializes the concurrent upload emissions. +TEST(CASEventDispatcher, LedgerEmissionOutsideLocks) +{ + auto b = std::make_shared(); + std::vector seen; /// declared before the Pool so it outlives any late background emit + std::mutex seen_mutex; + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const RootNamespace ns{"srv1/tbl"}; + const String ref = "all_0_0_0"; + publishOneBlobPart(s, ns.string(), ref, "the-resolvable-payload"); + + std::atomic reentered{false}; + s->setEventSink([&](CasEvent e) + { + { + std::lock_guard g(seen_mutex); + seen.push_back(e); + } + /// Re-enter a ledger read that takes `state_mutex`, exactly once (`Deferred` => this read + /// itself emits nothing, so there is no unbounded emit recursion). Under the pre-fix code the + /// outer `resolveRef` still holds `state_mutex` here, so this call self-deadlocks. + if (e.type == CasEventType::RefResolve && !reentered.exchange(true)) + (void)s->resolveRef(ns, ref, false, ResolveAudit::Deferred); + }); + + std::promise resolve_done; + auto resolve_future = resolve_done.get_future(); + std::thread resolver([&] + { + (void)s->resolveRef(ns, ref); /// ResolveAudit::Emit (default) -> emits RefResolve -> drives the sink + resolve_done.set_value(); + }); + + /// A second thread emits upload-task-style events concurrently with the resolve, so the dispatcher's + /// serialization is exercised alongside the reentrancy path. + std::thread uploader([&] + { + for (int i = 0; i < 32; ++i) + { + CasEvent up; + up.type = CasEventType::BlobPut; + up.object_hash = "up-" + std::to_string(i); + up.reason = "concurrent upload-task emission"; + s->emitEvent(std::move(up)); + } + }); + + /// Bounded wait: the resolve is two in-memory map lookups plus queue drains -- microseconds of + /// real work. 10 seconds is many orders of magnitude above that and only elapses if the + /// emit-under-lock defect self-deadlocks the worker on `state_mutex`. + const auto status = resolve_future.wait_for(std::chrono::seconds(10)); + ASSERT_EQ(status, std::future_status::ready) + << "resolveRef with a re-entrant sink did not complete: emission is happening under state_mutex"; + resolver.join(); + uploader.join(); + + EXPECT_TRUE(reentered.load()) << "the reentrant ledger read must have run"; + std::lock_guard g(seen_mutex); + size_t resolves = 0; + size_t uploads = 0; + for (const auto & e : seen) + { + if (e.type == CasEventType::RefResolve) + ++resolves; + else if (e.type == CasEventType::BlobPut) + ++uploads; + } + EXPECT_GE(resolves, 1u) << "the driving resolve emitted its RefResolve"; + EXPECT_EQ(uploads, 32u) << "every concurrent upload emission was delivered exactly once"; +} diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp new file mode 100644 index 000000000000..ae65ed27274b --- /dev/null +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -0,0 +1,672 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int NETWORK_ERROR; +} + +namespace DB::Cas +{ +void configureMountRenewObservability( + const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; +void reportMountRenewCompletion(const MountRenewResult & result) noexcept; +} + +namespace +{ + +class RenewalEventBackend final : public InMemoryBackend +{ +public: + using InMemoryBackend::get; + using InMemoryBackend::putOverwrite; + + bool throw_before_next_overwrite = false; + bool throw_nonretryable_next_overwrite = false; + bool vanish_on_next_overwrite = false; + + void armResolveProbe() + { + std::lock_guard lock(resolve_mutex); + observe_next_get = true; + resolve_started = false; + } + + bool resolveStarted() + { + std::lock_guard lock(resolve_mutex); + return resolve_started; + } + + std::optional get(const String & key, Range range) override + { + { + std::lock_guard lock(resolve_mutex); + if (observe_next_get) + { + resolve_started = true; + observe_next_get = false; + } + } + return InMemoryBackend::get(key, range); + } + + PutResult putOverwrite( + const String & key, + const String & bytes, + const Token & expected, + const ObjectMeta & meta) override + { + if (std::exchange(vanish_on_next_overwrite, false)) + { + (void)InMemoryBackend::deleteExact(key, expected); + return {PutOutcome::PreconditionFailed, {}}; + } + if (std::exchange(throw_nonretryable_next_overwrite, false)) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected deterministic renewal rejection"); + if (std::exchange(throw_before_next_overwrite, false)) + throw Poco::TimeoutException("injected renewal timeout before commit"); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + +private: + std::mutex resolve_mutex; + bool observe_next_get = false; + bool resolve_started = false; +}; + +CasRequestBudget renewalEventBudget() +{ + return CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = 2, + .lease_safety_margin_ms = 20, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }; +} + +PoolPtr openRenewalEventPool( + const std::shared_ptr & backend, + uint64_t & boot_ms, + CasRequestBudget budget = renewalEventBudget(), + String prefix = "renewal-events", + String server_root_id = "test") +{ + return Pool::open(backend, PoolConfig{ + .pool_prefix = std::move(prefix), + .server_root_id = std::move(server_root_id), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return boot_ms; }, + }); +} + +std::vector watermarkRenewEvents(const std::vector & events) +{ + std::vector result; + std::copy_if(events.begin(), events.end(), std::back_inserter(result), [](const CasEvent & event) + { + return event.type == CasEventType::WatermarkRenew; + }); + return result; +} + +} + +/// Round-B opt §6: `reason` is templated rationale (a handful of distinct strings repeated across +/// every row), unlike `object_hash`/`token` which are genuinely per-row varied -- it belongs alongside +/// the log's other LowCardinality columns (event_type/object_kind/outcome), not as a full String. +TEST(CASContentAddressedLog, ReasonColumnIsLowCardinality) +{ + const auto columns = DB::ContentAddressedLogElement::getColumnsDescription(); + const auto & reason_col = columns.get("reason"); + EXPECT_TRUE(typeid_cast(reason_col.type.get())) + << "reason column must be LowCardinality(String) (Round-B opt §6)"; +} +TEST(CASEvent, ConstructAndCopyAndName) +{ + CasEvent e; + e.type = CasEventType::BlobDelete; + e.object_kind = CasEventObjectKind::Blob; + e.object_hash = "abcd"; + e.token = "tok"; + e.round = 7; e.gen = 3; + e.reason = "in-degree 0 after strip"; + e.detail["freed"] = "10"; + CasEvent c = e; + EXPECT_EQ(c.type, CasEventType::BlobDelete); + EXPECT_EQ(c.object_hash, "abcd"); + EXPECT_EQ(c.detail.at("freed"), "10"); + EXPECT_EQ(toString(CasEventType::BlobDelete), "blob_delete"); + EXPECT_EQ(toString(CasEventType::IndegZero), "indegree_zero"); + EXPECT_EQ(toString(CasEventType::GcRecheckVerdict), "gc_recheck_verdict"); + EXPECT_EQ(toString(CasEventObjectKind::Manifest), "manifest"); +} + +TEST(CASEvent, PoolEmitsToSink) +{ + auto b = std::make_shared(); + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + CasEvent e; + e.type = CasEventType::BlobPut; + e.object_hash = "h"; + s->emitEvent(std::move(e)); + ASSERT_EQ(seen.size(), 1u); + EXPECT_EQ(seen[0].type, CasEventType::BlobPut); + /// null sink => no-op (no crash, no row); a fresh event, not the one already moved above. + s->setEventSink(nullptr); + CasEvent e2; + e2.type = CasEventType::BlobPut; + s->emitEvent(std::move(e2)); + EXPECT_EQ(seen.size(), 1u); +} + +TEST(CASEvent, FirstAttemptRenewalIsSilent) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + auto store = openRenewalEventPool(backend, boot_ms); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + + EXPECT_NO_THROW(store->renewWatermarkOnce()); + EXPECT_TRUE(watermarkRenewEvents(events).empty()); +} + +TEST(CASEvent, WatermarkRenewEventsAreBoundedAndComplete) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + auto store = openRenewalEventPool(backend, boot_ms); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + + backend->throw_before_next_overwrite = true; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + + const std::vector renewals = watermarkRenewEvents(events); + ASSERT_EQ(renewals.size(), 2u); + EXPECT_EQ(renewals[0].outcome, "retrying"); + EXPECT_EQ(renewals[1].outcome, "recovered"); + EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "1"); + EXPECT_EQ(renewals[1].detail.at("attempts_sent"), "2"); + EXPECT_EQ(renewals[0].detail.at("server_root_id"), "test"); + EXPECT_EQ(renewals[0].detail.at("writer_epoch"), std::to_string(store->writerEpoch())); + EXPECT_EQ(renewals[0].detail.at("seq"), "2"); + EXPECT_EQ(renewals[0].detail.at("write_attempt_id"), renewals[1].detail.at("write_attempt_id")); + EXPECT_FALSE(renewals[0].detail.at("write_attempt_id").empty()); + EXPECT_LT(renewals[0].detail.at("write_attempt_id").size(), 32u); + + for (const CasEvent & event : renewals) + { + for (const String & key : { + "server_root_id", + "writer_epoch", + "seq", + "write_attempt_id", + "attempts_sent", + "elapsed_ms", + "remaining_confirmed_budget_ms", + "unresolved_reason", + "deadline_source", + "stop_cause", + "classification"}) + EXPECT_TRUE(event.detail.contains(key)) << "missing detail key " << key; + } +} + +TEST(CASEvent, FirstAmbiguityIsVisibleWhileResolveIsInFlight) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::atomic retrying_events{0}; + auto store = openRenewalEventPool( + backend, boot_ms, renewalEventBudget(), "renewal-inflight-ambiguity"); + store->setEventSink([&](CasEvent event) + { + if (event.type == CasEventType::WatermarkRenew && event.outcome == "retrying") + { + retrying_events.fetch_add(1); + /// The diagnostic callback may consume the remaining recovery budget. The controller + /// must re-check its absolute deadline before starting the resolving GET. + boot_ms = 1081; + } + }); + + backend->throw_before_next_overwrite = true; + backend->armResolveProbe(); + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + + EXPECT_EQ(retrying_events.load(), 1u) + << "first ambiguity must be externally visible before the pre-resolve deadline gate"; + EXPECT_FALSE(backend->resolveStarted()) + << "a diagnostic sink that exhausts the budget must prevent the resolving GET from starting"; + EXPECT_EQ(retrying_events.load(), 1u) << "retrying delivery is bounded to the first ambiguity"; +} + +TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) +{ + constexpr size_t depth = 10; + std::array, depth> backends; + std::array, depth> layouts; + std::array, depth> keepers; + std::array server_root_ids; + std::array sinks; + uint64_t wall_ms = 100; + uint64_t boot_ms = 100; + std::optional deepest_result; + std::function renew_at; + + renew_at = [&](size_t index) + { + configureMountRenewObservability(&server_root_ids[index], &sinks[index], /*deferred=*/false); + MountRenewResult result = keepers[index]->renew( + CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = 1, + .lease_safety_margin_ms = 0, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }, + MountRenewOperationEnvironment{}); + reportMountRenewCompletion(result); + return result; + }; + + for (size_t index = 0; index < depth; ++index) + { + backends[index] = std::make_shared(); + layouts[index] = std::make_unique(fmt::format("deep-renewal-{}", index)); + server_root_ids[index] = fmt::format("deep-{}", index); + sinks[index] = [&, index](CasEvent event) + { + if (event.type == CasEventType::MountConflict && index + 1 < depth) + { + MountRenewResult child_result = renew_at(index + 1); + if (index + 2 == depth) + deepest_result = std::move(child_result); + } + }; + keepers[index] = std::make_unique( + backends[index], + *layouts[index], + server_root_ids[index], + UInt128(index + 1), + 7, + std::chrono::milliseconds(1000), + [&] { return wall_ms; }, + [] { return uint64_t{0}; }, + sinks[index], + std::chrono::milliseconds(0), + [&] { return boot_ms; }); + keepers[index]->start(); + + if (index + 1 < depth) + { + const String key = layouts[index]->mountKey(server_root_ids[index]); + auto observed = backends[index]->get(key); + ASSERT_TRUE(observed.has_value()); + MountLease foreign = decodeMountLease(observed->bytes); + foreign.server_uuid = UInt128(100 + index); + ASSERT_EQ( + backends[index]->putOverwrite(key, encodeMountLease(foreign), observed->token).outcome, + PutOutcome::Done); + } + } + backends.back()->throw_nonretryable_next_overwrite = true; + + const MountRenewResult outer_result = renew_at(0); + EXPECT_EQ(outer_result.outcome, MountRenewOutcome::Terminal); + ASSERT_TRUE(deepest_result.has_value()); + EXPECT_EQ(deepest_result->outcome, MountRenewOutcome::Terminal); + EXPECT_EQ(deepest_result->diagnostics.attempts_sent, 1u) + << "nesting beyond the rich-event stack must not erase physical attempt truth"; +} + +TEST(CASEvent, WatermarkRenewSinkFailureCannotChangeOutcome) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = openRenewalEventPool(backend, boot_ms); + const String mount_key = store->layout().mountKey("test"); + const uint64_t seq_before = decodeMountLease(backend->get(mount_key)->bytes).seq; + store->setEventSink([](const CasEvent & event) + { + if (event.type == CasEventType::WatermarkRenew) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected renewal event sink failure"); + }); + + backend->throw_before_next_overwrite = true; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + EXPECT_EQ(decodeMountLease(backend->get(mount_key)->bytes).seq, seq_before + 1); + EXPECT_TRUE(store->mayMutate()); +} + +TEST(CASEvent, TerminalRenewalDetailsPreservePhysicalTruthAndClassification) +{ + const auto one_failed_event = [](const std::vector & events) -> CasEvent + { + const std::vector renewals = watermarkRenewEvents(events); + const auto failed = std::find_if(renewals.begin(), renewals.end(), [](const CasEvent & event) + { + return event.outcome == "failed"; + }); + EXPECT_NE(failed, renewals.end()); + return failed == renewals.end() ? CasEvent{} : *failed; + }; + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-deterministic-details"); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + backend->throw_nonretryable_next_overwrite = true; + + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + const CasEvent failed = one_failed_event(events); + EXPECT_EQ(failed.detail.at("attempts_sent"), "1"); + EXPECT_EQ(failed.detail.at("unresolved_reason"), "not_unresolved"); + EXPECT_EQ(failed.detail.at("stop_cause"), "continue"); + EXPECT_EQ(failed.detail.at("classification"), "deterministic_failure"); + } + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + CasRequestBudget budget = renewalEventBudget(); + budget.max_attempts = 1; + auto store = openRenewalEventPool(backend, boot_ms, budget, "renewal-exhausted-details"); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + backend->throw_before_next_overwrite = true; + + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + const CasEvent failed = one_failed_event(events); + EXPECT_EQ(failed.detail.at("attempts_sent"), "1"); + EXPECT_EQ(failed.detail.at("unresolved_reason"), "attempts_exhausted"); + EXPECT_EQ(failed.detail.at("classification"), "attempts_exhausted"); + } + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-deadline-details"); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + boot_ms = 1071; + + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + const CasEvent failed = one_failed_event(events); + EXPECT_EQ(failed.detail.at("attempts_sent"), "0"); + EXPECT_EQ(failed.detail.at("unresolved_reason"), "no_attempt_sent"); + EXPECT_EQ(failed.detail.at("deadline_source"), "external_lease_safety"); + EXPECT_EQ(failed.detail.at("classification"), "external_lease_deadline"); + } +} + +TEST(CASEvent, ReentrantRenewalSinkPreservesOuterObservationIdentity) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + std::vector events; + PoolPtr store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-reentrant-sink"); + bool reentered = false; + store->setEventSink([&](CasEvent event) + { + if (event.type != CasEventType::WatermarkRenew) + return; + events.push_back(event); + if (event.outcome == "recovered" && !std::exchange(reentered, true)) + store->renewWatermarkOnce(); + }); + + backend->throw_before_next_overwrite = true; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + + ASSERT_TRUE(reentered); + ASSERT_EQ(events.size(), 2u); + EXPECT_EQ(events[0].outcome, "retrying"); + EXPECT_EQ(events[1].outcome, "recovered"); + EXPECT_EQ(events[0].detail.at("seq"), "2"); + EXPECT_EQ(events[1].detail.at("seq"), events[0].detail.at("seq")); + EXPECT_EQ(events[1].detail.at("write_attempt_id"), events[0].detail.at("write_attempt_id")); + EXPECT_EQ(decodeMountLease(backend->get(store->layout().mountKey("test"))->bytes).seq, 3u) + << "the nested first-attempt success must run without replacing the outer observation"; +} + +TEST(CASEvent, PreCompletionConflictReentrancyPreservesOuterTerminalObservation) +{ + auto inner_backend = std::make_shared(); + uint64_t inner_boot_ms = 100; + auto inner = openRenewalEventPool( + inner_backend, inner_boot_ms, renewalEventBudget(), "renewal-reentrant-inner", "inner"); + + auto outer_backend = std::make_shared(); + uint64_t outer_boot_ms = 100; + CasRequestBudget outer_budget = renewalEventBudget(); + outer_budget.max_attempts = 1; + auto outer = openRenewalEventPool( + outer_backend, outer_boot_ms, outer_budget, "renewal-reentrant-outer", "outer"); + std::vector outer_events; + bool reentered = false; + outer->setEventSink([&](CasEvent event) + { + outer_events.push_back(event); + if (event.type == CasEventType::MountConflict && !std::exchange(reentered, true)) + inner->renewWatermarkOnce(); + }); + + outer_backend->vanish_on_next_overwrite = true; + EXPECT_THROW(outer->renewWatermarkOnce(), DB::Exception); + + ASSERT_TRUE(reentered); + const std::vector renewals = watermarkRenewEvents(outer_events); + ASSERT_EQ(renewals.size(), 2u); + EXPECT_EQ(renewals[0].outcome, "retrying"); + EXPECT_EQ(renewals[1].outcome, "failed"); + EXPECT_EQ(renewals[0].detail.at("server_root_id"), "outer"); + EXPECT_EQ(renewals[1].detail.at("server_root_id"), "outer"); + EXPECT_EQ(renewals[1].detail.at("write_attempt_id"), renewals[0].detail.at("write_attempt_id")); + EXPECT_EQ(renewals[1].detail.at("classification"), "vanished"); +} + +/// Round-B opt §6: `emitEvent` takes the event BY VALUE (moved-through, not `const &`), so a +/// caller's local is genuinely moved-from -- not merely copied via a const reference -- by the time +/// the sink runs. Mirrors `makeCasEventSink`'s own move-out-of-the-by-value-event idiom (a small test +/// double stands in for the `ContentAddressedLogElement` it would normally build). +TEST(CASEvent, EmitEventMovesSourceIntoSink) +{ + auto b = std::make_shared(); + String captured_reason; + std::map captured_detail; + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + s->setEventSink([&](CasEvent ev) + { + captured_reason = std::move(ev.reason); + captured_detail = std::move(ev.detail); + }); + CasEvent e; + e.type = CasEventType::BlobPut; + e.reason = "sentinel-reason"; + e.detail["k"] = "v"; + s->emitEvent(std::move(e)); + EXPECT_EQ(captured_reason, "sentinel-reason"); + EXPECT_EQ(captured_detail.at("k"), "v"); + /// the source event must be MOVED-FROM after emit, not merely aliased/copied through -- reading + /// `e` here is the whole point of the test, not an oversight. + EXPECT_TRUE(e.reason.empty()); // NOLINT(bugprone-use-after-move, hicpp-invalid-access-moved) + EXPECT_TRUE(e.detail.empty()); // NOLINT(bugprone-use-after-move, hicpp-invalid-access-moved) +} + +namespace +{ + +/// A single-blob part: upload one blob, stage a one-entry manifest naming it, precommit + promote the +/// ref. Returns the blob's object_hash (lowercase hex) so the test can filter the captured rows by it. +String publishOneBlobPart(const PoolPtr & s, const String & ns, const String & ref, const String & payload) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(nsr, ref, build->buildId(), id); + /// Phase 3 (mixed-algo pools): every blob-content-hash event render is `blobIdOf(ref)` + /// (":"), never a bare hex -- the prime directive that a digest never appears + /// without its algo. + return DB::Cas::blobIdOf(e.ref); +} + +/// Whether the CURRENT retired list (any gc-shard) still holds an entry (ack-floor pipeline in flight). +bool anyRetiredPending(const PoolPtr & s) +{ + /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// separate retired list — reconstruct the in-flight set from the seal. + return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); +} + +/// Drive regular GC to a fixpoint over the ACK-FLOOR round (renew the store's mount ack after each round; +/// stay alive while any work counter is nonzero OR an in-flight retired entry remains). +void runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + for (size_t r = 0; r < max_rounds; ++r) + { + const RoundReport rep = DB::Cas::tests::runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(s)) + break; + } +} + +bool hasType(const std::vector & events, CasEventType t) +{ + for (const auto & e : events) + if (e.type == t) + return true; + return false; +} + +} + +/// B170 Task 4 acceptance: drive a full publish -> drop -> GC-to-delete lifecycle through a capturing +/// sink and assert (a) the taxonomy of events is emitted, (b) EVERY event carries a non-empty reason, +/// (c) filtering by a deleted blob's object_hash reconstructs its edge/retire/delete chain in order. +TEST(CASEvent, LifecycleReconstructionFromRows) +{ + auto b = std::make_shared(); + /// Declared BEFORE the Pool so they OUTLIVE it: the Pool's background retired-view syncer can emit + /// (e.g. a view-advance event) right up to the Pool's destructor, and a sink capturing locals that + /// die first is a use-after-scope (found by ASan 2026-07-09; the production sink captures the Context + /// shared_ptr by value and is immune). + std::vector events; + std::mutex events_mutex; + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + s->setEventSink([&](const CasEvent & e) + { + std::lock_guard lock(events_mutex); + events.push_back(e); + }); + + const RootNamespace ns{"srv1/tbl"}; + const String ref = "all_0_0_0"; + const String payload = "the-doomed-blob-payload"; + + /// publish -> the blob's whole closure is born and a ref names it. + const String blob_hash = publishOneBlobPart(s, ns.string(), ref, payload); + + /// drop the ref and advance the watermark so the now-unreferenced closure is collectable. + s->dropRef(ns, ref); + s->renewWatermarkOnce(); + + /// GC reclaims the tree and the blob to a fixpoint. + Gc gc(s, u128Of("gc-event-log")); + runGcToFixpoint(s, gc); + + /// The blob must actually be gone (the delete fired). + ASSERT_FALSE(b->head(s->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))})).exists) + << "GC must have deleted the now-unreferenced blob"; + + /// (a) the expected taxonomy was emitted across the lifecycle (manifest model: no standalone trees). + EXPECT_TRUE(hasType(events, CasEventType::BlobPut)); + EXPECT_TRUE(hasType(events, CasEventType::RootAdd)) + << "a fold must have recorded the manifest owner's blob edge (+1)"; + EXPECT_TRUE(hasType(events, CasEventType::RefDrop)); + EXPECT_TRUE(hasType(events, CasEventType::IndegZero)); + EXPECT_TRUE(hasType(events, CasEventType::GcRetireObserve) + || hasType(events, CasEventType::GcRetireDecision) + || hasType(events, CasEventType::GcRecheckVerdict)) + << "a GC retire/recheck transition must be recorded"; + EXPECT_TRUE(hasType(events, CasEventType::BlobDelete) || hasType(events, CasEventType::ManifestDelete)) + << "the single content-delete site must emit a delete row"; + + /// (b) completeness mandate: every emitted event has a non-empty reason (the human WHY). + for (const auto & e : events) + EXPECT_FALSE(e.reason.empty()) + << "event " << toString(e.type) << " (" << e.object_hash << ") has an empty reason"; + + /// (c) lifecycle reconstruction: filtering by the deleted blob's object_hash yields, in time + /// order, at least its in-degree-zero -> retire-observe -> delete chain — its whole story. + std::vector chain; + for (const auto & e : events) + if (e.object_hash == blob_hash) + chain.push_back(e.type); + + ASSERT_FALSE(chain.empty()) << "no rows reference the deleted blob " << blob_hash; + + /// The decisive ordering: the blob's in-degree hit 0 BEFORE GC observed/condemned it, which was + /// BEFORE it was deleted. Find the first index of each and assert the order. + auto firstIndexOf = [&](CasEventType t) -> int + { + for (size_t i = 0; i < chain.size(); ++i) + if (chain[i] == t) + return static_cast(i); + return -1; + }; + const int i_indeg = firstIndexOf(CasEventType::IndegZero); + const int i_observe = firstIndexOf(CasEventType::GcRetireObserve); + const int i_delete = firstIndexOf(CasEventType::BlobDelete); + ASSERT_GE(i_indeg, 0) << "the blob's indegree_zero must be in its chain"; + ASSERT_GE(i_observe, 0) << "the blob's gc_retire_observe must be in its chain"; + ASSERT_GE(i_delete, 0) << "the blob's blob_delete must be in its chain"; + EXPECT_LT(i_indeg, i_observe) << "in-degree hit 0 before GC observed it"; + EXPECT_LT(i_observe, i_delete) << "GC observed it before deleting it"; +} diff --git a/src/Disks/tests/gtest_cas_fence_generation.cpp b/src/Disks/tests/gtest_cas_fence_generation.cpp new file mode 100644 index 000000000000..be621ddd9dd0 --- /dev/null +++ b/src/Disks/tests/gtest_cas_fence_generation.cpp @@ -0,0 +1,447 @@ +#include +#include "cas_test_helpers.h" + +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; +} + +/// Task 4 (spec §1 "Gate lifetime [C2]"): every durable-effect path on the plain-object surface +/// (`CasPlainObjects::casPutObject`/`casRemoveObject`), the S3-native staging-buffer finalize +/// (`Cas::CaContentWriteBuffer`), and the part-write condemned-displacement raw writes capture the mount +/// runtime's fence generation at admission and re-check it -- and `mayMutate()` -- immediately before their +/// durable backend call, throwing the typed transient error (`NETWORK_ERROR` -- the upstream-retryable +/// class every CA write-plane transient uses) on a mismatch instead of letting a stale-incarnation write +/// land. +/// +/// These tests drive a real `Cas::Pool` over `InMemoryBackend` (the "Emulated"-style in-memory +/// backend) via `Pool::open`, exactly like `gtest_cas_mount.cpp`/`gtest_cas_s3_staging.cpp` -- the +/// fence is tripped/observed through `Pool`'s public forwarders (`tripMountLost`, `mayMutate`, +/// `fenceGeneration`, `checkFenceOrThrow`). + +using namespace DB::Cas; + +namespace +{ + +/// A backend whose `head()` call can trigger an injected side-effect exactly once -- deterministically +/// simulates a fence trip landing BETWEEN a durable-effect operation's admission and its durable +/// backend call, with no real concurrency at all (mirrors the injected-fault shape of +/// `TransportFaultBackend` in gtest_cas_sentinel_probe.cpp, but fires a callback instead of throwing). +class TripOnHeadBackend final : public InMemoryBackend +{ +public: + HeadResult head(const String & key) override + { + if (trigger) + std::exchange(trigger, {})(); + return InMemoryBackend::head(key); + } + + std::function trigger; +}; + +/// Same idea as `TripOnHeadBackend`, but fires on the SECOND `head()` call and forces a first-attempt +/// `PreconditionFailed` so the retry loop actually reaches a second iteration -- proves the fence +/// re-check runs on EVERY conditional-retry iteration, not just the admission-time first attempt. +class TripOnSecondHeadBackend final : public InMemoryBackend +{ +public: + using Backend::putIfAbsent; + + HeadResult head(const String & key) override + { + ++head_calls; + if (head_calls == 2 && trigger) + std::exchange(trigger, {})(); + return InMemoryBackend::head(key); + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (fail_first_put) + { + fail_first_put = false; + return PutResult{.outcome = PutOutcome::PreconditionFailed, .token = {}}; + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + int head_calls = 0; + /// Default false: `Pool::open`'s own capability probe issues `putIfAbsent` calls before the test + /// gets to arm this, and those must succeed normally. The test flips this to `true` only right + /// before driving the write it actually targets. + bool fail_first_put = false; + std::function trigger; +}; + +/// A minimal in-memory `WriteBufferFromFileBase` standing in for an object-store sink, trimmed to just +/// what these tests observe (whether `finalizeImpl` ran) -- mirrors `FakeStagingSink` in +/// gtest_cas_s3_staging.cpp (not reusable from here: that one lives in that file's own anonymous +/// namespace). +class RecordingSink final : public DB::WriteBufferFromFileBase +{ +public: + explicit RecordingSink(std::string key_) + : DB::WriteBufferFromFileBase(/*buf_size=*/8192, nullptr, 0), key(std::move(key_)) + { + } + + void sync() override {} + std::string getFileName() const override { return key; } + bool wasFinalizedForTest() const { return did_finalize; } + +protected: + void nextImpl() override + { + if (offset()) + written.append(working_buffer.begin(), offset()); + } + + void finalizeImpl() override + { + next(); + did_finalize = true; + } + + void cancelImpl() noexcept override { cancelled = true; } + +private: + std::string key; + std::string written; + bool did_finalize = false; + bool cancelled = false; +}; + +PoolPtr openTestPool(BackendPtr backend) +{ + return Pool::open(std::move(backend), PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +PartWriteTxnPtr precommittedBuildForBlob( + const PoolPtr & store, const RootNamespace & ns, const String & ref_name, const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref_name; + auto build = store->beginPartWrite(std::move(info)); + const ManifestId manifest = build->stageManifest( + {DB::Cas::tests::blobEntryFor("data.bin", DB::Cas::tests::u128Of(payload), payload.size())}); + build->precommitAdd(ns, ref_name, manifest); + return build; +} + +/// Trips the mount either while returning the mandatory blob `HEAD`, or immediately after an +/// unconditional publication has landed. These are the two sides of the writer's final pre-I/O fence +/// check: the first must send no publication; the second may leave equivalent debris but no proof. +class BlobPublicationFenceBackend final : public InMemoryBackend +{ +public: + enum class TripPoint : uint8_t + { + OnHead, + AfterPublication, + }; + + HeadResult head(const String & key) override + { + const HeadResult result = InMemoryBackend::head(key); + if (key == watched_key && trip_point == TripPoint::OnHead && trigger) + std::exchange(trigger, {})(); + return result; + } + + void publishBlob(const BlobPublishRequest & request) override + { + ++publish_calls; + InMemoryBackend::publishBlob(request); + if (request.destination_key == watched_key && trip_point == TripPoint::AfterPublication && trigger) + std::exchange(trigger, {})(); + } + + String watched_key; + TripPoint trip_point = TripPoint::OnHead; + std::function trigger; + size_t publish_calls = 0; +}; + +TEST(CASFenceGeneration, RearmPublishesTheNewGenerationBeforeOpeningTheFence) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const RootNamespace ns{"srv1/rearm-publication-order"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + store->tripMountLost(); + const uint64_t dead_generation = store->fenceGeneration(); + bool admitted_in_interposition = false; + store->setArmMountFenceInterpositionHookForTest([&] + { + EXPECT_EQ(store->fenceGeneration(), dead_generation + 1) + << "the fresh generation must be visible before the fence can become live"; + try + { + (void)store->namespaceLife(ns); + admitted_in_interposition = true; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + } + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u) + << "no runtime may be published in the re-arm interposition"; + }); + + store->armMountFence(DB::UInt128{0, 1}, store->writerEpoch(), store->bootMsNow() + 600000); + store->setArmMountFenceInterpositionHookForTest(nullptr); + + EXPECT_FALSE(admitted_in_interposition); + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + EXPECT_NO_THROW((void)store->namespaceLife(ns)); + EXPECT_EQ(store->refTableRuntimeAdmittedFenceGenerationForTest(ns), store->fenceGeneration()); +} + +} + +TEST(CASFenceGeneration, BlobPublicationFenceLossBeforeFinalCheckPublishesNothing) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const String payload = "fence-before-unconditional-publication"; + const BlobRef ref = DB::Cas::tests::idOf(payload); + auto build = precommittedBuildForBlob(store, RootNamespace{"srv1/fence-before"}, "part", payload); + backend->watched_key = store->layout().blobKey(ref); + backend->trip_point = BlobPublicationFenceBackend::TripPoint::OnHead; + backend->trigger = [&] { store->tripMountLost(); }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->putBlob(ref, BlobSource::fromString(payload)); + }); + + EXPECT_EQ(backend->publish_calls, 0u); + EXPECT_FALSE(backend->head(backend->watched_key).exists); + EXPECT_EQ(build->dependencyProof(ref), std::nullopt); +} + +TEST(CASFenceGeneration, BlobPublicationHeadTripAndRearmCannotAdoptNewFenceGeneration) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const String payload = "fence-trip-and-rearm-during-head"; + const BlobRef ref = DB::Cas::tests::idOf(payload); + auto build = precommittedBuildForBlob(store, RootNamespace{"srv1/fence-rearm-during-head"}, "part", payload); + backend->watched_key = store->layout().blobKey(ref); + backend->trip_point = BlobPublicationFenceBackend::TripPoint::OnHead; + const uint64_t admitted_generation = store->fenceGeneration(); + backend->trigger = [&] + { + store->tripMountLost(); + DB::Cas::tests::rearmMountFenceAfterAnomalyForTest(store); + EXPECT_TRUE(store->mayMutate()); + EXPECT_NE(store->fenceGeneration(), admitted_generation); + }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->putBlob(ref, BlobSource::fromString(payload)); + }); + + EXPECT_EQ(backend->publish_calls, 0u); + EXPECT_FALSE(backend->head(backend->watched_key).exists); + EXPECT_EQ(loadMeta(*backend, store->layout(), ref), std::nullopt) + << "the stale operation must not reconcile freshness metadata after trip-and-rearm"; + EXPECT_EQ(build->dependencyProof(ref), std::nullopt); +} + +TEST(CASFenceGeneration, BlobPublicationFenceLossAfterLandingReturnsNoProof) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const String payload = "fence-after-unconditional-publication"; + const BlobRef ref = DB::Cas::tests::idOf(payload); + auto build = precommittedBuildForBlob(store, RootNamespace{"srv1/fence-after"}, "part", payload); + backend->watched_key = store->layout().blobKey(ref); + backend->trip_point = BlobPublicationFenceBackend::TripPoint::AfterPublication; + backend->trigger = [&] { store->tripMountLost(); }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->putBlob(ref, BlobSource::fromString(payload)); + }); + + EXPECT_EQ(backend->publish_calls, 1u); + EXPECT_TRUE(backend->head(backend->watched_key).exists) + << "a publication that landed before fence loss is safe unreferenced debris"; + EXPECT_EQ(build->dependencyProof(ref), std::nullopt); +} + +/// (a) `casPutObject` (reached via `Pool::putNamespaceFile`) with the fence tripped BETWEEN admission +/// and the durable PUT: the typed transient refusal, and the object is never actually written. +TEST(CASFenceGeneration, PlainObjectPutAbortsWhenFenceTripsBetweenAdmissionAndDurableCall) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + ASSERT_TRUE(store->mayMutate()); + + const RootNamespace ns{"test/ns"}; + backend->trigger = [&] { store->tripMountLost(); }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "somefile", "hello"); + }); + + /// No durable write ever landed -- assert via the Emulated backend listing. + EXPECT_TRUE(store->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)).empty()); +} + +/// `casRemoveObject`'s delete sibling, same shape: the fence trips between admission and the durable +/// delete, so the victim object survives untouched. +TEST(CASFenceGeneration, PlainObjectRemoveAbortsWhenFenceTripsBetweenAdmissionAndDurableCall) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const RootNamespace ns{"test/ns"}; + + /// Seed the victim BEFORE arming the trigger -- the seeding write itself must not trip the fence. + store->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "victim", "still here"); + ASSERT_TRUE(store->mayMutate()); + + backend->trigger = [&] { store->tripMountLost(); }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->removeNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "victim"); + }); + + /// The durable delete never ran -- the object survives (reads are not fence-gated by this task). + const auto still_there = store->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "victim"); + ASSERT_TRUE(still_there.has_value()); + EXPECT_EQ(*still_there, "still here"); +} + +/// The fence re-check must run before EVERY conditional-retry iteration, not just the first attempt +/// (spec wording, verbatim): a synthetic `PreconditionFailed` forces a second loop iteration, and the +/// fence trips on the SECOND `head()` call. If the check ran only once, at admission, this write would +/// incorrectly succeed on the retry. +TEST(CASFenceGeneration, PlainObjectPutRechecksFenceOnEveryRetryIterationNotJustFirst) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + ASSERT_TRUE(store->mayMutate()); + + const RootNamespace ns{"test/ns"}; + backend->head_calls = 0; /// reset past whatever `Pool::open`'s own probe/mount claim already did + backend->fail_first_put = true; + backend->trigger = [&] { store->tripMountLost(); }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "somefile", "hello"); + }); + + EXPECT_EQ(backend->head_calls, 2); + EXPECT_TRUE(store->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)).empty()); +} + +/// (b) The S3-native staging-buffer finalize: the fence trips AFTER the buffer is constructed +/// (admission) but BEFORE `finalize()` reaches the durable `sink->finalize()` call -- same typed abort, +/// and the sink is never actually finalized (`on_finalized` never fires either, so the transaction +/// never learns of a promote-worthy hash/size for bytes that were never durable). +TEST(CASFenceGeneration, S3StagingFinalizeAbortsWhenFenceTripsBeforeDurableCall) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + ASSERT_TRUE(store->mayMutate()); + + const std::string staging_key = "staging/mount1/racer.tmp"; + auto * sink_ptr = new RecordingSink(staging_key); + std::unique_ptr sink(sink_ptr); + + bool on_finalized_called = false; + const uint64_t admitted_generation = store->fenceGeneration(); + + auto buf = std::make_unique( + std::move(sink), + staging_key, + /*envelope_header=*/std::string(), + BlobHashAlgo::CityHash128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string &, size_t, const std::string &) { on_finalized_called = true; }, + [store, admitted_generation] { store->checkFenceOrThrow(admitted_generation); }); + + const std::string payload = "some bytes that must never become durable"; + buf->write(payload.data(), payload.size()); + + /// The race this test targets: admission already captured `admitted_generation` above, and now the + /// fence trips before `finalize()` runs. + store->tripMountLost(); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { buf->finalize(); }); + + EXPECT_FALSE(on_finalized_called); + EXPECT_FALSE(sink_ptr->wasFinalizedForTest()); +} + +/// (d) Happy path unchanged: an ordinary plain-object write/read/remove, and an ordinary S3-staging +/// finalize, both succeed exactly as before when the fence stays live throughout. +TEST(CASFenceGeneration, HappyPathPlainObjectWriteReadRemoveUnaffected) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + const RootNamespace ns{"test/ns"}; + + store->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "a", "hello"); + const auto got = store->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "a"); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(*got, "hello"); + + store->removeNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "a"); + EXPECT_FALSE(store->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "a").has_value()); +} + +TEST(CASFenceGeneration, HappyPathS3StagingFinalizeUnaffected) +{ + auto backend = std::make_shared(); + auto store = openTestPool(backend); + + const std::string staging_key = "staging/mount1/happy.tmp"; + auto * sink_ptr = new RecordingSink(staging_key); + std::unique_ptr sink(sink_ptr); + + bool on_finalized_called = false; + const uint64_t admitted_generation = store->fenceGeneration(); + + auto buf = std::make_unique( + std::move(sink), + staging_key, + /*envelope_header=*/std::string(), + BlobHashAlgo::CityHash128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string &, size_t, const std::string &) { on_finalized_called = true; }, + [store, admitted_generation] { store->checkFenceOrThrow(admitted_generation); }); + + const std::string payload = "unaffected happy path bytes"; + buf->write(payload.data(), payload.size()); + buf->finalize(); + + EXPECT_TRUE(on_finalized_called); + EXPECT_TRUE(sink_ptr->wasFinalizedForTest()); +} diff --git a/src/Disks/tests/gtest_cas_fold_seal_codec.cpp b/src/Disks/tests/gtest_cas_fold_seal_codec.cpp new file mode 100644 index 000000000000..feee48a70069 --- /dev/null +++ b/src/Disks/tests/gtest_cas_fold_seal_codec.cpp @@ -0,0 +1,41 @@ +#include +#include + +using namespace DB::Cas; + +/// The GC-reclaim tests that used to live here (`AbandonedPrecommitOrphansManifestUntilFix`, +/// `ReclaimIsIdempotentAndSelfTerminating`, `SkipPreservedForLivePrecommitAndForNoPrecommit`, +/// `DoubleRemovalOfReclaimedPrecommitIsIdempotent`) were removed with the snapshot+log ref model. +/// They asserted that GC reclaims an abandoned precommit once the mount watermark proves it dead, and +/// that the token-diff Skip optimization self-terminates. Per spec §Responsibility Boundary, reclaiming +/// an abandoned precommit is now the WRITER's job (it appends the exact `owner_transition` removal), and +/// the token-diff Skip machinery (`computeDiscoverDecisions`/`discoverDecisionsForTest`) no longer exists +/// -- the "did it change" signal is simply logs above the durable cursor. There is no GC-side reclaim to +/// assert, so these tests are obsolete rather than adaptable. +/// +/// The live-precommit watermark fields (`has_live_precommit`/`min_live_precommit_*`) that fed that +/// removed reclaim were deleted from `RefCoverage` with it (T13). The still-meaningful fold-seal +/// assertion is the round-trip of `last_folded_ref_id` -- the per-table durable ref cursor that replaced +/// them in the same struct under the snapshot+log ref model. +TEST(CASFoldSealCodec, RefLifeCoverageRoundTripsLastFoldedRefId) +{ + CasFoldSeal seal; + seal.generation = 3; + seal.parent_generation = 2; + RefCoverage cov; + cov.classification = 1; + cov.last_folded_ref_id = RefTxnId{4, 11}; + constexpr UInt128 life_id{1}; + seal.ref_lives[life_id].coverage = cov; + + const CasFoldSeal back = decodeFoldSeal(encodeFoldSeal(seal)); + const RefCoverage & r = back.ref_lives.at(life_id).coverage; + EXPECT_EQ(r.last_folded_ref_id, (RefTxnId{4, 11})); + + /// Default (nothing folded) round-trips as {0,0}. + CasFoldSeal empty_seal; + constexpr UInt128 empty_life_id{2}; + empty_seal.ref_lives[empty_life_id].coverage = RefCoverage{}; + const CasFoldSeal e_back = decodeFoldSeal(encodeFoldSeal(empty_seal)); + EXPECT_EQ(e_back.ref_lives.at(empty_life_id).coverage.last_folded_ref_id, (RefTxnId{})); +} diff --git a/src/Disks/tests/gtest_cas_fold_seal_format.cpp b/src/Disks/tests/gtest_cas_fold_seal_format.cpp new file mode 100644 index 000000000000..afe6331a4ed8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_fold_seal_format.cpp @@ -0,0 +1,327 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int LOGICAL_ERROR; } + +namespace +{ +CasFoldSeal sampleFoldSeal() +{ + CasFoldSeal seal; + seal.generation = 7; + seal.parent_generation = 6; + seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{3, 4}}; + seal.ref_lives[UInt128{2}].coverage = RefCoverage{.classification = 1}; + seal.blob_target_runs.push_back(RunRef{.key = "gc/gen/7/blob_target/0/0", .checksum = UInt128(0xABCDEF)}); + return seal; +} + +void eraseRequiredField(String & encoded, std::string_view field) +{ + const size_t pos = encoded.find(field); + ASSERT_NE(pos, String::npos); + encoded.erase(pos, field.size()); +} +} + +TEST(CASFormatBattery, FoldSeal) +{ + CasFoldSeal seal; + seal.generation = 5; + seal.parent_generation = 4; + seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{7, 11}}; + seal.blob_target_runs.push_back(RunRef{.key = "r0", .checksum = UInt128(0x0f), .shard = 0, .generation = 5}); + seal.condemned_summary[0] = CondemnedSummary{.condemned_total = 3, .pending_total = 1, + .oldest_nonpending_condemn_round = 4}; + runFormatBattery({FormatId::FoldSeal, + [&] { return sealObject(FormatId::FoldSeal, encodeFoldSeal(seal)); }, + [](std::string_view s) { decodeFoldSeal(std::string(openObject(FormatId::FoldSeal, s))); }, + currentFormatHeader("cas_fold_seal") + + "{\"g\":\"5\",\"pg\":\"4\"}\n" + "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":2,\"lfe\":\"7\",\"lfs\":\"11\"}\n" + "{\"k\":\"btr\",\"key\":\"r0\",\"ck\":\"0000000000000000000000000000000f\",\"shard\":0,\"gen\":\"5\"}\n" + "{\"k\":\"cnd\",\"shard\":0,\"ct\":3,\"pt\":1,\"ocr\":\"4\"}\n" + "{\"n\":3}\n"}); +} + +TEST(CASFoldSealFormat, RoundTripsAllFields) +{ + const CasFoldSeal in = sampleFoldSeal(); + const CasFoldSeal out = decodeFoldSeal(encodeFoldSeal(in)); + + EXPECT_EQ(out.generation, in.generation); + EXPECT_EQ(out.parent_generation, in.parent_generation); + ASSERT_EQ(out.ref_lives.size(), in.ref_lives.size()); + EXPECT_EQ(out.ref_lives.at(UInt128{1}).coverage.classification, 2); + EXPECT_EQ(out.ref_lives.at(UInt128{1}).coverage.last_folded_ref_id, (RefTxnId{3, 4})); + ASSERT_EQ(out.blob_target_runs.size(), 1u); + EXPECT_EQ(out.blob_target_runs[0].key, "gc/gen/7/blob_target/0/0"); + EXPECT_EQ(out.blob_target_runs[0].checksum, UInt128(0xABCDEF)); + EXPECT_EQ(out, in); +} + +TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsTwoBlobTargetRunsForOneShard) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.generation = 7; + seal.parent_generation = 6; + seal.blob_target_runs = { + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}, + }; + seal.condemned_summary[0] = CondemnedSummary{}; + + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, /*gc_shards=*/1); }, + "duplicate blob-target shard"); +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASFoldSealFormatDeathTest, ProducerValidationRejectsMalformedSealBeforePut) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.blob_target_runs = { + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}}; + seal.condemned_summary[0] = CondemnedSummary{}; + EXPECT_DEATH({ validateFoldSealForWrite(seal, layout, 1); }, "duplicate blob-target shard"); +} +#else +TEST(CASFoldSealFormat, ProducerValidationRejectsMalformedSealBeforePut) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.blob_target_runs = { + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}}; + seal.condemned_summary[0] = CondemnedSummary{}; + cas_battery_detail::expectCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { validateFoldSealForWrite(seal, layout, 1); }, "duplicate blob-target shard"); +} +#endif + +TEST(CASFoldSealFormat, AuthoritativeDecodeRequiresEveryBlobTargetAndSummaryField) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.generation = 7; + seal.parent_generation = 6; + seal.blob_target_runs.push_back(RunRef{ + .key = layout.blobTargetRunKey(7, 1, 0, 0), + .checksum = UInt128{1}, + .shard = 0, + .generation = 7}); + seal.condemned_summary[0] = CondemnedSummary{}; + const String valid = encodeFoldSeal(seal); + + for (const std::string_view field : { + R"(,"key":"p/gc/gen/7/attempt/1/blob_target/0/0")", + R"(,"ck":"00000000000000000000000000000001")", + R"(,"gen":"7")", + ",\"ct\":0", + ",\"pt\":0", + R"(,"ocr":"18446744073709551615")"}) + { + String malformed = valid; + eraseRequiredField(malformed, field); + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(malformed, layout, 1); }, "missing"); + } + + /// `shard` occurs once on each row; remove each occurrence independently. + String missing_btr_shard = valid; + eraseRequiredField(missing_btr_shard, ",\"shard\":0"); + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(missing_btr_shard, layout, 1); }, "missing"); + + String missing_cnd_shard = valid; + const size_t first_shard = missing_cnd_shard.find(",\"shard\":0"); + ASSERT_NE(first_shard, String::npos); + const size_t second_shard = missing_cnd_shard.find(",\"shard\":0", first_shard + 1); + ASSERT_NE(second_shard, String::npos); + missing_cnd_shard.erase(second_shard, std::string_view(",\"shard\":0").size()); + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(missing_cnd_shard, layout, 1); }, "missing"); +} + +TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsNoncanonicalRowsAndIncompleteSummaryDomain) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.generation = 7; + seal.parent_generation = 6; + seal.blob_target_runs.push_back(RunRef{ + .key = layout.blobTargetRunKey(7, 1, 1, 0), + .checksum = UInt128{1}, + .shard = 1, + .generation = 7}); + seal.condemned_summary[0] = CondemnedSummary{}; + + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "outside"); + + seal.blob_target_runs[0].shard = 0; + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "not canonical"); + + seal.blob_target_runs.clear(); + seal.condemned_summary.clear(); + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "exactly 1"); + + seal.condemned_summary[0] = CondemnedSummary{}; + seal.condemned_summary[1] = CondemnedSummary{}; + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "exactly 1"); +} + +TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsContradictorySummaryCounts) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.condemned_summary[0] = CondemnedSummary{ + .condemned_total = 1, + .pending_total = 2, + .oldest_nonpending_condemn_round = 3}; + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "greater than"); + + seal.condemned_summary[0] = CondemnedSummary{ + .condemned_total = 2, + .pending_total = 1, + .oldest_nonpending_condemn_round = std::numeric_limits::max()}; + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encodeFoldSeal(seal), layout, 1); }, "real oldest"); +} + +TEST(CASFoldSealFormat, RejectsUnexpectedGeneration) +{ + CasFoldSeal seal; + seal.generation = 5; + const String encoded = encodeFoldSeal(seal); + + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(encoded, /*expected_generation=*/6); }, "unexpected generation"); + EXPECT_EQ(decodeFoldSeal(encoded, /*expected_generation=*/5).generation, 5); + EXPECT_EQ(decodeFoldSeal(encoded).generation, 5); +} + +TEST(CASFoldSeal, EncodingIsByteDeterministic) +{ + const CasFoldSeal in = sampleFoldSeal(); + EXPECT_EQ(encodeFoldSeal(in), encodeFoldSeal(in)); +} + +TEST(CASFoldSealFormat, TextIsByteDeterministic) +{ + CasFoldSeal a; + a.generation = 5; + a.parent_generation = 4; + a.blob_target_runs = {RunRef{"z", UInt128(2), 1, 5}, RunRef{"a", UInt128(1), 0, 5}}; + CasFoldSeal b = a; + std::reverse(b.blob_target_runs.begin(), b.blob_target_runs.end()); /// same set, different order + EXPECT_EQ(encodeFoldSeal(a), encodeFoldSeal(b)); /// encoder must sort runs by key +} + +TEST(CASFoldSeal, RejectsEmptyAndBadMagic) +{ + EXPECT_ANY_THROW(decodeFoldSeal("")); + EXPECT_ANY_THROW(decodeFoldSeal("not-a-seal")); +} + +TEST(CASFoldSeal, CoverageRecordsEveryCatalogLife) +{ + CasFoldSeal in = sampleFoldSeal(); + in.ref_lives[UInt128{3}].coverage = RefCoverage{.classification = 0}; + const CasFoldSeal out = decodeFoldSeal(encodeFoldSeal(in)); + EXPECT_TRUE(out.ref_lives.contains(UInt128{3})); + EXPECT_EQ(out.ref_lives.size(), 3u); +} + +TEST(CASFoldSeal, FoldSealCondemnedSummaryRoundTrips) +{ + /// A seal carrying a non-empty condemned_summary over 2 shards (one a zero entry) round-trips and + /// compares equal, and the UINT64_MAX "none" sentinel survives. + CasFoldSeal s; + s.generation = 9; + s.parent_generation = 8; + s.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2}; + s.blob_target_runs.push_back(RunRef{.key = "gc/gen/9/blob_target/0/0", .checksum = UInt128(0x77), + .shard = 0, .generation = 9}); + s.condemned_summary[0] = CondemnedSummary{.condemned_total = 3, .pending_total = 1, + .oldest_nonpending_condemn_round = 5}; + s.condemned_summary[1] = CondemnedSummary{}; /// explicit zero entry (totality over gc_shards) + + const CasFoldSeal out = decodeFoldSeal(encodeFoldSeal(s)); + EXPECT_EQ(out, s); + ASSERT_EQ(out.condemned_summary.size(), 2u); + EXPECT_EQ(out.condemned_summary.at(0).condemned_total, 3u); + EXPECT_EQ(out.condemned_summary.at(0).pending_total, 1u); + EXPECT_EQ(out.condemned_summary.at(0).oldest_nonpending_condemn_round, 5u); + EXPECT_EQ(out.condemned_summary.at(1).oldest_nonpending_condemn_round, + std::numeric_limits::max()); /// UINT64_MAX sentinel survives + + EXPECT_TRUE(decodeFoldSeal(encodeFoldSeal(CasFoldSeal{})).condemned_summary.empty()); +} + +/// Mutation caught: restoring separate `cov` and `nsc` rows, dropping the cleanup evidence, or +/// serializing the row under a logical namespace changes these literal generation-8 bytes. +TEST(CASFoldSealFormat, UnifiedRefLifeRowRoundTripsCoverageHoldAndCleanupEvidence) +{ + CasFoldSeal seal; + seal.generation = 8; + seal.parent_generation = 7; + const UInt128 life_id{0x1234}; + seal.ref_lives.emplace(life_id, RefLifeFoldState{ + .coverage = RefCoverage{ + .classification = 4, + .last_folded_ref_id = RefTxnId{3, 4}, + .hold = RefHold{ + .reason = HoldReason::ManifestBodyMissing, + .offending_position = RefTxnId{5, 6}, + .retry_count = 7, + .next_retry_round = 8}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{9, 10}}}); + + const String expected = currentFormatHeader("cas_fold_seal") + + "{\"g\":\"8\",\"pg\":\"7\"}\n" + "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000001234\",\"cls\":4," + "\"lfe\":\"3\",\"lfs\":\"4\",\"hr\":\"manifest_body_missing\",\"hpe\":\"5\"," + "\"hps\":\"6\",\"hrc\":7,\"hnr\":\"8\",\"rte\":\"9\",\"rts\":\"10\"}\n" + "{\"n\":1}\n"; + + EXPECT_EQ(encodeFoldSeal(seal), expected); + EXPECT_EQ(decodeFoldSeal(expected), seal); +} + +/// Mutation caught: accepting the generation-6 split coverage collection would leave a second +/// namespace-keyed source of lifecycle work in a generation-7 process. +TEST(CASFoldSealFormat, UnifiedCodecRejectsLegacyCoverageRecord) +{ + const String old = + "{\"type\":\"cas_fold_seal\",\"v\":7}\n" + "{\"g\":\"8\",\"pg\":\"7\"}\n" + "{\"k\":\"cov\",\"key\":\"name/0\",\"cls\":2,\"lfe\":\"3\",\"lfs\":\"4\"}\n" + "{\"n\":1}\n"; + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(old); }, "legacy coverage"); +} + +/// Mutation caught: accepting the generation-6 cleanup-item state would restore the independent +/// marker-driven `Pending`/`Completed` handshake. +TEST(CASFoldSealFormat, UnifiedCodecRejectsLegacyNamespaceCleanupRecord) +{ + const String old = + "{\"type\":\"cas_fold_seal\",\"v\":7}\n" + "{\"g\":\"8\",\"pg\":\"7\"}\n" + "{\"k\":\"nsc\",\"ns\":\"name\",\"rte\":\"3\",\"rts\":\"4\",\"st\":\"completed\"}\n" + "{\"n\":1}\n"; + cas_battery_detail::expectCode( + DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(old); }, "legacy namespace cleanup"); +} diff --git a/src/Disks/tests/gtest_cas_forget.cpp b/src/Disks/tests/gtest_cas_forget.cpp new file mode 100644 index 000000000000..2d8a72fc1850 --- /dev/null +++ b/src/Disks/tests/gtest_cas_forget.cpp @@ -0,0 +1,599 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Task 10 (rev.7 spec §5): `SYSTEM CAS FORGET` — the operator force-Vanish. FORGET drives a +/// content-addressed pool to `Vanished(forgotten)` with the fence-first protocol: (1) publish terminal +/// intent, (2) trip the local fence, (3+4) stop the GC scheduler, (5) join keeper/remount, drain, retire +/// the keeper WITHOUT an unearned clean farewell, (6) publish `Vanished(forgotten)` with the [D5] message +/// carrying the decommission timestamp. These tests exercise the Pool-level protocol body (`Pool::forgetDisk`) +/// and the end-to-end verb through a real `ContentAddressedMetadataStorage` (the six-class gate wired to the +/// new state). Harness patterns follow gtest_cas_lifecycle_condition.cpp and gtest_cas_operation_gate.cpp. + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +} + +using namespace DB; +using DB::Cas::PoolLifecycle; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +const String kSrid = "test"; + +/// A test-authored [D5] reason with a RECOGNIZABLE timestamp — the Pool-level tests assert this exact +/// string flows through `enterVanished` into the `throwIfLifecycleTerminal` message (the timestamp +/// threading the metadata storage does in production). It keeps the two [D5] substrings the gate relies on. +const String kForgetReason = + "decommissioned by SYSTEM CAS FORGET at 2099-01-02 03:04:05 UTC — erasure was NOT " + "verified; if this was a mistake the data may be intact (restart re-registers the name)"; + +/// Delete an existing key exactly (its current token comes from the same GET). Mirrors +/// gtest_cas_lifecycle_condition.cpp — used to drive a live pool into `IdentityLost`. +void deleteKeyExact(DB::Cas::Backend & backend, const String & key) +{ + const auto got = backend.get(key); + ASSERT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; + if (got) + backend.deleteExact(key, got->token); +} + +/// GC's fence-out applied directly to the mount lease (preserve the body, set `gc_fenced`, bump `seq`) — +/// a subsequent `tryRemountOnce` verdicts `Recover` and reclaims a FRESH incarnation immediately (no +/// lease-expiry wait), reaching `armMountFence`. Mirrors gtest_cas_lifecycle_condition.cpp's helper. +void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, + DB::Cas::PutOutcome::Done); +} + +/// A Backend decorator whose head/get/list throw an untyped transport error while `fail` is armed — so a +/// self-remount attempt verdicts `StayTransient` (fast, no lease-expiry wait) and the remount loop keeps +/// spinning. Starts DISARMED so `Pool::open` succeeds. Mirrors gtest_cas_lifecycle_condition.cpp's decorator. +class ToggleableTransportFaultBackend final : public DB::Cas::InMemoryBackend +{ +public: + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + DB::Cas::HeadResult head(const String & key) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::head(key); + } + std::optional get(const String & key, DB::Cas::Range range) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::get(key, range); + } + DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::list(prefix, cursor, limit); + } + + std::atomic fail{false}; +}; + +/// The message thrown by `fn`, or a failure if it did not throw a `DB::Exception`. +std::string messageOf(const std::function & fn) +{ + try + { + fn(); + } + catch (const Exception & e) + { + return std::string(e.message()); + } + ADD_FAILURE() << "expected a DB::Exception"; + return {}; +} + +/// A live table dir + committed part reused by the end-to-end gate test (the shape +/// gtest_cas_operation_gate.cpp uses). +const std::string kTableDir = "gg0/gg0gg0g0-0808-4808-8808-080808080808"; +const std::string kPartDir = kTableDir + "/all_1_1_0"; +const std::string kPartFile = kPartDir + "/data.bin"; + +std::shared_ptr openForgetStorage() +{ + auto settings = Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_forget_scratch"); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +void commitOnePart(ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kTableDir + "/tmp_insert_all_1_1_0/data.bin", 65536, WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kTableDir + "/tmp_insert_all_1_1_0", kPartDir); + tx->commit(NoCommitOptions{}); +} + +/// Deterministically interleave a real FORGET into the admission->lock window of a manual GC verb (the +/// I-1/I-2 admission TOCTOU), with BOUNDED condition-variable waits and never a sleep. The sequence pinned: +/// M (this thread, running `gc_verb`): passes the verb's pre-lock admission gate while `Live`, then the +/// installed seam signals `admitted` and blocks until `forget_done`, then M resumes to acquire +/// `gc_scheduler_mutex` and hit the under-lock re-check. +/// F (the FORGET thread): waits for `admitted`, runs the REAL `forgetDisk` (acquiring lifecycle + +/// gc_scheduler mutexes while M holds NEITHER -- M is parked in the seam BEFORE the lock), settling the +/// pool `Vanished(forgotten)`, then signals `forget_done`. +/// Returns the exception message `gc_verb` threw (via `messageOf`), so the caller asserts the typed [D5] +/// refusal. The 30s bounds trip ONLY on a genuine deadlock regression, never in the happy path. +std::string raceForgetIntoGcVerbWindow(ContentAddressedMetadataStorage & storage, + const std::function & gc_verb) +{ + std::mutex m; + std::condition_variable cv; + bool admitted = false; + bool forget_done = false; + + storage.setGcVerbAdmitWindowHookForTest([&] + { + { + std::lock_guard lk(m); + admitted = true; + } + cv.notify_all(); + std::unique_lock lk(m); + EXPECT_TRUE(cv.wait_for(lk, std::chrono::seconds(30), [&] { return forget_done; })) + << "the concurrent FORGET must complete within the bound (else the interleave deadlocked)"; + }); + + std::thread forgetter([&] + { + { + std::unique_lock lk(m); + EXPECT_TRUE(cv.wait_for(lk, std::chrono::seconds(30), [&] { return admitted; })) + << "the GC verb must reach the admission->lock window before FORGET runs"; + } + storage.forgetDisk(); + { + std::lock_guard lk(m); + forget_done = true; + } + cv.notify_all(); + }); + + const std::string msg = messageOf(gc_verb); + forgetter.join(); + storage.setGcVerbAdmitWindowHookForTest({}); /// clear the seam (references this frame's locals) + return msg; +} + +} + +/// (a) FORGET on a LIVE pool: the local fence is tripped, the injected GC-stop step runs, the pool settles +/// `Vanished(forgotten)`, and store-class access fails loud with the timestamped [D5] message. +TEST(CASForget, ForgetOnLivePoolTripsFenceAndVanishes) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + ASSERT_TRUE(store->mayMutate()); + + bool gc_stopped = false; + store->forgetDisk([&] { gc_stopped = true; }, kForgetReason); + + /// Step 3/4 ran (the GC-stop callback was invoked from inside the protocol). + EXPECT_TRUE(gc_stopped); + /// Terminal truth, fence tripped. + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_TRUE(store->isVanished()); + EXPECT_FALSE(store->mayMutate()); + + /// The [D5] message carries the operator's FORGET timestamp (threaded through the reason) and still + /// names the sub-state ("erasure was NOT verified"). + const std::string msg = messageOf([&] { store->throwIfLifecycleTerminal(); }); + EXPECT_NE(msg.find("SYSTEM CAS FORGET at "), std::string::npos) << msg; + EXPECT_NE(msg.find("2099-01-02 03:04:05 UTC"), std::string::npos) << msg; + EXPECT_NE(msg.find("erasure was NOT verified"), std::string::npos) << msg; +} + +/// (a') FORGET stops AND joins a real `CasGcScheduler`'s worker + heartbeat threads (the injected GC-stop +/// step). A long interval keeps any round from firing during the test window, so this isolates the +/// thread-lifecycle: `start()` spawns the two workers, FORGET's callback `stop()`s + joins them, and the +/// test completing (no hang) plus a clean `isQuiescent()` proves the join. +TEST(CASForget, ForgetStopsAndJoinsRealGcScheduler) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + Cas::CasGcScheduler sched(store, std::chrono::seconds(3600), "CasForgetTest", "forget-disk"); + sched.start(); + + bool gc_joined = false; + store->forgetDisk([&] { sched.stop(); gc_joined = true; }, kForgetReason); + + EXPECT_TRUE(gc_joined); + /// What this proves is the JOIN: `stop()` returned, so the worker + heartbeat threads are joined and + /// the test could not have hung; `isQuiescent()` confirms no round is in flight. NOTE: the callback is + /// only `sched.stop()`, which does NOT itself clear the in-process `i_am_leader` hint — the + /// metadata-storage handler clears leadership by DESTROYING the scheduler (see + /// `ContentAddressedMetadataStorage::forgetDisk`), so asserting `is_leader == false` here would be + /// vacuous (this 3600s scheduler never led) or, after a real round, wrong. + EXPECT_TRUE(sched.isQuiescent()) << "no GC round may be in flight after FORGET joined the scheduler"; + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); +} + +/// (c) Double FORGET is idempotent: the second call is a no-op (the pool is already `Vanished(forgotten)`), +/// so it never re-runs the protocol — the GC-stop callback is NOT invoked again, and the first reason wins. +TEST(CASForget, DoubleForgetIsIdempotent) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + int gc_stops = 0; + store->forgetDisk([&] { ++gc_stops; }, kForgetReason); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + ASSERT_EQ(gc_stops, 1); + + /// A second FORGET with a DIFFERENT reason must change nothing (first terminal transition wins) and + /// must NOT re-enter the teardown (idempotent short-circuit on `isVanished()`). + store->forgetDisk([&] { ++gc_stops; }, "a different reason that must be ignored"); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_EQ(gc_stops, 1) << "the idempotent second FORGET must not re-run the protocol"; + + const std::string msg = messageOf([&] { store->throwIfLifecycleTerminal(); }); + EXPECT_NE(msg.find("2099-01-02 03:04:05 UTC"), std::string::npos) + << "the first FORGET's reason must win: " << msg; +} + +/// (d) FORGET on an `IdentityLost` pool → `Vanished(forgotten)` — the escape hatch. `IdentityLost` is +/// non-absorbing and has no benign answer, so FORGET is the operator's way out. +TEST(CASForget, ForgetOnIdentityLostPoolVanishesForgotten) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + /// Delete both pool sentinels while other objects remain, then drive the identity gate: the pool enters + /// `IdentityLost` (never `Vanished`) — exactly gtest_cas_lifecycle_condition.cpp scenario (a). + deleteKeyExact(*backend, store->layout().poolMetaKey()); + deleteKeyExact(*backend, store->layout().ownerKey(kSrid)); + EXPECT_FALSE(store->tryRemountOnce()); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + ASSERT_FALSE(store->isVanished()); + + bool gc_stopped = false; + store->forgetDisk([&] { gc_stopped = true; }, kForgetReason); + + EXPECT_TRUE(gc_stopped); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_TRUE(store->isVanished()); +} + +/// (a'') The clean-farewell is EARNED, never unconditional: on a drained pool FORGET stamps the mount lease +/// with the terminated sentinel (`min_active == UINT64_MAX`) so a same-server restart reclaims immediately, +/// but with an UNSETTLED (wedged) ref lane it must NOT — the lease is left to expire by observation. +TEST(CASForget, ForgetCleanFarewellGatedOnDrain) +{ + using DB::Cas::decodeMountLease; + constexpr uint64_t kTerminated = std::numeric_limits::max(); + + /// Drained pool → clean farewell written (lease stamped terminated). + { + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const String mount_key = store->layout().mountKey(kSrid); + ASSERT_NE(decodeMountLease(backend->get(mount_key)->bytes).min_active, kTerminated); /// baseline + + store->forgetDisk([] {}, kForgetReason); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + + const auto got = backend->get(mount_key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(decodeMountLease(got->bytes).min_active, kTerminated) + << "a drained FORGET earns the clean-release farewell"; + } + + /// Unsettled (wedged) ref lane → NO clean farewell (the drain cannot certify a clean death). + { + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const String mount_key = store->layout().mountKey(kSrid); + + const DB::Cas::RootNamespace ns{"test/forget_wedge"}; + store->forceWedgeForTest(ns, /*writer_epoch*/ 1, /*ref_sequence*/ 1, "bogus/_log/key", "bogus-bytes"); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + store->forgetDisk([] {}, kForgetReason); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + + const auto got = backend->get(mount_key); + ASSERT_TRUE(got.has_value()) << "the lease object must still be present (expiry by observation)"; + EXPECT_NE(decodeMountLease(got->bytes).min_active, kTerminated) + << "an unearned clean farewell must NOT be written when the ref lanes did not drain"; + } +} + +/// (b1) BOUNDED COMPLETION: FORGET racing an ACTIVE persistent remount worker joins it without deadlock. Here the +/// faulting backend keeps every attempt at `StayTransient` (it never reaches `armMountFence`), so this +/// isolates the join/no-deadlock property; the fence re-arm path is covered by (b2) below. Uses a +/// `std::future` timeout wait (never a sleep) — the timeout only fires on a genuine deadlock regression. +TEST(CASForget, ForgetRacingActiveRemountThreadCompletesBounded) +{ + auto backend = std::make_shared(); + /// `background_watermark = true` so the persistent recovery worker exists (mirrors + /// gtest_cas_pool.cpp's ShutdownGuardRefusesToArmRemount setup). + auto store = DB::Cas::Pool::open(backend, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); + + /// Arm the fault so every remount attempt verdicts `StayTransient` fast (no lease-expiry wait), then + /// trip the fence and latch a recovery generation — the worker now loops `tryRemountOnce` against the fault. + backend->fail.store(true); + store->tripMountLost(); + ASSERT_TRUE(store->scheduleRemountForTest()) << "the recovery worker must accept the request and run"; + + /// FORGET from ANOTHER thread must join the active remount worker and finish in bounded time. + std::promise done; + auto fut = done.get_future(); + std::thread forgetter([&] + { + store->forgetDisk([] {}, kForgetReason); + done.set_value(); + }); + EXPECT_EQ(fut.wait_for(std::chrono::seconds(30)), std::future_status::ready) + << "FORGET must not deadlock against an in-flight self-remount"; + forgetter.join(); + + /// Disarm before ~Pool so its residual teardown is not fighting the injected fault. + backend->fail.store(false); + + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_FALSE(store->mayMutate()) << "the fence must stay latched even if a raced reclaim re-armed it"; +} + +/// (b2) FENCE RE-LATCH REGRESSION GUARD (the fix's raison d'être): a self-remount that reaches +/// `armMountFence` re-arms the local fence (`lost=false`) after FORGET has already tripped it. FORGET's +/// SECOND `tripMountLost` — placed AFTER the remount worker is joined — must override it. +/// +/// (b1)'s fault keeps every attempt at `StayTransient`, so it can NOT catch removal of that second trip. To +/// make EXACTLY ONE reclaim reach `armMountFence` inside FORGET's window, deterministically and without a +/// sleep, we drive a REAL `tryRemountOnce` from FORGET's own GC-stop step (invoked at spec §5 step 3/4, +/// strictly AFTER the fence trip): the mount is fenced-out so the reclaim succeeds fast and re-arms the +/// fence, and `tryRemountOnce`'s step-0 gate checks `isVanished()` — still false in this window — so it does +/// NOT bail. The re-arm therefore lands after trip#1 and before trip#2, exactly the interval trip#2 guards. +/// Verified to go RED when trip#2 is removed (see task-10-report.md — test_task10b_reddemo.log). +TEST(CASForget, ForgetReLatchesFenceAfterAReclaimReachesArmMountFence) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + /// Make the current mount claimable so a self-remount SUCCEEDS fast and reaches `armMountFence`. + fenceOutMount(*backend, store->layout().mountKey(kSrid)); + + bool reclaimed = false; + store->forgetDisk([&] { reclaimed = store->tryRemountOnce(); }, kForgetReason); + + /// Guard against a vacuous pass: if the injected reclaim did not actually succeed (reach + /// `armMountFence`), there is no re-arm for trip#2 to override and the test proves nothing. + ASSERT_TRUE(reclaimed) << "the injected reclaim must reach armMountFence, else this guard is vacuous"; + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_FALSE(store->mayMutate()) + << "FORGET's post-join fence re-latch (trip#2) must override the fence the reclaim re-armed"; +} + +/// (b3) PROMOTION-GUARD REGRESSION (spec §9 rev.8 item 7): with the erasure-proof excised, the natural +/// `Vanished(replaced)` verdict is the ONLY remaining mid-FORGET natural-terminal race. A `tryRemountOnce` +/// in flight during FORGET — one that passed step 0's `isVanished()` gate before FORGET published its intent +/// — must NOT settle `Vanished(replaced)` and mislabel the operator-visible reason; FORGET's +/// `Vanished(forgotten)` must win. We drive a REAL `tryRemountOnce` from FORGET's own GC-stop step (spec §5 +/// step 3/4, strictly AFTER the step-1 intent publish, BEFORE the step-6 settle), against a FOREIGN +/// `_pool_meta` (the `Replaced` verdict), and assert the guard bailed. +TEST(CASForget, ForgetIntentBlocksNaturalReplacedPromotion) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + /// Make the identity gate verdict `Replaced`: overwrite `_pool_meta` with a FOREIGN pool_id (present, + /// mismatched identity) — exactly gtest_cas_lifecycle_condition.cpp scenario (b). + const String meta_key = store->layout().poolMetaKey(); + const auto got = backend->get(meta_key); + ASSERT_TRUE(got.has_value()); + DB::Cas::PoolMeta foreign = DB::Cas::decodePoolMeta(got->bytes); + foreign.pool_id = foreign.pool_id + DB::UInt128(1); + ASSERT_EQ(backend->putOverwrite(meta_key, DB::Cas::encodePoolMeta(foreign), got->token).outcome, + DB::Cas::PutOutcome::Done); + + /// The in-flight gate (run from the GC-stop callback) reaches the `Replaced` verdict but must BAIL on the + /// already-published intent rather than settle `Vanished(replaced)`. + bool replaced_settled_midforget = false; + store->forgetDisk( + [&] + { + store->tryRemountOnce(); + replaced_settled_midforget = (store->lifecycle() == PoolLifecycle::VanishedReplaced); + }, + kForgetReason); + + EXPECT_FALSE(replaced_settled_midforget) + << "a mid-FORGET Replaced verdict must NOT settle — the intent guard bails before enterVanished(Replaced)"; + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten) + << "FORGET's Vanished(forgotten) must win (first terminal STATE transition)"; + const std::string msg = messageOf([&] { store->throwIfLifecycleTerminal(); }); + EXPECT_NE(msg.find("erasure was NOT verified"), std::string::npos) << msg; + EXPECT_EQ(msg.find("foreign pool"), std::string::npos) + << "the reason must NOT be the mislabeled Replaced text: " << msg; +} + +/// (e) End-to-end through the verb entry `ContentAddressedMetadataStorage::forgetDisk` and the six-class +/// gate: after FORGET, a Probe answers truth-absent, a Remove no-ops, and a content read throws the [D5] +/// message with the REAL decommission timestamp produced by the handler. +TEST(CASForget, ForgetEndToEndGatesTruthWithTimestampedMessage) +{ + auto storage = openForgetStorage(); + commitOnePart(*storage); + ASSERT_TRUE(storage->existsFile(kPartFile)); /// Live baseline + + storage->forgetDisk(); + + /// Probe → truth-absent (no throw): the committed part reads absent on a forgotten disk. + EXPECT_FALSE(storage->existsFile(kPartFile)); + EXPECT_FALSE(storage->existsDirectory(kPartDir)); + + /// Remove → no-op success (this is what lets a forgotten-disk table's DROP complete). + EXPECT_NO_THROW({ + auto tx = storage->createTransaction(); + tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr); + tx->commit(NoCommitOptions{}); + }); + + /// Content read → the typed [D5] message, with the handler's real UTC timestamp. + const std::string msg = messageOf([&] { storage->getFileSize(kPartFile); }); + EXPECT_NE(msg.find("SYSTEM CAS FORGET at "), std::string::npos) << msg; + EXPECT_NE(msg.find(" UTC"), std::string::npos) << msg; + EXPECT_NE(msg.find("erasure was NOT verified"), std::string::npos) << msg; +} + +/// (I-1 regression) A manual `SYSTEM CAS GC RUN` admitted while `Live` but that acquires +/// `gc_scheduler_mutex` strictly AFTER a concurrent FORGET completes must NOT recreate a `CasGcScheduler` +/// on the now-`Vanished` pool: the under-lock admission re-check refuses with the typed [D5] message. The +/// interleave is deterministic (bounded cv waits, no sleep) — the GC-verb seam parks the RUN in the +/// admission→lock window while the FORGET thread drives the real teardown. The lasting-damage observable is +/// `gcHealth()` staying empty: a resurrected scheduler (the pre-fix behavior) would make it non-empty. +/// Verified RED against the pre-fix ordering (see task-17-report.md). +TEST(CASForget, GcRunAdmittedWhileLiveRefusesAfterConcurrentForget) +{ + auto storage = openForgetStorage(); + /// Capture the pool while Live (store() is fail-closed once Vanished) to assert its terminal state after. + auto pool = storage->store(); + ASSERT_EQ(pool->lifecycle(), PoolLifecycle::Live); + ASSERT_FALSE(storage->gcHealth().has_value()) + << "no scheduler exists before the first GC round (unit-test null context creates none at startup)"; + + const std::string msg = raceForgetIntoGcVerbWindow( + *storage, [&] { storage->runGarbageCollectionRoundNow(); }); + + /// The refusal is the typed FORGET [D5] message (an under-lock admission throw), not a round-internal + /// error and not a silently-run round. + EXPECT_NE(msg.find("erasure was NOT verified"), std::string::npos) << msg; + /// The I-1 lasting-damage observable: NO scheduler was created on the decommissioned pool. + EXPECT_FALSE(storage->gcHealth().has_value()) + << "a GC RUN refused post-FORGET must NOT recreate the scheduler on a Vanished pool"; + EXPECT_EQ(pool->lifecycle(), PoolLifecycle::VanishedForgotten); +} + +/// (I-2 regression) A `SYSTEM CAS GC REBUILD` holds `gc_scheduler_mutex` for its whole +/// duration, so a concurrent FORGET must SERIALIZE behind it — FORGET cannot report the disk decommissioned +/// while the rebuild is still issuing durable `gc/`-plane writes. Deterministic (bounded cv waits + a bounded +/// negative future poll anchored by a positive control, never a sleep-to-fix-a-race): the in-lock seam parks +/// the rebuild WHILE it holds the mutex; a FORGET launched in that window must NOT complete until the rebuild +/// releases the lock. The pre-fix `runGcRebuildNow` took NO lock, so an in-flight rebuild was invisible to +/// FORGET and FORGET would complete immediately. Verified RED against the pre-fix code (see task-17-report.md). +TEST(CASForget, GcRebuildInFlightSerializesForget) +{ + auto storage = openForgetStorage(); + auto pool = storage->store(); /// captured while Live + ASSERT_EQ(pool->lifecycle(), PoolLifecycle::Live); + + std::mutex m; + std::condition_variable cv; + bool rebuild_holds_lock = false; + bool may_release = false; + + /// In-lock seam: fires WHILE the rebuild holds `gc_scheduler_mutex`. It parks there (bounded) until the + /// coordinator has verified FORGET is blocked, then lets the rebuild finish and release the lock. + storage->setGcVerbAdmitWindowHookForTest([&] + { + { + std::lock_guard lk(m); + rebuild_holds_lock = true; + } + cv.notify_all(); + std::unique_lock lk(m); + EXPECT_TRUE(cv.wait_for(lk, std::chrono::seconds(30), [&] { return may_release; })) + << "the coordinator must release the in-flight rebuild within the bound"; + }); + + /// The rebuild runs on its own thread; it holds the lock through the seam above. + std::promise rebuild_done_p; + auto rebuild_done = rebuild_done_p.get_future(); + std::thread rebuilder([&] + { + /// On release the pool may already be Vanished (RED path: FORGET ran unserialized) — `store()` then + /// throws; swallow it, the assertions below carry the verdict. + try { storage->runGcRebuildNow(/*force=*/false); } catch (...) {} // NOLINT(bugprone-empty-catch) + rebuild_done_p.set_value(); + }); + + /// Wait until the rebuild is genuinely in flight (holding the lock). + { + std::unique_lock lk(m); + ASSERT_TRUE(cv.wait_for(lk, std::chrono::seconds(30), [&] { return rebuild_holds_lock; })) + << "the rebuild must reach its in-lock seam"; + } + + /// Launch FORGET while the rebuild holds the lock. With the fix it BLOCKS on `gc_scheduler_mutex`; + /// without the fix (pre-fix rebuild took no lock) it runs straight through. + std::promise forget_done_p; + auto forget_done = forget_done_p.get_future(); + std::thread forgetter([&] { storage->forgetDisk(); forget_done_p.set_value(); }); + + /// The discriminator: FORGET must NOT complete while the rebuild holds the lock. This bounded negative + /// observation is anchored by the positive control below (FORGET DOES complete once the lock releases), + /// so the window's meaning is real, not a race hidden behind a sleep. + EXPECT_EQ(forget_done.wait_for(std::chrono::seconds(2)), std::future_status::timeout) + << "FORGET must serialize behind an in-flight GC rebuild (the rebuild holds gc_scheduler_mutex)"; + + /// Release the in-flight rebuild; it finishes and drops the lock, and FORGET can now proceed. + { + std::lock_guard lk(m); + may_release = true; + } + cv.notify_all(); + + ASSERT_EQ(rebuild_done.wait_for(std::chrono::seconds(30)), std::future_status::ready) + << "the rebuild must complete after release"; + ASSERT_EQ(forget_done.wait_for(std::chrono::seconds(30)), std::future_status::ready) + << "once the rebuild releases the lock, the serialized FORGET completes (positive control)"; + + rebuilder.join(); + forgetter.join(); + storage->setGcVerbAdmitWindowHookForTest({}); + + EXPECT_EQ(pool->lifecycle(), PoolLifecycle::VanishedForgotten) + << "FORGET settles the pool Vanished(forgotten) once it is no longer serialized behind the rebuild"; +} diff --git a/src/Disks/tests/gtest_cas_format.cpp b/src/Disks/tests/gtest_cas_format.cpp new file mode 100644 index 000000000000..9aa93e99f76f --- /dev/null +++ b/src/Disks/tests/gtest_cas_format.cpp @@ -0,0 +1,126 @@ +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int UNKNOWN_FORMAT_VERSION; + extern const int LOGICAL_ERROR; +} + +using namespace DB::Cas; + +TEST(CASFormat, ChangePointsExistForEveryClass) +{ + /// Every class that existed from the start has a non-empty, gen-1 baseline. + for (auto id : {FormatId::Blob, + FormatId::GcState, + FormatId::PoolMeta, FormatId::Roster, + FormatId::GcOutcomes, + FormatId::PartManifest, FormatId::RunFile, + FormatId::FoldSeal}) + { + auto cps = changePoints(id); + ASSERT_FALSE(cps.empty()); + EXPECT_EQ(cps.front().generation, 1u); + EXPECT_EQ(cps.front().min_reader, 1u); + } +} + +/// A class BORN after generation 1 begins its history at its birth generation, not at 1. `RefCkpt` +/// (spec INV-4) was introduced at generation 4: there is no such thing as a generation-1 `_ckpt`, and a +/// `{1, 1}` baseline would assert that a generation-1 reader could read one. Its history then gained a +/// three later breaking entries: generation 5 re-keyed it under `//`, generation 6 +/// moved it to opaque life-owned state, and generation 9 added the exact committed frontier. Neither +/// change touches the gen-1 baseline. Pinned because the decision is +/// invisible otherwise — nothing consults `changePoints` at decode time yet, so a wrong entry here +/// would sit unnoticed until the day a per-class reader floor is wired and starts admitting objects it +/// should refuse. +TEST(CASFormat, ChangePointsOfAClassBornAfterGenerationOneStartAtItsBirth) +{ + const auto cps = changePoints(FormatId::RefCkpt); + ASSERT_EQ(cps.size(), 4u); + EXPECT_EQ(cps.front().generation, kContiguousRefStreamsGeneration); + EXPECT_EQ(cps.front().min_reader, kContiguousRefStreamsGeneration); + EXPECT_GT(cps.front().generation, 1u) << "the point of this test is that it is NOT the gen-1 baseline"; + EXPECT_EQ(cps[1].generation, kNamespaceLifeKeyedGeneration); + EXPECT_EQ(cps[1].min_reader, kNamespaceLifeKeyedGeneration); + EXPECT_EQ(cps[2].generation, kOpaqueNamespaceLifeLayoutGeneration); + EXPECT_EQ(cps[2].min_reader, kOpaqueNamespaceLifeLayoutGeneration); + EXPECT_EQ(cps.back().generation, kCommittedRefFrontierGeneration); + EXPECT_EQ(cps.back().min_reader, kCommittedRefFrontierGeneration); +} + +TEST(CASFormat, PoolMetaTracksTheRecreateOnlyRecoveryFrontierGeneration) +{ + const auto cps = changePoints(FormatId::PoolMeta); + ASSERT_EQ(cps.size(), 4u); + EXPECT_EQ(cps.back().generation, kMountWriteAttemptIdGeneration); + EXPECT_EQ(cps.back().min_reader, kMountWriteAttemptIdGeneration); +} + +TEST(CASFormat, MountAttemptIdentityIsARecreateOnlyGenerationTenChange) +{ + EXPECT_EQ(G_BUILD, 10u); + EXPECT_EQ(kMountWriteAttemptIdGeneration, 10u); + + const auto mount_points = changePoints(FormatId::MountLease); + ASSERT_EQ(mount_points.back().generation, kMountWriteAttemptIdGeneration); + EXPECT_EQ(mount_points.back().min_reader, kMountWriteAttemptIdGeneration); + + const auto pool_points = changePoints(FormatId::PoolMeta); + ASSERT_EQ(pool_points.back().generation, kMountWriteAttemptIdGeneration); + EXPECT_EQ(pool_points.back().min_reader, kMountWriteAttemptIdGeneration); +} + +TEST(CASPoolMeta, GenerationNinePoolIsRejectedAtReaderFloor) +{ + PoolMeta meta; + meta.pool_id = UInt128{1}; + meta.blob_header_len = 256; + meta.gc_shards = 1; + meta.min_reader_generation = 10; + meta.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + String encoded = encodePoolMeta(meta); + const String current = "\"v\":10"; + const size_t version = encoded.find(current); + ASSERT_NE(version, String::npos); + encoded.replace(version, current.size(), "\"v\":9"); + + try + { + decodePoolMeta(encoded); + FAIL() << "expected UNKNOWN_FORMAT_VERSION"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + EXPECT_NE(e.message().find("generation-10 mount-attempt-identity"), String::npos); + } +} + +TEST(CASFormat, CurrentVersionsAreGBuild) +{ + EXPECT_EQ(currentWriterVersion(), G_BUILD); + EXPECT_EQ(currentCompatibilityVersion(), G_BUILD); +} + +TEST(CASFormat, CheckCompatibilityPassesWhenKnown) +{ + EXPECT_NO_THROW(checkCompatibility(1u, "manifest")); + EXPECT_NO_THROW(checkCompatibility(G_BUILD, "manifest")); +} + +TEST(CASFormat, CheckCompatibilityFailsClosedOnFuture) +{ + try + { + checkCompatibility(G_BUILD + 1, "manifest"); + FAIL() << "expected UNKNOWN_FORMAT_VERSION"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + } +} diff --git a/src/Disks/tests/gtest_cas_format_battery.cpp b/src/Disks/tests/gtest_cas_format_battery.cpp new file mode 100644 index 000000000000..f6ba9bcdbcd0 --- /dev/null +++ b/src/Disks/tests/gtest_cas_format_battery.cpp @@ -0,0 +1,23 @@ +#include "cas_format_test_battery.h" +#include +#include + +using namespace DB::Cas; + +/// The real cas_pool_meta case replaces the phase-1 toy proving instance. Every other control-plane +/// format registers its own battery row in its own gtest_cas__format.cpp file (Tasks 3-6). + +TEST(CASFormatBattery, PoolMeta) +{ + PoolMeta pm; + pm.pool_id = hexToU128("00112233445566778899aabbccddeeff"); + pm.blob_header_len = 256; + pm.min_reader_generation = 3; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + runFormatBattery(FormatBatteryCase{ + .id = FormatId::PoolMeta, + .encode = [&] { return sealObject(FormatId::PoolMeta, encodePoolMeta(pm)); }, + .decode = [](std::string_view s) { decodePoolMeta(std::string(openObject(FormatId::PoolMeta, s))); }, + .golden = currentFormatHeader("cas_pool_meta") + + "{\"pid\":\"00112233445566778899aabbccddeeff\",\"hln\":256,\"gcs\":1,\"mrg\":3,\"alg\":\"ch128\"}\n"}); +} diff --git a/src/Disks/tests/gtest_cas_fsck.cpp b/src/Disks/tests/gtest_cas_fsck.cpp new file mode 100644 index 000000000000..71abb917eaa2 --- /dev/null +++ b/src/Disks/tests/gtest_cas_fsck.cpp @@ -0,0 +1,1584 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ +constexpr uint64_t kWriterEpoch = 7; +const String kServerRoot = "00"; +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = kWriterEpoch, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} + +/// B207 race-simulation harness: `InMemoryBackend` is documented "not final: tests subclass it to +/// distort single behaviors". `runFsck`'s ref-walk and its physical blob listing (`listAll` over +/// `layout.blobsPrefix()`) are two separate calls to `Backend::list` minutes apart in production; here +/// we fire an injected mutation the FIRST time `list` is called against the armed prefix — i.e. +/// strictly AFTER the ref-walk has captured its (now stale) `reachable_blobs`/`blob_labels` view, and +/// strictly BEFORE the HEAD-confirm loop sees the physical listing. That reproduces the race +/// deterministically, without any real timing. +class RepublishOnListBackend : public InMemoryBackend +{ +public: + void armOnFirstList(String prefix, std::function mutation) + { + std::lock_guard lock(arm_mutex); + armed_prefix = std::move(prefix); + pending_mutation = std::move(mutation); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + std::function to_run; + { + std::lock_guard lock(arm_mutex); + if (pending_mutation && prefix == armed_prefix) + { + to_run = std::move(pending_mutation); + pending_mutation = nullptr; + } + } + if (to_run) + to_run(); + return InMemoryBackend::list(prefix, cursor, limit); + } + +private: + std::mutex arm_mutex; + String armed_prefix; + std::function pending_mutation; +}; + +/// Companion to `RepublishOnListBackend` for the MANIFEST phantom-dangle race: the ref-walk's +/// per-namespace recovery captures each committed `(ref -> manifest)` minutes before the per-ref +/// `backend.get(mkey)` that confirms the manifest body. This backend fires an injected mutation the +/// FIRST time `get` is called for the armed manifest key — strictly AFTER the walk captured its (now +/// stale) row and AT the GET that would otherwise read the manifest — reproducing "ref republished/ +/// dropped + old manifest legitimately GC-deleted" deterministically, with no real timing. +class MutateOnFirstGetBackend : public InMemoryBackend +{ +public: + void armOnFirstGet(String key, std::function mutation) + { + std::lock_guard lock(arm_mutex); + armed_key = std::move(key); + pending_mutation = std::move(mutation); + } + + std::optional get(const String & key, Range range) override + { + std::function to_run; + { + std::lock_guard lock(arm_mutex); + if (pending_mutation && key == armed_key) + { + to_run = std::move(pending_mutation); + pending_mutation = nullptr; + } + } + if (to_run) + to_run(); + return InMemoryBackend::get(key, range); + } + +private: + std::mutex arm_mutex; + String armed_key; + std::function pending_mutation; +}; + +enum class FsckListingMode : uint8_t +{ + Full, + Empty, + Partial, + Reordered, +}; + +/// Distort only one namespace stream LIST after fixture deposition. Exact GET/HEAD and every other +/// prefix retain ordinary backend semantics, so the test varies the hint and nothing authoritative. +class FsckListingBackend : public InMemoryBackend +{ +public: + void distort(String prefix_, FsckListingMode mode_) + { + prefix = std::move(prefix_); + mode = mode_; + } + + ListPage list(const String & listed_prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(listed_prefix, cursor, limit); + if (listed_prefix != prefix) + return page; + if (mode == FsckListingMode::Empty) + page.keys.clear(); + else if (mode == FsckListingMode::Partial && !page.keys.empty()) + page.keys.erase(page.keys.begin()); + else if (mode == FsckListingMode::Reordered) + std::reverse(page.keys.begin(), page.keys.end()); + return page; + } + +private: + String prefix; + FsckListingMode mode = FsckListingMode::Full; +}; + +/// Fail one exact GET without disturbing LIST or any other object read. This keeps the checkpoint +/// authority stable while proving that fsck distinguishes a transport failure from durable corruption. +class FailExactGetBackend : public InMemoryBackend +{ +public: + void fail(String key_) + { + key = std::move(key_); + } + + std::optional get(const String & requested_key, Range range) override + { + if (requested_key == key) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected exact GET failure"); + return InMemoryBackend::get(requested_key, range); + } + +private: + String key; +}; + +/// Publish the exact `_ckpt` authority an ordinary Live test life would have after its first committed +/// record. Raw ref-log helpers deliberately do not do this: several protocol tests need malformed or +/// pre-creation states. Fsck tests that exercise a recoverable Live life must make the durable authority +/// explicit instead of accidentally borrowing the legacy LIST-only recovery rule. +void writeFsckCheckpoint(Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId committed_through) +{ + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(backend, layout); + const auto it = std::find_if(cut.catalog.entries.begin(), cut.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; }); + ASSERT_NE(it, cut.catalog.entries.end()); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); + const String key = layout.refCkptKey(life); + const String body = encodeRefCkpt(RefCkpt{ + .life_epoch = committed_through.writer_epoch, + .committed_through = committed_through, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + const HeadResult current = backend.head(key); + const PutResult put = current.exists + ? backend.putOverwrite(key, body, current.token) + : backend.putIfAbsent(key, body); + ASSERT_EQ(put.outcome, PutOutcome::Done); +} + +void writeFsckCheckpointWithBase( + Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId base, + std::optional last_epoch_seal = std::nullopt) +{ + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(backend, layout, ns); + ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = base, + .checkpoint_snapshot_id = base, + .last_epoch_seal = last_epoch_seal})).outcome, PutOutcome::Done); +} + +void expectCheckpointBaseVerdict( + const FsckReport & report, const String & exact_base_key, FsckClass expected_class, + std::string_view expected_reason) +{ + EXPECT_EQ(report.chain_broken, expected_class == FsckClass::ChainBroken ? 1u : 0u); + EXPECT_EQ(report.unchecked, expected_class == FsckClass::Unchecked ? 1u : 0u); + EXPECT_EQ(report.clean(), expected_class != FsckClass::ChainBroken); + + size_t matching = 0; + for (const FsckObject & object : report.objects) + { + if (object.cls != expected_class || object.key != exact_base_key) + continue; + ++matching; + ASSERT_EQ(object.reachable_from.size(), 1u); + EXPECT_NE(object.reachable_from.front().find(expected_reason), String::npos); + } + EXPECT_EQ(matching, 1u) << "the checkpoint-base verdict must identify its exact named base and cause"; +} + +/// Test-only external catalog writer. It changes the namespace's current logical life after fsck took +/// its catalog cut, precisely the competing-cut mutation that fsck must not splice into its verdict. +void replaceCatalogLife(Backend & backend, const Layout & layout, const RootNamespace & ns, UInt128 incarnation) +{ + CasRefCatalog::Snapshot current = CasRefCatalog::read(backend, layout); + const auto it = std::find_if(current.catalog.entries.begin(), current.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; }); + ASSERT_NE(it, current.catalog.entries.end()); + it->incarnation = incarnation; + it->state = NsState::Live; + it->creator.reset(); + it->removal_started_round.reset(); + ASSERT_TRUE(current.token.has_value()); + ASSERT_EQ(backend.putOverwrite(layout.refCatalogKey(), encodeRefCatalog(current.catalog), *current.token).outcome, + PutOutcome::Done); +} + +FsckReport runFsckWithListingMode(FsckListingMode mode, std::string_view suffix) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/listing_" + String(suffix) + "@cas@"}; + const ManifestRef r1 = ref(1, 0xD1); + const ManifestRef r2 = ref(2, 0xD2); + const DB::UInt128 h1 = u128Of("fsck-listing-old-" + String(suffix)); + const DB::UInt128 h2 = u128Of("fsck-listing-new-" + String(suffix)); + writeBlobBody(*backend, layout, h1); + writeBlobBody(*backend, layout, h2); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", h1)}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("a", h2)}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r1); + const uint64_t frontier = publishCommittedTransition(*backend, layout, ns, "tbl", r1, r2); + writeFsckCheckpoint(*backend, layout, ns, RefTxnId{1, frontier}); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + backend->distort(layout.namespaceStreamPrefix(life), mode); + return runFsck(*store, /*detail=*/true); +} + +void expectListingIndependentFsck(const FsckReport & report) +{ + EXPECT_EQ(report.chain_broken, 0u); + EXPECT_EQ(report.unchecked, 0u); + EXPECT_EQ(report.dangling, 0u); + EXPECT_EQ(report.ref_records_walked, 2u); + EXPECT_EQ(report.reachable, 1u); +} + +struct FsckAuthorityVerdict +{ + bool clean = false; + uint64_t hard_findings = 0; + uint64_t chain_broken = 0; + uint64_t unchecked = 0; + uint64_t ref_records_walked = 0; + uint64_t reachable = 0; + uint64_t dangling = 0; + + bool operator==(const FsckAuthorityVerdict &) const = default; +}; + +FsckAuthorityVerdict authorityVerdict(const FsckReport & report) +{ + uint64_t hard_findings = 0; + for (const FsckHardFinding & finding : kFsckHardFindings) + hard_findings += report.*(finding.value); + return FsckAuthorityVerdict{ + .clean = report.clean(), + .hard_findings = hard_findings, + .chain_broken = report.chain_broken, + .unchecked = report.unchecked, + .ref_records_walked = report.ref_records_walked, + .reachable = report.reachable, + .dangling = report.dangling, + }; +} + +FsckReport runCheckpointBaseFsckWithListingMode( + FsckListingMode mode, std::string_view suffix, bool corrupt_exact_base) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/listing_base_" + String(suffix) + "@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefTxnId base{1, 1}; + const RefLogTxn birth{ + .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, birth); + RefTableState base_state; + applyRefLogTxn(base_state, birth); + writeRefSnapshotRaw(*backend, layout, snapshotOf(base_state, ns.string())); + writeFsckCheckpointWithBase(*backend, layout, ns, base); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + if (corrupt_exact_base) + { + const String base_snapshot_key = layout.refSnapshotKey(life, base); + const HeadResult head = backend->head(base_snapshot_key); + EXPECT_TRUE(head.exists); + if (head.exists) + EXPECT_EQ(backend->deleteExact(base_snapshot_key, head.token).kind, DeleteOutcome::Kind::Deleted); + } + else + { + /// This newer pair is deliberately outside `_ckpt.committed_through`. It is inert garbage: + /// changing whether LIST happens to reveal it must not add or remove an fsck finding. + const RefTxnId unadopted{1, 2}; + const RefOwnerBinding listed_binding{RefOwnerKind::Precommit, "listed", ref(9, 0xE9)}; + const RefLogTxn listed_log{ + .ns = ns.string(), + .txn_id = unadopted, + .ops = {ownerTransitionOp(std::nullopt, listed_binding)}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, listed_log); + RefTableState listed_state = base_state; + applyRefLogTxn(listed_state, listed_log); + RefTableSnapshot unadopted_snapshot = snapshotOf(listed_state, ns.string()); + unadopted_snapshot.precommits.push_back( + RefOwnerBinding{RefOwnerKind::Precommit, "unlisted", ref(10, 0xEA)}); + writeRefSnapshotRaw(*backend, layout, unadopted_snapshot); + } + + backend->distort(layout.namespaceStreamPrefix(life), mode); + return runFsck(*store, /*detail=*/true); +} +} + +/// A committed ref whose manifest body is present and whose blobs exist => clean. +TEST(CASFsck, CleanManifestPoolHasNoDangling) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.dangling, 0u); +} + +/// A committed ref naming a MISSING manifest body is an ERROR (Dangling). +TEST(CASFsck, OwnerVisibleMissingManifestBodyIsError) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + const uint64_t sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r); // no body written + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_FALSE(rep.clean()); + EXPECT_GE(rep.dangling, 1u); +} + +/// A committed ref whose blob body is missing is an ERROR (Dangling). +TEST(CASFsck, ReachableBlobMissingIsError) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); // no blob body + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_FALSE(rep.clean()); + EXPECT_GE(rep.dangling, 1u); +} + +/// fsck RECORDS AND CONTINUES over a key that names no namespace life. It is the forensic tool an +/// operator reaches for after something has already gone wrong, so one bad key must not make it report +/// NOTHING -- including about the healthy namespaces it would never reach. The finding is hard (an +/// un-incarnated key is corruption behind the format bump) and counted once per key, not once per sweep. +TEST(CASFsck, LifelessKeyIsRecordedAndTheHealthyNamespaceIsStillReported) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + + /// Hand-built: no helper can mint the un-incarnated shape any more. + const String lifeless = store->layout().casRefsPrefix() + ns.string() + "/_log/" + + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + ASSERT_EQ(backend->putIfAbsent(lifeless, "garbage").outcome, PutOutcome::Done); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)) + << "the audit must not be taken out by the damage it exists to report"; + + /// The finding, named, and counted ONCE even though several sweeps enumerate namespaces. + EXPECT_EQ(rep.lifeless_keys, 1u); + EXPECT_FALSE(rep.clean()); + bool saw = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::LifelessKey) + { + saw = true; + EXPECT_EQ(o.key, lifeless); + } + EXPECT_TRUE(saw) << "a counted finding with no row is a number nobody can act on"; + + /// And the healthy namespace was still reached: its committed ref resolved to a present manifest and + /// a present blob, which only a sweep that ran can report. + EXPECT_GE(rep.reachable, 1u); + EXPECT_EQ(rep.dangling, 0u); +} + +/// A COMPLETE, canonical namespace-life key (a real `_files` write under a real admitted life) whose +/// catalog row is then removed entirely -- exactly what a fenced GC's exact-CAS row deletion leaves +/// behind, before the perpetual namespace janitor's next page reaches it -- must classify as +/// `janitor_pending`, a SOFT finding, never `lifeless_keys`. The report stays clean. +TEST(CASFsck, CanonicalDeadLifeResidueIsJanitorPendingNotHardFinding) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const NamespaceLifeId life = store->namespaceLife(ns); + store->putNamespaceFile(life, "format_version.txt", "1\n"); + + /// Simulate a fenced GC's exact-CAS catalog-row deletion: the row is gone, the life-owned physical + /// object above survives it (the janitor's own job, not GC's own round). `casUpdate` deliberately + /// refuses to add or delete rows (there is no generic catalog remove-by-name API -- deletion is + /// only `deleteCompletedRemoving`/`cancelStalledCreating`, both requiring the full fenced-GC + /// protocol this fixture is not driving), so inject the post-deletion catalog snapshot directly, + /// mirroring `DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgresses` below. + { + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, store->layout()); + const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; }); + ASSERT_NE(it, snapshot.catalog.entries.end()); + snapshot.catalog.entries.erase(it); + const auto catalog_head = backend->head(store->layout().refCatalogKey()); + ASSERT_TRUE(catalog_head.exists); + ASSERT_EQ(backend->putOverwrite(store->layout().refCatalogKey(), encodeRefCatalog(snapshot.catalog), + catalog_head.token).outcome, PutOutcome::Done); + } + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)) + << "janitor-pending residue must never abort the scan"; + + EXPECT_EQ(rep.lifeless_keys, 0u); + EXPECT_GE(rep.namespace_janitor_pending, 1u); + EXPECT_EQ(rep.namespace_janitor_pending_lives, 1u); + EXPECT_TRUE(rep.clean()) << "janitor-pending residue is not a hard finding"; + bool saw = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::JanitorPending) + saw = true; + EXPECT_TRUE(saw) << "a counted soft finding with no row is a number nobody can act on"; +} + +/// The observe-then-cut race: a life admitted between fsck's namespace-tree LIST and the catalog cut +/// it takes AFTER that listing must NOT be misread as residue. Mirrors +/// `CASNamespaceJanitor.PostListCatalogCutProtectsConcurrentCreationWithOneGet` -- the same ordering, +/// the same reason: creation admits `Creating` before writing any life-owned object, so a life visible +/// only in the LATER cut cannot have raced this listing. +namespace +{ +class AdmitLifeAfterNamespaceListingBackend : public InMemoryBackend +{ +public: + explicit AdmitLifeAfterNamespaceListingBackend(NamespaceLifeId life_) : protected_life(std::move(life_)) {} + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + if (!published && prefix.ends_with("/cas/ns/")) + { + published = true; + CasRefCatalog::casAdmitEntry(*this, Layout("p"), /*gc_shards*/1, + CatalogEntry{.ns = protected_life.ns, .state = NsState::Live, + .incarnation = protected_life.incarnation}); + } + return page; + } + +private: + NamespaceLifeId protected_life; + bool published = false; +}; +} + +TEST(CASFsck, LifeAdmittedBetweenNamespaceListingAndLaterCutIsNotResidue) +{ + const RootNamespace ns{"00/late@cas@"}; + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128{909}); + auto backend = std::make_shared(life); + auto store = openPoolForTest(backend); + /// The physical object exists before the listing runs, exactly as a legitimate late admission would + /// leave it: written only after `casAdmitEntry` above, but here pre-seeded since the injected + /// backend admits the CATALOG row, not the physical file, on the list callback. + ASSERT_EQ(backend->putIfAbsent(store->layout().namespaceFilesPrefix(life) + "format_version.txt", "1\n").outcome, + PutOutcome::Done); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)); + EXPECT_EQ(rep.namespace_janitor_pending, 0u) + << "a life visible in the post-listing cut must not be misclassified as residue"; + EXPECT_EQ(rep.lifeless_keys, 0u); + EXPECT_TRUE(rep.clean()); +} + +/// Malformed or non-canonical namespace-tree shapes must stay HARD findings even after the +/// janitor-pending split: a dirty `_files` relative name (the parser-asymmetry fix), a zero life id, an +/// uppercase life id, and an unrecognized kind directory all name no current writer's grammar. +TEST(CASFsck, MalformedNamespaceTreeShapesStayHardFindings) +{ + const Layout layout("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"00/bb@cas@"}, UInt128{909}); + const struct { String key; String description; } cases[] = { + {layout.namespaceFilesPrefix(life) + "../escape", "dirty _files relative name"}, + {"p/cas/ns/state/" + String(32, '0') + "/_files/format_version.txt", "zero life id"}, + {"p/cas/ns/state/112233445566778899AABBCCDDEEFF01/_files/format_version.txt", "uppercase life id"}, + {"p/cas/ns/stream/" + renderIncarnation(UInt128{909}) + "/_unknown_kind/x.zst", "unknown kind directory"}, + }; + for (const auto & c : cases) + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + ASSERT_EQ(backend->putIfAbsent(c.key, "garbage").outcome, PutOutcome::Done) << c.description; + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)) << c.description; + EXPECT_GE(rep.lifeless_keys, 1u) << c.description; + EXPECT_EQ(rep.namespace_janitor_pending, 0u) << c.description; + EXPECT_FALSE(rep.clean()) << c.description; + } +} + +/// Mutation caught: calling the destructive consumer's global `throwIfAmbiguous` from fsck aborts +/// before the unique row is audited. The read-only tool reports the ambiguous physical id and +/// continues through an unrelated unique namespace. +TEST(CASFsck, DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgresses) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace unique_ns{"00/unique@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, unique_ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, layout, unique_ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, layout, unique_ns, RefTxnId{1, sequence}); + + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, layout); + snapshot.catalog.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"bad/a"}, .state = NsState::Live, .incarnation = UInt128{777}}); + snapshot.catalog.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"bad/b"}, + .state = NsState::Removing, + .incarnation = UInt128{777}, + .removal_started_round = 1}); + std::sort(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), + [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); + const auto catalog_head = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(catalog_head.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head.token).outcome, + PutOutcome::Done); + + FsckReport report; + ASSERT_NO_THROW(report = runFsck(*store, /*detail=*/true)); + EXPECT_GE(report.lifeless_keys, 1u); + EXPECT_GE(report.reachable, 1u) << "the unrelated unique namespace must still be audited"; + EXPECT_EQ(report.dangling, 0u); +} + +/// A physical namespace-life key whose life id is ambiguous in the POST-LISTING cut (two catalog rows +/// share one incarnation) must be recorded as a `lifeless_keys` finding and must NOT abort the scan: +/// `CatalogLifeIndex::resolve` throws `CORRUPTED_DATA` on a duplicate, and the janitor-pending +/// classification loop must catch it exactly like every other catalog-authority failure in this scan. +/// Mirrors `DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgresses`, but that fixture +/// has no physical object under the duplicated life id, so it never drives a candidate into the new +/// post-listing loop at all -- this is the case that actually exercises it. +TEST(CASFsck, AmbiguousLifeUnderAPhysicalKeyIsRecordedNotAborted) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace unique_ns{"00/unique@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, unique_ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, layout, unique_ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, layout, unique_ns, RefTxnId{1, sequence}); + + const NamespaceLifeId duplicated_life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"bad/a"}, UInt128{777}); + ASSERT_EQ(backend->putIfAbsent(layout.namespaceFilesPrefix(duplicated_life) + "format_version.txt", "1\n").outcome, + PutOutcome::Done); + + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, layout); + snapshot.catalog.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"bad/a"}, .state = NsState::Live, .incarnation = UInt128{777}}); + snapshot.catalog.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"bad/b"}, + .state = NsState::Removing, + .incarnation = UInt128{777}, + .removal_started_round = 1}); + std::sort(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), + [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); + const auto catalog_head = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(catalog_head.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head.token).outcome, + PutOutcome::Done); + + FsckReport report; + ASSERT_NO_THROW(report = runFsck(*store, /*detail=*/true)) + << "an ambiguous life under a physical key must be a recorded finding, never an abort"; + EXPECT_GE(report.lifeless_keys, 1u); + EXPECT_GE(report.reachable, 1u) << "the unrelated unique namespace must still be audited"; + EXPECT_EQ(report.dangling, 0u); +} + +/// Fsck's namespace universe is catalog-authoritative. Admit `ns`, publish one real ref-log record, +/// then hide its whole stream prefix from LIST. Fsck must still retain the namespace from the catalog +/// cut and use the checkpoint-anchored arithmetic walk to read the record. +/// +/// Proves `ref_records_walked`, not `dangling`/`clean()`: `checkRefStream` (which this proves runs) has +/// its own `_ckpt`-anchored arithmetic walk and so is reachable here. The distinct +/// `manifestStillReferenced` recheck now receives the same frozen catalog row and exact `_ckpt` authority; +/// its competing-cut regression is pinned separately by `MissingManifestRecheckStaysOnInitialCatalogCut`. +TEST(CASFsck, CatalogLiveNamespaceHiddenFromListIsStillWalked) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/hidden_from_list@cas@"}; + + fixture::admitLive(*backend, layout, ns); + const uint64_t sequence = appendRefLogSeed( + *backend, layout, ns, {}); // one real record: a birth-only ref-log transaction + + /// `casAdmitEntry` never publishes a `_ckpt` (by its own design), and the write above used + /// `appendRefLogSeed`'s hardcoded writer_epoch 1. `checkRefStream`'s own walk needs SOME anchor -- a + /// `_ckpt.life_epoch`, a listed snapshot, or a listed log -- to know where to start reading, and this + /// test is about to hide every listed one. Without an anchor the walk sees nothing at all and + /// correctly treats the namespace as never-born, the same "nothing to probe" trap the I4 replacement + /// controls hit and were restructured around (fold-before-hide). fsck has no "fold" step to run + /// first, so the anchor is published directly, by exact key, before the hide -- the exact-key GET + /// this enables is unaffected by list-hiding either way. + writeFsckCheckpoint(*backend, layout, ns, RefTxnId{1, sequence}); + const NamespaceLifeId life = fixture::fixtureLife(ns); + + backend->hidePrefix(layout.namespaceStreamPrefix(life)); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)); + EXPECT_GT(backend->holesServed(), 0u) + << "the hide must actually have been exercised by the stream LIST, or this test passes vacuously"; + EXPECT_GE(rep.ref_records_walked, 1u) + << "the namespace must be discovered and its stream actually read even when LIST omits every one " + "of its keys, or the catalog-authoritative universe supplement did not run"; +} + +TEST(CASFsckAuthority, FullListingDoesNotDefineStreamGeometry) +{ + expectListingIndependentFsck(runFsckWithListingMode(FsckListingMode::Full, "full")); +} + +TEST(CASFsckAuthority, EmptyListingDoesNotDefineStreamGeometry) +{ + expectListingIndependentFsck(runFsckWithListingMode(FsckListingMode::Empty, "empty")); +} + +TEST(CASFsckAuthority, PartialListingDoesNotDefineStreamGeometry) +{ + expectListingIndependentFsck(runFsckWithListingMode(FsckListingMode::Partial, "partial")); +} + +TEST(CASFsckAuthority, ReorderedListingDoesNotDefineStreamGeometry) +{ + expectListingIndependentFsck(runFsckWithListingMode(FsckListingMode::Reordered, "reordered")); +} + +/// Stream LIST is not fsck authority. The same exact catalog + `_ckpt` + checkpoint-base triple must +/// yield the same result when LIST is complete, empty, partial, or reordered. A newer unadopted log and +/// snapshot are inert garbage, while damage to the exact checkpoint base remains a hard finding under +/// every listing. Mutation caught: the old LIST-derived snapshot oracle makes only listings that reveal +/// the unadopted pair non-clean. +TEST(CASFsckAuthority, StreamListingDoesNotChangeCheckpointBaseVerdict) +{ + const std::array modes{ + FsckListingMode::Full, + FsckListingMode::Empty, + FsckListingMode::Partial, + FsckListingMode::Reordered, + }; + + std::optional clean_reference; + std::optional corrupt_reference; + for (size_t i = 0; i < modes.size(); ++i) + { + const FsckAuthorityVerdict clean = authorityVerdict(runCheckpointBaseFsckWithListingMode( + modes[i], "clean_" + std::to_string(i), /*corrupt_exact_base=*/false)); + if (!clean_reference) + clean_reference = clean; + EXPECT_EQ(clean, *clean_reference); + EXPECT_TRUE(clean.clean); + EXPECT_EQ(clean.hard_findings, 0u); + + const FsckAuthorityVerdict corrupt = authorityVerdict(runCheckpointBaseFsckWithListingMode( + modes[i], "corrupt_" + std::to_string(i), /*corrupt_exact_base=*/true)); + if (!corrupt_reference) + corrupt_reference = corrupt; + EXPECT_EQ(corrupt, *corrupt_reference); + EXPECT_FALSE(corrupt.clean); + EXPECT_EQ(corrupt.chain_broken, 1u); + EXPECT_EQ(corrupt.unchecked, 0u); + EXPECT_EQ(corrupt.hard_findings, 1u); + } +} + +/// A durable but unfrontiered F+1 is not part of this fsck cut. Mutation caught: probing one position +/// beyond `_ckpt.committed_through` walks the visible record and changes both coverage and reachability. +TEST(CASFsckAuthority, VisibleFPlusOneDoesNotAffectVerdict) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/visible_f_plus_one@cas@"}; + const ManifestRef committed_ref = ref(1, 0xE1); + const ManifestRef unfrontiered_ref = ref(2, 0xE2); + const DB::UInt128 committed_blob = u128Of("fsck-frontier-committed"); + writeBlobBody(*backend, layout, committed_blob); + writeManifestRaw(*backend, layout, ns, committed_ref, {blobEntryFor("a", committed_blob)}); + const uint64_t frontier = publishCommittedTransition( + *backend, layout, ns, "tbl", std::nullopt, committed_ref); + writeFsckCheckpoint(*backend, layout, ns, RefTxnId{1, frontier}); + + /// The object is durable and visible, but `_ckpt` is deliberately NOT advanced to it. The semantic + /// convenience wrapper advances `_ckpt`, so deposit this unfrontiered F+1 as the raw transaction + /// shape that a stopped writer can leave behind. Its missing manifest would become a false dangle if + /// either fsck leg adopted F+1. + std::vector unfrontiered_ops; + unfrontiered_ops.push_back(ownerTransitionOp( + RefOwnerBinding{RefOwnerKind::Committed, "tbl", committed_ref}, std::nullopt)); + const std::vector commit_ops = publishCommittedOps("tbl", unfrontiered_ref); + unfrontiered_ops.insert(unfrontiered_ops.end(), commit_ops.begin(), commit_ops.end()); + ASSERT_EQ(appendRefLogSeed(*backend, layout, ns, std::move(unfrontiered_ops)), frontier + 1); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.chain_broken, 0u); + EXPECT_EQ(report.unchecked, 0u); + EXPECT_EQ(report.ref_records_walked, 1u); + EXPECT_EQ(report.dangling, 0u); + EXPECT_EQ(report.reachable, 1u); +} + +/// INV-2 materializes every burned global epoch, including an empty one, as a sequence-1 seal. A +/// direct `{1,2}` -> `{7,1}` chain that omits `{2,1}` is therefore data loss, not a legal sparse epoch +/// transition. Mutation caught: accepting the later head as a shortcut blesses the missing seal. +TEST(CASFsckAuthority, MissingBurnedEpochSealIsChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/skipped_writer_epoch@cas@"}; + + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = {seal}, + .prev_epoch_seal = std::nullopt}); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + /// The codec now rejects this skip. Deposit its old on-disk corruption shape by changing only the + /// fixed-width epoch token of an otherwise encodable body, so fsck still proves that a missing + /// intermediate epoch is reported rather than treated as a sparse legal transition. + String skipped_bytes = encodeRefLogTxn(RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{7, 1}, .ops = {}, .prev_epoch_seal = RefTxnId{6, 1}}); + const String old_epoch_token = R"("!pse":"6")"; + const auto old_epoch = skipped_bytes.find(old_epoch_token); + ASSERT_NE(old_epoch, String::npos); + skipped_bytes.replace(old_epoch, old_epoch_token.size(), R"("!pse":"1")"); + ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, RefTxnId{7, 1}), + sealObject(FormatId::RefLog, skipped_bytes)).outcome, PutOutcome::Done); + + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{7, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{6, 1}})).outcome, PutOutcome::Done); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.chain_broken, 1u); + EXPECT_EQ(report.unchecked, 0u); + EXPECT_EQ(report.ref_records_walked, 2u); +} + +/// The checkpoint base is the inclusive frontier, so there is no replay tail in which another hole +/// could satisfy this test. The missing same-id log itself makes the stable exact authority corrupt. +/// Mutation caught: mapping every `readCheckpointSnapshotBase` failure to `Unchecked` leaves `clean` true. +TEST(CASFsckAuthority, MissingCheckpointBaseLogIsChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/missing_checkpoint_base_log@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefTxnId base{1, 1}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + writeFsckCheckpointWithBase(*backend, layout, ns, base); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.ref_records_walked, 0u); + expectCheckpointBaseVerdict( + report, layout.refSnapshotKey(life, base), FsckClass::ChainBroken, "has no matching log"); +} + +/// A present, valid non-seal base log rules out a stream hole; only its checkpoint-named same-id +/// snapshot is absent. Stable exact absence is damage, not lost diagnostic coverage. +TEST(CASFsckAuthority, MissingCheckpointBaseSnapshotIsChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/missing_checkpoint_base_snapshot@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefTxnId base{1, 1}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); + writeFsckCheckpointWithBase(*backend, layout, ns, base); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.ref_records_walked, 0u); + expectCheckpointBaseVerdict( + report, layout.refSnapshotKey(life, base), FsckClass::ChainBroken, + "is absent under the supplied immutable lifecycle authority"); +} + +/// `_ckpt.checkpoint_snapshot_id` names a state snapshot, never an `EpochSeal`. The forged base is an +/// OLDER seal, deliberately different from `last_epoch_seal`, so comparing checkpoint metadata cannot +/// expose it: the stream audit must exact-read the base log and reject it before recovery can bless the +/// same-id snapshot. +TEST(CASFsckAuthority, CheckpointSnapshotAtOlderEpochSealIsChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint_base_seal@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefLogTxn birth{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, birth); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + const RefLogTxn seal_txn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = {seal}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, seal_txn); + RefOp later_seal; + later_seal.kind = RefOpKind::EpochSeal; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{2, 1}, .ops = {later_seal}, + .prev_epoch_seal = RefTxnId{1, 2}}); + + RefTableState through_seal; + applyRefLogTxn(through_seal, birth); + applyRefLogTxn(through_seal, seal_txn); + writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.ref_records_walked, 0u) + << "the seal is rejected as the checkpoint base, not walked as a normal replay record"; + expectCheckpointBaseVerdict( + report, layout.refSnapshotKey(life, RefTxnId{1, 2}), FsckClass::ChainBroken, + "names an EpochSeal, not a snapshot base"); +} + +/// An unstable transport failure while exact-reading the same valid checkpoint base proves neither +/// presence nor absence. It remains the honest third answer and must not become a hard finding. +TEST(CASFsckAuthority, CheckpointBaseTransportFailureIsUnchecked) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint_base_transport@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefTxnId base{1, 1}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); + RefTableState state; + applyRefLogTxn(state, RefLogTxn{ + .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); + writeRefSnapshotRaw(*backend, layout, snapshotOf(state, ns.string())); + writeFsckCheckpointWithBase(*backend, layout, ns, base); + backend->fail(layout.refLogKey(life, base)); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.ref_records_walked, 0u); + expectCheckpointBaseVerdict( + report, layout.refSnapshotKey(life, base), FsckClass::Unchecked, "injected exact GET failure"); +} + +/// The sampled checkpoint is immutable input, but cleanup may advance `_ckpt` after that sample and +/// retire its old base before fsck exact-reads it. The miss is then authority instability, not evidence +/// that either durable checkpoint incarnation was internally corrupt. +/// Mutation caught: classifying `CORRUPTED_DATA` without rechecking the sampled checkpoint token. +TEST(CASFsckAuthority, CheckpointBaseVanishingAfterAuthorityAdvanceIsUnchecked) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint_base_advanced@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const RefTxnId old_base{1, 1}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + writeFsckCheckpointWithBase(*backend, layout, ns, old_base); + backend->armOnFirstGet(layout.refLogKey(life, old_base), [&] + { + const String ckpt_key = layout.refCkptKey(life); + const HeadResult head = backend->head(ckpt_key); + ASSERT_TRUE(head.exists); + ASSERT_EQ(backend->putOverwrite(ckpt_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}), head.token).outcome, PutOutcome::Done); + }); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_EQ(report.ref_records_walked, 0u); + expectCheckpointBaseVerdict( + report, layout.refSnapshotKey(life, old_base), FsckClass::Unchecked, + "checkpoint authority changed while validating its snapshot base"); +} + +/// A Live catalog row without `_ckpt` is not a recoverable table, even when a listing happens to show a +/// complete ref log. Mutation caught: replacing the authority-taking recovery with the old LIST replay +/// makes this audit look clean and silently blesses a life whose durable frontier is unknown. +TEST(CASFsck, LiveNamespaceWithoutCheckpointIsUnchecked) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/live_without_checkpoint@cas@"}; + const ManifestRef r = ref(1, 0xC1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + /// `publishCommittedTransition` correctly advances `_ckpt`; this test instead deposits the raw + /// missing-checkpoint corruption shape that fsck must refuse to recover. + std::vector ops = publishCommittedOps("tbl", r); + appendRefLogSeed(*backend, store->layout(), ns, std::move(ops)); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_GE(report.unchecked, 1u) + << "a Live life with no exact checkpoint cannot be recovered from a convenient LIST"; + EXPECT_EQ(report.reachable, 0u) + << "fsck must not consume refs after the mandatory recovery authority was absent"; +} + +/// fsck's missing-manifest recheck runs after its primary walk. If it re-resolves the name from a second +/// catalog cut, a concurrent rebirth can make the old durable owner disappear from the recheck and hide +/// a real dangle. The initial cut's row and exact checkpoint must remain the sole authority throughout +/// the whole fsck call. +/// +/// Mutation caught: re-resolve `ns` from `manifestStillReferenced`. The first manifest GET changes the +/// catalog to a fresh, empty life; the second resolution then sees no owner and suppresses the dangle. A +/// recovery from the original cut continues to see the original owner and reports it. +TEST(CASFsck, MissingManifestRecheckStaysOnInitialCatalogCut) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/frozen_fsck_cut@cas@"}; + const ManifestRef r = ref(1, 0xC2); + const uint64_t sequence = publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + /// `publishCommittedTransition`'s raw fixture log writer uses its documented epoch 1. + writeFsckCheckpoint(*backend, layout, ns, RefTxnId{1, sequence}); + + backend->armOnFirstGet(layout.manifestKey(ManifestId{ns, r}), [&] + { + replaceCatalogLife(*backend, layout, ns, UInt128{0xC3}); + }); + + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_GE(report.dangling, 1u) + << "the initial catalog-cut owner still names the absent manifest despite a later rebirth"; +} + +/// A pre-precommit body in an eligible prefix (no owner) is INFO (Unreachable), not an error. +TEST(CASFsck, ReclaimablePrePrecommitBodyIsInfo) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + /// Seed a birth-only ref log through the fixture helper, which also admits the catalog row, while + /// leaving NO committed owner; the manifest body below is orphan debris. + const uint64_t sequence = appendRefLogSeed(*backend, store->layout(), ns, {}); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); // body, no owner + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); // eligible + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); // not an error + EXPECT_GE(rep.unreachable, 1u); // counted as info/unreachable +} + +/// Pipeline classification (2026-07-02): a condemned-but-present blob is PendingGc — an EXPECTED +/// pipeline state (deletion is scheduled), never the suspicious "unreachable" lump beta testers +/// read as a leak. clean() is unaffected. +TEST(CASFsck, CondemnedBlobClassifiesPendingGc) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t publish_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, publish_sequence}); + Gc gc(store, hexToU128("00000000000000000000000000000001")); + gc.runRegularRound(); + const uint64_t drop_sequence = dropRefTransition(*backend, store->layout(), ns, "tbl", r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); + gc.runRegularRound(); /// -1 folds => zero => condemned into the retired list; blob still present + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.pending_gc, 1u); + EXPECT_EQ(rep.unaccounted, 0u); + bool saw = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::PendingGc) + { + saw = true; + ASSERT_FALSE(o.reachable_from.empty()); + EXPECT_NE(o.reachable_from[0].find("condemned at round"), String::npos); + } + EXPECT_TRUE(saw); +} + +/// A drop whose -1 has NOT folded yet: the blob's edges are still in the GC snapshot => AwaitingGc +/// (expected), not Unaccounted. +TEST(CASFsck, DroppedButUnfoldedBlobClassifiesAwaitingGc) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t publish_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, publish_sequence}); + Gc gc(store, hexToU128("00000000000000000000000000000001")); + gc.runRegularRound(); /// +1 folded into the snapshot + const uint64_t drop_sequence = dropRefTransition( + *backend, store->layout(), ns, "tbl", r); /// -1 NOT folded (no round) + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.awaiting_gc, 1u); + EXPECT_EQ(rep.unaccounted, 0u); +} + +/// Stale-edge cross-check, NEGATIVE side: the residual edge's source manifest body is still PRESENT in +/// the pool, so its removal still has a `-1` to fold (and the orphan sweep still has a body to reclaim). +/// That is a genuine mid-pipeline backlog and must keep the `AwaitingGc` verdict — the new check may +/// never turn an ordinary unfolded drop into a hard finding. +TEST(CASFsck, UnfoldedDropWithPresentSourceManifestStaysAwaitingGc) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t publish_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, publish_sequence}); + Gc gc(store, hexToU128("00000000000000000000000000000001")); + gc.runRegularRound(); /// +1 folded into the snapshot + const uint64_t drop_sequence = dropRefTransition( + *backend, store->layout(), ns, "tbl", r); /// -1 NOT folded; the BODY survives + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.awaiting_gc, 1u); + EXPECT_EQ(rep.stale_edge, 0u); + bool saw = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::AwaitingGc) + saw = true; + EXPECT_TRUE(saw); +} + +/// Stale-edge cross-check, POSITIVE side: the blob's only residual `+1` names a manifest that no longer +/// exists anywhere in the pool, so no `-1` is left to fold — the in-degree stays at 1 for every future +/// round and the incremental GC can never nominate the blob. It must NOT be labeled `AwaitingGc` +/// ("expected, no action needed", the sentence that hid 56 permanently retained blobs); it is the hard +/// `StaleEdge` finding and the report is not `clean()`. +TEST(CASFsck, ResidualEdgeNamingAnAbsentManifestClassifiesStaleEdge) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + const ManifestId id = writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t publish_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, publish_sequence}); + Gc gc(store, hexToU128("00000000000000000000000000000001")); + gc.runRegularRound(); /// +1 folded into the snapshot + const uint64_t drop_sequence = dropRefTransition( + *backend, store->layout(), ns, "tbl", r); /// the owner is gone ... + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); + deleteManifestBody(*backend, store->layout(), id); /// ... and so is the body, un-folded + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_EQ(rep.stale_edge, 1u); + EXPECT_EQ(rep.awaiting_gc, 0u); + EXPECT_EQ(rep.dangling, 0u) << "no committed ref names the manifest any more — this is not a dangle"; + EXPECT_FALSE(rep.clean()); + bool saw = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::StaleEdge) + { + saw = true; + ASSERT_FALSE(o.reachable_from.empty()); + EXPECT_NE(o.reachable_from[0].find("no longer exist"), String::npos); + } + EXPECT_TRUE(saw); +} + +/// GC never ran on the pool: nothing is classifiable through the GC view — everything unreferenced +/// is AwaitingGc ("GC has not run yet"), never a false Unaccounted alarm. +TEST(CASFsck, GcNeverRanClassifiesAwaitingGc) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + writeBlobBody(*backend, store->layout(), DB::UInt128(5)); /// present, never referenced, no gc/state + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.awaiting_gc, 1u); + EXPECT_EQ(rep.unaccounted, 0u); +} + +/// A blob outside the WHOLE GC view on a pool where GC runs: Unaccounted — expected only as a +/// transient (fast create+drop between rounds); persistent occurrences violate INV-2. +TEST(CASFsck, ForeignBlobClassifiesUnaccounted) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + Gc gc(store, hexToU128("00000000000000000000000000000001")); + gc.runRegularRound(); + + writeBlobBody(*backend, store->layout(), DB::UInt128(0xF0F0)); /// never referenced anywhere + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.unaccounted, 1u); + EXPECT_EQ(rep.pending_gc, 0u); +} + +/// A `.meta` descriptor whose body is missing is ADVISORY (meta_without_body), NOT a hard finding: +/// GC deletes the body FIRST and drops the `.meta` afterwards on a bounded, error-suppressed advisory +/// pool that may drop the op, so a single raw LIST legitimately observes a body-less `.meta` mid- +/// graduation and no finite grace makes a persistent one hard evidence. It is still counted/reported; +/// it must NOT be a `dangling` (nothing referenced it) and NOT one of the present-but-unreferenced blob +/// pipeline classes (the `.meta` key is excluded from body classification entirely). `clean()` stays TRUE. +TEST(CASFsck, MetaWithoutBodyIsAdvisoryNotHard) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const DB::UInt128 h = u128Of("meta-without-body"); + writeMetaClean(*backend, store->layout(), h, /*size*/ 10); /// meta only, no body written + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_GE(rep.meta_without_body, 1u); // still counted and reported in the full report + EXPECT_EQ(rep.dangling, 0u); + EXPECT_EQ(rep.unreachable, 0u); + EXPECT_EQ(rep.pending_gc, 0u); + EXPECT_EQ(rep.awaiting_gc, 0u); + EXPECT_EQ(rep.unaccounted, 0u); + EXPECT_TRUE(rep.clean()); // meta_without_body is advisory — excluded from clean() +} + +/// A body with no `.meta` sibling is a BENIGN not-yet-adopted (or crashed-birth) artifact — NOT a +/// dangle, and it must still classify through the ordinary present-but-unreferenced pipeline. +TEST(CASFsck, BodyWithoutMetaIsBenign) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const DB::UInt128 h = u128Of("body-without-meta"); + writeBlobBody(*backend, store->layout(), h); /// body only, no meta written + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_GE(rep.body_without_meta, 1u); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_EQ(rep.meta_without_body, 0u); + EXPECT_TRUE(rep.clean()); +} + +/// A scan whose deadline is already in the past: partial_on_deadline=false keeps the old +/// throw-on-timeout contract; partial_on_deadline=true returns the accumulated lower-bound counts +/// instead of failing empty-handed (the 2026-07-05 campaign lost 5 verdicts to this). +TEST(CASFsckPartial, DeadlineReturnsAccumulatedCountsInsteadOfThrowing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + + const auto past = std::chrono::steady_clock::now() - std::chrono::seconds(1); + /// partial_on_deadline=false keeps the old contract: + EXPECT_THROW(DB::Cas::runFsck(*store, /*detail=*/false, {}, past), DB::Exception); + /// partial_on_deadline=true returns a flagged report: + const auto report = DB::Cas::runFsck(*store, false, {}, past, /*partial_on_deadline=*/true); + EXPECT_TRUE(report.partial); + EXPECT_FALSE(report.partial_reason.empty()); +} + +/// A `namespace_prefix` scopes the scan to only the matching namespaces' refs (dangling-only): no +/// pool-wide unreachable/pending/awaiting/unaccounted classification, since that needs the whole pool. +TEST(CASFsckScoped, NamespacePrefixChecksOnlyMatchingRefsDanglingOnly) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + const RootNamespace ns_a{"nsa"}; + const ManifestRef r_a = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns_a, r_a, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t sequence_a = publishCommittedTransition( + *backend, store->layout(), ns_a, "tbl", std::nullopt, r_a); + writeFsckCheckpoint(*backend, store->layout(), ns_a, RefTxnId{1, sequence_a}); + + const RootNamespace ns_b{"nsb"}; + const ManifestRef r_b = ref(1, 0xB1); + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); + writeManifestRaw(*backend, store->layout(), ns_b, r_b, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t sequence_b = publishCommittedTransition( + *backend, store->layout(), ns_b, "tbl", std::nullopt, r_b); + writeFsckCheckpoint(*backend, store->layout(), ns_b, RefTxnId{1, sequence_b}); + + const auto scoped = DB::Cas::runFsck(*store, false, {}, {}, false, /*namespace_prefix=*/"nsa"); + EXPECT_EQ(scoped.dangling, 0u); + EXPECT_GT(scoped.reachable, 0u); + /// Scoped mode skips only the POOL-WIDE physical/pipeline classification; the manifest-debris + /// pass stays active for the scoped namespaces, so `unreachable` here counts THEIR orphan + /// manifest bodies — zero in this clean setup, legitimately nonzero on a churned pool. + EXPECT_EQ(scoped.unreachable, 0u); + EXPECT_EQ(scoped.pending_gc + scoped.awaiting_gc + scoped.unaccounted, 0u); +} + +/// B207: the ref-walk and the HEAD-confirm run minutes apart with no snapshot. A ref that gets +/// RE-PUBLISHED to a different manifest in that window, combined with a legitimate GC delete of the +/// blob it used to name, must NOT surface as a phantom `dangling` — only a CURRENT ref over an absent +/// object is a real dangle. +TEST(CASFsck, PhantomDanglingFromRepublishedRefIsReresolvedAway) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xA1); + const ManifestRef r2 = ref(2, 0xA2); + const DB::UInt128 h1 = u128Of("b207-phantom-old"); + const DB::UInt128 h2 = u128Of("b207-phantom-new"); + + writeBlobBody(*backend, store->layout(), h1); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", h1)}); + const uint64_t initial_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r1); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, initial_sequence}); + + /// Fires strictly between the ref-walk (which captures ref "tbl" -> r1, blob h1, as reachable) and + /// the HEAD-confirm's physical listing — exactly the window B207 is about. + backend->armOnFirstList(store->layout().blobsPrefix(), [&] + { + writeBlobBody(*backend, store->layout(), h2); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", h2)}); + const uint64_t repoint_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", r1, r2); /// re-publish + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, repoint_sequence}); + + const String old_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h1)}); + const HeadResult head = backend->head(old_key); + ASSERT_TRUE(head.exists); + backend->deleteExact(old_key, head.token); /// legitimate GC delete of the now-unreferenced blob + }); + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_TRUE(rep.clean()); +} + +/// Same race, but the ref is DROPPED (not re-published) in the window between the walk and the +/// HEAD-confirm — also must not surface as a phantom dangle. +TEST(CASFsck, PhantomDanglingFromDroppedRefIsReresolvedAway) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xA1); + const DB::UInt128 h1 = u128Of("b207-phantom-dropped"); + + writeBlobBody(*backend, store->layout(), h1); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", h1)}); + const uint64_t initial_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r1); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, initial_sequence}); + + backend->armOnFirstList(store->layout().blobsPrefix(), [&] + { + const uint64_t drop_sequence = dropRefTransition( + *backend, store->layout(), ns, "tbl", r1); /// ref dropped since the walk + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); + + const String old_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h1)}); + const HeadResult head = backend->head(old_key); + ASSERT_TRUE(head.exists); + backend->deleteExact(old_key, head.token); /// legitimate GC delete after the drop folds + }); + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_TRUE(rep.clean()); +} + +/// Companion: the fix must never HIDE a real loss. A blob that a CURRENT ref still names, but whose +/// object is genuinely gone (an operator error, a storage-layer bug — NOT a legitimate GC delete), +/// stays `dangling` after the re-resolve. +TEST(CASFsck, RealDanglingStillCaughtAfterReresolve) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + const DB::UInt128 h = u128Of("b207-real-dangle"); + + writeBlobBody(*backend, store->layout(), h); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", h)}); + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); + + const String key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}); + const HeadResult head = backend->head(key); + ASSERT_TRUE(head.exists); + backend->deleteExact(key, head.token); /// genuine loss — the ref is UNCHANGED, still names this blob + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_EQ(rep.dangling, 1u); + EXPECT_FALSE(rep.clean()); +} + +/// The MANIFEST analogue of the blob phantom-dangle. The ref-walk captures "tbl" -> r1's manifest, then +/// the ref is RE-PUBLISHED to a different manifest r2 and the OLD r1 manifest body is legitimately +/// GC-deleted before the per-ref body GET. The missing OLD manifest must be revalidated away — a fresh +/// re-resolve shows the CURRENT ref no longer names it — never surfacing as a phantom `dangling`. +TEST(CASFsck, PhantomDanglingManifestFromRepublishedRefIsReresolvedAway) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xA1); + const ManifestRef r2 = ref(2, 0xA2); + const DB::UInt128 h1 = u128Of("phantom-manifest-old"); + const DB::UInt128 h2 = u128Of("phantom-manifest-new"); + + writeBlobBody(*backend, store->layout(), h1); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", h1)}); + const uint64_t initial_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", std::nullopt, r1); + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, initial_sequence}); + + const String m1_key = store->layout().manifestKey(ManifestId{ns, r1}); + /// Fires strictly between the ref-walk (captures "tbl" -> r1) and the per-ref GET of r1's manifest: + /// re-publish "tbl" to r2 and legitimately GC-delete the now-superseded r1 manifest body. + backend->armOnFirstGet(m1_key, [&] + { + writeBlobBody(*backend, store->layout(), h2); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", h2)}); + const uint64_t repoint_sequence = publishCommittedTransition( + *backend, store->layout(), ns, "tbl", r1, r2); /// re-publish + writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, repoint_sequence}); + + const HeadResult head = backend->head(m1_key); + ASSERT_TRUE(head.exists); + backend->deleteExact(m1_key, head.token); /// legitimate GC delete of the superseded manifest + }); + + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_TRUE(rep.clean()); +} + +namespace +{ +/// Build a real `ContentAddressedMetadataStorage` over Local object storage and start it (Mounted) -- +/// the same harness gtest_cas_operation_gate.cpp uses. Each call gets an isolated pool root. +std::shared_ptr openRunningStorageForTest() +{ + auto settings = makeSettingsForTest("test", std::filesystem::temp_directory_path() / "ca_fsck_running_scratch"); + auto storage = std::make_shared( + makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// Commit one real part (tmp -> final rename -> commit) so a RUNNING FSCK has live committed content. +void commitOneRunningPart(DB::ContentAddressedMetadataStorage & storage) +{ + const std::string table_dir = "g80/g80g80g8-0808-4808-8808-080808080808"; + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(table_dir + "/tmp_insert_all_1_1_0/data.bin", 65536, DB::WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(table_dir + "/tmp_insert_all_1_1_0", table_dir + "/all_1_1_0"); + tx->commit(DB::NoCommitOptions{}); +} +} + +/// (rev.8) FSCK runs on a RUNNING disk: scanning a live pool with one committed part succeeds and reports +/// its content (the one-row summary the SQL verb renders from this report). +TEST(CASFsckRunning, FsckOnMountedDiskSucceeds) +{ + auto storage = openRunningStorageForTest(); + commitOneRunningPart(*storage); + + FsckReport rep; + EXPECT_NO_THROW(rep = storage->runFsckNow(/*detail=*/false)); + EXPECT_TRUE(rep.clean()); + EXPECT_GE(rep.distinct_blobs, 1u) << "the running scan must see the live committed part's blob"; + EXPECT_EQ(rep.dangling, 0u); +} + +/// (rev.8) FSCK is Admin-class: on a not-live pool (a lease blip / IdentityLost) it refuses before +/// scanning, exactly like the GC entry points -- an FSCK of a disk whose data root may be gone or replaced +/// is meaningless (the operator has the snapshot / FORGET path). The two states refuse in DIFFERENT +/// classes, and the pairing is the point: a lease blip is transient unavailability (upstream-retryable), +/// an identity loss is terminal (668). +TEST(CASFsckRunning, FsckOnNotLiveDiskRefusesTransientRetryableAndIdentityLostTerminal) +{ + for (const auto & [lc, code] : {std::pair{PoolLifecycle::TransientNotLive, DB::ErrorCodes::NETWORK_ERROR}, + std::pair{PoolLifecycle::IdentityLost, DB::ErrorCodes::INVALID_STATE}}) + { + auto storage = openRunningStorageForTest(); + storage->store()->setLifecycleForTest(lc); /// one force from Live; no later store() call + expectThrowsCode(code, [&] { storage->runFsckNow(/*detail=*/false); }); + } +} + +/// The summary line is the ONLY thing most consumers ever read: the soak harness parses it, an operator +/// eyeballs it, and `exit_code` gates CI on it. So a field that `clean()` treats as a hard finding but the +/// summary omits is invisible in practice, however faithfully it is counted -- which is exactly what +/// happened to `corrupted_runs`: counted since the seal check landed, part of `clean()`, rendered in +/// `--detail` rows, and absent from the summary, so no run has ever reported one. +/// +/// This test ITERATES `kFsckHardFindings` -- the list `clean()` is computed from -- and never names a +/// finding itself, so a term added to that list and not rendered fails HERE. It used to claim exactly +/// that while its body was a hand-listed set of five names, and the claim was false: `lifeless_keys` was +/// added to `clean()` and nothing failed anywhere, which is how it reached the SQL row's absence too. +/// `formatFsckSummary` exists to be testable at all: the line used to be built inline in +/// `CommandFsck::executeImpl`, where nothing could reach it. +/// +/// A per-finding DISTINCT value is what makes this more than a substring sweep: it catches a formatter +/// that prints the right names against the wrong counters. +TEST(CASFsckSummary, EveryHardFindingAppearsOnTheSummaryLine) +{ + FsckReport rep; + uint64_t value = 11; + for (const FsckHardFinding & finding : kFsckHardFindings) + { + rep.*finding.value = value; + value += 11; + } + + const String line = formatFsckSummary(rep); + + value = 11; + for (const FsckHardFinding & finding : kFsckHardFindings) + { + const String token = String(finding.name) + "=" + std::to_string(value); + EXPECT_NE(line.find(token), String::npos) + << "hard finding '" << finding.name << "' is missing from the summary line (expected `" + << token << "`); the line was: " << line; + value += 11; + } + + /// A report carrying these values is NOT clean; the line must not be mistakable for a clean one. + EXPECT_FALSE(rep.clean()); +} + +/// A zero must be PRINTED, not omitted. The harness's `stale_edge_verdict` fails closed on an absent key +/// precisely because "field missing" and "field zero" are different facts, and a formatter that skips +/// zeros would turn every clean pool into an unparseable one. +TEST(CASFsckSummary, ZeroValuedHardFindingsAreStillPrinted) +{ + const String line = formatFsckSummary(FsckReport{}); + /// Iterated for the same reason the test above is: a new hard finding printed only when nonzero is a + /// finding the harness's fail-closed-on-absence consumers would read as missing. + for (const FsckHardFinding & finding : kFsckHardFindings) + EXPECT_NE(line.find(String(finding.name) + "=0"), String::npos) + << "hard finding '" << finding.name << "' prints no zero; the line was: " << line; + EXPECT_EQ(line.find("partial="), String::npos) << "a non-partial report must not claim partial: " << line; +} + +/// A partial scan is a lower bound over the visited subset, so the flag and its reason must travel WITH +/// the counts -- a consumer that sees the numbers but not `partial=1` reads a truncated walk as the pool +/// truth. +TEST(CASFsckSummary, PartialFlagAndReasonTravelWithTheCounts) +{ + FsckReport rep; + rep.partial = true; + rep.partial_reason = "deadline exceeded after 180s"; + const String line = formatFsckSummary(rep); + EXPECT_NE(line.find("partial=1"), String::npos) << line; + EXPECT_NE(line.find("reason='deadline exceeded after 180s'"), String::npos) << line; +} diff --git a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp new file mode 100644 index 000000000000..9a6f92d3650c --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp @@ -0,0 +1,1289 @@ +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +namespace ProfileEvents +{ +extern const Event CASMetaDelete; +extern const Event CASGCCondemnMarkerUnconfirmedCarry; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +ManifestRef ref(const String &, uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} +bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +/// Publish one physical blob through the production durable-precommit ordering. The committed fixture +/// ref keeps the transaction complete; callers that need an initially unowned body drop that ref. +PutBlobResult publishBlobWithDurablePrecommit( + const PoolPtr & store, const RootNamespace & ns, const String & ref, + const BlobRef & blob_ref, const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + ManifestEntry entry; + entry.path = "data.bin"; + entry.placement = EntryPlacement::Blob; + entry.ref = blob_ref; + entry.blob_size = payload.size(); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, ref, id); + const PutBlobResult result = build->putBlob(blob_ref, BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return result; +} + +/// The current retired entry for `hash` (dereferenced through gc/state.retired_refs, shard 0), or nullopt. +std::optional currentEntryFor(Backend & backend, const Layout & layout, const UInt128 & hash) +{ + for (const RetiredEntry & e : currentRetiredSet(backend, layout, /*shard*/0)) + if (e.ref == BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}) + return e; + return std::nullopt; +} + +/// Decorator reproducing the rustfs quirk (observed 2026-07-11): a conditional exact-token delete against +/// an object that is ALREADY absent can answer HTTP 412 (precondition failed), which this backend layer +/// maps to `TokenMismatch` -- not the 404-shaped `NotFound` an in-memory backend naturally returns. For +/// keys marked via `quirkOnAbsent`, `deleteExact` forces exactly that answer whenever the underlying +/// object is gone, letting a test drive the GC redelete site through the disambiguation path +/// backend-agnostically (without guessing at real rustfs HTTP mappings). +class TokenMismatchOnAbsentBackend : public InMemoryBackend +{ +public: + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (quirk_keys.contains(key) && !InMemoryBackend::head(key).exists) + { + DeleteOutcome d; + d.kind = DeleteOutcome::Kind::TokenMismatch; + return d; + } + return InMemoryBackend::deleteExact(key, token); + } + + void quirkOnAbsent(const String & key) { quirk_keys.insert(key); } + +private: + std::set quirk_keys; +}; + +class CkptReplacementConflictBackend : public InMemoryBackend +{ +public: + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (conflict_once && key == watched_key) + { + conflict_once = false; + return CasResult{CasOutcome::Conflict, {}}; + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + String watched_key; + bool conflict_once = false; +}; +} + +TEST(CASSemanticRefFixture, WrapperCreatesInitialRecoverableCheckpoint) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/semantic-create@cas@"}; + const ManifestRef manifest = ref("srv-a:1", 1, 0xAB); + + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest); + const RefTxnId expected_id{manifest.writer_epoch, sequence}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + const auto ckpt = readCkpt(*backend, store->layout(), life); + + ASSERT_TRUE(ckpt.has_value()); + EXPECT_EQ(ckpt->ckpt.life_epoch, 1); + EXPECT_EQ(ckpt->ckpt.committed_through, expected_id); + EXPECT_FALSE(ckpt->ckpt.checkpoint_snapshot_id.has_value()); + EXPECT_FALSE(ckpt->ckpt.last_epoch_seal.has_value()); +} + +TEST(CASSemanticRefFixture, WrapperAdvancesCheckpointWithoutDiscardingSnapshot) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/semantic-advance@cas@"}; + const ManifestRef manifest = ref("srv-a:1", 1, 0xAC); + + const uint64_t publish_sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest); + const RefTxnId publish_id{manifest.writer_epoch, publish_sequence}; + writeRefSnapshotRaw(*backend, store->layout(), minimalLiveSnapshot(ns.string(), publish_id, {committedRow("tbl", manifest)})); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + const auto before_drop = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(before_drop.has_value()); + RefCkpt with_snapshot = before_drop->ckpt; + with_snapshot.checkpoint_snapshot_id = publish_id; + ASSERT_EQ(backend->casPut( + store->layout().refCkptKey(life), encodeRefCkpt(with_snapshot), before_drop->token).outcome, + CasOutcome::Committed); + + const uint64_t drop_sequence = dropRefTransition(*backend, store->layout(), ns, "tbl", manifest); + const RefTxnId drop_id{manifest.writer_epoch, drop_sequence}; + const auto ckpt = readCkpt(*backend, store->layout(), life); + + ASSERT_TRUE(ckpt.has_value()); + EXPECT_EQ(ckpt->ckpt.committed_through, drop_id); + EXPECT_EQ(ckpt->ckpt.checkpoint_snapshot_id, publish_id); + EXPECT_FALSE(ckpt->ckpt.last_epoch_seal.has_value()); +} + +TEST(CASSemanticRefFixture, CheckpointAdvanceRejectsNonMonotoneAndInvalidState) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/semantic-refusal@cas@"}; + const ManifestRef manifest = ref("srv-a:1", 1, 0xAD); + + const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest); + const RefTxnId id{manifest.writer_epoch, sequence}; + EXPECT_THROW(advanceRecoverableCkptForRawFixture(*backend, store->layout(), ns, id), DB::Exception); + + const RootNamespace invalid_ns{"00/semantic-invalid@cas@"}; + fixture::admitLive(*backend, store->layout(), invalid_ns); + const NamespaceLifeId invalid_life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), invalid_ns); + const String invalid_key = store->layout().refCkptKey(invalid_life); + ASSERT_EQ(backend->putIfAbsent(invalid_key, "not a checkpoint").outcome, PutOutcome::Done); + EXPECT_THROW(advanceRecoverableCkptForRawFixture(*backend, store->layout(), invalid_ns, id), DB::Exception); + EXPECT_EQ(backend->get(invalid_key)->bytes, "not a checkpoint"); +} + +TEST(CASRawRefFixture, RawLogWriteDoesNotCreateCheckpoint) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/raw-no-ckpt@cas@"}; + const RefTxnId id{1, 1}; + + fixture::writeRefLogRaw(*backend, store->layout(), RefLogTxn{ + .ns = ns.string(), + .txn_id = id, + .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt, + }); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + EXPECT_FALSE(readCkpt(*backend, store->layout(), life).has_value()); +} + +TEST(CASRawRefFixture, ReplaceRecoverableCheckpointWritesTheSuppliedFullState) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/replace-ckpt@cas@"}; + const ManifestRef manifest = ref("srv-a:1", 1, 0xAE); + const RefTxnId first_id{manifest.writer_epoch, + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest)}; + const RefTxnId seal_id{first_id.writer_epoch, first_id.ref_sequence + 1}; + writeRefSnapshotRaw(*backend, store->layout(), minimalLiveSnapshot(ns.string(), first_id, {committedRow("tbl", manifest)})); + writeSealAt(*backend, store->layout(), ns, seal_id); + + const RefCkpt next{ + .life_epoch = 1, + .committed_through = seal_id, + .checkpoint_snapshot_id = first_id, + .last_epoch_seal = seal_id, + }; + replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, next); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + const auto replaced = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(replaced.has_value()); + EXPECT_EQ(replaced->ckpt.life_epoch, next.life_epoch); + EXPECT_EQ(replaced->ckpt.committed_through, next.committed_through); + EXPECT_EQ(replaced->ckpt.checkpoint_snapshot_id, next.checkpoint_snapshot_id); + EXPECT_EQ(replaced->ckpt.last_epoch_seal, next.last_epoch_seal); +} + +TEST(CASRawRefFixture, ReplaceRecoverableCheckpointRejectsStaleRegressiveAndWrongLife) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/replace-ckpt-refusal@cas@"}; + const ManifestRef manifest = ref("srv-a:1", 1, 0xAF); + const RefTxnId id{manifest.writer_epoch, + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest)}; + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + const auto existing = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(existing.has_value()); + + RefCkpt wrong_life = existing->ckpt; + wrong_life.life_epoch = *wrong_life.life_epoch + 1; + wrong_life.committed_through = RefTxnId{*wrong_life.life_epoch, 1}; + EXPECT_THROW(replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, wrong_life), DB::Exception); + + RefCkpt regressive = existing->ckpt; + regressive.committed_through = std::nullopt; + EXPECT_THROW(replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, regressive), DB::Exception); + + backend->watched_key = store->layout().refCkptKey(life); + backend->conflict_once = true; + EXPECT_THROW(replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, existing->ckpt), DB::Exception); + EXPECT_EQ(readCkpt(*backend, store->layout(), life)->ckpt.committed_through, id); +} + +/// The owner-removed manifest body is deleted only after a full round (its decrement is sealed — #11). +TEST(CASGCRetire, ManifestBodyDeletedAfterDecrementsSealed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + runRegularRoundReclaiming(gc); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// A publish racing the pass (in-degree restored) is SPARED, not deleted (#14). +TEST(CASGCRecheck, PublishRacingFenceSparesBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + gc.runRegularRound(); + // Repoint the ref from r1 to r2 (both reference blob 1) in the same window before the next round + // folds. ONE repoint event {old=committed(r1), new=committed(r2)} — the -1 (r1's body) and +1 + // (r2's body) net to in-degree 1, so blob 1 is re-pinned and must be SPARED. (Not a separate drop + // THEN repoint — that would double-count the -1 on r1's body and over-delete.) + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", r1, r2); + gc.runRegularRound(); // net in-degree 1 => spared + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} + +/// A genuinely unreferenced blob is deleted with its exact token (the single content-delete site). The +/// delete is not one-round-after-drop: the entry condemns, graduates the round AFTER the condemning +/// round (round-paced, unconditional), then the NEXT pass executes the exact-token delete. +TEST(CASGCRecheck, UnreferencedBlobDeletedExactToken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + // The drop's -1 condemns blob 1; the retired-cursor pipeline (condemn -> graduate -> delete) reclaims it. + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), DB::UInt128(1))); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// Task 5 (spec 2026-07-09 §raw-body-refinement, v3): GC writes the writer's freshness meta ALONGSIDE +/// the unchanged ledger retire (RetiredEntry, body token) — the meta is the writer/promote gate's +/// point-read signal (Task 3/4), not a replacement for the ledger or the exact-token body delete. +TEST(CASGCRetire, CondemnWritesMetaCondemned) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); /// +1 folds; blob referenced (`writeBlobBody` never wrote a meta itself) + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + gc.runRegularRound(); /// -1 folds => in-degree 0 => condemned THIS round + + const auto lm = loadMetaForTest(*backend, store->layout(), DB::UInt128(1)); + ASSERT_TRUE(lm.has_value()) << "GC must write the freshness meta at condemn time (Task 5)"; + EXPECT_EQ(lm->meta.state, MetaState::Condemned); + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) << "condemned, NOT yet deleted"; +} + +/// Task 5: the round's exact-token body delete drops the meta alongside it (advisory, no tombstone — +/// an absent meta reads exactly like a Clean one for the writer's point-read gate). +TEST(CASGCRetire, DeleteRemovesBodyAndMeta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + /// §0 introspection: the meta drop below rides `deleteMetaExact` (`CASMetaDelete` choke point). + const auto delete_before = ProfileEvents::global_counters[ProfileEvents::CASMetaDelete].load(); + // condemn -> graduate (round-paced) -> delete (the retired-cursor pipeline). + ASSERT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), DB::UInt128(1))); + + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) << "body gone via exact-token delete"; + EXPECT_FALSE(loadMetaForTest(*backend, store->layout(), DB::UInt128(1)).has_value()) + << "the meta must be dropped alongside the exact-token body delete (Task 5)"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaDelete].load() - delete_before, 1); +} + +/// GC freshness meta is ADD-ONLY (spec 2026-07-11 deposed-leader `clearSparedMeta` fix): an entry whose +/// in-degree recovers before graduation is SPARED (unchanged ledger behavior) but GC must NEVER flip its +/// meta `Condemned -> Clean` on the spare. A deposed leader that cleared-then-lost the round would leave a +/// stray-`Clean` over a still-condemned body; a writer reading `Clean` would reuse the exact condemned +/// token, which a stale exact-token redelete then deletes (INV_NO_LOSS live-blob loss). +/// The spare leaves the meta `Condemned`; +/// ONLY a writer that displaces the body with a fresh incarnation token publishes `Clean`. +TEST(CASGCRetire, SpareLeavesMetaCondemned) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + /// A content-addressed body + Clean meta via a real fresh upload, so a later writer dedup-attempt + /// resolves to THIS exact hash (GC condemns it; the writer republishes it). Drop the fixture ref + /// immediately so only the raw owner transitions below govern its in-degree. + const String payload = "spare-add-only-payload"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const RootNamespace seed_ns{"00/spare-seed@cas@"}; + publishBlobWithDurablePrecommit(store, seed_ns, "seed", id, payload); + store->dropRef(seed_ns, "seed"); + store->renewWatermarkOnce(); + const Token t_seed = backend->head(store->layout().blobKey(id)).token; + + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", hash)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + gc.runRegularRound(); /// +1 folds + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + gc.runRegularRound(); /// -1 folds => in-degree 0 => condemned; meta flipped Condemned + ASSERT_TRUE(currentEntryFor(*backend, store->layout(), hash).has_value()); + { + const auto lm = loadMetaForTest(*backend, store->layout(), hash); + ASSERT_TRUE(lm.has_value()); + ASSERT_EQ(lm->meta.state, MetaState::Condemned); + } + + /// Re-reference the SAME blob (same body/token — never re-uploaded) via a fresh ref before + /// graduation. The pass merge nets in-degree back to 1: the prior retired entry is SPARED + /// (recovery wins, even past the floor) -- not the republication-supersede path (the token never changed). + const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", hash)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_detached", std::nullopt, r2); + gc.runRegularRound(); /// +1 folds => spared + + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), hash).has_value()) + << "the spared entry drops from the retired set"; + EXPECT_TRUE(blobExists(*backend, store->layout(), hash)); + EXPECT_EQ(backend->head(store->layout().blobKey(id)).token, t_seed) + << "spare does not touch the body — the incarnation token is unchanged"; + + /// ADD-ONLY: the spare must NOT clear the meta back to Clean (that is the deposed-leader hole). + { + const auto lm = loadMetaForTest(*backend, store->layout(), hash); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Condemned) + << "GC freshness meta is add-only: a spare leaves the meta Condemned (never -> Clean)"; + } + + /// Only a WRITER re-publishes Clean, and only by displacing the body with a fresh incarnation token: + /// a materialization attempt on the condemned hash republishes the writer's source — the body token CHANGES and + /// the meta flips to Clean WITH that token change. + const RootNamespace writer_ns{"00/spare-writer@cas@"}; + auto ref_w = publishBlobWithDurablePrecommit(store, writer_ns, "writer", id, payload); + EXPECT_EQ(ref_w.ref, id); + const Token t_resurrect = backend->head(store->layout().blobKey(id)).token; + EXPECT_NE(t_resurrect, t_seed) << "republication displaces the body with a fresh incarnation token"; + const auto lm_after = loadMetaForTest(*backend, store->layout(), hash); + ASSERT_TRUE(lm_after.has_value()); + EXPECT_EQ(lm_after->meta.state, MetaState::Clean) + << "the writer's republication path is the sole Condemned -> Clean transition"; +} + +/// Two-leader stale-redelete regression — the executable form of the deposed-leader spec §2. A stale +/// leader's pre-CAS exact-token redelete `deleteExact(h, t1)` must never delete a live reuse. With the +/// buggy clear-on-spare, a spare publishes `Clean`; a writer reads `Clean` and REUSES `t1`; the stale +/// `deleteExact(t1)` then deletes the LIVE body (INV_NO_LOSS). Add-only meta closes it: the spare leaves +/// `Condemned`, the writer resurrects to `t2`, and the stale `deleteExact(t1)` is a `TokenMismatch` no-op. +/// +/// Interleaving fidelity (APPROXIMATED): the deposed leader's destructive side effect is its pre-CAS +/// exact-token `deleteExact(h, t1)`. We reproduce it deterministically by CAPTURING `t1` at condemn time +/// (exactly the token a paused leader's `delete_pending` snapshot holds) and firing that exact +/// `deleteExact` AFTER the surviving leader's spare and the writer's republication — the faithful destructive +/// op, without a mid-round CAS-interrupt seam on the delete path (which the backend does not expose). +TEST(CASGCRetire, StaleRedeleteAfterSpareDoesNotDeleteLiveReuse) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + const String payload = "two-leader-stale-redelete-payload"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const String blob_key = store->layout().blobKey(id); + const RootNamespace seed_ns{"00/redelete-seed@cas@"}; + publishBlobWithDurablePrecommit(store, seed_ns, "seed", id, payload); + store->dropRef(seed_ns, "seed"); + store->renewWatermarkOnce(); + + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", hash)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + gc.runRegularRound(); /// +1 folds + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + gc.runRegularRound(); /// -1 => in-degree 0 => condemned at t1 + + /// The OLD leader L1's planned pre-CAS delete uses the EXACT token it observed at condemn: capture t1. + const auto condemned_entry = currentEntryFor(*backend, store->layout(), hash); + ASSERT_TRUE(condemned_entry.has_value()); + const Token t1 = condemned_entry->token; + ASSERT_EQ(backend->head(blob_key).token, t1); + + /// A NEW leader L2 folds a +1 that recovered h's in-degree and adopts a SPARE for h. + const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", hash)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_live", std::nullopt, r2); + gc.runRegularRound(); /// +1 => spared + + /// Add-only: the spare left the meta Condemned (the stale-redelete guard depends on it). + { + const auto lm = loadMetaForTest(*backend, store->layout(), hash); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Condemned) + << "add-only: a spare must not clear the meta (the writer must still see the hash condemned)"; + } + + /// A writer dedup-hits h. It point-reads Condemned and RESURRECTS to a fresh token t2 + /// from the writer's own source — it never reuses t1. + const RootNamespace writer_ns{"00/redelete-writer@cas@"}; + publishBlobWithDurablePrecommit(store, writer_ns, "writer", id, payload); + const Token t2 = backend->head(blob_key).token; + EXPECT_NE(t2, t1) << "the writer resurrected to a fresh incarnation, not a reuse of t1"; + + /// L1 resumes and executes its stale pre-CAS exact-token redelete `deleteExact(h, t1)`: it must be a + /// TokenMismatch no-op (the live body is now t2), NEVER a Deleted of the live reuse. + const DeleteOutcome stale = backend->deleteExact(blob_key, t1); + EXPECT_EQ(stale.kind, DeleteOutcome::Kind::TokenMismatch) + << "the stale exact-token redelete must miss the live reuse (add-only closes INV_NO_LOSS)"; + + /// The live body under t2 survives, stays reachable via the committed r2, and fsck sees no dangle. + const HeadResult hr = backend->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(hr.token, t2); + replaceRecoverableCkptForRawFixture( + *backend, store->layout(), ns, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + EXPECT_EQ(runFsck(*store, /*detail=*/false).dangling, 0u) + << "no live reference dangles: the stale redelete did not delete the reused body"; +} + +/// Copy-forward aftermath, republished arm (spec 2026-07-02-cas-copy-forward-condemned-evidence.md): +/// after a condemned incarnation (hash, t0) is displaced by a verified copy-forward (fresh token t1) +/// and the republished part's +1 lands, the listed (hash, t0) entry settles WITHOUT touching the new +/// incarnation: its exact-token delete is a mismatch no-op and the entry drops; the blob survives at t1. +TEST(CASGCRetire, CopyForwardedBlobSurvivesWhenRepublished) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + gc.runRegularRound(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + gc.runRegularRound(); /// -1 folds => in-degree 0 => entry (1, t0) condemned + ASSERT_TRUE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()); + + /// The raw equivalent of writer republication: displace exactly t0 with the + /// same verified bytes under a fresh token t1, then republish a part referencing the blob (the + /// promoted dst ref of a republishRef move). + const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); + const Token t0 = backend->head(blob_key).token; + const auto res = backend->putOverwrite(blob_key, backend->get(blob_key)->bytes, t0); + ASSERT_EQ(res.outcome, PutOutcome::Done); + const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_detached", std::nullopt, r2); + + /// The +1 folds => spared; the (1, t0) entry drops; the t1 incarnation is never deleted. + for (int i = 0; i < 4; ++i) + gc.runRegularRound(); + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()); + const HeadResult hr = backend->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_EQ(hr.token, res.token); +} + +/// Copy-forward aftermath, stale-entry arm: a listed (hash, t0) entry whose incarnation was +/// displaced (token now t1) with NO accompanying owner events. The entry graduates and its +/// exact-token delete MISMATCHES — a no-op, the entry drops, the t1 incarnation is NEVER +/// wrong-token-deleted (no wedge, no unsafe delete). This is a RAW-displacement model, stronger +/// than the real flow: in real `republishRef` the dst precommit + body are durable BEFORE the +/// promote pre-pass runs (reachability-before-content, B188), so an abandoned real copy-forward +/// is fully reclaimed by the pipeline (+1 spare -> reclaim -1 -> transition to zero -> fresh +/// (hash, t1) entry -> exact delete). The raw shape pins the GC-side invariant in isolation. +TEST(CASGCRetire, AbandonedCopyForwardDropsEntryWithoutWrongTokenDelete) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + runRegularRoundReclaiming(gc); + ASSERT_TRUE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()); + + const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); + const Token t0 = backend->head(blob_key).token; + const auto res = backend->putOverwrite(blob_key, backend->get(blob_key)->bytes, t0); + ASSERT_EQ(res.outcome, PutOutcome::Done); + + /// No events land at all (raw displacement). Drive rounds with the store's ack kept current so + /// the (1, t0) entry graduates; its exact-token delete mismatches t1 and the entry drops. + for (int i = 0; i < 6; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()) + << "the stale (hash, t0) entry must settle (mismatch redelete drops it), not wedge the list"; + const HeadResult hr = backend->head(blob_key); + ASSERT_TRUE(hr.exists) << "the fresh incarnation must never be deleted under the stale token"; + EXPECT_EQ(hr.token, res.token); +} + +/// A completed round adopts the SAME attempt its fold minted (the round's single gc/state CAS commits the +/// fold's (snap_generation, snap_attempt) together). Completion seals are a retired concept, so the durable +/// index of the adopted round is the FOLD seal at (snap_generation, snap_attempt). Across rounds each +/// `runRegularRound` re-acquires the lease (bumping `lease.seq`), so a later round mints a FRESH attempt. +TEST(CASGCRecheck, CompletionInheritsFoldAttempt) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + + gc.runRegularRound(); // round 1: one pass, single CAS commits (snap_generation, snap_attempt) + const auto after_round1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + // The round adopted the attempt of THIS round's fold: snap_attempt == the lease.seq that folded it. + EXPECT_EQ(after_round1.snap_attempt, after_round1.lease.seq); + EXPECT_GT(after_round1.snap_generation, 0u); + // The fold seal is durable under the adopted (snap_generation, snap_attempt) pair (no completion seal). + EXPECT_TRUE(backend->head(store->layout() + .foldSealKey(after_round1.snap_generation, after_round1.snap_attempt)).exists); + + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + gc.runRegularRound(); // round 2: re-acquire (bump lease.seq) -> fresh attempt at its fold + const auto after_round2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_EQ(after_round2.snap_attempt, after_round2.lease.seq); + EXPECT_GT(after_round2.snap_attempt, after_round1.snap_attempt); // per-round monotone attempt + EXPECT_GT(after_round2.snap_generation, after_round1.snap_generation); + EXPECT_TRUE(backend->head(store->layout() + .foldSealKey(after_round2.snap_generation, after_round2.snap_attempt)).exists); +} + +/// ---- round-paced graduation suite (spec 2026-07-02 + Task-9 amendment; re-keyed off acks in v3 Task 6) ---- + +/// A regular round performs NO writes to the ref objects: ref state is writer-owned (immutable +/// `_log`/`_snap`), and GC only reads it (plus deletes covered objects via ref-object cleanup, which +/// needs a covering snapshot -- none exists here). So a no-op round adds and removes NO ref object. +TEST(CASGCAckFloor, NoOpRoundDoesNotMutateRefShards) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, + ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}); + Gc gc(store, kGc); + gc.runRegularRound(); // first round folds the publish + + const auto listRefKeys = [&] + { + std::set keys; + String cursor; + for (;;) + { + const ListPage page = backend->list(store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + keys.insert(lk.key); + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return keys; + }; + + const std::set before = listRefKeys(); + ASSERT_FALSE(before.empty()) << "the publish must have written at least one ref object"; + + gc.runRegularRound(); // a second, no-op round must not add or remove any ref object + const std::set after = listRefKeys(); + EXPECT_EQ(before, after) << "a no-op GC round must not mutate the table's ref objects"; + // The registry object is gone (Task 4); the fence never existed to write it. + EXPECT_FALSE(backend->get("p/gc/registry").has_value()); +} + +/// The canonical pipeline: a blob condemned at round K stays present after the condemning round; the +/// VERY NEXT round graduates it (round-paced, unconditional — condemn_round < current_round the first +/// round current_round exceeds it) and publishes it delete_pending — the blob still exists; the round +/// AFTER THAT executes the exact-token delete and the blob becomes absent. This pins the critical +/// off-by-one: current_round MUST equal state.round + 1 (the SAME basis condemn_round is stamped at), +/// so an entry graduates exactly one round after it was condemned — never the same round, never never. +TEST(CASGCAckFloor, CondemnThenGraduatesNextRoundThenDeletes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + + runRegularRoundReclaiming(gc); // round 1: folds the +1; blob referenced + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + // The condemning round: the -1 drops in-degree to 0; the blob is condemned into the current retired + // list but NOT deleted. The entry is present and NOT yet pending. report.condemned counts it. + { + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep.condemned, 1u); // one blob condemned this round + EXPECT_EQ(rep.graduated, 0u); // must NOT graduate the same round it was condemned + EXPECT_EQ(rep.redeleted, 0u); // nothing pending to delete yet + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); + const auto e = currentEntryFor(*backend, store->layout(), blob); + ASSERT_TRUE(e.has_value()); + EXPECT_FALSE(e->delete_pending); // condemned, not yet graduated + } + + // The VERY NEXT round graduates it deterministically (no ack/heartbeat dependency). + { + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep.graduated, 1u); + EXPECT_EQ(rep.redeleted, 0u); // the delete lands on the NEXT pass, not this one + const auto e = currentEntryFor(*backend, store->layout(), blob); + ASSERT_TRUE(e.has_value()); + EXPECT_TRUE(e->delete_pending); + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); // pending: still present this pass + } + + // The pass AFTER the pending publish executes the exact-token delete; the blob becomes absent and the + // entry is dropped from the current retired list. report.redeleted counts the executed pending delete. + { + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep.redeleted, 1u); // the pending delete executed this round + EXPECT_FALSE(blobExists(*backend, store->layout(), blob)); + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); + } +} + +/// End-to-end through the real round driver (`Gc::runRegularRound`) rather than +/// `foldDeltasIntoGeneration` directly: a cohort well past `gc_round_redelete_budget` still drains +/// completely, but no single round's `redeleted` count exceeds the cap — the same convergence the +/// merge-level `CASThreeCursorMerge` budget tests pin, proven through the production entry point. +TEST(CASGCAckFloor, RedeleteBudgetCapsRoundDrainAndConverges) +{ + auto backend = std::make_shared(); + constexpr uint64_t kCap = 5; + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_round_redelete_budget = kCap}); + const RootNamespace ns{"00/aa@cas@"}; + + constexpr uint64_t kCohort = 20; + std::vector blobs; + std::vector refs; + for (uint64_t i = 1; i <= kCohort; ++i) + { + const UInt128 blob(i); + const ManifestRef r = ref("srv-a:1", i, static_cast(i)); + blobs.push_back(blob); + refs.push_back(r); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl" + std::to_string(i), std::nullopt, r); + } + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); // round 1: folds every publish, every blob referenced + + for (uint64_t i = 1; i <= kCohort; ++i) + dropRefTransition(*backend, store->layout(), ns, "tbl" + std::to_string(i), refs[i - 1]); + + { + const RoundReport rep = runRegularRoundReclaiming(gc); // condemning round + EXPECT_EQ(rep.condemned, kCohort); + EXPECT_EQ(rep.graduated, 0u); + EXPECT_EQ(rep.redeleted, 0u); + } + { + // Graduation has no budget configured in this test (default 0 = unbounded) — the whole + // cohort graduates together, isolating the redelete cap as the only thing under test. + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep.graduated, kCohort); + EXPECT_EQ(rep.redeleted, 0u); + } + + uint64_t total_redeleted = 0; + uint64_t rounds = 0; + while (total_redeleted < kCohort && rounds < 10) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_LE(rep.redeleted, kCap) << "a round must never redelete past gc_round_redelete_budget"; + total_redeleted += rep.redeleted; + ++rounds; + } + EXPECT_EQ(total_redeleted, kCohort) << "no entry lost to the cap across the whole drain"; + EXPECT_EQ(rounds, kCohort / kCap) << "ceil(20 / 5) rounds to fully drain the cohort"; + for (const UInt128 & blob : blobs) + EXPECT_FALSE(blobExists(*backend, store->layout(), blob)); +} + +/// A publish re-referencing the condemned blob before graduation is folded and SPARES the entry: the entry +/// is dropped (recovery wins even past graduation) and the blob survives. +TEST(CASGCAckFloor, PublishBeforeGraduationSpares) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); + const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + + gc.runRegularRound(); + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + gc.runRegularRound(); // condemns blob 1 (in-degree 0) + store->renewWatermarkOnce(); + ASSERT_TRUE(currentEntryFor(*backend, store->layout(), blob).has_value()); + + // Re-publish a committed ref pointing at the same blob BEFORE it graduates: the next pass folds the +1, + // the merge sees in-degree 1, and the entry is spared (dropped from the retired list). + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r2); + gc.runRegularRound(); + store->renewWatermarkOnce(); + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); // spared: entry dropped + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); + + // Keep running: the re-referenced blob must never be deleted. + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); +} + +/// A dead mount is fenced out by the round's heartbeat step: gc_fenced is set on its body (a +/// token-guarded rewrite that bumps seq). The fence is pure liveness (re-arms the write fence so a +/// resumed sleeper can never mutate again); reclaim itself no longer depends on any mount's heartbeat — +/// graduation paces on GC rounds. The fenced mount's own subsequent renew then fails closed, because the +/// fence invalidated the token it held. +/// +/// Rev.6 §token-stability observation (Task 9): the fence-out no longer trusts a bare wall-clock stamp +/// (`expires_at_ms`) against the GC's own clock — it fences ONLY once GC has watched a mount's write +/// token hold unchanged for the full threshold on its OWN monotonic clock. That takes (at least) two +/// `computeHeartbeatFloor` calls spanning the threshold, so this test drives the GC leader's own +/// (persistent) `mono_ms_fn` across two rounds: round 1 seeds the observation for both mounts; the +/// STORE's own mount is then renewed (as a live leader would) before round 2 crosses the threshold — +/// srid2, never renewed again after its one-shot claim, is the one that gets fenced. +TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) +{ + auto backend = std::make_shared(); + std::vector events; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + // srid2's keeper claims ONE lease via `start` and is never renewed again — tests never enable + // the runtime-owned renewal worker (`background_watermark` defaults to false), so this alone models a + // crashed process: a body that is live-shaped (not terminated, not fenced) but whose write token + // never changes again. + const String srid2 = "stale-server"; + MountLeaseKeeper srid2_keeper(backend, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, + std::chrono::milliseconds(100), [] { return 1000u; }, [] { return 0u; }, {}, + std::chrono::milliseconds(0), [] { return 0u; }); + srid2_keeper.start(); + ASSERT_FALSE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); + + // The fence-out threshold on the GC leader's OWN monotonic clock — mirrors the production formula + // in `Gc::runRegularRound` (ttl + 5% drift allowance + one round's worth of renewal slack). + const uint64_t ttl_ms = static_cast(store->poolConfig().mount_lease_ttl_ms.count()); + const uint64_t threshold_ms = ttl_ms + ttl_ms / 20 + + static_cast(store->poolConfig().mount_renew_period.count()); + + uint64_t gc_now = 1'000'000; // audit-only wall clock; never gates the fence decision + uint64_t gc_mono = 0; + Gc gc(store, kGc, [&] { return gc_now; }, [&] { return gc_mono; }); + + // Capture the emitted events so we can assert the round emits exactly one GcFenceOut row for srid2. + store->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(7); + writeBlobBody(*backend, layout, blob); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + + // Round 1 (mono 0): first sight of both mounts — observation starts, nothing fenced yet. + const RoundReport rep1 = gc.runRegularRound(); + EXPECT_EQ(rep1.fence_outs, 0u); + + // The store's OWN mount renews between rounds (as a live leader would); srid2 never does. + store->renewWatermarkOnce(); + gc_mono = threshold_ms; + + // Round 2 (mono == threshold): srid2's original token has held stable for the full threshold — + // fenced. The store's own (just-renewed) mount restarts its observation and stays live. + const RoundReport rep = gc.runRegularRound(); + + EXPECT_EQ(rep.fence_outs, 1u); // exactly one dead mount fenced-out this round + const MountLease fenced = decodeMountLease(backend->get(layout.mountKey(srid2))->bytes); + EXPECT_TRUE(fenced.gc_fenced); + + // Exactly one GcFenceOut audit row was emitted, naming srid2 in its detail. + size_t fence_out_rows = 0; + for (const CasEvent & e : events) + if (e.type == CasEventType::GcFenceOut) + { + ++fence_out_rows; + EXPECT_EQ(e.outcome, "fenced"); + EXPECT_FALSE(e.reason.empty()); + const auto it = e.detail.find("server_root_id"); + ASSERT_NE(it, e.detail.end()); + EXPECT_EQ(it->second, srid2); + } + EXPECT_EQ(fence_out_rows, 1u); + + // srid2's writer comes back and tries to renew: its held token was invalidated by the fence rewrite, + // so synchronous renewal returns a terminal failure. (It renews on its own clock; liveness is irrelevant — the token guard + // trips regardless.) + const MountRenewResult renewed = srid2_keeper.renew( + CasRequestBudget{.attempt_timeout_ms = 1, .operation_deadline_ms = 10, .max_attempts = 1, + .lease_safety_margin_ms = 0, .retry_initial_backoff_ms = 0, .retry_max_backoff_ms = 0}, + MountRenewOperationEnvironment{}); + ASSERT_EQ(renewed.outcome, MountRenewOutcome::Terminal); + ASSERT_NE(renewed.failure, nullptr); + EXPECT_THROW(std::rethrow_exception(renewed.failure), DB::Exception); + + // The fence-out is pure liveness cleanup: reclaim proceeds through the normal (round-paced) pipeline + // regardless of srid2's fate — fencing one stale mount must never wedge the reclaim pipeline. + dropRefTransition(*backend, layout, ns, "tbl", r); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, blob)); +} + +/// fix-round F6 (author-review: `Gc`'s own `mono_ms_fn` used to default to the RAW static `Pool:: +/// bootMs()`, bypassing the Pool's own injectable `config.boot_ms_fn` -- a time-controlled test can +/// desync the mount side's fake clock from the GC side's real one). This mirrors +/// `ExpiredMountFencedOutAndExcluded` exactly, except: the Pool is opened with an injected +/// `boot_ms_fn` driving a FAKE clock that barely advances in real time, and `Gc` is constructed WITHOUT +/// an explicit `mono_ms_fn` -- exercising the DEFAULT under test. If the default still read the real +/// wall clock, this round would see essentially zero elapsed mono time and never cross the fence-out +/// threshold; the fix makes it default to `store->bootMsNow()`, which tracks the SAME fake clock. +TEST(CASGCAckFloor, DefaultMonoClockTracksPoolsInjectedBootClockNotWallClock) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 0; + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .boot_ms_fn = [&] { return fake_boot; }}); + const Layout & layout = store->layout(); + + // A stale mount, exactly as `ExpiredMountFencedOutAndExcluded`: one claim, never renewed again. + const String srid2 = "stale-server"; + MountLeaseKeeper srid2_keeper(backend, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, + std::chrono::milliseconds(100), [] { return 1000u; }, [&] { return fake_boot; }); + srid2_keeper.start(); + ASSERT_FALSE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); + + const uint64_t ttl_ms = static_cast(store->poolConfig().mount_lease_ttl_ms.count()); + const uint64_t threshold_ms = ttl_ms + ttl_ms / 20 + + static_cast(store->poolConfig().mount_renew_period.count()); + + // `Gc` constructed with only `now_ms_fn` -- `mono_ms_fn` is left at its DEFAULT (the fix under test). + Gc gc(store, kGc, [] { return 1'000'000u; }); + + const RoundReport rep1 = gc.runRegularRound(); + EXPECT_EQ(rep1.fence_outs, 0u); + + store->renewWatermarkOnce(); + fake_boot = threshold_ms; // advance the FAKE clock only; this test runs in well under a millisecond + + const RoundReport rep2 = gc.runRegularRound(); + EXPECT_EQ(rep2.fence_outs, 1u) + << "Gc's default mono_ms_fn must track the Pool's injected boot clock, not the real wall clock"; + EXPECT_TRUE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); +} + +/// deleteExact against a blob the writer RECREATED (fresh incarnation, different token) between the pending +/// publish and the deleting pass lands TokenMismatch — a terminal-OK outcome recorded as a replace: the +/// fresh incarnation is a live object and survives. report.replaced counts it. +TEST(CASGCAckFloor, RecreatedBlobDeleteIsTokenMismatchOk) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + const BlobRef blob_id{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob)}; + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + runRegularRoundReclaiming(gc); // condemn (captures the ORIGINAL token) + store->renewWatermarkOnce(); + + // Drive rounds until the entry is delete_pending (the token it holds is the original observation). + bool pending = false; + for (int i = 0; i < 6 && !pending; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const auto e = currentEntryFor(*backend, store->layout(), blob); + pending = e && e->delete_pending; + } + ASSERT_TRUE(pending); + + // The writer recreates the blob with a FRESH incarnation before the deleting pass: the current token no + // longer matches the pending entry's captured token. + displaceBlobToken(*backend, store->layout(), blob_id); + + // The deleting pass issues deleteExact(entry.token) → TokenMismatch → Replaced. The fresh incarnation + // survives; the entry is dropped. + const RoundReport rep = runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + EXPECT_EQ(rep.replaced, 1u); + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); // the recreated incarnation is live + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); +} + +/// Idempotent replay of a crashed round: a fresh Gc instance (new lease seq = new attempt) re-runs a round +/// and completes; a delete that already landed under a prior pass replays onto NotFound (Absent outcome) +/// and the round still completes. We model the crash-after-delete-before-CAS replay by manually deleting +/// the pending blob (its exact token) BEFORE the deleting pass, then asserting the pass reports the delete +/// as absent (report.absent == 1) and completes (round advances). +TEST(CASGCAckFloor, ResumeAfterCrashBetweenRetiredPutAndStateCas) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + const BlobRef blob_id{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob)}; + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + // A fresh Gc per round (each acquires the lease, bumping lease.seq = a fresh attempt) — the replay + // property: no wedging, each round completes under its own fresh attempt. + { + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + RetiredEntry pending_entry; + bool pending = false; + for (int i = 0; i < 6 && !pending; ++i) + { + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const auto e = currentEntryFor(*backend, store->layout(), blob); + if (e && e->delete_pending) + { + pending = true; + pending_entry = *e; + } + } + ASSERT_TRUE(pending); + + // Simulate a crashed deleting pass that DID land the exact-token delete but crashed before the gc/state + // CAS. The next (fresh-attempt) pass replays the delete → the object is already gone → NotFound → the + // pass records Absent and completes. + ASSERT_EQ(backend->deleteExact(store->layout().blobKey(blob_id), pending_entry.token).kind, + DeleteOutcome::Kind::Deleted); + + const uint64_t round_before = decodeGcState(backend->get(store->layout().gcStateKey())->bytes).round; + Gc gc2(store, kGc); + const RoundReport rep = runRegularRoundReclaiming(gc2); + store->renewWatermarkOnce(); + EXPECT_EQ(rep.absent, 1u); // the replayed delete found the object already gone + const uint64_t round_after = decodeGcState(backend->get(store->layout().gcStateKey())->bytes).round; + EXPECT_GT(round_after, round_before); // the round completed (no wedge) + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); +} + +/// Backend-agnostic regression for the rustfs 412-on-absent quirk: a conditional exact-token delete +/// against an object that is ALREADY absent answers `TokenMismatch`, not `NotFound`, on this backend +/// (`TokenMismatchOnAbsentBackend` reproduces it deterministically). The redelete site must disambiguate +/// via a follow-up HEAD: the object is truly gone, so the outcome must settle as Absent (never Replaced) +/// and the `.meta` cleanup (gated on Deleted/NotFound) must still run. +TEST(CASGCAckFloor, TokenMismatchOnAbsentBlobSettlesAsAbsentAndDropsMeta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + const BlobRef blob_id{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob)}; + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + runRegularRoundReclaiming(gc); // condemn + store->renewWatermarkOnce(); + + // Drive rounds until the entry is delete_pending, capturing its exact condemn-time token. + RetiredEntry pending_entry; + bool pending = false; + for (int i = 0; i < 6 && !pending; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const auto e = currentEntryFor(*backend, store->layout(), blob); + if (e && e->delete_pending) + { + pending = true; + pending_entry = *e; + } + } + ASSERT_TRUE(pending); + { + const auto lm = loadMetaForTest(*backend, store->layout(), blob); + ASSERT_TRUE(lm.has_value()) << "the blob must still carry its Condemned freshness meta pre-delete"; + ASSERT_EQ(lm->meta.state, MetaState::Condemned); + } + + // The object is genuinely gone already (as if a prior crashed pass landed the delete); confirm that, + // then arm the quirk so the NEXT conditional delete against this now-absent key answers TokenMismatch + // instead of NotFound (the rustfs 412-on-absent behavior). + const String blob_key = store->layout().blobKey(blob_id); + ASSERT_EQ(backend->deleteExact(blob_key, pending_entry.token).kind, DeleteOutcome::Kind::Deleted); + ASSERT_FALSE(backend->head(blob_key).exists); + backend->quirkOnAbsent(blob_key); + + // The deleting pass replays deleteExact(entry.token): the backend answers TokenMismatch (quirk), but + // the follow-up HEAD shows the object absent, so the fix disambiguates the outcome to Absent and still + // runs the `.meta` cleanup. + const RoundReport rep = runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + EXPECT_EQ(rep.absent, 1u) << "the 412-on-absent quirk must settle as Absent, not Replaced"; + EXPECT_EQ(rep.replaced, 0u); + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); + EXPECT_FALSE(loadMetaForTest(*backend, store->layout(), blob).has_value()) + << ".meta cleanup (gated on Deleted/NotFound) must still run on the disambiguated Absent outcome"; +} + +/// ---- condemn-marker gate suite ---- +/// +/// The per-hash condemn marker is LOAD-BEARING for the delete edge: the writer's adopt gate point-reads +/// the meta and an ABSENT meta reads as Clean, so a blob whose condemn-marker write was swallowed can be +/// same-token adopted by a writer landing in the [discovery-LIST, deleteExact] window — invisible to the +/// graduating fold — and the exact-token redelete then deletes a body under a live committed edge +/// (dangling manifest). Graduation to `delete_pending` therefore requires CONFIRMED durable `Condemned` +/// evidence for the entry; absent evidence CARRIES the entry to the next round (fail-safe delay, never a +/// fail-open delete) and retries the marker so a healed backend restores liveness. + +/// A condemned entry whose marker write was swallowed must be CARRIED round after round — never +/// graduated, never deleted — until durable `Condemned` evidence exists. Once the backend heals, the +/// carry-time marker retry lands and the normal two-phase pipeline reclaims the blob (delay, not a leak). +TEST(CASGCCondemnMarker, SwallowedMarkerWriteCarriesEntryInsteadOfDeleting) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + store->setCasRetrySleepForTest([](uint64_t) {}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + + runRegularRoundReclaiming(gc); // +1 folds; blob referenced + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + runRegularRoundReclaiming(gc); // the condemning round; the controlled marker write exhausts as Unresolved + ASSERT_FALSE(loadMetaForTest(*backend, store->layout(), blob).has_value()) + << "precondition: the injected fault must have lost the condemn-marker write"; + ASSERT_TRUE(currentEntryFor(*backend, store->layout(), blob).has_value()) + << "precondition: the retired entry must have been committed despite the lost marker"; + + /// Rounds keep coming while the marker stays unwritable: without durable Condemned evidence the + /// entry must be CARRIED — a writer reading the absent meta may have adopted this exact token. + const auto carries_before = + ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry].load(); + for (int i = 0; i < 4; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)) + << "round " << i << " after condemn: deleted without a durable condemn marker"; + } + const auto e = currentEntryFor(*backend, store->layout(), blob); + ASSERT_TRUE(e.has_value()) << "the entry must remain retired (carried), not dropped"; + EXPECT_FALSE(e->delete_pending) << "graduation must be refused without a confirmed marker"; + EXPECT_FALSE(e->marker_confirmed); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry].load() + - carries_before, 4u) + << "every refused graduation must count one unconfirmed carry"; + + /// Heal the backend: the carry-time retry publishes the marker, the entry confirms + graduates, and + /// the pipeline reclaims the blob and drops the meta. + backend->fail_meta_writes.store(false); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), blob)); + EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); + EXPECT_FALSE(loadMetaForTest(*backend, store->layout(), blob).has_value()); +} + +/// The healthy-path counterpart: with the condemn-time marker write landing normally, the gate must not +/// change the canonical schedule — condemned at round K, graduated (delete_pending) at K+1, deleted at +/// K+2 — and the durable Condemned marker exists from the condemning round on. +TEST(CASGCCondemnMarker, DurableMarkerKeepsCanonicalGraduationSchedule) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + + runRegularRoundReclaiming(gc); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + { + const RoundReport rep = runRegularRoundReclaiming(gc); // condemn round K + EXPECT_EQ(rep.condemned, 1u); + const auto lm = loadMetaForTest(*backend, store->layout(), blob); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Condemned); + } + { + const RoundReport rep = runRegularRoundReclaiming(gc); // K+1: confirmed marker => graduates on schedule + EXPECT_EQ(rep.graduated, 1u); + const auto e = currentEntryFor(*backend, store->layout(), blob); + ASSERT_TRUE(e.has_value()); + EXPECT_TRUE(e->delete_pending); + EXPECT_TRUE(e->marker_confirmed) << "a delete_pending row must carry the confirmation bit"; + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); + } + { + const RoundReport rep = runRegularRoundReclaiming(gc); // K+2: the pending delete executes + EXPECT_EQ(rep.redeleted, 1u); + EXPECT_FALSE(blobExists(*backend, store->layout(), blob)); + } +} + +/// The leader-restart path: `condemn_markers_confirmed` is a process-local registry on `Gc`, lost on a +/// GC leader restart. Same idiom as the crash-replay tests above (`Gc gc2(store, kGc)` -- a fresh `Gc` +/// object under the SAME identity models a process restart that resumes its own lease, not a steal by a +/// different owner). The fresh instance must still confirm graduation via the ONE synchronous `loadMeta` +/// re-check: the durable `Condemned` meta observed now is sufficient evidence on its own, with no +/// in-process (hash, token) confirmation available at all. This proves the fallback branch -- not just +/// the in-process registry -- authorizes the delete. +TEST(CASGCCondemnMarker, LoadMetaFallbackConfirmsGraduationAfterLeaderRestart) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + { + /// The first (soon-to-be-gone) leader: seeds the blob, condemns it, and lets the marker write + /// land on the healthy backend. Its `condemn_markers_confirmed` registry dies with it. + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + const RoundReport rep = runRegularRoundReclaiming(gc); // condemn round + EXPECT_EQ(rep.condemned, 1u); + store->renewWatermarkOnce(); + } + const auto lm = loadMetaForTest(*backend, store->layout(), blob); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Condemned) + << "precondition: the durable marker must be on disk before the simulated restart"; + + /// A brand-new `Gc` object under the SAME identity -- an empty `condemn_markers_confirmed`, exactly + /// as after a process restart that resumes its own lease. It never observed the condemn round above, + /// so the in-process confirmation path (`condemnMarkerConfirmedInProcess`) has nothing to return true + /// for; only the `loadMeta` fallback can authorize graduation. + Gc gc2(store, kGc); + const RoundReport rep = runRegularRoundReclaiming(gc2); + EXPECT_EQ(rep.graduated, 1u) + << "the loadMeta fallback (leader-restart path) must authorize graduation from durable evidence " + "alone"; + const auto e = currentEntryFor(*backend, store->layout(), blob); + ASSERT_TRUE(e.has_value()); + EXPECT_TRUE(e->delete_pending); + EXPECT_TRUE(e->marker_confirmed) << "a delete_pending row confirmed via loadMeta still carries the bit"; + EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); +} diff --git a/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp new file mode 100644 index 000000000000..870bcdc20295 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp @@ -0,0 +1,595 @@ +#include + +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include + +/// ARITHMETIC REF INTAKE (spec 2026-07-27 "ref chain complete cut" §5). +/// +/// The GC fold used to walk the ids the round's LIST returned. That made the LIST a source of TRUTH +/// about which records exist, and an object store that omits a durable key from a listing -- observed +/// in production as the `0x1430c`/`0x1430d` shape -- silently skipped those records' owner edges and +/// then sealed a cursor ABOVE them, so their blobs looked unreferenced forever after. +/// +/// Under INV-1 (per-namespace contiguous ids) the ids within one `(namespace, writer_epoch)` are dense +/// `1..T`, so the next record's id is COMPUTABLE: `cursor + 1`. The fold therefore steps by arithmetic +/// and reads each expected id by EXACT key (the per-record GET was always owed -- the round read every +/// record's body anyway). The listing is demoted to a HINT with two jobs: it says which namespaces +/// exist, and it supplies the witnesses that make an absent expected-next decidable: +/// +/// * absent at `expected`, no listed id above it => the namespace's frontier this round (normal end) +/// * absent at `expected`, a listed id above it => IMPOSSIBLE under contiguity: the store is lying +/// or a durable record was lost. Hold the namespace +/// (classification 4), cursor unmoved. +/// +/// Epochs are crossed ONLY by consuming the `EpochSeal` that closes an epoch (INV-2). The seal folds as +/// an applied no-op (probe B2: `produced=false`), and the next epoch's start is `{E', 1}` -- reached +/// through the `prev_epoch_seal` back-chain, never guessed from the hint, so an epoch the hint omits +/// entirely is still walked. +/// +/// These tests drive REAL rounds over an in-memory pool whose LIST omits keys that are genuinely +/// present (`HintHoleBackend`) -- the production shape, reproduced. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +/// The lying store is `HintHoleBackend` from `cas_test_helpers.h`: LIST permanently omits keys that +/// stay readable by exact key, so these tests exercise the intake's arithmetic walk rather than any +/// one `list` call. + +/// The `RefCoverage` the newest fold seal recorded for `ns`'s opaque catalog life, or nullopt when the +/// seal has no entry for it. Scans downward from the adopted generation for the most recent fold seal, +/// mirroring `foldCursorOf`'s reasoning (a completed round's `gc/state` points at the recheck generation). +std::optional coverageOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const uint64_t gen = currentGenerationOf(backend, layout); + const uint64_t attempt = currentAttemptOf(backend, layout); + const UInt128 life_id = catalogLifeIdForTest(backend, layout, ns); + for (uint64_t g = gen; ; --g) + { + if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + { + const CasFoldSeal seal = decodeFoldSeal(got->bytes); + const auto it = seal.ref_lives.find(life_id); + if (it == seal.ref_lives.end()) + return std::nullopt; + return it->second.coverage; + } + if (g == 0) + return std::nullopt; + } +} + +RefTxnId cursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto cov = coverageOf(backend, layout, ns); + return cov ? cov->last_folded_ref_id : RefTxnId{}; +} + +uint8_t classificationOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto cov = coverageOf(backend, layout, ns); + return cov ? cov->classification : 0; +} + +/// The `fold_ref_intake` phase metrics of the round `sched` runs -- the only place probe B1's two +/// numbers are observable. +std::map runRoundAndReadIntakeMetrics(const PoolPtr & store) +{ + std::vector rows; + CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca", + [&](const GcRoundLogRecord & r) { rows.push_back(r); }); + EXPECT_TRUE(sched.runOneRoundNow(GcRoundLogRecord::Trigger::Manual).acquired_lease); + for (const GcRoundLogRecord & r : rows) + if (r.event_type == GcRoundLogRecord::EventType::Phase && r.phase == "fold_ref_intake") + return r.phase_metrics; + return {}; +} + +} + +/// ===================== THE BLOCKER, AS A UNIT TEST ===================== +/// +/// Five records, all durable and all readable by exact key; the hint omits the two in the MIDDLE. +/// Arithmetic intake never consults the hint for what to read next, so the omission is a non-event: +/// every record folds, every blob keeps its owner edge, and the cursor lands on the true tail. +/// +/// Under listing-driven intake this test fails on the blobs, not on the cursor: the cursor still +/// reached `{1, 5}` (the last LISTED id) while the two hidden records' edges were never folded -- the +/// exact damage shape the production blocker caused, since a cursor sealed above an unfolded record can +/// never be re-read. +TEST(CASGCArithmeticIntake, HintOmittingMiddleRecordsFoldsThroughUnnoticed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + for (uint64_t i = 1; i <= 5; ++i) + publishAt(*backend, layout, ns, RefTxnId{1, i}, "ref_" + std::to_string(i), i, + DB::UInt128(i), /*birth=*/i == 1); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 5}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 3})); + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 4})); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_GT(backend->holesServed(), 0u) << "the hint hole was never actually served"; + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 5})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 2) << "a folded namespace is `changed`"; + for (uint64_t i = 1; i <= 5; ++i) + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) + << "blob " << i << " lost its owner edge: its record was skipped because the hint omitted it"; +} + +/// A clean, hole-free namespace still ends its walk exactly where it should: at the first absent id, +/// with no witness above it and therefore no hold. +TEST(CASGCArithmeticIntake, WalkEndsAtFrontierWithoutHold) +{ + auto backend = std::make_shared(); + /// Fold every round: the default defer window would skip the second round entirely and leave the + /// first round's seal in place, so the assertion below would read a stale coverage record. + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + for (uint64_t i = 1; i <= 3; ++i) + publishAt(*backend, layout, ns, RefTxnId{1, i}, "ref_" + std::to_string(i), i, + DB::UInt128(i), /*birth=*/i == 1); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + + /// A second round over an unchanged namespace pays exactly one exact GET, finds the same frontier, + /// and neither advances nor holds. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 1) << "an unchanged namespace is `carried`"; +} + +/// ===================== EPOCHS ARE CROSSED ONLY BY CONSUMING A SEAL ===================== +/// +/// `{1,1} {1,2} seal{1,3} | {2,1} {2,2}`: the seal is applied as a table no-op, counted applied, and +/// the walk continues at `{2, 1}` -- whose `prev_epoch_seal` names the seal just consumed, which is what +/// makes the crossing provable rather than guessed. +TEST(CASGCArithmeticIntake, SealCrossesEpochAndIsAppliedAsNoOp) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + writeSealAt(*backend, layout, ns, RefTxnId{1, 3}); + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_3", 3, DB::UInt128(3), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 3}); + publishAt(*backend, layout, ns, RefTxnId{2, 2}, "ref_4", 4, DB::UInt128(4)); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 3}, + }); + + /// The hint hides the seal AND the new epoch's first record: neither the epoch boundary nor its + /// start may depend on the listing. + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 3})); + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{2, 1})); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_GT(backend->holesServed(), 0u); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{2, 2})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + for (uint64_t i = 1; i <= 4; ++i) + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) << "blob " << i; +} + +/// Two chained seals in ONE round, the middle epoch entirely EMPTY (its seal is its only record, at +/// sequence 1, carrying `prev_epoch_seal`). The hint omits that whole epoch, so the only way to reach it +/// is the back-chain: the record the hint DOES show names the seal that must be consumed first. +TEST(CASGCArithmeticIntake, ChainedEmptyEpochSealsBothConsumedInOneRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + writeSealAt(*backend, layout, ns, RefTxnId{2, 1}, /*prev_epoch_seal=*/RefTxnId{1, 2}); + publishAt(*backend, layout, ns, RefTxnId{3, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{2, 1}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{3, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{2, 1}, + }); + + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{2, 1})); + + const auto intake = runRoundAndReadIntakeMetrics(store); + ASSERT_FALSE(intake.empty()) << "no fold_ref_intake row"; + ASSERT_GT(backend->holesServed(), 0u); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{3, 1})); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); + + /// THE ASSERTION THAT MAKES THIS TEST ABOUT THE BACK-CHAIN. Every check above is also satisfied by + /// an id-ordered walk over the listed keys — which is why this case passed before the change. Two + /// crossings can only happen by consuming `seal{1,2}` and then `seal{2,1}`, and `{2,1}` is reachable + /// only through `{3,1}`'s `prev_epoch_seal`, since the hint never mentions it. + EXPECT_EQ(intake.at("epoch_crossings"), 2u) + << "the walk must cross TWICE, through a hidden epoch it can only reach by the seal chain"; + EXPECT_EQ(intake.at("logs_accounted"), intake.at("logs_applied")); + EXPECT_EQ(intake.at("logs_applied"), 4u) << "two records and two seals, all applied"; + /// The checkpoint names the complete, authoritative frontier, so the bounded walk has no absent + /// probes to perform. The hidden epoch remains reachable only through the seal chain. + EXPECT_EQ(intake.at("absent_probes"), 0u); +} + +/// A round that ends ON a seal (nothing above it yet) leaves the cursor there. The NEXT round must +/// still cross into the epoch that appears later -- the cursor sitting on a closed epoch's seal is the +/// ordinary steady state after a writer-epoch change, not a wedge. +TEST(CASGCArithmeticIntake, CursorRestingOnSealCrossesInALaterRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) << "the round consumed the seal"; + + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{2, 1}); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{2, 1})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); +} + +/// ===================== IMPOSSIBLE SHAPES HOLD THE NAMESPACE ===================== +/// +/// `{1,3}` is genuinely absent while `{1,4}` is present AND listed. Contiguity says that cannot happen, +/// so whatever sits behind the gap may be an acked `+1`: the namespace is held at classification 4 with +/// its cursor UNMOVED, rather than sealing past the gap. +/// +/// Listing-driven intake folded `{1,4}` and sealed the cursor at it -- permanently, since a record below +/// the cursor is never re-read. +TEST(CASGCArithmeticIntake, GapBelowWitnessHoldsNamespaceAtClassificationFour) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + /// {1,3} is never written -- the record that vanished. + publishAt(*backend, layout, ns, RefTxnId{1, 4}, "ref_4", 4, DB::UInt128(4)); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 4}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) + << "the cursor must not advance past a gap"; + EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(4)), 0) + << "the record above the gap was not folded"; +} + +/// A record of a LATER epoch reachable while the current epoch's seal was never consumed: the crossing +/// has no proof (the later epoch's `prev_epoch_seal` names a seal this cursor never reached), so the +/// namespace holds instead of jumping the boundary. +TEST(CASGCArithmeticIntake, UnconsumedSealCrossingHoldsNamespace) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + /// Epoch 1's missing `{1,2}` seal is the exact position epoch 2 claims to chain from. With no + /// same-epoch witness above it, the later epoch is the only witness and the crossing must hold. + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + const auto coverage = coverageOf(*backend, layout, ns); + ASSERT_TRUE(coverage && coverage->hold.has_value()); + EXPECT_EQ(coverage->hold->reason, HoldReason::UnconsumedSealCrossing); + EXPECT_EQ(coverage->hold->offending_position, (RefTxnId{1, 2})); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 0); +} + +/// The back-chain proves the IDENTITY of the position an epoch chains from; it does not, by itself, +/// prove that position is a SEAL. Here a writer names an ordinary record (`{1,2}`, epoch 1's last +/// record, never sealed) as `{2,1}`'s `prev_epoch_seal`. Identity matches, so a chain-only check would +/// grant the crossing and declare epoch 1 closed while its writer may still be appending -- any later +/// `{1,k}` would then land permanently below the cursor, which is exactly the damage the seal exists to +/// prevent. The walk applied that record itself this round, so it knows its kind for free: refuse. +TEST(CASGCArithmeticIntake, CrossingFromANonSealRecordIsRefusedEvenWhenTheChainMatches) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + /// `{1,2}` is an ordinary published record, NOT an `EpochSeal` -- and epoch 2 chains to it anyway. + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_3", 3, DB::UInt128(3), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) + << "epoch 1 was never sealed, so the cursor may not leave it"; + EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1) << "epoch 1's records still fold"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(3)), 0) + << "the record beyond the unsealed boundary must NOT be folded"; +} + +/// The crossing reads the epoch-start record to prove the chain, and the walk then reads it again to +/// fold it. A record that answers the first read and not the second would make the next iteration +/// re-derive the SAME crossing from the same unchanged cursor and resolve to the same position -- an +/// infinite spin inside one namespace's walk. The strict-progress guard turns that into a hold. +/// +/// The fixture is the only shape that reaches it: a key that alternates present/absent across reads, so +/// `crossFromSeal` keeps succeeding while the walk's own GET keeps failing. +TEST(CASGCArithmeticIntake, EpochStartThatAnswersOnlyEveryOtherReadHoldsInsteadOfSpinning) +{ + /// Answers `flaky` on odd-numbered reads and 404s on even ones. Nothing else is disturbed. + class AlternatingGetBackend : public InMemoryBackend + { + public: + using DB::Cas::Backend::get; + String flaky; + size_t reads = 0; + + std::optional get(const String & key, Range range) override + { + if (key == flaky && ++reads % 2 == 0) + return std::nullopt; + return InMemoryBackend::get(key, range); + } + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + /// A THIRD epoch is what keeps the unstable position from reading as a frontier: without a witness + /// strictly above it, an absent `{2,1}` is just "the namespace ends here" and the walk stops + /// normally. With `{3,1}` listed, the walk must keep trying to cross -- and the chain from `{3,1}` + /// leads back to `{2,1}` every time, which is the spin. + publishAt(*backend, layout, ns, RefTxnId{3, 1}, "ref_3", 3, DB::UInt128(3), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{2, 1}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{3, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{2, 1}, + }); + + /// Arm only after seeding, so the fixture's own writes are undisturbed and the read counter starts + /// at the round's first read of this key (`crossFromSeal`'s, which must succeed). + backend->flaky = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{2, 1}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); /// must RETURN -- the spin is the failure mode + ASSERT_GE(backend->reads, 3u) + << "the crossing must have re-proved the same position after the walk failed to read it; " + "fewer reads means the guard was never reached"; + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) + << "the cursor stops on the seal it consumed and never enters the unstable epoch"; + EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 0); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(3)), 0) + << "nothing above the unstable position may be folded either"; +} + +/// ===================== A PER-NAMESPACE FAILURE IS NOT A ROUND FAILURE ===================== +/// +/// Spec §5 narrows the whole-round abort to a key that cannot be attributed to any namespace. An +/// undecodable BODY belongs to exactly one namespace, so it clamps that namespace and nothing else: +/// `ns_a` holds at its last good record while `ns_b` folds and seals normally in the same round. +TEST(CASGCArithmeticIntake, CorruptBodyClampsOneNamespaceWhileAnotherFolds) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns_a{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns_a); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + const RootNamespace ns_b{"00/bb@cas@"}; + + publishAt(*backend, layout, ns_a, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + backend->putIfAbsent(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, 2}), "this is not a cas_ref_log object"); + writeRecoverableCkptForRawFixture(*backend, layout, ns_a, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + publishAt(*backend, layout, ns_b, RefTxnId{1, 1}, "ref_1", 11, DB::UInt128(11), /*birth=*/true); + publishAt(*backend, layout, ns_b, RefTxnId{1, 2}, "ref_2", 12, DB::UInt128(12)); + writeSealAt(*backend, layout, ns_b, RefTxnId{1, 3}); + writeRecoverableCkptForRawFixture(*backend, layout, ns_b, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 3}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns_a), (RefTxnId{1, 1})); + EXPECT_EQ(classificationOf(*backend, layout, ns_a), 4); + + EXPECT_EQ(cursorOf(*backend, layout, ns_b), (RefTxnId{1, 3})) + << "a sibling namespace's corrupt body must not stop this one"; + EXPECT_EQ(classificationOf(*backend, layout, ns_b), 2); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(11)), 1); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(12)), 1); +} + +/// ===================== PROBE B1 OVER AN ARITHMETIC CUT ===================== +/// +/// `logs_accounted` is recomputed from the SEALED cut -- the arithmetic distance the round's cursors +/// claim to cover -- and compared with the count the walk incremented once per applied record. The two +/// can only differ if a cursor moved over a position nothing applied, which is precisely the damage +/// listing-driven intake used to do silently. The identity must survive both a hint hole (positions +/// applied that the listing never mentioned) and a seal crossing (an applied no-op, and a cut that +/// spans two epochs). +TEST(CASGCArithmeticIntake, B1IdentityHoldsOverAHoleyCutThatCrossesASeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + writeSealAt(*backend, layout, ns, RefTxnId{1, 3}); + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_3", 3, DB::UInt128(3), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 3}); + publishAt(*backend, layout, ns, RefTxnId{2, 2}, "ref_4", 4, DB::UInt128(4)); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 3}, + }); + + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2})); + + const auto intake = runRoundAndReadIntakeMetrics(store); + ASSERT_FALSE(intake.empty()) << "no fold_ref_intake row"; + ASSERT_GT(backend->holesServed(), 0u); + + EXPECT_EQ(intake.at("logs_accounted"), intake.at("logs_applied")); + EXPECT_EQ(intake.at("logs_applied"), 5u) + << "four records and the seal: the seal is APPLIED, as a no-op"; + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{2, 2})); +} + +/// A namespace the hint omits ENTIRELY still folds through the checkpoint's authoritative frontier. +/// Every log key remains readable by exact key, and the cursor reaches `{1,3}` despite the empty hint. +/// When the hint reappears, it changes neither the cursor nor the owner edges. +TEST(CASGCArithmeticIntake, WhollyOmittedNamespaceFoldsThroughAuthoritativeCheckpoint) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + for (uint64_t i = 1; i <= 3; ++i) + publishAt(*backend, layout, ns, RefTxnId{1, i}, "ref_" + std::to_string(i), i, + DB::UInt128(i), /*birth=*/i == 1); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + for (uint64_t i = 1; i <= 3; ++i) + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, i})); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const auto hidden_cov = coverageOf(*backend, layout, ns); + ASSERT_TRUE(hidden_cov.has_value()) + << "the namespace is `Live` in the catalog, so it stays in the universe even fully hidden"; + EXPECT_EQ(hidden_cov->classification, 2) << "the checkpoint's frontier is folded by exact key"; + EXPECT_EQ(hidden_cov->last_folded_ref_id, (RefTxnId{1, 3})); + + /// The store stops lying: the already folded namespace reappears. + backend->revealAll(); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})); + for (uint64_t i = 1; i <= 3; ++i) + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) << "blob " << i; +} diff --git a/src/Disks/tests/gtest_cas_gc_attempt.cpp b/src/Disks/tests/gtest_cas_gc_attempt.cpp new file mode 100644 index 000000000000..b919298ae24e --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_attempt.cpp @@ -0,0 +1,189 @@ +#include + +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +} + +/// Unit-level GC-CONCURRENT-LEADER-LEAK regression (the original bug the attempt-scoped-generation fix +/// closes), ported to the one-pass ack-floor round. +/// +/// The historical wedge: two GC leaders fold the same generation. A DEPOSED leader writes its +/// `fold_seal(G_f)` to a FINAL `gc/gen//fold_seal` key just before its lease-guarded `gc/state` CAS +/// fails (lease lost mid-round). That orphaned write-once seal then poisons every future round: each +/// honest round recomputes `G_f`, hits the orphan's divergent bytes, throws "concurrent leader" +/// (`ABORTED`) forever — GC wedged, nothing reclaimed. +/// +/// The fix: every per-round `gc/gen` artifact is ATTEMPT-scoped (keyed by the folding leader's +/// `lease.seq`). A deposed leader writes its fold seal under its OWN attempt `a1`, which the failed +/// `gc/state` CAS never adopts — so it is pure unadopted debris, invisible to every reader resolving +/// only the adopted `(snap_generation, snap_attempt)`. The next honest round renews the lease (a fresh +/// `lease.seq`), folds under a DIFFERENT attempt, never collides, and drains. +/// +/// In the ONE-PASS round there is a SINGLE `gc/state` CAS per round (fold/publish/deletes all precede it), +/// so the deposition point is simply that single round-commit CAS. This test denies it once (leaving the +/// deposed fold seal under `a1`), then runs an honest GC to a fixpoint and asserts it drains the +/// now-unreachable blob to zero without wedging. + +namespace +{ + +const UInt128 kGcA = hexToU128("00000000000000000000000000000001"); + +ManifestRef ref(const String &, uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} + +/// Whether a blob's body object is present in the backend (HEADs the object key directly). +bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +/// Whether the CURRENT retired list (any gc-shard) still holds an entry — the ack-floor deletion pipeline +/// is in flight while this is true. +bool anyRetiredPending(const PoolPtr & s) +{ + /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// separate retired list — reconstruct the in-flight set from the seal. + return anyCondemnedInSeal(s->backend(), s->layout()); +} + +/// Drive regular GC to a fixpoint over the ACK-FLOOR round (advancing the store's own mount ack after each +/// round so the floor follows the committed round; stay alive while any work counter is nonzero OR the +/// current retired list still holds an in-flight entry). +size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(s)) + break; + } + return rounds; +} + +/// A backend that throws ONCE on the SINGLE round-commit `gc/state` CAS — the casPut that advances +/// snap_generation (the one-pass round has exactly one such CAS; the lease-acquire CAS does not advance +/// snap_generation, so "advances snap_generation" uniquely picks the round commit). +class InterruptRoundCasBackend : public InMemoryBackend +{ +public: + explicit InterruptRoundCasBackend(String gc_state_key_) : gc_state_key(std::move(gc_state_key_)) {} + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (arm_interrupt && key == gc_state_key) + { + const auto stored = get(key); + const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; + const uint64_t next_gen = decodeGcState(bytes).snap_generation; + if (next_gen > stored_gen) + { + arm_interrupt = false; /// one-shot: only depose the first round-commit CAS + throw DB::Exception(DB::ErrorCodes::ABORTED, + "test-injected: round-commit gc/state CAS denied (leader deposed mid-round; lease lost)"); + } + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + bool arm_interrupt = false; + +private: + String gc_state_key; +}; + +} + +/// A leader whose round-commit CAS is denied (lease lost mid-round) leaves its fold seal ONLY under its +/// own attempt `a1`; it never occupies the adopted attempt, so a subsequent honest round is not wedged +/// and drains the now-unreachable blob to zero. +TEST(CASGCAttempt, DeposedFoldAttemptDoesNotWedge) +{ + auto backend = std::make_shared(/*gc_state_key*/ "p/gc/state"); + auto store = openPoolForTest(backend); + ASSERT_EQ(store->layout().gcStateKey(), "p/gc/state"); // guard the injected key against layout drift + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGcA); + + // Round 1 (honest): fold the +1 so the blob is pinned in the in-degree generation, and adopt the + // first (snap_generation, snap_attempt). + runRegularRoundReclaiming(gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) << "blob pinned by the committed ref"; + const auto after_fold = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + ASSERT_EQ(after_fold.snap_attempt, after_fold.lease.seq); + ASSERT_GT(after_fold.snap_generation, 0u); + + // Drop the only ref and advance the watermark floor so the now-orphaned blob is not spared in-flight. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + store->renewWatermarkOnce(); + + // Round 2 (DEPOSED): the round folds the -1 and writes its fold seal under its own attempt `a1`, then + // its single round-commit CAS is DENIED (lease lost mid-round). The round must throw and must NOT + // advance the adopted (snap_generation, snap_attempt). + backend->arm_interrupt = true; + EXPECT_ANY_THROW(runRegularRoundReclaiming(gc)); // ABORTED: round-commit CAS denied + backend->arm_interrupt = false; + + const auto after_deposed = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_EQ(after_deposed.snap_generation, after_fold.snap_generation) + << "the denied round-commit CAS must NOT advance the adopted generation"; + EXPECT_EQ(after_deposed.snap_attempt, after_fold.snap_attempt) + << "the denied round-commit CAS must NOT advance the adopted attempt"; + + // The deposed leader DID write its fold seal under its OWN attempt `a1` (= the lease.seq it renewed + // for round 2, which is strictly past the still-adopted attempt) at its fold generation `G_f` + // (= snap_generation + 1; fold mints the next generation, the round-commit CAS adopts it). That orphan + // is pure debris: it is under an attempt that gc/state never adopted, so no reader resolving + // (snap_generation, snap_attempt) can see it. On a PRE-FIX tree this seal would instead sit at the + // FINAL `gc/gen//fold_seal` key and wedge every future round's fold at the same G_f. + const uint64_t a1 = after_fold.lease.seq + 1; // round 2 renewed the lease => seq bumped once + const uint64_t g_f = after_fold.snap_generation + 1; // the generation the deposed fold minted + EXPECT_NE(a1, after_deposed.snap_attempt) << "the deposed attempt must differ from the adopted one"; + EXPECT_TRUE(backend->head(store->layout().foldSealKey(g_f, a1)).exists) + << "the deposed leader's fold seal is durable under its own (unadopted) attempt a1"; + EXPECT_FALSE(backend->head(store->layout().foldSealKey(g_f, after_deposed.snap_attempt)).exists) + << "no fold seal exists under the still-adopted attempt at the deposed fold generation (orphan is invisible)"; + + // An HONEST GC to a fixpoint (CAS now allowed). The KEY property: with attempt-scoping this SUCCEEDS — + // the next honest fold mints a FRESH attempt (a different lease.seq), never collides with the deposed + // seal under a1, and drains the unreachable blob. On a pre-fix (final-key) tree, the next fold would + // adopt-collide with the deposed final-key seal's divergent bytes and throw forever (GC wedged). + EXPECT_NO_THROW(runGcToFixpoint(store, gc)); + + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the dropped blob must be reclaimed (GC drained past the deposed attempt)"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0) << "no stranded positive in-degree"; + EXPECT_EQ(runFsck(*store, /*detail=*/false).unreachable, 0u) + << "INV-NO-LEAK: the deposed fold attempt did not wedge GC; the pool fully drained"; + + // GC advanced past the deposed attempt: the adopted (snap_generation, snap_attempt) moved on, and the + // adopted attempt is a fresh one (never the deposed a1). + const auto after_drain = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_GT(after_drain.snap_generation, after_fold.snap_generation) << "completion advanced the generation"; + EXPECT_NE(after_drain.snap_attempt, a1) << "the drained round never adopted the deposed attempt a1"; +} diff --git a/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp b/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp new file mode 100644 index 000000000000..e9041dd7b223 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp @@ -0,0 +1,570 @@ +#include + +#include +#include +#include +#include +#include "cas_test_helpers.h" + +/// THE BOUNDED FOLD WALK. +/// +/// Arithmetic ref intake reads the next record by exact key -- `cursor + 1` -- and stops when that read +/// comes back absent. That is exact and immune to a lying listing, and it is also, on its own, a walk +/// with no last record: a namespace whose writer keeps appending never produces the absent read, so a +/// round's duration stopped being `backlog / walker_rate` and became `backlog / (walker - writer)`. It +/// diverges the moment a writer keeps up. Measured on a hot pool: ZERO completed GC rounds in 42 +/// minutes, and with them nothing that paces on rounds -- fold seal, cursors, the sampled store-quality +/// detector, ref-object cleanup -- ever ran again. +/// +/// The bound is `_ckpt.committed_through`, snapshotted once per namespace before the walk and never +/// re-read within the round, so the work is finite and fixed before the round began however fast the +/// writer appends. It is the AUTHORITY ceiling too -- a record above it is durable but is not logical +/// history yet -- so ONE comparison both terminates the round and refuses to fold uncommitted work. +/// +/// IT DOES NOT BOUND WHAT THE ROUND READS, and that is not an oversight. The read at `cursor + 1` +/// produces the frontier proof, an unproven namespace suppresses all +/// destructive work, and suppression stops the ref-object cleanup that would have drained the listing -- +/// so a namespace that stops being read can never become provable again by any route. "Skip the quiet +/// namespace entirely" therefore is not a cheaper version of this design, it is a GC that permanently +/// reclaims nothing; the saving is one `GET` and that `GET` is the proof. +/// +/// So the properties these tests pin are: +/// * a round folds through its round-start committed frontier and no further, whatever lands meanwhile; +/// * a namespace whose tail did not move folds NOTHING and a matching CTE proves its carried cursor; +/// * a round where no tail moved is skipped outright by the existing defer machinery; +/// * a raw record ABOVE the committed frontier neither extends the fold nor suppresses destruction; +/// * a manifest edge fold costs one GET and never a HEAD; +/// * a namespace that folded nothing still keeps its sealed coverage row. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +/// Composed over `CountingBackend` because these tests assert REQUEST COUNTS: "folds nothing and reads +/// once" is the claim, and only a counting backend can check it. +using CountingHintHoleBackend = DB::Cas::tests::HintHoleBackendOn; + +/// The `RefCoverage` the newest fold seal recorded for `ns`'s opaque catalog life. Scans downward from +/// the adopted generation for the most recent fold seal, mirroring `foldCursorOf`'s reasoning (a +/// completed round's `gc/state` points at the recheck generation, which writes a completion seal rather +/// than a fold seal). +std::optional coverageOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const uint64_t gen = currentGenerationOf(backend, layout); + const uint64_t attempt = currentAttemptOf(backend, layout); + const UInt128 life_id = catalogLifeIdForTest(backend, layout, ns); + for (uint64_t g = gen; ; --g) + { + if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + { + const CasFoldSeal seal = decodeFoldSeal(got->bytes); + const auto it = seal.ref_lives.find(life_id); + if (it == seal.ref_lives.end()) + return std::nullopt; + return it->second.coverage; + } + if (g == 0) + return std::nullopt; + } +} + +RefTxnId cursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto cov = coverageOf(backend, layout, ns); + return cov ? cov->last_folded_ref_id : RefTxnId{}; +} + +/// A phase metric, or 0 when the row does not carry it. Reading it this way rather than through +/// `std::map::at` is deliberate: against the unbounded walk these columns do not exist yet, and a +/// missing column should fail the assertion that names it, not abort the test with an exception. +UInt64 metric(const std::map & row, const String & name) +{ + const auto it = row.find(name); + return it == row.end() ? 0 : it->second; +} + +/// Every key the backend was asked to delete, so a failing zero-delete assertion names the site that +/// leaked instead of only reporting a count. +String deletedKeysMessage(const CountingBackend & backend) +{ + String out; + for (const String & key : backend.deletedKeys()) + out += "\n " + key; + return out.empty() ? String{" (none)"} : out; +} + +/// Every `_log/` GET this round issued against `ns` over ids `first..last` -- the "read once, fold +/// nothing" claim, made against the store rather than against a counter the fold keeps about itself. +uint64_t refLogGetsFor(const CountingBackend & backend, const Layout & layout, const RootNamespace & ns, + uint64_t first, uint64_t last, uint64_t epoch = 1) +{ + uint64_t total = 0; + for (uint64_t i = first; i <= last; ++i) + total += backend.getCount(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{epoch, i})); + return total; +} + +/// One round driven directly on `Gc`, capturing the `fold_ref_intake` phase row. Driving `Gc` rather +/// than `CasGcScheduler` is what lets a caller choose the universe policy, which the suppression test +/// needs. An EMPTY row means the round deferred and folded nothing at all. +std::map runRoundCapturingIntake(Gc & gc, UniversePolicy policy = UniversePolicy::kDefault) +{ + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = gc.runRegularRound({}, /*allow_steal*/true, policy); + gc.setPhaseSink({}); + EXPECT_TRUE(report.acquired_lease) << "the round must have run at all"; + return intake; +} + +/// A store whose writer keeps pace with the walker EXACTLY: every time the fold reads the newest record +/// by exact key, one more record lands above it. +/// +/// This is a mid-round appender expressed as a synchronous hook rather than as a thread, and the +/// determinism is the point. The property under test is "the round stops at the tail it froze, however +/// much arrives afterwards", and a thread can only make appends arrive at times the scheduler chooses -- +/// including, on an unlucky run, entirely after the walk has gone past. The hook reproduces the WORST +/// case (writer rate == walker rate, the rate at which the unbounded walk provably never terminates) on +/// every run, and `max_appends` bounds it so that the UNPATCHED walk still finishes and can be measured +/// rather than hanging the suite. +class ChasingWriterBackend : public CountingBackend +{ +public: + using CountingBackend::get; + + /// Start appending above `published_through` (writer epoch 1) whenever the tail is read, up to + /// `max_appends` further records. + void arm(const Layout * layout_, const RootNamespace & ns_, uint64_t published_through, uint64_t max_appends) + { + layout = layout_; + ns = ns_; + published = published_through; + limit = published_through + max_appends; + } + + /// Stop appending; the tail stands still from here on. + void disarm() { layout = nullptr; } + + uint64_t publishedThrough() const { return published; } + + std::optional get(const String & key, DB::Cas::Range range) override + { + auto result = CountingBackend::get(key, range); + if (!layout || appending || published >= limit) + return result; + if (key != layout->refLogKey(fixture::fixtureLife(ns), RefTxnId{1, published})) + return result; + + /// The walk just consumed the tail; the writer answers with the next record. Guarded against + /// re-entry because publishing issues backend calls of its own. + appending = true; + const uint64_t next = published + 1; + publishAt(*this, *layout, ns, RefTxnId{1, next}, "ref_" + std::to_string(next), next, DB::UInt128(next)); + published = next; + appending = false; + return result; + } + +private: + const Layout * layout = nullptr; + RootNamespace ns{}; + uint64_t published = 0; + uint64_t limit = 0; + bool appending = false; +}; + +/// Publish ids `first .. last` of `ns` in writer epoch 1, each pinning its own blob. +void publishRange(Backend & backend, const Layout & layout, const RootNamespace & ns, uint64_t first, uint64_t last) +{ + for (uint64_t i = first; i <= last; ++i) + publishAt(backend, layout, ns, RefTxnId{1, i}, "ref_" + std::to_string(i), i, DB::UInt128(i), + /*birth=*/i == 1); +} + +} + +/// ===================== (a) THE ROUND FOLDS THROUGH THE TAIL IT FROZE ===================== +/// +/// `planted` records are durable when the round starts. While it walks, the writer keeps pace exactly -- +/// every record the fold reads is answered with another one above it. The round must fold through the +/// tail it saw at round start and stop, leaving the stragglers to the round that lists them. +/// +/// Against the unbounded walk this fails on the cursor: it chases the appends and seals a cursor far +/// above the round-start tail. On a real pool nothing bounds that chase at all; the appender here stops +/// after `appended_mid_round` so the unpatched behaviour is measurable rather than a hang. +TEST(CASGCBoundedWalk, ARoundFoldsThroughItsRoundStartTailAndLeavesTheStragglers) +{ + const uint64_t planted = 6; + const uint64_t appended_mid_round = 40; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/hot@cas@"}; + + publishRange(*backend, layout, ns, 1, planted); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, planted}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + backend->arm(&layout, ns, planted, appended_mid_round); + + Gc gc(store, kGc); + const std::map intake = runRoundCapturingIntake(gc); + + ASSERT_GT(backend->publishedThrough(), planted) + << "the mid-round appender never fired, so this test proves nothing about a moving tail"; + + const auto cov = coverageOf(*backend, layout, ns); + ASSERT_TRUE(cov.has_value()) << "the round must seal a coverage row for the namespace it walked"; + EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, planted})) + << "the walk must fold through the round-start tail and no further -- it chased the writer"; + EXPECT_FALSE(cov->hold.has_value()) << "reaching the committed frontier is not a hold"; + EXPECT_NE(cov->classification, 4) << "reaching the committed frontier is not a clamp"; + EXPECT_EQ(metric(intake, "tails_advanced"), 1u); + EXPECT_EQ(metric(intake, "logs_applied"), planted) << "exactly the round-start backlog was folded"; + + /// The stragglers are not lost: the next round's listing has a higher tail and folds through it. + backend->disarm(); + const uint64_t total = backend->publishedThrough(); + ASSERT_GT(total, planted); + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, total}); + const std::map second = runRoundCapturingIntake(gc); + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, total})) + << "the records that landed mid-round are folded by the round that lists them"; + EXPECT_EQ(metric(second, "tails_advanced"), 1u); +} + +/// ===================== (b) A CTE-AUTHORIZED UNCHANGED NAMESPACE FOLDS NOTHING ===================== +/// +/// Two namespaces; only one gets a new record. The unchanged one must fold NOTHING. Its CTE already +/// authorizes the sealed cursor as a frontier, so an exact probe at `cursor + 1` would be redundant. +TEST(CASGCBoundedWalk, ACTEAuthorizedUnchangedNamespaceFoldsNothingWithoutAProbe) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace moved{"00/moved@cas@"}; + const RootNamespace still{"00/still@cas@"}; + + publishRange(*backend, layout, moved, 1, 2); + publishRange(*backend, layout, still, 1, 2); + writeRecoverableCkptForRawFixture(*backend, layout, moved, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + writeRecoverableCkptForRawFixture(*backend, layout, still, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + runRoundCapturingIntake(gc); + ASSERT_EQ(cursorOf(*backend, layout, still), (RefTxnId{1, 2})) << "the seeding round must fold both"; + ASSERT_EQ(cursorOf(*backend, layout, moved), (RefTxnId{1, 2})); + + /// Only `moved` advances. + publishAt(*backend, layout, moved, RefTxnId{1, 3}, "ref_3", 3, DB::UInt128(0x33)); + advanceRecoverableCkptForRawFixture(*backend, layout, moved, RefTxnId{1, 3}); + + backend->resetCounts(); + const std::map intake = runRoundCapturingIntake(gc); + + EXPECT_EQ(refLogGetsFor(*backend, layout, still, 1, 8), 0u) + << "the CTE already proves the unchanged namespace's sealed frontier"; + EXPECT_EQ(backend->getCount(layout.refLogKey(fixture::fixtureLife(still), RefTxnId{1, 3})), 0u) + << "a valid CTE needs no successor probe"; + EXPECT_EQ(metric(intake, "tails_unchanged"), 1u); + EXPECT_EQ(metric(intake, "tails_advanced"), 1u); + EXPECT_EQ(metric(intake, "logs_applied"), 1u) << "only the one new record was folded, pool-wide"; + EXPECT_EQ(cursorOf(*backend, layout, moved), (RefTxnId{1, 3})) << "the advanced namespace still folds"; +} + +/// ===================== (c) A ROUND WHERE NO TAIL MOVED IS SKIPPED OUTRIGHT ===================== +/// +/// The round-level skip is the EXISTING defer machinery, whose signal is already exactly this +/// comparison: `RefScanSummary::changed_shards` counts the namespaces whose greatest listed log sits +/// above their sealed cursor. So a round in which no tail moved folds nothing at all -- no intake phase, +/// no per-namespace walk, not even the probes -- and an append un-defers it. +TEST(CASGCBoundedWalk, ARoundWhereNoTailMovedIsDeferredAndAnAppendUnDefersIt) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); /// the DEFAULT defer window: this test is about the skip + const Layout & layout = store->layout(); + const RootNamespace a{"00/aa@cas@"}; + const RootNamespace b{"00/bb@cas@"}; + + publishRange(*backend, layout, a, 1, 2); + publishRange(*backend, layout, b, 1, 2); + writeRecoverableCkptForRawFixture(*backend, layout, a, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + writeRecoverableCkptForRawFixture(*backend, layout, b, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_FALSE(runRoundCapturingIntake(gc).empty()) << "the seeding round must actually fold"; + ASSERT_EQ(cursorOf(*backend, layout, a), (RefTxnId{1, 2})); + ASSERT_EQ(cursorOf(*backend, layout, b), (RefTxnId{1, 2})); + + backend->resetCounts(); + EXPECT_TRUE(runRoundCapturingIntake(gc).empty()) + << "no tail moved, so the round has no fold to run"; + EXPECT_EQ(refLogGetsFor(*backend, layout, a, 1, 8), 0u) << "a deferred round reads no ref log at all"; + EXPECT_EQ(refLogGetsFor(*backend, layout, b, 1, 8), 0u); + + /// One append, in `a` only. + publishAt(*backend, layout, a, RefTxnId{1, 3}, "ref_3", 3, DB::UInt128(0xaa3)); + advanceRecoverableCkptForRawFixture(*backend, layout, a, RefTxnId{1, 3}); + + backend->resetCounts(); + const std::map woken = runRoundCapturingIntake(gc); + ASSERT_FALSE(woken.empty()) << "an append must un-defer the round"; + EXPECT_EQ(metric(woken, "tails_advanced"), 1u) << "the appended-to namespace is walked again"; + EXPECT_EQ(metric(woken, "tails_unchanged"), 1u) << "and only that one"; + EXPECT_EQ(cursorOf(*backend, layout, a), (RefTxnId{1, 3})); + EXPECT_EQ(refLogGetsFor(*backend, layout, b, 1, 8), 0u) + << "the still-quiet namespace's CTE proves its carried frontier without a probe"; +} + +/// ============ (d) A RAW RECORD ABOVE THE COMMITTED FRONTIER PROVES AND SUPPRESSES NOTHING ============ +/// +/// This is the safety argument. It is stated under an explicit `Authoritative` policy so that the claim +/// is about the frontier terms and not about which policy the caller happened to pass. +/// +/// The CTE fixes the namespace's committed frontier at `{1,3}`. A raw record ABOVE that frontier was +/// never committed, so it is neither a reason to extend the fold nor a reason to suppress destruction: +/// LIST may observe it, but it cannot manufacture a later authoritative frontier. The dropped blob is +/// therefore reclaimable after the normal condemn/graduation pipeline. +TEST(CASGCBoundedWalk, ARawRecordBeyondTheCommittedFrontierCannotSuppressDestruction) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/hot@cas@"}; + const DB::UInt128 blob(0xd00d); + + /// Publish a blob and drop it, both within the round-start listing: its folded in-degree returns to + /// zero, so a round with a complete frontier would condemn it. + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "kept", 1, DB::UInt128(0x1), /*birth=*/true); + const ManifestRef doomed{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, blob); + writeManifestRaw(*backend, layout, ns, doomed, {blobEntryFor("data.bin", blob)}); + writeTxnAt(*backend, layout, ns, RefTxnId{1, 2}, publishCommittedOps("doomed", doomed)); + dropRefTransition(*backend, layout, ns, "doomed", doomed); + + /// One record lands mid-round, above the committed frontier. + backend->arm(&layout, ns, /*published_through*/ 3, /*max_appends*/ 1); + + /// From HERE the deletes are the ROUND's. Opening the pool runs a capability probe that writes and + /// deletes its own `_probe/` keys, and counting those against the round would make this assertion + /// fail on debris that has nothing to do with the destructive gate. + backend->resetCounts(); + + Gc gc(store, kGc); + const std::map intake = runRoundCapturingIntake(gc, UniversePolicy::Authoritative); + + ASSERT_EQ(backend->publishedThrough(), 4u) << "the mid-round appender never fired"; + ASSERT_EQ(metric(intake, "tails_advanced"), 1u) << "the namespace must have been walked"; + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})) << "and folded exactly its committed frontier"; + EXPECT_EQ(metric(intake, "frontier_proven"), 1u) + << "the CTE, not the raw record beyond it, fixes the namespace frontier"; + EXPECT_EQ(metric(intake, "frontier_namespaces"), 1u) << "it is still in the round's universe"; + EXPECT_EQ(backend->deleteTotal(), 1u) + << "the committed frontier permits the round's immediate manifest cleanup. Deleted:" + << deletedKeysMessage(*backend); + EXPECT_TRUE(backend->head(layout.blobKey(legacyMetaTestRef(blob))).exists); + + /// The raw F+1 record remains outside the CTE; it cannot defer the normal destructive pipeline. + backend->disarm(); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, blob, /*max_rounds*/ 8)) + << "the committed frontier must permit reclamation despite the raw record above it"; +} + +/// A store that hides a namespace's records from every LIST does not lose them: the namespace goes +/// QUIET, and a quiet namespace is exactly the shape the exact-key probe at `cursor + 1` exists for. It +/// has no listed tail at all, and no bound is taken from a listing anyway -- bounding a namespace by a +/// tail the liar refuses to admit to would hand it the omission it was hoping for. +TEST(CASGCBoundedWalk, AListHiddenTailIsCaughtAndFoldedByTheQuietProbePath) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/liar@cas@"}; + + publishRange(*backend, layout, ns, 1, 2); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + Gc gc(store, kGc); + runRoundCapturingIntake(gc); + ASSERT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})); + + /// A third record lands and the store stops listing the namespace at the same moment: its listed + /// tail is now nothing at all, while `{1, 3}` is durable and readable by exact key. + publishAt(*backend, layout, ns, RefTxnId{1, 3}, "ref_3", 3, DB::UInt128(0x1a3)); + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, 3}); + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(ns))); + + const std::map intake = runRoundCapturingIntake(gc); + EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})) + << "the exact-key probe sees what LIST omits, so the hidden record is folded, not lost"; + EXPECT_EQ(metric(intake, "unhinted_quiet_walked"), 1u) << "it is the quiet path that caught it"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(0x1a3)), 1) + << "the hidden record's owner edge must be folded, or its blob looks unreferenced"; +} + +/// ===================== (e) ONE ROUND TRIP PER MANIFEST EDGE ===================== +/// +/// The edge fold used to pay HEAD-then-GET: two serial round trips per manifest edge, on every folded +/// log, where the GET alone already answers "is it there". +TEST(CASGCBoundedWalk, ManifestEdgeFoldsPayAGetAndNeverAHead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const uint64_t records = 3; + publishRange(*backend, layout, ns, 1, records); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, records}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->resetCounts(); + Gc gc(store, kGc); + runRoundCapturingIntake(gc); + ASSERT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, records})) << "the round must have folded them"; + + uint64_t manifest_heads = 0; + uint64_t manifest_gets = 0; + for (uint64_t i = 1; i <= records; ++i) + { + const ManifestId id{ns, ManifestRef{.writer_epoch = 1, .build_sequence = i, .manifest_ordinal = 1}}; + manifest_heads += backend->headCount(layout.manifestKey(id)); + manifest_gets += backend->getCount(layout.manifestKey(id)); + } + EXPECT_EQ(manifest_heads, 0u) << "the fold must not HEAD a manifest body it is about to GET"; + EXPECT_GT(manifest_gets, 0u) << "the bodies were read, so the counters really are watching these keys"; +} + +/// An absent manifest body still takes the record-and-continue path -- the GET's own absence is the +/// signal the HEAD used to carry -- and still raises the fold barrier without ever HEADing the key. +TEST(CASGCBoundedWalk, AnAbsentManifestBodyStillHoldsWithoutAHead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishRange(*backend, layout, ns, 1, 3); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + const ManifestId gone{ns, ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}}; + deleteManifestBody(*backend, layout, gone); + + backend->resetCounts(); + Gc gc(store, kGc); + runRoundCapturingIntake(gc); + + const auto cov = coverageOf(*backend, layout, ns); + ASSERT_TRUE(cov.has_value()); + EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, 1})) + << "the cursor stays BELOW the log whose manifest body is missing"; + ASSERT_TRUE(cov->hold.has_value()) << "an absent committed manifest body raises the fold barrier"; + EXPECT_EQ(cov->hold->reason, HoldReason::ManifestBodyMissing); + EXPECT_EQ(cov->hold->offending_position, (RefTxnId{1, 2})); + EXPECT_EQ(cov->classification, 4); + EXPECT_EQ(backend->headCount(layout.manifestKey(gone)), 0u) + << "absence is decided by the GET, so the missing body costs no HEAD either"; +} + +/// ===================== (f) A NAMESPACE THAT FOLDED NOTHING KEEPS ITS COVERAGE ROW =================== +/// +/// The fold seal writes a row only for the namespaces the intake loop visits, so any future shortcut +/// that stops visiting a namespace with nothing to fold would DROP its cursor -- and a dropped cursor is +/// not a lost optimisation, it is a re-fold from `{0, 0}`: every owner edge counted a second time, every +/// blob's in-degree inflated, and the eventual correction mass-condemning live data. This pins the row +/// against exactly that. +TEST(CASGCBoundedWalk, ANamespaceThatFoldedNothingKeepsItsSealedCursor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace quiet{"00/quiet@cas@"}; + const RootNamespace moved{"00/moved@cas@"}; + + publishRange(*backend, layout, quiet, 1, 3); + publishRange(*backend, layout, moved, 1, 1); + writeRecoverableCkptForRawFixture(*backend, layout, quiet, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + writeRecoverableCkptForRawFixture(*backend, layout, moved, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + runRoundCapturingIntake(gc); + const auto before = coverageOf(*backend, layout, quiet); + ASSERT_TRUE(before.has_value()); + ASSERT_EQ(before->last_folded_ref_id, (RefTxnId{1, 3})); + ASSERT_FALSE(before->hold.has_value()); + + /// A second round in which `quiet` folds nothing and `moved` does, so the round really does write a + /// new seal that could have dropped the row. + publishAt(*backend, layout, moved, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(0x22)); + advanceRecoverableCkptForRawFixture(*backend, layout, moved, RefTxnId{1, 2}); + const std::map intake = runRoundCapturingIntake(gc); + ASSERT_EQ(metric(intake, "tails_unchanged"), 1u) << "the fixture must actually exercise the quiet case"; + + const auto after = coverageOf(*backend, layout, quiet); + ASSERT_TRUE(after.has_value()) + << "the coverage row was DROPPED -- the next round would re-fold this namespace from {0,0}"; + /// The CURSOR and the HOLD are what the next round trusts, and both ride unchanged. + /// `classification` legitimately moves from 2 ("this round folded records") to 1 ("unchanged"), + /// because that is what the round did — it is the one field that may differ, so it is the one field + /// asserted loosely. + EXPECT_EQ(after->last_folded_ref_id, before->last_folded_ref_id) + << "a namespace that folded nothing must keep the cursor it had"; + EXPECT_EQ(after->hold, before->hold); + EXPECT_NE(after->classification, 4) << "folding nothing is not a clamp"; + EXPECT_EQ(metric(intake, "frontier_namespaces"), 2u) + << "it stays in the round's universe, so its proof is still owed"; +} diff --git a/src/Disks/tests/gtest_cas_gc_fold.cpp b/src/Disks/tests/gtest_cas_gc_fold.cpp new file mode 100644 index 000000000000..c764aa0ac0b7 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_fold.cpp @@ -0,0 +1,624 @@ +#include + +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +namespace +{ +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +ManifestRef ref(const String &, uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} +} + +/// Committed new_manifest => +1 per blob entry (BlobInDegreeMatchesActiveManifests). +/// After a fold, gc/state records snap_attempt == the folding leader's lease.seq, and the fold seal +/// lives under (snap_generation, snap_attempt). +TEST(CASGCFold, FoldAdoptsAttemptEqualsLeaseSeq) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); + + const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_EQ(st.snap_attempt, st.lease.seq); + EXPECT_GT(st.snap_generation, 0u); + /// The one-pass round's fold seal is durable under (snap_generation, snap_attempt) — the adopted + /// attempt locates it (a seal under any other attempt would be unadopted debris). + EXPECT_TRUE(backend->head(store->layout().foldSealKey(st.snap_generation, st.snap_attempt)).exists); +} + +TEST(CASGCFold, CommittedAddEmitsPlusOnePerBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, + {blobEntryFor("a", DB::UInt128(1)), blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); + + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1); +} + +/// Owner removal => -1 per blob entry; in-degree returns to 0. +TEST(CASGCFold, RemovalEmitsMinusOne) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +/// Precommit with a PRESENT, valid body => +1. +TEST(CASGCFold, PrecommitBodyPresentEmitsPlusOne) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + addPrecommitTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); +} + +/// Precommit whose body is ABSENT => NO delta (control #4); the 404 must NOT throw. +TEST(CASGCFold, PrecommitMissingBodyEmitsNoDelta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + addPrecommitTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", std::nullopt, r); + Gc gc(store, kGc); + EXPECT_NO_THROW(gc.runRegularRound()); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +/// FOLD BARRIER (control #23): a LIVE precommit binding whose body is missing does NOT advance the +/// durable fold cursor past its activation event; when the body appears the cursor advances. +TEST(CASGCFold, FoldBarrierHaltsCursorAtLiveMissingBodyPrecommit) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const uint64_t v = addPrecommitTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", std::nullopt, r); + Gc gc(store, kGc); + EXPECT_NO_THROW(gc.runRegularRound()); + EXPECT_LT(foldCursorOf(*backend, store->layout(), ns, 0), v); // barrier: halted at the activation + + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_GE(foldCursorOf(*backend, store->layout(), ns, 0), v); // barrier lifted by activation +} + +/// Promote of an already-activated precommit is a PURE OWNER MOVE: NO delta, body not condemned. +TEST(CASGCFold, PromoteOfActivatedPrecommitEmitsNoDelta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + addPrecommitTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + promoteTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", r); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); // unchanged, still pinned + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); // not condemned +} + +/// Committed add naming a MISSING body (404) => clamp + anomaly, never a guessed +1, never a throw. +TEST(CASGCFold, CommittedMissingBodyClampsCursorAndRecordsAnomaly) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const uint64_t v = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); // no body + Gc gc(store, kGc); + RoundReport report; + EXPECT_NO_THROW(report = gc.runRegularRound()); + EXPECT_TRUE(report.hasAnomaly(ns, /*shard*/0)); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); + EXPECT_LT(foldCursorOf(*backend, store->layout(), ns, 0), v); +} + +/// A body whose self-ref disagrees (PRESENT but INVALID) => hard fail closed (controls #19/#20). +TEST(CASGCFold, RefMismatchFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + PartManifest bad; + bad.ref = ref("srv-a:1", 1, 0xBB); // != r + bad.root_namespace_id = ns; + bad.entries = {blobEntryFor("a", DB::UInt128(1))}; + bad.payload_digest = computePayloadDigest(bad); + backend->putIfAbsent(store->layout().manifestKey(ManifestId{ns, r}), encodePartManifest(bad)); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&]{ gc.runRegularRound(); }); +} + +/// Owner-removal whose OLD committed body is gone at removal-fold => clamp + anomaly, no partial -1. +TEST(CASGCFold, RemovalWithMissingOldBodyClampsAndRecordsAnomaly) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); // +1; blob 1 in-degree 1 + + const uint64_t removal_version = dropRefTransition(*backend, store->layout(), ns, "tbl", r); + deleteManifestBody(*backend, store->layout(), ManifestId{ns, r}); // body gone before its decrement + + RoundReport report; + EXPECT_NO_THROW(report = gc.runRegularRound()); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); // unchanged: no silent -1 + EXPECT_TRUE(report.hasAnomaly(ns, /*shard*/0)); + EXPECT_LT(foldCursorOf(*backend, store->layout(), ns, 0), removal_version); +} + +/// (The two `CASGCFold.IncarnationMismatchRestartsFoldAtZero*` tests were removed with the snapshot+log +/// ref model: they injected a stale per-shard fold cursor beyond the live mutable shard's version and +/// asserted the fold RESET the cursor to 0 on an incarnation mismatch. There is no mutable per-shard +/// cursor to stale-reset anymore -- the durable cursor is a strictly-increasing `RefTxnId`, and a +/// recreated namespace uses a GREATER `writer_epoch`, so the ABA hazard is impossible by construction. +/// The ref-model equivalent -- `remove_namespace` then a later `namespace_birth` with a greater id folds +/// normally -- is covered by `gtest_cas_gc_shard_incarnation.cpp` and `gtest_cas_ref_gc.cpp`.) + +/// T0 (2026-07-02 snapshot-streaming): an idle round — no journal changes, no retired entries — touches +/// ZERO run objects. After one populated round, reset the counters and run a no-op round; the fold must +/// carry the parent generation's `RunRef` verbatim into the new fold_seal (same key, same checksum, same +/// generation) and NOT read or write any `.../blob_target/...` object. +TEST(CASGCFold, EmptyDeltaShardCarriesParentRunRef) +{ + auto backend = std::make_shared(); + /// gc_fold_max_defer_rounds=0 forces fold-every-round: this test exercises the pure-ref-carry FOLD + /// path on an idle round; without it the round would DEFER (re-adopt the sealed generation) and never + /// mint the carried generation this test inspects. + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); // round 1: folds the +1, seals the gen-1 blob_target run + + const auto st1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto parent_seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + ASSERT_EQ(parent_seal.blob_target_runs.size(), 1u); + const RunRef parent_ref = parent_seal.blob_target_runs.front(); + + backend->resetCounts(); + gc.runRegularRound(); // round 2: no changes => pure ref-carry, zero run I/O + + EXPECT_EQ(backend->ioCountForKeysContaining("/blob_target/"), 0u) + << "idle round must not GET/getStream/PUT any blob_target run object"; + + const auto st2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_GT(st2.snap_generation, st1.snap_generation); + const auto new_seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); + ASSERT_EQ(new_seal.blob_target_runs.size(), 1u); + const RunRef carried = new_seal.blob_target_runs.front(); + EXPECT_EQ(carried.key, parent_ref.key) << "carried ref points at the PARENT generation's run key"; + EXPECT_EQ(carried.checksum, parent_ref.checksum); + EXPECT_EQ(carried.shard, 0u); + EXPECT_EQ(carried.generation, st1.snap_generation) + << "the carried ref names the generation whose key namespace physically holds the object"; +} + +/// The round AFTER a ref-carry, with a real delta, folds THROUGH the carried ref: the new generation's +/// run is produced from the OLD-generation run (resolved via the carried ref, not by key construction) +/// merged with the delta, and the resulting in-degree is correct. +TEST(CASGCFold, FoldResolvesThroughCarriedRef) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref("srv-a:1", 1, 0xAA); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + gc.runRegularRound(); // gen 1: blob 1 in-degree 1 + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + gc.runRegularRound(); // gen 2: no delta => carries the gen-1 ref + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) + << "in-degree resolves through the carried parent ref"; + + // A real delta on the NEXT round must fold through the carried ref and drop blob 1 to zero. + const ManifestRef r2 = ref("srv-a:2", 2, 0xBB); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", r1, r2); + + gc.runRegularRound(); // gen 3: -1 on blob 1 (old owner dropped), +1 on blob 2 + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0) + << "fold through the carried ref applied the -1 correctly"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1); +} + +/// previewDeletes resolves runs through the current seal's refs, not by key construction. After a +/// pure ref-carry round the current seal's `blob_target_runs` point at an OLDER generation's key; the +/// preview must open that physical object via the ref and report the correct in-degree — here blob 1 is +/// still referenced, so its carried-ref-resolved in-degree is 1 and it is NOT surfaced as a candidate. +/// (A carried ref that the preview failed to resolve would mis-open the run and either throw or spuriously +/// surface the still-referenced blob.) +TEST(CASGCFold, PreviewResolvesCarriedRef) +{ + auto backend = std::make_shared(); + /// gc_fold_max_defer_rounds=0 forces the idle second round to FOLD (pure ref-carry) rather than + /// DEFER, so the current seal's `blob_target_runs` point at the parent generation's key (the carried + /// ref this test resolves through). + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); // gen 1: blob referenced, in-degree 1 + const auto st1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + + gc.runRegularRound(); // gen 2: no delta, no retired => pure ref-carry (ref points back at gen 1) + const auto st2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + ASSERT_GT(st2.snap_generation, st1.snap_generation); + const auto seal2 = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); + ASSERT_EQ(seal2.blob_target_runs.size(), 1u); + ASSERT_EQ(seal2.blob_target_runs.front().generation, st1.snap_generation) + << "the current seal's ref physically lives at the parent generation (carried, not reconstructed)"; + + // The preview resolves the carried ref (a gen-1 physical key) and computes in-degree 1 => blob 1 is + // not a delete candidate. Resolution-by-ref is the property under test. + const auto preview = gc.previewDeletes(); + for (const auto & e : preview) + EXPECT_NE(e.ref, (DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})) << "still-referenced blob must not be surfaced (carried ref resolved to in-degree 1)"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), blob), 1) + << "in-degree through the carried parent ref is 1"; +} + +/// Per-consumer whole-file seal-checksum RED tests (codecs-v3 phase 5, Task 6) at the seal-driven +/// consumers. Setup: fold one referenced blob into a sealed generation, then corrupt the persisted +/// seal's blob_target_runs[0].checksum (the stored run bytes stay valid), so the abort comes from the +/// seal-checksum verify, not a row invariant. +namespace +{ +String corruptSealedRunChecksum(InMemoryBackend & backend, const Layout & layout, const GcState & st) +{ + const String sk = layout.foldSealKey(st.snap_generation, st.snap_attempt); + const auto existing = backend.get(sk); + auto seal = decodeFoldSeal(existing->bytes); + if (seal.blob_target_runs.empty()) + return {}; + const String run_key = seal.blob_target_runs.front().key; + seal.blob_target_runs.front().checksum = seal.blob_target_runs.front().checksum + 1; + backend.putOverwrite(sk, encodeFoldSeal(seal), existing->token); + return run_key; +} +} + +TEST(CASGCFold, PreviewDeletesSealChecksumMismatchFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); // seals gen-1 with one blob_target run + const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + ASSERT_FALSE(corruptSealedRunChecksum(*backend, store->layout(), st).empty()); + + // A deletion preview must never be derived from an unverified run: fail closed. + Gc gc2(store, kGc); // fresh read of the corrupted seal + EXPECT_THROW(gc2.previewDeletes(), DB::Exception); +} + +TEST(CASGCFold, FsckSealChecksumMismatchCataloguedAndAuditCompletes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + replaceRecoverableCkptForRawFixture( + *backend, store->layout(), ns, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + Gc gc(store, kGc); + gc.runRegularRound(); + const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + + /// A present-but-unreferenced blob (written AFTER the round so GC never touches it) is what makes + /// fsck enter its GC-pipeline classification path (guarded by a non-empty unreferenced set), which + /// is where it streams + seal-checksum-verifies the snapshot runs. + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); + + const String bad_run_key = corruptSealedRunChecksum(*backend, store->layout(), st); + ASSERT_FALSE(bad_run_key.empty()); + + // fsck is a read-only auditor: it must CATALOGUE the corrupt run and COMPLETE, not abort the scan. + FsckReport report; + EXPECT_NO_THROW(report = runFsck(*store, /*detail*/ true)); + EXPECT_GE(report.corrupted_runs, 1u); + bool catalogued = false; + for (const auto & o : report.objects) + if (o.cls == FsckClass::CorruptedRun && o.key == bad_run_key) + catalogued = true; + EXPECT_TRUE(catalogued) << "the corrupt run must be catalogued with its key"; +} + +/// A mid-log clamp must be RECOVERABLE (spec §Step 3 transaction atomicity). A single log carrying two +/// ops -- [drop committed A (a `-1` whose body is present at removal-fold), add precommit B (whose body is +/// transiently absent)] -- clamps on B. The `-1` on A must NOT be merged into the round's owner-removed +/// cleanup, because the post-CAS body delete would then reclaim A's body while A's edge stays unfolded +/// behind the clamp; the next re-fold of that same log would then find A's body missing and clamp forever +/// (a permanent pool-wide destructive freeze). With per-log staging, A's body survives the clamp round and +/// the log folds cleanly once B's body reappears. +TEST(CASGCFold, MidLogClampPreservesEarlierRemovalBodyAndRecovers) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef a = ref("srv-a:1", 1, 0xAA); + const ManifestRef b = ref("srv-a:2", 2, 0xBB); + + /// Round 0: commit A (references blob 1). A's body is present and folds a +1. + writeManifestRaw(*backend, store->layout(), ns, a, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "r1", std::nullopt, a); + Gc gc(store, kGc); + gc.runRegularRound(); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// ONE log with two ops: drop committed A (`-1`, body present), then add precommit B (`+1`, body + /// staged then removed => a transient 404 clamps the log after A's `-1` already folded). + writeManifestRaw(*backend, store->layout(), ns, b, {blobEntryFor("b", DB::UInt128(2))}); + deleteManifestBody(*backend, store->layout(), ManifestId{ns, b}); // B's body absent => clamp + const uint64_t log_seq = appendRefLogSeed(*backend, store->layout(), ns, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "r1", a}, std::nullopt), + ownerTransitionOp(std::nullopt, RefOwnerBinding{RefOwnerKind::Precommit, "r2", b})}); + advanceRecoverableCkptForRawFixture(*backend, store->layout(), ns, RefTxnId{1, log_seq}); + + const RoundReport clamp_report = gc.runRegularRound(); + EXPECT_TRUE(clamp_report.hasAnomaly(ns, /*shard*/0)) << "the missing B body must clamp this log"; + EXPECT_LT(foldCursorOf(*backend, store->layout(), ns, 0), log_seq) << "the clamp halts the cursor below the log"; + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, a})).exists) + << "A's body must survive the clamp round: its `-1` was staged, not merged, so no post-CAS delete " + "reclaimed it -- otherwise the re-fold would clamp on A's missing body forever"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) << "A's `-1` was not adopted (clamp)"; + + /// The transient 404 heals: B's body reappears. The next round re-folds the SAME log cleanly. + writeManifestRaw(*backend, store->layout(), ns, b, {blobEntryFor("b", DB::UInt128(2))}); + const RoundReport clean_report = gc.runRegularRound(); + EXPECT_FALSE(clean_report.hasAnomaly(ns, /*shard*/0)) << "with both bodies present the log folds; no clamp"; + EXPECT_GE(foldCursorOf(*backend, store->layout(), ns, 0), log_seq) << "the cursor advanced past the recovered log"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0) << "A's `-1` applied"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1) << "B's `+1` applied"; +} + +/// A `+1` precommit whose body is PERMANENTLY absent and whose build is below the durable watermark floor +/// (provably dead -- the exact fact the orphan sweep uses to reclaim the body) must be SKIPPED, not held on +/// the fold barrier forever. Without a terminal rule this table clamps every round with no resolution (a +/// late-predecessor precommit whose body was already reclaimed). The watermark is seeded so the precommit's +/// build is dead; the fold must advance the cursor past the log and record no clamp anomaly. +TEST(CASGCFold, DeadPrecommitWithMissingBodyIsSkippedNotClampedForever) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + /// The namespace's server-root prefix is "srv"; seed its watermark floor so build_sequence 5 is retired. + const RootNamespace ns{"srv/tbl"}; + setWatermarkMinActive(*backend, store->layout(), "srv", /*writer_epoch*/1, /*min_active*/10); + + /// A precommit naming a build (writer_epoch 1, build_sequence 5) whose body is never written. + const ManifestRef dead = ManifestRef{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; + const uint64_t log_seq = + addPrecommitTransition(*backend, store->layout(), ns, DB::UInt128(7), "r1", std::nullopt, dead); + + Gc gc(store, kGc); + const RoundReport report = gc.runRegularRound(); + EXPECT_FALSE(report.hasAnomaly(ns, /*shard*/0)) + << "a provably-dead precommit's missing body is skipped, not clamped"; + EXPECT_GE(foldCursorOf(*backend, store->layout(), ns, 0), log_seq) + << "the fold advanced past the log instead of holding the barrier forever"; + + /// A second identical round stays clean (terminal resolution, not a recurring clamp). + const RoundReport report2 = gc.runRegularRound(); + EXPECT_FALSE(report2.hasAnomaly(ns, /*shard*/0)) << "the resolution is terminal: no recurring clamp"; +} + +/// A10: a single clamp anomaly must suppress ALL destructive actions in the round — the merge-side +/// deletes AND the post-CAS ref/namespace cleanup — from ONE decision, not two independent recomputes +/// of !report.anomalies.empty() that a future edit could desync (over-delete class). This pins that a +/// clamped round reclaims nothing. +TEST(CASGCFold, SingleAnomalySuppressesEveryDestructiveActionInTheRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef a = ref("srv-a:1", 1, 0xAA); + const ManifestRef b = ref("srv-a:2", 2, 0xBB); + + /// Round 0: commit A (references blob 1); its body folds a +1. + writeManifestRaw(*backend, store->layout(), ns, a, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "r1", std::nullopt, a); + Gc gc(store, kGc); + gc.runRegularRound(); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// One log: drop committed A (`-1`, body present) then add precommit B whose body is absent -> the + /// missing B body clamps the log AFTER A's `-1` folded. + writeManifestRaw(*backend, store->layout(), ns, b, {blobEntryFor("b", DB::UInt128(2))}); + deleteManifestBody(*backend, store->layout(), ManifestId{ns, b}); + const uint64_t log_seq = appendRefLogSeed(*backend, store->layout(), ns, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "r1", a}, std::nullopt), + ownerTransitionOp(std::nullopt, RefOwnerBinding{RefOwnerKind::Precommit, "r2", b})}); + advanceRecoverableCkptForRawFixture(*backend, store->layout(), ns, RefTxnId{1, log_seq}); + + const RoundReport rep = gc.runRegularRound(); + ASSERT_TRUE(rep.hasAnomaly(ns, /*shard*/0)) << "the missing B body must clamp this round"; + /// The clamp suppresses the WHOLE destructive pipeline this round: no deletes, no redeletes, and + /// A's `-1` stays unadopted (its body must survive, else the re-fold clamps on it forever). + EXPECT_EQ(rep.deleted, 0u); + EXPECT_EQ(rep.redeleted, 0u); + EXPECT_EQ(rep.graduated, 0u); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, a})).exists); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); +} + +/// A10 follow-up: the round-side destructive gates -- the perpetual dead-life janitor AND +/// `cleanupRefObjects`' covered ref-object deletion -- must ALSO honor the round's ONE +/// `suppress_destructive` decision, not just fold()'s merge-side reducers pinned above. A clamp anomaly in +/// one namespace must suppress destructive cleanup POOL-WIDE: dead-life physical debris must not be +/// swept, and an unrelated live +/// table's snapshot-covered ref-log must not be deleted, in the SAME clamped round. A clean round +/// afterward proves the setup really was cleanup-eligible, not vacuously untouched. +TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJanitorWork) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + Gc gc(store, kGc); + + /// Namespace 1: the clamp trigger (same construction as + /// SingleAnomalySuppressesEveryDestructiveActionInTheRound above). + const RootNamespace ns_clamp{"00/aa@cas@"}; + const ManifestRef a = ref("srv-a:1", 1, 0xAA); + const ManifestRef b = ref("srv-a:2", 2, 0xBB); + writeManifestRaw(*backend, layout, ns_clamp, a, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns_clamp, "r1", std::nullopt, a); + runRegularRoundReclaiming(gc); /// folds A cleanly; establishes the baseline before the clamp + + /// Namespace 2: a namespace mid-removal with physical manifest and verbatim-file debris. Generation + /// 7 has no lifecycle-specific cleanup pass: terminal folding records evidence, while these bytes + /// remain inert work for the perpetual janitor and orphan-manifest sweep. + const RootNamespace ns_removed{"00/cc@cas@"}; + RefOp remove_op; + remove_op.kind = RefOpKind::RemoveNamespace; + const uint64_t removal_log_seq = appendRefLogSeed(*backend, layout, ns_removed, {remove_op}); + writeRecoverableCkptForRawFixture( + *backend, layout, ns_removed, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, removal_log_seq}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + /// Keyed at the life the CATALOG names for this namespace (`appendRefLogSeed` admitted it above), + /// which is the physical life that owns the eventual janitor work. Spelling the sentinel here instead + /// would plant debris under the wrong life and make the retention assertion vacuous. + const String debris_key + = layout.namespaceFilesPrefix(CasRefCatalog::lifeIfCataloged(*backend, layout, ns_removed).value()) + + "leftover_verbatim_file"; + backend->putIfAbsent(debris_key, "debris"); + const ManifestRef removed_body = ref("srv-r:1", 1, 0xEE); + writeManifestRaw(*backend, layout, ns_removed, removed_body, {blobEntryFor("r", DB::UInt128(9))}); + const String debris_manifest_key = layout.manifestKey(ManifestId{ns_removed, removed_body}); + + /// Namespace 3: a live table with an exact checkpoint-named recovery triple -- exactly what a + /// clamp-free round's `cleanupRefObjects` may clean below that base. + const RootNamespace ns_covered{"00/dd@cas@"}; + const ManifestRef c1 = ref("srv-c:1", 1, 0xCC); + const ManifestRef c2 = ref("srv-c:2", 2, 0xDD); + writeManifestRaw(*backend, layout, ns_covered, c1, {blobEntryFor("c", DB::UInt128(3))}); + writeManifestRaw(*backend, layout, ns_covered, c2, {blobEntryFor("d", DB::UInt128(4))}); + const uint64_t cv1 = publishCommittedTransition(*backend, layout, ns_covered, "t1", std::nullopt, c1); + const uint64_t cv2 = publishCommittedTransition(*backend, layout, ns_covered, "t2", std::nullopt, c2); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns_covered.string(), RefTxnId{1, cv2}, + {committedRow("t1", c1), committedRow("t2", c2)})); + replaceRecoverableCkptForRawFixture(*backend, layout, ns_covered, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, cv2}, + .checkpoint_snapshot_id = RefTxnId{1, cv2}, + .last_epoch_seal = std::nullopt, + }); + const String covered_log_key = layout.refLogKey(fixture::fixtureLife(ns_covered), RefTxnId{1, cv1}); + ASSERT_TRUE(backend->head(covered_log_key).exists); + + /// Trigger the clamp in ns_clamp: drop committed A, add precommit B whose body is absent. + writeManifestRaw(*backend, layout, ns_clamp, b, {blobEntryFor("b", DB::UInt128(2))}); + deleteManifestBody(*backend, layout, ManifestId{ns_clamp, b}); + const uint64_t clamp_log_seq = appendRefLogSeed(*backend, layout, ns_clamp, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "r1", a}, std::nullopt), + ownerTransitionOp(std::nullopt, RefOwnerBinding{RefOwnerKind::Precommit, "r2", b})}); + advanceRecoverableCkptForRawFixture(*backend, layout, ns_clamp, RefTxnId{1, clamp_log_seq}); + + const RoundReport rep = runRegularRoundReclaiming(gc); + ASSERT_TRUE(rep.hasAnomaly(ns_clamp, /*shard*/0)) << "the missing B body must clamp this round"; + EXPECT_EQ(rep.deleted, 0u); + EXPECT_EQ(rep.redeleted, 0u); + EXPECT_EQ(rep.graduated, 0u); + + /// Removal folding never performs lifecycle-specific physical cleanup, with or without a clamp. + EXPECT_TRUE(backend->head(debris_manifest_key).exists) + << "removed manifest debris remains ordinary orphan-sweep work"; + EXPECT_TRUE(backend->head(debris_key).exists) + << "removed verbatim-file debris remains ordinary janitor work"; + + /// `cleanupRefObjects` must not have deleted anything anywhere this round. + EXPECT_TRUE(backend->head(covered_log_key).exists) + << "a clamp anywhere in the round must suppress ref-log cleanup pool-wide, even for an unrelated live table"; + + /// Heal the clamp and run a clean round. Ordinary ref-log cleanup resumes, while removal debris + /// remains physically untouched by the lifecycle path. + writeManifestRaw(*backend, layout, ns_clamp, b, {blobEntryFor("b", DB::UInt128(2))}); + const RoundReport clean_rep = runRegularRoundReclaiming(gc); + EXPECT_FALSE(clean_rep.hasAnomaly(ns_clamp, /*shard*/0)); + EXPECT_TRUE(backend->head(debris_manifest_key).exists) + << "a clamp-free fold still performs no lifecycle-specific manifest deletion"; + EXPECT_TRUE(backend->head(debris_key).exists) + << "a clamp-free fold still performs no lifecycle-specific verbatim-file deletion"; + EXPECT_FALSE(backend->head(covered_log_key).exists) << "a clamp-free round cleans the covered ref-log"; +} diff --git a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp new file mode 100644 index 000000000000..13dc97ab826c --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp @@ -0,0 +1,3488 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; + extern const int CORRUPTED_DATA; +} + +/// THE DESTRUCTIVE-ROUND FRONTIER PROOF (spec 2026-07-27 "ref chain complete cut" §5). +/// +/// Reachability is a property of the WHOLE POOL. A blob is unreferenced only if no namespace anywhere +/// owns an edge to it, so a round that deletes one is asserting something about every namespace at +/// once -- including the ones it never looked at. Task 7 made the per-namespace half of that assertion +/// cheap and exact: one `GET` at the cursor's arithmetic successor, absent means end-of-stream. Task 8 +/// made a namespace that could NOT be walked say so durably. What neither can supply is the SET those +/// proofs have to cover, and that is what this task is about. +/// +/// So the gate has three terms, and a round destroys only when all three are clear: +/// +/// suppress_destructive = any anomaly this round +/// OR any hold the seal carries +/// OR the frontier is incomplete +/// +/// The second term is STRUCTURAL. Every hold recorded today also records an anomaly, so the first term +/// happens to imply it -- but the invariant is the hold SET, not that coincidence, and the gate reads +/// the seal directly so that a future change to anomaly recording cannot quietly open it. +/// +/// The third term is the SET, and only the catalog supplies it. The scenario that makes that so is the +/// one these tests open with: a hidden acked `+1` in a namespace no listing mentions and no sealed cursor +/// names, while a visible `-1` elsewhere drives the shared blob's OBSERVABLE in-degree to zero. Every +/// proof the round holds comes back clean and the blob is still owned. It survives because the round's +/// universe is the catalog's `Live`/`Removing` set, so that namespace is a member the round owes a proof +/// for and cannot supply one -- neither the listing's silence nor the missing cursor can shrink the set. +/// +/// The tests here come in two shapes. Ones whose subject is the OPEN gate run on the production path, +/// with no policy argument at all. Ones whose subject is a SUPPRESSOR either pass `StageA_Suppressed` +/// explicitly or arrange the suppressing condition on the pool, and assert every delete family inert PER +/// FAMILY -- an aggregate zero can hide one family running while another did not. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace ProfileEvents +{ +extern const Event CASGCRefWalkPlansBuilt; +extern const Event CASGCUnmatchedAdoptedParentLives; +extern const Event CASGCNamespaceCleanupLeaks; +} + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +/// The lying store, shared from `cas_test_helpers.h`: every key is served by exact GET while the +/// selected ones are HIDDEN from every LIST. That is the only way to build the cross-namespace +/// scenario -- the hidden namespace's records stay durable and readable, so a round that KNOWS to +/// look for them finds them, while a round that only enumerates never learns they exist. Composed +/// over `CountingBackend` because these tests also assert request counts. +using CountingHintHoleBackend = DB::Cas::tests::HintHoleBackendOn; + +class DrainRaceBackend final : public CountingBackend +{ +public: + using CountingBackend::casPut; + using CountingBackend::get; + using CountingBackend::putIfAbsent; + + void blockNextCatalogCas(const String & key) + { + std::lock_guard lock(control_mutex); + catalog_key = key; + block_next_catalog_cas = true; + } + + void loseNextCatalogCasResponse(const String & key) + { + std::lock_guard lock(control_mutex); + catalog_key = key; + lose_next_catalog_cas_response = true; + } + + void conflictNextCatalogCas(const String & key) + { + std::lock_guard lock(control_mutex); + catalog_key = key; + conflict_next_catalog_cas = true; + } + + void waitForBlockedCatalogCas() + { + std::unique_lock lock(control_mutex); + control_cv.wait(lock, [&] { return catalog_cas_blocked; }); + } + + void releaseBlockedCatalogCas() + { + std::lock_guard lock(control_mutex); + release_catalog_cas = true; + control_cv.notify_all(); + } + + void clearJournal() + { + std::lock_guard lock(journal_mutex); + journal.clear(); + } + + std::vector journalSnapshot() const + { + std::lock_guard lock(journal_mutex); + return journal; + } + + std::optional get(const String & key, Range range) override + { + record("get " + key); + return CountingBackend::get(key, range); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + record("list " + prefix); + return CountingBackend::list(prefix, cursor, limit); + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + record("put_begin " + key); + const PutResult result = CountingBackend::putIfAbsent(key, bytes, meta); + record("put_end " + key); + return result; + } + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + record("cas_begin " + key); + bool lose_response = false; + bool force_conflict = false; + { + std::unique_lock lock(control_mutex); + if (key == catalog_key && block_next_catalog_cas) + { + block_next_catalog_cas = false; + catalog_cas_blocked = true; + control_cv.notify_all(); + control_cv.wait(lock, [&] { return release_catalog_cas; }); + } + if (key == catalog_key && lose_next_catalog_cas_response) + { + lose_next_catalog_cas_response = false; + lose_response = true; + } + if (key == catalog_key && conflict_next_catalog_cas) + { + conflict_next_catalog_cas = false; + force_conflict = true; + } + } + if (force_conflict) + { + record("cas_forced_conflict " + key); + return {.outcome = CasOutcome::Conflict, .token = {}}; + } + const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); + record("cas_end " + key); + if (lose_response && result.outcome == CasOutcome::Committed) + { + record("cas_response_lost " + key); + throw std::runtime_error("injected lost catalog CAS response"); + } + return result; + } + +private: + void record(String entry) const + { + std::lock_guard lock(journal_mutex); + journal.push_back(std::move(entry)); + } + + mutable std::mutex journal_mutex; + mutable std::vector journal; + std::mutex control_mutex; + std::condition_variable control_cv; + String catalog_key; + bool block_next_catalog_cas = false; + bool catalog_cas_blocked = false; + bool release_catalog_cas = false; + bool lose_next_catalog_cas_response = false; + bool conflict_next_catalog_cas = false; +}; + +class PostFoldUnreadableTerminalBackend final : public CountingBackend +{ +public: + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = CountingBackend::list(prefix, cursor, limit); + if (prefix.ends_with("/cas/ns/")) + for (ListedKey & listed : page.keys) + listed.token.reset(); + return page; + } + + HeadResult head(const String & key) override + { + if (key == unreadable_key) + throw std::runtime_error("injected post-fold terminal read failure for " + key); + return CountingBackend::head(key); + } + + void makeUnreadable(String key) + { + unreadable_key = std::move(key); + } + + bool existsIgnoringFault(const String & key) + { + return CountingBackend::head(key).exists; + } + +private: + String unreadable_key; +}; + +class ScopedCasGcLogCapture +{ +public: + ScopedCasGcLogCapture() + : logger(getLogger("CasGc")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("warning"); + } + + ~ScopedCasGcLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const + { + return stream.str(); + } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +struct CompletedRemovingFixture +{ + RootNamespace ns; + UInt128 life_id{}; + String checkpoint_key; + String checkpoint_bytes; +}; + +CompletedRemovingFixture seedCompletedRemoving( + DrainRaceBackend & backend, const PoolPtr & store, const UInt128 & lease_owner) +{ + const Layout & layout = store->layout(); + CompletedRemovingFixture fixture{ + .ns = RootNamespace{"00/drain-race@cas@"}, + .life_id = UInt128{177}, + .checkpoint_key = {}, + .checkpoint_bytes = {}}; + CasRefCatalog::casAdmitEntry(backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + .ns = fixture.ns, .state = NsState::Live, .incarnation = fixture.life_id}); + fixture.checkpoint_key = layout.refCkptKey( + NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); + fixture.checkpoint_bytes = encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + backend.putIfAbsent(fixture.checkpoint_key, fixture.checkpoint_bytes); + EXPECT_TRUE(store->namespaceFilesLifeIfReadable(fixture.ns)); + CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + + CasFoldSeal parent; + parent.generation = 1; + parent.ref_lives.emplace(fixture.life_id, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); + for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) + parent.condemned_summary.emplace(shard, CondemnedSummary{}); + backend.putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)); + + GcState state; + state.round = 1; + state.gc_shards = store->poolConfig().gc_shards; + state.snap_generation = 1; + state.snap_attempt = 1; + state.lease = GcLease{.owner = lease_owner, .seq = 1}; + backend.putIfAbsent(layout.gcStateKey(), encodeGcState(state)); + + return fixture; +} + +void seedCompletedRemovingBatch( + DrainRaceBackend & backend, const PoolPtr & store, const UInt128 & lease_owner, size_t count) +{ + const Layout & layout = store->layout(); + std::vector entries; + entries.reserve(count); + for (size_t i = 0; i < count; ++i) + { + CatalogEntry entry{ + .ns = RootNamespace{fmt::format("00/drain-batch-{}@cas@", i)}, + .state = NsState::Live, + .incarnation = UInt128{200 + i}}; + CasRefCatalog::casAdmitEntry(backend, layout, store->poolConfig().gc_shards, entry); + entries.push_back(std::move(entry)); + } + CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + for (CatalogEntry & entry : next.entries) + { + entry.state = NsState::Removing; + entry.removal_started_round = 1; + } + return next; + }); + + CasFoldSeal parent; + parent.generation = 1; + for (const CatalogEntry & entry : entries) + parent.ref_lives.emplace(entry.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); + for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) + parent.condemned_summary.emplace(shard, CondemnedSummary{}); + ASSERT_EQ(backend.putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)).outcome, + PutOutcome::Done); + + GcState state; + state.round = 1; + state.gc_shards = store->poolConfig().gc_shards; + state.snap_generation = 1; + state.snap_attempt = 1; + state.lease = GcLease{.owner = lease_owner, .seq = 1}; + ASSERT_EQ(backend.putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); +} + +enum class CompetingCatalogOutcome : uint8_t +{ + Absent, + Replacement, +}; + +class CASGCCompletedRemovalFenceRace : public testing::TestWithParam +{ +}; + +void transferGcLease(DrainRaceBackend & backend, const Layout & layout, const UInt128 & new_owner) +{ + const auto got = backend.get(layout.gcStateKey()); + ASSERT_TRUE(got); + GcState state = decodeGcState(got->bytes); + state.lease.owner = new_owner; + ++state.lease.seq; + ASSERT_EQ(backend.casPut(layout.gcStateKey(), encodeGcState(state), got->token).outcome, + CasOutcome::Committed); +} + +size_t findJournalAfter(const std::vector & journal, const String & entry, size_t after) +{ + const auto it = std::find(journal.begin() + static_cast(after), journal.end(), entry); + return it == journal.end() ? journal.size() : static_cast(it - journal.begin()); +} + +/// A pool whose GC frontier-probe budget is set explicitly. Everything else matches `openPoolForTest`. +PoolPtr openPoolWithProbeBudget(std::shared_ptr backend, uint64_t budget) +{ + return Pool::open(std::move(backend), + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_frontier_probe_budget = budget, .gc_fold_max_defer_rounds = 0}); +} + +/// Publish `ref_name` in `ns` pinning `blob`, allocating the next ref-log id. Writes the blob body and +/// the manifest body too, so the published edge is one GC can actually fold. +ManifestRef publish(Backend & backend, const Layout & layout, const RootNamespace & ns, + const String & ref_name, uint64_t build_sequence, const DB::UInt128 & blob) +{ + const ManifestRef mref{.writer_epoch = 1, .build_sequence = build_sequence, .manifest_ordinal = 1}; + writeBlobBody(backend, layout, blob); + writeManifestRaw(backend, layout, ns, mref, {blobEntryFor("data.bin", blob)}); + publishCommittedTransition(backend, layout, ns, ref_name, std::nullopt, mref); + return mref; +} + +/// The blob key for a raw hash, as the tests spell it. +String blobKeyOf(const Layout & layout, const DB::UInt128 & hash) +{ + return layout.blobKey(legacyMetaTestRef(hash)); +} + +/// The sealed fold cursor for `ns` as a full `RefTxnId`. Every seed here allocates `writer_epoch = 1`, +/// which is what `foldCursorOf` (returning the sequence alone) assumes too. +RefTxnId sealedCursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + return RefTxnId{1, foldCursorOf(backend, layout, ns, /*shard*/ 0)}; +} + +/// Drive `rounds` GC rounds under the given policy, renewing the store's watermark between them the way +/// the production scheduler does. +void drive(const PoolPtr & store, Gc & gc, int rounds, UniversePolicy policy) +{ + for (int i = 0; i < rounds; ++i) + { + gc.runRegularRound({}, /*allow_steal*/true, policy); + store->renewWatermarkOnce(); + } +} + +/// Every key the backend was asked to delete, rendered for a failing assertion's message. +String deletedKeysMessage(const CountingBackend & backend) +{ + String out; + for (const String & key : backend.deletedKeys()) + out += "\n " + key; + return out.empty() ? String{" (none)"} : out; +} + +/// The gate's own verdict for one round, READ OFF THE PHASE ROWS rather than recomputed in the test. A +/// test that re-derived `frontier_complete` from the tally would agree with a wrong formula just as +/// readily as with the right one. +struct GateVerdict +{ + bool saw_fold = false; + bool frontier_complete = false; + bool suppress_destructive = false; + uint64_t frontier_namespaces = 0; + uint64_t frontier_proven = 0; + uint64_t frontier_unprobed_budget = 0; + uint64_t catalog_entries = 0; + bool catalog_proved_empty = false; +}; + +GateVerdict runRoundCapturingGate(const PoolPtr & store, Gc & gc, UniversePolicy policy) +{ + GateVerdict verdict; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + const auto value = [&](const char * name) -> std::optional + { + const auto it = rec.metrics.find(name); + return it == rec.metrics.end() ? std::nullopt : std::optional{it->second}; + }; + if (rec.phase == "fold_reduce") + { + if (const auto complete = value("frontier_complete")) + { + verdict.saw_fold = true; + verdict.frontier_complete = *complete != 0; + } + if (const auto suppress = value("suppress_destructive")) + verdict.suppress_destructive = *suppress != 0; + } + else if (rec.phase == "fold_ref_intake") + { + if (const auto total = value("frontier_namespaces")) + verdict.frontier_namespaces = *total; + if (const auto proven = value("frontier_proven")) + verdict.frontier_proven = *proven; + if (const auto unprobed = value("frontier_unprobed_budget")) + verdict.frontier_unprobed_budget = *unprobed; + if (const auto entries = value("catalog_entries")) + verdict.catalog_entries = *entries; + if (const auto proved_empty = value("catalog_proved_empty")) + verdict.catalog_proved_empty = *proved_empty != 0; + } + }); + gc.runRegularRound({}, /*allow_steal*/true, policy); + gc.setPhaseSink({}); + store->renewWatermarkOnce(); + return verdict; +} + +/// Every delete family a round can reach, asserted PER FAMILY: an aggregate zero can hide one family +/// running while another did not. +void expectEveryDeleteFamilyInert(const CountingBackend & backend, const char * where) +{ + EXPECT_EQ(backend.deleteCountForKeysContaining("/blobs/"), 0u) << where << ": blob delete"; + EXPECT_EQ(backend.deleteCountForKeysContaining("/cas/manifests/"), 0u) + << where << ": manifest-body delete"; + EXPECT_EQ(backend.deleteCountForKeysContaining("/gc/gen/"), 0u) + << where << ": generation prune and hand-off reclaim"; + EXPECT_EQ(backend.deleteCountForKeysContaining("/cas/ns/stream/"), 0u) + << where << ": covered-log / superseded-snapshot cleanup"; + EXPECT_EQ(backend.deleteTotal(), 0u) + << where << ": a family not named above also ran. Deleted:" << deletedKeysMessage(backend); +} + +} + +/// ===================== A HIDDEN `+1` IN AN UNKNOWN NAMESPACE ===================== +/// +/// Two namespaces share one blob. `visible` publishes it and then drops it, so the round observes +/// `+1` then `-1` and reads the blob's in-degree as zero. `hidden` also owns it -- durably, acked, +/// readable by exact key -- but is absent from the round's LIST hint. Its own publish still carries a +/// real checkpoint, so the arithmetic walk finds and folds its `+1` by exact key regardless of what the +/// LIST omits. +/// +/// The three arms below are the whole argument: the exact-key probe finds `hidden`'s edge and saves the +/// blob on a complete frontier; the blob still drains once `hidden` also honestly folds its own removal; +/// and a namespace inside the universe with a sealed cursor has its hidden `+1` found by the exact-key +/// probe the same way. + +namespace +{ +/// Build the shared-blob scenario. `hidden` owns `blob` and is hidden from every LIST; `visible` +/// publishes and drops it. Returns the pool. +PoolPtr buildCrossNamespaceScenario(const std::shared_ptr & backend, + const RootNamespace & hidden, const RootNamespace & visible, + const DB::UInt128 & blob, bool fold_hidden_first) +{ + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + if (fold_hidden_first) + { + /// Give the hidden namespace a sealed cursor, WITHOUT folding the edge under test. It publishes + /// an unrelated blob and one round folds that; from then on the namespace is in the universe via + /// its cursor even after the hint stops naming it. + /// + /// The unrelated blob is what makes this arm mean anything: if the shared blob's `+1` had + /// already been folded by the seeding round, the blob would survive on the DURABLE in-degree and + /// the test would pass whether or not the round probes anything. Publishing it only AFTER the + /// seal puts it strictly above the cursor, so the probe is the one and only thing that can find + /// it. + publish(*backend, layout, hidden, "seed_ref", 7, DB::UInt128(0x5eed)); + Gc seed(store, kGc); + seed.runRegularRound(); + store->renewWatermarkOnce(); + } + + publish(*backend, layout, hidden, "kept_ref", 1, blob); + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(hidden))); + + const ManifestRef dropped = publish(*backend, layout, visible, "dropped_ref", 2, blob); + dropRefTransition(*backend, layout, visible, "dropped_ref", dropped); + return store; +} +} + +/// Rounds on the PRODUCTION path -- no policy argument -- because that is what the claim is about. +/// `hidden`'s own publish left it a real checkpoint, so the arithmetic walk's exact-key probe finds and +/// folds its `+1` no matter what the LIST hides: the blob survives on its own complete, proven frontier, +/// not on the round declining to touch anything. +TEST(CASGCFrontierGate, AHiddenEdgeIsFoundByTheExactKeyProbeAndSavesTheBlobOnACompleteFrontier) +{ + auto backend = std::make_shared(); + const RootNamespace hidden{"00/hidden@cas@"}; + const RootNamespace visible{"00/visible@cas@"}; + const DB::UInt128 blob(0x5ade); + + auto store = buildCrossNamespaceScenario(backend, hidden, visible, blob, /*fold_hidden_first=*/false); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + backend->resetCounts(); + GateVerdict verdict; + for (int i = 0; i < 5; ++i) + { + const GateVerdict round = runRoundCapturingGate(store, gc, UniversePolicy::kDefault); + if (round.saw_fold) + verdict = round; + store->renewWatermarkOnce(); + } + + ASSERT_TRUE(verdict.saw_fold) << "no round folded, so none published a gate verdict"; + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the blob a hidden namespace still owns must survive"; + EXPECT_TRUE(verdict.frontier_complete) + << "the exact-key probe reads at `cursor + 1` and a LIST hole cannot hide an exact key, so the " + "catalog-named hidden namespace IS provable — if this is false the blob above survived on " + "suppression instead of on its own in-degree, which proves nothing about the edge"; + EXPECT_FALSE(verdict.suppress_destructive); +} + +/// The arm above asserts "nothing was deleted", which does not on its own distinguish the gate correctly +/// refusing from the round simply never deleting anything at all. This +/// is the positive control: `hidden` is genuinely folded through its OWN drop of the same +/// blob (an honest exact-key read of a record the LIST still hides -- the arithmetic-intake mechanism +/// this whole file is about), so its frontier is REALLY proven, not merely declared so, and the blob is +/// REALLY unreferenced by both namespaces. The round drains it -- the zero-deletion arm above would pass +/// identically if the round were simply incapable of ever deleting anything. +TEST(CASGCFrontierGate, TheSameBlobDrainsOnceHiddenGenuinelyProvesItsOwnFrontier) +{ + auto backend = std::make_shared(); + const RootNamespace hidden{"00/hidden@cas@"}; + const RootNamespace visible{"00/visible@cas@"}; + const DB::UInt128 blob(0x5ade); + + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + Gc gc(store, kGc); + + /// `hidden`'s BIRTH is folded (and its cursor SEALED) with everything still listed -- a namespace + /// with no `_ckpt` (the raw-fixture admission this file's helper uses never publishes one) has no + /// genesis signal EXCEPT a sealed cursor or a visible LIST, so a real fold first is what makes an + /// arithmetic (cursor-relative) genesis available at all for what follows. + const ManifestRef kept = publish(*backend, layout, hidden, "kept_ref", 1, blob); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + store->renewWatermarkOnce(); + + /// NOW `hidden` drops its own reference (written while still fully listed, so the raw fixture's own + /// LIST -- finding the greatest existing log id, to derive the next one -- sees the truth), and + /// ONLY THEN does its whole prefix vanish from every subsequent LIST. With a sealed cursor already + /// in hand the walk's genesis is arithmetic (`cursor + 1`), so this drop is found and folded by + /// exact key alone -- the arithmetic-intake mechanism this whole file is about, exercised honestly + /// rather than declared past by fiat. + dropRefTransition(*backend, layout, hidden, "kept_ref", kept); + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(hidden))); + + const ManifestRef dropped = publish(*backend, layout, visible, "dropped_ref", 2, blob); + dropRefTransition(*backend, layout, visible, "dropped_ref", dropped); + + drive(store, gc, /*rounds*/ 5, UniversePolicy::Authoritative); + + EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + << "both namespaces genuinely proved their frontier and the blob is genuinely unreferenced -- " + "the round must still be able to reclaim it"; +} + +/// AND THE PER-NAMESPACE LOGIC IS WHAT SAVES IT. Identical to the arm above except that the hidden +/// namespace was folded once first, so it carries a sealed cursor and is therefore IN the universe even +/// though the hint has gone silent about it. The round probes its expected-next by exact key, finds the +/// record the listing hid, folds the `+1`, and the blob is never condemned. +TEST(CASGCFrontierGate, AKnownNamespaceIsProbedByExactKeyAndItsHiddenEdgeSavesTheBlob) +{ + auto backend = std::make_shared(); + const RootNamespace hidden{"00/hidden@cas@"}; + const RootNamespace visible{"00/visible@cas@"}; + const DB::UInt128 blob(0x5ade); + + auto store = buildCrossNamespaceScenario(backend, hidden, visible, blob, /*fold_hidden_first=*/true); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + drive(store, gc, /*rounds*/ 5, UniversePolicy::Authoritative); + + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the cursor kept the namespace in the universe, so its frontier was probed and its edge folded"; +} + +/// ===================== THE GATE FORMULA, TERM BY TERM ===================== +/// +/// The healthy case first, because every suppressor arm below is only meaningful against it: a pool with +/// nothing hidden, nothing held, no anomaly, and every namespace walked to an honest end-of-stream OPENS +/// the gate and reclaims. The two booleans are read off the fold's own rows, so a formula that computed +/// them differently would fail here rather than agree with the test. +TEST(CASGCFrontierGate, AHealthyCatalogRoundOpensTheGateAndReclaims) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const DB::UInt128 blob(0xdead); + + const ManifestRef mref = publish(*backend, layout, ns, "ref_1", 1, blob); + dropRefTransition(*backend, layout, ns, "ref_1", mref); + + Gc gc(store, kGc); + backend->resetCounts(); + const GateVerdict verdict = runRoundCapturingGate(store, gc, UniversePolicy::kDefault); + + ASSERT_TRUE(verdict.saw_fold) << "the round did not fold, so it published no gate verdict"; + EXPECT_TRUE(verdict.frontier_complete) + << "every namespace in a healthy catalog universe reached a proven frontier"; + EXPECT_FALSE(verdict.suppress_destructive) + << "no anomaly, no hold, a complete frontier -- the gate has nothing left to refuse on"; + EXPECT_GT(verdict.frontier_namespaces, 0u) + << "a zero-namespace universe would satisfy the equality vacuously; this pool must not be one"; + EXPECT_EQ(verdict.frontier_proven, verdict.frontier_namespaces); + + /// And the gate being open is worth something: the condemned blob actually drains. + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, blob)) + << "an unsuppressed round must reclaim a blob no ref owns any more"; +} + +/// ===================== EVERY DESTRUCTIVE SITE, INDIVIDUALLY ===================== +/// +/// The inventory as an assertion. The pool below has real work waiting at every gated site: a +/// graduated blob to delete, an owner-removed manifest body to delete, aged generations to prune and +/// hand off, ref logs and snapshots covered by a durable snapshot, and a removed namespace with a +/// Pending cleanup item. A suppressed round issues ZERO deletes against ALL of them, and the per-site +/// assertions name which one leaked if any does. + +namespace +{ +/// A pool with destructive work pending at every site, plus a few completed rounds so generations have +/// aged past the retention floor. Returns the hash of a blob whose in-degree has dropped to zero. +DB::UInt128 buildPoolWithWorkAtEverySite(const std::shared_ptr & backend, + const PoolPtr & store, Gc & gc) +{ + const Layout & layout = store->layout(); + const RootNamespace live{"00/live@cas@"}; + const RootNamespace doomed{"00/doomed@cas@"}; + const DB::UInt128 blob(0xfeed); + + /// A long-lived namespace that keeps publishing, so snapshots and covered logs accumulate and + /// generations keep advancing past the retention floor. + for (uint64_t i = 1; i <= 4; ++i) + { + publish(*backend, layout, live, "ref_" + std::to_string(i), i, DB::UInt128(0x1000 + i)); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + + /// The condemnable blob: published in `doomed`, then dropped. Its manifest body becomes + /// owner-removed cleanup work at the same time. + const ManifestRef mref = publish(*backend, layout, doomed, "doomed_ref", 9, blob); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + dropRefTransition(*backend, layout, doomed, "doomed_ref", mref); + return blob; +} +} + +/// (3a) THE NEGATIVE-POLICY SEAM, and the per-site inventory at the same time. A caller that supplies no +/// universe suppresses on that term ALONE: this pool has no anomaly, no hold, and a frontier every +/// per-namespace probe proves -- the control at the end of the test is what says so, since the identical +/// pool drains on the production path. +TEST(CASGCFrontierGate, EveryInventoriedDestructiveSiteIsInertUnderSuppression) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + const DB::UInt128 blob = buildPoolWithWorkAtEverySite(backend, store, gc); + + /// From here on the rounds supply NO universe: every site has work queued and every site must + /// decline it. + backend->resetCounts(); + const GateVerdict verdict = runRoundCapturingGate(store, gc, UniversePolicy::StageA_Suppressed); + for (int i = 0; i < 5; ++i) + runRoundCapturingGate(store, gc, UniversePolicy::StageA_Suppressed); + + ASSERT_TRUE(verdict.saw_fold) << "the round did not fold, so it published no gate verdict"; + EXPECT_FALSE(verdict.frontier_complete) + << "with no universe supplied the frontier can never be complete, whatever the probes proved"; + EXPECT_TRUE(verdict.suppress_destructive); + expectEveryDeleteFamilyInert(*backend, "no universe supplied"); + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + + /// The control: the identical pool DOES reclaim at those sites on the production path, so the zeros + /// above are the gate at work and not an empty work queue -- and it is also what makes the "on that + /// term alone" claim above true rather than assumed. + drive(store, gc, /*rounds*/ 4, UniversePolicy::kDefault); + EXPECT_GT(backend->deleteTotal(), 0u) + << "the work queue was real -- a round with a universe drains it"; + EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists); +} + +/// (1) ONE ANOMALY. A namespace whose `_ckpt` is present but undecodable records the "no usable +/// checkpoint" anomaly, and the round declines every site on that. It leaves the frontier incomplete too, +/// so what this arm pins is "an anomaly suppresses", not "only the anomaly does" -- which is why every +/// assertion below is about inertness and not about which term fired. +TEST(CASGCFrontierGate, AnUndecodableCheckpointAnomalySuppressesEveryDeleteFamily) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + const DB::UInt128 blob = buildPoolWithWorkAtEverySite(backend, store, gc); + + const RootNamespace damaged{"00/damaged@cas@"}; + publish(*backend, layout, damaged, "damaged_ref", 42, DB::UInt128(0xda43)); + /// Resolved through the catalog, not minted from the namespace name: the corruption has to land on + /// the very object the round's own life resolution will read, or the round folds normally and this + /// test measures nothing. + const std::optional damaged_life = + CasRefCatalog::lifeIfCataloged(*backend, layout, damaged); + ASSERT_TRUE(damaged_life.has_value()) << "the publish must have left a catalog entry to resolve"; + const std::optional damaged_ckpt = readCkpt(*backend, layout, *damaged_life); + ASSERT_TRUE(damaged_ckpt.has_value()) << "the publish must have left a `_ckpt` to damage"; + ASSERT_EQ(backend->casPut(layout.refCkptKey(*damaged_life), "not a checkpoint", + damaged_ckpt->token).outcome, CasOutcome::Committed); + + backend->resetCounts(); + std::vector anomaly_counts; + for (int i = 0; i < 6; ++i) + { + anomaly_counts.push_back(gc.runRegularRound().anomalies.size()); + store->renewWatermarkOnce(); + } + + EXPECT_GT(anomaly_counts.front(), 0u) + << "the undecodable `_ckpt` must be RECORDED, not silently absorbed -- a silent exit would make " + "this test pass for the wrong reason"; + expectEveryDeleteFamilyInert(*backend, "one anomaly"); + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); +} + +/// (2) ONE CARRIED HOLD. The gate's second term reads the SEAL, not this round's anomaly list, so the +/// round that matters here is a LATER one: the hold was detected earlier, rides forward because its +/// offending position is still unresolved, and must suppress on its own. +TEST(CASGCFrontierGate, ACarriedHoldSuppressesEveryDeleteFamily) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + const DB::UInt128 blob = buildPoolWithWorkAtEverySite(backend, store, gc); + + /// A committed gap: the checkpoint says `{1,2}` is committed while only `{1,1}` was ever written, so + /// the walk reads `{1,2}` absent BELOW its authority ceiling and holds there. Nothing repairs it, so + /// every later round re-detects the same position and carries the same hold. + const RootNamespace gapped{"00/gapped@cas@"}; + publish(*backend, layout, gapped, "gapped_ref", 44, DB::UInt128(0x6a9)); + replaceRecoverableCkptForRawFixture(*backend, layout, gapped, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + gc.runRegularRound(); + gc.setPhaseSink({}); + store->renewWatermarkOnce(); + ASSERT_FALSE(intake.empty()); + ASSERT_GT(intake.at("tables_held"), 0u) + << "the gap must be HELD, or the later rounds carry nothing and this test proves nothing"; + + backend->resetCounts(); + for (int i = 0; i < 5; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + expectEveryDeleteFamilyInert(*backend, "one carried hold"); + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); +} + +/// (3c) THE PROBE BUDGET. A namespace with a sealed cursor, no `_ckpt` and no listing left can be proven +/// only by a successor probe; a zero budget denies it one, so it counts toward the universe and not toward +/// the proofs, and the equality fails. +TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSuppressesEveryDeleteFamily) +{ + auto backend = std::make_shared(); + auto store = openPoolWithProbeBudget(backend, /*budget*/ 0); + const Layout & layout = store->layout(); + + Gc gc(store, kGc); + const DB::UInt128 blob = buildPoolWithWorkAtEverySite(backend, store, gc); + + /// The budget is spent only on a namespace the round knows about and can reach NO other way. Three + /// conditions, and all three are load-bearing: a SEALED cursor (an unhinted namespace with no cursor + /// is a no-genesis shape the budget never reaches), NO listing (a hinted target is walked for free), + /// and NO readable `_ckpt` (a recoverable checkpoint proves the frontier without spending a probe -- + /// which is why publishing and hiding alone leaves the namespace provable and measures nothing). + const RootNamespace quiet{"00/quiet@cas@"}; + publish(*backend, layout, quiet, "quiet_ref", 43, DB::UInt128(0x9a1e)); + runRoundCapturingGate(store, gc, UniversePolicy::StageA_Suppressed); + ASSERT_NE(sealedCursorOf(*backend, layout, quiet), (RefTxnId{})) + << "without a sealed cursor the namespace never becomes a budget-spending probe target"; + + const std::optional quiet_life = + CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); + ASSERT_TRUE(quiet_life.has_value()); + const std::optional quiet_ckpt = readCkpt(*backend, layout, *quiet_life); + ASSERT_TRUE(quiet_ckpt.has_value()) << "there must be a `_ckpt` to remove"; + ASSERT_EQ(backend->deleteExact(layout.refCkptKey(*quiet_life), quiet_ckpt->token).kind, + DeleteOutcome::Kind::Deleted); + backend->hidePrefix(layout.namespaceStreamPrefix(*quiet_life)); + + backend->resetCounts(); + const GateVerdict verdict = runRoundCapturingGate(store, gc, UniversePolicy::kDefault); + for (int i = 0; i < 5; ++i) + runRoundCapturingGate(store, gc, UniversePolicy::kDefault); + + ASSERT_TRUE(verdict.saw_fold); + EXPECT_GT(verdict.frontier_unprobed_budget, 0u) + << "this arm must suppress on the BUDGET term; a zero here means some other suppressor fired and " + "the test would pass without ever exhausting a budget"; + EXPECT_LT(verdict.frontier_proven, verdict.frontier_namespaces) + << "the unprobed namespace must count toward the universe and not toward the proofs"; + EXPECT_FALSE(verdict.frontier_complete); + EXPECT_TRUE(verdict.suppress_destructive); + expectEveryDeleteFamilyInert(*backend, "exhausted probe budget"); + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); +} + +/// `frontier_proven == frontier_namespaces` is `0 == 0` -- vacuously TRUE -- on an empty universe, which +/// is not by itself a proof of anything: a fresh pool, a damaged catalog, and a genuinely emptied pool +/// all produce the same zeros. The gate's non-vacuity term therefore has TWO ways to be satisfied: +/// `frontier_namespaces > 0` (an ordinary nonempty pool, everything proven), or the round's own hot-scan +/// catalog cut positively proving the universe empty (present, token-bearing, decoded, zero rows of +/// every lifecycle state). The next two tests are that positive/negative pair. +TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndDrainsRetiredWork) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const DB::UInt128 blob(0xbead); + + /// Built to make `frontier_namespaces` GENUINELY zero by every source that feeds it -- no catalog + /// entry (this pool never admitted a namespace), no sealed cursor, no ref-log hint -- while a + /// condemned blob with a real, present body and in-degree 0 sits queued exactly the way a real + /// round leaves one (`injectRetire`). + writeBlobBody(*backend, layout, blob); + const BlobRef blob_ref = legacyMetaTestRef(blob); + const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); + store->renewWatermarkOnce(); + + ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()) + << "the scenario needs a genuinely, provably empty catalog, or this test measures nothing"; + + Gc gc(store, kGc); + backend->resetCounts(); + + /// `injectRetire` seeds the condemned entry WITHOUT the durable per-hash `Condemned` meta a real + /// condemn round writes (`fold`'s own `scheduleCondemnMarkerWrite` side effect) -- so this fixture's + /// round cadence is longer than the textbook condemn->graduate->delete: the lease is UNCLAIMED + /// (`injectRetire` writes `gc/state` directly, never through a real acquire), so the first round + /// only arms `acquireOrRenewLease`'s two-tick steal-safety window; the graduation gate's own + /// `confirm_condemned_marker` then fails its first sighting, schedules the meta write, and CARRIES + /// the entry (not yet delete_pending) while `meta_pool_wait` lands it durably by that round's end; + /// only the round after that sees the durable meta and actually graduates; and the physical delete + /// is the round after THAT. MEASURED (phase-sink instrumentation, not a guess): round 1 arms + /// (`saw_fold == false`), round 2 is the first to fold with the gate open while the blob is still + /// present (the carry round), round 3 graduates, round 4 executes the delete -- four rounds exactly. + /// Bound the drive at that plus one (5): enough slack for the fixture's own cadence to be measured + /// without hand-counting rounds against this gate, but tight enough that a real regression in the + /// confirm/retry cadence still fails loudly instead of silently absorbing into a generous loop. + ASSERT_TRUE(backend->head(layout.blobKey(blob_ref)).exists) + << "the scenario starts with the condemned blob present, or the loop below measures nothing"; + + constexpr int kMaxRounds = 5; /// measured cadence (4) + 1; see the comment above + /// Observe the TWO-PHASE PIPELINE explicitly: an open, unsuppressed gate while the blob was still + /// present (the graduate side), and only a STRICTLY LATER round removing it (the delete side). + /// Asserting only the final state and the last verdict would pass just as readily if the blob + /// vanished by some other path entirely. + int round_gate_opened_while_present = -1; /// the graduate side: FIRST round that folded, unsuppressed, + /// with the blob still present + int round_blob_vanished = -1; /// the delete side + GateVerdict last; + int rounds_run = 0; + for (int i = 0; i < kMaxRounds && backend->head(layout.blobKey(blob_ref)).exists; ++i) + { + last = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); + ++rounds_run; + const bool still_present = backend->head(layout.blobKey(blob_ref)).exists; + if (round_gate_opened_while_present < 0 && last.saw_fold && !last.suppress_destructive && still_present) + round_gate_opened_while_present = rounds_run; + if (round_blob_vanished < 0 && !still_present) + round_blob_vanished = rounds_run; + } + + ASSERT_GT(rounds_run, 0) << "the loop must actually run, or every assertion below is vacuous"; + ASSERT_LE(rounds_run, kMaxRounds) + << "the drain took more than the measured cadence -- this is a real regression in the " + "confirm/retry pacing, not something to hide by bumping the bound; re-derive the cadence"; + ASSERT_TRUE(last.saw_fold); + EXPECT_EQ(last.frontier_namespaces, 0u); + EXPECT_EQ(last.frontier_proven, 0u); + EXPECT_TRUE(last.catalog_proved_empty) + << "the catalog cut itself must be the proof, not the bare 0==0 equality"; + EXPECT_TRUE(last.frontier_complete); + EXPECT_FALSE(last.suppress_destructive); + ASSERT_GT(round_gate_opened_while_present, 0) + << "the gate must have opened (folded, unsuppressed) at least one round BEFORE the blob was " + "gone -- the graduate side of the two-phase pipeline -- not just at the round that deleted it"; + ASSERT_GT(round_blob_vanished, round_gate_opened_while_present) + << "the delete must be a round STRICTLY LATER than the one that opened the gate, never the same " + "round -- a round that both graduates and deletes in one step would hide the two-phase split"; + EXPECT_FALSE(backend->head(layout.blobKey(blob_ref)).exists) + << "a proved-empty universe is a COMPLETE frontier, not a suppressed one -- the condemned blob " + "must drain through the ordinary two-phase pipeline instead of leaking forever"; +} + +/// The negative half of the pair. A `Creating` row is a birth in progress (spec §3: no publication can +/// exist under it, so `live_incarnation`/`walk_plan.lives()` exclude it -- see the R10 comment above the +/// intake loop), not an empty universe -- but it produces the SAME `frontier_namespaces == +/// frontier_proven == 0` a genuinely empty catalog does. Only `catalog_cut_proved_empty` tells them +/// apart, because it reads `entries` (every lifecycle state), not the frontier counters. +TEST(CASGCFrontierGate, AZeroWalkableFrontierWithACreatingCatalogRowIsNotProvedEmpty) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const DB::UInt128 blob(0xbead); + + writeBlobBody(*backend, layout, blob); + const BlobRef blob_ref = legacyMetaTestRef(blob); + const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); + store->renewWatermarkOnce(); + + const RootNamespace stalled{"00/stalled@cas@"}; + CatalogEntry entry; + entry.ns = stalled; + entry.state = NsState::Creating; + entry.incarnation = hexToU128("00000000000000000000000000000042"); + entry.creator = CreatorFence{ + .server_root_id = "test-stalled-creator", .writer_epoch = 1, .fence_generation = 1}; + CasRefCatalog::casAdmitEntry(*backend, layout, /*gc_shards*/ 1, entry); + + Gc gc(store, kGc); + backend->resetCounts(); + /// `injectRetire` leaves `gc/state`'s lease unclaimed (owner 0); the first round only arms the + /// steal-safety window (see `ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndDrainsRetiredWork`) + /// and folds nothing, so it is spent here rather than counted among the assertions below. + const GateVerdict warm_up = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); + EXPECT_FALSE(warm_up.saw_fold); + for (int i = 0; i < 6; ++i) + { + const GateVerdict v = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); + ASSERT_TRUE(v.saw_fold); + EXPECT_EQ(v.frontier_namespaces, 0u); + EXPECT_EQ(v.frontier_proven, 0u); + EXPECT_FALSE(v.catalog_proved_empty) + << "a Creating-only catalog is a birth in progress, not proof of an empty universe"; + EXPECT_FALSE(v.frontier_complete); + EXPECT_TRUE(v.suppress_destructive); + } + expectEveryDeleteFamilyInert(*backend, "Creating-only catalog"); + EXPECT_TRUE(backend->head(layout.blobKey(blob_ref)).exists); +} + +/// The bootstrap-only absent-as-empty representation (`initializeEmptyForNewPool`) must never leak into +/// the operational round: an absent mandatory catalog is corruption, never an empty authority set. +TEST(CASGCFrontierGate, AnAbsentCatalogNeverReadsAsAnEmptyUniverse) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + const Token catalog_token = backend->head(layout.refCatalogKey()).token; + ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog_token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc(store, kGc); + backend->resetCounts(); + bool saw_fold = false; + gc.setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "fold_reduce") saw_fold = true; }); + try + { + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::Authoritative); + FAIL() << "expected the missing mandatory catalog to throw before any fold gate verdict"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + gc.setPhaseSink({}); + EXPECT_FALSE(saw_fold) << "an absent catalog must abort before the round computes any gate verdict"; + EXPECT_EQ(backend->deleteTotal(), 0u) << "no destructive work may run on an unauthorized round"; +} + +/// A malformed, truncated, wrong-typed, count-mismatched, or future-versioned catalog must not decode +/// into a `Snapshot` at all -- these are the replacement guards for R11's original "damaged catalog" +/// concern, now that the empty case has a positive proof to keep separate from a broken one. +/// One table-driven test: each row installs a different broken body at the mandatory key and expects +/// decode failure before any destructive work. +/// +/// NOT covered here: a header version BELOW `RefCatalog`'s own birth generation. `checkCompatibility` +/// today only rejects a version ABOVE `G_BUILD`; a version below a type's birth floor decodes as if it +/// were legal, and `decodeRefCatalog` discards the parsed header entirely, so "decoded successfully" +/// does not yet imply "legal version for this type". That gap is not load-bearing for THIS proof: the +/// proof is token-present + a full structural decode (type, complete records, matching count trailer, +/// no trailing bytes) + zero entries, and a well-formed-but-out-of-protocol empty catalog is already an +/// accepted residual under the trusted-store model (the token proves byte identity, not history) -- +/// closing the version floor would only shrink that residual, not remove it. Tracked separately as +/// `[cas-format-version-floor]` in `BACKLOG.md`; deliberately out of scope for this gate. +TEST(CASGCFrontierGate, AMalformedCatalogNeverDecodesIntoAnEmptyProof) +{ + /// A one-entry catalog's canonical bytes, the base every mutation below starts from. + RefCatalog one_entry; + CatalogEntry entry; + entry.ns = RootNamespace{"00/malformed-base@cas@"}; + entry.state = NsState::Live; + entry.incarnation = hexToU128("00000000000000000000000000000099"); + one_entry.entries.push_back(entry); + const String base = encodeRefCatalog(one_entry); + const String empty_base = encodeRefCatalog(RefCatalog{}); + + const String type_needle = fmt::format("\"type\":\"{}\"", traitsFor(FormatId::RefCatalog).type); + ASSERT_NE(empty_base.find(type_needle), String::npos); + const String version_needle = fmt::format("\"v\":{}", currentCompatibilityVersion()); + ASSERT_NE(empty_base.find(version_needle), String::npos); + ASSERT_NE(base.find("\"n\":1"), String::npos); + + const auto replaceOnce = [](const String & haystack, const String & needle, const String & replacement) -> String + { + const auto pos = haystack.find(needle); + EXPECT_NE(pos, String::npos) << "expected to find '" << needle << "'"; + String out = haystack; + out.replace(pos, needle.size(), replacement); + return out; + }; + + struct Case { const char * name; String bytes; }; + const std::vector cases = { + {"wrong-type", replaceOnce(empty_base, type_needle, "\"type\":\"cas_ref_ckpt\"")}, + /// The one version case the CURRENT (unmodified) gate actually enforces: a version ABOVE + /// `G_BUILD` is refused by `checkCompatibility` before decode proceeds. + {"future-version", replaceOnce(empty_base, version_needle, "\"v\":999999")}, + {"trailer-count-mismatch", replaceOnce(base, "\"n\":1", "\"n\":2")}, + /// The trailer line entirely gone: decode's post-entry loop expects another line and hits EOF. + {"missing-trailer", base.substr(0, base.rfind("{\"n\":1}\n"))}, + /// The trailer present but its own line has no terminator: EOF strictly inside a line. + {"truncated-mid-line", base.substr(0, base.size() - 2)}, + }; + + for (const Case & c : cases) + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const Token bootstrap_token = backend->head(layout.refCatalogKey()).token; + ASSERT_EQ(backend->casPut(layout.refCatalogKey(), c.bytes, bootstrap_token).outcome, + CasOutcome::Committed) << c.name; + + Gc gc(store, kGc); + backend->resetCounts(); + bool saw_fold = false; + gc.setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "fold_reduce") saw_fold = true; }); + EXPECT_THROW( + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::Authoritative), DB::Exception) + << c.name; + gc.setPhaseSink({}); + EXPECT_FALSE(saw_fold) << c.name << ": a broken catalog must abort before any fold gate verdict"; + EXPECT_EQ(backend->deleteTotal(), 0u) << c.name << ": no destructive work may run on it"; + } +} + +/// `StageA_Suppressed` refuses outright regardless of what the catalog proves -- a proved-empty cut +/// satisfies the frontier term but is not the only term the gate reads. +TEST(CASGCFrontierGate, AProvedEmptyCatalogUnderStageASuppressedStaysSuppressed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const DB::UInt128 blob(0xbead); + + writeBlobBody(*backend, layout, blob); + const BlobRef blob_ref = legacyMetaTestRef(blob); + const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); + store->renewWatermarkOnce(); + + Gc gc(store, kGc); + backend->resetCounts(); + /// `injectRetire` leaves `gc/state`'s lease unclaimed (owner 0); the first round only arms the + /// steal-safety window (see `ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndDrainsRetiredWork`) + /// and folds nothing, independent of policy -- lease acquisition precedes the destructive gate. + const GateVerdict warm_up = runRoundCapturingGate(store, gc, UniversePolicy::StageA_Suppressed); + EXPECT_FALSE(warm_up.saw_fold); + for (int i = 0; i < 6; ++i) + { + const GateVerdict v = runRoundCapturingGate(store, gc, UniversePolicy::StageA_Suppressed); + ASSERT_TRUE(v.saw_fold); + EXPECT_TRUE(v.catalog_proved_empty) + << "the catalog cut is still genuinely empty -- the fact does not depend on policy"; + EXPECT_FALSE(v.frontier_complete) + << "StageA_Suppressed refuses outright no matter what the catalog cut proves"; + EXPECT_TRUE(v.suppress_destructive); + } + expectEveryDeleteFamilyInert(*backend, "StageA_Suppressed over a proved-empty catalog"); + EXPECT_TRUE(backend->head(layout.blobKey(blob_ref)).exists); +} + +/// THE BIRTH-AFTER-EMPTY-CUT BLOB RACE. The proved-empty exception's soundness rests on one hard fact: +/// under this pool's protocol every live or live-precommit edge requires an exact `Live` catalog row +/// (INV-3), so a catalog cut with zero rows proves no namespace ANYWHERE holds one -- AT THAT INSTANT. +/// A namespace born strictly after the cut is invisible to the round that took it; safety for blob +/// CONTENT then rests entirely on the condemned-marker/resurrection protocol (EDGE-BEFORE-OBSERVE: +/// `ContentAddressedTransaction.cpp` persists the precommit edge before observing/uploading the pool +/// blob), never on the frontier proof, which by construction cannot see a birth postdating its own cut. +/// This test pins that: a real writer, through the production `createNamespace` lifecycle +/// (`precommitAdd` on a namespace that has never existed), lands a precommit edge to an +/// ALREADY-CONDEMNED blob strictly after round R's catalog cut but strictly before round R executes the +/// pending delete that cut licensed. +TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlobInstead) +{ + ensureBlobUploadPoolForTest(); + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace doomed{"00/doomed@cas@"}; + + const String payload = "empty-cut-birth-race-payload"; + const DB::UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const String key = layout.blobKey(id); + String raw_body(store->poolMeta().blob_header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*backend, layout, hash, raw_body); + + const ManifestRef mref{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}; + writeManifestRaw(*backend, layout, doomed, mref, {blobEntryFor("data.bin", hash)}); + publishCommittedTransition(*backend, layout, doomed, "ref_1", std::nullopt, mref); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); /// folds the +1 edge + store->renewWatermarkOnce(); + dropRefTransition(*backend, layout, doomed, "ref_1", mref); + runRegularRoundReclaiming(gc); /// condemns: durable Condemned meta + store->renewWatermarkOnce(); + const Token condemned_token = backend->head(key).token; + const auto condemned_meta = loadMetaForTest(*backend, layout, hash); + ASSERT_TRUE(condemned_meta.has_value()); + ASSERT_EQ(condemned_meta->meta.state, MetaState::Condemned) + << "the delete round R is about to execute must be backed by durable Condemned evidence"; + runRegularRoundReclaiming(gc); /// graduates: publishes delete_pending + store->renewWatermarkOnce(); + + /// Remove `doomed` entirely -- the ONLY way the catalog can become genuinely, provably empty. A raw + /// `RemoveNamespace` op plus the Removing-state CAS mirrors + /// `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`'s own recipe exactly. + RefOp remove_op; + remove_op.kind = RefOpKind::RemoveNamespace; + const uint64_t remove_seq = appendRefLogSeed(*backend, layout, doomed, {remove_op}); + publishRecoverableCkptForSemanticWrapper(*backend, layout, doomed, RefTxnId{1, remove_seq}); + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) -> RefCatalog + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), + [&](const CatalogEntry & e) { return e.ns == doomed; }); + EXPECT_NE(it, next.entries.end()); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + + /// A SUPPRESSED round folds `doomed` through its removal terminal and records `cleanup_evidence` in + /// the seal, WITHOUT executing the delete_pending the graduate round above published -- suppressed + /// rounds carry pending deletes forward untouched. `StageA_Suppressed` here is the test's OWN + /// control over timing, not the scenario under test: it exists only to keep blob X's delete pending + /// until round R below, rather than letting it drain the ordinary way while `doomed` is still Live. + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::StageA_Suppressed); + store->renewWatermarkOnce(); + EXPECT_TRUE(backend->head(key).exists) << "the pending delete must still be carried, not yet run"; + + /// Round R: its pre-fold drain (`drainCompletedRemoving`) reads the round just above's + /// `cleanup_evidence` and drops `doomed`'s catalog row BEFORE this round's own hot-scan `GET` -- + /// so round R's catalog cut is the first one that is genuinely, provably empty. The hook fires the + /// instant that cut is taken and races a real namespace birth into the window before round R's own + /// pre-CAS delete phase runs. + bool hook_fired = false; + Token fresh_token{}; + gc.setPostHotScanCatalogReadHookForTest([&]() + { + hook_fired = true; + ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()) + << "the race must land inside the window where the cut itself is already empty"; + + const RootNamespace newborn{"00/newborn@cas@"}; + auto build = store->beginPartWrite( + PartWriteInfo{.intended_ref = newborn.string() + "/ref_1", .intended_namespace = newborn}); + const ManifestId new_id = build->stageManifest({blobEntryFor("data.bin", hash)}); + build->precommitAdd(newborn, "ref_1", new_id); /// mints `newborn` via real createNamespace + const PutBlobResult uploaded = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(uploaded.ref, id); + fresh_token = backend->head(key).token; + EXPECT_NE(fresh_token, condemned_token) + << "the writer must have observed Condemned and resurrected -- a fresh token, not an adopt " + "of the dying incarnation"; + }); + + const GateVerdict verdict = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); + ASSERT_TRUE(hook_fired) << "the race hook never fired -- this test proves nothing about the race"; + ASSERT_TRUE(verdict.saw_fold); + EXPECT_TRUE(verdict.catalog_proved_empty); + EXPECT_TRUE(verdict.frontier_complete); + EXPECT_FALSE(verdict.suppress_destructive); + + EXPECT_TRUE(backend->head(key).exists) + << "the resurrected incarnation must survive round R's delete"; + EXPECT_EQ(backend->head(key).token, fresh_token) << "and it is still the writer's incarnation"; + EXPECT_EQ(backend->deleteExact(key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch) + << "the condemned token can never remove the fresh object (INV_NO_LOSS)"; + + /// A later round's own fresh catalog cut names `newborn`, folds its `+1`, and the blob's frontier is + /// intact going forward. + const GateVerdict later = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); + ASSERT_TRUE(later.saw_fold); + EXPECT_EQ(later.frontier_namespaces, 1u); + EXPECT_EQ(later.frontier_proven, 1u); + EXPECT_TRUE(backend->head(key).exists) << "the newly folded owner keeps the blob alive"; +} + +/// The generation prune's cursor must not move on a suppressed round either. It is a monotone +/// high-water mark that the wholesale prune never revisits, so a cursor that advanced past a generation +/// this round declined to delete would strand that generation's whole prefix with no reclaimer left. +TEST(CASGCFrontierGate, ASuppressedRoundDoesNotAdvanceTheGenerationPruneCursor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + Gc gc(store, kGc); + for (uint64_t i = 1; i <= 6; ++i) + { + publish(*backend, layout, ns, "ref_" + std::to_string(i), i, DB::UInt128(0x2000 + i)); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + const uint64_t pruned_through_before = + decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through; + + for (uint64_t i = 7; i <= 10; ++i) + { + publish(*backend, layout, ns, "ref_" + std::to_string(i), i, DB::UInt128(0x2000 + i)); + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::StageA_Suppressed); + store->renewWatermarkOnce(); + } + + EXPECT_EQ(decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through, + pruned_through_before) + << "the retention cursor is a high-water mark; it may not pass a generation nothing deleted"; +} + +/// THE HAND-OFF RECLAIM, WHICH THE INVENTORY TEST ABOVE CANNOT REACH. This site only fires for a +/// generation the wholesale prune SKIPPED while a live ref still pinned it (so the retention cursor +/// moved past it and will never revisit it) and which a later round's ref then moves off. Building that +/// takes a deliberately idle shard and a retention cursor driven past it, which is why it gets its own +/// test rather than riding on the inventory pool. +/// +/// It is reachable under suppression precisely because FOLDING still happens on a suppressed round: the +/// ref moves off the old generation exactly as it would otherwise, and only the reclaim is withheld. +TEST(CASGCFrontierGate, TheHandOffReclaimIsInertUnderSuppression) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_snapshot_generations_to_keep = 1, .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r1 = publish(*backend, layout, ns, "tbl", 1, DB::UInt128(0xa1)); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const uint64_t old_gen = decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_generation; + const String old_prefix = layout.gcGenPrefix(old_gen); + ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()); + + /// Idle-carry the ref until the retention cursor is strictly PAST its generation. Until then an + /// ordinary prune could still reclaim it and the hand-off would not be the load-bearing path. + for (int i = 0; i < 6; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + ASSERT_GT(decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through, old_gen) + << "the generation must be behind the retention cursor before the hand-off is exercised"; + ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + << "and still retained, because a live ref pins it"; + + /// A real delta moves the shard's run off the old generation. This is the round the hand-off would + /// reclaim it on -- and it supplies no universe. + const ManifestRef r2{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, DB::UInt128(0xb2)); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("data.bin", DB::UInt128(0xb2))}); + publishCommittedTransition(*backend, layout, ns, "tbl", r1, r2); + + backend->resetCounts(); + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::StageA_Suppressed); + + EXPECT_EQ(backend->deleteCountForKeysContaining("/gc/gen/"), 0u) + << "a suppressed round hands nothing off. Deleted:" << deletedKeysMessage(*backend); + EXPECT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + << "the superseded generation's prefix survives a suppressed round intact"; + + /// AND THE OPPORTUNITY IS CONSUMED, NOT DEFERRED -- the one place in this task where the gate + /// costs something permanent, so it is asserted here rather than left to be discovered later. + /// + /// The hand-off is a one-shot DIFFERENCE: it compares the PARENT seal's runs against the new + /// seal's, and the suppressed round above already folded the delta, so the next round's parent + /// seal no longer mentions the old generation. Nothing revisits it -- the retention cursor is + /// already past it and the prune never goes back. The prefix is left to `fsck`, which is exactly + /// the outcome the site's own doc comment already records for a crash in the same window ("the + /// cursor already advanced, so a plain retry will NOT re-attempt it; fsck is the backstop"). + /// Bounded (one small run per shard per occurrence) and not a correctness problem. + /// + /// The hand-off itself is not going untested: `CASGCRetention.HandOffDeletesSupersededRef` drives + /// the same transition on an authoritative round and asserts the prefix IS reclaimed. + runRegularRoundReclaiming(gc); + EXPECT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + << "the hand-off is a one-shot difference: the suppressed round consumed it, so the prefix is " + "now fsck's problem rather than a later round's"; +} + +/// THE ORPHAN-MANIFEST SWEEP, which the inventory pool above also cannot reach: it only deletes bodies +/// that no ref names AND whose build is provably dead by the durable watermark floor, so it needs a +/// pool seeded with exactly that -- orphan bodies and a floor above them. +/// +/// It is gated with its CURSOR, not just its deletes. The cursor paces a cold-prefix enumeration and +/// nothing revisits a range it passed, so advancing it on a round that swept nothing would silently +/// skip that range forever. A suppressed round therefore declines the whole pass. +TEST(CASGCFrontierGate, TheOrphanManifestSweepAndItsCursorAreInertUnderSuppression) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "gc-runner", + .manifest_sweep_list_budget_keys = 1, .manifest_sweep_delete_budget_keys = 1, + .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const RootNamespace ns{"test/aa@cas@"}; + /// The control arm below needs a recoverable catalog life whose frontier is exactly the carried + /// cursor. An empty non-seal transaction is a valid genesis that recovers to an empty table while + /// leaving the manifest epoch below the cursor's epoch. + fixture::admitLive(*backend, layout, ns); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = RefTxnId{6, 1}, + .ops = {}, + .prev_epoch_seal = std::nullopt, + }); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 6, + .committed_through = RefTxnId{6, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + /// Two manifest bodies no ref ever named, under a build the durable floor has already passed. + const ManifestRef r1{.writer_epoch = 5, .build_sequence = 0xCA01, .manifest_ordinal = 1}; + const ManifestRef r2{.writer_epoch = 5, .build_sequence = 0xCA02, .manifest_ordinal = 1}; + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(0xa1))}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(0xb2))}); + setWatermarkMinActive(*backend, layout, "test", r1.writer_epoch, /*min_active*/ 0xCA03); + /// The §6 deletion premise is a second precondition on the CONTROL arm below: a manifest of an + /// epoch-`E` build is deletable only once the namespace's sealed fold cursor sits in an epoch + /// strictly above `E`. Sealing that cursor here is what keeps this test about the GATE — without it + /// the control arm would stop deleting for the premise's reason, and a removed gate would no longer + /// show up as a difference between the two arms. A real round rewrites this row with the same cursor + /// (the namespace is known, quiet and unheld, so the walk probes `cursor+1`, finds the frontier and + /// carries the cursor), so the seeded fact survives every round below. + seedFoldCursorForTest(*backend, layout, ns, RefTxnId{r1.writer_epoch + 1, 1}); + + Gc gc(store, kGc); + backend->resetCounts(); + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::StageA_Suppressed); + store->renewWatermarkOnce(); + } + + EXPECT_EQ(backend->deleteCountForKeysContaining("/cas/manifests/"), 0u) + << "a suppressed round sweeps nothing. Deleted:" << deletedKeysMessage(*backend); + EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, r1})).exists); + EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, r2})).exists); + EXPECT_TRUE(decodeGcState(backend->get(layout.gcStateKey())->bytes).manifest_sweep_cursor.empty()) + << "the sweep cursor must not advance over a range the round declined to sweep -- nothing " + "revisits it"; + + /// The control: the same orphans ARE swept once the universe is authoritative. + for (int i = 0; i < 4; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + EXPECT_FALSE(backend->head(layout.manifestKey(ManifestId{ns, r1})).exists); + EXPECT_FALSE(backend->head(layout.manifestKey(ManifestId{ns, r2})).exists); +} + + +/// ===================== QUIET NAMESPACES AND THE PROBE BUDGET ===================== + +/// THE TALLY ARITHMETIC, at a PARTIAL budget — the case neither 0 nor the default reaches. +/// +/// `frontier_namespaces` is the denominator an operator reads as "the round's universe", and the +/// integration test reads it too. A valid checkpoint at every quiet namespace's carried cursor is +/// authoritative independently of LIST and the probe budget: all three lives are proven without +/// successor probes, so the budget leaves no namespace unprobed. +TEST(CASGCFrontierGate, APartialProbeBudgetPublishesATallyThatMatchesTheSealedSet) +{ + auto backend = std::make_shared(); + auto store = openPoolWithProbeBudget(backend, /*budget*/ 1); + const Layout & layout = store->layout(); + const RootNamespace a{"00/quiet_a@cas@"}; + const RootNamespace b{"00/quiet_b@cas@"}; + const RootNamespace c{"00/quiet_c@cas@"}; + + for (const RootNamespace & ns : {a, b, c}) + publish(*backend, layout, ns, "ref_1", 1, DB::UInt128(0x300 + ns.string().size())); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + for (const RootNamespace & ns : {a, b, c}) + ASSERT_NE(sealedCursorOf(*backend, layout, ns), (RefTxnId{})) << ns.string(); + + /// All three go unhinted at once. Their valid checkpoint frontiers still prove their carried + /// cursors, so this does not consume the successor-probe budget. + for (const RootNamespace & ns : {a, b, c}) + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(ns))); + + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + ASSERT_FALSE(intake.empty()) << "the intake phase must have emitted its row"; + EXPECT_EQ(intake["unhinted_quiet_walked"], 3u) + << "a valid checkpoint frontier makes every quiet life eligible without a successor probe"; + EXPECT_EQ(intake["frontier_unprobed_budget"], 0u) + << "the CTE authority, not the probe budget, decides these quiet lives"; + EXPECT_EQ(intake["frontier_proven"], 3u) + << "each carried cursor equals its valid checkpoint frontier"; + EXPECT_EQ(intake["frontier_namespaces"], 3u) + << "the denominator is the complete authoritative set of sealed quiet lives"; + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]) + << "a valid CTE frontier remains authoritative even when LIST omits every namespace"; + + /// And the seal really does carry all three rows — the denominator's claim, checked against the + /// object it describes rather than against another counter. + for (const RootNamespace & ns : {a, b, c}) + { + EXPECT_NE(sealedCursorOf(*backend, layout, ns), (RefTxnId{})) + << "every namespace in the tally must have a sealed cursor: " << ns.string(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const auto checkpoint = readCkpt(*backend, layout, life); + ASSERT_TRUE(checkpoint.has_value()); + EXPECT_EQ(checkpoint->ckpt.committed_through, (RefTxnId{1, 1})) + << "LIST omission and the probe budget do not alter a valid CTE"; + } +} + +/// A checkpoint boundary already equal to the carried cursor proves a quiet catalog life complete; +/// GC must not manufacture a successor `GET` merely because its LIST is empty. +TEST(CASGCFrontierGate, AQuietKnownNamespaceAtItsCheckpointFrontierCostsNoSuccessorGet) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace quiet{"00/quiet@cas@"}; + + publish(*backend, layout, quiet, "ref_1", 1, DB::UInt128(0x11)); + replaceRecoverableCkptForRawFixture(*backend, layout, quiet, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + + const RefTxnId sealed = sealedCursorOf(*backend, layout, quiet); + ASSERT_NE(sealed, (RefTxnId{})) << "the seeding round must have sealed a cursor to carry"; + + /// Now the store stops listing the namespace entirely. + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(quiet))); + backend->resetCounts(); + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + const String expected_next = + layout.refLogKey(fixture::fixtureLife(quiet), RefTxnId{sealed.writer_epoch, sealed.ref_sequence + 1}); + EXPECT_EQ(backend->getCount(expected_next), 0u) + << "the inclusive checkpoint boundary proves this quiet life without a successor probe"; + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]) + << "the inherited cursor already at the checkpoint boundary is destructive-eligible"; +} + +/// A checkpoint must never retreat below a sealed cursor. Its inclusive frontier can prove a cursor +/// already at that point, but cannot explain one that has advanced beyond it. +TEST(CASGCFrontierGate, CheckpointFrontierBehindAnInheritedCursorFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-behind-inherited-cursor@cas@"}; + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "first", 1, DB::UInt128(0xfb)); + publish(*backend, layout, ns, "second", 2, DB::UInt128(0xfc)); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String checkpoint_key = layout.refCkptKey(life); + const HeadResult checkpoint_head = backend->head(checkpoint_key); + ASSERT_TRUE(checkpoint_head.exists); + ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }), checkpoint_head.token).outcome, PutOutcome::Done); + + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + EXPECT_FALSE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_LT(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A carried `EpochSeal` may have its authoritative successor in the next epoch. The arithmetic +/// successor in the sealed epoch is absent by design, so the exact checkpoint frontier must nominate +/// the shared seal-chain crossing before that absence is classified as a same-epoch gap. +TEST(CASGCFrontierGate, CheckpointFrontierCrossesAnInheritedEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-inherited-seal-crossing@cas@"}; + const DB::UInt128 crossed_blob(0xfd); + + fixture::admitLive(*backend, layout, ns); + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "birth", 1, DB::UInt128(0xfe), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); + + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "crossed", 2, crossed_blob, + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String checkpoint_key = layout.refCkptKey(life); + const HeadResult checkpoint_head = backend->head(checkpoint_key); + ASSERT_TRUE(checkpoint_head.exists); + ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }), checkpoint_head.token).outcome, PutOutcome::Done); + + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + EXPECT_TRUE(report.anomalies.empty()); + EXPECT_GT(inDegreeOf(*backend, layout, crossed_blob), 0); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// The exact checkpoint successor must chain to the seal just consumed. Merely being in the next epoch +/// is insufficient: an incorrect predecessor would skip an unclosed history segment forever. +TEST(CASGCFrontierGate, CheckpointFrontierRejectsWrongPredecessorAfterFreshEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-wrong-fresh-seal-predecessor@cas@"}; + + fixture::admitLive(*backend, layout, ns); + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "birth", 1, DB::UInt128(0xff), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "wrong_predecessor", 2, DB::UInt128(0x100), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 1}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + EXPECT_FALSE(report.anomalies.empty()); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); + ASSERT_FALSE(intake.empty()); + EXPECT_LT(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A namespace that was WRONGLY quiet -- the hint hid a record that is durably there -- is walked this +/// round, not next: the probe finds the record and the walk continues from it. +TEST(CASGCFrontierGate, AWronglyQuietNamespaceIsWalkedTheSameRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace quiet{"00/quiet@cas@"}; + const DB::UInt128 late_blob(0x77); + + publish(*backend, layout, quiet, "ref_1", 1, DB::UInt128(0x11)); + replaceRecoverableCkptForRawFixture(*backend, layout, quiet, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const RefTxnId sealed_before = sealedCursorOf(*backend, layout, quiet); + + /// A second publish lands, and the store hides the namespace from every LIST at the same moment. + publish(*backend, layout, quiet, "ref_2", 2, late_blob); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); + const String checkpoint_key = layout.refCkptKey(life); + const HeadResult checkpoint_head = backend->head(checkpoint_key); + ASSERT_TRUE(checkpoint_head.exists); + ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }), checkpoint_head.token).outcome, PutOutcome::Done); + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(quiet))); + + runRegularRoundReclaiming(gc); + + EXPECT_LT(sealed_before, sealedCursorOf(*backend, layout, quiet)) + << "the probe found the hidden record, so the walk folded it and the cursor advanced"; + EXPECT_GT(inDegreeOf(*backend, layout, late_blob), 0) + << "the hidden publish's edge folded this round -- the hint never mentioned it"; +} + +/// The catalog life is grounded by its exact decoded `_ckpt`, not by the round's listing or a later +/// absent probe. A durable `F+1` is physically present but not committed history, so this fold may apply +/// only `F`; in particular it must not read `F+2`. Reaching `F` still proves the checkpoint-bounded +/// cut, so the physical successor cannot suppress otherwise eligible destructive work. +TEST(CASGCFrontierGate, CheckpointFrontierBoundsOrdinaryFoldBeforeDurableSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-bounds-fold@cas@"}; + const DB::UInt128 committed_blob(0xf1); + const DB::UInt128 beyond_frontier_blob(0xf2); + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "committed", 1, committed_blob); + const ManifestRef uncommitted{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, beyond_frontier_blob); + writeManifestRaw(*backend, layout, ns, uncommitted, {blobEntryFor("data.bin", beyond_frontier_blob)}); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = RefTxnId{1, 2}, + .ops = publishCommittedOps("durable_but_uncommitted", uncommitted), + .prev_epoch_seal = std::nullopt, + }); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + backend->resetCounts(); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_EQ(inDegreeOf(*backend, layout, beyond_frontier_blob), 0) + << "a durable log above `_ckpt.committed_through` is not foldable history"; + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 2})), 0u) + << "the checkpoint frontier stops the walk before `F+1`"; + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 3})), 0u) + << "a 404 above `F+1` must not authorize the destructive frontier"; + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]) + << "the consumed checkpoint frontier, not the physical successor, authorizes this cut"; +} + +/// With no physical successor at all, consuming the exact inclusive checkpoint frontier proves this +/// catalog life complete. This is the control for the same bounded-cut proof exercised with a durable +/// uncommitted successor above. +TEST(CASGCFrontierGate, ConsumedCheckpointFrontierProvesOrdinaryLifeWithoutSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-complete-fold@cas@"}; + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "committed", 1, DB::UInt128(0xf3)); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + backend->resetCounts(); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 2})), 0u) + << "the checkpoint boundary proves the cut without a post-frontier 404"; + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// The same cut remains complete when the durable uncommitted successor is hidden from every LIST. +/// Exact reads still serve that successor, but the checkpoint ceiling must leave it untouched and must +/// not let the list omission suppress the checkpoint-bounded destructive path. +TEST(CASGCFrontierGate, CheckpointFrontierProvesLifeWithHiddenDurableSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/checkpoint-hidden-successor@cas@"}; + const DB::UInt128 beyond_frontier_blob(0xf4); + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "committed", 1, DB::UInt128(0xf5)); + const ManifestRef uncommitted{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, beyond_frontier_blob); + writeManifestRaw(*backend, layout, ns, uncommitted, {blobEntryFor("data.bin", beyond_frontier_blob)}); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = RefTxnId{1, 2}, + .ops = publishCommittedOps("hidden_durable_but_uncommitted", uncommitted), + .prev_epoch_seal = std::nullopt, + }); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + backend->hide(layout.refLogKey(life, RefTxnId{1, 2})); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + backend->resetCounts(); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + EXPECT_GT(backend->holesServed(), 0u) << "the F+1 log must really be hidden from LIST"; + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_EQ(inDegreeOf(*backend, layout, beyond_frontier_blob), 0); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 2})), 0u) + << "the hidden durable successor is outside the checkpoint cut"; + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// The checkpoint's inclusive endpoint is itself a durable witness. If that exact log is absent, +/// the namespace is corrupt rather than complete; a 404 at the endpoint must not authorize cleanup. +TEST(CASGCFrontierGate, MissingCommittedCheckpointLogHoldsInsteadOfProvingTheFrontier) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/missing-committed-checkpoint-log@cas@"}; + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "first", 1, DB::UInt128(0xf6)); + publish(*backend, layout, ns, "missing_but_committed", 2, DB::UInt128(0xf7)); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String missing_key = layout.refLogKey(life, RefTxnId{1, 2}); + const HeadResult missing_head = backend->head(missing_key); + ASSERT_TRUE(missing_head.exists); + ASSERT_EQ(backend->deleteExact(missing_key, missing_head.token).kind, DeleteOutcome::Kind::Deleted); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_FALSE(report.anomalies.empty()) << "the missing committed checkpoint record is corruption"; + ASSERT_FALSE(intake.empty()); + EXPECT_LT(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A checkpoint may name a durable record that the round's LIST omitted. Exact GETs must still fold +/// that committed record; the frozen list tail is only a scheduling hint, never a history boundary. +TEST(CASGCFrontierGate, HiddenCommittedCheckpointLogIsFoldedThroughTheAuthorityCeiling) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/hidden-committed-checkpoint-log@cas@"}; + const DB::UInt128 hidden_blob(0xf8); + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "first", 1, DB::UInt128(0xf9)); + publish(*backend, layout, ns, "hidden_but_committed", 2, hidden_blob); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + backend->hide(layout.refLogKey(life, RefTxnId{1, 2})); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + EXPECT_GT(backend->holesServed(), 0u) << "the committed endpoint must really be omitted from LIST"; + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); + EXPECT_GT(inDegreeOf(*backend, layout, hidden_blob), 0); + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A valid checkpoint with no committed record is an authoritative empty history. It is complete for +/// a never-folded life without probing a fabricated first transaction. +TEST(CASGCFrontierGate, EmptyCheckpointFrontierProvesAnUnfoldedLife) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/empty-checkpoint-frontier@cas@"}; + + casAdmitRecoverableEntry(*backend, layout, ns); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + EXPECT_TRUE(report.anomalies.empty()); + ASSERT_FALSE(intake.empty()); + EXPECT_EQ(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// Empty history cannot explain an inherited cursor. An operator-corrupted checkpoint that erases its +/// own committed boundary must clamp the life rather than silently authorize destruction. +TEST(CASGCFrontierGate, EmptyCheckpointFrontierRejectsAnInheritedCursor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/empty-checkpoint-after-cursor@cas@"}; + + fixture::admitLive(*backend, layout, ns); + publish(*backend, layout, ns, "first", 1, DB::UInt128(0xfa)); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String checkpoint_key = layout.refCkptKey(life); + const HeadResult checkpoint_head = backend->head(checkpoint_key); + ASSERT_TRUE(checkpoint_head.exists); + ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }), checkpoint_head.token).outcome, PutOutcome::Done); + + std::map intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + gc.setPhaseSink({}); + + EXPECT_FALSE(report.anomalies.empty()) << "an empty checkpoint cannot explain a nonzero cursor"; + ASSERT_FALSE(intake.empty()); + EXPECT_LT(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A catalog `Live` life without its exact checkpoint cannot derive either its genesis or a frontier +/// from the ref LIST. Even a durable listed first log must be retained until the authority is repaired. +TEST(CASGCFrontierGate, CatalogLifeWithoutCheckpointDefersWithoutUsingListedFrontier) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/missing-checkpoint-fold@cas@"}; + const DB::UInt128 blob(0xc7); + + fixture::admitLive(*backend, layout, ns); + const ManifestRef manifest{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, blob); + writeManifestRaw(*backend, layout, ns, manifest, {blobEntryFor("data.bin", blob)}); + appendRefLogSeed(*backend, layout, ns, publishCommittedOps("must_remain_unfolded", manifest)); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_FALSE(readCkpt(*backend, layout, life).has_value()); + + std::map intake; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + intake = rec.metrics; + }); + backend->resetCounts(); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(foldCursorOf(*backend, layout, ns, /*shard=*/0), 0u); + EXPECT_EQ(inDegreeOf(*backend, layout, blob), 0) + << "a missing checkpoint must defer rather than fold the listed log"; + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 1})), 0u) + << "the listed log is not authority for a checkpoint-less catalog life"; + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 2})), 0u) + << "the next 404 is not authority for a checkpoint-less catalog life"; + EXPECT_EQ(backend->deleteTotal(), 0u) << deletedKeysMessage(*backend); + ASSERT_FALSE(intake.empty()); + EXPECT_LT(intake["frontier_proven"], intake["frontier_namespaces"]); +} + +/// A valid checkpoint frontier proves a quiet unhinted life without spending the successor-probe budget. +/// A zero budget therefore cannot suppress unrelated destructive work merely because this life is absent +/// from LIST. +TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSealsCursorsAndDeletesNothing) +{ + auto backend = std::make_shared(); + auto store = openPoolWithProbeBudget(backend, /*budget*/ 0); + const Layout & layout = store->layout(); + const RootNamespace quiet{"00/quiet@cas@"}; + const RootNamespace busy{"00/busy@cas@"}; + const DB::UInt128 blob(0xbeef); + + publish(*backend, layout, quiet, "quiet_ref", 1, DB::UInt128(0x11)); + const ManifestRef mref = publish(*backend, layout, busy, "busy_ref", 2, blob); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + const RefTxnId quiet_cursor = sealedCursorOf(*backend, layout, quiet); + ASSERT_NE(quiet_cursor, (RefTxnId{})); + + /// The quiet namespace goes unhinted and the budget is zero. Its CTE still proves the carried + /// cursor, while the busy namespace drops its ref and may proceed through reclamation. + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(quiet))); + dropRefTransition(*backend, layout, busy, "busy_ref", mref); + + backend->resetCounts(); + drive(store, gc, /*rounds*/ 5, UniversePolicy::Authoritative); + + EXPECT_GT(backend->deleteTotal(), 0u) + << "the quiet life's checkpoint authority leaves unrelated deletion eligible"; + EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + << "the busy life's removal remains reclaimable despite the quiet LIST omission"; + EXPECT_EQ(sealedCursorOf(*backend, layout, quiet), quiet_cursor) + << "the unprobed namespace's cursor rides verbatim -- it is never dropped"; + const NamespaceLifeId quiet_life = *CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); + const auto quiet_checkpoint = readCkpt(*backend, layout, quiet_life); + ASSERT_TRUE(quiet_checkpoint.has_value()); + EXPECT_EQ(quiet_checkpoint->ckpt.committed_through, quiet_cursor) + << "the quiet life's valid CTE is unaffected by LIST omission and a zero probe budget"; + EXPECT_GT(decodeGcState(backend->get(layout.gcStateKey())->bytes).round, 1u) + << "the round still commits; only its destructive half is withheld"; +} + +/// ===================== A COMMITTED GAP IS REDETECTED UNTIL REPAIRED ===================== +/// +/// A hold's committed checkpoint frontier remains a durable witness of its own gap. Hiding the later +/// log from LIST cannot make that gap quiet: every retry exact-reads the missing position, redetects the +/// hold, and suppresses destructive work until an operator repairs the record stream. +TEST(CASGCFrontierGate, ACommittedGapIsRedetectedAndSuppressesEveryRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace held{"00/held@cas@"}; + const RootNamespace busy{"00/busy@cas@"}; + const DB::UInt128 blob(0xbeef); + + /// {1,3} never existed while {1,4} is durable and listed. + publish(*backend, layout, held, "ref_1", 1, DB::UInt128(0x21)); + publish(*backend, layout, held, "ref_2", 2, DB::UInt128(0x22)); + const ManifestRef orphan_ref{.writer_epoch = 1, .build_sequence = 4, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, DB::UInt128(0x24)); + writeManifestRaw(*backend, layout, held, orphan_ref, {blobEntryFor("data.bin", DB::UInt128(0x24))}); + RefLogTxn txn; + txn.ns = held.string(); + txn.txn_id = RefTxnId{1, 4}; + txn.ops = publishCommittedOps("ref_4", orphan_ref); + fixture::writeRefLogRaw(*backend, layout, txn); + replaceRecoverableCkptForRawFixture(*backend, layout, held, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 4}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + std::map first_intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + first_intake = rec.metrics; + }); + const RoundReport first_round = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + store->renewWatermarkOnce(); + ASSERT_EQ(sealedCursorOf(*backend, layout, held), (RefTxnId{1, 2})) + << "round 1 must stop below the gap and hold there"; + ASSERT_FALSE(first_intake.empty()); + EXPECT_GT(first_intake["tables_clamped"], 0u); + EXPECT_GT(first_intake["tables_held"], 0u); + EXPECT_FALSE(first_round.anomalies.empty()); + + const NamespaceLifeId held_life = *CasRefCatalog::lifeIfCataloged(*backend, layout, held); + const auto held_checkpoint = readCkpt(*backend, layout, held_life); + ASSERT_TRUE(held_checkpoint.has_value()); + EXPECT_EQ(held_checkpoint->ckpt.committed_through, (RefTxnId{1, 4})); + + /// Hiding `{1,4}` from LIST does not hide the committed CTE frontier. The next round exact-reads + /// the missing `{1,3}`, re-detects the gap, and seals a fresh hold. + backend->hidePrefix(layout.refLogKey(fixture::fixtureLife(held), RefTxnId{1, 4})); + + /// Meanwhile a blob elsewhere becomes condemnable, so the round has real destructive work to decline. + const ManifestRef mref = publish(*backend, layout, busy, "busy_ref", 9, blob); + dropRefTransition(*backend, layout, busy, "busy_ref", mref); + + backend->resetCounts(); + std::map second_intake; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + second_intake = rec.metrics; + }); + const RoundReport second_round = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + store->renewWatermarkOnce(); + + ASSERT_FALSE(second_intake.empty()); + EXPECT_GT(second_intake["tables_clamped"], 0u) + << "the committed `{1,4}` frontier is a durable witness that re-detects the missing `{1,3}`"; + EXPECT_GT(second_intake["tables_held"], 0u) + << "the fresh clamp preserves the unresolved hold in the next sealed coverage"; + EXPECT_FALSE(second_round.anomalies.empty()); + + drive(store, gc, /*rounds*/ 4, UniversePolicy::Authoritative); + + EXPECT_EQ(backend->deleteTotal(), 0u) + << "the re-detected committed gap suppresses each round's destructive work. " + "Deleted:" << deletedKeysMessage(*backend); + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + EXPECT_EQ(sealedCursorOf(*backend, layout, held), (RefTxnId{1, 2})) + << "the committed gap remains unresolved and the cursor cannot advance through it"; + const auto final_checkpoint = readCkpt(*backend, layout, held_life); + ASSERT_TRUE(final_checkpoint.has_value()); + EXPECT_EQ(final_checkpoint->ckpt.committed_through, (RefTxnId{1, 4})); +} + +/// ===================== THE TEMPORAL LEMMA, ALL THREE ARMS ===================== +/// +/// The gate says WHEN a round may destroy. These say that even a round which may destroy cannot +/// destroy a blob some edge still owns, over the three interleavings that matter. + +/// ARM (a): a `+1` that lands after this round's probes and is followed by the SAME round's +/// condemnation. Round pacing makes it safe on its own: an entry condemned at round K cannot graduate +/// before K+1 and cannot be deleted before K+2, so the round that condemns never deletes. +TEST(CASGCFrontierGate, ABlobCondemnedThisRoundIsNeverDeletedThisRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const DB::UInt128 blob(0xc04d); + + const ManifestRef mref = publish(*backend, layout, ns, "ref_1", 1, blob); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + + dropRefTransition(*backend, layout, ns, "ref_1", mref); + backend->resetCounts(); + runRegularRoundReclaiming(gc); + + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the condemning round must not also delete"; + EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, blob)), 0u) + << "not merely still present -- the delete was never attempted"; +} + +/// ARM (c) of the temporal lemma is the delete-site in-degree re-read, and it is NORMATIVE (spec §5, +/// third arm): an edge folded AFTER the condemnation but BEFORE the delete pass spares the blob +/// outright, `indeg > 0` winning over `delete_pending` past the floor. The other two arms bound WHEN +/// and WHAT a delete may remove; only this one asks whether the blob is still referenced at the moment +/// the pass decides. +TEST(CASGCFrontierGate, ALateEdgeSparesADeletePendingBlobAtTheDeleteSite) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const DB::UInt128 blob(0x1a7e); + + const ManifestRef mref = publish(*backend, layout, ns, "ref_1", 1, blob); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + + /// Condemn it, then graduate it to delete_pending. + dropRefTransition(*backend, layout, ns, "ref_1", mref); + runRegularRoundReclaiming(gc); /// condemn + store->renewWatermarkOnce(); + runRegularRoundReclaiming(gc); /// graduate: delete_pending published + store->renewWatermarkOnce(); + + /// A new owner appears BEFORE the delete pass. The pass recomputes the in-degree from the merge it + /// just ran and finds it nonzero. + const ManifestRef revived{.writer_epoch = 1, .build_sequence = 42, .manifest_ordinal = 1}; + writeManifestRaw(*backend, layout, ns, revived, {blobEntryFor("data.bin", blob)}); + publishCommittedTransition(*backend, layout, ns, "revived_ref", std::nullopt, revived); + + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the delete-site in-degree re-read spares a blob a fresh edge re-referenced"; + EXPECT_GT(inDegreeOf(*backend, layout, blob), 0); +} + +/// ARM (b): a TOKENED adoption of an already-delete-pending blob. The writer's admit gate reads the +/// `Condemned` meta, refuses to adopt the dying incarnation, and rematerializes from its own source as +/// a FRESH incarnation -- so the delayed exact-token delete the previous round published finds a +/// different token and removes nothing. The blob's identity is preserved by re-upload, never by +/// reviving the condemned object. +TEST(CASGCFrontierGate, AResurrectedIncarnationSurvivesTheDelayedStaleTokenDelete) +{ + ensureBlobUploadPoolForTest(); + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + /// A REAL content-addressed blob, so the writer path below addresses exactly the object GC condemns. + const String payload = "frontier-gate-republish-payload"; + const DB::UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const String key = layout.blobKey(id); + String raw_body(store->poolMeta().blob_header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*backend, layout, hash, raw_body); + + /// Publish and drop it so GC condemns and then graduates it to delete_pending. + const ManifestRef mref{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}; + writeManifestRaw(*backend, layout, ns, mref, {blobEntryFor("data.bin", hash)}); + publishCommittedTransition(*backend, layout, ns, "ref_1", std::nullopt, mref); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + dropRefTransition(*backend, layout, ns, "ref_1", mref); + runRegularRoundReclaiming(gc); /// condemn: writes the durable Condemned meta + store->renewWatermarkOnce(); + runRegularRoundReclaiming(gc); /// graduate: publishes delete_pending against THIS token + store->renewWatermarkOnce(); + + const Token condemned_token = backend->head(key).token; + const auto condemned_meta = loadMetaForTest(*backend, layout, hash); + ASSERT_TRUE(condemned_meta.has_value()); + ASSERT_EQ(condemned_meta->meta.state, MetaState::Condemned) + << "the delete GC is about to execute must be backed by durable Condemned evidence"; + + /// A writer now adopts the blob through the REAL admit gate. It point-reads the Condemned meta, + /// refuses to adopt the dying incarnation, and rematerializes from its OWN source bytes -- never by + /// reading the condemned object. The key ends up holding a DIFFERENT incarnation. + PartWriteInfo info; + info.intended_ref = ns.string() + "/republished"; + auto build = store->beginPartWrite(info); + const ManifestId republished_manifest + = build->stageManifest({blobEntryFor("data.bin", hash, payload.size())}); + build->precommitAdd(ns, "republished", republished_manifest); + const PutBlobResult uploaded = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(uploaded.ref, id); + build->promote(ns, "republished", build->buildId(), republished_manifest); + const Token fresh_token = backend->head(key).token; + ASSERT_NE(fresh_token, condemned_token) << "republication must displace the condemned incarnation"; + + /// GC's delayed delete still names the OLD token. It cannot touch the new object. + drive(store, gc, /*rounds*/ 2, UniversePolicy::Authoritative); + + ASSERT_TRUE(backend->head(key).exists) + << "the resurrected incarnation survives the delete published against its predecessor"; + EXPECT_EQ(backend->head(key).token, fresh_token) << "and it is still the writer's incarnation"; + EXPECT_EQ(backend->deleteExact(key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch) + << "the condemned token can never remove the fresh object (INV-NO-RETURN)"; +} + +/// ARM (c): a TOKENLESS relink -- the receiver adopts by evidence, holding no token at all. Safety +/// then rests entirely on ORDER, so the operation journal has to show it: the receiver's `+1` is +/// durable BEFORE the source releases its own committed edge, and no point in the schedule leaves the +/// blob with zero durable owners. +TEST(CASGCFrontierGate, ATokenlessRelinkMakesTheReceiverEdgeDurableBeforeTheSourceReleases) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace source{"00/source@cas@"}; + const RootNamespace receiver{"00/receiver@cas@"}; + const DB::UInt128 blob(0x8e11); + + const ManifestRef source_ref = publish(*backend, layout, source, "part_1", 1, blob); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + ASSERT_GT(inDegreeOf(*backend, layout, blob), 0); + + /// The relink, in the only order the protocol permits: the receiver's manifest body and its + /// committed edge first (tokenless -- it never HEADs the blob), and only afterwards the source's + /// removal. Between the two writes the blob has TWO durable owners; it never has zero. + const ManifestRef receiver_ref{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; + writeManifestRaw(*backend, layout, receiver, receiver_ref, {blobEntryFor("data.bin", blob)}); + publishCommittedTransition(*backend, layout, receiver, "part_1", std::nullopt, receiver_ref); + + /// The round that observes ONLY the receiver's `+1` -- the exact midpoint of the schedule. + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + EXPECT_GE(inDegreeOf(*backend, layout, blob), 2) + << "at the midpoint both owners are durable; the handoff never dips to zero"; + + dropRefTransition(*backend, layout, source, "part_1", source_ref); + drive(store, gc, /*rounds*/ 4, UniversePolicy::Authoritative); + + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the source released its edge only after the receiver's was durable, so nothing may collect it"; + EXPECT_EQ(inDegreeOf(*backend, layout, blob), 1) + << "the receiver is the sole remaining owner"; +} + +/// ===================== CLEANUP RANGES ARE COMPUTED, NOT ENUMERATED ===================== +/// +/// `planRefCleanup` is pure, so the boundary arithmetic is pinned directly rather than inferred from a +/// round's side effects. Its sole coverage authority is the checkpoint-named base; a listed snapshot +/// is merely a physical observation until the same-id triple has been validated. + +TEST(CASGCFrontierGateCleanupRange, CoveredLogsStopAtTheMinimumOfCheckpointAndCursor) +{ + RefTableListing listing; + listing.logs = {{1, 1}, {1, 2}, {1, 3}, {1, 4}, {1, 5}}; + listing.snapshots = {{1, 5}}; + + /// No checkpoint means no recovery base at all. A snapshot PUT that has not reached the `_ckpt` + /// CAS must retain every listed object. + const RefCleanupPlan without = planRefCleanup(listing, RefTxnId{1, 4}); + EXPECT_TRUE(without.deletable_logs.empty()); + EXPECT_TRUE(without.deletable_snapshots.empty()); + + /// A checkpoint BELOW the cursor tightens it to {1,2}. Its exact `_log` witness must survive, + /// so cleanup may remove only the strictly older entry. + const RefCleanupPlan with = planRefCleanup(listing, RefTxnId{1, 4}, RefTxnId{1, 2}); + EXPECT_EQ(with.deletable_logs, (std::vector{{1, 1}})) + << "the checkpoint witness and everything above it must survive"; + + /// Once validation has established a later checkpoint base, its earlier covered history is + /// reclaimable even if the hot fold cursor has not yet reached that base. + const RefCleanupPlan ahead = planRefCleanup(listing, RefTxnId{1, 4}, RefTxnId{1, 9}); + EXPECT_EQ(ahead.deletable_logs, (std::vector{{1, 1}, {1, 2}, {1, 3}, {1, 4}})); + EXPECT_EQ(ahead.deletable_snapshots, (std::vector{{1, 5}})); +} + +TEST(CASGCFrontierGateCleanupRange, ASnapshotAtTheCheckpointSurvivesAndOnlyStrictlyOlderOnesGo) +{ + RefTableListing listing; + listing.logs = {{1, 1}, {1, 2}, {1, 3}}; + listing.snapshots = {{1, 1}, {1, 2}, {1, 3}}; + + /// A LIST-only newest snapshot is never a cleanup boundary. + const RefCleanupPlan without = planRefCleanup(listing, RefTxnId{1, 3}); + EXPECT_TRUE(without.deletable_snapshots.empty()); + + /// With the checkpoint AT {1,2}, only {1,1} is strictly below it. The snapshot the checkpoint names + /// is the one a recovering reader samples, so it must survive its own cleanup. + const RefCleanupPlan with = planRefCleanup(listing, RefTxnId{1, 3}, RefTxnId{1, 2}); + EXPECT_EQ(with.deletable_snapshots, (std::vector{{1, 1}})); + + /// The oldest checkpoint deletes nothing at all. + const RefCleanupPlan oldest = planRefCleanup(listing, RefTxnId{1, 3}, RefTxnId{1, 1}); + EXPECT_TRUE(oldest.deletable_snapshots.empty()); +} + +/// Cleanup shares recovery's validator rather than inferring its own authority from a LIST. The +/// missing-base case is the no-checkpoint range above; the three physical triple failures below must +/// each reject exactly the checkpoint-named candidate. +TEST(CASGCFrontierGateCleanupRange, CheckpointBaseValidatorRejectsMissingLogSnapshotAndSeal) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RefTxnId base{1, 1}; + const RefCkpt checkpoint{ + .life_epoch = 1, + .committed_through = base, + .checkpoint_snapshot_id = base, + .last_epoch_seal = std::nullopt}; + CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + + { + const RootNamespace ns{"00/cleanup-missing-base-log@cas@"}; + fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base)); + EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + } + { + const RootNamespace ns{"00/cleanup-missing-base-snapshot@cas@"}; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + } + { + const RootNamespace ns{"00/cleanup-seal-is-not-base@cas@"}; + writeSealAt(*backend, layout, ns, base); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base)); + EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + } +} + +TEST(CASGCFrontierGateCleanupRange, LaterEpochBaseWithoutItsContextualBacklinkCannotLicenseDeletion) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + const RefTxnId seal_id{1, 2}; + const RefTxnId base_id{2, 1}; + + const auto expect_no_deletion_authority = [&](const RootNamespace & ns, std::optional backlink) + { + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}); + writeSealAt(*backend, layout, ns, seal_id); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = base_id, .ops = {}, .prev_epoch_seal = backlink}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + + std::optional validated_base; + try + { + (void)readCheckpointSnapshotBase(*backend, layout, life, RefCkpt{ + .life_epoch = 1, + .committed_through = base_id, + .checkpoint_snapshot_id = base_id, + .last_epoch_seal = seal_id}); + validated_base = base_id; + } + catch (const DB::Exception &) // NOLINT(bugprone-empty-catch): failure to validate is the tested case -- `validated_base` deliberately stays nullopt + { + } + + RefTableListing listing; + listing.logs = {{1, 1}, seal_id, base_id}; + listing.snapshots = {{1, 1}, base_id}; + const RefCleanupPlan plan = planRefCleanup(listing, base_id, validated_base); + EXPECT_TRUE(plan.deletable_logs.empty()); + EXPECT_TRUE(plan.deletable_snapshots.empty()); + }; + + expect_no_deletion_authority(RootNamespace{"00/cleanup-base-missing-backlink@cas@"}, std::nullopt); + expect_no_deletion_authority(RootNamespace{"00/cleanup-base-wrong-backlink@cas@"}, RefTxnId{1, 99}); +} + +/// Folding a namespace terminal records evidence but performs no lifecycle-specific physical cleanup. +/// The checkpoint is inert debris for the perpetual janitor, and no `_cleanup` marker is published. +TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace removed{"00/removed@cas@"}; + const RefOp birth_op = namespaceBirthOp(); + RefOp remove_op; + remove_op.kind = RefOpKind::RemoveNamespace; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = removed.string(), .txn_id = RefTxnId{1, 1}, .ops = {birth_op}, .prev_epoch_seal = std::nullopt}); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = removed.string(), .txn_id = RefTxnId{1, 2}, .ops = {remove_op}, .prev_epoch_seal = std::nullopt}); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, removed).value(); + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == removed; + }); + EXPECT_NE(it, next.entries.end()); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + const String ckpt_key = layout.refCkptKey(life); + backend->putIfAbsent(ckpt_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + })); + + /// The removal evidence must arise from a replay-valid terminal lifecycle, rather than merely + /// from a raw terminal record that the recovery state machine refuses. + const RecoveredRefTable recovered = recoverRefTableDetailedAtCatalogCutForTest( + *backend, layout, CasRefCatalog::read(*backend, layout), removed); + EXPECT_EQ(recovered.state.getLifecycle(), RefLifecycle::Removed); + EXPECT_EQ(recovered.state.getRemoveTxnId(), (RefTxnId{1, 2})); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const auto row_it = seal.ref_lives.find(life.incarnation); + ASSERT_NE(row_it, seal.ref_lives.end()); + ASSERT_TRUE(row_it->second.cleanup_evidence.has_value()); + EXPECT_EQ(row_it->second.cleanup_evidence->remove_txn_id, (RefTxnId{1, 2})); + EXPECT_TRUE(backend->head(ckpt_key).exists); + for (const String & key : backend->touchedKeys()) + EXPECT_EQ(key.find("/_cleanup/"), String::npos) << key; + + /// Round 2 drops `removed`'s catalog row (the pre-fold drain, using round 1's `cleanup_evidence`), + /// which makes THIS round's own hot-scan catalog cut genuinely, provably empty -- so its destructive + /// gate opens for the first time (`catalog_cut_proved_empty`), and the namespace janitor -- a + /// separate `namespace_cleanup` phase the SAME round call also runs -- reclaims the now-orphaned + /// checkpoint. Reclaiming a removed namespace's `_ckpt` once the pool empties is exactly the + /// standstill this gate exists to fix, so the janitor running here is the fix working, not a + /// regression. What this test still pins is the DISCRIMINATION the title promises: the FOLD stage + /// itself performs no lifecycle-specific physical cleanup (asserted above, unchanged), and the + /// janitor is attributed the delete via its OWN phase counters -- never inferred from end-state + /// absence, which would not distinguish "the janitor did it" from "something else did". + std::map janitor_metrics; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "namespace_cleanup") + janitor_metrics = rec.metrics; + }); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)); + ASSERT_FALSE(janitor_metrics.empty()) << "the namespace_cleanup phase must have run this round"; + EXPECT_GE(janitor_metrics.at("janitor_deleted"), 1u) + << "the janitor's OWN counter must show the delete -- now that the proved-empty gate has " + "opened, not because some other site happened to remove the key"; + EXPECT_FALSE(backend->head(ckpt_key).exists); + EXPECT_EQ(backend->deleteCount(ckpt_key), 1); +} + +/// Once a terminal has folded, a later physical read failure is janitor debt, not lifecycle evidence +/// loss. Removing this per-key leak handling would either make the signal disappear or let one dead +/// object prevent the janitor from considering the rest of its page. +TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingProgress) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace removed{"00/post-fold-unreadable@cas@"}; + const RootNamespace progressing{"00/post-fold-progress@cas@"}; + + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = removed.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}); + RefOp remove_op; + remove_op.kind = RefOpKind::RemoveNamespace; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = removed.string(), .txn_id = RefTxnId{1, 2}, .ops = {remove_op}, + .prev_epoch_seal = std::nullopt}); + const NamespaceLifeId removed_life = CasRefCatalog::lifeIfCataloged(*backend, layout, removed).value(); + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == removed; + }); + if (it == next.entries.end()) + throw std::runtime_error("test fixture lost removing catalog row"); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(removed_life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + })).outcome, PutOutcome::Done); + + const DB::UInt128 blob(0xfeed); + const ManifestRef manifest = publish(*backend, layout, progressing, "victim", 1, blob); + const ManifestId manifest_id{progressing, manifest}; + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const GcState folded_state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal folded_seal = decodeFoldSeal( + backend->get(layout.foldSealKey(folded_state.snap_generation, folded_state.snap_attempt))->bytes); + const auto folded_row = folded_seal.ref_lives.find(removed_life.incarnation); + ASSERT_NE(folded_row, folded_seal.ref_lives.end()); + ASSERT_TRUE(folded_row->second.cleanup_evidence.has_value()); + + dropRefTransition(*backend, layout, progressing, "victim", manifest); + const String terminal_key = layout.refLogKey(removed_life, RefTxnId{1, 2}); + const String later_dead_residue = layout.refLogKey(removed_life, RefTxnId{1, 3}); + ASSERT_EQ(backend->putIfAbsent(later_dead_residue, "dead residue after the folded terminal").outcome, + PutOutcome::Done); + backend->makeUnreadable(terminal_key); + + std::map namespace_cleanup; + const uint64_t leaks_before + = ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks].load(); + gc.setPhaseSink([&](const GcPhaseRecord & record) + { + if (record.phase == "namespace_cleanup") + namespace_cleanup = record.metrics; + }); + ScopedCasGcLogCapture log_capture; + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + ASSERT_TRUE(report.acquired_lease); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)) + << "post-fold physical cleanup cannot gate catalog removal"; + EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, progressing)); + EXPECT_EQ(report.manifests_deleted, 1u) + << "the janitor leak cannot promote itself into pool-wide destructive suppression"; + EXPECT_FALSE(backend->head(layout.manifestKey(manifest_id)).exists); + EXPECT_TRUE(backend->existsIgnoringFault(terminal_key)); + EXPECT_FALSE(backend->existsIgnoringFault(later_dead_residue)) + << "one unreadable key cannot stop the perpetual janitor from deciding the rest of its page"; + ASSERT_FALSE(namespace_cleanup.empty()); + EXPECT_EQ(namespace_cleanup["leaked"], 1u); + EXPECT_EQ( + ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks].load() - leaks_before, + 1u); + const String captured = log_capture.captured(); + EXPECT_NE(captured.find(terminal_key), String::npos); + EXPECT_NE(captured.find("leak"), String::npos); +} + +TEST(CASGCFrontierGate, UnmatchedAdoptedParentLifeDoesNotSuppressAuthoritativeDeletion) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/unmatched-parent@cas@"}; + const DB::UInt128 blob(0xcafe); + const ManifestRef mref = publish(*backend, layout, ns, "victim", 1, blob); + const ManifestId manifest_id{ns, mref}; + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_TRUE(backend->head(layout.manifestKey(manifest_id)).exists); + + const GcState before = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const String parent_seal_key = layout.foldSealKey(before.snap_generation, before.snap_attempt); + const auto parent_object = backend->get(parent_seal_key); + ASSERT_TRUE(parent_object); + CasFoldSeal parent = decodeFoldSeal(parent_object->bytes, before.snap_generation); + const UInt128 unmatched_life = hexToU128("fedcba98765432100123456789abcdef"); + ASSERT_FALSE(parent.ref_lives.contains(unmatched_life)); + parent.ref_lives.emplace(unmatched_life, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{9, 9}}}); + ASSERT_EQ( + backend->putOverwrite(parent_seal_key, encodeFoldSeal(parent), parent_object->token).outcome, + PutOutcome::Done); + + dropRefTransition(*backend, layout, ns, "victim", mref); + const uint64_t events_before = + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load(); + const RoundReport report = runRegularRoundReclaiming(gc); + + ASSERT_TRUE(report.acquired_lease); + EXPECT_EQ( + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load() - events_before, + 1u); + EXPECT_EQ(report.manifests_deleted, 1u) + << "an unmatched adopted-parent row is observed and dropped, not promoted to pool-wide suppression"; + EXPECT_FALSE(backend->head(layout.manifestKey(manifest_id)).exists) + << "the valid manifest candidate must be physically deleted by the same authoritative round"; + + const GcState after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal successor = decodeFoldSeal( + backend->get(layout.foldSealKey(after.snap_generation, after.snap_attempt))->bytes, + after.snap_generation); + EXPECT_FALSE(successor.ref_lives.contains(unmatched_life)); +} + +TEST(CASCatalogLifecycleReconciler, EmptyCatalogReturnsAuthoritativeCompleteCut) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + ASSERT_TRUE(CasRefCatalog::initializeEmptyForNewPool(*backend, layout).catalog.entries.empty()); + + CasFoldSeal parent; + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [](uint64_t) + { + return CasRefCatalog::LeaderFenceStatus::Held; + }); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); + EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); + ASSERT_TRUE(result.final_catalog_cut); + EXPECT_TRUE(result.final_catalog_cut->catalog.entries.empty()); + EXPECT_TRUE(result.retired_lives.empty()); + EXPECT_EQ(result.deleted, 0); +} + +TEST(CASCatalogLifecycleReconciler, DeletesEligibleRowsFromReturnedResolutionCuts) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + constexpr size_t deletes = 3; + seedCompletedRemovingBatch(*backend, store, kGc, deletes); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + backend->clearJournal(); + backend->resetCounts(); + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [](uint64_t) + { + return CasRefCatalog::LeaderFenceStatus::Held; + }); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); + EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); + EXPECT_EQ(result.deleted, deletes); + ASSERT_EQ(result.retired_lives.size(), deletes); + ASSERT_TRUE(result.final_catalog_cut); + EXPECT_TRUE(result.final_catalog_cut->catalog.entries.empty()); + const std::vector journal = backend->journalSnapshot(); + const String catalog_get = "get " + layout.refCatalogKey(); + EXPECT_EQ(std::count(journal.begin(), journal.end(), catalog_get), deletes + 1); +} + +TEST(CASCatalogLifecycleReconciler, ReturnsRetiredLifeWhenAuthorityMovesAfterResolution) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + size_t fence_checks = 0; + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [&fence_checks](uint64_t) + { + ++fence_checks; + return fence_checks == 2 + ? CasRefCatalog::LeaderFenceStatus::Moved + : CasRefCatalog::LeaderFenceStatus::Held; + }); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + EXPECT_EQ(result.authority_status, AuthorityStatus::FencedOut); + EXPECT_EQ(result.catalog_resolution, CatalogResolution::ExactRowAbsent); + ASSERT_EQ(result.retired_lives.size(), 1); + EXPECT_EQ(result.retired_lives.front(), + NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); + EXPECT_EQ(result.deleted, 0); + EXPECT_FALSE(result.final_catalog_cut); +} + +TEST(CASCatalogLifecycleReconciler, InitialFenceLossReportsEligibleRowStillPresent) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + backend->resetCounts(); + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [](uint64_t) + { + return CasRefCatalog::LeaderFenceStatus::Moved; + }); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + EXPECT_EQ(result.authority_status, AuthorityStatus::FencedOut); + EXPECT_EQ(result.catalog_resolution, CatalogResolution::ExactRowStillPresent); + EXPECT_TRUE(result.retired_lives.empty()); + EXPECT_EQ(result.deleted, 0); + EXPECT_FALSE(result.final_catalog_cut); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 2) + << "the initial selection and mandatory erase-resolution cuts are the only catalog reads"; + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0); + EXPECT_EQ(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns), + NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); +} + +TEST(CASCatalogLifecycleReconciler, RetriesFromTheMandatoryConflictResolutionCut) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + seedCompletedRemoving(*backend, store, kGc); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + backend->clearJournal(); + backend->resetCounts(); + backend->conflictNextCatalogCas(layout.refCatalogKey()); + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [](uint64_t) + { + return CasRefCatalog::LeaderFenceStatus::Held; + }); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); + EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); + EXPECT_EQ(result.deleted, 1); + const std::vector journal = backend->journalSnapshot(); + const String catalog_get = "get " + layout.refCatalogKey(); + EXPECT_EQ(std::count(journal.begin(), journal.end(), catalog_get), 3) + << "the token-conflict retry must reuse its mandatory resolution cut"; +} + +TEST(CASCatalogLifecycleReconciler, PropagatesAuthorityFailureBeforeEraseCas) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + seedCompletedRemoving(*backend, store, kGc); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + size_t fence_checks = 0; + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [&fence_checks](uint64_t) + { + if (++fence_checks == 2) + throw std::runtime_error("injected reconciler authority failure before CAS"); + return CasRefCatalog::LeaderFenceStatus::Held; + }); + try + { + (void)reconciler.reconcile(); + FAIL() << "the authority exception must propagate"; + } + catch (const std::runtime_error & e) + { + EXPECT_STREQ(e.what(), "injected reconciler authority failure before CAS"); + } +} + +TEST(CASCatalogLifecycleReconciler, PropagatesAuthorityFailureAfterMandatoryResolution) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + ASSERT_TRUE(parent_object); + const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); + size_t fence_checks = 0; + + CatalogLifecycleReconciler reconciler( + *backend, + layout, + parent, + /*admitted_generation=*/1, + [&fence_checks](uint64_t) + { + if (++fence_checks == 3) + throw std::runtime_error("injected reconciler authority failure after resolution"); + return CasRefCatalog::LeaderFenceStatus::Held; + }); + try + { + (void)reconciler.reconcile(); + FAIL() << "the authority exception must propagate"; + } + catch (const std::runtime_error & e) + { + EXPECT_STREQ(e.what(), "injected reconciler authority failure after resolution"); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + } +} + +TEST(CASGCFrontierGate, HealthyRebuildUsesTheCatalogLifecycleReconciler) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + + Gc gc(store, kGc); + const RebuildReport result = gc.rebuildBaseline(/*force=*/true); + + EXPECT_TRUE(result.performed); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before + 1); +} + +TEST(CASGCFrontierGate, DamagedStateRebuildDoesNotDeleteCompletedRemovingRows) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/damaged-rebuild-removing@cas@"}; + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + .ns = ns, .state = NsState::Live, .incarnation = UInt128{901}}); + CasRefCatalog::casUpdate(*backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + next.entries.front().state = NsState::Removing; + next.entries.front().removal_started_round = 1; + return next; + }); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + + Gc gc(store, kGc); + const RebuildReport result = gc.rebuildBaseline(/*force=*/false); + + EXPECT_TRUE(result.performed); + EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before); +} + +TEST(CASGCFrontierGate, DeferredRoundDrainsCompletedRemovingBeforeReturning) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/100); + const Layout & layout = store->layout(); + const RootNamespace removed{"00/deferred-removed@cas@"}; + const UInt128 life_id{77}; + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + .ns = removed, .state = NsState::Live, .incarnation = life_id}); + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + + CasFoldSeal parent; + parent.generation = 1; + parent.ref_lives.emplace(life_id, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); + for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) + parent.condemned_summary.emplace(shard, CondemnedSummary{}); + ASSERT_EQ(backend->putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)).outcome, PutOutcome::Done); + GcState state; + state.round = 1; + state.gc_shards = store->poolConfig().gc_shards; + state.snap_generation = 1; + state.snap_attempt = 1; + state.lease = GcLease{.owner = kGc, .seq = 1}; + ASSERT_EQ(backend->putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); + + const String ckpt_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(removed, life_id)); + ASSERT_EQ(backend->putIfAbsent(ckpt_key, "inert checkpoint debris").outcome, PutOutcome::Done); + const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + + Gc gc(store, kGc); + const RoundReport report = runRegularRoundReclaiming(gc); + ASSERT_TRUE(report.acquired_lease); + EXPECT_TRUE(report.deferred); + EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before + 1); + EXPECT_TRUE(backend->head(ckpt_key).exists); + EXPECT_EQ(backend->deleteCount(ckpt_key), 0); +} + +TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const UInt128 leader_b = hexToU128("00000000000000000000000000000002"); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + backend->clearJournal(); + backend->blockNextCatalogCas(layout.refCatalogKey()); + + std::exception_ptr leader_a_failure; + std::thread leader_a([&] + { + try + { + Gc gc_a(store, kGc); + (void)runRegularRoundReclaiming(gc_a); + } + catch (...) + { + leader_a_failure = std::current_exception(); + } + }); + backend->waitForBlockedCatalogCas(); + + transferGcLease(*backend, layout, leader_b); + RoundReport report_b; + std::exception_ptr leader_b_failure; + /// `fixture.ns` is Removing with a durable `_ckpt`, same shape as + /// `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`: once leader_b's round drops its + /// catalog row, the resulting cut is genuinely, provably empty, the destructive gate opens, and the + /// namespace janitor -- a separate `namespace_cleanup` phase within this SAME round -- reclaims the + /// checkpoint. Captured so the assertions below can attribute the delete to the janitor rather than + /// assume survival. + std::map janitor_metrics_b; + try + { + Gc gc_b(store, leader_b); + gc_b.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "namespace_cleanup") + janitor_metrics_b = rec.metrics; + }); + report_b = runRegularRoundReclaiming(gc_b); + gc_b.setPhaseSink({}); + } + catch (...) + { + leader_b_failure = std::current_exception(); + } + + const std::vector before_a_release = backend->journalSnapshot(); + backend->releaseBlockedCatalogCas(); + leader_a.join(); + + ASSERT_FALSE(leader_b_failure); + ASSERT_TRUE(report_b.acquired_lease); + ASSERT_FALSE(report_b.deferred); + ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + + const size_t catalog_cas_end = findJournalAfter(before_a_release, "cas_end " + layout.refCatalogKey(), 0); + ASSERT_LT(catalog_cas_end, before_a_release.size()); + const size_t conclusive_rescan = findJournalAfter( + before_a_release, "get " + layout.refCatalogKey(), catalog_cas_end + 1); + ASSERT_LT(conclusive_rescan, before_a_release.size()); + const size_t stream_list = findJournalAfter( + before_a_release, "list " + layout.casRefsPrefix(), conclusive_rescan + 1); + ASSERT_LT(stream_list, before_a_release.size()); + const size_t fresh_catalog_cut = findJournalAfter( + before_a_release, "get " + layout.refCatalogKey(), stream_list + 1); + ASSERT_LT(fresh_catalog_cut, before_a_release.size()); + const GcState adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const String successor_seal_key = layout.foldSealKey(adopted.snap_generation, adopted.snap_attempt); + const size_t successor_seal_put = findJournalAfter( + before_a_release, "put_end " + successor_seal_key, fresh_catalog_cut + 1); + ASSERT_LT(successor_seal_put, before_a_release.size()); + const size_t successor_adoption = findJournalAfter( + before_a_release, "cas_end " + layout.gcStateKey(), successor_seal_put + 1); + ASSERT_LT(successor_adoption, before_a_release.size()); + EXPECT_LT(catalog_cas_end, conclusive_rescan); + EXPECT_LT(conclusive_rescan, stream_list); + EXPECT_LT(stream_list, fresh_catalog_cut); + /// The invariant this ordering must still prove: the fold's OWN walk plan is built from the single + /// hot-scan cut, taken immediately after the ref-object LIST, with no earlier catalog read sneaking + /// into that construction. `fresh_catalog_cut` is defined as the FIRST catalog `get` after + /// `stream_list` (the `findJournalAfter` search above), so that already holds by construction -- + /// the walk plan physically cannot have consumed an earlier one. + /// + /// What this test used to also assert -- no SECOND catalog read anywhere before the seal PUT -- is + /// no longer the right claim once the destructive gate can open on a proved-empty cut: other + /// destructive families this SAME round now also runs (the orphan-manifest sweep, the namespace + /// janitor) take their OWN separate catalog cuts by design, each after its own candidate listing, + /// to resolve authority against a fresh read rather than the fold's frozen one -- exactly the shape + /// measured here (`list p/cas/manifests/` immediately followed by a second `get + /// p/cas/ref_catalog`, before the seal PUT, from the orphan sweep). That is expected, not redundant, + /// so it is not asserted against; the fold's own single-cut plan construction is what remains pinned. + EXPECT_LT(fresh_catalog_cut, successor_seal_put); + EXPECT_LT(successor_seal_put, successor_adoption); + + ASSERT_TRUE(leader_a_failure); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + /// Same discrimination as `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`: leader_b's + /// round both drops `fixture.ns`'s catalog row AND, because the resulting cut is genuinely, + /// provably empty, opens the destructive gate -- so the namespace janitor reclaims the checkpoint + /// in this SAME round. Attribute the delete to the janitor's own counter rather than assume either + /// survival (the old expectation) or absence (which an unchecked dereference here cannot + /// distinguish from "never existed"). + ASSERT_FALSE(janitor_metrics_b.empty()) << "the namespace_cleanup phase must have run this round"; + EXPECT_GE(janitor_metrics_b.at("janitor_deleted"), 1u) + << "the janitor's OWN counter must show the delete, now that the proved-empty gate has opened"; + EXPECT_FALSE(backend->get(fixture.checkpoint_key).has_value()); + EXPECT_EQ(backend->deleteCount(fixture.checkpoint_key), 1); +} + +TEST(CASGCFrontierGate, LostCatalogCasResponseIsResolvedBeforeListing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + backend->clearJournal(); + backend->loseNextCatalogCasResponse(layout.refCatalogKey()); + + /// Same shape as `StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListing`: `fixture.ns` is + /// Removing with a durable `_ckpt`, so this round both drops its catalog row and, because the + /// resulting cut is genuinely, provably empty, opens the destructive gate -- the namespace janitor + /// (a separate `namespace_cleanup` phase within this SAME round) reclaims the checkpoint. Captured + /// so the assertions below attribute the delete to the janitor's own counter. + std::map janitor_metrics; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "namespace_cleanup") + janitor_metrics = rec.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + ASSERT_TRUE(report.acquired_lease); + ASSERT_FALSE(report.deferred); + + const std::vector journal = backend->journalSnapshot(); + const size_t response_lost = findJournalAfter( + journal, "cas_response_lost " + layout.refCatalogKey(), 0); + ASSERT_LT(response_lost, journal.size()); + const size_t conclusive_rescan = findJournalAfter( + journal, "get " + layout.refCatalogKey(), response_lost + 1); + ASSERT_LT(conclusive_rescan, journal.size()); + const size_t stream_list = findJournalAfter( + journal, "list " + layout.casRefsPrefix(), conclusive_rescan + 1); + ASSERT_LT(stream_list, journal.size()); + const size_t fresh_catalog_cut = findJournalAfter( + journal, "get " + layout.refCatalogKey(), stream_list + 1); + ASSERT_LT(fresh_catalog_cut, journal.size()); + EXPECT_LT(response_lost, conclusive_rescan); + EXPECT_LT(conclusive_rescan, stream_list); + EXPECT_LT(stream_list, fresh_catalog_cut); + + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + /// See the discrimination comment in `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`: + /// attribute the delete to the janitor's own counter, never to end-state absence alone, and never + /// assume survival -- both would be indistinguishable from a bug on this exact line (the old + /// unchecked `->bytes` here is what aborted the whole binary once the janitor started reclaiming). + ASSERT_FALSE(janitor_metrics.empty()) << "the namespace_cleanup phase must have run this round"; + EXPECT_GE(janitor_metrics.at("janitor_deleted"), 1u) + << "the janitor's OWN counter must show the delete, now that the proved-empty gate has opened"; + EXPECT_FALSE(backend->get(fixture.checkpoint_key).has_value()); + EXPECT_EQ(backend->deleteCount(fixture.checkpoint_key), 1); +} + +/// A stale leader may learn from its mandatory resolution read that the old life is gone, and must +/// invalidate that exact runtime, but loss of the leader fence remains the control outcome. It must +/// abort before the hot LIST and cannot build or publish any successor generation. +TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrReplacesLife) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const UInt128 leader_b = hexToU128("00000000000000000000000000000002"); + const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id); + ASSERT_TRUE(store->refTableRecoveredForTest(fixture.ns)) + << "the fixture must retain a resident predecessor runtime before removal"; + ASSERT_EQ(store->refTableLifeForTest(fixture.ns), predecessor_life); + const uint64_t predecessor_runtime = store->refTableRuntimeIdentityForTest(fixture.ns); + ASSERT_NE(predecessor_runtime, 0u); + backend->clearJournal(); + backend->blockNextCatalogCas(layout.refCatalogKey()); + + std::exception_ptr leader_a_failure; + std::thread leader_a([&] + { + try + { + Gc gc_a(store, kGc); + (void)runRegularRoundReclaiming(gc_a); + } + catch (...) + { + leader_a_failure = std::current_exception(); + } + }); + backend->waitForBlockedCatalogCas(); + + transferGcLease(*backend, layout, leader_b); + const CasRefCatalog::Snapshot observed = CasRefCatalog::read(*backend, layout); + RefCatalog winner_catalog; + if (GetParam() == CompetingCatalogOutcome::Replacement) + { + winner_catalog.entries.push_back(CatalogEntry{ + .ns = fixture.ns, + .state = NsState::Live, + .incarnation = UInt128{178}}); + /// Mirror production's publish-then-flip order: the successor life needs a readable `_ckpt` + /// before its catalog row can read `Live`, or `chooseRecoveryGrounding` rejects it. + const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(fixture.ns, UInt128{178}); + backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + })); + } + ASSERT_EQ(backend->casPut( + layout.refCatalogKey(), encodeRefCatalog(winner_catalog), observed.token).outcome, + CasOutcome::Committed); + + backend->clearJournal(); + const uint64_t plans_before /// NOLINT(clang-analyzer-deadcode.DeadStores) + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + backend->releaseBlockedCatalogCas(); + leader_a.join(); + + const std::vector journal = backend->journalSnapshot(); + ASSERT_TRUE(leader_a_failure); + try + { + std::rethrow_exception(leader_a_failure); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(e.message().find("pre-fold drain lost authority"), String::npos) << e.message(); + } + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 0u); + EXPECT_EQ(findJournalAfter(journal, "list " + layout.casRefsPrefix(), 0), journal.size()); + EXPECT_EQ(findJournalAfter(journal, "cas_begin " + layout.gcStateKey(), 0), journal.size()); + EXPECT_FALSE(std::any_of(journal.begin(), journal.end(), [](const String & entry) + { + return entry.starts_with("put_begin ") && entry.ends_with("/fold_seal"); + })); + EXPECT_LT(findJournalAfter(journal, "get " + layout.refCatalogKey(), 0), journal.size()) + << "the stale leader must still complete mandatory erase resolution"; + + (void)store->namespaceLife(fixture.ns); + EXPECT_NE(store->refTableRuntimeIdentityForTest(fixture.ns), 0u); + ASSERT_TRUE(store->refTableLifeForTest(fixture.ns)); + EXPECT_NE(store->refTableLifeForTest(fixture.ns), predecessor_life) + << "the next name-based resolution must not retain the retired predecessor life"; +} + +INSTANTIATE_TEST_SUITE_P( + CASWinnerShape, + CASGCCompletedRemovalFenceRace, + testing::Values(CompetingCatalogOutcome::Absent, CompetingCatalogOutcome::Replacement), + [](const testing::TestParamInfo & parameter) + { + return parameter.param == CompetingCatalogOutcome::Absent ? "Absent" : "Replacement"; + }); + +/// One initial full catalog read selects the first row; each successful erase's mandatory resolution +/// read becomes the next selection snapshot. Therefore N uncontended deletes cost N+1 reads before +/// the hot LIST. The round then takes one post-LIST walk-plan cut and, later in the separate +/// `namespace_cleanup` phase, one post-page janitor cut. +TEST(CASGCFrontierGate, CompletedRemovalDrainUsesNPlusOneCatalogReads) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + constexpr size_t deletes = 3; + seedCompletedRemovingBatch(*backend, store, kGc, deletes); + backend->clearJournal(); + backend->resetCounts(); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const std::vector journal = backend->journalSnapshot(); + const size_t stream_list = findJournalAfter(journal, "list " + layout.casRefsPrefix(), 0); + ASSERT_LT(stream_list, journal.size()); + const String catalog_get = "get " + layout.refCatalogKey(); + EXPECT_EQ(std::count(journal.begin(), journal.begin() + static_cast(stream_list), catalog_get), + deletes + 1); + const size_t walk_plan_cut = findJournalAfter(journal, catalog_get, stream_list); + ASSERT_LT(walk_plan_cut, journal.size()); + /// Between the hot walk-plan cut and the janitor's own page, the orphan-manifest sweep -- ANOTHER + /// destructive family the now-open gate also unlocks (the batch drain above empties the catalog, so + /// this round's frontier is proved empty and the sweep's own `!suppress_destructive` gate opens + /// too) -- lists its own manifest candidates and takes its OWN separate catalog cut to resolve + /// authority, exactly as the janitor does. Located explicitly so the final read count below states + /// what it counts rather than drifting silently the next time a family is unlocked. + const size_t orphan_sweep_list = findJournalAfter(journal, "list " + layout.casManifestsPrefix(), walk_plan_cut); + ASSERT_LT(orphan_sweep_list, journal.size()); + const size_t orphan_sweep_cut = findJournalAfter(journal, catalog_get, orphan_sweep_list); + ASSERT_LT(orphan_sweep_cut, journal.size()); + const size_t janitor_list + = findJournalAfter(journal, "list " + layout.namespaceRootPrefix(), orphan_sweep_cut); + ASSERT_LT(janitor_list, journal.size()); + const size_t janitor_cut = findJournalAfter(journal, catalog_get, janitor_list); + ASSERT_LT(janitor_cut, journal.size()); + EXPECT_EQ(findJournalAfter(journal, catalog_get, janitor_cut + 1), journal.size()) + << "one hot walk-plan cut, one orphan-sweep cut, and one janitor page cut are the only " + "post-drain catalog reads"; + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u); + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), deletes + 4) + << "N+1 drain reads, one post-hot-LIST walk-plan cut, one orphan-manifest-sweep cut (now that " + "the proved-empty gate has opened, unlocking that destructive family too), and one separate " + "post-janitor-page cut"; +} diff --git a/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp new file mode 100644 index 000000000000..19a63bc6842a --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp @@ -0,0 +1,1590 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +/// DURABLE HOLDS (spec 2026-07-27 "ref chain complete cut" §5). +/// +/// A namespace whose ref-log walk meets an IMPOSSIBLE shape stops there, and that stop has to survive +/// the round. Before this task the stop was a single bit — `classification == 4` — and everything that +/// explained it (what went wrong, and exactly WHERE) lived in a log line and an in-memory anomaly, both +/// gone by the next round. That is not enough for three separate reasons: +/// +/// * the next round could not RETRY the exact position, so a hold only survived while the round's +/// hint happened to keep mentioning the namespace; +/// * the hold could be cleared by an ABSENT — precisely the observation a lying store produces, and +/// precisely the shape that made the hold necessary in the first place; +/// * REBUILD rewrote coverage from owner state and silently dropped every hold, handing back a +/// baseline that looked proven when it was not. +/// +/// So the hold is now DURABLE and STRICTLY GRAMMARED: `{reason, offending_position, retry_count, +/// next_retry_round}` present if and only if `classification == 4`, rejected in both directions +/// otherwise. It rides the seal across rounds — including rounds whose hint omits the namespace +/// entirely — and across REBUILD, and it clears by exactly ONE event: the fold resolving the offending +/// position and that result being adopted in `gc/state`. +/// +/// The carried hold is also a WITNESS, and a better one than the listing: it is durable proof that the +/// walk once reached that position, so an absent below it is a gap rather than a frontier no matter +/// what the hint says this round. That is what makes "retry the exact offending position" work for a +/// hold that sits above an epoch boundary. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int LIMIT_EXCEEDED; +extern const int LOGICAL_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASGCRebuildVirginByEnumeration; +} + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +/// ===================== FIXTURES ===================== + +/// A backend that hides keys from every LIST while serving them by exact key (the observed lying-store +/// shape) AND counts reads. The hold tests need both: the hint has to go quiet while the exact GET the +/// hold forces stays observable. +class HintHoleCountingBackend : public CountingBackend +{ +public: + void hide(const String & key) + { + std::lock_guard lock(m); + hidden.insert(key); + } + + size_t holesServed() const + { + std::lock_guard lock(m); + return served; + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = CountingBackend::list(prefix, cursor, limit); + std::lock_guard lock(m); + if (hidden.empty()) + return page; + const size_t before = page.keys.size(); + std::erase_if(page.keys, [&](const ListedKey & k) { return hidden.contains(k.key); }); + if (page.keys.size() != before) + ++served; + return page; + } + +private: + mutable std::mutex m; + std::set hidden; + size_t served = 0; +}; + +/// Write the namespace's `_ckpt` naming `checkpoint` as its snapshot base, through the real codec — the +/// fold's second witness source is a decode of exactly these bytes, so a hand-rolled body would prove +/// nothing about the object the writers actually publish. +void writeCkptAt( + Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & checkpoint) +{ + writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = checkpoint, + .checkpoint_snapshot_id = checkpoint, + .last_epoch_seal = std::nullopt, + }); +} + +/// Establish only the immutable recovery frontier for a raw-log fixture. Unlike `writeCkptAt`, this +/// does not claim a snapshot exists: rebuild tests need to replay the log through this exact position. +void writeCommittedCkptAt( + Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & committed_through) +{ + writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = committed_through, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); +} + +/// The newest fold seal, scanning downward from the adopted generation (a completed round's gc/state +/// points at the recheck generation). +std::optional newestSeal(Backend & backend, const Layout & layout) +{ + const uint64_t gen = currentGenerationOf(backend, layout); + const uint64_t attempt = currentAttemptOf(backend, layout); + for (uint64_t g = gen; ; --g) + { + if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + return decodeFoldSeal(got->bytes); + if (g == 0) + return std::nullopt; + } +} + +std::optional coverageOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto seal = newestSeal(backend, layout); + if (!seal) + return std::nullopt; + const auto it = seal->ref_lives.find(catalogLifeIdForTest(backend, layout, ns)); + if (it == seal->ref_lives.end()) + return std::nullopt; + return it->second.coverage; +} + +/// The cursor `ns` was sealed at, or `{0, 0}` when the round sealed NO row for it at all. It never +/// dereferences a disengaged optional: a test that aborts the process takes every test after it in the +/// binary down with it, and "there is no coverage row" is exactly the shape a regression in the hold +/// carry produces — so it has to read as a failed expectation, not as a crash that hides the rest. +RefTxnId sealedCursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto cov = coverageOf(backend, layout, ns); + EXPECT_TRUE(cov.has_value()) << "no coverage row for " << ns.string(); + return cov ? cov->last_folded_ref_id : RefTxnId{}; +} + +/// The coverage row a round MUST have sealed for `ns`, held. Fails the test rather than returning an +/// empty optional, so every caller below reads a real hold. +RefHold holdOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const auto cov = coverageOf(backend, layout, ns); + EXPECT_TRUE(cov.has_value()) << "no coverage row for " << ns.string(); + if (!cov) + return RefHold{}; + EXPECT_EQ(cov->classification, 4) << "a held namespace is classification 4"; + EXPECT_TRUE(cov->hold.has_value()) << "classification 4 without a hold is the forbidden shape"; + return cov->hold ? *cov->hold : RefHold{}; +} + +UInt128 fixtureLifeId(std::string_view key) +{ + return key.ends_with("/1") ? UInt128{2} : UInt128{1}; +} + +RefCoverage & fixtureCoverage(CasFoldSeal & seal, std::string_view key) +{ + return seal.ref_lives[fixtureLifeId(key)].coverage; +} + +const RefCoverage & fixtureCoverage(const CasFoldSeal & seal, std::string_view key) +{ + return seal.ref_lives.at(fixtureLifeId(key)).coverage; +} + +/// A seal carrying exactly one held coverage row with every numeric at its maximum and a coverage key +/// that needs escaping — the widest row the per-row line budget has to survive. The caller supplies the +/// key, which is what the line-cap tests below grow byte by byte. +CasFoldSeal maximalHoldSeal(const String & map_key) +{ + CasFoldSeal seal; + seal.generation = std::numeric_limits::max(); + seal.parent_generation = std::numeric_limits::max(); + RefCoverage cov; + cov.classification = 4; + cov.last_folded_ref_id = RefTxnId{std::numeric_limits::max(), + std::numeric_limits::max()}; + cov.hold = RefHold{.reason = HoldReason::UnconsumedSealCrossing, /// the longest reason word + .offending_position = RefTxnId{std::numeric_limits::max(), + std::numeric_limits::max()}, + .retry_count = std::numeric_limits::max(), + .next_retry_round = std::numeric_limits::max()}; + fixtureCoverage(seal, map_key) = cov; + return seal; +} + +/// The `cov` line of an encoded seal (line 3: header, meta, then the single record). +String covLineOf(const String & encoded) +{ + size_t begin = encoded.find('\n') + 1; /// past the header + begin = encoded.find('\n', begin) + 1; /// past the meta line + return encoded.substr(begin, encoded.find('\n', begin) - begin); +} + +/// A one-row seal whose coverage is ordinary and CLEAN: folded through its cursor, nothing held. +CasFoldSeal cleanSeal(const String & map_key) +{ + CasFoldSeal seal; + seal.generation = 3; + seal.parent_generation = 2; + RefCoverage cov; + cov.classification = 2; + cov.last_folded_ref_id = RefTxnId{4, 5}; + fixtureCoverage(seal, map_key) = cov; + return seal; +} + +/// A one-row seal whose coverage is HELD at an exact position — the row every erasure shape below is +/// trying to make disappear. +CasFoldSeal heldSeal(const String & map_key) +{ + CasFoldSeal seal = cleanSeal(map_key); + RefCoverage & cov = fixtureCoverage(seal, map_key); + cov.classification = 4; + cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{4, 6}, + .retry_count = 7, .next_retry_round = 99}; + return seal; +} + +/// The header and meta lines (1 and 2) of an encoded seal, terminators included. +String headerAndMetaOf(const String & encoded) +{ + const size_t past_meta = encoded.find('\n', encoded.find('\n') + 1) + 1; + EXPECT_NE(past_meta, 0u); + return encoded.substr(0, past_meta); +} + +/// Assemble a raw seal object from `records` (one record per element, no terminators), on `prototype`'s +/// header and meta lines, closed by the trailer count those records imply. This is the ONLY way to put a +/// repeated record key on the wire: `CasFoldSeal` stores keyed maps, so a duplicate is not a value any +/// producer can hold — it is a shape a forged, truncated, or mis-merged object has. +String sealTextWith(const String & prototype, const std::vector & records) +{ + String text = headerAndMetaOf(prototype); + for (const String & record : records) + text += record + "\n"; + return text + "{\"n\":" + std::to_string(records.size()) + "}\n"; +} + +/// Replace the coverage row's `cls` value with `raw`, VERBATIM. The point is to write integers no +/// `RefCoverage` can hold: the field is a byte in the struct, so a wide value exists only on the wire, +/// which is exactly where a reader has to catch it. `cls` is never the last field of a `cov` record, so +/// the value always ends at a comma. +String withRawClassification(const String & encoded, std::string_view raw) +{ + const size_t at = encoded.find("\"cls\":"); + EXPECT_NE(at, String::npos); + const size_t begin = at + strlen("\"cls\":"); + const size_t end = encoded.find(',', begin); + EXPECT_NE(end, String::npos); + return encoded.substr(0, begin) + String{raw} + encoded.substr(end); +} + +/// Replace the FIRST occurrence of `field` with `replacement` (both are whole `"key":value` fragments), +/// so a test states the exact wire shape it is feeding the decoder. +String withField(const String & encoded, const String & field, const String & replacement) +{ + const size_t at = encoded.find(field); + EXPECT_NE(at, String::npos) << "the encoder does not emit " << field; + return encoded.substr(0, at) + replacement + encoded.substr(at + field.size()); +} + +/// Every coverage row the ENCODER must refuse, each paired with why producing it would be a bug in our +/// own fold rather than corruption arriving from a store. Shared by the two builds' assertions below so +/// the release expectation and the sanitizer death expectation can never drift apart. +std::vector> illFormedSealsTheEncoderMustRefuse() +{ + std::vector> out; + + /// The pairing, both ways round. + CasFoldSeal hold_on_folded = heldSeal("ns/0"); + fixtureCoverage(hold_on_folded, "ns/0").classification = 2; + out.emplace_back("a hold on a folded (2) row claims a stop that did not happen", hold_on_folded); + + CasFoldSeal clamped_without_hold = heldSeal("ns/0"); + fixtureCoverage(clamped_without_hold, "ns/0").hold.reset(); + out.emplace_back("a clamped (4) row with no hold is indistinguishable from a clean cursor once " + "durable", clamped_without_hold); + + /// The closed set. 3 is the dangerous one: it passes the sweep's `== 4` and `== 0` refusals and + /// reaches the deletion premise, which is a refusal written in terms of the set. + CasFoldSeal classification_three = cleanSeal("ns/0"); + fixtureCoverage(classification_three, "ns/0").classification = 3; + out.emplace_back("classification 3 is not one of {0,1,2,4} and passes every refusal stated in terms " + "of them", classification_three); + + CasFoldSeal classification_max = cleanSeal("ns/0"); + fixtureCoverage(classification_max, "ns/0").classification = 255; + out.emplace_back("classification 255 is not one of {0,1,2,4}", classification_max); + + /// The self-erasing hold, and its half-zero sibling. + CasFoldSeal hold_at_zero = heldSeal("ns/0"); + fixtureCoverage(hold_at_zero, "ns/0").hold->offending_position = RefTxnId{}; + out.emplace_back("a hold at {0,0} is cleared by the first record the next round folds", hold_at_zero); + + CasFoldSeal hold_zero_sequence = heldSeal("ns/0"); + fixtureCoverage(hold_zero_sequence, "ns/0").hold->offending_position = RefTxnId{7, 0}; + out.emplace_back("a hold position with a zero component is not a renderable id", hold_zero_sequence); + + return out; +} + +} + +/// ===================== THE SHARED BYTE ARITHMETIC ===================== +/// +/// Two caps, two predicates, one place they are computed. Stage B's catalog reuses THESE functions for +/// its additive "does one more entry still fit" question, so their boundary behaviour is pinned here +/// rather than re-derived per format: a cap is the largest PERMITTED value (equality fits), and every +/// sum saturates, because a wrapped sum answers "fits" for an object that does not — turning an +/// overflow into a durable object nothing can read. +TEST(CASGCHoldGrammarBudget, BothPredicatesAcceptEqualityAndRefuseOneMore) +{ + static_assert(fitsLineCap(64, 64)); + static_assert(!fitsLineCap(65, 64)); + static_assert(fitsObjectCap(40, 24, 64)); + static_assert(!fitsObjectCap(40, 25, 64)); + + EXPECT_TRUE(fitsLineCap(64, 64)); + EXPECT_FALSE(fitsLineCap(65, 64)); + EXPECT_TRUE(fitsObjectCap(64, 0, 64)); + EXPECT_FALSE(fitsObjectCap(64, 1, 64)); + + /// A cap of 0 means the format declares none (a streamed object never materialized whole). + EXPECT_TRUE(fitsLineCap(std::numeric_limits::max(), 0)); + EXPECT_TRUE(fitsObjectCap(std::numeric_limits::max(), 1, 0)); +} + +TEST(CASGCHoldGrammarBudget, SumsSaturateInsteadOfWrapping) +{ + constexpr uint64_t kMax = std::numeric_limits::max(); + static_assert(addByteBudget(kMax, 1) == kMax); + static_assert(addByteBudget(kMax, kMax) == kMax); + static_assert(addByteBudget(3, 4) == 7); + + /// The predicate that matters: a reservation that would wrap must REFUSE, not report a tiny sum. + EXPECT_FALSE(fitsObjectCap(kMax, 2, 256 * 1024 * 1024)); +} + +/// ===================== THE STRICT CLASSIFICATION-4 GRAMMAR ===================== + +TEST(CASGCHoldGrammar, EveryHoldReasonRoundTrips) +{ + for (const HoldReason reason : {HoldReason::GapBelowWitness, HoldReason::UnconsumedSealCrossing, + HoldReason::WitnessDisappeared, HoldReason::BodyUndecodable, + HoldReason::ManifestBodyMissing, HoldReason::CheckpointUndecodable}) + { + CasFoldSeal seal; + seal.generation = 3; + seal.parent_generation = 2; + RefCoverage cov; + cov.classification = 4; + cov.last_folded_ref_id = RefTxnId{4, 5}; + cov.hold = RefHold{.reason = reason, .offending_position = RefTxnId{4, 6}, + .retry_count = 7, .next_retry_round = 99}; + fixtureCoverage(seal, "ns/0") = cov; + + const CasFoldSeal back = decodeFoldSeal(encodeFoldSeal(seal)); + EXPECT_EQ(back, seal) << "hold reason " << static_cast(reason); + ASSERT_TRUE(fixtureCoverage(back, "ns/0").hold.has_value()); + EXPECT_EQ(fixtureCoverage(back, "ns/0").hold->reason, reason); + } +} + +/// THE ENCODER'S HALF OF THE GRAMMAR, in one place. Every shape here is OUR OWN fold handing the codec +/// a row it must never make durable, so the refusal is `LOGICAL_ERROR` — the code `encodeGcState` raises +/// for the same category of impossible input — and not the `CORRUPTED_DATA` reserved for bytes that +/// arrived from a store. Under a debug or sanitizer build that code ABORTS at construction +/// (`handle_error_code`), so the same table is asserted as a death expectation there; the contract +/// ("these bytes are never produced") is what both forms pin. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASGCHoldGrammar, TheEncoderRefusesEveryIllFormedCoverageRow) +{ + for (const auto & entry : illFormedSealsTheEncoderMustRefuse()) + { + SCOPED_TRACE(entry.first); + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeFoldSeal(entry.second); }); + } +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASGCHoldGrammarDeathTest, TheEncoderRefusesEveryIllFormedCoverageRow) +{ + for (const auto & entry : illFormedSealsTheEncoderMustRefuse()) + { + SCOPED_TRACE(entry.first); + EXPECT_DEATH({ (void)encodeFoldSeal(entry.second); }, ""); + } +} +#endif + +TEST(CASGCHoldGrammar, AHoldOnAnyOtherClassificationIsRefusedByTheDecoder) +{ + CasFoldSeal seal; + seal.generation = 1; + RefCoverage cov; + + /// Bytes some other producer wrote. Built by demoting a legitimate held row's classification, so the + /// hold fields are exactly the ones the encoder emits. + cov.classification = 4; + cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, + .retry_count = 0, .next_retry_round = 1}; + fixtureCoverage(seal, "ns/0") = cov; + String text = encodeFoldSeal(seal); + const size_t at = text.find("\"cls\":4"); + ASSERT_NE(at, String::npos); + text[at + 6] = '2'; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(text); }); +} + +TEST(CASGCHoldGrammar, ClassificationFourWithoutAHoldIsRefusedByTheDecoder) +{ + CasFoldSeal seal; + seal.generation = 1; + RefCoverage cov; + cov.classification = 4; + cov.last_folded_ref_id = RefTxnId{1, 1}; + + /// Every single hold field is REQUIRED: dropping any one of them is corruption, not a default. + cov.hold = RefHold{.reason = HoldReason::BodyUndecodable, .offending_position = RefTxnId{1, 2}, + .retry_count = 3, .next_retry_round = 4}; + fixtureCoverage(seal, "ns/0") = cov; + const String whole = encodeFoldSeal(seal); + for (const String & field : {String(R"("hr":"body_undecodable")"), String(R"("hpe":"1")"), + String(R"("hps":"2")"), String(R"("hrc":3)"), String(R"("hnr":"4")")}) + { + SCOPED_TRACE("without " + field); + const size_t at = whole.find(field); + ASSERT_NE(at, String::npos) << "the encoder does not emit " << field; + String without = whole; + without.erase(at - 1, field.size() + 1); /// the field and the ',' before it + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(without); }); + } +} + +TEST(CASGCHoldGrammar, DuplicateHoldKeyIsCorruptedData) +{ + CasFoldSeal seal; + seal.generation = 1; + RefCoverage cov; + cov.classification = 4; + cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, + .retry_count = 0, .next_retry_round = 5}; + fixtureCoverage(seal, "ns/0") = cov; + + const String whole = encodeFoldSeal(seal); + const String field = R"("hr":"gap_below_witness")"; + const size_t at = whole.find(field); + ASSERT_NE(at, String::npos); + /// The same key twice, with a DIFFERENT value: last-wins would silently rewrite the reason. + String doubled = whole; + doubled.insert(at, R"("hr":"witness_disappeared",)"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(doubled); }); +} + +TEST(CASGCHoldGrammar, UnknownHoldReasonWordIsCorruptedData) +{ + CasFoldSeal seal; + seal.generation = 1; + RefCoverage cov; + cov.classification = 4; + cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, + .retry_count = 0, .next_retry_round = 5}; + fixtureCoverage(seal, "ns/0") = cov; + + String text = encodeFoldSeal(seal); + const size_t at = text.find("gap_below_witness"); + ASSERT_NE(at, String::npos); + text.replace(at, strlen("gap_below_witness"), "gap_below_witnesX"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(text); }); +} + +/// ===================== THE THREE WAYS A SEAL CAN ERASE A HOLD ===================== +/// +/// The three shapes below are one finding, and it is about what a fold seal is FOR. The hold is the only +/// durable record that a namespace stopped and where; everything downstream reads the seal and nothing +/// re-derives the stop. So a seal that decodes into "no hold here" is not a lossy read, it is a licence +/// to delete: the sweep's §6 refusals are stated as `classification == 4` / `== 0` / `hold.has_value()`, +/// and a row that slips past all three reaches an irreversible delete of a manifest the fold never +/// accounted for. Each shape gets past a DIFFERENT one of the decoder's checks, which is why they are +/// pinned separately rather than as one "malformed seal" case. + +/// (1) The classification the reader never sees. `cls` is narrowed to a byte, so an integer on the wire +/// is truncated first and validated (if at all) afterwards: 258 becomes 2, "everything through the +/// cursor was folded". The value has to be judged WIDE, before the narrowing, or the wire can buy +/// coverage that no fold ever performed. +TEST(CASGCHoldGrammar, AClassificationOutsideTheGrammarIsCorruptedData) +{ + const String clean = encodeFoldSeal(cleanSeal("ns/0")); + ASSERT_EQ(fixtureCoverage(decodeFoldSeal(clean), "ns/0").classification, 2) + << "the unmodified row is the one every case below deviates from"; + + /// In-range bytes that are simply not classifications. 3 is the one the sweep's refusals miss. + for (const std::string_view raw : {"3", "5", "6", "255"}) + { + SCOPED_TRACE(String{"cls="} + String{raw}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withRawClassification(clean, raw)); }); + } + + /// Wide integers whose LOW BYTE lands inside the grammar: 258 -> 2 (fully folded), 256 -> 0 + /// (absent), 260 -> 4 (clamped). Each would decode as a row the fold never wrote. + for (const std::string_view raw : {"256", "258", "260", "18446744073709551615"}) + { + SCOPED_TRACE(String{"cls="} + String{raw}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withRawClassification(clean, raw)); }); + } +} + +/// And the field itself is required: an absent `cls` reads as 0, which is not "nothing was said about +/// this namespace" but the positive claim "no round folded it". +TEST(CASGCHoldGrammar, ACoverageRowWithoutAClassificationIsCorruptedData) +{ + const String clean = encodeFoldSeal(cleanSeal("ns/0")); + const size_t at = clean.find("\"cls\":2,"); + ASSERT_NE(at, String::npos); + String without = clean; + without.erase(at, strlen("\"cls\":2,")); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(without); }); +} + +/// (2) The hold that clears itself. `{0,0}` passes the completeness check — every field is present — and +/// then the carry rule drops it on the next round, because a hold rides forward only while the walk +/// stops BELOW its position and nothing is below zero. The namespace advances with no record that it was +/// ever held. A zero in EITHER component is the same defect, and is additionally unnameable: the sweep +/// renders the position when it reports what it retained, and `renderRefTxnId` refuses a zero component. +TEST(CASGCHoldGrammar, AHoldWhoseOffendingPositionHasAZeroComponentIsCorruptedData) +{ + const String held = encodeFoldSeal(heldSeal("ns/0")); + ASSERT_TRUE(fixtureCoverage(decodeFoldSeal(held), "ns/0").hold.has_value()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + decodeFoldSeal(withField(withField(held, R"("hpe":"4")", R"("hpe":"0")"), + R"("hps":"6")", R"("hps":"0")")); + }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withField(held, R"("hpe":"4")", R"("hpe":"0")")); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withField(held, R"("hps":"6")", R"("hps":"0")")); }); +} + +/// (3) The duplicate row. Two `cov` records for the same (namespace, shard) — held first, clean second — +/// used to be accepted with last-wins, so a single appended line erased a hold without touching the one +/// that recorded it. There is exactly one row per key, and a second one is corruption. +TEST(CASGCHoldGrammar, ASecondCoverageRowForTheSameKeyIsCorruptedData) +{ + const String held_line = covLineOf(encodeFoldSeal(heldSeal("ns/0"))); + const String clean_line = covLineOf(encodeFoldSeal(cleanSeal("ns/0"))); + const String other_clean_line = covLineOf(encodeFoldSeal(cleanSeal("ns/1"))); + const String prototype = encodeFoldSeal(heldSeal("ns/0")); + + /// The CONTROL first: the same two-record assembly with DIFFERENT keys decodes, so the refusal below + /// is about the repeated key and not about the way these bytes are forged. + const CasFoldSeal two_keys = decodeFoldSeal(sealTextWith(prototype, {held_line, other_clean_line})); + ASSERT_EQ(two_keys.ref_lives.size(), 2u); + ASSERT_TRUE(fixtureCoverage(two_keys, "ns/0").hold.has_value()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(sealTextWith(prototype, {held_line, clean_line})); }); + /// Order does not redeem it: a clean row followed by a held one is the same broken object. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(sealTextWith(prototype, {clean_line, held_line})); }); + /// Nor does repeating the identical row. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(sealTextWith(prototype, {held_line, held_line})); }); + + /// The ENCODER needs no matching check, and this is why: the seal stores keyed maps, so a second row + /// for a key is not a value any producer can construct — assigning it replaces the first. + CasFoldSeal seal = heldSeal("ns/0"); + fixtureCoverage(seal, "ns/0") = fixtureCoverage(cleanSeal("ns/0"), "ns/0"); + EXPECT_EQ(seal.ref_lives.size(), 1u); +} + +/// The same one-record-per-key rule applies to `cnd`: a repeated row rewrites a shard's condemned +/// totals, which graduation paces on. +TEST(CASGCHoldGrammar, ASecondCondemnedSummaryRecordIsCorruptedData) +{ + CasFoldSeal seal = cleanSeal("ns/0"); + seal.condemned_summary[0] = CondemnedSummary{.condemned_total = 5, .pending_total = 1, + .oldest_nonpending_condemn_round = 3}; + const String encoded = encodeFoldSeal(seal); + + /// Lines 3..4 are `rfl`, `cnd` in the encoder's fixed order. + std::vector lines; + for (size_t begin = headerAndMetaOf(encoded).size(); begin < encoded.size();) + { + const size_t end = encoded.find('\n', begin); + ASSERT_NE(end, String::npos); + lines.push_back(encoded.substr(begin, end - begin)); + begin = end + 1; + } + ASSERT_EQ(lines.size(), 3u) << "rfl, cnd and the trailer"; + const String ref_life_line = lines[0]; + const String cnd_line = lines[1]; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(sealTextWith(encoded, {ref_life_line, cnd_line, cnd_line})); }); + /// The unduplicated assembly is the control. + const std::vector one_of_each{ref_life_line, cnd_line}; + EXPECT_NO_THROW(decodeFoldSeal(sealTextWith(encoded, one_of_each))); +} + +/// Unified cleanup evidence still requires a canonical nonzero removal transaction id. Decoding +/// foreign bytes must fail this read, never the process. +TEST(CASGCHoldGrammar, CleanupEvidenceWithAZeroRemovalIdIsCorruptedData) +{ + CasFoldSeal seal = cleanSeal("ns/0"); + seal.ref_lives.at(fixtureLifeId("ns/0")).cleanup_evidence = + RefCleanupEvidence{.remove_txn_id = RefTxnId{2, 3}}; + const String encoded = encodeFoldSeal(seal); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withField(encoded, R"("rte":"2")", R"("rte":"0")")); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withField(encoded, R"("rts":"3")", R"("rts":"0")")); }); + /// Omitted entirely is the same thing: the fields default to zero. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeFoldSeal(withField(encoded, R"("rte":"2",)", "")); }); +} + +/// The OBJECT cap bounds the whole seal. Nothing on the fold-seal READ path enforces it (the seal +/// is read raw, never through `openObject`), so an oversized PUT would leave a durable seal that no +/// later round can decode — unrecoverable. The gate therefore sits before the bytes are handed out, and +/// equality is still accepted: the cap is the largest permitted size, not the first forbidden one. +TEST(CASGCHoldGrammar, ObjectCapAcceptsEqualityAndRefusesOneMoreByte) +{ + const uint64_t object_cap = foldSealCaps().object_cap; + ASSERT_EQ(object_cap, 256u * 1024 * 1024); + + EXPECT_NO_THROW(checkFoldSealObjectBytes(object_cap - 1)); + EXPECT_NO_THROW(checkFoldSealObjectBytes(object_cap)); + expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] { checkFoldSealObjectBytes(object_cap + 1); }); + + /// An ordinary seal is nowhere near it, so the gate costs a comparison and changes nothing. + EXPECT_NO_THROW(encodeFoldSeal(maximalHoldSeal("ns/0"))); +} + +/// ===================== HOLDS ARE CREATED WITH AN EXACT POSITION ===================== + +TEST(CASGCHoldGrammar, GapBelowWitnessNamesTheExactAbsentPosition) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + /// {1,3} never existed; {1,4} is durable AND listed, so the gap is impossible under contiguity. + publishAt(*backend, layout, ns, RefTxnId{1, 4}, "ref_4", 4, DB::UInt128(4)); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 4}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::GapBelowWitness); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 3})); + EXPECT_EQ(hold.retry_count, 0u) << "the round that creates a hold has retried nothing yet"; + EXPECT_GT(hold.next_retry_round, 0u); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); +} + +TEST(CASGCHoldGrammar, UnconsumedSealCrossingNamesTheAbsentPosition) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + /// Epoch 1 ends at {1,1} with NO seal, and epoch 2 chains to a seal at {1,3} that this cursor + /// never consumed (and that does not exist). The nearest witness above the absent {1,2} therefore + /// sits in another epoch, and the crossing has nothing to prove itself from. + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 3}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 3}, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::UnconsumedSealCrossing); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 2})) << "the hold names the position that read absent"; + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 0) + << "nothing beyond the unproven boundary may fold"; +} + +TEST(CASGCHoldGrammar, UndecodableBodyNamesTheRecordItCouldNotRead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + backend->putIfAbsent(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2}), "this is not a cas_ref_log object"); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 2}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::BodyUndecodable); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 2})); +} + +/// The fold barrier is a hold too, and it is the ONE hold whose ordinary cause is benign: a writer that +/// has appended its precommit record but not yet finished uploading the manifest body. It gets the same +/// durable treatment as the corruption shapes because it stops the namespace the same way — and because +/// a barrier that is durably named is one an operator can distinguish from a wedge. +TEST(CASGCHoldGrammar, MissingManifestBodyBarrierIsADurableHold) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + deleteManifestBody(*backend, layout, + ManifestId{ns, ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}}); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 2}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::ManifestBodyMissing); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 2})) << "the hold names the LOG whose edges could not fold"; + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); +} + +/// An above-cursor record that answered one GET and then stopped answering is CORRUPTION, not a +/// frontier: nothing may legitimately remove an object above the fold cursor. It is the one hold shape +/// that no amount of waiting can clear, and naming it durably is what stops a later round from reading +/// the same namespace as quiet and granting it a frontier proof. +TEST(CASGCHoldGrammar, AWitnessThatStopsAnsweringIsWitnessDisappeared) +{ + /// Answers on odd-numbered reads and 404s on even ones: `crossFromSeal` proves the position, and + /// the walk's own GET of it then fails. + class AlternatingGetBackend : public InMemoryBackend + { + public: + using DB::Cas::Backend::get; + String flaky; + size_t reads = 0; + + std::optional get(const String & key, Range range) override + { + if (key == flaky && ++reads % 2 == 0) + return std::nullopt; + return InMemoryBackend::get(key, range); + } + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeSealAt(*backend, layout, ns, RefTxnId{1, 2}); + publishAt(*backend, layout, ns, RefTxnId{2, 1}, "ref_2", 2, DB::UInt128(2), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + /// A third epoch keeps the unstable position from reading as a frontier. + publishAt(*backend, layout, ns, RefTxnId{3, 1}, "ref_3", 3, DB::UInt128(3), + /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{2, 1}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{3, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{2, 1}, + }); + backend->flaky = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{2, 1}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::WitnessDisappeared); + /// The walk crossed into epoch 2 on the record's first answer and then could not read it: the hold + /// names {2,1}, the position that stopped being readable, and the cursor stays on the seal below it. + EXPECT_EQ(hold.offending_position, (RefTxnId{2, 1})); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); +} + +/// ===================== THE SECOND WITNESS: `_ckpt.checkpoint` ===================== +/// +/// A listing is a SNAPSHOT: a record that became durable after the enumeration is invisible to that +/// round's probes, so an absent expected-next reads as a frontier when it is really a gap. The +/// namespace's own durable checkpoint decides the same question without asking the listing anything — +/// and this pair of pools is the proof, because they differ in nothing else. +TEST(CASGCHoldGrammar, CheckpointWitnessHoldsAGapTheHintIsSilentAbout) +{ + const RootNamespace ns{"00/aa@cas@"}; + /// Stage B (Task 4-C): no pin needed here -- `publishAt` below (draining into `writeRefLogTxnRaw`) + /// admits `ns` into the catalog itself, once per pool, inside each nested block's own `seed` call. + const auto seed = [&](HintHoleCountingBackend & backend, const Layout & layout) + { + publishAt(backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + /// {1,3} is missing and {1,4}, though durable, is invisible to every LIST. + publishAt(backend, layout, ns, RefTxnId{1, 4}, "ref_4", 4, DB::UInt128(4)); + backend.hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 4})); + }; + + /// Hint-only: nothing above {1,2} is visible, so the walk honestly reads a frontier and does not hold. + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + seed(*backend, store->layout()); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_GT(backend->holesServed(), 0u); + const auto cov = coverageOf(*backend, store->layout(), ns); + ASSERT_TRUE(cov.has_value()); + EXPECT_FALSE(cov->hold.has_value()) << "without a witness an absent IS the frontier"; + } + + /// Same pool, same hint, plus the checkpoint: the gap becomes decidable and holds at the same + /// position, with the same reason, as if the hint had shown the witness itself. + /// + /// The `_ckpt` object is hidden from every LIST as well, so the two pools' listings are byte-for-byte + /// the same and the only difference between them is an object reachable by EXACT KEY alone. That is + /// what makes this a proof of hint-INDEPENDENCE rather than of a richer hint. + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + seed(*backend, layout); + writeCkptAt(*backend, layout, ns, RefTxnId{1, 4}); + backend->hide(layout.refCkptKey(fixture::fixtureLife(ns))); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::GapBelowWitness); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 3})); + } +} + +/// The namespace whose second witness matters MOST: one the hint has stopped mentioning entirely, kept +/// in the round's universe by nothing but its CARRIED HOLD. Its checkpoint is why +/// `readCheckpointWitnesses` takes the parent cursors as well as the hint — the hold alone witnesses only +/// the position it stopped at, so a gap ABOVE that position, once the hold resolves, has no witness left. +TEST(CASGCHoldGrammar, CheckpointWitnessReachesAHeldNamespaceTheHintNoLongerNames) +{ + const RootNamespace ns{"00/aa@cas@"}; + /// Stage B (Task 4-C): no pin needed -- `publishAt` inside `seedPool` (draining into + /// `writeRefLogTxnRaw`) admits `ns` into each nested block's own pool. + + /// Round 1 in both pools: held at {1,3} by a gap below the listed witness {1,4}. Then the hint goes + /// silent about every one of the namespace's objects, {1,3} becomes readable (so the hold resolves and + /// the walk runs on), and a durable-but-unlisted {1,6} leaves a fresh gap at {1,5}. + const auto seedPool = [&](HintHoleCountingBackend & backend, const Layout & layout, Gc & gc) + { + publishAt(backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + publishAt(backend, layout, ns, RefTxnId{1, 4}, "ref_4", 4, DB::UInt128(4)); + writeCommittedCkptAt(backend, layout, ns, RefTxnId{1, 4}); + EXPECT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(holdOf(backend, layout, ns).offending_position, (RefTxnId{1, 3})); + + publishAt(backend, layout, ns, RefTxnId{1, 3}, "ref_3", 3, DB::UInt128(3)); + publishAt(backend, layout, ns, RefTxnId{1, 6}, "ref_6", 6, DB::UInt128(6)); + for (const RefTxnId & id : {RefTxnId{1, 1}, RefTxnId{1, 2}, RefTxnId{1, 3}, RefTxnId{1, 4}, + RefTxnId{1, 6}}) + backend.hide(layout.refLogKey(fixture::fixtureLife(ns), id)); + }; + + /// Hold-witness only: it witnesses {1,3}, which the walk has now passed, so the absent {1,5} above it + /// is an honest frontier and the namespace comes out clean. + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + Gc gc(store, kGc); + seedPool(*backend, store->layout(), gc); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const auto cov = coverageOf(*backend, store->layout(), ns); + ASSERT_TRUE(cov.has_value()); + EXPECT_FALSE(cov->hold.has_value()) << "a resolved hold witnesses nothing above itself"; + EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, 4})); + } + + /// Same pool, plus the checkpoint — read by exact key for a namespace THIS round's hint never names. + { + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + Gc gc(store, kGc); + seedPool(*backend, layout, gc); + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, 6}); + backend->hide(layout.refCkptKey(fixture::fixtureLife(ns))); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const RefHold hold = holdOf(*backend, layout, ns); + EXPECT_EQ(hold.reason, HoldReason::GapBelowWitness); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 5})); + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 4})); + } +} + +/// The second witness can also be UNREADABLE, and that is a different answer from absent. An absent +/// `_ckpt` says "this namespace published no checkpoint" and honestly contributes no witness; a present +/// one that will not decode says "this namespace HAS a checkpoint and we cannot read it", which no walk +/// may treat as no witness. +/// +/// It is still ONE NAMESPACE'S object. The fold used to fail the whole round closed on it — every +/// namespace's cursor, seal and cleanup stopped, every round, on one unreadable 4 KiB object, and the +/// exception named neither the namespace nor the key. The rule is the one §5 states for every other +/// per-namespace failure: hold the namespace that owns the object, fold everything else. +TEST(CASGCHoldGrammar, AnUndecodableCheckpointHoldsOnlyItsOwnNamespace) +{ + const RootNamespace bad{"00/aa@cas@"}; + /// Stage B (Task 4-C): no pin needed -- `publishAt(..., birth=true)` below (draining into + /// `writeRefLogTxnRaw`) admits `bad` into the catalog itself, pinned to the same sentinel this + /// test's own `fixture::fixtureLife(bad)` key computations already assume. + const RootNamespace good{"00/bb@cas@"}; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + Gc gc(store, kGc); + + publishAt(*backend, layout, bad, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, bad, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + publishAt(*backend, layout, good, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(11), /*birth=*/true); + writeCkptAt(*backend, layout, bad, RefTxnId{1, 2}); + writeCkptAt(*backend, layout, good, RefTxnId{1, 1}); + + /// Round 1 is the BASELINE both namespaces are measured against: each folds its whole stream and + /// seals a cursor, and neither holds. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_EQ(sealedCursorOf(*backend, layout, bad), (RefTxnId{1, 2})); + ASSERT_EQ(sealedCursorOf(*backend, layout, good), (RefTxnId{1, 1})); + ASSERT_FALSE(coverageOf(*backend, layout, bad)->hold.has_value()); + + /// Corrupt EXACTLY ONE OBJECT: the first namespace's `_ckpt` body. Nothing else in the pool changes, + /// so everything the next round does differently is attributable to this one object. + const String bad_ckpt_key = layout.refCkptKey(fixture::fixtureLife(bad)); + const HeadResult ckpt_head = backend->head(bad_ckpt_key); + ASSERT_TRUE(ckpt_head.exists); + ASSERT_EQ(backend->putOverwrite(bad_ckpt_key, "this is not a cas_ref_ckpt", ckpt_head.token).outcome, + PutOutcome::Done); + + /// Work only a round that COMPLETES can fold. + publishAt(*backend, layout, good, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(12)); + const String good_ckpt_key = layout.refCkptKey(fixture::fixtureLife(good)); + const HeadResult good_ckpt_head = backend->head(good_ckpt_key); + ASSERT_TRUE(good_ckpt_head.exists); + ASSERT_EQ(backend->putOverwrite(good_ckpt_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = std::nullopt, + }), good_ckpt_head.token).outcome, PutOutcome::Done); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + /// The namespace that owns the object is held, at the position its walk would have read next, and + /// its coverage row rides UNCHANGED — the cursor may not move while the hold stands. + const RefHold hold = holdOf(*backend, layout, bad); + EXPECT_EQ(hold.reason, HoldReason::CheckpointUndecodable); + EXPECT_EQ(hold.offending_position, (RefTxnId{1, 3})); + EXPECT_EQ(sealedCursorOf(*backend, layout, bad), (RefTxnId{1, 2})); + + /// The other namespace folded its new record. This is the whole point of the finding: one corrupt + /// object must not stop the pool. + const auto good_cov = coverageOf(*backend, layout, good); + ASSERT_TRUE(good_cov.has_value()); + EXPECT_FALSE(good_cov->hold.has_value()) << "the corrupt object belongs to the OTHER namespace"; + EXPECT_EQ(good_cov->last_folded_ref_id, (RefTxnId{1, 2})); + EXPECT_EQ(good_cov->classification, 2); + + /// And nothing was destroyed for the held namespace: a hold shuts the round's destructive gate, so + /// its ref objects — including the ones a cleanup range computed WITHOUT the unreadable checkpoint + /// would have widened onto — are all still there. + for (const RefTxnId & id : {RefTxnId{1, 1}, RefTxnId{1, 2}}) + EXPECT_TRUE(backend->head(layout.refLogKey(fixture::fixtureLife(bad), id)).exists) + << "ref log " << renderRefTxnId(id) << " of the held namespace was deleted"; +} + +/// THE OTHER ARM OF THE SAME RULE, and the one that must NOT mint a hold. +/// +/// A namespace can carry an undecodable `_ckpt` and offer the walk NO POSITION TO READ: never folded +/// (no sealed cursor) and no listed log. Two ways to get there, both real. A writer publishes the +/// object around its namespace's birth, so a `_ckpt` that lands before the birth log is durable is +/// exactly this shape. And `parseRefCkptKey` deliberately resolves anything of the form +/// `/_ckpt`, so a key with a stray segment names the checkpoint of a table that has no logs +/// and no snapshots and never will (`CasLayout.h`, "the phantom table it names ... the fold does +/// nothing for it") — which is precisely the object that used to halt GC for the entire pool. +/// +/// NO HOLD IS MINTED, and that is a positive design choice rather than a shortfall. A hold is not just +/// a stop flag: its `offending_position` is read by every later round as a DURABLE WITNESS that some +/// round once reached that position, which turns an absent below it into a gap rather than a frontier. +/// The walk here reached nothing, so any position would be invented — `{0, 0}` is rejected outright by +/// both codecs, and any canonical value would plant a permanent false witness under a namespace whose +/// records legitimately do not exist. The anomaly carries it instead, which is enough because it shuts +/// the same round-wide destructive gate a hold would, and because everything the checkpoint gates is a +/// no-op for a namespace the walk cannot even start on: nothing to fold, and an empty delete plan. +TEST(CASGCHoldGrammar, AnUndecodableCheckpointWithNoWalkPositionRecordsAnAnomalyAndMintsNoHold) +{ + const RootNamespace phantom{"00/aa@cas@"}; + const RootNamespace good{"00/bb@cas@"}; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + Gc gc(store, kGc); + + /// Stage B (Task 4-C): `phantom` gets no birth and no other production touch -- it is meant to + /// have no logs and no snapshots, ever. But `discoverUniverse` is now catalog-authoritative, so a + /// namespace absent from the catalog is invisible to the walk (R10 treats it as foreign-prefix-inert), + /// and this test's whole premise -- that GC still surfaces an anomaly for an uncataloged `_ckpt` -- + /// would be silently defeated. Admitting it here (still with no `_ckpt` of its own) is what keeps + /// `phantom` reachable by `readCheckpointWitnesses` without giving it the birth this test deliberately + /// withholds. + fixture::admitLive(*backend, layout, phantom); + + /// A lone `_ckpt` with an undecodable body, and NOTHING else under that namespace. + backend->putIfAbsent(layout.refCkptKey(fixture::fixtureLife(phantom)), "this is not a cas_ref_ckpt"); + publishAt(*backend, layout, good, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(11), /*birth=*/true); + writeCommittedCkptAt(*backend, layout, good, RefTxnId{1, 1}); + + const RoundReport report = gc.runRegularRound(); + ASSERT_TRUE(report.acquired_lease); + + /// The anomaly is the whole carrier here: it is what shuts the round's destructive gate, and the + /// gate is what keeps `cleanupRefObjects` from computing this namespace's delete range from an + /// ABSENT checkpoint — which is the WIDEST reading, not the safest. + /// Located by namespace and shard, then CHECKED ON ITS REASON. Asserting only that "some anomaly + /// exists for this namespace" is a pin any future unrelated anomaly would satisfy, and this test + /// would then stop testing anything; the reason check is what keeps it pinned to this arm. Note it + /// is also stricter than searching BY reason would be — it requires the FIRST anomaly recorded for + /// this namespace to be this one, not merely that one of them somewhere is. + const auto anomaly = std::find_if(report.anomalies.begin(), report.anomalies.end(), + [&](const RoundAnomaly & a) { return a.ns.string() == phantom.string() && a.shard == 0; }); + ASSERT_NE(anomaly, report.anomalies.end()) + << "an unreadable `_ckpt` must be surfaced even when there is no walk to stop"; + EXPECT_NE(anomaly->reason.find("_ckpt"), String::npos) + << "the anomaly must say WHAT stopped the namespace, not merely that something did"; + + const auto cov = coverageOf(*backend, layout, phantom); + ASSERT_TRUE(cov.has_value()); + EXPECT_FALSE(cov->hold.has_value()) << "a hold here could only name a position no round ever read"; + EXPECT_EQ(cov->classification, 1) << "nothing was folded, so the row is `unchanged`"; + EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{})); + + /// Same isolation as the held arm: the pool keeps working. + const auto good_cov = coverageOf(*backend, layout, good); + ASSERT_TRUE(good_cov.has_value()); + EXPECT_FALSE(good_cov->hold.has_value()); + EXPECT_EQ(good_cov->last_folded_ref_id, (RefTxnId{1, 1})); + + /// The unreadable object itself is never deleted as debris — repairing it is the operator's move, + /// and GC removing it would erase the only evidence of what stopped the namespace. + EXPECT_TRUE(backend->head(layout.refCkptKey(fixture::fixtureLife(phantom))).exists); +} + +/// ===================== THE HOLD IS DURABLE ===================== + +namespace +{ + +/// Seed a namespace held at {1,3} by a gap below the listed witness {1,4}, then make the hint forget +/// the namespace exists. Returns the round-1 hold. +RefHold seedHeldThenUnhinted( + const std::shared_ptr & backend, const PoolPtr & store, + const RootNamespace & ns, Gc & gc) +{ + const Layout & layout = store->layout(); + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + publishAt(*backend, layout, ns, RefTxnId{1, 4}, "ref_4", 4, DB::UInt128(4)); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 4}); + + EXPECT_TRUE(gc.runRegularRound().acquired_lease); + const RefHold hold = holdOf(*backend, layout, ns); + + /// Every one of the namespace's objects vanishes from every LIST while staying readable by key: + /// the round that follows has no hint entry for this namespace at all. + for (const RefTxnId & id : {RefTxnId{1, 1}, RefTxnId{1, 2}, RefTxnId{1, 4}}) + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), id)); + return hold; +} + +} + +TEST(CASGCHoldGrammar, HoldRidesARoundWhoseHintOmitsTheNamespace) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const RootNamespace ns{"00/aa@cas@"}; + Gc gc(store, kGc); + const RefHold first = seedHeldThenUnhinted(backend, store, ns, gc); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + ASSERT_GT(backend->holesServed(), 0u); + + const RefHold second = holdOf(*backend, store->layout(), ns); + EXPECT_EQ(second.reason, first.reason) << "a quiet hint must not rewrite why the namespace is held"; + EXPECT_EQ(second.offending_position, first.offending_position); + EXPECT_EQ(sealedCursorOf(*backend, store->layout(), ns), (RefTxnId{1, 2})) + << "the cursor may not advance while the hold stands"; + /// The one field that moves, and the reason it exists: it counts the rounds that retried and failed. + EXPECT_EQ(second.retry_count, first.retry_count + 1); +} + +TEST(CASGCHoldGrammar, HoldForcesAnExactRetryOfItsOffendingPositionWhenUnhinted) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + Gc gc(store, kGc); + seedHeldThenUnhinted(backend, store, ns, gc); + + const String offending = store->layout().refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 3}); + const uint64_t before = backend->getCount(offending); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_GT(backend->getCount(offending), before) + << "a carried hold must read its offending position by EXACT key; the hint cannot be asked, " + "because the hint no longer mentions the namespace at all"; +} + +/// The clearing rule, stated as a test: an absent proves nothing. The round below observes the +/// offending position absent AGAIN, with no witness anywhere — exactly the observation a lying store +/// produces — and the hold survives it. Only the record actually appearing, being folded, and the +/// result reaching `gc/state` clears it. +TEST(CASGCHoldGrammar, HoldClearsOnlyByFoldingThroughTheOffendingPosition) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + Gc gc(store, kGc); + seedHeldThenUnhinted(backend, store, ns, gc); + + /// Round 2: another absent, no witness. NOT a clearance. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(holdOf(*backend, layout, ns).offending_position, (RefTxnId{1, 3})); + + /// The record appears at last (still invisible to every LIST — the hold is the only thing that + /// knows to look there). + publishAt(*backend, layout, ns, RefTxnId{1, 3}, "ref_3", 3, DB::UInt128(3)); + backend->hide(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 3})); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const auto cov = coverageOf(*backend, layout, ns); + ASSERT_TRUE(cov.has_value()); + EXPECT_FALSE(cov->hold.has_value()) << "folding through the offending position is what clears a hold"; + EXPECT_EQ(cov->classification, 2); + EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, 4})) << "the walk resumed past the resolved gap"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(4)), 1) + << "the record above the gap finally contributed its owner edge"; +} + +/// ===================== REBUILD ===================== + +namespace +{ + +/// Rewrite the fold seal at an EXACT `(generation, attempt)`, applying `mutate` to it. Needed where +/// the seal under test is not the adopted one — a step-down test plants its hold in a generation the +/// pool has already moved past. +void mutateSealAt(Backend & backend, const Layout & layout, uint64_t generation, uint64_t attempt, + const std::function & mutate) +{ + const String key = layout.foldSealKey(generation, attempt); + CasFoldSeal seal = decodeFoldSeal(backend.get(key)->bytes); + mutate(seal); + backend.putOverwrite(key, encodeFoldSeal(seal), backend.head(key).token); +} + +/// Rewrite the adopted fold seal, applying `mutate` to it. Used to plant a hold that the rebuild must +/// then carry: planting it directly (rather than by holding a real round) keeps the REBUILD tests about +/// the carry, not about how the hold arose. +void mutateAdoptedSeal(Backend & backend, const Layout & layout, const std::function & mutate) +{ + const GcState st = decodeGcState(backend.get(layout.gcStateKey())->bytes); + const String key = layout.foldSealKey(st.snap_generation, st.snap_attempt); + CasFoldSeal seal = decodeFoldSeal(backend.get(key)->bytes); + mutate(seal); + backend.putOverwrite(key, encodeFoldSeal(seal), backend.head(key).token); +} + +RefHold plantedHold() +{ + return RefHold{.reason = HoldReason::WitnessDisappeared, .offending_position = RefTxnId{4, 9}, + .retry_count = 17, .next_retry_round = 23}; +} + +} + +/// A rebuild carries a hold only for the matching catalog life. A historical row whose id is absent +/// from the rebuild cut is dropped and cannot mint output work. +TEST(CASGCHoldGrammar, RebuildCarriesMatchingHoldAndDropsAbsentLife) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); + constexpr UInt128 absent_life_id{0xfeed}; + mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) + { + RefCoverage & cov = seal.ref_lives.at(life_id).coverage; + cov.classification = 4; + cov.hold = plantedHold(); + RefCoverage gone; + gone.classification = 4; + gone.last_folded_ref_id = RefTxnId{2, 2}; + gone.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{2, 3}, + .retry_count = 1, .next_retry_round = 2}; + seal.ref_lives[absent_life_id].coverage = gone; + }); + + const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); + ASSERT_TRUE(rep.performed) << rep.refusal; + + const auto rebuilt = newestSeal(*backend, layout); + ASSERT_TRUE(rebuilt.has_value()); + const auto rediscovered = rebuilt->ref_lives.find(life_id); + ASSERT_NE(rediscovered, rebuilt->ref_lives.end()); + EXPECT_EQ(rediscovered->second.coverage.classification, 4); + ASSERT_TRUE(rediscovered->second.coverage.hold.has_value()); + EXPECT_EQ(*rediscovered->second.coverage.hold, plantedHold()); + EXPECT_FALSE(rebuilt->ref_lives.contains(absent_life_id)); +} + +/// AN ORDINARY CRASH IS NOT A CORRUPT POOL. A round writes its runs during the reduce phase and its +/// fold seal only at phase 10/18, so a crash in between leaves the newest generation existing WITHOUT +/// a seal — the commonest shape there is. If discovery stopped at the listing's maximum it would find +/// no seal there, conclude it could enumerate nothing, and refuse — telling the operator to recreate a +/// pool whose holds are sitting readable one generation down. +/// +/// So discovery steps DOWN through the generations the listing itself reported until one carries a +/// seal. That spends no trust the maximum had not already been given. What it does NOT weaken is the +/// refusal above the maximum: that one stays terminal, because a seal found there is the listing +/// caught lying, not merely being incomplete about seals. +TEST(CASGCHoldGrammar, RebuildStepsDownPastACrashedNewestGenerationToTheSealBelowIt) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const GcState after_first = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const uint64_t older_generation = after_first.snap_generation; + const uint64_t older_attempt = after_first.snap_attempt; + const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); + + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, 2}); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const GcState after_second = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(after_second.snap_generation, older_generation) << "the fixture needs two generations"; + + /// The older generation is the one holding the pool's durable hold. + mutateSealAt(*backend, layout, older_generation, older_attempt, [&](CasFoldSeal & seal) + { + RefCoverage & cov = seal.ref_lives.at(life_id).coverage; + cov.classification = 4; + cov.hold = plantedHold(); + }); + + /// THE CRASH: the newest generation's run objects are there, its seal never got written. Then + /// `gc/state` is lost, which is this path's whole premise. + const String newest_seal = layout.foldSealKey(after_second.snap_generation, after_second.snap_attempt); + const HeadResult seal_head = backend->head(newest_seal); + ASSERT_TRUE(seal_head.exists); + ASSERT_EQ(backend->deleteExact(newest_seal, seal_head.token).kind, DeleteOutcome::Kind::Deleted); + ASSERT_FALSE(backend->list(layout.gcGenPrefix(after_second.snap_generation), "", 1).keys.empty()) + << "the crashed generation must still hold objects, or it is not the shape being modelled"; + const HeadResult sh = backend->head(layout.gcStateKey()); + ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc2(store, hexToU128("0000000000000000000000000000000c")); + const RebuildReport rep = gc2.rebuildBaseline(/*force=*/false); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_FALSE(rep.virgin_by_enumeration) << "a pool with a readable seal is not virgin"; + EXPECT_EQ(rep.adopted_seal_generation, older_generation) + << "the report must name WHICH generation the holds came from, so a step-down is visible"; + + const auto rebuilt = newestSeal(*backend, layout); + ASSERT_TRUE(rebuilt.has_value()); + const auto it = rebuilt->ref_lives.find(life_id); + ASSERT_NE(it, rebuilt->ref_lives.end()); + ASSERT_TRUE(it->second.coverage.hold.has_value()) + << "a crash between the run writes and the seal write turned into 'recreate the pool', and the " + "hold readable one generation down was thrown away with it"; + EXPECT_EQ(*it->second.coverage.hold, plantedHold()); +} + +/// With no readable prior seal there is nothing to carry, and the holds it may have contained are +/// unknowable. The rebuild refuses rather than blessing a baseline whose provenance it cannot state — +/// a pool-wide hold is not representable (there is no offending position anyone could ever fold +/// through), so the honest answer is the refusal, and the recovery path is pool recreation. +TEST(CASGCHoldGrammar, RebuildRefusesWithAMissingPriorSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(st.snap_generation, 0u); + const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); + const HeadResult sh = backend->head(seal_key); + ASSERT_TRUE(sh.exists); + ASSERT_EQ(backend->deleteExact(seal_key, sh.token).kind, DeleteOutcome::Kind::Deleted); + + /// FORCE does not buy past it either: force means "rebuild deliberately", never "drop the holds". + for (const bool force : {false, true}) + { + SCOPED_TRACE(force ? "force" : "plain"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.rebuildBaseline(force); }); + } + + const GcState after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + EXPECT_EQ(after.snap_generation, st.snap_generation) << "a refused rebuild adopts nothing"; +} + +TEST(CASGCHoldGrammar, RebuildRefusesWithAnUndecodablePriorSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); + backend->putOverwrite(seal_key, "{\"type\":\"cas_fold_seal\",\"v\":4}\nthis is not a seal body\n", + backend->head(seal_key).token); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.rebuildBaseline(/*force=*/true); }); +} + +/// LOSING THE POINTER IS NOT WEAKER THAN LOSING THE SEAL. `gc/state` names the adopted seal, and it is +/// the seal that carries the holds — so if the refusal only covered an unreadable seal, the *lesser* +/// corruption (the pointer is gone, every seal intact) would be treated more permissively than the +/// greater one, and the rebuild would write a baseline with no hold in it at all. +/// +/// That matters because holds are not re-derivable by the next walk. `WitnessDisappeared` names a +/// record that is *gone*: the next round reads a clean frontier and would hand the namespace exactly +/// the frontier proof the hold exists to deny. Same for any hold whose only witness was the checkpoint +/// or the hold itself. +/// +/// So with no adopted baseline named, the rebuild finds the newest fold seal OBJECT by enumeration and +/// carries its holds. This keeps the pool's disaster recovery intact — losing `gc/state` on a +/// lived-in pool is the scenario `REBUILD` exists for — while making it impossible to write a +/// hold-free baseline over a pool that had holds. +TEST(CASGCHoldGrammar, RebuildWithLostStateStillCarriesHoldsFromTheNewestSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); + mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) + { + RefCoverage & cov = seal.ref_lives.at(life_id).coverage; + cov.classification = 4; + cov.hold = plantedHold(); + }); + + /// The pointer vanishes; every seal object survives. + const HeadResult sh = backend->head(layout.gcStateKey()); + ASSERT_TRUE(sh.exists); + ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc2(store, hexToU128("00000000000000000000000000000009")); + const RebuildReport rep = gc2.rebuildBaseline(/*force=*/false); + ASSERT_TRUE(rep.performed) << rep.refusal; + + const auto rebuilt = newestSeal(*backend, layout); + ASSERT_TRUE(rebuilt.has_value()); + const auto it = rebuilt->ref_lives.find(life_id); + ASSERT_NE(it, rebuilt->ref_lives.end()); + ASSERT_TRUE(it->second.coverage.hold.has_value()) + << "the rebuild blessed a baseline with no hold in it, having read no seal at all"; + EXPECT_EQ(*it->second.coverage.hold, plantedHold()); +} + +/// ...and when that newest seal cannot be read either, there is nothing left to carry and no way to +/// know what was lost, so the rebuild refuses exactly as it does for an unreadable adopted seal. +TEST(CASGCHoldGrammar, RebuildRefusesWhenTheNewestSealIsUnreadableAndTheStateIsLost) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); + backend->putOverwrite(seal_key, "{\"type\":\"cas_fold_seal\",\"v\":4}\nthis is not a seal body\n", + backend->head(seal_key).token); + const HeadResult sh = backend->head(layout.gcStateKey()); + ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc2(store, hexToU128("0000000000000000000000000000000a")); + for (const bool force : {false, true}) + { + SCOPED_TRACE(force ? "force" : "plain"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc2.rebuildBaseline(force); }); + } +} + +/// NEWEST-NESS IS NOT READ OFF A LISTING. Taking the newest seal from the pool-wide enumeration would +/// put the same hole one layer up: a listing that omits the true newest seal hands back an OLDER one, +/// and every hold detected since that older seal is silently lost. Two narrow single-generation probes +/// above the listing's maximum ask whether it lied. +/// +/// And when it did lie, the answer is REFUSAL, not adoption of the newer seal. A store that misreports +/// its own enumeration DURING DISASTER RECOVERY does not get a second guess: adopting whatever the +/// second query happened to return would move the same trust one query along and prove nothing. +/// +/// The fixture is the production shape rather than a contrivance: the broad `gc/gen/` enumeration +/// omits the newest generation's objects while a listing scoped to that generation still returns +/// them — the same class of lie the arithmetic ref walk was built for one layer down. +TEST(CASGCHoldGrammar, RebuildRefusesWhenANarrowProbeFindsASealAboveTheListingMaximum) +{ + /// Omits keys from ONE enumeration prefix only. Every other query — including a listing scoped to + /// the generation itself — answers truthfully. + class BroadListHoleBackend : public InMemoryBackend + { + public: + String hide_under_prefix; + String hidden_key_infix; + size_t holes_served = 0; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + if (prefix != hide_under_prefix) + return page; + const size_t before = page.keys.size(); + std::erase_if(page.keys, + [&](const ListedKey & k) { return k.key.find(hidden_key_infix) != String::npos; }); + if (page.keys.size() != before) + ++holes_served; + return page; + } + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(st.snap_generation, 1u) << "the fixture needs a newer generation to hide"; + const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); + mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) + { + RefCoverage & cov = seal.ref_lives.at(life_id).coverage; + cov.classification = 4; + cov.hold = plantedHold(); + }); + + /// The pool-wide enumeration loses the newest generation entirely; the pointer to it is deleted. + const String gen_prefix = layout.gcGenPrefix(0); + backend->hide_under_prefix = gen_prefix.substr(0, gen_prefix.size() - 2); /// ".../gc/gen/" + backend->hidden_key_infix = layout.gcGenPrefix(st.snap_generation); + const HeadResult sh = backend->head(layout.gcStateKey()); + ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc2(store, hexToU128("0000000000000000000000000000000b")); + for (const bool force : {false, true}) + { + SCOPED_TRACE(force ? "force" : "plain"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc2.rebuildBaseline(force); }); + } + ASSERT_GT(backend->holes_served, 0u) << "the broad listing never actually lied"; + + /// Nothing was adopted: the refusal fires before the lease, so the pool is exactly as it was. + EXPECT_FALSE(backend->head(layout.gcStateKey()).exists) + << "a refused rebuild must not mint a baseline, nor a bootstrap body"; +} + +/// The virgin verdict, pinned so the refusal can never grow to swallow a fresh pool — and pinned as +/// what it actually is. It rests on THREE pieces of enumeration evidence (wide LIST empty, narrow +/// generation-1 probe empty, no `gc/state`) and on no point read at all, so it is COUNTED: an operator +/// reading a disaster-recovery run needs to see that the clean slate came from enumeration rather than +/// from proof. `CASGCRebuildVirginByEnumeration` on a pool that has ever completed a round means the +/// enumeration lied. +TEST(CASGCHoldGrammar, RebuildProceedsOnAPoolThatNeverSealedABaselineAndCountsTheVerdict) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + /// No round has run, so there is no `gc/state` and no seal — only owner state to rebuild from. + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); + writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); + ASSERT_FALSE(backend->head(layout.gcStateKey()).exists); + + using ProfileEvents::global_counters; + const auto virgin_before = global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration].load(); + + Gc gc(store, kGc); + const RebuildReport rep = gc.rebuildBaseline(/*force=*/false); + EXPECT_TRUE(rep.performed) << rep.refusal; + EXPECT_GT(global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration].load(), virgin_before) + << "a clean slate granted from enumeration alone must be visible to whoever reads the run"; +} diff --git a/src/Disks/tests/gtest_cas_gc_leak.cpp b/src/Disks/tests/gtest_cas_gc_leak.cpp new file mode 100644 index 000000000000..a65b225b8f9a --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_leak.cpp @@ -0,0 +1,525 @@ +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int FILE_DOESNT_EXIST; +extern const int ABORTED; +} + +/// NO-LEAK property suite (C++ verification of the R0 INV-NO-LEAK invariant for the root-local +/// part-manifest model). Every dropped/abandoned closure must be FULLY reclaimed: after GC reaches a +/// fixpoint, NO blob or manifest object may remain for the reclaimed part, and the in-degree generation +/// must hold no stranded positive counter for a now-unreferenced blob. +/// +/// The model has changed since the tree/snap era: a part is one immutable single-owner `ManifestId` +/// (only blobs stay content-addressed; manifests are NEVER shared across instances — backlog item B7). +/// The leak scenarios below therefore drive the REAL write flow (`stageManifest -> precommitAdd -> +/// putBlob -> promote`) and the real drop/abandon paths, then assert the reclaimed closure leaves no +/// debris. The old "adopt-by-tree relink" leak cases (B7) are REMOVED: there is no shared content id, +/// no subtree placement, `getPartTreeId` returns nullopt and `adoptPart` throws `NOT_IMPLEMENTED`; the +/// byte-stream-fallback relink is an ordinary publish covered by the no-leak displacement repros below. + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::inDegreeOf; + +namespace +{ + +PoolPtr openTestPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// Whether the CURRENT retired list (any gc-shard) still holds an entry — the ack-floor deletion pipeline +/// (condemn -> graduate -> delete) is in flight while this is true. +bool anyRetiredPending(const PoolPtr & s) +{ + /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// separate retired list — reconstruct the in-flight set from the seal. + return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); +} + +/// Drive regular GC to a fixpoint. A condemned blob is not deleted in the round that folds its removal: +/// it condemns, then graduates the round after (round-paced, unconditional), then the NEXT pass deletes +/// it. The loop renews the store's own heartbeat after each round (`renewWatermarkOnce`, unrelated to +/// graduation timing but keeping the build-watermark floor and lease current) and stays alive while ANY +/// work counter is nonzero OR the current retired list still holds an in-flight entry. +size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + const RoundReport rep = DB::Cas::tests::runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(s)) + break; + } + return rounds; +} + +/// A `ManifestEntry` for a Blob leaf at `path` referencing `payload`'s content hash. +ManifestEntry blobEntry(const String & path, const String & payload) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + return e; +} + +/// Publish ONE ref naming a two-blob part through the REAL writer transaction sequence — the exact order +/// the wiring drives (EDGE-BEFORE-OBSERVE): `beginPartWrite -> stageManifest(entries) -> precommitAdd -> +/// putBlob(each body) -> promote`. The durable precommit closure names every blob hash before putBlob +/// makes the first backend observation. Returns the published `ManifestId` so a caller can later HEAD +/// its body / assert reclaim. +ManifestId publishTwoBlobPart( + const PoolPtr & s, const RootNamespace & ns, const String & ref, + const String & payload_a, const String & payload_b) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + + const ManifestId id = build->stageManifest({blobEntry("data.bin", payload_a), + blobEntry("data.cmrk3", payload_b)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload_a), BlobSource::fromString(payload_a)); + build->putBlob(idOf(payload_b), BlobSource::fromString(payload_b)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// Publish ONE ref naming a single-blob part through the real writer sequence. Returns its ManifestId. +ManifestId publishOneBlobPart( + const PoolPtr & s, const RootNamespace & ns, const String & ref, const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({blobEntry("data.bin", payload)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// Whether a blob's body object is present in the backend (HEADs blobKey directly — the GC retire path +/// HEADs the object key, never the Pool's manifest decode cache). +bool blobPresent(const std::shared_ptr & b, const Layout & layout, const String & payload) +{ + return b->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))})).exists; +} + +/// Whether a manifest body object is present in the backend. +bool manifestPresent(const std::shared_ptr & b, const Layout & layout, const ManifestId & id) +{ + return b->head(layout.manifestKey(id)).exists; +} + +/// Replace the existing ref with partB through the real durable-precommit writer sequence. The +/// resulting `promote` is the production REPOINT: partA's committed owner is removed while partB's +/// committed owner is installed in the same ordered journal transition. +ManifestId publishPartBReplacement( + const PoolPtr & s, const RootNamespace & ns, const String & ref, + const String & payload_a, const String & payload_b) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({blobEntry("data.bin", payload_a), + blobEntry("data.cmrk3", payload_b)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload_a), BlobSource::fromString(payload_a)); + build->putBlob(idOf(payload_b), BlobSource::fromString(payload_b)); + build->promote(ns, ref, build->buildId(), id, /*allow_repoint=*/true); + return id; +} + +/// Reproduce displacement on the SAME (s, ns, ref) and run GC to a fixpoint. partB's distinct blobs +/// displace partA's via a REPOINT of the ref (one RootOwnerEvent old={Committed,ref,partA}/ +/// new={Committed,ref,partB}) — the real production shape of last-owner-wins, NOT a body delete. +/// +/// Crucially the test does NOT delete partA's manifest body. In the part-manifest model a true removal +/// (the repoint's -1) is derived by GC READING partA's body at removal-fold time; only GC may delete a +/// committed owner's body, and only AFTER the -1 is sealed (recheck cleanup, control #11). So GC folds +/// the repoint: -1 for partA's blobs (body present), +1 for partB's blobs, retires + deletes partA's +/// now-zero-in-degree blobs, and recheck cleanup deletes partA's owner-removed body. Returns the fsck +/// report so the caller can assert the no-leak end state (partA's blobs AND body gone, unreachable==0). +FsckReport displaceAndGc( + const PoolPtr & s, const std::shared_ptr & b, + const RootNamespace & ns, const String & ref, const ManifestId & part_a) +{ + /// Publish partB's full closure and atomically repoint the ref from partA to partB. + const ManifestId part_b = publishPartBReplacement(s, ns, ref, "data-B", "mark-B"); + + EXPECT_TRUE(b->head(s->layout().manifestKey(part_a)).exists) + << "partA manifest body must still be present so GC can read its -1 edges at removal-fold"; + + const auto resolved = s->resolveRef(ns, ref); + EXPECT_TRUE(resolved.has_value()); + if (resolved) + EXPECT_EQ(resolved->manifest_id, part_b) << "the real writer promotion must leave the ref on partB"; + + /// The repoint dropped partA's owner; advance the watermark floor so partA's now-orphaned blobs are + /// not spared as in-flight, then run GC to a fixpoint. + s->renewWatermarkOnce(); + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runGcToFixpoint(s, gc); + return runFsck(*s, /*detail=*/false); +} + +} + +/// NO-LEAK (S1, fold interleaved): partA is published and folded ONCE (its body present, +1 per blob), +/// then partB REPOINTS the ref away from partA (partA's body stays present so GC reads its -1 edges at +/// removal-fold; only GC deletes the owner-removed body, after the -1 is sealed). GC must reclaim partA's +/// blobs to a fixpoint: no blob/manifest object remains for partA and the in-degree generation holds no +/// stranded positive counter for partA's blobs. +TEST(CASGCLeak, DisplacedPartBlobsReclaimedFoldBetween) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String ref = "all_0_0_0"; + + const ManifestId part_a = publishTwoBlobPart(s, ns, ref, "data-A", "mark-A"); + + /// A GC fold runs HERE, before any displacement — partA's body is present, so the fold records +1 for + /// each of partA's blobs into the durable in-degree generation. + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runGcToFixpoint(s, gc); + } + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("data-A")), 1) << "partA's data blob is pinned (+1)"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("mark-A")), 1) << "partA's mark blob is pinned (+1)"; + + const FsckReport after = displaceAndGc(s, b, ns, ref, part_a); + + EXPECT_EQ(after.dangling, 0u) << "S1 INV-NO-LOSS: displacement must never lose a reachable object"; + EXPECT_GT(after.reachable, 0u) << "S1: the live ref points at partB; partB's closure is reachable"; + EXPECT_EQ(after.unreachable, 0u) + << "S1 INV-NO-LEAK: an interleaved fold recorded partA's edges; the removal -1 + retire must " + "reclaim partA's blobs (unreachable=" << after.unreachable << ")"; + + /// Backend-level no-debris: partA's blobs and body object are gone; the in-degree counters are 0. + EXPECT_FALSE(blobPresent(b, s->layout(), "data-A")) << "S1: partA data blob object must be deleted"; + EXPECT_FALSE(blobPresent(b, s->layout(), "mark-A")) << "S1: partA mark blob object must be deleted"; + EXPECT_FALSE(manifestPresent(b, s->layout(), part_a)) << "S1: partA manifest body must be gone"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("data-A")), 0) << "S1: no stranded positive in-degree"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("mark-A")), 0) << "S1: no stranded positive in-degree"; +} + +/// NO-LEAK (S2, NO fold interleaved — the decisive worst case): partA is published, then IMMEDIATELY +/// repointed to partB before ANY GC fold runs. The single fold therefore folds partA's activation (+1) +/// and its removal (-1, read from partA's still-present body) in one pass; the retire reclaims partA's +/// blobs and recheck cleanup deletes partA's owner-removed body. No debris may remain. +TEST(CASGCLeak, DisplacedPartBlobsReclaimedNoFoldBetween) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String ref = "all_0_0_0"; + + const ManifestId part_a = publishTwoBlobPart(s, ns, ref, "data-A", "mark-A"); + + const FsckReport after = displaceAndGc(s, b, ns, ref, part_a); + + EXPECT_EQ(after.dangling, 0u) << "S2 INV-NO-LOSS: displacement must never lose a reachable object"; + EXPECT_GT(after.reachable, 0u) << "S2: the live ref points at partB; partB's closure is reachable"; + EXPECT_EQ(after.unreachable, 0u) + << "S2 INV-NO-LEAK: partA's blobs must be reclaimed even with no interleaved fold — the recorded " + "owner edges drive the removal -1 + retire (unreachable=" << after.unreachable << ")"; + + EXPECT_FALSE(blobPresent(b, s->layout(), "data-A")) << "S2: partA data blob object must be deleted"; + EXPECT_FALSE(blobPresent(b, s->layout(), "mark-A")) << "S2: partA mark blob object must be deleted"; + EXPECT_FALSE(manifestPresent(b, s->layout(), part_a)) << "S2: partA manifest body must be gone"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("data-A")), 0) << "S2: no stranded positive in-degree"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("mark-A")), 0) << "S2: no stranded positive in-degree"; +} + +/// NO-LEAK (drop): a fully-committed part is published, folded (+1 per blob), then its ref is dropped. +/// GC must reclaim the WHOLE closure — both blobs and the manifest body — leaving no debris and no +/// stranded positive in-degree. +TEST(CASGCLeak, DroppedPartFullyReclaimed) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String ref = "all_1_1_0"; + + const ManifestId id = publishTwoBlobPart(s, ns, ref, "drop-data", "drop-mark"); + { + Gc gc(s, hexToU128("00000000000000000000000000000002")); + runGcToFixpoint(s, gc); + } + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("drop-data")), 1); + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("drop-mark")), 1); + + s->dropRef(ns, ref); + s->renewWatermarkOnce(); /// advance the floor so the now-unreferenced closure is not spared + + Gc gc(s, hexToU128("00000000000000000000000000000002")); + runGcToFixpoint(s, gc); + + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.dangling, 0u) << "drop INV-NO-LOSS: nothing reachable was lost"; + EXPECT_EQ(after.unreachable, 0u) + << "drop INV-NO-LEAK: the dropped closure's blobs + body must be fully reclaimed " + "(unreachable=" << after.unreachable << ")"; + EXPECT_FALSE(blobPresent(b, s->layout(), "drop-data")) << "dropped data blob must be deleted"; + EXPECT_FALSE(blobPresent(b, s->layout(), "drop-mark")) << "dropped mark blob must be deleted"; + EXPECT_FALSE(manifestPresent(b, s->layout(), id)) << "dropped manifest body must be gone"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("drop-data")), 0) << "no stranded positive in-degree"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of("drop-mark")), 0) << "no stranded positive in-degree"; +} + +/// NO-LEAK (republish): a blob incarnation A is published, dropped, and condemned by ONE GC +/// round (retired, NOT yet deleted — it is still mid-pipeline). A fresh build then dedup-hits the SAME +/// content hash: `putBlob` HEADs A, sees it condemned via the per-hash freshness meta point-read, and — +/// per INV-1 (revival-from-source) — re-uploads a DISTINCT incarnation B at the same content-addressed key +/// (fresh `incarnation_tag`, never a GET of the dying object A). B is referenced by a second ref, then +/// that ref is dropped too. GC must fold B's own activation/removal exactly like any other incarnation +/// and reclaim it to a fixpoint: no blob object may remain for the content hash and the in-degree +/// generation must hold no stranded positive counter. +/// +/// This reproduces RESURRECT-REUPLOAD-ORPHAN: if GC's bookkeeping keys off the content hash rather than +/// the (hash, token) incarnation identity, it may treat the hash as "already handled" from A's retire +/// cycle and never open a fresh condemn cycle for B once B's in-degree drops to zero — B then orphans +/// forever (unreachable > 0, its body never deleted). +TEST(CASGCLeak, ResurrectReplacedIncarnationReclaimed) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String P = "republish-payload"; + + /// 1. Publish ref r1 -> token A referenced; capture A. + publishOneBlobPart(s, ns, "r1", P); + const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hA.exists); + + /// 2. Drop r1 -> A dereferenced. + s->dropRef(ns, "r1"); + s->renewWatermarkOnce(); /// advance the floor so A is not spared as in-flight + + /// 3. ONE GC round: A transitions to in-degree 0 and is condemned (retired), NOT yet deleted. + Gc gc(s, hexToU128("00000000000000000000000000000004")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + { + const auto lm = DB::Cas::tests::loadMetaForTest(*b, s->layout(), u128Of(P)); + ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned) + << "precondition: token A must be condemned before republication"; + } + ASSERT_TRUE(blobPresent(b, s->layout(), P)) << "A not yet deleted (still in the pipeline)"; + + /// 4. RESURRECT: a fresh build dedup-hits P; putBlob sees A condemned -> re-uploads a DISTINCT + /// incarnation B at the same content-addressed key (INV-1 revival-from-source). + publishOneBlobPart(s, ns, "r2", P); + const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hB.exists); + ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a new incarnation token B"; + + /// 5. Drop r2 -> B dereferenced. + s->dropRef(ns, "r2"); + s->renewWatermarkOnce(); + + /// 6. Run GC to fixpoint. The replaced incarnation B MUST be reclaimed. + runGcToFixpoint(s, gc); + + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.dangling, 0u) << "republication INV-NO-LOSS: nothing reachable was lost"; + EXPECT_EQ(after.unreachable, 0u) + << "republication INV-NO-LEAK: the replaced incarnation B must not orphan " + "(unreachable=" << after.unreachable << ")"; + EXPECT_FALSE(blobPresent(b, s->layout(), P)) << "B's object must be deleted"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of(P)), 0) << "no stranded positive in-degree"; +} + +/// IDEMPOTENCY of the republication-orphan fold: drives the exact same condemn-A / republish-B / +/// drop-B / reclaim sequence as `ResurrectReplacedIncarnationReclaimed` above, then keeps running the +/// regular round PAST the fixpoint. The re-condemn that reclaims the replaced incarnation B +/// must fire exactly once: extra rounds on an already-reclaimed content hash must be no-ops (no +/// re-condemn churn, no duplicate retired entry) and must never manufacture fresh fsck debris. +TEST(CASGCLeak, ResurrectReplacedReclaimIsIdempotent) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String P = "republish-payload-idem"; + + /// 1. Publish ref r1 -> token A referenced, then drop it. + publishOneBlobPart(s, ns, "r1", P); + s->dropRef(ns, "r1"); + s->renewWatermarkOnce(); /// advance the floor so A is not spared as in-flight + + /// 2. ONE GC round: A transitions to in-degree 0 and is condemned (retired), NOT yet deleted. + Gc gc(s, hexToU128("00000000000000000000000000000005")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + + /// 3. RESURRECT: r2 dedup-hits P while A is condemned -> mints a fresh incarnation B. + publishOneBlobPart(s, ns, "r2", P); + s->dropRef(ns, "r2"); + s->renewWatermarkOnce(); + + /// 4. Reclaim B to a fixpoint (the RESURRECT-REUPLOAD-ORPHAN fold under test). + runGcToFixpoint(s, gc); + ASSERT_FALSE(blobPresent(b, s->layout(), P)) << "B must be reclaimed before the idempotency check"; + + /// 5. Extra rounds past the fixpoint: nothing is left to do for this hash. The fold must not + /// re-condemn it (that would be the churn/duplicate-entry bug) and must not republish any debris. + for (int round = 0; round < 3; ++round) + { + const RoundReport r = DB::Cas::tests::runRegularRoundReclaiming(gc); + if (r.acquired_lease) + EXPECT_EQ(r.condemned, 0u) << "no re-condemn of an already-reclaimed hash on extra round " << round; + s->renewWatermarkOnce(); + } + + EXPECT_FALSE(blobPresent(b, s->layout(), P)) << "stays deleted across extra rounds"; + EXPECT_EQ(inDegreeOf(*b, s->layout(), u128Of(P)), 0) << "no stranded positive in-degree"; + + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.unreachable, 0u) << "re-condemn churn must not manufacture a fresh unreachable object"; + EXPECT_EQ(after.dangling, 0u) << "idempotent extra rounds must never lose a reachable object"; +} + +/// WRITER-SIDE half of the republication-orphan fold: after the round that folds the replaced +/// replaced incarnation B's dereference re-condemns B, a fresh writer dedup-hitting the SAME content hash +/// must see B as condemned via the per-hash freshness meta point-read — never as an adoptable live token. +/// If GC's bookkeeping instead kept treating B as adopt-eligible (the pre-fix bug), a concurrent writer's +/// `putBlob` would adopt the being-reclaimed B rather than publish a fresh incarnation, racing the +/// delete pipeline. +/// +/// Depending on round timing, by the time the meta is checked B may be (a) still present and visibly +/// condemned, or (b) already physically deleted by the delete pipeline (meta dropped alongside it) — BOTH +/// outcomes prove B is never adoptable. The assertion only fails on the pre-fix shape: B present and NOT +/// condemned. +TEST(CASGCLeak, ResurrectReplacedTokenIsCondemnedInMeta) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + Gc gc(s, hexToU128("00000000000000000000000000000006")); + const String P = "republish-payload-view"; + + /// 1. Publish ref r1 -> token A referenced; capture A, then drop it and condemn via ONE GC round. + publishOneBlobPart(s, ns, "r1", P); + const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hA.exists); + s->dropRef(ns, "r1"); + s->renewWatermarkOnce(); /// advance the floor so A is not spared as in-flight + gc.runRegularRound(); + + /// 2. RESURRECT: r2 dedup-hits P while A is condemned -> mints a fresh incarnation B. + publishOneBlobPart(s, ns, "r2", P); + const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hB.exists); + ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a distinct incarnation"; + s->dropRef(ns, "r2"); + s->renewWatermarkOnce(); + + /// 3. The round that folds B's dereference re-condemns token B. + gc.runRegularRound(); + + const auto lm = DB::Cas::tests::loadMetaForTest(*b, s->layout(), u128Of(P)); + EXPECT_TRUE((lm.has_value() && lm->meta.state == MetaState::Condemned) || !blobPresent(b, s->layout(), P)) + << "the replaced incarnation B must be visible as condemned (or already reclaimed) so a " + "dedup-hitting writer resurrects, not adopts"; +} + +/// (The NO-LEAK-on-abandon test `CASGCLeak.AbandonedPrecommitReclaimsOwnBlobs` was removed with the +/// snapshot+log ref model: it asserted GC AUTOMATICALLY reclaims a crashed build's abandoned precommit and +/// collects its own unique blob. Per spec §Responsibility Boundary that reclaim is now the WRITER's job +/// (it appends the exact `owner_transition` removal on recovery); GC never scans for or removes precommit +/// bindings. The writer-side abandon/recovery cleanup is exercised by the writer tests.) + +/// REUSE-vs-GC race (no-LOSS half of the no-leak family): a build ADOPTS a committed blob B by tokenless +/// evidence (B present, not yet condemned), the committed ref pinning B is DROPPED, GC retires+deletes B +/// AND completes the round, and only THEN does the build try to publish a manifest naming B. +/// +/// The promote gate re-observes the loss (it re-HEADs every blob leaf and fails closed on a deleted dep, +/// throwing a retryable ABORTED) — it must NEVER silently commit a dangling ref. The assertion is the +/// no-LOSS guarantee: `dangling==0`. (A tokenless adopt has no body to re-upload, so a real caller would +/// re-derive B from source on retry; here we only confirm the gate fails closed.) +TEST(CASReuseGcRace, ReuseOfBlobDeletedBeforePublish) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"test/tbl"}; + const String B = "shared-blob-payload"; + + /// build1: commit part_1 -> manifest -> blob B. + publishOneBlobPart(s, ns, "part_1", B); + + /// build2: adopt B by tokenless evidence (no HEAD). It does NOT yet stage a manifest or precommit — + /// the scenario is that GC deletes B BEFORE build2 publishes a manifest + /// naming it. (Staging+precommitting BEFORE the drop would make the precommit's activating +1 PIN B — + /// B would never reach in-degree 0 and GC could not delete it, so the race could not be reproduced.) + PartWriteInfo info; + info.intended_ref = ns.string() + "/part_2"; + auto build2 = s->beginPartWrite(info); + + ManifestEntry eb; + eb.path = "data.bin"; + eb.placement = EntryPlacement::Blob; + eb.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(B))}; + + eb.blob_size = B.size(); + build2->adoptEvidence(eb); /// tokenless dep (no HEAD) + + /// Drop the committed pin on B and advance the watermark so B (owned by the finished build1) is not + /// spared. No owner names B now (build2 has not staged/precommitted), so GC folds B to in-degree 0 + /// once part_1 is dropped. + s->dropRef(ns, "part_1"); + s->renewWatermarkOnce(); + + /// GC reclaims build1's manifest and the now-unreferenced B to a fixpoint, completing the rounds. + { + Gc gc(s, u128Of("gc-reuse-race")); + runGcToFixpoint(s, gc); + } + ASSERT_FALSE(blobPresent(b, s->layout(), B)) + << "GC must have deleted the now-unreferenced reused blob B"; + + /// Only NOW does build2 publish a manifest naming the (just-deleted) B: stage the body + precommit. + const ManifestId id2 = build2->stageManifest({eb}); + build2->precommitAdd(ns, "part_2", id2); + + /// build2 promotes part_2 -> id2 -> {B}. §4 manifest-trust: B is a committed-source adopted leaf, + /// so the promote gate TRUSTS it (no HEAD/loadMeta probe) and commits — it does NOT re-observe the + /// deleted B. This is the accepted D4 trade-off. On the real reuse/relink path B CANNOT be deleted + /// while build2's precommit edge is live: precommitAdd durably appends the Precommit OwnerTransition + /// (CasPartWriteTxn.cpp precommitAdd) BEFORE promote, and promote re-proves that edge live (WPromote + /// owner==bld) BEFORE it trusts the leaf — so B has in-degree >= 1 and GC (the sole deleter) cannot + /// collect it. This test injects the loss DIRECTLY (raw GC-to-fixpoint after dropping EVERY owner, + /// with build2 not yet precommitted), which the live-precommit invariant excludes. So the dangle is + /// not prevented at promote under §4 — it is DETECTED by fsck (the backstop). + EXPECT_NO_THROW(build2->promote(ns, "part_2", build2->buildId(), id2)); + + /// THE BACKSTOP (INV-NO-DANGLE-via-fsck): fsck's reachable-but-absent scan reports the committed-yet- + /// deleted B as dangling. This is where an absent adopted blob surfaces under §4 — not at the promote + /// gate. Detection moved, it did not disappear. + const FsckReport rep = runFsck(*s, /*detail=*/true); + EXPECT_GE(rep.dangling, 1u) + << "§4 D4 backstop: promote trusts the adopted leaf and commits; the deleted B must surface as an " + "fsck dangling finding (dangling=" << rep.dangling << ", reachable=" << rep.reachable << ")"; +} diff --git a/src/Disks/tests/gtest_cas_gc_log.cpp b/src/Disks/tests/gtest_cas_gc_log.cpp new file mode 100644 index 000000000000..5dbd654343d9 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_log.cpp @@ -0,0 +1,489 @@ +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include + +/// Unit coverage for the CA GC scheduler's logging sink (the source of +/// `system.cas_gc_log`). The scheduler emits a Start + Finish +/// `GcRoundLogRecord` per round through the injected `GcRoundLogger`; here we capture the records in +/// a vector and assert their shape over a real (in-memory) Pool driven through a dropped-then- +/// collectable object — the same Pool/Backend fixture the B140 reclaim test uses. +/// +/// NOTE on ProfileEvents: `runOneRoundNow` runs on THIS (bare gtest) thread, which has no attached +/// `ThreadStatus`, so the scheduler's `CurrentThread::isInitialized()` guard skips per-round +/// ProfileEvents capture. The `profile_events` map is therefore EXPECTED to be empty here and this +/// test does NOT assert it non-empty (the on-server paths are attached; the functional/soak coverage +/// asserts non-empty there). + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; +using Rec = DB::Cas::GcRoundLogRecord; + +namespace +{ + +/// Publish one part `ref` with a single content blob whose payload is `payload`. Returns the manifest id. +ManifestId publishPart(const PoolPtr & s, const String & ns, const String & ref, const String & payload) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(nsr, ref, build->buildId(), id); + return id; +} + +} + +namespace +{ +/// One round emits a Start, then one Phase row per GC phase it reached, then a Finish. Tests that care +/// only about the round-outcome rows filter the phase rows out through this. +std::vector roundRowsOnly(const std::vector & rows) +{ + std::vector out; + for (const Rec & r : rows) + if (r.event_type != Rec::EventType::Phase) + out.push_back(r); + return out; +} +} + +/// The happy path: a marking round (candidates_marked > 0). Each `runOneRoundNow` must emit exactly one +/// Start, then its phase rows, then one Finish, with `disk_name`/`gc_id` set and `duration_ms` +/// populated on the Finish. +/// +/// It drives the PRODUCTION scheduler, so it covers both halves of the pipeline: a MARKING round +/// (candidates condemned, nothing deleted) and, some rounds later once the mount's ack floor graduates +/// them, a DELETING round whose Finish carries the count through. The ordering is asserted, because a +/// deletion reported before its marking would mean the row is not describing the round it names. +TEST(CASGCLog, EmitsStartFinishWithCounts) +{ + auto backend = std::make_shared(); + /// gc_fold_max_defer_rounds=0: this test drives up to 16 consecutive rounds through the scheduler + /// (no direct Gc handle to override per-instance) expecting each to fold; force fold-every-round + /// (Phase-4 Lever A would otherwise defer once the pool quiesces, stalling the mark-then-delete + /// pipeline within the round budget). + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"srv1/tbl"}; + + /// Publish a part, then drop it so its blob/tree become collectable. + publishPart(store, ns.string(), "all_0_0_0", "hello-cas-gc-log"); + store->dropRef(ns, "all_0_0_0"); + /// Advance the durable watermark floor past the build's seq so the build-watermark guard no + /// longer spares the now-dropped objects (the background renewer is off in this test). + store->renewWatermarkOnce(); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + /// Drive rounds until we observe both a marking round and a deletion round. Under the ack-floor + /// pipeline a candidate is marked (condemned) in one round and physically deleted a few rounds later, + /// once the mount's ack floor graduates it — so advance the store's own mount ack after each round + /// (renewWatermarkOnce runs the beat) and give the pipeline a generous round budget. Each + /// runOneRoundNow call appends a Start, the round's phase rows, and a Finish. + bool saw_marked = false; + bool saw_deleted = false; + size_t marking_finish_idx = 0; + size_t deleting_finish_idx = 0; + uint64_t total_deleted = 0; + constexpr size_t max_rounds = 16; + for (size_t round = 0; round < max_rounds; ++round) + { + const size_t before = rows.size(); + sched.runOneRoundNow(Rec::Trigger::Manual); + store->renewWatermarkOnce(); + + /// Each call emits exactly one Start (first) and one Finish (last), with the round's phase rows + /// in between. + ASSERT_GE(rows.size(), before + 2u) << "each round must emit at least a Start and a Finish"; + ASSERT_EQ(rows[before].event_type, Rec::EventType::Start); + ASSERT_EQ(rows.back().event_type, Rec::EventType::Finish); + for (size_t i = before + 1; i + 1 < rows.size(); ++i) + ASSERT_EQ(rows[i].event_type, Rec::EventType::Phase) + << "only Phase rows may sit between a round's Start and Finish"; + + const size_t finish_idx = rows.size() - 1; + const Rec & fin = rows[finish_idx]; + if (!saw_marked && fin.candidates_marked > 0) + { + saw_marked = true; + marking_finish_idx = finish_idx; + } + if (!saw_deleted && fin.objects_deleted > 0) + { + saw_deleted = true; + deleting_finish_idx = finish_idx; + } + total_deleted += fin.objects_deleted; + } + + ASSERT_TRUE(saw_marked) << "expected a round that marked at least one candidate"; + EXPECT_GT(rows[marking_finish_idx].candidates_marked, 0u); + EXPECT_GT(rows[marking_finish_idx].entries_condemned, 0u); + ASSERT_TRUE(saw_deleted) << "expected a round that physically deleted at least one object"; + EXPECT_GE(deleting_finish_idx, marking_finish_idx) + << "an object cannot be reported deleted before the round that condemned it"; + EXPECT_GT(total_deleted, 0u) + << "the deleted count must reach the Finish row, not stop inside the round"; + + /// Identity + timing fields are set on every record. + for (const Rec & r : rows) + { + EXPECT_EQ(r.disk_name, "ca"); + EXPECT_FALSE(r.gc_id.empty()); + EXPECT_EQ(r.trigger, Rec::Trigger::Manual); + } + /// The round-outcome rows alternate Start, Finish, Start, Finish, ... once the phase rows are + /// filtered out; `duration_ms` is meaningful on each Finish (populated unconditionally there). + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size() % 2, 0u); + for (size_t i = 0; i < round_rows.size(); ++i) + EXPECT_EQ(round_rows[i].event_type, + i % 2 == 0 ? Rec::EventType::Start : Rec::EventType::Finish); +} + +namespace +{ + +/// A backend that throws on `list`, the first thing the GC round does (namespace discovery via the +/// roots registry / listing). Used to drive the Aborted-Finish path: the round throws, the scheduler +/// emits an Aborted Finish with the exception text, and `runOneRoundNow` rethrows. +class ThrowingBackend : public InMemoryBackend +{ +public: + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (arm) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend list failure"); + return InMemoryBackend::list(prefix, cursor, limit); + } + + std::optional get(const String & key, Range range) override + { + if (arm) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend get failure"); + return InMemoryBackend::get(key, range); + } + + HeadResult head(const String & key) override + { + if (arm) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend head failure"); + return InMemoryBackend::head(key); + } + + /// Armed only after Pool::open, so opening (which reads/initialises gc state) succeeds. + std::atomic arm{false}; +}; + +} + +/// A7-HIGH-fix: the manual `SYSTEM ... GC` path (runOneRoundNow) reuses ONE stable Gc instance across +/// calls (A7 — the lease's observation-window steal protocol compares consecutive observations of the +/// SAME observer), but it must be OBSERVE-ONLY with respect to STEALING: the protocol's safety argument +/// requires the two observations that flag an incumbent "frozen" to be spaced by real wall time (>= the +/// heartbeat cadence H) so a live incumbent gets a chance to pulse in between — a guarantee only the +/// background loop's own interval-paced ticks provide. Two manual calls have no such guarantee (they +/// can land microseconds apart in a real query), so a manual round must NEVER execute the steal CAS, +/// no matter how many times it re-observes the same frozen tuple. Dead-incumbent recovery stays the +/// loop's job (bounded ~2*interval; covered by the CASGCLease loop-driven steal tests in +/// gtest_cas_gc_round.cpp, e.g. StealAfterObservedNonRenewalBumpsEpoch / FailoverStealOnceHeartbeatStops). +/// Deterministic: "time" is the order of runRegularRound calls; no sleep, no clock, no threads. +TEST(CASGCSchedulerSteal, ManualRoundNeverStealsEvenADeadIncumbent) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + /// A foreign incumbent takes the lease and then DIES (never renews, never heartbeats). + const UInt128 kIncumbent = hexToU128("00000000000000000000000000000abc"); + Gc incumbent(store, kIncumbent); + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + + DB::Cas::CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca"); + + /// obs #1: records the incumbent's (owner, seq, hb=absent). + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + /// obs #2 and #3: the same frozen (owner, seq, hb) observed repeatedly would be steal-eligible on + /// the loop path (see the Core-level test this mirrors), but the manual path keeps backing off. + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); +} + +/// Negative-control companion to the test above (reviewer-requested): with the incumbent visibly alive +/// (its heartbeat advancing between the manual round's observations, exactly like +/// CASGCLease.HeartbeatBlocksFalseStealOfAliveLeader at the Core level), the manual round must still +/// correctly back off — confirming the new observe-only branch didn't regress the PRE-EXISTING +/// incumbent_renewed/hb_alive liveness detection (this test would already pass on the protocol's own +/// terms even without the A7-HIGH-fix; it pins that the fix didn't break it). +TEST(CASGCSchedulerSteal, ManualRoundNeverStealsALiveHeartbeatingIncumbent) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const UInt128 kIncumbent = hexToU128("00000000000000000000000000000abc"); + Gc incumbent(store, kIncumbent); + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + + DB::Cas::CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca"); + + /// obs #1: records (owner=incumbent, seq, hb=absent). + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + Gc::pulseHeartbeat(*store, kIncumbent); /// the incumbent is alive and pulsing (hb 0->1) + /// obs #2: hb advanced since obs #1 => alive => no steal (never reaches the observe-only branch). + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + Gc::pulseHeartbeat(*store, kIncumbent); /// hb 1->2 + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); +} + +/// A round whose backend throws must produce a Finish with `outcome == Aborted` and a non-empty +/// `error`, and `runOneRoundNow` must rethrow the exception (the round failure is observable, not +/// swallowed — the logging sink itself is best-effort, but the round error propagates). +TEST(CASGCLog, AbortedFinishOnThrowingRound) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + backend->arm.store(true); + + EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + + /// A throwing round still emits a Start and an Aborted Finish. It also emits the phase row of the + /// phase it died in -- the timer is RAII, so it fires during unwinding, which is exactly the forensic + /// record a failed round needs. That is also why `round_id`, not `round`, is the correlator: this + /// round has no round number at all. + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size(), 2u) << "a throwing round still emits a Start and a (Aborted) Finish"; + EXPECT_EQ(round_rows[0].event_type, Rec::EventType::Start); + EXPECT_EQ(round_rows[1].event_type, Rec::EventType::Finish); + EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Failed); + EXPECT_FALSE(round_rows[1].error.empty()) << "a failed Finish must carry the exception text"; + EXPECT_EQ(round_rows[1].disk_name, "ca"); + EXPECT_FALSE(round_rows[1].gc_id.empty()); + EXPECT_FALSE(round_rows[1].round_id.empty()); + for (const Rec & r : rows) + EXPECT_EQ(r.round_id, round_rows[0].round_id) + << "every row of a FAILED round must still correlate through round_id"; +} + +/// Every row of one round -- its Start, each of its Phase rows, and its Finish -- carries the SAME +/// non-empty `round_id`, and two rounds carry DIFFERENT ones. That is the property the column exists +/// for: `round` is 0 on Start, is only known after the round's single `gc/state` CAS, and is absent on a +/// round that never led, so it cannot serve as the correlator. +TEST(CASGCLog, EveryRowOfARoundSharesOneRoundId) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"srv1/tbl"}; + publishPart(store, ns.string(), "all_0_0_0", "hello-round-id"); + store->dropRef(ns, "all_0_0_0"); + store->renewWatermarkOnce(); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + sched.runOneRoundNow(Rec::Trigger::Manual); + const size_t after_first = rows.size(); + ASSERT_GE(after_first, 2u); + const String first_id = rows.front().round_id; + EXPECT_FALSE(first_id.empty()); + for (size_t i = 0; i < after_first; ++i) + EXPECT_EQ(rows[i].round_id, first_id) << "row " << i << " of the first round has a different round_id"; + + store->renewWatermarkOnce(); + sched.runOneRoundNow(Rec::Trigger::Manual); + ASSERT_GT(rows.size(), after_first); + const String second_id = rows[after_first].round_id; + EXPECT_FALSE(second_id.empty()); + EXPECT_NE(second_id, first_id) << "two rounds must not share a round_id"; + for (size_t i = after_first; i < rows.size(); ++i) + EXPECT_EQ(rows[i].round_id, second_id); +} + +namespace +{ +/// The phase names of one round, in emission order. +std::vector phaseNames(const std::vector & rows, size_t from) +{ + std::vector out; + for (size_t i = from; i < rows.size(); ++i) + if (rows[i].event_type == Rec::EventType::Phase) + out.push_back(rows[i].phase); + return out; +} + +/// The `phase_metrics` of the named phase of one round. Fails the caller's expectation if absent. +std::map metricsOf(const std::vector & rows, size_t from, const String & phase) +{ + for (size_t i = from; i < rows.size(); ++i) + if (rows[i].event_type == Rec::EventType::Phase && rows[i].phase == phase) + return rows[i].phase_metrics; + return {}; +} +} + +/// A FOLDING round emits every phase, in execution order, and each phase's row carries the semantic +/// counts only that phase can compute. This is the test that would catch an instrumentation site +/// silently dropping out of the round -- a phase that stops emitting reads exactly like a phase that +/// costs nothing, which is the failure mode this whole change exists to prevent. +/// +/// ProfileEvents are deliberately NOT asserted: `runOneRoundNow` runs on the bare gtest thread, which +/// has no attached `ThreadStatus`, so per-phase capture degrades to an empty map exactly as the +/// round-level capture already does (see the note at the top of this file). +TEST(CASGCLog, FoldingRoundEmitsEveryPhaseInOrder) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"srv1/tbl"}; + publishPart(store, ns.string(), "all_0_0_0", "hello-cas-gc-phases"); + store->dropRef(ns, "all_0_0_0"); + store->renewWatermarkOnce(); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + ASSERT_TRUE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + + const std::vector expected = { + "lease", "pre_fold_ref_drain", "heartbeat_floor", "defer_decision", "parent_seal_read", + "fold_ref_group", "fold_seal_read", "fold_ref_intake", + "fold_reduce", "fold_seal_write", + "pending_deletes", "meta_pool_wait", "round_commit", "handoff_reclaim", + "manifest_deletes", "namespace_cleanup", "ref_object_cleanup", "orphan_sweep"}; + EXPECT_EQ(phaseNames(rows, 0), expected); + + /// Every phase row is a Phase row of THIS round and carries a duration field (0 is a legitimate + /// microsecond reading for a phase that did nothing, so only the shape is asserted). + for (const Rec & r : rows) + if (r.event_type == Rec::EventType::Phase) + { + EXPECT_EQ(r.round_id, rows.front().round_id); + EXPECT_FALSE(r.phase.empty()); + EXPECT_TRUE(r.error.empty()); + } + + /// The defer decision reports the signal it decided on, and the two fold-seal reads it paid for. + const auto defer = metricsOf(rows, 0, "defer_decision"); + EXPECT_EQ(defer.at("deferred"), 0u) << "this round folded, so it cannot report itself deferred"; + EXPECT_EQ(defer.at("fold_seal_reads"), 2u); + EXPECT_GT(defer.at("namespaces_seen"), 0u); + + const auto ref_group = metricsOf(rows, 0, "fold_ref_group"); + EXPECT_EQ(ref_group.at("ref_folding_aborted"), 0u); + EXPECT_GT(ref_group.at("ref_keys_listed"), 0u); + + /// Probe B1's identity, as an OBSERVABLE property of the table rather than an assumption in a + /// comment: the round sealed coverage over exactly the logs it folded. + const auto intake = metricsOf(rows, 0, "fold_ref_intake"); + EXPECT_EQ(intake.at("logs_accounted"), intake.at("logs_applied")); + EXPECT_GT(intake.at("logs_applied"), 0u); + EXPECT_GT(intake.at("deltas_emitted"), 0u); + + /// Probe B2's verdict. Nonzero would have thrown, so the row can only ever read 0 on a round that + /// reached its Finish -- which is the point: the column is the round's own attestation. + EXPECT_EQ(metricsOf(rows, 0, "fold_reduce").at("transactions_unapplied"), 0u); + + /// The honest gap: the meta pool's work runs on other threads, so this row's ProfileEvents delta is + /// empty by construction and these two counts are its ONLY signal. They must be real numbers. + const auto meta = metricsOf(rows, 0, "meta_pool_wait"); + EXPECT_GT(meta.at("jobs_scheduled"), 0u) << "this round condemns, so it schedules condemn-marker writes"; + EXPECT_EQ(meta.at("jobs_completed"), meta.at("jobs_scheduled")) + << "every scheduled job must have finished by the time the wait returns"; +} + +/// A round that never leads emits ONLY the phase it reached. `round` does not exist for such a round, +/// so `round_id` is the only thing tying its rows together -- which is why it is the correlator. +TEST(CASGCLog, NotALeaderRoundEmitsOnlyTheLeasePhase) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + /// A foreign incumbent holds the lease, so the scheduler's round backs off immediately. + Gc incumbent(store, hexToU128("00000000000000000000000000000abc")); + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + + EXPECT_EQ(phaseNames(rows, 0), (std::vector{"lease"})); + EXPECT_EQ(metricsOf(rows, 0, "lease").at("acquired"), 0u); + ASSERT_EQ(rows.size(), 3u); + EXPECT_EQ(rows.back().outcome, Rec::Outcome::NotALeader); + for (const Rec & r : rows) + EXPECT_EQ(r.round_id, rows.front().round_id); +} + +/// B3: the scheduler exposes per-disk GC health for system.cas_mounts (the process- +/// global CurrentMetrics gauges were clobbered with >= 2 CAS disks). Drive one leader round and +/// assert the health snapshot reflects leadership, the pending-reclaim backlog and a fresh success. +TEST(CASGCHealth, ReflectsLeadershipAndPendingReclaim) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"srv1/tbl"}; + publishPart(store, ns.string(), "all_0_0_0", "hello-cas-gc-health"); + store->dropRef(ns, "all_0_0_0"); + store->renewWatermarkOnce(); + + DB::Cas::CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca", {}); + + const auto h0 = sched.gcHealth(); + EXPECT_FALSE(h0.is_leader); + EXPECT_FALSE(h0.ever_succeeded); + EXPECT_EQ(h0.pending_reclaim, 0); + EXPECT_EQ(h0.wedged_namespace_count, 0u); + + const RoundReport rep = sched.runOneRoundNow(Rec::Trigger::Manual); + ASSERT_TRUE(rep.acquired_lease); + + const auto h1 = sched.gcHealth(); + EXPECT_TRUE(h1.is_leader); + EXPECT_TRUE(h1.ever_succeeded); + EXPECT_EQ(h1.pending_reclaim, + static_cast(rep.condemned) - static_cast(rep.redeleted)); + EXPECT_EQ(h1.wedged_namespace_count, 0u); + EXPECT_LT(h1.last_success_age_seconds, 60u); +} diff --git a/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp b/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp new file mode 100644 index 000000000000..158040be38c9 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp @@ -0,0 +1,202 @@ +#include "cas_test_helpers.h" +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LIMIT_EXCEEDED; + extern const int UNKNOWN_FORMAT_VERSION; +} + +namespace +{ +class FailingMaintenanceReadBackend : public InMemoryBackend +{ +public: + std::optional get(const String &, Range) override + { + throw std::runtime_error("injected maintenance read failure"); + } +}; +} + +TEST(CASGCMaintenanceStateFormat, RegistryLayoutAndCanonicalCodec) +{ + EXPECT_EQ(static_cast(FormatId::GcMaintenanceState), 25); + const auto points = changePoints(FormatId::GcMaintenanceState); + ASSERT_EQ(points.size(), 1u); + EXPECT_EQ(points[0].generation, 7); + EXPECT_EQ(points[0].min_reader, 7); + const FormatTraits & traits = traitsFor(FormatId::GcMaintenanceState); + EXPECT_EQ(traits.type, "cas_gc_maintenance_state"); + EXPECT_EQ(traits.family, TextFamily::Control); + EXPECT_EQ(traits.strictness, KeyStrictness::Strict); + EXPECT_EQ(traits.compression, CompressionPolicy::Never); + EXPECT_EQ(traits.object_cap, 512 * 1024); + EXPECT_EQ(traits.line_cap, 512 * 1024); + EXPECT_EQ(storedSuffix(FormatId::GcMaintenanceState), ""); + EXPECT_EQ(traitsForType("cas_gc_maintenance_state"), &traits); + + const Layout layout("p"); + EXPECT_EQ(layout.gcMaintenanceStateKey(), "p/gc/maintenance_state"); + EXPECT_NE(layout.gcMaintenanceStateKey(), layout.gcStateKey()); + EXPECT_NE(layout.gcMaintenanceStateKey(), layout.gcHbKey()); + + const GcMaintenanceState empty; + EXPECT_EQ(encodeGcMaintenanceState(empty), fmt::format( + "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"cur\":\"\"}}\n", currentCompatibilityVersion())); + const GcMaintenanceState state{.janitor_cursor = R"(cas/ns/a/"quoted"\\next)"}; + EXPECT_EQ(decodeGcMaintenanceState(encodeGcMaintenanceState(state)), state); +} + +TEST(CASGCMaintenanceStateFormat, RejectsMalformedAndBoundsCursor) +{ + const auto bad = [](std::string_view body) + { + return "{\"type\":\"cas_gc_maintenance_state\",\"v\":7}\n" + String(body); + }; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(bad("{}\n")); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\",\"cur\":\"b\"}\n")); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\",\"extra\":1}\n")); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\"}\nx")); }); + + const GcMaintenanceState at_limit{.janitor_cursor = String(kMaxGcMaintenanceCursorBytes, 'x')}; + EXPECT_EQ(decodeGcMaintenanceState(encodeGcMaintenanceState(at_limit)), at_limit); + const GcMaintenanceState over_limit{.janitor_cursor = String(kMaxGcMaintenanceCursorBytes + 1, 'x')}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, + [&] { (void)encodeGcMaintenanceState(over_limit); }); + const String raw = "{\"type\":\"cas_gc_maintenance_state\",\"v\":7}\n{\"cur\":\"" + over_limit.janitor_cursor + "\"}\n"; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(raw); }); + String oversized = R"({"type":"cas_gc_maintenance_state","v":7,"pad":")"; + oversized.append(448 * 1024, 'x'); + oversized += "\"}\n{\"cur\":\""; + oversized.append(kMaxGcMaintenanceCursorBytes, 'y'); + oversized += "\"}\n"; + ASSERT_GT(oversized.size(), traitsFor(FormatId::GcMaintenanceState).object_cap); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeGcMaintenanceState(oversized); }); +} + +TEST(CASGCMaintenanceState, ReadsAndCasWithoutAdoptingConflicts) +{ + InMemoryBackend backend; + const Layout layout("p"); + const String key = layout.gcMaintenanceStateKey(); + const GcMaintenanceReadResult absent = readGcMaintenanceState(backend, layout); + EXPECT_EQ(absent.status, GcMaintenanceReadStatus::Absent); + EXPECT_FALSE(absent.state); + EXPECT_FALSE(absent.token); + + const GcMaintenanceState first{.janitor_cursor = "cas/ns/first"}; + const GcMaintenanceCasResult created = casGcMaintenanceState(backend, layout, std::nullopt, first); + EXPECT_EQ(created.outcome, GcMaintenanceCasOutcome::Committed); + const GcMaintenanceReadResult valid = readGcMaintenanceState(backend, layout); + ASSERT_EQ(valid.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(valid.token); + ASSERT_TRUE(valid.state); + EXPECT_EQ(*valid.state, first); + + const GcMaintenanceCasResult advanced = casGcMaintenanceState(backend, layout, valid.token, + GcMaintenanceState{.janitor_cursor = "cas/ns/advanced"}); + ASSERT_EQ(advanced.outcome, GcMaintenanceCasOutcome::Committed); + const auto advanced_body = backend.get(key); + ASSERT_TRUE(advanced_body); + + ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), advanced_body->token).outcome, + CasOutcome::Committed); + const GcMaintenanceCasResult conflict = casGcMaintenanceState(backend, layout, valid.token, + GcMaintenanceState{.janitor_cursor = "loser"}); + EXPECT_EQ(conflict.outcome, GcMaintenanceCasOutcome::Conflict); + EXPECT_EQ(decodeGcMaintenanceState(backend.get(key)->bytes).janitor_cursor, "winner"); +} + +TEST(CASGCMaintenanceState, ClassifiesCorruptionAndResetsOnlyExactToken) +{ + InMemoryBackend backend; + const Layout layout("p"); + const String key = layout.gcMaintenanceStateKey(); + ASSERT_EQ(backend.putIfAbsent(key, "malformed").outcome, PutOutcome::Done); + const GcMaintenanceReadResult corrupt = readGcMaintenanceState(backend, layout); + ASSERT_EQ(corrupt.status, GcMaintenanceReadStatus::Corrupt); + ASSERT_TRUE(corrupt.token); + EXPECT_FALSE(corrupt.state); + EXPECT_FALSE(corrupt.diagnostic.empty()); + ASSERT_EQ(casGcMaintenanceState(backend, layout, corrupt.token, {}).outcome, GcMaintenanceCasOutcome::Committed); + EXPECT_EQ(decodeGcMaintenanceState(backend.get(key)->bytes), GcMaintenanceState{}); +} + +TEST(CASGCMaintenanceState, UsesExactlyOneReadOrCasAttempt) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const String key = layout.gcMaintenanceStateKey(); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent); + EXPECT_EQ(backend.getCount(key), 1u); + + backend.resetCounts(); + ASSERT_EQ(casGcMaintenanceState(backend, layout, std::nullopt, {}).outcome, + GcMaintenanceCasOutcome::Committed); + EXPECT_EQ(backend.casPutCount(key), 1u); + EXPECT_EQ(backend.getCount(key), 0u); + + backend.resetCounts(); + EXPECT_EQ(casGcMaintenanceState(backend, layout, std::nullopt, + GcMaintenanceState{.janitor_cursor = "loser"}).outcome, GcMaintenanceCasOutcome::Conflict); + EXPECT_EQ(backend.casPutCount(key), 1u); + EXPECT_EQ(backend.getCount(key), 0u); + + const auto current = backend.get(key); + ASSERT_TRUE(current); + ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), current->token).outcome, + CasOutcome::Committed); + backend.resetCounts(); + EXPECT_EQ(casGcMaintenanceState(backend, layout, current->token, + GcMaintenanceState{.janitor_cursor = "stale"}).outcome, GcMaintenanceCasOutcome::Conflict); + EXPECT_EQ(backend.casPutCount(key), 1u); + EXPECT_EQ(backend.getCount(key), 0u); + EXPECT_EQ(decodeGcMaintenanceState(backend.InMemoryBackend::get(key)->bytes).janitor_cursor, "winner"); +} + +TEST(CASGCMaintenanceState, FutureVersionPropagatesInsteadOfResetting) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const String key = layout.gcMaintenanceStateKey(); + ASSERT_EQ(backend.putIfAbsent(key, fmt::format( + "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"cur\":\"\"}}\n", currentCompatibilityVersion() + 1)).outcome, + PutOutcome::Done); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, + [&] { (void)readGcMaintenanceState(backend, layout); }); + EXPECT_EQ(backend.casPutCount(key), 0u); + + FailingMaintenanceReadBackend failing; + EXPECT_THROW((void)readGcMaintenanceState(failing, layout), std::runtime_error); +} + +TEST(CASGCMaintenanceState, LosingCorruptResetPreservesConcurrentWinner) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const String key = layout.gcMaintenanceStateKey(); + ASSERT_EQ(backend.putIfAbsent(key, "corrupt").outcome, PutOutcome::Done); + const auto corrupt = readGcMaintenanceState(backend, layout); + ASSERT_EQ(corrupt.status, GcMaintenanceReadStatus::Corrupt); + ASSERT_TRUE(corrupt.token); + ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), corrupt.token).outcome, + CasOutcome::Committed); + backend.resetCounts(); + EXPECT_EQ(casGcMaintenanceState(backend, layout, corrupt.token, {}).outcome, + GcMaintenanceCasOutcome::Conflict); + EXPECT_EQ(backend.casPutCount(key), 1u); + EXPECT_EQ(backend.getCount(key), 0u); + EXPECT_EQ(decodeGcMaintenanceState(backend.InMemoryBackend::get(key)->bytes).janitor_cursor, "winner"); +} diff --git a/src/Disks/tests/gtest_cas_gc_meta_writer.cpp b/src/Disks/tests/gtest_cas_gc_meta_writer.cpp new file mode 100644 index 000000000000..2e98c3477419 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_meta_writer.cpp @@ -0,0 +1,188 @@ +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +using namespace DB::Cas; +using DB::Cas::tests::MetaWriteLatchBackend; +using DB::Cas::tests::awaitLatchEntered; + +namespace +{ +constexpr auto kGcId = "0000000000000000000000000000002a"; + +static_assert(noexcept(std::declval().drainOnExitNoThrow()), + "round-exit meta-pool cleanup must not throw from the scope guard"); +} + +/// A real condemn-marker job may be in flight when its `Gc` is destroyed. The job holds everything it +/// touches, so the pool's join completes it correctly rather than racing member teardown -- and the +/// marker it was writing is durable afterwards. +/// +/// This asserts function, not ordering: the release may land before, during or after destruction +/// begins, and all three are sound. Nothing here detects a job that wrongly captured its owner -- +/// that is prevented by there being no API to write one. +TEST(CASGcMetaWriter, RealCondemnMarkerJobCompletesAcrossOwnerDestruction) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const BlobRef ref = DB::Cas::tests::idOf("1"); + const Token token{"tok-1"}; + + auto gc = std::make_unique(store, DB::Cas::tests::u128Of(kGcId)); + backend->arm(); + gc->metaWriterForTest().scheduleCondemnMarkerWrite(ref, token, /*condemn_round=*/1, /*size=*/128); + + awaitLatchEntered(*backend); + + std::thread releaser([&] { backend->release(); }); + gc.reset(); + releaser.join(); + + const auto meta = loadMeta(*backend, store->layout(), ref); + ASSERT_TRUE(meta) << "the condemn marker was lost across owner destruction"; + EXPECT_EQ(meta->meta.state, MetaState::Condemned); + EXPECT_EQ(meta->meta.condemn_round, 1u); +} + +/// The confirmation registry is written by the pool thread and read by the graduation gate. Assert it +/// on a `Gc` that is still alive, so the read is possible at all: after destruction there is no +/// registry left to consult, which is the documented behaviour a fresh leader relies on. +TEST(CASGcMetaWriter, CondemnMarkerConfirmationIsVisibleAfterDrain) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const BlobRef ref = DB::Cas::tests::idOf("1"); + const Token token{"tok-1"}; + + Gc gc(store, DB::Cas::tests::u128Of(kGcId)); + EXPECT_FALSE(gc.metaWriterForTest().condemnMarkerConfirmedInProcess(ref, token)); + + gc.metaWriterForTest().scheduleCondemnMarkerWrite(ref, token, /*condemn_round=*/1, /*size=*/128); + gc.metaWriterForTest().drain(); + + EXPECT_TRUE(gc.metaWriterForTest().condemnMarkerConfirmedInProcess(ref, token)); + EXPECT_EQ(gc.metaWriterForTest().scheduled(), gc.metaWriterForTest().completed()); +} + +/// Same lifetime property for the other production job. `deleteConfirmedMeta` RETURNS IMMEDIATELY when +/// no meta object exists (`Gc/CasGcMetaWriter.cpp`), so the meta must be seeded first -- otherwise the +/// job never reaches the latch and the wait above is waiting for something that will never happen. +TEST(CASGcMetaWriter, RealConfirmedMetaDeleteCompletesAcrossOwnerDestruction) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const BlobRef ref = DB::Cas::tests::idOf("2"); + ASSERT_EQ( + putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 64}).outcome, + CasOverwriteOutcome::Committed); + ASSERT_TRUE(loadMeta(*backend, store->layout(), ref)); + + auto gc = std::make_unique(store, DB::Cas::tests::u128Of(kGcId)); + backend->arm(); + gc->metaWriterForTest().scheduleConfirmedMetaDelete(ref); + + awaitLatchEntered(*backend); + + std::thread releaser([&] { backend->release(); }); + gc.reset(); + releaser.join(); + + EXPECT_FALSE(loadMeta(*backend, store->layout(), ref)) + << "the confirmed-meta delete was lost across owner destruction"; +} + +/// A round that throws must not leave its meta jobs running into the next round: their effects would +/// land in the registry the next round's graduation gate reads, and inside its counter deltas. +/// +/// The round is made to throw at its outcome-log write, with the confirmed-meta delete it scheduled a +/// few lines earlier held inside the backend. The round must then BLOCK, draining, until that job is +/// released -- so the test asserts the round has NOT returned while the job is still held, releases, +/// and only then joins. +TEST(CASGcMetaWriter, ThrowingRoundDrainsBeforeReturning) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + /// Fixture: one part written and dropped, then rounds driven until the NEXT round is the one that + /// deletes -- the round that both schedules a confirmed-meta delete and writes an outcome log. + const RootNamespace ns{"test/tbl"}; + const String ref_name = "all_0_0_0"; + const String payload = "round-drain-payload"; + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref_name; + auto build = store->beginPartWrite(info); + ManifestEntry entry; + entry.path = "data.bin"; + entry.placement = EntryPlacement::Blob; + entry.ref = DB::Cas::tests::idOf(payload); + entry.blob_size = payload.size(); + const ManifestId manifest_id = build->stageManifest({entry}); + build->precommitAdd(ns, ref_name, manifest_id); + build->putBlob(entry.ref, BlobSource::fromString(payload)); + build->promote(ns, ref_name, build->buildId(), manifest_id); + store->dropRef(ns, ref_name); + store->renewWatermarkOnce(); + + Gc gc(store, DB::Cas::tests::u128Of(kGcId)); + + size_t rounds = 0; + while (true) + { + bool delete_pending = false; + for (const auto & entry_to_delete : gc.previewDeletes()) + delete_pending |= entry_to_delete.reason == "delete_pending"; + if (delete_pending) + break; + + ASSERT_LT(++rounds, 16u) << "no round ever reached a pending delete -- fixture is wrong"; + ASSERT_NO_THROW(gc.runRegularRound()); + store->renewWatermarkOnce(); + } + + const uint64_t scheduled_before = gc.metaWriterForTest().scheduled(); + + backend->arm(); + backend->fail_outcome_logs.store(true); + + /// Return the outcome instead of asserting on the worker thread: a gtest assertion raised off the + /// main thread is not reliably reported, and this one distinguishes the two ways the test can go + /// wrong, so it must be visible. + auto round = std::async(std::launch::async, [&] + { + try + { + gc.runRegularRound(); + return false; + } + catch (...) + { + return true; + } + }); + + awaitLatchEntered(*backend); + EXPECT_GT(gc.metaWriterForTest().scheduled(), scheduled_before) + << "the faulted round scheduled no meta job -- it cannot be the deleting round"; + + EXPECT_EQ(round.wait_for(std::chrono::seconds(2)), std::future_status::timeout) + << "the round returned while a meta job was still in flight -- it did not drain on its " + "throwing exit"; + + backend->release(); + EXPECT_TRUE(round.get()) + << "the round completed normally -- the outcome-log fault never fired, so the timeout above " + "was the round blocking in its own `meta_pool_wait`, not in the drain under test"; + + EXPECT_EQ(gc.metaWriterForTest().scheduled(), gc.metaWriterForTest().completed()); +} diff --git a/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp b/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp new file mode 100644 index 000000000000..e6530f131299 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp @@ -0,0 +1,101 @@ +#include "cas_format_test_battery.h" +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ + +/// Same tiny inline copy as `gtest_cas_part_manifest_format.cpp`'s `expectThrowsCode`: stays clear +/// of `Disks/tests/cas_test_helpers.h`, which would drag in the whole CAS backend/store machinery +/// this file otherwise has no need for. +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} + +} + +TEST(CASFormatBattery, GcOutcomes) +{ + OutcomeLog log; + OutcomeEntry e; + e.kind = ObjectKind::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("00112233445566778899aabbccddeeff"))}; + e.token = Token{"e-1", TokenType::ETag}; + e.outcome = OutcomeKind::Deleted; + log.entries.push_back(e); + runFormatBattery({FormatId::GcOutcomes, + [&] { return sealObject(FormatId::GcOutcomes, encodeOutcomeLog(log)); }, + [](std::string_view d) { decodeOutcomeLog(std::string(openObject(FormatId::GcOutcomes, d))); }, + currentFormatHeader("cas_gc_outcomes") + + "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\"," + "\"tt\":\"etag\",\"tv\":\"e-1\",\"oc\":\"deleted\"}\n{\"n\":1}\n"}); +} + +TEST(CASGCOutcomesFormat, EmptyRoundTrips) +{ + EXPECT_EQ(decodeOutcomeLog(encodeOutcomeLog(OutcomeLog{})).entries.size(), 0u); +} + +TEST(CASGCOutcomesFormat, MultiEntryRoundTripAllOutcomes) +{ + OutcomeLog log; + log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("aa00000000000000000000000000000a"))}, + Token{"etag-1", TokenType::ETag}, OutcomeKind::Deleted}); + log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("bb00000000000000000000000000000b"))}, + Token{"7", TokenType::Emulated}, OutcomeKind::Spared}); + log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("cc00000000000000000000000000000c"))}, + Token{"8", TokenType::Emulated}, OutcomeKind::Replaced}); + log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("dd00000000000000000000000000000d"))}, + Token{"9", TokenType::Emulated}, OutcomeKind::Absent}); + const String text = encodeOutcomeLog(log); + const OutcomeLog d = decodeOutcomeLog(text); + ASSERT_EQ(d.entries.size(), 4u); + EXPECT_EQ(d.entries[0].ref, (BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("aa00000000000000000000000000000a"))})); + EXPECT_EQ(d.entries[0].outcome, OutcomeKind::Deleted); + EXPECT_EQ(d.entries[1].outcome, OutcomeKind::Spared); + EXPECT_EQ(d.entries[2].outcome, OutcomeKind::Replaced); + EXPECT_EQ(d.entries[3].outcome, OutcomeKind::Absent); + EXPECT_EQ(d.entries[0].token.value, "etag-1"); + EXPECT_EQ(d.entries[0].token.type, TokenType::ETag); + EXPECT_EQ(d.entries[3].token.value, "9"); + /// Insertion order + byte-stable text (the encoder is a pure function of the log). + EXPECT_EQ(encodeOutcomeLog(d), text); +} + +TEST(CASGCOutcomesFormat, GarbageAndUnknownWordsFailClosed) +{ + EXPECT_THROW(decodeOutcomeLog(String("")), DB::Exception); + EXPECT_THROW(decodeOutcomeLog(String("not a cas object\n")), DB::Exception); + /// A record with an unknown outcome word fails closed. + const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n" + "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\"," + "\"tt\":\"etag\",\"tv\":\"x\",\"oc\":\"bogus\"}\n{\"n\":1}\n"; + EXPECT_THROW(decodeOutcomeLog(bad), DB::Exception); + /// A trailer count mismatch fails closed. + const String miscount = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n{\"n\":5}\n"; + EXPECT_THROW(decodeOutcomeLog(miscount), DB::Exception); +} + +TEST(CASGCOutcomesFormat, DigestWidthMismatchFailsClosedWithCorruptedData) +{ + /// `ch128` (CityHash128) digests are 16 bytes = 32 hex chars; here the "h" field is truncated + /// to 30 hex chars. Must surface as CORRUPTED_DATA (malformed serialized input), not + /// `fromHex`'s BAD_ARGUMENTS. + const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n" + "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddee\"," + "\"tt\":\"etag\",\"tv\":\"x\",\"oc\":\"deleted\"}\n{\"n\":1}\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeOutcomeLog(bad); }); +} diff --git a/src/Disks/tests/gtest_cas_gc_rebuild.cpp b/src/Disks/tests/gtest_cas_gc_rebuild.cpp new file mode 100644 index 000000000000..aa896dbf68df --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_rebuild.cpp @@ -0,0 +1,681 @@ +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace ProfileEvents +{ +extern const Event CASGCRefWalkPlansBuilt; +} + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +namespace +{ +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} +} + +/// (`CASGCBaselineGuard.FreshStateOverTrimmedJournalsFailsClosed` was removed with the snapshot+log ref +/// model. It asserted that a fresh GC over a MUTABLE shard journal whose folded history had been TRIMMED +/// must refuse, lest it fold only the surviving tails and mass-delete live data. Immutable `_log`/`_snap` +/// objects are never trimmed in place: a fresh GC always reconstructs the FULL ref state via the recovery +/// equation (newest snapshot + later log tail), so the "trimmed history" hazard cannot arise. The +/// vanished-`gc/state` disaster-recovery path is covered by `CASGCRebuild.RecoversLostStateAndConverges`, +/// and the corrupt-bookkeeping guard by `CASGCBaselineGuard.AbsentAdoptedSealFailsClosed`.) + +/// A genuinely fresh pool (journals start at version 1) passes the guard — rounds run as today. +TEST(CASGCBaselineGuard, GenuinelyFreshPoolIsUnaffected) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + EXPECT_NO_THROW(gc.runRegularRound()); + EXPECT_NO_THROW(gc.runRegularRound()); +} + +/// (б) audit: snap_generation > 0 whose adopted fold seal is ABSENT must be CORRUPTED_DATA, +/// never silently treated as an empty baseline. +TEST(CASGCBaselineGuard, AbsentAdoptedSealFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + + /// Corrupt (б): delete the adopted fold seal out from under a healthy gc/state. + const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + ASSERT_GT(st.snap_generation, 0u); + const String seal_key = store->layout().foldSealKey(st.snap_generation, st.snap_attempt); + const HeadResult sh = backend->head(seal_key); + ASSERT_TRUE(sh.exists); + ASSERT_EQ(backend->deleteExact(seal_key, sh.token).kind, DeleteOutcome::Kind::Deleted); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.runRegularRound(); }); +} + +/// (а): lose gc/state on a lived-in pool -> guard blocks rounds -> rebuild -> rounds converge: the +/// live blob intact, the round minted strictly above the last one seen, and the REBUILT baseline is a +/// working one — a ref dropped AFTER it is still reclaimed by ordinary rounds. +/// +/// A blob dropped BEFORE the rebuild is a different matter, and this test pins it: the rebuild derives +/// edges from owner state, so a blob no owner names gets no row at all, and a rebuild CONDEMNS NOTHING +/// (spec §7 — the condemnation that used to catch this case was the r5-finding-4 data-loss vector). +/// Such a blob is retained until register R4's build/upload registry can enumerate it safely. That is +/// the NAMED Stage-A residual, and it is asserted here rather than left to be discovered. +TEST(CASGCRebuild, RecoversLostStateAndConverges) +{ + auto backend = std::make_shared(); + /// gc_fold_max_defer_rounds=0: this test drives MANY consecutive rounds via runRoundsUntilAbsent + /// expecting every one to fold (Phase-4 Lever A would otherwise defer once the pool quiesces, + /// stalling the reclaim loop below the 8-round budget); force fold-every-round. + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef live_r = ref(1, 0xA1); + const ManifestRef dead_r = ref(2, 0xA2); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); + writeManifestRaw(*backend, store->layout(), ns, live_r, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, dead_r, {blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_live", std::nullopt, live_r); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_dead", std::nullopt, dead_r); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + dropRefTransition(*backend, store->layout(), ns, "tbl_dead", dead_r); + runRegularRoundReclaiming(gc); /// -1 folds; eager trim cuts the journal + store->renewWatermarkOnce(); /// renews the lease + build-watermark floor + + /// Capture the round reached before gc/state is destroyed (the rebuild must mint strictly above it). + const auto pre_rebuild_got = backend->get(store->layout().gcStateKey()); + ASSERT_TRUE(pre_rebuild_got.has_value()); + const uint64_t pre_rebuild_round = decodeGcState(pre_rebuild_got->bytes).round; + ASSERT_EQ(backend->deleteExact(store->layout().gcStateKey(), pre_rebuild_got->token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc2(store, hexToU128("00000000000000000000000000000003")); + /// A fresh GC over the orphaned generation artifacts fails closed: re-folding from a fresh gc/state + /// collides with a leftover run object (divergent bytes) — the disaster is surfaced, never silently + /// double-applied. That first round also re-mints a superficially-healthy gc/state (snap_generation 0), + /// so the recovery is a DELIBERATE force-rebuild (the auto-rebuild correctly refuses to discard a + /// state that "looks healthy" without the operator's force). + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { runRegularRoundReclaiming(gc2); }); + + const RebuildReport rep = gc2.rebuildBaseline(/*force*/ true); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.committed_refs, 1u); + EXPECT_EQ(rep.namespaces, 1u); + + /// Round strictly above the fence/state/generation numbers seen so far. + EXPECT_GT(rep.round, pre_rebuild_round); + + /// The live blob is intact, and blob 2 — dropped BEFORE the rebuild, so invisible to a baseline + /// derived from owner state — is RETAINED. Retention, not loss: the named residual above. + for (int i = 0; i < 4; ++i) + { + runRegularRoundReclaiming(gc2); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).exists); + EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(2))})).exists) + << "a rebuild condemns nothing, so a pre-rebuild drop is retained — never reclaimed by a " + "substitute pass, and never lost"; + + /// The rebuilt baseline is a WORKING one: a ref published over it and then dropped still folds to + /// zero and is reclaimed by ordinary rounds. Without this the test would prove only that the + /// pipeline stopped deleting. + const ManifestRef post_r = ref(3, 0xA3); + writeBlobBody(*backend, store->layout(), DB::UInt128(3)); + writeManifestRaw(*backend, store->layout(), ns, post_r, {blobEntryFor("c", DB::UInt128(3))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_post", std::nullopt, post_r); + runRegularRoundReclaiming(gc2); + dropRefTransition(*backend, store->layout(), ns, "tbl_post", post_r); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc2, *backend, store->layout(), DB::UInt128(3))) + << "the rebuilt baseline must still reclaim what it can actually see"; +} + +/// (б): a run object named by a healthy state is lost -> the regular round fails closed -> the +/// PLAIN rebuild (no FORCE) recovers, and rounds converge afterwards. +TEST(CASGCRebuild, RecoversLostGenerationArtifact) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + + /// Lose one snapshot run object out from under the healthy state. + const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto seal = decodeFoldSeal(backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + ASSERT_FALSE(seal.blob_target_runs.empty()); + const String run_key = seal.blob_target_runs.front().key; + const HeadResult rh = backend->head(run_key); + ASSERT_TRUE(rh.exists); + ASSERT_EQ(backend->deleteExact(run_key, rh.token).kind, DeleteOutcome::Kind::Deleted); + + /// A pure ref-carry round would not read the lost run; land a REAL delta so the fold's + /// three-cursor merge must stream the prior run — and fails closed on its absence. + const ManifestRef r2 = ref(2, 0xB7); + writeBlobBody(*backend, store->layout(), DB::UInt128(3)); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("c", DB::UInt128(3))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.runRegularRound(); }); + + const RebuildReport rep = gc.rebuildBaseline(/*force*/ false); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_NO_THROW(gc.runRegularRound()); + EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).exists); +} + +/// FORCE: a healthy state refuses the plain rebuild; FORCE rebuilds; rounds run clean after. +TEST(CASGCRebuild, HealthyStateRequiresForce) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + + const RebuildReport refused = gc.rebuildBaseline(/*force*/ false); + EXPECT_FALSE(refused.performed); + EXPECT_NE(refused.refusal.find("FORCE"), String::npos); + + backend->resetCounts(); + const uint64_t plans_before + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + const RebuildReport forced = gc.rebuildBaseline(/*force*/ true); + ASSERT_TRUE(forced.performed) << forced.refusal; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u) + << "healthy FORCE REBUILD must share the same one-shot authoritative plan builder"; + EXPECT_EQ(backend->getCount(store->layout().refCatalogKey()), 2u) + << "healthy FORCE REBUILD may read the catalog for the conclusive drain and the one post-LIST cut only"; + EXPECT_EQ(backend->listCount(store->layout().namespaceStreamRootPrefix()), 1u); + EXPECT_NO_THROW(gc.runRegularRound()); +} + +/// The post-LIST catalog cut and its exact `_ckpt` are REBUILD's authority. A visible later log is +/// not admitted merely because a LIST would find it: it may be a durable-but-unfrontiered writer +/// attempt, and folding its missing manifest would turn a safe rebuild into a false refusal. +/// +/// This catches a regression back to the legacy `recoverRefTable` overload, whose full LIST-derived +/// replay consumes the second transaction and therefore refuses on `unfrontiered`'s missing body. +TEST(CASGCRebuild, FrozenCheckpointFrontierExcludesVisibleUnfrontieredTail) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/rebuild-frozen-frontier@cas@"}; + const UInt128 life_id{0xF001}; + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, life_id); + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); + + const ManifestRef admitted = ref(1, 0xA1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns, admitted, {blobEntryFor("a", DB::UInt128(1))}); + + std::vector first_ops{namespaceBirthOp()}; + const auto admitted_ops = publishCommittedOps("admitted", admitted); + first_ops.insert(first_ops.end(), admitted_ops.begin(), admitted_ops.end()); + fixture::writeRefLogRaw(*backend, layout, + RefLogTxn{.ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = std::move(first_ops), .prev_epoch_seal = std::nullopt}); + + /// This record is real and listable, but the writer never published it through `_ckpt`. + const ManifestRef unfrontiered = ref(2, 0xB2); + fixture::writeRefLogRaw(*backend, layout, + RefLogTxn{.ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = publishCommittedOps("unfrontiered", unfrontiered), + .prev_epoch_seal = std::nullopt}); + ASSERT_TRUE(backend->head(layout.refLogKey(life, RefTxnId{1, 2})).exists); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + Gc gc(store, kGc); + const RebuildReport report = gc.rebuildBaseline(/*force=*/false); + + ASSERT_TRUE(report.performed) << report.refusal; + EXPECT_EQ(report.committed_refs, 1u); +} + +/// A catalog-admitted life without its exact checkpoint has no bounded recovery frontier. REBUILD +/// must refuse rather than falling back to a list-derived history and publishing a baseline it cannot +/// prove complete. +TEST(CASGCRebuild, LiveCatalogLifeWithoutCheckpointFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/rebuild-missing-checkpoint@cas@"}; + const UInt128 life_id{0xF002}; + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); + + const ManifestRef admitted = ref(1, 0xA2); + writeBlobBody(*backend, layout, DB::UInt128(2)); + writeManifestRaw(*backend, layout, ns, admitted, {blobEntryFor("a", DB::UInt128(2))}); + std::vector ops{namespaceBirthOp()}; + const auto admitted_ops = publishCommittedOps("admitted", admitted); + ops.insert(ops.end(), admitted_ops.begin(), admitted_ops.end()); + fixture::writeRefLogRaw(*backend, layout, + RefLogTxn{.ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = std::move(ops), .prev_epoch_seal = std::nullopt}); + + Gc gc(store, kGc); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)gc.rebuildBaseline(/*force=*/false); }); + + const auto state = backend->get(layout.gcStateKey()); + ASSERT_TRUE(state); + EXPECT_EQ(decodeGcState(state->bytes).snap_generation, 0u) + << "a rejected recovery must not adopt a new baseline"; +} + +/// A syntactically valid snapshot at an OLDER `EpochSeal` id must not let REBUILD synthesize a +/// baseline. The forged base differs from `last_epoch_seal`, so metadata equality cannot reject it; +/// REBUILD must use the retained same-id log witness. +TEST(CASGCRebuild, CheckpointSnapshotAtOlderEpochSealFailsClosed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/rebuild-checkpoint-base-seal@cas@"}; + const UInt128 life_id{0xF003}; + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, life_id); + + const RefLogTxn birth{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, birth); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + const RefLogTxn seal_txn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = {seal}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, seal_txn); + RefOp later_seal; + later_seal.kind = RefOpKind::EpochSeal; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{2, 1}, .ops = {later_seal}, + .prev_epoch_seal = RefTxnId{1, 2}}); + RefTableState through_seal; + applyRefLogTxn(through_seal, birth); + applyRefLogTxn(through_seal, seal_txn); + writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); + + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + + Gc gc(store, kGc); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)gc.rebuildBaseline(/*force=*/false); }); + + const auto state = backend->get(layout.gcStateKey()); + ASSERT_TRUE(state); + EXPECT_EQ(decodeGcState(state->bytes).snap_generation, 0u) + << "a rejected checkpoint base must not publish a REBUILD baseline"; +} + +TEST(CASGCRebuild, DamagedGenerationZeroStatePerformsNoCatalogDrainMutation) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/removing-without-parent@cas@"}; + const UInt128 life_id{91}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + .ns = ns, .state = NsState::Live, .incarnation = life_id}); + CasRefCatalog::casUpdate(*backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(ns, life_id)), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + const uint64_t plans_before + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + + Gc gc(store, kGc); + const RebuildReport report = gc.rebuildBaseline(/*force*/ false); + ASSERT_TRUE(report.performed) << report.refusal; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(*backend, layout); + ASSERT_EQ(catalog.catalog.entries.size(), 1u); + EXPECT_EQ(catalog.catalog.entries[0].state, NsState::Removing); + EXPECT_EQ(catalog.catalog.entries[0].incarnation, life_id); +} + +/// Refusal: a committed owner with a MISSING manifest body is data loss — the rebuild refuses, +/// names the owner, and writes nothing (gc/state stays absent). +TEST(CASGCRebuild, MissingCommittedManifestRefuses) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef a = ref(1, 0xA1); + const ManifestRef b = ref(2, 0xB2); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, a, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, b, {blobEntryFor("b", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_a", std::nullopt, a); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_b", std::nullopt, b); + Gc gc(store, kGc); + gc.runRegularRound(); + gc.runRegularRound(); /// trim + + /// Disaster pair: gc/state lost AND tbl_b's manifest body lost. + const HeadResult st = backend->head(store->layout().gcStateKey()); + backend->deleteExact(store->layout().gcStateKey(), st.token); + const String mkey = store->layout().manifestKey(ManifestId{ns, b}); + const HeadResult mh = backend->head(mkey); + ASSERT_TRUE(mh.exists); + backend->deleteExact(mkey, mh.token); + + Gc gc2(store, hexToU128("00000000000000000000000000000004")); + const RebuildReport rep = gc2.rebuildBaseline(/*force*/ false); + EXPECT_FALSE(rep.performed); + EXPECT_NE(rep.refusal.find("tbl_b"), String::npos) << rep.refusal; + /// The lease acquire minted a gen-0 bootstrap body (that is the acquire's contract, not the + /// rebuild's); the rebuild's own contract is that NO baseline was blessed by the refusal. + const auto post = backend->get(store->layout().gcStateKey()); + ASSERT_TRUE(post.has_value()); + const GcState post_state = decodeGcState(post->bytes); + EXPECT_EQ(post_state.snap_generation, 0u) << "a refused rebuild must not adopt a baseline"; +} + +/// A live precommit with a durable body contributes edges (no clamp); the rebuilt baseline +/// protects its blob from condemnation. +TEST(CASGCRebuild, LivePrecommitEdgesIncluded) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef pre = ref(7, 0xC1); + writeBlobBody(*backend, store->layout(), DB::UInt128(9)); + writeManifestRaw(*backend, store->layout(), ns, pre, {blobEntryFor("p", DB::UInt128(9))}); + addPrecommitTransition( + *backend, store->layout(), ns, /*build_id*/ DB::UInt128(0x77), "part_pre", std::nullopt, pre); + + /// No round before the rebuild: the journal still carries the create-precommit event (a round's + /// eager trim would cut it — the trimmed-but-live case is the next test). gc/state absent => + /// the plain rebuild is allowed. + Gc gc2(store, hexToU128("00000000000000000000000000000005")); + const RebuildReport rep = gc2.rebuildBaseline(/*force*/ false); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.live_precommits, 1u); + EXPECT_EQ(rep.clamped_shards, 0u); + + /// The precommit's blob is edge-protected: rounds never reclaim it while the precommit lives. + for (int i = 0; i < 4; ++i) + { + gc2.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).exists); +} + +/// O(budget) attempt iteration: a tiny edge budget forces multi-batch folding; the rebuilt +/// baseline still protects every committed blob (same convergence as the single-batch path). +TEST(CASGCRebuild, BatchedRebuildProtectsAllRefs) +{ + auto backend = std::make_shared(); + constexpr uint64_t gc_shards = 2; + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = gc_shards}); + const RootNamespace ns{"00/aa@cas@"}; + constexpr uint64_t edge_budget = 2; + constexpr uint64_t refs_per_shard = edge_budget + 1; + std::vector blobs; + for (uint64_t shard = 0; shard < gc_shards; ++shard) + { + for (uint64_t i = 0; i < refs_per_shard; ++i) + { + const uint64_t sequence = blobs.size() + 1; + const UInt128 blob = (UInt128{shard + gc_shards * i} << 64) | UInt128{sequence}; + ASSERT_EQ(blobShard(legacyMetaTestRef(blob), gc_shards), shard); + blobs.push_back(blob); + writeBlobBody(*backend, store->layout(), blob); + const ManifestRef r = ref(sequence, 0xA0 + sequence); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("f", blob)}); + publishCommittedTransition( + *backend, store->layout(), ns, "tbl_" + std::to_string(sequence), std::nullopt, r); + } + } + Gc gc(store, kGc); + gc.runRegularRound(); + gc.runRegularRound(); + const HeadResult st = backend->head(store->layout().gcStateKey()); + backend->deleteExact(store->layout().gcStateKey(), st.token); + + Gc gc2(store, hexToU128("00000000000000000000000000000006")); + /// Every shard has `edge_budget + 1` live edges, so each independently crosses the flush budget; + /// a test with only a pool-wide excess would not prove multiple batches for every non-empty shard. + gc2.setRebuildEdgeBudgetForTest(edge_budget); + const RebuildReport rep = gc2.rebuildBaseline(/*force*/ false); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.committed_refs, blobs.size()); + + /// Multiple rebuild flushes still converge to one authoritative row domain: no more than one + /// canonical seq-0 `btr` per shard and exactly one `cnd` per shard. These are the cardinalities the + /// catalog admission reservation over-covers independently of catalog-entry count. + const GcState rebuilt_state = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const CasFoldSeal rebuilt_seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey( + rebuilt_state.snap_generation, rebuilt_state.snap_attempt))->bytes, + store->layout(), gc_shards); + ASSERT_EQ(rebuilt_seal.condemned_summary.size(), gc_shards); + bool run_seen[gc_shards] = {false, false}; + ASSERT_EQ(rebuilt_seal.blob_target_runs.size(), gc_shards); + for (const RunRef & run : rebuilt_seal.blob_target_runs) + { + ASSERT_LT(run.shard, gc_shards); + EXPECT_FALSE(run_seen[run.shard]); + run_seen[run.shard] = true; + const auto parsed = store->layout().parseBlobTargetRunKey(run.key); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->shard, run.shard); + EXPECT_EQ(parsed->generation, run.generation); + EXPECT_EQ(parsed->seq, 0u); + } + EXPECT_TRUE(run_seen[0]); + EXPECT_TRUE(run_seen[1]); + + for (int i = 0; i < 5; ++i) + { + gc2.runRegularRound(); + store->renewWatermarkOnce(); + } + for (const UInt128 blob : blobs) + EXPECT_TRUE(backend->head(store->layout().blobKey(legacyMetaTestRef(blob))).exists) + << "blob " << u128ToHex(blob); +} + +/// Trimmed-but-live (design delta 2): the precommit's journal evidence is gone (trim), the build +/// is NOT provably dead (a live build holds min_active down) — the unowned-alive sweep must +/// over-protect the manifest's edges. +TEST(CASGCRebuild, UnownedAliveManifestOverProtected) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + /// A LIVE build pins min_active at its build_seq, so higher build sequences are not provably dead. + auto live_build = store->beginPartWrite({}); + store->renewWatermarkOnce(); + + /// An unowned manifest from build_seq 7 (no journal events at all — the trimmed shape). + const ManifestRef pre = ref(7, 0xC1); + writeBlobBody(*backend, store->layout(), DB::UInt128(9)); + writeManifestRaw(*backend, store->layout(), ns, pre, {blobEntryFor("p", DB::UInt128(9))}); + /// The namespace must be discoverable: give it one committed ref on another manifest. + const ManifestRef anchor = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, anchor, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, anchor); + + Gc gc(store, hexToU128("00000000000000000000000000000007")); + const RebuildReport rep = gc.rebuildBaseline(/*force*/ false); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.unowned_alive_manifests, 1u); + + /// Over-protected: rounds never reclaim the unowned-alive manifest's blob. + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).exists); +} + +/// Task 4 (SYSTEM CAS GC REBUILD): a rebuild refuses when ANOTHER Gc instance holds +/// the lease, even under FORCE (FORCE bypasses the "healthy state" refusal, not the lease). Gc A's +/// runRegularRound freshly acquires/renews the lease; Gc B (a different gc_id) must see it as live. +TEST(CASGCRebuild, LeaseConflictRefuses) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc_a(store, kGc); + gc_a.runRegularRound(); /// Gc A acquires/renews the lease. + + Gc gc_b(store, hexToU128("00000000000000000000000000000002")); + const RebuildReport rep = gc_b.rebuildBaseline(/*force*/ true); + EXPECT_FALSE(rep.performed); + EXPECT_NE(rep.refusal.find("lease"), String::npos) << rep.refusal; + EXPECT_NE(rep.refusal.find("leader"), String::npos) << rep.refusal; +} + +/// A rebuild CONDEMNS NOTHING (spec §7). The zero-edge condemnation that used to live here — a +/// `blobs/` LIST whose every unreached body was condemned into the rebuilt run — was the +/// r5-finding-4 data-loss vector: the rebuild's own traversal is listing-driven, so a hidden +/// durable owner made this pass condemn acked data. Its removal, the NAMED residual it leaves +/// (manifest-less orphans are retained until register R4's build/upload registry), and the +/// hold-carry that had to survive the removal are all covered by +/// `gtest_cas_rebuild_condemn_nothing.cpp`. + +/// CLAMP SUPPRESSION regression (2026-07-03 night soak: 31 dangling blobs). A committed +1 for +/// blob X lands on a shard whose fold cursor is CLAMPED (behind a bodiless precommit — the fold +/// barrier), while X's only FOLDED edge (a committed ref on ANOTHER shard) drops. Without +/// suppression the pipeline condemns, graduates and DELETES X while its landed +1 sits unfolded +/// behind the clamp; the clamp release then folds the +1 into a DANGLING reference (the model's +/// SabotageSkipChangedShard, realized). With suppression a clamped pass neither graduates nor +/// redeletes; X survives until the clamp clears, after which the +1 folds and X is SPARED. +TEST(CASGCClampSuppression, LandedEdgeBehindClampNeverDeleted) +{ + auto backend = std::make_shared(); + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + /// Folded baseline: blob X referenced by committed tbl_a (manifest m1) on shard 1. + const ManifestRef m1 = ref(1, 0xA1); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, m1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_a", std::nullopt, m1, /*shard*/1); + Gc gc(store, kGc); + gc.runRegularRound(); + store->renewWatermarkOnce(); + + /// The CLAMP on shard 0: a bodiless precommit (fold barrier — its manifest body never written). + const ManifestRef pre = ref(9, 0xEE); + addPrecommitTransition(*backend, store->layout(), ns, /*build_id*/ DB::UInt128(0x99), "part_pre", + std::nullopt, pre, /*shard*/0); + + /// BEHIND the clamp: a committed +1 for X (manifest m2, tbl_b) on shard 0 — landed, unfoldable + /// until the barrier clears. Then tbl_a drops on shard 1 — X's only FOLDED edge disappears. + const ManifestRef m2 = ref(2, 0xB2); + writeManifestRaw(*backend, store->layout(), ns, m2, {blobEntryFor("b", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_b", std::nullopt, m2, /*shard*/0); + dropRefTransition(*backend, store->layout(), ns, "tbl_a", m1, /*shard*/1); + + /// Rounds with acks current: X reaches folded in-degree 0 and is condemned, but every pass is + /// CLAMPED (the bodiless precommit persists), so nothing may graduate or delete. + /// Observability (2026-07-03): every clamp emits a gc_fold_clamp event with the reason. + store->setEventSink([&](const CasEvent & e){ if (e.type == CasEventType::GcFoldClamp) seen.push_back(e); }); + const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); + for (int i = 0; i < 6; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + ASSERT_TRUE(backend->head(blob_key).exists) + << "round " << i << ": X was deleted while its landed +1 sat unfolded behind the clamp"; + } + + ASSERT_FALSE(seen.empty()) << "each clamped pass must emit a gc_fold_clamp event"; + EXPECT_NE(seen.front().reason.find("fold barrier"), String::npos); + /// Snapshot+log ref model: the clamp is per-table (one ref-log stream per namespace, no ref shards), + /// so the event names the clamped `log` and the `resolved_through` cursor rather than a shard number. + EXPECT_TRUE(seen.front().detail.contains("log")) + << "clamp event must name the clamped log id"; + EXPECT_TRUE(seen.front().detail.contains("resolved_through")) + << "clamp event must name the cursor it resolved through"; + store->setEventSink(nullptr); + + /// Release the clamp: the precommit's body lands (the build finished staging). The next rounds + /// fold through the barrier, m2's +1 lands, and X is SPARED (entry dropped, blob intact). + writeManifestRaw(*backend, store->layout(), ns, pre, {blobEntryFor("p", DB::UInt128(1))}); + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(backend->head(blob_key).exists); + /// And the pipeline is unwedged: a genuinely-unreferenced blob still gets reclaimed. + const ManifestRef m3 = ref(3, 0xC3); + writeBlobBody(*backend, store->layout(), DB::UInt128(5)); + writeManifestRaw(*backend, store->layout(), ns, m3, {blobEntryFor("c", DB::UInt128(5))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl_c", std::nullopt, m3, /*shard*/1); + gc.runRegularRound(); + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl_c", m3, /*shard*/1); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), DB::UInt128(5))); +} diff --git a/src/Disks/tests/gtest_cas_gc_resume.cpp b/src/Disks/tests/gtest_cas_gc_resume.cpp new file mode 100644 index 000000000000..6c707bdd60cd --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_resume.cpp @@ -0,0 +1,184 @@ +#include + +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +} + +namespace +{ +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} +bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +/// Whether the CURRENT retired list (any gc-shard) still holds an entry. +bool anyRetiredPending(const PoolPtr & s) +{ + /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// separate retired list — reconstruct the in-flight set from the seal. + return anyCondemnedInSeal(s->backend(), s->layout()); +} + +/// Drive regular GC to a fixpoint over the ACK-FLOOR round (renew the store's mount ack after each round; +/// stay alive while any work counter is nonzero OR an in-flight retired entry remains). +size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(s)) + break; + } + return rounds; +} + +/// A backend that denies ONCE the SINGLE round-commit `gc/state` CAS — the casPut that advances +/// snap_generation (the one-pass round has exactly one such CAS; the lease-acquire CAS does not advance +/// snap_generation). A denied round leaves only never-adopted attempt-scoped debris (fold seal / retired +/// list under an attempt gc/state never adopted); a fresh-attempt rerun is idempotent. +class InterruptRoundCasBackend : public InMemoryBackend +{ +public: + explicit InterruptRoundCasBackend(String gc_state_key_) : gc_state_key(std::move(gc_state_key_)) {} + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (arm_interrupt && key == gc_state_key) + { + const auto stored = get(key); + const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; + const uint64_t next_gen = decodeGcState(bytes).snap_generation; + if (next_gen > stored_gen) + { + arm_interrupt = false; /// one-shot: only depose the first round-commit CAS + throw DB::Exception(DB::ErrorCodes::ABORTED, + "test-injected: round-commit gc/state CAS denied (leader deposed mid-round)"); + } + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + bool arm_interrupt = false; + +private: + String gc_state_key; +}; +} + +/// (`CASGCRound.TrimDropsFoldedOwnerEvents` was removed with the snapshot+log ref model: it asserted GC +/// trims folded owner events out of a MUTABLE shard journal in place. Immutable `_log` objects are never +/// trimmed in place; the new-model equivalent -- ref-object cleanup deletes a covered `_log`/`_snap` key +/// once BOTH the durable cursor AND a checkpoint-named validated recovery triple cover it -- is exercised in +/// `gtest_cas_ref_gc.cpp` (`RefObjectCleanupRetainsCheckpointNamedTriple`).) + +/// A crashed round leaves only never-adopted attempt-scoped debris (there is no resume machinery in the +/// one-pass round). A fresh Gc simply re-runs the round under a fresh attempt and the deletion pipeline +/// converges idempotently — a delete that already landed replays onto NotFound. +TEST(CASGCReplay, FreshAttemptRerunCompletes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + // Drive the ack-floor pipeline to a fixpoint: the blob condemns, graduates, then is deleted. Every + // step is exact-token / write-once, so a replay is idempotent (the unit oracle for the crash-replay rule). + runGcToFixpoint(store, gc); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); + + // Re-running again is a clean no-op (idempotent): the blob stays gone, no throw. + EXPECT_NO_THROW(runRegularRoundReclaiming(gc)); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} + +/// Crash-replay idempotence: a round is deposed at its SINGLE round-commit CAS (lease lost mid-round), +/// leaving only never-adopted attempt-scoped debris (a fold seal + retired list under an attempt gc/state +/// never adopted). A SECOND leader (different id => a fresh lease.seq, hence a fresh attempt) re-runs the +/// round from scratch and completes: no wedge, no CORRUPTED_DATA, and the prior-round artifacts under the +/// old (unadopted) attempt are simply unreferenced. The pool drains to a fixpoint. +TEST(CASGCReplay, DeposedRoundRerunsUnderFreshAttempt) +{ + auto backend = std::make_shared(/*gc_state_key*/ "p/gc/state"); + auto store = openPoolForTest(backend); + ASSERT_EQ(store->layout().gcStateKey(), "p/gc/state"); // guard the injected key against layout drift + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + // First leader folds + adopts the first (snap_generation, snap_attempt). + Gc gc1(store, hexToU128("00000000000000000000000000000001")); + runRegularRoundReclaiming(gc1); + store->renewWatermarkOnce(); + const auto after_fold = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + ASSERT_EQ(after_fold.snap_attempt, after_fold.lease.seq); + ASSERT_GT(after_fold.snap_generation, 0u); + + // Drop the only ref, then drive the round whose single commit CAS is DENIED (leader deposed mid-round). + // The round folded under a FRESH attempt and published its fold seal + retired list under that attempt, + // but the commit never adopted them — pure unadopted debris. gc/state is unchanged. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + backend->arm_interrupt = true; + EXPECT_THROW(runRegularRoundReclaiming(gc1), DB::Exception); + backend->arm_interrupt = false; + + const auto after_interrupt = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_EQ(after_interrupt.snap_generation, after_fold.snap_generation) + << "the denied round-commit CAS must NOT advance the adopted generation"; + EXPECT_EQ(after_interrupt.snap_attempt, after_fold.snap_attempt) + << "the denied round-commit CAS must NOT advance the adopted attempt"; + // The deposed round's fold seal is durable under its OWN (unadopted) attempt — unreferenced by gc/state. + const uint64_t deposed_attempt = after_fold.lease.seq + 1; // round 2 renewed the lease once + const uint64_t deposed_gen = after_fold.snap_generation + 1; + EXPECT_TRUE(backend->head(store->layout().foldSealKey(deposed_gen, deposed_attempt)).exists) + << "the deposed round's fold seal is durable under its own unadopted attempt (harmless debris)"; + + // A DIFFERENT leader takes over. The lease steal protocol observes the stalled lease twice before + // stealing: the first round only observes and defers; the second steals and re-runs the round from + // scratch under its own fresh attempt. + Gc gc2(store, hexToU128("00000000000000000000000000000002")); + EXPECT_NO_THROW(runRegularRoundReclaiming(gc2)); // observe-and-defer (lease not yet provably stalled) + store->renewWatermarkOnce(); + + // From here gc2 owns the lease; drive it to a fixpoint. It must drain the unreachable blob WITHOUT + // wedging on the deposed attempt's debris (attempt-scoping keeps that debris invisible). + EXPECT_NO_THROW(runGcToFixpoint(store, gc2)); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); + + const auto after_drain = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_GT(after_drain.snap_generation, after_fold.snap_generation) << "the round completed under gc2"; + EXPECT_NE(after_drain.snap_attempt, deposed_attempt) << "the drained round never adopted the deposed attempt"; + + // A further round is a clean no-op. + EXPECT_NO_THROW(runRegularRoundReclaiming(gc2)); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} diff --git a/src/Disks/tests/gtest_cas_gc_round.cpp b/src/Disks/tests/gtest_cas_gc_round.cpp new file mode 100644 index 000000000000..b987cc39d8fb --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_round.cpp @@ -0,0 +1,1999 @@ +#include + +#include + +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int CORRUPTED_DATA; +extern const int ABORTED; +} + +namespace ProfileEvents +{ +extern const Event CASGCMetaOps; +extern const Event CASGCEnumerationPages; +extern const Event CASMountExclusivityViolation; +extern const Event CASGCRetiredSpared; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +/// ROUND-LEVEL end-to-end GC tests over the root-local part-manifest model (one-pass ack-floor round: +/// heartbeat floor -> fold with the three-cursor merge -> pre-CAS exact-token deletes -> single CAS -> trim). +/// +/// This file is the survivor of the old snap/cascade-based `gtest_cas_gc_round.cpp`. The per-STEP +/// behaviours it used to cover have moved to the dedicated GC-core suites and are intentionally NOT +/// re-tested here: +/// - fold edge dispatch (committed/precommit/promote/removal +/-1, 404 clamp/anomaly, fold barrier, +/// ref-mismatch fail-closed) -> gtest_cas_gc_fold.cpp +/// - condemn/graduate/delete + spare (manifest body deferred delete, publish racing the pass is spared, +/// unreferenced blob exact-token delete) -> gtest_cas_gc_ack_floor.cpp +/// - trim of folded owner events + idempotent crash replay -> gtest_cas_gc_resume.cpp +/// What remains here is what those step suites do NOT cover: the LEASE/leadership protocol (the round's +/// only stateful concurrency), the cursor-key codec, and the multi-round END-TO-END reclaim scenarios +/// driven to fixpoint (publish->drop->reclaim, multi-ref sharing, spare-on-recheck race, idempotent +/// fixpoint, split-brain duplicate-work-only). Every kept test keeps STRONG no-loss / no-dangle / no-leak +/// assertions. No test sleeps or reads a clock — "time" is the order of `runRegularRound` calls. + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +const UInt128 kGcA = hexToU128("0000000000000000000000000000000a"); +const UInt128 kGcB = hexToU128("0000000000000000000000000000000b"); +const UInt128 kGcC = hexToU128("0000000000000000000000000000000c"); + +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} + +bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +bool manifestExists(InMemoryBackend & b, const Layout & layout, const ManifestId & id) +{ + return b.head(layout.manifestKey(id)).exists; +} + +PoolPtr openTestPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +PoolPtr openTestPoolWithConfig(std::shared_ptr & out_backend, PoolConfig config) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, std::move(config)); +} + +/// Fault decorator for triage #5's regression test (`CASGCRetention.LosingRoundNeverDestroysParentSealGeneration` +/// below): the `fail_at_call`-th `casPut` against `faulted_key` returns `Conflict` instead of committing — +/// deterministically and single-threaded reproducing "this round's own gc/state CAS lost the race to a +/// concurrent leader," which is the only condition under which the pre-CAS wholesale prune's choice of +/// `referenced_generations` is externally observable (a round whose own CAS SUCCEEDS reclaims the same +/// generation moments later via the existing, unrelated post-CAS hand-off delete regardless of this fix, +/// so faulting the CAS is required, not optional, to pin the production call site). A call count, not a +/// one-shot arm flag: `Gc::acquireOrRenewLease` issues its OWN earlier `casPut` on the very same gc/state +/// key to renew the lease BEFORE a round folds — that renewal must SUCCEED (so the round actually reaches +/// the fold/prune it's meant to exercise), and only the round's LATER, final round-commit `casPut` must +/// be the one that loses. `fail_at_call` is 1-indexed and lets the test target that specific call exactly, +/// computed from `calls_to_faulted_key` observed so far rather than hardcoded. +class GcStateCasFaultBackend : public InMemoryBackend +{ +public: + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + CasResult casPut(const String & key, const String & bytes, + const std::optional & expected, const ObjectMeta & meta) override + { + if (key == faulted_key) + { + ++calls_to_faulted_key; + if (fail_at_call != 0 && calls_to_faulted_key == fail_at_call) + return CasResult{CasOutcome::Conflict, {}}; + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + String faulted_key; + size_t calls_to_faulted_key = 0; + size_t fail_at_call = 0; /// 0 = never fault; else fault exactly the Nth casPut to `faulted_key` +}; + +GcState readState(InMemoryBackend & b, const Pool & s) +{ + const auto got = b.get(s.layout().gcStateKey()); + if (!got) + { + ADD_FAILURE() << "gc/state absent"; + return {}; + } + return decodeGcState(got->bytes); +} + +/// Whether ANY gc-shard's adopted-seal run still holds a `kCondemned` row (retired-in-snapshot T4: the +/// retired state rides the snapshot run, not a separate retired-list object) — the ack-floor deletion +/// pipeline is still in flight while this is true. +bool anyRetiredPending(InMemoryBackend & b, const Pool & s) +{ + return anyCondemnedInSeal(b, s.layout()); +} + +/// Drive a Gc to fixpoint over the round-paced retired-cursor pipeline: run rounds, renewing the store's +/// own heartbeat after each (`renewWatermarkOnce` — keeps the lease + build-watermark floor current; +/// graduation itself paces on rounds alone). A condemned blob traverses the multi-round condemn -> +/// graduate -> delete pipeline, so "fixpoint" is reached only when a round did NO work AND the current +/// retired list is empty (nothing still in flight). Returns the number of rounds that held the lease and +/// did work. Bounded so a non-converging core fails downstream assertions rather than hanging. +size_t driveToFixpoint(InMemoryBackend & backend, const PoolPtr & store, Gc & gc) +{ + size_t working_rounds = 0; + for (size_t r = 0; r < 64; ++r) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + store->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(backend, *store)) + break; + if (!no_work) + ++working_rounds; + } + return working_rounds; +} + +/// A full key -> token snapshot of the backend, for the previewDeletes write-free invariant: any +/// put/casPut/overwrite mints a fresh token (or adds a key) and any delete removes one, so an unchanged +/// map across a call proves it performed NO writes. +std::map snapshotKeyTokens(InMemoryBackend & b) +{ + std::map out; + String cursor; + while (true) + { + const ListPage page = b.list("", cursor, 100000); + for (const ListedKey & k : page.keys) + out[k.key] = k.token ? k.token->value : String{}; + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return out; +} + +} + +/// ---- LEASE / leadership protocol (the round's only stateful concurrency) ---- +/// +/// The lease steal window is observation-based and deterministic (see CasGc.h): a contender becomes +/// steal-eligible when it observes the SAME (owner, seq) across two of its own consecutive round +/// attempts. The new model keeps gc/state {round, snap_generation, lease}, so these tests +/// are model-agnostic and were ported verbatim from the pre-redesign suite. + +TEST(CASGCLease, FreshPoolAcquiresAndRenews) +{ + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc(s, kGc); + + EXPECT_TRUE(gc.runRegularRound().acquired_lease); + const GcState st1 = readState(*b, *s); + EXPECT_EQ(st1.lease.owner, kGc); + const uint64_t seq1 = st1.lease.seq; + EXPECT_GE(seq1, 1u); + + EXPECT_TRUE(gc.runRegularRound().acquired_lease); /// renew + const GcState st2 = readState(*b, *s); + EXPECT_EQ(st2.lease.owner, kGc); + EXPECT_GT(st2.lease.seq, seq1); /// seq strictly advanced +} + +TEST(CASGCLease, ContenderBacksOffWhileIncumbentRenews) +{ + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// first sight: record observation + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// incumbent renews (seq advances) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// gc2 sees a NEW seq => incumbent alive + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// alive again - never steals while renewing + EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); +} + +TEST(CASGCLease, StealAfterObservedNonRenewalAdvancesLease) +{ + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + const GcState st0 = readState(*b, *s); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// observation recorded; gc1 then DIES + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// same (owner, seq) observed twice => steal + const GcState st = readState(*b, *s); + EXPECT_EQ(st.lease.owner, kGcB); + EXPECT_GT(st.lease.seq, st0.lease.seq); +} + +TEST(CASGCLease, HeartbeatBlocksFalseStealOfAliveLeader) +{ + /// B160: a slow-but-alive incumbent whose lease.seq is frozen for its (long) round must NOT be + /// stolen from, because its advisory heartbeat keeps advancing. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// gc1 leads (seq frozen for its round) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// gc2 observes (gc/hb absent yet) + + Gc::pulseHeartbeat(*s, kGcA); /// gc1 mid-round but heartbeating (hb 0->1) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// hb advanced => alive => NO steal + Gc::pulseHeartbeat(*s, kGcA); /// hb 1->2 + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// still no steal while heartbeating + EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); /// gc1 still owns the lease +} + +/// A7-HIGH-fix follow-up (residual timing window): an allow_steal=false observation of a foreign +/// incumbent (the manual `SYSTEM ... GC` path) must NOT arm the frozen-tuple comparison that the loop's +/// very next (allow_steal=true) call uses to decide whether to steal. Without this, a manual round's +/// observation at time t, immediately followed by an unluckily-timed scheduled tick at t+epsilon (no +/// real chance for a live incumbent to heartbeat in between), would see the SAME (owner, seq, hb) twice +/// and steal a LIVE leader — exactly the hazard the allow_steal gate alone does not close, since it only +/// stops the MANUAL call itself from executing the steal CAS, not from contaminating the shared `Gc` +/// instance's observation state that the next allow_steal=true call reads. +TEST(CASGCLease, ManualObservationNeverArmsTheLoopsStealDecision) +{ + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); /// plays the scheduler's ONE shared Gc, observed by both manual and loop calls + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// gc1 leads; never renews, never heartbeats + + /// Manual observation "at t": allow_steal=false. Would normally be obs #1, but must NOT record it. + EXPECT_FALSE(gc2.runRegularRound({}, /*allow_steal=*/false).acquired_lease); + + /// Loop-path call "immediately after" (allow_steal=true, the default): with the fix, gc2's + /// last_seen_* is UNTOUCHED by the manual call above, so this is still effectively obs #1 (first + /// sight) => must NOT steal. Pre-fix (manual observations armed the state), this would see the same + /// frozen tuple as "twice observed" and steal gc1's still-live lease. + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); + EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); /// gc1 keeps the lease + + /// The loop still recovers a genuinely dead incumbent across its OWN two spaced observations: the + /// call above was the loop's real obs #1 (now armed, since allow_steal=true); this one is obs #2 of + /// the same still-frozen tuple => steal-eligible => steals. Recovery is delayed, not disabled. + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); + EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); +} + +/// P3-B1 (2026-07-11 mid-switch soak wedge): CasGcScheduler used to flip its `i_am_leader` flag (which +/// gates the heartbeat thread's pulses) only AFTER `runRegularRound` RETURNS, while the lease is +/// acquired INSIDE the round, before the (potentially long) fold. A brand-new leader's FIRST round +/// therefore ran the whole fold with no heartbeat cover: a follower observing the frozen (owner, seq) +/// across two of its own ticks steals deterministically once that first round outlasts ~2 ticks - +/// mutual-steal livelock under a slow fold. The fix moves the "start heartbeating" action to the +/// INSTANT the lease is acquired (`Gc::runRegularRound`'s new `on_lease_acquired` hook), fired before +/// the fold begins. These two tests pin the protocol both ways at the `Gc` level (the scheduler itself +/// only wires `i_am_leader.store(true, ...)` + one `pulseHeartbeat` call into that hook - a thread-pacing +/// wire-up not practically unit-testable without sleeps; verified by code review + the full gtest run). + +TEST(CASGCLease, WithoutAcquireTimePulseFirstRoundStealsDeterministically) +{ + /// RED-before-the-fix scenario: gc1 acquires the lease and (simulating a long first round) never + /// pulses `gc/hb` and never renews - exactly what happened before `on_lease_acquired` existed. + /// gc2's SECOND observation of the same frozen (owner, seq, hb) steals, per the documented protocol. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// gc1 becomes leader; NO pulse follows (the bug) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1: records (owner=A, seq, hb=absent) + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// obs #2: unchanged => steal-eligible => STEALS + EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); +} + +TEST(CASGCLease, AcquireTimePulseProtectsNewLeadersFirstRound) +{ + /// GREEN-after-the-fix scenario: with the fix, `i_am_leader` flips true and the FIRST pulse fires + /// the instant gc1 acquires the lease - before B's first observation even happens - and the + /// (separately-threaded, out of scope here) `heartbeatLoop` keeps landing further pulses on its own + /// cadence for as long as `i_am_leader` stays true, i.e. for the whole duration of gc1's first round. + /// The net effect proven here is the one that matters: SOME pulse lands between B's two + /// observations (not just before both, and not only after both), so B's second observation sees hb + /// advanced relative to its first and backs off instead of stealing - exactly what never happened + /// pre-fix, when `i_am_leader` (and hence every pulse) was gated on the round having already returned. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// gc1 becomes leader (still mid-fold, seq frozen) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1: records (owner=A, seq, hb=absent) + Gc::pulseHeartbeat(*s, kGcA); /// a heartbeatLoop tick lands mid-round (hb 0->1) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #2: hb advanced since obs #1 => alive => NO steal + + EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); /// gc1 keeps the lease through its whole first round +} + +TEST(CASGCLease, StaleOwnerHeartbeatDoesNotEnableFalseSteal) +{ + /// A deposed leader's heartbeat thread keeps pulsing until its next round notices the lost lease + /// (`i_am_leader` is only reset there), and `pulseHeartbeat` stamps `owner = self` while a losing + /// CAS write silently vanishes — so a zombie old leader can keep `gc/hb.owner` pointing at ITSELF + /// even while the live new leader is pulsing too. The liveness gate must therefore treat ANY + /// movement of the observed (owner, hb_seq) pair between a follower's two ticks as "someone is + /// alive": comparing hb_seq is only meaningful against the SAME remembered hb owner. The old + /// predicate compared `hb.owner` with the LEASE owner instead, so a zombie-owned hb read as + /// "not the leader's heartbeat" on both ticks and a live, pulsing new leader got its lease stolen. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + Gc gc3(s, kGcC); + + /// gc1 leads, beats, then dies mid-round; gc2 legitimately steals the lease. + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + Gc::pulseHeartbeat(*s, kGcA); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1 of gc1's frozen tuple + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// obs #2: frozen lease + frozen hb => steal + ASSERT_EQ(readState(*b, *s).lease.owner, kGcB); + + /// gc2 is now mid-long-round (lease tuple frozen) and PULSING — but gc1's zombie heartbeat + /// thread interleaves after every gc2 pulse, so the follower gc3 only ever OBSERVES gc1-owned + /// heartbeats. The pair keeps moving, which is proof of life. + Gc::pulseHeartbeat(*s, kGcB); + Gc::pulseHeartbeat(*s, kGcA); /// zombie masks gc2's pulse + EXPECT_FALSE(gc3.runRegularRound().acquired_lease); /// obs #1: records (hb owner=A, seq) + Gc::pulseHeartbeat(*s, kGcB); + Gc::pulseHeartbeat(*s, kGcA); /// zombie masks again + EXPECT_FALSE(gc3.runRegularRound().acquired_lease); /// obs #2: hb pair MOVED => alive => NO steal + EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); /// the live leader keeps its lease + + /// Liveness is preserved: once everything genuinely freezes (gc2 dead, zombie gone), the next + /// tick completes the window — obs #2 above already re-armed on the now-frozen (lease, hb) pair. + EXPECT_TRUE(gc3.runRegularRound().acquired_lease); /// still frozen a full tick later => steal + EXPECT_EQ(readState(*b, *s).lease.owner, kGcC); +} + +TEST(CASGCLease, FailoverStealOnceHeartbeatStops) +{ + /// B160: once the incumbent stops heartbeating (it died), a follower observing the now-frozen + /// heartbeat steals — automatic failover is preserved. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1 + Gc::pulseHeartbeat(*s, kGcA); /// one last pulse (hb 0->1) + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// hb advanced => no steal; records hb=1 + /// gc1 now DEAD: no renew, no further pulse. hb stays at 1 == gc2's last observation. + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// hb frozen + seq frozen => STEAL + EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); +} + +TEST(CASGCLease, DeadIncumbentThenRevivedIncumbentWinsRace) +{ + /// A stalled incumbent that revives and renews BEFORE the contender's second look resets the + /// contender's window: gc2's second observation sees a NEW seq => NOT steal-eligible => backs off. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1 + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); /// gc1 revives and renews + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// new seq seen => window resets + EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); +} + +TEST(CASGCLease, ConcurrentStealLosesCas) +{ + /// The CAS-race horn: gc2 is steal-eligible and goes for the CAS, but gc/state moved under it + /// (injected one-shot conflict). It must back off (never acquired=true off a lost CAS) and the + /// owner on storage must be unperturbed. The injected conflict left the object unchanged, so gc2's + /// NEXT round is steal-eligible again and succeeds. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + const GcState st0 = readState(*b, *s); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1; gc1 stalls now + b->failNextCasPut(s->layout().gcStateKey()); /// inject: gc2's steal CAS conflicts + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// steal attempt loses the CAS => back off + const GcState st1 = readState(*b, *s); + EXPECT_EQ(st1.lease.owner, kGcA); /// unchanged + EXPECT_EQ(st1.lease.seq, st0.lease.seq); /// nothing clobbered + EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// still steal-eligible => succeeds now + EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); +} + +TEST(CASGCLease, CreateConflictReReadsWithinTheBound) +{ + /// The create-Conflict branch: a fresh pool where the create-if-absent CAS conflicts (one-shot). + /// The contender re-reads and falls through within its bounded (2) CAS attempts — the re-read still + /// finds the key absent, so the second attempt creates and acquires. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc(s, hexToU128("0000000000000000000000000000000c")); + + b->failNextCasPut(s->layout().gcStateKey()); + EXPECT_TRUE(gc.runRegularRound().acquired_lease); + const GcState st = readState(*b, *s); + EXPECT_EQ(st.lease.owner, hexToU128("0000000000000000000000000000000c")); + EXPECT_EQ(st.lease.seq, 1u); +} + +TEST(CASGCLease, CtorFailsClosedOnBadArguments) +{ + /// Guards: a null store and gc_id == 0 (reserved for "lease never held") are caller bugs. + std::shared_ptr b; + auto s = openTestPool(b); + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { Gc(nullptr, kGc); }); + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { Gc(s, DB::UInt128(0)); }); +} + +TEST(CASGCLease, IncumbentRenewConflictRetriesOnceAndAcquires) +{ + /// The incumbent's own renew CAS conflicts (one-shot). Re-read sees our own ownership => the renew + /// is retried ONCE within the bounded (2) CAS attempts => acquired. Never acquired=true without a + /// Committed CAS — storage must carry the seq the SECOND (committed) attempt wrote. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc(s, hexToU128("0000000000000000000000000000000d")); + + ASSERT_TRUE(gc.runRegularRound().acquired_lease); /// create: seq 1 + b->failNextCasPut(s->layout().gcStateKey()); /// inject: the renew CAS conflicts + EXPECT_TRUE(gc.runRegularRound().acquired_lease); /// re-read (still us) => retried once + const GcState st = readState(*b, *s); + EXPECT_EQ(st.lease.owner, hexToU128("0000000000000000000000000000000d")); + EXPECT_EQ(st.lease.seq, 2u); /// the committed retry's seq +} + +TEST(CASGCLease, VanishedStateAfterObservationFailsClosed) +{ + /// gc/state is never legally deleted - absent AFTER a recorded observation proves an out-of-model + /// deletion. Recreating a default state would reset round/cursors; the lease protocol + /// must fail closed (CORRUPTED_DATA) instead. + std::shared_ptr b; + auto s = openTestPool(b); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// gc2 records an observation + + const auto head = b->head(s->layout().gcStateKey()); /// out-of-model wipe (raw delete) + ASSERT_TRUE(head.exists); + ASSERT_EQ(b->deleteExact(s->layout().gcStateKey(), head.token).kind, DeleteOutcome::Kind::Deleted); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc2.runRegularRound(); }); +} + +/// ---- END-TO-END round scenarios driven to fixpoint (the headline value of this file) ---- + +/// publish -> drop -> GC-to-fixpoint reclaim: a committed ref names a blob; after the ref is dropped, +/// the round protocol collects the blob (exact-token delete) AND the owner-removed manifest body, and a +/// further round is a clean no-op. The strongest no-loss/no-leak oracle: while the ref is live the blob +/// is NEVER touched; once dropped, BOTH the blob and the manifest are gone and nothing dangles. +TEST(CASGCRound, PublishDropReclaimsBlobAndManifestToFixpoint) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + const ManifestId id{ns, r}; + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + /// While live: the blob's in-degree is 1 and NOTHING is collected (no-loss). + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + EXPECT_TRUE(manifestExists(*backend, store->layout(), id)); + + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + driveToFixpoint(*backend, store, gc); + /// After drop + fixpoint: the blob's only edge is gone, the blob is collected, the owner-removed + /// manifest body is collected, and the in-degree generation reflects zero (no-leak / no-dangle). + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); + EXPECT_FALSE(manifestExists(*backend, store->layout(), id)); + + /// Idempotent: re-running to fixpoint changes nothing and never throws. + EXPECT_NO_THROW(driveToFixpoint(*backend, store, gc)); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} + +/// retired-in-snapshot T4: after a round condemns one blob, the ADOPTED fold seal's per-shard +/// condemned_summary reflects it (condemned_total == 1, pending_total == 0) — distilled zero-I/O from the +/// kCondemned rows the fold sealed into the snapshot run. +TEST(CASGCRound, CondemnRoundSealSummaryCountsCondemned) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); /// folds the +1 + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); /// the -1 condemns it + + /// Drive rounds until the blob shows up condemned in the adopted-seal run; capture that seal. + bool condemned = false; + CasFoldSeal seal; + for (int i = 0; i < 6 && !condemned; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + const GcState st = readState(*backend, *store); + seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + for (const RetiredEntry & e : currentRetiredSet(*backend, store->layout(), /*shard*/0)) + if (e.ref == DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(DB::UInt128(1))}) + condemned = true; + } + ASSERT_TRUE(condemned) << "blob never condemned into the snapshot run"; + ASSERT_TRUE(seal.condemned_summary.contains(0)) << "seal summary must be total over gc_shards"; + EXPECT_EQ(seal.condemned_summary.at(0).condemned_total, 1u); + EXPECT_EQ(seal.condemned_summary.at(0).pending_total, 0u) + << "a freshly condemned entry is not yet delete_pending"; + EXPECT_LT(seal.condemned_summary.at(0).oldest_nonpending_condemn_round, + std::numeric_limits::max()) + << "a non-pending condemned entry records its condemn round"; +} + +/// retired-in-snapshot T5: `previewDeletes` streams the adopted seal's `kCondemned` rows and reports each +/// with the STORED condemn-time token — `awaiting_graduation` while newly condemned, then `delete_pending` +/// once graduated, and NOTHING once the exact-token redelete has removed the blob. The preview performs no +/// HEAD on the condemned rows (the token is durable in-run) and is WRITE-FREE throughout (spec §5 req 1). +TEST(CASGCRound, PreviewReportsCondemnedRowsAndIsWriteFree) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + const UInt128 blob = DB::UInt128(1); + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); /// round 1: folds the +1; blob referenced + EXPECT_TRUE(gc.previewDeletes().empty()) << "a live-referenced blob is never previewed for deletion"; + + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + runRegularRoundReclaiming(gc); /// condemning round: -1 => in-degree 0 => kCondemned row (not pending) + + /// Write-free contract: a full key->token snapshot must be identical across the previewDeletes call. + const auto before = snapshotKeyTokens(*backend); + const std::vector awaiting = gc.previewDeletes(); + const auto after = snapshotKeyTokens(*backend); + EXPECT_EQ(before, after) << "previewDeletes must perform NO writes (put/casPut/overwrite/delete)"; + + ASSERT_EQ(awaiting.size(), 1u) << "exactly the one condemned blob is previewed"; + EXPECT_EQ(awaiting[0].ref, (DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})); + EXPECT_EQ(awaiting[0].key, store->layout().blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})); + EXPECT_EQ(awaiting[0].reason, "awaiting_graduation"); + EXPECT_FALSE(awaiting[0].token.value.empty()) << "must carry the stored condemn-time token"; + EXPECT_GT(awaiting[0].condemn_round, 0u) << "must carry the stored condemn round"; + + runRegularRoundReclaiming(gc); /// graduation round: entry becomes delete_pending (blob still present) + const std::vector pending = gc.previewDeletes(); + ASSERT_EQ(pending.size(), 1u); + EXPECT_EQ(pending[0].ref, (DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})); + EXPECT_EQ(pending[0].reason, "delete_pending"); + EXPECT_FALSE(pending[0].token.value.empty()); + + runRegularRoundReclaiming(gc); /// redelete round: exact-token delete; entry dropped; blob gone + EXPECT_FALSE(blobExists(*backend, store->layout(), blob)); + EXPECT_TRUE(gc.previewDeletes().empty()) << "nothing to preview once the blob is redeleted"; +} + +/// A fully idle fold pure-carries every shard's authoritative rows verbatim. The parent is first made +/// non-vacuous with one live blob in each of two shards; the forced no-delta successor must preserve +/// both `btr` rows and the total `cnd` domain byte-for-byte. +TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_shards = 2, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + + Gc gc(store, kGc); + + const UInt128 shard0_blob{1}; + const UInt128 shard1_blob = (UInt128{1} << 64) | UInt128{1}; + ASSERT_EQ(blobShard(legacyMetaTestRef(shard0_blob), 2), 0u); + ASSERT_EQ(blobShard(legacyMetaTestRef(shard1_blob), 2), 1u); + + const ManifestRef r0 = ref(1, 0xAA); + const ManifestRef r1 = ref(2, 0xBB); + writeBlobBody(*backend, store->layout(), shard0_blob); + writeBlobBody(*backend, store->layout(), shard1_blob); + writeManifestRaw(*backend, store->layout(), ns, r0, {blobEntryFor("a", shard0_blob)}); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("b", shard1_blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl0", std::nullopt, r0); + publishCommittedTransition(*backend, store->layout(), ns, "tbl1", std::nullopt, r1); + gc.runRegularRound(); + const GcState st1 = readState(*backend, *store); + const CasFoldSeal seal1 = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes, + store->layout(), /*gc_shards=*/2); + + /// No state changes after the parent. The zero defer bound forces an actual fold rather than DEFER, + /// so every shard takes the production pure-carry path. + gc.runRegularRound(); + const GcState st2 = readState(*backend, *store); + const CasFoldSeal seal2 = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes, + store->layout(), /*gc_shards=*/2); + + /// TOTALITY: both seals carry a summary entry for every gc-shard. + ASSERT_EQ(seal1.condemned_summary.size(), 2u); + ASSERT_EQ(seal2.condemned_summary.size(), 2u); + EXPECT_TRUE(seal1.condemned_summary.contains(0) && seal1.condemned_summary.contains(1)); + EXPECT_TRUE(seal2.condemned_summary.contains(0) && seal2.condemned_summary.contains(1)); + + /// Capacity reserves one widest `btr` row per shard. Pin the production pure-carry seal to the + /// authoritative grammar that makes that bound sufficient: at most one in-range canonical seq-0 + /// run per shard, beside exactly one `cnd` row for every shard. + bool run_seen[2] = {false, false}; + ASSERT_EQ(seal1.blob_target_runs.size(), 2u); + ASSERT_EQ(seal2.blob_target_runs.size(), 2u); + for (const RunRef & run : seal2.blob_target_runs) + { + ASSERT_LT(run.shard, 2u); + EXPECT_FALSE(run_seen[run.shard]); + run_seen[run.shard] = true; + const auto parsed = store->layout().parseBlobTargetRunKey(run.key); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->shard, run.shard); + EXPECT_EQ(parsed->generation, run.generation); + EXPECT_EQ(parsed->seq, 0u); + } + EXPECT_TRUE(run_seen[0]); + EXPECT_TRUE(run_seen[1]); + for (uint64_t shard = 0; shard < 2; ++shard) + { + const auto parent_run = std::find_if( + seal1.blob_target_runs.begin(), seal1.blob_target_runs.end(), + [shard](const RunRef & run) { return run.shard == shard; }); + const auto carried_run = std::find_if( + seal2.blob_target_runs.begin(), seal2.blob_target_runs.end(), + [shard](const RunRef & run) { return run.shard == shard; }); + ASSERT_NE(parent_run, seal1.blob_target_runs.end()); + ASSERT_NE(carried_run, seal2.blob_target_runs.end()); + EXPECT_EQ(*carried_run, *parent_run); + } + + /// VERBATIM CARRY: nothing was ever condemned, so every shard's summary is the zero entry, carried + /// unchanged from parent to child across the fully idle fold. + for (uint64_t shard = 0; shard < 2; ++shard) + { + EXPECT_EQ(seal2.condemned_summary.at(shard), seal1.condemned_summary.at(shard)) + << "shard " << shard << " summary must be carried verbatim from the parent seal"; + EXPECT_EQ(seal2.condemned_summary.at(shard).condemned_total, 0u); + } +} + +/// Attempt-scoping (B2): a fold seal planted under a NON-adopted attempt at the adopted generation +/// must be INVISIBLE to every reader. A deposed leader writes its fold seal under its own (unadopted) +/// `lease.seq`; that artifact lives at `foldSealKey(snap_generation, snap_attempt + k)` and no decision +/// path may resolve it. `previewDeletes` reads the in-degree generation strictly at the adopted +/// `(snap_generation, snap_attempt)`, so the decoy must not change its output and must not throw. This +/// is the implementation-level complement to the TLA+ `INV_ONLY_ADOPTED_VIEWABLE` gate. +TEST(CASGCRound, NonAdoptedAttemptSealIgnored) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); + const GcState st = readState(*backend, *store); + ASSERT_GT(st.snap_generation, 0u); + + /// Control preview BEFORE the decoy (previewDeletes is write-free, so the result is deterministic). + const auto control = gc.previewDeletes(); + + /// Plant a decoy fold seal under a DIFFERENT attempt at the SAME generation (a deposed leader's + /// unadopted artifact). It must be invisible to the adopted-attempt readers. + backend->putIfAbsent(store->layout().foldSealKey(st.snap_generation, st.snap_attempt + 999), + "decoy-seal-bytes"); + + /// No reader resolves the non-adopted attempt: no throw, and the preview is unchanged by the decoy. + std::vector after; + EXPECT_NO_THROW(after = gc.previewDeletes()); + EXPECT_EQ(after.size(), control.size()) + << "a non-adopted attempt's fold seal must not influence previewDeletes"; + + /// A further full round must still proceed without throwing and without the decoy wedging it. + EXPECT_NO_THROW(gc.runRegularRound()); +} + +/// B11: the round summary must count manifest-body (tree) deletes separately from blob deletes. A drop +/// that reclaims one manifest body must report manifests_deleted >= 1 in the RoundReport of the +/// reclaiming round, while blobs and manifests remain separately countable. +TEST(CASGCRound, RoundSummaryCountsManifestBodyDeletes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xCC); + const ManifestId id{ns, r}; + + writeBlobBody(*backend, store->layout(), DB::UInt128(3)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(3))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); /// fold the publish; no delete yet + + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + /// §0 introspection: both counters are captured BEFORE the condemn+delete pipeline below, which + /// drives the round's meta pool (condemn/spare/delete) and its own orphan-sweep cursor pass. + const auto meta_ops_before = ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps].load(); + const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load(); + + /// Ack-floor drift: the owner-removed manifest body is deleted in the CONDEMNING round (post-CAS, + /// after its -1 is adopted), while the blob's exact-token delete happens a few rounds later once the + /// ack floor graduates its retired entry. So the two deletes fall in DIFFERENT reports now — accumulate + /// across the pipeline (renewing the ack each round so the floor advances) and assert both were counted. + uint64_t total_manifests_deleted = 0; + uint64_t total_blob_deleted = 0; + for (size_t i = 0; i < 64; ++i) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + store->renewWatermarkOnce(); + total_manifests_deleted += rep.manifests_deleted; + total_blob_deleted += rep.deleted; + if (total_manifests_deleted > 0 && total_blob_deleted > 0) + break; + } + + /// B11: the manifest-body delete must be counted separately from the blob delete. + EXPECT_GE(total_manifests_deleted, 1u) + << "round summary must count the owner-removed manifest body delete (B11 — manifests_deleted)"; + /// Blobs and manifests are separately countable: the blob delete (deleted >= 1) is independent. + EXPECT_GE(total_blob_deleted, 1u) + << "the blob exact-token delete must still be counted in deleted"; + /// The manifest body is gone and the blob is gone — no-leak / no-dangle. + EXPECT_FALSE(manifestExists(*backend, store->layout(), id)); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(3))); + + /// §0 introspection: the exact-token blob delete above scheduled at least one per-hash freshness-meta + /// op on the round's bounded meta pool, and every round ran its own orphan-manifest-sweep cursor pass + /// (default `manifest_sweep_list_budget_keys` is nonzero), fetching at least one LIST page directly. + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps].load() - meta_ops_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load() - pages_before, 1); +} + +/// Manifest-body cleanup (post-CAS `manifest_deletes` phase) has no cap: the ref-log intake cursor that +/// discovers each owner-removed manifest commits in the SAME round's CAS that produces `mf_cleanup`, so an +/// entry a cap declined would never be re-derived by this pipeline -- a bounded burst would become a +/// permanent leak. Five tables' manifests are all owner-removed in one fold; the round must delete all +/// five bodies in the same round, with nothing left un-deleted. +TEST(CASGCRound, ManifestCleanupDrainsEntireRoundWithNoSkips) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "gc-runner", + .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + constexpr int kManifests = 5; + + std::vector ids; + for (int i = 0; i < kManifests; ++i) + { + const ManifestRef r = ref(1, 0xD0 + i); + writeBlobBody(*backend, store->layout(), DB::UInt128(100 + i)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(100 + i))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl" + std::to_string(i), std::nullopt, r); + ids.push_back(ManifestId{ns, r}); + } + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); /// fold all five +1s; no manifest owner-removed yet + for (const ManifestId & id : ids) + ASSERT_TRUE(manifestExists(*backend, store->layout(), id)); + + /// Remove all five owners in one window; the next fold's intake sees all five `-1` edges together. + for (int i = 0; i < kManifests; ++i) + dropRefTransition(*backend, store->layout(), ns, "tbl" + std::to_string(i), ref(1, 0xD0 + i)); + + const RoundReport rep = runRegularRoundReclaiming(gc); + ASSERT_TRUE(rep.acquired_lease); + EXPECT_EQ(rep.manifests_deleted, kManifests) + << "manifest_deletes must drain the entire mf_cleanup vector in one round, not cap it"; + + for (const ManifestId & id : ids) + EXPECT_FALSE(manifestExists(*backend, store->layout(), id)) + << "an unbudgeted cleanup must leave nothing surviving the round it was discovered in"; +} + +/// §0 introspection follow-up: `CASGCEnumerationPages` must not depend on the orphan-manifest sweep alone +/// (`manifest_sweep_list_budget_keys` zeroed below disables that pass entirely). The mandatory per-round +/// `cas/ns/stream/` scan -- `listRefPrefix`'s pre-fold DEFER signal and the fold share its result -- +/// must still land at least one page each round. +TEST(CASGCRound, EnumerationPagesCountedEvenWithSweepBudgetZeroed) +{ + std::shared_ptr backend; + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.manifest_sweep_list_budget_keys = 0; /// disables the orphan sweep's own LIST entirely + config.gc_fold_max_defer_rounds = 0; /// force fold-every-round (Phase-4 Lever A would defer) + auto store = openTestPoolWithConfig(backend, config); + + Gc gc(store, kGc); + const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load(); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load() - pages_before, 1) + << "the round's own cas/ns/stream/ enumeration must count pages independent of " + "the orphan sweep"; +} + +/// M1 REGRESSION (cross-round fold cursor must survive independent of trim): a folded-but-untrimmed owner +/// event must NOT be re-folded by the next round. With eager trim the folded event is removed so the bug +/// (sealedCursorOf resetting to 0 after a completed round, because snap_generation points at the COMPLETION +/// generation whose fold_seal lives at the parent) is MASKED. Disable trim to expose it: the publish event +/// stays in the journal, so a round that re-folds from 0 emits a SECOND +1 and drives the blob's in-degree +/// to 2 (a silent over-pin => leak). The fix carries the per-shard fold cursor into the completion seal so +/// the next round recovers the exact cursor. Asserts in-degree stays EXACTLY 1 across >= 2 re-folds. +TEST(CASGCRound, FoldCursorSurvivesAcrossRoundsWithoutTrim) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.setTrimEnabledForTest(false); /// keep the folded publish event in the journal across rounds + + /// Round 1 folds the +1 edge: in-degree 1, blob pinned. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// Several more rounds. The publish event is STILL in the journal (trim off). Each round must + /// recover the exact sealed cursor and re-fold NOTHING for this shard — in-degree stays exactly 1. + for (int round = 0; round < 3; ++round) + { + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) + << "round " << round << ": a folded-but-untrimmed event was re-folded => blob in-degree double-counted"; + } + + /// No-loss throughout: the live blob and its owner body are intact. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + EXPECT_TRUE(manifestExists(*backend, store->layout(), ManifestId{ns, r})); +} + +/// Multi-ref sharing (INV-NO-LOSS): one blob referenced by TWO committed refs is spared until BOTH +/// drop. Dropping the first ref must NOT collect the blob (the second ref still pins it); only after the +/// second ref drops does the round collect it. +TEST(CASGCRound, SharedBlobSparedUntilBothRefsDrop) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xA1); + const ManifestRef r2 = ref(2, 0xA2); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + /// Two distinct manifests at two distinct refs, BOTH referencing the same shared blob 1. + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl1", std::nullopt, r1); + publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 2); /// two source edges + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + + /// Drop the FIRST ref: in-degree falls to 1, blob STILL pinned by tbl2 (spared). + dropRefTransition(*backend, store->layout(), ns, "tbl1", r1); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "shared blob must survive while a second ref still names it"; + + /// Drop the SECOND ref: in-degree reaches 0, blob is finally collected. + dropRefTransition(*backend, store->layout(), ns, "tbl2", r2); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} + +/// `gc_round_outcome_entry_budget` bounds only the `GcOutcomes` AUDIT row per spared decision, never the +/// decision itself. Five blobs are condemned (owner dropped, indegree 0, durable retired rows), then -- +/// BEFORE graduation -- a fresh manifest re-references all five (the `CASThreeCursorMerge.RecoverySpares` +/// shape, scaled up and driven through the real round path): recovery wins unconditionally for every one +/// of them. `CASGCRetiredSpared` and blob survival prove all five decisions happened regardless of the +/// budget, but with a budget of 2, only 2 of the 5 get a row in the round's `GcOutcomes` log, so +/// `RoundReport::spared` (tallied from that log) reports 2, not 5. +TEST(CASGCRound, OutcomeEntryBudgetCapsSparedLogRowsWithoutRecondemning) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_round_outcome_entry_budget = 2, + .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xC1); + const ManifestRef r2 = ref(2, 0xC2); + constexpr int kBlobs = 5; + + std::vector entries; + for (int i = 0; i < kBlobs; ++i) + { + writeBlobBody(*backend, store->layout(), DB::UInt128(i + 1)); + entries.push_back(blobEntryFor("p" + std::to_string(i), DB::UInt128(i + 1))); + } + writeManifestRaw(*backend, store->layout(), ns, r1, entries); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + for (int i = 0; i < kBlobs; ++i) + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(i + 1)), 1); + + /// Drop the only ref: one round later all five blobs are condemned (indegree 0, durable retired rows). + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + for (int i = 0; i < kBlobs; ++i) + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(i + 1)), 0); + store->renewWatermarkOnce(); + + /// BEFORE graduation, a fresh manifest re-references all five: the NEXT fold recomputes indegree 1 for + /// every one of them -- recovery wins over graduation for every entry, unconditionally. + writeManifestRaw(*backend, store->layout(), ns, r2, entries); + publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); + + const auto spared_events_before = ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared].load(); + const RoundReport rep = runRegularRoundReclaiming(gc); + ASSERT_TRUE(rep.acquired_lease); + const uint64_t total_spared_reported = rep.spared; + + /// THE LOAD-BEARING ASSERTION: the audit log under-reports (capped at the budget) while every + /// decision it under-reports still happened correctly. + EXPECT_EQ(total_spared_reported, 2u) + << "GcOutcomes rows must be capped at gc_round_outcome_entry_budget, not one per spared entry"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared].load() - spared_events_before, kBlobs) + << "every spared decision must still happen even when its audit row is capped"; + for (int i = 0; i < kBlobs; ++i) + { + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(i + 1))) + << "blob " << i << " must survive -- the cap must never re-condemn a spared entry"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(i + 1)), 1); + } +} + +/// Spare-during-the-pass, multi-blob discrimination: a drop condemns two blobs; in the SAME window +/// between rounds (before the next pass folds), one of them is re-referenced under a fresh ref. The pass +/// folds the racing publish and SPARES the re-referenced blob (recovery wins in the pass merge, dropping +/// its retired entry), while the genuinely-unreferenced blob proceeds through the condemn -> graduate -> +/// delete pipeline. The discriminating assertion: at fixpoint, one is spared (kept) and the other gone. +TEST(CASGCRound, RepublishDuringFenceWindowSparesOnlyReReferencedBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xB1); + const ManifestRef r2 = ref(2, 0xB2); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); /// kept (will be re-referenced) + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); /// genuinely dropped + writeManifestRaw(*backend, store->layout(), ns, r1, + {blobEntryFor("a", DB::UInt128(1)), blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1); + + /// Repoint the ref from r1 to r2 between rounds: ONE event {old=committed(r1), new=committed(r2)}. + /// The -1 (r1's body: blobs 1 AND 2) and +1 (r2's body: blob 1 only) net to in-degree 1 for blob 1 + /// (re-referenced => SPARED in the pass merge) and 0 for blob 2 (genuinely unreferenced => condemned, + /// then reclaimed by the ack-floor pipeline). (A separate drop THEN repoint would double-count the -1 + /// on r1's blobs and drive blob 2 to -1 — an undercount the in-degree fold fails closed on.) + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", r1, r2); + + driveToFixpoint(*backend, store, gc); + /// Blob 1 is re-referenced (net in-degree 1) => SPARED; blob 2 is genuinely unreferenced => GONE. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the racing republish must spare blob 1"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(2))) + << "the genuinely-unreferenced blob 2 must be collected"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 0); +} + +/// Idempotent fixpoint: once a pool is quiescent (all live refs folded, nothing to collect), repeated +/// rounds are pure no-ops — no blob is collected, no manifest disappears, the in-degree generation is +/// stable, and no round throws. The split-brain-safety bedrock: every step is idempotent. +TEST(CASGCRound, IdempotentRerunAtFixpointIsNoOp) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + const ManifestId id{ns, r}; + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + const uint64_t gen0 = currentGenerationOf(*backend, store->layout()); + + /// At quiescence: a fresh round does NO work (no candidates/deletes/spares) and changes nothing. + const RoundReport quiescent = gc.runRegularRound(); + EXPECT_TRUE(quiescent.acquired_lease); + EXPECT_EQ(quiescent.candidates, 0u); + EXPECT_EQ(quiescent.deleted, 0u); + EXPECT_EQ(quiescent.spared, 0u); + + EXPECT_NO_THROW(driveToFixpoint(*backend, store, gc)); + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); /// no-loss + EXPECT_TRUE(manifestExists(*backend, store->layout(), id)); /// no-loss + /// The CONTENT no-op invariant: the live blob's durable in-degree is unchanged (still pinned). The + /// generation POINTER advances every round by design (each fold seals a fresh generation for durable + /// cursor coverage, and recheck seals the completion generation), even when no edges change — so the + /// quiescence guarantee is "no candidates/deletes/spares + nothing lost", NOT a frozen generation. + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + EXPECT_GE(currentGenerationOf(*backend, store->layout()), gen0) + << "the generation pointer is monotone; a quiescent round never moves it backward"; +} + +/// Split-brain: two leaders racing the same pool only DUPLICATE WORK, never double-delete or lose data. +/// gc1 leads and folds the live publish; the ref is then dropped; gc2 steals the lease (stale leader) +/// and both contend to collect the now-unreferenced blob. The exact-token delete is the only destructive +/// authority, so the blob is removed exactly once and a losing/duplicate attempt is a harmless 404/412 — +/// no exception escapes, and the blob ends up gone exactly once with no dangling owner. +TEST(CASGCRound, SplitBrainLeadersOnlyDuplicateWork) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + const ManifestId id{ns, r}; + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc1(store, kGcA); + Gc gc2(store, kGcB); + + /// gc1 leads; fold the publish edge. + ASSERT_TRUE(runRegularRoundReclaiming(gc1).acquired_lease); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// The ref is dropped; gc1 stalls. gc2 observes the frozen lease twice and STEALS (new epoch). + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + EXPECT_FALSE(runRegularRoundReclaiming(gc2).acquired_lease); /// obs #1 + ASSERT_TRUE(runRegularRoundReclaiming(gc2).acquired_lease); /// obs #2 => steal + + /// Both leaders now drive rounds. The blob is collected exactly once; duplicate attempts are + /// harmless. No round throws. + EXPECT_NO_THROW(driveToFixpoint(*backend, store, gc2)); + EXPECT_NO_THROW(driveToFixpoint(*backend, store, gc1)); /// the revived stale leader backs off / duplicates harmlessly + + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the dropped blob must be collected exactly once across both leaders"; + EXPECT_FALSE(manifestExists(*backend, store->layout(), id)); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +/// (`CASGCRound.TrimOnlyBelowSealedCoverage` and the B12 lazy/batched-trim tests +/// `LazyTrimSkipsSmallJournalAndKeepsTokenStable`, `LazyTrimCompactsAtThresholdOrSoftLimit`, +/// `MaintenanceTrimCompactsEverythingOnce` were removed with the snapshot+log ref model. They asserted +/// GC compacts a MUTABLE shard journal in place (INV-JOURNAL-COVERAGE / `gc_trim_min_events` gates). +/// Immutable `_log` objects are never trimmed in place: covered `_log`/`_snap` keys are DELETED by +/// ref-object cleanup once BOTH the durable cursor AND a checkpoint-named validated recovery triple +/// cover them -- exercised in `gtest_cas_ref_gc.cpp` (`RefObjectCleanupRetainsCheckpointNamedTriple`).) + +/// ---- INTENTIONALLY NOT PORTED (covered elsewhere or obsolete in the manifest model) ---- +/// +/// The removed snap/cascade/tree cases and where their behaviour now lives: +/// - CasGcFold.{FreshUploadsAreNeverCandidates, DropZeroesTreeButChildStaysPinned, +/// RepublishSameRefIsLastOpWins, ExpansionIsOncePerTree, IncrementalSecondFoldOnlyNewRecords, +/// DurableSnapBeforeCursorAdvance, ForeignDivergentGenerationIsProbedPast, +/// GenerationProbeRecoversAfterLostCursorCas, SnapShardsOtherThanOneIsNotImplemented, +/// AbsentTree*, NoChurnRound*} — fold-step behaviour now in gtest_cas_gc_fold.cpp +/// (CommittedAdd/Removal/Precommit/FoldBarrier/Clamp+anomaly/RefMismatch). +/// - CasGcCorruptCommittedTree.MissingTreeOfLiveRefDoesNotHaltGc — now +/// CASGCFold.CommittedMissingBodyClampsCursorAndRecordsAnomaly. +/// - CASGCRetire.{Observes*, AbsentCandidate*, DeletedCandidate*, DeleteTimePrune*, BlobOnlyPrune*, +/// RetireForgets*, RetireSetsDurable*, Diverged*, BlobHeaderUnderflow*, RetireUsesFoldCommitted*, +/// RetireReplayAdoptsOwnCrashedAttempt} — retire-step behaviour now split between +/// gtest_cas_gc_ack_floor.cpp and the retire-view suite. +/// - CASGCRecheck.{SparedWhenPublishRacesTheFence, ReplacedWhenResurrectionWins, AbsentWhenAlreadyGone} +/// — now CASGCRecheck.{PublishRacingFenceSparesBlob, UnreferencedBlobDeletedExactToken}. +/// - CasGcFence.* / CasGcDiscovery.UsesRegistryNotList — the fence machinery is retired; the equivalent +/// no-op-round-does-not-mutate-ref-shards property is gtest_cas_gc_ack_floor.cpp:: +/// CASGCAckFloor.NoOpRoundDoesNotMutateRefShards (+ helper registerNamespaceRaw discovery is +/// exercised by every fold test). +/// - CasGcCascade.* — the cascade/closure model is REMOVED; in-degree is per-blob, so a shared +/// child surviving one parent's deletion is now CASGCRound.SharedBlobSparedUntilBothRefsDrop above, +/// and "never cascades on replaced" is CASGCRound.RepublishDuringFenceWindowSparesOnlyReReferencedBlob. +/// - CasGcTrim.* — now gtest_cas_gc_resume.cpp::CASGCRound.TrimDropsFoldedOwnerEvents. +/// - CasGcResume.{CompletesRoundAfterCrashBeforeFencePersist, AdoptsOutcomesAfterCrashBeforeCascadePersist} +/// — now gtest_cas_gc_resume.cpp::CasGcResume.ResumeFromDurableFoldSealCompletesRound. +/// - CasGcScenario.ZombieDeleteAfterResurrectIs412 — relied on the snap/tree publish path + held +/// in-flight deletes; the in-degree-spare equivalent is RepublishDuringFenceWindowSparesOnly... +/// above (exact-token delete is the sole authority; a zombie carrying a stale token 412s). +/// - CASGCRound.PreviewDeletesIsWriteFreeAndSubsetOfUnreachable — previewDeletes survives, but it is +/// covered by the fsck/preview suite; not duplicated here. +/// - CasGcWatermark.LiveBuildPrecommitHonoredAcrossGcRounds / +/// CASGCRetire.ReclaimsAbandonedPrecommitWhenFloorPasses — precommit removal is now the WRITER's job +/// (an exact `owner_transition` on abandon, or a fenced successor's stale-precommit sweep); GC no +/// longer reclaims abandoned precommits. Exercised by the orphan-manifest-sweep / build-root suites. + +/// B9 snap-generation retention, reimplemented over the run/generation model: after a generation is +/// adopted the GC prunes the per-generation seal/run/cleanup objects of generations at or below the +/// retention floor (snap_generation - gc_snapshot_generations_to_keep), advancing snap_pruned_through. This +/// test drives enough rounds to accumulate several generations, then asserts that everything at or below +/// the floor is GONE while the last `keep` generations (and the live current one) remain. +TEST(CASGCSnapRetention, PrunesOldGenerationsKeepingLastThree) +{ + auto backend = std::make_shared(); + /// keep the default 3 generations; one root shard so cursor keys are "ns/0". + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 3, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + /// Several quiescent rounds, each advancing the generation pointer (fold + completion). Enough to + /// push generations below the floor. + for (int i = 0; i < 8; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const GcState st = readState(*backend, *store); + const uint64_t keep = 3; + ASSERT_GT(st.snap_generation, keep); + const uint64_t floor = st.snap_generation - keep; + + /// snap_pruned_through reached the floor (bounded burst is large enough for this generation count). + EXPECT_EQ(st.snap_pruned_through, floor) + << "retention cursor must reach the floor (snap_generation - keep)"; + + /// Every generation at or below the floor is fully gone (fold seal absent). + for (uint64_t g = 1; g <= floor; ++g) + { + EXPECT_FALSE(backend->head(store->layout().foldSealKey(g, st.snap_attempt)).exists) + << "fold seal of pruned generation " << g << " must be gone"; + EXPECT_FALSE(backend->head(store->layout().blobTargetRunKey(g, st.snap_attempt, /*shard*/0, /*seq*/0)).exists) + << "blob-target run of pruned generation " << g << " must be gone"; + } + + /// The fold seal at the current generation survives (the live in-degree view). + EXPECT_TRUE(backend->head(store->layout().foldSealKey(st.snap_generation, st.snap_attempt)).exists) + << "the current generation's seal must NOT be pruned"; + + /// No-loss: the live blob and owner body are intact throughout retention pruning. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + EXPECT_TRUE(manifestExists(*backend, store->layout(), ManifestId{ns, r})); +} + +/// Task 9 (wholesale generation-retention): a generation may hold artifacts under MULTIPLE attempts +/// (each round mints a fresh `lease.seq`, and a deposed leader may have written debris under its own +/// unadopted attempt). When a generation ages past the retention floor it must be reclaimed WHOLESALE +/// — every attempt's artifacts (incl. the attempt-scoped `retired/` and `outcomes/` sets that now live +/// under `gc/gen//attempt//`), not just the final adopted attempt's. This test plants a retired +/// set AND a decoy fold seal under a NON-adopted attempt at an old generation, ages that generation out, +/// and asserts the whole `gc/gen//` subtree is gone (the per-key single-attempt prune leaked it). +TEST(CASGCSnapRetention, WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcomes) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 3, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + /// One round to establish the first completed generation and learn its adopted attempt (derive both + /// from gc/state — never hardcode a generation; the round folds and completes, so it is > 1). + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const GcState st1 = readState(*backend, *store); + ASSERT_GT(st1.snap_generation, 0u); + const uint64_t old_gen = st1.snap_generation; + const uint64_t adopted_attempt_g1 = st1.snap_attempt; + + /// Plant debris under a NON-adopted attempt of generation 1: a retired set, an outcomes log, a fold + /// seal, and a blob-target run — exactly the families a deposed leader would have written before its + /// CAS failed. The per-key single-attempt prune (keyed on the FINAL snap_attempt) never touches them. + const uint64_t decoy_attempt = adopted_attempt_g1 + 777; + const String decoy_outcomes = store->layout().outcomesKey(old_gen, decoy_attempt, /*round*/0, /*shard*/0); + const String decoy_seal = store->layout().foldSealKey(old_gen, decoy_attempt); + const String decoy_run = store->layout().blobTargetRunKey(old_gen, decoy_attempt, /*shard*/0, /*seq*/0); + backend->putIfAbsent(decoy_outcomes, "decoy-outcomes"); + backend->putIfAbsent(decoy_seal, "decoy-seal"); + backend->putIfAbsent(decoy_run, "decoy-run"); + + /// Drop the ref so the next fold writes a FRESH run under a newer generation and the adopted seal's + /// blob_target ref moves OFF `old_gen`. Under T0 reference-parent carry, a still-referenced generation + /// is deliberately retained (its run is live), so `old_gen` can only age out once nothing references + /// its run anymore — which the drop guarantees. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + /// Age generation 1 well past the retention floor (keep=3): several more quiescent rounds. + for (int i = 0; i < 8; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const GcState st = readState(*backend, *store); + ASSERT_GT(st.snap_generation, old_gen + 3) << "generation 1 must be below the retention floor"; + + /// The ENTIRE gc/gen// subtree — across ALL attempts — must be reclaimed. + EXPECT_FALSE(backend->head(decoy_outcomes).exists) << "non-adopted outcomes log leaked past retention"; + EXPECT_FALSE(backend->head(decoy_seal).exists) << "non-adopted fold seal leaked past retention"; + EXPECT_FALSE(backend->head(decoy_run).exists) << "non-adopted blob-target run leaked past retention"; + + /// Nothing remains under the old generation prefix at all. + const ListPage residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000); + EXPECT_TRUE(residue.keys.empty()) << "old generation prefix must be fully reclaimed; left " + << residue.keys.size() << " objects"; + + /// The drop was necessary to move the seal's blob_target ref off `old_gen` so it could age out (under + /// T0 a still-referenced generation is deliberately retained — see the comment above). The blob is + /// condemned by the drop and, being round-paced, graduates and is physically deleted well within the + /// 8 quiescent rounds above — retire drain is not the property under test here (generation retention + /// is), so this only asserts the reclaim pipeline is not itself broken by the retention plumbing. + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); + /// The owner-removed manifest body IS reclaimed by the part-manifest cleanup pass over the aging rounds. + EXPECT_FALSE(manifestExists(*backend, store->layout(), ManifestId{ns, r})); +} + +/// `deletePrefixWholesale`'s callers now draw from the round's shared +/// object-count budget instead of `UINT64_MAX`, and `snap_pruned_through` must advance only past a +/// FULLY drained generation -- never past one the budget cut short, or its undeleted remainder would +/// be stranded behind a cursor this loop never revisits. A tiny budget (2 objects/round) against a +/// generation carrying far more debris than that forces multiple rounds to fully drain it; the +/// invariant under test is that AT EVERY ROUND, `snap_pruned_through >= old_gen` implies the old +/// generation's prefix is already empty -- the cursor never claims completion it has not earned. +TEST(CASGCSnapRetention, PruneRespectsPrefixWholesaleBudgetAndNeverStrandsAPartialGeneration) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_snapshot_generations_to_keep = 3, + .gc_round_prefix_wholesale_budget = 2, + .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const GcState st1 = readState(*backend, *store); + const uint64_t old_gen = st1.snap_generation; + + /// Ten extra debris objects under generation 1's prefix -- far more than the 2-object round budget + /// can wholesale-delete in a single pass, regardless of whatever real fold artifacts already live + /// there. + for (int i = 0; i < 10; ++i) + backend->putIfAbsent(store->layout().gcGenPrefix(old_gen) + "debris" + std::to_string(i), "x"); + + /// Move the ref off `old_gen`'s run (as `WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcomes` + /// does) so the WHOLESALE RETENTION PRUNE -- not the one-shot post-CAS hand-off -- is what + /// eventually processes this generation once the cursor reaches it. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + size_t previous_residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000).keys.size(); + std::optional drain_start_round; /// first round the residue count actually DROPS + std::optional drain_done_round; /// first round the residue reaches zero + for (int i = 0; i < 40 && !drain_done_round; ++i) + { + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const GcState st = readState(*backend, *store); + const ListPage residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000); + + if (st.snap_pruned_through >= old_gen) + EXPECT_TRUE(residue.keys.empty()) + << "round " << i << ": snap_pruned_through (" << st.snap_pruned_through + << ") claims generation " << old_gen << " is behind it, but " << residue.keys.size() + << " object(s) remain -- the cursor advanced past a partially-drained prefix"; + + if (!drain_start_round && residue.keys.size() < previous_residue) + drain_start_round = i; + if (residue.keys.empty()) + drain_done_round = i; + previous_residue = residue.keys.size(); + } + + ASSERT_TRUE(drain_start_round.has_value()) << "the round loop never even started draining the debris"; + ASSERT_TRUE(drain_done_round.has_value()) + << "the budget-limited generation must eventually fully drain within a generous round bound"; + /// THE LOAD-BEARING ASSERTION: with a 2-object budget against 10+ debris objects, draining cannot + /// finish the SAME round it starts -- it must take several rounds. An unbounded + /// `deletePrefixWholesale` call (the mutation this pins) drains everything the round it starts, + /// collapsing this gap to zero. + EXPECT_GT(*drain_done_round, *drain_start_round) + << "draining finished the same round it started -- the per-round budget is not load-bearing"; +} + +/// Reclaim-VIA-RETENTION of a non-adopted current-generation attempt orphan (KISS prune model). A +/// deposed leader can write its fold seal under an attempt that lost CAS #1 to a higher-seq adopter — +/// debris at the FOLD generation under a NON-adopted attempt. There is NO per-round current-generation +/// attempt-sweep anymore (it cost a per-round LIST for a rare collision); the wholesale +/// generation-retention prune is the SOLE reclaimer. So such an orphan is NOT reclaimed within one +/// round; instead it waits until its generation ages past `keep` and the wholesale prefix-delete +/// reclaims the whole `gc/gen//` subtree — every attempt at once, including this orphan. This test +/// plants the orphan at a fold generation, ages that generation out, and asserts retention reclaims it. +TEST(CASGCSnapRetention, ReclaimsNonAdoptedCurrentGenAttemptViaRetention) +{ + auto backend = std::make_shared(); + /// keep=3 retention floor (matches WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcomes). + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 3, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + /// Drive a couple of rounds so snap_attempt is comfortably above 0 (a low orphan seq exists below it). + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const GcState st = readState(*backend, *store); + ASSERT_GT(st.snap_attempt, 0u) << "need snap_attempt > 0 so a strictly-older orphan attempt exists"; + + /// Plant a deposed competitor's debris at the next round's FOLD generation under an attempt strictly + /// older than that round's adopted attempt — exactly the orphan the old per-round sweep targeted. + const uint64_t orphan_gen = st.snap_generation + 1; + const uint64_t orphan_attempt = st.snap_attempt - 1; + const String orphan_seal = store->layout().foldSealKey(orphan_gen, orphan_attempt); + const String orphan_run = store->layout().blobTargetRunKey(orphan_gen, orphan_attempt, 0, 0); + backend->putIfAbsent(orphan_seal, "orphan-seal"); + backend->putIfAbsent(orphan_run, "orphan-run"); + + /// One more round folds into `orphan_gen` and completes. The orphan must SURVIVE this round — there + /// is no current-generation sweep; retention has not yet reached `orphan_gen`. + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + EXPECT_TRUE(backend->head(orphan_seal).exists) + << "orphan must survive its own round — there is no per-round current-gen sweep"; + + /// Age `orphan_gen` well past the retention floor (keep=3): several more quiescent rounds. The + /// wholesale generation-retention prune then reclaims the WHOLE `gc/gen//` subtree, + /// including this non-adopted attempt's debris. + for (int i = 0; i < 8; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const GcState st_after = readState(*backend, *store); + ASSERT_GT(st_after.snap_generation, orphan_gen + 3) << "orphan_gen must be below the retention floor"; + + EXPECT_FALSE(backend->head(orphan_seal).exists) + << "non-adopted attempt orphan must be reclaimed by wholesale retention once its generation ages out"; + EXPECT_FALSE(backend->head(orphan_run).exists) + << "the whole orphan subtree must be reclaimed by wholesale retention"; + + /// No-loss: the live data is intact throughout. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + EXPECT_TRUE(manifestExists(*backend, store->layout(), ManifestId{ns, r})); +} + +/// ---- Task 7 (2026-07-02 snapshot-streaming): ref-aware retention + post-CAS hand-off delete ---- + +/// Retention must NOT reclaim a generation whose run the live seal still references, EVEN once the +/// retention cursor (`snap_pruned_through`) has advanced past that generation. With `keep=1` and a live +/// ref that idle-carries across generations, `pruneSupersededGenerations` SKIPS gen-1's prefix every +/// round while advancing the cursor over it. The gen-1 run object (physically holding the seal's ref) +/// must survive, and folding/in-degree resolution THROUGH the carried ref must keep working. +TEST(CASGCRetention, PruneRetainsLiveReferencedRun) +{ + auto backend = std::make_shared(); + /// keep=1: the retention floor is aggressive so the cursor reaches gen-1's neighbourhood fast. + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 1, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); // gen 1: the blob's run is sealed under gen-1's key namespace + const GcState st1 = readState(*backend, *store); + const uint64_t ref_gen = st1.snap_generation; + + /// The gen-1 seal's ref names gen-1's physical run key — capture it so we can assert the OBJECT + /// (not just the generation number) survives retention. + const auto seal1 = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + ASSERT_EQ(seal1.blob_target_runs.size(), 1u); + const String referenced_run_key = seal1.blob_target_runs.front().key; + ASSERT_EQ(seal1.blob_target_runs.front().generation, ref_gen); + ASSERT_TRUE(backend->head(referenced_run_key).exists); + + /// Several idle rounds: no delta, no retired => pure ref-carry. Each round advances the generation + /// and, once adopted_generation > keep, drives the retention prune forward. gen-1 is referenced every + /// round, so it is SKIPPED (retained) even as `snap_pruned_through` climbs past it. + for (int i = 0; i < 6; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + const GcState st = readState(*backend, *store); + /// The cursor has advanced strictly past the referenced generation (the retention prune SKIPPED it + /// but still moved the high-water cursor forward) — this is the exact window Task 7 guards. + ASSERT_GT(st.snap_pruned_through, ref_gen) + << "the retention cursor must have advanced past the still-referenced generation"; + + /// The referenced run object is STILL ALIVE despite the cursor passing its generation. + EXPECT_TRUE(backend->head(referenced_run_key).exists) + << "a run referenced by the live seal must be retained even after the cursor passes its generation"; + + /// The current seal still references that same physical gen-1 object (carried, not reconstructed), + /// and in-degree resolution THROUGH the carried ref still works. + const auto seal_now = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + ASSERT_EQ(seal_now.blob_target_runs.size(), 1u); + EXPECT_EQ(seal_now.blob_target_runs.front().key, referenced_run_key); + EXPECT_EQ(seal_now.blob_target_runs.front().generation, ref_gen); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) + << "folding still resolves in-degree through the retained, carried parent ref"; + + /// No-loss end-to-end: the live blob is intact. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); +} + +/// When a later delta finally REPLACES the carried ref with a fresh run, the superseded old-generation +/// run — whose generation the retention cursor already passed while it was retained — is reclaimed by the +/// post-CAS HAND-OFF delete in `runRegularRound` (the wholesale prune never revisits a generation behind +/// its cursor, so the ordinary prune would leak it). The whole `gc/gen//` prefix must be gone. +TEST(CASGCRetention, HandOffDeletesSupersededRef) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 1, .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + runRegularRoundReclaiming(gc); // gen 1: run sealed under gen-1 + const GcState st1 = readState(*backend, *store); + const uint64_t old_gen = st1.snap_generation; + const String old_prefix = store->layout().gcGenPrefix(old_gen); + ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) << "gen-1 prefix must be populated"; + + /// Idle-carry the gen-1 ref until the retention cursor has advanced strictly PAST gen-1. Until it + /// does, a normal prune could still reclaim gen-1 when the ref moves — the hand-off is only load- + /// bearing once gen-1 is BEHIND the cursor. + for (int i = 0; i < 6; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_GT(readState(*backend, *store).snap_pruned_through, old_gen) + << "gen-1 must be behind the retention cursor before the hand-off is exercised"; + /// gen-1 is retained (referenced) even though the cursor passed it. + ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + << "the referenced gen-1 prefix must still exist before the ref moves off it"; + + /// A real delta: swap the ref to a new manifest naming a different blob. The next fold writes a FRESH + /// run under the new generation and the seal's shard-0 ref moves OFF gen-1. + const ManifestRef r2 = ref(2, 0xBB); + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", r1, r2); + + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); // folds through the carried ref; ref leaves gen-1 + + /// The seal no longer references gen-1 ... + const GcState st_after = readState(*backend, *store); + const auto seal_after = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st_after.snap_generation, st_after.snap_attempt))->bytes); + for (const RunRef & rr : seal_after.blob_target_runs) + EXPECT_NE(rr.generation, old_gen) << "the live seal must have moved its ref off gen-1"; + + /// ... and the post-CAS hand-off delete reclaimed gen-1's WHOLE prefix (not just the single run + /// object): seal, attempt subtree, run — all gone. The ordinary prune would have leaked it because its + /// cursor is already past gen-1. + const ListPage residue = backend->list(old_prefix, "", 1000); + EXPECT_TRUE(residue.keys.empty()) + << "the superseded gen-1 prefix must be hand-off deleted; left " << residue.keys.size() << " objects"; + + /// The now-referenced blob 2 is intact; folding through the fresh run resolves it. + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(2))); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1); +} + +/// The post-CAS hand-off draws from its OWN reserve, so a prune that spends its ENTIRE (separate, tiny) +/// budget in a round can never leave the hand-off with zero. Combines the two existing shapes: a +/// debris-heavy generation only the ordinary PRUNE ever touches (like +/// `PruneRespectsPrefixWholesaleBudgetAndNeverStrandsAPartialGeneration`, mid-drain over several rounds on +/// a starvation-small prune budget), running CONCURRENTLY with an idle-carried ref that finally moves off +/// its generation (like `HandOffDeletesSupersededRef`) in one of those very same mid-drain rounds. +TEST(CASGCRetention, HandoffOwnBudgetSurvivesAPruneHeavyRound) +{ + auto backend = std::make_shared(); + /// `gc_shards = 2` with the "keep" and "debris" blobs routed to DIFFERENT shards is load-bearing: with + /// the default single shard, ANY delta anywhere rewrites the pool's one shared run object every round, + /// which would drag the "keep" ref's physical run forward the moment the debris table is touched -- + /// destroying the idle-carry this test depends on. Two independent shards keep debris activity from + /// disturbing the kept ref's generation at all until its ref is explicitly moved. + /// `keep=5` (not the more aggressive `keep=1` other hand-off tests use) is ALSO load-bearing: the + /// debris generation must still be numerically AHEAD of the cursor at the moment its own drop folds, + /// or that fold's post-CAS hand-off phase -- not the ordinary prune -- would claim it (the same + /// one-round "parent-seal protects, then hand-off claims" shape `HandOffDeletesSupersededRef` relies + /// on, which this test must deliberately avoid for the debris generation). + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_snapshot_generations_to_keep = 5, + .gc_shards = 2, + .gc_round_prefix_wholesale_budget = 2, /// prune: starvation-small, shared by nothing else + .gc_round_handoff_prefix_wholesale_budget = 5, /// hand-off: its own separate reserve + .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"00/aa@cas@"}; + Gc gc(store, kGc); + /// `blobShard` uses only the digest's high 64 bits, so a small integer's `UInt128` (high bits zero) + /// always routes to shard 0 regardless of `gc_shards` -- these two differ in the high half so they + /// land in different shards of a 2-shard pool. + const DB::UInt128 blob_keep_1 = hexToU128("00000000000000010000000000000000"); + const DB::UInt128 blob_keep_2 = hexToU128("00000000000000010000000000000001"); + const DB::UInt128 blob_debris = hexToU128("00000000000000020000000000000000"); + + /// The HAND-OFF generation: ns "keep" idle-carries this ref for several rounds until the cursor has + /// advanced strictly past it (referenced generations are skipped for free -- no budget spent). + const ManifestRef r_keep_1 = ref(1, 0xE1); + writeBlobBody(*backend, store->layout(), blob_keep_1); + writeManifestRaw(*backend, store->layout(), ns, r_keep_1, {blobEntryFor("a", blob_keep_1)}); + publishCommittedTransition(*backend, store->layout(), ns, "keep", std::nullopt, r_keep_1); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const uint64_t handoff_gen = readState(*backend, *store).snap_generation; + const String handoff_prefix = store->layout().gcGenPrefix(handoff_gen); + ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()); + + for (int i = 0; i < 20; ++i) + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_GT(readState(*backend, *store).snap_pruned_through, handoff_gen) + << "the hand-off generation must be behind the cursor before this test is meaningful"; + ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()) + << "still referenced -- must survive despite the cursor having passed it"; + + /// The PRUNE-DEBRIS generation: a second table, on the OTHER shard, unreferenced from the start, + /// carrying far more debris than the tiny prune budget can drain in one round. Minted well AHEAD of + /// the current cursor (see the `keep=5` note above), so its own drop-fold is NOT immediately + /// hand-off-eligible. + const ManifestRef r_debris = ref(1, 0xE2); + writeBlobBody(*backend, store->layout(), blob_debris); + writeManifestRaw(*backend, store->layout(), ns, r_debris, {blobEntryFor("b", blob_debris)}); + publishCommittedTransition(*backend, store->layout(), ns, "debris", std::nullopt, r_debris); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const uint64_t debris_gen = readState(*backend, *store).snap_generation; + ASSERT_GT(debris_gen, readState(*backend, *store).snap_pruned_through) + << "the debris generation must still be ahead of the cursor when its drop folds, or the hand-off " + "(not the prune) would claim it"; + for (int i = 0; i < 10; ++i) + backend->putIfAbsent(store->layout().gcGenPrefix(debris_gen) + "debris" + std::to_string(i), "x"); + dropRefTransition(*backend, store->layout(), ns, "debris", r_debris); + + /// Drive rounds until the debris generation is MID-DRAIN (prune has started but not yet finished it -- + /// the round-budget of 2 against 10+ objects guarantees several such rounds exist). + bool mid_drain = false; + for (int i = 0; i < 20 && !mid_drain; ++i) + { + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + const size_t residue = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + mid_drain = residue > 0 && residue < 10; + } + ASSERT_TRUE(mid_drain) << "the debris generation never reached a partially-drained state to test against"; + ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()) + << "the hand-off generation must still be intact (untouched) going into the contended round"; + + /// NOW, in a round where the prune is busy mid-drain on the debris generation (spending its entire + /// small budget there), move the kept ref off the hand-off generation -- a fresh manifest replaces it. + const ManifestRef r_keep_2 = ref(2, 0xE3); + writeBlobBody(*backend, store->layout(), blob_keep_2); + writeManifestRaw(*backend, store->layout(), ns, r_keep_2, {blobEntryFor("a", blob_keep_2)}); + publishCommittedTransition(*backend, store->layout(), ns, "keep", r_keep_1, r_keep_2); + const size_t debris_residue_before = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + /// THE LOAD-BEARING ASSERTIONS: the prune spent its whole (separate) budget on the debris generation + /// this very round (proving the two really contended for I/O in the same round) ... + const size_t debris_residue_after = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + EXPECT_EQ(debris_residue_before - debris_residue_after, 2u) + << "the prune must have spent its entire per-round budget on the debris generation this round"; + /// ... and the hand-off, drawing from its OWN reserve, still fully reclaimed the generation the ref + /// just moved off -- zero, not starved to zero by the prune's consumption. + EXPECT_TRUE(backend->list(handoff_prefix, "", 1000).keys.empty()) + << "the hand-off must not be starved by a prune-heavy round that exhausted a SEPARATE budget"; +} + +/// triage #5, driven through the REAL call site (`Gc::runRegularRound`, not a test seam): a losing +/// leader's pre-CAS wholesale generation-retention prune must never destroy a generation the PARENT +/// (currently-adopted, pre-fold) seal still references, even when the round's own PROPOSED seal has +/// already moved off it and the round's own `gc/state` CAS then loses. `GcStateCasFaultBackend` makes +/// this round's own round-commit CAS return `Conflict` — deterministically and single-threaded standing +/// in for a concurrent leader winning first — which is the only condition under which the fix is +/// externally observable: a round whose own CAS SUCCEEDS reclaims the same generation moments later via +/// the existing (unrelated, unchanged) post-CAS hand-off delete regardless of this fix, so a plain +/// successful round cannot tell bug from fix apart. +TEST(CASGCRetention, LosingRoundNeverDestroysParentSealGeneration) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 1}); + const Layout & layout = store->layout(); + backend->faulted_key = layout.gcStateKey(); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xAA); + + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); // round 1: gen=1 adopted, referencing blob 1's run. + const GcState st1 = readState(*backend, *store); + const uint64_t g_parent = st1.snap_generation; + + const auto seal1 = decodeFoldSeal(backend->get(layout.foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + ASSERT_EQ(seal1.blob_target_runs.size(), 1u); + const String parent_run_key = seal1.blob_target_runs.front().key; + const String parent_gen_prefix = layout.gcGenPrefix(g_parent); + ASSERT_FALSE(backend->list(parent_gen_prefix, "", 1000).keys.empty()); + + /// A real delta: swap the ref to a new manifest naming a different blob. The next fold will move + /// shard 0's run OFF `g_parent` onto a fresh generation. + const ManifestRef r2 = ref(2, 0xBB); + writeBlobBody(*backend, layout, DB::UInt128(2)); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + publishCommittedTransition(*backend, layout, ns, "tbl", r1, r2); + + /// Arm the fault for the NEXT round's SECOND casPut on gc/state, not its first: the first is + /// `acquireOrRenewLease`'s own lease-renewal CAS (must SUCCEED, so the round actually folds), and the + /// second is the round's final round-commit CAS (the one that must LOSE, exactly as if a concurrent + /// leader had already committed a different seal first). + const size_t calls_before = backend->calls_to_faulted_key; + backend->fail_at_call = calls_before + 2; + bool threw_aborted = false; + try + { + gc.runRegularRound(); + } + catch (const DB::Exception & e) + { + threw_aborted = (e.code() == DB::ErrorCodes::ABORTED); + if (!threw_aborted) + throw; + } + ASSERT_TRUE(threw_aborted) << "the losing round's own gc/state CAS must fail and propagate ABORTED"; + EXPECT_EQ(backend->calls_to_faulted_key, calls_before + 2) + << "the round must have made exactly the expected two gc/state casPut attempts (renew + commit)"; + + /// GREEN evidence: the losing round's pre-CAS prune must NOT have destroyed `g_parent` — it is still + /// exactly what the (unreplaced, still-adopted) parent seal references. + EXPECT_FALSE(backend->list(parent_gen_prefix, "", 1000).keys.empty()) + << "a losing round must never destroy the generation the still-adopted parent seal references"; + EXPECT_TRUE(backend->head(parent_run_key).exists) + << "the parent seal's exact run object must survive a losing round's pre-CAS prune"; + + /// GC is NOT wedged: gc/state is unchanged (the CAS never committed) and the original blob still + /// resolves cleanly through the surviving parent run — no `CORRUPTED_DATA` from a dangling reference. + EXPECT_EQ(readState(*backend, *store).snap_generation, g_parent); + EXPECT_TRUE(blobExists(*backend, layout, DB::UInt128(1))); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1); + + /// A subsequent round (fault already disarmed) must succeed normally, AND must still reclaim the + /// losing round's own abandoned attempt debris — a generation referenced by NEITHER the parent nor + /// the new proposed seal — proving the fix does not turn pruning off altogether. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const uint64_t g_after = readState(*backend, *store).snap_generation; + ASSERT_GT(g_after, g_parent); + for (uint64_t g = g_parent + 1; g < g_after; ++g) + EXPECT_TRUE(backend->list(layout.gcGenPrefix(g), "", 1000).keys.empty()) + << "generation " << g << " (the losing round's own abandoned attempt debris, referenced by " + "neither the parent nor the new proposed seal) must still be reclaimed on a successful " + "round — the fix must not disable pruning"; + + EXPECT_TRUE(blobExists(*backend, layout, DB::UInt128(2))); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); +} + +/// keep == 0 is the forensics "keep ALL" mode: NO generation is pruned, snap_pruned_through stays 0. +TEST(CASGCSnapRetention, KeepZeroPrunesNothing) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_snapshot_generations_to_keep = 0}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + for (int i = 0; i < 6; ++i) + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const GcState st = readState(*backend, *store); + EXPECT_EQ(st.snap_pruned_through, 0u) << "keep==0 must prune nothing"; + + /// Every seal from generation 1 up to the current one remains. Each generation was sealed under the + /// attempt of the round that produced it (attempt == that round's lease.seq, which bumps every round), + /// so a historical generation's seal lives under an earlier attempt than the final snap_attempt — scan + /// all attempts up to snap_attempt and require the seal to survive under one of them. + for (uint64_t g = 1; g <= st.snap_generation; ++g) + { + bool seal_present = false; + for (uint64_t a = 0; a <= st.snap_attempt && !seal_present; ++a) + seal_present = backend->head(store->layout().foldSealKey(g, a)).exists; + EXPECT_TRUE(seal_present) << "keep==0: seal of generation " << g << " must remain"; + } +} + +TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) +{ + std::shared_ptr backend; + PoolConfig config; + config.pool_prefix = "p"; + /// The GC runner owns a different mount from the synthetic `test` watermark below. This keeps the + /// cursor-sweep assertions in the parent process without replacing its live keeper incarnation. + config.server_root_id = "gc-runner"; + config.manifest_sweep_list_budget_keys = 1; + config.manifest_sweep_delete_budget_keys = 1; + /// This test drives MANY consecutive rounds expecting each to sweep + persist the cursor; force + /// fold-every-round (Phase-4 Lever A would otherwise defer once the pool quiesces). + config.gc_fold_max_defer_rounds = 0; + auto store = openTestPoolWithConfig(backend, config); + + const RootNamespace ns{"test/aa@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r1 = ref(5, 0xCA01); + const ManifestRef r2 = ref(5, 0xCA02); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + setWatermarkMinActive(*backend, store->layout(), "test", r1.writer_epoch, /*min_active*/6); + + /// The §6 deletion premise is a second precondition on every sweep deletion: a manifest of an + /// epoch-`E` build is deletable only once the namespace's sealed fold cursor sits in an epoch + /// STRICTLY above `E`. The debris above is epoch 1, so the namespace's own ref log has to cross + /// into epoch 2 and the ROUND has to fold that crossing. + /// + /// The crossing is written record by record and folded by the round's own arithmetic intake, which + /// makes this the composition of the whole chain in one test: a real `EpochSeal` is minted at + /// `{1,2}`, `RefTableState::apply` consumes it as INV-2's chain link when the epoch-2 record names + /// it in `prev_epoch_seal`, the walk CROSSES on that back-chain, the round seals a cursor in epoch + /// 2, and the premise then admits a deletion for the crossed epoch. Nothing here is seeded: an + /// injected cursor would prove only that the premise reads a number, not that the number can be + /// produced. + /// + /// The live publications use build sequences ABOVE the watermark's `min_active`, so the only + /// sweep-ELIGIBLE manifests in the namespace remain the two debris bodies -- the premise, not the + /// watermark, is what this test varies. + publishAt(*backend, store->layout(), ns, RefTxnId{1, 1}, "tbl", /*build_sequence=*/7, + DB::UInt128(0xB10B1), /*birth=*/true); + writeRecoverableCkptForRawFixture(*backend, store->layout(), ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + EXPECT_TRUE(manifestExists(*backend, store->layout(), ManifestId{ns, r1})) + << "a cursor still INSIDE epoch 1 proves nothing about epoch 1's closing seal: the premise retains"; + EXPECT_TRUE(manifestExists(*backend, store->layout(), ManifestId{ns, r2})); + + /// The round above already persisted a mid-circuit cursor while deleting nothing, which is the + /// cursor half of this test's subject: the sweep examined a key, retained it, and durably recorded + /// where it got to. + EXPECT_FALSE(readState(*backend, *store).manifest_sweep_cursor.empty()) + << "the sweep persisted the cursor it examined to, even having deleted nothing"; + + /// Close epoch 1 and open epoch 2 over the seal it consumed. + writeSealAt(*backend, store->layout(), ns, RefTxnId{1, 2}); + publishAt(*backend, store->layout(), ns, RefTxnId{2, 1}, "tbl2", /*build_sequence=*/7, + DB::UInt128(0xB10B2), /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + const std::optional life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + ASSERT_TRUE(life.has_value()); + const String ckpt_key = store->layout().refCkptKey(*life); + const auto old_ckpt = backend->get(ckpt_key); + ASSERT_TRUE(old_ckpt.has_value()); + ASSERT_EQ(backend->putOverwrite(ckpt_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }), old_ckpt->token).outcome, PutOutcome::Done); + + /// The list budget is one key per round, so reclaiming both debris bodies takes a circuit. + for (int round = 0; round < 12; ++round) + { + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease) << "round " << round; + if (!manifestExists(*backend, store->layout(), ManifestId{ns, r1}) + && !manifestExists(*backend, store->layout(), ManifestId{ns, r2})) + break; + } + EXPECT_FALSE(manifestExists(*backend, store->layout(), ManifestId{ns, r1})); + EXPECT_FALSE(manifestExists(*backend, store->layout(), ManifestId{ns, r2})); + + /// What ADMITTED those deletions, stated rather than inferred: the round folded the crossing itself + /// and sealed a cursor in the epoch above the debris. Without this the two expectations above would + /// still pass if the premise ever stopped consulting the cursor at all. + { + const GcState st = readState(*backend, *store); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const auto it = seal.ref_lives.find(catalogLifeIdForTest(*backend, store->layout(), ns)); + ASSERT_NE(it, seal.ref_lives.end()) << "the round must have sealed a coverage row"; + EXPECT_FALSE(it->second.coverage.hold.has_value()) << "a held namespace can never reach the premise"; + EXPECT_EQ(it->second.coverage.last_folded_ref_id, (RefTxnId{2, 1})) + << "the cursor must sit in the epoch ABOVE the debris, reached by folding the seal at {1,2}"; + } + + /// Replacing a live Pool's own mount with a synthetic foreign watermark must make release fail + /// CLOSED. This was an `EXPECT_DEATH` pinning a `LOGICAL_ERROR` abort; the abort was the defect + /// (it fires from `~Pool`, defeating `finishTeardown`'s own catch by aborting at exception + /// construction, and it fires in ASan builds on any deposed writer's shutdown). What it was really + /// protecting is asserted directly now: the runtime never had a failed renewal, so it still + /// believed it owned the mount, which makes this the exclusivity-violation arm — refuse, leave the + /// occupant byte-for-byte untouched, and SURVIVE the teardown. + std::shared_ptr foreign_backend; + PoolConfig foreign_config = config; + foreign_config.server_root_id = "test"; + auto invalid_store = openTestPoolWithConfig(foreign_backend, std::move(foreign_config)); + const String foreign_mount_key = invalid_store->layout().mountKey("test"); + setWatermarkMinActive(*foreign_backend, invalid_store->layout(), "test", r1.writer_epoch, /*min_active*/6); + const auto occupant_before = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(occupant_before.has_value()); + const uint64_t violations_before + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + + invalid_store.reset(); /// must not abort, must not terminate + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + violations_before + 1) + << "a runtime that never observed a deposition must report the foreign occupant as a broken " + "single-writer guarantee"; + const auto occupant_after = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(occupant_after.has_value()) << "the release must never delete another incarnation's lease"; + EXPECT_EQ(occupant_after->bytes, occupant_before->bytes) + << "the release must leave the slot byte-for-byte untouched, never stamp our farewell over it"; +} + +/// Source-edge idempotency: re-folding the same blob activation does not double-count. +/// A blob activated twice from the SAME source edge (same ManifestId + path) has in-degree 1, not 2. +TEST(CASGCRound, FoldManifestEdgesEmitsOnePlusEdgePerBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) + << "a single published manifest must contribute exactly one source edge per blob"; + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the blob must still exist (in-degree > 0)"; +} + +/// Re-fold of a removal is idempotent: the fold barrier + source-edge set model ensure that +/// folding the same removal twice (the H1b scenario) does NOT drive the in-degree below zero. +TEST(CASGCRound, ReFoldOfRemovalIsIdempotent) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// Drop the ref and run to fixpoint. The blob should be reclaimed. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + EXPECT_NO_THROW(driveToFixpoint(*backend, store, gc)) + << "re-fold of a removal must be idempotent (source-edge set, never underflows)"; + + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the blob must be reclaimed after the only reference is dropped"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +/// Two distinct manifests referencing the same blob contribute TWO independent source edges. +/// Dropping one manifest leaves the other's edge intact (in-degree stays 1, blob is spared). +TEST(CASGCRound, TwoManifestsTwoSourceEdgesDropOneSpares) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r1 = ref(1, 0xAA); + const ManifestRef r2 = ref(2, 0xBB); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl1", std::nullopt, r1); + publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); + + Gc gc(store, kGc); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 2) + << "two distinct manifests referencing the same blob must each contribute one source edge"; + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))); + + /// Drop one of the two references; the other still pins the blob. + dropRefTransition(*backend, store->layout(), ns, "tbl1", r1); + driveToFixpoint(*backend, store, gc); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) + << "after dropping one of two references the in-degree must be 1"; + EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the blob must survive — the second reference still pins it"; +} diff --git a/src/Disks/tests/gtest_cas_gc_round_defer.cpp b/src/Disks/tests/gtest_cas_gc_round_defer.cpp new file mode 100644 index 000000000000..a33816a8bac2 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_round_defer.cpp @@ -0,0 +1,640 @@ +#include + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace ProfileEvents +{ +extern const Event CASGCRefWalkPlansBuilt; +} + +namespace +{ +const UInt128 kGc = UInt128(0xAB); +} + +TEST(CASGCRoundDefer, PredicateTruthTable) +{ + /// threshold=1 (default): defer ONLY when zero shards changed AND no graduation due AND within bound. + EXPECT_TRUE (shouldDeferRound(/*changed*/0, /*grad_due*/false, /*since*/0, /*threshold*/1, /*max*/8)); + EXPECT_FALSE(shouldDeferRound(1, false, 0, 1, 8)); // a shard changed => fold + EXPECT_FALSE(shouldDeferRound(0, true, 0, 1, 8)); // graduation due => force fold + EXPECT_FALSE(shouldDeferRound(0, false, 8, 1, 8)); // defer bound reached => force fold + + /// threshold=3 (batching): defer while accumulated changed shards < threshold, no grad, within bound. + EXPECT_TRUE (shouldDeferRound(2, false, 0, 3, 8)); + EXPECT_FALSE(shouldDeferRound(3, false, 0, 3, 8)); // reached threshold => fold + EXPECT_FALSE(shouldDeferRound(2, true, 0, 3, 8)); // graduation due => force fold regardless of size + EXPECT_FALSE(shouldDeferRound(2, false, 8, 3, 8)); // bound reached => force fold +} + +/// graduationDue (retired-in-snapshot T4): read ZERO-I/O from the adopted seal's condemned_summary. An +/// entry whose oldest non-pending condemn round crosses current_round forces it true; a delete_pending +/// entry forces it true regardless of the round; otherwise false. +TEST(CASGCRoundDefer, GraduationDueDetectsDuePendingAndRoundCrossing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + /// Adopt a seal whose shard-0 summary holds one condemned-but-not-yet-graduated entry (round 2). + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, + {{0, CondemnedSummary{.condemned_total = 1, .pending_total = 0, + .oldest_nonpending_condemn_round = 2}}}); + + Gc gc(store, kGc); + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + + EXPECT_FALSE(gc.graduationDueForTest(state, /*current_round=*/2)) + << "oldest non-pending condemn round (2) is not < current_round (2); not yet due to graduate"; + EXPECT_TRUE(gc.graduationDueForTest(state, /*current_round=*/3)) + << "oldest non-pending condemn round (2) < current_round (3) => due to graduate"; + + /// Re-adopt a seal whose summary entry is delete_pending: due regardless of the round. + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, + {{0, CondemnedSummary{.condemned_total = 1, .pending_total = 1, + .oldest_nonpending_condemn_round = std::numeric_limits::max()}}}); + const GcState state_pending = decodeGcState(backend->get(layout.gcStateKey())->bytes); + + EXPECT_TRUE(gc.graduationDueForTest(state_pending, /*current_round=*/0)) + << "a delete_pending entry must force graduationDue true regardless of current_round"; +} + +/// graduationDue fail-closed: when the adopted seal OBJECT is deleted out from under gc/state, the signal +/// must be TRUE (forces the fold so the round's own fail-closed path surfaces the corrupt bookkeeping), +/// never a silent defer. +TEST(CASGCRoundDefer, GraduationDueFailsClosedWhenSealMissing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, + {{0, CondemnedSummary{}}}); + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + + /// Delete the adopted seal object (corrupt destructive bookkeeping). + const String seal_key = layout.foldSealKey(state.snap_generation, state.snap_attempt); + const HeadResult h = backend->head(seal_key); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->deleteExact(seal_key, h.token).kind, DeleteOutcome::Kind::Deleted); + + Gc gc(store, kGc); + EXPECT_TRUE(gc.graduationDueForTest(state, /*current_round=*/5)) + << "a missing adopted seal must fail-closed to a forced fold"; +} + +/// graduationDue is FALSE on a TOTAL all-zero summary: nothing condemned in any shard => nothing due. +TEST(CASGCRoundDefer, GraduationDueFalseOnAllZeroSummary) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = 2}); + const Layout & layout = store->layout(); + + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/2, + {{0, CondemnedSummary{}}, {1, CondemnedSummary{}}}); + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + + Gc gc(store, kGc); + EXPECT_FALSE(gc.graduationDueForTest(state, /*current_round=*/9)) + << "an all-zero total summary means nothing is due to graduate"; + + /// Fail-closed if the summary is NOT total over gc_shards (shard 1 missing). + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/2, + {{0, CondemnedSummary{}}}); + const GcState partial = decodeGcState(backend->get(layout.gcStateKey())->bytes); + EXPECT_TRUE(gc.graduationDueForTest(partial, /*current_round=*/9)) + << "a summary not total over gc_shards is corrupt => fail-closed force-fold"; +} + +/// `listRefPrefix`'s `changed_shards`: with the fold seal covering shard s at its current token, a quiescent pool reports +/// 0; after one publish to a ref in shard s, it reports 1. +TEST(CASGCRoundDefer, ChangedShardCountIsZeroWhenQuiescent) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + + writeBlobBody(*backend, layout, UInt128(1)); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); /// fold; the round's own trim then rewrites the + /// shard (compacting the just-folded event), so + /// its sealed token is the PRE-trim snapshot. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); /// a second, work-free round: nothing left to + /// trim, so THIS round's fold seal finally + /// captures the shard's actual current token. + + const GcState quiescent_state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + EXPECT_EQ(gc.listRefPrefixForTest(quiescent_state).changed_shards, 0u) + << "a quiescent shard (listed token == sealed token) must not count as changed"; + + /// Publish a second ref into the SAME shard: its LISTED token now differs from what + /// `quiescent_state`'s adopted fold seal recorded. + const ManifestRef r2{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 0xBB}; + writeBlobBody(*backend, layout, UInt128(2)); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", UInt128(2))}); + publishCommittedTransition(*backend, layout, ns, "tbl2", std::nullopt, r2); + + EXPECT_EQ(gc.listRefPrefixForTest(quiescent_state).changed_shards, 1u) + << "one shard whose token advanced since the sealed generation must count as changed"; +} + +/// Mutation caught: widening the hot LIST from `cas/ns/stream/` to `cas/ns/` would offer `_ckpt` and +/// `_files` state objects to the fold. The backend-observed result set must contain both immutable +/// stream kinds and neither state kind. +TEST(CASGCRoundDefer, HotEnumerationOffersLogsAndSnapshotsButNeverCheckpointOrFiles) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"name-must-not-appear"}, UInt128{0x123}); + const RefTxnId id{1, 1}; + const String log_key = layout.refLogKey(life, id); + const String snap_key = layout.refSnapshotKey(life, id); + const String ckpt_key = layout.refCkptKey(life); + const String file_key = layout.namespaceFileKey(life, "f"); + ASSERT_EQ(backend->putIfAbsent(log_key, "log").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(snap_key, "snap").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(ckpt_key, "ckpt").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(file_key, "file").outcome, PutOutcome::Done); + backend->resetCounts(); + + Gc gc(store, kGc); + const RefScanSummary scan = gc.listRefPrefixForTest(GcState{}); + const std::set offered(scan.keys.begin(), scan.keys.end()); + EXPECT_EQ(offered, (std::set{log_key, snap_key})); + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u); + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 0u); + EXPECT_EQ(backend->listCount(layout.namespaceStateRootPrefix()), 0u); +} + +/// The authoritative cut follows the completed hot LIST. A listed life absent from that later cut is +/// inert dead-life debris: it is not admitted and does not defer the round or read the body. +TEST(CASGCRoundDefer, ListedLifeAbsentFromThePostListCatalogCutIsInertDebris) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const NamespaceLifeId unknown = NamespaceLifeId::fromCatalogEntry(RootNamespace{"cannot-authorize"}, UInt128{0x456}); + const String log_key = layout.refLogKey(unknown, RefTxnId{1, 1}); + ASSERT_EQ(backend->putIfAbsent(log_key, "not-read-on-defer").outcome, PutOutcome::Done); + backend->resetCounts(); + + Gc gc(store, kGc); + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + EXPECT_FALSE(report.deferred); + EXPECT_EQ(backend->getCount(log_key), 0u) + << "inert means the body is never read: the life is absent from the authoritative cut, so no " + "admission and no fold intake can touch it"; + /// The debris IS reclaimed in this round, and that is the janitor's designed job, not the fold's: + /// a life id absent from the catalog cut is a dead life, and the namespace janitor deletes its + /// objects by exact token behind the same fence. A round over a proved-empty catalog completes its + /// frontier, so nothing suppresses that reclaim any more -- the object is dropped without ever + /// being read or admitted, which is exactly what "inert debris" means here. + EXPECT_EQ(backend->deleteCount(log_key), 1u); +} + +/// The post-LIST cut classifies every immutable stream kind, not only logs. A snapshot belonging to a +/// life absent from that later cut is inert debris and its body is not read. +TEST(CASGCRoundDefer, SnapshotLifeAbsentFromThePostListCatalogCutIsInertDebris) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const NamespaceLifeId unknown = NamespaceLifeId::fromCatalogEntry(RootNamespace{"cannot-authorize"}, UInt128{0x457}); + const String snapshot_key = layout.refSnapshotKey(unknown, RefTxnId{1, 1}); + ASSERT_EQ(backend->putIfAbsent(snapshot_key, "not-read-on-defer").outcome, PutOutcome::Done); + backend->resetCounts(); + + Gc gc(store, kGc); + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + EXPECT_FALSE(report.deferred); + EXPECT_EQ(backend->getCount(snapshot_key), 0u) + << "inert means the body is never read, whatever immutable stream kind it is"; + /// As for the log above: the dead life's snapshot is reclaimed by the janitor by exact token, + /// never read and never admitted. + EXPECT_EQ(backend->deleteCount(snapshot_key), 1u); +} + +/// ---- Task 4: the DEFER short-circuit wired into runRegularRound ---- + +/// Idle round re-adopts: after a settled round, a subsequent round with zero changed shards and no +/// graduation due sets report.deferred=true and performs dramatically less generation-run I/O than a +/// real fold round (no `blob_target` run object touched at all -- the fold never runs). Snap +/// generation/attempt are untouched (the snapshot is not rebuilt). +/// +/// SETTLING NOTE: immutable `_log` objects are never trimmed in place (unlike the legacy mutable shard +/// journal, whose fold-then-trim token rewrite forced a second settling round), so the pool quiesces the +/// round AFTER the folding round -- the very next round defers. +TEST(CASGCRoundDefer, IdleRoundDefersAndReadsNoGeneration) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + writeBlobBody(*backend, store->layout(), UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + backend->resetCounts(); + const RoundReport fold_rep = gc.runRegularRound(); /// round 1: folds the +1 (no trim-lag, quiesces at once) + ASSERT_FALSE(fold_rep.deferred); + const uint64_t fold_round_gets = backend->getTotal(); + EXPECT_GT(fold_round_gets, 0u) << "sanity: a real fold round performs some GETs"; + + const auto st_before = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + + backend->resetCounts(); + const RoundReport rep = gc.runRegularRound(); /// round 2: genuinely quiesced now => must defer + const uint64_t defer_round_gets = backend->getTotal(); + + EXPECT_TRUE(rep.deferred) << "a settled idle round must re-adopt the sealed generation, not fold"; + /// A deferred round mints no new round (CasGc.cpp:runRegularRound's defer branch), so the honest + /// `report.round` is the round that was ALREADY adopted before this round started -- the same round + /// the preceding fold round committed. Guards against the bug where the defer path returned WITHOUT + /// ever assigning `report.round`, leaving it at its zero-initialized default and making every + /// deferred round print `CA GC round 0` regardless of how far GC had actually progressed. + EXPECT_NE(rep.round, 0u) << "a deferred round must report a truthful, nonzero round number"; + EXPECT_EQ(rep.round, fold_rep.round) + << "a deferred round re-adopts the already-committed round, not a fabricated new one"; + + const auto st_after = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + EXPECT_EQ(st_after.snap_generation, st_before.snap_generation) + << "a deferred round must not mint a new generation (snapshot rebuild elided)"; + EXPECT_EQ(st_after.snap_attempt, st_before.snap_attempt); + + /// SECONDARY (not over-fit to "exactly 0 gets" -- the decision itself pays a bounded retired-list + + /// discovery-LIST cost that may share the same get counter): the deferred round touches NO + /// blob_target run object at all (fold never runs, so foldDeltasIntoGeneration never executes), and + /// its total get volume sits far below a genuine fold round's. + EXPECT_EQ(backend->ioCountForKeysContaining("/blob_target/"), 0u) + << "a deferred round must never GET/getStream/PUT any blob_target run object"; + EXPECT_LT(defer_round_gets, fold_round_gets) + << "a deferred round's read volume must sit far below a real fold round's"; +} + +/// Every ordinary round constructs one complete catalog-authoritative walk plan after the hot LIST, +/// before deciding DEFER. A fold consumes that exact frozen plan; it must not build another one. +TEST(CASGCRoundDefer, FoldAndDeferEachBuildExactlyOneCompletePostListWalkPlan) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/one-walk-plan@cas@"}; + const ManifestRef ref{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}; + writeBlobBody(*backend, layout, UInt128{1}); + writeManifestRaw(*backend, layout, ns, ref, {blobEntryFor("a", UInt128{1})}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, ref); + + Gc gc(store, kGc); + std::vector phases; + gc.setPhaseSink([&](const GcPhaseRecord & phase) { phases.push_back(phase); }); + + backend->resetCounts(); + const uint64_t fold_builds_before + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + ASSERT_FALSE(gc.runRegularRound().deferred); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - fold_builds_before, 1u); + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) + << "the hot walk must enumerate the stream tree exactly once"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the bounded janitor page is a distinct ownership-tree enumeration"; + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 3u) + << "generation zero has no drain read: one cut builds the hot walk plan, one follows the janitor " + "page, and `planManifestCursorPage` takes its own"; + const auto fold_decision = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "defer_decision"; + }); + ASSERT_NE(fold_decision, phases.end()); + EXPECT_EQ(fold_decision->metrics.at("walk_plan_builds"), 1u); + EXPECT_EQ(fold_decision->metrics.at("walk_plan_rows"), 1u); + const auto fold_cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(fold_cleanup, phases.end()); + EXPECT_EQ(fold_cleanup->metrics.at("janitor_pages"), 1u); + + phases.clear(); + backend->resetCounts(); + const uint64_t defer_builds_before + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + ASSERT_TRUE(gc.runRegularRound().deferred); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - defer_builds_before, 1u); + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) + << "a deferred round still builds exactly one complete hot walk plan"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the janitor remains one separately paced ownership-tree page"; + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 3u) + << "one adopted-parent drain cut, one post-hot-LIST cut, and one post-janitor-page cut"; + const auto defer_decision = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "defer_decision"; + }); + ASSERT_NE(defer_decision, phases.end()); + EXPECT_EQ(defer_decision->metrics.at("walk_plan_builds"), 1u); + EXPECT_EQ(defer_decision->metrics.at("walk_plan_rows"), 1u); + EXPECT_EQ(defer_decision->metrics.at("walk_plan_dropped_parent_rows"), 0u); + EXPECT_EQ(defer_decision->metrics.at("walk_plan_dropped_listed_lives"), 0u); + EXPECT_EQ(defer_decision->metrics.at("walk_plan_dropped_tails"), 0u); + const auto defer_cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(defer_cleanup, phases.end()); + EXPECT_EQ(defer_cleanup->metrics.at("janitor_pages"), 1u); +} + +/// A maintenance cursor can be left between pages while the correctness state is already quiescent. +/// The next acquired round may DEFER its fold, but it has no authoritative destructive verdict. It +/// must therefore inspect exactly one janitor page without deleting OR advancing past it; the bounded +/// forced fold then retries the same page under its computed global gate and reclaims the debris. +TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutPublishingSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/1); + const Layout & layout = store->layout(); + const NamespaceLifeId dead_a + = NamespaceLifeId::fromCatalogEntry(RootNamespace{"dead/a"}, UInt128{0xDA}); + const NamespaceLifeId dead_b + = NamespaceLifeId::fromCatalogEntry(RootNamespace{"dead/b"}, UInt128{0xDB}); + const String key_a = layout.refCkptKey(dead_a); + const String key_b = layout.refCkptKey(dead_b); + ASSERT_EQ(backend->putIfAbsent(key_a, "dead-a").outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(key_b, "dead-b").outcome, PutOutcome::Done); + + /// Establish real opaque backend progress rather than fabricating a cursor value. One key remains + /// after this page and the durable cursor must be non-empty. + const NamespaceJanitorResult first_page + = NamespaceJanitor(*backend, layout, 1).runOnePage(false, [] { return true; }); + ASSERT_EQ(first_page.pages, 1u); + ASSERT_EQ(first_page.deleted, 1u); + const GcMaintenanceReadResult partial = readGcMaintenanceState(*backend, layout); + ASSERT_EQ(partial.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(partial.state); + ASSERT_FALSE(partial.state->janitor_cursor.empty()); + ASSERT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 1u); + + /// Give the forced fold a nonempty, fully proved authoritative universe. The R11 floor correctly + /// refuses to open the destructive gate for an empty 0-of-0 universe even in the test-only policy. + const RootNamespace live_namespace{"live/frontier@cas@"}; + fixture::admitLive(*backend, layout, live_namespace); + ASSERT_EQ(backend->putIfAbsent( + layout.refCkptKey(fixture::fixtureLife(live_namespace)), + encodeRefCkpt(RefCkpt{ + .life_epoch = std::optional{1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, + PutOutcome::Done); + + backend->resetCounts(); + std::vector phases; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & phase) { phases.push_back(phase); }); + const uint64_t plans_before + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + + const RoundReport report = gc.runRegularRound(); + + ASSERT_TRUE(report.acquired_lease); + ASSERT_TRUE(report.deferred); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u) + << "DEFER still constructs its one immutable hot walk plan, never a second janitor-derived plan"; + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u); + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the deferred round must inspect exactly one separately paced janitor page"; + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 2u) + << "generation zero pays one hot walk-plan cut and one post-janitor-page cut"; + const auto cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(cleanup, phases.end()); + EXPECT_EQ(cleanup->metrics.at("janitor_pages"), 1u); + EXPECT_GE(cleanup->metrics.at("janitor_keys"), 1u); + EXPECT_EQ(cleanup->metrics.at("janitor_deleted"), 0u); + + const GcMaintenanceReadResult deferred_progress = readGcMaintenanceState(*backend, layout); + ASSERT_EQ(deferred_progress.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(deferred_progress.state); + EXPECT_EQ(deferred_progress.state->janitor_cursor, partial.state->janitor_cursor) + << "a suppressed DEFER page is undecided and must remain selected for the authoritative fold"; + EXPECT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 1u); + + const auto gc_state = backend->get(layout.gcStateKey()); + ASSERT_TRUE(gc_state); + const GcState state = decodeGcState(gc_state->bytes); + EXPECT_EQ(state.snap_generation, 0u); + EXPECT_EQ(state.snap_attempt, 0u); + EXPECT_FALSE(backend->head(layout.foldSealKey(1, 1)).exists) + << "maintenance on DEFER must not publish a fold successor"; + + backend->resetCounts(); + phases.clear(); + const RoundReport folded = gc.runRegularRound({}, true, UniversePolicy::Authoritative); + ASSERT_TRUE(folded.acquired_lease); + ASSERT_FALSE(folded.deferred) + << "gc_fold_max_defer_rounds=1 forces the round immediately following one DEFER to fold"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the authoritative fold must run the janitor exactly once, not once per call site"; + const auto folded_cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(folded_cleanup, phases.end()); + EXPECT_EQ(folded_cleanup->metrics.at("janitor_pages"), 1u); + EXPECT_GE(folded_cleanup->metrics.at("janitor_keys"), 1u); + EXPECT_EQ(folded_cleanup->metrics.at("janitor_deleted"), 1u); + EXPECT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 0u) + << "the fold must retry and delete the exact page that DEFER left undecided"; + const GcMaintenanceReadResult completed = readGcMaintenanceState(*backend, layout); + ASSERT_EQ(completed.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(completed.state); + EXPECT_TRUE(completed.state->janitor_cursor.empty()); +} + +/// The same idle-defer property under a sharded blob-target GC (gc_shards=2): graduationDue's loop +/// over state.retired_refs and `listRefPrefix`'s discovery must both settle to "nothing due" once +/// quiesced, regardless of how many gc-shards partition the retired bookkeeping. +TEST(CASGCRoundDefer, IdleRoundDefersUnderShardedGc) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_shards = 2}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + writeBlobBody(*backend, store->layout(), UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + ASSERT_FALSE(gc.runRegularRound().deferred); /// round 1: folds the publish + + /// Immutable `_log` objects are never trimmed in place, so there is no fold-then-trim token-rewrite + /// lag: the pool quiesces after the folding round, and the very next round defers. + const RoundReport rep = gc.runRegularRound(); /// round 2: quiesced + EXPECT_TRUE(rep.deferred) << "idle pool under gc_shards=2 must defer once settled"; +} + +/// The +1 guard (mirror of the 2026-06-27 leak): a blob condemned + published delete_pending, then +/// re-referenced WHILE it is pending, must NOT be over-deleted -- the due graduation forces a fold +/// (never a defer) that sees the +1 and spares the blob. +TEST(CASGCRoundDefer, DueGraduationForcesFoldAndSparesReReferencedBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const UInt128 blob(1); + const ManifestRef r1{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + writeBlobBody(*backend, store->layout(), blob); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + + runRegularRoundReclaiming(gc); /// folds the +1; blob referenced + store->renewWatermarkOnce(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); /// the -1 condemns it + + runRegularRoundReclaiming(gc); /// the condemning round + store->renewWatermarkOnce(); + + /// Drive rounds until the entry graduates (published delete_pending) -- mirrors + /// CASGCAckFloor.CondemnThenDeleteNextRoundAfterAcks. It is still PRESENT at that pass, and the + /// ack floor is by construction already past its condemn_round (that is what graduated it). + bool saw_pending = false; + for (int i = 0; i < 6 && !saw_pending; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + for (const RetiredEntry & e : currentRetiredSet(*backend, store->layout(), /*shard*/0)) + if (e.ref == DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)} && e.delete_pending) + saw_pending = true; + } + ASSERT_TRUE(saw_pending) << "entry never reached delete_pending"; + ASSERT_FALSE(blobAbsent(*backend, store->layout(), blob)) << "pending: still present this pass"; + + /// While B sits delete_pending, a NEW manifest re-references it -- a genuine +1 racing the + /// already-published pending delete. + const ManifestRef r2{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 0xBB}; + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", blob)}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); + + /// The next pass would otherwise execute B's pending exact-token delete; graduationDue must force + /// a FOLD (never a DEFER) so the +1 is folded in and the blob is spared, not deleted. + const RoundReport rep = runRegularRoundReclaiming(gc); + EXPECT_FALSE(rep.deferred) << "a due graduation must force a fold, never defer"; + EXPECT_FALSE(blobAbsent(*backend, store->layout(), blob)) << "the re-referenced blob must survive"; + + const FsckReport fsck = runFsck(*store, /*detail*/true); + EXPECT_EQ(fsck.dangling, 0u); +} + +/// Companion to the test above: it proves `graduationDue` is the SOLE fold trigger at the assertion +/// round. `DueGraduationForcesFoldAndSparesReReferencedBlob` opens its store at the DEFAULT +/// `gc_fold_threshold` (1), so at its assertion round the +1 re-reference ALSO makes +/// `changed_shards (>= 1) >= fold_threshold (1)` true -- that branch of `shouldDeferRound` would force +/// the very same fold even if `graduationDue` were deleted or hard-wired false. Here `gc_fold_threshold` +/// and `gc_fold_max_defer_rounds` are both set to 1000, so neither the changed-shards branch (one +/// changed shard is nowhere near 1000) nor the liveness-bound branch (this is round 1) can fire -- +/// `graduationDue` is the ONLY thing in `shouldDeferRound` that can force this round's fold, making +/// `EXPECT_FALSE(rep.deferred)` below load-bearing for `graduationDue` specifically. +TEST(CASGCRoundDefer, DueGraduationIsSoleFoldTriggerAtHighThreshold) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_threshold = 1000, .gc_fold_max_defer_rounds = 1000}); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const UInt128 blob(1); + + Gc gc(store, kGc); + /// Warm-up round on the still-empty pool: `gc/state` does not exist yet, so lease acquisition takes + /// the create-fresh path and succeeds immediately (`gc_id` becomes the owner in storage). This + /// matters because the `injectCondemnedSummarySeal` seeding below writes `gc/state` directly, and a fresh `Gc` + /// object's FIRST-EVER `acquireOrRenewLease` call against a PRE-EXISTING lease it has never observed + /// refuses to steal it (two-observation safety against stealing from a live incumbent) -- it would + /// return `acquired_lease=false` and the round would bail out BEFORE the fold-decision code, making + /// `EXPECT_FALSE(rep.deferred)` below vacuously true regardless of `graduationDue`. Running this + /// warm-up round FIRST makes `gc_id` the observed incumbent, so the assertion round's lease RENEWAL + /// (not a steal) succeeds unconditionally and the round actually reaches the decision it's testing. + gc.runRegularRound(); + + writeBlobBody(*backend, layout, blob); + + /// Seed the adopted fold seal's condemned_summary with B already `delete_pending` (pending_total = 1), + /// mirroring `CASGCRoundDefer.GraduationDueDetectsDuePendingAndRoundCrossing`. Retired-in-snapshot + /// (T4): graduationDue reads this summary ZERO-I/O off the adopted seal — a delete_pending entry forces + /// it true regardless of the round. At `gc_fold_threshold = 1000` a real condemn -> graduate pipeline of + /// `runRegularRound` calls is not usable to set this up: every round before graduation would ITSELF + /// defer (nothing due yet, and changed_shards never nears 1000), so the due-pending summary is injected + /// directly instead of driven through real rounds. + injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, + {{0, CondemnedSummary{.condemned_total = 1, .pending_total = 1, + .oldest_nonpending_condemn_round = std::numeric_limits::max()}}}); + + /// The +1: a fresh manifest re-references B while it sits `delete_pending` -- one changed shard, + /// far below the threshold of 1000. + const ManifestRef r{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xBB}; + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", blob)}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + + const RoundReport rep = gc.runRegularRound(); + + /// DISCRIMINATING (load-bearing): with graduationDue intact, the due delete_pending entry forces + /// the fold. If graduationDue were broken/hard-wired false, changed_shards (1) < threshold (1000) + /// and the defer bound (1000) is nowhere near reached, so `shouldDeferRound` would return true and + /// this round would DEFER instead. + EXPECT_FALSE(rep.deferred) << "a due graduation must be the SOLE fold trigger at a high fold threshold"; + EXPECT_FALSE(blobAbsent(*backend, layout, blob)) << "the re-referenced blob must survive the forced fold"; + + const FsckReport fsck = runFsck(*store, /*detail*/true); + EXPECT_EQ(fsck.dangling, 0u); +} + +/// Bounded deferral: with a large fold_threshold and a small standing delta (one shard changed, +/// forever, since deferring never resolves it), at most gc_fold_max_defer_rounds consecutive rounds +/// defer, then one round forces a fold (the liveness bound). +TEST(CASGCRoundDefer, BoundedDeferralForcesFoldWithinWindow) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_threshold = 100, .gc_fold_max_defer_rounds = 3}); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + writeBlobBody(*backend, store->layout(), UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + for (int i = 0; i < 3; ++i) + { + const RoundReport rep = gc.runRegularRound(); + EXPECT_TRUE(rep.deferred) << "round " << (i + 1) << " is within the defer bound"; + } + const RoundReport rep4 = gc.runRegularRound(); + EXPECT_FALSE(rep4.deferred) << "the 4th round hits the defer bound and must force-fold"; +} diff --git a/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp b/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp new file mode 100644 index 000000000000..2efced9934c7 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp @@ -0,0 +1,534 @@ +#include + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; +using DB::Cas::tests::injectRetire; + +namespace +{ + +PoolPtr makePoolWithShards(std::shared_ptr & out_backend, uint64_t gc_shards = 1) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = gc_shards}); +} + +ManifestRef testRef(uint64_t seq) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = 1}; +} + +} + +/// Review I5: `discoverUniverse` is catalog-authoritative (Task 4-C), and this test used to survive +/// the switch from LIST-based discovery unchanged -- `publishCommittedTransition` admits a catalog +/// entry as its own side effect, so LIST-based and catalog-based discovery were indistinguishable to +/// it. Pins the three shapes that actually distinguish the two sources directly: +/// (a) a `Live` catalog entry with ZERO ref objects IS in the universe -- the catalog alone decides; +/// (b) a `Creating` entry is EXCLUDED -- spec §3, no publication can exist yet; +/// (c) a namespace with ref OBJECTS but NO catalog entry is EXCLUDED -- the C1 shape: the catalog is +/// the authority, so its absence is authoritative too, however much debris LIST would still find. +TEST(CASGCShardIncarnation, DiscoveryEqualsPresentShards) +{ + for (const uint64_t gc_shards : {1u, 4u}) + { + std::shared_ptr backend; + auto store = makePoolWithShards(backend, gc_shards); + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + const Layout & layout = store->layout(); + + const RootNamespace ns_live_empty{"srv1/tblLiveEmpty"}; + const RootNamespace ns_creating{"srv1/tblCreating"}; + const RootNamespace ns_uncataloged{"srv1/tblUncataloged"}; + + /// (a) Admitted Live, nothing else ever written under it. + fixture::admitLive(*backend, layout, ns_live_empty); + + /// (b) A genuinely Creating entry, admitted directly (step 1 alone -- never completed to Live). + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns_creating, .state = NsState::Creating, + .incarnation = UInt128(1), .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}}); + + /// (c) Ref objects present, but the catalog was never told (or has since forgotten): write + /// through the real path, which self-admits, then strip the entry back out to simulate "the + /// catalog does not name it" without touching the ref objects it left behind. + writeManifestRaw(*backend, layout, ns_uncataloged, testRef(1), {}); + publishCommittedTransition(*backend, layout, ns_uncataloged, "part_1", std::nullopt, testRef(1), /*shard=*/0); + { + CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + std::erase_if(snap.catalog.entries, [&](const CatalogEntry & e) { return e.ns.string() == ns_uncataloged.string(); }); + const HeadResult h = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, + PutOutcome::Done); + } + + const auto universe = gc.discoverUniverseForTest(); + + /// Stage B (Task 4-C): the universe is catalog-authoritative now, so it is a life per namespace + /// (there are no numeric shards to destructure -- see `NamespaceLifeId`), never a + /// `(namespace, shard)` pair. + bool found_live_empty = false; + for (const NamespaceLifeId & life : universe) + { + if (life.ns.string() == ns_live_empty.string()) + found_live_empty = true; + EXPECT_NE(life.ns.string(), ns_creating.string()) << "a Creating entry must never be discovered"; + EXPECT_NE(life.ns.string(), ns_uncataloged.string()) + << "ref objects with no catalog entry must not be discovered, however much debris LIST would find"; + } + EXPECT_TRUE(found_live_empty) << "a Live catalog entry with zero ref objects must still be discovered"; + + /// Confirm (b) really is still Creating (not merely absent from a differently-shaped universe). + const CasRefCatalog::Snapshot final_snap = CasRefCatalog::read(*backend, layout); + const auto creating_it = std::find_if(final_snap.catalog.entries.begin(), final_snap.catalog.entries.end(), + [&](const CatalogEntry & e) { return e.ns.string() == ns_creating.string(); }); + ASSERT_NE(creating_it, final_snap.catalog.entries.end()); + EXPECT_EQ(creating_it->state, NsState::Creating); + } +} + +/// Catalog ambiguity stops destructive GC and REBUILD before either can derive authority from a +/// first row. No attempted delete is allowed on the rejected regular round. +TEST(CASGCShardIncarnation, DuplicateLifeIdStopsDestructiveRoundAndRebuild) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = 1}); + const Layout & layout = store->layout(); + RefCatalog catalog; + catalog.entries = { + CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128{77}}, + CatalogEntry{ + .ns = RootNamespace{"b"}, + .state = NsState::Removing, + .incarnation = UInt128{77}, + .removal_started_round = 1}, + }; + const auto empty_catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(empty_catalog); + ASSERT_EQ(backend->putOverwrite( + layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->token).outcome, PutOutcome::Done); + backend->resetCounts(); + + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + EXPECT_THROW(gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative), DB::Exception); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_THROW(gc.rebuildBaseline(/*force=*/true), DB::Exception); +} + +/// A physical life id carries no reversible logical namespace component. Once the catalog moves a +/// logical name to a new life, the former stream is opaque debris: it cannot redirect GC to that name +/// or contribute an edge to the current-life fold. The separately paced janitor may reclaim its +/// unowned physical objects after that fold. +TEST(CASGCShardIncarnation, DeadLifeStreamIsOpaqueInertDebris) +{ + std::shared_ptr backend; + auto store = makePoolWithShards(backend, /*gc_shards=*/1); + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/tblIncarnationSwap"}; + + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, + .incarnation = UInt128(11), .creator = std::nullopt}); // Live forbids a creator fence + const ManifestRef dead_ref = testRef(1); + writeBlobBody(*backend, layout, UInt128(11)); + writeManifestRaw(*backend, layout, ns, dead_ref, {blobEntryFor("dead", UInt128(11))}); + std::vector dead_ops{namespaceBirthOp()}; + const auto dead_committed_ops = publishCommittedOps("part_dead", dead_ref); + dead_ops.insert(dead_ops.end(), dead_committed_ops.begin(), dead_committed_ops.end()); + appendRefLogSeed(*backend, layout, ns, std::move(dead_ops)); // real but unacknowledged record at incarnation 11 + const NamespaceLifeId dead_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(11)); + + { + CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + const auto it = std::find_if(snap.catalog.entries.begin(), snap.catalog.entries.end(), + [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); + ASSERT_NE(it, snap.catalog.entries.end()); + it->incarnation = UInt128(22); // "recreated" -- same name, different (empty) key space + const HeadResult h = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, + PutOutcome::Done); + } + + const NamespaceLifeId current_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(22)); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(current_life), encodeRefCkpt(RefCkpt{ + .life_epoch = std::optional{1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + const ManifestRef current_ref = testRef(2); + writeBlobBody(*backend, layout, UInt128(22)); + writeManifestRaw(*backend, layout, ns, current_ref, {blobEntryFor("current", UInt128(22))}); + std::vector current_ops{namespaceBirthOp()}; + const auto committed_ops = publishCommittedOps("part_current", current_ref); + current_ops.insert(current_ops.end(), committed_ops.begin(), committed_ops.end()); + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = std::move(current_ops), .prev_epoch_seal = std::nullopt}); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + EXPECT_TRUE(report.anomalies.empty()); + EXPECT_FALSE(report.deferred); + EXPECT_EQ(inDegreeOf(*backend, layout, UInt128(22)), 1); + EXPECT_EQ(inDegreeOf(*backend, layout, UInt128(11)), 0) + << "the unmatched old life must not contribute its unacknowledged edge to the current-life fold"; +} + +/// Checkpoints live in the state tree and are read by exact key from the catalog cut. They are never +/// discovered through the hot stream LIST, so hiding one from LIST must not affect the round. +TEST(CASGCShardIncarnation, CurrentLifeCheckpointIsReadByExactKeyOutsideHotList) +{ + auto backend = std::make_shared>(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", .gc_shards = 1, + .gc_fold_max_defer_rounds = 0}); + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/tblOrdinaryRebirth"}; + + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, + .incarnation = UInt128(11), .creator = std::nullopt}); + + CatalogEntry after_rebirth{.ns = ns, .state = NsState::Live, .incarnation = UInt128(22), .creator = std::nullopt}; + { + CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + const auto it = std::find_if(snap.catalog.entries.begin(), snap.catalog.entries.end(), + [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); + ASSERT_NE(it, snap.catalog.entries.end()); + *it = after_rebirth; // "recreated" -- same name, new (current) incarnation 22 + const HeadResult h = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, + PutOutcome::Done); + } + + /// The successor's own genesis `_ckpt`, published for the current physical life. Hiding it from + /// LIST must be irrelevant because the walk obtains state only through exact GETs. + const NamespaceLifeId current_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(22)); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(current_life), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + backend->hide(layout.refCkptKey(current_life)); + backend->resetCounts(); + std::vector phases; + gc.setPhaseSink([&](const GcPhaseRecord & phase) { phases.push_back(phase); }); + + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + EXPECT_FALSE(report.deferred) << "the forced catalog-only fold must reach checkpoint intake"; + EXPECT_GT(backend->getCount(layout.refCkptKey(current_life)), 0u) + << "the catalog-derived current life must drive an exact checkpoint GET"; + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) + << "the round must build exactly one hot stream plan"; + EXPECT_EQ(backend->listCount(layout.namespaceStateRootPrefix()), 0u) + << "checkpoint state must never receive its own hot LIST"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the only broader LIST is the separately paced janitor page"; + EXPECT_EQ(backend->holesServed(), 1u) + << "the hidden checkpoint is omitted only from the janitor's broad page, never from the hot stream LIST"; + EXPECT_TRUE(backend->head(layout.refCkptKey(current_life)).exists) + << "the post-page catalog cut retains the current life even when LIST omitted its checkpoint"; + const auto cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(cleanup, phases.end()); + EXPECT_EQ(cleanup->metrics.at("janitor_pages"), 1u); + EXPECT_EQ(cleanup->metrics.at("janitor_deleted"), 0u); + bool saw_anomaly_for_ns = false; + for (const RoundAnomaly & a : report.anomalies) + if (a.ns.string() == ns.string()) + saw_anomaly_for_ns = true; + EXPECT_FALSE(saw_anomaly_for_ns) + << "the current life's `_ckpt` is real and readable by its exact catalog-derived key"; +} + +/// A stream life absent from the immutable catalog cut cannot be attributed to any logical namespace. +/// It remains inert debris rather than producing a made-up name or a round anomaly. +TEST(CASGCShardIncarnation, UncatalogedStreamLifeDefersWithoutInventingNamespace) +{ + std::shared_ptr backend; + auto store = makePoolWithShards(backend, /*gc_shards=*/1); + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/tblForgotten"}; + + writeManifestRaw(*backend, layout, ns, testRef(1), {}); + publishCommittedTransition(*backend, layout, ns, "part_1", std::nullopt, testRef(1), /*shard=*/0); + const NamespaceLifeId forgotten_life = store->namespaceLife(ns); + + { + CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + std::erase_if(snap.catalog.entries, [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); + const HeadResult h = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, + PutOutcome::Done); + } + + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + EXPECT_TRUE(report.anomalies.empty()); + EXPECT_FALSE(backend->list(layout.namespaceStreamPrefix(forgotten_life), "", 100).keys.empty()); +} + +/// State-tree objects are point-addressed only. A stalled creator's checkpoint and an unowned opaque +/// checkpoint are both outside the hot stream scan and cannot manufacture logical namespace anomalies. +TEST(CASGCShardIncarnation, StateCheckpointsOutsideCatalogAreInertToHotWalk) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = 1}); + Gc gc(store, hexToU128("0000000000000000000000000000000a")); + const Layout & layout = store->layout(); + const RootNamespace creating_ns{"srv1/tblStalledBirth"}; + const RootNamespace unrelated_gone_ns{"srv1/tblGenuinelyGone"}; + + /// Step 1 of createNamespace: insert the Creating entry with a live creator fence. + CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = creating_ns, .state = NsState::Creating, + .incarnation = UInt128(33), + .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}}); + /// Step 2, without step 3: publish the genesis `_ckpt` directly, at the SAME incarnation the + /// Creating entry names -- exactly what `completeCreation` durably leaves behind if the creator + /// crashes between its own steps 2 and 3. + const NamespaceLifeId creating_life = NamespaceLifeId::fromCatalogEntry(creating_ns, UInt128(33)); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(creating_life), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + /// Opaque state debris with no corresponding catalog entry. + const NamespaceLifeId gone_life = NamespaceLifeId::fromCatalogEntry(unrelated_gone_ns, UInt128(44)); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(gone_life), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + /// Add one fully current stream so the round performs a fold rather than stopping at an empty + /// walk. Catalog and checkpoint admission keep this traffic out of the janitor's dead-life set, + /// isolating the one deliberately unowned checkpoint below. + const RootNamespace ordinary_ns{"srv1/tblOrdinaryTraffic"}; + fixture::admitLive(*backend, layout, ordinary_ns); + const NamespaceLifeId ordinary_life = fixture::fixtureLife(ordinary_ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(ordinary_life), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + appendRefLogSeed(*backend, layout, ordinary_ns, {}); + + backend->resetCounts(); + std::vector phases; + gc.setPhaseSink([&](const GcPhaseRecord & phase) { phases.push_back(phase); }); + + const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); + bool saw_stalled_birth_anomaly = false; + bool saw_genuinely_gone_anomaly = false; + for (const RoundAnomaly & a : report.anomalies) + { + if (a.ns.string() == creating_ns.string()) + saw_stalled_birth_anomaly = true; + if (a.ns.string() == unrelated_gone_ns.string()) + saw_genuinely_gone_anomaly = true; + } + EXPECT_FALSE(saw_stalled_birth_anomaly); + EXPECT_FALSE(saw_genuinely_gone_anomaly); + EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) + << "all hot intake must consume one immutable stream listing"; + EXPECT_EQ(backend->listCount(layout.namespaceStateRootPrefix()), 0u) + << "state checkpoints are never a hot discovery source"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) + << "the only ownership-tree listing belongs to the independently paced janitor"; + EXPECT_GT(backend->getCount(layout.refCkptKey(ordinary_life)), 0u) + << "the cataloged Live life is read by its exact checkpoint key"; + EXPECT_EQ(backend->getCount(layout.refCkptKey(creating_life)), 0u) + << "Creating is retained by the janitor cut but excluded from hot checkpoint intake"; + EXPECT_EQ(backend->getCount(layout.refCkptKey(gone_life)), 0u) + << "uncataloged state debris is classified by the janitor page, never exact-read by the hot walk"; + EXPECT_TRUE(backend->head(layout.refCkptKey(creating_life)).exists); + EXPECT_TRUE(backend->head(layout.refCkptKey(ordinary_life)).exists); + EXPECT_FALSE(backend->head(layout.refCkptKey(gone_life)).exists) + << "catalog absence is inert to the hot walk but authorizes the later janitor exact-token delete"; + const auto cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "namespace_cleanup"; + }); + ASSERT_NE(cleanup, phases.end()); + EXPECT_EQ(cleanup->metrics.at("janitor_pages"), 1u); + EXPECT_EQ(cleanup->metrics.at("janitor_deleted"), 1u); +} + +/// `listNamespaces` projects the authoritative catalog; physical streams never contribute names. +TEST(CASGCShardIncarnation, ListNamespacesFromCatalog) +{ + for (const uint64_t gc_shards : {1u, 4u}) + { + std::shared_ptr backend; + auto store = makePoolWithShards(backend, gc_shards); + + const RootNamespace ns_a{"srv1/tblA"}; + + EXPECT_TRUE(store->listNamespaces("").namespaces.empty()); + + /// The real writer path admits the catalog row for namespace A. + writeManifestRaw(*backend, store->layout(), ns_a, testRef(1), {}); + publishCommittedTransition(*backend, store->layout(), ns_a, "part_1", std::nullopt, testRef(1), /*shard=*/0); + + const auto nss = store->listNamespaces("").namespaces; + ASSERT_EQ(nss.size(), 1u); + EXPECT_EQ(nss[0], "srv1/tblA"); + + /// Prefix filter: no match. + EXPECT_TRUE(store->listNamespaces("srv2/").namespaces.empty()); + /// Prefix filter: match. + const auto filtered = store->listNamespaces("srv1/").namespaces; + ASSERT_EQ(filtered.size(), 1u); + EXPECT_EQ(filtered[0], "srv1/tblA"); + } +} + +/// Task 5: THM-NO-RETURN create-race. A NEWBORN ref-shard is born fenced to the current GC round +/// (self-floor: `fence_round` self-floors to `currentGcRound()` on the create-if-absent branch). +/// +/// Scenario (registry-free create-race): +/// 1. Open a Pool (gc/state absent). +/// 2. Write blob b1's body directly to the backend (present, not yet condemned). +/// 3. Inject gc/state at round 1 with b1 condemned (its current token in the retired set). +/// b1's body is still PRESENT — this simulates GC having fenced+retired b1 but not yet +/// deleted it (the retired-but-body-present window). +/// 4. A writer for NEWBORN ns B calls `precommitAdd` → reads `currentGcRound() = 1` → +/// the NEWBORN shard is born with `fence_round = 1` (self-floor). +/// 5. `promote` binds the condemned-but-present tokenless leaf AS IS (spec +/// 2026-07-09-cas-writer-gc-simplification D5: there is no writer-side view refresh at promote +/// any more). This is safe because the precommit closure's edge is journal-durable BEFORE +/// promote returns (EDGE-BEFORE-OBSERVE): the NEXT GC fold sees net in-degree >= 1 for b1 and +/// SPARES the entry, regardless of when it would otherwise graduate — the condemnation is +/// doomed, never the blob. INV-NO-DANGLE holds (dangling=0 in fsck). +/// +/// Both gc_shards=1 and gc_shards>1 are exercised. The self-floor and promote gate are independent +/// of the blob-hash-prefix sharding axis (fence_round lives in the ROOT shard). +TEST(CASGCShardIncarnation, NewbornPrecommitProtectsDedupBlobAgainstConcurrentDrop) +{ + for (const uint64_t gc_shards : {1u, 4u}) + { + std::shared_ptr backend; + auto store = makePoolWithShards(backend, gc_shards); + const RootNamespace ns_b{"srv1/tblB"}; + + /// --- Phase 1: Write b1's body before any GC. --- + /// Mint b1 under the pool streaming-hash id through a complete durable-precommit fixture, then + /// drop that fixture ref so the newborn owner below is the edge whose safety matters. + const String b1_payload = "shared-blob-b1"; + const String b1_hex = streamingHexOf(b1_payload); + const BlobRef b1_ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(b1_hex))}; + { + const RootNamespace seed_ns{"srv1/seed"}; + PartWriteInfo seed_info; + seed_info.intended_ref = seed_ns.string() + "/seed"; + auto seed = store->beginPartWrite(seed_info); + ManifestEntry seed_entry; + seed_entry.path = "data.bin"; + seed_entry.placement = EntryPlacement::Blob; + seed_entry.ref = b1_ref; + seed_entry.blob_size = b1_payload.size(); + const ManifestId seed_manifest = seed->stageManifest({seed_entry}); + seed->precommitAdd(seed_ns, "seed", seed_manifest); + seed->putBlob(b1_ref, BlobSource::fromString(b1_payload)); + seed->promote(seed_ns, "seed", seed->buildId(), seed_manifest); + store->dropRef(seed_ns, "seed"); + store->renewWatermarkOnce(); + } + const String b1_key = store->layout().blobKey(b1_ref); + ASSERT_TRUE(backend->head(b1_key).exists) + << "b1 body must be present after the seed putBlob"; + const Token b1_token = backend->head(b1_key).token; + + /// --- Phase 2: Inject gc/state at round 1 with b1 CONDEMNED (body still present). --- + /// This simulates GC having advanced to round 1 and retired b1 (condemned token recorded + /// in the retired set) but not yet deleted b1's body object. + injectRetire(*backend, store->layout(), /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = b1_ref, + .token = b1_token, .size = static_cast(b1_payload.size())}}); + + /// Sanity: currentGcRound() reads gc/state fresh and returns 1. + ASSERT_EQ(store->currentGcRound(), 1u) + << "currentGcRound() must return the injected round"; + + /// --- Phase 3: Writer for NEWBORN ns B — b1 condemned but body present --- + PartWriteInfo info_b; + info_b.intended_ref = ns_b.string() + "/part_b1"; + auto build_b = store->beginPartWrite(info_b); + + /// Adopt b1 by tokenless evidence (simulating the dedup case: the writer observed b1 + /// present BEFORE the GC round — no HEAD here, just evidence). + ManifestEntry dep_b1; + dep_b1.path = "data.bin"; + dep_b1.placement = EntryPlacement::Blob; + dep_b1.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hexToU128(b1_hex))}; + + dep_b1.blob_size = b1_payload.size(); + build_b->adoptEvidence(dep_b1); + + const ManifestId id_b = build_b->stageManifest({dep_b1}); + + /// precommitAdd: NEWBORN shard does not exist yet. Reads currentGcRound() = 1 → stamps + /// fence_round = 1 (self-floor). An existing shard would keep its old fence_round. + build_b->precommitAdd(ns_b, "part_b1", id_b); + + /// --- Phase 4: promote — the safety assertion (Phase-A contract) --- + /// Spec 2026-07-09-cas-writer-gc-simplification D5: there is NO writer-side view refresh at + /// promote any more — the K3 gate binds the condemned-but-present token AS IS. This is SAFE + /// because the precommit closure's edge has been journal-durable since precommitAdd (BEFORE + /// promote returns), so the NEXT GC fold sees net in-degree >= 1 for b1 and SPARES the entry + /// (EDGE-BEFORE-OBSERVE) regardless of round-paced graduation timing — the condemnation is + /// doomed, never the blob. (No GC round runs in this test at all; the argument is what makes + /// deferring the round safe, not something this test drives to completion.) + /// The former behavior (self-floor-forced refresh → in-closure copy-forward → fresh incarnation) + /// was TLA+-Gate-A-verified redundant; the shard's fence_round stamp itself (THM-NO-RETURN birth + /// floor) remains and is asserted by the sibling shard-incarnation tests. + EXPECT_NO_THROW(build_b->promote(ns_b, "part_b1", build_b->buildId(), id_b)) + << "gc_shards=" << gc_shards << ": promote must commit — the durable edge protects the " + "condemned-but-present tokenless leaf without any refresh or copy-forward"; + EXPECT_TRUE(store->resolveRef(ns_b, "part_b1").has_value()) + << "gc_shards=" << gc_shards << ": the ref must commit"; + /// The condemned token is bound UNCHANGED — no displacement happens (and none is needed). + EXPECT_EQ(backend->head(b1_key).token, b1_token) + << "gc_shards=" << gc_shards << ": no copy-forward under the Phase-A contract — the token " + "stays; the folded edge will spare it at the next fold (no round runs here to delete it)"; + + /// INV-NO-DANGLE: the body is present and no GC round ever runs in this test to fold the + /// precommit/committed edge; a real deployment's next fold would see net in-degree >= 1 and + /// spare the entry. A regression that let the delete pipeline race a live durable edge would + /// produce dangling=1 here. + const FsckReport rep = runFsck(*store, /*detail=*/false); + EXPECT_EQ(rep.dangling, 0u) + << "gc_shards=" << gc_shards << ": INV-NO-DANGLE violated — a committed ref names a " + "missing blob (dangling=" << rep.dangling << ", reachable=" << rep.reachable << ")"; + } +} + +/// The five shard-OBJECT-reclaim tests that used to follow (`DroppedShardObjectIsReclaimed`, +/// `IdleButLiveShardNotReclaimed`, `RecreateAfterReclaimFoldsFromZero`, `ActivatedPrecommitBlocksShardReclaim`, +/// `ReviveRacesReclaimAborts`) were removed with the snapshot+log ref model. They asserted GC reclaims / +/// token-guards a MUTABLE per-namespace ref-shard object at `rootShardKey(ns, shard)`. There is no such +/// mutable object anymore: a namespace's ref state is its immutable `_log`/`_snap` objects, physical +/// reclamation belongs to the perpetual namespace janitor, and ABA safety is structural -- a recreated +/// namespace uses a different opaque life id. The still-meaningful reincarnation case (a terminal old +/// life followed by a new life folds without inheriting the old cursor) is covered by +/// `gtest_cas_ref_gc.cpp`; lifecycle completion itself requires only folded terminal evidence and the +/// exact catalog-row mutation. diff --git a/src/Disks/tests/gtest_cas_gc_shard_plan.cpp b/src/Disks/tests/gtest_cas_gc_shard_plan.cpp new file mode 100644 index 000000000000..c45cd3ff772d --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_shard_plan.cpp @@ -0,0 +1,637 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +TEST(CASGCShardConfig, DefaultIsSingleShard) +{ + PoolConfig cfg; + EXPECT_EQ(cfg.gc_shards, 1u); + EXPECT_EQ(cfg.manifest_sweep_list_budget_keys, 1000u); + EXPECT_EQ(cfg.manifest_sweep_delete_budget_keys, 100u); +} + +TEST(CASGCShardConfig, GcStateRoundTripPreservesShardCount) +{ + GcState s; + s.gc_shards = 4; + s.round = 7; + const GcState d = decodeGcState(encodeGcState(s)); + EXPECT_EQ(d.gc_shards, 4u); + EXPECT_EQ(d.round, 7u); +} + +/// ---- blobShard tests (Phase 4, Task 3) ---- + +TEST(CASGCShardScatter, DeterministicAndStable) +{ + /// A fixed hash — the same bytes every run. blobShard must return the same value twice, + /// must be strictly less than gc_shards=4, and must be 0 when gc_shards=1. + const UInt128 h = hexToU128("0102030405060708090a0b0c0d0e0f10"); + const BlobRef hd{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}; + + const uint64_t s4a = blobShard(hd, 4); + const uint64_t s4b = blobShard(hd, 4); + + EXPECT_EQ(s4a, s4b) << "blobShard must be deterministic"; + EXPECT_LT(s4a, 4u) << "blobShard result must be < gc_shards"; + EXPECT_EQ(blobShard(hd, 1), 0u) << "gc_shards==1 must route every hash to shard 0"; +} + +TEST(CASGCShardScatter, DisjointCoverageOverManyHashes) +{ + /// Over 4096 spread-out hashes with gc_shards=4: every result in [0,4) and every shard + /// gets at least one hash (no dead shard). + constexpr uint64_t kNumHashes = 4096; + constexpr uint64_t kShards = 4; + + std::vector seen(kShards, false); + for (uint64_t i = 0; i < kNumHashes; ++i) + { + /// Spread: use i in the high and low halves to avoid clustering. + const UInt128 h = (static_cast(i * 0x9e3779b97f4a7c15ULL) << 64) + | static_cast(i * 0x6c62272e07bb0142ULL); + const uint64_t s = blobShard(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}, kShards); + ASSERT_LT(s, kShards) << "blobShard out of range at i=" << i; + seen[s] = true; + } + + for (uint64_t s = 0; s < kShards; ++s) + EXPECT_TRUE(seen[s]) << "shard " << s << " received no hashes (dead shard)"; +} + +/// ---- ShardReducer tests (Phase 4, Task 4) ---- + +/// Build two blob hashes that route to DIFFERENT shards under gc_shards=2. +/// Returns {hash_for_shard0, hash_for_shard1}. +static std::pair makeTwoShardHashes() +{ + /// Scan pairs (i, j): find hash_a -> shard 0, hash_b -> shard 1 under gc_shards=2. + /// We construct candidates by setting the high 64 bits and leaving the low 64 bits zero + /// so blobShard = high64 % 2. i=0 => shard 0, i=1 => shard 1. + const UInt128 h0 = static_cast(0ULL) << 64; /// high64=0 => shard 0 + const UInt128 h1 = static_cast(1ULL) << 64; /// high64=1 => shard 1 + return {BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h0)}, + BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h1)}}; +} + +/// `ShardReducer::reduce` merges deltas into the correct per-shard in-degree run. +/// +/// Scenario: scatter (+1 b1, +1 b1, -1 b1, +1 b2) across two shards. +/// - b1 routes to shard 0; net = +2 - 1 = 1; in-degree after reduce = 1. +/// - b2 routes to shard 1; net = +1; in-degree after reduce = 1. +/// - Each reducer touches ONLY its own shard's key space. +TEST(CASGCShardReducer, MergesDeltasToInDegree) +{ + const auto [b1, b2] = makeTwoShardHashes(); + ASSERT_EQ(blobShard(b1, 2), 0u) << "b1 must route to shard 0"; + ASSERT_EQ(blobShard(b2, 2), 1u) << "b2 must route to shard 1"; + + /// Construct source-edge deltas directly (the production fold produces these via + /// `foldManifestEdges`, bucketed by `blobShard`): + /// b1 shard=0: source 1 activates, source 2 activates, source 1 removes => 1 active edge + /// b2 shard=1: source 3 activates => 1 active edge + std::vector> buckets(2); + buckets[0] = { + BlobDelta{.ref = b1, .source_id = UInt128(1), .remove = false}, + BlobDelta{.ref = b1, .source_id = UInt128(2), .remove = false}, + BlobDelta{.ref = b1, .source_id = UInt128(1), .remove = true}, + }; + buckets[1] = { + BlobDelta{.ref = b2, .source_id = UInt128(3), .remove = false}, + }; + + /// Verify bucket net effects (source 2 survives for b1; source 3 survives for b2). + ASSERT_EQ(buckets.size(), 2u); + { + int64_t net_b1 = 0; + for (const auto & d : buckets[0]) + if (d.ref == b1) + net_b1 += d.remove ? -1 : +1; + EXPECT_EQ(net_b1, 1) << "shard-0 bucket net delta for b1 must be +1"; + } + { + int64_t net_b2 = 0; + for (const auto & d : buckets[1]) + if (d.ref == b2) + net_b2 += d.remove ? -1 : +1; + EXPECT_EQ(net_b2, 1) << "shard-1 bucket net delta for b2 must be +1"; + } + + /// Reduce: each reducer merges its shard's deltas into generation 1 (prior = 0 = fresh). + auto backend = std::make_shared(); + const Layout layout("p"); + + ShardReducer r0(0, 2); + ShardReducer r1(1, 2); + + EXPECT_TRUE(r0.owns(b1)) << "r0 must own b1"; + EXPECT_FALSE(r0.owns(b2)) << "r0 must not own b2"; + EXPECT_TRUE(r1.owns(b2)) << "r1 must own b2"; + EXPECT_FALSE(r1.owns(b1)) << "r1 must not own b1"; + + const auto runs0 = r0.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + std::move(buckets[0])); + const auto runs1 = r1.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + std::move(buckets[1])); + + ASSERT_EQ(runs0.size(), 1u) << "shard-0 reduce must produce exactly one RunRef"; + ASSERT_EQ(runs1.size(), 1u) << "shard-1 reduce must produce exactly one RunRef"; + + /// The keys must be distinct (disjoint shard namespaces). + EXPECT_NE(runs0[0].key, runs1[0].key) << "shard-0 and shard-1 run keys must be distinct"; + + /// Read back in-degree from the sealed runs (resolved via each reduce's returned refs). + const int64_t indeg_b1 = inDegreeInRuns(*backend, runs0, b1); + const int64_t indeg_b2 = inDegreeInRuns(*backend, runs1, b2); + EXPECT_EQ(indeg_b1, 1) << "b1 in-degree after reduce must be 1"; + EXPECT_EQ(indeg_b2, 1) << "b2 in-degree after reduce must be 1"; + + /// Cross-shard reads: shard-0's run must not contain b2; shard-1's run must not contain b1. + EXPECT_EQ(inDegreeInRuns(*backend, runs0, b2), 0) + << "shard-0 run must not mention b2"; + EXPECT_EQ(inDegreeInRuns(*backend, runs1, b1), 0) + << "shard-1 run must not mention b1"; +} + +/// `ShardReducer::owns` partitions the blob hash space: for any hash, exactly ONE reducer among +/// {r0, r1} owns it (union == all, intersection == empty). +TEST(CASGCShardReducer, TwoReducersCoverDisjointShards) +{ + constexpr uint64_t kNumHashes = 4096; + constexpr uint64_t kGcShards = 2; + + ShardReducer r0(0, kGcShards); + ShardReducer r1(1, kGcShards); + + for (uint64_t i = 0; i < kNumHashes; ++i) + { + const UInt128 h = (static_cast(i * 0x9e3779b97f4a7c15ULL) << 64) + | static_cast(i * 0x6c62272e07bb0142ULL); + const BlobRef href{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}; + const bool o0 = r0.owns(href); + const bool o1 = r1.owns(href); + + /// Exactly one of the two reducers must own every hash. + ASSERT_TRUE(o0 || o1) + << "hash " << i << " is owned by neither shard (gap in coverage)"; + ASSERT_FALSE(o0 && o1) + << "hash " << i << " is owned by BOTH shards (overlap in coverage)"; + } +} + +/// ---- manifestCleanupShard tests (Phase 4, Task 5) ---- + +/// Two `ManifestId`s with the SAME `ManifestRef` but DIFFERENT namespaces must be unequal (proving +/// qualified identity), and `manifestCleanupShard` must depend on the namespace — not just the ref. +/// +/// Phase 0 `SabotageKeyByRefNotId`: if routing used only the `ManifestRef`, two namespaces sharing +/// the same ref would land on the same worker, merging cleanup work that belongs to distinct objects. +TEST(CASGCShardCleanup, RoutesByQualifiedManifestIdNotRef) +{ + /// Shared ManifestRef: identical across both ManifestIds. + const ManifestRef shared_ref{ + .writer_epoch = 1, + .build_sequence = 7, + .manifest_ordinal = 1, + }; + + const ManifestId id_a{RootNamespace("ns_alpha"), shared_ref}; + const ManifestId id_b{RootNamespace("ns_beta"), shared_ref}; + + /// The two ids are unequal (different namespace => different qualified identity). + EXPECT_NE(id_a, id_b) << "ManifestIds with different namespaces must be unequal"; + + /// Both results must be in range. + constexpr uint64_t kShards = 4; + const uint64_t shard_a = manifestCleanupShard(id_a, kShards); + const uint64_t shard_b = manifestCleanupShard(id_b, kShards); + EXPECT_LT(shard_a, kShards) << "shard for id_a must be < gc_shards"; + EXPECT_LT(shard_b, kShards) << "shard for id_b must be < gc_shards"; + + /// Deterministic: same id always routes to the same shard. + EXPECT_EQ(manifestCleanupShard(id_a, kShards), shard_a) << "manifestCleanupShard must be deterministic"; + EXPECT_EQ(manifestCleanupShard(id_b, kShards), shard_b) << "manifestCleanupShard must be deterministic"; + + /// Single-shard equivalence: gc_shards==1 routes everything to shard 0. + EXPECT_EQ(manifestCleanupShard(id_a, 1), 0u) << "gc_shards==1 must route to shard 0"; + EXPECT_EQ(manifestCleanupShard(id_b, 1), 0u) << "gc_shards==1 must route to shard 0"; + + /// KEY ASSERTION: routing depends on the namespace, not the ref alone. + /// Scan namespace-pair candidates (varying only the namespace string) until we find two that + /// route to different shards under gc_shards=8. This directly demonstrates that + /// `manifestCleanupShard` is NOT a function of `ManifestRef` alone. + bool found_namespace_split = false; + for (uint64_t i = 0; i < 256 && !found_namespace_split; ++i) + { + const ManifestId probe_a{RootNamespace("namespace_probe_" + std::to_string(i)), shared_ref}; + for (uint64_t j = i + 1; j < 256 && !found_namespace_split; ++j) + { + const ManifestId probe_b{RootNamespace("namespace_probe_" + std::to_string(j)), shared_ref}; + if (manifestCleanupShard(probe_a, 8) != manifestCleanupShard(probe_b, 8)) + found_namespace_split = true; + } + } + EXPECT_TRUE(found_namespace_split) + << "could not find two namespace variants of the same ManifestRef that route to different " + "shards — routing is not namespace-sensitive (SabotageKeyByRefNotId hazard)"; +} + +/// Over many `ManifestId`s with `gc_shards=4`: every owner shard is covered, and each id lands in +/// exactly one shard (total, disjoint coverage). +TEST(CASGCShardCleanup, DisjointWorkerCoverage) +{ + constexpr uint64_t kNumIds = 4096; + constexpr uint64_t kShards = 4; + + std::vector seen(kShards, false); + for (uint64_t i = 0; i < kNumIds; ++i) + { + /// Vary both namespace and ManifestRef fields to spread the distribution. + const ManifestId id{ + RootNamespace("ns_" + std::to_string(i % 16)), + ManifestRef{ + .writer_epoch = 1 + i / 16, + .build_sequence = i, + .manifest_ordinal = static_cast(i % kMaxManifestOrdinal + 1), + }, + }; + + const uint64_t s = manifestCleanupShard(id, kShards); + ASSERT_LT(s, kShards) << "manifestCleanupShard out of range at i=" << i; + seen[s] = true; + } + + for (uint64_t s = 0; s < kShards; ++s) + EXPECT_TRUE(seen[s]) << "owner shard " << s << " received no ManifestIds (dead shard)"; +} + +/// The sharded fold (gc_shards > 1) partitions a flat `BlobDelta` stream by `blobShard` and folds +/// each bucket via its own `ShardReducer`, exactly as `Gc::fold` does. This test replicates that +/// partition-and-reduce step over `gc_shards = 2` and asserts each blob's in-degree lands in its +/// owning shard's run and nowhere else. (The full two-replica round is covered by Task 8.) +TEST(CASGCShardCoordinator, ShardedFoldRoutesDeltasToOwningShards) +{ + constexpr uint64_t kGcShards = 2; + const auto [b0, b1] = makeTwoShardHashes(); + ASSERT_EQ(blobShard(b0, kGcShards), 0u); + ASSERT_EQ(blobShard(b1, kGcShards), 1u); + + /// A flat delta stream as produced by `foldManifestEdges`: b0 net +1 (two +1, one -1), b1 net +1. + std::vector deltas{ + BlobDelta{.ref = b0, .source_id = UInt128(1), .remove = false}, + BlobDelta{.ref = b1, .source_id = UInt128(2), .remove = false}, + BlobDelta{.ref = b0, .source_id = UInt128(3), .remove = false}, + BlobDelta{.ref = b0, .source_id = UInt128(1), .remove = true}, + }; + + /// Partition by blobShard — the exact step the sharded fold runs before reducing. + std::vector> buckets(kGcShards); + for (BlobDelta & d : deltas) + buckets[blobShard(d.ref, kGcShards)].push_back(d); + + auto backend = std::make_shared(); + const Layout layout("p"); + + std::vector> shard_runs(kGcShards); + for (uint64_t shard = 0; shard < kGcShards; ++shard) + { + ShardReducer reducer{shard, kGcShards}; + shard_runs[shard] = reducer.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + std::move(buckets[shard])); + } + + EXPECT_EQ(inDegreeInRuns(*backend, shard_runs[0], b0), 1) + << "b0 must fold into shard-0 with in-degree 1"; + EXPECT_EQ(inDegreeInRuns(*backend, shard_runs[1], b1), 1) + << "b1 must fold into shard-1 with in-degree 1"; + EXPECT_EQ(inDegreeInRuns(*backend, shard_runs[1], b0), 0) + << "b0 must NOT appear in shard-1's run"; + EXPECT_EQ(inDegreeInRuns(*backend, shard_runs[0], b1), 0) + << "b1 must NOT appear in shard-0's run"; +} + +/// ---- Phase 4, Task 7: single-shard equivalence ---- +/// +/// Prove that the sharded partition+reduce path (gc_shards=2, all blobs routing to shard 0) produces +/// the SAME per-blob in-degrees as the single-shard (gc_shards=1, Phase 1d) fold over an IDENTICAL +/// journal. This is approach (a) from the spec: choose blob hashes whose high64 % 2 == 0 so shard 1's +/// bucket is always empty; the sharded path's shard-0 reducer and the single-shard path both call +/// `foldDeltasIntoGeneration` with the same delta stream (one routing into shard 0 of 2, the other +/// into shard 0 of 1). +/// +/// NOTE ON SEAL-BYTE EQUALITY: byte-for-byte equality of the `CasFoldSeal` is NOT asserted here. The +/// fold seal records the `blobTargetRunKey(gen, shard, seq)` path, which embeds the shard number. The +/// single-shard path writes `blobTargetRunKey(g, 0, 0)` for gc_shards=1, while the sharded path writes +/// `blobTargetRunKey(g, 0, 0)` for the shard-0 run AND `blobTargetRunKey(g, 1, 0)` for the (empty) +/// shard-1 run. The per-blob in-degree (the load-bearing property — it drives the spare/delete +/// decision) is identical; the seal's key-set legitimately differs by shard count. +TEST(CASGCShardEquivalence, SingleShardMatchesPhase1dInDegree) +{ + /// Build three blob hashes that ALL route to shard 0 under gc_shards=2 (high64 % 2 == 0). + /// high64=0 => shard 0, high64=2 => shard 0, high64=4 => shard 0. + const UInt128 hA = static_cast(0ULL) << 64; /// high64=0, routes to shard 0 + const UInt128 hB = static_cast(2ULL) << 64; /// high64=2, routes to shard 0 + const UInt128 hC = static_cast(4ULL) << 64; /// high64=4, routes to shard 0 + + const BlobRef refA{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hA)}; + const BlobRef refB{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hB)}; + const BlobRef refC{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hC)}; + ASSERT_EQ(blobShard(refA, 2), 0u) << "hA must route to shard 0 under gc_shards=2"; + ASSERT_EQ(blobShard(refB, 2), 0u) << "hB must route to shard 0 under gc_shards=2"; + ASSERT_EQ(blobShard(refC, 2), 0u) << "hC must route to shard 0 under gc_shards=2"; + ASSERT_EQ(blobShard(refA, 1), 0u) << "hA must route to shard 0 under gc_shards=1"; + ASSERT_EQ(blobShard(refB, 1), 0u) << "hB must route to shard 0 under gc_shards=1"; + ASSERT_EQ(blobShard(refC, 1), 0u) << "hC must route to shard 0 under gc_shards=1"; + + /// Construct the journal: hA gets net +2 (published twice), hB gets net +1, hC gets net 0 (publish + /// then drop => transitions to zero). This exercises all three outcomes (>1, =1, =0) for the + /// equivalence proof. + /// + /// Note: net +2 is unrealistic for production (two DISTINCT manifests can share a blob, each + /// contributing +1 independently) but is valid for the fold math test. It directly verifies that + /// accumulators sum correctly under both paths. + const RootNamespace ns{"ns-equiv"}; + const ManifestRef rA1{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = static_cast(0x1)}; + const ManifestRef rA2{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = static_cast(0x2)}; + const ManifestRef rB{.writer_epoch = 1, .build_sequence = 3, .manifest_ordinal = static_cast(0x3)}; + const ManifestRef rC{.writer_epoch = 1, .build_sequence = 4, .manifest_ordinal = static_cast(0x4)}; + + /// Helper lambda that sets up a fresh backend + store with the shared scripted journal, runs one GC + /// round with the given gc_shards, and returns the per-blob in-degrees in the sealed generation. + /// Returns {indeg_A, indeg_B, indeg_C}. + auto runJournalAndGetInDegrees = [&](uint64_t gc_shards) -> std::tuple + { + auto backend = std::make_shared(); + const Layout layout("p"); + /// Raw journal fixtures model an already-created pool and therefore establish both mandatory + /// controls before writing residual data. + seedPoolMetaForRestart(*backend); + + /// Write blob bodies so HEAD returns a token (GC retires zero-in-degree blobs only if present). + writeBlobBody(*backend, layout, hA); + writeBlobBody(*backend, layout, hB); + writeBlobBody(*backend, layout, hC); + + /// Write manifests: rA1 references hA once; rA2 also references hA once; rB references hB; + /// rC references hC. Each publication contributes +1 per referenced blob. + writeManifestRaw(*backend, layout, ns, rA1, {blobEntryFor("a", hA)}); + writeManifestRaw(*backend, layout, ns, rA2, {blobEntryFor("a", hA)}); + writeManifestRaw(*backend, layout, ns, rB, {blobEntryFor("b", hB)}); + writeManifestRaw(*backend, layout, ns, rC, {blobEntryFor("c", hC)}); + + /// Publish all four refs (tbl1=rA1, tbl2=rA2, tbl3=rB, tbl4=rC). + publishCommittedTransition(*backend, layout, ns, "tbl1", std::nullopt, rA1); + publishCommittedTransition(*backend, layout, ns, "tbl2", std::nullopt, rA2); + publishCommittedTransition(*backend, layout, ns, "tbl3", std::nullopt, rB); + publishCommittedTransition(*backend, layout, ns, "tbl4", std::nullopt, rC); + /// Drop tbl4 (hC net = 0): rC removed from the live set. + dropRefTransition(*backend, layout, ns, "tbl4", rC); + + /// Open a store with the given `gc_shards` over the pre-seeded restart state. + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = gc_shards}); + const UInt128 gc_id = UInt128(0xDEADBEEF42ULL); + Gc gc(store, gc_id); + EXPECT_TRUE(gc.runRegularRound().acquired_lease); + + /// The fold seal for new_generation (== snap_generation after fold) holds the in-degree runs. + /// After runRegularRound the snap_generation points at the COMPLETION generation; the fold + /// generation is snap_generation - 1 for the first full round. Use inDegreeOf (which reads + /// currentGenerationOf = completion generation) for the final in-degrees. + const std::vector shard0 = runsForShard(*backend, layout, /*shard=*/0); + const int64_t iA = inDegreeInRuns(*backend, shard0, refA); + const int64_t iB = inDegreeInRuns(*backend, shard0, refB); + const int64_t iC = inDegreeInRuns(*backend, shard0, refC); + return {iA, iB, iC}; + }; + + const auto [a1, b1_indeg, c1] = runJournalAndGetInDegrees(/*gc_shards=*/1); + const auto [a2, b2_indeg, c2] = runJournalAndGetInDegrees(/*gc_shards=*/2); + + /// The in-degree values must match exactly between the two runs. + EXPECT_EQ(a1, a2) + << "hA in-degree must match: gc_shards=1 gives " << a1 << ", gc_shards=2 gives " << a2; + EXPECT_EQ(b1_indeg, b2_indeg) + << "hB in-degree must match: gc_shards=1 gives " << b1_indeg << ", gc_shards=2 gives " << b2_indeg; + EXPECT_EQ(c1, c2) + << "hC in-degree must match: gc_shards=1 gives " << c1 << ", gc_shards=2 gives " << c2; + + /// Cross-check the known correct values (derivable from the scripted journal). + /// hA: +1 (tbl1/rA1) + 1 (tbl2/rA2) = 2. + EXPECT_EQ(a1, 2) << "hA in-degree must be 2 (two distinct live refs both citing hA)"; + /// hB: +1 (tbl3/rB) = 1. + EXPECT_EQ(b1_indeg, 1) << "hB in-degree must be 1"; + /// hC: +1 (tbl4/rC publish) - 1 (tbl4 drop) = 0. + EXPECT_EQ(c1, 0) << "hC in-degree must be 0 (publish then drop; net zero)"; +} + +/// ---- Phase 4, Task 8: two-replica disjoint-shard concurrency ---- +/// +/// With gc_shards=2 over a shared `InMemoryBackend`: +/// (a) DISJOINTNESS: a shard-0 reducer's product covers only hashes routing to shard 0; shard-1 +/// covers only hashes routing to shard 1 (`owns` check). +/// (b) PER-SHARD RUNS: each reducer writes its own write-once blob-target run; the runs for the two +/// shards are disjoint object keys and durably present after each `ShardReducer::reduce`. +/// (c) MERGED IN-DEGREE: the merged in-degrees across both shards equal the expected edge multiset, +/// and each blob is absent from the other shard's run (cross-shard disjointness). +/// +/// Interleaving: driven entirely from the test thread (no threads, no sleeps). The two reducers are +/// constructed and called sequentially from the test thread. This proves the protocol is correct even +/// when reducer work interleaves arbitrarily — the key-space disjointness is static. +TEST(CASGCShardTwoReplica, DisjointShardsConcurrentPerShardRuns) +{ + constexpr uint64_t kGcShards = 2; + constexpr uint64_t kNewGen = 1; + constexpr uint64_t kAttempt = 0; + + /// b0 routes to shard 0, b1 routes to shard 1 (from makeTwoShardHashes). + const auto [b0, b1] = makeTwoShardHashes(); + ASSERT_EQ(blobShard(b0, kGcShards), 0u) << "b0 must route to shard 0"; + ASSERT_EQ(blobShard(b1, kGcShards), 1u) << "b1 must route to shard 1"; + + auto backend = std::make_shared(); + const Layout layout("p"); + + /// (a) DISJOINTNESS — verify `owns` predicate before any reduce. + ShardReducer r0(0, kGcShards); + ShardReducer r1(1, kGcShards); + + EXPECT_TRUE(r0.owns(b0)) << "shard-0 reducer must own b0"; + EXPECT_FALSE(r0.owns(b1)) << "shard-0 reducer must NOT own b1"; + EXPECT_TRUE(r1.owns(b1)) << "shard-1 reducer must own b1"; + EXPECT_FALSE(r1.owns(b1) && r0.owns(b1)) << "no hash may be owned by both reducers"; + + /// Construct disjoint delta streams: b0 gets net +2 in shard 0; b1 gets net +1 in shard 1. + /// In production these buckets are produced by `foldManifestEdges` and partitioned by `blobShard` + /// (two distinct manifests both referencing b0 contribute two source edges; one manifest + /// referencing b1 contributes one source edge). + std::vector bucket0 = { + BlobDelta{.ref = b0, .source_id = UInt128(1), .remove = false}, + BlobDelta{.ref = b0, .source_id = UInt128(2), .remove = false}, + }; + std::vector bucket1 = { + BlobDelta{.ref = b1, .source_id = UInt128(3), .remove = false}, + }; + + /// (b) PER-SHARD RUNS — drive both reducers. + /// + /// Run shard-0 reducer (simulates the shard-0 replica's work). + const auto runs0 = r0.reduce(*backend, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket0)); + ASSERT_FALSE(runs0.empty()) << "shard-0 reducer must produce at least one RunRef"; + + /// Run shard-1 reducer (simulates the shard-1 replica's work, interleaved from the test thread). + const auto runs1 = r1.reduce(*backend, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket1)); + ASSERT_FALSE(runs1.empty()) << "shard-1 reducer must produce at least one RunRef"; + + /// The blob-target runs for both shards are durably present (the reducer's write-once `putIfAbsent`), + /// at disjoint object keys. + EXPECT_TRUE(backend->head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/0, /*seq=*/0)).exists) + << "shard-0 blob-target run must be durably written by r0.reduce"; + EXPECT_TRUE(backend->head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/1, /*seq=*/0)).exists) + << "shard-1 blob-target run must be durably written by r1.reduce"; + + /// (c) MERGED IN-DEGREE — the merged in-degrees across both shards equal the expected edge multiset. + EXPECT_EQ(inDegreeInRuns(*backend, runs0, b0), 2) + << "b0 in-degree must be 2 in shard-0 run"; + EXPECT_EQ(inDegreeInRuns(*backend, runs1, b1), 1) + << "b1 in-degree must be 1 in shard-1 run"; + /// Cross-shard: each blob must be absent from the other shard's run. + EXPECT_EQ(inDegreeInRuns(*backend, runs0, b1), 0) + << "b1 must NOT appear in shard-0's run (cross-shard disjointness)"; + EXPECT_EQ(inDegreeInRuns(*backend, runs1, b0), 0) + << "b0 must NOT appear in shard-1's run (cross-shard disjointness)"; +} + +/// ---- Phase 4 regression: gc_shards>1 retire-drain (High #1) ---- +/// +/// A FULL round-protocol regression that drives publish -> drop -> reclaim end-to-end under +/// `gc_shards = 2` with a droppable blob owned by a NON-zero shard. The fold/`ShardReducer` write one +/// in-degree run PER shard, so a zero-in-degree blob owned by shard 1..N is only ever retired (and +/// then exact-token deleted) if `retire`/`previewDeletes` scan EVERY blob-target shard. Before +/// `5f5fa5f7906` both hardcoded shard 0: a shard-1 candidate was never scanned, never retired, and +/// leaked forever. After the fix both shards are drained. +/// +/// The test plants TWO droppable blobs in the SAME round — one owned by shard 0, one owned by shard 1 +/// (verified via `blobShard(hash, 2)`) — and asserts BOTH are reclaimed. The shard-0 blob proves the +/// round works at all; the shard-1 blob is the regression's teeth (it would leak pre-fix while shard-0 +/// still drained, so a single-blob test could pass even with the bug). +/// +/// HOW IT WOULD LEAK PRE-FIX: under the old shard-0-only `retire`, the round folds the drop (shard-1 +/// blob's in-degree -> 0 in shard 1's run) but `retire` only reads shard 0's in-degree run and only +/// writes shard 0's retired set, so the shard-1 zero-in-degree blob is never proposed for retirement. +/// `previewDeletes` (also shard-0-only pre-fix) never lists it, the recheck never spares-or-deletes it, +/// and `blobExists(b1)` stays true at fixpoint. The shard-0 blob would still be reclaimed — which is +/// exactly why the existing in-degree-equivalence tests (all blobs route to shard 0) did not catch it. +TEST(CASGCShardRetireDrain, ReclaimsDroppableBlobOwnedByNonZeroShard) +{ + constexpr uint64_t kGcShards = 2; + + /// Two blob hashes routing to DIFFERENT shards under gc_shards=2. blobShard = high64 % 2. + /// high64=0 => shard 0; high64=1 => shard 1. + const UInt128 blob_shard0 = (static_cast(0ULL) << 64) | static_cast(7ULL); /// high64=0 => shard 0 + const UInt128 blob_shard1 = (static_cast(1ULL) << 64) | static_cast(7ULL); /// high64=1 => shard 1 + ASSERT_EQ(blobShard(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard0)}, kGcShards), 0u) << "blob_shard0 must route to shard 0"; + ASSERT_EQ(blobShard(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard1)}, kGcShards), 1u) << "blob_shard1 must route to shard 1 (regression teeth)"; + + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_shards = kGcShards}); + const Layout & layout = store->layout(); + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r0{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = static_cast(0xA0)}; + const ManifestRef r1{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = static_cast(0xA1)}; + const ManifestId id0{ns, r0}; + const ManifestId id1{ns, r1}; + + /// Local blobExists (the round-level helper is file-local to gtest_cas_gc_round.cpp). + auto blobExists = [&](const UInt128 & hash) + { + return backend->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + }; + auto manifestExists = [&](const ManifestId & id) + { + return backend->head(layout.manifestKey(id)).exists; + }; + /// Whether ANY gc-shard still holds an in-flight condemned entry (the ack-floor deletion pipeline is + /// in flight while this is true). Retired-in-snapshot (T4): reconstructed from the adopted fold seal's + /// kCondemned rows across all shards, not a separate retired list. + auto anyRetiredPending = [&] + { + return anyCondemnedInSeal(*backend, layout); + }; + /// Drive to a fixpoint over the ACK-FLOOR round: advance the store's mount ack each round (so the floor + /// follows the committed round) and stay alive while any work counter is nonzero OR an in-flight + /// retired entry remains in ANY shard. + auto driveToFixpoint = [&](Gc & gc) + { + for (size_t r = 0; r < 64; ++r) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + store->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending()) + break; + } + }; + + /// Publish: ref r0 names the shard-0 blob, ref r1 names the shard-1 blob (distinct refs => distinct + /// edges, each contributing +1 to its blob's in-degree in its OWNING shard's run). + writeBlobBody(*backend, layout, blob_shard0); + writeBlobBody(*backend, layout, blob_shard1); + writeManifestRaw(*backend, layout, ns, r0, {blobEntryFor("a", blob_shard0)}); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("b", blob_shard1)}); + publishCommittedTransition(*backend, layout, ns, "tbl0", std::nullopt, r0); + publishCommittedTransition(*backend, layout, ns, "tbl1", std::nullopt, r1); + + const UInt128 gc_id = UInt128(0xDEADBEEF42ULL); + Gc gc(store, gc_id); + driveToFixpoint(gc); + + /// While both refs are live: each blob's in-degree is 1 in its OWNING shard's run, and nothing is + /// collected (no-loss). Derive generation/attempt from gc/state — never hardcode. + const GcState live = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(live.snap_generation, 0u); + ASSERT_EQ(live.gc_shards, kGcShards) << "the pool must be running with gc_shards=2"; + EXPECT_EQ(inDegreeInRuns(*backend, runsForShard(*backend, layout, /*shard=*/0), BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard0)}), 1) + << "shard-0 blob in-degree must be 1 while live"; + EXPECT_EQ(inDegreeInRuns(*backend, runsForShard(*backend, layout, /*shard=*/1), BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard1)}), 1) + << "shard-1 blob in-degree must be 1 while live"; + EXPECT_TRUE(blobExists(blob_shard0)); + EXPECT_TRUE(blobExists(blob_shard1)); + + /// Drop BOTH refs: each blob's only edge goes away (in-degree -> 0 in its owning shard's run). + dropRefTransition(*backend, layout, ns, "tbl0", r0); + dropRefTransition(*backend, layout, ns, "tbl1", r1); + driveToFixpoint(gc); + + /// After drop + fixpoint: BOTH blobs are retired and exact-token deleted, and BOTH owner-removed + /// manifest bodies are collected. The shard-1 blob is the regression's teeth — pre-`5f5fa5f` it + /// would still exist here because retire/previewDeletes never scanned shard 1. + EXPECT_EQ(inDegreeInRuns(*backend, runsForShard(*backend, layout, /*shard=*/0), BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard0)}), 0) + << "shard-0 blob in-degree must be 0 after drop"; + EXPECT_EQ(inDegreeInRuns(*backend, runsForShard(*backend, layout, /*shard=*/1), BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard1)}), 0) + << "shard-1 blob in-degree must be 0 after drop"; + EXPECT_FALSE(blobExists(blob_shard0)) << "shard-0 droppable blob must be reclaimed"; + EXPECT_FALSE(blobExists(blob_shard1)) + << "shard-1 droppable blob must be reclaimed (High #1: retire must scan ALL shards, not just shard 0)"; + EXPECT_FALSE(manifestExists(id0)) << "shard-0 owner-removed manifest body must be reclaimed"; + EXPECT_FALSE(manifestExists(id1)) << "shard-1 owner-removed manifest body must be reclaimed"; + + /// Idempotent: another fixpoint changes nothing and never throws. + EXPECT_NO_THROW(driveToFixpoint(gc)); + EXPECT_FALSE(blobExists(blob_shard0)); + EXPECT_FALSE(blobExists(blob_shard1)); +} diff --git a/src/Disks/tests/gtest_cas_gc_source_edge.cpp b/src/Disks/tests/gtest_cas_gc_source_edge.cpp new file mode 100644 index 000000000000..5fffe7ad2a97 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_source_edge.cpp @@ -0,0 +1,84 @@ +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +TEST(CASSourceEdge, IdIsDeterministicAndPathSensitive) +{ + const ManifestId id{RootNamespace{"00/aa@cas@"}, ManifestRef{.writer_epoch = 1, .build_sequence = 15, .manifest_ordinal = 1}}; + EXPECT_EQ(sourceEdgeId(id, "a.bin"), sourceEdgeId(id, "a.bin")); // deterministic + EXPECT_NE(sourceEdgeId(id, "a.bin"), sourceEdgeId(id, "b.bin")); // path-sensitive + const ManifestId id2{id.root_namespace, ManifestRef{.writer_epoch = 1, .build_sequence = 31, .manifest_ordinal = 1}}; + EXPECT_NE(sourceEdgeId(id, "a.bin"), sourceEdgeId(id2, "a.bin")); // ref-sensitive +} + +TEST(CASSourceEdge, RunKeyRoundTripsAndOrdersByBlobThenSource) +{ + const BlobRef b1{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(1))}; + const BlobRef b2{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(2))}; + const UInt128 s1(10); + const UInt128 s2(20); + + BlobRef gb; + UInt128 gs; + SourceEdgeKeyCodec::parse(SourceEdgeKeyCodec::key(b1, s1), gb, gs); + EXPECT_EQ(gb, b1); + EXPECT_EQ(gs, s1); + EXPECT_LT(SourceEdgeKeyCodec::key(b1, s2), SourceEdgeKeyCodec::key(b2, s1)); // ref is the primary sort + EXPECT_LT(SourceEdgeKeyCodec::key(b1, s1), SourceEdgeKeyCodec::key(b1, s2)); // source_id is the secondary sort +} + +TEST(CASSourceEdge, KeyCodecSha256RoundTripAndRejectsBadSizes) +{ + /// sha256 (32-byte digest) round trip: key is 1 + 32 + 16 = 49 bytes, parse recovers the full ref. + BlobDigest d32{}; + for (size_t i = 0; i < d32.bytes.size(); ++i) + d32.bytes[i] = static_cast(i + 1); + const BlobRef sha_ref{BlobHashAlgo::Sha256, d32}; + const UInt128 sid(0xABCDu); + const String key32 = SourceEdgeKeyCodec::key(sha_ref, sid); + ASSERT_EQ(key32.size(), 49u); + BlobRef gb; + UInt128 gs; + SourceEdgeKeyCodec::parse(key32, gb, gs); + EXPECT_EQ(gb, sha_ref); + EXPECT_EQ(gs, sid); + + /// ch128 (16-byte digest): key is 1 + 16 + 16 = 33 bytes. + const BlobRef ch_ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(0x0102030405060708ULL))}; + const String key16 = SourceEdgeKeyCodec::key(ch_ref, sid); + ASSERT_EQ(key16.size(), 33u); + EXPECT_EQ(key16.substr(1), String(reinterpret_cast(ch_ref.digest.bytes.data()), 16) + u128ToBytesBE(sid)); + + /// Fail-close: a wrong-size key throws CORRUPTED_DATA, never a silent false. `key16` truncated by + /// one byte still declares algo=ch128 (33-byte width expected) but is only 32 bytes. + EXPECT_THROW(SourceEdgeKeyCodec::parse(key16.substr(0, key16.size() - 1), gb, gs), DB::Exception); + EXPECT_THROW(SourceEdgeKeyCodec::parse(String(20, '\0'), gb, gs), DB::Exception); + + /// Unknown algo byte -> NOT_IMPLEMENTED (fail closed). + String bad_key = key32; + bad_key[0] = static_cast(99); + EXPECT_THROW(SourceEdgeKeyCodec::parse(bad_key, gb, gs), DB::Exception); +} + +TEST(CASSourceEdge, KeyOrderSentinelFirstAtLen32) +{ + /// At sha256 width, the sentinel (source_id 0) sorts before any nonzero source_id for the same + /// digest, and digest magnitude order is preserved (big-endian raw-byte lexicographic order == + /// numeric magnitude order for a width-homogeneous run — the consult's load-bearing fact). + BlobDigest d{}; + d.bytes[0] = 0x10; + const BlobRef ref{BlobHashAlgo::Sha256, d}; + EXPECT_LT(SourceEdgeKeyCodec::key(ref, UInt128(0)), SourceEdgeKeyCodec::key(ref, UInt128(1))); + + BlobDigest d_small{}; + d_small.bytes[0] = 0x01; + BlobDigest d_large{}; + d_large.bytes[0] = 0x02; + const BlobRef ref_small{BlobHashAlgo::Sha256, d_small}; + const BlobRef ref_large{BlobHashAlgo::Sha256, d_large}; + EXPECT_LT(SourceEdgeKeyCodec::key(ref_small, UInt128(5)), SourceEdgeKeyCodec::key(ref_large, UInt128(5))); +} diff --git a/src/Disks/tests/gtest_cas_gc_state_format.cpp b/src/Disks/tests/gtest_cas_gc_state_format.cpp new file mode 100644 index 000000000000..aa661429f813 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_state_format.cpp @@ -0,0 +1,179 @@ +#include "cas_format_test_battery.h" +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; +} + +TEST(CASFormatBattery, GcState) +{ + GcState s; + s.round = 4; + s.gc_shards = 1; + s.snap_generation = 9; + s.snap_pruned_through = 7; + s.snap_attempt = 3; + s.manifest_sweep_cursor = ""; + s.lease = GcLease{UInt128(1), 12}; + runFormatBattery({FormatId::GcState, + [&] { return sealObject(FormatId::GcState, encodeGcState(s)); }, + [](std::string_view d) { decodeGcState(std::string(openObject(FormatId::GcState, d))); }, + currentFormatHeader("cas_gc_state") + + "{\"rnd\":\"4\",\"gcs\":1,\"sg\":\"9\",\"spt\":\"7\",\"sa\":\"3\",\"msc\":\"\"," + "\"lo\":\"00000000000000000000000000000001\",\"ls\":\"12\"}\n"}); +} + +TEST(CASFormatBattery, GcHeartbeat) +{ + GcHeartbeat hb{UInt128(1), 1741}; + runFormatBattery({FormatId::GcHeartbeat, + [&] { return sealObject(FormatId::GcHeartbeat, encodeGcHeartbeat(hb)); }, + [](std::string_view d) { decodeGcHeartbeat(std::string(openObject(FormatId::GcHeartbeat, d))); }, + currentFormatHeader("cas_gc_hb") + + "{\"by\":\"00000000000000000000000000000001\",\"seq\":\"1741\"}\n"}); +} + +/// ---------- field round-trips (migrated from gtest_cas_gc_formats.cpp, re-pointed at the text codec) ---------- + +TEST(CASGCStateFormat, RoundTripsCoreFields) +{ + GcState s; + s.round = 7; + s.gc_shards = 1; + s.snap_generation = 12; + s.lease.owner = hexToU128("00000000000000000000000000000005"); + s.lease.seq = 5; + auto d = decodeGcState(encodeGcState(s)); + EXPECT_EQ(d.round, 7u); + EXPECT_EQ(d.gc_shards, 1u); + EXPECT_EQ(d.snap_generation, 12u); + EXPECT_EQ(d.lease.owner, hexToU128("00000000000000000000000000000005")); + EXPECT_EQ(d.lease.seq, 5u); +} + +TEST(CASGCStateFormat, SnapPrunedThroughAndAttemptAndCursorRoundTrip) +{ + GcState s; + s.gc_shards = 2; + s.snap_generation = 42; + s.snap_pruned_through = 38; + s.snap_attempt = 7; + s.manifest_sweep_cursor = "p/cas/manifests/server/store/abc/table@cas@/writer/42/aa/id"; + auto d = decodeGcState(encodeGcState(s)); + EXPECT_EQ(d.snap_pruned_through, 38u); + EXPECT_EQ(d.snap_attempt, 7u); + EXPECT_EQ(d.manifest_sweep_cursor, s.manifest_sweep_cursor); +} + +TEST(CASGCStateFormat, DefaultsRoundTrip) +{ + GcState s; /// gc_shards defaults to 1 + EXPECT_EQ(s.gc_shards, 1u); + auto d = decodeGcState(encodeGcState(s)); + EXPECT_EQ(d.round, 0u); + EXPECT_EQ(d.snap_attempt, 0u); + EXPECT_TRUE(d.manifest_sweep_cursor.empty()); + EXPECT_EQ(d.lease.owner, UInt128{}); +} + +TEST(CASGCStateFormat, RejectsZeroGcShards) +{ + /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes + /// the header gate, which is the point — the BODY is what has to fail here. + const String bad = "{\"type\":\"cas_gc_state\",\"v\":3}\n" + "{\"rnd\":\"0\",\"gcs\":0,\"sg\":\"0\",\"spt\":\"0\",\"sa\":\"0\",\"msc\":\"\"," + "\"lo\":\"00000000000000000000000000000000\",\"ls\":\"0\"}\n"; + EXPECT_THROW(decodeGcState(bad), DB::Exception); +} + +#ifndef DEBUG_OR_SANITIZER_BUILD +/// encodeGcState(gc_shards=0) throws LOGICAL_ERROR, which aborts the whole process in debug/sanitizer +/// builds instead of behaving like a catchable exception -- CASGCStateFormatDeathTest below proves the +/// abort positively in those builds instead. +TEST(CASGCStateFormat, RejectsZeroGcShardsOnEncode) +{ + GcState state; + state.gc_shards = 0; + + try + { + encodeGcState(state); + FAIL() << "expected exception code " << DB::ErrorCodes::LOGICAL_ERROR; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + } +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASGCStateFormatDeathTest, RejectsZeroGcShardsOnEncodeAborts) +{ + GcState state; + state.gc_shards = 0; + EXPECT_DEATH({ (void)encodeGcState(state); }, ""); +} +#endif + +TEST(CASGCStateFormat, RejectsAbsentGcShards) +{ + /// An absent gcs key must fail closed (the writer always emits it) rather than silently defaulting + /// to the struct's gc_shards = 1 — a missing shard count means a corrupt object, not "use the floor". + /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes + /// the header gate, which is the point — the BODY is what has to fail here. + const String bad = "{\"type\":\"cas_gc_state\",\"v\":3}\n" + "{\"rnd\":\"0\",\"sg\":\"0\",\"spt\":\"0\",\"sa\":\"0\",\"msc\":\"\"," + "\"lo\":\"00000000000000000000000000000000\",\"ls\":\"0\"}\n"; + EXPECT_THROW(decodeGcState(bad), DB::Exception); +} + +TEST(CASGCStateFormat, GarbageFailsClosed) +{ + EXPECT_THROW(decodeGcState(String("")), DB::Exception); + EXPECT_THROW(decodeGcState(String("not a cas object\n")), DB::Exception); +} + +TEST(CASGCHeartbeatFormat, RoundTripAndBoundaries) +{ + GcHeartbeat hb; + hb.owner = hexToU128("0123456789abcdeffedcba9876543210"); + hb.hb_seq = 12345; + GcHeartbeat d = decodeGcHeartbeat(encodeGcHeartbeat(hb)); + EXPECT_EQ(d.owner, hb.owner); + EXPECT_EQ(d.hb_seq, 12345u); + + GcHeartbeat z; + z.owner = hexToU128("ffffffffffffffffffffffffffffffff"); + z.hb_seq = 0; + EXPECT_EQ(decodeGcHeartbeat(encodeGcHeartbeat(z)).owner, z.owner); + EXPECT_THROW(decodeGcHeartbeat(String("short")), DB::Exception); +} + +TEST(CASGCHeartbeatFormat, RejectsMissingIdentityFields) +{ + /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes + /// the header gate, which is the point — the BODY is what has to fail here. + const String header = "{\"type\":\"cas_gc_hb\",\"v\":3}\n"; + + const auto expectCorrupted = [](const String & data) + { + try + { + decodeGcHeartbeat(data); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + }; + + expectCorrupted(header + "{\"seq\":\"1741\"}\n"); + expectCorrupted(header + "{\"by\":\"00000000000000000000000000000001\"}\n"); +} diff --git a/src/Disks/tests/gtest_cas_gc_stop_start.cpp b/src/Disks/tests/gtest_cas_gc_stop_start.cpp new file mode 100644 index 000000000000..fa4c864cc8de --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_stop_start.cpp @@ -0,0 +1,492 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Task 11 (rev.7 spec §6): `SYSTEM CAS GC STOP` / `GC START` -- granular operator control +/// of ONLY the background GC scheduler. STOP is STOP-IN-PLACE: it joins the worker + heartbeat threads and +/// clears the in-process leadership hint, but RETAINS the scheduler object so a later START restarts the +/// SAME instance (its `gc_id` + lease-observation history preserved). The disk stays fully usable (reads/ +/// writes unaffected) while GC is stopped. START refuses on a decommissioned/uncertain pool (typed 668). +/// +/// These tests exercise the scheduler-level behavior directly (`CasGcScheduler::stop`/`start`) and the +/// end-to-end verbs through a real `ContentAddressedMetadataStorage`. Harness patterns follow +/// gtest_cas_forget.cpp and gtest_cas_gc_log.cpp. + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +} + +using namespace DB; +using DB::Cas::CasGcScheduler; +using DB::Cas::GcRoundLogRecord; +using DB::Cas::InMemoryBackend; +using DB::Cas::PoolLifecycle; +using DB::Cas::RoundReport; +using DB::Cas::tests::openPoolForTest; + +namespace +{ + +/// A live table dir + committed part reused by the "reads/writes unaffected while stopped" test (the shape +/// gtest_cas_forget.cpp / gtest_cas_operation_gate.cpp use). +const std::string kTableDir = "gg0/gg0gg0g0-0808-4808-8808-080808080808"; +const std::string kPartDir = kTableDir + "/all_1_1_0"; +const std::string kPartFile = kPartDir + "/data.bin"; + +/// The Pool-level `server_root_id` `openPoolForTest` mints (mirrors gtest_cas_lifecycle_condition.cpp). +const std::string kSrid = "test"; + +/// GC's fence-out applied directly to the mount lease (preserve the body, set `gc_fenced`, bump `seq`) so a +/// subsequent `tryRemountOnce` verdicts `Recover` and reclaims a FRESH incarnation immediately (no +/// lease-expiry wait), driving a transient-not-live pool back to `Live`. Mirrors +/// gtest_cas_lifecycle_condition.cpp's helper — used by the operator-STOP-persistence test below. +void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, + DB::Cas::PutOutcome::Done); +} + +/// A real `ContentAddressedMetadataStorage` over a fresh, unique local object storage. `context == nullptr` +/// (a unit-test mount), so `startup()` creates NO GC scheduler -- the GC entry points, and `gcStart`, create +/// one lazily. GC is enabled by default (`gc_enabled == true`, `gc_interval_sec == 60`), so no background +/// round fires during the sub-second test window. Mirrors gtest_cas_forget.cpp's `openForgetStorage`. +std::shared_ptr openGcStorage() +{ + static std::atomic counter{0}; + const auto scratch = std::filesystem::temp_directory_path() + / ("ca_gc_stopstart_scratch_" + std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1))); + auto settings = Cas::tests::makeSettingsForTest("test", scratch); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +void commitOnePart(ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kTableDir + "/tmp_insert_all_1_1_0/data.bin", 65536, WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kTableDir + "/tmp_insert_all_1_1_0", kPartDir); + tx->commit(NoCommitOptions{}); +} + +/// A thread-safe sink for the scheduler's per-round log records, with a condition variable so a test can +/// WAIT (never sleep) for a background round to land. `waitForSuccessFinish` blocks until a Finish record +/// with `outcome == Success` (the round acquired/renewed the GC lease) appears at index >= `from`, or the +/// timeout trips (only on a genuine hang/regression -- the round is sub-millisecond on an in-memory pool). +class RoundLogSink +{ +public: + Cas::GcRoundLogger logger() + { + return [this](const GcRoundLogRecord & r) + { + std::lock_guard lock(mutex); + records.push_back(r); + cv.notify_all(); + }; + } + + /// Index one past the current end of the record log -- the "from" watermark for a subsequent wait. + size_t mark() + { + std::lock_guard lock(mutex); + return records.size(); + } + + /// The first Success Finish record at index >= `from`, waiting up to `timeout`. Returns nullopt on + /// timeout so the caller asserts with a clear message rather than hanging. + std::optional waitForSuccessFinish(size_t from, std::chrono::milliseconds timeout) + { + std::unique_lock lock(mutex); + const bool ok = cv.wait_for(lock, timeout, [&] + { + for (size_t i = from; i < records.size(); ++i) + if (records[i].event_type == GcRoundLogRecord::EventType::Finish + && records[i].outcome == GcRoundLogRecord::Outcome::Success) + return true; + return false; + }); + if (!ok) + return std::nullopt; + for (size_t i = from; i < records.size(); ++i) + if (records[i].event_type == GcRoundLogRecord::EventType::Finish + && records[i].outcome == GcRoundLogRecord::Outcome::Success) + return records[i]; + return std::nullopt; + } + +private: + std::mutex mutex; + std::condition_variable cv; + std::vector records; +}; + +/// A generous wait bound for a background round to land -- trips only on a real deadlock/regression. +constexpr std::chrono::milliseconds kRoundWait{60000}; + +/// Bound for the [C1] self-exit observation: comfortably above the 1s pacing interval (so a slow CI box +/// still sees the loop tick + observe the terminal state) yet short enough that the RED demo (self-exit +/// removed) fails fast rather than hanging for `kRoundWait`. +constexpr std::chrono::milliseconds kSelfExitWait{15000}; + +/// A bounded OBSERVATION window (not a sleep-to-fix-a-race) for the "stays stopped across recovery" test: +/// comfortably above the 1s pacing interval so a running scheduler would have filled it with several +/// rounds, yet short enough to keep the negative assertion cheap. Its meaning is anchored by a positive +/// control (an explicit START right after DOES produce a round through the same sink). +constexpr std::chrono::milliseconds kStayStoppedWindow{3000}; + +} + +/// (C1) A NATURAL terminal transition (`VanishedReplaced`, or here `VanishedForgotten` forced via the test +/// seam) is never accompanied by a `stop()` on this scheduler — only `~Pool`/FORGET join it. The scheduler's +/// OWN loops must observe the terminal lifecycle at their next tick and self-exit, +/// so the pacing loop stops spamming Failed rounds (the G2 zombie) and the steal-capable loop can never +/// fold/condemn a foreign pool's prefix. Drive it while RUNNING, then vanish it, then prove BOTH loops +/// self-exit (bounded cv wait, no sleep) and that no further round-log rows appear. +TEST(CASGCStopStart, SchedulerSelfExitsOnNaturalVanished) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + RoundLogSink sink; + /// 1s interval: the loop ticks ~1s; the cv wait below (never a sleep) synchronizes on real records. + CasGcScheduler sched(store, std::chrono::seconds(1), "CasGcSelfExitTest", "ca-disk", sink.logger()); + sched.start(); + + /// Prove the loop is genuinely RUNNING first: a background round must land and acquire the lease. + ASSERT_TRUE(sink.waitForSuccessFinish(/*from=*/0, kRoundWait).has_value()) + << "the scheduler must be pacing rounds before we drive it terminal"; + + /// A natural terminal transition (forced here via the seam; in production `VanishedReplaced` and + /// `IdentityLost` arrive identically, WITHOUT anyone calling stop() on this scheduler). + store->setLifecycleForTest(PoolLifecycle::VanishedForgotten); + + ASSERT_TRUE(sched.waitForTerminalSelfExitForTest(kSelfExitWait)) + << "both the pacing and heartbeat loops must self-exit once the pool is Vanished"; + + /// No further round-log rows appear after the self-exit: both loops have returned, so capture the + /// count, reap them with stop() (a hang/double-terminate here would fail the test), and assert stable. + const size_t count_at_exit = sink.mark(); + sched.stop(); + EXPECT_EQ(sink.mark(), count_at_exit) << "a self-exited pacing loop must emit no further round records"; + EXPECT_FALSE(sched.gcHealth().is_leader); +} + +/// (C1, rev.8 §9 item 8) `IdentityLost` is now a fail-loud TERMINAL state, so the scheduler must self-exit +/// there exactly as it does on `Vanished` — a scheduler ticking against a half-erased pool is a pure zombie +/// (eternal `CORRUPTED_DATA` retries against the vanished `gc/state`). Prove BOTH loops self-exit and that no +/// further round-log rows appear, and that leadership is cleared. +TEST(CASGCStopStart, SchedulerSelfExitsOnIdentityLost) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + RoundLogSink sink; + CasGcScheduler sched(store, std::chrono::seconds(1), "CasGcIdentityLostTest", "ca-disk", sink.logger()); + sched.start(); + + ASSERT_TRUE(sink.waitForSuccessFinish(/*from=*/0, kRoundWait).has_value()); + + store->setLifecycleForTest(PoolLifecycle::IdentityLost); + + ASSERT_TRUE(sched.waitForTerminalSelfExitForTest(kSelfExitWait)) + << "IdentityLost is terminal (rev.8): both the pacing and heartbeat loops must self-exit"; + + const size_t count_at_exit = sink.mark(); + sched.stop(); + EXPECT_EQ(sink.mark(), count_at_exit) << "a self-exited pacing loop must emit no further round records"; + EXPECT_FALSE(sched.gcHealth().is_leader) << "a self-exited scheduler must report it no longer leads"; +} + +/// (C1 cleanup hygiene) After BOTH loops self-exit on a terminal transition, `stop()` must cleanly reap +/// the already-finished (joinable) threads, a second `stop()` is a safe no-op, and destruction (scope exit +/// → ~CasGcScheduler → stop()) runs clean — the ThreadFromGlobalPool join/reset contract holds for a +/// self-exited thread exactly as for a stop()-signalled one. +TEST(CASGCStopStart, StopAndDestroyCleanAfterSelfExit) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + { + RoundLogSink sink; + CasGcScheduler sched(store, std::chrono::seconds(1), "CasGcSelfExitCleanupTest", "ca-disk", sink.logger()); + sched.start(); + ASSERT_TRUE(sink.waitForSuccessFinish(/*from=*/0, kRoundWait).has_value()); + + store->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + ASSERT_TRUE(sched.waitForTerminalSelfExitForTest(kSelfExitWait)); + + EXPECT_NO_THROW(sched.stop()) << "stop() must cleanly join the self-exited threads"; + EXPECT_NO_THROW(sched.stop()) << "a second stop() after self-exit is a safe no-op"; + /// Destruction at scope exit runs stop() a third time — also clean (test completing proves it). + } + SUCCEED(); +} + +/// (a + e) STOP joins the worker + heartbeat threads and clears the in-process leadership hint. The T10 +/// lesson: make the assertion REAL -- acquire leadership via a manual round FIRST, so `is_leader` is +/// genuinely true before STOP for the clear to prove anything (otherwise `EXPECT_FALSE` would be vacuous). +TEST(CASGCStopStart, StopJoinsWorkersAndClearsLeadershipHint) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + /// A long interval keeps any BACKGROUND round from firing; the manual round below is what leads. + CasGcScheduler sched(store, std::chrono::seconds(3600), "CasGcStopStartTest", "ca-disk"); + sched.start(); + + /// Acquire REAL leadership: a manual round on a free lease acquires it. + const RoundReport rep = sched.runOneRoundNow(); + ASSERT_TRUE(rep.acquired_lease) << "a manual round on a fresh pool must acquire the free GC lease"; + ASSERT_TRUE(sched.gcHealth().is_leader) << "leadership must be true BEFORE stop for the clear to prove anything"; + ASSERT_TRUE(sched.isQuiescent()) << "the manual round completed; nothing is in flight"; + + sched.stop(); /// joins loop + heartbeat threads (the test completing without hanging proves the join) + + EXPECT_TRUE(sched.isQuiescent()) << "no GC round may be in flight after stop joined the workers"; + EXPECT_FALSE(sched.gcHealth().is_leader) + << "stop must clear the in-process leadership hint (the disk no longer leads GC)"; +} + +/// (b) START after STOP restarts the SAME scheduler: background rounds resume, they carry the SAME gc_id +/// (identity preserved across the restart), and leadership is re-entered via the next round's NORMAL +/// acquisition (is_leader becomes true only after the restarted background round re-acquires the lease). +/// Deterministic and sleep-free: a condition variable fed by the round logger waits for each background +/// Finish. This also exercises `start()`'s post-join re-entrancy -- a bug there would hang the wait. +TEST(CASGCStopStart, StartAfterStopResumesBackgroundRoundsWithSameGcId) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + RoundLogSink sink; + /// 1s interval: the background loop's first round fires ~1s after start(); the cv wait (not a sleep) + /// synchronizes on the actual Finish record. + CasGcScheduler sched(store, std::chrono::seconds(1), "CasGcStopStartTest", "ca-disk", sink.logger()); + + /// First run: background rounds start and one acquires the lease. + sched.start(); + const auto first = sink.waitForSuccessFinish(/*from=*/0, kRoundWait); + ASSERT_TRUE(first.has_value()) << "the background scheduler must run a round and acquire the lease after start()"; + EXPECT_TRUE(sched.gcHealth().is_leader) << "leadership is held after the first background round"; + const std::string gc_id_before = first->gc_id; + EXPECT_FALSE(gc_id_before.empty()); + + /// Stop: leadership hint cleared, threads joined. + sched.stop(); + EXPECT_FALSE(sched.gcHealth().is_leader) << "stop clears the leadership hint"; + const size_t after_stop = sink.mark(); + + /// Restart the SAME instance: a NEW background round must land, re-acquiring the lease, and it must + /// carry the SAME gc_id (proving the instance -- and its lease observer -- survived the restart). + sched.start(); + const auto second = sink.waitForSuccessFinish(/*from=*/after_stop, kRoundWait); + ASSERT_TRUE(second.has_value()) << "background rounds must resume after START (start() is re-enterable post-join)"; + EXPECT_EQ(second->gc_id, gc_id_before) << "the restarted scheduler must preserve its gc_id (same instance)"; + EXPECT_TRUE(sched.gcHealth().is_leader) + << "leadership is re-entered via the restarted round's normal lease acquisition"; + + sched.stop(); +} + +/// (c) STOP and START are both idempotent: a second STOP on an already-stopped scheduler is a safe no-op, +/// and a second START on a running one is a no-op that leaves it running (a manual round still works). +TEST(CASGCStopStart, StopAndStartAreIdempotent) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + CasGcScheduler sched(store, std::chrono::seconds(3600), "CasGcStopStartTest", "ca-disk"); + + sched.start(); + EXPECT_NO_THROW(sched.start()) << "a second START on a running scheduler is a no-op"; + + sched.stop(); + EXPECT_NO_THROW(sched.stop()) << "a second STOP on a stopped scheduler is a safe no-op"; + EXPECT_TRUE(sched.isQuiescent()); + EXPECT_FALSE(sched.gcHealth().is_leader); + + /// After the double-stop, START still restarts the same instance and it runs a round. + sched.start(); + const RoundReport rep = sched.runOneRoundNow(); + EXPECT_TRUE(rep.acquired_lease) << "the restarted scheduler still runs rounds after idempotent stop/start"; + sched.stop(); +} + +/// (d) START refuses on a Vanished disk with the typed 668 (`INVALID_STATE`) error -- restarting GC on a +/// decommissioned pool is meaningless and would only spin failing rounds -- while STOP on the SAME +/// Vanished disk (with a live scheduler present) SUCCEEDS: stopping the reclaimer on a sick disk is a +/// legitimate operator action, so STOP never consults the operation gate. +TEST(CASGCStopStart, StartRefusesOnVanishedButStopSucceeds) +{ + /// START on a Vanished disk -> typed 668. No scheduler needed: the gate refuses before touching it. + { + auto storage = openGcStorage(); + auto pool = storage->store(); /// captured while Live (store() throws once Vanished) + pool->setLifecycleForTest(PoolLifecycle::VanishedForgotten); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->gcStart(); }); + } + + /// STOP on a Vanished disk WITH a live scheduler -> succeeds. + { + auto storage = openGcStorage(); + storage->gcStart(); /// Live: lazily creates + starts a scheduler + ASSERT_TRUE(storage->gcHealth().has_value()) << "gcStart must have created a scheduler on a Live disk"; + + auto pool = storage->store(); + pool->setLifecycleForTest(PoolLifecycle::VanishedForgotten); + + EXPECT_NO_THROW(storage->gcStop()) << "stopping GC on a Vanished disk is legitimate operator action"; + } +} + +/// (f) The disk stays fully usable while its GC scheduler is stopped: a store()-path write + read succeed +/// after `gcStop`. STOP controls ONLY the GC pacer, not the disk's data plane. +TEST(CASGCStopStart, DiskReadsWritesUnaffectedWhileGcStopped) +{ + auto storage = openGcStorage(); + storage->gcStart(); /// create + start the scheduler + storage->gcStop(); /// stop it in place (scheduler retained, threads joined) + + /// A write (commit a part) and a read (existsFile) both succeed with GC stopped. + EXPECT_NO_THROW(commitOnePart(*storage)); + EXPECT_TRUE(storage->existsFile(kPartFile)) << "reads/writes must be unaffected while the GC scheduler is stopped"; + + /// And START brings the scheduler back (idempotent, re-enterable) without disturbing the data. + EXPECT_NO_THROW(storage->gcStart()); + EXPECT_TRUE(storage->existsFile(kPartFile)); + storage->gcStop(); +} + +/// (T11 M3, acceptance matrix) Two threads hammering `gcStop`/`gcStart` on the SAME storage concurrently. +/// The verbs serialize on `lifecycle_mutex` (then `gc_scheduler_mutex`, always in that order — so there is +/// no lock-order inversion and hence no deadlock), so each call is atomic: the barrage interleaves in any +/// order but never tears the retained scheduler pointer or its worker-thread set. We bound each worker with +/// a `std::future` timeout (never a sleep) so a deadlock regression fails FAST instead of hanging the suite, +/// and — since the final serialized call determines the resting state — a single quiet STOP then START at +/// the end lands the object in a well-defined, usable state (last call wins). ASan/TSan running this proves +/// the racing start()/stop() thread spawns+joins never race the shared members. +TEST(CASGCStopStart, ConcurrentStopStartFromTwoThreadsStaysConsistent) +{ + auto storage = openGcStorage(); + + /// 200 iterations each, opposite phase, so the two threads spend the whole run contending on the + /// lifecycle mutex with one about to START while the other is about to STOP. + constexpr int kIters = 200; + auto worker = [&](bool start_first) + { + for (int i = 0; i < kIters; ++i) + { + if (start_first) { storage->gcStart(); storage->gcStop(); } + else { storage->gcStop(); storage->gcStart(); } + } + }; + + auto a = std::async(std::launch::async, worker, true); + auto b = std::async(std::launch::async, worker, false); + ASSERT_EQ(a.wait_for(std::chrono::seconds(60)), std::future_status::ready) + << "two-thread GC stop/start must not deadlock (both verbs lock lifecycle_mutex then gc_scheduler_mutex)"; + ASSERT_EQ(b.wait_for(std::chrono::seconds(60)), std::future_status::ready) + << "two-thread GC stop/start must not deadlock"; + a.get(); + b.get(); + + /// No torn state: a scheduler exists (both workers created/re-entered one) and its health snapshot is + /// coherently queryable rather than reading a half-published pointer. + ASSERT_TRUE(storage->gcHealth().has_value()) << "the scheduler must exist and report coherent health after the barrage"; + + /// Last call wins: once contention ends, one serialized STOP lands it stopped (leadership cleared, + /// quiescent), and one serialized START lands it running again — each observed deterministically. + storage->gcStop(); + ASSERT_TRUE(storage->gcHealth().has_value()); + EXPECT_FALSE(storage->gcHealth()->is_leader) << "a final serialized STOP clears leadership -- last call wins"; + + storage->gcStart(); + EXPECT_TRUE(storage->gcHealth().has_value()) << "a final serialized START leaves the scheduler present"; + + /// The data plane is unharmed by the whole barrage: a write + read still succeed. + EXPECT_NO_THROW(commitOnePart(*storage)); + EXPECT_TRUE(storage->existsFile(kPartFile)); + storage->gcStop(); +} + +/// (T11 cannot-verify, acceptance matrix) Operator intent PERSISTS across a transient recovery: after the +/// operator STOPs GC, the disk loses its mount lease (transient-not-live) and self-remounts back to Live — +/// and NOTHING restarts the GC scheduler. Recovery is a Pool-internal operation with no reference to the +/// scheduler; only an explicit START (`SYSTEM CAS GC START`) resumes it. We prove the scheduler +/// was genuinely running+leading first, STOP it, drive a real transient→Live recovery on the pool, then show +/// it stays stopped across a bounded observation window (a running 1s-paced scheduler would have produced +/// several rounds), and finally that an explicit START — the ONLY resumption path — brings rounds back on the +/// SAME instance (`gc_id` preserved). The positive control makes the negative meaningful: the sink IS live. +TEST(CASGCStopStart, OperatorStopPersistsAcrossTransientRecovery) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + RoundLogSink sink; + /// 1s interval so a RUNNING scheduler would pace rounds within the observation window below. + CasGcScheduler sched(store, std::chrono::seconds(1), "CasGcStopPersistTest", "ca-disk", sink.logger()); + + /// The operator has GC running and leading. + sched.start(); + ASSERT_TRUE(sink.waitForSuccessFinish(/*from=*/0, kRoundWait).has_value()) + << "the scheduler must be pacing rounds and leading before the operator stops it"; + ASSERT_TRUE(sched.gcHealth().is_leader); + + /// The operator STOPs GC (stop-in-place: threads joined, leadership hint cleared). + sched.stop(); + ASSERT_FALSE(sched.gcHealth().is_leader); + const size_t after_stop = sink.mark(); + + /// The disk now suffers a transient mount-lease loss and self-remounts back to Live (a fresh + /// incarnation), WITHOUT any operator action — exactly the recovery §4 describes. + store->tripMountLost(); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive); + fenceOutMount(*backend, store->layout().mountKey(kSrid)); + ASSERT_TRUE(store->tryRemountOnce()) << "the self-remount must reclaim a fresh incarnation"; + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live) << "the pool must auto-recover to Live"; + + /// The operator's STOP persists: recovery restarted NOTHING. The scheduler is still not leading and + /// still quiescent, and NO background round appears across a window a running scheduler would have + /// filled many times over. + EXPECT_FALSE(sched.gcHealth().is_leader); + EXPECT_TRUE(sched.isQuiescent()); + EXPECT_FALSE(sink.waitForSuccessFinish(after_stop, kStayStoppedWindow).has_value()) + << "a self-remount recovery must NOT restart an operator-STOPped GC scheduler"; + + /// Positive control: only an explicit START resumes rounds, on the SAME instance (gc_id preserved). + /// This also proves the sink WOULD have caught a round, so the negative above is meaningful. + sched.start(); + const auto resumed = sink.waitForSuccessFinish(after_stop, kRoundWait); + ASSERT_TRUE(resumed.has_value()) << "an explicit START must resume background rounds after the recovery"; + EXPECT_TRUE(sched.gcHealth().is_leader); + sched.stop(); +} diff --git a/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp b/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp new file mode 100644 index 000000000000..430e2faa9293 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp @@ -0,0 +1,416 @@ +#include + +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include + +/// Regression suite for the soak S04 / S04b undercount failures: +/// Code: 246 CORRUPTED_DATA: CAS blob in-degree: merged in-degree -1 < 0 for a blob ... +/// +/// H1 (DeposedFoldAdopt) and H1b (FenceWindowReRemoval) guard against the fence-window re-fold +/// undercount. Fixed STRUCTURALLY by replacing the persisted integer in-degree with an idempotent +/// source-edge SET: re-folding a fence-window removal across generations is a set-difference no-op, +/// so the underflow cannot occur (NOT by patching the sealed cursor — that approach was rejected). +/// +/// H2 (DuplicateRemovalIdempotent) guards against the duplicate-remove undercount that existed when +/// in-degree was a persisted integer: two events both carrying `old=committed(r1)` subtracted -1 +/// twice from a blob's count, driving it to -1. Same fix — the second removal of an already-absent +/// edge is a no-op. (Formerly staged the second event as a `{old=committed(r1), +/// new=committed(r2)}` "repoint" -- a single op naming DIFFERENT manifests in its old/new bindings. +/// Post-classifier (see Pool/CasRefProtocol.cpp's `classifyOwnerTransitionShape`) that single-op shape +/// is not representable at all: `manifestEdgesOfTxn` now throws `CORRUPTED_DATA` on it, same as the +/// state machine always has. The ACTUALLY representable duplicate-removal hazard -- two SEPARATE +/// remove-committed events both naming `old=committed(r1)`, which the GC fold extracts blindly without +/// replaying the state machine -- is what this test exercises instead.) + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int ABORTED; +} + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +const UInt128 kGcA = hexToU128("0000000000000000000000000000000a"); +const UInt128 kGcB = hexToU128("0000000000000000000000000000000b"); + +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} + +bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +/// A committed `RefOwnerBinding` for a raw `owner_transition` op. The raw appender is now +/// `tests::appendOwnerEvent`, which writes ONE `owner_transition` ref-log transaction via +/// `writeRefLogTxnRaw` at the next `RefTxnId` -- the GC fold EXTRACTS edges from each log +/// (`manifestEdgesOfTxn`) and never replays them through the state machine, so a SHAPE-legal +/// `old_binding` (an exact `remove committed` op) that no longer names the table's current committed +/// owner is still folded -- it is not caught until (and unless) the full state machine replays the +/// log. That is exactly the "duplicate removal of an already-removed committed ref" hazard H2 below +/// exercises: two SEPARATE remove-committed events for the same `(ref_name, manifest_ref)`, each +/// individually shape-legal (`classifyOwnerTransitionShape` accepts every one), but the second is a +/// stale repeat the idempotent source-edge set must absorb rather than double-subtract. +RefOwnerBinding committed(const String & ref_name, const ManifestRef & r) +{ + return RefOwnerBinding{RefOwnerKind::Committed, ref_name, r}; +} + +} + +/// ============================ H2: DUPLICATE COMMITTED REMOVAL IS IDEMPOTENT (REGRESSION GUARD) ======== +/// +/// Two SEPARATE journal events both carry the EXACT same `old = committed(r1)` removal: +/// v2: DROP r1 {old=committed(r1), new=none} => removes r1's source-edge to {1,2} +/// v3: DUPLICATE DROP r1 {old=committed(r1), new=none} => removes r1's source-edge to {1,2} AGAIN +/// +/// Each event is individually SHAPE-legal (`classifyOwnerTransitionShape` accepts a bare +/// `old=Committed, new=none` removal unconditionally; it has no state to check that the removal is +/// still live). The GC fold extracts edges from each log directly, without replaying the state machine +/// (which alone would notice the second removal names an owner that is no longer bound), so both +/// events fold. Under the OLD integer in-degree model this drove blob 2 to prior(1) + (-1) + (-1) = -1 +/// and threw CORRUPTED_DATA. Under the FIXED idempotent source-edge SET model the second removal of +/// r1's edge to a blob is a no-op: each source edge is present or absent, and removing an +/// already-absent edge is silent. +/// +/// Correct post-fix behaviour: GC must NOT throw, and both blobs -- owned only by r1, which has no +/// live owner after the (idempotent) drop -- become collectible (in-degree 0, keys gone). +TEST(CASGCUndercount, H2DuplicateCommittedRemovalIsIdempotentNoUnderflow) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r1 = ref(1, 0xB1); + + /// r1 pins blobs {1,2}. + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeBlobBody(*backend, store->layout(), DB::UInt128(2)); + writeManifestRaw(*backend, store->layout(), ns, r1, + {blobEntryFor("a", DB::UInt128(1)), blobEntryFor("b", DB::UInt128(2))}); + + /// v1: publish r1 (owner: none -> committed(r1)). Fold it so blobs 1 and 2 are each pinned at 1. + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 1); + + /// Stage TWO distinct transactions, each carrying the SAME removal event for r1, in ONE fold window + /// (r1's body is NOT deleted until recheck, so both events are resolved at fold time): + /// v2: DROP r1 {old=committed(r1), new=none} + /// v3: DUPLICATE DROP r1 {old=committed(r1), new=none} + /// The second removal of r1's edges is a no-op under the idempotent set model. + appendOwnerEvent(*backend, store->layout(), ns, 0, committed("tbl", r1), std::nullopt); + const uint64_t duplicate_removal_sequence + = appendOwnerEvent(*backend, store->layout(), ns, 0, committed("tbl", r1), std::nullopt); + advanceRecoverableCkptForRawFixture( + *backend, store->layout(), ns, RefTxnId{1, duplicate_removal_sequence}); + + /// Drive GC to fixpoint (advancing the mount ack each round so the ack floor graduates the condemned + /// blobs): must complete without throwing and collect both blobs. + ASSERT_NO_THROW({ + for (int i = 0; i < 12; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + }) << "H2 regression: duplicate removal of an already-removed committed ref must NOT underflow; " + << "the idempotent edge set absorbs the duplicate removal"; + + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0) + << "blob 1 is unreferenced (r1 dropped) and must be collected"; + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "blob 1 must be physically removed from the store"; + + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(2)), 0) + << "blob 2 is unreferenced (r1 dropped) and must be collected"; + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(2))) + << "blob 2 must be physically removed from the store"; +} + +/// ============================ H1: CURSOR RE-FOLD UNDER ABORT ============================ +/// +/// Hypothesis H1: a removal `-1` is folded, but the SINGLE round-commit CAS that would durably advance the +/// cursor past that removal LOSES to a concurrent leader (ABORTED). The cursor is NOT advanced, so a later +/// honest round RE-FOLDS the same removal against a parent generation whose in-degree for that blob has +/// already reached 0 => -1. +/// +/// We reproduce the deposed-round-commit injection from gtest_cas_gc_attempt.cpp +/// (DeposedFoldAttemptDoesNotWedge): deny the SINGLE round-commit gc/state CAS (the one that advances +/// snap_generation) of the round that folds the drop's -1. The deposed round left only never-adopted +/// attempt-scoped debris, so the retry re-folds the -1 against the still-adopted parent (in-degree 1), +/// producing a clean 0 — never a double-applied -1. +class InterruptRoundCasBackend : public InMemoryBackend +{ +public: + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (arm_interrupt && key == gc_state_key) + { + const auto stored = get(key); + const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; + const uint64_t next_gen = decodeGcState(bytes).snap_generation; + if (next_gen > stored_gen) + { + arm_interrupt = false; + throw DB::Exception(DB::ErrorCodes::ABORTED, + "test-injected: round-commit gc/state CAS denied (leader deposed mid-round; lease lost)"); + } + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + bool arm_interrupt = false; + String gc_state_key = "p/gc/state"; +}; + +TEST(CASGCUndercount, H1DrainAfterDeposedRemovalFoldDoesNotUnderflow) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + ASSERT_EQ(store->layout().gcStateKey(), "p/gc/state"); + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + /// Round 1 (honest): fold +1, pin blob 1. + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// Drop the only ref and advance the watermark floor. + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + store->renewWatermarkOnce(); + + /// Round 2 (DEPOSED): fold the -1, then the round-commit CAS is denied (ABORTED). The adopted + /// (snap_generation, snap_attempt) must NOT advance. + backend->arm_interrupt = true; + EXPECT_ANY_THROW(runRegularRoundReclaiming(gc)); + backend->arm_interrupt = false; + + /// Honest drive to fixpoint (advancing the mount ack each round). H1 predicts the re-fold of the -1 + /// underflows; the current code predicts a clean drain. Capture whichever happens. + bool threw_undercount = false; + try + { + for (int i = 0; i < 32; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + } + catch (const DB::Exception & e) + { + threw_undercount = (e.code() == DB::ErrorCodes::CORRUPTED_DATA + && e.message().find("merged in-degree -1 < 0") != String::npos); + if (!threw_undercount) + throw; + std::cerr << "H1 captured exception: " << e.message() << "\n"; + } + + if (threw_undercount) + { + FAIL() << "H1 REPRODUCED: the deposed removal fold underflowed on re-fold"; + } + else + { + /// H1 did NOT reproduce with a single deposed round: the drain is clean. + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "H1-not-reproduced: the pool drained cleanly (single deposed round is idempotent)"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); + } +} + +/// ==================== H1b: A CONCURRENT-DROP REMOVAL IS FOLDED ONCE (IDEMPOTENCE) ==================== +/// +/// The idempotence claim that survives the redesign, without the (retired) fence-window framing: a removal +/// that lands AFTER a round's fold sealed its cursor but BEFORE that round's single commit CAS must be +/// folded EXACTLY ONCE by a later round — never re-folded to drive the blob in-degree below zero. +/// +/// In the one-pass round there is a single gc/state CAS (fold -> publish -> commit). We inject the drop +/// just before that commit CAS lands, so the event (v2) is above the fold's sealed cursor (v1) this round. +/// The committed round adopts the fold seal at cursor v1; the next round folds (v1, v2] as an ordinary -1 +/// against the still-live parent (blob 1 at in-degree 1 => 0). The source-edge SET model makes a re-fold +/// of the same removal a set-difference no-op, so the in-degree never underflows. +class DropAtCommitBackend : public InMemoryBackend +{ +public: + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + /// The one-pass round has a SINGLE gc/state CAS that advances snap_generation. Fire the injected + /// drop ONCE, just before that CAS commits — so the drop event is above this round's sealed cursor. + if (arm_drop && key == gc_state_key) + { + const auto stored = get(key); + if (stored) + { + const GcState prev = decodeGcState(stored->bytes); + const GcState next = decodeGcState(bytes); + if (next.snap_generation > prev.snap_generation) + { + arm_drop = false; + if (on_commit) + on_commit(); + } + } + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + bool arm_drop = false; + String gc_state_key = "p/gc/state"; + std::function on_commit; +}; + +TEST(CASGCUndercount, H1bFenceWindowRemovalReFoldedNextRoundUnderflows) +{ + auto backend = std::make_shared(); + /// gc_fold_max_defer_rounds=0 forces fold-every-round: the injected drop fires from `on_commit`, + /// which only runs on the round-commit CAS that ADVANCES snap_generation. With immutable logs an idle + /// round DEFERS (never advancing the generation), so a default store would never fire the injection -- + /// forcing a fold each round keeps the fence-window injection point (and its re-fold) reachable. + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + ASSERT_EQ(store->layout().gcStateKey(), "p/gc/state"); + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + /// Round 1 (honest): fold +1, pin blob 1 at in-degree 1. Cursor sealed at v1. + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); + + /// Round 2: just before the round-commit CAS lands (after the fold sealed its cursor at v1), inject the + /// DROP as v2. The fold this round saw only up to v1 (no change), so the sealed cursor stays v1; the + /// drop event v2 is above it and survives trim. The NEXT round folds (v1, v2] => -1 on blob 1 against + /// the still-live parent (in-degree 1 => 0). It must NOT be re-folded a second time. + backend->arm_drop = true; + backend->on_commit = [&] + { + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + }; + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + backend->arm_drop = false; + + /// CORRECT behaviour: the concurrently-dropped blob is reclaimed exactly once and GC stays quiescent — + /// the removal folds ONCE (idempotent source-edge set), NEVER driving the in-degree below zero. Advance + /// the mount ack each round so the ack floor graduates and deletes the condemned blob. + EXPECT_NO_THROW({ + for (int i = 0; i < 12; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + }) << "undercount: a concurrent-drop removal was re-folded and drove the blob in-degree < 0"; + + EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) + << "the concurrently-dropped blob must be reclaimed"; + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +/// ============== UNRECOGNIZED owner_transition SHAPE ABORTS THE ROUND, NEVER DELETES ================= +/// +/// A decodable ref log whose `owner_transition` op is SHAPE-illegal (here: neither `old_binding` nor +/// `new_binding` set) is exactly what `classifyOwnerTransitionShape` (Pool/CasRefProtocol.cpp) throws +/// `CORRUPTED_DATA` on. `writeRefLogTxnRaw` -- the same codec real writers use -- never checks op-shape +/// legality at encode/decode time, so this body is perfectly decodable; only `manifestEdgesOfTxn`'s +/// shape classification rejects it, at GC fold time. +/// +/// `Gc::fold` extracts edges inside the SAME try-block as `decodeRefLogTxn` +/// (Gc/CasGc.cpp), so the throw gets the identical "ref log body invalid: ref folding aborted this +/// round" treatment as an undecodable body: no cursor advance for ANY table (not just the corrupt +/// one), no ref delta lands, and the recorded anomaly drives `suppress_destructive`, which gates OFF +/// every graduated/pending blob delete for the WHOLE round -- including a blob in a namespace the +/// corrupt log never touched. +TEST(CASGCUndercount, UnrecognizedOwnerTransitionShapeAbortsRoundNeverDeletes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const RootNamespace corrupt_ns{"00/bb@cas@"}; + const ManifestRef r1 = ref(1, 0xC1); + + writeBlobBody(*backend, store->layout(), DB::UInt128(9)); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(9))}); + + /// Publish r1 (pins blob 9) and drop it again -- an ordinary, LEGAL removal that, absent + /// corruption, condemns blob 9 and (over a few more rounds, matching H1/H2 above) physically + /// deletes it. + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r1); + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + ASSERT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(9)), 1); + dropRefTransition(*backend, store->layout(), ns, "tbl", r1); + + /// Drive rounds until blob 9 first reaches in-degree 0 (condemned) -- still physically present: + /// deletion is two-phase (a later round graduates it to `delete_pending`, a later round still + /// executes the delete), so a freshly condemned blob is never deleted in the same round. + bool condemned = false; + for (int i = 0; i < 12 && !condemned; ++i) + { + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + condemned = (inDegreeOf(*backend, store->layout(), DB::UInt128(9)) == 0); + } + ASSERT_TRUE(condemned) << "setup: blob 9 must reach in-degree 0 (condemned) before injecting corruption"; + ASSERT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(9))) + << "setup: a freshly condemned blob must still be physically present"; + + const uint64_t cursor_before = foldCursorOf(*backend, store->layout(), ns, /*shard*/0); + + /// A LEGAL, foldable log in `ns` ITSELF, staged AFTER capturing `cursor_before`. Without it, `ns` + /// has nothing new to fold, so the "ns cursor did not advance" assertion below is vacuous -- it + /// would pass even if the abort were per-table rather than round-wide. A duplicate remove-committed + /// of r1 is shape-legal and foldable (idempotent on the source-edge set, so it does not disturb blob + /// 9's already-condemned state), so absent the round-wide abort, folding `ns` WOULD advance its + /// cursor past this log -- making the pin below load-bearing. + appendOwnerEvent(*backend, store->layout(), ns, 0, committed("tbl", r1), std::nullopt); + + /// A decodable but SHAPE-illegal owner_transition (neither binding) in an UNRELATED table. + appendRefLogSeed(*backend, store->layout(), corrupt_ns, {ownerTransitionOp(std::nullopt, std::nullopt)}); + + /// Drive MANY more rounds with the corrupt log present. Absent corruption blob 9 -- already + /// condemned -- would graduate and be physically deleted within a handful more rounds (exactly + /// what H1/H2 above demonstrate for an equivalent drop). Every round here must instead: not throw, + /// leave every table's cursor exactly where it was (including `ns`, which the corrupt log never + /// touched -- ref-folding abort is round-wide, never per-table), record the anomaly, and never + /// physically delete blob 9. + for (int i = 0; i < 20; ++i) + { + RoundReport rep; + ASSERT_NO_THROW(rep = runRegularRoundReclaiming(gc)) + << "round " << i << ": an unrecognized owner_transition shape must abort ref folding, " + "never throw out of the round"; + store->renewWatermarkOnce(); + EXPECT_FALSE(rep.anomalies.empty()) + << "round " << i << ": the round must record the unrecognized-shape anomaly"; + EXPECT_EQ(foldCursorOf(*backend, store->layout(), ns, /*shard*/0), cursor_before) + << "round " << i << ": ns's cursor must not advance on a round whose ref folding aborted"; + ASSERT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(9))) + << "round " << i << ": a previously-eligible (condemned) blob must NOT be deleted while " + "ref folding is aborted -- destructive work is suppressed for the whole round"; + } +} diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp new file mode 100644 index 000000000000..c32b03e23074 --- /dev/null +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -0,0 +1,940 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; + extern const int ABORTED; +} + +using namespace DB::Cas; + + +/// MountLeaseKeeper behavior: the per-server mount lease and the merged build-watermark floor ride the +/// SAME slot, renewed by one beat. The keeper anchors durably before return, adopts a slot already +/// written by `claimMount` (same uuid+epoch), re-reads the callback on each renew and bumps `seq`, +/// stamps the farewell sentinel (`min_active = UINT64_MAX`, `expires_at_ms <= now`) on `release`, and +/// returns typed terminal results on any foreign touch. + +namespace +{ +/// The normal steady-state flow: `claimMount` writes the live (uuid, epoch) mount, THEN the keeper +/// adopts it. Seed that claim so `start` adopts instead of self-tripping the double-start guard. +void seedOwnClaim(Backend & b, const Layout & l, const String & srid, UInt128 uuid, uint64_t epoch, + uint64_t now_ms, uint64_t ttl_ms) +{ + ASSERT_EQ(claimMount(b, l, srid, uuid, epoch, now_ms, ttl_ms).kind, MountClaimResult::Claimed); +} + +class RenewalScriptBackend final : public InMemoryBackend +{ +public: + enum class Action : uint8_t + { + Delegate, + ThrowBefore, + LandThenThrow, + ReturnThenCancel, + ThrowBeforeThenLandAfterResolve, + }; + + struct Attempt + { + String key; + String bytes; + Token expected; + }; + + using InMemoryBackend::get; + using InMemoryBackend::putOverwrite; + + std::deque actions; + std::vector attempts; + std::function cancel_after_write; + uint64_t get_calls = 0; + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + attempts.push_back({key, bytes, expected}); + const Action action = actions.empty() ? Action::Delegate : actions.front(); + if (!actions.empty()) + actions.pop_front(); + + if (action == Action::ThrowBefore || action == Action::ThrowBeforeThenLandAfterResolve) + { + if (action == Action::ThrowBeforeThenLandAfterResolve) + pending = Attempt{key, bytes, expected}; + throw Poco::TimeoutException("injected renewal response uncertainty before a result"); + } + + PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + if (action == Action::LandThenThrow) + { + if (cancel_after_write) + cancel_after_write(); + throw Poco::TimeoutException("injected renewal response loss after commit"); + } + if (action == Action::ReturnThenCancel && cancel_after_write) + cancel_after_write(); + return result; + } + + std::optional get(const String & key, Range range) override + { + ++get_calls; + std::optional result = InMemoryBackend::get(key, range); + if (pending && pending->key == key) + { + const Attempt delayed = *pending; + pending.reset(); + const PutResult landed = InMemoryBackend::putOverwrite(delayed.key, delayed.bytes, delayed.expected, {}); + if (landed.outcome != PutOutcome::Done) + throw DB::Exception(DB::ErrorCodes::ABORTED, "injected delayed renewal did not land"); + } + return result; + } + +private: + std::optional pending; +}; + +CasRequestBudget renewalBudget(uint32_t max_attempts = 3) +{ + return CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = max_attempts, + .lease_safety_margin_ms = 20, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }; +} + +MountRenewOperationEnvironment renewalEnvironment( + uint64_t & boot_ms, + const std::function & stop_cause = {}) +{ + return MountRenewOperationEnvironment{ + .boot_ms = [&boot_ms] { return boot_ms; }, + .stop_cause = stop_cause ? stop_cause : [] { return CasOverwriteStopCause::Continue; }, + .wait_before_retry = [](uint64_t) { return true; }, + .observe = {}, + }; +} + +DB::Exception terminalException(const MountRenewResult & result) +{ + EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_NE(result.failure, nullptr); + try + { + std::rethrow_exception(result.failure); + } + catch (const DB::Exception & e) + { + return e; + } + catch (...) + { + ADD_FAILURE() << "terminal keeper failure was not a typed DB::Exception"; + } + return DB::Exception(DB::ErrorCodes::ABORTED, "missing terminal exception"); +} + +void renewKeeperOrThrow(MountLeaseKeeper & keeper) +{ + const MountRenewResult result = keeper.renew(renewalBudget(), MountRenewOperationEnvironment{}); + if (result.outcome == MountRenewOutcome::Terminal) + std::rethrow_exception(result.failure); + if (result.outcome != MountRenewOutcome::Committed) + throw DB::Exception(DB::ErrorCodes::ABORTED, "keeper renewal was not attempted"); +} +} + +TEST(CASHeartbeat, AnchorCarriesFloor) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t min_active_now = 5; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [&] { return min_active_now; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + auto hr = backend->head(layout.mountKey(srid)); + ASSERT_TRUE(hr.exists); + auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); + EXPECT_EQ(m.writer_epoch, 9u); + EXPECT_EQ(m.min_active, 5u); + EXPECT_EQ(m.seq, 1u); + EXPECT_FALSE(m.gc_fenced); +} + +TEST(CASHeartbeat, RenewRereadsCallbackAndBumpsSeq) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t min_active_now = 5; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [&] { return min_active_now; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + /// The dynamic field moves; the renewal re-reads it off the callback and bumps seq. + now_ms = 1500; + min_active_now = 8; + renewKeeperOrThrow(keeper); + + auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); + EXPECT_EQ(m.min_active, 8u); + EXPECT_EQ(m.seq, 2u); + EXPECT_EQ(m.expires_at_ms, 1500u + 100u); +} + +TEST(CASHeartbeat, StopStampsExpiredAndFarewellSentinel) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + now_ms = 2000; + keeper.release(); + + auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); + /// Terminal body stamps the lease already-expired (so a same-server reopen reclaims immediately) + /// AND folds the watermark farewell into it (min_active = UINT64_MAX). + EXPECT_LE(m.expires_at_ms, now_ms); + EXPECT_EQ(m.min_active, std::numeric_limits::max()); +} + +/// Phase A (spec rev.4 2026-07-24): a confirmed renewal mismatch whose re-read shows OUR OWN +/// (uuid, epoch), unfenced, is state UNCERTAINTY (an ambiguous landed renewal of ours, or a +/// same-pair twin after epoch-state loss) — fail closed via fence + self-remount, never an +/// exception that aborts debug/ASan builds at construction. +TEST(CASHeartbeat, SameEpochUnfencedTouchIsUncertainNotFatal) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + /// The slot advances past our held token under our own pair (the ambiguous-landed-renewal shape). + const HeadResult h = backend->head(layout.mountKey(srid)); + ASSERT_TRUE(h.exists); + MountLease advanced; + advanced.server_uuid = uuid; + advanced.writer_epoch = 9; + advanced.seq = 99; + advanced.write_attempt_id = UInt128{99}; + backend->putOverwrite(layout.mountKey(srid), encodeMountLease(advanced), h.token); + + try + { + renewKeeperOrThrow(keeper); + FAIL() << "renew must return a terminal conflict"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED) << e.message(); + EXPECT_NE(e.message().find("state uncertain"), String::npos) << e.message(); + /// Forensics must ride in the message: the observed seq and our local seq. + EXPECT_NE(e.message().find("seq=99"), String::npos) << e.message(); + /// The local-seq fragment specifically -- not just any "seq=99" substring (which the + /// OBSERVED holder's own describeMountHolder text could also satisfy on its own). + EXPECT_NE(e.message().find("vs our seq="), String::npos) << e.message(); + } +} + +/// A body under our own uuid but a NEWER writer_epoch is proven supersession — a normal fencing +/// outcome (the TLA model's localLost), fail closed but never an abort. +TEST(CASHeartbeat, SupersededTouchIsFailClosedNotFatal) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + const HeadResult h = backend->head(layout.mountKey(srid)); + ASSERT_TRUE(h.exists); + MountLease successor; + successor.server_uuid = uuid; + successor.writer_epoch = 10; + successor.seq = 1; + successor.write_attempt_id = UInt128{1}; + backend->putOverwrite(layout.mountKey(srid), encodeMountLease(successor), h.token); + + try + { + renewKeeperOrThrow(keeper); + FAIL() << "renew must return a terminal conflict"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED) << e.message(); + EXPECT_NE(e.message().find("superseded by a newer incarnation"), String::npos) << e.message(); + } +} + +/// A foreign server holding our mount slot must FAIL CLOSED — and must not take the process with it. +/// +/// This test used to be `ForeignUuidTouchStillDies`, an `EXPECT_DEATH` that pinned the abort. The abort +/// was the defect: the arm raised `LOGICAL_ERROR`, which aborts at CONSTRUCTION in debug/ASan builds, +/// and the runtime consumes it on its renewal worker — so an environment-reachable condition (clear the +/// prefix, recreate under a different server id, and the survivor's next renewal lands there; see +/// `CASRefContiguousAlloc.SurvivingWriterIsFencedByTheRecreatedPoolsMount`, which drives exactly that) +/// took the whole server down, and took the ASan gate down with it. +/// +/// What must NOT change is the outcome, which is what this test now pins: synchronous renewal returns +/// a terminal failure that, when propagated, throws; the exception +/// carries the foreign holder's identity, and it is classified `ABORTED` — the same mount-lost class the +/// sibling fencing arms use, which the runtime terminal consumer turns into a latched write fence. The +/// `abort_on_logical_error` arming is deliberately kept: with it ON, a `LOGICAL_ERROR` would still abort, +/// so reaching the `EXPECT_THROW` at all is the proof that this condition is no longer classified as one. +TEST(CASHeartbeat, ForeignUuidTouchFailsClosedWithoutAborting) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + + const HeadResult h = backend->head(layout.mountKey(srid)); + ASSERT_TRUE(h.exists); + MountLease foreign; + foreign.server_uuid = UInt128(0x9999); + foreign.writer_epoch = 1; + foreign.seq = 1; + foreign.write_attempt_id = UInt128{1}; + backend->putOverwrite(layout.mountKey(srid), encodeMountLease(foreign), h.token); + + /// Restored on every exit: this flag is process-global and every later test in this binary would + /// inherit it. + const bool armed_before = DB::abort_on_logical_error.load(std::memory_order_relaxed); + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + SCOPE_EXIT({ DB::abort_on_logical_error.store(armed_before, std::memory_order_relaxed); }); + + String message; + int code = 0; + try + { + renewKeeperOrThrow(keeper); + FAIL() << "a foreign holder must fail the renewal closed, not be silently taken over"; + } + catch (const DB::Exception & e) + { + message = e.message(); + code = e.code(); + } + EXPECT_NE(message.find("held by a foreign server"), String::npos) << message; + EXPECT_EQ(code, DB::ErrorCodes::ABORTED) + << "the mount-lost class the runtime terminal consumer latches the write fence on -- and, critically, not " + "LOGICAL_ERROR, which would abort the renewal worker and the whole process with it"; +} + +/// Mount-slot writer audit (the P1 "foreign writer" instrument): every mount-slot WRITE and every +/// OBSERVED foreign/conflicting body becomes an event, carrying the conflicting body's identity — +/// the payload the chronic "touched by a foreign writer" collisions need to be diagnosable. +TEST(CASMountAudit, ClaimReleaseAndForeignConflictEmitEvents) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + std::vector seen; + CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; + + const uint64_t now_ms = 1'000'000; + /// mint for uuid 1 -> one mount_claim + ASSERT_EQ(claimMount(*backend, layout, "a", UInt128{1}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink).kind, + MountClaimResult::Claimed); + ASSERT_EQ(seen.size(), 1u); + EXPECT_EQ(seen[0].type, CasEventType::MountClaim); + EXPECT_EQ(seen[0].detail.at("server_root_id"), "a"); + EXPECT_EQ(seen[0].detail.at("branch"), "mint"); + + /// a FOREIGN uuid claiming a live slot -> mount_conflict carrying the current holder's identity + seen.clear(); + (void)claimMount(*backend, layout, "a", UInt128{2}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink); + ASSERT_FALSE(seen.empty()); + EXPECT_EQ(seen.back().type, CasEventType::MountConflict); + EXPECT_EQ(seen.back().detail.at("server_root_id"), "a"); + /// The conflict must carry the ORIGINAL holder's identity (uuid 1, the minter) — not the + /// foreign claimer's (uuid 2). + EXPECT_EQ(seen.back().detail.at("holder_uuid"), u128ToHex(UInt128{1})); + EXPECT_NE(seen.back().detail.at("holder_uuid"), u128ToHex(UInt128{2})); +} + +/// The MountLeaseKeeper wiring: `start` adopting an already-claimed slot emits mount_claim, `stop` +/// (the farewell write) emits mount_release. +TEST(CASMountAudit, KeeperAdoptEmitsClaimAndTerminateEmitsRelease) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + std::vector seen; + CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0)); + keeper.start(); + + ASSERT_EQ(seen.size(), 1u); + EXPECT_EQ(seen[0].type, CasEventType::MountClaim); + EXPECT_EQ(seen[0].detail.at("branch"), "adopt"); + + seen.clear(); + now_ms = 2000; + keeper.release(); + + ASSERT_EQ(seen.size(), 1u); + EXPECT_EQ(seen[0].type, CasEventType::MountRelease); + EXPECT_EQ(seen[0].detail.at("branch"), "farewell"); +} + +/// Keeper-level foreign-conflict refusal: the mount slot is already held by a FOREIGN uuid (X) when +/// a keeper for a DIFFERENT uuid (Y) tries to claim it. This must fail closed and — since the +/// mount-audit sink is not yet installed at first-open — name X in the exception's message text +/// (the only identity carrier in err.log at that point). MountConflict payload coverage is above. +TEST(CASMountAudit, KeeperForeignConflictRefusesAndNamesHolder) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid_x(0x1111); + const UInt128 uuid_y(0x2222); + uint64_t now_ms = 1000; + + /// Foreign holder X claims the slot first. + ASSERT_EQ(claimMount(*backend, layout, srid, uuid_x, /*our_epoch=*/1, now_ms, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + + MountLeaseKeeper keeper(backend, layout, srid, uuid_y, /*writer_epoch=*/1, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }); + + /// The enriched refusal message must name the OBSERVED holder (X), not the caller (Y). + const String holder_uuid = u128ToHex(uuid_x); + DB::Cas::tests::expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + holder_uuid, + [&] { keeper.start(); }); +} + +/// `Pool::open` can fail before/inside `doStart` (e.g. a foreign-conflict refusal, see +/// `KeeperForeignConflictRefusesAndNamesHolder` above) — the keeper is destroyed without ever having +/// claimed anything. Teardown must not throw "release before start"; there is nothing to release. A +/// stop AFTER a successful start still performs the farewell (covered by +/// `StopStampsExpiredAndFarewellSentinel` above); a genuinely-started DOUBLE terminate stays loud. +TEST(CASMountAudit, KeeperAdoptRefusesFencedSelfWithTypedError) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + + /// mint (uuid, epoch 9), then fence it in place (what computeHeartbeatFloor does on expiry): + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + { + auto got = backend->get(layout.mountKey(srid)); + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + fenced.seq += 1; + ASSERT_EQ(backend->putOverwrite(layout.mountKey(srid), encodeMountLease(fenced), got->token).outcome, + PutOutcome::Done); + } + + std::vector seen; + CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; + /// A keeper for the SAME (uuid, epoch) tries to adopt the now-fenced slot. + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, sink); + + bool threw = false; + try + { + keeper.start(); + } + catch (const MountFencedException & e) + { + threw = true; + EXPECT_NE(e.message().find("fenced by GC"), String::npos) << e.message(); + EXPECT_EQ(e.message().find("foreign writer"), String::npos) << e.message(); + } + EXPECT_TRUE(threw); + + ASSERT_FALSE(seen.empty()); + EXPECT_EQ(seen.back().type, CasEventType::MountConflict); + EXPECT_EQ(seen.back().detail.at("branch"), "fenced_by_gc"); +} + +/// A renew mismatch is classified by BODY, not blamed on "a foreign writer" by default: the GC can +/// fence our OWN (uuid, epoch) mount slot after our lease expires (a late renewal beat racing the +/// GC's fence-out). The keeper must re-read and recognize this as its OWN incarnation being fenced — +/// a recoverable `MountFencedException`, not the generic single-writer-violation text. +TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + std::vector seen; + CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; + MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), + [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0)); + keeper.start(); + seen.clear(); + + /// Mid-run: the GC fences our own (uuid, epoch) mount slot in place (as `computeHeartbeatFloor` + /// does on an expired lease), preserving the whole body — a token-guarded putOverwrite, exactly + /// as the GC's own fence-out does it. + { + const auto got = backend->get(layout.mountKey(srid)); + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + fenced.seq += 1; + ASSERT_EQ(backend->putOverwrite(layout.mountKey(srid), encodeMountLease(fenced), got->token).outcome, + PutOutcome::Done); + } + + /// The renewal must classify the fence honestly — not "foreign writer": + try + { + renewKeeperOrThrow(keeper); + FAIL() << "renew over a fenced slot must be terminal"; + } + catch (const MountFencedException & e) + { + EXPECT_TRUE(e.message().find("fenced by GC") != String::npos); + EXPECT_TRUE(e.message().find("foreign writer") == String::npos); + } + /// and the capture sink saw mount_conflict branch=fenced_by_gc with the fenced body's identity. + ASSERT_FALSE(seen.empty()); + EXPECT_EQ(seen.back().type, CasEventType::MountConflict); + EXPECT_EQ(seen.back().detail.at("branch"), "fenced_by_gc"); + EXPECT_EQ(seen.back().detail.at("holder_uuid"), u128ToHex(uuid)); +} + +TEST(CASHeartbeat, KeeperStateAllowsOnlyActiveReleaseOrTerminal) +{ +#if defined(DEBUG_OR_SANITIZER_BUILD) +#define EXPECT_KEEPER_STATE_REJECTION(statement) EXPECT_DEATH({ statement; }, "allowed only in") +#else +#define EXPECT_KEEPER_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) +#endif + + Layout layout("pool"); + const UInt128 uuid{0x1234}; + + { + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "released", uuid, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "released", uuid, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::New); + EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); + EXPECT_KEEPER_STATE_REJECTION(keeper.release()); + EXPECT_EQ(keeper.start(), 100u); + EXPECT_KEEPER_STATE_REJECTION(keeper.start()); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Active); + keeper.release(); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Released); + EXPECT_KEEPER_STATE_REJECTION(keeper.start()); + EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); + EXPECT_KEEPER_STATE_REJECTION(keeper.release()); + } + + { + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "terminal", uuid, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; + const MountRenewResult result = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_NE(result.failure, nullptr); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); + EXPECT_KEEPER_STATE_REJECTION(keeper.start()); + EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); + EXPECT_KEEPER_STATE_REJECTION(keeper.release()); + } + +#undef EXPECT_KEEPER_STATE_REJECTION +} + +TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid{0x1234}; + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, srid, uuid, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, srid, uuid, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + + backend->attempts.clear(); + backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::Delegate}; + MountRenewResult retried = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + ASSERT_EQ(retried.outcome, MountRenewOutcome::Committed); + ASSERT_EQ(backend->attempts.size(), 2u); + EXPECT_EQ(backend->attempts[0].key, backend->attempts[1].key); + EXPECT_EQ(backend->attempts[0].bytes, backend->attempts[1].bytes); + EXPECT_EQ(backend->attempts[0].expected, backend->attempts[1].expected); + const MountLease retry_body = decodeMountLease(backend->attempts[0].bytes); + EXPECT_NE(retry_body.write_attempt_id, UInt128{}); + + backend->attempts.clear(); + backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; + MountRenewResult adopted = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + EXPECT_EQ(adopted.outcome, MountRenewOutcome::Committed); + EXPECT_TRUE(adopted.diagnostics.resolved_by_get); + EXPECT_EQ(adopted.diagnostics.attempts_sent, 1u); + EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey(srid))->bytes).write_attempt_id, + decodeMountLease(backend->attempts.front().bytes).write_attempt_id); +} + +TEST(CASHeartbeat, DeadlineBeforeSendTerminalizesWithTypedFailure) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 100); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->attempts.clear(); + backend->get_calls = 0; + boot_ms = 180; + const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const DB::Exception failure = terminalException(result); + EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(failure.message().find("no attempt was sent"), String::npos) << failure.message(); + EXPECT_EQ(result.diagnostics.unresolved_reason, CasUnresolvedReason::NoAttemptSent); + EXPECT_TRUE(backend->attempts.empty()); + EXPECT_EQ(backend->get_calls, 0u) << "a pre-send terminal deadline must perform no diagnostic GET"; +} + +TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->attempts.clear(); + backend->get_calls = 0; + const auto cancelled = [] { return CasOverwriteStopCause::Cancelled; }; + const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms, cancelled)); + EXPECT_EQ(result.outcome, MountRenewOutcome::NotAttempted); + EXPECT_EQ(result.failure, nullptr); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Active); + EXPECT_TRUE(backend->attempts.empty()); + EXPECT_NO_THROW(keeper.release()); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Released); +} + +TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + bool cancelled = false; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->attempts.clear(); + backend->get_calls = 0; + backend->cancel_after_write = [&] { cancelled = true; }; + backend->actions = {RenewalScriptBackend::Action::ReturnThenCancel}; + const MountRenewResult result = keeper.renew( + renewalBudget(), renewalEnvironment(boot_ms, [&] { + return cancelled ? CasOverwriteStopCause::Cancelled : CasOverwriteStopCause::Continue; + })); + const DB::Exception failure = terminalException(result); + EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_EQ(result.diagnostics.unresolved_reason, CasUnresolvedReason::FenceLostPostWrite); + EXPECT_EQ(backend->get_calls, 0u) << "post-write cancellation must not start a diagnostic GET"; + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); + const String bytes_before = backend->get(layout.mountKey("test"))->bytes; + EXPECT_FALSE(keeper.canRelease()); + EXPECT_EQ(backend->get(layout.mountKey("test"))->bytes, bytes_before); +} + +TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + boot_ms = 150; + backend->cancel_after_write = [&] { boot_ms = 400; }; + backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; + const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_EQ(result.attempt_start_boot_ms, 150u); + EXPECT_EQ(keeper.lastCommittedAttemptStartBootMs(), 150u); +} + +TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) +{ + const auto run_case = [](UInt128 current_uuid, uint64_t current_epoch, UInt128 current_attempt) + { + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + seedOwnClaim(*backend, layout, "test", uuid, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", uuid, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + auto got = backend->get(layout.mountKey("test")); + MountLease current = decodeMountLease(got->bytes); + current.server_uuid = current_uuid; + current.writer_epoch = current_epoch; + current.write_attempt_id = current_attempt; + ++current.seq; + ASSERT_EQ(backend->putOverwrite(layout.mountKey("test"), encodeMountLease(current), got->token).outcome, + PutOutcome::Done); + backend->get_calls = 0; + const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const DB::Exception failure = terminalException(result); + EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); + EXPECT_EQ(backend->get_calls, 1u) << "the controller's resolving GET must be the only terminal read"; + }; + + run_case(UInt128{1}, 9, UInt128{0xAAAA}); + run_case(UInt128{2}, 9, UInt128{0xBBBB}); + run_case(UInt128{1}, 10, UInt128{0xCCCC}); +} + +TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->attempts.clear(); + backend->actions = { + RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve, + RenewalScriptBackend::Action::Delegate, + }; + const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_TRUE(result.diagnostics.resolved_by_get); + ASSERT_EQ(backend->attempts.size(), 2u); + EXPECT_EQ(backend->attempts[0].bytes, backend->attempts[1].bytes); + EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey("test"))->bytes).write_attempt_id, + decodeMountLease(backend->attempts[0].bytes).write_attempt_id); +} + +TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) +{ + const auto run_case = [](bool vanish) + { + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + const String key = layout.mountKey("test"); + auto got = backend->get(key); + if (vanish) + ASSERT_EQ(backend->deleteExact(key, got->token).kind, DeleteOutcome::Kind::Deleted); + else + { + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + ++fenced.seq; + ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), got->token).outcome, PutOutcome::Done); + } + const DB::Exception failure = terminalException(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); + EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); + EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); + }; + run_case(false); + run_case(true); +} + +TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) +{ + const auto make_terminal = [](const std::shared_ptr & backend, + const Layout & layout, const String & srid, + uint64_t & wall_ms, uint64_t & boot_ms) + { + seedOwnClaim(*backend, layout, srid, UInt128{1}, 9, wall_ms, 1000); + auto keeper = std::make_unique( + backend, layout, srid, UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper->start(); + backend->actions = {RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve}; + const MountRenewResult result = keeper->renew(renewalBudget(1), renewalEnvironment(boot_ms)); + EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_EQ(keeper->state(), MountLeaseKeeperState::RenewalTerminal); + return keeper; + }; + + Layout layout("pool"); + { + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + auto keeper = make_terminal(backend, layout, "before-reclaim", wall_ms, boot_ms); + const MountLease landed = decodeMountLease(backend->get(layout.mountKey("before-reclaim"))->bytes); + EXPECT_EQ(landed.writer_epoch, 9u); + EXPECT_EQ(keeper->state(), MountLeaseKeeperState::RenewalTerminal); + } + { + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "after-successor", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; + const MountRenewResult result = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); + ASSERT_FALSE(backend->attempts.empty()); + const auto delayed = backend->attempts.back(); + auto current = backend->get(delayed.key); + MountLease fenced = decodeMountLease(current->bytes); + fenced.gc_fenced = true; + ++fenced.seq; + ASSERT_EQ(backend->InMemoryBackend::putOverwrite(delayed.key, encodeMountLease(fenced), current->token, {}).outcome, + PutOutcome::Done); + ASSERT_EQ(claimMount(*backend, layout, "after-successor", UInt128{1}, 10, wall_ms, 1000).kind, + MountClaimResult::Claimed); + MountLeaseKeeper successor( + backend, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + successor.start(); + EXPECT_EQ(backend->InMemoryBackend::putOverwrite(delayed.key, delayed.bytes, delayed.expected, {}).outcome, + PutOutcome::PreconditionFailed); + EXPECT_EQ(decodeMountLease(backend->get(delayed.key)->bytes).writer_epoch, 10u); + } +} + +TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseKeeper keeper( + backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), + [&] { return boot_ms; }); + keeper.start(); + + wall_ms = 9'000'000; + EXPECT_EQ(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + wall_ms = 1; + EXPECT_EQ(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + + backend->attempts.clear(); + boot_ms += 10'000; + const MountRenewResult suspended = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const DB::Exception failure = terminalException(suspended); + EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_TRUE(backend->attempts.empty()) << "suspend-sized BOOTTIME overshoot must close admission"; +} diff --git a/src/Disks/tests/gtest_cas_holey_list_detector.cpp b/src/Disks/tests/gtest_cas_holey_list_detector.cpp new file mode 100644 index 000000000000..6c1728d3ba31 --- /dev/null +++ b/src/Disks/tests/gtest_cas_holey_list_detector.cpp @@ -0,0 +1,306 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +/// SKIPPED-TRANSACTION suite. The defect class: a GC round's fold cursor advances past a ref +/// transaction the round never applied. Once the cursor is sealed above a record, that record can +/// never be folded again, so BOTH directions of the damage are permanent: +/// +/// - RETENTION: a skipped `-1` leaves a residual `+1`, so the blob is never reclaimed (a leak); +/// - DELETION: a skipped `+1` hides a live owner, so GC deletes a blob a committed manifest still +/// references (data loss). +/// +/// The suspected MECHANISM (a `LIST` page that omits a durable key) is UNCONFIRMED — a holey page was +/// never directly observed, it survives by elimination, and `CaRelinkConfirmCore.tla` `_sab_holeylist` +/// proves the mechanism is SUFFICIENT, not that it is what happened. These tests therefore use the +/// holey listing only as the cheapest way to make the EFFECT executable; nothing here may key on how +/// the hole was produced. + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// A backend that drops ONE chosen key from ONE chosen `list` call, while leaving exact `get`/`head` +/// of that key working. This is the minimal realisation of "the store returned an incomplete answer": +/// the record is durable and readable, it is simply absent from one enumeration. The mechanism is +/// deliberately NOT modelled (no page split, no cursor games) — the arithmetic intake under test must +/// not depend on how the hole was produced. +/// +/// WHICH call is explicit and load-bearing. A GC round enumerates the ref prefix ONCE, in +/// `Gc::listRefPrefix`, and the fold regroups that same enumeration -- so `nth = 0` is the walk whose +/// hole the fold would have to survive, and it is the one every test here arms. `nth` counts, from the +/// moment `omitFromNthListCall` is called, only those `list` calls that WOULD have returned the key — so +/// unrelated prefix enumerations do not shift it. +/// Arm the sabotage AFTER every seeding write: the writer's own sequence allocation lists the +/// namespace prefix and would otherwise consume a qualifying call. +/// +/// Erasing a key from the page never disturbs pagination: `ListPage::next_cursor` is the LAST key the +/// underlying backend returned and is computed before the erase, so the next page still resumes +/// strictly after it. +class HoleyListBackend : public InMemoryBackend +{ +public: + /// Omit `key` from the `nth` (0-based) subsequent qualifying `list` call. Resets the counter. + void omitFromNthListCall(const String & key, size_t nth) + { + std::lock_guard lock(m); + omitted = key; + target_call = nth; + seen_calls = 0; + served = false; + } + + /// Whether the hole was actually served. Every test asserts this, so a mis-typed key or a + /// miscounted `nth` cannot let a test pass vacuously. + bool holeServed() const + { + std::lock_guard lock(m); + return served; + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + std::lock_guard lock(m); + if (omitted.empty()) + return page; + auto it = std::find_if(page.keys.begin(), page.keys.end(), + [&](const ListedKey & k) { return k.key == omitted; }); + if (it == page.keys.end()) + return page; /// not a qualifying call — do not count it + if (seen_calls++ != target_call) + return page; + page.keys.erase(it); + served = true; + omitted.clear(); /// one hole only + return page; + } + +private: + mutable std::mutex m; + String omitted; + size_t target_call = 0; + size_t seen_calls = 0; + bool served = false; +}; + +PoolPtr openHoleyPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// A `ManifestEntry` for a Blob leaf at `path` referencing `payload`'s content hash. +ManifestEntry blobEntry(const String & path, const String & payload) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}; + e.blob_size = payload.size(); + return e; +} + +/// Publish one single-blob part through the REAL writer sequence and return its `ManifestId`. +ManifestId publishOneBlobPart(const PoolPtr & s, const RootNamespace & ns, const String & ref, + const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({blobEntry("data.bin", payload)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +bool blobPresent(const std::shared_ptr & b, const Layout & layout, const String & payload) +{ + return b->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, + BlobDigest::fromU128(u128Of(payload))})).exists; +} + +/// Every ref object key of one namespace. Used to identify WHICH objects a publish appended, rather +/// than guessing a sequence number. +std::set listRefKeys(Backend & b, const Layout & layout, const RootNamespace & ns) +{ + /// Stage B (Task 4-C): `ns` is born through the REAL append lane here, so its objects sit at a + /// real catalog-minted incarnation, not the Stage-A sentinel. + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(b, layout, ns).value(); + std::set keys; + forEachListedKey(b, layout.namespaceStreamPrefix(life), [&](const ListedKey & k) { keys.insert(k.key); }); + return keys; +} + +/// The keys present in `after` and not in `before`. +std::vector addedKeys(const std::set & before, const std::set & after) +{ + std::vector added; + std::set_difference(after.begin(), after.end(), before.begin(), before.end(), + std::back_inserter(added)); + return added; +} + +/// Among `candidates`, the ONE ref-log key whose transaction emits an edge of sign `change` naming +/// `manifest_id`. +/// +/// Selecting the key by DECODING is load-bearing. One logical publish appends SEVERAL ref-log +/// transactions (`precommitAdd`, then `promote`), and it is the `precommitAdd` that carries the `+1` +/// activation — a promote is an owner move at the same `manifest_ref` and emits no edge at all. Picking +/// "the greatest new key" would therefore omit the wrong object and the sabotage would be a no-op that +/// still let the test pass. +String refLogKeyEmittingEdge(Backend & b, const Layout & layout, const RootNamespace & ns, + const std::vector & candidates, const ManifestId & manifest_id, + int change) +{ + std::vector hits; + for (const String & key : candidates) + { + const auto parsed = layout.parseRefObjectKey(key); + if (!parsed || parsed->kind != RefObjectKind::Log) + continue; + const auto got = b.get(key); + if (!got) + continue; + const RefLogTxn txn = + decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), parsed->txn_id); + for (const RefManifestEdge & e : manifestEdgesOfTxn(txn)) + if (e.change == change && e.manifest_id == manifest_id) + { + hits.push_back(key); + break; + } + } + EXPECT_EQ(hits.size(), 1u) << "expected exactly one ref log emitting a " << change + << " edge for the manifest, found " << hits.size(); + return hits.empty() ? String{} : hits.front(); +} + +void runRounds(const PoolPtr & s, Gc & gc, int rounds) +{ + for (int i = 0; i < rounds; ++i) + { + DB::Cas::tests::runRegularRoundReclaiming(gc); + s->renewWatermarkOnce(); + } +} + +} + +/// RETENTION DIRECTION (the RCA's primary reproduction). A ref-log record omitted from a listing used to +/// sort at or below the cursor forever, so restoring the listing could not recover it: the blob's `-1` +/// never folded and the blob was retained permanently. Under arithmetic intake the omitted record is +/// reached by exact key on the very round that was lied to, so the removal folds and the blob dies on +/// the normal schedule — no abort, and no waiting for the store to become honest again. +TEST(CASHoleyListDetector, OmittedRemoveRecordIsSkippedForever) +{ + std::shared_ptr b; + auto s = openHoleyPool(b); + const Layout & layout = s->layout(); + const RootNamespace ns{"test/tbl"}; + const String payload = "holey-payload"; + + /// A: publish the part (its `+1` edges). Folded by the rounds below. + const ManifestId part = publishOneBlobPart(s, ns, "part_a", payload); + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runRounds(s, gc, 2); + ASSERT_TRUE(blobPresent(b, layout, payload)); + + /// R: drop the ref (the `-1`). Through the REAL writer API, never the raw ref-log helper: the raw + /// helper allocates a sequence by listing, which collides with the ledger's own in-memory sequence + /// as soon as the same namespace is written through the writer again (`part_h` below). + const std::set before_drop = listRefKeys(*b, layout, ns); + s->dropRef(ns, "part_a"); + const std::set after_drop = listRefKeys(*b, layout, ns); + const String remove_key = + refLogKeyEmittingEdge(*b, layout, ns, addedKeys(before_drop, after_drop), part, -1); + ASSERT_FALSE(remove_key.empty()); + + /// H: a later, unrelated record so the cursor has a reason to advance past R even when R is not + /// returned. + publishOneBlobPart(s, ns, "part_h", "harmless-payload"); + s->renewWatermarkOnce(); /// advance the floor so the dropped closure is not spared as in-flight + + /// nth = 0: the round's own enumeration of the ref prefix — the one the fold regroups, and the only + /// walk whose hole the intake has to survive. Armed LAST, after every seeding write, so no + /// writer-side namespace listing consumes a qualifying call. + b->omitFromNthListCall(remove_key, /*nth=*/0); + + runRounds(s, gc, 1); + ASSERT_TRUE(b->holeServed()) << "the sabotage never fired — the omitted key was never listed"; + + /// Drive to the reclaim. The point is that the FIRST of these rounds — the one served the hole — + /// already folded the removal; the rest are the condemn/graduate/delete pacing. + runRounds(s, gc, 12); + + EXPECT_FALSE(blobPresent(b, layout, payload)) + << "the removal was hidden from one enumeration and never folded — the cursor advanced past a " + "record the round never applied, which is the skipped-transaction defect itself"; +} + +/// DELETION DIRECTION (the mirror safety test from the RCA). Two owners share ONE deduplicated blob. +/// The SECOND owner's `+1` is omitted from one listing while the FIRST owner's `-1` folds normally, +/// so GC sees zero edges for a blob a live manifest still references. THIS MUST NEVER DELETE THE BLOB. +TEST(CASHoleyListDetector, OmittedActivationNeverPermitsDeletingALiveBlob) +{ + std::shared_ptr b; + auto s = openHoleyPool(b); + const Layout & layout = s->layout(); + const RootNamespace ns{"test/tbl"}; + const String payload = "shared-payload"; + + /// M1 owns the token. Fold it so its `+1` is durable in the in-degree generation. + const ManifestId m1 = publishOneBlobPart(s, ns, "part_1", payload); + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runRounds(s, gc, 2); + ASSERT_TRUE(blobPresent(b, layout, payload)); + ASSERT_TRUE(b->head(layout.manifestKey(m1)).exists) + << "M1's body must still be present so its `-1` edges are readable at removal-fold"; + + /// M2 adopts the SAME deduplicated blob (`putBlob` of an identical payload dedups). Learn WHICH + /// ref-log object carries M2's ACTIVATION by diffing the namespace's ref prefix around the publish + /// and decoding the new objects — do NOT guess a sequence number and do not append a probe + /// transaction (that would perturb the very stream under test). + const std::set before = listRefKeys(*b, layout, ns); + const ManifestId m2 = publishOneBlobPart(s, ns, "part_2", payload); + const std::set after = listRefKeys(*b, layout, ns); + const String m2_key = refLogKeyEmittingEdge(*b, layout, ns, addedKeys(before, after), m2, +1); + ASSERT_FALSE(m2_key.empty()); + + /// M1's removal folds normally. Through the REAL writer API (see the retention test's note). + s->dropRef(ns, "part_1"); + s->renewWatermarkOnce(); /// advance the floor so the removed closure is not spared as in-flight + + /// nth = 0: the round's own walk (see the note in the retention test). Armed LAST so the writer's + /// own namespace listings cannot shift the count. + b->omitFromNthListCall(m2_key, /*nth=*/0); + + runRounds(s, gc, 12); /// condemn -> graduate -> delete needs several rounds + /// The anti-vacuity check, and it is the RIGHT one now: a run that merely happened not to delete the + /// blob must not pass for the wrong reason, and what makes this run non-trivial is that the hole was + /// actually SERVED to the enumeration the fold works from. (It used to be "and the detector fired", + /// which was only ever a proxy for that — and is now a property of a different, sampled mechanism, + /// pinned in `CASRetirementSweep`.) + ASSERT_TRUE(b->holeServed()) << "the sabotage never fired — the omitted key was never listed"; + + EXPECT_TRUE(blobPresent(b, layout, payload)) + << "GC deleted a blob that manifest " << manifestRefDebugString(m2.ref) + << " still references — the skipped-transaction DATA-LOSS class, reproduced"; +} diff --git a/src/Disks/tests/gtest_cas_ids.cpp b/src/Disks/tests/gtest_cas_ids.cpp new file mode 100644 index 000000000000..046696124bd3 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ids.cpp @@ -0,0 +1,41 @@ +#include +#include +#include + +using namespace DB::Cas; + +TEST(CASIds, StrongTypingAndContainers) +{ + /// Test the strong-typed-string class `RootNamespace`. + /// (`BlobId` was deleted in the mixed-algo-pools refactor; `TreeId` was part of the + /// standalone-tree layer excised in the rev. 15 `PartManifest` redesign.) + RootNamespace ns1{"srv1"}; + RootNamespace ns2{"srv1"}; + RootNamespace ns3{"srv2"}; + EXPECT_EQ(ns1, ns2); + EXPECT_NE(ns1, ns3); + std::unordered_set s{ns1, ns3}; + EXPECT_EQ(s.size(), 2u); +} + +TEST(CASIds, HexU128RoundTrip) +{ + // UInt128 is a global typedef (wide::integer<128,unsigned>), not in DB:: namespace. + const UInt128 v = (UInt128(0x0123456789abcdefULL) << 64) | 0xfedcba9876543210ULL; + const auto hex = u128ToHex(v); + EXPECT_EQ(hex.size(), 32u); + EXPECT_EQ(hexToU128(hex), v); + EXPECT_THROW(hexToU128("zz"), DB::Exception); // not hex + EXPECT_THROW(hexToU128("0123"), DB::Exception); // wrong length +} + +TEST(CASToken, Basics) +{ + Token a{"etag-1", TokenType::ETag}; + Token b{"etag-1", TokenType::ETag}; + Token c{"etag-2", TokenType::ETag}; + EXPECT_EQ(a, b); + EXPECT_NE(a, c); + EXPECT_TRUE(Token{}.empty()); + EXPECT_FALSE(a.empty()); +} diff --git a/src/Disks/tests/gtest_cas_inline_placement.cpp b/src/Disks/tests/gtest_cas_inline_placement.cpp new file mode 100644 index 000000000000..5b4000636fa0 --- /dev/null +++ b/src/Disks/tests/gtest_cas_inline_placement.cpp @@ -0,0 +1,28 @@ +#include +#include + +using DB::Cas::partFileMustStayBlob; + +TEST(CASInlinePlacement, ColumnAndMarkFilesStayBlob) +{ + EXPECT_TRUE(partFileMustStayBlob("data.bin")); + EXPECT_TRUE(partFileMustStayBlob("data.mrk")); + EXPECT_TRUE(partFileMustStayBlob("data.mrk2")); + EXPECT_TRUE(partFileMustStayBlob("data.mrk3")); + EXPECT_TRUE(partFileMustStayBlob("data.cmrk")); + EXPECT_TRUE(partFileMustStayBlob("data.cmrk2")); + EXPECT_TRUE(partFileMustStayBlob("data.cmrk3")); + EXPECT_TRUE(partFileMustStayBlob("primary.idx")); // potentially large; stays blob (follow-up tuning) +} + +TEST(CASInlinePlacement, EagerMetadataFilesAreInlineCandidates) +{ + EXPECT_FALSE(partFileMustStayBlob("checksums.txt")); + EXPECT_FALSE(partFileMustStayBlob("columns.txt")); + EXPECT_FALSE(partFileMustStayBlob("count.txt")); + EXPECT_FALSE(partFileMustStayBlob("serialization.json")); + EXPECT_FALSE(partFileMustStayBlob("metadata_version.txt")); + EXPECT_FALSE(partFileMustStayBlob("partition.dat")); + EXPECT_FALSE(partFileMustStayBlob("minmax_date.idx")); + EXPECT_FALSE(partFileMustStayBlob("default_compression_codec.txt")); +} diff --git a/src/Disks/tests/gtest_cas_inspect.cpp b/src/Disks/tests/gtest_cas_inspect.cpp new file mode 100644 index 000000000000..d807bec3f931 --- /dev/null +++ b/src/Disks/tests/gtest_cas_inspect.cpp @@ -0,0 +1,222 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; + +namespace +{ + +ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +BlobRef bh(uint64_t n) +{ + return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(n))}; +} + +} + +/// Stage-1 T12 (spec §4 "RefOp payload removal"): `cas-inspect` renders the renamed `SetPublishedAt` +/// op kind, and neither the ref-log nor the ref-snapshot rendering carries a `payload_size` key -- +/// `RefOp`/`RefCommittedRow` no longer have a `payload` field to size. + +TEST(CASInspect, RendersSetPublishedAtOpWithNoPayloadSizeKey) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + const RefTxnId id{7, 9}; + + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = id; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "all_1_1_0"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + op.published_at_ms = 42; + txn.ops.push_back(op); + + const String key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id); + const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); + + const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("kind":"SetPublishedAt")"), String::npos) << json; + EXPECT_EQ(json.find("payload"), String::npos) << json; +} + +/// Task-1 review finding M5: `cas inspect` renders the new `EpochSeal` op kind and the txn-level +/// `prev_epoch_seal` chain field, needed to debug INV-2 seal chains without a raw byte dump. +TEST(CASInspect, RendersEpochSealTxnWithPrevEpochSeal) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + const RefTxnId id{3, 1}; + + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = id; + txn.prev_epoch_seal = RefTxnId{2, 9}; + RefOp op; + op.kind = RefOpKind::EpochSeal; + txn.ops.push_back(op); + + const String key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id); + const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); + + const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("kind":"EpochSeal")"), String::npos) << json; + EXPECT_NE(json.find(R"("prev_epoch_seal":{"writer_epoch":2,"ref_sequence":9})"), String::npos) << json; +} + +TEST(CASInspect, RendersCommittedRowWithNoPayloadSizeKey) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + const RefTxnId id{7, 9}; + + RefTableSnapshot snap; + snap.ns = ns.string(); + snap.snapshot_id = id; + RefCommittedRow row; + row.ref_name = "all_1_1_0"; + row.manifest_ref = manifestRef(1, 1, 1); + row.published_at_ms = 42; + snap.committed.push_back(row); + + const String key = layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), id); + const String bytes = sealObject(FormatId::RefSnapshot, encodeRefTableSnapshot(snap)); + + const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_EQ(json.find("payload"), String::npos) << json; + EXPECT_EQ(json.find("lifecycle"), String::npos) << json; + EXPECT_EQ(json.find("remove_txn_id"), String::npos) << json; + EXPECT_NE(json.find(R"("published_at_ms":42)"), String::npos) << json; +} + +/// A blob-target source-edge run segment (`Layout::blobTargetRunKey`) is the ground truth for every +/// in-degree question; `cas-inspect` decodes it with the typed `SourceEdgeRunView` reader (not by hand) +/// and must distinguish an active edge from a condemned sentinel row, decoding the latter's fields. +TEST(CASInspect, RendersBlobTargetRunEdgeAndCondemnedRows) +{ + const Layout layout("p"); + + /// `SourceEdgeRunWriter::append` requires non-decreasing `(ref, source_id)` order; the condemned + /// sentinel sorts first for its blob (source_id 0), and `bh(1) < bh(2)`, so appending in this + /// order already satisfies it. + SourceEdgeRecord condemned_rec; + condemned_rec.ref = bh(1); + condemned_rec.source_id = UInt128{0}; + condemned_rec.marker = kCondemned; + condemned_rec.delete_pending = true; + condemned_rec.token = Token{.value = "etag-1", .type = TokenType::Emulated}; + condemned_rec.size = 123; + condemned_rec.condemn_round = 7; + condemned_rec.marker_confirmed = true; + + SourceEdgeRecord edge_rec; + edge_rec.ref = bh(2); + edge_rec.source_id = UInt128(9); + edge_rec.marker = kEdgeActive; + + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(condemned_rec); + writer.append(edge_rec); + writer.finish(); + out.finalize(); + const String bytes = out.str(); + + const String key = layout.blobTargetRunKey(/*generation*/2, /*attempt*/0, /*shard*/0, /*seq*/0); + + const String json = caInspectToJson(layout, key, bytes); + EXPECT_NE(json.find(R"("object":"blob_target_run")"), String::npos) << json; + EXPECT_NE(json.find(R"("generation":2)"), String::npos) << json; + EXPECT_NE(json.find(R"("kind":"edge")"), String::npos) << json; + EXPECT_NE(json.find(R"("kind":"condemned")"), String::npos) << json; + EXPECT_NE(json.find(R"("delete_pending":true)"), String::npos) << json; + EXPECT_NE(json.find(R"("condemn_round":7)"), String::npos) << json; + EXPECT_NE(json.find(R"("value":"etag-1")"), String::npos) << json; + EXPECT_NE(json.find(R"("rows":2)"), String::npos) << json; + EXPECT_NE(json.find(R"("distinct_blobs":2)"), String::npos) << json; + EXPECT_NE(json.find(R"("edges":1)"), String::npos) << json; + EXPECT_NE(json.find(R"("condemned":1)"), String::npos) << json; + EXPECT_NE(json.find(R"("zero_markers":0)"), String::npos) << json; +} + +/// Stage A task 5 (spec INV-4): the `_ckpt` renders as its own object kind. It is point-addressed in +/// `cas/ns/state/` with no transaction id, so it has a separate dispatch from stream objects and once +/// fell through to +/// `BAD_ARGUMENTS` for it -- and it is precisely the object an operator reaches for when asking "what +/// is recovery's base" or "why is cleanup not reclaiming anything". +TEST(CASInspect, RendersRefCkptWithEveryFieldPresent) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + + const RefCkpt ckpt{.life_epoch = std::optional{7}, + .committed_through = RefTxnId{7, 9}, + .checkpoint_snapshot_id = RefTxnId{7, 9}, + .last_epoch_seal = RefTxnId{6, 4}}; + + const String json = caInspectToJson( + layout, layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(ckpt), + DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("object":"ref_ckpt")"), String::npos) << json; + /// The namespace comes from the KEY: a `_ckpt` body does not name it. + EXPECT_NE(json.find(R"("namespace":"srv1/db/tbl")"), String::npos) << json; + EXPECT_NE(json.find(R"("life_epoch":7)"), String::npos) << json; + EXPECT_NE(json.find(R"("committed_through":{"writer_epoch":7,"ref_sequence":9})"), String::npos) << json; + EXPECT_NE(json.find(R"("writer_epoch":7,"ref_sequence":9)"), String::npos) << json; + EXPECT_NE(json.find(R"("writer_epoch":6,"ref_sequence":4)"), String::npos) << json; +} + +/// The absences are the interesting readings, so they render as explicit `null`s rather than missing +/// keys: no checkpoint means recovery has no base AND nothing is deletable, which is a very different +/// report from "the key is there and I could not tell you what is in it". +TEST(CASInspect, RendersRefCkptAbsencesAsExplicitNulls) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/fresh"}; + + const String json = caInspectToJson( + layout, layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(RefCkpt{}), + DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("object":"ref_ckpt")"), String::npos) << json; + EXPECT_NE(json.find(R"("life_epoch":null)"), String::npos) << json; + EXPECT_NE(json.find(R"("checkpoint_snapshot_id":null)"), String::npos) << json; + EXPECT_NE(json.find(R"("last_epoch_seal":null)"), String::npos) << json; +} + +/// A listed physical id cannot supply a namespace. Inspect must receive the unique catalog join, and +/// a different logical spelling at the same id is rejected by the decoded object's own namespace. +TEST(CASInspect, RefObjectRequiresTheExactCatalogResolution) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128{91}); + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = RefTxnId{1, 1}; + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + txn.ops = {birth}; + const String key = layout.refLogKey(life, txn.txn_id); + const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); + + EXPECT_THROW(caInspectToJson(layout, key, bytes), DB::Exception); + EXPECT_THROW(caInspectToJson( + layout, key, bytes, NamespaceLifeId::fromCatalogEntry(RootNamespace{"redirected"}, life.incarnation)), + DB::Exception); + EXPECT_NO_THROW(caInspectToJson(layout, key, bytes, life)); +} diff --git a/src/Disks/tests/gtest_cas_json_writer.cpp b/src/Disks/tests/gtest_cas_json_writer.cpp new file mode 100644 index 000000000000..f04b89255818 --- /dev/null +++ b/src/Disks/tests/gtest_cas_json_writer.cpp @@ -0,0 +1,214 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; +using namespace DB::Cas; + +TEST(CASJsonWriter, KeyValueSequenceMatchesCanonicalShape) +{ + CasJsonWriter w; + bool first = true; + w.key("we", first); + w.u64StringValue(7); + w.key("mo", first); + w.u64Number(3); + w.key("ok", first); + w.boolValue(true); + w.key("o", "me", first); + w.u64StringValue(1); + w.closeObject(first); + w.newline(); + EXPECT_EQ(std::move(w).take(), "{\"we\":\"7\",\"mo\":3,\"ok\":true,\"ome\":\"1\"}\n"); +} + +TEST(CASJsonWriter, EmptyObjectAndClear) +{ + CasJsonWriter w; + bool first = true; + w.closeObject(first); + EXPECT_EQ(w.view(), "{}"); + w.clear(); + EXPECT_EQ(w.size(), 0u); +} + +TEST(CASJsonWriter, Hex128MatchesU128ToHex) +{ + const UInt128 v = (UInt128(0x0123456789abcdefULL) << 64) | UInt128(0xfedcba9876543210ULL); + CasJsonWriter w; + w.hex128Value(v); + EXPECT_EQ(std::move(w).take(), "\"" + u128ToHex(v) + "\""); +} + +TEST(CASJsonWriter, U64Extremes) +{ + CasJsonWriter w; + w.u64Number(0); + w.appendChar(' '); + w.u64Number(UINT64_MAX); + EXPECT_EQ(std::move(w).take(), "0 18446744073709551615"); +} + +namespace +{ +String referenceJson(std::string_view s) +{ + DB::FormatSettings settings; + settings.json.escape_forward_slashes = false; /// the pinned CAS canon + DB::WriteBufferFromOwnString out; + DB::writeJSONString(s, out, settings); + out.finalize(); + return out.str(); +} + +String writerJson(std::string_view s) +{ + DB::Cas::CasJsonWriter w; + w.stringValue(s); + return std::move(w).take(); +} +} + +TEST(CASJsonWriterEscaping, TargetedCorpusMatchesWriteJSONString) +{ + const std::vector corpus = { + "", + "plain_safe_ref_name_20260101_0_1_1_1", + "roots/pin", /// '/' must stay UNESCAPED + "quote\"inside", "back\\slash", "both\\\"x", + String("\b\f\n\r\t"), + String(1, '\0'), String("a") + '\0' + "b", + String("\x01\x02\x03\x1e\x1f"), + "\xE2\x80\xA8", "\xE2\x80\xA9", /// U+2028 / U+2029 -> / + "x\xE2\x80\xA8" "y", // NOLINT(bugprone-suspicious-missing-comma): deliberate adjacent-literal concatenation, testing a U+2028 sequence split across two source literals + "\xE2", /// truncated lead byte at end + "\xE2\x80", /// truncated pair at end + "\xE2\x21\x21", /// 0xE2 + non-continuation bytes + "\xE2\x80\x21", + "\xE2\xE2\x80\xA8", /// lead byte immediately before a real sequence + "\xC3\xA9\xF0\x9F\x98\x80", /// ordinary multi-byte UTF-8 passes through + "\xff\xfe invalid utf8 \x80", + String(1000, 'a'), /// long safe run (vector path) + String(1000, '"'), /// special-dense + }; + for (const String & s : corpus) + EXPECT_EQ(writerJson(s), referenceJson(s)) << "input bytes: " << s.size(); +} + +TEST(CASJsonWriterEscaping, FuzzMatchesWriteJSONString) +{ + std::mt19937 rng(20260720); // NOLINT(cert-msc32-c, cert-msc51-cpp) + for (int iter = 0; iter < 5000; ++iter) + { + const size_t len = rng() % 200; + String s(len, '\0'); + const int mode = iter % 3; + for (auto & c : s) + { + if (mode == 0) + c = static_cast(rng() % 256); /// full byte range + else if (mode == 1) + c = static_cast('a' + rng() % 26); /// safe-only + else + { + static constexpr char specials[] = {'"', '\\', '\n', '\x01', '\xE2', '\x80', '\xA8', 'z'}; + c = specials[rng() % (sizeof(specials))]; /// special-dense + } + } + ASSERT_EQ(writerJson(s), referenceJson(s)) << "iter " << iter; + } +} + +/// ---- CasJsonWriter overloads of the shared vocabulary (Task 4) ---- +/// +/// The production WriteBuffer vocabulary was retired in Task 9 (CasJsonWriter is now the only CAS +/// text writer). `reference_vocab` below is a verbatim copy of the retired implementation, kept +/// test-local so these differential tests keep an independent oracle instead of comparing +/// CasJsonWriter against itself. +namespace reference_vocab +{ +namespace +{ +/// Verbatim copy of the retired WriteBuffer-based CAS vocabulary (CasTextFormat.cpp pre-CasJsonWriter), +/// kept as the differential reference. jsonWriteSettings is inlined: escape_forward_slashes=false. +const DB::FormatSettings & settings() +{ + static const DB::FormatSettings s = [] + { + DB::FormatSettings fs; + fs.json.escape_forward_slashes = false; + return fs; + }(); + return s; +} + +void writeKey(DB::WriteBuffer & out, std::string_view key, bool & first) +{ + DB::writeChar(first ? '{' : ',', out); + first = false; + DB::writeChar('"', out); + out.write(key.data(), key.size()); + DB::writeChar('"', out); + DB::writeChar(':', out); +} + +void writeStringValue(DB::WriteBuffer & out, std::string_view s) { DB::writeJSONString(s, out, settings()); } + +void writeHex128Value(DB::WriteBuffer & out, const UInt128 & v) +{ + DB::writeChar('"', out); + const String hex = DB::Cas::u128ToHex(v); + out.write(hex.data(), hex.size()); + DB::writeChar('"', out); +} + +void writeU64StringValue(DB::WriteBuffer & out, uint64_t v) +{ + DB::writeChar('"', out); + DB::writeIntText(v, out); + DB::writeChar('"', out); +} + +void writeBoolValue(DB::WriteBuffer & out, bool v) { writeCString(v ? "true" : "false", out); } + +void closeObject(DB::WriteBuffer & out, bool & first) +{ + if (first) + DB::writeChar('{', out); + first = false; + DB::writeChar('}', out); +} +} +} + +TEST(CASJsonWriterVocab, MatchesReferenceVocabulary) +{ + using namespace DB::Cas; + const UInt128 h = (UInt128(0xdeadbeefULL) << 64) | UInt128(42); + + DB::WriteBufferFromOwnString ref; + CasJsonWriter w; + bool rf = true; + bool wf = true; + + reference_vocab::writeKey(ref, "a", rf); writeKey(w, "a", wf); + reference_vocab::writeStringValue(ref, "x/\"y"); writeStringValue(w, "x/\"y"); + reference_vocab::writeKey(ref, "h", rf); writeKey(w, "h", wf); + reference_vocab::writeHex128Value(ref, h); writeHex128Value(w, h); + reference_vocab::writeKey(ref, "u", rf); writeKey(w, "u", wf); + reference_vocab::writeU64StringValue(ref, UINT64_MAX); writeU64StringValue(w, UINT64_MAX); + reference_vocab::writeKey(ref, "b", rf); writeKey(w, "b", wf); + reference_vocab::writeBoolValue(ref, false); writeBoolValue(w, false); + reference_vocab::writeKey(ref, "n", rf); writeKey(w, "n", wf); + DB::writeIntText(uint64_t(12345), ref); writeIntText(uint64_t(12345), w); + reference_vocab::closeObject(ref, rf); closeObject(w, wf); + DB::writeChar('\n', ref); writeChar('\n', w); + ref.finalize(); + EXPECT_EQ(std::move(w).take(), ref.str()); +} diff --git a/src/Disks/tests/gtest_cas_layout.cpp b/src/Disks/tests/gtest_cas_layout.cpp new file mode 100644 index 000000000000..1006e87188a2 --- /dev/null +++ b/src/Disks/tests/gtest_cas_layout.cpp @@ -0,0 +1,382 @@ +#include +#include +#include +#include +#include "cas_test_helpers.h" + +using namespace DB::Cas; + +namespace +{ +/// A `BlobRef` at `algo` whose first bytes are `0x00, 0xaa, 0xbb` (the rest zero) -- for key-shape +/// tests that need a stable, recognizable hex prefix. `Layout` no longer captures an algo (Phase 3 +/// T2/T3): every blob key is built from a `BlobRef` alone, so key-shape tests construct one directly. +BlobRef prefixedRef(BlobHashAlgo algo) +{ + BlobDigest d{}; + d.bytes[0] = 0x00; d.bytes[1] = 0xaa; d.bytes[2] = 0xbb; + return BlobRef{algo, d}; +} +} + +TEST(CASLayout, KeyShapes) +{ + /// Per design §10 EVERY algo carries an explicit path segment: `blobs/ch128/...`, not the legacy + /// `blobs/...`. + Layout l{"p"}; + const BlobRef ref = prefixedRef(BlobHashAlgo::CityHash128); + const String hex = codecFor(BlobHashAlgo::CityHash128).toHex(ref.digest); + EXPECT_EQ(l.blobKey(ref), "p/blobs/ch128/" + hex.substr(0, 2) + "/" + hex); + EXPECT_EQ(l.gcStateKey(), "p/gc/state"); + EXPECT_EQ(l.outcomesKey(4, 42, 7, 1), "p/gc/gen/4/attempt/42/outcomes/7/1.zst"); + EXPECT_EQ(l.poolMetaKey(), "p/_pool_meta"); +} + +TEST(CASLayout, BlobKeyCarriesAlgoSegment) +{ + /// Every algo gets its own segment (design §3/§10), so two algos can never collide in the key + /// space even after a config change on a fresh pool. `Layout` itself carries no algo anymore -- + /// the segment comes from the `BlobRef` passed to `blobKey`/`blobMetaKey`. + const Layout l("p"); + + const BlobRef ch128_ref = prefixedRef(BlobHashAlgo::CityHash128); + const String ch128_hex = codecFor(BlobHashAlgo::CityHash128).toHex(ch128_ref.digest); + EXPECT_EQ(l.blobKey(ch128_ref), "p/blobs/ch128/" + ch128_hex.substr(0, 2) + "/" + ch128_hex); + EXPECT_EQ(l.blobMetaKey(ch128_ref), l.blobKey(ch128_ref) + ".meta"); + + const BlobRef xxh3_ref = prefixedRef(BlobHashAlgo::XXH3_128); + const String xxh3_hex = codecFor(BlobHashAlgo::XXH3_128).toHex(xxh3_ref.digest); + EXPECT_EQ(l.blobKey(xxh3_ref), "p/blobs/xxh3/" + xxh3_hex.substr(0, 2) + "/" + xxh3_hex); + EXPECT_EQ(l.blobMetaKey(xxh3_ref), l.blobKey(xxh3_ref) + ".meta"); + + const BlobRef sha256_ref = prefixedRef(BlobHashAlgo::Sha256); + const String sha256_hex = codecFor(BlobHashAlgo::Sha256).toHex(sha256_ref.digest); + EXPECT_EQ(l.blobKey(sha256_ref), "p/blobs/sha256/" + sha256_hex.substr(0, 2) + "/" + sha256_hex); + + /// Trees/manifests/refs are UNCHANGED -- only blob-body keys gain the algo segment. + EXPECT_EQ(l.blobsPrefix(), "p/blobs/"); +} + +TEST(CASLayout, RootNamespaceKeys) +{ + Layout l("p"); + RootNamespace ns{"srv1/3f2e-uuid"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + EXPECT_EQ(l.namespaceStreamPrefix(ns_id), + "p/cas/ns/stream/" + renderIncarnation(ns_id.incarnation) + "/"); + EXPECT_EQ(l.namespaceFileKey(ns_id, "format_version.txt"), + "p/cas/ns/state/" + renderIncarnation(ns_id.incarnation) + "/_files/format_version.txt"); + EXPECT_EQ(l.namespaceFilesPrefix(ns_id), + "p/cas/ns/state/" + renderIncarnation(ns_id.incarnation) + "/_files/"); +} + +TEST(CASLayout, OpaqueLifeIdSeparatesStreamFromState) +{ + /// This catches a builder that accidentally puts the logical namespace back into a life-owned + /// key. The two different names deliberately share one physical id: object identity is the id, + /// while the name remains catalog-only. + Layout l("p"); + const UInt128 life_id = UInt128(0x1234); + const NamespaceLifeId first = NamespaceLifeId::fromCatalogEntry(RootNamespace{"root/first"}, life_id); + const NamespaceLifeId second = NamespaceLifeId::fromCatalogEntry(RootNamespace{"root/second"}, life_id); + const RefTxnId txn{7, 9}; + + EXPECT_EQ(l.namespaceStreamPrefix(first), "p/cas/ns/stream/00000000000000000000000000001234/"); + EXPECT_EQ(l.namespaceStatePrefix(first), "p/cas/ns/state/00000000000000000000000000001234/"); + EXPECT_EQ(l.refLogKey(first, txn), "p/cas/ns/stream/00000000000000000000000000001234/_log/0000000000000007-0000000000000009.zst"); + EXPECT_EQ(l.refSnapshotKey(first, txn), "p/cas/ns/stream/00000000000000000000000000001234/_snap/0000000000000007-0000000000000009.zst"); + EXPECT_EQ(l.refCkptKey(first), "p/cas/ns/state/00000000000000000000000000001234/_ckpt"); + EXPECT_EQ(l.namespaceFileKey(first, "nested/file"), "p/cas/ns/state/00000000000000000000000000001234/_files/nested/file"); + + EXPECT_EQ(l.refLogKey(second, txn), l.refLogKey(first, txn)); + EXPECT_EQ(l.namespaceFileKey(second, "nested/file"), l.namespaceFileKey(first, "nested/file")); +} + +TEST(CASLayout, RelocatedRefAndManifestKeys) +{ + Layout l("p"); + const RootNamespace ns{"srid/store/ab/uuid@cas@"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + EXPECT_EQ(l.namespaceStreamPrefix(ns_id), + "p/cas/ns/stream/" + renderIncarnation(ns_id.incarnation) + "/"); + EXPECT_EQ(l.casRefsPrefix(), "p/cas/ns/stream/"); + /// All manifests of a namespace: cas/manifests// (replaces roots//_manifests/). + EXPECT_EQ(l.manifestNamespacePrefix(ns), "p/cas/manifests/srid/store/ab/uuid@cas@/"); + + /// manifestKey: canonical hex build directory, under cas/manifests// (no /_manifests/ infix). + ManifestId id; + id.root_namespace = ns; + id.ref.writer_epoch = 1; + id.ref.build_sequence = 1042; + id.ref.manifest_ordinal = 1; + const String key = l.manifestKey(id); + EXPECT_EQ(key, "p/cas/manifests/srid/store/ab/uuid@cas@/" + "0000000000000001-0000000000000412/000001.zst"); + EXPECT_EQ(key.find("/_manifests/"), String::npos) << key; +} + +TEST(CASLayout, RootNamespaceValidation) +{ + Layout l("p"); + /// Opaque physical life keys deliberately do not inspect the logical namespace. Namespace-bearing + /// families such as manifests remain responsible for validating it. + EXPECT_THROW(l.manifestNamespacePrefix(RootNamespace{""}), DB::Exception); + EXPECT_THROW(l.manifestNamespacePrefix(RootNamespace{"/lead"}), DB::Exception); + EXPECT_THROW(l.manifestNamespacePrefix(RootNamespace{"trail/"}), DB::Exception); + /// File names may be NESTED relative paths (M-W T2: deduplication_logs/...); only unclean + /// shapes are rejected (empty, leading/trailing '/', empty segments, '..' escapes). + const NamespaceLifeId ok_id = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"ok"}); + EXPECT_NO_THROW(l.namespaceFileKey(ok_id, "a/b")); + EXPECT_THROW(l.namespaceFileKey(ok_id, ""), DB::Exception); + EXPECT_THROW(l.namespaceFileKey(ok_id, "/lead"), DB::Exception); + EXPECT_THROW(l.namespaceFileKey(ok_id, "trail/"), DB::Exception); + EXPECT_THROW(l.namespaceFileKey(ok_id, "a//b"), DB::Exception); + EXPECT_THROW(l.namespaceFileKey(ok_id, "../up"), DB::Exception); + EXPECT_THROW(l.namespaceFileKey(ok_id, "a/../b"), DB::Exception); + + EXPECT_THROW(l.manifestNamespacePrefix(RootNamespace{"a//b"}), DB::Exception); + EXPECT_THROW(l.manifestNamespacePrefix(RootNamespace{"srv1/_files/x"}), DB::Exception); + EXPECT_NO_THROW(l.manifestNamespacePrefix(RootNamespace{"my_files/tbl"})); +} + +TEST(CASLayout, GenerationAndRootsKeys) +{ + Layout l("p"); + /// rev. 15: gc/snap is gone; generations carry write-once seals + blob-target / cleanup runs. + /// rev. 16: every per-round artifact is attempt-scoped under gc/gen//attempt//. + EXPECT_EQ(l.foldSealKey(12, 0), "p/gc/gen/12/attempt/0/fold_seal"); + EXPECT_EQ(l.blobTargetRunKey(12, 0, 0, 0), "p/gc/gen/12/attempt/0/blob_target/0/0"); + EXPECT_EQ(l.namespaceRootPrefix(), "p/cas/ns/"); + EXPECT_EQ(l.rootsPrefix(), "p/roots/"); +} + +TEST(CASLayout, AttemptScopedGenKeys) +{ + DB::Cas::Layout layout("p"); + EXPECT_EQ(layout.foldSealKey(4, 42), "p/gc/gen/4/attempt/42/fold_seal"); + EXPECT_EQ(layout.blobTargetRunKey(4, 42, 3, 0), "p/gc/gen/4/attempt/42/blob_target/3/0"); + EXPECT_EQ(layout.outcomesKey(5, 42, 7, 3), "p/gc/gen/5/attempt/42/outcomes/7/3.zst"); + EXPECT_EQ(layout.gcGenPrefix(4), "p/gc/gen/4/"); + EXPECT_EQ(layout.gcGenAttemptPrefix(4, 42), "p/gc/gen/4/attempt/42/"); +} + +TEST(CASLayout, RegistryDeletedGcDiscoveryViaList) +{ + /// Task 4: the namespace registry (`gc/registry`) is deleted; discovery authority moved to LIST. + /// The `_registry` namespace segment is not reserved (it was only reserved while the registry lived + /// under `roots/_registry`, which was already relocated to `gc/registry` before being deleted). + Layout l("p"); + EXPECT_NO_THROW(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"a/_registry@cas@"}))); + /// Opaque stream keys are independent of namespace-segment reservations. + EXPECT_NO_THROW(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"a/_files"}))); +} + +TEST(CASLayout, CasArchiveSuffixConstant) +{ + EXPECT_EQ(DB::Cas::kCasArchiveSuffix, "@cas@"); +} + +TEST(CASVfsPaths, MirroredArchiveNamespace) +{ + using DB::Cas::mirroredArchiveNamespace; + /// Atomic: bare uuid -> store//@cas@ + EXPECT_EQ(mirroredArchiveNamespace("3f2a0000-0000-0000-0000-000000000001"), + "store/3f2/3f2a0000-0000-0000-0000-000000000001@cas@"); + /// Non-Atomic: a full data/db/tbl path is used verbatim, @cas@ appended to the last segment. + EXPECT_EQ(mirroredArchiveNamespace("data/mydb/events"), + "data/mydb/events@cas@"); +} + +TEST(CASLayout, ManifestKeyShape) +{ + Layout l("p"); + ManifestId id; + id.root_namespace = RootNamespace("srv-a/3f2e-uuid@cas@"); + id.ref.writer_epoch = 7; + id.ref.build_sequence = 1042; + id.ref.manifest_ordinal = 1; + const String key = l.manifestKey(id); + EXPECT_EQ(key, + "p/cas/manifests/srv-a/3f2e-uuid@cas@/" + "0000000000000007-0000000000000412/000001.zst"); +} + +TEST(CASLayout, ManifestsSegmentReserved) +{ + Layout l("p"); + ManifestId bad; + bad.root_namespace = RootNamespace("srv-a/_manifests/x"); + EXPECT_THROW(l.manifestKey(bad), DB::Exception); + /// Opaque life prefixes ignore the logical spelling; manifests still enforce the reservation. + EXPECT_NO_THROW(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv-a/_manifests/tbl"}))); + EXPECT_NO_THROW(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"my_manifests/tbl"}))); +} + +TEST(CASLayout, ManifestKeyHexRoundTrip) +{ + Layout l("p"); + ManifestId id; + id.root_namespace = RootNamespace("srv-a/3f2e-uuid@cas@"); + id.ref.writer_epoch = 7; + id.ref.build_sequence = 0x8e; + id.ref.manifest_ordinal = 42; + const String key = l.manifestKey(id); + EXPECT_EQ(key, + "p/cas/manifests/srv-a/3f2e-uuid@cas@/" + "0000000000000007-000000000000008e/000042.zst"); + + const auto parsed = l.parseManifestKey(key); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->root_namespace, id.root_namespace); + EXPECT_EQ(parsed->ref, id.ref); + + /// The old two-directory decimal shape (`//.zst`) is no + /// longer canonical: the segment right before the file is a plain decimal number, not two + /// fixed-width hex fields joined by '-', so `parseRefTxnId` rejects it. + EXPECT_FALSE(l.parseManifestKey("p/cas/manifests/srv-a/3f2e-uuid@cas@/7/142/000042.zst").has_value()); + /// Foreign prefix, missing build segment, non-registered-suffix file, and out-of-range ordinal + /// are all rejected. + EXPECT_FALSE(l.parseManifestKey("p/cas/refs/srv-a/3f2e-uuid@cas@/" + "0000000000000007-000000000000008e/000042.zst").has_value()); + EXPECT_FALSE(l.parseManifestKey("p/cas/manifests/0000000000000007-000000000000008e/000042.zst").has_value()); + EXPECT_FALSE(l.parseManifestKey("p/cas/manifests/srv-a/3f2e-uuid@cas@/" + "0000000000000007-000000000000008e/000042.bin").has_value()); + EXPECT_FALSE(l.parseManifestKey("p/cas/manifests/srv-a/3f2e-uuid@cas@/" + "0000000000000007-000000000000008e/000000.zst").has_value()); + EXPECT_FALSE(l.parseManifestKey("p/cas/manifests/srv-a/3f2e-uuid@cas@/" + "0000000000000007-000000000000008E/000042.zst").has_value()); /// uppercase hex +} + +TEST(CASLayout, RefObjectKeyRoundTrips) +{ + Layout l("p"); + const RootNamespace ns{"srv1/tbl@cas@"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + const RefTxnId id{7, 0x8e}; + const String life = "p/cas/ns/stream/" + renderIncarnation(ns_id.incarnation) + "/"; + + const String log_key = l.refLogKey(ns_id, id); + EXPECT_EQ(log_key, life + "_log/0000000000000007-000000000000008e.zst"); + const auto parsed_log = l.parseRefObjectKey(log_key); + ASSERT_TRUE(parsed_log.has_value()); + EXPECT_EQ(parsed_log->life_id, ns_id.incarnation); + EXPECT_EQ(parsed_log->kind, RefObjectKind::Log); + EXPECT_EQ(parsed_log->txn_id, id); + + const String snap_key = l.refSnapshotKey(ns_id, id); + EXPECT_EQ(snap_key, life + "_snap/0000000000000007-000000000000008e.zst"); + const auto parsed_snap = l.parseRefObjectKey(snap_key); + ASSERT_TRUE(parsed_snap.has_value()); + EXPECT_EQ(parsed_snap->life_id, ns_id.incarnation); + EXPECT_EQ(parsed_snap->kind, RefObjectKind::Snap); + EXPECT_EQ(parsed_snap->txn_id, id); + +} + +TEST(CASLayout, RefObjectKeyLexicalOrder) +{ + Layout l("p"); + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/tbl@cas@"}); + const RefTxnId id{7, 0x8e}; + EXPECT_LT(l.refLogKey(ns_id, id), l.refSnapshotKey(ns_id, id)); +} + +TEST(CASLayout, ParseRefObjectKeyRejections) +{ + Layout l("p"); + const RootNamespace ns{"srv1/tbl@cas@"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + const RefTxnId id{7, 0x8e}; + const String log_key = l.refLogKey(ns_id, id); + const String snap_key = l.refSnapshotKey(ns_id, id); + + /// Foreign top-level prefix. + EXPECT_FALSE(l.parseRefObjectKey("p/cas/manifests/srv1/tbl@cas@/_log/" + renderRefTxnId(id)).has_value()); + /// Unknown kind directory (also covers the removed numeric-shard ref-key shape, which has no kind dir). + EXPECT_FALSE(l.parseRefObjectKey("p/cas/ns/stream/00000000000000000000000000000001/_bogus/" + renderRefTxnId(id)).has_value()); + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceStreamPrefix(ns_id) + "3").has_value()); + /// Uppercase hex and a short id are non-canonical RefTxnId renders. The id is judged BEFORE the + /// life segment, so these stay "not ours" rather than becoming an incarnation refusal. + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceStreamPrefix(ns_id) + "_log/" + "0000000000000007-000000000000008E").has_value()); + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceStreamPrefix(ns_id) + "_log/7-8e").has_value()); + /// `_snap` without its stored suffix, and WITH a stray one, are both rejected. The suffix is taken + /// from the registry rather than spelled out: it was `.proto` when this test was written and is + /// `.zst` today, and stripping the wrong number of characters would have tested nothing. + const String snap_suffix{storedSuffix(FormatId::RefSnapshot)}; + EXPECT_FALSE(l.parseRefObjectKey(snap_key.substr(0, snap_key.size() - snap_suffix.size())).has_value()); + EXPECT_FALSE(l.parseRefObjectKey(log_key + ".proto").has_value()); + /// Trailing garbage after the id. + EXPECT_FALSE(l.parseRefObjectKey(log_key + "/extra").has_value()); + EXPECT_FALSE(l.parseRefObjectKey(snap_key + "/extra").has_value()); + /// Missing namespace segment entirely. + EXPECT_FALSE(l.parseRefObjectKey("p/cas/ns/stream/_log/" + renderRefTxnId(id)).has_value()); + /// The `_ckpt` (spec INV-4) has no kind directory and no transaction id, so the id-bearing parser + /// must not claim it. Every sweep over the ref prefix has to consult `parseRefCkptKey` as well -- + /// `groupRefKeys` treats a key neither parser recognizes as corruption that aborts ref folding. + EXPECT_FALSE(l.parseRefObjectKey(l.refCkptKey(ns_id)).has_value()); +} + +/// Stage A task 5 (spec INV-4): `refCkptKey` and `parseRefCkptKey` are inverses, and the `_ckpt` +/// parser is exactly as strict as its id-bearing sibling -- it claims OUR checkpoint keys and nothing +/// else. A key that is not one of ours at all still yields `std::nullopt` rather than an exception, +/// for the same reason `parseRefObjectKey` does: classifying an untrusted listed key is an ordinary +/// "is this ours" question. Refusal is reserved for a key that IS ours but names no life -- +/// `gtest_cas_ref_namespace_id.cpp` owns that half. +TEST(CASLayout, RefCkptKeyRoundTripsAndRejectsEverythingElse) +{ + Layout l("p"); + const RootNamespace ns{"srv1/tbl@cas@"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + const RefTxnId id{7, 0x8e}; + + /// The state prefix plus the bare leaf, with no compression suffix (the format is raw), so the key + /// is exactly `cas/ns/state//_ckpt`. + EXPECT_EQ(l.refCkptKey(ns_id), l.namespaceStatePrefix(ns_id) + "_ckpt"); + EXPECT_EQ(l.parseRefCkptKey(l.refCkptKey(ns_id)), ns_id.incarnation); + const NamespaceLifeId deep = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"a/b/c"}); + EXPECT_EQ(l.parseRefCkptKey(l.refCkptKey(deep)), deep.incarnation); + + /// Foreign pool prefix. + EXPECT_FALSE(l.parseRefCkptKey("q/cas/ns/state/00000000000000000000000000000001/_ckpt").has_value()); + /// The two id-bearing kinds are not checkpoints. + EXPECT_FALSE(l.parseRefCkptKey(l.refLogKey(ns_id, id)).has_value()); + EXPECT_FALSE(l.parseRefCkptKey(l.refSnapshotKey(ns_id, id)).has_value()); + /// A suffix the registry does not put there, and trailing garbage. + EXPECT_FALSE(l.parseRefCkptKey(l.refCkptKey(ns_id) + ".zst").has_value()); + EXPECT_FALSE(l.parseRefCkptKey(l.refCkptKey(ns_id) + "/extra").has_value()); + /// A near-miss leaf name. + EXPECT_FALSE(l.parseRefCkptKey(l.namespaceStatePrefix(ns_id) + "_ckp").has_value()); + EXPECT_FALSE(l.parseRefCkptKey(l.namespaceStatePrefix(ns_id) + "_ckpt2").has_value()); + /// Missing namespace segment entirely. + EXPECT_FALSE(l.parseRefCkptKey("p/cas/ns/state/_ckpt").has_value()); + /// The mirror of the rejection above: `_ckpt` is not a canonical `RefTxnId` render, so a key that + /// puts it inside a kind directory is claimed by NEITHER parser. + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceStreamPrefix(ns_id) + "_log/_ckpt").has_value()); + /// The same key used to be READ by this parser as the checkpoint of a phantom namespace named + /// `srv1/tbl@cas@//_log`, because a namespace is an OPAQUE multi-segment string and nothing + /// distinguished a deeper real namespace from a shallower one with a stray segment. The life + /// segment closes that: `_log` is not a canonical incarnation, so the key is now REFUSED instead + /// of quietly naming a table that cannot exist. + EXPECT_FALSE(l.parseRefCkptKey(l.namespaceStatePrefix(ns_id) + "_log/_ckpt").has_value()); +} + +/// C3: blobKey/parseBlobKey are inverses; pins the grammar before relocating the definitions +/// from CasPartWriteTxn.cpp to CasLayout.cpp (relocation must not change a single byte of output). +TEST(CASLayout, BlobKeyRoundTripsThroughParse) +{ + DB::Cas::Layout layout("pool0"); + const DB::Cas::BlobRef ref{DB::Cas::BlobHashAlgo::XXH3_128, + DB::Cas::codecFor(DB::Cas::BlobHashAlgo::XXH3_128).fromHex(std::string(32, 'a'))}; + const String body = layout.blobKey(ref); + const String meta = layout.blobMetaKey(ref); + EXPECT_EQ(meta, body + ".meta"); + + auto parsed_body = layout.parseBlobKey(body); + auto parsed_meta = layout.parseBlobKey(meta); /// body and .meta parse to the SAME BlobRef + ASSERT_TRUE(parsed_body.has_value()); + ASSERT_TRUE(parsed_meta.has_value()); + EXPECT_EQ(*parsed_body, ref); + EXPECT_EQ(*parsed_meta, ref); + EXPECT_FALSE(layout.parseBlobKey("pool0/blobs/unknown-algo/aa/aa00").has_value()); /// foreign => nullopt +} diff --git a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp new file mode 100644 index 000000000000..b93166c88eac --- /dev/null +++ b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp @@ -0,0 +1,258 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +/// Task 5 (spec §§1-3): the pool lifecycle condition + the identity gate at step 0 of `tryRemountOnce`. +/// These tests open a real writable `Pool` over the in-memory ("Emulated"-style) backend, manipulate the +/// pool sentinels behind the pool's back, then drive the gate through the synchronous `tryRemountOnce` +/// seam and assert the resulting lifecycle condition + the store()-class refusal. They follow +/// gtest_cas_sentinel_probe.cpp's harness patterns; the op counter is `tests::CountingBackend`. + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +const String kSrid = "test"; + +/// Delete an existing key exactly (its current token comes from the same GET). Returns the deleted body +/// so a test can restore it verbatim later (scenario d). +String deleteKeyReturningBody(Backend & backend, const String & key) +{ + const auto got = backend.get(key); + EXPECT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; + if (!got) + return {}; + backend.deleteExact(key, got->token); + return got->bytes; +} + +/// GC's fence-out applied directly to the mount lease: preserve the body, set `gc_fenced`, bump `seq` +/// (token-guarded). A subsequent `tryRemountOnce` whose identity gate verdicts `Recover` then reclaims a +/// fresh incarnation and returns true. Mirrors gtest_cas_pool.cpp's `fenceOutMount`. +void fenceOutMount(Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + MountLease m = decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, PutOutcome::Done); +} + +/// A Backend decorator whose head/get/list throw an untyped transport error while `fail` is armed. Starts +/// DISARMED so `Pool::open` succeeds; a test arms it only to make the identity probe inconclusive. Mirrors +/// gtest_cas_sentinel_probe.cpp's `TransportFaultBackend`, but toggleable AFTER open. +class ToggleableTransportFaultBackend final : public InMemoryBackend +{ +public: + /// Unhide the base convenience overloads, matching every other Backend subclass in this suite. + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + HeadResult head(const String & key) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::head(key); + } + + std::optional get(const String & key, Range range) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::get(key, range); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::list(prefix, cursor, limit); + } + + std::atomic fail{false}; +}; + +} + +/// (a) `_pool_meta` + the owner anchor authoritatively absent → the gate enters `IdentityLost` (never +/// `Vanished`) and store()-class access fails loud. rev.8: `IdentityLost` is a fail-loud TERMINAL state — +/// `isVanished()` still reads false (it is a distinct terminal), but a direct gate re-probe refuses without +/// ever claiming/allocating/writing (the thread-exit behavior of the background observer is covered by +/// `RemountThreadSelfExitsOnceIdentityLost` below). +TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + + const String meta_key = store->layout().poolMetaKey(); + const String owner_key = store->layout().ownerKey(kSrid); + + /// Both sentinels gone (other objects may or may not remain — rev.8 does not distinguish). + deleteKeyReturningBody(*backend, meta_key); + deleteKeyReturningBody(*backend, owner_key); + + /// Even from `Live` (no fence trip), a direct remount attempt transitions through `TransientNotLive` + /// and enters `IdentityLost` at step 0 — WITHOUT reaching `claimOwnerOrThrow`. + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + EXPECT_FALSE(store->isVanished()) << "IdentityLost is a distinct terminal, not a Vanished state"; + + /// store()-class access now fails loud with the typed lifecycle error. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->throwIfLifecycleTerminal(); }); + + /// A direct gate re-probe still refuses without mutating: it probes the sentinels authoritatively and + /// performs ZERO writes (never claims/allocates/mounts on a terminal pool). + backend->resetCounts(); + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + EXPECT_EQ(backend->putTotal(), 0u) << "a terminal-IdentityLost gate probe must never claim, allocate, or write"; + EXPECT_GE(backend->headCount(meta_key), 1u) << "the gate still probes _pool_meta authoritatively"; +} + +/// (a2) rev.8 worker-exit: `IdentityLost` is terminal, so the persistent self-remount worker must self-exit +/// — mirroring how a `Vanished` pool refuses to latch work. With `background_watermark = true`, `scheduleRemount` +/// must REFUSE to latch a recovery generation once the pool is `IdentityLost` (`remountTerminal` covers it), +/// exactly as it refuses on a published `Vanished` intent. +TEST(CASLifecycleCondition, RemountThreadSelfExitsOnceIdentityLost) +{ + auto backend = std::make_shared(); + /// `background_watermark = true` so the persistent recovery worker exists in production mode + /// (mirrors gtest_cas_pool.cpp's ShutdownGuardRefusesToArmRemount setup). + auto store = DB::Cas::Pool::open(backend, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); + + /// Drive the pool terminal (`IdentityLost`) synchronously before latching any recovery work. + deleteKeyReturningBody(*backend, store->layout().poolMetaKey()); + deleteKeyReturningBody(*backend, store->layout().ownerKey(kSrid)); + EXPECT_FALSE(store->tryRemountOnce()); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + + /// The runtime terminal consumer (or any direct `scheduleRemount`) must now refuse: no worker runs on a + /// terminal pool. + EXPECT_FALSE(store->scheduleRemountForTest()) + << "an IdentityLost pool is terminal (rev.8) — scheduleRemount must not latch recovery work"; +} + +/// (b) `_pool_meta` present but its `pool_id` is foreign → `Vanished(replaced)` immediately. +TEST(CASLifecycleCondition, PoolMetaForeignPoolIdEntersVanishedReplacedImmediately) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + + /// Overwrite `_pool_meta` with a FOREIGN pool_id (identity replaced); the object stays present. + const String meta_key = store->layout().poolMetaKey(); + const auto got = backend->get(meta_key); + ASSERT_TRUE(got.has_value()); + PoolMeta foreign = decodePoolMeta(got->bytes); + foreign.pool_id = foreign.pool_id + DB::UInt128(1); + ASSERT_EQ(backend->putOverwrite(meta_key, encodePoolMeta(foreign), got->token).outcome, PutOutcome::Done); + + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedReplaced); + EXPECT_TRUE(store->isVanished()); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->throwIfLifecycleTerminal(); }); +} + +/// (c) [B6] trap: `_pool_meta` present, pool_id + blob_header_len match, but `algos_used` differs → NOT a +/// replacement (`algos_used` is legally mutable); the existing recovery proceeds and the pool returns to +/// `Live`. +TEST(CASLifecycleCondition, PoolMetaAlgosUsedDifferIsNotReplacementRecoveryProceeds) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + const String meta_key = store->layout().poolMetaKey(); + const auto got = backend->get(meta_key); + ASSERT_TRUE(got.has_value()); + PoolMeta mutated = decodePoolMeta(got->bytes); + /// pool_id + blob_header_len UNCHANGED; only `algos_used` gains a member (a mutable field, [B6]). + const auto extra = static_cast(BlobHashAlgo::XXH3_128); + ASSERT_FALSE(std::binary_search(mutated.algos_used.begin(), mutated.algos_used.end(), extra)); + mutated.algos_used.push_back(extra); + std::sort(mutated.algos_used.begin(), mutated.algos_used.end()); + ASSERT_EQ(backend->putOverwrite(meta_key, encodePoolMeta(mutated), got->token).outcome, PutOutcome::Done); + + /// Fence out the mount so the (correctly non-replacement) recovery cleanly reclaims a fresh incarnation. + fenceOutMount(*backend, store->layout().mountKey(kSrid)); + + /// A differing `algos_used` must NOT read as a foreign pool: the gate verdicts `Recover`, recovery + /// completes, and the pool is `Live` — never `Vanished`. + EXPECT_TRUE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); + EXPECT_FALSE(store->isVanished()); +} + +/// (d) [D3] no auto-revival: from `IdentityLost`, restoring both sentinels with matching identity does NOT +/// bring the disk back — the observer stays fail-loud; only a restart recovers. +TEST(CASLifecycleCondition, IdentityLostDoesNotAutoReviveWhenSentinelsRestored) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + const String meta_key = store->layout().poolMetaKey(); + const String owner_key = store->layout().ownerKey(kSrid); + + const String meta_body = deleteKeyReturningBody(*backend, meta_key); + const String owner_body = deleteKeyReturningBody(*backend, owner_key); + + EXPECT_FALSE(store->tryRemountOnce()); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + + /// Restore both sentinels verbatim (a backup restore with matching identity). + ASSERT_EQ(backend->putIfAbsent(meta_key, meta_body).outcome, PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(owner_key, owner_body).outcome, PutOutcome::Done); + + /// The gate now sees Present+match, but the state is `IdentityLost`, so it stays fail-loud. + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->throwIfLifecycleTerminal(); }); +} + +/// (e) Transport error from the probe → the pool stays `TransientNotLive` (recoverable); absence is never +/// proven, so no terminal transition fires and store()-class access does NOT throw the terminal lifecycle +/// error (the transient class stays fence-gated until Task 8). +TEST(CASLifecycleCondition, ProbeTransportErrorStaysTransientAndRetries) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + + /// Arm the transport fault: the identity probe's head/get/list now throw → Indeterminate. + backend->fail.store(true); + + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive); + EXPECT_FALSE(store->isVanished()); + EXPECT_NO_THROW(store->throwIfLifecycleTerminal()); + + /// A second attempt with the fault still armed remains transient (retries continue). + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive); + + /// Disarm before teardown so `~Pool()`'s clean-farewell write is not fighting the injected fault. + backend->fail.store(false); +} diff --git a/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp new file mode 100644 index 000000000000..154c365a2a6d --- /dev/null +++ b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp @@ -0,0 +1,236 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +/// Task 12 (rev.7 spec §7, [C5]-visibility): the NON-GATED lifecycle snapshot backing +/// `system.cas_mounts`. A Factory-class read (spec §1): I/O-free, no `store()`/`poolAccess`, +/// truthful in EVERY state — so a not-live / stopped / vanished / never-started disk stays VISIBLE to the +/// operator instead of silently missing from the table. These tests exercise the accessor directly (the +/// SQL-level assertions land in Task 14): `ContentAddressedMetadataStorage::lifecycleSnapshot` at the +/// storage level, and `Pool::lifecycleSnapshot` at the pool level (including the zero-backend-op proof). +/// Harness patterns follow gtest_cas_operation_gate.cpp / gtest_cas_forget.cpp. + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +} + +using namespace DB; +using DB::Cas::PoolLifecycle; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +const std::string kSrid = "test"; + +/// A live table dir + committed part reused by the storage-level tests (the shape +/// gtest_cas_operation_gate.cpp / gtest_cas_forget.cpp use). +const std::string kTableDir = "sn0/sn0sn0s0-0808-4808-8808-080808080808"; +const std::string kPartDir = kTableDir + "/all_1_1_0"; +const std::string kPartFile = kPartDir + "/data.bin"; + +std::shared_ptr openSnapshotStorage() +{ + auto settings = Cas::tests::makeSettingsForTest( + kSrid, std::filesystem::temp_directory_path() / "ca_snapshot_scratch"); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +void commitOnePart(ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kTableDir + "/tmp_insert_all_1_1_0/data.bin", 65536, WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kTableDir + "/tmp_insert_all_1_1_0", kPartDir); + tx->commit(NoCommitOptions{}); +} + +/// Delete an existing key exactly (its current token comes from the same GET) — used to drive a live pool +/// into a NATURAL `IdentityLost`. Mirrors gtest_cas_forget.cpp / gtest_cas_lifecycle_condition.cpp. +void deleteKeyExact(DB::Cas::Backend & backend, const String & key) +{ + const auto got = backend.get(key); + ASSERT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; + if (got) + backend.deleteExact(key, got->token); +} + +} + +/// (a) Live: the snapshot reads `live` with no reason and no `since`, and always carries the disk's +/// last-known identity (pool_id + server_root_id). +TEST(CASLifecycleSnapshot, LiveIsTruthfulWithIdentity) +{ + auto storage = openSnapshotStorage(); + commitOnePart(*storage); + + const CasLifecycleSnapshot snap = storage->lifecycleSnapshot(); + EXPECT_EQ(snap.lifecycle, "live"); + EXPECT_TRUE(snap.reason.empty()) << snap.reason; + EXPECT_TRUE(snap.detail.empty()) << snap.detail; + EXPECT_EQ(snap.since, 0) << "a live pool has no lifecycle `since`"; + EXPECT_EQ(snap.server_root_id, storage->serverRootId()); + EXPECT_FALSE(snap.pool_id.empty()) << "a started disk knows its pool identity"; + EXPECT_EQ(snap.pool_id, storage->getPoolUUID()); +} + +/// (b) IdentityLost (forced from Live on the captured handle, the gate-test idiom): the snapshot names the +/// non-auto-recovering `identity_lost` state with the [D5] detail present and `since` set. The enum-clean +/// `reason` word is empty here — it carries only the `vanished` sub-state, and `identity_lost` is already +/// fully named by the `lifecycle` column. +TEST(CASLifecycleSnapshot, IdentityLostHasDetailAndSince) +{ + auto storage = openSnapshotStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live (store() is fail-closed on a terminal pool) + + pool->setLifecycleForTest(PoolLifecycle::IdentityLost); + + const CasLifecycleSnapshot snap = storage->lifecycleSnapshot(); + EXPECT_EQ(snap.lifecycle, "identity_lost"); + EXPECT_TRUE(snap.reason.empty()) << "reason is the vanish sub-state word only: " << snap.reason; + EXPECT_NE(snap.detail.find("identity lost"), std::string::npos) << snap.detail; + EXPECT_NE(snap.since, 0) << "a not-live state carries the wall-clock instant it was entered"; + /// Identity survives a terminal state — the disk stays introspectable under it. + EXPECT_EQ(snap.pool_id, storage->getPoolUUID()); +} + +/// (c) VanishedForgotten via the REAL verb (`storage->forgetDisk()`): the snapshot reads `vanished` with the +/// enum-clean `reason` word `forgotten` (so Task 14's `lifecycle || '(' || lifecycle_reason || ')'` reads +/// EXACTLY `vanished(forgotten)`), the [D5] `detail` carrying the operator's decommission timestamp, `since` +/// set, and the identity still present. +TEST(CASLifecycleSnapshot, VanishedForgottenIsEnumCleanWithTimestampedDetail) +{ + auto storage = openSnapshotStorage(); + commitOnePart(*storage); + const String pool_id_before = storage->getPoolUUID(); + + storage->forgetDisk(); + + const CasLifecycleSnapshot snap = storage->lifecycleSnapshot(); + EXPECT_EQ(snap.lifecycle, "vanished"); + EXPECT_EQ(snap.reason, "forgotten"); + /// Task 14's teardown check depends on this exact concatenation. + EXPECT_EQ(snap.lifecycle + "(" + snap.reason + ")", "vanished(forgotten)"); + EXPECT_NE(snap.detail.find("SYSTEM CAS FORGET at "), std::string::npos) << snap.detail; + EXPECT_NE(snap.detail.find("erasure was NOT verified"), std::string::npos) << snap.detail; + EXPECT_NE(snap.since, 0); + /// The disk stays registered and introspectable under its identity after FORGET. + EXPECT_EQ(snap.pool_id, pool_id_before); + EXPECT_EQ(snap.server_root_id, storage->serverRootId()); +} + +/// (d) A null pool never crashes the accessor and reports the storage-level lifecycle: `constructing` +/// before the first startup, `shutdown` after teardown. reason/since stay empty/0 (no terminal cause). +TEST(CASLifecycleSnapshot, NullPoolReportsConstructingThenShutdown) +{ + auto settings = Cas::tests::makeSettingsForTest( + kSrid, std::filesystem::temp_directory_path() / "ca_snapshot_null_scratch"); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + + /// Constructed but never started: no pool published. + const CasLifecycleSnapshot before = storage->lifecycleSnapshot(); + EXPECT_EQ(before.lifecycle, "constructing"); + EXPECT_TRUE(before.reason.empty()); + EXPECT_TRUE(before.detail.empty()); + EXPECT_EQ(before.since, 0); + EXPECT_TRUE(before.pool_id.empty()) << "no identity before startup"; + EXPECT_EQ(before.server_root_id, kSrid) << "the identity is known from config even pre-startup"; + + storage->startup(); + ASSERT_EQ(storage->lifecycleSnapshot().lifecycle, "live"); + + storage->shutdown(); + const CasLifecycleSnapshot after = storage->lifecycleSnapshot(); + EXPECT_EQ(after.lifecycle, "shutdown") << "a torn-down disk is distinguishable from a never-started one"; + EXPECT_FALSE(after.pool_id.empty()) << "the last-known identity survives shutdown"; +} + +/// (e) The accessor is I/O-free (spec §1 Factory class): NO backend op runs, in any lifecycle state. Proven +/// against a `CountingBackend` — the totals recorded after open do not move across snapshot reads, whether +/// the pool is Live or forced terminal. +TEST(CASLifecycleSnapshot, PerformsZeroBackendOps) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + const uint64_t head0 = backend->headTotal(); + const uint64_t get0 = backend->getTotal(); + const uint64_t put0 = backend->putTotal(); + const uint64_t getstream0 = backend->getStreamTotal(); + const uint64_t list0 = backend->listTotal(); + + const auto assertNoIo = [&](const char * where) + { + EXPECT_EQ(backend->headTotal(), head0) << where; + EXPECT_EQ(backend->getTotal(), get0) << where; + EXPECT_EQ(backend->putTotal(), put0) << where; + EXPECT_EQ(backend->getStreamTotal(), getstream0) << where; + EXPECT_EQ(backend->listTotal(), list0) << where; + }; + + /// Live snapshot: zero I/O. + (void)store->lifecycleSnapshot(); + assertNoIo("live snapshot must not touch the backend"); + + /// Forced terminal snapshot (the very state the store()-class surface refuses): still zero I/O. + store->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + const DB::Cas::Pool::LifecycleSnapshot vanished = store->lifecycleSnapshot(); + assertNoIo("a vanished-pool snapshot must not touch the backend"); + EXPECT_EQ(vanished.lifecycle, PoolLifecycle::VanishedReplaced); +} + +/// (f) A NATURAL transition (not the forced setter) captures the detail + `since`, and the snapshot's detail +/// is EXACTLY the [D5] text `throwIfLifecycleTerminal` throws (minus the pool-name prefix) — the spec §1 +/// "same reason strings in the snapshot and the error" guarantee, so the two can never drift. +TEST(CASLifecycleSnapshot, NaturalIdentityLostMatchesThrowDetail) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + + /// Delete both pool sentinels while other objects remain, then drive the identity gate → IdentityLost + /// (never Vanished), exactly gtest_cas_lifecycle_condition.cpp scenario (a). + deleteKeyExact(*backend, store->layout().poolMetaKey()); + deleteKeyExact(*backend, store->layout().ownerKey(kSrid)); + EXPECT_FALSE(store->tryRemountOnce()); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); + + const DB::Cas::Pool::LifecycleSnapshot snap = store->lifecycleSnapshot(); + EXPECT_EQ(snap.lifecycle, PoolLifecycle::IdentityLost); + EXPECT_NE(snap.since, 0) << "the natural enterIdentityLost transition stamps the wall-clock `since`"; + EXPECT_FALSE(snap.detail.empty()); + + /// The snapshot detail is the SAME [D5] text the typed error surfaces: the throw is + /// "content-addressed pool '' ", so the error message must contain the snapshot detail. + std::string thrown; + try + { + store->throwIfLifecycleTerminal(); + ADD_FAILURE() << "IdentityLost must throw from throwIfLifecycleTerminal"; + } + catch (const Exception & e) + { + thrown = std::string(e.message()); + } + EXPECT_NE(thrown.find(snap.detail), std::string::npos) + << "snapshot detail and the typed error must not drift\n detail: " << snap.detail + << "\n thrown: " << thrown; +} diff --git a/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp b/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp new file mode 100644 index 000000000000..1b48718663cb --- /dev/null +++ b/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp @@ -0,0 +1,624 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include +#include + +/// THE 2026-07-25 RELEASE BLOCKER, AS A PERMANENT REGRESSION. +/// +/// The defect the object store actually exhibited (`reports/2026-07-26-list-incompleteness-proof/`): +/// objects that were durable, acked, and readable by exact key were OMITTED from enumeration, while a +/// LATER key under the same prefix was listed. Nothing was lost and nothing was corrupt -- the store +/// simply under-reported what it held. +/// +/// Every CAS reader that treated a listing as a CENSUS then drew a false conclusion from it, and the +/// two that mattered drew ruinous ones. The GC fold walked the ids the listing returned, so it skipped +/// the omitted records' owner edges AND sealed a cursor above them -- and nothing ever re-reads below a +/// sealed cursor, so those edges were lost permanently: blobs that were still referenced looked +/// unreferenced forever after. Recovery replayed the listing, so a table came back missing an ACKED +/// transaction while looking perfectly healthy. +/// +/// The answer is that a listing is a HINT and arithmetic is the census. Ids are dense `1..T` +/// within `(namespace, writer_epoch)` (INV-1), so the next record's id is COMPUTABLE and every record +/// is read by EXACT KEY. A hidden-but-durable contiguous id is then a NON-EVENT -- the walk finds it +/// anyway -- while a genuinely absent expected id is a durable HOLD, never a silent skip. +/// +/// This file is that claim stated end to end, against a store that lies exactly the way the real one +/// did. `setListOmissions` names the omitted keys; `get`/`head`/`putIfAbsent`/`casPut`/`deleteExact` +/// keep serving them honestly. Each test below asserts the lie changed NOTHING -- not the folded +/// edges, not the cursor, not the recovered table, not fsck's verdict -- and the arms that are about +/// reclamation additionally assert that reclamation still happens, so "nothing was deleted" can never +/// pass for "the lie was harmless". +/// +/// The unit-level statements about the walk itself live in `gtest_cas_gc_arithmetic_intake.cpp`, and +/// the destructive gate's own inventory lives in `gtest_cas_gc_frontier_gate.cpp`. This file is the +/// INTEGRATION of the two: real rounds, real recovery, real fsck, one lying store. +/// +/// The suite name is prefixed `Cas` so the `Cas*` unit-test gate filter covers it. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +/// The lying store: LIST omits the named keys, everything else serves them honestly. Composed over +/// `CountingBackend` so the arms whose subject is reclamation can assert on DELETES rather than only on +/// what survived. +using LiarBackend = HintHoleBackendOn; + +String blobKeyOf(const Layout & layout, const DB::UInt128 & hash) +{ + return layout.blobKey(legacyMetaTestRef(hash)); +} + +/// The sealed fold cursor for `ns` as a full `RefTxnId`. Every fixture here writes ids inside writer +/// epoch 1, which is the assumption `foldCursorOf` (returning the sequence alone) already makes. +RefTxnId sealedCursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + return RefTxnId{1, foldCursorOf(backend, layout, ns, /*shard*/ 0)}; +} + +/// Drop the committed ref `ref_name` (currently naming `old_ref`) as ONE transaction at EXACTLY `id`. +/// The `dropRefTransition` helper allocates its id by LISTING, which a fixture that hides keys must +/// never do -- it would allocate over a hidden record. Every id in this file is therefore chosen. +void dropAt(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & id, + const String & ref_name, const ManifestRef & old_ref) +{ + writeTxnAt(backend, layout, ns, id, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, ref_name, old_ref}, std::nullopt)}); +} + +/// The manifest `publishAt` mints for a given (id, build_sequence) -- needed to drop that ref later. +ManifestRef publishedManifest(const RefTxnId & id, uint64_t build_sequence) +{ + return ManifestRef{.writer_epoch = id.writer_epoch, .build_sequence = build_sequence, .manifest_ordinal = 1}; +} + +/// One round plus everything a verdict in this file is allowed to rest on: the report (which carries +/// the anomaly list), the two intake phase rows (which carry the hold count and the one remaining +/// whole-round ref abort), and the gate's own verdict off `fold_reduce` (read, not recomputed, so a test +/// cannot agree with a wrong formula just as readily as with the right one). +struct RoundEvidence +{ + RoundReport report; + std::map intake; /// `fold_ref_intake` + std::map group; /// `fold_ref_group` + bool saw_fold = false; + bool frontier_complete = false; + bool suppress_destructive = false; +}; + +RoundEvidence runRoundCapturing(Gc & gc, UniversePolicy policy) +{ + RoundEvidence evidence; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + evidence.intake = rec.metrics; + else if (rec.phase == "fold_ref_group") + evidence.group = rec.metrics; + else if (rec.phase == "fold_reduce") + { + evidence.saw_fold = true; + if (const auto it = rec.metrics.find("frontier_complete"); it != rec.metrics.end()) + evidence.frontier_complete = it->second != 0; + if (const auto it = rec.metrics.find("suppress_destructive"); it != rec.metrics.end()) + evidence.suppress_destructive = it->second != 0; + } + }); + evidence.report = gc.runRegularRound({}, /*allow_steal*/ true, policy); + gc.setPhaseSink({}); + return evidence; +} + +/// "ZERO ANOMALIES", spelled out once so every test means the same thing by it: the round recorded no +/// anomaly, sealed no hold, and did not abort ref folding. A lie the walk absorbs must be invisible in +/// all three -- a hold in particular would be a WRONG (if safe) answer, since it would suppress the +/// round's destructive half over records that were durable all along. +void expectNoAnomalies(const RoundEvidence & evidence, const char * where) +{ + EXPECT_TRUE(evidence.report.anomalies.empty()) + << where << ": the round recorded " << evidence.report.anomalies.size() + << " anomaly/anomalies; a hidden-but-durable contiguous id is a NON-EVENT"; + ASSERT_FALSE(evidence.intake.empty()) << where << ": no `fold_ref_intake` row was emitted"; + EXPECT_EQ(evidence.intake.at("tables_held"), 0u) + << where << ": a namespace was HELD -- the walk mistook an omitted-but-durable record for a gap"; + EXPECT_EQ(evidence.intake.at("ref_folding_aborted"), 0u) << where; + ASSERT_FALSE(evidence.group.empty()) << where << ": no `fold_ref_group` row was emitted"; + EXPECT_EQ(evidence.group.at("ref_folding_aborted"), 0u) << where; +} + +/// The pool's view of a table, rendered so a failing comparison prints something a human can read. +std::map refsOf(const PoolPtr & store, const RootNamespace & ns) +{ + std::map out; + for (const auto & [ref_name, resolved] : store->listRefs(ns)) + out[ref_name] = std::to_string(resolved.manifest_id.ref.writer_epoch) + "/" + + std::to_string(resolved.manifest_id.ref.build_sequence) + "/" + + std::to_string(resolved.manifest_id.ref.manifest_ordinal); + return out; +} + +/// THE STREAM UNDER TEST, written identically into any backend: five ordinary publishes at +/// `{1,1}..{1,5}`, each pinning its own blob, plus the `_ckpt` a recovering reader starts from. Shared +/// so the oracle arms can seed a lying store and an honest one from the SAME code and compare outcomes +/// rather than compare against a hand-written expectation. +void seedFiveRecordStream(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + seedPoolMetaForRestart(backend, layout.poolPrefix()); + for (uint64_t i = 1; i <= 5; ++i) + publishAt(backend, layout, ns, RefTxnId{1, i}, "ref_" + std::to_string(i), i, + DB::UInt128(i), /*birth=*/i == 1); + writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 5}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); +} + +/// The exact defect shape: ids 3 and 4 invisible while the LATER id 5 is visible. +std::vector hiddenMiddleOf(const Layout & layout, const RootNamespace & ns) +{ + return {layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 3}), + layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 4})}; +} + +PoolConfig recoveryPoolConfig() +{ + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.server_id = DB::UInt128(1); + /// No background publication: a threshold-triggered snapshot would move the base under the + /// comparison these tests make about what recovery reconstructed. + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + return config; +} + +PoolPtr openRecoveryPool(const std::shared_ptr & backend) +{ + seedPoolMetaForRestart(*backend, "p"); + return Pool::open(backend, recoveryPoolConfig()); +} + +} + +/// ===================== THE BLOCKER, FULL PIPELINE ===================== +/// +/// Five durable records; the store lists 1, 2 and 5 and pretends 3 and 4 do not exist. Arithmetic +/// intake never asks the listing what to read next, so all five fold, every blob keeps its owner edge, +/// and the cursor lands on the true tail. +/// +/// Under listing-driven intake this fails on the BLOBS, not on the cursor: the cursor still reaches +/// `{1,5}` (the last listed id) while records 3 and 4 were never folded -- and since nothing re-reads +/// below a sealed cursor, their edges are gone for good. That is the production damage, exactly. +TEST(CASListLiarEndToEnd, TheHiddenMiddleOfTheStreamFoldsThroughUnnoticed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/blocker@cas@"}; + + seedFiveRecordStream(*backend, layout, ns); + backend->setListOmissions(hiddenMiddleOf(layout, ns)); + + Gc gc(store, kGc); + const RoundEvidence evidence = runRoundCapturing(gc, UniversePolicy::kDefault); + ASSERT_TRUE(evidence.report.acquired_lease); + ASSERT_GT(backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 5})) + << "the walk must reach the true tail of the stream"; + for (uint64_t i = 1; i <= 5; ++i) + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) + << "blob " << i << " lost its owner edge: its record was skipped because the store hid it"; + + expectNoAnomalies(evidence, "hidden middle"); + EXPECT_EQ(evidence.intake.at("logs_applied"), 5u) << "all five records are APPLIED, not three"; + EXPECT_EQ(evidence.intake.at("logs_accounted"), evidence.intake.at("logs_applied")) + << "probe B1: the arithmetic cut the cursors claim must equal what the walk applied"; +} + +/// RECOVERY, UNDER THE SAME LIE, AGAINST AN HONEST ORACLE. The comparison is against a second pool +/// seeded by the SAME code over a store that does not lie -- not against a hand-written expectation, +/// which could encode the same mistake the code makes. +TEST(CASListLiarEndToEnd, RecoveryUnderTheSameLieReconstructsExactlyTheTruth) +{ + const Layout layout("p"); + const RootNamespace ns{"00/recover@cas@"}; + + auto honest_backend = std::make_shared(); + seedFiveRecordStream(*honest_backend, layout, ns); + auto honest = openRecoveryPool(honest_backend); + const std::map truth = refsOf(honest, ns); + + auto lying_backend = std::make_shared(); + seedFiveRecordStream(*lying_backend, layout, ns); + lying_backend->setListOmissions(hiddenMiddleOf(layout, ns)); + auto lying = openRecoveryPool(lying_backend); + const std::map recovered = refsOf(lying, ns); + + ASSERT_GT(lying_backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + EXPECT_EQ(truth.size(), 5u) << "the oracle itself must see all five published refs"; + EXPECT_EQ(recovered, truth) + << "a table recovered under an omitting listing must be byte-identical to the truth; the " + "blocker's recovery came back missing an ACKED transaction and looked healthy"; +} + +/// ===================== THE DATA-LOSS ARM ===================== +/// +/// A blob with two owners. The `+1` that publishes the SECOND owner rides the hidden id; the `-1` that +/// releases the first is visible and lands above it. The arithmetic fold reads both, so the blob's +/// in-degree is 1 and it is never condemned. +/// +/// Listing-driven intake folds the visible `-1`, never folds the hidden `+1`, and seals the cursor +/// above it: the blob's in-degree reads zero while a live ref still names it, and the round deletes +/// data that is referenced. That is the data loss, and it is why this arm asserts the blob was never +/// even offered for deletion rather than merely that it is still present. +TEST(CASListLiarEndToEnd, AHiddenPlusOneKeepsItsBlobWhenAVisibleMinusOneLandsLater) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/dataloss@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + const DB::UInt128 shared(0x5ade); + + /// `ref_a` and `ref_b` both pin `shared`; `ref_b`'s publish is the record the store will hide. + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_a", 1, shared, /*birth=*/true); + publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_b", 2, shared); + dropAt(*backend, layout, ns, RefTxnId{1, 3}, "ref_a", publishedManifest(RefTxnId{1, 1}, 1)); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->setListOmissions({layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2})}); + + Gc gc(store, kGc); + const RoundEvidence first = runRoundCapturing(gc, UniversePolicy::Authoritative); + ASSERT_TRUE(first.report.acquired_lease); + ASSERT_GT(backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + expectNoAnomalies(first, "hidden +1"); + + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 3})); + EXPECT_EQ(inDegreeOf(*backend, layout, shared), 1) + << "the hidden publish's `+1` must be folded: `ref_b` still owns this blob"; + + /// Rounds that are ALLOWED to reclaim, and would, if the in-degree were wrong. + for (int i = 0; i < 5; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(backend->head(blobKeyOf(layout, shared)).exists) + << "a blob a live ref still names was DELETED -- the hidden `+1` was never folded"; + EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, shared)), 0u) + << "not merely still present: the delete was never even attempted"; +} + +/// ===================== THE LEAK ARM ===================== +/// +/// The mirror image, and the reason the arm above is not the whole story. Here the hidden record +/// carries the `-1` that releases the blob's last owner, and a visible record lands above it. The +/// arithmetic fold reads the `-1`, so the in-degree reaches zero and the blob is actually reclaimed. +/// +/// Listing-driven intake skips the `-1` and seals the cursor above it, so the blob keeps a phantom +/// owner forever: not data loss, but an object no incremental round can ever reclaim. Asserting the +/// blob DOES go away is also what stops the data-loss arm above from being satisfiable by a fold that +/// simply never deletes anything. +TEST(CASListLiarEndToEnd, AHiddenMinusOneIsStillFoldedSoTheBlobIsActuallyReclaimed) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/leak@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + const DB::UInt128 released(0xdea1); + const DB::UInt128 unrelated(0xb00c); + + publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_a", 1, released, /*birth=*/true); + dropAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_a", publishedManifest(RefTxnId{1, 1}, 1)); + /// A VISIBLE record above the hidden one. Without it the hidden id would be the stream's tail, and + /// a listing-driven walk would merely stop below it -- deferring the `-1` rather than sealing past + /// it, which is not the permanent damage this arm is about. + publishAt(*backend, layout, ns, RefTxnId{1, 3}, "ref_c", 3, unrelated); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->setListOmissions({layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2})}); + + Gc gc(store, kGc); + const RoundEvidence condemning = runRoundCapturing(gc, UniversePolicy::Authoritative); + ASSERT_TRUE(condemning.report.acquired_lease); + ASSERT_GT(backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + expectNoAnomalies(condemning, "hidden -1"); + + EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 3})); + EXPECT_EQ(inDegreeOf(*backend, layout, released), 0) + << "the hidden `-1` must be folded: nothing owns this blob any more"; + EXPECT_TRUE(backend->head(blobKeyOf(layout, released)).exists) + << "round pacing: the round that CONDEMNS never also deletes"; + + store->renewWatermarkOnce(); + EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, released)) + << "the blob was never reclaimed -- the hidden `-1` left it pinned by a phantom owner"; + EXPECT_TRUE(backend->head(blobKeyOf(layout, unrelated)).exists) + << "and the still-owned blob is untouched"; +} + +/// ===================== THE CROSS-NAMESPACE SHOT ===================== +/// +/// The shape that is not about walking a single namespace: it is about a namespace the store hides in +/// its ENTIRETY, never just one record inside it. +/// +/// Two namespaces share a blob. `visible` publishes it and then drops it, so the round observes `+1` +/// then `-1` and reads the blob's in-degree as zero. `hidden` also owns it -- durably, acked, readable +/// by exact key -- but the store omits its ENTIRE ref stream, so no listing mentions it. `hidden`'s own +/// publish still leaves it a real `_ckpt`, and a `_ckpt` is read by exact key, so the arithmetic walk's +/// first probe finds and folds `hidden`'s `+1` regardless of what the listing omits: the blob survives +/// on its own complete, folded frontier. +/// +/// This is NOT a duplicate of `gtest_cas_gc_frontier_gate.cpp`'s twin: that file's backend hides only a +/// hint prefix, while this one is the end-to-end LIST-liar backend from this file's own header -- the +/// distinct thing this test proves is that arithmetic intake reads a record the backend actively hides +/// from every enumeration, in the full pipeline (real pool, real recovery-shaped checkpoints), not that +/// the gate's universe/count terms hold. `gtest_cas_gc_frontier_gate.cpp` owns those terms: its +/// (3a)/(3b)/(3c) suppressor arms are what pin `universe_authoritative`, the empty-universe floor, and +/// the probe budget -- terms this fixture cannot exercise, because grounding both namespaces here makes +/// `frontier_namespaces > 0` and `universe_authoritative` true unconditionally. + +namespace +{ +/// Build the shared-blob scenario and return the manifest `visible` will drop. Both namespaces are +/// grounded with a real `_ckpt` reflecting what was actually published (`writeRecoverableCkptForRawFixture`, +/// the idiom every other test in this file uses): otherwise neither namespace has a usable checkpoint at +/// all, the round suppresses on that anomaly alone, and the scenario proves nothing about `hidden` +/// specifically. +ManifestRef buildKillShot(const std::shared_ptr & backend, const Layout & layout, + const RootNamespace & hidden, const RootNamespace & visible, + const DB::UInt128 & blob) +{ + publishAt(*backend, layout, hidden, RefTxnId{1, 1}, "kept_ref", 1, blob, /*birth=*/true); + writeRecoverableCkptForRawFixture(*backend, layout, hidden, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + publishAt(*backend, layout, visible, RefTxnId{1, 1}, "dropped_ref", 2, blob, /*birth=*/true); + const ManifestRef dropped = publishedManifest(RefTxnId{1, 1}, 2); + dropAt(*backend, layout, visible, RefTxnId{1, 2}, "dropped_ref", dropped); + writeRecoverableCkptForRawFixture(*backend, layout, visible, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + /// The whole of `hidden`'s ref stream goes invisible -- the namespace itself is what the listing + /// stops mentioning, not a record inside it. Its `_ckpt` stays readable by exact key, which is what + /// lets the arithmetic walk find and fold its birth despite the omission. + backend->setListOmissions({layout.refLogKey(fixture::fixtureLife(hidden), RefTxnId{1, 1}), + layout.refCkptKey(fixture::fixtureLife(hidden))}); + return dropped; +} +} + +/// Rounds on the PRODUCTION path -- no policy argument anywhere -- because that is the posture the +/// claim is about: the arithmetic walk's exact-key probe reaches `hidden`'s birth despite the store +/// hiding its whole stream from every listing, so the frontier it proves is complete and the blob +/// survives on its own folded in-degree, not on a caller declining to supply a universe. +TEST(CASListLiarEndToEnd, AHiddenNamespacesBirthIsFoundByExactKeyAndSavesTheBlobOnACompleteFrontier) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace hidden{"00/hidden@cas@"}; + const RootNamespace visible{"00/visible@cas@"}; + const DB::UInt128 blob(0x5ade); + + buildKillShot(backend, layout, hidden, visible, blob); + + Gc gc(store, kGc); + backend->resetCounts(); + RoundEvidence evidence; + for (int i = 0; i < 5; ++i) + { + const RoundEvidence round = runRoundCapturing(gc, UniversePolicy::kDefault); + if (round.saw_fold) + evidence = round; + store->renewWatermarkOnce(); + } + + ASSERT_TRUE(evidence.saw_fold) << "no round folded, so none published a gate verdict"; + ASSERT_GT(backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + << "the blob a hidden namespace still owns must survive"; + EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, blob)), 0u) + << "not merely still present: the blob must never even be offered for deletion"; + EXPECT_TRUE(evidence.frontier_complete) + << "`hidden`'s own `_ckpt` is read by exact key, so its frontier is provable despite the " + "listing omission -- if this is false the blob above survived on suppression instead of on " + "its own in-degree, which proves nothing about the edge"; + EXPECT_FALSE(evidence.suppress_destructive); +} + +/// The arm above asserts "nothing was deleted", which on its own does not distinguish the gate correctly +/// refusing from the round simply never deleting anything. Positive control: +/// `hidden` drops its OWN reference too (still by exact key, still hidden from every listing), so its +/// frontier is REALLY proven by the arithmetic-intake exact-key probe -- never declared so by fiat -- +/// and the blob is REALLY unreferenced by both namespaces. The round drains it. +TEST(CASListLiarEndToEnd, TheSameBlobDrainsOnceHiddenGenuinelyProvesItsOwnFrontier) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace hidden{"00/hidden@cas@"}; + const RootNamespace visible{"00/visible@cas@"}; + const DB::UInt128 blob(0x5ade); + + publishAt(*backend, layout, hidden, RefTxnId{1, 1}, "kept_ref", 1, blob, /*birth=*/true); + const ManifestRef kept = publishedManifest(RefTxnId{1, 1}, 1); + writeRecoverableCkptForRawFixture(*backend, layout, hidden, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + Gc gc(store, kGc); + + /// `hidden`'s birth is folded (and its cursor SEALED) while everything is still listed. Its + /// checkpoint proves that exact initial frontier; the real fold then makes the arithmetic + /// (cursor-relative) genesis available for what follows. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + store->renewWatermarkOnce(); + + /// NOW `hidden` drops its own reference, and ONLY THEN does its whole prefix vanish from LIST. With + /// a sealed cursor already in hand, the walk's genesis for `hidden` is arithmetic (`cursor + 1`), so + /// this drop is found and folded by exact key alone -- the arithmetic-intake mechanism this whole + /// file is about, exercised honestly rather than declared past by fiat. + dropAt(*backend, layout, hidden, RefTxnId{1, 2}, "kept_ref", kept); + advanceRecoverableCkptForRawFixture(*backend, layout, hidden, RefTxnId{1, 2}); + backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(hidden))); + + publishAt(*backend, layout, visible, RefTxnId{1, 1}, "dropped_ref", 2, blob, /*birth=*/true); + const ManifestRef dropped = publishedManifest(RefTxnId{1, 1}, 2); + dropAt(*backend, layout, visible, RefTxnId{1, 2}, "dropped_ref", dropped); + writeRecoverableCkptForRawFixture(*backend, layout, visible, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + for (int i = 0; i < 5; ++i) + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + + ASSERT_GT(backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + << "both namespaces genuinely proved their frontier and the blob is genuinely unreferenced -- " + "the round must still be able to reclaim it"; +} + +/// ===================== FSCK ===================== +/// +/// fsck runs two checkpoint-grounded passes over a namespace's ref stream. +/// +/// * `checkRefStream` walks arithmetically by exact key. An omitted-but-durable record is a +/// non-event to it, exactly as it is to the GC fold. That is the pass the arms below pin. +/// * the reachability pass uses the same catalog row and exact `_ckpt` to recover the ref table +/// without stream enumeration. +/// +/// Both arms are written against an HONEST TWIN seeded by the same code, not against hand-written +/// expectations: the claim is "identical to the truth", and a pass that quietly examined fewer records +/// would satisfy a hand-written "clean" just as well. + +TEST(CASListLiarEndToEnd, FsckArithmeticStreamAuditIsUnmovedByAHiddenMiddle) +{ + const Layout layout("p"); + const RootNamespace ns{"00/fsck@cas@"}; + + auto honest_backend = std::make_shared(); + seedFiveRecordStream(*honest_backend, layout, ns); + auto honest = openRecoveryPool(honest_backend); + const FsckReport truth = runFsck(*honest, /*detail=*/true); + + auto lying_backend = std::make_shared(); + seedFiveRecordStream(*lying_backend, layout, ns); + lying_backend->setListOmissions(hiddenMiddleOf(layout, ns)); + auto lying = openRecoveryPool(lying_backend); + const FsckReport under_lie = runFsck(*lying, /*detail=*/true); + + ASSERT_GT(lying_backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + + EXPECT_TRUE(truth.clean()) << "the oracle itself must be clean, or the comparison means nothing"; + EXPECT_GT(truth.ref_records_walked, 0u); + EXPECT_GT(truth.reachable, 0u) + << "the honest oracle must recover at least one live object, or reachability equality is vacuous"; + + /// The arithmetic pass: a hidden record is a non-event, and no finding is manufactured out of it. + EXPECT_TRUE(under_lie.clean()) + << "fsck must not manufacture a finding out of an omitted-but-durable record"; + EXPECT_EQ(under_lie.chain_broken, 0u) + << "a record the listing hid is NOT a broken chain: the walk reads it by exact key"; + EXPECT_EQ(under_lie.dangling, 0u); + EXPECT_EQ(under_lie.ref_records_walked, truth.ref_records_walked) + << "the arithmetic walk must read the SAME number of records under the lie"; + + EXPECT_EQ(under_lie.unchecked, 0u) + << "an omitted durable record must not turn a healthy checkpoint-bounded namespace unchecked"; + EXPECT_EQ(under_lie.reachable, truth.reachable) + << "the reachability recovery must observe the same exact committed frontier under the lie"; +} + +/// A hidden tail record is the silent variant of the historical residual: a LIST-driven replay could +/// return a plausible but short table. Checkpoint-bounded recovery must produce the honest table even +/// though the list omission is served. +TEST(CASListLiarEndToEnd, FsckReachabilityRecoveryMatchesTruthUnderAHiddenTailTransaction) +{ + const Layout layout("p"); + const RootNamespace ns{"00/fsck_tail@cas@"}; + /// Stage B (Task 4-C): no pin needed -- `seedFiveRecordStream` below calls `publishAt` (draining + /// into `writeRefLogTxnRaw`), which admits `ns` into each of the two independent backends' own + /// catalogs itself. + + auto honest_backend = std::make_shared(); + seedFiveRecordStream(*honest_backend, layout, ns); + auto honest = openRecoveryPool(honest_backend); + const FsckReport truth = runFsck(*honest, /*detail=*/true); + + auto lying_backend = std::make_shared(); + seedFiveRecordStream(*lying_backend, layout, ns); + lying_backend->setListOmissions({layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 5})}); + auto lying = openRecoveryPool(lying_backend); + const FsckReport under_lie = runFsck(*lying, /*detail=*/true); + + ASSERT_GT(lying_backend->holesServed(), 0u) + << "the omission was never actually served -- the test would pass vacuously"; + EXPECT_GT(truth.reachable, 0u) + << "the honest oracle must recover at least one live object, or reachability equality is vacuous"; + + /// The arithmetic pass is unmoved here too: it probes `{1,5}` by exact key and finds it. + EXPECT_EQ(under_lie.chain_broken, 0u); + EXPECT_EQ(under_lie.ref_records_walked, truth.ref_records_walked) + << "the arithmetic walk reads the hidden tail by exact key, so it counts the same records"; + + EXPECT_EQ(under_lie.unchecked, 0u) + << "a LIST omission must not make a checkpoint-bounded namespace unchecked"; + EXPECT_EQ(under_lie.reachable, truth.reachable) + << "the exact committed frontier must include the hidden tail transaction"; +} diff --git a/src/Disks/tests/gtest_cas_manifest_id.cpp b/src/Disks/tests/gtest_cas_manifest_id.cpp new file mode 100644 index 000000000000..8be89457de93 --- /dev/null +++ b/src/Disks/tests/gtest_cas_manifest_id.cpp @@ -0,0 +1,86 @@ +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ + +ManifestRef ref(uint64_t w, uint64_t seq, uint64_t m) +{ + return ManifestRef{w, seq, static_cast(m)}; +} + +ManifestId id(const char * ns, uint64_t w, uint64_t seq, uint64_t m) +{ + return ManifestId{RootNamespace(ns), ref(w, seq, m)}; +} + +} + +TEST(CASManifestId, RefEqualityAndOrdering) +{ + EXPECT_EQ(ref(1, 2, 3), ref(1, 2, 3)); + EXPECT_NE(ref(1, 2, 3), ref(1, 2, 4)); + /// Strict total order: distinct by manifest_ordinal, then build_sequence, then writer_epoch. + EXPECT_LT(ref(1, 2, 3), ref(1, 2, 4)); + EXPECT_LT(ref(1, 2, 9), ref(1, 3, 0)); + EXPECT_LT(ref(1, 9, 9), ref(2, 0, 0)); + EXPECT_FALSE(ref(1, 2, 3) < ref(1, 2, 3)); +} + +TEST(CASManifestId, IdIsNamespaceQualified) +{ + /// Same ref tuple, different namespace => DIFFERENT ids (the SabotageKeyByRefNotId guard). + EXPECT_NE(id("nsA", 1, 1, 1), id("nsB", 1, 1, 1)); + EXPECT_EQ(id("nsA", 1, 1, 1), id("nsA", 1, 1, 1)); + /// Ordering separates by namespace first. + EXPECT_LT(id("nsA", 9, 9, 9), id("nsB", 0, 0, 0)); +} + +TEST(CASManifestId, UsableAsMapAndSetKey) +{ + std::set s; + s.insert(id("nsA", 1, 1, 1)); + s.insert(id("nsB", 1, 1, 1)); /// distinct namespace -> distinct key + s.insert(id("nsA", 1, 1, 1)); /// duplicate -> no growth + EXPECT_EQ(s.size(), 2u); + + std::map m; + m[ref(1, 1, 1)] = 10; + m[ref(1, 1, 2)] = 20; + EXPECT_EQ(m.size(), 2u); + EXPECT_EQ(m[ref(1, 1, 1)], 10); +} + +TEST(CASManifestId, UsableInUnorderedContainers) +{ + /// std::hash / std::hash let the read-path cache (Phase 1c) and GC use + /// unordered_map/set. Equal values => equal hash; distinct values => (overwhelmingly) distinct. + std::unordered_set s; + s.insert(id("nsA", 1, 1, 1)); + s.insert(id("nsB", 1, 1, 1)); /// distinct namespace -> distinct key + s.insert(id("nsA", 1, 1, 1)); /// duplicate -> no growth + EXPECT_EQ(s.size(), 2u); + + std::unordered_map m; + m[ref(1, 1, 1)] = 10; + m[ref(1, 1, 1)] = 11; /// same key overwrites + m[ref(1, 1, 2)] = 20; + EXPECT_EQ(m.size(), 2u); + EXPECT_EQ(m.at(ref(1, 1, 1)), 11); + + EXPECT_EQ(std::hash{}(id("nsA", 1, 1, 1)), std::hash{}(id("nsA", 1, 1, 1))); +} + +TEST(CASManifestId, ManifestOrdinalFileName) +{ + EXPECT_EQ(manifestOrdinalFileName(1), "000001.zst"); + EXPECT_EQ(manifestOrdinalFileName(999999), "999999.zst"); + EXPECT_THROW(manifestOrdinalFileName(0), DB::Exception); + EXPECT_THROW(manifestOrdinalFileName(1000000), DB::Exception); +} diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp new file mode 100644 index 000000000000..55af475d2048 --- /dev/null +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -0,0 +1,1947 @@ +#include +#include "cas_test_helpers.h" +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; + extern const int CORRUPTED_DATA; + extern const int FILE_DOESNT_EXIST; + extern const int LOGICAL_ERROR; +} + +namespace ProfileEvents +{ + extern const Event CASMountLeaseLost; + extern const Event CASMountExclusivityViolation; + extern const Event CASMountRenewalAttempts; + extern const Event CASMountRenewalRetries; +} + +using namespace DB::Cas; + +namespace +{ + +const ObserveRefCatalog & emptyCatalogObservation() +{ + static const ObserveRefCatalog observe = [] { return RefCatalog{}; }; + return observe; +} + +RefCatalog catalogOwning(const String & ns, NsState state) +{ + CatalogEntry entry{.ns = RootNamespace{ns}, .state = state, .incarnation = UInt128{42}}; + if (state == NsState::Creating) + entry.creator = CreatorFence{.server_root_id = "root/x", .writer_epoch = 1, .fence_generation = 1}; + return RefCatalog{.entries = {std::move(entry)}}; +} + +void renewKeeperOrThrow(MountLeaseKeeper & keeper) +{ + const MountRenewResult result = keeper.renew( + CasRequestBudget{.attempt_timeout_ms = 1, .operation_deadline_ms = 100, .max_attempts = 2, + .lease_safety_margin_ms = 0, .retry_initial_backoff_ms = 0, .retry_max_backoff_ms = 0}, + MountRenewOperationEnvironment{}); + if (result.outcome == MountRenewOutcome::Terminal) + std::rethrow_exception(result.failure); + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); +} + +class OwnerConflictRevealsManifestBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::putIfAbsent; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (!fired && key == "p/gc/server-roots/root/x/owner") + { + fired = true; + InMemoryBackend::putIfAbsent("p/cas/manifests/root/x/table/debris", "x"); + return {PutOutcome::PreconditionFailed, {}}; + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + bool fired = false; +}; + +class EpochConflictRevealsManifestBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::casPut; + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (!fired && key == "p/gc/server-roots/root/x/epoch") + { + fired = true; + /// Install the competing allocator's winning epoch before revealing owned work. The + /// retry must not accept that now-present epoch without rechecking the entire emptiness + /// bundle that authorized the original absent-epoch attempt. + const CasResult winner = InMemoryBackend::casPut( + key, encodeServerEpoch(ServerEpoch{.next_writer_epoch = 2}), expected, meta); + winner_installed = winner.outcome == CasOutcome::Committed; + InMemoryBackend::putIfAbsent("p/cas/manifests/root/x/table/debris", "x"); + return {CasOutcome::Conflict, {}}; + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + + bool fired = false; + bool winner_installed = false; +}; + +class RenewalLogBackend final : public InMemoryBackend +{ +public: + using InMemoryBackend::putOverwrite; + + bool throw_before_next_overwrite = false; + + void armBlockedRetry() + { + std::lock_guard lock(mutex); + blocked_retry_armed = true; + renewal_puts = 0; + second_put_arrived = false; + release_second_put = false; + } + + bool waitForSecondPut() + { + std::unique_lock lock(mutex); + return cv.wait_for(lock, std::chrono::seconds(2), [&] { return second_put_arrived; }); + } + + void releaseSecondPut() + { + std::lock_guard lock(mutex); + release_second_put = true; + cv.notify_all(); + } + + PutResult putOverwrite( + const String & key, + const String & bytes, + const Token & expected, + const ObjectMeta & meta) override + { + { + std::unique_lock lock(mutex); + if (blocked_retry_armed) + { + ++renewal_puts; + if (renewal_puts == 1) + throw Poco::TimeoutException("injected renewal timeout before blocked retry"); + if (renewal_puts == 2) + { + second_put_arrived = true; + cv.notify_all(); + if (!cv.wait_for(lock, std::chrono::seconds(20), [&] { return release_second_put; })) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "blocked renewal retry was not released"); + blocked_retry_armed = false; + } + } + } + if (std::exchange(throw_before_next_overwrite, false)) + throw Poco::TimeoutException("injected renewal timeout before commit"); + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + +private: + std::mutex mutex; + std::condition_variable cv; + bool blocked_retry_armed = false; + uint32_t renewal_puts = 0; + bool second_put_arrived = false; + bool release_second_put = false; +}; + +class BlockingRenewalDebugChannel final : public Poco::Channel +{ +public: + void log(const Poco::Message & message) override + { + if (message.getText().find("physical retry attempt") == String::npos) + return; + std::unique_lock lock(mutex); + cv.wait_for(lock, std::chrono::seconds(20), [&] { return released; }); + } + + void unblock() + { + std::lock_guard lock(mutex); + released = true; + cv.notify_all(); + } + +private: + std::mutex mutex; + std::condition_variable cv; + bool released = false; +}; + +class ScopedBlockingRenewalDebugLog +{ +public: + ScopedBlockingRenewalDebugLog() + : logger(getLogger("CasMountLeaseKeeper")) + , channel(new BlockingRenewalDebugChannel) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("debug"); + } + + ~ScopedBlockingRenewalDebugLog() + { + channel->unblock(); + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + void release() { channel->unblock(); } + +private: + LoggerPtr logger; + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +class ScopedRenewalLogCapture +{ +public: + explicit ScopedRenewalLogCapture(const String & level) + : logger(getLogger("CasMountLeaseKeeper")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel(level); + } + + ~ScopedRenewalLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +size_t countRenewalLogText(const String & haystack, std::string_view needle) +{ + size_t count = 0; + for (size_t pos = 0; (pos = haystack.find(needle, pos)) != String::npos; pos += needle.size()) + ++count; + return count; +} + +CasRequestBudget renewalLogBudget(uint32_t max_attempts = 2) +{ + return CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = max_attempts, + .lease_safety_margin_ms = 20, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }; +} + +} + +TEST(CASMountAudit, RenewalDefaultLogsAreBounded) +{ + const auto open_store = [](const std::shared_ptr & backend, uint64_t & boot_ms, const String & prefix) + { + return Pool::open(backend, PoolConfig{ + .pool_prefix = prefix, + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = renewalLogBudget(), + .boot_ms_fn = [&] { return boot_ms; }, + }); + }; + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = open_store(backend, boot_ms, "renewal-log-silent"); + ScopedRenewalLogCapture capture("information"); + EXPECT_NO_THROW(store->renewWatermarkOnce()); + EXPECT_EQ(countRenewalLogText(capture.captured(), "CAS mount renewal"), 0u); + } + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = open_store(backend, boot_ms, "renewal-log-recovered"); + ScopedRenewalLogCapture capture("information"); + backend->throw_before_next_overwrite = true; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + const String output = capture.captured(); + EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 2u) << output; + EXPECT_EQ(countRenewalLogText(output, "entered retry"), 1u) << output; + EXPECT_EQ(countRenewalLogText(output, "recovered"), 1u) << output; + EXPECT_EQ(countRenewalLogText(output, "physical retry attempt"), 0u) << output; + } + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = open_store(backend, boot_ms, "renewal-log-debug"); + ScopedRenewalLogCapture capture("debug"); + backend->throw_before_next_overwrite = true; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + EXPECT_EQ(countRenewalLogText(capture.captured(), "physical retry attempt 2"), 1u); + } + + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = open_store(backend, boot_ms, "renewal-log-fenced"); + ScopedRenewalLogCapture capture("information"); + boot_ms = 1071; + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + const String output = capture.captured(); + EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 1u) << output; + EXPECT_EQ(countRenewalLogText(output, "fenced"), 1u) << output; + EXPECT_EQ(countRenewalLogText(output, "entered retry"), 0u) << output; + } +} + +TEST(CASMountAudit, PhysicalRetryCannotBeDelayedByDebugLogging) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "renewal-debug-order", + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = renewalLogBudget(), + .boot_ms_fn = [&] { return boot_ms; }, + }); + + const uint64_t attempts_before + = ProfileEvents::global_counters[ProfileEvents::CASMountRenewalAttempts].load(); + const uint64_t retries_before + = ProfileEvents::global_counters[ProfileEvents::CASMountRenewalRetries].load(); + backend->armBlockedRetry(); + ScopedBlockingRenewalDebugLog blocked_log; + auto renewal = std::async(std::launch::async, [&] { store->renewWatermarkOnce(); }); + + const bool retry_reached_backend = backend->waitForSecondPut(); + EXPECT_TRUE(retry_reached_backend) + << "diagnostic logging after retry admission must not delay the backend request"; + EXPECT_EQ( + ProfileEvents::global_counters[ProfileEvents::CASMountRenewalAttempts].load(), + attempts_before + 2) + << "physical attempt visibility must precede completion of the in-flight retry"; + EXPECT_EQ( + ProfileEvents::global_counters[ProfileEvents::CASMountRenewalRetries].load(), + retries_before + 1); + + blocked_log.release(); + backend->releaseSecondPut(); + EXPECT_NO_THROW(renewal.get()); +} + +TEST(CASServerRootId, ValidationAcceptsCleanPathsRejectsBad) +{ + EXPECT_NO_THROW(validateServerRootId("replica-a")); + EXPECT_NO_THROW(validateServerRootId("shard-01/replica-a")); + EXPECT_THROW(validateServerRootId(""), DB::Exception); + EXPECT_THROW(validateServerRootId("/replica"), DB::Exception); + EXPECT_THROW(validateServerRootId("replica/"), DB::Exception); + EXPECT_THROW(validateServerRootId("a//b"), DB::Exception); + EXPECT_THROW(validateServerRootId("a/../b"), DB::Exception); + EXPECT_THROW(validateServerRootId("a/_files/b"), DB::Exception); +} + +TEST(CASServerRoot, KeysAndCodecsRoundTrip) +{ + Layout layout("p"); + + /// Layout keys under gc/server-roots//. + EXPECT_EQ(layout.serverRootPrefix("replica-a"), "p/gc/server-roots/replica-a/"); + EXPECT_EQ(layout.ownerKey("replica-a"), "p/gc/server-roots/replica-a/owner"); + EXPECT_EQ(layout.epochKey("replica-a"), "p/gc/server-roots/replica-a/epoch"); + EXPECT_EQ(layout.mountKey("replica-a"), "p/gc/server-roots/replica-a/mount"); + + /// Owner round-trip. + { + OwnerObject o; + o.server_uuid = (UInt128(0x0123456789abcdefULL) << 64) | UInt128(0xfedcba9876543210ULL); + const OwnerObject back = decodeOwner(encodeOwner(o)); + EXPECT_EQ(back.server_uuid, o.server_uuid); + } + + /// ServerEpoch round-trip. + { + ServerEpoch e; + e.next_writer_epoch = 4242; + const ServerEpoch back = decodeServerEpoch(encodeServerEpoch(e)); + EXPECT_EQ(back.next_writer_epoch, e.next_writer_epoch); + } + + /// MountLease round-trip. + { + MountLease m; + m.server_uuid = (UInt128(0xdeadbeefcafef00dULL) << 64) | UInt128(0x0011223344556677ULL); + m.writer_epoch = 7; + m.hostname = "host-1.example.com"; + m.pid = 12345; + m.started_at_ms = 1700000000000ULL; + m.seq = 99; + m.expires_at_ms = 1700000030000ULL; + m.write_attempt_id = UInt128{1}; + const MountLease back = decodeMountLease(encodeMountLease(m)); + EXPECT_EQ(back.server_uuid, m.server_uuid); + EXPECT_EQ(back.writer_epoch, m.writer_epoch); + EXPECT_EQ(back.hostname, m.hostname); + EXPECT_EQ(back.pid, m.pid); + EXPECT_EQ(back.started_at_ms, m.started_at_ms); + EXPECT_EQ(back.seq, m.seq); + EXPECT_EQ(back.expires_at_ms, m.expires_at_ms); + } + + /// Fail-closed decode on garbage bytes. + EXPECT_THROW(decodeOwner("not-a-proto-with-magic"), DB::Exception); + EXPECT_THROW(decodeServerEpoch(""), DB::Exception); + EXPECT_THROW(decodeMountLease(""), DB::Exception); +} + +TEST(CASServerRootClaim, OwnerStickyAndForeignFailsClosed) +{ + auto b = std::make_shared(); + Layout l("p"); + EXPECT_NO_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation())); // fresh empty root → claim + EXPECT_NO_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation())); // same uuid → ok + try + { + claimOwnerOrThrow(*b, l, "r", UInt128(2), emptyCatalogObservation()); + FAIL() << "expected a foreign owner to fail closed"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_NE(e.message().find(""), String::npos) << e.message(); + } +} + +TEST(CASServerRootClaim, TombstonedSameOwnerFailsClosed) +{ + auto b = std::make_shared(); + Layout l("p"); + b->putIfAbsent(l.ownerKey("r"), encodeOwner(OwnerObject{ + .server_uuid = UInt128(1), + .retired_at_ms = 1752537600000ULL, + })); + + try + { + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + FAIL() << "expected a tombstoned owner claim to fail closed"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_NE(e.message().find("decommissioned"), String::npos) << e.message(); + EXPECT_EQ(e.message().find("owned by a different server"), String::npos) << e.message(); + } +} + +TEST(CASServerRootEpoch, AllocatorIsMonotoneAndSurvivesMountConcept) +{ + auto b = std::make_shared(); + Layout l("r"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + const uint64_t e1 = allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); + const uint64_t e2 = allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); + EXPECT_GE(e1, 1u); // 0 is a reserved sentinel + EXPECT_GT(e2, e1); // strictly increasing + + /// Deleting the (separate) mount object must NOT reset the epoch. No mount has been written in + /// Task 4, so deleteExact of a non-existent mount is a NotFound no-op that touches nothing. + const auto del = b->deleteExact(l.mountKey("r"), b->head(l.mountKey("r")).token); + EXPECT_EQ(del.kind, DeleteOutcome::Kind::NotFound); + EXPECT_GT(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), e2); +} + +/// Phase C (spec rev.4): an ABSENT epoch object over a PRESENT mount object means durable epoch +/// state was lost while a mount is live/recent — re-minting epoch 1 there is how a same-(uuid, +/// epoch) twin is born. Refuse. +TEST(CASMount, EpochRemintOverExistingMountRefuses) +{ + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, + MountClaimResult::Claimed); + /// The epoch object is ABSENT (never created in this sequence) while the mount exists: + EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); /// CORRUPTED_DATA +} + +TEST(CASMount, EpochRemintAuthoritativeAbsenceMints) +{ + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// fresh root: both control objects absent + EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present now: normal CAS bump, no probe +} + +/// The probe outcome gates the mint: anything short of authoritative KeyAbsent fails closed. +TEST(CASMount, EpochRemintIndeterminateProbeFailsClosed) +{ + class IndeterminateProbeBackend final : public InMemoryBackend + { + public: + SentinelProbeResult probeSentinelRaw(const String &) override + { + return {.outcome = ProbeOutcome::Indeterminate, .body = std::nullopt}; + } + }; + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); +} + +/// Decommission over a TERMINAL (expired/fenced) mount with a lost epoch object proceeds and mints +/// an epoch DISTINCT from the surviving mount's — the same-pair state is unrepresentable. +TEST(CASMount, DecommissionRemintOverTerminalMountMintsDistinctEpoch) +{ + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/3, /*now_ms=*/1000, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + /// now_ms=5000: the ttl_ms=100 lease above is long expired -> terminal. + EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/5000, emptyCatalogObservation()), 4u); +} + +/// Decommission over a LIVE mount with a lost epoch refuses — the blind bypass would recreate the +/// forbidden pair (codex round-3 finding 1) and defeat CASDecommission.RefusesLiveMember. +TEST(CASMount, DecommissionRemintOverLiveMountRefuses) +{ + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, + MountClaimResult::Claimed); + EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/2000, emptyCatalogObservation()), + DB::Exception); /// ABORTED: live member +} + +/// The steady-state path (epoch object PRESENT) must never pay the probe — pins the zero +/// normal-path cost the spec claims. +TEST(CASMount, EpochBumpWithPresentEpochIssuesNoProbe) +{ + class ProbeCountingBackend final : public InMemoryBackend + { + public: + int probes = 0; + SentinelProbeResult probeSentinelRaw(const String & k) override + { + ++probes; + return InMemoryBackend::probeSentinelRaw(k); + } + }; + auto b = std::make_shared(); + Layout l("p"); + claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// bootstrap: ONE probe (absent-epoch branch) + const int probes_after_bootstrap = b->probes; + EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present: normal CAS bump... + EXPECT_EQ(b->probes, probes_after_bootstrap) << "...must not probe the mount key"; +} + +TEST(CASServerRootClaim, MissingOwnerOverNonEmptyRootIsCorrupted) +{ + auto b = std::make_shared(); + Layout l("p"); + /// Simulate existing data without an owner (identity lost): plant a key under roots//. + b->putIfAbsent(l.serverRootDataPrefix("r") + "some-data", "x"); + EXPECT_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()), DB::Exception); +} + +TEST(CASServerRootSafety, EveryCatalogLifecycleStateBlocksOwnerAndEpochRecreation) +{ + const Layout layout("p"); + for (const NsState state : {NsState::Creating, NsState::Live, NsState::Removing}) + { + RefCatalog catalog = catalogOwning("root/x/table", state); + const ObserveRefCatalog observe = [catalog] { return catalog; }; + + InMemoryBackend owner_backend; + EXPECT_THROW(claimOwnerOrThrow(owner_backend, layout, "root/x", UInt128{1}, observe), DB::Exception); + EXPECT_FALSE(owner_backend.head(layout.ownerKey("root/x")).exists); + + InMemoryBackend epoch_backend; + EXPECT_THROW(allocateWriterEpoch( + epoch_backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, observe), DB::Exception); + EXPECT_FALSE(epoch_backend.head(layout.epochKey("root/x")).exists); + } +} + +TEST(CASServerRootSafety, OwnershipUsesAPathComponentBoundary) +{ + InMemoryBackend backend; + const Layout layout("p"); + EXPECT_TRUE(serverRootSubtreeEmpty( + backend, layout, "root/x", catalogOwning("root/xy/table", NsState::Live))); + EXPECT_FALSE(serverRootSubtreeEmpty( + backend, layout, "root/x", catalogOwning("root/x/table", NsState::Live))); +} + +TEST(CASServerRootSafety, OpaqueStreamAndStateDebrisAloneDoesNotBlockRecreation) +{ + InMemoryBackend backend; + const Layout layout("p"); + const NamespaceLifeId dead = NamespaceLifeId::fromCatalogEntry(RootNamespace{"unowned"}, UInt128{99}); + ASSERT_EQ(backend.putIfAbsent(layout.refLogKey(dead, RefTxnId{1, 1}), "debris").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(dead), "debris").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(layout.namespaceFileKey(dead, "f"), "debris").outcome, PutOutcome::Done); + + EXPECT_NO_THROW(claimOwnerOrThrow(backend, layout, "root/x", UInt128{1}, emptyCatalogObservation())); + EXPECT_EQ(allocateWriterEpoch( + backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); +} + +TEST(CASServerRootSafety, ManifestAndLooseRootDebrisStillBlockRecreation) +{ + const Layout layout("p"); + for (const String & key : { + layout.casManifestsServerPrefix("root/x") + "table/debris", + layout.serverRootDataPrefix("root/x") + "loose"}) + { + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent(key, "x").outcome, PutOutcome::Done); + EXPECT_THROW(claimOwnerOrThrow( + backend, layout, "root/x", UInt128{1}, emptyCatalogObservation()), DB::Exception); + EXPECT_THROW(allocateWriterEpoch( + backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + } +} + +TEST(CASServerRootSafety, UnreadableCatalogNeverFallsBackToPhysicalGuesses) +{ + InMemoryBackend backend; + const Layout layout("p"); + const ObserveRefCatalog unreadable = []() -> RefCatalog + { + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected unreadable catalog"); + }; + EXPECT_THROW(claimOwnerOrThrow(backend, layout, "root/x", UInt128{1}, unreadable), DB::Exception); + EXPECT_THROW(allocateWriterEpoch( + backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, unreadable), DB::Exception); + EXPECT_FALSE(backend.head(layout.ownerKey("root/x")).exists); + EXPECT_FALSE(backend.head(layout.epochKey("root/x")).exists); +} + +TEST(CASServerRootSafety, OwnerConflictRecomputesTheWholeEmptinessBundle) +{ + OwnerConflictRevealsManifestBackend backend; + const Layout layout("p"); + EXPECT_THROW(claimOwnerOrThrow( + backend, layout, "root/x", UInt128{1}, emptyCatalogObservation()), DB::Exception); + EXPECT_TRUE(backend.fired); + EXPECT_FALSE(backend.head(layout.ownerKey("root/x")).exists); +} + +TEST(CASServerRootSafety, EpochConflictRecomputesTheWholeEmptinessBundle) +{ + EpochConflictRevealsManifestBackend backend; + const Layout layout("p"); + EXPECT_THROW(allocateWriterEpoch( + backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + EXPECT_TRUE(backend.fired); + ASSERT_TRUE(backend.winner_installed); + const auto epoch = backend.get(layout.epochKey("root/x")); + ASSERT_TRUE(epoch.has_value()); + EXPECT_EQ(decodeServerEpoch(epoch->bytes).next_writer_epoch, 2u) + << "the rejected allocator must not consume an epoch from the conflict winner"; +} + +TEST(CASMountLease, AbsentClaimThenRenewBumpsSeq) +{ + auto b = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + auto r = claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100); + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, + [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); + k.start(); + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).seq, 1u); + renewKeeperOrThrow(k); + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).seq, 2u); +} + +TEST(CASMountLease, HolderBodiesMintFreshAttemptIdsAndFenceCopiesIt) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 7, now, 100).kind, MountClaimResult::Claimed); + const String key = layout.mountKey("r"); + const MountLease claimed = decodeMountLease(backend->get(key)->bytes); + + MountLeaseKeeper keeper(backend, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), [&] { return now; }, + [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); + keeper.start(); + renewKeeperOrThrow(keeper); + const MountLease renewed = decodeMountLease(backend->get(key)->bytes); + EXPECT_NE(claimed.write_attempt_id, UInt128{}); + EXPECT_NE(renewed.write_attempt_id, UInt128{}); + EXPECT_NE(claimed.write_attempt_id, renewed.write_attempt_id); + + auto observed = backend->get(key); + ASSERT_TRUE(observed.has_value()); + MountLease fenced = decodeMountLease(observed->bytes); + fenced.gc_fenced = true; + ++fenced.seq; + ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), observed->token).outcome, PutOutcome::Done); + EXPECT_EQ(decodeMountLease(backend->get(key)->bytes).write_attempt_id, renewed.write_attempt_id); +} + +TEST(CASMountLease, ReclaimAndSuccessorBodiesMintNewAttemptIds) +{ + auto backend = std::make_shared(); + Layout layout("p"); + const String key = layout.mountKey("r"); + ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 7, 1000, 100).kind, MountClaimResult::Claimed); + const MountLease first = decodeMountLease(backend->get(key)->bytes); + + auto observed = backend->get(key); + ASSERT_TRUE(observed.has_value()); + MountLease fenced = decodeMountLease(observed->bytes); + fenced.gc_fenced = true; + ++fenced.seq; + ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), observed->token).outcome, PutOutcome::Done); + const MountLease fence = decodeMountLease(backend->get(key)->bytes); + EXPECT_EQ(fence.write_attempt_id, first.write_attempt_id); + + ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 8, 2000, 100).kind, MountClaimResult::Claimed); + const MountLease successor = decodeMountLease(backend->get(key)->bytes); + EXPECT_NE(successor.write_attempt_id, first.write_attempt_id); + EXPECT_NE(successor.write_attempt_id, UInt128{}); +} + +/// STID 3982-3b48: `rm -rf` of the pool dir under a live mount deletes the mount slot object out from +/// under a running keeper. The next synchronous renewal must return terminal WITHOUT constructing a +/// `LOGICAL_ERROR` -- that aborts debug/ASan builds at +/// exception construction, and there is no foreign writer here to fail closed against, only an +/// environmental condition. +TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) +{ + auto b = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, + [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); + k.start(); + + const String mount_key = l.mountKey("r"); + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); /// NOLINT(clang-analyzer-deadcode.DeadStores) + + /// Simulate `rm -rf` of the backing store: the mount slot object is gone, but the keeper still + /// holds a (now stale) token for it. + ASSERT_EQ(b->deleteExact(mount_key, b->head(mount_key).token).kind, DeleteOutcome::Kind::Deleted); + + try + { + renewKeeperOrThrow(k); + FAIL() << "renew against a vanished mount object must throw"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::FILE_DOESNT_EXIST) << e.message(); + EXPECT_NE(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + } + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) + << "keeper classification is metric-free; the runtime records operational loss"; +} + +/// STID 3982-3b48 (part 1b): the terminal/clean-release counterpart to the renewal fix above. When +/// the backing store vanishes (`rm -rf` of the pool dir), the renewal side already stops non-fatally +/// (see the previous test); teardown then runs the terminal release (`stop()` -> `terminate()`), +/// which used to unconditionally throw `LOGICAL_ERROR` once the token-guarded farewell PUT observed +/// an absent object. The desired end state of a release ("no live lease object") is already true, so +/// this must be a no-op, never a `LOGICAL_ERROR` (which aborts debug/ASan builds). +/// +/// Driven WITHOUT a prior failed renew, so the count is deterministic: this is the only place along +/// this path that increments `CASMountLeaseLost`, so we expect exactly +1 (not +2, since renewal was +/// never invoked here). +TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) +{ + auto b = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, + [] { return uint64_t{0}; }); + k.start(); + + const String mount_key = l.mountKey("r"); + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + + /// Simulate `rm -rf` of the backing store: the mount slot object is gone before we ever attempt + /// a renewal, so `terminate()`'s token-guarded farewell PUT is the first thing to observe it. + ASSERT_EQ(b->deleteExact(mount_key, b->head(mount_key).token).kind, DeleteOutcome::Kind::Deleted); + + EXPECT_NO_THROW(k.release()) + << "clean release against a vanished store must be a no-op, not a LOGICAL_ERROR abort"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before); +} + +/// rev.6: a bare `claimMount` (no `proven_dead_token`) NEVER reclaims a same-uuid, different-epoch +/// lease off a wall-clock-looking-expired stamp — only `claimMountAwaitingExpiry`'s observation loop +/// can turn that into a reclaim. Renamed from `...ExpiredReclaims` to describe the corrected behavior. +TEST(CASMountLease, SameUuidLiveFailsForeignFailsExpiredStillLiveDoubleStart) +{ + auto b = std::make_shared(); + Layout l("p"); + claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100); // A live until 1100 + // same uuid, lease still live → double-start guard: + EXPECT_EQ(claimMount(*b, l, "r", UInt128(1), 8, 1050, 100).kind, MountClaimResult::LiveDoubleStart); + // foreign uuid, even after expiry → fail closed: + EXPECT_EQ(claimMount(*b, l, "r", UInt128(2), 1, 1200, 100).kind, MountClaimResult::ForeignOwner); + // same uuid, even after the stamp LOOKS expired on our wall clock → still LiveDoubleStart: no + // proven_dead_token was supplied, so there is no certificate of death to reclaim on. + EXPECT_EQ(claimMount(*b, l, "r", UInt128(1), 9, 1200, 100).kind, MountClaimResult::LiveDoubleStart); +} + +TEST(CASMountMessage, DoubleStartTextHasIdentityAndRemediation) +{ + MountLease m; + m.server_uuid = (UInt128(0xdeadbeefcafef00dULL) << 64) | UInt128(0x0011223344556677ULL); + m.writer_epoch = 7; + m.hostname = "host-9.example.com"; + m.pid = 4242; + m.seq = 13; + m.expires_at_ms = 1700000030000ULL; + + const std::string msg = mountDoubleStartMessage("replica-a", m); + + /// Identity / existing-holder fields. + EXPECT_NE(msg.find(""), std::string::npos); + EXPECT_NE(msg.find("'replica-a'"), std::string::npos); + EXPECT_NE(msg.find("hostname=host-9.example.com"), std::string::npos); + EXPECT_NE(msg.find("pid=4242"), std::string::npos); + EXPECT_NE(msg.find("last_seq=13"), std::string::npos); + EXPECT_NE(msg.find("expires_at_ms=1700000030000"), std::string::npos); + /// New wait-aware remediation (this server already waited; the lease kept being renewed). + EXPECT_NE(msg.find("waited"), std::string::npos); + EXPECT_NE(msg.find("unique"), std::string::npos); + EXPECT_NE(msg.find("reclaim the mount on restart"), std::string::npos); + EXPECT_NE(msg.find("uuid file"), std::string::npos); + /// Clock-skew caveat + manual mount-object delete escape hatch. + EXPECT_NE(msg.find("CLOCK SKEW"), std::string::npos); + EXPECT_NE(msg.find("NTP"), std::string::npos); + EXPECT_NE(msg.find("manually delete the mount"), std::string::npos); + EXPECT_NE(msg.find("gc/server-roots/replica-a/mount"), std::string::npos); +} + +/// rev.6: a stamped `expires_at_ms` that already looks past-due on our wall clock must NOT shortcut +/// the observation wait — the old "instant, zero-sleep" reclaim this test name described was exactly +/// the cross-node wall-clock trust rev.6 removes. Renamed to describe the CORRECTED behavior: the +/// wall-clock-looking-expired stamp buys nothing, the full threshold is still observed. +TEST(CASMountAwaitExpiry, PastExpiryStillPaysTheFullObservationThreshold) +{ + auto b = std::make_shared(); + Layout l("p"); + /// A prior incarnation (uuid=1, epoch=7) claimed a lease live until 1100. + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + + uint64_t wall = 1200; // already past 1100 on wall clock — irrelevant to the decision + uint64_t mono = 0; + int sleeps = 0; + auto now_fn = [&] { return wall; }; + auto mono_fn = [&] { return mono; }; + auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; ++sleeps; }; + + const auto r = claimMountAwaitingExpiry( + *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + EXPECT_GT(sleeps, 0); // NOT instant — no wall-clock trust + EXPECT_GE(mono, 100 + 100 / 20 + 25); // full observation threshold paid + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 8u); // reclaimed as us +} + +TEST(CASMountAwaitExpiry, FutureExpiryReclaimsAfterClockAdvances) +{ + auto b = std::make_shared(); + Layout l("p"); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + + uint64_t wall = 1000; // lease looks live until 1100, holder does NOT renew + uint64_t mono = 0; + auto now_fn = [&] { return wall; }; + auto mono_fn = [&] { return mono; }; + auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; }; + + const auto r = claimMountAwaitingExpiry( + *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 50, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + const auto body = decodeMountLease(b->get(l.mountKey("r"))->bytes); + EXPECT_EQ(body.writer_epoch, 8u); + EXPECT_EQ(body.seq, 2u); // reclaim continues seq (prev 1 + 1) +} + +/// rev.6: a genuinely live twin now times out via BOUNDED OBSERVATION RESTARTS (its every renewal +/// bumps the write-token, forcing a restart each poll), never via a wall-clock deadline. +TEST(CASMountAwaitExpiry, LiveRenewingTwinTimesOutAsDoubleStart) +{ + auto b = std::make_shared(); + Layout l("p"); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + + uint64_t wall = 1000; + uint64_t mono = 0; + auto now_fn = [&] { return wall; }; + auto mono_fn = [&] { return mono; }; + /// Each poll: both clocks advance AND the live holder (uuid=1, epoch=7) renews its own lease — + /// the observed write-token changes on EVERY poll, forcing a restart every time. + auto sleep_fn = [&](uint64_t ms) + { + wall += ms; + mono += ms; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, wall, 100).kind, MountClaimResult::Claimed); + }; + + const auto r = claimMountAwaitingExpiry( + *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart); + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 7u); // still the holder's +} + +namespace +{ +/// fix-round F5 harness: makes the mount key vanish to EVERY `get()`, unconditionally, while the real +/// underlying object stays put -- forcing `claimMount`'s own internal GET to take the absent-slot race +/// branch every call (its `putIfAbsent` then fails against the real, still-present object, returning +/// `LiveDoubleStart` with no token -- fix-round F8 leaves `.token` unset on exactly this branch, since +/// no re-read was done). That in turn forces `claimMountAwaitingExpiry`'s F8 fallback re-GET, which +/// ALSO sees the slot as vanished -- deterministically reproducing "the slot vanished between +/// claimMount's own GET and ours" on EVERY loop iteration, not just a lucky one-shot race. +class AlwaysVanishesBackend final : public DB::Cas::Backend +{ +public: + explicit AlwaysVanishesBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} + String watched_key; + + std::optional get(const String & k, DB::Cas::Range r) override + { + if (k == watched_key) + return std::nullopt; + return inner->get(k, r); + } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + std::shared_ptr inner; +}; +} + +/// fix-round F5 (author-review: `!got -> continue` in the observation loop, with no sleep and outside +/// the restart limit, spins `get`/`claimMount`/`put` at backend RTT under persistent slot churn). A +/// backend that makes the mount slot look vanished to every GET must still terminate (bounded restarts, +/// not an infinite loop) AND must pace itself (the injected `sleep_fn` must actually fire) rather than +/// busy-spin. +TEST(CASMountAwaitExpiry, PersistentSlotVanishPacesAndBoundsRestartsInsteadOfSpinning) +{ + auto inner = std::make_shared(); + Layout l("p"); + /// A real slot exists underneath (uuid 1, epoch 7) so `claimMount`'s absent-slot `putIfAbsent` + /// genuinely fails every time (never accidentally re-mints). + ASSERT_EQ(claimMount(*inner, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + + auto vanishing = std::make_shared(inner); + vanishing->watched_key = l.mountKey("r"); + + uint64_t wall = 1000; + uint64_t mono = 0; + int sleeps = 0; + auto now_fn = [&] { return wall; }; + auto mono_fn = [&] { return mono; }; + auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; ++sleeps; }; + + const auto r = claimMountAwaitingExpiry( + *vanishing, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart) << "must terminate (bounded), not loop forever"; + EXPECT_GT(sleeps, 0) << "a persistently vanishing slot must still pace via sleep_fn, not busy-spin"; + /// The real epoch-7 lease is untouched -- every `putIfAbsent` attempt against it genuinely fails + /// (the object is still there), so it is never accidentally re-minted over. + EXPECT_EQ(decodeMountLease(inner->get(l.mountKey("r"))->bytes).writer_epoch, 7u); +} + +TEST(CASMountAwaitExpiry, ForeignUuidFailsClosedImmediately) +{ + auto b = std::make_shared(); + Layout l("p"); + /// A foreign server (uuid=2) holds the mount. + ASSERT_EQ(claimMount(*b, l, "r", UInt128(2), 1, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + + uint64_t now = 1000; + int sleeps = 0; + auto now_fn = [&] { return now; }; + auto mono_fn = [&] { return uint64_t{0}; }; + auto sleep_fn = [&](uint64_t ms) { now += ms; ++sleeps; }; + + const auto r = claimMountAwaitingExpiry( + *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::ForeignOwner); + EXPECT_EQ(sleeps, 0); // never waits across UUIDs +} + +/// rev.6: the predecessor's own stamped `expires_at_ms` (however skewed) is NEVER consulted for the +/// reclaim decision any more — the wait is bounded purely by OUR OWN `ttl_ms`-derived threshold. A +/// prior incarnation minted with an absurdly large `ttl` (so its own stamp claims aliveness for +/// ~100000ms) still reclaims within the SAME small threshold as any other case, because that stamp is +/// never read for timing. +TEST(CASMountAwaitExpiry, SkewedFarFutureExpiryHasNoEffectOnObservationThreshold) +{ + auto b = std::make_shared(); + Layout l("p"); + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100000).kind, MountClaimResult::Claimed); + + uint64_t wall = 1000; + uint64_t mono = 0; + auto now_fn = [&] { return wall; }; + auto mono_fn = [&] { return mono; }; + auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; }; + + const auto r = claimMountAwaitingExpiry( + *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + EXPECT_LE(mono, 100u + 100u / 20 + 20u + 20u); // bounded by OUR threshold, not the predecessor's stamp + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 8u); // reclaimed +} + +TEST(CASMountLease, KeeperStartAdoptsOurOwnClaimNotDoubleStart) +{ + auto b = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + // The normal flow: claimMount writes the live mount under (uuid=1, epoch=7), THEN keeper.start(). + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseKeeper k(b, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), [&] { return now; }, + [] { return uint64_t{0}; }); + EXPECT_NO_THROW(k.start()); // adopts our own live (uuid=1,epoch=7) mount — NOT a double-start + EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 7u); +} + +TEST(CASMountFence, SupersededWriterRefusedNoS3Read) +{ + auto b = std::make_shared(); + auto store = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "r"}); + + /// Permissive default: a Pool that has NOT armed the fence allows mutations. + EXPECT_TRUE(store->mayMutate()); + + /// Latching loss: once the renewer trips the fence it stays lost (purely local — no S3 read). + store->tripMountLost(); + EXPECT_FALSE(store->mayMutate()); + + /// A real mutate entrypoint that funnels through mutateShard now fails closed at the gate, BEFORE + /// the mutate lambda runs (so this is the ABORTED gate throw, not a FILE_DOESNT_EXIST from inside). + const RootNamespace ns{"srv1/tbl"}; + EXPECT_THROW(store->dropRef(ns, "any_ref"), DB::Exception); +} + +TEST(CASMountStartup, SecondServerSameRootFailsClosed) +{ + auto b = std::make_shared(); + auto s1 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); + /// A second server (different uuid) on the SAME server_root_id + same backend → fail closed + /// (the owner gate rejects the foreign uuid before any mount/epoch mutation). + EXPECT_THROW( + Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(2), .server_root_id = "r"}), + DB::Exception); +} + +TEST(CASMountStartup, WriterEpochStrictlyIncreasesAcrossReopen) +{ + auto b = std::make_shared(); + auto s1 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); + const uint64_t e1 = s1->writerEpoch(); + + /// Simulate shutdown: the Pool dtor stops the keeper, whose terminate() retires the lease + /// (stamps it already-expired). The owner + the durable epoch object stay sticky. + s1.reset(); + + /// Same server reopen → reclaims the (now-expired, different-epoch) mount and allocates a strictly + /// higher durable writer_epoch. + auto s2 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); + const uint64_t e2 = s2->writerEpoch(); + EXPECT_GT(e2, e1); +} + +TEST(CASMountStartup, FreshWritablePoolBootstrapsAnExplicitEmptyCatalog) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .skip_access_check = true}); + + const auto catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog.has_value()); + EXPECT_TRUE(decodeRefCatalog(catalog->bytes).entries.empty()); +} + +TEST(CASMountStartup, ExistingPoolWithoutCatalogFailsBeforeSlotMutation) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + { + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .skip_access_check = true}); + } + + /// Old raw fixtures did not persist an empty catalog. Make this an explicit existing-pool + /// fixture before removing the mandatory object whose loss the mount must reject. + if (!backend->head(layout.refCatalogKey()).exists) + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, + PutOutcome::Done); + const HeadResult catalog_head = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(catalog_head.exists); + ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog_head.token).kind, + DeleteOutcome::Kind::Deleted); + + const auto owner_before = backend->get(layout.ownerKey("r")); + const auto epoch_before = backend->get(layout.epochKey("r")); + const auto mount_before = backend->get(layout.mountKey("r")); + ASSERT_TRUE(owner_before.has_value()); + ASSERT_TRUE(epoch_before.has_value()); + ASSERT_TRUE(mount_before.has_value()); + + EXPECT_THROW(Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .skip_access_check = true}), DB::Exception); + + const auto owner_after = backend->get(layout.ownerKey("r")); + const auto epoch_after = backend->get(layout.epochKey("r")); + const auto mount_after = backend->get(layout.mountKey("r")); + ASSERT_TRUE(owner_after.has_value()); + ASSERT_TRUE(epoch_after.has_value()); + ASSERT_TRUE(mount_after.has_value()); + EXPECT_EQ(owner_after->bytes, owner_before->bytes); + EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); + EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(mount_after->bytes, mount_before->bytes); + EXPECT_EQ(mount_after->token, mount_before->token); +} + +TEST(CASMountReadOnly, ForeignOwnedPoolOpensWithoutMutation) +{ + auto b = std::make_shared(); + Layout l("p"); + + /// Server A claims the pool (writable): owner = uuid(1), a durable epoch + a live mount lease. + auto a = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); + + /// Capture the control objects BEFORE the read-only open so we can prove it mutated nothing. + const auto owner_before = b->get(l.ownerKey("r")); + const auto mount_before = b->get(l.mountKey("r")); + const auto epoch_before = b->get(l.epochKey("r")); + ASSERT_TRUE(owner_before.has_value()); + ASSERT_TRUE(mount_before.has_value()); + ASSERT_TRUE(epoch_before.has_value()); + + /// A READ-ONLY observer with a DIFFERENT server_id on the SAME backend/server_root_id must NOT + /// throw — a read-only mount never participates in the owner/epoch/mount protocol, so a pool + /// owned by another server_uuid is freely observable. + PoolPtr ro; + EXPECT_NO_THROW( + ro = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(2), .server_root_id = "r", + .read_only = true})); + EXPECT_NE(ro, nullptr); + + /// And it mutated nothing: owner still decodes to A's uuid, the mount body is still A's, and the + /// raw bytes of owner/epoch/mount are byte-for-byte unchanged (no second owner, no re-claim). + const auto owner_after = b->get(l.ownerKey("r")); + const auto mount_after = b->get(l.mountKey("r")); + const auto epoch_after = b->get(l.epochKey("r")); + ASSERT_TRUE(owner_after.has_value()); + ASSERT_TRUE(mount_after.has_value()); + ASSERT_TRUE(epoch_after.has_value()); + + EXPECT_EQ(decodeOwner(owner_after->bytes).server_uuid, UInt128(1)); + EXPECT_EQ(decodeMountLease(mount_after->bytes).server_uuid, UInt128(1)); + + EXPECT_EQ(owner_after->bytes, owner_before->bytes); + EXPECT_EQ(mount_after->bytes, mount_before->bytes); + EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); +} + +/// Pool::open must call validateCasRequestBudget itself (not just the free function in isolation — +/// see gtest_cas_request_control.cpp for that): an inconsistent cas_request_budget must refuse a +/// writable mount end-to-end (RFC cas-s3-timeout-retry-control §required-timeout-model), never mount +/// silently with a budget that could let a controlled attempt outlive the lease it is fenced under. +TEST(CASMountStartup, RefusesWritableOpenWithInconsistentCasRequestBudget) +{ + auto b = std::make_shared(); + + /// attempt_timeout_ms + lease_safety_margin_ms == mount_lease_ttl_ms below (30000): not STRICTLY + /// less, so this must be rejected. + const CasRequestBudget bad_budget{ + .attempt_timeout_ms = 25000, .operation_deadline_ms = 30000, .max_attempts = 3, .lease_safety_margin_ms = 5000}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .cas_request_budget = bad_budget}); + }); +} + +TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) +{ + auto b = std::make_shared(); + + /// Server A opens writable with a SHORT lease TTL and no background renewer (`background_watermark` + /// defaults false). The test captures its live mount body, destroys the real Pool cleanly, then + /// replays that body to simulate a crashed process whose lease survives but is never renewed. + /// This test's short lease TTL is far below the CasRequestBudget defaults (RFC + /// cas-s3-timeout-retry-control §required-timeout-model requires attempt_timeout + safety_margin < + /// lease TTL), so it also scales down cas_request_budget to fit — the budget itself is not + /// exercised here, only Pool::open's validateCasRequestBudget startup gate. + const CasRequestBudget tiny_budget{ + .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + auto a = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .mount_lease_ttl_ms = std::chrono::milliseconds(300), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = tiny_budget}); + ASSERT_NE(a, nullptr); + const uint64_t e1 = a->writerEpoch(); + const String mount_key = a->layout().mountKey("r"); + const auto stale_mount = b->get(mount_key); + ASSERT_TRUE(stale_mount.has_value()); + + /// Preserve A's live lease as if its process disappeared without running C++ teardown. Destroying + /// the real Pool first keeps the parent process valid; replaying the saved body recreates the exact + /// durable stale-lease state that a crashed process would leave behind. + a.reset(); + const auto farewell = b->get(mount_key); + ASSERT_TRUE(farewell.has_value()); + ASSERT_EQ(b->putOverwrite(mount_key, stale_mount->bytes, farewell->token).outcome, PutOutcome::Done); + + /// A restart of the SAME server (same uuid) must NOT abort: it waits out the stale lease (<= ~300ms) + /// and reclaims the mount, coming up with a strictly higher durable writer_epoch. The replayed live + /// body hides A's clean farewell, so the reclaim is `MountPriorState::UncleanObserved`. Inject a + /// fake `boot_ms_fn` + `wait_sleep_fn` (mirroring + /// `CASMountOpenWaits.UncleanOpenPaysOnlyTheObservationWindow`) so the observation window resolves + /// instantly instead of blocking this test on real time. + uint64_t a2_fake_boot = 0; + PoolPtr a2; + EXPECT_NO_THROW( + a2 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .mount_lease_ttl_ms = std::chrono::milliseconds(300), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = tiny_budget, + .boot_ms_fn = [&a2_fake_boot] { return a2_fake_boot; }, + .wait_sleep_fn = [&a2_fake_boot](uint64_t ms) { a2_fake_boot += ms; }})); + ASSERT_NE(a2, nullptr); + EXPECT_GT(a2->writerEpoch(), e1); + + /// The original live-object overlap: a first Pool is still alive when a replacement reclaims its + /// slot, so the first one's release meets a stranger. This was an `EXPECT_DEATH` pinning a + /// `LOGICAL_ERROR` abort — which fires from `~Pool`, defeating `finishTeardown`'s own catch by + /// aborting at exception construction. The first Pool never observed a deposition (nothing failed + /// its renewal; the slot was reclaimed underneath it), so this is the exclusivity-violation arm: + /// refuse, leave the reclaimer's slot untouched, latch the fence, and SURVIVE. + auto overlap_backend = std::make_shared(); + auto first = Pool::open(overlap_backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .mount_lease_ttl_ms = std::chrono::milliseconds(300), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = tiny_budget}); + const String overlap_mount_key = first->layout().mountKey("r"); + + uint64_t overlap_fake_boot = 0; + auto replacement = Pool::open(overlap_backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", + .mount_lease_ttl_ms = std::chrono::milliseconds(300), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = tiny_budget, + .boot_ms_fn = [&overlap_fake_boot] { return overlap_fake_boot; }, + .wait_sleep_fn = [&overlap_fake_boot](uint64_t ms) { overlap_fake_boot += ms; }}); + ASSERT_NE(replacement, nullptr); + + const auto reclaimer_slot_before = overlap_backend->get(overlap_mount_key); + ASSERT_TRUE(reclaimer_slot_before.has_value()); + const uint64_t overlap_violations_before + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + + first.reset(); /// must not abort, must not terminate + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + overlap_violations_before + 1); + const auto reclaimer_slot_after = overlap_backend->get(overlap_mount_key); + ASSERT_TRUE(reclaimer_slot_after.has_value()); + EXPECT_EQ(reclaimer_slot_after->bytes, reclaimer_slot_before->bytes) + << "the deposed Pool's release must not retire the reclaimer's lease"; + EXPECT_TRUE(replacement->mayMutate()) << "and must not disturb the live reclaimer"; +} + +TEST(CASMountLease, BodyCarriesFloorAndFence) +{ + MountLease m; + m.server_uuid = UInt128(0xAB); + m.writer_epoch = 7; + m.hostname = "h"; + m.pid = 42; + m.started_at_ms = 1000; + m.seq = 3; + m.expires_at_ms = 2000; + m.min_active = 5; + m.gc_fenced = true; + m.write_attempt_id = UInt128{1}; + const MountLease d = decodeMountLease(encodeMountLease(m)); + EXPECT_EQ(d.min_active, 5u); + EXPECT_TRUE(d.gc_fenced); + EXPECT_EQ(d.writer_epoch, 7u); +} + +TEST(CASMountLease, RetiredSentinelRoundTrips) +{ + MountLease m; + m.min_active = std::numeric_limits::max(); + m.write_attempt_id = UInt128{1}; + EXPECT_EQ(decodeMountLease(encodeMountLease(m)).min_active, + std::numeric_limits::max()); +} + +/// ---- Task 7 / Task 9: GC heartbeat classification with token-guarded, observation-based fence-out ---- + +namespace +{ +/// A fixed, fake "now" — no real clocks in these tests. Lease timestamps are chosen relative to it. +/// Rev.6 §token-stability observation removed the wall clock from the fence DECISION; `kNowMs` below +/// is threaded through only as `computeHeartbeatFloor`'s audit-only `now_ms`. +constexpr uint64_t kNowMs = 1'000'000; +/// The fence-out threshold measured on the LEADER's OWN monotonic clock (`mono_now_ms`), independent +/// of any lease's stamped `expires_at_ms`. +constexpr uint64_t kStableThresholdMs = 10'000; + +/// Seed one mount body under mountKey(srid) via the on-storage codec (`encodeMountLease` + +/// `putIfAbsent`) — the same interface the keeper writes through. +MountLease seedMount( + Backend & b, const Layout & l, const String & srid, + uint64_t expires_at_ms, bool gc_fenced, uint64_t min_active, uint64_t seq = 1) +{ + MountLease m; + m.server_uuid = UInt128(srid.back()); // distinct per srid; content is irrelevant to the gate + m.writer_epoch = 1; + m.hostname = "h-" + srid; + m.pid = 100; + m.started_at_ms = kNowMs; + m.seq = seq; + m.expires_at_ms = expires_at_ms; + m.min_active = min_active; + m.gc_fenced = gc_fenced; + m.write_attempt_id = UInt128{1}; + b.putIfAbsent(l.mountKey(srid), encodeMountLease(m)); + return m; +} + +/// Simulate a keeper's real renewal between two `computeHeartbeatFloor` calls: a token-guarded +/// overwrite that bumps `seq` (and so mints a fresh backend token), leaving everything else as-is. +/// Models the one thing the observation-based fence cares about: the write token changed, so any +/// in-progress observation of the OLD token must restart. +void renewMount(Backend & b, const Layout & l, const String & srid) +{ + const auto got = b.get(l.mountKey(srid)); + ASSERT_TRUE(got.has_value()); + MountLease m = decodeMountLease(got->bytes); + m.seq += 1; + const PutResult res = b.putOverwrite(l.mountKey(srid), encodeMountLease(m), got->token); + ASSERT_EQ(res.outcome, PutOutcome::Done); +} +} + +TEST(CASHeartbeatFloor, FirstSightNeverFencesEvenIfStampLooksExpired) +{ + auto b = std::make_shared(); + Layout l("p"); + + /// A stamp that would have read as long-expired under the old skew-margin comparison — under + /// rev.6 observation the stamp is never even consulted for the fence decision. + seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + + MountObservationMap obs; + const HeartbeatFloor floor = computeHeartbeatFloor(*b, l, /*now_ms*/ kNowMs, /*mono_now_ms*/ 0, + kStableThresholdMs, obs); + + EXPECT_EQ(floor.fenced_now, 0u); + EXPECT_EQ(floor.live, 1u); + ASSERT_TRUE(obs.contains("s1")); + EXPECT_EQ(obs.at("s1").first_seen_mono_ms, 0u); +} + +TEST(CASHeartbeatFloor, StableTokenPastThresholdIsFenced) +{ + auto b = std::make_shared(); + Layout l("p"); + seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + + MountObservationMap obs; + const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + EXPECT_EQ(floor_before.fenced_now, 0u); + + const MountLease before = decodeMountLease(b->get(l.mountKey("s1"))->bytes); + + /// No renewal in between: the SAME token, observed since mono 0, is now stable for the full + /// threshold on the leader's own clock. + const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + kStableThresholdMs, obs); + + EXPECT_EQ(floor2.fenced_now, 1u); + EXPECT_EQ(floor2.fenced_srids, std::vector{"s1"}); + const MountLease fenced = decodeMountLease(b->get(l.mountKey("s1"))->bytes); + EXPECT_TRUE(fenced.gc_fenced); + EXPECT_EQ(fenced.seq, before.seq + 1); +} + +TEST(CASHeartbeatFloor, RenewalBetweenRoundsRestartsObservation) +{ + auto b = std::make_shared(); + Layout l("p"); + seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + + MountObservationMap obs; + computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + ASSERT_TRUE(obs.contains("s1")); + const Token first_token = obs.at("s1").token; + + renewMount(*b, l, "s1"); + const Token renewed_token = b->get(l.mountKey("s1"))->token; + EXPECT_NE(renewed_token, first_token); + + const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + kStableThresholdMs, obs); + + EXPECT_EQ(floor2.fenced_now, 0u); + ASSERT_TRUE(obs.contains("s1")); + EXPECT_EQ(obs.at("s1").token, renewed_token); + EXPECT_EQ(obs.at("s1").first_seen_mono_ms, kStableThresholdMs); +} + +/// fix-round F7 (author-review: `Gc::mount_obs` not pruned for srids gone from LIST -> slow unbounded +/// growth on a long-lived leader, worsened by pool-member decommission). A srid whose `/mount` key is +/// removed ENTIRELY (not merely fenced/terminated -- those already `obs.erase` themselves mid-loop) is +/// never visited by a later LIST pass again, so its observation entry must be pruned at end-of-round, +/// not linger in `obs` forever. +TEST(CASHeartbeatFloor, UnseenSridPrunedFromObservationMap) +{ + auto b = std::make_shared(); + Layout l("p"); + seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + seedMount(*b, l, "s2", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + + MountObservationMap obs; + computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + ASSERT_TRUE(obs.contains("s1")); + ASSERT_TRUE(obs.contains("s2")); + + /// s2's `/mount` key is removed entirely -- e.g. `SYSTEM CAS DROP POOL MEMBER` -- so + /// no future LIST pass will ever visit it again. s1 renews (a live keeper would), so its OWN + /// observation restarts and it stays `live` -- isolating this test to the pruning behavior alone, + /// not confounding it with s1 also becoming fence-eligible (which would erase its `obs` entry too, + /// for an unrelated reason). + renewMount(*b, l, "s1"); + const auto s2_key = l.mountKey("s2"); + const auto got = b->get(s2_key); + ASSERT_TRUE(got.has_value()); + ASSERT_EQ(b->deleteExact(s2_key, got->token).kind, DeleteOutcome::Kind::Deleted); + + computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); + EXPECT_TRUE(obs.contains("s1")); + EXPECT_FALSE(obs.contains("s2")) + << "a srid removed from the LIST entirely must be pruned from obs, not linger forever"; +} + +TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) +{ + auto b = std::make_shared(); + Layout l("p"); + + /// two live mounts — genuinely renewing between the two rounds below, so their observation never + /// stabilizes. + seedMount(*b, l, "s1", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active*/ 0); + seedMount(*b, l, "s2", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active*/ 0); + /// dead — no renewal between the two rounds below — must be fenced-out by the second call. + seedMount(*b, l, "s3", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active*/ 0); + /// already-fenced — excluded, body byte-identical after both calls (no PUT). + seedMount(*b, l, "s4", /*expires*/ kNowMs - 60'000, /*fenced*/ true, /*min_active*/ 0); + /// terminated (min_active == UINT64_MAX) with expired-looking timestamps — excluded, not fenced. + seedMount(*b, l, "s5", /*expires*/ kNowMs - 60'000, /*fenced*/ false, + /*min_active*/ std::numeric_limits::max()); + + MountObservationMap obs; + + /// Round 1 (mono 0): first sight of every non-terminal mount — nothing is fence-eligible yet. + const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + EXPECT_EQ(floor_before.live, 3u); // s1, s2, s3: observation just started + EXPECT_EQ(floor_before.terminated, 1u); // s5 + EXPECT_EQ(floor_before.fenced_now, 0u); + EXPECT_EQ(floor_before.already_fenced, 1u); // s4 + + /// s1 and s2 renew between rounds (as a live keeper would); s3 does not (it crashed). + renewMount(*b, l, "s1"); + renewMount(*b, l, "s2"); + + const auto s3_before = b->get(l.mountKey("s3")); + const auto s4_before = b->get(l.mountKey("s4")); + ASSERT_TRUE(s3_before.has_value()); + ASSERT_TRUE(s4_before.has_value()); + + /// Round 2 (mono == threshold): s1/s2's renewed tokens restart their observation (still live); + /// s3's original token has now held stable for the full threshold -> fenced. + const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + kStableThresholdMs, obs); + + EXPECT_EQ(floor2.live, 2u); // s1, s2: renewed, observation restarted + EXPECT_EQ(floor2.terminated, 1u); // s5 + EXPECT_EQ(floor2.fenced_now, 1u); // s3 + EXPECT_EQ(floor2.already_fenced, 1u); // s4 + + /// The dead body was fenced: gc_fenced set, seq bumped, the rest of the body preserved. + const auto s3_after = b->get(l.mountKey("s3")); + ASSERT_TRUE(s3_after.has_value()); + const MountLease s3_prev = decodeMountLease(s3_before->bytes); + const MountLease s3_now = decodeMountLease(s3_after->bytes); + EXPECT_TRUE(s3_now.gc_fenced); + EXPECT_EQ(s3_now.seq, s3_prev.seq + 1); + EXPECT_EQ(s3_now.server_uuid, s3_prev.server_uuid); + EXPECT_EQ(s3_now.writer_epoch, s3_prev.writer_epoch); + EXPECT_EQ(s3_now.hostname, s3_prev.hostname); + EXPECT_EQ(s3_now.expires_at_ms, s3_prev.expires_at_ms); + + /// The already-fenced body was not touched (no PUT) across either call. + const auto s4_after = b->get(l.mountKey("s4")); + ASSERT_TRUE(s4_after.has_value()); + EXPECT_EQ(s4_after->bytes, s4_before->bytes); +} + +namespace +{ +/// A delegating backend whose `putOverwrite` of the target mount key first performs an inner renewal +/// (a real, token-correct overwrite that pushes expiry far into the future) and THEN delegates — so +/// the caller's fence-out overwrite lands on a stale token and returns PreconditionFailed. The inner +/// renewal runs exactly once (`renewed`), modelling a holder that renews concurrently in the window +/// between the function's GET and its fence-out PUT. +class RenewOnFenceBackend : public InMemoryBackend +{ +public: + RenewOnFenceBackend(String target_key_, uint64_t renewed_expires_ms_) + : target_key(std::move(target_key_)), renewed_expires_ms(renewed_expires_ms_) + { + } + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, + const ObjectMeta & meta) override + { + if (key == target_key && !renewed) + { + renewed = true; + /// The holder renews under the real current token: fresh far-future expiry. + const auto got = InMemoryBackend::get(key, {}); + MountLease m = decodeMountLease(got->bytes); + m.seq += 1; + m.expires_at_ms = renewed_expires_ms; + const PutResult renew = InMemoryBackend::putOverwrite(key, encodeMountLease(m), got->token); + EXPECT_EQ(renew.outcome, PutOutcome::Done); + } + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + +private: + String target_key; + uint64_t renewed_expires_ms; + bool renewed = false; +}; +} + +TEST(CASHeartbeatFloor, FenceOutLosesTokenRaceReclassifiesLive) +{ + Layout l("p"); + auto b = std::make_shared( + l.mountKey("s1"), /*renewed_expires*/ kNowMs + 120'000); + + seedMount(*b, l, "s1", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active*/ 0); + + MountObservationMap obs; + /// Round 1: first sight, observation starts — never reaches the fence-out path (the race + /// decorator stays armed for round 2). + const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + EXPECT_EQ(floor_before.fenced_now, 0u); + + /// Round 2: the token has been stable past threshold, so the function attempts the fence-out. + /// The decorator renews concurrently under the real token, the PUT hits PreconditionFailed, the + /// function re-GETs and reclassifies it as live (observation restarted on the new token) — never + /// fenced. + const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + kStableThresholdMs, obs); + + EXPECT_EQ(floor2.fenced_now, 0u); + EXPECT_EQ(floor2.live, 1u); + + const auto after = b->get(l.mountKey("s1")); + ASSERT_TRUE(after.has_value()); + EXPECT_FALSE(decodeMountLease(after->bytes).gc_fenced); +} + +TEST(CASHeartbeatFloor, EmptyPrefixYieldsNoLiveMounts) +{ + auto b = std::make_shared(); + Layout l("p"); + + MountObservationMap obs; + const HeartbeatFloor floor = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + + EXPECT_EQ(floor.live, 0u); + EXPECT_EQ(floor.terminated, 0u); + EXPECT_EQ(floor.fenced_now, 0u); + EXPECT_EQ(floor.already_fenced, 0u); +} + +/// ---- Task 1 (Phase 2): `listMounts` — read-only mount-slot enumeration for introspection ---- + +TEST(CASListMounts, ClassifiesEveryStateReadOnly) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const uint64_t now_ms = 1'000'000; + const uint64_t ttl_ms = 10'000; + + /// live: fresh claim for srid "a" + ASSERT_EQ(claimMount(*backend, layout, "a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, + MountClaimResult::Claimed); + /// expired: claim for "b" whose lease ran out long before now_ms + ASSERT_EQ(claimMount(*backend, layout, "b", UInt128{2}, 1, now_ms - 100'000, ttl_ms).kind, + MountClaimResult::Claimed); + /// corrupt: garbage bytes in "c"'s mount slot + backend->putIfAbsent(layout.mountKey("c"), "garbage-not-a-proto", {}); + + auto mounts = listMounts(*backend, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); + ASSERT_EQ(mounts.size(), 3u); + std::map by_srid; + for (const auto & m : mounts) + by_srid[m.srid] = m.state; + EXPECT_EQ(by_srid["a"], "live"); + EXPECT_EQ(by_srid["b"], "expired"); + EXPECT_EQ(by_srid["c"], "corrupt"); + + /// READ-ONLY guarantee: "b" is expired but must NOT be fenced by listMounts + /// (computeHeartbeatFloor would stamp gc_fenced=true; the introspection view must not). + auto again = listMounts(*backend, layout, now_ms, ttl_ms / 2); + for (const auto & m : again) + if (m.srid == "b") + { + EXPECT_FALSE(m.lease.gc_fenced); + EXPECT_EQ(m.state, "expired"); + } +} + +/// A `srid` may itself contain `/` (e.g. `shard-01/replica-a` — legal per +/// `CASServerRootId.ValidationAcceptsCleanPathsRejectsBad`). Slicing the key by the last `/` before +/// the `/mount` suffix (as opposed to by `serverRootsPrefix()` length) truncates it to `replica-a`. +TEST(CASListMounts, NestedSridIsNotTruncated) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const uint64_t now_ms = 1'000'000; + const uint64_t ttl_ms = 10'000; + + ASSERT_EQ(claimMount(*backend, layout, "shard-01/replica-a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, + MountClaimResult::Claimed); + + auto mounts = listMounts(*backend, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); + ASSERT_EQ(mounts.size(), 1u); + EXPECT_EQ(mounts[0].srid, "shard-01/replica-a"); + EXPECT_EQ(mounts[0].state, "live"); +} + +/// "A fence costs an epoch": a same-(uuid, epoch) re-claim must NOT refresh a `gc_fenced` body in +/// place — that would reactivate a fenced incarnation. It is terminal for THIS epoch; only a +/// DIFFERENT (fresh) epoch may reclaim the slot. +TEST(CASClaimMount, SameEpochFencedIsNotRefreshable) +{ + using namespace DB::Cas; + auto backend = std::make_shared(); + Layout layout("pool"); + /// mint for (uuid 1, epoch 1), then fence it in place (what computeHeartbeatFloor does): + ASSERT_EQ(claimMount(*backend, layout, "a", DB::UInt128{1}, 1, 1000, 10'000).kind, + MountClaimResult::Claimed); + { + auto got = backend->get(layout.mountKey("a")); + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + fenced.seq += 1; + ASSERT_EQ(backend->putOverwrite(layout.mountKey("a"), encodeMountLease(fenced), got->token).outcome, + PutOutcome::Done); + } + /// Same (uuid, epoch) re-claim must NOT refresh a fenced body — a fence costs an epoch: + const auto r = claimMount(*backend, layout, "a", DB::UInt128{1}, 1, 2000, 10'000); + EXPECT_EQ(r.kind, MountClaimResult::FencedSelf); + /// The body on the backend is still the fenced one (no write happened): + EXPECT_TRUE(decodeMountLease(backend->get(layout.mountKey("a"))->bytes).gc_fenced); + /// A DIFFERENT epoch reclaims immediately (existing branch, unchanged): + EXPECT_EQ(claimMount(*backend, layout, "a", DB::UInt128{1}, 2, 2000, 10'000).kind, + MountClaimResult::Claimed); +} + +/// ---- rev.6 Task 4: observation-based lease reclaim (no cross-node wall-clock trust) ---- + +/// A same-uuid, different-epoch lease whose STAMPED `expires_at_ms` looks long expired on OUR wall +/// clock must NOT be reclaimed by that comparison alone — a clock-skewed or simply late-observing +/// caller must never trust a bare wall-clock read across incarnations. `claimMount` (without a +/// `proven_dead_token`) always reports `LiveDoubleStart` for this branch now; only the observation +/// loop (`claimMountAwaitingExpiry`) may turn it into a reclaim, and only after proving death on ITS +/// OWN clock. +TEST(CASMountObservation, ExpiredLookingLeaseIsNotReclaimedByWallClock) +{ + auto b = std::make_shared(); + Layout l{"p"}; + /// Predecessor epoch 7 stamped expires_at_ms = 1000; our wall clock says 999999 (long past). + auto first = claimMount(*b, l, "r", UInt128(1), 7, /*now_ms=*/500, /*ttl_ms=*/500); + ASSERT_EQ(first.kind, MountClaimResult::Claimed); + auto r = claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/8, /*now_ms=*/999999, 500); + EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart); /// no wall-clock trust +} + +/// The observation loop reclaims once the write-token has held stable for the FULL rate-bound +/// threshold (`ttl_ms + ttl_ms/20 + poll_interval_ms`) on its OWN (injected, fake) clock — never +/// short-circuiting on the wall clock, which this test drives to an irrelevant, already-expired value. +TEST(CASMountObservation, TokenStableForThresholdThenReclaimed) +{ + auto b = std::make_shared(); + Layout l{"p"}; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); + uint64_t mono = 0; + std::vector sleeps; + auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), 8, + []{ return uint64_t{999999}; }, /// wall clock: irrelevant + [&]{ return mono; }, /// observation clock + /*ttl_ms=*/500, /*poll_interval_ms=*/50, + [&](uint64_t ms){ sleeps.push_back(ms); mono += ms; }); + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + EXPECT_EQ(r.prior, MountPriorState::UncleanObserved); + EXPECT_GE(mono, 500 + 500 / 20 + 50); /// full threshold actually waited +} + +/// A renewal DURING the observation window (the real holder is still alive) bumps the write-token — +/// the loop must detect the mismatch and RESTART the observation from the new token, never reclaiming +/// off a window that started watching a now-superseded token. +TEST(CASMountObservation, RenewalDuringObservationRestartsIt) +{ + auto b = std::make_shared(); + Layout l{"p"}; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); + + /// The real (still-alive) holder's keeper for epoch 7: `start()` adopts the slot `claimMount` just + /// wrote (no seq bump, per the ADOPT RULE), then synchronous renewal bumps the token mid-observation. + uint64_t keeper_wall = 500; + MountLeaseKeeper keeper(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), + [&] { return keeper_wall; }, [] { return uint64_t{0}; }, {}, + std::chrono::milliseconds(0)); + keeper.start(); + + const uint64_t threshold_ms = 500 + 500 / 20 + 50; /// = 575 + uint64_t mono = 0; + bool renewed = false; + int wait_starts = 0; + auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), 8, + []{ return uint64_t{999999}; }, /// wall clock: irrelevant + [&]{ return mono; }, /// observation clock + /*ttl_ms=*/500, /*poll_interval_ms=*/50, + [&](uint64_t ms) + { + mono += ms; + /// Renew once, close to (but before) the first window's threshold would complete — + /// almost the whole first window is wasted, forcing a near-full second window. + if (!renewed && mono >= threshold_ms - 50) + { + renewed = true; + renewKeeperOrThrow(keeper); + } + }, + /*on_wait_start=*/[&](const MountLease &, uint64_t) { ++wait_starts; }); + + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + EXPECT_EQ(r.prior, MountPriorState::UncleanObserved); + EXPECT_EQ(wait_starts, 2); /// the renewal forced exactly one restart + /// The restart's own window did not begin until at least (threshold - poll) had already elapsed, + /// so total elapsed time is well over a single threshold window. + EXPECT_GE(mono, (threshold_ms - 50) + threshold_ms); +} + +/// A GC-fenced lease is a terminal, already-threshold-gated certificate of death (the fence-out +/// itself cost the predecessor an epoch) — the observation loop must reclaim it on the FIRST attempt, +/// with zero polling/sleeping. +TEST(CASMountObservation, GcFencedIsReclaimedInstantlyWithPriorFenced) +{ + auto b = std::make_shared(); + Layout l{"p"}; + ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 1000, 500).kind, MountClaimResult::Claimed); + + /// Fence it manually (what `computeHeartbeatFloor`'s fence-out does): gc_fenced=true, seq+1, + /// token-guarded. + { + auto got = b->get(l.mountKey("r")); + ASSERT_TRUE(got.has_value()); + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + fenced.seq += 1; + ASSERT_EQ(b->putOverwrite(l.mountKey("r"), encodeMountLease(fenced), got->token).outcome, + PutOutcome::Done); + } + + int sleeps = 0; + auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), /*our_epoch=*/8, + []{ return uint64_t{999999}; }, + []{ return uint64_t{0}; }, + /*ttl_ms=*/500, /*poll_interval_ms=*/50, + [&](uint64_t) { ++sleeps; }); + + EXPECT_EQ(r.kind, MountClaimResult::Claimed); + EXPECT_EQ(r.prior, MountPriorState::Fenced); + EXPECT_EQ(sleeps, 0); +} + +/// ---- Stage B Task 3: `isCreatorFenceTerminal` -- the cross-process terminality predicate +/// `CasRefCatalog::reconcileStaleCreator` gates on. Built from `writer_epoch` plus the SAME two +/// clock-free certificates `probeNonTerminalMountSlots`/`computeHeartbeatFloor` already use, PLUS a +/// third certificate available only here: a currently-live DIFFERENT `writer_epoch` at the slot. ---- + +TEST(CASFenceTerminal, AbsentMountSlotIsNotTerminal) +{ + InMemoryBackend b; + Layout l{"p"}; + EXPECT_FALSE(isCreatorFenceTerminal(b, l, "never-mounted", 1)) + << "absence proves nothing about liveness -- never waved through"; +} + +TEST(CASFenceTerminal, UndecodableMountBodyIsNotTerminal) +{ + InMemoryBackend b; + Layout l{"p"}; + b.putIfAbsent(l.mountKey("r"), "garbage-not-a-lease", {}); + EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 1)) + << "an unreadable lease of some other format generation must block, never wave through"; +} + +TEST(CASFenceTerminal, GcFencedIsTerminal) +{ + InMemoryBackend b; + Layout l{"p"}; + ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); + auto got = b.get(l.mountKey("r")); + ASSERT_TRUE(got.has_value()); + MountLease fenced = decodeMountLease(got->bytes); + fenced.gc_fenced = true; + ASSERT_EQ(b.putOverwrite(l.mountKey("r"), encodeMountLease(fenced), got->token).outcome, PutOutcome::Done); + + EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)); +} + +TEST(CASFenceTerminal, CleanFarewellIsTerminal) +{ + InMemoryBackend b; + Layout l{"p"}; + ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); + auto got = b.get(l.mountKey("r")); + ASSERT_TRUE(got.has_value()); + MountLease retired = decodeMountLease(got->bytes); + retired.min_active = std::numeric_limits::max(); + ASSERT_EQ(b.putOverwrite(l.mountKey("r"), encodeMountLease(retired), got->token).outcome, PutOutcome::Done); + + EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)); +} + +TEST(CASFenceTerminal, ADifferentLiveWriterEpochIsTerminalForTheOldOne) +{ + InMemoryBackend b; + Layout l{"p"}; + /// Slot now held at epoch 8 -- epoch 7's incarnation is superseded regardless of ITS OWN + /// certificate (neither fenced nor farewelled). + ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/8, 1000, 500).kind, MountClaimResult::Claimed); + + EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)) + << "a different epoch is currently live at this slot -- epoch 7 can never reclaim it"; + EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 8)) + << "epoch 8 IS the current live epoch -- not terminal"; +} + +/// A merely EXPIRED lease (wall-clock past `expires_at_ms`, same epoch, no certificate) must NOT be +/// treated as terminal -- mirrors `claimMount`'s own refusal to trust a bare timestamp comparison. +TEST(CASFenceTerminal, ExpiredButSameEpochAndUncertifiedIsNotTerminal) +{ + InMemoryBackend b; + Layout l{"p"}; + /// A lease whose stamped expiry is already far in the past, same epoch throughout. + ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, /*now_ms=*/0, /*ttl_ms=*/1).kind, + MountClaimResult::Claimed); + + EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 7)) + << "expiry alone is never a certificate of death, exactly like claimMount's own discipline"; +} diff --git a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp new file mode 100644 index 000000000000..42dae767213a --- /dev/null +++ b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp @@ -0,0 +1,186 @@ +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +} + +using namespace DB::Cas; +using DB::Cas::tests::MountSlotRaceBackend; +using DB::Cas::tests::expectThrowsCodeWithMessage; + +namespace +{ + +/// One keeper for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. +MountLeaseKeeper makeKeeper( + const std::shared_ptr & backend, + uint64_t & now, + DB::UInt128 uuid = DB::UInt128(1), + uint64_t epoch = 7) +{ + return MountLeaseKeeper( + backend, + Layout("p"), + "r", + uuid, + epoch, + std::chrono::milliseconds(100), + [&now] { return now; }, + [] { return uint64_t{0}; }); +} + +void markMountGcFenced(MountSlotRaceBackend & backend, const Layout & layout, const String & server_root_id) +{ + const String key = layout.mountKey(server_root_id); + const auto got = backend.get(key); + ASSERT_TRUE(got); + MountLease lease = decodeMountLease(got->bytes); + lease.gc_fenced = true; + const PutResult result = backend.putOverwrite(key, encodeMountLease(lease), got->token); + ASSERT_EQ(result.outcome, PutOutcome::Done); +} + +} + +TEST(CASMountClaimConflicts, SlotAppearedBetweenHeadAndPutIfAbsent) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + /// Empty at `head`; another process mints it before our `putIfAbsent` lands. + backend->before_put_if_absent = [&] + { + claimMount(*backend, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100); + }; + auto keeper = makeKeeper(backend, now); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "appeared between head and putIfAbsent", + [&] { keeper.start(); }); +} + +TEST(CASMountClaimConflicts, SlotVanishedBetweenHeadAndGet) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + backend->before_get = [&] + { + const auto got = backend->get(layout.mountKey("r")); + ASSERT_TRUE(got); + backend->deleteExact(layout.mountKey("r"), got->token); + }; + auto keeper = makeKeeper(backend, now); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "vanished between head and get while claiming", + [&] { keeper.start(); }); +} + +TEST(CASMountClaimConflicts, SlotHeldByForeignServer) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + auto keeper = makeKeeper(backend, now); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "held by a foreign server", + [&] { keeper.start(); }); +} + +TEST(CASMountClaimConflicts, SlotHeldByDifferentWriterEpoch) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + auto keeper = makeKeeper(backend, now, DB::UInt128(1), /*epoch=*/8); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "held by a different writer_epoch", + [&] { keeper.start(); }); +} + +TEST(CASMountClaimConflicts, SlotChangedInsideAdoptionWindow) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + /// Rewrite the slot under a NEW token after our `get`, so our adoption `putOverwrite` conflicts. + backend->before_put_overwrite = [&] + { + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now + 1, /*ttl_ms=*/100); + }; + auto keeper = makeKeeper(backend, now); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "changed while adopting our own mount slot", + [&] { keeper.start(); }); +} + +TEST(CASMountClaimConflicts, SlotVanishedInsideAdoptionWindow) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + backend->before_put_overwrite = [&] + { + const auto got = backend->get(layout.mountKey("r")); + ASSERT_TRUE(got); + backend->deleteExact(layout.mountKey("r"), got->token); + }; + auto keeper = makeKeeper(backend, now); + expectThrowsCodeWithMessage( + DB::ErrorCodes::ABORTED, + "vanished while adopting our own mount slot", + [&] { keeper.start(); }); +} + +/// The two fenced branches keep their own type, and keep PRECEDENCE over the conflicts above: the +/// mount-open loop catches `MountFencedException` by type and recovers with a fresh writer epoch, so +/// a fence reported as a plain conflict would turn a recoverable state into a failed mount. +TEST(CASMountClaimConflicts, FencedBeforeAdoptionRaisesMountFenced) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + markMountGcFenced(*backend, layout, "r"); + auto keeper = makeKeeper(backend, now); + EXPECT_THROW(keeper.start(), MountFencedException); +} + +TEST(CASMountClaimConflicts, FencedInsideAdoptionWindowRaisesMountFencedNotAborted) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + ASSERT_EQ( + claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + /// The slot changes inside the adoption window AND the new body is fenced: the fenced branch must + /// win over the "changed while adopting" one. + backend->before_put_overwrite = [&] { markMountGcFenced(*backend, layout, "r"); }; + auto keeper = makeKeeper(backend, now); + EXPECT_THROW(keeper.start(), MountFencedException); +} diff --git a/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp b/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp new file mode 100644 index 000000000000..bfb62f817ecb --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp @@ -0,0 +1,558 @@ +#include +#include "cas_test_helpers.h" + +#include +#include +#include + +/// Per-TU declaration of the one setting this file overrides, following the pattern `cas_test_helpers.h` +/// documents for `server_root_id`/`scratch_path`: defined once in `ContentAddressedSettings.cpp`, declared +/// by each consumer for what it actually references. +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsBool gc_enabled; +} + +#include +#include +#include +#include +#include + +/// The namespace-file REQUEST PROFILE gate (directive §dedup-performance-constraint) -- for the +/// `Pool` namespace-file surface, which is the layer that must be read carefully below. +/// +/// `MergeTreeDeduplicationLog` rotates namespace files on the insert path because the CA disk cannot +/// append, so every namespace-file operation's request profile is insert latency. The directive's +/// constraint has five clauses: no catalog request per file operation, no ref-log append, no blob +/// upload, no folder-manifest rewrite, and unchanged direct-object backend request counts. +/// +/// WHAT THIS FILE PINS: the last clause, per key. The counts below were READ OFF this tree before any +/// key change and pasted as literals, which is the whole point of the file -- expectations re-derived +/// after a change measure the change against itself. Incarnation qualification changes the KEY a +/// namespace file is stored under, so the keys are derived from `Layout` rather than spelled out; what +/// must not move is the count per key and the set of keys touched. +/// +/// TWO KINDS OF GATE LIVE IN THIS FILE, and mistaking one for the other would misread what it proves. +/// The per-operation COUNTS above are baseline-anchored: they were read off the tree BEFORE the key change +/// and pasted as literals, so they detect DRIFT from a measured past. The four negatives below are a +/// FORWARD ALARM: a zero has no baseline to drift from, and the disk-layer cases could not have existed +/// before the life resolution they measure was put on that path. They are no weaker for it -- they fail +/// the moment a catalog, ref-log, blob or manifest request appears where none belongs -- but they are not +/// evidence that anything was "unchanged". +/// +/// WHERE THE OTHER FOUR CLAUSES ARE FENCED, and why they could not be fenced by the cases above. Every +/// pool-layer case drives `Pool::putNamespaceFile`/`getNamespaceFile`/`removeNamespaceFile`/ +/// `listNamespaceFiles`, which reach `CasPlainObjects` and have no catalog, ref-log, blob or manifest +/// path to take -- so at THAT layer the four negatives hold by construction of the call and measuring +/// them proves nothing. The layer where they can be violated is the DISK operation above, where +/// `ContentAddressedTransaction::writeFile` resolves the namespace's life; that is exactly where Task 4b +/// put a life resolution, so that is where a per-operation catalog GET would appear. The two +/// `CASNamespaceFileDiskProfile` cases at the bottom of this file fence it there, through a recording +/// `IObjectStorage` (the metadata storage builds its own `Backend` from an `ObjectStoragePtr`, so the +/// object storage is the injectable seam -- no production surface is widened for the test's benefit). + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const String kNsString = "test/req_profile@cas@"; +const String kFile = "format_version.txt"; +/// A NESTED relative name, which is what the dedup log actually stores (its segments live in a +/// table-level subdirectory), so the profile is captured on the shape the constraint is about. +const String kSegment1 = "deduplication_logs/deduplication_log_1.txt"; +const String kSegment2 = "deduplication_logs/deduplication_log_2.txt"; + +/// The identity every case below operates under. `fixture::fixtureLife` is the transitional mint Task 6 +/// deletes; what matters to this file is only that ONE life is used throughout, so a count is not +/// split across two prefixes. +NamespaceLifeId testLife() +{ + return fixture::fixtureLife(RootNamespace{kNsString}); +} + +/// A pool over `CountingBackend`, with the counts reset AFTER open: `Pool::open` runs its own +/// capability probe and mount claim, and those requests belong to no file operation. +PoolPtr openCountedPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + PoolPtr store = openPoolForTest(out_backend); + out_backend->resetCounts(); + return store; +} + +} + +/// CREATE (the key is absent) and REWRITE (the key is present) are different request shapes on the +/// same call, and the profile pins both: one HEAD to learn the token, then the create-if-absent or the +/// token-conditioned replacement that HEAD selected. +TEST(CASNamespaceFileRequestProfile, CreateThenRewrite) +{ + std::shared_ptr backend; + PoolPtr store = openCountedPool(backend); + const NamespaceLifeId life = testLife(); + const String key = store->layout().namespaceFileKey(life, kFile); + + store->putNamespaceFile(life, kFile, "1\n"); + + EXPECT_EQ(backend->headCount(key), 1u); + EXPECT_EQ(backend->putCount(key), 1u); /// putIfAbsent -- the key was absent + EXPECT_EQ(backend->putOverwriteCount(key), 0u); + EXPECT_EQ(backend->getCount(key), 0u); + EXPECT_EQ(backend->deleteCount(key), 0u); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->touchedKeys(), std::vector{key}); + + backend->resetCounts(); + store->putNamespaceFile(life, kFile, "2\n"); + + EXPECT_EQ(backend->headCount(key), 1u); + EXPECT_EQ(backend->putOverwriteCount(key), 1u); /// token-conditioned replacement -- it existed + EXPECT_EQ(backend->putCount(key), 0u); + EXPECT_EQ(backend->getCount(key), 0u); + EXPECT_EQ(backend->deleteCount(key), 0u); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->touchedKeys(), std::vector{key}); +} + +/// A plain read is one whole-object GET and nothing else. +TEST(CASNamespaceFileRequestProfile, Read) +{ + std::shared_ptr backend; + PoolPtr store = openCountedPool(backend); + const NamespaceLifeId life = testLife(); + const String key = store->layout().namespaceFileKey(life, kFile); + + store->putNamespaceFile(life, kFile, "1\n"); + backend->resetCounts(); + + EXPECT_EQ(store->getNamespaceFile(life, kFile), String("1\n")); + + EXPECT_EQ(backend->getCount(key), 1u); + EXPECT_EQ(backend->wholeGetCount(key), 1u); + EXPECT_EQ(backend->headCount(key), 0u); + EXPECT_EQ(backend->putCount(key), 0u); + EXPECT_EQ(backend->putOverwriteCount(key), 0u); + EXPECT_EQ(backend->touchedKeys(), std::vector{key}); +} + +/// APPEND on a CA disk is serviced by read-modify-rewrite, and its request shape is the composition of +/// the two calls that implement it: a GET of the current body, then a whole-body PUT of base+delta. +/// Driven here as that composition against the same key, which is the shape whose count must not move. +TEST(CASNamespaceFileRequestProfile, ReadModifyRewriteAppend) +{ + std::shared_ptr backend; + PoolPtr store = openCountedPool(backend); + const NamespaceLifeId life = testLife(); + const String key = store->layout().namespaceFileKey(life, kSegment1); + + store->putNamespaceFile(life, kSegment1, "base"); + backend->resetCounts(); + + const std::optional carried = store->getNamespaceFile(life, kSegment1); + ASSERT_TRUE(carried.has_value()); + store->putNamespaceFile(life, kSegment1, *carried + "-delta"); + + EXPECT_EQ(backend->getCount(key), 1u); + EXPECT_EQ(backend->headCount(key), 1u); + EXPECT_EQ(backend->putOverwriteCount(key), 1u); + EXPECT_EQ(backend->putCount(key), 0u); + EXPECT_EQ(backend->deleteCount(key), 0u); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->touchedKeys(), std::vector{key}); + EXPECT_EQ(store->getNamespaceFile(life, kSegment1), String("base-delta")); +} + +/// REMOVE is exact-token deletion, so it is one HEAD for the token plus one delete against it. +TEST(CASNamespaceFileRequestProfile, Remove) +{ + std::shared_ptr backend; + PoolPtr store = openCountedPool(backend); + const NamespaceLifeId life = testLife(); + const String key = store->layout().namespaceFileKey(life, kFile); + + store->putNamespaceFile(life, kFile, "1\n"); + backend->resetCounts(); + + store->removeNamespaceFile(life, kFile); + + EXPECT_EQ(backend->headCount(key), 1u); + EXPECT_EQ(backend->deleteCount(key), 1u); + EXPECT_EQ(backend->getCount(key), 0u); + EXPECT_EQ(backend->putCount(key), 0u); + EXPECT_EQ(backend->putOverwriteCount(key), 0u); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->touchedKeys(), std::vector{key}); + EXPECT_FALSE(store->getNamespaceFile(life, kFile).has_value()); +} + +/// ROTATION is the sequence the constraint names: the retiring segment is enumerated, the new segment +/// is created, and the retired one is removed. One LIST of the files prefix serves the enumeration (a +/// single page here), and each segment carries its own create or remove shape. +TEST(CASNamespaceFileRequestProfile, DedupLogRotation) +{ + std::shared_ptr backend; + PoolPtr store = openCountedPool(backend); + const NamespaceLifeId life = testLife(); + const String prefix = store->layout().namespaceFilesPrefix(life); + const String old_key = store->layout().namespaceFileKey(life, kSegment1); + const String new_key = store->layout().namespaceFileKey(life, kSegment2); + + store->putNamespaceFile(life, kSegment1, "segment-1-records"); + backend->resetCounts(); + + const std::vector before = store->listNamespaceFiles(life); + ASSERT_EQ(before, std::vector{kSegment1}); + store->putNamespaceFile(life, kSegment2, "segment-2-records"); + store->removeNamespaceFile(life, kSegment1); + + EXPECT_EQ(backend->listCount(prefix), 1u); + EXPECT_EQ(backend->listTotal(), 1u); + EXPECT_EQ(backend->headCount(new_key), 1u); + EXPECT_EQ(backend->putCount(new_key), 1u); + EXPECT_EQ(backend->putOverwriteCount(new_key), 0u); + EXPECT_EQ(backend->headCount(old_key), 1u); + EXPECT_EQ(backend->deleteCount(old_key), 1u); + EXPECT_EQ(backend->getTotal(), 0u); /// rotation reads no body + EXPECT_EQ(backend->casPutTotal(), 0u); + /// Sorted, and the files prefix is a proper prefix of both segment keys, so it comes first. + EXPECT_EQ(backend->touchedKeys(), (std::vector{prefix, old_key, new_key})); + + EXPECT_EQ(store->listNamespaceFiles(life), std::vector{kSegment2}); +} + + +/// ===================== THE FOUR NEGATIVES, AT THE DISK LAYER ===================== +/// +/// Constraint 16's other four clauses: a namespace-file operation performs no catalog request, no +/// ref-log append, no blob upload and no folder-manifest rewrite. They are asserted here rather than +/// above because only here is there a life resolution to get wrong. +/// +/// WHAT MAKES THE CLAIM NON-TRIVIAL. Task 4b's read and write paths resolve a catalog-minted life. That +/// resolution is per TABLE-OPEN -- `CasRefLedger` caches it on the table's runtime -- so the steady-state +/// operation pays nothing for it. If it ever became per-operation, `format_version.txt` and every +/// dedup-log rotation on the insert path would carry a catalog round trip, and nothing else in the suite +/// would notice. `SteadyStateFileOperationsTouchNoCatalogRefBlobOrManifestKey` is that alarm, and +/// `TheLifeResolutionIsPaidOncePerTableOpen` is the other half: it shows the birth cost EXISTS and is +/// paid exactly once, so the steady-state zeros are a real property rather than an artifact of a fixture +/// that never triggered a resolution at all. + +namespace +{ + +/// A `LocalObjectStorage` that records every key it is asked about, per operation family. Used to ask +/// "was any key under these four families touched at all", which is a question about WHICH keys an +/// operation reaches -- not about counts -- so recording the key set is the whole instrument. +/// +/// It overrides every method `CasObjectStorageBackend` reaches: a family left un-overridden would be an +/// unrecorded path, and an assertion of "nothing touched it" would then be silently satisfied by the +/// gap rather than by the behaviour. +class RecordingObjectStorage : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + bool exists(const DB::StoredObject & object) const override + { + record(object.remote_path, /*is_write*/ false); + return DB::LocalObjectStorage::exists(object); + } + + std::unique_ptr readObject( + const DB::StoredObject & object, const DB::ReadSettings & read_settings, + std::optional read_hint, bool use_external_buffer, + bool restrict_seek) const override + { + record(object.remote_path, /*is_write*/ false); + return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); + } + + std::unique_ptr writeObject( + const DB::StoredObject & object, DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + record(object.remote_path, /*is_write*/ true); + return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } + + void removeObjectIfExists(const DB::StoredObject & object) override + { + record(object.remote_path, /*is_write*/ true); + DB::LocalObjectStorage::removeObjectIfExists(object); + } + + void removeObjectsIfExist(const DB::StoredObjects & objects) override + { + for (const DB::StoredObject & object : objects) + record(object.remote_path, /*is_write*/ true); + DB::LocalObjectStorage::removeObjectsIfExist(objects); + } + + DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override + { + record(path, /*is_write*/ false); + return DB::LocalObjectStorage::getObjectMetadata(path, with_tags); + } + + std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override + { + record(path, /*is_write*/ false); + return DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + } + + void listObjects(const std::string & path, DB::RelativePathsWithMetadata & children, size_t max_keys) const override + { + record(path, /*is_write*/ false); + DB::LocalObjectStorage::listObjects(path, children, max_keys); + } + + bool existsOrHasAnyChild(const std::string & path) const override + { + record(path, /*is_write*/ false); + return DB::LocalObjectStorage::existsOrHasAnyChild(path); + } + + void copyObject( + const DB::StoredObject & object_from, const DB::StoredObject & object_to, + const DB::ReadSettings & read_settings, const DB::WriteSettings & write_settings, + std::optional object_to_attributes) override + { + record(object_from.remote_path, /*is_write*/ false); + record(object_to.remote_path, /*is_write*/ true); + DB::LocalObjectStorage::copyObject(object_from, object_to, read_settings, write_settings, object_to_attributes); + } + + /// Every recorded key containing `needle`, in first-touch order, so a failure names the offender. + std::vector touchedContaining(std::string_view needle) const + { + std::lock_guard lock(mutex); + std::vector out; + for (const String & key : touched) + if (key.find(needle) != String::npos) + out.push_back(key); + return out; + } + + std::vector writtenContaining(std::string_view needle) const + { + std::lock_guard lock(mutex); + std::vector out; + for (const String & key : written) + if (key.find(needle) != String::npos) + out.push_back(key); + return out; + } + + void resetRecords() + { + std::lock_guard lock(mutex); + touched.clear(); + written.clear(); + } + +private: + void record(const std::string & key, bool is_write) const + { + std::lock_guard lock(mutex); + touched.push_back(key); + if (is_write) + written.push_back(key); + } + + mutable std::mutex mutex; + mutable std::vector touched; + mutable std::vector written; +}; + +const std::string kTableUuid = "a11a11a1-1111-4111-8111-111111111111"; +const std::string kTablePath = "a11/a11a11a1-1111-4111-8111-111111111111"; + +/// The four families the constraint forbids a file operation from touching, as key substrings. Taken +/// from `Layout` where a helper exists rather than spelled out, so a layout change breaks this by +/// failing to compile or by moving the substring, not by silently matching nothing. +struct ForbiddenFamily +{ + String needle; + String clause; +}; + +std::vector forbiddenFamilies(const DB::Cas::Layout & layout) +{ + return { + {layout.refCatalogKey(), "no catalog request"}, + {layout.casRefsPrefix(), "no ref-log append"}, + {layout.blobsPrefix(), "no blob upload"}, + {layout.casManifestsPrefix(), "no folder-manifest rewrite"}, + }; +} + +std::shared_ptr openRecordingStorage( + std::shared_ptr & out_object_storage) +{ + static std::atomic counter{0}; + const String unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_ns_file_profile_" + unique)).string(); + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + out_object_storage = std::make_shared( + DB::LocalObjectStorageSettings("test", root, /*read_only_=*/false)); + + auto settings = makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / ("cas_ns_file_profile_scratch_" + unique)); + /// A GC round touches `cas/ref_catalog`, `cas/ns/stream/` and `cas/manifests/`, so with the + /// background scheduler enabled these zeros would hold only because the first tick (60s) outlives the + /// test. A timer is not a fence. + settings[DB::ContentAddressedSetting::gc_enabled] = false; + auto storage = std::make_shared( + out_object_storage, "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// One verbatim namespace file written through the REAL disk write path (the buffer whose finalize +/// callback reaches `putNamespaceFile`), not through the pool surface. +void writeVerbatimThroughDisk( /// ASSERT_* inside -> must return void + DB::ContentAddressedMetadataStorage & storage, const std::string & path, const String & bytes, + DB::WriteMode mode = DB::WriteMode::Rewrite) +{ + /// `tryCreateWriteBuffer` is the interface entry the disk itself uses, so this drives the same + /// buffer construction (and the same autocommit-on-finalize contract for verbatim files) that a real + /// write does. `owner` is null here: only a part-blob buffer's deferred finalize needs the pin, and + /// a verbatim file finalizes inline, inside this call's scope. + auto tx = storage.createTransaction(); + auto buf = tx->tryCreateWriteBuffer( + /*owner*/ nullptr, path, DB::DBMS_DEFAULT_BUFFER_SIZE, mode, {}, /*autocommit*/ true); + ASSERT_TRUE(buf != nullptr); + DB::writeString(bytes, *buf); + buf->finalize(); +} + +} + +/// The steady state: with the table open and its life already resolved, no namespace-file operation -- +/// rewrite, append, read, rotation, remove -- touches a catalog, ref, blob or manifest key. +TEST(CASNamespaceFileDiskProfile, SteadyStateFileOperationsTouchNoCatalogRefBlobOrManifestKey) +{ + std::shared_ptr object_storage; + auto storage = openRecordingStorage(object_storage); + const DB::Cas::Layout & layout = storage->store()->layout(); + + /// Open the table by doing the first file operation, which is what resolves (and here mints) the + /// life. Everything measured below happens after it. + writeVerbatimThroughDisk(*storage, kTablePath + "/format_version.txt", "1\n"); + ASSERT_TRUE(storage->existsFile(kTablePath + "/format_version.txt")); + + object_storage->resetRecords(); + + /// A whole-file rewrite, the read-modify-rewrite append, a read, and a dedup-log rotation + /// (create the new segment, enumerate, drop the retired one) -- the four shapes the constraint names. + writeVerbatimThroughDisk(*storage, kTablePath + "/format_version.txt", "2\n"); + writeVerbatimThroughDisk(*storage, kTablePath + "/deduplication_logs/deduplication_log_1.txt", "a"); + writeVerbatimThroughDisk( + *storage, kTablePath + "/deduplication_logs/deduplication_log_1.txt", "b", DB::WriteMode::Append); + EXPECT_EQ(storage->tryGetInManifestBytes(kTablePath + "/deduplication_logs/deduplication_log_1.txt"), + std::optional("ab")); + writeVerbatimThroughDisk(*storage, kTablePath + "/deduplication_logs/deduplication_log_2.txt", "c"); + storage->createTransaction()->unlinkFile( + kTablePath + "/deduplication_logs/deduplication_log_1.txt", /*if_exists*/ false, /*remove_metadata_only*/ false); + + for (const ForbiddenFamily & family : forbiddenFamilies(layout)) + EXPECT_EQ(object_storage->touchedContaining(family.needle), std::vector{}) + << "Constraint 16, '" << family.clause << "': a namespace-file operation reached " << family.needle; + + /// A positive control on the instrument itself: the operations above DID reach the store, so the + /// four empty answers are the absence of those families and not a recorder that recorded nothing. + EXPECT_FALSE(object_storage->writtenContaining("/_files/").empty()) + << "the recorder must have seen the file writes themselves"; +} + +/// The other half: the life resolution is real and is paid ONCE per table-open. Without this, the zeros +/// above could be produced by a fixture in which no resolution ever happened. +TEST(CASNamespaceFileDiskProfile, TheLifeResolutionIsPaidOncePerTableOpen) +{ + std::shared_ptr object_storage; + auto storage = openRecordingStorage(object_storage); + const DB::Cas::Layout & layout = storage->store()->layout(); + + /// The FIRST namespace-file operation on a never-opened table resolves the life from the catalog, + /// minting the namespace when it names none -- so it DOES reach the catalog. That is the per-open + /// cost, and the reason the steady-state case above resets its records after this point. + writeVerbatimThroughDisk(*storage, kTablePath + "/format_version.txt", "1\n"); + EXPECT_FALSE(object_storage->touchedContaining(layout.refCatalogKey()).empty()) + << "the first file operation must resolve a life, which reaches the catalog"; + + object_storage->resetRecords(); + + /// The second operation on the SAME open table resolves nothing: the life is cached on the table's + /// runtime. This is the assertion that says "per table-open", and it is the one that would fail if a + /// future change moved the resolution onto the operation. + writeVerbatimThroughDisk(*storage, kTablePath + "/format_version.txt", "2\n"); + EXPECT_EQ(object_storage->touchedContaining(layout.refCatalogKey()), std::vector{}) + << "a second file operation must not re-resolve the life"; +} + + +/// THE REMOVAL PATHS MUST NOT CREATE A NAMESPACE — the case that regressed silently in this task's first +/// round, so it is pinned on the catalog rather than on the file outcome. +/// +/// Why the file outcome cannot pin it: `unlinkFile`/`removeRecursive` against a never-opened table +/// answer "absent" both before and after the defect, because a freshly minted namespace has no files +/// either. The only observable difference is the catalog write, so that is what is asserted. And it +/// matters twice over: `unlinkFile(..., if_exists = true)` is called from cleanup paths whose contract is +/// to be a no-op, and the catalog is ONE pool-wide object under a capacity-admission predicate — a +/// removal that admits an entry per never-created table grows it without bound. +TEST(CASNamespaceFileDiskProfile, RemovalOnANeverOpenedTableLeavesTheCatalogUntouched) +{ + std::shared_ptr object_storage; + auto storage = openRecordingStorage(object_storage); + const DB::Cas::Layout & layout = storage->store()->layout(); + + /// A valid pool already owns its explicit empty mandatory catalog. Nothing has opened this table: + /// no namespace file written, no part published, and no ref operation has changed that object. + const auto catalog_before = storage->store()->backend().get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_before); + EXPECT_TRUE(decodeRefCatalog(catalog_before->bytes).entries.empty()); + object_storage->resetRecords(); + + /// Three removal shapes, all against paths under a table that does not exist. + storage->createTransaction()->unlinkFile( + kTablePath + "/format_version.txt", /*if_exists*/ true, /*remove_metadata_only*/ false); + storage->createTransaction()->removeRecursive( + kTablePath + "/deduplication_logs", DB::IMetadataTransaction::ShouldRemoveObjectsPredicate{}); + /// The table directory ITSELF, which is a different arm from the subdirectory above: it is the one + /// that reaches the ref layer's namespace drop and its ref enumeration, rather than only the + /// namespace-file resolver. + storage->createTransaction()->removeRecursive( + kTablePath, DB::IMetadataTransaction::ShouldRemoveObjectsPredicate{}); + EXPECT_FALSE(storage->existsFile(kTablePath + "/format_version.txt")); + + EXPECT_EQ(object_storage->writtenContaining(layout.refCatalogKey()), std::vector{}) + << "a removal must not write the catalog: it must not birth the namespace it is removing from"; + const auto catalog_after_removal = storage->store()->backend().get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_after_removal); + EXPECT_EQ(catalog_after_removal->bytes, catalog_before->bytes); + EXPECT_EQ(catalog_after_removal->token, catalog_before->token) + << "the mandatory catalog must remain byte-for-byte and token-for-token unchanged"; + + /// Not vacuous: the SAME operations on the same table after a write do reach the file, so the zeros + /// above are the absence of a birth and not the absence of any work. + writeVerbatimThroughDisk(*storage, kTablePath + "/format_version.txt", "1\n"); + ASSERT_TRUE(storage->existsFile(kTablePath + "/format_version.txt")); + storage->createTransaction()->unlinkFile( + kTablePath + "/format_version.txt", /*if_exists*/ false, /*remove_metadata_only*/ false); + EXPECT_FALSE(storage->existsFile(kTablePath + "/format_version.txt")); + /// Positive control: the write really did birth the namespace and mutate the same catalog object + /// whose stability the removal assertions pin above. + const auto catalog_after_birth = storage->store()->backend().get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_after_birth); + EXPECT_NE(catalog_after_birth->bytes, catalog_after_removal->bytes); + EXPECT_NE(catalog_after_birth->token, catalog_after_removal->token); +} diff --git a/src/Disks/tests/gtest_cas_namespace_janitor.cpp b/src/Disks/tests/gtest_cas_namespace_janitor.cpp new file mode 100644 index 000000000000..7a3455738bc4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_janitor.cpp @@ -0,0 +1,607 @@ +#include "cas_test_helpers.h" +#include +#include + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +class OrderedJanitorBackend : public CountingBackend +{ +public: + using CountingBackend::get; + std::vector events; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (prefix.ends_with("/cas/ns/")) + events.push_back("list"); + return CountingBackend::list(prefix, cursor, limit); + } + + std::optional get(const String & key, Range range) override + { + if (key.ends_with("/cas/ref_catalog")) + events.push_back("catalog"); + return CountingBackend::get(key, range); + } +}; + +class OmitFirstNamespacePageBackend : public CountingBackend +{ +public: + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (omit && prefix.ends_with("/cas/ns/")) + { + omit = false; + return {}; + } + return CountingBackend::list(prefix, cursor, limit); + } +private: + bool omit = true; +}; + +class ReplaceBeforeJanitorDeleteBackend : public CountingBackend +{ +public: + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (!replaced) + { + replaced = true; + const auto current = InMemoryBackend::get(key); + if (current) + (void)InMemoryBackend::casPut(key, "winner", current->token); + } + return CountingBackend::deleteExact(key, token); + } +private: + bool replaced = false; +}; + +class TokenlessListBackend : public CountingBackend +{ +public: + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = CountingBackend::list(prefix, cursor, limit); + for (ListedKey & key : page.keys) + key.token.reset(); + return page; + } + + bool supportsListTokens() const override { return false; } + + HeadResult head(const String & key) override + { + HeadResult result = CountingBackend::head(key); + if (!replaced && result.exists && key == replace_on_head) + { + replaced = true; + (void)InMemoryBackend::casPut(key, "winner", result.token); + } + return result; + } + + String replace_on_head; + +private: + bool replaced = false; +}; + +class FenceLossDuringHeadBackend : public TokenlessListBackend +{ +public: + HeadResult head(const String & key) override + { + HeadResult result = TokenlessListBackend::head(key); + fence_held = false; + return result; + } + + bool fence_held = true; +}; + +class CatalogAfterListBackend : public CountingBackend +{ +public: + explicit CatalogAfterListBackend(NamespaceLifeId life_) : protected_life(std::move(life_)) {} + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = CountingBackend::list(prefix, cursor, limit); + if (!published && prefix.ends_with("/cas/ns/")) + { + published = true; + const String catalog_key = "p/cas/ref_catalog"; + /// This models a CONCURRENT actor's read, not the janitor's own -- counting it here would + /// make `PostListCatalogCutProtectsConcurrentCreationWithOneGet`'s "exactly one get" assertion + /// count this simulated actor's read as the janitor's, defeating the point of that assertion. + const auto current = InMemoryBackend::get(catalog_key, {}); // NOLINT(bugprone-parent-virtual-call) + if (current) + { + RefCatalog catalog; + catalog.entries.push_back(CatalogEntry{.ns = protected_life.ns, .state = NsState::Live, + .incarnation = protected_life.incarnation}); + (void)InMemoryBackend::casPut(catalog_key, encodeRefCatalog(catalog), current->token); + } + } + return page; + } +private: + NamespaceLifeId protected_life; + bool published = false; +}; + +class RejectCursorBackend : public CountingBackend +{ +public: + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (prefix.ends_with("/cas/ns/") && !cursor.empty()) + throw std::runtime_error("backend rejected cursor"); + return CountingBackend::list(prefix, cursor, limit); + } +}; + +class FailMaintenancePublicationBackend : public CountingBackend +{ +public: + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (fail_publication && key.ends_with("/gc/maintenance_state")) + throw std::runtime_error("maintenance publication failed"); + return CountingBackend::casPut(key, bytes, expected, meta); + } + bool fail_publication = false; +}; + +void seedCatalog(CountingBackend & backend, const Layout & layout, RefCatalog catalog = {}) +{ + ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(catalog)).outcome, PutOutcome::Done); +} + +NamespaceLifeId life(const char * name, uint64_t id) +{ + const RootNamespace ns{name}; + return NamespaceLifeId::fromCatalogEntry(ns, UInt128{id}); +} + +} + +TEST(CASNamespaceJanitor, DeletesDeadFilesAndCheckpointFromOnePostListCatalogCut) +{ + CountingBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const auto dead = life("dead", 41); + const String file = layout.namespaceFilesPrefix(dead) + "part/data.bin"; + const String ckpt = layout.refCkptKey(dead); + ASSERT_EQ(backend.putIfAbsent(file, "file-bytes").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(ckpt, "ckpt-bytes").outcome, PutOutcome::Done); + backend.resetCounts(); + + NamespaceJanitor janitor(backend, layout, 100); + const NamespaceJanitorResult result = janitor.runOnePage(false, [] { return true; }); + + EXPECT_EQ(result.pages, 1u); + EXPECT_EQ(result.keys, 2u); + EXPECT_EQ(result.deleted, 2u); + EXPECT_FALSE(backend.get(file)); + EXPECT_FALSE(backend.get(ckpt)); + EXPECT_EQ(backend.listCount(layout.namespaceRootPrefix()), 1u); + EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(readGcMaintenanceState(backend, layout).state, GcMaintenanceState{}); +} + +TEST(CASNamespaceJanitor, RetainsEveryCurrentLifecycleAndSuppressesAmbiguousCut) +{ + CountingBackend backend; + const Layout layout("p"); + RefCatalog catalog; + CatalogEntry creating{.ns = RootNamespace{"creating"}, .state = NsState::Creating, .incarnation = UInt128{51}, + .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}}; + CatalogEntry live{.ns = RootNamespace{"live"}, .state = NsState::Live, .incarnation = UInt128{52}}; + CatalogEntry removing{.ns = RootNamespace{"removing"}, .state = NsState::Removing, .incarnation = UInt128{53}, + .removal_started_round = 1}; + catalog.entries = {creating, live, removing}; + seedCatalog(backend, layout, catalog); + for (const auto & entry : catalog.entries) + ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey( + NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation)), "keep").outcome, PutOutcome::Done); + + NamespaceJanitor janitor(backend, layout, 100); + const auto result = janitor.runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.deleteTotal(), 0u); +} + +TEST(CASNamespaceJanitor, CatalogFirstCreatingRetainsEveryObjectOfTheNewLife) +{ + CountingBackend backend; + const Layout layout("p"); + const CatalogEntry creating{ + .ns = RootNamespace{"catalog-first"}, + .state = NsState::Creating, + .incarnation = UInt128{54}, + .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 2, .fence_generation = 3}}; + seedCatalog(backend, layout, RefCatalog{.entries = {creating}}); + + /// The production creation order is the point: the catalog row is durable before either object. + const NamespaceLifeId creating_life + = NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation); + const String ckpt = layout.refCkptKey(creating_life); + const String file = layout.namespaceFilesPrefix(creating_life) + "data"; + ASSERT_EQ(backend.putIfAbsent(ckpt, "checkpoint").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(file, "file").outcome, PutOutcome::Done); + backend.resetCounts(); + + const NamespaceJanitorResult result + = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); + EXPECT_TRUE(backend.get(ckpt)); + EXPECT_TRUE(backend.get(file)); +} + +TEST(CASNamespaceJanitor, CancelledCreatingCheckpointIsReclaimedThroughPublicLifecycle) +{ + CountingBackend backend; + const Layout layout("p"); + const CatalogEntry creating{ + .ns = RootNamespace{"cancelled"}, + .state = NsState::Creating, + .incarnation = UInt128{55}, + .creator = CreatorFence{.server_root_id = "dead-srv", .writer_epoch = 4, .fence_generation = 5}}; + seedCatalog(backend, layout, RefCatalog{.entries = {creating}}); + const String ckpt = layout.refCkptKey( + NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation)); + ASSERT_EQ(backend.putIfAbsent(ckpt, "cancelled-checkpoint").outcome, PutOutcome::Done); + + ASSERT_EQ(CasRefCatalog::cancelStalledCreating( + backend, layout, creating, [](const CreatorFence &) { return true; }, + /*admitted_generation=*/7, [](uint64_t) {}), + CasRefCatalog::StalledCreatingCancelOutcome::Cancelled); + EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); + + const NamespaceJanitorResult result + = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(backend.get(ckpt)); +} + +TEST(CASNamespaceJanitor, SuppressionAndFenceLossDeleteNothing) +{ + CountingBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String first = layout.refCkptKey(life("dead-a", 61)); + const String second = layout.refCkptKey(life("dead-b", 62)); + ASSERT_EQ(backend.putIfAbsent(first, "first").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(second, "second").outcome, PutOutcome::Done); + + NamespaceJanitor janitor(backend, layout, 1); + EXPECT_EQ(janitor.runOnePage(true, [] { return true; }).deleted, 0u); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + << "a globally suppressed page is undecided and must not mint cleanup progress"; + EXPECT_EQ(backend.putCount(layout.gcMaintenanceStateKey()), 0u); + EXPECT_EQ(backend.casPutCount(layout.gcMaintenanceStateKey()), 0u); + EXPECT_EQ(janitor.runOnePage(false, [] { return false; }).deleted, 0u); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + << "fence loss must not mint progress past a page whose deletion was not authorized"; + EXPECT_TRUE(backend.get(first)); + EXPECT_TRUE(backend.get(second)); + EXPECT_EQ(backend.deleteTotal(), 0u); +} + +TEST(CASNamespaceJanitor, FenceLossOnRetainedOnlyPageDoesNotAdvanceCursor) +{ + CountingBackend backend; + const Layout layout("p"); + const CatalogEntry current{ + .ns = RootNamespace{"current"}, .state = NsState::Live, .incarnation = UInt128{63}}; + seedCatalog(backend, layout, RefCatalog{.entries = {current}}); + const NamespaceLifeId current_life + = NamespaceLifeId::fromCatalogEntry(current.ns, current.incarnation); + const String ckpt = layout.refCkptKey(current_life); + const String file = layout.namespaceFilesPrefix(current_life) + "data"; + ASSERT_EQ(backend.putIfAbsent(ckpt, "checkpoint").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(file, "file").outcome, PutOutcome::Done); + + const NamespaceJanitorResult result + = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return false; }); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_TRUE(backend.get(ckpt)); + EXPECT_TRUE(backend.get(file)); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + << "a tenure that observes fence loss cannot publish progress even when every object was retained"; +} + +TEST(CASNamespaceJanitor, FenceLossAfterLastDeleteRetainsCursorWithoutRollingBackDelete) +{ + CountingBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead = layout.refCkptKey(life("dead-after-delete", 64)); + ASSERT_EQ(backend.putIfAbsent(dead, "dead").outcome, PutOutcome::Done); + uint64_t fence_checks = 0; + + const NamespaceJanitorResult result + = NamespaceJanitor(backend, layout, 1).runOnePage(false, [&] { return fence_checks++ == 0; }); + + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(backend.get(dead)) + << "the exact delete completed under the fence and is never rolled back"; + EXPECT_EQ(fence_checks, 2u) + << "the fence must be checked before deletion and again immediately before cursor publication"; + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + << "losing the fence after the delete keeps this page selected for an idempotent retry"; +} + +TEST(CASNamespaceJanitor, CursorResumesThenResetsAtEnd) +{ + CountingBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const auto dead = life("dead", 71); + ASSERT_EQ(backend.putIfAbsent(layout.namespaceFilesPrefix(dead) + "a", "a").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(layout.namespaceFilesPrefix(dead) + "b", "b").outcome, PutOutcome::Done); + + NamespaceJanitor first_process(backend, layout, 1); + EXPECT_EQ(first_process.runOnePage(false, [] { return true; }).deleted, 1u); + const auto mid = readGcMaintenanceState(backend, layout); + ASSERT_EQ(mid.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(mid.state); + EXPECT_FALSE(mid.state->janitor_cursor.empty()); + NamespaceJanitor restarted_process(backend, layout, 1); + EXPECT_EQ(restarted_process.runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_TRUE(readGcMaintenanceState(backend, layout).state->janitor_cursor.empty()); +} + +TEST(CASNamespaceJanitor, TakesOneCatalogCutAfterListingAndContinuesPastMalformedKey) +{ + OrderedJanitorBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const auto dead = life("dead", 81); + const String valid = layout.namespaceFilesPrefix(dead) + "data"; + const String malformed = layout.namespaceStreamRootPrefix() + "not-a-life/_log/1-1.zst"; + const String malformed_state = layout.namespaceStateRootPrefix() + "not-a-life/_ckpt"; + ASSERT_EQ(backend.putIfAbsent(valid, "v").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(malformed, "bad").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(malformed_state, "bad-state").outcome, PutOutcome::Done); + backend.resetCounts(); + backend.events.clear(); + + const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(result.anomalies.empty()); + EXPECT_TRUE(backend.get(malformed)); + EXPECT_TRUE(backend.get(malformed_state)); + ASSERT_EQ(backend.events.size(), 2u); + EXPECT_EQ(backend.events[0], "list"); + EXPECT_EQ(backend.events[1], "catalog"); + EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); +} + +TEST(CASNamespaceJanitor, MalformedKeyIsFinalAndAdvancesCursor) +{ + CountingBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String first = layout.namespaceStreamRootPrefix() + "bad-a/_log/1-1.zst"; + const String second = layout.namespaceStreamRootPrefix() + "bad-b/_log/1-1.zst"; + ASSERT_EQ(backend.putIfAbsent(first, "first").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(second, "second").outcome, PutOutcome::Done); + + const NamespaceJanitorResult result + = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_FALSE(result.anomalies.empty()); + EXPECT_TRUE(backend.get(first)); + EXPECT_TRUE(backend.get(second)); + const GcMaintenanceReadResult progress = readGcMaintenanceState(backend, layout); + ASSERT_EQ(progress.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(progress.state); + EXPECT_FALSE(progress.state->janitor_cursor.empty()) + << "malformed keys are surfaced and skipped, but do not pin the cleanup cycle"; +} + +TEST(CASNamespaceJanitor, DuplicateCurrentLifeSuppressesWholePage) +{ + CountingBackend backend; + const Layout layout("p"); + RefCatalog catalog; + catalog.entries = { + CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128{91}}, + CatalogEntry{.ns = RootNamespace{"b"}, .state = NsState::Live, .incarnation = UInt128{91}}}; + seedCatalog(backend, layout, catalog); + const String dead_a = layout.refCkptKey(life("dead-a", 92)); + const String dead_b = layout.refCkptKey(life("dead-b", 93)); + ASSERT_EQ(backend.putIfAbsent(dead_a, "a").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(dead_b, "b").outcome, PutOutcome::Done); + const auto result = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_TRUE(backend.get(dead_a)); + EXPECT_TRUE(backend.get(dead_b)); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + << "an ambiguous catalog cut leaves the selected page undecided for an authoritative retry"; +} + +TEST(CASNamespaceJanitor, CorruptProgressResetsWithoutDeletingAndFilesOnlyOmittedCycleRetries) +{ + OmitFirstNamespacePageBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead = layout.namespaceFilesPrefix(life("dead", 101)) + "only-residue"; + ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(layout.gcMaintenanceStateKey(), "corrupt").outcome, PutOutcome::Done); + EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_TRUE(backend.get(dead)); + EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Valid); + EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_TRUE(backend.get(dead)); + EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_FALSE(backend.get(dead)); +} + +TEST(CASNamespaceJanitor, ExactTokenMismatchRetainsConcurrentReplacement) +{ + ReplaceBeforeJanitorDeleteBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead = layout.refCkptKey(life("dead-a", 111)); + const String later = layout.refCkptKey(life("dead-b", 112)); + ASSERT_EQ(backend.putIfAbsent(dead, "old").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(later, "later").outcome, PutOutcome::Done); + const auto result = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 0u); + ASSERT_TRUE(backend.get(dead)); + EXPECT_EQ(backend.get(dead)->bytes, "winner"); + EXPECT_TRUE(backend.get(later)); + const GcMaintenanceReadResult progress = readGcMaintenanceState(backend, layout); + ASSERT_EQ(progress.status, GcMaintenanceReadStatus::Valid); + ASSERT_TRUE(progress.state); + EXPECT_FALSE(progress.state->janitor_cursor.empty()) + << "an exact-token mismatch retains the rewrite but completes this page's decision"; +} + +TEST(CASNamespaceJanitor, TokenlessListHeadsDeadKeysAndRetainsConcurrentReplacement) +{ + TokenlessListBackend backend; + const Layout layout("p"); + const CatalogEntry current{ + .ns = RootNamespace{"current"}, .state = NsState::Live, .incarnation = UInt128{161}}; + seedCatalog(backend, layout, RefCatalog{.entries = {current}}); + const String live_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(current.ns, current.incarnation)); + const String dead_key = layout.refCkptKey(life("dead", 162)); + const String raced_key = layout.namespaceFilesPrefix(life("raced", 163)) + "data"; + ASSERT_EQ(backend.putIfAbsent(live_key, "live").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(dead_key, "dead").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(raced_key, "old").outcome, PutOutcome::Done); + backend.replace_on_head = raced_key; + backend.resetCounts(); + + const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + + EXPECT_EQ(result.deleted, 1u); + EXPECT_TRUE(result.anomalies.empty()); + EXPECT_TRUE(backend.get(live_key)); + EXPECT_FALSE(backend.get(dead_key)); + ASSERT_TRUE(backend.get(raced_key)); + EXPECT_EQ(backend.get(raced_key)->bytes, "winner"); + EXPECT_EQ(backend.headCount(live_key), 0u); + EXPECT_EQ(backend.headCount(dead_key), 1u); + EXPECT_EQ(backend.headCount(raced_key), 1u); + EXPECT_EQ(backend.deleteCount(dead_key), 1u); + EXPECT_EQ(backend.deleteCount(raced_key), 1u); +} + +TEST(CASNamespaceJanitor, TokenlessListRechecksFenceAfterHeadBeforeDelete) +{ + FenceLossDuringHeadBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead_key = layout.refCkptKey(life("dead", 164)); + ASSERT_EQ(backend.putIfAbsent(dead_key, "dead").outcome, PutOutcome::Done); + backend.resetCounts(); + + const auto result = NamespaceJanitor(backend, layout, 100).runOnePage( + false, [&] { return backend.fence_held; }); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.headCount(dead_key), 1u); + EXPECT_EQ(backend.deleteCount(dead_key), 0u); + EXPECT_TRUE(backend.get(dead_key)); +} + +TEST(CASNamespaceJanitor, PostListCatalogCutProtectsConcurrentCreationWithOneGet) +{ + const auto created = life("created", 121); + CatalogAfterListBackend backend(created); + const Layout layout("p"); + seedCatalog(backend, layout); + const String first = layout.refCkptKey(created); + const String second = layout.namespaceFilesPrefix(created) + "data"; + ASSERT_EQ(backend.putIfAbsent(first, "ckpt").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(second, "file").outcome, PutOutcome::Done); + backend.resetCounts(); + const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); + EXPECT_TRUE(backend.get(first)); + EXPECT_TRUE(backend.get(second)); +} + +TEST(CASNamespaceJanitor, BackendRejectedCursorResetsExactlyAndDeletesNothing) +{ + RejectCursorBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead = layout.refCkptKey(life("dead", 131)); + ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); + ASSERT_EQ(backend.putIfAbsent(layout.gcMaintenanceStateKey(), + encodeGcMaintenanceState({.janitor_cursor = "rejected"})).outcome, PutOutcome::Done); + EXPECT_THROW(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }), std::runtime_error); + EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_TRUE(backend.get(dead)); + EXPECT_TRUE(readGcMaintenanceState(backend, layout).state->janitor_cursor.empty()); +} + +TEST(CASNamespaceJanitor, CursorPublicationFailureIsLeakOnly) +{ + FailMaintenancePublicationBackend backend; + const Layout layout("p"); + seedCatalog(backend, layout); + const String dead = layout.refCkptKey(life("dead", 141)); + ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); + backend.fail_publication = true; + const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(result.anomalies.empty()); + EXPECT_FALSE(backend.get(dead)); +} + +TEST(CASNamespaceJanitorIntegration, RegularGcRoundDeletesDeadNamespaceBytes) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace live_namespace{"00/live@cas@"}; + fixture::admitLive(*backend, layout, live_namespace); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(fixture::fixtureLife(live_namespace)), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})).outcome, + PutOutcome::Done); + const String dead = layout.refCkptKey(life("dead", 151)); + ASSERT_EQ(backend->putIfAbsent(dead, "checkpoint").outcome, PutOutcome::Done); + + std::map namespace_cleanup; + Gc gc(store, UInt128{152}); + gc.setPhaseSink([&](const GcPhaseRecord & record) + { + if (record.phase == "namespace_cleanup") + namespace_cleanup = record.metrics; + }); + const RoundReport report = runRegularRoundReclaiming(gc); + gc.setPhaseSink({}); + + ASSERT_TRUE(report.acquired_lease); + EXPECT_FALSE(backend->get(dead)); + ASSERT_FALSE(namespace_cleanup.empty()); + EXPECT_EQ(namespace_cleanup["janitor_pages"], 1u); + EXPECT_GE(namespace_cleanup["janitor_keys"], 1u); + EXPECT_EQ(namespace_cleanup["janitor_deleted"], 1u); +} diff --git a/src/Disks/tests/gtest_cas_namespace_life_id.cpp b/src/Disks/tests/gtest_cas_namespace_life_id.cpp new file mode 100644 index 000000000000..9c2156d52aea --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_life_id.cpp @@ -0,0 +1,429 @@ +#include +#include +#include +#include +#include +/// Explicit rather than relying on a transitive path: `DEBUG_OR_SANITIZER_BUILD` (used below to gate +/// the `*DeathTest` split) must resolve in THIS translation unit. +#include + +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +namespace +{ + +/// The two incarnations of ONE namespace used throughout: distinct, nonzero, and rendering to two +/// hex segments that differ in the first character, so a key that carried the wrong one is visible +/// in the failure message rather than hidden in the tail of 32 digits. +UInt128 incarnationA() +{ + return (static_cast(0x1122'3344'5566'7788ULL) << 64) | static_cast(0x99aa'bbcc'ddee'ff01ULL); +} + +UInt128 incarnationB() +{ + return (static_cast(0xfedc'ba98'7654'3210ULL) << 64) | static_cast(0x0123'4567'89ab'cdefULL); +} + +const String kNs = "srv1/tbl@cas@"; +const String kHexA = "112233445566778899aabbccddeeff01"; +const String kHexB = "fedcba98765432100123456789abcdef"; +const String kTxn = "0000000000000007-000000000000008e"; + +/// Asserts that `body` refuses with CORRUPTED_DATA and that the message names `key` -- the refusal is +/// only useful to an operator if it says which object was rejected (the CI-observability rule). +template +void expectRefusalNaming(F && body, const String & key) +{ + try + { + std::forward(body)(); + FAIL() << "expected a refusal for key '" << key << "', got none"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA) << "for key '" << key << "'"; + EXPECT_NE(e.message().find(key), String::npos) + << "refusal does not name the offending key '" << key << "'; message: " << e.message(); + } +} + +/// The compile-time half of spec §9 r9-5 #3, one pair per migrated helper. The NEGATIVE proves the +/// namespace-only overload is gone; the POSITIVE proves the concept is actually looking at a real +/// member, so a typo in the requires-clause cannot make the negative pass vacuously. Both halves are +/// genuinely templated on `L`, so a missing member is a substitution failure rather than a hard error. +template +concept HasNamespaceOnlyRefsNamespacePrefix = + requires(const L & l, const RootNamespace & ns) { l.namespaceStreamPrefix(ns); }; +template +concept HasIncarnationRefsNamespacePrefix = + requires(const L & l, const NamespaceLifeId & id) { l.namespaceStreamPrefix(id); }; + +template +concept HasNamespaceOnlyRefLogKey = + requires(const L & l, const RootNamespace & ns, const RefTxnId & id) { l.refLogKey(ns, id); }; +template +concept HasIncarnationRefLogKey = + requires(const L & l, const NamespaceLifeId & ns_id, const RefTxnId & id) { l.refLogKey(ns_id, id); }; + +template +concept HasNamespaceOnlyRefSnapshotKey = + requires(const L & l, const RootNamespace & ns, const RefTxnId & id) { l.refSnapshotKey(ns, id); }; +template +concept HasIncarnationRefSnapshotKey = + requires(const L & l, const NamespaceLifeId & ns_id, const RefTxnId & id) { l.refSnapshotKey(ns_id, id); }; + +template +concept HasNamespaceOnlyRefCkptKey = + requires(const L & l, const RootNamespace & ns) { l.refCkptKey(ns); }; +template +concept HasIncarnationRefCkptKey = + requires(const L & l, const NamespaceLifeId & id) { l.refCkptKey(id); }; + +/// The namespace-FILE half of the same pattern (directive §1: "Delete all namespace-only ref and +/// namespace-file key overloads"), paired the same way. +template +concept HasNamespaceOnlyNamespaceFileKey = + requires(const L & l, const RootNamespace & ns, const String & n) { l.namespaceFileKey(ns, n); }; +template +concept HasIncarnationNamespaceFileKey = + requires(const L & l, const NamespaceLifeId & life, const String & n) { l.namespaceFileKey(life, n); }; + +template +concept HasNamespaceOnlyNamespaceFilesPrefix = + requires(const L & l, const RootNamespace & ns) { l.namespaceFilesPrefix(ns); }; +template +concept HasIncarnationNamespaceFilesPrefix = + requires(const L & l, const NamespaceLifeId & life) { l.namespaceFilesPrefix(life); }; + +/// The two OUT-OF-SCOPE families (Constraint 12, directive §2 "Keep these unchanged"): loose +/// mountpoint objects and part manifests keep the identity they have today. Each is paired in the +/// opposite direction from the migrated helpers -- the POSITIVE is the un-life-scoped overload that +/// must survive, the NEGATIVE is the life-scoped overload that must never appear. +template +concept HasUnscopedMountpointObjectKey = + requires(const L & l, const String & key) { l.mountpointObjectKey(key); }; +template +concept HasLifeScopedMountpointObjectKey = + requires(const L & l, const NamespaceLifeId & life, const String & key) { l.mountpointObjectKey(life, key); }; + +template +concept HasNamespaceOnlyManifestNamespacePrefix = + requires(const L & l, const RootNamespace & ns) { l.manifestNamespacePrefix(ns); }; +template +concept HasLifeScopedManifestNamespacePrefix = + requires(const L & l, const NamespaceLifeId & life) { l.manifestNamespacePrefix(life); }; + +} + +/// Every ref-layer key names one life by an opaque fixed-width physical id. The logical namespace is +/// intentionally absent and must be supplied by a catalog join. +TEST(CASNamespaceLifeId, KeysCarryTheIncarnationSegment) +{ + Layout l("p"); + const NamespaceLifeId id = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + const RefTxnId txn{7, 0x8e}; + const String life = "p/cas/ns/stream/" + kHexA + "/"; + const String state = "p/cas/ns/state/" + kHexA + "/"; + + EXPECT_EQ(l.namespaceStreamPrefix(id), life); + EXPECT_EQ(l.refLogKey(id, txn), life + "_log/" + kTxn + ".zst"); + EXPECT_EQ(l.refSnapshotKey(id, txn), life + "_snap/" + kTxn + ".zst"); + EXPECT_EQ(l.refCkptKey(id), state + "_ckpt"); + + const auto parsed_log = l.parseRefObjectKey(l.refLogKey(id, txn)); + ASSERT_TRUE(parsed_log.has_value()); + EXPECT_EQ(parsed_log->life_id, id.incarnation); + EXPECT_EQ(parsed_log->kind, RefObjectKind::Log); + EXPECT_EQ(parsed_log->txn_id, txn); + + const auto parsed_snap = l.parseRefObjectKey(l.refSnapshotKey(id, txn)); + ASSERT_TRUE(parsed_snap.has_value()); + EXPECT_EQ(parsed_snap->life_id, id.incarnation); + EXPECT_EQ(parsed_snap->kind, RefObjectKind::Snap); + + EXPECT_EQ(l.parseRefCkptKey(l.refCkptKey(id)), id.incarnation); +} + +/// The property the type exists for: two lives of the SAME namespace name share no key at all, so a +/// reborn namespace can neither read nor delete the previous life's objects by name. +TEST(CASNamespaceLifeId, TwoLivesOfOneNamespaceShareNoKeys) +{ + Layout l("p"); + const RefTxnId txn{7, 0x8e}; + const NamespaceLifeId first = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + const NamespaceLifeId second = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationB()); + + EXPECT_EQ(first.ns, second.ns); + EXPECT_NE(first, second); + EXPECT_NE(l.namespaceStreamPrefix(first), l.namespaceStreamPrefix(second)); + EXPECT_NE(l.refLogKey(first, txn), l.refLogKey(second, txn)); + EXPECT_NE(l.refCkptKey(first), l.refCkptKey(second)); + + /// Neither life's prefix covers the other: a LIST of one enumerates only its own objects. + EXPECT_FALSE(l.refLogKey(second, txn).starts_with(l.namespaceStreamPrefix(first))); + EXPECT_FALSE(l.refLogKey(first, txn).starts_with(l.namespaceStreamPrefix(second))); + + /// A key spelling the other life parses back to the OTHER id -- the parser reports what the key + /// says; it is the catalog, not the parser, that decides which lives are current. + const auto parsed = l.parseRefObjectKey(l.refLogKey(second, txn)); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->life_id, second.incarnation); + EXPECT_NE(parsed->life_id, first.incarnation); +} + +/// Zero is not a wildcard and not "the namespace itself": it can never be constructed, so it can +/// never reach a key builder. +/// +/// Both throws below raise `LOGICAL_ERROR`, which aborts the process in debug/sanitizer builds +/// instead of behaving like a catchable exception (`Common/Exception.cpp`'s `handle_error_code`) -- +/// `CASNamespaceLifeIdDeathTest.ZeroIncarnationIsUnconstructibleAborts` below proves the abort +/// positively in those builds instead. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASNamespaceLifeId, ZeroIncarnationIsUnconstructible) +{ + EXPECT_THROW(NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, UInt128{0}), DB::Exception); + EXPECT_THROW(renderIncarnation(UInt128{0}), DB::Exception); + EXPECT_NO_THROW(NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA())); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASNamespaceLifeIdDeathTest, ZeroIncarnationIsUnconstructibleAborts) +{ + EXPECT_DEATH({ (void)NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, UInt128{0}); }, "incarnation must be nonzero"); + EXPECT_DEATH({ (void)renderIncarnation(UInt128{0}); }, "incarnation must be nonzero"); + EXPECT_NO_THROW(NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA())); +} +#endif + +/// Generation-5 namespace-bearing keys are outside the generation-6 parser roots altogether. Pool +/// admission rejects their generation before any listed-key parser is involved. +TEST(CASNamespaceLifeId, GenerationFiveNamespaceBearingKeysAreOutsideTheFinalGrammar) +{ + Layout l("p"); + const String legacy_log = "p/cas/refs/" + kNs + "/_log/" + kTxn + ".zst"; + const String legacy_snap = "p/cas/refs/" + kNs + "/_snap/" + kTxn + ".zst"; + const String legacy_cleanup = "p/cas/refs/" + kNs + "/_cleanup/" + kTxn; + const String legacy_ckpt = "p/cas/refs/" + kNs + "/_ckpt"; + /// A single-segment namespace leaves nothing at all where the incarnation belongs. + const String legacy_single_segment = "p/cas/refs/srv1/_log/" + kTxn + ".zst"; + + EXPECT_FALSE(l.parseRefObjectKey(legacy_log)); + EXPECT_FALSE(l.parseRefObjectKey(legacy_snap)); + EXPECT_FALSE(l.parseRefObjectKey(legacy_cleanup)); + EXPECT_FALSE(l.parseRefObjectKey(legacy_single_segment)); + EXPECT_FALSE(l.parseRefCkptKey(legacy_ckpt)); +} + +/// An all-zero incarnation segment is well-formed hex naming no life, so it is corruption on the read +/// side exactly as it is unconstructible on the write side. +TEST(CASNamespaceLifeId, ParsersRefuseAZeroIncarnation) +{ + Layout l("p"); + const String zeros(32, '0'); + const String zero_log = "p/cas/ns/stream/" + zeros + "/_log/" + kTxn + ".zst"; + const String zero_ckpt = "p/cas/ns/state/" + zeros + "/_ckpt"; + + expectRefusalNaming([&] { l.parseRefObjectKey(zero_log); }, zero_log); + expectRefusalNaming([&] { l.parseRefCkptKey(zero_ckpt); }, zero_ckpt); +} + +/// The incarnation segment has ONE canonical spelling. A key that is nearly right -- wrong width, +/// upper case, a non-hex digit -- is refused rather than repaired, so two spellings of one life can +/// never both exist. +TEST(CASNamespaceLifeId, ParsersRefuseAMalformedIncarnationSegment) +{ + Layout l("p"); + const String upper = "112233445566778899AABBCCDDEEFF01"; + const String too_short = kHexA.substr(0, 31); + const String too_long = kHexA + "0"; + const String non_hex = kHexA.substr(0, 31) + "z"; + + for (const String & bad : {upper, too_short, too_long, non_hex}) + { + const String log_key = "p/cas/ns/stream/" + bad + "/_log/" + kTxn + ".zst"; + const String ckpt_key = "p/cas/ns/state/" + bad + "/_ckpt"; + expectRefusalNaming([&] { l.parseRefObjectKey(log_key); }, log_key); + expectRefusalNaming([&] { l.parseRefCkptKey(ckpt_key); }, ckpt_key); + } +} + +/// The boundary between "corrupt" and "not ours". Refusal is reserved for keys the parser has already +/// recognized as OUR ref objects; anything else keeps returning `std::nullopt`, because classifying an +/// untrusted listed key remains an ordinary "is this ours" question and a sweep must be able to walk +/// past foreign debris without an exception. +TEST(CASNamespaceLifeId, ForeignAndUnrecognizedKeysStayInert) +{ + Layout l("p"); + const NamespaceLifeId id = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + const RefTxnId txn{7, 0x8e}; + + /// Foreign top-level prefix (another pool, another subtree). + EXPECT_FALSE(l.parseRefObjectKey("q/cas/refs/" + kNs + "/" + kHexA + "/_log/" + kTxn + ".zst").has_value()); + EXPECT_FALSE(l.parseRefCkptKey("q/cas/refs/" + kNs + "/" + kHexA + "/_ckpt").has_value()); + EXPECT_FALSE(l.parseRefObjectKey("p/cas/manifests/" + kNs + "/" + kHexA + "/_log/" + kTxn).has_value()); + /// An unrecognized kind directory is not one of our ref objects, so its incarnation segment is + /// never even reached. + EXPECT_FALSE(l.parseRefObjectKey("p/cas/refs/" + kNs + "/_bogus/" + kTxn).has_value()); + /// A non-canonical transaction id likewise loses the key before the incarnation is judged. + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceStreamPrefix(id) + "_log/7-8e").has_value()); + /// The two parsers stay disjoint: neither claims the other's objects. + EXPECT_FALSE(l.parseRefObjectKey(l.refCkptKey(id)).has_value()); + EXPECT_FALSE(l.parseRefCkptKey(l.refLogKey(id, txn)).has_value()); + /// No namespace and no incarnation at all. + EXPECT_FALSE(l.parseRefObjectKey("p/cas/refs/_log/" + kTxn + ".zst").has_value()); + EXPECT_FALSE(l.parseRefCkptKey("p/cas/refs/_ckpt").has_value()); +} + +/// Namespace files are life-keyed too: `cas/ns/state//_files/`. The +/// round trip covers a flat name and a NESTED one, because the dedup log's segments live in a +/// table-level subdirectory and the nested shape is the one on the insert path. +TEST(CASNamespaceLifeId, NamespaceFileKeysCarryTheIncarnationSegment) +{ + Layout l("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + const String files = "p/cas/ns/state/" + kHexA + "/_files/"; + + EXPECT_EQ(l.namespaceFilesPrefix(life), files); + EXPECT_EQ(l.namespaceFileKey(life, "format_version.txt"), files + "format_version.txt"); + EXPECT_EQ(l.namespaceFileKey(life, "deduplication_logs/deduplication_log_1.txt"), + files + "deduplication_logs/deduplication_log_1.txt"); + + for (const String & name : {String("format_version.txt"), String("deduplication_logs/deduplication_log_1.txt")}) + { + const auto parsed = l.parseNamespaceFileKey(l.namespaceFileKey(life, name)); + ASSERT_TRUE(parsed.has_value()) << "for name '" << name << "'"; + EXPECT_EQ(parsed->life_id, life.incarnation); + EXPECT_EQ(parsed->relative_name, name); + } + + /// Two lives of one namespace share no file key either, and neither files prefix covers the other. + const NamespaceLifeId second = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationB()); + EXPECT_NE(l.namespaceFileKey(life, "format_version.txt"), l.namespaceFileKey(second, "format_version.txt")); + EXPECT_FALSE(l.namespaceFileKey(second, "format_version.txt").starts_with(l.namespaceFilesPrefix(life))); +} + +/// Generation-5 namespace-bearing file keys are outside the final parser root. Malformed ids under the +/// final state root are corruption and name the offending key. +TEST(CASNamespaceLifeId, NamespaceFileParserRefusesLegacyAndMalformedIncarnations) +{ + Layout l("p"); + const String zeros(32, '0'); + const String legacy = "p/roots/" + kNs + "/_files/format_version.txt"; + /// A single-segment namespace leaves nothing at all where the incarnation belongs. + const String legacy_single_segment = "p/roots/srv1/_files/format_version.txt"; + const String zero_inc = "p/cas/ns/state/" + zeros + "/_files/format_version.txt"; + + EXPECT_FALSE(l.parseNamespaceFileKey(legacy)); + EXPECT_FALSE(l.parseNamespaceFileKey(legacy_single_segment)); + expectRefusalNaming([&] { l.parseNamespaceFileKey(zero_inc); }, zero_inc); + + const String upper = "112233445566778899AABBCCDDEEFF01"; + const String too_short = kHexA.substr(0, 31); + const String too_long = kHexA + "0"; + const String non_hex = kHexA.substr(0, 31) + "z"; + for (const String & bad : {upper, too_short, too_long, non_hex}) + { + const String key = "p/cas/ns/state/" + bad + "/_files/format_version.txt"; + expectRefusalNaming([&] { l.parseNamespaceFileKey(key); }, key); + } +} + +/// The corrupt/not-ours boundary for file keys, mirroring the ref parsers': refusal is reserved for +/// keys already identified as OUR namespace files by their reserved `_files` segment. A loose +/// mountpoint object has no such segment and is a legitimate inhabitant of `roots/`, so it must parse +/// as `std::nullopt` and never as damage. +TEST(CASNamespaceLifeId, ForeignAndMountpointKeysStayInertForTheFileParser) +{ + Layout l("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + + EXPECT_FALSE(l.parseNamespaceFileKey("q/roots/" + kNs + "/" + kHexA + "/_files/x").has_value()); + EXPECT_FALSE(l.parseNamespaceFileKey(l.mountpointObjectKey("srv1/clickhouse_access_check_abc")).has_value()); + EXPECT_FALSE(l.parseNamespaceFileKey("p/cas/refs/" + kNs + "/" + kHexA + "/_files/x").has_value()); + /// The files prefix itself names no file: there is no relative name after the reserved segment. + EXPECT_FALSE(l.parseNamespaceFileKey(l.namespaceFilesPrefix(life)).has_value()); + /// And the ref parsers do not claim a file key. + EXPECT_FALSE(l.parseRefObjectKey(l.namespaceFileKey(life, "x")).has_value()); + EXPECT_FALSE(l.parseRefCkptKey(l.namespaceFileKey(life, "x")).has_value()); +} + +/// Physical file keys use only `life_id`, so changing the logical spelling cannot redirect a key. A +/// relative name may still contain `_files` and round-trips after the fixed delimiter. +TEST(CASNamespaceLifeId, PhysicalFileKeysIgnoreLogicalNamespaceSpelling) +{ + Layout l("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{kNs}, incarnationA()); + const NamespaceLifeId differently_named = NamespaceLifeId::fromCatalogEntry( + RootNamespace{"different/_files/spelling"}, incarnationA()); + EXPECT_EQ(l.namespaceFilesPrefix(life), l.namespaceFilesPrefix(differently_named)); + const String nested_name = "deduplication_logs/_files/log_1.txt"; + const auto parsed = l.parseNamespaceFileKey(l.namespaceFileKey(life, nested_name)); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(parsed->life_id, life.incarnation); + EXPECT_EQ(parsed->relative_name, nested_name); +} + +/// The "cannot compile" half of spec §9 r9-5 #3: after this task there is no way to reach a ref-layer +/// key from a namespace alone, so dropping the incarnation is a compile error rather than an aliasing +/// bug. Each helper is asserted twice -- the namespace-only form absent, the incarnation form present. +TEST(CASNamespaceLifeId, NamespaceOnlyKeyHelpersDoNotExist) +{ + static_assert(!HasNamespaceOnlyRefsNamespacePrefix); + static_assert(HasIncarnationRefsNamespacePrefix); + + static_assert(!HasNamespaceOnlyRefLogKey); + static_assert(HasIncarnationRefLogKey); + + static_assert(!HasNamespaceOnlyRefSnapshotKey); + static_assert(HasIncarnationRefSnapshotKey); + + static_assert(!HasNamespaceOnlyRefCkptKey); + static_assert(HasIncarnationRefCkptKey); + + static_assert(!HasNamespaceOnlyNamespaceFileKey); + static_assert(HasIncarnationNamespaceFileKey); + + static_assert(!HasNamespaceOnlyNamespaceFilesPrefix); + static_assert(HasIncarnationNamespaceFilesPrefix); + + SUCCEED(); +} + +/// Directive §1's remaining requirements on the type, fenced rather than fixed: the type declares no +/// conversion operator and no `RootNamespace` constructor takes a `NamespaceLifeId`, so nothing +/// interconverts in either direction today and only an explicit `.ns` crosses. Without these +/// assertions a later convenience conversion would land unnoticed, and dropping the incarnation would +/// become representable again -- which is the property the whole re-keying rests on. +TEST(CASNamespaceLifeId, NamespaceLifeIdAndRootNamespaceDoNotInterconvert) +{ + static_assert(!std::convertible_to); + static_assert(!std::constructible_from); + static_assert(!std::is_default_constructible_v); + + SUCCEED(); +} + +/// The out-of-scope fences, and they are POSITIVE on purpose: Constraint 12 keeps loose mountpoint +/// objects and part manifests on the identity they have today, so this task must NOT have qualified +/// them. If a negative here fails, someone added a life-scoped overload to a family the amendment +/// explicitly excluded; if a positive fails, someone removed the un-scoped one those callers use. +TEST(CASNamespaceLifeId, MountpointObjectsAndManifestsStayUnqualified) +{ + static_assert(HasUnscopedMountpointObjectKey); + static_assert(!HasLifeScopedMountpointObjectKey); + + static_assert(HasNamespaceOnlyManifestNamespacePrefix); + static_assert(!HasLifeScopedManifestNamespacePrefix); + + SUCCEED(); +} diff --git a/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp b/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp new file mode 100644 index 000000000000..7acab161e63d --- /dev/null +++ b/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp @@ -0,0 +1,530 @@ +#include "cas_test_helpers.h" +#include +#include +#include +/// Explicit rather than relying on a transitive path: `DEBUG_OR_SANITIZER_BUILD` (used below to gate +/// the `*DeathTest` split) must resolve in THIS translation unit. +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; + extern const int NETWORK_ERROR; +} + +/// Stage B Task 3 (spec §3, the ref-chain catalog's creation lifecycle): the three-conditional-write +/// sequence that carries a namespace from nothing to `Live` -- +/// 1. catalog CAS: insert `{ns, Creating, fresh incarnation, creator}` (`CasRefCatalog::createNamespace`, +/// built on Task 2's `casAdmitEntry`); +/// 2. `_ckpt` create (`CasRefCatalog::completeCreation`, step 2 -- Stage A's `publishCkpt` unchanged); +/// 3. catalog CAS: `Creating -> Live`, re-presenting the creator's admission GENERATION and +/// value-CASing the OBSERVED entry (`completeCreation`, step 3 -- the "ZombieGoLive" guard) -- +/// plus stale-`Creating` reconciliation (`CasRefCatalog::reconcileStaleCreator`) and the publication +/// gate (`checkPublicationAdmittedOrThrow`). +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace +{ + +/// A fence that never refuses, for tests whose subject is not the fence -- same helper, same intent, +/// as `gtest_cas_ref_ckpt.cpp`'s identically-named constant (not shared: each `_ckpt`/catalog test file +/// defines its own copy, matching that file's own precedent). +const std::function ALWAYS_ADMITTED = [](uint64_t) {}; + +/// A deadline far enough out that only the test's own contention decides the outcome -- mirrors +/// `gtest_cas_ref_ckpt.cpp`'s `generousDeadline`. +CkptDeadline generousDeadline() +{ + return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; +} + +CreatorFence creatorFence(const String & srid, uint64_t writer_epoch, uint64_t fence_generation = 1) +{ + return CreatorFence{.server_root_id = srid, .writer_epoch = writer_epoch, .fence_generation = fence_generation}; +} + +/// A `is_creator_fence_terminal` stub that answers the same fixed verdict for every fence -- for tests +/// whose subject is not terminality itself (that predicate's own tests live in `gtest_cas_mount.cpp`, +/// next to `isCreatorFenceTerminal`, the real implementation this stub stands in for). +std::function fixedTerminality(bool terminal) +{ + return [terminal](const CreatorFence &) { return terminal; }; +} + +const CatalogEntry * findEntryForTest(const RefCatalog & catalog, const RootNamespace & ns) +{ + for (const CatalogEntry & e : catalog.entries) + if (e.ns.string() == ns.string()) + return &e; + return nullptr; +} + +/// Raw lifecycle tests operate below `Pool::open`, so model an already-bootstrapped pool explicitly. +class InitializedCatalogBackend : public InMemoryBackend +{ +public: + InitializedCatalogBackend() + { + CasRefCatalog::initializeEmptyForNewPool(*this, Layout("p")); + } +}; + +} + +/// --------------------------------------------------------------------------------------------- +/// Happy path: all three writes land +/// --------------------------------------------------------------------------------------------- + +TEST(CASNsCreationLifecycle, HappyPathReachesLiveWithADurableCkptAndAStableIncarnation) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence creator = creatorFence("srv1", /*writer_epoch=*/5); + + const auto outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, creator, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Live); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Live); + EXPECT_EQ(entry->creator, std::nullopt) << "creator is forbidden outside Creating (strict grammar)"; + const UInt128 incarnation = entry->incarnation; + EXPECT_NE(incarnation, UInt128(0)); + + const std::optional ckpt = readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, incarnation)); + ASSERT_TRUE(ckpt.has_value()) << "step 2's _ckpt must be durable"; + EXPECT_EQ(ckpt->ckpt.life_epoch, 5u) << "INV-4's genesis epoch is the creator's writer_epoch"; + + /// Re-reading the catalog again must show the SAME incarnation -- nothing mints a second one. + EXPECT_EQ(CasRefCatalog::read(backend, layout).catalog.entries.at(0).incarnation, incarnation); +} + +/// --------------------------------------------------------------------------------------------- +/// `createNamespace` refuses a namespace that already has an entry (Task 2 review's own note: this +/// is Task 3's job, not `casAdmitEntry`'s duplicate-namespace grammar refusal). +/// --------------------------------------------------------------------------------------------- + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASNsCreationLifecycle, CreateNamespaceRejectsAnAlreadyExistingEntry) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence creator = creatorFence("srv1", 1); + ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creator, 1, ALWAYS_ADMITTED, generousDeadline()), + CasRefCatalog::NamespaceCreationOutcome::Live); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv2", 2), 1, ALWAYS_ADMITTED, generousDeadline()); + }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASNsCreationLifecycleDeathTest, CreateNamespaceRejectsAnAlreadyExistingEntryAborts) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence creator = creatorFence("srv1", 1); + ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creator, 1, ALWAYS_ADMITTED, generousDeadline()), + CasRefCatalog::NamespaceCreationOutcome::Live); + + EXPECT_DEATH( + { + CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv2", 2), 1, ALWAYS_ADMITTED, generousDeadline()); + }, + "already carries a catalog entry"); +} +#endif + +/// --------------------------------------------------------------------------------------------- +/// `Creating` forbids publication +/// --------------------------------------------------------------------------------------------- + +TEST(CASNsCreationLifecycle, CreatingForbidsPublication) +{ + RefCatalog catalog; + catalog.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, + .incarnation = UInt128(1), .creator = creatorFence("srv1", 1)}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { CasRefCatalog::checkPublicationAdmittedOrThrow(catalog, RootNamespace{"a"}); }); +} + +TEST(CASNsCreationLifecycle, LiveAndRemovingAndAbsentAllAdmitPublication) +{ + RefCatalog catalog; + catalog.entries.push_back(CatalogEntry{.ns = RootNamespace{"live"}, .state = NsState::Live, .incarnation = UInt128(1)}); + catalog.entries.push_back(CatalogEntry{.ns = RootNamespace{"removing"}, .state = NsState::Removing, .incarnation = UInt128(2)}); + EXPECT_NO_THROW(CasRefCatalog::checkPublicationAdmittedOrThrow(catalog, RootNamespace{"live"})); + EXPECT_NO_THROW(CasRefCatalog::checkPublicationAdmittedOrThrow(catalog, RootNamespace{"removing"})); + EXPECT_NO_THROW(CasRefCatalog::checkPublicationAdmittedOrThrow(catalog, RootNamespace{"never-heard-of"})); +} + +/// --------------------------------------------------------------------------------------------- +/// ZombieGoLive: fenced-out between the `_ckpt` publish and the `Creating -> Live` CAS +/// --------------------------------------------------------------------------------------------- + +/// A fence callback that admits its FIRST call (spent by step 2's `publishCkpt`) and refuses every +/// call after (spent by step 3's `mutate`) -- deterministically reproducing "fenced out between the +/// `_ckpt` create and the `Creating -> Live` CAS" without a second thread or fault injection. +namespace +{ +std::function admittedOnceThenFenced() +{ + auto calls = std::make_shared(0); + return [calls](uint64_t admitted) + { + if (++*calls > 1) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); + }; +} +} + +TEST(CASNsCreationLifecycle, FencedOutBetweenTheCkptPublishAndGoLiveRefusesAndLeavesEntryCreating) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence creator = creatorFence("srv1", 5); + + const auto outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, creator, /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()); + EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Creating) << "step 3 never ran its CAS -- ZombieGoLive refuses before sending it"; + ASSERT_TRUE(entry->creator.has_value()); + EXPECT_EQ(*entry->creator, creator); + + /// Step 2's _ckpt DID land (it is not what the fence check gates) -- CKPT-FAILED-BIRTH-DEBRIS is a + /// different mechanism (the OLD `RefOpKind::NamespaceBirth` writer, `Pool/CasRefLedger.cpp`); this + /// driver's own `_ckpt` is simply left in place for whichever actor next reconciles this entry. + EXPECT_TRUE(readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation)).has_value()); +} + +/// Regression (CI PR#2073, `03611_freeze_partition_parallel_verbose` under `amd_tsan, cas s3 storage`): +/// sibling openers of the SAME namespace race `resolveNamespaceLife`'s "no entry" read the same way +/// concurrent per-part `ALTER TABLE ... FREEZE` threads race the table's one shadow-store namespace. +/// The loser's own `createNamespace` read lands AFTER the winner's step 1, observing `Creating` -- that +/// must send the loser back through the resume loop (`Superseded`), never abort the server. +TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsStillCreatingEntryReportsSupersededNotAbort) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence winner = creatorFence("srv1", 1); + + /// Leaves the entry in `Creating` without reaching `Live` -- the same shape `resolveNamespaceLife` + /// observes when a sibling thread's `casAdmitEntry` has landed but its `completeCreation` has not. + const auto winner_outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, winner, /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()); + ASSERT_EQ(winner_outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut); + ASSERT_EQ(CasRefCatalog::read(backend, layout).catalog.entries.at(0).state, NsState::Creating); + + /// The loser: a second call, as if a sibling thread's own outer "no entry" read had raced ahead of + /// this one. Same fence as the winner (sibling threads of one query share a mount's fence) -- + /// exercising exactly the case `resolveNamespaceLife`'s "own fence -> completeCreation" branch is + /// built to resume, never a `LOGICAL_ERROR` abort. + const auto loser_outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, winner, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + EXPECT_EQ(loser_outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); + + /// Nothing about the winner's own still-`Creating` entry was disturbed by the loser's refused call. + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Creating); + ASSERT_TRUE(entry->creator.has_value()); + EXPECT_EQ(*entry->creator, winner); +} + +/// Second catch-point of the same CI PR#2073 race, distinct from the test above. That test starts the +/// winner FIRST, so the loser's own outer pre-check read (`createNamespace`'s `read(...)` before step +/// 1) already observes `Creating` and takes the fast top-of-function refusal. This test instead lands +/// the winner's ENTIRE `createNamespace` call inside the window between the loser's pre-check read +/// (which observes NOTHING) and the loser's own step 1 read -- the shape CI actually hit as an +/// encode-time `LOGICAL_ERROR` ("entries are not canonically ordered ... no duplicate namespace"), not +/// the top-of-function one: both openers pass the pre-check, so both proceed to admit a row for the +/// same namespace, and only `createNamespaceStep1`'s own per-read recheck (not `createNamespace`'s +/// single upfront read) can catch it. +TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsFullCreateBetweenPreCheckAndStep1ReportsSupersededNotAbort) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence winner = creatorFence("srv1", 1); + const CreatorFence loser = creatorFence("srv1", 2); + + /// Fires exactly once, inside the LOSER's `createNamespace` call, after its pre-check read already + /// observed no entry -- synchronously runs the winner's own full `createNamespace` to completion + /// (all the way to `Live`) before the loser's step 1 performs its own first read. The production + /// call site swaps the hook into a local before invoking it, so the global is already empty by the + /// time this body runs: the winner's own nested call, and every later call in the test, run + /// hook-free without this body needing to clear it itself. + CasRefCatalog::setCreateNamespaceStep1PreReadHookForTest([&] + { + const auto winner_outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, winner, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + ASSERT_EQ(winner_outcome, CasRefCatalog::NamespaceCreationOutcome::Live); + }); + + const auto loser_outcome = CasRefCatalog::createNamespace( + backend, layout, 1, ns, loser, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + EXPECT_EQ(loser_outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); + + /// Exactly one row for `ns`, owned by the winner, at `Live` -- the loser's refused admission left + /// no trace and did not disturb it. + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + size_t rows_for_ns = 0; + for (const CatalogEntry & e : snap.catalog.entries) + if (e.ns.string() == ns.string()) + ++rows_for_ns; + EXPECT_EQ(rows_for_ns, 1u); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Live); + EXPECT_FALSE(entry->creator.has_value()) << "Live entries carry no creator fence"; +} + +/// --------------------------------------------------------------------------------------------- +/// Token-stale: the observed entry no longer matches at the `Creating -> Live` CAS +/// --------------------------------------------------------------------------------------------- + +TEST(CASNsCreationLifecycle, EntryStolenByAConcurrentReconcilerRefusesGoLiveAndLeavesTheStolenEntryAlone) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence original_creator = creatorFence("srv1", 5); + const CreatorFence thief = creatorFence("srv2", 9); + + /// Write 1 only -- models "crash after write 1": no _ckpt yet, entry still Creating. + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(42), .creator = original_creator}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + + /// `check_fence_or_throw` is the seam this driver calls on EVERY attempt -- once inside step 2's + /// `publishCkpt`, once more inside step 3's own `mutate` -- so smuggling a REAL concurrent write + /// into it (rather than faking the outcome) has to land on the SECOND call specifically, or the + /// steal itself would run twice (and the second run would see its own first result and refuse). + /// This reproduces "stolen between the creator's _ckpt publish and its Creating -> Live CAS" + /// without a second thread. The steal itself must succeed (asserted), so the mismatch + /// `completeCreation` sees below is the entry ACTUALLY changing, not a contrived stub. + auto calls = std::make_shared(0); + const std::function steal_before_the_go_live_cas = [&, calls](uint64_t) + { + if (++*calls == 2) + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, thief, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + }; + + const auto outcome = CasRefCatalog::completeCreation( + backend, layout, entry, /*admitted_generation=*/1, steal_before_the_go_live_cas, generousDeadline()); + EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); + + /// `read`'s `Snapshot` is bound to a name here, not chained through a temporary: a `const + /// CatalogEntry *` taken from `.catalog` of an unbound temporary dangles the instant the full + /// expression ends, which every other site in this file (and the copy/paste that spread it) got + /// wrong until ASan caught it. + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); + ASSERT_NE(after, nullptr); + EXPECT_EQ(after->state, NsState::Creating) << "the ORIGINAL creator's attempt wrote nothing -- only the thief's CAS did"; + ASSERT_TRUE(after->creator.has_value()); + EXPECT_EQ(*after->creator, thief) << "the entry is exactly what the thief left it as, untouched by our refused attempt"; + EXPECT_EQ(after->incarnation, entry.incarnation) << "reconciliation never mints a fresh incarnation"; +} + +/// --------------------------------------------------------------------------------------------- +/// Both stale at once: fence moved AND the entry was stolen -- refused (fence checked first). +/// --------------------------------------------------------------------------------------------- + +TEST(CASNsCreationLifecycle, BothFenceAndEntryStaleRefusesGoLiveViaTheFenceCheckWhichRunsFirst) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence original_creator = creatorFence("srv1", 5); + const CreatorFence thief = creatorFence("srv2", 9); + + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(42), .creator = original_creator}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + + /// Same steal as the test above, landing on the SECOND `check_fence_or_throw` call (step 3's own + /// `mutate`, not step 2's `publishCkpt`) -- but this one ALSO throws on that same second call, so + /// both axes go stale in the SAME `mutate` invocation. `completeCreation`'s fence check runs before + /// its entry check (documented ordering), so this is reported `FencedOut`; the assertions below + /// confirm the entry ALSO changed, so the test is not merely re-proving the fence-only case above. + auto calls = std::make_shared(0); + const std::function steal_and_fence_before_the_go_live_cas = [&, calls](uint64_t admitted) + { + if (++*calls == 2) + { + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, thief, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); + } + }; + + const auto outcome = CasRefCatalog::completeCreation( + backend, layout, entry, /*admitted_generation=*/1, steal_and_fence_before_the_go_live_cas, generousDeadline()); + EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut) + << "both checks would refuse; the fence check speaks first by this driver's fixed ordering"; + + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); + ASSERT_NE(after, nullptr); + ASSERT_TRUE(after->creator.has_value()); + EXPECT_EQ(*after->creator, thief) << "confirms the entry axis really did go stale too, not just the fence"; +} + +/// --------------------------------------------------------------------------------------------- +/// Stale-`Creating` reconciliation +/// --------------------------------------------------------------------------------------------- + +TEST(CASNsCreationLifecycle, ReconcileRefusedWhileTheOriginalCreatorFenceIsStillLive) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), + .creator = creatorFence("srv1", 5)}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + + const auto outcome = CasRefCatalog::reconcileStaleCreator( + backend, layout, entry, creatorFence("srv2", 9), fixedTerminality(false), /*admitted_generation=*/1, ALWAYS_ADMITTED); + EXPECT_EQ(outcome, CasRefCatalog::ReconcileCreatorOutcome::CreatorFenceStillLive); + + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); + ASSERT_NE(after, nullptr); + EXPECT_EQ(*after, entry) << "refused -- nothing written"; +} + +TEST(CASNsCreationLifecycle, ReconcileSucceedsTokenExactlyAfterTheOriginalCreatorFenceIsTerminalThenResumesToLive) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CreatorFence original_creator = creatorFence("srv1", 5); + const CreatorFence new_creator = creatorFence("srv2", 9); + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), .creator = original_creator}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); /// "crash after write 1" -- no _ckpt yet + + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, new_creator, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + + CatalogEntry taken_over = entry; + taken_over.creator = new_creator; + const CasRefCatalog::Snapshot snap_mid = CasRefCatalog::read(backend, layout); + const CatalogEntry * mid = findEntryForTest(snap_mid.catalog, ns); + ASSERT_NE(mid, nullptr); + EXPECT_EQ(*mid, taken_over) << "creator moved to the new actor; state and incarnation unchanged"; + + const auto outcome = CasRefCatalog::completeCreation( + backend, layout, taken_over, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Live); + + const CasRefCatalog::Snapshot snap_final = CasRefCatalog::read(backend, layout); + const CatalogEntry * final_entry = findEntryForTest(snap_final.catalog, ns); + ASSERT_NE(final_entry, nullptr); + EXPECT_EQ(final_entry->state, NsState::Live); + EXPECT_EQ(final_entry->incarnation, entry.incarnation) << "the SAME incarnation throughout -- resumption, not rebirth"; + const std::optional ckpt = readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(final_entry->ns, final_entry->incarnation)); + ASSERT_TRUE(ckpt.has_value()); + EXPECT_EQ(ckpt->ckpt.life_epoch, new_creator.writer_epoch) + << "the RESUMING actor's writer_epoch is the genesis epoch that actually landed"; +} + +/// "Stale token at reconciliation -> fail closed": a SECOND reconciler racing the first, both reading +/// the SAME stale `observed` before either writes. +TEST(CASNsCreationLifecycle, ReconcileFailsClosedWhenTheEntryAlreadyChanged) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const RootNamespace ns{"a"}; + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), + .creator = creatorFence("srv1", 5)}; + CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + + const CreatorFence first_reconciler = creatorFence("srv2", 9); + const CreatorFence second_reconciler = creatorFence("srv3", 11); + + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, first_reconciler, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + + /// The second reconciler still holds the ORIGINAL `entry` it read before either of them wrote -- + /// token-exactness must refuse it even though the terminality predicate would still say yes. + const auto outcome = CasRefCatalog::reconcileStaleCreator( + backend, layout, entry, second_reconciler, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + EXPECT_EQ(outcome, CasRefCatalog::ReconcileCreatorOutcome::EntryChanged); + + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); + ASSERT_NE(after, nullptr); + ASSERT_TRUE(after->creator.has_value()); + EXPECT_EQ(*after->creator, first_reconciler) << "the second reconciler's refused attempt changed nothing"; +} + +/// --------------------------------------------------------------------------------------------- +/// Preconditions: `completeCreation`/`reconcileStaleCreator` refuse anything but a `Creating` entry +/// with a creator fence -- a caller bug, not a race, hence `LOGICAL_ERROR`. +/// --------------------------------------------------------------------------------------------- + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASNsCreationLifecycle, CompleteCreationRejectsANonCreatingEntry) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + CasRefCatalog::completeCreation(backend, layout, live, 1, ALWAYS_ADMITTED, generousDeadline()); + }); +} + +TEST(CASNsCreationLifecycle, ReconcileStaleCreatorRejectsANonCreatingEntry) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + CasRefCatalog::reconcileStaleCreator(backend, layout, live, creatorFence("srv2", 2), fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASNsCreationLifecycleDeathTest, CompleteCreationRejectsANonCreatingEntryAborts) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; + EXPECT_DEATH( + { CasRefCatalog::completeCreation(backend, layout, live, 1, ALWAYS_ADMITTED, generousDeadline()); }, + "not a Creating entry"); +} + +TEST(CASNsCreationLifecycleDeathTest, ReconcileStaleCreatorRejectsANonCreatingEntryAborts) +{ + InitializedCatalogBackend backend; + Layout layout("p"); + const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; + EXPECT_DEATH( + { + CasRefCatalog::reconcileStaleCreator(backend, layout, live, creatorFence("srv2", 2), fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + }, + "not a Creating entry"); +} +#endif diff --git a/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp b/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp new file mode 100644 index 000000000000..901b223cb90f --- /dev/null +++ b/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp @@ -0,0 +1,279 @@ +#include + +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" +#include + +namespace DB::ErrorCodes +{ + extern const int UNKNOWN_FORMAT_VERSION; +} + +/// Namespace files are keyed by an opaque LIFE, not by its name: `cas/ns/state//_files/` +/// (Stage B Task 4b, directive design change 2). This file pins the three properties that re-key exists +/// to produce, and the one it must NOT produce. +/// +/// THE HOLE IT CLOSES. Before the re-key, a namespace file lived at a name-keyed prefix shared by every +/// life of that name. A file the store's LIST omitted therefore survived namespace removal -- nothing +/// enumerated it, so nothing deleted it -- and then became VISIBLE to the next namespace created under +/// the same name, because that namespace read the same prefix. Deletion was load-bearing for +/// correctness, and deletion depends on enumeration, which is the one thing an object store is allowed +/// to be late about (`HintHoleBackendOn` is that lateness as an interface -- see its doc). +/// +/// WHY THE KEY IS THE FIX AND THE DELETE IS NOT. After the re-key the old file is at a prefix the new +/// life cannot name. It is unreachable whether or not it was ever deleted, so a blind LIST costs +/// STORAGE and nothing else -- the directive's "LIST omission may only leak storage, never visibility, +/// rebirth or deletion safety". `ColdReaderUsesCatalogCutWhileOldFileSurvivesRemoval` asserts exactly +/// that split by leaving the old object physically present and byte-intact through the real lifecycle. +/// +/// WHAT REBIRTH NO LONGER WAITS FOR. Catalog removal depends on folded terminal evidence for the old +/// opaque life, not on a physical-empty proof. `RebirthDoesNotWaitForFilesToBeEmpty` keeps old `_files` +/// bytes present while that evidence is adopted; their later reclamation belongs to the perpetual +/// janitor and is not a precondition for same-name reuse. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const String kNsString = "00/aa@cas@"; +const String kFile = "format_version.txt"; + +const UInt128 kGcId = hexToU128("00000000000000000000000000000001"); + +/// Create a real catalog life and a replay-valid `Live` ref table through the production writer path. +void publishWithProductionBirth(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +} + +/// THE COUPLED HEADLINE. A real removal reaches a catalog-absent cut even when LIST permanently omits +/// an old-life file. A cold reader follows that catalog cut rather than the physical residue, while an +/// already-held exact life remains stale-or-NotFound and can never cross into the successor life. +TEST(CASNsFileIncarnation, ColdReaderUsesCatalogCutWhileOldFileSurvivesRemoval) +{ + auto backend = std::make_shared>(); + PoolPtr store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + const RootNamespace ns{kNsString}; + const String old_bytes = "old-life\n"; + const String successor_bytes = "successor-life\n"; + + publishWithProductionBirth(store, ns, "predecessor"); + const std::optional old_life = store->namespaceFilesLifeIfReadable(ns); + ASSERT_TRUE(old_life); + store->putNamespaceFile(*old_life, kFile, old_bytes); + const String old_key = layout.namespaceFileKey(*old_life, kFile); + backend->hide(old_key); + + ASSERT_TRUE(backend->head(old_key).exists) << "the lie must be in LIST only -- the object is durable"; + ASSERT_TRUE(store->listNamespaceFiles(*old_life).empty()) + << "precondition: enumeration omits the file, so no cleanup pass can ever find it"; + const size_t holes_before_gc = backend->holesServed(); + + store->dropNamespace(ns); + ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)); + + Gc gc(store, kGcId); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "N: the production terminal must fold"; + ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)) + << "the terminal fold alone must not erase its catalog row"; + (void)runRegularRoundReclaiming(gc); + ASSERT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)) + << "N+1: the pre-fold drain must erase the exact completed Removing row"; + ASSERT_GT(backend->holesServed(), holes_before_gc) + << "the GC janitor must observe the injected LIST hole after the explicit precondition LIST"; + + const auto old_head = backend->head(old_key); + ASSERT_TRUE(old_head.exists) << "logical removal must not depend on physical empty"; + const auto old_object = backend->get(old_key); + ASSERT_TRUE(old_object); + EXPECT_EQ(old_object->bytes, old_bytes); + + PoolPtr cold = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "cold-reader", .gc_fold_max_defer_rounds = 0}); + EXPECT_FALSE(cold->namespaceFilesLifeIfReadable(ns)) + << "a fresh reader follows the absent catalog row, not discoverable or exact-key old bytes"; + + publishWithProductionBirth(store, ns, "successor"); + const std::optional successor_life = cold->namespaceFilesLifeIfReadable(ns); + ASSERT_TRUE(successor_life); + ASSERT_NE(successor_life->incarnation, old_life->incarnation); + cold->putNamespaceFile(*successor_life, kFile, successor_bytes); + EXPECT_EQ(cold->getNamespaceFile(*successor_life, kFile), successor_bytes); + + const std::optional retained_old = store->getNamespaceFile(*old_life, kFile); + EXPECT_TRUE(!retained_old || *retained_old == old_bytes) + << "an exact predecessor life may be stale or NotFound, but never aliases successor bytes"; + EXPECT_NE(retained_old, std::optional{successor_bytes}); + + for (const String & key : backend->touchedKeys()) + EXPECT_EQ(key.find("/_cleanup/"), String::npos) << key; +} + +/// The non-minting reader assignment site accepts exactly a catalog `Live` row. `Creating`, +/// `Removing`, and absence neither install a runtime life nor mutate durable catalog/stream state. +TEST(CASNsFileIncarnation, FreshReaderAssignsOnlyLiveCatalogLifeWithoutMutation) +{ + auto backend = std::make_shared(); + PoolPtr store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace creating{"00/creating@cas@"}; + const RootNamespace live{"00/live@cas@"}; + const RootNamespace removing{"00/removing@cas@"}; + const RootNamespace absent{"00/absent@cas@"}; + + CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + .ns = creating, + .state = NsState::Creating, + .incarnation = UInt128{31}, + .creator = CreatorFence{.server_root_id = "foreign", .writer_epoch = 7, .fence_generation = 1}}); + CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + .ns = live, .state = NsState::Live, .incarnation = UInt128{32}}); + CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + .ns = removing, .state = NsState::Live, .incarnation = UInt128{33}}); + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == removing; + }); + chassert(it != next.entries.end()); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + + /// Only a `Live` catalog row is readable. Give that exact life the empty checkpoint authority + /// that production creation publishes; the other rows deliberately remain raw lifecycle states. + writeRecoverableCkptForRawFixture(*backend, layout, live, RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->resetCounts(); + EXPECT_FALSE(store->namespaceFilesLifeIfReadable(creating)); + EXPECT_FALSE(store->namespaceFilesLifeIfReadable(removing)); + EXPECT_FALSE(store->namespaceFilesLifeIfReadable(absent)); + const std::optional readable = store->namespaceFilesLifeIfReadable(live); + ASSERT_TRUE(readable); + EXPECT_EQ(readable->incarnation, UInt128{32}); + + EXPECT_FALSE(store->refTableLifeForTest(creating)); + EXPECT_FALSE(store->refTableLifeForTest(removing)); + EXPECT_FALSE(store->refTableLifeForTest(absent)); + ASSERT_TRUE(store->refTableLifeForTest(live)); + EXPECT_EQ(store->refTableLifeForTest(live)->incarnation, UInt128{32}); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +/// A real GC fold records terminal evidence for the previous life while its namespace-file debris +/// remains physically present. Lifecycle completion is therefore independent of `_files` enumeration; +/// the perpetual janitor may reclaim the bytes later without participating in the removal proof. +TEST(CASNsFileIncarnation, RebirthDoesNotWaitForFilesToBeEmpty) +{ + auto backend = std::make_shared(); + PoolPtr store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{kNsString}; + + /// A removed namespace (a bare `remove_namespace` transaction -- no committed refs, so no + /// owner-removal edge confounds this with an unconditional delete path) whose only surviving + /// physical objects are namespace files: one flat, one nested in the dedup-log shape. + { + RefOp remove_op; + remove_op.kind = RefOpKind::RemoveNamespace; + appendRefLogSeed(*backend, layout, ns, {remove_op}); + } + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + const String debris_key = layout.namespaceFileKey(life, kFile); + backend->putIfAbsent(debris_key, "1\n"); + backend->putIfAbsent(layout.namespaceFileKey(life, "deduplication_logs/deduplication_log_1.txt"), "records"); + + Gc gc(store, kGcId); + gc.runRegularRound(); + + /// Folding the terminal records positive evidence on the same life row even though files remain. + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(state.snap_generation, 0u); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + const auto row_it = seal.ref_lives.find(life.incarnation); + ASSERT_NE(row_it, seal.ref_lives.end()); + ASSERT_TRUE(row_it->second.cleanup_evidence.has_value()); + EXPECT_EQ(row_it->second.cleanup_evidence->remove_txn_id, (RefTxnId{1, 1})); + EXPECT_TRUE(backend->head(debris_key).exists) << "cleanup evidence does not gate on physical deletion"; +} + +/// An old-format pool carrying unqualified `roots//_files/x` keys is REFUSED AT OPEN. It is not +/// read, not migrated, and not silently re-keyed: the file layer rides Task 4's format bump B, and the +/// pool-open floor is what makes "there is nothing to migrate" true rather than merely intended. +/// +/// Asserted at OPEN rather than at the parser on purpose: `Layout` has no unqualified key constructor +/// at all (a compile-time concept check in `gtest_cas_namespace_life_id.cpp` pins that, and +/// `parseNamespaceFileKey`'s refusal of a legacy key is pinned there too), so the only reachable +/// question left is whether a pool that CONTAINS such keys can be opened. It cannot. +TEST(CASNsFileIncarnation, LegacyUnqualifiedFileKeyIsRefusedAtOpen) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + + /// A generation-5 `_pool_meta`: the current encoder's output with its header generation moved back + /// one, so every other byte is exactly what that generation really wrote. + PoolMeta meta; + meta.pool_id = hexToU128("0123456789abcdef0123456789abcdef"); + meta.blob_header_len = 256; + meta.min_reader_generation = kNamespaceLifeKeyedGeneration - 1; + meta.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + String encoded = encodePoolMeta(meta); + const String current_v = "\"v\":" + std::to_string(G_BUILD); + const String legacy_v = "\"v\":" + std::to_string(kNamespaceLifeKeyedGeneration); + const size_t at = encoded.find(current_v); + /// Guard the substitution itself: a silent no-op here would leave a CURRENT-generation pool and the + /// test would pass by opening a pool it believes it downgraded. + ASSERT_NE(at, String::npos) << "pool-meta header no longer spells its generation as " << current_v; + encoded.replace(at, current_v.size(), legacy_v); + ASSERT_NE(encoded.find(legacy_v), String::npos); + backend->putIfAbsent(layout.poolMetaKey(), encoded); + + /// The legacy artifact this task removes: a namespace file keyed by NAME ONLY, with no incarnation + /// segment. Written as raw bytes because no code path in the tree can produce this key any more. + backend->putIfAbsent("p/roots/" + kNsString + "/_files/" + kFile, "1\n"); + + try + { + openPoolForTest(backend); + FAIL() << "an old-format pool must fail closed at open, naming recreation"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + EXPECT_NE(e.message().find("recreate"), String::npos) + << "the refusal must tell the operator what to do; got: " << e.message(); + } +} diff --git a/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp b/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp new file mode 100644 index 000000000000..3166230f5ec8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp @@ -0,0 +1,247 @@ +#include + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsBool gc_enabled; +} + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; + extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const String kTableUuid = "a11a11a1-1111-4111-8111-111111111111"; +const String kTablePath = "a11/a11a11a1-1111-4111-8111-111111111111"; +const String kFile = "format_version.txt"; +const String kFilePath = kTablePath + "/" + kFile; +const UInt128 kLife2Id = hexToU128("22222222222222222222222222222222"); + +struct DiskFixture +{ + DB::ObjectStoragePtr object_storage; + std::shared_ptr storage; +}; + +DiskFixture openDiskFixture() +{ + static std::atomic counter{0}; + const String unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + auto settings = makeSettingsForTest( + "srv1", std::filesystem::temp_directory_path() / ("cas_ns_file_contract_scratch_" + unique)); + settings[DB::ContentAddressedSetting::gc_enabled] = false; + + DiskFixture fixture; + fixture.object_storage = makeLocalObjectStorageForTest(); + fixture.storage = std::make_shared( + fixture.object_storage, "pool", "srv1", "", nullptr, settings); + fixture.storage->startup(); + return fixture; +} + +void writeVerbatimThroughDisk( + DB::ContentAddressedMetadataStorage & storage, const String & path, const String & bytes) +{ + auto transaction = storage.createTransaction(); + auto & ca_transaction = dynamic_cast(*transaction); + auto buffer = ca_transaction.writeFile(path, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteMode::Rewrite, {}); + ASSERT_NE(dynamic_cast(buffer.get()), nullptr); + DB::writeString(bytes, *buffer); + buffer->finalize(); +} + +/// Delete the current catalog life through the production exact-removal authority while retaining all +/// old physical bytes and the original process's already-resident runtime. +void deleteCatalogLife( + DB::ContentAddressedMetadataStorage & storage, const NamespaceLifeId & life1) +{ + Backend & backend = storage.store()->backend(); + const Layout & layout = storage.store()->layout(); + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == life1.ns && entry.incarnation == life1.incarnation; + }); + if (it == next.entries.end()) + throw DB::Exception( + DB::ErrorCodes::LOGICAL_ERROR, "Missing fixture catalog life '{}'", life1.ns.string()); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + + const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(backend, layout); + const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == life1.ns && entry.incarnation == life1.incarnation; + }); + if (it == snapshot.catalog.entries.end()) + throw DB::Exception( + DB::ErrorCodes::LOGICAL_ERROR, "Missing Removing fixture catalog life '{}'", life1.ns.string()); + + CasFoldSeal parent; + parent.ref_lives.emplace(life1.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); + if (CasRefCatalog::deleteCompletedRemoving( + backend, layout, *it, parent, 1, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }) + != CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted) + throw DB::Exception( + DB::ErrorCodes::LOGICAL_ERROR, "Failed to delete fixture catalog life '{}'", life1.ns.string()); + +} + +NamespaceLifeId admitReplacementLife( + DB::ContentAddressedMetadataStorage & storage, const NamespaceLifeId & life1) +{ + if (life1.incarnation == kLife2Id) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Fixture life ids unexpectedly collide"); + const NamespaceLifeId life2 = NamespaceLifeId::fromCatalogEntry(life1.ns, kLife2Id); + CasRefCatalog::casAdmitEntry( + storage.store()->backend(), storage.store()->layout(), storage.store()->poolConfig().gc_shards, CatalogEntry{ + .ns = life2.ns, .state = NsState::Live, .incarnation = life2.incarnation}); + return life2; +} + +NamespaceLifeId replaceCatalogLife( + DB::ContentAddressedMetadataStorage & storage, const NamespaceLifeId & life1) +{ + deleteCatalogLife(storage, life1); + return admitReplacementLife(storage, life1); +} + +NamespaceLifeId currentLife(DB::ContentAddressedMetadataStorage & storage) +{ + const RootNamespace ns = storage.liveNamespace(kTableUuid); + const auto life = storage.store()->namespaceFilesLifeIfReadable(ns); + if (!life) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Fixture namespace '{}' has no readable life", ns.string()); + return *life; +} + +} + +/// This storage already holds life 1. Reusing its warm runtime after same-name rebirth is a retained +/// life-handle operation, not a fresh logical-name admission: it may still see predecessor bytes (or +/// answer absent), but the opaque physical life id makes successor bytes structurally unreachable. +TEST(CASNamespaceFileReadContract, HeldLifeAfterSameNameRebirthNeverSeesSuccessorBytes) +{ + DiskFixture fixture = openDiskFixture(); + writeVerbatimThroughDisk(*fixture.storage, kFilePath, "life-1\n"); + const NamespaceLifeId life1 = currentLife(*fixture.storage); + const NamespaceLifeId life2 = replaceCatalogLife(*fixture.storage, life1); + fixture.storage->store()->putNamespaceFile(life2, kFile, "life-2\n"); + + const std::optional held_read = fixture.storage->tryGetInManifestBytes(kFilePath); + EXPECT_NE(held_read, std::optional{"life-2\n"}); + EXPECT_TRUE(!held_read || held_read == std::optional{"life-1\n"}); + EXPECT_EQ(fixture.storage->store()->getNamespaceFile(life1, kFile), std::optional{"life-1\n"}); +} + +/// Mutation caught: capturing only the namespace name and resolving it when the buffer finalizes would +/// overwrite life 2. The real buffer must retain the exact life admitted when it was opened. +TEST(CASNamespaceFileReadContract, DelayedInlineFinalizeCannotChangeSuccessorTokenOrBytes) +{ + DiskFixture fixture = openDiskFixture(); + writeVerbatimThroughDisk(*fixture.storage, kFilePath, "life-1-before\n"); + const NamespaceLifeId life1 = currentLife(*fixture.storage); + + auto delayed_transaction = fixture.storage->createTransaction(); + auto & ca_transaction = dynamic_cast(*delayed_transaction); + auto delayed_buffer = ca_transaction.writeFile( + kFilePath, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteMode::Rewrite, {}); + ASSERT_NE(dynamic_cast(delayed_buffer.get()), nullptr); + DB::writeString("life-1-delayed\n", *delayed_buffer); + + const NamespaceLifeId life2 = replaceCatalogLife(*fixture.storage, life1); + fixture.storage->store()->putNamespaceFile(life2, kFile, "life-2-stable\n"); + Backend & backend = fixture.storage->store()->backend(); + const Layout & layout = fixture.storage->store()->layout(); + const String life1_key = layout.namespaceFileKey(life1, kFile); + const String life2_key = layout.namespaceFileKey(life2, kFile); + ASSERT_TRUE(std::filesystem::exists(nativeKeyUnder(fixture.object_storage, life2_key))); + const HeadResult life2_before = backend.head(life2_key); + ASSERT_TRUE(life2_before.exists); + const auto life2_body_before = backend.get(life2_key); + ASSERT_TRUE(life2_body_before.has_value()); + ASSERT_EQ(life2_body_before->bytes, "life-2-stable\n"); + + bool stale_failure = false; + try + { + delayed_buffer->finalize(); + } + catch (const DB::Exception & e) + { + stale_failure = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(e.message().find("retrying later"), String::npos); + } + + const HeadResult life2_after = backend.head(life2_key); + ASSERT_TRUE(life2_after.exists); + EXPECT_EQ(life2_after.token, life2_before.token); + const auto life2_body_after = backend.get(life2_key); + ASSERT_TRUE(life2_body_after.has_value()); + EXPECT_EQ(life2_body_after->bytes, "life-2-stable\n"); + + if (!stale_failure) + { + ASSERT_TRUE(std::filesystem::exists(nativeKeyUnder(fixture.object_storage, life1_key))); + const auto life1_body = backend.get(life1_key); + ASSERT_TRUE(life1_body.has_value()); + EXPECT_EQ(life1_body->bytes, "life-1-delayed\n"); + } +} + +/// `listNamespaceFiles` derives its LIST prefix from `layout.namespaceFilesPrefix(life)` -- a physical +/// life-scoped stream, not the catalog. Listing under a held life must cost exactly one LIST of the +/// files prefix and nothing else. +TEST(CASNamespaceFileReadContract, ListThroughHeldLifeIssuesZeroCatalogRequests) +{ + auto backend = std::make_shared(); + PoolPtr store = openPoolForTest(backend); + const NamespaceLifeId life = fixture::fixtureLife(RootNamespace{"00/ns_file_list_contract@cas@"}); + const String prefix = store->layout().namespaceFilesPrefix(life); + + store->putNamespaceFile(life, "a.txt", "a\n"); + store->putNamespaceFile(life, "b.txt", "b\n"); + backend->resetCounts(); + + const std::vector names = store->listNamespaceFiles(life); + + std::vector sorted_names = names; + std::sort(sorted_names.begin(), sorted_names.end()); + EXPECT_EQ(sorted_names, (std::vector{"a.txt", "b.txt"})); + + /// Positive control: the journal recorded exactly one LIST against the namespace-file stream + /// prefix and nothing else, so the touched-set assertion below names an absence, not a + /// recorder that never saw anything. + EXPECT_EQ(backend->listCount(prefix), 1u); + EXPECT_EQ(backend->listTotal(), 1u); + EXPECT_EQ(backend->touchedKeys(), std::vector{prefix}); +} diff --git a/src/Disks/tests/gtest_cas_observability.cpp b/src/Disks/tests/gtest_cas_observability.cpp new file mode 100644 index 000000000000..bf989f1e3206 --- /dev/null +++ b/src/Disks/tests/gtest_cas_observability.cpp @@ -0,0 +1,492 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event CASGCRetiredCondemned; +extern const Event CASGCRetireReplaced; +extern const Event CASMountRenewalAttempts; +extern const Event CASMountRenewalRetries; +extern const Event CASMountRenewalResolved; +extern const Event CASMountRenewalRecovered; +extern const Event CASMountRenewalDeadlineExceeded; +} + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::currentRetiredSet; + +namespace +{ + +PoolPtr openPool(std::shared_ptr & b) +{ + b = std::make_shared(); + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +class RenewalCounterBackend final : public InMemoryBackend +{ +public: + enum class Fault : uint8_t + { + None, + ThrowBefore, + LandThenThrow, + }; + + using InMemoryBackend::putOverwrite; + + Fault fault = Fault::None; + + PutResult putOverwrite( + const String & key, + const String & bytes, + const Token & expected, + const ObjectMeta & meta) override + { + const Fault current = std::exchange(fault, Fault::None); + if (current == Fault::ThrowBefore) + throw Poco::TimeoutException("injected renewal timeout before commit"); + + PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + if (current == Fault::LandThenThrow) + throw Poco::TimeoutException("injected renewal response loss after commit"); + return result; + } +}; + +CasRequestBudget renewalCounterBudget(uint32_t max_attempts = 2) +{ + return CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = max_attempts, + .lease_safety_margin_ms = 20, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }; +} + +struct RenewalCounterSnapshot +{ + uint64_t attempts; + uint64_t retries; + uint64_t resolved; + uint64_t recovered; + uint64_t deadline_exceeded; +}; + +RenewalCounterSnapshot renewalCounters() +{ + using ProfileEvents::global_counters; + return { + .attempts = global_counters[ProfileEvents::CASMountRenewalAttempts].load(), + .retries = global_counters[ProfileEvents::CASMountRenewalRetries].load(), + .resolved = global_counters[ProfileEvents::CASMountRenewalResolved].load(), + .recovered = global_counters[ProfileEvents::CASMountRenewalRecovered].load(), + .deadline_exceeded = global_counters[ProfileEvents::CASMountRenewalDeadlineExceeded].load(), + }; +} + +void expectRenewalCounterDelta( + const RenewalCounterSnapshot & before, + const RenewalCounterSnapshot & after, + uint64_t attempts, + uint64_t retries, + uint64_t resolved, + uint64_t recovered, + uint64_t deadline_exceeded) +{ + EXPECT_EQ(after.attempts - before.attempts, attempts); + EXPECT_EQ(after.retries - before.retries, retries); + EXPECT_EQ(after.resolved - before.resolved, resolved); + EXPECT_EQ(after.recovered - before.recovered, recovered); + EXPECT_EQ(after.deadline_exceeded - before.deadline_exceeded, deadline_exceeded); +} + +/// Publish ONE ref naming a single-blob part through the real writer sequence (mirrors +/// `publishOneBlobPart` in `gtest_cas_gc_leak.cpp`, duplicated here because that helper has internal +/// linkage in its own translation unit). +ManifestId publishOneBlobPart( + const PoolPtr & s, const RootNamespace & ns, const String & ref, const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + DB::Cas::ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob -> promote. + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +TEST(CASObservability, RenewalCountersHaveExactPhysicalAndLogicalDeltas) +{ + const auto run = [](RenewalCounterBackend::Fault fault, uint64_t attempts, uint64_t retries, uint64_t resolved, uint64_t recovered) + { + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "renewal-counter-" + std::to_string(attempts) + "-" + std::to_string(resolved), + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = renewalCounterBudget(), + .boot_ms_fn = [&] { return boot_ms; }, + }); + backend->fault = fault; + const RenewalCounterSnapshot before = renewalCounters(); + EXPECT_NO_THROW(store->renewWatermarkOnce()); + const RenewalCounterSnapshot after = renewalCounters(); + expectRenewalCounterDelta(before, after, attempts, retries, resolved, recovered, 0); + }; + + run(RenewalCounterBackend::Fault::None, /*attempts=*/1, /*retries=*/0, /*resolved=*/0, /*recovered=*/0); + run(RenewalCounterBackend::Fault::ThrowBefore, /*attempts=*/2, /*retries=*/1, /*resolved=*/0, /*recovered=*/1); + run(RenewalCounterBackend::Fault::LandThenThrow, /*attempts=*/1, /*retries=*/0, /*resolved=*/1, /*recovered=*/1); +} + +TEST(CASObservability, ExternalLeaseDeadlineCountsOnceWithoutReconstructingAttempts) +{ + auto backend = std::make_shared(); + uint64_t boot_ms = 100; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "renewal-deadline-counter", + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = renewalCounterBudget(), + .boot_ms_fn = [&] { return boot_ms; }, + }); + + /// The confirmed external safety deadline is 1080. At 1071 a ten-millisecond physical attempt + /// no longer fits, so the logical renewal ends without reconstructing a sent attempt. + boot_ms = 1071; + const RenewalCounterSnapshot before = renewalCounters(); + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + const RenewalCounterSnapshot after = renewalCounters(); + expectRenewalCounterDelta( + before, after, /*attempts=*/0, /*retries=*/0, /*resolved=*/0, /*recovered=*/0, + /*deadline_exceeded=*/1); +} + +} + +/// B170/Task 1 (Part A audit events): `PartWriteTxn::stageManifest` writes a part-manifest body but never +/// emitted an audit row for it — the log could not answer "when was this manifest written." Verifies +/// the emitted `ManifestPut` event (exactly once per successful stage). +TEST(CASObservability, StageManifestEmitsManifestPut) +{ + std::shared_ptr b; + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto s = openPool(b); + s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + + const RootNamespace ns{"srv/tbl@cas@"}; + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/all_0_0_0", .intended_namespace = ns}); + ManifestEntry e; + e.path = "f"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "AAA"; + const ManifestId id = build->stageManifest({e}); + s->setEventSink(nullptr); + + EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + [](const CasEvent & x){ return x.type == CasEventType::ManifestPut; }), 1); + + const auto it = std::find_if(seen.begin(), seen.end(), + [](const CasEvent & x){ return x.type == CasEventType::ManifestPut; }); + ASSERT_NE(it, seen.end()); + EXPECT_EQ(it->object_kind, CasEventObjectKind::Manifest); + EXPECT_EQ(it->object_hash, manifestRefDebugString(id.ref)); + EXPECT_FALSE(it->token.empty()); +} + +/// `PartWriteTxn::abandon` removes a live precommit's owner binding (the correctness-bearing step) but never +/// audited the removal — the log could not distinguish "never precommitted" from "precommitted then +/// abandoned." Verifies the emitted `PrecommitRemoved` event (exactly once, only when a precommit was +/// actually live). +TEST(CASObservability, AbandonEmitsPrecommitRemoved) +{ + std::shared_ptr b; + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto s = openPool(b); + + const RootNamespace ns{"srv/tbl@cas@"}; + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/all_0_0_0", .intended_namespace = ns}); + ManifestEntry e; + e.path = "f"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "AAA"; + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(ns, "all_0_0_0", id); + + s->setEventSink([&](const CasEvent & x){ seen.push_back(x); }); + build->abandon(); + s->setEventSink(nullptr); + + EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }), 1); + + const auto it = std::find_if(seen.begin(), seen.end(), + [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }); + ASSERT_NE(it, seen.end()); + EXPECT_EQ(it->namespace_, ns.string()); + EXPECT_EQ(it->ref_name, "all_0_0_0"); + EXPECT_EQ(it->object_kind, CasEventObjectKind::Root); + EXPECT_EQ(it->object_hash, manifestRefDebugString(id.ref)); +} + +/// A build that never precommitted has nothing to remove: `abandon` must not fabricate a +/// `PrecommitRemoved` row for a binding that was never live. +TEST(CASObservability, AbandonWithoutPrecommitEmitsNoPrecommitRemoved) +{ + std::shared_ptr b; + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto s = openPool(b); + + const RootNamespace ns{"srv/tbl@cas@"}; + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/all_0_0_0", .intended_namespace = ns}); + ManifestEntry e; + e.path = "f"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "AAA"; + build->stageManifest({e}); /// staged, never precommitted + + s->setEventSink([&](const CasEvent & x){ seen.push_back(x); }); + build->abandon(); + s->setEventSink(nullptr); + + EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }), 0); +} + +/// Task 2 (Part A audit fix, 2026-07-08): the republication-supersede branch inside `closeBlob` +/// (`CasBlobInDegree.cpp`) used to peek the current token via `head_blob` — the FRESH-CONDEMN +/// observation hook — which double-emitted `blob_retire` alongside `blob_retire_replaced` and +/// double-counted `CASGCRetiredCondemned` for what is really one physical condemnation (republication +/// replaced a stale retired entry with the current token). Drives the same condemn-A / republish-B / +/// drop-B sequence as `CASGCLeak.ResurrectReplacedIncarnationReclaimed`, then isolates the ONE round +/// that folds B's create+drop and supersedes A's stale retired entry: that round must emit exactly one +/// `blob_retire_replaced` (carrying the STALE token A in `detail["superseded_token"]`), ZERO +/// `blob_retire` for this hash, one `CASGCRetireReplaced` increment, and NO `CASGCRetiredCondemned` +/// double-count. +TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) +{ + std::shared_ptr b; + std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto s = openPool(b); + const RootNamespace ns{"test/tbl"}; + const String P = "republish-payload-audit"; + + /// 1. Publish ref r1 -> token A referenced; drop it; ONE GC round condemns A (retired, not deleted). + publishOneBlobPart(s, ns, "r1", P); + const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hA.exists); + s->dropRef(ns, "r1"); + s->renewWatermarkOnce(); + + Gc gc(s, hexToU128("000000000000000000000000000000ab")); + { + const RoundReport rep = gc.runRegularRound(); + ASSERT_TRUE(rep.acquired_lease); + } + { + const auto lm = DB::Cas::tests::loadMetaForTest(*b, s->layout(), u128Of(P)); + ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned) + << "precondition: token A must be condemned before republication"; + } + + /// 2. RESURRECT: r2 dedup-hits P while A is condemned -> mints a fresh incarnation B; drop it too. + publishOneBlobPart(s, ns, "r2", P); + const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); + ASSERT_TRUE(hB.exists); + ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a new incarnation token B"; + s->dropRef(ns, "r2"); + s->renewWatermarkOnce(); + + /// 3. The NEXT round folds r2's create+drop in one pass and must SUPERSEDE A's stale retired entry + /// with a fresh condemn of B (peek, not the fresh-condemn `head_blob` hook). Capture events + the + /// counters for exactly THIS round. + using ProfileEvents::global_counters; + const auto condemned_before = global_counters[ProfileEvents::CASGCRetiredCondemned].load(); + const auto replaced_before = global_counters[ProfileEvents::CASGCRetireReplaced].load(); + + s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + const RoundReport rep = gc.runRegularRound(); + s->setEventSink(nullptr); + ASSERT_TRUE(rep.acquired_lease); + + const auto condemned_after = global_counters[ProfileEvents::CASGCRetiredCondemned].load(); + const auto replaced_after = global_counters[ProfileEvents::CASGCRetireReplaced].load(); + + /// Phase 3 (mixed-algo pools): event `object_hash` renders are `blobIdOf(ref)` (":"), + /// never a bare hex. + const String hash_hex = DB::Cas::blobIdOf(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(P))}); + const auto is_this_blob = [&](const CasEvent & e){ return e.object_hash == hash_hex; }; + + EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + [&](const CasEvent & e){ return is_this_blob(e) && e.type == CasEventType::BlobRetire; }), 0) + << "supersede must not also emit blob_retire (that is the fresh-condemn hook's event)"; + + std::vector replaced_events; + std::copy_if(seen.begin(), seen.end(), std::back_inserter(replaced_events), + [&](const CasEvent & e){ return is_this_blob(e) && e.type == CasEventType::BlobRetireReplaced; }); + ASSERT_EQ(replaced_events.size(), 1u) << "exactly one blob_retire_replaced for the supersede"; + EXPECT_EQ(replaced_events[0].token, hB.token.value) << "the event's own token is the fresh CURRENT token B"; + ASSERT_TRUE(replaced_events[0].detail.count("superseded_token")); + EXPECT_FALSE(replaced_events[0].detail.at("superseded_token").empty()); + EXPECT_EQ(replaced_events[0].detail.at("superseded_token"), hA.token.value) + << "superseded_token must name the stale token (A) that republication replaced"; + + EXPECT_EQ(replaced_after - replaced_before, 1u) << "CASGCRetireReplaced increments exactly once"; + EXPECT_EQ(condemned_after - condemned_before, 0u) + << "supersede peek must not fresh-condemn -- CASGCRetiredCondemned must not double-count"; + + /// Size-unit regression guard (audit fix, 2026-07-08): `peek_head` used to return the RAW + /// `backend.head(...)` size (physical, header-included), while the fresh-condemn hook `head_blob` + /// strips the pool's fixed blob header via `retiredLogicalSize` before the size lands in + /// `RetiredEntry.size`. That mismatch meant supersede-minted entries and fresh-condemn entries carried + /// two different unit conventions in the SAME persisted `RetiredSet`. The superseded entry (now naming + /// the fresh token B) must carry the LOGICAL size -- i.e. the payload length, with the pool's blob + /// header already stripped -- exactly like a fresh condemn of the same blob would. + const std::vector retired = currentRetiredSet(*b, s->layout(), /*shard*/0); + const auto it = std::find_if(retired.begin(), retired.end(), + [&](const RetiredEntry & e){ return e.kind == ObjectKind::Blob && e.ref == DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(P))}; }); + ASSERT_NE(it, retired.end()) << "the superseded entry must be present in the current retired set"; + EXPECT_EQ(it->token.value, hB.token.value) << "the persisted entry names the fresh CURRENT token B"; + EXPECT_EQ(it->size, P.size()) + << "supersede must persist the LOGICAL size (payload length, header stripped), matching what " + "a fresh condemn of the same blob would carry -- not the raw physical (header-included) size"; +} + +/// Task 3 (Part B, `clickhouse-disks cas-inspect`): `caInspectToJson` is a FREE function (no +/// disk/backend involved) that decodes any CA bucket object at `key` and renders it as JSON, purely +/// by matching `key` against `Layout`'s prefixes/key-shapes and calling the matching `decode*`. +/// These tests drive it directly against real encoder output (one per recognized key shape) plus the +/// unknown-key fail-closed path — the same function the CLI command (`CommandCaInspect.cpp`) calls. + +/// The legacy mutable ref-shard object is gone (snapshot+log ref model); inspect now decodes the two +/// immutable ref objects. A `_snap/.proto` renders as a ref-table snapshot... +TEST(CASObservability, CaInspectDecodesRefSnapshotToJson) +{ + using DB::Cas::tests::committedRow; + using DB::Cas::tests::minimalLiveSnapshot; + Layout layout("p"); + const RootNamespace ns{"srv/tbl@cas@"}; + const RefTxnId snap_id{1, 7}; + const RefTableSnapshot snap = minimalLiveSnapshot(ns.string(), snap_id, + {committedRow("all_0_0_0", ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1})}); + const String key = layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), snap_id); + const String json = caInspectToJson( + layout, key, encodeRefTableSnapshot(snap), DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("object":"ref_snapshot")"), String::npos) << json; + EXPECT_NE(json.find(R"("namespace":"srv/tbl@cas@")"), String::npos) << json; + EXPECT_NE(json.find(R"("snapshot_id":{"writer_epoch":1,"ref_sequence":7})"), String::npos) << json; + EXPECT_NE(json.find(R"("ref_name":"all_0_0_0")"), String::npos) << json; + EXPECT_NE(json.find(R"("precommits":[])"), String::npos) << json; + EXPECT_EQ(json.find("\"lifecycle\""), String::npos) + << "generation-8 snapshot inspection must not recreate lifecycle state retired from the snapshot DTO"; +} + +/// ...and a `_log/` renders as a ref-transaction log. +TEST(CASObservability, CaInspectDecodesRefLogToJson) +{ + Layout layout("p"); + const RootNamespace ns{"srv/tbl@cas@"}; + const RefTxnId txn_id{1, 8}; + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = txn_id; + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "all_0_0_0", + ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 1}}; + txn.ops = {add}; + const String key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), txn_id); + const String json = caInspectToJson( + layout, key, encodeRefLogTxn(txn), DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find("ref_log"), String::npos); + EXPECT_NE(json.find("OwnerTransition"), String::npos); + EXPECT_NE(json.find("all_0_0_0"), String::npos); +} + +TEST(CASObservability, CaInspectDecodesPartManifestToJson) +{ + Layout layout("p"); + const RootNamespace ns{"srv/tbl@cas@"}; + + PartManifest m; + m.ref = ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 3}; + m.root_namespace_id = ns; + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "hello"; + m.entries = {e}; + m.payload_digest = computePayloadDigest(m); + + const ManifestId id{.root_namespace = ns, .ref = m.ref}; + const String key = layout.manifestKey(id); + const String json = caInspectToJson(layout, key, encodePartManifest(m)); + EXPECT_NE(json.find("\"root_namespace_id\""), String::npos); + EXPECT_NE(json.find("data.bin"), String::npos); + EXPECT_NE(json.find("\"manifest_ordinal\":3"), String::npos); +} + +TEST(CASObservability, CaInspectDecodesMountLeaseToJson) +{ + Layout layout("p"); + MountLease lease; + lease.server_uuid = hexToU128("000000000000000000000000000000ab"); + lease.writer_epoch = 5; + lease.write_attempt_id = hexToU128("00112233445566778899aabbccddeeff"); + lease.hostname = "host1"; + lease.pid = 123; + + const String key = layout.mountKey("srid1"); + const String json = caInspectToJson(layout, key, encodeMountLease(lease)); + EXPECT_NE(json.find("\"writer_epoch\":5"), String::npos); + EXPECT_NE(json.find("\"write_attempt_id\":\"00112233445566778899aabbccddeeff\""), String::npos); + EXPECT_NE(json.find("host1"), String::npos); +} + +TEST(CASObservability, CaInspectDecodesGcStateToJson) +{ + Layout layout("p"); + GcState state; + state.round = 42; + state.gc_shards = 4; + + const String key = layout.gcStateKey(); + const String json = caInspectToJson(layout, key, encodeGcState(state)); + EXPECT_NE(json.find("\"round\":42"), String::npos); + EXPECT_NE(json.find("\"gc_shards\":4"), String::npos); +} + +TEST(CASObservability, CaInspectUnknownKeyThrows) +{ + Layout layout("p"); + EXPECT_THROW(caInspectToJson(layout, "p/not/a/ca/object", "xxxx"), DB::Exception); /// BAD_ARGUMENTS +} diff --git a/src/Disks/tests/gtest_cas_operation_gate.cpp b/src/Disks/tests/gtest_cas_operation_gate.cpp new file mode 100644 index 000000000000..6d16c3946fae --- /dev/null +++ b/src/Disks/tests/gtest_cas_operation_gate.cpp @@ -0,0 +1,436 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +/// Task 8 (rev.7 spec §1): the central six-class operation gate (`checkOpAdmitted`), the `Vanished` truth +/// semantics, and the [D5] per-reason typed messages. These tests build a real +/// `ContentAddressedMetadataStorage` over a Local object storage (the same harness as +/// gtest_ca_transaction.cpp), commit a real part, then force the pool lifecycle condition directly via the +/// Task-5 setter (`Pool::setLifecycleForTest`) to pin each class × state cell of the spec §1 table and +/// assert what every public entry does. +/// +/// NOTE the harness idiom: `store()` itself is fail-closed on a terminal pool (it throws), so a test +/// captures the `PoolPtr` ONCE while the pool is still `Live` and drives `setLifecycleForTest` on that +/// captured handle -- the SAME object the metadata storage's `cas_store` points at -- rather than calling +/// `store()` again after forcing a terminal state. + +namespace DB::ErrorCodes +{ +extern const int INVALID_STATE; +extern const int NETWORK_ERROR; +extern const int FILE_DOESNT_EXIST; +} + +using namespace DB; +using DB::Cas::PoolLifecycle; + +namespace +{ + +/// A live table dir + part reused across the tests (the exact shape gtest_ca_transaction.cpp uses). +const std::string kTableDir = "g80/g80g80g8-0808-4808-8808-080808080808"; +const std::string kPartDir = kTableDir + "/all_1_1_0"; +const std::string kPartFile = kPartDir + "/data.bin"; + +std::shared_ptr openGateStorage() +{ + auto settings = Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_op_gate_scratch"); + auto storage = std::make_shared( + Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// Commit one real part (tmp -> final rename -> commit), leaving `kPartFile` durable and `kPartDir`/ +/// `kTableDir` non-empty. Every op below runs against this committed state. +void commitOnePart(ContentAddressedMetadataStorage & storage) +{ + auto tx = storage.createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + auto buf = ca_tx.writeFile(kTableDir + "/tmp_insert_all_1_1_0/data.bin", 65536, WriteMode::Rewrite, {}); + const std::string bytes = "content-of-the-part"; + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + tx->moveDirectory(kTableDir + "/tmp_insert_all_1_1_0", kPartDir); + tx->commit(NoCommitOptions{}); +} + +std::string messageOf(const std::function & fn) +{ + try + { + fn(); + } + catch (const Exception & e) + { + return std::string(e.message()); + } + ADD_FAILURE() << "expected a DB::Exception"; + return {}; +} + +/// The thrown exception itself, for the tests that assert against an upstream CLASSIFIER rather than +/// against an error code. NEVER returns a null `exception_ptr`: every consumer feeds the result to a +/// classifier that rethrows it, and `std::rethrow_exception(nullptr)` is undefined behaviour that takes +/// the whole binary down instead of failing one test. On the nothing-was-thrown path the failure is +/// recorded and a SENTINEL is returned -- the test has already failed by then, and the sentinel merely +/// keeps the assertion that follows harmless. +std::exception_ptr exceptionOf(const std::function & fn) +{ + try + { + fn(); + } + catch (...) + { + return std::current_exception(); + } + ADD_FAILURE() << "expected a DB::Exception, nothing was thrown"; + return std::make_exception_ptr(std::runtime_error("exceptionOf sentinel: nothing was thrown")); +} + +/// The Pool-level `server_root_id` a test mount uses (mirrors gtest_cas_lifecycle_condition.cpp). +const std::string kSrid = "test"; + +/// GC's fence-out applied to the mount lease (preserve the body, set `gc_fenced`, bump `seq`) so a +/// subsequent `tryRemountOnce` verdicts `Recover` and reclaims a FRESH incarnation immediately, driving a +/// transient-not-live pool back to `Live` without a lease-expiry wait. Mirrors +/// gtest_cas_lifecycle_condition.cpp's helper. +void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, + DB::Cas::PutOutcome::Done); +} + +} + +/// (a) Probes on a Vanished disk answer the truth: absent/empty, WITHOUT reaching the pool. +TEST(CASOperationGate, ProbesOnVanishedAnswerAbsentEmpty) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live + + /// Live baseline: the probes see the committed part. + ASSERT_TRUE(storage->existsFile(kPartFile)); + ASSERT_TRUE(storage->existsDirectory(kPartDir)); + ASSERT_TRUE(storage->existsFileOrDirectory(kPartFile)); + ASSERT_FALSE(storage->isDirectoryEmpty(kTableDir)); + ASSERT_FALSE(storage->listDirectory(kTableDir).empty()); + ASSERT_TRUE(storage->getStorageObjectsIfExist(kPartFile).has_value()); + + pool->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + + EXPECT_FALSE(storage->existsFile(kPartFile)); + EXPECT_FALSE(storage->existsDirectory(kPartDir)); + EXPECT_FALSE(storage->existsFileOrDirectory(kPartFile)); + EXPECT_TRUE(storage->listDirectory(kTableDir).empty()); + EXPECT_FALSE(storage->iterateDirectory(kTableDir)->isValid()); + EXPECT_TRUE(storage->isDirectoryEmpty(kTableDir)); + EXPECT_FALSE(storage->getStorageObjectsIfExist(kPartFile).has_value()); + /// The offender `liveTreeDirHasChildren` hardcoded-true is now truthful too: the disk root reads absent. + EXPECT_FALSE(storage->liveTreeDirHasChildren("")); +} + +/// (b) Removes on a Vanished disk are no-op SUCCESS and never touch the backend: after restoring Live the +/// part is still there. This is what lets a vanished-disk table's DROP complete. +TEST(CASOperationGate, RemovesOnVanishedAreNoOpSuccessBackendUntouched) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live + ASSERT_TRUE(storage->existsDirectory(kPartDir)); + + pool->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + + /// A whole-table removeRecursive + commit (the DROP shape): both no-op-succeed. + { + auto tx = storage->createTransaction(); + EXPECT_NO_THROW(tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr)); + EXPECT_NO_THROW(tx->commit(NoCommitOptions{})); /// empty parts -> Remove -> no-op success + } + /// A single removeDirectory of the part dir + commit: no-op-succeed. + { + auto tx = storage->createTransaction(); + EXPECT_NO_THROW(tx->removeDirectory(kPartDir)); + EXPECT_NO_THROW(tx->commit(NoCommitOptions{})); + } + + /// Truth check: nothing was actually removed. Back on Live the part is intact. + pool->setLifecycleForTest(PoolLifecycle::Live); + EXPECT_TRUE(storage->existsDirectory(kPartDir)) << "a remove on a Vanished disk must not touch the backend"; + EXPECT_TRUE(storage->existsFile(kPartFile)); +} + +/// (c) A content read on a Vanished disk throws the typed per-reason [D5] message -- the exact substring +/// names the ACTUAL sub-state (replaced / forgotten), never a wrong diagnosis. +TEST(CASOperationGate, ContentReadOnVanishedThrowsTypedPerReasonMessage) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live + + pool->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + EXPECT_NE(messageOf([&] { storage->getFileSize(kPartFile); }).find("foreign pool"), std::string::npos); + EXPECT_NE(messageOf([&] { storage->getStorageObjects(kPartFile); }).find("foreign pool"), std::string::npos); + + pool->setLifecycleForTest(PoolLifecycle::VanishedForgotten); + EXPECT_NE(messageOf([&] { storage->getFileSize(kPartFile); }).find("erasure was NOT verified"), + std::string::npos); +} + +/// (d1) Every class but Factory refuses on `TransientNotLive` — and the refusal carries the TRANSIENT +/// class (`NETWORK_ERROR`), not the terminal 668. The split from `IdentityLost` (test d2) is the whole +/// point: a lease blip is unavailability, an identity loss is damage, and consumers outside CAS act on +/// the difference. `ReplicatedMergeTreePartCheckThread` declares a part broken and detaches it for any +/// refusal its `isRetryableException` hatch does not recognise, so coding a blip 668 made healthy parts +/// look corrupt (BACKLOG {#lease-blip-part-check-collapse}). +TEST(CASOperationGate, EveryClassThrowsRetryableTransientOnTransientNotLive) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + storage->store()->setLifecycleForTest(PoolLifecycle::TransientNotLive); /// one force from Live + + /// Probe + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->existsFile(kPartFile); }); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->existsDirectory(kPartDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->listDirectory(kTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->isDirectoryEmpty(kTableDir); }); + /// ContentRead + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->getFileSize(kPartFile); }); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->getStorageObjects(kPartFile); }); + /// Write (via a transaction) + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + ca_tx.writeFile(kTableDir + "/tmp_x/data.bin", 65536, WriteMode::Rewrite, {}); + }); + /// Remove (via a transaction) + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { + auto tx = storage->createTransaction(); + tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr); + }); + /// Admin + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->runOneGcRoundForTest(); }); + + /// The coarser code buys retryability at the cost of precision, so the MESSAGE carries the whole + /// truth: which CA condition, and that it is transient rather than an error of record. + const std::string msg = messageOf([&] { storage->getFileSize(kPartFile); }); + EXPECT_NE(msg.find("mount lease not held"), std::string::npos) << msg; + EXPECT_NE(msg.find("TRANSIENT"), std::string::npos) << msg; + EXPECT_NE(msg.find("recovers to Live"), std::string::npos) << msg; +} + +/// (d2) `IdentityLost` is TERMINAL — the sentinels are gone, nothing auto-recovers — so it keeps the 668 +/// (`INVALID_STATE`) class and its own richer [D5] diagnosis ("identity lost … restart or FORGET"). +/// Nothing about the transient re-coding may leak here: a terminal state that read as retryable would +/// make every consumer spin forever on a disk that will never come back. +TEST(CASOperationGate, EveryClassThrows668OnIdentityLost) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + storage->store()->setLifecycleForTest(PoolLifecycle::IdentityLost); /// one force from Live + + /// Probe + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->existsFile(kPartFile); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->existsDirectory(kPartDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->listDirectory(kTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->isDirectoryEmpty(kTableDir); }); + /// ContentRead + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->getFileSize(kPartFile); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->getStorageObjects(kPartFile); }); + /// Write (via a transaction) + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { + auto tx = storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + ca_tx.writeFile(kTableDir + "/tmp_x/data.bin", 65536, WriteMode::Rewrite, {}); + }); + /// Remove (via a transaction) + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { + auto tx = storage->createTransaction(); + tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr); + }); + /// Admin + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->runOneGcRoundForTest(); }); + + EXPECT_NE(messageOf([&] { storage->getFileSize(kPartFile); }).find("identity lost"), std::string::npos); +} + +/// (d3) The contract the d1/d2 split exists to satisfy, asserted against upstream's OWN predicate instead +/// of a code number: `isRetryableException` is what `ReplicatedMergeTreePartCheckThread::checkPartImpl` +/// consults before declaring a part broken. A transient CA refusal must satisfy it (the part stays +/// queued); a terminal one must not (the disk is genuinely unusable and must surface). Pinning the +/// predicate rather than `NETWORK_ERROR` keeps this test meaningful if upstream's list ever moves. +TEST(CASOperationGate, TransientRefusalIsUpstreamRetryableTerminalIsNot) +{ + { + auto storage = openGateStorage(); + commitOnePart(*storage); + storage->store()->setLifecycleForTest(PoolLifecycle::TransientNotLive); + EXPECT_TRUE(isRetryableException(exceptionOf([&] { storage->getFileSize(kPartFile); }))) + << "a lease blip must not read as part damage to the part-check thread"; + } + { + auto storage = openGateStorage(); + commitOnePart(*storage); + storage->store()->setLifecycleForTest(PoolLifecycle::IdentityLost); + /// Pin WHICH error is being classified before classifying it: `EXPECT_FALSE` alone passes for any + /// non-retryable error, so a future regression that threw something else entirely here -- or threw + /// from the wrong site -- would slip through as a pass. + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->getFileSize(kPartFile); }); + EXPECT_NE(messageOf([&] { storage->getFileSize(kPartFile); }).find("identity lost"), std::string::npos); + EXPECT_FALSE(isRetryableException(exceptionOf([&] { storage->getFileSize(kPartFile); }))) + << "a terminal identity loss must NOT be retried forever as if it were transient"; + } +} + +/// (e) `createTransaction` (Factory: I/O-free) and the capability/introspection getters construct fine on +/// a Vanished disk -- so a vanished-disk table's DROP can allocate its removal transaction. +TEST(CASOperationGate, FactoryClassWorksOnVanished) +{ + auto storage = openGateStorage(); + storage->store()->setLifecycleForTest(PoolLifecycle::VanishedForgotten); /// one force from Live + + EXPECT_NO_THROW({ auto tx = storage->createTransaction(); (void)tx; }); + EXPECT_EQ(storage->getType(), MetadataStorageType::CAS); + EXPECT_NO_THROW((void)storage->getPath()); + EXPECT_NO_THROW((void)storage->isContentAddressed()); +} + +/// (f) `tryGetInManifestBytes` PROPAGATES the typed refusal — terminal 668 on a `Vanished` disk, the +/// transient class in a lease gap — rather than converting either into a silent-absent `std::nullopt` +/// (the narrowed catch). RED before the narrowing. +TEST(CASOperationGate, TryGetInManifestBytesPropagatesTypedError) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live + + pool->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + /// Never FILE_DOESNT_EXIST, never a swallowed nullopt -- the typed INVALID_STATE escapes. + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { + storage->tryGetInManifestBytes(kTableDir + "/format_version.txt"); + }); + + pool->setLifecycleForTest(PoolLifecycle::TransientNotLive); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { + storage->tryGetInManifestBytes(kTableDir + "/format_version.txt"); + }); +} + +/// (g) (rev.8, Task 15) Null-pool fail-loud: the Dormant/UNMOUNT rollback replaced the transitional +/// not-Mounted branch (which answered `Probe`->benign-absent) with a null-pool fail-loud. A storage whose +/// pool is torn down (`shutdown()`) refuses EVERY class, `Probe` included, with `INVALID_STATE` +/// ("not started") -- there is no benign-absent answer for a not-started disk; only a genuinely `Vanished` +/// POOL answers truth-absent. Replaces the deleted `DormantDiskKeepsOldBenignAbsent_RemoveAtTask15`. +TEST(CASOperationGate, NullPoolFailsLoudForEveryClass) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + ASSERT_TRUE(storage->existsDirectory(kPartDir)); + + storage->shutdown(); /// null pool -- the ShutDown storage lifecycle + + /// Probes now THROW (not started), NOT the transitional benign-absent answer. + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->existsFile(kPartFile); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->existsDirectory(kPartDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { (void)storage->listDirectory(kTableDir); }); + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->isDirectoryEmpty(kTableDir); }); + /// Store-class ops throw the same INVALID_STATE ("not started"), not the typed Vanished message. + Cas::tests::expectThrowsCode(ErrorCodes::INVALID_STATE, [&] { storage->getFileSize(kPartFile); }); +} + +/// (h) The raw GC round entry points refuse on a not-live pool (Admin class): typed [D5] reason once Vanished. +TEST(CASOperationGate, GcEntryPointsRefuseOnNotLive) +{ + auto storage = openGateStorage(); + auto pool = storage->store(); /// captured while Live + + pool->setLifecycleForTest(PoolLifecycle::VanishedReplaced); + EXPECT_NE(messageOf([&] { storage->runOneGcRoundForTest(); }).find("foreign pool"), std::string::npos); + + pool->setLifecycleForTest(PoolLifecycle::TransientNotLive); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->runOneGcRoundForTest(); }); +} + +/// (i) `CasGcScheduler::isQuiescent` reflects the round-in-flight flag: a round in flight => not quiescent. +/// (This is the join-completion signal the FORGET / GC-STOP tests rely on.) +TEST(CASOperationGate, GcSchedulerIsQuiescentReflectsRoundInFlight) +{ + auto backend = std::make_shared(); + auto pool = Cas::tests::openPoolForTest(backend); + auto scheduler = std::make_shared( + pool, std::chrono::seconds(3600), "op-gate-test-gc", "disk", Cas::GcRoundLogger{}); + EXPECT_TRUE(scheduler->isQuiescent()); + scheduler->setRoundInFlightForTest(true); + EXPECT_FALSE(scheduler->isQuiescent()) << "a round in flight must NOT read as GC-quiescent"; + scheduler->setRoundInFlightForTest(false); + EXPECT_TRUE(scheduler->isQuiescent()); +} + +/// (j) (acceptance matrix — transient auto-recovery / DROP-drain round-trip) The full §4 recovery arc on ONE +/// storage: a Remove-class op (the DROP shape) throws the typed transient refusal while the mount lease is +/// lost, then SUCCEEDS and actually drains once the disk self-remounts back to Live — no operator action, +/// no restart. Where test (d) forces `TransientNotLive` via the setter to pin the gap, this drives a REAL +/// transient→Live recovery (`tripMountLost` → fence-out → `tryRemountOnce`) so the throw-then-drain is one +/// continuous arc on the same pool. Closes the "access throws in the gap, auto-recovers, a Remove re-queues +/// and drains" matrix row end-to-end (the per-table DROP re-queue itself is the MergeTree caller's job; the +/// CAS contract is exactly this: refuse in the gap, admit after recovery). +TEST(CASOperationGate, RemoveThrowsDuringTransientAndDrainsAfterRecovery) +{ + auto storage = openGateStorage(); + commitOnePart(*storage); + auto pool = storage->store(); /// captured while Live (store() is fail-closed once not-live) + ASSERT_EQ(pool->lifecycle(), PoolLifecycle::Live); + ASSERT_TRUE(storage->existsDirectory(kPartDir)); /// Live baseline: the part is present. + + /// The mount lease is transiently lost — the pool goes TransientNotLive. + pool->tripMountLost(); + ASSERT_EQ(pool->lifecycle(), PoolLifecycle::TransientNotLive); + + /// In the gap, EVERY store-class access throws the typed transient refusal — the Remove (DROP shape) + /// included, and a content read too. Nothing is answered benign, nothing is silently dropped. + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { + auto tx = storage->createTransaction(); + tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr); + }); + Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->getFileSize(kPartFile); }); + const std::string gap_msg = messageOf([&] { storage->getFileSize(kPartFile); }); + EXPECT_NE(gap_msg.find("mount lease not held"), std::string::npos) + << "the gap message must name the transient (auto-recovering) condition: " << gap_msg; + + /// The lease is restored: the disk self-remounts a fresh incarnation and auto-recovers to Live. + fenceOutMount(pool->backend(), pool->layout().mountKey(kSrid)); + ASSERT_TRUE(pool->tryRemountOnce()) << "the self-remount must reclaim a fresh incarnation"; + ASSERT_EQ(pool->lifecycle(), PoolLifecycle::Live) << "the pool must auto-recover to Live"; + + /// After recovery the SAME Remove drains: it commits cleanly and actually removes the part. + { + auto tx = storage->createTransaction(); + EXPECT_NO_THROW(tx->removeRecursive(kTableDir, /*should_remove_objects=*/nullptr)); + EXPECT_NO_THROW(tx->commit(NoCommitOptions{})); + } + EXPECT_FALSE(storage->existsDirectory(kPartDir)) + << "the re-queued removal must drain (really remove the part) once the disk recovers to Live"; + EXPECT_FALSE(storage->existsFile(kPartFile)); +} diff --git a/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp b/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp new file mode 100644 index 000000000000..45f82c1a9e33 --- /dev/null +++ b/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp @@ -0,0 +1,678 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_sweep_test_support.h" +#include "cas_test_helpers.h" +#include +#include +#include +#include +#include + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ +constexpr uint64_t kWriterEpoch = 7; +const String kServerRoot = "00"; +ManifestRef ref(uint64_t seq, uint64_t inst) +{ + return ManifestRef{.writer_epoch = kWriterEpoch, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; +} + +/// The §6 deletion premise (`manifestDeletionPremise`) is a SECOND precondition on every deletion below, +/// alongside the watermark eligibility these tests are about: a manifest of an epoch-`E` build is +/// deletable only once the namespace's sealed fold cursor sits in an epoch strictly above `E`. Tests +/// whose subject is the eligibility or ownership rule therefore have to establish it, or they would +/// assert a deletion the premise (not the rule under test) prevented. Tests whose subject is RETENTION +/// deliberately do NOT call this — see `CASSweepDeletionPremise` for the premise's own coverage. +void seedConsumedSealCursor(InMemoryBackend & backend, const Layout & layout, const RootNamespace & ns) +{ + seedFoldCursorForTest(backend, layout, ns, RefTxnId{kWriterEpoch + 1, 1}); +} + +/// The catalog cut and `_ckpt` are recovery's sole authority. A fixture that expects a catalog-named +/// life to be swept must establish the same empty, fully readable recovery state a real completed +/// creation would have, rather than relying on the retired sentinel fallback. +void seedEmptyRecoveryAuthority(InMemoryBackend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + const auto entry = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), + [&](const CatalogEntry & candidate) { return candidate.ns == ns; }); + ASSERT_NE(entry, catalog.catalog.entries.end()); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = std::optional{kWriterEpoch}, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); +} + +/// Replaces the catalog row immediately before its second read after arming. The legacy orphan path +/// reads a catalog cut for coverage, then resolves the name again inside recovery; the second read can +/// splice a successor life into the old coverage decision. An authority-threaded path has no second +/// catalog read, so this seam must remain dormant. +class CatalogChangingOnSecondReadBackend : public InMemoryBackend +{ +public: + using Backend::get; + + void arm(const Layout & layout, CatalogEntry predecessor_, CatalogEntry successor_) + { + catalog_key = layout.refCatalogKey(); + predecessor = std::move(predecessor_); + successor = std::move(successor_); + catalog_reads = 0; + armed = true; + } + + bool didSwitch() const { return did_switch; } + + std::optional get(const String & key, Range range) override + { + if (armed && key == catalog_key && ++catalog_reads == 2) + { + const auto current = InMemoryBackend::get(key, range); + if (!current) + throw std::runtime_error("test catalog disappeared"); + RefCatalog next = decodeRefCatalog(current->bytes); + const auto it = std::find(next.entries.begin(), next.entries.end(), predecessor); + if (it == next.entries.end()) + throw std::runtime_error("test predecessor catalog row disappeared"); + *it = successor; + if (InMemoryBackend::casPut(key, encodeRefCatalog(next), current->token).outcome != CasOutcome::Committed) + throw std::runtime_error("test catalog replacement conflicted"); + did_switch = true; + } + return InMemoryBackend::get(key, range); + } + +private: + String catalog_key; + CatalogEntry predecessor; + CatalogEntry successor; + uint64_t catalog_reads = 0; + bool armed = false; + bool did_switch = false; +}; + +/// Rewrites a listed manifest after the page captured it but before the page takes its lifecycle cut. +/// The old implementation performed its candidate GET after that cut and would delete this replacement +/// with its new token. The fixed path may nominate the old observation, but exact-token deletion loses. +class ReplacingManifestAfterObservationBackend : public InMemoryBackend +{ +public: + using Backend::get; + + void arm(const Layout & layout, String manifest_key_) + { + catalog_key = layout.refCatalogKey(); + manifests_prefix = layout.casManifestsPrefix(); + manifest_key = std::move(manifest_key_); + listed_page = false; + replaced_manifest = false; + armed = true; + } + + bool didReplace() const { return replaced_manifest; } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + const ListPage page = InMemoryBackend::list(prefix, cursor, limit); + if (armed && prefix == manifests_prefix) + listed_page = true; + return page; + } + + std::optional get(const String & key, Range range) override + { + const auto result = InMemoryBackend::get(key, range); + if (armed && listed_page && !replaced_manifest && key == catalog_key) + { + const auto current = InMemoryBackend::get(manifest_key); + if (!current) + throw std::runtime_error("test manifest disappeared before replacement"); + if (InMemoryBackend::casPut(manifest_key, current->bytes, current->token).outcome != CasOutcome::Committed) + throw std::runtime_error("test manifest replacement conflicted"); + replaced_manifest = true; + } + return result; + } + +private: + String catalog_key; + String manifests_prefix; + String manifest_key; + bool armed = false; + bool listed_page = false; + bool replaced_manifest = false; +}; +} + +/// A staged-but-unowned body in an ELIGIBLE prefix, absent from the owner view, is deleted (#7). +TEST(CASOrphanManifestSweep, EligibleAndUnownedIsDeleted) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); // body, no owner + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active*/6); // 6 > 5 => eligible + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// The orphan sweep must not turn a forged same-id snapshot at an OLDER `EpochSeal` into an empty owner +/// view. The base differs from `last_epoch_seal`, so metadata equality cannot catch it; the candidate +/// remains retained until the checkpoint is repaired. +TEST(CASOrphanManifestSweep, CheckpointSnapshotAtOlderEpochSealSkipsDeletion) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/sweep-checkpoint-base-seal@cas@"}; + fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + + const RefLogTxn birth{ + .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, birth); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + const RefLogTxn seal_txn{ + .ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = {seal}, + .prev_epoch_seal = std::nullopt}; + fixture::writeRefLogRaw(*backend, layout, seal_txn); + RefOp later_seal; + later_seal.kind = RefOpKind::EpochSeal; + fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), .txn_id = RefTxnId{2, 1}, .ops = {later_seal}, + .prev_epoch_seal = RefTxnId{1, 2}}); + RefTableState through_seal; + applyRefLogTxn(through_seal, birth); + applyRefLogTxn(through_seal, seal_txn); + writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + + const ManifestRef candidate = ref(5, 0xAC); + const String candidate_key = layout.manifestKey(ManifestId{ns, candidate}); + writeManifestRaw(*backend, layout, ns, candidate, {blobEntryFor("a", DB::UInt128(1))}); + setWatermarkMinActive(*backend, layout, kServerRoot, kWriterEpoch, /*min_active=*/6); + seedConsumedSealCursor(*backend, layout, ns); + + std::vector warnings; + EXPECT_EQ(sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}, &warnings), 0u); + EXPECT_TRUE(backend->head(candidate_key).exists); + ASSERT_FALSE(warnings.empty()); +} + +/// A body that IS in the owner view (committed) is NEVER swept (#8). +TEST(CASOrphanManifestSweep, OwnedBodyIsSkipped) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); // now owned + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// GC-WEDGE regression (2026-07-10): a COMMITTED ref that has been DROPPED but whose removal `-1` is NOT +/// yet sealed (transition_version above the sealed fold cursor, which is 0 for this fresh pool) must +/// SURVIVE the sweep — the GC fold still needs the body to emit the `-1` (delete-after-sealed-decrements). +/// A promoted build retires its build_seq, so the prefix is watermark-eligible; before the fix the sweep +/// deleted the body in the dropRef→fold window → the removal-fold then clamped FOREVER on the missing +/// committed body → pool-wide GC stop. The pending-removal protection now covers COMMITTED (not only +/// PRECOMMIT) removals. +/// +/// SINCE THE §6 PREMISE, this shape is held by TWO independent facts: the tail-removal protection this +/// test is named for, and the premise's rule (1) — the fixture seals no fold cursor, so epoch +/// `kWriterEpoch`'s closing seal is not consumed either. They cannot be separated HERE: the removal log +/// sits in a lower epoch than the build, so any cursor high enough to satisfy rule (1) would also sit +/// above the log and stop the tail scan from reading it at all. The case where the tail-removal +/// protection is the ONLY thing standing — a removal in a LATER epoch, which is the direction removals +/// actually cross — is `CASSweepDeletionPremise.AnUnconsumedTailRemovalRetainsItsTarget`. +TEST(CASOrphanManifestSweep, PendingCommittedRemovalBodyIsSkipped) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); // committed owner + dropRefTransition(*backend, store->layout(), ns, "tbl", r); // dropped: pending committed removal, -1 unsealed + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); // 6 > 5 => prefix eligible + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + << "a dropped-but-unsealed committed manifest body must survive the sweep (delete-after-sealed-" + "decrements) — else the removal-fold clamps forever on the missing body (GC-WEDGE-2026-07-10)"; +} + +/// The sweep emits NO blob deltas: the in-degree generation is unchanged. +TEST(CASOrphanManifestSweep, EmitsNoBlobDeltas) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + // The sweep must not advance the in-degree generation: capture it AFTER the fixture's own seal. + const uint64_t gen_before = currentGenerationOf(*backend, store->layout()); + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + EXPECT_EQ(currentGenerationOf(*backend, store->layout()), gen_before); + EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 0); +} + +TEST(CASOrphanManifestSweep, CursorPageAdvancesAndWrapsWithListBudget) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r1 = ref(5, 0xE1); + const ManifestRef r2 = ref(5, 0xE2); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active*/6); + + const ManifestSweepResult first = sweepManifestCursorPageForTest(*store, "", /*list_budget*/1, /*delete_budget*/0); + EXPECT_EQ(first.listed, 1u); + EXPECT_FALSE(first.wrapped); + EXPECT_FALSE(first.next_cursor.empty()); + + const ManifestSweepResult second = sweepManifestCursorPageForTest(*store, first.next_cursor, /*list_budget*/100, /*delete_budget*/0); + EXPECT_GE(second.listed, 1u); + EXPECT_TRUE(second.wrapped); + EXPECT_TRUE(second.next_cursor.empty()); +} + +/// A NON-eligible prefix (no watermark fact) deletes NOTHING (#9: frozen-seq is not authority). +TEST(CASOrphanManifestSweep, NoWatermarkIsNotAuthority) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(5, 0xAB); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + // No setWatermarkMinActive — no durable fact => not eligible. + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +TEST(CASOrphanManifestSweep, CursorPageDeletesEligibleUnownedBody) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r = ref(5, 0xAC); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); + EXPECT_GE(result.listed, 1u); + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +TEST(CASOrphanManifestSweep, CursorPageRespectsDeleteBudget) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r1 = ref(5, 0xAD); + const ManifestRef r2 = ref(5, 0xAE); + writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/1); + EXPECT_EQ(result.deleted, 1u); + const bool first_exists = backend->head(store->layout().manifestKey(ManifestId{ns, r1})).exists; + const bool second_exists = backend->head(store->layout().manifestKey(ManifestId{ns, r2})).exists; + EXPECT_NE(first_exists, second_exists); +} + +/// A physical manifest captured before a catalog cut which omits its name is dead-life debris: a live +/// creation cannot publish a life-owned object before its catalog row. It therefore has an eventual +/// page-sweep owner without trying to reconstruct a deleted incarnation from the key. +TEST(CASOrphanManifestSweep, CursorPageDeletesObservedBodyWhenCatalogOmitsNamespace) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/catalog-absent-debris@cas@"}; + const ManifestRef r = ref(5, 0xA9); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("debris", DB::UInt128(9))}); + /// No mount lease/watermark exists: after legal catalog-row deletion there may be no server-root + /// state left to supply one. The post-observation absent row is the complete dead-life proof. + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); + + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// The candidate body and token must be frozen before the later catalog cut. A concurrent same-key +/// replacement after that observation is a new physical incarnation and must lose the old-token delete. +TEST(CASOrphanManifestSweep, CursorPageCannotDeleteManifestReplacedAfterObservation) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/replace-after-observation@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r = ref(5, 0xAA); + const String key = store->layout().manifestKey(ManifestId{ns, r}); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("body", DB::UInt128(10))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + backend->arm(store->layout(), key); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); + + EXPECT_TRUE(backend->didReplace()); + EXPECT_EQ(result.deleted, 0u); + EXPECT_TRUE(backend->head(key).exists); +} + +/// Any duplicate current life id makes the catalog-to-physical join ambiguous. The cursor page is +/// destructive, so the whole cut must be rejected before it can nominate even an unrelated body. +TEST(CASOrphanManifestSweep, CursorPageRefusesAmbiguousCatalogLifeIndex) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/ambiguous-life@cas@"}; + registerNamespaceRaw(*backend, store->layout(), ns); + const ManifestRef r = ref(5, 0xAB); + const String key = store->layout().manifestKey(ManifestId{ns, r}); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("body", DB::UInt128(11))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + seedEmptyRecoveryAuthority(*backend, store->layout(), ns); + + const CasRefCatalog::Snapshot before = CasRefCatalog::read(*backend, store->layout()); + RefCatalog damaged = before.catalog; + CatalogEntry duplicate = damaged.entries.front(); + duplicate.ns = RootNamespace{"00/ambiguous-life-twin@cas@"}; + damaged.entries.push_back(duplicate); + std::sort(damaged.entries.begin(), damaged.entries.end(), + [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); + ASSERT_EQ(backend->casPut(store->layout().refCatalogKey(), encodeRefCatalog(damaged), before.token).outcome, + CasOutcome::Committed); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); }); + EXPECT_TRUE(backend->head(key).exists); +} + +TEST(CASOrphanManifestSweep, CursorPageSkipsOwnedBody) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(5, 0xAF); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); + EXPECT_EQ(result.deleted, 0u); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); +} + +/// A catalog-named life cannot be treated as an empty table merely because its mandatory recovery +/// checkpoint is missing. The orphan sweep is destructive, so it must retain the body until the +/// caller can recover from the same frozen catalog row and its exact `_ckpt`. +TEST(CASOrphanManifestSweep, MissingRequiredCheckpointSuppressesDestructiveDecision) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/authority-required@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + const ManifestRef r = ref(5, 0xB0); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + + const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + ASSERT_FALSE(readCkpt(*backend, store->layout(), NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation))); + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + << "without the exact _ckpt required by a Live catalog row, the sweep must retain rather than " + "derive an empty owner set"; +} + +/// A decoded fold cursor at an `EpochSeal` advances to the next GLOBAL writer epoch. Even when this +/// namespace was inactive, every intermediate epoch exists as a chained sequence-1 empty seal, so the +/// exact tail begins at `{E+1, 1}` and must consume each one before reaching a later removal. The +/// removal's target stays protected while an unrelated eligible body remains deletable; retaining both +/// would hide a false missing-log failure at the first intermediate seal. +TEST(CASOrphanManifestSweep, EpochSealFoldCursorCrossesTailByExactDecodedSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/seal-cursor-tail@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + + const ManifestRef removed{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; + const ManifestRef unowned{.writer_epoch = 1, .build_sequence = 6, .manifest_ordinal = 1}; + const ManifestRef still_owned{.writer_epoch = 2, .build_sequence = 1, .manifest_ordinal = 1}; + publishAt(*backend, store->layout(), ns, RefTxnId{1, 1}, "dropped", removed.build_sequence, DB::UInt128(0xA1), /*birth=*/true); + writeSealAt(*backend, store->layout(), ns, RefTxnId{1, 2}); + writeTxnAt(*backend, store->layout(), ns, RefTxnId{2, 1}, publishCommittedOps("still-owned", still_owned), RefTxnId{1, 2}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{2, 2}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{3, 1}, RefTxnId{2, 2}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{4, 1}, RefTxnId{3, 1}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{5, 1}, RefTxnId{4, 1}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{6, 1}, RefTxnId{5, 1}); + for (uint64_t epoch = 3; epoch <= 6; ++epoch) + ASSERT_TRUE(backend->head(store->layout().refLogKey(life, RefTxnId{epoch, 1})).exists) + << "fixture must deposit every intermediate exact successor in the catalog life"; + writeTxnAt(*backend, store->layout(), ns, RefTxnId{7, 1}, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "dropped", removed}, std::nullopt)}, + RefTxnId{6, 1}); + writeSealAt(*backend, store->layout(), ns, RefTxnId{7, 2}); + writeManifestRaw(*backend, store->layout(), ns, unowned, {blobEntryFor("unowned", DB::UInt128(0xA2))}); + ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{7, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{7, 2}})).outcome, PutOutcome::Done); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 7); + seedFoldCursorForTest(*backend, store->layout(), ns, RefTxnId{2, 2}); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); + + EXPECT_EQ(result.deleted, 1u); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, removed})).exists) + << "the exact successor of the folded epoch seal contains this body's unconsumed -1"; + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, unowned})).exists) + << "an unrelated eligible body must still drain; retaining it would mask a geometry failure"; +} + +/// A cleaned inherited cursor does not let a later epoch backlink skip the mandatory immediately-next +/// global epoch. `{3,1}` is missing here, so the direct `{7,1} -> {2,2}` link cannot authorize a tail +/// scan; the whole namespace must fail closed and retain even an otherwise unowned eligible body. +TEST(CASOrphanManifestSweep, MissingImmediateEpochAfterCleanedCursorCannotBeSkippedByLaterBacklink) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/missing-next-epoch@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + + const RefTxnId cursor{2, 2}; + writeSealAt(*backend, store->layout(), ns, cursor); + const HeadResult cursor_head = backend->head(store->layout().refLogKey(life, cursor)); + ASSERT_TRUE(cursor_head.exists); + ASSERT_EQ(classifyDeleteOutcome( + backend->deleteExact(store->layout().refLogKey(life, cursor), cursor_head.token)), DeleteClass::Deleted); + + const ManifestRef phantom{.writer_epoch = 7, .build_sequence = 1, .manifest_ordinal = 1}; + /// The codec refuses this skipped predecessor when a writer tries to create it. Inject the malformed + /// physical lure explicitly: a reader must still not enumerate forward to it when `{3,1}` is absent. + RefLogTxn direct_later_link{ + .ns = ns.string(), + .txn_id = RefTxnId{7, 1}, + .ops = publishCommittedOps("phantom", phantom), + .prev_epoch_seal = RefTxnId{6, 1}}; + String malformed_later_link = encodeRefLogTxn(direct_later_link); + const String encoded_predecessor{R"("!pse":"6")"}; + const size_t predecessor_pos = malformed_later_link.find(encoded_predecessor); + ASSERT_NE(predecessor_pos, String::npos); + malformed_later_link.replace( + predecessor_pos, encoded_predecessor.size(), R"("!pse":"2")"); + ASSERT_EQ(backend->putIfAbsent( + store->layout().refLogKey(life, RefTxnId{7, 1}), sealObject(FormatId::RefLog, malformed_later_link)).outcome, + PutOutcome::Done); + writeRefSnapshotRaw(*backend, store->layout(), RefTableSnapshot{ + .ns = ns.string(), + .snapshot_id = RefTxnId{7, 1}, + .committed = {committedRow("phantom", phantom)}, + .precommits = {}}); + ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{7, 1}, + .checkpoint_snapshot_id = RefTxnId{7, 1}, + .last_epoch_seal = RefTxnId{6, 1}})).outcome, PutOutcome::Done); + + const ManifestRef victim{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; + writeManifestRaw(*backend, store->layout(), ns, victim, {blobEntryFor("victim", DB::UInt128(0xC1))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedFoldCursorForTest(*backend, store->layout(), ns, cursor); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, victim})).exists); +} + +/// Control for the cleaned-cursor path: the exact immediately-next epoch head exists and names the +/// deleted seal, so the tail is readable. Its `-1` protects the removed body while an unrelated eligible +/// body proves the namespace was scanned rather than retained wholesale. The checkpoint base is the +/// following same-epoch transaction: recovery therefore has a retained exact anchor without turning the +/// deliberately cleaned predecessor seal into part of that anchor's proof. +TEST(CASOrphanManifestSweep, CleanedCursorCrossesOnlyThroughExactImmediateEpochHead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/exact-next-epoch@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + + const RefTxnId cursor{2, 2}; + writeSealAt(*backend, store->layout(), ns, cursor); + const HeadResult cursor_head = backend->head(store->layout().refLogKey(life, cursor)); + ASSERT_TRUE(cursor_head.exists); + ASSERT_EQ(classifyDeleteOutcome( + backend->deleteExact(store->layout().refLogKey(life, cursor), cursor_head.token)), DeleteClass::Deleted); + + const ManifestRef removed{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; + writeTxnAt(*backend, store->layout(), ns, RefTxnId{3, 1}, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "removed", removed}, std::nullopt)}, cursor); + const ManifestRef absent_anchor{.writer_epoch = 1, .build_sequence = 7, .manifest_ordinal = 1}; + writeTxnAt(*backend, store->layout(), ns, RefTxnId{3, 2}, + {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "absent-anchor", absent_anchor}, std::nullopt)}); + writeRefSnapshotRaw(*backend, store->layout(), RefTableSnapshot{ + .ns = ns.string(), .snapshot_id = RefTxnId{3, 2}, .committed = {}, .precommits = {}}); + ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{3, 2}, + .checkpoint_snapshot_id = RefTxnId{3, 2}, + .last_epoch_seal = cursor})).outcome, PutOutcome::Done); + + const ManifestRef unowned{.writer_epoch = 1, .build_sequence = 6, .manifest_ordinal = 1}; + writeManifestRaw(*backend, store->layout(), ns, removed, {blobEntryFor("removed", DB::UInt128(0xC2))}); + writeManifestRaw(*backend, store->layout(), ns, unowned, {blobEntryFor("unowned", DB::UInt128(0xC3))}); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 7); + seedFoldCursorForTest(*backend, store->layout(), ns, cursor); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); + + EXPECT_EQ(result.deleted, 1u); + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, removed})).exists); + EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, unowned})).exists); +} + +/// The catalog row used to obtain coverage and the life used to recover ownership must be ONE frozen +/// authority cut. A later catalog row for the same name may not make the old life's committed manifest +/// look orphaned and therefore deletable. +TEST(CASOrphanManifestSweep, LaterCatalogCutCannotSpliceOwnershipAuthority) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/frozen-catalog-cut@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + const CatalogEntry predecessor = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + + const ManifestRef r = ref(5, 0xB1); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + ASSERT_EQ(publishCommittedTransition(*backend, store->layout(), ns, "live", std::nullopt, r), 1u); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); + seedConsumedSealCursor(*backend, store->layout(), ns); + + CatalogEntry successor = predecessor; + successor.incarnation = DB::UInt128(0xBEEF); + backend->arm(store->layout(), predecessor, successor); + + sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); + + EXPECT_FALSE(backend->didSwitch()) + << "the sweep must not resolve a second catalog cut after it starts using the frozen entry"; + EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + << "the committed predecessor manifest must remain protected by the same frozen authority cut"; +} + +/// The LIST-based late-log detector that lived here is RETIRED with the sentinel seal, and it is worth +/// recording why rather than leaving a hole in this file's coverage story. +/// +/// It existed because the old seal was a SNAPSHOT at a synthetic `{E-1, UINT64_MAX}` id: that object +/// occupied no `_log` key, so a dying predecessor's in-flight PUT could still land in the dead epoch and +/// the only possible response was to notice it afterwards and report it. INV-2's seal is a +/// TRANSACTION at exactly `{E, T+1}` -- the key that ghost would take -- so the store's own write-once +/// create refuses it. There is nothing left to detect at that shape: an id above the seal cannot be +/// minted either, because ids are state-derived and a writer that could derive `{E, T+2}` would have had +/// to observe the seal first. +/// +/// `CasEventType::RefLateLogDetected` is retired WITH the detector -- pre-release, so a vocabulary entry +/// nothing can emit is just dead surface. Soak scenario S38 (`s38_late_put_injection.py`) keeps its +/// injection and FLIPS its assertion: from "the detection fired" to "the fence held" -- the late PUT's +/// conditional create must LOSE to the occupied slot, with zero data loss and the namespace folding +/// normally. diff --git a/src/Disks/tests/gtest_cas_orphan_nomination.cpp b/src/Disks/tests/gtest_cas_orphan_nomination.cpp new file mode 100644 index 000000000000..d79f10bf73f7 --- /dev/null +++ b/src/Disks/tests/gtest_cas_orphan_nomination.cpp @@ -0,0 +1,327 @@ +#include + +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +constexpr uint64_t kCandidateEpoch = 1; +constexpr uint64_t kCandidateBuild = 5; +const UInt128 kGcId = hexToU128("000000000000000000000000000000d8"); + +ManifestRef candidateRef() +{ + return ManifestRef{.writer_epoch = kCandidateEpoch, .build_sequence = kCandidateBuild, .manifest_ordinal = 1}; +} + +bool manifestExists(Backend & backend, const Layout & layout, const ManifestId & id) +{ + return backend.head(layout.manifestKey(id)).exists; +} + +bool activeSourceExists(Backend & backend, const Layout & layout, const UInt128 & source_id) +{ + const auto state_got = backend.get(layout.gcStateKey()); + if (!state_got) + return false; + const GcState state = decodeGcState(state_got->bytes); + const auto seal_got = backend.get(layout.foldSealKey(state.snap_generation, state.snap_attempt)); + if (!seal_got) + return false; + const CasFoldSeal seal = decodeFoldSeal(seal_got->bytes); + for (const RunRef & run : seal.blob_target_runs) + { + SourceEdgeRunView view = openSourceEdgeRun(backend, run.key); + String key; + String payload; + while (view.next(key, payload)) + { + if (payload.empty() || payload[0] != kEdgeActive) + continue; + BlobRef ref; + UInt128 row_source{}; + SourceEdgeKeyCodec::parse(key, ref, row_source); + if (row_source == source_id) + return true; + } + view.verifyAgainst(run.checksum); + } + return false; +} + +size_t condemnedCount(Backend & backend, const Layout & layout) +{ + size_t count = 0; + const GcState state = decodeGcState(backend.get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend.get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + for (const RunRef & run : seal.blob_target_runs) + { + SourceEdgeRunView view = openSourceEdgeRun(backend, run.key); + String key; + String payload; + while (view.next(key, payload)) + count += !payload.empty() && payload[0] == kCondemned; + view.verifyAgainst(run.checksum); + } + return count; +} + +class NominationBackend : public InMemoryBackend +{ +public: + using Backend::deleteExact; + using Backend::get; + using Backend::putOverwrite; + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (key == watched_manifest_key) + { + source_absent_when_delete_started = !activeSourceExists(*this, layout, watched_source_id); + if (replace_manifest_before_delete) + { + const auto got = get(key); + if (got) + putOverwrite(key, got->bytes, got->token); + } + } + return InMemoryBackend::deleteExact(key, token); + } + + Layout layout{"p"}; + String watched_manifest_key; + UInt128 watched_source_id{}; + bool source_absent_when_delete_started = false; + bool replace_manifest_before_delete = false; +}; + +struct ReadyFixture +{ + std::shared_ptr backend; + PoolPtr store; + std::unique_ptr gc; + RootNamespace ns{"test/aa@cas@"}; + ManifestId candidate{ns, candidateRef()}; + std::vector blobs; +}; + +ReadyFixture makeReadyFixture() +{ + ReadyFixture f; + f.backend = std::make_shared(); + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "gc-runner"; + config.manifest_sweep_list_budget_keys = 100; + config.manifest_sweep_delete_budget_keys = 100; + config.gc_fold_max_defer_rounds = 0; + f.store = Pool::open(f.backend, config); + f.backend->layout = f.store->layout(); + f.gc = std::make_unique(f.store, kGcId); + + /// Establish a real catalog life and fold its cursor across epoch 1 before introducing the orphan. + publishAt(*f.backend, f.store->layout(), f.ns, RefTxnId{1, 1}, "live-a", /*build_sequence=*/7, + UInt128(0x7001), /*birth=*/true); + EXPECT_TRUE(runRegularRoundReclaiming(*f.gc).acquired_lease); + writeSealAt(*f.backend, f.store->layout(), f.ns, RefTxnId{1, 2}); + publishAt(*f.backend, f.store->layout(), f.ns, RefTxnId{2, 1}, "live-b", /*build_sequence=*/7, + UInt128(0x7002), /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); + /// Raw log helpers intentionally do not manufacture lifecycle authority. This fixture's durable + /// frontier includes the predecessor seal and the epoch-2 start, so nomination is exercised rather + /// than being (correctly) skipped for a missing `_ckpt`. + writeRecoverableCkptForRawFixture(*f.backend, f.store->layout(), f.ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + EXPECT_TRUE(runRegularRoundReclaiming(*f.gc).acquired_lease); + setWatermarkMinActive(*f.backend, f.store->layout(), "test", kCandidateEpoch, /*min_active=*/6); + + std::vector entries; + std::vector seeded_edges; + for (uint64_t i = 0; i < 6; ++i) + { + const UInt128 digest = UInt128(0x8000 + i); + const BlobRef blob = legacyMetaTestRef(digest); + f.blobs.push_back(blob); + writeBlobBody(*f.backend, f.store->layout(), digest); + const String path = "blob-" + std::to_string(i); + entries.push_back(blobEntryFor(path, digest)); + seeded_edges.push_back(BlobDelta{ + .ref = blob, + .source_id = sourceEdgeId(f.candidate, path), + .remove = false}); + if (i < 4) + seeded_edges.push_back(BlobDelta{ + .ref = blob, + .source_id = UInt128(0x9000 + i), + .remove = false}); + } + writeManifestRaw(*f.backend, f.store->layout(), f.ns, f.candidate.ref, entries); + + /// Seed the exact S42 precondition: the candidate manifest's `+1` edges are already in the adopted + /// run, yet the recovered owner view does not name the body. Four blobs also have another source. + const auto state_got = f.backend->get(f.store->layout().gcStateKey()); + EXPECT_TRUE(state_got.has_value()); + GcState state = decodeGcState(state_got->bytes); + const auto parent_got = f.backend->get( + f.store->layout().foldSealKey(state.snap_generation, state.snap_attempt)); + EXPECT_TRUE(parent_got.has_value()); + CasFoldSeal seal = decodeFoldSeal(parent_got->bytes); + const uint64_t new_generation = state.snap_generation + 1; + const uint64_t new_attempt = state.snap_attempt + 1000; + std::vector runs; + RetiredMergeResult retired; + foldDeltasIntoGeneration( + *f.backend, f.store->layout(), seal.blob_target_runs, + new_generation, new_attempt, /*shard=*/0, std::move(seeded_edges), runs, + /*current_round=*/state.round, /*condemn_round=*/state.round, + {}, {}, {}, &retired, /*suppress_destructive=*/false, nullptr); + seal.parent_generation = state.snap_generation; + seal.generation = new_generation; + seal.blob_target_runs = std::move(runs); + seal.condemned_summary[0] = CondemnedSummary{}; + putDeterministicArtifact( + *f.backend, f.store->layout().foldSealKey(new_generation, new_attempt), encodeFoldSeal(seal)); + state.snap_generation = new_generation; + state.snap_attempt = new_attempt; + f.backend->putOverwrite(f.store->layout().gcStateKey(), encodeGcState(state), state_got->token); + + f.backend->watched_manifest_key = f.store->layout().manifestKey(f.candidate); + f.backend->watched_source_id = sourceEdgeId(f.candidate, "blob-0"); + return f; +} + +} + +/// S42: sweeping an aborted precommit must retire that manifest's exact source edges before deleting +/// the body. Other sources stay intact, and only the two uniquely-owned blobs enter retirement. +TEST(CASOrphanNomination, RetiresExactManifestSourcesBeforeDelete) +{ + ReadyFixture f = makeReadyFixture(); + + /// The nominating round's own `fold_reduce` phase carries probe B1/B2's per-round verdict; capture + /// it so the orphan-sourced retirement can be proven accounting-neutral on the real end-to-end path, + /// not only on the synthetic `foldDeltasIntoGeneration` call `SourceRetirementIsAccountingNeutral` + /// drives below. + std::optional fold_reduce; + f.gc->setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "fold_reduce") fold_reduce = rec; }); + + ASSERT_TRUE(runRegularRoundReclaiming(*f.gc).acquired_lease); + + EXPECT_FALSE(manifestExists(*f.backend, f.store->layout(), f.candidate)); + EXPECT_TRUE(f.backend->source_absent_when_delete_started) + << "the adopted in-degree run must retire the manifest source before exact deletion begins"; + for (size_t i = 0; i < f.blobs.size(); ++i) + { + EXPECT_FALSE(activeSourceExists( + *f.backend, f.store->layout(), sourceEdgeId(f.candidate, "blob-" + std::to_string(i)))); + EXPECT_EQ(inDegreeInRuns(*f.backend, runsForShard(*f.backend, f.store->layout(), 0), f.blobs[i]), + i < 4 ? 1 : 0); + } + EXPECT_EQ(condemnedCount(*f.backend, f.store->layout()), 2u); + + ASSERT_TRUE(fold_reduce.has_value()); + EXPECT_EQ(fold_reduce->metrics.at("unmatched_removes"), 0u) + << "the orphan source retirements are exact removes against a present edge, never an unmatched one"; + EXPECT_EQ(fold_reduce->metrics.at("transactions_unapplied"), 0u) + << "the retirement input rides the reducer alongside ordinary deltas without stranding a " + "committed+produced ref transaction unapplied"; +} + +/// A nomination must exact-GET and decode the manifest before it can derive any source-edge identity. +/// An undecodable body is retained and surfaced without aborting the rest of the round. +TEST(CASOrphanNomination, CorruptManifestIsRetainedAndSurfaced) +{ + ReadyFixture f = makeReadyFixture(); + const auto got = f.backend->get(f.backend->watched_manifest_key); + ASSERT_TRUE(got.has_value()); + f.backend->putOverwrite(f.backend->watched_manifest_key, "not a sealed manifest", got->token); + + std::optional orphan_sweep; + f.gc->setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "orphan_sweep") orphan_sweep = rec; }); + + RoundReport report; + ASSERT_NO_THROW(report = runRegularRoundReclaiming(*f.gc)); + EXPECT_TRUE(report.acquired_lease); + ASSERT_TRUE(orphan_sweep.has_value()); + EXPECT_EQ(orphan_sweep->metrics.at("undecodable"), 1u); + EXPECT_TRUE(f.backend->head(f.backend->watched_manifest_key).exists); +} + +/// Manifest identities are immutable. A changed token at the same key is illegal ABA, not an ordinary +/// exact-delete race that may be silently treated as spared. +TEST(CASOrphanNomination, TokenAbaIsRetainedAndSurfaced) +{ + ReadyFixture f = makeReadyFixture(); + f.backend->replace_manifest_before_delete = true; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { runRegularRoundReclaiming(*f.gc); }); + EXPECT_TRUE(f.backend->head(f.backend->watched_manifest_key).exists); +} + +/// Nomination PLANNING itself is gated on `!suppress_destructive` +/// (`Gc::fold`'s orphan_sweep call site), not merely its eventual delete -- a suppressed pass must +/// never even LIST candidates. The suppressed universe is selected explicitly, because that is the +/// subject: a round on the production default would open the gate and sweep. +TEST(CASOrphanNomination, SuppressedRoundNominatesNothing) +{ + ReadyFixture f = makeReadyFixture(); + + std::optional orphan_sweep; + f.gc->setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "orphan_sweep") orphan_sweep = rec; }); + + ASSERT_TRUE(f.gc->runRegularRound({}, /*allow_steal*/true, + UniversePolicy::StageA_Suppressed).acquired_lease); + + ASSERT_TRUE(orphan_sweep.has_value()); + EXPECT_EQ(orphan_sweep->metrics.at("suppressed"), 1u); + EXPECT_EQ(orphan_sweep->metrics.at("listed"), 0u) + << "planning is gated on !suppress_destructive; a suppressed pass must not even LIST candidates"; + EXPECT_EQ(orphan_sweep->metrics.at("deleted"), 0u); + EXPECT_TRUE(manifestExists(*f.backend, f.store->layout(), f.candidate)) + << "the orphan body must survive a suppressed round"; +} + +/// The retirement input is deliberately outside both ref-transaction accounting mechanisms: a +/// matching edge disappears, an already-absent one stays an idempotent no-op, and neither can alter B2. +TEST(CASOrphanNomination, SourceRetirementIsAccountingNeutral) +{ + InMemoryBackend backend; + const Layout layout{"p"}; + const BlobRef blob = legacyMetaTestRef(UInt128(0xA001)); + const UInt128 source = UInt128(0xA002); + std::vector parent_runs; + foldDeltasIntoGeneration( + backend, layout, {}, /*new_generation=*/1, /*attempt=*/1, /*shard=*/0, + {BlobDelta{.ref = blob, .source_id = source, .remove = false}}, parent_runs); + + std::vector next_runs; + RetiredMergeResult retired; + std::vector applied{0x5A}; + foldDeltasIntoGeneration( + backend, layout, parent_runs, /*new_generation=*/2, /*attempt=*/2, /*shard=*/0, + {}, next_runs, /*current_round=*/1, /*condemn_round=*/1, + {}, {}, {}, &retired, /*suppress_destructive=*/false, &applied, + {BlobSourceRetirement{.ref = blob, .source_id = source}, + BlobSourceRetirement{.ref = blob, .source_id = UInt128(0xA003)}}); + + EXPECT_EQ(inDegreeInRuns(backend, next_runs, blob), 0); + EXPECT_EQ(retired.unmatched_removes, 0u); + EXPECT_EQ(applied, (std::vector{0x5A})); +} diff --git a/src/Disks/tests/gtest_cas_parallel_commit.cpp b/src/Disks/tests/gtest_cas_parallel_commit.cpp new file mode 100644 index 000000000000..7ae5ab676712 --- /dev/null +++ b/src/Disks/tests/gtest_cas_parallel_commit.cpp @@ -0,0 +1,308 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Task 2 of the CAS parallel-write-path plan (docs/superpowers/sdd): `promoteBuild`/`repointRef` +/// return an exact, in-lane-derived `Cas::CommitOutcome` instead of `void`/`bool`, and +/// `dropRefIfMatches` gives a future rollback a conditional drop keyed on that exact outcome instead +/// of the unsafe-under-concurrency `dropRef` (which removes whatever manifest currently occupies the +/// ref name). This suite grows across the later parallel-commit tasks; here it only proves the +/// outcome is exact and that the conditional drop is a true guard -- still single-threaded commit, no +/// concurrency yet. +/// +/// Task 3 reworks `ContentAddressedTransaction::commit()`'s rollback to be EXACT (per-part +/// `Cas::CommitOutcome` slots + `dropRefIfMatches`) while the commit loop stays single-threaded -- +/// correctness-first, before Task 5 adds concurrency. `CasCommitRollback` below drives real +/// `ContentAddressedTransaction`s (not the bare pool primitives `CaWiringFixture` above exercises) +/// through the exact `publishStaging` call path production `commit()` uses, so the fault seams +/// (`armPromoteFailure`/`armAfterPromoteHook`) fire from the real thing. + +using namespace DB; +using namespace DB::Cas::tests; + +namespace +{ + +/// Fixture mirroring `gtest_cas_part_folder_access.cpp`'s `publishPart`/`cacheOn` helpers: a fresh +/// in-memory pool + a `CachedPartFolderAccess` facade over it, plus the minimal staging helpers this +/// suite's tests need (stage a simple one-file part without promoting it; stage-and-promote it in one +/// call; repoint an already-committed ref onto a fresh manifest, modeling a later writer). +struct CaWiringFixture +{ + std::shared_ptr backend = std::make_shared(); + Cas::PoolPtr store = openPoolForTest(backend); + Cas::CachedPartFolderAccess access{store}; + Cas::RootNamespace namespace_{"srv/t1"}; + int content_counter = 0; + + const Cas::RootNamespace & ns() const { return namespace_; } + Cas::CachedPartFolderAccess & partAccess() { return access; } + + static Cas::ManifestEntry inlineEntry(const String & path, const String & bytes) + { + Cas::ManifestEntry e; + e.path = path; + e.placement = Cas::EntryPlacement::Inline; + e.ref = Cas::BlobRef{Cas::BlobHashAlgo::CityHash128, Cas::BlobDigest::fromU128(u128Of(bytes))}; + e.blob_size = bytes.size(); + e.inline_bytes = bytes; + return e; + } + + struct Staged + { + Cas::PartWriteTxnPtr build; + Cas::ManifestId id; + }; + + /// Stages a fresh build (manifest + precommit) for `key` over `blobs` inline entries, WITHOUT + /// promoting it -- the caller drives `promoteBuild` itself so it can observe the exact + /// `CommitOutcome` the promote primitive derives. + Staged stageSimplePart(const Cas::PartRefKey & key, int blobs) const + { + std::vector entries; + for (int i = 0; i < blobs; ++i) + entries.push_back(inlineEntry(fmt::format("f{}", i), fmt::format("payload-{}-{}", key.ref, i))); + auto build = store->beginPartWrite(Cas::PartWriteInfo{ + .intended_ref = key.ns.string() + "/" + key.ref, .intended_namespace = key.ns, .op = Cas::ProvenanceOp::Insert}); + const Cas::ManifestId id = build->stageManifest(entries); + build->precommitAdd(key.ns, key.ref, id); + return {std::move(build), id}; + } + + /// Stages and promotes one simple part end-to-end, returning the exact `CommitOutcome`. + Cas::CommitOutcome commitSimplePart(const Cas::PartRefKey & key, int blobs) + { + auto staged = stageSimplePart(key, blobs); + return access.promoteBuild(*staged.build, key, staged.build->buildId(), staged.id); + } + + /// Repoints an already-committed `key` onto a fresh manifest (different content), through the + /// public `repointRef` primitive -- models "another writer" rebinding the ref after this + /// fixture's own `commitSimplePart`. + Cas::CommitOutcome repointToFreshManifest(const Cas::PartRefKey & key) + { + return access.repointRef(key, {inlineEntry("f0", fmt::format("repoint-{}", ++content_counter))}, + Cas::ProvenanceOp::Other); + } +}; + +} + +TEST(CASCommitOutcome, PromoteReportsCreatedAndManifest) +{ + CaWiringFixture fx; + const Cas::PartRefKey key{fx.ns(), "20260101_1_1_0"}; + auto staged = fx.stageSimplePart(key, /*blobs=*/1); + + const Cas::CommitOutcome oc = fx.partAccess().promoteBuild(*staged.build, key, staged.build->buildId(), staged.id); + + EXPECT_TRUE(oc.created); + EXPECT_EQ(oc.ns.string(), key.ns.string()); + EXPECT_EQ(oc.ref, key.ref); + EXPECT_EQ(oc.manifest_ref, staged.id.ref); +} + +TEST(CASCommitOutcome, DropRefIfMatchesRemovesOnlyExact) +{ + CaWiringFixture fx; + const Cas::PartRefKey key{fx.ns(), "20260101_2_2_0"}; + const Cas::CommitOutcome oc1 = fx.commitSimplePart(key, /*blobs=*/1); + EXPECT_TRUE(oc1.created); + + /// Rebind key -> M2 (a legitimate repoint by "another writer"). + const Cas::CommitOutcome oc2 = fx.repointToFreshManifest(key); + EXPECT_FALSE(oc2.created); + ASSERT_NE(oc1.manifest_ref, oc2.manifest_ref); + + /// Conditional drop keyed on the STALE M1 must NOT remove the current M2 binding. + EXPECT_FALSE(fx.partAccess().dropRefIfMatches(key, oc1.manifest_ref)); + EXPECT_TRUE(fx.partAccess().existsRef(key, Cas::Freshness::ForceFresh)); + + /// Conditional drop keyed on the CURRENT M2 removes it. + EXPECT_TRUE(fx.partAccess().dropRefIfMatches(key, oc2.manifest_ref)); + EXPECT_FALSE(fx.partAccess().existsRef(key, Cas::Freshness::ForceFresh)); +} + +TEST(CASCommitOutcome, DropRefIfMatchesOnAbsentRefIsANoOp) +{ + CaWiringFixture fx; + const Cas::PartRefKey key{fx.ns(), "20260101_3_3_0"}; + Cas::ManifestRef bogus; + EXPECT_FALSE(fx.partAccess().dropRefIfMatches(key, bogus)) << "no committed ref at all: nothing to match"; + EXPECT_FALSE(fx.partAccess().existsRef(key, Cas::Freshness::ForceFresh)); +} + +/// `repointRef`'s byte-equal candidate is a documented ZERO-pool-mutation no-op (it must not mint a +/// fresh manifest just to compare it). The returned `CommitOutcome` must still describe reality: the +/// CURRENTLY committed manifest, unchanged, `created=false`. +TEST(CASCommitOutcome, RepointRefByteEqualNoOpReportsCurrentManifestNotCreated) +{ + CaWiringFixture fx; + const Cas::PartRefKey key{fx.ns(), "20260101_4_4_0"}; + const Cas::CommitOutcome oc1 = fx.commitSimplePart(key, /*blobs=*/1); + + const Cas::CommitOutcome oc_noop = fx.partAccess().repointRef( + key, {CaWiringFixture::inlineEntry("f0", fmt::format("payload-{}-0", key.ref))}, Cas::ProvenanceOp::Other); + EXPECT_FALSE(oc_noop.created); + EXPECT_EQ(oc_noop.manifest_ref, oc1.manifest_ref); +} + +namespace +{ + +/// Fixture for the `CasCommitRollback` suite: wraps a real `ContentAddressedMetadataStorage` and +/// drives ordinary `ContentAddressedTransaction`s through disk paths, so the fault seams under test +/// (`ContentAddressedMetadataStorage::armPromoteFailureForTest`/`setAfterPromoteHookForTest`, the +/// minimal test-only hooks this task adds) fire from the SAME `publishStaging` call path production +/// `commit()` uses -- unlike `CaWiringFixture` above, which pokes the bare pool primitives directly. +/// Every part in one fixture instance shares ONE fixed table uuid (and therefore one `RootNamespace`), +/// matching every test's single `fx.ns()`. +struct CaTxnRollbackFixture +{ + static constexpr const char * kTableUuid = "c3c3c3c3-0000-4000-8000-c3c3c3c3c3c3"; + + std::shared_ptr storage; + Cas::RootNamespace namespace_; + Cas::ManifestRef last_repoint_manifest; + int content_counter = 0; + + static std::string tablePrefix() + { + return std::string(kTableUuid).substr(0, 3) + "/" + kTableUuid; + } + + const Cas::RootNamespace & ns() const { return namespace_; } + Cas::CachedPartFolderAccess & partAccess() const { return *storage->partAccess(); } + + DB::MetadataTransactionPtr beginTxn() const { return storage->createTransaction(); } + + /// Stages `blobs` small distinct files for `key` under a tmp build dir and re-keys them to the + /// final ref name -- the standard MergeTree-insert shape (`gtest_ca_transaction.cpp`'s + /// `writeFileTx` + `moveDirectory` idiom) this storage's routing expects; `key.ns` must be `ns()`. + void stageInto(const DB::MetadataTransactionPtr & txn, const Cas::PartRefKey & key, int blobs) + { + auto & ca_tx = dynamic_cast(*txn); + const std::string tmp_dir = tablePrefix() + "/tmp_insert_" + key.ref; + for (int i = 0; i < blobs; ++i) + { + auto buf = ca_tx.writeFile(fmt::format("{}/f{}.bin", tmp_dir, i), 65536, DB::WriteMode::Rewrite, {}); + const std::string bytes = fmt::format("payload-{}-{}", key.ref, i); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + } + txn->moveDirectory(tmp_dir, tablePrefix() + "/" + key.ref); + } + + /// Stages and commits one part end-to-end in its own transaction -- sets up a pre-existing + /// committed ref before the transaction under test begins. + void commitSimplePart(const Cas::PartRefKey & key, int blobs) + { + auto txn = beginTxn(); + stageInto(txn, key, blobs); + txn->commit(DB::NoCommitOptions{}); + } + + /// Repoints an already-committed `key` onto a fresh manifest through the public `repointRef` + /// primitive directly -- models "another writer" rebinding the ref concurrently with the + /// transaction under test. Records the manifest for `lastRepointManifest()`. + void repointToFreshManifest(const Cas::PartRefKey & key) + { + const std::string bytes = fmt::format("repoint-{}", ++content_counter); + Cas::ManifestEntry e; + e.path = "f0.bin"; + e.placement = Cas::EntryPlacement::Inline; + e.ref = Cas::BlobRef{Cas::BlobHashAlgo::CityHash128, Cas::BlobDigest::fromU128(u128Of(bytes))}; + e.blob_size = bytes.size(); + e.inline_bytes = bytes; + const auto oc = partAccess().repointRef(key, {e}, Cas::ProvenanceOp::Other); + last_repoint_manifest = oc.manifest_ref; + } + + /// The manifest CURRENTLY bound to `key`, or a default-constructed (zero) `ManifestRef` when `key` + /// has no committed ref at all. + Cas::ManifestRef currentManifest(const Cas::PartRefKey & key) const + { + auto view = partAccess().getView(key, Cas::Freshness::ForceFresh); + return view ? view->manifestId().ref : Cas::ManifestRef{}; + } + + Cas::ManifestRef lastRepointManifest() const { return last_repoint_manifest; } + + /// Test-only fault seam (see `ContentAddressedMetadataStorage::armPromoteFailureForTest`): the + /// NEXT `publishStaging` promote/repoint for `key` (the full `(ns, ref)` routed identity) throws + /// instead of committing. + void armPromoteFailure(const Cas::PartRefKey & key) const { storage->armPromoteFailureForTest(key); } + /// Test-only hook (see `ContentAddressedMetadataStorage::setAfterPromoteHookForTest`): runs once, + /// synchronously, immediately after `key`'s promote/repoint confirms. + void armAfterPromoteHook(const Cas::PartRefKey & key, std::function hook) const + { + storage->setAfterPromoteHookForTest(key, std::move(hook)); + } +}; + +CaTxnRollbackFixture makeCaWiringFixture() +{ + static std::atomic counter{0}; + const auto scratch = std::filesystem::temp_directory_path() + / fmt::format("ca_commit_rollback_scratch_{}_{}", ::getpid(), counter.fetch_add(1)); + auto settings = DB::Cas::tests::makeSettingsForTest("test", scratch); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + + CaTxnRollbackFixture fx; + fx.storage = storage; + fx.namespace_ = storage->liveNamespace(CaTxnRollbackFixture::kTableUuid); + return fx; +} + +} + +/// [TXN-ONE-PIPELINE] Task 3: `commit()` publishes `new_a` (created=true) then fails on `new_b`'s +/// promote. The rollback must drop the just-created `new_a` (absent afterward) but never touch the +/// unrelated `pre_existing` ref committed by an EARLIER, already-finished transaction. +TEST(CASCommitRollback, AbsentBeforeDroppedPreExistingUntouched) +{ + auto fx = makeCaWiringFixture(); + const Cas::PartRefKey pre{fx.ns(), "pre_existing_1_1_0"}; + fx.commitSimplePart(pre, 1); // a pre-existing ref, must survive + // A transaction that commits one NEW part then fails on a second part's promote. + auto txn = fx.beginTxn(); + fx.stageInto(txn, {fx.ns(), "new_a_1_1_0"}, 1); + fx.stageInto(txn, {fx.ns(), "new_b_1_1_0"}, 1); + fx.armPromoteFailure({fx.ns(), "new_b_1_1_0"}); // fault injection in publishStaging's promote + EXPECT_ANY_THROW(txn->commit({})); + EXPECT_FALSE(fx.partAccess().existsRef({fx.ns(), "new_a_1_1_0"}, Cas::Freshness::ForceFresh)); // rolled back + EXPECT_TRUE (fx.partAccess().existsRef(pre, Cas::Freshness::ForceFresh)); // untouched +} + +/// [TXN-ONE-PIPELINE] Task 3: T1 (this transaction) promotes `shared` (M1), then a concurrent writer +/// (modeled by the after-promote hook) repoints it to M2 BEFORE T1's own commit later fails on +/// `poison`'s promote. Rollback must use `dropRefIfMatches(M1)`: M1 != the now-current M2, so the +/// conditional drop must leave `shared` bound to M2 untouched. +/// +/// `commit()` publishes `parts` in the map's own (ns, ref) sort order -- so the "shared" part is named +/// `a_shared_...` and the "poison" part `z_poison_...` here purely so `'a' < 'z'` makes "shared" +/// publish (and get repointed by the hook) deterministically BEFORE "poison" fails; this is a test +/// naming choice, not a production ordering guarantee. +TEST(CASCommitRollback, RepointByOtherWriterSurvivesRollback) +{ + auto fx = makeCaWiringFixture(); + const Cas::PartRefKey key{fx.ns(), "a_shared_1_1_0"}; + auto txn = fx.beginTxn(); + fx.stageInto(txn, key, 1); // T1 will create R -> M1 + fx.armAfterPromoteHook(key, [&]{ fx.repointToFreshManifest(key); }); // T2 repoints R -> M2 right after T1's promote + fx.stageInto(txn, {fx.ns(), "z_poison_1_1_0"}, 1); + fx.armPromoteFailure({fx.ns(), "z_poison_1_1_0"}); + EXPECT_ANY_THROW(txn->commit({})); + // T1's rollback used dropRefIfMatches(M1); M2 != M1 so it must survive. + EXPECT_TRUE(fx.partAccess().existsRef(key, Cas::Freshness::ForceFresh)); + EXPECT_EQ(fx.currentManifest(key), fx.lastRepointManifest()); +} diff --git a/src/Disks/tests/gtest_cas_part_folder_access.cpp b/src/Disks/tests/gtest_cas_part_folder_access.cpp new file mode 100644 index 000000000000..69fc028aeda4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_part_folder_access.cpp @@ -0,0 +1,1285 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int FILE_DOESNT_EXIST; + extern const int ABORTED; + extern const int BAD_ARGUMENTS; + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; + extern const int MEMORY_LIMIT_EXCEEDED; + extern const int NETWORK_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASRefRollbackBestEffortDropFailed; +extern const Event CASPartFolderValidateSkipped; +} + +using namespace DB; +using namespace DB::Cas::tests; + +namespace +{ + +Cas::ManifestEntry inlineEntry(const String & path, const String & bytes) +{ + Cas::ManifestEntry e; + e.path = path; + e.placement = Cas::EntryPlacement::Inline; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(bytes))}; + + e.blob_size = bytes.size(); + e.inline_bytes = bytes; + return e; +} + +/// Publish `entries` as committed ref `ns/ref` through the real writer protocol. +Cas::ManifestId publishPart(const Cas::PoolPtr & store, const Cas::RootNamespace & ns, + const String & ref, std::vector entries) +{ + auto build = store->beginPartWrite(Cas::PartWriteInfo{.intended_ref = ns.string() + "/" + ref, + .intended_namespace = ns, .op = Cas::ProvenanceOp::Insert}); + const Cas::ManifestId id = build->stageManifest(entries); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +Cas::CachedPartFolderAccess::CacheParams cacheOn() +{ + return {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, + .explain_enabled = true, .validate = {}}; +} + +/// Mirrors gtest_cas_s3_staging.cpp's helper of the same shape: the shape a real CAS disk config +/// has under `storage_configuration.disks.`, so `config_prefix = "disk"` reads exactly like +/// the disk factory's `config_prefix`. Used to unit-test `parsePartFolderValidate` standalone. +Poco::AutoPtr configWithDiskSection(const std::string & inner_xml) +{ + std::istringstream xml_stream( // STYLE_CHECK_ALLOW_STD_STRING_STREAM + "" + inner_xml + ""); + return new Poco::Util::XMLConfiguration(xml_stream); +} + +/// Every mutating backend op throws once armed — models a correlated backend outage during the +/// transaction's compensating rollback (dropRef must append a removal, which mutates the backend). +class RollbackFaultBackend final : public Cas::InMemoryBackend +{ +public: + std::atomic armed{false}; + + Cas::PutResult putIfAbsent(const String & k, const String & b, const Cas::ObjectMeta & m) override + { + failIfArmed(); + return InMemoryBackend::putIfAbsent(k, b, m); + } + + Cas::PutResult putOverwrite(const String & k, const String & b, const Cas::Token & e, const Cas::ObjectMeta & m) override + { + failIfArmed(); + return InMemoryBackend::putOverwrite(k, b, e, m); + } + + Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const Cas::ObjectMeta & m) override + { + failIfArmed(); + return InMemoryBackend::casPut(k, b, e, m); + } + + Cas::DeleteOutcome deleteExact(const String & k, const Cas::Token & t) override + { + failIfArmed(); + return InMemoryBackend::deleteExact(k, t); + } + +private: + void failIfArmed() + { + if (armed.load()) + throw Exception(ErrorCodes::ABORTED, "injected backend outage"); + } +}; + +/// Task 7 (`publishEntries` abandons its build on exception): forces publishEntries's PROMOTE step +/// specifically -- not the earlier stageManifest/precommitAdd writes -- to observe a proven ref-log +/// conflict. `skip` lets the FIRST matching '_log/' PUT (precommitAdd's OwnerTransition-to-Precommit) +/// land normally; the fault then fires on the SECOND (promote's atomic precommit->committed move). +/// Mirrors `RefWriterTestBackend::corrupt_key_substr` (gtest_cas_ref_writer.cpp, reproduced locally +/// because that class lives in a different translation unit): landing a DIFFERENT object at the +/// intended key makes `putIfAbsentControlled`'s resolve-before-reissue observe a proven conflict +/// (CORRUPTED_DATA) rather than the ambiguous-timeout shape, which would instead wedge the whole +/// table's append lane. +class PromoteConflictOnceBackend final : public Cas::InMemoryBackend +{ +public: + String fault_key_substr; + int skip = 0; + int fault_count = 0; + /// Every create ATTEMPTED at a matching key, faulted or not. It is how a test observes that a + /// cleanup path ran its ref-log append at all, on a table where that append can no longer succeed. + int matching_put_attempts = 0; + + Cas::PutResult putIfAbsent(const String & key, const String & bytes, const Cas::ObjectMeta & meta) override + { + if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + { + ++matching_put_attempts; + if (skip > 0) + --skip; + else if (fault_count > 0) + { + --fault_count; + /// The 3-arg qualified call bypasses virtual dispatch entirely (unlike a 2-arg + /// convenience overload, which would re-enter this very override through the vtable). + InMemoryBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT"), meta); + throw Poco::TimeoutException("PromoteConflictOnceBackend: a foreign different object landed; response lost"); + } + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } +}; + +/// The same shape as `PromoteConflictOnceBackend`, except its fault is a whitelisted SYNCHRONOUS +/// REJECTION -- an S3-classified malformed request, which `classifyConditionalWriteResult` proves was +/// never applied. That distinction is the whole reason this second backend exists: a proven DIFFERENT +/// OBJECT is a breach of mount write-exclusivity and fences the whole mount closed, so every cleanup +/// append after it is refused at the gate and becomes unobservable. A definite rejection is an ordinary +/// failed write -- nothing is fenced, nothing is wedged, the table stays usable -- so the cleanup +/// appends that follow DO reach the store and can be counted. +class PromoteDefiniteFailureBackend final : public Cas::InMemoryBackend +{ +public: + String fault_key_substr; + int skip = 0; + int fault_count = 0; + int matching_put_attempts = 0; + + Cas::PutResult putIfAbsent(const String & key, const String & bytes, const Cas::ObjectMeta & meta) override + { + if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + { + ++matching_put_attempts; + if (skip > 0) + --skip; + else if (fault_count > 0) + { + --fault_count; + throw DB::S3Exception("PromoteDefiniteFailureBackend: simulated malformed request", + Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); + } + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } +}; + +} + +TEST(CASPartFolderAccess, RetainedHitSkipsManifestHead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + backend->resetCounts(); + for (int i = 0; i < 5; ++i) + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + + /// The one-GET goal (spec acceptance 4): ONE body GET, ONE mandatory HEAD (the cold build); + /// every subsequent CachedForLoad call is a validated hit — zero manifest ops. + EXPECT_EQ(backend->getCount(manifest_key), 1u); + EXPECT_EQ(backend->headCount(manifest_key), 1u); + EXPECT_TRUE(access.explain(key).retained); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::Hit); +} + +TEST(CASPartFolderAccess, HitPathJournalEmptyAndCheapWhenExplainDisabled) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + /// Retention ON, explain journal OFF (the production default): the hit path must take neither the + /// per-disk explain mutex nor write a journal entry (B2). + Cas::CachedPartFolderAccess access(store, + {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, + .explain_enabled = false, .validate = {}}); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + backend->resetCounts(); + for (int i = 0; i < 5; ++i) + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + + /// Same request oracle as RetainedHitSkipsManifestHead — one cold build, then validated hits. + EXPECT_EQ(backend->getCount(manifest_key), 1u); + EXPECT_EQ(backend->headCount(manifest_key), 1u); + /// The journal is never written when disabled. + EXPECT_EQ(access.explainJournalSizeForTest(), 0u); + /// explain() still reports live retention truthfully, but the decision defaults to Miss (unwritten). + EXPECT_TRUE(access.explain(key).retained); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::Miss); +} + +TEST(CASPartFolderAccess, GetViewServesCommittedFolder) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + publishPart(store, ns, "part_1", + {inlineEntry("checksums.txt", "cs"), inlineEntry("count.txt", "1"), inlineEntry("txn_version.txt", "v1")}); + + Cas::CachedPartFolderAccess access(store); + const Cas::PartRefKey key{ns, "part_1"}; + + auto view = access.getView(key, Cas::Freshness::CachedForLoad); + ASSERT_NE(view, nullptr); + EXPECT_NE(view->findFile("checksums.txt"), nullptr); + EXPECT_EQ(view->inlineBytes("txn_version.txt"), std::optional("v1")); + + /// Absent ref => nullptr, never an exception, never retained (nothing to retain in Phase 2). + EXPECT_EQ(access.getView({ns, "absent"}, Cas::Freshness::CachedForLoad), nullptr); + EXPECT_TRUE(access.existsRef(key, Cas::Freshness::CachedForLoad)); + EXPECT_FALSE(access.existsRef({ns, "absent"}, Cas::Freshness::ForceFresh)); + ASSERT_TRUE(access.resolve(key, Cas::Freshness::ForceFresh).has_value()); +} + +TEST(CASPartFolderAccess, GetViewFailsClosedOnMissingBody) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + + /// Physically delete the live manifest body (a protocol violation) — every getView mode must + /// surface INV-NO-DANGLE as FILE_DOESNT_EXIST in Phase 2 (there is no retained view to hit). + /// Retention is off (the single-arg ctor below), so this is the `always` (default) part_folder_validate + /// mode under test regardless — the `never`/`age` skip is proven by the ValidateNever/ValidateAge + /// tests further down, which turn retention ON. + deleteManifestBody(*backend, layout, id); + + Cas::CachedPartFolderAccess access(store); + const Cas::PartRefKey key{ns, "part_1"}; + for (auto freshness : {Cas::Freshness::CachedForLoad, + Cas::Freshness::ForceFresh, + Cas::Freshness::StrictValidate}) + expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, [&] { access.getView(key, freshness); }); +} + +TEST(CASPartFolderAccess, WritePrimitivesRoundTrip) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store); + const Cas::PartRefKey key{ns, "part_1"}; + + /// promoteBuild: the transaction's terminal publish step, through the facade. + auto build = store->beginPartWrite(Cas::PartWriteInfo{.intended_ref = ns.string() + "/part_1", + .intended_namespace = ns, .op = Cas::ProvenanceOp::Insert}); + const Cas::ManifestId id = build->stageManifest({inlineEntry("checksums.txt", "cs")}); + build->precommitAdd(ns, "part_1", id); + access.promoteBuild(*build, key, build->buildId(), id); + ASSERT_TRUE(access.existsRef(key, Cas::Freshness::ForceFresh)); + + /// dropRefIfPresent: replay-safe (absent ref is success, not failure). + access.dropRefIfPresent(key); + EXPECT_FALSE(access.existsRef(key, Cas::Freshness::ForceFresh)); + access.dropRefIfPresent(key); /// second drop: no-op, no throw + access.dropRefBestEffort(key); /// noexcept even when absent + + /// dropNamespace clears the whole namespace. + publishPart(store, ns, "part_2", {inlineEntry("checksums.txt", "cs")}); + access.dropNamespace(ns); + EXPECT_FALSE(access.existsRef({ns, "part_2"}, Cas::Freshness::ForceFresh)); +} + +TEST(CASPartFolderAccess, RepublishRefMovesCommittedRef) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store); + publishPart(store, ns, "src_part", {inlineEntry("checksums.txt", "cs"), inlineEntry("txn_version.txt", "v1")}); + + EXPECT_FALSE(access.republishRef({ns, "absent"}, {ns, "dst"})); /// absent source: nothing written + + ASSERT_TRUE(access.republishRef({ns, "src_part"}, {ns, "dst_part"})); + EXPECT_FALSE(access.existsRef({ns, "src_part"}, Cas::Freshness::ForceFresh)); + auto view = access.getView({ns, "dst_part"}, Cas::Freshness::ForceFresh); + ASSERT_NE(view, nullptr); + EXPECT_NE(view->findFile("checksums.txt"), nullptr); + EXPECT_EQ(view->inlineBytes("txn_version.txt"), std::optional("v1")); /// carried over +} + +TEST(CASPartFolderAccess, RepublishRefIdempotentRedriveAndConflict) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store); + + /// Re-drive: dst already committed with the SAME content (a prior attempt's promote landed, + /// only dropRef(src) was interrupted) -- idempotent-skip: drop src, dst's manifest is untouched + /// (all-tree-part-files Task 9: there is no separate mutable payload left to drift/re-sync -- + /// identical `entries` is the whole idempotency contract now). + publishPart(store, ns, "src", {inlineEntry("f", "same")}); + publishPart(store, ns, "dst", {inlineEntry("f", "same")}); + const auto dst_id_before = access.resolve({ns, "dst"}, Cas::Freshness::ForceFresh)->manifest_id; + ASSERT_TRUE(access.republishRef({ns, "src"}, {ns, "dst"})); + EXPECT_FALSE(access.existsRef({ns, "src"}, Cas::Freshness::ForceFresh)); + auto resolved = access.resolve({ns, "dst"}, Cas::Freshness::ForceFresh); + EXPECT_EQ(resolved->manifest_id, dst_id_before) << "idempotent re-drive must not mint a fresh manifest"; + + /// Conflict: dst committed with DIFFERENT content — fail closed, src untouched. + publishPart(store, ns, "src2", {inlineEntry("f", "one")}); + publishPart(store, ns, "dst2", {inlineEntry("f", "two")}); + expectThrowsCode(ErrorCodes::ABORTED, [&] { access.republishRef({ns, "src2"}, {ns, "dst2"}); }); + EXPECT_TRUE(access.existsRef({ns, "src2"}, Cas::Freshness::ForceFresh)); +} + +/// Task 7: `publishEntries`'s `catch (...) { build->abandon(); throw; }` must leave no live-epoch +/// precommit binding behind when its promote fails -- only `abandon()` removes it (the build +/// destructor merely retires the build seq; GC never touches a live precommit). Drives the failure +/// through `republishRef` -> `publishEntries`, with the fault isolated to promote's own ref-log +/// append (precommitAdd's own append is let through first via `skip`). +TEST(CASPartFolderAccess, PublishEntriesAbandonsBuildOnPromoteFailure) +{ + auto backend = std::make_shared(); + auto store = Cas::Pool::open(backend, Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const Cas::RootNamespace ns{"srv/t1"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + Cas::CachedPartFolderAccess access(store); + + publishPart(store, ns, "src", {inlineEntry("f", "same")}); + + backend->fault_key_substr = store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)) + "_log/"; + backend->skip = 1; /// let precommitAdd's own ref-log append land normally + backend->fault_count = 1; /// fault exactly promote's ref-log append + const int attempts_before = backend->matching_put_attempts; + + /// republishRef(src, dst) drives publishEntries(dst, ...): precommitAdd succeeds, promote's + /// appendRefOps observes a proven conflict and throws CORRUPTED_DATA -- publishEntries's catch must + /// abandon() the build before rethrowing. + expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { access.republishRef({ns, "src"}, {ns, "dst"}); }); + /// The anomaly fenced this runtime, so a post-fence `ForceFresh` read must refuse rather than + /// authorizing its stale generation. The backend assertions below prove directly that `dst` never + /// committed and that no append skipped around the damaged slot. + expectThrowsCode(ErrorCodes::NETWORK_ERROR, + [&] { (void)access.existsRef({ns, "dst"}, Cas::Freshness::ForceFresh); }); + + /// EXACTLY two ref-log create attempts reach the store: precommitAdd's own append and the promote's + /// faulted one. Both of the cleanup appends that follow -- `promote`'s catch-abandon and the handle + /// destructor's backstop -- are refused at the mount-fence gate before they reach the store, because + /// proving a different object at our own key now fences the mount closed and schedules a remount + /// (review I5: the append site self-heals like the wedge-resolve site instead of leaving this table + /// blocked until a manual remount). + /// + /// COVERAGE NOTE, deliberately explicit: this count no longer DISCRIMINATES whether the catch-abandon + /// ran. It used to (four attempts with it, three without), and that only worked because the + /// catch-abandon could still reach the store and fail there, making the destructor retry. With the + /// fence closed both cleanups are refused identically and unobservably, so the assertion below is a + /// shape check, not the regression guard it was. The guard cannot be restored in THIS scenario -- + /// nothing the cleanup does is observable once the mount is fenced -- and it is not silently + /// dropped: the property it protected is stated here, and reclaiming the binding is now the + /// scheduled remount's job (a fresh incarnation re-derives the table and the stale-precommit sweep + /// reclaims), not this best-effort abandon's. + EXPECT_EQ(backend->matching_put_attempts, attempts_before + 2) + << "only precommitAdd's append and the promote's faulted one may reach the store; every cleanup " + "append after the anomaly is refused at the fence"; + EXPECT_FALSE(store->mayMutate()) << "the proven conflict must fence this mount closed"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) + << "and must schedule exactly one remount -- the self-heal that replaces the manual one"; + + /// And nothing was written ABOVE the damage: the occupant is the GREATEST log id in the namespace + /// (keys render the id in fixed-width hex, so lexical order is id order). An append that carved a + /// fresh id to get past the foreign object would sort above it. + String greatest_key; + size_t foreign_objects = 0; + for (String cursor;;) + { + const Cas::ListPage page = backend->list(backend->fault_key_substr, cursor, 1000); + for (const auto & listed : page.keys) + { + if (listed.key > greatest_key) + greatest_key = listed.key; + const auto body = backend->get(listed.key); + if (body && body->bytes.find("_FOREIGN_DIFFERENT") != String::npos) + ++foreign_objects; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + EXPECT_EQ(foreign_objects, 1u) << "the foreign object must still own the key it took"; + ASSERT_FALSE(greatest_key.empty()); + const auto greatest_body = backend->get(greatest_key); + ASSERT_TRUE(greatest_body.has_value()); + EXPECT_NE(greatest_body->bytes.find("_FOREIGN_DIFFERENT"), String::npos) + << "the foreign occupant must still be the highest id in this table's stream: a log object above " + "it would mean an append carved a fresh id around the damage instead of failing closed"; +} + +/// The DISCRIMINATING guard for the same duty, on the path where it can still be observed: a promote +/// failure that is an ordinary failed write rather than a breach of mount write-exclusivity. Nothing is +/// fenced and nothing is wedged, so both cleanup appends reach the store and the two worlds separate. +/// +/// The fault covers TWO appends, and that is the whole construction: +/// with `promote`'s catch-abandon -- precommitAdd lands (skipped), promote's append is refused, +/// the catch-abandon's append is refused too, and the handle DESTRUCTOR's backstop retries and lands: +/// FOUR attempts, and no binding is left behind; +/// without it -- precommitAdd lands, promote's append is refused, and the destructor's backstop takes +/// the second fault and is refused: THREE attempts, and the precommit binding LEAKS. +/// So the count and the end state disagree between the two worlds, which is what makes this a guard +/// rather than a shape check. `livePrecommitsForTest` is the direct statement of the property -- +/// `publishEntries` must not walk away from a live precommit binding -- and the count is what pins +/// WHERE the cleanup came from. +TEST(CASPartFolderAccess, PublishEntriesAbandonsBuildOnARetryablePromoteFailure) +{ + auto backend = std::make_shared(); + auto store = Cas::Pool::open(backend, Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const Cas::RootNamespace ns{"srv/t1"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + Cas::CachedPartFolderAccess access(store); + + publishPart(store, ns, "src", {inlineEntry("f", "same")}); + + backend->fault_key_substr = store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)) + "_log/"; + backend->skip = 1; /// let precommitAdd's own ref-log append land normally + backend->fault_count = 2; /// fault promote's append AND the cleanup append that follows it + const int attempts_before = backend->matching_put_attempts; + + /// A definite rejection is reported to the caller as a retry-later failure, not as corruption. + expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { access.republishRef({ns, "src"}, {ns, "dst"}); }); + EXPECT_FALSE(access.existsRef({ns, "dst"}, Cas::Freshness::ForceFresh)) << "the failed promote never committed dst"; + + EXPECT_TRUE(store->mayMutate()) << "an ordinary failed write must not fence the mount"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), 0u) << "and must not schedule a remount"; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a definite rejection is proven non-durable: no wedge"; + + EXPECT_EQ(backend->matching_put_attempts, attempts_before + 4) + << "three attempts means only the destructor backstop ran -- publishEntries stopped abandoning " + "the build at the promote site"; + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()) + << "publishEntries must not walk away from a live precommit binding"; +} + + +/// ==== Task 12: the prepared-part-write handle (spec §relink-handle) ==== +/// `prepareEntries` stops after `precommitAdd`, so the durable-but-unpromoted state -- the window the +/// relink confirm round-trip has to sit inside -- becomes an OWNED object instead of an interval inside +/// one call. Every test below pins one half of that ownership contract. + +/// Prepare-then-promote must be indistinguishable from today's atomic `publishEntries`, and the state +/// BETWEEN the two halves must be exactly one live precommit and no committed ref. +TEST(CASPartFolderAccess, PrepareThenPromoteMatchesPublishEntries) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + const std::vector entries{inlineEntry("f", "one"), inlineEntry("g", "two")}; + const Cas::CommitOutcome published = access.publishEntries({ns, "via_publish"}, entries, Cas::ProvenanceOp::Insert); + + auto prepared = access.prepareEntries({ns, "via_prepare"}, entries, Cas::ProvenanceOp::Insert); + + /// The interposition point: the manifest is durable and owned by a LIVE precommit, but nothing is + /// committed yet. This is precisely the state the confirm round-trip runs in. + EXPECT_TRUE(store->livePrecommitsForTest(ns).contains({"via_prepare", prepared.manifestId().ref})) + << "prepareEntries must leave the precommit binding live -- it is the durable `+1`"; + EXPECT_FALSE(access.existsRef({ns, "via_prepare"}, Cas::Freshness::ForceFresh)) + << "prepareEntries must not commit the ref"; + + const Cas::CommitOutcome promoted = prepared.promote(); + EXPECT_EQ(promoted.ns.string(), ns.string()); + EXPECT_EQ(promoted.ref, "via_prepare"); + EXPECT_EQ(promoted.manifest_ref, prepared.manifestId().ref); + EXPECT_TRUE(promoted.created); + EXPECT_EQ(promoted.created, published.created) << "the split must reproduce publishEntries's outcome shape"; + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()) << "promote moves the binding out of the precommit view"; + + auto view = access.getView({ns, "via_prepare"}, Cas::Freshness::ForceFresh); + ASSERT_NE(view, nullptr); + EXPECT_EQ(view->inlineBytes("f"), std::optional("one")); + EXPECT_EQ(view->inlineBytes("g"), std::optional("two")); +} + +/// Abort is not "drop the handle": it must APPEND the exact precommit removal. An abandoned precommit +/// that keeps its `+1` is the retention-leak class (`BACKLOG {#unmatched-minus-one-retention-leak}`), +/// and the stale-precommit sweep is prior-epoch-scoped, so a same-epoch leak is never reclaimed. +/// Asserted through the ledger's own precommit view rather than inferred from a later `precommitAdd`. +TEST(CASPartFolderAccess, PrepareThenAbortAppendsThePrecommitRemoval) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + auto prepared = access.prepareEntries({ns, "part_1"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + const Cas::ManifestId id = prepared.manifestId(); + ASSERT_TRUE(store->livePrecommitsForTest(ns).contains({"part_1", id.ref})); + + prepared.abort(); + + EXPECT_FALSE(access.existsRef({ns, "part_1"}, Cas::Freshness::ForceFresh)) << "an aborted prepare commits nothing"; + EXPECT_FALSE(store->livePrecommitsForTest(ns).contains({"part_1", id.ref})) + << "abort must append the EXACT precommit removal; a same-epoch precommit left behind retains its " + "blobs forever (the prior-epoch-scoped stale sweep never reclaims it)"; + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); + /// The precommit BODY survives (delete-after-sealed-decrements) -- the removal queues GC's `-1`, + /// it does not writer-delete the manifest. Mirrors + /// `CASPartWriteTxn.AbandonAppendsPrecommitRemovalAndKeepsLivePrecommitBody`. + EXPECT_TRUE(backend->head(store->layout().manifestKey(id)).exists); +} + +/// A forgotten terminal must be impossible, not merely discouraged: `~PartWriteTxn` only retires the +/// build sequence, so the handle's own destructor is the last-resort owner of the precommit removal. +TEST(CASPartFolderAccess, DestroyingAnUnfinishedPreparedPartWriteAborts) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + std::optional id; + { + auto prepared = access.prepareEntries({ns, "part_1"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + id = prepared.manifestId(); + ASSERT_TRUE(store->livePrecommitsForTest(ns).contains({"part_1", id->ref})); + } /// neither promoted nor aborted + + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()) + << "destruction without a terminal must still append the precommit removal"; + EXPECT_FALSE(access.existsRef({ns, "part_1"}, Cas::Freshness::ForceFresh)); +} + +/// The terminal flag is explicit and one-shot: a second `promote`/`abort` is a caller bug, not an +/// idempotent no-op, and must never re-drive the (already dead) transaction. +/// +/// The rejection throws LOGICAL_ERROR, which aborts the whole process in debug/sanitizer builds +/// (Exception.cpp's handle_error_code) instead of behaving like a catchable exception -- so the +/// expectThrowsCode form only makes sense in a plain release build, and the DeathTest variant below +/// proves the SAME rejections positively abort under debug/sanitizer builds instead (same pattern as +/// CASWiringOpsDeathTest in gtest_ca_wiring.cpp). +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASPartFolderAccess, PreparedPartWriteRejectsASecondTerminal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + auto promoted = access.prepareEntries({ns, "promoted"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + EXPECT_FALSE(promoted.isTerminal()); + promoted.promote(); + EXPECT_TRUE(promoted.isTerminal()); + expectThrowsCode(ErrorCodes::LOGICAL_ERROR, [&] { promoted.promote(); }); + expectThrowsCode(ErrorCodes::LOGICAL_ERROR, [&] { promoted.abort(); }); + + auto aborted = access.prepareEntries({ns, "aborted"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + aborted.abort(); + EXPECT_TRUE(aborted.isTerminal()); + expectThrowsCode(ErrorCodes::LOGICAL_ERROR, [&] { aborted.abort(); }); + expectThrowsCode(ErrorCodes::LOGICAL_ERROR, [&] { aborted.promote(); }); + + EXPECT_TRUE(access.existsRef({ns, "promoted"}, Cas::Freshness::ForceFresh)); + EXPECT_FALSE(access.existsRef({ns, "aborted"}, Cas::Freshness::ForceFresh)); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} +#else +TEST(CASPartFolderAccessDeathTest, PreparedPartWriteRejectsASecondTerminalAborts) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + auto promoted = access.prepareEntries({ns, "promoted"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + promoted.promote(); + EXPECT_TRUE(promoted.isTerminal()); + EXPECT_DEATH(promoted.promote(), "owes exactly one terminal operation"); + EXPECT_DEATH(promoted.abort(), "owes exactly one terminal operation"); + + auto aborted = access.prepareEntries({ns, "aborted"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + aborted.abort(); + EXPECT_TRUE(aborted.isTerminal()); + EXPECT_DEATH(aborted.abort(), "owes exactly one terminal operation"); + EXPECT_DEATH(aborted.promote(), "owes exactly one terminal operation"); + + EXPECT_TRUE(access.existsRef({ns, "promoted"}, Cas::Freshness::ForceFresh)); + EXPECT_FALSE(access.existsRef({ns, "aborted"}, Cas::Freshness::ForceFresh)); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} +#endif + +/// Move-only, and the move transfers the terminal duty in full: the moved-from handle is already +/// terminal (its destructor must not re-abort a transaction the destination now owns), while the +/// destination still owes exactly one terminal. +TEST(CASPartFolderAccess, PreparedPartWriteMoveTransfersTheTerminalDuty) +{ + static_assert(!std::is_copy_constructible_v); + static_assert(!std::is_copy_assignable_v); + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + + auto source = access.prepareEntries({ns, "part_1"}, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + const Cas::ManifestId id = source.manifestId(); + { + Cas::PreparedPartWrite moved = std::move(source); + /// NOLINTNEXTLINE(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + EXPECT_TRUE(source.isTerminal()) << "a moved-from handle owes nothing"; +#ifndef DEBUG_OR_SANITIZER_BUILD + expectThrowsCode(ErrorCodes::LOGICAL_ERROR, [&] { source.abort(); }); +#else + /// LOGICAL_ERROR aborts the process in debug/sanitizer builds; EXPECT_DEATH forks, so the + /// parent's state (and the rest of this test) is unaffected. + EXPECT_DEATH(source.abort(), "owes exactly one terminal operation"); +#endif + EXPECT_TRUE(store->livePrecommitsForTest(ns).contains({"part_1", id.ref})) + << "the moved-from handle must not have aborted the transaction it handed over"; + EXPECT_EQ(moved.manifestId().ref, id.ref); + moved.promote(); + } /// the moved-from handle's destructor also runs here: it must be a no-op, not a second abort + + EXPECT_TRUE(access.existsRef({ns, "part_1"}, Cas::Freshness::ForceFresh)); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} + +TEST(CASPartFolderAccess, ExplainRecordsDecisions) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, {.explain_enabled = true, .validate = {}}); + publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + + access.getView(key, Cas::Freshness::CachedForLoad); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::Miss); /// cold build + EXPECT_FALSE(access.explain(key).retained); /// Phase 3: never + + access.getView(key, Cas::Freshness::ForceFresh); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::ForceFreshRead); + + access.getView(key, Cas::Freshness::StrictValidate); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::StrictBypass); + + access.dropRef(key); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::Invalidated); + EXPECT_GT(access.explain(key).estimated_bytes, 0u); +} + +TEST(CASPartFolderAccess, BaselineRequestCountsWithoutRetention) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + backend->resetCounts(); + constexpr int n = 5; + for (int i = 0; i < n; ++i) + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + + /// The Phase-3 baseline (retention off): one manifest-body GET (the decode cache absorbs the + /// rest) but a mandatory manifest HEAD per call. Phase 4's validated hits remove the HEADs; + /// this test pins the numbers Phase 4 improves. + EXPECT_EQ(backend->getCount(manifest_key), 1u); + EXPECT_EQ(backend->headCount(manifest_key), static_cast(n)); +} + +/// ==== Phase 4 (retention) semantics battery: spec §Testing acceptance criteria ==== + +/// REMOVED (all-tree-part-files Task 9): +/// `MutableRefreshWithoutManifestRead` and `WriteThroughEraseThenRebuild` proved the cache facade's +/// `LastDecision::MutableRefresh` fast path -- a cheap re-check that could serve a retained view whose +/// manifest was unchanged but whose separate mutable payload had drifted, without a manifest re-read. +/// That whole two-tier freshness model is gone: every per-part file is an ordinary manifest entry now, +/// so ANY content change is a manifest change (`repointRef`) and the existing manifest-id staleness +/// check (`getView`'s `cached->manifestId() == resolved->manifest_id` compare) is the only freshness +/// check left -- there is no cheaper "payload-only" path to test separately. Coverage that remains +/// valid: `MismatchRebuildAfterRepublish` below proves the cache correctly rebuilds when the manifest +/// id changes under a retained view (the one case the deleted tests' "erase => cold rebuild" half also +/// exercised); `gtest_cas_repoint.cpp` (Task 3) proves `repointRef` erases the affected view on success. +TEST(CASPartFolderAccess, MismatchRebuildAfterRepublish) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + publishPart(store, ns, "part_1", {inlineEntry("f", "orig")}); + const Cas::PartRefKey key{ns, "part_1"}; + + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); /// retained + + /// Drop + republish the SAME ref name with DIFFERENT content through the raw Core protocol (no + /// facade => no write-through erase): the retained entry survives with a manifest_id that no + /// longer resolves — the next CachedForLoad hits the manifest-changed compare (step 2c). + store->dropRef(ns, "part_1"); + const auto id2 = publishPart(store, ns, "part_1", {inlineEntry("f", "DIFFERENT")}); + const String manifest_key2 = layout.manifestKey(id2); + backend->resetCounts(); + + auto view = access.getView(key, Cas::Freshness::CachedForLoad); + ASSERT_NE(view, nullptr); + EXPECT_NE(view->findFile("f"), nullptr); + EXPECT_EQ(view->findFile("f")->inline_bytes, "DIFFERENT"); /// never the stale view + EXPECT_EQ(backend->getCount(manifest_key2), 1u); /// one new manifest GET + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::Miss); /// rebuilt, now retained + EXPECT_TRUE(access.explain(key).retained); +} + +TEST(CASPartFolderAccess, ForceFreshFailsClosedWhileRetainedViewExists) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); + const Cas::PartRefKey key{ns, "part_1"}; + + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); /// retained + deleteManifestBody(*backend, layout, id); /// protocol violation: live body vanishes + + /// Write-evidence and strict paths surface INV-NO-DANGLE immediately (mandatory HEAD)... + expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, + [&] { access.getView(key, Cas::Freshness::ForceFresh); }); + expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, + [&] { access.getView(key, Cas::Freshness::StrictValidate); }); + + /// ...while a validated CachedForLoad hit still serves the immutable decode — the documented + /// residual delta (spec §Staleness Equivalence): detection deferred, never for write evidence. + EXPECT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); +} + +/// ==== §3 (part_folder_validate): the ForceFresh body re-proof HEAD is configurable ==== + +TEST(CASPartFolderAccess, ValidateNeverServesRetainedViewWithoutBodyHead) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + + auto params = cacheOn(); + params.validate = {Cas::PartFolderValidate::Mode::Never, 0}; + Cas::CachedPartFolderAccess access(store, params); + const Cas::PartRefKey key{ns, "part_1"}; + + /// Prime the retained view (pays the HEAD once). + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + /// Body vanishes (a protocol violation the net would normally catch)... + deleteManifestBody(*backend, layout, id); + const auto skips_before = ProfileEvents::global_counters[ProfileEvents::CASPartFolderValidateSkipped].load(); + /// ...but `never` serves the retained view, no HEAD, no throw. + EXPECT_NO_THROW(access.getView(key, Cas::Freshness::ForceFresh)); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASPartFolderValidateSkipped].load() - skips_before, 1); +} + +TEST(CASPartFolderAccess, ValidateAlwaysStillHeadsEveryForceFresh) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + + Cas::CachedPartFolderAccess access(store, cacheOn()); /// default = Always + const Cas::PartRefKey key{ns, "part_1"}; + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + deleteManifestBody(*backend, layout, id); + /// `always` re-proves the body every ForceFresh — the deleted body surfaces as FILE_DOESNT_EXIST. + expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, + [&] { access.getView(key, Cas::Freshness::ForceFresh); }); +} + +TEST(CASPartFolderAccess, ValidateAgeSkipsWithinWindowThenHeadsAfter) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + + auto params = cacheOn(); + params.validate = {Cas::PartFolderValidate::Mode::Age, /*age_seconds=*/5}; + /// An injected clock (spec §3 TDD requirement): the SAME function stamps the retained view's + /// validated_at_ms (buildView) and drives the age-window comparison (getView), so the test controls + /// both sides of the comparison deterministically -- no real sleep. + std::atomic fake_now_ms{1'000'000}; + Cas::CachedPartFolderAccess access(store, params, [&] { return fake_now_ms.load(); }); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + /// Prime the retained view (pays the HEAD once) at fake_now_ms. + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + const uint64_t heads_after_prime = backend->headCount(manifest_key); + + /// +2s: still inside the 5s window — served from the retained view, no new HEAD. + fake_now_ms += 2000; + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + EXPECT_EQ(backend->headCount(manifest_key), heads_after_prime); + + /// +6s from the ORIGINAL stamp (past the 5s window): re-proves the body via a fresh HEAD. + fake_now_ms += 4000; + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + EXPECT_GT(backend->headCount(manifest_key), heads_after_prime); +} + +/// ==== §3: `parsePartFolderValidate` config parsing, standalone (mirrors CASS3Staging's +/// parseStagingBackend coverage) -- review finding: std::stoull silently accepted a leading '-' +/// (unsigned wraparound), so a malformed `age -5` never hit the parser's own fail-closed throw. +/// These pin the fixed `std::from_chars`-based parsing directly, with no disk/store needed. ==== + +TEST(CASPartFolderValidateParse, DefaultConfigParsesToAlways) +{ + /// No `part_folder_validate` key at all -- the byte-for-byte-pre-§3-behavior default. + auto config = configWithDiskSection("/tmp/whatever"); + const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); + EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Always); +} + +TEST(CASPartFolderValidateParse, ParsesAlways) +{ + auto config = configWithDiskSection("always"); + const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); + EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Always); +} + +TEST(CASPartFolderValidateParse, ParsesNever) +{ + auto config = configWithDiskSection("never"); + const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); + EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Never); +} + +TEST(CASPartFolderValidateParse, ParsesPositiveAge) +{ + auto config = configWithDiskSection("age 5"); + const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); + EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Age); + EXPECT_EQ(v.age_seconds, 5u); +} + +TEST(CASPartFolderValidateParse, AcceptsAgeZeroAsADegenerateButValidWindow) +{ + /// `age 0` is accepted, not rejected: it is a well-formed (if degenerate -- effectively an + /// almost-always-expired window) configuration, not malformed input. Only genuinely malformed + /// suffixes (negative, non-digit, empty, trailing garbage) fail closed below. + auto config = configWithDiskSection("age 0"); + const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); + EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Age); + EXPECT_EQ(v.age_seconds, 0u); +} + +TEST(CASPartFolderValidateParse, NegativeAgeThrows) +{ + /// The bug this regression-guards: std::stoull("-5") used to return 18446744073709551611 + /// (unsigned wraparound) instead of rejecting the leading '-'. + auto config = configWithDiskSection("age -5"); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, + [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); +} + +TEST(CASPartFolderValidateParse, NonDigitAgeThrows) +{ + auto config = configWithDiskSection("age abc"); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, + [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); +} + +TEST(CASPartFolderValidateParse, TrailingGarbageAfterAgeThrows) +{ + auto config = configWithDiskSection("age 5abc"); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, + [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); +} + +TEST(CASPartFolderValidateParse, EmptyAgeSuffixThrows) +{ + auto config = configWithDiskSection("age "); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, + [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); +} + +TEST(CASPartFolderValidateParse, UnknownValueThrows) +{ + /// Fail-closed: an unrecognized value must NEVER silently become `never`/`always`. + auto config = configWithDiskSection("sometimes"); + expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, + [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); +} + +TEST(CASPartFolderAccess, AbsenceIsNeverRetained) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); + const Cas::PartRefKey key{ns, "part_1"}; + + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); /// retained + access.dropRef(key); + EXPECT_EQ(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); /// absent: nullptr, never retained + + /// Re-publish under the SAME ref name: immediately visible, no stale absence remembered. + publishPart(store, ns, "part_1", {inlineEntry("f", "y")}); + auto view = access.getView(key, Cas::Freshness::CachedForLoad); + ASSERT_NE(view, nullptr); + EXPECT_EQ(view->inlineBytes("f"), std::optional("y")); +} + +/// Task 23 (URF plan phase 7): `getView` emits a `RefResolve` audit event only when the access does +/// real resolve work -- a warm `CachedForLoad` hit whose retained view already matches the fresh +/// resolve serves the call with no new information, so it must add no row. `resolveRef` itself defers +/// the emit on this call path (`ResolveAudit::Deferred`, `CachedPartFolderAccess::resolve`), and +/// `getView` re-emits the identical event on every OTHER path -- cold builds and `ForceFresh`. +TEST(CASPartFolderAccess, GetViewEmitsRefResolveOnlyOnRealResolveWork) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + + std::vector seen; + store->setEventSink([&](const Cas::CasEvent & e) { seen.push_back(e); }); + Cas::CachedPartFolderAccess access(store, cacheOn()); /// retention on, validate == Always (default) + + const auto refResolveCount = [&] + { + return std::count_if(seen.begin(), seen.end(), + [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::RefResolve; }); + }; + + /// Cold CachedForLoad build: real resolve work -> exactly one RefResolve. + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + EXPECT_EQ(refResolveCount(), 1); + + /// Warm hit: the retained view still matches the fresh resolve, so this call serves the SAME + /// manifest with no new information -- before this fix it would emit a SECOND RefResolve + /// (resolveRef emitted unconditionally); after the fix it must add none. + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + EXPECT_EQ(refResolveCount(), 1) << "a warm view-cache hit must not add a RefResolve row"; + + /// ForceFresh always re-proves the manifest body under the default Always validation policy, so + /// this is real resolve work again -> +1. + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); + EXPECT_EQ(refResolveCount(), 2); + + store->setEventSink(nullptr); +} + +TEST(CASPartFolderAccess, OversizedViewServedNotRetained) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + /// max_entry_bytes = 1: every real view (>= the 256-byte fixed overhead alone) is oversized. + Cas::CachedPartFolderAccess access(store, + Cas::CachedPartFolderAccess::CacheParams{ + .cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 1, + .explain_enabled = true, .validate = {}}); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + auto view1 = access.getView(key, Cas::Freshness::CachedForLoad); + ASSERT_NE(view1, nullptr); + EXPECT_FALSE(access.explain(key).retained); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::OversizedBypass); + + const uint64_t head_before = backend->headCount(manifest_key); + auto view2 = access.getView(key, Cas::Freshness::CachedForLoad); + ASSERT_NE(view2, nullptr); + EXPECT_GT(backend->headCount(manifest_key), head_before); /// not retained: re-HEADs every call + EXPECT_FALSE(access.explain(key).retained); +} + +TEST(CASPartFolderAccess, DisabledModeKeepsBaseline) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + /// CacheParams{} (cache_bytes == 0): the explicit disable switch, same as the single-arg ctor. + Cas::CachedPartFolderAccess access(store, Cas::CachedPartFolderAccess::CacheParams{}); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + backend->resetCounts(); + constexpr int n = 5; + for (int i = 0; i < n; ++i) + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + + /// Exactly the Phase-3 baseline: bytes=0 restores the no-retention call graph byte-for-byte. + EXPECT_EQ(backend->getCount(manifest_key), 1u); + EXPECT_EQ(backend->headCount(manifest_key), static_cast(n)); + EXPECT_FALSE(access.explain(key).retained); +} + +TEST(CASPartFolderAccess, SingleFlightColdBuild) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::Layout layout("p"); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); + const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); + + backend->resetCounts(); + constexpr int k = 8; + std::latch start_gate(k); + std::vector threads; + std::vector> results(k); + for (int i = 0; i < k; ++i) + threads.emplace_back([&, i] + { + start_gate.arrive_and_wait(); + results[i] = access.getView(key, Cas::Freshness::CachedForLoad); + }); + for (auto & t : threads) + t.join(); + + for (const auto & r : results) + EXPECT_NE(r, nullptr); + EXPECT_EQ(backend->getCount(manifest_key), 1u); /// single-flight: ONE body GET for the burst +} + +TEST(CASPartFolderAccess, DropNamespaceErasesAllViews) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + /// Review C2: deliberately NOT pinned -- `ns` gets a REAL, random catalog incarnation from + /// `publishPart` below, which is what this test drives production's namespace-drop/recreate + /// terminal snapshot or retirement checkpoint at. Pinning + /// it to the sentinel would make production's real-incarnation path untested by the one test that + /// exercises it end-to-end (the exact gap C2 named). + Cas::CachedPartFolderAccess access(store, cacheOn()); + publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); + publishPart(store, ns, "part_2", {inlineEntry("f", "y")}); + const Cas::PartRefKey key1{ns, "part_1"}; + const Cas::PartRefKey key2{ns, "part_2"}; + + ASSERT_NE(access.getView(key1, Cas::Freshness::CachedForLoad), nullptr); /// retained + ASSERT_NE(access.getView(key2, Cas::Freshness::CachedForLoad), nullptr); /// retained + EXPECT_TRUE(access.explain(key1).retained); + EXPECT_TRUE(access.explain(key2).retained); + + access.dropNamespace(ns); + + /// dropNamespace removes the namespace via the ref-log `remove_namespace` transaction AND erases every + /// cached view: the dropped entries must not masquerade as "retained", and no stale key1/key2 view may + /// be served. + EXPECT_FALSE(access.explain(key1).retained); + EXPECT_FALSE(access.explain(key2).retained); /// dropped too, even though never re-touched + + /// A fresh getView on the removed namespace is a COLD MISS (nullptr) -- never a stale hit on the + /// dropped manifest. A residual retained entry would instead be served here without ever going through + /// validate-on-hit, exactly the masquerade this guards against. + EXPECT_EQ(access.getView(key1, Cas::Freshness::CachedForLoad), nullptr); + EXPECT_EQ(access.getView(key2, Cas::Freshness::CachedForLoad), nullptr); + +} + +TEST(CASPartFolderAccess, BestEffortRollbackDropCountsAndSurvivesABackendOutage) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + Cas::CachedPartFolderAccess access(store, cacheOn()); + + const Cas::RootNamespace ns_a{"srv/ta"}; + const Cas::RootNamespace ns_b{"srv/tb"}; + publishPart(store, ns_a, "part_a", {inlineEntry("checksums.txt", "cs")}); + publishPart(store, ns_b, "part_b", {inlineEntry("checksums.txt", "cs")}); + + backend->armed = true; + /// Sanity: with the backend armed, a real dropRef propagates (so the fault reaches the catch). + EXPECT_ANY_THROW(store->dropRef(ns_a, "part_a")); + + using ProfileEvents::global_counters; + const auto before = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed].load(); + /// The compensating-rollback path must NOT throw (noexcept) and MUST record the swallowed failure. + access.dropRefBestEffort(Cas::PartRefKey{ns_b, "part_b"}); + const auto after = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed].load(); + EXPECT_EQ(after, before + 1); + + backend->armed = false; /// let store teardown release its lease cleanly +} + +namespace +{ + +/// A pool whose ref lane makes ONE attempt per append. That is what turns a single lost-response fault +/// into a conclusive `Unresolved`: with retries allowed the controller's resolve-before-reissue would +/// settle the ambiguity inside the same attempt and the lane would never wedge. Same budget shape, and +/// the same reason, as `gtest_cas_ref_install_safety.cpp`'s `openPoolSingleAttempt`. +Cas::PoolPtr openPoolSingleAttempt(const std::shared_ptr & backend) +{ + Cas::PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; + Cas::CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + return Cas::Pool::open(backend, cfg); +} + +} + +/// Part B review, MAJOR 3a: a promote whose ref-log append did not resolve MUST NOT be reported as +/// "nothing was committed". +/// +/// `PreparedRelinkOverPartWrite::promote` maps a `NETWORK_ERROR` to `MechanismFallbackAllowed`, which +/// tells the interserver receiver to fetch the part's bytes from the same sender instead. That is sound +/// only when the promote is PROVEN not to have committed. It is not proven here: the promotion object +/// landed and only its acknowledgement was lost, so the ref below IS committed while `promote` reports +/// failure -- and a byte fetch on top of it is a sequential double publication of one logical fetch. +/// +/// The transaction therefore records the distinction where it is knowable (around its own append) +/// rather than leaving it to be guessed from an error code, which cannot carry it: the SAME +/// `NETWORK_ERROR` is raised by a promote rejected before the append (proof of the negative) and by one +/// whose append never resolved. +TEST(CASPartFolderAccess, AnUnresolvedPromoteIsNotReportedAsDefinitelyNotCommitted) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const Cas::RootNamespace ns{"srv/t1"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + Cas::CachedPartFolderAccess access(store, cacheOn()); + const Cas::PartRefKey key{ns, "part_1"}; + + auto prepared = access.prepareEntries(key, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + ASSERT_FALSE(prepared.commitIsUnresolved()) << "no promote has been attempted yet"; + + /// The promotion's own ref-log object lands; only the acknowledgement, and the controller's + /// verifying read, are lost. Scoped to this namespace's ref log so nothing else consumes the fault. + backend->fault_substr = store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)) + "_log/"; + backend->mode = Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { prepared.promote(); }); + + EXPECT_TRUE(prepared.commitIsUnresolved()) + << "a promote whose append may have landed must not be classified as a mechanism failure -- the " + "receiver would fetch the bytes and publish the same part a second time"; + + /// The hazard itself, stated as an assertion: the promote DID commit. Any further append into this + /// table resolves the wedge first, which is what makes the committed row visible. + backend->mode = Cas::tests::ChunkFaultBackend::Mode::None; + access.prepareEntries({ns, "flush_driver"}, {inlineEntry("f", "two")}, Cas::ProvenanceOp::Insert).abort(); + EXPECT_TRUE(access.existsRef(key, Cas::Freshness::ForceFresh)) + << "the promotion object landed, so 'the promote failed' says nothing about the ref"; +} + +/// Part B review, MAJOR 3b: nothing after a durable commit may throw before the handle records it. +/// +/// `promoteBuild` used to assemble its `CommitOutcome` -- two `String` copies -- and invalidate the +/// cached view AFTER the durable append and BEFORE `PreparedPartWrite::promote` set `terminal`. An +/// allocation failure in that window therefore entered the failed-promote catch with the ref already +/// committed, where the handle abandons its build and reports the promote as failed. The outcome's +/// strings are now copied BEFORE the append and the commit is recorded in an allocation-free region +/// immediately after it, so the window is empty by construction; the probe below fires just past it. +TEST(CASPartFolderAccess, APostCommitFailureLeavesTheHandleTerminal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Cas::RootNamespace ns{"srv/t1"}; + Cas::CachedPartFolderAccess access(store, cacheOn()); + const Cas::PartRefKey key{ns, "part_1"}; + + auto prepared = access.prepareEntries(key, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); + + std::vector seen; + store->setEventSink([&](const Cas::CasEvent & e) { seen.push_back(e); }); + + /// `MEMORY_LIMIT_EXCEEDED` -- what a tracked allocation failure actually raises -- and deliberately + /// not `LOGICAL_ERROR`, which aborts at construction in debug/sanitizer builds. + access.setPostCommitProbeForTest([] + { + throw Exception(ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure in the post-commit work of promoteBuild"); + }); + expectThrowsCode(ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { prepared.promote(); }); + access.setPostCommitProbeForTest(nullptr); + store->setEventSink(nullptr); + + EXPECT_TRUE(prepared.isTerminal()) + << "the commit is durable, so the handle owes nothing"; + EXPECT_TRUE(access.existsRef(key, Cas::Freshness::ForceFresh)) << "the promote really did commit"; + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); + + /// The discriminating assertion. `isTerminal` alone is not one: the old code reached the catch, + /// abandoned an ALREADY PROMOTED build -- which succeeds, because a promoted build no longer owes a + /// precommit removal -- and so ended up terminal too, by accident. What the abandon leaves behind is + /// the audit trail of a publish that is reported as thrown away while its ref is committed. + const auto build_aborts = std::count_if(seen.begin(), seen.end(), + [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::BuildAbort; }); + EXPECT_EQ(build_aborts, 0) + << "a build whose promote is DURABLE was abandoned by the failed-promote catch: the handle had " + "not yet recorded the commit when the post-commit work threw"; + EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::BuildPublish; }), 1); +} + +/// Part B review, MAJOR 4: move ASSIGNMENT is deleted rather than implemented. +/// +/// It cannot be implemented correctly. Overwriting a handle that still owes a terminal must first +/// discharge that duty, and `abandon` appends through the ref lane, so it can FAIL -- which a move +/// assignment has no way to report. The old implementation overwrote the destination's build even when +/// `abandonBuildBestEffort` returned false, permanently dropping a cleanup owner: a live-epoch precommit +/// that no sweep and no GC ever reclaims. Nothing needs the operator (the interserver relink's handle is +/// move CONSTRUCTED into place), and a contract that cannot be relied on is worse than none. +TEST(CASPartFolderAccess, PreparedPartWriteIsNotMoveAssignable) +{ + EXPECT_FALSE(std::is_move_assignable_v) + << "a move assignment cannot discharge a terminal duty that may fail to be discharged"; + EXPECT_TRUE(std::is_move_constructible_v); +} diff --git a/src/Disks/tests/gtest_cas_part_folder_view.cpp b/src/Disks/tests/gtest_cas_part_folder_view.cpp new file mode 100644 index 000000000000..be9b9e91de91 --- /dev/null +++ b/src/Disks/tests/gtest_cas_part_folder_view.cpp @@ -0,0 +1,105 @@ +#include +#include + +using namespace DB; + +TEST(CASPartRefKey, CacheKeyIsUnambiguous) +{ + /// Refs may contain '/' (the `detached/` fold, B181); the '\0' join keeps + /// (ns="a", ref="b/c") distinct from (ns="a/b", ref="c"). + const Cas::PartRefKey k1{Cas::RootNamespace{"a"}, "b/c"}; + const Cas::PartRefKey k2{Cas::RootNamespace{"a/b"}, "c"}; + EXPECT_NE(k1.cacheKey(), k2.cacheKey()); + EXPECT_FALSE(k1 == k2); + EXPECT_TRUE((k1 == Cas::PartRefKey{Cas::RootNamespace{"a"}, "b/c"})); +} + +#include +#include + +namespace +{ + +using namespace DB; + +std::shared_ptr makeView() +{ + auto manifest = std::make_shared(); + auto add = [&](const char * path, Cas::EntryPlacement placement, const char * bytes, uint64_t blob_size) + { + Cas::ManifestEntry e; + e.path = path; + e.placement = placement; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(UInt128(manifest->entries.size() + 1))}; + + e.blob_size = blob_size; + e.inline_bytes = bytes; + manifest->entries.push_back(e); + }; + /// Canonical (sorted) order — the ctor chasserts it. All-tree-part-files Task 9: `txn_version.txt` + /// is an ordinary Inline entry now, not a separate mutable payload. + add("checksums.txt", Cas::EntryPlacement::Inline, "cs", 2); + add("data.bin", Cas::EntryPlacement::Blob, "", 100); + add("p.proj/checksums.txt", Cas::EntryPlacement::Inline, "pc", 2); + add("p.proj/data.bin", Cas::EntryPlacement::Blob, "", 50); + add("txn_version.txt", Cas::EntryPlacement::Inline, "ver", 3); + + return std::make_shared( + Cas::PartRefKey{Cas::RootNamespace{"srv/t"}, "part_1"}, + Cas::ManifestId{Cas::RootNamespace{"srv/t"}, Cas::ManifestRef{1, 2, 3}}, + /*manifest_size=*/1000, manifest, + /*validated_at_ms=*/42); +} + +std::vector sorted(std::vector v) { std::sort(v.begin(), v.end()); return v; } + +} + +TEST(CASPartFolderView, FindFileAndHasFile) +{ + auto v = makeView(); + ASSERT_NE(v->findFile("data.bin"), nullptr); + EXPECT_EQ(v->findFile("data.bin")->blob_size, 100u); + EXPECT_EQ(v->findFile("absent.bin"), nullptr); + EXPECT_TRUE(v->hasFile("p.proj/data.bin")); + EXPECT_TRUE(v->hasFile("txn_version.txt")); /// an ordinary Inline entry + EXPECT_FALSE(v->hasFile("p.proj")); /// a directory, not a file +} + +TEST(CASPartFolderView, ListChildrenCollapsesFirstComponent) +{ + auto v = makeView(); + EXPECT_EQ(sorted(v->listChildren("")), + sorted({"checksums.txt", "data.bin", "p.proj", "txn_version.txt"})); + EXPECT_EQ(sorted(v->listChildren("p.proj/")), sorted({"checksums.txt", "data.bin"})); + EXPECT_TRUE(v->listChildren("q.proj/").empty()); +} + +TEST(CASPartFolderView, HasDirectory) +{ + auto v = makeView(); + EXPECT_TRUE(v->hasDirectory("p.proj/")); + EXPECT_FALSE(v->hasDirectory("q.proj/")); +} + +TEST(CASPartFolderView, SizesAndBytes) +{ + auto v = makeView(); + EXPECT_EQ(v->fileSize("checksums.txt"), std::optional(2)); /// inline: bytes size + EXPECT_EQ(v->fileSize("data.bin"), std::optional(100)); /// blob: blob_size + EXPECT_EQ(v->fileSize("txn_version.txt"), std::optional(3)); /// inline: bytes size + EXPECT_EQ(v->fileSize("absent"), std::nullopt); + EXPECT_EQ(v->inlineBytes("checksums.txt"), std::optional("cs")); + EXPECT_EQ(v->inlineBytes("data.bin"), std::nullopt); /// blob has no inline bytes + EXPECT_EQ(v->inlineBytes("txn_version.txt"), std::optional("ver")); + EXPECT_GE(v->estimatedBytes(), 1000u); /// >= manifest_size +} + +TEST(CASPartFolderView, ProjectionDirPrefixRecognizer) +{ + using V = Cas::PartFolderView; + EXPECT_EQ(V::projectionDirPrefix("p.proj"), std::optional("p.proj/")); + EXPECT_EQ(V::projectionDirPrefix("a/b.tmp_proj"), std::optional("a/b.tmp_proj/")); + EXPECT_EQ(V::projectionDirPrefix("data.bin"), std::nullopt); + EXPECT_EQ(V::projectionDirPrefix(""), std::nullopt); +} diff --git a/src/Disks/tests/gtest_cas_part_manifest_format.cpp b/src/Disks/tests/gtest_cas_part_manifest_format.cpp new file mode 100644 index 000000000000..abb81c1928aa --- /dev/null +++ b/src/Disks/tests/gtest_cas_part_manifest_format.cpp @@ -0,0 +1,523 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ + +/// NOT `Disks/tests/cas_test_helpers.h`'s `DB::Cas::tests::expectThrowsCode`: pulling in that header +/// drags along a large chunk of the CAS backend/store machinery this file has no other need for, so it +/// stays clear of `cas_test_helpers.h` entirely and inlines its own copy of the same tiny assertion +/// instead. +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} + +/// One Blob + one Inline entry, matching the plan's §text-shape illustration verbatim (codecs-v3 +/// phase 6): deliberately NOT path-sorted on input, so the round trip also exercises canonical +/// path-order encoding. +PartManifest sample() +{ + PartManifest m; + m.ref = ManifestRef{5, 15, 1}; + m.root_namespace_id = RootNamespace("00/aa@cas@"); + + ManifestEntry inl; + inl.path = "c/small.txt"; + inl.placement = EntryPlacement::Inline; + inl.inline_bytes = "hello world!"; /// 12 raw bytes, no embedded '\n' + + ManifestEntry blob; + blob.path = "a/b.bin"; + blob.placement = EntryPlacement::Blob; + blob.ref = BlobRef{BlobHashAlgo::CityHash128, codecFor(BlobHashAlgo::CityHash128).fromHex("00112233445566778899aabbccddeeff")}; + blob.blob_size = 4096; + + m.entries = {inl, blob}; /// deliberately out of canonical order + /// Set LAST, after all other fields (matches gtest_cas_manifest_codec.cpp's + /// makeTwoEntryManifestForOrderTest): decode now recomputes + verifies this, so a placeholder + /// value here would make every test that round-trips `sample()` through decode fail closed. + m.payload_digest = computePayloadDigest(m); + return m; +} + +} + +TEST(CASFormatBattery, PartManifest) +{ + const PartManifest m = sample(); + /// Interpolate the REAL digest (never hand-compute a CityHash128 hex by hand) so the golden text + /// stays self-consistent with whatever sample() produces, now that decode verifies payload_digest. + const String golden = + currentFormatHeader("cas_part_manifest") + + "{\"me\":\"5\",\"mb\":\"15\",\"mo\":1,\"ns\":\"00/aa@cas@\",\"pd\":\"" + u128ToHex(m.payload_digest) + "\"}\n" // NOLINT(modernize-raw-string-literal): mixes '\"' quoting with '\n' line endings across this concatenated literal; a raw string can't hold the newline as-is. + "{\"p\":\"a/b.bin\",\"pm\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\",\"sz\":4096}\n" + "{\"p\":\"c/small.txt\",\"pm\":\"inline\",\"il\":12}\n" + "{\"n\":2}\n" + "==> \"c/small.txt\" il=12 <==\n" + "hello world!\n"; + runFormatBattery({FormatId::PartManifest, + [&] { return sealObject(FormatId::PartManifest, encodePartManifest(m)); }, + [](std::string_view d) { decodePartManifest(std::string(openObject(FormatId::PartManifest, d))); }, + golden}); +} + +TEST(CASPartManifestFormat, RoundTripDescriptorAndEntries) +{ + const PartManifest m = sample(); + const PartManifest got = decodePartManifest(encodePartManifest(m)); + EXPECT_EQ(got.ref, m.ref); + EXPECT_EQ(got.root_namespace_id, m.root_namespace_id); + EXPECT_EQ(got.payload_digest, m.payload_digest); + ASSERT_EQ(got.entries.size(), 2u); + + /// canonical path order: "a/b.bin" < "c/small.txt" + EXPECT_EQ(got.entries[0].path, "a/b.bin"); + EXPECT_EQ(got.entries[0].placement, EntryPlacement::Blob); + EXPECT_EQ(got.entries[0].ref, m.entries[1].ref); + EXPECT_EQ(got.entries[0].blob_size, 4096u); + + EXPECT_EQ(got.entries[1].path, "c/small.txt"); + EXPECT_EQ(got.entries[1].placement, EntryPlacement::Inline); + /// The payload-zone round trip: exact raw bytes recovered from the banner+bytes+'\n' zone. + EXPECT_EQ(got.entries[1].inline_bytes, "hello world!"); +} + +TEST(CASPartManifestFormat, EmptyEntriesRoundTrips) +{ + PartManifest m = sample(); + m.entries.clear(); + m.payload_digest = computePayloadDigest(m); /// recompute: content changed, sample()'s digest is stale + const PartManifest got = decodePartManifest(encodePartManifest(m)); + EXPECT_TRUE(got.entries.empty()); + EXPECT_EQ(got.ref, m.ref); + /// No payload zone at all when there are no Inline entries. + EXPECT_FALSE(encodePartManifest(m).contains("==>")); +} + +TEST(CASPartManifestFormat, PlacementWordsRenderAndRejectUnknown) +{ + const String text = encodePartManifest(sample()); + EXPECT_NE(text.find("\"pm\":\"blob\""), String::npos); + EXPECT_NE(text.find("\"pm\":\"inline\""), String::npos); + + /// An unknown placement word fails closed. + String bad = text; + const size_t pos = bad.find(R"("pm":"blob")"); + ASSERT_NE(pos, String::npos); + bad.replace(pos, String(R"("pm":"blob")").size(), R"("pm":"bogus")"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); +} + +/// Proves the payload zone, not JSON-string escaping: an Inline entry whose bytes contain an +/// embedded '\n', a NUL byte, and a '"' character round-trip byte-faithfully. If this content were +/// carried as a JSON string value it would need escaping (or would be flatly invalid for the NUL +/// byte); the payload zone instead carries it as raw length-delimited bytes. +TEST(CASPartManifestFormat, InlineBytesWithEmbeddedSpecialCharsRoundTripByteFaithfully) +{ + PartManifest m; + m.ref = ManifestRef{7, 21, 2}; + m.root_namespace_id = RootNamespace("00/bb@cas@"); + + ManifestEntry e; + e.path = "weird.bin"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "line1\nline2"; + e.inline_bytes.push_back('\0'); + e.inline_bytes += "after-nul\"quoted\"end"; + m.entries = {e}; + m.payload_digest = computePayloadDigest(m); + + const PartManifest got = decodePartManifest(encodePartManifest(m)); + ASSERT_EQ(got.entries.size(), 1u); + EXPECT_EQ(got.entries[0].inline_bytes, m.entries[0].inline_bytes); + EXPECT_EQ(got.entries[0].inline_bytes.size(), e.inline_bytes.size()); +} + +/// The path is written twice: escaped into the entry-record line, and -- before this fix -- raw into the +/// payload-zone banner. Only a byte that breaks the banner's physical line framing actually corrupts the +/// object, which today means LF alone; the rest of these cases pin the round trip so a future escaping +/// change cannot quietly start mangling them. +TEST(CASPartManifestFormat, InlineEntryPathSurvivesEveryEscapableByte) +{ + const std::vector paths{ + String("p\nq.proj/columns.txt"), /// the reported reproducer: LF splits the banner line + String("a\rb.txt"), + String("tab\there.txt"), + String("quote\"and\\slash.txt"), + String("nul\0byte.txt", 12), /// length-explicit, or the NUL is lost to the terminator + }; + for (const String & path : paths) + { + SCOPED_TRACE(path); + PartManifest m; + m.ref = ManifestRef{17, 66, 7}; + m.root_namespace_id = RootNamespace("00/ff@cas@"); + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "hello world!"; + m.entries = {e}; + m.payload_digest = computePayloadDigest(m); + + const PartManifest got = decodePartManifest(encodePartManifest(m)); + ASSERT_EQ(got.entries.size(), 1u); + EXPECT_EQ(got.entries[0].path, path); + EXPECT_EQ(got.entries[0].inline_bytes, "hello world!"); + } +} + +/// The banner quotes and escapes the path with the SAME writer the entry-record line uses. Pin the byte +/// shape, so a future hand-rolled escaper here cannot silently diverge from the record line again. +TEST(CASPartManifestFormat, InlineBannerCarriesTheEscapedPath) +{ + PartManifest m; + m.ref = ManifestRef{17, 66, 7}; + m.root_namespace_id = RootNamespace("00/ff@cas@"); + ManifestEntry e; + e.path = "p\nq.proj/c.txt"; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "x"; + m.entries = {e}; + m.payload_digest = computePayloadDigest(m); + + EXPECT_NE(encodePartManifest(m).find("==> \"p\\nq.proj/c.txt\" il=1 <=="), String::npos); +} + +TEST(CASPartManifestFormat, ByteDeterminism) +{ + const PartManifest m = sample(); + /// Encode twice -> identical bytes. Also encode a copy with entries pre-shuffled into the other + /// order -> still identical, because the encoder sorts canonically. + PartManifest m2 = m; + std::swap(m2.entries[0], m2.entries[1]); + EXPECT_EQ(encodePartManifest(m), encodePartManifest(m)); + EXPECT_EQ(encodePartManifest(m), encodePartManifest(m2)); +} + +TEST(CASPartManifestFormat, MixedAlgoEntriesRoundTrip) +{ + PartManifest m; + m.ref = ManifestRef{9, 33, 4}; + m.root_namespace_id = RootNamespace("00/cc@cas@"); + + ManifestEntry e16; + e16.path = "a/ch128.bin"; + e16.placement = EntryPlacement::Blob; + e16.ref = BlobRef{BlobHashAlgo::CityHash128, + codecFor(BlobHashAlgo::CityHash128).fromHex("00112233445566778899aabbccddeeff")}; + e16.blob_size = 100; + + ManifestEntry e32; + e32.path = "b/sha256.bin"; + e32.placement = EntryPlacement::Blob; + e32.ref = BlobRef{BlobHashAlgo::Sha256, codecFor(BlobHashAlgo::Sha256).fromHex(String(64, 'a'))}; + e32.blob_size = 200; + + m.entries = {e16, e32}; + m.payload_digest = computePayloadDigest(m); + + const PartManifest got = decodePartManifest(encodePartManifest(m)); + ASSERT_EQ(got.entries.size(), 2u); + EXPECT_EQ(got.entries[0].path, "a/ch128.bin"); + EXPECT_EQ(got.entries[0].ref, e16.ref); + EXPECT_EQ(got.entries[0].blob_size, 100u); + EXPECT_EQ(got.entries[1].path, "b/sha256.bin"); + EXPECT_EQ(got.entries[1].ref, e32.ref); + EXPECT_EQ(got.entries[1].blob_size, 200u); +} + +/// Builds a single-Blob-entry manifest whose entry path is exactly `path` -- `encodePartManifest` +/// itself does not validate path shape (only ordering/duplicates), so this lets the negative cases +/// below reach `decodePartManifest`'s shape check unobstructed. +static PartManifest manifestWithSinglePath(std::string_view path) +{ + PartManifest m; + m.ref = ManifestRef{17, 66, 7}; + m.root_namespace_id = RootNamespace("00/ff@cas@"); + + ManifestEntry e; + e.path = String(path); + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, + codecFor(BlobHashAlgo::CityHash128).fromHex("00112233445566778899aabbccddeeff")}; + e.blob_size = 10; + m.entries = {e}; + m.payload_digest = computePayloadDigest(m); + return m; +} + +/// T11: manifest bytes arrive over the interserver relink channel, so decode enforces the same path +/// hygiene as CasLayout::checkNamespace -- relative, no empty/'.'/'..' segments, no leading '/'. +/// `encodePartManifest` does not itself reject these (see `manifestWithSinglePath`), so each case +/// must fail closed at decode time instead. +TEST(CASPartManifestFormat, DecodeRejectsMalformedEntryPaths) +{ + for (const char * path : {"../evil", "/abs", "", "a//b", "a/./b"}) + { + SCOPED_TRACE(path); + const String encoded = encodePartManifest(manifestWithSinglePath(path)); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(encoded); }); + } +} + +/// Legal projection subdirectories (`.proj/`) must not be caught by the shape +/// check above -- it is syntactic only, not a directory-depth restriction. +TEST(CASPartManifestFormat, DecodeAcceptsLegalProjectionSubdirPath) +{ + const PartManifest m = manifestWithSinglePath("proj.proj/data.bin"); + const PartManifest got = decodePartManifest(encodePartManifest(m)); + ASSERT_EQ(got.entries.size(), 1u); + EXPECT_EQ(got.entries[0].path, "proj.proj/data.bin"); +} + +TEST(CASPartManifestFormat, DuplicatePathRejectedOnEncode) +{ + PartManifest m = sample(); + ManifestEntry dup = m.entries[0]; /// same path as an existing entry + m.entries.push_back(dup); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodePartManifest(m); }); +} + +/// Hand-forge two valid entry-record LINES swapped out of canonical order (no CRC-patching forge +/// helpers needed - this is a text format, lines carry no per-line checksum). Both entries are Blob +/// (no payload-zone bytes), so the swap cannot disturb payload-zone alignment - it isolates exactly +/// the ordering check. +TEST(CASPartManifestFormat, DecodeRejectsOutOfOrderEntries) +{ + PartManifest m; + m.ref = ManifestRef{11, 44, 5}; + m.root_namespace_id = RootNamespace("00/dd@cas@"); + + auto mkBlob = [](std::string_view path) + { + ManifestEntry e; + e.path = String(path); + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, + codecFor(BlobHashAlgo::CityHash128).fromHex("00112233445566778899aabbccddeeff")}; + e.blob_size = 10; + return e; + }; + /// "a/one.bin" and "b/two.bin" are the same length, so swapping their record lines in place + /// does not shift any other byte offset in the text. + m.entries = {mkBlob("a/one.bin"), mkBlob("b/two.bin"), mkBlob("c/three.bin")}; + m.payload_digest = computePayloadDigest(m); + + const String text = encodePartManifest(m); + const size_t pos_a = text.find(R"("p":"a/one.bin")"); + const size_t pos_b = text.find(R"("p":"b/two.bin")"); + ASSERT_NE(pos_a, String::npos); + ASSERT_NE(pos_b, String::npos); + + const size_t a_start = text.rfind('\n', pos_a) + 1; + const size_t a_end = text.find('\n', pos_a) + 1; + const size_t b_start = text.rfind('\n', pos_b) + 1; + const size_t b_end = text.find('\n', pos_b) + 1; + const String a_line = text.substr(a_start, a_end - a_start); + const String b_line = text.substr(b_start, b_end - b_start); + ASSERT_EQ(a_line.size(), b_line.size()); + + String forged = text; + forged.replace(a_start, a_line.size(), b_line); + forged.replace(b_start, b_line.size(), a_line); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(forged); }); +} + +/// a < b < c in canonical order; forge entry c's path to equal entry a's path. A naive "only check +/// adjacent pairs" implementation would miss this (c is only ever compared against b, never against +/// a); requiring strict ascending order against just the immediately-preceding entry still catches +/// it, because the forged c(=a's path) is no longer greater than b either. +TEST(CASPartManifestFormat, DecodeRejectsNonAdjacentDuplicatePath) +{ + PartManifest m; + m.ref = ManifestRef{13, 55, 6}; + m.root_namespace_id = RootNamespace("00/ee@cas@"); + + auto mkBlob = [](std::string_view path) + { + ManifestEntry e; + e.path = String(path); + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, + codecFor(BlobHashAlgo::CityHash128).fromHex("00112233445566778899aabbccddeeff")}; + e.blob_size = 10; + return e; + }; + m.entries = {mkBlob("aaa/one.bin"), mkBlob("bbb/two.bin"), mkBlob("ccc/three.bin")}; + m.payload_digest = computePayloadDigest(m); + + String forged = encodePartManifest(m); + const String needle = R"("p":"ccc/three.bin")"; + const size_t pos = forged.find(needle); + ASSERT_NE(pos, String::npos); + forged.replace(pos, needle.size(), R"("p":"aaa/one.bin")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(forged); }); +} + +TEST(CASPartManifestFormat, UnknownEntryAlgoFailsClosed) +{ + String bad = encodePartManifest(sample()); + const String needle = R"("ha":"ch128")"; + const size_t pos = bad.find(needle); + ASSERT_NE(pos, String::npos); + bad.replace(pos, needle.size(), R"("ha":"bogus")"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); +} + +/// `DigestCodec::fromHex` throws BAD_ARGUMENTS (not CORRUPTED_DATA) on a width mismatch; decode must +/// check the width itself first so this fails closed with the same code every other decode error +/// here uses. +TEST(CASPartManifestFormat, DigestHexWidthMismatchFailsClosedNotBadArguments) +{ + String bad = encodePartManifest(sample()); + const String key = R"("h":")"; + const size_t key_pos = bad.find(key); + ASSERT_NE(key_pos, String::npos); + const size_t hex_start = key_pos + key.size(); + const size_t hex_end = bad.find('"', hex_start); + ASSERT_NE(hex_end, String::npos); + ASSERT_EQ(hex_end - hex_start, 32u); /// ch128: 16-byte digest -> 32 hex chars + bad.erase(hex_start, 1); /// drop one hex char -> width mismatch (31 chars) + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); +} + +/// Pure-function properties of computePayloadDigest, independent of decode-time verification: stable +/// across calls for identical content, independent of the payload_digest field's own value, and +/// content-sensitive (changes when real content changes). +TEST(CASPartManifestFormat, PayloadDigestStableAndContentSensitive) +{ + const PartManifest m = sample(); + PartManifest with_different_stored_digest = m; + with_different_stored_digest.payload_digest = UInt128(0x1234); + EXPECT_EQ(computePayloadDigest(m), computePayloadDigest(m)); + EXPECT_EQ(computePayloadDigest(m), computePayloadDigest(with_different_stored_digest)); + + /// m.entries[1] is the Blob entry (m.entries[0] is Inline, whose blob_size is unused on the + /// wire) - changing its blob_size changes the canonical encoding and therefore the digest. + ASSERT_EQ(m.entries[1].placement, EntryPlacement::Blob); + PartManifest changed = m; + changed.entries[1].blob_size += 1; + EXPECT_NE(computePayloadDigest(m), computePayloadDigest(changed)); +} + +/// No-smuggling: one extra trailing byte after the last payload-zone segment (or after the trailer, +/// when there are no Inline entries) must be rejected - exercises the final `!in.eof()` check. +TEST(CASPartManifestFormat, TrailingByteAfterPayloadZoneFailsClosed) +{ + String bad = encodePartManifest(sample()); + bad += "X"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); +} + +/// An Inline entry's record "il" disagrees with what the payload zone's banner+bytes actually +/// declare (the banner and bytes are left as originally written; only the record line's "il" is +/// edited). The record's declared `il` is what decode uses both to build the expected banner text +/// and to know how many bytes to read from the zone, so this must fail closed rather than silently +/// reading the wrong byte count. +TEST(CASPartManifestFormat, InlineRecordIlMismatchWithPayloadZoneBannerFailsClosed) +{ + String bad = encodePartManifest(sample()); + const String needle = "\"il\":12"; + const size_t pos = bad.find(needle); + ASSERT_NE(pos, String::npos); + bad.replace(pos, needle.size(), "\"il\":13"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); +} + +/// ==== migrated from gtest_cas_manifest_codec.cpp (deleted in the phase-6 binary->text cutover, +/// Task 3): these exercise refMatchesBody/manifestNamespaceMatches/findEntry/entryRange, pure +/// functions carried over verbatim from the retired binary codec (untouched by the wire-shape +/// migration) — reusing this file's own sample() fixture instead of reintroducing a second one. ==== + +TEST(CASPartManifestFormat, RefMatchesBodyAcceptsExactRef) +{ + const PartManifest m = sample(); + /// The journal ref equals the body ref -> true. + EXPECT_TRUE(refMatchesBody(m.ref, m)); +} + +TEST(CASPartManifestFormat, RefMatchesBodyRejectsEachFieldMismatch) +{ + const PartManifest m = sample(); + ManifestRef wrong_writer = m.ref; wrong_writer.writer_epoch = m.ref.writer_epoch + 1; + ManifestRef wrong_seq = m.ref; wrong_seq.build_sequence = m.ref.build_sequence + 1; + ManifestRef wrong_inst = m.ref; wrong_inst.manifest_ordinal = m.ref.manifest_ordinal + 1; + EXPECT_FALSE(refMatchesBody(wrong_writer, m)); + EXPECT_FALSE(refMatchesBody(wrong_seq, m)); + EXPECT_FALSE(refMatchesBody(wrong_inst, m)); +} + +TEST(CASPartManifestFormat, ManifestNamespaceMatchesAcceptsOwningNs) +{ + const PartManifest m = sample(); + EXPECT_TRUE(manifestNamespaceMatches(m.root_namespace_id, m)); +} + +TEST(CASPartManifestFormat, ManifestNamespaceMatchesRejectsForeignNs) +{ + const PartManifest m = sample(); + /// sample()'s namespace is "00/aa@cas@" — pick a genuinely foreign one and a strict-prefix one. + EXPECT_FALSE(manifestNamespaceMatches(RootNamespace("00/bb@cas@"), m)); + /// A namespace that is a prefix but not equal is still a mismatch (no loose comparison). + EXPECT_FALSE(manifestNamespaceMatches(RootNamespace("00/aa"), m)); +} + +TEST(CASPartManifestFormat, FindEntryBinarySearch) +{ + std::vector entries; + for (const char * p : {"a.txt", "b/inner.txt", "b/z.txt", "c.txt"}) + { + ManifestEntry e; + e.path = p; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "v"; + entries.push_back(e); + } + EXPECT_NE(findEntry(entries, "a.txt"), nullptr); + EXPECT_EQ(findEntry(entries, "a.txt")->path, "a.txt"); + EXPECT_NE(findEntry(entries, "c.txt"), nullptr); /// last element + EXPECT_EQ(findEntry(entries, "b"), nullptr); /// prefix of a path, not a path + EXPECT_EQ(findEntry(entries, "zzz"), nullptr); /// past the end + EXPECT_EQ(findEntry({}, "a"), nullptr); /// empty +} + +TEST(CASPartManifestFormat, EntryRangeContiguousPrefix) +{ + std::vector entries; + for (const char * p : {"a.txt", "p.proj/data.bin", "p.proj/x.txt", "q.txt"}) + { + ManifestEntry e; + e.path = p; + e.placement = EntryPlacement::Inline; + e.inline_bytes = "v"; + entries.push_back(e); + } + auto [first, last] = entryRange(entries, "p.proj/"); + ASSERT_EQ(last - first, 2); + EXPECT_EQ(first->path, "p.proj/data.bin"); + EXPECT_EQ((last - 1)->path, "p.proj/x.txt"); + + auto [w1, w2] = entryRange(entries, ""); /// empty prefix = whole span + EXPECT_EQ(w2 - w1, 4); + + auto [n1, n2] = entryRange(entries, "zzz/"); /// no match + EXPECT_EQ(n1, n2); +} diff --git a/src/Disks/tests/gtest_cas_part_write.cpp b/src/Disks/tests/gtest_cas_part_write.cpp new file mode 100644 index 000000000000..7a426676cbc8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_part_write.cpp @@ -0,0 +1,2604 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event CASMetaPut; +extern const Event CASMetaCompareSwap; +extern const Event CASMetaCreateClean; +extern const Event CASMetaAdoptBackfill; +extern const Event CASMetaResurrectClean; +extern const Event CASBlobAdoptTrusted; +} + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int FILE_DOESNT_EXIST; +extern const int INVALID_STATE; +extern const int LOGICAL_ERROR; +extern const int NOT_IMPLEMENTED; +extern const int ABORTED; +extern const int CORRUPTED_DATA; +extern const int LIMIT_EXCEEDED; +extern const int NETWORK_ERROR; +extern const int UNKNOWN_EXCEPTION; +} + +using namespace DB::Cas; +using DB::Cas::tests::condemnMeta; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::idOf; +using DB::Cas::tests::injectRetire; +using DB::Cas::tests::loadMetaForTest; +using DB::Cas::tests::streamingHexOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::writeMetaClean; +using DB::Cas::tests::writeRawBlobBody; + +namespace +{ + +PoolPtr openPool(const std::shared_ptr & b) +{ + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// Start a build whose owning manifest namespace + final ref name are `ns`/`ref` (promote/stageManifest +/// derive the manifest namespace by splitting PartWriteInfo::intended_ref on the LAST '/'). +PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + return s->beginPartWrite(info); +} + +/// A one-entry Blob ManifestEntry for `payload` at `path` (the build's stageManifest entry). +ManifestEntry blobManifestEntry(const String & path, const String & payload) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + return e; +} + +/// The streaming (production-convention) `BlobRef` of `payload` — CityHash128 at the write width. +BlobRef streamRefOf(const String & payload) +{ + return BlobRef{BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hexToU128(streamingHexOf(payload)))}; +} + +ManifestEntry blobManifestEntryStreaming(const String & path, const String & payload) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Blob; + e.ref = streamRefOf(payload); + + e.blob_size = payload.size(); + return e; +} + +PartWriteTxnPtr precommittedBuildForPayload( + const PoolPtr & store, const RootNamespace & ns, const String & ref, const String & payload) +{ + auto build = startBuildFor(store, ns, ref); + const ManifestId manifest = build->stageManifest({blobManifestEntry("data.bin", payload)}); + build->precommitAdd(ns, ref, manifest); + return build; +} + +ManifestId durablyPrecommit( + const PartWriteTxnPtr & build, + const RootNamespace & ns, + const String & ref, + std::vector entries) +{ + const ManifestId manifest = build->stageManifest(std::move(entries)); + build->precommitAdd(ns, ref, manifest); + return manifest; +} + +/// The full single-blob write flow (EDGE-BEFORE-OBSERVE wiring order): +/// stageManifest(one entry) -> precommitAdd -> putBlob -> promote. Returns the committed ManifestId. +ManifestId publishOneBlobPart( + const PoolPtr & s, const RootNamespace & ns, const String & ref, const String & path, const String & payload) +{ + auto build = startBuildFor(s, ns, ref); + const ManifestId id = build->stageManifest({blobManifestEntry(path, payload)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// A one-shot backend hook (mirrors the WriteCountingBackend delegation pattern in gtest_cas_pool.cpp): +/// it delegates every op to a wrapped Backend, but the FIRST time head(target_key) is called it fires a +/// deleteExact(target_key, condemned_token) AFTER computing the (present) HEAD result and BEFORE returning +/// it — simulating GC's exact-token content delete landing in the writer's HEAD->GET window (B136). +class HeadThenDeleteOnceBackend final : public DB::Cas::Backend +{ +public: + HeadThenDeleteOnceBackend(BackendPtr inner_, String target_key_, DB::Cas::Token condemned_) + : inner(std::move(inner_)), target_key(std::move(target_key_)), condemned(condemned_) {} + + DB::Cas::HeadResult head(const String & k) override + { + const DB::Cas::HeadResult hr = inner->head(k); + if (k == target_key && !fired) + { + fired = true; + /// GC's single content-delete site, landing in the HEAD->GET window. + inner->deleteExact(target_key, condemned); + } + return hr; + } + + std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { return inner->putIfAbsent(k, b, meta); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { return inner->putOverwrite(k, b, e, meta); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { return inner->casPut(k, b, e, meta); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + BackendPtr inner; + String target_key; + DB::Cas::Token condemned; + bool fired = false; +}; + +/// A delegating backend that counts head()/get() calls per key. Lets a test assert the promote gate +/// performs ZERO per-file probes on a TRUSTED adopted leaf (§4 manifest-trust): no presence HEAD on +/// the blob key, no loadMeta GET on the blob-meta key. +class KeyCountingBackend final : public DB::Cas::Backend +{ +public: + explicit KeyCountingBackend(BackendPtr inner_) : inner(std::move(inner_)) {} + + size_t headCountFor(const String & k) const { auto it = head_counts.find(k); return it == head_counts.end() ? 0 : it->second; } + size_t getCountFor(const String & k) const { auto it = get_counts.find(k); return it == get_counts.end() ? 0 : it->second; } + + DB::Cas::HeadResult head(const String & k) override { ++head_counts[k]; return inner->head(k); } + std::optional get(const String & k, DB::Cas::Range r) override { ++get_counts[k]; return inner->get(k, r); } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::ListPage list(const String & pfx, const String & c, size_t l) override { return inner->list(pfx, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { return inner->putIfAbsent(k, b, meta); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { return inner->putOverwrite(k, b, e, meta); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { return inner->casPut(k, b, e, meta); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + BackendPtr inner; + std::map head_counts; + std::map get_counts; +}; + +/// Forces two writers to complete their absent `HEAD` observations before either can publish. +class RacingBlobPublicationBackend final : public InMemoryBackend +{ +public: + void watch(String key_) + { + std::lock_guard lock(mutex); + key = std::move(key_); + head_calls = 0; + publish_calls = 0; + } + + HeadResult head(const String & requested_key) override + { + if (requested_key != key) + return InMemoryBackend::head(requested_key); + + const HeadResult observed = InMemoryBackend::head(requested_key); + std::unique_lock lock(mutex); + ++head_calls; + cv.notify_all(); + cv.wait_for(lock, std::chrono::seconds(5), [&] { return head_calls >= 2; }); + return observed; + } + + void publishBlob(const BlobPublishRequest & request) override + { + if (request.destination_key == key) + { + std::lock_guard lock(mutex); + ++publish_calls; + } + InMemoryBackend::publishBlob(request); + } + + String key; + std::mutex mutex; + std::condition_variable cv; + size_t head_calls = 0; + size_t publish_calls = 0; +}; + +} + +TEST(CASPartWrite, RacingWritersBothHeadMissAndPublishEquivalentBodies) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const String payload = "two-racing-mandatory-head-writers"; + const BlobRef ref = idOf(payload); + auto first = precommittedBuildForPayload(store, RootNamespace{"srv1/racing-a"}, "part", payload); + auto second = precommittedBuildForPayload(store, RootNamespace{"srv1/racing-b"}, "part", payload); + backend->watch(store->layout().blobKey(ref)); + + std::exception_ptr first_error; + std::exception_ptr second_error; + std::thread first_thread([&] + { + try + { + first->putBlob(ref, BlobSource::fromString(payload)); + } + catch (...) + { + first_error = std::current_exception(); + } + }); + std::thread second_thread([&] + { + try + { + second->putBlob(ref, BlobSource::fromString(payload)); + } + catch (...) + { + second_error = std::current_exception(); + } + }); + first_thread.join(); + second_thread.join(); + + EXPECT_EQ(first_error, nullptr); + EXPECT_EQ(second_error, nullptr); + EXPECT_EQ(backend->head_calls, 2u); + EXPECT_EQ(backend->publish_calls, 2u) + << "both equivalent writers may publish after racing absent observations"; + EXPECT_EQ(first->dependencyProof(ref), BlobDependencyProof::Materialized); + EXPECT_EQ(second->dependencyProof(ref), BlobDependencyProof::Materialized); + const auto stored = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(stored.has_value()); + EXPECT_EQ(stored->bytes.substr(store->poolMeta().blob_header_len), payload); +} + +TEST(CASPartWrite, WrongSizeSourcePublishesNothing) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const String expected_payload = "declared-eleven-bytes"; + const BlobRef ref = idOf(expected_payload); + auto build = precommittedBuildForPayload( + store, RootNamespace{"srv1/wrong-size-publication"}, "part", expected_payload); + + BlobSource source; + source.size = 11; + source.open = []() -> std::unique_ptr + { + return std::make_unique(String("short")); + }; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + build->putBlob(ref, std::move(source)); + }); + EXPECT_FALSE(backend->head(store->layout().blobKey(ref)).exists); +} + +TEST(CASPartWriteTxn, PutBlobWritesEnvelopeWithFixedHeader) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/envelope"}; + auto build = precommittedBuildForPayload(s, ns, "part", "hello world"); + auto ref = build->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + EXPECT_EQ(ref.size, 11u); + + auto raw = b->get(s->layout().blobKey(ref.ref)); + ASSERT_TRUE(raw.has_value()); + auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(h.header_len, s->poolMeta().blob_header_len); /// 256 + /// `logical_size`/`logical_hash` were dropped 2026-07-11, and `domain_id` in codecs-v3 phase 7 + /// (the pool id no longer travels in the envelope) — identity is the content key and the payload + /// starts at the fixed offset `header_len`. + EXPECT_EQ(h.build_id, build->buildId()); + EXPECT_NE(h.incarnation_tag, UInt128{}); + EXPECT_EQ(raw->bytes.substr(h.header_len), "hello world"); +} + +TEST(CASPartWriteTxn, StageManifestUsesPerBuildOrdinals) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"test/tbl@cas@"}; + + auto build = startBuildFor(s, ns, "all_1_1_0"); + const ManifestId first = build->stageManifest({blobManifestEntry("a.bin", "a")}); + const ManifestId second = build->stageManifest({blobManifestEntry("b.bin", "b")}); + + EXPECT_EQ(first.ref.writer_epoch, s->writerEpoch()); + EXPECT_EQ(first.ref.build_sequence, build->buildSeq()); + EXPECT_EQ(first.ref.manifest_ordinal, 1u); + EXPECT_EQ(second.ref.writer_epoch, first.ref.writer_epoch); + EXPECT_EQ(second.ref.build_sequence, first.ref.build_sequence); + EXPECT_EQ(second.ref.manifest_ordinal, 2u); + /// Canonical hex build directory (spec §Manifest Identifier): `-/`. + const String build_segment = renderRefTxnId(RefTxnId{s->writerEpoch(), build->buildSeq()}); + EXPECT_EQ(s->layout().manifestKey(first), "p/cas/manifests/test/tbl@cas@/" + build_segment + "/000001.zst"); + EXPECT_EQ(s->layout().manifestKey(second), "p/cas/manifests/test/tbl@cas@/" + build_segment + "/000002.zst"); + + auto next_build = startBuildFor(s, ns, "all_2_2_0"); + const ManifestId next = next_build->stageManifest({blobManifestEntry("c.bin", "c")}); + EXPECT_EQ(next.ref.writer_epoch, first.ref.writer_epoch); + EXPECT_NE(next.ref.build_sequence, first.ref.build_sequence); + EXPECT_EQ(next.ref.manifest_ordinal, 1u); +} + +/// B171: the `cas_owner` owner-triple stamping (`PartWriteTxn::ownerMeta`) was DELETED — protection is now +/// the build-root precommit edge (reachability), not revocable object metadata GC reads per-candidate. +/// The old `CASPartWriteTxn.BlobCarriesOwnerTripleInMetadata` asserted that stamping; its coverage is replaced +/// by the build-root precommit/reclaim tests (`CASPartWriteTxnRoot*`, `CASPartWriteTxnRootDangle*`), which prove a +/// written-but-unreferenced object is protected by a live precommit and collectable once it is abandoned. + +TEST(CASPartWriteTxn, PutBlobDedupSecondWriterAdopts) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + /// First writer publishes under its durable precommit edge. + auto build_a = precommittedBuildForPayload(s, RootNamespace{"srv/tbl-a"}, "ref_a", "dup"); + auto ref_a = build_a->putBlob(idOf("dup"), BlobSource::fromString("dup")); + const Token token_a = b->head(s->layout().blobKey(ref_a.ref)).token; + + /// Second writer ADOPTS — the adopt must happen under a durable precommit edge (EDGE-BEFORE-OBSERVE: + /// stageManifest -> precommitAdd -> putBlob), so give build_b the wiring order. + const RootNamespace ns_b{"srv/tbl"}; + auto build_b = startBuildFor(s, ns_b, "ref_b"); + const ManifestId id_b = build_b->stageManifest({blobManifestEntry("data.bin", "dup")}); + build_b->precommitAdd(ns_b, "ref_b", id_b); + auto ref_b = build_b->putBlob(idOf("dup"), BlobSource::fromString("dup")); + + EXPECT_EQ(ref_b.ref, ref_a.ref); + /// A's incarnation survives — the second writer adopts, nothing was overwritten. + EXPECT_EQ(b->head(s->layout().blobKey(ref_a.ref)).token, token_a); +} + +/// Task 3 (spec §meta-protocols v3): the writer's dedup gate no longer consults the RetireView for the +/// condemned decision — it point-reads the per-hash freshness meta instead. A fresh (absent -> present) +/// upload must WRITE that meta as Clean so future point-readers (other writers, GC) can see it. +TEST(CASPartWriteTxn, PutBlobFreshUploadWritesCleanMeta) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "fresh-meta-payload"; + auto build = precommittedBuildForPayload(s, RootNamespace{"srv1/fresh-meta"}, "part", payload); + auto ref = build->putBlob(idOf(payload), BlobSource::fromString(payload)); + EXPECT_EQ(ref.size, payload.size()); + + const auto lm = loadMetaForTest(*b, s->layout(), u128Of(payload)); + ASSERT_TRUE(lm.has_value()) << "a fresh upload must write a Clean meta descriptor (writer point-read protocol)"; + EXPECT_EQ(lm->meta.state, MetaState::Clean); + EXPECT_EQ(lm->meta.size, payload.size()); +} + +/// §0 introspection: a fresh body upload writes the Clean meta exactly once through the +/// `putMetaIfAbsent` choke point (`CASMetaPut`), tagged with its reason (`CASMetaCreateClean`). +TEST(CASPartWriteTxnMetaCounters, CreateCleanAndChokePointCountOnFreshBody) +{ + /// Fresh body upload writes the Clean meta exactly once: CASMetaPut +1 (choke point) + /// and CASMetaCreateClean +1 (reason). Reuse the fixture of the nearest putBlob test. + const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load(); + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean].load(); + + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "fresh-meta-payload-counters"; + auto build = precommittedBuildForPayload(s, RootNamespace{"srv1/fresh-meta-counters"}, "part", payload); + auto ref = build->putBlob(idOf(payload), BlobSource::fromString(payload)); + EXPECT_EQ(ref.size, payload.size()); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load() - put_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean].load() - reason_before, 1); +} + +/// §0 introspection: an adopt of a pre-existing body that has NO meta at all (a pre-protocol blob, or a +/// lost race with a concurrent fresh-uploader's own meta write) backfills a Clean meta through the +/// `putMetaIfAbsent` choke point (`CASMetaPut`), tagged with its reason (`CASMetaAdoptBackfill`). No +/// existing test elsewhere in the suite drives this branch: every other pre-seeded raw body in this file +/// pairs `writeRawBlobBody` with `writeMetaClean`, which skips the `!lm` backfill branch entirely. +TEST(CASPartWriteTxnMetaCounters, AdoptBackfillCountsChokePointAndReason) +{ + const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load(); + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill].load(); + + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "adopt-backfill-payload-counters"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + + /// Pre-seed a present body big enough that `ensureBlobPresent`'s logical-size guard does not underflow — + /// deliberately WITHOUT any meta (unlike PutBlobAdoptsWhenMetaCleanNoRetireView), so the adopt reaches + /// the `!lm` backfill branch. + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + + /// Adopt must happen under a durable precommit edge (EDGE-BEFORE-OBSERVE). + const RootNamespace ns{"srv/tbl"}; + auto build = startBuildFor(s, ns, "ref_adopt_backfill"); + const ManifestId manifest_id = build->stageManifest({blobManifestEntry("data.bin", payload)}); + build->precommitAdd(ns, "ref_adopt_backfill", manifest_id); + auto ref = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(ref.ref, id); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load() - put_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill].load() - reason_before, 1); + + const auto lm = loadMetaForTest(*b, s->layout(), hash); + ASSERT_TRUE(lm.has_value()) << "the adopt-backfill must leave a Clean meta for future point-readers"; + EXPECT_EQ(lm->meta.state, MetaState::Clean); +} + +/// The adopt decision is driven PURELY by the meta point-read — no RetireView is ever seeded in this +/// test. A pre-existing body plus an independent Clean meta must be adopted (no putOverwrite/re-upload: +/// the pre-seeded incarnation's token survives untouched), and the meta stays Clean. +TEST(CASPartWriteTxn, PutBlobAdoptsWhenMetaCleanNoRetireView) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "adopt-meta-payload"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const String blob_key = s->layout().blobKey(id); + + /// Pre-seed a body big enough that `ensureBlobPresent`'s logical-size guard (hr.size - header_len) + /// does not underflow, plus an INDEPENDENT Clean meta — deliberately NOT via a real putBlob (so the + /// adopt decision below cannot be riding on THIS task's own fresh-upload meta write). + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + writeMetaClean(*b, s->layout(), hash, payload.size()); + const Token t0 = b->head(blob_key).token; + + /// Adopt must happen under a durable precommit edge (EDGE-BEFORE-OBSERVE), mirroring + /// PutBlobDedupSecondWriterAdopts above. + const RootNamespace ns{"srv/tbl"}; + auto build = startBuildFor(s, ns, "ref_adopt"); + const ManifestId manifest_id = build->stageManifest({blobManifestEntry("data.bin", payload)}); + build->precommitAdd(ns, "ref_adopt", manifest_id); + auto ref = build->putBlob(id, BlobSource::fromString(payload)); + + EXPECT_EQ(ref.ref, id); + /// Adopted: the pre-seeded incarnation survives untouched — no putOverwrite/re-upload happened. + EXPECT_EQ(b->head(blob_key).token, t0); + + const auto lm = loadMetaForTest(*b, s->layout(), hash); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean) << "an adopt must leave the meta Clean"; +} + +/// A putBlob call without the mandatory durable precommit must fail closed before either observing or +/// publishing a body. This is the intentional negative fixture for the writer-readiness contract. +TEST(CASPartWriteTxn, AdoptBeforePrecommitFailsClosed) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "adopt-before-precommit-payload"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + + /// Pre-seed a present body (padded past the pool header so the logical-size guard does not + /// underflow) + an independent Clean meta, so putBlob's upload conflicts on the present object and + /// takes the observation/adoption branch of `ensureBlobPresent` — mirroring PutBlobAdoptsWhenMetaCleanNoRetireView. + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + writeMetaClean(*b, s->layout(), hash, payload.size()); + + /// Start a build but DO NOT call precommitAdd. + const RootNamespace ns{"srv/tbl"}; + auto build = startBuildFor(s, ns, "ref_adopt"); + + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->putBlob(id, BlobSource::fromString(payload)); + }, + "durable precommit required"); +} + +/// The condemned-body replacement decision is likewise driven purely by the metadata point-read +/// — again, no RetireView is seeded. A condemned meta must cause putBlob to displace the body (a fresh +/// token, the old one never returns — INV-NO-RETURN, unchanged body mechanics) AND flip the meta back +/// to Clean. +TEST(CASPartWriteTxn, PutBlobRepublishesWhenMetaCondemned) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "republish-meta-payload"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + const String blob_key = s->layout().blobKey(id); + + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + writeMetaClean(*b, s->layout(), hash, payload.size()); + condemnMeta(*b, s->layout(), hash, /*condemn_round*/ 1); + const Token t0 = b->head(blob_key).token; + + /// No retire-view seeding: the replacement is decided from the metadata point-read. + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/republish-meta"}, "part", payload); + auto ref = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(ref.ref, id); + + /// Resurrected: the condemned incarnation was displaced by a fresh one. + const HeadResult hr = b->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_NE(hr.token, t0) << "a condemned incarnation must be displaced by a fresh publication"; + EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch) + << "the condemned token must never return (INV-NO-RETURN)"; + + const auto lm = loadMetaForTest(*b, s->layout(), hash); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean) << "republishing must flip the metadata back to Clean"; +} + +/// §0 introspection: the condemned-displacement metadata flip goes through the `casMeta` +/// choke point (`CASMetaCompareSwap`), tagged with its reason (`CASMetaResurrectClean`). +TEST(CASPartWriteTxnMetaCounters, CondemnedRepublicationCountsCasAndReason) +{ + const auto cas_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap].load(); + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean].load(); + + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "republish-meta-payload-counters"; + const UInt128 hash = u128Of(payload); + const BlobRef id = idOf(payload); + + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + writeMetaClean(*b, s->layout(), hash, payload.size()); + condemnMeta(*b, s->layout(), hash, /*condemn_round*/ 1); + + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/republish-meta-counters"}, "part", payload); + auto ref = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(ref.ref, id); + + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap].load() - cas_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean].load() - reason_before, 1); +} + +TEST(CASPartWriteTxn, PutBlobWrongSizeFailsClosed) +{ + auto b = std::make_shared(); + auto s = openPool(b); + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/wrong-size"}, "part", "hello world"); + + BlobSource lying; + lying.size = 11; /// declares 11 but writes 5 + lying.open = []() -> std::unique_ptr + { return std::make_unique(String("short")); }; + + const BlobRef id = idOf("hello world"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + build->putBlob(id, std::move(lying)); + }); + /// The cancelled stream created nothing. + EXPECT_FALSE(b->head(s->layout().blobKey(id)).exists); +} + +/// The happy-path upload STREAMS the source directly into the put sink — it does NOT pre-materialize the +/// whole blob into an in-memory String before the I/O. We assert this by counting `open` +/// invocations: a single fresh upload must invoke it EXACTLY ONCE (streamed straight into the sink). The +/// previous implementation buffered the whole blob into a `String source_bytes` first (a full in-memory +/// copy whose peak grew ~linearly with the blob size — the OOM); that pass would invoke `open` +/// before the sink write. One invocation here is the streaming-not-materializing guarantee. +TEST(CASPartWriteTxn, PutBlobStreamsSourceOnceNoFullMaterialization) +{ + auto b = std::make_shared(); + auto s = openPool(b); + + const String payload = "streamed-not-materialized"; + auto build = precommittedBuildForPayload(s, RootNamespace{"srv1/streaming"}, "part", payload); + int invocations = 0; + BlobSource source; + source.size = payload.size(); + source.open = [&invocations, &payload]() -> std::unique_ptr + { + ++invocations; + return std::make_unique(payload); + }; + + auto ref = build->putBlob(idOf(payload), std::move(source)); + EXPECT_EQ(ref.size, payload.size()); + EXPECT_EQ(invocations, 1) << "happy-path upload must stream the source exactly once (no pre-materialization pass)"; + + /// And the object really landed with the streamed payload (at the fixed header offset). + auto raw = b->get(s->layout().blobKey(ref.ref)); + ASSERT_TRUE(raw.has_value()); + auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(raw->bytes.substr(h.header_len), payload); +} + +/// B190: reuseBlob is removed (it had no production callers post-B188). Its behaviors are now covered by: +/// - trusted adopted leaf at gate: PromoteTrustsAdoptedLeafNoProbeManifestTrust (CasPartWriteTxn) — §4 +/// manifest-trust: a committed-source adopted leaf publishes with NO per-file probe; a materialized leaf +/// is edge-protected (Phase A) and never re-observed at the gate. +/// - absent adopted leaf trusted: PromoteTrustsAdoptedLeafEvenIfBackendRaced (CasPartWriteTxn) — the D4 +/// trade-off (a genuinely-absent adopted blob is caught by fsck, not the promote gate). +/// - explicit evidence: DependencyProofDistinguishesMaterializedAndTrustedManifest. +/// - no-dependency staging bug: MissingDependencyProofFailsClosed. + +TEST(CASPartWriteTxnReuseBlob, DependencyProofDistinguishesMaterializedAndTrustedManifest) +{ + /// A successful publication records physical evidence without retaining the backend token in + /// writer readiness state. + auto b = std::make_shared(); + auto s = openPool(b); + + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/dependency-proof/materialized"}, "written", "written"); + + /// A fresh upload is materialized. + build->putBlob(idOf("written"), BlobSource::fromString("written")); + EXPECT_EQ(build->dependencyProof(idOf("written")), BlobDependencyProof::Materialized); + + /// Observing that same live blob under a durable precommit is also materialized. + const RootNamespace ns{"srv1/dependency-proof"}; + auto observer = startBuildFor(s, ns, "observed"); + const ManifestEntry observed_entry = blobManifestEntry("data.bin", "written"); + const ManifestId observed_manifest = observer->stageManifest({observed_entry}); + observer->precommitAdd(ns, "observed", observed_manifest); + observer->putBlob(idOf("written"), BlobSource::fromString("written")); + EXPECT_EQ(observer->dependencyProof(idOf("written")), BlobDependencyProof::Materialized); + + /// A committed-source manifest supplies trusted-manifest evidence without physical I/O. + build->adoptEvidence(blobManifestEntry("f", "adopted")); + EXPECT_EQ(build->dependencyProof(idOf("adopted")), BlobDependencyProof::TrustedManifest); + + /// An unknown hash has no proof; absence is distinct from both accepted states. + EXPECT_EQ(build->dependencyProof(idOf("unknown")), std::nullopt); +} + +/// B190: ReuseBlobCondemnedThrowsAbortedRetryable is removed (reuseBlob is gone). §4 manifest-trust: a +/// committed-source adopted leaf is TRUSTED at the promote gate (no HEAD/loadMeta probe), so a condemned +/// pool blob no longer surfaces at promote — covered by PromoteTrustsAdoptedLeafNoProbeManifestTrust. + +TEST(CASPartWriteTxn, PutBlobRepublishesVanishedBodyFromHeldSource) +{ + auto b = std::make_shared(); + + /// 1. Write payload-X via a throwaway build to create the blob; capture its token t0. + BlobRef id; + Token t0; + { + auto s0 = openPool(b); + auto build0 = precommittedBuildForPayload( + s0, RootNamespace{"srv1/republish-vanished-seed"}, "part", "payload-X"); + id = build0->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")).ref; + t0 = b->head(s0->layout().blobKey(id)).token; + build0->abandon(); + } + + /// 2. Condemn (Blob, hash(X), t0) in the retire view. + DB::Cas::Layout layout("p"); + const String blob_key = layout.blobKey(id); + /// v3: the writer's condemned decision is a per-hash meta point-read (not the retire-view). Condemn the + /// meta; t0 stays as the body token the delete-hook below fires with. + condemnMeta(*b, layout, u128Of("payload-X"), /*condemn_round*/ 1); + + /// 3. Wrap the backend so the NEXT head(blob_key) returns the (present) result and THEN fires + /// deleteExact(blob_key, t0) exactly once — GC's delete in the HEAD->GET window. Open a FRESH + /// Pool over the hook so its retire view (refreshed at open) sees the condemnation. + auto hook = std::make_shared(b, blob_key, t0); + auto s = Pool::open(hook, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/republish-vanished"}, "part", "payload-X"); + + /// 4. putBlob with a re-invokable body. + /// The mandatory `HEAD` observes the condemned body and the hook deletes it before returning. + /// Publication then recreates the body under a fresh token without reading the condemned object. + auto ref = build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); + EXPECT_EQ(ref.ref, id); + + /// 5. The blob is present again under a FRESH token, with the same payload; and the condemned token + /// never returns (INV-NO-RETURN). + const HeadResult hr = b->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_NE(hr.token, t0); + + auto raw = b->get(blob_key); + ASSERT_TRUE(raw.has_value()); + auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(h.header_len, s->poolMeta().blob_header_len); + EXPECT_EQ(raw->bytes.substr(h.header_len), "payload-X"); + + EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); + + /// The freshness meta must be reconciled to Clean too, not left stale at Condemned: the fresh + /// re-upload's meta write (writeFreshMetaClean) must find and fix the pre-existing Condemned + /// marker via the same reload-and-reconcile path used after condemned replacement, not discard the + /// conflict (a stale Condemned marker would otherwise mislead every future point-reader). + const auto lm = loadMetaForTest(*b, s->layout(), u128Of("payload-X")); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean) + << "a fresh re-upload over a stale Condemned marker must reconcile it back to Clean"; +} + +/// A persistently-failing freshness-meta write (every attempt of every outer reload-retry) must +/// surface as a controlled retry-later signal, not silently succeed with the marker left stale +/// (S22 RCA). The blob body PUT +/// itself is unaffected (MetaWriteFaultBackend only faults `.meta` keys) -- only the meta write +/// exhausts, and that exhaustion must reach putBlob's caller as NETWORK_ERROR. +TEST(CASPartWriteTxn, PutBlobFreshMetaExhaustionThrowsRetryLater) +{ + /// Short budget + zero backoff: keep the test fast. Each of the metadata reconciliation loop's 8 outer + /// attempts calls putMetaIfAbsent, which itself retries up to max_attempts times internally — + /// with max_attempts=1 the controller gives up on the first faulted attempt each time. + CasRequestBudget budget; + budget.max_attempts = 1; + budget.retry_initial_backoff_ms = 0; + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + + const String payload = "fresh-meta-exhaustion-payload"; + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/fresh-meta-exhaustion"}, "part", payload); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + }); + + /// The body itself landed (only .meta writes are faulted) -- confirming the failure is + /// specifically the freshness marker, not the blob body. + const HeadResult hr = b->head(s->layout().blobKey(idOf(payload))); + EXPECT_TRUE(hr.exists) << "the body PUT is unaffected by the meta-only fault"; +} + +/// INV-1 (revival-from-source): a condemned blob is NEVER read via GET to revive it. +/// putBlob on a condemned-dedup hit must re-upload from its OWN source bytes — never calling +/// backend().get(blob_key). This test counts backend GETs on the blob key and asserts zero. +TEST(CASPartWriteTxn, PutBlobCondemnedDedupNeverGetsTheDyingObject) +{ + /// A delegating backend that counts get() calls on a specific key to assert INV-1. + struct GetCountingBackend final : public DB::Cas::Backend + { + explicit GetCountingBackend(BackendPtr inner_, String watched_key_) + : inner(std::move(inner_)), watched_key(std::move(watched_key_)) {} + size_t get_count = 0; + + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + std::optional get(const String & k, DB::Cas::Range r) override + { + if (k == watched_key) + ++get_count; + return inner->get(k, r); + } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & bts, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, bts, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & bts, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & bts, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & tok) override { return inner->deleteExact(k, tok); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + private: + BackendPtr inner; + String watched_key; + }; + + auto b = std::make_shared(); + + /// 1. Upload blob Y via a throwaway build; capture the token t0. + BlobRef id; + Token t0; + { + auto s0 = openPool(b); + auto build0 = precommittedBuildForPayload( + s0, RootNamespace{"srv1/condemned-absent-seed"}, "part", "payload-Y"); + id = build0->putBlob(idOf("payload-Y"), BlobSource::fromString("payload-Y")).ref; + t0 = b->head(s0->layout().blobKey(id)).token; + build0->abandon(); + } + + /// 2. Condemn (Blob, hash(Y), t0) in the retire view, then GC-delete the object so it is absent + /// (simulates GC completing the delete before the writer's dedup hit). + DB::Cas::Layout layout("p"); + const String blob_key = layout.blobKey(id); + injectRetire(*b, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-Y"))}, .token = t0, .size = 9}}); + b->deleteExact(blob_key, t0); + ASSERT_FALSE(b->head(blob_key).exists); + + /// 3. Open a fresh Pool over a GET-counting wrapper; the retire view sees the condemnation at open. + auto counting = std::make_shared(b, blob_key); + auto s = Pool::open(counting, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/condemned-absent"}, "part", "payload-Y"); + + /// 4. The object is absent, so `putBlob` observes absence and publishes from the held source. + /// Even if a racing re-creation happens between observation and publication, the condemned branch + /// must never call `Backend::get` on the blob key. + auto ref = build->putBlob(idOf("payload-Y"), BlobSource::fromString("payload-Y")); + EXPECT_EQ(ref.ref, id); + EXPECT_EQ(counting->get_count, 0u) << "INV-1: putBlob must not GET the dying object to revive it"; + + const HeadResult hr = b->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_NE(hr.token, t0) << "a fresh incarnation must have a new token"; + const auto raw = b->get(blob_key); + ASSERT_TRUE(raw.has_value()); + const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(raw->bytes.substr(hdr.header_len), "payload-Y"); +} + +/// INV-1 variant: blob is PRESENT and condemned (GC hasn't fired the delete yet). putBlob dedup-hits +/// it via PreconditionFailed, sees condemned token, and must re-upload from source — NEVER GET. +TEST(CASPartWriteTxn, PutBlobCondemnedDedupPresentNeverGetsTheDyingObject) +{ + struct GetCountingBackend final : public DB::Cas::Backend + { + explicit GetCountingBackend(BackendPtr inner_, String watched_key_) + : inner(std::move(inner_)), watched_key(std::move(watched_key_)) {} + size_t get_count = 0; + + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + std::optional get(const String & k, DB::Cas::Range r) override + { + if (k == watched_key) + ++get_count; + return inner->get(k, r); + } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & bts, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, bts, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & bts, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & bts, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & tok) override { return inner->deleteExact(k, tok); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + private: + BackendPtr inner; + String watched_key; + }; + + auto b = std::make_shared(); + + /// 1. Upload blob Z via a throwaway build; capture the token t0. + BlobRef id; + Token t0; + { + auto s0 = openPool(b); + auto build0 = precommittedBuildForPayload( + s0, RootNamespace{"srv1/condemned-present-seed"}, "part", "payload-Z"); + id = build0->putBlob(idOf("payload-Z"), BlobSource::fromString("payload-Z")).ref; + t0 = b->head(s0->layout().blobKey(id)).token; + build0->abandon(); + } + + /// 2. Condemn (Blob, hash(Z), t0) — object still PRESENT (GC condemned but not yet deleted). + DB::Cas::Layout layout("p"); + const String blob_key = layout.blobKey(id); + /// v3: condemn via the per-hash meta (the writer's freshness point-read), object still PRESENT. + condemnMeta(*b, layout, u128Of("payload-Z"), /*condemn_round*/ 1); + ASSERT_TRUE(b->head(blob_key).exists) << "blob must be PRESENT for the condemned-present path"; + + /// 3. Open a fresh Pool over a GET-counting wrapper. + auto counting = std::make_shared(b, blob_key); + auto s = Pool::open(counting, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + auto build = precommittedBuildForPayload( + s, RootNamespace{"srv1/condemned-present"}, "part", "payload-Z"); + + /// 4. `putBlob` observes the condemned metadata and republishes from the held source without a GET. + auto ref = build->putBlob(idOf("payload-Z"), BlobSource::fromString("payload-Z")); + EXPECT_EQ(ref.ref, id); + EXPECT_EQ(counting->get_count, 0u) << "INV-1: putBlob must not GET the condemned object"; + + const HeadResult hr = b->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_NE(hr.token, t0) << "condemned incarnation must be displaced by a fresh token"; + const auto raw = b->get(blob_key); + ASSERT_TRUE(raw.has_value()); + const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(raw->bytes.substr(hdr.header_len), "payload-Z"); +} + +TEST(CASPartWriteTxn, PromoteTrustsAdoptedLeafNoProbeManifestTrust) +{ + /// §4 manifest-trust: a committed-source adoptEvidence leaf is TRUSTED at the promote gate — the live + /// source pins the blob (in-degree >= 1, not condemnable) and this build's precommit edge is durable, + /// so promote publishes with NO per-file HEAD (presence) and NO loadMeta GET (the condemned point-read) + /// and NO copy-forward. The durable manifest edge is the liveness evidence. `CASBlobAdoptTrusted` counts + /// the trusted leaf. A KeyCountingBackend proves zero probes on the blob key and the blob-meta key. + auto raw = std::make_shared(); + auto counting = std::make_shared(raw); + auto s = Pool::open(counting, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl"}; + + /// A committed-source blob lives in the shared pool (seeded via a throwaway build on the same store). + { + const RootNamespace seed_ns{"srv1/trusted-seed"}; + auto seed = startBuildFor(s, seed_ns, "part"); + const ManifestId seed_manifest = durablyPrecommit( + seed, + seed_ns, + "part", + {blobManifestEntryStreaming("data.bin", "payload-TR")}); + seed->putBlob(streamRefOf("payload-TR"), BlobSource::fromString("payload-TR")); + seed->promote(seed_ns, "part", seed->buildId(), seed_manifest); + } + const String blob_key = s->layout().blobKey(streamRefOf("payload-TR")); + const String meta_key = s->layout().blobMetaKey(streamRefOf("payload-TR")); + + auto build = startBuildFor(s, ns, "part_1"); + const ManifestEntry entry = blobManifestEntryStreaming("data.bin", "payload-TR"); + build->adoptEvidence(entry); + EXPECT_EQ(build->dependencyProof(entry.ref), BlobDependencyProof::TrustedManifest); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + const auto trusted_before = ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted].load(); + const size_t head_before = counting->headCountFor(blob_key); + const size_t meta_get_before = counting->getCountFor(meta_key); + + build->promote(ns, "part_1", build->buildId(), id); + + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted].load() - trusted_before, 1); + EXPECT_EQ(counting->headCountFor(blob_key) - head_before, 0u) << "trust must not HEAD the adopted blob"; + EXPECT_EQ(counting->getCountFor(meta_key) - meta_get_before, 0u) << "trust must not loadMeta the adopted blob"; +} + +TEST(CASPartWriteTxn, PromotionAcceptsBothDependencyProofs) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/proof-promotion"}; + + /// Seed the committed-source body used by the trusted-manifest branch. + { + publishOneBlobPart( + s, RootNamespace{"srv1/proof-promotion-seed"}, "part", "data.bin", "trusted-body"); + } + + auto build = startBuildFor(s, ns, "part_1"); + const ManifestEntry materialized = blobManifestEntry("data.bin", "materialized-body"); + const ManifestEntry trusted = blobManifestEntry("data.cmrk3", "trusted-body"); + const ManifestId id = build->stageManifest({materialized, trusted}); + build->precommitAdd(ns, "part_1", id); + + build->putBlob(materialized.ref, BlobSource::fromString("materialized-body")); + build->adoptEvidence(trusted); + ASSERT_EQ(build->dependencyProof(materialized.ref), BlobDependencyProof::Materialized); + ASSERT_EQ(build->dependencyProof(trusted.ref), BlobDependencyProof::TrustedManifest); + + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); +} + +TEST(CASPartWriteTxn, InvalidDependencyProofFailsClosed) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/invalid-proof"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestEntry entry = blobManifestEntry("data.bin", "invalid-proof-body"); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + BlobUploadResult invalid{ + entry.ref, + BlobDepRecord{ObjectKind::Blob, static_cast(2), entry.blob_size}, + BlobUploadDiagnostics{ + BlobMaterializationAction::Published, + BlobPublicationReason::Absent, + BlobPublicationTransport::Streaming}}; + build->mergeBlobUploadResults(std::span(&invalid, 1)); + + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->promote(ns, "part_1", build->buildId(), id); + }, + "unnamed dependency proof"); +} + +TEST(CASPartWriteTxn, PromoteTrustsAdoptedLeafEvenIfBackendRaced) +{ + /// §4 manifest-trust trade-off (D4 relink interserver-trust model): a committed-source adopted leaf is + /// published WITHOUT a presence probe. Even if the pool object raced to absent between adopt and + /// promote, promote does NOT re-observe it — the ref publishes. A genuinely-absent adopted blob is an + /// invariant violation detected by fsck (or an actual body GET on read), not caught at the promote gate. + /// This is the deliberate reduction from the pre-§4 "absent adopted leaf => ABORTED at gate". + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Seed X, adopt it, then delete it out from under the build (a landed GC delete in the adopt->promote + /// window). The pre-§4 gate HEADed X, found it absent, and threw ABORTED; §4 trusts the durable edge. + { + const RootNamespace seed_ns{"srv1/race-seed"}; + auto seed = startBuildFor(s, seed_ns, "part"); + const ManifestId seed_manifest = durablyPrecommit( + seed, + seed_ns, + "part", + {blobManifestEntryStreaming("data.bin", "payload-RACE")}); + seed->putBlob(streamRefOf("payload-RACE"), BlobSource::fromString("payload-RACE")); + seed->promote(seed_ns, "part", seed->buildId(), seed_manifest); + } + const String blob_key = s->layout().blobKey(streamRefOf("payload-RACE")); + const Token t0 = b->head(blob_key).token; + + auto build = startBuildFor(s, ns, "part_1"); + const ManifestEntry entry = blobManifestEntryStreaming("data.bin", "payload-RACE"); + build->adoptEvidence(entry); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + ASSERT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::Deleted); + ASSERT_FALSE(b->head(blob_key).exists); + + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); +} + +TEST(CASPartWriteTxn, PromoteSwallowsPostDurableEventSinkFailure) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + ManifestEntry entry; + entry.path = "data.bin"; + entry.placement = EntryPlacement::Inline; + entry.ref = idOf("payload"); + entry.blob_size = 7; + entry.inline_bytes = "payload"; + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + /// UNKNOWN_EXCEPTION (not LOGICAL_ERROR): this simulates an arbitrary observer/sink callback + /// failing, not a CAS invariant violation -- LOGICAL_ERROR would abort the whole process under + /// debug/sanitizer builds instead of behaving like a catchable exception. + s->setEventSink([](const CasEvent & e) + { + if (e.type == CasEventType::BuildPublish) + throw DB::Exception(DB::ErrorCodes::UNKNOWN_EXCEPTION, "injected post-durable event sink failure"); + }); + + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + const auto resolved = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(resolved); + EXPECT_EQ(resolved->manifest_id, id); + s->setEventSink(nullptr); +} + +TEST(CASPartWriteTxn, MissingDependencyProofFailsClosed) +{ + /// A manifest blob leaf with NO recorded dep (a staging-bug shape: neither putBlob nor adoptEvidence + /// recorded it) must fail closed at the promote gate because neither accepted proof exists, so §4 + /// never silently publishes it. Under manifest-trust there is NO per-file + /// probe, so the fail-closed is a LOGICAL_ERROR decided from the dep set alone — it fires regardless of + /// the pool blob's presence/condemnation (here the blob is even condemned, but that is never observed). + /// The no-dep shape is reachable through the public build API (stageManifest names the leaf without any + /// dep having been recorded), so no test accessor for the private predicate is needed. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// X exists (streaming-keyed) but THIS build records NO dep for it (no putBlob, no adoptEvidence). + { + const RootNamespace seed_ns{"srv1/missing-proof-seed"}; + auto seed = startBuildFor(s, seed_ns, "part"); + const ManifestId seed_manifest = durablyPrecommit( + seed, + seed_ns, + "part", + {blobManifestEntryStreaming("data.bin", "payload-NODEP")}); + seed->putBlob(streamRefOf("payload-NODEP"), BlobSource::fromString("payload-NODEP")); + seed->promote(seed_ns, "part", seed->buildId(), seed_manifest); + } + const String blob_key = s->layout().blobKey(streamRefOf("payload-NODEP")); + const Token t0 = b->head(blob_key).token; + + auto build = startBuildFor(s, ns, "part_1"); + const ManifestEntry entry = blobManifestEntryStreaming("data.bin", "payload-NODEP"); + /// NB: no `adoptEvidence` call — dependencies stay empty for this hash. + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + /// Condemn X (present) via the meta — under §4 the gate never point-reads it (no probe on a non-trusted + /// leaf), so this only confirms the fail-closed does not depend on the leaf being clean. + condemnMeta(*b, s->layout(), hexToU128(streamingHexOf("payload-NODEP")), /*condemn_round*/ 1); + + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->promote(ns, "part_1", build->buildId(), id); + }, + "no dependency proof"); + EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); + /// The pool blob was never touched (no probe, no displacement). + EXPECT_EQ(b->head(blob_key).token, t0); +} + +TEST(CASPartWriteTxn, PromoteRevalidatesBlobPresenceFailClosed) +{ + /// Port of the old W-TREE-BUILD bottom-up enforcement (PutTreeEnforcesBottomUp): the surviving + /// "a committed ref never names a missing dependency" invariant. In the part-manifest model + /// stageManifest does not validate its entries' bodies. §4 manifest-trust: the fail-closed authority at + /// the promote gate is now the DEP SET, not a backend HEAD — a leaf named by the manifest with NO + /// dependency proof (neither `putBlob` nor `adoptEvidence` recorded one) is a staging bug and fails + /// closed with LOGICAL_ERROR (a real write always records a dep for every leaf). No per-file probe. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Stage + precommit a manifest naming a blob hash that was NEVER uploaded (no dep recorded). + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId mid = build->stageManifest({blobManifestEntry("data.bin", "never-uploaded")}); + build->precommitAdd(ns, "part_1", mid); + + /// Promotion must fail closed because the leaf has no dependency proof. No ref is committed. + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->promote(ns, "part_1", build->buildId(), mid); + }, + "no dependency proof"); + EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); + + /// After uploading the blob, a fresh build's promote succeeds — the same manifest content is now + /// fully present. + auto build2 = startBuildFor(s, ns, "part_1"); + const ManifestId mid2 = build2->stageManifest({blobManifestEntry("data.bin", "never-uploaded")}); + build2->precommitAdd(ns, "part_1", mid2); + build2->putBlob(idOf("never-uploaded"), BlobSource::fromString("never-uploaded")); + EXPECT_NO_THROW(build2->promote(ns, "part_1", build2->buildId(), mid2)); + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); +} + +TEST(CASPartWriteTxn, AdoptEvidenceRecordsTrustedManifestProof) +{ + /// Port of AdoptFromTreeRecordsEvidence. `adoptEvidence` records `TrustedManifest` directly + /// from a resolved `ManifestEntry`; a Blob entry has a proof, while an Inline entry + /// records nothing. §4: whether the dep is a committed-source adopt vs absent (adopted vs no-dep) is + /// asserted end-to-end at the promote gate by PromoteTrustsAdoptedLeafNoProbeManifestTrust (positive: + /// adopted leaf ⇒ trusted, no probe) and PromoteCondemnedLeafWithoutDepAbortsFailClosed (negative + /// control: no dep ⇒ fail closed, LOGICAL_ERROR). + auto b = std::make_shared(); + auto s = openPool(b); + auto build = s->beginPartWrite({}); + + const ManifestEntry adopted = blobManifestEntry("data.bin", "source-blob"); + build->adoptEvidence(adopted); + EXPECT_EQ(build->dependencyProof(adopted.ref), BlobDependencyProof::TrustedManifest); + + /// An Inline entry references no standalone object and records no proof. + ManifestEntry inline_entry; + inline_entry.path = "small"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.inline_bytes = "abc"; + build->adoptEvidence(inline_entry); + EXPECT_EQ(build->dependencyProof(idOf("abc")), std::nullopt); +} + +TEST(CASPartWriteTxn, AbandonRemovesStagedDebrisAndDisables) +{ + /// Port of AbandonLeavesDebrisAndDisables to the new abandon semantics (CasPartWriteTxn.cpp abandon): + /// abandon best-effort exact-token-DELETEs this build's STAGED manifest debris, leaves blob bodies + /// (full GC's job via min_active), and disables the build (further ops throw via requireAlive). + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "ref"); + + const ManifestId owner = build->stageManifest({blobManifestEntry("owner", "kept")}); + build->precommitAdd(ns, "ref", owner); + auto blob_ref = build->putBlob(idOf("kept"), BlobSource::fromString("kept")); + const ManifestId mid = build->stageManifest({blobManifestEntry("f", "kept")}); + + /// The staged manifest body and the blob are present before abandon. + EXPECT_TRUE(b->head(s->layout().blobKey(blob_ref.ref)).exists); + EXPECT_TRUE(b->head(s->layout().manifestKey(mid)).exists); + + build->abandon(); + + /// Blob stays (debris — full GC reclaims it). The staged manifest debris is best-effort cleaned now. + EXPECT_TRUE(b->head(s->layout().blobKey(blob_ref.ref)).exists); + EXPECT_FALSE(b->head(s->layout().manifestKey(mid)).exists) + << "abandon must best-effort delete this build's staged manifest debris"; + + /// Further operations throw via requireAlive. + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->putBlob(idOf("after"), BlobSource::fromString("after")); + }, + "has been abandoned"); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->stageManifest({blobManifestEntry("g", "kept")}); + }, + "has been abandoned"); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->precommitAdd(ns, "ref", mid); + }, + "has been abandoned"); +} + +TEST(CASPartWriteTxn, PublishHappyPathRoundTrip) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId id = build->stageManifest({blobManifestEntry("data.bin", "hello world")}); + build->precommitAdd(ns, "part_1", id); + auto blob = build->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + EXPECT_EQ(blob.size, 11u); + + build->promote(ns, "part_1", build->buildId(), id); + + auto r = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->manifest_id, id); + + /// Read the manifest back and locate its single blob leaf. + const PartManifest manifest = s->readManifest(id); + ASSERT_EQ(manifest.entries.size(), 1u); + const auto * entry = findEntry(manifest.entries, "data.bin"); + ASSERT_TRUE(entry != nullptr); + const auto loc = s->locate(*entry); + auto got = b->get(loc.key, Range{loc.offset, loc.length}); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, "hello world"); +} + +TEST(CASPartWriteTxn, PromoteCrossNamespaceManifestFailsClosed) +{ + /// Port of PublishRequiresTreeInDepSet. The W-DEP-SET "root must be a built/adopted dep" authority + /// is gone (the tree object model it guarded is gone); the surviving fail-closed authority that + /// refuses an inconsistent commit target is the namespace consistency check in precommitAdd/promote + /// (CasPartWriteTxn.cpp): a manifest whose root_namespace != the target namespace is a bug ⇒ LOGICAL_ERROR. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + const RootNamespace other_ns{"srv1/other"}; + + auto build = startBuildFor(s, ns, "part_1"); + /// The manifest is minted in `ns` (derived from intended_ref). Promoting/precommitting it into a + /// DIFFERENT namespace must fail closed. + const ManifestId id = build->stageManifest({blobManifestEntry("data.bin", "hello world")}); + + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->precommitAdd(other_ns, "part_1", id); + }, + "precommitAdd: manifest namespace"); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->promote(other_ns, "part_1", build->buildId(), id); + }, + "promote: manifest namespace"); +} + +/// (CASPartWriteTxn.PublishOwnThreadConflictRetries was removed with the legacy mutable ref-shard lane: it +/// injected a Conflict on the promote's shard `casPut` and asserted the shard re-read/retry. The ref model +/// has no shard CAS -- promote appends a write-once ref-log object via `putIfAbsentControlled`, and an +/// uncertain create is resolved by exact-key observation, covered by the ref-writer uncertain-result tests +/// (`gtest_cas_ref_writer.cpp`), not a CAS retry.) + +TEST(CASPartWriteTxn, PublishIntoSecondNamespaceSameBlob) +{ + /// Port of PublishIntoSecondNamespaceSameTree. A part manifest is single-owner and namespace-qualified + /// (precommitAdd/promote enforce id.root_namespace == target_ns), so the SAME ManifestId cannot be + /// published into two namespaces — each namespace gets its OWN manifest. The invariant the original + /// test protected is preserved at the BLOB plane: the shared blob is uploaded ONCE and adopted by the + /// second build (its token is unchanged after the second publish). + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns1{"srv1/tbl"}; + const RootNamespace ns2{"srv1/tbl/detached"}; + + /// First build publishes part_1 in ns1, uploading the blob. + auto build1 = startBuildFor(s, ns1, "part_1"); + const ManifestId id1 = build1->stageManifest({blobManifestEntry("data.bin", "hello world")}); + build1->precommitAdd(ns1, "part_1", id1); + auto blob = build1->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + const String blob_key = s->layout().blobKey(blob.ref); + const Token blob_token = b->head(blob_key).token; + build1->promote(ns1, "part_1", build1->buildId(), id1); + + /// Second build publishes part_1 in ns2 referencing the SAME blob: putBlob dedup-hits and ADOPTS the + /// present incarnation (no re-upload), so the blob token is unchanged. Wiring order + /// (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob -> promote. + auto build2 = startBuildFor(s, ns2, "part_1"); + const ManifestId id2 = build2->stageManifest({blobManifestEntry("data.bin", "hello world")}); + build2->precommitAdd(ns2, "part_1", id2); + build2->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + build2->promote(ns2, "part_1", build2->buildId(), id2); + + auto r1 = s->resolveRef(ns1, "part_1"); + auto r2 = s->resolveRef(ns2, "part_1"); + ASSERT_TRUE(r1.has_value()); + ASSERT_TRUE(r2.has_value()); + EXPECT_EQ(r1->manifest_id, id1); + EXPECT_EQ(r2->manifest_id, id2); + + /// The blob object was uploaded once: its token is unchanged after both publishes. + EXPECT_EQ(b->head(blob_key).token, blob_token); +} + +/// Task 10: refs are no longer sharded (one whole-table cache per namespace, spec §Table State), so +/// there is no more "same shard" CAS-conflict-retry to force — two builds publishing into the SAME +/// TABLE now serialize through the append lane's per-namespace batching queue instead (exercised by +/// gtest_cas_ref_writer.cpp's co-batch/queue tests). What remains a real regression to guard is the +/// end-to-end outcome: two builds publishing distinct refs into one namespace both land correctly. +TEST(CASPartWriteTxn, TwoBuildsPublishToSameNamespaceBothLand) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + const String ref1 = "a"; + const String ref2 = "b"; + + auto build_a = startBuildFor(s, ns, ref1); + const ManifestId id_a = build_a->stageManifest({blobManifestEntry("data.bin", "content-a")}); + build_a->precommitAdd(ns, ref1, id_a); + build_a->putBlob(idOf("content-a"), BlobSource::fromString("content-a")); + build_a->promote(ns, ref1, build_a->buildId(), id_a); + + auto build_b = startBuildFor(s, ns, ref2); + const ManifestId id_b = build_b->stageManifest({blobManifestEntry("data.bin", "content-b")}); + build_b->precommitAdd(ns, ref2, id_b); + build_b->putBlob(idOf("content-b"), BlobSource::fromString("content-b")); + build_b->promote(ns, ref2, build_b->buildId(), id_b); + + auto r1 = s->resolveRef(ns, ref1); + auto r2 = s->resolveRef(ns, ref2); + ASSERT_TRUE(r1.has_value()); + ASSERT_TRUE(r2.has_value()); + EXPECT_EQ(r1->manifest_id, id_a); + EXPECT_EQ(r2->manifest_id, id_b); + EXPECT_EQ(s->listRefs(ns).size(), 2u); +} + +TEST(CASPartWriteTxn, FirstPublishMakesNamespaceDiscoverable) +{ + /// After Task 4 the registry is deleted; the first publication admits the namespace to the + /// authoritative catalog before appending its stream record. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv9/fresh"}; + + EXPECT_TRUE(s->listNamespaces("").namespaces.empty()); + publishOneBlobPart(s, ns, "part_1", "f", "reg-payload"); + + /// The namespace is now discoverable from the catalog -- no registry write needed. + const auto all = s->listNamespaces("").namespaces; + ASSERT_EQ(all.size(), 1u); + EXPECT_EQ(all[0], "srv9/fresh"); +} + +TEST(CASPartWriteTxn, AdoptEvidenceRecordsTrustedManifestDependencyProofWithoutIO) +{ + /// B188: `adoptEvidence` records `TrustedManifest` from an already resolved `ManifestEntry` + /// WITHOUT any backend call (no HEAD, no GET, no PUT). + /// + /// Two behavioural assertions: + /// 1. No backend op fires during adoptEvidence (counted via a delegating wrapper). + /// 2. The recorded dependency proof is `TrustedManifest`. + + /// A delegating wrapper that counts every backend access path. + struct LocalCountingBackend final : public Backend + { + explicit LocalCountingBackend(BackendPtr inner_) : inner(std::move(inner_)) {} + size_t heads = 0; + size_t puts = 0; + size_t gets = 0; + + HeadResult head(const String & k) override { ++heads; return inner->head(k); } + void publishBlob(const BlobPublishRequest & request) override + { + ++puts; + inner->publishBlob(request); + } + std::optional get(const String & k, Range r) override { ++gets; return inner->get(k, r); } + std::optional getStream(const String & k, Range r) override { return inner->getStream(k, r); } + ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + PutResult putIfAbsent(const String & k, const String & bts, const ObjectMeta & m) override + { + ++puts; + return inner->putIfAbsent(k, bts, m); + } + PutResult putOverwrite(const String & k, const String & bts, const Token & e, const ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } + CasResult casPut(const String & k, const String & bts, const std::optional & e, const ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } + DeleteOutcome deleteExact(const String & k, const Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + private: + BackendPtr inner; + }; + + auto raw = std::make_shared(); + auto counting = std::make_shared(raw); + auto s = Pool::open(counting, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + auto build = s->beginPartWrite({}); + + /// A Blob ManifestEntry. adoptEvidence is called on a hand-crafted entry — that IS the B188 interface. + const ManifestEntry entry = blobManifestEntry("b188.bin", "b188-content"); + + /// Reset the counters after Pool::open (which may HEAD/GET gc/server-roots etc. during startup). + counting->heads = 0; + counting->puts = 0; + counting->gets = 0; + + /// adoptEvidence — must record the dep WITHOUT touching the backend. + EXPECT_NO_THROW(build->adoptEvidence(entry)); + EXPECT_EQ(counting->heads, 0u) << "adoptEvidence must not HEAD the backend"; + EXPECT_EQ(counting->puts, 0u) << "adoptEvidence must not PUT to the backend"; + EXPECT_EQ(counting->gets, 0u) << "adoptEvidence must not GET from the backend"; + + /// The dep is recorded as trusted manifest evidence; the no-backend-op counts above are the B188 + /// contract's primary guard. + EXPECT_EQ(build->dependencyProof(entry.ref), BlobDependencyProof::TrustedManifest); + + /// Inline entry: adoptEvidence records nothing (Inline has no standalone object) and no backend op. + ManifestEntry inline_entry; + inline_entry.path = "small"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.inline_bytes = "xy"; + EXPECT_NO_THROW(build->adoptEvidence(inline_entry)); + EXPECT_EQ(counting->heads, 0u); + EXPECT_EQ(counting->puts, 0u); + EXPECT_EQ(counting->gets, 0u); + EXPECT_EQ(build->dependencyProof(idOf("xy")), std::nullopt); +} + +TEST(CASPartWriteTxn, ConvergesUnderProductiveGc) +{ + /// B167/B171 LIVENESS — the re-upload/condemn livelock, now closed by the build-root precommit edge. + /// + /// THE BUG (before the fix): a blob H was referenced, dropped, and GC-condemned (everEdged ∧ InDeg=0, + /// condemned in the retire view). A NEW build dedup-HITS H by content and must re-upload it from + /// source — it re-streams a FRESH incarnation of H. But the productive GC, re-deriving H as a + /// zero-in-degree candidate every round, kept RE-CONDEMNING and exact-token-DELETING that fresh + /// incarnation in the build's upload→commit window. The build never converged → livelock. + /// + /// THE FIX (B171): protection is the build-root PRECOMMIT EDGE. PartWriteTxn B precommits its manifest (naming + /// H) BEFORE the adversarial loop, so the GC fold lifts H to in-degree ≥ 1 — H is never even a + /// zero-in-degree candidate and is SPARED every round until B promotes (the committed ref then pins H). + /// + /// FORM: full adversarial loop. A real Gc drives complete runRegularRound rounds against the same + /// pool while build B holds an active watermark covering H's incarnation. We assert H is SPARED + /// every round and that B promotes within a BOUNDED number of GC rounds, after which H reads back. + auto b = std::make_shared(); + const RootNamespace ns{"srv1/tbl"}; + + PoolConfig cfg; + cfg.pool_prefix = "p"; + cfg.server_root_id = "test"; + cfg.server_id = UInt128(0xAB); + cfg.background_watermark = false; + const String content = "shared-content"; + + /// 1. PartWriteTxn A creates H ("shared-content"), publishes a part referencing it, then drops the ref. + /// Capture H's first incarnation token so we can condemn exactly it. + BlobRef h; + Token h_token0; + { + auto s0 = Pool::open(b, cfg); + publishOneBlobPart(s0, ns, "part_1", "f", content); + h = idOf(content); + h_token0 = b->head(s0->layout().blobKey(h)).token; + s0->dropRef(ns, "part_1"); + } + + /// 2. Condemn (Blob, H, h_token0): `injectRetire` seeds the LEDGER (a real round's later settle/spare + /// of h_token0 rides this entry — the adversarial loop below still exercises that), and v3's + /// `condemnMeta` seeds the per-hash META (the writer's condemned decision is now a point-read of + /// it, not the retire-view — Task 3). `publishOneBlobPart` already created H's meta as Clean, so + /// condemnMeta's read-modify-CAS finds it. Together these reproduce exactly what a real GC condemn + /// now writes (Task 5), without driving a full round just to observe H at in-degree 0. + DB::Cas::Layout layout("p"); + injectRetire(*b, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(content))}, .token = h_token0, + .size = content.size()}}); + condemnMeta(*b, layout, u128Of(content), /*condemn_round*/ 1); + + /// 3. Open the live Pool and start build B. B observes condemned H and republishes from + /// its own source through `putBlob`: a fresh incarnation, a NEW token. B stays ACTIVE for + /// the whole adversarial loop — its build_seq is never retired below. + auto s = Pool::open(b, cfg); + const String blob_key = s->layout().blobKey(h); + auto build_b = startBuildFor(s, ns, "part_2"); + + /// Establish the reachability edge before observing/replacing H. + const ManifestId mid_b = build_b->stageManifest({blobManifestEntry("f", content)}); + build_b->precommitAdd(ns, "part_2", mid_b); + + /// B190: use `putBlob` (holds source bytes). It detects the condemned observation and publishes + /// unconditionally — no GET of the dying object. + const auto ref_b = build_b->putBlob(h, BlobSource::fromString(content)); + ASSERT_EQ(ref_b.ref, h); + + const HeadResult after_reupload = b->head(blob_key); + ASSERT_TRUE(after_reupload.exists); + EXPECT_NE(after_reupload.token, h_token0); /// a genuinely fresh incarnation + + /// 4. THE ADVERSARIAL LOOP. A real, productive GC keeps trying to reclaim. It reclaims the now- + /// unreferenced part_1 manifest (build A's, UNprotected) but H stays pinned by B's PRECOMMIT edge + /// (in-degree ≥ 1), so H is never even a zero-in-degree candidate. We drive far more rounds than B + /// needs to promote; H must survive ALL of them. Each round renews B's watermark so the crash + /// detector keeps judging B live. + Gc gc(s, hexToU128("00000000000000000000000000000001")); + constexpr int MAX_GC_ROUNDS = 8; + + const auto driveRoundAndAssertHSpared = [&](int round_no) + { + /// A LIVE server renews its watermark continuously. Renew once per GC round so B's watermark seq + /// ADVANCES between rounds — that is precisely what distinguishes a live server from a crashed one + /// (a frozen B would have its precommit reclaimed; an advancing seq keeps it). + s->renewWatermarkOnce(); + gc.runRegularRound(); + const HeadResult hr = b->head(blob_key); + ASSERT_TRUE(hr.exists) << "H was deleted by GC at round " << round_no + << " despite being pinned by the live build B's precommit (B167 livelock would do this)"; + const auto raw = b->get(blob_key); + ASSERT_TRUE(raw.has_value()); + const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); + EXPECT_EQ(raw->bytes.substr(hdr.header_len), content) + << "H's content was lost/corrupted at round " << round_no; + }; + + /// Phase 1 — the livelock window. H is referenced by NO committed TABLE ref (B has not promoted yet) + /// but IS named by B's precommit, so the build-root fold lifts it to in-degree ≥ 1. Drive several + /// full rounds; the precommit edge must SPARE H's fresh incarnation every round. + int rounds_run = 0; + constexpr int PRE_PUBLISH_ROUNDS = 4; + for (int i = 0; i < PRE_PUBLISH_ROUNDS; ++i) + { + driveRoundAndAssertHSpared(++rounds_run); + if (::testing::Test::HasFatalFailure()) + return; + } + + /// Phase 2 — converge. With H still alive (spared through the whole window), build B promotes a part + /// referencing it. The promote gate sees H present + live (fresh incarnation uploaded above), so it + /// commits. This MUST succeed — the build converges in bounded steps. + build_b->promote(ns, "part_2", build_b->buildId(), mid_b); + const bool published = true; + + /// Phase 3 — keep the GC hammering after promote. H is now pinned by the committed ref's manifest + /// edge; the GC must keep sparing it as a genuinely-reachable node. + while (rounds_run < MAX_GC_ROUNDS) + { + driveRoundAndAssertHSpared(++rounds_run); + if (::testing::Test::HasFatalFailure()) + return; + } + + /// 6. ASSERT convergence: promote SUCCEEDED within the bounded budget, and H reads back intact. + ASSERT_TRUE(published) << "build B never published — the B167 livelock is back"; + EXPECT_LE(rounds_run, MAX_GC_ROUNDS); + + const auto resolved = s->resolveRef(ns, "part_2"); + ASSERT_TRUE(resolved.has_value()); + EXPECT_EQ(resolved->manifest_id, mid_b); + + const PartManifest manifest = s->readManifest(mid_b); + ASSERT_EQ(manifest.entries.size(), 1u); + const auto * entry = findEntry(manifest.entries, "f"); + ASSERT_TRUE(entry != nullptr); + const auto loc = s->locate(*entry); + const auto got = b->get(loc.key, Range{loc.offset, loc.length}); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, content); +} + +/// BUG 1 (WPromote owner==bld): promote is a PURE owner MOVE (Δ=0 — it restores no blob in-degree). The +/// TLA+ `WPromote` requires the precommit to STILL be the live owner of the ref before the move (`owner[m] +/// = bld`). If the precommit was removed/reclaimed (an abandon or GC reclaim appended a removal event), a +/// Δ=0 move would re-publish a committed ref over blobs whose in-degree was already decremented to 0 — GC +/// then deletes them ⇒ a reachable committed manifest with dangling blobs (INV_NO_DANGLE violation). +/// promote MUST fail closed (ABORTED) unless the precommit is the current live owner binding of the ref. +TEST(CASPartWriteTxn, PromoteFailsClosedWhenPrecommitNoLongerLiveOwner) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId id = build->stageManifest({blobManifestEntry("data.bin", "hello world")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + + /// Make the precommit NO LONGER the live owner: append an exact precommit-removal ref-log + /// transaction exactly as an abandon / GC reclaim would (spec §Remove Precommit) -- via the SAME + /// public append lane a real abandon/reclaim would use, simulating an external actor this build + /// object does not know about (not this build's own `abandon()`, which would also retire it and + /// mask the "precommit no longer live" guard behind requireAlive()'s own rejection). + s->appendRefOps(ns, MutationScope::ref("part_1"), + [&](const RefTableState &) -> std::vector + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "part_1", id.ref}; + return {op}; + }, + RootMutationOrigin::Writer, RootMutationKind::Abandon); + + /// promote must fail closed: the precommit is no longer the live owner, so a Δ=0 move would dangle. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { build->promote(ns, "part_1", build->buildId(), id); }); + /// No ref committed. + EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); +} + +/// BUG 1 happy path: a promote whose precommit is STILL the live owner succeeds (the guard must not +/// reject the normal commit). Distinct from PublishHappyPathRoundTrip in that it pins the WPromote guard. +TEST(CASPartWriteTxn, PromoteSucceedsWhenPrecommitIsLiveOwner) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId id = build->stageManifest({blobManifestEntry("data.bin", "hello world")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); + + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + ASSERT_TRUE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_EQ(s->resolveRef(ns, "part_1")->manifest_id, id); +} + +/// all-tree-part-files Task 2 (TLA+ `WRepoint`): +/// `promote`'s existing unique-ref guard (BUG 1a) refuses to overwrite a committed ref naming a +/// DIFFERENT manifest -- correct for an ACCIDENTAL double-publish, but there is no way to perform an +/// INTENDED repoint (a standalone write/remove on an already-committed part) without it. `allow_repoint` +/// opts into exactly that: the guard's throw is skipped, and the committed-transition RefOp (old = +/// the currently-committed manifest, new = the incoming one) is appended in the SAME ref-log record as +/// the ordinary precommit->committed promotion -- the C++ realization of `WRepoint`'s one-event +/// old-binding/new-binding shape (Phase 0, task-1 gate). Without the flag, behavior is BYTE-IDENTICAL +/// to today (BUG 1a still fires). +TEST(CASPartWriteTxnRepoint, PromoteRepointsCommittedRef) +{ + auto b = std::make_shared(); + /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. + std::vector events; + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Publish ref "part_1" -> M1 through the normal build path. + auto build1 = startBuildFor(s, ns, "part_1"); + const ManifestId m1_id = build1->stageManifest({blobManifestEntry("data.bin", "m1")}); + build1->precommitAdd(ns, "part_1", m1_id); + build1->putBlob(idOf("m1"), BlobSource::fromString("m1")); + build1->promote(ns, "part_1", build1->buildId(), m1_id); + ASSERT_TRUE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_EQ(s->resolveRef(ns, "part_1")->manifest_id, m1_id); + + /// A second build stages M2 (one extra entry) onto the SAME ref. + auto build2 = startBuildFor(s, ns, "part_1"); + const ManifestId m2_id = build2->stageManifest({blobManifestEntry("data.bin", "m2"), blobManifestEntry("extra.bin", "m2x")}); + build2->precommitAdd(ns, "part_1", m2_id); + build2->putBlob(idOf("m2"), BlobSource::fromString("m2")); + build2->putBlob(idOf("m2x"), BlobSource::fromString("m2x")); + + /// allow_repoint = false (the default) -> NETWORK_ERROR (CAS write-retry-later), existing invariant + /// untouched; M1 still resolves. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { build2->promote(ns, "part_1", build2->buildId(), m2_id); }); + EXPECT_EQ(s->resolveRef(ns, "part_1")->manifest_id, m1_id); + + /// The failed no-flag attempt threw BEFORE appendRefOps returned, so build2's precommit is still the + /// live owner (no removal was appended) -- the SAME build/manifest can be retried with the flag. + s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + EXPECT_NO_THROW(build2->promote(ns, "part_1", build2->buildId(), m2_id, /*allow_repoint=*/true)); + auto resolved = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(resolved); + EXPECT_EQ(resolved->manifest_id.ref, m2_id.ref); + + /// Every effective repoint is loud (spec §4): exactly one RefRepoint event, naming the ref and the + /// old manifest it replaced. + size_t repoint_events = 0; + for (const CasEvent & e : events) + if (e.type == CasEventType::RefRepoint) + { + ++repoint_events; + EXPECT_EQ(e.ref_name, "part_1"); + EXPECT_EQ(e.detail.at("old_manifest"), manifestRefDebugString(m1_id.ref)); + } + EXPECT_EQ(repoint_events, 1u); +} + +/// BUG 2 (WAbandonPrecommit; delete-after-sealed-decrements): once `precommitAdd` has made a manifest a +/// LIVE precommit owner input, `abandon` must NOT writer-delete its body. The TLA+ `WAbandonPrecommit` +/// appends a REMOVAL event (`old = precommit(build_id, final_ref, T)`, `new = none`) and NEVER deletes +/// the body — GC decrements the precommit's blob edges and deletes the body only after the decrement is +/// sealed. Writer-deleting a live precommit body strands GC's fold barrier (live precommit, missing body +/// → clamp forever) or loses the activating +1. +TEST(CASPartWriteTxn, AbandonAppendsPrecommitRemovalAndKeepsLivePrecommitBody) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId mid = build->stageManifest({blobManifestEntry("data.bin", "kept")}); + const String manifest_key = s->layout().manifestKey(mid); + const UInt128 abandoned_build_id = build->buildId(); + build->precommitAdd(ns, "part_1", mid); + build->putBlob(idOf("kept"), BlobSource::fromString("kept")); + + /// The precommit manifest body is present before abandon. + ASSERT_TRUE(b->head(manifest_key).exists); + + build->abandon(); + + /// (a) the LIVE precommit body must SURVIVE abandon (left for GC after the sealed decrement). + EXPECT_TRUE(b->head(manifest_key).exists) + << "abandon must NOT writer-delete a live precommit body (delete-after-sealed-decrements)"; + + /// (b) the exact precommit binding is gone (spec §Remove Precommit: an exact owner_transition + /// removal, old=Precommit new=none). Proven black-box: a FRESH precommitAdd for the ref must + /// succeed -- if abandon had failed to remove the exact binding, this would instead throw + /// CORRUPTED_DATA ("add precommit ... already exists"). The probe manifest must be freshly staged + /// BY `rebuild` itself rather than re-precommitting `mid` (which `build`, a different transaction, + /// staged): A3 mint-tightening now refuses an unowned `ManifestId` from any transaction other than + /// the one that minted it, regardless of whether abandon's removal landed, so re-using `mid` here + /// would no longer distinguish the property under test. Content identity is irrelevant to the + /// ref-level owner-slot check this proves, so a fresh manifest is just as conclusive a probe. + (void)abandoned_build_id; + auto rebuild = startBuildFor(s, ns, "part_1"); + const ManifestId rebuild_mid = rebuild->stageManifest({blobManifestEntry("data.bin", "kept")}); + EXPECT_NO_THROW(rebuild->precommitAdd(ns, "part_1", rebuild_mid)); +} + +/// BUG 2 regression for the never-precommitted path: a manifest that was STAGED but never precommitted is +/// still best-effort writer-deleted by abandon (pre-precommit debris) — only a LIVE precommit body is +/// spared. Confirms the fix narrows the skip to the precommitted manifest exactly. +TEST(CASPartWriteTxn, AbandonStillDeletesNeverPrecommittedStagedDebris) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + + /// Two staged manifests: one becomes the precommit, the other is pure pre-precommit debris. + const ManifestId debris = build->stageManifest({blobManifestEntry("debris.bin", "kept")}); + const ManifestId precommitted = build->stageManifest({blobManifestEntry("data.bin", "kept")}); + build->precommitAdd(ns, "part_1", precommitted); + build->putBlob(idOf("kept"), BlobSource::fromString("kept")); + + build->abandon(); + + /// The never-precommitted debris is best-effort deleted; the live precommit body survives. + EXPECT_FALSE(b->head(s->layout().manifestKey(debris)).exists) + << "never-precommitted staged debris must still be best-effort deleted by abandon"; + EXPECT_TRUE(b->head(s->layout().manifestKey(precommitted)).exists) + << "the live precommit body must be spared"; +} + +/// Task 6 (review finding 2): `abandon()`'s three audit `EventEmitter{*store}.emit(...)` calls are each +/// wrapped `try { ... } catch (...) { tryLogCurrentException(...); }`, mirroring `promote`'s own +/// post-durable emit guard -- a throwing sink (e.g. a bad_alloc growing the system-log queue, or a +/// Context/log-shutdown edge) must never turn an otherwise-successful abandon into a reported failure. +TEST(CASPartWriteTxn, AbandonSwallowsThrowingEventSink) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl_abandon_sink"}; + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId mid = build->stageManifest({blobManifestEntry("data.bin", "kept")}); + build->precommitAdd(ns, "part_1", mid); + build->putBlob(idOf("kept"), BlobSource::fromString("kept")); + + /// UNKNOWN_EXCEPTION (not LOGICAL_ERROR): mirrors `PromoteSwallowsPostDurableEventSinkFailure` + /// above -- this simulates an arbitrary observer/sink callback failing, not a CAS invariant + /// violation. LOGICAL_ERROR would abort the whole process under debug/sanitizer builds instead of + /// behaving like a catchable exception. + s->setEventSink([](const CasEvent &) + { + throw DB::Exception(DB::ErrorCodes::UNKNOWN_EXCEPTION, "injected event sink failure"); + }); + + EXPECT_NO_THROW(build->abandon()); + s->setEventSink(nullptr); + + /// The precommit binding is gone despite the sink failure -- proven black-box exactly like + /// `AbandonAppendsPrecommitRemovalAndKeepsLivePrecommitBody` above: a FRESH precommitAdd for the + /// ref must succeed (it would instead throw CORRUPTED_DATA "add precommit ... already exists" had + /// the throwing sink aborted the removal). The probe manifest is freshly staged BY `rebuild` + /// itself, not `mid` (staged by `build`, a different transaction) -- A3 mint-tightening now refuses + /// a foreign id unconditionally, so re-using `mid` would no longer isolate the property under test. + auto rebuild = startBuildFor(s, ns, "part_1"); + const ManifestId rebuild_mid = rebuild->stageManifest({blobManifestEntry("data.bin", "kept")}); + EXPECT_NO_THROW(rebuild->precommitAdd(ns, "part_1", rebuild_mid)); +} + +namespace +{ + +/// Forces the SINGLE ref-log ('_log/' key) PUT that `abandon()`'s precommit-removal `appendRefOps` +/// issues to observe a PROVEN conflict instead of a genuine ambiguity. Mirrors +/// `RefWriterTestBackend::corrupt_key_substr` (gtest_cas_ref_writer.cpp, reproduced locally because +/// that class lives in a different translation unit): landing a DIFFERENT object at the intended key +/// makes `putIfAbsentControlled`'s resolve-before-reissue observe a real conflict and throw +/// CORRUPTED_DATA -- a CONCLUSIVE rejection ("do NOT wedge: the cache is unchanged and nothing of ours +/// is durable", CasRefLedger.cpp's `commitRefChunk`), unlike a genuinely-ambiguous timeout, which would +/// instead WEDGE the whole table's append lane (`rt->append_attempt`) until the SAME key resolves durable -- a +/// state a one-shot fault can never itself clear, since wedge resolution only re-GETs the intended key +/// and never re-PUTs it (proven by +/// `CASRefWriterAppendLane.WedgedLaneBlocksSameTableWhileOtherTableProceeds`). A conflict leaves the cached +/// ref-table state untouched, so the SAME logical retry reaches its append again -- and under INV-1 that +/// retry re-derives the SAME id from that unchanged state, so it meets the same occupant rather than +/// carving a fresh id around it. +class RefLogConflictOnceBackend final : public InMemoryBackend +{ +public: + String corrupt_key_substr; + int corrupt_count = 0; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (corrupt_count > 0 && !corrupt_key_substr.empty() && key.find(corrupt_key_substr) != String::npos) + { + --corrupt_count; + /// The 3-arg qualified call bypasses virtual dispatch entirely (unlike the 2-arg + /// convenience overload, which would re-enter this very override through the vtable). + InMemoryBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT"), meta); + throw Poco::TimeoutException("RefLogConflictOnceBackend: a foreign different object landed; response lost"); + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } +}; + +} + +/// Task 6 (review finding 2): `alive` now flips to false only AFTER the correctness-bearing precommit +/// removal's `appendRefOps` succeeds, so a caller that catches an append failure can retry `abandon()` +/// on the SAME object. Before the fix, `alive = false` ran unconditionally before that append, so a +/// retry would hit `requireAlive`'s "has been abandoned" LOGICAL_ERROR instead. +TEST(CASPartWriteTxn, AbandonRetryableAfterAppendFailure) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl_abandon_retry"}; + /// Stage B (Task 4-C): pin `ns`'s real incarnation to the Stage-A sentinel BEFORE the first real + /// append, so the corruption injected below (at a key computed from that sentinel) actually lands + /// on the path production writes to -- otherwise `precommitAdd` mints an unrelated random + /// incarnation and the corruption below misses it entirely. + DB::Cas::tests::casAdmitRecoverableEntry(*b, s->layout(), ns, s->liveWriterEpoch()); + auto build = startBuildFor(s, ns, "part_1"); + + const ManifestId mid = build->stageManifest({blobManifestEntry("data.bin", "kept")}); + build->precommitAdd(ns, "part_1", mid); + build->putBlob(idOf("kept"), BlobSource::fromString("kept")); + + b->corrupt_key_substr = s->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + b->corrupt_count = 1; + + /// First abandon(): the precommit-removal appendRefOps' single PUT observes a foreign object at its + /// exact key (a proven conflict) -> CORRUPTED_DATA propagates. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { build->abandon(); }); + + /// The proven conflict fences the mount closed and schedules a remount (the append site routes + /// through the anomaly policy exactly as the wedge-resolve site does). Re-arming only the test + /// fence does not replace this runtime, so its immutable admitted generation remains stale and its + /// terminal `Faulted` lane remains blocked behind that outer refusal. + DB::Cas::tests::rearmMountFenceAfterAnomalyForTest(s); + + /// The retryability under test: the SAME object accepts a second abandon() -- `alive` was not + /// flipped by the failed append, so this is not the "has been abandoned" condition the unfixed code + /// produced. It reaches immutable-runtime admission and is refused by the stale generation; only a + /// real remount may replace that runtime and reach a fresh lane. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->abandon(); }); + + /// The removal never landed, and nothing was written around the occupant: the precommit binding this + /// build owns is still live, exactly where the failed abandon left it. That is the honest end state + /// under the fail-closed contract -- the old proof (a fresh `precommitAdd` for the same ref + /// succeeding, which showed the binding gone) needed one more successful append on a table that can + /// no longer take one. + String greatest_key; + size_t foreign_objects = 0; + for (String cursor;;) + { + const ListPage page = b->list(b->corrupt_key_substr, cursor, 1000); + for (const auto & listed : page.keys) + { + if (listed.key > greatest_key) + greatest_key = listed.key; + const auto body = b->get(listed.key); + if (body && body->bytes.find("_FOREIGN_DIFFERENT") != String::npos) + ++foreign_objects; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + EXPECT_EQ(foreign_objects, 1u) << "the foreign object must still own the key it took"; + ASSERT_FALSE(greatest_key.empty()); + const auto greatest_body = b->get(greatest_key); + ASSERT_TRUE(greatest_body.has_value()); + EXPECT_NE(greatest_body->bytes.find("_FOREIGN_DIFFERENT"), String::npos) + << "the foreign occupant must still be the highest id in this table's stream: a log object above " + "it would mean an append carved a fresh id around the damage instead of failing closed"; +} + +/// ------------------------------------------------------------------------------------------------ +/// OQ7 manifest-cap fail-close (S07): the scenario suite tried to reach `stageManifest`'s encoded-bytes +/// cap through a wide-column SQL `INSERT`, but the cap sits 3+ orders of magnitude above what dev SQL +/// can reach in reasonable time (confirmed: even a 20000-column full-scale insert cannot get there). So +/// this P0 safety path is not scenario-testable and is exercised directly here instead. +/// ------------------------------------------------------------------------------------------------ + +namespace +{ + +/// Mirrors `CasPartWriteTxn.cpp`'s private `kMaxManifestEncodedBytes` (256 MiB). There is no way to read a +/// file-local `constexpr` from a different translation unit, so this is kept in sync by hand — if that +/// cap ever changes, update this one to match. +constexpr uint64_t kExpectedManifestEncodedCap = 256ULL << 20; + +/// The exact encoded size `PartWriteTxn::stageManifest` would compute for a single Blob-placement entry whose +/// path is `path_len` bytes long, staged under `ns` — measured through the SAME `encodePartManifest` +/// codec `stageManifest` calls, so this is an exact reproduction rather than a hand-derived estimate. +/// `ref` and `payload_digest` are fixed-width fields (20 and 16 bytes respectively): their VALUES don't +/// affect the encoded size, only their presence does, so the zero-valued placeholders here reproduce +/// the exact same byte count `stageManifest` would produce with its real (non-zero) values. +size_t manifestEncodedSizeForPathLen(const RootNamespace & ns, size_t path_len) +{ + PartManifest probe; + probe.root_namespace_id = ns; + ManifestEntry e; + e.path = String(path_len, 'a'); + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(UInt128{})}; + + e.blob_size = 12345; + probe.entries = {std::move(e)}; + return encodePartManifest(probe).size(); +} + +/// Finds the exact boundary: the SMALLEST `path_len` whose single-entry manifest encodes to MORE than +/// `kExpectedManifestEncodedCap` bytes under `ns`. `path_len - 1` therefore encodes to AT MOST the cap +/// (the encoding is monotonic in `path_len` — a longer path can only grow the encoded size). Starts +/// from a linear estimate (the encoding is affine in `path_len`: fixed framing overhead plus a constant +/// number of bytes per path byte) and walks to the exact crossing, so this stays correct even if the +/// framing overhead changes, without needing a full binary search over a ~256 MiB range. +size_t findManifestEncodedCapBoundaryPathLen(const RootNamespace & ns) +{ + constexpr size_t probe_lo = 1000; + constexpr size_t probe_hi = 2'000'000; + const size_t size_lo = manifestEncodedSizeForPathLen(ns, probe_lo); + const size_t size_hi = manifestEncodedSizeForPathLen(ns, probe_hi); + const double slope = static_cast(size_hi - size_lo) / static_cast(probe_hi - probe_lo); + const double intercept = static_cast(size_lo) - slope * static_cast(probe_lo); + + size_t path_len = static_cast(std::ceil( + (static_cast(kExpectedManifestEncodedCap) - intercept) / slope)) + 1; + + while (manifestEncodedSizeForPathLen(ns, path_len) <= kExpectedManifestEncodedCap) + ++path_len; + while (path_len > 1 && manifestEncodedSizeForPathLen(ns, path_len - 1) > kExpectedManifestEncodedCap) + --path_len; + return path_len; +} + +/// A one-entry Blob ManifestEntry with a synthetic `path_len`-byte path (used only to inflate the +/// encoded manifest size towards the OQ7 cap). +ManifestEntry wideBlobManifestEntry(size_t path_len) +{ + ManifestEntry e; + e.path = String(path_len, 'a'); + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(UInt128{0x42})}; + + e.blob_size = 12345; + return e; +} + +} + +/// Boundary case 1/2: a manifest whose encoded size is the LARGEST that still fits under the cap stages +/// successfully. Proves the cap enforcement isn't overly conservative — a real just-under-the-limit +/// manifest is not mistakenly rejected. +TEST(CASPartWriteTxn, ManifestCapEncodedBytesJustUnderStagesSuccessfully) +{ + auto b = std::make_shared(); + /// A frozen boot_ms_fn (not the shared openPool helper): this test's manifest sits just under the + /// 256 MiB cap, so encodePartManifest/sealObject do real, sizeable CPU work before the single + /// InMemoryBackend put (which always succeeds deterministically, no faults). Under heavy + /// instrumentation (TSan) that encode+seal step alone can take long enough in real wall-clock time + /// to cross the mount lease's fence margin (CasMountRuntime::refAppendFenceOk) and the CAS request + /// controller's own deadline (both consult the SAME injected clock, CasRefLedger.cpp) before the + /// attempt even resolves -- a sanitizer-speed artifact unrelated to what this test verifies. Freezing + /// the clock decouples the outcome from real execution speed: the single attempt now succeeds or + /// fails purely on the backend's own (deterministic) behavior, on any build. + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .boot_ms_fn = [] { return uint64_t{0}; }}); + const RootNamespace ns{"srv1/tbl"}; + + const size_t path_len_over = findManifestEncodedCapBoundaryPathLen(ns); + ASSERT_GT(path_len_over, 1u); + const size_t path_len_under = path_len_over - 1; + ASSERT_LE(manifestEncodedSizeForPathLen(ns, path_len_under), kExpectedManifestEncodedCap); + + auto build = startBuildFor(s, ns, "wide_part"); + const ManifestId id = build->stageManifest({wideBlobManifestEntry(path_len_under)}); + EXPECT_EQ(id.root_namespace, ns); + EXPECT_TRUE(b->head(s->layout().manifestKey(id)).exists) + << "a just-under-cap manifest must actually be written"; +} + +/// Boundary case 2/2: a manifest whose encoded size exceeds the cap by the smallest possible margin +/// (one more path byte than the passing case above) throws `LIMIT_EXCEEDED` fail-closed, BEFORE the body +/// write — no manifest object lands in the backend for the rejected attempt. +TEST(CASPartWriteTxn, ManifestCapEncodedBytesOverThrowsBeforeBodyWrite) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + const size_t path_len_over = findManifestEncodedCapBoundaryPathLen(ns); + ASSERT_GT(manifestEncodedSizeForPathLen(ns, path_len_over), kExpectedManifestEncodedCap); + + auto build = startBuildFor(s, ns, "wide_part"); + + const size_t keys_before = b->list("", "", 100).keys.size(); + bool threw = false; + try + { + build->stageManifest({wideBlobManifestEntry(path_len_over)}); + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::LIMIT_EXCEEDED); + EXPECT_NE(e.message().find("exceeds cap"), String::npos) << e.message(); + } + EXPECT_TRUE(threw) << "an over-cap manifest must throw, not silently truncate or accept"; + + /// Fail-closed BEFORE the body write: the over-cap attempt must not have created ANY new object + /// (no partial state, no orphaned blob/manifest debris for a manifest that was never accepted). + const size_t keys_after = b->list("", "", 100).keys.size(); + EXPECT_EQ(keys_before, keys_after) + << "stageManifest must fail closed before writing the manifest body, leaving no new objects"; +} + +/// spec §9.9 (mixed-algo pools, Phase 3 T2) — the W-DEP-SET cross-satisfaction crux: a manifest with +/// two entries carrying the SAME digest VALUE under TWO DIFFERENT algos (`ch128:X` / `xxh3:X`). Only +/// `ch128:X`'s body is ever putBlob'd; `xxh3:X`'s body never lands anywhere. Promote MUST fail closed — +/// the materialized `ch128:X` proof must never satisfy the missing `xxh3:X` proof. +/// This test is RED (wrongly passes / silently promotes) if `PartWriteTxn::deps` (the W-DEP-SET) were keyed on +/// a bare digest instead of the full `BlobRef` pair: both entries would collapse to the SAME map key +/// (the digest alone), so the proof query would report the xxh3 leaf as edge-protected via the ch128 +/// entry's putBlob and promote would skip its revalidation (and hence its absence) entirely. +TEST(CASPartWriteTxn, WDepSetCrossAlgoSatisfactionFailsClosed) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl"}; + + const BlobDigest shared_digest = BlobDigest::fromU128(u128Of("shared-digest-value")); + + ManifestEntry e_ch128; + e_ch128.path = "a.bin"; + e_ch128.placement = EntryPlacement::Blob; + e_ch128.ref = BlobRef{BlobHashAlgo::CityHash128, shared_digest}; + e_ch128.blob_size = 3; + + ManifestEntry e_xxh3; + e_xxh3.path = "b.bin"; + e_xxh3.placement = EntryPlacement::Blob; + e_xxh3.ref = BlobRef{BlobHashAlgo::XXH3_128, shared_digest}; /// SAME digest bytes, DIFFERENT algo + e_xxh3.blob_size = 3; + + auto build = startBuildFor(s, ns, "part_mixed"); + const ManifestId id = build->stageManifest({e_ch128, e_xxh3}); + build->precommitAdd(ns, "part_mixed", id); + + /// Only the ch128 leaf's body is ever uploaded — its BlobId hex is the digest at the ch128 width, + /// which addresses EXACTLY `e_ch128`'s object key (`blobs/ch128/...`), a DISTINCT key from + /// `e_xxh3`'s (`blobs/xxh3/...`), even though the raw digest bytes are identical. + build->putBlob(BlobRef{BlobHashAlgo::CityHash128, shared_digest}, BlobSource::fromString("abc")); + + /// Promotion must fail closed: the xxh3:X leaf has no dependency proof — never silently + /// satisfied by the ch128:X entry's materialized proof (same digest bytes, distinct object key). §4 + /// manifest-trust catches an unsatisfied leaf by the dependency set and + /// fails closed with LOGICAL_ERROR — a staging bug — without any backend probe on the xxh3 key. + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->promote(ns, "part_mixed", build->buildId(), id); + }, + "no dependency proof"); + + /// No committed ref appears — the promote aborted before installing one. + EXPECT_FALSE(s->resolveRef(ns, "part_mixed").has_value()); +} + +/// ===================================================================================== +/// Task B (chaos-tolerance-report §Task B): stageManifest's part-manifest conditional PUT rides the +/// shared CasRequestController — budgeted attempts + resolve-before-reissue — instead of the old +/// single bare attempt (which a 19s object-store pause killed while every read path survived). +/// ===================================================================================== + +namespace +{ + +/// Faults the part-manifest body PUT (`/cas/manifests/` keys) with an ambiguous +/// (Unresolved-classified) timeout a bounded number of times, mirroring RefWriterTestBackend's fault +/// seam (gtest_cas_ref_writer.cpp). Part manifests use the small-object `putIfAbsent` primitive. +class ManifestPutFaultBackend final : public InMemoryBackend +{ +public: + int fault_count = 0; /// remaining ambiguous faults on matching body PUTs + bool land_despite_fault = false; /// the faulted attempt's own write actually lands (response lost) + String plant_different_on_fault; /// a FOREIGN different body lands at the key before the fault + int put_attempts = 0; /// matching body-PUT attempts observed + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (!isManifestBodyKey(key)) + return InMemoryBackend::putIfAbsent(key, bytes, meta); + ++put_attempts; + maybeFault(key, bytes); + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + +private: + static bool isManifestBodyKey(const String & key) { return key.find("/cas/manifests/") != String::npos; } + + /// One fault: apply the configured server-side effect, then lose the response. + void maybeFault(const String & key, const String & bytes) + { + if (fault_count <= 0) + return; + --fault_count; + if (!plant_different_on_fault.empty()) + InMemoryBackend::putIfAbsent(key, plant_different_on_fault, {}); + else if (land_despite_fault) + InMemoryBackend::putIfAbsent(key, bytes, {}); + throw Poco::TimeoutException("ManifestPutFaultBackend: simulated ambiguous result (response lost)"); + } +}; + +} + +/// The Task B core: two consecutive ambiguous timeouts on the part-manifest body PUT (each resolved +/// to "absent" by the controller's exact-GET), then a clean third attempt. The old single-attempt +/// path fails the whole stage on the FIRST timeout (the observed 19s-pause INSERT kill); the +/// controller path must ride its attempt budget and succeed. +TEST(CASPartWriteTxnStageManifestRetry, AmbiguousTimeoutsThenCommitSucceedsWithinBudget) +{ + /// Zero backoff: the retry semantics are under test here, not the (controller-level-tested) + /// inter-attempt sleep schedule — keep the suite free of real sleeps. + CasRequestBudget budget; + budget.retry_initial_backoff_ms = 0; + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + const RootNamespace ns{"srv/tbl"}; + + auto build = startBuildFor(s, ns, "part_retry"); + b->fault_count = 2; + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", "a")}); + + EXPECT_EQ(b->put_attempts, 3) << "two faulted attempts + the committing third"; + const auto got = b->get(s->layout().manifestKey(id)); + ASSERT_TRUE(got.has_value()) << "the staged manifest body must be durable"; + EXPECT_EQ(decodePartManifest(openObject(FormatId::PartManifest, got->bytes)).ref, id.ref); +} + +/// Ambiguous-but-landed: the FIRST attempt's response is lost AFTER the write actually landed +/// server-side. Resolve-before-reissue's exact-GET observes the identical bytes and reports +/// Committed — the stage succeeds WITHOUT a reissue (no duplicate PUT of the object), and the +/// `ManifestPut` audit event carries the landed incarnation's token (from the resolve GET). +TEST(CASPartWriteTxnStageManifestRetry, AmbiguousLandedWriteResolvesToCommittedWithoutReissue) +{ + auto b = std::make_shared(); + /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. + std::vector events; + auto s = openPool(b); + const RootNamespace ns{"srv/tbl"}; + + s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + + auto build = startBuildFor(s, ns, "part_landed"); + b->fault_count = 1; + b->land_despite_fault = true; + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", "a")}); + + EXPECT_EQ(b->put_attempts, 1) << "a landed ambiguous attempt must be resolved, never reissued"; + const String key = s->layout().manifestKey(id); + ASSERT_TRUE(b->get(key).has_value()); + + const auto ev = std::find_if(events.begin(), events.end(), + [](const CasEvent & e) { return e.type == CasEventType::ManifestPut; }); + ASSERT_NE(ev, events.end()) << "the stage must still emit its ManifestPut audit event"; + EXPECT_EQ(ev->token, b->head(key).token.value) + << "the audit token must be the landed incarnation's token"; +} + +/// A DIFFERENT object at the exact staged key (a foreign body ahead of our ambiguous attempt) is a +/// proven conflict — the NoManifestIdReuse invariant broke — and must stay the loud CORRUPTED_DATA +/// class: never a retry signal, never silently adopted. +TEST(CASPartWriteTxnStageManifestRetry, DifferentObjectAtKeyStaysLoudConflict) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl"}; + + auto build = startBuildFor(s, ns, "part_conflict"); + b->fault_count = 1; + b->plant_different_on_fault = "a-foreign-different-manifest-body"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + build->stageManifest({blobManifestEntry("a.bin", "a")}); + }); + EXPECT_EQ(b->put_attempts, 1) << "a proven conflict is never retried"; +} + +/// Budget exhaustion: EVERY attempt is ambiguous and nothing ever lands. The controller reports +/// Unresolved after `max_attempts` and stageManifest maps it to NETWORK_ERROR (fix #37 phase 2) — +/// the same retryable abort class the ref-log lane's exhausted budget maps to. Nothing was durably +/// named: the caller re-stages with a fresh ManifestId. +TEST(CASPartWriteTxnStageManifestRetry, BudgetExhaustionMapsToNetworkError) +{ + CasRequestBudget budget; + budget.max_attempts = 3; + budget.retry_initial_backoff_ms = 0; /// no real sleeps; the backoff schedule has its own tests + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + const RootNamespace ns{"srv/tbl"}; + + auto build = startBuildFor(s, ns, "part_exhausted"); + b->fault_count = 1000; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->stageManifest({blobManifestEntry("a.bin", "a")}); + }); + EXPECT_EQ(b->put_attempts, 3) << "attempts must be bounded by the configured budget"; +} + +/// ===================================================================================== +/// Blob publication retries re-stream from the writer's replayable source. A retry after a failed +/// server-side copy retags and streams from the intact source instead of repeating the copy. +/// ===================================================================================== + +namespace +{ + +/// Faults unconditional blob publications with an ambiguous timeout a bounded number of times. +/// Blob metadata writes (`.meta` keys, plain `putIfAbsent`) are never faulted. +class BlobPutFaultBackend final : public InMemoryBackend +{ +public: + int fault_count = 0; /// remaining ambiguous faults on matching create attempts + bool land_despite_fault = false; /// the faulted attempt's own write actually lands (response lost) + int publish_stream_attempts = 0; /// unconditional streaming publications observed + int publish_copy_attempts = 0; /// unconditional native-copy publications observed + int blob_head_attempts = 0; /// transaction-level blob observations + + HeadResult head(const String & key) override + { + if (isBlobBodyKey(key)) + ++blob_head_attempts; + return InMemoryBackend::head(key); + } + + void publishBlob(const BlobPublishRequest & request) override + { + if (std::holds_alternative(request.publication)) + ++publish_stream_attempts; + else + ++publish_copy_attempts; + + if (fault_count > 0) + { + --fault_count; + if (land_despite_fault) + InMemoryBackend::publishBlob(request); + else if (const auto * streaming = std::get_if(&request.publication)) + (void)streaming->open_payload(); + throw Poco::TimeoutException("BlobPutFaultBackend: simulated ambiguous publication (response lost)"); + } + InMemoryBackend::publishBlob(request); + } + +private: + static bool isBlobBodyKey(const String & key) + { + return key.find("/blobs/") != String::npos && !key.ends_with(".meta"); + } + +}; + +/// Zero-backoff store over a BlobPutFaultBackend: the sleep schedule has its own controller-level +/// tests; these Pool-level tests pin the retry/resolve/abort semantics without real sleeps. +PoolPtr openBlobFaultPool(const std::shared_ptr & b, uint32_t max_attempts = CasRequestBudget{}.max_attempts) +{ + CasRequestBudget budget; + budget.max_attempts = max_attempts; + budget.retry_initial_backoff_ms = 0; + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); +} + +/// A replayable BlobSource that COUNTS its own re-streams — pins INV-1's "retry = fresh re-stream +/// from the writer's own source" (never a GET of the dying/failed object). +BlobSource countingSource(const String & payload, int & payload_streams) +{ + BlobSource source; + source.size = payload.size(); + source.open = [payload, &payload_streams]() -> std::unique_ptr + { + ++payload_streams; + return std::make_unique(payload); + }; + return source; +} + +} + +/// The core ride: two consecutive ambiguous timeouts on the blob-body streaming PUT (each resolved +/// "absent" by the controller's occupancy HEAD), then a clean third attempt. The old single-attempt +/// path failed the whole INSERT on the FIRST timeout (the raw Poco::TimeoutException escaped +/// putBlob); the controller path rides its budget, RE-STREAMING the payload from the writer's own +/// replayable source on every attempt. +TEST(CASPartWrite, AmbiguousTimeoutsThenCommitRestreamsFromSource) +{ + auto b = std::make_shared(); + auto s = openBlobFaultPool(b); + const RootNamespace ns{"srv/tbl"}; + const String payload = "blob-payload-A"; + + auto build = startBuildFor(s, ns, "part_blob_retry"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_blob_retry", id); + + int payload_streams = 0; + b->fault_count = 2; + const PutBlobResult res = build->putBlob(idOf(payload), countingSource(payload, payload_streams)); + EXPECT_EQ(res.size, payload.size()); + + EXPECT_EQ(b->publish_stream_attempts, 3) << "two ambiguous publications + the committing third"; + EXPECT_EQ(b->blob_head_attempts, 3) << "every outer retry restarts from a fresh blob HEAD"; + EXPECT_EQ(payload_streams, 3) << "every reissue must RE-STREAM from the writer's own source (INV-1)"; + EXPECT_TRUE(b->head(s->layout().blobKey(idOf(payload))).exists) << "the blob body must be durable"; +} + +/// Ambiguous-but-landed: the FIRST attempt's response is lost AFTER the write actually landed +/// server-side. The occupancy resolve observes the key present and the existing 412 machinery takes +/// over — the occupant is ADOPTED (content-addressed identity: any occupant of this key IS the +/// content), with NO reissue and NO second body upload. +TEST(CASPartWrite, AmbiguousLandedWriteAdoptsOccupantWithoutReupload) +{ + auto b = std::make_shared(); + /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. + std::vector events; + auto s = openBlobFaultPool(b); + const RootNamespace ns{"srv/tbl"}; + const String payload = "blob-payload-B"; + + s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + + auto build = startBuildFor(s, ns, "part_blob_landed"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_blob_landed", id); + + int payload_streams = 0; + b->fault_count = 1; + b->land_despite_fault = true; + const PutBlobResult res = build->putBlob(idOf(payload), countingSource(payload, payload_streams)); + EXPECT_EQ(res.size, payload.size()); + + EXPECT_EQ(b->publish_stream_attempts, 1) << "a landed ambiguous attempt must be observed, never reissued"; + EXPECT_EQ(b->blob_head_attempts, 2) << "the ambiguity is resolved by restarting from HEAD"; + EXPECT_EQ(payload_streams, 1); + + const String key = s->layout().blobKey(idOf(payload)); + const auto adopt = std::find_if(events.begin(), events.end(), + [](const CasEvent & e) { return e.type == CasEventType::BlobReuseAdopt; }); + ASSERT_NE(adopt, events.end()) << "the landed occupant must be ADOPTED (the standard dedup leg)"; + EXPECT_EQ(adopt->token, b->head(key).token.value) << "the adopted token must be the landed incarnation's"; + EXPECT_EQ(std::count_if(events.begin(), events.end(), + [](const CasEvent & e) { return e.type == CasEventType::BlobPut; }), 0) + << "no fresh-upload event: the body was never re-uploaded"; +} + +/// Budget exhaustion: EVERY attempt is ambiguous and nothing ever lands. The controller reports the +/// uncertainty and `ensureBlobPresent` maps it to `NETWORK_ERROR` -- the same retryable +/// abort class stageManifest and the ref-log lane map their exhausted budgets to. Unlike the OLD +/// ABORTED mapping, putBlob's bounded condemned-churn loop (8 rounds) does NOT re-drive this: it only +/// catches ABORTED, so a NETWORK_ERROR escapes on the FIRST attempt -- desirable (no point hammering a +/// lost fence locally 8 times; the caller's own backoff, e.g. the merge queue's, is what should retry). +TEST(CASPartWrite, AmbiguousNonLandingPublicationStopsAtOuterBound) +{ + auto b = std::make_shared(); + auto s = openBlobFaultPool(b, /*max_attempts=*/3); + const RootNamespace ns{"srv/tbl"}; + const String payload = "blob-payload-C"; + + auto build = startBuildFor(s, ns, "part_blob_exhausted"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_blob_exhausted", id); + + int payload_streams = 0; + b->fault_count = 1000000; + bool threw = false; + try + { + build->putBlob(idOf(payload), countingSource(payload, payload_streams)); + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(e.message().find("ambiguous"), String::npos) << e.message(); + } + EXPECT_TRUE(threw); + EXPECT_EQ(b->publish_stream_attempts, 8) << "the writer's correctness retry loop is bounded"; + EXPECT_EQ(b->blob_head_attempts, 8) << "every ambiguous retry is preceded by a new observation"; + EXPECT_EQ(payload_streams, 8); +} + +/// A server-side copy publication is ambiguous-but-landed: its response is lost after the +/// destination was created. The occupancy resolve observes the destination present and the occupant +/// is adopted — Committed-in-effect WITHOUT a re-copy. +TEST(CASPartWrite, AmbiguousCopyLandedAdoptsDestinationWithoutRecopy) +{ + auto b = std::make_shared(); + /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. + std::vector events; + auto s = openBlobFaultPool(b); + const RootNamespace ns{"srv/tbl"}; + const String payload = "staged-payload-A"; + /// The staging object: [pool-fixed-length envelope header][payload], promoted VERBATIM by the copy. + const String staging_key = "p/staging/test/blob-a"; + const String staging_bytes = String(s->poolMeta().blob_header_len, 'h') + payload; + ASSERT_EQ(b->putIfAbsent(staging_key, staging_bytes).outcome, PutOutcome::Done); + + s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + + auto build = startBuildFor(s, ns, "part_copy_landed"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_copy_landed", id); + + BlobSource source; + source.size = payload.size(); + source.server_side_copy_from = staging_key; + b->fault_count = 1; + b->land_despite_fault = true; + const PutBlobResult res = build->putBlob(idOf(payload), std::move(source)); + EXPECT_EQ(res.size, payload.size()); + + EXPECT_EQ(b->publish_copy_attempts, 1) << "a landed ambiguous copy must be resolved, never re-copied"; + EXPECT_EQ(b->publish_stream_attempts, 0); + EXPECT_EQ(b->blob_head_attempts, 2); + const String key = s->layout().blobKey(idOf(payload)); + const auto got = b->get(key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, staging_bytes) << "the destination is the staging object's verbatim copy"; + EXPECT_NE(std::find_if(events.begin(), events.end(), + [](const CasEvent & e) { return e.type == CasEventType::BlobReuseAdopt; }), + events.end()) << "the landed destination must be ADOPTED"; +} + +/// A server-side copy publication is ambiguous-and-absent: the first copy attempt times out with +/// nothing landing; the resolve observes the destination absent and the copy is REISSUED from the +/// (intact, still-staged) source object — the second attempt commits. +TEST(CASPartWrite, AmbiguousCopyAbsentReattemptsAndCommits) +{ + auto b = std::make_shared(); + auto s = openBlobFaultPool(b); + const RootNamespace ns{"srv/tbl"}; + const String payload = "staged-payload-B"; + const String staging_key = "p/staging/test/blob-b"; + const String staging_bytes = String(s->poolMeta().blob_header_len, 'h') + payload; + ASSERT_EQ(b->putIfAbsent(staging_key, staging_bytes).outcome, PutOutcome::Done); + + auto build = startBuildFor(s, ns, "part_copy_retry"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_copy_retry", id); + + BlobSource source; + source.size = payload.size(); + source.server_side_copy_from = staging_key; + source.open = [payload]() -> std::unique_ptr + { + return std::make_unique(payload); + }; + b->fault_count = 1; + const PutBlobResult res = build->putBlob(idOf(payload), std::move(source)); + EXPECT_EQ(res.size, payload.size()); + + EXPECT_EQ(b->publish_copy_attempts, 1) << "only the first absent observation may select verbatim copy"; + EXPECT_EQ(b->publish_stream_attempts, 1) << "the absent retry must retag and stream"; + EXPECT_EQ(b->blob_head_attempts, 2); + const String key = s->layout().blobKey(idOf(payload)); + const auto got = b->get(key); + ASSERT_TRUE(got.has_value()); + EXPECT_NE(got->bytes, staging_bytes); + EXPECT_EQ(got->bytes.substr(s->poolMeta().blob_header_len), payload); +} diff --git a/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp b/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp new file mode 100644 index 000000000000..b1022505585d --- /dev/null +++ b/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp @@ -0,0 +1,256 @@ +#include +#include +#include +#include +#include +#include +#include + +#include + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// Mirrors the B140 repro. +PoolPtr openTestPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +size_t runGcToFixpoint(Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + RoundReport rep; + try + { + rep = gc.runRegularRound(); + } + catch (const DB::Exception &) + { + break; + } + if (!rep.acquired_lease) + continue; + if (rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0) + break; + } + return rounds; +} + +ManifestEntry blobEntry(const String & name, const String & payload) +{ + ManifestEntry e; + e.path = name; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + return e; +} + +} + +/// B171 build-root / precommit, RED repro of the B140-dangle at unit level driven entirely through the +/// public PartWriteTxn/Pool/Gc API (no snap injection): +/// +/// PartWriteTxn A uploads blob P and publishes refA -> t1 -> { data.bin: P }. A is then RELEASED (dtor), +/// retiring its build_seq so the GC watermark `min_active` advances PAST A. P now carries A's +/// `cas_owner` and is no longer protected by any in-flight build. +/// +/// PartWriteTxn B starts and ADOPTS the same blob P via tokenless evidence (adoptEvidence — the cross-node +/// adopt case), assembles t2 -> { other.bin: P }, and `precommit(t2)` — which publishes a durable +/// build-root ref so GC's fold lifts the in-degree of P's closure. +/// +/// refA is dropped + watermark renewed; GC runs to fixpoint. P is protected by B's precommit edge, +/// so GC must NOT delete it. PartWriteTxn B then publishes refB -> t2 successfully. +/// +/// THE POSITIVE INVARIANT: the whole flow must succeed AND P must survive, because B's precommit pins +/// P's closure across A's retire + GC (B171 two-phase commit; `checkAndResolveDeps` proves closure +/// present at publish time). +TEST(CASPartWriteTxnRootDangle, SharedBlobSurvivesSourceDropDuringBuild) +{ + std::shared_ptr backend; + auto s = openTestPool(backend); + const RootNamespace ns{"test/tbl"}; + const String P = "shared-blob-payload-P"; + + /// PartWriteTxn A: upload P, publish refA -> manifest -> { data.bin: P }, then release A so its build_seq + /// retires and min_active advances past it. + { + PartWriteInfo info; + info.intended_ref = ns.string() + "/refA"; + auto a = s->beginPartWrite(info); + const ManifestId id = a->stageManifest({blobEntry("data.bin", P)}); + a->precommitAdd(ns, "refA", id); + a->putBlob(idOf(P), BlobSource::fromString(P)); + a->promote(ns, "refA", a->buildId(), id); + } + s->renewWatermarkOnce(); /// A is gone; min_active now advances past A's build_seq + + /// PartWriteTxn B: adopt the SAME blob P (cross-node adopt — tokenless evidence via adoptEvidence), assemble + /// its manifest, and precommitAdd it. The precommit pins P's closure (fold +1 edge) for the build. + PartWriteInfo binfo; + binfo.intended_ref = ns.string() + "/refB"; + auto b = s->beginPartWrite(binfo); + const ManifestEntry pe = blobEntry("other.bin", P); + b->adoptEvidence(pe); + const ManifestId t2 = b->stageManifest({pe}); + b->precommitAdd(ns, "refB", t2); + + /// The source ref disappears, and the watermark is renewed so the closure looks collectable. + s->dropRef(ns, "refA"); + s->renewWatermarkOnce(); + + /// GC to fixpoint. P must survive: the live precommit binding for refB activates a +1 blob edge on + /// P during the fold, so P never reaches in-degree 0 (B171 two-phase commit). + Gc gc(s, u128Of("gc-b171")); + runGcToFixpoint(gc); + + /// PartWriteTxn B commits refB by promoting its precommit. Should succeed end-to-end; if it throws (e.g. + /// ABORTED because the blob is gone) that is itself the RED outcome. + ASSERT_NO_THROW(b->promote(ns, "refB", b->buildId(), t2)) + << "B171: PartWriteTxn B's promote must succeed — the precommit should have kept P alive"; + + /// The blob B references must still be present (no dangle), and refB must resolve. + ASSERT_TRUE(backend->head(s->layout().blobKey(idOf(P))).exists) + << "B171-dangle: GC deleted the shared blob P that PartWriteTxn B adopted — its cas_owner was the " + << "retired PartWriteTxn A and the stub precommit published no build-root edge, so inDeg(P) hit 0 " + << "and the single content-delete site removed it. refB now dangles."; + ASSERT_TRUE(s->resolveRef(ns, "refB").has_value()) + << "B171: refB must resolve to its committed manifest"; +} + +/// B171 INV-COMMIT-FAILCLOSED: even if the build-root precommit is PREMATURELY RECLAIMED mid-build +/// (e.g. a live build whose watermark renewer froze and was falsely judged dead), the real commit must +/// NEVER publish a table ref over a missing dependency. It must fail closed — abort — never dangle. +/// +/// Setup mirrors the primary repro: PartWriteTxn A publishes refA -> t1 -> { data.bin: P } then retires; PartWriteTxn +/// B adopts P, assembles t2 -> { other.bin: P }, and precommits t2 (a real build-root edge now protects +/// P). We then SIMULATE the premature reclaim by manually dropping the build-root ref (as GC's reclaim +/// would) AND dropping refA, then renew the watermark and run GC to fixpoint. With P's only protection +/// (the precommit edge) gone and its owner retired, GC deletes P. PartWriteTxn B's publish must now ABORT +/// (`checkAndResolveDeps` finds the adopted blob absent and not re-creatable) instead of committing a dangle. +TEST(CASPartWriteTxnRootDangle, PrematureReclaimCommitFailsClosed) +{ + std::shared_ptr backend; + auto s = openTestPool(backend); + const RootNamespace ns{"test/tbl"}; + const String P = "shared-blob-payload-P-reclaim"; + + /// PartWriteTxn A: upload P, publish refA -> manifest, retire A so min_active advances past it. + { + PartWriteInfo info; + info.intended_ref = ns.string() + "/refA"; + auto a = s->beginPartWrite(info); + const ManifestId id = a->stageManifest({blobEntry("data.bin", P)}); + a->precommitAdd(ns, "refA", id); + a->putBlob(idOf(P), BlobSource::fromString(P)); + a->promote(ns, "refA", a->buildId(), id); + } + s->renewWatermarkOnce(); + + /// PartWriteTxn B: adopt P via tokenless evidence, assemble its manifest, precommitAdd it (the precommit + /// owner binding for refB now protects P with a +1 fold edge). + PartWriteInfo binfo; + binfo.intended_ref = ns.string() + "/refB"; + auto b = s->beginPartWrite(binfo); + const ManifestEntry pe2 = blobEntry("other.bin", P); + b->adoptEvidence(pe2); + const ManifestId t2 = b->stageManifest({pe2}); + b->precommitAdd(ns, "refB", t2); + + /// SIMULATE a premature reclaim having already collected P: had the precommit binding been wrongly + /// reclaimed with no other owner, GC would condemn+delete P's closure. Reproduce that END STATE + /// directly by deleting P's blob object. (The durable ref-log stream is owned by the live writer, so a + /// RAW removal append would collide with the writer's own `RefTxnId` sequence allocation on the next + /// flush; the property under test is the COMMIT gate's fail-closed behavior against a missing + /// dependency, not the reclaim mechanics -- so we go straight to the reclaimed state.) + { + const String pkey = s->layout().blobKey(idOf(P)); + const HeadResult h = backend->head(pkey); + ASSERT_TRUE(h.exists) << "P must be present before the simulated reclaim"; + ASSERT_EQ(backend->deleteExact(pkey, h.token).kind, DeleteOutcome::Kind::Deleted); + } + /// Drop the source ref too (the state a real premature reclaim leaves: P unprotected and gone). + s->dropRef(ns, "refA"); + s->renewWatermarkOnce(); + + /// The shared blob must be GONE (the premature reclaim collected it). + ASSERT_FALSE(backend->head(s->layout().blobKey(idOf(P))).exists) + << "premature-reclaim setup invalid: P should have been collected after losing its precommit"; + + /// §4 manifest-trust (test name is legacy — B171 INV-COMMIT-FAILCLOSED for an ADOPTED leaf now moves to + /// fsck): P is a committed-source adopted leaf, so PartWriteTxn B's promote TRUSTS it (no HEAD/loadMeta probe) + /// and COMMITS refB. On the real reuse/relink path this dangle is UNREACHABLE: precommitAdd durably + /// appended refB's Precommit OwnerTransition (CasPartWriteTxn.cpp precommitAdd) BEFORE promote, and promote + /// re-proves that edge is the LIVE owner (WPromote owner==bld) BEFORE trusting P — so P has in-degree + /// >= 1 and GC (the sole deleter) cannot collect it. This test injects the collection DIRECTLY (a raw + /// deleteExact while refB's precommit is still live), which the live-precommit invariant excludes. So + /// promote SUCCEEDS; the dangle is not prevented at promote but DETECTED by fsck (the backstop). + ASSERT_NO_THROW(b->promote(ns, "refB", b->buildId(), t2)) + << "§4: an adopted leaf is trusted at promote — a missing dependency is not re-observed here"; + + /// Trust never fabricates the missing blob (it never touches P); refB IS committed (naming absent P). + ASSERT_FALSE(backend->head(s->layout().blobKey(idOf(P))).exists) + << "trust never fabricates the missing blob — P stays absent"; + ASSERT_TRUE(s->resolveRef(ns, "refB").has_value()) + << "§4: refB commits under trust (the D4 trade-off); the dangle is caught by fsck, below"; + + /// THE BACKSTOP (INV-NO-DANGLE-via-fsck): fsck's reachable-but-absent scan reports refB's absent P as + /// dangling — this is where the B171 guarantee lives under §4. Detection moved, it did not disappear. + const FsckReport rep = runFsck(*s, /*detail=*/true); + EXPECT_GE(rep.dangling, 1u) + << "§4 D4 backstop: refB committed over the deleted P; fsck must report it dangling (dangling=" + << rep.dangling << ", reachable=" << rep.reachable << ")"; +} + +/// (The GC-reclaim test `CASPartWriteTxnRoot.AbandonedPrecommitReclaimed` -- which asserted GC AUTOMATICALLY +/// reclaims an abandoned precommit of a judged-dead build and then collects its closure -- was removed +/// with the snapshot+log ref model. Per spec §Responsibility Boundary, reclaiming an abandoned precommit +/// is now the WRITER's job (it appends the exact `owner_transition` removal on recovery); GC never scans +/// for or removes precommit bindings, and there is no mutable shard journal to append a `PrecommitRemove` +/// into. The `precommitRemovalAppended` shard-journal probe it shared with `LivePrecommitNotReclaimed` +/// went with it.) + +/// B8 CONSERVATISM (liveness-correctness guard): a live in-flight build's precommit binding (and its +/// pinned blobs) must survive a full GC run, and the build must still be able to promote it. In the +/// snapshot+log model GC never reclaims a precommit at all, so this is purely a liveness pin: the live +/// precommit's `+1` fold edge keeps its exclusively-owned blob alive across GC. +TEST(CASPartWriteTxnRoot, LivePrecommitNotReclaimed) +{ + std::shared_ptr backend; + auto s = openTestPool(backend); + const RootNamespace ns{"test/tbl"}; + const String Q = "live-build-blob-payload-Q"; + + /// PartWriteTxn B stays ALIVE: upload Q, assemble, precommitAdd — and we DO NOT retire its seq. So + /// `min_active <= build_seq` (B is in-flight) and the watermark keeps a live, advancing seq. + PartWriteInfo binfo; + binfo.intended_ref = ns.string() + "/refLive"; + auto b = s->beginPartWrite(binfo); + const ManifestId t = b->stageManifest({blobEntry("data.bin", Q)}); + b->precommitAdd(ns, "refLive", t); + b->putBlob(idOf(Q), BlobSource::fromString(Q)); + s->renewWatermarkOnce(); + ASSERT_LE(s->minActive(), b->buildSeq()) << "precondition: B must be in-flight (min_active <= seq)"; + + /// GC to fixpoint while B is live. + Gc gc(s, u128Of("gc-b8-live")); + runGcToFixpoint(gc); + + /// Q must still be present (the live precommit's +1 edge pins it across GC). + ASSERT_TRUE(backend->head(s->layout().blobKey(idOf(Q))).exists) + << "B8 conservatism: the live precommit must keep its blob alive across GC"; + + /// B can still commit (the precommit is intact). + ASSERT_NO_THROW(b->promote(ns, "refLive", b->buildId(), t)) + << "B8 conservatism: a live build must still be able to promote its untouched precommit"; +} diff --git a/src/Disks/tests/gtest_cas_pluggable_hash.cpp b/src/Disks/tests/gtest_cas_pluggable_hash.cpp new file mode 100644 index 000000000000..dca4081c0041 --- /dev/null +++ b/src/Disks/tests/gtest_cas_pluggable_hash.cpp @@ -0,0 +1,944 @@ +#include + +/// P1-T2 (CAS pluggable-blob-hash Phase 1): +/// `PoolMeta` records the pool-wide `blob_hash_algo` and `PoolMeta::createOrValidate` fail-closes on a +/// disk config that disagrees with an existing pool's recorded algo -- the pool-wide durability +/// invariant (never silently re-hash an existing pool). +/// +/// Phase 3 T4 RELAXES that single fail-closed +/// value into `PoolMeta::algos_used` (sorted, append-only): a config algo already a MEMBER is +/// accepted with no write (steady state); a non-member is admitted via a CAS-union ONLY when the +/// disk opts in (`blob_hash_allow_new`), and refused (`BAD_ARGUMENTS`, same as before) otherwise -- +/// a changed config alone must never silently turn a pool mixed. See `AdmissionIsFlagGated` and +/// `ConcurrentAdmissionUnions` below. +/// +/// P1-T3a (this file, extended): the pool's `blob_hash_algo` is threaded into the three hash sites +/// (spec §5/§6) -- `Cas::CaContentWriteBuffer` (streaming blob-body hash), +/// `PartWriteTxn`'s envelope `hash_algo` field, and (transitively, via `Cas::blobHashHexOneShot`) the +/// `poolContentHash` content-key mint on the write path. `poolContentHash` itself is a static +/// helper in `CasPartWriteTxn.cpp` and not directly reachable from a gtest; its production callers already +/// exercise the default `CityHash128` path, and it delegates to the SAME `Cas::blobHashHexOneShot` +/// this file tests directly below. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include + + +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int UNKNOWN_FORMAT_VERSION; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + + +namespace +{ + +/// A deterministic, non-repeating-byte payload spanning several `DBMS_DEFAULT_HASHING_BLOCK_SIZE` +/// (2048 B) blocks, so a chunked-vs-one-shot divergence (the CityHash128 pitfall documented on +/// `poolContentHash`) would not accidentally go unnoticed. +std::string makeMultiBlockPayload(size_t size = 5000) +{ + std::string s; + s.reserve(size); + for (size_t i = 0; i < size; ++i) + s.push_back(static_cast('a' + (i % 23))); + return s; +} + +/// A blob written at its OWN algo's content key, plus the key it landed at. +struct SeededBlob +{ + BlobRef ref; + String key; +}; + +/// Write a blob body of `algo` at its content key, reference it from a committed ref, and DROP that +/// ref — so the blob reaches a folded in-degree of zero and the ORDINARY pipeline condemns it by +/// transition-to-zero. The caller then runs the rounds that fold the `+1` and the `-1`. +/// +/// These tests used to seed a blob no manifest ever named and lean on `rebuildBaseline`'s LIST/HEAD +/// sweep, which was the only path that could condemn such a blob. That sweep is GONE (spec §7: a +/// rebuild condemns nothing — it was the r5-finding-4 data-loss vector), and a blob nothing names is +/// now retained by design. What these tests actually guard — that a blob is recognized under its OWN +/// `` path segment by the fold's key codec, `previewDeletes`, the exact-token delete and fsck, +/// rather than silently skipped as foreign — is unaffected, and lives on the PRODUCTION path, which is +/// where it is now exercised. Reverting either per-algo port still turns these red. +SeededBlob seedReferencedBlob(Pool & store, Backend & backend, const RootNamespace & ns, BlobHashAlgo algo, + uint64_t build_sequence, size_t payload_size, const String & ref_name) +{ + const std::string payload = makeMultiBlockPayload(payload_size); + const BlobRef ref{algo, codecFor(algo).fromHex(blobHashHexOneShot(algo, payload))}; + const String key = store.layout().blobKey(ref); + + EnvelopeHeader header; + header.kind = ObjectKind::Blob; + header.incarnation_tag = UInt128(0x1234); + header.build_id = UInt128(0x5678); + backend.putIfAbsent(key, encodeEnvelopeHeader(header, static_cast(store.poolMeta().blob_header_len)) + payload); + + ManifestEntry entry; + entry.path = "data_" + std::to_string(build_sequence) + ".bin"; + entry.placement = EntryPlacement::Blob; + entry.ref = ref; /// the entry carries the blob's OWN algo, not the pool's write algo + entry.blob_size = 1; + + const ManifestRef mref{.writer_epoch = 1, .build_sequence = build_sequence, .manifest_ordinal = 1}; + writeManifestRaw(backend, store.layout(), ns, mref, {entry}); + publishCommittedTransition(backend, store.layout(), ns, ref_name, std::nullopt, mref); + return SeededBlob{ref, key}; +} + +/// Drop the committed ref `seedReferencedBlob` published, so the blob's only edge disappears. +void dropSeededRef(Pool & store, Backend & backend, const RootNamespace & ns, uint64_t build_sequence, + const String & ref_name) +{ + const ManifestRef mref{.writer_epoch = 1, .build_sequence = build_sequence, .manifest_ordinal = 1}; + dropRefTransition(backend, store.layout(), ns, ref_name, mref); +} + +} + +TEST(CASPluggableHash, PoolMetaRoundTripsAlgosUsed) +{ + PoolMeta pm; + pm.pool_id = u128Of("pool-a"); + pm.blob_header_len = 256; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128), static_cast(BlobHashAlgo::XXH3_128)}; + + const PoolMeta back = decodePoolMeta(encodePoolMeta(pm)); + EXPECT_EQ(back.algos_used, pm.algos_used); + EXPECT_EQ(back.blob_header_len, 256u); +} + +TEST(CASPluggableHash, CreateOrValidateRecordsConfigAlgoOnFreshPool) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); + + /// Reopening with the SAME algo is a no-op reopen: the recorded value comes back unchanged. + const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128); + EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); + EXPECT_EQ(reopened.pool_id, pm.pool_id); +} + +TEST(CASPluggableHash, CreateOrValidateDefaultsToCityHash128) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + + const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); +} + +/// Phase 3 T4 (spec §5, replaces the Phase 1/2 unconditional-fail-close test of the same shape): +/// admission of a NEW algo is EXPLICIT OPT-IN -- the default reopen with a non-member algo still +/// fails closed (`BAD_ARGUMENTS`), but the message names `` and the pool +/// is truly extensible with the flag set. See `AdmissionIsFlagGated` below for the full flow. +TEST(CASPluggableHash, CreateOrValidateFailsClosedOnAlgoMismatchWithoutFlag) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + + expectThrowsCodeWithMessage( + DB::ErrorCodes::BAD_ARGUMENTS, + "1", + [&] + { + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false); + }); + + /// The pool is untouched by the refused reopen: a subsequent open with the ORIGINAL algo still + /// succeeds and returns the same pool_id. + const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128); + EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); +} + +/// spec §9.1 at the unit level: admission of a new algo requires the flag; once admitted, membership +/// alone is the steady-state check (the flag is not needed again for the same algo). +TEST(CASPluggableHash, AdmissionIsFlagGated) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + + /// without the flag: refuse, pool untouched + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, false); }); + + /// with the flag: admitted + const PoolMeta admitted = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, true); + EXPECT_EQ(admitted.algos_used, (std::vector{1, 3})); + + /// steady state: admitted algo reopens WITHOUT the flag + const PoolMeta steady = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, false); + EXPECT_EQ(steady.algos_used, (std::vector{1, 3})); +} + +TEST(CASPluggableHash, ConcurrentAdmissionUnions) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, false, /*allow_mint*/ true); + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, true); + PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, true); + const PoolMeta final_pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, false); + EXPECT_EQ(final_pm.algos_used, (std::vector{1, 2, 3})); /// union, sorted, nothing lost +} + +/// ---- P1-T3a: the pool's blob_hash_algo threaded into the streaming write-buffer hash site ---- + +/// `Cas::CaContentWriteBuffer`'s LOCAL-staging constructor (the everyday spill-to-temp-file +/// mode `ContentAddressedTransaction::writeFile` uses), built with `BlobHashAlgo::XXH3_128`, must hash +/// the streamed payload with xxh3 -- agreeing with the standalone `blobHashHexOneShot` one-shot helper +/// (the same convention `poolContentHash`'s re-hash uses). +TEST(CASPluggableHash, ContentWriteBufferLocalModeHashesWithSelectedAlgoXxh3) +{ + const std::string payload = makeMultiBlockPayload(); + const auto temp_dir = (std::filesystem::temp_directory_path() / "cas_pluggable_hash_xxh3_local").string(); + + std::string got_hash_hex; + size_t got_size = 0; + auto buf = std::make_unique( + temp_dir, + BlobHashAlgo::XXH3_128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string & hash_hex, size_t size, const std::string &) + { + got_hash_hex = hash_hex; + got_size = size; + }); + + /// Write in two chunks so more than one nextImpl flush happens (exercises the streaming state, not + /// just a single call). + buf->write(payload.data(), 1234); + buf->write(payload.data() + 1234, payload.size() - 1234); + buf->finalize(); + + EXPECT_EQ(got_size, payload.size()); + EXPECT_EQ(got_hash_hex, blobHashHexOneShot(BlobHashAlgo::XXH3_128, payload)); + /// A wrong-but-plausible result (e.g. accidentally still hashing with cityHash128) would silently + /// produce a DIFFERENT hex string -- pin that the two algos disagree on this payload, so the + /// assertion above is actually discriminating. + EXPECT_NE(got_hash_hex, blobHashHexOneShot(BlobHashAlgo::CityHash128, payload)); +} + +/// The DEFAULT algo (`CityHash128`) through the SAME write buffer must stay byte-for-byte unchanged -- +/// the CAS pluggable-blob-hash invariant (spec §8). Compares against `blobHashHexOneShot`, which +/// `gtest_cas_blob_hasher.cpp`'s `CityHash128ByteIdenticalToHashingWriteBuffer` already proves is +/// byte-identical to the pre-existing plain `HashingWriteBuffer` convention. +TEST(CASPluggableHash, ContentWriteBufferLocalModeCityHash128Unchanged) +{ + const std::string payload = makeMultiBlockPayload(); + const auto temp_dir = (std::filesystem::temp_directory_path() / "cas_pluggable_hash_ch128_local").string(); + + std::string got_hash_hex; + auto buf = std::make_unique( + temp_dir, + BlobHashAlgo::CityHash128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string & hash_hex, size_t, const std::string &) + { + got_hash_hex = hash_hex; + }); + + buf->write(payload.data(), payload.size()); + buf->finalize(); + + EXPECT_EQ(got_hash_hex, blobHashHexOneShot(BlobHashAlgo::CityHash128, payload)); +} + +/// (codecs-v3 phase 7) The two former `Pool...StampsEnvelopeHashAlgo...` tests were REMOVED: the v3 +/// blob envelope no longer carries a `hash_algo` field (the algo identity lives in the blob KEY, spec +/// §blob-envelope). Algo correctness for the write path is covered by the P1-T3b blob-body-PATH-key +/// tests below (they assert the blob key uses the pool's algo), which is the surviving source of truth. + +/// ---- P1-T3b: the pool's blob_hash_algo threaded into blob-body PATH keys (spec §3/§10) ---- + +/// A blob written and promoted through a live ref on an xxh3-128 pool lands under the +/// `blobs/xxh3//` path segment (not the bare `blobs//` shape), is readable at +/// that key, and `runFsck`'s LIST-based discovery (`Layout::blobsPrefix`, deliberately algo-agnostic) +/// finds it reachable and clean -- proving the GC/fsck key-parse (which takes only the LAST path +/// component as the hex digest, `CasGc.cpp`/`CasFsck.cpp`) still works with the extra segment. +TEST(CASPluggableHash, Xxh3BlobLandsUnderAlgoSegmentAndIsDiscoveredCleanByFsck) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::XXH3_128}); + + const RootNamespace ns{"srv1/tbl"}; + const std::string payload = makeMultiBlockPayload(); + const BlobRef id{BlobHashAlgo::XXH3_128, codecFor(BlobHashAlgo::XXH3_128).fromHex(blobHashHexOneShot(BlobHashAlgo::XXH3_128, payload))}; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/rb"; + auto build = store->beginPartWrite(info); + + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = id; + + e.blob_size = payload.size(); + const ManifestId mid = build->stageManifest({e}); + build->precommitAdd(ns, "rb", mid); + build->putBlob(id, BlobSource::fromString(payload)); + + /// The blob body landed under the algo-segmented path -- readable there, not at the legacy + /// no-segment shape. + const String blob_key = store->layout().blobKey(id); + EXPECT_NE(blob_key.find("/blobs/xxh3/"), String::npos) << blob_key; + EXPECT_EQ(blob_key.find("/blobs/ch128/"), String::npos) << blob_key; + EXPECT_TRUE(backend->head(blob_key).exists); + + build->promote(ns, "rb", build->buildId(), mid); + store->renewWatermarkOnce(); + + const FsckReport rep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_GE(rep.reachable, 1u); + + /// Not merely "clean by omission" (e.g. a bug that silently LISTed nothing): the physical listing + /// actually walked the algo-segmented key. + const bool found = std::any_of(rep.objects.begin(), rep.objects.end(), + [](const FsckObject & o) { return o.key.find("/blobs/xxh3/") != String::npos; }); + EXPECT_TRUE(found); +} + +/// ============================================================================================ +/// CAS pluggable-blob-hash Phase 2 Task 5 -- THE CRUX (anti-silent-leak regression gate). +/// +/// Two sites classify a blob by parsing its object-key hex into a hash set: `CasGc.cpp`'s condemn +/// path (the fold's transition-to-zero, and — until spec §7 removed it — `Gc::rebuildBaseline`'s +/// LIST/HEAD sweep) and `CasFsck.cpp`'s +/// present-but-unreferenced classification. Both used to route through the bare, fixed-width +/// `hexToU128` (32-hex-only) inside a `catch(...) continue` / no-catch-at-all — so a 64-hex `sha256` +/// key either (a) fell into the "foreign key shape — not ours" catch and was silently treated as +/// debris (the condemn sweep: the blob is NEVER condemned — a permanent GC leak), or (b) threw +/// uncaught out of fsck's present-but-unreferenced loop (a hard fsck failure on a live sha256 pool). +/// Phase 2 Task 5 ports both to the pool-scoped `DigestCodec::fromHex`, which parses a CORRECT-WIDTH +/// key (16 OR 32 bytes) — a genuinely foreign key shape (e.g. a `.meta` sibling) still falls into +/// the catch, but a real sha256 blob no longer does. +/// +/// This test constructs a `sha256`-algo pool DIRECTLY via `PoolConfig` (this bypasses only the +/// disk-config *factory* guard in `MetadataStorageFactory.cpp`, which Task 6 removes — `Pool::open` +/// itself has never gated on algo) and writes a blob body straight at its 64-hex content-addressed key +/// (bypassing `PartWriteTxn::putBlob`, whose OWN internal `logical_hash` stays a fixed 128-bit +/// representation until a later task — see the Task 5 report), references it, and drops the reference +/// so the fold condemns it. It then drives BOTH crux sites and asserts the blob is CLASSIFIED, not +/// silently skipped as foreign. +/// +/// MUST GO RED if either port is reverted to `hexToU128`: reverting `CasGc.cpp`'s fold leaves +/// `condemned_total == 0` (never condemned) and `previewDeletes()` empty; reverting `CasFsck.cpp`'s +/// sites either throws out of `runFsck` or leaves the blob unclassified/absent from `unreachable`. +TEST(CASPluggableHash, Sha256BlobSeenByCondemnSweepAndFsckNotSilentlySkipped) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::Sha256, .gc_fold_max_defer_rounds = 0}); + ASSERT_EQ(blobHashLenFor(store->writeAlgo()), 32u) << "sha256 must derive a 32-byte digest width"; + + const DigestCodec codec = codecFor(store->writeAlgo()); + const std::string payload = makeMultiBlockPayload(); + const std::string hex = blobHashHexOneShot(BlobHashAlgo::Sha256, payload); + ASSERT_EQ(hex.size(), 64u) << "sha256 renders 64 lowercase hex chars"; + const BlobDigest digest = codec.fromHex(hex); // round-trip sanity: must not throw at width 32 + + /// Reference the blob from a committed ref, then drop that ref: the fold sees `+1` then `-1`, the + /// blob transitions to in-degree zero, and the ORDINARY condemn path claims it. Every per-algo + /// parse this test guards sits on that path. + const RootNamespace ns{"00/aa@cas@"}; + Gc gc(store, UInt128(1)); + const SeededBlob seeded = seedReferencedBlob(*store, *backend, ns, BlobHashAlgo::Sha256, + /*build_sequence=*/1, /*payload_size=*/5000, "tbl_sha"); + const BlobRef id = seeded.ref; + const String blob_key = seeded.key; + EXPECT_NE(blob_key.find("/blobs/sha256/"), String::npos) << blob_key; + ASSERT_TRUE(backend->head(blob_key).exists) << "the sha256 blob body must be present before the fold"; + ASSERT_EQ(codecFor(store->writeAlgo()).fromHex(hex), digest) << "fixture sanity: the seeded digest is ours"; + + /// ---- Site 1: the fold's condemn path ---- + runRegularRoundReclaiming(gc); /// folds the +1 + dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_sha"); + runRegularRoundReclaiming(gc); /// folds the -1: transition to zero => condemned + + const auto state_bytes = backend->get(store->layout().gcStateKey()); + ASSERT_TRUE(state_bytes.has_value()); + const GcState state = decodeGcState(state_bytes->bytes); + ASSERT_GT(state.snap_generation, 0u); + const auto seal_bytes = backend->get(store->layout().foldSealKey(state.snap_generation, state.snap_attempt)); + ASSERT_TRUE(seal_bytes.has_value()); + const CasFoldSeal seal = decodeFoldSeal(seal_bytes->bytes); + ASSERT_TRUE(seal.condemned_summary.contains(0)) << "the seal's condemned_summary must be total over gc_shards"; + EXPECT_EQ(seal.condemned_summary.at(0).condemned_total, 1u) + << "THE CRUX: the sha256 blob must be condemned by the fold -- a silent-leak regression (a " + "reverted CasGc.cpp codec.fromHex port) leaves this at 0"; + + /// previewDeletes streams the SAME adopted seal via the run's own SourceEdgeKeyCodec (never pool + /// meta) and must report exactly our blob, at its real 32-byte digest. + const std::vector preview = gc.previewDeletes(); + ASSERT_EQ(preview.size(), 1u) << "THE CRUX: previewDeletes must surface the condemned sha256 blob"; + EXPECT_EQ(preview[0].ref, id); + EXPECT_EQ(preview[0].key, blob_key); + + /// ---- Site 2: fsck's present-but-unreferenced classification ---- + /// Must complete without throwing (a reverted port either throws BAD_ARGUMENTS out of the + /// no-try/catch parse sites, or silently drops the blob from every classified set) and must + /// physically account for the blob. + FsckReport frep; + ASSERT_NO_THROW(frep = runFsck(*store, /*detail=*/true)); + EXPECT_GE(frep.unreachable, 1u) + << "THE CRUX: fsck's physical listing must count the sha256 blob as unreachable-but-present, " + "not silently omit it"; + const auto oit = std::find_if(frep.objects.begin(), frep.objects.end(), + [&](const FsckObject & o) { return o.key == blob_key; }); + ASSERT_NE(oit, frep.objects.end()) << "the sha256 blob must appear in fsck's detailed object list"; + /// The fold above already condemned it into the GC snapshot, so fsck's GC-pipeline-view + /// classification (not the generic Unaccounted bucket -- reachable only by width-correctly pairing + /// the fsck-side hash against the run's kCondemned row hash) must recognize it as known-to-GC. + EXPECT_EQ(oit->cls, FsckClass::PendingGc) + << "THE CRUX: fsck must pair the sha256 blob against the GC snapshot's kCondemned row (a " + "silent-leak regression in CasFsck.cpp's unref_hashes/in_run_hashes/retired_by_hash port " + "leaves this as the generic Unaccounted bucket instead)"; +} + +/// ============================================================================================ +/// CAS pluggable-blob-hash Phase 2 Task 6 -- end-to-end sha256 WRITE path (in-memory; the real +/// wiring-level integration + soak is Task 7). +/// +/// Before this task, `PartWriteTxn`'s OWN write-path internals stayed a fixed 128-bit representation +/// downstream of the mint (`poolContentHash`/`PartWriteTxn::putBlob`'s `logical_hash`, the `deps` map key, the +/// event-log `object_hash` render, and `objectKey`) -- safe only because the disk-config factory guard +/// (`MetadataStorageFactory.cpp`) blocked any real sha256 pool from reaching `PartWriteTxn` at all (see the +/// Task 5 report and the "Task 6+" comments this task removes). Task 6 finishes those sites AND lifts +/// the guard in the SAME commit. This test drives a REAL `PartWriteTxn` (`putBlob` -> `stageManifest` -> +/// `precommitAdd` -> `promote`) on a `Sha256` pool and asserts: +/// 1. the blob lands under `blobs/sha256/<64-hex>` and the manifest entry's `blob_hash`, read back via +/// `decodePartManifest`, is the FULL 32-byte digest (bytes beyond 16 are non-zero for a real sha256 +/// digest, i.e. NOT truncated to `.toU128()`'s low 16 bytes); +/// 2. an inline file and a standalone blob of IDENTICAL content get the SAME 32-byte `file_hash` under +/// sha256 -- mirroring the (fixed) `ContentAddressedTransaction.cpp` inline-candidate formula +/// (`blobHashHexOneShot(pool_algo, bytes)` -> pool-scoped `DigestCodec::fromHex`) directly at the +/// Core level, since exercising the wiring itself is Task 7's job; +/// 3. `runFsck` on the pool is clean (no dangling, no foreign) -- the whole write -> GC -> fsck loop +/// agrees on the 64-hex key. +TEST(CASPluggableHash, Sha256BuildWritesFullWidthDigestAndInlineEqualsBlob) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::Sha256}); + ASSERT_EQ(blobHashLenFor(store->writeAlgo()), 32u) << "sha256 must derive a 32-byte digest width"; + const DigestCodec codec = codecFor(store->writeAlgo()); + + const RootNamespace ns{"srv1/tbl"}; + const std::string payload = makeMultiBlockPayload(); + const std::string hex = blobHashHexOneShot(BlobHashAlgo::Sha256, payload); + ASSERT_EQ(hex.size(), 64u) << "sha256 renders 64 lowercase hex chars"; + const BlobRef id{BlobHashAlgo::Sha256, codec.fromHex(hex)}; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/part1"; + auto build = store->beginPartWrite(info); + + /// Mirror the (fixed) inline-candidate hash site directly: same content, same pool algo, via the + /// SAME public formula ContentAddressedTransaction.cpp's writeFile now uses -- NOT the old hardcoded + /// CityHash128 (which would produce a DIFFERENT, 128-bit-then-zero-padded value here). + const BlobDigest inline_hash = codec.fromHex(blobHashHexOneShot(BlobHashAlgo::Sha256, payload)); + const BlobDigest blob_hash = codec.fromHex(hex); + EXPECT_EQ(inline_hash, blob_hash) << "inline == blob: identical content must hash identically under sha256"; + + /// THE CRUX (width): a genuine 32-byte sha256 digest must NOT be zero-padded past byte 16 -- the + /// shape `BlobDigest::fromU128` (or a reverted hardcoded-CityHash128 inline site) would produce. + const bool tail_nonzero = std::any_of(blob_hash.bytes.begin() + 16, blob_hash.bytes.end(), + [](uint8_t b) { return b != 0; }); + EXPECT_TRUE(tail_nonzero) << "a genuine sha256 digest must not be zero-padded past byte 16"; + + ManifestEntry blob_entry; + blob_entry.path = "data.bin"; + blob_entry.placement = EntryPlacement::Blob; + blob_entry.ref = BlobRef{BlobHashAlgo::Sha256, blob_hash}; + blob_entry.blob_size = payload.size(); + + ManifestEntry inline_entry; + inline_entry.path = "checksums.txt"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.ref = BlobRef{BlobHashAlgo::Sha256, inline_hash}; + inline_entry.blob_size = payload.size(); + inline_entry.inline_bytes = payload; + + const ManifestId mid = build->stageManifest({blob_entry, inline_entry}); + build->precommitAdd(ns, "part1", mid); + const PutBlobResult ref = build->putBlob(id, BlobSource::fromString(payload)); + EXPECT_EQ(ref.size, payload.size()); + + /// THE CRUX (blob side): the blob body lands under the sha256-segmented path, addressed by the + /// FULL 64-hex key -- `PartWriteTxn::putBlob`'s internal `logical_hash` must not have silently narrowed it + /// to a 32-hex (128-bit) key before this task. + const String blob_key = store->layout().blobKey(id); + EXPECT_NE(blob_key.find("/blobs/sha256/"), String::npos) << blob_key; + ASSERT_TRUE(backend->head(blob_key).exists); + + build->promote(ns, "part1", build->buildId(), mid); + store->renewWatermarkOnce(); + + /// Read the committed manifest back -- the on-disk `blob_hash` must be the FULL 32-byte digest, not + /// truncated by the manifest codec or by anything upstream of `stageManifest`. + const auto manifest_bytes = backend->get(store->layout().manifestKey(mid)); + ASSERT_TRUE(manifest_bytes.has_value()); + const PartManifest read_back = decodePartManifest(openObject(FormatId::PartManifest, manifest_bytes->bytes)); + ASSERT_EQ(read_back.entries.size(), 2u); + const auto read_blob_it = std::find_if(read_back.entries.begin(), read_back.entries.end(), + [](const ManifestEntry & e) { return e.placement == EntryPlacement::Blob; }); + ASSERT_NE(read_blob_it, read_back.entries.end()); + EXPECT_EQ(read_blob_it->ref.digest, blob_hash); + const bool read_tail_nonzero = std::any_of(read_blob_it->ref.digest.bytes.begin() + 16, + read_blob_it->ref.digest.bytes.end(), [](uint8_t b) { return b != 0; }); + EXPECT_TRUE(read_tail_nonzero) << "the manifest's on-disk blob_hash must not be truncated either"; + + /// The write -> GC -> fsck loop must agree end-to-end on the 64-hex key: clean, no dangling. + const FsckReport rep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_GE(rep.reachable, 1u); +} + +/// ============================================================================================ +/// CAS mixed-algo pools Phase 3 T5: +/// path-derived `BlobRef` in the sweep/fsck (`Layout::parseBlobKey`) and per-entry admission +/// validation at `foldManifestEdges` with refresh-on-miss. +/// ============================================================================================ + +/// spec §9.8 -- THE race regression this task exists to close. Each `Pool`'s `admitted_algos` cache +/// is a MONOTONE snapshot seeded once at `Pool::open` and never re-read on its own; if node A admits +/// a brand-new algo and publishes a manifest naming it, node B's stale cache must NOT fail the fold +/// closed forever -- `foldManifestEdges` must refresh `_pool_meta` on the very first miss and accept +/// once the fresh read proves the algo genuinely admitted. Node B is opened BEFORE node A performs the +/// admission on purpose: constructing B afterward would seed its cache already-fresh and never +/// exercise the race the fix targets. +TEST(CASPluggableHash, StaleAlgoRegistryRefreshOnMiss) +{ + auto backend = std::make_shared(); + + /// Node B opens FIRST -- its admitted-cache seeds at {ch128} only, before sha256 exists anywhere. + auto store_b = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "b", + .blob_hash_algo = BlobHashAlgo::CityHash128}); + ASSERT_TRUE(store_b->isAlgoAdmitted(BlobHashAlgo::CityHash128)); + ASSERT_FALSE(store_b->isAlgoAdmitted(BlobHashAlgo::Sha256)); + + /// Node A opens SECOND, admits sha256 via the opt-in flag, and publishes a manifest naming a + /// sha256 blob through the real PartWriteTxn path (putBlob -> stageManifest -> precommitAdd -> promote). + auto store_a = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "a", + .blob_hash_algo = BlobHashAlgo::Sha256, .blob_hash_allow_new = true}); + ASSERT_TRUE(store_a->isAlgoAdmitted(BlobHashAlgo::Sha256)); + + const RootNamespace ns{"srv1/tbl"}; + const std::string payload = makeMultiBlockPayload(); + const BlobRef id{BlobHashAlgo::Sha256, codecFor(BlobHashAlgo::Sha256).fromHex(blobHashHexOneShot(BlobHashAlgo::Sha256, payload))}; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/part1"; + auto build = store_a->beginPartWrite(info); + + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = id; + e.blob_size = payload.size(); + const ManifestId mid = build->stageManifest({e}); + build->precommitAdd(ns, "part1", mid); + build->putBlob(id, BlobSource::fromString(payload)); + build->promote(ns, "part1", build->buildId(), mid); + store_a->renewWatermarkOnce(); + + /// B's cache is STILL stale here -- it has never re-read `_pool_meta` since open. + ASSERT_FALSE(store_b->isAlgoAdmitted(BlobHashAlgo::Sha256)); + + /// B folds the committed ref naming the sha256 entry: without refresh-on-miss this throws + /// CORRUPTED_DATA ("manifest entry algo sha256 not admitted"); with it, the miss triggers exactly + /// one `refreshAdmittedAlgos()` and the fold proceeds. + Gc gc(store_b, UInt128(1)); + const RebuildReport rep = gc.rebuildBaseline(/*force*/ true); + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.committed_refs, 1u); + EXPECT_TRUE(store_b->isAlgoAdmitted(BlobHashAlgo::Sha256)) << "the miss must have unioned B's cache"; +} + +/// spec §9.4 half: an object whose key names an algo THIS BUILD has never heard of (a genuinely +/// foreign top-level segment, e.g. planted by a different/future tool) must never be treated as one +/// of ours -- the GC must skip it (never condemn or delete it) and fsck must classify it into the +/// generic `Unaccounted` bucket (never throw, never silently drop it from the physical listing). +/// In the SAME pass, a 2-algo pool's OWN blobs under `blobs/ch128/` and `blobs/sha256/` must both +/// still be classified normally -- the foreign segment must not make the fold/fsck narrow to one +/// algo or blind them to the others. +TEST(CASPluggableHash, ForeignAlgoSegmentIsDebrisNotOurs) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::CityHash128, .gc_fold_max_defer_rounds = 0}); + /// Admit sha256 into the SAME pool from a second mount, then pull the union into `store`'s cache. + Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test2", + .blob_hash_algo = BlobHashAlgo::Sha256, .blob_hash_allow_new = true}); + store->refreshAdmittedAlgos(); + ASSERT_TRUE(store->isAlgoAdmitted(BlobHashAlgo::Sha256)); + + /// Two of the pool's OWN blobs -- one per algo -- each referenced by a committed ref that is then + /// DROPPED, so the fold condemns both by transition-to-zero. + const RootNamespace ns{"00/aa@cas@"}; + Gc gc(store, UInt128(1)); + const SeededBlob ch = seedReferencedBlob(*store, *backend, ns, BlobHashAlgo::CityHash128, + /*build_sequence=*/1, /*payload_size=*/5001, "tbl_ch"); + const SeededBlob sh = seedReferencedBlob(*store, *backend, ns, BlobHashAlgo::Sha256, + /*build_sequence=*/2, /*payload_size=*/5002, "tbl_sh"); + const BlobRef ch_ref = ch.ref; + const BlobRef sh_ref = sh.ref; + const String ch_key = ch.key; + const String sh_key = sh.key; + + /// A FOREIGN object under an algo segment `blobHashAlgoName` never renders ("md5") -- not one of + /// ours under any circumstance. + const String foreign_key = store->layout().blobsPrefix() + "md5/aa/" + std::string(32, 'a'); + backend->putIfAbsent(foreign_key, std::string("not a real envelope")); + + runRegularRoundReclaiming(gc); /// folds both +1s + dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_ch"); + dropSeededRef(*store, *backend, ns, /*build_sequence=*/2, "tbl_sh"); + runRegularRoundReclaiming(gc); /// folds both -1s: both transition to zero + + /// The fold condemns exactly the two OWN blobs -- never the foreign object. + const std::vector preview = gc.previewDeletes(); + ASSERT_EQ(preview.size(), 2u); + std::unordered_set condemned_refs; + for (const auto & p : preview) + { + condemned_refs.insert(p.ref); + EXPECT_NE(p.key, foreign_key); + } + EXPECT_TRUE(condemned_refs.count(ch_ref)); + EXPECT_TRUE(condemned_refs.count(sh_ref)); + EXPECT_TRUE(backend->head(foreign_key).exists) << "the foreign object must never be touched by the fold"; + + const FsckReport frep = runFsck(*store, /*detail=*/true); + /// The physical listing counts all THREE unreferenced objects (two ours + one foreign). + EXPECT_EQ(frep.unreachable, 3u); + const auto foreign_obj = std::find_if(frep.objects.begin(), frep.objects.end(), + [&](const FsckObject & o) { return o.key == foreign_key; }); + ASSERT_NE(foreign_obj, frep.objects.end()) << "the foreign object must still appear in the physical listing"; + /// ... but classified as generic Unaccounted -- it can never pair against the GC snapshot, which + /// only ever knows about OUR two algo-segmented refs. + EXPECT_EQ(foreign_obj->cls, FsckClass::Unaccounted); + + /// The two OWN blobs are recognized under their OWN algo segment in the SAME pass. + const auto ch_obj = std::find_if(frep.objects.begin(), frep.objects.end(), + [&](const FsckObject & o) { return o.key == ch_key; }); + const auto sh_obj = std::find_if(frep.objects.begin(), frep.objects.end(), + [&](const FsckObject & o) { return o.key == sh_key; }); + ASSERT_NE(ch_obj, frep.objects.end()); + ASSERT_NE(sh_obj, frep.objects.end()); + EXPECT_EQ(ch_obj->cls, FsckClass::PendingGc); + EXPECT_EQ(sh_obj->cls, FsckClass::PendingGc); +} + +/// ============================================================================================ +/// CAS reader-generation gate (`Core/Formats/CasFormat.h`'s `G_BUILD`) was raised to 4 for +/// per-namespace contiguous ref-log ids (INV-1) and has since moved again, to 5, for Stage B's +/// namespace-life-keyed ref layer ("format bump B", `kNamespaceLifeKeyedGeneration`) -- this test's +/// assertions read `G_BUILD` itself rather than a hardcoded generation number for exactly that reason, +/// so a THIRD bump does not silently make them false. `PoolMeta::createOrValidate`'s open-time +/// CAS-raise targets `G_BUILD`, and `decodePoolMeta` fail-closes BOTH on a FUTURE +/// `min_reader_generation` AND on a BACKWARD pool whose header `compatibility_version` is below +/// `kNamespaceLifeKeyedGeneration` (which, being the LATER of the two historical breaking-change +/// floors, subsumes `kContiguousRefStreamsGeneration` -- see `CasPoolMetaFormat.cpp`). +/// ============================================================================================ + +TEST(CASPluggableHash, ReaderGenerationIsRaisedToGBuild) +{ + EXPECT_GE(G_BUILD, kNamespaceLifeKeyedGeneration) + << "the reader-generation gate must be at least the namespace-life-keyed floor it enforces"; + + /// A freshly opened/created pool records `min_reader_generation == G_BUILD` (the open-time + /// CAS-raise, `PoolMeta::createOrValidate`, always targets this build's own floor). + { + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + EXPECT_EQ(store->poolMeta().min_reader_generation, G_BUILD); + + const auto meta_bytes = backend->get(store->layout().poolMetaKey()); + ASSERT_TRUE(meta_bytes.has_value()); + EXPECT_EQ(decodePoolMeta(meta_bytes->bytes).min_reader_generation, G_BUILD); + } + + /// FORWARD gate: a pool-meta carrying `min_reader_generation == G_BUILD + 1` (one generation past + /// THIS build's floor) still fails closed at open -- the startup gate (`decodePoolMeta`) rejects it + /// even though generation 4 is now understood. + { + auto backend = std::make_shared(); + const Layout layout("p"); + PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + pm.min_reader_generation = G_BUILD + 1; + ASSERT_TRUE(backend->casPut(layout.poolMetaKey(), encodePoolMeta(pm), backend->get(layout.poolMetaKey())->token).outcome == CasOutcome::Committed); + + expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] + { Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); + } + + /// BACKWARD floor: a pool whose header `v` (compatibility_version) is BELOW `G_BUILD` was written + /// by an older build this reader can no longer trust -- today that is one generation short of + /// `kNamespaceLifeKeyedGeneration`, a pool whose ref-object keys carry no incarnation segment, + /// which this build's parsers refuse as corruption rather than read. Craft it at the text layer: + /// take a fresh pool-meta and rewrite its line-1 version gate down to `G_BUILD - 1` (an older + /// build would have stamped exactly that). + { + auto backend = std::make_shared(); + const Layout layout("p"); + PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + const String fresh_bytes = encodePoolMeta(pm); + + const String from = "\"v\":" + std::to_string(G_BUILD); + const String to = "\"v\":" + std::to_string(G_BUILD - 1); + const auto pos = fresh_bytes.find(from); + ASSERT_NE(pos, String::npos); // sanity: a fresh pool stamps the header at the floor + String downgraded = fresh_bytes; + downgraded.replace(pos, from.size(), to); + ASSERT_TRUE(backend->casPut(layout.poolMetaKey(), downgraded, backend->get(layout.poolMetaKey())->token).outcome == CasOutcome::Committed); + + /// `decodePoolMeta`'s backward floor rejects the downgraded bytes directly... + expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { decodePoolMeta(downgraded); }); + /// ...and so does a full `Pool::open` (decoding the pool-meta is its first fail-closed step). + expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] + { Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); + } +} + +/// ============================================================================================ +/// CAS mixed-algo pools Phase 3 T6: +/// cross-cutting cruxes over a pool that genuinely mixes algos end-to-end (reclaim + distinctness). +/// The no-bare-digest grep gates (design Step 3) are run separately, not as gtest bodies. +/// ============================================================================================ + +/// spec §9.3 -- THE reclaim crux. A pool admits BOTH `ch128` and `sha256`; a blob body is +/// planted directly under EACH algo's segment (mirrors `Sha256BlobSeenByCondemnSweepAndFsckNotSilentlySkipped`'s +/// fixture, widened to two algos). The fold must condemn BOTH into the SAME baseline +/// (`previewDeletes` surfaces both refs), and driving the round-paced pipeline to completion (graduate, +/// then the exact-token delete) must reclaim BOTH bodies -- the backend ends up holding ZERO blob +/// bytes of EITHER algo, and fsck reports clean. +/// +/// MUST GO RED if any settlement/graduation/delete path silently narrows to one algo -- e.g. a fold +/// that only accounts `blobs/ch128/`, a graduation/delete loop that iterates a digest-only set and +/// coalesces the two algos' entries, or an fsck reachability check that stops after the first algo it +/// sees. +TEST(CASPluggableHash, TwoAlgoBlobsBothFullyReclaimed) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::CityHash128, .gc_fold_max_defer_rounds = 0}); + /// Admit sha256 into the SAME pool from a second mount, then pull the union into `store`'s cache + /// (mirrors `ForeignAlgoSegmentIsDebrisNotOurs`'s admission fixture). + Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test2", + .blob_hash_algo = BlobHashAlgo::Sha256, .blob_hash_allow_new = true}); + store->refreshAdmittedAlgos(); + ASSERT_TRUE(store->isAlgoAdmitted(BlobHashAlgo::Sha256)); + + /// Two of the pool's OWN blobs -- one per algo -- referenced then dropped, so the fold condemns + /// both by transition-to-zero. + const RootNamespace ns{"00/aa@cas@"}; + Gc gc(store, UInt128(1)); + const SeededBlob ch = seedReferencedBlob(*store, *backend, ns, BlobHashAlgo::CityHash128, + /*build_sequence=*/1, /*payload_size=*/5001, "tbl_ch"); + const SeededBlob sh = seedReferencedBlob(*store, *backend, ns, BlobHashAlgo::Sha256, + /*build_sequence=*/2, /*payload_size=*/5002, "tbl_sh"); + const String ch_key = ch.key; + const String sh_key = sh.key; + ASSERT_TRUE(backend->head(ch_key).exists); + ASSERT_TRUE(backend->head(sh_key).exists); + + runRegularRoundReclaiming(gc); /// folds both +1s + dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_ch"); + dropSeededRef(*store, *backend, ns, /*build_sequence=*/2, "tbl_sh"); + runRegularRoundReclaiming(gc); /// folds both -1s: both condemned in the same round + + /// previewDeletes covers BOTH refs from the adopted seal -- never just one algo. + { + const std::vector preview = gc.previewDeletes(); + ASSERT_EQ(preview.size(), 2u); + std::unordered_set refs; + for (const auto & p : preview) + refs.insert(p.ref); + EXPECT_TRUE(refs.count(ch.ref)); + EXPECT_TRUE(refs.count(sh.ref)); + } + + /// Drive the round-paced pipeline to actual physical deletion: the fold condemned both at its + /// round; the VERY NEXT round graduates them (unconditionally, round-paced); the round after that + /// executes the exact-token delete for both. + { + const RoundReport rep1 = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep1.graduated, 2u) << "both algos' blobs must graduate together in one round"; + EXPECT_TRUE(backend->head(ch_key).exists); // pending: still present this pass + EXPECT_TRUE(backend->head(sh_key).exists); + } + { + const RoundReport rep2 = runRegularRoundReclaiming(gc); + EXPECT_EQ(rep2.redeleted, 2u) << "both algos' pending deletes must execute together in one round"; + } + + /// THE CRUX: after graduation the backend holds ZERO blob bodies of EITHER algo. + EXPECT_FALSE(backend->head(ch_key).exists) << "the ch128 blob must be physically reclaimed"; + EXPECT_FALSE(backend->head(sh_key).exists) << "the sha256 blob must be physically reclaimed"; + + const FsckReport frep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(frep.clean()); + EXPECT_EQ(frep.dangling, 0u); +} + +/// spec §9.5 -- same-digest-different-algo end-to-end. `ch128:X` and `xxh3:X` share the SAME 16-byte +/// digest VALUE but are DISTINCT blob identities (`BlobRef` is the pair): distinct object keys, distinct +/// `.meta`, distinct bodies, distinct settlement rows (fold both -> distinct in-degree per ref), and +/// dropping ONE ref's committed manifest reclaims ONLY that algo's blob -- the other stays fully +/// readable throughout. +/// +/// MUST GO RED if anything upstream of `BlobRef` ever collapses identity to the bare digest (e.g. a +/// settlement/meta/condemn site keyed on `BlobDigest` alone) -- the two blobs would alias into one row +/// and dropping one ref would (wrongly) reclaim or corrupt the other. +TEST(CASPluggableHash, SameDigestDifferentAlgoDistinctBodiesAndSettlement) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .blob_hash_algo = BlobHashAlgo::CityHash128, .gc_fold_max_defer_rounds = 0}); + Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test2", + .blob_hash_algo = BlobHashAlgo::XXH3_128, .blob_hash_allow_new = true}); + store->refreshAdmittedAlgos(); + ASSERT_TRUE(store->isAlgoAdmitted(BlobHashAlgo::XXH3_128)); + + /// SAME 16-byte digest VALUE under two different algos -- deliberately NOT derived from either + /// body's real content hash: the crux under test is identity distinctness (the pair), not hash + /// correctness (already covered by the sha256/xxh3 write-path tests above). + const BlobDigest shared_digest = BlobDigest::fromU128(UInt128(0xC0FFEE)); + const BlobRef ref_ch{BlobHashAlgo::CityHash128, shared_digest}; + const BlobRef ref_xx{BlobHashAlgo::XXH3_128, shared_digest}; + /// Distinct content, not merely distinct length: `makeMultiBlockPayload` at two different sizes + /// would make the shorter body a byte-for-byte PREFIX of the longer one (same repeating pattern + /// from the same phase), which would defeat the "must not contain" assertions below. + const std::string body_ch = makeMultiBlockPayload(4001); + std::string body_xx = makeMultiBlockPayload(4002); + std::reverse(body_xx.begin(), body_xx.end()); + ASSERT_NE(body_ch, body_xx); + + const RootNamespace ns{"srv1/tbl"}; + + PartWriteInfo info_a; + info_a.intended_ref = ns.string() + "/part_a"; + auto build_a = store->beginPartWrite(info_a); + ManifestEntry e_a; + e_a.path = "a.bin"; e_a.placement = EntryPlacement::Blob; e_a.ref = ref_ch; e_a.blob_size = body_ch.size(); + const ManifestId mid_a = build_a->stageManifest({e_a}); + build_a->precommitAdd(ns, "part_a", mid_a); + build_a->putBlob(ref_ch, BlobSource::fromString(body_ch)); + build_a->promote(ns, "part_a", build_a->buildId(), mid_a); + + PartWriteInfo info_b; + info_b.intended_ref = ns.string() + "/part_b"; + auto build_b = store->beginPartWrite(info_b); + ManifestEntry e_b; + e_b.path = "b.bin"; e_b.placement = EntryPlacement::Blob; e_b.ref = ref_xx; e_b.blob_size = body_xx.size(); + const ManifestId mid_b = build_b->stageManifest({e_b}); + build_b->precommitAdd(ns, "part_b", mid_b); + build_b->putBlob(ref_xx, BlobSource::fromString(body_xx)); + build_b->promote(ns, "part_b", build_b->buildId(), mid_b); + store->renewWatermarkOnce(); + + /// Distinct object keys and distinct bodies despite the SAME digest value. + const String key_ch = store->layout().blobKey(ref_ch); + const String key_xx = store->layout().blobKey(ref_xx); + EXPECT_NE(key_ch, key_xx); + const auto raw_ch = backend->get(key_ch); + const auto raw_xx = backend->get(key_xx); + ASSERT_TRUE(raw_ch.has_value()); + ASSERT_TRUE(raw_xx.has_value()); + EXPECT_NE(raw_ch->bytes.find(body_ch), String::npos); + EXPECT_NE(raw_xx->bytes.find(body_xx), String::npos); + EXPECT_EQ(raw_ch->bytes.find(body_xx), String::npos) << "the ch128 body must not contain the xxh3 payload"; + EXPECT_EQ(raw_xx->bytes.find(body_ch), String::npos) << "the xxh3 body must not contain the ch128 payload"; + + /// Distinct `.meta` objects. + const String meta_ch = store->layout().blobMetaKey(ref_ch); + const String meta_xx = store->layout().blobMetaKey(ref_xx); + EXPECT_NE(meta_ch, meta_xx); + EXPECT_TRUE(backend->head(meta_ch).exists); + EXPECT_TRUE(backend->head(meta_xx).exists); + + /// Distinct settlement (in-degree per ref, keyed on the FULL `BlobRef` pair -- never the shared + /// bare digest, which would alias the two rows into one). + Gc gc(store, UInt128(1)); + runRegularRoundReclaiming(gc); + { + const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + EXPECT_EQ(inDegreeInRuns(*backend, seal.blob_target_runs, ref_ch), 1); + EXPECT_EQ(inDegreeInRuns(*backend, seal.blob_target_runs, ref_xx), 1); + } + + /// Dropping ONLY `part_a`'s committed ref condemns+reclaims ONLY `ch128:X`; `xxh3:X` (the SAME + /// digest value, a DIFFERENT algo) stays referenced and fully readable throughout. + store->dropRef(ns, "part_a"); + runRegularRoundReclaiming(gc); // condemns ch128:X (in-degree drops to 0); xxh3:X is untouched (still ref'd) + runRegularRoundReclaiming(gc); // graduates ch128:X + runRegularRoundReclaiming(gc); // executes the exact-token delete for ch128:X + + EXPECT_FALSE(backend->head(key_ch).exists) << "ch128:X must be reclaimed once its ref is dropped"; + EXPECT_TRUE(backend->head(key_xx).exists) + << "THE CRUX: xxh3:X (same digest value, different algo) must remain readable after ch128:X " + "is reclaimed -- a digest-only settlement would have condemned/deleted both together"; + const auto still_readable = backend->get(key_xx); + ASSERT_TRUE(still_readable.has_value()); + EXPECT_NE(still_readable->bytes.find(body_xx), String::npos); + + const FsckReport frep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(frep.clean()); + EXPECT_EQ(frep.dangling, 0u); +} diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp new file mode 100644 index 000000000000..b0ad1b340afd --- /dev/null +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -0,0 +1,3646 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +extern const int BAD_ARGUMENTS; +extern const int CORRUPTED_DATA; +extern const int NOT_IMPLEMENTED; +extern const int UNKNOWN_FORMAT_VERSION; +extern const int FILE_DOESNT_EXIST; +extern const int UNKNOWN_EXCEPTION; +extern const int NETWORK_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASRefRecoveryEpochSealed; +extern const Event CASMountExclusivityViolation; +extern const Event CASMountLeaseLost; +extern const Event CASMountReleaseSkippedForeignOccupant; +extern const Event CASRemountAttempts; +extern const Event CASRemountSucceeded; +extern const Event CASRemountFailed; +} + +using namespace DB::Cas; +using DB::Cas::tests::blobEntryFor; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ +/// Counts mutating backend calls so a test can assert an open path is write-free. +class WriteCountingBackend final : public DB::Cas::Backend +{ +public: + explicit WriteCountingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} + size_t writes = 0; + + std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->putIfAbsent(k, b, meta); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + ++writes; + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->putOverwrite(k, b, e, meta); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->casPut(k, b, e, meta); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { ++writes; return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } +private: + std::shared_ptr inner; +}; + +/// Publish one part `ref` through the REAL PartWriteTxn write path: stage a manifest holding a single content +/// blob whose payload is `payload`, precommit-add into the owning shard, then promote precommit -> +/// committed. Returns the published ManifestId. This is the canonical write-side fixture for the +/// read-path tests (the same shape as `publishPart` in gtest_cas_gc_log.cpp). The manifest entry path +/// is `data.bin` unless `entry_path` overrides it. +ManifestId publishPart( + const PoolPtr & s, const String & ns, const String & ref, const String & payload, + const String & entry_path = "data.bin") +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + + ManifestEntry e; + e.path = entry_path; + e.placement = EntryPlacement::Blob; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + e.blob_size = payload.size(); + + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(nsr, ref, build->buildId(), id); + return id; +} + +/// A ManifestRef carrying a unique instance id derived from `tag` (all fields explicit so the +/// missing-designated-field-initializer warning never fires). The writer/build fields are stable test +/// constants — the read path keys identity by the full ref, so any consistent choice works here. +ManifestRef manifestRefFor(const String & tag) +{ + uint32_t ordinal = 1; + for (char c : tag) + ordinal = ordinal * 131 + static_cast(c); + ordinal = ordinal % 999999 + 1; + return ManifestRef{ + .writer_epoch = 1, + .build_sequence = 1, + .manifest_ordinal = ordinal}; +} + +/// Publish a part holding the given manifest entries verbatim through the real PartWriteTxn. Used by read-path +/// lookup/list tests that want a precise multi-entry manifest. Each Blob entry's body MUST be present at +/// promote: the promote gate revalidates EVERY blob leaf with a HEAD and fails closed on an absent body. +/// So write a blob body for each Blob entry (addressed by its hash) and record it as W-EVIDENCE before +/// staging. Inline entries need no body. Returns the published ManifestId. +ManifestId publishPartWithEntries( + const PoolPtr & s, const String & ns, const String & ref, std::vector entries) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + for (const auto & e : entries) + if (e.placement == EntryPlacement::Blob) + { + /// Materialize the blob body so the promote-time HEAD revalidation succeeds, then record the + /// tokenless W-EVIDENCE dep (the gate re-observes the current token at promote). + DB::Cas::tests::writeBlobBody(s->backend(), s->layout(), e.ref.digest.toU128()); + build->adoptEvidence(e); + } + const ManifestId id = build->stageManifest(std::move(entries)); + build->precommitAdd(nsr, ref, id); + build->promote(nsr, ref, build->buildId(), id); + return id; +} +} + +TEST(CASPool, ReadOnlyOpenSkipsProbe) +{ + auto shared = std::make_shared(); + + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "test"; + /// Writable open: creates _pool_meta and runs the probe (which writes+cleans up). + DB::Cas::Pool::open(std::make_shared(shared), cfg); + + /// Read-only re-open over the SAME data must perform ZERO writes (no probe, meta already present). + auto counter = std::make_shared(shared); + DB::Cas::PoolConfig ro = cfg; + ro.read_only = true; + auto store = DB::Cas::Pool::open(counter, ro); + EXPECT_EQ(counter->writes, 0u); + ASSERT_NE(store, nullptr); +} + +namespace +{ +/// Records whether any MUTATING op touched a `_probe/` key, so a test can assert an open ran (or +/// skipped) the capability probe. Mirrors WriteCountingBackend above but keys on the probe subtree. +class ProbeWatchingBackend final : public DB::Cas::Backend +{ +public: + explicit ProbeWatchingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} + bool probe_touched = false; + + std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { note(k); return inner->putIfAbsent(k, b, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + note(request.destination_key); + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { note(k); return inner->putOverwrite(k, b, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { note(k); return inner->casPut(k, b, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { note(k); return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } +private: + void note(const String & k) { if (k.find("/_probe/") != String::npos) probe_touched = true; } + std::shared_ptr inner; +}; +} + +TEST(CASPool, SkipAccessCheckOpenSkipsProbeButStaysWritable) +{ + auto shared = std::make_shared(); + + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "srv-1"; + + /// Baseline: a normal writable open runs the capability probe (PUT+delete of `_probe/` keys). + { + auto watch = std::make_shared(shared); + auto s = DB::Cas::Pool::open(watch, cfg); + ASSERT_NE(s, nullptr); + EXPECT_TRUE(watch->probe_touched) << "the probe must run by default"; + } + + /// skip_access_check open ("start now, fix later"): NO probe I/O, yet still a WRITABLE mount + /// (owner/epoch/mount/watermark bootstrap writes still happen — unlike a read_only open, which is + /// a total no-op). Distinct root over the same (now-created) pool. + { + auto watch = std::make_shared(shared); + DB::Cas::PoolConfig sac = cfg; + sac.server_id = DB::UInt128(2); + sac.server_root_id = "srv-2"; + sac.skip_access_check = true; + auto s = DB::Cas::Pool::open(watch, sac); + ASSERT_NE(s, nullptr); + EXPECT_FALSE(watch->probe_touched) << "skip_access_check must perform no probe I/O"; + + /// Prove the mount is genuinely WRITABLE, not merely non-null — a read_only open would also + /// satisfy the two assertions above. Publish a part through the real PartWriteTxn write path + /// (beginPartWrite/putBlob/stageManifest/precommitAdd/promote) and read it back. + publishPart(s, "srv-2/tbl", "part_1", "payload-x"); + const auto r = s->resolveRef(DB::Cas::RootNamespace{"srv-2/tbl"}, "part_1"); + ASSERT_TRUE(r.has_value()) << "skip_access_check open must accept real writes, not just open"; + } +} + +namespace +{ +/// Delegates every storage operation to `inner` and leaves the mount-time capability gates at their +/// permissive defaults, so a subclass can make exactly ONE gate throw and a test can attribute a +/// refused mount to that gate alone. +class ForwardingBackend : public DB::Cas::Backend +{ +public: + explicit ForwardingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} + + std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + std::shared_ptr inner; +}; + +/// A backend whose checkConditionalWriteSingleAttemptSupport ALWAYS throws — a stand-in for a +/// Native-mode backend with no working single-attempt client (see +/// ObjectStorageBackend::checkConditionalWriteSingleAttemptSupport). Pins that skip_access_check does +/// NOT bypass this gate: the regression this guards is reverting Pool::open's skip_access_check +/// branch back to the naive "wrap the whole probe" shape, which would silently skip this check too. +class ThrowingSingleAttemptBackend final : public ForwardingBackend +{ +public: + using ForwardingBackend::ForwardingBackend; + + void checkConditionalWriteSingleAttemptSupport() override + { + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "test: no single-attempt client"); + } +}; + +/// A backend that forbids skipping the access-check battery — a stand-in for the writable +/// generation-dialect (GCS) backend (see ObjectStorageBackend::checkSkipAccessCheckSupport). +class ThrowingSkipAccessCheckBackend final : public ForwardingBackend +{ +public: + using ForwardingBackend::ForwardingBackend; + + void checkSkipAccessCheckSupport() override + { + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "test: this backend forbids skip_access_check"); + } +}; + +DB::Cas::PoolConfig writablePoolConfigForTest() +{ + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "test"; + return cfg; +} +} + +TEST(CASPool, SkipAccessCheckStillEnforcesSingleAttemptGate) +{ + auto backend = std::make_shared(std::make_shared()); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + cfg.skip_access_check = true; + + /// skip_access_check must NOT bypass checkConditionalWriteSingleAttemptSupport (RFC + /// cas-s3-timeout-retry-control): a writable open still refuses to mount on a backend that cannot + /// prove single-attempt conditional-write support, exactly as it does without skip_access_check. + EXPECT_THROW(DB::Cas::Pool::open(backend, cfg), DB::Exception); +} + +/// A backend that forbids skipping the battery refuses the writable mount outright. Asserting the +/// gate's own message, not merely that open threw: Pool::open has many other refusals, and a mount +/// that failed for one of those would satisfy a bare EXPECT_THROW. +TEST(CASPool, SkipAccessCheckRefusedByBackendFailsTheWritableMount) +{ + auto backend = std::make_shared(std::make_shared()); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + cfg.skip_access_check = true; + + try + { + DB::Cas::Pool::open(backend, cfg); + FAIL() << "expected the skip_access_check gate to refuse the mount"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + EXPECT_NE(e.message().find("forbids skip_access_check"), std::string::npos) << "actual message: " << e.message(); + } +} + +/// The discriminator for the test above: the SAME backend opens fine without the flag, so that +/// refusal came from the new gate rather than from anything else in the open path. It also pins the +/// gate's scope — it is consulted only where skip_access_check is honoured, so a mount that runs the +/// battery is unaffected. +TEST(CASPool, BackendForbiddingSkipAccessCheckStillOpensWhenTheBatteryRuns) +{ + auto backend = std::make_shared(std::make_shared()); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + cfg.background_watermark = false; + ASSERT_FALSE(cfg.skip_access_check); + + auto store = DB::Cas::Pool::open(backend, cfg); + ASSERT_NE(store, nullptr); +} + +TEST(CASPool, MinActiveTracksInFlightBuilds) +{ + auto backend = std::make_shared(); + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "test"; + cfg.background_watermark = false; + auto store = DB::Cas::Pool::open(backend, cfg); + + ASSERT_EQ(store->minActive(), store->peekNextBuildSeq()); /// no builds: floor == next seq + auto b1 = store->beginPartWrite({}); /// seq 1 + auto b2 = store->beginPartWrite({}); /// seq 2 + ASSERT_EQ(store->minActive(), 1u); + b1->abandon(); /// finishes seq 1 + ASSERT_EQ(store->minActive(), 2u); /// floor advances + b2->abandon(); + ASSERT_EQ(store->minActive(), store->peekNextBuildSeq()); /// empty again +} + +/// A throwing audit sink must NOT break a storage operation. The single reentrancy-safe event +/// dispatcher (stage-1 §1, Task 2) CONTAINS sink exceptions ("never throws through"), so an arbitrary +/// observer/sink callback failing during `beginPartWrite` is swallowed and construction succeeds -- +/// consistent with `CASPartWriteTxn.AbandonSwallowsThrowingEventSink` and +/// `PromoteSwallowsPostDurableEventSinkFailure`, which already establish that an audit-sink failure +/// never aborts the operation. Before Task 2 the sink was invoked directly and its exception +/// propagated out of construction (audit-log backpressure breaking a write); the dispatcher removes +/// that. The build_seq lifecycle is still exercised: the in-flight build holds the `minActive` GC +/// floor and is retired on `abandon`. +TEST(CASPool, BeginPartWriteSwallowsThrowingEventSink) +{ + auto backend = std::make_shared(); + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "test"; + cfg.background_watermark = false; + auto store = DB::Cas::Pool::open(backend, cfg); + + const uint64_t next_seq = store->peekNextBuildSeq(); + /// UNKNOWN_EXCEPTION (not LOGICAL_ERROR): this simulates an arbitrary observer/sink callback + /// failing, not a CAS invariant violation -- LOGICAL_ERROR would abort the whole process under + /// debug/sanitizer builds instead of behaving like a catchable exception. + store->setEventSink([](const CasEvent & e) + { + if (e.type == CasEventType::BuildStart) + throw DB::Exception(DB::ErrorCodes::UNKNOWN_EXCEPTION, "injected audit sink failure"); + }); + + PartWriteTxnPtr build; + ASSERT_NO_THROW({ build = store->beginPartWrite({}); }) + << "a throwing audit sink must be contained by the dispatcher, not fail construction"; + store->setEventSink(nullptr); + + EXPECT_EQ(build->buildSeq(), next_seq); + EXPECT_EQ(store->peekNextBuildSeq(), next_seq + 1); + EXPECT_EQ(store->minActive(), build->buildSeq()); /// the in-flight build holds the floor + build->abandon(); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); /// retired on abandon +} + +TEST(CASPool, BuildSeqIsStrictlyMonotone) +{ + auto backend = std::make_shared(); + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(1); + cfg.server_root_id = "test"; + cfg.background_watermark = false; + auto store = DB::Cas::Pool::open(backend, cfg); + auto a = store->beginPartWrite({}); + auto sa = a->buildSeq(); + a->abandon(); + auto b = store->beginPartWrite({}); + ASSERT_GT(b->buildSeq(), sa); /// never reused, never lower +} + +TEST(CASPoolMeta, CreateThenReopen) +{ + auto b = std::make_shared(); + Layout layout("p"); + PoolMeta created = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 256, + BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_NE(created.pool_id, UInt128{}); + PoolMeta reopened = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512); + EXPECT_EQ(reopened.pool_id, created.pool_id); /// pool is authoritative — config ignored on reopen + EXPECT_EQ(reopened.blob_header_len, 256u); +} + +TEST(CASPoolMeta, FailClosed) +{ + Layout layout("p"); + /// Garbage bytes are not a valid cas_pool_meta text object => CORRUPTED_DATA at the header line + /// (createOrValidate path). The future-version fail-closed (v > G_BUILD => UNKNOWN_FORMAT_VERSION) + /// is exercised at the codec level by the battery's per-row v+1 gate. + auto b2 = std::make_shared(); + b2->putIfAbsent(layout.poolMetaKey(), "garbage"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { PoolMeta::createOrValidate(*b2, layout, 256); }); +} + +TEST(CASPoolMeta, RoundTripAndReadability) +{ + PoolMeta pm; + pm.pool_id = hexToU128("0123456789abcdeffedcba9876543210"); + pm.blob_header_len = 256; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + + const String encoded = encodePoolMeta(pm); + /// v3 text form: a header line + one JSON body object, human-readable (jq/less friendly). No binary + /// magic; the object starts with '{' and names its type so a reader can identify it by eye. + ASSERT_GE(encoded.size(), 8u); + EXPECT_EQ(encoded.front(), '{'); + EXPECT_NE(encoded.find(String("cas_pool_meta")), String::npos); + EXPECT_EQ(encoded.find(String("CAPM")), String::npos); + + PoolMeta decoded = decodePoolMeta(encoded); + EXPECT_EQ(decoded.pool_id, pm.pool_id); + EXPECT_EQ(decoded.blob_header_len, pm.blob_header_len); +} + +TEST(CASPoolMeta, RejectsBadConstantsAtCreation) +{ + auto b = std::make_shared(); + Layout layout("p"); + + /// not 8-aligned (above the floor, so it is the alignment rule that rejects it) + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, + [&] { PoolMeta::createOrValidate(*b, layout, 250); }); + /// below the v3 envelope floor (240) but 8-aligned: rejected by the floor, not the alignment rule. + /// Without the raised floor this pool would pass creation and LOGICAL_ERROR on the first blob write. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, + [&] { PoolMeta::createOrValidate(*b, layout, 128); }); + /// well below the floor + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, + [&] { PoolMeta::createOrValidate(*b, layout, 64); }); + /// above the 16 KiB ceiling + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, + [&] { PoolMeta::createOrValidate(*b, layout, 17 * 1024); }); + + /// A creation that fails config validation must not have written anything. + EXPECT_FALSE(b->get(layout.poolMetaKey()).has_value()); +} + +TEST(CASPoolMeta, RejectsBadConstantsOnDecode) +{ + auto b = std::make_shared(); + Layout layout("p"); + /// Encode a PoolMeta with blob_header_len=100 (not 8-aligned); decode must reject it as CORRUPTED_DATA. + PoolMeta bad_pm; + bad_pm.pool_id = hexToU128("00000000000000000000000000000001"); + bad_pm.blob_header_len = 100; /// violates 8-alignment invariant + b->putIfAbsent(layout.poolMetaKey(), encodePoolMeta(bad_pm)); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { PoolMeta::createOrValidate(*b, layout, 256); }); +} + +TEST(CASPoolMeta, DecodeGarbageFails) +{ + /// Any non-CAPM framing byte sequence => CORRUPTED_DATA. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { decodePoolMeta(String("garbage")); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { decodePoolMeta(String("")); }); +} + +TEST(CASPoolMeta, ConcurrentCreateRace) +{ + auto b = std::make_shared(); + Layout layout("p"); + + /// A racing creator already wrote a valid foreign pool_id. createOrValidate must NOT overwrite it: + /// it re-reads (after losing the create-if-absent CAS, or seeing it present) and returns the + /// foreign pool_id, validated like a reopen. + const UInt128 foreign = hexToU128("0123456789abcdeffedcba9876543210"); + PoolMeta foreign_pm; + foreign_pm.pool_id = foreign; + foreign_pm.blob_header_len = 256; + foreign_pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + b->putIfAbsent(layout.poolMetaKey(), encodePoolMeta(foreign_pm)); + + PoolMeta result = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512); + EXPECT_EQ(result.pool_id, foreign); + EXPECT_EQ(result.blob_header_len, 256u); /// the foreign pool's constants win +} + +TEST(CASPoolMeta, CasConflictReReadsWinner) +{ + /// The subtlest branch: the initial GET sees ABSENT, so createOrValidate proceeds to the + /// create-if-absent casPut — and loses, because a racing creator committed in between. The loser + /// must then re-read and return the WINNER's pool identity, not LOGICAL_ERROR. A single-threaded + /// `failNextCasPut` alone cannot exercise this: it returns Conflict without leaving the object + /// readable, so the re-read would fire the LOGICAL_ERROR guard. We model the real interleaving + /// with a backend whose casPut commits the winner's object (via the public putIfAbsent) and THEN + /// reports Conflict — exactly what the loser observes. + class RacingBackend : public InMemoryBackend + { + public: + String winner_bytes; + CasResult casPut(const String & key, const String & bytes, + const std::optional & expected, const ObjectMeta & meta) override + { + if (!winner_committed) + { + winner_committed = true; + /// The winner lands first; our create-if-absent now necessarily conflicts. + putIfAbsent(key, winner_bytes); + return {CasOutcome::Conflict, {}}; + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + private: + bool winner_committed = false; + }; + + const UInt128 winner = hexToU128("0123456789abcdeffedcba9876543210"); + PoolMeta winner_pm; + winner_pm.pool_id = winner; + winner_pm.blob_header_len = 256; + winner_pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + + auto b = std::make_shared(); + b->winner_bytes = encodePoolMeta(winner_pm); + Layout layout("p"); + + /// Our config (512) is what we WOULD have minted, but we lose the race and inherit the winner. + PoolMeta result = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512, + BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + EXPECT_EQ(result.pool_id, winner); + EXPECT_EQ(result.blob_header_len, 256u); +} + +TEST(CASPool, OpenFailsClosedOnNonEnforcingBackend) +{ + auto b = std::make_shared(); + b->setEnforceTokens(false); + expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, + [&] { Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); /// the probe error contract +} + +TEST(CASPool, OpenCreatesPoolMetaAndReopens) +{ + auto b = std::make_shared(); + /// Two CONCURRENT opens over the same POOL: a shared pool is the multi-server model, so each + /// mounts a DISTINCT server_root_id (and a distinct server_id) — same-root same-uuid co-mounting + /// is correctly fail-closed by the mount-safety protocol. This test only asserts that pool-meta is + /// pool-authoritative and shared across opens. + auto s1 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "srv-1"}); + auto s2 = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(2), .server_root_id = "srv-2"}); + EXPECT_EQ(s1->poolMeta().pool_id, s2->poolMeta().pool_id); /// pool authoritative +} + +TEST(CASPool, OpenWithExplicitConstantsCreatesThem) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .blob_header_len = 512}); + EXPECT_EQ(s->poolMeta().blob_header_len, 512u); /// config applies at creation +} + +TEST(CASPool, VerbatimFilesLifecycle) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt", "1\n"); + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "uuid.txt", "abc"); + EXPECT_EQ(s->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt"), String("1\n")); + EXPECT_FALSE(s->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "absent").has_value()); + auto names = s->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_EQ(names, (std::vector{"format_version.txt", "uuid.txt"})); + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "uuid.txt", "def"); /// overwrite allowed (head + putOverwrite) + EXPECT_EQ(s->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "uuid.txt"), String("def")); +} + +TEST(CASPool, ListNamespaceFilesEmpty) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + EXPECT_TRUE(s->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)).empty()); +} + +/// ---------- read side (spec §6): resolveRef / readManifest / findEntry / entryRange / listRefs ---------- + +/// Phase 1c read path: a published ref resolves to a ManifestId; readManifest returns the immutable +/// body; locate yields a ranged blob read; an Inline entry has no location. Replaces the old +/// resolveRef().tree_id / readTree round trip (the tree model is gone — a part is a single ManifestId). +TEST(CASPool, ResolveReturnsManifestId) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl"}; + + /// blob "hello world" + an inline file, published through the real PartWriteTxn write path. + const String payload = "hello world"; + PartWriteInfo info; + info.intended_ref = ns.string() + "/part_1"; + auto build = s->beginPartWrite(info); + + ManifestEntry blob_entry; + blob_entry.path = "data.bin"; + blob_entry.placement = EntryPlacement::Blob; + blob_entry.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload))}; + + blob_entry.blob_size = payload.size(); + ManifestEntry inline_entry; + inline_entry.path = "small.txt"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.inline_bytes = "tiny\n"; + + const ManifestId id = build->stageManifest({blob_entry, inline_entry}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, "part_1", build->buildId(), id); + + auto r = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->manifest_id, id); /// resolve yields the published ManifestId + + auto manifest = s->readManifest(r->manifest_id); + ASSERT_EQ(manifest.entries.size(), 2u); + + /// "data.bin" sorts before "small.txt" (canonical path order). + const auto * data = findEntry(manifest.entries, "data.bin"); + ASSERT_TRUE(data != nullptr); + auto loc = s->locate(*data); + EXPECT_EQ(loc.offset, s->poolMeta().blob_header_len); + EXPECT_EQ(loc.length, payload.size()); + + auto bytes = b->get(loc.key, Range{loc.offset, loc.length}); + ASSERT_TRUE(bytes.has_value()); + EXPECT_EQ(bytes->bytes, payload); /// ranged read, no header touch + + const auto * small = findEntry(manifest.entries, "small.txt"); + ASSERT_TRUE(small != nullptr); + EXPECT_THROW(s->locate(*small), DB::Exception); /// Inline has no location +} + +/// readManifest fail-closes on a body whose self-described `ref`/`root_namespace_id` does NOT match the +/// resolved ManifestId — the ref is addressing the wrong object / a cross-namespace dangle. We stage a +/// body raw (writeManifestRaw, the on-storage write fixture) at a ManifestId, then resolve through a +/// committed binding that names a DIFFERENT ManifestRef pointing at the SAME object key — so the head +/// succeeds, the body decodes, but refMatchesBody fails => CORRUPTED_DATA. +TEST(CASPool, ReadManifestValidatesBodyAndFailsClosed) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl"}; + Layout layout("p"); + + /// (1) ref/namespace mismatch: the BODY self-describes namespace `srv1/other`, but it is addressed + /// as a manifest of `srv1/tbl` => manifestNamespaceMatches fails => CORRUPTED_DATA. We craft an id + /// whose key lives under `srv1/tbl` but whose body carries the foreign namespace. + { + const ManifestRef ref = manifestRefFor("mismatch-ns"); + const ManifestId addressed{.root_namespace = ns, .ref = ref}; + /// Encode a body that claims a DIFFERENT namespace than `addressed.root_namespace`. + PartManifest body; + body.ref = ref; /// ref matches + body.root_namespace_id = RootNamespace{"srv1/other"}; /// namespace does NOT + body.entries = {blobEntryFor("f", u128Of("x"), 1)}; + body.payload_digest = computePayloadDigest(body); + b->putIfAbsent(layout.manifestKey(addressed), encodePartManifest(body)); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(addressed); }); + } + + /// (2) ref mismatch: the body self-describes a DIFFERENT ManifestRef than the id addressing it => + /// refMatchesBody fails => CORRUPTED_DATA. + { + const ManifestRef addressed_ref = manifestRefFor("addressed-ref"); + const ManifestRef body_ref = manifestRefFor("body-ref-other"); + const ManifestId addressed{.root_namespace = ns, .ref = addressed_ref}; + PartManifest body; + body.ref = body_ref; /// ref does NOT match `addressed` + body.root_namespace_id = ns; /// namespace matches + body.entries = {blobEntryFor("f", u128Of("y"), 1)}; + body.payload_digest = computePayloadDigest(body); + b->putIfAbsent(layout.manifestKey(addressed), encodePartManifest(body)); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(addressed); }); + } + + /// (3) a committed ref naming a manifest with NO body present => readManifest throws + /// FILE_DOESNT_EXIST (INV-NO-DANGLE surfaced on the read path). resolveRef itself SUCCEEDS — refs + /// are pure manifest state. A raw ref-log fixture (not the real PartWriteTxn path, which validates the + /// body exists at promote) is the only way to construct this state. + { + const ManifestRef missing_ref = manifestRefFor("never-staged"); + DB::Cas::tests::fixture::writeRefLogRaw(*b, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, + {DB::Cas::tests::namespaceBirthOp(), DB::Cas::tests::publishCommittedOps("part_dangle", missing_ref)[0], + DB::Cas::tests::publishCommittedOps("part_dangle", missing_ref)[1]}, std::nullopt}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*b, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + auto r = s->resolveRef(ns, "part_dangle"); + ASSERT_TRUE(r.has_value()); + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { s->readManifest(r->manifest_id); }); + } +} + +/// findEntry and entryRange over a decoded part manifest's canonical-path-ordered entries. +TEST(CASPool, LookupAndListOverManifestEntries) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl"}; + + /// A multi-file/multi-directory part: top-level + a projection subdir. + std::vector entries; + entries.push_back(blobEntryFor("columns.txt", u128Of("cols"), 4)); + entries.push_back(blobEntryFor("data.bin", u128Of("data"), 8)); + entries.push_back(blobEntryFor("p.proj/data.bin", u128Of("proj-data"), 6)); + entries.push_back(blobEntryFor("p.proj/columns.txt", u128Of("proj-cols"), 5)); + const ManifestId id = publishPartWithEntries(s, ns.string(), "all_1_1_0", entries); + + auto r = s->resolveRef(ns, "all_1_1_0"); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->manifest_id, id); + auto manifest = s->readManifest(r->manifest_id); + ASSERT_EQ(manifest.entries.size(), 4u); + + /// findEntry: exact-path hit + miss. + const auto * hit = findEntry(manifest.entries, "data.bin"); + ASSERT_TRUE(hit != nullptr); + EXPECT_EQ(hit->ref.digest.toU128(), u128Of("data")); + EXPECT_TRUE(findEntry(manifest.entries, "no_such_file") == nullptr); + + /// entryRange under "p.proj/" yields exactly the two projection files, in canonical order. + auto [proj_first, proj_last] = entryRange(manifest.entries, "p.proj/"); + std::vector proj(proj_first, proj_last); + ASSERT_EQ(proj.size(), 2u); + EXPECT_EQ(proj[0].path, "p.proj/columns.txt"); + EXPECT_EQ(proj[1].path, "p.proj/data.bin"); + + /// The empty prefix lists everything (all four), still in canonical order. + auto [all_first, all_last] = entryRange(manifest.entries, ""); + std::vector all(all_first, all_last); + ASSERT_EQ(all.size(), 4u); + EXPECT_EQ(all[0].path, "columns.txt"); + EXPECT_EQ(all[3].path, "p.proj/data.bin"); +} + +/// The Phase 1c manifest decode cache is keyed by (ManifestId, Token). Resolve+read the same ref twice: +/// the second readManifest must be served from the cache (no second GET of the body). A fresh publish +/// under a DIFFERENT ref name mints a NEW ManifestId (and a new shard token), so the cache misses and +/// the body is fetched again. A CountingBackend asserts the body GET count. +TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"srv1/tbl"}; + Layout layout("p"); + + const ManifestId id1 = publishPart(s, ns.string(), "part_1", "payload-1"); + const String key1 = layout.manifestKey(id1); + + /// First read: a body GET populates the (id1, token) cache entry. + { + auto r = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(r.has_value()); + auto m = s->readManifest(r->manifest_id); + ASSERT_EQ(m.entries.size(), 1u); + } + const uint64_t gets_after_first = b->getCount(key1); + ASSERT_GE(gets_after_first, 1u); /// the first read DID fetch the body + + /// Second read of the SAME id: the (id, token) cache must serve it — NO additional body GET. + { + auto r = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(r.has_value()); + EXPECT_EQ(r->manifest_id, id1); + auto m = s->readManifest(r->manifest_id); + ASSERT_EQ(m.entries.size(), 1u); + } + EXPECT_EQ(b->getCount(key1), gets_after_first) + << "second readManifest re-GET the body for the same (ManifestId, Token) — cache miss"; + + /// A fresh publish under a DIFFERENT ref name mints a NEW ManifestId: the cache (keyed by id) misses. + /// (Promoting a different manifest over the SAME committed ref is a distinct promote-over-committed + /// leak that `PartWriteTxn::promote` now forbids — see the CASPromoteRepublish tests.) + const ManifestId id2 = publishPart(s, ns.string(), "part_2", "payload-2"); + EXPECT_FALSE(id2 == id1); /// a new publish never reuses a ManifestId + const String key2 = layout.manifestKey(id2); + + auto r2 = s->resolveRef(ns, "part_2"); + ASSERT_TRUE(r2.has_value()); + EXPECT_EQ(r2->manifest_id, id2); /// resolve now sees the new manifest + auto m2 = s->readManifest(r2->manifest_id); + ASSERT_EQ(m2.entries.size(), 1u); + EXPECT_GE(b->getCount(key2), 1u) /// the new id's body WAS fetched (cache miss) + << "fresh publish (new ManifestId) should miss the id-keyed manifest cache"; +} + +/// Phase 5 (part-folder cache spec): manifest_cache is now a byte-weighted CacheBase LRU instead of a +/// count-only bound, since decoded manifests carry inline bytes and can each be megabytes. +TEST(CASPool, ManifestDecodeCacheIsByteBounded) +{ + auto backend = std::make_shared(); + const DB::Cas::Layout layout("p"); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + const DB::Cas::RootNamespace ns{"srv/t1"}; + + /// 8 manifests x ~1 MiB of inline bytes; a 2 MiB decode-cache bound must hold while every + /// read stays correct (evicted decodes just re-GET + re-decode). + std::vector ids; + std::vector birth_ops{DB::Cas::tests::namespaceBirthOp()}; + for (int i = 0; i < 8; ++i) + { + const DB::Cas::ManifestRef ref{.writer_epoch = 1, .build_sequence = static_cast(i + 1), + .manifest_ordinal = 1}; + DB::Cas::ManifestEntry e; + e.path = "big.txt"; + e.placement = DB::Cas::EntryPlacement::Inline; + e.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(DB::UInt128(i + 1))}; + + e.inline_bytes = String(1 << 20, static_cast('a' + i)); + e.blob_size = e.inline_bytes.size(); + ids.push_back(DB::Cas::tests::writeManifestRaw(*backend, layout, ns, ref, {e})); + + const String ref_name = "part_" + std::to_string(i); + std::vector ops = i == 0 ? birth_ops : std::vector{}; + const auto committed_ops = DB::Cas::tests::publishCommittedOps(ref_name, ref); + ops.insert(ops.end(), committed_ops.begin(), committed_ops.end()); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, static_cast(i + 1)}, ops, std::nullopt}); + } + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 8}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + DB::Cas::PoolConfig config{.pool_prefix = "p", .server_root_id = "test"}; + config.manifest_decode_cache_bytes = 2ULL << 20; + auto store = DB::Cas::Pool::open(backend, std::move(config)); + + uint64_t total_gets = 0; + for (int round = 0; round < 2; ++round) + for (int i = 0; i < 8; ++i) + { + auto resolved = store->resolveRef(ns, "part_" + std::to_string(i)); + ASSERT_TRUE(resolved.has_value()); + auto m = store->readManifestShared(resolved->manifest_id); + ASSERT_EQ(m->entries.size(), 1u); + EXPECT_EQ(m->entries[0].inline_bytes[0], static_cast('a' + i)); /// always correct + } + for (const auto & id : ids) + total_gets += backend->getCount(layout.manifestKey(id)); + + /// The bound forces re-GETs (16 reads over a 2 MiB window of ~1 MiB decodes cannot all hit), + /// proving eviction actually happens... + EXPECT_GT(total_gets, 8u); + /// ...and the cache reports an in-bound retained size. + EXPECT_LE(store->manifestDecodeCacheBytesForTest(), 2ULL << 20); +} + +TEST(CASPool, ResolveDecodeCacheInvalidatesOnWrite) +{ + /// B113: resolveRef uses a token-validated shard-manifest decode cache. A write to the shard + /// mints a new token, so a subsequent resolve must observe the change (cache must NOT serve a + /// stale decoded manifest). Without token invalidation this would still see the dropped ref. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + publishPart(s, ns.string(), "part_1", "payload-1"); + + /// First resolve decodes + caches; second is a cache hit — both must see part_1. + ASSERT_TRUE(s->resolveRef(ns, "part_1").has_value()); + ASSERT_TRUE(s->resolveRef(ns, "part_1").has_value()); + + /// Write through the Pool (mutateShard => new shard token), removing part_1. + s->dropRef(ns, "part_1"); + + /// The cache must invalidate on the token change: resolve now reflects the drop. + EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_TRUE(s->listRefs(ns).empty()); +} + +TEST(CASPool, ResolveAbsentRefAndAbsentNamespace) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + /// A freshly-opened pool has no shard manifests: an absent shard is an empty manifest, so resolve + /// yields nullopt and listRefs is empty (NOT an error). + EXPECT_FALSE(s->resolveRef(ns, "anything").has_value()); + EXPECT_TRUE(s->listRefs(ns).empty()); +} + +TEST(CASPool, ListRefsMergesAllShards) +{ + /// Task 10: refs are no longer sharded (the snapshot+log protocol caches one coherent table state + /// per namespace, not one manifest per shard) -- this now proves listRefs returns every committed + /// ref of a table built from a single multi-owner transaction, the closest surviving analogue of + /// the old "merges refs spread across shards" contract. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + RootNamespace ns{"srv1/tbl"}; + + std::vector ops{DB::Cas::tests::namespaceBirthOp()}; + for (char c = 'a'; c <= 'h'; ++c) + { + const String ref(1, c); + const auto committed_ops = DB::Cas::tests::publishCommittedOps(ref, manifestRefFor("manifest-" + ref)); + ops.insert(ops.end(), committed_ops.begin(), committed_ops.end()); + } + DB::Cas::tests::fixture::writeRefLogRaw(*b, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, ops, std::nullopt}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*b, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + auto refs = s->listRefs(ns); + ASSERT_EQ(refs.size(), 8u); + for (char c = 'a'; c <= 'h'; ++c) + { + const String ref(1, c); + ASSERT_TRUE(refs.count(ref)); + EXPECT_EQ(refs.at(ref).manifest_id.ref, manifestRefFor("manifest-" + ref)); + EXPECT_EQ(refs.at(ref).manifest_id.root_namespace.string(), ns.string()); + } +} + +/// An empty namespace recovers from its exact `_ckpt` authority and exact successor GET. It performs +/// ZERO LISTs and ZERO HEADs: recovery no longer enumerates the stream, and it never probes a shard +/// fan-out. Measure deltas around `listRefs`; `Pool::open` and fixture admission have their own metadata +/// traffic. +TEST(CASPool, ListRefsEmptyNamespaceCostsZeroListsAndHeads) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + /// EMPTY, but EXISTING and recoverable. A namespace the catalog does not name is answered from the + /// catalog and never reaches recovery; that separate shape is measured by the case below. + DB::Cas::tests::casAdmitRecoverableEntry(*b, Layout("p"), ns); + + const uint64_t heads_before = b->headTotal(); + const uint64_t lists_before = b->listTotal(); + + auto refs = s->listRefs(ns); + + EXPECT_TRUE(refs.empty()); + EXPECT_EQ(b->headTotal() - heads_before, 0u) + << "empty-namespace listRefs must not HEAD any shard"; + EXPECT_EQ(b->listTotal() - lists_before, 0u) + << "checkpoint-grounded recovery reads exact keys and must not LIST the ref stream"; +} + +/// The other shape: a namespace that was never born. A read must not be what brings one into existence, +/// so the answer comes from the catalog alone -- no recovery, and therefore not even the one LIST the +/// case above pins. +TEST(CASPool, ListRefsOnANeverBornNamespaceCostsNoListAndNoHead) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + const uint64_t heads_before = b->headTotal(); + const uint64_t lists_before = b->listTotal(); + const uint64_t gets_before = b->getTotal(); + + auto refs = s->listRefs(ns); + + EXPECT_TRUE(refs.empty()); + EXPECT_EQ(b->listTotal() - lists_before, 0u) + << "a never-born namespace has no ref stream to LIST"; + EXPECT_EQ(b->headTotal() - heads_before, 0u); + /// Positive control: the zeros above are the answer coming from the catalog, not from a call that + /// did nothing at all. + EXPECT_GT(b->getTotal() - gets_before, 0u) + << "the answer must come from a catalog read"; +} + +/// listRefs must return every committed ref of a table, correctly, regardless of how many refs the +/// table holds (Task 10: there is no more shard fan-out to discover -- see the comment inside). +TEST(CASPool, ListRefsReturnsSameContentAsBefore) +{ + /// Task 10: there is no more per-shard HEAD fan-out to bound (a warm listRefs costs ZERO requests; + /// a cold empty one costs zero LISTs and HEADs, already covered by + /// `ListRefsEmptyNamespaceCostsZeroListsAndHeads`) -- this now just proves the returned content is + /// correct for a multi-ref table built from a single raw ref-log fixture. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + RootNamespace ns{"srv1/tbl"}; + + std::vector ops{DB::Cas::tests::namespaceBirthOp()}; + for (const String & ref : {String("a"), String("m"), String("z")}) + { + const auto committed_ops = DB::Cas::tests::publishCommittedOps(ref, manifestRefFor("manifest-" + ref)); + ops.insert(ops.end(), committed_ops.begin(), committed_ops.end()); + } + DB::Cas::tests::fixture::writeRefLogRaw(*b, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, ops, std::nullopt}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*b, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + auto refs = s->listRefs(ns); + + ASSERT_EQ(refs.size(), 3u); + for (const String & ref : {String("a"), String("m"), String("z")}) + { + ASSERT_TRUE(refs.count(ref)); + EXPECT_EQ(refs.at(ref).manifest_id.ref, manifestRefFor("manifest-" + ref)); + EXPECT_EQ(refs.at(ref).manifest_id.root_namespace.string(), ns.string()); + } +} + +/// A stray key under the namespace's ref-object prefix that does not parse as one of Task 10's +/// `_log`/`_snap` kinds (a foreign/corrupt object) must not break listRefs — it is skipped +/// defensively, listRefs still returns the legit refs and never throws. +TEST(CASPool, ListRefsSkipsForeignKeys) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + RootNamespace ns{"srv1/tbl"}; + + const String ref = "legit"; + const ManifestRef mref = manifestRefFor("manifest-" + ref); + DB::Cas::tests::fixture::writeRefLogRaw(*b, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, + {DB::Cas::tests::namespaceBirthOp(), DB::Cas::tests::publishCommittedOps(ref, mref)[0], + DB::Cas::tests::publishCommittedOps(ref, mref)[1]}, std::nullopt}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*b, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + /// A stray key directly under the namespace's ref-object prefix that is not `_log`/ + /// `_snap` shaped (also covers the legacy shard-number layout GC/dropNamespace still write). + b->putIfAbsent(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "garbage", "not-a-ref-object"); + + std::map refs; + EXPECT_NO_THROW(refs = s->listRefs(ns)); + ASSERT_EQ(refs.size(), 1u); + ASSERT_TRUE(refs.count(ref)); + EXPECT_EQ(refs.at(ref).manifest_id.ref, mref); +} + +/// readManifest fails CLOSED on a corrupt or kind-mismatched manifest body addressed by a live id. +TEST(CASPool, ReadManifestFailsClosed) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + const RootNamespace ns{"srv1/tbl"}; + + /// (1) Garbage bytes at the manifest key => decodePartManifest throws CORRUPTED_DATA. + { + const ManifestRef ref = manifestRefFor("garbage-body"); + const ManifestId id{.root_namespace = ns, .ref = ref}; + b->putIfAbsent(layout.manifestKey(id), "not a valid manifest body"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(id); }); + } + + /// (2) A ref naming a manifest id with NO object present => readManifest throws FILE_DOESNT_EXIST + /// (INV-NO-DANGLE), carrying the manifest key. + { + const ManifestRef ref = manifestRefFor("absent-body"); + const ManifestId id{.root_namespace = ns, .ref = ref}; + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { s->readManifest(id); }); + } +} + +/// ---------- ref lifecycle: dropRef / updateRefPublishedAt / dropNamespace ---------- + +TEST(CASPool, DropRefAppendsJournalAtomically) +{ + /// Task 10: the OLD shared-journal record assertions are gone (there is no shared mutable journal + /// object anymore — dropRef appends its OWN immutable ref-log transaction); the surviving + /// behavioral contract is: the drop is atomic (visible to resolveRef only once durable), and + /// dropping a missing ref is fail-closed, never a silent no-op. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + publishPart(s, ns.string(), "part_1", "payload-1"); + ASSERT_TRUE(s->resolveRef(ns, "part_1").has_value()); + + s->dropRef(ns, "part_1"); + EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_TRUE(s->listRefs(ns).empty()); + + /// Dropping a missing ref is fail-closed, never a silent no-op. + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { s->dropRef(ns, "no_such_ref"); }); +} + +/// Task 10 renamed this from "...WithoutJournal": updateRefPublishedAt now DOES append an immutable +/// `set_published_at` ref-log transaction (spec §Update Payload) -- the old journal-free in-place field +/// mutation had no equivalent once persistence is an append-only log; every change, even timestamp-only, +/// must be a logged operation to be part of the ordered history. All-tree-part-files Task 9: the +/// carrier's mutable-file map is gone -- `published_at_ms` is the only field left to mutate. The +/// surviving contract is the user-visible one: a `published_at_ms` update is observable through +/// resolveRef and the manifest edge cannot change on this path -- the `RefPublishedAtUpdate` carrier +/// deliberately has no `manifest_ref` field, so a reachability change is structurally impossible here +/// (it goes through publish/drop/repoint instead). +TEST(CASPool, UpdateRefPublishedAtUpdatesPublishedAtMs) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + const ManifestId id = publishPart(s, ns.string(), "part_1", "payload-1"); + const ManifestRef manifest_ref = id.ref; + + s->updateRefPublishedAt(ns, "part_1", [](RefPublishedAtUpdate & r) { r.published_at_ms = 1; }); + s->updateRefPublishedAt(ns, "part_1", [](RefPublishedAtUpdate & r) { r.published_at_ms = 7; }); + + auto after = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(after.has_value()); + EXPECT_EQ(after->published_at_ms, 7u); + EXPECT_EQ(after->manifest_id.ref, manifest_ref); +} + +/// Task 11: dropNamespace removes every owner through the ref-log `remove_namespace` transaction and +/// performs NO physical deletion at all -- verbatim files survive until GC's perpetual janitor +/// reclaims the dead life. So after the drop every ref resolves away and +/// `listRefs` is empty, but the verbatim files remain readable. +TEST(CASPool, DropNamespaceRemovesEveryOwnerButLeavesFilesForGc) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + RootNamespace ns{"srv1/tbl"}; + + const std::vector ref_names{"alpha", "bravo", "charlie"}; + for (const String & name : ref_names) + publishPart(s, ns.string(), name, "payload-" + name); + for (const String & name : ref_names) + ASSERT_TRUE(s->resolveRef(ns, name).has_value()); + + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt", "1\n"); + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "uuid.txt", "abc"); + + s->dropNamespace(ns); + + for (const String & name : ref_names) + EXPECT_FALSE(s->resolveRef(ns, name).has_value()); + EXPECT_TRUE(s->listRefs(ns).empty()); + + /// The writer performs NO physical deletion; verbatim files survive until the perpetual janitor + /// reclaims the dead life. + EXPECT_TRUE(s->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt").has_value()); + EXPECT_TRUE(s->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "uuid.txt").has_value()); + + /// Repeated drop is idempotent: no throw, no second transaction (nothing left to observe changing). + EXPECT_NO_THROW(s->dropNamespace(ns)); + + /// Ordinary mutations on a cataloged `Removing` life are rejected with typed retry-later until + /// the terminal fold and catalog-only drain complete. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { s->dropRef(ns, "alpha"); }); +} + +TEST(CASPool, ListNamespacesFromCatalog) +{ + /// `listNamespaces` projects logical names from the authoritative catalog. Physical life keys + /// contain no namespace spelling and therefore cannot participate in this enumeration. + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + EXPECT_TRUE(s->listNamespaces("").namespaces.empty()); /// fresh pool: empty catalog + + /// The real publication path admits each namespace before writing its stream. + DB::Cas::tests::publishCommittedTransition(*b, s->layout(), RootNamespace{"srv1/tbl"}, + "ref1", std::nullopt, DB::Cas::ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}); + DB::Cas::tests::publishCommittedTransition(*b, s->layout(), RootNamespace{"srv1/shadow/bk1/tbl"}, + "ref1", std::nullopt, DB::Cas::ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}); + DB::Cas::tests::publishCommittedTransition(*b, s->layout(), RootNamespace{"srv1/shadow/bk2/tbl"}, + "ref1", std::nullopt, DB::Cas::ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}); + + const auto all = s->listNamespaces("").namespaces; + EXPECT_EQ(all.size(), 3u); + const auto shadows = s->listNamespaces("srv1/shadow/").namespaces; + ASSERT_EQ(shadows.size(), 2u); + /// listNamespaces returns results from an unordered_set; sort for deterministic comparison. + auto sorted_shadows = shadows; + std::sort(sorted_shadows.begin(), sorted_shadows.end()); + EXPECT_EQ(sorted_shadows[0], "srv1/shadow/bk1/tbl"); + EXPECT_EQ(sorted_shadows[1], "srv1/shadow/bk2/tbl"); + EXPECT_TRUE(s->listNamespaces("nope/").namespaces.empty()); +} + +/// Physical namespace files carry only an opaque life id and cannot mint a logical catalog row. +TEST(CASPool, ListNamespacesDoesNotMintLogicalNamesFromFileKeys) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"test/tbl@cas@"}; + + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt", "1\n"); + /// A second life of the SAME name, written by exact key because no helper mints two lives yet. + const NamespaceLifeId other = NamespaceLifeId::fromCatalogEntry(ns, DB::UInt128(0x5eed)); + ASSERT_EQ(b->putIfAbsent(s->layout().namespaceFileKey(other, "format_version.txt"), "1\n").outcome, + PutOutcome::Done); + + const NamespaceListing listing = s->listNamespaces(""); + EXPECT_TRUE(listing.skipped.empty()); + EXPECT_TRUE(listing.namespaces.empty()); +} + +/// Catalog discovery neither adopts nor reports malformed physical debris. Diagnostic ownership-tree +/// scans, not ordinary logical enumeration, classify those keys. +TEST(CASPool, ListNamespacesDoesNotTreatPhysicalDebrisAsCatalogAuthority) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const RootNamespace ns{"test/tbl@cas@"}; + + /// One well-formed key per family, so the namespace is attributable either way. + DB::Cas::tests::publishCommittedTransition(*b, s->layout(), ns, + "ref1", std::nullopt, DB::Cas::ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}); + s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt", "1\n"); + + /// Hand-built un-incarnated keys: no helper can mint either shape any more. + const String lifeless_ref = s->layout().casRefsPrefix() + ns.string() + "/_log/" + + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + const String lifeless_file = s->layout().rootsPrefix() + ns.string() + "/_files/format_version.txt"; + ASSERT_EQ(b->putIfAbsent(lifeless_ref, "garbage").outcome, PutOutcome::Done); + ASSERT_EQ(b->putIfAbsent(lifeless_file, "garbage").outcome, PutOutcome::Done); + + NamespaceListing listing; + ASSERT_NO_THROW(listing = s->listNamespaces("")) + << "one un-attributable key must not abort the enumeration for every consumer of it"; + + /// The healthy namespace is still listed -- attribution is per key, so a namespace disappears only + /// when every key that would name it is unattributable. + ASSERT_EQ(listing.namespaces.size(), 1u); + EXPECT_EQ(listing.namespaces[0], ns.string()); + + EXPECT_TRUE(listing.skipped.empty()); + EXPECT_TRUE(b->head(lifeless_ref).exists); + EXPECT_TRUE(b->head(lifeless_file).exists); +} + +TEST(CASPool, ListMirroredChildren) +{ + using namespace DB::Cas; + auto b = std::make_shared(); + auto store = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + /// Seed two catalog-authoritative shadow archives; physical files alone carry no logical path. + DB::Cas::tests::fixture::admitLive(*b, store->layout(), RootNamespace{"srv1/shadow/bk1/store/3f2/3f2a-uuid@cas@"}); + DB::Cas::tests::fixture::admitLive(*b, store->layout(), RootNamespace{"srv1/shadow/bk2/store/3f2/3f2a-uuid@cas@"}); + auto children = store->listMirroredChildren("srv1/shadow/"); + std::sort(children.begin(), children.end()); + ASSERT_EQ(children.size(), 2u); + EXPECT_EQ(children[0], "bk1"); + EXPECT_EQ(children[1], "bk2"); +} + +namespace +{ + +/// Delegating backend that fences the mount slot IN PLACE the first time a `get` returns a present +/// body for the armed key — reproducing the S13 window: the GC's token-guarded fence-out lands +/// between the keeper adopt's GET and its CAS. The caller's subsequent token-guarded `putOverwrite` +/// then fails `PreconditionFailed`, the adopt re-reads, sees `gc_fenced`, and throws +/// `MountFencedException` — which `Pool::open`'s fence-recovery loop must turn into a fresh-epoch +/// retry rather than a permanent wedge (P3.1 vector C). +class FenceInAdoptWindowBackend final : public DB::Cas::Backend +{ +public: + explicit FenceInAdoptWindowBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} + String fence_key; /// empty = fault disarmed; set to the mount key to arm the one-shot fence + + std::optional get(const String & k, DB::Cas::Range r) override + { + auto got = inner->get(k, r); + if (!fence_key.empty() && k == fence_key && got.has_value()) + { + /// One-shot: fence the slot in place exactly as `computeHeartbeatFloor` does (preserve the + /// body, gc_fenced = true, seq + 1, token-guarded against the value we just read), then + /// disarm so the retry can adopt cleanly. + DB::Cas::MountLease fenced = DB::Cas::decodeMountLease(got->bytes); + fenced.gc_fenced = true; + fenced.seq += 1; + inner->putOverwrite(k, DB::Cas::encodeMountLease(fenced), got->token); + fence_key.clear(); + } + return got; + } + std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } + DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } + DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } + DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + inner->publishBlob(request); + } + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } + DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } + DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + +private: + std::shared_ptr inner; +}; + +} + +TEST(CASPoolMountFence, OpenRecoversFromFenceInAdoptWindowWithFreshEpoch) +{ + auto inner = std::make_shared(); + auto fencing = std::make_shared(inner); + /// Arm the one-shot fence on the mount slot. Pool::open first claims the mount (fresh mint), then + /// the keeper adopts it — the adopt's GET trips the fence, its CAS fails, and open must recover. + const DB::Cas::Layout layout("p"); + fencing->fence_key = layout.mountKey("test"); + + /// The retry that recovers from the fence reclaims a same-uuid, different-epoch, `gc_fenced` body + /// -> `MountPriorState::Fenced` (a fenced prior is reclaimed on the first attempt, with no + /// observation polling -- see `CASMountOpenWaits.FencedPriorReclaimsWithoutAnyWait`). The injected + /// `boot_ms_fn`/`wait_sleep_fn` below keep this test off the real clock regardless. + uint64_t fake_boot = 0; + DB::Cas::PoolPtr store; + ASSERT_NO_THROW( + store = DB::Cas::Pool::open(fencing, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .boot_ms_fn = [&fake_boot] { return fake_boot; }, + .wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }})) + << "open must recover from a fence in the adopt window, not wedge (exit-49 S13 bug)"; + ASSERT_TRUE(store); + + /// The final live lease is unfenced and at a HIGHER writer_epoch than the first attempt (a fence + /// costs an epoch): the first claim took epoch 1, got fenced, the retry took epoch 2 and mounted. + const auto got = inner->get(layout.mountKey("test")); + ASSERT_TRUE(got.has_value()); + const MountLease final_lease = decodeMountLease(got->bytes); + EXPECT_FALSE(final_lease.gc_fenced); + EXPECT_GT(final_lease.writer_epoch, 1u) << "recovery must draw a fresh writer_epoch"; + EXPECT_TRUE(fencing->fence_key.empty()) << "the one-shot fence must have fired"; +} + +/// Task 12: the write-fence deadline is a CLOCK_BOOTTIME instant (boottime includes VM-suspend time, +/// so a resumed sleeper sees its fence expired — unlike CLOCK_MONOTONIC, which freezes across suspend). +/// A CLOCK_MONOTONIC freeze cannot be simulated in a unit test, so we exercise the injected-fn seam: a +/// fake boot clock that we advance past the ttl must flip mayMutate to false and make a gated mutate +/// fail closed with ABORTED. +TEST(CASPool, WriteFenceUsesInjectedBootClock) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 1'000'000; /// arbitrary boottime origin (ms) + auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ + .pool_prefix = "p", + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .boot_ms_fn = [&] { return fake_boot; }, + }); + + /// Freshly armed at open (deadline = fake_boot + ttl): well within the ttl, mutations are allowed. + EXPECT_TRUE(store->mayMutate()); + + /// Advance the boot clock just short of the deadline — still armed. + fake_boot += 29999; + EXPECT_TRUE(store->mayMutate()); + + /// Cross the deadline (ttl elapsed with no renew — a resumed sleeper's view). The fence must expire. + /// (The "a gated mutate then fails closed with ABORTED" leg used `mutateShardForTest` -- the held + /// Phase-E shard lane -- and moves there; here we pin the boot-clock fence flip itself.) + fake_boot += 2; /// now fake_boot = origin + 30001 > origin + 30000 + EXPECT_FALSE(store->mayMutate()); +} + +/// ==== self-remount after GC fence-out (liveness counterpart of the fence-out safety rule) ==== + +namespace +{ + +/// GC's fence-out, applied directly: preserve the body, set gc_fenced, bump seq (token-guarded). +void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + MountLease m = decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, + DB::Cas::PutOutcome::Done); +} + +} + +TEST(CASPoolRemount, FenceOutThenSelfRemountRestoresWrites) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const String mount_key = store->layout().mountKey("test"); + const uint64_t epoch_before = decodeMountLease(backend->get(mount_key)->bytes).writer_epoch; + EXPECT_EQ(store->liveWriterEpoch(), epoch_before); + + fenceOutMount(*backend, mount_key); + + /// The keeper's next renewal fails closed (foreign touch — never re-mint). + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + + /// Self-remount claims a FRESH incarnation: epoch bumped, gc_fenced cleared, writes restored. + ASSERT_TRUE(store->tryRemountOnce()); + const MountLease after = decodeMountLease(backend->get(mount_key)->bytes); + EXPECT_EQ(after.writer_epoch, epoch_before + 1); + EXPECT_FALSE(after.gc_fenced); + EXPECT_EQ(store->liveWriterEpoch(), epoch_before + 1); + + /// The renewal path works again (the new keeper owns the slot). (The follow-on "...and so does a + /// ref-shard mutation" check used `mutateShardForTest` -- the held Phase-E shard lane -- and moves + /// to Phase E's own tests; the self-remount liveness assertion above is the point of this test.) + EXPECT_NO_THROW(store->renewWatermarkOnce()); +} + +TEST(CASPoolRemount, OldEpochBuildFailsClosedAfterRemount) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + auto build = store->beginPartWrite({}); + + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + + /// The build was minted under the superseded incarnation — every further step fails closed. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { build->putBlob(DB::Cas::tests::idOf("x"), DB::Cas::BlobSource::fromString("x")); }); + + /// A FRESH build under the live incarnation works once its publication edge is durable. + const RootNamespace ns{"srv/remount"}; + PartWriteInfo info; + info.intended_ref = ns.string() + "/fresh"; + auto fresh = store->beginPartWrite(info); + const ManifestId id = fresh->stageManifest({blobEntryFor("data.bin", DB::Cas::tests::u128Of("y"))}); + fresh->precommitAdd(ns, "fresh", id); + EXPECT_NO_THROW(fresh->putBlob(DB::Cas::tests::idOf("y"), DB::Cas::BlobSource::fromString("y"))); + fresh->abandon(); +} + +TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const String mount_key = store->layout().mountKey("test"); + + /// A genuinely foreign uuid holds the mount (live or not — foreign is terminal for the claim). + const auto got = backend->get(mount_key); + MountLease foreign = decodeMountLease(got->bytes); + foreign.server_uuid = foreign.server_uuid + DB::UInt128(1); + foreign.seq += 1; + ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, + DB::Cas::PutOutcome::Done); + + EXPECT_FALSE(store->tryRemountOnce()); + /// The foreign body is untouched (no takeover, ever). + EXPECT_EQ(decodeMountLease(backend->get(mount_key)->bytes).server_uuid, foreign.server_uuid); + + /// Move the parent fixture to the production-recognized fenced terminal state before explicitly + /// destroying its superseded keeper. The unfenced foreign-release guard is covered separately below. + fenceOutMount(*backend, mount_key); + store.reset(); + + /// A foreign owner is never taken over — at remount OR at release. This was an `EXPECT_DEATH` + /// pinning a `LOGICAL_ERROR` abort on the release half; the abort fired from `~Pool` and defeated + /// `finishTeardown`'s own catch by aborting at exception construction. The runtime never observed a + /// deposition (the slot was overwritten out of band), so the release takes the + /// exclusivity-violation arm: refuse, leave the foreign occupant untouched, and SURVIVE teardown. + auto foreign_backend = std::make_shared(); + auto invalid_store = DB::Cas::tests::openPoolForTest(foreign_backend); + const String foreign_mount_key = invalid_store->layout().mountKey("test"); + const auto foreign_got = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(foreign_got.has_value()); + MountLease foreign_lease = decodeMountLease(foreign_got->bytes); + foreign_lease.server_uuid = foreign_lease.server_uuid + DB::UInt128(1); + foreign_lease.seq += 1; + ASSERT_EQ( + foreign_backend->putOverwrite(foreign_mount_key, encodeMountLease(foreign_lease), foreign_got->token).outcome, + DB::Cas::PutOutcome::Done); + const auto occupant_before = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(occupant_before.has_value()); + + EXPECT_FALSE(invalid_store->tryRemountOnce()) << "a foreign owner is never taken over at remount"; + + const uint64_t violations_before + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + invalid_store.reset(); /// must not abort, must not terminate + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + violations_before + 1) + << "the release must report the broken single-writer guarantee rather than dying on it"; + const auto occupant_after = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(occupant_after.has_value()) << "nor is it taken over at release"; + EXPECT_EQ(occupant_after->bytes, occupant_before->bytes) + << "the slot must be left byte-for-byte as the foreign owner wrote it"; +} + +TEST(CASPoolRemount, ShutdownGuardRefusesToArmRemount) +{ + auto backend = std::make_shared(); + /// `background_watermark = true` so `scheduleRemount` can latch a recovery generation for the + /// persistent worker in production mode (the same gate both runtime workers check). + auto store = DB::Cas::Pool::open(backend, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); + + /// Teardown has begun: `Pool` latches this before joining either persistent worker. + store->beginShutdownForTest(); + + /// A lease-renewal failure firing during teardown re-enters `scheduleRemount`. With the guard it + /// must refuse to latch another generation after the workers are stopping. + EXPECT_FALSE(store->scheduleRemountForTest()) + << "scheduleRemount must not latch recovery work once teardown has begun"; +} + +namespace +{ +/// A sequenced fake boot clock: the first N `bootMsNow()` calls return the values queued via +/// `.queue`, in order; every call after the queue drains returns `.steady`. `CasMountRuntime::bootMsNow` +/// re-invokes `PoolConfig::boot_ms_fn` on EVERY call, with zero memoization -- so a plain call-counter +/// deterministically distinguishes an early (anchor) reading from a later (response-time) one, with no +/// real sleep and no threads. +struct SequencedBootClock +{ + std::vector queue; + size_t next = 0; + uint64_t steady = 0; + + uint64_t operator()() + { + if (next < queue.size()) + return queue[next++]; + return steady; + } +}; +} + +/// Phase B addendum 2 (task 5b review, reviewer's probe): the self-remount arm must anchor at the +/// claim attempt's pre-I/O instant (`remount_anchor_boot_ms`, captured right after `installKeeper` +/// and right before `keeperStart()` in `Pool::tryRemountOnce`), never at a later reading taken after +/// `keeperStart`/`quiesceRefTablesForRemount` have already run. +/// +/// The two `bootMsNow()` calls of interest, in the ORDER each code version issues them: +/// - FIXED code: call #1 = the new anchor (`remount_anchor_boot_ms`, before `keeperStart`); +/// call #2 = `MountLeaseKeeper::prepareRenew`'s own internal boot read inside `keeperStart`'s +/// `doStart` (feeds only the keeper's OWN internal `confirmed_deadline_ms` -- unrelated to the +/// Pool-level arm -- so its value is irrelevant to the arm post-fix). +/// - PRE-FIX code (no anchor line): call #1 = that SAME `prepareRenew` read (now the first boot +/// call of the attempt, since nothing reads the clock before `keeperStart`); call #2 = the +/// arm-site's own `mount_runtime.bootMsNow()`, read AFTER `keeperStart` returns -- the stale, +/// response-time reading this whole fix exists to stop using. +/// A sequenced clock returning 10000 then 11000 (a later response-time reading that remains inside +/// the normal renewal window) therefore arms the FIXED code from 10000 and the PRE-FIX code from +/// 11000, regardless of which call site reads which value -- letting a single deterministic probe +/// (`mayMutate()` at boot == 10000+ttl) tell +/// them apart with no sleep and no thread. (TDD evidence for both branches is recorded in the task-5 +/// report, not re-asserted here: this test body only encodes the FIXED expectation.) +TEST(CASPoolRemount, RemountArmAnchorsAtClaimAttemptNotResponseTime) +{ + SequencedBootClock clock; + auto backend = std::make_shared(); + auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30'000), + .boot_ms_fn = [&] { return clock(); }, + }); + ASSERT_TRUE(store); + + /// Trip the fence exactly as every other remount test in this file does. + fenceOutMount(*backend, store->layout().mountKey("test")); + + /// Arm the sequence for the upcoming remount attempt: the initial `open` above already drained + /// an unrelated number of `bootMsNow()` calls (all served from `.steady = 0` -- irrelevant, since + /// nothing probes the resulting arm before this point). Reset the counter so the FIRST call from + /// here on is the remount attempt's own call #1. + clock.queue = {10000, 11000}; + clock.next = 0; + + ASSERT_TRUE(store->tryRemountOnce()); + + /// Probe at boot == anchor + ttl (10000 + 30000 = 40000): the fixed code armed from the anchor + /// (10000), so the fence has JUST expired here -- `mayMutate` must be false. (The pre-fix code + /// would still read `mayMutate` as true here, armed from 11000 + 30000 -- see the TDD run in the + /// report.) + clock.steady = 40000; + EXPECT_FALSE(store->mayMutate()) + << "the remount arm must anchor at the claim attempt's pre-I/O instant, not a later " + "response-time reading taken after keeperStart/quiesceRefTablesForRemount"; +} + +/// ==== rev.6 Task 5: clean-release drain gates the farewell marker ==== + +namespace +{ +/// Forces the FIRST `putIfAbsent` whose key contains `fault_key_substr` to throw an ambiguous +/// (Unresolved-classified) exception, `fault_count` times -- the minimal one-shot subset of +/// `RefWriterTestBackend`'s fault injection (gtest_cas_ref_writer.cpp) this file's shutdown test needs +/// to drive a ref-log append into the `Unresolved`/wedge outcome, with `max_attempts = 1` in the budget +/// so the single failed attempt exhausts the retry budget immediately. +class UnresolvedPutBackend final : public DB::Cas::tests::CountingBackend +{ +public: + String fault_key_substr; + int fault_count = 0; + + DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + { + if (fault_count > 0 && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + { + --fault_count; + throw Poco::TimeoutException("UnresolvedPutBackend: simulated ambiguous result (response lost)"); + } + return DB::Cas::tests::CountingBackend::putIfAbsent(key, bytes, meta); + } +}; + +class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend +{ +public: + enum class Fault : uint8_t + { + None, + ThrowBefore, + LandThenThrow, + BlockThenDelegate, + BlockThenThrow, + }; + + using DB::Cas::tests::CountingBackend::putOverwrite; + + Fault fault = Fault::None; + DB::Cas::tests::ManualBarrier * barrier = nullptr; + std::function after_commit; + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + const Fault current = std::exchange(fault, Fault::None); + if (current == Fault::BlockThenDelegate || current == Fault::BlockThenThrow) + { + if (!barrier) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "runtime renewal barrier is absent"); + barrier->arriveAndWait(); + } + if (current == Fault::ThrowBefore || current == Fault::BlockThenThrow) + throw Poco::TimeoutException("injected runtime renewal ambiguity before result"); + + PutResult result = DB::Cas::tests::CountingBackend::putOverwrite(key, bytes, expected, meta); + if (after_commit) + after_commit(); + if (current == Fault::LandThenThrow) + throw Poco::TimeoutException("injected runtime renewal response loss after commit"); + return result; + } +}; + +CasRequestBudget runtimeRenewBudget(uint32_t max_attempts); + +enum class ForeignConflictSinkBehavior : uint8_t +{ + ReenterSameRuntime, + Throw, +}; + +void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behavior) +{ + auto backend = std::make_shared(); + const Layout layout( + behavior == ForeignConflictSinkBehavior::ReenterSameRuntime + ? "runtime-reentrant-foreign-conflict" + : "runtime-throwing-foreign-conflict"); + const String server_root_id = "test"; + const String key = layout.mountKey(server_root_id); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, server_root_id, uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + + std::vector events; + bool reentered = false; + std::optional reentrant_lifecycle; + std::optional reentrant_may_mutate; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink = [&](CasEvent event) + { + const bool foreign_conflict + = event.type == CasEventType::MountConflict && event.outcome == "foreign_writer"; + events.push_back(event); + if (!foreign_conflict) + return; + if (behavior == ForeignConflictSinkBehavior::ReenterSameRuntime) + { + if (!std::exchange(reentered, true)) + { + reentrant_lifecycle = runtime_ptr->lifecycle(); + reentrant_may_mutate = runtime_ptr->mayMutate(); + throw std::runtime_error("injected reentrant mount diagnostic sink failure"); + } + } + else + { + throw std::runtime_error("injected mount diagnostic sink failure"); + } + }; + CasMountRuntime runtime( + backend, + layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .boot_ms_fn = [&] { return boot_ms; }, + }, + server_root_id, + sink, + runtimeRenewBudget(1), + [] { return false; }); + runtime_ptr = &runtime; + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + + auto ours = backend->get(key); + ASSERT_TRUE(ours.has_value()); + MountLease successor = decodeMountLease(ours->bytes); + successor.server_uuid = UInt128{2}; + successor.writer_epoch = 9; + successor.seq += 1; + ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(successor), ours->token).outcome, PutOutcome::Done); + const uint64_t skipped_before + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + const uint64_t violations_before + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + + int failure_code = 0; + String failure_message; + try + { + runtime.renewWatermarkOnce(); + ADD_FAILURE() << "authoritative foreign successor must terminalize renewal"; + } + catch (const DB::Exception & e) + { + failure_code = e.code(); + failure_message = e.message(); + } + + EXPECT_EQ(reentered, behavior == ForeignConflictSinkBehavior::ReenterSameRuntime); + if (behavior == ForeignConflictSinkBehavior::ReenterSameRuntime) + { + ASSERT_TRUE(reentrant_lifecycle.has_value()); + EXPECT_EQ(*reentrant_lifecycle, PoolLifecycle::Live); + ASSERT_TRUE(reentrant_may_mutate.has_value()); + EXPECT_TRUE(*reentrant_may_mutate); + } + else + { + EXPECT_FALSE(reentrant_lifecycle.has_value()); + EXPECT_FALSE(reentrant_may_mutate.has_value()); + } + EXPECT_EQ(failure_code, DB::ErrorCodes::ABORTED) << failure_message; + EXPECT_NE(failure_message.find("held by a foreign server"), String::npos) << failure_message; + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + skipped_before + 1); + const auto failed = std::find_if(events.begin(), events.end(), [](const CasEvent & event) + { + return event.type == CasEventType::WatermarkRenew && event.outcome == "failed"; + }); + EXPECT_NE(failed, events.end()); + if (failed != events.end()) + EXPECT_EQ(failed->detail.at("classification"), "conflict"); + + const auto successor_before_teardown = backend->get(key); + ASSERT_TRUE(successor_before_teardown.has_value()); + const uint64_t heads_before_teardown = backend->headCount(key); + const uint64_t gets_before_teardown = backend->getCount(key); + const uint64_t writes_before_teardown = backend->putOverwriteCount(key); + const uint64_t skipped_before_teardown + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + runtime.finishTeardown(true); + EXPECT_EQ(backend->headCount(key), heads_before_teardown); + EXPECT_EQ(backend->getCount(key), gets_before_teardown); + EXPECT_EQ(backend->putOverwriteCount(key), writes_before_teardown); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + skipped_before_teardown); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), violations_before); + const auto successor_after_teardown = backend->get(key); + ASSERT_TRUE(successor_after_teardown.has_value()); + EXPECT_EQ(successor_after_teardown->bytes, successor_before_teardown->bytes); +} + +TEST(CASPoolRemount, SameRuntimeReentrantForeignConflictSinkCannotReplaceTerminalOutcome) +{ + verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior::ReenterSameRuntime); +} + +TEST(CASPoolRemount, ThrowingForeignConflictSinkCannotReplaceTerminalOutcome) +{ + verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior::Throw); +} + +class RemountStepBackend final : public DB::Cas::tests::CountingBackend +{ +public: + using DB::Cas::tests::CountingBackend::get; + + void failNextGet(String key) + { + failed_key = std::move(key); + } + + std::optional get(const String & key, Range range) override + { + if (!failed_key.empty() && key == failed_key) + { + failed_key.clear(); + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected remount probe failure"); + } + return DB::Cas::tests::CountingBackend::get(key, range); + } + +private: + String failed_key; +}; + +class ScopedRemountLogCapture +{ +public: + ScopedRemountLogCapture() + : logger(getLogger("CasPool")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("information"); + } + + ~ScopedRemountLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +class ScopedParkedRenewalLogCapture +{ +public: + ScopedParkedRenewalLogCapture() + : logger(getLogger("CasMountLeaseKeeper")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("information"); + } + + ~ScopedParkedRenewalLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +size_t countRemountFinalLogs(const String & output) +{ + constexpr std::string_view needle = "CAS whole-chain remount attempt"; + size_t count = 0; + for (size_t pos = 0; (pos = output.find(needle, pos)) != String::npos; pos += needle.size()) + ++count; + return count; +} + +class WorkerExitLatch +{ +public: + void recordExit() + { + std::lock_guard lock(mutex); + ++exits; + cv.notify_all(); + } + + bool waitForAtLeast(uint64_t expected) + { + std::unique_lock lock(mutex); + return cv.wait_for(lock, std::chrono::seconds(20), [&] { return exits >= expected; }); + } + + uint64_t count() const + { + std::lock_guard lock(mutex); + return exits; + } + +private: + mutable std::mutex mutex; + std::condition_variable cv; + uint64_t exits = 0; +}; + +CasRequestBudget runtimeRenewBudget(uint32_t max_attempts = 1) +{ + return CasRequestBudget{ + .attempt_timeout_ms = 10, + .operation_deadline_ms = 500, + .max_attempts = max_attempts, + .lease_safety_margin_ms = 20, + .retry_initial_backoff_ms = 0, + .retry_max_backoff_ms = 0, + }; +} +} + +TEST(CASPoolShutdown, CleanStopDrainsAndWritesFarewell) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + publishPart(store, "srv/clean_stop", "x", "payload"); + + const String mount_key = store->layout().mountKey("test"); + store.reset(); /// drives ~Pool(): with no in-flight ref-log PUT, the drain must succeed. + + const auto got = backend->get(mount_key); + ASSERT_TRUE(got.has_value()); + const MountLease lease = decodeMountLease(got->bytes); + EXPECT_EQ(lease.min_active, std::numeric_limits::max()) + << "a clean drain (no in-flight ref-log PUT) must write the farewell marker"; +} + +TEST(CASPoolShutdown, UnresolvedWedgeSkipsFarewell) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + /// By value: `layout` is used after `store.reset()` below, a reference would dangle. + const Layout layout = store->layout(); + const RootNamespace ns{"srv/wedge_shutdown"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so the fault + /// injected below (computed from that same sentinel) lands on the key production actually writes + /// to -- otherwise the real append mints an unrelated random incarnation and the fault misses. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + publishPart(store, ns.string(), "x", "payload"); + + /// Force the ref-log append the drop below performs into the Unresolved/wedge outcome (as in the + /// wedge tests in gtest_cas_ref_writer.cpp): the single attempt the budget allows fails ambiguously. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + const String mount_key = store->layout().mountKey("test"); + store.reset(); /// drives ~Pool(): the still-wedged lane must skip the farewell marker. + + const auto got = backend->get(mount_key); + ASSERT_TRUE(got.has_value()); + const MountLease lease = decodeMountLease(got->bytes); + EXPECT_NE(lease.min_active, std::numeric_limits::max()) + << "an unresolved ref-log PUT must skip the clean-release farewell marker"; + EXPECT_FALSE(lease.gc_fenced); + + /// A successor claimMount on this body must return LiveDoubleStart (unclean path): no certificate of + /// death (not fenced, not the clean farewell marker, no proven-dead observation) justifies a + /// same-uuid, different-epoch reclaim. + const MountClaimResult claim = claimMount(*backend, layout, "test", lease.server_uuid, + lease.writer_epoch + 1, /*now_ms=*/1, /*ttl_ms=*/30000); + EXPECT_EQ(claim.kind, MountClaimResult::LiveDoubleStart); +} + +/// ==== What a writable mount open may block on ==== +/// +/// Exactly one thing: the token-stability observation window, and only when the predecessor's death +/// has to be OBSERVED rather than certified. The post-reclaim materialization grace (`T_mat`) that +/// used to run beside it is retired -- it existed so a straggler conditional `PUT` from the dying +/// epoch would settle before the successor trusted its recovery LISTINGS, and recovery does not trust +/// listings any more (it walks arithmetically and fences the straggler with an in-band `EpochSeal`). +/// These three tests pin the surviving shape from all three directions: observed-dead, certified-dead, +/// and cleanly departed. + +TEST(CASMountOpenWaits, UncleanOpenPaysOnlyTheObservationWindow) +{ + auto b = std::make_shared(); + Layout l{"p"}; + DB::Cas::tests::seedPoolMetaForRestart(*b); + /// Predecessor: claim epoch 7, no farewell (simulate crash: just drop the keeper) -- a bare + /// `claimMount` plants the lease directly, with no clean-farewell `min_active` marker and no + /// `gc_fenced`, so the successor below has no certificate of death until it observes one itself. + ASSERT_EQ(claimMount(*b, l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, + MountClaimResult::Claimed); + /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs + /// before the mount claim); seed that durable epoch object here too, or the successor's own + /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). + b->putIfAbsent(l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + + /// A 500ms lease TTL is far below the default `cas_request_budget` (RFC + /// cas-s3-timeout-retry-control §required-timeout-model requires attempt_timeout + safety_margin < + /// lease TTL), so scale the budget down to fit -- mirrors `CasMountStartup::StaleSelfMountReclaimedAfterWait`. + const CasRequestBudget tiny_budget{ + .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + + uint64_t fake_boot = 0; + std::vector waits; + PoolPtr store; + ASSERT_NO_THROW( + store = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = tiny_budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + })); + ASSERT_TRUE(store); + + /// The token-stability observation window (>= the 500ms ttl) is paid, because this predecessor's + /// death was never certified -- only observed. + uint64_t total = 0; + for (uint64_t w : waits) + total += w; + EXPECT_GE(total, 500u) << "the observation window must have been paid"; + /// And NOTHING is paid on top of it. Every recorded wait is a poll of that window, bounded by the + /// lease TTL; a wait longer than the whole window can only be a reintroduced grace period. + for (uint64_t w : waits) + EXPECT_LE(w, 500u) + << "an unclean reclaim must not block on any wait beyond the observation poll -- the " + "straggler it used to wait out is fenced by the recovery seal instead"; +} + +TEST(CASMountOpenWaits, CleanOpenSkipsAllWaits) +{ + auto b = std::make_shared(); + /// Predecessor released cleanly (drain + farewell from Task 5): open, then reset() drives ~Pool(), + /// which -- with nothing in flight -- writes the farewell marker (min_active == UINT64_MAX). + auto predecessor = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test"}); + predecessor.reset(); + + std::vector waits; + PoolPtr successor; + ASSERT_NO_THROW( + successor = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .wait_sleep_fn = [&](uint64_t ms) { waits.push_back(ms); }, + })); + ASSERT_TRUE(successor); + + EXPECT_TRUE(waits.empty()) + << "a clean farewell (Task 5) needs no observation window"; +} + +TEST(CASMountOpenWaits, FencedPriorReclaimsWithoutAnyWait) +{ + auto b = std::make_shared(); + Layout l{"p"}; + DB::Cas::tests::seedPoolMetaForRestart(*b); + ASSERT_EQ(claimMount(*b, l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, + MountClaimResult::Claimed); + /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs + /// before the mount claim); seed that durable epoch object here too, or the successor's own + /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). + b->putIfAbsent(l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + /// Predecessor lease carries gc_fenced=true: fence it directly, exactly as `computeHeartbeatFloor`'s + /// fence-out does (preserve the body, gc_fenced = true, seq + 1, token-guarded). + fenceOutMount(*b, l.mountKey("test")); + + /// See UncleanOpenPaysOnlyTheObservationWindow above: a 500ms TTL needs a scaled-down budget too. + const CasRequestBudget tiny_budget{ + .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + + std::vector waits; + PoolPtr store; + ASSERT_NO_THROW( + store = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .cas_request_budget = tiny_budget, + .wait_sleep_fn = [&](uint64_t ms) { waits.push_back(ms); }, + })); + ASSERT_TRUE(store); + + /// A GC-fenced prior is a terminal, already-threshold-gated certificate of death -- reclaimed on the + /// FIRST attempt, with no observation polling. It is also an UNCLEAN prior, which used to mean it + /// paid the materialization grace; nothing is owed now, so this open blocks on nothing at all. + EXPECT_TRUE(waits.empty()) + << "a certified-dead predecessor needs neither the observation window nor any grace period"; +} + +namespace +{ +/// Stalls the CLAIM ITSELF past the lease TTL, and counts what the open writes afterwards. +/// +/// The mount key is written twice before the write fence arms: once by `claimMount`'s reclaim, then +/// once by the keeper's adopt -- and the fence's anchor is taken BETWEEN them. So advancing the +/// injected boot clock on the SECOND write models exactly the thing the Phase B redo exists for: the +/// claim's own I/O outliving the lease it is about to arm a fence under. (This used to be modelled by +/// a materialization grace long enough to consume the TTL; that wait is retired, and the guard it +/// motivated is not -- a stalled socket can still outlive a validated request budget.) +class StalledMountClaimBackend final : public DB::Cas::InMemoryBackend +{ +public: + String mount_key; + std::function on_second_mount_write; + std::atomic mount_writes{0}; + std::atomic mount_writes_after_stall{0}; + + DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, + const DB::Cas::ObjectMeta & m) override + { + if (k == mount_key) + { + const int n = ++mount_writes; + if (n == 2 && on_second_mount_write) + on_second_mount_write(); + else if (n > 2) + ++mount_writes_after_stall; + } + return InMemoryBackend::putOverwrite(k, b, e, m); + } +}; +} + +/// Phase B startup-arm (spec rev.4, codex round-3 finding 2): a claim path that consumed the lease TTL +/// must force ONE fresh conditional lease write before arming — the fence must never arm from an anchor +/// that has already expired (a successor could have legally reclaimed meanwhile). +TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) +{ + auto backend = std::make_shared(); + DB::Cas::Layout layout("pool"); + DB::Cas::tests::seedPoolMetaForRestart(*backend, "pool"); + const String srid = "s"; + const DB::UInt128 uuid(0x42); + backend->mount_key = layout.mountKey(srid); + + /// Seed a FENCED, expired predecessor body under a DIFFERENT epoch (7, matching + /// `FencedPriorPaysOnlyTmat`'s convention). The durable epoch object seeded a few lines below + /// carries `next_writer_epoch = 8`, so THIS pool's own first-allocated `writer_epoch` is 8 -- + /// non-colliding with the seeded epoch-7 prior by construction. With no collision the first + /// (and only) claim attempt reclaims directly with MountPriorState::Fenced, with no silent + /// FencedSelf fence-recovery detour to account for -- so the mount key is written exactly twice + /// before the arm, which is what the stall hook counts on. + { + DB::Cas::MountLease prior; + prior.server_uuid = uuid; + prior.writer_epoch = 7; + prior.seq = 7; + prior.expires_at_ms = 1; /// long expired + prior.gc_fenced = true; + prior.write_attempt_id = DB::UInt128{7}; + backend->putIfAbsent(layout.mountKey(srid), DB::Cas::encodeMountLease(prior)); + } + /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs + /// before the mount claim); seed that durable epoch object here too, or `Pool::open`'s own + /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). + backend->putIfAbsent(layout.epochKey(srid), DB::Cas::encodeServerEpoch(DB::Cas::ServerEpoch{.next_writer_epoch = 8})); + uint64_t fake_boot_ms = 10'000; + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = uuid; + cfg.server_root_id = srid; + cfg.background_watermark = true; + cfg.mount_lease_ttl_ms = std::chrono::milliseconds(30'000); + cfg.boot_ms_fn = [&] { return fake_boot_ms; }; + /// The keeper's adopt write stalls for 15 s of boot clock. That consumes the publication horizon + /// (one 10 s cadence plus one 5 s attempt) while leaving one physical attempt admissible inside + /// the old lease's safety window, so the synchronous redo can safely re-anchor. + backend->on_second_mount_write = [&] { fake_boot_ms += 15'000; }; + + auto store = DB::Cas::Pool::open(backend, cfg); + ASSERT_NE(store, nullptr); + + ASSERT_EQ(backend->mount_writes.load(), 3) + << "the fixture assumes exactly two mount writes before the redo (the reclaim and the keeper's " + "adopt, with the fence anchor between them); a different sequence would make the stall land " + "somewhere else and this test would stop testing the redo"; + EXPECT_EQ(backend->mount_writes_after_stall.load(), 1) + << "a TTL-consuming claim must be followed by exactly ONE fresh conditional lease write " + "(the re-anchoring redo) before the write fence arms"; +} + +/// ==== What a self-remount may block on ==== +/// +/// Nothing an operator configures. The remount used to consult `refLanesSettledForRemount` and pay the +/// materialization grace whenever a ref lane still held an undecided `PUT`; both are retired, because +/// the undecided `PUT` is settled by the protocol rather than waited out — recovery closes the dead +/// epoch with an in-band `EpochSeal` written as a conditional create, and the straggler's own create +/// loses to it. `gtest_cas_retirement_sweep.cpp` proves that conflict directly; these two pin that the +/// wait is gone from both the drained and the still-wedged path. + +TEST(CASRemountWaits, DrainedRemountPaysNoWait) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 1'000'000; + std::vector waits; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + }); + ASSERT_TRUE(store); + EXPECT_TRUE(waits.empty()) << "a fresh mount (no predecessor) pays no wait at open"; + + /// Trip the fence: advance the local boot clock past the deadline (as in `WriteFenceUsesInjectedBootClock` + /// above) and mark the durable lease `gc_fenced` (the certificate `claimMountAwaitingExpiry` reclaims + /// on its FIRST attempt, no observation polling -- avoids a real sleep in this test). + fake_boot += 30001; + fenceOutMount(*backend, store->layout().mountKey("test")); + + /// No in-flight ref-log PUT at all -- the easy direction. + ASSERT_TRUE(store->tryRemountOnce()); + + EXPECT_TRUE(waits.empty()) + << "a drained self-remount must pay no wait"; +} + +TEST(CASRemountWaits, UnresolvedWedgeRemountPaysNoWaitEither) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + uint64_t fake_boot = 1'000'000; + std::vector waits; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + }); + ASSERT_TRUE(store); + EXPECT_TRUE(waits.empty()) << "a fresh mount (no predecessor) pays no wait at open"; + + const Layout & layout = store->layout(); + const RootNamespace ns{"srv/remount_wedge"}; + /// Stage B (Task 4-C): see `CASPoolShutdown.UnresolvedWedgeSkipsFarewell`'s identical comment. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + publishPart(store, ns.string(), "x", "payload"); + + /// Force the ref-log append `dropRef` below performs into the Unresolved/wedge outcome (as in + /// `CASPoolShutdown.UnresolvedWedgeSkipsFarewell`): the single attempt the budget allows fails + /// ambiguously. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + /// Trip the fence exactly as in `DrainedRemountSkipsGrace` above. + fake_boot += 30001; + fenceOutMount(*backend, store->layout().mountKey("test")); + + /// THE HARD DIRECTION, and the one the retired wait existed for: a ref lane that still holds an + /// UNDECIDED conditional PUT when the fence trips. It used to buy a 30 s grace. It buys nothing now + /// -- the remount proceeds straight through, and the undecided PUT is decided by the seal the next + /// recovery writes into its slot. + ASSERT_TRUE(store->tryRemountOnce()); + + EXPECT_TRUE(waits.empty()) + << "an unresolved ref-lane wedge must not make the remount block: the straggler it describes is " + "fenced by the recovery seal, not waited out"; +} + +/// Sealing is decided by ARITHMETIC -- `epoch < live_epoch` -- and by nothing else. This test used to +/// pin the opposite ("a table recovered under a later CLEAN boundary must not seal"), which was the +/// right rule while a seal was a synthetic SNAPSHOT published only to close an unclean handover: such a +/// seal after a clean shutdown was pure parasitic cost, so it was gated on the per-epoch unclean flag. +/// +/// INV-2's seal is not that object. It is the chain link that makes a MISSING epoch detectable across a +/// transition, and a chain that skips every epoch whose mount happened to shut down cleanly is not a +/// chain -- the next sequence-1 transaction would have no `prev_epoch_seal` to name, and no reader could +/// tell "epoch 2 was empty" from "epoch 2's records are gone". So a late-touched table now closes EVERY +/// dead epoch below the live one, however its predecessors died, and this test pins that plus the two +/// things that must still be true: the seals land IN-BAND (at log keys, at the slot a straggler would +/// have taken) and no synthetic seal SNAPSHOT is written anywhere. +TEST(CASRemountWaits, ALateTouchedTableClosesEveryDeadEpochInBandHoweverItsPredecessorsDied) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + uint64_t fake_boot = 1'000'000; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; }, + }); + ASSERT_TRUE(store); + + const Layout & layout = store->layout(); + const RootNamespace ns1{"srv/table_a"}; + const RootNamespace ns2{"srv/table_b"}; + /// Stage B (Task 4-C): `ns1` is pinned because the fault below targets its key by exact sentinel + /// match. `ns2` must ALSO be pinned: the epoch-close assertions further down read its ref-log keys + /// directly at `DB::Cas::tests::fixture::fixtureLife(ns2)`. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns1, store->liveWriterEpoch()); + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns2, store->liveWriterEpoch()); + publishPart(store, ns1.string(), "x", "payload-a"); + /// ns2's epoch-1 data: never touched again by this incarnation until the final check below, well + /// after both remounts -- the "table recovered for the first time, late" the fix must not over-seal. + /// Distinct content from ns1's part: identical payloads collide on the same blob and race + /// `PartWriteTxn::ensureBlobPresent`'s mandatory observation, unrelated to what this test is about. + publishPart(store, ns2.string(), "y", "payload-b"); + + /// Force ns1's ref-log append into the Unresolved/wedge outcome (mirrors + /// `UnresolvedWedgeRemountPaysNoWaitEither` above). + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns1)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns1, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns1)); + + /// Self-remount #1: UNCLEAN (the wedge above). Epoch 1 -> 2. + fake_boot += 30001; + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + ASSERT_EQ(store->liveWriterEpoch(), 2u); + + /// Self-remount #2: CLEAN (no wedge left behind -- `quiesceRefTablesForRemount` already cleared the + /// cache). Epoch 2 -> 3. + fake_boot += 30001; + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + ASSERT_EQ(store->liveWriterEpoch(), 3u); + + using ProfileEvents::global_counters; + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + + /// ns2's FIRST recovery under this incarnation happens now, at epoch 3 -- strictly after both + /// remounts. Its only data is at epoch 1, so epochs 1 and 2 are both dead for it. + EXPECT_EQ(store->listRefs(ns2).size(), 1u); + + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2) + << "both dead epochs must be closed -- the chain link is what a later reader needs to tell an " + "EMPTY epoch from a LOST one, and that is independent of how each mount ended"; + EXPECT_TRUE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{1, 2})).has_value()) + << "epoch 1 closes at the slot right after its last durable id, in-band"; + EXPECT_TRUE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{2, 1})).has_value()) + << "empty epoch 2 closes at its own sequence 1, chained to the epoch-1 seal"; + const RefTxnId retired_sentinel_id{2, std::numeric_limits::max()}; + EXPECT_FALSE(backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns2), retired_sentinel_id)).has_value()) + << "and NO synthetic seal snapshot is written: that shape is retired"; +} + +TEST(CASPool, ReadManifestSharedReturnsSharedDecodeWithoutCopy) +{ + auto backend = std::make_shared(); + const DB::Cas::Layout layout("p"); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + const DB::Cas::RootNamespace ns{"srv/t1"}; + const DB::Cas::ManifestRef ref{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}; + const auto id = DB::Cas::tests::writeManifestRaw(*backend, layout, ns, ref, + {DB::Cas::tests::blobEntryFor("data.bin", DB::UInt128(7))}); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, + {DB::Cas::tests::namespaceBirthOp(), DB::Cas::tests::publishCommittedOps("part_1", ref)[0], + DB::Cas::tests::publishCommittedOps("part_1", ref)[1]}, std::nullopt}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + auto store = DB::Cas::Pool::open(backend, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const auto resolved = store->resolveRef(ns, "part_1"); + ASSERT_TRUE(resolved.has_value()); + + const String manifest_key = layout.manifestKey(id); + backend->resetCounts(); + + auto m1 = store->readManifestShared(resolved->manifest_id); + auto m2 = store->readManifestShared(resolved->manifest_id); + EXPECT_EQ(m1.get(), m2.get()); /// the SAME shared decode, no copy + EXPECT_EQ(backend->getCount(manifest_key), 1u); /// one body GET + EXPECT_EQ(backend->headCount(manifest_key), 2u); /// mandatory HEAD per call (unchanged) + ASSERT_EQ(m1->entries.size(), 1u); + EXPECT_EQ(m1->entries[0].path, "data.bin"); +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +#define EXPECT_RUNTIME_STATE_REJECTION(statement) EXPECT_DEATH({ statement; }, "CAS mount runtime") +#else +#define EXPECT_RUNTIME_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) +#endif + +TEST(CASPoolRemount, DirectRenewCannotRaceWorkerStartOrKeeperReplacement) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-direct"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + + DB::Cas::tests::ManualBarrier barrier; + backend->barrier = &barrier; + backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; + auto direct = std::async(std::launch::async, [&] { runtime.renewWatermarkOnce(); }); + barrier.waitUntilArrived(); + EXPECT_RUNTIME_STATE_REJECTION(runtime.startBackgroundWorkers(std::chrono::milliseconds(10))); + EXPECT_RUNTIME_STATE_REJECTION(runtime.installKeeper(uuid, 2, [&] { return wall_ms; })); + EXPECT_RUNTIME_STATE_REJECTION(runtime.keeperReset()); + barrier.release(); + EXPECT_NO_THROW(direct.get()); + runtime.finishTeardown(true); +} + +TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeParkRequest) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-admission-park"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier admitted; + DB::Cas::tests::ManualBarrier remount; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, + "test", sink, runtimeRenewBudget(), [&] + { + remount.arriveAndWait(); + return false; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + admitted.waitUntilArrived(); + runtime.tripMountLost(); + runtime.scheduleRemount(); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::ParkRequested); + admitted.release(); + remount.waitUntilArrived(); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); + remount.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeStop) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-admission-stop"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier admitted; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + admitted.waitUntilArrived(); + auto stop = std::async(std::launch::async, [&] { runtime.stopBackgroundWorkers(); }); + runtime.waitForRenewalDriverStateForTest(RenewalDriverState::Stopping); + admitted.release(); + EXPECT_NO_THROW(stop.get()); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Dormant); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-direct-after-stop"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.stopBackgroundWorkers(); + EXPECT_RUNTIME_STATE_REJECTION(runtime.renewWatermarkOnce()); + runtime.finishTeardown(true); +} + +#undef EXPECT_RUNTIME_STATE_REJECTION + +TEST(CASPoolRemount, RemountWaitsForRenewalParkedBeforeReplacement) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-park"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 10'000; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier renewal_barrier; + DB::Cas::tests::ManualBarrier remount_barrier; + std::atomic remount_calls{0}; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + remount_barrier.arriveAndWait(); + return false; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + backend->barrier = &renewal_barrier; + backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + renewal_barrier.waitUntilArrived(); + runtime.tripMountLost(); + runtime.scheduleRemount(); + runtime.waitForRenewalDriverStateForTest(RenewalDriverState::ParkRequested); + EXPECT_EQ(remount_calls.load(), 0u) << "replacement callback must wait until renewal has parked"; + renewal_barrier.release(); + remount_barrier.waitUntilArrived(); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); + remount_barrier.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-join"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + std::atomic worker_exits{0}; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + ++worker_exits; + }); + }; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.stopBackgroundWorkers(); + EXPECT_EQ(worker_exits.load(), 2u); + runtime.finishTeardown(true); + EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey("test"))->bytes).min_active, + std::numeric_limits::max()); +} + +TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit) +{ + for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) + { + auto backend = std::make_shared(); + const Layout layout(terminal == PoolLifecycle::IdentityLost + ? "runtime-natural-identity-lost" + : "runtime-natural-vanished"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + WorkerExitLatch exits; + DB::Cas::tests::ManualBarrier transitioned; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [&] + { + if (terminal == PoolLifecycle::IdentityLost) + runtime_ptr->enterIdentityLost(); + else + runtime_ptr->enterVanished(PoolLifecycle::VanishedReplaced, "injected natural replacement"); + transitioned.arriveAndWait(); + return false; + }); + runtime_ptr = &runtime; + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.tripMountLost(); + runtime.scheduleRemount(); + transitioned.waitUntilArrived(); + transitioned.release(); + const bool both_exited_without_stop = exits.waitForAtLeast(2); + runtime.stopBackgroundWorkers(); + EXPECT_TRUE(both_exited_without_stop); + EXPECT_EQ(exits.count(), 2u); + runtime.finishTeardown(false); + } +} + +TEST(CASPoolRemount, ParkedRenewalCannotMissNaturalTerminalPublication) +{ + for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) + { + auto backend = std::make_shared(); + const Layout layout(terminal == PoolLifecycle::IdentityLost + ? "runtime-parked-terminal-identity-lost" + : "runtime-parked-terminal-vanished"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + WorkerExitLatch exits; + std::latch renewal_before_driver_lock{1}; + std::latch release_renewal{1}; + std::once_flag pause_renewal_once; + std::latch parked_predicate_sampled_false{1}; + std::latch release_parked_predicate{1}; + std::latch terminal_pre_lock_reached{1}; + std::latch terminal_post_lock_reached{1}; + std::once_flag release_once; + std::atomic renewal_holds_driver_mutex{false}; + std::atomic terminal_reached_post_lock_while_renewal_held{false}; + const auto release_parked = [&] + { + std::call_once(release_once, [&] { release_parked_predicate.count_down(); }); + }; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .worker_factory = factory, + .remount_parked_hook_for_test = [&] + { + release_renewal.count_down(); + }, + .renewal_before_driver_lock_hook_for_test = [&] + { + std::call_once(pause_renewal_once, [&] + { + renewal_before_driver_lock.count_down(); + release_renewal.wait(); + }); + }, + .renewal_parked_predicate_false_hook_for_test = [&] + { + renewal_holds_driver_mutex.store(true, std::memory_order_release); + parked_predicate_sampled_false.count_down(); + release_parked_predicate.wait(); + renewal_holds_driver_mutex.store(false, std::memory_order_release); + }, + .terminal_publication_waiting_for_driver_lock_hook_for_test = [&] + { + terminal_pre_lock_reached.count_down(); + }, + .terminal_publication_driver_lock_contended_hook_for_test = [&] + { + release_parked(); + }, + .terminal_publication_driver_lock_acquired_hook_for_test = [&] + { + if (renewal_holds_driver_mutex.load(std::memory_order_acquire)) + terminal_reached_post_lock_while_renewal_held.store(true, std::memory_order_release); + release_parked(); + terminal_post_lock_reached.count_down(); + }}, + "test", sink, runtimeRenewBudget(), [&] + { + parked_predicate_sampled_false.wait(); + if (terminal == PoolLifecycle::IdentityLost) + runtime_ptr->enterIdentityLost(); + else + runtime_ptr->enterVanished(PoolLifecycle::VanishedReplaced, "injected parked-wait replacement"); + /// Before the fix, terminal publication does not wait for `driver_mutex`, so it reaches + /// this release only after its notification has raced ahead of the renewal worker's wait. + /// After the fix, the pre-lock hook above releases the waiter before publication blocks. + release_parked(); + return false; + }); + runtime_ptr = &runtime; + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + renewal_before_driver_lock.wait(); + runtime.tripMountLost(); + runtime.scheduleRemount(); + terminal_pre_lock_reached.wait(); + terminal_post_lock_reached.wait(); + const bool violated_serialization + = terminal_reached_post_lock_while_renewal_held.load(std::memory_order_acquire); + const bool both_exited_without_stop = violated_serialization ? false : exits.waitForAtLeast(2); + runtime.stopBackgroundWorkers(); + EXPECT_FALSE(violated_serialization); + EXPECT_TRUE(both_exited_without_stop); + EXPECT_EQ(exits.count(), 2u); + runtime.finishTeardown(false); + } +} + +TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRetryable) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-vanished-reason-preparation"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + WorkerExitLatch exits; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + std::atomic preparation_calls{0}; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .worker_factory = factory, + .vanished_reason_prepare_hook_for_test = [&] + { + if (preparation_calls.fetch_add(1) == 0) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected vanished-reason preparation failure"); + }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.tripMountLost(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + runtime.enterVanished(PoolLifecycle::VanishedReplaced, "must-not-publish"); + }); + EXPECT_FALSE(runtime.vanishedIntentPublished()); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); + EXPECT_TRUE(runtime.vanishedReason().empty()); + + runtime.enterVanished(PoolLifecycle::VanishedReplaced, "retry-completed"); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::VanishedReplaced); + EXPECT_EQ(runtime.vanishedReason(), "retry-completed"); + runtime.enterVanished(PoolLifecycle::VanishedForgotten, "must-remain-ignored"); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::VanishedReplaced); + EXPECT_EQ(runtime.vanishedReason(), "retry-completed"); + const bool both_exited_without_stop = exits.waitForAtLeast(2); + runtime.stopBackgroundWorkers(); + EXPECT_TRUE(both_exited_without_stop); + EXPECT_EQ(exits.count(), 2u); + EXPECT_EQ(preparation_calls.load(), 2u); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, WorkerConstructionRollbackFailsOpenClosed) +{ + for (uint64_t throw_on : {1u, 2u}) + { + auto backend = std::make_shared(); + const Layout layout("runtime-worker-failure-" + std::to_string(throw_on)); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + uint64_t factory_calls = 0; + RuntimeWorkerFactory factory = [&](std::function fn) + { + if (++factory_calls == throw_on) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected runtime worker construction failure"); + return ThreadFromGlobalPool(std::move(fn)); + }; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + EXPECT_THROW(runtime.startBackgroundWorkers(std::chrono::milliseconds(10)), DB::Exception); + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_FALSE(runtime.workersRunningForTest()); + runtime.finishTeardown(false); + } +} + +TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-external-loss"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100'000; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier renewal_barrier; + DB::Cas::tests::ManualBarrier remount_barrier; + std::atomic remount_calls{0}; + std::atomic fresh_epochs{0}; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + ++fresh_epochs; + fenceOutMount(*backend, layout.mountKey("test")); + const MountClaimResult fresh = claimMount(*backend, layout, "test", uuid, 2, wall_ms, 1000); + EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); + if (fresh.kind != MountClaimResult::Claimed) + return false; + runtime.installKeeper(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = runtime.startKeeper(); + runtime.setProcessEpoch(2, std::memory_order_release); + runtime.setLiveWriterEpoch(2); + runtime.armMountFence(uuid, 2, fresh_anchor + 1000); + runtime.noteRemounted(); + remount_barrier.arriveAndWait(); + return true; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + backend->barrier = &renewal_barrier; + backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + renewal_barrier.waitUntilArrived(); + runtime.tripMountLost(); + runtime.scheduleRemount(); + renewal_barrier.release(); + remount_barrier.waitUntilArrived(); + EXPECT_EQ(remount_calls.load(), 1u); + EXPECT_EQ(fresh_epochs.load(), 1u); + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before + 1); + remount_barrier.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, TerminalDepositionDoesNotTouchKeeperAfterReplacement) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-terminal-replacement"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier terminal_deposited; + DB::Cas::tests::ManualBarrier remount; + std::atomic replaced{false}; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_terminal_deposited_hook_for_test = [&] + { + runtime_ptr->keeperReset(); + runtime_ptr->installKeeper(uuid, 2, [&] { return wall_ms; }); + runtime_ptr->keeperReset(); + replaced.store(true, std::memory_order_release); + terminal_deposited.arriveAndWait(); + }}, + "test", sink, runtimeRenewBudget(), [&] + { + remount.arriveAndWait(); + return false; + }); + runtime_ptr = &runtime; + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + terminal_deposited.waitUntilArrived(); + EXPECT_TRUE(replaced.load(std::memory_order_acquire)); + terminal_deposited.release(); + remount.waitUntilArrived(); + remount.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-generations"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier first; + DB::Cas::tests::ManualBarrier second; + std::atomic calls{0}; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + const uint64_t call = ++calls; + (call == 1 ? first : second).arriveAndWait(); + return true; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.tripMountLost(); + runtime.scheduleRemount(); + first.waitUntilArrived(); + runtime.scheduleRemount(); + first.release(); + second.waitUntilArrived(); + EXPECT_EQ(calls.load(), 2u); + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 2u); + second.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-catchup"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 10'000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier first; + DB::Cas::tests::ManualBarrier second; + std::atomic calls{0}; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(10'000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + const uint64_t call = ++calls; + if (call == 1) + { + fenceOutMount(*backend, layout.mountKey("test")); + const MountClaimResult fresh = claimMount(*backend, layout, "test", uuid, 2, wall_ms, 10'000); + EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); + if (fresh.kind != MountClaimResult::Claimed) + return false; + runtime.installKeeper(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 2, fresh_anchor + 10'000); + runtime.noteRemounted(); + boot_ms = 2'000; + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + first.arriveAndWait(); + return true; + } + second.arriveAndWait(); + return false; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 10'000); + runtime.startBackgroundWorkers(std::chrono::milliseconds(1000)); + runtime.tripMountLost(); + runtime.scheduleRemount(); + first.waitUntilArrived(); + first.release(); + second.waitUntilArrived(); + EXPECT_EQ(calls.load(), 2u); + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 2u); + second.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPoolRemount, StaleRemountAnchorPerformsParkedRedo) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 100; + DB::Cas::tests::ManualBarrier committed; + PoolConfig config{ + .pool_prefix = "stale-remount-anchor", + .server_root_id = "test", + .background_watermark = true, + .event_sink = [&](const CasEvent & event) + { + if (event.type == CasEventType::MountRemount && event.outcome == "ok") + committed.arriveAndWait(); + }, + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = runtimeRenewBudget(), + .boot_ms_fn = [&] { return fake_boot; }, + .remount_quiesce_hook_for_test = [&] { fake_boot += 900; }, + }; + auto store = Pool::open(backend, config); + const String key = store->layout().mountKey("test"); + fenceOutMount(*backend, key); + const uint64_t writes_before = backend->putOverwriteCount(key); + ASSERT_TRUE(store->scheduleRemountForTest()); + committed.waitUntilArrived(); + EXPECT_GE(backend->putOverwriteCount(key), writes_before + 3) + << "claim, keeper start, and the stale-anchor parked redo must all write"; + committed.release(); +} + +TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 100; + std::promise result_observed; + std::future result_future = result_observed.get_future(); + std::atomic result_published{false}; + std::mutex events_mutex; + std::vector events; + PoolConfig config{ + .pool_prefix = "parked-redo-recovered-observability", + .server_root_id = "test", + .background_watermark = true, + .event_sink = [&](CasEvent event) + { + const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "ok"; + { + std::lock_guard lock(events_mutex); + events.push_back(std::move(event)); + } + if (final_remount && !result_published.exchange(true)) + result_observed.set_value(); + }, + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = runtimeRenewBudget(2), + .boot_ms_fn = [&] { return fake_boot; }, + .remount_quiesce_hook_for_test = [&] + { + fake_boot += 900; + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + }, + }; + auto store = Pool::open(backend, config); + std::weak_ptr store_lifetime = store; + ScopedParkedRenewalLogCapture renewal_logs; + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->scheduleRemountForTest()); + ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); + + std::vector observed; + { + std::lock_guard lock(events_mutex); + observed = events; + } + const auto retrying = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) + { + return event.type == CasEventType::WatermarkRenew && event.outcome == "retrying"; + }); + const auto recovered = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) + { + return event.type == CasEventType::WatermarkRenew && event.outcome == "recovered"; + }); + const auto remounted = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) + { + return event.type == CasEventType::MountRemount && event.outcome == "ok"; + }); + ASSERT_NE(retrying, observed.end()); + ASSERT_NE(recovered, observed.end()); + ASSERT_NE(remounted, observed.end()); + EXPECT_LT(std::distance(observed.begin(), retrying), std::distance(observed.begin(), recovered)); + EXPECT_LT(std::distance(observed.begin(), recovered), std::distance(observed.begin(), remounted)); + EXPECT_EQ(retrying->detail.at("remount_attempt_no"), remounted->detail.at("attempt_no")); + EXPECT_EQ(recovered->detail.at("remount_attempt_no"), remounted->detail.at("attempt_no")); + EXPECT_EQ(recovered->detail.at("classification"), "committed_after_retry"); + EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' recovered"), String::npos); + + /// `~Pool` stops and joins both persistent runtime workers. Make that quiescence boundary part of + /// the test, before any event/log capture state referenced by those workers can leave scope. + store.reset(); + EXPECT_TRUE(store_lifetime.expired()); +} + +TEST(CASPoolRemount, ParkedRedoFailureObservabilityPrecedesRemountResult) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 100; + std::promise result_observed; + std::future result_future = result_observed.get_future(); + std::atomic result_published{false}; + std::mutex events_mutex; + std::vector events; + PoolConfig config{ + .pool_prefix = "parked-redo-failed-observability", + .server_root_id = "test", + .background_watermark = true, + .event_sink = [&](CasEvent event) + { + const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "failed"; + { + std::lock_guard lock(events_mutex); + events.push_back(std::move(event)); + } + if (final_remount && !result_published.exchange(true)) + result_observed.set_value(); + }, + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = runtimeRenewBudget(1), + .boot_ms_fn = [&] { return fake_boot; }, + .remount_quiesce_hook_for_test = [&] + { + fake_boot += 900; + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + }, + }; + auto store = Pool::open(backend, config); + std::weak_ptr store_lifetime = store; + ScopedParkedRenewalLogCapture renewal_logs; + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->scheduleRemountForTest()); + ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); + + std::vector observed; + { + std::lock_guard lock(events_mutex); + observed = events; + } + const auto failed_renew = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) + { + return event.type == CasEventType::WatermarkRenew && event.outcome == "failed"; + }); + const auto failed_remount = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) + { + return event.type == CasEventType::MountRemount && event.outcome == "failed"; + }); + ASSERT_NE(failed_renew, observed.end()); + ASSERT_NE(failed_remount, observed.end()); + EXPECT_LT(std::distance(observed.begin(), failed_renew), std::distance(observed.begin(), failed_remount)); + EXPECT_EQ(failed_renew->detail.at("remount_attempt_no"), failed_remount->detail.at("attempt_no")); + EXPECT_EQ(failed_renew->detail.at("attempts_sent"), "1"); + EXPECT_EQ(failed_renew->detail.at("classification"), "attempts_exhausted"); + EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' fenced"), String::npos); + + /// A ready final-result future proves publication order; destruction additionally proves the + /// background renewal/remount threads are joined before the fixture's captured state is destroyed. + store.reset(); + EXPECT_TRUE(store_lifetime.expired()); +} + +TEST(CASPoolRemount, ThrowingEventSinkAfterCommitLeavesRuntimeLive) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "throwing-remount-event", .server_root_id = "test", .background_watermark = true}); + DB::Cas::tests::ManualBarrier committed; + store->setEventSink([&](const CasEvent & event) + { + if (event.type == CasEventType::MountRemount && event.outcome == "ok") + { + committed.arriveAndWait(); + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected remount event sink failure"); + } + }); + fenceOutMount(*backend, store->layout().mountKey("test")); + ASSERT_TRUE(store->scheduleRemountForTest()); + committed.waitUntilArrived(); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); + EXPECT_TRUE(store->mayMutate()); + committed.release(); + EXPECT_NO_THROW(store.reset()); +} + +TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) +{ + const auto run = [](bool ambiguous) + { + auto backend = std::make_shared(); + const Layout layout(ambiguous ? "shutdown-ambiguous" : "shutdown-presend"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + const MountClaimResult claim = claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000); + EXPECT_EQ(claim.kind, MountClaimResult::Claimed); + if (claim.kind != MountClaimResult::Claimed) + return uint64_t{0}; + DB::Cas::tests::ManualBarrier barrier; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + if (ambiguous) + { + backend->barrier = &barrier; + backend->fault = RuntimeRenewBackend::Fault::BlockThenThrow; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + barrier.waitUntilArrived(); + auto stop = std::async(std::launch::async, [&] { runtime.stopBackgroundWorkers(); }); + barrier.release(); + stop.get(); + } + else + { + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.stopBackgroundWorkers(); + } + runtime.finishTeardown(true); + return decodeMountLease(backend->get(layout.mountKey("test"))->bytes).min_active; + }; + + EXPECT_EQ(run(false), std::numeric_limits::max()); + EXPECT_NE(run(true), std::numeric_limits::max()); +} + +TEST(CASPool, DirectAndStartupTerminalFailuresRethrowTypedExceptions) +{ + enum class Refusal : uint8_t { PreAttemptDeadline, CancelledAfterSend, FenceLostAfterSend }; + const auto run = [](bool startup, Refusal refusal) + { + auto backend = std::make_shared(); + const Layout layout(startup ? "typed-startup" : "typed-direct"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + std::atomic stop_cause{CasOverwriteStopCause::Continue}; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_stop_cause_for_test = [&] { return stop_cause.load(std::memory_order_acquire); }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + if (refusal == Refusal::PreAttemptDeadline) + boot_ms = 1071; + else + backend->after_commit = [&] + { + stop_cause.store( + refusal == Refusal::CancelledAfterSend + ? CasOverwriteStopCause::Cancelled + : CasOverwriteStopCause::FenceOrLifecycleLost, + std::memory_order_release); + }; + try + { + if (startup) + (void)runtime.renewKeeperForStartupOnce(); + else + runtime.renewWatermarkOnce(); + ADD_FAILURE() << "terminal renewal did not propagate"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR) << e.message(); + } + runtime.finishTeardown(false); + }; + for (bool startup : {true, false}) + for (Refusal refusal : {Refusal::PreAttemptDeadline, Refusal::CancelledAfterSend, Refusal::FenceLostAfterSend}) + run(startup, refusal); +} + +TEST(CASPool, BackgroundCadenceMustFitLeaseBeforeWritablePublication) +{ + auto backend = std::make_shared(); + PoolConfig config{ + .pool_prefix = "invalid-renew-cadence", + .server_root_id = "test", + .background_watermark = true, + .mount_lease_ttl_ms = std::chrono::milliseconds(100), + .mount_renew_period = std::chrono::milliseconds(80), + .cas_request_budget = runtimeRenewBudget(), + }; + EXPECT_THROW((void)Pool::open(backend, config), DB::Exception); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +TEST(CASPool, DecommissionCadenceValidationPrecedesAuthorityWrites) +{ + auto backend = std::make_shared(); + { + auto victim = Pool::open(backend, PoolConfig{.pool_prefix = "invalid-decommission-cadence", .server_root_id = "victim"}); + } + backend->resetCounts(); + PoolConfig config{ + .pool_prefix = "invalid-decommission-cadence", + .server_root_id = "admin", + .mount_lease_ttl_ms = std::chrono::milliseconds(100), + .mount_renew_period = std::chrono::milliseconds(80), + .cas_request_budget = runtimeRenewBudget(), + }; + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + (void)Pool::openForDecommission(backend, config, "victim"); + }); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +TEST(CASPool, DisabledBackgroundDoesNotReserveRenewalCadence) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 100; + PoolConfig config{ + .pool_prefix = "disabled-renew-cadence", + .server_root_id = "test", + .background_watermark = false, + .mount_lease_ttl_ms = std::chrono::milliseconds(100), + .mount_renew_period = std::chrono::hours(24), + .cas_request_budget = runtimeRenewBudget(), + .boot_ms_fn = [&] { return fake_boot; }, + }; + auto store = Pool::open(backend, config); + const String key = store->layout().mountKey("test"); + EXPECT_EQ(backend->putOverwriteCount(key), 1u) + << "a disabled worker cadence must not force a synchronous startup redo"; +} + +TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) +{ + auto backend = std::make_shared(); + const Layout layout("worker-failure"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier remount_entered; + CasEventSink sink; + CasMountRuntime runtime( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + remount_entered.arriveAndWait(); + return false; + }); + runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startKeeper(); + runtime.armMountFence(uuid, 1, anchor + 1000); + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + remount_entered.waitUntilArrived(); + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); + remount_entered.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) +{ + auto backend = std::make_shared(); + uint64_t fake_boot = 100; + PoolConfig config{ + .pool_prefix = "direct-renew", + .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = runtimeRenewBudget(), + .boot_ms_fn = [&] { return fake_boot; }, + }; + auto store = Pool::open(backend, config); + fake_boot = 500; + EXPECT_NO_THROW(store->renewWatermarkOnce()); + fake_boot = 1200; + EXPECT_TRUE(store->mayMutate()) << "direct success must refresh the local fence from attempt start"; + + backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + const uint64_t schedules_before = store->scheduleRemountCallCountForTest(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->renewWatermarkOnce(); }); + EXPECT_FALSE(store->mayMutate()); + EXPECT_EQ(store->scheduleRemountCallCountForTest(), schedules_before + 1); +} + +TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) +{ + auto backend = std::make_shared(); + std::vector events; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "remount-observability", + .server_root_id = "test", + }); + store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + ScopedRemountLogCapture logs; + + store->tripMountLost(); + backend->failNextGet(store->layout().poolMetaKey()); + const uint64_t attempts_before = ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts].load(); + const uint64_t succeeded_before = ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(); + const uint64_t failed_before = ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(); + EXPECT_FALSE(store->tryRemountOnce()); + + fenceOutMount(*backend, store->layout().mountKey("test")); + EXPECT_TRUE(store->tryRemountOnce()); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts].load(), attempts_before + 2); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(), succeeded_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(), failed_before + 1); + + std::vector remounts; + std::copy_if(events.begin(), events.end(), std::back_inserter(remounts), [](const CasEvent & event) + { + return event.type == CasEventType::MountRemount; + }); + ASSERT_EQ(remounts.size(), 2u); + EXPECT_EQ(remounts[0].outcome, "failed"); + EXPECT_EQ(remounts[0].detail.at("step"), "pool_identity_probe"); + EXPECT_EQ(remounts[1].outcome, "ok"); + EXPECT_EQ(remounts[1].detail.at("step"), "publish_live"); + const uint64_t first_attempt = std::stoull(remounts[0].detail.at("attempt_no")); + const uint64_t second_attempt = std::stoull(remounts[1].detail.at("attempt_no")); + EXPECT_EQ(second_attempt, first_attempt + 1); + EXPECT_EQ(countRemountFinalLogs(logs.captured()), 2u) << logs.captured(); +} + +TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "lease-loss-owner", + .server_root_id = "test", + }); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + + store->tripMountLost(); + store->tripMountLost(); + backend->failNextGet(store->layout().poolMetaKey()); + EXPECT_FALSE(store->tryRemountOnce()); + store->beginShutdownForTest(); + store->tripMountLost(); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before + 1); +} + +TEST(CASPoolRemount, LiveForgetDoesNotCountOperationalLeaseLoss) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "forget-is-not-lease-loss", + .server_root_id = "test", + }); + ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + + store->forgetDisk([] {}, "deliberate test decommission"); + + EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) + << "a deliberate terminal decommission is not an operational recovery generation"; +} + +/// Coverage gap (Task 13a): restores the get/exists/remove roundtrip for the mount access-check probe +/// object. The old `CASPool.MountpointObjectRoundTrip` was dropped in the refactor; the wiring test only +/// exercises `putMountpointObject` + `existsFile`, leaving `getMountpointObject`'s value round-trip and +/// `removeMountpointObject` unasserted even though both `Pool` methods remain live. +TEST(CASPool, MountpointObjectRoundTrip) +{ + auto b = std::make_shared(); + auto store = DB::Cas::Pool::open(b, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const String key = "srv1/clickhouse_access_check_abc"; + EXPECT_FALSE(store->getMountpointObject(key).has_value()); + EXPECT_FALSE(store->mountpointObjectExists(key)); + store->putMountpointObject(key, "probe-bytes"); + EXPECT_TRUE(store->mountpointObjectExists(key)); + auto got = store->getMountpointObject(key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(*got, "probe-bytes"); + store->removeMountpointObject(key); + EXPECT_FALSE(store->getMountpointObject(key).has_value()); + EXPECT_FALSE(store->mountpointObjectExists(key)); +} diff --git a/src/Disks/tests/gtest_cas_probe.cpp b/src/Disks/tests/gtest_cas_probe.cpp new file mode 100644 index 000000000000..8c2beb7d4503 --- /dev/null +++ b/src/Disks/tests/gtest_cas_probe.cpp @@ -0,0 +1,360 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} +} + +using namespace DB::Cas; + +TEST(CASProbe, PassesOnEnforcingBackend) +{ + auto b = std::make_shared(); + EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); + EXPECT_TRUE(b->list("p/.cas_probe", "", 10).keys.empty()); // probe cleans up after itself +} + +/// AWS S3 answers 400 InvalidArgument to a conditional DELETE with an EMPTY If-Match, and the +/// probe's exit cleanup used to issue exactly that (deleteExact with the absent HeadResult's empty +/// token) after step 8 had already deleted the probe keys — two scary AWSClient log lines +/// on every real-S3 mount. The cleanup must HEAD-gate the delete instead of firing blindly. +class EmptyTokenDeleteRecorder : public InMemoryBackend +{ +public: + size_t empty_token_deletes = 0; + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (token.empty()) + ++empty_token_deletes; + return InMemoryBackend::deleteExact(key, token); + } +}; + +TEST(CASProbe, CleanupNeverDeletesWithEmptyToken) +{ + auto b = std::make_shared(); + EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); + EXPECT_EQ(b->empty_token_deletes, 0u); +} + +TEST(CASProbe, FailsClosedOnNonEnforcingDelete) +{ + auto b = std::make_shared(); + b->setEnforceTokens(false); // the MinIO-OSS failure mode + EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); +} + +TEST(CASProbe, FailsClosedOnDeleteMarkers) +{ + auto b = std::make_shared(); + b->setSimulateDeleteMarkers(true); // versioning enabled on the prefix + EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); +} + +TEST(CASProbe, PassesOnEmulatedLocal) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); +} + +/// B135: two servers mounting the SAME shared CA pool concurrently must not race on the probe keys. +/// We simulate "a concurrent mounter's probe is in flight" by PRE-SEEDING the fixed-name probe key +/// `/_probe/token` over a shared backend, then opening the Pool. With the OLD fixed-key probe +/// the open's `putIfAbsent("/_probe/token", …)` returns PreconditionFailed and `Pool::open` +/// throws NOT_IMPLEMENTED ("putIfAbsent on a fresh key returned PreconditionFailed"). With the +/// per-mount unique probe prefix `/_probe//token`, the seeded key does not collide and +/// the open succeeds — exactly the concurrent-shared-pool-mount behaviour we need. +TEST(CASProbe, ConcurrentMountsDoNotCollide) +{ + auto b = std::make_shared(); + + /// Simulate a concurrent mounter whose probe object under the legacy fixed key is still present. + ASSERT_EQ(b->putIfAbsent("p/_probe/token", "concurrent-mounter-in-flight").outcome, PutOutcome::Done); + + /// A real (second) mount over the same shared pool must still succeed — its probe runs under a + /// fresh per-mount-unique prefix and never touches the seeded fixed key. + EXPECT_NO_THROW(Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + + /// And two genuinely-concurrent mounts (distinct unique prefixes) both succeed over one backend. + EXPECT_NO_THROW(Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + + /// The seeded fixed-key artifact is untouched (the probe never collided with it). + EXPECT_TRUE(b->get("p/_probe/token").has_value()); +} + +/// The probe must consult the backend's store-preconditions hook BEFORE the op battery: a +/// generation-dialect store on a VERSIONED bucket passes every conditional-op check, but its +/// token-exact DELETEs archive noncurrent generations instead of reclaiming storage — only the +/// hook can see that, so a throwing hook must fail the probe closed. +class PreconditionRefusingBackend : public InMemoryBackend +{ +public: + void checkPoolPreconditions() override + { + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "test: store precondition violated (e.g. bucket versioning enabled)"); + } +}; + +TEST(CASProbe, FailsClosedOnPoolPreconditions) +{ + auto b = std::make_shared(); + EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); + /// The hook fires FIRST: no probe keys may have been written. + EXPECT_TRUE(b->list("p/.cas_probe", "", 10).keys.empty()); +} + +/// `Pool::open` wraps the pool backend in `InstrumentedBackend` BEFORE calling `runCapabilityProbe` +/// (see CasPool.cpp), so the hook must actually fire THROUGH the wrapper on the real mount path — +/// not just on a raw backend, which `FailsClosedOnPoolPreconditions` above already covers. +TEST(CASProbe, PoolPreconditionsFireThroughInstrumentedWrapper) +{ + auto inner = std::make_shared(); + InstrumentedBackend wrapped(inner); + EXPECT_THROW(runCapabilityProbe(wrapped, "p/.cas_probe"), DB::Exception); + /// The hook fires FIRST: no probe keys may have been written to the inner backend. + EXPECT_TRUE(inner->list("p/.cas_probe", "", 10).keys.empty()); +} + +/// RFC cas-s3-timeout-retry-control: a Native-mode mount over an object storage that does not support +/// the SingleAttempt retry profile must never silently proceed under the disk's default (~500-attempt) +/// transparent retry policy — see Backend::checkConditionalWriteSingleAttemptSupport. +/// LocalObjectStorage never supports the profile (IObjectStorage::supportsRetryProfile's default +/// implementation only answers true for Default), so Native mode over it is exactly the case this must +/// refuse. EmulatedSingleProcess is exempt: it never claims single-attempt S3 semantics in the first +/// place (PassesOnEmulatedLocal above). +TEST(CASProbe, FailsClosedOnUnsupportedSingleAttemptProfile) +{ + auto native = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + EXPECT_THROW(native->checkConditionalWriteSingleAttemptSupport(), DB::Exception); + + auto emulated = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + EXPECT_NO_THROW(emulated->checkConditionalWriteSingleAttemptSupport()); +} + +/// The same fail-closed refusal through the actual capability probe (Step 0b) — the real gate a +/// writable Pool::open goes through, not just the hook in isolation above. +TEST(CASProbe, MissingSingleAttemptClientFailsCapabilityProbe) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + /// Native mode passes the key to the object storage verbatim, so the probe prefix must be anchored + /// under this storage's own root: a bare prefix lands beside the test process, where an object left + /// by another run answers the LIST below and an unrooted LIST answers "no keys" for free. + const String probe_prefix = DB::Cas::tests::nativeKeyUnder(storage, "p/.cas_probe"); + + auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + EXPECT_THROW(runCapabilityProbe(*b, probe_prefix), DB::Exception); + /// The hook fires before the op battery: no probe keys may have been written. + EXPECT_TRUE(b->list(probe_prefix, "", 10).keys.empty()); + + /// The same LIST can see a key that IS under the prefix — otherwise the emptiness above would be + /// indistinguishable from a prefix this backend can never enumerate. + ASSERT_EQ(b->putIfAbsent(probe_prefix + "/token", "probe-v1").outcome, PutOutcome::Done); + EXPECT_FALSE(b->list(probe_prefix, "", 10).keys.empty()); +} + +/// Mirrors PoolPreconditionsFireThroughInstrumentedWrapper: the real mount path wraps the backend in +/// InstrumentedBackend BEFORE calling runCapabilityProbe, so this check must fire through it too. +TEST(CASProbe, MissingSingleAttemptClientFiresThroughInstrumentedWrapper) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto inner = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + InstrumentedBackend wrapped(inner); + EXPECT_THROW(runCapabilityProbe(wrapped, DB::Cas::tests::nativeKeyUnder(storage, "p/.cas_probe")), DB::Exception); +} + +namespace +{ + +/// Honors every conditional WRITE but ignores the token on a token-exact DELETE. This is what a GCS +/// delete degenerates to when its numeric generation leaves as a raw `If-Match` — no +/// `x-goog-if-generation-match` — and the service ignores the header it does not recognise. +class IgnoresDeleteTokenBackend : public InMemoryBackend +{ +public: + DeleteOutcome deleteExact(const String & key, const Token &) override + { + return InMemoryBackend::deleteExact(key, head(key).token); + } +}; + +/// The other half of that degeneracy: the service refuses the unrecognised header outright, so even +/// the correct token never removes anything. +class RejectsDeleteTokenBackend : public InMemoryBackend +{ +public: + DeleteOutcome deleteExact(const String &, const Token &) override + { + DeleteOutcome d; + d.kind = DeleteOutcome::Kind::TokenMismatch; + return d; + } +}; + +} + +/// A GCS mount whose exact deletes lost their generation semantics can fail in either direction, and +/// the probe's delete battery must reject the mount both times. Both backends enforce every +/// conditional write, so every step before the battery passes and only step 6's wrong-token +/// preservation check and step 8's correct-token deletion check can be what fires — +/// `PassesOnEnforcingBackend` above is the control showing the same probe succeeds when only +/// `deleteExact` is left alone. +/// +/// This is about the battery, not about the marking: that the `NativeConditional` mode actually +/// reaches the production request object is proven where the request is built, not here. +TEST(CASProbe, ExactDeleteBatteryDetectsMissingGenerationMode) +{ + IgnoresDeleteTokenBackend ignores; + EXPECT_THROW(runCapabilityProbe(ignores, "p/.cas_probe"), DB::Exception); + + RejectsDeleteTokenBackend rejects; + EXPECT_THROW(runCapabilityProbe(rejects, "p/.cas_probe"), DB::Exception); +} + +namespace +{ + +/// Models the exact shape of the trust-flip this suite must catch a regression of +/// (codex-review-triage §3.18, Critical): like the production `ObjectStorageBackend` in Native mode, +/// this backend mints and expects tokens under a dialect (`TokenType::ETag`) OTHER than +/// `TokenType::Emulated`, and rejects a foreign-dialect `expected`/`token` argument LOCALLY -- +/// before the value it carries ever reaches the real conditional-compare beneath the gate (`inner`, +/// a genuinely enforcing `InMemoryBackend`, standing in for "the wire"). Every gated method counts +/// how many times it actually delegated to `inner`, so a test can tell "rejected by the dialect +/// gate" apart from "rejected by the real enforcement" -- the exact distinction `Cas::Probe` exists +/// to prove, and the one the №19 hardening risked collapsing (see CasProbe.cpp step 3/5c/6). +class DialectGatedCountingBackend final : public Backend +{ +public: + std::optional get(const String & key, Range range) override { return inner.get(key, range); } + + std::optional getStream(const String & key, Range range) override { return inner.getStream(key, range); } + + HeadResult head(const String & key) override + { + HeadResult r = inner.head(key); + if (r.exists) + r.token.type = TokenType::ETag; + return r; + } + + bool supportsListTokens() const override { return inner.supportsListTokens(); } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + /// No `expected` token to gate -- matches production (ObjectStorageBackend::putIfAbsent has + /// no dialect check either). + PutResult r = inner.putIfAbsent(key, bytes, meta); + if (r.outcome == PutOutcome::Done) + r.token.type = TokenType::ETag; + return r; + } + + void publishBlob(const BlobPublishRequest & request) override + { + inner.publishBlob(request); + } + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + if (expected.type != TokenType::ETag) + return {PutOutcome::PreconditionFailed, {}}; /// dialect-gated: never reaches `inner` + ++overwrite_reached; + PutResult r = inner.putOverwrite(key, bytes, Token{expected.value, TokenType::Emulated}, meta); + if (r.outcome == PutOutcome::Done) + r.token.type = TokenType::ETag; + return r; + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override + { + if (expected.has_value() && expected->type != TokenType::ETag) + return {CasOutcome::Conflict, {}}; /// dialect-gated: never reaches `inner` + ++casput_reached; + std::optional retyped; + if (expected.has_value()) + retyped = Token{expected->value, TokenType::Emulated}; + CasResult r = inner.casPut(key, bytes, retyped, meta); + if (r.outcome == CasOutcome::Committed) + r.token.type = TokenType::ETag; + return r; + } + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + if (token.type != TokenType::ETag) + { + DeleteOutcome d; + d.kind = DeleteOutcome::Kind::TokenMismatch; /// dialect-gated: never reaches `inner` + return d; + } + ++delete_reached; + return inner.deleteExact(key, Token{token.value, TokenType::Emulated}); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage p = inner.list(prefix, cursor, limit); + for (auto & k : p.keys) + if (k.token) + k.token->type = TokenType::ETag; + return p; + } + + /// Number of times putOverwrite/casPut(with expected)/deleteExact actually delegated to `inner` + /// (i.e. reached the real enforcement) rather than being short-circuited by the dialect gate. + int overwrite_reached = 0; + int casput_reached = 0; + int delete_reached = 0; + +private: + InMemoryBackend inner; +}; + +} + +/// codex-review-triage §3.18, Critical: `runCapabilityProbe`'s three wrong-token sites (step 3 +/// putOverwrite, step 5c casPut, step 6 deleteExact) must send a token in the LIVE dialect this +/// backend mints (t1.type / ct1.type / t2.type), not a hardcoded `TokenType::Emulated`. A backend +/// whose native dialect differs from Emulated -- exactly what `ObjectStorageBackend` mints in Native +/// mode -- would otherwise reject the old hardcoded tokens LOCALLY via a dialect gate, never +/// exercising the real conditional enforcement those three steps exist to validate; the probe would +/// still report success (the outcome enums match either way), so a regression here is invisible +/// unless something counts whether the real enforcement was ever reached. `DialectGatedCountingBackend` +/// enforces real (correct) conditional semantics AND gates on dialect exactly like the production +/// risk, so `runCapabilityProbe` runs to completion (unlike a real Native-mode ObjectStorageBackend +/// over LocalObjectStorage, which cannot even reach this point -- see +/// MissingSingleAttemptClientFailsCapabilityProbe and the fact that LocalObjectStorage does not honor +/// WriteSettings conditions at all); the exact reached-counts below pin down that every wrong-token +/// site got past the gate: a probe that regressed to the hardcoded-Emulated construction would still +/// pass (no throw) but under-count here by exactly one at each of the three sites, since the dialect +/// gate would swallow that one call before `inner` ever saw it. +TEST(CASProbe, WrongTokenAttemptsReachTheBackendPastTheDialectGate) +{ + DialectGatedCountingBackend b; + EXPECT_NO_THROW(runCapabilityProbe(b, "p/.cas_probe")); + + /// putOverwrite: step 3 (wrong token) + step 4 (correct token) -- both live-dialect, both gated + /// through to `inner`. + EXPECT_EQ(b.overwrite_reached, 2); + /// casPut: 5a (create), 5b (conflict-on-exists, no expected token to gate), 5c (wrong token, + /// live-dialect), 5d (correct token) -- all four reach `inner`. + EXPECT_EQ(b.casput_reached, 4); + /// deleteExact: step 6 (wrong token, live-dialect) + step 8 (correct token) + step 9 cleanup + /// (correct token for cas_key) -- all three reach `inner`. + EXPECT_EQ(b.delete_reached, 3); +} diff --git a/src/Disks/tests/gtest_cas_promote_republish.cpp b/src/Disks/tests/gtest_cas_promote_republish.cpp new file mode 100644 index 000000000000..8c6b504535f4 --- /dev/null +++ b/src/Disks/tests/gtest_cas_promote_republish.cpp @@ -0,0 +1,402 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// RED/characterization tests for the promote-over-committed leak fix: +/// BUG 1a (PROMOTE-OVER-COMMITTED-LEAK): `PartWriteTxn::promote` silently overwrites `refs[final_ref_name]` +/// when it already names a DIFFERENT committed manifest, orphaning the old manifest (leak). The fix +/// (Task 2) makes this throw `ABORTED` instead. +/// BUG 1c: `republishRef`'s only idempotency gate is "source absent" -- a re-drive after a crash +/// between `promote(dst)` and `dropRef(src)` finds dst ALREADY committed with the (same) content it is +/// about to re-publish, but re-stages+re-promotes anyway, minting a fresh manifest and orphaning the +/// first attempt's manifest. The fix (Task 3) makes the re-drive idempotent (content-keyed, not +/// ManifestId-keyed) when dst matches, and fail-closed (`ABORTED`) when dst holds different content. +/// +/// These tests are EXPECTED TO FAIL pre-fix -- that failure IS the bug reproducing. They must not be +/// weakened to pass; Tasks 2/3 make them pass. + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +extern const int NETWORK_ERROR; +extern const int LOGICAL_ERROR; +} + +using namespace DB::Cas; + +namespace +{ + +PoolPtr openPool(const std::shared_ptr & b) +{ + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// One inline-entry manifest naming `path` with content `bytes` (distinct bytes => distinct content). +/// EntryPlacement::Inline means `promote`'s blob-leaf revalidation skips it entirely -- no real blob +/// objects are needed for these tests. +std::vector inlineEntries(const String & path, const String & bytes) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Inline; + e.inline_bytes = bytes; + return {e}; +} + +/// The full write flow for an INLINE-only manifest: stageManifest -> precommitAdd -> promote. Returns +/// the committed ManifestId. +ManifestId publishCommitted(const PoolPtr & s, const RootNamespace & ns, const String & ref, + const std::vector & entries) +{ + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = build->stageManifest(entries); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// The ContentAddressedTransaction fixture (mirrors gtest_ca_transaction.cpp's openTxStorage / +/// writeFileTx): a real disk-layer storage + transaction, used to drive `republishRef` through its +/// ONLY caller (`ContentAddressedTransaction::moveDirectory`'s committed-source-ref-move branch), +/// since `republishRef` itself is private. +std::shared_ptr openTxStorage() +{ + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_tx_promote_republish_scratch"); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +void writeFileTx(DB::IMetadataTransaction & tx, const std::string & path, const std::string & bytes) +{ + auto & ca_tx = dynamic_cast(tx); + auto buf = ca_tx.writeFile(path, 65536, DB::WriteMode::Rewrite, {}); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); +} + +} + +/// BUG 1a: promoting a DIFFERENT manifest onto an already-committed ref must fail closed (ABORTED), +/// not silently overwrite (which orphans the old manifest, PROMOTE-OVER-COMMITTED-LEAK). +/// PRE-FIX: promote() does not throw -- this test FAILS (RED), which IS the leak reproducing. +TEST(CASPromoteRepublish, PromoteOverDifferentCommittedRefFailsClosed) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl@cas@"}; + const String ref = "all_0_0_0"; + + publishCommitted(s, ns, ref, inlineEntries("f", "AAA")); // committed T_old + + auto build2 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id2 = build2->stageManifest(inlineEntries("f", "BBB")); // DIFFERENT content + build2->precommitAdd(ns, ref, id2); + + try + { + build2->promote(ns, ref, build2->buildId(), id2); + FAIL() << "PRE-FIX: promote silently overwrote a committed ref (PROMOTE-OVER-COMMITTED-LEAK); " + "POST-FIX must throw a CAS write-retry-later NETWORK_ERROR"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + } +} + +/// Re-promoting the SAME manifest_ref onto its own committed ref must NOT throw (idempotent +/// re-promote): the fix's guard keys on a DIFFERENT manifest_ref, not merely "ref already committed". +/// This is expected to pass BOTH pre- and post-fix (it is not part of the bug). +TEST(CASPromoteRepublish, PromoteSameManifestIsIdempotent) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl@cas@"}; + const String ref = "all_0_0_0"; + const ManifestId id = publishCommitted(s, ns, ref, inlineEntries("f", "AAA")); + + /// Re-precommit + re-promote the SAME id onto the same ref: allowed (same manifest_ref). + auto build2 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + build2->precommitAdd(ns, ref, id); + EXPECT_NO_THROW(build2->promote(ns, ref, build2->buildId(), id)); +} + +/// Sanity companion to BUG 1a: promote over an ABSENT ref (the normal insert path) must succeed +/// unconditionally -- the fail-close guard must only fire for an EXISTING different committed ref. +TEST(CASPromoteRepublish, PromoteOverAbsentRefSucceeds) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl@cas@"}; + const String ref = "all_0_0_0"; + + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = build->stageManifest(inlineEntries("f", "AAA")); + build->precommitAdd(ns, ref, id); + EXPECT_NO_THROW(build->promote(ns, ref, build->buildId(), id)); +} + +/// BUG 1c: a `republishRef` re-drive where the destination is ALREADY committed with the SAME content +/// (the crash-before-`dropRef(src)` state) must be idempotent: skip the re-stage/re-promote, drop src, +/// and leave dst's manifest UNCHANGED (no fresh manifest minted for identical content). +/// +/// PRE-FIX: republishRef's only idempotency gate is "source absent" -- it re-stages+re-promotes +/// unconditionally, minting a FRESH manifest id at dst even though the content is identical, orphaning +/// the first attempt's manifest. This test asserts dst's ManifestId is UNCHANGED across the re-drive -- +/// PRE-FIX this FAILS (RED: the id changes, proving the orphaning leak). +TEST(CASPromoteRepublish, RepublishReDriveOverCommittedDstIsIdempotent) +{ + auto storage = openTxStorage(); + const auto ns = storage->liveNamespace("b09b09b0-0909-4909-8909-090909090909"); + const String src_ref = "all_1_1_0"; + const String dst_ref = "detached_all_1_1_0"; + const String src_path = "b09/b09b09b0-0909-4909-8909-090909090909/" + src_ref; + const String dst_path = "b09/b09b09b0-0909-4909-8909-090909090909/" + dst_ref; + + /// 1. Publish a committed src part via the normal write flow (tmp -> final rename, B151 + /// publish-at-rename), exactly as gtest_ca_transaction.cpp's fixtures do. + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_" + src_ref + "/data.bin", "payload-A"); + tx->moveDirectory("b09/b09b09b0-0909-4909-8909-090909090909/tmp_insert_" + src_ref, src_path); + tx->commit(DB::NoCommitOptions{}); + } + ASSERT_TRUE(storage->store()->resolveRef(ns, src_ref).has_value()); + + /// 2. Construct the "crash-before-dropRef(src)" state of a PRIOR republishRef drive by replaying + /// its exact body (resolve src -> adoptEvidence every entry -> stageManifest(same entries) -> + /// precommitAdd -> promote) WITHOUT the trailing dropRef(src). This leaves BOTH src and dst + /// committed, with dst holding the SAME content as src -- precisely the state a re-driven + /// republishRef must handle idempotently (ContentAddressedTransaction.cpp:143-169). + const auto resolved_src = storage->store()->resolveRef(ns, src_ref); + ASSERT_TRUE(resolved_src.has_value()); + const PartManifest src_manifest = storage->store()->readManifest(resolved_src->manifest_id); + { + auto build = storage->store()->beginPartWrite( + PartWriteInfo{.intended_ref = ns.string() + "/" + dst_ref, .intended_namespace = ns}); + for (const auto & entry : src_manifest.entries) + build->adoptEvidence(entry); + const ManifestId id = build->stageManifest(src_manifest.entries); + build->precommitAdd(ns, dst_ref, id); + build->promote(ns, dst_ref, build->buildId(), id); + /// Deliberately NO dropRef(ns, src_ref) here -- this is the simulated crash. + } + ASSERT_TRUE(storage->store()->resolveRef(ns, src_ref).has_value()) + << "src must still be committed (the simulated crash happened before dropRef)"; + const auto resolved_dst_before = storage->store()->resolveRef(ns, dst_ref); + ASSERT_TRUE(resolved_dst_before.has_value()); + const ManifestId dst_id_before = resolved_dst_before->manifest_id; + + /// 3. RE-DRIVE the same rename through the real transaction path: both endpoints are already + /// committed-ref part paths (not a table-level rename, no staged source in this fresh + /// transaction) -- moveDirectory's "move any COMMITTED source ref" branch calls + /// republishRef(src, dst) for real (the only way to reach the private method). + { + auto tx = storage->createTransaction(); + tx->moveDirectory(src_path, dst_path); + tx->commit(DB::NoCommitOptions{}); + } + + /// 4. Idempotency: src dropped, dst unchanged (SAME ManifestId -- no second manifest minted for + /// identical content, so nothing orphaned). + EXPECT_FALSE(storage->store()->resolveRef(ns, src_ref).has_value()) + << "src ref must be dropped by the re-drive"; + const auto resolved_dst_after = storage->store()->resolveRef(ns, dst_ref); + ASSERT_TRUE(resolved_dst_after.has_value()); + EXPECT_EQ(resolved_dst_after->manifest_id, dst_id_before) + << "PRE-FIX: republishRef re-drive mints a FRESH manifest for identical content, orphaning the " + "first attempt's manifest (BUG 1c leak). POST-FIX: idempotent no-op, same manifest."; + EXPECT_EQ(storage->getFileSize(dst_path + "/data.bin"), 9u); +} + +/// REMOVED (all-tree-part-files Task 9): +/// `RepublishReDriveResyncsDriftedMutableFiles` proved that `republishRef`'s idempotent-skip path +/// re-synced dst's `mutable_files` from src's CURRENT resolve when src's mutable payload drifted +/// between the crashed attempt and the re-drive. That side channel is gone -- `metadata_version.txt` +/// etc. are ordinary manifest entries now, so a src drift of that kind changes `entries`, and +/// `republishRef`'s idempotency check (`dst_manifest->entries != src_manifest->entries`) now correctly +/// treats it as a genuine content conflict (ABORTED) rather than silently resyncing a side payload -- +/// there is no longer a "same content, drifted sidecar" state to re-sync. `RepublishReDriveOver- +/// CommittedDstIsIdempotent` above remains the live coverage for the idempotent-skip path itself. + +/// Companion conflict case: a re-drive where dst is committed to DIFFERENT content than src is a +/// genuine conflict (an ATTACH-onto-existing-name collision), not a re-drive -- it must fail closed +/// (ABORTED), never silently drop src (which would lose src's content) nor silently overwrite dst. +/// This scenario reaches the SAME `promote`-over-different-committed-ref guard as BUG 1a, so pre-fix it +/// behaves the same way BUG 1a does: no throw (silent overwrite), which is also a leak/data-loss risk. +TEST(CASPromoteRepublish, RepublishReDriveOverDifferentContentDstFailsClosed) +{ + auto storage = openTxStorage(); + const auto ns = storage->liveNamespace("b0ab0ab0-0a0a-4a0a-8a0a-0a0a0a0a0a0a"); + const String src_ref = "all_2_2_0"; + const String dst_ref = "detached_all_2_2_0"; + const String src_path = "b0a/b0ab0ab0-0a0a-4a0a-8a0a-0a0a0a0a0a0a/" + src_ref; + const String dst_path = "b0a/b0ab0ab0-0a0a-4a0a-8a0a-0a0a0a0a0a0a/" + dst_ref; + + /// src committed with content "payload-SRC". + { + auto tx = storage->createTransaction(); + writeFileTx(*tx, "b0a/b0ab0ab0-0a0a-4a0a-8a0a-0a0a0a0a0a0a/tmp_insert_" + src_ref + "/data.bin", "payload-SRC"); + tx->moveDirectory("b0a/b0ab0ab0-0a0a-4a0a-8a0a-0a0a0a0a0a0a/tmp_insert_" + src_ref, src_path); + tx->commit(DB::NoCommitOptions{}); + } + /// dst ALREADY committed with genuinely DIFFERENT content (not a re-drive artifact -- a real + /// name collision), via a completely independent build. + publishCommitted(storage->store(), ns, dst_ref, inlineEntries("data.bin", "different-content")); + ASSERT_TRUE(storage->store()->resolveRef(ns, src_ref).has_value()); + ASSERT_TRUE(storage->store()->resolveRef(ns, dst_ref).has_value()); + + try + { + auto tx = storage->createTransaction(); + tx->moveDirectory(src_path, dst_path); + tx->commit(DB::NoCommitOptions{}); + FAIL() << "PRE-FIX: republishRef silently overwrote dst's different content " + "(promote-over-committed leak); POST-FIX must throw ABORTED and leave src intact"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED); + } +} + +/// BUG 2: `abandon` must emit its precommit removal (an exact `owner_transition`) BEFORE retiring the +/// build_seq, so the build stays active until that removal is durable and no freshness-window consumer +/// judges the manifest build-dead while an un-removed precommit still names it (GC no longer reclaims +/// abandoned precommits — the writer removes them itself). +/// +/// A3 mint-tightening INVERTS this +/// test's original tail assertion. Before A3, this test's black-box PROOF that `abandon()` had really +/// removed the exact precommit binding was that a FRESH `precommitAdd` for the SAME (ref_name, +/// manifest_ref) succeeded -- a still-live binding would instead throw CORRUPTED_DATA ("add precommit +/// ... already exists"). That proof mechanism no longer works: `rebuild` is a DIFFERENT `PartWriteTxn` +/// from `build` and never staged `id` itself (`build` did), so `precommitAdd` now refuses it +/// UNCONDITIONALLY under A3 -- regardless of whether abandon's removal ever landed. Re-owning a +/// dropped identity from a transaction that did not mint it would let a later relink confirm's exact +/// `ManifestRef` equality (Part B of the same design) compare true against a token whose blobs may +/// already be reclaimed -- an ABA the whole publish-confirm design depends on being structurally +/// impossible. The removal-before-retire property this test used to prove is unaffected by A3 and +/// stays covered by the TLA+ `WAbandonPrecommit` model; `PrecommitAddRejectsAnIdThisTxnDidNotStage` +/// below is the dedicated A3 regression pin. +/// +/// `rebuild->precommitAdd(ns, ref, id)` below throws `LOGICAL_ERROR`, which aborts the whole process in +/// debug/sanitizer builds instead of behaving like a catchable exception -- `CASPromoteRepublishDeathTest. +/// AbandonEmitsRemovalBeforeRetireAborts` below proves the abort positively in those builds instead. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASPromoteRepublish, AbandonEmitsRemovalBeforeRetire) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl@cas@"}; + const String ref = "all_0_0_0"; + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = build->stageManifest(inlineEntries("f", "AAA")); + build->precommitAdd(ns, ref, id); + build->abandon(); + + /// `rebuild` never staged `id` -- A3 refuses it on that basis alone, before ever reaching the + /// ledger-state check that would otherwise distinguish "removed" from "still live". + auto rebuild = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + try + { + rebuild->precommitAdd(ns, ref, id); + FAIL() << "A3 mint-tightening: precommitAdd must refuse an id 'rebuild' never staged, even one " + "'build' legitimately staged and precommitted before dropping it via abandon()"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + } +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASPromoteRepublishDeathTest, AbandonEmitsRemovalBeforeRetireAborts) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl@cas@"}; + const String ref = "all_0_0_0"; + auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = build->stageManifest(inlineEntries("f", "AAA")); + build->precommitAdd(ns, ref, id); + build->abandon(); + + auto rebuild = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + EXPECT_DEATH({ rebuild->precommitAdd(ns, ref, id); }, ""); +} +#endif + +/// A3 mint-tightening's dedicated regression pin: an unowned `ManifestId` may enter ownership ONLY +/// from the transaction that freshly staged it. Without this, a dropped identity could be re-owned +/// later, which would make the relink confirm's exact-`ManifestRef` equality an ABA (the same token +/// could then name a manifest whose blobs were already reclaimed). No production path performs this +/// transition -- every real caller precommits an id it JUST staged itself, on the SAME `PartWriteTxn` +/// (`ContentAddressedTransaction.cpp:358,412`, `PartFolderAccess.cpp:352`). +/// +/// `txn2->precommitAdd(ns, ref, id)` below throws `LOGICAL_ERROR`, which aborts the whole process in +/// debug/sanitizer builds instead of behaving like a catchable exception -- `CASPromoteRepublishDeathTest. +/// PrecommitAddRejectsAnIdThisTxnDidNotStageAborts` below proves the abort positively in those builds +/// instead (it cannot also re-check the post-throw ref-log-tail/resolveRef state this test verifies, +/// since there IS no post-abort state in a real debug/sanitizer build). +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASPromoteRepublish, PrecommitAddRejectsAnIdThisTxnDidNotStage) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl_mint_tighten@cas@"}; + const String ref = "all_0_0_0"; + + /// txn1 mints `id` and abandons before ever precommitting it -- a genuinely unowned identity (it + /// was never even a live precommit), the simplest form A3 must still refuse for a foreign txn. + auto txn1 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = txn1->stageManifest(inlineEntries("f", "AAA")); + txn1->abandon(); + + const size_t tail_before = s->tailSinceSnapshotCountForTest(ns); + + /// txn2 never staged `id` -- only the transaction that minted an id may precommit it. + auto txn2 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + try + { + txn2->precommitAdd(ns, ref, id); + FAIL() << "A3 mint-tightening: precommitAdd must refuse an id staged by a DIFFERENT transaction"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + } + + /// Nothing was appended: the ref-log tail is unchanged and `ref` still has no owner at all. + EXPECT_EQ(s->tailSinceSnapshotCountForTest(ns), tail_before); + EXPECT_FALSE(s->resolveRef(ns, ref).has_value()); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASPromoteRepublishDeathTest, PrecommitAddRejectsAnIdThisTxnDidNotStageAborts) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv/tbl_mint_tighten@cas@"}; + const String ref = "all_0_0_0"; + + auto txn1 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + const ManifestId id = txn1->stageManifest(inlineEntries("f", "AAA")); + txn1->abandon(); + + auto txn2 = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, .intended_namespace = ns}); + EXPECT_DEATH({ txn2->precommitAdd(ns, ref, id); }, ""); +} +#endif diff --git a/src/Disks/tests/gtest_cas_protocol_scenarios.cpp b/src/Disks/tests/gtest_cas_protocol_scenarios.cpp new file mode 100644 index 000000000000..d70cca40a8ea --- /dev/null +++ b/src/Disks/tests/gtest_cas_protocol_scenarios.cpp @@ -0,0 +1,640 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Multi-actor protocol scenarios for the root-local part-manifest model (CA GC redesign rev. 15). +/// Ported from the removed tree/closure model. The single-call `publish(ns, ref, tree, RefPayload{})` +/// gate is gone; a write is now the four-step flow (EDGE-BEFORE-OBSERVE order): +/// stageManifest(entries) -> precommitAdd(ns, ref, id) -> putBlob(...) -> promote(ns, ref, build_id, id) +/// The fail-closed publish gate that those scenarios exercise now lives in TWO places (Phase A of spec +/// 2026-07-09-cas-writer-gc-simplification): +/// • putBlob: INV-1 condemned-dedup re-upload from the writer's OWN source bytes (never GETs the +/// dying object); +/// • promote: `Materialized` leaves (this build putBlob'd them) are EDGE-PROTECTED and NOT re-validated — the +/// precommit closure named them before putBlob observed them, so a condemnation in the +/// putBlob→promote window is doomed (the next fold spares it). promote commits without touching +/// the blob's current token. `TrustedManifest` leaves are accepted through their durable source +/// manifest edge with no per-file observation. +/// These scenarios assert the no-dangle / no-loss / fail-closed protocol properties faithfully on that +/// flow. The strong safety assertions are preserved. +/// +/// DELETED (Phase A): `RevalidateAbsentTokenedBlobResurrectsFromSource`. Its premise — a putBlob'd +/// (`Materialized`) blob body hand-deleted before the gate, then resurrected — is protocol-unreachable under +/// EDGE-BEFORE-OBSERVE: a materialized leaf under a durable precommit closure cannot be GC-deleted in the +/// putBlob→promote window, and promote no longer revalidates materialized leaves at all. Deleting a +/// putBlob'd body out-of-band is corruption, which is `cas-fsck`'s domain, not the promote gate's. + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +extern const int FILE_DOESNT_EXIST; +extern const int LOGICAL_ERROR; +} + +using namespace DB::Cas; +using DB::Cas::tests::blobEntryFor; +using DB::Cas::tests::condemnMeta; +using DB::Cas::tests::displaceBlobToken; +using DB::Cas::tests::idOf; +using DB::Cas::tests::injectRetire; +using DB::Cas::tests::loadMetaForTest; +using DB::Cas::tests::streamingHexOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::writeBlobRaw; + +namespace +{ + +PoolPtr openPool(const std::shared_ptr & b) +{ + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// A single-blob manifest entry naming `payload` at `path` (the entry the part's manifest carries). +ManifestEntry blobEntry(const String & path, const String & payload) +{ + return blobEntryFor(path, u128Of(payload), payload.size()); +} + +/// Start a build whose `intended_ref` is "ns/ref" — REQUIRED: stageManifest derives the manifest's +/// owning namespace by splitting intended_ref on the LAST '/'. (See PartWriteTxn::manifestNamespace.) +PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + return s->beginPartWrite(info); +} + +/// Seed an arbitrary blob identity through a complete writer transaction, preserving the production +/// ordering that makes physical publication legal. +void seedBlobWithDurablePrecommit( + const PoolPtr & store, const BlobRef & blob_ref, const String & payload) +{ + const RootNamespace ns{"fixture/seed"}; + auto build = startBuildFor(store, ns, "blob"); + ManifestEntry entry; + entry.path = "data.bin"; + entry.placement = EntryPlacement::Blob; + entry.ref = blob_ref; + entry.blob_size = payload.size(); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "blob", id); + build->putBlob(blob_ref, BlobSource::fromString(payload)); + build->promote(ns, "blob", build->buildId(), id); +} + +/// The full write flow for a part whose only file is `payload` at `path` (blob placement). Uploads the +/// blob via putBlob, stages the manifest, precommits, then promotes. Returns the committed ManifestId. +/// Mirrors what the old single-call `publish` did on the tree model. +ManifestId publishBlobPart( + const PoolPtr & s, const RootNamespace & ns, const String & ref, const String & path, const String & payload) +{ + auto build = startBuildFor(s, ns, ref); + /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob -> promote. + const ManifestId id = build->stageManifest({blobEntry(path, payload)}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// Read the part's blob back through the full read stack (resolveRef → readManifest → findEntry → +/// locate → ranged GET) and assert it returns `payload`. This is the INV-NO-DANGLE check: every named +/// object resolves and reads. +void assertPartReads( + const std::shared_ptr & b, const PoolPtr & s, + const RootNamespace & ns, const String & ref, const String & path, const String & payload) +{ + auto r = s->resolveRef(ns, ref); + ASSERT_TRUE(r.has_value()); + + const PartManifest manifest = s->readManifest(r->manifest_id); + const auto * entry = findEntry(manifest.entries, path); + ASSERT_TRUE(entry != nullptr); + auto loc = s->locate(*entry); + auto got = b->get(loc.key, Range{loc.offset, loc.length}); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, payload); +} + +} + +TEST(CASProtocol, FenceConflictCondemnedTokenedBlobCommitsWithTokenUnchanged) +{ + /// EDGE-BEFORE-OBSERVE (spec 2026-07-09-cas-writer-gc-simplification, Phase A): a blob leaf whose + /// CURRENT token is condemned at the promote gate, but which THIS build putBlob'd (`Materialized` proof under + /// the durable precommit closure), is EDGE-PROTECTED — the condemnation is doomed (the next fold spares + /// it) and promote does NOT revalidate or re-upload the materialized leaf. promote COMMITS with the blob's + /// token UNCHANGED; the premature condemn is invisible to the client. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Wiring order: stage + precommit (durable edge) BEFORE putBlob observes X (records token t0). + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); + + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + + /// GC condemns X at t0 in round 1 and fences the namespace to round 1. + injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + + /// promote: mutateShard refreshes the view (fence_round 1 > view round 0), but the materialized leaf is + /// edge-protected — skipped, not re-validated ⇒ commit, token unchanged. + build->promote(ns, "part_1", build->buildId(), id); + + /// The ref is committed and reads back; the blob still rides t0 (no re-upload). + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + EXPECT_EQ(b->head(blob_key).token, t0); +} + +TEST(CASProtocol, RevalidateReObservesStaleTokenKeepsWhenUnchanged) +{ + /// A blob dedup-adopted (`Materialized` proof) under the precommit closure; an EMPTY retire set at round 1. + /// Under EDGE-BEFORE-OBSERVE the materialized leaf is NOT re-observed at the promote gate at all — it is + /// edge-protected — so promote commits in place with the token UNCHANGED (no HEAD, no rewrite). + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// X pre-exists out-of-band; the build dedup-adopts it via putBlob (records the current token t0). + writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + + /// Wiring order: stage + precommit (durable edge) BEFORE the adopting putBlob. + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 + + /// GC advanced the round to 1 with an EMPTY retired set; fence to 1. X is NOT condemned and its + /// token is unchanged. + injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, {}); + + /// promote: the materialized leaf is edge-protected (not re-observed) ⇒ commit in place (KEEP). + build->promote(ns, "part_1", build->buildId(), id); + + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + /// No rewrite happened — the materialized leaf was never touched, so its object token stays at t0. + EXPECT_EQ(b->head(blob_key).token, t0); +} + +TEST(CASProtocol, RevalidateReObservesStaleTokenAdoptsWhenDisplaced) +{ + /// A blob displaced out-of-band to a fresh live token t1 before promote. Phase-A contract: the leaf is + /// `Materialized` (putBlob-adopted), so promote SKIPS it entirely (edge-protected — EDGE-BEFORE-OBSERVE); no + /// re-HEAD happens. The commit still rides the displaced object correctly because the manifest names + /// the HASH, not a token — this is the black-box "displaced object still reads by content key" check. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + + auto build = startBuildFor(s, ns, "part_1"); + /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob. + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 + + /// Another writer displaces X out-of-band ⇒ a new current token t1 (same payload, fresh tag). + const Token t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); + EXPECT_NE(t1, t0); + + /// GC advanced to round 1 with an EMPTY retired set; fence to 1. + injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, {}); + + /// promote refreshes ⇒ revalidate X ⇒ HEAD current t1 not condemned ⇒ commit. The dep rides t1. + build->promote(ns, "part_1", build->buildId(), id); + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + EXPECT_EQ(b->head(blob_key).token, t1); + + /// Black-box proof the part reads the t1 incarnation: re-publish the same blob into a SECOND + /// namespace with NO new GC injection. The blob is already present at t1; nothing is re-uploaded. + publishBlobPart(s, RootNamespace{"srv1/tbl/copy"}, "part_2", "data.bin", "payload-X"); + EXPECT_EQ(b->head(blob_key).token, t1); + assertPartReads(b, s, RootNamespace{"srv1/tbl/copy"}, "part_2", "data.bin", "payload-X"); + + /// Independent discriminator that the blob rides t1, not the stale t0: t0 is DEAD. A deleteExact + /// against t0 must TokenMismatch (INV-NO-RETURN — t0 was displaced and can never be current again). + EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); +} + +TEST(CASProtocol, RevalidateAdoptsLiveTokenWhenOnlyPhantomCondemnedAtDifferentToken) +{ + /// A blob whose OWN current token t0 is LIVE, but a DIFFERENT phantom token t_other for the same + /// hash IS condemned. The build records `Materialized` proof for t0, so promote does not re-observe it + /// (edge-protected) and commits in place: the blob keeps t0 (no upload, no displacement). The phantom + /// condemnation is for a different incarnation and never touches t0. + auto b = std::make_shared(); + const RootNamespace ns{"srv1/tbl"}; + + DB::Cas::Layout layout("p"); + { + auto s0 = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + writeBlobRaw(*b, s0->layout(), "payload-X", s0->poolMeta().blob_header_len, s0->poolMeta().pool_id); + } + const String blob_key = layout.blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + const Token t_other{"emulated-phantom", DB::Cas::TokenType::Emulated}; + ASSERT_NE(t_other, t0); + + injectRetire(*b, layout, /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t_other, .size = 9}}); + /// Fence to round 1 BEFORE opening the store, so the store's open-time refresh lands the view at + /// round 1 already populated. + + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); /// open-time refresh ⇒ view round 1 + /// Wiring order: stage + precommit (durable edge) BEFORE the adopting putBlob. + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 + + /// promote: the materialized leaf is edge-protected (not re-observed) ⇒ commit. Lands. t0 untouched. + build->promote(ns, "part_1", build->buildId(), id); + + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + + /// The object was NOT displaced — it STAYS at t0 (no re-upload, only re-validated). + EXPECT_EQ(b->head(blob_key).token, t0); +} + +/// (DELETED, Phase A) RevalidateAbsentTokenedBlobResurrectsFromSource — see the file-header note: a +/// hand-deleted putBlob'd (`Materialized`) body is protocol-unreachable under EDGE-BEFORE-OBSERVE (a +/// materialized leaf under a durable precommit closure cannot be GC-deleted in the putBlob→promote window, +/// and promote no longer revalidates materialized leaves). Out-of-band body deletion is `cas-fsck`'s domain. + +TEST(CASProtocol, EvidenceHitCondemnedPresentBlobCopiesForwardInClosure) +{ + /// `TrustedManifest` proof on a blob X whose hash is condemned-but-PRESENT. §4 manifest-trust + /// (test name is legacy — there is no copy-forward any more): a committed-source adopted leaf is TRUSTED + /// at the promote gate. The gate does NOT observe X — no HEAD, no meta point-read, no displacement — it + /// publishes on the strength of the durable manifest edge (D4 relink trust). So promote SUCCEEDS and X's + /// existing incarnation is left EXACTLY as-is: the token is UNCHANGED (never displaced) and the condemned + /// meta is NOT flipped (the gate never reads or writes it). Here the source-manifest proof is trusted; + /// materialized leaves are independently edge-protected. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// X pre-exists with token t0; the manifest names it as a tokenless adopted leaf. + const String hex = streamingHexOf("payload-X"); + const BlobRef seeded_ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(hex))}; + seedBlobWithDurablePrecommit(s, seeded_ref, "payload-X"); + const String blob_key = s->layout().blobKey(seeded_ref); + const Token t0 = b->head(blob_key).token; + + auto build = startBuildFor(s, ns, "part_1"); + ManifestEntry entry = blobEntry("data.bin", "payload-X"); + entry.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hexToU128(hex))}; /// streaming-convention id (matches the minted blob) + build->adoptEvidence(entry); /// tokenless W-EVIDENCE dep on X (no HEAD, no upload) + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + /// GC condemns X's hash in round 1 via the meta — under §4 the promote gate never reads it. + condemnMeta(*b, s->layout(), hexToU128(hex), /*condemn_round*/ 1); + + /// promote: the adopted leaf is trusted ⇒ commit, no probe, no displacement. + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + + /// The ref stands; X rides its ORIGINAL token t0 (trust never displaces a trusted leaf). + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); + EXPECT_EQ(b->head(blob_key).token, t0) << "trust must not displace the adopted blob"; + + /// The meta is untouched — still Condemned (the gate never reads or flips it under trust). + const auto lm_after = loadMetaForTest(*b, s->layout(), hexToU128(hex)); + ASSERT_TRUE(lm_after.has_value()); + EXPECT_EQ(lm_after->meta.state, MetaState::Condemned) << "trust must not flip the meta"; +} + +TEST(CASProtocol, WedgedHeartbeatCondemnedTokenedBlobCommitsWithTokenUnchanged) +{ + /// A build whose watermark never renews finds its OWN putBlob'd upload condemned by full GC while its + /// precommit is STILL the live owner (this setup injects only the retire set + fence, no owner-removal + /// — the false-positive-freeze window BEFORE any GC reclaim). The materialized leaf is EDGE-PROTECTED: the + /// precommit closure named it before putBlob observed it, so the condemnation is doomed and promote + /// does NOT re-validate it — promote COMMITS with the token UNCHANGED, closing the window invisibly. + /// The genuine dead-build case (precommit reclaimed ⇒ owner check aborts, NO re-upload) is covered + /// separately by CaWiringResurrect.PromoteAbandonedPrecommitAbortsWithoutResurrect. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Wiring order: stage + precommit (durable edge) BEFORE putBlob observes X. + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); + + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + + /// Full GC condemned the build's OWN upload. + injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + + /// promote: the materialized leaf is edge-protected — skipped, not revalidated ⇒ commit, token unchanged. + build->promote(ns, "part_1", build->buildId(), id); + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + EXPECT_EQ(b->head(blob_key).token, t0); +} + +TEST(CASProtocol, AbandonLeavesDebrisAndDisables) +{ + /// abandon leaves the uploaded blob + staged manifest body as debris (reaped by the orphan sweep); + /// no owner transition is touched, and further build ops fail LOGICAL_ERROR (requireAlive). + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + auto build = startBuildFor(s, ns, "part_1"); + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + auto blob = build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); + + build->abandon(); + + /// Both bodies remain as debris: once the manifest has named a durable precommit edge, its body + /// must survive until GC folds the matching owner removal. + EXPECT_TRUE(b->head(s->layout().blobKey(blob.ref)).exists); + EXPECT_TRUE(b->head(s->layout().manifestKey(id)).exists); + EXPECT_TRUE(s->listRefs(ns).empty()); + + /// Further build ops ⇒ LOGICAL_ERROR (requireAlive). + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + build->stageManifest({blobEntry("data.bin", "payload-X")}); + }, + "PartWriteTxn has been abandoned"); +} + +TEST(CASProtocol, DropReattachThroughDetachedNamespace) +{ + /// ATTACH choreography (design §4): publish part_1 in ns; re-publish into ns/detached + drop part_1 + /// from ns; then re-publish part_1 back in ns + drop from detached. The BLOB is never re-uploaded + /// (its token is stable throughout); each namespace gets its own single-owner manifest. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + const RootNamespace detached{"srv1/tbl/detached"}; + + publishBlobPart(s, ns, "part_1", "data.bin", "payload-X"); + + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token blob_tok = b->head(blob_key).token; + + EXPECT_TRUE(s->listRefs(ns).contains("part_1")); + EXPECT_TRUE(s->listRefs(detached).empty()); + + /// Move to detached: re-publish into detached (adopting the live blob), drop from ns. + publishBlobPart(s, detached, "part_1", "data.bin", "payload-X"); + s->dropRef(ns, "part_1"); + + EXPECT_TRUE(s->listRefs(ns).empty()); + ASSERT_TRUE(s->listRefs(detached).contains("part_1")); + assertPartReads(b, s, detached, "part_1", "data.bin", "payload-X"); + + /// Re-attach: re-publish part_1 back in ns, drop from detached. + publishBlobPart(s, ns, "part_1", "data.bin", "payload-X"); + s->dropRef(detached, "part_1"); + + ASSERT_TRUE(s->listRefs(ns).contains("part_1")); + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + EXPECT_TRUE(s->listRefs(detached).empty()); + + /// The blob was never re-uploaded (token stable throughout — every publish dedup-adopted it). + EXPECT_EQ(b->head(blob_key).token, blob_tok); +} + +TEST(CASProtocol, FreezeIntoShadowNamespace) +{ + /// `FREEZE` survives the table's part lifecycle (design §4): a shadow ref is a reachability root + /// that outlives the dropped live ref. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + const RootNamespace shadow{"srv1/shadow/backup1/tbl"}; + + publishBlobPart(s, ns, "part_1", "data.bin", "payload-X"); + + /// Freeze into the shadow namespace (adopting the live blob), then drop the live ref. + publishBlobPart(s, shadow, "part_1", "data.bin", "payload-X"); + s->dropRef(ns, "part_1"); + + EXPECT_TRUE(s->listRefs(ns).empty()); + /// The shadow ref still resolves and reads after the live ref is gone. + assertPartReads(b, s, shadow, "part_1", "data.bin", "payload-X"); +} + +TEST(CASProtocol, DisplacedToLiveTokenCommitsAtCurrentIncarnation) +{ + /// (Ported from the former ResurrectLosesRace scenario.) A blob displaced to a LIVE t1 (while its old + /// t0 is condemned for a now-defunct incarnation) is SAFE to commit: the committed manifest names a + /// blob HASH, the live t1 incarnation backs it, and GC's exact-token delete of t0 only TokenMismatches. + /// Phase-A contract: the leaf is TOKENED, so promote does not re-HEAD it at all (edge-protected — + /// EDGE-BEFORE-OBSERVE); the commit is correct by content addressing, not by revalidation. The old + /// conservative ABORTED has no manifest-model analog. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); + const String blob_key = s->layout().blobKey(idOf("payload-X")); + const Token t0 = b->head(blob_key).token; + + auto build = startBuildFor(s, ns, "part_1"); + /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob. + const ManifestId id = build->stageManifest({blobEntry("data.bin", "payload-X")}); + build->precommitAdd(ns, "part_1", id); + build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 + + /// Another writer displaces X to t1 (uncondemned) before our gate runs. + const Token t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); + ASSERT_NE(t1, t0); + + /// The view still condemns the OLD t0 at round 1, fenced. + injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + + /// promote: revalidate X ⇒ HEAD current t1 (NOT condemned; only the defunct t0 is) ⇒ commit. + build->promote(ns, "part_1", build->buildId(), id); + + /// The blob lives at t1 (the displacing writer's incarnation) and the part reads. + EXPECT_EQ(b->head(blob_key).token, t1); + assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); + + /// NO-LOSS / NO-RETURN: t0 is dead — a deleteExact against it TokenMismatches (the GC delete of the + /// condemned t0 spares the live t1). + EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); +} + +TEST(CASProtocol, NewNamespacePublishGatedByShardFenceFloor) +{ + /// Regression test (test name is legacy — the fence machinery is gone): build B adopts a blob, the + /// ack-floor GC pipeline retires + deletes it, then B publishes into a fresh namespace. §4 manifest- + /// trust: B's leaf is a committed-source adopted leaf, so promote TRUSTS it (no HEAD/loadMeta probe) and + /// COMMITS. On the real path this dangle is UNREACHABLE — B's precommit edge pins the blob at in-degree + /// >= 1 through promote (CasPartWriteTxn.cpp precommitAdd → promote's WPromote owner==bld re-proof precedes the + /// trust), so GC cannot delete it; here the test drives GC to delete the blob while B has NOT yet + /// precommitted, which the live-precommit invariant excludes. The dangle is DETECTED by fsck's + /// reachable-but-absent scan (the backstop), not prevented at promote. + auto b = std::make_shared(); + auto s = openPool(b); + + /// 1. part_1 → a blob in namespace A, through the real PartWriteTxn. + const RootNamespace ns_a{"srv1/tbl"}; + auto build_a = startBuildFor(s, ns_a, "part_1"); + const ManifestId id_a = build_a->stageManifest({blobEntry("data.bin", "floor-payload")}); + build_a->precommitAdd(ns_a, "part_1", id_a); + build_a->putBlob(idOf("floor-payload"), BlobSource::fromString("floor-payload")); + build_a->promote(ns_a, "part_1", build_a->buildId(), id_a); + const String blob_key = s->layout().blobKey(idOf("floor-payload")); + + /// 2. build B adopts the blob (tokenless W-EVIDENCE) while the view is still at round 0. + auto build_b = startBuildFor(s, RootNamespace{"srv2/new"}, "part_x"); + build_b->adoptEvidence(blobEntry("data.bin", "floor-payload")); + + /// 3. drop part_1 from A; the ack-floor GC pipeline retires the blob at t0 and deletes it. build_a + /// finished, so advancing the watermark floor condemns the blob. Drive rounds advancing the store's + /// own mount ack after each (so the floor graduates the condemned entry and the delete lands). + s->dropRef(ns_a, "part_1"); + build_a.reset(); + s->renewWatermarkOnce(); + Gc gc(s, hexToU128("00000000000000000000000000000001")); + for (size_t r = 0; r < 16; ++r) + { + const RoundReport rep = DB::Cas::tests::runRegularRoundReclaiming(gc); + s->renewWatermarkOnce(); + if (!b->head(blob_key).exists) + break; + } + /// The blob (unreachable) was deleted at t0. + EXPECT_FALSE(b->head(blob_key).exists); + + /// 4. build B publishes into a BRAND-NEW namespace. §4 manifest-trust: the adopted leaf is trusted at + /// promote (no probe) ⇒ promote SUCCEEDS and commits a manifest naming the deleted blob (the dangle). + const ManifestId id_b = build_b->stageManifest({blobEntry("data.bin", "floor-payload")}); + build_b->precommitAdd(RootNamespace{"srv2/new"}, "part_x", id_b); + EXPECT_NO_THROW(build_b->promote(RootNamespace{"srv2/new"}, "part_x", build_b->buildId(), id_b)); + + /// The ref committed over the deleted blob (the D4 trade-off); the backstop is fsck's reachable-but- + /// absent scan (INV-NO-DANGLE-via-fsck). + EXPECT_TRUE(s->resolveRef(RootNamespace{"srv2/new"}, "part_x").has_value()); + const FsckReport rep = runFsck(*s, /*detail=*/true); + EXPECT_GE(rep.dangling, 1u) << "§4 D4 backstop: part_x committed over the GC-deleted blob; fsck must " + "report it dangling (dangling=" << rep.dangling << ")"; +} + +TEST(CASProtocol, FreshEvidenceDepWithViewHitIsResolvedByGate) +{ + /// §4 manifest-trust (test name is legacy — the gate no longer "resolves" a tokenless leaf by observing + /// it): a committed-source adopted leaf whose blob is condemned-but-PRESENT is TRUSTED at the promote + /// gate. There is NO per-file probe (no HEAD, no meta point-read) and NO copy-forward — the durable + /// manifest edge is the liveness evidence (D4 relink trust). promote SUCCEEDS and X keeps its ORIGINAL + /// incarnation: the token t0 is UNCHANGED (never displaced). A materialized leaf is edge-protected; + /// this committed-source adopt instead carries trusted-manifest proof. + auto b = std::make_shared(); + + DB::Cas::Layout layout("p"); + const String hex = streamingHexOf("payload-fresh-ev"); + { + auto s0 = openPool(b); + seedBlobWithDurablePrecommit( + s0, + BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(hex))}, + "payload-fresh-ev"); + } + const String blob_key = layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(hex))}); + const Token t0 = b->head(blob_key).token; + condemnMeta(*b, layout, hexToU128(hex), /*condemn_round*/ 1); + + auto s = openPool(b); + + const RootNamespace ns{"srv1/tbl"}; + auto build = startBuildFor(s, ns, "part_1"); + /// adoptEvidence records a TOKENLESS dep. + ManifestEntry entry = blobEntry("data.bin", "payload-fresh-ev"); + entry.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hexToU128(hex))}; /// streaming-convention id (matches the minted blob) + build->adoptEvidence(entry); + const ManifestId id = build->stageManifest({entry}); + build->precommitAdd(ns, "part_1", id); + + /// promote trusts the adopted leaf ⇒ commit, no probe, no displacement. + EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); + + EXPECT_EQ(b->head(blob_key).token, t0) << "trust must not displace the adopted blob"; + EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); +} + +TEST(CASProtocol, AdoptedLeafCarriesRealBlobSize) +{ + /// B92 round-trip (re-expressed on the manifest model): an adopted leaf must carry its real + /// blob_size, NOT 0. PartWriteTxn A publishes a blob; build B adopts that leaf into a second ref. The + /// adopted manifest's entry must report the same non-zero blob_size as the original. + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// PartWriteTxn A: a blob with a real payload so blob_size > 0. + const ManifestId id_a = publishBlobPart(s, ns, "ref_a", "data.bin", "payload-B92"); + + const PartManifest manifest_a = s->readManifest(id_a); + const auto * entry_a = findEntry(manifest_a.entries, "data.bin"); + ASSERT_TRUE(entry_a != nullptr); + const uint64_t size_a = entry_a->blob_size; + EXPECT_NE(size_a, 0u) << "ref A blob_size must be non-zero"; + EXPECT_EQ(size_a, String("payload-B92").size()); + + /// PartWriteTxn B: adopt the same leaf, publish as ref_b (no re-upload). + auto build_b = startBuildFor(s, ns, "ref_b"); + ASSERT_TRUE(entry_a != nullptr); + build_b->adoptEvidence(*entry_a); + const ManifestId id_b = build_b->stageManifest({*entry_a}); + build_b->precommitAdd(ns, "ref_b", id_b); + build_b->promote(ns, "ref_b", build_b->buildId(), id_b); + + /// Resolve ref B: the adopted leaf's blob_size must match ref A (round-trip invariant for B92). + const PartManifest manifest_b = s->readManifest(s->resolveRef(ns, "ref_b")->manifest_id); + const auto * entry_b = findEntry(manifest_b.entries, "data.bin"); + ASSERT_TRUE(entry_b != nullptr); + EXPECT_NE(entry_b->blob_size, 0u) << "adopted leaf blob_size must not be 0 (B92)"; + EXPECT_EQ(entry_b->blob_size, size_a) << "adopted-leaf blob_size mismatch (B92 round-trip)"; +} + +/// ---- Genuinely-obsolete pure-tree-model scenarios (no manifest analog) ---- + +TEST(CASProtocol, DISABLED_RevalidateAbsentTreeDepRecreates) +{ + GTEST_SKIP() << "Obsolete (tree model). The gate's 'absent tree dep recreated from retained " + "payload' behavior has no manifest analog: a part manifest body is staged ONCE by " + "stageManifest and promote never re-creates it — an absent/invalid body at promote " + "fails closed (ABORTED). The blob-leaf absent-recreate case is covered by putBlob's " + "INV-1 re-upload-from-source path, not by the publish gate."; +} + +TEST(CASProtocol, DISABLED_AdoptTreeOfReclaimedTreeFailsClosedAtAdoptTime) +{ + GTEST_SKIP() << "Obsolete (tree model). adoptTree's fail-closed observe-at-adopt-time (one HEAD, " + "FILE_DOESNT_EXIST on an absent detached tree) has no manifest analog: the manifest " + "model's adoptEvidence is deliberately TOKENLESS and performs NO backend call — the " + "no-dangle guarantee for an adopted-but-reclaimed leaf is enforced at the promote " + "gate (unconditional blob revalidation ⇒ ABORTED), covered by " + "NewNamespacePublishGatedByShardFenceFloor and FreshEvidenceDepWithViewHitIsResolvedByGate."; +} diff --git a/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp b/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp new file mode 100644 index 000000000000..afe5ed26fb44 --- /dev/null +++ b/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp @@ -0,0 +1,636 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include +#include +#include +#include + +/// REBUILD CONDEMNS NOTHING, AND fsck WALKS STREAMS BY ARITHMETIC (spec 2026-07-27 "ref chain complete +/// cut" §7). +/// +/// REBUILD used to end with a LIST of `blobs/` and condemn every listed body its traversal had not +/// reached. That is the r5-finding-4 data-loss vector: the traversal itself is listing-driven, so a +/// store that omits a durable ref-log or manifest key from a LIST hides a LIVE owner, and the very same +/// pass then condemns the blob that owner pins. One lying enumeration, and acked data is scheduled for +/// deletion. The condemnation is GONE — REBUILD rebuilds cursors and edges and reclaims nothing. +/// +/// The NAMED residual that removal creates (Stage-A staging contract, register R4): a blob whose +/// manifest no longer exists anywhere is unreclaimable until the build/upload registry can enumerate +/// in-flight uploads. No substitute reclamation is added in its place — a quiet one would be the same +/// vector wearing a different hat (Constraint 3: no fallback). +/// +/// fsck's half is the other side of the same rule: it may not rest a verdict on a listing either. It +/// walks each namespace's ref stream by ARITHMETIC from `_ckpt.checkpoint` upward, reading every id by +/// exact key, and reports one verdict per namespace — `chain-broken` (a 404 below a CONFIRMED durable +/// same-epoch id: a hole, fatal in the summary AND in the exit code), `unchecked` (could not prove it +/// either way), or nothing at all. A finding is RECORDED, never thrown: an fsck that dies on the first +/// bad namespace says nothing about the ones it never reached. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); + +const RootNamespace kNsA{"00/aa@cas@"}; +const RootNamespace kNsB{"00/zz@cas@"}; + +/// Makes a second REBUILD catalog GET observe a different authority set. The command must take one +/// immutable cut at entry, so a correct implementation never triggers the mutation. +class CatalogChangesOnSecondReadBackend : public CountingBackend +{ +public: + using Backend::get; + + void armCatalogMutation(const String & key) + { + catalog_key = key; + catalog_reads = 0; + armed = true; + } + + size_t catalogReads() const { return catalog_reads; } + + std::optional get(const String & key, Range range) override + { + auto got = CountingBackend::get(key, range); + if (!armed || key != catalog_key) + return got; + + ++catalog_reads; + if (catalog_reads != 2) + return got; + if (!got) + throw std::runtime_error("catalog mutation fixture: second catalog read found absence"); + + const PutResult put = CountingBackend::putOverwrite( + key, encodeRefCatalog(RefCatalog{}), got->token, {}); + if (put.outcome != PutOutcome::Done) + throw std::runtime_error("catalog mutation fixture: catalog rewrite conflicted"); + return CountingBackend::get(key, range); + } + +private: + String catalog_key; + size_t catalog_reads = 0; + bool armed = false; +}; + +BlobRef blobRefOf(const DB::UInt128 & hash) +{ + return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}; +} + +bool blobPresent(Backend & backend, const Layout & layout, const DB::UInt128 & hash) +{ + return backend.head(layout.blobKey(blobRefOf(hash))).exists; +} + +/// Whether ANY run the newest fold seal references carries a `kCondemned` row for `hash`. This is where +/// a rebuild used to put its zero-edge condemnations, so "nothing was condemned" is checked HERE rather +/// than by watching for a deletion several rounds later. +bool condemnedInSealedRuns(Backend & backend, const Layout & layout, const DB::UInt128 & hash) +{ + const GcState st = decodeGcState(backend.get(layout.gcStateKey())->bytes); + const auto sealed = backend.get(layout.foldSealKey(st.snap_generation, st.snap_attempt)); + if (!sealed) + return false; + const CasFoldSeal seal = decodeFoldSeal(sealed->bytes); + for (const RunRef & r : seal.blob_target_runs) + { + auto reader = openSourceEdgeRun(backend, r.key); + String k; + String p; + while (reader.next(k, p)) + { + BlobRef ref; + UInt128 sid; + SourceEdgeKeyCodec::parse(k, ref, sid); + if (p.empty() || p[0] != kCondemned) + continue; + if (ref.digest.toU128() == hash) + return true; + } + } + return false; +} + +/// Publish `ref_name` -> a fresh manifest pinning `blob` at exactly `id`, and return the manifest's key +/// so a test can hide it from the listing. +String publishAtReturningManifestKey( + Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & id, + const String & ref_name, uint64_t build_sequence, const DB::UInt128 & blob, bool birth = false, + std::optional prev_epoch_seal = std::nullopt) +{ + publishAt(backend, layout, ns, id, ref_name, build_sequence, blob, birth, prev_epoch_seal); + return layout.manifestKey(ManifestId{ns, ManifestRef{.writer_epoch = id.writer_epoch, + .build_sequence = build_sequence, + .manifest_ordinal = 1}}); +} + +/// Publish the exact `_ckpt` that makes a raw fixture recoverable. The real writers go through +/// `publishCkpt`, which merges by semantic maximum and additionally refuses a `life_epoch` below the +/// durable one; this helper instead makes the fixture's one admissible recovery frontier explicit. +void writeCkptRaw(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefCkpt & ckpt) +{ + writeRecoverableCkptForRawFixture(backend, layout, ns, ckpt); +} + +/// The table state after applying exactly `ids`, through the same builder as recovery — so a snapshot +/// built from it is what the codec itself would have published. +RefTableState stateAfter(Backend & backend, const Layout & layout, const RootNamespace & ns, + const std::vector & ids) +{ + RefReplayBuilder builder(std::nullopt); + for (const RefTxnId & id : ids) + { + const auto got = backend.get(layout.refLogKey(fixture::fixtureLife(ns), id)); + if (!got) + throw std::runtime_error("stateAfter: fixture log " + std::to_string(id.writer_epoch) + "-" + + std::to_string(id.ref_sequence) + " is missing"); + builder.applyOne(decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id), + got->bytes.size()); + } + return std::move(builder).finish().state; +} + +/// Two writer epochs joined by a real seal: `{1,1} {1,2}` then the `{1,3}` seal, then `{2,1}` naming it +/// as its `prev_epoch_seal` and `{2,2}` after it. The exact-authority walk must handle this crossing. +void seedSealedTwoEpochStream(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + publishAt(backend, layout, ns, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(backend, layout, ns, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + writeSealAt(backend, layout, ns, RefTxnId{1, 3}); + publishAt(backend, layout, ns, RefTxnId{2, 1}, "ref_c", 1, DB::UInt128(3), /*birth=*/false, + /*prev_epoch_seal=*/RefTxnId{1, 3}); + publishAt(backend, layout, ns, RefTxnId{2, 2}, "ref_d", 2, DB::UInt128(4)); +} + +/// The number of rows in `cls` whose note mentions `needle`, over the whole report. +size_t rowsMentioning(const FsckReport & rep, FsckClass cls, const String & needle) +{ + size_t n = 0; + for (const FsckObject & o : rep.objects) + { + if (o.cls != cls) + continue; + for (const String & note : o.reachable_from) + if (note.find(needle) != String::npos) + ++n; + } + return n; +} + +} + +/// ---- REBUILD condemns nothing ---- + +/// THE REGRESSION TEST FOR r5-finding-4. A blob pinned by a COMMITTED ref, whose ref-log record and +/// whose manifest body the store both omit from every LIST while serving them perfectly by exact key. +/// The catalog row plus `_ckpt` frontier make the ref-log record authoritative, so REBUILD must recover +/// both owners despite the lying hint. The hidden manifest LIST still cannot justify condemnation: an +/// omitted object costs retention and never data. +TEST(CASRebuildCondemnNothing, HiddenLiveManifestBlobIsNotCondemned) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + /// Visible owner: ref_a pins blob 1. + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, DB::UInt128(1), /*birth=*/true); + /// HIDDEN owner: ref_b pins blob 2, and neither its record nor its manifest is ever listed. + const String hidden_manifest = + publishAtReturningManifestKey(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", /*build_sequence=*/2, DB::UInt128(2)); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + backend->hide(layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2})); + backend->hide(hidden_manifest); + + /// Precondition: both objects really are durable and really are hidden. + ASSERT_TRUE(backend->get(layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2})).has_value()); + ASSERT_TRUE(backend->get(hidden_manifest).has_value()); + + Gc gc(store, kGc); + const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); + ASSERT_TRUE(rep.performed) << rep.refusal; + ASSERT_GT(backend->holesServed(), 0u) << "the hidden keys were never actually omitted from a LIST"; + EXPECT_EQ(rep.committed_refs, 2u) + << "the immutable checkpoint frontier, not the lying LIST, defines both committed owners"; + + EXPECT_FALSE(condemnedInSealedRuns(*backend, layout, DB::UInt128(2))) + << "the hidden owner's blob was condemned — that is acked data scheduled for deletion"; + EXPECT_FALSE(loadMetaForTest(*backend, layout, DB::UInt128(2)).has_value()) + << "a rebuild condemns nothing, so it publishes no condemn marker"; + + /// And it survives the pipeline: rounds run, nothing reclaims it. + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) << "acked data was deleted after a rebuild"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(1))); +} + +/// An ORPHAN blob — one no manifest anywhere names — is likewise left alone. This is the NAMED residual +/// (register R4) stated as a test rather than as prose: until the build/upload registry can enumerate +/// in-flight uploads, a manifest-less blob is unreclaimable, and the rebuild does NOT get to guess. The +/// blob a live ref pins and the blob nothing pins are indistinguishable from a LIST, which is exactly +/// why the old pass could not tell them apart either. +TEST(CASRebuildCondemnNothing, OrphanBlobIsRetainedNotCondemned) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, DB::UInt128(1), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + writeBlobBody(*backend, layout, DB::UInt128(2)); /// orphan: present, named by nothing + + Gc gc(store, kGc); + const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); + ASSERT_TRUE(rep.performed) << rep.refusal; + + EXPECT_FALSE(condemnedInSealedRuns(*backend, layout, DB::UInt128(2))); + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + ASSERT_TRUE(seal.condemned_summary.contains(0)) << "the summary stays TOTAL over gc_shards"; + EXPECT_EQ(seal.condemned_summary.at(0).condemned_total, 0u) << "a rebuild condemns nothing"; + + for (int i = 0; i < 4; ++i) + { + gc.runRegularRound(); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) + << "the residual is RETENTION: an orphan is kept, not quietly reclaimed by a substitute pass"; +} + +/// The rebuild is the `gc/state` disaster-recovery command, so it is the LAST thing that may refuse to +/// run over a pool holding one bad key. A name-bearing segment under the opaque stream root is not a +/// canonical physical life id. The key must be skipped -- no catalog entry can claim it -- and the +/// rebuild must continue over unrelated cataloged lives. +TEST(CASRebuildCondemnNothing, NonCanonicalLifeKeyDoesNotAbortTheRebuild) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, DB::UInt128(1), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + /// Hand-built: no helper can mint this shape any more. + const String noncanonical_life = + layout.casRefsPrefix() + kNsA.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + ASSERT_EQ(backend->putIfAbsent(noncanonical_life, "garbage").outcome, PutOutcome::Done); + + Gc gc(store, kGc); + RebuildReport rep; + ASSERT_NO_THROW(rep = gc.rebuildBaseline(/*force=*/true)) + << "the recovery command must not be taken out by the damage it exists to recover from"; + ASSERT_TRUE(rep.performed) << rep.refusal; + + /// The live namespace was still discovered and folded -- the bad key was skipped, not the pool. + EXPECT_EQ(rep.committed_refs, 1u) << "the malformed key must not hide the live namespace"; +} + +/// The SECOND way the same damage can reach the rebuild, and it is a different code path from the one +/// above: the gen-0 health check LISTs each namespace's own life prefix and groups those keys to decide +/// whether any table proves cleaned logs. `groupRefKeys` refuses a key that names no life, so a NESTED +/// shape under the life prefix (`/x/_log/.zst`) reaches the refusal there instead of at +/// `discoverUniverse`, which absorbs it. Same rule, same reason: the recovery command must not be taken +/// out by the damage it exists to recover from. +/// +/// A decodable `gc/state` at generation 0 is what makes that branch run at all -- with no state object +/// the health check never reaches it. The state here comes from one real round (so the lease belongs to +/// this identity) with `snap_generation` written back to 0. +TEST(CASRebuildCondemnNothing, NestedLifelessKeyUnderTheLifePrefixDoesNotAbortTheRebuild) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, DB::UInt128(1), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + { + const auto got = backend->get(layout.gcStateKey()); + ASSERT_TRUE(got.has_value()); + GcState st = decodeGcState(got->bytes); + st.snap_generation = 0; + ASSERT_EQ(backend->putOverwrite(layout.gcStateKey(), encodeGcState(st), got->token).outcome, + PutOutcome::Done); + } + + /// Hand-built, and planted AFTER the round so the round itself is clean: one segment too deep under + /// the life prefix, so the segment where the incarnation belongs holds `x`. No helper mints this. + const String nested = layout.namespaceStreamPrefix(fixture::fixtureLife(kNsA)) + + "x/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + ASSERT_EQ(backend->putIfAbsent(nested, "garbage").outcome, PutOutcome::Done); + + RebuildReport rep; + ASSERT_NO_THROW(rep = gc.rebuildBaseline(/*force=*/false)) + << "the recovery command must not be taken out by the damage it exists to recover from"; + /// FORCE is deliberately NOT passed: a listing the check could not group proves nothing about the ref + /// baseline, so the pool cannot be declared healthy, and the un-forced rebuild must therefore RUN + /// rather than refuse. + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.committed_refs, 1u) << "the live namespace must still be folded"; +} + +TEST(CASRebuildCondemnNothing, OneCatalogCutDrivesHealthCheckAndRebuild) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, + DB::UInt128(1), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + /// A decoded generation-0 state exercises the health check's namespace walk before the rebuild + /// universe is consumed. The backend would erase the authority set on a second catalog GET. + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + { + const auto got = backend->get(layout.gcStateKey()); + ASSERT_TRUE(got); + GcState state = decodeGcState(got->bytes); + state.snap_generation = 0; + ASSERT_EQ(backend->putOverwrite( + layout.gcStateKey(), encodeGcState(state), got->token).outcome, PutOutcome::Done); + } + + backend->armCatalogMutation(layout.refCatalogKey()); + const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); + EXPECT_EQ(backend->catalogReads(), 1u) + << "REBUILD must use its entry cut for both the generation-0 health check and the rebuild"; + ASSERT_TRUE(rep.performed) << rep.refusal; + EXPECT_EQ(rep.committed_refs, 1u); +} + +/// Removing the condemnation must not disturb the other thing a rebuild owes: every hold in the prior +/// seal rides through VERBATIM (Task 8). Asserted together with the condemn-nothing rule because the +/// two used to be produced by the same pass, and a hold dropped here would hand back a baseline that +/// claims a frontier proof it does not have. +TEST(CASRebuildCondemnNothing, CarriesHoldsVerbatimWhileCondemningNothing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", /*build_sequence=*/1, DB::UInt128(1), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + writeBlobBody(*backend, layout, DB::UInt128(2)); /// an orphan alongside the held namespace + + /// One real round first: it establishes the pool's `gc/state` and takes the lease under THIS + /// identity, so the rebuild below is the disaster-recovery path and not a lease conflict. + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + const RefHold planted{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{4, 9}, + .retry_count = 17, .next_retry_round = 23}; + { + const GcState adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + seedFoldCursorForTest(*backend, layout, kNsA, RefTxnId{1, 1}, planted, + adopted.snap_generation, adopted.snap_attempt); + } + + const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); + ASSERT_TRUE(rep.performed) << rep.refusal; + + const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const auto it = seal.ref_lives.find(catalogLifeIdForTest(*backend, layout, kNsA)); + ASSERT_NE(it, seal.ref_lives.end()); + EXPECT_EQ(it->second.coverage.classification, 4); + ASSERT_TRUE(it->second.coverage.hold.has_value()); + EXPECT_EQ(*it->second.coverage.hold, planted) + << "a rebuild retried nothing, so it rewrites nothing about the hold"; + + EXPECT_FALSE(condemnedInSealedRuns(*backend, layout, DB::UInt128(2))); +} + +/// ---- fsck: arithmetic streams ---- + +/// A pool with nothing wrong reports nothing: no hole, no unproven namespace, a clean bill of health +/// and a zero exit. `unchecked` is not a resting state — it is a verdict a healthy pool never reaches. +TEST(CASRebuildCondemnNothingFsck, HealthyArithmeticPoolIsClean) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 3}, "ref_c", 3, DB::UInt128(3)); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + const FsckReport rep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(rep.clean()) << formatFsckSummary(rep); + EXPECT_EQ(rep.chain_broken, 0u); + EXPECT_EQ(rep.unchecked, 0u) << "a healthy namespace is PROVEN, not merely uncomplained-about"; + EXPECT_EQ(rep.ref_records_walked, 3u); + EXPECT_EQ(rep.dangling, 0u); +} + +/// A 404 BELOW a durable same-epoch id. Ids are dense `1..T` within `(namespace, epoch)` (INV-1), so +/// this cannot be the end of a stream: a durable record is missing and every transaction above it is +/// unreachable. The verdict is FATAL — it appears in the machine-parseable summary line and it makes +/// the report unclean, which is what turns into the command's nonzero exit. +TEST(CASRebuildCondemnNothingFsck, MidChainHoleBelowAWitnessIsChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 3}, "ref_c", 3, DB::UInt128(3)); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + /// Punch the hole: {1,2} is gone while {1,3} stays durable and listed. + const String holed = layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2}); + const HeadResult h = backend->head(holed); + ASSERT_TRUE(h.exists); + backend->deleteExact(holed, h.token); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail=*/true)) + << "a finding is RECORDED, never thrown — an fsck that dies reports nothing"; + EXPECT_EQ(rep.chain_broken, 1u); + EXPECT_FALSE(rep.clean()) << "chain-broken is a hard finding"; + EXPECT_NE(formatFsckSummary(rep).find("chain_broken=1"), String::npos) << formatFsckSummary(rep); + + bool row = false; + for (const FsckObject & o : rep.objects) + if (o.cls == FsckClass::ChainBroken) + row = true; + EXPECT_TRUE(row) << "the fatal must name the position it was detected at"; +} + +/// The tail ABOVE `_ckpt.checkpoint` is WALKED, not assumed. Here the store lists neither of the two +/// records above the checkpoint, so a listing-driven audit would see an empty tail and report a clean +/// pool it never read. Arithmetic reads them by exact key: they are walked, counted, and the namespace +/// comes back PROVEN — not `unchecked`, which is reserved for what cannot be proved at all. +TEST(CASRebuildCondemnNothingFsck, TailAboveTheCheckpointIsWalkedNotUnchecked) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 3}, "ref_c", 3, DB::UInt128(3)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 4}, "ref_d", 4, DB::UInt128(4)); + + /// A published snapshot at {1,2}, named by the checkpoint. Its bytes are the codec's own view of + /// the state at {1,2}, so exact checkpoint-base validation accepts it. + writeRefSnapshotRaw(*backend, layout, + snapshotOf(stateAfter(*backend, layout, kNsA, {RefTxnId{1, 1}, RefTxnId{1, 2}}), kNsA.string())); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 4}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, .last_epoch_seal = std::nullopt}); + + /// The store stops listing the tail. It stays perfectly readable by exact key. + backend->hide(layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 3})); + backend->hide(layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 4})); + + const FsckReport rep = runFsck(*store, /*detail=*/true); + ASSERT_GT(backend->holesServed(), 0u) << "the tail was never actually hidden from a LIST"; + EXPECT_EQ(rep.ref_records_walked, 2u) << "the two records above the checkpoint must be read by exact key"; + EXPECT_EQ(rep.unchecked, 0u) << "a walked tail is PROVEN; `unchecked` is not a default"; + EXPECT_EQ(rep.chain_broken, 0u); + EXPECT_TRUE(rep.clean()) << formatFsckSummary(rep); +} + +/// A SEALED multi-epoch stream, walked. The epoch boundary is crossed the way the protocol proves it — +/// through the next epoch's `prev_epoch_seal` back-chain — never by guessing `epoch + 1`. The +/// checkpoint sits below the seal, so the tail the walk owes covers the boundary itself. +TEST(CASRebuildCondemnNothingFsck, SealedStreamIsWalkedAcrossTheEpochBoundary) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + seedSealedTwoEpochStream(*backend, layout, kNsA); + writeRefSnapshotRaw(*backend, layout, + snapshotOf(stateAfter(*backend, layout, kNsA, {RefTxnId{1, 1}, RefTxnId{1, 2}}), kNsA.string())); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{2, 2}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, .last_epoch_seal = RefTxnId{1, 3}}); + + const FsckReport rep = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(rep.clean()) << formatFsckSummary(rep); + EXPECT_EQ(rep.chain_broken, 0u); + EXPECT_EQ(rep.unchecked, 0u) << "a PROVED crossing is not an unproven one"; + EXPECT_EQ(rep.ref_records_walked, 3u) << "the seal plus both records of the epoch it opened"; +} + +/// An exact `_ckpt` frontier turns an impossible epoch crossing into a hard chain break. Here `ns_a`'s +/// epoch 1 ends at the PRESENT ordinary record `{1,3}`, while both `_ckpt` and `{2,1}` falsely claim +/// that position as the closing seal. The finite range is complete and proves the contradiction: this +/// is NOT `unchecked`, and there is no earlier missing record that could make the test pass instead. +/// The healthy `ns_b` in the same pool is unaffected — one broken namespace never spreads. +TEST(CASRebuildCondemnNothingFsck, ExactFrontierMakesAnUnsealedEpochCrossingChainBroken) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 3}, "ref_c", 3, DB::UInt128(3)); + /// `{1,3}` exists but is an ordinary owner transaction, not an `EpochSeal`. Claiming it as the + /// predecessor must not authorize the transition to epoch 2. + publishAt(*backend, layout, kNsA, RefTxnId{2, 1}, "ref_d", 1, DB::UInt128(4), /*birth=*/false, + /*prev_epoch_seal=*/RefTxnId{1, 3}); + + publishAt(*backend, layout, kNsB, RefTxnId{1, 1}, "ref_z", 1, DB::UInt128(9), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{1, 3}}); + writeCkptRaw(*backend, layout, kNsB, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail=*/true)); + EXPECT_EQ(rep.unchecked, 0u) << "the exact frontier proves this crossing inconsistent"; + EXPECT_EQ(rep.chain_broken, 1u) << "exactly the malformed namespace must be reported"; + const String missing_key = layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 4}); + EXPECT_EQ(std::count_if(rep.objects.begin(), rep.objects.end(), [&](const FsckObject & object) + { + return object.cls == FsckClass::ChainBroken && object.key == missing_key; + }), 1u) << "the present ordinary 1-3 cannot close epoch 1, so arithmetic continuation requires 1-4"; + EXPECT_GE(rowsMentioning(rep, FsckClass::ChainBroken, "checkpoint requires id 1-4"), 1u) + << "the verdict must expose that the claimed ordinary predecessor did not authorize a crossing"; + EXPECT_GE(rowsMentioning(rep, FsckClass::ChainBroken, "inclusive frontier 2-1"), 1u) + << "the verdict must name the authority that made the absence a proven chain break"; +} + +/// W2 (Task-3 review): a holed namespace used to make the WHOLE scan throw — `applyOne` raises +/// `CORRUPTED_DATA` on a non-contiguous replay and nothing caught it, so one bad table aborted the +/// audit and every namespace after it went unexamined. For recovery, throwing is the correct +/// fail-close; for a read-only diagnostic it violates "record and continue, never wedge". +TEST(CASRebuildCondemnNothingFsck, OneBadNamespaceDoesNotAbortTheAudit) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + const Layout & layout = store->layout(); + + /// `ns_a` sorts FIRST, so a scan that dies on it never reaches `ns_b`. + publishAt(*backend, layout, kNsA, RefTxnId{1, 1}, "ref_a", 1, DB::UInt128(1), /*birth=*/true); + publishAt(*backend, layout, kNsA, RefTxnId{1, 2}, "ref_b", 2, DB::UInt128(2)); + publishAt(*backend, layout, kNsA, RefTxnId{1, 3}, "ref_c", 3, DB::UInt128(3)); + writeCkptRaw(*backend, layout, kNsA, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + const String holed = layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2}); + const HeadResult h = backend->head(holed); + ASSERT_TRUE(h.exists); + backend->deleteExact(holed, h.token); + + publishAt(*backend, layout, kNsB, RefTxnId{1, 1}, "ref_z", 1, DB::UInt128(9), /*birth=*/true); + writeCkptRaw(*backend, layout, kNsB, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + FsckReport rep; + ASSERT_NO_THROW(rep = runFsck(*store, /*detail=*/true)); + EXPECT_EQ(rep.chain_broken, 1u); + EXPECT_GE(rep.reachable, 1u) << "the namespace AFTER the broken one must still have been examined"; + EXPECT_EQ(rep.dangling, 0u) << "`ns_b` is healthy; a wedged scan would have reported nothing about it"; +} diff --git a/src/Disks/tests/gtest_cas_record_stream_format.cpp b/src/Disks/tests/gtest_cas_record_stream_format.cpp new file mode 100644 index 000000000000..42e44585c357 --- /dev/null +++ b/src/Disks/tests/gtest_cas_record_stream_format.cpp @@ -0,0 +1,255 @@ +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; + extern const int UNKNOWN_FORMAT_VERSION; +} + +namespace +{ + +BlobRef chRef(uint64_t n) +{ + return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(n))}; +} + +SourceEdgeRecord edge(const BlobRef & ref, uint64_t source_id) +{ + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(source_id), .marker = kEdgeActive}; +} + +SourceEdgeRecord zero(const BlobRef & ref) +{ + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = kZeroMarker}; +} + +SourceEdgeRecord condemned(const BlobRef & ref, const Token & token, uint64_t size, uint64_t round, bool pend) +{ + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = kCondemned, + .delete_pending = pend, .token = token, .size = size, .condemn_round = round}; +} + +/// Encode a run from records already in (ref, source_id) order. +String encodeRun(const std::vector & recs) +{ + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + for (const auto & r : recs) + writer.append(r); + writer.finish(); + out.finalize(); + return out.str(); +} + +/// Stream a run back to records; verifies the trailer count as a side effect. +std::vector decodeRun(const String & bytes) +{ + ReadBufferFromMemory in(bytes.data(), bytes.size()); + SourceEdgeRunReader reader(in); + std::vector out; + SourceEdgeRecord r; + while (reader.next(r)) + out.push_back(r); + return out; +} + +} + +TEST(CASRecordStream, EmptyRunRoundTripsAndChecksumMatches) +{ + const String bytes = encodeRun({}); + EXPECT_EQ(bytes, fmt::format( + "{{\"type\":\"cas_run\",\"v\":{},\"kind\":\"source_edge\"}}\n{{\"n\":0}}\n", currentCompatibilityVersion())); + + ReadBufferFromMemory in(bytes.data(), bytes.size()); + SourceEdgeRunReader reader(in); + SourceEdgeRecord r; + EXPECT_FALSE(reader.next(r)); + /// The read-side accumulated hash equals the write-side helper over the same bytes. + reader.verifyAgainst(sourceEdgeRunChecksum(bytes)); +} + +TEST(CASRecordStream, EdgeZeroCondemnedRoundTrip) +{ + const BlobRef a = chRef(1); + const BlobRef b = chRef(2); + const BlobRef c = chRef(3); + /// Sorted by (ref, source_id): b's condemned sentinel is at source_id 0 (sorts first for b); a has + /// an edge; c has a zero marker. Blobs ascend a < b < c, so the sequence is already non-decreasing. + std::vector recs = { + edge(a, 10), + condemned(b, Token{"e-1", TokenType::ETag}, 4242, 7, /*pend*/ true), + zero(c), + }; + const String bytes = encodeRun(recs); + const std::vector back = decodeRun(bytes); + ASSERT_EQ(back.size(), 3u); + + EXPECT_EQ(back[0].ref, a); + EXPECT_EQ(back[0].source_id, UInt128(10)); + EXPECT_EQ(back[0].marker, kEdgeActive); + + EXPECT_EQ(back[1].ref, b); + EXPECT_EQ(back[1].source_id, UInt128(0)); + EXPECT_EQ(back[1].marker, kCondemned); + EXPECT_TRUE(back[1].delete_pending); + EXPECT_EQ(back[1].token, (Token{"e-1", TokenType::ETag})); + EXPECT_EQ(back[1].size, 4242u); + EXPECT_EQ(back[1].condemn_round, 7u); + + EXPECT_EQ(back[2].ref, c); + EXPECT_EQ(back[2].marker, kZeroMarker); +} + +TEST(CASRecordStream, WriterIsByteDeterministic) +{ + std::vector recs = { + edge(chRef(1), 5), + edge(chRef(1), 9), + condemned(chRef(2), Token{"t/with/slashes", TokenType::ETag}, 1, 2, false), + }; + EXPECT_EQ(encodeRun(recs), encodeRun(recs)); /// pure function of the sorted record set +} + +TEST(CASRecordStream, SortOrderAcrossAlgosFollowsAlgoByte) +{ + /// b = . The algo byte leads, so string-sorting b reproduces the + /// binary (algo, digest, source_id) order: ch128 (01) < xxh3 (02) < sha256 (03). + BlobDigest d16 = BlobDigest::fromU128(UInt128(7)); + BlobDigest d32{}; + d32.bytes[0] = 0x10; + const BlobRef ch{BlobHashAlgo::CityHash128, d16}; + const BlobRef xx{BlobHashAlgo::XXH3_128, d16}; + const BlobRef sha{BlobHashAlgo::Sha256, d32}; + + /// Accepted in algo-byte order without an out-of-order throw. + const String bytes = encodeRun({edge(ch, 1), edge(xx, 1), edge(sha, 1)}); + const std::vector back = decodeRun(bytes); + ASSERT_EQ(back.size(), 3u); + EXPECT_EQ(back[0].ref.algo, BlobHashAlgo::CityHash128); + EXPECT_EQ(back[1].ref.algo, BlobHashAlgo::XXH3_128); + EXPECT_EQ(back[2].ref.algo, BlobHashAlgo::Sha256); +} + +TEST(CASRecordStream, AppendOutOfOrderThrows) +{ + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(edge(chRef(2), 1)); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + writer.append(edge(chRef(1), 1)); + }, + "records appended out of"); /// ref regression +} + +TEST(CASRecordStream, SourceIdRendersAs32Hex) +{ + const String bytes = encodeRun({edge(chRef(1), 10)}); + /// The source id 10 is a 32-char lowercase hex string ending in 'a'. + EXPECT_NE(bytes.find("\"s\":\"0000000000000000000000000000000a\""), String::npos); + /// The record key `b` for a ch128 ref is the algo byte 01 + a 32-hex digest (34 chars total). + EXPECT_NE(bytes.find("\"b\":\"01"), String::npos); +} + +TEST(CASRecordStream, SealChecksumMismatchFailsClosed) +{ + const String bytes = encodeRun({edge(chRef(1), 10), edge(chRef(1), 20)}); + const UInt128 good = sourceEdgeRunChecksum(bytes); + + /// A correct verify passes. + { + ReadBufferFromMemory in(bytes.data(), bytes.size()); + SourceEdgeRunReader reader(in); + SourceEdgeRecord r; + while (reader.next(r)) {} + reader.verifyAgainst(good); + } + + /// Any byte flip either fails the parse or the whole-file checksum — never silently trusted. + String flipped = bytes; + flipped[flipped.size() / 2] ^= 0x20; + EXPECT_NE(sourceEdgeRunChecksum(flipped), good); + EXPECT_THROW({ + ReadBufferFromMemory in(flipped.data(), flipped.size()); + SourceEdgeRunReader reader(in); + SourceEdgeRecord r; + while (reader.next(r)) {} + reader.verifyAgainst(good); + }, DB::Exception); +} + +TEST(CASRecordStream, TrailerCountMismatchIsCorruptData) +{ + String bytes = encodeRun({edge(chRef(1), 10)}); + /// Rewrite the trailer count 1 -> 2. + const String from = "{\"n\":1}\n"; + const String to = "{\"n\":2}\n"; + const size_t at = bytes.rfind(from); + ASSERT_NE(at, String::npos); + bytes.replace(at, from.size(), to); + EXPECT_THROW(decodeRun(bytes), DB::Exception); +} + +TEST(CASRecordStream, TruncationAtLineBoundaryFailsClosed) +{ + const String bytes = encodeRun({edge(chRef(1), 10), edge(chRef(1), 20)}); + /// Drop the trailer line entirely (truncate after the last record's newline). + const size_t trailer = bytes.rfind("{\"n\":"); + ASSERT_NE(trailer, String::npos); + EXPECT_THROW(decodeRun(bytes.substr(0, trailer)), DB::Exception); +} + +TEST(CASRecordStream, HeaderGates) +{ + /// Wrong type. + { + const String s = "{\"type\":\"cas_pool_meta\",\"v\":3,\"kind\":\"source_edge\"}\n{\"n\":0}\n"; + EXPECT_THROW(decodeRun(s), DB::Exception); + } + /// Wrong kind. + { + const String s = "{\"type\":\"cas_run\",\"v\":3,\"kind\":\"blob_delta\"}\n{\"n\":0}\n"; + EXPECT_THROW(decodeRun(s), DB::Exception); + } + /// Future version -> UNKNOWN_FORMAT_VERSION. + { + const String s = fmt::format( + "{{\"type\":\"cas_run\",\"v\":{},\"kind\":\"source_edge\"}}\n{{\"n\":0}}\n", currentCompatibilityVersion() + 1); + ReadBufferFromMemory in(s.data(), s.size()); + try + { + SourceEdgeRunReader reader(in); + FAIL() << "expected UNKNOWN_FORMAT_VERSION"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + } + } + /// An out-of-range version must not narrow to a valid low u32 value. + { + const String s = "{\"type\":\"cas_run\",\"v\":4294967299,\"kind\":\"source_edge\"}\n{\"n\":0}\n"; + try + { + decodeRun(s); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + } +} diff --git a/src/Disks/tests/gtest_cas_recovery_grounding.cpp b/src/Disks/tests/gtest_cas_recovery_grounding.cpp new file mode 100644 index 000000000000..5362813bdfa6 --- /dev/null +++ b/src/Disks/tests/gtest_cas_recovery_grounding.cpp @@ -0,0 +1,701 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int INVALID_STATE; +} + +using namespace DB::Cas; + +namespace +{ + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::minimalLiveSnapshot; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; +using DB::Cas::tests::seedPoolMetaForRestart; +using DB::Cas::tests::writeRefSnapshotRaw; + +enum class ListingMode : uint8_t +{ + Full, + Empty, + Partial, + Reordered, +}; + +class RecoveryListingBackend : public CountingBackend +{ +public: + explicit RecoveryListingBackend(ListingMode mode_) : mode(mode_) { seedPoolMetaForRestart(*this); } + + size_t list_calls = 0; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ++list_calls; + ListPage page = CountingBackend::list(prefix, cursor, limit); + if (mode == ListingMode::Empty) + page.keys.clear(); + else if (mode == ListingMode::Partial) + { + page.keys.erase(std::remove_if(page.keys.begin(), page.keys.end(), [](const ListedKey & key) + { + return key.key.find("/_log/") != String::npos; + }), page.keys.end()); + } + else if (mode == ListingMode::Reordered) + std::reverse(page.keys.begin(), page.keys.end()); + return page; + } + +private: + ListingMode mode; +}; + +RefLogTxn txn(const RootNamespace & ns, RefTxnId id, std::vector ops, + std::optional previous_seal = std::nullopt) +{ + return RefLogTxn{.ns = ns.string(), .txn_id = id, .ops = std::move(ops), .prev_epoch_seal = previous_seal}; +} + +std::map committedOf(const RefTableState & state) +{ + std::map result; + for (const auto [name, row] : state.getCommitted()) + result.emplace(name, row.manifest_ref); + return result; +} + +void seedAuthoritativeStream(Backend & backend, const Layout & layout, const RootNamespace & ns, + RefTxnId committed_through, bool include_f_plus_one = false) +{ + const ManifestRef first{1, 1, 1}; + std::vector birth{namespaceBirthOp()}; + const auto first_publish = publishCommittedOps("a", first); + birth.insert(birth.end(), first_publish.begin(), first_publish.end()); + const RefLogTxn first_txn = txn(ns, {1, 1}, std::move(birth)); + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, first_txn); + + if (committed_through > RefTxnId{1, 1}) + { + RefOp seal_op; + seal_op.kind = RefOpKind::EpochSeal; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, txn(ns, {1, 2}, {std::move(seal_op)})); + const ManifestRef second{2, 1, 1}; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, + txn(ns, {2, 1}, publishCommittedOps("b", second), RefTxnId{1, 2})); + } + if (include_f_plus_one) + { + const ManifestRef extra{1, 2, 1}; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, txn(ns, {1, 2}, publishCommittedOps("uncommitted", extra))); + } + + RefTableState snapshot_state; + applyRefLogTxn(snapshot_state, first_txn); + writeRefSnapshotRaw(backend, layout, snapshotOf(snapshot_state, ns.string())); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(backend, layout, ns); + const RefCkpt authority{ + .life_epoch = 1, + .committed_through = committed_through, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = committed_through.writer_epoch > 1 + ? std::optional{RefTxnId{1, 2}} : std::nullopt}; + backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(authority)); +} + +/// This is deliberately caller-side plumbing, not a convenience overload in `CasRefProtocol`: production +/// callers obtain `entry` from their frozen `RefPlan::catalogCut` and sample `_ckpt` in the same plan. +/// The API under test receives those exact values and performs no catalog or checkpoint resolution itself. +RecoveredRefTable recoverFromCurrentCatalogCut(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(backend, layout); + std::optional entry; + for (const CatalogEntry & candidate : cut.catalog.entries) + { + if (candidate.ns == ns) + { + entry = candidate; + break; + } + } + std::optional checkpoint; + if (entry) + { + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + if (const std::optional sample = readCkpt(backend, layout, life)) + checkpoint = sample->ckpt; + } + return recoverRefTableDetailedFromAuthority(backend, layout, entry, checkpoint); +} + +CatalogEntry catalog(NsState state) +{ + return CatalogEntry{.ns = RootNamespace{"srv1/recovery_grounding"}, .state = state, .incarnation = 1}; +} + +RefCkpt ckpt(uint64_t life_epoch, std::optional committed_through, + std::optional checkpoint_snapshot_id = std::nullopt, + std::optional last_epoch_seal = std::nullopt) +{ + return RefCkpt{.life_epoch = life_epoch, + .committed_through = committed_through, + .checkpoint_snapshot_id = checkpoint_snapshot_id, + .last_epoch_seal = last_epoch_seal}; +} + +void expectCode(const std::function & f, int code) +{ + try + { + f(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), code); + } +} + +TEST(CASRecoveryGrounding, CreatingAndAbsentCatalogEntriesAreNotRecovered) +{ + expectCode([&] { chooseRecoveryGrounding(catalog(NsState::Creating), ckpt(7, RefTxnId{7, 3})); }, + DB::ErrorCodes::INVALID_STATE); + expectCode([&] { chooseRecoveryGrounding(std::nullopt, ckpt(7, RefTxnId{7, 3})); }, + DB::ErrorCodes::INVALID_STATE); +} + +TEST(CASRecoveryGrounding, LiveAndRemovingRequireCheckpointAndLifeEpoch) +{ + expectCode([&] { chooseRecoveryGrounding(catalog(NsState::Live), std::nullopt); }, + DB::ErrorCodes::CORRUPTED_DATA); + expectCode([&] { chooseRecoveryGrounding(catalog(NsState::Removing), RefCkpt{}); }, + DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, MissingFrontierMeansNoCommittedTransaction) +{ + const RecoveryGrounding grounding = chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, std::nullopt)); + EXPECT_FALSE(grounding.base); + EXPECT_FALSE(grounding.committed_through); +} + +TEST(CASRecoveryGrounding, ChoosesCheckpointBaseAndArithmeticWalkStart) +{ + const RecoveryGrounding grounding = chooseRecoveryGrounding( + catalog(NsState::Live), ckpt(7, RefTxnId{7, 8}, RefTxnId{7, 4})); + EXPECT_EQ(grounding.base, (RefTxnId{7, 4})); + EXPECT_EQ(grounding.walk_from, (RefTxnId{7, 5})); + EXPECT_EQ(grounding.committed_through, (RefTxnId{7, 8})); +} + +TEST(CASRecoveryGrounding, BaseAtFrontierStillStartsAtItsExactSuccessor) +{ + /// A writer recovery probes exactly this slot for its sole possible unfrontiered successor. The + /// grounding contract must supply the arithmetic start even when the committed replay tail is empty. + const RecoveryGrounding grounding = chooseRecoveryGrounding( + catalog(NsState::Live), ckpt(7, RefTxnId{7, 8}, RefTxnId{7, 8})); + + EXPECT_EQ(grounding.base, (RefTxnId{7, 8})); + EXPECT_EQ(grounding.walk_from, (RefTxnId{7, 9})); + EXPECT_EQ(grounding.committed_through, (RefTxnId{7, 8})); +} + +TEST(CASRecoveryGrounding, WalksFromLifeEpochWithoutCheckpointBase) +{ + const RecoveryGrounding grounding = chooseRecoveryGrounding(catalog(NsState::Removing), ckpt(9, RefTxnId{9, 3})); + EXPECT_EQ(grounding.walk_from, (RefTxnId{9, 1})); +} + +TEST(CASRecoveryGrounding, RejectsBaseWithoutARepresentableSuccessor) +{ + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), + ckpt(7, RefTxnId{8, 1}, RefTxnId{7, std::numeric_limits::max()}, RefTxnId{8, 1})); + }, DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, RejectsCheckpointFieldsAboveCommittedFrontier) +{ + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, RefTxnId{7, 3}, RefTxnId{7, 4})); + }, DB::ErrorCodes::CORRUPTED_DATA); + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, RefTxnId{7, 3}, std::nullopt, RefTxnId{7, 4})); + }, DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, RejectsIncoherentEpochBoundaryInCheckpointAuthority) +{ + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, RefTxnId{10, 1}, std::nullopt, RefTxnId{7, 9})); + }, DB::ErrorCodes::CORRUPTED_DATA); + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, RefTxnId{8, 5}, std::nullopt, RefTxnId{8, 1})); + }, DB::ErrorCodes::CORRUPTED_DATA); + expectCode([&] + { + chooseRecoveryGrounding(catalog(NsState::Live), ckpt(7, RefTxnId{8, 1})); + }, DB::ErrorCodes::CORRUPTED_DATA); +} + +/// A life starts in its own writer epoch. Letting it start after the checkpoint's writer epoch makes +/// `walk_from > committed_through`, so recovery silently returns an empty table instead of refusing the +/// impossible authority. The codec and pure grounding entry point must reject the same sabotage. +TEST(CASRecoveryGrounding, RejectsLifeEpochAboveCommittedFrontierOnDecodeAndGrounding) +{ + const RefCkpt invalid = ckpt(2, RefTxnId{1, 5}); + String encoded = encodeRefCkpt(ckpt(1, RefTxnId{1, 5})); + const size_t life_epoch = encoded.find(R"("le":"1")"); + ASSERT_NE(life_epoch, String::npos); + encoded.replace(life_epoch, String{R"("le":"1")"}.size(), R"("le":"2")"); + + expectCode([&] { (void)decodeRefCkpt(encoded); }, DB::ErrorCodes::CORRUPTED_DATA); + expectCode([&] { (void)chooseRecoveryGrounding(catalog(NsState::Live), invalid); }, + DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, RecoveryIsEquivalentUnderFullEmptyPartialAndReorderedList) +{ + struct Observation + { + std::map committed; + RefTxnId greatest_applied; + std::optional last_epoch_seal; + RefTxnId next_id; + uint64_t log_gets = 0; + uint64_t snapshot_gets = 0; + uint64_t list_calls = 0; + }; + + std::vector observations; + for (const ListingMode mode : {ListingMode::Full, ListingMode::Empty, ListingMode::Partial, ListingMode::Reordered}) + { + auto backend = std::make_shared(mode); + const Layout layout("p"); + const RootNamespace ns{"srv1/list_equivalence"}; + const RefTxnId frontier{2, 1}; + seedAuthoritativeStream(*backend, layout, ns, frontier); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + backend->resetCounts(); + backend->list_calls = 0; + + const RecoveredRefTable recovered = recoverFromCurrentCatalogCut(*backend, layout, ns); + const uint64_t log_gets = backend->getCount(layout.refLogKey(life, {1, 1})) + + backend->getCount(layout.refLogKey(life, {1, 2})) + + backend->getCount(layout.refLogKey(life, {2, 1})); + const uint64_t snapshot_gets = backend->getCount(layout.refSnapshotKey(life, {1, 1})); + observations.push_back(Observation{ + .committed = committedOf(recovered.state), + .greatest_applied = recovered.state.getGreatestApplied(), + .last_epoch_seal = recovered.last_epoch_seal, + .next_id = recovered.state.nextTxnId(/*live_epoch=*/3), + .log_gets = log_gets, + .snapshot_gets = snapshot_gets, + .list_calls = backend->list_calls}); + } + + ASSERT_EQ(observations.size(), 4u); + for (size_t i = 1; i < observations.size(); ++i) + { + EXPECT_EQ(observations[i].committed, observations[0].committed); + EXPECT_EQ(observations[i].greatest_applied, observations[0].greatest_applied); + EXPECT_EQ(observations[i].last_epoch_seal, observations[0].last_epoch_seal); + EXPECT_EQ(observations[i].next_id, observations[0].next_id); + } + for (const Observation & observation : observations) + EXPECT_EQ(observation.log_gets, 3u) + << "recovery must fetch every exact log in the checkpoint-bounded frontier"; + for (const Observation & observation : observations) + EXPECT_EQ(observation.snapshot_gets, 0u) + << "a snapshot not named by `_ckpt` is not a recovery base"; + for (const Observation & observation : observations) + EXPECT_EQ(observation.list_calls, 0u) + << "recovery must not enumerate a stream whose exact checkpoint already supplies its base and frontier"; +} + +TEST(CASRecoveryGrounding, CatalogLifecycleAndCheckpointAreMandatoryForReadOnlyRecovery) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/mandatory_authority"}; + + { + auto backend = std::make_shared(ListingMode::Full); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, {namespaceBirthOp()})); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); + } + { + auto backend = std::make_shared(ListingMode::Full); + CasRefCatalog::casAdmitEntry( + *backend, layout, 1, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = 8}); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); + } + { + auto backend = std::make_shared(ListingMode::Full); + const CatalogEntry live{.ns = ns, .state = NsState::Live, .incarnation = 9}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, live); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(live.ns, live.incarnation); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), "not a sealed checkpoint").outcome, + PutOutcome::Done); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); + } + { + auto backend = std::make_shared(ListingMode::Full); + CatalogEntry creating{.ns = ns, .state = NsState::Creating, .incarnation = 7, + .creator = CreatorFence{"srv1", 1, 1}}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + backend->putIfAbsent(layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation)), + encodeRefCkpt(RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::INVALID_STATE); + } + { + auto backend = std::make_shared(ListingMode::Full); + backend->putIfAbsent(layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), + encodeRefCkpt(RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::INVALID_STATE); + } +} + +TEST(CASRecoveryGrounding, NonrecoverableAuthorityPerformsNoBackendRecoveryIo) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/nonrecoverable_authority"}; + const RefCkpt valid_ckpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}; + + { + auto backend = std::make_shared(ListingMode::Full); + backend->resetCounts(); + const CatalogEntry creating{ + .ns = ns, .state = NsState::Creating, .incarnation = 1, .creator = CreatorFence{"srv1", 1, 1}}; + expectCode( + [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, creating, valid_ckpt); }, + DB::ErrorCodes::INVALID_STATE); + EXPECT_EQ(backend->list_calls, 0u); + EXPECT_EQ(backend->getTotal(), 0u); + } + { + auto backend = std::make_shared(ListingMode::Full); + backend->resetCounts(); + const CatalogEntry live{.ns = ns, .state = NsState::Live, .incarnation = 2}; + expectCode( + [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, live, std::nullopt); }, + DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(backend->list_calls, 0u); + EXPECT_EQ(backend->getTotal(), 0u); + } + { + auto backend = std::make_shared(ListingMode::Full); + backend->resetCounts(); + expectCode( + [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, std::nullopt, valid_ckpt); }, + DB::ErrorCodes::INVALID_STATE); + EXPECT_EQ(backend->list_calls, 0u); + EXPECT_EQ(backend->getTotal(), 0u); + } +} + +TEST(CASRecoveryGrounding, ReadOnlyRecoveryNeverAdoptsFPlusOne) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RootNamespace ns{"srv1/read_only_excludes_f_plus_one"}; + seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 1}, /*include_f_plus_one=*/true); + + const RecoveredRefTable recovered = recoverFromCurrentCatalogCut(*backend, layout, ns); + EXPECT_EQ(recovered.state.getGreatestApplied(), (RefTxnId{1, 1})); + EXPECT_TRUE(recovered.state.getCommitted().contains("a")); + EXPECT_FALSE(recovered.state.getCommitted().contains("uncommitted")); +} + +/// A well-formed snapshot can describe a real but uncommitted transaction. If recovery merely treated +/// `LIST` as a performance hint, it could still select this false base and skip the exact first log. +/// The checkpoint names no snapshot, so every listing behaviour must leave the forged object unread. +TEST(CASRecoveryGrounding, ForgedWellFormedListedSnapshotIsUnobservedAndRecoveryDoesNotList) +{ + for (const ListingMode mode : {ListingMode::Full, ListingMode::Empty, ListingMode::Partial, ListingMode::Reordered}) + { + auto backend = std::make_shared(mode); + const Layout layout("p"); + const RootNamespace ns{"srv1/forged_listed_snapshot"}; + seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 1}, /*include_f_plus_one=*/true); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + + RefTableState forged_state; + std::vector birth{namespaceBirthOp()}; + const auto first_publish = publishCommittedOps("a", ManifestRef{1, 1, 1}); + birth.insert(birth.end(), first_publish.begin(), first_publish.end()); + applyRefLogTxn(forged_state, txn(ns, {1, 1}, std::move(birth))); + applyRefLogTxn(forged_state, txn(ns, {1, 2}, publishCommittedOps("uncommitted", ManifestRef{1, 2, 1}))); + writeRefSnapshotRaw(*backend, layout, snapshotOf(forged_state, ns.string())); + const String forged_key = layout.refSnapshotKey(life, {1, 2}); + + backend->resetCounts(); + backend->list_calls = 0; + const RecoveredRefTable recovered = recoverFromCurrentCatalogCut(*backend, layout, ns); + + EXPECT_EQ(backend->list_calls, 0u); + EXPECT_EQ(backend->getCount(forged_key), 0u); + EXPECT_EQ(recovered.state.getGreatestApplied(), (RefTxnId{1, 1})); + EXPECT_TRUE(recovered.state.getCommitted().contains("a")); + EXPECT_FALSE(recovered.state.getCommitted().contains("uncommitted")); + } +} + +/// A checkpoint-named snapshot is immutable lifecycle authority, not a list candidate. Its exact GET +/// and semantic decode must therefore fail closed rather than falling back to replaying the same log. +TEST(CASRecoveryGrounding, SemanticallyMalformedCheckpointSnapshotIsCorruptionAfterExactRead) +{ + auto backend = std::make_shared(ListingMode::Empty); + const Layout layout("p"); + const RootNamespace ns{"srv1/semantically_malformed_checkpoint"}; + const ManifestRef manifest{1, 1, 1}; + std::vector ops{namespaceBirthOp()}; + const auto publish = publishCommittedOps("committed", manifest); + ops.insert(ops.end(), publish.begin(), publish.end()); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, std::move(ops))); + + RefTableSnapshot malformed = minimalLiveSnapshot( + ns.string(), {1, 1}, {DB::Cas::tests::committedRow("committed", manifest)}); + malformed.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "precommit", manifest}); + writeRefSnapshotRaw(*backend, layout, malformed); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String snapshot_key = layout.refSnapshotKey(life, {1, 1}); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = RefTxnId{1, 1}, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + backend->resetCounts(); + try + { + (void)recoverFromCurrentCatalogCut(*backend, layout, ns); + FAIL() << "expected checkpoint-named malformed snapshot to fail closed"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_NE(String(e.message()).find("stateFromSnapshot"), String::npos); + } + EXPECT_EQ(backend->getCount(snapshot_key), 1u) + << "the corruption must come from the checkpoint snapshot's exact decode"; +} + +TEST(CASRecoveryGrounding, CheckpointSnapshotEqualToLastEpochSealIsRejectedBeforeReadingItsLog) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RootNamespace ns{"srv1/checkpoint_base_seal"}; + /// The checkpoint directly contradicts itself: its sole snapshot base names its terminal seal. + seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 2}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + + RefTableState through_seal; + std::vector birth{namespaceBirthOp()}; + const auto first_publish = publishCommittedOps("a", ManifestRef{1, 1, 1}); + birth.insert(birth.end(), first_publish.begin(), first_publish.end()); + applyRefLogTxn(through_seal, txn(ns, {1, 1}, std::move(birth))); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + applyRefLogTxn(through_seal, txn(ns, {1, 2}, {std::move(seal)})); + writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); + + const CkptSample before = *readCkpt(*backend, layout, life); + const RefCkpt with_sealed_base{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = RefTxnId{1, 2}}; + ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(with_sealed_base), before.token).outcome, + CasOutcome::Committed); + + backend->resetCounts(); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, {1, 2})), 0u) + << "the contradictory checkpoint metadata is rejected before any matching-log read"; + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, {1, 2})), 0u) + << "the seal-kind witness must be checked before reading the forged same-id snapshot"; +} + +/// An `EpochSeal` terminates its numeric epoch. A checkpoint frontier one sequence later in that +/// same epoch is not an empty tail: no record can occupy that slot. Recovery must diagnose the +/// malformed authority instead of advancing to `{E+1,1}` and terminating because that id sorts above +/// the bogus same-epoch frontier. +TEST(CASRecoveryGrounding, SameEpochFrontierAfterDecodedEpochSealIsCorruption) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RootNamespace ns{"srv1/frontier_after_seal"}; + + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, {namespaceBirthOp()})); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 2}, {std::move(seal)})); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + String malformed_ckpt = encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}}); + const size_t frontier_sequence = malformed_ckpt.find(R"("cts":"2")"); + ASSERT_NE(frontier_sequence, String::npos); + malformed_ckpt.replace(frontier_sequence, String{R"("cts":"2")"}.size(), R"("cts":"3")"); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), malformed_ckpt).outcome, PutOutcome::Done); + + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, OlderCheckpointSnapshotAtSealIsCorruption) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RootNamespace ns{"srv1/older_checkpoint_base_seal"}; + seedAuthoritativeStream(*backend, layout, ns, RefTxnId{2, 1}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + + RefOp second_seal; + second_seal.kind = RefOpKind::EpochSeal; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {2, 2}, {std::move(second_seal)})); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, + txn(ns, {3, 1}, publishCommittedOps("c", ManifestRef{3, 1, 1}), RefTxnId{2, 2})); + + RefTableState through_first_seal; + std::vector birth{namespaceBirthOp()}; + const auto first_publish = publishCommittedOps("a", ManifestRef{1, 1, 1}); + birth.insert(birth.end(), first_publish.begin(), first_publish.end()); + applyRefLogTxn(through_first_seal, txn(ns, {1, 1}, std::move(birth))); + RefOp first_seal; + first_seal.kind = RefOpKind::EpochSeal; + applyRefLogTxn(through_first_seal, txn(ns, {1, 2}, {std::move(first_seal)})); + writeRefSnapshotRaw(*backend, layout, snapshotOf(through_first_seal, ns.string())); + + const CkptSample before = *readCkpt(*backend, layout, life); + const RefCkpt with_old_sealed_base{ + .life_epoch = 1, + .committed_through = RefTxnId{3, 1}, + .checkpoint_snapshot_id = RefTxnId{1, 2}, + .last_epoch_seal = RefTxnId{2, 2}}; + ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(with_old_sealed_base), before.token).outcome, + CasOutcome::Committed); + + backend->resetCounts(); + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, {1, 2})), 1u) + << "the old seal differs from `last_epoch_seal`, so only the matching-log proof can reject it"; + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, {1, 2})), 0u) + << "the old seal must be rejected before the forged same-id snapshot is read"; +} + +TEST(CASRecoveryGrounding, TerminalGapBelowFrontierIsCorruptionNotARebirth) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RootNamespace ns{"srv1/terminal_gap"}; + const RefLogTxn birth = txn(ns, {1, 1}, {namespaceBirthOp()}); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, birth); + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 2}, {std::move(remove)})); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {2, 1}, {namespaceBirthOp()})); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + String malformed_ckpt = encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + const size_t frontier_epoch = malformed_ckpt.find(R"("cte":"1")"); + ASSERT_NE(frontier_epoch, String::npos); + malformed_ckpt.replace(frontier_epoch, String{R"("cte":"1")"}.size(), R"("cte":"2")"); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), malformed_ckpt).outcome, PutOutcome::Done); + + expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); +} + +TEST(CASRecoveryGrounding, LaterEpochCheckpointBaseRequiresItsContextualBacklink) +{ + auto backend = std::make_shared(ListingMode::Full); + const Layout layout("p"); + const RefTxnId seal_id{1, 2}; + const RefTxnId base_id{2, 1}; + + const auto expect_rejected = [&](const RootNamespace & ns, std::optional backlink) + { + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, {namespaceBirthOp()})); + RefOp seal; + seal.kind = RefOpKind::EpochSeal; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, seal_id, {std::move(seal)})); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, base_id, {}, backlink)); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = base_id, + .checkpoint_snapshot_id = base_id, + .last_epoch_seal = seal_id})).outcome, PutOutcome::Done); + + expectCode( + [&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, + DB::ErrorCodes::CORRUPTED_DATA); + }; + + expect_rejected(RootNamespace{"srv1/base_missing_backlink"}, std::nullopt); + expect_rejected(RootNamespace{"srv1/base_wrong_backlink"}, RefTxnId{1, 99}); + + const auto expect_predecessor_rejected = [&](const RootNamespace & ns, bool write_ordinary_predecessor) + { + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, {namespaceBirthOp()})); + if (write_ordinary_predecessor) + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, seal_id, {})); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, base_id, {}, seal_id)); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = base_id, + .checkpoint_snapshot_id = base_id, + .last_epoch_seal = seal_id})).outcome, PutOutcome::Done); + + expectCode( + [&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, + DB::ErrorCodes::CORRUPTED_DATA); + }; + + expect_predecessor_rejected(RootNamespace{"srv1/base_predecessor_absent"}, false); + expect_predecessor_rejected(RootNamespace{"srv1/base_predecessor_not_seal"}, true); +} + +} diff --git a/src/Disks/tests/gtest_cas_recovery_streaming.cpp b/src/Disks/tests/gtest_cas_recovery_streaming.cpp new file mode 100644 index 000000000000..ad2dbbcb9342 --- /dev/null +++ b/src/Disks/tests/gtest_cas_recovery_streaming.cpp @@ -0,0 +1,647 @@ +#include + +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include + +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int S3_ERROR; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +/// A deterministic accountant for the streaming-recovery memory probe: each recovery loop reports +/// `+footprint` while one decoded transaction is resident and `-footprint` once it is discarded, so +/// `peak` is the maximum summed decoded-transaction footprint ever resident at one instant. Streaming +/// holds one transaction; the retired whole-tail materialiser -- and the test-local control that stands +/// in for it -- held the entire tail. Deterministic (it accounts the footprints the probe is handed, a +/// pure function of decoded content, not RSS), so it is stable under ASan quarantine noise. +struct PeakTracker +{ + std::atomic alive_bytes{0}; + std::atomic peak_bytes{0}; + + std::function probe() + { + return [this](int64_t delta) + { + const int64_t now = alive_bytes.fetch_add(delta, std::memory_order_relaxed) + delta; + int64_t prev = peak_bytes.load(std::memory_order_relaxed); + while (now > prev && !peak_bytes.compare_exchange_weak(prev, now, std::memory_order_relaxed)) + { + } + }; + } + + int64_t peak() const { return peak_bytes.load(std::memory_order_relaxed); } + int64_t alive() const { return alive_bytes.load(std::memory_order_relaxed); } +}; + +/// A distinct manifest per call: `build_sequence` carries the identity so every generated +/// `(ref_name, manifest_ref)` add-precommit is a legal transition (no manifest is owned twice). +ManifestRef mref(uint64_t seq) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = 1}; +} + +/// One maximum-shaped ref-log transaction: `num_ops` add-precommit ops over distinct +/// `(ref_name, manifest_ref)` pairs (plus a leading `namespace_birth` for the first transaction of a +/// never-born table). Each pair is unique across the whole tail (the running `manifest_seq`), so the +/// tail replays cleanly and the candidate state simply grows -- the point is a large decoded body per +/// transaction, which is what makes the whole-tail vector's resident footprint N times a single +/// transaction's. +RefLogTxn makeBigTxn(const String & ns, RefTxnId id, size_t num_ops, uint64_t & manifest_seq, bool birth) +{ + RefLogTxn txn; + txn.ns = ns; + txn.txn_id = id; + if (birth) + txn.ops.push_back(namespaceBirthOp()); + for (size_t i = 0; i < num_ops; ++i) + { + const String ref_name = "rs_" + std::to_string(id.ref_sequence) + "_" + std::to_string(i); + txn.ops.push_back(ownerTransitionOp( + std::nullopt, RefOwnerBinding{RefOwnerKind::Precommit, ref_name, mref(manifest_seq)})); + ++manifest_seq; + } + return txn; +} + +/// Seed `num_txns` maximum-shaped transactions at ids {1,1}..{1,num_txns} directly into `ns`'s `_log/` +/// stream, and return the resident DECODED footprint (`decodedRefLogTxnFootprint`) of the largest single +/// transaction plus the total across all. The largest single footprint is the streaming peak (one +/// transaction resident at a time); the total is what a whole-tail materialiser holds resident at once. +/// Footprint -- not the compressed stored size -- is the bound's currency: it is what actually sits in +/// memory and what a materialising regression accumulates N-fold, and it is a deterministic function of +/// the decoded content (identical whether computed on the built or the decoded transaction). +struct SeededTail +{ + uint64_t max_single_footprint = 0; + uint64_t total_footprint = 0; +}; + +SeededTail seedBigTail( + InMemoryBackend & backend, const Layout & layout, const RootNamespace & ns, + size_t num_txns, size_t ops_per_txn, uint64_t & manifest_seq) +{ + SeededTail seeded; + for (size_t t = 0; t < num_txns; ++t) + { + const RefLogTxn txn = makeBigTxn(ns.string(), RefTxnId{1, t + 1}, ops_per_txn, manifest_seq, /*birth=*/t == 0); + const uint64_t footprint = decodedRefLogTxnFootprint(txn); + seeded.max_single_footprint = std::max(seeded.max_single_footprint, footprint); + seeded.total_footprint += footprint; + fixture::writeRefLogRaw(backend, layout, txn); + } + /// This helper always builds a recoverable `Live` life. Tests that need the distinct missing- + /// checkpoint corruption shape use the lower-level raw writers directly instead. + writeRecoverableCkptForRawFixture( + backend, layout, ns, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, static_cast(num_txns)}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + return seeded; +} + +/// Bounded busy-wait on a test-observable predicate (the established `yield()`-poll idiom for recovery +/// waiters, see `CasPool::refRecoveryWaitersForTest`). Bound is a generous wall-clock ceiling that only +/// trips on a genuine hang, never in the normal fast path; returns false on timeout so the caller can +/// release any blocked threads before asserting. +template +bool pollUntil(Pred pred) +{ + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(30); + while (!pred()) + { + if (std::chrono::steady_clock::now() > deadline) + return false; + std::this_thread::yield(); + } + return true; +} + +/// Backend that drops one selected `_log/` object on its FIRST GET (a concurrent-cleanup vanish), +/// then serves it normally, and counts fresh (cursor-empty) LISTs of the ref prefix so a test can +/// prove a stable checkpoint verdict did not spin on the advisory listing. +class VanishMidTailOnceBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::get; /// keep the one-arg convenience overload visible past our override + + String target_log_key; + String refs_prefix; + std::atomic armed{false}; + std::atomic vanished{false}; + std::atomic fresh_list_count{0}; + + std::optional get(const String & key, Range range) override + { + if (armed.load() && key == target_log_key && !vanished.exchange(true)) + return std::nullopt; /// selected object gone between LIST and GET; recovery must re-LIST + return InMemoryBackend::get(key, range); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (armed.load() && prefix == refs_prefix && cursor.empty()) + fresh_list_count.fetch_add(1, std::memory_order_relaxed); + return InMemoryBackend::list(prefix, cursor, limit); + } +}; + +/// Backend that replaces one selected `_log/` object's body with a valid-but-foreign ref-log object +/// (a different namespace in the body): decoding it fails with CORRUPTED_DATA (body/key mismatch), the +/// durable-corruption class recovery must fail fast on -- no re-LIST loop. +class CorruptLogOnGetBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::get; /// keep the one-arg convenience overload visible past our override + + String target_log_key; + String corrupt_bytes; + String refs_prefix; + std::atomic armed{false}; + std::atomic refs_list_count{0}; + + std::optional get(const String & key, Range range) override + { + auto got = InMemoryBackend::get(key, range); + if (armed.load() && got && key == target_log_key) + got->bytes = corrupt_bytes; + return got; + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (armed.load() && prefix == refs_prefix && cursor.empty()) + refs_list_count.fetch_add(1, std::memory_order_relaxed); + return InMemoryBackend::list(prefix, cursor, limit); + } +}; + +/// Backend that blocks the first exact log GET while recovery holds no state lock. A concurrent second +/// caller can then reach `recovery_cv`, while the LIST counter proves neither caller enumerates the +/// recovery stream. +class BlockingFirstLogGetBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::get; + + String refs_prefix; + String target_log_key; + std::atomic armed{false}; + std::atomic blocked{false}; + std::atomic list_calls{0}; + std::function on_first_target_get; + + std::optional get(const String & key, Range range) override + { + if (armed.load() && key == target_log_key && !blocked.exchange(true)) + on_first_target_get(); + return InMemoryBackend::get(key, range); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (armed.load() && prefix == refs_prefix && cursor.empty()) + list_calls.fetch_add(1, std::memory_order_relaxed); + return InMemoryBackend::list(prefix, cursor, limit); + } +}; + +} + +/// Test 14 (load-bearing memory bound): a long tail of maximum-shaped transactions replays under a hard +/// peak bound (twice the largest single transaction's decoded footprint) that a whole-tail materialiser +/// -- which holds every decoded transaction resident at once -- provably exceeds. The bound is computed +/// from the fixture's own footprints and the whole-tail total is asserted to exceed it, so the bound is +/// a property of the fixture, not a lucky constant. Its materialising counterpart, +/// `MaterializingControlExceedsMemoryBound`, trips this same bound. +TEST(CASRecoveryStreaming, LongTailReplaysUnderMemoryBound) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + seedPoolMetaForRestart(*backend); + const RootNamespace ns{"00/aa@cas@"}; + + constexpr size_t kTxns = 24; + constexpr size_t kOpsPerTxn = 250; + uint64_t manifest_seq = 1; + const SeededTail seeded = seedBigTail(*backend, layout, ns, kTxns, kOpsPerTxn, manifest_seq); + + const uint64_t bound = 2 * seeded.max_single_footprint; + ASSERT_GT(seeded.total_footprint, bound) + << "fixture must make the whole tail (" << seeded.total_footprint + << " B) provably exceed the bound (" << bound << " B)"; + + PeakTracker tracker; + setRecoveryReplayMemoryProbeForTest(tracker.probe()); + SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); + + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + const RefTableState state = recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, catalog_cut, ns).state; + EXPECT_EQ(state.getPrecommits().size(), kTxns * kOpsPerTxn) << "the whole tail must have replayed"; + EXPECT_LE(tracker.peak(), static_cast(bound)) + << "streaming recovery must hold at most ~one decoded transaction (peak " << tracker.peak() + << " B) not the whole " << kTxns << "-transaction tail (" << seeded.total_footprint << " B)"; + /// Lower bound (the accountant's fail-close): the peak must reach at least one whole decoded + /// transaction's footprint. This couples the assertion to the production report calls -- delete them + /// and the peak collapses to zero, failing HERE instead of passing vacuously under the upper bound. + EXPECT_GE(tracker.peak(), static_cast(seeded.max_single_footprint)) + << "the probe must observe at least one whole decoded transaction resident (peak " << tracker.peak() + << " B, one transaction " << seeded.max_single_footprint + << " B) -- a zero peak means the production report calls were removed and the bound guards nothing"; + EXPECT_EQ(tracker.alive(), 0) << "every decoded transaction must be discarded after it is applied"; +} + +/// Test 14 (materialising RED control): the discriminating counterpart to the streaming bound above. A +/// control that GETs+decodes the WHOLE tail into a vector BEFORE applying it -- the retired whole-tail +/// shape -- holds every decoded transaction resident at once. Driven through the SAME memory probe as +/// streaming recovery, its peak must EXCEED the same bound the streaming path stays under. This is the +/// regression the memory guard exists to catch, and the guard discriminates precisely because the probe +/// now accounts the caller's whole resident set (each decoded transaction for the span it is held), not +/// one apply in isolation. Under the retired stored-byte-in-`applyOne` probe this control's peak stayed +/// at one transaction (see the RED capture in the round-2 fix report); it now correctly trips. +TEST(CASRecoveryStreaming, MaterializingControlExceedsMemoryBound) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + seedPoolMetaForRestart(*backend); + const RootNamespace ns{"00/aa@cas@"}; + + constexpr size_t kTxns = 24; + constexpr size_t kOpsPerTxn = 250; + uint64_t manifest_seq = 1; + const SeededTail seeded = seedBigTail(*backend, layout, ns, kTxns, kOpsPerTxn, manifest_seq); + + const uint64_t bound = 2 * seeded.max_single_footprint; + ASSERT_GT(seeded.total_footprint, bound) + << "fixture must make the whole tail (" << seeded.total_footprint + << " B) provably exceed the bound (" << bound << " B)"; + + PeakTracker tracker; + setRecoveryReplayMemoryProbeForTest(tracker.probe()); + SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); + + /// Materialise the WHOLE tail first (retired shape): every decoded transaction stays resident in + /// `resident_txns`, and its footprint is reported to the probe up front, released only AFTER the whole + /// tail has been applied -- exactly the memory profile streaming recovery replaced. + std::vector resident_txns; + int64_t held = 0; + for (size_t t = 1; t <= kTxns; ++t) + { + const auto got = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, t})); + ASSERT_TRUE(got.has_value()); + RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), RefTxnId{1, t}); + const int64_t footprint = static_cast(decodedRefLogTxnFootprint(txn)); + reportReplayMemoryDelta(footprint); + held += footprint; + resident_txns.push_back(std::move(txn)); + } + + RefReplayBuilder builder(std::nullopt); + for (RefLogTxn & txn : resident_txns) + builder.applyOne(std::move(txn), 0); + const RecoveryResult result = std::move(builder).finish(); + EXPECT_EQ(result.state.getPrecommits().size(), kTxns * kOpsPerTxn) << "the whole tail must have applied"; + + EXPECT_GT(tracker.peak(), static_cast(bound)) + << "the materialising control holds the whole tail resident; the probe must exceed the " + "single-transaction bound (peak " << tracker.peak() << " B, bound " << bound << " B)"; + + reportReplayMemoryDelta(-held); /// release the whole tail + EXPECT_EQ(tracker.alive(), 0); +} + +/// Test 14 (vanished-selected-object leg): once the exact `_ckpt` commits a finite frontier, a missing +/// record inside it is not something a fresh LIST may reinterpret as a shorter stream. With the same +/// checkpoint token still durable, recovery fails closed immediately instead of accepting incomplete +/// state or spinning on an advisory enumeration. +TEST(CASRecoveryStreaming, MidTailVanishedObjectFailsClosedAgainstStableAuthority) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns{"00/aa@cas@"}; + + const uint64_t seq1 = publishCommittedTransition(*backend, layout, ns, "a", std::nullopt, mref(1)); + const uint64_t seq2 = publishCommittedTransition(*backend, layout, ns, "b", std::nullopt, mref(2)); + const uint64_t seq3 = publishCommittedTransition(*backend, layout, ns, "c", std::nullopt, mref(3)); + ASSERT_LT(seq1, seq2); + ASSERT_LT(seq2, seq3); + /// Semantic publication already durably advances the exact checkpoint frontier to `seq3`. + + auto store = openPoolForTest(backend); + backend->refs_prefix = layout.namespaceStreamPrefix(fixture::fixtureLife(ns)); + backend->target_log_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, seq2}); /// vanish a mid-tail object + backend->armed = true; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); + EXPECT_TRUE(backend->vanished.load()) << "the selected committed object must actually have vanished"; + EXPECT_EQ(backend->fresh_list_count.load(), 0) + << "the exact checkpoint frontier makes recovery stream enumeration unnecessary"; +} + +/// Test 14 (durable-corruption leg): a `_log/` object whose body decodes to a foreign namespace is +/// durable corruption, not a transient vanish -- recovery discards the candidate and fails fast with +/// no re-LIST loop. Asserts the throw, zero restarts, and a single LIST. +TEST(CASRecoveryStreaming, CorruptObjectFailsFast) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns{"00/aa@cas@"}; + + publishCommittedTransition(*backend, layout, ns, "a", std::nullopt, mref(1)); + const uint64_t seq2 = publishCommittedTransition(*backend, layout, ns, "b", std::nullopt, mref(2)); + publishCommittedTransition(*backend, layout, ns, "c", std::nullopt, mref(3)); + /// Semantic publication already durably advances the exact checkpoint frontier. + + /// A structurally valid ref-log object for a DIFFERENT namespace: it decompresses and parses, but + /// its body namespace does not match the key, which `decodeRefLogTxn` rejects as CORRUPTED_DATA. + RefLogTxn foreign; + foreign.ns = "99/zz@cas@"; + foreign.txn_id = RefTxnId{1, seq2}; + foreign.ops = {namespaceBirthOp()}; + backend->corrupt_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(foreign)); + + auto store = openPoolForTest(backend); + backend->refs_prefix = layout.namespaceStreamPrefix(fixture::fixtureLife(ns)); + backend->target_log_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, seq2}); + backend->armed = true; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->resolveRef(ns, "a"); }); + /// The exact checkpoint frontier determines this GET. A corrupt committed object fails fast without + /// asking a stream enumeration to reinterpret the durable recovery boundary. + EXPECT_EQ(backend->refs_list_count.load(), 0) << "durable corruption must not trigger recovery LIST"; +} + +/// Test 14 (concurrent-waiter leg): while one caller is blocked in recovery's unlocked exact-log GET, +/// a second caller for the same table parks on `recovery_cv` and is woken exactly once when recovery +/// completes. Neither caller may race an independent stream LIST. +TEST(CASRecoveryStreaming, ConcurrentWaiterUnblockedOnce) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns{"00/aa@cas@"}; + + publishCommittedTransition(*backend, layout, ns, "x", std::nullopt, mref(1)); + publishCommittedTransition(*backend, layout, ns, "y", std::nullopt, mref(2)); + /// Semantic publication already durably advances the exact checkpoint frontier. + + auto store = openPoolForTest(backend); + backend->refs_prefix = layout.namespaceStreamPrefix(fixture::fixtureLife(ns)); + + /// Gate the first exact replay GET. The leader reaches it with `state_mutex` released, which is + /// the window in which a second caller must be able to park on `recovery_cv`. + std::atomic get_entered{false}; + std::promise entered_promise; + std::promise release_promise; + std::shared_future release_future = release_promise.get_future().share(); + backend->target_log_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 1}); + backend->on_first_target_get = [&] + { + if (!get_entered.exchange(true)) + entered_promise.set_value(); + release_future.wait(); + }; + backend->armed = true; + + std::thread t1([&] { store->listRefs(ns); }); /// exact GET blocks with recovery unlocked + entered_promise.get_future().wait(); + + std::thread t2([&] { store->listRefs(ns); }); /// second caller must park on recovery_cv + const bool parked = pollUntil([&] { return store->refRecoveryWaitersForTest(ns) >= 1; }); + + release_promise.set_value(); + t1.join(); + t2.join(); + + EXPECT_TRUE(parked) << "the second caller must reach recovery_cv while the first is in the retry window"; + EXPECT_EQ(store->refRecoveryWaitersForTest(ns), 0u) << "no phantom waiter after recovery completes"; + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()); + EXPECT_TRUE(store->resolveRef(ns, "y").has_value()); + EXPECT_EQ(backend->list_calls.load(), 0) + << "the leader and parked waiter must both recover without stream enumeration"; +} + +/// Test 14 (other materializers leg): the orphan-sweep recovery (`recoverRefTableDetailedFromAuthority`) +/// and fsck's exact-authority recovery stream through the SAME builder and hold under the SAME +/// per-transaction bound as primary recovery. +TEST(CASRecoveryStreaming, OrphanSweepAndFsckSameBound) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns_sweep{"00/sweep@cas@"}; + const RootNamespace ns_fsck{"00/fsck@cas@"}; + + constexpr size_t kTxns = 16; + constexpr size_t kOpsPerTxn = 250; + uint64_t manifest_seq = 1; + const SeededTail sweep_tail = seedBigTail(*backend, layout, ns_sweep, kTxns, kOpsPerTxn, manifest_seq); + const SeededTail fsck_tail = seedBigTail(*backend, layout, ns_fsck, kTxns, kOpsPerTxn, manifest_seq); + + const uint64_t max_single = std::max(sweep_tail.max_single_footprint, fsck_tail.max_single_footprint); + const uint64_t bound = 2 * max_single; + ASSERT_GT(sweep_tail.total_footprint, bound); + ASSERT_GT(fsck_tail.total_footprint, bound); + + auto store = openPoolForTest(backend); + + { + PeakTracker tracker; + setRecoveryReplayMemoryProbeForTest(tracker.probe()); + SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); + const CasRefCatalog::Snapshot sweep_catalog_cut = CasRefCatalog::read(*backend, layout); + const RecoveredRefTable recovered = + recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, sweep_catalog_cut, ns_sweep); + EXPECT_EQ(recovered.state.getPrecommits().size(), kTxns * kOpsPerTxn); + EXPECT_LE(tracker.peak(), static_cast(bound)) + << "orphan-sweep recovery must stream: peak " << tracker.peak() << " B, bound " << bound << " B"; + EXPECT_GE(tracker.peak(), static_cast(sweep_tail.max_single_footprint)) + << "the probe must observe at least one decoded sweep transaction (peak " << tracker.peak() + << " B) -- a zero peak means the orphan-sweep report call was silently removed"; + EXPECT_EQ(tracker.alive(), 0); + } + + { + PeakTracker tracker; + setRecoveryReplayMemoryProbeForTest(tracker.probe()); + SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); + const FsckReport report = runFsck(*store, /*detail=*/true); + EXPECT_TRUE(report.clean()); + EXPECT_LE(tracker.peak(), static_cast(bound)) + << "fsck exact-authority recovery must stream: peak " << tracker.peak() << " B, bound " << bound << " B"; + EXPECT_GE(tracker.peak(), static_cast(fsck_tail.max_single_footprint)) + << "the probe must observe at least one decoded fsck-recovery transaction (peak " << tracker.peak() + << " B) -- a zero peak means fsck stopped recovering catalog-authoritative namespaces"; + EXPECT_EQ(tracker.alive(), 0); + } +} + +/// Test 14 (writer-ledger leg -- the production recovery path): the writer ledger's OWN recovery loop +/// (`CasRefLedger::ensureRefTableRecovered`, reached through any Pool touch) must stream the tail under +/// the SAME per-transaction bound the free recovery does. This is the exact production path the original +/// memory finding named; `LongTailReplaysUnderMemoryBound` above exercises the free authoritative recovery, +/// NOT the ledger loop, so the ledger could regress to whole-tail materialisation while every other +/// bound stayed green. Recovery is driven through the production non-minting namespace-file read path, +/// which does NOT dispatch the stale-precommit sweep `listRefs` would (that sweep +/// would append removals over the seeded epoch-1 precommit bindings and perturb both the count and the +/// probe). The whole tail sits above a never-born base, so the retained tail count equals the whole tail. +TEST(CASRecoveryStreaming, LedgerRecoveryReplaysUnderMemoryBound) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns{"00/ledger@cas@"}; + + constexpr size_t kTxns = 24; + constexpr size_t kOpsPerTxn = 250; + uint64_t manifest_seq = 1; + const SeededTail seeded = seedBigTail(*backend, layout, ns, kTxns, kOpsPerTxn, manifest_seq); + + const uint64_t bound = 2 * seeded.max_single_footprint; + ASSERT_GT(seeded.total_footprint, bound) + << "fixture must make the whole tail (" << seeded.total_footprint + << " B) provably exceed the bound (" << bound << " B)"; + + auto store = openPoolForTest(backend); + + PeakTracker tracker; + setRecoveryReplayMemoryProbeForTest(tracker.probe()); + SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); + + /// The production Task 4b reader drives `CasRefLedger::ensureRefTableRecovered` without the + /// stale-precommit sweep; the resident-only observer then reads the retained tail count. + ASSERT_TRUE(store->namespaceFilesLifeIfReadable(ns)); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), kTxns) + << "the whole tail must have replayed through the ledger's own recovery loop"; + EXPECT_LE(tracker.peak(), static_cast(bound)) + << "ledger recovery must hold at most ~one decoded transaction (peak " << tracker.peak() + << " B) not the whole " << kTxns << "-transaction tail (" << seeded.total_footprint << " B)"; + /// Lower bound (the accountant's fail-close): a zero peak means the ledger loop's production report + /// call was removed and this bound would guard nothing -- the exact silent-decoupling this leg exists + /// to catch on the production path. + EXPECT_GE(tracker.peak(), static_cast(seeded.max_single_footprint)) + << "the probe must observe at least one whole decoded transaction resident (peak " << tracker.peak() + << " B, one transaction " << seeded.max_single_footprint + << " B) -- a zero peak means the ledger's production report call was removed and the bound guards nothing"; + EXPECT_EQ(tracker.alive(), 0) << "every decoded transaction must be discarded after it is applied"; +} + +/// Test 15 (publication inventory): after streaming recovery of a table with a non-trivial snapshot +/// base, precommit bindings, and a tail of committed transactions, EVERY field the +/// recovery publication seeds is asserted -- not just the two a prose inventory would keep. This is a +/// regression guard: streaming recovery must install exactly what the whole-tail recovery installed. +TEST(CASRecoveryStreaming, RecoveryResultInventoryComplete) +{ + auto backend = std::make_shared(); + seedPoolMetaForRestart(*backend); + const Layout layout("p"); + const RootNamespace ns{"00/inv@cas@"}; + + /// A non-trivial base snapshot: two committed rows plus a stale predecessor precommit binding. + RefTableSnapshot base; + base.ns = ns.string(); + base.snapshot_id = RefTxnId{1, 5}; + base.committed = {committedRow("c_one", mref(11)), committedRow("c_two", mref(12))}; + base.precommits = {RefOwnerBinding{RefOwnerKind::Precommit, "p_stale", mref(13)}}; + RefLogTxn base_txn; + base_txn.ns = ns.string(); + base_txn.txn_id = base.snapshot_id; + base_txn.ops = publishCommittedOps("c_two", mref(12)); + fixture::writeRefLogRaw(*backend, layout, base_txn); + writeRefSnapshotRaw(*backend, layout, base); + const auto base_got = backend->get(layout.refSnapshotKey(fixture::fixtureLife(ns), base.snapshot_id)); + ASSERT_TRUE(base_got.has_value()); + const uint64_t base_stored_bytes = base_got->bytes.size(); + + /// Two committed transactions strictly above the base -- the tail. + RefLogTxn t6; + t6.ns = ns.string(); + t6.txn_id = RefTxnId{1, 6}; + t6.ops = publishCommittedOps("c_three", mref(21)); + fixture::writeRefLogRaw(*backend, layout, t6); + RefLogTxn t7; + t7.ns = ns.string(); + t7.txn_id = RefTxnId{1, 7}; + t7.ops = publishCommittedOps("c_four", mref(22)); + fixture::writeRefLogRaw(*backend, layout, t7); + + writeRecoverableCkptForRawFixture( + *backend, layout, ns, RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 7}, + .checkpoint_snapshot_id = RefTxnId{1, 5}, + .last_epoch_seal = std::nullopt}); + + const uint64_t tail6 = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 6}))->bytes.size(); + const uint64_t tail7 = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 7}))->bytes.size(); + + backend->resetCounts(); + auto store = openPoolForTest(backend); + + /// Drive recovery via the production namespace-file reader WITHOUT the stale-precommit sweep that + /// `resolveRef`/`listRefs` dispatch (that + /// sweep would clear `needs_stale_precommit_sweep` before it could be observed). Every inventory + /// field below is then read straight off the seeded runtime, and the read-path state assertion is + /// left for LAST (after `needs_stale_precommit_sweep` has been observed). + + ASSERT_TRUE(store->namespaceFilesLifeIfReadable(ns)); + const NamespaceLifeId life = fixture::fixtureLife(ns); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, base.snapshot_id)), 1u) + << "recovery must validate the selected base's matching ordinary log"; + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, base.snapshot_id)), 1u) + << "the inventory must come from the selected snapshot, not a pre-snapshot failure"; + + /// newest snapshot identity: the recovered base id, no seal on this clean mount. + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::optional(base.snapshot_id)); + + /// last_epoch_seal: this mount's live epoch is the one the seeded stream was written in, so the + /// CAS-walk crossed no epoch transition and installed no chain link. Pins that the field IS part of + /// the published inventory (its non-empty counterpart lives in + /// `CASRefRecoveryCasWalk.DeadEpochIsClosedByOurOwnSealAtTPlusOne`). + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt); + + /// stale-precommit sweep: recovery always arms it (asserted BEFORE any read-side sweep runs). + EXPECT_TRUE(store->needsStalePrecommitSweepForTest(ns)); + + /// tail count / bytes: exactly the two transactions above the base and their stored sizes. + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u); + EXPECT_EQ(store->refTailBytesSinceSnapshotForTest(ns), tail6 + tail7); + + /// base snapshot bytes: the encoded body size of the recovered base snapshot. + EXPECT_EQ(store->refBaseSnapshotBytesForTest(ns), base_stored_bytes); + + /// admission budgets: the raw hard limits minus this table's wire overhead and the safety margin. + const uint64_t overhead = 4 + ns.string().size() + 4096; + const uint64_t expected_budget = 64ULL * 1024 * 1024 - overhead; + EXPECT_EQ(store->refSnapshotBudgetForTest(ns), expected_budget); + EXPECT_EQ(store->refRemovalBudgetForTest(ns), expected_budget); + + /// state: four committed rows (two from the base, two from the tail). This read dispatches the + /// read-side sweep, hence it comes last -- after the sweep flag has been observed above. + const auto refs = store->listRefs(ns); + EXPECT_EQ(refs.size(), 4u); + EXPECT_TRUE(store->resolveRef(ns, "c_one").has_value()); + EXPECT_TRUE(store->resolveRef(ns, "c_two").has_value()); + EXPECT_TRUE(store->resolveRef(ns, "c_three").has_value()); + EXPECT_TRUE(store->resolveRef(ns, "c_four").has_value()); +} diff --git a/src/Disks/tests/gtest_cas_ref_carve.cpp b/src/Disks/tests/gtest_cas_ref_carve.cpp new file mode 100644 index 000000000000..e12706d64f69 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_carve.cpp @@ -0,0 +1,366 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Task 7 (stage-1 §2): the ref-flush two-phase carve and the validation loop's publish ordering. +/// +/// Two exception-safety windows are pinned here, both driven through the `setCarveHookForTest` fault +/// seam (which fires `std::bad_alloc` at named carve/validation phase points): +/// +/// - The carve must PLAN (scan `pending` without popping; build the selection and every reservation) +/// and only then PUBLISH (pop + append under the same continuous `ref_queue_mutex` hold, using only +/// non-throwing moves/copies). A throw anywhere in the plan must leave the queue byte-for-byte intact +/// so no already-selected item is stranded (removed from `pending` yet never completed) and no waiter +/// hangs. The pre-fix carve interleaved pops with the allocating `seen_refs`/`batch` growth, so a +/// throw after the first pop stranded popped items and hung their waiters forever — the behavioural +/// signature this suite demonstrates. +/// - The per-item validation loop must reserve `final_ops`/`survivors` growth BEFORE applying the item +/// to `working`, and publish only past all throwing points. The pre-fix loop moved `working` before +/// those allocations, so a failure there left a failed item's effects in `working` and — when the +/// throw fell between the two accumulator writes — its ops already in the durably-committed +/// transaction while its own caller was told the append failed. +/// +/// The suite name is prefixed `RefWriter` so it is covered by the `RefWriter*` unit-test gate filter. + +using namespace DB::Cas; +using Phase = CasRefLedger::CarvePhaseForTest; + +namespace +{ + +PoolPtr openPool(const BackendPtr & backend) +{ + /// A fresh pool with no residue: `seedPoolMetaForRestart` is idempotent and a no-op here (mirrors the + /// ref-writer suite's `openPool`), it just lets `beginPartWrite` bootstrap over a valid `_pool_meta`. + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// A legal blob-free part: stage an empty manifest, precommit, promote — enough to leave one committed +/// ref in `ns` that a later `dropRef` can co-batch. Mirrors the ref-writer suite's `publishEmptyPart`. +/// +/// Stage B (Task 4-C): pin `ns` to the sentinel before the first real touch, mirroring the same fix in +/// `gtest_cas_ref_writer.cpp`'s `startBuildFor` and `gtest_cas_ref_chunked_flush.cpp`'s +/// `publishEmptyPart` -- every test in this file births its namespace here before any fault +/// injection/verification that separately computes a key via `DB::Cas::tests::fixture::fixtureLife(ns)`. +void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +/// Shared, heap-backed synchronisation state for one case. Heap-backing (captured by `shared_ptr` into +/// the hooks and caller threads) is what makes a leaked/detached hung thread safe on the RED path: the +/// thread keeps its own references alive, so nothing it touches is destroyed underneath it. +struct CaseSync +{ + std::mutex m; + std::condition_variable cv; + bool entered = false; /// guarded by m: the first flush reached the pre-carve hook + /// Per-`CarvePhaseForTest` invocation counter, indexed by `static_cast(phase)`. Sized off the + /// enum's last enumerator rather than a literal: the array was already undersized once (it predates + /// `ChunkReseed`; `PostDurableInstall` and `PostInstallPreAck` followed), and it is out of bounds only + /// because every call site happens to filter to a lower-numbered phase first -- a trap for the next + /// phase added. + std::atomic phase_hits[static_cast(CasRefLedger::CarvePhaseForTest::PostInstallPreAck) + 1] = {}; +}; + +/// The newest `_log/` transaction currently present for `ns`, decoded from the backend directly (no Pool +/// cache). Used to inspect exactly what a flush durably committed. +std::optional newestLogTxn(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const RootNamespace & ns) +{ + std::optional newest; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation + && parsed->kind == RefObjectKind::Log + && (!newest || *newest < parsed->txn_id)) + newest = parsed->txn_id; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + if (!newest) + return std::nullopt; + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest)); + if (!got) + return std::nullopt; + return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), *newest); +} + +/// Counts `OwnerTransition` removal ops (old binding present, no new binding) naming `ref_name` across +/// EVERY committed `_log/` transaction of `ns`. +size_t committedRemovalCountForRef(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, + const RootNamespace & ns, const String & ref_name) +{ + size_t count = 0; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (!parsed || parsed->life_id != DB::Cas::tests::fixture::fixtureLife(ns).incarnation + || parsed->kind != RefObjectKind::Log) + continue; + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), parsed->txn_id)); + if (!got) + continue; + const RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), parsed->txn_id); + for (const RefOp & op : txn.ops) + if (op.kind == RefOpKind::OwnerTransition && op.old_binding.has_value() + && !op.new_binding.has_value() && op.old_binding->ref_name == ref_name) + ++count; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return count; +} + +/// One queued append driven on its own thread, with a future that becomes ready only when the caller's +/// `dropRef` RETURNS (normally or by throwing). A caller whose item was stranded never returns, so its +/// future stays not-ready — a bounded `wait_for` on it is the hung-waiter detector. +struct Caller +{ + std::thread t; + std::future fut; + String ref; +}; + +Caller launchDrop(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + auto prom = std::make_shared>(); + std::future fut = prom->get_future(); + std::thread t([store, ns, ref, prom] + { + std::exception_ptr err; + try { store->dropRef(ns, ref); } + catch (...) { err = std::current_exception(); } + prom->set_value(err); + }); + return Caller{std::move(t), std::move(fut), ref}; +} + +/// Stages three compatible drops (leader "a" plus followers "b","c") into one carve, injects +/// `std::bad_alloc` at `target_phase` on its `target_ordinal`-th firing, then asserts that no caller +/// hangs and the queue drains. Returns true iff a caller hung (the stranded-item signature). +/// +/// On the fixed (two-phase) carve, a plan-phase throw pops nothing: the leader's own item is failed by +/// the leadership-exit guard and the untouched followers commit on the next leader's retry. On the +/// pre-fix interleaved carve, the same throw strands already-popped followers, whose waiters hang. +bool runPlanPointCase(Phase target_phase, int target_ordinal, const char * label) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{String("srv1/carve_") + label}; + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); + publishEmptyPart(store, ns, "c"); + + auto sync = std::make_shared(); + /// Block the first flush's leader in the pre-carve window until all three items are queued, forcing a + /// deterministic three-item batch. Heap-backed captures (see `CaseSync`) keep this safe even if a + /// stranded follower later spins through it on the RED path. + store->setRefPreCarveHookForTest([sync, store, ns] + { + std::unique_lock lk(sync->m); + if (sync->entered) + return; /// only the first carve blocks; retries proceed straight through + sync->entered = true; + sync->cv.notify_all(); + /// Bounded (10s) so a staging bug bounds the wait instead of blocking the whole suite; the + /// predicate is normally satisfied well before the deadline. + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return store->refQueuePendingForTest(ns) >= 3; }); + }); + store->setCarveHookForTest([sync, target_phase, target_ordinal](Phase ph) + { + if (ph != target_phase) + return; + if (sync->phase_hits[static_cast(ph)].fetch_add(1) + 1 == target_ordinal) + throw std::bad_alloc{}; + }); + + Caller ca = launchDrop(store, ns, "a"); + { + std::unique_lock lk(sync->m); + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return sync->entered; }); + } + Caller cb = launchDrop(store, ns, "b"); + Caller cc = launchDrop(store, ns, "c"); + { + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->refQueuePendingForTest(ns) < 3 && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); + } + sync->cv.notify_all(); /// release the pre-carve hook now its (>=3 pending) predicate holds + + bool hung = false; + std::vector callers = {&ca, &cb, &cc}; + for (Caller * c : callers) + { + /// Bounded (5s): a stranded item never completes, so this is where the hang surfaces. + if (c->fut.wait_for(std::chrono::seconds(5)) != std::future_status::ready) + { + hung = true; + EXPECT_TRUE(false) << label << ": caller for ref '" << c->ref + << "' never returned within 5s — its item was stranded (removed from " + "pending, never completed) and the waiter hung"; + /// Cannot join a permanently-hung thread. Detach it; its heap-backed captures (including a + /// `store` copy) keep everything it touches alive until the process exits. Leave the hooks + /// installed — clearing them here would race the detached thread's `std::function` read. + c->t.detach(); + } + else + { + c->t.join(); + } + } + if (hung) + return true; + + /// GREEN: everything joined, so clearing the hooks now cannot race any live flush. + store->setRefPreCarveHookForTest(nullptr); + store->setCarveHookForTest(nullptr); + EXPECT_EQ(store->refQueuePendingForTest(ns), 0u) << label << ": the append queue did not fully drain"; + /// The leader ("a") is the item whose plan threw; the guard fails it, so its drop must NOT commit. + EXPECT_TRUE(store->resolveRef(ns, "a").has_value()) + << label << ": the failed leader's drop must not have committed"; + /// The untouched followers commit on retry. + EXPECT_FALSE(store->resolveRef(ns, "b").has_value()) << label << ": survivor 'b' must have been dropped"; + EXPECT_FALSE(store->resolveRef(ns, "c").has_value()) << label << ": survivor 'c' must have been dropped"; + return false; +} + +} + +/// Test 7: a throw at any plan-phase point of the carve must leave the queue intact and hang no waiter. +/// RED on the pre-fix interleaved carve (a plan-point throw after the first pop strands followers whose +/// waiters then hang, tripping the 5s bounded wait); GREEN on the two-phase carve. +TEST(CASRefWriterCarve, CarveThrowLeavesQueueIntact) +{ + /// `PlanSeenRefs`/`PlanBatchGrow` fire once per scanned item; injecting on the third firing strands a + /// popped follower under the old carve. `PlanReserveOwned` fires once, after the whole selection is + /// (under the old carve) already popped, stranding every follower. `ASSERT_FALSE` stops at the first + /// demonstrated hang so at most one case leaks a detached thread on the RED path. + ASSERT_FALSE(runPlanPointCase(Phase::PlanSeenRefs, 3, "PlanSeenRefs")); + ASSERT_FALSE(runPlanPointCase(Phase::PlanBatchGrow, 3, "PlanBatchGrow")); + ASSERT_FALSE(runPlanPointCase(Phase::PlanReserveOwned, 1, "PlanReserveOwned")); +} + +/// Test 8: an allocation failure at the per-item accumulation point must leave the failed item's effects +/// out of BOTH `working` (the in-memory committed state) and the durable transaction. RED on the pre-fix +/// loop (which moved `working` and appended `final_ops` before the throwing point, so the failed drop +/// committed while its caller was told it failed); GREEN on the reserve-before-publish loop. +TEST(CASRefWriterCarve, ValidationAllocFailureLeavesWorkingClean) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const DB::Cas::Layout & layout = store->layout(); + const RootNamespace ns{"srv1/carve_validate"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + ASSERT_TRUE(store->resolveRef(ns, "x").has_value()); + ASSERT_TRUE(store->resolveRef(ns, "y").has_value()); + + auto sync = std::make_shared(); + store->setRefPreCarveHookForTest([sync, store, ns] + { + std::unique_lock lk(sync->m); + if (sync->entered) + return; + sync->entered = true; + sync->cv.notify_all(); + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + /// Fail the FIRST admitted item (the leader's own drop of "x") at its accumulation point. + store->setCarveHookForTest([sync](Phase ph) + { + if (ph != Phase::ValidateFinalOps) + return; + if (sync->phase_hits[static_cast(ph)].fetch_add(1) + 1 == 1) + throw std::bad_alloc{}; + }); + + Caller cx = launchDrop(store, ns, "x"); + { + std::unique_lock lk(sync->m); + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return sync->entered; }); + } + Caller cy = launchDrop(store, ns, "y"); + { + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->refQueuePendingForTest(ns) < 2 && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); + } + sync->cv.notify_all(); + + ASSERT_EQ(cx.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "drop x must not hang"; + ASSERT_EQ(cy.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "drop y must not hang"; + const std::exception_ptr x_err = cx.fut.get(); + const std::exception_ptr y_err = cy.fut.get(); + cx.t.join(); + cy.t.join(); + store->setRefPreCarveHookForTest(nullptr); + store->setCarveHookForTest(nullptr); + + /// The injected item's own caller was told the append failed. + ASSERT_TRUE(x_err != nullptr) << "drop x's caller must observe the injected allocation failure"; + /// The co-batched survivor committed cleanly. + EXPECT_TRUE(y_err == nullptr) << "the co-batched survivor drop y must commit"; + + /// (1) `working`/committed state stays clean: x's drop, whose caller failed, must NOT have taken + /// effect — x remains resolvable. + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) + << "the failed item's drop leaked into the committed state — `working` was not kept clean"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()) << "survivor drop y must be committed"; + + /// (2) Decode the committed object: no committed ref-log transaction may carry x's removal op, and + /// exactly the survivor's removal must be present. + EXPECT_EQ(committedRemovalCountForRef(*backend, layout, ns, "x"), 0u) + << "the failed item's removal op leaked into a durably-committed ref-log object"; + EXPECT_EQ(committedRemovalCountForRef(*backend, layout, ns, "y"), 1u) + << "the survivor's removal op must be present in exactly one committed ref-log object"; + + const auto newest = newestLogTxn(*backend, layout, ns); + ASSERT_TRUE(newest.has_value()); + for (const RefOp & op : newest->ops) + if (op.kind == RefOpKind::OwnerTransition && op.old_binding.has_value() && !op.new_binding.has_value()) + EXPECT_NE(op.old_binding->ref_name, String("x")) + << "the newest committed transaction must not contain the failed item's removal"; +} diff --git a/src/Disks/tests/gtest_cas_ref_catalog.cpp b/src/Disks/tests/gtest_cas_ref_catalog.cpp new file mode 100644 index 000000000000..11a4c029f139 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_catalog.cpp @@ -0,0 +1,1718 @@ +#include "cas_format_test_battery.h" +#include "cas_test_helpers.h" +#include +#include +#include +#include +/// Explicit rather than relying on a transitive path: `DEBUG_OR_SANITIZER_BUILD` (used below to gate +/// the `*DeathTest` split) must resolve in THIS translation unit. +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace ProfileEvents +{ + extern const Event CASGCUnmatchedAdoptedParentLives; + extern const Event CASGCStuckRemovals; +} + +namespace DB::Cas::tests +{ + +/// This friend-only compile pin is derived from the actual private production member pointers. It +/// fails if a raw round carrier becomes separately pairable with `fold`. +class GcRoundPlanSignatureAccess +{ +public: + using FoldSignature = decltype(&Gc::fold); + using ExpectedFoldSignature = Gc::FoldResult (Gc::*)( + GcState &, Token &, RoundReport &, uint64_t, const RefPlan &, UniversePolicy, GcRoundWorkBudget &); + using BuilderSignature = decltype(&buildRefWalkPlan); + using ExpectedBuilderSignature = RefPlan (*)(RoundInput &&); + + static_assert(std::is_same_v); + static_assert(std::is_same_v); +}; + +} + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; + extern const int LIMIT_EXCEEDED; + extern const int NETWORK_ERROR; + extern const int BAD_ARGUMENTS; +} + +namespace +{ + +/// Hand-builds one raw "ent" line, bypassing `encodeRefCatalog` entirely -- used by the decode-side +/// rejection tests, which must exercise bytes the encoder itself would refuse to produce. +String rawEntLine(const String & ns, const String & state, const String & inc_hex, + std::optional> creator = std::nullopt) +{ + if (!creator) + return fmt::format(R"({{"k":"ent","ns":"{}","st":"{}","inc":"{}"}})", ns, state, inc_hex); + const auto & [srid, we, fg] = *creator; + return fmt::format(R"({{"k":"ent","ns":"{}","st":"{}","inc":"{}","csr":"{}","cwe":"{}","cfg":"{}"}})", + ns, state, inc_hex, srid, we, fg); +} + +/// Wraps `ent_lines` in the header/trailer a real `cas_ref_catalog` object carries. `v:1` always +/// passes the header gate (any version <= the build's `G_BUILD` does), matching the convention +/// `gtest_cas_fold_seal_format.cpp`'s `RejectsOutOfRangeNsCleanupState` uses for the same reason. +String rawCatalog(const std::vector & ent_lines) +{ + String out = R"({"type":"cas_ref_catalog","v":1})" "\n"; + for (const String & l : ent_lines) + out += l + "\n"; + out += fmt::format("{{\"n\":{}}}\n", ent_lines.size()); + return out; +} + +String withRemovalStartedRound(String line, uint64_t round) +{ + const size_t close = line.rfind('}'); + EXPECT_NE(close, String::npos); + line.insert(close, fmt::format(R"(,"rsr":"{}")", round)); + return line; +} + +CatalogEntry liveEntry(const String & ns, uint64_t inc) +{ + return CatalogEntry{.ns = RootNamespace{ns}, .state = NsState::Live, .incarnation = UInt128(inc)}; +} + +CatalogEntry entryInState(const String & ns, NsState state, uint64_t inc) +{ + CatalogEntry entry{.ns = RootNamespace{ns}, .state = state, .incarnation = UInt128(inc)}; + if (state == NsState::Creating) + entry.creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + if (state == NsState::Removing) + entry.removal_started_round = 1; + return entry; +} + +class EraseWinnerBackend final : public DB::Cas::tests::CountingBackend +{ +public: + using CountingBackend::casPut; + using CountingBackend::get; + + void replaceOnNextCatalogCas(const String & key, std::optional replacement_) + { + catalog_key = key; + replacement = std::move(replacement_); + armed = true; + } + + bool fenceMoved() const { return fence_moved; } + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (armed && key == catalog_key) + { + armed = false; + const auto current = CountingBackend::get(key); + if (!current) + throw std::runtime_error("test fixture lost mandatory catalog"); + RefCatalog winner_catalog; + if (replacement) + winner_catalog.entries.push_back(*replacement); + const CasResult winner = CountingBackend::casPut( + key, encodeRefCatalog(winner_catalog), current->token, meta); + if (winner.outcome != CasOutcome::Committed) + throw std::runtime_error("test fixture winner failed to replace catalog"); + fence_moved = true; + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + +private: + String catalog_key; + std::optional replacement; + bool armed = false; + bool fence_moved = false; +}; + +class CasPutThrowsOnceBackend final : public DB::Cas::tests::CountingBackend +{ +public: + using CountingBackend::casPut; + + void armCasPutThrow(const String & key) + { + throw_key = key; + armed = true; + } + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (armed && key == throw_key) + { + armed = false; + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, + "injected casPut failure during completed-removal erase"); + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + +private: + String throw_key; + bool armed = false; +}; + +class ScopedCasGcLogCapture +{ +public: + ScopedCasGcLogCapture() + : logger(getLogger("CasGc")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("warning"); + } + + ~ScopedCasGcLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +} + +/// ---------- format-battery registration ---------- + +TEST(CASFormatBattery, RefCatalog) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, + .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = 5, .fence_generation = 2}}); + c.entries.push_back(liveEntry("b", 2)); + runFormatBattery({FormatId::RefCatalog, + [&] { return sealObject(FormatId::RefCatalog, encodeRefCatalog(c)); }, + [](std::string_view s) { decodeRefCatalog(std::string(openObject(FormatId::RefCatalog, s))); }, + currentFormatHeader("cas_ref_catalog") + + "{\"k\":\"ent\",\"ns\":\"a\",\"st\":\"creating\",\"inc\":\"00000000000000000000000000000001\"," + "\"csr\":\"srv1\",\"cwe\":\"5\",\"cfg\":\"2\"}\n" + "{\"k\":\"ent\",\"ns\":\"b\",\"st\":\"live\",\"inc\":\"00000000000000000000000000000002\"}\n" + "{\"n\":2}\n"}); +} + +/// ---------- codec round-trip ---------- + +TEST(CASRefCatalogFormat, RoundTripsAllThreeStates) +{ + RefCatalog in; + in.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, + .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = 5, .fence_generation = 2}}); + in.entries.push_back(liveEntry("b", 2)); + in.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"c"}, + .state = NsState::Removing, + .incarnation = UInt128(3), + .removal_started_round = 11}); + + const RefCatalog out = decodeRefCatalog(encodeRefCatalog(in)); + EXPECT_EQ(out, in); + EXPECT_EQ(out.entries[0].state, NsState::Creating); + EXPECT_EQ(out.entries[1].state, NsState::Live); + EXPECT_EQ(out.entries[2].state, NsState::Removing); +} + +/// Mutation caught: making removal age caller-local or optional would let an adopted `Removing` row +/// lose the immutable round from which stuck-removal diagnostics measure. +TEST(CASRefCatalogFormat, RemovalStartedRoundIsRequiredExactlyForRemoving) +{ + CatalogEntry removing{ + .ns = RootNamespace{"removing"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 19}; + const RefCatalog catalog{.entries = {removing}}; + const String encoded = encodeRefCatalog(catalog); + EXPECT_NE(encoded.find("\"rsr\":\"19\""), String::npos); + EXPECT_EQ(decodeRefCatalog(encoded), catalog); + + const String inc = "00000000000000000000000000000009"; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefCatalog(rawCatalog({rawEntLine("missing", "removing", inc)})); }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + (void)decodeRefCatalog(rawCatalog({withRemovalStartedRound(rawEntLine("forbidden", "live", inc), 21)})); + }); +} + +TEST(CASRefCatalogFormat, EmptyCatalogRoundTrips) +{ + EXPECT_EQ(decodeRefCatalog(encodeRefCatalog(RefCatalog{})), RefCatalog{}); +} + +/// Mutation caught: replacing the reverse index with `emplace`-and-ignore would make the first row +/// win. Every lifecycle state participates, both duplicate ids are unresolvable, and an unrelated +/// unique row remains usable by point resolution. +TEST(CASRefCatalogLifeIndex, DuplicatePhysicalIdsAreAmbiguousWithoutPoisoningUniquePointResolution) +{ + RefCatalog catalog; + catalog.entries = { + entryInState("a-creating", NsState::Creating, 7), + entryInState("b-live", NsState::Live, 7), + entryInState("c-removing", NsState::Removing, 8), + entryInState("d-live", NsState::Live, 8), + entryInState("e-unique", NsState::Live, 9), + }; + + const CatalogLifeIndex index(catalog); + EXPECT_TRUE(index.isAmbiguous(UInt128{7})); + EXPECT_TRUE(index.isAmbiguous(UInt128{8})); + EXPECT_THROW(index.resolve(UInt128{7}), DB::Exception); + EXPECT_THROW(index.resolve(UInt128{8}), DB::Exception); + const auto unique = index.resolve(UInt128{9}); + ASSERT_TRUE(unique); + EXPECT_EQ(unique->ns.string(), "e-unique"); +} + +/// Catalog mutation is destructive authority: any ambiguous current id stops the mutation before a +/// candidate can be written. An unrelated unique point lookup remains available from the same cut. +TEST(CASRefCatalogLifeIndex, AmbiguityStopsCatalogMutationButNotUnrelatedPointLookup) +{ + InMemoryBackend backend; + const Layout layout("p"); + RefCatalog catalog; + catalog.entries = { + entryInState("a", NsState::Live, 7), + entryInState("b", NsState::Removing, 7), + entryInState("c", NsState::Live, 9), + }; + ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(catalog)).outcome, PutOutcome::Done); + const auto before = backend.get(layout.refCatalogKey()); + ASSERT_TRUE(before); + + EXPECT_THROW(CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) { return current; }), DB::Exception); + const auto after = backend.get(layout.refCatalogKey()); + ASSERT_TRUE(after); + EXPECT_EQ(after->token, before->token); + EXPECT_EQ(after->bytes, before->bytes); + + const auto unique = CasRefCatalog::lifeIfCataloged(backend, layout, RootNamespace{"c"}); + ASSERT_TRUE(unique); + EXPECT_EQ(unique->incarnation, UInt128{9}); +} + +TEST(CASRefCatalogFormat, NamespaceAtExactByteBoundRoundTrips) +{ + RefCatalog c; + c.entries.push_back(liveEntry(String(kMaxNamespaceBytes, 'a'), 1)); + const RefCatalog out = decodeRefCatalog(encodeRefCatalog(c)); + EXPECT_EQ(out, c); +} + +/// ---------- strict rejections: encode side (LOGICAL_ERROR -- our own state, not yet durable) ---------- + +/// Every `expectThrowsCode(LOGICAL_ERROR, ...)` in this block aborts the process in debug/sanitizer +/// builds instead of behaving like a catchable exception (`Common/Exception.cpp`'s +/// `handle_error_code`), so each test is split: the throw-and-catch form below runs only on a plain +/// release build, and its `...DeathTest` counterpart (grouped after this block) proves the abort +/// positively on debug/sanitizer builds instead. +#ifndef DEBUG_OR_SANITIZER_BUILD + +TEST(CASRefCatalogFormat, EncodeRejectsDuplicateNamespace) +{ + RefCatalog c; + c.entries.push_back(liveEntry("a", 1)); + c.entries.push_back(liveEntry("a", 2)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsNonCanonicalOrder) +{ + RefCatalog c; + c.entries.push_back(liveEntry("b", 1)); + c.entries.push_back(liveEntry("a", 2)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsCreatorPresentOnLive) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsCreatorAbsentOnCreating) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, .incarnation = UInt128(1)}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsZeroIncarnation) +{ + RefCatalog c; + c.entries.push_back(liveEntry("a", 0)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsNameOverByteBound) +{ + RefCatalog c; + c.entries.push_back(liveEntry(String(kMaxNamespaceBytes + 1, 'a'), 1)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +TEST(CASRefCatalogFormat, EncodeRejectsEmptyNamespace) +{ + RefCatalog c; + c.entries.push_back(liveEntry("", 1)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { encodeRefCatalog(c); }); +} + +/// Mutation caught: making removal age caller-local or optional would let an adopted `Removing` row +/// lose the immutable round from which stuck-removal diagnostics measure. +TEST(CASRefCatalogFormat, EncodeRejectsLiveWithRemovalStartedRound) +{ + CatalogEntry live_with_round = liveEntry("live", 8); + live_with_round.removal_started_round = 20; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { (void)encodeRefCatalog(RefCatalog{.entries = {live_with_round}}); }); +} + +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsDuplicateNamespaceAborts) +{ + RefCatalog c; + c.entries.push_back(liveEntry("a", 1)); + c.entries.push_back(liveEntry("a", 2)); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "not canonically ordered"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsNonCanonicalOrderAborts) +{ + RefCatalog c; + c.entries.push_back(liveEntry("b", 1)); + c.entries.push_back(liveEntry("a", 2)); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "not canonically ordered"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsCreatorPresentOnLiveAborts) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}}); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "carries a creator fence"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsCreatorAbsentOnCreatingAborts) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, .incarnation = UInt128(1)}); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "lacks a creator fence"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsZeroIncarnationAborts) +{ + RefCatalog c; + c.entries.push_back(liveEntry("a", 0)); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "zero incarnation"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsNameOverByteBoundAborts) +{ + RefCatalog c; + c.entries.push_back(liveEntry(String(kMaxNamespaceBytes + 1, 'a'), 1)); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "admission bound"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsEmptyNamespaceAborts) +{ + RefCatalog c; + c.entries.push_back(liveEntry("", 1)); + EXPECT_DEATH({ (void)encodeRefCatalog(c); }, "namespace must not be empty"); +} + +TEST(CASRefCatalogFormatDeathTest, EncodeRejectsLiveWithRemovalStartedRoundAborts) +{ + CatalogEntry live_with_round = liveEntry("live", 8); + live_with_round.removal_started_round = 20; + EXPECT_DEATH( + { (void)encodeRefCatalog(RefCatalog{.entries = {live_with_round}}); }, "removal_started_round"); +} + +#endif + +/// A namespace + creator server_root_id that both max out at their respective byte bounds (512 + +/// 255), escaped worst-case, land one "ent" line over the 4 KiB line cap (~4.7 KiB) -- reachable +/// because neither this codec nor `validateServerRootId` restricts the charset, only the length. +/// The refusal must be `LIMIT_EXCEEDED` (a capacity refusal), not `LOGICAL_ERROR` (a bug report) -- +/// `encodeFoldSeal`'s own `checkLineBytes` raises `LIMIT_EXCEEDED` for the identical shape of gate. +TEST(CASRefCatalogFormat, EncodeLineOverCapRaisesLimitExceeded) +{ + RefCatalog c; + c.entries.push_back(CatalogEntry{ + .ns = RootNamespace{String(kMaxNamespaceBytes, '\x01')}, + .state = NsState::Creating, + .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = String(255, '\x01'), .writer_epoch = 1, .fence_generation = 1}}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] { encodeRefCatalog(c); }); +} + +/// ---------- strict rejections: decode side (CORRUPTED_DATA -- bytes may have come from anywhere) ---------- + +TEST(CASRefCatalogFormat, DecodeRejectsDuplicateNamespace) +{ + const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(1))), + rawEntLine("a", "live", u128ToHex(UInt128(2)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsNonCanonicalOrder) +{ + const String bad = rawCatalog({rawEntLine("b", "live", u128ToHex(UInt128(1))), + rawEntLine("a", "live", u128ToHex(UInt128(2)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsCreatorPresentOnLive) +{ + const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(1)), + std::make_tuple(String("srv"), uint64_t(1), uint64_t(1)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsCreatorAbsentOnCreating) +{ + const String bad = rawCatalog({rawEntLine("a", "creating", u128ToHex(UInt128(1)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsZeroIncarnation) +{ + const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(0)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsNameOverByteBound) +{ + const String too_long_ns(kMaxNamespaceBytes + 1, 'a'); + const String bad = rawCatalog({rawEntLine(too_long_ns, "live", u128ToHex(UInt128(1)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsUnknownState) +{ + const String bad = rawCatalog({rawEntLine("a", "bogus", u128ToHex(UInt128(1)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsEmptyNamespace) +{ + const String bad = rawCatalog({rawEntLine("", "live", u128ToHex(UInt128(1)))}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +TEST(CASRefCatalogFormat, DecodeRejectsMissingNamespaceKey) +{ + /// No "ns" key at all -- must be refused exactly like an explicit empty one, not read as "". + const String bad = rawCatalog({R"({"k":"ent","st":"live","inc":")" + u128ToHex(UInt128(1)) + "\"}"}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); +} + +/// `nsStateToWord`'s only reachable input is either a live `NsState` or one `nsStateFromWord` already +/// validated on decode, so an unrecognized value is a bug in THIS process -- `LOGICAL_ERROR`, matching +/// this file's own stated taxonomy for the encode-side helper it (indirectly, via `creatorPairingOk`'s +/// error message) serves. Aborts under debug/sanitizer builds -- split like the block above; +/// `CASRefCatalogFormatDeathTest.NsStateToWordRaisesLogicalErrorOnImpossibleValueAborts` covers it there. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalogFormat, NsStateToWordRaisesLogicalErrorOnImpossibleValue) +{ + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { nsStateToWord(static_cast(99)); }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogFormatDeathTest, NsStateToWordRaisesLogicalErrorOnImpossibleValueAborts) +{ + EXPECT_DEATH({ (void)nsStateToWord(static_cast(99)); }, "unknown ns state"); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange): the whole point of this test is an impossible enum value +} +#endif + +/// ---------- registry row / raw-storage tripwire ---------- + +/// The registry row is part of the contract, mirroring `gtest_cas_ref_ckpt.cpp`'s +/// `RegistryRowIsControlStrictWithTightCaps`: Control/Strict decides how the decoder treats unknown +/// keys, and the caps are the first thing that fires if a foreign object ever lands at the key. +TEST(CASRefCatalogFormat, RegistryRowIsControlStrictWithRawStorage) +{ + const FormatTraits & traits = traitsFor(FormatId::RefCatalog); + EXPECT_EQ(traits.type, "cas_ref_catalog"); + EXPECT_EQ(traits.family, TextFamily::Control); + EXPECT_EQ(traits.strictness, KeyStrictness::Strict); + EXPECT_EQ(traits.object_cap, 256u * 1024u * 1024u); + EXPECT_EQ(traits.line_cap, 4u * 1024u); + EXPECT_EQ(traitsForType("cas_ref_catalog"), &traits); + /// Raw, so the key has no suffix: `Pool/CasRefCatalog.cpp` hands bytes to/from the backend + /// directly, bypassing `sealObject`/`openObject` because both are the identity under + /// `CompressionPolicy::Never`. This line is the TRIPWIRE for that shortcut -- a policy flip to + /// `Always` would silently write uncompressed bodies under a `.zst` key, which this assertion + /// catches first (see `CasRefCatalogFormat.h`'s comment on `encodeRefCatalog`). + EXPECT_EQ(storedSuffix(FormatId::RefCatalog), ""); + EXPECT_EQ(traits.compression, CompressionPolicy::Never); +} + +/// ---------- capacity admission: per-predicate boundary tests [codex r2/r3 finding 9] ---------- + +TEST(CASRefCatalogAdmission, Predicate1AcceptsEqualityRefusesCapPlusOne) +{ + const uint64_t cap = traitsFor(FormatId::RefCatalog).object_cap; + const RootNamespace ns{"admitted"}; + EXPECT_NO_THROW(checkCatalogObjectBytes(cap, ns)); + try + { + checkCatalogObjectBytes(cap + 1, ns); + FAIL() << "expected LIMIT_EXCEEDED"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LIMIT_EXCEEDED); + EXPECT_NE(e.message().find("predicate 1"), String::npos) << e.message(); + EXPECT_NE(e.message().find(ns.string()), String::npos) << e.message(); + } +} + +TEST(CASRefCatalogAdmission, Predicate2AcceptsEqualityRefusesOneEntryOver) +{ + const Layout layout("p"); + constexpr uint64_t gc_shards = 1; + /// The exact boundary is expressed in ENTRIES (predicate (2) is a sum over admitted entries), so + /// the boundary count is derived from the real registry constants rather than assumed. + const uint64_t cap = foldSealCaps().object_cap; + const uint64_t fixed = foldSealFixedBytes(); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + ASSERT_GT(reservation, 0u); + const uint64_t nonentry = widestBlobTargetRunReservationBytes(layout, gc_shards) + + widestCondemnedSummaryReservationBytes(gc_shards); + const uint64_t max_entries = (cap - fixed - nonentry) / reservation; + + const RootNamespace ns{"admitted"}; + EXPECT_NO_THROW(checkFoldSealReservation(max_entries, gc_shards, layout, ns)); + try + { + checkFoldSealReservation(max_entries + 1, gc_shards, layout, ns); + FAIL() << "expected LIMIT_EXCEEDED"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LIMIT_EXCEEDED); + EXPECT_NE(e.message().find("predicate 2"), String::npos) << e.message(); + EXPECT_NE(e.message().find(ns.string()), String::npos) << e.message(); + } +} + +/// `entry_count * worstCaseEntryFoldReservationBytes()` must saturate, not wrap: choosing +/// `entry_count` as the SMALLEST value whose true (unbounded) product with `reservation` crosses +/// 2^64, an unsaturated `uint64_t` multiplication wraps to a remainder SMALLER than `reservation` +/// itself (a few KiB) -- which reads as trivially "fits" a 256 MiB cap even though the real +/// reservation this many entries demands is astronomically larger. A saturating multiply refuses it +/// regardless of the wraparound arithmetic underneath. +TEST(CASRefCatalogAdmission, Predicate2SaturatesEntryCountReservationInsteadOfWrapping) +{ + const Layout layout("p"); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + const uint64_t entry_count = std::numeric_limits::max() / reservation + 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, + [&] { checkFoldSealReservation(entry_count, 1, layout, RootNamespace{"huge"}); }); +} + +TEST(CASRefCatalogAdmission, CombinedAdmissionPropagatesCandidateEntryCount) +{ + /// `checkCatalogAdmission` runs predicate (1) then predicate (2) against the SAME candidate; for + /// an ordinary small catalog both hold slack and it returns the exact bytes `encodeRefCatalog` + /// would produce. + RefCatalog candidate; + candidate.entries.push_back(liveEntry("a", 1)); + candidate.entries.push_back(liveEntry("b", 2)); + const Layout layout("p"); + const String encoded = checkCatalogAdmission(candidate, 1, layout, RootNamespace{"b"}); + EXPECT_EQ(encoded, encodeRefCatalog(candidate)); +} + +TEST(CASRefCatalogAdmission, ReservationCoversActualWidestLegalRowsAcrossDecimalTransitions) +{ + const Layout layout("p/quoted-\"prefix"); + constexpr uint64_t gc_shards = 100; + constexpr uint64_t max = std::numeric_limits::max(); + + for (const uint64_t entry_count : {9, 10, 99, 100}) + { + CasFoldSeal seal; + seal.generation = max; + seal.parent_generation = max; + for (uint64_t i = 0; i < entry_count; ++i) + { + seal.ref_lives.emplace(std::numeric_limits::max() - i, RefLifeFoldState{ + .coverage = RefCoverage{ + .classification = 4, + .last_folded_ref_id = RefTxnId{max, max}, + .hold = RefHold{ + .reason = HoldReason::UnconsumedSealCrossing, + .offending_position = RefTxnId{max, max}, + .retry_count = std::numeric_limits::max(), + .next_retry_round = max}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{max, max}}}); + } + for (uint64_t shard = 0; shard < gc_shards; ++shard) + { + /// Predicate 2 charges exactly `gc_shards` widest `btr` rows. This fixture is the maximum + /// legal cardinality, not an optimistic producer convention: authoritative fold-seal + /// grammar permits at most one run per shard and requires its canonical key to use seq 0. + seal.blob_target_runs.push_back(RunRef{ + .key = layout.blobTargetRunKey(max, max, shard, 0), + .checksum = std::numeric_limits::max(), + .shard = shard, + .generation = max}); + seal.condemned_summary.emplace(shard, CondemnedSummary{ + .condemned_total = max, + .pending_total = max, + .oldest_nonpending_condemn_round = max}); + } + + ASSERT_EQ(seal.blob_target_runs.size(), gc_shards); + EXPECT_NO_THROW(validateFoldSealForWrite(seal, layout, gc_shards)); + + const uint64_t bound = foldSealFixedBytes() + + entry_count * worstCaseEntryFoldReservationBytes() + + gc_shards * widestBlobTargetRunReservationBytes(layout, gc_shards) + + gc_shards * widestCondemnedSummaryReservationBytes(gc_shards); + EXPECT_LE(encodeFoldSeal(seal).size(), bound) << "entry_count=" << entry_count; + } +} + +/// ---------- Constraint 13: removal is never refused, even at the admission boundary ---------- + +TEST(CASRefCatalogAdmission, RemovalNeverRefusedEvenAtCapacity) +{ + const Layout layout("p"); + constexpr uint64_t gc_shards = 1; + const uint64_t cap = foldSealCaps().object_cap; + const uint64_t fixed = foldSealFixedBytes(); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + const uint64_t nonentry = widestBlobTargetRunReservationBytes(layout, gc_shards) + + widestCondemnedSummaryReservationBytes(gc_shards); + const uint64_t max_entries = (cap - fixed - nonentry) / reservation; + + /// Confirm the boundary is real: one entry beyond it is refused through admission. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, + [&] { checkFoldSealReservation(max_entries + 1, gc_shards, layout, RootNamespace{"z"}); }); + + /// Build a catalog carrying exactly `max_entries` Live entries -- as full as admission ever + /// permits -- directly (a fixture, not itself an admission call). + RefCatalog full; + full.entries.reserve(max_entries); + for (uint64_t i = 0; i < max_entries; ++i) + full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); + + InMemoryBackend backend; + backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(full)); + + /// The removal transition (Live -> Removing) on one entry goes through the PLAIN update path + /// (`casUpdate`, which runs no admission check at all) and succeeds even though the catalog is + /// already at the point where ANY growth would be refused. + const RefCatalog after = CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) + { + RefCatalog next = cur; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + EXPECT_EQ(after.entries.size(), max_entries); + EXPECT_EQ(after.entries[0].state, NsState::Removing); +} + +/// ---------- Pool/CasRefCatalog: token-CAS read / create / update / conflict-retry ---------- + +TEST(CASRefCatalog, ReadAbsentFailsClosed) +{ + InMemoryBackend backend; + Layout layout("p"); + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)CasRefCatalog::read(backend, layout); }); +} + +TEST(CASRefCatalog, CasUpdateRefusesWhenAbsent) +{ + InMemoryBackend backend; + Layout layout("p"); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) { return cur; }); + }); + EXPECT_FALSE(backend.head(layout.refCatalogKey()).exists); +} + +TEST(CASRefCatalog, CasUpdateAppliesOnTopOfExistingState) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + + const RefCatalog updated = CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) + { + RefCatalog next = cur; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + ASSERT_EQ(updated.entries.size(), 1u); + EXPECT_EQ(updated.entries[0].ns.string(), "a"); + EXPECT_EQ(updated.entries[0].state, NsState::Removing); +} + +/// `CasRefCatalog::casUpdate`'s identity-preserving refusal throws `LOGICAL_ERROR`, which aborts the +/// whole process in debug/sanitizer builds (`Common/Exception.cpp`'s `handle_error_code`) instead of +/// behaving like a catchable exception -- so the throw-and-catch form below runs only on a plain +/// release build, and `CASRefCatalogDeathTest.GenericCasUpdateCannotDeleteOrReplaceCatalogIdentityAborts` +/// proves the abort positively on debug/sanitizer builds instead. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalog, GenericCasUpdateCannotDeleteOrReplaceCatalogIdentity) +{ + const Layout layout("p"); + { + InMemoryBackend backend; + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog &) { return RefCatalog{}; }); + }); + } + + { + InMemoryBackend backend; + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + next.entries[0] = liveEntry("b", 2); + return next; + }); + }); + } +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogDeathTest, GenericCasUpdateCannotDeleteOrReplaceCatalogIdentityAborts) +{ + const Layout layout("p"); + { + InMemoryBackend backend; + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + EXPECT_DEATH( + { (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog &) { return RefCatalog{}; }); }, + "cannot add or delete catalog entries"); + } + + { + InMemoryBackend backend; + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + EXPECT_DEATH( + { + (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + { + RefCatalog next = current; + next.entries[0] = liveEntry("b", 2); + return next; + }); + }, + "cannot replace catalog identity"); + } +} +#endif + +TEST(CASRefCatalog, CasUpdateRetriesOnConflictAgainstFreshState) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + + backend.failNextCasPut(layout.refCatalogKey()); /// one-shot artificial Conflict on the next write + + int mutate_calls = 0; + const RefCatalog result = CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + { + ++mutate_calls; + RefCatalog next = cur; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + + EXPECT_EQ(mutate_calls, 2); /// first attempt hit the injected conflict; the retry succeeded + ASSERT_EQ(result.entries.size(), 1u); + EXPECT_EQ(result.entries[0].state, NsState::Removing); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + EXPECT_EQ(snap.catalog, result); +} + +TEST(CASRefCatalog, BeginRemovingRechecksFenceAfterCatalogCasConflict) +{ + InMemoryBackend backend; + const Layout layout("p"); + const CatalogEntry observed = liveEntry("a", 1); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, observed); + + uint64_t current_fence_generation = 7; + size_t fence_checks = 0; + const auto outcome = CasRefCatalog::beginRemoving( + backend, layout, observed, /*removal_started_round*/ 13, /*admitted_generation*/ 7, + [&](uint64_t admitted_generation) + { + ++fence_checks; + if (admitted_generation != current_fence_generation) + throw std::runtime_error("stale catalog mutation fence"); + if (fence_checks == 1) + { + /// Move the caller fence after the first admission check and force that attempt's + /// catalog CAS to conflict. The next attempt must check the fence again before writing. + current_fence_generation = 8; + backend.failNextCasPut(layout.refCatalogKey()); + } + }); + + EXPECT_EQ(outcome, CasRefCatalog::BeginRemovingOutcome::FencedOut); + EXPECT_EQ(fence_checks, 2u); + const CasRefCatalog::Snapshot after = CasRefCatalog::read(backend, layout); + EXPECT_EQ(after.catalog.entries, std::vector{observed}); +} + +/// A re-read that finds the catalog genuinely ABSENT after it was previously observed present is a +/// real concurrent delete, not a bootstrap -- `casUpdate` must refuse rather than silently create a +/// fresh catalog containing only this one mutation's entry (which would drop every other namespace). +/// Reproduced with a REAL delete (no fault injection needed): `mutate`'s first invocation deletes the +/// seeded object using the token `casUpdate`'s own initial read observed, so the loop's own `casPut` +/// against that now-stale token gets a genuine `Conflict`, and the follow-up re-read genuinely finds +/// the key absent. +/// Missing mandatory authority raises `CORRUPTED_DATA`; the split remains only because the debug +/// variant historically lived in the death-test suite. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalog, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTheCatalog) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(backend, layout); + ASSERT_TRUE(seeded.token.has_value()); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + { + backend.deleteExact(layout.refCatalogKey(), *seeded.token); + RefCatalog next = cur; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + }); + + /// Nothing was written by the failed attempt: the object is exactly as the delete left it + /// (absent), never a fresh single-entry catalog. + EXPECT_FALSE(backend.head(layout.refCatalogKey()).exists); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogDeathTest, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTheCatalogAborts) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(backend, layout); + ASSERT_TRUE(seeded.token.has_value()); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + { + backend.deleteExact(layout.refCatalogKey(), *seeded.token); + RefCatalog next = cur; + next.entries[0].state = NsState::Removing; + next.entries[0].removal_started_round = 1; + return next; + }); + }); +} +#endif + +/// The retry loop is bounded (the same live-lock brake `publishCkpt`/`allocateWriterEpoch` use on +/// their own contended token-CAS singletons) and ends in the typed retryable error, not an infinite +/// spin. `mutate` re-arms the one-shot conflict injection on every call, so every attempt fails. +TEST(CASRefCatalog, CasUpdateGivesUpAfterBoundedAttemptsWithRetryLaterError) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + + int mutate_calls = 0; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + { + ++mutate_calls; + backend.failNextCasPut(layout.refCatalogKey()); + RefCatalog next = cur; + return next; + }); + }); + EXPECT_GT(mutate_calls, 1); /// genuinely retried, not a single-shot failure +} + +TEST(CASRefCatalog, CasAdmitEntryAcceptsAnOrdinaryCreation) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + + const RefCatalog created = CasRefCatalog::casAdmitEntry(backend, layout, 1, + CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, .incarnation = UInt128(1), + .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}}); + ASSERT_EQ(created.entries.size(), 1u); + EXPECT_EQ(created.entries[0].state, NsState::Creating); +} + +TEST(CASRefCatalog, CasAdmitEntryInsertsAtCanonicalPosition) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("b", 1)); + const RefCatalog after = CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); + ASSERT_EQ(after.entries.size(), 2u); + EXPECT_EQ(after.entries[0].ns.string(), "a"); /// inserted BEFORE "b", not appended + EXPECT_EQ(after.entries[1].ns.string(), "b"); +} + +/// Caught by `encodeRefCatalog`'s own canonical-order/no-duplicate grammar check, inside +/// `checkCatalogAdmission` -- no separate duplicate check needed here. That `LOGICAL_ERROR` aborts +/// under debug/sanitizer builds -- split like the blocks above. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalog, CasAdmitEntryRejectsADuplicateNamespace) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogDeathTest, CasAdmitEntryRejectsADuplicateNamespaceAborts) +{ + InMemoryBackend backend; + Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + EXPECT_DEATH({ CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); }, "not canonically ordered"); +} +#endif + +TEST(CASRefCatalog, CasAdmitEntryRefusesOverCapacity) +{ + InMemoryBackend backend; + Layout layout("p"); + + const uint64_t cap = foldSealCaps().object_cap; + const uint64_t fixed = foldSealFixedBytes(); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + const uint64_t nonentry = widestBlobTargetRunReservationBytes(layout, 1) + + widestCondemnedSummaryReservationBytes(1); + const uint64_t max_entries = (cap - fixed - nonentry) / reservation; + + /// Seed the catalog directly at the admission boundary (a fixture -- not itself an admission call). + RefCatalog full; + full.entries.reserve(max_entries); + for (uint64_t i = 0; i < max_entries; ++i) + full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); + backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(full)); + + /// Admitting ONE more namespace is refused -- the additive predicate is checked BEFORE the write, + /// so the backend object is untouched. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] + { + CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("zzz", 999999999)); + }); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + EXPECT_EQ(snap.catalog.entries.size(), max_entries); +} + +TEST(CASRefCatalogRemoval, DeleteCompletedRemovingRequiresExactAdoptedProofAndLeaderFence) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, + PutOutcome::Done); + + CasFoldSeal held_parent; + held_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{ + .classification = 4, + .last_folded_ref_id = RefTxnId{1, 2}, + .hold = RefHold{.offending_position = RefTxnId{1, 3}}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, held_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); + + CasFoldSeal mismatched_parent; + mismatched_parent.ref_lives.emplace(UInt128{8}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, mismatched_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); + + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + CatalogEntry live = removing; + live.state = NsState::Live; + live.removal_started_round.reset(); + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, live, ready_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); + CatalogEntry creating = live; + creating.state = NsState::Creating; + creating.creator = CreatorFence{.server_root_id = "server", .writer_epoch = 3, .fence_generation = 4}; + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, creating, ready_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); + + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Moved; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::FencedOut); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, [](uint64_t generation) + { + EXPECT_EQ(generation, 5); + return CasRefCatalog::LeaderFenceStatus::Held; + }), + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 1); + EXPECT_EQ(backend.listTotal(), 0); + EXPECT_EQ(backend.deleteTotal(), 0); +} + +TEST(CASRefCatalogRemoval, ExactDeletionRefusesChangedEntryAndAdmissionCannotCarryRemoval) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; +#ifndef DEBUG_OR_SANITIZER_BUILD + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { (void)CasRefCatalog::casAdmitEntry(backend, layout, 1, removing); }); +#endif + + const CatalogEntry current{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 14}; + ASSERT_EQ(backend.putIfAbsent("unrelated", "sentinel").outcome, PutOutcome::Done); + ASSERT_EQ(backend.casPut(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {current}}), + CasRefCatalog::read(backend, layout).token).outcome, CasOutcome::Committed); + + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + CasRefCatalog::CompletedRemovingDeleteOutcome::EntryChanged); + EXPECT_EQ(CasRefCatalog::read(backend, layout).catalog.entries, std::vector{current}); +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogRemovalDeathTest, AdmissionCannotCarryRemovalAborts) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + CasRefCatalog::initializeEmptyForNewPool(backend, layout); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + + EXPECT_DEATH( + { (void)CasRefCatalog::casAdmitEntry(backend, layout, 1, removing); }, + "cannot admit namespace.*directly as Removing"); +} +#endif + +/// Mutation caught: deriving the control outcome from the resolution snapshot would turn a stale +/// leader's `FencedOut` into `Deleted` or `EntryChanged`. Resolution may prove the old life dead and +/// carry its invalidation, but it cannot restore the caller's authority to continue the GC round. +TEST(CASRefCatalogRemoval, FenceLossRemainsControlOutcomeWhenWinnerRemovesOrReplacesLife) +{ + for (const bool replace : {false, true}) + { + EraseWinnerBackend backend; + const Layout layout(replace ? "replacement" : "absence"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + ASSERT_EQ(backend.putIfAbsent( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, + PutOutcome::Done); + + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + std::optional replacement; + if (replace) + replacement = CatalogEntry{ + .ns = removing.ns, + .state = NsState::Live, + .incarnation = UInt128{8}}; + backend.replaceOnNextCatalogCas(layout.refCatalogKey(), replacement); + + const CasRefCatalog::CompletedRemovingDeleteResult result + = CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, [&](uint64_t) + { + if (backend.fenceMoved()) + return CasRefCatalog::LeaderFenceStatus::Moved; + return CasRefCatalog::LeaderFenceStatus::Held; + }); + + EXPECT_EQ(result.outcome, CasRefCatalog::CompletedRemovingDeleteOutcome::FencedOut); + ASSERT_TRUE(result.invalidated_life); + EXPECT_EQ(*result.invalidated_life, + NamespaceLifeId::fromCatalogEntry(removing.ns, removing.incarnation)); + const RefCatalog current = CasRefCatalog::read(backend, layout).catalog; + if (replace) + EXPECT_EQ(current.entries, std::vector{*replacement}); + else + EXPECT_TRUE(current.entries.empty()); + } +} + +/// Mutation caught: treating every authority-check exception as a moved fence hides corruption and +/// backend/decode failures. Before any CAS, inability to evaluate authority must propagate unchanged. +TEST(CASRefCatalogRemoval, NonFenceAuthorityExceptionPropagatesBeforeEraseCas) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("pre-cas-authority-error"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + ASSERT_EQ(backend.putIfAbsent( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, + PutOutcome::Done); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + (void)CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, [](uint64_t) -> CasRefCatalog::LeaderFenceStatus + { + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "injected authority read failure before erase CAS"); + }); + }); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0u); +} + +/// The post-CAS authority check is distinct: the erase may already be durable and its mandatory +/// resolution complete, but inability to evaluate authority is still the original error, not +/// `FencedOut`. +TEST(CASRefCatalogRemoval, NonFenceAuthorityExceptionPropagatesAfterEraseResolution) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("post-cas-authority-error"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + ASSERT_EQ(backend.putIfAbsent( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, + PutOutcome::Done); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + size_t authority_checks = 0; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + (void)CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, [&](uint64_t) + { + if (++authority_checks == 2) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "injected authority read failure after erase resolution"); + return CasRefCatalog::LeaderFenceStatus::Held; + }); + }); + EXPECT_EQ(authority_checks, 2u); + EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); +} + +/// Mutation caught: swallowing a synchronous `casPut` exception raised during the erase attempt +/// itself (as opposed to the authority/fence check) and treating it as ordinary non-convergence +/// would hide a real backend fault behind ProofRefused/EntryChanged, and would skip the mandatory +/// resolution read that this branch's siblings above already prove runs before any conclusion. +TEST(CASRefCatalogRemoval, CasPutExceptionPropagatesAfterMandatoryResolution) +{ + CasPutThrowsOnceBackend backend; + const Layout layout("cas-put-throw"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + ASSERT_EQ(backend.putIfAbsent( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, + PutOutcome::Done); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + backend.armCasPutThrow(layout.refCatalogKey()); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + (void)CasRefCatalog::deleteCompletedRemoving( + backend, layout, removing, ready_parent, 5, [](uint64_t) + { + return CasRefCatalog::LeaderFenceStatus::Held; + }); + }); + /// The mandatory resolution read ran before the rethrow: the exact old row is still present, + /// unchanged by the failed attempt. + const RefCatalog current = CasRefCatalog::read(backend, layout).catalog; + EXPECT_EQ(current.entries, std::vector{removing}); +} + +TEST(CASRefCatalogRemoval, CancelStalledCreatingRequiresExactRowAndTerminalCreatorFence) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const CatalogEntry creating{ + .ns = RootNamespace{"a"}, + .state = NsState::Creating, + .incarnation = UInt128{7}, + .creator = CreatorFence{.server_root_id = "server", .writer_epoch = 3, .fence_generation = 4}}; + ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {creating}})).outcome, + PutOutcome::Done); + + EXPECT_EQ(CasRefCatalog::cancelStalledCreating( + backend, layout, creating, [](const CreatorFence &) { return false; }, 5, [](uint64_t) {}), + CasRefCatalog::StalledCreatingCancelOutcome::CreatorFenceStillLive); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + + CatalogEntry stale = creating; + stale.creator->writer_epoch = 2; + EXPECT_EQ(CasRefCatalog::cancelStalledCreating( + backend, layout, stale, [](const CreatorFence &) { return true; }, 5, [](uint64_t) {}), + CasRefCatalog::StalledCreatingCancelOutcome::EntryChanged); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + + EXPECT_EQ(CasRefCatalog::cancelStalledCreating( + backend, layout, creating, [](const CreatorFence &) { return true; }, 5, [](uint64_t) {}), + CasRefCatalog::StalledCreatingCancelOutcome::Cancelled); + EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); + EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 1); + EXPECT_EQ(backend.listTotal(), 0); + EXPECT_EQ(backend.deleteTotal(), 0); +} + +TEST(CASGCRefWalkPlan, CatalogIsSoleRowAdmissionAuthorityAcrossOrdinaryAndRebuildInputs) +{ + RefCatalog catalog; + catalog.entries = { + CatalogEntry{ + .ns = RootNamespace{"creating"}, + .state = NsState::Creating, + .incarnation = UInt128{1}, + .creator = CreatorFence{.server_root_id = "server", .writer_epoch = 1, .fence_generation = 1}}, + liveEntry("live", 2), + CatalogEntry{ + .ns = RootNamespace{"removing"}, + .state = NsState::Removing, + .incarnation = UInt128{3}, + .removal_started_round = 8}, + }; + const CasRefCatalog::Snapshot cut{ + .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + + RefScanSummary ordinary_scan; + ordinary_scan.parent_ref_lives.emplace(UInt128{1}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}}); + ordinary_scan.parent_ref_lives.emplace(UInt128{3}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{3, 3}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{3, 3}}}); + ordinary_scan.parent_ref_lives.emplace(UInt128{4}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{4, 4}}}); + ordinary_scan.listed_lives = {UInt128{1}, UInt128{2}, UInt128{4}}; + ordinary_scan.holds.emplace(UInt128{1}, RefHold{.offending_position = RefTxnId{1, 2}}); + ordinary_scan.holds.emplace(UInt128{2}, RefHold{.offending_position = RefTxnId{2, 2}}); + ordinary_scan.checkpoint_observations.emplace(UInt128{1}, RefTxnId{1, 9}); + ordinary_scan.checkpoint_observations.emplace(UInt128{2}, RefTxnId{2, 9}); + ordinary_scan.max_log_by_life.emplace(UInt128{1}, RefTxnId{1, 10}); + ordinary_scan.max_log_by_life.emplace(UInt128{2}, RefTxnId{2, 10}); + + RefScanSummary rebuild_scan; + rebuild_scan.parent_ref_lives.emplace(UInt128{1}, ordinary_scan.parent_ref_lives.at(UInt128{1})); + rebuild_scan.parent_ref_lives.emplace(UInt128{5}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{5, 5}}}); + rebuild_scan.listed_lives = {UInt128{1}, UInt128{3}, UInt128{5}}; + rebuild_scan.holds.emplace(UInt128{1}, RefHold{.offending_position = RefTxnId{1, 3}}); + rebuild_scan.holds.emplace(UInt128{3}, RefHold{.offending_position = RefTxnId{3, 4}}); + rebuild_scan.checkpoint_observations.emplace(UInt128{1}, RefTxnId{1, 11}); + rebuild_scan.checkpoint_observations.emplace(UInt128{3}, RefTxnId{3, 11}); + rebuild_scan.max_log_by_life.emplace(UInt128{1}, RefTxnId{1, 12}); + rebuild_scan.max_log_by_life.emplace(UInt128{3}, RefTxnId{3, 12}); + + const RefPlan ordinary = tests::buildRefWalkPlanForTest(ordinary_scan, cut); + const RefPlan rebuild = tests::buildRefWalkPlanForTest(rebuild_scan, cut); + const auto ordinary_parent_states = ordinary.parentFoldStates(); + const auto rebuild_parent_states = rebuild.parentFoldStates(); + const auto ordinary_successor_states = ordinary.successorFoldStates(); + const auto rebuild_successor_states = rebuild.successorFoldStates(); + EXPECT_EQ(ordinary_parent_states.size(), 1u); + EXPECT_TRUE(ordinary_parent_states.contains(UInt128{3})); + EXPECT_FALSE(ordinary_parent_states.contains(UInt128{2})); + EXPECT_TRUE(ordinary_successor_states.contains(UInt128{2})); + EXPECT_TRUE(ordinary_successor_states.contains(UInt128{3})); + EXPECT_TRUE(rebuild_parent_states.empty()); + EXPECT_TRUE(rebuild_successor_states.contains(UInt128{2})); + EXPECT_TRUE(rebuild_successor_states.contains(UInt128{3})); + const std::set expected{UInt128{2}, UInt128{3}}; + EXPECT_EQ(ordinary.lifeIds(), expected); + EXPECT_EQ(rebuild.lifeIds(), expected); + + EXPECT_TRUE(ordinary.row(UInt128{2}).listed_hint); + ASSERT_TRUE(ordinary.row(UInt128{2}).fold_state.coverage.hold); + EXPECT_EQ(ordinary.row(UInt128{2}).checkpoint_observation, (RefTxnId{2, 9})); + EXPECT_EQ(ordinary.row(UInt128{2}).tail_observation, (RefTxnId{2, 10})); + const std::optional cleanup_evidence{ + RefCleanupEvidence{.remove_txn_id = RefTxnId{3, 3}}}; + EXPECT_EQ(ordinary.row(UInt128{3}).fold_state.cleanup_evidence, cleanup_evidence); + EXPECT_EQ(ordinary.row(UInt128{3}).removal_started_round, 8u); + EXPECT_FALSE(ordinary.contains(UInt128{1})); + EXPECT_FALSE(ordinary.contains(UInt128{4})); + + EXPECT_TRUE(rebuild.row(UInt128{3}).listed_hint); + ASSERT_TRUE(rebuild.row(UInt128{3}).fold_state.coverage.hold); + EXPECT_EQ(rebuild.row(UInt128{3}).checkpoint_observation, (RefTxnId{3, 11})); + EXPECT_EQ(rebuild.row(UInt128{3}).tail_observation, (RefTxnId{3, 12})); + EXPECT_FALSE(rebuild.contains(UInt128{1})); + EXPECT_FALSE(rebuild.contains(UInt128{5})); +} + +TEST(CASGCStuckRemoval, ThresholdAndRestartUseOnlyDurableRounds) +{ + const Layout layout("p"); + RefWalkPlanRow row{ + .life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"removing"}, UInt128{7}), + .fold_state = {}, + .removal_started_round = 10, + .has_parent_fold_state = false, + .listed_hint = false, + .checkpoint_observation = std::nullopt, + .tail_observation = std::nullopt}; + + EXPECT_FALSE(stuckRemovalWarning(row, /*current_round=*/12, /*threshold_rounds=*/3, layout)); + const auto at_threshold = stuckRemovalWarning(row, /*current_round=*/13, /*threshold_rounds=*/3, layout); + const auto next_round = stuckRemovalWarning(row, /*current_round=*/14, /*threshold_rounds=*/3, layout); + ASSERT_TRUE(at_threshold); + ASSERT_TRUE(next_round); + EXPECT_NE(at_threshold->find("age_rounds=3"), String::npos); + EXPECT_NE(next_round->find("age_rounds=4"), String::npos); + + /// A fresh process given the same durable catalog row and adopted round produces the same signal. + EXPECT_EQ(stuckRemovalWarning(row, 13, 3, layout), at_threshold); + + row.fold_state.cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}; + EXPECT_FALSE(stuckRemovalWarning(row, 100, 3, layout)); +} + +TEST(CASGCStuckRemoval, BoundaryAndAbsentVersusUnreadableMessagesAreExact) +{ + const Layout layout("p"); + RefWalkPlanRow row{ + .life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"removing"}, UInt128{7}), + .fold_state = {}, + .removal_started_round = std::numeric_limits::max(), + .has_parent_fold_state = false, + .listed_hint = false, + .checkpoint_observation = std::nullopt, + .tail_observation = std::nullopt}; + EXPECT_FALSE(stuckRemovalWarning(row, 0, 1, layout)); + + row.removal_started_round = 1; + const auto absent = stuckRemovalWarning(row, 2, 1, layout); + ASSERT_TRUE(absent); + EXPECT_NE(absent->find("terminal has not folded"), String::npos); + EXPECT_EQ(absent->find("/_log/"), String::npos) << "an absent terminal has no exact id to name"; + + row.fold_state.coverage.classification = 4; + row.fold_state.coverage.hold = RefHold{ + .reason = HoldReason::BodyUndecodable, + .offending_position = RefTxnId{5, 6}}; + const auto unreadable = stuckRemovalWarning(row, 2, 1, layout); + ASSERT_TRUE(unreadable); + EXPECT_NE(unreadable->find(layout.refLogKey(row.life, RefTxnId{5, 6})), String::npos); + EXPECT_NE(unreadable->find("is unreadable"), String::npos); + EXPECT_NE(unreadable->find("restore the exact object"), String::npos); + EXPECT_NE(unreadable->find("recreate the pool"), String::npos); + EXPECT_EQ(unreadable->find("REBUILD"), String::npos) + << "the diagnostic must not promise a command that cannot recover this exact object"; +} + +TEST(CASGCStuckRemoval, DiagnosticDoesNotAppendOrMutateBackend) +{ + DB::Cas::tests::CountingBackend backend; + const Layout layout("p"); + const RefWalkPlanRow row{ + .life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"removing"}, UInt128{7}), + .fold_state = {}, + .removal_started_round = 1, + .has_parent_fold_state = false, + .listed_hint = false, + .checkpoint_observation = std::nullopt, + .tail_observation = std::nullopt}; + const uint64_t puts_before = backend.putTotal(); + const uint64_t cas_before = backend.casPutTotal(); + EXPECT_TRUE(stuckRemovalWarning(row, 11, 10, layout)); + EXPECT_EQ(backend.putTotal(), puts_before); + EXPECT_EQ(backend.casPutTotal(), cas_before); + EXPECT_EQ(backend.deleteTotal(), 0u); +} + +TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_root_id = "test", + .gc_stuck_removal_rounds = 10}); + const Layout & layout = store->layout(); + const UInt128 gc_id{99}; + const UInt128 life_id{7}; + + const CatalogEntry removing{ + .ns = RootNamespace{"removing"}, + .state = NsState::Removing, + .incarnation = life_id, + .removal_started_round = 1}; + const auto catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog); + ASSERT_EQ(backend->casPut( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}}), catalog->token).outcome, + CasOutcome::Committed); + + CasFoldSeal seal; + seal.generation = 1; + seal.ref_lives.emplace(life_id, RefLifeFoldState{ + .coverage = RefCoverage{ + .classification = 4, + .hold = RefHold{ + .reason = HoldReason::BodyUndecodable, + .offending_position = RefTxnId{5, 6}, + .retry_count = 0, + .next_retry_round = 12}}}); + seal.condemned_summary[0] = CondemnedSummary{}; + ASSERT_EQ(backend->putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(seal)).outcome, PutOutcome::Done); + + GcState state; + state.lease = GcLease{.owner = gc_id, .seq = 1}; + state.round = 11; + state.gc_shards = 1; + state.snap_generation = 1; + state.snap_attempt = 1; + ASSERT_EQ(backend->putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); + + const uint64_t signals_before + = ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(removing.ns, life_id); + const String unreadable_ref_log_key = layout.refLogKey(life, RefTxnId{5, 6}); + const uint64_t append_puts_before = backend->putCount(unreadable_ref_log_key); + ScopedCasGcLogCapture log_capture; + Gc first_process(store, gc_id); + EXPECT_TRUE(first_process.runRegularRound().acquired_lease); + Gc restarted_process(store, gc_id); + EXPECT_TRUE(restarted_process.runRegularRound().acquired_lease); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load() - signals_before, 2u); + EXPECT_EQ(backend->putCount(unreadable_ref_log_key), append_puts_before) + << "the diagnostic cannot append the unreadable ref log"; + const String captured = log_capture.captured(); + EXPECT_EQ(std::count(captured.begin(), captured.end(), '\n'), 2u); + EXPECT_NE(captured.find(unreadable_ref_log_key), String::npos); + EXPECT_NE(captured.find("is unreadable"), String::npos); +} + +TEST(CASGCStuckRemoval, ZeroThresholdIsRefusedAtGcConstruction) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_root_id = "test", + .gc_stuck_removal_rounds = 0}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, + [&] { Gc gc(store, UInt128{1}); }); +} + +TEST(CASGCRefWalkPlan, UnmatchedAdoptedParentLifeIsObservedWithoutEnteringThePlan) +{ + const NamespaceLifePhysicalId current_life{2}; + const NamespaceLifePhysicalId unmatched_life = + hexToU128("fedcba98765432100123456789abcdef"); + RefCatalog catalog{.entries = {liveEntry("live", 2)}}; + const CasRefCatalog::Snapshot cut{ + .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + RefScanSummary scan; + scan.parent_ref_lives.emplace(current_life, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{2, 3}}}); + scan.parent_ref_lives.emplace(unmatched_life, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{9, 9}}}); + + const uint64_t events_before = + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load(); + const RefPlan plan = tests::buildRefWalkPlanForTest(scan, cut); + + EXPECT_EQ( + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load() - events_before, + 1u); + EXPECT_EQ(plan.droppedParentRows(), 1u); + EXPECT_EQ(plan.size(), 1u); + EXPECT_TRUE(plan.contains(current_life)); + EXPECT_FALSE(plan.contains(unmatched_life)); + EXPECT_FALSE(plan.parentFoldStates().contains(unmatched_life)); + EXPECT_FALSE(plan.successorFoldStates().contains(unmatched_life)); +} + +TEST(CASGCRefPlan, RoundInputOwnsObservationsAndSuccessorStateCannotChangePlan) +{ + /// This catches a plan that borrows the post-LIST observations or lets its successor state alias a + /// row. Replacing the owning `RoundInput`/`RefPlan` boundary with the former loose inputs, or + /// returning plan storage for the successor, must make this fail. + static_assert(!std::is_constructible_v); + static_assert(!std::is_default_constructible_v); + static_assert(!std::is_default_constructible_v); + static_assert(!std::is_assignable_v); + static_assert(!std::is_assignable_v); + + RefCatalog catalog; + catalog.entries = {liveEntry("live", 2)}; + CasRefCatalog::Snapshot cut{ + .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + + RefScanSummary observations; + observations.max_log_by_life.emplace(UInt128{2}, RefTxnId{2, 7}); + observations.parent_ref_lives.emplace(UInt128{2}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{2, 3}}}); + + const RefPlan plan = tests::buildRefWalkPlanForTest(observations, cut); + + /// The caller may reuse and mutate the sources after its one post-LIST/catalog observation and + /// plan construction. Those mutations cannot retarget the plan DEFER, fold, and publication use. + observations.max_log_by_life.at(UInt128{2}) = RefTxnId{2, 99}; + observations.parent_ref_lives.at(UInt128{2}).coverage.last_folded_ref_id = RefTxnId{2, 88}; + cut.catalog.entries.clear(); + + ASSERT_TRUE(plan.contains(UInt128{2})); + EXPECT_EQ(plan.row(UInt128{2}).tail_observation, (RefTxnId{2, 7})); + EXPECT_EQ(plan.row(UInt128{2}).fold_state.coverage.last_folded_ref_id, (RefTxnId{2, 3})); + + /// A fold/rebuild successor starts as a copy. It can earn a new cleanup state without changing the + /// immutable input that DEFER, the fold, and publication all consume. + auto successor_lives = plan.successorFoldStates(); + successor_lives.at(UInt128{2}).coverage.last_folded_ref_id = RefTxnId{2, 9}; + successor_lives.emplace(UInt128{9}, RefLifeFoldState{}); + EXPECT_EQ(plan.row(UInt128{2}).fold_state.coverage.last_folded_ref_id, (RefTxnId{2, 3})); + EXPECT_FALSE(plan.contains(UInt128{9})); +} diff --git a/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp new file mode 100644 index 000000000000..1ea4b8096e4d --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp @@ -0,0 +1,512 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int NETWORK_ERROR; +} + +/// Stage B Task 4-C: production birth wiring. `CasRefLedger::resolveNamespaceLife`, called from +/// `ensureRefTableRecovered`, resolves a namespace's real catalog life ONCE per table-open -- +/// create-if-absent, adopt an existing `Live`/`Removing` entry, or reconcile a stale `Creating` one via +/// `CasRefCatalog::reconcileStaleCreator` + `isCreatorFenceTerminal` -- so every ref-layer object a +/// mounted writer produces is keyed at a real, catalog-proven incarnation (spec INV-3), never the +/// Stage-A sentinel. +/// +/// OBLIGATION 3 (carried from Task 3's review, closed here): Task 3 could only enforce "`Creating` +/// forbids publication" (`CasRefCatalog::checkPublicationAdmittedOrThrow`) AT THE CATALOG LEVEL, because +/// nothing on the production ref-write path consulted the catalog at all. The refusal this suite pins +/// below rests on CONSTRUCTION, not a check: there is no `if (state == Creating) throw` anywhere in +/// `appendRefOps`'s path. `ensureRefTableRecovered` simply cannot make a table's runtime usable +/// (`rt.recovered` never becomes `true`, `rt.life` never gets set) while the catalog entry is `Creating` +/// under a fence that is not provably dead -- so no append can reach `commitRefChunk` for such a +/// namespace, by construction, stronger than any per-write check could prove. Stated here so nobody +/// later greps for a check and concludes the gap Task 3's review flagged is still open. +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +/// Fault the mandatory catalog's very first bootstrap write before it reaches durable storage. This +/// models a definite write failure, distinct from the acknowledgement-loss shape below: a retry must +/// still be allowed to prove a new pool, and the failed first attempt must not have published +/// `_pool_meta` without the catalog it makes mandatory. +class CatalogBootstrapPutFailsOnceBackend final : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (fail_once && key == Layout{"p"}.refCatalogKey()) + { + fail_once = false; + throw Poco::TimeoutException("CatalogBootstrapPutFailsOnceBackend: catalog PUT did not land"); + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } + +private: + bool fail_once = true; +}; + +class CatalogCancellationRaceBackend final : public CountingBackend +{ +public: + using CountingBackend::casPut; + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (race_armed && key == Layout{"p"}.refCatalogKey()) + { + race_armed = false; + on_catalog_cas(); + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + + bool race_armed = false; + std::function on_catalog_cas; +}; + +PoolPtr openPoolForBirthTest(const BackendPtr & backend, const String & server_root_id = "test") +{ + seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = server_root_id}); +} + +const CatalogEntry * findEntry(const RefCatalog & catalog, const RootNamespace & ns) +{ + for (const CatalogEntry & e : catalog.entries) + if (e.ns.string() == ns.string()) + return &e; + return nullptr; +} + +/// The same one-transaction publish `gtest_cas_ref_ckpt.cpp`'s `publishRef` drives: a namespace's first +/// append through the REAL append lane, which is also what triggers `resolveNamespaceLife`. +RefTxnId publishBirth(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + return store->appendRefOps(ns, MutationScope::ref(ref), + [&ref](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps(ref, ManifestRef{1, 1, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +} + +/// The happy path: nothing to reconcile, no pre-existing entry. The first append mints a fresh `Live` +/// catalog entry and keys the birth transaction at it -- not at the Stage-A sentinel. +TEST(CASRefCatalogBirthWiring, FirstOpenMintsALiveCatalogEntryAndKeysTheBirthAtIt) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const RootNamespace ns{"srv1/birth_wiring"}; + + const RefTxnId id = publishBirth(store, ns, "a"); + EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, store->layout()); + const CatalogEntry * entry = findEntry(snap.catalog, ns); + ASSERT_NE(entry, nullptr) << "the first open must mint a catalog entry"; + EXPECT_EQ(entry->state, NsState::Live); + EXPECT_NE(entry->incarnation, UInt128(0)); + EXPECT_EQ(entry->creator, std::nullopt) << "creator is forbidden outside Creating (strict grammar)"; + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + EXPECT_TRUE(backend->head(store->layout().refLogKey(life, id)).exists) + << "the birth transaction must be keyed at the REAL minted incarnation, not the Stage-A sentinel"; + EXPECT_FALSE(backend->head(store->layout().refLogKey(fixture::fixtureLife(ns), id)).exists) + << "and must NOT be keyed at the sentinel any more"; +} + +TEST(CASRefCatalogBirthWiring, CatalogLossAfterMountCannotRecreateAOneRowAuthority) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + + publishBirth(store, RootNamespace{"srv1/existing"}, "old"); + const auto catalog = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog); + ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog->token).kind, + DeleteOutcome::Kind::Deleted); + backend->resetCounts(); + + EXPECT_THROW(publishBirth(store, RootNamespace{"srv1/new"}, "new"), DB::Exception); + EXPECT_FALSE(backend->head(layout.refCatalogKey()).exists) + << "runtime loss must not be repaired with a one-row replacement authority"; + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->putTotal(), 0u) + << "the failed birth must not publish a checkpoint or ref-log body"; + EXPECT_EQ(backend->putOverwriteTotal(), 0u); +} + +TEST(CASRefCatalogBirthWiring, FailedCatalogBootstrapDoesNotPublishPoolMetaAndRetryConverges) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + + EXPECT_ANY_THROW(Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists) + << "a failed mandatory catalog bootstrap must leave no authoritative pool meta behind"; + EXPECT_FALSE(backend->head(layout.refCatalogKey()).exists); + + PoolPtr retry; + ASSERT_NO_THROW(retry = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); + EXPECT_TRUE(backend->head(layout.refCatalogKey()).exists); +} + +TEST(CASRefCatalogBirthWiring, LostCatalogBootstrapAcknowledgementLeavesOnlyRetryableCatalogResidue) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + backend->key_substr = layout.refCatalogKey(); + + EXPECT_ANY_THROW(Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); + EXPECT_TRUE(backend->head(layout.refCatalogKey()).exists) + << "the injected write must land before its acknowledgement is lost"; + + PoolPtr retry; + ASSERT_NO_THROW(retry = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); +} + +TEST(CASRefCatalogBirthWiring, BootstrapConflictExactReadsTheCanonicalEmptyCatalog) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const String canonical_empty = encodeRefCatalog(RefCatalog{}); + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), canonical_empty).outcome, PutOutcome::Done); + backend->resetCounts(); + + const CasRefCatalog::Snapshot snap = CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + EXPECT_TRUE(snap.catalog.entries.empty()); + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u) + << "a concurrent bootstrap winner must be exact-read before acceptance"; +} + +TEST(CASRefCatalogBirthWiring, BootstrapConflictRefusesANonemptyCatalog) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RefCatalog nonempty{.entries = {CatalogEntry{ + .ns = RootNamespace{"test/nonempty"}, .state = NsState::Live, .incarnation = UInt128{1}, .creator = std::nullopt}}}; + ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(nonempty)).outcome, PutOutcome::Done); + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { CasRefCatalog::initializeEmptyForNewPool(*backend, layout); }); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); +} + +TEST(CASRefCatalogBirthWiring, ExistingPoolMetaWithMissingCatalogStillFailsClosed) +{ + auto backend = std::make_shared(); + PoolPtr first = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const auto catalog = backend->get(first->layout().refCatalogKey()); + ASSERT_TRUE(catalog); + ASSERT_EQ(backend->deleteExact(first->layout().refCatalogKey(), catalog->token).kind, + DeleteOutcome::Kind::Deleted); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); +} + +TEST(CASRefCatalogBirthWiring, RestartFixturePreservesItsExistingNonemptyCatalog) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + seedPoolMetaForRestart(*backend); + const auto empty = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(empty); + + const RefCatalog nonempty{.entries = {CatalogEntry{ + .ns = RootNamespace{"test/preserved"}, .state = NsState::Live, .incarnation = UInt128{1}, .creator = std::nullopt}}}; + const String bytes = encodeRefCatalog(nonempty); + ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), bytes, empty->token).outcome, PutOutcome::Done); + const auto before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(before); + + seedPoolMetaForRestart(*backend); + const auto after = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(after); + EXPECT_EQ(after->bytes, before->bytes); + EXPECT_EQ(after->token, before->token); +} + +/// A namespace whose catalog entry is ALREADY `Live` (e.g. admitted by an earlier mount that this +/// runtime never cached) must be ADOPTED, never re-minted: `CasRefCatalog::createNamespace` refuses +/// outright once any entry exists, so `resolveNamespaceLife` has no create branch left to take here -- +/// only the adopt branch can succeed. +TEST(CASRefCatalogBirthWiring, AnExistingLiveEntryIsAdoptedRatherThanReminted) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/adopt_live"}; + + const CatalogEntry entry{.ns = ns, .state = NsState::Live, .incarnation = UInt128(0xcafe), + .creator = std::nullopt}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = store->writerEpoch(), + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + const RefTxnId id = publishBirth(store, ns, "a"); + EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + + /// The read result must outlive the returned pointer -- findEntry points into its entries. + const auto after_cut = CasRefCatalog::read(*backend, layout); + const CatalogEntry * after = findEntry(after_cut.catalog, ns); + ASSERT_NE(after, nullptr); + EXPECT_EQ(after->incarnation, UInt128(0xcafe)) << "adopted, not re-minted"; + EXPECT_EQ(after->state, NsState::Live); + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(0xcafe)); + EXPECT_TRUE(backend->head(layout.refLogKey(life, id)).exists); +} + +/// OBLIGATION 3, pinned through the PRODUCTION path: a `Creating` entry left by a DIFFERENT, still-live +/// (or at least not provably dead) actor refuses every append -- no test-only seam, no direct call to +/// `resolveNamespaceLife`/`reconcileStaleCreator`, just an ordinary `appendRefOps`. +TEST(CASRefCatalogBirthWiring, ANamespaceStuckCreatingUnderALiveForeignFenceRefusesProductionPublicationByConstruction) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/stuck_creating"}; + + /// A DIFFERENT actor's `Creating` entry naming a server root that never mounted at all -- + /// `isCreatorFenceTerminal`'s own doc: an ABSENT mount slot answers nothing about liveness, so it + /// is treated as NOT terminal (fail closed), never as proof of death. + const CreatorFence foreign_creator{.server_root_id = "ghost-server", .writer_epoch = 9, .fence_generation = 1}; + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(0xdead), + .creator = foreign_creator}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishBirth(store, ns, "a"); }); + + /// Nothing was written: the entry is exactly as observed, still Creating, still the foreign fence. + /// The read result must outlive the returned pointer -- findEntry points into its entries. + const auto still_cut = CasRefCatalog::read(*backend, layout); + const CatalogEntry * still = findEntry(still_cut.catalog, ns); + ASSERT_NE(still, nullptr); + EXPECT_EQ(*still, entry) << "a refused resolution must write nothing"; + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +/// The mirror image, and Task 3's own deferred obligation ("wire `reconcileStaleCreator` and pin it +/// with a test that drives reconciliation through the discovery path rather than by calling the +/// primitive directly"): a dead predecessor's `Creating` entry is reconciled onto THIS mount and +/// completed to `Live`, over the SAME incarnation -- resumption, not rebirth. +TEST(CASRefCatalogBirthWiring, AStaleCreatingEntryFromATerminatedForeignFenceIsReconciledThroughTheProductionPath) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend, "this-server"); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/reconciled"}; + + /// A dead predecessor's `Creating` entry: its mount lease carries the clean-farewell sentinel + /// (`min_active == UINT64_MAX`), one of `isCreatorFenceTerminal`'s three certificates of death. + const CreatorFence dead_creator{.server_root_id = "dead-server", .writer_epoch = 3, .fence_generation = 1}; + const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(0xbeef), + .creator = dead_creator}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + setWatermarkMinActive(*backend, layout, "dead-server", /*writer_epoch=*/3, + /*min_active=*/std::numeric_limits::max()); + + /// The production path resumes creation itself: reconciles the stale entry onto THIS mount's own + /// fence and completes it to `Live`, over the SAME incarnation the dead creator minted. + const RefTxnId id = publishBirth(store, ns, "a"); + EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + + /// The read result must outlive the returned pointer -- findEntry points into its entries. + const auto live_cut = CasRefCatalog::read(*backend, layout); + const CatalogEntry * live = findEntry(live_cut.catalog, ns); + ASSERT_NE(live, nullptr); + EXPECT_EQ(live->state, NsState::Live); + EXPECT_EQ(live->incarnation, UInt128(0xbeef)) << "the SAME incarnation throughout -- resumption, not rebirth"; + EXPECT_EQ(live->creator, std::nullopt); + + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(0xbeef)); + EXPECT_TRUE(backend->head(layout.refLogKey(life, id)).exists); +} + +TEST(CASRefCatalogBirthWiring, DropRefusesLiveCreatingFenceWithZeroCatalogMutation) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"drop_live_creator"}; + const CatalogEntry creating{ + .ns = ns, + .state = NsState::Creating, + .incarnation = UInt128{0xd001}, + .creator = CreatorFence{.server_root_id = "unproven-live", .writer_epoch = 7, .fence_generation = 1}}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{creating}); +} + +TEST(CASRefCatalogBirthWiring, DropDeletesTerminalCreatingExactlyAndLeavesCkptForJanitor) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"drop_terminal_creator"}; + const CatalogEntry creating{ + .ns = ns, + .state = NsState::Creating, + .incarnation = UInt128{0xd002}, + .creator = CreatorFence{.server_root_id = "dead-creator", .writer_epoch = 8, .fence_generation = 1}}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + setWatermarkMinActive(*backend, layout, "dead-creator", 8, std::numeric_limits::max()); + const NamespaceLifeId old_life = NamespaceLifeId::fromCatalogEntry(ns, creating.incarnation); + const String ckpt_key = layout.refCkptKey(old_life); + ASSERT_EQ(backend->putIfAbsent(ckpt_key, "stalled-ckpt").outcome, PutOutcome::Done); + backend->resetCounts(); + + store->dropNamespace(ns); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); + EXPECT_TRUE(backend->head(ckpt_key).exists); + EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + + const NamespaceLifeId reborn = store->namespaceLife(ns); + EXPECT_NE(reborn.incarnation, old_life.incarnation); +} + +TEST(CASRefCatalogBirthWiring, DropLosesExactCreatingRaceToReconciliationWithoutDeletingCkpt) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"drop_reconcile_race"}; + const CreatorFence old_creator{ + .server_root_id = "dead-racing-creator", .writer_epoch = 9, .fence_generation = 1}; + const CatalogEntry creating{ + .ns = ns, .state = NsState::Creating, .incarnation = UInt128{0xd003}, .creator = old_creator}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + setWatermarkMinActive( + *backend, layout, old_creator.server_root_id, old_creator.writer_epoch, + std::numeric_limits::max()); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, creating.incarnation); + const String ckpt_key = layout.refCkptKey(life); + ASSERT_EQ(backend->putIfAbsent(ckpt_key, "stalled-ckpt").outcome, PutOutcome::Done); + backend->on_catalog_cas = [&] + { + EXPECT_EQ(CasRefCatalog::reconcileStaleCreator( + *backend, layout, creating, + CreatorFence{.server_root_id = "replacement", .writer_epoch = 10, .fence_generation = 1}, + [](const CreatorFence &) { return true; }, store->fenceGeneration(), + [](uint64_t) {}), CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + }; + backend->race_armed = true; + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); + EXPECT_TRUE(backend->head(ckpt_key).exists); + const CasRefCatalog::Snapshot after = CasRefCatalog::read(*backend, layout); + ASSERT_EQ(after.catalog.entries.size(), 1u); + ASSERT_TRUE(after.catalog.entries.front().creator); + EXPECT_EQ(after.catalog.entries.front().creator->server_root_id, "replacement"); +} + +TEST(CASRefCatalogBirthWiring, FencedDropCannotCancelTerminalCreating) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"fenced_drop_terminal_creator"}; + const CatalogEntry creating{ + .ns = ns, + .state = NsState::Creating, + .incarnation = UInt128{0xd004}, + .creator = CreatorFence{.server_root_id = "dead-fenced-creator", .writer_epoch = 11, .fence_generation = 1}}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + setWatermarkMinActive(*backend, layout, "dead-fenced-creator", 11, std::numeric_limits::max()); + backend->resetCounts(); + + /// The first cancellation attempt passes its fence check, then loses its catalog CAS while the + /// local mount is re-armed at a new fence generation. The retry must re-check the caller fence and + /// refuse before another catalog mutation attempt. + backend->on_catalog_cas = [&] + { + rearmMountFenceAfterAnomalyForTest(store); + backend->failNextCasPut(layout.refCatalogKey()); + }; + backend->race_armed = true; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{creating}); +} + +TEST(CASRefCatalogBirthWiring, ExactOldLifeCannotCancelReplacementTerminalCreating) +{ + auto backend = std::make_shared(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"exact_old_life_terminal_creator"}; + const NamespaceLifeId predecessor = NamespaceLifeId::fromCatalogEntry(ns, UInt128{0xd005}); + const CatalogEntry successor{ + .ns = ns, + .state = NsState::Creating, + .incarnation = UInt128{0xd006}, + .creator = CreatorFence{.server_root_id = "dead-successor-creator", .writer_epoch = 12, .fence_generation = 1}}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, successor); + setWatermarkMinActive(*backend, layout, "dead-successor-creator", 12, std::numeric_limits::max()); + const String ckpt_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation)); + ASSERT_EQ(backend->putIfAbsent(ckpt_key, "successor-ckpt").outcome, PutOutcome::Done); + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(predecessor); }); + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); + EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{successor}); +} diff --git a/src/Disks/tests/gtest_cas_ref_chunk_preparation.cpp b/src/Disks/tests/gtest_cas_ref_chunk_preparation.cpp new file mode 100644 index 000000000000..c9cc317ab813 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_chunk_preparation.cpp @@ -0,0 +1,276 @@ +#include + +#include +#include +#include +#include "cas_test_helpers.h" +#include +#include + +#include + +#include +#include +#include +#include + +/// `prepareRefChunk` is the pure half of `commitRefChunk` (Stage B directive +/// `{#extract-prepare-ref-chunk}`): everything the append lane DECIDES before this chunk can have any +/// durable effect. This TU is where that purity is exercised, and it is deliberately backend-free -- +/// nothing below names a backend, a pool, a ledger instance or a clock, and nothing constructs one. The +/// mechanical guarantee is `static` on `prepareRefChunk` itself: with no `this` there is no member +/// backend, runtime, clock or lock reachable from inside it, so a future edit cannot quietly reach for +/// one and still compile here. +/// +/// What that buys is exactly what shows up below: every case is a direct call, so INV-2's chain-link +/// grammar is swept as a cross product -- including its negatives -- instead of being probed through +/// I/O. +/// +/// The value the extraction protects is pinned elsewhere on purpose: the equivalence fences in +/// `gtest_cas_ref_ckpt.cpp` assert that the durable key, the sealed bytes and the per-key request +/// counts a REAL append produces are unchanged. Those need a backend, so they live there and this TU +/// stays pure. + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; + +namespace +{ + +const RootNamespace kNs{"srv1/prep@cas@"}; +const Layout kLayout{"p"}; +/// `prepareRefChunk` takes a resolved catalog life (Stage B, Task 4-C), not a bare namespace; this TU +/// is deliberately backend-free (no catalog to resolve one from), so it threads the Stage-A sentinel +/// through EXPLICITLY as its own test input -- the same value production minted internally before +/// Task 4-C, so every golden byte/key assertion below is unchanged. +const NamespaceLifeId kLife = DB::Cas::tests::fixture::fixtureLife(kNs); + +RefOp birthOp() +{ + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + return op; +} + +RefOp epochSealOp() +{ + RefOp op; + op.kind = RefOpKind::EpochSeal; + return op; +} + +/// A minimal content op: the `AddPrecommit` shape (a pure add of a PRECOMMIT owner). A committed owner +/// is only ever reached by promoting a precommit, so this is the smallest legal content transition. +RefOp addPrecommitOp(const String & ref_name, const ManifestRef & manifest) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref_name, manifest}; + return op; +} + +ManifestRef mref(uint64_t seq) +{ + return ManifestRef{1, seq, 1}; +} + +/// One live namespace, born at `{1,1}`, as the state a later chunk prepares against. +RefTableState bornState() +{ + RefTableState state; + applyRefLogTxn(state, RefLogTxn{kNs.string(), RefTxnId{1, 1}, {birthOp()}, std::nullopt}); + return state; +} + +/// Asserts that preparation REFUSES with `CORRUPTED_DATA` -- the code every ref-log grammar violation +/// normalises to -- and that the message names the chain link, so a row cannot pass because some +/// unrelated validator happened to throw first. +template +void expectGrammarRefusal(F && body, const char * what) +{ + try + { + std::forward(body)(); + FAIL() << "expected a grammar refusal: " << what; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA) << what; + EXPECT_NE(e.message().find("prev_epoch_seal"), String::npos) + << what << " -- refused, but not by the chain-link rule; message: " << e.message(); + } +} + +/// `prepareRefChunk` CONSUMES its state, so this copies -- which also lets every caller below assert +/// afterwards that its own state was left alone. +CasRefLedger::PreparedRefChunk prepare(const RefTableState & state, const RefTxnId & id, + const std::optional & chain_link, + const std::vector & ops, uint64_t admitted_generation = 7) +{ + return CasRefLedger::prepareRefChunk(kLayout, kLife, state, id, chain_link, ops, admitted_generation); +} + +} + +/// The two things that actually become durable -- the key and the sealed body -- are both derivable +/// before any request, and both round-trip: the key parses back to the life and id it names, and the +/// bytes decode back to the very transaction that was prepared. +TEST(CASRefChunkPreparation, PreparedKeyAndSealedBytesAreCanonical) +{ + const RefTxnId id{1, 2}; /// the contiguous successor of the born state's `1-1` + const auto prepared = prepare(bornState(), id, std::nullopt, {addPrecommitOp("r1", mref(3))}); + + const auto parsed = kLayout.parseRefObjectKey(prepared.prepared_attempt.key); + ASSERT_TRUE(parsed.has_value()) << "the prepared key must be one of OUR ref-object keys"; + EXPECT_EQ(parsed->life_id, kLife.incarnation); + EXPECT_EQ(parsed->kind, RefObjectKind::Log); + EXPECT_EQ(parsed->txn_id, id); + EXPECT_EQ(prepared.prepared_attempt.key, kLayout.refLogKey(kLife, id)); + + const RefLogTxn decoded = decodeRefLogTxn( + openObject(FormatId::RefLog, prepared.prepared_attempt.bytes), kNs.string(), id); + EXPECT_EQ(decoded, prepared.chunk_txn) << "the sealed bytes must decode back to the prepared transaction"; + EXPECT_EQ(decoded.ns, kNs.string()); + EXPECT_EQ(decoded.txn_id, id); + ASSERT_EQ(decoded.ops.size(), 1u); + EXPECT_EQ(decoded.ops.front().kind, RefOpKind::OwnerTransition); +} + +/// The base id a later install re-presents is the greatest-applied of the state preparation STARTED +/// from -- not of the candidate it produced. Getting this backwards would let an install adopt a +/// candidate over a state that had moved on. +TEST(CASRefChunkPreparation, CandidateBaseIdIsGreatestApplied) +{ + const RefTableState state = bornState(); + const RefTxnId base = state.getGreatestApplied(); + ASSERT_EQ(base, (RefTxnId{1, 1})) << "precondition: the born state's greatest-applied is its birth"; + + const auto prepared = prepare(state, RefTxnId{1, 2}, std::nullopt, {addPrecommitOp("r1", mref(3))}); + EXPECT_EQ(prepared.candidate_base_id, base) << "the base id describes the state prepared FROM"; + EXPECT_EQ(prepared.candidate.getGreatestApplied(), (RefTxnId{1, 2})) + << "the candidate itself has this chunk applied"; + /// `prepare` handed over a COPY, so the caller's state cannot have been advanced -- the property the + /// real caller relies on when it re-presents `candidate_base_id` at install time. + EXPECT_EQ(state.getGreatestApplied(), base) << "preparation must not mutate the caller's state"; +} + +/// INV-2's chain-link grammar across the full cross product. Preparation runs the real validators, so +/// this sweeps both directions: where the link is required or forbidden, an ill-formed combination must +/// be REFUSED here -- before anything is durable -- rather than sealed into bytes and PUT. That +/// two-sided sweep is what the extraction buys: it needs no backend, so there is no reason not to cover +/// the negatives too. +/// +/// The base state is built per row, because a transaction id is only meaningful as the contiguous +/// successor of some stream (INV-1): a row cannot just assert a grammar rule on an id the stream would +/// never reach. +/// +/// Note which validator each row lands on, because the two halves of the rule are DISJOINT and live in +/// different steps of preparation: the required-iff half is `validateEpochSealGrammarContextual`, run by +/// the candidate apply; the forbidden-off-sequence-1 half is `validateEpochSealGrammarStructural`, run +/// by `encodeRefLogTxn` during the seal. Both are inside preparation, which is the point -- a chunk that +/// passes one and fails the other still fails before any durable effect. +TEST(CASRefChunkPreparation, ChainLinkRequiredExactlyOnSequenceOneOfNonGenesisEpoch) +{ + const std::vector ops{addPrecommitOp("r1", mref(3))}; + const RefTxnId epoch1_seal{1, 5}; /// the seal that closed epoch 1 + + /// From a namespace born at `1-1` (so `life_epoch == 1`). + /// Sequence > 1 of the genesis epoch: the link is FORBIDDEN. + EXPECT_NO_THROW(prepare(bornState(), RefTxnId{1, 2}, std::nullopt, ops)) + << "seq >1 with no link is the ordinary case"; + expectGrammarRefusal([&] { prepare(bornState(), RefTxnId{1, 2}, epoch1_seal, ops); }, + "a link at sequence >1 is forbidden and must be refused before any durable effect"); + + /// Sequence 1 of an epoch ABOVE genesis: the link is REQUIRED. + EXPECT_NO_THROW(prepare(bornState(), RefTxnId{2, 1}, epoch1_seal, ops)) + << "seq 1 of a higher epoch names the seal that closed the previous one"; + expectGrammarRefusal([&] { prepare(bornState(), RefTxnId{2, 1}, std::nullopt, ops); }, + "seq 1 of a higher epoch without a link must be refused -- 'no seal' is a fact " + "about the stream, not a defaulted field"); + + /// Genesis itself: sequence 1 of the birth epoch has nothing to name, so a link is FORBIDDEN. + EXPECT_NO_THROW(prepare(RefTableState{}, RefTxnId{3, 1}, std::nullopt, {birthOp()})) + << "a genesis birth at sequence 1 finds nothing to name"; + expectGrammarRefusal([&] { prepare(RefTableState{}, RefTxnId{3, 1}, epoch1_seal, {birthOp()}); }, + "a link on the birth transaction itself must be refused"); + + /// Whatever the grammar admitted, the sealed bytes carry exactly that link and nothing else. + const auto linked = prepare(bornState(), RefTxnId{2, 1}, epoch1_seal, ops); + ASSERT_TRUE(linked.chunk_txn.prev_epoch_seal.has_value()); + EXPECT_EQ(*linked.chunk_txn.prev_epoch_seal, epoch1_seal); + const RefLogTxn decoded = decodeRefLogTxn( + openObject(FormatId::RefLog, linked.prepared_attempt.bytes), kNs.string(), RefTxnId{2, 1}); + EXPECT_EQ(decoded.prev_epoch_seal, linked.chunk_txn.prev_epoch_seal) + << "the link must survive into the bytes that would become durable"; +} + +/// The birth `_ckpt` contribution is PREPARED here and published by `commitRefChunk`, because +/// publishing it is a birth chunk's first durable effect. Preparation therefore owes two things: the +/// value only for a birth, and the one fact no later writer can recover -- `life_epoch`. +TEST(CASRefChunkPreparation, BirthContributionSetOnlyForNamespaceBirth) +{ + /// A birth chunk at epoch 3: the contribution exists and names THIS transaction's writer epoch. + const RefTxnId birth_id{3, 1}; + const auto born = prepare(RefTableState{}, birth_id, std::nullopt, {birthOp()}); + ASSERT_TRUE(born.birth_contribution.has_value()); + ASSERT_TRUE(born.birth_contribution->life_epoch.has_value()); + EXPECT_EQ(*born.birth_contribution->life_epoch, birth_id.writer_epoch); + EXPECT_FALSE(born.birth_contribution->checkpoint_snapshot_id.has_value()) + << "the birth contributes life_epoch and nothing else -- the publisher owns the checkpoint field"; + EXPECT_FALSE(born.birth_contribution->last_epoch_seal.has_value()); + + /// An ordinary content chunk contributes nothing: a second `_ckpt` write here would be a request the + /// append lane does not owe. + const auto ordinary = prepare(bornState(), RefTxnId{1, 2}, std::nullopt, {addPrecommitOp("r1", mref(3))}); + EXPECT_FALSE(ordinary.birth_contribution.has_value()); + + /// A birth op mixed into a larger chunk still counts -- the check is over the whole chunk. + const auto mixed = prepare(RefTableState{}, RefTxnId{5, 1}, std::nullopt, + {birthOp(), addPrecommitOp("r1", mref(3))}); + ASSERT_TRUE(mixed.birth_contribution.has_value()); + EXPECT_EQ(*mixed.birth_contribution->life_epoch, 5u); +} + +TEST(CASRefChunkPreparation, CommitContributionCarriesFrontierAndOnlyMatchingSeal) +{ + const RefTxnId ordinary_id{1, 2}; + const auto ordinary = prepare(bornState(), ordinary_id, std::nullopt, + {addPrecommitOp("r1", mref(3))}); + EXPECT_EQ(ordinary.commit_contribution.committed_through, ordinary_id); + EXPECT_FALSE(ordinary.commit_contribution.life_epoch.has_value()); + EXPECT_FALSE(ordinary.commit_contribution.checkpoint_snapshot_id.has_value()); + EXPECT_FALSE(ordinary.commit_contribution.last_epoch_seal.has_value()); + + const RefTxnId seal_id{1, 2}; + const auto seal = prepare(bornState(), seal_id, std::nullopt, {epochSealOp()}); + EXPECT_EQ(seal.commit_contribution.committed_through, seal_id); + EXPECT_EQ(seal.commit_contribution.last_epoch_seal, seal_id) + << "an epoch seal and its committed frontier must be one checkpoint contribution"; + EXPECT_FALSE(seal.commit_contribution.life_epoch.has_value()); + EXPECT_FALSE(seal.commit_contribution.checkpoint_snapshot_id.has_value()); +} + +/// The attempt exists so that an `Unresolved` PUT -- an object that may be durable -- can be recorded by +/// a MOVE and nothing else. That only holds if every field is already populated before the request goes +/// out, so this asserts the whole struct is complete at the end of preparation. +TEST(CASRefChunkPreparation, PreparedAttemptIsCompleteBeforeAnyDurableEffect) +{ + const RefTxnId id{1, 2}; + const auto prepared = prepare(bornState(), id, std::nullopt, {addPrecommitOp("r1", mref(3))}, /*admitted_generation=*/42); + + EXPECT_EQ(prepared.prepared_attempt.txn_id, id); + EXPECT_FALSE(prepared.prepared_attempt.key.empty()); + EXPECT_FALSE(prepared.prepared_attempt.bytes.empty()); + EXPECT_EQ(prepared.prepared_attempt.admitted_fence_generation, 42u) + << "the attempt carries the generation it was ADMITTED under, not a current reading"; + + /// Nothing left to build: the key and body the request will read are already the canonical ones, so + /// the arming block's only remaining work really is the move it declares itself to be. + EXPECT_EQ(prepared.prepared_attempt.key, kLayout.refLogKey(kLife, id)); + EXPECT_EQ(prepared.prepared_attempt.bytes, + sealObject(FormatId::RefLog, encodeRefLogTxn(prepared.chunk_txn))); +} diff --git a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp new file mode 100644 index 000000000000..2cdc32ff1851 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp @@ -0,0 +1,919 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int LIMIT_EXCEEDED; +extern const int CORRUPTED_DATA; +extern const int NETWORK_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASRefBatchFlushes; +extern const Event CASRefBatchedMutations; +extern const Event CASRefSnapshotPublishDispatched; +} + +/// Task 8 (stage-1 §3 "Budget: counts only, chunked flush"): the counts-only admission caps -- +/// `ref_txn_max_ops` (5000), the carve item cap `kMaxRefBatch` (1000), and the per-op size cap +/// `ref_op_max_bytes` (4096 bytes on normal-class ops) -- plus their failure-isolation contract: a +/// single item whose own op count, or whose one op's encoded size, exceeds its cap fails ALONE; a +/// neighbor co-batched into the same flush still commits. `ref_txn_max_ops` is checked exactly (the +/// `build_ops` result's size), and the per-op cap is checked by encoding exactly one op at a time -- +/// no accumulation, matching the admission machinery this replaces. T9 (removal-class detection by +/// op inspection) and T10 (chunked flush across a whole-batch op-count overflow) extend this file; +/// this task adds only the per-item / per-op isolation tests and the canonical round-trip leg of +/// test 12 (the maximum legally-admissible normal-class transaction). +/// +/// The suite name is prefixed `RefWriter` so it is covered by the `RefWriter*` unit-test gate filter. + +using namespace DB::Cas; +using DB::Cas::tests::committedRow; +using DB::Cas::tests::minimalLiveSnapshot; +using DB::Cas::tests::writeRefSnapshotRaw; + +namespace +{ + +PoolPtr openPool(const BackendPtr & backend) +{ + /// A fresh pool with no residue, mirroring the T7 carve suite's `openPool`. + /// + /// The FROZEN clock is load-bearing, not hygiene. `CasMountRuntime::refAppendFenceOk` gates every + /// controlled attempt against `boot_ms_fn`, and with the compiled defaults (mount_lease_ttl_ms + /// 30000, safety margin 7000) a pool opened on the REAL clock fences itself ~23s later — no + /// background renewal advances that deadline in a unit-test pool. Every test in this suite is + /// about chunking and op-caps, none about wall-clock lease behaviour, so any of them that runs + /// long enough simply dies of an unrelated fence trip: `DropNamespaceOverOpCapSucceeds` (5200 + /// refs) takes 43-65s under a sanitizer and failed deterministically on all three sanitizer CI + /// builds with `txn is UNCERTAIN (retry budget exhausted)` — the pre-attempt fence reject, not a + /// real retry exhaustion. Same artifact, same fix as + /// `CASPartWriteTxn.ManifestCapEncodedBytesJustUnderStagesSuccessfully` (2026-07-18): decouple the + /// fence from execution speed. The waits in this file are `steady_clock` timeouts on futures and + /// condvars, which are unaffected by this injection. + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", .boot_ms_fn = [] { return uint64_t{0}; }}); +} + +/// A legal blob-free part: stage an empty manifest, precommit, promote -- enough to leave one +/// committed ref (and a `Live` table) that a later co-batched item can join. +/// +/// Stage B (Task 4-C): pin `ns` to the sentinel before the first real touch -- the ONE choke point +/// every test in this file uses to birth its namespace, before any `launchAppendOps`/`launchAppend`/ +/// `launchDrop` call. Several tests separately compute an expected key via +/// `DB::Cas::tests::fixture::fixtureLife(ns)` for verification/fault injection; without this the real +/// production birth mints a random incarnation and those computed keys land nowhere real. +void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +/// One queued append (or drop) driven on its own thread; the future becomes ready only when the call +/// RETURNS (normally or by throwing). Mirrors `gtest_cas_ref_carve.cpp`'s `Caller`/`launchDrop`. +struct Caller +{ + std::thread t; + std::future fut; +}; + +Caller launchAppend(const PoolPtr & store, const RootNamespace & ns, MutationScope scope, + std::function(const RefTableState &)> build_ops) +{ + auto prom = std::make_shared>(); + std::future fut = prom->get_future(); + std::thread t([store, ns, scope, build_ops, prom] + { + std::exception_ptr err; + try { store->appendRefOps(ns, scope, build_ops, RootMutationOrigin::Writer, RootMutationKind::Publish); } + catch (...) { err = std::current_exception(); } + prom->set_value(err); + }); + return Caller{std::move(t), std::move(fut)}; +} + +Caller launchDrop(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + auto prom = std::make_shared>(); + std::future fut = prom->get_future(); + std::thread t([store, ns, ref, prom] + { + std::exception_ptr err; + try { store->dropRef(ns, ref); } + catch (...) { err = std::current_exception(); } + prom->set_value(err); + }); + return Caller{std::move(t), std::move(fut)}; +} + +/// `n` filler ops for a `build_ops` result whose only purpose is to overflow the per-item op-count +/// cap. They are NOT inert-when-applied: a default-constructed op is a `NamespaceBirth`, which throws +/// `CORRUPTED_DATA` ("namespace_birth while already Live") if it were ever applied to the pre-published +/// namespace. The load-bearing safety property is that the count check fires BEFORE any of these ops is +/// applied or otherwise inspected. +std::vector fillerOps(size_t n) +{ + return std::vector(n, RefOp{}); +} + +/// A zero-padded ref name for index `i`, so `kTotalRefs` names sort in the same order as their index +/// (the snapshot fixture's committed rows must already be sorted by `ref_name`). +String paddedRefName(size_t i) +{ + String s = std::to_string(i); + return "ref_" + String(6 - s.size(), '0') + s; +} + +/// A single `SetPublishedAt` op whose `ref_name` is padded so its OWN encoded size (`encodedOpSize`) +/// is exactly `target_bytes`. Every added 'a' is one un-escaped byte in the JSON ref-name string, so +/// the size grows one-for-one; `checkCanonicalRefName` imposes no length limit, so this stays a +/// valid, merely over-long, canonical ref name. +RefOp paddedSetPublishedAtOp(size_t target_bytes) +{ + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = ManifestRef{1, 1, 1}; + op.published_at_ms = 0; + const size_t base = encodedOpSize(op); + op.ref_name = "r" + String(target_bytes - base, 'a'); + return op; +} + +/// Blocks the FIRST flush's leader in the pre-carve window until `expected_pending` items are queued, +/// forcing a deterministic multi-item batch (mirrors `gtest_cas_ref_carve.cpp`'s `CaseSync`/pre-carve +/// hook pattern). Only the first carve blocks; retries proceed straight through. +struct CaseSync +{ + std::mutex m; + std::condition_variable cv; + bool entered = false; +}; + +void armPreCarveBlock(const PoolPtr & store, const RootNamespace & ns, const std::shared_ptr & sync, size_t expected_pending) +{ + store->setRefPreCarveHookForTest([sync, store, ns, expected_pending] + { + std::unique_lock lk(sync->m); + if (sync->entered) + return; + sync->entered = true; + sync->cv.notify_all(); + /// Bounded (10s) so a staging bug bounds the wait instead of blocking the whole suite. + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return store->refQueuePendingForTest(ns) >= expected_pending; }); + }); +} + +void waitEntered(const std::shared_ptr & sync) +{ + std::unique_lock lk(sync->m); + sync->cv.wait_for(lk, std::chrono::seconds(10), [&] { return sync->entered; }); +} + +void waitPendingAtLeast(const PoolPtr & store, const RootNamespace & ns, size_t n) +{ + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->refQueuePendingForTest(ns) < n && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); +} + +/// Asserts `err` is non-null and carries EXACTLY `expected_code` -- distinguishes the new counts-only +/// admission checks (`LIMIT_EXCEEDED`) from any other per-item validation failure. +void expectFailedWithCode(const std::exception_ptr & err, int expected_code, const char * what) +{ + ASSERT_TRUE(err != nullptr) << what << ": the caller must observe the admission-cap error"; + try + { + std::rethrow_exception(err); + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code) << what; + } +} + +} + +/// Test 10 (spec §3 "Oversized item / oversized op fail alone"): an item whose OWN op count exceeds +/// `ref_txn_max_ops` fails alone -- its ops never enter the batch's transaction -- and a co-batched +/// neighbor still commits. +TEST(CASRefWriterChunkedFlush, OversizedItemFailsAlone) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/chunked_oversized_item"}; + publishEmptyPart(store, ns, "neighbor"); + ASSERT_TRUE(store->resolveRef(ns, "neighbor").has_value()); + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + + Caller oversized = launchAppend(store, ns, MutationScope::ref("oversized"), + [](const RefTableState &) -> std::vector { return fillerOps(ref_txn_max_ops + 1); }); + waitEntered(sync); + Caller neighbor = launchDrop(store, ns, "neighbor"); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); /// release the pre-carve hook now its (>=2 pending) predicate holds + + ASSERT_EQ(oversized.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "oversized item must not hang"; + ASSERT_EQ(neighbor.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "neighbor must not hang"; + const std::exception_ptr oversized_err = oversized.fut.get(); + const std::exception_ptr neighbor_err = neighbor.fut.get(); + oversized.t.join(); + neighbor.t.join(); + store->setRefPreCarveHookForTest(nullptr); + + expectFailedWithCode(oversized_err, DB::ErrorCodes::LIMIT_EXCEEDED, "oversized item (op count)"); + EXPECT_TRUE(neighbor_err == nullptr) << "the co-batched neighbor must commit despite the oversized item"; + EXPECT_FALSE(store->resolveRef(ns, "neighbor").has_value()) << "neighbor's drop must have committed"; +} + +/// Test 10, second leg: one op whose OWN encoded size exceeds `ref_op_max_bytes` (a maximum-length +/// ref name -- `checkCanonicalRefName` imposes no length limit) fails only its item; a co-batched +/// neighbor still commits. +TEST(CASRefWriterChunkedFlush, OversizedOpFailsItsItemAlone) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/chunked_oversized_op"}; + publishEmptyPart(store, ns, "neighbor"); + ASSERT_TRUE(store->resolveRef(ns, "neighbor").has_value()); + + const RefOp oversized_op = paddedSetPublishedAtOp(ref_op_max_bytes + 1); + ASSERT_GT(encodedOpSize(oversized_op), ref_op_max_bytes); + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + + Caller oversized = launchAppend(store, ns, MutationScope::ref("oversized_op"), + [oversized_op](const RefTableState &) -> std::vector { return {oversized_op}; }); + waitEntered(sync); + Caller neighbor = launchDrop(store, ns, "neighbor"); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); + + ASSERT_EQ(oversized.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "oversized op item must not hang"; + ASSERT_EQ(neighbor.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "neighbor must not hang"; + const std::exception_ptr oversized_err = oversized.fut.get(); + const std::exception_ptr neighbor_err = neighbor.fut.get(); + oversized.t.join(); + neighbor.t.join(); + store->setRefPreCarveHookForTest(nullptr); + + expectFailedWithCode(oversized_err, DB::ErrorCodes::LIMIT_EXCEEDED, "oversized op"); + EXPECT_TRUE(neighbor_err == nullptr) << "the co-batched neighbor must commit despite the oversized op"; + EXPECT_FALSE(store->resolveRef(ns, "neighbor").has_value()) << "neighbor's drop must have committed"; +} + +/// Test 12, canonical round-trip leg: the maximum legally-admissible normal-class transaction under +/// the new counts-only caps -- `ref_txn_max_ops` ops, each padded to exactly `ref_op_max_bytes` -- +/// round-trips comfortably under the whole-transaction `ref_txn_max_bytes` decode cap (5000 * 4096 = +/// 20,480,000 bytes, with framing headroom to spare). Pure codec-level: proves the two counts-only +/// caps compose without ever approaching the byte cap the encode-side estimation machinery used to +/// police. +TEST(CASRefWriterChunkedFlush, CanonicalMaxTransactionRoundTrips) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + txn.ops.reserve(ref_txn_max_ops); + for (size_t i = 0; i < ref_txn_max_ops; ++i) + { + RefOp op = paddedSetPublishedAtOp(ref_op_max_bytes); + ASSERT_EQ(encodedOpSize(op), ref_op_max_bytes); + txn.ops.push_back(std::move(op)); + } + + const String bytes = encodeRefLogTxn(txn); + /// Every op contributes exactly `ref_op_max_bytes`; header/meta/trailer framing adds strictly + /// more on top, and the whole thing still stays well under the 20 MiB decode cap. + EXPECT_GT(bytes.size(), ref_txn_max_ops * ref_op_max_bytes); + EXPECT_LT(bytes.size(), ref_txn_max_bytes); + + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded.ops.size(), ref_txn_max_ops); + EXPECT_EQ(decoded, txn); +} + +/// Test 11 (spec §3 "Removal-class detection, falsifiably"): `dropNamespace` over a table with +/// > `ref_txn_max_ops` committed refs builds ONE transaction whose ops (one `owner_transition` +/// removal per ref, plus a terminal `remove_namespace`) exceed the normal-class op-count cap -- +/// and must still succeed, because removal-class is byte-budgeted (`ref_removal_max_bytes`, 64 MiB) +/// and has no op-count cap. Seeded via a raw snapshot (not `kTotalRefs` individual writer round-trips +/// through `publishEmptyPart`) so the fixture stays fast; the writer never touches these rows until +/// `dropNamespace` itself builds the one removal transaction. +TEST(CASRefWriterChunkedFlush, DropNamespaceOverOpCapSucceeds) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/dropns_over_cap"}; + constexpr size_t kTotalRefs = static_cast(ref_txn_max_ops) + 200; + + /// Open the store FIRST (still untouched for `ns`) so the seeded snapshot can use THIS mount's own + /// writer_epoch: namespace recovery is per-namespace and lazy (first touch), so writing the raw + /// fixture directly to `backend` after open, but before `ns` is ever touched, is observed identically + /// to writing it before open. + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + /// Stage B (Task 4-C): pin `ns` to the sentinel now, before the raw snapshot below -- `listRefs`/ + /// `dropNamespace` further down are real production reads that trigger `resolveNamespaceLife`, + /// which for an UNADMITTED namespace mints a fresh RANDOM incarnation rather than adopting the + /// sentinel the raw fixture wrote at. Pinning first makes them adopt it instead. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + + /// Ids are PER-NAMESPACE and derived from the table's own `greatest_applied` (INV-1), so seeding + /// `ns` at `{epoch, 1}` is all this fixture has to do: the `dropNamespace` below derives `{epoch, 2}` + /// from the seeded snapshot, and no other namespace's traffic can move it. + std::vector committed; + committed.reserve(kTotalRefs); + for (size_t i = 0; i < kTotalRefs; ++i) + committed.push_back(committedRow(paddedRefName(i), ManifestRef{epoch, i + 1, 1})); + ASSERT_GT(committed.size(), ref_txn_max_ops); + + /// Recovery's checkpoint anchor includes the same-id ordinary log. The synthetic snapshot stands + /// for a long prior history, while this genesis record supplies the retained non-seal witness the + /// real publisher would necessarily leave at the selected id. + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = RefTxnId{epoch, 1}, + .ops = {DB::Cas::tests::namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), RefTxnId{epoch, 1}, committed)); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = epoch, + .committed_through = RefTxnId{epoch, 1}, + .checkpoint_snapshot_id = RefTxnId{epoch, 1}, + .last_epoch_seal = std::nullopt, + }); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + backend->resetCounts(); + ASSERT_EQ(store->listRefs(ns).size(), kTotalRefs); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{epoch, 1})), 1u); + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, RefTxnId{epoch, 1})), 1u); + + DropNamespaceStats stats; + EXPECT_NO_THROW(stats = store->dropNamespace(ns)); + EXPECT_EQ(stats.committed_refs, kTotalRefs); + EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries.front().state, NsState::Removing); +} + +/// Test 11, second leg: `WholeShard` scope ALONE is not the removal-class discriminator -- the +/// stale-precommit reclaim sweep is also `WholeShard`-scoped but is not removal-class +/// (`CasRefLedger.cpp` ~:1979). Only a SYNTHETIC item can pin this: the production stale-precommit +/// sweep self-limits its own chunk size to the op cap, so running it proves nothing (spec's own +/// warning). This item drives `MutationScope::wholeShard()` directly with ops that contain NO +/// `RemoveNamespace` op -- if classification were keyed on scope instead of op inspection, this would +/// be wrongly treated as removal-class and admitted; op-inspection correctly rejects it under the +/// ordinary normal-class op-count cap, exactly like `OversizedItemFailsAlone` above. +TEST(CASRefWriterChunkedFlush, SyntheticWholeShardNonRemovalRejected) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/synthetic_wholeshard_nonremoval"}; + + Caller synthetic = launchAppend(store, ns, MutationScope::wholeShard(), + [](const RefTableState &) -> std::vector { return fillerOps(ref_txn_max_ops + 1); }); + ASSERT_EQ(synthetic.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "synthetic WholeShard item must not hang"; + const std::exception_ptr err = synthetic.fut.get(); + synthetic.t.join(); + + expectFailedWithCode(err, DB::ErrorCodes::LIMIT_EXCEEDED, + "synthetic WholeShard-scoped item with non-removal ops over the op cap"); +} + +/// =================================================================================== +/// Task 10 (spec §3 "Chunked flush, where each chunk is a complete commit boundary"): when admitting +/// the next item's ops would exceed `ref_txn_max_ops`, the leader commits the accumulated chunk as a +/// COMPLETE ref-log transaction (real id, PUT, apply, tail, metrics, survivor completion + waiter +/// wakeups, snapshot scheduling), reseeds `working`/the trial-id high-water mark from the now-live +/// state, and continues into a fresh chunk -- so one tenure can emit several transactions, each a valid +/// persisted prefix. The failure-isolation and tenure-containment contracts are pinned below. +/// =================================================================================== + +namespace +{ + +/// The `_log/`-PUT fault seam these tests are built on now lives in `cas_test_helpers.h`, next to +/// `CountingBackend` it derives from: `gtest_cas_ref_install_safety.cpp` needs the SAME seam (spec §A1 +/// sites 2 and 3 both turn on what happens when a `_log/` PUT's response is lost), and two copies of a +/// fault backend would drift apart. +using DB::Cas::tests::ChunkFaultBackend; + +PoolPtr openPoolWith(const BackendPtr & backend, PoolConfig cfg) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + cfg.pool_prefix = "p"; + cfg.server_root_id = "test"; + /// Same frozen clock as `openPool` above, and for the same reason — see its comment. Defaulted + /// rather than forced, so a future test that IS about lease timing can still supply its own. + if (!cfg.boot_ms_fn) + cfg.boot_ms_fn = [] { return uint64_t{0}; }; + return Pool::open(backend, cfg); +} + +/// `num_pairs` add-then-remove precommit op pairs (2 * `num_pairs` ops total) for distinct refs +/// (`prefix` + zero-padded index) each naming a distinct valid manifest. Every pair adds a precommit +/// binding and immediately removes it, so the LIVE state (the `precommits` set, the committed COW map, +/// the owned-manifest index) stays ~empty throughout the whole transaction -- keeping the per-op +/// `admits` preview and the sanitizer-only body-counter assert O(1), so validating a maximal chunk of +/// thousands of ops stays O(ops), not O(ops^2). It is the OP COUNT (not the resident state) that drives +/// the chunk split under test; each op is tiny (well under `ref_op_max_bytes`), so the whole run is +/// admissible on a `Live` namespace. The durable transaction still carries every op verbatim, so a +/// chunk's ops can be compared against the exact expected vector. +std::vector addRemovePrecommitPairs(const String & prefix, size_t num_pairs, uint64_t manifest_epoch) +{ + std::vector ops; + ops.reserve(num_pairs * 2); + for (size_t i = 0; i < num_pairs; ++i) + { + const String ref = prefix + paddedRefName(i); + const ManifestRef manifest{manifest_epoch, i + 1, 1}; + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref, manifest}; + ops.push_back(std::move(add)); + RefOp remove; + remove.kind = RefOpKind::OwnerTransition; + remove.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref, manifest}; + ops.push_back(std::move(remove)); + } + return ops; +} + +/// Every durable `_log/` transaction for `ns`, decoded, sorted ascending by transaction id. Undecodable +/// objects (e.g. the foreign bytes a `ForeignConflict` fault lands) are skipped so a corrupt object never +/// breaks the inventory. Reads the backend directly (no Pool cache). +std::vector listLogTxns(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const RootNamespace & ns) +{ + std::vector ids; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation + && parsed->kind == RefObjectKind::Log) + ids.push_back(parsed->txn_id); + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + std::sort(ids.begin(), ids.end(), [](const RefTxnId & a, const RefTxnId & b) { return a < b; }); + std::vector txns; + for (const RefTxnId & id : ids) + { + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + if (!got) + continue; + try + { + txns.push_back(decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id)); + } + catch (...) // NOLINT(bugprone-empty-catch): best-effort helper -- an undecodable txn is simply skipped, not asserted on + { + } + } + return txns; +} + +/// One queued append driven on its own thread, capturing BOTH the committed transaction id (on success) +/// and the exception (on failure); `build_calls` (when non-null) counts `build_ops` invocations to pin +/// the at-most-once contract across chunk boundaries. The ops are precomputed and returned verbatim, so a +/// second invocation (a bug) is caught by the counter, not masked by a state-dependent rebuild. +struct AppendResult +{ + std::exception_ptr err; + RefTxnId id{}; +}; + +struct AppendCaller +{ + std::thread t; + std::future fut; +}; + +AppendCaller launchAppendOps(const PoolPtr & store, const RootNamespace & ns, MutationScope scope, + std::vector ops, std::shared_ptr> build_calls) +{ + auto prom = std::make_shared>(); + std::future fut = prom->get_future(); + auto build_ops = [captured_ops = std::move(ops), build_calls](const RefTableState &) -> std::vector + { + if (build_calls) + build_calls->fetch_add(1); + return captured_ops; + }; + std::thread t([store, ns, scope, build_ops, prom] + { + AppendResult r; + try { r.id = store->appendRefOps(ns, scope, build_ops, RootMutationOrigin::Writer, RootMutationKind::Publish); } + catch (...) { r.err = std::current_exception(); } + prom->set_value(r); + }); + return AppendCaller{std::move(t), std::move(fut)}; +} + +} + +/// Test 9 (happy path): a carve whose total ops exceed `ref_txn_max_ops` emits >= 2 ref-log transactions +/// in ONE leader tenure. Three items (2000 ops each = 6000 > 5000) split into chunk 1 = {item_a,item_b} +/// (4000 ops, one id) and chunk 2 = {item_c} (2000 ops, the next id). Per-chunk assertions: committed +/// ids (co-chunk survivors share one real id; the next chunk allocates the next), tail counters (one per +/// chunk), per-chunk metrics (`CASRefBatchFlushes` once per chunk, `CASRefBatchedMutations` counting +/// survivors per chunk), follower wakeups (both followers return their correct real id -> completed + +/// woken at their chunk's commit), `build_ops` at-most-once (invocation counters == 1), and folded state +/// == the sequential result (the two durable transactions carry exactly item_a++item_b, then item_c). +TEST(CASRefWriterChunkedFlush, ChunkedFlushCommitsPerChunk) +{ + auto backend = std::make_shared(); + /// Default thresholds: this handful of transactions never crosses the snapshot-publish threshold, so + /// no background publish interleaves and the tail/metric deltas below are exact. + auto store = openPool(backend); + const DB::Cas::Layout & layout = store->layout(); + const RootNamespace ns{"srv1/chunked_commits_per_chunk"}; + publishEmptyPart(store, ns, "seed"); + ASSERT_TRUE(store->resolveRef(ns, "seed").has_value()); + + /// 2000 ops per item (1000 add/remove pairs) -> 6000 > ref_txn_max_ops (5000): chunk 1 = + /// {item_a,item_b} (4000), chunk 2 = {item_c} (2000). + const std::vector ops1 = addRemovePrecommitPairs("aaa_", 1000, 900000001); + const std::vector ops2 = addRemovePrecommitPairs("bbb_", 1000, 900000002); + const std::vector ops3 = addRemovePrecommitPairs("ccc_", 1000, 900000003); + auto c1 = std::make_shared>(0); + auto c2 = std::make_shared>(0); + auto c3 = std::make_shared>(0); + + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + const uint64_t flushes_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes].load(); + const uint64_t mutations_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations].load(); + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 3); + /// Serialise the enqueue order so the batch is exactly [item_a(leader), item_b, item_c]. + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), ops1, c1); + waitEntered(sync); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), ops2, c2); + waitPendingAtLeast(store, ns, 2); + AppendCaller c = launchAppendOps(store, ns, MutationScope::ref("item_c"), ops3, c3); + waitPendingAtLeast(store, ns, 3); + sync->cv.notify_all(); + + ASSERT_EQ(a.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "item_a must not hang"; + ASSERT_EQ(b.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "item_b must not hang"; + ASSERT_EQ(c.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "item_c must not hang"; + const AppendResult ra = a.fut.get(); + const AppendResult rb = b.fut.get(); + const AppendResult rc = c.fut.get(); + a.t.join(); + b.t.join(); + c.t.join(); + store->setRefPreCarveHookForTest(nullptr); + + ASSERT_TRUE(ra.err == nullptr) << "item_a must commit"; + ASSERT_TRUE(rb.err == nullptr) << "item_b must commit"; + ASSERT_TRUE(rc.err == nullptr) << "item_c must commit"; + + /// `build_ops` ran exactly once per item -- including item_c, the overflowing item validated once in + /// the fresh chunk it lands in. + EXPECT_EQ(c1->load(), 1); + EXPECT_EQ(c2->load(), 1); + EXPECT_EQ(c3->load(), 1); + + /// Committed ids per chunk: item_a and item_b share chunk 1's real id (co-chunk survivors, both + /// woken with it); item_c gets chunk 2's id, exactly one sequence step above chunk 1. + EXPECT_EQ(ra.id, rb.id) << "co-chunk survivors must complete with the SAME real transaction id"; + EXPECT_EQ(rc.id.writer_epoch, ra.id.writer_epoch); + EXPECT_EQ(rc.id.ref_sequence, ra.id.ref_sequence + 1) << "chunk 2 must allocate the id after chunk 1"; + + /// >= 2 durable transactions in the tenure, and the split is exactly the sequential result: chunk 1 + /// carries item_a's then item_b's ops (survivor order), chunk 2 carries item_c's. + const std::vector logs = listLogTxns(*backend, layout, ns); + std::optional chunk1_txn; + std::optional chunk2_txn; + for (const RefLogTxn & txn : logs) + { + if (txn.txn_id == ra.id) + chunk1_txn = txn; + if (txn.txn_id == rc.id) + chunk2_txn = txn; + } + ASSERT_TRUE(chunk1_txn.has_value()) << "chunk 1 must be durable"; + ASSERT_TRUE(chunk2_txn.has_value()) << "chunk 2 must be durable (a second transaction in one tenure)"; + std::vector expect_chunk1 = ops1; + expect_chunk1.insert(expect_chunk1.end(), ops2.begin(), ops2.end()); + EXPECT_EQ(chunk1_txn->ops, expect_chunk1); + EXPECT_EQ(chunk2_txn->ops, ops3); + + /// Tail counters advanced once per committed chunk. + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before + 2); + + /// Per-chunk metrics: one batch-flush per chunk (2), survivors counted per chunk (2 + 1 = 3). The + /// snapshot-scheduling trigger is the final step of the SAME committed arm that increments + /// `CASRefBatchFlushes`, so == 2 also proves the scheduler was invoked per chunk; + /// `SnapshotPublisherLatchedAcrossChunks` proves that trigger actually re-fires across chunks. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes].load() - flushes_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations].load() - mutations_before, 3u); +} + +namespace +{ + +/// Shared body for the three chunk-failure variants: two items (3000 ops each) -> chunk 1 = {item_a} +/// (the leader's own item), chunk 2 = {item_b}. `mode` faults ONLY chunk 2's `_log/` PUT (skip chunk 1). +/// In every variant chunk 1 commits and the leader's own call returns chunk 1's real id, while chunk 2's +/// caller fails. Returns the two callers' results plus chunk 1's id for the per-variant assertions. +struct ChunkFailureOutcome +{ + AppendResult leader; /// item_a, chunk 1 + AppendResult follower; /// item_b, chunk 2 + RefTxnId chunk1_id{}; + std::shared_ptr backend; + PoolPtr store; +}; + +ChunkFailureOutcome runChunkFailureCase(const String & ns_suffix, ChunkFaultBackend::Mode mode) +{ + auto backend = std::make_shared(); + PoolConfig cfg; + /// Single-attempt budget: one ambiguous PUT is a conclusive Unresolved (wedge) / DefiniteFailure, + /// with no inter-attempt sleep to serve. + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + auto store = openPoolWith(backend, cfg); + const DB::Cas::Layout & layout = store->layout(); + const RootNamespace ns{String("srv1/") + ns_suffix}; + publishEmptyPart(store, ns, "seed"); + + /// Fault ONLY chunk 2's `_log/` PUT: skip chunk 1's (the first match), fault the second. Armed AFTER + /// the seed so only the flush's two log PUTs are counted. + backend->fault_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = mode; + backend->fault_skip = 1; + backend->fault_count = 1; + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + /// 3000 ops per item (1500 add/remove pairs) -> 6000 > ref_txn_max_ops: chunk 1 = {item_a}, + /// chunk 2 = {item_b}. + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), nullptr); + waitEntered(sync); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), nullptr); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); + + EXPECT_EQ(a.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "leader must not hang"; + EXPECT_EQ(b.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "follower must not hang"; + ChunkFailureOutcome out; + out.leader = a.fut.get(); + out.follower = b.fut.get(); + a.t.join(); + b.t.join(); + store->setRefPreCarveHookForTest(nullptr); + out.chunk1_id = out.leader.id; + out.backend = backend; + out.store = store; + return out; +} + +} + +/// Test 9 (chunk-failure variant a -- definite failure): chunk 2's PUT is conclusively rejected +/// (`CasWriteOutcome::DefiniteFailure`). Chunk 1's caller (the leader's own item) observes SUCCESS with +/// chunk 1's real id; chunk 2's caller fails; the lane does NOT wedge (a definite rejection is a safe +/// gap, not an uncertain PUT). +TEST(CASRefWriterChunkedFlush, ChunkFailureDefinite) +{ +#if !USE_AWS_S3 + GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; +#endif + ChunkFailureOutcome out = runChunkFailureCase("chunk_fail_definite", ChunkFaultBackend::Mode::Definite); + ASSERT_TRUE(out.leader.err == nullptr) << "chunk-1 caller must observe success even though chunk 2 failed"; + ASSERT_TRUE(out.follower.err != nullptr) << "chunk-2 caller must observe the definite failure"; + EXPECT_FALSE(out.store->refLaneWedgedForTest(RootNamespace{"srv1/chunk_fail_definite"})) + << "a definite failure is proven non-durable and must NOT wedge the lane"; + + const auto logs = listLogTxns(*out.backend, out.store->layout(), RootNamespace{"srv1/chunk_fail_definite"}); + bool saw_chunk1 = false; + for (const RefLogTxn & txn : logs) + if (txn.txn_id == out.chunk1_id) + saw_chunk1 = true; + EXPECT_TRUE(saw_chunk1) << "chunk 1 must be durably committed"; +} + +/// Test 9 (chunk-failure variant b -- unresolved wedge): chunk 2's PUT is ambiguous and exhausts the +/// budget, wedging the lane. Chunk 1's caller observes SUCCESS; chunk 2's caller fails; the wedge holds +/// ONLY chunk 2's key (chunk 1 + 1), and chunk 1's object is durable while chunk 2's was never written. +TEST(CASRefWriterChunkedFlush, ChunkFailureWedge) +{ + const RootNamespace ns{"srv1/chunk_fail_wedge"}; + ChunkFailureOutcome out = runChunkFailureCase("chunk_fail_wedge", ChunkFaultBackend::Mode::Unresolved); + ASSERT_TRUE(out.leader.err == nullptr) << "chunk-1 caller must observe success even though chunk 2 wedged"; + ASSERT_TRUE(out.follower.err != nullptr) << "chunk-2 caller must observe the append failure"; + + EXPECT_TRUE(out.store->refLaneWedgedForTest(ns)) << "chunk 2's unresolved PUT must wedge the lane"; + RefTxnId chunk2_id = out.chunk1_id; + ++chunk2_id.ref_sequence; + EXPECT_EQ(out.store->wedgedKeyForTest(ns), out.store->layout().refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), chunk2_id)) + << "the wedge must contain ONLY chunk 2's key"; + + const auto logs = listLogTxns(*out.backend, out.store->layout(), ns); + bool saw_chunk1 = false; + bool saw_chunk2 = false; + for (const RefLogTxn & txn : logs) + { + if (txn.txn_id == out.chunk1_id) + saw_chunk1 = true; + if (txn.txn_id == chunk2_id) + saw_chunk2 = true; + } + EXPECT_TRUE(saw_chunk1) << "chunk 1 must be durably committed"; + EXPECT_FALSE(saw_chunk2) << "chunk 2's wedged object was never durably written"; +} + +/// Test 9 (chunk-failure variant c -- a throw): chunk 2's PUT surfaces a proven conflict (CORRUPTED_DATA +/// thrown by the controller). Chunk 1's caller observes SUCCESS; chunk 2's caller fails with +/// CORRUPTED_DATA; the lane does NOT wedge (a conclusive rejection). +TEST(CASRefWriterChunkedFlush, ChunkFailureThrow) +{ + const RootNamespace ns{"srv1/chunk_fail_throw"}; + ChunkFailureOutcome out = runChunkFailureCase("chunk_fail_throw", ChunkFaultBackend::Mode::ForeignConflict); + ASSERT_TRUE(out.leader.err == nullptr) << "chunk-1 caller must observe success even though chunk 2 threw"; + ASSERT_TRUE(out.follower.err != nullptr) << "chunk-2 caller must observe the thrown failure"; + expectFailedWithCode(out.follower.err, DB::ErrorCodes::CORRUPTED_DATA, "chunk-2 proven-conflict throw"); + EXPECT_FALSE(out.store->refLaneWedgedForTest(ns)) << "a proven conflict is conclusive and must NOT wedge"; + + const auto logs = listLogTxns(*out.backend, out.store->layout(), ns); + bool saw_chunk1 = false; + for (const RefLogTxn & txn : logs) + if (txn.txn_id == out.chunk1_id) + saw_chunk1 = true; + EXPECT_TRUE(saw_chunk1) << "chunk 1 must be durably committed"; +} + +/// Test 9 (containment variant 1): the leader's OWN item lands in chunk 1; a throw is injected at the +/// chunk boundary (simulating a reseed allocation failure) AFTER chunk 1 committed. Tenure containment +/// (spec §3): the leader's own `appendRefOps` returns chunk 1's real id -- NOT the later exception -- +/// while the unattempted remainder (item_b) fails. This exercises the reworked outer catch, which no +/// longer rethrows unconditionally over a durable own item. +TEST(CASRefWriterChunkedFlush, LeaderOwnItemCommittedBeforeThrow) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const DB::Cas::Layout & layout = store->layout(); + const RootNamespace ns{"srv1/chunk_leader_own_committed"}; + publishEmptyPart(store, ns, "seed"); + + auto c1 = std::make_shared>(0); + auto c2 = std::make_shared>(0); + + /// Throw once at the first chunk boundary -- after chunk 1 (the leader's own item) is durable and + /// before the reseed completes. + auto boundary_hits = std::make_shared>(0); + store->setCarveHookForTest([boundary_hits](CasRefLedger::CarvePhaseForTest ph) + { + if (ph == CasRefLedger::CarvePhaseForTest::ChunkReseed && boundary_hits->fetch_add(1) == 0) + throw std::bad_alloc{}; + }); + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + /// 3000 ops per item (1500 add/remove pairs) -> chunk 1 = {item_a}, boundary throw before chunk 2. + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), c1); + waitEntered(sync); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), c2); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); + + ASSERT_EQ(a.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "leader must not hang"; + ASSERT_EQ(b.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "follower must not hang"; + const AppendResult ra = a.fut.get(); + const AppendResult rb = b.fut.get(); + a.t.join(); + b.t.join(); + store->setCarveHookForTest(nullptr); + store->setRefPreCarveHookForTest(nullptr); + + ASSERT_TRUE(ra.err == nullptr) + << "the leader's own committed-chunk item must return success, not the later boundary throw"; + ASSERT_TRUE(rb.err != nullptr) << "the unattempted remainder must fail"; + + const std::vector logs = listLogTxns(*backend, layout, ns); + std::optional chunk1_txn; + for (const RefLogTxn & txn : logs) + if (txn.txn_id == ra.id) + chunk1_txn = txn; + ASSERT_TRUE(chunk1_txn.has_value()) << "chunk 1 must be durable"; + EXPECT_EQ(chunk1_txn->ops, addRemovePrecommitPairs("aaa_", 1500, 900000001)); + /// item_a's build_ops ran once (chunk 1); item_b's ran once (before the boundary throw preempted its + /// validation) and is NOT re-invoked -- the at-most-once contract holds through the failed tenure. + EXPECT_EQ(c1->load(), 1); + EXPECT_EQ(c2->load(), 1); +} + +/// Test 9 (containment variant 2 -- snapshot coalescing): a snapshot publisher dispatched by chunk 1 is +/// latched at its PUT AFTER capturing chunk 1's prefix; chunk 2 then commits and its publish trigger is +/// discarded by the single-in-flight gate. When the latched publisher settles, settlement must re-fire +/// the dropped trigger so a FOLLOW-UP publication covers chunk 2 -- otherwise chunk 2 would stay +/// unsnapshotted until an unrelated later mutation. The chunk boundary is gated until the publisher has +/// parked, so its captured candidate is provably chunk 1's prefix only. +TEST(CASRefWriterChunkedFlush, SnapshotPublisherLatchedAcrossChunks) +{ + auto backend = std::make_shared(); + PoolConfig cfg; + cfg.snapshot_log_count_threshold = 0; /// every committed chunk crosses the tail-count threshold + auto store = openPoolWith(backend, cfg); + const DB::Cas::Layout & layout = store->layout(); + const RootNamespace ns{"srv1/chunk_snapshot_coalesce"}; + publishEmptyPart(store, ns, "seed"); + store->waitForSnapshotPublishSettleForTest(ns); /// drain the seed's publish chain -> tail == 0 + + /// Latch the FIRST `_snap/` PUT (chunk 1's publisher) at its conditional PUT -- i.e. AFTER it has + /// captured chunk 1's prefix under state_mutex. + backend->armBlock(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_snap/"); + /// Gate the leader at the chunk boundary until that publisher has parked on its PUT, so its captured + /// candidate is EXACTLY chunk 1's prefix (not chunk 1 + chunk 2). + store->setCarveHookForTest([backend](CasRefLedger::CarvePhaseForTest ph) + { + if (ph == CasRefLedger::CarvePhaseForTest::ChunkReseed) + backend->awaitBlockEntered(); + }); + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + /// 3000 ops per item (1500 add/remove pairs) -> chunk 1 = {item_a}, chunk 2 = {item_b}. + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), nullptr); + waitEntered(sync); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), nullptr); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); + + ASSERT_EQ(a.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "leader must not hang"; + ASSERT_EQ(b.fut.wait_for(std::chrono::seconds(20)), std::future_status::ready) << "follower must not hang"; + const AppendResult ra = a.fut.get(); + const AppendResult rb = b.fut.get(); + a.t.join(); + b.t.join(); + ASSERT_TRUE(ra.err == nullptr) << "chunk 1 must commit"; + ASSERT_TRUE(rb.err == nullptr) << "chunk 2 must commit"; + RefTxnId chunk2_id = ra.id; + ++chunk2_id.ref_sequence; + EXPECT_EQ(rb.id, chunk2_id); + + /// Release the latched chunk-1 publisher. Its settlement must re-fire the chunk-2 trigger the + /// single-flight gate dropped -> a follow-up publication covers chunk 2. + backend->releaseBlock(); + store->waitForSnapshotPublishSettleForTest(ns); + store->setCarveHookForTest(nullptr); + store->setRefPreCarveHookForTest(nullptr); + + const std::optional newest = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(newest.has_value()) << "at least one snapshot must have been published"; + EXPECT_FALSE(*newest < chunk2_id) + << "settlement must re-fire the dropped chunk-2 trigger so a snapshot covers chunk 2 (no lost trigger)"; +} diff --git a/src/Disks/tests/gtest_cas_ref_ckpt.cpp b/src/Disks/tests/gtest_cas_ref_ckpt.cpp new file mode 100644 index 000000000000..9b63fa4dfc92 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_ckpt.cpp @@ -0,0 +1,1313 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include + +/// Stage A task 5 (INV-4): the `_ckpt` object. +/// +/// `_ckpt` exists because prefix cleaning made the ref stream unreadable from a LIST alone, so it is +/// simultaneously the thing recovery point-reads to find its base AND the gate on what cleanup may +/// delete. Both roles are only safe while three properties hold, and this suite pins exactly those: +/// +/// 1. the codec is STRICT in both directions -- a body that only partly decoded would be a cleanup +/// decision taken from a partly-read object; +/// 2. there is ONE merge, by semantic maximum per field, used by BOTH writers -- a writer that +/// wrote back the value it sampled earlier regresses the other writer's progress, which is TLC +/// counterexample `_sab_sealclobbersbase` and costs an acked transaction; +/// 3. every CAS attempt re-checks the admitted fence generation AFTER its read and BEFORE its write, +/// so a writer whose mount incarnation moved advances nothing. +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int MEMORY_LIMIT_EXCEEDED; +extern const int NETWORK_ERROR; +extern const int UNKNOWN_FORMAT_VERSION; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; + +namespace +{ + +const RefTxnId ID_1_1{1, 1}; +const RefTxnId ID_1_2{1, 2}; +const RefTxnId ID_2_1{2, 1}; + +PoolPtr openPool(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// The same one-transaction publish the other ref suites drive, so a namespace reaches `Live` through +/// the REAL append lane (which is also what creates its `_ckpt`). +RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const String & ref, uint64_t ordinal) +{ + return store->appendRefOps(ns, MutationScope::ref(ref), + [&ref, ordinal](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps(ref, ManifestRef{1, ordinal, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +/// A fence that never refuses, for the tests whose subject is not the fence. +const std::function ALWAYS_ADMITTED = [](uint64_t) {}; + +/// A deadline far enough out that only the test's own contention decides the outcome. The clock is +/// frozen (a constant `now`), which is what makes every non-exhaustion test independent of wall time. +CkptDeadline generousDeadline() +{ + return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; +} + +/// Reads `life`'s `_ckpt` and returns its body, or a default-constructed one after failing the +/// current test when the object is absent. Every assertion below goes through this rather than +/// dereferencing the optional directly: a bare `->` on a disengaged optional ABORTS the whole test +/// binary, so one regression would take every later suite's result with it instead of failing a test. +RefCkpt readCkptOrFail(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +{ + const std::optional sample = readCkpt(backend, layout, life); + if (!sample) + { + ADD_FAILURE() << "expected a _ckpt for namespace '" << life.ns.string() << "', found none"; + return RefCkpt{}; + } + return sample->ckpt; +} + +/// Stage B (Task 4-C): the incarnation `store`'s production birth wiring minted for `ns`, learned back +/// from the catalog exactly as a real reader would (`NamespaceLifeId::fromCatalogEntry`) -- once a real +/// `Pool`/`CasRefLedger` has opened the table, its ref-layer objects are no longer keyed at the +/// Stage-A sentinel, so every test below that drives the REAL append lane must ask the catalog what +/// incarnation it minted rather than assume the sentinel. Fails the current test (rather than +/// dereferencing a disengaged optional) if the catalog carries no entry for `ns` -- e.g. called before +/// the namespace's first append. +NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + for (const CatalogEntry & entry : snap.catalog.entries) + if (entry.ns.string() == ns.string()) + return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + ADD_FAILURE() << "expected a catalog entry for namespace '" << ns.string() << "', found none"; + return DB::Cas::tests::fixture::fixtureLife(ns); +} + +/// Replaces the whole body of one key, minting a new incarnation -- how a test installs a deliberately +/// malformed or concurrently-advanced object. +void overwriteObject(Backend & backend, const String & key, const String & bytes) +{ + const HeadResult h = backend.head(key); + ASSERT_TRUE(h.exists) << "overwriteObject expects " << key << " to exist"; + ASSERT_EQ(backend.putOverwrite(key, bytes, h.token).outcome, PutOutcome::Done); +} + +/// Runs `on_get` right after every `get` of `watched_key` -- the deterministic way to act inside +/// another component's read-then-write window without a sleep or a second thread. The hook is a public +/// member rather than a constructor argument so it can be installed AFTER the backend exists (every +/// interesting hook writes through that same backend) and only once the test's setup writes are done. +class GetHookBackend : public CountingBackend +{ +public: + using CountingBackend::get; + + explicit GetHookBackend(String watched_key_) : watched_key(std::move(watched_key_)) {} + + /// Stage B (Task 4-C): a test that must watch a namespace's `_ckpt` key can no longer compute it + /// before the pool exists -- the real incarnation is minted only once the namespace's first open + /// resolves it, which requires the pool (and so this backend) to already be constructed. Lets a + /// test retarget the watch once it has learned the real key, strictly before arming `on_get`. + void setWatchedKey(String watched_key_) { watched_key = std::move(watched_key_); } + + std::function on_get; + + std::optional get(const String & key, Range range) override + { + auto result = CountingBackend::get(key, range); + if (key == watched_key && on_get) + on_get(); + return result; + } + +private: + String watched_key; +}; + +/// Records the exact `_ckpt` recovery protocol and injects an ambiguous CAS response. The fault is +/// armed only after fixture setup, so the journal contains solely the operation under test. +class AmbiguousCkptBackend : public CountingBackend +{ +public: + enum class Fault : uint8_t + { + None, + CommitThenThrow, + ThrowWithoutCommit, + AlwaysThrowWithoutCommit, + }; + + using CountingBackend::casPut; + using CountingBackend::get; + + String watched_key; + Fault fault = Fault::None; + String dominating_bytes; + bool fail_resolution_get = false; + std::function after_ambiguous_cas; + std::function before_resolution_get; + std::function after_resolution_get; + std::vector journal; + + void arm(const String & key, Fault fault_) + { + watched_key = key; + fault = fault_; + watched_get_count = 0; + journal.clear(); + } + + std::optional get(const String & key, Range range) override + { + if (key != watched_key) + return CountingBackend::get(key, range); + + journal.push_back("GET"); + ++watched_get_count; + if (watched_get_count >= 2 && before_resolution_get) + before_resolution_get(); + if (watched_get_count == 2 && fail_resolution_get) + throw Poco::TimeoutException("AmbiguousCkptBackend: exact-read response lost"); + auto result = CountingBackend::get(key, range); + if (watched_get_count >= 2 && after_resolution_get) + after_resolution_get(); + return result; + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (key != watched_key) + return CountingBackend::casPut(key, bytes, expected, meta); + + journal.push_back("CAS"); + if (fault == Fault::None) + return CountingBackend::casPut(key, bytes, expected, meta); + const Fault this_fault = fault; + if (fault != Fault::AlwaysThrowWithoutCommit) + fault = Fault::None; + if (this_fault == Fault::CommitThenThrow) + { + const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); + if (result.outcome == CasOutcome::Committed && !dominating_bytes.empty()) + { + const HeadResult head_result = CountingBackend::head(key); + EXPECT_EQ(CountingBackend::putOverwrite(key, dominating_bytes, head_result.token).outcome, + PutOutcome::Done); + } + } + if (after_ambiguous_cas) + after_ambiguous_cas(); + throw Poco::TimeoutException("AmbiguousCkptBackend: CAS response lost"); + } + +private: + size_t watched_get_count = 0; +}; + +} + +/// --------------------------------------------------------------------------------------------- +/// The codec +/// --------------------------------------------------------------------------------------------- + +/// Every combination of the frontier and existing optionals survives a round trip. Both-absent is the shape a namespace +/// carries from creation until its first snapshot, so it is a real state and not a degenerate one. +TEST(CASRefCheckpoint, RoundTripsEveryFieldCombination) +{ + const std::vector cases = { + RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = ID_2_1}, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = ID_2_1}, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = ID_2_1}, + }; + for (const RefCkpt & ckpt : cases) + EXPECT_EQ(decodeRefCkpt(encodeRefCkpt(ckpt)), ckpt); +} + +TEST(CASRefCheckpoint, CommittedThroughHasCanonicalExactWireEncoding) +{ + const RefCkpt ckpt{.life_epoch = std::optional{7}, + .committed_through = RefTxnId{9, 11}, + .checkpoint_snapshot_id = RefTxnId{9, 10}, + .last_epoch_seal = RefTxnId{8, 12}}; + const String expected = R"({"type":"cas_ref_ckpt","v":10} +{"le":"7","cte":"9","cts":"11","cse":"9","css":"10","lse":"8","lss":"12"} +)"; + + EXPECT_EQ(encodeRefCkpt(ckpt), expected); + EXPECT_EQ(decodeRefCkpt(expected), ckpt); +} + +/// `last_epoch_seal` is chain evidence, not an arbitrary lower bound. It either names the frontier +/// itself when that frontier is the terminal seal, or closes the immediately preceding numeric epoch. +/// Accepting a gap or a later same-epoch frontier would manufacture a boundary that INV-2 never proved. +TEST(CASRefCheckpoint, CodecRejectsIncoherentCommittedFrontierAndSealEpochs) +{ + const RefCkpt valid{.life_epoch = 7, .committed_through = RefTxnId{8, 5}, + .checkpoint_snapshot_id = RefTxnId{7, 4}, .last_epoch_seal = RefTxnId{7, 9}}; + EXPECT_NO_THROW(encodeRefCkpt(valid)); + + const RefCkpt skipped_epoch{.life_epoch = 7, .committed_through = RefTxnId{10, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{7, 9}}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefCkpt(skipped_epoch); }); + + const RefCkpt frontier_after_same_epoch_seal{.life_epoch = 7, .committed_through = RefTxnId{8, 5}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{8, 1}}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { encodeRefCkpt(frontier_after_same_epoch_seal); }); + + const RefCkpt unsealed_non_genesis{.life_epoch = 7, .committed_through = RefTxnId{8, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefCkpt(unsealed_non_genesis); }); + + String malformed = encodeRefCkpt(valid); + const size_t cte = malformed.find(R"("cte":"8")"); + ASSERT_NE(cte, String::npos); + malformed.replace(cte, String{R"("cte":"8")"}.size(), R"("cte":"10")"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(malformed); }); +} + +/// STRICT means an unknown key is corruption, not something to skip. A `_ckpt` decides deletions, so a +/// reader that ignored a field it did not understand would be authorizing them from a body it only +/// partly read. +TEST(CASRefCheckpoint, RejectsAnUnknownKey) +{ + const String good = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, + .last_epoch_seal = std::nullopt}); + String with_unknown = good; + with_unknown.replace(with_unknown.rfind('}'), 1, R"(,"zz":"1"})"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(with_unknown); }); + + /// A `!`-prefixed key is a REQUIRED extension and reports the version, not corruption -- the + /// distinction is what lets an operator tell "this build is too old" from "this object is broken". + String with_critical = good; + with_critical.replace(with_critical.rfind('}'), 1, R"(,"!zz":"1"})"); + expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { decodeRefCkpt(with_critical); }); +} + +/// A duplicate key has no single meaning, so it can never be resolved by a reader's preference. +TEST(CASRefCheckpoint, RejectsADuplicateKey) +{ + const String good = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + String duplicated = good; + duplicated.replace(duplicated.rfind('}'), 1, R"(,"le":"9"})"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(duplicated); }); +} + +/// Truncation in each of its shapes. Half an optional pair is the dangerous one: silently dropping it +/// would turn a truncated body into a well-formed `_ckpt` with NO checkpoint, which reads as +/// "recovery has no base" and would be trusted. +TEST(CASRefCheckpoint, RejectsTruncation) +{ + const String good = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, + .last_epoch_seal = std::nullopt}); + + const String header_only = good.substr(0, good.find('\n') + 1); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(header_only); }); + + /// The body line without its terminator: a read that stopped mid-object. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(good.substr(0, good.size() - 1)); }); + + /// An EMPTY body is not truncation, it is the legitimate "nobody knows anything yet" object -- the + /// shape a namespace carries between its creation and its first checkpoint. Asserted here, next to + /// the truncation cases, because the two are one character apart on the wire. + const String empty_body = good.substr(0, good.find('\n') + 1) + "{}\n"; + EXPECT_EQ(decodeRefCkpt(empty_body), RefCkpt{}); + + const String half_pair = good.substr(0, good.find('\n') + 1) + R"({"le":"7","cse":"1"})" + "\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(half_pair); }); + + const String other_half = good.substr(0, good.find('\n') + 1) + R"({"le":"7","lss":"2"})" + "\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(other_half); }); + + const String frontier_half = good.substr(0, good.find('\n') + 1) + R"({"le":"7","cte":"1"})" + "\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(frontier_half); }); +} + +TEST(CASRefCheckpoint, RejectsTrailingBytes) +{ + const String good = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(good + "junk\n"); }); +} + +/// The field-validity rule runs in BOTH directions: a struct this build refuses to read can never be +/// written by it either, so a bug on the write side surfaces at the writer and not as an unreadable +/// object discovered by a future recovery. +TEST(CASRefCheckpoint, RejectsInvalidFieldsOnEncodeAndOnDecode) +{ + /// PRESENT means REAL: an absent field is legal, a present-but-impossible one is not. A zero + /// `life_epoch` would give the field two meanings ("unknown" and "epoch zero") on an object that + /// gates deletions. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [] { encodeRefCkpt(RefCkpt{.life_epoch = std::optional{0}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [] { encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = RefTxnId{1, 0}, .last_epoch_seal = std::nullopt}); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [] { encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{0, 1}}); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [] { encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}); }); + + const String header = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + const String prefix = header.substr(0, header.find('\n') + 1); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(prefix + R"({"le":"0"})" + "\n"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefCkpt(prefix + R"({"le":"7","cse":"1","css":"0"})" + "\n"); }); +} + +/// The registry row is part of the contract: Control/Strict decides how the decoder treats unknown +/// keys, and the caps are the first thing that fires if a foreign object ever lands at the key. +TEST(CASRefCheckpoint, RegistryRowIsControlStrictWithTightCaps) +{ + const FormatTraits & traits = traitsFor(FormatId::RefCkpt); + EXPECT_EQ(traits.type, "cas_ref_ckpt"); + EXPECT_EQ(traits.family, TextFamily::Control); + EXPECT_EQ(traits.strictness, KeyStrictness::Strict); + EXPECT_EQ(traits.object_cap, 64u * 1024u); + EXPECT_EQ(traits.line_cap, 4u * 1024u); + EXPECT_EQ(traitsForType("cas_ref_ckpt"), &traits); + /// Raw, so the key has no suffix -- the Stage A shape is exactly `/_ckpt`. This line is also + /// the TRIPWIRE for the codec's shortcut: `encodeRefCkpt`/`decodeRefCkpt` hand bytes to and from + /// the backend directly, bypassing `sealObject`/`openObject` because both are the identity under + /// `CompressionPolicy::Never`. Flip the policy to `Always` and that bypass would silently write + /// uncompressed bodies under a `.zst` key -- which this assertion catches first. + EXPECT_EQ(storedSuffix(FormatId::RefCkpt), ""); + EXPECT_EQ(traits.compression, CompressionPolicy::Never); +} + +/// --------------------------------------------------------------------------------------------- +/// The key +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefCheckpoint, KeyIsTheLifeLeafAndParsesBack) +{ + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_key"}; + const NamespaceLifeId ns_id = DB::Cas::tests::fixture::fixtureLife(ns); + EXPECT_EQ(layout.refCkptKey(ns_id), + "p/cas/ns/state/" + renderIncarnation(ns_id.incarnation) + "/_ckpt"); + EXPECT_EQ(layout.parseRefCkptKey(layout.refCkptKey(ns_id)), ns_id.incarnation); + + /// `_ckpt` has no kind directory, so the id-bearing parser must NOT claim it -- and the `_ckpt` + /// parser must not claim the id-bearing keys either. Each key has exactly one classifier. + EXPECT_FALSE(layout.parseRefObjectKey(layout.refCkptKey(ns_id)).has_value()); + EXPECT_FALSE(layout.parseRefCkptKey(layout.refLogKey(ns_id, ID_1_1)).has_value()); + EXPECT_FALSE(layout.parseRefCkptKey(layout.refSnapshotKey(ns_id, ID_1_1)).has_value()); + EXPECT_FALSE(layout.parseRefCkptKey(layout.refCkptKey(ns_id) + ".zst").has_value()); + EXPECT_FALSE(layout.parseRefCkptKey("p/cas/ns/state/_ckpt").has_value()); + EXPECT_FALSE(layout.parseRefCkptKey("q" + layout.refCkptKey(ns_id).substr(1)).has_value()); +} + +/// The hot stream grouping accepts logs and snapshots while ignoring a checkpoint from the separate +/// state tree. An unrecognized key inside the stream tree still aborts the round. +TEST(CASRefCheckpoint, GroupRefKeysScopesHotIntakeToTheStreamTree) +{ + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_group"}; + const std::vector keys = { + layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), ID_1_1), + layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), ID_1_1), + layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), + }; + + const auto grouped = groupRefKeys(layout, keys); + ASSERT_EQ(grouped.size(), 1u); + const RefTableListing & listing = grouped.at(DB::Cas::tests::fixture::fixtureLife(ns).incarnation); + EXPECT_EQ(listing.logs, std::vector{ID_1_1}); + EXPECT_EQ(listing.snapshots, std::vector{ID_1_1}); + + /// A genuinely unrecognizable key inside this life stream is still corruption. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { groupRefKeys(layout, {layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_bogus"}); }); +} + +/// --------------------------------------------------------------------------------------------- +/// The merge -- per field, both directions +/// --------------------------------------------------------------------------------------------- + +/// The per-field table the ledger obligation from the TLA phase asks for: each field independently +/// newer on either side, plus both-absent and equal bodies. A merge that is not per-field would pass +/// some rows and fail others, which is the point of enumerating them. +TEST(CASRefCheckpoint, MergeTakesThePerFieldSemanticMaximum) +{ + const RefCkpt low{.life_epoch = std::optional{3}, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = ID_1_1}; + const RefCkpt high_ckpt{.life_epoch = std::optional{3}, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = ID_1_1}; + const RefCkpt high_seal{.life_epoch = std::optional{3}, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = ID_2_1}; + const RefCkpt high_life{.life_epoch = std::optional{9}, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = ID_1_1}; + + /// Each field newer on the RIGHT, then the same case mirrored to the LEFT: the merge is symmetric, + /// which is exactly why the two writers need no ordering between them. + /// + /// The `life_epoch` rows stay mirrored, and that is a deliberate statement rather than an oversight: + /// a `life_epoch` that FALLS is refused, but the refusal lives in `publishCkpt`, which knows which + /// side is durable, and NOT here. This function stays commutative, so both directions must keep + /// yielding the maximum. See `CASRefCheckpointJoin` (`gtest_cas_ref_ckpt_join.cpp`) for the refusal itself + /// and for why it cannot be expressed at this level. + EXPECT_EQ(mergeCkpt(low, high_ckpt), high_ckpt); + EXPECT_EQ(mergeCkpt(high_ckpt, low), high_ckpt); + EXPECT_EQ(mergeCkpt(low, high_seal), high_seal); + EXPECT_EQ(mergeCkpt(high_seal, low), high_seal); + EXPECT_EQ(mergeCkpt(low, high_life), high_life); + EXPECT_EQ(mergeCkpt(high_life, low), high_life); + + /// Fields advance INDEPENDENTLY: a merge of two bodies each newer in a different field keeps both. + const RefCkpt both = mergeCkpt(high_ckpt, high_seal); + EXPECT_EQ(both.checkpoint_snapshot_id, ID_1_2); + EXPECT_EQ(both.last_epoch_seal, ID_2_1); + + /// An absent optional loses to a present one, whichever side it is on, and two absents stay absent. + const RefCkpt none{.life_epoch = std::optional{3}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(mergeCkpt(none, low), low); + EXPECT_EQ(mergeCkpt(low, none), low); + EXPECT_EQ(mergeCkpt(none, none), none); + + /// Identical bodies merge to themselves -- the property `publishCkpt` turns into "no write". + EXPECT_EQ(mergeCkpt(low, low), low); + + /// A contribution that knows NOTHING about `life_epoch` (the snapshot publisher's shape) must not + /// erase it. This is the case a plain assignment would get wrong. + const RefCkpt publisher_only{.life_epoch = std::nullopt, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; + const RefCkpt advanced = mergeCkpt(low, publisher_only); + EXPECT_EQ(advanced.life_epoch, 3u); + EXPECT_EQ(advanced.checkpoint_snapshot_id, ID_1_2); + EXPECT_EQ(advanced.last_epoch_seal, ID_1_1) << "the publisher knows nothing about the seal and must " + "not drag it backwards"; +} + +/// --------------------------------------------------------------------------------------------- +/// publishCkpt +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefCheckpoint, CreatesTheObjectWhenItIsAbsent) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_create"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const RefCkpt birth{.life_epoch = std::optional{5}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + + EXPECT_EQ(publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const auto sample = readCkpt(*backend, layout, life); + ASSERT_TRUE(sample.has_value()); + EXPECT_EQ(sample->ckpt, birth); +} + +/// Any writer may CREATE the object, and none of them may complete it. A publisher knows only the +/// checkpoint, so it creates an object that knows only the checkpoint; the birth transaction's +/// `life_epoch` merges in afterwards. Order does not matter -- the merge is a per-field maximum, and +/// no writer ever supplies a field it does not know (a guess here would be permanent, since the merge +/// can never lower it). +TEST(CASRefCheckpoint, EachWriterCreatesWithOnlyWhatItKnowsAndTheOtherFieldsMergeInLater) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_partial_create"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const RefCkpt publisher{.life_epoch = std::nullopt, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; + + ASSERT_EQ(publishCkpt(*backend, layout, life, publisher, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const auto created = readCkpt(*backend, layout, life); + ASSERT_TRUE(created.has_value()); + EXPECT_EQ(created->ckpt.checkpoint_snapshot_id, ID_1_1); + EXPECT_FALSE(created->ckpt.life_epoch.has_value()) << "the publisher must not invent a genesis epoch"; + + const RefCkpt birth{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const auto completed = readCkpt(*backend, layout, life); + ASSERT_TRUE(completed.has_value()); + EXPECT_EQ(completed->ckpt.life_epoch, 1u); + EXPECT_EQ(completed->ckpt.checkpoint_snapshot_id, ID_1_1) << "and must not lose the checkpoint on the way in"; +} + +/// The conflict path is the whole reason the algorithm re-READS instead of retrying its bytes: the +/// winner's field must survive the loser's retry. Here a concurrent writer advances the seal between +/// our read and our CAS; our retry must merge onto the new body, not overwrite it. +TEST(CASRefCheckpoint, TokenConflictRereadsAndMergesOntoTheWinner) +{ + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_conflict"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + + auto backend = std::make_shared(key); + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + + /// The concurrent sealer lands exactly ONCE, immediately after our first read -- so our first CAS + /// carries a token that is no longer current, and our retry has to merge onto its body. + bool interfered = false; + backend->on_get = [&] + { + if (interfered) + return; + interfered = true; + const RefCkpt sealer{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = ID_2_1}; + const HeadResult h = backend->head(key); + ASSERT_EQ(backend->putOverwrite(key, encodeRefCkpt(mergeCkpt(base, sealer)), h.token).outcome, + PutOutcome::Done); + }; + + const RefCkpt publisher{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(publishCkpt(*backend, layout, life, publisher, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + + const auto sample = readCkpt(*backend, layout, life); + ASSERT_TRUE(sample.has_value()); + EXPECT_EQ(sample->ckpt.checkpoint_snapshot_id, ID_1_2) << "our own contribution must land"; + EXPECT_EQ(sample->ckpt.last_epoch_seal, ID_2_1) + << "the concurrent writer's seal must survive our retry -- a retry that reused the body read " + "before the conflict would silently drop it (TLC `_sab_sealclobbersbase`)"; + EXPECT_EQ(sample->ckpt.life_epoch, 1u); + EXPECT_GE(backend->casPutCount(key), 2u) << "the first CAS must have been rejected, not skipped"; +} + +/// A contribution that adds nothing issues NO write. This is a correctness property, not a saving: +/// both writers publish on every snapshot and every seal, and a no-op write would mint a fresh token +/// each time, turning every other writer's in-flight CAS into a conflict for identical bytes. +TEST(CASRefCheckpoint, AnIdenticalMergedBodyIssuesNoWrite) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_noop"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.refCkptKey(life); + const RefCkpt full{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = ID_2_1}; + + ASSERT_EQ(publishCkpt(*backend, layout, life, full, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const uint64_t writes_after_create = backend->casPutCount(key); + const Token token_after_create = backend->head(key).token; + + /// The same contribution again, and a strictly OLDER one: neither adds anything. + EXPECT_EQ(publishCkpt(*backend, layout, life, full, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::IdenticalSkip); + const RefCkpt older{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(publishCkpt(*backend, layout, life, older, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::IdenticalSkip); + + EXPECT_EQ(backend->casPutCount(key), writes_after_create) << "a skip must issue no CAS at all"; + EXPECT_EQ(backend->head(key).token, token_after_create) << "and must not mint a new incarnation"; +} + +/// The fence is re-checked AFTER the read and BEFORE the write, on every attempt. A generation that +/// moved means this writer's lease incarnation is gone, so its merged body is stale even if the fence +/// is live again under a fresh incarnation. +TEST(CASRefCheckpoint, AFenceBumpBetweenTheReadAndTheCasWritesNothing) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_fenced"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(*backend, layout, life, base, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const Token token_before = backend->head(key).token; + const uint64_t writes_before = backend->casPutCount(key); + + /// The callback the pool wires from `CasMountRuntime::checkFenceOrThrow`: it throws when the + /// generation moved since admission. Mirrors the real site's class (the transient, upstream-retryable + /// one) so the stub cannot drift into testing a shape production never produces. + const auto moved_fence = [](uint64_t admitted) + { + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, + "fence generation moved since admission ({})", admitted); + }; + + const RefCkpt advance{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(publishCkpt(*backend, layout, life, advance, 1, moved_fence, generousDeadline()), + CkptPublishOutcome::FencedOut); + EXPECT_EQ(backend->casPutCount(key), writes_before) << "the check precedes the CAS, so nothing is sent"; + EXPECT_EQ(backend->head(key).token, token_before); + EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); +} + +/// Persistent contention fails CLOSED and says so. There is no partial state to clean up -- every +/// attempt either committed the complete merged body or changed nothing -- but the caller must be told +/// its contribution is unpublished rather than left to assume it landed. +TEST(CASRefCheckpoint, AnExhaustedDeadlineUnderPersistentConflictThrowsRetryLater) +{ + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_exhausted"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; + + auto backend = std::make_shared(key); + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + + /// Every read is followed by a rewrite of the SAME body under a fresh incarnation, so the token this + /// call holds is always stale and every CAS it issues conflicts. The clock advances one step per + /// read, so the DEADLINE is what ends the loop -- deterministically, with no sleeping and well + /// before the live-lock brake. + uint64_t now = 0; + backend->on_get = [&] + { + ++now; + const HeadResult h = backend->head(key); + if (h.exists) + backend->putOverwrite(key, encodeRefCkpt(base), h.token); + }; + + const RefCkpt advance{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { publishCkpt(*backend, layout, life, advance, 1, ALWAYS_ADMITTED, CkptDeadline{[&] { return now; }, 5}); }); + EXPECT_EQ(readCkptOrFail(*backend, layout, life), base) << "no partial state: every attempt either " + "committed the complete merged body or wrote nothing"; +} + +TEST(CASRefCheckpoint, AmbiguousCommittedCasIsResolvedByOneExactReadWithoutBlindRetry) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_committed"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt contribution{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + + EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); + EXPECT_EQ(readCkptOrFail(*backend, layout, life).committed_through, ID_1_2); +} + +TEST(CASRefCheckpoint, AmbiguousUncommittedCasRetriesAgainstTheExactReadToken) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_retry"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt contribution{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + backend->arm(key, AmbiguousCkptBackend::Fault::ThrowWithoutCommit); + + EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET", "CAS"})); + EXPECT_EQ(readCkptOrFail(*backend, layout, life).committed_through, ID_1_2); +} + +TEST(CASRefCheckpoint, AmbiguousCasAcceptsAValidDominatingDurableFrontier) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_dominating"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt contribution{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt dominating{.life_epoch = 1, .committed_through = ID_2_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = ID_2_1}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + backend->dominating_bytes = encodeRefCkpt(dominating); + backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + + EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); + EXPECT_EQ(readCkptOrFail(*backend, layout, life), dominating); +} + +TEST(CASRefCheckpoint, FailedExactReadAfterAmbiguousCasFailsClosedWithoutAnotherCas) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_read_failed"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + backend->fail_resolution_get = true; + backend->arm(key, AmbiguousCkptBackend::Fault::ThrowWithoutCommit); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + publishCkpt(*backend, layout, life, RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, 7, + ALWAYS_ADMITTED, generousDeadline()); + }); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); + EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); +} + +TEST(CASRefCheckpoint, FenceMovementAroundAmbiguityResolutionMakesTheExactReadInert) +{ + for (const bool move_before_read : {true, false}) + { + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife( + RootNamespace{move_before_read ? "srv1/ckpt_fence_before_resolution" : "srv1/ckpt_fence_after_resolution"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + bool admitted = true; + const auto move_fence = [&] { admitted = false; }; + if (move_before_read) + backend->before_resolution_get = move_fence; + else + backend->after_resolution_get = move_fence; + backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + + const auto check_admission = [&](uint64_t) + { + if (!admitted) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence moved"); + }; + EXPECT_EQ(publishCkpt(*backend, layout, life, + RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, 7, + check_admission, generousDeadline()), CkptPublishOutcome::FencedOut); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); + } +} + +TEST(CASRefCheckpoint, AdmissionLostWithTheAmbiguousCasPreventsItsResolutionGet) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife( + RootNamespace{"srv1/ckpt_admission_lost_before_resolution"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + + bool admitted = true; + backend->after_ambiguous_cas = [&] { admitted = false; }; + backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + const auto admit_request = [&] + { + if (!admitted) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "admission withdrawn"); + }; + + EXPECT_EQ(publishCkpt(*backend, layout, life, + RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, + 7, ALWAYS_ADMITTED, generousDeadline(), admit_request), CkptPublishOutcome::FencedOut); + EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS"})) + << "publishCkpt started its ambiguity-resolution GET after admission was withdrawn"; +} + +TEST(CASRefCheckpoint, ContinuedAmbiguityStopsAtTheDeadlineAndNeverIssuesConsecutiveCasAttempts) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguity_deadline"}); + const String key = layout.refCkptKey(life); + const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + uint64_t now = 0; + backend->after_resolution_get = [&] { ++now; }; + backend->arm(key, AmbiguousCkptBackend::Fault::AlwaysThrowWithoutCommit); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + publishCkpt(*backend, layout, life, + RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, + 7, ALWAYS_ADMITTED, CkptDeadline{[&] { return now; }, 3}); + }); + EXPECT_EQ(backend->journal, + (std::vector{"GET", "CAS", "GET", "CAS", "GET", "CAS", "GET"})); + EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); +} + +/// A `_ckpt` that does not decode is NEVER overwritten. It is the only record of recovery's base and +/// of what cleanup may delete, so replacing it with a body derived from the contribution alone would +/// erase the base and leave a well-formed object a reader would trust. +TEST(CASRefCheckpoint, ACorruptCheckpointIsNeverOverwritten) +{ + auto backend = std::make_shared(); + const Layout layout{"p"}; + const RootNamespace ns{"srv1/ckpt_corrupt"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.refCkptKey(life); + ASSERT_EQ(publishCkpt(*backend, layout, life, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}, + 1, ALWAYS_ADMITTED, generousDeadline()), CkptPublishOutcome::Published); + + const String garbage = "not a cas object\n"; + overwriteObject(*backend, key, garbage); + + const RefCkpt birth{.life_epoch = std::optional{5}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()); }); + EXPECT_EQ(backend->get(key)->bytes, garbage) << "corruption must be surfaced, never laundered into a " + "well-formed object"; +} + +/// --------------------------------------------------------------------------------------------- +/// The reader-side rules Task 6 and the cleanup call sites consume +/// --------------------------------------------------------------------------------------------- + +/// INV-4's three-way revalidation of a base that turned out to be missing. +TEST(CASRefCheckpoint, AMissingSampledBaseRestartsOnAnAdvancedTokenAndIsCorruptionOnAnUnchangedOne) +{ + const Token sampled{"t1", TokenType::Emulated}; + const Token advanced{"t2", TokenType::Emulated}; + + EXPECT_EQ(classifyMissingSampledBase(sampled, advanced), MissingBaseVerdict::RestartRecovery) + << "cleanup legitimately moved the checkpoint while we read; restart from the newer base"; + EXPECT_EQ(classifyMissingSampledBase(sampled, sampled), MissingBaseVerdict::Corrupted) + << "the checkpoint still names an object that is not there, which the strictly-below deletion " + "gate makes unreachable in an honest run"; + EXPECT_EQ(classifyMissingSampledBase(sampled, std::nullopt), MissingBaseVerdict::Corrupted) + << "a namespace with a sampled base and no checkpoint at all is worse, not better"; +} + +/// The deletion gate is STRICTLY below, because the checkpoint names the snapshot a recovery is +/// entitled to fetch by exact key. At-or-below is TLC counterexample `_sab_staleckptcorruption`. +TEST(CASRefCheckpoint, SnapshotsAreDeletableStrictlyBelowTheCheckpoint) +{ + EXPECT_TRUE(snapshotDeletableUnderCkpt(ID_1_1, ID_1_2)); + EXPECT_FALSE(snapshotDeletableUnderCkpt(ID_1_2, ID_1_2)) << "the checkpoint's own base is off limits"; + EXPECT_FALSE(snapshotDeletableUnderCkpt(ID_2_1, ID_1_2)); + /// Fail closed: a namespace with no checkpoint has established no covering base, so nothing is + /// deletable -- a stale or absent pointer may only ever under-clean. + EXPECT_FALSE(snapshotDeletableUnderCkpt(ID_1_1, std::nullopt)); +} + +/// --------------------------------------------------------------------------------------------- +/// The REAL call sites, through the ledger +/// --------------------------------------------------------------------------------------------- + +/// The namespace-birth transaction creates the checkpoint, and it is the only writer that can: the +/// `life_epoch` is this transaction's own writer epoch. +TEST(CASRefCheckpoint, NamespaceBirthCreatesTheCheckpointCarryingItsLifeEpoch) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/ckpt_birth"}; + + /// Stage B (Task 4-C): the catalog carries no entry for `ns` before its first open, and the + /// namespace's real incarnation does not exist to name a key with yet -- the pre-birth analog of + /// "nothing exists" is "nothing is even NAMED", checked at the catalog rather than at a key this + /// test cannot yet compute. + EXPECT_TRUE(CasRefCatalog::read(*backend, store->layout()).catalog.entries.empty()) + << "nothing exists before the birth"; + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const auto sample = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(sample.has_value()) << "spec §3 creates the _ckpt before the namespace becomes Live"; + EXPECT_EQ(sample->ckpt.life_epoch, store->writerEpoch()); + EXPECT_FALSE(sample->ckpt.checkpoint_snapshot_id.has_value()) << "a newborn namespace has no base yet"; + EXPECT_FALSE(sample->ckpt.last_epoch_seal.has_value()); +} + +/// The snapshot publisher is INV-4's second writer: the body PUT commits, then the checkpoint names it. +TEST(CASRefCheckpoint, ACommittedSnapshotPublishAdvancesTheCheckpoint) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/ckpt_publish"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + ASSERT_FALSE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + const auto published = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(published.has_value()); + + const auto sample = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(sample.has_value()); + EXPECT_EQ(sample->ckpt.checkpoint_snapshot_id, published); + EXPECT_EQ(sample->ckpt.life_epoch, epoch) << "the publisher contributes nothing about life_epoch, so " + "the merge must preserve what the birth wrote"; + /// And the snapshot body it names really is there -- the checkpoint may never point at a key that + /// does not exist, which is the premise the missing-base rule reasons from. + EXPECT_TRUE(backend->head(store->layout().refSnapshotKey(life, *published)).exists); +} + +/// The body-PUT/cleanup/`_ckpt` race, decided by the ORDER of the two writes: cleanup planned in the +/// window between the snapshot body PUT and the checkpoint CAS still reads the OLD checkpoint, and the +/// gate is strictly below it -- so it cannot delete the snapshot just published. +TEST(CASRefCheckpoint, CleanupPlannedBetweenTheBodyPutAndTheCkptCasCannotDeleteTheNewSnapshot) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/ckpt_race"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + const RefTxnId first_snapshot = *store->newestPublishedSnapshotIdForTest(ns); + + ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{epoch, 2})); + /// The checkpoint a cleanup pass sampled BEFORE the second publication -- the stale reading the + /// race hands it. + const std::optional stale_checkpoint = readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id; + ASSERT_EQ(stale_checkpoint, first_snapshot); + + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + const RefTxnId second_snapshot = *store->newestPublishedSnapshotIdForTest(ns); + ASSERT_LT(first_snapshot, second_snapshot); + + /// Planning against the STALE checkpoint: the just-published snapshot is not deletable, and neither + /// is the one the stale checkpoint itself names. A stale pointer can only under-clean. + EXPECT_FALSE(snapshotDeletableUnderCkpt(second_snapshot, stale_checkpoint)); + EXPECT_FALSE(snapshotDeletableUnderCkpt(first_snapshot, stale_checkpoint)); + /// Once the checkpoint is re-read, the older snapshot becomes reclaimable and the base does not. + const std::optional fresh_checkpoint = readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id; + EXPECT_TRUE(snapshotDeletableUnderCkpt(first_snapshot, fresh_checkpoint)); + EXPECT_FALSE(snapshotDeletableUnderCkpt(second_snapshot, fresh_checkpoint)); +} + +/// One `_ckpt` write per publication and not one more: the checkpoint is written where the snapshot is +/// published, and a publisher with nothing above its newest snapshot touches it at all. +TEST(CASRefCheckpoint, TheCheckpointIsWrittenOncePerPublicationAndNotOnIdleAttempts) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/ckpt_republish"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String key = store->layout().refCkptKey(life); + const uint64_t writes_after_birth = backend->casPutCount(key); + EXPECT_EQ(writes_after_birth, 2u) + << "birth publishes `life_epoch` before its log, then the durable log's committed frontier"; + + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_EQ(backend->casPutCount(key), writes_after_birth + 1) << "one publication, one checkpoint CAS"; + const uint64_t writes_after_publish = backend->casPutCount(key); + const auto after_publish = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(after_publish.has_value()); + + /// Nothing was appended since, so there is nothing above the newest snapshot: the publisher declines + /// before it reaches the checkpoint at all, and repeating the attempt changes nothing. + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_EQ(backend->casPutCount(key), writes_after_publish); + EXPECT_EQ(readCkptOrFail(*backend, store->layout(), life), after_publish->ckpt); +} + +TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/no_snapshot_at_seal"}; + uint64_t predecessor_epoch = 0; + { + auto predecessor = openPool(backend); + predecessor_epoch = predecessor->writerEpoch(); + ASSERT_EQ(publishRef(predecessor, ns, "ref_1", 1), (RefTxnId{predecessor_epoch, 1})); + } + + auto store = openPool(backend); + ASSERT_GT(store->writerEpoch(), predecessor_epoch); + ASSERT_EQ(store->listRefs(ns).size(), 1u) << "recovery must close the predecessor epoch before publishing"; + + const RefTxnId seal_id{predecessor_epoch, 2}; + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String snapshot_key = store->layout().refSnapshotKey(life, seal_id); + const String ckpt_key = store->layout().refCkptKey(life); + ASSERT_EQ(store->lastEpochSealForTest(ns), std::make_optional(seal_id)); + ASSERT_EQ(readCkptOrFail(*backend, store->layout(), life).committed_through, std::make_optional(seal_id)); + const uint64_t snapshot_puts_before = backend->putCount(snapshot_key); + const uint64_t ckpt_cas_before = backend->casPutCount(ckpt_key); + + /// Recovery installed the epoch seal as the runtime's greatest applied transaction. The publisher + /// must decline it without reaching either durable write. + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_EQ(backend->putCount(snapshot_key), snapshot_puts_before); + EXPECT_EQ(backend->casPutCount(ckpt_key), ckpt_cas_before); + + /// Once an ordinary transaction advances the candidate beyond the seal, normal publication resumes. + ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 1})); + EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); +} + +/// Publication replays a `NeedsRecovery` lane before it captures a snapshot and advances `_ckpt`. +TEST(CASRefCheckpoint, NeedsRecoveryReplaysBeforeCheckpointAdvance) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/ckpt_poisoned"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String key = store->layout().refCkptKey(life); + const auto before = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(before.has_value()); + ASSERT_FALSE(before->ckpt.checkpoint_snapshot_id.has_value()); + const uint64_t writes_before = backend->casPutCount(key); + + /// Enter `NeedsRecovery`: an install throws after its transaction is + /// durable, leaving this cached table missing a transaction the log contains. + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); + expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { publishRef(store, ns, "ref_2", 2); }); + store->setInstallRegionProbeForTest(nullptr); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + /// The publish entry point recovers first, so the snapshot covers the stranded transaction. + EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_TRUE(store->resolveRef(ns, "ref_2", /*allow_stale=*/false).has_value()) + << "the stranded transaction is durable; the re-derivation must have applied it"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_GT(backend->casPutCount(key), writes_before) + << "and the checkpoint advances -- truthfully, over a snapshot that is not missing anything"; + EXPECT_TRUE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + +} + +/// A publish admitted under an incarnation that is replaced mid-attempt advances NOTHING, and does not +/// adopt the snapshot either -- adopting would suppress every later publication for it while the +/// checkpoint still pointed below it, leaving recovery on an older base with nothing to fix it. +TEST(CASRefCheckpoint, APublishFencedOutMidAttemptDoesNotAdvanceTheCheckpoint) +{ + const RootNamespace ns{"srv1/ckpt_stale_gen"}; + + /// The watched key cannot be computed yet -- the real incarnation is minted only once the pool + /// exists and this namespace's first open resolves it (`setWatchedKey` below, once it has). + auto backend = std::make_shared(""); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolPtr store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String ckpt_key = store->layout().refCkptKey(life); + backend->setWatchedKey(ckpt_key); + + const auto before = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(before.has_value()); + ASSERT_FALSE(before->ckpt.checkpoint_snapshot_id.has_value()); + const uint64_t writes_before = backend->casPutCount(ckpt_key); + + /// Arm only after the precondition read above. The next watched `_ckpt` read is therefore the one + /// inside this publish's read-then-CAS window, after the attempt captured its immutable runtime + /// generation. Arming before `readCkpt` would stale the runtime before the operation began and test + /// entry admission instead of the intended mid-attempt recheck. + bool hook_fired = false; + backend->on_get = [&] + { + if (hook_fired) + return; + hook_fired = true; + DB::Cas::tests::rearmMountFenceAfterAnomalyForTest(store); + }; + + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "a publish whose checkpoint could not be advanced must not report success"; + EXPECT_TRUE(hook_fired) << "the checkpoint read-then-CAS seam was never exercised"; + backend->on_get = nullptr; + EXPECT_EQ(backend->casPutCount(ckpt_key), writes_before) << "nothing may be sent after the fence moved"; + EXPECT_FALSE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + EXPECT_FALSE(store->newestPublishedSnapshotIdForTest(ns).has_value()) + << "the snapshot must not be adopted as the newest while its checkpoint is unpublished"; +} + +/// =================================================================================== +/// Equivalence fences for the `prepareRefChunk` extraction (Stage B `{#extract-prepare-ref-chunk}`) +/// =================================================================================== +/// +/// An extraction is only safe to review if something pins what crosses its boundary. These three +/// fences are deliberately NOT red-first: they pass on the PRE-extraction tree and must keep passing +/// after it, which is the whole point -- the literals below were captured from a real append on the +/// pre-extraction tree and pasted in, so re-deriving them afterwards cannot silently measure the +/// change against itself. +/// +/// They live in this TU rather than beside the pure preparation tests because all three need a real +/// backend and the real append lane, which this suite already drives through `publishRef` (including +/// the namespace birth, the one chunk shape whose first durable effect is the `_ckpt` and not the +/// ref-log `PUT`). +TEST(CASRefCheckpoint, CommitRefChunkDurableBytesUnchangedByExtraction) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"test/golden@cas@"}; + + const RefTxnId id = publishRef(store, ns, "gold_ref", 7); + ASSERT_EQ(id.writer_epoch, 1u); + ASSERT_EQ(id.ref_sequence, 1u); + + /// The KEY carries the namespace incarnation, so its life segment is rendered rather than pasted + /// (Task 1c re-keys it); every other segment is literal. Stage B (Task 4-C): the incarnation is now + /// a REAL, randomly minted catalog value rather than the Stage-A sentinel, so it is learned back + /// from the catalog (`liveLifeOrFail`) rather than pasted as a literal -- the shape assertion below + /// is unaffected, since it names every OTHER segment literally and renders this one dynamically. + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String key = store->layout().refLogKey(life, id); + EXPECT_EQ(key, "p/cas/ns/stream/" + renderIncarnation(life.incarnation) + + "/_log/0000000000000001-0000000000000001.zst") + << "the canonical ref-log key the append lane derives"; + + /// The BODY is checked as exact length plus a 128-bit SipHash of it -- not literally byte for byte, + /// but any change that survives both is a 128-bit collision at a fixed length, which is the trade for + /// keeping the assertion readable. It is a function of `{format generation, ns, id, ops, + /// chain_link}` only -- no incarnation reaches it. Generation 10 changed the shared format header; + /// the plaintext discriminator below removes only that change and pins every remaining byte to the + /// generation-9 fixture before accepting the new deterministic compressed size and hash. + const auto got = backend->get(key); + ASSERT_TRUE(got.has_value()) << "the birth chunk must be durable at its canonical key"; + String as_generation_9 = openObject(FormatId::RefLog, got->bytes); + const String generation_10_header = R"({"type":"cas_ref_log","v":10})"; + ASSERT_TRUE(as_generation_9.starts_with(generation_10_header)); + as_generation_9.replace(0, generation_10_header.size(), R"({"type":"cas_ref_log","v":9})"); + EXPECT_EQ(as_generation_9, R"({"type":"cas_ref_log","v":9} +{"ns":"test/golden@cas@","we":"1","rs":"1"} +{"op":"namespace_birth"} +{"op":"owner_transition","nbk":"precommit","nrn":"gold_ref","nme":"1","nmb":"7","nmo":1} +{"op":"owner_transition","obk":"precommit","orn":"gold_ref","ome":"1","omb":"7","omo":1,"nbk":"committed","nrn":"gold_ref","nme":"1","nmb":"7","nmo":1} +{"n":3} +)") << "generation 10 must change only the self-describing header of this ref-log fixture"; + EXPECT_EQ(got->bytes.size(), 179u) << "the sealed ref-log body changed size"; + SipHash body_hash; + body_hash.update(got->bytes.data(), got->bytes.size()); + EXPECT_EQ(getHexUIntLowercase(body_hash.get128()), "ada75a83638e933c98d731183a46b7b7") + << "the sealed ref-log body changed content -- preparation must seal the same bytes it sealed " + "before the extraction"; +} + +/// The directive's "preserve backend request counts", asserted rather than assumed: preparation is pure, +/// so lifting it out must not add or remove a single request. One birth chunk = exactly one write-once +/// `PUT` at the ref-log key, no read-back, plus the two ordered `_ckpt` CASes required by the protocol: +/// creation publishes `life_epoch` before the log and the append lane publishes `committed_through` +/// after the log is durable. +/// +/// COUNTS per key. Request ORDER is not checked here and cannot be with these counters; the ordering +/// that matters for a birth -- `_ckpt` before the ref-log `PUT` -- is argued at the call site and would +/// need a sequence-recording backend to pin. +TEST(CASRefCheckpoint, AppendRequestCountUnchangedByExtraction) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"test/req@cas@"}; + + const RefTxnId id = publishRef(store, ns, "req_ref", 1); + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const String log_key = store->layout().refLogKey(life, id); + const String ckpt_key = store->layout().refCkptKey(life); + + EXPECT_EQ(backend->putCount(log_key), 1u) << "exactly one write-once PUT per committed chunk"; + /// ONE GET, not zero, since Stage B (Task 4-C): `resolveNamespaceLife`'s `completeCreation` call + /// publishes this life's `_ckpt.life_epoch` BEFORE the birth chunk is prepared, so this table's + /// OWN recovery walk (also inside this `appendRefOps`, ahead of the commit) grounds itself at the + /// genesis position `_ckpt` now names and confirms it absent by exact key -- which is `log_key` + /// itself, the position the birth chunk is about to occupy. That GET precedes the Committed PUT; + /// the PUT itself still owes no read-back. + EXPECT_EQ(backend->getCount(log_key), 1u) << "one grounding probe from recovery, before the birth PUT"; + EXPECT_EQ(backend->casPutCount(ckpt_key), 2u) + << "the birth contributes `life_epoch` before its log and `committed_through` after the durable " + "log; these are two different ordering obligations, not a duplicate publication"; +} + +/// The post-durable install region is the reason preparation has to happen where it does: once "this +/// object may be durable", recording it must not fail. The extraction moves work EARLIER, never into that +/// region. +/// +/// The guarded region is NOT the whole window: between the `Committed` outcome and the swap, +/// `carve_hook_for_test(PostDurableInstall)`, the `state_mutex` acquisition and the `state_unchanged` +/// evaluation all run OUTSIDE `DENY_ALLOCATIONS_IN_SCOPE`. This fence does not prove allocation-freedom +/// for them or for anything else. +/// +/// WHAT THIS TEST PROVES, and what it does NOT -- stated precisely, because a fence trusted for more +/// than it checks is worse than no fence. +/// +/// `DENY_ALLOCATIONS_IN_SCOPE` is `static_assert(true)` unless `!defined(NDEBUG)` (`MemoryTracker.h`), +/// so it is inert in every build that leaves `NDEBUG` defined -- which includes this gate and CI's +/// sanitizer lanes, since those configure `CMAKE_BUILD_TYPE=None` and `CMakeLists.txt` maps that to +/// `RelWithDebInfo`. Only a `Debug` build, or a tidy lane (which adds `-UNDEBUG`), has the +/// no-allocation half live. Nothing here proves the region does not allocate. +/// +/// What is left is weaker than "the install is still guarded": `install_region_probe_for_test` fires +/// as the FIRST statement inside the guarded scope, BEFORE `rt->state.swap(*candidate)`, and the SAME +/// probe is shared by BOTH probe-instrumented post-durable install regions (`CasRefLedger.cpp`: the +/// wedge-resolution adoption and `commitRefChunk`'s `Committed` install). So `probe_hits > 0` proves +/// only that SOME probe-instrumented region was entered on this path -- which on this path can only be +/// the commit install, since nothing here wedges. It goes red if the region stops being entered at all +/// (a lost commit path, a skipped install arm); a refactor that lifted the swap out of the scope while +/// leaving the guard shell and the probe behind would keep it GREEN. +TEST(CASRefCheckpoint, PostDurableInstallRegionStillEnteredAfterExtraction) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"test/region@cas@"}; + + unsigned probe_hits = 0; + store->setInstallRegionProbeForTest([&probe_hits] { ++probe_hits; }); + const RefTxnId id = publishRef(store, ns, "region_ref", 1); + store->setInstallRegionProbeForTest(nullptr); + + EXPECT_GT(probe_hits, 0u) + << "no probe-instrumented post-durable install region was entered on a committing append -- " + "the `Committed` install arm was not reached at all"; + const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + EXPECT_TRUE(backend->get(store->layout().refLogKey(life, id)).has_value()); +} diff --git a/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp b/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp new file mode 100644 index 000000000000..e50603d9258d --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp @@ -0,0 +1,549 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +/// The `_ckpt` JOIN law and its `O(1)` SIZE invariant. +/// +/// `mergeCkpt` already has a suite (`CasRefCkpt` in `gtest_cas_ref_ckpt.cpp`) covering it as one step of +/// the publish algorithm. This suite's subject is narrower and different: the JOIN LAW itself, per +/// field, stated so that a later change to any one field's rule fails here rather than being absorbed +/// into a publish-path assertion; plus the size invariant, which no existing test constrains at all. +/// +/// The size half is a REGRESSION FENCE, not a fix -- nothing about today's `_ckpt` is non-`O(1)`. It +/// exists to fail the day someone adds a map, a collection, or any per-ref/per-file term to an object +/// that has no repair path and gates destructive cleanup. +/// +/// Constraint 15 names four dimensions (refs, files, transactions, writer epochs) and they do NOT +/// behave the same way, so they get two different assertions rather than one claim covering both: +/// +/// - REFS and FILES never enter the body in any form, so the encoded size is BYTE-EQUAL between a +/// namespace holding one and a namespace holding ten thousand. That is `EncodedCkptSizeIs...` +/// below, and it drives the REAL append lane on purpose: a hand-built pair of `RefCkpt` structs +/// would leave a newly-added collection field EMPTY in both and the equality would still hold, +/// so the fence would not fire on the very change it exists to catch. Only a real producer +/// populates a real field. +/// - TRANSACTIONS and WRITER EPOCHS enter as the DECIMAL WIDTH of the two id pairs. That is not +/// equality: `{cse=1,css=1}` and `{cse=1,css=10000}` differ by four bytes. It is `O(1)` because +/// the fields are `uint64_t` and so the width is ceilinged at twenty digits, which is a bound a +/// test asserts on a constructed worst case -- `EncodedCkptSizeHasAConstantCeiling...` below. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; + +/// Constraint 15's COMPILE-TIME half: `_ckpt` is a fixed-size product of scalar monotone facts. Any +/// field that owns heap storage -- a map, a vector, a `String` -- makes `RefCkpt` non-trivially-copyable +/// and fails the build here, which is the earliest and cheapest place the constraint can be enforced. +/// The two runtime size tests below are the rest of the fence: this one cannot see a fixed-capacity +/// array, and they cannot see a field that is never populated by the producers they drive. +static_assert(std::is_trivially_copyable_v, + "Constraint 15: _ckpt is a fixed-size product of scalar monotone facts, so its encoded size is " + "O(1) in refs, files, transactions and writer epochs. A field with heap storage (a map, a " + "vector, a String) breaks that and belongs in a separate immutable object or ledger."); + +namespace +{ + +constexpr uint64_t U64_MAX = std::numeric_limits::max(); + +/// Constraint 15's bound, as a number: the encoded size of the WIDEST `_ckpt` this build can produce +/// (all three fields present, every integer component at `UINT64_MAX`). Pinned as a literal so that +/// adding a field, or widening one, fails a test rather than quietly moving the bound. Generation 10 +/// added one byte to the shared format-version header (`9` became `10`); the scalar body is unchanged. +constexpr size_t CKPT_WORST_CASE_ENCODED_BYTES = 235; + +/// The high-cardinality side of the size fence, in ONE transaction. Bounded above by the append lane's +/// 5000-operation cap on a normal-class item (`publishCommittedOps` emits two ops per ref), and kept at +/// one transaction on purpose: spreading a larger namespace over several of them costs tens of seconds +/// in a debug build, and a size fence that cannot finish inside the harness budget fences nothing. Any +/// per-ref term in `_ckpt` is as visible at this count as at any larger one. +constexpr size_t MANY_REFS = 2000; + +/// Fixed-width, so the refs themselves cannot be what differs between the two namespaces: the claim +/// under test is that ref cardinality does not reach `_ckpt`, and a name that grew with `i` would +/// confound a size comparison if it ever did. +String refName(size_t i) +{ + return fmt::format("r{:08}", i); +} + +/// A fence that never refuses, and a deadline far enough out that only the test's own contention +/// decides the outcome -- each `_ckpt`/catalog test file defines its own copy, matching the precedent +/// `gtest_cas_ns_creation_lifecycle.cpp` states explicitly. +const std::function ALWAYS_ADMITTED = [](uint64_t) {}; + +CkptDeadline generousDeadline() +{ + return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; +} + +/// Admits the FIRST call (spent by `completeCreation`'s step-2 `publishCkpt`) and refuses every call +/// after (step 3's own `mutate`): "fenced out between the `_ckpt` create and the `Creating -> Live` +/// CAS", deterministically and without a second thread. That is the durable shape a stalled creator +/// leaves behind, and the starting state the resumption test needs. +std::function admittedOnceThenFenced() +{ + auto calls = std::make_shared(0); + return [calls](uint64_t admitted) + { + if (++*calls > 1) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); + }; +} + +CreatorFence creatorFence(const String & srid, uint64_t writer_epoch, uint64_t fence_generation = 1) +{ + return CreatorFence{.server_root_id = srid, .writer_epoch = writer_epoch, .fence_generation = fence_generation}; +} + +/// A `is_creator_fence_terminal` stub answering one fixed verdict: terminality itself is not this +/// suite's subject (its tests live next to the real predicate in `gtest_cas_mount.cpp`). +std::function fixedTerminality(bool terminal) +{ + return [terminal](const CreatorFence &) { return terminal; }; +} + +const CatalogEntry * findEntryForTest(const RefCatalog & catalog, const RootNamespace & ns) +{ + for (const CatalogEntry & e : catalog.entries) + if (e.ns.string() == ns.string()) + return &e; + return nullptr; +} + +/// `life`'s durable `life_epoch`, failing the current test rather than dereferencing a disengaged +/// optional -- a bare `->` on one aborts the whole binary and takes every later suite's result with it. +uint64_t lifeEpochOrFail(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +{ + const std::optional sample = readCkpt(backend, layout, life); + if (!sample || !sample->ckpt.life_epoch) + { + ADD_FAILURE() << "expected a _ckpt carrying a life_epoch for namespace '" << life.ns.string() << "'"; + return 0; + } + return *sample->ckpt.life_epoch; +} + +/// `boot_ms_fn` defaults to the real clock. A caller whose test body does enough CPU-bound work +/// against ONE open pool to risk outrunning `mount_lease_ttl_ms` on a slow sanitizer build should +/// pass a frozen one instead of widening the TTL: the mount fence and the ref-log request controller +/// both read time through this same seam (see `CasRefLedger`'s `controller_boot_ms_fn`), so freezing +/// it removes the wall-clock race rather than merely giving it more room. +PoolPtr openPool(const BackendPtr & backend, std::function boot_ms_fn = {}) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test"}; + config.boot_ms_fn = std::move(boot_ms_fn); + return Pool::open(backend, std::move(config)); +} + +/// The incarnation the production birth wiring minted for `ns`, learned back from the catalog the way a +/// real reader does. Fails the current test rather than dereferencing a disengaged optional, so one +/// regression cannot abort the binary and take every later suite's result with it. +NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + for (const CatalogEntry & entry : snap.catalog.entries) + if (entry.ns.string() == ns.string()) + return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + ADD_FAILURE() << "expected a catalog entry for namespace '" << ns.string() << "', found none"; + return DB::Cas::tests::fixture::fixtureLife(ns); +} + +/// Births `ns` and publishes `ref_count` committed refs through the REAL append lane, in ONE +/// transaction, and returns that namespace's durable `_ckpt` as encoded bytes. +/// +/// One transaction also holds every OTHER dimension fixed while `ref_count` varies: two namespaces +/// built this way end at the same transaction id, so a difference in their `_ckpt` bodies can only be +/// the refs. `ref_count` must therefore stay within the append lane's per-item operation cap. +String encodedCkptOfNamespaceWithRefs(const PoolPtr & store, Backend & backend, const Layout & layout, + const RootNamespace & ns, size_t ref_count) +{ + store->appendRefOps(ns, MutationScope::wholeShard(), + [ref_count](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (size_t i = 0; i < ref_count; ++i) + for (const RefOp & op : publishCommittedOps(refName(i), ManifestRef{1, i + 1, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + + const NamespaceLifeId life = liveLifeOrFail(backend, layout, ns); + const std::optional sample = readCkpt(backend, layout, life); + if (!sample) + { + ADD_FAILURE() << "expected a _ckpt for namespace '" << ns.string() << "' after its birth transaction"; + return {}; + } + return encodeRefCkpt(sample->ckpt); +} + +} + +/// --------------------------------------------------------------------------------------------- +/// The join law, per field +/// --------------------------------------------------------------------------------------------- + +/// An absence is "this writer knew nothing", never "this writer says none". Exactly one writer ever +/// knows a namespace's genesis epoch, so every other contribution is `nullopt` and must leave what is +/// on record alone -- in BOTH argument orders, because the two `_ckpt` writers have no ordering +/// between them and the merge is what makes that safe. +TEST(CASRefCheckpointJoin, JoinUnknownLifeEpochWithPresentYieldsPresent) +{ + const RefCkpt unknown{.life_epoch = std::nullopt, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt present{.life_epoch = 7, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + + EXPECT_EQ(mergeCkpt(unknown, present).life_epoch, std::optional{7}); + EXPECT_EQ(mergeCkpt(present, unknown).life_epoch, std::optional{7}) + << "the merge is commutative -- a writer that knows nothing must not be able to erase the " + "genesis epoch, whichever side it is on"; + + /// The other half of "absent loses": two absences stay absent. `life_epoch` has no floor to fall + /// back to, and a fabricated one is permanent -- the semantic-max merge can never lower it again. + EXPECT_EQ(mergeCkpt(unknown, unknown).life_epoch, std::nullopt); +} + +/// The ordinary steady state: both writers agree. Asserted for its own sake because it is what +/// `publishCkpt`'s may-not-decrease rule must keep admitting -- an equal republish is not a decrease -- +/// and it is also what `publishCkpt` turns into "no write at all". +TEST(CASRefCheckpointJoin, JoinEqualLifeEpochsYieldsSame) +{ + const RefCkpt a{.life_epoch = 9, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt b{.life_epoch = 9, .checkpoint_snapshot_id = RefTxnId{9, 4}, .last_epoch_seal = std::nullopt}; + + EXPECT_EQ(mergeCkpt(a, b).life_epoch, std::optional{9}); + EXPECT_EQ(mergeCkpt(b, a).life_epoch, std::optional{9}); + const std::optional b_checkpoint = RefTxnId{9, 4}; + EXPECT_EQ(mergeCkpt(a, b).checkpoint_snapshot_id, b_checkpoint) + << "an equal life_epoch must not disturb the other fields' own join"; +} + +TEST(CASRefCheckpointJoin, CrossEpochFrontierRequiresAnImmediatelyAdjacentSeal) +{ + const RefCkpt older{.life_epoch = std::nullopt, .committed_through = RefTxnId{7, 9}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + const RefCkpt transitioned{.life_epoch = std::nullopt, .committed_through = RefTxnId{8, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{8, 1}}; + EXPECT_EQ(mergeCkpt(older, transitioned).committed_through, transitioned.committed_through); + EXPECT_EQ(mergeCkpt(transitioned, older).committed_through, transitioned.committed_through); + + /// Every committed epoch is materialized. A later frontier may advance only to the immediately + /// following numeric writer epoch, otherwise a missing epoch would be mistaken for a proved + /// boundary. The log grammar rejects this same skip at the record boundary; `_ckpt` must not + /// reintroduce it through its semantic merge. + const RefCkpt skipped_epoch{.life_epoch = std::nullopt, .committed_through = RefTxnId{10, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{7, 9}}; + EXPECT_THROW(mergeCkpt(older, skipped_epoch), DB::Exception); + + const RefCkpt advanced{.life_epoch = std::nullopt, .committed_through = RefTxnId{8, 5}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{7, 9}}; + EXPECT_EQ(mergeCkpt(advanced, older).committed_through, advanced.committed_through); + + const RefCkpt unsealed{.life_epoch = std::nullopt, .committed_through = RefTxnId{8, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + EXPECT_THROW(mergeCkpt(older, unsealed), DB::Exception); + const RefCkpt stale_prior_seal{.life_epoch = std::nullopt, .committed_through = RefTxnId{10, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{7, 8}}; + EXPECT_THROW(mergeCkpt(older, stale_prior_seal), DB::Exception) + << "a seal below the lower durable frontier does not connect the two histories"; + const RefCkpt seal_above_frontier{.life_epoch = std::nullopt, .committed_through = RefTxnId{8, 5}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{8, 6}}; + EXPECT_THROW(mergeCkpt(older, seal_above_frontier), DB::Exception); +} + +/// THE FIRST OF THE TWO SEQUENCES THAT RAISE `life_epoch` HONESTLY, end to end through the production +/// primitives rather than at the merge: a creator publishes `_ckpt` at E1 (`completeCreation` step 2) +/// and dies before its `Creating -> Live` CAS (step 3), and a later actor reconciles the stalled entry +/// and resumes over the SAME incarnation, contributing E2. Two different present values in one +/// incarnation, and NOT a conflict -- the stored value must simply become E2, which is also what +/// `CASNsCreationLifecycle.ReconcileSucceedsTokenExactlyAfterTheOriginalCreatorFenceIsTerminalThenResumesToLive` +/// already pins from the catalog side ("the RESUMING actor's writer_epoch is the genesis epoch that +/// actually landed"). Had the directive's literal rule landed, this sequence would raise +/// `CORRUPTED_DATA` and, since `_ckpt` has no repair path, wedge the namespace forever. +/// +/// Both fences share ONE `server_root_id` on purpose: every live namespace is rooted at its own pool +/// member's `server_root_id`, so a creator and its reconciler are always actors of the same server root +/// and draw from the same durable-monotone epoch counter. That is the whole basis for "contributions +/// only ever rise", so the fixture must not quietly model two roots. +TEST(CASRefCheckpointJoin, ResumedCreationRaisesLifeEpochWithoutRefusal) +{ + InMemoryBackend backend; + Layout layout("p"); + DB::Cas::tests::seedPoolMetaForRestart(backend); + const RootNamespace ns{"a"}; + + ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv1", 5), + /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()), + CasRefCatalog::NamespaceCreationOutcome::FencedOut); + + /// Bound to a name, never chained through a temporary: a `const CatalogEntry *` taken from an + /// unbound `Snapshot` dangles the instant the full expression ends. + const CasRefCatalog::Snapshot stalled = CasRefCatalog::read(backend, layout); + const CatalogEntry * entry = findEntryForTest(stalled.catalog, ns); + ASSERT_NE(entry, nullptr); + ASSERT_EQ(entry->state, NsState::Creating); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 5u) << "step 2 landed before the creator stalled"; + + const CreatorFence resumer = creatorFence("srv1", 9); + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, *entry, resumer, fixedTerminality(true), + /*admitted_generation=*/1, ALWAYS_ADMITTED), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + + CatalogEntry resumed = *entry; + resumed.creator = resumer; + EXPECT_EQ(CasRefCatalog::completeCreation(backend, layout, resumed, /*admitted_generation=*/1, + ALWAYS_ADMITTED, generousDeadline()), + CasRefCatalog::NamespaceCreationOutcome::Live) + << "the resumption must not be refused by the join"; + EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u) + << "the genesis epoch that actually landed is the resuming actor's, and the join must let it rise"; +} + +/// THE SECOND SEQUENCE, at the seam where the two `life_epoch`-knowing writers actually meet -- both of +/// them reach this object only through `publishCkpt`, so driving that twice over one key IS the +/// production interleaving, not a stand-in for it. `completeCreation` contributes the catalog creator's +/// epoch; the mount's writer epoch then advances (a restart, a remount); the first precommit's birth +/// chunk contributes the `NamespaceBirth` record's epoch. CREATE TABLE, restart, INSERT. +TEST(CASRefCheckpointJoin, RestartBetweenCreationAndFirstWriteRaisesLifeEpochWithoutRefusal) +{ + InMemoryBackend backend; + Layout layout("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); + + const RefCkpt from_creation{.life_epoch = 4, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(backend, layout, life, from_creation, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + + const RefCkpt from_birth_chunk{.life_epoch = 7, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(publishCkpt(backend, layout, life, from_birth_chunk, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published) + << "the birth chunk's later epoch must be publishable, not refused as a conflict"; + EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 7u); +} + +/// THE REFUSAL, and the state it constructs IS UNREACHABLE ON ANY HONEST PATH -- that is the point of +/// the test, not a caveat on it. `writer_epoch` is durable-monotone per server root +/// (`allocateWriterEpoch` CAS-bumps `/gc/server-roots//epoch`) and a namespace belongs to +/// exactly one server root, so no live writer can contribute an epoch below one already durable. The +/// only way to reach this is for the fence discipline itself to have failed and a SUPERSEDED writer's +/// contribution to have landed anyway. +/// +/// So this test does not model an operating condition; it asserts what happens if the guarantee above +/// is ever violated -- `publishCkpt` refuses and names both values, rather than absorbing the violation +/// into a maximum and leaving no trace. The state is built by publishing the two contributions in the +/// order the fence discipline is supposed to prevent, which needs no seam that manufactures impossible +/// states: `publishCkpt` is a public entry point and the order of two calls is the test's to choose. +/// +/// It is driven through `publishCkpt` for a second reason, not just convenience: that IS where the rule +/// lives and the only place it CAN live. `mergeCkpt` is commutative -- the stated reason the two writers +/// need no ordering between them -- so it cannot tell a decrease from an increase, having no idea which +/// of its arguments is durable. There is deliberately no merge-level counterpart to this test. +TEST(CASRefCheckpointJoin, JoinDecreasingLifeEpochIsCorruptionAndPublishesNothing) +{ + CountingBackend backend; + Layout layout("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); + const String key = layout.refCkptKey(life); + + const RefCkpt durable{.life_epoch = 9, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(backend, layout, life, durable, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const uint64_t cas_puts_before = backend.casPutCount(key); + + const RefCkpt superseded{.life_epoch = 3, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + String message; + try + { + publishCkpt(backend, layout, life, superseded, 1, ALWAYS_ADMITTED, generousDeadline()); + ADD_FAILURE() << "a contribution below the durable life_epoch must not be published"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + message = e.message(); + } + + /// BOTH values, not just the offending one: an operator reading this has to be able to tell which + /// writer is the superseded one without going to the object. Matched as RENDERED substrings rather + /// than as bare digits -- a lone "9" would also be satisfied by a key or a byte count that happened + /// to contain it, so a bare-digit match would keep passing after the message stopped saying this. + EXPECT_NE(message.find("9 is durable"), String::npos) << "the durable value must be named: " << message; + EXPECT_NE(message.find("contributed 3"), String::npos) << "the contributed value must be named: " << message; + EXPECT_NE(message.find(key), String::npos) << "the key must be named: " << message; + /// And that the object cannot be repaired in place, which is the part an operator cannot derive + /// from the two numbers. + EXPECT_NE(message.find("NO in-place repair"), String::npos) + << "the message must say the object has no in-place repair: " << message; + + /// And nothing was written. The refusal is decided before the body is built, so the durable object + /// is untouched and no write was even attempted. + EXPECT_EQ(backend.casPutCount(key), cas_puts_before) << "the publisher must not CAS on a refused publish"; + EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u) << "the durable value is unchanged"; +} + +/// The other half of the refusal, and the reason it consults the fence before classifying: the SAME +/// decrease from a writer the fence is about to refuse is not corruption. That writer landed nothing +/// anywhere, so what it gets is the transient control signal every other refusal in `publishCkpt` +/// returns rather than throws. Reporting corruption for it would turn "your incarnation moved, retry" +/// into a permanent verdict on the namespace, which is the opposite of what the detector means: the +/// violation is a STILL-ADMITTED writer contributing a superseded epoch. +TEST(CASRefCheckpointJoin, ADecreasingLifeEpochFromAFencedOutWriterIsReportedFencedOutNotCorruption) +{ + CountingBackend backend; + Layout layout("p"); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); + const String key = layout.refCkptKey(life); + + const RefCkpt durable{.life_epoch = 9, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(backend, layout, life, durable, 1, ALWAYS_ADMITTED, generousDeadline()), + CkptPublishOutcome::Published); + const uint64_t cas_puts_before = backend.casPutCount(key); + + const std::function always_fenced = [](uint64_t admitted) + { + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); + }; + const RefCkpt superseded{.life_epoch = 3, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; + EXPECT_EQ(publishCkpt(backend, layout, life, superseded, 1, always_fenced, generousDeadline()), + CkptPublishOutcome::FencedOut); + + EXPECT_EQ(backend.casPutCount(key), cas_puts_before); + EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u); +} + +/// `checkpoint_snapshot_id` and `last_epoch_seal` continue to merge by SEMANTIC MAXIMUM. Unlike +/// `life_epoch` these two genuinely advance over a namespace's life, and the max is what stops a writer +/// that sampled an older body from regressing the other writer's progress (TLC counterexample +/// `_sab_sealclobbersbase`, which costs an acked transaction). Both directions and present-beats-absent, +/// since the two writers have no ordering between them. +TEST(CASRefCheckpointJoin, CheckpointAndSealStillMergeBySemanticMaximum) +{ + const RefCkpt lower{.life_epoch = std::nullopt, .checkpoint_snapshot_id = RefTxnId{3, 5}, .last_epoch_seal = RefTxnId{3, 4}}; + const RefCkpt higher{.life_epoch = std::nullopt, .checkpoint_snapshot_id = RefTxnId{4, 1}, .last_epoch_seal = RefTxnId{4, 2}}; + + const std::optional higher_checkpoint = higher.checkpoint_snapshot_id; + const std::optional higher_seal = higher.last_epoch_seal; + const std::optional lower_checkpoint = lower.checkpoint_snapshot_id; + const std::optional lower_seal = lower.last_epoch_seal; + + /// Ordered by writer_epoch FIRST: `{4,1}` beats `{3,5}` even though its sequence is smaller, which + /// is the intended timeline across an epoch restart that resets the sequence. + EXPECT_EQ(mergeCkpt(lower, higher).checkpoint_snapshot_id, higher_checkpoint); + EXPECT_EQ(mergeCkpt(higher, lower).checkpoint_snapshot_id, higher_checkpoint); + EXPECT_EQ(mergeCkpt(lower, higher).last_epoch_seal, higher_seal); + EXPECT_EQ(mergeCkpt(higher, lower).last_epoch_seal, higher_seal); + + /// Present beats absent, both directions and both fields. + const RefCkpt nothing; + EXPECT_EQ(mergeCkpt(nothing, lower).checkpoint_snapshot_id, lower_checkpoint); + EXPECT_EQ(mergeCkpt(lower, nothing).checkpoint_snapshot_id, lower_checkpoint); + EXPECT_EQ(mergeCkpt(nothing, lower).last_epoch_seal, lower_seal); + EXPECT_EQ(mergeCkpt(lower, nothing).last_epoch_seal, lower_seal); + EXPECT_EQ(mergeCkpt(nothing, nothing).checkpoint_snapshot_id, std::nullopt); + EXPECT_EQ(mergeCkpt(nothing, nothing).last_epoch_seal, std::nullopt); +} + +/// --------------------------------------------------------------------------------------------- +/// Constraint 15: the `O(1)` size invariant +/// --------------------------------------------------------------------------------------------- + +/// REFS and FILES: byte-equal, because they never enter the body. Driven through the REAL append lane +/// (see `encodedCkptOfNamespaceWithRefs` on why a hand-built struct pair would not fence anything). +TEST(CASRefCheckpointJoin, EncodedCkptSizeIsIndependentOfCardinality) +{ + auto backend = std::make_shared(); + /// `MANY_REFS` committed through ONE `appendRefOps` call is CPU-bound encoding, not I/O -- on a + /// slow sanitizer build (msan in particular) it can outrun the real-clock `mount_lease_ttl_ms` + /// this pool was opened under and trip the mount fence mid-publish. Freeze the pool's clock + /// instead of racing it (see `openPool`'s doc comment). + auto store = openPool(backend, [] { return uint64_t{0}; }); + Layout layout("p"); + + const String one = encodedCkptOfNamespaceWithRefs(store, *backend, layout, RootNamespace{"srv1/one"}, 1); + const String many = encodedCkptOfNamespaceWithRefs(store, *backend, layout, RootNamespace{"srv1/many"}, MANY_REFS); + + ASSERT_FALSE(one.empty()); + ASSERT_FALSE(many.empty()); + EXPECT_EQ(one, many) + << "not merely equal in SIZE: refs and files reach `_ckpt` in no form at all, so the two bodies " + "are byte-identical."; + /// The same claim stated so that it does not depend on the chosen cardinality at all: no ref + /// PUBLISHED into the namespace appears anywhere in its `_ckpt`. A count-based comparison can only + /// catch a term that grows; this catches one that is merely there. + EXPECT_EQ(many.find(refName(0)), String::npos) + << "a published ref's NAME appears in `_ckpt`: " << many; + EXPECT_EQ(many.find(refName(MANY_REFS - 1)), String::npos) + << "a published ref's NAME appears in `_ckpt`: " << many; + EXPECT_EQ(one.size(), many.size()) + << "Constraint 15: `_ckpt`'s encoded size must not grow with the number of refs or files in the " + "namespace. A collection or per-ref term was added to an object that has NO repair path and " + "gates destructive cleanup; it belongs in a separate immutable object or ledger instead.\n" + " 1 ref: " << one + << " " << MANY_REFS << " refs: " << many; +} + +/// TRANSACTIONS and WRITER EPOCHS: not equality -- they enter as the decimal width of the id pairs -- +/// but ceilinged, because the fields are `uint64_t`. The worst case is constructible exactly (every +/// field present at `UINT64_MAX`), so the bound is asserted on it rather than believed about it. +TEST(CASRefCheckpointJoin, EncodedCkptSizeHasAConstantCeilingAcrossTransactionsAndEpochs) +{ + /// The true worst case over every namespace history: all three fields present, every component at + /// the widest value its type can hold. No real `_ckpt` can encode larger, because there is no field + /// that is not one of these five integers. + const RefCkpt worst{.life_epoch = U64_MAX, + .committed_through = RefTxnId{U64_MAX, U64_MAX}, + .checkpoint_snapshot_id = RefTxnId{U64_MAX, U64_MAX}, + .last_epoch_seal = RefTxnId{U64_MAX, U64_MAX}}; + const size_t worst_bytes = encodeRefCkpt(worst).size(); + + /// Pinned as a literal, not merely compared against itself: this is the number Constraint 15's + /// `O(1)` claim reduces to, and a change to it means a field was added, removed or rewidened. + EXPECT_EQ(worst_bytes, CKPT_WORST_CASE_ENCODED_BYTES) + << "the widest `_ckpt` this build can encode changed size -- a field was added, removed, or " + "given a wider type. Constraint 15's O(1) bound is exactly this constant."; + + /// The growth term is the decimal width, and it is bounded by that ceiling rather than proportional + /// to the number of transactions: four orders of magnitude of `ref_sequence` cost four bytes. + const RefCkpt at_sequence_1{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = RefTxnId{1, 1}, .last_epoch_seal = RefTxnId{1, 1}}; + const RefCkpt at_sequence_10k{.life_epoch = 1, .committed_through = RefTxnId{1, 10000}, .checkpoint_snapshot_id = RefTxnId{1, 10000}, .last_epoch_seal = RefTxnId{1, 10000}}; + EXPECT_EQ(encodeRefCkpt(at_sequence_10k).size(), encodeRefCkpt(at_sequence_1).size() + 12); + EXPECT_LE(encodeRefCkpt(at_sequence_10k).size(), worst_bytes); + EXPECT_LE(encodeRefCkpt(at_sequence_1).size(), worst_bytes); + + /// And the ceiling is far below the format registry's own object cap, so the cap is what it is + /// documented to be -- a corruption brake this object cannot approach -- and never the thing that + /// makes the size bounded. + EXPECT_LT(worst_bytes, traitsFor(FormatId::RefCkpt).object_cap); +} diff --git a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp new file mode 100644 index 000000000000..bc911e4ad6db --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp @@ -0,0 +1,641 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event CASMountReleaseSkippedForeignOccupant; +extern const Event CASMountExclusivityViolation; +} + +/// Stage A task 3 (INV-1): ref-log transaction ids are PER-NAMESPACE and CONTIGUOUS. +/// +/// The id an append persists is not drawn from a counter at all -- it is DERIVED from the table's own +/// durable state: `{live_epoch, greatest_applied.ref_sequence + 1}` within one epoch, `{live_epoch, 1}` +/// at an epoch change. Two consequences this suite pins, both of which the pool-wide counter this +/// replaced made impossible: +/// +/// 1. namespaces are independent -- a busy table cannot push another table's ids up, so `(namespace, +/// epoch)` ids are dense `1..T` and a reader can tell "this stream is complete" from the ids alone; +/// 2. an attempt that provably sent nothing consumes nothing -- the next caller re-derives the SAME +/// id, so a refusal leaves no hole behind it. +/// +/// The read side enforces exactly what the allocator produces: `RefTableState::applyTxnInPlace` rejects +/// a non-successor id as `CORRUPTED_DATA`, so a hole can never become durable even if some future +/// writer path forgot the rule. +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int MEMORY_LIMIT_EXCEEDED; +extern const int UNKNOWN_FORMAT_VERSION; +} + +using namespace DB::Cas; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; + +namespace +{ + +PoolPtr openPool(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// The fence-controlled pool of `gtest_cas_ref_install_safety.cpp`, for the pre-attempt refusal: the +/// boot clock is frozen so `setMountDeadline` alone decides both fence predicates, renewal is parked an +/// hour out so nothing re-arms the deadline underneath the test, and the single-attempt budget makes +/// `attempt_timeout_ms + lease_safety_margin_ms` (200 ms) the window between "the flush is admitted" +/// and "an attempt may start". +PoolPtr openPoolFenceControlled(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; + cfg.boot_ms_fn = [] { return uint64_t{0}; }; + cfg.mount_renew_period = std::chrono::milliseconds{3600000}; + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + return Pool::open(backend, cfg); +} + +constexpr uint64_t FENCE_DEADLINE_HEALTHY_MS = 30000; +constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 100; + +/// A bare `Pool::open` with no `_pool_meta` seeded: the path an operator's pool RECREATION takes, and +/// the only one that runs the bootstrap residual + quiesce gates (`seedPoolMetaForRestart` mints the +/// metadata directly and would bypass them). +PoolPtr openPoolWithoutSeeding(const BackendPtr & backend, const String & srid) +{ + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = srid}); +} + +/// Deletes every object whose key contains `substr` ("" = the whole prefix), as an operator clearing +/// the prefix would. Returns how many were removed. +size_t eraseKeysContaining(Backend & backend, const String & substr) +{ + size_t removed = 0; + String cursor; + std::vector keys; + while (true) + { + const ListPage page = backend.list("", cursor, 1000); + for (const ListedKey & listed : page.keys) + if (substr.empty() || listed.key.find(substr) != String::npos) + keys.push_back(listed.key); + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + for (const String & key : keys) + { + const HeadResult h = backend.head(key); + if (h.exists && backend.deleteExact(key, h.token).kind == DeleteOutcome::Kind::Deleted) + ++removed; + } + return removed; +} + +String messageOfThrow(const std::function & fn) +{ + try + { + fn(); + } + catch (const DB::Exception & e) + { + return e.message(); + } + return {}; +} + +/// One ordinary publish transaction, driven straight through the append lane so the committed id is +/// observable: `namespace_birth` while the table is not yet `Live`, then the precommit+promote pair for +/// `ref`. Returns the id the append persisted under. +RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const String & ref, uint64_t ordinal) +{ + return store->appendRefOps(ns, MutationScope::ref(ref), + [&ref, ordinal](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps(ref, ManifestRef{1, ordinal, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +} + +/// INV-1, first half: each namespace has its OWN stream. Two tables are written strictly alternately, +/// so a pool-wide counter would hand them 1,3,5 and 2,4 -- every id unique across the pool and dense +/// nowhere. Per-namespace derivation gives each table 1,2,3.. of its own. +TEST(CASRefContiguousAlloc, TwoNamespacesAllocateIndependently) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns_a{"srv1/contig_ns_a"}; + const RootNamespace ns_b{"srv1/contig_ns_b"}; + + const RefTxnId a1 = publishRef(store, ns_a, "ref_1", 1); + const RefTxnId b1 = publishRef(store, ns_b, "ref_1", 1); + const RefTxnId a2 = publishRef(store, ns_a, "ref_2", 2); + const RefTxnId b2 = publishRef(store, ns_b, "ref_2", 2); + const RefTxnId a3 = publishRef(store, ns_a, "ref_3", 3); + + EXPECT_EQ(a1, (RefTxnId{epoch, 1})); + EXPECT_EQ(a2, (RefTxnId{epoch, 2})); + EXPECT_EQ(a3, (RefTxnId{epoch, 3})) + << "ns_a's third transaction must be its own third id -- the two ns_b transactions interleaved " + "between them belong to a different stream and must not push it up"; + EXPECT_EQ(b1, (RefTxnId{epoch, 1})); + EXPECT_EQ(b2, (RefTxnId{epoch, 2})); +} + +/// INV-1, second half (the free half of the every-attempt rule): a refusal that PROVES nothing was sent +/// consumes no id. The pre-attempt gate refuses while the flush is still admitted -- no fault injection, +/// nothing reaches the backend -- and the very next append on that table commits under the SAME id the +/// refused one would have used. Under the pool-wide counter that id was burned as a "safe gap", which is +/// precisely what makes a durable stream unreadable as a contiguous chain. +TEST(CASRefContiguousAlloc, PreAttemptRefusalConsumesNoId) +{ + auto backend = std::make_shared(); + auto store = openPoolFenceControlled(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/contig_no_gap"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + + store->setMountDeadline(FENCE_DEADLINE_REFUSES_ATTEMPT_MS); + ASSERT_TRUE(store->mayMutate()) << "the flush must still be ADMITTED, or this exercises the " + "top-of-flush gate instead of the pre-attempt one"; + const String refusal = messageOfThrow([&] { publishRef(store, ns, "ref_2", 2); }); + ASSERT_NE(refusal, String()) << "the pre-attempt gate must refuse this append"; + /// Pin WHICH refusal this is. The id-reuse below is only meaningful for a refusal that proves + /// nothing was sent; a different failure (an ambiguous PUT, say) would be free to have landed, and + /// re-deriving its id would then be a collision rather than the no-gap property under test. + EXPECT_NE(refusal.find("was refused BEFORE any request was sent"), String::npos) + << "this test is about the provably-sent-nothing refusal specifically: " << refusal; + EXPECT_NE(refusal.find("the txn id is not consumed"), String::npos) << refusal; + ASSERT_FALSE(store->refLaneWedgedForTest(ns)) << "a refusal that sent nothing must not wedge"; + + store->setMountDeadline(FENCE_DEADLINE_HEALTHY_MS); + EXPECT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{epoch, 2})) + << "the refused attempt sent nothing, so the next caller must re-derive the SAME id -- a refusal " + "must never leave a hole in the durable stream"; +} + +/// The epoch component is the second half of the id, and the sequence is dense WITHIN an epoch: a new +/// mount incarnation restarts its table's sequence at 1 rather than continuing the dead incarnation's +/// numbering. `{E1, 2}` -> `{E2, 1}` is therefore not a gap, and the apply-side check must admit it. +TEST(CASRefContiguousAlloc, EpochChangeRestartsTheSequenceAtOne) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/contig_epoch_reset"}; + + uint64_t e1 = 0; + { + auto predecessor = openPool(backend); + e1 = predecessor->writerEpoch(); + ASSERT_EQ(publishRef(predecessor, ns, "ref_1", 1), (RefTxnId{e1, 1})); + ASSERT_EQ(publishRef(predecessor, ns, "ref_2", 2), (RefTxnId{e1, 2})); + } /// predecessor destroyed: its mount lease is released + + auto successor = openPool(backend); + const uint64_t e2 = successor->writerEpoch(); + ASSERT_GT(e2, e1); + EXPECT_EQ(publishRef(successor, ns, "ref_3", 3), (RefTxnId{e2, 1})) + << "a fresh incarnation starts this table's sequence over at 1"; + EXPECT_EQ(publishRef(successor, ns, "ref_4", 4), (RefTxnId{e2, 2})); +} + +/// The read side is what makes INV-1 an invariant rather than a convention: a transaction whose id is +/// not the successor of `greatest_applied` is CORRUPTED_DATA, naming both ids. Before this task the +/// state machine checked strict increase only, so a stream with a hole applied cleanly and no reader +/// could tell a complete chain from a truncated one. +TEST(CASRefContiguousAlloc, NonSuccessorIdIsRejectedOnApply) +{ + const String ns = "srv1/contig_density"; + constexpr uint64_t kEpoch = 7; + + RefTableState state = replay(DB::Cas::tests::minimalLiveSnapshot(ns, RefTxnId{kEpoch, 1}), {}); + ASSERT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch, 1})); + + /// Strictly greater, but skips {7,2}: admitted before this task, rejected now. + try + { + applyRefLogTxn(state, RefLogTxn{ns, RefTxnId{kEpoch, 3}, publishCommittedOps("r", ManifestRef{1, 1, 1}), std::nullopt}); + FAIL() << "a non-successor id must be rejected"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_NE(e.message().find("7-3"), String::npos) << "the offending id must be named: " << e.message(); + EXPECT_NE(e.message().find("7-1"), String::npos) << "the greatest applied id must be named: " << e.message(); + } + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch, 1})) << "the rejected apply must change nothing"; + + /// A new epoch must ALSO start at 1: continuing the previous epoch's numbering is a hole in the new + /// epoch's stream, which reads exactly like a lost first transaction. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + applyRefLogTxn(state, RefLogTxn{ns, RefTxnId{kEpoch + 1, 2}, publishCommittedOps("r", ManifestRef{1, 1, 1}), std::nullopt}); + }); + + /// The two shapes the allocator can produce are the two the checker admits -- and `nextRefTxnId` is + /// the single rule both sides use, so they cannot drift apart. + EXPECT_EQ(nextRefTxnId(state.getGreatestApplied(), kEpoch), (RefTxnId{kEpoch, 2})); + EXPECT_NO_THROW(applyRefLogTxn(state, RefLogTxn{ns, nextRefTxnId(state.getGreatestApplied(), kEpoch), + publishCommittedOps("r", ManifestRef{1, 1, 1}), std::nullopt})); + ASSERT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch, 2})); + + EXPECT_EQ(nextRefTxnId(state.getGreatestApplied(), kEpoch + 1), (RefTxnId{kEpoch + 1, 1})); + /// The id is admissible, but a Live table crossing into a new epoch also owes INV-2's chain link -- + /// the seal that closed the epoch below, at the slot one past its last durable id. + EXPECT_NO_THROW(applyRefLogTxn(state, RefLogTxn{ns, nextRefTxnId(state.getGreatestApplied(), kEpoch + 1), + publishCommittedOps("r2", ManifestRef{1, 2, 1}), RefTxnId{kEpoch, 3}})); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch + 1, 1})); +} + +/// The format floor. A pool written before contiguous ref streams holds ref logs whose ids this build +/// would read as a corrupt (holed) chain, so opening it must fail closed at the pool metadata, naming +/// recreation as the migration -- CAS is pre-release and has no in-place migration path. +TEST(CASRefContiguousAlloc, OldPoolFormatIsRefusedNamingRecreation) +{ + PoolMeta pm; + pm.pool_id = UInt128{1, 2}; + pm.blob_header_len = 256; + pm.min_reader_generation = G_BUILD; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + + const String current = encodePoolMeta(pm); + EXPECT_NO_THROW(decodePoolMeta(current)); + + /// Rewrite the header-line generation to the last pre-contiguous one, exactly as an older build + /// would have stamped it. + const String from = "\"v\":" + std::to_string(G_BUILD); + const String to = "\"v\":" + std::to_string(kContiguousRefStreamsGeneration - 1); + const size_t at = current.find(from); + ASSERT_NE(at, String::npos); + String old_format = current; + old_format.replace(at, from.size(), to); + + try + { + decodePoolMeta(old_format); + FAIL() << "a pre-contiguous pool must not open"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + EXPECT_NE(e.message().find(fmt::format("CAS pool format {} predates generation-10 mount-attempt-identity floor", + kContiguousRefStreamsGeneration - 1)), String::npos) + << "the message must name the migration: " << e.message(); + } +} + +/// Generation 6 is a recreate-only physical-layout cut. A generation-5 pool has contiguous, +/// incarnation-qualified streams but still repeats the logical namespace in every key; accepting it +/// would silently run the generation-6 parsers over a different grammar. +TEST(CASRefContiguousAlloc, GenerationFiveNamespaceBearingPoolIsRefusedNamingRecreation) +{ + PoolMeta pm; + pm.pool_id = UInt128{1, 2}; + pm.blob_header_len = 256; + pm.min_reader_generation = G_BUILD; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + + const String current = encodePoolMeta(pm); + EXPECT_NO_THROW(decodePoolMeta(current)); + + /// Rewrite the header to the immediately preceding generation, which used + /// `cas/refs///...`. + const String from = "\"v\":" + std::to_string(G_BUILD); + const String to = "\"v\":" + std::to_string(kNamespaceLifeKeyedGeneration); + const size_t at = current.find(from); + ASSERT_NE(at, String::npos); + String old_format = current; + old_format.replace(at, from.size(), to); + ASSERT_EQ(kNamespaceLifeKeyedGeneration + 1, kOpaqueNamespaceLifeLayoutGeneration) + << "this test pins the immediately preceding namespace-bearing generation"; + + try + { + decodePoolMeta(old_format); + FAIL() << "a generation-5 namespace-bearing pool must not open"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + EXPECT_NE(e.message().find(fmt::format("CAS pool format {} predates generation-10 mount-attempt-identity floor", + kNamespaceLifeKeyedGeneration)), String::npos) + << "the message must name the migration: " << e.message(); + } +} + +/// Mutation caught: leaving the pool floor at generation 6 would admit a seal whose independent +/// name-keyed coverage and cleanup collections this build no longer has. Generation 7 is a +/// recreate-only grammar cut, so the immediately preceding generation must fail at pool open. +TEST(CASRefContiguousAlloc, GenerationSixSplitFoldSealPoolIsRefusedNamingRecreation) +{ + PoolMeta pm; + pm.pool_id = UInt128{1, 2}; + pm.blob_header_len = 256; + pm.min_reader_generation = G_BUILD; + pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + + const String current = encodePoolMeta(pm); + const String from = "\"v\":" + std::to_string(G_BUILD); + const String to = "\"v\":6"; + const size_t at = current.find(from); + ASSERT_NE(at, String::npos); + String old_format = current; + old_format.replace(at, from.size(), to); + + try + { + decodePoolMeta(old_format); + FAIL() << "a generation-6 split ref-life fold seal pool must not open"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); + EXPECT_NE(e.message().find("CAS pool format 6 predates generation-10 mount-attempt-identity floor"), String::npos) + << "the message must name the recreate-only grammar cut: " << e.message(); + } +} + +TEST(CASPoolMeta, GcShardsIsPersistedAndOverridesMismatchedReopenConfig) +{ + InMemoryBackend backend; + const Layout layout("p"); + const PoolMeta created = PoolMeta::createOrValidate( + backend, layout, /*blob_header_len=*/256, /*gc_shards=*/4, + BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/true); + EXPECT_EQ(created.gc_shards, 4u); + + const PoolMeta reopened = PoolMeta::createOrValidate( + backend, layout, /*blob_header_len=*/256, /*gc_shards=*/1, + BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/false); + EXPECT_EQ(reopened.gc_shards, 4u); + EXPECT_EQ(decodePoolMeta(backend.get(layout.poolMetaKey())->bytes).gc_shards, 4u); +} + +/// The one path where "an attempt that provably sent nothing consumes nothing" does not hold, and the +/// A known-durable install failure must replay before the next id is derived. Replay installs the +/// stranded transaction, so the next append derives its real contiguous successor. +TEST(CASRefContiguousAlloc, NeedsRecoveryReplaysBeforeAllocatingTheNextId) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/contig_durable_floor"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + + /// One-shot throw inside the post-durable install region: txn {epoch, 2} commits durably and is + /// never installed. The exception is built OUTSIDE the region (building it inside would trip + /// `DENY_ALLOCATIONS_IN_SCOPE` and test the guard instead of the recovery transition). + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + [&] { publishRef(store, ns, "ref_2", 2); }); + store->setInstallRegionProbeForTest(nullptr); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + /// The next append first recovers `{epoch, 2}`, then lands at `{epoch, 3}`. + EXPECT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{epoch, 3})) + << "the stranded transaction is durable, so the next id must be its successor, not itself"; + EXPECT_TRUE(store->resolveRef(ns, "ref_3", /*allow_stale=*/false).has_value()) + << "the append may proceed only after recovery has repaired the cached state"; + + EXPECT_TRUE(store->resolveRef(ns, "ref_2", /*allow_stale=*/false).has_value()) + << "the stranded transaction is back in this cache, which is what repairs the divergence"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + /// The durable stream itself is dense: `1`, `2`, `3` all exist as objects. `ns` was born through + /// the REAL append lane (Stage B Task 4-C), so its objects sit at a real catalog-minted incarnation, + /// not the Stage-A sentinel -- resolve it the same way production discovery does. + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + for (uint64_t seq = 1; seq <= 3; ++seq) + EXPECT_TRUE(backend->head(store->layout().refLogKey(life, RefTxnId{epoch, seq})).exists) + << "log object " << epoch << "-" << seq << " must exist: the durable stream has no hole"; +} + +/// Snapshot publication also recovers a `NeedsRecovery` lane before it captures state. +TEST(CASRefContiguousAlloc, NeedsRecoveryReplaysBeforeSnapshotPublication) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const uint64_t epoch = store->writerEpoch(); + const RootNamespace ns{"srv1/contig_poison_publish"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) << "a healthy table must publish, or the refusal " + "asserted below would prove nothing"; + const auto published_before = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(published_before.has_value()); + + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + [&] { publishRef(store, ns, "ref_2", 2); }); + store->setInstallRegionProbeForTest(nullptr); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + /// The append entry point replays before admitting this transaction. + EXPECT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{epoch, 3})); + + /// Publication is safe because recovery installed the stranded transaction first. + EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + EXPECT_TRUE(store->resolveRef(ns, "ref_2", /*allow_stale=*/false).has_value()) + << "the stranded transaction is durable and was re-derived -- publishing is safe precisely " + "because there is nothing left to omit"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_NE(store->newestPublishedSnapshotIdForTest(ns), published_before); +} + +/// Recreation quiesce, refusal leg. Refusing to OPEN an old-format pool fences nothing: the server that +/// mounted it before the operator acted is still running, still holds its mount lease, and still has +/// queued writes. If "recreate the pool" is followed literally -- clear the prefix, start fresh -- that +/// writer's next flush lands its old-format transactions inside the NEW pool. So a recreation over a +/// prefix whose mount slots are not terminal must fail closed, and must say why, BEFORE the operator +/// clears anything. +TEST(CASRefContiguousAlloc, RecreationRefusedWhileAMountSlotIsStillHeld) +{ + auto backend = std::make_shared(); + auto holder = openPool(backend); + const RootNamespace ns{"srv1/contig_quiesce"}; + ASSERT_EQ(publishRef(holder, ns, "ref_1", 1), (RefTxnId{holder->writerEpoch(), 1})); + + /// The operator removes the pool identity, intending to recreate -- but the holder is still up. + ASSERT_EQ(eraseKeysContaining(*backend, "_pool_meta"), 1u); + + const String message = messageOfThrow([&] { openPoolWithoutSeeding(backend, "test2"); }); + EXPECT_NE(message.find("mount lease(s) under this prefix are still held"), String::npos) + << "the refusal must name the held lease, not merely the residual data: " << message; + EXPECT_NE(message.find("do NOT clear the prefix first"), String::npos) + << "the remedy ordering is the whole point of this gate: " << message; + EXPECT_NE(message.find("server root 'test'"), String::npos) + << "the holder must be identified so the operator knows what to stop: " << message; + + /// And the holder is untouched by the refused recreation: its own stream continues contiguously. + EXPECT_EQ(publishRef(holder, ns, "ref_2", 2), (RefTxnId{holder->writerEpoch(), 2})); +} + +/// Recreation quiesce, acceptance leg. Once the holder is gone its slot carries the graceful-farewell +/// marker -- one of the two clock-free certificates of death the mount protocol already recognises -- +/// so the quiesce gate stops firing and the ordinary bootstrap rules take over: clear the prefix, and +/// the recreation mints a fresh pool. +TEST(CASRefContiguousAlloc, RecreationProceedsOnceTheHolderIsTerminal) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/contig_quiesce_ok"}; + { + auto holder = openPool(backend); + publishRef(holder, ns, "ref_1", 1); + } /// destroyed: the keeper stamps the farewell, making the slot terminal + ASSERT_EQ(eraseKeysContaining(*backend, "_pool_meta"), 1u); + + /// The prefix still holds this pool's data, so the bootstrap still refuses -- but on the ORDINARY + /// residual rule, not the quiesce gate. That difference is the whole assertion: nothing is being + /// held any more. + const String residual = messageOfThrow([&] { openPoolWithoutSeeding(backend, "test2"); }); + EXPECT_EQ(residual.find("still held"), String::npos) + << "a terminal slot must not block recreation: " << residual; + EXPECT_NE(residual.find("refusing to bootstrap over residual data"), String::npos) << residual; + + /// The operator now clears the prefix -- in the order the refusal prescribed -- and the recreation + /// mints a fresh pool that starts its own ref stream at 1. + ASSERT_GT(eraseKeysContaining(*backend, ""), 0u); + auto recreated = openPoolWithoutSeeding(backend, "test"); + EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->writerEpoch(), 1})); +} + +/// The other half of the rule: if the prefix IS cleared while a writer survives (the mistake the +/// refusal above exists to prevent, or a writer that was already mid-flight), the recreated pool's +/// ordinary mount claim is what stops it. The survivor's next lease renewal finds a slot it can no +/// longer hold, its local fence latches shut, and every later write is refused -- so a straggler can +/// never append into the new pool. +/// +/// The recreating mount here is a DIFFERENT server (its own `server_id`), which is what makes the +/// survivor's renewal conclusive. Clearing the prefix also resets the durable writer-epoch counter, so +/// a recreation by the SAME server uuid can be handed the very same `(uuid, epoch)` the survivor still +/// holds -- and the two are then indistinguishable to the lease protocol, which reads the survivor's +/// renewal as its own keeper adopting a refreshed body. That is precisely why the refusal above is the +/// primary defence and this fence is only the backstop: quiescing the holder BEFORE the prefix is +/// cleared is what keeps the ambiguous case from arising at all. +TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) +{ + auto backend = std::make_shared(); + /// The survivor uses the runtime-owned renewal worker, as a real mount does: the runtime terminal + /// consumer is what latches the write fence when a renewal fails, so a keeper-only call would + /// reproduce the failure but not the lifecycle effect it causes. + PoolConfig survivor_cfg{.pool_prefix = "p", .server_root_id = "test"}; + survivor_cfg.background_watermark = true; + survivor_cfg.mount_renew_period = std::chrono::milliseconds{50}; + DB::Cas::tests::seedPoolMetaForRestart(*backend); + auto survivor = Pool::open(backend, survivor_cfg); + const RootNamespace ns{"srv1/contig_survivor"}; + ASSERT_EQ(publishRef(survivor, ns, "ref_1", 1), (RefTxnId{survivor->writerEpoch(), 1})); + const uint64_t skipped_before + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + const uint64_t violations_before + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + + /// The prefix is cleared and the pool recreated underneath the still-running survivor. + ASSERT_GT(eraseKeysContaining(*backend, ""), 0u); + PoolConfig recreated_cfg{.pool_prefix = "p", .server_root_id = "test"}; + recreated_cfg.server_id = UInt128{7, 7}; + auto recreated = Pool::open(backend, recreated_cfg); + ASSERT_TRUE(recreated->mayMutate()); + + /// The survivor's next renewal finds a slot held by a foreign server and fails closed, and the loop + /// latches the local write fence. Bounded wait: a real hang fails the test instead of stalling it. + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (survivor->mayMutate() && std::chrono::steady_clock::now() < deadline) + std::this_thread::sleep_for(std::chrono::milliseconds(20)); + EXPECT_FALSE(survivor->mayMutate()) + << "a survivor whose slot was reclaimed must be fenced closed by its own failing renewal, not " + "left writing into the new pool"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + skipped_before + 1) + << "the conclusive foreign-successor observation must be counted when deposition is detected"; + EXPECT_NE(messageOfThrow([&] { publishRef(survivor, ns, "ref_2", 2); }), String()) + << "the survivor's queued write must be refused"; + + /// The recreated pool is unaffected and owns the stream from 1. + EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->writerEpoch(), 1})); + + /// The survivor's TEARDOWN is the other half, and it is asserted here rather than left to the + /// destructor at scope exit. A terminal keeper must skip release without backend I/O: the renewal + /// conflict already counted the conclusive foreign successor, and teardown must neither double-count + /// it nor stamp a farewell over the successor's slot. + const String survivor_mount_key = recreated->layout().mountKey("test"); + const auto successor_slot_before = backend->get(survivor_mount_key); + ASSERT_TRUE(successor_slot_before.has_value()); + const uint64_t skipped_after_deposition + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + + survivor.reset(); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + skipped_after_deposition) + << "terminal teardown must not count the already-observed successor twice"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + violations_before) + << "and must NOT report an exclusivity violation: this is a failover, not a broken guarantee"; + const auto successor_slot_after = backend->get(survivor_mount_key); + ASSERT_TRUE(successor_slot_after.has_value()); + EXPECT_EQ(successor_slot_after->bytes, successor_slot_before->bytes) + << "the deposed writer must not stamp its farewell over the successor's lease"; + EXPECT_TRUE(recreated->mayMutate()) << "and must not disturb the live successor"; +} diff --git a/src/Disks/tests/gtest_cas_ref_cow_manifest_set.cpp b/src/Disks/tests/gtest_cas_ref_cow_manifest_set.cpp new file mode 100644 index 000000000000..5e1ddea64568 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_cow_manifest_set.cpp @@ -0,0 +1,392 @@ +#include +#include +#include +#include + +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +ManifestRef mref(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +} + +/// =================================================================================== +/// Keyed ops: contains/insert/erase across base+overlay (the "E2 owned-manifest index" work). +/// =================================================================================== + +TEST(CASRefCowManifestSet, EmptySetHasNoMembers) +{ + RefCowManifestSet s; + EXPECT_TRUE(s.empty()); + EXPECT_EQ(s.size(), 0u); + EXPECT_FALSE(s.contains(mref(1, 1, 1))); +} + +TEST(CASRefCowManifestSet, InsertThenContains) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 1u); + EXPECT_FALSE(s.contains(mref(2, 2, 2))); +} + +TEST(CASRefCowManifestSet, InsertMultipleThenContainsEachIndependently) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.insert(mref(1, 1, 2)); + s.insert(mref(2, 1, 1)); + EXPECT_EQ(s.size(), 3u); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_TRUE(s.contains(mref(1, 1, 2))); + EXPECT_TRUE(s.contains(mref(2, 1, 1))); + EXPECT_FALSE(s.contains(mref(3, 3, 3))); +} + +TEST(CASRefCowManifestSet, EraseRemovesAnOverlayOnlyMember) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.erase(mref(1, 1, 1)); + EXPECT_FALSE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 0u); + EXPECT_TRUE(s.empty()); + EXPECT_EQ(s.overlayEntriesForTest(), 0u); /// pure-overlay member: erase removes it outright +} + +TEST(CASRefCowManifestSet, TombstoneThenReinsertWhilePurelyInOverlay) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.erase(mref(1, 1, 1)); + s.insert(mref(1, 1, 1)); /// re-insert -- must not be treated as "still present" + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 1u); +} + +/// =================================================================================== +/// materialize() +/// =================================================================================== + +TEST(CASRefCowManifestSet, MaterializeFoldsOverlayIntoBaseAndEmptiesOverlay) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.insert(mref(1, 1, 2)); + EXPECT_GT(s.overlayEntriesForTest(), 0u); + + s.materialize(); + EXPECT_EQ(s.overlayEntriesForTest(), 0u); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_TRUE(s.contains(mref(1, 1, 2))); + EXPECT_EQ(s.size(), 2u); +} + +TEST(CASRefCowManifestSet, MaterializeOnAnEmptyOverlayIsANoOp) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); + const int64_t use_count_before = s.baseUseCountForTest(); + s.materialize(); /// overlay is already empty + EXPECT_EQ(s.baseUseCountForTest(), use_count_before); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); +} + +TEST(CASRefCowManifestSet, EraseAfterMaterializeTombstonesABaseMember) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.insert(mref(1, 1, 2)); + s.materialize(); /// both now live in `base` + + s.erase(mref(1, 1, 1)); + EXPECT_FALSE(s.contains(mref(1, 1, 1))); + EXPECT_TRUE(s.contains(mref(1, 1, 2))); + EXPECT_EQ(s.size(), 1u); + + s.materialize(); /// tombstone folds away; base member actually removed + EXPECT_FALSE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 1u); +} + +TEST(CASRefCowManifestSet, TombstoneThenReinsertAcrossMaterializedBase) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); /// mref(1,1,1) now lives in `base` + + s.erase(mref(1, 1, 1)); /// tombstone shadowing the base member + s.insert(mref(1, 1, 1)); /// revive the tombstone -- must read as present again + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 1u); + + s.materialize(); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_EQ(s.size(), 1u); +} + +/// =================================================================================== +/// materialize() fast path: fold into a uniquely-owned base IN PLACE, no O(N) copy (E5). +/// =================================================================================== + +TEST(CASRefCowManifestSet, MaterializeReusesBaseWhenUniquelyOwned) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); /// mref(1,1,1) now in base; base is uniquely owned + const void * base_before = s.baseIdentityForTest(); + ASSERT_EQ(s.baseUseCountForTest(), 1); + + s.insert(mref(2, 2, 2)); /// pure-overlay addition + s.erase(mref(1, 1, 1)); /// tombstone a base member + s.materialize(); + + EXPECT_EQ(s.baseIdentityForTest(), base_before); /// folded in place: same base allocation + EXPECT_EQ(s.overlayEntriesForTest(), 0u); + EXPECT_FALSE(s.contains(mref(1, 1, 1))); /// tombstone erased from base + EXPECT_TRUE(s.contains(mref(2, 2, 2))); + EXPECT_EQ(s.size(), 1u); /// net_delta reset, size still exact +} + +TEST(CASRefCowManifestSet, MaterializeBuildsFreshBaseWhenBaseIsShared) +{ + RefCowManifestSet original; + original.insert(mref(1, 1, 1)); + original.materialize(); + const void * shared_base = original.baseIdentityForTest(); + + RefCowManifestSet writer = original; /// shares the base (use_count 2) + ASSERT_EQ(writer.baseUseCountForTest(), 2); + writer.insert(mref(9, 9, 9)); + writer.erase(mref(1, 1, 1)); + writer.materialize(); /// base is shared -> must build a fresh one, mutate nothing shared + + /// Load-bearing correctness pin: the OTHER holder's view is byte-unchanged. + EXPECT_EQ(original.baseIdentityForTest(), shared_base); + EXPECT_TRUE(original.contains(mref(1, 1, 1))); + EXPECT_FALSE(original.contains(mref(9, 9, 9))); + EXPECT_EQ(original.size(), 1u); + + /// The writer folded its overlay into a fresh base of its own. + EXPECT_NE(writer.baseIdentityForTest(), shared_base); + EXPECT_FALSE(writer.contains(mref(1, 1, 1))); + EXPECT_TRUE(writer.contains(mref(9, 9, 9))); + EXPECT_EQ(writer.size(), 1u); +} + +TEST(CASRefCowManifestSet, MaterializeEmptyOverlayIsANoOpEvenWhenUniquelyOwned) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); + ASSERT_EQ(s.baseUseCountForTest(), 1); + const void * base_before = s.baseIdentityForTest(); + s.materialize(); /// overlay already empty: no fold, no reallocation + EXPECT_EQ(s.baseIdentityForTest(), base_before); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); +} + +/// =================================================================================== +/// Copy-on-write isolation + O(1)-copy assertion. +/// =================================================================================== + +TEST(CASRefCowManifestSet, CopyIsIsolatedFromOriginal) +{ + RefCowManifestSet original; + original.insert(mref(1, 1, 1)); + original.materialize(); + + RefCowManifestSet copy = original; + copy.insert(mref(9, 9, 9)); + copy.erase(mref(1, 1, 1)); + + EXPECT_TRUE(original.contains(mref(1, 1, 1))); + EXPECT_FALSE(original.contains(mref(9, 9, 9))); + + EXPECT_FALSE(copy.contains(mref(1, 1, 1))); + EXPECT_TRUE(copy.contains(mref(9, 9, 9))); +} + +TEST(CASRefCowManifestSet, CopySharesBaseUntilEitherSideMaterializesANewOne) +{ + RefCowManifestSet original; + original.insert(mref(1, 1, 1)); + original.materialize(); + + RefCowManifestSet copy = original; + /// A copy shares the SAME base object (refcount bump, no per-element allocation) until a write + /// forces a new base into existence via `materialize()`. + EXPECT_EQ(original.baseUseCountForTest(), 2); + EXPECT_EQ(copy.baseUseCountForTest(), 2); + + copy.insert(mref(2, 2, 2)); /// writes go to `copy`'s overlay; `base` is untouched + EXPECT_EQ(original.baseUseCountForTest(), 2); + EXPECT_EQ(copy.baseUseCountForTest(), 2); + EXPECT_FALSE(original.contains(mref(2, 2, 2))); + + copy.materialize(); /// NOW `copy` points at a fresh base of its own + EXPECT_EQ(original.baseUseCountForTest(), 1); + EXPECT_EQ(copy.baseUseCountForTest(), 1); +} + +/// =================================================================================== +/// size()/net_delta correctness across a longer op sequence, mixing base and overlay changes. +/// =================================================================================== + +TEST(CASRefCowManifestSet, SizeTracksNetDeltaAcrossMixedOps) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.insert(mref(1, 1, 2)); + s.insert(mref(1, 1, 3)); + EXPECT_EQ(s.size(), 3u); + s.materialize(); + EXPECT_EQ(s.size(), 3u); + + s.erase(mref(1, 1, 2)); /// base member removed via overlay tombstone + EXPECT_EQ(s.size(), 2u); + s.insert(mref(1, 1, 4)); /// pure-overlay addition + EXPECT_EQ(s.size(), 3u); + s.erase(mref(1, 1, 4)); /// pure-overlay addition removed outright + EXPECT_EQ(s.size(), 2u); + s.insert(mref(1, 1, 2)); /// revive the earlier tombstone + EXPECT_EQ(s.size(), 3u); + + s.materialize(); + EXPECT_EQ(s.size(), 3u); + EXPECT_TRUE(s.contains(mref(1, 1, 1))); + EXPECT_TRUE(s.contains(mref(1, 1, 2))); + EXPECT_TRUE(s.contains(mref(1, 1, 3))); + EXPECT_FALSE(s.contains(mref(1, 1, 4))); +} + +/// =================================================================================== +/// Drift-detection misuse (throws `CORRUPTED_DATA` in EVERY build, post-consult -- previously a +/// debug-only `chassert`): `insert` requires absence, `erase` requires presence. The ref table's own +/// uniqueness invariant guarantees both before either is ever called, so a violation here means the +/// index has drifted, not that a legitimate caller can trigger it. Failing closed (rather than a silent +/// release-build `net_delta` drift) is what keeps a corrupted history from later hiding a still-live +/// owner. +/// =================================================================================== + +TEST(CASRefCowManifestSet, InsertThrowsWhenAlreadyPresentInOverlay) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s.insert(mref(1, 1, 1)); }); +} + +TEST(CASRefCowManifestSet, InsertThrowsWhenAlreadyPresentInBase) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s.insert(mref(1, 1, 1)); }); +} + +TEST(CASRefCowManifestSet, EraseThrowsWhenAbsent) +{ + RefCowManifestSet s; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s.erase(mref(1, 1, 1)); }); +} + +TEST(CASRefCowManifestSet, EraseThrowsWhenAlreadyTombstoned) +{ + RefCowManifestSet s; + s.insert(mref(1, 1, 1)); + s.materialize(); + s.erase(mref(1, 1, 1)); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s.erase(mref(1, 1, 1)); }); +} + +/// =================================================================================== +/// Fast-vs-forced-slow materialize parity (E5 xhigh review): the in-place fold (uniquely-owned base) +/// and the build-fresh-and-swap fold (a copy still shares the base) must agree on membership and size +/// across randomized op sequences. No iteration surface here, so membership is probed over a fixed +/// keyspace. insert/erase preconditions are respected (guarded by the shared membership) so the two +/// sets never drift and never trip the fail-closed CORRUPTED_DATA guards. +/// =================================================================================== + +TEST(CASRefCowManifestSet, FastAndForcedSlowMaterializeAgreeOverRandomOps) +{ + std::mt19937 rng(20260722); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed for reproducible coverage. + + std::vector keyspace; + for (uint64_t k = 0; k < 10; ++k) + keyspace.push_back(mref(1, k, 1)); + + for (int trial = 0; trial < 60; ++trial) + { + RefCowManifestSet fast; /// never copied -> in-place (uniquely-owned) materialize + RefCowManifestSet slow; /// a live copy is held across each materialize -> forced fresh-base path + + for (int step = 0; step < 150; ++step) + { + const ManifestRef m = keyspace[rng() % keyspace.size()]; + const bool present = fast.contains(m); /// identical in both sets by construction + switch (rng() % 5) + { + case 0: + if (!present) /// respect the insert precondition (absent) + { + fast.insert(m); + slow.insert(m); + } + break; + case 1: + if (present) /// respect the erase precondition (present) + { + fast.erase(m); + slow.erase(m); + } + break; + case 2: /// materialize both, each via its intended path + { + ASSERT_EQ(fast.baseUseCountForTest(), 1) << "fast set must be uniquely owned"; + fast.materialize(); /// in-place fast path + { + RefCowManifestSet pin = slow; /// shares slow's base + ASSERT_EQ(slow.baseUseCountForTest(), 2) << "slow set must be forced onto the copy path"; + slow.materialize(); /// build-fresh-and-swap slow path + } + EXPECT_EQ(fast.overlayEntriesForTest(), 0u) << "trial " << trial << " step " << step; + EXPECT_EQ(slow.overlayEntriesForTest(), 0u) << "trial " << trial << " step " << step; + break; + } + default: + break; /// accumulate overlay without materializing + } + + ASSERT_EQ(fast.size(), slow.size()) << "trial " << trial << " step " << step; + for (const auto & probe : keyspace) + ASSERT_EQ(fast.contains(probe), slow.contains(probe)) << "trial " << trial << " step " << step; + } + + fast.materialize(); + { + RefCowManifestSet pin = slow; + slow.materialize(); + } + EXPECT_EQ(fast.overlayEntriesForTest(), 0u) << "trial " << trial; + EXPECT_EQ(slow.overlayEntriesForTest(), 0u) << "trial " << trial; + EXPECT_EQ(fast.size(), slow.size()) << "trial " << trial; + for (const auto & probe : keyspace) + EXPECT_EQ(fast.contains(probe), slow.contains(probe)) << "trial " << trial; + } +} diff --git a/src/Disks/tests/gtest_cas_ref_cow_map.cpp b/src/Disks/tests/gtest_cas_ref_cow_map.cpp new file mode 100644 index 000000000000..40e4cb6b9cb5 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_cow_map.cpp @@ -0,0 +1,516 @@ +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ + +RefCommittedRow row(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + RefCommittedRow r; + r.manifest_ref = ManifestRef{epoch, seq, ordinal}; + return r; +} + +} + +/// =================================================================================== +/// Keyed ops +/// =================================================================================== + +TEST(CASRefCowMap, EmptyMapHasNoEntries) +{ + RefCowMap m; + EXPECT_TRUE(m.empty()); + EXPECT_EQ(m.size(), 0u); + EXPECT_FALSE(m.contains("a")); + EXPECT_FALSE(m.contains("a")); +} + +TEST(CASRefCowMap, EmplaceThenFind) +{ + RefCowMap m; + const auto [it, inserted] = m.emplace("a", row(1, 1, 1)); + EXPECT_TRUE(inserted); + EXPECT_EQ(m.size(), 1u); + ASSERT_TRUE(m.contains("a")); + EXPECT_EQ(it->second.manifest_ref, (ManifestRef{1, 1, 1})); + EXPECT_EQ(m.at("a").manifest_ref, (ManifestRef{1, 1, 1})); +} + +TEST(CASRefCowMap, EmplaceDoesNotOverwriteExisting) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + const auto [it, inserted] = m.emplace("a", row(2, 2, 2)); + EXPECT_FALSE(inserted); + EXPECT_EQ(it->second.manifest_ref, (ManifestRef{1, 1, 1})); + EXPECT_EQ(m.at("a").manifest_ref, (ManifestRef{1, 1, 1})); /// unchanged +} + +TEST(CASRefCowMap, InsertOrAssignOverwritesExisting) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + const auto [it, inserted] = m.insert_or_assign("a", row(2, 2, 2)); + EXPECT_FALSE(inserted); + EXPECT_EQ(it->second.manifest_ref, (ManifestRef{2, 2, 2})); + EXPECT_EQ(m.at("a").manifest_ref, (ManifestRef{2, 2, 2})); +} + +TEST(CASRefCowMap, InsertOrAssignInsertsWhenAbsent) +{ + RefCowMap m; + const auto [it, inserted] = m.insert_or_assign("a", row(1, 1, 1)); + EXPECT_TRUE(inserted); + EXPECT_EQ(m.size(), 1u); + EXPECT_EQ(it->second.manifest_ref, (ManifestRef{1, 1, 1})); +} + +TEST(CASRefCowMap, EraseByKey) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + EXPECT_EQ(m.erase("a"), 1u); + EXPECT_FALSE(m.contains("a")); + EXPECT_EQ(m.size(), 0u); + EXPECT_EQ(m.erase("a"), 0u); /// already gone: no-op + EXPECT_EQ(m.erase("nonexistent"), 0u); +} + +TEST(CASRefCowMap, AtThrowsOnMissingKey) +{ + RefCowMap m; + EXPECT_THROW(m.at("missing"), std::out_of_range); +} + +TEST(CASRefCowMap, CountMatchesContains) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + EXPECT_EQ(m.count("a"), 1u); + EXPECT_EQ(m.count("b"), 0u); +} + +/// =================================================================================== +/// Ordered iteration -- overlay overrides/tombstones a materialized base (spec: "Ordered +/// iteration: merge-iterate base and overlay ... a standard two-sorted-range merge"). +/// =================================================================================== + +TEST(CASRefCowMap, OrderedIterationOverAllBaseRowsIsSorted) +{ + RefCowMap m; + m.emplace("c", row(1, 3, 1)); + m.emplace("a", row(1, 1, 1)); + m.emplace("b", row(1, 2, 1)); + + std::vector names; + for (const auto [name, r] : m) + names.push_back(name); + EXPECT_EQ(names, (std::vector{"a", "b", "c"})); +} + +TEST(CASRefCowMap, MergedIterationAppliesTombstonesAndOverrides) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.emplace("b", row(1, 2, 1)); + m.emplace("c", row(1, 3, 1)); + m.materialize(); /// a, b, c now live in `base` + + m.insert_or_assign("b", row(9, 9, 9)); /// override b via the overlay + m.erase("c"); /// tombstone c via the overlay + m.emplace("d", row(9, 9, 2)); /// pure-overlay addition (not in base) + + std::vector> seen; + for (const auto [name, r] : m) + seen.emplace_back(name, r.manifest_ref); + + const std::vector> expected = { + {"a", ManifestRef{1, 1, 1}}, + {"b", ManifestRef{9, 9, 9}}, + {"d", ManifestRef{9, 9, 2}}, + }; + EXPECT_EQ(seen, expected); + EXPECT_EQ(m.size(), 3u); +} + +TEST(CASRefCowMap, FindOverlayOnlyKeyIteratesIntoBase) +{ + RefCowMap m; + m.emplace("A", row(1, 1, 1)); + m.emplace("D", row(1, 4, 1)); + m.materialize(); /// A, D now live in `base` + + m.insert_or_assign("B", row(2, 2, 1)); /// overlay-only key between base keys "A" and "D" + + auto it = m.find("B"); + ASSERT_NE(it, m.end()); + EXPECT_EQ(it->first, "B"); + ++it; + ASSERT_NE(it, m.end()); /// must land on "D", not collapse straight to end() + EXPECT_EQ(it->first, "D"); +} + +TEST(CASRefCowMap, EraseByIteratorReturnsNextAndRemovesTheRow) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.emplace("b", row(1, 2, 1)); + m.emplace("c", row(1, 3, 1)); + + auto it = m.find("b"); + ASSERT_TRUE(it != m.end()); + auto next = m.erase(it); + ASSERT_TRUE(next != m.end()); + EXPECT_EQ(next->first, "c"); + EXPECT_FALSE(m.contains("b")); + EXPECT_EQ(m.size(), 2u); +} + +TEST(CASRefCowMap, EraseByIteratorOfLastElementReturnsEnd) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + auto it = m.find("a"); + auto next = m.erase(it); + EXPECT_TRUE(next == m.end()); + EXPECT_TRUE(m.empty()); +} + +/// =================================================================================== +/// materialize() (spec §Materialization) +/// =================================================================================== + +TEST(CASRefCowMap, MaterializeFoldsOverlayIntoFreshBaseAndKeepsValuesUnchanged) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.emplace("b", row(1, 2, 1)); + m.erase("a"); + EXPECT_GT(m.overlayEntriesForTest(), 0u); + + m.materialize(); + EXPECT_EQ(m.overlayEntriesForTest(), 0u); + EXPECT_FALSE(m.contains("a")); + ASSERT_TRUE(m.contains("b")); + EXPECT_EQ(m.at("b").manifest_ref, (ManifestRef{1, 2, 1})); + EXPECT_EQ(m.size(), 1u); +} + +TEST(CASRefCowMap, MaterializeOnAnEmptyOverlayIsANoOp) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.materialize(); + const int64_t use_count_before = m.baseUseCountForTest(); + m.materialize(); /// overlay is already empty + EXPECT_EQ(m.baseUseCountForTest(), use_count_before); + EXPECT_TRUE(m.contains("a")); +} + +TEST(CASRefCowMap, MaterializeDoesNotAffectACopyTakenBeforeIt) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + RefCowMap snapshot_before = m; /// copy shares m's pre-materialize base, owns its own overlay + m.insert_or_assign("a", row(2, 2, 2)); + m.materialize(); + + EXPECT_EQ(m.at("a").manifest_ref, (ManifestRef{2, 2, 2})); + EXPECT_EQ(snapshot_before.at("a").manifest_ref, (ManifestRef{1, 1, 1})); +} + +/// =================================================================================== +/// materialize() fast path: fold into a uniquely-owned base IN PLACE, no O(N) copy (E5). +/// =================================================================================== + +TEST(CASRefCowMap, MaterializeReusesBaseWhenUniquelyOwned) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.materialize(); /// "a" now in base; base is uniquely owned + const void * base_before = m.baseIdentityForTest(); + ASSERT_EQ(m.baseUseCountForTest(), 1); + + m.insert_or_assign("b", row(2, 2, 2)); /// pure-overlay addition + m.erase("a"); /// tombstone a base member + m.materialize(); + + EXPECT_EQ(m.baseIdentityForTest(), base_before); /// folded in place: same base allocation + EXPECT_EQ(m.overlayEntriesForTest(), 0u); + EXPECT_FALSE(m.contains("a")); /// tombstone erased from base + ASSERT_TRUE(m.contains("b")); + EXPECT_EQ(m.at("b").manifest_ref, (ManifestRef{2, 2, 2})); + EXPECT_EQ(m.size(), 1u); /// net_delta reset, size still exact +} + +TEST(CASRefCowMap, MaterializeBuildsFreshBaseWhenBaseIsShared) +{ + RefCowMap original; + original.emplace("a", row(1, 1, 1)); + original.materialize(); + const void * shared_base = original.baseIdentityForTest(); + + RefCowMap writer = original; /// shares the base (use_count 2) + ASSERT_EQ(writer.baseUseCountForTest(), 2); + writer.insert_or_assign("a", row(9, 9, 9)); + writer.emplace("b", row(9, 9, 2)); + writer.materialize(); /// base is shared -> must build a fresh one, mutate nothing shared + + /// Load-bearing correctness pin: the OTHER holder's view is byte-unchanged. + EXPECT_EQ(original.baseIdentityForTest(), shared_base); + EXPECT_EQ(original.at("a").manifest_ref, (ManifestRef{1, 1, 1})); + EXPECT_FALSE(original.contains("b")); + EXPECT_EQ(original.size(), 1u); + + /// The writer folded its overlay into a fresh base of its own. + EXPECT_NE(writer.baseIdentityForTest(), shared_base); + EXPECT_EQ(writer.at("a").manifest_ref, (ManifestRef{9, 9, 9})); + EXPECT_TRUE(writer.contains("b")); + EXPECT_EQ(writer.size(), 2u); +} + +TEST(CASRefCowMap, MaterializeEmptyOverlayIsANoOpEvenWhenUniquelyOwned) +{ + RefCowMap m; + m.emplace("a", row(1, 1, 1)); + m.materialize(); + ASSERT_EQ(m.baseUseCountForTest(), 1); + const void * base_before = m.baseIdentityForTest(); + m.materialize(); /// overlay already empty: no fold, no reallocation + EXPECT_EQ(m.baseIdentityForTest(), base_before); + EXPECT_TRUE(m.contains("a")); +} + +TEST(CASRefCowMap, EqualityComparesEffectiveContentsNotInternalLayout) +{ + RefCowMap a; + a.emplace("x", row(1, 1, 1)); + a.materialize(); /// "x" lives in `base` + + RefCowMap b; + b.emplace("x", row(1, 1, 1)); /// same logical content, but lives entirely in `overlay` + + EXPECT_EQ(a.overlayEntriesForTest(), 0u); + EXPECT_GT(b.overlayEntriesForTest(), 0u); + EXPECT_TRUE(a == b); +} + +/// =================================================================================== +/// Copy-on-write isolation + O(1)-copy assertion (spec §Correctness & testing) +/// =================================================================================== + +TEST(CASRefCowMap, CopyIsIsolatedFromOriginal) +{ + RefCowMap original; + original.emplace("a", row(1, 1, 1)); + original.materialize(); + + RefCowMap copy = original; + copy.insert_or_assign("a", row(9, 9, 9)); + copy.emplace("b", row(9, 9, 9)); + + EXPECT_EQ(original.at("a").manifest_ref, (ManifestRef{1, 1, 1})); + EXPECT_FALSE(original.contains("b")); + + EXPECT_EQ(copy.at("a").manifest_ref, (ManifestRef{9, 9, 9})); + EXPECT_TRUE(copy.contains("b")); +} + +TEST(CASRefCowMap, CopySharesBaseUntilEitherSideMaterializesANewOne) +{ + RefCowMap original; + original.emplace("a", row(1, 1, 1)); + original.materialize(); + + RefCowMap copy = original; + /// A copy shares the SAME base object (refcount bump, no per-row allocation) until a write + /// forces a new base into existence via `materialize()` (spec §Mechanism: "Copy = O(1)"). + EXPECT_EQ(original.baseUseCountForTest(), 2); + EXPECT_EQ(copy.baseUseCountForTest(), 2); + + copy.insert_or_assign("a", row(2, 2, 2)); /// writes go to `copy`'s overlay; `base` is untouched + EXPECT_EQ(original.baseUseCountForTest(), 2); + EXPECT_EQ(copy.baseUseCountForTest(), 2); + + copy.materialize(); /// NOW `copy` points at a fresh base of its own + EXPECT_EQ(original.baseUseCountForTest(), 1); + EXPECT_EQ(copy.baseUseCountForTest(), 1); +} + +/// =================================================================================== +/// Randomized exactness property test: RefCowMap must behave IDENTICALLY to +/// std::map across randomized op sequences (spec §Correctness & +/// testing: "random op sequences ... including copy-then-mutate isolation ... and +/// tombstone/override correctness on the merged iterator"). +/// =================================================================================== + +TEST(CASRefCowMap, PropertyMatchesStdMapOverRandomOps) +{ + std::mt19937 rng(20260717); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + + for (int trial = 0; trial < 50; ++trial) + { + RefCowMap actual; + std::map oracle; + + for (int step = 0; step < 200; ++step) + { + const String key = "ref" + std::to_string(rng() % 12); + const uint32_t action = rng() % 6; + switch (action) + { + case 0: /// emplace + { + RefCommittedRow r = row(1, static_cast(step) + 1, 1); + const bool oracle_inserted = oracle.emplace(key, r).second; + const bool actual_inserted = actual.emplace(key, r).second; + EXPECT_EQ(oracle_inserted, actual_inserted) << "trial " << trial << " step " << step; + break; + } + case 1: /// insert_or_assign + { + RefCommittedRow r = row(2, static_cast(step) + 1, 2); + oracle[key] = r; + actual.insert_or_assign(key, r); + break; + } + case 2: /// erase by key + { + const size_t oracle_erased = oracle.erase(key); + const size_t actual_erased = actual.erase(key); + EXPECT_EQ(oracle_erased, actual_erased) << "trial " << trial << " step " << step; + break; + } + case 3: /// find/contains/at (read-only) + { + EXPECT_EQ(oracle.contains(key), actual.contains(key)) << "trial " << trial << " step " << step; + if (oracle.contains(key)) + EXPECT_EQ(oracle.at(key), actual.at(key)) << "trial " << trial << " step " << step; + break; + } + case 4: /// erase via a found iterator + { + if (auto it = actual.find(key); it != actual.end()) + { + oracle.erase(key); + actual.erase(it); + } + break; + } + case 5: /// materialize -- must not change observable content + { + actual.materialize(); + break; + } + default: + UNREACHABLE(); + } + + ASSERT_EQ(oracle.size(), actual.size()) << "trial " << trial << " step " << step; + + auto oit = oracle.begin(); + auto ait = actual.begin(); + for (; oit != oracle.end() && ait != actual.end(); ++oit, ++ait) + { + ASSERT_EQ(oit->first, ait->first) << "trial " << trial << " step " << step; + ASSERT_EQ(oit->second, ait->second) << "trial " << trial << " step " << step; + } + ASSERT_TRUE(oit == oracle.end()) << "trial " << trial << " step " << step; + ASSERT_TRUE(ait == actual.end()) << "trial " << trial << " step " << step; + } + } +} + +/// =================================================================================== +/// Fast-vs-forced-slow materialize parity (E5 xhigh review): the in-place fold (uniquely-owned base) +/// and the build-fresh-and-swap fold (a copy still shares the base) must produce IDENTICAL merged +/// content, size, and empty overlay across randomized op sequences. This pins that the two code paths +/// -- which handle `net_delta`, tombstones, and overrides differently -- never diverge. +/// =================================================================================== + +TEST(CASRefCowMap, FastAndForcedSlowMaterializeAgreeOverRandomOps) +{ + std::mt19937 rng(20260722); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed for reproducible coverage. + + for (int trial = 0; trial < 60; ++trial) + { + RefCowMap fast; /// never copied -> `materialize` always takes the in-place (uniquely-owned) path + RefCowMap slow; /// a live copy is held across each `materialize` -> forced fresh-base path + + for (int step = 0; step < 150; ++step) + { + const String key = "ref" + std::to_string(rng() % 10); + switch (rng() % 5) + { + case 0: + { + RefCommittedRow r = row(1, static_cast(step) + 1, 1); + fast.emplace(key, r); + slow.emplace(key, r); + break; + } + case 1: + { + RefCommittedRow r = row(2, static_cast(step) + 1, 2); + fast.insert_or_assign(key, r); + slow.insert_or_assign(key, r); + break; + } + case 2: + { + fast.erase(key); + slow.erase(key); + break; + } + case 3: /// materialize both, each via its intended path + { + ASSERT_EQ(fast.baseUseCountForTest(), 1) << "fast map must be uniquely owned"; + fast.materialize(); /// in-place fast path + { + RefCowMap pin = slow; /// shares slow's base + ASSERT_EQ(slow.baseUseCountForTest(), 2) << "slow map must be forced onto the copy path"; + slow.materialize(); /// build-fresh-and-swap slow path + } + EXPECT_EQ(fast.overlayEntriesForTest(), 0u) << "trial " << trial << " step " << step; + EXPECT_EQ(slow.overlayEntriesForTest(), 0u) << "trial " << trial << " step " << step; + break; + } + default: + break; /// accumulate overlay without materializing + } + + /// Content + size parity holds at EVERY step, materialized or not. + ASSERT_EQ(fast.size(), slow.size()) << "trial " << trial << " step " << step; + auto fi = fast.begin(); + auto si = slow.begin(); + for (; fi != fast.end() && si != slow.end(); ++fi, ++si) + { + ASSERT_EQ(fi->first, si->first) << "trial " << trial << " step " << step; + ASSERT_EQ(fi->second, si->second) << "trial " << trial << " step " << step; + } + ASSERT_TRUE(fi == fast.end() && si == slow.end()) << "trial " << trial << " step " << step; + } + + /// A final materialize of both via their two paths must leave identical, fully-folded state. + fast.materialize(); + { + RefCowMap pin = slow; + slow.materialize(); + } + EXPECT_EQ(fast.overlayEntriesForTest(), 0u) << "trial " << trial; + EXPECT_EQ(slow.overlayEntriesForTest(), 0u) << "trial " << trial; + EXPECT_TRUE(fast == slow) << "trial " << trial; + EXPECT_EQ(fast.size(), slow.size()) << "trial " << trial; + } +} diff --git a/src/Disks/tests/gtest_cas_ref_decode_bounds.cpp b/src/Disks/tests/gtest_cas_ref_decode_bounds.cpp new file mode 100644 index 000000000000..a2301f88bb08 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_decode_bounds.cpp @@ -0,0 +1,136 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +/// Stage-1 T11 (spec §3 "Byte limits: encode-side estimation machinery is what dies; the decode-side +/// cap stays"). Two closures: +/// +/// 1. `openObject`'s raw (uncompressed) arm skipped `object_cap` entirely -- only the zstd arm checked +/// the declared decompressed content size against it. A tolerated-unknown-field-padded or raw-body +/// object up to `object_cap` would decode as though it were within budget just because it skipped +/// compression. Fixed by gating the raw arm on the SAME cap. +/// 2. The writer's post-encode budget check (`checkBudget`, called from `encodeRefLogTxn`) must be a +/// real `if`+`throw` (CORRUPTED_DATA), never a debug-only `chassert` -- verified here, not +/// re-implemented (it was already a runtime throw as of stage-1 T8). + +namespace +{ + +/// A single `SetPublishedAt` op whose `ref_name` is padded so its own encoded size (`encodedOpSize`) +/// is exactly `target_bytes` -- same construction as `gtest_cas_ref_chunked_flush.cpp`'s helper of +/// the same shape (not shared: each test file owns its small fixture helpers). +RefOp paddedSetPublishedAtOp(size_t target_bytes) +{ + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = ManifestRef{1, 1, 1}; + op.published_at_ms = 0; + const size_t base = encodedOpSize(op); + op.ref_name = "r" + String(target_bytes - base, 'a'); + return op; +} + +} + +/// --------------------------------------------------------------------------------------------- +/// `openObject`: object_cap must gate a raw (uncompressed) body exactly as it gates a zstd frame's +/// declared content size -- skipping compression must never also skip the size cap. +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefDecodeBounds, RawOverCapObjectRejected) +{ + const FormatTraits & t = traitsFor(FormatId::RefLog); + ASSERT_NE(t.object_cap, 0u); + + /// A raw body strictly larger than the format's object cap. It carries no valid header at all -- + /// the raw arm returns bytes verbatim (or, once fixed, rejects them by size) before any JSON + /// parsing happens, so the content need not be well-formed. + const String oversized(t.object_cap + 1, 'x'); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { openObject(FormatId::RefLog, oversized); }); +} + +TEST(CASRefDecodeBounds, RawAtCapObjectAccepted) +{ + /// The boundary itself must stay legal: exactly `object_cap` bytes, raw, still opens unchanged. + const FormatTraits & t = traitsFor(FormatId::RefLog); + const String at_cap(t.object_cap, 'x'); + EXPECT_EQ(openObject(FormatId::RefLog, at_cap), at_cap); +} + +/// --------------------------------------------------------------------------------------------- +/// `checkBudget` (decode side): the whole-object byte cap is measured over the ACTUAL decoded bytes, +/// not accumulated per-op, so padding smuggled through a tolerant unknown field is caught exactly like +/// padding smuggled through an oversized raw body. +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefDecodeBounds, PaddedNormalTxnOver20MiBRejected) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = ManifestRef{1, 1, 1}; + op.published_at_ms = 1; + txn.ops.push_back(op); + + const String text = encodeRefLogTxn(txn); + ASSERT_GE(text.size(), 2u); + ASSERT_EQ(text[text.size() - 1], '\n'); + ASSERT_EQ(text[text.size() - 2], '}'); + + /// Pad the trailer line with an unknown tolerant field ("zz") -- legal per the wire's evolution + /// policy (`skipUnknown`) -- inflating the decoded object well past `ref_txn_max_bytes` without + /// touching a single op line or the op count. Padding an op line would only trip the per-op cap + /// and prove nothing about this (much larger) whole-transaction bound. + constexpr size_t pad_bytes = ref_txn_max_bytes + (1 << 20); + String padded = text.substr(0, text.size() - 2); + padded += ",\"zz\":\"" + String(pad_bytes, 'A') + "\"}\n"; // NOLINT(modernize-raw-string-literal): mixes '\"' quoting with '\n' line endings across this concatenated literal; a raw string can't hold the newline as-is. + ASSERT_GT(padded.size(), ref_txn_max_bytes); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(padded, txn.ns, txn.txn_id); }); +} + +/// --------------------------------------------------------------------------------------------- +/// Writer side: the post-encode budget check is a real `if`+`throw`, never a debug-only `chassert` -- +/// a release build must reject an over-cap encode, not silently persist it. +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefDecodeBounds, WriterPostEncodeThrowIsRuntime) +{ + /// Constructed directly at the codec level -- bypassing the ledger's op-count admission gate + /// (`ref_txn_max_ops`) -- so the transaction's total encoded size alone drives the outcome: the + /// canonical writer can never reach this state through admission (at most `ref_txn_max_ops` ops at + /// `ref_op_max_bytes` each stays under `ref_txn_max_bytes`), but `encodeRefLogTxn`'s own post-encode + /// `checkBudget` call must still catch a direct over-cap construction as a real exception, not an + /// assert that a release build would silently skip. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + + constexpr size_t op_count = ref_txn_max_bytes / ref_op_max_bytes + 16; + txn.ops.reserve(op_count); + for (size_t i = 0; i < op_count; ++i) + txn.ops.push_back(paddedSetPublishedAtOp(ref_op_max_bytes)); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} diff --git a/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp b/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp new file mode 100644 index 000000000000..5b6fa2070c46 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp @@ -0,0 +1,491 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include +#include + +/// v3 text codec tests for the `EpochSeal` record kind + strict seal grammar added to `cas_ref_log` +/// (stage A task 1, spec INV-2). Split into its own file per the plan's "prefer NEW test files" +/// constraint, rather than extending `gtest_cas_ref_log_format.cpp`. Covers: the new op kind's round +/// trip (including the meta-line `prev_epoch_seal` field), the context-free structural grammar +/// (`validateEpochSealGrammarStructural`, run by both `encodeRefLogTxn` and `decodeRefLogTxn`), and +/// the contextual required-iff rule (`validateEpochSealGrammarContextual`, exercised directly against +/// explicit `life_epoch` values -- its writer-runtime call sites land in later tasks). + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +RefOp epochSealOp() +{ + RefOp op; + op.kind = RefOpKind::EpochSeal; + return op; +} + +RefOp namespaceBirthOp() +{ + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + return op; +} + +} + +/// =================================================================================== +/// refLogTxnIsEpochSeal / refLogTxnIsRemovalClass classification +/// =================================================================================== + +TEST(CASRefEpochSealFormat, IsEpochSealTrueForSoleSealOp) +{ + RefLogTxn txn; + txn.ops.push_back(epochSealOp()); + EXPECT_TRUE(refLogTxnIsEpochSeal(txn)); +} + +TEST(CASRefEpochSealFormat, IsEpochSealFalseForSealPlusOtherOp) +{ + RefLogTxn txn; + txn.ops.push_back(epochSealOp()); + txn.ops.push_back(namespaceBirthOp()); + EXPECT_FALSE(refLogTxnIsEpochSeal(txn)); +} + +TEST(CASRefEpochSealFormat, IsEpochSealFalseForNonSealOp) +{ + RefLogTxn txn; + txn.ops.push_back(namespaceBirthOp()); + EXPECT_FALSE(refLogTxnIsEpochSeal(txn)); +} + +TEST(CASRefEpochSealFormat, IsEpochSealFalseForEmptyOps) +{ + RefLogTxn txn; + EXPECT_FALSE(refLogTxnIsEpochSeal(txn)); +} + +/// Step 3's explicit regression note: an `EpochSeal`-only op vector is not removal-class. +TEST(CASRefEpochSealFormat, RemovalClassIsFalseForEpochSeal) +{ + std::vector ops{epochSealOp()}; + EXPECT_FALSE(refLogTxnIsRemovalClass(ops)); +} + +/// =================================================================================== +/// Round trip +/// =================================================================================== + +TEST(CASRefEpochSealFormat, RoundTripSealAtSequenceOneWithPrevEpochSeal) +{ + /// An empty dead epoch (3) closes with a sequence-1 seal, which is therefore itself required to + /// carry `prev_epoch_seal` chaining to the seal that closed epoch 2 (spec INV-2's grammar: required + /// on exactly sequence 1 of every epoch above genesis). + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{2, 9}; + txn.ops.push_back(epochSealOp()); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + ASSERT_TRUE(decoded.prev_epoch_seal.has_value()); + EXPECT_EQ(*decoded.prev_epoch_seal, (RefTxnId{2, 9})); + ASSERT_EQ(decoded.ops.size(), 1u); + EXPECT_EQ(decoded.ops[0].kind, RefOpKind::EpochSeal); +} + +TEST(CASRefEpochSealFormat, RoundTripSealWithoutPrevEpochSeal) +{ + /// The common case: epoch 2 had real records (greatest applied sequence 5), so its closing seal + /// lands at sequence 6 -- not sequence 1 -- and therefore must NOT carry `prev_epoch_seal`. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{2, 6}; + txn.ops.push_back(epochSealOp()); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + EXPECT_FALSE(decoded.prev_epoch_seal.has_value()); +} + +/// A re-encode of a decoded seal transaction is byte-identical (the encoder is a pure function of the +/// txn), matching the pin `gtest_cas_ref_log_format.cpp` keeps for the other op kinds. +TEST(CASRefEpochSealFormat, ByteIdenticalReencodeWithPrevEpochSeal) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{2, 9}; + txn.ops.push_back(epochSealOp()); + + const String bytes1 = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes1, txn.ns, txn.txn_id); + const String bytes2 = encodeRefLogTxn(decoded); + EXPECT_EQ(bytes1, bytes2); +} + +/// =================================================================================== +/// Structural grammar (validateEpochSealGrammarStructural, via encode/decode -- context-free) +/// =================================================================================== + +TEST(CASRefEpochSealFormat, EncodeRejectsSealTxnWithTwoSealOps) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + txn.ops.push_back(epochSealOp()); + txn.ops.push_back(epochSealOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefEpochSealFormat, EncodeRejectsSealTxnWithSecondNonSealOp) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + txn.ops.push_back(epochSealOp()); + txn.ops.push_back(namespaceBirthOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// Decode-side pin for the same op-count rule (review finding I1): `encodeRefLogTxn` can never +/// produce a 2-op seal body, so only a decode-only splice proves `decodeRefLogTxn` independently +/// re-derives the rule rather than trusting whatever the encoder produced -- deleting the structural +/// validator's call site inside `decodeRefLogTxn` would leave this the only failing test. +TEST(CASRefEpochSealFormat, DecodeRejectsSealTxnWithTwoOpsSpliced) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{2, 6}; + txn.ops.push_back(epochSealOp()); + const String bytes = encodeRefLogTxn(txn); + + const String op_line = "{\"op\":\"epoch_seal\"}\n"; + const auto op_pos = bytes.find(op_line); + ASSERT_NE(op_pos, String::npos); + String tampered = bytes; + tampered.insert(op_pos, op_line); /// two consecutive "epoch_seal" op lines now + + const String old_trailer = "{\"n\":1}\n"; + const auto trailer_pos = tampered.find(old_trailer); + ASSERT_NE(trailer_pos, String::npos); + tampered.replace(trailer_pos, old_trailer.size(), "{\"n\":2}\n"); /// keep the trailer honest + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// Decode-side pin for the same op-count rule, with a DIFFERENT second op kind -- proves the rule +/// rejects any companion op, not just a second `epoch_seal`. +TEST(CASRefEpochSealFormat, DecodeRejectsSealTxnWithSecondNonSealOpSpliced) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{2, 6}; + txn.ops.push_back(epochSealOp()); + const String bytes = encodeRefLogTxn(txn); + + const String op_line = "{\"op\":\"epoch_seal\"}\n"; + const auto op_pos = bytes.find(op_line); + ASSERT_NE(op_pos, String::npos); + String tampered = bytes; + tampered.insert(op_pos + op_line.size(), "{\"op\":\"namespace_birth\"}\n"); + + const String old_trailer = "{\"n\":1}\n"; + const auto trailer_pos = tampered.find(old_trailer); + ASSERT_NE(trailer_pos, String::npos); + tampered.replace(trailer_pos, old_trailer.size(), "{\"n\":2}\n"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealAtNonUnitSequence) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 2}; + txn.prev_epoch_seal = RefTxnId{1, 1}; + txn.ops.push_back(namespaceBirthOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// Decode-side pin for the sequence-1-only rule (review finding I1). `prev_epoch_seal`'s +/// writer_epoch (1) is strictly below the transaction's own (5), satisfying the I3 chain-direction +/// rule, so this isolates the sequence-1 rule specifically rather than incidentally also tripping I3. +TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealAtNonUnitSequenceSpliced) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 2}; + txn.ops.push_back(namespaceBirthOp()); + const String bytes = encodeRefLogTxn(txn); + + const String needle = R"("rs":"2")"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.insert(pos + needle.size(), R"(,"!pse":"1","!pss":"1")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// Well-formedness (review finding M2): a zero component inside `prev_epoch_seal` is rejected the +/// same way a zero component in the primary `txn_id` is (`checkRefTxnIdNonzero`, shared code path). +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealWithZeroWriterEpoch) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{0, 9}; + txn.ops.push_back(epochSealOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealWithZeroRefSequence) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{2, 0}; + txn.ops.push_back(epochSealOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// Decode-side splice: `prev_epoch_seal` present as only one of its two wire fields ("!pse" without +/// "!pss") -- a shape only reachable via corrupted bytes, since the encoder always writes both +/// together. Boundary-plus-one for the additive-field decode contract (Constraint 7). +TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealMissingPssComponent) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{2, 9}; + txn.ops.push_back(epochSealOp()); + const String bytes = encodeRefLogTxn(txn); + + const String needle = R"(,"!pss":"9")"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.erase(pos, needle.size()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// Chain direction (review finding I3): a seal closing epoch E always has id `{E, T+1}`, and the +/// sequence-1 transaction in the next numeric epoch must name it. This remains context-free (a +/// property of one transaction), so it belongs in the structural half; Tasks 2/6 walk this pointer +/// backwards over untrusted decoded bodies and must not have to re-derive the rule themselves. +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealPointingAtSameEpoch) +{ + /// Self-pointer: prev_epoch_seal names the SAME epoch this transaction is in. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.prev_epoch_seal = RefTxnId{5, 3}; + txn.ops.push_back(epochSealOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealPointingAtFutureEpoch) +{ + /// Forward-pointer: prev_epoch_seal names an epoch AFTER this transaction's own. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.prev_epoch_seal = RefTxnId{9, 3}; + txn.ops.push_back(epochSealOp()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// INV-2 materializes every global writer epoch for an existing life. A sequence-1 transaction in +/// epoch E therefore chains to the seal of exactly E-1: accepting an older link would make an omitted +/// epoch look like a proved boundary and let a fold bypass its missing seal. +TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealSkippingImmediateEpoch) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.prev_epoch_seal = RefTxnId{3, 7}; + txn.ops.push_back(epochSealOp()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// A damaged object bypasses the encoder, so the decoder must independently reject the same skipped +/// link before any GC or recovery walker can treat it as boundary evidence. +TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealSkippingImmediateEpochSpliced) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.ops.push_back(namespaceBirthOp()); + const String bytes = encodeRefLogTxn(txn); + + const String needle = R"("rs":"1")"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.insert(pos + needle.size(), R"(,"!pse":"3","!pss":"1")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// Decode-side pin for the chain-direction rule (review finding I3): the encoder's own check would +/// refuse to produce this shape (the two Encode* tests above pin that direction), so a splice into an +/// otherwise-valid sequence-1 body proves decode re-derives the rule independently. +TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealPointingAtSameOrFutureEpochSpliced) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.ops.push_back(namespaceBirthOp()); + const String bytes = encodeRefLogTxn(txn); /// valid: sequence 1, no prev_epoch_seal + + const String needle = R"("rs":"1")"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.insert(pos + needle.size(), R"(,"!pse":"5","!pss":"1")"); /// self-pointer + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// =================================================================================== +/// Contextual grammar (validateEpochSealGrammarContextual, called directly against explicit +/// life_epoch values -- the writer-runtime call sites are wired by later tasks) +/// =================================================================================== + +TEST(CASRefEpochSealFormat, ContextualRejectsMissingPrevEpochSealWhenRequired) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.ops.push_back(namespaceBirthOp()); + /// life_epoch 1 < writer_epoch 3: a sequence-1 txn above genesis MUST carry prev_epoch_seal. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { validateEpochSealGrammarContextual(txn, /*life_epoch=*/1); }); +} + +TEST(CASRefEpochSealFormat, ContextualRejectsPrevEpochSealWhenForbidden) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.prev_epoch_seal = RefTxnId{4, 3}; + txn.ops.push_back(namespaceBirthOp()); + /// life_epoch == writer_epoch == 5: this IS the namespace's genesis sequence-1 txn, so + /// prev_epoch_seal is forbidden -- there is no preceding epoch to chain to. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { validateEpochSealGrammarContextual(txn, /*life_epoch=*/5); }); +} + +/// codex r2 finding 2: "genesis" is per-namespace. A namespace first born at global epoch 5 (not +/// epoch 1) appends {5, 1} with NO prev_epoch_seal -- that IS its genesis, not a transition. +TEST(CASRefEpochSealFormat, ContextualAllowsGenesisBirthAboveEpochOneWithoutPrevEpochSeal) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{5, 1}; + txn.ops.push_back(namespaceBirthOp()); + EXPECT_NO_THROW(validateEpochSealGrammarContextual(txn, /*life_epoch=*/5)); +} + +/// Review finding I2: the `ref_sequence != 1` early return is load-bearing for Task 4's encode call +/// site, which calls this on every txn it mints, including ordinary sequence->=2 transactions in a +/// post-transition epoch that legitimately carry no `prev_epoch_seal`. Pinned on both sides of the +/// life_epoch relation to prove the early return fires regardless of it. +TEST(CASRefEpochSealFormat, ContextualPassesThroughNonSequenceOneAboveLifeEpoch) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 5}; + txn.ops.push_back(namespaceBirthOp()); + /// writer_epoch(3) > life_epoch(1): would be REQUIRED if this were sequence 1. + EXPECT_NO_THROW(validateEpochSealGrammarContextual(txn, /*life_epoch=*/1)); +} + +TEST(CASRefEpochSealFormat, ContextualPassesThroughNonSequenceOneAtOrBelowLifeEpoch) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 5}; + txn.ops.push_back(namespaceBirthOp()); + /// writer_epoch(1) == life_epoch(1): would be FORBIDDEN-if-present if this were sequence 1. + EXPECT_NO_THROW(validateEpochSealGrammarContextual(txn, /*life_epoch=*/1)); +} + +/// =================================================================================== +/// Criticality of the prev_epoch_seal wire fields (review finding M4) +/// =================================================================================== + +/// `!pse`/`!pss` are `!`-prefixed CRITICAL keys: `prev_epoch_seal` is INV-2 chain evidence, and a +/// build that silently dropped it would still pass the structural grammar (absent field => no check) +/// while losing the chain link. Proven here by splicing in a DIFFERENT, genuinely-unrecognized +/// `!`-key (simulating a future critical field this build predates) rather than `!pse`/`!pss` +/// themselves, which this build DOES recognize: `JsonObjectReader::skipUnknown` rejects any +/// unrecognized `!`-prefixed key with `UNKNOWN_FORMAT_VERSION` (never a silent skip), so this pins +/// the general mechanism the meta-line reader relies on to keep `!pse`/`!pss` safe against a decoder +/// that doesn't (yet, or anymore) understand them. +TEST(CASRefEpochSealFormat, DecodeRejectsUnknownCriticalKeyInMetaLine) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + txn.ops.push_back(namespaceBirthOp()); + const String bytes = encodeRefLogTxn(txn); + + const String needle = R"("rs":"1")"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.insert(pos + needle.size(), R"(,"!future_critical_field":"1")"); + + expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// =================================================================================== +/// Regression guard: existing unknown-op-word behavior stays intact after adding "epoch_seal" +/// =================================================================================== + +TEST(CASRefEpochSealFormat, DecodeRejectsUnknownOpWordRegressionGuard) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + txn.ops.push_back(epochSealOp()); + const String bytes = encodeRefLogTxn(txn); + + const String needle = "\"epoch_seal\""; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + String tampered = bytes; + tampered.replace(pos, needle.size(), "\"totally_bogus_op\""); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +/// =================================================================================== +/// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) +/// =================================================================================== + +TEST(CASRefEpochSealFormat, FormatBatteryEpochSeal) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{3, 1}; + txn.prev_epoch_seal = RefTxnId{2, 9}; + txn.ops.push_back(epochSealOp()); + + const String ns = txn.ns; + const RefTxnId id = txn.txn_id; + runFormatBattery({FormatId::RefLog, + [txn] { return sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); }, + [ns, id](std::string_view s) { decodeRefLogTxn(openObject(FormatId::RefLog, s), ns, id); }, + "{\"type\":\"cas_ref_log\",\"v\":10}\n" + "{\"ns\":\"ns\",\"we\":\"3\",\"rs\":\"1\",\"!pse\":\"2\",\"!pss\":\"9\"}\n" + "{\"op\":\"epoch_seal\"}\n" + "{\"n\":1}\n"}); +} diff --git a/src/Disks/tests/gtest_cas_ref_gc.cpp b/src/Disks/tests/gtest_cas_ref_gc.cpp new file mode 100644 index 000000000000..97d637982af6 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_gc.cpp @@ -0,0 +1,987 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include + +#include + +/// Task 12 required GC tests over the snapshot+log ref model (spec 2026-07-11-cas-ref-table-snapshot-log-design). +/// Every fixture produces REAL wire-format ref logs (via the writer or `writeRefLogTxnRaw`, never hand-rolled +/// bytes), and every test proves the fold actually consumed them (cursor advanced / nonzero in-degree), so a +/// silent no-op fold cannot pass vacuously. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +namespace ProfileEvents +{ +extern const Event CASRefGlobalListPages; +extern const Event CASRefLogBodyGets; +extern const Event CASRefManifestBodyFoldGets; +extern const Event CASRefEmittedEdges; +extern const Event CASRefCleanupObjectsDeleted; +} + +namespace +{ +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +const UInt128 kGc2 = hexToU128("00000000000000000000000000000002"); + +ManifestRef mref(uint64_t seq, uint32_t ord = 1) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = ord}; +} + +/// Append a committed-ref log at an EXPLICIT sequence (no per-call LIST) -- fast bulk seeding of a +/// >1000-key stream. The ops are replay-valid (birth on the first, then add-precommit + promote). +void seedCommittedAt( + Backend & backend, const Layout & layout, const RootNamespace & ns, uint64_t seq, + const String & ref_name, const ManifestRef & mr, bool birth) +{ + std::vector ops; + if (birth) + ops.push_back(namespaceBirthOp()); + const std::vector commit_ops = publishCommittedOps(ref_name, mr); + ops.insert(ops.end(), commit_ops.begin(), commit_ops.end()); + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = RefTxnId{1, seq}; + txn.ops = std::move(ops); + fixture::writeRefLogRaw(backend, layout, txn); +} + +/// Drive regular rounds, renewing the mount ack after each, until quiescent or `max_rounds`. +size_t runToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyCondemnedInSeal(s->backend(), s->layout())) + break; + } + return rounds; +} + +bool blobPresent(Backend & b, const Layout & layout, const UInt128 & hash) +{ + return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; +} + +/// Denies ONCE the single round-commit `gc/state` CAS that advances `snap_generation` (the losing +/// leader deposed mid-round). The denied round leaves only never-adopted attempt-scoped debris. +class DeposeRoundCommitBackend : public InMemoryBackend +{ +public: + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (arm && key == "p/gc/state") + { + const auto stored = get(key); + const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; + if (decodeGcState(bytes).snap_generation > stored_gen) + { + arm = false; + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "test-injected: round-commit gc/state CAS denied (losing leader deposed mid-round)"); + } + } + return InMemoryBackend::casPut(key, bytes, expected, meta); + } + bool arm = false; +}; + +/// Moves one of the two authorities `cleanupRefObjects` must revalidate at a precise ref-log delete +/// boundary. The target object's own token is untouched, so only an authority check can refuse it. +class RefCleanupAuthorityRaceBackend : public CountingBackend +{ +public: + enum class Authority : uint8_t + { + Catalog, + GcFence, + }; + + enum class Timing : uint8_t + { + BeforeFirstDelete, + AfterFirstDelete, + }; + + void arm( + Authority authority_, Timing timing_, const Layout & layout, + const String & first_cleanup_key_) + { + authority = authority_; + timing = timing_; + catalog_key = layout.refCatalogKey(); + gc_state_key = layout.gcStateKey(); + first_cleanup_key = first_cleanup_key_; + armed = true; + } + + HeadResult head(const String & key) override + { + HeadResult result = CountingBackend::head(key); + if (armed && timing == Timing::BeforeFirstDelete && key == first_cleanup_key) + moveAuthority(); + return result; + } + + DeleteOutcome deleteExact(const String & key, const Token & token) override + { + DeleteOutcome result = CountingBackend::deleteExact(key, token); + if (armed && timing == Timing::AfterFirstDelete && key == first_cleanup_key) + moveAuthority(); + return result; + } + +private: + void moveAuthority() + { + armed = false; + const String & key = authority == Authority::Catalog ? catalog_key : gc_state_key; + const auto got = CountingBackend::get(key); + if (!got) + throw std::runtime_error("test-injected cleanup authority object is absent"); + + String bytes = got->bytes; + if (authority == Authority::GcFence) + { + GcState moved = decodeGcState(bytes); + ++moved.lease.seq; + bytes = encodeGcState(moved); + } + if (CountingBackend::casPut(key, bytes, got->token).outcome != CasOutcome::Committed) + throw std::runtime_error("test-injected cleanup authority move lost its CAS"); + } + + Authority authority = Authority::Catalog; + Timing timing = Timing::BeforeFirstDelete; + String catalog_key; + String gc_state_key; + String first_cleanup_key; + bool armed = false; +}; + +struct RefCleanupFixture +{ + String first_log_key; + String second_log_key; +}; + +RefCleanupFixture seedTwoCoveredLogs( + RefCleanupAuthorityRaceBackend & backend, const Layout & layout, + const RootNamespace & ns) +{ + fixture::admitLive(backend, layout, ns); + const ManifestRef r1 = mref(1); + const ManifestRef r2 = mref(2); + const ManifestRef r3 = mref(3); + writeManifestRaw(backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + writeManifestRaw(backend, layout, ns, r3, {blobEntryFor("c", DB::UInt128(3))}); + const uint64_t v1 = publishCommittedTransition(backend, layout, ns, "t1", std::nullopt, r1); + const uint64_t v2 = publishCommittedTransition(backend, layout, ns, "t2", std::nullopt, r2); + const uint64_t v3 = publishCommittedTransition(backend, layout, ns, "t3", std::nullopt, r3); + writeRefSnapshotRaw(backend, layout, + minimalLiveSnapshot(ns.string(), RefTxnId{1, v3}, + {committedRow("t1", r1), committedRow("t2", r2), committedRow("t3", r3)})); + replaceRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, v3}, + .checkpoint_snapshot_id = RefTxnId{1, v3}, + .last_epoch_seal = std::nullopt, + }); + const NamespaceLifeId life = fixture::fixtureLife(ns); + return { + .first_log_key = layout.refLogKey(life, RefTxnId{1, v1}), + .second_log_key = layout.refLogKey(life, RefTxnId{1, v2})}; +} +} + +/// (1) A >1000-key ref scan folds every pre-existing log exactly once: the cursor advances to the greatest +/// id and every referenced blob has in-degree exactly 1 (folded once, not skipped, not doubled). +TEST(CASRefGc, LargeRefScanFoldsEveryLogExactlyOnce) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + constexpr uint64_t N = 1200; /// > 1000: forces multi-page LIST paging in the fold's global scan + for (uint64_t i = 1; i <= N; ++i) + { + const ManifestRef mr = mref(i); + writeManifestRaw(*backend, layout, ns, mr, {blobEntryFor("data", DB::UInt128(i))}); + seedCommittedAt(*backend, layout, ns, /*seq*/ i, "t" + std::to_string(i), mr, /*birth*/ i == 1); + } + writeRecoverableCkptForRawFixture( + *backend, layout, ns, + RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, N}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + + Gc gc(store, kGc); + ASSERT_NO_THROW(gc.runRegularRound()); + + /// The durable cursor advanced to the greatest log id. + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), N) + << "the fold must advance the per-table cursor to the greatest pre-existing log id"; + + /// Every referenced blob folded EXACTLY once (in-degree 1). Spot-check a spread across the >1000 set. + for (uint64_t i : {uint64_t{1}, uint64_t{2}, uint64_t{999}, uint64_t{1000}, uint64_t{1001}, N}) + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) + << "blob " << i << " must be folded exactly once (not skipped, not doubled)"; +} + +/// (2) A concurrent log appended AFTER the round's scan has passed its table is NOT skipped: the sealed +/// cursor stays below it, and the next round folds it. +TEST(CASRefGc, ConcurrentLogAfterScanIsFoldedNextRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r1 = mref(1); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r1); + + Gc gc(store, kGc); + gc.runRegularRound(); /// round 1 folds v1 + ASSERT_EQ(foldCursorOf(*backend, layout, ns, 0), v1); + ASSERT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1); + + /// A NEW log lands after the round sealed its cursor at v1 (a concurrent writer). + const ManifestRef r2 = mref(2); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "tbl2", std::nullopt, r2); + ASSERT_GT(v2, v1); + + /// The sealed cursor is still v1 (< v2) -- the new log was never skipped past. + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), v1) + << "a log that landed after the scan must remain below the durable cursor, never skipped"; + + gc.runRegularRound(); /// round 2 folds v2 + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), v2); + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1) + << "the next round must fold the concurrently-appended log"; +} + +/// (3) Fold barrier: a live precommit whose manifest body is absent clamps the table cursor below its +/// log (an anomaly is recorded), then folds once the body appears. +TEST(CASRefGc, FoldBarrierClampsBelowMissingBodyThenFoldsOnAppear) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef pre = mref(7); + /// No writeManifestRaw for `pre`: its body is intentionally absent (the live precommit's barrier). + const uint64_t v = addPrecommitTransition(*backend, layout, ns, DB::UInt128(9), "part", std::nullopt, pre); + + Gc gc(store, kGc); + RoundReport report; + ASSERT_NO_THROW(report = gc.runRegularRound()); + EXPECT_TRUE(report.hasAnomaly(ns, /*shard*/0)) << "a missing live-precommit body must record an anomaly"; + EXPECT_LT(foldCursorOf(*backend, layout, ns, 0), v) + << "the barrier must clamp the durable cursor BELOW the bodiless-precommit log"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 0); + + /// The body appears (the build finished staging): the next fold passes the barrier. + writeManifestRaw(*backend, layout, ns, pre, {blobEntryFor("p", DB::UInt128(1))}); + gc.runRegularRound(); + EXPECT_GE(foldCursorOf(*backend, layout, ns, 0), v) << "the barrier lifts once the body lands"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1); +} + +/// (4) Edge cancellation: a manifest added then removed across a batch nets to zero in-degree and the +/// exclusively-owned blob is reclaimed. +TEST(CASRefGc, EdgeCancellationAddThenRemoveReclaimsBlob) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = mref(1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); /// +1 for r's blob + dropRefTransition(*backend, layout, ns, "tbl", r); /// -1: the add is cancelled + + Gc gc(store, kGc); + ASSERT_TRUE(runToFixpoint(store, gc) < 64u) << "the add+remove batch must converge to a fixpoint"; + + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 0) + << "an added-then-removed manifest nets to zero in-degree"; + EXPECT_FALSE(blobPresent(*backend, layout, DB::UInt128(1))) + << "the net-zero blob is reclaimed"; +} + +/// (5) A losing generation commit adopts nothing and deletes nothing: a round whose single round-commit +/// `gc/state` CAS is denied (deposed mid-round) must NOT advance the adopted (snap_generation, snap_attempt) +/// and must NOT delete the condemned-but-unadopted blob. Its fold seal is durable only under its OWN +/// never-adopted attempt (harmless debris). +TEST(CASRefGc, LosingGenerationCommitAdoptsNothingDeletesNothing) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = mref(1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + gc.runRegularRound(); /// round 1: folds the +1 and adopts it cleanly + store->renewWatermarkOnce(); + const auto adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + ASSERT_GT(adopted.snap_generation, 0u); + + /// Drop the ref, then run the round whose commit is DENIED (losing leader). + dropRefTransition(*backend, layout, ns, "tbl", r); + backend->arm = true; + EXPECT_ANY_THROW(gc.runRegularRound()); + backend->arm = false; + + /// The deposed round adopted NOTHING: the durable pointers are unchanged... + const auto after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + EXPECT_EQ(after.snap_generation, adopted.snap_generation) + << "a denied round-commit CAS must not advance the adopted generation"; + EXPECT_EQ(after.snap_attempt, adopted.snap_attempt); + /// ...and it deleted NOTHING: the blob its unadopted fold condemned is still present. + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(1))) + << "a losing generation commit must never delete a blob against an unadopted fold"; +} + +/// (6) Ref-object cleanup trusts only a checkpoint-named recovery triple: an older `_log` and `_snap` +/// are deleted after the durable cursor reaches them, while that triple remains intact. +TEST(CASRefGc, RefObjectCleanupRetainsCheckpointNamedTriple) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + /// Two committed publishes -> logs {1,1} and {1,2}. + const ManifestRef r1 = mref(1); + const ManifestRef r2 = mref(2); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "t1", std::nullopt, r1); + const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "t2", std::nullopt, r2); + + /// Two observed snapshots: an OLD one covering only v1, and the NEWEST covering v2. Both are real + /// wire-format snapshot objects (the recovery codec reads them). + RefTableSnapshot old_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v1}, + {committedRow("t1", r1)}); + RefTableSnapshot new_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v2}, + {committedRow("t1", r1), committedRow("t2", r2)}); + writeRefSnapshotRaw(*backend, layout, old_snap); + writeRefSnapshotRaw(*backend, layout, new_snap); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, v2}, + .checkpoint_snapshot_id = RefTxnId{1, v2}, + .last_epoch_seal = std::nullopt, + }); + + const String log_v1_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, v1}); + const String log_v2_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, v2}); + const String old_snap_key = layout.refSnapshotKey(fixture::fixtureLife(ns), RefTxnId{1, v1}); + const String new_snap_key = layout.refSnapshotKey(fixture::fixtureLife(ns), RefTxnId{1, v2}); + ASSERT_TRUE(backend->head(log_v1_key).exists); + ASSERT_TRUE(backend->head(log_v2_key).exists); + ASSERT_TRUE(backend->head(old_snap_key).exists); + + Gc gc(store, kGc); + runToFixpoint(store, gc); /// folds v1,v2 (cursor -> v2) then cleans covered ref objects post-CAS + + /// The old log lies below both the durable cursor and the validated checkpoint base => DELETED. + EXPECT_FALSE(backend->head(log_v1_key).exists) + << "a log below the checkpoint-named snapshot base and durable cursor must be deleted"; + /// The same-id ordinary log is part of recovery's triple and must survive. + EXPECT_TRUE(backend->head(log_v2_key).exists) + << "the checkpoint-named non-seal log must survive with its snapshot"; + /// The older snapshot is deleted; the checkpoint-named snapshot is retained. + EXPECT_FALSE(backend->head(old_snap_key).exists) << "an older snapshot must be deleted"; + EXPECT_TRUE(backend->head(new_snap_key).exists) << "the checkpoint-named snapshot must be retained"; +} + +/// `cleanupRefObjects`'s per-round cap. Five deletable logs share one +/// namespace with a tiny `gc_round_ref_cleanup_budget`; the per-key fail-close validation +/// (`deleteRefObject`'s catalog/lease revalidation before every exact delete) is untouched -- it is +/// NOT amortized, only the cohort size per round is capped. `planRefCleanup` recomputes the same +/// remaining candidates from durable state every round, so the excess needs no cursor of its own. +TEST(CASRefGc, RefObjectCleanupRespectsRoundBudgetAndConvergesAcrossRounds) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_round_ref_cleanup_budget = 1, + .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, layout, ns); + + /// Six sequential replacements of the SAME ref -> six committed logs {1,1}..{1,6}. + constexpr int kLogs = 6; + std::optional prev; + ManifestRef latest{}; + uint64_t last_seq = 0; + for (int i = 1; i <= kLogs; ++i) + { + const ManifestRef r = mref(i); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a" + std::to_string(i), DB::UInt128(static_cast(i)))}); + last_seq = publishCommittedTransition(*backend, layout, ns, "t", prev, r); + prev = r; + latest = r; + } + + /// A snapshot + checkpoint naming the LATEST row: every earlier log is below the checkpoint base. + RefTableSnapshot snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, last_seq}, {committedRow("t", latest)}); + writeRefSnapshotRaw(*backend, layout, snap); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, last_seq}, + .checkpoint_snapshot_id = RefTxnId{1, last_seq}, + .last_epoch_seal = std::nullopt, + }); + + std::vector deletable_log_keys; + for (int i = 1; i < kLogs; ++i) /// {1,1}..{1,5}: strictly below the checkpoint base, hence deletable + deletable_log_keys.push_back(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, static_cast(i)})); + + Gc gc(store, kGc); + auto countSurviving = [&] + { + size_t n = 0; + for (const String & k : deletable_log_keys) + if (backend->head(k).exists) + ++n; + return n; + }; + ASSERT_EQ(countSurviving(), deletable_log_keys.size()) + << "nothing cleaned before the first round even runs"; + + /// The SAME round that folds the whole tail also runs post-CAS cleanup, and with + /// `gc_round_ref_cleanup_budget = 1` deletes exactly one of the five deletable candidates. + runRegularRoundReclaiming(gc); + EXPECT_EQ(countSurviving(), deletable_log_keys.size() - 1) + << "a round with gc_round_ref_cleanup_budget=1 must delete exactly one ref object"; + + /// Repeated budgeted rounds converge: the whole deletable tail eventually drains, none stranded. + for (int i = 0; i < 10 && countSurviving() > 0; ++i) + runRegularRoundReclaiming(gc); + EXPECT_EQ(countSurviving(), 0u) + << "the whole deletable tail must eventually drain under repeated budgeted rounds"; +} + +TEST(CASRefGc, RefObjectCleanupRetainsCheckpointPredecessorSealProof) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/cross-epoch-cleanup@cas@"}; + fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = fixture::fixtureLife(ns); + const RefTxnId birth_id{1, 1}; + const RefTxnId seal_id{1, 2}; + const RefTxnId base_id{2, 1}; + + const RefLogTxn birth{ + .ns = ns.string(), + .txn_id = birth_id, + .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}; + RefOp seal_op; + seal_op.kind = RefOpKind::EpochSeal; + const RefLogTxn seal{ + .ns = ns.string(), + .txn_id = seal_id, + .ops = {std::move(seal_op)}, + .prev_epoch_seal = std::nullopt}; + const RefLogTxn base{ + .ns = ns.string(), + .txn_id = base_id, + .ops = {}, + .prev_epoch_seal = seal_id}; + fixture::writeRefLogRaw(*backend, layout, birth); + fixture::writeRefLogRaw(*backend, layout, seal); + fixture::writeRefLogRaw(*backend, layout, base); + + RefTableState state; + applyRefLogTxn(state, birth); + writeRefSnapshotRaw(*backend, layout, snapshotOf(state, ns.string())); + applyRefLogTxn(state, seal); + applyRefLogTxn(state, base); + writeRefSnapshotRaw(*backend, layout, snapshotOf(state, ns.string())); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = base_id, + .checkpoint_snapshot_id = base_id, + .last_epoch_seal = seal_id}); + + Gc gc(store, kGc); + runToFixpoint(store, gc); + + EXPECT_TRUE(backend->head(layout.refLogKey(life, seal_id)).exists) + << "cleanup must retain the predecessor seal that proves the checkpoint base's epoch transition"; + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(*backend, layout); + const auto entry = std::find_if(cut.catalog.entries.begin(), cut.catalog.entries.end(), + [&](const CatalogEntry & candidate) { return candidate.ns == ns; }); + ASSERT_NE(entry, cut.catalog.entries.end()); + const std::optional checkpoint = readCkpt(*backend, layout, life); + ASSERT_TRUE(checkpoint); + EXPECT_NO_THROW((void)recoverRefTableDetailedFromAuthority(*backend, layout, *entry, checkpoint->ckpt)); +} + +TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBeforeFirstDeleteRefusesEveryRefObjectDelete) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); + backend->arm( + RefCleanupAuthorityRaceBackend::Authority::Catalog, + RefCleanupAuthorityRaceBackend::Timing::BeforeFirstDelete, layout, keys.first_log_key); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + EXPECT_TRUE(backend->head(keys.first_log_key).exists); + EXPECT_TRUE(backend->head(keys.second_log_key).exists); + EXPECT_EQ(backend->deleteCount(keys.first_log_key), 0u); + EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); +} + +TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBetweenKeysAllowsFirstAndRefusesSecondDelete) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); + backend->arm( + RefCleanupAuthorityRaceBackend::Authority::Catalog, + RefCleanupAuthorityRaceBackend::Timing::AfterFirstDelete, layout, keys.first_log_key); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + EXPECT_FALSE(backend->head(keys.first_log_key).exists); + EXPECT_TRUE(backend->head(keys.second_log_key).exists); + EXPECT_EQ(backend->deleteCount(keys.first_log_key), 1u); + EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); +} + +TEST(CASRefGcCleanupAuthority, GcFenceMoveBeforeFirstDeleteRefusesEveryRefObjectDelete) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); + backend->arm( + RefCleanupAuthorityRaceBackend::Authority::GcFence, + RefCleanupAuthorityRaceBackend::Timing::BeforeFirstDelete, layout, keys.first_log_key); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + EXPECT_TRUE(backend->head(keys.first_log_key).exists); + EXPECT_TRUE(backend->head(keys.second_log_key).exists); + EXPECT_EQ(backend->deleteCount(keys.first_log_key), 0u); + EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); +} + +TEST(CASRefGcCleanupAuthority, GcFenceMoveBetweenKeysAllowsFirstAndRefusesSecondDelete) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); + backend->arm( + RefCleanupAuthorityRaceBackend::Authority::GcFence, + RefCleanupAuthorityRaceBackend::Timing::AfterFirstDelete, layout, keys.first_log_key); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + EXPECT_FALSE(backend->head(keys.first_log_key).exists); + EXPECT_TRUE(backend->head(keys.second_log_key).exists); + EXPECT_EQ(backend->deleteCount(keys.first_log_key), 1u); + EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); +} + +/// Task 13 (spec §implementation-impact / §GC Budget): one fold+clean round increments every ref-intake +/// observability counter -- global LIST pages (Q), log-body GETs (K), manifest-body fold GETs (H), emitted +/// manifest edges, and cleaned old ref objects (D). Before/after deltas prove each site actually fires. +TEST(CASRefGc, RefIntakeIncrementsObservabilityCounters) +{ + using ProfileEvents::global_counters; + const auto list_pages_before = global_counters[ProfileEvents::CASRefGlobalListPages].load(); + const auto log_gets_before = global_counters[ProfileEvents::CASRefLogBodyGets].load(); + const auto mf_gets_before = global_counters[ProfileEvents::CASRefManifestBodyFoldGets].load(); + const auto edges_before = global_counters[ProfileEvents::CASRefEmittedEdges].load(); + const auto cleaned_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r1 = mref(1); + const ManifestRef r2 = mref(2); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "t1", std::nullopt, r1); + const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "t2", std::nullopt, r2); + /// A checkpoint-named snapshot base makes older listed objects eligible for cleanup once folded. + writeRefSnapshotRaw(*backend, layout, + minimalLiveSnapshot(ns.string(), RefTxnId{1, v2}, {committedRow("t1", r1), committedRow("t2", r2)})); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, v2}, + .checkpoint_snapshot_id = RefTxnId{1, v2}, + .last_epoch_seal = std::nullopt, + }); + (void)v1; + + Gc gc(store, kGc); + runToFixpoint(store, gc); + + EXPECT_GT(global_counters[ProfileEvents::CASRefGlobalListPages].load(), list_pages_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefLogBodyGets].load(), log_gets_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefManifestBodyFoldGets].load(), mf_gets_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefEmittedEdges].load(), edges_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(), cleaned_before); +} + +/// Task 13 e2e (in-process regression twin of the rustfs integration test): the whole snapshot+log +/// lifecycle over real wire-format objects and real GC rounds -- publish committed refs across two +/// tables, replace one (dropping a blob), publish a covering snapshot, drive GC to a fixpoint, and +/// assert the fold + ref-object cleanup + snapshot lifecycle plus the two read-only consumers: +/// `runFsck(*store).clean()` (the fsck CLI's verdict, oracle included) and `gc.previewDeletes().empty()` +/// (what `cas-gc-dryrun` reports). This is the deterministic permanent twin the unit sweep keeps running. +TEST(CASRefGc, RefSnaplogLifecycleE2E) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns_a{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns_a); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + const RootNamespace ns_b{"00/bb@cas@"}; + + /// Two tables with committed refs naming present manifests + blobs (insert-like). ns_a's ref is then + /// re-published to a second manifest, dropping the first manifest's blob (a replace: -1 old, +1 new). + const ManifestRef a1 = mref(1); + const ManifestRef a2 = mref(2); + const ManifestRef b1 = mref(3); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeBlobBody(*backend, layout, DB::UInt128(2)); + writeBlobBody(*backend, layout, DB::UInt128(3)); + writeManifestRaw(*backend, layout, ns_a, a1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns_a, a2, {blobEntryFor("a", DB::UInt128(2))}); + writeManifestRaw(*backend, layout, ns_b, b1, {blobEntryFor("b", DB::UInt128(3))}); + const uint64_t va1 = publishCommittedTransition(*backend, layout, ns_a, "t", std::nullopt, a1); + const uint64_t va2 = publishCommittedTransition(*backend, layout, ns_a, "t", a1, a2); /// replace a1 -> a2 + publishCommittedTransition(*backend, layout, ns_b, "t", std::nullopt, b1); + /// The semantic transition helper has already published the exact CTE for each life. + + /// The writer's compaction: a snapshot of ns_a covering its greatest log (va2), the same + /// deterministic bytes the oracle recomputes. + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + const RefTableState sa = recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, catalog_cut, ns_a).state; + writeRefSnapshotRaw(*backend, layout, snapshotOf(sa, ns_a.string())); + const NamespaceLifeId life_a = store->namespaceLife(ns_a); + const CkptSample before_snapshot_publish = *readCkpt(*backend, layout, life_a); + RefCkpt after_snapshot_publish = before_snapshot_publish.ckpt; + after_snapshot_publish.checkpoint_snapshot_id = RefTxnId{1, va2}; + ASSERT_EQ(backend->casPut( + layout.refCkptKey(life_a), encodeRefCkpt(after_snapshot_publish), before_snapshot_publish.token).outcome, + CasOutcome::Committed); + + Gc gc(store, kGc); + runToFixpoint(store, gc); + + /// Snapshot lifecycle: the covering snapshot is retained; the covered logs (folded + snapshot-covered) + /// are cleaned; the replaced manifest's blob is reclaimed while the live blobs survive. + EXPECT_TRUE(backend->head(layout.refSnapshotKey(fixture::fixtureLife(ns_a), RefTxnId{1, va2})).exists) + << "covering snapshot retained"; + EXPECT_FALSE(backend->head(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, va1})).exists) << "covered log cleaned"; + EXPECT_FALSE(blobPresent(*backend, layout, DB::UInt128(1))) << "replaced blob reclaimed"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) << "live blob survives"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(3))) << "other table's blob survives"; + + /// Read-only consumers agree: fsck recovers through the exact checkpoint base and reports no dangle, + /// while cas-gc-dryrun has no pending content deletes. Covered LIST debris is not diagnostic authority. + const FsckReport rep = runFsck(*store, /*detail*/true); + EXPECT_TRUE(rep.clean()); + EXPECT_EQ(rep.dangling, 0u); + EXPECT_TRUE(gc.previewDeletes().empty()) << "cas-gc-dryrun equivalent: no pending content deletes"; +} + +/// (8) A malformed/adversarial ref key aborts ref folding for the round: no partial delta, no cursor +/// advance. The malformed key is a real object under `cas/ns/stream/` whose `RefTxnId` render is invalid. +TEST(CASRefGc, MalformedRefKeyAbortsRefFoldingNoPartialDelta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = mref(1); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + /// The semantic transition helper has already published the exact CTE. + + /// Plant a malformed ref key under the ref prefix (a `_log` with a non-canonical id render). + const NamespaceLifeId life = store->namespaceLife(ns); + backend->putIfAbsent(layout.namespaceStreamPrefix(life) + "_log/not-a-valid-txn-id", "garbage"); + + Gc gc(store, kGc); + /// The fold's `groupRefKeys` rejects the unrecognized key and ABORTS ref folding for the round (spec + /// §Step 2: a malformed key cannot produce a partial ref delta or authorize destructive work). The + /// round CATCHES this internally and survives -- it must not propagate, and must not fold anything. + ASSERT_NO_THROW(gc.runRegularRound()); + + /// No partial delta, no cursor advance: the valid log's blob was NOT folded. + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 0) + << "a malformed ref key must abort the round before any partial ref delta lands"; + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), 0u) + << "the durable cursor must not advance on an aborted round"; +} + +/// (8b) A non-canonical physical life segment is the OTHER way a ref key can be malformed, and it must +/// land on exactly the path (8) pins -- abort ref folding, record the anomaly, COMPLETE the round. +/// +/// It gets its own test because the failure mode is worse than a lost round. The parser REFUSES this +/// shape by name rather than returning `std::nullopt`, so it is the one malformed key that can throw +/// from the round's global `cas/ns/stream/` enumeration, which runs in `defer_decision` -- before the fold, +/// and outside the fold's catch. Escaping there does not merely fail one round: GC is the only thing +/// that could ever delete the key, so a round that dies on it dies on it again every time, forever. +/// The enumeration must therefore absorb the refusal per key and leave the key unindexed in +/// `scan.keys`, exactly as it already does for every other malformed shape, and let `groupRefKeys` +/// raise it once where the round is ready to catch it. +TEST(CASRefGc, NonCanonicalLifeKeyAbortsRefFoldingWithoutWedgingTheRound) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + + const ManifestRef r = mref(1); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + + /// A ref log whose supposed life segment contains logical namespace text rather than one canonical + /// opaque id. Only a foreign or corrupt writer can put this key here, and the pool must survive it. + const String noncanonical_life = + layout.casRefsPrefix() + ns.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; + ASSERT_EQ(backend->putIfAbsent(noncanonical_life, "garbage").outcome, PutOutcome::Done); + + Gc gc(store, kGc); + RoundReport rep; + ASSERT_NO_THROW(rep = gc.runRegularRound()) + << "the round must COMPLETE: a key GC alone could remove must never abort the round that would"; + EXPECT_TRUE(rep.hasAnomaly(RootNamespace{}, /*shard*/ 0)) + << "the refusal must surface as the fold's abort anomaly, not vanish"; + EXPECT_EQ(rep.deleted, 0u); + EXPECT_EQ(rep.redeleted, 0u); + + /// Same fail-close as (8): no partial delta, no cursor advance. + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 0) + << "an aborted ref fold must land no partial ref delta"; + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), 0u) + << "the durable cursor must not advance on an aborted round"; + + /// The wedge is only visible over time: the key is still there (nothing deletes it), so a second + /// round meets it again. It must survive that one too. + ASSERT_TRUE(backend->head(noncanonical_life).exists) << "precondition: nothing removed the key"; + ASSERT_NO_THROW(gc.runRegularRound()) << "a round that dies on this key would die on it forever"; +} + +/// Coverage gap (Task 13a): a ref log at a CANONICAL key but with an undecodable BODY -- distinct from a +/// malformed *key* (which aborts earlier at the group step, above). This exercises the +/// GET-then-decode-throw path. +/// +/// Its blast radius is the NAMESPACE, not the round (spec §5: the whole-round abort survives only for a +/// key that cannot be attributed to any namespace). The body sits at the position the arithmetic walk +/// reads next, so the walk stops there: everything below it stays folded (a transaction applies +/// atomically -- there is no partial delta either way), the cursor never moves past it, and the recorded +/// anomaly suppresses every destructive step of the round, so nothing the unfolded tail might still +/// reference can be reclaimed. +TEST(CASRefGc, InvalidRefLogBodyHoldsNamespaceNoPartialDelta) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + const ManifestRef r = mref(1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns, "tbl", std::nullopt, r); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1) << "published and folded"; + + /// Now DROP the ref, so the blob is genuinely unreferenced once that record folds, and only then + /// plant the invalid body at the walk's very next position. This ordering is what makes the + /// suppression assertion below mean something: asserting that a LIVE blob survives a held round + /// proves nothing, since a live blob is never reclaimable in the first place. + const uint64_t dropped = dropRefTransition(*backend, layout, ns, "tbl", r); + + /// A canonical `_log` key (groupRefKeys accepts it) whose body cannot be decoded: the fold GETs it + /// and `decodeRefLogTxn` throws. + const String garbage_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, dropped + 1}); + backend->putIfAbsent(garbage_key, "garbage-not-a-valid-reflog-body"); + /// The corruption claims the next committed position. Advance only the durable frontier, not the + /// log body, so recovery must exact-GET and hold this malformed object instead of ignoring F+1. + advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, dropped + 1}); + + /// Eight rounds under the hold. Each one catches the hold internally and survives. + for (int i = 0; i < 8; ++i) + { + ASSERT_NO_THROW(runRegularRoundReclaiming(gc)); + store->renewWatermarkOnce(); + } + + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), dropped) + << "the durable cursor must stop BELOW the invalid record, and never advance past it"; + EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 0) + << "the complete transaction below the invalid body folded -- the drop applied, so the blob is " + "unreferenced and would be reclaimed by any unsuppressed round"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(1))) + << "the held namespace's anomaly suppresses graduation and pending deletes: an unreferenced " + "blob is NOT reclaimed while any namespace is held, because the unfolded tail behind the " + "hold may still name it"; + + /// DELETING THE EVIDENCE DOES NOT RELEASE THE HOLD. The hold is durable and clears by exactly one + /// event -- the fold resolving its offending position -- so an object that stops answering does not + /// turn the gap into a frontier. It is the same observation a lying store produces, and it is + /// precisely what made the hold necessary; if an absent could clear it, the whole mechanism would + /// be defeated by the corruption it exists to survive. (Before durable holds this delete DID + /// release the namespace, which is the hole Task 8 closed.) + const HeadResult h = backend->head(garbage_key); + ASSERT_TRUE(h.exists); + ASSERT_EQ(backend->deleteExact(garbage_key, h.token).kind, DeleteOutcome::Kind::Deleted); + + for (int i = 0; i < 4; ++i) + { + ASSERT_NO_THROW(runRegularRoundReclaiming(gc)); + store->renewWatermarkOnce(); + } + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(1))) + << "the hold still stands: nothing resolved the offending position, an absent proved nothing"; + + /// REPAIR is the release: a DECODABLE record at the offending position. The fold reads it, folds + /// through it, seals a cursor above it -- and only then does the namespace stop being held and + /// destruction resumes. The CTE already claims this position, so this must replace the repaired + /// body at its exact id rather than use the semantic wrapper, which would attempt a non-monotone + /// checkpoint advance. + const ManifestRef r2 = mref(2); + writeBlobBody(*backend, layout, DB::UInt128(2)); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + writeTxnAt(*backend, layout, ns, RefTxnId{1, dropped + 1}, publishCommittedOps("tbl2", r2)); + + ASSERT_TRUE(runToFixpoint(store, gc) < 64u) << "the released namespace must converge"; + EXPECT_EQ(foldCursorOf(*backend, layout, ns, 0), dropped + 1) << "the walk folded through the hold"; + EXPECT_FALSE(blobPresent(*backend, layout, DB::UInt128(1))) + << "once the hold clears, the unreferenced blob is reclaimed -- so the survival above was the " + "suppression doing its job, not the blob being unreclaimable"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) << "the repair's own blob is referenced"; +} + +/// Coverage gap (Task 13a): the per-table baseline guard (spec §Offline Recovery) has no positive-trip +/// test at HEAD -- the adapted successor of the retired CASGCBaselineGuard.FreshStateOverTrimmedJournals +/// contract. A table whose logs at/below its newest snapshot are gone and that has no sealed fold cursor +/// is the "a prior fold advanced+cleaned covered logs, then gc/state was lost" signature: folding it from +/// {0,0} would emit no edges and mass-condemn its still-referenced blob. GC must refuse the round before +/// any delete. The existing CASGCBaselineGuard tests cover only the genuinely-fresh pass case and the +/// adopted-seal-missing guard, not this branch. +TEST(CASRefGc, BaselineGuardRefusesWhenSnapshotSurvivesWithoutLogsOrCursor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + + /// Table A is healthy (a committed ref with its manifest+blob, no snapshot), giving GC a normal table + /// to fold in the same round. + const RootNamespace ns_a{"00/aa@cas@"}; + const ManifestRef ra = mref(1); + writeBlobBody(*backend, layout, DB::UInt128(1)); + writeManifestRaw(*backend, layout, ns_a, ra, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, layout, ns_a, "ta", std::nullopt, ra); + + /// Table B is poisoned: a durable snapshot survives, but its logs at/below it are GONE and B has no + /// sealed cursor (first round -> no adopted parent cursors). This is the exact baseline-guard input. + const RootNamespace ns_b{"00/bb@cas@"}; + /// Stage B (Task 4-C): `writeRefSnapshotRaw` deliberately does NOT self-admit (several fixtures + /// build a table with no catalog entry on purpose), so without this `ns_b` would never enter the + /// catalog at all and would be invisible to the round -- the baseline guard below could then never + /// fire, since it never runs on a namespace outside the universe. + fixture::admitLive(*backend, layout, ns_b); + const ManifestRef rb = mref(2); + writeBlobBody(*backend, layout, DB::UInt128(2)); + writeManifestRaw(*backend, layout, ns_b, rb, {blobEntryFor("b", DB::UInt128(2))}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns_b.string(), RefTxnId{1, 5}, + {committedRow("tb", rb)})); + + /// The baseline guard must fail closed BEFORE any destructive step (first round: no prior fold seal, + /// so the failure can only come from the baseline guard, not the seal-divergence guard). + Gc gc(store, kGc); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.runRegularRound(); }); + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(1))) << "table A's blob survives the refusal"; + EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) + << "table B's blob must NOT be condemned -- the guard fires before any delete"; +} + +/// A catalog-admitted life without a parent cursor is a valid fresh fold target when it has no +/// snapshot or logs. The fold must seed its successor seal from every plan row, not only the +/// parent-cursor subset used by the baseline guard. +TEST(CASRefGc, CatalogAdmittedFreshLifeWithoutParentSeedsSuccessorSeal) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, layout, ns); + + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + ASSERT_EQ(catalog_cut.catalog.entries.size(), 1u); + const UInt128 life_id = catalog_cut.catalog.entries.front().incarnation; + + Gc gc(store, kGc); + ASSERT_NO_THROW(gc.runRegularRound()); + + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + EXPECT_TRUE(seal.ref_lives.contains(life_id)); +} diff --git a/src/Disks/tests/gtest_cas_ref_install_safety.cpp b/src/Disks/tests/gtest_cas_ref_install_safety.cpp new file mode 100644 index 000000000000..3c3c58b480d8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_install_safety.cpp @@ -0,0 +1,959 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +/// Task 3 (spec §A1, site 1): the region of `CasRefLedger::commitRefChunk` between "this chunk's +/// ref-log object is durable" and "the runtime records it". +/// +/// Before the fix that region ran `applyRefLogTxn(rt->state, chunk_txn)`, which allocates (the COW +/// containers build an overlay) and can therefore throw `MEMORY_LIMIT_EXCEEDED`. A throw there left the +/// transaction durable but invisible to the writer -- and because a later transaction only needs +/// `greatest_applied < its own id` (contiguity is never checked), a snapshot published afterwards is +/// labelled with that LATER id, so recovery skips the stranded transaction permanently while GC, which +/// folds the ref logs themselves, still applies it. That divergence loses data (a stranded removal +/// leaves the writer holding a ref whose blobs GC deleted), which is why the install is now a +/// prepared-candidate swap: allocation-free, hence non-throwing, and enforced as such by +/// `DENY_ALLOCATIONS_IN_SCOPE`. +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int NETWORK_ERROR; +extern const int CORRUPTED_DATA; +extern const int LOGICAL_ERROR; +extern const int MEMORY_LIMIT_EXCEEDED; +} + +namespace ProfileEvents +{ +extern const Event CASRefNeedsRecovery; +} + +using namespace DB::Cas; + +namespace +{ + +PoolPtr openPool(const BackendPtr & backend) +{ + /// A fresh pool with no residue, mirroring `gtest_cas_ref_chunked_flush.cpp`'s `openPool`. + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// As `openPool`, but with a SINGLE-attempt request budget, which is what makes one ambiguous `PUT` +/// conclusive: with retries allowed the controller's resolve-before-reissue would either re-`PUT` (the +/// object never landed) or prove the object durable (it did) and report `Committed`, and neither of the +/// wedge arms under test would ever be reached. Same budget shape as +/// `gtest_cas_ref_chunked_flush.cpp`'s `runChunkFailureCase`, including the short timeouts so there is +/// no inter-attempt sleep to serve. +PoolPtr openPoolSingleAttempt(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; + CasRequestBudget budget; + /// ONE attempt is the whole mechanism these tests need: it is what turns an injected lost + /// acknowledgement into `Unresolved` instead of a transparent retry, and it does so independently of + /// how fast the machine is. + /// + /// The operation deadline must therefore NOT sit at `attempt_timeout_ms`, which is where it used to. + /// The controller's pre-send gate (`putIfAbsentControlled`: `now + attempt_timeout > deadline` + /// returns `Unresolved` WITHOUT sending) is then a zero-width race that passes only if no + /// millisecond tick elapses between the deadline capture and the gate. Under parallel-build load it + /// loses: the gate fires first, nothing is sent, the injected fault is never reached, and the flush + /// fails CLEAN -- so the product correctly does NOT wedge the lane and the wedge expectations flip. + /// `UncertainPrecommitKeepsItsCleanupOwnerAndItsBody` was observed failing exactly that way (Task 9, + /// `refLaneWedgedForTest` false at the wedge assertion), and every test on this fixture carries the + /// same razor. Same root cause and same fix as `8f9e63c7a19` for the sweep-interruption test. + /// + /// A WIDE deadline keeps the request always actually sent, so the injected fault decides the outcome + /// rather than the scheduler. Tests that want the pre-send REFUSAL instead use + /// `openPoolFenceControlled`, where a frozen clock makes that refusal deterministic rather than raced. + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + return Pool::open(backend, cfg); +} + +/// The mount-fence deadlines the pre-attempt tests drive, in the FROZEN boot clock of +/// `openPoolFenceControlled` (which is pinned at 0, so these are also the remaining lease budgets). +/// +/// `CasMountRuntime` has TWO fence predicates and they are deliberately not the same: +/// `mayMutate` -- `now < deadline`; the top-of-flush gate in `flushRefBatch`. +/// `refAppendFenceOk` -- additionally `attempt_timeout_ms + lease_safety_margin_ms < deadline - now`, +/// i.e. "there is room for one whole controlled attempt"; the `fence_ok` +/// `commitRefChunk` hands to `putIfAbsentControlled`. +/// With `openPoolFenceControlled`'s budget below that margin is 100 + 100 = 200 ms, so a 100 ms +/// remaining lease sits BETWEEN them: the flush is admitted and then its very first pre-attempt gate +/// refuses. That is +/// exactly the production shape this task is about (a lease too short to start a write, not a lost +/// one), and it needs no fault injection at all -- which is the point: nothing is sent. +constexpr uint64_t FENCE_DEADLINE_HEALTHY_MS = 30000; +constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 100; + +/// A legal blob-free part: stage an empty manifest, precommit, promote -- enough to drive real +/// ref-log transactions through the append lane. +void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +/// As `openPoolSingleAttempt`, but with the mount fence under the TEST's control instead of the wall +/// clock's: +/// - the boot clock is FROZEN at 0, so `setMountDeadline` alone decides both fence predicates and no +/// elapsed real time can flip one of them mid-test (the same load-bearing injection, for the same +/// reason, as `gtest_cas_ref_chunked_flush.cpp`'s `openPool`); +/// - lease renewal is parked an hour out, so the runtime-owned renewal worker cannot re-arm the deadline +/// underneath a test that just shortened it. Ten seconds (the default) would be enough in practice +/// and flaky in principle; this removes the race rather than betting on it. +PoolPtr openPoolFenceControlled(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; + cfg.boot_ms_fn = [] { return uint64_t{0}; }; + cfg.mount_renew_period = std::chrono::milliseconds{3600000}; + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + return Pool::open(backend, cfg); +} + +/// Runs `f`, requires it to throw the ref lane's retry-later condition, and returns the message so a +/// caller can assert WHICH condition it was. The message is the only place the `CasUnresolvedReason` +/// surfaces -- there is no accessor for it, by design (it is a diagnostic, not state) -- so this is how +/// a test proves the reason actually reached the decision site instead of defaulting. +String retryLaterMessageOf(const std::function & f) +{ + try + { + f(); + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR) << e.message(); + return e.message(); + } + ADD_FAILURE() << "expected the CAS retry-later condition, but nothing was thrown"; + return {}; +} + +/// Installs a ONE-SHOT throwing probe into the post-durable install regions (spec §A2): the next region +/// entered throws, every later one runs normally -- which is what lets a terminality test drive a +/// successful flush after the recovery transition. +/// +/// The exception is built HERE, outside the region, and the probe only rethrows it: constructing a +/// `DB::Exception` inside the region would allocate and trip `DENY_ALLOCATIONS_IN_SCOPE`, so the test +/// would be exercising the guard instead of the recovery transition. `MEMORY_LIMIT_EXCEEDED` (what a +/// real tracked allocation failure raises) is used deliberately instead of `LOGICAL_ERROR`, which +/// aborts at construction in debug/sanitizer builds. +/// +/// With §A1 landed this seam is the only way to reach `NeedsRecovery` from an install region: +/// install regions are allocation-free and therefore cannot throw on their own. +void armOneShotInstallFailure(const PoolPtr & store) +{ + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + /// The throw itself allocates its exception object through `malloc`, which the memory tracker + /// does not see, so it would not trip the guard anyway -- re-allowing allocations for the + /// duration of the throw makes that a stated property of the test rather than a bet on a libc++ + /// implementation detail. + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); +} + +} + +/// The post-durable install seam exists, is reached by an ordinary commit, and every transaction that +/// reaches it is RECORDED: the tail counter advances exactly once per install, and the ref resolves. +/// The equality is the point -- it is the invariant the old code could break, since there the install +/// was an allocating apply that could throw between the durable `PUT` and the counter bump. +TEST(CASRefInstallSafety, PostDurableInstallIsAllocationFree) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/install_safety_seam"}; + + /// Fired on the calling thread by the flush leader, which is this thread; `atomic` regardless, so + /// the assertions below cannot be read as depending on that. + std::atomic installs{0}; + store->setCarveHookForTest([&installs](CasRefLedger::CarvePhaseForTest phase) + { + if (phase == CasRefLedger::CarvePhaseForTest::PostDurableInstall) + installs.fetch_add(1); + }); + + publishEmptyPart(store, ns, "part_a"); + store->setCarveHookForTest(nullptr); + + const size_t seen = installs.load(); + EXPECT_GT(seen, 0u) << "the post-durable install seam must be reached by an ordinary commit"; + /// No snapshot publish can interfere: the thresholds are 256 logs / 1 MiB and this part is a + /// handful of tiny transactions, so the tail counter still holds every one of them. + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), seen) + << "every durable transaction that entered the install region must be recorded in the tail"; + EXPECT_TRUE(store->resolveRef(ns, "part_a", /*allow_stale=*/false).has_value()); +} + +/// Task 4 (spec §A1, site 3). An `Unresolved` PUT must ALWAYS leave the lane wedged with the exact +/// {id, key, bytes} of the in-doubt object -- the wedge is the only record that can ever resolve it, and +/// on this path the object may already be durable. Building the wedge AFTER the PUT copies two `String`s +/// and could therefore fail on allocation, recording NEITHER the transaction nor the wedge: strictly +/// worse than a wedge, because the next append then mints a fresh id and proceeds against a state that +/// is missing a landed transaction. It is now preconstructed before the PUT and installed by a +/// non-throwing move. +/// +/// `Mode::Unresolved` deliberately lands NOTHING, so this test also pins the other half of the wedge +/// contract: an ambiguous outcome wedges even when the object turns out never to have existed. The tail +/// counter must NOT advance -- an unproven transaction is not a recorded one. +TEST(CASRefInstallSafety, UnresolvedAlwaysRecordsTheWedge) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/unresolved_wedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + /// Scoped to THIS namespace's ref log, so nothing else the part publish writes (the manifest, the + /// pool's own metadata) can consume the single fault. + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + + const String message = retryLaterMessageOf([&] { publishEmptyPart(store, ns, "part_a"); }); + + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "an Unresolved PUT must always leave a wedge"; + const String wedged_key = store->wedgedKeyForTest(ns); + EXPECT_FALSE(wedged_key.empty()) << "the wedge must retain the in-doubt object's key"; + EXPECT_TRUE(wedged_key.starts_with(store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/")) + << "the wedged key must be this namespace's ref-log object, not some other key: " << wedged_key; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) + << "an UNPROVEN transaction must not be recorded as applied"; + + /// Task 18. The contrast half of `PreAttemptRefusalDoesNotWedgeTheLane` below, asserted on the ONE + /// artifact that carries the distinction: an attempt WAS sent here (the fault is thrown by the + /// backend's `putIfAbsent`, so the request reached it), the single-attempt budget is then spent, and + /// the lane wedges. The message must say so -- and must NOT say "no attempt was sent", which is the + /// only shape allowed to skip the wedge. + EXPECT_NE(message.find("attempt budget was exhausted"), String::npos) + << "the reason must reach the wedge message rather than defaulting: " << message; + EXPECT_EQ(message.find("no attempt was sent"), String::npos) + << "an ambiguous PUT is not a pre-attempt refusal: " << message; +} + +/// Task 18 (finding #37 defect 3, behavioural half). A `NoAttemptSent` `Unresolved` must NOT wedge. +/// +/// The wedge exists because an ambiguous PUT MAY HAVE LANDED, so the durable log may or may not contain +/// the transaction and only an exact-key GET can settle it. That reasoning needs an attempt to have been +/// SENT. Here both pre-attempt gates reject on the FIRST iteration -- the remaining lease has no room +/// for one controlled attempt -- so nothing reaches the backend, the key is provably unwritten, and a +/// wedge would protect against nothing while costing the table every ref append (inserts included) +/// until a remount: an exact-key GET of a key that was never written reports `Unresolved` forever, so +/// such a wedge can never clear itself. +/// +/// No fault injection anywhere in this test, deliberately: the ZERO ref-log I/O assertion below is the +/// direct proof that nothing was sent, and it would be meaningless if a fault backend were swallowing +/// the request. +TEST(CASRefInstallSafety, PreAttemptRefusalDoesNotWedgeTheLane) +{ + auto backend = std::make_shared(); + auto store = openPoolFenceControlled(backend); + const RootNamespace ns{"srv1/pre_attempt_refusal"}; + + publishEmptyPart(store, ns, "part_a"); + const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); + const String log_prefix = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + const uint64_t log_io_after_seed = backend->ioCountForKeysContaining(log_prefix); + + /// Shorten the lease to the window where the flush is admitted but no attempt may start. + store->setMountDeadline(FENCE_DEADLINE_REFUSES_ATTEMPT_MS); + /// Half of "the PRE-ATTEMPT gate is what refuses" is asserted here (the flush is admitted, so this + /// is not the top-of-flush `mayMutate` gate); the other half is asserted below, by the message + /// naming `NoAttemptSent` and by the ref-log I/O count not moving. `refAppendFenceOk` itself is + /// private to `Pool`, and is not worth widening for a test that can prove the same thing from the + /// outside. + ASSERT_TRUE(store->mayMutate()) << "the flush must still be ADMITTED, or this exercises the " + "top-of-flush gate instead of the pre-attempt one"; + + const String message = retryLaterMessageOf([&] { store->dropRef(ns, "part_a"); }); + + EXPECT_NE(message.find("no attempt was sent"), String::npos) + << "the caller must be told WHY, and this is the reason the no-wedge decision rests on: " << message; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) + << "nothing was sent, so nothing can be durable: there is no ambiguity for a wedge to resolve"; + EXPECT_TRUE(store->wedgedKeyForTest(ns).empty()); + EXPECT_EQ(backend->ioCountForKeysContaining(log_prefix), log_io_after_seed) + << "the refusal must be PRE-attempt: not one ref-log object may have been touched"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "no apply is owed for a transaction that was never sent -- leaving the marker pending would " + "claim this table may be missing a durable transaction for the rest of its life"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed) + << "nothing was committed, so nothing may be recorded"; + EXPECT_TRUE(store->resolveRef(ns, "part_a", /*allow_stale=*/false).has_value()) + << "the refused drop must not have taken effect"; + + /// The availability half of the claim: the lane is usable the moment the lease is healthy again -- + /// no remount, no wedge resolution, nothing to clear. Before this task the same sequence left a + /// wedge over a key that was never written, and this append would have failed forever. + store->setMountDeadline(FENCE_DEADLINE_HEALTHY_MS); + store->dropRef(ns, "part_a"); + EXPECT_FALSE(store->resolveRef(ns, "part_a", /*allow_stale=*/false).has_value()) + << "the retry on the same lane must commit"; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); +} + +/// Task 18, the same pair on the WEDGE-RESOLUTION path -- the negative half. +/// +/// One flush does both things: it resolves an outstanding wedge over a genuinely durable object (the +/// resolving GET proves it, so the transaction is installed and the lane unwedged), and then commits +/// its own new chunk, which the pre-attempt gate refuses. The lane must come out CLEAN. +/// +/// This is the worst pre-fix shape and the reason the case is worth its own test: the flush had just +/// converted a resolvable wedge into a recorded transaction, and the old code immediately re-wedged the +/// lane over an id whose object was never written -- turning a wedge that WOULD have cleared into one +/// that never can. +TEST(CASRefInstallSafety, PreAttemptRefusalAfterAWedgeResolutionLeavesTheLaneClean) +{ + auto backend = std::make_shared(); + auto store = openPoolFenceControlled(backend); + const RootNamespace ns{"srv1/pre_attempt_after_unwedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); + + /// Wedge over an object that IS durable: the write lands, its acknowledgement is lost, and the + /// controller's own verifying read is lost too (the only mode that reaches the resolution install). + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed); + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + /// The wedge resolution is itself a conditional CREATE under the every-attempt rule, so it is + /// fence-gated like any other write: shortening the lease BEFORE the flush would refuse the + /// resolution too, and there would be no "after a wedge resolution" left to test. Shorten it + /// BETWEEN the two instead -- the pre-carve hook fires exactly there, after the wedge block and + /// before the batch is carved. + store->setRefPreCarveHookForTest([&] { store->setMountDeadline(FENCE_DEADLINE_REFUSES_ATTEMPT_MS); }); + const String message = retryLaterMessageOf([&] { store->dropRef(ns, "y"); }); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_NE(message.find("no attempt was sent"), String::npos) << message; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) + << "the wedge that existed was RESOLVED, and the chunk that followed it was never sent -- the " + "lane must be left clean, not re-wedged over an id that can never resolve"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed + 1) + << "the resolved wedge must still have been installed exactly once"; + EXPECT_FALSE(store->resolveRef(ns, "x", /*allow_stale=*/false).has_value()) + << "the wedged drop was proven durable, so its removal must be visible"; + EXPECT_TRUE(store->resolveRef(ns, "y", /*allow_stale=*/false).has_value()) + << "the refused chunk must not have taken effect"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + store->setMountDeadline(FENCE_DEADLINE_HEALTHY_MS); + store->dropRef(ns, "y"); + EXPECT_FALSE(store->resolveRef(ns, "y", /*allow_stale=*/false).has_value()); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); +} + +/// Task 18, the same pair on the wedge-resolution path -- the POSITIVE half, so the test above cannot +/// pass by the fix having weakened the wedge generally. Same flush shape (resolve a durable wedge, then +/// commit a new chunk), except the new chunk's PUT is genuinely ambiguous: an attempt WAS sent, so the +/// lane must wedge again, now over the NEW transaction. +TEST(CASRefInstallSafety, AmbiguousChunkAfterAWedgeResolutionRewedgesTheLane) +{ + auto backend = std::make_shared(); + auto store = openPoolFenceControlled(backend); + const RootNamespace ns{"srv1/ambiguous_after_unwedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String first_wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(first_wedged_key.empty()); + + /// The resolution is a conditional CREATE at the wedged key now, and that key already holds our + /// own landed object, so it conflicts and the follow-up read adopts it (`LandedThenLost`'s one-shot + /// lost read was consumed inside the previous attempt, so this read succeeds). `fault_skip` lets + /// that create through and puts the fault on this flush's OWN chunk PUT, which is the subject. + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_skip = 1; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "y"); }); + + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) + << "an attempt was sent for the new chunk, so its object may be durable: the lane must wedge"; + EXPECT_NE(store->wedgedKeyForTest(ns), first_wedged_key) + << "the new wedge must describe the NEW transaction, not the resolved one"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed + 1) + << "only the resolved wedge is recorded; the in-doubt chunk is not"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) + << "a wedged lane may hold a durable transaction the runtime has not recorded"; +} + +/// Task 18's regression guard, asserted on the mapping itself rather than through six pieces of fault +/// choreography. `unresolvedProvesNothingWasSent` is the whole decision: the ledger wedges unless it +/// answers true, so this table IS the protocol. +/// +/// What protects a future contributor who adds a `CasUnresolvedReason` member and forgets this file: +/// the predicate is a switch with NO `default`, so the addition is a `-Wswitch` build error (a forced +/// decision, not a silent one), and its trailing `return false` makes the runtime answer "wedge" even +/// if that diagnostic is ever suppressed. Both directions fail closed; neither can widen the allow-list +/// by omission. The `static_assert`s make the mapping a compile-time fact, and the `EXPECT`s repeat it +/// so a break names the offending value in the test report. +TEST(CASRefInstallSafety, OnlyNoAttemptSentMaySkipTheWedge) +{ + static_assert(unresolvedProvesNothingWasSent(CasUnresolvedReason::NoAttemptSent)); + static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::NotUnresolved)); + static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostMidWay)); + static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::DeadlineMidWay)); + static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostPostWrite)); + static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::AttemptsExhausted)); + + EXPECT_TRUE(unresolvedProvesNothingWasSent(CasUnresolvedReason::NoAttemptSent)) + << "the pre-attempt gates rejected before the first request: the key is provably unwritten"; + /// `NotUnresolved` is reachable at the decision site if any path ever returns `Unresolved` without + /// recording a reason, so it is listed here as a real case, not as enum hygiene. + EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::NotUnresolved)) + << "an unrecorded reason proves nothing and must keep wedging"; + EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostMidWay)) + << "an attempt was already sent: its object may be durable"; + EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::DeadlineMidWay)) + << "an attempt was already sent: its object may be durable"; + EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostPostWrite)) + << "the attempt COMMITTED and only the fence was lost afterwards -- the most durable case of all"; + EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::AttemptsExhausted)) + << "every attempt is a candidate for having landed"; +} + +/// Task 5 (spec §A1, site 2). Resolving a wedge is a post-durable install too: the resolving GET PROVES +/// the object landed, so the transaction MUST be recorded -- and recording it must be inseparable from +/// clearing the wedge. It was not: the apply and the `materializeCommitted` fold sat between them +/// WITHOUT the ordinary commit arm's swallow, so a fold failure left the transaction applied and the +/// wedge still set, and the next resolution re-applied the same transaction and DOUBLE-bumped the tail +/// counters. The candidate is now built before the GET and installed by a `noexcept` swap that clears +/// the wedge in the same allocation-free region, with the fold outside it and swallowing. +/// +/// Drives the real thing end to end (no seam beyond the `LandedThenLost` backend mode): a drop whose +/// object landed but whose acknowledgement -- and whose immediate verification read -- were both lost, +/// then a second append whose flush resolves it. The tail counter is the "exactly once" witness: it is +/// bumped once per install, so a re-applied transaction shows up as one extra. +TEST(CASRefInstallSafety, WedgeResolutionInstallsExactlyOnce) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/wedge_resolution"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + /// No snapshot publish can interfere and reset these: the thresholds are 256 logs / 1 MiB and this + /// whole test is a handful of tiny transactions, so every delta below is exact. + const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); + + /// Drop "x" through a PUT that LANDS and then loses its response, plus the one-shot lost read that + /// keeps the controller's own resolve-before-reissue from settling it inside the same attempt. One + /// attempt, so the lane wedges over an object that is genuinely durable -- the only way in. + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "the lost-response drop must wedge the lane"; + ASSERT_FALSE(store->wedgedKeyForTest(ns).empty()); + ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed) + << "the wedged transaction is durable but not yet PROVEN, so it must not be recorded yet"; + + /// A second append into the same table: its flush resolves the wedge first (+1 install) and then + /// commits its own transaction (+1). Nothing else can add a transaction in between. + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + store->dropRef(ns, "y"); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a wedge proven durable must be cleared"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed + 2) + << "the resolved transaction must be recorded EXACTLY once: +1 for it and +1 for the append that " + "resolved it (a double-apply would show as +3)"; + /// Both drops took effect -- the wedged one via the resolution install, which is what proves that + /// install happened at all rather than the wedge merely being discarded. + EXPECT_FALSE(store->resolveRef(ns, "x", /*allow_stale=*/false).has_value()) + << "the wedged drop was proven durable, so its removal must be visible in the cached state"; + EXPECT_FALSE(store->resolveRef(ns, "y", /*allow_stale=*/false).has_value()); +} + +/// Negative control, part 1 of 2: does `DENY_ALLOCATIONS_IN_SCOPE` actually fire on an allocation in +/// THIS binary? Gated on `MEMORY_TRACKER_DEBUG_CHECKS`, because that is the macro the guard itself is +/// gated on (`MemoryTracker.h`: defined only under `!NDEBUG`, i.e. plain debug builds; everywhere else +/// `DENY_ALLOCATIONS_IN_SCOPE` compiles to `static_assert(true)` and there is nothing to observe). +/// An earlier version dispatched on `DEBUG_OR_SANITIZER_BUILD` instead — but sanitizer builds define +/// NDEBUG, so the guard is a no-op there and the death test "failed to die" on all three sanitizer CI +/// lanes. Note the implication chain: `MEMORY_TRACKER_DEBUG_CHECKS` ⇒ `!NDEBUG` ⇒ +/// `DEBUG_OR_SANITIZER_BUILD`, so whenever the guard exists its `LOGICAL_ERROR` aborts at Exception +/// construction (`Exception.cpp`) — death is the only observable outcome, and a throw-only variant is +/// dead code. +#if defined(MEMORY_TRACKER_DEBUG_CHECKS) +TEST(CASRefInstallSafetyDeathTest, DenyGuardStopsAnAllocation) +{ + EXPECT_DEATH( + { + DENY_ALLOCATIONS_IN_SCOPE; + volatile auto * p = new char[64]; + (void)p; + }, + ""); +} +#endif + +/// Negative control, part 2 of 2: the region the guard protects is actually ENTERED, and the guard is +/// armed at that exact point. A probe that only reads flags proves both without allocating, so unlike +/// part 1 this assertion is immune to how a build type renders a `LOGICAL_ERROR`. Together the two +/// parts give what a single death test was meant to give, and a failure now names WHICH half broke: +/// "the guard does not fire" versus "the install region is never reached / not armed". +TEST(CASRefInstallSafety, InstallRegionProbeIsInvokedAndTheGuardIsArmed) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/install_probe_diag"}; + + bool probe_ran = false; + [[maybe_unused]] bool guard_armed_when_probe_ran = false; + store->setInstallRegionProbeForTest([&] + { + probe_ran = true; +#if defined(MEMORY_TRACKER_DEBUG_CHECKS) + guard_armed_when_probe_ran = memory_tracker_always_throw_logical_error_on_allocation; +#endif + }); + + publishEmptyPart(store, ns, "part_a"); + store->setInstallRegionProbeForTest(nullptr); + + EXPECT_TRUE(probe_ran) << "the install-region probe was never invoked"; +#if defined(MEMORY_TRACKER_DEBUG_CHECKS) + EXPECT_TRUE(guard_armed_when_probe_ran) << "the probe ran but DENY_ALLOCATIONS_IN_SCOPE was not armed"; +#endif +} + +/// =================================================================================== +/// The append lane state machine. +/// =================================================================================== + +/// The exact attempt is visible as `Writing` after durability and before installation. +TEST(CASRefInstallSafety, WritingOwnsTheAttemptUntilInstall) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/apply_state_commit"}; + + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Closed) + << "a resident-only observer must not materialize a runtime for an untouched name"; + + /// Fired on the calling thread by the flush leader, which is this thread; `atomic` regardless, as in + /// `PostDurableInstallIsAllocationFree` above, so no assertion here reads as depending on that. + std::atomic observations{0}; + std::atomic pending_observations{0}; + store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::PostDurableInstall) + return; + observations.fetch_add(1); + if (store->laneStateForTest(ns) == RefLaneState::Writing) + pending_observations.fetch_add(1); + }); + + publishEmptyPart(store, ns, "part_a"); + store->setCarveHookForTest(nullptr); + + EXPECT_GT(observations.load(), 0u) << "the post-durable seam must be reached by an ordinary commit"; + EXPECT_EQ(pending_observations.load(), observations.load()) + << "every durable-but-not-yet-installed transaction remains Writing"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "a completed install owes no apply: the marker must be back to Clean"; +} + +/// Ambiguity transfers the same exact attempt from `Writing` to `Wedged`. +TEST(CASRefInstallSafety, UnresolvedTransfersWritingToWedged) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_wedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "part_a"); }); + + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) + << "a wedged lane may hold a durable transaction the runtime has not recorded"; +} + +/// Durable resolution installs the attempt and returns the lane to `Ready`. +TEST(CASRefInstallSafety, WedgeResolutionReturnsReady) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_unwedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + /// The one mode that wedges over a GENUINELY durable object (see `ChunkFaultBackend`): the write + /// lands, its acknowledgement is lost, and the controller's own verifying read is lost too. + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + store->dropRef(ns, "y"); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "the wedged transaction was proven durable AND installed, so nothing is owed any more"; +} + +/// A foreign occupant is a terminal `Faulted` verdict. +TEST(CASRefInstallSafety, ConclusiveForeignConflictFaultsTheLane) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_conflict"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::ForeignConflict; + backend->fault_count = 1; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { publishEmptyPart(store, ns, "part_a"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a proven conflict is conclusive and must not wedge"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) + << "a foreign occupant is a terminal protocol verdict, not a retryable attempt"; +} + +/// `DefiniteFailure` proves nothing became durable and returns the lane to `Ready`. Needs S3 error +/// classification: that is the only exception family +/// `classifyConditionalWriteResult` will ever call definite (everything else is fail-safe Unresolved). +TEST(CASRefInstallSafety, DefiniteFailureReturnsReady) +{ +#if !USE_AWS_S3 + GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; +#else + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_definite"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Definite; + backend->fault_count = 1; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "part_a"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a definite failure is proven non-durable and must not wedge"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "a definitively rejected PUT is proven non-durable, so no apply is owed"; +#endif +} + +/// A foreign occupant is the conclusive negative wedge resolution: +/// the wedged (write-once) key prove our body never landed there. `resolveByExactGet` never reports a +/// plain "absent" verdict -- absent or unreadable is `Unresolved`, since another attempt may still be +/// legal -- so this arm is the whole of "a resolution that proves the key is not ours". +/// +/// Runs in every build: the arm reports `CORRUPTED_DATA` (storage-controlled input must never be able +/// to abort the server), where it used to raise the process-aborting `LOGICAL_ERROR` and this test had +/// to be release-only with a death-test twin standing in. The marker is cleared BEFORE the anomaly +/// reaction, which is what this test pins; the fence/audit half is +/// `CASAnomalyPolicy.ForeignBytesAtWedgeKeyTripFenceAndRemount`'s. +TEST(CASRefInstallSafety, WedgeResolutionProvenForeignFaultsTheLane) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_foreign_wedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + + /// Out of band, a foreign writer lands DIFFERENT bytes at the exact wedged key. The fault mode is + /// off first so this write is not itself intercepted. + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(wedged_key.empty()); + ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) + << "foreign interference is a terminal verdict, not an unresolved attempt"; +} + +/// `Writing -> NeedsRecovery` at the ordinary candidate install. Unreachable +/// in production with §A1 landed -- the region allocates nothing -- so the probe seam simulates the +/// post-durable failure. The transaction is durable at that point, but the runtime has not installed it: +/// exactly the condition that `NeedsRecovery` names. +TEST(CASRefInstallSafety, PostDurableInstallFailureRequiresRecovery) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/apply_state_poison"}; + + publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load(); + armOneShotInstallFailure(store); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "x"); }); + store->setInstallRegionProbeForTest(nullptr); + + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery) + << "an install that failed AFTER its object was durable must be visible, not silent"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load() - needs_recovery_before, 1u) + << "the transition to NeedsRecovery must be exported exactly once"; +} + +/// `Wedged -> NeedsRecovery` at the wedge-resolution install. Same class of failure +/// one region over: the resolving GET already PROVED the object durable, so an install that does not +/// complete there leaves the same missing transaction -- and the wedge survives, because the swap that +/// would have cleared it is in the same region that threw. +TEST(CASRefInstallSafety, WedgeResolutionInstallFailureRequiresRecovery) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/apply_state_poison_unwedge"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + armOneShotInstallFailure(store); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "y"); }); + store->setInstallRegionProbeForTest(nullptr); + + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) + << "known durability transfers ownership from the attempt to recovery"; +} + +/// A later append may proceed only after the top-of-flush recovery has replayed the known-durable +/// transaction and returned the lane to `Ready`. +TEST(CASRefInstallSafety, NeedsRecoveryReplaysBeforeALaterFlush) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/apply_state_poison_terminal"}; + + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load(); + armOneShotInstallFailure(store); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "x"); }); + store->setInstallRegionProbeForTest(nullptr); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + /// A perfectly ordinary, fully successful append afterwards. + store->dropRef(ns, "y"); + EXPECT_FALSE(store->resolveRef(ns, "y", /*allow_stale=*/false).has_value()) + << "the later flush must really have committed AND installed -- otherwise the assertion below " + "would pass for the wrong reason"; + + /// A flush's own success is not evidence that the stranded transaction was installed. What returns + /// the lane to `Ready` is the re-derivation that `ensureRefTableRecovered` performs at the top of + /// that same flush: the walk reads the stranded transaction from the durable log, then installs the + /// recovered state. + /// + /// Both halves are asserted, because only together do they mean the repair happened rather than the + /// state being relabeled: `x` really is gone (the stranded drop is applied at last), and the lane is + /// `Ready`. + EXPECT_FALSE(store->resolveRef(ns, "x", /*allow_stale=*/false).has_value()) + << "the stranded drop of 'x' is durable, so the re-derivation must install it before returning " + "the lane to Ready"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load() - needs_recovery_before, 1u) + << "the event counts transitions, so the successful flush must not have added another"; +} + +/// Part B review, BLOCKER 2: an UNCERTAIN `precommitAdd` must keep its cleanup owner and its body. +/// +/// `PartWriteTxn::precommitAdd` used to record `precommit_*` and set its `precommitted` flag only AFTER +/// `appendRefOps` returned. But an `Unresolved` append MAY HAVE LANDED -- the `Unresolved` arm of +/// `commitRefChunk` says exactly that, and wedges the lane for precisely that reason -- so on that path +/// `abandon` ran against an object that believed it had never precommitted. It therefore queued NO +/// removal, and `cleanupStagedManifestDebrisBestEffort`, deciding from the same unset state, DELETED the +/// manifest body. When the wedge later resolved as committed, the table gained a live precommit with no +/// cleanup owner and no body -- which clamps GC's fold barrier (a live precommit whose body is missing) +/// forever. +/// +/// The fix is the same discipline the wedge itself uses: record the intent BEFORE the ambiguous +/// operation. Both assertions below fail against the old code -- the body is gone, and the precommit is +/// still live once the wedge resolves. +TEST(CASRefInstallSafety, UncertainPrecommitKeepsItsCleanupOwnerAndItsBody) +{ + auto backend = std::make_shared(); + auto store = openPoolSingleAttempt(backend); + const RootNamespace ns{"srv1/uncertain_precommit"}; + /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so + /// the fault injected below (computed from that same sentinel) lands on the key production + /// actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/part_a"; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + const String manifest_key = store->layout().manifestKey(id); + ASSERT_TRUE(backend->head(manifest_key).exists) << "the staged body must exist before the precommit"; + + /// Scoped to THIS namespace's ref log so the manifest body's own PUT cannot consume the fault. The + /// object LANDS and only its acknowledgement is lost, which with the single-attempt budget wedges + /// the lane over a genuinely durable precommit -- the exact shape the old code mishandled. + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->precommitAdd(ns, "part_a", id); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "the lost-response precommit must wedge the lane"; + EXPECT_EQ(build->precommitState(), PartWriteTxn::PrecommitState::Uncertain) + << "an append that may have landed is neither 'never precommitted' nor 'durably precommitted'"; + + /// The cleanup owner survives the uncertainty: this `abandon` resolves the wedge (proving the + /// precommit durable) and appends the exact removal in the same flush. + build->abandon(); + + EXPECT_TRUE(backend->head(manifest_key).exists) + << "abandon writer-deleted the body of a precommit that may be live -- GC's fold barrier would " + "clamp on it forever"; + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()) + << "the uncertain precommit landed, so abandon owed its exact removal"; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "the abandon's own flush must have resolved the wedge"; +} + +/// The other side of the same state, and the reason it is a STATE and not just an extra bool: an +/// `Uncertain` precommit that in fact never landed must not make `abandon` fail forever. +/// +/// `RefTableState::applyOwnerTransition` rejects a removal whose `old_binding` names an absent +/// precommit, so the removal is NOT unconditionally idempotent (the review's "it is idempotent" is only +/// true with the presence check this test pins). Here the append is refused BEFORE any request is sent, +/// which is provably-nothing-durable, and yet the transaction has already recorded the intent -- so the +/// removal it owes must resolve to a no-op rather than to `CORRUPTED_DATA`. +TEST(CASRefInstallSafety, UncertainPrecommitThatNeverLandedStillAbandonsCleanly) +{ + auto backend = std::make_shared(); + auto store = openPoolFenceControlled(backend); + const RootNamespace ns{"srv1/uncertain_precommit_absent"}; + + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/part_a"; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + + /// A lease with room for the flush but not for one whole controlled attempt: the pre-attempt gate + /// refuses, nothing is sent, and no wedge forms (`PreAttemptRefusalDoesNotWedgeTheLane`). + store->setMountDeadline(FENCE_DEADLINE_REFUSES_ATTEMPT_MS); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->precommitAdd(ns, "part_a", id); }); + ASSERT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(build->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + + store->setMountDeadline(FENCE_DEADLINE_HEALTHY_MS); + build->abandon(); /// must not throw: there is no binding to remove, and that is not an anomaly here + EXPECT_EQ(build->precommitState(), PartWriteTxn::PrecommitState::Settled); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} diff --git a/src/Disks/tests/gtest_cas_ref_intake.cpp b/src/Disks/tests/gtest_cas_ref_intake.cpp new file mode 100644 index 000000000000..49ae6bed4137 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_intake.cpp @@ -0,0 +1,260 @@ +#include +#include +#include "cas_test_helpers.h" +#include +#include + +using namespace DB::Cas; + +namespace +{ + +ManifestRef mr(uint64_t epoch, uint64_t seq, uint32_t ordinal = 1) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +RefTxnId rid(uint64_t epoch, uint64_t seq) +{ + return RefTxnId{epoch, seq}; +} + +RefOp addOwner(RefOwnerKind kind, const String & ref, const ManifestRef & manifest) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{kind, ref, manifest}; + return op; +} + +RefOp removeOwner(RefOwnerKind kind, const String & ref, const ManifestRef & manifest) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{kind, ref, manifest}; + return op; +} + +RefOp promote(const String & ref, const ManifestRef & manifest) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, ref, manifest}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, ref, manifest}; + return op; +} + +/// A raw `owner_transition` op from explicit optional bindings, bypassing every shape-builder above -- +/// used by the rejection tests to construct shapes `classifyOwnerTransitionShape` does not recognize. +RefOp rawOwnerTransition(std::optional old_binding, std::optional new_binding) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = std::move(old_binding); + op.new_binding = std::move(new_binding); + return op; +} + +RefLogTxn txn(const String & ns, RefTxnId id, std::vector ops) +{ + RefLogTxn t; + t.ns = ns; + t.txn_id = id; + t.ops = std::move(ops); + return t; +} + +} + +/// spec §gc-step-produce-manifest-edge-delta: each explicit operation states its own edge change. +TEST(CASRefIntake, ManifestEdgesPerOperationShape) +{ + /// Add precommit => one +1. + { + const auto edges = manifestEdgesOfTxn(txn("db/t", rid(1, 1), {addOwner(RefOwnerKind::Precommit, "p", mr(1, 5))})); + ASSERT_EQ(edges.size(), 1u); + EXPECT_EQ(edges[0].change, 1); + EXPECT_EQ(edges[0].manifest_id, (ManifestId{RootNamespace{"db/t"}, mr(1, 5)})); + EXPECT_EQ(edges[0].op_ordinal, 0u); + EXPECT_EQ(edges[0].edge_ordinal, 1u); + } + /// Remove committed => one -1. + { + const auto edges = manifestEdgesOfTxn(txn("db/t", rid(1, 2), {removeOwner(RefOwnerKind::Committed, "p", mr(1, 5))})); + ASSERT_EQ(edges.size(), 1u); + EXPECT_EQ(edges[0].change, -1); + EXPECT_EQ(edges[0].manifest_id, (ManifestId{RootNamespace{"db/t"}, mr(1, 5)})); + } + /// Remove precommit => one -1 (the fourth classified shape, distinct from remove committed only by + /// `old_binding.kind`). + { + const auto edges = manifestEdgesOfTxn(txn("db/t", rid(1, 25), {removeOwner(RefOwnerKind::Precommit, "p", mr(1, 5))})); + ASSERT_EQ(edges.size(), 1u); + EXPECT_EQ(edges[0].change, -1); + EXPECT_EQ(edges[0].owner_kind, RefOwnerKind::Precommit); + EXPECT_EQ(edges[0].manifest_id, (ManifestId{RootNamespace{"db/t"}, mr(1, 5)})); + } + /// Promote same manifest => no net edge (spec §Promote). + { + const auto edges = manifestEdgesOfTxn(txn("db/t", rid(1, 3), {promote("p", mr(1, 5))})); + EXPECT_TRUE(edges.empty()); + } + /// set_published_at / namespace_birth / remove_namespace => no edge. + { + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + set_published_at.ref_name = "p"; + set_published_at.expected_manifest_ref = mr(1, 5); + EXPECT_TRUE(manifestEdgesOfTxn(txn("db/t", rid(1, 4), {set_published_at})).empty()); + + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + EXPECT_TRUE(manifestEdgesOfTxn(txn("db/t", rid(1, 5), {birth})).empty()); + } + /// Replace one manifest by a different one (two explicit ops) => -1 old, +1 new. + { + const auto edges = manifestEdgesOfTxn(txn("db/t", rid(1, 6), + {removeOwner(RefOwnerKind::Committed, "p", mr(1, 5)), addOwner(RefOwnerKind::Precommit, "p", mr(1, 6))})); + ASSERT_EQ(edges.size(), 2u); + EXPECT_EQ(edges[0].change, -1); + EXPECT_EQ(edges[0].manifest_id.ref, mr(1, 5)); + EXPECT_EQ(edges[1].change, 1); + EXPECT_EQ(edges[1].manifest_id.ref, mr(1, 6)); + } +} + +/// `manifestEdgesOfTxn` rejects every `owner_transition` shape outside the four `classifyOwnerTransitionShape` +/// recognizes (Pool/CasRefProtocol.cpp) -- it must never silently assign edge meaning to a shape the +/// writer/replay state machine would refuse to apply. Each case throws `CORRUPTED_DATA`. +TEST(CASRefIntake, ManifestEdgesRejectsUnrecognizedShapes) +{ + /// Neither binding: a degenerate owner_transition that names no owner change at all. + EXPECT_THROW(manifestEdgesOfTxn(txn("db/t", rid(1, 1), {rawOwnerTransition(std::nullopt, std::nullopt)})), + DB::Exception); + + /// old+new naming DIFFERENT manifests in ONE op (the never-legal "replace" shape; an atomic + /// manifest replace is always two ops -- an explicit removal then a same-manifest promote). + EXPECT_THROW(manifestEdgesOfTxn(txn("db/t", rid(1, 2), + {rawOwnerTransition(RefOwnerBinding{RefOwnerKind::Committed, "p", mr(1, 5)}, + RefOwnerBinding{RefOwnerKind::Precommit, "p", mr(1, 6)})})), + DB::Exception); + + /// Promote-shaped kinds (old=Precommit, new=Committed) but with MISMATCHED ref_names. + EXPECT_THROW(manifestEdgesOfTxn(txn("db/t", rid(1, 3), + {rawOwnerTransition(RefOwnerBinding{RefOwnerKind::Precommit, "p", mr(1, 5)}, + RefOwnerBinding{RefOwnerKind::Committed, "q", mr(1, 5)})})), + DB::Exception); + + /// Add with new.kind == Committed (only Precommit is a legal add target). + EXPECT_THROW(manifestEdgesOfTxn(txn("db/t", rid(1, 4), + {rawOwnerTransition(std::nullopt, RefOwnerBinding{RefOwnerKind::Committed, "p", mr(1, 5)})})), + DB::Exception); + + /// old+new both Committed, same manifest: not a promote (promote requires old.kind == Precommit). + EXPECT_THROW(manifestEdgesOfTxn(txn("db/t", rid(1, 5), + {rawOwnerTransition(RefOwnerBinding{RefOwnerKind::Committed, "p", mr(1, 5)}, + RefOwnerBinding{RefOwnerKind::Committed, "p", mr(1, 5)})})), + DB::Exception); +} + +/// Namespaces are edge-distinct even with identical ManifestRef tuples (spec §gc-inputs-and-output). +TEST(CASRefIntake, EdgesAreNamespaceQualified) +{ + const auto a = manifestEdgesOfTxn(txn("db/a", rid(1, 1), {addOwner(RefOwnerKind::Precommit, "p", mr(1, 5))})); + const auto b = manifestEdgesOfTxn(txn("db/b", rid(1, 1), {addOwner(RefOwnerKind::Precommit, "p", mr(1, 5))})); + ASSERT_EQ(a.size(), 1u); + ASSERT_EQ(b.size(), 1u); + EXPECT_NE(a[0].manifest_id, b[0].manifest_id); +} + +TEST(CASRefIntake, RemovalTxnIdDetection) +{ + RefOp remove_ns; + remove_ns.kind = RefOpKind::RemoveNamespace; + const auto with_removal = txn("db/t", rid(3, 8), {removeOwner(RefOwnerKind::Committed, "p", mr(1, 5)), remove_ns}); + ASSERT_TRUE(removalTxnId(with_removal).has_value()); + EXPECT_EQ(*removalTxnId(with_removal), rid(3, 8)); + + const auto ordinary = txn("db/t", rid(3, 9), {addOwner(RefOwnerKind::Precommit, "p", mr(1, 5))}); + EXPECT_FALSE(removalTxnId(ordinary).has_value()); +} + +/// spec §Step 1: one global LIST groups by table, split by kind, sorted; the reconstructed namespace is +/// re-validated (VERIFY-AT-T12) and a malformed ref key aborts ref folding (throws). +TEST(CASRefIntake, GroupRefKeys) +{ + const Layout layout{"p"}; + const RootNamespace ns{"db/t"}; + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + + std::vector keys{ + layout.refSnapshotKey(life, rid(1, 4)), + layout.refLogKey(life, rid(1, 5)), + layout.refLogKey(life, rid(1, 3)), + layout.refCkptKey(life), /// state-family keys are outside the hot stream LIST + "p/cas/manifests/db/t/foo", /// outside the ref prefix -> ignored + }; + const auto grouped = groupRefKeys(layout, keys); + ASSERT_EQ(grouped.size(), 1u); + const RefTableListing & t = grouped.at(life.incarnation); + EXPECT_EQ(t.logs, (std::vector{rid(1, 3), rid(1, 5)})); + EXPECT_EQ(t.snapshots, (std::vector{rid(1, 4)})); + /// Checkpoints live under `cas/ns/state/` and are deliberately absent from the hot stream listing. + + /// A key under the ref prefix that is not a valid ref object aborts (a leftover old-format shard key). + EXPECT_THROW(groupRefKeys(layout, {"p/cas/ns/stream/0"}), DB::Exception); + /// A malformed physical id under a valid stream prefix aborts. + EXPECT_THROW(groupRefKeys(layout, {"p/cas/ns/stream/not-an-id/_log/" + renderRefTxnId(rid(1, 1))}), DB::Exception); +} + +/// A LIST can observe a snapshot after its PUT but before the `_ckpt` CAS makes it a recovery base. +/// That physical object proves nothing by itself: without a checkpoint-named triple, cleanup leaks +/// rather than deleting either the genesis log or the unacknowledged snapshot. +TEST(CASRefIntake, PlanRefCleanupRequiresCheckpointNamedBase) +{ + RefTableListing listing; + listing.logs = {rid(1, 1), rid(1, 2), rid(1, 3)}; + listing.snapshots = {rid(1, 2)}; /// newest observed snapshot X = (1,2) + + /// Even a complete-looking listing and cursor do not license cleanup without the checkpoint's + /// exact base. This is the snapshot-PUT-before-checkpoint-CAS sabotage. + { + const auto plan = planRefCleanup(listing, rid(1, 3), {}); + EXPECT_TRUE(plan.deletable_logs.empty()); + EXPECT_TRUE(plan.deletable_snapshots.empty()); + } + /// A smaller cursor cannot turn that incomplete authority into a cleanup range. + { + const auto plan = planRefCleanup(listing, rid(1, 1), {}); + EXPECT_TRUE(plan.deletable_logs.empty()); + } + /// Nor may a newer listed snapshot reclaim an older listed snapshot before `_ckpt` names a base. + { + RefTableListing two_snaps = listing; + two_snaps.snapshots = {rid(1, 1), rid(1, 2)}; + const auto plan = planRefCleanup(two_snaps, rid(1, 3), {}); + EXPECT_TRUE(plan.deletable_snapshots.empty()); + } + /// No snapshot => no coverage boundary => empty plan (condition 2). + { + RefTableListing no_snap; + no_snap.logs = {rid(1, 1)}; + const auto plan = planRefCleanup(no_snap, rid(1, 5), {}); + EXPECT_TRUE(plan.deletable_logs.empty()); + EXPECT_TRUE(plan.deletable_snapshots.empty()); + } +} + +/// The checkpoint recovery anchor is a triple: `_ckpt`, its same-id `_snap`, and the same-id ordinary +/// `_log` that proves the id is not an `EpochSeal`. Cleanup may reclaim older covered logs, but must +/// retain that one witness for recovery and fsck. +TEST(CASRefIntake, PlanRefCleanupRetainsCheckpointBaseLog) +{ + RefTableListing listing; + listing.logs = {rid(1, 1), rid(1, 2), rid(1, 3)}; + listing.snapshots = {rid(1, 2)}; + + const RefCleanupPlan plan = planRefCleanup(listing, rid(1, 3), rid(1, 2)); + EXPECT_EQ(plan.deletable_logs, (std::vector{rid(1, 1)})); + EXPECT_TRUE(plan.deletable_snapshots.empty()); +} diff --git a/src/Disks/tests/gtest_cas_ref_lane_exception_safety.cpp b/src/Disks/tests/gtest_cas_ref_lane_exception_safety.cpp new file mode 100644 index 000000000000..a88838dd7049 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_lane_exception_safety.cpp @@ -0,0 +1,217 @@ +#include + +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +/// Task 1: ref-lane exception-safety. A queue leader that throws BEFORE carving its compatible batch +/// must not leave its own enqueued item stranded in `rt->pending`. If it does, a later leader (a woken +/// follower) carves the stranded item and runs its `build_ops` closure long after the original caller's +/// stack -- which the production `[&]` closures capture by reference -- has unwound: a use-after-free. +/// +/// These tests drive the fault through the SAME pre-carve injection point production leaders pass +/// (`setRefPreCarveHookForTest`, invoked inside `flushRefBatch` immediately before the batch is carved). +/// The suite name is prefixed `RefWriter` so it is covered by the `RefWriter*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; + +namespace +{ + +PoolPtr openPoolForRefLane(const BackendPtr & backend) +{ + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +} + +/// A SOLO faulted caller must not leave its own item behind in the pending queue. Before the fix, the +/// leader's `appendRefOps` catch reset `leader_active` and rethrew but never completed / de-pended the +/// leader's own item, so it was stranded in `rt->pending` with `done == false` forever (nothing left to +/// carve it) -- the deterministic, sanitizer-independent shape of the stranded-item defect. +TEST(CASRefWriterLaneExceptionSafety, SoloLeaderThrowBeforeCarveDrainsOwnItem) +{ + auto backend = std::make_shared(); + auto store = openPoolForRefLane(backend); + const RootNamespace ns{"srv1/reflane_solo"}; + + std::atomic fault_armed{1}; + store->setRefPreCarveHookForTest([&] + { + if (fault_armed.exchange(0) == 1) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected pre-carve fault"); + }); + + bool threw = false; + try + { + store->appendRefOps(ns, MutationScope::ref("ref_solo"), + [](const RefTableState &) -> std::vector { return {}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + } + catch (const DB::Exception &) + { + threw = true; + } + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_TRUE(threw) << "the faulted solo caller must observe the injected error"; + EXPECT_EQ(store->refQueuePendingForTest(ns), 0u) + << "the leader's own item was left stranded in rt->pending after it threw before carving"; +} + +/// Two concurrent callers on one namespace. The first flush's leader throws before carving; a woken +/// follower then leads. Before the fix, the follower carved the faulted leader's STILL-pending item and +/// ran its `build_ops` closure -- the use-after-free window. This asserts, sanitizer-independently, that +/// the follower never invokes the faulted caller's closure, that the queue drains, and that the +/// non-faulted caller still completes. +TEST(CASRefWriterLaneExceptionSafety, FollowerNeverRunsStrandedLeaderClosure) +{ + auto backend = std::make_shared(); + auto store = openPoolForRefLane(backend); + const RootNamespace ns{"srv1/reflane_follower"}; + + std::atomic fault_armed{1}; + /// The leader parks HERE, at the pre-carve point, until the main thread has queued the follower + /// behind it -- and only then throws. Parking (rather than letting the leader race ahead while the + /// main thread polls the queue depth) is what makes the interleaving this test is about -- + /// "the leader throws WHILE a follower is waiting for the baton" -- deterministic. The previous + /// formulation polled `refQueuePendingForTest(ns) >= 1` from the main thread AFTER starting t1, + /// which loses a race the scheduler decides: t1 could enqueue, take the baton, throw, and have its + /// item erased by `completeOwnedItemsAndReleaseLeadership` before the main thread was ever + /// scheduled to sample -- after which `pending` is 0 forever and the poll spins until the harness + /// is killed. Invisible on an idle 32-core box (10/10 green) and a guaranteed hang under + /// contention (8/8 when pinned to one CPU with `taskset -c 3`). + std::mutex hook_mutex; + std::condition_variable hook_cv; + bool leader_parked_at_precarve = false; /// guarded by hook_mutex + bool release_leader = false; /// guarded by hook_mutex + store->setRefPreCarveHookForTest([&] + { + /// Park+throw only on the FIRST leader flush, so the follower (or a re-drive) can proceed. + if (fault_armed.exchange(0) != 1) + return; + { + std::lock_guard announce(hook_mutex); + leader_parked_at_precarve = true; + } + hook_cv.notify_all(); + { + std::unique_lock wait_for_follower(hook_mutex); + hook_cv.wait(wait_for_follower, [&] { return release_leader; }); + } + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected pre-carve fault"); + }); + + /// Set by the faulted caller's own closure iff a DIFFERENT thread (a follower leader) ever runs it -- + /// i.e. the stranded item was carved by someone other than its owner. This is the direct, portable + /// signature of the use-after-free the fix prevents. + std::atomic faulted_owner{}; + std::atomic faulted_closure_ran_on_follower{false}; + + std::atomic ok{0}; + auto caller = [&](int seq, bool is_faulted) + { + try + { + store->appendRefOps(ns, MutationScope::ref("ref_" + std::to_string(seq)), + [&, is_faulted](const RefTableState &) -> std::vector + { + if (is_faulted && std::this_thread::get_id() != faulted_owner.load()) + faulted_closure_ran_on_follower.store(true); + return {}; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + ok.fetch_add(1); + } + catch (const DB::Exception &) // NOLINT(bugprone-empty-catch) + { + /// The faulted caller may see the injected error; that is expected. + } + }; + + /// Serialize the two callers so the fault deterministically lands on the FIRST one to lead: t1 + /// enqueues, takes the baton, and PARKS at the pre-carve hook; t2 then queues behind it as a + /// follower; only then is t1 released to throw. + std::thread t1([&] + { + faulted_owner.store(std::this_thread::get_id()); + caller(1, /*is_faulted=*/true); + }); + { + std::unique_lock wait_for_leader(hook_mutex); + hook_cv.wait(wait_for_leader, [&] { return leader_parked_at_precarve; }); + } + std::thread t2([&] { caller(2, /*is_faulted=*/false); }); + /// The parked leader cannot drain anything, so this poll cannot miss its window: t1's own item is + /// already in `pending`, and the count reaches 2 as soon as t2 has enqueued. + while (store->refQueuePendingForTest(ns) < 2) + std::this_thread::yield(); + { + std::lock_guard release(hook_mutex); + release_leader = true; + } + hook_cv.notify_all(); + + t1.join(); + t2.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_FALSE(faulted_closure_ran_on_follower.load()) + << "a follower leader carved and ran the stranded faulted caller's build_ops closure (use-after-free)"; + EXPECT_EQ(store->refQueuePendingForTest(ns), 0u) << "an item was stranded in rt->pending"; + EXPECT_GE(ok.load(), 1) << "the non-faulted caller must complete cleanly"; +} + +/// codex stage-1 review (Important): an allocation exception at the PRE-TENURE point -- the first +/// allocation that builds the leader's responsibility set, BEFORE `leader_active` is published -- must +/// not permanently strand the append-lane baton. Before the fix the throwing allocation fired AFTER +/// `leader_active = true` (and after the queue mutex was released), leaving the baton held with no live +/// leader and the caller's item stuck in `pending`: every later writer on the namespace would wait +/// forever at the leader-election cv, and shutdown draining could only time out. This drives the fault +/// through the dedicated pre-tenure seam and asserts, deterministically (no hang), that the lane is left +/// idle: the item is un-enqueued and the baton is un-taken. +TEST(CASRefWriterLaneExceptionSafety, PreTenureAllocFailureReleasesBaton) +{ + auto backend = std::make_shared(); + auto store = openPoolForRefLane(backend); + const RootNamespace ns{"srv1/reflane_pretenure"}; + + std::atomic fault_armed{1}; + store->setRefPreTenureHookForTest([&] + { + if (fault_armed.exchange(0) == 1) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected pre-tenure fault"); + }); + + bool threw = false; + try + { + store->appendRefOps(ns, MutationScope::ref("ref_pretenure"), + [](const RefTableState &) -> std::vector { return {}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + } + catch (const DB::Exception &) + { + threw = true; + } + store->setRefPreTenureHookForTest(nullptr); + + EXPECT_TRUE(threw) << "the faulted caller must observe the injected error"; + EXPECT_EQ(store->refQueuePendingForTest(ns), 0u) + << "a pre-tenure allocation failure left the caller's item stranded in rt->pending"; + EXPECT_FALSE(store->refLeaderActiveForTest(ns)) + << "a pre-tenure allocation failure left the append-lane baton held with no live leader"; +} diff --git a/src/Disks/tests/gtest_cas_ref_log_format.cpp b/src/Disks/tests/gtest_cas_ref_log_format.cpp new file mode 100644 index 000000000000..d5186e77aca1 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_log_format.cpp @@ -0,0 +1,790 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include +#include +#include + +/// v3 text codec tests for `cas_ref_log` (codecs-v3 phase 3). Split out of the retired +/// `gtest_cas_ref_codecs.cpp` and re-pointed at the TEXT codec: the encoder-side validation tests are +/// format-agnostic (they only assert `encodeRefLogTxn` throws) and carry over verbatim; the old +/// binary-offset byte-patch decode tests (`bytes[k] = 99`) are gone — the shape-level corruption +/// classes (truncation, `v`+1 forward-gate, wrong type, leading garbage) are now covered by the +/// `CASFormatBattery.RefLog` row below. `RefTxnId` render/parse coverage lives here too (it rode in +/// the same suite and is independent of either ref codec). + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +} + +/// =================================================================================== +/// RefTxnId: render / parse +/// =================================================================================== + +TEST(CASRefCodec, RenderCanonicalForm) +{ + EXPECT_EQ(renderRefTxnId(RefTxnId{7, 0x8e}), "0000000000000007-000000000000008e"); + EXPECT_EQ(renderRefTxnId(RefTxnId{1, 1}), "0000000000000001-0000000000000001"); + EXPECT_EQ(renderRefTxnId(RefTxnId{0xffffffffffffffffULL, 0xffffffffffffffffULL}), + "ffffffffffffffff-ffffffffffffffff"); +} + +TEST(CASRefCodec, RenderRejectsZeroComponent) +{ + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + renderRefTxnId(RefTxnId{0, 1}); + }, + "RefTxnId: writer_epoch and ref_sequence must both be nonzero"); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + renderRefTxnId(RefTxnId{1, 0}); + }, + "RefTxnId: writer_epoch and ref_sequence must both be nonzero"); + EXPECT_DEATH( + { + DB::abort_on_logical_error.store(true, std::memory_order_relaxed); + renderRefTxnId(RefTxnId{0, 0}); + }, + "RefTxnId: writer_epoch and ref_sequence must both be nonzero"); +} + +TEST(CASRefCodec, ParseRoundTrip) +{ + for (const RefTxnId id : {RefTxnId{7, 0x8e}, RefTxnId{1, 1}, RefTxnId{255, 2}, RefTxnId{0x100000000ULL, 3}, + RefTxnId{0x8000000000000000ULL, 0x8000000000000000ULL}, + RefTxnId{0xffffffffffffffffULL, 0xffffffffffffffffULL}}) + { + const String rendered = renderRefTxnId(id); + const auto parsed = parseRefTxnId(rendered); + ASSERT_TRUE(parsed.has_value()); + EXPECT_EQ(*parsed, id); + } +} + +TEST(CASRefCodec, ParseRejectsShort) +{ + EXPECT_FALSE(parseRefTxnId("000000000000007-000000000000008e").has_value()); /// 32 chars, one short + EXPECT_FALSE(parseRefTxnId("7-8e").has_value()); + EXPECT_FALSE(parseRefTxnId("").has_value()); +} + +TEST(CASRefCodec, ParseRejectsLong) +{ + EXPECT_FALSE(parseRefTxnId("00000000000000007-000000000000008e").has_value()); /// 34 chars, one long + EXPECT_FALSE(parseRefTxnId("0000000000000007-000000000000008e0").has_value()); +} + +TEST(CASRefCodec, ParseRejectsUppercase) +{ + EXPECT_FALSE(parseRefTxnId("0000000000000007-00000000000000AE").has_value()); + EXPECT_FALSE(parseRefTxnId("0000000000000007-000000000000008E").has_value()); + EXPECT_FALSE(parseRefTxnId("0000000000000007-00000000000000Ae").has_value()); /// mixed case +} + +TEST(CASRefCodec, ParseRejectsZeroComponent) +{ + EXPECT_FALSE(parseRefTxnId("0000000000000000-000000000000008e").has_value()); + EXPECT_FALSE(parseRefTxnId("0000000000000007-0000000000000000").has_value()); + EXPECT_FALSE(parseRefTxnId("0000000000000000-0000000000000000").has_value()); +} + +TEST(CASRefCodec, ParseRejectsNonHexGarbage) +{ + EXPECT_FALSE(parseRefTxnId("000000000000000g-000000000000008e").has_value()); + EXPECT_FALSE(parseRefTxnId("!!!!!!!!!!!!!!!!-000000000000008e").has_value()); + EXPECT_FALSE(parseRefTxnId("0000000000000007_000000000000008e").has_value()); /// wrong separator +} + +TEST(CASRefCodec, ParseRejectsMisplacedSeparator) +{ + /// 17 hex digits then '-' then 15: same total length (33), dash at the wrong index -- the kind of + /// shape that, read naively without a fixed dash position, could be mistaken for an in-range but + /// overflowing first component. + EXPECT_FALSE(parseRefTxnId("00000000000000078-00000000000000e").has_value()); +} + +TEST(CASRefCodec, OrderMatchesLexicalOrderOfRender) +{ + const std::vector values{1, 2, 255, 1ULL << 32, 1ULL << 63}; + std::vector ids; + for (uint64_t epoch : values) + for (uint64_t seq : values) + ids.push_back(RefTxnId{epoch, seq}); + + std::mt19937 rng(42); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + for (int iter = 0; iter < 200; ++iter) + { + const RefTxnId & a = ids[rng() % ids.size()]; + const RefTxnId & b = ids[rng() % ids.size()]; + const String ra = renderRefTxnId(a); + const String rb = renderRefTxnId(b); + EXPECT_EQ(a < b, ra < rb) << ra << " vs " << rb; + EXPECT_EQ(a == b, ra == rb); + } +} + +/// =================================================================================== +/// RefLogTxn: round trip +/// =================================================================================== + +TEST(CASRefCodec, RoundTripNamespaceBirth) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +TEST(CASRefCodec, RoundTripRemoveNamespace) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{1, 2}; + RefOp op; + op.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +TEST(CASRefCodec, RoundTripSetPublishedAt) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{3, 5}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "all_1_1_0"; + op.expected_manifest_ref = manifestRef(3, 4, 1); + op.published_at_ms = 1717000000000ULL; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +/// No-tolerance decode pin (codex round-2, finding 3): the `"pl"` (payload) field was removed from the +/// ref-op wire in stage-1 T12. Although the retired `set_payload` op WORD is already rejected by +/// `opKindFromWord`, the generic op-record reader reads all field keys before switching on kind, so a +/// `"pl"` field paired with a still-recognized op word would otherwise be `skipUnknown`'d. It is a +/// removed field, not a genuinely-unknown one: decoding an op record that still carries `"pl"` must FAIL +/// with `CORRUPTED_DATA` naming the removed field. +TEST(CASRefCodec, DecodeRejectsRemovedPayloadFieldInOpRecord) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{3, 5}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "all_1_1_0"; + op.expected_manifest_ref = manifestRef(3, 4, 1); + op.published_at_ms = 1717000000000ULL; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + /// Splice the retired `"pl"` field back into the op record, just before its `"ts"` field. + const String needle = ",\"ts\":"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + const String tampered = bytes.substr(0, pos) + R"(,"pl":"deadbeef")" + bytes.substr(pos); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); +} + +TEST(CASRefCodec, RoundTripSetPublishedAtZeroTimestamp) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + op.published_at_ms = 0; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +TEST(CASRefCodec, RoundTripOwnerTransitionAdd) +{ + /// new-only = add: no old_binding, a fresh new_binding. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "all_1_1_0", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + ASSERT_TRUE(decoded.ops[0].new_binding.has_value()); + EXPECT_FALSE(decoded.ops[0].old_binding.has_value()); +} + +TEST(CASRefCodec, RoundTripOwnerTransitionRemoval) +{ + /// old-only = removal: an old_binding, no new_binding. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "all_1_1_0", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + EXPECT_FALSE(decoded.ops[0].new_binding.has_value()); + ASSERT_TRUE(decoded.ops[0].old_binding.has_value()); +} + +TEST(CASRefCodec, RoundTripOwnerTransitionReplace) +{ + /// both present = replace. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "all_1_1_0", manifestRef(1, 1, 1)}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "all_1_1_0", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + ASSERT_TRUE(decoded.ops[0].old_binding.has_value()); + ASSERT_TRUE(decoded.ops[0].new_binding.has_value()); +} + +TEST(CASRefCodec, RoundTripMultipleOpsInOneTransaction) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{9, 100}; + + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(birth); + + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "a/b/c", manifestRef(9, 1, 1)}; + txn.ops.push_back(add); + + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + set_published_at.ref_name = "a/b/c"; + set_published_at.expected_manifest_ref = manifestRef(9, 1, 1); + set_published_at.published_at_ms = 42; + txn.ops.push_back(set_published_at); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); + EXPECT_EQ(decoded.ops.size(), 3u); +} + +/// A re-encode of a decoded transaction is byte-identical (the encoder is a pure function of the txn). +TEST(CASRefCodec, ByteIdenticalReencode) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{9, 100}; + + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(birth); + + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "a/b/c", manifestRef(9, 1, 1)}; + add.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "a/b/c", manifestRef(9, 1, 1)}; + txn.ops.push_back(add); + + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + set_published_at.ref_name = "a/b/c"; + set_published_at.expected_manifest_ref = manifestRef(9, 1, 1); + set_published_at.published_at_ms = 1717000000000ULL; + txn.ops.push_back(set_published_at); + + const String bytes1 = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes1, txn.ns, txn.txn_id); + const String bytes2 = encodeRefLogTxn(decoded); + EXPECT_EQ(bytes1, bytes2); +} + +/// =================================================================================== +/// RefLogTxn: validation rejections (encoder-side + key/body binding + truncation) +/// =================================================================================== + +TEST(CASRefCodec, EncodeRejectsZeroTxnId) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{0, 1}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, DecodeRejectsTruncatedBuffer) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + const String bytes = encodeRefLogTxn(txn); + + /// Dropping the trailing bytes leaves the final line without its '\n' terminator -> fail closed. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefLogTxn(bytes.substr(0, bytes.size() - 3), txn.ns, txn.txn_id); }); +} + +TEST(CASRefCodec, DecodeRejectsBodyNamespaceMismatch) +{ + RefLogTxn txn; + txn.ns = "ns-a"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + const String bytes = encodeRefLogTxn(txn); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefLogTxn(bytes, "ns-b", txn.txn_id); }); +} + +TEST(CASRefCodec, DecodeRejectsBodyTxnIdMismatch) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + const String bytes = encodeRefLogTxn(txn); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefLogTxn(bytes, txn.ns, RefTxnId{1, 2}); }); +} + +TEST(CASRefCodec, EncodeRejectsEmptyRefName) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = ""; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsDotRefName) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "."; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsDotDotSegment) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "a/../b"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsRepeatedSeparator) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "a//b"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsLeadingSlash) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "/a"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsTrailingSlash) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "a/"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsBackslash) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "a\\b"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsNonCanonicalOwnerBindingRefName) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "..", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsEmbeddedNulRefName) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = String("a\0b", 3); /// embedded NUL byte -- never legitimate in a ref name + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsTooManyOps) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + for (size_t i = 0; i < ref_txn_max_ops + 1; ++i) + { + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + } + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeAllowsExactlyMaxOps) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + for (size_t i = 0; i < ref_txn_max_ops; ++i) + { + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + } + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded.ops.size(), ref_txn_max_ops); +} + +TEST(CASRefCodec, EncodeRejectsOversizedNormalTransaction) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r" + String(ref_txn_max_bytes + 1, 'x'); + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, RemovalClassTransactionLiftsByteBudgetAboveNormalLimit) +{ + /// A RemoveNamespace transaction carrying a ref_name bigger than the NORMAL limit but within the + /// REMOVAL limit must succeed -- proving the removal-class flag actually lifts the byte budget + /// rather than merely being ignored. The single set_published_at op here is also vastly bigger + /// than `ref_op_max_bytes`, so this doubles as proof that removal-class ops are exempt from the + /// per-op cap too. + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(remove); + + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + set_published_at.ref_name = "r" + String(ref_txn_max_bytes + 1024, 'x'); + set_published_at.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(set_published_at); + + const String bytes = encodeRefLogTxn(txn); + EXPECT_GT(bytes.size(), ref_txn_max_bytes); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +TEST(CASRefCodec, RemovalClassTransactionStillRejectsBeyondRemovalLimit) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(remove); + + RefOp set_published_at; + set_published_at.kind = RefOpKind::SetPublishedAt; + set_published_at.ref_name = "r" + String(ref_removal_max_bytes + 1, 'x'); + set_published_at.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(set_published_at); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, RemovalClassTransactionNotCappedOnOpCount) +{ + /// A removal-class transaction may exceed `ref_txn_max_ops` -- only the (much larger) byte budget + /// bounds it, per spec ("its operation count is bounded by that byte limit"). + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(remove); + for (size_t i = 0; i < ref_txn_max_ops + 10; ++i) + { + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(op); + } + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded.ops.size(), txn.ops.size()); +} + +/// Stage-1 T8 (spec §3 "Budget: counts only, chunked flush") retires the scenario this test used to +/// pin: ONE op carrying almost the whole `ref_txn_max_bytes` budget in its payload. The new per-op +/// cap (`ref_op_max_bytes`, `EncodeAllowsExactlyMaxPerOpBytes` below) makes that construction illegal +/// for a normal-class transaction — no single op may exceed `ref_op_max_bytes` regardless of the +/// whole-transaction budget — so the exact-boundary pin moves to the per-op cap, the boundary a +/// legally-admitted normal-class transaction can actually reach (`ref_txn_max_ops * ref_op_max_bytes` +/// stays comfortably under `ref_txn_max_bytes`, pinned by `CanonicalMaxTransactionRoundTrips` in +/// `gtest_cas_ref_chunked_flush.cpp`). + +TEST(CASRefCodec, EncodeAllowsExactlyMaxPerOpBytes) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + + const size_t base_size = encodedOpSize(op); + ASSERT_LE(base_size, ref_op_max_bytes); + /// Every added 'a' is one un-escaped byte inside the JSON ref-name string, so the encoded op size + /// grows one-for-one to exactly the per-op cap. + txn.ops[0].ref_name = "r" + String(ref_op_max_bytes - base_size, 'a'); + ASSERT_EQ(encodedOpSize(txn.ops[0]), ref_op_max_bytes); + + const String bytes = encodeRefLogTxn(txn); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +TEST(CASRefCodec, EncodeRejectsOversizedOpOnNormalTransaction) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(op); + + const size_t base_size = encodedOpSize(op); + ASSERT_LE(base_size, ref_op_max_bytes); + /// One byte past the per-op cap, well within the whole-transaction byte cap -- isolates the + /// per-op check from the (much larger) whole-transaction one. + txn.ops[0].ref_name = "r" + String(ref_op_max_bytes - base_size + 1, 'a'); + ASSERT_GT(encodedOpSize(txn.ops[0]), ref_op_max_bytes); + ASSERT_LT(encodedOpSize(txn.ops[0]), ref_txn_max_bytes); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeAllowsExactlyMaxRemovalBytes) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + txn.ops.push_back(remove); + RefOp ts_op; + ts_op.kind = RefOpKind::SetPublishedAt; + ts_op.ref_name = "r"; + ts_op.expected_manifest_ref = manifestRef(1, 1, 1); + txn.ops.push_back(ts_op); + + const size_t base_size = encodeRefLogTxn(txn).size(); + ASSERT_LE(base_size, ref_removal_max_bytes); + /// Every added 'x' is one un-escaped byte inside the JSON ref-name string, so the encoded size + /// grows one-for-one to exactly the cap; the base "r" contributes 1 byte already counted in + /// base_size, so appending (rather than replacing) reaches the target exactly. + txn.ops[1].ref_name = "r" + String(ref_removal_max_bytes - base_size, 'x'); + + const String bytes = encodeRefLogTxn(txn); + EXPECT_EQ(bytes.size(), ref_removal_max_bytes); + const RefLogTxn decoded = decodeRefLogTxn(bytes, txn.ns, txn.txn_id); + EXPECT_EQ(decoded, txn); +} + +/// ManifestRef field validation, enforced by the log codec (spec's "invalid identifiers are rejected" +/// binds both codecs). Encoder-side only -- the decode path re-runs the identical checks and is +/// covered by the round-trips + the battery. + +TEST(CASRefCodec, EncodeRejectsZeroManifestRefWriterEpochInOwnerBinding) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "r", manifestRef(0, 1, 1)}; + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsZeroManifestRefBuildSequenceInOwnerBinding) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "r", manifestRef(1, 0, 1)}; + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsOutOfRangeManifestOrdinalInOwnerBinding) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "r", manifestRef(1, 1, 0)}; + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +TEST(CASRefCodec, EncodeRejectsZeroManifestRefInSetPublishedAt) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = manifestRef(1, 1, 0); + txn.ops.push_back(op); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); +} + +/// =================================================================================== +/// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) +/// =================================================================================== + +TEST(CASFormatBattery, RefLog) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "all_1_1_0"; + op.expected_manifest_ref = manifestRef(1, 1, 1); + op.published_at_ms = 42; + txn.ops.push_back(op); + + const String ns = txn.ns; + const RefTxnId id = txn.txn_id; + runFormatBattery({FormatId::RefLog, + [txn] { return sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); }, + [ns, id](std::string_view s) { decodeRefLogTxn(openObject(FormatId::RefLog, s), ns, id); }, + currentFormatHeader("cas_ref_log") + + "{\"ns\":\"ns\",\"we\":\"1\",\"rs\":\"1\"}\n" + "{\"op\":\"set_published_at\",\"rn\":\"all_1_1_0\",\"me\":\"1\",\"mb\":\"1\",\"mo\":1,\"ts\":42}\n" + "{\"n\":1}\n"}); +} diff --git a/src/Disks/tests/gtest_cas_ref_read_contract.cpp b/src/Disks/tests/gtest_cas_ref_read_contract.cpp new file mode 100644 index 000000000000..a86d2530d4e8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_read_contract.cpp @@ -0,0 +1,256 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +/// The ref-side read contract (the ten `CasRefCatalog::read` sites in `CasRefLedger.cpp` were +/// classified elsewhere: every site is a mutation/admission authority or a per-key destructive +/// revalidation, and every live table reader reaches `acquireReadableRefTableRuntime`, whose warm path +/// returns the resident runtime before any catalog read). These are COVERAGE PINS for a contract the +/// classification predicted already holds, not a fix: a held reader runtime answers stale-or-absent +/// across a same-name rebirth, a warm read costs no catalog request, and the one held ref-writer seam +/// that exists (`dropNamespace(const NamespaceLifeId &)`) refuses across the same rebirth rather than +/// touching the successor. + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; + extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +/// One committed ref, born and published through the REAL production write path (`beginPartWrite` / +/// `stageManifest` / `precommitAdd` / `promote`) -- this is what mints `ns`'s catalog life for real, +/// exactly as an ordinary insert would, rather than a fixture sentinel. +ManifestId publishRefThroughPool(const PoolPtr & store, const RootNamespace & ns, const String & ref_name) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref_name; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref_name, id); + build->promote(ns, ref_name, build->buildId(), id); + return id; +} + +/// Delete the current catalog life through the production exact-removal authority (`casUpdate` to +/// `Removing`, then `deleteCompletedRemoving` under a held fence), retaining every old physical byte +/// and any already-resident runtime. Mirrors `gtest_cas_ns_file_read_contract.cpp`'s +/// `deleteCatalogLife` -- lifecycle-real, not a raw sentinel overwrite. +void deleteCatalogLife(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +{ + CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == life.ns && entry.incarnation == life.incarnation; + }); + if (it == next.entries.end()) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Missing fixture catalog life '{}'", life.ns.string()); + it->state = NsState::Removing; + it->removal_started_round = 1; + return next; + }); + + const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(backend, layout); + const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == life.ns && entry.incarnation == life.incarnation; + }); + if (it == snapshot.catalog.entries.end()) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Missing Removing fixture catalog life '{}'", life.ns.string()); + + CasFoldSeal parent; + parent.ref_lives.emplace(life.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); + if (CasRefCatalog::deleteCompletedRemoving( + backend, layout, *it, parent, 1, + [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }) + != CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Failed to delete fixture catalog life '{}'", life.ns.string()); +} + +/// Admit a fresh `Live` catalog row at the SAME logical name as `predecessor` -- the same-name +/// rebirth every test below drives. Mirrors `gtest_cas_ns_file_read_contract.cpp`'s +/// `admitReplacementLife`. +NamespaceLifeId admitReplacementLife( + Backend & backend, const Layout & layout, uint64_t gc_shards, + const NamespaceLifeId & predecessor, UInt128 successor_incarnation) +{ + if (predecessor.incarnation == successor_incarnation) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Fixture life ids unexpectedly collide"); + const NamespaceLifeId successor = NamespaceLifeId::fromCatalogEntry(predecessor.ns, successor_incarnation); + CasRefCatalog::casAdmitEntry(backend, layout, gc_shards, CatalogEntry{ + .ns = successor.ns, .state = NsState::Live, .incarnation = successor.incarnation}); + return successor; +} + +} + +/// This reader's runtime already holds life 1. Reusing it after a same-name rebirth is a retained +/// life-handle operation, not a fresh logical-name admission: it may still answer life 1's committed +/// value (or absent), but it must never surface life 2's -- the opaque physical life id makes the +/// successor's bytes structurally unreachable through an unrefreshed handle. +TEST(CASRefReadContract, HeldRuntimeAfterSameNameRebirthReadsStaleOrNotFoundNeverSuccessorRefs) +{ + auto backend = std::make_shared(); + /// A 1-byte whole-table cache budget is the production knob (`CASRefTableCacheEviction`) that lets a + /// single store instance both HOLD a table's runtime and, later, genuinely forget it by touching a + /// different table -- so the "fresh resolution" positive control below is a real re-recovery, not + /// a second mount. + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .ref_table_cache_bytes = 1}); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/ref_read_contract_rebirth@cas@"}; + const RootNamespace throwaway_ns{"00/ref_read_contract_rebirth_evictor@cas@"}; + const String ref_name = "part_1"; + + const ManifestId life1_manifest = publishRefThroughPool(store, ns, ref_name); + + /// Hold the reader runtime resident: one read. + const auto held_before = store->resolveRef(ns, ref_name); + ASSERT_TRUE(held_before.has_value()); + EXPECT_EQ(held_before->manifest_id, life1_manifest); + ASSERT_TRUE(store->refTableLifeForTest(ns).has_value()); + const NamespaceLifeId life1 = *store->refTableLifeForTest(ns); + + /// Drop and re-admit under the SAME logical name, bypassing this store's own ledger entirely -- + /// exactly as an independent actor's drop/rebirth would look from this reader's point of view. + deleteCatalogLife(*backend, layout, life1); + const NamespaceLifeId life2 = admitReplacementLife(*backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc123}); + ASSERT_NE(life1.incarnation, life2.incarnation); + + const ManifestRef life2_ref{/*writer_epoch*/ 1, /*build_sequence*/ 777, /*manifest_ordinal*/ 1}; + publishCommittedTransition(*backend, layout, ns, ref_name, std::nullopt, life2_ref); + const ManifestId life2_manifest{ns, life2_ref}; + ASSERT_NE(life2_manifest, life1_manifest); + + /// The held runtime never re-validates the catalog: it answers from its resident cache -- stale or + /// not-found -- but never the successor's value. + const auto held_after = store->resolveRef(ns, ref_name); + EXPECT_NE( + held_after.has_value() ? std::optional(held_after->manifest_id) : std::nullopt, + std::optional(life2_manifest)); + if (held_after.has_value()) + EXPECT_EQ(held_after->manifest_id, life1_manifest); + + /// Force the cached runtime out: touch a different namespace under the 1-byte cache budget (the + /// production whole-table eviction path), so the NEXT access to `ns` re-recovers from scratch. + (void)publishRefThroughPool(store, throwaway_ns, "evict"); + ASSERT_FALSE(store->refTableCachedForTest(ns)); + + /// Positive control: a fresh resolution -- through the SAME Pool, now cold -- sees life 2. Not + /// vacuous: the value really did move, and an unrefreshed handle really would have missed it. + const auto fresh = store->resolveRef(ns, ref_name); + ASSERT_TRUE(fresh.has_value()); + EXPECT_EQ(fresh->manifest_id, life2_manifest); +} + +/// The disjoint half of the read-side contract: once a table's runtime is resident, an ordinary read +/// costs no catalog request at all -- the recovered-and-cached `RefTableState` is this process's +/// sole authority for a table it has already opened. +TEST(CASRefReadContract, HotRefReadsThroughHeldRuntimeIssueZeroCatalogRequests) +{ + auto backend = std::make_shared(); + PoolPtr store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/ref_read_contract_hot@cas@"}; + const String ref_name = "part_1"; + + const ManifestId published = publishRefThroughPool(store, ns, ref_name); + const auto warm = store->resolveRef(ns, ref_name); + ASSERT_TRUE(warm.has_value()); + EXPECT_EQ(warm->manifest_id, published); + ASSERT_TRUE(store->refTableLifeForTest(ns).has_value()); + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + + /// Positive control, captured BEFORE the reset below: the cold admission above really did reach + /// the catalog and this namespace's own ref stream, so the upcoming zero is an absence and not a + /// recorder that never saw anything. + EXPECT_GT( + backend->headCount(layout.refCatalogKey()) + backend->getCount(layout.refCatalogKey()) + + backend->casPutCount(layout.refCatalogKey()), + 0u) << "the cold admission above must have reached the catalog at least once"; + EXPECT_GT(backend->getCount(layout.refCkptKey(life)), 0u) + << "the cold recovery above must have read this namespace's own checkpoint at least once"; + + backend->resetCounts(); + + (void)store->resolveRef(ns, ref_name); + (void)store->listRefs(ns); + (void)store->hasAnyRefWithPrefix(ns, ""); + + EXPECT_EQ(backend->headCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); + /// Stronger than the catalog-only clauses above: a warm ref read is a pure map lookup over the + /// recovered state (`ensureRefTableRecovered`'s early return once `rt.recovered`), so it issues no + /// backend request whatsoever, not merely none against the catalog. + EXPECT_TRUE(backend->touchedKeys().empty()) + << "a warm ref read must issue no backend requests at all"; +} + +/// The one held ref-WRITER seam the classification found: `dropNamespace(const NamespaceLifeId &)`'s +/// exact-incarnation guard. A stale holder can only be refused, never allowed to act on the successor +/// -- there is no path by which it could target life 2's row or its ref data. +TEST(CASRefReadContract, StaleLifeDropRefusesAfterRebirthAndNeverTouchesSuccessor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/ref_read_contract_stale_drop@cas@"}; + const String ref_name = "part_1"; + + (void)publishRefThroughPool(store, ns, ref_name); + ASSERT_TRUE(store->refTableLifeForTest(ns).has_value()); + const NamespaceLifeId life1 = *store->refTableLifeForTest(ns); + + deleteCatalogLife(*backend, layout, life1); + const NamespaceLifeId life2 = admitReplacementLife(*backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc456}); + + const ManifestRef life2_ref{/*writer_epoch*/ 1, /*build_sequence*/ 999, /*manifest_ordinal*/ 1}; + publishCommittedTransition(*backend, layout, ns, ref_name, std::nullopt, life2_ref); + const ManifestId life2_manifest{ns, life2_ref}; + + const HeadResult catalog_head_before = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(catalog_head_before.exists); + const auto catalog_get_before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_get_before.has_value()); + + /// The held life-1 handle names an incarnation the catalog no longer carries: refused, not + /// resolved against the current (life-2) row. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(life1); }); + + const HeadResult catalog_head_after = backend->head(layout.refCatalogKey()); + ASSERT_TRUE(catalog_head_after.exists); + EXPECT_EQ(catalog_head_after.token, catalog_head_before.token) + << "a refused stale-life drop must not touch the catalog object at all"; + const auto catalog_get_after = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_get_after.has_value()); + EXPECT_EQ(catalog_get_after->bytes, catalog_get_before->bytes); + + /// Life 2's ref data is untouched: a fresh resolution (a separate mount over the same backend, + /// exactly like `CASRefWriterRuntimeIdentity.ColdReadRejectsReplacementByExternalPoolActor`'s + /// `external_store`) still sees exactly the value published above. + auto verify_store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "verify"}); + const auto resolved = verify_store->resolveRef(ns, ref_name); + ASSERT_TRUE(resolved.has_value()); + EXPECT_EQ(resolved->manifest_id, life2_manifest); +} diff --git a/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp new file mode 100644 index 000000000000..2b75ae13ada0 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp @@ -0,0 +1,2010 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// Stage A task 6: recovery is `_ckpt` + an ARITHMETIC tail + a seal CAS-walk, and it installs nothing +/// without presenting the fence generation it was admitted under. +/// +/// The one sentence this suite exists to defend: **recovery performs no stream `LIST`.** Everything the +/// old recovery knew about a table's durable stream came from one `LIST`, so a listing that silently +/// omitted a key produced a table missing an ACKED transaction and looked perfectly healthy. The +/// checkpoint now supplies the only base and finite frontier; arithmetic exact GETs decide recovery. +/// These list-liar fixtures are retained as sentinels: hiding or fabricating a listed key cannot affect +/// recovery because recovery sends zero stream LIST requests. +/// +/// The other half is INV-2: a dead epoch is closed IN-BAND, by a seal transaction the store's own +/// conditional create places at exactly `{E, T+1}` -- the key a dying predecessor's in-flight PUT would +/// have taken. That is why the walk WRITES, and why every write it performs is gated on the ONE fence +/// generation captured when the recovery was admitted (slot-occupy, the `_ckpt` CAS, and the install +/// recheck -- one capture, three checks). +/// +/// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int NETWORK_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASRefRecoveryRestarts; +extern const Event CASRefRecoveryEpochSealed; +extern const Event CASRefRecoveryEpochSealAdopted; +extern const Event CASRefRecoveryStragglerAdopted; +extern const Event CASRefRecoveryCancelled; +extern const Event CASRefCheckpointPublished; +} + +using namespace DB::Cas; +using DB::Cas::tests::committedRow; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::minimalLiveSnapshot; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; +using DB::Cas::tests::rearmMountFenceAfterAnomalyForTest; +using DB::Cas::tests::writeRefSnapshotRaw; + +namespace +{ + +ManifestRef manifestRef(uint64_t epoch, uint64_t build_sequence, uint32_t ordinal) +{ + return ManifestRef{epoch, build_sequence, ordinal}; +} + +/// Make the durable mount immediately reclaimable so a test that deliberately moved the local fence +/// generation can drive the production remount boundary without paying a live-lease expiry wait. +void fenceOutMountForRemount(Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + MountLease mount = decodeMountLease(got->bytes); + mount.gc_fenced = true; + mount.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(mount), got->token).outcome, + PutOutcome::Done); +} + +/// A backend whose `LIST` can lie by omission. `hidden_keys` remain readable by exact key, so these +/// fixtures prove the stronger modern rule: recovery sends no stream `LIST` at all and therefore cannot +/// be affected by an enumeration inconsistency. +/// +/// Deliberately NOT a "delete the object" fixture: an object that is genuinely gone is a different +/// (and already covered) case. The blocker is an object that EXISTS and is invisible to enumeration. +class HidingListBackend : public CountingBackend +{ +public: + explicit HidingListBackend(bool seed_pool_meta = true) + { + if (seed_pool_meta) + DB::Cas::tests::seedPoolMetaForRestart(*this); + } + + using CountingBackend::get; + using CountingBackend::list; + using CountingBackend::putIfAbsent; + using CountingBackend::casPut; + + std::set hidden_keys; + std::set phantom_list_keys; + + /// Every `putIfAbsent` of a key containing this substring throws a PLAIN (non-`DB::Exception`) + /// error, which `classifyConditionalWriteResult` can only ever classify `Unresolved` -- never + /// `DefiniteFailure`. Persistent rather than one-shot on purpose: the subject is what recovery does + /// when the store KEEPS refusing to say whether the write landed. + String ambiguous_put_substr; + + /// Persistent thrown response for a matching mutable checkpoint CAS. The ref-log PUT has already + /// completed when tests arm this, producing the exact one-successor recovery window. + String ambiguous_cas_substr; + int ambiguous_cas_count = 0; + + /// Runs after a checkpoint publisher read its expected token but before that publisher presents + /// its CAS. This is the exact window in which another admitted writer can advance the frontier. + std::function &)> before_cas_put; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = CountingBackend::list(prefix, cursor, limit); + std::vector kept; + kept.reserve(page.keys.size()); + for (ListedKey & lk : page.keys) + if (!hidden_keys.contains(lk.key)) + kept.push_back(std::move(lk)); + if (cursor.empty()) + { + for (const String & key : phantom_list_keys) + { + if (key.starts_with(prefix)) + kept.push_back(ListedKey{.key = key, .size = 0, .token = std::nullopt}); + } + } + page.keys = std::move(kept); + return page; + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (!ambiguous_put_substr.empty() && key.find(ambiguous_put_substr) != String::npos) + throw std::runtime_error("injected ambiguous putIfAbsent"); + return CountingBackend::putIfAbsent(key, bytes, meta); + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (before_cas_put) + before_cas_put(key, bytes, expected); + if (ambiguous_cas_count > 0 && !ambiguous_cas_substr.empty() + && key.find(ambiguous_cas_substr) != String::npos) + { + --ambiguous_cas_count; + throw Poco::TimeoutException("HidingListBackend: simulated ambiguous checkpoint CAS"); + } + return CountingBackend::casPut(key, bytes, expected, meta); + } +}; + +/// Fires `on_key` immediately AFTER a `putIfAbsent` whose key contains `watched_substr` -- the +/// deterministic way to act inside recovery's own write window (bump a fence, land a straggler) with no +/// sleep and no second thread. `skip` lets a test target the Nth such write. +class PutHookBackend : public HidingListBackend +{ +public: + using HidingListBackend::putIfAbsent; + + using HidingListBackend::casPut; + + String watched_substr; + uint64_t skip = 0; + std::function on_key; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + PutResult result = HidingListBackend::putIfAbsent(key, bytes, meta); + fireIfWatched(key); + return result; + } + + /// The `_ckpt` advance is a token-CAS, not a create, whenever the object already exists -- which is + /// the normal case, since the namespace birth creates it. Hooking only `putIfAbsent` would silently + /// never fire for it. + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + CasResult result = HidingListBackend::casPut(key, bytes, expected, meta); + fireIfWatched(key); + return result; + } + +private: + void fireIfWatched(const String & key) + { + if (!on_key || watched_substr.empty() || key.find(watched_substr) == String::npos) + return; + if (skip > 0) + { + --skip; + return; + } + auto hook = on_key; + on_key = nullptr; /// one-shot: a hook that re-enters its own trigger would recurse + hook(); + } +}; + +/// Materializes `late_bytes` at `late_key` at the instant the walk READS that key and finds it absent -- +/// i.e. strictly between the read and the conditional create that follows it. +/// +/// This is the only faithful way to construct the race the `Occupied` arms exist for. Seeding the object +/// up front does NOT work, and finding that out is the point: the walk fetches every id by EXACT KEY, so +/// an object hidden from the listing is simply FOUND by the read and applied there. To meet it as an +/// OCCUPANT of the slot, it has to arrive after the read said absent -- which is exactly what a +/// straggler, or a concurrent recoverer's seal, does. +class LateMaterializeBackend : public HidingListBackend +{ +public: + using HidingListBackend::get; + + String late_key; + String late_bytes; + + std::optional get(const String & key, Range range) override + { + std::optional result = HidingListBackend::get(key, range); + if (!result && !late_key.empty() && key == late_key) + { + CountingBackend::putIfAbsent(late_key, late_bytes); + late_key.clear(); /// one-shot: the walk must see it present from here on + } + return result; + } +}; + +/// Fires `on_key` immediately BEFORE a `get` whose key contains `watched_substr`, and can additionally +/// FAULT that read with a transient object-store error -- the I/O seam the remount-barrier test pauses +/// recovery at. +class GetSeamBackend : public HidingListBackend +{ +public: + using HidingListBackend::get; + + String watched_substr; + + /// Assigned from the test thread and read from whatever thread the recovery runs on, so the + /// read-and-move below is guarded. Today's tests all assign before starting the recovery thread and + /// clear after joining it, so there is no race to fix -- but this is the same seam that already + /// produced one use-after-free, and "the current tests happen not to race" is not a property a + /// future test author can see. The mutex makes the constraint enforced rather than remembered. + std::mutex hook_mutex; + std::function on_key; + + std::optional get(const String & key, Range range) override + { + std::unique_lock hook_lock(hook_mutex); + if (on_key && !watched_substr.empty() && key.find(watched_substr) != String::npos) + { + /// ONE-SHOT by moving the callback OUT before invoking it, and that is a correctness + /// requirement rather than a convenience. A hook that cleared `on_key` from inside its own + /// body would destroy the `std::function` whose closure it is still executing, and every + /// by-reference capture it touched afterwards would read freed heap. That is not + /// theoretical: it is what the first version of these tests did, and the ASan gate caught + /// it as a `heap-use-after-free` while a hook was parked on a condition variable. + auto hook = std::move(on_key); + on_key = nullptr; + /// Released before the hook runs: it parks on a condition variable, and holding the seam's + /// own mutex across that would deadlock the very thread meant to release it. + hook_lock.unlock(); + hook(key); + } + return HidingListBackend::get(key, range); + } +}; + +/// Fires once after an exact GET has already fixed its result. This is the recovery authority seam: +/// another actor advances the log+checkpoint after the walk observed its old end, but before the walk +/// performs its final catalog/checkpoint validation. +class AfterGetHookBackend : public HidingListBackend +{ +public: + using HidingListBackend::get; + + String watched_key; + std::function after_get; + + std::optional get(const String & key, Range range) override + { + std::optional result = HidingListBackend::get(key, range); + if (after_get && key == watched_key) + { + auto hook = std::move(after_get); + after_get = nullptr; + hook(); + } + return result; + } +}; + +CasRequestBudget tinyBudget() +{ + return CasRequestBudget{ + .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; +} + +PoolConfig walkTestConfig() +{ + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.server_id = DB::UInt128(1); + config.cas_request_budget = tinyBudget(); + config.wait_sleep_fn = [](uint64_t) {}; + /// No background publication: every test here drives its own, so a threshold-triggered snapshot can + /// never move the base under an assertion about which base recovery chose. + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + return config; +} + +PoolPtr openWalkPool(const BackendPtr & backend, PoolConfig config = walkTestConfig()) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend, config.pool_prefix); + return Pool::open(backend, std::move(config)); +} + +/// Burns durable writer epochs so a subsequent `Pool::open` allocates `target_live_epoch`. Epochs are +/// minted, never reclaimed (`CasPool.cpp`'s allocator), so this is exactly what a pool that has been +/// mounted `n` times looks like -- including the burned epochs in which nothing was ever written, which +/// the seal chain must cross. +void burnEpochsUpTo(Backend & backend, const Layout & layout, uint64_t target_live_epoch) +{ + for (uint64_t e = 1; e < target_live_epoch; ++e) + allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); +} + +/// One ordinary transaction at `id`, publishing `ref` (prepending the birth op when `birth`). +RefLogTxn makeOrdinaryTxn(const RootNamespace & ns, RefTxnId id, const String & ref, bool birth, + std::optional prev_epoch_seal = std::nullopt) +{ + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = id; + if (birth) + txn.ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps(ref, manifestRef(id.writer_epoch, id.ref_sequence, 1u))) + txn.ops.push_back(op); + txn.prev_epoch_seal = prev_epoch_seal; + return txn; +} + +/// The terminal `remove_namespace` op (this project's warning set requires every field named, so it is +/// built field-by-field rather than by designated init). +RefOp removeNamespaceOp() +{ + RefOp op; + op.kind = RefOpKind::RemoveNamespace; + return op; +} + +/// One EPOCH SEAL transaction at `id` -- what a concurrent recoverer leaves behind. +RefLogTxn makeSealTxn(const RootNamespace & ns, RefTxnId id, + std::optional prev_epoch_seal = std::nullopt) +{ + RefLogTxn seal; + seal.ns = ns.string(); + seal.txn_id = id; + RefOp op; + op.kind = RefOpKind::EpochSeal; + seal.ops.push_back(op); + seal.prev_epoch_seal = prev_epoch_seal; + return seal; +} + +void seedTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId id, + const String & ref, bool birth) +{ + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, makeOrdinaryTxn(ns, id, ref, birth)); +} + +/// Seeds the `_ckpt` a real namespace birth would have created, so recovery can ground its walk at the +/// namespace's `life_epoch` without consulting the (untrusted) listing. Raw, because these fixtures +/// never run a birth through the append lane. +void seedCkpt(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefCkpt & ckpt) +{ + backend.putIfAbsent(layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(ckpt)); +} + +RefCkpt lifeEpochCkpt(uint64_t life_epoch, std::optional committed_through = std::nullopt) +{ + return RefCkpt{.life_epoch = std::optional{life_epoch}, + .committed_through = committed_through, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}; +} + +/// The decoded transaction at `id`, or `nullopt` when the object is absent. Never dereferences a +/// disengaged optional: an aborted binary would take every later suite's result with it. +std::optional readLogTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId id) +{ + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + if (!got) + return std::nullopt; + return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id); +} + +uint64_t counterOf(ProfileEvents::Event event) +{ + return ProfileEvents::global_counters[event].load(); +} + +NamespaceLifeId catalogLife(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + for (const CatalogEntry & entry : catalog.catalog.entries) + if (entry.ns == ns) + return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); + throw std::runtime_error("test namespace has no catalog life"); +} + +NamespaceLifeId strandOneUnfrontieredSuccessor( + HidingListBackend & backend, const PoolPtr & store, const Layout & layout, const RootNamespace & ns) +{ + store->appendRefOps(ns, MutationScope::ref("a"), + [](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps("a", manifestRef(1, 1, 1))) + ops.push_back(op); + return ops; + }, RootMutationOrigin::Writer, RootMutationKind::Publish); + + const NamespaceLifeId life = catalogLife(backend, layout, ns); + backend.ambiguous_cas_substr = layout.refCkptKey(life); + backend.ambiguous_cas_count = 200; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->appendRefOps(ns, MutationScope::ref("b"), + [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + backend.ambiguous_cas_count = 0; + return life; +} + +CatalogEntry replaceCatalogLifeForTest( + Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) +{ + const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + RefCatalog without_predecessor = before_delete.catalog; + std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) + { + return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; + }); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to retire exact predecessor catalog life"); + + CatalogEntry successor{ + .ns = predecessor.ns, + .state = NsState::Live, + .incarnation = successor_incarnation, + .creator = std::nullopt}; + const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + RefCatalog reborn = after_delete.catalog; + reborn.entries.push_back(successor); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to publish successor catalog life"); + return successor; +} + +} + +/// --------------------------------------------------------------------------------------------- +/// The checkpoint-bounded arithmetic tail: no recovery LIST +/// --------------------------------------------------------------------------------------------- + +/// The durable stream is `{1,1} {1,2} {1,3}` while the backend hides the middle key from LIST. +/// Recovery must make zero stream LIST requests and recover the same exact checkpoint range. +TEST(CASRefRecoveryCasWalk, HiddenMiddleLogDoesNotAffectCheckpointRecovery) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/hint_middle"}; + + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 3})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + seedTxn(*backend, layout, ns, RefTxnId{1, 2}, "b", /*birth=*/false); + seedTxn(*backend, layout, ns, RefTxnId{1, 3}, "c", /*birth=*/false); + backend->hidden_keys.insert(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2})); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + backend->resetCounts(); + + const auto refs = store->listRefs(ns); + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns))), 0u); + EXPECT_EQ(refs.size(), 3u) << "the arithmetic walk must fetch {1,2} by exact key"; + EXPECT_TRUE(refs.contains("a")); + EXPECT_TRUE(refs.contains("b")) << "'b' is the ref the omitted transaction published"; + EXPECT_TRUE(refs.contains("c")); +} + +/// The same sentinel at the tail. A hidden tail key is still found by the bounded exact walk, not by a +/// stream enumeration. +TEST(CASRefRecoveryCasWalk, HiddenTailLogDoesNotAffectCheckpointRecovery) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/hint_tail"}; + + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 2})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + seedTxn(*backend, layout, ns, RefTxnId{1, 2}, "b", /*birth=*/false); + backend->hidden_keys.insert(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2})); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + backend->resetCounts(); + + const auto refs = store->listRefs(ns); + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns))), 0u); + EXPECT_EQ(refs.size(), 2u) << "an omitted TAIL id is indistinguishable from the end of the stream to a " + "listing; recovery never enumerates it and exact-reads the checkpoint range"; + EXPECT_TRUE(refs.contains("b")); +} + +/// Hiding the checkpoint base snapshot from LIST cannot matter: the checkpoint names it, recovery +/// exact-reads its matching non-seal log first, then exact-reads the snapshot. +TEST(CASRefRecoveryCasWalk, CkptNamedBaseIsRecoveredWithoutStreamList) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/hint_snap"}; + + const RefTxnId base{1, 1}; + seedTxn(*backend, layout, ns, base, "a", /*birth=*/true); + writeRefSnapshotRaw(*backend, layout, + minimalLiveSnapshot(ns.string(), base, {committedRow("a", manifestRef(1, 1, 1))})); + seedTxn(*backend, layout, ns, RefTxnId{1, 2}, "c", /*birth=*/false); + seedCkpt(*backend, layout, ns, RefCkpt{.life_epoch = std::optional{1}, + .committed_through = base, + .checkpoint_snapshot_id = base, + .last_epoch_seal = std::nullopt}); + backend->hidden_keys.insert(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), base)); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + backend->resetCounts(); + + const auto refs = store->listRefs(ns); + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns))), 0u); + EXPECT_EQ(refs.size(), 2u) << "the checkpoint names the base; the listing's omission is irrelevant"; + EXPECT_TRUE(refs.contains("a")) << "'a' exists inside the checkpoint-named snapshot"; + EXPECT_TRUE(refs.contains("c")); +} + +TEST(CASRefRecoveryCasWalk, MissingExactIdAtOrBelowCommittedFrontierIsCorruption) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/missing_below_frontier"}; + const RefTxnId frontier{1, 2}; + + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = frontier, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + const auto ckpt_before = readCkpt(*backend, layout, life); + ASSERT_TRUE(ckpt_before); + + auto store = openWalkPool(backend); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); + + const auto ckpt_after = readCkpt(*backend, layout, life); + ASSERT_TRUE(ckpt_after); + EXPECT_EQ(ckpt_after->token, ckpt_before->token) + << "an unchanged checkpoint makes the missing committed id corruption, not a shorter stream"; +} + +TEST(CASRefRecoveryCasWalk, UncommittedSnapshotIsUnobservedWithoutStreamList) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/hint_above_frontier"}; + const RefTxnId frontier{1, 1}; + const RefTxnId uncommitted_snapshot_id{1, 2}; + + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + seedTxn(*backend, layout, ns, frontier, "committed", /*birth=*/true); + writeRefSnapshotRaw(*backend, layout, + minimalLiveSnapshot(ns.string(), uncommitted_snapshot_id, + {committedRow("laundered", manifestRef(1, 2, 1))})); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = frontier, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + auto store = openWalkPool(backend); + backend->resetCounts(); + const auto refs = store->listRefs(ns); + + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(life)), 0u); + EXPECT_TRUE(refs.contains("committed")); + EXPECT_FALSE(refs.contains("laundered")) + << "a physical snapshot not named by `_ckpt` cannot raise the recovered cut"; +} + +TEST(CASRefRecoveryCasWalk, ListingShapeDoesNotAffectCheckpointRecovery) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/list_equivalence"}; + const RefTxnId base{1, 1}; + const RefTxnId frontier{1, 2}; + auto seed = std::make_shared(); + + DB::Cas::tests::fixture::admitLive(*seed, layout, ns); + seedTxn(*seed, layout, ns, base, "a", /*birth=*/true); + writeRefSnapshotRaw(*seed, layout, + minimalLiveSnapshot(ns.string(), base, {committedRow("a", manifestRef(1, 1, 1))})); + seedTxn(*seed, layout, ns, frontier, "b", /*birth=*/false); + seedCkpt(*seed, layout, ns, RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = frontier, + .checkpoint_snapshot_id = base, + .last_epoch_seal = std::nullopt}); + const NamespaceLifeId life = catalogLife(*seed, layout, ns); + + const auto clone_seed = [&]() -> std::shared_ptr + { + /// A clone starts empty: constructing the normal fixture would pre-seed independent pool-meta + /// bytes before this loop could copy the source's identical durable image. + auto backend = std::make_shared(/*seed_pool_meta=*/false); + String cursor; + do + { + const ListPage page = seed->list("", cursor, 1000); + for (const ListedKey & listed : page.keys) + { + const auto object = seed->get(listed.key); + if (!object) + throw std::runtime_error("seed LIST returned a key that exact GET could not read"); + const auto existing = backend->get(listed.key); + if (existing) + { + if (existing->bytes != object->bytes || existing->attributes != object->attributes) + throw std::runtime_error("clone backend constructor disagreed with seeded object"); + } + else if (backend->putIfAbsent(listed.key, object->bytes, object->attributes).outcome != PutOutcome::Done) + throw std::runtime_error("clone backend failed to copy seeded object"); + } + cursor = page.next_cursor; + } while (!cursor.empty()); + return backend; + }; + const auto recover = [&](const std::shared_ptr & backend) + { + auto store = openWalkPool(backend); + backend->resetCounts(); + const auto refs = store->listRefs(ns); + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(life)), 0u); + return refs; + }; + + const auto full = recover(clone_seed()); + auto empty_backend = clone_seed(); + empty_backend->hidden_keys.insert(layout.refSnapshotKey(life, base)); + empty_backend->hidden_keys.insert(layout.refLogKey(life, frontier)); + const auto empty = recover(empty_backend); + ASSERT_EQ(full.size(), empty.size()); + for (const auto & [name, resolved] : full) + { + ASSERT_TRUE(empty.contains(name)); + EXPECT_EQ(resolved.manifest_id, empty.at(name).manifest_id); + EXPECT_EQ(resolved.manifest_size, empty.at(name).manifest_size); + EXPECT_EQ(resolved.published_at_ms, empty.at(name).published_at_ms); + } + EXPECT_TRUE(empty.contains("a")); + EXPECT_TRUE(empty.contains("b")); +} + +TEST(CASRefRecoveryCasWalk, PhantomListedSnapshotIsUnobserved) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/stale_snapshot_hint"}; + const RefTxnId checkpoint_base{1, 1}; + const RefTxnId frontier{1, 2}; + + seedTxn(*backend, layout, ns, checkpoint_base, "a", /*birth=*/true); + writeRefSnapshotRaw(*backend, layout, + minimalLiveSnapshot(ns.string(), checkpoint_base, {committedRow("a", manifestRef(1, 1, 1))})); + seedTxn(*backend, layout, ns, frontier, "b", /*birth=*/false); + seedCkpt(*backend, layout, ns, RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = frontier, + .checkpoint_snapshot_id = checkpoint_base, + .last_epoch_seal = std::nullopt}); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + backend->phantom_list_keys.insert(layout.refSnapshotKey(life, frontier)); + + auto store = openWalkPool(backend); + backend->resetCounts(); + const auto refs = store->listRefs(ns); + + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(life)), 0u); + EXPECT_TRUE(refs.contains("a")); + EXPECT_TRUE(refs.contains("b")); +} + +TEST(CASRefRecoveryCasWalk, ListedFPlusTwoWithoutFPlusOneIsInertUncommittedDebris) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/listed_uncommitted_debris"}; + const RefTxnId frontier{1, 1}; + + seedTxn(*backend, layout, ns, frontier, "a", /*birth=*/true); + seedTxn(*backend, layout, ns, RefTxnId{1, 3}, "debris", /*birth=*/false); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, frontier)); + + auto store = openWalkPool(backend); + backend->resetCounts(); + const auto refs = store->listRefs(ns); + + EXPECT_EQ(backend->listCount(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns))), 0u); + EXPECT_TRUE(refs.contains("a")); + EXPECT_FALSE(refs.contains("debris")); +} + +TEST(CASRefRecoveryCasWalk, DuplicateCatalogLifeIsCorruptionBeforeColdRuntimeAdmission) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/ambiguous_life_a"}; + auto store = openWalkPool(backend); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + const CasRefCatalog::Snapshot sampled = CasRefCatalog::read(*backend, layout); + ASSERT_EQ(sampled.catalog.entries.size(), 1u); + RefCatalog ambiguous = sampled.catalog; + ambiguous.entries.push_back(CatalogEntry{ + .ns = RootNamespace{"srv1/ambiguous_life_b"}, + .state = NsState::Live, + .incarnation = life.incarnation}); + std::sort(ambiguous.entries.begin(), ambiguous.entries.end(), + [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); + ASSERT_EQ(backend->casPut(layout.refCatalogKey(), encodeRefCatalog(ambiguous), sampled.token).outcome, + CasOutcome::Committed); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + auto cold_store = openWalkPool(backend); + (void)cold_store->listRefs(ns); + }); +} + +TEST(CASRefRecoveryCasWalk, CheckpointAdvanceAfterLastLogProbeRestartsBeforeInstall) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/final_authority_validation"}; + const RefTxnId initial_frontier{1, 1}; + const RefTxnId concurrent_frontier{1, 2}; + + seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + backend->watched_key = layout.refLogKey(life, concurrent_frontier); + backend->after_get = [&] + { + seedTxn(*backend, layout, ns, concurrent_frontier, "b", /*birth=*/false); + const auto sampled = readCkpt(*backend, layout, life); + ASSERT_TRUE(sampled); + RefCkpt advanced = sampled->ckpt; + advanced.committed_through = concurrent_frontier; + ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(advanced), sampled->token).outcome, + CasOutcome::Committed); + }; + + auto store = openWalkPool(backend); + const uint64_t restarts_before = store->refRecoveryRestartsForTest(ns); + const auto refs = store->listRefs(ns); + + EXPECT_TRUE(refs.contains("a")); + EXPECT_TRUE(refs.contains("b")) + << "the old private cut must be discarded when final exact authority moved after its last probe"; + EXPECT_GT(store->refRecoveryRestartsForTest(ns), restarts_before) + << "the final authority observation is recovery's linearization point"; +} + +TEST(CASRefRecoveryCasWalk, LiveCatalogLifeWithoutReadableCheckpointIsCorruption) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/live_without_ckpt"}; + + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + seedTxn(*backend, layout, ns, RefTxnId{7, 1}, "hint-must-not-be-genesis", /*birth=*/true); + ASSERT_FALSE(readCkpt(*backend, layout, life)); + + auto store = openWalkPool(backend); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); +} + +/// A 404 BELOW the exact committed frontier is not the end of the stream -- it is a HOLE, and a hole +/// in a dense stream is corruption. Recovery exact-reads the checkpoint token once to distinguish a +/// concurrently moved cut from durable-data loss, then FAILS CLOSED while that token is unchanged. It +/// must never fold what it has: that is precisely how an acknowledged transaction disappears. +TEST(CASRefRecoveryCasWalk, AbsentIdBelowADurableHigherIdFailsClosed) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/hole"}; + + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 3})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + /// {1,2} is MISSING while {1,3} is durable and listed: the listing itself witnesses the hole. + seedTxn(*backend, layout, ns, RefTxnId{1, 3}, "c", /*birth=*/false); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + store->setCasRetrySleepForTest([](uint64_t) {}); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); +} + +/// --------------------------------------------------------------------------------------------- +/// The CAS-walk: closing dead epochs in-band +/// --------------------------------------------------------------------------------------------- + +/// The ordinary case: one dead epoch, closed by OUR seal at `{E, T+1}` -- the exact key a dying +/// predecessor's in-flight PUT would have taken, which is what makes the store's conditional create the +/// fence (INV-2) rather than a detector after the fact. +TEST(CASRefRecoveryCasWalk, DeadEpochIsClosedByOurOwnSealAtTPlusOne) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/seal_created"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 2u); + + const uint64_t sealed_before = counterOf(ProfileEvents::CASRefRecoveryEpochSealed); + ASSERT_EQ(store->listRefs(ns).size(), 1u); + EXPECT_EQ(counterOf(ProfileEvents::CASRefRecoveryEpochSealed), sealed_before + 1); + + const RefTxnId seal_id{1, 2}; + const auto seal = readLogTxn(*backend, layout, ns, seal_id); + ASSERT_TRUE(seal.has_value()) << "epoch 1 is dead and must be closed at {1,2}"; + EXPECT_TRUE(refLogTxnIsEpochSeal(*seal)); + EXPECT_EQ(seal->prev_epoch_seal, std::nullopt) << "sequence 2 never carries a chain link"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::optional(seal_id)) + << "the chain link the next epoch's sequence-1 transaction must name"; +} + +/// A concurrent recoverer got there first. Its seal is already at `{E, T+1}`, so our conditional create +/// loses -- and the right reaction is to ADOPT it, not to treat a peer's correct write as interference. +/// The adopted seal is the same chain link ours would have been. +TEST(CASRefRecoveryCasWalk, ConcurrentRecoverersSealIsAdoptedNotContested) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/seal_adopt"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + /// The peer's seal lands between our read of {1,2} and our create of it, so we meet it as an + /// OCCUPANT rather than as a tail entry. + backend->late_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2}); + backend->late_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(makeSealTxn(ns, RefTxnId{1, 2}))); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + const uint64_t adopted_before = counterOf(ProfileEvents::CASRefRecoveryEpochSealAdopted); + ASSERT_EQ(store->listRefs(ns).size(), 1u); + EXPECT_GT(counterOf(ProfileEvents::CASRefRecoveryEpochSealAdopted), adopted_before); + EXPECT_EQ(store->lastEpochSealForTest(ns), std::optional(RefTxnId{1, 2})) + << "an adopted seal is this namespace's chain link exactly as a minted one is"; +} + +/// A STRAGGLER: an ordinary transaction of the dead epoch landed at `{E, T+1}` after our read of the +/// tail and before our seal. The rule is state-derived ids (INV-2): adopt the transaction, advance `T` +/// by ONE, and re-seal at the NEW `T+1`. Never mint `T+2` around it -- that writes a hole into the +/// durable stream that no later reader can tell from a lost object. +TEST(CASRefRecoveryCasWalk, StragglerAtTPlusOneIsAdoptedAndResealedAtTheNewTPlusOne) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/straggler"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + /// The dying epoch's last append materializes between our read of {1,2} and our create of it -- the + /// straggler, arriving exactly where the every-attempt rule says it can. + backend->late_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2}); + backend->late_bytes = sealObject(FormatId::RefLog, + encodeRefLogTxn(makeOrdinaryTxn(ns, RefTxnId{1, 2}, "late", /*birth=*/false))); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + const uint64_t straggler_before = counterOf(ProfileEvents::CASRefRecoveryStragglerAdopted); + const auto refs = store->listRefs(ns); + EXPECT_EQ(refs.size(), 2u) << "the straggler's transaction is durable and must be applied, not skipped"; + EXPECT_TRUE(refs.contains("late")); + EXPECT_GT(counterOf(ProfileEvents::CASRefRecoveryStragglerAdopted), straggler_before); + + const auto seal = readLogTxn(*backend, layout, ns, RefTxnId{1, 3}); + ASSERT_TRUE(seal.has_value()) << "the epoch must be re-sealed at the NEW T+1 = {1,3}, never at a blindly minted T+2"; + EXPECT_TRUE(refLogTxnIsEpochSeal(*seal)); + EXPECT_EQ(store->lastEpochSealForTest(ns), std::optional(RefTxnId{1, 3})); +} + +TEST(CASRefRecoveryCasWalk, RecoveryPublishesEveryOccupiedObjectBeforeAdvancingPastIt) +{ + struct Case + { + String suffix; + uint64_t live_epoch; + RefTxnId occupant; + RefTxnId forbidden_successor; + bool occupant_is_seal; + }; + const std::vector cases{ + {"seal", 3, {1, 2}, {2, 1}, true}, + {"straggler", 2, {1, 2}, {1, 3}, false}, + }; + + for (const Case & test_case : cases) + { + SCOPED_TRACE(test_case.suffix); + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/occupied_frontier_" + test_case.suffix}; + const RefTxnId initial_frontier{1, 1}; + + burnEpochsUpTo(*backend, layout, test_case.live_epoch); + seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + backend->late_key = layout.refLogKey(life, test_case.occupant); + const RefLogTxn occupant = test_case.occupant_is_seal + ? makeSealTxn(ns, test_case.occupant) + : makeOrdinaryTxn(ns, test_case.occupant, "late", /*birth=*/false); + backend->late_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(occupant)); + + uint64_t fake_now = 1'000'000; + PoolConfig config = walkTestConfig(); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget.recovery_retry_budget_ms = 1; + config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; + config.cas_request_budget.recovery_retry_max_backoff_ms = 1; + auto store = openWalkPool(backend, config); + store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + + backend->ambiguous_cas_substr = layout.refCkptKey(life); + backend->ambiguous_cas_count = 100'000; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); + + EXPECT_TRUE(backend->get(layout.refLogKey(life, test_case.occupant))); + EXPECT_FALSE(backend->get(layout.refLogKey(life, test_case.forbidden_successor))) + << "recovery advanced before exact _ckpt certified the occupied object"; + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)); + } +} + +/// Two BURNED epochs -- mounted, never written to, and abandoned. `CasPool`'s epoch allocator mints and +/// never reclaims, so this is the normal shape of a pool that has restarted a few times, not an +/// anomaly. Each empty epoch is closed by its own sequence-1 seal, and each carries the previous seal as +/// its `prev_epoch_seal`: the chain is what makes a MISSING epoch detectable, which arithmetic within an +/// epoch cannot do. +TEST(CASRefRecoveryCasWalk, TwoBurnedEmptyEpochsProduceTwoChainedSequenceOneSeals) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/burned"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/4); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 4u); + + ASSERT_EQ(store->listRefs(ns).size(), 1u); + + const auto seal1 = readLogTxn(*backend, layout, ns, RefTxnId{1, 2}); + ASSERT_TRUE(seal1.has_value()) << "epoch 1 closes at {1,2}"; + EXPECT_EQ(seal1->prev_epoch_seal, std::nullopt); + + const auto seal2 = readLogTxn(*backend, layout, ns, RefTxnId{2, 1}); + ASSERT_TRUE(seal2.has_value()) << "empty epoch 2 still closes -- at its sequence 1"; + EXPECT_TRUE(refLogTxnIsEpochSeal(*seal2)); + EXPECT_EQ(seal2->prev_epoch_seal, std::optional(RefTxnId{1, 2})) + << "a sequence-1 seal MUST name the seal that closed the previous epoch"; + + const auto seal3 = readLogTxn(*backend, layout, ns, RefTxnId{3, 1}); + ASSERT_TRUE(seal3.has_value()) << "empty epoch 3 closes too"; + EXPECT_EQ(seal3->prev_epoch_seal, std::optional(RefTxnId{2, 1})); + + EXPECT_FALSE(readLogTxn(*backend, layout, ns, RefTxnId{4, 1}).has_value()) + << "epoch 4 is LIVE -- sealing it would close the epoch this mount writes in"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::optional(RefTxnId{3, 1})); +} + +TEST(CASRefRecoveryCasWalk, RecoveryPublishesEachCreatedSealBeforeCreatingTheNextEpochSeal) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/seal_frontier_before_next"}; + const RefTxnId initial_frontier{1, 1}; + const RefTxnId first_seal{1, 2}; + const RefTxnId second_seal{2, 1}; + const RefTxnId cold_remount_frontier{3, 1}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/3); + seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + + uint64_t fake_now = 1'000'000; + PoolConfig config = walkTestConfig(); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget.recovery_retry_budget_ms = 1; + config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; + config.cas_request_budget.recovery_retry_max_backoff_ms = 1; + auto store = openWalkPool(backend, config); + ASSERT_EQ(store->liveWriterEpoch(), 3u); + store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + + backend->ambiguous_cas_substr = layout.refCkptKey(life); + backend->ambiguous_cas_count = 100'000; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); + + EXPECT_TRUE(backend->get(layout.refLogKey(life, first_seal))) + << "the first recovery seal became durable before its frontier attempt"; + EXPECT_FALSE(backend->get(layout.refLogKey(life, second_seal))) + << "recovery may not create a second object while the first is still above exact _ckpt"; + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)); + + /// Restart cold, without the failed mount's `NeedsRecovery` attempt. The first seal is durable but + /// still outside `_ckpt`; the remount must recover and certify it before it may create `{2,1}`. + backend->ambiguous_cas_count = 0; + store.reset(); + auto cold_store = openWalkPool(backend); + ASSERT_EQ(cold_store->liveWriterEpoch(), 4u); + ASSERT_EQ(cold_store->listRefs(ns).size(), 1u); + EXPECT_TRUE(backend->get(layout.refLogKey(life, second_seal))); + EXPECT_TRUE(backend->get(layout.refLogKey(life, cold_remount_frontier))); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, cold_remount_frontier); +} + +/// A straggler is not an exception to the recovered-successor rule. When it materializes in the seal +/// slot, recovery adopts it as the one object above its accepted checkpoint and must certify that exact +/// frontier before it can create the following seal at the new `T+1`. +TEST(CASRefRecoveryCasWalk, RecoveryPublishesAnAdoptedStragglerBeforeCreatingItsFollowingSeal) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/straggler_frontier_before_seal"}; + const RefTxnId initial_frontier{1, 1}; + const RefTxnId straggler{1, 2}; + const RefTxnId following_seal{1, 3}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + backend->late_key = layout.refLogKey(life, straggler); + backend->late_bytes = sealObject(FormatId::RefLog, + encodeRefLogTxn(makeOrdinaryTxn(ns, straggler, "late", /*birth=*/false))); + + uint64_t fake_now = 1'000'000; + PoolConfig config = walkTestConfig(); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget.recovery_retry_budget_ms = 1; + config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; + config.cas_request_budget.recovery_retry_max_backoff_ms = 1; + auto store = openWalkPool(backend, config); + store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + + backend->ambiguous_cas_substr = layout.refCkptKey(life); + backend->ambiguous_cas_count = 100'000; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); + + EXPECT_TRUE(backend->get(layout.refLogKey(life, straggler))) + << "the straggler occupied the recovery seal slot"; + EXPECT_FALSE(backend->get(layout.refLogKey(life, following_seal))) + << "recovery may not create a seal after an adopted straggler above exact _ckpt"; + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)); + + backend->ambiguous_cas_count = 0; + ASSERT_EQ(store->listRefs(ns).size(), 2u); + EXPECT_TRUE(backend->get(layout.refLogKey(life, following_seal))); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, following_seal); +} + +/// GENESIS. A namespace born at epoch 5 has no epochs 1-4 of its own: they are not "empty epochs it +/// failed to close", they are epochs before it existed. The walk starts at the namespace's `life_epoch` +/// and writes no phantom seals below it, and with no transition ever having happened it installs NO +/// chain link -- `nullopt` means genesis and must mean it exactly, or the table's first transaction +/// would be required to name a seal that never existed. +TEST(CASRefRecoveryCasWalk, GenesisAtEpochFiveWritesNoPhantomSealsBelowLifeEpoch) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/genesis5"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/5); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(5, RefTxnId{5, 1})); + seedTxn(*backend, layout, ns, RefTxnId{5, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 5u); + + ASSERT_EQ(store->listRefs(ns).size(), 1u); + + for (uint64_t e = 1; e <= 4; ++e) + EXPECT_FALSE(readLogTxn(*backend, layout, ns, RefTxnId{e, 1}).has_value()) + << "no seal may be written for epoch " << e << ", which predates this namespace"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) + << "no transition ever happened for this namespace: nullopt means GENESIS and must mean it exactly"; +} + +/// --------------------------------------------------------------------------------------------- +/// The trio: ONE captured generation, three checks +/// --------------------------------------------------------------------------------------------- + +/// The GENERIC mid-walk bump: the fence moves while recovery is doing I/O, so the incarnation that +/// admitted this work is gone. Nothing may be installed -- the recovered view belongs to a mount that no +/// longer owns the namespace. The table stays unrecovered, and a retry under the CURRENT generation +/// succeeds, which is what makes this a refusal rather than a wedge. +TEST(CASRefRecoveryCasWalk, FenceBumpedMidWalkRefusesTheInstallAndTheRetrySucceeds) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/bump_midwalk"}; + + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + std::atomic bumped{false}; + backend->watched_substr = "_log/"; + backend->on_key = [&](const String &) + { + if (!bumped.exchange(true)) + rearmMountFenceAfterAnomalyForTest(store); + }; + + EXPECT_ANY_THROW(store->listRefs(ns)) << "a recovery whose I/O window straddled a fence bump must install nothing"; + + backend->on_key = nullptr; + fenceOutMountForRemount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()) + << "a generation bump cannot rebind the captured runtime; the production remount must publish " + "a distinct runtime at the accepted generation"; + EXPECT_EQ(store->listRefs(ns).size(), 1u) << "the retry through the remounted runtime succeeds"; +} + +TEST(CASRefRecoveryCasWalk, RetiredLifePausedInRealRecoveryIoWritesAndInstallsNothing) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery-retired-mid-io"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "predecessor", /*birth=*/true); + const CatalogEntry predecessor = CasRefCatalog::read(*backend, layout).catalog.entries.front(); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(predecessor.ns, predecessor.incarnation); + const auto predecessor_ckpt_before = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_before); + + auto store = openWalkPool(backend); + ASSERT_EQ(store->liveWriterEpoch(), 2u); + const uint64_t recovery_installs_before = store->recoveryInstallCountForTest(); + + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + backend->watched_substr = "_log/"; + backend->on_key = [&](const String &) + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }; + + std::exception_ptr recovery_error; + std::thread recovery([&] + { + try + { + (void)store->listRefs(ns); + } + catch (...) + { + recovery_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + const CatalogEntry successor = replaceCatalogLifeForTest(*backend, layout, predecessor, UInt128{0x5152}); + const NamespaceLifeId successor_life + = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(lifeEpochCkpt(2))).outcome, + PutOutcome::Done); + const auto successor_ckpt_before = backend->get(layout.refCkptKey(successor_life)); + ASSERT_TRUE(successor_ckpt_before); + store->invalidateRemovedCatalogLife(predecessor_life); + backend->resetCounts(); + + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + recovery.join(); + + EXPECT_TRUE(recovery_error) << "the predecessor recovery must be refused, not exposed"; + EXPECT_EQ(backend->putCount(layout.refLogKey(predecessor_life, RefTxnId{1, 2})), 0u) + << "no predecessor seal retry may be sent after exact retirement"; + EXPECT_EQ(backend->casPutCount(layout.refCkptKey(predecessor_life)), 0u) + << "no predecessor checkpoint CAS may be sent after exact retirement"; + const auto predecessor_ckpt_after = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_after); + EXPECT_EQ(predecessor_ckpt_after->token, predecessor_ckpt_before->token); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "the detached predecessor result was installed"; + EXPECT_EQ(store->recoveryInstallCountForTest(), recovery_installs_before) + << "the detached predecessor reached the recovery publication point"; + const String successor_prefix = layout.namespaceStreamPrefix(successor_life); + for (const String & key : backend->touchedKeys()) + EXPECT_EQ(key.find(successor_prefix), String::npos) + << "predecessor recovery retargeted storage I/O into successor key " << key; + const auto successor_ckpt_after = backend->get(layout.refCkptKey(successor_life)); + ASSERT_TRUE(successor_ckpt_after); + EXPECT_EQ(successor_ckpt_after->token, successor_ckpt_before->token); + EXPECT_EQ(successor_ckpt_after->bytes, successor_ckpt_before->bytes); +} + +/// Bump point 1 of the trio's two interior seams: AFTER the slot-occupy landed, BEFORE the `_ckpt` CAS. +/// The seal is durable (it was written under a generation that was still valid), but the checkpoint must +/// NOT advance and nothing may be installed. This is the seam a single "check the fence at entry" would +/// miss entirely. +TEST(CASRefRecoveryCasWalk, FenceBumpedAfterSlotOccupyBeforeCkptCasAdvancesNoCheckpoint) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/bump_after_seal"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + const auto ckpt_before = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + ASSERT_TRUE(ckpt_before.has_value()); + + backend->watched_substr = "_log/"; + backend->on_key = [&] { rearmMountFenceAfterAnomalyForTest(store); }; + + EXPECT_ANY_THROW(store->listRefs(ns)); + + const auto ckpt_after = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->ckpt.last_epoch_seal, std::nullopt) + << "the seal is durable but the checkpoint must not record it under a generation that moved"; + EXPECT_EQ(ckpt_after->token, ckpt_before->token) << "no CAS was sent at all"; +} + +/// Bump point 2: AFTER the `_ckpt` CAS, BEFORE the install. The checkpoint advance is harmless (the +/// merge is a semantic maximum, so the retry re-derives the same or a greater value), but the STATE must +/// not be published: this runtime's view belongs to a dead incarnation. Today there is no such recheck +/// at all -- that gap is the whole reason this test exists. +TEST(CASRefRecoveryCasWalk, FenceBumpedAfterCkptCasBeforeInstallPublishesNoState) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/bump_after_ckpt"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + /// The `_ckpt` CAS is the LAST write recovery performs, so hooking it fires strictly between the + /// checkpoint advance and the install recheck. + backend->watched_substr = "/_ckpt"; + backend->on_key = [&] { rearmMountFenceAfterAnomalyForTest(store); }; + + EXPECT_ANY_THROW(store->listRefs(ns)) << "the install recheck must refuse a result from a moved generation"; + + const auto ckpt_after = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->ckpt.last_epoch_seal, std::optional(RefTxnId{1, 2})) + << "the checkpoint advance already landed and is harmless -- the merge is a semantic maximum"; + + backend->on_key = nullptr; + fenceOutMountForRemount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()) + << "a generation bump cannot rebind the captured runtime; the production remount must publish " + "a distinct runtime at the accepted generation"; + EXPECT_EQ(store->listRefs(ns).size(), 1u) << "the retry through the remounted runtime installs normally"; +} + +/// --------------------------------------------------------------------------------------------- +/// The self-remount barrier +/// --------------------------------------------------------------------------------------------- + +/// Spec §3: "self-remount cancels or waits out recovery before rearming." The install recheck alone is +/// not that rule -- it protects the install, not the WINDOW. A recovery paused in its I/O while the +/// fence is re-armed would still be holding an admitted generation that is about to be superseded, and +/// the barrier is what guarantees no `_ckpt` CAS and no install can follow the re-arm. +/// +/// Driven at a real I/O seam: recovery blocks inside a `get`, the remount barrier is invoked from +/// another thread and must BLOCK, the recovery is released, acknowledges the cancellation, and only then +/// does the barrier return. +TEST(CASRefRecoveryCasWalk, RemountBarrierBlocksUntilAPausedRecoveryAcknowledgesCancellation) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/remount_barrier"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + std::mutex m; + std::condition_variable cv; + bool recovery_parked = false; + bool release_recovery = false; + + backend->watched_substr = "_log/"; + /// `GetSeamBackend` moves the hook out before calling it, so this parks exactly once without the + /// hook having to clear itself -- see its `get` for why self-clearing is a use-after-free. + backend->on_key = [&](const String &) + { + std::unique_lock lock(m); + recovery_parked = true; + cv.notify_all(); + cv.wait(lock, [&] { return release_recovery; }); + }; + + const uint64_t ckpt_before = counterOf(ProfileEvents::CASRefCheckpointPublished); + const uint64_t cancelled_before = counterOf(ProfileEvents::CASRefRecoveryCancelled); + + std::thread recovery([&] { try { store->listRefs(ns); } catch (...) {} }); // NOLINT(bugprone-empty-catch): the outcome is asserted below via the ProfileEvents counters, not this thread's exception + + { + std::unique_lock lock(m); + cv.wait(lock, [&] { return recovery_parked; }); + } + + std::atomic barrier_returned{false}; + std::thread barrier([&] + { + store->cancelRefRecoveriesAndAwaitQuiescence(); + barrier_returned.store(true); + }); + + /// Wait for the barrier's REQUEST to be visible before touching anything else. Releasing the parked + /// recovery any earlier would race it past a flag set a moment too late, and the test would observe + /// an ordinary completion and call it a missing cancellation. + while (!store->refRecoveryCancelRequestedForTest(ns)) + std::this_thread::yield(); + + /// The request is published and the recovery is still parked, so the barrier is now provably inside + /// its wait. It must not have returned: fence re-arm may not proceed while a recovery is in flight. + EXPECT_FALSE(barrier_returned.load()); + + { + std::lock_guard lock(m); + release_recovery = true; + } + cv.notify_all(); + + barrier.join(); + recovery.join(); + EXPECT_TRUE(barrier_returned.load()); + + EXPECT_GT(counterOf(ProfileEvents::CASRefRecoveryCancelled), cancelled_before) + << "the released recovery must observe the cancellation rather than run to completion"; + EXPECT_EQ(counterOf(ProfileEvents::CASRefCheckpointPublished), ckpt_before) + << "a cancelled recovery performs ZERO _ckpt CASes"; + EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "and ZERO installs"; +} + +/// A `NeedsRecovery` lane replays the known-durable transaction before returning to `Ready`. +TEST(CASRefRecoveryCasWalk, NeedsRecoveryReplaysTheStrandedTxn) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/poisoned"}; + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + /// Publish one ref through the real lane, then fail the next commit's install region: the + /// transaction is durable and the install that would have recorded it throws. + ASSERT_NO_THROW(store->appendRefOps(ns, MutationScope::ref("a"), + [](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps("a", manifestRef(1, 1, 1))) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish)); + + /// Built outside the region. One-shot, and re-allowing allocations for + /// the duration of the throw: `std::rethrow_exception` allocates through libc++'s + /// `__cxa_rethrow_primary_exception`, which the debug build's `DENY_ALLOCATIONS_IN_SCOPE` aborts on. + /// (Found by the debug gate -- the first cut of this probe took the whole binary down there.) Same + /// shape as `gtest_cas_ref_install_safety.cpp`'s `armOneShotInstallFailure`. + auto planned_failure = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "install probe")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned_failure, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned_failure); + }); + EXPECT_ANY_THROW(store->appendRefOps(ns, MutationScope::ref("b"), + [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, + RootMutationOrigin::Writer, RootMutationKind::Publish)); + store->setInstallRegionProbeForTest(nullptr); + + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + /// "Durable but not applied here", stated as the two facts it is made of: the object IS in the + /// store, and this runtime's floor is what keeps the allocator off its id. `ns` was born through + /// the REAL production lane (`appendRefOps`), not the raw `seedTxn`/`casAdmitEntry` fixtures this + /// file's OTHER tests use -- so its ref-layer objects sit at a REAL, catalog-minted incarnation, + /// not the Stage-A sentinel `readLogTxn` assumes. Resolved here rather than through `readLogTxn`. + { + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + const CatalogEntry * entry = nullptr; + for (const CatalogEntry & e : snap.catalog.entries) + if (e.ns.string() == ns.string()) + entry = &e; + ASSERT_NE(entry, nullptr) << "the birth above must have minted a catalog entry for " << ns.string(); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 2})).has_value()) + << "the stranded transaction must be durable -- otherwise recovery is not owed"; + } + + /// The next touch drives recovery again -- this is the structural closure Task 3 deferred here. + const auto refs = store->listRefs(ns); + EXPECT_EQ(refs.size(), 2u); + EXPECT_TRUE(refs.contains("b")) << "the walk re-derived the stranded transaction from the durable log"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "only a completed recovery install returns the lane to Ready"; +} + +TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsOneExactUnfrontieredSuccessorAndPublishesItsFrontier) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_one_successor"}; + auto store = openWalkPool(backend); + + ASSERT_NO_THROW(store->appendRefOps(ns, MutationScope::ref("a"), + [](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps("a", manifestRef(1, 1, 1))) + ops.push_back(op); + return ops; + }, RootMutationOrigin::Writer, RootMutationKind::Publish)); + + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const String ckpt_key = layout.refCkptKey(life); + ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + backend->ambiguous_cas_substr = ckpt_key; + backend->ambiguous_cas_count = 200; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->appendRefOps(ns, MutationScope::ref("b"), + [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 2}))) + << "the sole deterministic successor must be durable before recovery"; + ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + + backend->ambiguous_cas_count = 0; + const auto refs = store->listRefs(ns); + + EXPECT_TRUE(refs.contains("b")); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 2})) + << "the successor is not installable until the current admitted fence publishes its frontier"; +} + +TEST(CASRefRecoveryCasWalk, ColdWriterRecoveryPublishesOneExactUnfrontieredSuccessorBeforeSealing) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_cold_successor"}; + auto store = openWalkPool(backend); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + store.reset(); + std::vector checkpoint_cas_bodies; + backend->before_cas_put = [&](const String & key, const String & bytes, const std::optional &) + { + if (key == layout.refCkptKey(life)) + checkpoint_cas_bodies.push_back(decodeRefCkpt(bytes)); + }; + + /// A remount/process restart has no in-memory `RefAppendAttempt`; the writer recovery entry point + /// still owns its one exact F+1 adoption duty from the durable checkpoint and log alone. + auto cold_store = openWalkPool(backend); + const auto refs = cold_store->listRefs(ns); + + EXPECT_TRUE(refs.contains("b")); + EXPECT_EQ(cold_store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_TRUE(std::any_of(checkpoint_cas_bodies.begin(), checkpoint_cas_bodies.end(), + [](const RefCkpt & ckpt) { return ckpt.committed_through == std::make_optional(RefTxnId{1, 2}); })) + << "the exact F+1 frontier must publish before the remount seals its dead epoch"; + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 3})); +} + +TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsFirstCommittedTxnAboveLifeEpochOnlyCheckpoint) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_first_unfrontiered"}; + auto store = openWalkPool(backend); + + /// Create the catalog life explicitly, then retain exactly the checkpoint fragment published by + /// production birth before its first log. This makes `{1,1}` the first durable transaction above a + /// readable checkpoint whose `committed_through` is absent. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(lifeEpochCkpt(1))).outcome, + PutOutcome::Done); + ASSERT_TRUE(readCkpt(*backend, layout, life)->ckpt.life_epoch); + ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, std::nullopt); + + backend->ambiguous_cas_substr = layout.refCkptKey(life); + backend->ambiguous_cas_count = 200; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->appendRefOps(ns, MutationScope::ref("a"), + [](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps("a", manifestRef(1, 1, 1))) + ops.push_back(op); + return ops; + }, RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 1}))); + ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, std::nullopt); + ASSERT_FALSE(backend->get(layout.refSnapshotKey(life, RefTxnId{1, 1}))) + << "the grounding test must exercise the exact log successor, not a hinted snapshot"; + + backend->ambiguous_cas_count = 0; + const auto refs = store->listRefs(ns); + + EXPECT_TRUE(refs.contains("a")); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); +} + +TEST(CASRefRecoveryCasWalk, WriterRecoveryRestartsWhenCheckpointAdvancesPastPrivateCandidate) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_checkpoint_moves"}; + auto store = openWalkPool(backend); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const String ckpt_key = layout.refCkptKey(life); + const RefLogTxn later = makeOrdinaryTxn(ns, RefTxnId{1, 3}, "c", /*birth=*/false); + bool injected = false; + + /// Recovery has fetched `{1,2}` and proved `{1,3}` absent. Another admitted writer can then append + /// `{1,3}` and publish its frontier before recovery's own checkpoint CAS. The stale private + /// candidate contains only `b`; it must restart and replay `c`, not accept an `IdenticalSkip` and + /// install below the exact checkpoint it just observed. + backend->before_cas_put = [&](const String & key, const String &, const std::optional & expected) + { + if (injected || key != ckpt_key) + return; + injected = true; + ASSERT_TRUE(expected); + ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, later.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(later))).outcome, + PutOutcome::Done); + const auto current = backend->get(key); + ASSERT_TRUE(current); + ASSERT_EQ(current->token, *expected); + const RefCkpt advanced = mergeCkpt( + decodeRefCkpt(current->bytes), + RefCkpt{.life_epoch = std::nullopt, + .committed_through = later.txn_id, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt}); + ASSERT_EQ(backend->putOverwrite(key, encodeRefCkpt(advanced), current->token).outcome, PutOutcome::Done); + }; + + const auto refs = store->listRefs(ns); + + EXPECT_TRUE(injected); + EXPECT_TRUE(refs.contains("b")); + EXPECT_TRUE(refs.contains("c")) << "recovery must restart from the newer exact frontier"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, later.txn_id); +} + +TEST(CASRefRecoveryCasWalk, WriterRecoveryRejectsTwoUnfrontieredSuccessorsAfterExactCheckpointReread) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_two_successors"}; + auto store = openWalkPool(backend); + + ASSERT_NO_THROW(store->appendRefOps(ns, MutationScope::ref("a"), + [](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps("a", manifestRef(1, 1, 1))) + ops.push_back(op); + return ops; + }, RootMutationOrigin::Writer, RootMutationKind::Publish)); + + const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const String ckpt_key = layout.refCkptKey(life); + backend->ambiguous_cas_substr = ckpt_key; + backend->ambiguous_cas_count = 200; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + store->appendRefOps(ns, MutationScope::ref("b"), + [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + const RefLogTxn second_successor = makeOrdinaryTxn(ns, RefTxnId{1, 3}, "c", /*birth=*/false); + ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, second_successor.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(second_successor))).outcome, + PutOutcome::Done); + backend->ambiguous_cas_count = 0; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})) + << "corruption must not launder either successor into the frontier"; +} + +TEST(CASRefRecoveryCasWalk, WriterRecoveryRejectsDifferentOrdinaryBytesAtTheRetainedSuccessorSlot) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_different_successor"}; + auto store = openWalkPool(backend); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + const String successor_key = layout.refLogKey(life, RefTxnId{1, 2}); + const auto original = backend->get(successor_key); + ASSERT_TRUE(original); + const RefLogTxn different = makeOrdinaryTxn(ns, RefTxnId{1, 2}, "different", /*birth=*/false); + ASSERT_EQ(backend->putOverwrite(successor_key, + sealObject(FormatId::RefLog, encodeRefLogTxn(different)), + original->token).outcome, + PutOutcome::Done); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); +} + +TEST(CASRefRecoveryCasWalk, RetainedOldWriterAttemptLosesConclusiveToASuccessorSeal) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/recovery_successor_seal"}; + auto store = openWalkPool(backend); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + const RefTxnId successor_id{1, 2}; + const String successor_key = layout.refLogKey(life, successor_id); + const auto original = backend->get(successor_key); + ASSERT_TRUE(original); + const RefLogTxn successor_seal = makeSealTxn(ns, successor_id); + ASSERT_EQ(backend->putOverwrite(successor_key, + sealObject(FormatId::RefLog, encodeRefLogTxn(successor_seal)), + original->token).outcome, + PutOutcome::Done); + + const auto refs = store->listRefs(ns); + + EXPECT_FALSE(refs.contains("b")) << "the old writer's retained ordinary bytes lost at the sealed slot"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::make_optional(successor_id)); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, successor_id); + EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.last_epoch_seal, successor_id); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); +} + +/// --------------------------------------------------------------------------------------------- +/// Fail-closed on an unresolved slot +/// --------------------------------------------------------------------------------------------- + +/// `Unresolved` from the slot-occupy means the store will not say whether our seal landed. That is not a +/// state to guess about: recovery takes the transient-retry path and, once its budget is spent, fails +/// closed with the table left unrecovered. Exposing a table whose dead epoch may or may not be closed is +/// the one outcome that must be impossible. +TEST(CASRefRecoveryCasWalk, UnresolvedSealSlotFailsClosedWithoutInstalling) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/unresolved"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + /// The backoff sleep ADVANCES the same fake clock the budget is measured against, so the retry + /// envelope is spent in a handful of iterations instead of spinning against a frozen clock. Not + /// cosmetic: with a frozen clock this test burns ~700k retries and the same number of log lines, + /// which is how a real regression in this arm would become invisible in the noise. + uint64_t fake_now = 1'000'000; + PoolConfig config = walkTestConfig(); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + auto store = openWalkPool(backend, config); + ASSERT_TRUE(store); + + store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + backend->ambiguous_put_substr = "/_log/"; + + EXPECT_ANY_THROW(store->listRefs(ns)); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)) + << "a table whose dead epoch may or may not be closed must never be exposed as recovered"; +} + +/// --------------------------------------------------------------------------------------------- +/// Carried forward from the retired `RefWriterRecoverySeal` suite +/// --------------------------------------------------------------------------------------------- + +/// THE property the whole in-band design exists for, and the one the retired suite could only +/// approximate with a detector: the Late Predecessor PUT is REFUSED, by the store, at the key it wanted. +/// +/// A dying writer of epoch 1 has an append in flight for `{1,2}`. Recovery closes epoch 1 by occupying +/// exactly that slot. When the ghost's conditional create finally reaches the store there is nothing for +/// it to do -- the key is write-once and taken. The old sentinel seal was a SNAPSHOT at a synthetic id, +/// which left `{1,2}` free: the ghost landed, and all anyone could do was notice afterwards. +TEST(CASRefRecoveryCasWalk, ALatePredecessorPutAtTheSealedSlotIsRefusedByTheStore) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/ghost"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + ASSERT_EQ(store->listRefs(ns).size(), 1u); + + /// The ghost: the exact append the dead epoch's writer had in flight, arriving late. + const RefTxnId ghost_id{1, 2}; + const String ghost_bytes = sealObject(FormatId::RefLog, + encodeRefLogTxn(makeOrdinaryTxn(ns, ghost_id, "ghost", /*birth=*/false))); + const PutResult put = backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), ghost_id), ghost_bytes); + EXPECT_EQ(put.outcome, PutOutcome::PreconditionFailed) + << "the seal occupies the ghost's own key, so the store itself is the fence"; + + /// And the object at that key is still the seal, byte for byte -- nothing adopted the ghost. + const auto occupant = readLogTxn(*backend, layout, ns, ghost_id); + ASSERT_TRUE(occupant.has_value()); + EXPECT_TRUE(refLogTxnIsEpochSeal(*occupant)); +} + +/// An occupant at the seal slot that this build cannot decode is NOT a straggler to adopt and NOT a +/// peer's seal to defer to: it is an object at a key this namespace exclusively owns whose meaning is +/// unknown. Recovery fails closed on it -- and, just as importantly, stays RESTARTABLE: the throw must +/// leave `recovery_in_progress` cleared, or the table would be unrecoverable for the mount's life and +/// every later toucher would park forever on a condition variable nobody will signal. +TEST(CASRefRecoveryCasWalk, UndecodableOccupantAtTheSealSlotFailsClosedAndLeavesRecoveryRestartable) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/foreign_slot"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + backend->late_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2}); + backend->late_bytes = "not a ref-log object at all"; + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)); + + /// Restartable: a second touch runs a WHOLE new attempt (it fails the same way, which is the point -- + /// it reaches the failure again rather than hanging on a stuck `recovery_in_progress`). + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); +} + +/// A second caller that arrives while a recovery is mid-walk WAITS for it rather than racing an +/// independent walk of its own. Two concurrent walks would both try to occupy the same seal slot, and +/// while the loser adopts correctly, they would also both replay the whole tail and one would install a +/// state the other's install immediately replaces -- work and I/O for nothing, on the path that is +/// already the most expensive one in the system. +TEST(CASRefRecoveryCasWalk, ASecondCallerWaitsForTheWalkInsteadOfRacingIt) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/serialized"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + std::mutex m; + std::condition_variable cv; + bool parked = false; + bool release = false; + + backend->watched_substr = "_log/"; + backend->on_key = [&](const String &) + { + std::unique_lock lock(m); + parked = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }; + + const uint64_t adopted_before = counterOf(ProfileEvents::CASRefRecoveryEpochSealAdopted); + std::thread first([&] { store->listRefs(ns); }); + { + std::unique_lock lock(m); + cv.wait(lock, [&] { return parked; }); + } + + /// The second caller blocks on `recovery_in_progress`. Its own recovery would have to LIST, and the + /// walk holds no lock while parked, so nothing but the serialization flag can be keeping it out. + std::atomic second_done{false}; + std::thread second([&] { store->listRefs(ns); second_done.store(true); }); + for (int i = 0; i < 50 && !second_done.load(); ++i) + std::this_thread::yield(); + EXPECT_FALSE(second_done.load()) << "a second caller must wait out the in-flight walk, not race it"; + + { + std::lock_guard lock(m); + release = true; + } + cv.notify_all(); + first.join(); + second.join(); + + EXPECT_TRUE(store->refTableRecoveredForTest(ns)); + /// Exactly ONE walk minted the seal, and no second walk ever met it as an occupant. Adopting is the + /// CORRECT outcome for a concurrent recoverer -- it is just work this serialization exists to avoid + /// paying inside one process, so observing zero adoptions is what proves the second caller waited. + EXPECT_EQ(counterOf(ProfileEvents::CASRefRecoveryEpochSealAdopted), adopted_before); + const auto seal = readLogTxn(*backend, layout, ns, RefTxnId{1, 2}); + ASSERT_TRUE(seal.has_value()); + EXPECT_TRUE(refLogTxnIsEpochSeal(*seal)); +} + +/// Checkpoint-grounded recovery starts at the recreated life's own genesis and does not replay or +/// extend the predecessor life's stream. +/// +/// The old same-stream fixture claimed that recovery walked through the epoch-1 removal into epoch 2. +/// With authoritative `_ckpt.life_epoch=2`, epoch 2 is instead the current life's genesis and the walk +/// begins at `{2,1}`. Epoch-1 objects are inert predecessor-life debris: they neither supply state nor +/// receive a recovery seal. +TEST(CASRefRecoveryCasWalk, RecoveryStartsAtRecreatedLifeGenesisAndLeavesPredecessorStreamUntouched) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/removed_then_reborn"}; + + burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/3); + seedCkpt(*backend, layout, ns, lifeEpochCkpt(2, RefTxnId{2, 1})); + seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); + + /// Epoch 1 ends with the terminal record: the ref is removed, then the namespace. + RefLogTxn removal; + removal.ns = ns.string(); + removal.txn_id = RefTxnId{1, 2}; + removal.ops = {DB::Cas::tests::ownerTransitionOp( + RefOwnerBinding{RefOwnerKind::Committed, "a", manifestRef(1, 1, 1u)}, std::nullopt), + removeNamespaceOp()}; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, removal); + + /// The current life starts in epoch 2. Its birth is sequence 1 of its own genesis epoch, so it + /// carries no chain link to the predecessor life. + seedTxn(*backend, layout, ns, RefTxnId{2, 1}, "reborn", /*birth=*/true); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 3u); + + const auto refs = store->listRefs(ns); + EXPECT_EQ(refs.size(), 1u); + EXPECT_TRUE(refs.contains("reborn")) << "recovery must begin at the recreated life's genesis"; + + EXPECT_FALSE(readLogTxn(*backend, layout, ns, RefTxnId{1, 3}).has_value()) + << "recovery of life epoch 2 must not extend the predecessor-life stream"; + const auto seal2 = readLogTxn(*backend, layout, ns, RefTxnId{2, 2}); + ASSERT_TRUE(seal2.has_value()) << "epoch 2 IS live again by the time it dies, so it closes normally"; + EXPECT_TRUE(refLogTxnIsEpochSeal(*seal2)); + EXPECT_EQ(seal2->prev_epoch_seal, std::nullopt) << "sequence 2 carries no chain link"; +} + +/// `PutHookBackend::casPut` must route through its immediate parent `HidingListBackend::casPut`, not +/// past it to `CountingBackend`, so that a test arming BOTH layers on one `PutHookBackend` instance +/// gets both behaviors composed rather than one silently disabled by the other. +TEST(CASRefRecoveryCasWalk, PutHookBackendComposesHidingListBackendCasPutFaultInjection) +{ + auto backend = std::make_shared(); + + bool before_cas_put_fired = false; + backend->before_cas_put = [&](const String &, const String &, const std::optional &) + { + before_cas_put_fired = true; + }; + + backend->watched_substr = "probe"; + bool on_key_fired = false; + backend->on_key = [&] { on_key_fired = true; }; + + ASSERT_EQ(backend->casPut("p/probe", "x", std::nullopt).outcome, CasOutcome::Committed); + + EXPECT_TRUE(before_cas_put_fired) + << "HidingListBackend's before_cas_put hook must still fire for a PutHookBackend instance"; + EXPECT_TRUE(on_key_fired) << "PutHookBackend's own on_key hook must still fire on top of it"; +} diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp new file mode 100644 index 000000000000..2293c0bc166c --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp @@ -0,0 +1,436 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include + +/// v3 text codec tests for `cas_ref_snap` (codecs-v3 phase 3). Split out of the retired +/// `gtest_cas_ref_codecs.cpp` and re-pointed at the TEXT codec. The encoder-side validation tests are +/// format-agnostic and carry over verbatim; the old binary-offset byte-patch decode tests +/// (`bytes[k] = 99`) are gone -- the shape-level corruption classes (truncation, `v`+1 forward-gate, +/// wrong type, leading garbage) are covered by the `CASFormatBattery.RefSnapshot` row below, which also +/// subsumes the old `DecodeRejectsFutureFormatVersion`/`DecodeRejectsFormatVersionOne` pair (there is +/// no `format_version` byte any more -- the header `v` gate is the single forward-compat mechanism). + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +RefTableSnapshot makeLiveSnapshot() +{ + RefTableSnapshot s; + s.ns = "srv1/db/table@cas@"; + s.snapshot_id = RefTxnId{5, 200}; + + RefCommittedRow c1; + c1.ref_name = "all_1_1_0"; + c1.manifest_ref = manifestRef(5, 10, 1); + c1.published_at_ms = 1717000000000ULL; + s.committed.push_back(c1); + + RefCommittedRow c2; + c2.ref_name = "all_2_2_0"; + c2.manifest_ref = manifestRef(5, 11, 1); + c2.published_at_ms = 1717000000001ULL; + s.committed.push_back(c2); + + RefOwnerBinding p1{RefOwnerKind::Precommit, "all_3_3_0", manifestRef(5, 12, 1)}; + s.precommits.push_back(p1); + + return s; +} + +} + +/// =================================================================================== +/// RefTableSnapshot: round trip +/// =================================================================================== + +TEST(CASRefSnapshotCodec, RoundTripLive) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + const String bytes = encodeRefTableSnapshot(s); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); + EXPECT_EQ(decoded, s); +} + +TEST(CASRefSnapshotCodec, DecodeRequiresLifecycleField) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + String bytes = encodeRefTableSnapshot(s); + const String field = R"(,"lc":"live")"; + const size_t at = bytes.find(field); + ASSERT_NE(at, String::npos); + bytes.erase(at, field.size()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsTerminalLifecycleWord) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + String bytes = encodeRefTableSnapshot(s); + const String live = R"("lc":"live")"; + const size_t at = bytes.find(live); + ASSERT_NE(at, String::npos); + bytes.replace(at, live.size(), R"("lc":"removed")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnEpochField) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + String bytes = encodeRefTableSnapshot(s); + const String live = R"("lc":"live")"; + const size_t at = bytes.find(live); + ASSERT_NE(at, String::npos); + bytes.replace(at, live.size(), live + R"(,"rte":"7")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnSequenceField) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + String bytes = encodeRefTableSnapshot(s); + const String live = R"("lc":"live")"; + const size_t at = bytes.find(live); + ASSERT_NE(at, String::npos); + bytes.replace(at, live.size(), live + R"(,"rts":"9")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnFieldPair) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + String bytes = encodeRefTableSnapshot(s); + const String live = R"("lc":"live")"; + const size_t at = bytes.find(live); + ASSERT_NE(at, String::npos); + bytes.replace(at, live.size(), live + R"(,"rte":"7","rts":"9")"); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); +} + +/// No-tolerance decode pin (codex round-2, finding 3): the `"pl"` (payload) field was removed from the +/// committed-row wire in stage-1 T12. It is NOT a genuinely-unknown future field the tolerant reader may +/// skip -- silently discarding a persisted payload would lose data -- so decoding a committed row that +/// still carries `"pl"` must FAIL with `CORRUPTED_DATA` naming the removed field, not `skipUnknown` it. +TEST(CASRefSnapshotCodec, DecodeRejectsRemovedPayloadFieldInCommittedRow) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow c; + c.ref_name = "all_1_1_0"; + c.manifest_ref = manifestRef(5, 10, 1); + c.published_at_ms = 1717000000000ULL; + s.committed.push_back(c); + + const String bytes = encodeRefTableSnapshot(s); + /// Splice the retired `"pl"` field back into the committed record, just before its `"ts"` field. + const String needle = ",\"ts\":"; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + const String tampered = bytes.substr(0, pos) + R"(,"pl":"deadbeef")" + bytes.substr(pos); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(tampered, s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, RoundTripLiveEmpty) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + + const String bytes = encodeRefTableSnapshot(s); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); + EXPECT_EQ(decoded, s); + EXPECT_TRUE(decoded.committed.empty()); + EXPECT_TRUE(decoded.precommits.empty()); +} + +TEST(CASRefSnapshotCodec, ByteIdenticalReencode) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + const String bytes1 = encodeRefTableSnapshot(s); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes1, s.ns, s.snapshot_id); + const String bytes2 = encodeRefTableSnapshot(decoded); + EXPECT_EQ(bytes1, bytes2); +} + +TEST(CASRefSnapshotCodec, RoundTripPrecommitsSameNameDifferentManifest) +{ + /// Two builds racing for the same final ref name: same ref_name, different manifest_ref, sorted + /// by manifest_ref as the tiebreak. + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 1, 1)}); + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 2, 1)}); + + const String bytes = encodeRefTableSnapshot(s); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); + EXPECT_EQ(decoded, s); + EXPECT_EQ(decoded.precommits.size(), 2u); +} + +/// =================================================================================== +/// Large ids +/// =================================================================================== + +/// `ref_sequence` is a 64-bit counter and the codec writes it as a decimal STRING, so the top of the +/// range survives a round trip without JSON's number semantics getting involved. This used to be pinned +/// through the retired sentinel seal, whose synthetic `{E-1, UINT64_MAX}` id was the only place such a +/// value arose; the representation guarantee is what actually mattered and it is pinned directly here. +TEST(CASRefSnapshotFormat, MaximalRefSequenceRoundTripsAsADecimalString) +{ + RefTableSnapshot m; + m.ns = "ns"; + m.snapshot_id = RefTxnId{5, std::numeric_limits::max()}; + + const String text = encodeRefTableSnapshot(m); + const RefTableSnapshot back = decodeRefTableSnapshot(text, m.ns, m.snapshot_id); + EXPECT_EQ(back.snapshot_id.ref_sequence, std::numeric_limits::max()); + EXPECT_NE(text.find("\"rs\":\"18446744073709551615\""), String::npos); +} + +/// =================================================================================== +/// RefTableSnapshot: validation rejections (encoder-side + key/body binding + truncation) +/// =================================================================================== + +TEST(CASRefSnapshotCodec, EncodeRejectsZeroSnapshotId) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{0, 1}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsUnsortedCommitted) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow a; + a.ref_name = "b"; + a.manifest_ref = manifestRef(1, 1, 1); + RefCommittedRow b; + b.ref_name = "a"; + b.manifest_ref = manifestRef(1, 2, 1); + s.committed.push_back(a); + s.committed.push_back(b); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsDuplicateCommittedRefName) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow a; + a.ref_name = "same"; + a.manifest_ref = manifestRef(1, 1, 1); + RefCommittedRow b; + b.ref_name = "same"; + b.manifest_ref = manifestRef(1, 2, 1); + s.committed.push_back(a); + s.committed.push_back(b); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsUnsortedPrecommits) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "b", manifestRef(1, 1, 1)}); + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "a", manifestRef(1, 2, 1)}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsPrecommitsSameNameWrongManifestOrder) +{ + /// Same ref_name but the manifest_ref tiebreak is descending -- must be rejected even though the + /// names alone look sorted. + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 2, 1)}); + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 1, 1)}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsDuplicatePrecommitBinding) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 1, 1)}); + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "same", manifestRef(1, 1, 1)}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsNonCanonicalCommittedRefName) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + row.ref_name = "a/../b"; + row.manifest_ref = manifestRef(1, 1, 1); + s.committed.push_back(row); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsNonCanonicalPrecommitRefName) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "", manifestRef(1, 1, 1)}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsPrecommitWrongKind) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + s.precommits.push_back(RefOwnerBinding{RefOwnerKind::Committed, "r", manifestRef(1, 1, 1)}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, EncodeRejectsZeroManifestRefFields) +{ + { + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + row.ref_name = "r"; + row.manifest_ref = manifestRef(0, 1, 1); + s.committed.push_back(row); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); + } + { + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + row.ref_name = "r"; + row.manifest_ref = manifestRef(1, 1, 0); /// ordinal 0 is out of range + s.committed.push_back(row); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); + } +} + +TEST(CASRefSnapshotCodec, EncodeRejectsOversizedSnapshot) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + /// `ref_name` has no length limit (`checkCanonicalRefName`), so it is the padding field now that + /// `payload` is gone: a run of un-escaped 'x' bytes inflates the encoded row one-for-one. + row.ref_name = String(ref_snapshot_max_bytes + 1, 'x'); + row.manifest_ref = manifestRef(1, 1, 1); + s.committed.push_back(row); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefTableSnapshot(s); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsTruncatedBuffer) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + const String bytes = encodeRefTableSnapshot(s); + /// Dropping the trailing bytes leaves the final line without its '\n' terminator -> fail closed. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(bytes.substr(0, bytes.size() - 3), s.ns, s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsNamespaceMismatch) +{ + RefTableSnapshot s; + s.ns = "ns-a"; + s.snapshot_id = RefTxnId{1, 1}; + const String bytes = encodeRefTableSnapshot(s); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(bytes, "ns-b", s.snapshot_id); }); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsSnapshotIdMismatch) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + const String bytes = encodeRefTableSnapshot(s); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(bytes, s.ns, RefTxnId{1, 2}); }); +} + +TEST(CASRefSnapshotCodec, EncodeAllowsExactlySnapshotMaxBytes) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + row.ref_name = "r"; + row.manifest_ref = manifestRef(1, 1, 1); + s.committed.push_back(row); + + const size_t base_size = encodeRefTableSnapshot(s).size(); + ASSERT_LE(base_size, ref_snapshot_max_bytes); + /// Every added 'x' is one un-escaped byte inside the JSON ref_name string, so the encoded size + /// grows one-for-one to exactly the cap; +1 accounts for the base row's own 1-byte ref_name "r" + /// already counted in base_size. + s.committed[0].ref_name = String(ref_snapshot_max_bytes - base_size + 1, 'x'); + + const String bytes = encodeRefTableSnapshot(s); + EXPECT_EQ(bytes.size(), ref_snapshot_max_bytes); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); + EXPECT_EQ(decoded, s); +} + +TEST(CASRefSnapshotCodec, DecodeRejectsOversizedBufferDirectly) +{ + /// A body with no line terminator inside the first `line_cap` bytes fails closed before any field + /// parsing (the text `readLine` line-cap guard, the text-codec analogue of the old early size guard). + const String oversized(ref_snapshot_max_bytes + 1, 'x'); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(oversized, "ns", RefTxnId{1, 1}); }); +} + +/// =================================================================================== +/// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) +/// =================================================================================== + +TEST(CASFormatBattery, RefSnapshot) +{ + const RefTableSnapshot s = makeLiveSnapshot(); + const String ns = s.ns; + const RefTxnId id = s.snapshot_id; + runFormatBattery({FormatId::RefSnapshot, + [s] { return sealObject(FormatId::RefSnapshot, encodeRefTableSnapshot(s)); }, + [ns, id](std::string_view d) { decodeRefTableSnapshot(openObject(FormatId::RefSnapshot, d), ns, id); }, + currentFormatHeader("cas_ref_snap") + + "{\"ns\":\"srv1/db/table@cas@\",\"we\":\"5\",\"rs\":\"200\",\"lc\":\"live\"}\n" + "{\"k\":\"c\",\"rn\":\"all_1_1_0\",\"me\":\"5\",\"mb\":\"10\",\"mo\":1,\"ts\":1717000000000}\n" + "{\"k\":\"c\",\"rn\":\"all_2_2_0\",\"me\":\"5\",\"mb\":\"11\",\"mo\":1,\"ts\":1717000000001}\n" + "{\"k\":\"p\",\"rn\":\"all_3_3_0\",\"me\":\"5\",\"mb\":\"12\",\"mo\":1}\n" + "{\"n\":3}\n"}); +} diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp new file mode 100644 index 000000000000..ca34681f7b93 --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp @@ -0,0 +1,561 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event CASRefSnapshotPublishDispatched; +extern const Event CASRefSnapshotPublishBackoff; +} + +namespace DB::ErrorCodes +{ +extern const int NETWORK_ERROR; +} + +/// Task 6b remainder (Stage B, `{#t2}`): the publication-ordering coverage that Task 6b's rename left +/// undone. This suite PINS existing behavior of `CasRefLedger::tryPublishSnapshotAndAdvanceCheckpointOnce` +/// (the one retry unit), `admitSnapshotPublishUnderStateLock`, `advancePublishBackoff`/ +/// `resetPublishBackoff`, and `dispatchSnapshotPublisher`/`settleSnapshotPublish`. +/// +/// Normative ordering: (1) the immutable snapshot body becomes durable; (2) `_ckpt` advances; (3) the new +/// snapshot is adopted in this cache's memory. `NeedsRecovery` (this campaign's `Poisoned`) blocks +/// publication -- a durable transaction may be missing from the cached view -- and forces +/// `ensureRefTableRecovered` to re-walk the durable stream on the very next touch. +/// +/// The suite name is prefixed `CAS` so it is covered by the `CAS*` unit-test gate filter. + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::OrderedFaultBackend; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; + +namespace +{ + +PoolPtr openPool(const std::shared_ptr & backend, PoolConfig config = {}) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, std::move(config)); +} + +/// The same one-transaction publish every other ref suite drives, so a namespace reaches `Live` through +/// the REAL append lane (which is also what creates its `_ckpt`). +RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const String & ref, uint64_t ordinal) +{ + return store->appendRefOps(ns, MutationScope::ref(ref), + [&ref, ordinal](const RefTableState & state) + { + std::vector ops; + if (state.getLifecycle() != RefLifecycle::Live) + ops.push_back(namespaceBirthOp()); + for (const RefOp & op : publishCommittedOps(ref, ManifestRef{1, ordinal, 1})) + ops.push_back(op); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish); +} + +void forceAdoptablePublishWedge( + const PoolPtr & store, const RootNamespace & ns, uint64_t ref_sequence, const String & ref, uint64_t ordinal) +{ + const RefTxnId txn_id{store->writerEpoch(), ref_sequence}; + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = txn_id; + txn.ops = publishCommittedOps(ref, ManifestRef{1, ordinal, 1}); + const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + store->forceWedgeForTest( + ns, txn_id.writer_epoch, txn_id.ref_sequence, store->layout().refLogKey(life, txn_id), bytes); +} + +} + +/// --------------------------------------------------------------------------------------------- +/// 1. Snapshot body durable strictly before `_ckpt` advances +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefSnapshotPublishOrdering, SnapshotBodyIsDurableBeforeCheckpointAdvances) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/order_body_before_ckpt"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->writerEpoch(), 1}); + const String ckpt_key = store->layout().refCkptKey(life); + + /// The birth transaction above already CAS'd `_ckpt` itself (once for its own `life_epoch`, once for + /// its committed frontier) -- ordinary append-commit traffic that has nothing to do with the snapshot + /// publisher. The comparison below must therefore look only at what happens FROM this offset, or it + /// would find the birth's ckpt writes (which precede the snapshot body by construction) and conclude + /// nothing about the publisher's own ordering. + const size_t offset = backend->journalSize(); + const uint64_t put_before = backend->putCount(snapshot_key); + const uint64_t cas_before = backend->casPutCount(ckpt_key); + + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "a healthy Ready-lane table with an uncovered tail must publish"; + + /// Positive control: this attempt touched each key exactly once (no retry, no redundant write) -- + /// which is what makes the index comparison below meaningful rather than an artifact of a busy log. + EXPECT_EQ(backend->putCount(snapshot_key) - put_before, 1u); + EXPECT_EQ(backend->casPutCount(ckpt_key) - cas_before, 1u); + + const auto body_index = backend->firstIndexFrom(OrderedFaultBackend::Op::Put, snapshot_key, offset); + const auto ckpt_index = backend->firstIndexFrom(OrderedFaultBackend::Op::Cas, ckpt_key, offset); + ASSERT_TRUE(body_index.has_value()) << "the snapshot body must have been PUT"; + ASSERT_TRUE(ckpt_index.has_value()) << "the checkpoint must have been CAS-advanced"; + EXPECT_LT(*body_index, *ckpt_index) + << "INV-4's second `_ckpt` writer runs strictly after the immutable body is durable"; +} + + +/// --------------------------------------------------------------------------------------------- +/// 2. Adoption happens last, and only once both durable effects landed +/// --------------------------------------------------------------------------------------------- + +TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEffects) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/order_adoption_after_both"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->writerEpoch(), 1}); + const String ckpt_key = store->layout().refCkptKey(life); + + /// Fail every one of the (attempt-bounded) 100 `_ckpt` CAS attempts `publishCkpt` will make: the + /// body PUT still commits (dedup: an identical, already-durable body resolves as `Committed` without + /// re-sending), but the checkpoint never advances within this call. + backend->armCasConflict(ckpt_key, 100); + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "a persistently conflicting checkpoint CAS must not be reported as a successful publish"; + + EXPECT_EQ(backend->putCount(snapshot_key), 1u) << "the body is durable regardless of the ckpt outcome"; + EXPECT_FALSE(store->newestPublishedSnapshotIdForTest(ns).has_value()) + << "in-memory adoption must NOT happen while the checkpoint has not advanced"; + + /// Disarm the fault and retry (the one retry unit): the retry issues its OWN `putIfAbsent` attempt at + /// the same content-addressed key with the same bytes (so `putCount`, a call counter, becomes 2 -- + /// not a "no write happened" 1), but the backend resolves it as `Committed` against the already-durable + /// object rather than sending a distinct object, and the checkpoint CAS now succeeds. + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "the retry, with the fault cleared, must publish"; + EXPECT_EQ(backend->putCount(snapshot_key), 2u) + << "the retry's body PUT is its own attempt, resolved via dedup against identical, " + "already-durable bytes rather than writing a second object"; + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::make_optional(RefTxnId{store->writerEpoch(), 1})) + << "adoption happens exactly once, after both effects are durable"; +} + +/// --------------------------------------------------------------------------------------------- +/// 3. `NeedsRecovery` ("Poisoned") lane: recovery precedes any snapshot publication +/// --------------------------------------------------------------------------------------------- + +/// `Poisoned` is this task's plan's name for what the code spells `RefLaneState::NeedsRecovery` -- the +/// state the header documents as "a transaction is known durable but cannot be installed in this cache +/// ... a hard write and certification fence until replay completes". Recorded here as the vocabulary +/// correction for later tasks: there is no state literally named `Poisoned` anywhere in `CasRefLedger`. +/// This test is the plan's `PoisonedRefusesPublicationAndTriggersReRecovery`, renamed to state the actual +/// pinned behavior precisely (recovery precedes publication, rather than an outright refusal). +/// +/// It is reached here the same way `gtest_cas_ref_writer.cpp`'s +/// `CASRefWriterAppendLane.CheckpointConflictAfterLogCommitRequiresRecoveryWithoutInstall` reaches it: a +/// mutation's ref-log body commits durably while its OWN checkpoint-frontier CAS (`commitRefChunk`'s +/// `commit_contribution`, not the snapshot publisher's) conflicts persistently. +/// +/// The INVARIANT this pins (not a raw write count): a snapshot must never be published FROM AN +/// UNRECOVERED CACHE -- a durable transaction may be missing from the cached view, and advancing `_ckpt` +/// onto a snapshot built from that stale view is the data-loss shape `NeedsRecovery` exists to prevent. +/// `tryPublishSnapshotAndAdvanceCheckpointOnce` calls `ensureRefTableRecovered` unconditionally, and that +/// function re-walks the durable stream whenever the lane is `NeedsRecovery`, regardless of `recovered`. +/// Recovery's own `_ckpt` catch-up write is NOT a violation of this invariant -- it is the remedy: it is +/// how the cache stops being stale before anything is allowed to read it for a snapshot. So a request +/// against a poisoned lane recovers first and MAY legitimately go on to publish (this table had never +/// published a snapshot, so once recovered it has a real, uncovered candidate) -- "inert refusal with +/// zero writes" is NOT what production implements, and recover-then-proceed is the correct behavior, not +/// a deviation from it. What this test pins is: (a) no snapshot-publish effect (body PUT, publisher's own +/// checkpoint-advance CAS) can ever appear in the journal before recovery's reconciliation CAS; (b) +/// re-recovery is an observable state transition, never a silent skip; (c) if a snapshot IS published, it +/// reflects the RECOVERED frontier -- the durable transaction the stale cache was missing is actually +/// covered by it, not merely "some snapshot, from whichever view". +TEST(CASRefSnapshotPublishOrdering, NeedsRecoveryLaneRecoversBeforeAnySnapshotPublication) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/order_poisoned_refuses"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + const String ckpt_key = store->layout().refCkptKey(life); + /// The durable transaction the stale cache will be missing: `dropRef`'s removal, sequence 2. + const RefTxnId missing_durable_txn{store->writerEpoch(), 2}; + const String next_snapshot_key = store->layout().refSnapshotKey(life, missing_durable_txn); + + /// Drive the very next mutation's OWN checkpoint-frontier CAS into persistent conflict: the log PUT + /// for `missing_durable_txn` commits durably, but its checkpoint never advances within this call, and + /// the lane is left `NeedsRecovery` rather than installing an uncertain result -- so the cached view + /// still reflects `ref_1` present, while the durable log already reflects it removed. + backend->armCasConflict(ckpt_key, 100); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "ref_1"); }); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + + const uint64_t recovery_installs_before = store->recoveryInstallCountForTest(); + backend->armCasConflict(ckpt_key, 0); /// clear the fault so re-recovery's OWN catch-up CAN succeed + const size_t offset = backend->journalSize(); + + EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "recovery reconciles the durable gap and this table has never published a snapshot, so the " + "same call legitimately goes on to publish one -- see the invariant note above the test"; + + /// Re-recovery WAS triggered as an observable state transition (not a silent skip): the lane left + /// `NeedsRecovery`, and `recoveryInstallCountForTest` -- a counter of exact recovery-result + /// publications -- advanced. + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "ensureRefTableRecovered must have re-walked the durable stream and cleared the fence"; + EXPECT_GT(store->recoveryInstallCountForTest(), recovery_installs_before) + << "a re-recovery install must be observable, not indistinguishable from never having run"; + + /// ORDER, not a global zero: recovery's OWN checkpoint catch-up CAS is the boundary marker. NO + /// snapshot-publish effect (the new snapshot's body PUT, nor the publisher's own checkpoint-advance + /// CAS) may appear at or before it. + const auto ckpt_cas_indices = backend->indicesFrom(OrderedFaultBackend::Op::Cas, ckpt_key, offset); + const auto snap_put_indices = backend->indicesFrom(OrderedFaultBackend::Op::Put, next_snapshot_key, offset); + ASSERT_GE(ckpt_cas_indices.size(), 2u) + << "expected one checkpoint CAS from recovery's catch-up and one from the snapshot publisher"; + const size_t recovery_catchup_index = ckpt_cas_indices.front(); + const size_t publisher_ckpt_index = ckpt_cas_indices.back(); + ASSERT_FALSE(snap_put_indices.empty()) << "the recovered, uncovered candidate must have been published"; + for (const size_t snap_put_index : snap_put_indices) + EXPECT_GT(snap_put_index, recovery_catchup_index) + << "no snapshot-publish body PUT may precede recovery's own checkpoint reconciliation"; + EXPECT_LT(snap_put_indices.front(), publisher_ckpt_index) + << "the snapshot publisher's own checkpoint CAS still runs after ITS OWN body PUT (INV-4), even " + "immediately following recovery"; + + /// STRONGEST form: the published snapshot is not merely "some snapshot from whichever view" -- it + /// covers EXACTLY the recovered, previously-missing-from-cache frontier. Its id names the durable + /// removal transaction, and the recovered cache (which the snapshot was built from) no longer + /// resolves the removed ref. + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::make_optional(missing_durable_txn)) + << "the published snapshot's frontier IS the durable transaction the stale cache was missing"; + EXPECT_FALSE(store->resolveRef(ns, "ref_1").has_value()) + << "the recovered (and now snapshotted) cache reflects the durable removal the stale view lacked"; +} + +/// --------------------------------------------------------------------------------------------- +/// 4. Publish backoff: characterized against a controlled clock (`PoolConfig::boot_ms_fn`) +/// --------------------------------------------------------------------------------------------- + +/// `admitSnapshotPublishUnderStateLock`, `advancePublishBackoff` and `resetPublishBackoff` are private +/// to `CasRefLedger`, so they can only be characterized through the public dispatch surface +/// (`appendRefOps`/`resolveRef` triggering `maybeScheduleSnapshotPublish`, and +/// `waitForSnapshotPublishSettleForTest`/`ProfileEvents::CASRefSnapshotPublishDispatched` as the +/// observables). `CASRequestControllerBackoff` is a DIFFERENT mechanism (the request controller's +/// per-attempt retry backoff); this characterizes ONLY the per-table snapshot-publish dispatch backoff. +/// +/// A controlled clock (`PoolConfig::boot_ms_fn`) DOES exist for this seam (`gtest_cas_ref_writer.cpp`'s +/// `C4BackoffDefersThenRetriesAndPublishes` already relies on it) -- so unlike the plan's anticipated +/// fallback, this pins literal accept/refuse decisions against exact clock offsets rather than only +/// attempt counts. +TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + + /// A single-attempt request budget, exactly as `gtest_cas_ref_writer.cpp`'s + /// `C4BackoffDefersThenRetriesAndPublishes` uses: with `max_attempts = 1` a faulted PUT resolves to a + /// definite, non-`Committed` outcome on its own attempt, with no internal retry loop and so no + /// wall-clock wait. + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 100; + + uint64_t fake_now = 1'000'000; + PoolConfig config; + config.snapshot_log_count_threshold = 0; /// any nonempty tail is over-threshold + config.snapshot_log_bytes_threshold = 1ULL << 40; + config.snapshot_publish_backoff_initial_ms = 1000; + config.snapshot_publish_backoff_max_ms = 4000; + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget = budget; + auto store = openPool(backend, config); + const RootNamespace ns{"srv1/order_backoff"}; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + store->waitForSnapshotPublishSettleForTest(ns); /// drain the birth's own auto-dispatched publish + /// The birth's own auto-dispatch already published a snapshot at this point (threshold 0); the + /// baseline every "no new publish yet" check below compares against. + const auto snapshot_after_birth = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(snapshot_after_birth.has_value()); + + /// Fault the snapshot BODY put (never the `_ckpt` CAS -- an append-commit's OWN checkpoint write + /// shares that key, and faulting it would drive the append lane into `NeedsRecovery` instead of + /// exercising the snapshot-publish backoff this test targets). Exactly 3 failures: the next 3 + /// automatic dispatch attempts fail (arming, then doubling, then re-doubling the backoff); the 4th + /// finds the fault disarmed and succeeds. + backend->armPutFailure("_snap/", 3); + + const auto dispatchCount = [&] { return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); }; + + /// Attempt 1: admitted immediately (no backoff armed yet). Fails -> backoff armed at the initial 1000ms. + ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 2})); + store->waitForSnapshotPublishSettleForTest(ns); + const uint64_t d1 = dispatchCount(); + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), snapshot_after_birth) + << "the failed attempt must not have advanced the published snapshot"; + + /// Still within the 1000ms window: a further trigger must NOT re-dispatch. + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1) << "a read within the initial backoff window must not re-dispatch"; + + /// Cross the 1000ms deadline: exactly one retry dispatches (and fails again, doubling to 2000ms). + fake_now += 1000; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 1) << "past the first deadline, exactly one retry dispatches"; + + /// Short of the DOUBLED (2000ms) deadline: still refused. + fake_now += 1000; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 1) + << "advancePublishBackoff doubled the interval to 2000ms; 1000ms elapsed is not enough"; + + /// Cross the doubled deadline: one more retry dispatches (and fails again -- the third and last armed + /// failure -- doubling to the 4000ms cap). + fake_now += 1000; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 2) << "past the doubled deadline, exactly one more retry dispatches"; + + /// Pin the 4000ms cap FROM BELOW: without this probe, a regression that stopped doubling at + /// 2000ms, or that read `initial` where it means `max`, would still pass -- the only check so far + /// is AT the +4000 crossing below. 2000ms past the doubled deadline is still short of the capped + /// 4000ms backoff, so no third retry may dispatch yet. + fake_now += 2000; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 2) + << "2000ms past the doubled deadline is still short of the capped 4000ms backoff"; + + /// Cross the (capped) 4000ms deadline: the retry's fault budget is exhausted, so this attempt + /// succeeds, and `resetPublishBackoff` clears the cooldown -- proved by the NEXT trigger dispatching + /// with no wait at all. + fake_now += 2000; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 3) << "past the second (capped) deadline, the retry dispatches and succeeds"; + EXPECT_NE(store->newestPublishedSnapshotIdForTest(ns), snapshot_after_birth) + << "the fault budget is exhausted, so this attempt actually advances the published snapshot"; + + ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->writerEpoch(), 3})); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d1 + 4) + << "resetPublishBackoff must have cleared the cooldown: the very next over-threshold trigger, at " + "the SAME clock reading as the successful publish, dispatches immediately with no wait"; + + /// The assertion just above cannot tell a real reset from a no-op: the successful publish and this + /// next trigger share one `fake_now`, so `now >= until` would still hold even with the stale + /// (pre-reset) deadline in place. Arm one more failure and check that the schedule restarts from + /// the INITIAL 1000ms interval rather than continuing from the 4000ms cap -- refused short of + /// 1000ms, admitted at 1000ms -- which a no-op reset cannot produce (it would refuse both probes, + /// since the stale deadline is still far in the future). + backend->armPutFailure("_snap/", 1); + ASSERT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->writerEpoch(), 4})); + store->waitForSnapshotPublishSettleForTest(ns); + const uint64_t d2 = dispatchCount(); + fake_now += 500; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d2) << "short of 1000ms since the reset, no retry may dispatch yet"; + fake_now += 500; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatchCount(), d2 + 1) + << "resetPublishBackoff must have restarted the schedule at the INITIAL 1000ms interval, not " + "left it continuing from the 4000ms cap"; +} + +TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurablePublish) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 100; + + uint64_t fake_now = 2'000'000; + PoolConfig config; + config.snapshot_log_count_threshold = 0; + config.snapshot_log_bytes_threshold = 1ULL << 40; + config.snapshot_publish_backoff_initial_ms = 200; + config.snapshot_publish_backoff_max_ms = 30'000; + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget = budget; + auto store = openPool(backend, config); + const RootNamespace ns{"srv1/order_not_ready_backoff"}; + + const auto dispatch_count = [&] + { + return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + }; + const auto backoff_count = [&] + { + return global_counters[ProfileEvents::CASRefSnapshotPublishBackoff].load(); + }; + + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + store->waitForSnapshotPublishSettleForTest(ns); + ASSERT_EQ( + store->newestPublishedSnapshotIdForTest(ns), + std::make_optional(RefTxnId{store->writerEpoch(), 1})); + + /// Direct calls remain one attempt per invocation even while a cooldown is armed. This first + /// refusal is also the non-hanging RED discriminator: without the production fix the backoff + /// counter is unchanged, so the fatal assertion stops before settlement can redispatch forever. + forceAdoptablePublishWedge(store, ns, 2, "ref_2", 2); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + const uint64_t warmup_backoffs = backoff_count(); + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + ASSERT_EQ(backoff_count(), warmup_backoffs + 1) + << "one admitted NotReady refusal must arm the initial snapshot-publish backoff"; + EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + ASSERT_EQ(backoff_count(), warmup_backoffs + 2) + << "a direct call is still one admitted attempt per invocation and doubles the cooldown"; + + /// Resolve the exact wedge through the real append-lane adoption path. Its adopted txn and + /// the caller's own txn raise the table above threshold, but the warm-up cooldown prevents an + /// automatic publish while the fixture prepares one uncovered tail entry. + ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->writerEpoch(), 3})); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + bool appended_during_capture = false; + store->setSnapshotAfterCaptureHookForTest([&] + { + if (appended_during_capture) + return; + appended_during_capture = true; + EXPECT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->writerEpoch(), 4})); + }); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + store->setSnapshotAfterCaptureHookForTest(nullptr); + ASSERT_TRUE(appended_during_capture); + ASSERT_EQ( + store->newestPublishedSnapshotIdForTest(ns), + std::make_optional(RefTxnId{store->writerEpoch(), 3})); + + /// One uncovered tail entry now exists with no cooldown. Make the lane non-Ready before the read + /// trigger, so the first production dispatch is an admitted refusal rather than a body PUT. + forceAdoptablePublishWedge(store, ns, 5, "ref_5", 5); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + const uint64_t production_dispatches = dispatch_count(); + const uint64_t production_backoffs = backoff_count(); + + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + ASSERT_EQ(dispatch_count(), production_dispatches + 1) + << "the over-threshold table must dispatch one admitted refusal"; + ASSERT_EQ(backoff_count(), production_backoffs + 1) + << "the admitted refusal must arm exactly one 200ms cooldown"; + + /// Settlement re-evaluates immediately. The armed deadline must stop that handoff from becoming a + /// second dispatch, and an ordinary trigger at the same BOOTTIME instant must also remain refused. + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatch_count(), production_dispatches + 1); + EXPECT_EQ(backoff_count(), production_backoffs + 1); + + uint64_t admitted_retries = 0; + uint64_t delay_ms = 200; + const std::vector next_delays{ + 400, 800, 1600, 3200, 6400, 12'800, 25'600, 30'000, 30'000}; + for (const uint64_t next_delay_ms : next_delays) + { + fake_now += delay_ms - 1; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatch_count(), production_dispatches + 1 + admitted_retries) + << "no retry may dispatch one millisecond before the current deadline"; + EXPECT_EQ(backoff_count(), production_backoffs + 1 + admitted_retries); + + ++fake_now; + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + ++admitted_retries; + EXPECT_EQ(dispatch_count(), production_dispatches + 1 + admitted_retries) + << "exactly one retry must dispatch at the BOOTTIME deadline"; + EXPECT_EQ(backoff_count(), production_backoffs + 1 + admitted_retries) + << "each admitted NotReady retry advances the same bounded cooldown once"; + delay_ms = next_delay_ms; + } + + /// The last two intervals are both 30 seconds: the retry at the first capped deadline must arm the + /// same cap, rather than overflow, reset, or continue doubling. + EXPECT_EQ(delay_ms, 30'000u); + + /// Adopt the outstanding wedge through production and publish durably. The hook commits one later + /// txn after capture while the capped cooldown is still armed; a correct durable publication resets + /// that cooldown, so an immediate same-clock read dispatches the leftover tail without waiting. + ASSERT_EQ(publishRef(store, ns, "ref_6", 6), (RefTxnId{store->writerEpoch(), 6})); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + bool appended_after_reset_capture = false; + store->setSnapshotAfterCaptureHookForTest([&] + { + if (appended_after_reset_capture) + return; + appended_after_reset_capture = true; + EXPECT_EQ(publishRef(store, ns, "ref_7", 7), (RefTxnId{store->writerEpoch(), 7})); + }); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + store->setSnapshotAfterCaptureHookForTest(nullptr); + ASSERT_TRUE(appended_after_reset_capture); + ASSERT_EQ( + store->newestPublishedSnapshotIdForTest(ns), + std::make_optional(RefTxnId{store->writerEpoch(), 6})); + + const uint64_t dispatches_before_reset_probe = dispatch_count(); + store->resolveRef(ns, "ref_1"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(dispatch_count(), dispatches_before_reset_probe + 1) + << "durable publication must clear the capped cooldown for an immediate same-clock trigger"; + EXPECT_EQ( + store->newestPublishedSnapshotIdForTest(ns), + std::make_optional(RefTxnId{store->writerEpoch(), 7})); +} diff --git a/src/Disks/tests/gtest_cas_ref_statemachine.cpp b/src/Disks/tests/gtest_cas_ref_statemachine.cpp new file mode 100644 index 000000000000..af79232cde2e --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_statemachine.cpp @@ -0,0 +1,1445 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +/// =================================================================================== +/// Small builders (mirrors gtest_cas_ref_codecs.cpp's local helpers) +/// =================================================================================== + +ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +RefLogTxn makeTxn(const String & ns, RefTxnId id, std::vector ops) +{ + RefLogTxn txn; + txn.ns = ns; + txn.txn_id = id; + txn.ops = std::move(ops); + return txn; +} + +RefOp birthOp() +{ + RefOp op; + op.kind = RefOpKind::NamespaceBirth; + return op; +} + +RefOp addPrecommitOp(const String & name, const ManifestRef & mref) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, name, mref}; + return op; +} + +RefOp removePrecommitOp(const String & name, const ManifestRef & mref) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, name, mref}; + return op; +} + +RefOp promoteOp(const String & name, const ManifestRef & mref) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, name, mref}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, name, mref}; + return op; +} + +RefOp removeCommittedOp(const String & name, const ManifestRef & mref) +{ + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Committed, name, mref}; + return op; +} + +RefOp setPublishedAtOp(const String & name, const ManifestRef & mref, uint64_t ts = 0) +{ + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = name; + op.expected_manifest_ref = mref; + op.published_at_ms = ts; + return op; +} + +RefOp removeNamespaceOp() +{ + RefOp op; + op.kind = RefOpKind::RemoveNamespace; + return op; +} + +/// Field-by-field comparison (via getters) rather than a `RefTableState::operator==` addition: the +/// class is the plan's verbatim-normative interface and gains no member beyond what it specifies. +void expectStatesEqual(const RefTableState & a, const RefTableState & b) +{ + EXPECT_EQ(a.getLifecycle(), b.getLifecycle()); + EXPECT_EQ(a.getRemoveTxnId(), b.getRemoveTxnId()); + EXPECT_EQ(a.getGreatestApplied(), b.getGreatestApplied()); + EXPECT_EQ(a.getCommitted(), b.getCommitted()); + EXPECT_EQ(a.getPrecommits(), b.getPrecommits()); + /// Also compare the incremental budget counters: in release builds (no `debugAssertBodyCounters`) + /// this is the only cross-check that catches counter drift between two equal-looking states. + EXPECT_EQ(a.getSnapshotBodyBytes(), b.getSnapshotBodyBytes()); + EXPECT_EQ(a.getRemovalBodyBytes(), b.getRemovalBodyBytes()); +} + +/// The spec's own construction for a hypothetical `remove_namespace` transaction (§Remove Namespace): +/// an exact owner-removal op for every committed ref and precommit, then `remove_namespace`. Built +/// independently of `CasRefStateMachine.cpp`'s internal helper of the same shape, purely from the +/// public `RefTableState` fields, so the admission-budget property tests below measure against a +/// ground truth this test file derives on its own. +RefLogTxn buildRemovalTxnForTest(const RefTableState & state, const String & ns, RefTxnId id) +{ + std::vector ops; + for (const auto [name, row] : state.getCommitted()) + ops.push_back(removeCommittedOp(name, row.manifest_ref)); + for (const auto & [name, mref] : state.getPrecommits()) + ops.push_back(removePrecommitOp(name, mref)); + ops.push_back(removeNamespaceOp()); + return makeTxn(ns, id, std::move(ops)); +} + +constexpr const char * kNs = "srv1/db/table@cas@"; + +/// A validated state with "a" committed to manifest (1,1,1) -- the base the fail-closed replay/append +/// tests below reuse to build a tail whose add-precommit op would collide cross-owner (name the SAME +/// manifest under a DIFFERENT ref_name). +RefTableSnapshot buildCollidingBaseSnapshotForTest() +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + return snapshotOf(state, kNs); +} + +} + +/// =================================================================================== +/// NamespaceBirth +/// =================================================================================== + +TEST(CASRefStateMachine, BirthFromNeverBornAccepts) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Live); + EXPECT_FALSE(state.getRemoveTxnId().has_value()); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{1, 1})); +} + +TEST(CASRefStateMachine, BirthWhileLiveRejectedAndStateUnchanged) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {birthOp()})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, BirthAfterRemovalAccepts) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeNamespaceOp()})); + ASSERT_EQ(state.getLifecycle(), RefLifecycle::Removed); + ASSERT_TRUE(state.getRemoveTxnId().has_value()); + + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {birthOp()})); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Live); + EXPECT_FALSE(state.getRemoveTxnId().has_value()); +} + +/// =================================================================================== +/// Ops rejected outside Live (never-born and Removed) except birth +/// =================================================================================== + +TEST(CASRefStateMachine, OwnerTransitionWhileNeverBornRejected) +{ + RefTableState state; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); +} + +TEST(CASRefStateMachine, SetPublishedAtWhileNeverBornRejected) +{ + RefTableState state; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {setPublishedAtOp("a", manifestRef(1, 1, 1))})); }); +} + +TEST(CASRefStateMachine, RemoveNamespaceWhileNeverBornRejected) +{ + RefTableState state; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {removeNamespaceOp()})); }); +} + +TEST(CASRefStateMachine, OpsWhileRemovedRejectedExceptBirth) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeNamespaceOp()})); + const RefTableState after_removal = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(after_removal, state); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 4}, {setPublishedAtOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(after_removal, state); + + /// Repeated removal is corruption at THIS layer (spec §Remove Namespace: idempotent-success is + /// the API layer's job, not the state machine's). + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 5}, {removeNamespaceOp()})); }); + expectStatesEqual(after_removal, state); +} + +/// =================================================================================== +/// Add precommit (spec §Add Precommit) +/// =================================================================================== + +TEST(CASRefStateMachine, AddPrecommitAccepts) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + EXPECT_TRUE(state.getPrecommits().contains({"a", manifestRef(1, 1, 1)})); +} + +TEST(CASRefStateMachine, AddPrecommitRejectsExactDuplicate) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, AddPrecommitRejectsConflictingManifestUnderDifferentName) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + /// Same manifest_ref, a DIFFERENT ref_name: "no conflicting owner may name the same manifest". + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("b", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, AddPrecommitRejectsManifestAlreadyCommittedElsewhere) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + ASSERT_TRUE(state.getCommitted().contains("a")); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("b", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, AddPrecommitAllowsDifferentManifestsRacingForSameName) +{ + /// Two builds racing for the same final ref name (same shape gtest_cas_ref_codecs.cpp's + /// RoundTripPrecommitsSameNameDifferentManifest round-trips): distinct manifest_ref, no conflict. + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("same", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("same", manifestRef(1, 2, 1))})); + EXPECT_TRUE(state.getPrecommits().contains({"same", manifestRef(1, 1, 1)})); + EXPECT_TRUE(state.getPrecommits().contains({"same", manifestRef(1, 2, 1)})); +} + +/// =================================================================================== +/// Remove precommit / remove committed (spec §Remove Precommit, §Remove Committed Ref) +/// =================================================================================== + +TEST(CASRefStateMachine, RemovePrecommitAccepts) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removePrecommitOp("a", manifestRef(1, 1, 1))})); + EXPECT_TRUE(state.getPrecommits().empty()); +} + +TEST(CASRefStateMachine, RemovePrecommitRejectsAbsentBinding) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removePrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, RemovePrecommitRejectsWrongManifest) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removePrecommitOp("a", manifestRef(1, 2, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, RemoveCommittedAccepts) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeCommittedOp("a", manifestRef(1, 1, 1))})); + EXPECT_TRUE(state.getCommitted().empty()); +} + +TEST(CASRefStateMachine, RemoveCommittedRejectsAbsentRef) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeCommittedOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, RemoveCommittedRejectsWrongManifest) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeCommittedOp("a", manifestRef(9, 9, 9))})); }); + expectStatesEqual(before, state); +} + +/// =================================================================================== +/// Promote (spec §Promote): exact precommit required, atomicity, invalid shapes +/// =================================================================================== + +TEST(CASRefStateMachine, PromoteRejectsAbsentPrecommit) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, PromoteAtomicityNoOwnerlessIntermediate) +{ + /// A bare promote (no set_published_at in the same transaction) is itself a complete, valid, and + /// OBSERVABLE transaction -- there is no partial-op state exposed here, only the choice of + /// whether the timestamp arrives in this txn or a later one (spec §Promote). + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))})); + + EXPECT_FALSE(state.getPrecommits().contains({"a", manifestRef(1, 1, 1)})); + ASSERT_TRUE(state.getCommitted().contains("a")); + EXPECT_EQ(state.getCommitted().at("a").manifest_ref, manifestRef(1, 1, 1)); + EXPECT_EQ(state.getCommitted().at("a").published_at_ms, 0u); +} + +TEST(CASRefStateMachine, PromoteWithSetPublishedAtInSameTxnInstallsTimestamp) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {promoteOp("a", manifestRef(1, 1, 1)), setPublishedAtOp("a", manifestRef(1, 1, 1), 42)})); + + ASSERT_TRUE(state.getCommitted().contains("a")); + EXPECT_EQ(state.getCommitted().at("a").published_at_ms, 42u); +} + +TEST(CASRefStateMachine, PromoteRejectsDisplacingAnotherCommittedManifest) +{ + /// A challenger precommit under the SAME ref_name as an already-committed (different) manifest is + /// legal to stage (spec §Add Precommit only restricts manifest identity, not ref_name), but a bare + /// promote of it must not silently displace the stale committed row. + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 2, 1))})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {promoteOp("a", manifestRef(1, 2, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, PromoteAcceptsAfterExplicitRemovalOfStaleCommitted) +{ + /// The correct atomic-replace sequence: an explicit removal of the old committed row, followed by + /// the promote, in the SAME transaction -- both ops are recorded, so GC sees the old manifest's + /// "-1" edge explicitly rather than losing it to a silent displacement. + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 2, 1))})); + + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, + {removeCommittedOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 2, 1))})); + + ASSERT_TRUE(state.getCommitted().contains("a")); + EXPECT_EQ(state.getCommitted().at("a").manifest_ref, manifestRef(1, 2, 1)); + EXPECT_FALSE(state.getPrecommits().contains({"a", manifestRef(1, 2, 1)})); +} + +TEST(CASRefStateMachine, OwnerTransitionRejectsInvalidCombinations) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + + /// old=None, new=Committed: not a recognized shape (committed rows are only reached via promote). + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "a", manifestRef(1, 1, 1)}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {op})); }); + } + + /// A promote-shaped op (Precommit -> Committed) with mismatched ref_name is not a legal promote. + /// The rejected transaction above left `greatest_applied` untouched, so THIS one is still {1, 2}. + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "a", manifestRef(1, 1, 1)}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "b", manifestRef(1, 1, 1)}; + const RefTableState before = state; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {op})); }); + expectStatesEqual(before, state); + } + + /// old=Committed, new=Precommit: moving a committed ref "backwards" is not a recognized shape. + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Committed, "a", manifestRef(1, 1, 1)}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "a", manifestRef(1, 1, 1)}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {op})); }); + } +} + +/// =================================================================================== +/// SetPublishedAt (spec §Update Payload) +/// =================================================================================== + +TEST(CASRefStateMachine, SetPublishedAtRejectsWhenRefAbsent) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {setPublishedAtOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, SetPublishedAtRejectsManifestMismatch) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {setPublishedAtOp("a", manifestRef(9, 9, 9))})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, SetPublishedAtAcceptsAndReplacesTimestamp) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {setPublishedAtOp("a", manifestRef(1, 1, 1), 10)})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {setPublishedAtOp("a", manifestRef(1, 1, 1), 20)})); + + EXPECT_EQ(state.getCommitted().at("a").published_at_ms, 20u); + EXPECT_EQ(state.getCommitted().at("a").manifest_ref, manifestRef(1, 1, 1)); /// unchanged: no edge move +} + +/// =================================================================================== +/// RemoveNamespace ordering lens (spec §Remove Namespace; codec deliberately doesn't check this) +/// =================================================================================== + +TEST(CASRefStateMachine, RemoveNamespaceAloneOnEmptyTableAccepted) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeNamespaceOp()})); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Removed); + ASSERT_TRUE(state.getRemoveTxnId().has_value()); + EXPECT_EQ(*state.getRemoveTxnId(), (RefTxnId{1, 2})); +} + +TEST(CASRefStateMachine, CatalogedNeverBornLifeAcceptsAtomicEmptyBirthAndRemoval) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), removeNamespaceOp()})); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Removed); + ASSERT_TRUE(state.getRemoveTxnId().has_value()); + EXPECT_EQ(*state.getRemoveTxnId(), (RefTxnId{1, 1})); + EXPECT_TRUE(state.getCommitted().empty()); + EXPECT_TRUE(state.getPrecommits().empty()); +} + +TEST(CASRefStateMachine, RemoveNamespaceDrainingOwnersInSameTxnAccepted) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), addPrecommitOp("b", manifestRef(1, 2, 1)), + promoteOp("b", manifestRef(1, 2, 1))})); + ASSERT_TRUE(state.getPrecommits().contains({"a", manifestRef(1, 1, 1)})); + ASSERT_TRUE(state.getCommitted().contains("b")); + + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {removePrecommitOp("a", manifestRef(1, 1, 1)), removeCommittedOp("b", manifestRef(1, 2, 1)), + removeNamespaceOp()})); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Removed); + EXPECT_TRUE(state.getCommitted().empty()); + EXPECT_TRUE(state.getPrecommits().empty()); +} + +TEST(CASRefStateMachine, RemoveNamespaceRejectsWhenOwnersRemain) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), addPrecommitOp("b", manifestRef(1, 2, 1))})); + const RefTableState before = state; + + /// Only "a" is drained; "b" remains -- remove_namespace's own precondition (empty owner sets) + /// must fail, and the WHOLE transaction (including the "a" removal) must not apply. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {removePrecommitOp("a", manifestRef(1, 1, 1)), removeNamespaceOp()})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, RemoveNamespaceMustBeFinalOp) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeNamespaceOp(), birthOp()})); }); + expectStatesEqual(before, state); +} + +TEST(CASRefStateMachine, RemoveNamespaceRejectsNonRemovalEarlierOp) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + const RefTableState before = state; + + /// set_published_at before remove_namespace: not an owner-removal transition. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {setPublishedAtOp("a", manifestRef(1, 1, 1)), removeCommittedOp("a", manifestRef(1, 1, 1)), + removeNamespaceOp()})); }); + expectStatesEqual(before, state); + + /// An ADD (not a removal) owner_transition before remove_namespace: also rejected. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, + {addPrecommitOp("c", manifestRef(1, 3, 1)), removeCommittedOp("a", manifestRef(1, 1, 1)), + removeNamespaceOp()})); }); + expectStatesEqual(before, state); +} + +/// =================================================================================== +/// Whole-transaction atomicity: a failing LAST op leaves the whole txn (and earlier ops) unapplied +/// =================================================================================== + +TEST(CASRefStateMachine, WholeTxnAtomicityLastOpFailureLeavesStateUntouched) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + /// ops[0] (add "a") would succeed in isolation; ops[1] (remove absent "b") fails -- the whole + /// transaction, including "a", must be rejected. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {addPrecommitOp("a", manifestRef(1, 1, 1)), removePrecommitOp("b", manifestRef(9, 9, 9))})); }); + + expectStatesEqual(before, state); + EXPECT_FALSE(state.getPrecommits().contains({"a", manifestRef(1, 1, 1)})); +} + +/// =================================================================================== +/// Contiguous txn ids (INV-1) +/// =================================================================================== + +/// A table's durable ids are DENSE within `(namespace, epoch)`: the only admissible id is the one +/// `nextRefTxnId` derives from `greatest_applied`, which is also the only id the writer ever mints. +/// Equal, lower, and skipped ids are all corruption -- the last of those is what makes "I can see ids +/// 1..T" mean "nothing is missing", the property the whole invariant exists to provide. +TEST(CASRefStateMachine, ContiguousTxnIdsRejectEqualLowerAndSkipped) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableState before = state; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{0, 999}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); + + /// Strictly greater but SKIPPED: admitted before INV-1, corruption now. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); + + /// A new epoch restarts the sequence, so it must start at 1 -- carrying the previous epoch's + /// numbering forward would read exactly like a lost first transaction of the new stream. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{2, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); + + /// The successor applies; then the next epoch's first id does. + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{1, 2})); + + /// Crossing into a new epoch needs INV-2's chain link as well as INV-1's id: without it the reader + /// cannot tell an EMPTY epoch from a lost one, so a sequence-1 transaction that names no seal is + /// refused on a Live table. Both halves are pinned, since either alone would be silently weaker. + RefLogTxn crossing = makeTxn(kNs, RefTxnId{2, 1}, {addPrecommitOp("b", manifestRef(2, 1, 1))}); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { applyRefLogTxn(state, crossing); }); + crossing.prev_epoch_seal = RefTxnId{1, 3}; /// the seal that closed epoch 1, one past its last id + applyRefLogTxn(state, crossing); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{2, 1})); +} + +/// =================================================================================== +/// snapshotOf: canonical sort + terminal-state refusal +/// =================================================================================== + +TEST(CASRefStateMachine, SnapshotOfSortsCommittedAndPrecommits) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("zzz", manifestRef(1, 3, 1)), addPrecommitOp("aaa", manifestRef(1, 1, 1)), + promoteOp("aaa", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("mmm", manifestRef(1, 2, 1))})); + + const RefTableSnapshot snap = snapshotOf(state, kNs); + ASSERT_EQ(snap.committed.size(), 1u); + EXPECT_EQ(snap.committed[0].ref_name, "aaa"); + ASSERT_EQ(snap.precommits.size(), 2u); + EXPECT_EQ(snap.precommits[0].ref_name, "mmm"); + EXPECT_EQ(snap.precommits[1].ref_name, "zzz"); + EXPECT_EQ(snap.snapshot_id, (RefTxnId{1, 2})); + + /// The result must actually be encodable (canonical shape) -- a real round trip through the codec. + const String bytes = encodeRefTableSnapshot(snap); + const RefTableSnapshot decoded = decodeRefTableSnapshot(bytes, kNs, snap.snapshot_id); + EXPECT_EQ(decoded, snap); +} + +TEST(CASRefStateMachine, SnapshotOfRefusesTerminalState) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {removeNamespaceOp()})); + + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Removed); + ASSERT_TRUE(state.getRemoveTxnId().has_value()); + EXPECT_EQ(*state.getRemoveTxnId(), (RefTxnId{1, 2})); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { snapshotOf(state, kNs); }); +} + +/// =================================================================================== +/// replay: TableState = Replay(S_X.state, tail(X)) +/// =================================================================================== + +TEST(CASRefStateMachine, ReplayFromNoSnapshot) +{ + std::vector tail{ + makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))}), + makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))}), + }; + const RefTableState state = replay(std::nullopt, tail); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Live); + EXPECT_TRUE(state.getCommitted().contains("a")); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{1, 2})); +} + +TEST(CASRefStateMachine, ReplayFromSnapshotPlusTail) +{ + RefTableState built; + applyRefLogTxn(built, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + const RefTableSnapshot snap = snapshotOf(built, kNs); + + std::vector tail{makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))})}; + const RefTableState state = replay(snap, tail); + EXPECT_TRUE(state.getCommitted().contains("a")); + EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{1, 2})); +} + +TEST(CASRefStateMachine, StateFromSnapshotConstructsLiveState) +{ + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + + const RefTableState state = stateFromSnapshot(snap); + EXPECT_EQ(state.getLifecycle(), RefLifecycle::Live); + EXPECT_FALSE(state.getRemoveTxnId().has_value()); +} + +TEST(CASRefStateMachine, ReplayRejectsTailNsMismatchAgainstSnapshot) +{ + RefTableState built; + applyRefLogTxn(built, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + const RefTableSnapshot snap = snapshotOf(built, kNs); + + std::vector tail{makeTxn("other-ns", RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, tail); }); +} + +TEST(CASRefStateMachine, ReplayRejectsTailNsMismatchAcrossEntries) +{ + std::vector tail{ + makeTxn("ns-a", RefTxnId{1, 1}, {birthOp()}), + makeTxn("ns-b", RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))}), + }; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(std::nullopt, tail); }); +} + +TEST(CASRefStateMachine, ReplayRejectsHandBuiltSnapshotWithDuplicateCommittedName) +{ + /// A hand-built RefTableSnapshot (never passed through decodeRefTableSnapshot -- exactly what + /// fsck hands to replay) with two committed rows sharing one ref_name must be rejected, not + /// silently collapsed to one row via std::map::emplace (the phantom-alive class of bug fixed in + /// stateFromSnapshot). + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row1; + row1.ref_name = "a"; + row1.manifest_ref = manifestRef(1, 1, 1); + RefCommittedRow row2; + row2.ref_name = "a"; + row2.manifest_ref = manifestRef(1, 2, 1); + snap.committed.push_back(row1); + snap.committed.push_back(row2); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, {}); }); +} + +TEST(CASRefStateMachine, ReplayRejectsHandBuiltSnapshotWithUnsortedPrecommits) +{ + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "b", manifestRef(1, 1, 1)}); + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "a", manifestRef(1, 2, 1)}); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, {}); }); +} + +/// Randomized replay equation: replay(snapshotOf(mid-state), tail) == full replay (spec §Table State). +TEST(CASRefStateMachine, ReplayEquationPropertyTest) +{ + std::mt19937 rng(4242); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + const std::vector names{"a", "b", "c"}; + + for (int trial = 0; trial < 30; ++trial) + { + std::vector history; + uint64_t seq = 1; + history.push_back(makeTxn(kNs, RefTxnId{1, seq++}, {birthOp()})); + + /// Track our own model of legal next actions so every generated op is guaranteed valid -- + /// this test exercises the replay equation, not the rejection paths (covered above). + std::vector> open_precommits; + std::vector> open_committed; + uint64_t next_build_seq = 1; + + const int steps = 15; + for (int step = 0; step < steps; ++step) + { + const uint32_t choice = rng() % 4; + if (choice == 0 || (open_precommits.empty() && open_committed.empty())) + { + /// Add precommit under a fresh manifest_ref (never collides, so always legal). + const String & name = names[rng() % names.size()]; + const ManifestRef mref = manifestRef(1, next_build_seq++, 1); + history.push_back(makeTxn(kNs, RefTxnId{1, seq++}, {addPrecommitOp(name, mref)})); + open_precommits.emplace_back(name, mref); + } + else if (choice == 1 && !open_precommits.empty()) + { + /// Only a name NOT already committed is eligible for a BARE promote: promoting into an + /// already-committed name requires an explicit prior removal in the same transaction + /// (spec §Promote; see PromoteRejectsDisplacingAnotherCommittedManifest) -- a distinct + /// scenario from the one this equation test exercises. + std::vector eligible; + for (size_t i = 0; i < open_precommits.size(); ++i) + { + const bool already_committed = std::any_of(open_committed.begin(), open_committed.end(), + [&](const auto & c) { return c.first == open_precommits[i].first; }); + if (!already_committed) + eligible.push_back(i); + } + if (!eligible.empty()) + { + const size_t idx = eligible[rng() % eligible.size()]; + const auto [name, mref] = open_precommits[idx]; + open_precommits.erase(open_precommits.begin() + static_cast(idx)); + history.push_back(makeTxn(kNs, RefTxnId{1, seq++}, {promoteOp(name, mref)})); + open_committed.emplace_back(name, mref); + } + } + else if (choice == 2 && !open_committed.empty()) + { + const size_t idx = rng() % open_committed.size(); + const auto & [name, mref] = open_committed[idx]; + const uint64_t this_id = seq++; + history.push_back(makeTxn(kNs, RefTxnId{1, this_id}, + {setPublishedAtOp(name, mref, this_id)})); + } + else if (!open_precommits.empty()) + { + const size_t idx = rng() % open_precommits.size(); + const auto [name, mref] = open_precommits[idx]; + open_precommits.erase(open_precommits.begin() + static_cast(idx)); + history.push_back(makeTxn(kNs, RefTxnId{1, seq++}, {removePrecommitOp(name, mref)})); + } + else if (!open_committed.empty()) + { + const size_t idx = rng() % open_committed.size(); + const auto [name, mref] = open_committed[idx]; + open_committed.erase(open_committed.begin() + static_cast(idx)); + history.push_back(makeTxn(kNs, RefTxnId{1, seq++}, {removeCommittedOp(name, mref)})); + } + } + + const RefTableState full = replay(std::nullopt, history); + + const size_t cut = rng() % (history.size() + 1); + const std::vector head(history.begin(), history.begin() + static_cast(cut)); + const std::vector tail(history.begin() + static_cast(cut), history.end()); + const RefTableState mid = replay(std::nullopt, head); + const std::optional mid_snapshot = + cut == 0 ? std::nullopt : std::make_optional(snapshotOf(mid, kNs)); + const RefTableState resumed = replay(mid_snapshot, tail); + + expectStatesEqual(full, resumed); + } +} + +/// =================================================================================== +/// Fail-closed replay + snapshot validation: a corrupted history or snapshot naming one manifest under +/// two owners must be REJECTED in EVERY build (post-consult). The cross-owner uniqueness check is O(1) +/// via `owned_manifests`, so it runs unconditionally -- on the writer's append path AND on replay -- +/// rather than being elided into a debug-only assertion. `stateFromSnapshot` enforces the same +/// invariant across snapshot rows (the codec never did). +/// =================================================================================== + +/// (Add path, committed collision) The writer's append-time contract rejects a fresh precommit that +/// names a manifest already committed under a DIFFERENT ref_name, and leaves the state unchanged. +TEST(CASRefStateMachine, LiveAppendRejectsAddPrecommitCollidingWithCommitted) +{ + RefTableState state = stateFromSnapshot(buildCollidingBaseSnapshotForTest()); + const RefTableState before = state; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("b", manifestRef(1, 1, 1))})); }); + expectStatesEqual(before, state); +} + +/// (Replay path, committed collision) A tail whose add-precommit collides cross-owner with an existing +/// committed owner makes `replay` THROW -- it must NOT be silently accepted. This is the exact behavior +/// the deleted `TrustedReplaySkipsCrossOwnerScanInRelease` test pinned as *desired*; post-consult it is +/// the opposite: fail closed. +TEST(CASRefStateMachine, ReplayRejectsTailAddPrecommitCollidingWithCommitted) +{ + const RefTableSnapshot snap = buildCollidingBaseSnapshotForTest(); + const std::vector tail{makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("b", manifestRef(1, 1, 1))})}; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, tail); }); +} + +/// (Replay path, precommit collision) The same, but the base already holds a PRECOMMIT for the manifest +/// and the tail adds a second precommit for it under another ref_name (precommit/precommit collision). +TEST(CASRefStateMachine, ReplayRejectsTailAddPrecommitCollidingWithPrecommit) +{ + const std::vector tail{ + makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))}), + makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("b", manifestRef(1, 1, 1))}), // collides cross-owner + }; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(std::nullopt, tail); }); +} + +/// (Snapshot validation, committed/committed) A hand-built snapshot with two committed rows naming ONE +/// manifest passes the codec (it checks only sortedness + no-duplicate ref_name) but must be rejected by +/// `stateFromSnapshot`/`replay` as semantically corrupt. +TEST(CASRefStateMachine, ReplayRejectsSnapshotWithTwoCommittedRowsNamingOneManifest) +{ + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row1; + row1.ref_name = "a"; + row1.manifest_ref = manifestRef(1, 1, 1); + RefCommittedRow row2; + row2.ref_name = "b"; // distinct ref_name (codec-legal)... + row2.manifest_ref = manifestRef(1, 1, 1); // ...but the SAME manifest (corrupt) + snap.committed.push_back(row1); + snap.committed.push_back(row2); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, {}); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)stateFromSnapshot(snap); }); +} + +/// (Snapshot validation, committed/precommit) A committed row and a precommit binding sharing one +/// manifest -- also codec-legal (different owner kinds, sorted independently) but corrupt. +TEST(CASRefStateMachine, ReplayRejectsSnapshotWithCommittedAndPrecommitSharingManifest) +{ + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow row; + row.ref_name = "a"; + row.manifest_ref = manifestRef(1, 1, 1); + snap.committed.push_back(row); + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "b", manifestRef(1, 1, 1)}); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, {}); }); +} + +/// (Snapshot validation, precommit/precommit) Two precommit bindings under different ref_names naming +/// one manifest -- sorted by (ref_name, manifest_ref), so codec-legal, but corrupt. +TEST(CASRefStateMachine, ReplayRejectsSnapshotWithTwoPrecommitsSharingManifest) +{ + RefTableSnapshot snap; + snap.ns = kNs; + snap.snapshot_id = RefTxnId{1, 1}; + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "a", manifestRef(1, 1, 1)}); + snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "b", manifestRef(1, 1, 1)}); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(snap, {}); }); +} + +/// Positive equivalence: a VALID tail replayed via `replay` (the in-place trusted path) produces a state +/// byte-identical (getters + encoded snapshot) to the same tail applied via the public strong-guarantee +/// `applyRefLogTxn` -- the apply strategy changes nothing a legal transaction produces. +TEST(CASRefStateMachine, TrustedReplayEquivalentToLiveAppendOnValidTail) +{ + const std::vector tail{ + makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))}), + makeTxn(kNs, RefTxnId{1, 2}, + {promoteOp("a", manifestRef(1, 1, 1)), addPrecommitOp("b", manifestRef(1, 2, 1))}), + makeTxn(kNs, RefTxnId{1, 3}, {setPublishedAtOp("a", manifestRef(1, 1, 1), 7)}), + makeTxn(kNs, RefTxnId{1, 4}, {promoteOp("b", manifestRef(1, 2, 1))}), + }; + + RefTableState full_state; + for (const RefLogTxn & txn : tail) + applyRefLogTxn(full_state, txn); // LiveAppend (default) + + const RefTableState trusted_state = replay(std::nullopt, tail); // replay uses the in-place trusted path internally + + expectStatesEqual(full_state, trusted_state); + EXPECT_EQ(encodeRefTableSnapshot(snapshotOf(full_state, kNs)), + encodeRefTableSnapshot(snapshotOf(trusted_state, kNs))); +} + +/// =================================================================================== +/// E3: apply strategy per validation mode +/// - LiveAppend: two-phase scratch copy, "throw => state byte-for-byte unchanged" +/// - TrustedReplay (replay): in-place, poison-on-throw, discarded by the sole caller +/// =================================================================================== + +namespace +{ +/// A populated, MATERIALIZED Live state -- committed "a"->(1,1,1) plus a pending precommit +/// ("p",(1,2,1)) -- built through the public LiveAppend path, then materialized so its COW overlays are +/// empty (exactly the shape the writer's live state has at each flush boundary). The E3 LiveAppend-path +/// tests mutate a COPY of this and assert the original-equivalent captured bytes/getters are intact +/// after a rejected transaction. +RefTableState buildPopulatedLiveState() +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1)), + addPrecommitOp("p", manifestRef(1, 2, 1))})); + state.materializeCommitted(); + return state; +} +} + +/// LiveAppend-path atomicity, LATER-op throw ("populated" abort path): the first two ops touch committed, +/// precommits, the owned-manifest index and the body counters; the third is illegal. The whole +/// transaction is rejected and `state` is byte-for-byte unchanged -- getters AND encoded-snapshot +/// bytes. This is the writer's live-state contract, preserved verbatim by E3's `LiveAppend` arm. +TEST(CASRefStateMachine, E3LiveAppendLaterOpThrowLeavesPopulatedStateByteIdentical) +{ + RefTableState state = buildPopulatedLiveState(); + const RefTableState before = state; + const String before_bytes = encodeRefTableSnapshot(snapshotOf(state, kNs)); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, { + addPrecommitOp("q", manifestRef(1, 3, 1)), // touches precommits + index + counters + removeCommittedOp("a", manifestRef(1, 1, 1)), // touches committed + index + counters + removePrecommitOp("absent", manifestRef(9, 9, 9)) // ILLEGAL: exact binding absent -> throws + })); }); + + expectStatesEqual(before, state); + EXPECT_EQ(before_bytes, encodeRefTableSnapshot(snapshotOf(state, kNs))); + /// Neither surviving-looking earlier op leaked into the live state. + EXPECT_FALSE(state.getPrecommits().contains({"q", manifestRef(1, 3, 1)})); + EXPECT_TRUE(state.getCommitted().contains("a")); +} + +/// LiveAppend-path atomicity, FIRST-op throw ("empty" abort path -- nothing applied before the throw): the +/// symmetric guarantee still holds. Distinct from the case above because no op ever mutated the +/// scratch, exercising the throw-before-any-effect branch. +TEST(CASRefStateMachine, E3LiveAppendFirstOpThrowLeavesPopulatedStateByteIdentical) +{ + RefTableState state = buildPopulatedLiveState(); + const RefTableState before = state; + const String before_bytes = encodeRefTableSnapshot(snapshotOf(state, kNs)); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, { + removeCommittedOp("absent", manifestRef(9, 9, 9)), // ILLEGAL first op + addPrecommitOp("q", manifestRef(1, 3, 1)) + })); }); + + expectStatesEqual(before, state); + EXPECT_EQ(before_bytes, encodeRefTableSnapshot(snapshotOf(state, kNs))); +} + +/// `admits` previews an op against `state` and must leave it byte-for-byte unchanged whether the op +/// fits (true) or overflows (false) -- it is a pure query. Verified against both getters and encoded +/// bytes, for both the accept and the reject verdicts. +TEST(CASRefStateMachine, E3AdmitsPreviewLeavesStateByteIdentical) +{ + RefTableState state = buildPopulatedLiveState(); + /// Must be an independent snapshot -- `state` is queried and potentially mutated below, and + /// comparing against a reference would make the check vacuous. + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization) + const RefTableState before = state; + const String before_bytes = encodeRefTableSnapshot(snapshotOf(state, kNs)); + + const RefOp grow = addPrecommitOp("q", manifestRef(1, 3, 1)); + + /// Accept verdict (ample budget): state untouched. + EXPECT_TRUE(admits(state, grow, 1'000'000, 1'000'000)); + expectStatesEqual(before, state); + EXPECT_EQ(before_bytes, encodeRefTableSnapshot(snapshotOf(state, kNs))); + + /// Reject verdict (snapshot budget one byte short of the grown size): state STILL untouched. + RefTableState grown = state; + applyRefLogTxn(grown, makeTxn(kNs, RefTxnId{1, 2}, {grow})); + const size_t grown_size = encodeRefTableSnapshot(snapshotOf(grown, "")).size(); + EXPECT_FALSE(admits(state, grow, grown_size - 1, 1'000'000)); + expectStatesEqual(before, state); + EXPECT_EQ(before_bytes, encodeRefTableSnapshot(snapshotOf(state, kNs))); +} + +/// TrustedReplay in-place apply, SUCCESS path across every `applyOp` arm: a tail that births, adds, +/// promotes, replaces a committed manifest, removes a committed and a precommit, restamps a timestamp, +/// and finally removes the namespace, replayed via `replay` (TrustedReplay, in place) must produce a +/// state byte-identical to the SAME tail applied op-by-op through `LiveAppend` (scratch copy). This is the +/// test only E3's in-place machinery can fail: a mis-maintained counter, a dropped owned-manifest +/// index entry, or a lost `greatest_applied` update on the no-copy path would diverge here. +TEST(CASRefStateMachine, E3TrustedReplayInPlaceMatchesLiveAppendAcrossAllArms) +{ + const std::vector tail{ + makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), + addPrecommitOp("a", manifestRef(1, 1, 1)), addPrecommitOp("b", manifestRef(1, 2, 1))}), + makeTxn(kNs, RefTxnId{1, 2}, { + promoteOp("a", manifestRef(1, 1, 1)), // precommit -> committed + setPublishedAtOp("a", manifestRef(1, 1, 1), 42)}), // restamp published_at_ms + makeTxn(kNs, RefTxnId{1, 3}, {removePrecommitOp("b", manifestRef(1, 2, 1))}), // drop precommit + makeTxn(kNs, RefTxnId{1, 4}, { + removeCommittedOp("a", manifestRef(1, 1, 1)), // evict stale committed... + addPrecommitOp("a", manifestRef(1, 9, 1)), // ...then re-add under same name + promoteOp("a", manifestRef(1, 9, 1))}), // and promote the replacement + makeTxn(kNs, RefTxnId{1, 5}, { + removeCommittedOp("a", manifestRef(1, 9, 1)), // drain the last owner... + removeNamespaceOp()}), // ...then remove the namespace + }; + + RefTableState full_state; + for (const RefLogTxn & txn : tail) + applyRefLogTxn(full_state, txn); // LiveAppend (default): two-phase scratch copy + + const RefTableState replayed = replay(std::nullopt, tail); // TrustedReplay in-place + + expectStatesEqual(full_state, replayed); + EXPECT_THROW(encodeRefTableSnapshot(snapshotOf(full_state, kNs)), DB::Exception); + EXPECT_THROW(encodeRefTableSnapshot(snapshotOf(replayed, kNs)), DB::Exception); + EXPECT_EQ(replayed.getLifecycle(), RefLifecycle::Removed); + EXPECT_EQ(replayed.getRemoveTxnId(), std::make_optional(RefTxnId{1, 5})); +} + +/// TrustedReplay in-place apply, THROW path: a tail whose LAST transaction is illegal makes `replay` +/// throw `CORRUPTED_DATA`. The in-place apply poisons a state that is entirely internal to the failed +/// `replay` call (it is never assigned to a caller on a throw), so an INDEPENDENT replay of just the +/// valid prefix is completely unaffected -- pinning that the poison never escapes. +TEST(CASRefStateMachine, E3TrustedReplayPoisonOnBadTailIsInternal) +{ + const std::vector good_prefix{ + makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))}), + makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))}), + }; + std::vector bad_tail = good_prefix; + /// A third txn whose op removes an absent precommit -- legal txn_id ordering, illegal effect, so it + /// throws mid-apply AFTER the good prefix has already been applied in place to the internal state. + bad_tail.push_back(makeTxn(kNs, RefTxnId{1, 3}, {removePrecommitOp("absent", manifestRef(9, 9, 9))})); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { replay(std::nullopt, bad_tail); }); + + /// The failed replay's poisoned internal state never leaked: a fresh replay of the valid prefix is + /// byte-identical to one built entirely via LiveAppend, and reflects exactly the prefix. + const RefTableState from_prefix = replay(std::nullopt, good_prefix); + RefTableState full_prefix; + for (const RefLogTxn & txn : good_prefix) + applyRefLogTxn(full_prefix, txn); + expectStatesEqual(full_prefix, from_prefix); + EXPECT_TRUE(from_prefix.getCommitted().contains("a")); + EXPECT_EQ(from_prefix.getGreatestApplied(), (RefTxnId{1, 2})); +} + +/// =================================================================================== +/// admits(): dual-bound admission budget (spec §Snapshot Format) +/// =================================================================================== + +TEST(CASRefStateMachine, AdmitsAcceptsWellUnderBudget) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + EXPECT_TRUE(admits(state, addPrecommitOp("a", manifestRef(1, 1, 1)), 1'000'000, 1'000'000)); +} + +TEST(CASRefStateMachine, AdmitsRejectsGrowthPastSnapshotBudgetOwnerTransitionAdd) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + + const RefOp op = addPrecommitOp("a", manifestRef(1, 1, 1)); + RefTableState scratch = state; + applyRefLogTxn(scratch, makeTxn(kNs, RefTxnId{1, 2}, {op})); + const size_t true_size = encodeRefTableSnapshot(snapshotOf(scratch, "")).size(); + + EXPECT_TRUE(admits(state, op, true_size, 1'000'000)); + EXPECT_FALSE(admits(state, op, true_size - 1, 1'000'000)); +} + +TEST(CASRefStateMachine, AdmitsRejectsGrowthPastSnapshotBudgetSetPublishedAt) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + + const RefOp op = setPublishedAtOp("a", manifestRef(1, 1, 1), 1700000000000ull); + RefTableState scratch = state; + applyRefLogTxn(scratch, makeTxn(kNs, RefTxnId{1, 2}, {op})); + const size_t true_size = encodeRefTableSnapshot(snapshotOf(scratch, "")).size(); + + EXPECT_TRUE(admits(state, op, true_size, 1'000'000)); + EXPECT_FALSE(admits(state, op, true_size - 1, 1'000'000)); +} + +TEST(CASRefStateMachine, AdmitsRejectsGrowthPastSnapshotBudgetPromoteWithSetPublishedAt) +{ + /// The "promote-with-set_published_at" growth class: the owner_transition half of a promote is + /// admitted cheaply (published_at_ms starts unset), but the immediately-following set_published_at + /// that installs the REAL initial timestamp is where the growth actually happens. + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {promoteOp("a", manifestRef(1, 1, 1))})); + ASSERT_TRUE(state.getCommitted().contains("a")); + ASSERT_EQ(state.getCommitted().at("a").published_at_ms, 0u); + + const RefOp op = setPublishedAtOp("a", manifestRef(1, 1, 1), 1700000000099ull); + RefTableState scratch = state; + applyRefLogTxn(scratch, makeTxn(kNs, RefTxnId{1, 3}, {op})); + const size_t true_size = encodeRefTableSnapshot(snapshotOf(scratch, "")).size(); + + EXPECT_TRUE(admits(state, op, true_size, 1'000'000)); + EXPECT_FALSE(admits(state, op, true_size - 1, 1'000'000)); +} + +TEST(CASRefStateMachine, AdmitsRejectsGrowthPastRemovalBudget) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), addPrecommitOp("a", manifestRef(1, 1, 1)), promoteOp("a", manifestRef(1, 1, 1))})); + + const RefOp op = setPublishedAtOp("a", manifestRef(1, 1, 1), 1700000000300ull); + RefTableState scratch = state; + applyRefLogTxn(scratch, makeTxn(kNs, RefTxnId{1, 2}, {op})); + const String removal_bytes = encodeRefLogTxn(buildRemovalTxnForTest(scratch, "", RefTxnId{1, 1})); + const size_t true_removal_size = removal_bytes.size(); + + /// A generous snapshot budget isolates the removal-budget bound specifically. + EXPECT_TRUE(admits(state, op, 1'000'000, true_removal_size)); + EXPECT_FALSE(admits(state, op, 1'000'000, true_removal_size - 1)); +} + +/// Randomized exactness property test: admits()'s internal size computation must exactly match the +/// real encoders' output, for both bounds, across randomized states and candidate growing ops. +TEST(CASRefStateMachine, AdmitsExactnessPropertyTest) +{ + std::mt19937 rng(777); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed is required for reproducible property coverage. + + for (int trial = 0; trial < 20; ++trial) + { + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + uint64_t seq = 2; + uint64_t next_build_seq = 1; + std::vector> open_precommits; + std::vector> open_committed; + + /// Build up a random but valid mid-state (a handful of precommits/committed rows/timestamps). + const int setup_steps = 1 + static_cast(rng() % 5); + for (int i = 0; i < setup_steps; ++i) + { + const String name = "ref" + std::to_string(rng() % 4); + const ManifestRef mref = manifestRef(1, next_build_seq++, 1); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, seq++}, {addPrecommitOp(name, mref)})); + open_precommits.emplace_back(name, mref); + + /// A bare promote may not target a name already committed under a different manifest + /// (spec §Promote; see PromoteRejectsDisplacingAnotherCommittedManifest) -- skip promoting + /// this iteration's precommit when an earlier iteration already committed the same name. + const bool name_already_committed = std::any_of(open_committed.begin(), open_committed.end(), + [&](const auto & c) { return c.first == name; }); + if (!name_already_committed && rng() % 2 == 0) + { + const auto [pname, pmref] = open_precommits.back(); + open_precommits.pop_back(); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, seq++}, {promoteOp(pname, pmref)})); + open_committed.emplace_back(pname, pmref); + } + } + + /// Pick a random candidate growing op against this state. + RefOp candidate; + const uint32_t kind = rng() % 3; + if (kind == 0 || open_committed.empty()) + { + candidate = addPrecommitOp("fresh-" + std::to_string(trial), manifestRef(1, next_build_seq++, 1)); + } + else if (kind == 1) + { + const auto & [name, mref] = open_committed[rng() % open_committed.size()]; + candidate = setPublishedAtOp(name, mref, rng()); + } + else + { + /// A genuinely distinct third shape: a racing precommit under an ALREADY-committed name + /// (legal -- spec §Add Precommit only restricts manifest identity, never ref_name). + const String & name = open_committed[rng() % open_committed.size()].first; + candidate = addPrecommitOp(name, manifestRef(1, next_build_seq++, 1)); + } + + RefTableState scratch = state; + applyRefLogTxn(scratch, makeTxn(kNs, RefTxnId{1, seq}, {candidate})); + const size_t true_snapshot_size = encodeRefTableSnapshot(snapshotOf(scratch, "")).size(); + const size_t true_removal_size = + encodeRefLogTxn(buildRemovalTxnForTest(scratch, "", RefTxnId{1, 1})).size(); + + EXPECT_TRUE(admits(state, candidate, true_snapshot_size, true_removal_size)); + EXPECT_FALSE(admits(state, candidate, true_snapshot_size - 1, true_removal_size)); + EXPECT_FALSE(admits(state, candidate, true_snapshot_size, true_removal_size - 1)); + } +} + +/// =================================================================================== +/// Snapshot size helpers: framing + Σ per-row must equal a full encode, byte for byte. +/// =================================================================================== +TEST(CASRefSnapshotSizeHelpers, FramingPlusRowsEqualsFullEncode) +{ + /// Build a non-trivial Live table: two committed rows (one with a stamped published_at_ms) and one + /// precommit. + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), + addPrecommitOp("alpha", manifestRef(1, 1, 1)), promoteOp("alpha", manifestRef(1, 1, 1)), + addPrecommitOp("beta", manifestRef(1, 2, 1)), promoteOp("beta", manifestRef(1, 2, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, + {setPublishedAtOp("alpha", manifestRef(1, 1, 1), 42)})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {addPrecommitOp("gamma", manifestRef(1, 3, 1))})); + + const RefTableSnapshot snap = snapshotOf(state, ""); + const size_t full = encodeRefTableSnapshot(snap).size(); + + size_t rebuilt = snapshotFramingSize("", snap.snapshot_id, snap.committed.size() + snap.precommits.size()); + for (const RefCommittedRow & row : snap.committed) + rebuilt += committedRowEncodedSize(row); + for (const RefOwnerBinding & pc : snap.precommits) + rebuilt += precommitRowEncodedSize(pc); + + EXPECT_EQ(rebuilt, full); +} + +/// =================================================================================== +/// Removal-txn size helpers: framing + Σ per-owner-op must equal a full removal-txn encode. +/// =================================================================================== +TEST(CASRefLogSizeHelpers, FramingPlusOpsEqualsFullRemovalEncode) +{ + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, + {birthOp(), + addPrecommitOp("alpha", manifestRef(1, 1, 1)), promoteOp("alpha", manifestRef(1, 1, 1)), + addPrecommitOp("beta", manifestRef(1, 2, 1))})); + + /// Ground truth: the whole-namespace removal txn this test file already builds independently. + const RefLogTxn removal = buildRemovalTxnForTest(state, "", RefTxnId{1, 1}); + const size_t full = encodeRefLogTxn(removal).size(); + + size_t rebuilt = removalFramingSize("", RefTxnId{1, 1}, + state.getCommitted().size() + state.getPrecommits().size() + 1); + for (const auto [name, row] : state.getCommitted()) + rebuilt += removalOpEncodedSize(RefOwnerKind::Committed, name, row.manifest_ref); + for (const auto & [name, mref] : state.getPrecommits()) + rebuilt += removalOpEncodedSize(RefOwnerKind::Precommit, name, mref); + + EXPECT_EQ(rebuilt, full); +} + +/// =================================================================================== +/// Body-byte counters: snapshot_body_bytes / removal_body_bytes are a pure function of the rows. +/// =================================================================================== +namespace +{ +uint64_t recomputeSnapshotBody(const RefTableState & s) +{ + uint64_t total = 0; + for (const auto [name, row] : s.getCommitted()) + total += committedRowEncodedSize(row); + for (const auto & [name, mref] : s.getPrecommits()) + total += precommitRowEncodedSize(RefOwnerBinding{RefOwnerKind::Precommit, name, mref}); + return total; +} +uint64_t recomputeRemovalBody(const RefTableState & s) +{ + uint64_t total = 0; + for (const auto [name, row] : s.getCommitted()) + total += removalOpEncodedSize(RefOwnerKind::Committed, name, row.manifest_ref); + for (const auto & [name, mref] : s.getPrecommits()) + total += removalOpEncodedSize(RefOwnerKind::Precommit, name, mref); + return total; +} +} + +TEST(CASRefStateCounters, CountersTrackRowsThroughEveryOpKind) +{ + RefTableState state; + EXPECT_EQ(state.getSnapshotBodyBytes(), 0u); + EXPECT_EQ(state.getRemovalBodyBytes(), 0u); + + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 2}, {addPrecommitOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 3}, {promoteOp("a", manifestRef(1, 1, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 4}, + {setPublishedAtOp("a", manifestRef(1, 1, 1), 5)})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 5}, {addPrecommitOp("b", manifestRef(1, 2, 1))})); + EXPECT_EQ(state.getSnapshotBodyBytes(), recomputeSnapshotBody(state)); + EXPECT_EQ(state.getRemovalBodyBytes(), recomputeRemovalBody(state)); + + /// Shrink back down: remove the precommit, then the committed row. + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 6}, {removePrecommitOp("b", manifestRef(1, 2, 1))})); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 7}, {removeCommittedOp("a", manifestRef(1, 1, 1))})); + EXPECT_EQ(state.getSnapshotBodyBytes(), recomputeSnapshotBody(state)); + EXPECT_EQ(state.getRemovalBodyBytes(), recomputeRemovalBody(state)); + EXPECT_EQ(state.getSnapshotBodyBytes(), 0u); + EXPECT_EQ(state.getRemovalBodyBytes(), 0u); +} + +/// =================================================================================== +/// Budget-size accessors equal the real encoders across randomized states. +/// =================================================================================== +TEST(CASRefBudgetSize, AccessorsEqualFullEncodeRandomized) +{ + std::mt19937 rng(1234); // NOLINT(cert-msc32-c,cert-msc51-cpp): deterministic seed for reproducibility. + for (int trial = 0; trial < 30; ++trial) + { + RefTableState state; + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, 1}, {birthOp()})); + uint64_t seq = 2; + uint64_t build = 1; + std::vector> committed_names; + + const int steps = 1 + static_cast(rng() % 6); + for (int i = 0; i < steps; ++i) + { + const String name = "r" + std::to_string(rng() % 5); + const ManifestRef mref = manifestRef(1, build++, 1); + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, seq++}, {addPrecommitOp(name, mref)})); + const bool already = std::any_of(committed_names.begin(), committed_names.end(), + [&](const auto & c) { return c.first == name; }); + if (!already && rng() % 2 == 0) + { + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, seq++}, {promoteOp(name, mref)})); + committed_names.emplace_back(name, mref); + if (rng() % 2 == 0) + applyRefLogTxn(state, makeTxn(kNs, RefTxnId{1, seq++}, + {setPublishedAtOp(name, mref, rng())})); + } + } + + const size_t true_snapshot = encodeRefTableSnapshot(snapshotOf(state, "")).size(); + const size_t true_removal = encodeRefLogTxn(buildRemovalTxnForTest(state, "", RefTxnId{1, 1})).size(); + EXPECT_EQ(encodedSnapshotBudgetSize(state), true_snapshot); + EXPECT_EQ(encodedRemovalBudgetSize(state), true_removal); + } +} diff --git a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp new file mode 100644 index 000000000000..ace02d24836e --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp @@ -0,0 +1,1476 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +/// ================================================================================================ +/// Task 4 (2026-07-28 CAS ref-chain Stage A streams, spec INV-1's every-attempt rule + INV-2's seal): +/// the writer wedge. +/// +/// An id is freed only when NOTHING WAS SENT, or when every sent attempt has its own CONCLUSIVE +/// rejection. That is what these tests are about, and the second half is the part that changed: an +/// ambiguous attempt used to be resolved by a bare exact GET, which can only ever report "absent" -- +/// and absent is not a rejection, because the ambiguous attempt may still land afterwards. The lane +/// therefore stayed wedged FOREVER over a key nothing had written. The rule now runs one bounded +/// `slotOccupy` per later caller's flush: the ref-log key is write-once, so a conditional CREATE of +/// the SAME bytes either makes the transaction durable (adopt it) or conflicts with whatever is +/// there, which the follow-up read then names -- our own earlier write (adopt), a successor's +/// `EpochSeal` (the operation is conclusively rejected and never was acked), or a foreign object +/// (impossible under mount-lease exclusivity: fail loud). +/// +/// Two cross-cutting rules are exercised throughout rather than in one place: +/// - the ADMISSION FENCE: a wedge carries the mount-fence generation it was admitted under, every +/// retry is gated on THAT generation (never the current one), and every result is re-checked +/// under `state_mutex` before anything acts on it. A result that returns after a fence +/// bump/re-arm, or after the wedge it belonged to was replaced, must be INERT. +/// - `prev_epoch_seal`: a seal observed at the wedged key IS this namespace's epoch-closing record, +/// so it becomes the `prev_epoch_seal` the next sequence-1 append carries. `nullopt` means +/// genesis, and means it exactly. +/// ================================================================================================ + +namespace ProfileEvents +{ +extern const Event CASRefAppendSealRejected; +extern const Event CASRefAppendOccupantUnreadable; +extern const Event CASRefAppendWedged; +extern const Event CASRefAppendDefiniteFailure; +} + +namespace DB::ErrorCodes +{ +extern const int BAD_ARGUMENTS; +extern const int CORRUPTED_DATA; +extern const int INVALID_STATE; +extern const int MEMORY_LIMIT_EXCEEDED; +extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::LandedButAckLostOnceBackend; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +PoolPtr openPool(const BackendPtr & backend, CasRequestBudget budget = {}) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); +} + +/// The budget every wedge test uses: ONE attempt, so a single injected ambiguity is the whole +/// operation and the lane wedges deterministically instead of retrying its way out. +CasRequestBudget singleAttemptBudget() +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) + budget.lease_safety_margin_ms = 100; + return budget; +} + +/// TWO attempts of one logical operation, with the inter-attempt backoff disabled. Everything about the +/// call-level verdict rule lives BETWEEN two attempts of a single call, so it cannot be reached with the +/// one-attempt budget the tests above use; the backoff is switched off because the schedule is +/// `gtest_cas_request_control.cpp`'s subject and a real sleep here would only slow the suite. +/// +/// `[[maybe_unused]]`: both of its callers need a real S3-classified rejection to script their second +/// attempt, so they compile away entirely in a build without S3. +[[maybe_unused]] CasRequestBudget twoAttemptBudget() +{ + CasRequestBudget budget = singleAttemptBudget(); + budget.max_attempts = 2; + budget.retry_initial_backoff_ms = 0; + budget.retry_max_backoff_ms = 0; + return budget; +} + +PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + return s->beginPartWrite(info); +} + +void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + auto build = startBuildFor(s, ns, ref); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +/// A `CountingBackend` with the exact seams these tests need, all keyed by substring so a whole Pool's +/// bootstrap traffic never consumes a fault meant for a `_log/` PUT. +class WedgeTestBackend : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + using CountingBackend::get; + + /// One-shot ambiguity that writes NOTHING: the response is lost and the key stays absent, which is + /// the input that makes a later `slotOccupy` report `Created`. + String ambiguous_substr; + int ambiguous_count = 0; + + /// One-shot DETERMINISTIC LOCAL failure (`BAD_ARGUMENTS`, in `isDeterministicLocalFailure`'s set), + /// which `slotOccupy` rethrows unchanged -- a definite refusal of THIS attempt. Portable stand-in + /// for the S3 `DefiniteFailure` shape, which needs `USE_AWS_S3`; both are "proven never applied". + String definite_substr; + int definite_count = 0; + + /// One-shot WHITELISTED SYNCHRONOUS REJECTION: the ONLY shape `classifyConditionalWriteResult` + /// answers `DefiniteFailure` for, and therefore the only way to drive the append lane's definite + /// arm. Distinct from `definite_substr` above on purpose -- that one is a deterministic LOCAL + /// failure, which `slotOccupy` rethrows but `putIfAbsentControlled` (no such special case) merely + /// classifies Unresolved, so it cannot script this arm at all. + String s3_definite_substr; + int s3_definite_count = 0; + + /// A SUCCESSOR lands `conflict_bytes` at the key and only then is our response lost, so the + /// controller's resolve-before-reissue reads a different object and proves the conflict. This is how + /// the ordinary append site meets an occupant at the id it derived. + String conflict_substr; + int conflict_count = 0; + String conflict_bytes; + + /// Fail GETs of matching keys after skipping the first `fail_get_skip` of them -- the resolve read + /// that PROVES the conflict must succeed, so only the adjudication read that follows it is faulted. + String fail_get_substr; + int fail_get_skip = 0; + int fail_get_count = 0; + + String fail_cas_substr; + int fail_cas_count = 0; + + std::optional get(const String & key, Range range) override + { + if (fail_get_count > 0 && !fail_get_substr.empty() && key.find(fail_get_substr) != String::npos) + { + if (fail_get_skip > 0) + --fail_get_skip; + else + { + --fail_get_count; + throw Poco::TimeoutException("WedgeTestBackend: simulated lost GET (read response never arrived)"); + } + } + return CountingBackend::get(key, range); + } + + CasResult casPut(const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + if (fail_cas_count > 0 && !fail_cas_substr.empty() && key.find(fail_cas_substr) != String::npos) + { + --fail_cas_count; + throw Poco::TimeoutException("WedgeTestBackend: simulated ambiguous checkpoint CAS"); + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + + /// Park a matching PUT until `releaseBlock()`, notifying `awaitBlockEntered()` on arrival, so a + /// test can drive a fence bump or a successor's write into the exact I/O window. + void armBlock(const String & substr) + { + std::lock_guard g(block_mutex); + block_substr = substr; + block_armed = true; + block_entered = false; + } + void awaitBlockEntered() + { + std::unique_lock lk(block_mutex); + block_cv.wait(lk, [&] { return block_entered; }); + } + void releaseBlock() + { + { + std::lock_guard g(block_mutex); + block_armed = false; + } + block_cv.notify_all(); + } + + /// Write straight through, bypassing every fault and block seam above -- how a test models what a + /// SUCCESSOR (another process entirely) put at a key. Using the faulting entry point instead would + /// park the test's own write on the very gate it is trying to drive a scenario through. The + /// qualification must name the THREE-argument overload: `Backend`'s two-argument convenience + /// forwards to the VIRTUAL one, so `CountingBackend::putIfAbsent(key, bytes)` would dispatch right + /// back into the override above and deadlock the test against its own block. + PutResult putAsSuccessor(const String & key, const String & bytes) + { + return CountingBackend::putIfAbsent(key, bytes, ObjectMeta{}); + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (ambiguous_count > 0 && !ambiguous_substr.empty() && key.find(ambiguous_substr) != String::npos) + { + --ambiguous_count; + throw Poco::TimeoutException("WedgeTestBackend: simulated ambiguous PUT (response lost, nothing landed)"); + } + if (definite_count > 0 && !definite_substr.empty() && key.find(definite_substr) != String::npos) + { + --definite_count; + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "WedgeTestBackend: scripted deterministic local failure"); + } + /// AFTER the ambiguity seam, so arming both scripts one call's attempts in order: the first + /// attempt goes ambiguous, the reissue is definitively refused. + if (s3_definite_count > 0 && !s3_definite_substr.empty() && key.find(s3_definite_substr) != String::npos) + { + --s3_definite_count; +#if USE_AWS_S3 + throw DB::S3Exception("WedgeTestBackend: simulated malformed request", + Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); +#else + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, + "WedgeTestBackend: DefiniteFailure requires S3 error classification (USE_AWS_S3 off)"); +#endif + } + if (conflict_count > 0 && !conflict_substr.empty() && key.find(conflict_substr) != String::npos) + { + --conflict_count; + CountingBackend::putIfAbsent(key, conflict_bytes, meta); + throw Poco::TimeoutException("WedgeTestBackend: a successor's object landed; our response was lost"); + } + { + std::unique_lock lk(block_mutex); + if (block_armed && !block_substr.empty() && key.find(block_substr) != String::npos) + { + block_entered = true; + block_cv.notify_all(); + /// Bounded so a wiring bug bounds the wait instead of hanging the suite. + block_cv.wait_for(lk, std::chrono::seconds(20), [&] { return !block_armed; }); + } + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } + +private: + std::mutex block_mutex; + std::condition_variable block_cv; + String block_substr; + bool block_armed = false; + bool block_entered = false; +}; + +/// The `_log/` key prefix of one namespace -- what every fault seam here matches on. +String logPrefix(const PoolPtr & store, const RootNamespace & ns) +{ + return store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; +} + +CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns; + }); + if (it == catalog.entries.end()) + throw std::runtime_error("catalog entry missing from wedge fixture"); + return *it; +} + +/// Wedge tests address raw ref-log keys at Stage A's deterministic sentinel identity, but their +/// catalog fixture must still use production's `Creating -> _ckpt -> Live` birth order. A fixed +/// creator identity makes the durable genesis checkpoint deterministic too. +void admitProperlyBornEntry(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const CatalogEntry creating{ + .ns = ns, + .state = NsState::Creating, + .incarnation = DB::Cas::tests::fixture::fixtureLife(ns).incarnation, + .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}, + }; + CasRefCatalog::casAdmitEntry(backend, layout, /*gc_shards=*/1, creating); + + const CkptDeadline deadline{.now_ms = [] { return uint64_t{1000}; }, .deadline_ms = 60000}; + ASSERT_EQ( + CasRefCatalog::completeCreation( + backend, layout, creating, /*admitted_generation=*/1, [](uint64_t) {}, deadline), + CasRefCatalog::NamespaceCreationOutcome::Live); +} + +CatalogEntry replaceCatalogLifeForWedgeRace( + Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) +{ + const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + RefCatalog without_predecessor = before_delete.catalog; + std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) + { + return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; + }); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to retire exact predecessor catalog life"); + + CatalogEntry successor{ + .ns = predecessor.ns, + .state = NsState::Live, + .incarnation = successor_incarnation, + .creator = std::nullopt}; + const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + RefCatalog reborn = after_delete.catalog; + reborn.entries.push_back(successor); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to publish successor catalog life"); + return successor; +} + +/// Decode the ref-log object at `id`, through the SAME codec the writer's recovery uses (never a +/// hand-rolled parse), so an assertion about `prev_epoch_seal` is an assertion about the WIRE. +RefLogTxn readRefLogTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & id) +{ + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + if (!got) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "no ref-log object at {}-{}", id.writer_epoch, id.ref_sequence); + return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id); +} + +/// The bytes of a real `EpochSeal` transaction closing `id.writer_epoch` at `id` -- what a SUCCESSOR +/// writes into the dead epoch's next slot (spec INV-2). Grammar: exactly one `EpochSeal` op, and +/// `prev_epoch_seal` on sequence 1 only, so callers pass it exactly when `id.ref_sequence == 1`. +String epochSealBytes(const RootNamespace & ns, const RefTxnId & id, std::optional prev_epoch_seal = std::nullopt) +{ + RefOp op; + op.kind = RefOpKind::EpochSeal; + const RefLogTxn txn{ns.string(), id, {op}, prev_epoch_seal}; + return sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); +} + +/// Re-arm the mount fence WITHOUT a self-remount: the generation moves (twice -- trip, then re-arm) +/// while the cached runtime survives, which is what isolates the generation check as the sole +/// detector. A real self-remount also quiesces the runtimes; that path is Task 6's. +void bumpFenceGeneration(const PoolPtr & store, uint64_t writer_epoch) +{ + store->tripMountLost(); + store->armMountFence(DB::UInt128{0, 1}, writer_epoch, store->bootMsNow() + 600000); + /// The fence re-arm alone moves the GENERATION; the live incarnation's writer epoch is a separate + /// publication (`tryRemountOnce` does both), and the append lane derives its ids from that one. + store->setLiveWriterEpochForTest(writer_epoch); +} + +/// Arm a one-shot throw inside the post-durable install regions. The exception is built OUTSIDE the +/// region (building it inside would trip `DENY_ALLOCATIONS_IN_SCOPE` and test the guard instead), and +/// `MEMORY_LIMIT_EXCEEDED` is what a real tracked allocation failure raises. Same shape as +/// `gtest_cas_ref_install_safety.cpp`'s helper. +void armOneShotInstallFailure(const PoolPtr & store) +{ + auto planned = std::make_exception_ptr(DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, + "simulated allocation failure inside the post-durable install region")); + auto fired = std::make_shared>(false); + store->setInstallRegionProbeForTest([planned, fired] + { + if (fired->exchange(true)) + return; + ALLOW_ALLOCATIONS_IN_SCOPE; + std::rethrow_exception(planned); + }); +} + +} + +/// =================================================================================== +/// The every-attempt rule: an ambiguous attempt is resolved by a bounded CREATE, not a read +/// =================================================================================== + +/// The headline change. Nothing landed, so the old bare-GET resolution reported "absent" forever and +/// the lane never recovered without a remount. One conditional create of the SAME bytes settles it: +/// the object becomes durable and the wedged transaction is adopted -- applied EXACTLY once, before +/// the flush that resolved it allocates any new id. +TEST(CASRefWedgeEveryAttempt, AmbiguousPutWedgesTheLaneAndTheNextFlushsCreateAdoptsItExactlyOnce) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_created"}; + /// Stage B (Task 4-C): `logPrefix` below computes its fault-injection match at the sentinel; + /// pinning `ns` there BEFORE the first real touch keeps the real production birth landing on the + /// same key the fault targets. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "a wedged transaction is not applied"; + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(backend->get(wedged_key).has_value()) << "the ambiguous attempt wrote nothing"; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + /// The next caller's flush resolves the wedge with ONE create, adopts it, and only then carves and + /// commits its own transaction. + EXPECT_NO_THROW(store->dropRef(ns, "y")); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the adopted wedge applied its drop"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()) << "the resolving flush committed its own drop"; + EXPECT_TRUE(backend->get(wedged_key).has_value()) << "the wedged transaction is durable at its own key"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before + 2) + << "the adopted wedge and the ordinary commit must each join the tail exactly once"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); +} + +TEST(CASRefWedgeEveryAttempt, DurableCreatedWedgeNeedsRecoveryWhenItsFrontierCannotBePublished) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_created_frontier_failed"}; + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(backend->get(wedged_key)); + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + const NamespaceLifeId life = *store->refTableLifeForTest(ns); + const String ckpt_key = store->layout().refCkptKey(life); + const RefCkpt ckpt_before = decodeRefCkpt(backend->get(ckpt_key)->bytes); + backend->fail_cas_substr = ckpt_key; + backend->fail_cas_count = 200; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "y"); }); + + EXPECT_TRUE(backend->get(wedged_key)) << "the exact wedged log was proven durable"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery) + << "a durable log without a confirmed frontier must not return to Ready"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) + << "the unfrontiered wedge must not be installed into the resident table"; + EXPECT_EQ(decodeRefCkpt(backend->get(ckpt_key)->bytes), ckpt_before); +} + +TEST(CASRefWedgeEveryAttempt, RetiredLifeRefusesWedgeRetryBeforeAnyRequestOrAdoption) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge-retired-before-retry"}; + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, store->layout(), ns); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(predecessor.ns, predecessor.incarnation); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(backend->get(wedged_key)); + + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + store->setWedgeBeforeSlotOccupyHookForTest([&] + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }); + + std::exception_ptr retry_error; + std::thread retry([&] + { + try + { + store->dropRef(ns, "y"); + } + catch (...) + { + retry_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + const CatalogEntry successor + = replaceCatalogLifeForWedgeRace(*backend, store->layout(), predecessor, UInt128{0x71f2}); + const NamespaceLifeId successor_life + = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); + ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + .life_epoch = store->liveWriterEpoch(), + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + store->invalidateRemovedCatalogLife(predecessor_life); + backend->resetCounts(); + + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + retry.join(); + store->setWedgeBeforeSlotOccupyHookForTest(nullptr); + + EXPECT_TRUE(retry_error); + EXPECT_EQ(backend->putCount(wedged_key), 0u) << "retirement must refuse before the retry send"; + EXPECT_EQ(backend->getCount(wedged_key), 0u) << "a refused retry needs no occupant resolution read"; + EXPECT_FALSE(backend->get(wedged_key)) << "the predecessor wedge was adopted or made durable"; + EXPECT_NO_THROW((void)store->listRefs(ns)); + ASSERT_TRUE(store->refTableLifeForTest(ns)); + EXPECT_EQ(*store->refTableLifeForTest(ns), successor_life); +} + +/// The other adoption input, and the one that proves the identity rule is about BYTES: our own +/// earlier attempt DID land (only its ack, and the controller's own resolve read, were lost). The +/// retry's create conflicts with our own object, the follow-up read returns bytes equal to the +/// wedge's, and the transaction is adopted -- ONCE, not once per attempt. +TEST(CASRefWedgeEveryAttempt, OwnLandedAttemptIsAdoptedFromOccupiedWithoutDoubleApply) +{ + auto backend = std::make_shared(); + /// Disarmed while the fixture is built: the one-shot fault matches ANY key until a substring is + /// set, and the pool's own bootstrap PUT would otherwise consume it. + backend->fired = true; + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_occupied_mine"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->key_substr = logPrefix(store, ns); + backend->lose_resolve_read = true; + backend->fired = false; /// armed: the next `_log/` PUT lands and loses its ack + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_TRUE(backend->get(wedged_key).has_value()) << "this fault LANDS the write; only the ack was lost"; + ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "durable, but not applied while wedged"; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + const uint64_t puts_before = backend->putCount(wedged_key); + + EXPECT_NO_THROW(store->dropRef(ns, "y")); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the landed transaction is adopted on resolution"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before + 2) + << "adopted exactly once: a double-apply would bump the tail twice for one transaction"; + EXPECT_EQ(backend->putCount(wedged_key), puts_before + 1) + << "the resolution costs exactly ONE conditional create at the wedged key"; +} + +/// `ambiguous-then-definite`, the control the phase-0 model singles out: a definite refusal of a LATER +/// attempt says nothing about the EARLIER ambiguous one, which may still be in flight. The lane must +/// stay wedged -- unwedging here is how an acked-then-lost transaction gets written around. +TEST(CASRefWedgeEveryAttempt, DefiniteRefusalOfARetryAttemptKeepsTheLaneWedged) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_ambiguous_then_definite"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + const RefTxnId wedged_id = store->layout().parseRefObjectKey(wedged_key)->txn_id; + + /// The retry's own create is definitively refused. + backend->definite_substr = logPrefix(store, ns); + backend->definite_count = 1; + EXPECT_ANY_THROW(store->dropRef(ns, "y")); + + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "a definite refusal AFTER an ambiguous attempt must not unwedge"; + EXPECT_EQ(store->wedgedKeyForTest(ns), wedged_key) << "the SAME wedge, not a fresh one"; + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "nothing was adopted"; + EXPECT_FALSE(backend->get(wedged_key).has_value()) << "and nothing became durable"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) + << "a wedged lane's steady state is 'may be durable, not applied'"; + + /// Still the same id afterwards: the definite refusal consumed nothing. + backend->definite_count = 0; + EXPECT_NO_THROW(store->dropRef(ns, "y")); + EXPECT_EQ(store->layout().parseRefObjectKey( + store->layout().refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), wedged_id))->txn_id, wedged_id); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "the create-based resolution still settles it afterwards"; +} + +/// The SAME rule one level down, and the level where it was actually broken. The test above splits the +/// two attempts across two CALLS, which the wedge already handles. Inside ONE call the controller used +/// to report the LAST attempt's outcome: an ambiguous attempt followed by a definitively refused reissue +/// came back `DefiniteFailure` -- the verdict that means "the key is provably unwritten". It is not. The +/// refusal proves only that the SECOND request never applied; the first may still be in flight and may +/// still land, and `unresolvedProvesNothingWasSent` is false for exactly that reason. So the CALL is +/// unresolved, and a definite verdict is only ever the whole call's. +/// +/// The new reason lands on the fail-close side of the predicate the ledger acts on. Asserted at compile +/// time, beside the behaviour, because a member added to the enum without classifying it is precisely +/// how the wedge would silently stop happening. +static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::DefiniteFailureAfterAmbiguity)); + +TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalCannotSpeakForAnEarlierAmbiguousAttemptOfTheSameCall) +{ +#if !USE_AWS_S3 + GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; +#else + auto backend = std::make_shared(); + CasRequestController controller(backend, twoAttemptBudget()); + const std::function fence_ok = [] { return true; }; + + /// One call, two attempts: ambiguous, then definitively refused. + backend->ambiguous_substr = "key/"; + backend->ambiguous_count = 1; + backend->s3_definite_substr = "key/"; + backend->s3_definite_count = 1; + + CasUnresolvedReason reason = CasUnresolvedReason::NotUnresolved; + const CasWriteOutcome outcome = + controller.putIfAbsentControlled("key/haunted", "bytes", fence_ok, /*out_token=*/nullptr, &reason); + + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved) + << "a definite refusal of the SECOND attempt cannot retire the first attempt's ambiguity"; + EXPECT_EQ(reason, CasUnresolvedReason::DefiniteFailureAfterAmbiguity); + EXPECT_FALSE(unresolvedProvesNothingWasSent(reason)) + << "the caller must keep protecting itself: an earlier attempt was sent and may yet land"; + EXPECT_FALSE(backend->get("key/haunted").has_value()) + << "and the key is still empty -- which is exactly why an absent read settles nothing"; + + /// THE CONTROL. Aggregation must not soften a definite refusal that speaks for the whole call: with + /// no ambiguous predecessor, the first attempt's whitelisted rejection is still `DefiniteFailure`, + /// and the ledger may still free the id on it. + backend->s3_definite_count = 1; + CasUnresolvedReason clean_reason = CasUnresolvedReason::NotUnresolved; + EXPECT_EQ(controller.putIfAbsentControlled("key/clean", "bytes", fence_ok, /*out_token=*/nullptr, &clean_reason), + CasWriteOutcome::DefiniteFailure); + EXPECT_EQ(clean_reason, CasUnresolvedReason::NotUnresolved); +#endif +} + +/// The ledger-side twin of the same call: what the append lane does with that verdict. On +/// `DefiniteFailure` it returns the lane to `Ready` and tells callers the txn id was never used, +/// so the next append re-derives that id -- which, with an earlier attempt still possibly in flight, is +/// how an acked-then-lost transaction gets written around. The lane must wedge instead and stay pending +/// until the key itself resolves. +TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCallStillWedgesTheLane) +{ +#if !USE_AWS_S3 + GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; +#else + auto backend = std::make_shared(); + auto store = openPool(backend, twoAttemptBudget()); + const RootNamespace ns{"srv1/wedge_one_call_ambiguous_then_definite"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + backend->s3_definite_substr = logPrefix(store, ns); + backend->s3_definite_count = 1; + + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) + << "one call whose first attempt is unresolved leaves an object that may become durable"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) + << "the id must NOT be declared never-used: the marker stands until the key itself resolves"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(), definite_before) + << "this append was never definitively rejected -- only one of its attempts was"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1); + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "nothing is applied while the lane is wedged"; + const String wedged_key = store->wedgedKeyForTest(ns); + EXPECT_FALSE(backend->get(wedged_key).has_value()) << "and nothing became durable"; + + /// And it still recovers by the ordinary route: the next flush's bounded create lands the wedged + /// transaction and adopts it, so wedging costs availability only until the next caller arrives. + EXPECT_NO_THROW(store->dropRef(ns, "y")); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the adopted wedge applied its drop"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); +#endif +} + +/// =================================================================================== +/// A successor's `EpochSeal` is the conclusive rejection (spec INV-2) +/// =================================================================================== + +/// The seal is the ONLY thing that can prove our transaction will never be durable: the key is +/// write-once and a successor put its epoch-closing record there. The operation was never acked, so +/// its callers get a permanent error; the wedge is cleared; and the seal becomes this namespace's +/// `prev_epoch_seal`, which the first append of the NEXT epoch carries on the wire. +TEST(CASRefWedgeEveryAttempt, SuccessorSealAtTheWedgedKeyRejectsConclusivelyAndSourcesPrevEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/wedge_sealed"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + const uint64_t epoch = store->liveWriterEpoch(); + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + const RefTxnId seal_id = layout.parseRefObjectKey(wedged_key)->txn_id; + ASSERT_EQ(seal_id.writer_epoch, epoch); + ASSERT_GT(seal_id.ref_sequence, 1u) << "this namespace already has records, so its seal is not at sequence 1"; + ASSERT_EQ(store->lastEpochSealForTest(ns), std::nullopt) << "nothing has closed an epoch for this namespace yet"; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + /// A successor closes our epoch at exactly the slot our attempt was aiming at. + ASSERT_EQ(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)).outcome, PutOutcome::Done); + + /// The next caller's resolution meets the seal. Its own items fail -- permanently, not "retry + /// later": nothing about this lane's epoch will ever accept a write again. + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "y"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a conclusive rejection clears the wedge"; + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "the rejected transaction was never applied"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) << "and never joined the tail"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Closed) + << "the successor seal closes this epoch's lane"; + ASSERT_EQ(store->lastEpochSealForTest(ns), std::make_optional(seal_id)) + << "the observed seal is this namespace's epoch-closing record"; + + /// INV-2's fence, stated as behaviour: a dying lane that observed the seal keeps deriving the SAME + /// `T+1` and keeps colliding with it -- it never mints `T+2` and writes its stream past the record + /// that closed its epoch. Ids are state-derived, so this falls out rather than being enforced. + /// + /// And the collision is adjudicated as the CONCLUSIVE REJECTION it is, not as foreign interference: + /// this is the designed path, so it must not fence the mount or raise an anomaly. The append site + /// reads the occupant and tells a seal of this namespace from a genuine breach, exactly as the + /// wedge-resolve site does. + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "y"); }); + EXPECT_EQ(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch, seal_id.ref_sequence + 1})), std::nullopt) + << "nothing of ours may exist above the seal in the closed epoch"; + EXPECT_TRUE(store->mayMutate()) << "meeting a successor's seal is the protocol working, not an anomaly"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) + << "and must not schedule a remount"; + + /// Merely changing the epoch counters does not reopen a cached runtime. Its immutable admitted + /// generation is stale, so the outer retry-safe fence refusal wins before the still-Closed lane is + /// consulted. Production reaches a new epoch through remount, which replaces the runtime and + /// recovers its chain link. + bumpFenceGeneration(store, epoch + 1); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "y"); }); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Closed); +} + +/// The wire round trip of the same rule, driven from the OTHER producer of `last_epoch_seal`: +/// recovery's CAS-walk (Task 6), stood in for here by its test seam. The point is the encode call +/// site, which is this task's. +TEST(CASRefWedgeEveryAttempt, OrdinaryFirstAppendAfterASealedTransitionCarriesTheExactPrevEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/prev_epoch_seal_roundtrip"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `readRefLogTxn` above + /// reads that exact key. + admitProperlyBornEntry(*backend, store->layout(), ns); + + const uint64_t epoch = store->liveWriterEpoch(); + publishEmptyPart(store, ns, "x"); + const RefTxnId seal_id{epoch, 42}; + + /// A recovery that walked the dead epoch installs the seal it wrote; model its later epoch without + /// moving the mount-fence generation. This test is about the wire link, not runtime supersession. + store->setLastEpochSealForTest(ns, seal_id); + store->setLiveWriterEpochForTest(epoch + 1); + + EXPECT_NO_THROW(store->dropRef(ns, "x")); + + const RefLogTxn written = readRefLogTxn(*backend, layout, ns, RefTxnId{epoch + 1, 1}); + EXPECT_EQ(written.prev_epoch_seal, std::make_optional(seal_id)); + + /// And it is carried on sequence 1 ONLY: the next transaction of the same epoch must not repeat it. + publishEmptyPart(store, ns, "z"); + const RefLogTxn second = readRefLogTxn(*backend, layout, ns, RefTxnId{epoch + 1, 2}); + EXPECT_EQ(second.prev_epoch_seal, std::nullopt) + << "prev_epoch_seal is required on sequence 1 of a non-genesis epoch and forbidden everywhere else"; +} + +/// GENESIS: `last_epoch_seal` is `nullopt` exactly for a namespace whose stream starts here, and a +/// genesis birth carries NO `prev_epoch_seal` even though its epoch is far above 1. Nothing about the +/// global epoch number makes a namespace non-genesis -- only a transition of its OWN stream does. +TEST(CASRefWedgeEveryAttempt, GenesisBirthAtAHighEpochCarriesNoPrevEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/genesis_at_five"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `readRefLogTxn` above + /// reads that exact key. + admitProperlyBornEntry(*backend, store->layout(), ns); + + bumpFenceGeneration(store, 5); + ASSERT_EQ(store->liveWriterEpoch(), 5u); + + publishEmptyPart(store, ns, "x"); + + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) + << "a namespace with no recovered seal, whose greatest applied id is at its own life epoch, is genesis"; + const RefLogTxn birth = readRefLogTxn(*backend, layout, ns, RefTxnId{5, 1}); + EXPECT_EQ(birth.prev_epoch_seal, std::nullopt) << "a genesis stream opens; it does not continue one"; +} + +/// =================================================================================== +/// A foreign occupant is impossible, so it is loud -- and the mount self-heals +/// =================================================================================== + +/// Under mount-lease exclusivity the wedged key is exclusively ours, so a foreign non-seal object at +/// it is corruption or a protocol breach. Fail closed with `CORRUPTED_DATA`, KEEP the wedge for +/// inspection, and route the anomaly so the mount remounts itself rather than staying stuck until +/// someone notices. +TEST(CASRefWedgeEveryAttempt, ForeignNonSealOccupantIsCorruptedDataAndSchedulesARemount) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_foreign"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + + /// Something that is neither our bytes nor a seal occupies the slot. + ASSERT_EQ(backend->putAsSuccessor(wedged_key, "not a ref-log object at all").outcome, PutOutcome::Done); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted); + EXPECT_GT(store->scheduleRemountCallCountForTest(), remounts_before) + << "the impossible-interference route must schedule a remount"; +} + +/// I5, the OTHER site with the same shape: the ordinary append's own conditional create can prove a +/// different object sits at the id it derived. Task 3 made that fail closed -- correctly -- but it +/// left the mount stuck there until a manual remount, unlike the wedge-resolution site. Both are the +/// same impossibility and both must self-heal by remount. +TEST(CASRefWedgeEveryAttempt, AppendSiteProvenDifferentObjectAlsoSchedulesARemount) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/append_site_foreign"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + + /// Occupy the id the next append will derive with a foreign object, so its create conflicts and + /// the controller's resolve-before-reissue proves the occupant is not ours. + const RefTxnId next{store->liveWriterEpoch(), 3}; + ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next), + "a different object entirely").outcome, PutOutcome::Done); + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a proven different object is conclusive, never a wedge"; + EXPECT_GT(store->scheduleRemountCallCountForTest(), remounts_before) + << "the append site must route through the same impossible-interference reaction as the wedge site"; +} + +/// =================================================================================== +/// The admission fence +/// =================================================================================== + +/// The old-generation-retry-inert rule. A wedge admitted under one mount incarnation may not send an +/// attempt under another: the retry is refused BEFORE anything reaches the store, so the key is +/// provably untouched and the wedge is intact for whoever recovers the lane properly. +TEST(CASRefWedgeEveryAttempt, RetryUnderAnOlderAdmissionGenerationSendsNothing) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_old_generation"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + const uint64_t epoch = store->liveWriterEpoch(); + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_EQ(store->wedgedAdmittedGenerationForTest(ns), store->fenceGeneration()) + << "the wedge records the generation it was admitted under"; + const uint64_t puts_before = backend->putCount(wedged_key); + + /// The lease incarnation moves under the wedge; the mount is writable again, but not the same one. + bumpFenceGeneration(store, epoch); + ASSERT_NE(store->wedgedAdmittedGenerationForTest(ns), store->fenceGeneration()); + + EXPECT_ANY_THROW(store->dropRef(ns, "y")); + + EXPECT_EQ(backend->putCount(wedged_key), puts_before) + << "the retry must be refused pre-attempt: nothing may reach the store under a foreign generation"; + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "and the wedge is untouched"; + EXPECT_FALSE(backend->get(wedged_key).has_value()); +} + +/// The post-I/O recheck, deterministically. The retry's create is parked mid-flight; while it is +/// parked the fence is lost and re-armed AND a successor seals the slot. The released result is a +/// perfectly real `Occupied`(seal) -- but it belongs to an incarnation that no longer exists, so this +/// runtime must act on NOTHING: no acknowledgement, no unwedge, no install, and no adoption of the +/// seal it just read. +TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterAFenceBumpAndSuccessorSealIsInert) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/wedge_blocked_io"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + const uint64_t epoch = store->liveWriterEpoch(); + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + const RefTxnId seal_id = layout.parseRefObjectKey(wedged_key)->txn_id; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + backend->armBlock(logPrefix(store, ns)); + std::exception_ptr caller_error; + std::thread resolver([&] + { + try { store->dropRef(ns, "y"); } + catch (...) { caller_error = std::current_exception(); } + }); + backend->awaitBlockEntered(); + + /// Everything that makes this runtime superseded happens INSIDE the I/O window. + ASSERT_EQ(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)).outcome, PutOutcome::Done); + bumpFenceGeneration(store, epoch + 1); + backend->releaseBlock(); + resolver.join(); + + ASSERT_TRUE(caller_error != nullptr) << "no acknowledgement: the caller must not be told this succeeded"; + /// And it must be the RETRY-SAFE class. A moved incarnation is usually a routine lease blip, and the + /// storage layer classifies retry-safety on exactly `ABORTED || NETWORK_ERROR` — surfacing the fence + /// check's own `INVALID_STATE` here would turn every blip into a hard failure for the caller. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { std::rethrow_exception(caller_error); }); + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "no unwedge"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) << "no install"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) + << "and no adoption of the seal a superseded runtime happened to read"; +} + +/// The same recheck, on the identity leg rather than the generation leg. The fence never moves; only +/// the installed wedge's BYTES change while the create is parked. Generation-equality alone would let +/// the released result install a candidate built from the OTHER attempt's transaction -- the aliasing +/// bug the phase-0 model found, which is why identity is (generation, id, bytes) and not any one of +/// them. Production cannot reach this (one leader per table mutates a lane), so this is a white-box +/// guard on the rule, driven through the force-wedge seam. +TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterTheWedgeIdentityChangedIsInert) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_identity_changed"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + const RefTxnId wedged_id = store->layout().parseRefObjectKey(wedged_key)->txn_id; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + backend->armBlock(logPrefix(store, ns)); + std::exception_ptr caller_error; + std::thread resolver([&] + { + try { store->dropRef(ns, "y"); } + catch (...) { caller_error = std::current_exception(); } + }); + backend->awaitBlockEntered(); + + /// Same id, same generation, DIFFERENT bytes. + store->forceWedgeForTest(ns, wedged_id.writer_epoch, wedged_id.ref_sequence, wedged_key, "different attempt bytes"); + backend->releaseBlock(); + resolver.join(); + + ASSERT_TRUE(caller_error != nullptr); + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "no unwedge"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) + << "no install: the released result described a wedge that is no longer installed"; + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()); +} + +/// A resolution that proves the exact attempt durable but cannot install it has one successor: +/// `NeedsRecovery`. It drops the attempt and forbids another write until replay catches the cache up. +TEST(CASRefWedgeEveryAttempt, KnownDurableInstallFailureMovesDirectlyToRecovery) +{ + auto backend = std::make_shared(); + backend->fired = true; /// disarmed while the fixture is built (see the adoption test above) + auto store = openPool(backend, singleAttemptBudget()); + const RootNamespace ns{"srv1/wedge_floor"}; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches + /// its fault at that key. + admitProperlyBornEntry(*backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->key_substr = logPrefix(store, ns); + backend->lose_resolve_read = true; + backend->fired = false; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_TRUE(backend->get(wedged_key).has_value()) << "the wedged transaction is durable"; + /// The adoption reaches its install region and the install throws. + armOneShotInstallFailure(store); + expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "y"); }); + store->setInstallRegionProbeForTest(nullptr); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + ASSERT_FALSE(store->refLaneWedgedForTest(ns)) + << "known durability transfers ownership to recovery; no uncertain attempt remains"; + /// Do not call the tail-count seam here: it intentionally forces recovery, which is the transition + /// this assertion is proving has not happened yet. + + /// The next flush first replays the durable drop of `x`, then admits the drop of `y`. + EXPECT_NO_THROW(store->dropRef(ns, "y")); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) + << "replay must install the already-durable drop of `x` before the next write"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) + << "only completed replay returns the lane to Ready"; +} + +/// =================================================================================== +/// The append site owes the SAME three-way adjudication as the wedge site +/// =================================================================================== + +/// No wedge is involved here at all: an ordinary append derives its next id and finds a successor's +/// epoch seal sitting on it. That is not interference — it is INV-2's designed outcome for a lane that +/// has been deposed without being told, and the lane will keep re-deriving that same id forever. So it +/// must be adjudicated as the conclusive rejection it is: a permanent error for the callers, the seal +/// recorded as this namespace's epoch-closing record, and NO fence and NO remount. +TEST(CASRefWedgeEveryAttempt, AppendSiteMeetingASuccessorSealIsAConclusiveRejectionNotInterference) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/append_site_seal"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + + const uint64_t epoch = store->liveWriterEpoch(); + const RefTxnId next{epoch, 3}; + ASSERT_EQ(store->lastEpochSealForTest(ns), std::nullopt); + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + const uint64_t sealed_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected].load(); + + /// The successor's seal lands at exactly the id this table's next append derives. + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->conflict_bytes = epochSealBytes(ns, next); + backend->conflict_count = 1; + + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a conclusive rejection is not an uncertain outcome"; + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "the rejected transaction never applied"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::make_optional(next)) + << "the observed seal IS this namespace's epoch-closing record, whichever site observed it"; + EXPECT_TRUE(store->mayMutate()) << "the designed path must not fence the mount"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedule a remount"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected].load(), sealed_before + 1) + << "a deposed writer must still be COUNTED: this is the protocol working, and also the signal " + "that this mount has lost its lease and does not know it"; +} + +/// [CKPT-FAILED-BIRTH-DEBRIS] REVERSED (increment review Critical B; BACKLOG `{#ckpt-failed-birth-debris}` +/// reopened, `{#ckpt-neverborn-gc-backstop}` filed). This test used to pin the OPPOSITE of what it now +/// asserts: Task 3's `cleanupOrphanedBirthCkptBestEffort` deleted `_ckpt` here by a FRESH `head()` read +/// at cleanup time, not a token captured from this attempt's own publish, and every branch that called it +/// -- this one included -- had just PROVEN a different object occupies the derived key, directly +/// contradicting the "reachable only while the ref-log has never durably held anything" argument that +/// made the delete look safe. A successor that legitimately owns the same live incarnation (an ordinary +/// INV-2 epoch-seal handoff, e.g. after a remount) may already have read this SAME `_ckpt` for its own +/// recovery before this cleanup could run, and the delete could destroy the one genesis record +/// (`life_epoch`) that successor's own future recovery still needs, with no way to tell that case apart +/// from ordinary debris at cleanup time. The cleanup was removed entirely rather than patched (see +/// `CasRefLedger.cpp`'s comment at the removed call sites for why a captured token does not close the +/// gap either). The trade, named rather than hidden: a creation `_ckpt` whose first ref-log +/// `NamespaceBirth` is conclusively rejected now SURVIVES -- a drained server root carrying it will refuse +/// decommission (`claimOwnerOrThrow` -> `CORRUPTED_DATA`) until `{#ckpt-neverborn-gc-backstop}` lands -- +/// which is the right side of the trade against an unrecoverable delete of a live successor's only +/// genesis record. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesAConclusiveFirstRefLogRejection) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/birth_ckpt_debris"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + const RefTxnId genesis{store->liveWriterEpoch(), 1}; + const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; + + /// A successor's epoch seal lands at exactly the id this first `NamespaceBirth` transaction derives. + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); + backend->conflict_bytes = epochSealBytes(ns, genesis); + backend->conflict_count = 1; + + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { publishEmptyPart(store, ns, "x"); }); + + const auto ckpt_after = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) + << "the creation checkpoint must survive a conclusively rejected first ref-log PUT unchanged"; +} + +/// A creation `_ckpt` belonging to a `Live` namespace must survive no matter how a LATER transaction +/// on that same namespace fails -- true unconditionally now that increment review Critical B removed +/// the only code that ever deleted `_ckpt` on this path at all, but kept as its own pin: the fixture +/// has already made `ns` `Live`; then two initial ref-log chunks (precommit-add, then promote) and a +/// THIRD chunk meet the +/// identical successor-seal conflict the test above exercises -- same conclusive rejection -- and the +/// creation `_ckpt` must survive it byte-for-byte. `_ckpt` has no repair +/// path (BACKLOG `{#ckpt-damage-no-repair-path}`), so this is the row that would catch a future +/// reintroduction of the removed cleanup landing back on an already-Live namespace's `_ckpt`. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesALaterConclusiveRejection) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/birth_ckpt_survives_live"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + /// ONE `publishEmptyPart` reaches sequence 2 (the precommit-add chunk at seq 1 carries the first + /// `NamespaceBirth`, the promote chunk lands at seq 2), so `next` + /// below is the SAME `{epoch, 3}` the sibling `AppendSiteMeetingASuccessorSealIsAConclusiveRejectionNotInterference` + /// test derives from the identical one-call setup -- copying THAT test's two-call variant here + /// (from a different test in this file) would derive a different id and never trigger the conflict. + publishEmptyPart(store, ns, "x"); + + const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation step must have published a real _ckpt"; + + const RefTxnId next{store->liveWriterEpoch(), 3}; + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->conflict_bytes = epochSealBytes(ns, next); + backend->conflict_count = 1; + + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); + + const auto ckpt_after = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_after.has_value()) << "a Live namespace's _ckpt must never be deleted by this path"; + EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) + << "not merely present but UNCHANGED -- no code anywhere on this path deletes _ckpt any more " + "(increment review Critical B), so it must not have been touched at all"; +} + +/// THE OTHER negative row (review C2): the AMBIGUOUS branch -- `Writing` -> `Wedged` -- is deliberately +/// EXCLUDED from the cleanup call (see the lambda's own comment), and that exclusion is the +/// load-bearing half of the whole safety story: it is the one branch where the ref-log bytes MIGHT +/// still have landed. Nothing pinned that exclusion before this row; an edit that added the call here +/// would be caught by no test. A one-shot ambiguous PUT on the first ref-log `NamespaceBirth` +/// (`prepared->birth_contribution` set) writes NOTHING (the response is lost, the key +/// stays absent) and wedges the lane -- `AmbiguousPutWedgesTheLaneAndTheNextFlushsCreateAdoptsItExactlyOnce` +/// is the precedent this mirrors, adapted to a namespace's FIRST-ever transaction instead of its third. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesWhenTheFirstNamespaceBirthIsAmbiguous) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, singleAttemptBudget()); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/birth_ckpt_ambiguous"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; + + backend->ambiguous_substr = logPrefix(store, ns); + backend->ambiguous_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "x"); }); + + ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "an ambiguous outcome must WEDGE the lane, not " + "resolve into one of the conclusive branches"; + const auto ckpt_after = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) + << "the creation checkpoint must survive an ambiguous first ref-log outcome unchanged"; +} + +/// Final review F5: the two other removed call sites, given their own first-`NamespaceBirth` survival rows. +/// The reversed test above pins the `SuccessorSeal` branch; the sibling below it pins the ambiguous +/// branch (never called it in the first place). The remaining two -- occupant-unreadable +/// (`CORRUPTED_DATA` from a failed adjudication read) and genuine foreign interference -- had no +/// first-`NamespaceBirth` row at all: `AppendSiteFaultsWhenTheOccupantCannotBeRead`, +/// `ForeignNonSealOccupantIsCorruptedDataAndSchedulesARemount`, and `WellFormedNonSealOccupantIsStillForeign` +/// all `publishEmptyPart` FIRST, so none of them ever carries a `birth_contribution` -- a reinstated +/// GUARDED cleanup at either of these two sites would pass the whole suite with no first-transaction case to +/// catch it. Mirrors `WellFormedNonSealOccupantIsStillForeign`'s occupant shape (a decodable, well-formed +/// NON-seal transaction at the derived key), moved to sequence 1 of a namespace with no prior ref-log +/// transaction, so this attempt's own PUT is its first. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesWhenTheFirstNamespaceBirthOccupantCannotBeRead) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/birth_ckpt_occupant_unreadable"}; + admitProperlyBornEntry(*backend, store->layout(), ns); + + const RefTxnId genesis{store->liveWriterEpoch(), 1}; + const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; + + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); + backend->conflict_bytes = epochSealBytes(ns, genesis); + backend->conflict_count = 1; + /// Proper birth makes recovery first probe this absent log key. Skip that probe and the resolve + /// read that PROVES the conflict; fail only the adjudication read after it, so the occupant's + /// identity (seal vs. breach) cannot be determined. + backend->fail_get_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); + backend->fail_get_skip = 2; + backend->fail_get_count = 1; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { publishEmptyPart(store, ns, "x"); }); + + const auto ckpt_after = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) + << "the creation checkpoint must survive an occupant-unreadable first ref-log outcome unchanged"; +} + +/// The other former call site: a genuine breach of write-exclusivity at the first `NamespaceBirth` +/// id, mirroring `WellFormedNonSealOccupantIsStillForeign`'s occupant shape but with no prior ref-log publish. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesFirstNamespaceBirthForeignInterference) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/birth_ckpt_foreign_interference"}; + admitProperlyBornEntry(*backend, store->layout(), ns); + + const RefTxnId genesis{store->liveWriterEpoch(), 1}; + const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; + + /// A perfectly decodable transaction for this exact namespace and id -- just not an epoch seal, and + /// not this attempt's own birth. + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + const RefLogTxn foreign_txn{ns.string(), genesis, {birth}, std::nullopt}; + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); + backend->conflict_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(foreign_txn)); + backend->conflict_count = 1; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { publishEmptyPart(store, ns, "x"); }); + + const auto ckpt_after = backend->get(ckpt_key); + ASSERT_TRUE(ckpt_after.has_value()); + EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) + << "the creation checkpoint must survive a foreign-interference first ref-log outcome unchanged"; +} + +/// The same conflict, but the read that would tell a seal from a breach fails. We must then decide +/// NEITHER: fencing the mount would be a guess, and reporting a conclusive rejection would acknowledge +/// a deposition nobody observed. The id is not consumed, so the next attempt re-derives it and +/// classifies again — deferring costs one round trip and decides nothing wrongly. +TEST(CASRefWedgeEveryAttempt, AppendSiteFaultsWhenTheOccupantCannotBeRead) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/append_site_unreadable"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + + const RefTxnId next{store->liveWriterEpoch(), 3}; + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->conflict_bytes = epochSealBytes(ns, next); + backend->conflict_count = 1; + /// Skip the resolve read that PROVES the conflict; fail only the adjudication read after it. + backend->fail_get_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->fail_get_skip = 1; + backend->fail_get_count = 1; + + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(), deferred_before + 1) + << "the deferral is the one quiet arm here -- it must be counted or a starved loud path is invisible"; + EXPECT_TRUE(store->mayMutate()) << "the table faults without guessing that the whole mount is corrupt"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedule a remount"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) + << "nor record a deposition that was never actually observed"; + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "nothing of ours became durable, so nothing is wedged"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted); + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); +} + +/// A WELL-FORMED ref-log transaction of this namespace at this id, which simply is not a seal, must be +/// adjudicated `Foreign` on CONTENT — not because it failed to decode. The sibling test above reaches +/// the same verdict through an undecodable body, so without this one the classifier could be deciding +/// "foreign" purely from decode failures and nothing would notice. +TEST(CASRefWedgeEveryAttempt, WellFormedNonSealOccupantIsStillForeign) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/append_site_wellformed_foreign"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + + const RefTxnId next{store->liveWriterEpoch(), 3}; + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + + /// A perfectly decodable transaction for this exact namespace and id — just not an epoch seal. + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + const RefLogTxn foreign_txn{ns.string(), next, {birth}, std::nullopt}; + backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->conflict_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(foreign_txn)); + backend->conflict_count = 1; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + + EXPECT_GT(store->scheduleRemountCallCountForTest(), remounts_before) + << "a well-formed non-seal occupant is still a breach of write-exclusivity"; + EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) << "and is emphatically not an epoch seal"; +} + +/// The deposed-lane self-pointer. A successor that seals an EMPTY epoch writes its record at sequence 1 +/// of that epoch, and a lane still live there re-derives exactly that id. Stamping the seal as its own +/// `prev_epoch_seal` would be a self-pointer, which the structural grammar (strictly-less by +/// construction) refuses at ENCODE — so the lane would fail with a self-inflicted `CORRUPTED_DATA` on +/// every attempt and never reach the seal collision that is supposed to fence it. The stamp is +/// therefore conditioned on the seal's epoch being strictly BELOW the id's. +TEST(CASRefWedgeEveryAttempt, ALiveEpochSealIsNeverStampedAsItsOwnPrevEpochSeal) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/live_epoch_seal"}; + admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + publishEmptyPart(store, ns, "x"); + + const uint64_t epoch = store->liveWriterEpoch(); + /// Keep this a local wire/encoder test: moving the mount-fence generation would correctly make the + /// immutable runtime stale before the self-pointer guard was reached. + store->setLiveWriterEpochForTest(epoch + 1); + /// A seal of the LIVE epoch — the deposed-lane shape the wedge rejection arm can record. + store->setLastEpochSealForTest(ns, RefTxnId{epoch + 1, 7}); + + /// The lane now holds NOTHING it can legally write. Its next id is sequence 1 of the new epoch, which + /// owes a link to the seal that closed the epoch BELOW -- and the only seal it has is of the epoch it + /// is trying to open. Stamping that one would be a self-pointer the ENCODER refuses; stamping nothing + /// leaves a crossing the READER refuses. So the append fails closed, locally, before anything is sent. + /// + /// That is a strictly better outcome than the one this test originally pinned (stamp nothing, send, + /// and let the successor's seal reject the attempt at the key): the deposed lane spends no request to + /// learn what it can already prove about itself. The property the test exists for is unchanged and is + /// asserted below in its strongest form -- NO object is written at that id at all, so no self-pointer + /// can have been stamped anywhere. + /// The refusal must reach the SAME TERMINAL OUTCOME the collision produced, not merely "an error". + /// Skipping the request must not skip the conclusion, so all four halves are pinned: + /// + /// 1. the class is INVALID_STATE -- the conclusive-rejection class the successor-seal arm uses, + /// NOT the retry-later class. This is the one that matters most: a retryable error here would + /// have every caller re-derive the same impossible transaction forever, and the deposition + /// would never surface anywhere; + /// 2. the message says the lane resumes only under a later epoch -- that IS the deposition, + /// reported to the caller and the operator in the same words the collision reported it; + /// 3. NOTHING is written, so no self-pointer can have been stamped and no request was spent; + /// 4. a SECOND flush behaves identically instead of looping or degrading. + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); /// NOLINT(clang-analyzer-deadcode.DeadStores) + const size_t puts_before = backend->putCount(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})); /// NOLINT(clang-analyzer-deadcode.DeadStores) + try + { + store->dropRef(ns, "x"); + FAIL() << "the deposed lane must reject conclusively, not succeed"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::INVALID_STATE) << "got: " << e.message(); + EXPECT_NE(e.message().find("resumes only under a later epoch"), String::npos) + << "the deposition must be surfaced, not just the failure: " << e.message(); + } + EXPECT_FALSE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})).has_value()) + << "nothing may be written: the lane could not construct a legal transaction, so it sent none"; + EXPECT_EQ(backend->putCount(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})), puts_before) + << "and no request was spent learning what the lane could already prove about itself"; + + /// The second flush: same conclusive answer, still no traffic. A lane that re-derived and re-sent + /// here would be exactly the spin this arm exists to prevent. + expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); + EXPECT_EQ(backend->putCount(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})), puts_before); + + /// NO remount is scheduled, matching the collision arm exactly. A successor closing our epoch is a + /// legitimate handover, not an anomaly to react to: the mount lease is what resolves it, and + /// scheduling a remount from here would turn every ordinary deposition into a self-inflicted + /// re-claim storm. + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before); +} diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp new file mode 100644 index 000000000000..48414871269b --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -0,0 +1,5421 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include + +/// Task 10: the writer's ref persistence on the snapshot+log protocol. Covers the plan's Task 10 +/// failing-test list: empty+birth recovery; snapshot+tail recovery; recovery restart on a vanished +/// object (converging on a newer snapshot); the append lane's wedge semantics (blocks the same table, +/// leaves other tables free, applies a later-observed-durable append before unwedging); invalid batch +/// entries failing in isolation; and the S3 request-cost contract (one create for a warm isolated +/// mutation, one create shared by a compatible batch). + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +extern const int FILE_DOESNT_EXIST; +extern const int CORRUPTED_DATA; +extern const int INVALID_STATE; +extern const int LOGICAL_ERROR; +extern const int NETWORK_ERROR; +extern const int S3_ERROR; +} + +namespace ProfileEvents +{ +extern const Event CASRefSweepDeferred; +extern const Event CASRefSweepRearmed; +extern const Event CASRefStalePrecommitsReclaimed; +extern const Event CASRefSnapshotPutBytes; +extern const Event CASRefSnapshotTailLogs; +extern const Event CASRefSnapshotPublishDispatched; +extern const Event CASRefSnapshotPublishBackoff; +extern const Event CASConditionalWriteFenceLostPostWrite; +extern const Event CASRefRecoveryEpochSealed; +extern const Event CASRefRecoveryRetries; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::committedRow; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::minimalLiveSnapshot; +using DB::Cas::tests::namespaceBirthOp; +using DB::Cas::tests::publishCommittedOps; +using DB::Cas::tests::runRegularRoundReclaiming; +using DB::Cas::tests::writeRefSnapshotRaw; +using DB::Cas::tests::writeSealAt; + +namespace +{ + +/// The operation deadline every SINGLE-ATTEMPT fixture in this file uses, and the reason it is +/// deliberately NOT `attempt_timeout_ms`. +/// +/// Those fixtures exist to make one injected ambiguous response conclusive, and `max_attempts = 1` +/// alone achieves that: with retries allowed the controller would resolve-before-reissue and report a +/// definite outcome instead. The deadline contributes nothing to that -- but setting it EQUAL to the +/// attempt timeout collapses the controller's pre-send gate into a race. The deadline is captured as +/// `now + operation_deadline_ms` and the gate asks `now + attempt_timeout_ms > deadline` +/// (`CasRequestControl.cpp`), so equal values reduce it to `now_2 > now_1`: ONE elapsed millisecond +/// between the two clock reads refuses the operation with NOTHING SENT, the injected fault is never +/// reached, and the product correctly does not wedge -- flipping every wedge expectation downstream. +/// +/// That is not hypothetical. It took down +/// `CASRefWriterStalePrecommitSweep.BoundedBatchesAndInterruptionResumeAcrossMounts` on 5 of 6 sanitizer +/// CI runs (fixed in `8f9e63c7a19`), `CASRefInstallSafety.UncertainPrecommitKeepsItsCleanupOwnerAndItsBody` +/// under parallel-build load, and `CASRefWriterAppendLane.WedgedLaneBlocksSameTableWhileOtherTableProceeds` +/// in a full-binary ASan run -- the last one with the mechanism named verbatim in the thrown message +/// ("refused BEFORE any request was sent ... the operation deadline rejected before the first request"). +/// +/// A wide deadline keeps the request always actually sent, so what the test observes is the fault it +/// injected rather than the machine it ran on. A fixture that genuinely wants the pre-send REFUSAL +/// must drive it deterministically with a frozen clock (see `gtest_cas_ref_install_safety.cpp`'s +/// `openPoolFenceControlled`), never by racing the wall clock. +constexpr uint64_t kSingleAttemptDeadlineMs = 5000; + +/// A `CasEvent` sink safe to hand to `Pool::setEventSink`: the emit runs on whatever thread the pool's +/// background syncer happens to be on, and the test reads the accumulated events afterward from the +/// main test thread with no other ordering between the two -- a bare `std::vector` there is a real data +/// race (the class this file's four `setEventSink` call sites all had, hidden because a debug/ASan build +/// doesn't reliably catch an unsynchronized push_back/iterator-read pair on a small vector). `add` takes +/// the lock only around the push; `snapshot` copies out under the lock and returns, so a caller iterating +/// the result never holds the mutex across anything that could call back into the pool (which an +/// event-sink callback legitimately can, on other seams in this file). +class SynchronizedEventLog +{ +public: + void add(const CasEvent & e) + { + std::lock_guard lock(mutex); + events.push_back(e); + } + std::vector snapshot() const + { + std::lock_guard lock(mutex); + return events; + } +private: + mutable std::mutex mutex; + std::vector events; +}; + +PoolPtr openPool(const BackendPtr & backend, CasRequestBudget budget = {}) +{ + /// Recovery tests seed ref-log/snapshot residue before opening; a pool with such residue always has a + /// `_pool_meta` in production, so establish it first (Task 7's zero-write bootstrap check refuses to + /// mint a fresh identity over residual data — see `seedPoolMetaForRestart`). Idempotent, and a no-op + /// for the fresh-open tests that seed nothing (the subsequent open validates the just-created meta). + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); +} + +/// Task 11: like `openPool`, but the caller supplies (and owns) the rest of the config -- snapshot +/// thresholds, grace age, a fake `boot_ms_fn`, etc. `pool_prefix`/`server_root_id` are pinned so every +/// test in this file addresses the same pool shape. +PoolPtr openPoolWithConfig(const BackendPtr & backend, PoolConfig config) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + DB::Cas::tests::seedPoolMetaForRestart(*backend); /// see `openPool` above + return Pool::open(backend, std::move(config)); +} + +/// Mirrors gtest_cas_part_write.cpp's startBuildFor/publishOneBlobPart, minus the blob (an empty-entry +/// manifest is a legal, blob-free part -- the ref-writer tests only care about ref/manifest identity). +/// +/// Stage B (Task 4-C): pin `ns` to the sentinel before the first real touch -- ONE choke point for +/// every test in this file, since every real-path setup here funnels through `startBuildFor` (directly, +/// or via `publishEmptyPart` below). Many of this file's tests separately compute an expected key via +/// `DB::Cas::tests::fixture::fixtureLife(ns)` for fault injection/verification; without this the real +/// production birth mints a random incarnation and those computed keys land nowhere real. +PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + return s->beginPartWrite(info); +} + +ManifestId publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + auto build = startBuildFor(s, ns, ref); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +void publishWithProductionBirth(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns; + }); + if (it == catalog.entries.end()) + throw std::runtime_error("catalog entry missing from test fixture"); + return *it; +} + +CatalogEntry replaceCatalogLifeForRuntimeRace( + Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) +{ + const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + RefCatalog without_predecessor = before_delete.catalog; + std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) + { + return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; + }); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to retire exact predecessor catalog life"); + + CatalogEntry successor{ + .ns = predecessor.ns, + .state = NsState::Live, + .incarnation = successor_incarnation, + .creator = std::nullopt}; + const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + RefCatalog reborn = after_delete.catalog; + reborn.entries.push_back(successor); + if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome + != CasOutcome::Committed) + throw std::runtime_error("test failed to publish successor catalog life"); + return successor; +} + +std::optional listGreatestLogIdForTest( + Backend & backend, const Layout & layout, const RootNamespace & ns); + +std::optional listGreatestLogIdForLifeForTest( + Backend & backend, const Layout & layout, const NamespaceLifeId & life) +{ + std::optional greatest; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(life), cursor, 1000); + for (const ListedKey & listed : page.keys) + { + const auto parsed = layout.parseRefObjectKey(listed.key); + if (parsed && parsed->life_id == life.incarnation && parsed->kind == RefObjectKind::Log + && (!greatest || *greatest < parsed->txn_id)) + greatest = parsed->txn_id; + } + if (page.next_cursor.empty()) + return greatest; + cursor = page.next_cursor; + } +} + +struct CompletedRemovingFixture +{ + CatalogEntry predecessor; + uint64_t writer_epoch = 0; + uint64_t runtime_identity = 0; +}; + +CompletedRemovingFixture prepareResidentRemovalForDrain( + const PoolPtr & store, Backend & backend, const RootNamespace & ns, Gc & gc) +{ + publishWithProductionBirth(store, ns, "predecessor"); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, store->layout(), ns); + const uint64_t writer_epoch = store->liveWriterEpoch(); + const uint64_t runtime_identity = store->refTableRuntimeIdentityForTest(ns); + + if (runRegularRoundReclaiming(gc).deferred) + throw std::runtime_error("fixture publish unexpectedly deferred"); + store->dropNamespace(ns); + const CatalogEntry removing = catalogEntryOrThrow(backend, store->layout(), ns); + if (removing.state != NsState::Removing || removing.incarnation != predecessor.incarnation) + throw std::runtime_error("fixture removal did not publish the expected exact Removing row"); + if (runRegularRoundReclaiming(gc).deferred) + throw std::runtime_error("fixture terminal fold unexpectedly deferred"); + + const GcState state = decodeGcState(backend.get(store->layout().gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend.get(store->layout().foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + const auto row = seal.ref_lives.find(predecessor.incarnation); + if (row == seal.ref_lives.end() || !row->second.cleanup_evidence) + throw std::runtime_error("fixture terminal fold produced no cleanup evidence"); + return {predecessor, writer_epoch, runtime_identity}; +} + +ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) +{ + return ManifestRef{epoch, seq, ordinal}; +} + +/// Task 11: an INDEPENDENT ground truth for "cache-replay equivalence" tests -- lists every `_log/` +/// key under `ns` directly off the backend (ignoring any snapshot), decodes and replays them in id +/// order via the SAME shared state machine the writer uses, and returns the resulting state. A +/// published snapshot's bytes must equal `encodeRefTableSnapshot(snapshotOf(replay-through-X, ns))` +/// for this oracle's replay truncated at `X`. +RefTableState independentFullReplayForTest(Backend & backend, const Layout & layout, const RootNamespace & ns, + std::optional up_to = std::nullopt) +{ + std::vector ids; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation && parsed->kind == RefObjectKind::Log + && (!up_to || !(*up_to < parsed->txn_id))) + ids.push_back(parsed->txn_id); + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + std::sort(ids.begin(), ids.end()); + + RefTableState state; + for (const RefTxnId & id : ids) + { + const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + applyRefLogTxn(state, decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id)); + } + return state; +} + +/// The greatest `_snap/.proto` key currently present for `ns`, found via a fresh LIST (independent +/// of the Pool's own cached bookkeeping). +std::optional listGreatestSnapshotIdForTest(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + std::optional greatest; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation && parsed->kind == RefObjectKind::Snap + && (!greatest || *greatest < parsed->txn_id)) + greatest = parsed->txn_id; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return greatest; +} + +/// A backend that can (a) force one `get()` on a chosen exact key to return absent exactly once +/// (simulating an object vanishing after recovery sampled its exact checkpoint, with an optional side effect +/// fired at that exact moment -- e.g. publishing a covering newer snapshot, mirroring a concurrent GC +/// cleanup+republish race), and (b) force `putIfAbsent` on keys matching a chosen substring to throw an +/// ambiguous (Unresolved-classified) exception a bounded number of times, optionally still capturing +/// the (key, bytes) so a test can later "deliver" it -- simulating a request whose RESPONSE was lost +/// even though the write eventually landed server-side. +class RefWriterTestBackend : public CountingBackend +{ +public: + RefWriterTestBackend() + { + DB::Cas::tests::seedPoolMetaForRestart(*this); + } + + using CountingBackend::get; + using CountingBackend::getStream; + using CountingBackend::putIfAbsent; + using CountingBackend::putOverwrite; + using CountingBackend::casPut; + + void clearRequestJournal() + { + std::lock_guard lock(request_journal_mutex); + request_journal.clear(); + } + + void recordRequestJournalEvent(String event) + { + std::lock_guard lock(request_journal_mutex); + request_journal.push_back(std::move(event)); + } + + std::vector requestJournal() const + { + std::lock_guard lock(request_journal_mutex); + return request_journal; + } + + std::set vanish_once_keys; + std::function on_vanish_fire; + + enum class CatalogCasFault : uint8_t + { + None, + CommitThenThrow, + OtherWriterReplacement, + }; + CatalogCasFault catalog_cas_fault = CatalogCasFault::None; + String catalog_fault_key; + String catalog_replacement_bytes; + int catalog_resolution_get_fault_count = 0; + bool catalog_cas_fault_fired = false; + /// Fail one selected catalog GET after allowing an exact number of earlier catalog GETs through. + /// This reaches the removal lane's post-close observation without faulting its initial discovery. + int catalog_gets_before_fault = -1; + int catalog_get_fault_count = 0; + + String fault_key_substr; + int fault_count = 0; + /// Let the first `fault_skip` matching PUTs through untouched before `fault_count` starts faulting. + /// Needed now that recovery's in-band epoch seal (INV-2) shares the `_log/` prefix with every other + /// write under a namespace: a test that wants to fault something LATER in the same prefix (e.g. the + /// stale-precommit sweep's removal chunk) must skip past recovery's own seal writes first. Same + /// seam as `ChunkFaultBackend::fault_skip` in `cas_test_helpers.h`. + int fault_skip = 0; + std::optional> pending_delayed_write; + + /// (I1) On a matching `putIfAbsent`, a FOREIGN writer lands a DIFFERENT object at the exact key and + /// then this attempt's response is lost -- so the controller's resolve-before-reissue GET observes + /// different bytes and must raise CORRUPTED_DATA (a proven conflict, never a retry signal). + /// By default the foreign object is the attempt's own bytes plus a trailing marker -- UNDECODABLE + /// for zstd-framed objects (the frame size no longer matches), which is exactly right for tests + /// that pin fail-closed handling of a corrupt object. Tests that instead need a VALID foreign + /// object (e.g. a real cross-process seal to be adopted on retry) set `corrupt_foreign_bytes`. + String corrupt_key_substr; + int corrupt_count = 0; + String corrupt_foreign_bytes; + + String ckpt_conflict_key; + size_t ckpt_conflict_count = 0; + String ckpt_get_hook_key; + std::function ckpt_get_hook; + + /// Force a stream `LIST` to throw a transient object-store error (S3_ERROR) a bounded number of + /// times. Recovery must not consume this injection; callers that intentionally enumerate still do. + int list_fault_count = 0; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (list_fault_count > 0) + { + --list_fault_count; + throw DB::Exception(DB::ErrorCodes::S3_ERROR, "RefWriterTestBackend: simulated transient LIST failure"); + } + return CountingBackend::list(prefix, cursor, limit); + } + + std::optional get(const String & key, Range range) override + { + recordRequestJournalEvent("GET " + key); + if (key == ckpt_get_hook_key && ckpt_get_hook) + { + auto hook = std::exchange(ckpt_get_hook, nullptr); + hook(); + } + if (key == catalog_fault_key && catalog_get_fault_count > 0 && catalog_gets_before_fault >= 0) + { + if (catalog_gets_before_fault == 0) + { + --catalog_get_fault_count; + throw std::runtime_error("RefWriterTestBackend: simulated catalog admission read failure"); + } + --catalog_gets_before_fault; + } + if (catalog_cas_fault_fired && key == catalog_fault_key && catalog_resolution_get_fault_count > 0) + { + --catalog_resolution_get_fault_count; + throw std::runtime_error("RefWriterTestBackend: simulated catalog resolution read failure"); + } + const auto it = vanish_once_keys.find(key); + if (it != vanish_once_keys.end()) + { + vanish_once_keys.erase(it); + if (on_vanish_fire) + { + auto fire = std::move(on_vanish_fire); + on_vanish_fire = nullptr; + fire(); + } + return std::nullopt; + } + return CountingBackend::get(key, range); + } + + CasResult casPut( + const String & key, const String & bytes, const std::optional & expected, + const ObjectMeta & meta) override + { + recordRequestJournalEvent("CAS " + key); + if (key == ckpt_conflict_key && ckpt_conflict_count > 0) + { + --ckpt_conflict_count; + return {CasOutcome::Conflict, {}}; + } + if (key == catalog_fault_key && catalog_cas_fault != CatalogCasFault::None) + { + const CatalogCasFault fault = std::exchange(catalog_cas_fault, CatalogCasFault::None); + catalog_cas_fault_fired = true; + if (fault == CatalogCasFault::CommitThenThrow) + { + const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); + if (result.outcome != CasOutcome::Committed) + return result; + throw Poco::TimeoutException( + "RefWriterTestBackend: catalog CAS committed but its response was lost"); + } + + CasResult replacement = CountingBackend::casPut( + key, catalog_replacement_bytes, expected, meta); + if (replacement.outcome != CasOutcome::Committed) + return replacement; + return {CasOutcome::Conflict, {}}; + } + return CountingBackend::casPut(key, bytes, expected, meta); + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + recordRequestJournalEvent("PUT " + key); + if (corrupt_count > 0 && !corrupt_key_substr.empty() && key.find(corrupt_key_substr) != String::npos) + { + --corrupt_count; + /// A foreign writer lands a DIFFERENT object at this exact key; then our own response is lost. + CountingBackend::putIfAbsent( + key, corrupt_foreign_bytes.empty() ? bytes + String("\x01_FOREIGN_DIFFERENT") : corrupt_foreign_bytes); + throw Poco::TimeoutException("RefWriterTestBackend: a foreign different object landed; response lost"); + } + if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + { + if (fault_skip > 0) + { + --fault_skip; + } + else if (fault_count > 0) + { + --fault_count; + pending_delayed_write = {key, bytes}; + throw Poco::TimeoutException("RefWriterTestBackend: simulated ambiguous result (response lost)"); + } + } + { + std::unique_lock lk(block_mutex); + bool block_this = false; + if (block_armed && key.find(block_substr) != String::npos) + { + if (!block_first_match_only) + block_this = true; /// block EVERY matching put (the original mode) + else if (blocked_key.empty()) + { + blocked_key = key; /// first match: capture and block exactly this key + block_this = true; + } + else if (key == blocked_key) + block_this = true; /// the SAME captured key retried: keep blocking it + /// a DIFFERENT matching key under first-match-only mode falls through unblocked + } + /// (I1) Independent per-key blocking: every matching key parks on its OWN release, unlike + /// `block_armed` above (one shared gate released all-at-once). Lets a test park two DISTINCT + /// `_snap/` PUTs concurrently and release them in a chosen order. + if (independent_block_armed && key.contains(independent_block_substr)) + { + independent_blocked_keys.insert(key); + block_cv.notify_all(); + block_cv.wait(lk, [&] { return independent_released_keys.contains(key); }); + } + if (block_this) + { + block_entered = true; + block_cv.notify_all(); + block_cv.wait(lk, [&] { return !block_armed; }); + /// fix-round F3-1a (CRITICAL, unlock-throw race harness): on release, behave like + /// `corrupt_key_substr` above instead of proceeding normally -- a foreign writer landed + /// DIFFERENT bytes at this exact key while we were parked, so our own attempt is a + /// PROVEN conflict once `putIfAbsentControlled`'s resolve-before-reissue GETs it. Lets a + /// test make the recovery seal's PUT throw CORRUPTED_DATA from INSIDE the unlocked + /// window, deterministically, instead of merely returning a non-Committed outcome. + if (block_throw_corrupted_on_release) + { + lk.unlock(); + CountingBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT")); + { + std::lock_guard g(block_mutex); + block_call_completed = true; + } + block_cv.notify_all(); + throw Poco::TimeoutException( + "RefWriterTestBackend: a foreign different object landed on release; response lost"); + } + } + } + const PutResult r = CountingBackend::putIfAbsent(key, bytes, meta); + { + std::lock_guard g(block_mutex); + block_call_completed = true; + } + block_cv.notify_all(); + return r; + } + /// See `putIfAbsent`'s `block_this` branch. Set before spawning any thread that could race + /// `putIfAbsent`, like `corrupt_key_substr`/`fault_key_substr` above -- not itself lock-protected. + bool block_throw_corrupted_on_release = false; + + /// "Deliver" the earlier ambiguous write: the request DID eventually land server-side, the caller + /// just never saw the ack. No-op if no fault has fired since the last delivery. + void materializePendingDelayedWrite() + { + if (pending_delayed_write) + { + CountingBackend::putIfAbsent(pending_delayed_write->first, pending_delayed_write->second); + pending_delayed_write.reset(); + } + } + + /// Task 11: blocks EVERY `putIfAbsent()` whose key contains `armed_block_substr` until + /// `releaseBlock()` is called, notifying `awaitBlockEntered()` the first time one is reached. Used + /// to prove snapshot publication never holds up an unrelated concurrent append. + void armPutBlock(const String & substr) + { + std::lock_guard g(block_mutex); + block_substr = substr; + block_armed = true; + block_entered = false; + block_call_completed = false; + block_first_match_only = false; + blocked_key.clear(); + } + + /// Task 11 (monotonic-adoption harness): block ONLY the FIRST `putIfAbsent` whose key contains + /// `substr`, capturing that exact key; every LATER put -- including a DIFFERENT `_snap/` key -- + /// proceeds unblocked. Lets a test pin one in-flight publish's PUT mid-flight while a second, + /// higher-id publish runs to completion, deterministically forcing the out-of-order overlap. + void armPutBlockFirstMatchOnly(const String & substr) + { + std::lock_guard g(block_mutex); + block_substr = substr; + block_armed = true; + block_entered = false; + block_call_completed = false; + block_first_match_only = true; + blocked_key.clear(); + } + void awaitBlockEntered() + { + std::unique_lock lk(block_mutex); + block_cv.wait(lk, [&] { return block_entered; }); + } + void releaseBlock() + { + { + std::lock_guard g(block_mutex); + block_armed = false; + } + block_cv.notify_all(); + } + /// Blocks until the PREVIOUSLY-blocked `putIfAbsent` call has actually RETURNED (not merely been + /// unblocked) -- i.e. its underlying `CountingBackend::putIfAbsent` has completed. Deterministic, + /// sleep-free way to observe a detached background caller's own work finishing when the TEST no + /// longer holds anything (e.g. a Pool handle) that call would otherwise let it wait on. + void awaitBlockedCallCompleted() + { + std::unique_lock lk(block_mutex); + block_cv.wait(lk, [&] { return block_call_completed; }); + } + + /// (I1 regression harness) Arms independent per-key blocking for every `putIfAbsent` matching + /// `substr`: unlike `armPutBlock`/`armPutBlockFirstMatchOnly` (one shared release gate), each + /// blocked key parks on ITS OWN release (`releaseKey`), so two distinct `_snap/` PUTs can be + /// parked concurrently -- both past their capture point, neither yet adopted -- and released in a + /// chosen order. Needed to construct the small-candidate-adopts-before-a-larger-one-already-in-flight + /// ordering that exercises `clampedCounterSub`'s actual clamp branch. + void armPutBlockIndependently(const String & substr) + { + std::lock_guard g(block_mutex); + independent_block_substr = substr; + independent_block_armed = true; + independent_blocked_keys.clear(); + independent_released_keys.clear(); + } + /// Blocks until at least `n` distinct matching keys are currently parked. + void awaitAtLeastNKeysBlocked(size_t n) + { + std::unique_lock lk(block_mutex); + block_cv.wait(lk, [&] { return independent_blocked_keys.size() >= n; }); + } + /// A snapshot of the keys currently parked under independent blocking. + std::set blockedKeysSnapshot() + { + std::lock_guard g(block_mutex); + return independent_blocked_keys; + } + /// Releases exactly the given key; every OTHER independently-blocked key stays parked. + void releaseKey(const String & key) + { + { + std::lock_guard g(block_mutex); + independent_released_keys.insert(key); + } + block_cv.notify_all(); + } + +private: + mutable std::mutex request_journal_mutex; + std::vector request_journal; + std::mutex block_mutex; + std::condition_variable block_cv; + String block_substr; + bool block_armed = false; + bool block_entered = false; + bool block_call_completed = false; + bool block_first_match_only = false; + String blocked_key; + String independent_block_substr; + bool independent_block_armed = false; + std::set independent_blocked_keys; + std::set independent_released_keys; +}; + +} + +/// =================================================================================== +/// Recovery (spec §Recovery / exact checkpoint grounding) +/// =================================================================================== + +TEST(CASRefWriterRecovery, EmptyNamespaceRecoversToEmptyState) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/never_touched"}; + + EXPECT_TRUE(store->listRefs(ns).empty()); + EXPECT_FALSE(store->resolveRef(ns, "anything").has_value()); + EXPECT_EQ(store->refRecoveryRestartsForTest(ns), 0u); +} + +TEST(CASRefWriterNonMinting, ListRefsOnAbsentNamespaceDoesNotMutateCatalog) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/list_absent_non_minting"}; + const auto catalog_before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_before); + backend->resetCounts(); + + EXPECT_TRUE(store->listRefs(ns).empty()); + + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); + const auto catalog_after = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_after); + EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); + EXPECT_EQ(catalog_after->token, catalog_before->token); +} + +TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsCatalogLifeReplacedWithoutLocalInvalidation) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/cold-read-catalog-aba"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + store->setReadableCatalogAfterObservationHookForTest([&] + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }); + + std::exception_ptr stale_error; + std::thread stale_reader([&] + { + try + { + (void)store->listRefs(ns); + } + catch (...) + { + stale_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + const CatalogEntry successor + = replaceCatalogLifeForRuntimeRace(*backend, layout, predecessor, UInt128{0xabc002}); + const NamespaceLifeId successor_life + = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + .life_epoch = store->liveWriterEpoch(), + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + stale_reader.join(); + store->setReadableCatalogAfterObservationHookForTest(nullptr); + + EXPECT_TRUE(stale_error) << "the stale catalog life was published instead of refused"; + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + EXPECT_NO_THROW((void)store->listRefs(ns)); + ASSERT_TRUE(store->refTableLifeForTest(ns)); + EXPECT_EQ(*store->refTableLifeForTest(ns), successor_life); +} + +TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsReplacementByExternalPoolActor) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + PoolConfig external_config{.pool_prefix = "p", .server_root_id = "external-runtime-race"}; + auto external_store = Pool::open(backend, std::move(external_config)); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/external-catalog-runtime-publication"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + CatalogEntry successor; + store->setReadableCatalogAfterObservationHookForTest([&] + { + successor = replaceCatalogLifeForRuntimeRace( + external_store->backend(), external_store->layout(), predecessor, UInt128{0xabc003}); + const NamespaceLifeId successor_life + = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); + if (external_store->backend().putIfAbsent( + external_store->layout().refCkptKey(successor_life), + encodeRefCkpt(RefCkpt{ + .life_epoch = external_store->liveWriterEpoch(), + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome != PutOutcome::Done) + throw std::runtime_error("test failed to publish external successor checkpoint"); + }); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); + store->setReadableCatalogAfterObservationHookForTest(nullptr); + + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + EXPECT_NO_THROW((void)store->listRefs(ns)); + ASSERT_TRUE(store->refTableLifeForTest(ns)); + EXPECT_EQ(*store->refTableLifeForTest(ns), + NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation)); +} + +/// The cold-reader revalidation is about THIS namespace's row, not whole-catalog stillness: an +/// unrelated namespace admitted between the two observations must not refuse the admission (that +/// refusal starved cold admissions under a parallel workload sharing one pool), while the target's +/// own row staying identical still publishes the runtime against the observed life. +TEST(CASRefWriterRuntimeIdentity, ColdReadAdmitsThroughUnrelatedCatalogMutationBetweenObservations) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/unrelated-catalog-runtime-publication"}; + const RootNamespace unrelated{"srv1/unrelated-catalog-row"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + store->setReadableCatalogAfterObservationHookForTest([&] + { + DB::Cas::tests::fixture::admitLive(*backend, layout, unrelated); + }); + + EXPECT_NO_THROW((void)store->listRefs(ns)); + store->setReadableCatalogAfterObservationHookForTest(nullptr); + + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), 0u); +} + +/// The per-row narrowing must not skip the second cut's ambiguity validation: an ALIASING incarnation +/// admitted between the two observations (another namespace stealing this life's incarnation -- +/// physical life-owned keys use only the incarnation) leaves the target's own row byte-identical yet +/// must still refuse the admission. The whole-catalog comparison refused this implicitly; the +/// narrowed check must refuse it explicitly. +TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsAliasingIncarnationAdmittedBetweenObservations) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/aliasing-incarnation-target"}; + const RootNamespace alias{"srv1/aliasing-incarnation-thief"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + const CatalogEntry target_row = catalogEntryOrThrow(*backend, layout, ns); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + store->setReadableCatalogAfterObservationHookForTest([&] + { + CatalogEntry thief; + thief.ns = alias; + thief.state = NsState::Creating; + thief.incarnation = target_row.incarnation; + thief.creator = CreatorFence{ + .server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, thief); + }); + + EXPECT_THROW((void)store->listRefs(ns), DB::Exception); + store->setReadableCatalogAfterObservationHookForTest(nullptr); + + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); +} + +TEST(CASRefWriterRuntimeIdentity, WarmReadableRuntimeDoesNotReadCatalog) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/warm-runtime-zero-catalog-get"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); + + EXPECT_NO_THROW((void)store->listRefs(ns)); + ASSERT_NE(store->refTableRuntimeIdentityForTest(ns), 0u); + backend->resetCounts(); + + EXPECT_NO_THROW((void)store->listRefs(ns)); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 0u); +} + +/// `DROP DETACHED PART` reaches this point lookup for a part that may already be absent. Its probe +/// must not turn a missing table namespace into a new catalog life. +TEST(CASRefWriterNonMinting, ResolveRefOnAbsentNamespaceDoesNotMutateCatalog) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/resolve_absent_non_minting"}; + const auto catalog_before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_before); + backend->resetCounts(); + + EXPECT_FALSE(store->resolveRef(ns, "detached_part").has_value()); + + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); + const auto catalog_after = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_after); + EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); + EXPECT_EQ(catalog_after->token, catalog_before->token); +} + +TEST(CASRefWriterNonMinting, DropNamespaceOnAbsentNamespaceDoesNotMutateCatalog) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/drop_absent_non_minting"}; + const auto catalog_before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_before); + backend->resetCounts(); + + store->dropNamespace(ns); + + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); + const auto catalog_after = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(catalog_after); + EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); + EXPECT_EQ(catalog_after->token, catalog_before->token); +} + +/// A table born by a log tail alone (no snapshot yet): `namespace_birth` with nothing else is a legal +/// Live-but-empty table. +TEST(CASRefWriterRecovery, BirthOnlyLogNoSnapshotRecoversToEmptyLiveTable) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/birth_only"}; + + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, {namespaceBirthOp()}, std::nullopt}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + auto store = openPool(backend); + EXPECT_TRUE(store->listRefs(ns).empty()); +} + +/// Empty base + birth log recovery (spec unit test list): birth and the first precommit->promote span +/// TWO separate log transactions with no snapshot at all. +TEST(CASRefWriterRecovery, BirthPlusPrecommitPromoteAcrossTwoLogsNoSnapshot) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/birth_then_promote"}; + const ManifestRef m1 = manifestRef(1, 1, 1); + + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, + {namespaceBirthOp(), publishCommittedOps("part_1", m1)[0]}, std::nullopt}); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 2}, + {publishCommittedOps("part_1", m1)[1]}, std::nullopt}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + auto store = openPool(backend); + const auto resolved = store->resolveRef(ns, "part_1"); + ASSERT_TRUE(resolved.has_value()); + EXPECT_EQ(resolved->manifest_id.ref, m1); + EXPECT_EQ(resolved->manifest_id.root_namespace, ns); + + const auto refs = store->listRefs(ns); + ASSERT_EQ(refs.size(), 1u); + EXPECT_TRUE(refs.contains("part_1")); +} + +TEST(CASRefWriterRecovery, TerminalGapBelowCheckpointFrontierIsCorruptionNotSameLifeRebirth) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/writer_terminal_gap"}; + + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + ns.string(), RefTxnId{1, 1}, {namespaceBirthOp()}, std::nullopt}); + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + ns.string(), RefTxnId{1, 2}, {std::move(remove)}, std::nullopt}); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + ns.string(), RefTxnId{2, 1}, {namespaceBirthOp()}, std::nullopt}); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }); + + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const String next_log_key = layout.refLogKey(life, RefTxnId{2, 2}); + auto store = openPool(backend); + const uint64_t installs_before = store->recoveryInstallCountForTest(); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); + EXPECT_FALSE(store->refTableRecoveredForTest(ns)); + EXPECT_EQ(store->recoveryInstallCountForTest(), installs_before); + + EXPECT_ANY_THROW((void)publishEmptyPart(store, ns, "must_not_allocate")); + EXPECT_EQ(backend->putCount(next_log_key), 0u) + << "an unrecovered malformed life must not allocate the next writer position"; +} + +/// Latest snapshot plus tail recovery (spec unit test list): a snapshot covering ref "a", a tail that +/// drops "a" and publishes "b", and a STALE log at/below the snapshot id that must be ignored (its +/// content, if replayed, would corrupt the result -- proving the "ignore log keys at or below the +/// selected snapshot" rule). +TEST(CASRefWriterRecovery, SnapshotPlusTailRecovery) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/snap_tail"}; + const ManifestRef ma = manifestRef(1, 1, 1); + const ManifestRef mb = manifestRef(1, 2, 1); + + /// A stale log BELOW the snapshot id would, if wrongly replayed, try to add "a" a second time + /// (the snapshot already contains it) and throw -- proving it must be ignored, not merely benign. + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 3}, + {namespaceBirthOp(), publishCommittedOps("a", ma)[0], publishCommittedOps("a", ma)[1]}, std::nullopt}); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = RefTxnId{1, 5}, + .ops = publishCommittedOps("a", ma), + .prev_epoch_seal = std::nullopt}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), RefTxnId{1, 5}, {committedRow("a", ma)})); + + std::vector tail_ops; + tail_ops.push_back([&] { RefOp op; op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Committed, "a", ma}; return op; }()); + tail_ops.push_back(publishCommittedOps("b", mb)[0]); + tail_ops.push_back(publishCommittedOps("b", mb)[1]); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 6}, tail_ops, std::nullopt}); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 6}, + .checkpoint_snapshot_id = RefTxnId{1, 5}, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + backend->resetCounts(); + auto store = openPool(backend); + EXPECT_FALSE(store->resolveRef(ns, "a").has_value()); + const auto b = store->resolveRef(ns, "b"); + ASSERT_TRUE(b.has_value()); + EXPECT_EQ(b->manifest_id.ref, mb); + EXPECT_EQ(store->listRefs(ns).size(), 1u); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 5})), 1u) + << "recovery must validate the selected base's retained ordinary log"; + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, RefTxnId{1, 5})), 1u) + << "the fixture must reach and decode the selected base snapshot"; +} + +/// Restart-on-vanish (spec §Recovery): the checkpoint-named snapshot vanishes during its exact GET +/// while concurrent cleanup publishes a newer checkpoint base. Recovery must restart from the newer +/// exact checkpoint, not treat the vanish as corruption. +TEST(CASRefWriterRecovery, RestartOnVanishConvergesOnNewerSnapshot) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/vanish_race"}; + const ManifestRef ma = manifestRef(1, 1, 1); + const ManifestRef mb = manifestRef(1, 2, 1); + + /// Stage B (Task 4-C): pin `ns` to the sentinel before the raw snapshot below -- `store->resolveRef` + /// further down is a real production read that triggers `resolveNamespaceLife`, which for an + /// UNADMITTED namespace mints a fresh RANDOM incarnation rather than adopting the sentinel the raw + /// fixture writes at. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const RefTxnId snap_x{1, 10}; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = snap_x, + .ops = publishCommittedOps("a", ma), + .prev_epoch_seal = std::nullopt}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), snap_x, {committedRow("a", ma)})); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 10}, + .checkpoint_snapshot_id = snap_x, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + backend->vanish_once_keys.insert(layout.refSnapshotKey(life, snap_x)); + bool vanish_fired = false; + backend->on_vanish_fire = [&] + { + vanish_fired = true; + const RefTxnId snap_y{1, 20}; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = snap_y, + .ops = publishCommittedOps("b", mb), + .prev_epoch_seal = std::nullopt}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), snap_y, {committedRow("b", mb)})); + const auto before = backend->get(layout.refCkptKey(life)); + ASSERT_TRUE(before); + ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 20}, + .checkpoint_snapshot_id = snap_y, + .last_epoch_seal = std::nullopt}), before->token).outcome, CasOutcome::Committed); + }; + + backend->resetCounts(); + auto store = openPool(backend); + const auto b = store->resolveRef(ns, "b"); + ASSERT_TRUE(b.has_value()); + EXPECT_EQ(b->manifest_id.ref, mb); + EXPECT_FALSE(store->resolveRef(ns, "a").has_value()) << "must converge on snapshot Y, not a mix of X and Y"; + EXPECT_EQ(store->refRecoveryRestartsForTest(ns), 1u); + EXPECT_TRUE(vanish_fired) << "the fixture must reach the old snapshot GET and fire the replacement hook"; + EXPECT_FALSE(backend->vanish_once_keys.contains(layout.refSnapshotKey(life, snap_x))); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, snap_x)), 1u); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 20})), 1u); +} + +/// A DIFFERENT valid object at the exact snapshot key (not merely absent) is corruption, never a +/// restart signal -- pins the boundary between "vanished" (restart) and "corrupt" (fail closed). +TEST(CASRefWriterRecovery, DifferentBytesAtSelectedSnapshotIsCorruptionNotRestart) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/corrupt_snap"}; + const RefTxnId snap_x{1, 10}; + + /// A structurally-valid snapshot BODY, but for a DIFFERENT namespace, placed under `ns`'s own key + /// (a copy-under-the-wrong-prefix scenario) -- decodeRefTableSnapshot's key/body cross-check must + /// reject it, never treat it as a restart signal. + /// Stage B (Task 4-C): pin `ns` to the sentinel before the raw write below -- `store->resolveRef` + /// further down is a real production read that would otherwise mint a fresh RANDOM incarnation + /// for this unadmitted namespace instead of adopting the sentinel the raw fixture writes at. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const RootNamespace other_ns{"srv1/other"}; + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ + .ns = ns.string(), + .txn_id = snap_x, + .ops = publishCommittedOps("anchor", manifestRef(1, 10, 1)), + .prev_epoch_seal = std::nullopt}); + DB::Cas::RefTableSnapshot foreign; + foreign.ns = other_ns.string(); + foreign.snapshot_id = snap_x; + const String snapshot_key = layout.refSnapshotKey(life, snap_x); + ASSERT_EQ(backend->putIfAbsent(snapshot_key, + DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(foreign))).outcome, + PutOutcome::Done); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = snap_x, + .checkpoint_snapshot_id = snap_x, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + + auto store = openPool(backend); + backend->resetCounts(); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->resolveRef(ns, "anything"); }); + EXPECT_EQ(backend->getCount(layout.refLogKey(life, snap_x)), 1u) + << "the matching ordinary log must be validated before the selected snapshot"; + EXPECT_EQ(backend->getCount(snapshot_key), 1u) + << "the corruption must come from decoding the required checkpoint snapshot"; +} + +/// =================================================================================== +/// Append lane: request cost + batching (spec §Common Mutation Path / §Local Batching Queue) +/// =================================================================================== + +TEST(CASRefWriterAppendLane, CommittedChunkPublishesFrontierBeforeInstallAndAck) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/warm"}; + publishEmptyPart(store, ns, "part_1"); + publishEmptyPart(store, ns, "part_2"); + ASSERT_TRUE(store->resolveRef(ns, "part_1").has_value()); + ASSERT_TRUE(store->resolveRef(ns, "part_2").has_value()); + + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const String log_prefix = store->layout().namespaceStreamPrefix(life) + "_log/"; + const String ckpt_key = store->layout().refCkptKey(life); + const auto ckpt_before = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(ckpt_before); + ASSERT_TRUE(ckpt_before->ckpt.committed_through); + const RefTxnId expected_frontier{ + ckpt_before->ckpt.committed_through->writer_epoch, + ckpt_before->ckpt.committed_through->ref_sequence + 1}; + backend->clearRequestJournal(); + const uint64_t list_before = backend->listTotal(); + const uint64_t put_before = backend->putTotal(); + const uint64_t ckpt_get_before = backend->getCount(ckpt_key); + const uint64_t ckpt_cas_before = backend->casPutCount(ckpt_key); + + std::mutex mutex; + std::condition_variable cv; + bool pre_carve_entered = false; + bool post_install_entered = false; + bool release_post_install = false; + bool follower_returned = false; + store->setRefPreCarveHookForTest([&] + { + std::unique_lock lock(mutex); + if (pre_carve_entered) + return; + pre_carve_entered = true; + cv.notify_all(); + cv.wait(lock, [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::PostInstallPreAck) + return; + backend->recordRequestJournalEvent("INSTALL"); + std::unique_lock lock(mutex); + post_install_entered = true; + cv.notify_all(); + cv.wait(lock, [&] { return release_post_install; }); + }); + + std::exception_ptr leader_error; + std::exception_ptr follower_error; + std::thread leader([&] + { + try + { + store->dropRef(ns, "part_1"); + } + catch (...) + { + leader_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return pre_carve_entered; }); + } + std::thread follower([&] + { + try + { + store->dropRef(ns, "part_2"); + backend->recordRequestJournalEvent("FOLLOWER ACK"); + { + std::lock_guard lock(mutex); + follower_returned = true; + } + cv.notify_all(); + } + catch (...) + { + follower_error = std::current_exception(); + } + }); + while (store->refQueuePendingForTest(ns) < 2) + std::this_thread::yield(); + cv.notify_all(); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return post_install_entered; }); + } + + bool follower_returned_before_release = false; + { + std::lock_guard lock(mutex); + follower_returned_before_release = follower_returned; + } + std::exception_ptr observation_error; + bool part_1_visible = true; + bool part_2_visible = true; + try + { + part_1_visible = store->resolveRef(ns, "part_1").has_value(); + part_2_visible = store->resolveRef(ns, "part_2").has_value(); + } + catch (...) + { + observation_error = std::current_exception(); + } + { + std::lock_guard lock(mutex); + release_post_install = true; + } + cv.notify_all(); + leader.join(); + follower.join(); + store->setRefPreCarveHookForTest(nullptr); + store->setCarveHookForTest(nullptr); + + EXPECT_FALSE(follower_returned_before_release) + << "a co-batched waiter returned before the installed transaction was acknowledged"; + EXPECT_FALSE(observation_error); + EXPECT_FALSE(part_1_visible); + EXPECT_FALSE(part_2_visible) + << "both co-batched mutations must be visible before either waiter can return success"; + EXPECT_FALSE(leader_error); + EXPECT_FALSE(follower_error); + EXPECT_EQ(backend->listTotal(), list_before) << "a warm mutation performs no LIST"; + EXPECT_EQ(backend->putTotal(), put_before + 1) << "exactly one body PUT with create-if-absent"; + EXPECT_EQ(backend->getCount(ckpt_key), ckpt_get_before + 1) + << "one committed chunk pays exactly one checkpoint GET"; + EXPECT_EQ(backend->casPutCount(ckpt_key), ckpt_cas_before + 1) + << "one committed chunk pays exactly one checkpoint CAS"; + + const std::vector journal = backend->requestJournal(); + ASSERT_EQ(journal.size(), 5u); + EXPECT_EQ(journal[0].find("PUT " + log_prefix), 0u) << journal[0]; + EXPECT_EQ(journal[1], "GET " + ckpt_key); + EXPECT_EQ(journal[2], "CAS " + ckpt_key); + EXPECT_EQ(journal[3], "INSTALL"); + EXPECT_EQ(journal[4], "FOLLOWER ACK"); + + const auto durable_ckpt = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(durable_ckpt); + EXPECT_EQ(durable_ckpt->ckpt.committed_through, expected_frontier); +} + +TEST(CASRefWriterAppendLane, CheckpointConflictAfterLogCommitRequiresRecoveryWithoutInstall) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/frontier-conflict"}; + publishEmptyPart(store, ns, "x"); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const String ckpt_key = store->layout().refCkptKey(life); + const auto before = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(before); + ASSERT_TRUE(before->ckpt.committed_through); + const RefTxnId candidate{before->ckpt.committed_through->writer_epoch, + before->ckpt.committed_through->ref_sequence + 1}; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + backend->ckpt_conflict_key = ckpt_key; + backend->ckpt_conflict_count = 100; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) + << "the durable log was not installed or acknowledged"; + const auto after = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(after); + EXPECT_EQ(after->ckpt.committed_through, before->ckpt.committed_through); + EXPECT_TRUE(backend->get(store->layout().refLogKey(life, candidate))) + << "the log PUT committed before checkpoint publication failed"; + EXPECT_FALSE(backend->get(store->layout().refLogKey( + life, RefTxnId{candidate.writer_epoch, candidate.ref_sequence + 1}))) + << "no later id may be allocated above an unfrontiered durable transaction"; +} + +TEST(CASRefWriterAppendLane, FenceMovementAtCheckpointPublicationRequiresRecoveryWithoutInstall) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/frontier-fenced"}; + publishEmptyPart(store, ns, "x"); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const String ckpt_key = store->layout().refCkptKey(life); + const auto before = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(before); + ASSERT_TRUE(before->ckpt.committed_through); + const RefTxnId candidate{before->ckpt.committed_through->writer_epoch, + before->ckpt.committed_through->ref_sequence + 1}; + const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + + backend->ckpt_get_hook_key = ckpt_key; + backend->ckpt_get_hook = [&] { store->tripMountLost(); }; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) + << "the fenced frontier attempt must not install or acknowledge the durable log"; + const auto after = readCkpt(*backend, store->layout(), life); + ASSERT_TRUE(after); + EXPECT_EQ(after->ckpt.committed_through, before->ckpt.committed_through); + EXPECT_TRUE(backend->get(store->layout().refLogKey(life, candidate))); + EXPECT_FALSE(backend->get(store->layout().refLogKey( + life, RefTxnId{candidate.writer_epoch, candidate.ref_sequence + 1}))); +} + +/// Phase 3 (reftable-cow-map materialization): each of these N +/// publishes is its own isolated (unbatched) flush touching exactly one NEW ref -- if +/// `flushRefBatch` did not materialize `rt->state.committed` after installing each flush's +/// transaction, the overlay would grow by ~1 entry per flush and this would read back ~N, +/// defeating the whole point of the COW map for a long-running table. +TEST(CASRefWriterAppendLane, MaterializeKeepsOverlaySmallAcrossManyIsolatedFlushes) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/cowmap"}; + + constexpr int kRefs = 20; + for (int i = 0; i < kRefs; ++i) + publishEmptyPart(store, ns, "ref" + std::to_string(i)); + + EXPECT_LE(store->committedOverlayEntriesForTest(ns), 1u); + EXPECT_EQ(store->listRefs(ns).size(), static_cast(kRefs)); /// sanity: all N really committed +} + +/// `B` compatible queued mutations share one create (spec §Writer Budget). +TEST(CASRefWriterAppendLane, CompatibleMutationsShareOneCreate) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/cobatch"}; + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); + ASSERT_TRUE(store->resolveRef(ns, "a").has_value()); + ASSERT_TRUE(store->resolveRef(ns, "b").has_value()); + + std::mutex m; + std::condition_variable cv; + bool entered = false; + store->setRefPreCarveHookForTest([&] + { + std::unique_lock lk(m); + if (entered) + return; /// only the leader's own first carve blocks; a second flush (if any) proceeds + entered = true; + cv.notify_all(); + cv.wait(lk, [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + + const uint64_t put_before = backend->putTotal(); + std::thread t_a([&] { store->dropRef(ns, "a"); }); + { + std::unique_lock lk(m); + cv.wait(lk, [&] { return entered; }); + } + std::thread t_b([&] { store->dropRef(ns, "b"); }); + while (store->refQueuePendingForTest(ns) < 2) + std::this_thread::yield(); + cv.notify_all(); /// wakes the pre-carve hook's own wait once its predicate (>=2 pending) holds + t_a.join(); + t_b.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_EQ(backend->putTotal(), put_before + 1) << "both drops must land in ONE created log object"; + EXPECT_FALSE(store->resolveRef(ns, "a").has_value()); + EXPECT_FALSE(store->resolveRef(ns, "b").has_value()); +} + +/// An invalid queued request returns its own exception without entering the transaction; the +/// co-batched neighbor still lands, in the SAME one create. +TEST(CASRefWriterAppendLane, InvalidBatchEntryGetsOwnExceptionBatchSurvives) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/invalid_entry"}; + publishEmptyPart(store, ns, "good"); + + std::mutex m; + std::condition_variable cv; + bool entered = false; + store->setRefPreCarveHookForTest([&] + { + std::unique_lock lk(m); + if (entered) + return; + entered = true; + cv.notify_all(); + cv.wait(lk, [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + + const uint64_t put_before = backend->putTotal(); + std::exception_ptr bad_error; + std::thread t_bad([&] + { + try { store->dropRef(ns, "does_not_exist"); } + catch (...) { bad_error = std::current_exception(); } + }); + { + std::unique_lock lk(m); + cv.wait(lk, [&] { return entered; }); + } + std::thread t_good([&] { store->dropRef(ns, "good"); }); + while (store->refQueuePendingForTest(ns) < 2) + std::this_thread::yield(); + cv.notify_all(); + t_bad.join(); + t_good.join(); + store->setRefPreCarveHookForTest(nullptr); + + ASSERT_TRUE(bad_error != nullptr) << "the invalid item's OWN caller must receive its exception"; + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { std::rethrow_exception(bad_error); }); + EXPECT_EQ(backend->putTotal(), put_before + 1) << "the survivor's own transaction still costs one create"; + EXPECT_FALSE(store->resolveRef(ns, "good").has_value()) << "the innocent co-batched drop must land"; +} + +/// =================================================================================== +/// Append lane: wedge semantics (spec §Writer-Side Linearization) +/// =================================================================================== + +TEST(CASRefWriterAppendLane, WedgedLaneBlocksSameTableWhileOtherTableProceeds) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns_a{"srv1/wedge_a"}; + const RootNamespace ns_b{"srv1/wedge_b"}; + publishEmptyPart(store, ns_a, "x"); + publishEmptyPart(store, ns_a, "x_second"); + publishEmptyPart(store, ns_b, "y"); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_a)) + "_log/"; + backend->fault_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_a, "x"); }); + EXPECT_TRUE(store->refLaneWedgedForTest(ns_a)); + + /// A different table proceeds normally while ns_a stays wedged. + EXPECT_NO_THROW(store->dropRef(ns_b, "y")); + EXPECT_FALSE(store->resolveRef(ns_b, "y").has_value()); + + /// Retrying ns_a does not allocate a later id -- it re-attempts the SAME one. The wedge's key was + /// never actually written (the fault never wrote through), and under the every-attempt rule the + /// retry is a conditional CREATE of the same bytes rather than a bare read: it lands, which makes + /// the wedged transaction durable and adopts it. That is the point of the rule -- a bare read could + /// only ever report "absent", which is not a rejection, and the lane would stay wedged forever over + /// a key nothing had written. See `gtest_cas_ref_wedge_every_attempt.cpp` for the full rule. + EXPECT_NO_THROW(store->dropRef(ns_a, "x_second")); + EXPECT_FALSE(store->refLaneWedgedForTest(ns_a)) << "the retry's own create resolves the lane"; + EXPECT_FALSE(store->resolveRef(ns_a, "x").has_value()) << "the wedged drop was adopted on resolution"; +} + +TEST(CASRefWriterAppendLane, WedgedAppendObservedDurableAppliesBeforeNextId) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/wedge_unwedge"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "not yet applied while wedged"; + + /// The earlier request eventually lands server-side; the caller just never saw the ack. + backend->materializePendingDelayedWrite(); + + /// A later mutation on the SAME table first resolves the wedge (applying "drop x" to cache) BEFORE + /// allocating its own next id (which drops "y"). + EXPECT_NO_THROW(store->dropRef(ns, "y")); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the wedged drop was applied on resolution"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()) << "the next mutation committed normally afterward"; +} + +/// Wedge tail-counter accounting across the three states (xhigh review, item F): an UNRESOLVED wedge +/// applied nothing, so it must NOT bump the applied-above-snapshot tail counters; a RESOLVED wedge is a +/// commit like any other and MUST bump them (exactly once) alongside the ordinary commit that resolves +/// it; and the resolution must fold its applied overlay in place (no residual committed overlay). Under +/// the default 256-log / 1 MiB snapshot thresholds this handful of txns never triggers a publish, so the +/// tail counter is a stable running count. +TEST(CASRefWriterAppendLane, WedgeResolutionJoinsTailCountersAndFoldsOverlay) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/wedge_tail"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + const size_t tail_after_setup = store->tailSinceSnapshotCountForTest(ns); + + /// Wedge the lane: the single-attempt budget turns the ambiguous log PUT into an Unresolved outcome. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "not applied while merely wedged"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_setup) + << "an UNRESOLVED wedge applied nothing and must not join the tail counters"; + + /// The wedged PUT actually landed server-side; a later mutation resolves the wedge (applying drop x) + /// before committing its own drop y. + backend->materializePendingDelayedWrite(); + EXPECT_NO_THROW(store->dropRef(ns, "y")); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the wedged drop was applied on resolution"; + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_setup + 2) + << "the RESOLVED wedge (drop x) and the ordinary commit (drop y) must each bump the tail once"; + EXPECT_EQ(store->committedOverlayEntriesForTest(ns), 0u) + << "both the wedge resolution and the ordinary commit fold their overlay in place at install"; +} + +/// B3: `Pool::wedgedRefLaneCount()` (the accessor `CasGcScheduler::gcHealth()` reads for +/// `system.cas_mounts.wedged_namespace_count`) must count EXACTLY the tables with a live +/// wedge -- neither a cached-but-healthy table nor an unrelated table's own successful mutation may move +/// it, and it must track the wedge's full lifecycle (0 -> 1 -> 0), not just a one-shot snapshot. +TEST(CASRefWriterAppendLane, WedgedRefLaneCountTracksExactlyTheWedgedTableThroughItsLifecycle) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns_a{"srv1/wedge_count_a"}; + const RootNamespace ns_b{"srv1/wedge_count_b"}; + publishEmptyPart(store, ns_a, "x"); + publishEmptyPart(store, ns_a, "y"); + publishEmptyPart(store, ns_b, "p"); + ASSERT_EQ(store->wedgedRefLaneCount(), 0u) << "both tables cached and healthy before the fault"; + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_a)) + "_log/"; + backend->fault_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_a, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns_a)); + EXPECT_EQ(store->wedgedRefLaneCount(), 1u); + + /// ns_b's own mutation succeeds and must not be swept into the count. + EXPECT_NO_THROW(store->dropRef(ns_b, "p")); + EXPECT_EQ(store->wedgedRefLaneCount(), 1u) << "an unrelated table's successful mutation must not move the count"; + + /// The earlier request eventually lands server-side; resolving ns_a's wedge on its next mutation + /// drops the count back to zero. + backend->materializePendingDelayedWrite(); + EXPECT_NO_THROW(store->dropRef(ns_a, "y")); + EXPECT_FALSE(store->refLaneWedgedForTest(ns_a)); + EXPECT_EQ(store->wedgedRefLaneCount(), 0u); +} + +/// =================================================================================== +/// I1: a CORRUPTED_DATA from the retry controller (resolve-before-reissue observed a DIFFERENT object at +/// the exact key) must be surfaced LOUDLY to the caller and never hang the table's append queue. The +/// unfixed code let the throw propagate through the leader loop with `leader_active` still true, so every +/// queued and future caller for that table blocked forever in `cv.wait`. +/// =================================================================================== + +/// Append-site CORRUPTED_DATA: the offending caller gets the error, the lane is NOT wedged (a proven +/// different-object conflict is conclusive, not uncertain), and no caller HANGS -- the queue's leader +/// bookkeeping is restored, proven by a bounded wait on both a same-table and an independent-table +/// append. +/// +/// The reaction is now the mount's, not the table's [review I5]: a foreign object at a key that +/// mount-lease exclusivity says is exclusively ours contradicts the exclusivity itself, so the append +/// site routes through `reportImpossibleInterference` exactly as the wedge-resolve site does -- fence +/// closed, remount scheduled. Before this task it failed closed and stayed closed, blocking the table +/// until somebody remounted by hand. So there are two separate scopes to keep straight, and this test +/// pins both: +/// the FENCE is mount-wide -- while it is closed EVERY lane is refused, including untouched ones; +/// the DAMAGE is per-namespace -- a real remount replaces both immutable runtimes, then recovery of +/// the damaged stream still refuses while the unrelated table commits normally. +TEST(CASRefWriterAppendLane, I1AppendCorruptionSurfacesAndFencesTheMountForRemount) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/i1_append"}; + const RootNamespace other{"srv1/i1_other"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, other, "z"); + + /// The next `_log` PUT for `ns` has a foreign different object land at its key; resolve-before-reissue + /// then observes the mismatch and raises CORRUPTED_DATA. + backend->corrupt_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->corrupt_count = 1; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a proven different-object conflict must not wedge the lane"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted); + EXPECT_FALSE(store->mayMutate()) << "the impossible-interference reaction must fence this mount closed"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) + << "the append site must schedule the remount that re-derives this table from the durable log"; + + /// Mount-wide while fenced -- and, crucially, PROMPT: a real cv hang would time out this wait, which + /// is the regression this test was written for. + auto fenced = std::async(std::launch::async, [&] { store->dropRef(other, "z"); }); + ASSERT_EQ(fenced.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "the independent-table append hung -- the queue's leader bookkeeping was not restored"; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { fenced.get(); }); + + /// Drive the scheduled production recovery boundary. A direct fence re-arm is intentionally NOT a + /// substitute anymore: immutable runtimes retain the generation that admitted them and cannot be + /// rebound to the new one. + const String mount_key = layout.mountKey("test"); + const auto mount = backend->get(mount_key); + ASSERT_TRUE(mount); + MountLease fenced_mount = decodeMountLease(mount->bytes); + fenced_mount.gc_fenced = true; + fenced_mount.seq += 1; + ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(fenced_mount), mount->token).outcome, + PutOutcome::Done); + ASSERT_TRUE(store->tryRemountOnce()); + + auto same = std::async(std::launch::async, [&] { store->dropRef(ns, "x"); }); + ASSERT_EQ(same.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "the same-table append hung -- the queue's leader bookkeeping was not restored"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { same.get(); }); + + auto indep = std::async(std::launch::async, [&] { store->dropRef(other, "z"); }); + ASSERT_EQ(indep.wait_for(std::chrono::seconds(10)), std::future_status::ready); + indep.get(); + EXPECT_FALSE(store->resolveRef(other, "z").has_value()) + << "an unrelated table's stream is independent and must be entirely unaffected by the damage"; +} + +/// Wedge-resolve-site foreign interference: a wedged lane whose key a foreign writer overwrote must +/// surface the anomaly to the triggering caller and fault the lane, without hanging. +/// rev.6 Task 11 (spec §anomaly-policy): under the mount-lease exclusivity model this is no longer a +/// possible protocol outcome (the wedged key is exclusively ours) -- it routes through +/// `reportImpossibleInterference`, which fences the mount and schedules a remount. +/// +/// It surfaces as `CORRUPTED_DATA`. It was `LOGICAL_ERROR` between rev.6 and the every-attempt rule, +/// and that had a cost this test used to carry: `LOGICAL_ERROR` ABORTS the process in debug/sanitizer +/// builds, so the whole test had to be release-only with a death-test twin standing in elsewhere. +/// Storage-controlled input must never be able to abort the server, so the arm now reports the +/// occupant for what it is -- corruption -- and one test covers every build. +TEST(CASRefWriterAppendLane, I1WedgeResolveCorruptionSurfacesAndFaultsLane) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/i1_wedge"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + /// Wedge the lane with an ambiguous PUT that never landed. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + /// A foreign writer lands a DIFFERENT object at the exact wedged key; the next append's wedge resolve + /// observes the mismatch and must raise `CORRUPTED_DATA` to that caller while faulting the lane. + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(wedged_key.empty()); + ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + + auto fut = std::async(std::launch::async, [&] + { + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); + }); + ASSERT_EQ(fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "the wedge-resolve anomaly hung the queue instead of surfacing to the caller"; + fut.get(); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) + << "foreign interference is a terminal lane verdict"; + + /// The queue's leader bookkeeping was restored, so a SUBSEQUENT same-table caller does not hang: it + /// observes the terminal state and returns promptly (a real cv hang would time out this bounded + /// wait). This is the leg the unfixed code left blocked forever. + auto fut2 = std::async(std::launch::async, [&] + { + try + { + store->dropRef(ns, "y"); + } + catch (...) // NOLINT(bugprone-empty-catch) + { + /// The anomaly is expected here; this future only verifies that the caller does not hang. + } + }); + ASSERT_EQ(fut2.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "a later same-table append hung -- the leader bookkeeping was not restored after the anomaly"; + fut2.get(); +} + +/// =================================================================================== +/// rev.6 Task 11: wedge hard contract + anomaly policy (spec §anomaly-policy) +/// =================================================================================== + +/// Foreign bytes at a wedge key (see `I1WedgeResolveCorruptionSurfacesAndFaultsLane` above for the +/// hang-freedom coverage) must ALSO trip the local write fence closed and audit a `ForeignInterference` +/// event -- the full anomaly-policy reaction, not just the throw. It runs in every build now that the +/// arm reports `CORRUPTED_DATA` instead of the process-aborting `LOGICAL_ERROR`; the death twin that +/// used to stand in for debug/sanitizer builds went with it. +TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/anomaly_wedge"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + store->setEventSink([&](const CasEvent & e) { seen.add(e); }); + + /// Wedge the lane with an ambiguous PUT that never landed. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_TRUE(store->mayMutate()) << "the fence must not be tripped yet -- only an ordinary Unresolved wedge so far"; + ASSERT_EQ(store->scheduleRemountCallCountForTest(), 0u) << "no remount must have been scheduled yet by the ordinary wedge alone"; + + /// Out-of-band, a foreign writer lands DIFFERENT bytes at the exact wedged key. + const String wedged_key = store->wedgedKeyForTest(ns); + ASSERT_FALSE(wedged_key.empty()); + ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + + /// The next append's wedge resolve observes the mismatch: CORRUPTED_DATA, the fence trips closed, + /// and a ForeignInterference event is audited. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) + << "foreign interference must fault the lane"; + EXPECT_FALSE(store->mayMutate()) << "the local write fence must trip closed on the anomaly"; + /// Positively pins that `reportImpossibleInterference` called `scheduleRemount` (not just + /// `tripMountLost`, which alone already accounts for `mayMutate() == false` above). Counted at + /// `scheduleRemount`'s own entry regardless of `background_watermark` -- see that accessor's + /// comment for why this test deliberately does NOT enable `background_watermark` to observe a real + /// automatic recovery: doing so was tried and makes the store's self-remount attempt race its own + /// still-live keeper for 30+ seconds per call (confirmed while building this test), which is not + /// something a fast unit test should be driving. + EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) + << "reportImpossibleInterference must have called scheduleRemount exactly once"; + + const std::vector observed = seen.snapshot(); + const auto has_event = std::any_of(observed.begin(), observed.end(), + [](const CasEvent & e) { return e.type == CasEventType::ForeignInterference; }); + EXPECT_TRUE(has_event) << "a ForeignInterference CasEvent must be audited"; +} + +/// An impossible non-`Ready` state at new-id allocation must refuse before minting an id, fault the +/// lane, and trigger the anomaly policy. The synthetic wedge is injected after the top-of-flush +/// resolver gate, so it represents an internal lifecycle contradiction rather than a normal wedge. +TEST(CASAnomalyPolicy, NonReadyAtNewIdAllocationFaultsAndFailsClosed) +{ + auto backend = std::make_shared(); + SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/wedge_contract"}; + publishEmptyPart(store, ns, "x"); + + store->setEventSink([&](const CasEvent & e) { seen.add(e); }); + + store->setRefPreCarveHookForTest([&] + { + store->forceWedgeForTest(ns, /*writer_epoch*/ 1, /*ref_sequence*/ 1, "bogus/_log/key", "bogus-bytes"); + }); + + /// Ground truth: no NEW `_log` object may appear -- the guard must refuse BEFORE any id is minted + /// or PUT attempted. + auto countLogObjects = [&] + { + size_t n = 0; + String cursor; + for (;;) + { + const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation && parsed->kind == RefObjectKind::Log) + ++n; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return n; + }; + const size_t log_objects_before = countLogObjects(); + + ASSERT_TRUE(store->mayMutate()) << "the fence must be armed BEFORE the wedge-contract violation, or the guard would trivially pass for the wrong reason"; + ASSERT_EQ(store->scheduleRemountCallCountForTest(), 0u) << "no remount must have been scheduled yet"; + + /// BACKLOG `{#lane-terminal-reported-as-retryable}`: `Faulted` is a TERMINAL lane state, the same + /// one every OTHER `Faulted` arm in `commitRefChunk` reports as `CORRUPTED_DATA` -- reporting it as + /// `NETWORK_ERROR`/retry-later would tell the caller a state the lane can never leave on its own is + /// transient. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + + EXPECT_EQ(countLogObjects(), log_objects_before) << "the release guard must refuse before allocating/PUTting a new _log object"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) + << "the invariant violation must have one explicit terminal state"; + EXPECT_FALSE(store->mayMutate()) << "the local write fence must trip closed on the wedge-contract violation"; + /// See the sibling test's comment on why this checks the call-count seam (never + /// `background_watermark` plus automatic recovery -- that combination makes the store's self-remount + /// race its own still-live keeper). + EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) + << "reportImpossibleInterference must have called scheduleRemount exactly once"; + + const std::vector observed = seen.snapshot(); + const auto has_event = std::any_of(observed.begin(), observed.end(), + [](const CasEvent & e) { return e.type == CasEventType::ForeignInterference; }); + EXPECT_TRUE(has_event) << "a ForeignInterference CasEvent must be audited"; +} + +/// A failed best-effort diagnostic dispatch must not replace the fail-closed exception raised by the +/// mutation that discovered the anomaly, even when preparing the diagnostic log fails too. +TEST(CASAnomalyPolicy, DiagnosticDispatchLoggingCannotReplaceFailClosedException) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.detached_dispatch_fault_for_test = DetachedDispatchFault::ThrowBeforeLaunch; + config.diagnostic_dispatch_error_hook_for_test + = [] { throw std::runtime_error("injected: preparing the diagnostic dispatch log failed"); }; + auto store = openPoolWithConfig(backend, std::move(config)); + const RootNamespace ns{"srv1/diagnostic_dispatch_log_failure"}; + publishEmptyPart(store, ns, "x"); + + store->setRefPreCarveHookForTest([&] + { + store->forceWedgeForTest(ns, /*writer_epoch*/ 1, /*ref_sequence*/ 1, "bogus/_log/key", "bogus-bytes"); + }); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); +} + +/// I3: a conditional write whose attempt classified Committed but whose FINAL post-write fence check +/// failed (the mount fence was lost after the write may have landed) is counted separately, not folded +/// into the generic Unresolved classifier (spec §Late Predecessor PUT best-effort diagnostic). +TEST(CASRequestControllerFenceLoss, I3PostWriteFenceLossIsCounted) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + CasRequestBudget budget; + budget.max_attempts = 3; + CasRequestController ctrl(backend, budget, [] { return static_cast(0); }); // fixed clock + + /// `fence_ok` holds for the pre-attempt check, then is lost by the post-write check. + int calls = 0; + auto fence_ok = [&calls] { return ++calls <= 1; }; + + const auto before = global_counters[ProfileEvents::CASConditionalWriteFenceLostPostWrite].load(); + const CasWriteOutcome outcome = ctrl.putIfAbsentControlled("k", "v", fence_ok); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved) << "a post-write fence loss must never be reported as Committed"; + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteFenceLostPostWrite].load(), before + 1); +} + +/// Task B (stageManifest rides the controller): a Committed return surfaces the committed +/// incarnation's token — from the attempt's own PutResult, and equally from a resolve that proves an +/// earlier ambiguous attempt landed — so audit emitters (`PartWriteTxn::stageManifest`'s `ManifestPut` +/// event) keep their token without a follow-up HEAD. +TEST(CASRequestController, CommittedSurfacesTokenFromPutAndFromResolve) +{ + auto backend = std::make_shared(); + CasRequestController ctrl(backend, CasRequestBudget{}, [] { return static_cast(0); }); + const auto fence_ok = [] { return true; }; + + Token direct_token; + ASSERT_EQ(ctrl.putIfAbsentControlled("k1", "v1", fence_ok, &direct_token), CasWriteOutcome::Committed); + EXPECT_EQ(direct_token, backend->head("k1").token) << "the direct-commit token is the PutResult's"; + + /// k2 already holds the IDENTICAL bytes (an earlier ambiguous attempt that landed): the attempt's + /// PreconditionFailed collapses to Unresolved and the resolve GET proves Committed — the token must + /// be the observed incarnation's, and no second incarnation is ever created. + const Token pre_existing = backend->putIfAbsent("k2", "v2").token; + Token resolved_token; + ASSERT_EQ(ctrl.putIfAbsentControlled("k2", "v2", fence_ok, &resolved_token), CasWriteOutcome::Committed); + EXPECT_EQ(resolved_token, pre_existing) << "the resolve-commit token is the observed incarnation's"; +} + +/// =================================================================================== +/// Task 13: whole-table ref-cache eviction (spec §Byte, Memory, And CPU Budget) +/// =================================================================================== + +/// A tiny cache budget forces WHOLE-TABLE eviction: publishing to several tables in turn keeps only the +/// most-recently-touched one resident, and an evicted table re-recovers its exact committed state on the +/// next touch (spec §Startup And Recovery: "Evicting the table drops the entire object; the next access +/// repeats recovery"). +TEST(CASRefTableCacheEviction, WholeTableEvictionUnderBudgetReRecovers) +{ + auto backend = std::make_shared(); + auto store = openPoolWithConfig(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .ref_table_cache_bytes = 1}); + const RootNamespace ns_a{"srv1/evict_a"}; + const RootNamespace ns_b{"srv1/evict_b"}; + const RootNamespace ns_c{"srv1/evict_c"}; + + publishEmptyPart(store, ns_a, "x"); + publishEmptyPart(store, ns_b, "y"); + publishEmptyPart(store, ns_c, "z"); + + /// A 1-byte budget is below one table's weight, so each new table evicts the prior idle ones: only + /// the last-touched table stays resident (the just-recovered table is never evicted). + EXPECT_EQ(store->refTablesCachedCountForTest(), 1u); + EXPECT_TRUE(store->refTableCachedForTest(ns_c)); + EXPECT_FALSE(store->refTableCachedForTest(ns_a)); + EXPECT_FALSE(store->refTableCachedForTest(ns_b)); + + /// The evicted table re-recovers its exact committed state on next touch. + const auto resolved = store->resolveRef(ns_a, "x"); + ASSERT_TRUE(resolved.has_value()) << "an evicted table must re-recover its committed ref"; + /// That touch, in turn, evicted the previously-resident table under the same budget. + EXPECT_TRUE(store->refTableCachedForTest(ns_a)); + EXPECT_FALSE(store->refTableCachedForTest(ns_c)); +} + +/// A zero budget disables eviction entirely: every touched table stays resident. +TEST(CASRefTableCacheEviction, ZeroBudgetDisablesEviction) +{ + auto backend = std::make_shared(); + auto store = openPoolWithConfig(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .ref_table_cache_bytes = 0}); + for (const String & n : {String("srv1/keep_a"), String("srv1/keep_b"), String("srv1/keep_c")}) + publishEmptyPart(store, RootNamespace{n}, "x"); + EXPECT_EQ(store->refTablesCachedCountForTest(), 3u); +} + +/// A table with a WEDGED append lane is never evicted, even when idle and over budget: its uncertain +/// in-flight PUT is not reconstructable from the durable objects (spec §Writer-Side Linearization), so +/// re-recovery must not be allowed to drop and re-materialize it (which could re-allocate an id). +TEST(CASRefTableCacheEviction, WedgedTableIsNeverEvicted) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPoolWithConfig(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .cas_request_budget = budget, .ref_table_cache_bytes = 1}); + const Layout & layout = store->layout(); + const RootNamespace ns_w{"srv1/wedged"}; + publishEmptyPart(store, ns_w, "x"); + + /// Wedge ns_w's append lane with one ambiguous (Unresolved) PUT that exhausts the single-attempt budget. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_w)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_w, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns_w)); + + /// Pressure the cache with other tables. ns_w is idle and over the 1-byte budget, but its wedged lane + /// makes it non-evictable, so its wedge state survives (a fresh runtime would report no wedge). + publishEmptyPart(store, RootNamespace{"srv1/other_a"}, "y"); + publishEmptyPart(store, RootNamespace{"srv1/other_b"}, "z"); + + EXPECT_TRUE(store->refTableCachedForTest(ns_w)) << "a wedged table must never be evicted"; + EXPECT_TRUE(store->refLaneWedgedForTest(ns_w)) << "and its wedge state survives"; +} + +/// =================================================================================== +/// Task 11: snapshot publication (spec §writer-snapshot-publication) +/// =================================================================================== + +/// The count threshold fires a background publish covering the whole retained tail; its bytes must +/// equal an INDEPENDENT oracle's replay of the same logs through the published id (cache-replay +/// equivalence), and the retained tail must be fully pruned afterward (spec: "Publication is +/// background and never blocks an append"). +TEST(CASRefWriterSnapshotPublish, ThresholdTriggerPublishesCacheReplayEquivalentBytes) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const RootNamespace ns{"srv1/threshold_publish"}; + + publishEmptyPart(store, ns, "a"); /// tail: 2 (birth+add, promote) + publishEmptyPart(store, ns, "b"); /// tail: 3, then 4 (4 > 3 -> dispatches ONE background publish) + + store->waitForSnapshotPublishSettleForTest(ns); + + const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(snap_id.has_value()) << "the threshold trigger must have published a snapshot"; + EXPECT_TRUE(store->newestPublishedSnapshotIdForTest(ns) == snap_id); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) << "a snapshot covering everything prunes the whole tail"; + + const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + ASSERT_TRUE(got.has_value()); + + /// The independent oracle: replay every `_log/` object directly, ignoring the snapshot entirely. + const RefTableState oracle = independentFullReplayForTest(*backend, layout, ns, snap_id); + const String expected_bytes = encodeRefTableSnapshot(snapshotOf(oracle, ns.string())); + EXPECT_EQ(openObject(FormatId::RefSnapshot, got->bytes), expected_bytes) + << "published snapshot bytes must equal replay(logs through X)"; +} + +/// A publisher owns the runtime it captured, not the logical name. If that exact life is deleted and +/// the name is reborn while snapshot bytes are still only local, the old attempt must become inert: in +/// particular it must not recreate the predecessor's `_snap` or `_ckpt` after the GC retired them. +TEST(CASRefWriterSnapshotPublish, CapturedPredecessorCannotPublishAfterSameNameRebirth) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/publisher-predecessor-rebirth"}; + + publishWithProductionBirth(store, ns, "predecessor"); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); + const auto predecessor_snapshot + = listGreatestLogIdForLifeForTest(*backend, layout, predecessor_life); + ASSERT_TRUE(predecessor_snapshot); + + Gc gc(store, UInt128{105}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + + std::mutex mutex; + std::condition_variable cv; + bool captured = false; + bool release = false; + store->setSnapshotAfterCaptureHookForTest([&] + { + std::unique_lock lock(mutex); + captured = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }); + + auto publisher = std::async(std::launch::async, [&]() -> bool + { + return store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); + }); + SCOPE_EXIT({ + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + }); + { + std::unique_lock lock(mutex); + ASSERT_TRUE(cv.wait_for(lock, std::chrono::seconds(10), [&] { return captured; })); + } + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); + EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); + const RefCatalog after_removal = CasRefCatalog::read(*backend, layout).catalog; + EXPECT_TRUE(std::none_of(after_removal.entries.begin(), after_removal.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; })); + + publishWithProductionBirth(store, ns, "successor"); + const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + EXPECT_NE(successor.incarnation, predecessor.incarnation); + const auto predecessor_ckpt_before_resume = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_before_resume) + << "the removal protocol leaves this checkpoint as janitor-owned predecessor debris"; + + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + ASSERT_EQ(publisher.wait_for(std::chrono::seconds(10)), std::future_status::ready); + EXPECT_FALSE(publisher.get()); + store->setSnapshotAfterCaptureHookForTest(nullptr); + + EXPECT_FALSE(backend->get(layout.refSnapshotKey(predecessor_life, *predecessor_snapshot))) + << "a stale publisher recreated the retired predecessor snapshot"; + const auto predecessor_ckpt_after_resume = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_after_resume); + EXPECT_EQ(predecessor_ckpt_after_resume->token, predecessor_ckpt_before_resume->token) + << "a stale publisher replaced the retired predecessor checkpoint"; + EXPECT_EQ(predecessor_ckpt_after_resume->bytes, predecessor_ckpt_before_resume->bytes) + << "a stale publisher changed the retired predecessor checkpoint"; + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); + EXPECT_TRUE(store->resolveRef(ns, "successor")); +} + +/// The runtime admission check belongs inside every retrying `_ckpt` CAS attempt, not merely before +/// calling the checkpoint helper. Retirement in the body-PUT/checkpoint gap leaves the already-written +/// snapshot as harmless debris but must not advance or recreate the predecessor checkpoint. +TEST(CASRefWriterSnapshotPublish, RetiredPredecessorCannotAdvanceCkptAfterSnapshotPut) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/publisher-predecessor-ckpt-race"}; + + publishWithProductionBirth(store, ns, "predecessor"); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); + const auto candidate_id = listGreatestLogIdForLifeForTest(*backend, layout, predecessor_life); + ASSERT_TRUE(candidate_id); + + Gc gc(store, UInt128{106}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + + std::mutex mutex; + std::condition_variable cv; + bool before_ckpt_cas = false; + bool release = false; + store->setSnapshotBeforeCkptCasHookForTest([&] + { + std::unique_lock lock(mutex); + before_ckpt_cas = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }); + + auto publisher = std::async(std::launch::async, [&]() -> bool + { + return store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); + }); + SCOPE_EXIT({ + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + }); + { + std::unique_lock lock(mutex); + ASSERT_TRUE(cv.wait_for(lock, std::chrono::seconds(10), [&] { return before_ckpt_cas; })); + } + EXPECT_TRUE(backend->get(layout.refSnapshotKey(predecessor_life, *candidate_id))) + << "the hook must run after the snapshot body PUT and immediately before `_ckpt` admission"; + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); + EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); + const RefCatalog after_removal = CasRefCatalog::read(*backend, layout).catalog; + EXPECT_TRUE(std::none_of(after_removal.entries.begin(), after_removal.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; })); + + publishWithProductionBirth(store, ns, "successor"); + const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + EXPECT_NE(successor.incarnation, predecessor.incarnation); + const auto predecessor_ckpt_before_resume = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_before_resume); + + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + ASSERT_EQ(publisher.wait_for(std::chrono::seconds(10)), std::future_status::ready); + EXPECT_FALSE(publisher.get()); + store->setSnapshotBeforeCkptCasHookForTest(nullptr); + + const auto predecessor_ckpt_after_resume = backend->get(layout.refCkptKey(predecessor_life)); + ASSERT_TRUE(predecessor_ckpt_after_resume); + EXPECT_EQ(predecessor_ckpt_after_resume->token, predecessor_ckpt_before_resume->token); + EXPECT_EQ(predecessor_ckpt_after_resume->bytes, predecessor_ckpt_before_resume->bytes); + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); + EXPECT_TRUE(store->resolveRef(ns, "successor")); +} + +/// A read that already owns the predecessor runtime does not consult the name slot again after a +/// same-name successor is published. Because removal applies the terminal state before retirement, a +/// reader paused immediately before its state lock resumes with `NotFound`, never successor data. +TEST(CASRefWriterRuntimeIdentity, CapturedReaderCannotRetargetSameNameSuccessor) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/captured-reader-rebirth"}; + + publishWithProductionBirth(store, ns, "shared"); + ASSERT_TRUE(store->resolveRef(ns, "shared")); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + Gc gc(store, UInt128{107}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + + std::mutex mutex; + std::condition_variable cv; + bool captured = false; + bool release = false; + store->setReadBeforeStateLockHookForTest([&] + { + std::unique_lock lock(mutex); + if (captured) + return; + captured = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }); + auto reader = std::async(std::launch::async, [&] + { + return store->resolveRef(ns, "shared"); + }); + SCOPE_EXIT({ + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + }); + { + std::unique_lock lock(mutex); + ASSERT_TRUE(cv.wait_for(lock, std::chrono::seconds(10), [&] { return captured; })); + } + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); + EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); + publishWithProductionBirth(store, ns, "shared"); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + EXPECT_NE(successor.incarnation, predecessor.incarnation); + const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); + const auto successor_ref = store->resolveRef(ns, "shared"); + ASSERT_TRUE(successor_ref); + + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + ASSERT_EQ(reader.wait_for(std::chrono::seconds(10)), std::future_status::ready); + EXPECT_FALSE(reader.get()) << "the captured predecessor reader retargeted through the name slot"; + store->setReadBeforeStateLockHookForTest(nullptr); + + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); + const auto successor_after = store->resolveRef(ns, "shared"); + ASSERT_TRUE(successor_after); + EXPECT_EQ(successor_after->manifest_id.ref, successor_ref->manifest_id.ref); +} + +/// Ordinary append admission also owns the runtime it captured. If removal and rebirth complete before +/// enqueue, the predecessor's closed lane returns retry-later; it cannot enqueue into or mutate the +/// successor even when the successor deliberately reuses the same logical ref name. +TEST(CASRefWriterRuntimeIdentity, CapturedAppendCannotEnqueueIntoSameNameSuccessor) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/captured-append-rebirth"}; + + publishWithProductionBirth(store, ns, "shared"); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + Gc gc(store, UInt128{108}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + + std::mutex mutex; + std::condition_variable cv; + bool captured = false; + bool release = false; + store->setAppendAfterRuntimeCaptureHookForTest([&] + { + std::unique_lock lock(mutex); + if (captured) + return; + captured = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }); + auto append = std::async(std::launch::async, [&] + { + store->dropRef(ns, "shared"); + }); + SCOPE_EXIT({ + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + }); + { + std::unique_lock lock(mutex); + ASSERT_TRUE(cv.wait_for(lock, std::chrono::seconds(10), [&] { return captured; })); + } + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); + EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); + publishWithProductionBirth(store, ns, "shared"); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + EXPECT_NE(successor.incarnation, predecessor.incarnation); + const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); + const auto successor_ref = store->resolveRef(ns, "shared"); + ASSERT_TRUE(successor_ref); + + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + ASSERT_EQ(append.wait_for(std::chrono::seconds(10)), std::future_status::ready); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { append.get(); }); + store->setAppendAfterRuntimeCaptureHookForTest(nullptr); + + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); + const auto successor_after = store->resolveRef(ns, "shared"); + ASSERT_TRUE(successor_after); + EXPECT_EQ(successor_after->manifest_id.ref, successor_ref->manifest_id.ref); +} + +/// Exact retirement is pointer/key scoped. A delayed notification for the predecessor may arrive after +/// its same-name successor is already attached; it must not erase or poison that successor slot. +TEST(CASRefWriterRuntimeIdentity, LatePredecessorInvalidationLeavesSuccessorAttached) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/late-predecessor-invalidation"}; + + publishWithProductionBirth(store, ns, "predecessor"); + const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const NamespaceLifeId predecessor_life + = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); + Gc gc(store, UInt128{109}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + store->dropNamespace(ns); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + ASSERT_TRUE(runRegularRoundReclaiming(gc).deferred); + + publishWithProductionBirth(store, ns, "successor"); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + ASSERT_NE(successor.incarnation, predecessor.incarnation); + const NamespaceLifeId successor_life + = NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation); + const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); + const auto successor_ref = store->resolveRef(ns, "successor"); + ASSERT_TRUE(successor_ref); + + store->invalidateRemovedCatalogLife(predecessor_life); + + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); + EXPECT_EQ(store->refTableLifeForTest(ns), successor_life); + const auto successor_after = store->resolveRef(ns, "successor"); + ASSERT_TRUE(successor_after); + EXPECT_EQ(successor_after->manifest_id.ref, successor_ref->manifest_id.ref); +} + +/// Task 13 (spec §implementation-impact): a threshold snapshot publish increments the writer-side +/// observability counters -- snapshot PUT bytes and the tail-logs-compacted count +/// (logs-per-table-after-snapshot). Before/after deltas prove both sites fire. +TEST(CASRefWriterSnapshotPublish, PublishIncrementsSnapshotCounters) +{ + using ProfileEvents::global_counters; + const auto bytes_before = global_counters[ProfileEvents::CASRefSnapshotPutBytes].load(); + const auto logs_before = global_counters[ProfileEvents::CASRefSnapshotTailLogs].load(); + + auto backend = std::make_shared(); + const Layout layout("p"); + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + const RootNamespace ns{"srv1/counter_publish"}; + + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); + store->waitForSnapshotPublishSettleForTest(ns); + + ASSERT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()) + << "the threshold trigger must have published a snapshot"; + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPutBytes].load(), bytes_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotTailLogs].load(), logs_before); +} + +/// A fresh mount that recovers a large PRE-EXISTING tail (left by a predecessor whose own thresholds +/// never fired) retains that tail as trigger debt. Recovery ends at a terminal epoch seal, which is not +/// snapshot-serializable; one ordinary successor makes the inherited over-threshold tail publishable. +/// The single successor alone is below the threshold, so the dispatch still proves the mount-time tail +/// was retained rather than forgotten during recovery. +TEST(CASRefWriterSnapshotPublish, MountTimeRecoveredLargeTailPublishesAfterOrdinarySuccessor) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/mount_time_publish"}; + + { + /// Predecessor: default (high) thresholds, so nothing publishes yet. 3 parts -> 6 tail entries. + auto predecessor = openPool(backend); + publishEmptyPart(predecessor, ns, "a"); + publishEmptyPart(predecessor, ns, "b"); + publishEmptyPart(predecessor, ns, "c"); + } /// mount released; the tail is durable but nothing has published it + + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto successor = openPoolWithConfig(backend, config); + + /// A mere read triggers recovery. The recovered tail is already above threshold, but its greatest + /// applied record is the terminal seal, so there is deliberately no snapshot candidate yet. + EXPECT_EQ(successor->listRefs(ns).size(), 3u); + successor->waitForSnapshotPublishSettleForTest(ns); + EXPECT_GT(successor->tailSinceSnapshotCountForTest(ns), config.snapshot_log_count_threshold) + << "the mount must retain the predecessor's large uncovered tail"; + EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()) + << "a terminal recovery seal is not snapshot-serializable"; + + /// One ordinary transaction above the seal reopens the candidate. It cannot cross the threshold + /// by itself; publication therefore depends on the recovered mount-time tail asserted above. + successor->dropRef(ns, "c"); + successor->waitForSnapshotPublishSettleForTest(ns); + + const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(snap_id.has_value()) + << "the ordinary successor must make the inherited mount-time tail publishable"; + EXPECT_EQ(successor->tailSinceSnapshotCountForTest(ns), 0u); + + const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + ASSERT_TRUE(got.has_value()); + const RefTableState oracle = independentFullReplayForTest(*backend, layout, ns, snap_id); + EXPECT_EQ(openObject(FormatId::RefSnapshot, got->bytes), encodeRefTableSnapshot(snapshotOf(oracle, ns.string()))); +} + +/// =================================================================================== +/// rev.6 Task 10 (spec §publish-from-live): the grace-window machinery +/// (`snapshot_min_log_age_ms`, the tail-replay-from-`snapshot_base_state` copy-once path, +/// `CasRefLatePredecessorObserved`) is DELETED. The Task 8 recovery-seal plus the Task 6 +/// recovery seal already makes a late-arriving predecessor write born-covered for every +/// observer by the time this writer could ever see it, so a young committed txn has nothing left to +/// wait out -- it is immediately publish-eligible, with no time manipulation anywhere below. +/// =================================================================================== + +/// A just-committed txn is covered by a publish forced immediately afterward -- no fake clock, no +/// aging, no waiting: the OLD grace-window code would have published nothing here at all. +TEST(CASRefWriterPublishFromLive, YoungTxnIsCoveredImmediately) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/publish_from_live_young"}; + auto store = openPool(backend); + + /// Setup: birth the namespace and add a precommit (not the txn under test). + auto build = startBuildFor(store, ns, "a"); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, "a", id); + + /// The ONE committed txn under test. + build->promote(ns, "a", build->buildId(), id); + + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) + << "publish-from-live: a just-committed txn is immediately coverable, with no grace window"; + const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(snap_id.has_value()); + const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + ASSERT_TRUE(got.has_value()); + const RefTableSnapshot snap = decodeRefTableSnapshot(openObject(FormatId::RefSnapshot, got->bytes), ns.string(), *snap_id); + ASSERT_EQ(snap.committed.size(), 1u); + EXPECT_EQ(snap.committed.front().ref_name, "a") + << "the published snapshot body contains the just-promoted row"; +} + +/// The count trigger fires purely off the tail counters -- no aging involved -- even under a boot +/// clock that never advances (the old code REQUIRED aging past `snapshot_min_log_age_ms` to fire). +TEST(CASRefWriterSnapshotPublish, TriggerFiresOnCountAboveThresholdWithoutAging) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/publish_from_live_trigger"}; + uint64_t fake_now = 1'000'000; /// frozen: never advances + + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + auto store = openPoolWithConfig(backend, config); + + const auto before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + publishEmptyPart(store, ns, "a"); /// tail: 2 + publishEmptyPart(store, ns, "b"); /// tail: 4 > 3 -> dispatches, clock frozen throughout + store->waitForSnapshotPublishSettleForTest(ns); + + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), before) + << "the count trigger must fire without any aging, even under a frozen clock"; +} + +/// Adoption subtracts EXACTLY the counters captured at copy time, not whatever the counters read at +/// adoption time: while a publish's PUT is in flight (captured count/bytes fixed), more commits land +/// on the live counters. After adoption, the counters must equal precisely the amount appended AFTER +/// the copy -- not zero (would drop the new txns from the next publish trigger) and not negative/ +/// wrapped (an unsigned underflow). +TEST(CASRefWriterSnapshotPublish, AdoptionSubtractsCapturedCountersUnderConcurrentAppends) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/publish_from_live_adoption"}; + auto store = openPool(backend); + + publishEmptyPart(store, ns, "a"); + ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u); + + backend->armPutBlock("_snap/"); + std::thread publisher([&] { store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); }); + backend->awaitBlockEntered(); /// the candidate (count=2) is captured; the PUT is now in flight, no lock held + + publishEmptyPart(store, ns, "b"); /// +2 more commits land WHILE the publish's PUT is in flight + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 4u); + + backend->releaseBlock(); + publisher.join(); + + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u) + << "adoption must subtract only the CAPTURED count (2), leaving exactly the 2 txns appended " + "after the copy"; +} + +/// Publication must never block a concurrent append on the SAME table (spec: "Publication is +/// background and never blocks an append"): while a dispatched background publish is stuck mid-PUT, an +/// ordinary mutation on the table must still complete promptly (a real deadlock would hang this test). +TEST(CASRefWriterSnapshotPublish, PublicationNeverBlocksConcurrentAppend) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/publish_no_block"}; + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + backend->armPutBlock("_snap/"); + + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); /// tail reaches 4 (> 3) -> dispatches a background publish + + backend->awaitBlockEntered(); /// the dispatched attempt is now stuck mid-PUT on the snapshot key + + /// An unrelated mutation on the SAME table must complete without waiting for the stuck publish. + EXPECT_NO_THROW(store->dropRef(ns, "a")); + EXPECT_FALSE(store->resolveRef(ns, "a").has_value()); + + backend->releaseBlock(); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); +} + +/// Review caution (T10 review): a dispatched background publish must never outlive the Pool object +/// it operates on -- `maybeScheduleSnapshotPublish` captures `shared_from_this()` BY VALUE into the +/// dispatch lambda specifically to guarantee this (the classic "background thread references a +/// dangling owner" shutdown segfault, avoided here since a shared_ptr copy keeps the object alive for +/// as long as the thread holds it, regardless of what every OTHER holder does). Proves it directly: +/// blocks a dispatched publish mid-PUT, drops the TEST's own (only) Pool handle while still blocked, +/// and confirms via a `weak_ptr` that the Pool demonstrably survives on the blocked thread's own +/// reference alone. Then unblocks it with no live Pool handle anywhere in this test any more -- a +/// dangling-pointer crash here would abort the whole test binary, the strongest possible signal for +/// this specific hazard. +TEST(CASRefWriterSnapshotPublish, PublishThreadOutlivesDroppedPoolHandleWithoutCrashing) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/publish_outlives_store"}; + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + backend->armPutBlock("_snap/"); + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); /// tail reaches 4 (> 3) -> dispatches a background publish + backend->awaitBlockEntered(); /// stuck mid-PUT, holding its OWN shared_ptr copy + + std::weak_ptr weak_store = store; + store.reset(); /// drop the ONLY Pool handle this test holds + EXPECT_FALSE(weak_store.expired()) + << "the blocked background thread's own shared_ptr copy must keep the Pool alive"; + + backend->releaseBlock(); + /// Deterministic, sleep-free: waits for the blocked call to actually RETURN (not merely unblock), + /// entirely through the backend -- this test holds no Pool handle to wait on any more. + backend->awaitBlockedCallCompleted(); +} + +/// Review (T11) — CRITICAL: publishes are NOT serialized, so two overlapping attempts can finish out of +/// order. An OLDER-candidate publish that lands its `_snap` PUT AFTER a newer one already adopted must +/// NOT regress `newest_snapshot_id` back to its older id, and its (monotonically-skipped) adoption must +/// NOT touch the tail counters a newer attempt already reset -- either would drop the txns committed in +/// between, so the NEXT published snapshot would silently omit committed transactions and recovery +/// would lose refs. Deterministic, sleep-free: the fake backend blocks publish #1's PUT (capturing +/// exactly its key) while a higher-id publish #2 runs to completion, then unblocks #1. +TEST(CASRefWriterSnapshotPublish, ConcurrentOutOfOrderPublishDoesNotRegressBaseNorDropCommittedTxns) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/concurrent_publish_monotonic"}; + PoolConfig config; + /// High thresholds: NO automatic background dispatch -- we drive + /// `tryPublishSnapshotAndAdvanceCheckpointOnce` directly for full determinism. + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + publishEmptyPart(store, ns, "a"); /// tail: 2 (birth+add, promote) + publishEmptyPart(store, ns, "b"); /// tail: 4 -- greatest_applied is publish #1's candidate + + /// Block ONLY publish #1's own `_snap` PUT (its exact key is captured on first match); a later, + /// different `_snap/` key proceeds unblocked. + backend->armPutBlockFirstMatchOnly("_snap/"); + + std::thread publisher1([&] { store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); }); + backend->awaitBlockEntered(); /// #1 is parked mid-PUT on `_snap/`, holding no lock + + /// While #1 is parked, commit more txns and run publish #2 to COMPLETION: it PUTs a strictly higher + /// `_snap/` (unblocked) and adopts it, resetting the tail counters through its own candidate. + publishEmptyPart(store, ns, "c"); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + const auto newest_after_2 = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(newest_after_2.has_value()); + ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) << "publish #2 covers everything committed so far"; + + /// Release #1: on the BUGGY code it now adopts its OLDER candidate, regressing newest below #2 + /// and/or double-subtracting from the counters #2 already reset. The monotonic guard must skip + /// that adoption entirely -- both the `newest_snapshot_id` write and the counter subtraction. + backend->releaseBlock(); + publisher1.join(); + + const auto newest_after_1 = store->newestPublishedSnapshotIdForTest(ns); + ASSERT_TRUE(newest_after_1.has_value()); + EXPECT_FALSE(*newest_after_1 < *newest_after_2) + << "a late-finishing OLDER publish must not regress newest_snapshot_id below the adopted newer one"; + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) + << "publish #1's skipped (monotonically-superseded) adoption must not subtract from counters " + "publish #2 already reset -- an unguarded subtraction here would corrupt or underflow them"; + + /// Independent proof no committed txn was lost: the NEXT publish's bytes must equal a full log replay. + /// A regressed base would omit the txns committed while publish #1 was parked. + publishEmptyPart(store, ns, "d"); + ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(snap_id.has_value()); + const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + ASSERT_TRUE(got.has_value()); + const RefTableState oracle = independentFullReplayForTest(*backend, layout, ns, snap_id); + EXPECT_EQ(openObject(FormatId::RefSnapshot, got->bytes), encodeRefTableSnapshot(snapshotOf(oracle, ns.string()))) + << "published snapshot bytes must equal replay(all logs through X) -- a regressed base drops txns"; +} + +/// (I1, review of commit 9093482176a) `clampedCounterSub`'s actual clamp-to-zero branch -- the exact +/// hazard it exists for -- was previously unpinned: `AdoptionSubtractsCapturedCountersUnderConcurrentAppends` +/// subtracts from a counter that never goes below the captured amount (no clamp needed), and in +/// `ConcurrentOutOfOrderPublish...` above the SMALLER candidate is the one parked, so its adoption is +/// skipped entirely by the T11 monotonic guard BEFORE it would ever reach the subtraction -- the clamp +/// is never exercised either way. This test forces the one ordering the guard does NOT catch: the +/// SMALLER candidate adopts (and subtracts) FIRST, then a LARGER candidate -- captured earlier, while +/// the counter still held the region the smaller one just subtracted -- adopts second. Its captured +/// count therefore double-counts that already-subtracted region, and `clampedCounterSub` must clamp +/// to zero rather than wrap a `uint64_t` to ~`UINT64_MAX` (which would permanently re-latch the C4 +/// storm trigger in a release build -- no `chassert` to catch it). Deterministic, sleep-free: two +/// `_snap` PUTs are parked independently (both past their own capture, neither yet adopted) via +/// `armPutBlockIndependently`, then released in the specific order that reproduces the hazard. +TEST(CASRefWriterSnapshotPublish, ClampedCounterSubClampsInsteadOfUnderflowingOnOutOfOrderAdoption) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/clamp_out_of_order"}; + PoolConfig config; + /// High thresholds: NO automatic background dispatch -- we drive + /// `tryPublishSnapshotAndAdvanceCheckpointOnce` directly for full determinism. + config.snapshot_log_count_threshold = 1ULL << 40; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + publishEmptyPart(store, ns, "a"); /// tail: 2 -- publisher A's (smaller) candidate + + backend->armPutBlockIndependently("_snap/"); + + std::thread publisher_a([&] { store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); }); + backend->awaitAtLeastNKeysBlocked(1); /// A has captured (candidate=2 txns, count=2) and parked mid-PUT + const String key_a = *backend->blockedKeysSnapshot().begin(); + + publishEmptyPart(store, ns, "b"); /// tail: 4 -- publisher B's (larger) candidate, captured BELOW + + std::thread publisher_b([&] { store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); }); + backend->awaitAtLeastNKeysBlocked(2); /// B has ALSO captured (candidate=4 txns, count=4) and parked + const auto blocked = backend->blockedKeysSnapshot(); + ASSERT_EQ(blocked.size(), 2u) << "both publishers must be parked past their own capture before either adopts"; + String key_b; + for (const auto & k : blocked) + if (k != key_a) + key_b = k; + ASSERT_FALSE(key_b.empty()); + + /// Release the SMALLER candidate first: its monotonic guard passes (newest is still unset), so it + /// adopts -- newest becomes A's candidate, and the count drops from the live 4 to 2 (4 - captured_A=2). + backend->releaseKey(key_a); + publisher_a.join(); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u) + << "publisher A (smaller candidate) adopts first and subtracts its own captured count safely"; + + /// Release the LARGER candidate: its monotonic guard ALSO passes (newest=A's candidate < B's + /// candidate), so it reaches the subtraction with `captured_count_B == 4` -- but the live counter + /// is now only 2 (A's adoption already removed the overlapping region). A plain `fetch_sub` here + /// would wrap to ~UINT64_MAX; `clampedCounterSub` must clamp to 0 instead. + backend->releaseKey(key_b); + publisher_b.join(); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) + << "clampedCounterSub must clamp to 0, not underflow/wrap, when B's captured count (4) already " + "includes the region A's earlier adoption already subtracted"; + + /// A wrapped counter would read as ~UINT64_MAX, permanently latching `over_threshold` (the C4 + /// storm regression). With the huge threshold configured above, a dispatch firing here can ONLY + /// mean the counter is corrupted -- a clamped counter of 0 never crosses it. + const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + for (int i = 0; i < 5; ++i) + store->resolveRef(ns, "a"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_before) + << "a correctly-clamped counter must never latch the threshold trigger"; +} + +/// =================================================================================== +/// C4: bound the read-triggered snapshot-publish dispatch (spec §writer-snapshot-publication). A +/// fold-heavy reader must not turn every ref read into a re-dispatched full-snapshot encode+PUT: an +/// in-flight gate admits at most one publish per table, and a non-Committed outcome arms a bounded +/// per-table backoff instead of re-triggering on the next read. The unfixed code dispatched a new +/// publish on every trigger and never backed off, producing the soak's 46 GB/hr `_snap` PUT storm. +/// =================================================================================== + +/// Under a saturated backend (every `_snap` PUT is Unresolved), the read path must NOT re-dispatch a +/// publish on each read: the failure arms the backoff, and while it holds no read re-dispatches. +TEST(CASRefWriterSnapshotPublish, C4LatchBoundedUnderSustainedNonCommittedPublish) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/c4_latch"}; + + CasRequestBudget budget; /// one attempt per publish so a failure is a single PUT + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + uint64_t fake_now = 1'000'000; + PoolConfig config; + config.snapshot_log_count_threshold = 1; + config.snapshot_log_bytes_threshold = 1ULL << 40; + config.snapshot_publish_backoff_initial_ms = 5000; /// the frozen clock keeps the backoff armed + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget = budget; + auto store = openPoolWithConfig(backend, config); + + /// Every `_snap` PUT throws Unresolved (backend saturated), from the very first publish attempt. + backend->fault_key_substr = "_snap/"; + backend->fault_count = 100000; + + publishEmptyPart(store, ns, "a"); /// crosses the threshold -> one dispatch -> fails -> backoff armed + store->waitForSnapshotPublishSettleForTest(ns); + + const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + for (int i = 0; i < 30; ++i) + { + store->resolveRef(ns, "a"); + store->waitForSnapshotPublishSettleForTest(ns); + } + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_before) + << "reads within the backoff window must not re-dispatch a publish (the storm latch is broken)"; +} + +/// Recovery can leave the runtime at an epoch seal with a tail already above the threshold. The seal +/// is not snapshot-serializable, so admission itself must reject it: letting execution reject it would +/// make settlement immediately dispatch another identical background attempt. A later ordinary record +/// must re-enable the same scheduler. +TEST(CASRefWriterSnapshotPublish, RecoveredSealAboveThresholdDoesNotRedispatchUntilOrdinarySuccessor) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/recovered_seal_no_storm"}; + + { + PoolConfig predecessor_config; + predecessor_config.snapshot_log_count_threshold = 1ULL << 40; + predecessor_config.snapshot_log_bytes_threshold = 1ULL << 40; + auto predecessor = openPoolWithConfig(backend, predecessor_config); + DB::Cas::tests::fixture::admitLive(*backend, predecessor->layout(), ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, predecessor->layout(), ns); + ASSERT_EQ(backend->putIfAbsent(predecessor->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + .life_epoch = predecessor->liveWriterEpoch(), + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + publishEmptyPart(predecessor, ns, "before_seal"); + const auto before = backend->get(predecessor->layout().refCkptKey(life)); + ASSERT_TRUE(before); + const RefCkpt before_seal = decodeRefCkpt(before->bytes); + ASSERT_TRUE(before_seal.committed_through); + const RefTxnId seal_id{before_seal.committed_through->writer_epoch, + before_seal.committed_through->ref_sequence + 1}; + writeSealAt(*backend, predecessor->layout(), ns, seal_id); + + RefCkpt recovered_seal = before_seal; + recovered_seal.committed_through = seal_id; + recovered_seal.last_epoch_seal = seal_id; + ASSERT_EQ(backend->casPut(predecessor->layout().refCkptKey(life), encodeRefCkpt(recovered_seal), before->token).outcome, + CasOutcome::Committed); + } + + PoolConfig successor_config; + successor_config.snapshot_log_count_threshold = 0; + successor_config.snapshot_log_bytes_threshold = 1ULL << 40; + auto successor = openPoolWithConfig(backend, successor_config); + + EXPECT_TRUE(successor->resolveRef(ns, "before_seal").has_value()); + successor->waitForSnapshotPublishSettleForTest(ns); + const auto dispatched_at_seal = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + for (int i = 0; i < 5; ++i) + EXPECT_TRUE(successor->resolveRef(ns, "before_seal").has_value()); + successor->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_at_seal) + << "a recovered seal must not dispatch or re-dispatch an unpublishable snapshot candidate"; + + /// One ordinary append transaction above the recovered seal must reopen the scheduler. `dropRef` + /// is exactly one ordinary ref-log append, unlike `publishEmptyPart`'s two-phase part publication. + successor->dropRef(ns, "before_seal"); + successor->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_at_seal + 1) + << "an ordinary successor above the seal must make the threshold candidate publishable again"; +} + +/// While one background publish is in flight (blocked mid-PUT), further reads must NOT dispatch a +/// second: the single-in-flight gate holds `pending_snapshot_publishes` at one per table. +TEST(CASRefWriterSnapshotPublish, C4InFlightGateAdmitsAtMostOne) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/c4_gate"}; + PoolConfig config; + config.snapshot_log_count_threshold = 0; /// any nonempty tail triggers + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + publishEmptyPart(store, ns, "a"); + store->waitForSnapshotPublishSettleForTest(ns); /// drain the setup publishes; tail is compacted + + /// Block the first `_snap` PUT so one publisher parks in flight. + backend->armPutBlockFirstMatchOnly("_snap/"); + std::thread mutator([&] { store->dropRef(ns, "a"); }); /// its detached publisher blocks mid-PUT + backend->awaitBlockEntered(); + + /// Many more reads while it is blocked must not admit a second publisher. + for (int i = 0; i < 20; ++i) + store->resolveRef(ns, "a"); + EXPECT_EQ(store->pendingSnapshotPublishesForTest(ns), 1) + << "the in-flight gate must hold background publishes to at most one per table"; + + backend->releaseBlock(); + mutator.join(); + store->waitForSnapshotPublishSettleForTest(ns); +} + +/// A non-Committed publish defers the next dispatch by the backoff, then a read past the backoff +/// deadline dispatches exactly one retry that publishes a durable snapshot (freshness preserved). +TEST(CASRefWriterSnapshotPublish, C4BackoffDefersThenRetriesAndPublishes) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/c4_backoff"}; + + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + uint64_t fake_now = 1'000'000; + PoolConfig config; + config.snapshot_log_count_threshold = 1; + config.snapshot_log_bytes_threshold = 1ULL << 40; + config.snapshot_publish_backoff_initial_ms = 1000; + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.cas_request_budget = budget; + auto store = openPoolWithConfig(backend, config); + + /// Fail ONLY the first `_snap` PUT (arms the backoff); later PUTs succeed. + backend->fault_key_substr = "_snap/"; + backend->fault_count = 1; + + publishEmptyPart(store, ns, "a"); /// dispatch -> publish fails -> backoff armed + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); + + /// A read within the backoff window (frozen clock) must not re-dispatch. + const auto d1 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + store->resolveRef(ns, "a"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d1) + << "a read within the backoff window must not re-dispatch"; + EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); + + /// Advance past the backoff: exactly one retry is dispatched and it publishes. + fake_now += 2000; + store->resolveRef(ns, "a"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d1 + 1) + << "after the backoff elapses exactly one retry is dispatched"; + EXPECT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()) + << "the retry publishes a durable snapshot (freshness preserved)"; +} + +/// =================================================================================== +/// rev.6 Task 10 (spec §publish-from-live): the tail counters count ONLY applied txns strictly above +/// `newest_snapshot_id` -- incremented per commit, subtracted (clamped) exactly by adoption. This +/// pins that a successful publish's adoption RESETS the counters rather than merely reducing them: a +/// buggy "subtract a fixed prune count" scheme could let the table's already-covered history keep +/// contributing to the trigger forever. +/// =================================================================================== + +/// After a successful publish adopts `newest_snapshot_id`, the trigger arithmetic must restart from +/// zero above it, not keep counting the table's already-covered history: 4 covered + 2 fresh entries +/// must read as 2 (below a 3 threshold), never as 6. +TEST(CASRefWriterSnapshotPublish, TriggerIgnoresEntriesCoveredByNewestSnapshot) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/trigger_covered"}; + PoolConfig config; + config.snapshot_log_count_threshold = 3; + config.snapshot_log_bytes_threshold = 1ULL << 40; + auto store = openPoolWithConfig(backend, config); + + const auto d0 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + + /// Drive ONE successful publish: 4 entries (4 > 3). + publishEmptyPart(store, ns, "a"); + publishEmptyPart(store, ns, "b"); + store->waitForSnapshotPublishSettleForTest(ns); + ASSERT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 1); + const auto first_snap = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(first_snap.has_value()); + EXPECT_TRUE(store->newestPublishedSnapshotIdForTest(ns) == first_snap); + ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u); + + /// 2 fresh entries: 2 <= 3, while the covered history (4 entries at/below the snapshot) would push + /// a covered-counting trigger to 6 > 3. Must not dispatch. + publishEmptyPart(store, ns, "c"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 1) + << "entries covered by the newest snapshot must not count toward the trigger"; + EXPECT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns) == first_snap); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u); + + /// Crossing the threshold with the fresh tail alone (4 > 3) dispatches exactly one more publish, + /// and it covers the whole uncovered tail. + publishEmptyPart(store, ns, "d"); + store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 2); + const auto second_snap = listGreatestSnapshotIdForTest(*backend, layout, ns); + ASSERT_TRUE(second_snap.has_value()); + EXPECT_TRUE(*first_snap < *second_snap); + EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u); +} + +/// =================================================================================== +/// Task 11: successor stale-precommit cleanup (spec §Clean Up Old Precommits) +/// =================================================================================== + +/// A predecessor's dangling (never-promoted) precommits are swept by the successor mount's first touch +/// of the table; a precommit the SUCCESSOR itself adds under its OWN (current) epoch must survive. +TEST(CASRefWriterStalePrecommitSweep, SweepsOnlyStaleEpochPrecommitsKeepsCurrentEpoch) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/precommit_sweep_basic"}; + + { + /// A predecessor writer leaves THREE precommits dangling (a crash before promote). + auto predecessor = openPool(backend); + for (const String & name : {"stale_a", "stale_b", "stale_c"}) + { + auto build = startBuildFor(predecessor, ns, name); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, name, id); + /// no promote -- left dangling, as a crashed build would leave it + } + } /// predecessor destroyed: its mount lease is released + + /// The successor allocates a strictly higher durable writer_epoch; its own FRESH precommit must + /// survive the sweep its very first touch of the table triggers. + auto successor = openPool(backend); + auto build = startBuildFor(successor, ns, "fresh_x"); + const ManifestId fresh_id = build->stageManifest({}); + build->precommitAdd(ns, "fresh_x", fresh_id); /// this call's own appendRefOps hoists the sweep first + + const RefTableState replayed = independentFullReplayForTest(*backend, successor->layout(), ns); + EXPECT_EQ(replayed.getLifecycle(), RefLifecycle::Live); + EXPECT_TRUE(replayed.getCommitted().empty()); + ASSERT_EQ(replayed.getPrecommits().size(), 1u); + EXPECT_EQ(replayed.getPrecommits().begin()->first, "fresh_x"); + EXPECT_EQ(replayed.getPrecommits().begin()->second, fresh_id.ref); +} + +/// The sweep chunks its removal to `ref_txn_max_ops` stale precommits per transaction (spec +/// §Clean Up Old Precommits), and an interruption (an uncertain PUT, wedging the lane) leaves the +/// remainder harmlessly for a LATER mount's own fresh recovery to finish -- "each chunk re-reads the +/// LIVE state, so a partial sweep just leaves fewer stale bindings for the next chunk (a later retry +/// on this mount, or the next mount's recovery) to find." (Same-mount retry is pinned separately by +/// `FailedSweepRearmsAndRetriesUntilClean`.) +TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossMounts) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/precommit_sweep_bounded"}; + /// Derived from `ref_txn_max_ops` (not a literal) so a future cap change cannot silently drop this + /// back to a single removal chunk: still > the cap, forcing at least two removal chunks. + constexpr int kTotalStale = static_cast(ref_txn_max_ops) + 200; + + uint64_t e1 = 0; + { + auto predecessor = openPool(backend); + e1 = predecessor->writerEpoch(); + } /// predecessor released; only its epoch is needed -- the stale precommits are seeded raw below + + /// Seed kTotalStale precommits directly (bypassing any Pool) under the predecessor's epoch, + /// spread over two raw log objects (each within the per-transaction op ENCODE cap) so recovery + /// costs only two GETs, not kTotalStale of them. + { + std::vector ops1; + ops1.push_back(namespaceBirthOp()); + for (int i = 0; i < 700; ++i) + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "stale_" + std::to_string(i), manifestRef(e1, static_cast(i) + 1, 1)}; + ops1.push_back(op); + } + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{e1, 1}, ops1, std::nullopt}); + + std::vector ops2; + for (int i = 700; i < kTotalStale; ++i) + { + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "stale_" + std::to_string(i), manifestRef(e1, static_cast(i) + 1, 1)}; + ops2.push_back(op); + } + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{e1, 2}, ops2, std::nullopt}); + } + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = e1, + .committed_through = RefTxnId{e1, 2}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + /// The successor: a tight retry budget so ONE simulated ambiguous response wedges rather than + /// transparently retries away. `8f9e63c7a19` widened `kSingleAttemptDeadlineMs` off a zero-width + /// race (equal attempt/operation deadlines), but it still measures the capture-to-gate window -- + /// encoding the removal chunk (up to `ref_txn_max_ops` ops) -- against the REAL wall clock, so it + /// recurred (3 of 3 sanitizer lanes) once that encode step got slow enough on its own, independent + /// of scheduler contention: msan in particular. `ref_request_controller` reads its clock through + /// the same injectable seam as the mount fence (`CasRefLedger`'s `controller_boot_ms_fn` is the + /// pool's `boot_ms_fn`), so freeze it here instead of racing it -- the fault-injecting PUT below + /// still reaches the backend synchronously; only the deadline arithmetic stops moving. + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + PoolConfig config; + config.cas_request_budget = budget; + config.boot_ms_fn = [] { return uint64_t{0}; }; + auto successor = openPoolWithConfig(backend, config); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + /// The successor's own recovery runs first and mints one in-band seal for the dead predecessor + /// epoch `e1` (its durable ids are `{e1,1}` and `{e1,2}`, so the seal lands at `{e1,3}`) -- that PUT + /// shares this same `_log/` prefix, so it would eat the fault before the sweep ever gets a chance. + /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. + backend->fault_skip = 1; + backend->fault_count = 1; /// hits exactly the sweep's FIRST removal chunk's PUT + + /// The sweep is piggybacked on this mount's very first touch; its (uncertain) failure is INSULATED + /// from the read (resolveRef/listRefs call `sweepStalePrecommitsForRead`, not + /// `maybeSweepStalePrecommits` directly): the read itself still succeeds, the failure is counted. + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); + EXPECT_NO_THROW(successor->listRefs(ns)); + const uint64_t deferred_after = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); + EXPECT_EQ(deferred_after, deferred_before + 1) + << "the read-only caller must observe (and count) the deferred sweep failure, not throw"; + EXPECT_TRUE(successor->refLaneWedgedForTest(ns)); + + /// The first chunk's request actually landed server-side; the caller just never saw the ack. + backend->materializePendingDelayedWrite(); + successor.reset(); /// abandoned mid-sweep WITHOUT ever resolving its own wedge in-memory + + /// A THIRD mount (successor-of-the-successor): fresh recovery replays the two raw seed logs PLUS the + /// first chunk's now-durable removal, sees `needs_stale_precommit_sweep` armed again, and finishes + /// the remaining stale precommits in exactly one further chunk (<= 1000 remain). `successor` was + /// abandoned mid-wedge above -- Task 5's drain fails closed on an unresolved PUT, so no clean + /// farewell was written -> this reclaim is `MountPriorState::UncleanObserved` (rev.6 Task 4), which + /// pays a real ~36.5s token-stability observation wait here. Inject a fake `boot_ms_fn` + + /// `wait_sleep_fn` (mirroring `CASMountOpenWaits.UncleanOpenPaysOnlyTheObservationWindow`) so it + /// resolves instantly. + uint64_t resumer_fake_boot = 0; + PoolConfig resumer_config; + resumer_config.boot_ms_fn = [&resumer_fake_boot] { return resumer_fake_boot; }; + resumer_config.wait_sleep_fn = [&resumer_fake_boot](uint64_t ms) { resumer_fake_boot += ms; }; + auto resumer = openPoolWithConfig(backend, resumer_config); + EXPECT_NO_THROW(resumer->listRefs(ns)); + + const RefTableState final_state = independentFullReplayForTest(*backend, layout, ns); + EXPECT_EQ(final_state.getLifecycle(), RefLifecycle::Live); + EXPECT_TRUE(final_state.getPrecommits().empty()) << "every stale precommit must eventually be swept"; + + /// Bounded batches: exactly THREE NEW `_log/` objects (epoch > e1) were needed -- never kTotalStale + /// individual removals, and one more than before INV-2 went in-band. In order: `{e1+1,1}` the + /// successor's own (delayed-delivered) FIRST removal chunk; `{e1+1,2}` the epoch seal that closes + /// the successor's own epoch once IT becomes dead in turn -- minted by `resumer`'s recovery, since + /// `successor` was abandoned mid-sweep without a clean farewell and never sealed itself; and + /// `{e1+2,1}` the resumer's own SECOND removal chunk, finishing the remaining stale precommits. + size_t new_log_objects = 0; + { + String cursor; + for (;;) + { + const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation + && parsed->kind == RefObjectKind::Log && parsed->txn_id.writer_epoch != e1) + ++new_log_objects; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + } + EXPECT_EQ(new_log_objects, 3u); +} + +/// S13 regression fix (triage `.superpowers/sdd/s13-triage-report.md`, run 20260713T172032_S13_seed42): +/// a FAILED sweep attempt must NOT consume the once-per-mount shot. The failure re-arms +/// `needs_stale_precommit_sweep` (with a bounded backoff, so a saturated backend is not stormed), the +/// read that piggybacked the sweep still succeeds (existing `CASRefSweepDeferred` contract), and a later +/// trigger -- here a mutation -- retries until a pass completes verified clean, clearing the flag +/// permanently. Each reclaimed binding is audited: one `precommit_reclaim` CA-log event + one +/// `CASRefStalePrecommitsReclaimed` increment, exactly per binding. +TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/precommit_sweep_retry"}; + + /// One shared injected clock for both incarnations. The successor's wait hook below advances this + /// same clock, so both mount observation and the later sweep-backoff deadline are deterministic. + uint64_t fake_now = 1'000'000; + size_t mount_wait_calls = 0; + const auto fake_clock = [&fake_now] { return fake_now; }; + + { + /// A predecessor writer leaves THREE precommits dangling (a crash before promote). + PoolConfig pred_config; + pred_config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + pred_config.boot_ms_fn = fake_clock; + auto predecessor = openPoolWithConfig(backend, pred_config); + std::vector predecessor_builds; + for (const String & name : {"stale_a", "stale_b", "stale_c"}) + { + auto build = startBuildFor(predecessor, ns, name); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, name, id); + /// no promote -- left dangling, as a crashed build would leave it + predecessor_builds.push_back(std::move(build)); + } + } /// all three cleanup duties remain pending; predecessor publishes no clean farewell + + /// The successor: a tight retry budget so ONE simulated ambiguous response wedges rather than + /// transparently retries away (mirrors the wedge-semantics tests in this file exactly). + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + PoolConfig config; + config.cas_request_budget = budget; + config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); + config.boot_ms_fn = fake_clock; + config.wait_sleep_fn = [&fake_now, &mount_wait_calls](uint64_t ms) + { + ++mount_wait_calls; + fake_now += ms; + }; + SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto successor = openPoolWithConfig(backend, config); + EXPECT_GT(mount_wait_calls, 0u) + << "the unclean predecessor must exercise the injected mount-observation wait"; + + successor->setEventSink([&](const CasEvent & e) { seen.add(e); }); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + /// The successor's own recovery runs first and mints one in-band seal for the predecessor's now-dead + /// epoch (its three precommits are its only durable ids, so the seal takes the very next slot) -- + /// that PUT shares this same `_log/` prefix, so it would eat the fault before the sweep gets a turn. + /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. + backend->fault_skip = 1; + backend->fault_count = 1; /// hits exactly the sweep's FIRST removal chunk's PUT + + /// FIRST trigger (read path): the sweep's removal PUT is uncertain -> the lane wedges; the read + /// itself still succeeds and counts the deferral (existing contract) -- but the shot must NOT be + /// consumed: the flag is re-armed for a later trigger. + const uint64_t deferred_before = global_counters[ProfileEvents::CASRefSweepDeferred].load(); + const uint64_t rearmed_before = global_counters[ProfileEvents::CASRefSweepRearmed].load(); + const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(); + EXPECT_NO_THROW(successor->listRefs(ns)); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before + 1); + EXPECT_TRUE(successor->refLaneWedgedForTest(ns)); + EXPECT_TRUE(successor->needsStalePrecommitSweepForTest(ns)) + << "a failed sweep must re-arm needs_stale_precommit_sweep, not consume the once-per-mount shot"; + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepRearmed].load(), rearmed_before + 1); + + /// Within the backoff window (the injected clock has not advanced) a read must NOT re-attempt -- + /// the bounded-backoff storm latch: no new deferral, flag still armed. + EXPECT_NO_THROW(successor->listRefs(ns)); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before + 1) + << "within the backoff window the sweep must not re-attempt (PUT-storm latch)"; + EXPECT_TRUE(successor->needsStalePrecommitSweepForTest(ns)); + + /// The lost response later lands server-side; past the backoff deadline the NEXT trigger (a + /// mutation this time) retries: the lane resolves its wedge (the first chunk's removals become + /// durable and applied), the re-pass verifies clean, and the flag clears permanently. + backend->materializePendingDelayedWrite(); + fake_now += 60'000; /// beyond any armed backoff (initial 200 ms, max 30 s) + EXPECT_NO_THROW(publishEmptyPart(successor, ns, "fresh")); + EXPECT_FALSE(successor->refLaneWedgedForTest(ns)); + EXPECT_FALSE(successor->needsStalePrecommitSweepForTest(ns)) + << "a verified-clean sweep clears the flag permanently"; + + /// Ground truth: every stale binding reclaimed; the successor's own committed work intact. + const RefTableState final_state = independentFullReplayForTest(*backend, layout, ns); + EXPECT_EQ(final_state.getLifecycle(), RefLifecycle::Live); + EXPECT_TRUE(final_state.getPrecommits().empty()); + EXPECT_TRUE(final_state.getCommitted().contains("fresh")); + + /// Audit (INTROSPECTION-1): exactly ONE `precommit_reclaim` event per reclaimed stale binding -- + /// this is what makes the S13 card's "abandoned precommits reclaimed" counter falsifiable. + std::vector reclaimed_refs; + for (const CasEvent & e : seen.snapshot()) + if (e.type == CasEventType::PrecommitReclaim) + reclaimed_refs.push_back(e.ref_name); + std::sort(reclaimed_refs.begin(), reclaimed_refs.end()); + EXPECT_EQ(reclaimed_refs, (std::vector{"stale_a", "stale_b", "stale_c"})); + EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(), reclaimed_before + 3); +} + +/// Verified-clean semantics: a sweep that finds NOTHING stale clears the flag on its very first pass +/// and emits no reclaim event (so "no abandons" and "reclaim broken" stay distinguishable in the +/// audit log). +TEST(CASRefWriterStalePrecommitSweep, VerifiedCleanSweepClearsFlagWithoutEvents) +{ + using ProfileEvents::global_counters; + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/precommit_sweep_clean"}; + + { + auto predecessor = openPool(backend); + publishEmptyPart(predecessor, ns, "committed_x"); /// committed work only; nothing dangles + } + + SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + auto successor = openPool(backend); + successor->setEventSink([&](const CasEvent & e) { seen.add(e); }); + + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); + const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(); + EXPECT_NO_THROW(successor->listRefs(ns)); + EXPECT_FALSE(successor->needsStalePrecommitSweepForTest(ns)) + << "a clean first pass IS the verified-clean sweep: the flag clears without any removal"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before); + EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(), reclaimed_before); + const std::vector observed = seen.snapshot(); + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), + [](const CasEvent & e) { return e.type == CasEventType::PrecommitReclaim; }), 0); +} + +/// =================================================================================== +/// C1: self-remount establishes a fresh ref-protocol incarnation (spec §Startup And Recovery / +/// §write-fence). A self-remount bumps the durable writer_epoch, so every ref transaction it stamps +/// afterward sorts strictly above any log a dead-incarnation or same-uuid twin left durable under an +/// older epoch, and it drops its stale in-memory cache so the next touch re-recovers under the new +/// epoch. The unfixed code kept the open-time `process_epoch` and the cached tables across the fence-out. +/// =================================================================================== + +namespace +{ + +/// Fence out the mount lease so `tryRemountOnce` reclaims a fresh incarnation (mirrors +/// gtest_cas_pool.cpp's fenceOutMount, without its ASSERT_ macros so it can run outside a fixture). +void fenceOutRefMount(Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + MountLease m = decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + backend.putOverwrite(mount_key, encodeMountLease(m), got->token); +} + +/// The greatest `_log/` transaction id currently present for `ns` (independent of any Pool cache). +std::optional listGreatestLogIdForTest(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + std::optional greatest; + String cursor; + for (;;) + { + const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation && parsed->kind == RefObjectKind::Log + && (!greatest || *greatest < parsed->txn_id)) + greatest = parsed->txn_id; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return greatest; +} + +/// Seed a same-uuid TWIN incarnation that bumped the durable writer_epoch and durably DROPPED `ref_name` +/// (its committed binding `old_ref`) at `{twin_epoch, 1}` -- an id that sorts strictly above every log a +/// Pool wrote under its own (lower) open-time epoch. Returns the twin's epoch. +/// `prev_epoch_seal` is NOT optional decoration here. The twin's drop is sequence 1 of a new epoch, so +/// INV-2's grammar requires it to name the seal that closed the epoch below -- and the reader enforces +/// that, so a twin seeded without it describes a stream with an uncertified epoch boundary, which is +/// exactly what recovery must refuse. The link names the id the recovering pool's own CAS-walk will mint +/// for the dead epoch: one past that epoch's greatest durable id, which is what `seal_of_previous_epoch` +/// derives by listing rather than hard-coding, so the fixture cannot drift from the walk's arithmetic. +uint64_t seedTwinDrop(Backend & backend, const Layout & layout, const RootNamespace & ns, + const String & ref_name, const ManifestRef & old_ref) +{ + uint64_t greatest_in_previous_epoch = 0; + uint64_t previous_epoch = 0; + forEachListedKey(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), [&](const ListedKey & lk) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (!parsed || parsed->kind != RefObjectKind::Log) + return; + if (parsed->txn_id.writer_epoch > previous_epoch + || (parsed->txn_id.writer_epoch == previous_epoch && parsed->txn_id.ref_sequence > greatest_in_previous_epoch)) + { + previous_epoch = parsed->txn_id.writer_epoch; + greatest_in_previous_epoch = parsed->txn_id.ref_sequence; + } + }, 1000); + + const uint64_t twin_epoch = allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); + RefLogTxn twin; + twin.ns = ns.string(); + twin.txn_id = RefTxnId{twin_epoch, 1}; + twin.prev_epoch_seal = RefTxnId{previous_epoch, greatest_in_previous_epoch + 1}; + RefOp drop; + drop.kind = RefOpKind::OwnerTransition; + drop.old_binding = RefOwnerBinding{RefOwnerKind::Committed, ref_name, old_ref}; + twin.ops = {drop}; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, twin); + return twin_epoch; +} + +} + +/// A fence-loss generation is a rejection marker, not a runtime admission token. If remount then loses +/// to a foreign owner, neither a warm name nor a never-seen name may select/materialize a runtime under +/// that intermediate generation; the predecessor remains only as a detached diagnostic object. +TEST(CASRefWriterRemount, FailedRemountPublishesNoRuntimeUnderFenceLossGeneration) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace existing{"srv1/failed-remount-existing"}; + const RootNamespace never_seen{"srv1/failed-remount-never-seen"}; + + publishWithProductionBirth(store, existing, "a"); + const uint64_t predecessor_runtime = store->refTableRuntimeIdentityForTest(existing); + const uint64_t predecessor_generation + = store->refTableRuntimeAdmittedFenceGenerationForTest(existing); + const size_t cached_before = store->refTablesCachedCountForTest(); + ASSERT_NE(predecessor_runtime, 0u); + ASSERT_EQ(predecessor_generation, store->fenceGeneration()); + + const String mount_key = layout.mountKey("test"); + const auto got = backend->get(mount_key); + ASSERT_TRUE(got); + MountLease foreign = decodeMountLease(got->bytes); + foreign.server_uuid = foreign.server_uuid + UInt128{1}; + foreign.seq += 1; + ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, + PutOutcome::Done); + + store->tripMountLost(); + const uint64_t rejected_generation = store->fenceGeneration(); + ASSERT_NE(rejected_generation, predecessor_generation); + EXPECT_FALSE(store->tryRemountOnce()); + EXPECT_FALSE(store->mayMutate()); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->resolveRef(existing, "a"); }); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->resolveRef(never_seen, "a"); }); + EXPECT_EQ(store->refTablesCachedCountForTest(), cached_before); + EXPECT_EQ(store->refTableRuntimeIdentityForTest(existing), predecessor_runtime); + EXPECT_EQ(store->refTableRuntimeAdmittedFenceGenerationForTest(existing), predecessor_generation); + EXPECT_EQ(store->refTableRuntimeIdentityForTest(never_seen), 0u); + EXPECT_NE(store->refTableRuntimeAdmittedFenceGenerationForTest(existing), rejected_generation); + + /// Make the foreign occupant terminal before teardown; it remains foreign and is never taken over. + fenceOutRefMount(*backend, mount_key); +} + +/// C1/N1 (stale cache): a warm table whose committed ref a twin durably dropped must re-recover to the +/// twin's view after a self-remount. The unfixed code kept the stale cache and still resolved the ref. +TEST(CASRefWriterRemount, ReRecoversStaleCacheToTwinDrop) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remount_twin_view"}; + + const ManifestId a_id = publishEmptyPart(store, ns, "a"); + ASSERT_TRUE(store->resolveRef(ns, "a").has_value()); + const uint64_t e1 = store->liveWriterEpoch(); + const uint64_t predecessor_runtime = store->refTableRuntimeIdentityForTest(ns); + const NamespaceLifeId predecessor_life = *store->refTableLifeForTest(ns); + const uint64_t predecessor_generation = store->refTableRuntimeAdmittedFenceGenerationForTest(ns); + + /// A same-uuid twin bumped the durable epoch and durably dropped "a"; this Pool's warm cache never + /// observed it. + const uint64_t twin_epoch = seedTwinDrop(*backend, layout, ns, "a", a_id.ref); + ASSERT_GT(twin_epoch, e1); + ASSERT_TRUE(store->resolveRef(ns, "a").has_value()) << "precondition: the warm cache is stale"; + + fenceOutRefMount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + EXPECT_GT(store->liveWriterEpoch(), twin_epoch); + + /// The remount dropped the stale runtime: the next read re-recovers from the durable objects and + /// adopts the twin's drop -- "a" is gone. + EXPECT_FALSE(store->resolveRef(ns, "a").has_value()) + << "a self-remount must re-recover the table under the new epoch, adopting the twin's drop"; + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), predecessor_runtime); + EXPECT_EQ(store->refTableLifeForTest(ns), predecessor_life); + EXPECT_NE(store->refTableRuntimeAdmittedFenceGenerationForTest(ns), predecessor_generation); + EXPECT_EQ(store->refTableRuntimeAdmittedFenceGenerationForTest(ns), store->fenceGeneration()); +} + +/// C1/N2 (epoch routing + ordering): a post-remount append must stamp its log with the fresh +/// incarnation's live epoch, landing strictly above a twin's durable log (the pagination premise +/// "a new log is never inserted at or below an already durable table log id"). The unfixed code stamped +/// the stale open-time epoch, which sorts BELOW a higher-epoch twin log. +TEST(CASRefWriterRemount, PostRemountAppendCarriesLiveEpochSortingAboveTwinLogs) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remount_epoch_order"}; + + const ManifestId a_id = publishEmptyPart(store, ns, "a"); + const uint64_t e1 = store->liveWriterEpoch(); + const uint64_t twin_epoch = seedTwinDrop(*backend, layout, ns, "a", a_id.ref); + ASSERT_GT(twin_epoch, e1); + + fenceOutRefMount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + const uint64_t e2 = store->liveWriterEpoch(); + ASSERT_GT(e2, twin_epoch); + + publishEmptyPart(store, ns, "b"); + const auto greatest = listGreatestLogIdForTest(*backend, layout, ns); + ASSERT_TRUE(greatest.has_value()); + EXPECT_EQ(greatest->writer_epoch, e2) + << "the newest ref log must carry the fresh incarnation's epoch and be the greatest id"; + EXPECT_GT(*greatest, (RefTxnId{twin_epoch, 1})) + << "the post-remount append must sort strictly above the twin's log"; +} + +/// C1 (wedge disposition): a wedged append lane's runtime (and its wedge) is dropped on a self-remount, +/// the next touch re-recovers a clean lane, and appends resume without hanging. The unfixed code kept the +/// wedged runtime cached across the remount. The drop is a plain cache detach and needs to certify +/// nothing: the undecided PUT the wedge describes is settled by the seal the next recovery writes into +/// its slot -- see `quiesceRefTablesForRemount`'s doc comment (`CasPool.h`). +TEST(CASRefWriterRemount, DiscardsWedgeAndLaneRemainsUsable) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + /// The self-remount below blocks on nothing (see + /// `CASRemountWaits.UnresolvedWedgeRemountPaysNoWaitEither`, `gtest_cas_pool.cpp`); the injected + /// `boot_ms_fn`/`wait_sleep_fn` keep this test off the real clock anyway. + uint64_t fake_boot = 0; + PoolConfig config; + config.cas_request_budget = budget; + config.boot_ms_fn = [&fake_boot] { return fake_boot; }; + config.wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remount_wedge"}; + publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "y"); + + /// Wedge the lane with an ambiguous PUT that never landed server-side. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + fenceOutRefMount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) + << "a self-remount discards the in-memory wedge with the detached runtime"; + + /// The lane is usable, not hung: a fresh append completes and carries the live epoch. + EXPECT_NO_THROW(store->dropRef(ns, "y")); + EXPECT_FALSE(store->resolveRef(ns, "y").has_value()); + const auto greatest = listGreatestLogIdForTest(*backend, layout, ns); + ASSERT_TRUE(greatest.has_value()); + EXPECT_EQ(greatest->writer_epoch, store->liveWriterEpoch()); +} + +/// C1 residual: a flush leader that passed the top-of-flush gate BEFORE a self-remount and stalled +/// mid-flush (here parked at the pre-carve hook, post-top-gate / pre-allocate) across the whole +/// fence-loss + remount window must NOT, on resume, allocate an id and PUT a transaction validated +/// against its now-stale detached cache. The pre-allocate `superseded_by_remount` re-check fails it +/// closed: the caller gets the failure and no backend `_log` object is created. +TEST(CASRefWriterRemount, SupersededLeaderMidFlushFailsClosedCreatesNoObject) +{ + auto backend = std::make_shared(); + /// This test parks a flush leader (`leader_active` stays true) across the ENTIRE `tryRemountOnce` + /// call below by construction (`release` is only set AFTER `tryRemountOnce` returns). Keep the + /// request budget at the file's usual tiny-wedge-test values so every bounded wait on that path + /// stays well under a second. + CasRequestBudget budget; + budget.attempt_timeout_ms = 100; + budget.lease_safety_margin_ms = 100; + uint64_t fake_boot = 0; + PoolConfig config; + config.cas_request_budget = budget; + config.boot_ms_fn = [&fake_boot] { return fake_boot; }; + config.wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remount_midflush"}; + publishEmptyPart(store, ns, "x"); + const auto greatest_before = listGreatestLogIdForTest(*backend, layout, ns); + ASSERT_TRUE(greatest_before.has_value()); + + /// Park the next flush leader at the pre-carve hook (post-top-gate, pre-allocate). Fires once. + std::mutex m; + std::condition_variable cv; + bool entered = false; + bool release = false; + std::atomic hook_fired{false}; + store->setRefPreCarveHookForTest([&] + { + if (hook_fired.exchange(true)) + return; + std::unique_lock lk(m); + entered = true; + cv.notify_all(); + cv.wait(lk, [&] { return release; }); + }); + + auto fut = std::async(std::launch::async, [&]() -> std::string + { + try { store->dropRef(ns, "x"); return "committed"; } + catch (const DB::Exception & e) { return e.message(); } + }); + { std::unique_lock lk(m); cv.wait(lk, [&] { return entered; }); } /// leader parked mid-flush + + /// The remount completes while the leader is parked (the quiesce does not wait for leaders); it marks + /// the table superseded and re-arms the fence. + fenceOutRefMount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + + /// Unpark: the leader resumes, re-checks the flag before allocating, and fails closed. + { std::lock_guard lk(m); release = true; } + cv.notify_all(); + + ASSERT_EQ(fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "the superseded leader hung instead of failing closed"; + const std::string result = fut.get(); + EXPECT_NE(result.find("superseded by a self-remount"), std::string::npos) + << "expected a superseded fail-closed, got: " << result; + + /// No new ref-log object was created: the greatest durable log id is unchanged. + const auto greatest_after = listGreatestLogIdForTest(*backend, layout, ns); + ASSERT_TRUE(greatest_after.has_value()); + EXPECT_EQ(*greatest_after, *greatest_before) + << "a superseded leader must allocate no id and PUT no object"; +} + +/// =================================================================================== +/// Task 11: namespace removal (spec §Namespace Removal) +/// =================================================================================== + +/// A cached writer paused after its ordinary gates must re-check the exact catalog life immediately +/// before id allocation. A concurrent `Live -> Removing` transition therefore admits no late owner. +TEST(CASRefWriterNamespaceRemoval, CachedPositiveWriterCannotAppendAfterRemovingIsPublished) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/removing_blocks_cached_writer"}; + publishEmptyPart(store, ns, "existing"); + + const CasRefCatalog::Snapshot before = CasRefCatalog::read(*backend, layout); + const auto observed = std::find_if(before.catalog.entries.begin(), before.catalog.entries.end(), + [&](const CatalogEntry & entry) { return entry.ns == ns; }); + ASSERT_NE(observed, before.catalog.entries.end()); + ASSERT_EQ(observed->state, NsState::Live); + const CatalogEntry & exact_live = *observed; + const auto greatest_before = listGreatestLogIdForTest(*backend, layout, ns); + ASSERT_TRUE(greatest_before); + + std::mutex mutex; + std::condition_variable cv; + bool entered = false; + bool release = false; + std::atomic hook_fired{false}; + store->setRefPreCarveHookForTest([&] + { + if (hook_fired.exchange(true)) + return; + std::unique_lock lock(mutex); + entered = true; + cv.notify_all(); + cv.wait(lock, [&] { return release; }); + }); + + auto writer = std::async(std::launch::async, [&]() -> String + { + try + { + publishEmptyPart(store, ns, "late"); + return "committed"; + } + catch (const DB::Exception & e) + { + return e.message(); + } + }); + + bool writer_parked = false; + { + std::unique_lock lock(mutex); + writer_parked = cv.wait_for(lock, std::chrono::seconds(10), [&] { return entered; }); + } + if (writer_parked) + { + CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + { + RefCatalog next = current; + const auto it = std::find(next.entries.begin(), next.entries.end(), exact_live); + if (it == next.entries.end()) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "exact Live row changed during test transition"); + it->state = NsState::Removing; + it->removal_started_round = 0; + return next; + }); + } + { + std::lock_guard lock(mutex); + release = true; + } + cv.notify_all(); + + EXPECT_TRUE(writer_parked) << "cached writer did not reach the deterministic pre-carve seam"; + ASSERT_EQ(writer.wait_for(std::chrono::seconds(10)), std::future_status::ready); + const String result = writer.get(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_NE(result, "committed") << "cached positive ownership appended after `Removing` became visible"; + EXPECT_FALSE(store->resolveRef(ns, "late")); + EXPECT_EQ(listGreatestLogIdForTest(*backend, layout, ns), greatest_before); +} + +/// dropNamespace's ONE body transaction names an exact removal for every committed ref AND every +/// dangling precommit, with `remove_namespace` as the FINAL op -- never any other shape. +TEST(CASRefWriterNamespaceRemoval, TxnNamesEveryOwnerThenRemoveNamespace) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remove_shape"}; + + publishEmptyPart(store, ns, "committed_1"); + publishEmptyPart(store, ns, "committed_2"); + /// One precommit left dangling (never promoted) so the removal txn must ALSO name it. + auto build = startBuildFor(store, ns, "dangling"); + const ManifestId dangling_id = build->stageManifest({}); + build->precommitAdd(ns, "dangling", dangling_id); + + store->dropNamespace(ns); + + /// The newest `_log/` object for `ns` is the removal transaction. + std::optional newest_log; + { + String cursor; + for (;;) + { + const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & lk : page.keys) + { + const auto parsed = layout.parseRefObjectKey(lk.key); + if (parsed && parsed->life_id == DB::Cas::tests::fixture::fixtureLife(ns).incarnation && parsed->kind == RefObjectKind::Log + && (!newest_log || *newest_log < parsed->txn_id)) + newest_log = parsed->txn_id; + } + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + } + ASSERT_TRUE(newest_log.has_value()); + const auto got = backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest_log)); + ASSERT_TRUE(got.has_value()); + const RefLogTxn removal_txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), *newest_log); + + ASSERT_FALSE(removal_txn.ops.empty()); + EXPECT_EQ(removal_txn.ops.back().kind, RefOpKind::RemoveNamespace); + size_t owner_removals = 0; + for (size_t i = 0; i + 1 < removal_txn.ops.size(); ++i) + { + const RefOp & op = removal_txn.ops[i]; + EXPECT_EQ(op.kind, RefOpKind::OwnerTransition); + EXPECT_TRUE(op.old_binding.has_value()); + EXPECT_FALSE(op.new_binding.has_value()); + ++owner_removals; + } + EXPECT_EQ(owner_removals, 3u) << "2 committed + 1 dangling precommit"; +} + +/// The terminal transaction is the only durable removal record. Generation 7 never publishes a +/// terminal `Removed` snapshot; the ordinary cleanup/janitor paths own old immutable stream debris. +TEST(CASRefWriterNamespaceRemoval, RemovalPublishesTerminalLogWithoutTerminalSnapshot) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remove_snapshot"}; + + publishEmptyPart(store, ns, "a"); + const auto snapshot_before = store->newestPublishedSnapshotIdForTest(ns); + store->dropNamespace(ns); + + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), snapshot_before) + << "removal must not publish a terminal snapshot"; + EXPECT_GT(store->tailSinceSnapshotCountForTest(ns), 0u) + << "the terminal transaction remains ordinary immutable stream work until GC folds it"; + + size_t terminal_logs = 0; + for (const ListedKey & listed : backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), "", 1000).keys) + { + const auto parsed = layout.parseRefObjectKey(listed.key); + if (!parsed || parsed->kind != RefObjectKind::Log) + continue; + const auto got = backend->get(listed.key); + ASSERT_TRUE(got.has_value()); + const RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), parsed->txn_id); + if (!txn.ops.empty() && txn.ops.back().kind == RefOpKind::RemoveNamespace) + ++terminal_logs; + } + EXPECT_EQ(terminal_logs, 1u); +} + +/// Review fix (prerequisite to this task's dropNamespace rewiring): `flushRefBatch`'s per-item +/// validation previously previewed each op as its OWN single-op trial transaction, so a +/// whole-transaction-shape rule ("remove_namespace must be the FINAL op") trivially passed on every +/// singleton slice regardless of an item's REAL combined shape -- a malformed item would only have +/// been caught by the post-persist apply, AFTER its transaction object was already durable (bricking +/// the table on every future recovery and permanently wedging this table's lane). Drives +/// `appendRefOps` directly with a deliberately malformed multi-op item (remove_namespace not last) to +/// prove the whole-item shape check now rejects it BEFORE any backend object is created. +TEST(CASRefWriterNamespaceRemoval, MalformedShapeWithRemoveNamespaceNotFinalRejectedBeforeAnyCreate) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/malformed_shape"}; + publishEmptyPart(store, ns, "a"); /// births the table so the malformed item isn't ALSO rejected + /// for the unrelated reason "namespace_birth was needed first" + + const uint64_t put_before = backend->putTotal(); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + store->appendRefOps(ns, MutationScope::wholeShard(), + [](const RefTableState &) -> std::vector + { + RefOp remove_ns_1; + remove_ns_1.kind = RefOpKind::RemoveNamespace; + RefOp remove_ns_2; + remove_ns_2.kind = RefOpKind::RemoveNamespace; + return {remove_ns_1, remove_ns_2}; /// remove_namespace NOT the final op -- malformed + }, + /// Deliberately mislabel the malformed terminal as an ordinary mutation: this bypasses + /// the public removal-capability preflight and proves the txn-wide shape check itself + /// rejects the object before the later capability check or any backend mutation. + RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + + EXPECT_EQ(backend->putTotal(), put_before) << "the malformed shape must be rejected before any object is created"; + ASSERT_TRUE(store->resolveRef(ns, "a").has_value()) << "the malformed attempt left no trace on the table"; +} + +/// A caller cannot turn the generic append surface into a second namespace-removal capability, even +/// when it disguises terminal operations as an ordinary mutation kind. Only `dropNamespace` may carry +/// the exact runtime ownership established by the durable `Live -> Removing` transition. +TEST(CASRefWriterNamespaceRemoval, GenericAppendCannotWriteTerminalWhileCatalogIsLive) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/unauthorized_terminal"}; + publishEmptyPart(store, ns, "owned"); + const CatalogEntry live = catalogEntryOrThrow(*backend, layout, ns); + ASSERT_EQ(live.state, NsState::Live); + const auto greatest_before = listGreatestLogIdForLifeForTest( + *backend, layout, NamespaceLifeId::fromCatalogEntry(ns, live.incarnation)); + ASSERT_TRUE(greatest_before); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + store->appendRefOps(ns, MutationScope::wholeShard(), + [](const RefTableState & state) -> std::vector + { + std::vector ops; + for (const auto [ref_name, row] : state.getCommitted()) + { + RefOp remove_owner; + remove_owner.kind = RefOpKind::OwnerTransition; + remove_owner.old_binding = RefOwnerBinding{ + RefOwnerKind::Committed, ref_name, row.manifest_ref}; + ops.push_back(std::move(remove_owner)); + } + RefOp terminal; + terminal.kind = RefOpKind::RemoveNamespace; + ops.push_back(terminal); + return ops; + }, + RootMutationOrigin::Writer, RootMutationKind::Publish, + /*skip_stale_precommit_sweep=*/true); + }); + + EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns), live); + EXPECT_EQ(listGreatestLogIdForLifeForTest( + *backend, layout, NamespaceLifeId::fromCatalogEntry(ns, live.incarnation)), greatest_before) + << "an unauthorized terminal must allocate no id and create no ref-log object"; + EXPECT_TRUE(store->resolveRef(ns, "owned")); +} + +/// The public generic surface must reject the terminal-capable operation kind before resolving or +/// creating a life. Otherwise an absent name can acquire a catalog row and checkpoint before the +/// internal terminal capability check rejects the actual operations. +TEST(CASRefWriterNamespaceRemoval, GenericTerminalOnAbsentNamePerformsZeroDurableMutation) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/absent_unauthorized_terminal"}; + const CasRefCatalog::Snapshot catalog_before = CasRefCatalog::read(*backend, store->layout()); + + backend->resetCounts(); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + store->appendRefOps(ns, MutationScope::wholeShard(), + [](const RefTableState &) -> std::vector + { + RefOp terminal; + terminal.kind = RefOpKind::RemoveNamespace; + return {terminal}; + }, + RootMutationOrigin::Writer, RootMutationKind::DropNamespace, + /*skip_stale_precommit_sweep=*/true); + }); + + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); + const CasRefCatalog::Snapshot catalog_after = CasRefCatalog::read(*backend, store->layout()); + EXPECT_EQ(catalog_after.token, catalog_before.token); + EXPECT_EQ(catalog_after.catalog, catalog_before.catalog); + EXPECT_FALSE(store->refTableLifeForTest(ns)); +} + +/// A namespace file births a catalog life and checkpoint without necessarily creating a ref stream. +/// Removing that table must still publish terminal evidence and let GC retire the catalog row. +TEST(CASRefWriterNamespaceRemoval, CatalogedNamespaceFilesOnlyLifeCompletesRemoval) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.gc_fold_max_defer_rounds = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/files_only"}; + const NamespaceLifeId life = store->namespaceLife(ns); + store->putNamespaceFile(life, "format_version.txt", "1\n"); + ASSERT_TRUE(backend->list(layout.namespaceStreamPrefix(life), "", 100).keys.empty()); + ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Live); + + EXPECT_NO_THROW(store->dropNamespace(ns)); + ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + + const ListPage terminal_page = backend->list(layout.namespaceStreamPrefix(life), "", 100); + ASSERT_EQ(terminal_page.keys.size(), 1u); + const auto parsed = layout.parseRefObjectKey(terminal_page.keys.front().key); + ASSERT_TRUE(parsed); + const auto terminal_body = backend->get(terminal_page.keys.front().key); + ASSERT_TRUE(terminal_body); + const RefLogTxn terminal = decodeRefLogTxn( + openObject(FormatId::RefLog, terminal_body->bytes), ns.string(), parsed->txn_id); + ASSERT_EQ(terminal.ops.size(), 2u); + EXPECT_EQ(terminal.ops[0].kind, RefOpKind::NamespaceBirth); + EXPECT_EQ(terminal.ops[1].kind, RefOpKind::RemoveNamespace); + + Gc gc(store, UInt128{181}); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + (void)runRegularRoundReclaiming(gc); + const RefCatalog after = CasRefCatalog::read(*backend, layout).catalog; + EXPECT_TRUE(std::none_of(after.entries.begin(), after.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns; + })); +} + +/// If the first catalog read after closing the positive lane fails, the catch-side authoritative read +/// is still allowed to prove the exact original `Live` row and reopen admission. +TEST(CASRefWriterNamespaceRemoval, PredurableCatalogReadFailureReopensExactLiveLane) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/predurable_read_failure"}; + publishEmptyPart(store, ns, "owned"); + const CatalogEntry live = catalogEntryOrThrow(*backend, layout, ns); + + backend->catalog_fault_key = layout.refCatalogKey(); + backend->catalog_gets_before_fault = 1; /// initial discovery succeeds; post-close observation fails + backend->catalog_get_fault_count = 1; + EXPECT_THROW(store->dropNamespace(ns), std::runtime_error); + EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns), live); + + EXPECT_NO_THROW(store->updateRefPublishedAt(ns, "owned", [](RefPublishedAtUpdate & update) + { + update.published_at_ms = 17; + })) << "a fresh exact Live observation must reopen the lane after a pre-durable failure"; + EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Live); +} + +/// spec §Namespace Removal (writer, line 666): "After the transaction is durable, it applies the same +/// operations to memory, cancels local builds, and rejects further ordinary mutations." An in-flight +/// build for the removed namespace must be cancelled once (and only once) the removal is durable: its +/// next operation throws (ABORTED, from requireAlive) rather than promoting a fresh committed ref into +/// the just-removed namespace. +TEST(CASRefWriterNamespaceRemoval, DropNamespaceCancelsInFlightBuildAndNextOpThrows) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/remove_cancels_build"}; + + publishEmptyPart(store, ns, "committed"); /// births the table + one committed ref + + /// An in-flight build for ns: staged + precommit-added, never promoted. + auto build = startBuildFor(store, ns, "inflight"); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, "inflight", id); + + store->dropNamespace(ns); + + /// The build is cancelled: EVERY subsequent operation fails fast at `requireAlive` with + /// NETWORK_ERROR (fix #37 phase 2's CAS write-retry-later reroute). `stageManifest` is the + /// discriminator -- it has NO namespace-lifecycle gate, so an UN-cancelled build would happily + /// execute it (staging more debris into a dead namespace); only cancellation stops it. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->stageManifest({}); }); + /// And it certainly cannot promote a fresh committed ref into the removed namespace (the important + /// invariant -- though the old WPromote "precommit removed" guard also blocked this, less directly). + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->promote(ns, "inflight", build->buildId(), id); }); + + /// The cancelled build did not recreate anything in the removed namespace. + EXPECT_FALSE(store->resolveRef(ns, "inflight").has_value()); + EXPECT_FALSE(store->resolveRef(ns, "committed").has_value()) << "the whole namespace was removed"; +} + +/// The catalog transition precedes the terminal append. If that append is unresolved, the namespace +/// remains `Removing`, positive ownership is refused, and a retry of the same removal resolves the wedge. +TEST(CASRefWriterNamespaceRemoval, RemovalAppendFailureLeavesRemovingAndRetryCompletes) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/remove_fault_keeps_build"}; + + publishEmptyPart(store, ns, "committed"); + + auto build = startBuildFor(store, ns, "inflight"); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, "inflight", id); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + + EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + EXPECT_FALSE(store->resolveRef(ns, "committed")) + << "a fresh name lookup must not expose a catalog-Removing life"; + /// The build was NOT cancelled: a non-append operation (`stageManifest` -- it never touches the now + /// wedged ref-append lane) still succeeds; it would throw ABORTED had the build been cancelled. + EXPECT_NO_THROW(build->stageManifest({})); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->promote(ns, "inflight", build->buildId(), id); + }); + + EXPECT_NO_THROW(store->dropNamespace(ns)); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->stageManifest({}); }); +} + +/// `namespaceStillLogicallyPresent` must stay `true` for the entire window between the catalog's +/// durable `Live -> Removing` transition and the terminal `remove_namespace` append actually landing -- +/// the crash-shaped case the fix exists for. Reuses the injected stream-write fault shape from +/// `RemovalAppendFailureLeavesRemovingAndRetryCompletes`. +TEST(CASRefWriterNamespaceRemoval, PresenceProbeStaysTrueThroughRemovingUntilTerminalRetrySucceeds) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/presence_removing_no_terminal"}; + + publishEmptyPart(store, ns, "committed"); + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + + ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)) + << "the catalog transitioned but the terminal append never landed -- cleanup is unproven"; + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(store->namespaceStillLogicallyPresent(ns)) + << "the retried removal's terminal is now durable"; +} + +/// A `Creating` row is conservative in both directions -- present, and removal refuses to cancel it +/// while its creator fence cannot be proven dead, then succeeds once a terminal certificate (here, a +/// GC-fenced lease for the same server root) makes the fence provably terminal. +TEST(CASRefWriterNamespaceRemoval, PresenceProbeCreatingIsPresentAndRemovalWaitsForCreatorFenceTerminality) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/presence_still_creating"}; + + CatalogEntry entry; + entry.ns = ns; + entry.state = NsState::Creating; + entry.incarnation = UInt128(99); + entry.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); + + /// The creator fence names an unmounted server root: `isCreatorFenceTerminal` cannot certify it + /// dead (absence proves nothing), so removal fails closed rather than cancelling a `Creating` row a + /// live writer might still publish into. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Creating); + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); + + /// Publish a GC-fenced lease for the SAME server root -- one of `isCreatorFenceTerminal`'s accepted + /// certificates -- and removal now cancels the row outright (no `Removing` transition for a + /// namespace that never reached `Live`). + MountLease dead; + dead.writer_epoch = store->liveWriterEpoch(); + dead.gc_fenced = true; + dead.seq = 1; + dead.write_attempt_id = UInt128{1}; + backend->putIfAbsent(layout.mountKey("srv1"), encodeMountLease(dead)); + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(store->namespaceStillLogicallyPresent(ns)); +} + +/// A "no catalog row" observation must never be turned into `false` by a race. Pausing the probe +/// right after its first catalog read and admitting a fresh `Creating` row before it resumes must +/// answer present -- the second read sees the born row; a stale absent answer is the one forbidden +/// outcome. A namespace absent from both reads legitimately settles on absent. +TEST(CASRefWriterNamespaceRemoval, PresenceProbeNoRowObservationRevalidatesRatherThanRacingToAbsent) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace racing{"srv1/presence_no_row_races_birth"}; + const RootNamespace stable{"srv1/presence_no_row_stays_absent"}; + + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + store->setNamespacePresenceProbeAfterFirstReadHookForTest([&] + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }); + + std::exception_ptr raced_error; + bool raced_answer = false; + std::thread racer([&] + { + try + { + raced_answer = store->namespaceStillLogicallyPresent(racing); + } + catch (...) + { + raced_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + CatalogEntry born; + born.ns = racing; + born.state = NsState::Creating; + born.incarnation = UInt128(1234); + born.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, born); + + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + racer.join(); + store->setNamespacePresenceProbeAfterFirstReadHookForTest(nullptr); + + if (raced_error) + std::rethrow_exception(raced_error); + EXPECT_TRUE(raced_answer) << "a namespace born after the first read must never resolve to a stale absent"; + + /// Negative control: an unraced, genuinely absent namespace settles on `false`. + EXPECT_FALSE(store->namespaceStillLogicallyPresent(stable)); +} + +/// The starvation regression: proving THIS row absent must not require the WHOLE catalog to hold +/// still. An unrelated namespace admitted between the probe's two reads changes the catalog token and +/// content, yet the target -- absent from both reads -- settles on `false` instead of a retry storm +/// (observed live as a 193/194 retry-later loop under a parallel workload sharing one pool). +TEST(CASRefWriterNamespaceRemoval, PresenceProbeIgnoresUnrelatedCatalogChurnBetweenReads) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace target{"srv1/presence_churn_target_stays_absent"}; + const RootNamespace unrelated{"srv1/presence_churn_unrelated_born"}; + + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + store->setNamespacePresenceProbeAfterFirstReadHookForTest([&] + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }); + + std::exception_ptr probe_error; + bool answer = true; + std::thread prober([&] + { + try + { + answer = store->namespaceStillLogicallyPresent(target); + } + catch (...) + { + probe_error = std::current_exception(); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + CatalogEntry born; + born.ns = unrelated; + born.state = NsState::Creating; + born.incarnation = UInt128(5678); + born.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; + CasRefCatalog::casAdmitEntry(*backend, layout, 1, born); + + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + prober.join(); + store->setNamespacePresenceProbeAfterFirstReadHookForTest(nullptr); + + if (probe_error) + std::rethrow_exception(probe_error); + EXPECT_FALSE(answer) << "unrelated churn between the two reads must not force a retry or a wrong present"; +} + +/// Every unreadable or ambiguous observation must throw, never answer `false`. Covers a catalog `GET` +/// failure on the probe's very first read, and a lost mount fence discovered mid-probe. Not covered +/// here: a missing checkpoint for a `Removing` row, and an ambiguous incarnation -- both would need a +/// raw-catalog-write test helper this suite does not currently expose. +TEST(CASRefWriterNamespaceRemoval, PresenceProbeCatalogReadFailurePropagatesRatherThanAnsweringAbsent) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/presence_catalog_read_fault"}; + + backend->catalog_fault_key = store->layout().refCatalogKey(); + backend->catalog_gets_before_fault = 0; + backend->catalog_get_fault_count = 1; + EXPECT_THROW((void)store->namespaceStillLogicallyPresent(ns), std::runtime_error); +} + +TEST(CASRefWriterNamespaceRemoval, PresenceProbeFenceLossPropagatesRatherThanAnsweringAbsent) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/presence_fence_loss"}; + publishEmptyPart(store, ns, "x"); + + const String mount_key = store->layout().mountKey("test"); + const auto got = backend->get(mount_key); + ASSERT_TRUE(got); + MountLease foreign = decodeMountLease(got->bytes); + foreign.server_uuid = foreign.server_uuid + UInt128{1}; + foreign.seq += 1; + ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, PutOutcome::Done); + store->tripMountLost(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->namespaceStillLogicallyPresent(ns); }); +} + +/// Pin all three facade states against each other on one namespace -- `Live` (present, content +/// readable), incomplete `Removing` (present, content deliberately unreadable), and terminal `Removing` +/// (absent, immediately, no GC). +TEST(CASRefWriterNamespaceRemoval, PresenceProbeFacadeConsistencyAcrossRemovalLifecycle) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = kSingleAttemptDeadlineMs; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + auto store = openPool(backend, budget); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/presence_facade_consistency"}; + + publishEmptyPart(store, ns, "committed"); + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); + EXPECT_FALSE(store->listRefs(ns).empty()); + + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + + EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)) + << "present for cleanup, even though content below is about to prove unreadable"; + EXPECT_FALSE(store->namespaceFilesLifeIfReadable(ns).has_value()); + EXPECT_FALSE(store->resolveRef(ns, "committed").has_value()); + + EXPECT_NO_THROW(store->dropNamespace(ns)); + EXPECT_FALSE(store->namespaceStillLogicallyPresent(ns)); + EXPECT_FALSE(store->namespaceFilesLifeIfReadable(ns).has_value()); +} + +/// Fix-verify review finding: `namespaceStillLogicallyPresent`'s `Removing` branch proved the OBSERVED +/// incarnation's terminal and returned `false` for the name without re-checking whether the catalog had +/// moved on since. Proving one incarnation terminal is not proof the CURRENT logical namespace is +/// absent: GC can delete the now-terminal row and a successor can be born under the same name while the +/// probe's own recovery call (real I/O, no upper bound) is still in flight. Drives that exact +/// interleaving deterministically via the terminal-proven hook: pause right after the predecessor's +/// terminal is proven, drain GC to actually delete its row, birth a successor under the same name, then +/// resume and require `true` (present) -- never the stale `false` the unfixed probe would answer. +TEST(CASRefWriterNamespaceRemoval, PresenceProbeRevalidatesAfterTerminalProvenRatherThanRacingToStaleAbsent) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.gc_fold_max_defer_rounds = 8; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/presence_terminal_revalidate"}; + const UInt128 gc_id = hexToU128("00000000000000000000000000000001"); + + const auto catalog_entry = [&]() -> std::optional + { + const RefCatalog catalog = CasRefCatalog::read(*backend, layout).catalog; + const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns; + }); + return it == catalog.entries.end() ? std::nullopt : std::optional{*it}; + }; + + /// `publishEmptyPart` pins its catalog entry to `fixture::fixtureLife(ns)`, a life derived from + /// `ns` alone -- the SAME value every time for the same name, which is exactly wrong for a test + /// whose whole point is that the successor's incarnation must differ from the predecessor's. + /// `publishWithProductionBirth` goes through the real birth path (`resolveNamespaceLife`'s random + /// mint), so both incarnations below are independently random. + publishWithProductionBirth(store, ns, "predecessor"); + const std::optional predecessor = catalog_entry(); + ASSERT_TRUE(predecessor.has_value()); + + Gc gc(store, gc_id); + store->dropNamespace(ns); /// terminal durable; catalog row still Removing until GC deletes it + + /// Pause the probe right after it proves the predecessor's terminal, before its revalidation read. + std::mutex mutex; + std::condition_variable cv; + bool paused = false; + bool resume = false; + store->setNamespacePresenceProbeAfterTerminalProvenHookForTest([&] + { + std::unique_lock lock(mutex); + paused = true; + cv.notify_all(); + cv.wait(lock, [&] { return resume; }); + }); + + std::optional probe_result; + std::exception_ptr probe_error; + std::thread prober([&] + { + try + { + probe_result = store->namespaceStillLogicallyPresent(ns); + } + catch (...) + { + probe_error = std::current_exception(); + } + }); + /// A fatal assertion below (predecessor/successor state, GC round shape) must not skip joining + /// `prober` -- it is still blocked on `cv` at that point, and destructing a joinable `std::thread` + /// calls `std::terminate`, aborting the whole binary and hiding every test queued after this one. + bool prober_joined = false; + SCOPE_EXIT({ + if (!prober_joined) + { + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + prober.join(); + store->setNamespacePresenceProbeAfterTerminalProvenHookForTest(nullptr); + } + }); + { + std::unique_lock lock(mutex); + cv.wait(lock, [&] { return paused; }); + } + + /// While the probe is paused: drain GC to actually delete the predecessor's catalog row (fold the + /// terminal, then a drain-only round to adopt the evidence and delete the row -- same two-round + /// shape `SameNameSameWriterEpochRebirth...` uses), then birth a successor under the SAME name. The + /// row must be gone before a fresh creation is admitted at all, so this also proves the row really + /// was deleted, not merely that the test raced ahead of production's own invariants. + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "the terminal delta must fold"; + ASSERT_TRUE(runRegularRoundReclaiming(gc).deferred) << "the drain-only round must adopt the evidence seal"; + ASSERT_FALSE(catalog_entry().has_value()) << "control: the predecessor's row is really gone before rebirth"; + publishWithProductionBirth(store, ns, "successor"); + const std::optional successor = catalog_entry(); + ASSERT_TRUE(successor.has_value()); + ASSERT_NE(successor->incarnation, predecessor->incarnation); + + { + std::lock_guard lock(mutex); + resume = true; + } + cv.notify_all(); + prober.join(); + store->setNamespacePresenceProbeAfterTerminalProvenHookForTest(nullptr); + prober_joined = true; + + ASSERT_FALSE(probe_error) << "a successor born under the same name must never surface as an error"; + ASSERT_TRUE(probe_result.has_value()); + EXPECT_TRUE(*probe_result) + << "the predecessor's proven terminal must not answer false once a successor occupies its name"; +} + +/// `StorageJoin`/`StorageSet::truncate` call `disk->removeRecursive(path)` then `disk->createDirectories +/// (path)`. `createDirectories` is a CAS no-op (`ContentAddressedTransaction::createDirectory` only +/// checks write admission, it never touches the catalog), so the actual re-mint happens lazily on the +/// FIRST subsequent write, which resolves through `namespaceLife` exactly like this test does directly. +/// Right after `TRUNCATE` the catalog row is still `Removing` -- GC has not yet folded and deleted it -- +/// so that first write throws a typed retry-later error rather than silently wedging or corrupting +/// anything: the same self-healing window the presence-probe revalidation above depends on. Before the +/// `existsDirectory` fix, the directory never reported as present in the first place, so `TRUNCATE` +/// silently skipped `removeRecursive` entirely and the table kept its OLD contents -- a different, +/// quieter wrong answer than this one, not a newly introduced break. +TEST(CASRefWriterNamespaceRemoval, FilesOnlyNamespaceTruncateThrowsRetryLaterUntilGcReclaimsThenRebirths) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.gc_fold_max_defer_rounds = 8; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const RootNamespace ns{"srv1/truncate_retry_then_rebirth"}; + const UInt128 gc_id = hexToU128("00000000000000000000000000000002"); + + /// Birth a files-only namespace: `StorageJoin`/`StorageSet` never publish a MergeTree part, their + /// table root only ever carries plain table files (`putNamespaceFile`'s shape, not a manifest ref). + const NamespaceLifeId predecessor_life = store->namespaceLife(ns); + store->putNamespaceFile(predecessor_life, "data.bin", "predecessor-contents"); + + Gc gc(store, gc_id); + store->dropNamespace(ns); /// the TRUNCATE-shaped removeRecursive: terminal durable, row still Removing + + /// The very next write CAS would attempt (the `createDirectories` no-op already ran; this is the + /// first real write) must not be told the namespace is gone, and must not silently mint into a + /// row still occupied by the predecessor -- it throws retry-later. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->namespaceLife(ns); }); + + /// Drain GC (fold the terminal, then a drain-only round to delete the now-evidenced row) -- same + /// two-round shape the presence-probe revalidation test above uses. + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "the terminal delta must fold"; + ASSERT_TRUE(runRegularRoundReclaiming(gc).deferred) << "the drain-only round must adopt the evidence seal"; + + /// Self-healed: the same logical name now mints a fresh incarnation and accepts writes again, + /// exactly what a retried `INSERT` (or a retried `TRUNCATE`) after the CAS write's retry-later gets. + const NamespaceLifeId successor_life = store->namespaceLife(ns); + EXPECT_NE(successor_life.incarnation, predecessor_life.incarnation); + store->putNamespaceFile(successor_life, "data.bin", "successor-contents"); + const auto successor_contents = store->getNamespaceFile(successor_life, "data.bin"); + ASSERT_TRUE(successor_contents.has_value()); + EXPECT_EQ(*successor_contents, "successor-contents"); +} + +/// Cancellation is namespace-scoped: dropping namespace N must not cancel an in-flight build targeting a +/// DIFFERENT namespace M -- that build promotes normally. +TEST(CASRefWriterNamespaceRemoval, DropNamespaceDoesNotCancelBuildsInOtherNamespaces) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns_dropped{"srv1/remove_me"}; + const RootNamespace ns_other{"srv1/keep_me"}; + + publishEmptyPart(store, ns_dropped, "x"); + + /// An in-flight build in a DIFFERENT namespace. + auto other_build = startBuildFor(store, ns_other, "y"); + const ManifestId id = other_build->stageManifest({}); + other_build->precommitAdd(ns_other, "y", id); + + store->dropNamespace(ns_dropped); + + /// The other namespace's build is untouched: it promotes successfully and its ref resolves. + EXPECT_NO_THROW(other_build->promote(ns_other, "y", other_build->buildId(), id)); + EXPECT_TRUE(store->resolveRef(ns_other, "y").has_value()); +} + +/// A writer-side create/resolution cannot reuse the predecessor while its catalog row is `Removing`. +/// Both the resident-runtime and fresh-runtime paths return typed retry-later without a durable write. +TEST(CASRefWriterNamespaceRemoval, CreateAgainstRemovingRetriesWithoutMutation) +{ + auto backend = std::make_shared(); + const RootNamespace ns{"srv1/create-while-removing"}; + + { + auto store = openPool(backend); + publishEmptyPart(store, ns, "predecessor"); + store->dropNamespace(ns); + + backend->resetCounts(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->namespaceLife(ns); }); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); + } + + auto fresh_store = openPool(backend); + backend->resetCounts(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)fresh_store->namespaceLife(ns); }); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +/// Same-name rebirth must not inherit the predecessor's physical life or folded cursor even when the +/// writer mount and its per-name runtime stay resident throughout the complete real removal sequence. +TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResidentLifeAndStartsAtZeroCoverage) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.gc_fold_max_defer_rounds = 8; + config.ref_table_cache_bytes = 0; /// unbounded: no eviction can explain a fresh resolution + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/same-name-rebirth"}; + const UInt128 gc_id = hexToU128("00000000000000000000000000000001"); + + const auto publish_without_fixture_admission = [&](const String & ref) + { + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); + }; + const auto catalog_entry = [&]() -> std::optional + { + const RefCatalog catalog = CasRefCatalog::read(*backend, layout).catalog; + const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns; + }); + return it == catalog.entries.end() ? std::nullopt : std::optional{*it}; + }; + + publish_without_fixture_admission("predecessor"); + const CatalogEntry predecessor = *catalog_entry(); + const uint64_t writer_epoch = store->liveWriterEpoch(); + ASSERT_EQ(store->refTableLifeForTest(ns)->incarnation, predecessor.incarnation); + const uint64_t runtime_identity = store->refTableRuntimeIdentityForTest(ns); + ASSERT_NE(runtime_identity, 0u); + + Gc gc(store, gc_id); + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + const auto predecessor_row = seal.ref_lives.find(predecessor.incarnation); + ASSERT_NE(predecessor_row, seal.ref_lives.end()); + ASSERT_NE(predecessor_row->second.coverage.last_folded_ref_id, RefTxnId{}); + + const uint64_t puts_before_drop = backend->putTotal(); + store->dropNamespace(ns); + ASSERT_GT(backend->putTotal(), puts_before_drop) + << "control: the real removal call returned after durably writing its terminal artifacts"; + std::optional terminal_id; + for (const ListedKey & listed : backend->list( + layout.namespaceStreamPrefix(NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation)), + "", 1000).keys) + { + const auto parsed = layout.parseRefObjectKey(listed.key); + if (parsed && parsed->kind == RefObjectKind::Log + && (!terminal_id || *terminal_id < parsed->txn_id)) + terminal_id = parsed->txn_id; + } + ASSERT_TRUE(terminal_id.has_value()); + const auto terminal_body = backend->get(layout.refLogKey( + NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation), *terminal_id)); + ASSERT_TRUE(terminal_body.has_value()); + const RefLogTxn terminal = decodeRefLogTxn( + openObject(FormatId::RefLog, terminal_body->bytes), ns.string(), *terminal_id); + ASSERT_FALSE(terminal.ops.empty()); + ASSERT_EQ(terminal.ops.back().kind, RefOpKind::RemoveNamespace) + << "control: the newest old-life log is the production terminal record"; + const std::optional removing = catalog_entry(); + ASSERT_TRUE(removing.has_value()); + ASSERT_EQ(removing->state, NsState::Removing) + << "the real terminal returned durable, but its catalog row stayed Live"; + ASSERT_EQ(removing->incarnation, predecessor.incarnation); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), runtime_identity) + << "the removal path must invalidate the resident runtime's life, not pass through eviction"; + + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "the terminal delta must fold"; + state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + ASSERT_TRUE(seal.ref_lives.at(predecessor.incarnation).cleanup_evidence.has_value()); + + const RoundReport drain = runRegularRoundReclaiming(gc); + ASSERT_TRUE(drain.deferred) << "the drain-only idle invocation must leave the evidence seal adopted"; + ASSERT_FALSE(catalog_entry().has_value()); + ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u) + << "catalog deletion must detach the predecessor from the name slot"; + + publish_without_fixture_admission("successor"); + const CatalogEntry successor = *catalog_entry(); + ASSERT_EQ(successor.ns, predecessor.ns) << "the exact same logical namespace must be reused"; + ASSERT_NE(successor.incarnation, predecessor.incarnation); + ASSERT_EQ(store->liveWriterEpoch(), writer_epoch) << "rebirth must use the same mounted writer epoch"; + ASSERT_NE(store->refTableRuntimeIdentityForTest(ns), runtime_identity) + << "rebirth must publish a distinct successor runtime, not reset the predecessor"; + ASSERT_EQ(store->refTableLifeForTest(ns)->incarnation, successor.incarnation); + + const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation); + const ListPage successor_stream = backend->list(layout.namespaceStreamPrefix(successor_life), "", 1000); + ASSERT_FALSE(successor_stream.keys.empty()) << "the real successor writer produced foldable stream work"; + std::vector successor_phases; + gc.setPhaseSink([&](const GcPhaseRecord & phase) { successor_phases.push_back(phase); }); + const RoundReport successor_round = runRegularRoundReclaiming(gc); + const auto decision = std::find_if(successor_phases.begin(), successor_phases.end(), [](const GcPhaseRecord & phase) + { + return phase.phase == "defer_decision"; + }); + ASSERT_NE(decision, successor_phases.end()); + ASSERT_FALSE(successor_round.deferred) + << "successor stream keys=" << successor_stream.keys.size() + << ", changed_shards=" << decision->metrics.at("changed_shards") + << ", dead_life_debris=" << decision->metrics.at("dead_life_debris"); + state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + EXPECT_FALSE(seal.ref_lives.contains(predecessor.incarnation)); + const auto successor_row = seal.ref_lives.find(successor.incarnation); + ASSERT_NE(successor_row, seal.ref_lives.end()); + EXPECT_EQ(successor_row->second.coverage.last_folded_ref_id.writer_epoch, writer_epoch); + EXPECT_EQ(successor_row->second.coverage.last_folded_ref_id.ref_sequence, 2u) + << "the successor starts at its own birth+publish stream, not the predecessor's cursor"; + for (const String & key : backend->touchedKeys()) + EXPECT_EQ(key.find("/_cleanup/"), String::npos) << key; +} + +/// Losing the response to an erase that committed must not strand the same resident writer runtime +/// behind its old removal-admission gate. A complete resolution read proves the exact old row absent, +/// so the same name can be born immediately under a fresh incarnation without inheriting coverage. +TEST(CASRefWriterNamespaceRemoval, CommitThenThrowEraseResolvesAndRebindsResidentRuntimeImmediately) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/removal-erase-lost-response"}; + Gc gc(store, UInt128{101}); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + + backend->catalog_fault_key = layout.refCatalogKey(); + backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; + EXPECT_NO_THROW((void)runRegularRoundReclaiming(gc)); + + const RefCatalog after_erase = CasRefCatalog::read(*backend, layout).catalog; + EXPECT_TRUE(std::none_of(after_erase.entries.begin(), after_erase.entries.end(), [&](const CatalogEntry & entry) + { + return entry.ns == ns && entry.incarnation == ready.predecessor.incarnation; + })); + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); + + EXPECT_NO_THROW(publishWithProductionBirth(store, ns, "successor")); + const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + EXPECT_NE(successor.incarnation, ready.predecessor.incarnation); + EXPECT_EQ(store->liveWriterEpoch(), ready.writer_epoch); + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); + + ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); + const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal( + backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + EXPECT_FALSE(seal.ref_lives.contains(ready.predecessor.incarnation)); + ASSERT_TRUE(seal.ref_lives.contains(successor.incarnation)); + EXPECT_EQ(seal.ref_lives.at(successor.incarnation).coverage.last_folded_ref_id, + (RefTxnId{ready.writer_epoch, 2})); +} + +/// If another actor wins the erase race by replacing the exact old row, `EntryChanged` still proves +/// the predecessor life dead and must invalidate its resident runtime. +TEST(CASRefWriterNamespaceRemoval, OtherWinnerReplacementInvalidatesExactPredecessorLife) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/removal-other-winner-replacement"}; + Gc gc(store, UInt128{102}); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + + const CatalogEntry replacement{ + .ns = ns, + .state = NsState::Live, + .incarnation = UInt128{0xfeed}, + .creator = std::nullopt}; + ASSERT_NE(replacement.incarnation, ready.predecessor.incarnation); + const NamespaceLifeId replacement_life = NamespaceLifeId::fromCatalogEntry(ns, replacement.incarnation); + ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(replacement_life), encodeRefCkpt(RefCkpt{ + .life_epoch = ready.writer_epoch, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + backend->catalog_fault_key = layout.refCatalogKey(); + backend->catalog_replacement_bytes = encodeRefCatalog(RefCatalog{.entries = {replacement}}); + backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::OtherWriterReplacement; + + EXPECT_NO_THROW((void)runRegularRoundReclaiming(gc)); + EXPECT_EQ(store->namespaceLife(ns), replacement_life); + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); +} + +/// Failure to read the catalog while resolving a lost erase response is not success. A later fresh +/// name lookup nevertheless observes the old exact row absent and reconciles the resident runtime. +TEST(CASRefWriterNamespaceRemoval, LaterNameLookupReconcilesAfterEraseResolutionReadFailure) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/removal-resolution-read-failure-lookup"}; + Gc gc(store, UInt128{103}); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + + backend->catalog_fault_key = layout.refCatalogKey(); + backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; + backend->catalog_resolution_get_fault_count = 1; + EXPECT_THROW((void)runRegularRoundReclaiming(gc), std::runtime_error); + EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + + std::optional successor; + EXPECT_NO_THROW(successor = store->namespaceLife(ns)); + ASSERT_TRUE(successor.has_value()); + EXPECT_NE(successor->incarnation, ready.predecessor.incarnation); + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); +} + +/// The normal post-LIST catalog cut is also a reconciliation point. It repairs a missed local +/// invalidation before any later writer touches the name. +TEST(CASRefWriterNamespaceRemoval, PostListCatalogCutReconcilesMissedEraseInvalidation) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.gc_fold_threshold = 1; + config.ref_table_cache_bytes = 0; + auto store = openPoolWithConfig(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/removal-resolution-read-failure-post-list"}; + Gc gc(store, UInt128{104}); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + + backend->catalog_fault_key = layout.refCatalogKey(); + backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; + backend->catalog_resolution_get_fault_count = 1; + EXPECT_THROW((void)runRegularRoundReclaiming(gc), std::runtime_error); + EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + + EXPECT_NO_THROW((void)runRegularRoundReclaiming(gc)); + EXPECT_NO_THROW(publishWithProductionBirth(store, ns, "successor")); + EXPECT_NE(catalogEntryOrThrow(*backend, layout, ns).incarnation, ready.predecessor.incarnation); + EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); +} + +/// =================================================================================== +/// Task 11: namespace birth / the recreation gate (spec §Namespace Birth) +/// =================================================================================== + +/// The writer assignment site may pin an already-`Live` catalog life, but recovering that empty life +/// performs no catalog or stream mutation. It must install the exact incarnation from the observed row. +TEST(CASRefWriterNamespaceBirth, ExistingLiveCatalogRowPinsExactLifeWithoutMutation) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/existing-live-assignment"}; + CasRefCatalog::casAdmitEntry(*backend, store->layout(), 1, CatalogEntry{ + .ns = ns, .state = NsState::Live, .incarnation = UInt128{41}}); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, store->layout(), ns, RefCkpt{ + .life_epoch = store->liveWriterEpoch(), + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + backend->resetCounts(); + const NamespaceLifeId life = store->namespaceLife(ns); + EXPECT_EQ(life.incarnation, UInt128{41}); + ASSERT_TRUE(store->refTableLifeForTest(ns)); + EXPECT_EQ(store->refTableLifeForTest(ns)->incarnation, UInt128{41}); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->casPutTotal(), 0u); +} + +/// A read of a never-born name may observe the catalog, but it must not allocate the local name slot +/// or a life runtime. Otherwise arbitrary read traffic can fill the cache with identity-less runtimes, +/// and a later birth has to mutate one of those objects into a different identity. +TEST(CASRefWriterNamespaceBirth, NeverBornReadAllocatesNoRuntime) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/never-born-read-no-runtime"}; + + ASSERT_EQ(store->refTablesCachedCountForTest(), 0u); + EXPECT_FALSE(store->resolveRef(ns, "missing").has_value()); + EXPECT_EQ(store->refTablesCachedCountForTest(), 0u) + << "catalog absence must be decided before a runtime is constructed"; + EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); +} + +/// A never-born namespace follows the ordinary catalog-first birth path. +TEST(CASRefWriterNamespaceBirth, BirthFromNeverBornUsesOrdinaryPath) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/virgin"}; + + EXPECT_NO_THROW(publishEmptyPart(store, ns, "first")); + EXPECT_TRUE(store->resolveRef(ns, "first").has_value()); +} + +/// Coverage gap (Task 13a): the "one op per ref name per batch" cut in `flushRefBatch` (the `seen_refs` +/// guard, `CASRefBatchScopeCuts`) had no test after the shard-lane `CasShardQueue.SameRefMutations +/// SplitAcrossFlushes` was retired. Two payload mutations of the SAME committed ref, made co-pending by +/// the pre-carve hook (mirrors `CompatibleMutationsShareOneCreate`), must NOT co-batch: per-request undo +/// validates each op against the pre-batch state, so the batch carries at most one op per ref name and +/// the two flush as two separate `_log` objects. +TEST(CASRefWriterAppendLane, SameRefMutationsSplitAcrossFlushes) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/samerefsplit"}; + publishEmptyPart(store, ns, "a"); + ASSERT_TRUE(store->resolveRef(ns, "a").has_value()); + + std::mutex m; + std::condition_variable cv; + bool entered = false; + store->setRefPreCarveHookForTest([&] + { + std::unique_lock lk(m); + if (entered) + return; /// only the leader's own first carve blocks; the second flush proceeds + entered = true; + cv.notify_all(); + cv.wait(lk, [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + + const uint64_t put_before = backend->putTotal(); + std::thread t_a([&] { store->updateRefPublishedAt(ns, "a", [](RefPublishedAtUpdate & r) { r.published_at_ms = 1; }); }); + { + std::unique_lock lk(m); + cv.wait(lk, [&] { return entered; }); + } + std::thread t_b([&] { store->updateRefPublishedAt(ns, "a", [](RefPublishedAtUpdate & r) { r.published_at_ms = 2; }); }); + while (store->refQueuePendingForTest(ns) < 2) + std::this_thread::yield(); + cv.notify_all(); + t_a.join(); + t_b.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_EQ(backend->putTotal(), put_before + 2) << "same-ref mutations must flush as two separate logs"; + /// Neither mutation was lost or corrupted -- the ref still resolves with one of the two writes. + const auto resolved = store->resolveRef(ns, "a"); + ASSERT_TRUE(resolved.has_value()); + EXPECT_TRUE(resolved->published_at_ms == 1 || resolved->published_at_ms == 2); +} + +/// =================================================================================== +/// rev.6 Task 8: the recovery seal (spec §recovery-seal / §seal-id / §seal-soundness). At an UNCLEAN +/// mount, `ensureRefTableRecovered` must close every dead epoch it discovers with an immediate +/// snapshot -- published at the UPPER BOUND of the dead-epoch region, `{liveWriterEpoch() - 1, +/// UINT64_MAX}` -- BEFORE the table is exposed as recovered, so no late predecessor PUT from any dead +/// epoch can ever surface to a cold fold or a fresh recovery. +/// =================================================================================== + +namespace +{ + +/// Seeds crash-style predecessor debris for the seal tests: two DEAD epochs (1 and 2) of durable logs +/// under `ns` -- epoch 1 births ref "a", epoch 2 adds ref "b" -- with no snapshot, and burns the +/// durable epoch counter to exactly 2 so a subsequent `Pool::open` allocates epoch 3 (both dead +/// epochs land strictly below the fresh writer's own, as `dead_region_nonempty` requires). +void seedSealFixtureDeadEpochs(Backend & backend, const Layout & layout, const RootNamespace & ns) +{ + allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); /// burns epoch 1 + allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); /// burns epoch 2 + + RefLogTxn birth; + birth.ns = ns.string(); + birth.txn_id = RefTxnId{1, 1}; + birth.ops = {namespaceBirthOp(), publishCommittedOps("a", manifestRef(1, 1, 1))[0], + publishCommittedOps("a", manifestRef(1, 1, 1))[1]}; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, birth); + + RefLogTxn mut; + mut.ns = ns.string(); + mut.txn_id = RefTxnId{2, 1}; + /// Sequence 1 of a new epoch names the seal that closed the one below -- `{1,2}`, the slot right + /// after epoch 1's only durable id, which is where the recovering mount's CAS-walk puts it. The seal + /// OBJECT is deliberately not seeded: this fixture's subject is a recovery that has to mint it. + mut.prev_epoch_seal = RefTxnId{1, 2}; + mut.ops = {publishCommittedOps("b", manifestRef(2, 1, 1))[0], + publishCommittedOps("b", manifestRef(2, 1, 1))[1]}; + DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, mut); + DB::Cas::tests::writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + /// The namespace was born in epoch 1 and only `{1,1}` is fronted initially. Recovery must mint + /// the missing required seal `{1,2}` before it may adopt the already durable `{2,1}` successor. + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); +} + +/// Plants a same-uuid, UNCLEAN (crash-style, no farewell) predecessor mount lease at `epoch`: a bare +/// `claimMount` followed by a GC fence-out -- mirrors `CASMountOpenWaits.FencedPriorReclaimsWithoutAnyWait`. A fenced +/// prior is an immediate certificate of death (`claimMountAwaitingExpiry` reclaims it on its FIRST +/// attempt, no observation polling), so a fake-clocked successor `Pool::open` above it becomes +/// unclean deterministically, without any real sleep. +void seedUncleanPredecessorMount(Backend & backend, const Layout & layout, uint64_t epoch) +{ + claimMount(backend, layout, "test", UInt128(1), epoch, /*now_ms=*/1000, /*ttl_ms=*/500); + fenceOutRefMount(backend, layout.mountKey("test")); +} + +/// The budget every seal test's successor `Pool::open` uses: a 500ms lease TTL needs a scaled-down +/// budget (RFC cas-s3-timeout-retry-control §required-timeout-model: attempt_timeout + safety_margin < +/// lease TTL) -- mirrors `CASMountOpenWaits.FencedPriorReclaimsWithoutAnyWait` exactly. +CasRequestBudget sealTestTinyBudget() +{ + return CasRequestBudget{ + .attempt_timeout_ms = 50, .operation_deadline_ms = kSingleAttemptDeadlineMs, .max_attempts = 1, + .lease_safety_margin_ms = 50}; +} + +} + +/// The `RefWriterRecoverySeal` suite is RETIRED with the sentinel seal it pinned, and the replacement is +/// `gtest_cas_ref_recovery_cas_walk.cpp` (`CASRefRecoveryCasWalk`), which covers the same duties against +/// the in-band mechanism: a dead epoch closed at `{E, T+1}`, a concurrent recoverer's seal adopted, a +/// straggler adopted and re-sealed at the new `T+1`, chained seals across burned epochs, and genesis. +/// +/// Three of its properties changed MEANING rather than mechanism, and a reader looking for them here +/// should know where they went: +/// +/// - "a clean boundary does not seal" is GONE as a rule. Sealing is now decided by `epoch < live_epoch` +/// alone, never by how the predecessor died: the seal is the chain link that makes a MISSING epoch +/// detectable, and a chain with holes in it wherever a mount shut down cleanly is not a chain. +/// - "a late log below the seal is invisible to recovery" is replaced by something stronger, and the +/// replacement is what makes the detector unnecessary: the seal occupies the ghost's own log key, so +/// a late PUT is REFUSED by the store instead of landing somewhere a reader must learn to ignore. +/// - the `sealed_from` inventory assertions are gone with the field; the chain link recovery installs +/// is `last_epoch_seal`, asserted in the new suite and in `CASRecoveryStreaming`'s inventory test. +/// +/// The fixtures above (`seedSealFixtureDeadEpochs`, `seedUncleanPredecessorMount`, `sealTestTinyBudget`) +/// are KEPT: the recovery-retry suite below drives the same dead-epoch shape. + +/// =================================================================================== +/// Layer 1 of the stuck-table-load fix: `ensureRefTableRecovered` retries a whole recovery attempt +/// after a TRANSIENT object-store NETWORK_ERROR (bounded by `recovery_retry_budget_ms`), instead of +/// failing the table's async load permanently. Non-transient errors and the terminal vanish-race +/// brake still fail fast. +/// =================================================================================== + +TEST(CASRefWriterRecoveryRetry, TransientSealFailureIsRetriedThenSucceeds) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_ok"}; + + seedSealFixtureDeadEpochs(*backend, layout, ns); + seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + + uint64_t fake_now = 1'000'000; + + PoolConfig config; + config.server_id = UInt128(1); + config.mount_lease_ttl_ms = std::chrono::milliseconds(500); + config.cas_request_budget = sealTestTinyBudget(); + config.cas_request_budget.recovery_retry_budget_ms = 120000; + config.cas_request_budget.recovery_retry_initial_backoff_ms = 1000; + config.cas_request_budget.recovery_retry_max_backoff_ms = 30000; + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.wait_sleep_fn = [](uint64_t) {}; + auto store = openPoolWithConfig(backend, config); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 3u); + + /// No-op backoff and a frozen clock: retries run until the transient faults are exhausted, and the + /// frozen clock keeps the mount fence alive across them (advancing it past the tiny lease TTL would + /// drop the fence and abort recovery -- exercising the fence path, which is the budget test's job). + store->setCasRetrySleepForTest([](uint64_t) {}); + + /// Fail the epoch seal's conditional create twice with a transient (timeout) error; the third + /// attempt lands. The seal is a LOG transaction at `{2,2}` -- the slot after the dead epoch's last + /// durable id -- because INV-2 closes an epoch in-band, at the key a straggler would have taken. + const RefTxnId seal_id{2, 2}; + backend->fault_key_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), seal_id); + backend->fault_count = 2; + + using ProfileEvents::global_counters; + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + + EXPECT_EQ(store->listRefs(ns).size(), 2u) << "recovery must succeed after retrying past the faults"; + + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before + 2); + /// TWO dead epochs (1 and 2) are closed by this walk, and a whole attempt is re-driven per transient + /// failure -- so the seals of the epochs a failed attempt already closed are ADOPTED on the retry + /// rather than minted again. Exactly two are minted in total. + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2); +} + +TEST(CASRefWriterRecoveryRetry, RecoveryDoesNotEnumerateItsStream) +{ + /// A recovery stream LIST used to be a transient failure leg. The checkpoint now names both the + /// base and frontier, so the same injected failures must remain untouched while recovery seals. + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_list"}; + + seedSealFixtureDeadEpochs(*backend, layout, ns); + seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + + uint64_t fake_now = 1'000'000; + + PoolConfig config; + config.server_id = UInt128(1); + config.mount_lease_ttl_ms = std::chrono::milliseconds(500); + config.cas_request_budget = sealTestTinyBudget(); + config.cas_request_budget.recovery_retry_budget_ms = 120000; + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.wait_sleep_fn = [](uint64_t) {}; + auto store = openPoolWithConfig(backend, config); + ASSERT_TRUE(store); + ASSERT_EQ(store->liveWriterEpoch(), 3u); + + store->setCasRetrySleepForTest([](uint64_t) {}); + + /// If recovery ever reintroduces a stream LIST, this injection turns the attempt into a retry and + /// consumes the counter. `namespaceFilesLifeIfReadable` reaches writer recovery without performing + /// the unrelated user-facing `listRefs` enumeration. + backend->list_fault_count = 2; + + using ProfileEvents::global_counters; + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + + ASSERT_TRUE(store->namespaceFilesLifeIfReadable(ns)); + EXPECT_EQ(backend->list_fault_count, 2); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2) + << "two dead epochs (1 and 2) are closed without enumerating their stream"; +} + +TEST(CASRefWriterRecoveryRetry, TransientFailureLongerThanBudgetPropagates) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_budget"}; + + seedSealFixtureDeadEpochs(*backend, layout, ns); + seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + + uint64_t fake_now = 1'000'000; + + PoolConfig config; + config.server_id = UInt128(1); + /// Lease TTL >> the recovery budget so the CLOCK-advancing backoff below trips the budget check, + /// not the mount fence -- this test specifically exercises the budget-exhaustion path. + config.mount_lease_ttl_ms = std::chrono::milliseconds(600000); + config.cas_request_budget = sealTestTinyBudget(); + config.cas_request_budget.recovery_retry_budget_ms = 5000; /// small, deterministic + config.cas_request_budget.recovery_retry_initial_backoff_ms = 1000; + config.cas_request_budget.recovery_retry_max_backoff_ms = 30000; + config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.wait_sleep_fn = [](uint64_t) {}; + auto store = openPoolWithConfig(backend, config); + ASSERT_TRUE(store); + store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + + /// The seal is an in-band LOG transaction at the slot after the dead epoch's last durable id, not a + /// snapshot at a synthetic id: epoch 1 closes at `{1,2}`, which is the FIRST write the walk attempts. + const RefTxnId seal_id{1, 2}; + backend->fault_key_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), seal_id); + backend->fault_count = 1000; /// never stops failing within the budget + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->listRefs(ns); }); +} + +TEST(CASRefWriterRecoveryRetry, NonNetworkErrorIsNotRetried) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_fatal"}; + + seedSealFixtureDeadEpochs(*backend, layout, ns); + seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + + PoolConfig config; + config.server_id = UInt128(1); + config.mount_lease_ttl_ms = std::chrono::milliseconds(500); + config.cas_request_budget = sealTestTinyBudget(); + config.wait_sleep_fn = [](uint64_t) {}; + auto store = openPoolWithConfig(backend, config); + ASSERT_TRUE(store); + + size_t sleep_calls = 0; + store->setCasRetrySleepForTest([&sleep_calls](uint64_t) { ++sleep_calls; }); + + /// A foreign writer lands DIFFERENT valid bytes at the seal key; resolve-before-reissue then throws + /// CORRUPTED_DATA (a real cross-process seal conflict), which must NOT be retried. + const RefTxnId seal_id{1, 2}; + backend->corrupt_key_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), seal_id); + backend->corrupt_count = 1; + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); + EXPECT_EQ(backend->corrupt_count, 0) + << "the test must reach the injected foreign seal conflict, not fail on fixture validation"; + EXPECT_EQ(sleep_calls, 0u) << "a non-transient error must fail fast with zero backoff sleeps"; +} + +TEST(CASRefWriterRecoveryRetry, VanishBrakeStaysTerminalNotRetried) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_vanish"}; + const ManifestRef ma = manifestRef(1, 1, 1); + + /// Stage B (Task 4-C): pin `ns` to the sentinel before the raw snapshot below -- `store->listRefs` + /// further down is a real production read that would otherwise mint a fresh RANDOM incarnation for + /// this unadmitted namespace instead of adopting the sentinel the raw fixture writes at. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + const RefTxnId snap_x{1, 10}; + std::vector base_ops{namespaceBirthOp()}; + const auto publish_a = publishCommittedOps("a", ma); + base_ops.insert(base_ops.end(), publish_a.begin(), publish_a.end()); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), snap_x, std::move(base_ops), std::nullopt}); + writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), snap_x, {committedRow("a", ma)})); + writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = snap_x, + .checkpoint_snapshot_id = snap_x, + .last_epoch_seal = std::nullopt, + }); + + auto store = openPool(backend); + + size_t sleep_calls = 0; + store->setCasRetrySleepForTest([&sleep_calls](uint64_t) { ++sleep_calls; }); + + /// A checkpoint-named snapshot belongs to the caller's immutable authority cut. If that exact + /// object is absent, recovery must report corruption immediately; it must neither reinterpret a + /// transient disappearance as a new authority cut nor enter the outer transient-retry loop. + const String vkey = layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), snap_x); + backend->vanish_once_keys.insert(vkey); + + using ProfileEvents::global_counters; + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); + + EXPECT_FALSE(backend->vanish_once_keys.contains(vkey)) + << "the test must reach the checkpoint-named snapshot GET, not fail on earlier fixture validation"; + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before) + << "missing immutable checkpoint authority is terminal; the outer transient-retry loop must NOT re-drive it"; + EXPECT_EQ(sleep_calls, 0u) << "no backoff sleep for missing immutable checkpoint authority"; +} + +TEST(CASRefWriterRecoveryRetry, ThrowingBackoffSleepDoesNotWedgeRecovery) +{ + /// If the backoff sleep itself throws (e.g. a clock syscall failure), the retry loop must re-acquire + /// state_mutex before unwinding so the SCOPE_EXIT that clears `recovery_in_progress` runs LOCKED -- + /// otherwise a later touch would hang forever on the never-cleared flag. This drives that path and + /// then proves a second touch can still recover (the lane is not wedged). + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/retry_sleep_throw"}; + + seedSealFixtureDeadEpochs(*backend, layout, ns); + seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + + PoolConfig config; + config.server_id = UInt128(1); + config.mount_lease_ttl_ms = std::chrono::milliseconds(500); + config.cas_request_budget = sealTestTinyBudget(); + config.cas_request_budget.recovery_retry_budget_ms = 120000; + config.wait_sleep_fn = [](uint64_t) {}; + auto store = openPoolWithConfig(backend, config); + ASSERT_TRUE(store); + + /// First touch: the seal PUT fails transiently -> the loop enters backoff -> the sleep THROWS. + bool sleep_should_throw = true; + store->setCasRetrySleepForTest([&sleep_should_throw](uint64_t) + { + if (sleep_should_throw) + throw std::runtime_error("injected backoff-sleep failure"); + }); + const RefTxnId seal_id{1, 2}; + backend->fault_key_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), seal_id); + backend->fault_count = 1; + + EXPECT_ANY_THROW(store->listRefs(ns)); /// the sleep failure propagates + + /// The lane must NOT be wedged: with the fault now spent and the sleep no longer throwing, a second + /// touch recovers cleanly. If recovery_in_progress had leaked (SCOPE_EXIT run unlocked / not run), a + /// concurrent-safe second recovery would deadlock or mis-behave. + sleep_should_throw = false; + EXPECT_EQ(store->listRefs(ns).size(), 2u) << "a second touch must recover; the retry lane is not wedged"; +} + +/// =================================================================================== +/// Task 16: `hasAnyRefWithPrefix` -- pure existence probe, same recovery preamble as `listRefs` but +/// without materializing the full ref map (an early-exit scan). +/// =================================================================================== + +TEST(CASRefWriterListRefs, HasAnyRefWithPrefixMatchesListRefsEmptiness) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/prefix_probe"}; + const RootNamespace empty_ns{"srv1/prefix_probe_empty"}; + + EXPECT_FALSE(store->hasAnyRefWithPrefix(empty_ns, "")) << "a never-touched namespace has no refs"; + + publishEmptyPart(store, ns, "all_1_1_0"); + publishEmptyPart(store, ns, "detached-x"); + + EXPECT_TRUE(store->hasAnyRefWithPrefix(ns, "")) << "empty prefix means \"any ref at all\""; + EXPECT_TRUE(store->hasAnyRefWithPrefix(ns, "detached-")); + EXPECT_FALSE(store->hasAnyRefWithPrefix(ns, "moving-")) << "no ref carries this prefix"; + + store->dropNamespace(ns); + EXPECT_FALSE(store->hasAnyRefWithPrefix(ns, "")) << "a tombstoned namespace has no committed refs"; +} diff --git a/src/Disks/tests/gtest_cas_repoint.cpp b/src/Disks/tests/gtest_cas_repoint.cpp new file mode 100644 index 000000000000..9044a8eda895 --- /dev/null +++ b/src/Disks/tests/gtest_cas_repoint.cpp @@ -0,0 +1,107 @@ +#include +#include +#include +#include +#include + +/// Task 3 (all-tree-part-files plan, 2026-07-15): `CachedPartFolderAccess::repointRef` -- the audited +/// primitive a standalone write/remove on an already-COMMITTED part must go through once the mutable +/// per-part file set is empty. It republishes +/// the whole manifest with the new entry set, riding `PartWriteTxn::promote`'s `allow_repoint` mode (Task 2). + +namespace ProfileEvents +{ +extern const Event CASRefRepoint; +} + +using namespace DB::Cas; + +namespace +{ + +ManifestEntry inlineEntry(const String & path, const String & bytes) +{ + ManifestEntry e; + e.path = path; + e.placement = EntryPlacement::Inline; + e.ref = BlobRef{BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(DB::Cas::tests::u128Of(bytes))}; + + e.blob_size = bytes.size(); + e.inline_bytes = bytes; + return e; +} + +/// Publish `entries` as committed ref `ns/ref` through the real writer protocol. +ManifestId publishPart(const PoolPtr & store, const RootNamespace & ns, const String & ref, + std::vector entries) +{ + auto build = store->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/" + ref, + .intended_namespace = ns, .op = ProvenanceOp::Insert}); + const ManifestId id = build->stageManifest(entries); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +} + +/// Byte-equal candidate: the exact same entries republished onto an already-committed ref must be a +/// ZERO-mutation no-op -- no fresh manifest staged, no ref-log record appended, no `RefRepoint` event. +/// `stageManifest` mints a non-content-derived `ManifestRef` AND durably PUTs the body on every call +/// (CasPartWriteTxn.cpp), so this can only hold if the no-op check compares candidate `entries` directly +/// against the currently-committed manifest's DECODED entries -- never by staging first (the same +/// structural comparison `republishRef`'s BUG 1c fix uses). +TEST(CASRepoint, ByteEqualIsNoOp) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const RootNamespace ns{"srv/t1"}; + DB::Cas::CachedPartFolderAccess access(store); + const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const DB::Cas::PartRefKey key{ns, "part_1"}; + + backend->resetCounts(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const DB::Cas::CommitOutcome oc = access.repointRef(key, {inlineEntry("checksums.txt", "cs")}, ProvenanceOp::Other); + EXPECT_FALSE(oc.created); + EXPECT_EQ(oc.manifest_ref, id.ref) << "the byte-equal outcome must name the manifest ALREADY committed, unchanged"; + + EXPECT_EQ(backend->putTotal(), 0u) << "byte-equal repoint must perform ZERO pool mutations"; + EXPECT_EQ(store->resolveRef(ns, "part_1")->manifest_id, id) + << "the committed manifest identity must be untouched"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before); +} + +/// A genuinely different entry set on an already-committed ref republishes the manifest: the returned +/// `CommitOutcome` names a FRESH manifest (`created` still false -- the ref was already committed), +/// the new content resolves, and the repoint is loud (ProfileEvent + the ref's cached view erased so a +/// subsequent read serves the new manifest, not a stale retained one). +TEST(CASRepoint, AddFileRepoints) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const RootNamespace ns{"srv/t1"}; + DB::Cas::CachedPartFolderAccess access( + store, {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, + .explain_enabled = false, .validate = {}}); + const auto id_before = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); + const DB::Cas::PartRefKey key{ns, "part_1"}; + /// Warm the retained view so the erase-on-success cache discipline is actually exercised. + ASSERT_NE(access.getView(key, DB::Cas::Freshness::CachedForLoad), nullptr); + + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const std::vector new_entries{inlineEntry("checksums.txt", "cs"), inlineEntry("metadata_version.txt", "7")}; + const DB::Cas::CommitOutcome oc = access.repointRef(key, new_entries, ProvenanceOp::Other); + EXPECT_FALSE(oc.created); + EXPECT_NE(oc.manifest_ref, id_before.ref); + + const auto resolved = store->resolveRef(ns, "part_1"); + ASSERT_TRUE(resolved.has_value()); + EXPECT_NE(resolved->manifest_id, id_before) << "a genuine content change must mint a fresh manifest"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + + /// The view a caller reads next must reflect the new file, not a stale retained one. + auto view = access.getView(key, DB::Cas::Freshness::CachedForLoad); + ASSERT_NE(view, nullptr); + EXPECT_TRUE(view->hasFile("metadata_version.txt")); +} diff --git a/src/Disks/tests/gtest_cas_request_control.cpp b/src/Disks/tests/gtest_cas_request_control.cpp new file mode 100644 index 000000000000..ad7801634f39 --- /dev/null +++ b/src/Disks/tests/gtest_cas_request_control.cpp @@ -0,0 +1,1494 @@ +#include + +#include "config.h" + +#include +#include + +#include + +#if USE_AWS_S3 +#include +#include +#include +#include +#include +#endif + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; + extern const int ABORTED; +} + +namespace ProfileEvents +{ + extern const Event CASConditionalWriteAttempts; + extern const Event CASConditionalWriteCommitted; + extern const Event CASConditionalWriteDefiniteFailure; + extern const Event CASConditionalWriteUnresolved; +} + +#if USE_AWS_S3 +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int BAD_ARGUMENTS; + extern const int LOGICAL_ERROR; + extern const int UNKNOWN_EXCEPTION; +} +#endif + +/// The success path (buf.finalize() returned without throwing) is always Committed. No exception +/// object is needed — the caller distinguishes success from failure before calling either overload. +TEST(CASRequestControl, SuccessIsAlwaysCommitted) +{ + EXPECT_EQ(classifyConditionalWriteResult(), CasWriteOutcome::Committed); +} + +/// Fix #37 phase 2: the retry-later throw must be NETWORK_ERROR, never ABORTED -- ABORTED is silently +/// swallowed by ReplicatedMergeMutateTaskBase (no backoff, no last_exception), which is exactly the +/// defect this fix closes. +TEST(CASWriteRetryLater, ThrowsNetworkErrorNotAborted) +{ + bool threw = false; + try + { + throwCasWriteRetryLater("test cause"); + FAIL() << "throwCasWriteRetryLater must always throw"; + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(e.code(), DB::ErrorCodes::ABORTED); + EXPECT_NE(e.message().find("test cause"), String::npos) << e.message(); + EXPECT_NE(e.message().find("retrying later"), String::npos) << e.message(); + } + EXPECT_TRUE(threw); +} + +/// The exception_ptr twin (for call sites that fail a pending future/promise rather than throw +/// directly, e.g. CasRefLedger's queued-append completion paths) must carry the SAME classification. +TEST(CASWriteRetryLater, ExceptionPtrVariantCarriesSameClassification) +{ + const std::exception_ptr eptr = makeCasWriteRetryLaterExceptionPtr("another cause"); + bool threw = false; + try + { + std::rethrow_exception(eptr); + FAIL() << "expected a thrown exception"; + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + EXPECT_NE(e.message().find("another cause"), String::npos) << e.message(); + } + EXPECT_TRUE(threw); +} + +#if USE_AWS_S3 + +/// One row per RFC cas-s3-timeout-retry-control §operation-classes classification. PreconditionFailed +/// is NEVER DefiniteFailure — it means the key exists, not that the request was rejected — and every +/// unrecognized/ambiguous error also falls to Unresolved, never to a false DefiniteFailure. +TEST(CASRequestControl, ClassifiesPreconditionFailedAsUnresolved) +{ + DB::S3Exception e("412 from backend", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); +} + +TEST(CASRequestControl, ClassifiesTimeoutAsUnresolved) +{ + Poco::TimeoutException e("simulated client-side receive timeout"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); +} + +TEST(CASRequestControl, ClassifiesConnectionResetAsUnresolved) +{ + Poco::Net::ConnectionResetException e("simulated connection reset"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); +} + +TEST(CASRequestControl, Classifies5xxAsUnresolved) +{ + DB::S3Exception e("simulated internal error", Aws::S3::S3Errors::INTERNAL_FAILURE, "InternalError"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); + /// SlowDown / ServiceUnavailable are also 5xx-class and equally Unresolved. + DB::S3Exception slow_down("simulated throttle", Aws::S3::S3Errors::SLOW_DOWN, "SlowDown"); + EXPECT_EQ(classifyConditionalWriteResult(slow_down), CasWriteOutcome::Unresolved); +} + +TEST(CASRequestControl, ClassifiesMalformedRequestAsDefiniteFailure) +{ + DB::S3Exception e("bad xml", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); + /// The modeled-enum path (no canonical name attached) must classify identically. + DB::S3Exception by_code("bad argument", Aws::S3::S3Errors::INVALID_REQUEST); + EXPECT_EQ(classifyConditionalWriteResult(by_code), CasWriteOutcome::DefiniteFailure); +} + +TEST(CASRequestControl, ClassifiesEntityTooLargeAsDefiniteFailure) +{ + DB::S3Exception e("body exceeds the maximum object size", Aws::S3::S3Errors::UNKNOWN, "EntityTooLarge"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); +} + +TEST(CASRequestControl, ClassifiesAccessDeniedAsDefiniteFailure) +{ + DB::S3Exception e("simulated 403", Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied"); + EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); + /// The modeled-enum path (no canonical name attached) must classify identically. + DB::S3Exception by_code("simulated 403, no name", Aws::S3::S3Errors::ACCESS_DENIED); + EXPECT_EQ(classifyConditionalWriteResult(by_code), CasWriteOutcome::DefiniteFailure); +} + +/// Anything the classifier does not recognize (an unmodeled/unnamed S3 error, or an entirely +/// unrelated exception type) must fail toward Unresolved — never toward a false DefiniteFailure or a +/// false Committed (RFC §resolve-before-reissuing: ambiguity always resolves toward "resolve before +/// reissuing"). +TEST(CASRequestControl, UnrecognizedErrorsFailSafeToUnresolved) +{ + DB::S3Exception unknown_named("weird service error", Aws::S3::S3Errors::UNKNOWN, "SomeFutureErrorCode"); + EXPECT_EQ(classifyConditionalWriteResult(unknown_named), CasWriteOutcome::Unresolved); + + /// UNKNOWN_EXCEPTION (not LOGICAL_ERROR): any arbitrary non-S3 exception type works here -- the + /// point is that the classifier doesn't recognize it, not which specific code it carries. + /// LOGICAL_ERROR would abort the whole process under debug/sanitizer builds merely by being + /// constructed (Exception's constructor calls handle_error_code unconditionally). + DB::Exception unrelated(DB::ErrorCodes::UNKNOWN_EXCEPTION, "not an S3 error at all"); + EXPECT_EQ(classifyConditionalWriteResult(unrelated), CasWriteOutcome::Unresolved); +} + +/// recordConditionalWriteAttemptStarted / recordConditionalWriteOutcome bump the per-class counters +/// (RFC §observability): attempts, and exactly one of Committed/DefiniteFailure/Unresolved per call. +TEST(CASRequestControl, CountersHookupIncrementsPerClass) +{ + using ProfileEvents::global_counters; + const auto attempts_before = global_counters[ProfileEvents::CASConditionalWriteAttempts].load(); + const auto committed_before = global_counters[ProfileEvents::CASConditionalWriteCommitted].load(); + const auto definite_before = global_counters[ProfileEvents::CASConditionalWriteDefiniteFailure].load(); + const auto unresolved_before = global_counters[ProfileEvents::CASConditionalWriteUnresolved].load(); + + recordConditionalWriteAttemptStarted(); + recordConditionalWriteOutcome(CasWriteOutcome::Committed); + recordConditionalWriteAttemptStarted(); + recordConditionalWriteOutcome(CasWriteOutcome::DefiniteFailure); + recordConditionalWriteAttemptStarted(); + recordConditionalWriteOutcome(CasWriteOutcome::Unresolved); + +#if !WITH_COVERAGE + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteAttempts].load() - attempts_before, 3u); + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteCommitted].load() - committed_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteDefiniteFailure].load() - definite_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteUnresolved].load() - unresolved_before, 1u); +#else + (void)attempts_before; (void)committed_before; (void)definite_before; (void)unresolved_before; +#endif +} + +/// Wiring smoke test: a real conditional write through ObjectStorageBackend (Native mode) counts one +/// attempt and one Committed outcome via the SAME instrumented call site nativeConditionalPut uses — +/// see finalizeConditionalWriteInstrumented in CasObjectStorageBackend.cpp. +TEST(CASRequestControl, NativeConditionalPutCountsOneAttemptAndCommitted) +{ + using ProfileEvents::global_counters; + const auto attempts_before = global_counters[ProfileEvents::CASConditionalWriteAttempts].load(); + const auto committed_before = global_counters[ProfileEvents::CASConditionalWriteCommitted].load(); + + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + EXPECT_EQ(b->putIfAbsent("p/rc/one", "v1").outcome, PutOutcome::Done); + +#if !WITH_COVERAGE + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteAttempts].load() - attempts_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteCommitted].load() - committed_before, 1u); +#else + (void)attempts_before; (void)committed_before; +#endif +} + +/// Mechanism property (RFC §disable-transparent-conditional-write-retries), tested at the layer +/// actually reachable from a unit-test binary: NO live/fake S3 endpoint is available here (the Native +/// conditional-write path is exercised end-to-end only at M-W against RustFS — see the HONEST NOTE in +/// CasObjectStorageBackend.cpp), so driving a real socket-level retry against a real client is not +/// reachable from this binary. What IS reachable and asserted here: every Native conditional write +/// selects the SingleAttempt object-storage retry profile, and a non-S3 backend such as +/// LocalObjectStorage reports it as UNSUPPORTED via IObjectStorage::supportsRetryProfile — the property +/// checkConditionalWriteSingleAttemptSupport's fail-closed mount-time gate relies on. +TEST(CASRequestControl, SingleAttemptProfileRequestedAndLocalBackendRejected) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + const auto ws = b->conditionalWriteSettingsForTest(); + EXPECT_EQ(ws.object_storage_retry_profile, DB::ObjectStorageRetryProfile::SingleAttempt); + /// LocalObjectStorage does not implement the profile — the capability check must say no. + EXPECT_FALSE(DB::Cas::tests::makeLocalObjectStorageForTest()->supportsRetryProfile(DB::ObjectStorageRetryProfile::SingleAttempt)); +} + +/// The SECOND retry-affecting layer above the S3 client (review finding): WriteBufferFromS3's OWN +/// makeSinglepartUpload/completeMultipartUpload retry loop reissues the identical conditional request +/// on a NO_SUCH_KEY response, driven by S3RequestSetting::max_unexpected_write_error_retries (default +/// 4) — a client-level override alone does not bound it (see WriteSettings:: +/// s3_max_unexpected_write_error_retries_override). Asserted at the reachable seam: no live/fake S3 +/// endpoint exists in this binary to drive the retry loop itself, so this proves the settings +/// plumbing conditionalWriteSettings() -> WriteSettings produces the override value that +/// S3ObjectStorage::writeObject then applies to request_settings — NOT a real single-attempt +/// assertion against a live wire attempt. +TEST(CASRequestControl, ConditionalWriteSettingsForceSingleUnexpectedWriteErrorRetry) +{ + auto b = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + const auto ws = b->conditionalWriteSettingsForTest(); + EXPECT_EQ(ws.s3_max_unexpected_write_error_retries_override, 1u); +} + +/// ================================================================================================ +/// Task 5: CasRequestController — retry controller (deadlines, fence gating, exact-key resolution) +/// ================================================================================================ + +namespace +{ + +/// A per-call scripted Backend for CasRequestController tests: `putIfAbsent` optionally throws a +/// caller-supplied exception (models one classified HTTP-attempt outcome) or returns a forced +/// `PutOutcome` directly (models a `PreconditionFailed` observed WITHOUT an exception); with neither +/// set it delegates to the real in-memory conditional-write semantics. `get` optionally returns a +/// forced result, independent of what `putIfAbsent` actually did, so a test can drive exact-key +/// resolution (identical / different / absent) without the scripted put and the resolve GET needing to +/// agree on a shared, real backing store. +class ScriptedControllerBackend : public InMemoryBackend +{ +public: + std::function put_thrower; + std::optional put_forced_outcome; + std::atomic put_attempts{0}; + + std::function put_overwrite_thrower; + std::function put_overwrite_handler; + std::optional put_overwrite_forced_outcome; + std::atomic put_overwrite_attempts{0}; + + std::function(const String &, Range)> get_handler; + std::atomic get_attempts{0}; + bool get_overridden = false; + std::optional get_override_value; /// meaningful only when get_overridden + + void setGetOverride(std::optional value) + { + get_overridden = true; + get_override_value = std::move(value); + } + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + ++put_attempts; + if (put_thrower) + put_thrower(); + if (put_forced_outcome) + return {*put_forced_outcome, {}}; + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } + + PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + { + ++put_overwrite_attempts; + if (put_overwrite_handler) + return put_overwrite_handler(key, bytes, expected, meta); + if (put_overwrite_thrower) + put_overwrite_thrower(); + if (put_overwrite_forced_outcome) + return {*put_overwrite_forced_outcome, {}}; + return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + + std::optional get(const String & key, Range range) override + { + ++get_attempts; + if (get_handler) + return get_handler(key, range); + if (get_overridden) + return get_override_value; + return InMemoryBackend::get(key, range); + } +}; + +GetResult resultWithBytes(const String & bytes) +{ + return GetResult{.bytes = bytes, .token = Token{"t", TokenType::Emulated}, .attributes = {}}; +} + +CasOverwriteOperationContext overwriteContext( + uint64_t absolute_deadline_ms, + CasOverwriteDeadlineSource deadline_source = CasOverwriteDeadlineSource::RequestBudget, + std::function stop_cause = {}, + std::function wait_before_retry = {}, + std::function observe = {}) +{ + return CasOverwriteOperationContext{ + .absolute_deadline_ms = absolute_deadline_ms, + .deadline_source = deadline_source, + .stop_cause = stop_cause ? std::move(stop_cause) : [] { return CasOverwriteStopCause::Continue; }, + .wait_before_retry = wait_before_retry ? std::move(wait_before_retry) : [](uint64_t) { return true; }, + .observe = observe ? std::move(observe) : [](const CasOverwriteProgress &) {}, + }; +} + +void expectOverwriteDiagnostics( + const CasOverwriteResult & result, + uint32_t attempts_sent, + bool resolved_by_get, + CasUnresolvedReason unresolved_reason, + CasOverwriteDeadlineSource deadline_source, + CasOverwriteStopCause stop_cause) +{ + EXPECT_EQ(result.diagnostics.attempts_sent, attempts_sent); + EXPECT_EQ(result.diagnostics.resolved_by_get, resolved_by_get); + EXPECT_EQ(result.diagnostics.unresolved_reason, unresolved_reason); + EXPECT_EQ(result.diagnostics.deadline_source, deadline_source); + EXPECT_EQ(result.diagnostics.stop_cause, stop_cause); +} + +} + +TEST(CASRequestController, UncertainResolvesIdenticalAsCommitted) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(resultWithBytes("payload")); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::Committed); + EXPECT_EQ(backend->put_attempts.load(), 1u); +} + +TEST(CASRequestController, UncertainResolvesDifferentThrowsCorruption) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(resultWithBytes("someone-elses-bytes")); + + CasRequestController controller(backend, CasRequestBudget{}); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + { + controller.putIfAbsentControlled("k", "payload", [] { return true; }); + }); +} + +/// GET-absent NEVER yields DefiniteFailure (spec §writer-side-linearization): the SAME (key, bytes) is +/// retried up to `max_attempts`, and only THEN does the call give up with Unresolved. +TEST(CASRequestController, UncertainResolvesAbsentRetriesSameKeyWithinBudget) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); /// absent on every resolve + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.retry_initial_backoff_ms = 0; /// backoff behavior is pinned by its own tests below + CasRequestController controller(backend, budget); + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 3u); /// every attempt targeted the SAME key/bytes +} + +/// The operation deadline — not just the attempt-count budget — cuts a retry loop short: a fake clock +/// advances by a fixed step per now_ms() call (no sleeps), and max_attempts is generous enough that only +/// the deadline check can be what stops the loop. +TEST(CASRequestController, OperationDeadlineExhaustionReturnsUnresolvedBeforeMaxAttempts) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); /// absent on every resolve + + uint64_t clock = 0; + auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 200; return t; }; + + CasRequestBudget budget; + budget.max_attempts = 10; + budget.attempt_timeout_ms = 50; + budget.operation_deadline_ms = 450; + budget.retry_initial_backoff_ms = 0; /// isolate the deadline check from the backoff's own deadline guard + CasRequestController controller(backend, budget, now_ms); + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 2u); /// cut off well before the 10-attempt budget +} + +/// WHY `attempt_timeout_ms == operation_deadline_ms` IS REJECTED AT STARTUP, demonstrated on the +/// mechanism itself before the rejection is asserted below. +/// +/// The deadline is captured as `now + operation_deadline_ms` and the pre-send gate asks +/// `now + attempt_timeout_ms > deadline`. Equal values collapse that to `now_2 > now_1`: ONE elapsed +/// millisecond between the capture and the gate refuses the whole operation with NOTHING SENT. That is +/// not a bounded operation, it is a coin flip on the scheduler -- "mostly works, occasionally refuses +/// having sent nothing", which is the flakiness class validation exists to prevent. Single-attempt +/// semantics is what `max_attempts = 1` is for; the equality contributes only the race. +/// +/// The controller is constructed DIRECTLY here, bypassing `validateCasRequestBudget`, because the +/// point is to show the behaviour the validator now forbids. Three tests were flaky on exactly this +/// before it was forbidden: `8f9e63c7a19`'s sweep-interruption test, +/// `CASRefInstallSafety.UncertainPrecommitKeepsItsCleanupOwnerAndItsBody`, and +/// `CASRefWriterAppendLane.WedgedLaneBlocksSameTableWhileOtherTableProceeds`. +TEST(CASRequestController, EqualAttemptTimeoutAndDeadlineWouldRefuseAfterASingleTick) +{ + auto backend = std::make_shared(); + + /// The smallest possible passage of time: one millisecond per clock read. + uint64_t clock = 0; + auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1; return t; }; + + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 100; + CasRequestController razor(backend, budget, now_ms); + EXPECT_EQ(razor.putIfAbsentControlled("k", "payload", [] { return true; }), CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 0u) + << "the refusal came from the clock, not from the backend: nothing was sent at all"; + + /// STRICTLY LESS -- the shape the validator now requires -- sends the request over the SAME one-tick + /// clock. So what the inequality buys is the request actually happening, not merely a bigger number. + clock = 0; + budget.operation_deadline_ms = 5000; + CasRequestController wide(backend, budget, now_ms); + EXPECT_EQ(wide.putIfAbsentControlled("k", "payload", [] { return true; }), CasWriteOutcome::Committed); + EXPECT_EQ(backend->put_attempts.load(), 1u); +} + +/// And the same equality is refused at startup, so no budget can reach the controller in that shape. +/// The boundary is asserted from BOTH sides: equality throws, one millisecond more is accepted. +TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutEqualToOperationDeadline) +{ + CasRequestBudget budget; + budget.attempt_timeout_ms = 5000; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 1000; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); + + budget.operation_deadline_ms = 5001; + EXPECT_NO_THROW(validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000)) + << "one millisecond of headroom is the whole requirement -- the rule is strictness, not size"; +} + +TEST(CASRequestController, OverwriteAmbiguousResolvesIntendedBytesAsCommitted) +{ + auto backend = std::make_shared(); + bool first_attempt = true; + backend->put_overwrite_thrower = [&first_attempt] + { + if (first_attempt) + { + first_attempt = false; + throw Poco::TimeoutException("scripted: ambiguous"); + } + }; + backend->setGetOverride(resultWithBytes("new-payload")); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, [] { return true; }); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(result.token, (Token{"t", TokenType::Emulated})); +} + +TEST(CASRequestController, OverwriteAmbiguousResolvesExpectedTokenAndRetriesWithinBudget) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget); + const auto result = controller.putOverwriteControlled("k", "new-payload", expected, [] { return true; }); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 3u); +} + +TEST(CASRequestController, OverwriteAmbiguousResolvesDifferentTokenAndBytesAsConflict) +{ + auto backend = std::make_shared(); + bool first_attempt = true; + backend->put_overwrite_thrower = [&first_attempt] + { + if (first_attempt) + { + first_attempt = false; + throw Poco::TimeoutException("scripted: ambiguous"); + } + }; + backend->setGetOverride(GetResult{ + .bytes = "someone-elses-payload", .token = Token{"other", TokenType::Emulated}, .attributes = {}}); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, [] { return true; }); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Conflict); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); +} + +TEST(CASRequestController, OverwriteOperationDeadlineExhaustionReturnsUnresolvedBeforeMaxAttempts) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + + uint64_t clock = 0; + auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 200; return t; }; + + CasRequestBudget budget; + budget.max_attempts = 10; + budget.attempt_timeout_ms = 50; + budget.operation_deadline_ms = 450; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget, now_ms); + const auto result = controller.putOverwriteControlled("k", "new-payload", expected, [] { return true; }); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); +} + +TEST(CASRequestController, AbsoluteDeadlineCannotBeReanchoredAfterPreemption) +{ + auto backend = std::make_shared(); + uint64_t clock = 60; + + CasRequestBudget budget; + budget.attempt_timeout_ms = 50; + budget.operation_deadline_ms = 10000; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); + expectOverwriteDiagnostics( + result, + 0, + false, + CasUnresolvedReason::NoAttemptSent, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, MaxAttemptsOneStillResolvesLostResponseByGet) +{ + auto backend = std::make_shared(); + const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); + ASSERT_EQ(predecessor.outcome, PutOutcome::Done); + + Token landed_token; + auto * backend_ptr = backend.get(); + backend->put_overwrite_handler = [backend_ptr, &landed_token]( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult + { + const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); + landed_token = landed.token; + throw Poco::TimeoutException("scripted: response lost after overwrite landed"); + }; + std::vector> progress; + + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + auto context = overwriteContext( + 1000, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + {}, + {}, + [&progress](const CasOverwriteProgress & event) { progress.emplace_back(event.kind, event.attempt_no); }); + const auto result = controller.putOverwriteControlled("k", "new-payload", predecessor.token, context); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); + EXPECT_EQ(result.token, landed_token); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + true, + CasUnresolvedReason::NotUnresolved, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); + const std::vector> expected_progress{ + {CasOverwriteProgressKind::PutStarted, 1}, + {CasOverwriteProgressKind::BecameAmbiguous, 1}, + {CasOverwriteProgressKind::ResolveStarted, 1}, + {CasOverwriteProgressKind::ResolvedByGet, 1}, + }; + EXPECT_EQ(progress, expected_progress); +} + +TEST(CASRequestController, StopBeforeFirstPutReportsExactCause) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}, [] { return static_cast(0); }); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext( + 10000, + CasOverwriteDeadlineSource::RequestBudget, + [] { return CasOverwriteStopCause::Cancelled; })); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); + EXPECT_EQ(backend->get_attempts.load(), 0u); + expectOverwriteDiagnostics( + result, + 0, + false, + CasUnresolvedReason::NoAttemptSent, + CasOverwriteDeadlineSource::RequestBudget, + CasOverwriteStopCause::Cancelled); +} + +TEST(CASRequestController, StopAfterPutSuppressesResolveAndReportsMidWay) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + uint32_t stop_samples = 0; + auto stop_cause = [&stop_samples] + { + ++stop_samples; + return stop_samples == 1 ? CasOverwriteStopCause::Continue : CasOverwriteStopCause::Cancelled; + }; + + CasRequestBudget budget; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(10000, CasOverwriteDeadlineSource::RequestBudget, stop_cause)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 0u); + expectOverwriteDiagnostics( + result, + 1, + false, + CasUnresolvedReason::FenceLostMidWay, + CasOverwriteDeadlineSource::RequestBudget, + CasOverwriteStopCause::Cancelled); +} + +TEST(CASRequestController, StopAfterResolvedCommitReportsPostWrite) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(resultWithBytes("new-payload")); + uint32_t stop_samples = 0; + auto stop_cause = [&stop_samples] + { + ++stop_samples; + return stop_samples < 3 ? CasOverwriteStopCause::Continue : CasOverwriteStopCause::FenceOrLifecycleLost; + }; + + CasRequestController controller(backend, CasRequestBudget{}, [] { return static_cast(0); }); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(10000, CasOverwriteDeadlineSource::RequestBudget, stop_cause)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + true, + CasUnresolvedReason::FenceLostPostWrite, + CasOverwriteDeadlineSource::RequestBudget, + CasOverwriteStopCause::FenceOrLifecycleLost); +} + +TEST(CASRequestController, FenceWinsCancellationAndExternalDeadlineWinsDeadlineTie) +{ + auto backend = std::make_shared(); + uint64_t clock = 100; + CasRequestBudget budget; + budget.attempt_timeout_ms = 10; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + + bool cancelled = true; + bool fenced = true; + const auto simultaneous_stop = [&] + { + if (fenced) + return CasOverwriteStopCause::FenceOrLifecycleLost; + if (cancelled) + return CasOverwriteStopCause::Cancelled; + return CasOverwriteStopCause::Continue; + }; + const auto stopped = controller.putOverwriteControlled( + "stopped", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety, simultaneous_stop)); + EXPECT_EQ(stopped.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + stopped, + 0, + false, + CasUnresolvedReason::NoAttemptSent, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::FenceOrLifecycleLost); + + const auto deadline_tie = controller.putOverwriteControlled( + "deadline", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + EXPECT_EQ(deadline_tie.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + deadline_tie, + 0, + false, + CasUnresolvedReason::NoAttemptSent, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); +} + +TEST(CASRequestController, InterruptedWaitResamplesStopCause) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + CasOverwriteStopCause stop = CasOverwriteStopCause::Continue; + uint32_t waits = 0; + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 100; + budget.retry_max_backoff_ms = 100; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + auto context = overwriteContext( + 10000, + CasOverwriteDeadlineSource::RequestBudget, + [&stop] { return stop; }, + [&stop, &waits](uint64_t wait_ms) + { + EXPECT_EQ(wait_ms, 100u); + ++waits; + stop = CasOverwriteStopCause::Cancelled; + return false; + }); + const auto result = controller.putOverwriteControlled("k", "new-payload", expected, context); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(waits, 1u); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + false, + CasUnresolvedReason::FenceLostMidWay, + CasOverwriteDeadlineSource::RequestBudget, + CasOverwriteStopCause::Cancelled); +} + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRequestController, InterruptedWaitWithoutPublishedStopIsAProgrammingException) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + uint32_t stop_samples = 0; + uint32_t waits = 0; + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 100; + budget.retry_max_backoff_ms = 100; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + auto context = overwriteContext( + 10000, + CasOverwriteDeadlineSource::RequestBudget, + [&stop_samples] + { + ++stop_samples; + return CasOverwriteStopCause::Continue; + }, + [&waits](uint64_t wait_ms) + { + EXPECT_EQ(wait_ms, 100u); + ++waits; + return false; + }); + + bool threw = false; + try + { + (void)controller.putOverwriteControlled("k", "new-payload", expected, context); + FAIL() << "an interrupted wait must publish a non-Continue stop cause"; + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + EXPECT_EQ( + e.message(), + "CasRequestController: wait_before_retry returned false while stop_cause remained Continue"); + } + + EXPECT_TRUE(threw); + EXPECT_EQ(stop_samples, 5u) << "the false wait was followed by an authoritative stop-cause resample"; + EXPECT_EQ(waits, 1u); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRequestControllerDeathTest, InterruptedWaitWithoutPublishedStopIsAProgrammingExceptionAborts) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 100; + budget.retry_max_backoff_ms = 100; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + auto context = overwriteContext( + 10000, + CasOverwriteDeadlineSource::RequestBudget, + [] { return CasOverwriteStopCause::Continue; }, + [](uint64_t) { return false; }); + + EXPECT_DEATH( + { (void)controller.putOverwriteControlled("k", "new-payload", expected, context); }, + "wait_before_retry returned false while stop_cause remained Continue"); +} +#endif + +TEST(CASRequestController, CompletedWaitCrossingDeadlineSendsNoRetry) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + const Token expected{"old", TokenType::Emulated}; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + uint64_t clock = 0; + uint32_t waits = 0; + + CasRequestBudget budget; + budget.max_attempts = 3; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 50; + budget.retry_max_backoff_ms = 50; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + auto context = overwriteContext( + 100, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + {}, + [&clock, &waits](uint64_t wait_ms) + { + EXPECT_EQ(wait_ms, 50u); + ++waits; + clock = 101; + return true; + }); + const auto result = controller.putOverwriteControlled("k", "new-payload", expected, context); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_TRUE(result.token.empty()); + EXPECT_EQ(waits, 1u); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + false, + CasUnresolvedReason::DeadlineMidWay, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, DirectPutCompletingAtDeadlineIsNotAccepted) +{ + auto backend = std::make_shared(); + const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); + ASSERT_EQ(predecessor.outcome, PutOutcome::Done); + uint64_t clock = 0; + Token landed_token; + auto * backend_ptr = backend.get(); + backend->put_overwrite_handler = [backend_ptr, &clock, &landed_token]( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult + { + const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); + landed_token = landed.token; + clock = 100; + return landed; + }; + + CasRequestBudget budget; + budget.attempt_timeout_ms = 10; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + const auto result = controller.putOverwriteControlled( + "k", + "new-payload", + predecessor.token, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_TRUE(result.token.empty()); + ASSERT_FALSE(landed_token.empty()); + const auto durable = backend->InMemoryBackend::get("k", Range{}); + ASSERT_TRUE(durable.has_value()); + EXPECT_EQ(durable->bytes, "new-payload"); + EXPECT_EQ(durable->token, landed_token); + EXPECT_NE(durable->token, predecessor.token); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 0u); + expectOverwriteDiagnostics( + result, + 1, + false, + CasUnresolvedReason::DeadlineMidWay, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, ReadProofCompletingAtDeadlineIsNotAccepted) +{ + auto backend = std::make_shared(); + const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); + ASSERT_EQ(predecessor.outcome, PutOutcome::Done); + uint64_t clock = 0; + Token landed_token; + auto * backend_ptr = backend.get(); + backend->put_overwrite_handler = [backend_ptr, &landed_token]( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult + { + const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); + landed_token = landed.token; + throw Poco::TimeoutException("scripted: response lost after overwrite landed"); + }; + backend->get_handler = [backend_ptr, &clock](const String & key, Range range) -> std::optional + { + const auto durable = backend_ptr->InMemoryBackend::get(key, range); + clock = 100; + return durable; + }; + + CasRequestBudget budget; + budget.attempt_timeout_ms = 10; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + const auto result = controller.putOverwriteControlled( + "k", + "new-payload", + predecessor.token, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_TRUE(result.token.empty()); + ASSERT_FALSE(landed_token.empty()); + const auto durable = backend->InMemoryBackend::get("k", Range{}); + ASSERT_TRUE(durable.has_value()); + EXPECT_EQ(durable->bytes, "new-payload"); + EXPECT_EQ(durable->token, landed_token); + EXPECT_NE(durable->token, predecessor.token); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + true, + CasUnresolvedReason::DeadlineMidWay, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, ObserverFailureCannotChangeOutcome) +{ + auto backend = std::make_shared(); + const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); + ASSERT_EQ(predecessor.outcome, PutOutcome::Done); + auto * backend_ptr = backend.get(); + bool first_attempt = true; + backend->put_overwrite_handler = [backend_ptr, &first_attempt]( + const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult + { + if (first_attempt) + { + first_attempt = false; + throw Poco::TimeoutException("scripted: transient failure before landing"); + } + return backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); + }; + uint32_t observer_calls = 0; + + CasRequestBudget budget; + budget.max_attempts = 2; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget, [] { return static_cast(0); }); + auto context = overwriteContext( + 10000, + CasOverwriteDeadlineSource::RequestBudget, + {}, + {}, + [&observer_calls](const CasOverwriteProgress &) + { + ++observer_calls; + throw DB::Exception(DB::ErrorCodes::UNKNOWN_EXCEPTION, "scripted observer failure"); + }); + + CasOverwriteResult result; + EXPECT_NO_THROW(result = controller.putOverwriteControlled("k", "new-payload", predecessor.token, context)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); + EXPECT_EQ(observer_calls, 5u); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 2, + false, + CasUnresolvedReason::NotUnresolved, + CasOverwriteDeadlineSource::RequestBudget, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, ResolveFailuresExhaustDeadlineWithoutSendingLatePut) +{ + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + uint64_t clock = 0; + backend->get_handler = [&clock](const String &, Range) -> std::optional + { + clock = 30; + throw Poco::TimeoutException("scripted: resolving GET failed at deadline"); + }; + + CasRequestBudget budget; + budget.max_attempts = 5; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 0; + CasRequestController controller(backend, budget, [&clock] { return clock; }); + const auto result = controller.putOverwriteControlled( + "k", "new-payload", Token{"old", TokenType::Emulated}, + overwriteContext(30, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); + EXPECT_EQ(backend->get_attempts.load(), 1u); + expectOverwriteDiagnostics( + result, + 1, + false, + CasUnresolvedReason::DeadlineMidWay, + CasOverwriteDeadlineSource::ExternalLeaseSafety, + CasOverwriteStopCause::Continue); +} + +TEST(CASRequestController, EveryTerminalShapeReportsExactDiagnostics) +{ + const Token expected{"old", TokenType::Emulated}; + auto make_controller = [](const std::shared_ptr & backend, uint64_t & clock, uint32_t max_attempts = 2) + { + CasRequestBudget budget; + budget.max_attempts = max_attempts; + budget.attempt_timeout_ms = 10; + budget.retry_initial_backoff_ms = 0; + return CasRequestController(backend, budget, [&clock] { return clock; }); + }; + + for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) + { + auto backend = std::make_shared(); + uint64_t clock = 0; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "pre-stop", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [stop] { return stop; })); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 0, false, CasUnresolvedReason::NoAttemptSent, + CasOverwriteDeadlineSource::RequestBudget, stop); + } + + for (const auto source : {CasOverwriteDeadlineSource::RequestBudget, CasOverwriteDeadlineSource::ExternalLeaseSafety}) + { + auto backend = std::make_shared(); + uint64_t clock = 100; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "pre-deadline", "new-payload", expected, overwriteContext(100, source)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 0, false, CasUnresolvedReason::NoAttemptSent, source, CasOverwriteStopCause::Continue); + } + + for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) + { + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + uint64_t clock = 0; + uint32_t stop_samples = 0; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "mid-stop", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [&stop_samples, stop] + { + ++stop_samples; + return stop_samples == 1 ? CasOverwriteStopCause::Continue : stop; + })); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 1, false, CasUnresolvedReason::FenceLostMidWay, + CasOverwriteDeadlineSource::RequestBudget, stop); + } + + for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) + { + auto backend = std::make_shared(); + backend->put_overwrite_forced_outcome = PutOutcome::Done; + uint64_t clock = 0; + uint32_t stop_samples = 0; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "post-stop", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [&stop_samples, stop] + { + ++stop_samples; + return stop_samples == 1 ? CasOverwriteStopCause::Continue : stop; + })); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 1, false, CasUnresolvedReason::FenceLostPostWrite, + CasOverwriteDeadlineSource::RequestBudget, stop); + } + + for (const auto source : {CasOverwriteDeadlineSource::RequestBudget, CasOverwriteDeadlineSource::ExternalLeaseSafety}) + { + auto backend = std::make_shared(); + uint64_t clock = 0; + backend->put_overwrite_handler = [&clock]( + const String &, const String &, const Token &, const ObjectMeta &) -> PutResult + { + clock = 100; + throw Poco::TimeoutException("scripted: ambiguous at deadline"); + }; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "mid-deadline", "new-payload", expected, overwriteContext(100, source)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 1, false, CasUnresolvedReason::DeadlineMidWay, source, CasOverwriteStopCause::Continue); + } + + { + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); + uint64_t clock = 0; + auto controller = make_controller(backend, clock, 1); + const auto result = controller.putOverwriteControlled( + "attempts", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 1, false, CasUnresolvedReason::AttemptsExhausted, + CasOverwriteDeadlineSource::ExternalLeaseSafety, CasOverwriteStopCause::Continue); + } + + { + auto backend = std::make_shared(); + auto * backend_ptr = backend.get(); + backend->put_overwrite_handler = [backend_ptr]( + const String &, const String &, const Token &, const ObjectMeta &) -> PutResult + { + if (backend_ptr->put_overwrite_attempts.load() == 1) + throw Poco::TimeoutException("scripted: first attempt remains ambiguous"); + throw DB::S3Exception("scripted: later attempt definitely refused", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); + }; + backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); + uint64_t clock = 0; + auto controller = make_controller(backend, clock); + const auto result = controller.putOverwriteControlled( + "definite-after-ambiguity", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); + expectOverwriteDiagnostics( + result, 2, false, CasUnresolvedReason::DefiniteFailureAfterAmbiguity, + CasOverwriteDeadlineSource::RequestBudget, CasOverwriteStopCause::Continue); + } + + { + auto backend = std::make_shared(); + backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + uint64_t clock = 0; + backend->get_handler = [&clock](const String &, Range) -> std::optional + { + clock = 95; + return std::nullopt; + }; + auto controller = make_controller(backend, clock, 1); + const auto result = controller.putOverwriteControlled( + "deadline-before-attempt-limit", "new-payload", expected, + overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); + EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); + expectOverwriteDiagnostics( + result, 1, false, CasUnresolvedReason::DeadlineMidWay, + CasOverwriteDeadlineSource::ExternalLeaseSafety, CasOverwriteStopCause::Continue); + } +} + +TEST(CASRequestController, FenceLostBeforeAttemptSendsNoAttempt) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return false; }); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 0u); +} + +/// The write itself may have landed, but a fence lost between the write and this call's own final +/// check must never surface as Committed (RFC §ack-and-cache-rules: no ACK, no cache update on that +/// path) — the caller sees Unresolved and must not treat the operation as acknowledged. +TEST(CASRequestController, FenceLostAfterWriteNeverReturnsCommitted) +{ + auto backend = std::make_shared(); /// real in-memory commit path + int fence_calls = 0; + auto fence_ok = [&fence_calls] { return fence_calls++ == 0; }; /// true once, then false + + CasRequestController controller(backend, CasRequestBudget{}); + const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 1u); /// the write itself DID happen + EXPECT_TRUE(backend->head("k").exists); /// ...it is durable; never claimed as Committed here +} + +TEST(CASRequestController, DefiniteFailurePropagatesImmediatelyWithoutResolve) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw DB::S3Exception("scripted: malformed", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); }; + + CasRequestController controller(backend, CasRequestBudget{}); + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::DefiniteFailure); + EXPECT_EQ(backend->put_attempts.load(), 1u); /// no retry, no resolve GET issued +} + +/// ================================================================================================ +/// Inter-attempt backoff (chaos-tolerance-report §Task B follow-up / stagefix-review M3): the +/// controller paces reissues with a capped-exponential, fence-gated, deadline-aware sleep instead of +/// hammering a recovering store with immediate retries. +/// ================================================================================================ + +/// The full event-ordered schedule: fence checked before EVERY attempt AND before EVERY sleep, sleeps +/// strictly between attempts, capped exponential (initial 100ms, cap 200ms), no sleep after the final +/// attempt. The exact interleaving is the contract — a sleep served before its fence check would keep +/// a fenced writer dozing past its lease. +TEST(CASRequestControllerBackoff, CappedExponentialSleepsAreFenceCheckedAndOrdered) +{ + auto backend = std::make_shared(); + std::vector events; + backend->put_thrower = [&] { events.emplace_back("put"); throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); /// absent on every resolve + + CasRequestBudget budget; + budget.max_attempts = 5; + budget.attempt_timeout_ms = 1; + budget.operation_deadline_ms = 1000000; /// never the binding constraint here + budget.retry_initial_backoff_ms = 100; + budget.retry_max_backoff_ms = 200; + CasRequestController controller( + backend, budget, + /*now_ms=*/[] { return static_cast(0); }, + /*sleep_ms=*/[&](uint64_t ms) { events.push_back("sleep:" + std::to_string(ms)); }); + + const auto fence_ok = [&] { events.emplace_back("fence"); return true; }; + const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 5u); + + const std::vector expected{ + "fence", "put", "fence", "sleep:100", + "fence", "put", "fence", "sleep:200", + "fence", "put", "fence", "sleep:200", + "fence", "put", "fence", "sleep:200", + "fence", "put"}; /// budget spent: no fence-for-sleep, no sleep after the last attempt + EXPECT_EQ(events, expected); +} + +/// A fence lost between an ambiguous attempt's resolve and its backoff sleep aborts INSTANTLY: no +/// sleep is served, no further attempt is sent, and the outcome is Unresolved (never a false +/// Committed, never a retry under a lost lease). +TEST(CASRequestControllerBackoff, FenceLostBeforeSleepAbortsWithoutSleeping) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); + + CasRequestBudget budget; + budget.max_attempts = 5; + budget.retry_initial_backoff_ms = 100; + budget.retry_max_backoff_ms = 200; + uint64_t sleeps = 0; + int fence_calls = 0; + CasRequestController controller( + backend, budget, /*now_ms=*/[] { return static_cast(0); }, + /*sleep_ms=*/[&](uint64_t) { ++sleeps; }); + + /// True for the pre-attempt check (call 1), lost by the pre-sleep check (call 2). + const auto fence_ok = [&fence_calls] { return ++fence_calls <= 1; }; + const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 1u) << "no attempt may be sent after the fence is lost"; + EXPECT_EQ(sleeps, 0u) << "a fence lost mid-backoff must abort BEFORE the sleep, not after it"; + EXPECT_EQ(fence_calls, 2); +} + +/// A backoff sleep the operation deadline cannot afford is never served: when sleep + one more +/// attempt would cross the deadline, the loop gives up immediately (Unresolved) instead of sleeping +/// into a guaranteed exhaustion. +TEST(CASRequestControllerBackoff, SleepThatWouldCrossOperationDeadlineIsSkipped) +{ + auto backend = std::make_shared(); + backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; + backend->setGetOverride(std::nullopt); + + uint64_t clock = 0; + CasRequestBudget budget; + budget.max_attempts = 10; + budget.attempt_timeout_ms = 10; + budget.operation_deadline_ms = 100; + budget.retry_initial_backoff_ms = 1000; /// any sleep would blow the 100ms deadline + budget.retry_max_backoff_ms = 1000; + uint64_t sleeps = 0; + CasRequestController controller( + backend, budget, /*now_ms=*/[&clock] { return clock; }, + /*sleep_ms=*/[&](uint64_t) { ++sleeps; }); + + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); + EXPECT_EQ(backend->put_attempts.load(), 1u); + EXPECT_EQ(sleeps, 0u) << "the deadline guard must refuse the sleep, not serve it and then fail"; +} + +/// THE ENVELOPE CONTRACT (chaos-tolerance-report §Task B follow-up): the DEFAULT budget rides a +/// simulated 60-second S3 outage — every conditional-write attempt fails (≈3s adaptive first-attempt +/// timeout each, the observed incident shape) until the store recovers at t=60s, then the next +/// attempt commits, all inside the default 90s operation deadline and 16-attempt budget. The fake +/// clock advances 3s per failed attempt and by each backoff sleep, so this test pins the arithmetic +/// documented on CasRequestBudget without any wall-clock waiting. +TEST(CASRequestControllerBackoff, DefaultBudgetRidesSixtySecondOutage) +{ + auto backend = std::make_shared(); + uint64_t clock = 0; + backend->put_thrower = [&clock] + { + if (clock < 60000) + { + clock += 3000; /// the failed attempt's own ~3s adaptive receive timeout + throw Poco::TimeoutException("scripted: store paused"); + } + /// store recovered: fall through to the real in-memory conditional write (Done) + }; + backend->setGetOverride(std::nullopt); /// nothing ever landed while the store was paused + + CasRequestController controller( + backend, CasRequestBudget{}, /*now_ms=*/[&clock] { return clock; }, + /*sleep_ms=*/[&clock](uint64_t ms) { clock += ms; }); + + const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); + EXPECT_EQ(outcome, CasWriteOutcome::Committed) << "the default budget must absorb a 60s outage"; + /// Schedule: attempts fail at 3s each with sleeps 0.2,0.4,0.8,1.6,3.2 then 5s (cap); the first + /// attempt scheduled at clock >= 60000 (attempt 11, t=61.2s) commits — well inside 16 attempts + /// and the 90s deadline. + EXPECT_EQ(backend->put_attempts.load(), 11u); + EXPECT_LT(clock, CasRequestBudget{}.operation_deadline_ms); +} + +/// Startup validation (RFC §required-timeout-model): a consistent default budget is accepted silently; +/// either inequality violated on its own is rejected with BAD_ARGUMENTS. +TEST(CASRequestController, ValidateBudgetAcceptsConsistentDefaults) +{ + EXPECT_NO_THROW(validateCasRequestBudget(CasRequestBudget{}, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000)); +} + +TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutPlusMarginAtOrAboveLeaseTtl) +{ + CasRequestBudget budget; + budget.attempt_timeout_ms = 25000; + budget.lease_safety_margin_ms = 5000; /// sums to EXACTLY the lease TTL below — not strictly less + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); +} + +TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutAboveOperationDeadline) +{ + CasRequestBudget budget; + budget.attempt_timeout_ms = 6000; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 1000; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); +} + +/// max_attempts == 0 would let putIfAbsentControlled return Unresolved without ever sending an +/// attempt — reject at startup rather than silently accepting a no-op budget. +TEST(CASRequestController, ValidateBudgetRejectsZeroMaxAttempts) +{ + CasRequestBudget budget; + budget.max_attempts = 0; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); +} + +/// A capped-exponential backoff whose cap sits below its own starting value is inconsistent — reject +/// at startup (0/0 disables backoff and stays accepted, covered by the defaults test above since the +/// defaults are nonzero and consistent). +TEST(CASRequestController, ValidateBudgetRejectsInitialBackoffAboveMaxBackoff) +{ + CasRequestBudget budget; + budget.retry_initial_backoff_ms = 500; + budget.retry_max_backoff_ms = 100; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); +} + +/// attempt_timeout_ms + lease_safety_margin_ms must not be computed by a wrapping uint64 sum: absurd +/// config values near UINT64_MAX must fail validation (correctly, as inconsistent), never wrap around +/// to a spuriously small sum that would pass the "< lease TTL" check. +TEST(CASRequestController, ValidateBudgetRejectsOverflowingSumRatherThanWrapping) +{ + CasRequestBudget budget; + budget.attempt_timeout_ms = std::numeric_limits::max() - 10; + budget.lease_safety_margin_ms = 20; /// sum would wrap past UINT64_MAX to a tiny value + budget.operation_deadline_ms = std::numeric_limits::max(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); + }); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_retirement_sweep.cpp b/src/Disks/tests/gtest_cas_retirement_sweep.cpp new file mode 100644 index 000000000000..7300f02d0c7c --- /dev/null +++ b/src/Disks/tests/gtest_cas_retirement_sweep.cpp @@ -0,0 +1,420 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +/// THE RETIREMENT SWEEP's executable half. +/// +/// Two mechanisms of the pre-v9 ref protocol lost their premise when the ref stream became a +/// contiguous, arithmetically-walkable chain, and this file is where each retirement is proved rather +/// than asserted in a comment: +/// +/// 1. PROBE A's ABORT. The detector compared the round's two enumerations of `cas/ns/stream/` and, on any +/// disagreement, aborted ref folding for the whole round -- because the fold ITERATED the listing, +/// so a hole in it meant a record was about to be skipped forever. The intake reads by exact key +/// now, so a hole folds through; a detector that still aborted would be halting a round that is +/// provably doing the right thing. It was demoted to a sampled store-quality detector and then +/// deleted outright: the round enumerates `cas/ns/stream/` exactly once, on every round. +/// 2. THE MATERIALIZATION GRACE (`T_mat`). A post-reclaim sleep, long enough for a straggler +/// conditional `PUT` from a dying epoch to land or exhaust its retries BEFORE the successor +/// trusted its recovery LISTINGS. Recovery does not trust listings; it closes every dead epoch +/// with an in-band `EpochSeal` written as a conditional create, and the straggler's own create +/// loses to it. The wait is deleted outright, setting and all -- the feature never shipped, so +/// there is no config to protect and no parsed-but-inert period to serve. +/// +/// The retirement rationale (premise / verdict / replacement / evidence, one row per retired item) +/// is captured in `docs/en/antalya/cas/architecture/design-history.md`. + +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// A backend that drops ONE chosen key from ONE chosen `list` call while exact `get`/`head` of that key +/// keep working: the minimal realisation of "the store returned an incomplete answer". WHICH call is +/// load-bearing here, because the round enumerates the ref prefix exactly once -- so `nth = 0` is +/// always the round's own enumeration, the one the fold regroups. +/// +/// `nth` counts only those `list` calls that WOULD have returned the key, so unrelated prefix +/// enumerations cannot shift it; arm it AFTER every seeding write, since the writer's own namespace +/// listings would otherwise consume a qualifying call. +class HoleyListBackend : public InMemoryBackend +{ +public: + void omitFromNthListCall(const String & key, size_t nth) + { + std::lock_guard lock(m); + omitted = key; + target_call = nth; + seen_calls = 0; + served = false; + } + + /// Whether the hole was actually served. Asserted by every test that plants one, so a mistyped key + /// or a miscounted `nth` cannot let a test pass vacuously. + bool holeServed() const + { + std::lock_guard lock(m); + return served; + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + ListPage page = InMemoryBackend::list(prefix, cursor, limit); + std::lock_guard lock(m); + if (omitted.empty()) + return page; + auto it = std::find_if(page.keys.begin(), page.keys.end(), + [&](const ListedKey & k) { return k.key == omitted; }); + if (it == page.keys.end()) + return page; /// not a qualifying call -- do not count it + if (seen_calls++ != target_call) + return page; + page.keys.erase(it); + served = true; + omitted.clear(); /// one hole only + return page; + } + +private: + mutable std::mutex m; + String omitted; + size_t target_call = 0; + size_t seen_calls = 0; + bool served = false; +}; + +/// Counts full enumerations of the ref prefix -- one increment per `list` call whose prefix is EXACTLY +/// `cas/ns/stream/`, which is how "the round lists this prefix once" becomes an assertion instead of a +/// claim. `janitor_prefix_lists` counts the bounded `cas/ns/` janitor page separately, by the same exact +/// match: the two prefixes are distinct strings (`cas/ns/stream/` vs `cas/ns/`), so a hot scan and the +/// janitor's own bounded page can never be conflated by this counter. +class RefPrefixListCountingBackend : public InMemoryBackend +{ +public: + String refs_prefix; + String janitor_prefix; + std::atomic ref_prefix_lists{0}; + std::atomic janitor_prefix_lists{0}; + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (!refs_prefix.empty() && prefix == refs_prefix) + ++ref_prefix_lists; + if (!janitor_prefix.empty() && prefix == janitor_prefix) + ++janitor_prefix_lists; + return InMemoryBackend::list(prefix, cursor, limit); + } +}; + +/// Forces the FIRST `putIfAbsent` whose key contains `fault_key_substr` to throw an ambiguous +/// (Unresolved-classified) exception, `fault_count` times -- the minimal fault injection needed to drive +/// a ref-log append into the `Unresolved`/wedge outcome, with `max_attempts = 1` in the budget so the +/// single failed attempt exhausts the retry budget immediately. (Same shape as `gtest_cas_pool.cpp`'s +/// file-local backend of the same name; both are three lines of `throw` over `InMemoryBackend`, and +/// hoisting a shared one would couple two suites' fault models for no gain.) +class UnresolvedPutBackend final : public InMemoryBackend +{ +public: + using Backend::putIfAbsent; + + String fault_key_substr; + int fault_count = 0; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (fault_count > 0 && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + { + --fault_count; + throw Poco::TimeoutException("UnresolvedPutBackend: simulated ambiguous result (response lost)"); + } + return InMemoryBackend::putIfAbsent(key, bytes, meta); + } +}; + +/// GC's fence-out applied directly to the mount lease: preserve the body, set `gc_fenced`, bump `seq` +/// (token-guarded). A subsequent `tryRemountOnce` then reclaims a fresh incarnation. +void fenceOutMount(Backend & backend, const String & mount_key) +{ + const auto got = backend.get(mount_key); + ASSERT_TRUE(got.has_value()); + MountLease m = decodeMountLease(got->bytes); + m.gc_fenced = true; + m.seq += 1; + ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, PutOutcome::Done); +} + +/// Publish one part `ref` with a single content blob whose payload is `payload`. +ManifestId publishOneBlobPart(const PoolPtr & s, const RootNamespace & ns, const String & ref, + const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + auto build = s->beginPartWrite(info); + + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}; + e.blob_size = payload.size(); + + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(ns, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(ns, ref, build->buildId(), id); + return id; +} + +/// Every ref-log key of `ns` currently listed, in key order. +std::set listRefLogKeys(Backend & b, const Layout & l, const RootNamespace & ns) +{ + std::set out; + String cursor; + while (true) + { + const ListPage page = b.list(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + for (const ListedKey & k : page.keys) + if (const auto parsed = l.parseRefObjectKey(k.key); parsed && parsed->kind == RefObjectKind::Log) + out.insert(k.key); + if (page.next_cursor.empty()) + break; + cursor = page.next_cursor; + } + return out; +} + +/// The greatest ref-log id of `ns` in the current listing. +RefTxnId greatestLoggedId(Backend & b, const Layout & l, const RootNamespace & ns) +{ + RefTxnId best{}; + for (const String & key : listRefLogKeys(b, l, ns)) + if (const auto parsed = l.parseRefObjectKey(key); parsed && best < parsed->txn_id) + best = parsed->txn_id; + return best; +} + +Poco::AutoPtr makeDiskConfig(const std::string & inner) +{ + std::istringstream iss("" + inner + ""); + return new Poco::Util::XMLConfiguration(iss); +} + +} + +/// The blob the hidden removal releases must actually be reclaimed. This is the retention half of the +/// skipped-transaction class, and it is the outcome the abort used to buy at the price of a lost round: +/// under arithmetic intake the very round that was served the hole folds the removal, so the blob dies +/// on the normal schedule rather than waiting for a listing to become honest again. +TEST(CASRetirementSweep, AHiddenRemovalStillReclaimsItsBlob) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0, + }); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv/tbl"}; + const String payload = "reclaimed-payload"; + /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `listRefLogKeys` below + /// lists at that exact prefix. The raw `Live` row also needs the same empty checkpoint authority + /// as a completed production birth before `publishOneBlobPart` invokes recovery. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + publishOneBlobPart(store, ns, "part_a", payload); + Gc gc(store, hexToU128("00000000000000000000000000000012")); + /// Reclaiming rounds: Stage A's destructive gate refuses a universe it cannot enumerate, and this + /// test's subject IS the reclamation (see `runRegularRoundReclaiming`). + ASSERT_TRUE(DB::Cas::tests::runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + const String blob_key = layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, + BlobDigest::fromU128(u128Of(payload))}); + ASSERT_TRUE(backend->head(blob_key).exists); + + const std::set before_drop = listRefLogKeys(*backend, layout, ns); + store->dropRef(ns, "part_a"); + String removal_key; + for (const String & k : listRefLogKeys(*backend, layout, ns)) + if (!before_drop.contains(k)) + removal_key = k; + ASSERT_FALSE(removal_key.empty()); + store->renewWatermarkOnce(); + + backend->omitFromNthListCall(removal_key, /*nth=*/0); + + /// condemn -> graduate -> exact-token delete needs several rounds; the first of them is the one + /// served the hole. + for (int i = 0; i < 12; ++i) + { + ASSERT_TRUE(DB::Cas::tests::runRegularRoundReclaiming(gc).acquired_lease); + store->renewWatermarkOnce(); + } + ASSERT_TRUE(backend->holeServed()) << "the sabotage never fired"; + + EXPECT_FALSE(backend->head(blob_key).exists) + << "the removal was hidden from one enumeration and never folded -- the retention half of the " + "skipped-transaction class, which arithmetic intake is supposed to close"; +} + +/// THE RETIREMENT's whole point, made observable: a folding round enumerates `cas/ns/stream/` exactly +/// ONCE, on EVERY round of a multi-round run -- not just the first, since a regression that quietly +/// reintroduced a second enumeration only on a LATER round would pass a one-round check for the wrong +/// reason. The bounded `cas/ns/` janitor page is a separate exact-string prefix and runs every round +/// too; it must never be counted as, or mistaken for, a hot scan of the ref prefix. 32 rounds exercises +/// the deleted detector's own cadence (every 16th folding round) twice over, so a regression that only +/// reintroduces the second enumeration on that cadence cannot hide inside a shorter run. +TEST(CASRetirementSweep, TheRoundEnumeratesTheRefPrefixExactlyOnce) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0, + }); + backend->refs_prefix = store->layout().casRefsPrefix(); + backend->janitor_prefix = store->layout().namespaceRootPrefix(); + const RootNamespace ns{"srv/tbl"}; + publishOneBlobPart(store, ns, "part_a", "counted-payload"); + store->renewWatermarkOnce(); + + Gc gc(store, hexToU128("00000000000000000000000000000013")); + for (int round = 0; round < 32; ++round) + { + backend->ref_prefix_lists.store(0); + backend->janitor_prefix_lists.store(0); + const RoundReport report = gc.runRegularRound(); + ASSERT_TRUE(report.acquired_lease); + ASSERT_FALSE(report.deferred); + EXPECT_EQ(backend->ref_prefix_lists.load(), 1u) + << "round " << round << " enumerated cas/ns/stream/ a number of times other than once"; + EXPECT_GT(backend->janitor_prefix_lists.load(), 0u) + << "round " << round << " never took the bounded cas/ns/ janitor page"; + store->renewWatermarkOnce(); + } +} + + +/// ==================== item 2: the materialization grace, retired ==================== + +/// THE MECHANISM THAT REPLACED THE WAIT, tested directly. A ref lane is left holding an UNDECIDED +/// conditional `PUT` when the fence trips -- the exact state `T_mat` was introduced to wait out. The +/// remount proceeds with no wait at all, the next recovery closes the dead epoch with an in-band +/// `EpochSeal` at the slot the straggler would have taken, and the straggler's own conditional create +/// then LOSES to it. +/// +/// The assertion is the conflict itself, not the absence of damage: "nothing bad happened" would also +/// be true of a run where the straggler simply never arrived. +TEST(CASRetirementSweep, AStragglerFromTheDyingEpochLosesItsCreateToTheRecoverySeal) +{ + CasRequestBudget budget; + budget.max_attempts = 1; + budget.attempt_timeout_ms = 100; + budget.operation_deadline_ms = 5000; + budget.lease_safety_margin_ms = 100; + + auto backend = std::make_shared(); + uint64_t fake_boot = 1'000'000; + std::vector waits; + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(30000), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + }); + ASSERT_TRUE(store); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv/straggler"}; + /// Pin to the transition life before the first real touch, and give that raw `Live` row the exact + /// empty checkpoint authority that production birth would have published before recovery. + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = std::nullopt, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + + publishOneBlobPart(store, ns, "x", "straggler-payload"); + ASSERT_EQ(store->liveWriterEpoch(), 1u); + + /// Drive the next ref-log append into the Unresolved/wedge outcome: the single attempt the budget + /// allows fails ambiguously, so this process can never learn whether its conditional PUT landed. + /// That undecidability is the whole reason the resolution is a conditional CREATE and not a GET. + backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + + /// The id the straggler would occupy: one past the greatest record that is actually durable in the + /// dying epoch. That is also, by construction, where the recovery seal goes. + const RefTxnId greatest = greatestLoggedId(*backend, layout, ns); + ASSERT_EQ(greatest.writer_epoch, 1u); + const RefTxnId straggler_slot{greatest.writer_epoch, greatest.ref_sequence + 1}; + ASSERT_FALSE(backend->head(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot)).exists) + << "the slot must be empty before recovery -- otherwise this test proves nothing about who won"; + + /// Fence and remount. No wait: this is the case that used to cost 30 seconds. + fake_boot += 30001; + fenceOutMount(*backend, layout.mountKey("test")); + ASSERT_TRUE(store->tryRemountOnce()); + ASSERT_EQ(store->liveWriterEpoch(), 2u); + EXPECT_TRUE(waits.empty()) + << "the remount blocked on an operator-configured wait; the grace is supposed to be gone"; + + /// Touch the namespace so it re-recovers under the new epoch: the walk closes epoch 1 in band. The + /// ref itself is still THERE -- the removal's PUT was the undecided one and (in this fixture) never + /// landed, which is precisely the state that leaves a straggler outstanding. + backend->fault_key_substr.clear(); + EXPECT_EQ(store->listRefs(ns).size(), 1u); + ASSERT_TRUE(backend->head(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot)).exists) + << "recovery did not seal the dead epoch at the slot a straggler would take -- without that " + "seal there is nothing for the straggler's create to lose to"; + + /// THE STRAGGLER ARRIVES. Its conditional create is refused, whenever it happens to land. + const PutResult put = backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot), "ghost-body"); + EXPECT_EQ(put.outcome, PutOutcome::PreconditionFailed) + << "the dying epoch's straggler overwrote (or joined) a slot the successor had already sealed"; +} + +/// THE FAIL-CLOSE THAT REPLACES THE DELETED SETTING. `materialization_grace_ms` is gone from the +/// settings table outright -- no parsed-but-inert period, no deprecation log -- so a config that still +/// asks for the wait is refused at disk open by the generic unknown-key path, loudly, instead of being +/// silently ignored by a server that no longer honours it. The feature never shipped, so there is no +/// deployed config this can break. +TEST(CASRetirementSweep, AConfigStillAskingForTheMaterializationGraceIsRejected) +{ + auto cfg = makeDiskConfig( + "srv130000"); + DB::ContentAddressedSettings s; + EXPECT_THROW( + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", [](const std::string & v) { return v; }), + DB::Exception) + << "a retired setting must fail the disk open, not be quietly accepted and ignored"; +} diff --git a/src/Disks/tests/gtest_cas_s3_staging.cpp b/src/Disks/tests/gtest_cas_s3_staging.cpp new file mode 100644 index 000000000000..92c4542b1a32 --- /dev/null +++ b/src/Disks/tests/gtest_cas_s3_staging.cpp @@ -0,0 +1,1196 @@ +#include +#include "cas_test_helpers.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +#include "config.h" +#if USE_AWS_S3 +#include +#include +#endif + +/// `staging_backend` defaults to `local`; explicit `s3` selection requires native-copy capability +/// on writable mounts. + +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsString staging_backend; +} + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int FILE_DOESNT_EXIST; +} + +namespace +{ + +/// Build a `Poco::Util::XMLConfiguration` with `inner_xml` nested under a `` element (mirrors +/// the shape a real CAS disk config has under `storage_configuration.disks.`, so +/// `config_prefix = "disk"` reads exactly like the disk factory's `config_prefix`). +Poco::AutoPtr configWithDiskSection(const std::string & inner_xml) +{ + std::istringstream xml_stream( // STYLE_CHECK_ALLOW_STD_STRING_STREAM + "" + inner_xml + ""); + return new Poco::Util::XMLConfiguration(xml_stream); +} + +/// A local test store whose copy-mode capability is configurable. Its ordinary `copyObject` +/// implementation remains the real local implementation; only the advertised transport capability +/// differs so mount selection can be tested independently of a live S3 service. +class FakeNativeCopyObjectStorage final : public DB::LocalObjectStorage +{ +public: + FakeNativeCopyObjectStorage(DB::LocalObjectStorageSettings settings_, bool native_only_copy_supported_) + : DB::LocalObjectStorage(std::move(settings_)) + , native_only_copy_supported(native_only_copy_supported_) + { + } + + bool supportsCopyMode(DB::ObjectStorageCopyMode mode) const override + { + return mode == DB::ObjectStorageCopyMode::Default + || (mode == DB::ObjectStorageCopyMode::NativeOnly && native_only_copy_supported); + } + +private: + const bool native_only_copy_supported; +}; + +std::shared_ptr makeFakeNativeCopyStorage(bool native_only_copy_supported) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_s3_staging_native_copy_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings), native_only_copy_supported); +} + +/// A fake object-store sink for Task 4 of the S3-native staging plan (`DB::Cas::CaContentWriteBuffer`'s +/// S3-staging constructor): an in-memory `WriteBufferFromFileBase` that records every byte written to +/// it, plus whether `cancelImpl`/`finalizeImpl` ran. This is enough to prove the S3-staging mode +/// streams to the SINK (not to a local temp file) while hashing, without needing a real object storage +/// — the end-to-end wiring (`writeFile` choosing this mode, the promote path) lands in later tasks. +class FakeStagingSink : public DB::WriteBufferFromFileBase +{ +public: + explicit FakeStagingSink(std::string key_) + : DB::WriteBufferFromFileBase(/*buf_size=*/8192, nullptr, 0), key(std::move(key_)) + { + } + + void sync() override {} + std::string getFileName() const override { return key; } + + const std::string & writtenBytes() const { return written; } + bool wasCancelled() const { return cancelled; } + bool wasFinalizedForTest() const { return did_finalize; } + +protected: + void nextImpl() override + { + if (!offset()) + return; + written.append(working_buffer.begin(), offset()); + } + + void finalizeImpl() override + { + next(); + did_finalize = true; + } + + void cancelImpl() noexcept override + { + cancelled = true; + } + +private: + std::string key; + std::string written; + bool cancelled = false; + bool did_finalize = false; +}; + +/// Records whether each unconditional publication used verbatim native copy or a retagged stream. +/// Stream reads are counted separately so condemned-destination tests can prove they read only the +/// writer-owned staging object. +class RecordingStagingBackend : public DB::Cas::InMemoryBackend +{ +public: + struct CopyCall + { + std::string from; + std::string to; + bool server_side_copy; + }; + + std::vector copy_calls; + + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + if (const auto * copy = std::get_if(&request.publication)) + copy_calls.push_back({copy->object_key, request.destination_key, true}); + else + copy_calls.push_back({String{}, request.destination_key, false}); + DB::Cas::InMemoryBackend::publishBlob(request); + } + + /// Every key read as a stream, with a count. Republishing opens its source with `getStream`, so + /// this counts exactly those reads -- and deliberately not the materializing + /// `get`, which the assertions themselves use to inspect bodies. + std::map reads_of; + + using DB::Cas::InMemoryBackend::getStream; + std::optional getStream(const String & key, DB::Cas::Range range) override + { + ++reads_of[key]; + return DB::Cas::InMemoryBackend::getStream(key, range); + } + + + size_t streamingPublicationCount() const + { + size_t n = 0; + for (const CopyCall & c : copy_calls) + n += c.server_side_copy ? 0 : 1; + return n; + } +}; + +/// Models an ETag store faithfully enough for the staged-envelope regressions: a blob token is a +/// deterministic digest of the complete object bytes, so copying the same staging object again would +/// reproduce the same token. Each script injects a different ambiguity transition from the design. +class EtagFaithfulPublicationBackend final : public DB::Cas::InMemoryBackend +{ +public: + enum class FaultScript : uint8_t + { + CopyLandsThenCondemned, + CopyLandsThenDeletedBeforeAbsentRetry, + FirstCondemnedStreamLandsThenDeleted, + }; + + explicit EtagFaithfulPublicationBackend(FaultScript script_) : script(script_) {} + + DB::Cas::HeadResult head(const String & key) override + { + DB::Cas::HeadResult result = DB::Cas::InMemoryBackend::head(key); + if (result.exists && isBlobBodyKey(key)) + { + const auto body = DB::Cas::InMemoryBackend::get(key); + chassert(body.has_value()); + result.token = DB::Cas::Token{sipHash128String(body->bytes), DB::Cas::TokenType::ETag}; + } + return result; + } + + DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + { + if (!isBlobBodyKey(key)) + return DB::Cas::InMemoryBackend::deleteExact(key, token); + + const DB::Cas::HeadResult current = head(key); + if (!current.exists) + return DB::Cas::DeleteOutcome{.kind = DB::Cas::DeleteOutcome::Kind::NotFound}; + if (current.token != token) + return DB::Cas::DeleteOutcome{.kind = DB::Cas::DeleteOutcome::Kind::TokenMismatch}; + return DB::Cas::InMemoryBackend::deleteExact(key, DB::Cas::InMemoryBackend::head(key).token); + } + + void publishBlob(const DB::Cas::BlobPublishRequest & request) override + { + const bool is_copy = std::holds_alternative(request.publication); + if (is_copy) + ++copy_publications; + else + ++streaming_publications; + + if (!fault_fired + && ((script == FaultScript::CopyLandsThenCondemned && is_copy) + || (script == FaultScript::CopyLandsThenDeletedBeforeAbsentRetry && is_copy) + || (script == FaultScript::FirstCondemnedStreamLandsThenDeleted && !is_copy))) + { + fault_fired = true; + DB::Cas::InMemoryBackend::publishBlob(request); + queued_delete_token = head(request.destination_key).token; + + if (script != FaultScript::CopyLandsThenCondemned) + first_delete = deleteExact(request.destination_key, queued_delete_token); + + throw Poco::TimeoutException("ETag-faithful staged publication response lost"); + } + + DB::Cas::InMemoryBackend::publishBlob(request); + } + + FaultScript script; + bool fault_fired = false; + size_t copy_publications = 0; + size_t streaming_publications = 0; + DB::Cas::Token queued_delete_token; + DB::Cas::DeleteOutcome first_delete; + +private: + static bool isBlobBodyKey(const String & key) + { + return key.find("/blobs/") != String::npos && !key.ends_with(".meta"); + } +}; + +DB::Cas::PoolPtr openStagingPool(const std::shared_ptr & b) +{ + return DB::Cas::Pool::open(b, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// A build whose owning manifest namespace / final ref name are `ns`/`ref` (mirrors gtest_cas_build's +/// `startBuildFor`: promote/stageManifest derive the namespace by splitting `intended_ref` on the LAST '/'). +DB::Cas::PartWriteTxnPtr startStagingBuild(const DB::Cas::PoolPtr & s, const DB::Cas::RootNamespace & ns, const String & ref) +{ + DB::Cas::PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + return s->beginPartWrite(info); +} + +/// Stage a one-blob manifest and precommit it (so the EDGE-BEFORE-OBSERVE fail-closed check in +/// `PartWriteTxn::ensureBlobPresent` holds), returning the build ready for `putBlob` on `hash`. +DB::Cas::PartWriteTxnPtr precommittedBuildFor( + const DB::Cas::PoolPtr & s, const DB::Cas::RootNamespace & ns, const String & ref, + const DB::UInt128 & hash, uint64_t blob_size) +{ + DB::Cas::PartWriteTxnPtr build = startStagingBuild(s, ns, ref); + const DB::Cas::ManifestId id = build->stageManifest({DB::Cas::tests::blobEntryFor("col.bin", hash, blob_size)}); + build->precommitAdd(ns, ref, id); + return build; +} + +DB::Cas::BlobSource reReadableStagedSource( + const DB::Cas::BackendPtr & backend, const std::string & staging_key, uint64_t payload_size, uint64_t header_len) +{ + DB::Cas::BlobSource source; + source.size = payload_size; + source.server_side_copy_from = staging_key; + source.open = [backend, staging_key, header_len]() -> std::unique_ptr + { + auto staged = backend->getStream(staging_key); + if (!staged) + throw DB::Exception(DB::ErrorCodes::FILE_DOESNT_EXIST, "staging object {} is absent", staging_key); + + String encoded_header(header_len, '\0'); + staged->stream->readStrict(encoded_header.data(), encoded_header.size()); + const DB::Cas::EnvelopeHeader decoded + = DB::Cas::decodeEnvelopeHeader(encoded_header, encoded_header.size(), DB::Cas::ObjectKind::Blob); + if (decoded.header_len != header_len) + throw DB::Exception( + DB::ErrorCodes::CORRUPTED_DATA, + "staging object {} uses envelope length {}, expected {}", + staging_key, + decoded.header_len, + header_len); + return std::move(staged->stream); + }; + return source; +} + +String stagedBytes(uint64_t header_len, const String & payload, DB::UInt128 tag) +{ + DB::Cas::EnvelopeHeader header; + header.kind = DB::Cas::ObjectKind::Blob; + header.incarnation_tag = tag; + return DB::Cas::encodeEnvelopeHeader(header, static_cast(header_len)) + payload; +} + +} + +TEST(CASS3Staging, StagedCopyCondemnedRetryRetagsBeforeQueuedDelete) +{ + auto backend = std::make_shared( + EtagFaithfulPublicationBackend::FaultScript::CopyLandsThenCondemned); + auto store = DB::Cas::Pool::open( + backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const String payload = "etag-copy-condemned-retry"; + const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); + const String staging_key = "p/staging/mount1/etag-condemned.tmp"; + const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{101}); + backend->putIfAbsent(staging_key, staging_bytes); + DB::Cas::tests::writeMetaClean(*backend, store->layout(), DB::Cas::tests::u128Of(payload), payload.size()); + DB::Cas::tests::condemnMeta(*backend, store->layout(), DB::Cas::tests::u128Of(payload), 31); + auto build = precommittedBuildFor( + store, DB::Cas::RootNamespace{"srv1/etag-condemned"}, "part", + DB::Cas::tests::u128Of(payload), payload.size()); + + build->putBlob( + ref, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + EXPECT_EQ(backend->copy_publications, 1u); + EXPECT_EQ(backend->streaming_publications, 1u); + EXPECT_EQ( + backend->deleteExact(store->layout().blobKey(ref), backend->queued_delete_token).kind, + DB::Cas::DeleteOutcome::Kind::TokenMismatch); + const auto current = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(current.has_value()); + EXPECT_NE(current->bytes, staging_bytes); + EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); +} + +TEST(CASS3Staging, StagedCopyDeletedBeforeAbsentRetryRetagsBeforeQueuedDelete) +{ + auto backend = std::make_shared( + EtagFaithfulPublicationBackend::FaultScript::CopyLandsThenDeletedBeforeAbsentRetry); + auto store = DB::Cas::Pool::open( + backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const String payload = "etag-copy-deleted-before-retry"; + const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); + const String staging_key = "p/staging/mount1/etag-deleted.tmp"; + const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{202}); + backend->putIfAbsent(staging_key, staging_bytes); + auto build = precommittedBuildFor( + store, DB::Cas::RootNamespace{"srv1/etag-deleted"}, "part", + DB::Cas::tests::u128Of(payload), payload.size()); + + build->putBlob( + ref, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + EXPECT_EQ(backend->first_delete.kind, DB::Cas::DeleteOutcome::Kind::Deleted); + EXPECT_EQ(backend->copy_publications, 1u) + << "the absent retry must not copy the original staged envelope again"; + EXPECT_EQ(backend->streaming_publications, 1u); + EXPECT_EQ( + backend->deleteExact(store->layout().blobKey(ref), backend->queued_delete_token).kind, + DB::Cas::DeleteOutcome::Kind::TokenMismatch) + << "the second queued exact delete for the copied ETag must miss the retagged replacement"; + const auto current = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(current.has_value()); + EXPECT_NE(current->bytes, staging_bytes); + EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); +} + +TEST(CASS3Staging, FirstCondemnedAttemptThenAbsentRetryNeverRecopies) +{ + auto backend = std::make_shared( + EtagFaithfulPublicationBackend::FaultScript::FirstCondemnedStreamLandsThenDeleted); + auto store = DB::Cas::Pool::open( + backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + const String payload = "etag-first-condemned-then-absent"; + const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); + const String staging_key = "p/staging/mount1/etag-first-condemned.tmp"; + const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{303}); + backend->putIfAbsent(staging_key, staging_bytes); + backend->putIfAbsent(store->layout().blobKey(ref), staging_bytes); + DB::Cas::tests::writeMetaClean(*backend, store->layout(), DB::Cas::tests::u128Of(payload), payload.size()); + DB::Cas::tests::condemnMeta(*backend, store->layout(), DB::Cas::tests::u128Of(payload), 37); + const DB::Cas::Token original_staged_etag = backend->head(store->layout().blobKey(ref)).token; + auto build = precommittedBuildFor( + store, DB::Cas::RootNamespace{"srv1/etag-first-condemned"}, "part", + DB::Cas::tests::u128Of(payload), payload.size()); + + build->putBlob( + ref, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + EXPECT_EQ(backend->first_delete.kind, DB::Cas::DeleteOutcome::Kind::Deleted); + EXPECT_EQ(backend->copy_publications, 0u) + << "a first condemned publication and every later absent retry must stream, never copy"; + EXPECT_EQ(backend->streaming_publications, 2u); + EXPECT_EQ( + backend->deleteExact(store->layout().blobKey(ref), original_staged_etag).kind, + DB::Cas::DeleteOutcome::Kind::TokenMismatch); + const auto current = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(current.has_value()); + EXPECT_NE(current->bytes, staging_bytes); + EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); +} + +TEST(CASS3Staging, ParsesS3BackendFromConfig) +{ + auto config = configWithDiskSection("s3"); + + EXPECT_EQ(DB::ContentAddressedMetadataStorage::parseStagingBackend(*config, "disk"), DB::Cas::StagingBackend::S3); +} + +TEST(CASS3Staging, DefaultConfigParsesToLocalBackend) +{ + /// No `staging_backend` key at all — the OFF BY DEFAULT arm. + auto config = configWithDiskSection("/tmp/whatever"); + + EXPECT_EQ(DB::ContentAddressedMetadataStorage::parseStagingBackend(*config, "disk"), DB::Cas::StagingBackend::Local); +} + +TEST(CASS3Staging, UnknownBackendValueThrows) +{ + auto config = configWithDiskSection("nfs"); + EXPECT_THROW(DB::ContentAddressedMetadataStorage::parseStagingBackend(*config, "disk"), DB::Exception); +} + +TEST(CASS3Staging, DefaultConstructedStorageReportsLocal) +{ + /// Constructed with no staging-related args at all (mirrors the existing gtest call sites, e.g. + /// gtest_ca_wiring.cpp, which stop at `context_`): the accessors must reflect the same + /// byte-for-byte-current-behavior defaults the config parser produces above. + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "cas_s3_staging_default_scratch"); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + + EXPECT_EQ(storage->stagingBackend(), DB::Cas::StagingBackend::Local); +} + +TEST(CASS3Staging, DefaultObjectStorageRejectsNativeOnlyCopyMode) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + + EXPECT_TRUE(storage->supportsCopyMode(DB::ObjectStorageCopyMode::Default)); + EXPECT_FALSE(storage->supportsCopyMode(DB::ObjectStorageCopyMode::NativeOnly)); +} + +/// Task 4 of the S3-native staging plan: `CaContentWriteBuffer`'s S3-staging constructor streams +/// directly to an already-opened object-store sink while hashing, instead of spilling to a local temp +/// file (see the constructor's doc comment in ContentAddressedWriteBuffers.h). These two tests +/// exercise the buffer directly over a `FakeStagingSink` — no real object storage, disk, or +/// `ContentAddressedTransaction` needed; `writeFile` choosing this mode is exercised together with the +/// promote path in later tasks (S3 mode is off by default and not enabled by any existing test). + +TEST(CASS3Staging, ContentWriteBufferS3ModeStreamsToSinkAndFinalizes) +{ + const std::string staging_key = "staging/mount1/abc123.tmp"; + auto * sink_ptr = new FakeStagingSink(staging_key); + std::unique_ptr sink(sink_ptr); + + std::string got_hash_hex; + size_t got_size = 0; + std::string got_key; + int on_finalized_calls = 0; + + /// S3-native staging fix 2026-07-11: the S3 constructor takes a fixed-length envelope header that is + /// written to the sink FIRST, UNHASHED and excluded from the reported size. A distinctive 256-byte + /// filler stands in for the real CABL header here (this test exercises the buffer mechanics, not the + /// envelope encoder). + const std::string envelope_header(256, 'H'); + + auto buf = std::make_unique( + std::move(sink), + staging_key, + envelope_header, + DB::Cas::BlobHashAlgo::CityHash128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string & hash_hex, size_t size, const std::string & key) + { + ++on_finalized_calls; + got_hash_hex = hash_hex; + got_size = size; + got_key = key; + }); + + /// Write in two chunks (exercises more than one nextImpl flush) and finalize. + const std::string payload_part1(4000, 'x'); + const std::string payload_part2(1234, 'y'); + buf->write(payload_part1.data(), payload_part1.size()); + buf->write(payload_part2.data(), payload_part2.size()); + buf->finalize(); + + const std::string payload = payload_part1 + payload_part2; + + /// (a) the sink received the ENVELOPE HEADER FIRST, then EXACTLY the payload bytes — the staging + /// object holds `[header][payload]` so the promote can stay a verbatim server-side copy. + EXPECT_EQ(sink_ptr->writtenBytes(), envelope_header + payload); + EXPECT_TRUE(sink_ptr->wasFinalizedForTest()); + EXPECT_FALSE(sink_ptr->wasCancelled()); + + /// (b) on_finalized fired exactly once with the correct cityHash128 hex, size, and staging key. + /// The pool-wide content hash is the STREAMING `HashingWriteBuffer` convention (chunked + /// cityHash128, block = 2048 B), which diverges from a one-shot `CityHash_v1_0_2::CityHash128` + /// call for a payload spanning more than one block (see `gtest_cas_part_write.cpp`'s + /// `CopyForwardMultiBlockPayloadVerifies`, which documents and exercises the same divergence). + /// This payload (5234 bytes) spans multiple 2048-byte blocks, so the expected hash must be + /// recomputed with the SAME streaming convention via `HashingReadBuffer`, not a one-shot call. + DB::ReadBufferFromMemory expected_in(payload.data(), payload.size()); + DB::HashingReadBuffer expected_hashing(expected_in); + expected_hashing.ignoreAll(); + const std::string expected_hash_hex = getHexUIntLowercase(expected_hashing.getHash()); + EXPECT_EQ(on_finalized_calls, 1); + EXPECT_EQ(got_hash_hex, expected_hash_hex); + EXPECT_EQ(got_size, payload.size()); + EXPECT_EQ(got_key, staging_key); + EXPECT_EQ(buf->getFileName(), staging_key); +} + +TEST(CASS3Staging, ContentWriteBufferS3ModeCancelCancelsSinkAndSkipsFinalize) +{ + const std::string staging_key = "staging/mount1/cancelled.tmp"; + auto * sink_ptr = new FakeStagingSink(staging_key); + std::unique_ptr sink(sink_ptr); + + bool on_finalized_called = false; + + auto buf = std::make_unique( + std::move(sink), + staging_key, + /*envelope_header=*/std::string(256, 'H'), + DB::Cas::BlobHashAlgo::CityHash128, + /*buf_size=*/8192, + /*use_adaptive_buffer_size=*/false, + /*adaptive_buffer_initial_size=*/0, + [&](const std::string &, size_t, const std::string &) + { + on_finalized_called = true; + }); + + const std::string payload = "some bytes that must never be promoted"; + buf->write(payload.data(), payload.size()); + buf->cancel(); + + /// (c) cancel() before finalize cancels the sink and on_finalized is NEVER called — no partial + /// finalize (no promote-worthy hash/size is ever handed to the transaction for cancelled bytes). + EXPECT_TRUE(sink_ptr->wasCancelled()); + EXPECT_FALSE(sink_ptr->wasFinalizedForTest()); + EXPECT_FALSE(on_finalized_called); + + /// The buffer's destructor calls cancel() again (defensive backstop) — already-cancelled, so this + /// must stay a no-op: still no on_finalized call, and no attempt to fs::remove a remote key. + buf.reset(); + EXPECT_FALSE(on_finalized_called); +} + +/// The ordinary staged cases below pin the same mandatory-`HEAD` selection used by the ETag-faithful +/// regressions: native verbatim copy only after a first absent observation, no publication for a live +/// body, and retagged streaming for `Condemned`. + +/// (a) Fresh blob key ⇒ the first-plus-absent native copy publishes verbatim and records `Materialized`. +TEST(CASS3Staging, PromoteViaServerSideCopyCreatesFreshBlobMaterializedProof) +{ + auto backend = std::make_shared(); + auto store = openStagingPool(backend); + const DB::Cas::RootNamespace ns{"srv1/nsA"}; + const std::string ref = "part_a"; + + const std::string payload(300, 'a'); + const DB::UInt128 hash = DB::Cas::tests::u128Of(payload); + const DB::Cas::BlobRef blob_id{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; + const std::string blob_key = store->layout().blobKey(blob_id); + const std::string staging_key = "p/staging/mount1/aaa.tmp"; + const std::string staging_bytes = stagedBytes( + store->poolMeta().blob_header_len, payload, DB::UInt128{0xA}); + backend->putIfAbsent(staging_key, staging_bytes); + + auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); + const DB::Cas::PutBlobResult bref = build->putBlob( + blob_id, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + /// Exactly one native verbatim publication from staging to the blob key. + ASSERT_EQ(backend->copy_calls.size(), 1u); + EXPECT_TRUE(backend->copy_calls[0].server_side_copy); + EXPECT_EQ(backend->copy_calls[0].from, staging_key); + EXPECT_EQ(backend->copy_calls[0].to, blob_key); + EXPECT_EQ(backend->streamingPublicationCount(), 0u); + + /// Successful publication records materialized evidence; the backend still owns the destination token. + EXPECT_EQ(build->dependencyProof(blob_id), DB::Cas::BlobDependencyProof::Materialized); + const DB::Cas::HeadResult hr = backend->head(blob_key); + ASSERT_TRUE(hr.exists); + EXPECT_FALSE(hr.token.empty()); + EXPECT_EQ(bref.size, payload.size()); + + /// The promoted blob body IS the staging bytes (server-side copy moved them verbatim). + const auto got = backend->get(blob_key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, staging_bytes); +} + +/// (b) Blob key already exists and is `Clean` ⇒ the writer observes it without publication. +TEST(CASS3Staging, PromoteOverExistingCleanBlobAdoptsAndNeverOverwrites) +{ + auto backend = std::make_shared(); + auto store = openStagingPool(backend); + const DB::Cas::RootNamespace ns{"srv1/nsB"}; + const std::string ref = "part_b"; + + const std::string payload(300, 'b'); + const DB::UInt128 hash = DB::Cas::tests::u128Of(payload); + const DB::Cas::BlobRef blob_id{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; + const std::string blob_key = store->layout().blobKey(blob_id); + const std::string staging_key = "p/staging/mount1/bbb.tmp"; + backend->putIfAbsent( + staging_key, + stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{0xB})); + + /// A pre-existing, well-formed, CLEAN blob (envelope + payload) already at the content key. + backend->putIfAbsent( + blob_key, + stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{0xBB})); + DB::Cas::tests::writeMetaClean(*backend, store->layout(), hash, payload.size()); + const DB::Cas::HeadResult before = backend->head(blob_key); + ASSERT_TRUE(before.exists); + + auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); + build->putBlob( + blob_id, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + /// Mandatory `HEAD` observes the live body, so no transport call is made. + EXPECT_TRUE(backend->copy_calls.empty()); + EXPECT_EQ(backend->streamingPublicationCount(), 0u); + + /// The existing incarnation is untouched: same token, same bytes. + const DB::Cas::HeadResult after = backend->head(blob_key); + EXPECT_EQ(after.token, before.token); + + /// Observing the existing incarnation records materialized evidence without retaining its token. + EXPECT_EQ(build->dependencyProof(blob_id), DB::Cas::BlobDependencyProof::Materialized); +} + +/// (c) Blob key exists but is CONDEMNED ⇒ the writer republishes its OWN staging PAYLOAD +/// under a FRESH-tagged envelope header — NEVER a read/copy of the condemned blob key +/// and the replacement body DIFFERS from the condemned incarnation +/// (INV-NO-RETURN: a verbatim copy would reproduce identical bytes ⇒ identical ETag ⇒ the queued +/// exact-token delete of the condemned incarnation would kill the live resurrection = data loss). +TEST(CASS3Staging, PublishOverCondemnedBlobUsesFreshTagNotVerbatim) +{ + auto backend = std::make_shared(); + auto store = openStagingPool(backend); + const DB::Cas::RootNamespace ns{"srv1/nsC"}; + const std::string ref = "part_c"; + + const std::string payload(300, 'c'); + const DB::UInt128 hash = DB::Cas::tests::u128Of(payload); + const DB::Cas::BlobRef blob_id{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; + const std::string blob_key = store->layout().blobKey(blob_id); + const std::string staging_key = "p/staging/mount1/ccc.tmp"; + + /// The staging object holds `[header][payload]` (as `writeFile` now emits it). The staging header is + /// a fixed 256-byte CABL envelope with its OWN incarnation_tag. + DB::Cas::EnvelopeHeader staging_h; + staging_h.kind = DB::Cas::ObjectKind::Blob; + staging_h.incarnation_tag = DB::UInt128(0xC0FFEE); /// the create-time tag + const std::string staging_header = DB::Cas::encodeEnvelopeHeader( + staging_h, static_cast(store->poolMeta().blob_header_len)); + ASSERT_EQ(staging_header.size(), store->poolMeta().blob_header_len); + const std::string staging_bytes = staging_header + payload; + backend->putIfAbsent(staging_key, staging_bytes); + + /// Seed the condemned blob body = EXACTLY what a verbatim promote of this staging object would have + /// produced (the writer's OWN create, later observed condemned). This is the adversarial shape: a + /// verbatim republication WOULD reproduce these identical bytes ⇒ identical ETag ⇒ collision. + backend->putIfAbsent(blob_key, staging_bytes); + DB::Cas::tests::writeMetaClean(*backend, store->layout(), hash, /*size=*/payload.size()); + DB::Cas::tests::condemnMeta(*backend, store->layout(), hash, /*condemn_round=*/5); + const DB::Cas::HeadResult before = backend->head(blob_key); + ASSERT_TRUE(before.exists); + + auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); + build->putBlob( + blob_id, + reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); + + /// A present `Condemned` destination selects exactly one retagged streaming publication. It never + /// attempts the verbatim-copy transport. + ASSERT_EQ(backend->copy_calls.size(), 1u); + EXPECT_FALSE(backend->copy_calls[0].server_side_copy); + EXPECT_EQ(backend->copy_calls[0].to, blob_key); + /// INV: republication reads the STAGING object and never the condemned blob key. Asserted on the + /// reads themselves rather than on a source argument, because the caller now opens the reader. + EXPECT_GT(backend->reads_of[staging_key], 0u) << "republication must read the writer's own staging object"; + EXPECT_EQ(backend->reads_of[blob_key], 0u) << "the condemned blob key must never be read"; + EXPECT_EQ(backend->streamingPublicationCount(), 1u); + + /// The incarnation token is REFRESHED (a fresh incarnation displaced the condemned one). + const DB::Cas::HeadResult after = backend->head(blob_key); + EXPECT_NE(after.token, before.token); + ASSERT_TRUE(after.exists); + + const auto got = backend->get(blob_key); + ASSERT_TRUE(got.has_value()); + const uint64_t header_len = store->poolMeta().blob_header_len; + + /// INV-NO-RETURN — THE fresh-tag property: the replacement body is NOT byte-identical to the + /// condemned incarnation (a verbatim copy would have been). The PAYLOAD is preserved exactly (the + /// writer read it from OUR staging object, skipping the staging header), but the envelope HEADER + /// differs — the writer minted a FRESH incarnation_tag — so on a real content-addressed store the + /// replacement ETag differs and the queued exact-token delete of the condemned incarnation cannot + /// match the live replacement. + EXPECT_NE(got->bytes, staging_bytes); + ASSERT_GE(got->bytes.size(), header_len); + EXPECT_EQ(got->bytes.substr(header_len), payload); /// payload preserved + EXPECT_NE(got->bytes.substr(0, header_len), staging_header); /// header freshly re-tagged + + /// The republication recorded materialized evidence and flipped the meta back to `Clean`. + EXPECT_EQ(build->dependencyProof(blob_id), DB::Cas::BlobDependencyProof::Materialized); + const auto lm = DB::Cas::tests::loadMetaForTest(*backend, store->layout(), hash); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, DB::Cas::MetaState::Clean); +} + +/// =========================================================================================== +/// Task 6 of the S3-native staging plan: staging cleanup after commit, read-your-writes over an S3 +/// pending blob, and the mount-lease-scoped sweeper (`CASStagingSweeper.h`). +/// +/// The wiring-level tests below drive the real metadata storage and transaction over a local test +/// store that advertises native copy. Its real `getType` stays `Local`, so the CAS core uses +/// `EmulatedSingleProcess`; these cases stop before a native-mode staged publication is required. + +namespace +{ + +/// Construct a `ContentAddressedMetadataStorage` with `staging_backend=s3` over `object_storage`, +/// mirroring `DefaultConstructedStorageReportsLocal`'s settings defaults for every +/// field this test suite does not care about — only `server_root_id` (the mount identity that names +/// the staging prefix) and `staging_backend` differ. +std::shared_ptr makeS3StagingMetadataStorageForTest( + const DB::ObjectStoragePtr & object_storage, const std::string & server_root_id) +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto scratch = std::filesystem::temp_directory_path() + / ("cas_s3_staging_wiring_" + server_root_id + "_" + unique); + auto settings = DB::Cas::tests::makeSettingsForTest(server_root_id, scratch); + settings[DB::ContentAddressedSetting::staging_backend] = "s3"; + settings.validate(); + return std::make_shared( + object_storage, "pool", "srv1", /*disk_name_=*/"", /*context_=*/nullptr, settings); +} + +/// Mirrors gtest_ca_wiring.cpp's helper of the same shape. +void writeThroughS3Transaction(DB::ContentAddressedTransaction & tx, const std::string & path, const std::string & bytes) +{ + auto buf = tx.writeFile(path, 65536, DB::WriteMode::Rewrite, {}); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); +} + +} + +TEST(CASS3Staging, WritableS3StagingRequiresNativeOnlyCopy) +{ + auto object_storage = makeFakeNativeCopyStorage(/*native_only_copy_supported=*/true); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountNative"); + metadata_storage->startup(); + + auto tx = metadata_storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + writeThroughS3Transaction( + ca_tx, + "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", + "native-staging"); + + DB::RelativePathsWithMetadata staged; + object_storage->listObjects(metadata_storage->stagingKeyPrefix(), staged, /*max_keys=*/0); + EXPECT_EQ(staged.size(), 1u); +} + +TEST(CASS3Staging, UnsupportedNativeOnlyCopyDoesNotFallBackToLocal) +{ + auto object_storage = makeFakeNativeCopyStorage(/*native_only_copy_supported=*/false); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountUnsupported"); + + DB::Cas::tests::expectThrowsCodeWithMessage( + DB::ErrorCodes::NOT_IMPLEMENTED, + "cas_staging_backend=s3", + [&] + { + metadata_storage->startup(); + }); +} + +/// (a) A successful commit removes the S3 staging object of a pending blob it staged. Uses the B189 +/// orphan shape (the pending blob's entry is unlinked before commit) so `publishStaging` never calls +/// `putBlob` for it — only `cleanupPendingTempFiles`'s Task 6 branch ever touches this staging object, +/// which is exactly the seam this test targets. +TEST(CASS3Staging, SuccessfulCommitRemovesOrphanedS3StagingObject) +{ + auto object_storage = makeFakeNativeCopyStorage(/*native_only_copy_supported=*/true); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountA"); + metadata_storage->startup(); + + auto tx = metadata_storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + + /// orphan.bin forces the S3-staging blob path (a ".bin" suffix always stays a blob, per + /// `partFileMustStayBlob`); it is unlinked below before commit. + writeThroughS3Transaction(ca_tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin", std::string(300, 'x')); + /// checksums.txt is small and NOT blob-forcing: an INLINE entry that gives the part's PartWriteTxn a real + /// (non-orphaned) manifest entry, so `publishStaging` takes its normal path (not the early-return + /// mutable-only/no-PartWriteTxn branch). + writeThroughS3Transaction(ca_tx, "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/checksums.txt", "sums"); + + DB::RelativePathsWithMetadata staged_before; + object_storage->listObjects(metadata_storage->stagingKeyPrefix(), staged_before, /*max_keys=*/0); + ASSERT_EQ(staged_before.size(), 1u) << "exactly orphan.bin's S3 staging object should exist pre-commit"; + + tx->unlinkFile("a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/orphan.bin", false, false); + + tx->commit(DB::NoCommitOptions{}); + + DB::RelativePathsWithMetadata staged_after; + object_storage->listObjects(metadata_storage->stagingKeyPrefix(), staged_after, /*max_keys=*/0); + EXPECT_TRUE(staged_after.empty()) + << "cleanupPendingTempFiles must remove the orphaned S3 staging object after a successful commit"; +} + +/// (b) Read-your-writes over an S3 pending blob (before commit) returns the staged bytes from the S3 +/// staging object, not a local temp file. +TEST(CASS3Staging, ReadYourWritesReturnsStagedBytesFromS3StagingObject) +{ + auto object_storage = makeFakeNativeCopyStorage(/*native_only_copy_supported=*/true); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountB"); + metadata_storage->startup(); + + auto tx = metadata_storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + + const std::string path = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"; + const std::string payload(5000, 'z'); + writeThroughS3Transaction(ca_tx, path, payload); + + auto read_buf = tx->tryReadFileInFlight(path, DB::ReadSettings{}, {}); + ASSERT_NE(read_buf, nullptr); + std::string got; + DB::readStringUntilEOF(got, *read_buf); + EXPECT_EQ(got, payload); +} + +/// (c) `sweepOwnMountStaging` removes only objects under the given mount prefix and leaves a DIFFERENT +/// mount's staging objects untouched (the lease-fence — `CASStagingSweeper.h`). +TEST(CASStagingSweeper, RemovesOnlyObjectsUnderGivenMountPrefix) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + const std::string root = storage->getCommonKeyPrefix(); + + auto put = [&](const std::string & key, const std::string & bytes) + { + auto buf = storage->writeObject(DB::StoredObject(key), DB::WriteMode::Rewrite); + buf->write(bytes.data(), bytes.size()); + buf->finalize(); + }; + + put(root + "/p/staging/mountA/one.tmp", "a1"); + put(root + "/p/staging/mountA/two.tmp", "a2"); + put(root + "/p/staging/mountB/three.tmp", "b1"); /// a DIFFERENT mount's staging — must survive + + DB::Cas::sweepOwnMountStaging(*storage, root + "/p/staging/mountA/"); + + EXPECT_FALSE(storage->exists(DB::StoredObject(root + "/p/staging/mountA/one.tmp"))); + EXPECT_FALSE(storage->exists(DB::StoredObject(root + "/p/staging/mountA/two.tmp"))); + EXPECT_TRUE(storage->exists(DB::StoredObject(root + "/p/staging/mountB/three.tmp"))); +} + +/// (d) GC's blob-discovery LISTs ONLY `Layout::blobsPrefix()` (`/blobs/`) — a top-level prefix +/// strictly disjoint from the S3-staging area (`/staging//`), so a staging object can +/// never be listed, HEAD'd, or condemned as an orphan blob by GC's fold (`CasGc.cpp`, `CasFsck.cpp`). +/// This is a prefix-separation assertion (the GC fold itself is not unit-testable in isolation from a +/// full round — see `gtest_cas_gc_fold.cpp` for that machinery); it pins the invariant a refactor that +/// nested `staging/` under `blobs/` (or vice versa) would violate. +TEST(CASS3Staging, GcBlobDiscoveryPrefixExcludesStagingObjects) +{ + const DB::Cas::Layout layout("p"); + const std::string blobs_prefix = layout.blobsPrefix(); + const std::string staging_prefix = "p/staging/mountA/"; + const std::string staging_key = staging_prefix + "aaa.tmp"; + + EXPECT_EQ(blobs_prefix, "p/blobs/"); + EXPECT_FALSE(staging_prefix.starts_with(blobs_prefix)); + EXPECT_FALSE(blobs_prefix.starts_with(staging_prefix)); + EXPECT_FALSE(staging_key.starts_with(blobs_prefix)); +} + +#if USE_AWS_S3 + +namespace +{ + +/// A `LocalObjectStorage` that reports the GCS generation dialect +/// (`conditionalOpsUseGenerationTokens() == true`) and a non-`Local` `getType()`, so +/// `ContentAddressedMetadataStorage::openPoolView` builds its backend in `Mode::Native` with +/// `native_token_type == TokenType::Generation`. The fake also advertises native copy so generation +/// token mode can exercise explicit S3 staging without endpoint/provider heuristics. +/// +/// Holds every object entirely in memory, keyed by the BARE CAS key exactly as `Backend` hands it to +/// `object_storage` (e.g. `"pool/_probe//token"`). Native mode never asks `object_storage` to +/// resolve that key against anything (`ContentAddressedMetadataStorage::physicalKey` is a documented +/// no-op for Native, since a real S3 client resolves a bucket-relative key against its own bucket/prefix +/// configuration internally) -- so this fake never needs a notion of "resolve a key to a location" at +/// all, unlike a real filesystem-backed object storage would. That sidesteps the class of bug a +/// resolve-then-strip round trip through the real `LocalObjectStorage` file/list implementation is prone +/// to (a key is a key, with no round trip to get wrong), and it never touches the real filesystem, so it +/// cannot leak files into the test process's working directory either. +/// +/// A writable Native-mode mount always runs the mandatory capability battery (`CasProbe.cpp`), which +/// requires REAL conditional-write enforcement: the precondition is evaluated when a write completes, +/// signaled by an `S3Exception` carrying the canonical `PreconditionFailed` name (see +/// `ObjectStorageBackend::finalizeConditionalWrite`). `writeObject` therefore buffers bytes in memory and +/// defers both the precondition check and the commit to `finalize`, mirroring how a real object store +/// only commits -- and only then can reject -- on PUT completion. +class FakeGenerationObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + DB::ObjectStorageType getType() const override { return DB::ObjectStorageType::S3; } + bool conditionalOpsUseGenerationTokens() const override { return true; } + std::optional isBucketVersioningEnabled() const override { return false; } + bool supportsRetryProfile(DB::ObjectStorageRetryProfile) const override { return true; } + bool supportsCopyMode(DB::ObjectStorageCopyMode mode) const override + { + return mode == DB::ObjectStorageCopyMode::Default || mode == DB::ObjectStorageCopyMode::NativeOnly; + } + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional /*attributes*/, + size_t /*buf_size*/, + const DB::WriteSettings & write_settings) override + { + if (mode != DB::WriteMode::Rewrite) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "FakeGenerationObjectStorage only supports Rewrite"); + return std::make_unique( + *this, object.remote_path, write_settings.object_storage_write_if_none_match, + write_settings.object_storage_write_if_match); + } + + bool exists(const DB::StoredObject & object) const override + { + std::lock_guard lock(mutex); + return objects.contains(object.remote_path); + } + + std::unique_ptr readObject( + const DB::StoredObject & object, + const DB::ReadSettings & /*read_settings*/, + std::optional /*read_hint*/, + bool /*use_external_buffer*/, + bool /*restrict_seek*/) const override + { + std::lock_guard lock(mutex); + auto it = objects.find(object.remote_path); + if (it == objects.end()) + /// `RESOURCE_NOT_FOUND`, not a plain `DB::Exception`: `Backend::probeSentinelRaw`'s Native + /// path (`CasObjectStorageBackend.cpp`) classifies absence ONLY from a caught `S3Exception` + /// carrying `NO_SUCH_KEY`/`RESOURCE_NOT_FOUND` (or the matching exception name) -- anything + /// else, including an unrecognized exception TYPE, falls through to `ProbeOutcome:: + /// Indeterminate` (fail-closed). A bodyless HEAD on a real absent S3 key throws exactly this + /// code, since the SDK cannot parse a `` from a body that was never sent. + throw DB::S3Exception("FakeGenerationObjectStorage: object does not exist", + Aws::S3::S3Errors::RESOURCE_NOT_FOUND); + /// Copies the bytes, matching a real remote read (no shared ownership with the stored entry, so + /// a later overwrite of this key cannot mutate bytes a caller is still reading). + return std::make_unique(object.remote_path, it->second.bytes); + } + + DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override + { + auto metadata = tryGetObjectMetadata(path, with_tags); + if (!metadata) + throw DB::S3Exception("FakeGenerationObjectStorage: object does not exist", + Aws::S3::S3Errors::RESOURCE_NOT_FOUND); + return *metadata; + } + + std::optional tryGetObjectMetadata(const std::string & path, bool /*with_tags*/) const override + { + std::lock_guard lock(mutex); + auto it = objects.find(path); + if (it == objects.end()) + return std::nullopt; + DB::ObjectMetadata metadata; + metadata.size_bytes = it->second.bytes.size(); + metadata.etag = std::to_string(it->second.generation); + return metadata; + } + + std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const override + { + return tryGetObjectMetadata(path, with_tags); + } + + void removeObjectIfExists(const DB::StoredObject & object) override + { + std::lock_guard lock(mutex); + objects.erase(object.remote_path); + } + + void removeObjectsIfExist(const DB::StoredObjects & objects_to_remove) override + { + std::lock_guard lock(mutex); + for (const auto & object : objects_to_remove) + objects.erase(object.remote_path); + } + + /// `path` is a bare CAS-relative prefix (e.g. `"pool/"` or `"pool/_probe/"`) -- exactly what + /// `Backend::list`'s Native-mode path passes through unchanged (it applies no prefix stripping of + /// its own there) and expects back on every listed key. A plain string-prefix scan over the + /// in-memory keys already IS that key space, so there is no separate "physical" representation to + /// resolve to or strip back off. + void listObjects(const std::string & path, DB::RelativePathsWithMetadata & children, size_t max_keys) const override + { + std::lock_guard lock(mutex); + for (const auto & [key, entry] : objects) + { + if (!key.starts_with(path)) + continue; + DB::ObjectMetadata metadata; + metadata.size_bytes = entry.bytes.size(); + metadata.etag = std::to_string(entry.generation); + children.push_back(std::make_shared(key, std::move(metadata))); + if (max_keys != 0 && children.size() >= max_keys) + break; + } + } + + DB::ConditionalRemoveResult removeObjectIfTokenMatches(const DB::StoredObject & object, const std::string & etag) override + { + std::lock_guard lock(mutex); + DB::ConditionalRemoveResult result; + auto it = objects.find(object.remote_path); + if (it == objects.end()) + { + result.outcome = DB::ConditionalRemoveOutcome::NotFound; + return result; + } + if (std::to_string(it->second.generation) != etag) + { + result.outcome = DB::ConditionalRemoveOutcome::TokenMismatch; + return result; + } + objects.erase(it); + result.outcome = DB::ConditionalRemoveOutcome::Removed; + return result; + } + + /// Checks the write-once/exact-token precondition against the current generation and, on success, + /// stores `bytes` and mints the next generation. Throws an `S3Exception` naming `PreconditionFailed` + /// on a lost condition -- the one signal `finalizeConditionalWrite` classifies as + /// `PutOutcome::PreconditionFailed` rather than an ordinary failure. + void commitConditionalWrite(const std::string & key, const std::string & bytes, + const std::string & if_none_match, const std::string & if_match) + { + std::lock_guard lock(mutex); + auto it = objects.find(key); + const bool exists_now = it != objects.end(); + if (!if_none_match.empty() && exists_now) + throw DB::S3Exception("FakeGenerationObjectStorage: if-none-match precondition failed", + Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); + if (!if_match.empty() && (!exists_now || std::to_string(it->second.generation) != if_match)) + throw DB::S3Exception("FakeGenerationObjectStorage: if-match precondition failed", + Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); + + objects[key] = Entry{bytes, next_generation++}; + } + +private: + struct Entry + { + std::string bytes; + uint64_t generation; + }; + + /// Buffers the whole body in memory (like `FakeStagingSink` above) so the entry is committed exactly + /// once, at `finalize`, and only after the precondition has been checked. + class ConditionalWriteBuffer final : public DB::WriteBufferFromFileBase + { + public: + ConditionalWriteBuffer(FakeGenerationObjectStorage & storage_, std::string key_, + std::string if_none_match_, std::string if_match_) + : DB::WriteBufferFromFileBase(/*buf_size=*/8192, nullptr, 0) + , storage(storage_), key(std::move(key_)) + , if_none_match(std::move(if_none_match_)), if_match(std::move(if_match_)) + { + } + + void sync() override {} + std::string getFileName() const override { return key; } + + protected: + void nextImpl() override + { + if (!offset()) + return; + buffered.append(working_buffer.begin(), offset()); + } + + void finalizeImpl() override + { + next(); + storage.commitConditionalWrite(key, buffered, if_none_match, if_match); + } + + private: + FakeGenerationObjectStorage & storage; + std::string key; + std::string if_none_match; + std::string if_match; + std::string buffered; + }; + + mutable std::mutex mutex; + std::map objects; + uint64_t next_generation = 1; +}; + +std::shared_ptr makeFakeGenerationObjectStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_s3_staging_generation_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +} + +TEST(CASS3Staging, GenerationBackendMayUseNativeOnlyCopy) +{ + auto object_storage = makeFakeGenerationObjectStorageForTest(); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountGenerationNative"); + metadata_storage->startup(); + + auto tx = metadata_storage->createTransaction(); + auto & ca_tx = dynamic_cast(*tx); + writeThroughS3Transaction( + ca_tx, + "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin", + "generation-native-staging"); + + DB::RelativePathsWithMetadata staged; + object_storage->listObjects(metadata_storage->stagingKeyPrefix(), staged, /*max_keys=*/0); + EXPECT_EQ(staged.size(), 1u); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_sentinel_probe.cpp b/src/Disks/tests/gtest_cas_sentinel_probe.cpp new file mode 100644 index 000000000000..05411f0c3848 --- /dev/null +++ b/src/Disks/tests/gtest_cas_sentinel_probe.cpp @@ -0,0 +1,263 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +#if USE_AWS_S3 +#include +#endif + +using namespace DB::Cas; + +/// Task 3 (spec §2): the typed sentinel probe below `Backend` must never conflate a transport error +/// with absence. These tests exercise the free-function entry point `probeSentinel` against the generic +/// `Backend::probeSentinelRaw` default (via `InMemoryBackend`, the "Emulated"-style in-memory backend used +/// by CAS tests) and against `ObjectStorageBackend`'s `EmulatedSingleProcess` override, which is the REAL +/// production mode for a content-addressed disk over `object_storage_type=local`. + +namespace +{ + +using DB::Cas::tests::nativeKeyUnder; + +/// A Backend decorator whose head/get/list all throw an untyped runtime error when armed — modelling +/// a backend with no sharper evidence than "something went wrong" (a network timeout, a 5xx, an +/// unclassifiable failure). Mirrors the existing MetaWriteFaultBackend fault-injection pattern +/// (cas_test_helpers.h): every other operation delegates to InMemoryBackend unchanged. +class TransportFaultBackend final : public InMemoryBackend +{ +public: + /// Unhide the base convenience overloads, matching every other Backend subclass in this suite. + using Backend::get; + using Backend::getStream; + using Backend::putIfAbsent; + using Backend::putOverwrite; + using Backend::casPut; + + HeadResult head(const String & key) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::head(key); + } + + std::optional get(const String & key, Range range) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::get(key, range); + } + + ListPage list(const String & prefix, const String & cursor, size_t limit) override + { + if (fail.load()) + throw std::runtime_error("injected fault: transport error"); + return InMemoryBackend::list(prefix, cursor, limit); + } + + std::atomic fail{true}; +}; + +} + +/// (a) A present key probes Present and carries the materialized body. +TEST(CASSentinelProbe, PresentKeyReturnsPresentWithBody) +{ + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("k", "hello").outcome, PutOutcome::Done); + + const auto result = probeSentinel(backend, "k"); + EXPECT_EQ(result.outcome, ProbeOutcome::Present); + ASSERT_TRUE(result.body.has_value()); + EXPECT_EQ(*result.body, "hello"); +} + +/// (b) A deleted (never-written) key probes KeyAbsent while the container/backend is otherwise alive. +TEST(CASSentinelProbe, AbsentKeyWithContainerAliveReturnsKeyAbsent) +{ + InMemoryBackend backend; + ASSERT_EQ(backend.putIfAbsent("other", "x").outcome, PutOutcome::Done); // proves the backend is alive + + const auto result = probeSentinel(backend, "missing"); + EXPECT_EQ(result.outcome, ProbeOutcome::KeyAbsent); + EXPECT_FALSE(result.body.has_value()); +} + +/// (c) `ObjectStorageBackend::EmulatedSingleProcess` is the REAL production backend for a +/// content-addressed disk over `object_storage_type=local` (ContentAddressedMetadataStorage.cpp +/// selects it whenever the underlying storage is Local). Removing the WHOLE configured container +/// directory (the disk root) must probe `ContainerAbsent`, distinct from an ordinary absent key — +/// `LocalObjectStorage::listObjects` silently reports zero children for BOTH a missing directory and +/// an empty one, so the distinction only exists because `probeSentinelRaw` stats the container first. +TEST(CASSentinelProbe, ContainerDirectoryRemovedReturnsContainerAbsent) +{ + auto storage = tests::makeLocalObjectStorageForTest(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + + ASSERT_EQ(backend.putIfAbsent("k", "hello").outcome, PutOutcome::Done); + + /// Sanity, container alive: Present vs. KeyAbsent are genuinely distinct before we remove anything. + EXPECT_EQ(probeSentinel(backend, "k").outcome, ProbeOutcome::Present); + EXPECT_EQ(probeSentinel(backend, "missing").outcome, ProbeOutcome::KeyAbsent); + + std::filesystem::remove_all(storage->getCommonKeyPrefix()); + + const auto result = probeSentinel(backend, "k"); + EXPECT_EQ(result.outcome, ProbeOutcome::ContainerAbsent); + EXPECT_FALSE(result.body.has_value()); +} + +/// `ObjectStorageBackend::Mode::Native` over a plain `LocalObjectStorage` (the same construction +/// `gtest_cas_backend.cpp`'s Native-mode tests use to exercise the Native code path without a live S3 +/// endpoint): a present key must probe `Present` and carry the materialized body via the raw-HEAD -> +/// `get` path, not just the EmulatedSingleProcess path already covered above. +TEST(CASSentinelProbe, NativePresentKeyReturnsPresentWithBody) +{ + auto storage = tests::makeLocalObjectStorageForTest(); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + const String key = nativeKeyUnder(storage, "some/key"); + + ASSERT_EQ(backend.putIfAbsent(key, "native body").outcome, PutOutcome::Done); + + const auto result = probeSentinel(backend, key); + EXPECT_EQ(result.outcome, ProbeOutcome::Present); + ASSERT_TRUE(result.body.has_value()); + EXPECT_EQ(*result.body, "native body"); +} + +/// (d) A backend forced to throw a transport error must probe Indeterminate — NEVER KeyAbsent, even +/// though the failure looks superficially like "nothing there" from the caller's point of view. +TEST(CASSentinelProbe, TransportErrorNeverClassifiesAsAbsent) +{ + TransportFaultBackend backend; + const auto result = probeSentinel(backend, "k"); + EXPECT_EQ(result.outcome, ProbeOutcome::Indeterminate); + EXPECT_FALSE(result.body.has_value()); +} + +#if USE_AWS_S3 + +namespace +{ + +/// A `LocalObjectStorage` whose `getObjectMetadata` can be armed to throw a configurable synthetic +/// `S3Exception` — the same technique `gtest_cas_backend.cpp`'s `NativeReadThrowsNoSuchKeyObjectStorage` +/// uses to exercise S3 error codes without a live S3 endpoint. Constructing `ObjectStorageBackend` in +/// `Mode::Native` over this fake is the established pattern for testing the Native/S3 raw-error classifier +/// in isolation (see also `gtest_cas_backend.cpp`'s `NativeRejectsWrongDialectTokenBeforeTouchingTheWire`). +class ThrowingS3MetadataObjectStorage final : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + void throwOnGetObjectMetadata(Aws::S3::S3Errors code) { metadata_error = code; } + + DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override + { + if (metadata_error) + throw DB::S3Exception("injected fault: " + path, *metadata_error); + return DB::LocalObjectStorage::getObjectMetadata(path, with_tags); + } + +private: + std::optional metadata_error; +}; + +DB::ObjectStoragePtr makeThrowingS3MetadataStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_sentinel_probe_unit_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + +} + +/// The full S3 IAM permutation table (spec §2): a raw NO_SUCH_KEY/NO_SUCH_BUCKET/ACCESS_DENIED HEAD +/// error must classify EXACTLY, and anything unmodeled must fail closed to Indeterminate. +TEST(CASSentinelProbe, NativeClassifiesNoSuchKeyAsKeyAbsent) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_KEY); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::KeyAbsent); +} + +/// A real S3 HEAD's 404 has no response body, so the SDK cannot parse a `NoSuchKey` `` and +/// instead derives `RESOURCE_NOT_FOUND` straight from the HTTP status (see `isNotFoundError`, +/// `src/IO/S3/getObjectInfo.cpp`) — THIS is the code a genuinely absent key throws on real S3, not +/// `NO_SUCH_KEY`. Without classifying it, every real-S3 absence would be `Indeterminate` forever. +TEST(CASSentinelProbe, NativeClassifiesResourceNotFoundAsKeyAbsent) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::RESOURCE_NOT_FOUND); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::KeyAbsent); +} + +TEST(CASSentinelProbe, NativeClassifiesNoSuchBucketAsContainerAbsent) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_BUCKET); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::ContainerAbsent); +} + +TEST(CASSentinelProbe, NativeClassifiesAccessDeniedAsAccessDenied) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::ACCESS_DENIED); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::AccessDenied); +} + +TEST(CASSentinelProbe, NativeClassifiesUnmodeledErrorAsIndeterminate) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::SERVICE_UNAVAILABLE); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + + EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::Indeterminate); +} + +/// Production wiring (`Pool::open`) ALWAYS wraps the real backend in `InstrumentedBackend` before +/// anything calls it. `InstrumentedBackend` must forward `probeSentinelRaw` to `inner`, not fall +/// through to `Backend::probeSentinelRaw`'s generic head/get-based default — the default would derive +/// its answer from THIS object's own (correctly delegating, but non-typed) `head`/`get` overrides, +/// silently discarding `ObjectStorageBackend`'s real S3-error classification. NO_SUCH_BUCKET is chosen +/// deliberately: the generic default cannot produce `ContainerAbsent` at all (it only ever returns +/// Present/KeyAbsent/Indeterminate), so this test can ONLY pass if the typed override is actually +/// reached through the wrapper. +TEST(CASSentinelProbe, InstrumentedBackendForwardsToInnerClassification) +{ + auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); + storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_BUCKET); + auto inner = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + InstrumentedBackend instrumented(inner); + + EXPECT_EQ(probeSentinel(instrumented, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::ContainerAbsent); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_server_root_format.cpp b/src/Disks/tests/gtest_cas_server_root_format.cpp new file mode 100644 index 000000000000..c2d1474dc29b --- /dev/null +++ b/src/Disks/tests/gtest_cas_server_root_format.cpp @@ -0,0 +1,152 @@ +#include "cas_format_test_battery.h" +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +TEST(CASFormatBattery, Owner) +{ + OwnerObject o; + o.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); + const String golden = currentFormatHeader("cas_owner") + + "{\"su\":\"0123456789abcdeffedcba9876543210\"}\n"; + EXPECT_EQ(encodeOwner(o), golden); + EXPECT_FALSE(decodeOwner(golden).retired_at_ms.has_value()); + runFormatBattery({FormatId::Owner, + [&] { return sealObject(FormatId::Owner, encodeOwner(o)); }, + [](std::string_view s) { decodeOwner(std::string(openObject(FormatId::Owner, s))); }, + golden}); +} + +TEST(CASOwnerFormat, RetiredAtRoundTrip) +{ + OwnerObject o; + o.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); + o.retired_at_ms = 1752537600000ULL; + + const OwnerObject back = decodeOwner(encodeOwner(o)); + EXPECT_EQ(back.server_uuid, o.server_uuid); + EXPECT_EQ(back.retired_at_ms, o.retired_at_ms); +} + +TEST(CASFormatBattery, ServerEpoch) +{ + ServerEpoch e; + e.next_writer_epoch = 7; + runFormatBattery({FormatId::ServerEpoch, + [&] { return sealObject(FormatId::ServerEpoch, encodeServerEpoch(e)); }, + [](std::string_view s) { decodeServerEpoch(std::string(openObject(FormatId::ServerEpoch, s))); }, + currentFormatHeader("cas_epoch") + "{\"nwe\":\"7\"}\n"}); +} + +TEST(CASFormatBattery, MountLease) +{ + MountLease m{hexToU128("0123456789abcdeffedcba9876543210"), 7, "host-1", 4242, + 1752537600000ULL, 5, 1752537630000ULL, 9, false, + hexToU128("00112233445566778899aabbccddeeff")}; + runFormatBattery({FormatId::MountLease, + [&] { return sealObject(FormatId::MountLease, encodeMountLease(m)); }, + [](std::string_view s) { decodeMountLease(std::string(openObject(FormatId::MountLease, s))); }, + currentFormatHeader("cas_mount_lease") + + "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"host-1\",\"pid\":4242," + "\"sat\":1752537600000,\"seq\":\"5\",\"eat\":1752537630000,\"ma\":\"9\",\"fen\":false," + "\"write_attempt_id\":\"00112233445566778899aabbccddeeff\"}\n"}); +} + +TEST(CASMountLeaseFormat, FarewellSentinelAndFencedSurvive) +{ + MountLease m{hexToU128("0123456789abcdeffedcba9876543210"), 7, "h", 1, + 1, 5, 2, std::numeric_limits::max(), true, + hexToU128("00112233445566778899aabbccddeeff")}; + const MountLease back = decodeMountLease(encodeMountLease(m)); + EXPECT_EQ(back.min_active, std::numeric_limits::max()); + EXPECT_TRUE(back.gc_fenced); + EXPECT_EQ(back.hostname, "h"); + EXPECT_EQ(back.writer_epoch, 7u); + EXPECT_EQ(back.seq, 5u); +} + +TEST(CASMountLeaseFormat, WriteAttemptIdIsRequiredAndCanonical) +{ + MountLease m; + m.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); + m.writer_epoch = 7; + m.write_attempt_id = hexToU128("00112233445566778899aabbccddeeff"); + + const String encoded = encodeMountLease(m); + EXPECT_NE(encoded.find("\"write_attempt_id\":\"00112233445566778899aabbccddeeff\""), String::npos); + EXPECT_EQ(decodeMountLease(encoded).write_attempt_id, m.write_attempt_id); + + const String without_attempt_id = currentFormatHeader("cas_mount_lease") + + "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"\",\"pid\":0," + "\"sat\":0,\"seq\":\"0\",\"eat\":0,\"ma\":\"0\",\"fen\":false}\n"; + try + { + decodeMountLease(without_attempt_id); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } +} + +TEST(CASMountLeaseFormat, ZeroWriteAttemptIdIsRejected) +{ + const String data = currentFormatHeader("cas_mount_lease") + + "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"\",\"pid\":0," + "\"sat\":0,\"seq\":\"0\",\"eat\":0,\"ma\":\"0\",\"fen\":false," + "\"write_attempt_id\":\"00000000000000000000000000000000\"}\n"; + try + { + decodeMountLease(data); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } +} + +TEST(CASMountLeaseFormat, UnknownFieldsRemainTolerated) +{ + MountLease m; + m.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); + m.writer_epoch = 7; + m.write_attempt_id = hexToU128("00112233445566778899aabbccddeeff"); + String encoded = encodeMountLease(m); + const size_t end = encoded.find("}\n"); + ASSERT_NE(end, String::npos); + encoded.insert(end, ",\"future_mount_field\":true"); + EXPECT_EQ(decodeMountLease(encoded).write_attempt_id, m.write_attempt_id); +} + +TEST(CASMountLeaseFormat, RejectsMissingIdentityFields) +{ + const String header = "{\"type\":\"cas_mount_lease\",\"v\":3}\n"; + const String fields = "\"hn\":\"host-1\",\"pid\":4242,\"sat\":1752537600000," + "\"seq\":\"5\",\"eat\":1752537630000,\"ma\":\"9\",\"fen\":false}"; + + const auto expectCorrupted = [](const String & data) + { + try + { + decodeMountLease(data); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } + }; + + expectCorrupted(header + R"({"we":"7",)" + fields + "\n"); + expectCorrupted(header + R"({"su":"0123456789abcdeffedcba9876543210",)" + fields + "\n"); +} diff --git a/src/Disks/tests/gtest_cas_settings.cpp b/src/Disks/tests/gtest_cas_settings.cpp new file mode 100644 index 000000000000..46d18480e55b --- /dev/null +++ b/src/Disks/tests/gtest_cas_settings.cpp @@ -0,0 +1,529 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; + +namespace DB::ErrorCodes +{ + extern const int NO_ELEMENTS_IN_CONFIG; + extern const int BAD_ARGUMENTS; + extern const int UNKNOWN_SETTING; +} + +/// Per-TU extern declarations for the `ContentAddressedSetting` entries this file uses -- the +/// established pattern for `BaseSettings`-derived classes in this codebase (see e.g. +/// `RegisterDiskCache.cpp`'s `namespace FileCacheSetting` block): the entries are DEFINED once in +/// `ContentAddressedSettings.cpp`, and each consumer TU declares only the ones it references. +namespace DB::ContentAddressedSetting +{ + extern const ContentAddressedSettingsBool gc_enabled; + extern const ContentAddressedSettingsUInt64 gc_shards; + extern const ContentAddressedSettingsUInt64 gc_interval_sec; + extern const ContentAddressedSettingsString scratch_path; +} + +namespace +{ +Poco::AutoPtr makeConfig(const std::string & inner) +{ + std::istringstream iss("" + inner + ""); + return new Poco::Util::XMLConfiguration(iss); +} + +const auto identity_macros = [](const std::string & s) { return s; }; + +class ScopedCasSettingsLogCapture +{ +public: + ScopedCasSettingsLogCapture() + : logger(getLogger("ContentAddressedSettings")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("warning"); + } + + ~ScopedCasSettingsLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const + { + return stream.str(); + } + +private: + LoggerPtr logger; + std::ostringstream stream; + Poco::AutoPtr channel; + /// `shared=true` above is load-bearing: `AutoPtr(ptr)` would STEAL a reference the fixture + /// never owned, undercounting the previous channel once per capture. The extra reference also + /// keeps the parked channel alive while the capture channel is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + +size_t countOccurrences(const String & haystack, const String & needle) +{ + size_t n = 0; + for (size_t at = haystack.find(needle); at != String::npos; at = haystack.find(needle, at + 1)) + ++n; + return n; +} + +void expectLoadFailureWithExactMessage(const String & config, int code, const String & message) +{ + auto cfg = makeConfig(config); + ContentAddressedSettings settings; + try + { + settings.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected settings load to fail"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), code); + EXPECT_EQ(e.message(), message); + } +} +} + +TEST(CASContentAddressedSettings, DefaultsAndOverridesLand) +{ + auto cfg = makeConfig("srv14"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/data", "/data/default_scratch", identity_macros); + EXPECT_EQ(s[ContentAddressedSetting::gc_shards].value, 4u); + EXPECT_EQ(s[ContentAddressedSetting::gc_interval_sec].value, 60u); /// table default + /// Absent key -> the verbatim default (never touches the anchor). + EXPECT_EQ(s[ContentAddressedSetting::scratch_path].value, "/data/default_scratch"); +} + +TEST(CASContentAddressedSettings, RemovedCacheSettingsAreRejected) +{ + for (const std::string & suffix : {"cache_bytes", "head_first_min_bytes"}) + { + const std::string setting = "cas_deduplication_" + suffix; + SCOPED_TRACE(setting); + auto cfg = makeConfig( + "srv1<" + setting + ">4096"); + ContentAddressedSettings settings; + try + { + settings.loadFromConfig(*cfg, "disk", "/data", "/data/scratch", identity_macros); + FAIL() << "expected removed setting " << setting << " to be rejected as unknown"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::UNKNOWN_SETTING); + } + } +} + +TEST(CASContentAddressedSettings, UnknownKeyRejected) +{ + expectLoadFailureWithExactMessage( + "srv14", + ErrorCodes::UNKNOWN_SETTING, + "Unknown setting 'cas_gc_shardz'"); +} + +TEST(CASContentAddressedSettings, MissingRequiredSettingNamesExternalConfigKey) +{ + expectLoadFailureWithExactMessage( + "1", + ErrorCodes::NO_ELEMENTS_IN_CONFIG, + "Expected `cas_server_root_id` in config for a content-addressed disk"); +} + +TEST(CASContentAddressedSettings, InvalidBoundsDiagnosticNamesExternalConfigKeys) +{ + expectLoadFailureWithExactMessage( + "srv10", + ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: cas_gc_interval_sec and cas_gc_shards must be >= 1 (got 60, 0)"); +} + +TEST(CASContentAddressedSettings, InvalidEnumDiagnosticsNameExternalConfigKeys) +{ + expectLoadFailureWithExactMessage( + "srv1md5", + ErrorCodes::BAD_ARGUMENTS, + "parseBlobHashAlgo: unknown cas_blob_hash config value 'md5' (expected one of cityhash128|xxh3-128|sha256)"); + expectLoadFailureWithExactMessage( + "srv1remote", + ErrorCodes::BAD_ARGUMENTS, + "Unknown cas_staging_backend value 'remote' (expected 'local' or 's3')"); + expectLoadFailureWithExactMessage( + "srv1sometimes", + ErrorCodes::BAD_ARGUMENTS, + "Unknown cas_part_folder_validate value 'sometimes' (expected 'always', 'never', or 'age ')"); +} + +/// The point of this test is that none of these names appears anywhere in CAS code. It is not an +/// enumeration to be extended when a backend adds a setting; it samples the classes that a +/// name-based skip-list provably cannot cover. +TEST(CASContentAddressedSettings, ForeignKeysAreNeverInspected) +{ + auto cfg = makeConfig( + "srv1" + "object_storages3" + "cashttp://x/y" + "cas_pool/cas_test_disk1" + "60" + "100" + "1000t" + "7100" + "
X-A: 1
X-B: 2
" + "X-C: 3alice" + "http://proxy:8080" + "k" + "acctc" + "DefaultEndpointsProtocol=http;"); + ContentAddressedSettings s; + EXPECT_NO_THROW(s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros)); +} + +TEST(CASContentAddressedSettings, LegacySpellingStillLoadsDuringMigrationWindow) +{ + auto cfg = makeConfig("srv14"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + EXPECT_EQ(s[ContentAddressedSetting::gc_shards].value, 4u); +} + +TEST(CASContentAddressedSettings, PartialMigrationLoadsAndReportsEveryLegacyKey) +{ + auto cfg = makeConfig( + "srv1" + "47"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + captured = capture.captured(); + } + EXPECT_EQ(s[ContentAddressedSetting::gc_shards].value, 4u); + EXPECT_EQ(s[ContentAddressedSetting::gc_interval_sec].value, 7u); + EXPECT_EQ(countOccurrences(captured, "superseded unprefixed spelling"), 1u); + EXPECT_NE(captured.find("gc_shards"), String::npos); + EXPECT_NE(captured.find("gc_interval_sec"), String::npos); +} + +TEST(CASContentAddressedSettings, FullyMigratedBlockWarnsAboutNothing) +{ + auto cfg = makeConfig("srv14"); + ContentAddressedSettings s; + ScopedCasSettingsLogCapture capture; + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + EXPECT_EQ(capture.captured().find("superseded"), String::npos); +} + +TEST(CASContentAddressedSettings, BothSpellingsOfOneSettingRejected) +{ + auto cfg = makeConfig( + "srv1" + "48"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the ambiguous pair to be rejected"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + } +} + +TEST(CASContentAddressedSettings, MalformedRepeatedPrefixedKeyIsRejectedBeforeParsing) +{ + auto cfg = makeConfig( + "srv1" + "not-a-number8"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the repeated key to be rejected before parsing its value"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + EXPECT_NE(String(e.message()).find("set more than once"), String::npos); + } +} + +TEST(CASContentAddressedSettings, MalformedPrefixedValueCannotMaskBothSpellingsConflict) +{ + auto cfg = makeConfig( + "srv1" + "not-a-number8"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the ambiguous pair to be rejected before parsing its values"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + EXPECT_NE(String(e.message()).find("both"), String::npos); + } +} + +TEST(CASContentAddressedSettings, AmbiguousConfigDoesNotWarnOrPartiallyApplySettings) +{ + auto cfg = makeConfig( + "srv1" + "48"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + EXPECT_THROW(s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros), Exception); + captured = capture.captured(); + } + EXPECT_EQ(captured.find("are applied"), String::npos); + EXPECT_FALSE(s[ContentAddressedSetting::gc_shards].changed); +} + +TEST(CASContentAddressedSettings, UnknownPrefixedKeyDoesNotWarnOrPartiallyApplySettings) +{ + auto cfg = makeConfig( + "30" + "8"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the unknown prefixed key to be rejected"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::UNKNOWN_SETTING); + } + captured = capture.captured(); + } + EXPECT_EQ(captured.find("are applied"), String::npos); + EXPECT_FALSE(s[ContentAddressedSetting::gc_shards].changed); + EXPECT_FALSE(s[ContentAddressedSetting::gc_enabled].changed); +} + +TEST(CASContentAddressedSettings, MalformedPrefixedKeyDoesNotWarnOrPartiallyApplySettings) +{ + auto cfg = makeConfig( + "30" + "not-a-number"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the malformed prefixed key to be rejected"; + } + catch (const Exception &) + { + } + captured = capture.captured(); + } + EXPECT_EQ(captured.find("are applied"), String::npos); + EXPECT_FALSE(s[ContentAddressedSetting::gc_shards].changed); + EXPECT_FALSE(s[ContentAddressedSetting::gc_enabled].changed); +} + +TEST(CASContentAddressedSettings, ValidMixedConfigCommitsAfterAllValuesValidate) +{ + auto cfg = makeConfig( + "srv13" + "0"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + EXPECT_EQ(s[ContentAddressedSetting::gc_shards].value, 3u); + EXPECT_FALSE(s[ContentAddressedSetting::gc_enabled].value); +} + +TEST(CASContentAddressedSettings, SemanticInvalidPrefixedKeyDoesNotWarnOrPartiallyApplySettings) +{ + auto cfg = makeConfig( + "00"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the semantically invalid prefixed key to be rejected"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + } + captured = capture.captured(); + } + EXPECT_EQ(captured.find("are applied"), String::npos); + EXPECT_FALSE(s[ContentAddressedSetting::gc_enabled].changed); + EXPECT_FALSE(s[ContentAddressedSetting::gc_shards].changed); +} + +TEST(CASContentAddressedSettings, InvalidEnumDoesNotWarnOrPartiallyApplySettings) +{ + auto cfg = makeConfig( + "srv10" + "md5"); + ContentAddressedSettings s; + String captured; + { + ScopedCasSettingsLogCapture capture; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the invalid hash algorithm to be rejected"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + } + captured = capture.captured(); + } + EXPECT_EQ(captured.find("are applied"), String::npos); + EXPECT_FALSE(s[ContentAddressedSetting::gc_enabled].changed); +} + +/// Poco renders a repeated element as `name`, `name[1]`. A key of ours that appears twice must be +/// recognized by its base name rather than passed over as foreign, or the first value would silently win. +TEST(CASContentAddressedSettings, RepeatedKeyRejectedInEitherSpelling) +{ + for (const std::string & spelling : {std::string("gc_shards"), std::string("cas_gc_shards")}) + { + SCOPED_TRACE(spelling); + auto cfg = makeConfig( + "srv1" + "<" + spelling + ">4" + "<" + spelling + ">8"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected a repeated key to be rejected"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + } + } +} + +TEST(CASContentAddressedSettings, SkipAccessCheckKeepsItsBareSpelling) +{ + auto with = makeConfig( + "srv1" + "1"); + ContentAddressedSettings s; + s.loadFromConfig(*with, "disk", "/scratch", "/scratch", identity_macros); + EXPECT_TRUE(s.skipAccessCheck()); + + auto prefixed = makeConfig( + "srv1" + "1"); + ContentAddressedSettings rejected; + try + { + rejected.loadFromConfig(*prefixed, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected `cas_skip_access_check` to be unknown"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::UNKNOWN_SETTING); + } +} + +TEST(CASContentAddressedSettings, PrefixedGcsCapIsNotACasSetting) +{ + auto cfg = makeConfig( + "srv1" + "4096"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected the prefixed cap name to be unknown"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::UNKNOWN_SETTING); + } +} + +TEST(CASContentAddressedSettings, ValidateFailsClosed) +{ + { + auto cfg = makeConfig("1"); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected an exception"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::NO_ELEMENTS_IN_CONFIG); + } + } + { + auto cfg = makeConfig(""); + ContentAddressedSettings s; + try + { + s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros); + FAIL() << "expected an exception"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::BAD_ARGUMENTS); + } + } + { + auto cfg = makeConfig("srv10"); + ContentAddressedSettings s; + EXPECT_THROW(s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros), Exception); + } + { + auto cfg = makeConfig("srv1md5"); + ContentAddressedSettings s; + EXPECT_THROW(s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros), Exception); + } +} + +TEST(CASContentAddressedSettings, RelativeScratchPathAnchored) +{ + auto cfg = makeConfig("srv1rel/dir"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/data", "/data/disks/x/cas_scratch", identity_macros); + EXPECT_EQ(s[ContentAddressedSetting::scratch_path].value, "/data/rel/dir"); +} + +TEST(CASContentAddressedSettings, AbsentScratchPathUsesDefaultVerbatim) +{ + auto cfg = makeConfig("srv1"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/data", "/data/disks/x/cas_scratch", identity_macros); + EXPECT_EQ(s[ContentAddressedSetting::scratch_path].value, "/data/disks/x/cas_scratch"); +} diff --git a/src/Disks/tests/gtest_cas_shutdown_context.cpp b/src/Disks/tests/gtest_cas_shutdown_context.cpp new file mode 100644 index 000000000000..6ceb58b1188b --- /dev/null +++ b/src/Disks/tests/gtest_cas_shutdown_context.cpp @@ -0,0 +1,205 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ContentAddressedSetting +{ +extern const ContentAddressedSettingsBool gc_enabled; +} + +namespace ProfileEvents +{ +extern const Event CASEventDroppedContextExpired; +} + +namespace +{ + +/// Reuse the single `ContextSharedPart` owned by the global gtest environment, exactly as the +/// interpreter tests do. Each returned `Context` is independently owned, so a test can release or +/// reset its copy without disturbing the process-global context or the other tests. +DB::ContextMutablePtr makeTestContext() +{ + return DB::Context::createCopy(getContext().context); +} + +std::shared_ptr openTestStorage( + const DB::ContextPtr & context = {}, bool startup = true) +{ + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "cas_shutdown_context_scratch"); + /// These tests exercise the event sink synchronously. Keeping the GC scheduler off avoids adding + /// unrelated worker activity while preserving the real pool event sink installed at `startup`. + settings[DB::ContentAddressedSetting::gc_enabled] = false; + settings.validate(); + + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", context, settings); + if (startup) + storage->startup(); + return storage; +} + +void emitTestEvent(DB::ContentAddressedMetadataStorage & storage) +{ + auto pool = storage.poolForTest(); + if (!pool) + throw std::runtime_error("test storage has no pool"); + EventEmitter{*pool}.emit([](CasEvent & event) + { + event.type = CasEventType::Exception; + event.reason = "test event"; + }); +} + +/// Open a pool, arm one teardown phase to throw, destroy it, and report whether the clean-release +/// marker was written. Runs inside the subprocess of each exit test below. +[[noreturn]] void tearDownWithThrowingPhase(int phase) +{ + auto backend = std::make_shared(); + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + auto thrower = [] { throw std::runtime_error("injected teardown phase failure"); }; + if (phase == 1) + config.teardown_phase1_throw_for_test = thrower; + else if (phase == 2) + config.teardown_phase2_throw_for_test = thrower; + else + config.teardown_phase3_throw_for_test = thrower; + + { + auto store = Pool::open(backend, config); + (void)store; + } /// `~Pool` runs here. + + /// A failed ref-lane drain must not leave a clean-release marker behind. That marker lets a + /// successor skip the observation window, so a phase-2 failure must leave it absent. + const auto mount = backend->get(Layout(config.pool_prefix).mountKey(config.server_root_id)); + const bool clean_release = mount + && decodeMountLease(mount->bytes).min_active == std::numeric_limits::max(); + const bool marker_must_be_absent = phase == 2; + std::_Exit(marker_must_be_absent && clean_release ? 1 : 0); +} + +} + +/// These exit tests run their statement in a fresh re-executed process ("threadsafe" style), not a +/// fork of the test runner. The default fork inherits a heavily multithreaded, sanitizer-instrumented +/// process, where filesystem calls made while building the storage inside the child can fail +/// spuriously (seen under TSan as `weakly_canonical: Invalid argument` from the bootstrap LIST). +/// gtest restores the flag after each test. +TEST(CASShutdownExitTest, TeardownPhase1ThrowExitsCleanly) +{ + GTEST_FLAG_SET(death_test_style, "threadsafe"); + EXPECT_EXIT(tearDownWithThrowingPhase(1), ::testing::ExitedWithCode(0), ""); +} + +TEST(CASShutdownExitTest, TeardownPhase2ThrowExitsCleanlyAndSkipsTheMarker) +{ + GTEST_FLAG_SET(death_test_style, "threadsafe"); + EXPECT_EXIT(tearDownWithThrowingPhase(2), ::testing::ExitedWithCode(0), ""); +} + +TEST(CASShutdownExitTest, TeardownPhase3ThrowExitsCleanly) +{ + GTEST_FLAG_SET(death_test_style, "threadsafe"); + EXPECT_EXIT(tearDownWithThrowingPhase(3), ::testing::ExitedWithCode(0), ""); +} + +/// `Server.cpp` calls `resetSharedContext` immediately before releasing the context. An event emitted +/// in that window must be skipped safely, not dereference a null `shared`. +TEST(CASShutdownExitTest, EmitAfterResetSharedContextExitsCleanly) +{ + GTEST_FLAG_SET(death_test_style, "threadsafe"); + EXPECT_EXIT( + { + auto context = makeTestContext(); + auto storage = openTestStorage(context); + ASSERT_TRUE(storage->poolForTest()->hasEventSink()); + emitTestEvent(*storage); + context->resetSharedContext(); + emitTestEvent(*storage); + std::_Exit(0); + }, + ::testing::ExitedWithCode(0), ""); +} + +/// An EXPIRED weak reference is the one case that is counted. +TEST(CASShutdownContext, ExpiredContextDropsTheEventAndCountsIt) +{ + auto context = makeTestContext(); + auto storage = openTestStorage(context); + ASSERT_TRUE(storage->poolForTest()->hasEventSink()); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + + context.reset(); + emitTestEvent(*storage); + + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + EXPECT_EQ(after - before, 1u); +} + +/// `nullptr` at construction means the integration is off. Nothing is emitted and NOTHING is counted -- +/// several existing suites construct the storage this way. +TEST(CASShutdownContext, DisabledIntegrationCountsNothing) +{ + auto storage = openTestStorage(); + ASSERT_FALSE(storage->poolForTest()->hasEventSink()); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + + emitTestEvent(*storage); + + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + EXPECT_EQ(after - before, 0u); +} + +/// A live context whose system log is not configured is ordinary steady state: no emit, no count. +TEST(CASShutdownContext, MissingSystemLogCountsNothing) +{ + auto context = makeTestContext(); + auto storage = openTestStorage(context); + ASSERT_TRUE(storage->poolForTest()->hasEventSink()); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + + emitTestEvent(*storage); + + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + EXPECT_EQ(after - before, 0u); +} + +/// The storage must no longer keep the context alive. This is the property `Server.cpp` relies on when +/// it destroys the context explicitly. +TEST(CASShutdownContext, StorageDoesNotExtendContextLifetime) +{ + auto context = makeTestContext(); + std::weak_ptr weak_context = context; + auto storage = openTestStorage(context); + + context.reset(); + + EXPECT_EQ(weak_context.use_count(), 0L); +} + +/// An expired reference supplied at `startup` is an error, not the disabled path. +TEST(CASShutdownContext, ExpiredContextAtStartupFails) +{ + auto context = makeTestContext(); + auto storage = openTestStorage(context, /*startup=*/false); + context.reset(); + + EXPECT_ANY_THROW(storage->startup()); +} diff --git a/src/Disks/tests/gtest_cas_slot_occupy.cpp b/src/Disks/tests/gtest_cas_slot_occupy.cpp new file mode 100644 index 000000000000..d248509c2d31 --- /dev/null +++ b/src/Disks/tests/gtest_cas_slot_occupy.cpp @@ -0,0 +1,358 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::ChunkFaultBackend; +using DB::Cas::tests::LandedButAckLostOnceBackend; + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +/// ================================================================================================ +/// Task 2 (2026-07-28 CAS ref-chain Stage A streams, spec INV-2): CasRequestController::slotOccupy -- +/// the dedicated RAW slot-occupy primitive every seal writer and wedge retry uses. ONE conditional +/// create; on conflict, ONE raw exact GET of the occupant -- NEVER retries internally, NEVER lists, +/// and NEVER composes putIfAbsentControlled (which retries the same (key, bytes) internally) or +/// resolveByExactGet (which compares against an expected body and throws CORRUPTED_DATA on a +/// mismatch) [codex finding 3]. Adjudicating whether an Occupied occupant is "mine" is entirely the +/// CALLER's job (Task 4/6, the CaCasMountCore `mine` contract) -- these tests only pin the +/// primitive's own three-way outcome and its op-count contract (Created=1, Occupied=2, +/// Unresolved<=2 backend ops). +/// ================================================================================================ + +namespace +{ + +/// Deletes the key the INSTANT its own conditional create conflicts, modelling "the occupant that +/// caused the conflict vanished before slotOccupy's single resolve GET" -- a race a real backend can +/// produce (e.g. GC reclaiming an already-condemned object) that the primitive must survive by +/// reporting Unresolved, NEVER a fabricated Created. +class VanishOnConflictBackend : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + PutResult result = CountingBackend::putIfAbsent(key, bytes, meta); + if (result.outcome == PutOutcome::PreconditionFailed) + { + const HeadResult h = head(key); + if (h.exists) + deleteExact(key, h.token); + } + return result; + } +}; + +/// Throws a deterministic LOCAL failure (BAD_ARGUMENTS, in isDeterministicLocalFailure's set) on the +/// first putIfAbsent -- models a backend-level programming bug, distinct from ChunkFaultBackend's +/// Mode::Definite below, which is a whitelisted SYNCHRONOUS REJECTION +/// (classifyConditionalWriteResult's DefiniteFailure). slotOccupy must rethrow both, unchanged, never +/// folding either into Unresolved (SlotOccupyResult::Kind has no DefiniteFailure member to carry it). +class LocalFailureOnceBackend : public CountingBackend +{ +public: + using CountingBackend::putIfAbsent; + bool fail_once = true; + + PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + { + if (fail_once) + { + fail_once = false; + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "scripted deterministic local failure"); + } + return CountingBackend::putIfAbsent(key, bytes, meta); + } +}; + +/// (`LandedButAckLostOnceBackend` -- "the write LANDS, then the ack is lost" -- was lifted into +/// `cas_test_helpers.h` for Task 4, whose wedge-adoption tests need the identical seam through a whole +/// Pool. Its `key_substr` defaults to empty, which is exactly this file's original behaviour: fault the +/// first `putIfAbsent` of any key.) +/// Delegates the FIRST putIfAbsent for a key to CountingBackend -- so the write actually LANDS -- and +/// only THEN throws an ambiguous exception, modelling "our own PUT committed but its response was lost" +/// (the Task-4 adoption input: plan's "Occupied + bytes == wedge.bytes -> an earlier attempt landed -> +/// adopt"). Distinct from InMemoryBackend::injectAmbiguousPutIfAbsent, which never touches the store at +/// all -- that hook models an attempt that did NOT land; this one models an attempt that DID. +/// One-shot per backend instance: review finding I2 asked specifically for a ~10-line local backend rather than +/// reusing ChunkFaultBackend::Mode::LandedThenLost, which also arms a one-shot lost-GET fault that would +/// obscure whether slotOccupy's OWN immediate resolve (not just a later caller's retry) is correct too. + +} + +/// ---- Step 1 required scenarios ---- + +TEST(CASSlotOccupy, AbsentKeyCreatesWithOneOp) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + + const auto result = controller.slotOccupy("k", "payload", [] { return true; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Created); + EXPECT_TRUE(result.occupant_bytes.empty()); + EXPECT_TRUE(result.occupant_token.empty()) << "occupant_token is Occupied-only; must stay default on Created"; + EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NotUnresolved); + + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->getCount("k"), 0u); + EXPECT_EQ(backend->headCount("k"), 0u); + + const auto landed = backend->get("k"); + ASSERT_TRUE(landed.has_value()); + EXPECT_EQ(landed->bytes, "payload"); +} + +TEST(CASSlotOccupy, PreExistingKeyOccupiedWithExactBytesAndTokenTwoOps) +{ + auto backend = std::make_shared(); + const PutResult seeded = backend->putIfAbsent("k", "occupant-bytes"); + ASSERT_EQ(seeded.outcome, PutOutcome::Done); + backend->resetCounts(); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto result = controller.slotOccupy("k", "my-attempt-bytes", [] { return true; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Occupied); + EXPECT_EQ(result.occupant_bytes, "occupant-bytes"); + EXPECT_EQ(result.occupant_token, seeded.token); + + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->getCount("k"), 1u); + EXPECT_EQ(backend->headCount("k"), 0u) << "exactly PUT+GET -- a HEAD-then-GET implementation must fail this"; + + /// A conflict never overwrites or appends -- the pre-existing object is untouched. + const auto current = backend->get("k"); + ASSERT_TRUE(current.has_value()); + EXPECT_EQ(current->bytes, "occupant-bytes"); +} + +TEST(CASSlotOccupy, InjectedAmbiguousPutResolvesUnresolvedWhenGetFindsNothing) +{ + auto backend = std::make_shared(); + backend->injectAmbiguousPutIfAbsent("k"); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto result = controller.slotOccupy("k", "payload", [] { return true; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); + /// An attempt WAS sent (the ambiguous PUT itself) -- this is never the pre-attempt NoAttemptSent + /// case. Of the existing CasUnresolvedReason values, AttemptsExhausted is the one documented as + /// "the genuine case the 'retry budget exhausted' wording describes" -- exactly this call's single + /// (and only) attempt having nothing left to give once its resolve GET came up empty. + EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::AttemptsExhausted); + EXPECT_FALSE(unresolvedProvesNothingWasSent(result.unresolved_reason)); + + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->getCount("k"), 1u); + EXPECT_FALSE(backend->head("k").exists) << "the injected fault must not actually create anything"; +} + +TEST(CASSlotOccupy, ConflictThenVanishResolvesUnresolved) +{ + auto backend = std::make_shared(); + const auto seeded = backend->putIfAbsent("k", "occupant-bytes"); + ASSERT_EQ(seeded.outcome, PutOutcome::Done); + backend->resetCounts(); + + CasRequestController controller(backend, CasRequestBudget{}); + const auto result = controller.slotOccupy("k", "my-attempt-bytes", [] { return true; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); + EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::AttemptsExhausted); + + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->getCount("k"), 1u); + /// No headCount assertion here (unlike the sibling Occupied test above): VanishOnConflictBackend's + /// OWN fixture issues a HEAD internally (to fetch the token before deleteExact) -- that HEAD belongs + /// to the test's vanish mechanism, not to slotOccupy, so asserting headCount==0 would be wrong, not + /// stronger. slotOccupy itself never calls head(); only put+get are its own ops. + EXPECT_FALSE(backend->head("k").exists) << "the occupant vanished between the conflict and the resolve GET"; +} + +TEST(CASSlotOccupy, FenceFlipMidCallRefusesPreAttemptNeverLiesCreated) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + + const auto result = controller.slotOccupy("k", "payload", [] { return false; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); + /// The pre-attempt reason: fence_ok refused before anything was sent to the backend. + EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NoAttemptSent); + EXPECT_TRUE(unresolvedProvesNothingWasSent(result.unresolved_reason)); + + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_FALSE(backend->head("k").exists) << "never a lie of Created -- the key must be untouched"; +} + +/// ---- Bonus coverage: the deadline pre-gate (the OTHER half of "fence/deadline-gated"), and the two +/// rethrow paths this primitive shares with its sibling controlled ops. ---- + +/// The deadline gate is the SAME pre-attempt refusal as the fence gate above -- a fake clock proves it +/// fires from elapsed time alone, with a fence that always says yes. +TEST(CASSlotOccupy, OperationDeadlineExhaustedRefusesPreAttempt) +{ + auto backend = std::make_shared(); + uint64_t clock = 0; + auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1000; return t; }; + + CasRequestBudget budget; + budget.attempt_timeout_ms = 50; + budget.operation_deadline_ms = 500; /// entry now_ms()==0 -> deadline_ms=500; the gate's OWN + /// now_ms() call then returns 1000 -> 1000+50 > 500 -> refuse + CasRequestController controller(backend, budget, now_ms); + + const auto result = controller.slotOccupy("k", "payload", [] { return true; }); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); + EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NoAttemptSent); + EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u) << "zero ops total -- the deadline gate must refuse before any I/O, same as the fence gate"; +} + +/// A whitelisted synchronous rejection (classifyConditionalWriteResult's DefiniteFailure) PROVES the +/// request was never applied -- slotOccupy must surface it unchanged rather than resolving or folding +/// it into Unresolved. Guarded to USE_AWS_S3 builds ONLY [review M6]: DefiniteFailure classification is +/// structurally unreachable without it (classifyConditionalWriteResult's whitelist is entirely inside +/// its own `#if USE_AWS_S3`), so on a no-S3 build ChunkFaultBackend::Mode::Definite instead throws a +/// plain CORRUPTED_DATA DB::Exception -- which is in isDeterministicLocalFailure's set, meaning this +/// test would silently exercise the SAME slotOccupy branch as DeterministicLocalFailurePropagatesWithoutResolve +/// below rather than the DefiniteFailure branch it claims to cover. Better a visibly-absent test on that +/// config than a passing one that isn't testing what its name says. +#if USE_AWS_S3 +TEST(CASSlotOccupy, DefiniteFailurePropagatesWithoutResolve) +{ + auto backend = std::make_shared(); + backend->fault_substr = "k"; + backend->mode = ChunkFaultBackend::Mode::Definite; + backend->fault_count = 1; + + CasRequestController controller(backend, CasRequestBudget{}); + EXPECT_THROW(controller.slotOccupy("k", "payload", [] { return true; }), DB::Exception); + /// ChunkFaultBackend's fault check throws BEFORE delegating to CountingBackend::putIfAbsent, so + /// putCount stays 0 on this path -- fault_count reaching 0 is this backend's own proof the (one) + /// attempt was made and consumed the fault. + EXPECT_EQ(backend->fault_count, 0); + EXPECT_EQ(backend->getCount("k"), 0u) << "a whitelisted definite rejection must never trigger a resolve GET"; +} +#endif + +/// A deterministic LOCAL failure (isDeterministicLocalFailure's set) is the OTHER rethrow path -- +/// distinct from DefiniteFailure above, and checked first in the implementation, so it needs its own +/// backend-level fault to prove both branches are wired, not just one masking the other. +TEST(CASSlotOccupy, DeterministicLocalFailurePropagatesWithoutResolve) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + + bool threw = false; + try + { + controller.slotOccupy("k", "payload", [] { return true; }); + } + catch (const DB::Exception & e) + { + threw = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::BAD_ARGUMENTS) << "the ORIGINAL exception must propagate unchanged"; + } + EXPECT_TRUE(threw) << "a deterministic local failure must propagate, never return an outcome"; + /// LocalFailureOnceBackend throws BEFORE delegating to CountingBackend::putIfAbsent (same shape as + /// ChunkFaultBackend above), so putCount stays 0 here too -- fail_once flipping is this backend's + /// own proof the attempt was made. + EXPECT_FALSE(backend->fail_once); + EXPECT_EQ(backend->getCount("k"), 0u); +} + +/// ---- Fix round 1 (review findings I1, I2): the two gaps the reviewer required landed before Task 4 +/// consumes this primitive. Both guard the design decisions the review approved -- see +/// task-2-review.md concern (a) and finding I2's Task-4-adoption note. ---- + +/// I1: pins the single-`fence_ok`-call `Created` design (concern (a)) so a future contributor cannot +/// silently "fix the inconsistency" by re-adding the sibling ops' post-write fence recheck. That change +/// would break Task 4's old-generation-retry semantics (resolveWedgeOnce deliberately calls slotOccupy under +/// the wedge's ORIGINAL admitted_fence_generation, and relies on ITS OWN post-I/O checkFenceOrThrow, +/// not a second internal check here, to decide whether the result is still relevant). A counting +/// fence_ok that only answers true on its FIRST call: if slotOccupy ever called it again after the +/// write landed, this test would see Unresolved instead of Created, OR (if the outcome happened to +/// still read Created some other way) the call-count assertion below would catch the extra invocation +/// either way. +TEST(CASSlotOccupy, CreatedNeverRechecksFenceAfterTheWrite) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + + int fence_calls = 0; + const auto fence_ok = [&fence_calls] + { + ++fence_calls; + return fence_calls == 1; + }; + + const auto result = controller.slotOccupy("k", "payload", fence_ok); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Created); + EXPECT_EQ(fence_calls, 1) << "slotOccupy must call fence_ok() exactly ONCE (pre-attempt only) -- " + "a post-write recheck would falsely report Unresolved here (fence_calls's " + "SECOND answer is false) and would break Task 4's old-generation-retry design"; +} + +/// A conflict needs a second backend request to resolve its occupant. Admission may disappear while +/// the conditional create is in flight; in that case the resolver must fail closed before starting +/// the `GET`, while preserving the one-check `Created` contract above. +TEST(CASSlotOccupy, AdmissionLostAfterConflictPreventsTheResolveGet) +{ + auto backend = std::make_shared(); + ASSERT_EQ(backend->putIfAbsent("k", "existing").outcome, PutOutcome::Done); + CasRequestController controller(backend, CasRequestBudget{}); + + int admission_checks = 0; + const auto admitted = [&admission_checks] + { + ++admission_checks; + return admission_checks == 1; + }; + + const auto result = controller.slotOccupy("k", "attempt", admitted); + EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); + EXPECT_EQ(admission_checks, 2); + EXPECT_EQ(backend->putCount("k"), 2u); + EXPECT_EQ(backend->getCount("k"), 0u) + << "slotOccupy started its ambiguity-resolution GET after admission was withdrawn"; +} + +/// I2: proves Occupied is reachable for an occupant that is OUR OWN earlier ambiguous write, not only +/// for a foreign pre-seeded one (PreExistingKeyOccupiedWithExactBytesAndTokenTwoOps above always seeds +/// via a plain, unambiguous putIfAbsent). This is the exact input shape Task 4's resolveWedgeOnce +/// adjudicates: "Occupied + bytes == wedge.bytes -> an earlier attempt landed -> adopt" (plan :329). +TEST(CASSlotOccupy, OwnLandedAmbiguousWriteObservedAsOccupiedOnRetry) +{ + auto backend = std::make_shared(); + CasRequestController controller(backend, CasRequestBudget{}); + + /// Call 1 -- the original attempt: the PUT's own response is lost, but the write DID commit, and + /// THIS call's own resolve GET (unfaulted) observes it immediately -- Occupied with OUR bytes, + /// proving the same-call resolve path works for a landed ambiguous write, not only a foreign one. + const auto first = controller.slotOccupy("k", "my-bytes", [] { return true; }); + EXPECT_EQ(first.kind, SlotOccupyResult::Kind::Occupied); + EXPECT_EQ(first.occupant_bytes, "my-bytes"); + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->getCount("k"), 1u); + + /// Call 2 -- Task 4's resolveWedgeOnce pattern: a LATER caller's flush resolving the SAME logical + /// attempt via a FRESH slotOccupy call. The fault is already consumed (one-shot), so this PUT + /// conflicts cleanly (PreconditionFailed) and the resolve GET observes OUR OWN earlier bytes again -- + /// the exact adoption input Task 4 is built on, and the SAME incarnation both calls saw. + const auto second = controller.slotOccupy("k", "my-bytes", [] { return true; }); + EXPECT_EQ(second.kind, SlotOccupyResult::Kind::Occupied); + EXPECT_EQ(second.occupant_bytes, "my-bytes"); + EXPECT_EQ(second.occupant_token, first.occupant_token) << "both calls must observe the SAME landed incarnation"; + EXPECT_EQ(backend->putCount("k"), 2u); + EXPECT_EQ(backend->getCount("k"), 2u); +} diff --git a/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp b/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp new file mode 100644 index 000000000000..4c0000b46236 --- /dev/null +++ b/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp @@ -0,0 +1,504 @@ +#include + +#include +#include +#include +#include "cas_sweep_test_support.h" +#include "cas_test_helpers.h" + +#include + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +/// Spec §6, the sweep deletion premise. A manifest of an epoch-`E` build is deletable only when the +/// namespace cursor has consumed epoch `E`'s seal AND no unconsumed tail record above the cursor names +/// it as a removal target; on ANY uncertainty the sweep RETAINS and says why. +/// +/// WHY THE CURSOR AND NOT A LISTING. The sweep's pre-existing protection view is assembled from an +/// enumeration of the namespace's ref objects, and arithmetic ref intake demoted exactly that +/// enumeration to a hint: a store may omit a durable key from a `LIST`. A hidden `+1` above the cursor +/// therefore makes an owned manifest look unowned, and deleting it is data loss; a hidden `-1` makes a +/// removal target look unprotected, and deleting it clamps the fold forever on the missing body. The +/// premise closes the first by arithmetic (grants do not cross epochs, and an epoch is left only over +/// its consumed seal) and the second by refusing whenever the tail is not decidable. +namespace +{ + +/// The build's epoch. The namespace's seeded ref log lives at writer epoch 1 (`appendRefLogSeed`), so +/// naming the build's epoch 1 as well keeps the fixture coherent: a cursor at `{2, _}` is then a cursor +/// that genuinely crossed epoch 1's closing seal, not an invented number above an unrelated stream. +constexpr uint64_t kBuildEpoch = 1; +const String kServerRoot = "00"; + +ManifestRef ref(uint64_t seq, uint64_t ordinal) +{ + return ManifestRef{.writer_epoch = kBuildEpoch, .build_sequence = seq, + .manifest_ordinal = static_cast(ordinal)}; +} + +BuildPrefix buildPrefix(uint64_t seq) +{ + return BuildPrefix{.writer_epoch = kBuildEpoch, .build_sequence = seq}; +} + +/// A pool with ONE eligible-but-unowned manifest body under build sequence 5: the shape the sweep is +/// meant to reclaim, so that every test below differs only in the durable fold state. +struct OrphanFixture +{ + std::shared_ptr backend = std::make_shared(); + PoolPtr store; + RootNamespace ns{"00/aa@cas@"}; + ManifestRef orphan = ref(5, 0xAB); + + OrphanFixture() + { + store = openPoolForTest(backend); + /// This fixture has no ref transaction, but it is a normal empty catalog life rather than the + /// deliberate missing-checkpoint corruption shape. State that empty recovery frontier before + /// exercising the independent sweep-deletion premise. + casAdmitRecoverableEntry(*backend, store->layout(), ns); + writeManifestRaw(*backend, store->layout(), ns, orphan, {blobEntryFor("a", DB::UInt128(1))}); + /// min_active 6 > build_sequence 5: the durable watermark fact makes the prefix ELIGIBLE, which + /// is the half the premise sits on top of. + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active*/6); + } + + String orphanKey() const { return store->layout().manifestKey(ManifestId{ns, orphan}); } + bool orphanExists() const { return backend->head(orphanKey()).exists; } +}; + +/// The same admissible orphan shape as `OrphanFixture`, but without its legal manifest write: the +/// test below must plant undecodable bytes as the key's first and only incarnation. +struct UndecodableOrphanFixture +{ + std::shared_ptr backend = std::make_shared(); + PoolPtr store; + RootNamespace ns{"00/aa@cas@"}; + ManifestRef orphan = ref(5, 0xAB); + + UndecodableOrphanFixture() + { + store = openPoolForTest(backend); + casAdmitRecoverableEntry(*backend, store->layout(), ns); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active*/6); + } + + String orphanKey() const { return store->layout().manifestKey(ManifestId{ns, orphan}); } + bool orphanExists() const { return backend->head(orphanKey()).exists; } +}; + +} + +/// Rule (1), the load-bearing case. The cursor is still INSIDE the build's own epoch, so epoch 1's +/// closing seal is not proven consumed and an unfolded `+1` naming this build may still exist above the +/// cursor. The body survives, and the sweep says so through its `warnings` out-param. +TEST(CASSweepDeletionPremise, AnUnconsumedEpochSealRetainsTheBuildsManifests) +{ + OrphanFixture f; + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch, 3}); + + std::vector warnings; + const uint64_t deleted = sweepNamespace(*f.store, f.ns, buildPrefix(5), &warnings); + + EXPECT_EQ(deleted, 0u); + EXPECT_TRUE(f.orphanExists()) + << "the cursor sits at {1,3}, inside the build's own epoch -- epoch 1's closing seal is not " + "consumed, so a grant naming this build may still be unfolded above the cursor"; + ASSERT_EQ(warnings.size(), 1u) << "a retained manifest is a visible decision, not a silent one"; + EXPECT_NE(warnings[0].find(f.orphanKey()), String::npos); + EXPECT_NE(warnings[0].find("seal"), String::npos); +} + +/// Rule (1) satisfied and the tail clean: the ordinary reclaim still happens, by exact token. +TEST(CASSweepDeletionPremise, AConsumedEpochSealWithACleanTailDeletes) +{ + OrphanFixture f; + /// A cursor at `{2, 1}` is in an epoch strictly above the build's. An epoch is left ONLY over its + /// consumed `EpochSeal`, so this cursor is durable proof that every epoch-1 record is folded. + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch + 1, 1}); + + std::vector warnings; + const uint64_t deleted = sweepNamespace(*f.store, f.ns, buildPrefix(5), &warnings); + + EXPECT_EQ(deleted, 1u); + EXPECT_FALSE(f.orphanExists()); + EXPECT_TRUE(warnings.empty()) << "nothing was retained, so nothing is warned about"; +} + +/// Uncertainty rule, hold arm. The cursor HAS consumed epoch 1's seal, so rule (1) alone would let the +/// body go -- but the namespace is held, which means the fold could not account for everything at or +/// above the held position. A held namespace retains everything under it. +TEST(CASSweepDeletionPremise, AHeldNamespaceRetainsEvenAboveAConsumedSeal) +{ + OrphanFixture f; + const RefHold hold{.reason = HoldReason::GapBelowWitness, + .offending_position = RefTxnId{kBuildEpoch + 1, 4}, + .retry_count = 2, .next_retry_round = 9}; + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch + 1, 3}, hold); + + std::vector warnings; + const uint64_t deleted = sweepNamespace(*f.store, f.ns, buildPrefix(5), &warnings); + + EXPECT_EQ(deleted, 0u); + EXPECT_TRUE(f.orphanExists()); + ASSERT_EQ(warnings.size(), 1u); + EXPECT_NE(warnings[0].find("held"), String::npos); + EXPECT_NE(warnings[0].find(String{holdReasonToWord(HoldReason::GapBelowWitness)}), String::npos) + << "the retain reason names WHAT stopped the namespace, not just that something did"; +} + +/// Uncertainty rule, unreached-frontier arm in its most complete form: the adopted seal carries no row +/// for this namespace at all, so no round has ever sealed a cursor for it and nothing about its ref +/// stream is proven. This is also the state of a pool whose GC has never run. +TEST(CASSweepDeletionPremise, ANamespaceWithNoSealedCursorRetains) +{ + OrphanFixture f; + /// A seal exists and is adopted, but it covers a DIFFERENT namespace. + seedFoldCursorForTest(*f.backend, f.store->layout(), RootNamespace{"00/zz@cas@"}, + RefTxnId{kBuildEpoch + 1, 1}); + + std::vector warnings; + const uint64_t deleted = sweepNamespace(*f.store, f.ns, buildPrefix(5), &warnings); + + EXPECT_EQ(deleted, 0u); + EXPECT_TRUE(f.orphanExists()); + ASSERT_EQ(warnings.size(), 1u); + EXPECT_NE(warnings[0].find("coverage"), String::npos); +} + +/// Rule (2). Removals cross epochs, so a record in a LATER epoch can name an earlier epoch's build as a +/// removal target; deleting the body before that `-1` folds clamps the fold forever on the missing body. +/// The predicate is exercised directly here because the sweep's own protection view already spares a +/// listed tail removal before the premise is ever consulted -- the point of the rule is that the SAME +/// answer is reached by the predicate both paths share, so neither path can lose it. +TEST(CASSweepDeletionPremise, AnUnconsumedTailRemovalRetainsItsTarget) +{ + OrphanFixture f; + const String key = f.orphanKey(); + + NamespaceFoldView view; + RefCoverage cov; + cov.classification = 2; + cov.last_folded_ref_id = RefTxnId{kBuildEpoch + 1, 1}; /// rule (1) satisfied + view.coverage = cov; + view.tail_removal_targets.insert(key); + + String reason; + EXPECT_FALSE(manifestDeletionPremise(view, ManifestKey{key, buildPrefix(5)}, &reason)); + EXPECT_NE(reason.find("removal"), String::npos); + + /// The same view without the removal target admits the deletion, so the retention above is the + /// removal target's doing and nothing else's. + view.tail_removal_targets.clear(); + reason.clear(); + EXPECT_TRUE(manifestDeletionPremise(view, ManifestKey{key, buildPrefix(5)}, &reason)); + EXPECT_TRUE(reason.empty()); +} + +/// Both sweep paths call the ONE predicate: the cursor-paced page must refuse the same body the +/// per-namespace sweep refuses, for the same reason. +TEST(CASSweepDeletionPremise, TheCursorPagePathHonoursTheSamePremise) +{ + OrphanFixture f; + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch, 3}); + + const ManifestSweepResult held = sweepManifestCursorPageForTest(*f.store, "", /*list_budget*/100, /*delete_budget*/10); + EXPECT_EQ(held.deleted, 0u); + EXPECT_GE(held.skipped, 1u); + EXPECT_TRUE(f.orphanExists()); + + /// Consume the seal and the very same page deletes it. + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch + 1, 1}); + const ManifestSweepResult freed = sweepManifestCursorPageForTest(*f.store, "", /*list_budget*/100, /*delete_budget*/10); + EXPECT_EQ(freed.deleted, 1u); + EXPECT_FALSE(f.orphanExists()); +} + +/// One undecodable body must be retained without preventing the same page from deciding a later key. +TEST(CASSweepDeletionPremise, AnUndecodableManifestDoesNotWedgeTheCursorPage) +{ + UndecodableOrphanFixture f; + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch + 1, 1}); + + /// A payload-zone banner that no longer matches its entry path: the exact shape the reproducer + /// produced -- built by hand, because a correct encoder never emits a banner that disagrees with + /// its own entry record. + ManifestEntry inline_entry; + inline_entry.path = "a.txt"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.inline_bytes = "payload"; + PartManifest good; + good.ref = f.orphan; + good.root_namespace_id = f.ns; + good.entries = {inline_entry}; + good.payload_digest = computePayloadDigest(good); + String bytes = encodePartManifest(good); + /// The canonical banner quotes the path. Searching for the former unquoted spelling would fail + /// before the sweep is called. + const size_t at = bytes.find("==> \"a.txt\""); + ASSERT_NE(at, String::npos) << "no banner line to corrupt -- the entry must be Inline, not Blob"; + bytes[at + 5] = 'X'; /// Inside the quoted path, same length, so no other offset shifts. + const PutResult put = f.backend->putIfAbsent(f.orphanKey(), sealObject(FormatId::PartManifest, bytes)); + /// `putIfAbsent` over an existing key writes nothing and reports `PreconditionFailed`, so a + /// silently legal body would make every assertion below pass against the wrong object. + ASSERT_EQ(put.outcome, PutOutcome::Done) << "the poison body was not the one planted"; + + const ManifestId legal = writeManifestRaw( + *f.backend, f.store->layout(), f.ns, ref(5, 0xCD), {blobEntryFor("b", DB::UInt128(2))}); + const String legal_key = f.store->layout().manifestKey(legal); + /// Both keys have the same epoch/build prefix; fixed-width ordinal `0xCD` sorts after poison + /// ordinal `0xAB`, so reaching `legal_key` proves that the page walked beyond the poison key. + ASSERT_LT(f.orphanKey(), legal_key); + + ManifestSweepResult result; + ASSERT_NO_THROW(result = sweepManifestCursorPageForTest(*f.store, "", /*list_budget*/8, /*delete_budget*/8)); + + EXPECT_EQ(result.undecodable, 1u) << "the anomaly must be recorded, not silently swallowed"; + EXPECT_GE(result.skipped, 1u) << "a key the sweep declined to nominate counts as skipped"; + EXPECT_TRUE(f.orphanExists()) << "an undecodable body is retained, never deleted on a guess"; + /// The page reached the end of the keyspace, so the cursor did not stall on the poison key. Assert + /// `wrapped` rather than a moved `next_cursor`: `InMemoryBackend` leaves `next_cursor` empty when no + /// keys remain, so a moved-cursor assertion fails after a correct fix, not before it. + EXPECT_TRUE(result.wrapped); + /// And the strong form: the object beyond the poison key was still decided this page. + EXPECT_FALSE(f.backend->head(legal_key).exists) + << "the sweep stopped at the poison key instead of walking past it"; +} + +/// WHAT THE PREMISE COSTS, pinned so it is a stated behaviour rather than something a later reader +/// discovers. The pure pre-precommit orphan -- a manifest body staged by a writer that crashed before +/// appending any ref record for it -- lives under a namespace whose ref stream may not exist at all. +/// Such a namespace never enters the fold's universe, so no round ever seals a cursor for it, so no +/// epoch's closing seal is ever consumed for it, so the premise retains its debris INDEFINITELY. That +/// is the safe direction and it is deliberate, but it is not "delay": reclaiming this class needs the +/// sweep's own rework (registers R2/R3, Stage B) -- the writer duty queue that knows what it staged, and +/// the nomination path. The premise ships as the safety floor, not as the reclaim policy. +TEST(CASSweepDeletionPremise, DebrisUnderANamespaceTheFoldNeverWalksIsRetainedIndefinitely) +{ + OrphanFixture f; + /// No ref stream, no coverage row -- repeated passes change nothing. + std::vector warnings; + for (int pass = 0; pass < 3; ++pass) + EXPECT_EQ(sweepNamespace(*f.store, f.ns, buildPrefix(5), &warnings), 0u) << "pass " << pass; + + EXPECT_TRUE(f.orphanExists()); + EXPECT_EQ(warnings.size(), 3u) << "every pass reports the retention rather than going quiet"; +} + +/// RETENTION IS VISIBLE ON THE PATH THAT ACTUALLY SWEEPS. `planManifestCursorPage` has no `warnings` +/// out-param -- the background sweep answers to nobody but its phase row -- so the premise's refusals +/// have to leave the process as COUNTERS or not at all. In Stage A that is nearly the whole story of +/// the sweep, because rule (1) is satisfiable only for a closed-and-folded epoch. +/// +/// Non-vacuous by construction: two namespaces on ONE page are retained for DIFFERENT reasons, so a +/// counter wired to the wrong class, or one bucket catching everything, changes the answer. A +/// single-reason page would pass against a single mislabelled counter. +TEST(CASSweepDeletionPremise, DistinctRetainReasonsLandInDistinctCounters) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + /// Namespace A: a cursor still INSIDE the build's epoch -> rule (1), `unconsumed_seal`. + const RootNamespace ns_a{"00/aa@cas@"}; + const ManifestRef ref_a = ref(5, 0xA1); + casAdmitRecoverableEntry(*backend, layout, ns_a); + writeManifestRaw(*backend, layout, ns_a, ref_a, {blobEntryFor("a", DB::UInt128(1))}); + seedFoldCursorForTest(*backend, layout, ns_a, RefTxnId{kBuildEpoch, 3}); + + /// Namespace B: a HELD row whose cursor is well above the build's epoch, so rule (1) is satisfied + /// and the hold is demonstrably what retained it. + const RootNamespace ns_b{"00/bb@cas@"}; + const ManifestRef ref_b = ref(5, 0xB1); + casAdmitRecoverableEntry(*backend, layout, ns_b); + writeManifestRaw(*backend, layout, ns_b, ref_b, {blobEntryFor("b", DB::UInt128(2))}); + const RefHold hold{.reason = HoldReason::BodyUndecodable, + .offending_position = RefTxnId{kBuildEpoch + 1, 9}, + .retry_count = 1, .next_retry_round = 4}; + seedFoldCursorForTest(*backend, layout, ns_b, RefTxnId{kBuildEpoch + 1, 8}, hold); + + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/6); + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); + + EXPECT_EQ(result.deleted, 0u); + EXPECT_EQ(result.retained_unconsumed_seal, 1u) << "namespace A's cursor is inside its build's epoch"; + EXPECT_EQ(result.retained_hold, 1u) << "namespace B is held"; + EXPECT_EQ(result.retained_no_coverage, 0u); + EXPECT_EQ(result.retained_tail_removal, 0u); + EXPECT_GE(result.skipped, 2u) << "both retentions are also ordinary skips"; + + /// The rollup an operator reads. The two classes tie at one each and the tie resolves by enum + /// order, which is what keeps an unchanged pool reporting an unchanged verdict pass after pass. + const auto top = result.topRetainReason(); + EXPECT_EQ(top.second, 1u); + EXPECT_EQ(top.first, SweepRetainClass::Hold); + EXPECT_EQ(String{sweepRetainClassName(SweepRetainClass::UnconsumedSeal)}, "unconsumed_seal"); + + /// A page with no candidates reports nothing: the counters carry the premise's own refusals, not + /// ordinary skips. + auto empty_backend = std::make_shared(); + auto empty_store = openPoolForTest(empty_backend); + const ManifestSweepResult nothing = + sweepManifestCursorPageForTest(*empty_store, "", /*list_budget*/100, /*delete_budget*/10); + EXPECT_EQ(nothing.topRetainReason().first, SweepRetainClass::None); + EXPECT_EQ(nothing.topRetainReason().second, 0u); +} + +/// STAGE B SEAM (registers R2/R3). The premise is the per-manifest SAFETY floor and nothing else: it +/// says when a body may go, never who nominates it or when. The sweep's own rework -- the writer duty +/// queue that reclaims its own live epoch's debris, and the nomination path -- attaches here, and must +/// satisfy this predicate rather than replace it. + +/// Uncertainty rule, budget arm. A candidate the page never DECIDED on -- the delete budget ran out +/// before it -- is retained, and the cursor must not step over it: the sweep's cursor is a +/// cleanup-progress hint whose skipped range is not revisited until a full wrap, so advancing past an +/// undecided candidate converts "retained this round" into "unexamined for a whole cycle". +TEST(CASSweepDeletionPremise, AnExhaustedDeleteBudgetRetainsAndDoesNotStepOverTheRest) +{ + OrphanFixture f; + const ManifestRef second = ref(5, 0xAC); + const ManifestRef third = ref(5, 0xAD); + writeManifestRaw(*f.backend, f.store->layout(), f.ns, second, {blobEntryFor("b", DB::UInt128(2))}); + writeManifestRaw(*f.backend, f.store->layout(), f.ns, third, {blobEntryFor("c", DB::UInt128(3))}); + seedFoldCursorForTest(*f.backend, f.store->layout(), f.ns, RefTxnId{kBuildEpoch + 1, 1}); + + const ManifestSweepResult first = sweepManifestCursorPageForTest(*f.store, "", /*list_budget*/100, /*delete_budget*/1); + EXPECT_EQ(first.deleted, 1u); + EXPECT_FALSE(first.wrapped) + << "the page stopped on an exhausted budget with candidates left, so it did not reach the end"; + ASSERT_FALSE(first.next_cursor.empty()); + + /// Resume: the two survivors are still ahead of the cursor, so a budgeted continuation reaches them. + const ManifestSweepResult second_page = + sweepManifestCursorPageForTest(*f.store, first.next_cursor, /*list_budget*/100, /*delete_budget*/10); + EXPECT_EQ(second_page.deleted, 2u) + << "the cursor must not have stepped over the candidates the exhausted budget left undecided"; + + size_t surviving = 0; + for (const ManifestRef & r : {f.orphan, second, third}) + if (f.backend->head(f.store->layout().manifestKey(ManifestId{f.ns, r})).exists) + ++surviving; + EXPECT_EQ(surviving, 0u); +} + +/// MANDATORY liveness proof: a namespace whose committed-tail recovery walk +/// can never finish within one round's `sweep_recovery_op_budget` must not wedge the cursor page for +/// every subsequent round. Six eligible candidates share ONE namespace whose tail is ~200 unrelated +/// committed transactions above the fold cursor -- far more than the tiny per-round recovery-op budget +/// can traverse -- so `activeManifestKeys` reports `recovery_incomplete` on every attempt, every one of +/// this namespace's candidates is retained (never nominated, never deleted), yet the page still DECIDES +/// them (a retained candidate is a decision) and the cursor advances across pages until the whole +/// keyspace is covered. +TEST(CASSweepDeletionPremise, RecoveryWorkBudgetRetainsAndConvergesWithoutWedgingTheCursor) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + /// `casAdmitEntry` (bare, no `_ckpt`) rather than `casAdmitRecoverableEntry` (which pre-seeds an + /// EMPTY `_ckpt`, `committed_through = nullopt`): the tail below is built entirely from real + /// `publishCommittedTransition` calls, whose first call needs `readCkpt` to see NOTHING yet so it + /// takes the fresh-`_ckpt` `putIfAbsent` path instead of `advanceRecoverableCkptForRawFixture`'s + /// monotonic-advance-from-existing-value path (which throws on a null `committed_through`). + casAdmitEntry(*backend, layout, ns); + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/1000); + + /// Six orphan candidates, all eligible (build_sequence << min_active), none owned by any ref. + constexpr int kCandidates = 6; + for (int i = 1; i <= kCandidates; ++i) + writeManifestRaw(*backend, layout, ns, ref(i, 1), + {blobEntryFor("c" + std::to_string(i), DB::UInt128(static_cast(i)))}); + + /// A committed tail of ~200 UNRELATED transactions above the fold cursor. None of these need a + /// manifest body of their own -- the recovery walk only GETs and decodes the ref-log transactions, + /// never the bodies they name. + constexpr int kTailSize = 200; + for (int i = 0; i < kTailSize; ++i) + publishCommittedTransition(*backend, layout, ns, "tail" + std::to_string(i), std::nullopt, ref(2000 + i, 1)); + seedFoldCursorForTest(*backend, layout, ns, RefTxnId{kBuildEpoch, 1}); + + uint64_t total_retained_work_budget = 0; + uint64_t total_skipped = 0; + uint64_t total_deleted = 0; + String cursor; + bool wrapped = false; + int pages = 0; + for (; pages < 10 && !wrapped; ++pages) + { + /// A FRESH budget every page, exactly like production's one-instance-per-round contract -- + /// the same namespace's recovery walk re-attempts and re-exhausts every time, by design (a + /// pathological namespace does not get to starve every OTHER page of budget forever). + GcRoundWorkBudget budget; + budget.max_sweep_recovery_ops = 5; /// far below the ~200-record tail + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, cursor, /*list_budget*/3, /*delete_budget*/10, &budget); + + /// LIVENESS: every page decides at least one candidate (a retained one counts) or wraps. + EXPECT_TRUE(result.skipped > 0 || result.deleted > 0 || result.wrapped) + << "page " << pages << " decided nothing and did not wrap -- a wedge"; + /// Every namespace this page touches hits the recovery-op-exhausted cause AT LEAST once -- + /// only the FIRST candidate of an errored namespace on a page carries the specific retain-class + /// counter (the SAME pre-existing convention `retained_no_coverage`/`retained_hold` already + /// use); every other candidate of that namespace still lands in the generic `skipped` tally. + EXPECT_GE(result.retained_work_budget, 1u) + << "page " << pages << " never attributed a candidate to the recovery-budget cause"; + + total_retained_work_budget += result.retained_work_budget; + total_skipped += result.skipped; + total_deleted += result.deleted; + wrapped = result.wrapped; + ASSERT_NE(cursor, result.next_cursor) << "page " << pages << " made no cursor progress"; + cursor = result.next_cursor; + } + + EXPECT_TRUE(wrapped) << "the whole small keyspace must be fully covered well within 10 pages"; + EXPECT_EQ(total_deleted, 0u) << "the pathological namespace's candidates are never safe to nominate"; + EXPECT_EQ(total_skipped, static_cast(kCandidates)) + << "every one of the six candidates was decided (skipped), none silently dropped from the page"; + EXPECT_GE(total_retained_work_budget, 1u); + for (int i = 1; i <= kCandidates; ++i) + EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, ref(i, 1)})).exists) + << "candidate " << i << " must survive: it was never proven safe to delete"; +} + +/// The per-page NAMESPACE cap. Two otherwise-independently-deletable +/// namespaces share one page; with `max_sweep_namespaces = 1`, only the first namespace this page +/// touches gets a protection view built at all -- the second is retained under the work-budget cause, +/// never given a partial or best-effort view. +TEST(CASSweepDeletionPremise, NamespaceWorkBudgetCapsDistinctViewsPerPage) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const Layout & layout = store->layout(); + + const RootNamespace ns_a{"00/aa@cas@"}; + const RootNamespace ns_b{"00/bb@cas@"}; + const ManifestRef ref_a = ref(5, 0xA1); + const ManifestRef ref_b = ref(5, 0xB1); + casAdmitRecoverableEntry(*backend, layout, ns_a); + casAdmitRecoverableEntry(*backend, layout, ns_b); + writeManifestRaw(*backend, layout, ns_a, ref_a, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns_b, ref_b, {blobEntryFor("b", DB::UInt128(2))}); + /// Both namespaces satisfy rule (1) (cursor past the build's own epoch) and have no committed tail + /// at all (`_ckpt.committed_through` unset), so absent the namespace cap BOTH would delete. + seedFoldCursorForTest(*backend, layout, ns_a, RefTxnId{kBuildEpoch + 1, 1}); + seedFoldCursorForTest(*backend, layout, ns_b, RefTxnId{kBuildEpoch + 1, 1}); + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/6); + + GcRoundWorkBudget budget; + budget.max_sweep_namespaces = 1; + + const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10, &budget); + + EXPECT_EQ(result.deleted, 1u) << "exactly one namespace's view could be built this page"; + EXPECT_EQ(result.retained_work_budget, 1u) + << "the other namespace's candidate is retained, never decided from a missing view"; + EXPECT_EQ(budget.sweep_namespaces_used, 1u); + + size_t surviving = 0; + for (const auto & p : std::vector>{{ns_a, ref_a}, {ns_b, ref_b}}) + if (backend->head(layout.manifestKey(ManifestId{p.first, p.second})).exists) + ++surviving; + EXPECT_EQ(surviving, 1u) << "exactly one candidate remains -- the one whose namespace had no budget left"; +} diff --git a/src/Disks/tests/gtest_cas_text_format.cpp b/src/Disks/tests/gtest_cas_text_format.cpp new file mode 100644 index 000000000000..4371afb3f9e8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_text_format.cpp @@ -0,0 +1,292 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int UNKNOWN_FORMAT_VERSION; +} + +namespace +{ +/// Run `f` and require a DB::Exception with exactly `code`. +template +void expectCode(int code, F && f) +{ + try + { + f(); + FAIL() << "expected exception code " << code; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), code); + } +} +} + +/// ---- Task 2: FormatId entries for refsnaplog / blob meta / heartbeat ---- + +TEST(CASFormatIds, NewIdsExistWithFrozenValues) +{ + EXPECT_EQ(static_cast(FormatId::RefLog), 19); + EXPECT_EQ(static_cast(FormatId::RefSnapshot), 20); + EXPECT_EQ(static_cast(FormatId::BlobMeta), 21); + EXPECT_EQ(static_cast(FormatId::GcHeartbeat), 22); + /// Every id, old and new, has a change-point ladder (BASELINE until a real bump). + for (auto id : {FormatId::RefLog, FormatId::RefSnapshot, FormatId::BlobMeta, FormatId::GcHeartbeat}) + EXPECT_FALSE(changePoints(id).empty()); +} + +/// ---- Task 3: per-format traits registry ---- + +TEST(CASFormatTraits, CompleteUniqueAndGated) +{ + /// Completeness: every FormatId except the reserved Roster has traits. + const FormatId all[] = {FormatId::Blob, FormatId::GcState, FormatId::PoolMeta, + FormatId::GcOutcomes, FormatId::PartManifest, FormatId::RunFile, + FormatId::FoldSeal, FormatId::Owner, FormatId::ServerEpoch, FormatId::MountLease, + FormatId::RefLog, FormatId::RefSnapshot, FormatId::BlobMeta, FormatId::GcHeartbeat, + FormatId::RefCkpt, FormatId::RefCatalog, FormatId::GcMaintenanceState}; + std::set types; + for (FormatId id : all) + { + const FormatTraits & t = traitsFor(id); + EXPECT_EQ(t.id, id); + EXPECT_TRUE(t.type.starts_with("cas_")) << t.type; + EXPECT_TRUE(types.insert(t.type).second) << "duplicate type " << t.type; + EXPECT_EQ(traitsForType(t.type), &t); + } + EXPECT_EQ(traitsForType("cas_nope"), nullptr); +#ifndef DEBUG_OR_SANITIZER_BUILD + /// traitsFor(Roster) throws LOGICAL_ERROR (a reserved/unreachable FormatId), which aborts the + /// whole process in debug/sanitizer builds instead of behaving like a catchable exception -- + /// CASFormatTraitsDeathTest below proves the abort positively in those builds instead. + EXPECT_THROW(traitsFor(FormatId::Roster), DB::Exception); +#endif + /// Deterministic formats are pinned raw + strict; spot-check the two. + EXPECT_EQ(traitsFor(FormatId::RunFile).compression, CompressionPolicy::PinnedRaw); + EXPECT_EQ(traitsFor(FormatId::RunFile).strictness, KeyStrictness::Strict); + EXPECT_EQ(traitsFor(FormatId::FoldSeal).compression, CompressionPolicy::PinnedRaw); + EXPECT_EQ(traitsFor(FormatId::FoldSeal).strictness, KeyStrictness::Strict); + /// .zst key suffix is exactly the Always set (can-grow-large types). + EXPECT_EQ(storedSuffix(FormatId::RefSnapshot), ".zst"); + EXPECT_EQ(storedSuffix(FormatId::RefLog), ".zst"); + EXPECT_EQ(storedSuffix(FormatId::PartManifest), ".zst"); + EXPECT_EQ(storedSuffix(FormatId::GcOutcomes), ".zst"); + EXPECT_EQ(storedSuffix(FormatId::PoolMeta), ""); + EXPECT_EQ(storedSuffix(FormatId::FoldSeal), ""); + EXPECT_EQ(storedSuffix(FormatId::RunFile), ""); +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +/// Debug/sanitizer-build counterpart to CompleteUniqueAndGated's Roster check: LOGICAL_ERROR aborts +/// the process here instead of throwing a catchable exception, so the check must be a death test +/// (same pattern as CASBlobDigestDeathTest in gtest_cas_blob_digest.cpp). +TEST(CASFormatTraitsDeathTest, TraitsForRosterAborts) +{ + EXPECT_DEATH({ (void)traitsFor(FormatId::Roster); }, ""); +} +#endif + +/// ---- Task 4: JSON micro-vocabulary + JsonObjectReader ---- + +TEST(CASJsonVocab, WriteAndReadBack) +{ + CasJsonWriter out; + bool first = true; + writeKey(out, "tag", first); + writeHex128Value(out, hexToU128("000102030405060708090a0b0c0d0e0f")); + writeKey(out, "seq", first); + writeU64StringValue(out, 18446744073709551615ULL); + writeKey(out, "n", first); + writeIntText(7, out); + writeKey(out, "ref", first); + writeStringValue(out, "t-1/all_1_2_0\n\"quoted\""); + closeObject(out, first); + const String rendered = std::move(out).take(); + EXPECT_EQ(rendered.substr(0, 45), R"({"tag":"000102030405060708090a0b0c0d0e0f","se)"); + + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Strict, "test"); + String key; + ASSERT_TRUE(r.nextKey(key)); EXPECT_EQ(key, "tag"); + EXPECT_EQ(r.readHex128(), hexToU128("000102030405060708090a0b0c0d0e0f")); + ASSERT_TRUE(r.nextKey(key)); EXPECT_EQ(key, "seq"); + EXPECT_EQ(r.readU64String(), 18446744073709551615ULL); + ASSERT_TRUE(r.nextKey(key)); EXPECT_EQ(key, "n"); + EXPECT_EQ(r.readU64Number(), 7u); + ASSERT_TRUE(r.nextKey(key)); EXPECT_EQ(key, "ref"); + EXPECT_EQ(r.readString(), "t-1/all_1_2_0\n\"quoted\""); + EXPECT_FALSE(r.nextKey(key)); +} + +TEST(CASJsonVocab, FailClosedRules) +{ + auto reader = [](std::string_view text, KeyStrictness s, auto && consume) + { + DB::ReadBufferFromMemory in(text.data(), text.size()); + JsonObjectReader r(in, s, "test"); + consume(r); + }; + /// duplicate key + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"a":1,"a":2})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + while (r.nextKey(k)) r.readU64Number(); + }); }); + /// unknown key: Tolerant skips (nested value), Strict rejects + reader(R"({"zz":{"deep":[1,2]},"n":5})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + ASSERT_TRUE(r.nextKey(k)); r.skipUnknown(k); + ASSERT_TRUE(r.nextKey(k)); EXPECT_EQ(r.readU64Number(), 5u); + EXPECT_FALSE(r.nextKey(k)); + }); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"zz":1})", KeyStrictness::Strict, [](auto & r) + { + String k; + ASSERT_TRUE(r.nextKey(k)); r.skipUnknown(k); + }); }); + /// critical key fails closed regardless of strictness + expectCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { reader(R"({"!x":1})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + ASSERT_TRUE(r.nextKey(k)); r.skipUnknown(k); + }); }); + /// whitespace is not canonical + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({ "a":1})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + r.nextKey(k); + }); }); + /// bad hex width / junk in u64 string + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"h":"0102"})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + r.nextKey(k); r.readHex128(); + }); }); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"s":"12x"})", KeyStrictness::Tolerant, [](auto & r) + { + String k; + r.nextKey(k); r.readU64String(); + }); }); +} + +/// ---- Task 5: header line, trailer line, readLine ---- + +TEST(CASTextHeader, WriteExpectSniffGate) +{ + CasJsonWriter out; + writeHeaderLine(out, FormatId::PoolMeta); + const String rendered = std::move(out).take(); + EXPECT_EQ(rendered, fmt::format("{{\"type\":\"cas_pool_meta\",\"v\":{}}}\n", currentCompatibilityVersion())); + + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + const TextHeader h = expectHeaderLine(in, FormatId::PoolMeta); + EXPECT_EQ(h.type, "cas_pool_meta"); + EXPECT_EQ(h.v, currentCompatibilityVersion()); + EXPECT_TRUE(in.eof()); + + const auto sniffed = sniffHeaderLine(rendered); + ASSERT_TRUE(sniffed.has_value()); + EXPECT_EQ(sniffed->type, "cas_pool_meta"); + EXPECT_FALSE(sniffHeaderLine("PAR1 not a cas object").has_value()); + + /// wrong type -> CORRUPTED_DATA; future v -> UNKNOWN_FORMAT_VERSION + /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes + /// the header gate, which is the point — the BODY is what has to fail here. + const String wrong = "{\"type\":\"cas_owner\",\"v\":3}\n"; + DB::ReadBufferFromMemory in2(wrong.data(), wrong.size()); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { expectHeaderLine(in2, FormatId::PoolMeta); }); + const String future = fmt::format("{{\"type\":\"cas_pool_meta\",\"v\":{}}}\n", currentCompatibilityVersion() + 1); + DB::ReadBufferFromMemory in3(future.data(), future.size()); + expectCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { expectHeaderLine(in3, FormatId::PoolMeta); }); + + const String out_of_range = "{\"type\":\"cas_pool_meta\",\"v\":4294967299}\n"; + DB::ReadBufferFromMemory in4(out_of_range.data(), out_of_range.size()); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { expectHeaderLine(in4, FormatId::PoolMeta); }); +} + +TEST(CASTextLines, ReadLineAndTrailer) +{ + CasJsonWriter out; + writeTrailerLine(out, 42); + EXPECT_EQ(std::move(out).take(), "{\"n\":42}\n"); + + const String two = "abc\ndef\n"; + DB::ReadBufferFromMemory in(two.data(), two.size()); + EXPECT_EQ(readLine(in, 16, "test"), "abc"); + EXPECT_EQ(readLine(in, 16, "test"), "def"); + /// missing terminator and over-cap both fail closed + const String noterm = "abc"; + DB::ReadBufferFromMemory in2(noterm.data(), noterm.size()); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { readLine(in2, 16, "test"); }); + DB::ReadBufferFromMemory in3(two.data(), two.size()); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { readLine(in3, 2, "test"); }); +} + +/// ---- Task 6: the zstd arm ---- + +TEST(CASZstdArm, SealOpenPolicyAndCaps) +{ + /// Always types compress regardless of size (no threshold — the .zst key must be + /// constructible without knowing the body); a raw body is still readable (repair path). + /// `v:3` here is NOT the "any version <= G_BUILD passes" case the other negative bodies rely on: + /// `cas_ref_snap`'s own `changePoints` floor is generation 4, so a generation-3 ref snapshot is not + /// readable by this build in principle. It passes the header gate only because nothing consults + /// `changePoints` at decode time yet -- the gate is `v > G_BUILD` alone. Once a per-class floor is + /// wired in, this literal must move to `G_BUILD`; the test's subject is the truncated BODY, not the + /// version. + const String small = "{\"type\":\"cas_ref_snap\",\"v\":3}\n{}\n"; + const String sealed_small = sealObject(FormatId::RefSnapshot, small); + ASSERT_TRUE(looksZstd(sealed_small)); + EXPECT_EQ(openObject(FormatId::RefSnapshot, sealed_small), small); + EXPECT_EQ(openObject(FormatId::RefSnapshot, small), small); + + String big = "{\"type\":\"cas_ref_snap\",\"v\":3}\n{\"pad\":\""; + big += String(8192, 'a'); + big += "\"}\n"; + const String sealed = sealObject(FormatId::RefSnapshot, big); + ASSERT_TRUE(looksZstd(sealed)); + EXPECT_LT(sealed.size(), big.size()); + EXPECT_EQ(openObject(FormatId::RefSnapshot, sealed), big); + + /// Never and PinnedRaw formats never compress on write and reject compressed input on read. + EXPECT_EQ(sealObject(FormatId::FoldSeal, big), big); + EXPECT_EQ(sealObject(FormatId::PoolMeta, big), big); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { openObject(FormatId::FoldSeal, sealed); }); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { openObject(FormatId::PoolMeta, sealed); }); + + /// Declared content size over the cap fails BEFORE the output allocation: 65 MiB of text + /// against RefSnapshot's 64 MiB cap (compresses to ~nothing, so the test is cheap on disk + /// bytes; the 65 MiB source string is the only big allocation). + const String over(65 * 1024 * 1024, 'b'); + const String sealed_over = sealObject(FormatId::RefSnapshot, over); + ASSERT_TRUE(looksZstd(sealed_over)); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { openObject(FormatId::RefSnapshot, sealed_over); }); + + /// A flipped byte inside the frame is caught by zstd (frame checksum is on). + String corrupted = sealed; + corrupted[corrupted.size() / 2] ^= 0x01; + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { openObject(FormatId::RefSnapshot, corrupted); }); +} + +TEST(CASTextValueEscaping, ForwardSlashPinnedUnescaped) +{ + /// Goes RED if the global escape_forward_slashes default ever leaks back into CAS string values. + /// CAS values are dense with '/' (ref-paths, fold-seal keys); their bytes must be CAS-owned so + /// cas_fold_seal byte-determinism and every golden text file are independent of the global default. + CasJsonWriter out; + writeStringValue(out, "ns/shard/all_1_2_0"); + EXPECT_EQ(std::move(out).take(), "\"ns/shard/all_1_2_0\""); +} diff --git a/src/Disks/tests/gtest_cas_truncate_reclaim.cpp b/src/Disks/tests/gtest_cas_truncate_reclaim.cpp new file mode 100644 index 000000000000..0a76c338cdb5 --- /dev/null +++ b/src/Disks/tests/gtest_cas_truncate_reclaim.cpp @@ -0,0 +1,282 @@ +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +/// B140 regression guard. The soak's Phase-1 sync run did a `TRUNCATE TABLE` at op 450 and then +/// observed fsck `unreachable` STUCK above zero (1751) while the incremental GC reported +/// `candidates=0` — i.e. the GC believed it was done while orphaned blobs remained. This file +/// reproduces the soak shape at the CORE level (no server, no docker): publish many parts that +/// SHARE blobs (dedup), interleave regular GC rounds with the publishes (so trees get expanded +/// into the durable snap exactly as they would during a steady-state insert workload), then +/// perform the SAME removal a Replicated TRUNCATE issues — a per-ref `dropRef` for every part — +/// and drive the GC to a fixpoint. The invariant under test: after the drops are folded and the +/// cascade runs, `runFsck().unreachable` reaches 0 (every shared blob is reclaimed). +/// +/// A `dropNamespace` variant is included as well (the path `removeRecursive` takes for a whole +/// table dir, e.g. DROP TABLE): it journals one Remove per former ref, so the cascade should fold +/// it identically. + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; + +namespace +{ + +PoolPtr openTestPool(std::shared_ptr & out_backend) +{ + out_backend = std::make_shared(); + return Pool::open(out_backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// Publish one part `ref` with TWO content files whose payloads are passed in. Identical payloads +/// across parts dedup to the SAME blob object (the soak's dedup_ratio ~3.8 comes from exactly this +/// sharing). Returns the manifest id. +ManifestId publishPart2( + const PoolPtr & s, const String & ns, const String & ref, + const String & payload_a, const String & payload_b) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + + ManifestEntry ea; + ea.path = "data.bin"; + ea.placement = EntryPlacement::Blob; + ea.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload_a))}; + + ea.blob_size = payload_a.size(); + + ManifestEntry eb; + eb.path = "data.cmrk3"; + eb.placement = EntryPlacement::Blob; + eb.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(payload_b))}; + + eb.blob_size = payload_b.size(); + + /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob -> promote. + const ManifestId id = build->stageManifest({ea, eb}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload_a), BlobSource::fromString(payload_a)); + build->putBlob(idOf(payload_b), BlobSource::fromString(payload_b)); + build->promote(nsr, ref, build->buildId(), id); + return id; +} + +/// Whether the CURRENT retired list (any gc-shard) still holds an entry — the ack-floor deletion pipeline +/// (condemn -> graduate -> delete) is in flight while this is true. +bool anyRetiredPending(const PoolPtr & s) +{ + /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// separate retired list — reconstruct the in-flight set from the seal. + return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); +} + +/// Run regular GC rounds until a fixpoint over the ACK-FLOOR round. A condemned blob is deleted only a +/// few rounds after its removal folds (condemn -> graduate once the ack floor passes it -> delete), so the +/// loop advances the store's own mount ack after each round (`renewWatermarkOnce` runs the beat) and stays +/// alive while ANY work counter is nonzero OR the current retired list still holds an in-flight entry. +size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) +{ + size_t rounds = 0; + for (; rounds < max_rounds; ++rounds) + { + const RoundReport rep = DB::Cas::tests::runRegularRoundReclaiming(gc); + if (!rep.acquired_lease) + continue; + s->renewWatermarkOnce(); + const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 + && rep.replaced == 0 && rep.spared == 0; + if (no_work && !anyRetiredPending(s)) + break; + } + return rounds; +} + +} + +/// The faithful soak repro: many parts sharing blobs, GC interleaved with the publishes, then a +/// per-ref drop of EVERY ref (Replicated TRUNCATE), then GC to a fixpoint. fsck.unreachable must +/// reach 0 — no orphaned blob may survive. +TEST(CASTruncateReclaim, PerRefDropOfSharedBlobsReclaimsToZero) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"srv1/tbl"}; + + constexpr int N = 32; + + /// Publish N parts. Payloads are chosen so blobs are SHARED across parts: data.bin cycles + /// through 8 distinct contents, data.cmrk3 through 4 — heavy dedup, like the soak. + std::vector refs; + for (int i = 0; i < N; ++i) + { + const String ref = "all_" + std::to_string(i) + "_" + std::to_string(i) + "_0"; + refs.push_back(ref); + const String pa = "data-" + std::to_string(i % 8); + const String pb = "mark-" + std::to_string(i % 4); + publishPart2(s, ns.string(), ref, pa, pb); + + /// Interleave a GC round every few publishes, so the live trees get EXPANDED into the + /// durable snap during the insert phase (steady-state GC, as in the soak). + if (i % 5 == 4) + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + } + } + + /// Steady-state GC has nothing to reclaim while the refs are live. + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runGcToFixpoint(s, gc); + const FsckReport before = runFsck(*s, /*detail=*/false); + EXPECT_EQ(before.unreachable, 0u) << "live pool must have no unreachable debris"; + EXPECT_EQ(before.dangling, 0u); + EXPECT_GT(before.reachable, 0u); + } + + /// TRUNCATE: a Replicated TRUNCATE removes each part dir, which routes to dropRef per ref. + for (const String & ref : refs) + s->dropRef(ns, ref); + + /// Every publishing build finished; advance the durable watermark floor past their seqs so the + /// Task 10 build-watermark guard no longer spares the now-dropped objects (production does this + /// via the background renewer ~2s; here the renewer is off, so drive it explicitly). + s->renewWatermarkOnce(); + + /// Drive GC to a fixpoint and require full reclamation — this is the B140 assertion. + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + const size_t rounds = runGcToFixpoint(s, gc); + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.dangling, 0u) << "TRUNCATE must never lose a reachable object"; + EXPECT_EQ(after.unreachable, 0u) + << "B140: orphaned blobs survived TRUNCATE after " << rounds + << " GC rounds (reachable=" << after.reachable + << ", unreachable=" << after.unreachable << ")"; + EXPECT_EQ(after.reachable, 0u) << "no refs remain, so nothing should be reachable"; + } +} + +/// Mirrors the soak exactly: TRUNCATE at "op 450" (drop every live ref), then CONTINUE inserting +/// (the soak's ops 451..599 had min_op=451) while the GC keeps running, then a final drive to a +/// fixpoint. The post-truncate inserts must not stall reclamation of the pre-truncate orphans. +/// Also asserts a TIGHT bound on the number of rounds reclamation needs (the soak's 180s budget at +/// gc_interval=30s only buys ~6 rounds, so the core must reach a fixpoint well inside that). +TEST(CASTruncateReclaim, TruncateThenKeepInsertingStillReclaims) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"srv1/tbl"}; + + /// Pre-truncate generation (the soak's ops < 451). + std::vector pre_refs; + for (int i = 0; i < 24; ++i) + { + const String ref = "pre_" + std::to_string(i); + pre_refs.push_back(ref); + publishPart2(s, ns.string(), ref, "p-data-" + std::to_string(i % 6), "p-mark-" + std::to_string(i % 3)); + if (i % 5 == 4) + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + } + } + + /// TRUNCATE: drop every pre-truncate ref (per-ref dropRef). + for (const String & ref : pre_refs) + s->dropRef(ns, ref); + + /// Continue inserting AFTER the truncate (the soak's ops 451..599), interleaving GC rounds. + for (int i = 0; i < 24; ++i) + { + publishPart2(s, ns.string(), "post_" + std::to_string(i), + "q-data-" + std::to_string(i % 6), "q-mark-" + std::to_string(i % 3)); + if (i % 5 == 4) + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + } + } + + /// All publishing builds finished; advance the durable watermark floor past their seqs so the + /// Task 10 build-watermark guard no longer spares the dropped objects (the background renewer is + /// off in this test, so drive it explicitly — production renews ~2s off the write path). + s->renewWatermarkOnce(); + + /// Drive to a fixpoint. unreachable must reach 0 (the pre-truncate orphans are gone) while the + /// post-truncate refs stay reachable. + Gc gc(s, hexToU128("00000000000000000000000000000001")); + const size_t rounds = runGcToFixpoint(s, gc); + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.dangling, 0u); + EXPECT_EQ(after.unreachable, 0u) + << "B140: pre-truncate orphans survived after " << rounds << " GC rounds"; + EXPECT_GT(after.reachable, 0u) << "post-truncate refs must stay reachable"; + /// Round bound: the ack-floor pipeline adds a bounded, constant number of rounds over the old + /// fold+delete (condemn -> graduate once the ack floor passes -> delete, with the ack kept current + /// each round). The dead subgraph still drains in a small, constant number of rounds — not O(orphans). + EXPECT_LE(rounds, 8u) << "reclamation took too many rounds (ack-floor pipeline is a small constant)"; +} + +/// The DROP TABLE path: removeRecursive of a table dir calls dropNamespace, which journals one +/// Remove per former ref. Same reclamation invariant. +TEST(CASTruncateReclaim, DropNamespaceLeavesSharedBlobDebrisForPerpetualSweep) +{ + std::shared_ptr b; + auto s = openTestPool(b); + const RootNamespace ns{"srv1/tbl"}; + + constexpr int N = 32; + for (int i = 0; i < N; ++i) + { + const String ref = "all_" + std::to_string(i) + "_" + std::to_string(i) + "_0"; + const String pa = "data-" + std::to_string(i % 8); + const String pb = "mark-" + std::to_string(i % 4); + publishPart2(s, ns.string(), ref, pa, pb); + if (i % 5 == 4) + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + DB::Cas::tests::runRegularRoundReclaiming(gc); + } + } + + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + runGcToFixpoint(s, gc); + } + + /// DROP TABLE: the whole namespace is tombstoned at once (one Remove per ref in the journal). + s->dropNamespace(ns); + + /// Every publishing build finished; advance the durable watermark floor past their seqs so the + /// Task 10 build-watermark guard no longer spares the dropped objects (renewer off here). + s->renewWatermarkOnce(); + + { + Gc gc(s, hexToU128("00000000000000000000000000000001")); + const size_t rounds = runGcToFixpoint(s, gc); + const FsckReport after = runFsck(*s, /*detail=*/false); + EXPECT_EQ(after.dangling, 0u); + /// Removal still performs no lifecycle-specific physical cleanup -- the perpetual sweep and the + /// janitor own the orphaned bytes. What changed is that they can now FINISH: dropping the last + /// namespace leaves an authoritative catalog that decodes to zero entries, which is a positive + /// proof of no live edge rather than the vacuous 0 == 0, so the round's frontier completes and + /// the sweep is no longer suppressed on an emptied pool. + EXPECT_EQ(after.unreachable, 0u) + << "an emptied pool must drain instead of standing still; the sweep owned these blobs and " + "reclaimed them within " << rounds << " GC rounds"; + EXPECT_EQ(after.reachable, 0u); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(s->backend(), s->layout(), ns)) + << "physical debris must not keep the logical namespace life cataloged"; + } +} diff --git a/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp b/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp new file mode 100644 index 000000000000..80d262854555 --- /dev/null +++ b/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp @@ -0,0 +1,159 @@ +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ +BlobRef bh(uint64_t n) { return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(n))}; } +DB::UInt128 s(uint64_t n) { return DB::UInt128(n); } +} + +/// The ledger is pure round-local bookkeeping; test it directly rather than trying to fabricate a +/// lost bucket inside a real fold. The fold-side wiring is covered by the gate: every existing GC +/// test now runs with the ledger armed and would throw if a delta went missing. +TEST(CASTxnApplyLedger, HealthyRoundReportsNothingUnapplied) +{ + TxnApplyLedger ledger; + const uint32_t a = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + const uint32_t b = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 2}); + ledger.markProduced(a); + ledger.markCommitted(a); + ledger.markApplied(a); + ledger.markCommitted(b); /// committed but produced no blob deltas — legitimate + EXPECT_TRUE(ledger.unapplied().empty()); +} + +TEST(CASTxnApplyLedger, CommittedAndProducedButNeverAppliedIsReported) +{ + TxnApplyLedger ledger; + const uint32_t a = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + ledger.markProduced(a); + ledger.markCommitted(a); + ASSERT_EQ(ledger.unapplied().size(), 1u); + EXPECT_EQ(ledger.unapplied().front(), a); +} + +TEST(CASTxnApplyLedger, ClampedTransactionIsNotReported) +{ + /// A clamped log emits deltas into the per-log staging buffer that is then DISCARDED; it is never + /// committed, so it must not be reported unapplied. + TxnApplyLedger ledger; + const uint32_t a = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + ledger.markProduced(a); + EXPECT_TRUE(ledger.unapplied().empty()); +} + +/// The reducers mark `applied` by indexing the raw vector with `BlobDelta::txn_ordinal`, so the +/// ledger's own vectors must stay index-parallel with the ordinals it hands out. Pin that: the +/// ordinal is the position, and every parallel vector grows with it. +TEST(CASTxnApplyLedger, OrdinalsIndexTheParallelVectors) +{ + TxnApplyLedger ledger; + EXPECT_EQ(ledger.open(RootNamespace{"a"}, RefTxnId{1, 7}), 0u); + EXPECT_EQ(ledger.open(RootNamespace{"b"}, RefTxnId{2, 3}), 1u); + ASSERT_EQ(ledger.applied.size(), 2u); + ASSERT_EQ(ledger.produced.size(), 2u); + ASSERT_EQ(ledger.committed.size(), 2u); + ASSERT_EQ(ledger.namespaces.size(), 2u); + EXPECT_EQ(ledger.namespaces[1], "b"); + EXPECT_EQ(ledger.txns[1], (RefTxnId{2, 3})); +} + +/// PROBE B2's reach, pinned as a property rather than left to prose: a delta consumed by a reducer +/// clears its transaction, and only the transaction whose ordinal was never written stays reported. +/// This is the exact shape a delta lost in gc-shard routing produces. +TEST(CASTxnApplyLedger, OnlyTheTransactionWhoseDeltasVanishedIsReported) +{ + TxnApplyLedger ledger; + const uint32_t routed = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + const uint32_t lost = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 2}); + for (const uint32_t o : {routed, lost}) + { + ledger.markProduced(o); + ledger.markCommitted(o); + } + /// The reducer's own write: a raw byte at the delta's ordinal, exactly as + /// `foldDeltasIntoGeneration` performs it. + ledger.applied[routed] = 1; + + ASSERT_EQ(ledger.unapplied().size(), 1u); + EXPECT_EQ(ledger.unapplied().front(), lost); +} + +/// The reducer-side half of probe B2, proven POSITIVELY rather than by the absence of a throw. The +/// three tests above exercise the ledger's own arithmetic; this one exercises the write that +/// `foldDeltasIntoGeneration` performs inside its delta-consumption loop — the only new code on the +/// fold's hot path — and pins that a routed delta marks its ordinal while an ordinal no delta carries +/// stays unmarked. Without this the fold-side wiring would only ever be covered negatively (the gate +/// does not throw), which cannot distinguish "the probe is correct" from "the probe is inert". +TEST(CASTxnApplyLedger, ReducerMarksTheOrdinalOfEveryDeltaItConsumes) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + TxnApplyLedger ledger; + const uint32_t routed = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + const uint32_t absent = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 2}); + + /// Only `routed`'s transaction emitted deltas. `absent`'s ordinal is live in the ledger but no + /// delta carries it — exactly the shape a delta lost before the reducer produces. + std::vector deltas{ + {bh(1), s(1), /*remove*/false, routed}, + {bh(2), s(1), /*remove*/false, routed}, + }; + std::vector runs; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, + /*shard*/0, deltas, runs, + /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, + /*confirm_condemned_marker*/{}, /*out_retired*/nullptr, + /*suppress_destructive*/false, &ledger.applied); + + EXPECT_EQ(ledger.applied[routed], 1) << "the reducer consumed this transaction's deltas but did " + "not mark its ordinal — probe B2 is inert"; + EXPECT_EQ(ledger.applied[absent], 0) << "an ordinal no delta carries must never be marked"; + + /// And the verdict follows from those bits: a committed+produced transaction whose deltas never + /// arrived is the one reported. + for (const uint32_t o : {routed, absent}) + { + ledger.markProduced(o); + ledger.markCommitted(o); + } + ASSERT_EQ(ledger.unapplied().size(), 1u); + EXPECT_EQ(ledger.unapplied().front(), absent); +} + +/// The reducer must mark a REMOVAL delta too. Removals are the direction that can legitimately +/// collapse to nothing inside the set merge (an unmatched `-1` changes no state and emits no row), so +/// a mark placed at run flush instead of at consumption would silently skip exactly this case and +/// report a healthy round as lossy. +TEST(CASTxnApplyLedger, ReducerMarksAnUnmatchedRemovalDelta) +{ + InMemoryBackend backend; + Layout layout{"pool"}; + + TxnApplyLedger ledger; + const uint32_t removal = ledger.open(RootNamespace{"ns"}, RefTxnId{1, 1}); + ledger.markProduced(removal); + ledger.markCommitted(removal); + + /// A `-1` for an edge no prior run ever activated: a per-key no-op by design. + std::vector deltas{{bh(1), s(1), /*remove*/true, removal}}; + std::vector runs; + RetiredMergeResult merged; + foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, + /*shard*/0, deltas, runs, + /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, + /*confirm_condemned_marker*/{}, &merged, + /*suppress_destructive*/false, &ledger.applied); + + EXPECT_EQ(merged.unmatched_removes, 1u) << "the fixture must actually stage an unmatched removal"; + EXPECT_EQ(ledger.applied[removal], 1); + EXPECT_TRUE(ledger.unapplied().empty()) + << "a legitimate no-op removal must not read as a lost transaction"; +} diff --git a/src/Disks/tests/gtest_cas_upload_detached.cpp b/src/Disks/tests/gtest_cas_upload_detached.cpp new file mode 100644 index 000000000000..6f6f341f4bb6 --- /dev/null +++ b/src/Disks/tests/gtest_cas_upload_detached.cpp @@ -0,0 +1,732 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::loadMetaForTest; +using DB::Cas::tests::writeMetaClean; +using DB::Cas::tests::condemnMeta; +using DB::Cas::tests::blobEntryFor; +using DB::Cas::tests::expectThrowsCode; // NOLINT(misc-unused-using-decls): only used inside `#ifndef DEBUG_OR_SANITIZER_BUILD` -- unused in a sanitizer build's TU, used in a release build's + +namespace ProfileEvents +{ +extern const Event CASBlobBodyPutAvoided; +} + +namespace DB::ErrorCodes +{ +extern const int FILE_DOESNT_EXIST; +extern const int LOGICAL_ERROR; +} + +namespace +{ + +/// Open a Pool over `b`. +PoolPtr openUploadPool(const std::shared_ptr & b) +{ + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +BlobSource reReadableStagedSource( + const BackendPtr & backend, const String & staging_key, uint64_t payload_size, uint64_t header_len) +{ + BlobSource source; + source.size = payload_size; + source.server_side_copy_from = staging_key; + source.open = [backend, staging_key, header_len, payload_size]() -> std::unique_ptr + { + auto staged = backend->getStream(staging_key); + if (!staged) + throw DB::Exception(DB::ErrorCodes::FILE_DOESNT_EXIST, "staging object {} is absent", staging_key); + String encoded_header(header_len, '\0'); + staged->stream->readStrict(encoded_header.data(), encoded_header.size()); + (void)decodeEnvelopeHeader(encoded_header, header_len + payload_size, ObjectKind::Blob); + return std::move(staged->stream); + }; + return source; +} + +/// Stage a one-blob manifest for `payload` and durably precommit it before materialization. +PartWriteTxnPtr precommitBuildFor( + const PoolPtr & s, const RootNamespace & ns, const String & ref, const String & payload) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + PartWriteTxnPtr build = s->beginPartWrite(std::move(info)); + const ManifestId id = build->stageManifest({blobEntryFor("col.bin", u128Of(payload), payload.size())}); + build->precommitAdd(ns, ref, id); + return build; +} + +/// Seed a present, well-formed blob body whose LOGICAL bytes are exactly `payload` (a fixed envelope +/// header followed by the payload), so a later HEAD returns a token and a logical size of `payload.size()`. +void seedPresentBody( + InMemoryBackend & b, const Layout & layout, const PoolMeta & pm, const BlobRef & ref, const String & payload) +{ + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xABCD); + h.build_id = DB::UInt128(0x1111); + const String head = encodeEnvelopeHeader(h, static_cast(pm.blob_header_len)); + b.putIfAbsent(layout.blobKey(ref), head + payload); +} + +/// The logical payload stored at `key` (object body minus the fixed blob header), or empty when absent. +String logicalPayloadAt(InMemoryBackend & b, const String & key, uint64_t header_len) +{ + const auto got = b.get(key); + if (!got || got->bytes.size() < header_len) + return {}; + return got->bytes.substr(header_len); +} + +/// The blob's meta state, or nullopt when the meta object is absent. +std::optional metaStateAt(InMemoryBackend & b, const Layout & layout, const String & payload) +{ + const auto lm = loadMetaForTest(b, layout, u128Of(payload)); + return lm ? std::optional(lm->meta.state) : std::nullopt; +} + +/// Records only the watched blob lane, so pool-open and precommit traffic cannot obscure the +/// transaction-level ordering asserted below. +class ProtocolRecordingBackend final : public InMemoryBackend +{ +public: + void watch(String blob_key_, String meta_key_) + { + blob_key = std::move(blob_key_); + meta_key = std::move(meta_key_); + operations.clear(); + blob_heads = 0; + meta_gets = 0; + publish_calls = 0; + meta_gets_before_first_publish.reset(); + } + + HeadResult head(const String & key) override + { + if (key == blob_key) + { + ++blob_heads; + operations.emplace_back("head"); + } + return InMemoryBackend::head(key); + } + + std::optional get(const String & key, Range range) override + { + if (key == meta_key) + { + ++meta_gets; + operations.emplace_back("meta-get"); + } + return InMemoryBackend::get(key, range); + } + + void publishBlob(const BlobPublishRequest & request) override + { + if (request.destination_key == blob_key) + { + ++publish_calls; + operations.emplace_back("publish"); + if (!meta_gets_before_first_publish) + meta_gets_before_first_publish = meta_gets; + } + InMemoryBackend::publishBlob(request); + } + + String blob_key; + String meta_key; + std::vector operations; + size_t blob_heads = 0; + size_t meta_gets = 0; + size_t publish_calls = 0; + std::optional meta_gets_before_first_publish; +}; + +} + +TEST(CASUploadDetached, FreshMissHeadsThenPublishesWithoutPrepublicationMetaGet) +{ + const String payload = "mandatory-head-fresh-miss"; + const BlobRef ref = idOf(payload); + auto backend = std::make_shared(); + auto store = openUploadPool(backend); + auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-fresh"}, "part", payload); + const String blob_key = store->layout().blobKey(ref); + backend->watch(blob_key, store->layout().blobMetaKey(ref)); + + const BlobUploadResult result = build->uploadBlobDetached( + BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(result.dep.proof, BlobDependencyProof::Materialized); + ASSERT_FALSE(backend->operations.empty()); + EXPECT_EQ(backend->operations.front(), "head"); + EXPECT_EQ(backend->blob_heads, 1u); + EXPECT_EQ(backend->publish_calls, 1u); + ASSERT_TRUE(backend->meta_gets_before_first_publish.has_value()); + EXPECT_EQ(*backend->meta_gets_before_first_publish, 0u); +} + +TEST(CASUploadDetached, ExistingCleanHeadsAndObservesWithoutPublication) +{ + const String payload = "mandatory-head-existing-clean"; + const BlobRef ref = idOf(payload); + auto backend = std::make_shared(); + auto store = openUploadPool(backend); + seedPresentBody(*backend, store->layout(), store->poolMeta(), ref, payload); + writeMetaClean(*backend, store->layout(), u128Of(payload), payload.size()); + auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-clean"}, "part", payload); + backend->watch(store->layout().blobKey(ref), store->layout().blobMetaKey(ref)); + const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(); + + const BlobUploadResult result = build->uploadBlobDetached( + BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(result.dep.proof, BlobDependencyProof::Materialized); + ASSERT_FALSE(backend->operations.empty()); + EXPECT_EQ(backend->operations.front(), "head"); + EXPECT_EQ(backend->blob_heads, 1u); + EXPECT_EQ(backend->meta_gets, 1u); + EXPECT_EQ(backend->publish_calls, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(), avoided_before + 1); +} + +TEST(CASUploadDetached, ExistingBodyWithoutMetadataBackfillsWithoutPublication) +{ + const String payload = "mandatory-head-metadata-backfill"; + const BlobRef ref = idOf(payload); + auto backend = std::make_shared(); + auto store = openUploadPool(backend); + seedPresentBody(*backend, store->layout(), store->poolMeta(), ref, payload); + auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-backfill"}, "part", payload); + backend->watch(store->layout().blobKey(ref), store->layout().blobMetaKey(ref)); + + const BlobUploadResult result = build->uploadBlobDetached( + BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(result.dep.proof, BlobDependencyProof::Materialized); + ASSERT_FALSE(backend->operations.empty()); + EXPECT_EQ(backend->operations.front(), "head"); + EXPECT_EQ(backend->blob_heads, 1u); + EXPECT_EQ(backend->meta_gets, 1u); + EXPECT_EQ(backend->publish_calls, 0u); + const auto meta = loadMetaForTest(*backend, store->layout(), u128Of(payload)); + ASSERT_TRUE(meta.has_value()); + EXPECT_EQ(meta->meta.state, MetaState::Clean); + EXPECT_EQ(meta->meta.size, payload.size()); +} + +TEST(CASUploadDetached, AbsentBodyWithStaleCondemnedPublishesBeforeMetadataRead) +{ + const String payload = "mandatory-head-absent-stale-condemned"; + const BlobRef ref = idOf(payload); + auto backend = std::make_shared(); + auto store = openUploadPool(backend); + writeMetaClean(*backend, store->layout(), u128Of(payload), payload.size()); + condemnMeta(*backend, store->layout(), u128Of(payload), 17); + auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-stale"}, "part", payload); + backend->watch(store->layout().blobKey(ref), store->layout().blobMetaKey(ref)); + + const BlobUploadResult result = build->uploadBlobDetached( + BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(result.dep.proof, BlobDependencyProof::Materialized); + ASSERT_FALSE(backend->operations.empty()); + EXPECT_EQ(backend->operations.front(), "head"); + EXPECT_EQ(backend->blob_heads, 1u); + EXPECT_EQ(backend->publish_calls, 1u); + ASSERT_TRUE(backend->meta_gets_before_first_publish.has_value()); + EXPECT_EQ(*backend->meta_gets_before_first_publish, 0u); + EXPECT_GT(backend->meta_gets, 0u) << "the stale marker is read only while reconciling after publication"; + EXPECT_EQ(metaStateAt(*backend, store->layout(), payload), std::optional(MetaState::Clean)); +} + +TEST(CASUploadDetached, PresentCondemnedPublishesFreshAndQueuedOldDeleteMisses) +{ + const String payload = "mandatory-head-present-condemned"; + const BlobRef ref = idOf(payload); + auto backend = std::make_shared(); + auto store = openUploadPool(backend); + seedPresentBody(*backend, store->layout(), store->poolMeta(), ref, payload); + writeMetaClean(*backend, store->layout(), u128Of(payload), payload.size()); + condemnMeta(*backend, store->layout(), u128Of(payload), 19); + auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-condemned"}, "part", payload); + const String blob_key = store->layout().blobKey(ref); + const Token condemned_token = backend->head(blob_key).token; + backend->watch(blob_key, store->layout().blobMetaKey(ref)); + const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(); + + const BlobUploadResult result = build->uploadBlobDetached( + BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(result.dep.proof, BlobDependencyProof::Materialized); + ASSERT_FALSE(backend->operations.empty()); + EXPECT_EQ(backend->operations.front(), "head"); + EXPECT_EQ(backend->blob_heads, 1u); + EXPECT_EQ(backend->publish_calls, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(), avoided_before); + EXPECT_EQ(backend->deleteExact(blob_key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_TRUE(backend->head(blob_key).exists); +} + +/// A present body with absent metadata is observed and backfilled `Clean` without publication. +TEST(CASUploadDetached, PresentBodyWithoutMetadataBackfills) +{ + const RootNamespace ns{"srv1/nsAdopt"}; + const String ref_name = "part"; + const String payload = "head-miss-adopt-payload"; + const BlobRef blob = idOf(payload); + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build) + { + b = std::make_shared(); + s = openUploadPool(b); + seedPresentBody(*b, s->layout(), s->poolMeta(), blob, payload); /// body present, no meta: backfill + build = precommitBuildFor(s, ns, ref_name, payload); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + arrange(b1, s1, build1); + const String key = s1->layout().blobKey(blob); + + ASSERT_FALSE(metaStateAt(*b1, s1->layout(), payload).has_value()); /// precondition: meta absent + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + + const BlobUploadResult r = build1->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(r.diagnostics.action, BlobMaterializationAction::Observed); + EXPECT_EQ(r.diagnostics.reason, std::nullopt); + EXPECT_EQ(r.diagnostics.transport, std::nullopt); + EXPECT_EQ(r.dep.proof, BlobDependencyProof::Materialized); + EXPECT_EQ(r.dep.size, payload.size()); + + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + /// The point-read backfilled a Clean meta. + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + arrange(b2, s2, build2); + build2->putBlob(blob, BlobSource::fromString(payload)); + EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); + + EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), + logicalPayloadAt(*b2, key, s2->poolMeta().blob_header_len)); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), metaStateAt(*b2, s2->layout(), payload)); +} + +/// Fresh local streaming: mandatory `HEAD` observes absence, then unconditional publication creates +/// the body and reconciles `Clean` metadata. +TEST(CASUploadDetached, FreshLocalStreaming) +{ + const RootNamespace ns{"srv1/nsFresh"}; + const String ref_name = "part"; + const String payload = "fresh-local-streaming-payload"; + const BlobRef blob = idOf(payload); + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build) + { + b = std::make_shared(); + s = openUploadPool(b); + build = precommitBuildFor(s, ns, ref_name, payload); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + arrange(b1, s1, build1); + const String key = s1->layout().blobKey(blob); + + ASSERT_FALSE(b1->head(key).exists); /// precondition: absent + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + + const BlobUploadResult r = build1->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(r.diagnostics.action, BlobMaterializationAction::Published); + EXPECT_EQ(r.diagnostics.reason, BlobPublicationReason::Absent); + EXPECT_EQ(r.diagnostics.transport, BlobPublicationTransport::Streaming); + EXPECT_EQ(r.ref, blob); + EXPECT_EQ(r.dep.proof, BlobDependencyProof::Materialized); + EXPECT_EQ(r.dep.size, payload.size()); + + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + EXPECT_TRUE(b1->head(key).exists); + EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), payload); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + arrange(b2, s2, build2); + const PutBlobResult pr = build2->putBlob(blob, BlobSource::fromString(payload)); + EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); + EXPECT_EQ(pr.size, r.dep.size); + + /// The envelope's fresh incarnation tag differs per upload, but the LOGICAL payload and meta match. + EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), + logicalPayloadAt(*b2, key, s2->poolMeta().blob_header_len)); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), metaStateAt(*b2, s2->layout(), payload)); +} + +/// A first absent observation with an S3-staged source selects verbatim native copy. +TEST(CASUploadDetached, S3StagingPromotion) +{ + const RootNamespace ns{"srv1/nsStaging"}; + const String ref_name = "part"; + const String payload = "s3-staging-promotion-payload"; + const BlobRef blob = idOf(payload); + const String staging_key = "p/staging/mount1/promote.tmp"; + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build, String & staging_bytes) + { + b = std::make_shared(); + s = openUploadPool(b); + /// The staging object holds [header][payload], exactly as the S3-staging writer emits it. + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; + b->putIfAbsent(staging_key, staging_bytes); + build = precommitBuildFor(s, ns, ref_name, payload); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + String staging_bytes1; + arrange(b1, s1, build1, staging_bytes1); + const String key = s1->layout().blobKey(blob); + + ASSERT_FALSE(b1->head(key).exists); + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + + const BlobUploadResult r = build1->uploadBlobDetached( + BlobUploadRequest{ + blob, + reReadableStagedSource(b1, staging_key, payload.size(), s1->poolMeta().blob_header_len), + payload.size()}); + + EXPECT_EQ(r.diagnostics.action, BlobMaterializationAction::Published); + EXPECT_EQ(r.diagnostics.reason, BlobPublicationReason::Absent); + EXPECT_EQ(r.diagnostics.transport, BlobPublicationTransport::ServerSideCopy); + EXPECT_EQ(r.dep.proof, BlobDependencyProof::Materialized); + EXPECT_EQ(r.dep.size, payload.size()); + + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + ASSERT_TRUE(b1->head(key).exists); + /// The server-side copy moved the staging bytes verbatim to the blob key. + const auto got = b1->get(key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, staging_bytes1); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + String staging_bytes2; + arrange(b2, s2, build2, staging_bytes2); + build2->putBlob( + blob, + reReadableStagedSource(b2, staging_key, payload.size(), s2->poolMeta().blob_header_len)); + EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); + + const auto got2 = b2->get(key); + ASSERT_TRUE(got2.has_value()); + EXPECT_EQ(got->bytes, got2->bytes); +} + +/// Condemned-local replacement: a present body observed condemned via the metadata point-read is displaced +/// by a fresh incarnation streamed from the writer's OWN source, never a read of the dying object. +/// Diagnostics are `Published` + `Condemned` + `Streaming`; the token changes and metadata returns to `Clean`. +TEST(CASUploadDetached, CondemnedLocalResurrection) +{ + const RootNamespace ns{"srv1/nsResLocal"}; + const String ref_name = "part"; + const String payload = "condemned-local-republish-payload"; + const BlobRef blob = idOf(payload); + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build) + { + b = std::make_shared(); + s = openUploadPool(b); + seedPresentBody(*b, s->layout(), s->poolMeta(), blob, payload); + writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); + condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/7); + build = precommitBuildFor(s, ns, ref_name, payload); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + arrange(b1, s1, build1); + const String key = s1->layout().blobKey(blob); + const Token condemned_token = b1->head(key).token; + + ASSERT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Condemned)); + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + + const BlobUploadResult r = build1->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()}); + + EXPECT_EQ(r.diagnostics.action, BlobMaterializationAction::Published); + EXPECT_EQ(r.diagnostics.reason, BlobPublicationReason::Condemned); + EXPECT_EQ(r.diagnostics.transport, BlobPublicationTransport::Streaming); + EXPECT_EQ(r.dep.proof, BlobDependencyProof::Materialized); + EXPECT_EQ(r.dep.size, payload.size()); + + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + /// The condemned incarnation was displaced by a fresh one (token changed) and the meta is Clean again. + const Token after_token = b1->head(key).token; + EXPECT_NE(after_token.value, condemned_token.value); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); + EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), payload); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + arrange(b2, s2, build2); + build2->putBlob(blob, BlobSource::fromString(payload)); + EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); + + EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), + logicalPayloadAt(*b2, key, s2->poolMeta().blob_header_len)); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), metaStateAt(*b2, s2->layout(), payload)); +} + +/// Condemned-S3 replacement: a present body observed condemned with an S3 staging source is displaced +/// by an unconditional retagged stream from that writer-owned staging payload, never a read/copy of +/// the condemned blob key. Diagnostics are `Published` + `Condemned` + `Streaming`. +TEST(CASUploadDetached, CondemnedS3Resurrection) +{ + const RootNamespace ns{"srv1/nsResS3"}; + const String ref_name = "part"; + const String payload = "condemned-s3-republish-payload"; + const BlobRef blob = idOf(payload); + const String staging_key = "p/staging/mount1/republish.tmp"; + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build) + { + b = std::make_shared(); + s = openUploadPool(b); + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; + b->putIfAbsent(staging_key, staging_bytes); + /// Seed the condemned blob body = exactly a verbatim promote of the staging object would produce. + b->putIfAbsent(s->layout().blobKey(blob), staging_bytes); + writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); + condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/9); + build = precommitBuildFor(s, ns, ref_name, payload); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + arrange(b1, s1, build1); + const String key = s1->layout().blobKey(blob); + const Token condemned_token = b1->head(key).token; + + ASSERT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Condemned)); + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + + const BlobUploadResult r = build1->uploadBlobDetached( + BlobUploadRequest{ + blob, + reReadableStagedSource(b1, staging_key, payload.size(), s1->poolMeta().blob_header_len), + payload.size()}); + + EXPECT_EQ(r.diagnostics.action, BlobMaterializationAction::Published); + EXPECT_EQ(r.diagnostics.reason, BlobPublicationReason::Condemned); + EXPECT_EQ(r.diagnostics.transport, BlobPublicationTransport::Streaming); + EXPECT_EQ(r.dep.proof, BlobDependencyProof::Materialized); + EXPECT_EQ(r.dep.size, payload.size()); + + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + /// A fresh incarnation displaced the condemned one (INV-NO-RETURN: fresh tag ⇒ different token). + const Token after_token = b1->head(key).token; + EXPECT_NE(after_token.value, condemned_token.value); + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + arrange(b2, s2, build2); + build2->putBlob( + blob, + reReadableStagedSource(b2, staging_key, payload.size(), s2->poolMeta().blob_header_len)); + EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); + + EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), metaStateAt(*b2, s2->layout(), payload)); +} + +/// `mergeBlobUploadResults` folds N detached results in ONE call to EXACTLY the same deps a serial +/// putBlob fold would produce. Both worlds run the identical sequence of backend calls (same +/// precommit and same blobs in the same order). The merge path adds no backend calls of its own, only +/// in-memory bookkeeping, so a deep dependency-map comparison is exact. +TEST(CASUploadDetached, MergeAppliesAllDeps) +{ + const RootNamespace ns{"srv1/nsMergeAll"}; + const String ref_name = "part"; + const std::vector payloads = {"merge-fresh-a", "merge-fresh-b", "merge-fresh-c"}; + + auto arrange = [&](std::shared_ptr & b, PoolPtr & s, PartWriteTxnPtr & build) + { + b = std::make_shared(); + s = openUploadPool(b); + build = precommitBuildFor(s, ns, ref_name, "manifest-seed"); + }; + + std::shared_ptr b1; + PoolPtr s1; + PartWriteTxnPtr build1; + arrange(b1, s1, build1); + + std::vector results; + for (const auto & payload : payloads) + { + const BlobRef blob = idOf(payload); + EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); + results.push_back(build1->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()})); + } + /// Still untouched before the merge -- uploadBlobDetached folds nothing. + for (const auto & payload : payloads) + EXPECT_EQ(build1->dependencyProof(idOf(payload)), std::nullopt); + + build1->mergeBlobUploadResults(results); + + for (const auto & payload : payloads) + EXPECT_EQ(build1->dependencyProof(idOf(payload)), BlobDependencyProof::Materialized); + + std::shared_ptr b2; + PoolPtr s2; + PartWriteTxnPtr build2; + arrange(b2, s2, build2); + for (const auto & payload : payloads) + build2->putBlob(idOf(payload), BlobSource::fromString(payload)); + + EXPECT_EQ(build1->depsSnapshotForTest(), build2->depsSnapshotForTest()); +} + +/// Merge exception safety (spec Test 16): a hook injected between per-result applications throws +/// after the FIRST result would have applied; the SECOND result must never reach `deps`, and neither +/// may a PRE-EXISTING unrelated dep be disturbed -- a DEEP snapshot (the whole map, not one ref probed +/// at a time) proves the build is byte-for-byte at its pre-merge state, all-or-nothing observed. +TEST(CASUploadDetached, MergeFailureLeavesBuildUntouched) +{ + const RootNamespace ns{"srv1/nsMergeFail"}; + const String ref_name = "part"; + const String payload_existing = "merge-fail-existing"; + const String payload_a = "merge-fail-a"; + const String payload_b = "merge-fail-b"; + + auto b = std::make_shared(); + auto s = openUploadPool(b); + auto build = precommitBuildFor(s, ns, ref_name, "manifest-seed"); + + /// A pre-existing folded dep the merge must leave completely alone. + build->putBlob(idOf(payload_existing), BlobSource::fromString(payload_existing)); + ASSERT_EQ(build->dependencyProof(idOf(payload_existing)), BlobDependencyProof::Materialized); + + std::vector results; + results.push_back(build->uploadBlobDetached( + BlobUploadRequest{idOf(payload_a), BlobSource::fromString(payload_a), payload_a.size()})); + results.push_back(build->uploadBlobDetached( + BlobUploadRequest{idOf(payload_b), BlobSource::fromString(payload_b), payload_b.size()})); + + const auto pre_merge_snapshot = build->depsSnapshotForTest(); + ASSERT_EQ(pre_merge_snapshot.size(), 1u); /// only the pre-existing dep; the detached uploads folded nothing + + build->setMergeHookForTest([](size_t applied_so_far) + { + if (applied_so_far == 1) + throw std::bad_alloc(); + }); + + EXPECT_THROW(build->mergeBlobUploadResults(results), std::bad_alloc); + + EXPECT_EQ(build->depsSnapshotForTest(), pre_merge_snapshot); + EXPECT_EQ(build->dependencyProof(idOf(payload_a)), std::nullopt); + EXPECT_EQ(build->dependencyProof(idOf(payload_b)), std::nullopt); +} + +/// Duplicate-grouping consistency: two results for the SAME ref with conflicting sizes are rejected +/// as a staging bug (LOGICAL_ERROR) BEFORE any result applies -- the fan-out's one-task-per-unique-ref +/// invariant means this should never happen upstream, so merge itself is the backstop. LOGICAL_ERROR +/// aborts the whole process in debug/sanitizer builds instead of behaving like a catchable exception +/// (`Common/Exception.cpp`'s `handle_error_code`) -- `CASUploadDetachedDeathTest` below proves the +/// abort positively in those builds instead (it cannot also verify the build-untouched postcondition, +/// since there is no continuation after a real abort). +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASUploadDetached, MergeValidatesSizes) +{ + const RootNamespace ns{"srv1/nsMergeSizes"}; + const String ref_name = "part"; + const String payload = "merge-size-conflict"; + const BlobRef blob = idOf(payload); + + auto b = std::make_shared(); + auto s = openUploadPool(b); + auto build = precommitBuildFor(s, ns, ref_name, "manifest-seed"); + + const BlobUploadResult r = build->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()}); + ASSERT_EQ(build->dependencyProof(blob), std::nullopt); + + BlobUploadResult conflicting = r; + conflicting.dep.size = r.dep.size + 1; /// same ref, conflicting declared size + + const auto pre_merge_snapshot = build->depsSnapshotForTest(); + + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + build->mergeBlobUploadResults(std::vector{r, conflicting}); + }); + + EXPECT_EQ(build->depsSnapshotForTest(), pre_merge_snapshot); + EXPECT_EQ(build->dependencyProof(blob), std::nullopt); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASUploadDetachedDeathTest, MergeValidatesSizesAborts) +{ + const RootNamespace ns{"srv1/nsMergeSizes"}; + const String ref_name = "part"; + const String payload = "merge-size-conflict"; + const BlobRef blob = idOf(payload); + + auto b = std::make_shared(); + auto s = openUploadPool(b); + auto build = precommitBuildFor(s, ns, ref_name, "manifest-seed"); + + const BlobUploadResult r = build->uploadBlobDetached( + BlobUploadRequest{blob, BlobSource::fromString(payload), payload.size()}); + + BlobUploadResult conflicting = r; + conflicting.dep.size = r.dep.size + 1; /// same ref, conflicting declared size + + EXPECT_DEATH( + { build->mergeBlobUploadResults(std::vector{r, conflicting}); }, ""); +} +#endif diff --git a/src/Disks/tests/gtest_cas_upload_fanout.cpp b/src/Disks/tests/gtest_cas_upload_fanout.cpp new file mode 100644 index 000000000000..1371d8a70fc7 --- /dev/null +++ b/src/Disks/tests/gtest_cas_upload_fanout.cpp @@ -0,0 +1,984 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB::Cas; +using DB::Cas::tests::idOf; +using DB::Cas::tests::u128Of; +using DB::Cas::tests::blobEntryFor; +using DB::Cas::tests::writeMetaClean; +using DB::Cas::tests::condemnMeta; +using DB::Cas::tests::loadMetaForTest; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::runRoundsUntilAbsent; +using DB::Cas::tests::blobAbsent; +using DB::Cas::tests::CountingBackend; + +namespace DB::ErrorCodes +{ +extern const int LOGICAL_ERROR; +extern const int INCORRECT_DATA; +extern const int NOT_IMPLEMENTED; +} + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} + +namespace +{ + +/// A local upload pool of a chosen size. Task 5 takes the pool as a parameter (rather than reaching +/// for the server-wide `Cas::blobUploadPool()`) precisely so a test can run the SAME fan-out through a +/// size-1 pool (the serial reference) and a size-N pool (the fanned-out world) in ONE process -- the +/// server-wide pool is once-only per binary and cannot be re-sized. The calling thread only submits +/// and joins (it never occupies a pool slot), so size 1 is a valid fully-serial configuration. +std::unique_ptr makePool(size_t size) +{ + return std::make_unique( + CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, CurrentMetrics::LocalThreadScheduled, size); +} + +/// Open a Pool over any InMemoryBackend-derived backend (the plain one, or the CountingBackend that +/// records per-key GET counts). +PoolPtr openPool(const std::shared_ptr & b) +{ + return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// Stage a one-blob seed manifest and precommit it, so every adopt branch of `uploadBlobDetached` +/// passes its EDGE-BEFORE-OBSERVE fail-closed gate (which only checks the `precommitted` flag). One +/// precommit covers an arbitrary number of subsequently-uploaded blobs, mirroring +/// `precommitBuildFor`/`MergeAppliesAllDeps` in the detached suite. +PartWriteTxnPtr precommitBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_ref = ns.string() + "/" + ref; + PartWriteTxnPtr build = s->beginPartWrite(std::move(info)); + const String seed = "seed-manifest-" + ns.string() + "/" + ref; + const ManifestId id = build->stageManifest({blobEntryFor("col.bin", u128Of(seed), seed.size())}); + build->precommitAdd(ns, ref, id); + return build; +} + +/// Seed a present, well-formed blob body whose LOGICAL bytes are exactly `payload`. +void seedPresentBody(InMemoryBackend & b, const Layout & layout, const PoolMeta & pm, const String & payload) +{ + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xABCD); + h.build_id = DB::UInt128(0x1111); + const String head = encodeEnvelopeHeader(h, static_cast(pm.blob_header_len)); + b.putIfAbsent(layout.blobKey(idOf(payload)), head + payload); +} + +/// The logical payload stored at a blob key (object body minus the fixed blob header), or empty when absent. +String logicalPayloadAt(InMemoryBackend & b, const String & key, uint64_t header_len) +{ + const auto got = b.get(key); + if (!got || got->bytes.size() < header_len) + return {}; + return got->bytes.substr(header_len); +} + +/// The blob's meta state, or nullopt when the meta object is absent. +std::optional metaStateAt(InMemoryBackend & b, const Layout & layout, const String & payload) +{ + const auto lm = loadMetaForTest(b, layout, u128Of(payload)); + return lm ? std::optional(lm->meta.state) : std::nullopt; +} + +/// A local streaming source for `payload`, exactly as `ContentAddressedTransaction::uploadPendingBlobs` +/// builds for a Local-staging pending blob. +BlobUploadRequest localRequest(const String & payload) +{ + return BlobUploadRequest{idOf(payload), BlobSource::fromString(payload), payload.size()}; +} + +/// An S3-staging source: the bytes already live at `staging_key` and the upload is a server-side copy. +BlobUploadRequest s3Request(const String & payload, const String & staging_key) +{ + BlobSource src; + src.size = payload.size(); + src.server_side_copy_from = staging_key; + src.open = [payload]() -> std::unique_ptr + { + return std::make_unique(payload); + }; + return BlobUploadRequest{idOf(payload), std::move(src), payload.size()}; +} + +/// The stable dependency state is independent of backend incarnation tokens: all successful upload +/// branches establish `Materialized`, regardless of serial or parallel token-mint ordering. +using StableDep = std::tuple; +std::map stableDeps(const PartWriteTxn & build) +{ + std::map out; + for (const auto & [ref, dep] : build.depsSnapshotForTest()) + out.emplace(ref, StableDep{dep.kind, dep.size, dep.proof}); + return out; +} + +/// The backend end state for a set of blob refs: (logical payload, meta state) per ref. Deterministic +/// (content is the payload; meta settles to Clean), so it is compared byte-for-byte across worlds. +using BackendState = std::map>>; +BackendState backendState(InMemoryBackend & b, const PoolPtr & s, const std::vector & payloads) +{ + BackendState out; + for (const auto & p : payloads) + out.emplace(idOf(p), + std::make_pair(logicalPayloadAt(b, s->layout().blobKey(idOf(p)), s->poolMeta().blob_header_len), + metaStateAt(b, s->layout(), p))); + return out; +} + +/// A one-shot event with a BOUNDED wait. Not a sleep-sequencer: the wait blocks only until the event +/// fires; the bound exists solely so a design regression surfaces as a fast test failure instead of an +/// infinite hang. +struct BoundedEvent +{ + std::mutex m; + std::condition_variable cv; + bool fired = false; + void fire() + { + { + std::lock_guard l(m); + fired = true; + } + cv.notify_all(); + } + bool wait(std::chrono::milliseconds bound) + { + std::unique_lock l(m); + return cv.wait_for(l, bound, [&] { return fired; }); + } +}; + +/// Records the peak number of tasks simultaneously "inside" the rendezvous. A task calls `enter(want)` +/// from the fan-out's in-task seam; it blocks (BOUNDED) until `want` tasks are inside together, OR every +/// dispatched task has entered (so a final straggler is never stranded when the pool cannot form another +/// pair), OR the bound elapses. A pool that CANNOT muster `want` concurrent tasks (size 1, where THIS +/// task occupies the single worker) times out on the first waiter, marks the run serial, and every later +/// task skips the wait -- so a too-small pool fails FAST and the whole run stays bounded, never +/// deadlocked. `total` (the dispatched task count) is set before dispatch. +struct ConcurrencyProbe +{ + std::mutex m; + std::condition_variable cv; + int current = 0; + int peak = 0; + int entered = 0; + int total = 0; + bool timed_out = false; + void enter(int want, std::chrono::milliseconds bound) + { + std::unique_lock l(m); + ++current; + ++entered; + peak = std::max(peak, current); + cv.notify_all(); + const bool ok = cv.wait_for(l, bound, + [&] { return current >= want || entered == total || timed_out; }); + if (!ok) + timed_out = true; /// the pool cannot reach `want`; later tasks skip the wait + cv.notify_all(); + --current; + } +}; + +/// A deterministic native-copy rejection used to prove that the logical source's publication state +/// survives every request copy made by the fan-out. The first call must propagate; a later request +/// copied from the same source may only stream a newly tagged envelope, never retry verbatim copy. +class RejectFirstStagedCopyBackend final : public InMemoryBackend +{ +public: + void publishBlob(const BlobPublishRequest & request) override + { + if (std::holds_alternative(request.publication)) + { + ++copy_publications; + if (reject_copy) + { + reject_copy = false; + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "test rejects the first staged copy"); + } + } + else + { + ++streaming_publications; + } + InMemoryBackend::publishBlob(request); + } + + bool reject_copy = true; + size_t copy_publications = 0; + size_t streaming_publications = 0; +}; + +} + +TEST(CASUploadFanout, CopiedAndMovedRequestsSharePublicationAttemptedState) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/shared-publication-state"}; + auto build = precommitBuildFor(store, ns, "part"); + const String payload = "shared-publication-attempted-payload"; + const BlobRef ref = idOf(payload); + const String staging_key = "p/staging/mount1/shared-attempt.tmp"; + + EnvelopeHeader header; + header.kind = ObjectKind::Blob; + header.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging_bytes + = encodeEnvelopeHeader(header, static_cast(store->poolMeta().blob_header_len)) + payload; + backend->putIfAbsent(staging_key, staging_bytes); + + BlobSource source; + source.size = payload.size(); + source.server_side_copy_from = staging_key; + source.open = [payload]() -> std::unique_ptr + { + return std::make_unique(payload); + }; + + BlobUploadRequest original{ref, source, payload.size()}; + BlobUploadRequest first_copy = original; + BlobUploadRequest fanout_copy = original; + + expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, [&] + { + build->uploadBlobDetached(first_copy); + }); + + std::vector requests; + requests.emplace_back(std::move(fanout_copy)); + auto pool = makePool(1); + fanOutBlobUploads(*build, requests, *pool); + + EXPECT_EQ(backend->copy_publications, 1u) + << "only the source's first publication may attempt verbatim staged copy"; + EXPECT_EQ(backend->streaming_publications, 1u) + << "the request copied and moved through fan-out must retain the consumed first-attempt state"; + EXPECT_EQ(build->dependencyProof(ref), BlobDependencyProof::Materialized); + const auto stored = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(stored.has_value()); + EXPECT_EQ(stored->bytes.substr(store->poolMeta().blob_header_len), payload); +} + +/// Test 1 (spec §1 "serial-vs-parallel equivalence for successful runs"): a multi-blob part that +/// exercises every branch of `uploadBlobDetached` produces IDENTICAL recorded deps and IDENTICAL backend +/// end state whether the fan-out runs serially (pool size 1) or in parallel (pool size 4). It covers +/// present-clean observation, metadata backfill, fresh local publication, staging copy, and local and +/// staged condemned-body republication. +namespace +{ + +/// Arrange the six-branch world and return the payloads it uploads. Every branch +/// is seeded on a DISTINCT ref so the one-task-per-unique-ref fan-out runs six independent tasks. +struct WorldA +{ + std::shared_ptr b; + PoolPtr s; + PartWriteTxnPtr build; + std::vector requests; + std::vector payloads; +}; + +const char * const kObserved = "fanoutA-observed-clean"; +const char * const kAdopt = "fanoutA-head-miss-adopt"; +const char * const kFresh = "fanoutA-fresh-local"; +const char * const kStaging = "fanoutA-s3-staging"; +const char * const kResLocal = "fanoutA-condemned-local"; +const char * const kResS3 = "fanoutA-condemned-s3"; + +WorldA arrangeWorldA() +{ + WorldA w; + w.b = std::make_shared(); + w.s = openPool(w.b); + const RootNamespace ns{"srv1/nsFanoutA"}; + w.build = precommitBuildFor(w.s, ns, "part"); + + /// Present body with `Clean` metadata: safe observation avoids publication. + seedPresentBody(*w.b, w.s->layout(), w.s->poolMeta(), kObserved); + writeMetaClean(*w.b, w.s->layout(), u128Of(kObserved), std::string(kObserved).size()); + + /// HEAD-miss then 412-path live adopt with meta backfill: present body, NO meta, not cached. + seedPresentBody(*w.b, w.s->layout(), w.s->poolMeta(), kAdopt); + + /// fresh local streaming: nothing present. + + /// S3-native staging promotion: bytes live in a staging object, blob key absent. + { + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging = encodeEnvelopeHeader(h, static_cast(w.s->poolMeta().blob_header_len)) + kStaging; + w.b->putIfAbsent("p/staging/mount1/A-staging.tmp", staging); + } + + /// condemned-local resurrection: present body + condemned meta, local source. + seedPresentBody(*w.b, w.s->layout(), w.s->poolMeta(), kResLocal); + writeMetaClean(*w.b, w.s->layout(), u128Of(kResLocal), std::string(kResLocal).size()); + condemnMeta(*w.b, w.s->layout(), u128Of(kResLocal), /*condemn_round=*/7); + + /// condemned-S3 resurrection: present body (= a verbatim promote of the staging object) + condemned + /// meta, S3 staging source. + { + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging = encodeEnvelopeHeader(h, static_cast(w.s->poolMeta().blob_header_len)) + kResS3; + w.b->putIfAbsent("p/staging/mount1/A-republish.tmp", staging); + w.b->putIfAbsent(w.s->layout().blobKey(idOf(kResS3)), staging); + writeMetaClean(*w.b, w.s->layout(), u128Of(kResS3), std::string(kResS3).size()); + condemnMeta(*w.b, w.s->layout(), u128Of(kResS3), /*condemn_round=*/9); + } + + w.requests = { + localRequest(kObserved), + localRequest(kAdopt), + localRequest(kFresh), + s3Request(kStaging, "p/staging/mount1/A-staging.tmp"), + localRequest(kResLocal), + s3Request(kResS3, "p/staging/mount1/A-republish.tmp"), + }; + w.payloads = {kObserved, kAdopt, kFresh, kStaging, kResLocal, kResS3}; + return w; +} + +} + +TEST(CASUploadFanout, DependencyProofEquivalentAcrossFanoutBranches) +{ + /// Serial reference: pool size 1. + WorldA serial = arrangeWorldA(); + auto serial_pool = makePool(1); + fanOutBlobUploads(*serial.build, serial.requests, *serial_pool); + const auto serial_deps = stableDeps(*serial.build); + const auto serial_backend = backendState(*serial.b, serial.s, serial.payloads); + + /// Fanned-out: pool size 4, same inputs, freshly arranged world. + WorldA fanned = arrangeWorldA(); + auto fanned_pool = makePool(4); + fanOutBlobUploads(*fanned.build, fanned.requests, *fanned_pool); + const auto fanned_deps = stableDeps(*fanned.build); + const auto fanned_backend = backendState(*fanned.b, fanned.s, fanned.payloads); + + EXPECT_EQ(serial_deps.size(), 6u) << "one dep per unique ref"; + EXPECT_EQ(serial_deps, fanned_deps) << "recorded deps must match across serial and fanned runs"; + EXPECT_EQ(serial_backend, fanned_backend) << "backend end state must match across serial and fanned runs"; + + /// Every successful upload branch records materialized evidence only after the fan-out joins. + for (const auto & [ref, dep] : serial_deps) + { + EXPECT_EQ(std::get<0>(dep), ObjectKind::Blob); + EXPECT_EQ(std::get<2>(dep), BlobDependencyProof::Materialized); + } + for (const auto & p : serial.payloads) + EXPECT_EQ(metaStateAt(*fanned.b, fanned.s->layout(), p), std::optional(MetaState::Clean)); + +} + +/// Test 1, GET-observability (routed from T3 review (a)): the republication invariant is that a condemned +/// object is NEVER GET (revival is a fresh re-upload from the writer's own source). With a +/// CountingBackend, assert ZERO get/getStream against the condemned blob keys through the whole fan-out. +TEST(CASUploadFanout, CondemnedBranchesNeverGet) +{ + auto counting = std::make_shared(); + auto s = openPool(counting); + const RootNamespace ns{"srv1/nsNoGet"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String local_payload = "noget-condemned-local"; + const String s3_payload = "noget-condemned-s3"; + const String s3_staging = "p/staging/mount1/noget-republish.tmp"; + + seedPresentBody(*counting, s->layout(), s->poolMeta(), local_payload); + writeMetaClean(*counting, s->layout(), u128Of(local_payload), local_payload.size()); + condemnMeta(*counting, s->layout(), u128Of(local_payload), /*condemn_round=*/3); + + { + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + s3_payload; + counting->putIfAbsent(s3_staging, staging); + counting->putIfAbsent(s->layout().blobKey(idOf(s3_payload)), staging); + writeMetaClean(*counting, s->layout(), u128Of(s3_payload), s3_payload.size()); + condemnMeta(*counting, s->layout(), u128Of(s3_payload), /*condemn_round=*/5); + } + + counting->resetCounts(); /// count only the fan-out's own backend traffic + + std::vector reqs{localRequest(local_payload), s3Request(s3_payload, s3_staging)}; + auto pool = makePool(2); + fanOutBlobUploads(*build, reqs, *pool); + + /// INV-1: revival is a fresh re-upload from the writer's own source; the condemned BODY object is + /// never read. (The fan-out DOES read the two condemned-META objects -- the meta point-read is how it + /// LEARNS an incarnation is condemned -- so the invariant is per-body-key, not a global GET count.) + const String local_key = s->layout().blobKey(idOf(local_payload)); + const String s3_key = s->layout().blobKey(idOf(s3_payload)); + EXPECT_EQ(counting->getCount(local_key), 0u) << "INV-1: the condemned local body is never read"; + EXPECT_EQ(counting->getStreamCount(local_key), 0u) << "INV-1: the condemned local body is never streamed"; + EXPECT_EQ(counting->getCount(s3_key), 0u) << "INV-1: the condemned S3 body is never read"; + EXPECT_EQ(counting->getStreamCount(s3_key), 0u) << "INV-1: the condemned S3 body is never streamed"; + + /// Both resurrections still completed to Clean. + EXPECT_EQ(metaStateAt(*counting, s->layout(), local_payload), std::optional(MetaState::Clean)); + EXPECT_EQ(metaStateAt(*counting, s->layout(), s3_payload), std::optional(MetaState::Clean)); +} + +/// Test 2: duplicate refs (staged-hardlink copies push a duplicate PendingBlob record) collapse to ONE +/// task, and the merged build records exactly one dep for the ref. +TEST(CASUploadFanout, DuplicateRefsLaunchOneTask) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDup"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "dup-fresh-payload"; + std::vector reqs{localRequest(payload), localRequest(payload)}; /// same ref twice + + std::atomic dispatched{0}; + std::map per_ref; + std::mutex per_ref_m; + BlobUploadFanoutHooksForTest hooks; + hooks.on_dispatch = [&](const BlobRef & ref) + { + ++dispatched; + std::lock_guard l(per_ref_m); + ++per_ref[ref]; + }; + + auto pool = makePool(4); + fanOutBlobUploads(*build, reqs, *pool, &hooks); + + EXPECT_EQ(dispatched.load(), 1) << "two pending-blob records for one ref launch exactly one task"; + EXPECT_EQ(per_ref[idOf(payload)], 1); + EXPECT_EQ(build->dependencyProof(idOf(payload)), BlobDependencyProof::Materialized) + << "the one task's dep was merged"; + EXPECT_EQ(build->depsSnapshotForTest().size(), 1u) << "exactly one dep for the unique ref"; +} + +/// Test 2, conflicting-size backstop: two records for the SAME ref with different declared sizes are a +/// staging bug -- rejected with LOGICAL_ERROR before any task runs. LOGICAL_ERROR aborts under +/// debug/sanitizer builds, so the abort is proven positively there (DeathTest) and the exception + +/// build-untouched postcondition in a release build. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASUploadFanout, ConflictingDuplicateSizesRejected) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDupConflict"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "dup-conflict-payload"; + BlobUploadRequest a = localRequest(payload); + BlobUploadRequest c = localRequest(payload); + c.declared_size = a.declared_size + 1; /// same ref, conflicting declared size + c.source.size = c.declared_size; /// keep declared == source so only the group conflict trips + + const auto before = build->depsSnapshotForTest(); + auto pool = makePool(4); + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + fanOutBlobUploads(*build, std::vector{a, c}, *pool); + }); + EXPECT_EQ(build->depsSnapshotForTest(), before) << "a rejected fan-out merges nothing"; +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASUploadFanoutDeathTest, ConflictingDuplicateSizesAbort) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDupConflict"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "dup-conflict-payload"; + BlobUploadRequest a = localRequest(payload); + BlobUploadRequest c = localRequest(payload); + c.declared_size = a.declared_size + 1; + c.source.size = c.declared_size; + + auto pool = makePool(4); + EXPECT_DEATH({ fanOutBlobUploads(*build, std::vector{a, c}, *pool); }, ""); +} +#endif + +/// Test 2, declared_size == source.size fail-close (routed from T3 review (b)): a request whose grouping +/// key (declared_size) disagrees with its streaming authority (source.size) is a wiring bug -- rejected +/// with LOGICAL_ERROR before dispatch (DeathTest split for debug/sanitizer builds). +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASUploadFanout, DeclaredSizeMustMatchSourceSize) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDeclared"}; + auto build = precommitBuildFor(s, ns, "part"); + + BlobUploadRequest r = localRequest("declared-mismatch-payload"); + r.declared_size = r.source.size + 7; /// diverge the grouping key from the streaming authority + + auto pool = makePool(2); + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + fanOutBlobUploads(*build, std::vector{r}, *pool); + }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASUploadFanoutDeathTest, DeclaredSizeMismatchAborts) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDeclared"}; + auto build = precommitBuildFor(s, ns, "part"); + + BlobUploadRequest r = localRequest("declared-mismatch-payload"); + r.declared_size = r.source.size + 7; + + auto pool = makePool(2); + EXPECT_DEATH({ fanOutBlobUploads(*build, std::vector{r}, *pool); }, ""); +} +#endif + +/// The condemned-LOCAL displacement, end to end on the new unconditional streaming shape: the +/// resurrected body is [fresh_header][payload], its token differs from the condemned one, and the +/// meta flips back to Clean -- which is exactly what a later attempt reads to adopt instead of +/// re-writing. +TEST(CASUploadFanout, CondemnedLocalResurrectStreamsAndFlipsMetaClean) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsResLocalStream"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "condemned-local-streamed-payload"; + seedPresentBody(*b, s->layout(), s->poolMeta(), payload); + writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); + condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/13); + + const String blob_key = s->layout().blobKey(idOf(payload)); + const Token condemned_token = b->head(blob_key).token; + + std::vector reqs{localRequest(payload)}; + auto pool = makePool(2); + fanOutBlobUploads(*build, reqs, *pool, nullptr); + + /// A fresh incarnation displaced the condemned one; INV-NO-RETURN: the queued exact-token delete + /// of the condemned incarnation must miss the resurrection. + const HeadResult after = b->head(blob_key); + ASSERT_TRUE(after.exists); + EXPECT_NE(after.token, condemned_token); + EXPECT_EQ(b->deleteExact(blob_key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_TRUE(b->head(blob_key).exists); + + /// The payload survived verbatim under the fresh header. + const auto got = b->get(blob_key); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes.substr(s->poolMeta().blob_header_len), payload); + + /// The meta flipped back to Clean -- the signal a later attempt adopts on. + const auto lm = loadMetaForTest(*b, s->layout(), u128Of(payload)); + ASSERT_TRUE(lm.has_value()); + EXPECT_EQ(lm->meta.state, MetaState::Clean); +} + +/// `open` is the per-publication unit of re-readability. A present `Condemned` observation selects +/// exactly one unconditional stream; the mandatory `HEAD` itself never opens the source. +TEST(CASUploadFanout, CondemnedLocalPublicationOpensSourceOnce) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsResLocalOpens"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "condemned-local-open-count-payload"; + seedPresentBody(*b, s->layout(), s->poolMeta(), payload); + writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); + condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/17); + + int opens = 0; + BlobSource source; + source.size = payload.size(); + source.open = [&opens, payload]() -> std::unique_ptr + { + ++opens; + return std::make_unique(payload); + }; + + std::vector reqs{BlobUploadRequest{idOf(payload), std::move(source), payload.size()}}; + auto pool = makePool(2); + fanOutBlobUploads(*build, reqs, *pool, nullptr); + + EXPECT_EQ(opens, 1) << "the mandatory `HEAD` selects one unconditional streaming publication"; + EXPECT_EQ(build->dependencyProof(idOf(payload)), BlobDependencyProof::Materialized); +} + +/// Test 2, condemned-S3 duplicate pair resurrects content-correctly: two duplicate S3-staging records +/// for one condemned ref collapse to ONE republication task; the fresh incarnation displaces the condemned +/// one (token changes, meta returns to Clean) and the content is the staging object's payload. +TEST(CASUploadFanout, DuplicateCondemnedS3ResurrectsCorrectly) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDupResS3"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String payload = "dup-condemned-s3-payload"; + const String staging_key = "p/staging/mount1/dup-republish.tmp"; + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = DB::UInt128(0xC0FFEE); + const String staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; + b->putIfAbsent(staging_key, staging_bytes); + b->putIfAbsent(s->layout().blobKey(idOf(payload)), staging_bytes); + writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); + condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/11); + + const Token condemned_token = b->head(s->layout().blobKey(idOf(payload))).token; + + std::atomic dispatched{0}; + BlobUploadFanoutHooksForTest hooks; + hooks.on_dispatch = [&](const BlobRef &) { ++dispatched; }; + + std::vector reqs{s3Request(payload, staging_key), s3Request(payload, staging_key)}; + auto pool = makePool(4); + fanOutBlobUploads(*build, reqs, *pool, &hooks); + + EXPECT_EQ(dispatched.load(), 1) << "duplicate condemned records collapse to one republication task"; + EXPECT_EQ(build->dependencyProof(idOf(payload)), BlobDependencyProof::Materialized); + const Token after_token = b->head(s->layout().blobKey(idOf(payload))).token; + EXPECT_NE(after_token.value, condemned_token.value) << "a fresh incarnation displaced the condemned one"; + EXPECT_EQ(metaStateAt(*b, s->layout(), payload), std::optional(MetaState::Clean)); + EXPECT_EQ(logicalPayloadAt(*b, s->layout().blobKey(idOf(payload)), s->poolMeta().blob_header_len), payload); +} + +/// Test 3: one task fails (a poisoned source), one sibling succeeds. Merge-nothing means the build stays +/// at its pre-fan-out state; the abandoned precommit turns the successful sibling's uploaded body into +/// ORDINARY GC-reclaimable debris (NOT a new orphan class) -- a GC round reclaims it. +TEST(CASUploadFanout, PendingFanoutFailureCreatesNoDependency) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsMergeNothing"}; + + const String good = "merge-nothing-good-sibling"; + const String poisoned = "merge-nothing-poisoned"; + + PartWriteInfo info; + info.intended_ref = ns.string() + "/part"; + auto build = s->beginPartWrite(info); + /// The precommit names BOTH blobs (the durable manifest edge the real writer establishes before any + /// upload), so the successful sibling's body is edge-protected until the precommit is abandoned. + const ManifestId id = build->stageManifest({blobEntryFor("data.bin", u128Of(good), good.size()), + blobEntryFor("data.cmrk3", u128Of(poisoned), poisoned.size())}); + build->precommitAdd(ns, "part", id); + EXPECT_EQ(build->dependencyProof(idOf(good)), std::nullopt); + EXPECT_EQ(build->dependencyProof(idOf(poisoned)), std::nullopt); + + /// Poison the failing sibling via the in-task seam: throw a plain (non-LOGICAL, non-ABORTED) + /// exception so it is neither retried nor an abort under sanitizer builds. + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&](const BlobRef & ref) + { + if (ref == idOf(poisoned)) + throw DB::Exception(DB::ErrorCodes::INCORRECT_DATA, "poisoned upload source (test)"); + }; + + std::vector reqs{localRequest(good), localRequest(poisoned)}; + auto pool = makePool(4); + expectThrowsCode(DB::ErrorCodes::INCORRECT_DATA, [&] + { + fanOutBlobUploads(*build, reqs, *pool, &hooks); + }); + + /// Merge-nothing: the build recorded NO dep, even though the good sibling's body was uploaded. + EXPECT_EQ(build->dependencyProof(idOf(good)), std::nullopt); + EXPECT_EQ(build->dependencyProof(idOf(poisoned)), std::nullopt); + EXPECT_EQ(build->depsSnapshotForTest().size(), 0u); + + /// Abandon the precommit (the existing failure path), then GC reclaims the orphaned sibling body. + build->abandon(); + s->renewWatermarkOnce(); + Gc gc(s, DB::Cas::hexToU128("00000000000000000000000000000001")); + EXPECT_TRUE(runRoundsUntilAbsent(s, gc, *b, s->layout(), u128Of(good))) + << "the successful sibling's body is ordinary GC-reclaimable debris after abandon"; + EXPECT_TRUE(blobAbsent(*b, s->layout(), u128Of(poisoned))) << "the poisoned sibling never uploaded a body"; +} + +/// Two distinct refs publish concurrently and establish materialized dependencies. +TEST(CASUploadFanout, ConcurrentPublicationsEstablishProof) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsPublicationRace"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String pa = "publication-race-a"; + const String pb = "publication-race-b"; + + /// A latch of 2 crossed with a pool of 2 cannot hang: both tasks are guaranteed to run concurrently + /// (the calling thread never occupies a pool slot), so both reach the latch and release together. + std::latch both_in{2}; + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&](const BlobRef &) { both_in.arrive_and_wait(); }; + + std::vector reqs{localRequest(pa), localRequest(pb)}; + auto pool = makePool(2); + fanOutBlobUploads(*build, reqs, *pool, &hooks); + + EXPECT_EQ(build->dependencyProof(idOf(pa)), BlobDependencyProof::Materialized); + EXPECT_EQ(build->dependencyProof(idOf(pb)), BlobDependencyProof::Materialized); +} + +/// Test 5: pool saturation is bounded. Eight blobs run through a pool of 2 (peak concurrency 2 is +/// observed) and a pool of 1 (peak concurrency 1 -- the single worker cannot self-wait for a second +/// concurrent task, so it FAILS FAST via the bounded wait). Both configurations complete every upload: +/// pool size 1 correctly degenerates to serial without deadlock. +TEST(CASUploadFanout, PoolSaturationBounded) +{ + constexpr int kBlobs = 8; + /// Bounds are DECOUPLED by pool size. Pool 2 will reach the 2-task rendezvous in microseconds under + /// any realistic load, so its bound is generous (10s) purely as a hang guard -- it is essentially + /// never waited out (and `entered == total` releases any final straggler). Pool 1 CANNOT form a pair + /// (its single worker is occupied by the waiting task while the caller thread only joins), so its + /// first waiter must time out; 500ms is far above the microseconds a real pair needs, yet keeps the + /// serial run fast. + auto runEight = [](size_t pool_size, ConcurrencyProbe & probe, std::chrono::milliseconds bound) + { + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsSaturate"}; + auto build = precommitBuildFor(s, ns, "part"); + + std::vector reqs; + std::vector payloads; + for (int i = 0; i < kBlobs; ++i) + { + payloads.push_back("saturate-payload-" + std::to_string(i)); + reqs.push_back(localRequest(payloads.back())); + } + probe.total = kBlobs; + + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&, bound](const BlobRef &) { probe.enter(2, bound); }; + + auto pool = makePool(pool_size); + fanOutBlobUploads(*build, reqs, *pool, &hooks); + + for (const auto & p : payloads) + EXPECT_EQ(build->dependencyProof(idOf(p)), BlobDependencyProof::Materialized) + << "every blob uploaded (pool_size=" << pool_size << ")"; + }; + + ConcurrencyProbe probe2; + runEight(2, probe2, std::chrono::seconds(10)); + EXPECT_EQ(probe2.peak, 2) << "pool of 2 runs two blob uploads concurrently"; + EXPECT_FALSE(probe2.timed_out) << "pool of 2 forms a pair without hitting the bound"; + + ConcurrencyProbe probe1; + runEight(1, probe1, std::chrono::milliseconds(500)); + EXPECT_EQ(probe1.peak, 1) << "pool of 1 degenerates to serial (never occupies the caller thread's slot)"; + EXPECT_TRUE(probe1.timed_out) << "the single worker fails fast on the bounded wait instead of hanging"; +} + +/// Test 6a: even when one task fails immediately, the join drains EVERY task before the failure surfaces. +/// A failing task counts down an event and throws; a sibling waits for that event, then uploads. The +/// fan-out rethrows only after the join, so the sibling's body is present in the backend by the time the +/// caller observes the failure. +TEST(CASUploadFanout, DrainPrecedesUnwind) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDrain"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String failing = "drain-failing"; + const String slow = "drain-slow-sibling"; + + BoundedEvent failing_threw; + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&](const BlobRef & ref) + { + if (ref == idOf(failing)) + { + failing_threw.fire(); + throw DB::Exception(DB::ErrorCodes::INCORRECT_DATA, "drain-test failing task"); + } + else + { + /// 5-second bound: the failing task fires the event in microseconds; the bound only guards + /// against a hang if the failing task never runs (a design regression). + (void)failing_threw.wait(std::chrono::seconds(5)); + } + }; + + std::vector reqs{localRequest(failing), localRequest(slow)}; + auto pool = makePool(2); + expectThrowsCode(DB::ErrorCodes::INCORRECT_DATA, [&] + { + fanOutBlobUploads(*build, reqs, *pool, &hooks); + }); + + EXPECT_TRUE(b->head(s->layout().blobKey(idOf(slow))).exists) + << "the sibling's upload was drained by the join before the failure surfaced"; + EXPECT_EQ(build->dependencyProof(idOf(slow)), std::nullopt) + << "merge-nothing: the drained sibling's dep is not merged"; +} + +/// Test 6b: a throw injected DURING the dispatch loop (before all tasks are enqueued) still drains the +/// tasks already scheduled -- the fan-out drains every already-scheduled task on the unwinding path +/// before the captured storage is destroyed (the B90 lesson). The first task's body is present after the +/// dispatch throw is caught. +/// +/// The throw is GATED on the first task actually entering its body: the runner marks tasks that are +/// still SCHEDULED as CANCELLED during unwind and skips waiting for a cancelled task, so a throw fired +/// before the first task's body ran could cancel it and leave its body ABSENT -- a real flakiness the +/// gate removes. Once the first task's `in_task` hook has fired, that task is past SCHEDULED (RUNNING), +/// so it can no longer be cancelled and the drain deterministically waits for its upload. +TEST(CASUploadFanout, DispatchThrowStillDrains) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsDispatchThrow"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String first = "dispatch-throw-first"; + const String second = "dispatch-throw-second"; + /// Dispatch runs in ascending-ref order, so the SMALLER-ref payload is the one enqueued before the + /// second dispatch throws. + const String enqueued = (idOf(first) < idOf(second)) ? first : second; + + BoundedEvent first_task_running; + std::atomic dispatch_calls{0}; + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&](const BlobRef & ref) + { + if (ref == idOf(enqueued)) + first_task_running.fire(); + }; + hooks.on_dispatch = [&](const BlobRef &) + { + if (++dispatch_calls == 2) + { + /// Wait until the first task's body has entered before throwing, so it is RUNNING (not + /// SCHEDULED) and the unwind cannot cancel it. Pool size 2 guarantees the first task gets a + /// worker while this (dispatch) thread waits; 10s is a pure hang guard, never a sequencer. + EXPECT_TRUE(first_task_running.wait(std::chrono::seconds(10))) + << "the first dispatched task must reach its body before the dispatch throw"; + throw DB::Exception(DB::ErrorCodes::INCORRECT_DATA, "dispatch-loop throw (test)"); + } + }; + + std::vector reqs{localRequest(first), localRequest(second)}; + auto pool = makePool(2); + expectThrowsCode(DB::ErrorCodes::INCORRECT_DATA, [&] + { + fanOutBlobUploads(*build, reqs, *pool, &hooks); + }); + + EXPECT_EQ(dispatch_calls.load(), 2) << "the throw fired on the second dispatch"; + /// The already-RUNNING first task was drained before the stack unwound, so its body is present + /// although nothing was merged. + EXPECT_TRUE(b->head(s->layout().blobKey(idOf(enqueued))).exists) + << "the already-dispatched task was drained before the stack unwound"; + EXPECT_EQ(build->depsSnapshotForTest().size(), 0u) << "merge-nothing on a dispatch throw"; +} + + + + +/// Test 6c (codex stage-1 review, Critical): a throw at the TRACKING-PUBLICATION seam still drains every +/// already-scheduled task before the captured `results` storage is destroyed. In the broken form a task +/// could be scheduled-but-untracked at the throw and run later against freed `results` (a +/// heap-use-after-free); the fix schedules-and-tracks in ONE no-throw step (pre-reserved handle vector) +/// and joins via a scope-exit drain guard, so a seam throw finds every scheduled task already tracked and +/// drains it. The throw is gated on the first task RUNNING (same reason as `DispatchThrowStillDrains`) so +/// its drain is deterministic; under ASan this run is UAF-clean -- the regression signature of a +/// scheduled-but-untracked task is a heap-use-after-free on `results` here. +TEST(CASUploadFanout, TrackingSeamThrowStillDrains) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const RootNamespace ns{"srv1/nsTrackSeam"}; + auto build = precommitBuildFor(s, ns, "part"); + + const String first = "track-seam-first"; + const String second = "track-seam-second"; + /// Dispatch is ascending-ref, so the smaller ref is enqueued first. + const String smaller = (idOf(first) < idOf(second)) ? first : second; + + BoundedEvent first_task_running; + std::atomic enqueue_calls{0}; + BlobUploadFanoutHooksForTest hooks; + hooks.in_task = [&](const BlobRef & ref) + { + if (ref == idOf(smaller)) + first_task_running.fire(); + }; + hooks.after_enqueue = [&](const BlobRef &) + { + /// Throw at the tracking seam of the SECOND enqueue -- by then both tasks are scheduled and (in + /// the fixed code) tracked, so the drain guard must join both. Gate on the first task RUNNING so + /// the drain cannot race a still-SCHEDULED cancellation; 10s is a pure hang guard. + if (++enqueue_calls == 2) + { + EXPECT_TRUE(first_task_running.wait(std::chrono::seconds(10))) + << "the first task must be RUNNING before the tracking-seam throw"; + throw DB::Exception(DB::ErrorCodes::INCORRECT_DATA, "tracking-seam throw (test)"); + } + }; + + std::vector reqs{localRequest(first), localRequest(second)}; + auto pool = makePool(2); + expectThrowsCode(DB::ErrorCodes::INCORRECT_DATA, [&] + { + fanOutBlobUploads(*build, reqs, *pool, &hooks); + }); + + /// The already-scheduled first task was drained before `results` was destroyed, so its body is + /// present; nothing was merged (merge-nothing on any fan-out throw). + EXPECT_TRUE(b->head(s->layout().blobKey(idOf(smaller))).exists) + << "an already-scheduled task was not drained before the stack unwound"; + EXPECT_EQ(build->depsSnapshotForTest().size(), 0u) << "merge-nothing on a tracking-seam throw"; +} diff --git a/src/Disks/tests/gtest_cas_wire_vocab.cpp b/src/Disks/tests/gtest_cas_wire_vocab.cpp new file mode 100644 index 000000000000..efbe3da7a0ae --- /dev/null +++ b/src/Disks/tests/gtest_cas_wire_vocab.cpp @@ -0,0 +1,53 @@ +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; } + +TEST(CASWireVocab, EnumWordsRoundTrip) +{ + for (TokenType t : {TokenType::ETag, TokenType::Generation, TokenType::Emulated}) + EXPECT_EQ(tokenTypeFromWord(tokenTypeToWord(t), "t"), t); + for (BlobHashAlgo a : {BlobHashAlgo::CityHash128, BlobHashAlgo::XXH3_128, BlobHashAlgo::Sha256}) + EXPECT_EQ(blobHashAlgoFromWord(blobHashAlgoName(a), "a"), a); + EXPECT_EQ(objectKindFromWord(objectKindToWord(ObjectKind::Blob), "k"), ObjectKind::Blob); + EXPECT_THROW(tokenTypeFromWord("nope", "t"), DB::Exception); + EXPECT_THROW(blobHashAlgoFromWord("nope", "a"), DB::Exception); +} + +TEST(CASWireVocab, SiblingFieldsWriteAndReadBack) +{ + CasJsonWriter out; + bool first = true; + writeTokenFields(out, first, Token{"etag-abc\"x", TokenType::ETag}); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("00112233445566778899aabbccddeeff"))}; + writeBlobRefFields(out, first, ref); + closeObject(out, first); + const String rendered = std::move(out).take(); + EXPECT_EQ(rendered, + R"({"tt":"etag","tv":"etag-abc\"x","ha":"ch128","h":"00112233445566778899aabbccddeeff"})"); + + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + String key; + String tv; + String ha; + String h; + TokenType tt{}; + while (r.nextKey(key)) + { + if (key == "tt") tt = tokenTypeFromWord(r.readString(), "t"); + else if (key == "tv") tv = r.readString(); + else if (key == "ha") ha = r.readString(); + else if (key == "h") h = r.readString(); + else r.skipUnknown(key); + } + EXPECT_EQ(tt, TokenType::ETag); + EXPECT_EQ(tv, "etag-abc\"x"); + const BlobRef back{blobHashAlgoFromWord(ha, "a"), codecFor(blobHashAlgoFromWord(ha, "a")).fromHex(h)}; + EXPECT_EQ(back, ref); +} diff --git a/src/Disks/tests/gtest_cas_writer_duties.cpp b/src/Disks/tests/gtest_cas_writer_duties.cpp new file mode 100644 index 000000000000..662c0daa165e --- /dev/null +++ b/src/Disks/tests/gtest_cas_writer_duties.cpp @@ -0,0 +1,545 @@ +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int NETWORK_ERROR; +} + +using namespace DB::Cas; + +namespace +{ + +PoolConfig singleAttemptConfig() +{ + PoolConfig config{ + .pool_prefix = "p", + .server_root_id = "test", + .background_watermark = false, + }; + config.cas_request_budget.max_attempts = 1; + config.cas_request_budget.attempt_timeout_ms = 100; + config.cas_request_budget.operation_deadline_ms = 5000; + config.cas_request_budget.lease_safety_margin_ms = 100; + return config; +} + +PoolPtr openSingleAttemptPool(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + return Pool::open(backend, singleAttemptConfig()); +} + +PoolPtr openFrozenSingleAttemptPool(const BackendPtr & backend) +{ + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig config = singleAttemptConfig(); + config.boot_ms_fn = [] { return uint64_t{0}; }; + config.mount_renew_period = std::chrono::hours{1}; + return Pool::open(backend, config); +} + +PartWriteTxnPtr stageEmptyManifest( + const PoolPtr & store, const RootNamespace & ns, const String & ref_name, ManifestId & id) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref_name; + auto build = store->beginPartWrite(std::move(info)); + id = build->stageManifest({}); + return build; +} + +void publishEmptyRef(const PoolPtr & store, const RootNamespace & ns, const String & ref_name) +{ + ManifestId id; + auto build = stageEmptyManifest(store, ns, ref_name, id); + build->precommitAdd(ns, ref_name, id); + build->promote(ns, ref_name, build->buildId(), id); +} + +uint64_t leaveRejectedCleanupDuty(const PoolPtr & store, const RootNamespace & ns) +{ + ManifestId rejected_id; + auto rejected = stageEmptyManifest(store, ns, "rejected", rejected_id); + const uint64_t rejected_seq = rejected->buildSeq(); + + store->setMountDeadline(100); + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + EXPECT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + + rejected.reset(); + EXPECT_EQ(store->minActive(), rejected_seq); + store->setMountDeadline(30000); + return rejected_seq; +} + +} + +/// Removing the deferred-cleanup transfer from `~PartWriteTxn` makes this test fail at the first +/// `minActive` assertion: the old unconditional destructor retirement advances the build floor while +/// the owner-grant outcome is still unknown. The later assertions pin the other half of the duty: the +/// next mutation resolves the durable wedge, removes the exact old precommit, and only then retires it. +TEST(CASWriterDuties, UncertainAdoptedGrantStaysActiveUntilTheNextMutationRemovesIt) +{ + auto backend = std::make_shared(); + auto store = openSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_adopt"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + ManifestId abandoned_id; + auto abandoned = stageEmptyManifest(store, ns, "abandoned", abandoned_id); + const uint64_t abandoned_seq = abandoned->buildSeq(); + const String abandoned_manifest_key = store->layout().manifestKey(abandoned_id); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { abandoned->precommitAdd(ns, "abandoned", abandoned_id); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(abandoned->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + + abandoned.reset(); + EXPECT_EQ(store->minActive(), abandoned_seq) + << "an unresolved owner grant must keep its build active after the transaction object is gone"; + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + ManifestId successor_id; + auto successor = stageEmptyManifest(store, ns, "successor", successor_id); + const uint64_t successor_seq = successor->buildSeq(); + successor->precommitAdd(ns, "successor", successor_id); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->minActive(), successor_seq) + << "the abandoned build retires only after its exact cleanup duty settles"; + EXPECT_EQ( + store->livePrecommitsForTest(ns), + (std::set>{{"successor", successor_id.ref}})); + EXPECT_TRUE(backend->head(abandoned_manifest_key).exists) + << "the removed precommit body remains GC-owned until its decrement is sealed"; + + successor->abandon(); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} + +/// Removing the absent-owner arm makes the deferred duty either stay forever or try to remove an owner +/// that was never transmitted. A controller pre-attempt refusal proves the grant absent; the next +/// healthy mutation must drain that duty as a no-op and retire the old build before publishing itself. +TEST(CASWriterDuties, ProvenAbsentGrantDrainsAsNoOpBeforeTheNextMutation) +{ + auto backend = std::make_shared(); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + PoolConfig config = singleAttemptConfig(); + config.boot_ms_fn = [] { return uint64_t{0}; }; + config.mount_renew_period = std::chrono::hours{1}; + auto store = Pool::open(backend, config); + const RootNamespace ns{"srv1/writer_duty_reject"}; + + ManifestId rejected_id; + auto rejected = stageEmptyManifest(store, ns, "rejected", rejected_id); + const uint64_t rejected_seq = rejected->buildSeq(); + + store->setMountDeadline(100); + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + ASSERT_FALSE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + + rejected.reset(); + EXPECT_EQ(store->minActive(), rejected_seq) + << "the destructor cannot retire even an uncertain grant whose rejection has not been consumed"; + + store->setMountDeadline(30000); + ManifestId successor_id; + auto successor = stageEmptyManifest(store, ns, "successor", successor_id); + const uint64_t successor_seq = successor->buildSeq(); + successor->precommitAdd(ns, "successor", successor_id); + + EXPECT_EQ(store->minActive(), successor_seq); + EXPECT_EQ( + store->livePrecommitsForTest(ns), + (std::set>{{"successor", successor_id.ref}})); + + successor->abandon(); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} + +/// The model gate recorded both wedge-resolution witnesses (adopt and reject); the C++ suite drove +/// only the adopt arm above. This drives an uncertain grant into an ACTUAL wedged lane -- unlike +/// `ProvenAbsentGrantDrainsAsNoOpBeforeTheNextMutation`'s controller pre-attempt refusal, which never +/// wedges at all -- and resolves it as REJECT: `Mode::Unresolved` lands nothing, so the next attempt's +/// resolve-before-reissue GET proves the key absent. The duty must then drain as a no-op: no +/// `OwnerTransition` removal is owed for an absent precommit, the wedge clears, and `minActive` advances +/// past the rejected build exactly as the no-wedge reject arm does. +TEST(CASWriterDuties, WedgeResolvedAsRejectDrainsTheDutyAsNoOp) +{ + auto backend = std::make_shared(); + auto store = openSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_wedge_reject"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + + ManifestId rejected_id; + auto rejected = stageEmptyManifest(store, ns, "rejected", rejected_id); + const uint64_t rejected_seq = rejected->buildSeq(); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + + rejected.reset(); + EXPECT_EQ(store->minActive(), rejected_seq) + << "an unresolved owner grant must keep its build active after the transaction object is gone"; + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + ManifestId successor_id; + auto successor = stageEmptyManifest(store, ns, "successor", successor_id); + const uint64_t successor_seq = successor->buildSeq(); + successor->precommitAdd(ns, "successor", successor_id); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->minActive(), successor_seq) + << "the rejected build retires only after its exact cleanup duty settles as a no-op"; + EXPECT_EQ( + store->livePrecommitsForTest(ns), + (std::set>{{"successor", successor_id.ref}})); + + successor->abandon(); + EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); +} + +/// Removing `mutateRefsAfterWriterCleanup` from the `dropRef` delegate leaves the rejected build at +/// `minActive` even though the ref removal succeeds. The observable floor proves the direct API +/// serviced the inherited cleanup duty before performing its own mutation. +TEST(CASWriterDuties, DropRefServicesPendingDutyBeforeRemovingTheRef) +{ + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_drop_ref"}; + publishEmptyRef(store, ns, "target"); + leaveRejectedCleanupDuty(store, ns); + + store->dropRef(ns, "target"); + + EXPECT_FALSE(store->resolveRef(ns, "target").has_value()); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); +} + +/// Removing the shared drain seam from `updateRefPublishedAt` lets the timestamp mutation overtake a +/// pending writer duty. The update remains observable, while the independent watermark assertion +/// catches that bypass. +TEST(CASWriterDuties, UpdateRefPublishedAtServicesPendingDutyBeforeUpdatingTheRef) +{ + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_update_ref"}; + publishEmptyRef(store, ns, "target"); + leaveRejectedCleanupDuty(store, ns); + + store->updateRefPublishedAt(ns, "target", [](RefPublishedAtUpdate & update) { update.published_at_ms = 17; }); + + const auto resolved = store->resolveRef(ns, "target"); + ASSERT_TRUE(resolved.has_value()); + EXPECT_EQ(resolved->published_at_ms, 17); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); +} + +/// Each public namespace-removal overload has its own Pool delegate. Omitting the shared seam from +/// either one still removes the namespace but strands the rejected build at the active floor, so the +/// two independent cases protect both forwarding paths. +TEST(CASWriterDuties, DropNamespaceOverloadsServicePendingDutyBeforeRemoval) +{ + { + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_drop_namespace"}; + publishEmptyRef(store, ns, "target"); + leaveRejectedCleanupDuty(store, ns); + + store->dropNamespace(ns); + + EXPECT_TRUE(store->listRefs(ns).empty()); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); + } + + { + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_drop_namespace_life"}; + publishEmptyRef(store, ns, "target"); + const NamespaceLifeId life = store->namespaceLife(ns); + leaveRejectedCleanupDuty(store, ns); + + store->dropNamespace(life); + + EXPECT_TRUE(store->listRefs(ns).empty()); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); + } +} + +/// The explicit snapshot/checkpoint attempt is the audited sibling mutation: without the common seam +/// it may publish ledger state while leaving the older writer duty pinned. Its return value is allowed +/// to be false; advancing the active floor is the cleanup contract under test. +TEST(CASWriterDuties, SnapshotAttemptServicesPendingDutyBeforePublishingLedgerState) +{ + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_snapshot"}; + publishEmptyRef(store, ns, "target"); + leaveRejectedCleanupDuty(store, ns); + + static_cast(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); + + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()); +} + +/// Removing the pending-duty term from `Pool` teardown makes this test fail at the farewell +/// assertion: a clean marker would falsely certify that the durable precommit below has no remaining +/// writer work. The unclean handoff forces a fresh writer epoch; its arithmetic recovery seal then +/// makes the ordinary stale-precommit sweep the crash-remnant cleanup path. +TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRemnant) +{ + auto backend = std::make_shared(); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + const CasRequestBudget budget{ + .attempt_timeout_ms = 50, + .operation_deadline_ms = 500, + .max_attempts = 1, + .lease_safety_margin_ms = 50, + }; + const RootNamespace ns{"srv1/writer_duty_crash"}; + + auto predecessor = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_id = UInt128(1), + .server_root_id = "test", + .background_watermark = false, + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = budget, + }); + DB::Cas::tests::casAdmitRecoverableEntry( + *backend, predecessor->layout(), ns, predecessor->liveWriterEpoch()); + + ManifestId abandoned_id; + auto abandoned = stageEmptyManifest(predecessor, ns, "abandoned", abandoned_id); + abandoned->precommitAdd(ns, "abandoned", abandoned_id); + const uint64_t predecessor_epoch = predecessor->writerEpoch(); + const Layout layout = predecessor->layout(); + const String mount_key = layout.mountKey("test"); + + abandoned.reset(); + predecessor.reset(); + + const auto mount = backend->get(mount_key); + ASSERT_TRUE(mount.has_value()); + EXPECT_NE(decodeMountLease(mount->bytes).min_active, std::numeric_limits::max()) + << "a live writer-cleanup duty forbids the clean-release certificate"; + + uint64_t fake_boot = 0; + std::vector waits; + auto successor_store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_id = UInt128(1), + .server_root_id = "test", + .background_watermark = false, + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + }); + ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); + ASSERT_FALSE(waits.empty()) << "the predecessor supplied no clean-death certificate"; + + ManifestId successor_id; + auto successor = stageEmptyManifest(successor_store, ns, "successor", successor_id); + successor->precommitAdd(ns, "successor", successor_id); + + EXPECT_EQ( + successor_store->livePrecommitsForTest(ns), + (std::set>{{"successor", successor_id.ref}})); + const auto seal = successor_store->lastEpochSealForTest(ns); + ASSERT_TRUE(seal.has_value()); + EXPECT_EQ(seal->writer_epoch, predecessor_epoch); + + successor->abandon(); +} + +/// The duty queue above only ever resolves the ref +/// table's precommit BINDING -- a rejected grant's manifest BODY is orphan from birth (no owner ever +/// named it, so the edge-before-observe `+1` a durable precommit would have folded never landed +/// either) and its reclaim is entirely the orphan sweep's job, gated on the one thing the duty queue +/// cannot give it: the build's own epoch durably closed. This drives that closure (the same crash +/// pattern as `PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRemnant`, but the predecessor's +/// build is REJECTED rather than adopted) and then runs real GC rounds until the body is gone. +TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) +{ + auto backend = std::make_shared(); + DB::Cas::tests::seedPoolMetaForRestart(*backend); + const CasRequestBudget budget{ + .attempt_timeout_ms = 50, + .operation_deadline_ms = 500, + .max_attempts = 1, + .lease_safety_margin_ms = 50, + }; + /// Rooted under the POOL's OWN `server_root_id` ("test", unlike this file's other fixtures, which + /// stay under "srv1" precisely because they never drive the orphan sweep): `prefixEligible`'s + /// watermark floor is looked up by walking the NAMESPACE's own prefix segments for a live mount + /// lease, so a namespace rooted under any other server-root would find no floor and retain forever + /// regardless of epoch/coverage. + const RootNamespace ns{"test/writer_duty_rejected_sweep"}; + + auto predecessor = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_id = UInt128(1), + .server_root_id = "test", + .manifest_sweep_list_budget_keys = 100, + .manifest_sweep_delete_budget_keys = 100, + .gc_fold_max_defer_rounds = 0, + .background_watermark = false, + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = budget, + }); + + /// A real, fully-promoted ref through the ordinary production write path (no seeded catalog/ckpt) + /// gives the namespace genuine epoch-1 content, so the successor's recovery below has something + /// real to close -- unlike a pre-attempt-refused grant, which never touches the backend at all and + /// so leaves the namespace's fold coverage exactly where it started. + publishEmptyRef(predecessor, ns, "anchor"); + + ManifestId rejected_id; + auto rejected = stageEmptyManifest(predecessor, ns, "rejected", rejected_id); + const String rejected_manifest_key = predecessor->layout().manifestKey(rejected_id); + ASSERT_TRUE(backend->head(rejected_manifest_key).exists) + << "stageManifest's body write is unconditional; only the owner grant is refused below"; + + /// `Unresolved` lands nothing, so the wedge it leaves resolves as a conclusive REJECT once the + /// successor's own recovery walks past it -- unlike the ADOPT-arm crash-remnant test, this + /// manifest never becomes a live owner in any epoch. `anchor`'s real birth just above minted a + /// genuine (random) incarnation, so the fault key is computed from the namespace's ACTUAL life, + /// not the deterministic `fixtureLife` fallback a raw, never-touched fixture would use. + backend->fault_substr = predecessor->layout().namespaceStreamPrefix(predecessor->namespaceLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + ASSERT_TRUE(predecessor->refLaneWedgedForTest(ns)); + ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); + const uint64_t predecessor_epoch = predecessor->writerEpoch(); + + rejected.reset(); + predecessor.reset(); + + uint64_t fake_boot = 0; + std::vector waits; + auto successor_store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", + .server_id = UInt128(1), + .server_root_id = "test", + .manifest_sweep_list_budget_keys = 100, + .manifest_sweep_delete_budget_keys = 100, + .gc_fold_max_defer_rounds = 0, + .background_watermark = false, + .mount_lease_ttl_ms = std::chrono::milliseconds(500), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = budget, + .boot_ms_fn = [&] { return fake_boot; }, + .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + }); + ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); + ASSERT_FALSE(waits.empty()) << "the predecessor supplied no clean-death certificate"; + + /// An ordinary successor mutation both drains the inherited duty as a no-op (the rejected grant + /// was never durable) and forces the predecessor's dead epoch to close with an arithmetic seal -- + /// the fact rule (1) of the sweep's deletion premise reads. + ManifestId successor_id; + auto successor = stageEmptyManifest(successor_store, ns, "successor", successor_id); + successor->precommitAdd(ns, "successor", successor_id); + + EXPECT_EQ( + successor_store->livePrecommitsForTest(ns), + (std::set>{{"successor", successor_id.ref}})); + const auto seal = successor_store->lastEpochSealForTest(ns); + ASSERT_TRUE(seal.has_value()); + EXPECT_EQ(seal->writer_epoch, predecessor_epoch); + + Gc gc(successor_store, hexToU128("000000000000000000000000000000e1")); + for (int round = 0; round < 16 && backend->head(rejected_manifest_key).exists; ++round) + DB::Cas::tests::runRegularRoundReclaiming(gc); + + EXPECT_FALSE(backend->head(rejected_manifest_key).exists) + << "the rejected attempt's orphan manifest must eventually be nominated and swept once its " + "build epoch is durably closed"; + + successor->abandon(); +} + +/// The settlement's own ordering is load-bearing: append the exact `OwnerTransition` removal (or +/// observe conclusive absence), only then retire the build seq, only then drop the duty -- a throw +/// between those steps must leave the duty owned by nobody but the queue. Faulting the SETTLEMENT's +/// append (not the original grant, which is a plain pre-attempt refusal here) proves the retry path +/// directly: the duty survives the throw and the mutation it was blocking aborts with it, then the +/// very next drain -- once the fault clears -- settles the duty and lets that mutation proceed. +TEST(CASWriterDuties, DutySurvivesSettlementFailureForRetry) +{ + auto backend = std::make_shared(); + auto store = openFrozenSingleAttemptPool(backend); + const RootNamespace ns{"srv1/writer_duty_settlement_retry"}; + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); + publishEmptyRef(store, ns, "target"); + + /// A plain, unfaulted precommit that is simply destroyed without promote/abandon: Durable, never + /// settled, and (unlike a proven-absent grant) its duty's own settlement owes a REAL + /// `OwnerTransition` removal -- exactly the append this test needs to fault. + ManifestId durable_id; + auto durable = stageEmptyManifest(store, ns, "durable", durable_id); + const uint64_t durable_seq = durable->buildSeq(); + durable->precommitAdd(ns, "durable", durable_id); + durable.reset(); + ASSERT_TRUE(store->writerCleanupDutiesPendingForTest()); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_count = 1; + DB::Cas::tests::expectThrowsCode( + DB::ErrorCodes::NETWORK_ERROR, + [&] { store->dropRef(ns, "target"); }); + + EXPECT_TRUE(store->writerCleanupDutiesPendingForTest()) + << "a settlement that throws must retain the duty for retry, never lose it"; + EXPECT_TRUE(store->resolveRef(ns, "target").has_value()) + << "the settlement's failure must abort the mutation it was blocking too, not just its own append"; + EXPECT_EQ(store->minActive(), durable_seq); + + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; + store->dropRef(ns, "target"); + + EXPECT_FALSE(store->writerCleanupDutiesPendingForTest()); + EXPECT_FALSE(store->resolveRef(ns, "target").has_value()); + EXPECT_EQ(store->minActive(), store->peekNextBuildSeq()) + << "the retried drain settles the retained duty and lets the mutation proceed"; +} diff --git a/src/IO/ReadBufferFromFileView.cpp b/src/IO/ReadBufferFromFileView.cpp index 1304f372df06..22de4731673f 100644 --- a/src/IO/ReadBufferFromFileView.cpp +++ b/src/IO/ReadBufferFromFileView.cpp @@ -19,11 +19,13 @@ ReadBufferFromFileView::ReadBufferFromFileView( , file_offset_of_buffer_end(left_bound_) , original_working_buffer(working_buffer) { - /// Seek to the begin of file. + /// Seek to the begin of file. The impl still owns its native buffer state here (no swap yet), + /// so its buffer-end offset can be read directly after the seek. impl->seek(left_bound, SEEK_SET); + const size_t impl_buffer_end = impl->getPosition() + impl->available(); swap(*impl); - file_offset_of_buffer_end += available(); + file_offset_of_buffer_end = impl_buffer_end; original_working_buffer = working_buffer; resizeWorkingBuffer(); } @@ -40,14 +42,31 @@ void ReadBufferFromFileView::setReadUntilPosition(size_t position) throw Exception(ErrorCodes::ARGUMENT_OUT_OF_BOUND, "Cannot read until position: {}. File size is {}", position, getFileSize()); - executeWithOriginalBuffer([&]{ impl->setReadUntilPosition(*read_until_position); }); + /// The impl is allowed to DISCARD its working buffer here (e.g. `ReadBufferFromS3` rebases its + /// offset to the consumer position and resets the buffer when the range changes), so the view's + /// buffer-end offset MUST be rebased from the impl's post-op state - keeping the stale value + /// over a replaced buffer silently shifts the reported position by the discarded bytes. + size_t impl_buffer_end = 0; + executeWithOriginalBuffer([&] + { + impl->setReadUntilPosition(*read_until_position); + impl_buffer_end = impl->getPosition() + impl->available(); + }); + file_offset_of_buffer_end = impl_buffer_end; resizeWorkingBuffer(); } void ReadBufferFromFileView::setReadUntilEnd() { read_until_position.reset(); - executeWithOriginalBuffer([&]{ impl->setReadUntilPosition(right_bound); }); + /// Same rebase contract as setReadUntilPosition. + size_t impl_buffer_end = 0; + executeWithOriginalBuffer([&] + { + impl->setReadUntilPosition(right_bound); + impl_buffer_end = impl->getPosition() + impl->available(); + }); + file_offset_of_buffer_end = impl_buffer_end; resizeWorkingBuffer(); } @@ -63,11 +82,19 @@ bool ReadBufferFromFileView::nextImpl() return false; bool result = false; - executeWithOriginalBuffer([&] { result = impl->next(); }); + size_t impl_buffer_end = 0; + executeWithOriginalBuffer([&] + { + result = impl->next(); + impl_buffer_end = impl->getPosition() + impl->available(); + }); if (result) { - file_offset_of_buffer_end += available(); + /// Rebase from the impl's own accounting instead of incrementing: the view's previous + /// buffer-end may have been clamped by resizeWorkingBuffer below the impl's real one, and + /// the impl continues from ITS position - incrementing would mislabel the new chunk. + file_offset_of_buffer_end = impl_buffer_end; resizeWorkingBuffer(); } @@ -87,7 +114,12 @@ off_t ReadBufferFromFileView::seek(off_t off, int whence) throw Exception(ErrorCodes::ARGUMENT_OUT_OF_BOUND, "ReadBufferFromFileView::seek expects SEEK_SET or SEEK_CUR as whence"); off_t result = 0; - executeWithOriginalBuffer([&] { result = impl->seek(new_pos, SEEK_SET); }); + size_t impl_buffer_end = 0; + executeWithOriginalBuffer([&] + { + result = impl->seek(new_pos, SEEK_SET); + impl_buffer_end = impl->getPosition() + impl->available(); + }); if (result < 0) throw Exception(ErrorCodes::SEEK_POSITION_OUT_OF_BOUND, "Seek position ({}) underflow", result); @@ -96,7 +128,7 @@ off_t ReadBufferFromFileView::seek(off_t off, int whence) throw Exception(ErrorCodes::SEEK_POSITION_OUT_OF_BOUND, "Seek position ({}) is out of bound. Available range: [{}, {}]", result, left_bound, right_bound); - file_offset_of_buffer_end = result + available(); + file_offset_of_buffer_end = impl_buffer_end; resizeWorkingBuffer(); return result - left_bound; @@ -110,7 +142,20 @@ void ReadBufferFromFileView::executeWithOriginalBuffer(Op && op) /// Set working buffer and other internal into impl. swap(*impl); - op(); + try + { + op(); + } + catch (...) + { + /// The swap MUST be undone even if `op` throws — otherwise `this` and `impl` are left holding + /// each other's working buffers (and a stale `original_working_buffer`), so any subsequent + /// read or seek over-reads / serves wrong bytes. `op` can throw (e.g. setReadUntilPosition / + /// seek bound checks), so restore-on-exception is required for the view to stay consistent. + swap(*impl); + original_working_buffer = working_buffer; + throw; + } swap(*impl); original_working_buffer = working_buffer; diff --git a/src/IO/ReadBufferFromMemory.cpp b/src/IO/ReadBufferFromMemory.cpp index 882f8b6a07d3..9f3c20fc51e2 100644 --- a/src/IO/ReadBufferFromMemory.cpp +++ b/src/IO/ReadBufferFromMemory.cpp @@ -76,7 +76,10 @@ ReadBufferFromMemoryFileBase::ReadBufferFromMemoryFileBase(bool owns_memory, { chassert(data.size() == internal_buffer.size()); - if (owns_memory) + /// memcpy's pointers are __attribute__((nonnull)) even when the length is 0. An empty file yields + /// data.data() == nullptr, so guard on non-empty to avoid the nonnull-attribute UB the asan_ubsan + /// lane aborts on (STID 5930-5afa). Nothing to copy when empty. + if (owns_memory && !data.empty()) std::memcpy(internal_buffer.begin(), data.data(), data.size()); working_buffer = internal_buffer; diff --git a/src/IO/ReadBufferFromS3.cpp b/src/IO/ReadBufferFromS3.cpp index d904e749d224..df009a32cabf 100644 --- a/src/IO/ReadBufferFromS3.cpp +++ b/src/IO/ReadBufferFromS3.cpp @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -359,6 +360,12 @@ bool ReadBufferFromS3::processException(size_t read_offset, size_t attempt) cons bucket, key, version_id.empty() ? "Latest" : version_id, read_offset, attempt, request_settings[S3RequestSetting::max_single_read_retries].value, getCurrentExceptionMessage(/* with_stacktrace = */ false)); + /// Stop retrying once the query is cancelled (B117): otherwise a killed query's reads keep + /// retrying a transient error (e.g. a dropped connection) for many attempts with backoff, + /// zombying for minutes and adding load. The SDK's own RetryStrategy makes the same check + /// (src/IO/S3/Client.cpp), but this outer ReadBufferFromS3 retry loop did not. + if (CurrentThread::isInitialized() && CurrentThread::get().isQueryCanceled()) + return false; if (auto * s3_exception = current_exception_cast()) { diff --git a/src/IO/ReadPipeline.cpp b/src/IO/ReadPipeline.cpp index 0329c87d2a22..0ffe16612b18 100644 --- a/src/IO/ReadPipeline.cpp +++ b/src/IO/ReadPipeline.cpp @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -164,9 +165,21 @@ void ReadPipeline::needDecryption(String path, size_t buffer_size, KeyFinderFunc .key_finder = std::move(key_finder)}); } +<<<<<<< HEAD void ReadPipeline::needLongConnectionLimit(std::shared_ptr limit) { long_connection_limit = std::move(limit); +======= +void ReadPipeline::needFileView(String file_name, size_t left_bound, size_t right_bound) +{ + if (right_bound < left_bound) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "ReadPipeline: file view right bound ({}) is below the left bound ({})", right_bound, left_bound); + file_view = FileViewStage{ + .file_name = std::move(file_name), + .left_bound = left_bound, + .right_bound = right_bound}; +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr ReadPipeline::build() const @@ -194,7 +207,8 @@ std::unique_ptr ReadPipeline::build() const impl = wrapMemoryCache(std::move(impl)); // Stage 4 impl = wrapAsyncPrefetch(std::move(impl)); // Stage 5 - impl = wrapDecryption(std::move(impl)); // Stage 6 (encryption) + impl = wrapFileView(std::move(impl)); // Stage 6 (byte window) + impl = wrapDecryption(std::move(impl)); // Stage 7 (encryption) return impl; } @@ -205,6 +219,7 @@ std::unique_ptr ReadPipeline::tryBuildReaderExecutor() c if (!settings.reader_executor.enabled) return nullptr; +<<<<<<< HEAD /// The executor implements neither async prefetch nor the distributed cache, so fall back rather /// than silently drop those stages. Decryption, the filesystem cache, and the page (memory) cache /// ARE supported (fed below). @@ -213,6 +228,17 @@ std::unique_ptr ReadPipeline::tryBuildReaderExecutor() c LOG_DEBUG(log, "use_reader_executor: falling back to the legacy read path " "(distributed cache or async prefetch not supported by the executor)"); +======= + /// The executor does not implement caches, decryption, async prefetch, the + /// distributed cache, or a file_view byte window, so fall back rather than + /// silently drop a configured stage. + if (distributed_cache || memory_cache || !filesystem_caches.empty() + || !decryption_stages.empty() || async_prefetch || file_view) + { + LOG_DEBUG(log, + "use_reader_executor: falling back to the legacy read path " + "(caches/decryption/file_view not yet supported by the executor)"); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) return nullptr; } @@ -772,6 +798,19 @@ std::unique_ptr ReadPipeline::wrapAsyncPrefetch(std::uni async_prefetch->prefetches_log); } +std::unique_ptr ReadPipeline::wrapFileView(std::unique_ptr impl) const +{ + /// -- Stage 6: File view -- + /// The view translates the consumer's positions/right bounds by `left_bound` and forwards + /// them down the chain, so `MergeTreeReaderStream::adjustRightMark` bounds reach the + /// object-storage reader and its range requests stay drainable (connection-pool friendly). + if (!file_view) + return impl; + + return std::make_unique( + std::move(impl), file_view->file_name, file_view->left_bound, file_view->right_bound); +} + std::unique_ptr ReadPipeline::wrapDecryption(std::unique_ptr impl) const { /// -- Stage 6: Decryption (may have multiple layers for double encryption) -- @@ -832,6 +871,8 @@ String ReadPipeline::describe() const append("MemoryCache"); if (async_prefetch) append("AsyncPrefetch"); + if (file_view) + append("FileView"); if (!decryption_stages.empty()) append("Decrypt"); diff --git a/src/IO/ReadPipeline.h b/src/IO/ReadPipeline.h index 58dc0d2210ae..0526237d0bc3 100644 --- a/src/IO/ReadPipeline.h +++ b/src/IO/ReadPipeline.h @@ -49,7 +49,8 @@ using FilesystemReadPrefetchesLogPtr = std::shared_ptr limit); @@ -168,6 +170,16 @@ class ReadPipeline /// disks on random-object-key backends (see `DiskEncrypted::prepareRead`). Deterministic-path /// backends and url or external reads leave it null, so a reused key cannot serve a stale header. void needEncryptionHeaderCache(std::shared_ptr cache) { encryption_header_cache = std::move(cache); } +======= + /// -- File view stage -- + /// Exposes ONLY the byte window [left_bound, right_bound) of the underlying chain as a + /// standalone file named `file_name` (ReadBufferFromFileView). Used by content-addressed + /// blob reads, where a logical file is a payload window inside a shared blob (the blob's + /// envelope header occupies [0, left_bound)). Sits outside async prefetch — the window's + /// seeks and right bounds are translated and forwarded down the standard chain — but + /// inside decryption, which operates on logical-file bytes. + void needFileView(String file_name, size_t left_bound, size_t right_bound); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// -- Build the final ReadBuffer chain -- /// Uses the ReadSettings stored in the source stage. @@ -221,6 +233,13 @@ class ReadPipeline KeyFinderFunc key_finder; }; + struct FileViewStage + { + String file_name; + size_t left_bound = 0; + size_t right_bound = 0; + }; + struct DistributedCacheStage { @@ -235,8 +254,12 @@ class ReadPipeline std::optional distributed_cache; std::optional async_prefetch; VectorWithMemoryTracking decryption_stages; +<<<<<<< HEAD /// Global encryption-header cache for the executor; null unless a random-object-key disk set it. std::shared_ptr encryption_header_cache; +======= + std::optional file_view; +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) LoggerPtr log = getLogger("ReadPipeline"); @@ -261,6 +284,7 @@ class ReadPipeline std::unique_ptr buildSingleObjectStage(const std::string & query_id) const; std::unique_ptr wrapMemoryCache(std::unique_ptr impl) const; std::unique_ptr wrapAsyncPrefetch(std::unique_ptr impl) const; + std::unique_ptr wrapFileView(std::unique_ptr impl) const; std::unique_ptr wrapDecryption(std::unique_ptr impl) const; }; diff --git a/src/IO/S3/Client.cpp b/src/IO/S3/Client.cpp index 7f16eb8fb0db..d8aa66d4873d 100644 --- a/src/IO/S3/Client.cpp +++ b/src/IO/S3/Client.cpp @@ -25,8 +25,10 @@ #include #include +#include #include +#include #include #include #include @@ -65,6 +67,8 @@ namespace ProfileEvents extern const Event S3Clients; extern const Event TinyS3Clients; + + extern const Event S3SingleAttemptRetryConsultations; } namespace CurrentMetrics @@ -104,6 +108,13 @@ bool Client::RetryStrategy::ShouldRetry(const Aws::Client::AWSError= config.max_retries) return false; @@ -183,6 +194,13 @@ void Client::RetryStrategy::RequestBookkeeping( RequestBookkeeping(httpResponseOutcome); } +/// NOLINTNEXTLINE(google-runtime-int) +bool SingleAttemptRetryStrategy::ShouldRetry(const Aws::Client::AWSError &, long) const +{ + ProfileEvents::increment(ProfileEvents::S3SingleAttemptRetryConsultations); + return false; +} + namespace { @@ -291,7 +309,14 @@ Client::Client( /// find credential keys we can simply behave as the underlying storage is S3 /// otherwise, we need to be aware we are making requests to GCS /// and replace all headers with a valid prefix when needed - if (credentials_provider) + if (Poco::toLower(client_configuration.http_client) == "gcs_hmac") + { + /// GOOG4-HMAC mode: all requests are re-signed with x-goog headers at the HTTP layer, + /// so the SDK-side GCS accommodations (x-amz header renames, x-amz-api-version + /// deletion) must be active even though credentials are present. + api_mode = ApiMode::GCS; + } + else if (credentials_provider) { auto credentials = credentials_provider->GetAWSCredentials(); if (credentials.IsEmpty()) @@ -518,6 +543,12 @@ Model::GetObjectTaggingOutcome Client::GetObjectTagging(GetObjectTaggingRequest doRequest(request, [this](const Model::GetObjectTaggingRequest & req) { return GetObjectTagging(req); })); } +Model::GetBucketVersioningOutcome Client::GetBucketVersioning(GetBucketVersioningRequest & request) const +{ + return processRequestResult( + doRequest(request, [this](const Model::GetBucketVersioningRequest & req) { return GetBucketVersioning(req); })); +} + Model::ListObjectsV2Outcome Client::ListObjectsV2(ListObjectsV2Request & request) const { return doRequestWithRetryNetworkErrors( @@ -967,6 +998,17 @@ bool Client::supportsMultiPartCopy() const return provider_type != ProviderType::GCS; } +bool httpClientImpliesGcsGenerationDialect(const String & http_client) +{ + const auto lowered = Poco::toLower(http_client); + return lowered == "gcp_oauth" || lowered == "gcs_hmac"; +} + +bool Client::supportsGcsNativeConditionalRequests() const +{ + return httpClientImpliesGcsGenerationDialect(client_configuration.http_client); +} + void Client::BuildHttpRequest(const Aws::AmazonWebServiceRequest& request, const std::shared_ptr& httpRequest) const { @@ -979,6 +1021,15 @@ void Client::BuildHttpRequest(const Aws::AmazonWebServiceRequest& request, /// note that "amz-sdk-invocation-id" and "amz-sdk-request" are preserved httpRequest->DeleteHeader("x-amz-api-version"); } + + /// Re-derived on every attempt: a retry or redirect discards the old HTTP request and builds a + /// fresh one (see AWSClient::AttemptExhaustively), so the bit cannot be left to survive on it. + if (auto * extended_http_request = dynamic_cast(httpRequest.get())) + { + const auto * wrapper = dynamic_cast(&request); + extended_http_request->setNativeConditional( + wrapper && wrapper->isNativeConditional() && supportsGcsNativeConditionalRequests()); + } } std::string Client::getGCSOAuthToken() const @@ -1309,6 +1360,9 @@ std::unique_ptr ClientFactory::create( // NOLINT auto credentials_provider = getCredentialsProvider(client_configuration, credentials, credentials_configuration); + if (Poco::toLower(client_configuration.http_client) == "gcs_hmac") + client_configuration.gcs_hmac_credentials_provider = credentials_provider; + /// Disable per-thread retry loops if global retry coordination is in use. if (client_configuration.s3_slow_all_threads_after_retryable_error) { diff --git a/src/IO/S3/Client.h b/src/IO/S3/Client.h index 4f679588689a..a53e981b1382 100644 --- a/src/IO/S3/Client.h +++ b/src/IO/S3/Client.h @@ -113,6 +113,13 @@ struct ClientSettings bool is_s3express_bucket = false; }; +/// True for the two `http_client` values (case-insensitive) that select a GCS-native HTTP layer: +/// `gcp_oauth` and `gcs_hmac`. The single source of truth for what "GCS generation dialect" means +/// from configuration alone -- `Client::supportsGcsNativeConditionalRequests` below is this applied to +/// a constructed client's own configuration; a caller that needs the answer before a client exists +/// (e.g. deciding whether a config change would flip the dialect) calls this directly instead. +bool httpClientImpliesGcsGenerationDialect(const String & http_client); + /// Client that improves the client from the AWS SDK /// - inject region and URI into requests so they are rerouted to the correct destination if needed /// - automatically detect endpoint and regions for each bucket and cache them @@ -142,7 +149,7 @@ class Client : private Aws::S3::S3Client std::unique_ptr clone() const; - std::unique_ptr cloneWithConfigurationOverride(const PocoHTTPClientConfiguration & client_configuration_override) const; + virtual std::unique_ptr cloneWithConfigurationOverride(const PocoHTTPClientConfiguration & client_configuration_override) const; Client & operator=(const Client &) = delete; @@ -206,6 +213,7 @@ class Client : private Aws::S3::S3Client Model::HeadObjectOutcome HeadObject(HeadObjectRequest & request) const; Model::GetObjectTaggingOutcome GetObjectTagging(GetObjectTaggingRequest & request) const; + Model::GetBucketVersioningOutcome GetBucketVersioning(GetBucketVersioningRequest & request) const; Model::ListObjectsV2Outcome ListObjectsV2(ListObjectsV2Request & request) const; Model::ListObjectsOutcome ListObjects(ListObjectsRequest & request) const; Model::GetObjectOutcome GetObject(GetObjectRequest & request) const; @@ -250,6 +258,10 @@ class Client : private Aws::S3::S3Client const PocoHTTPClientConfiguration & getClientConfiguration() const { return client_configuration; } + /// True when this client's HTTP layer can honor the typed `NativeConditional` request mode + /// (http_client = gcs_hmac or gcp_oauth), independent of whether any given request opts in. + bool supportsGcsNativeConditionalRequests() const; + /// For testing purposes only ClientCache * getRawCache() const { return cache.get(); } @@ -271,6 +283,7 @@ class Client : private Aws::S3::S3Client /// otherwise region and endpoint redirection won't work using Aws::S3::S3Client::HeadObject; using Aws::S3::S3Client::GetObjectTagging; + using Aws::S3::S3Client::GetBucketVersioning; using Aws::S3::S3Client::ListObjectsV2; using Aws::S3::S3Client::ListObjects; using Aws::S3::S3Client::GetObject; @@ -346,6 +359,17 @@ class Client : private Aws::S3::S3Client LoggerPtr log; }; +/// Refuses every SDK-transparent retry and counts each consultation. Used by the +/// ObjectStorageRetryProfile::SingleAttempt per-write profile (conditional writes whose retry +/// decisions live ABOVE the SDK: the caller must resolve an uncertain PUT before reissuing). +class SingleAttemptRetryStrategy final : public Aws::Client::RetryStrategy +{ +public: + bool ShouldRetry(const Aws::Client::AWSError &, long) const override; // NOLINT(google-runtime-int) + long CalculateDelayBeforeNextRetry(const Aws::Client::AWSError &, long) const override { return 0; } // NOLINT(google-runtime-int) + long GetMaxAttempts() const override { return 1; } // NOLINT(google-runtime-int) +}; + class ClientFactory { public: diff --git a/src/IO/S3/GCSConditionalDialect.cpp b/src/IO/S3/GCSConditionalDialect.cpp new file mode 100644 index 000000000000..faa454f8737e --- /dev/null +++ b/src/IO/S3/GCSConditionalDialect.cpp @@ -0,0 +1,256 @@ +#include + +#if USE_AWS_S3 + +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; +} + +namespace DB::S3 +{ + +namespace +{ + +constexpr std::string_view AMZ_PREFIX = "x-amz-"; +constexpr std::string_view GOOG_PREFIX = "x-goog-"; +constexpr std::string_view AMZ_META_PREFIX = "x-amz-meta-"; +constexpr std::string_view GOOG_META_PREFIX = "x-goog-meta-"; + +/// What both GCS authentication modes clear first: the SigV4 signature and the headers it was +/// computed over, which describe a canonical request neither GOOG4 nor Bearer authentication sends, +/// plus `x-amz-api-version`, which GCS rejects and which the SDK layer only removes when it +/// recognised the endpoint as GCS. +constexpr std::array AWS_HEADERS_CLEARED_BEFORE_GCS_AUTHENTICATION{ + "authorization", "x-amz-date", "x-amz-content-sha256", "x-amz-security-token", "x-amz-api-version"}; + +bool isAllDigits(const std::string & s) +{ + return !s.empty() && std::all_of(s.begin(), s.end(), [](char c) { return c >= '0' && c <= '9'; }); +} + +std::string stripQuotes(const std::string & s) +{ + if (s.size() >= 2 && s.front() == '"' && s.back() == '"') + return s.substr(1, s.size() - 2); + return s; +} + +std::string toLower(std::string_view s) +{ + std::string out{s}; + std::transform(out.begin(), out.end(), out.begin(), [](unsigned char c) { return static_cast(std::tolower(c)); }); + return out; +} + +/// Rename `name` to its `x-goog-` counterpart, refusing to pick a winner when the target already +/// carries a different value. +void renameToGoogPrefix(Aws::Http::HttpRequest & request, const std::string & name) +{ + const std::string value = request.GetHeaderValue(name.c_str()); + const std::string goog_name = std::string{GOOG_PREFIX} + name.substr(AMZ_PREFIX.size()); + if (request.HasHeader(goog_name.c_str()) && request.GetHeaderValue(goog_name.c_str()) != value) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "GCS request adaptation: '{}' and '{}' carry different values, so renaming would silently " + "discard one of them", name, goog_name); + request.DeleteHeader(name.c_str()); + request.SetHeaderValue(goog_name.c_str(), value); +} + +/// What GOOG4 authentication does with one `x-amz-*` request header. +enum class Goog4Disposition : uint8_t +{ + Rename, /// GCS accepts the same semantics under the x-goog- prefix + Consume, /// meaningful only to the AWS SDK; drop it, the wire request is unaffected + Reject, /// GCS cannot honor it and dropping it would change what the request means +}; + +struct Goog4HeaderRule +{ + std::string_view name; /// matched as a prefix when `is_prefix`, otherwise exactly + bool is_prefix; + Goog4Disposition disposition; +}; + +/// Every `x-amz-*` header ClickHouse's S3 requests can carry, with the reason for its fate. A header +/// absent from this table is rejected: this path deliberately signs a request whose prefixes are all +/// `x-goog-`, so guessing a translation or passing one through are both worse than an error naming +/// the header. +constexpr std::array GOOG4_HEADER_RULES{ + /// GCS object metadata is `x-goog-meta-*`; the storage class, copy source and metadata directive + /// are the same headers under the other prefix. + Goog4HeaderRule{AMZ_META_PREFIX, true, Goog4Disposition::Rename}, + Goog4HeaderRule{"x-amz-storage-class", false, Goog4Disposition::Rename}, + Goog4HeaderRule{"x-amz-copy-source", false, Goog4Disposition::Rename}, + Goog4HeaderRule{"x-amz-copy-source-range", false, Goog4Disposition::Rename}, + Goog4HeaderRule{"x-amz-metadata-directive", false, Goog4Disposition::Rename}, + + /// Flexible checksums are an S3 protocol feature: the algorithm selector and the computed value + /// mean nothing to the GCS XML API, and the body they describe is sent unchanged either way. + Goog4HeaderRule{"x-amz-sdk-checksum-algorithm", false, Goog4Disposition::Consume}, + Goog4HeaderRule{"x-amz-checksum-", true, Goog4Disposition::Consume}, + + /// These two announce `aws-chunked` body framing, which GCS cannot parse. Dropping them would + /// leave the framed body on the wire described as a plain one, so refuse instead. + Goog4HeaderRule{"x-amz-trailer", false, Goog4Disposition::Reject}, + Goog4HeaderRule{"x-amz-decoded-content-length", false, Goog4Disposition::Reject}, +}; + +std::optional goog4DispositionFor(std::string_view name) +{ + for (const auto & rule : GOOG4_HEADER_RULES) + if (!rule.is_prefix && name == rule.name) + return rule.disposition; + for (const auto & rule : GOOG4_HEADER_RULES) + if (rule.is_prefix && name.starts_with(rule.name)) + return rule.disposition; + return std::nullopt; +} + +} + +void applyGcsConditionalDialectToRequest(Aws::Http::HttpRequest & request) +{ + const auto query_params = request.GetUri().GetQueryStringParameters(); + const bool is_complete_multipart = request.GetMethod() == Aws::Http::HttpMethod::HTTP_POST + && query_params.contains("uploadId") && !query_params.contains("partNumber"); + + /// --- Conditional headers -> x-goog-if-generation-match --- + std::optional generation_match; + if (request.HasHeader("if-none-match")) + { + const auto value = request.GetHeaderValue("if-none-match"); + if (value != "*") + throw Exception(ErrorCodes::LOGICAL_ERROR, + "GCS native-conditional request: If-None-Match with a value other than '*' has no GCS " + "equivalent (got '{}') — refusing to silently change semantics", value); + generation_match = "0"; + request.DeleteHeader("if-none-match"); + } + if (request.HasHeader("if-match")) + { + const auto value = stripQuotes(request.GetHeaderValue("if-match")); + if (!isAllDigits(value)) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "GCS native-conditional request: If-Match value '{}' is not a generation number, so it " + "cannot name an incarnation on this backend", value); + generation_match = value; + request.DeleteHeader("if-match"); + } + if (generation_match) + { + if (is_complete_multipart) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "GCS native-conditional request: a CONDITIONAL CompleteMultipartUpload was about to be sent. " + "GCS silently ignores preconditions on CompleteMultipartUpload (measured 2026-07-03) — " + "this would be silent data loss. Conditional writes must use the single-PUT path."); + request.SetHeaderValue("x-goog-if-generation-match", *generation_match); + } + + /// --- Object metadata: x-amz-meta-* -> x-goog-meta-*, the prefix GCS documents --- + std::vector meta_headers; + for (const auto & header : request.GetHeaders()) + { + if (toLower(header.first).starts_with(AMZ_META_PREFIX)) + meta_headers.push_back(header.first); + } + for (const auto & name : meta_headers) + renameToGoogPrefix(request, name); +} + +void prepareGcsRequestForOAuthAuthentication(Aws::Http::HttpRequest & request) +{ + for (const auto * header : AWS_HEADERS_CLEARED_BEFORE_GCS_AUTHENTICATION) + request.DeleteHeader(header); +} + +void prepareGcsRequestForGoog4Authentication(Aws::Http::HttpRequest & request) +{ + for (const auto * header : AWS_HEADERS_CLEARED_BEFORE_GCS_AUTHENTICATION) + request.DeleteHeader(header); + + std::vector remaining; + for (const auto & header : request.GetHeaders()) + { + if (toLower(header.first).starts_with(AMZ_PREFIX)) + remaining.push_back(header.first); + } + + for (const auto & name : remaining) + { + const auto disposition = goog4DispositionFor(toLower(name)); + if (!disposition) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "GOOG4 authentication: header '{}' has no known GCS XML API counterpart, so it cannot be " + "translated. Sending it unchanged would mix the x-amz- and x-goog- prefixes in one " + "GOOG4-signed request, and whether GCS accepts that has not been established -- so it is " + "refused rather than guessed at. Remove it from the disk configuration, or use an " + "AWS-compatible endpoint.", + name); + + switch (*disposition) + { + case Goog4Disposition::Rename: + renameToGoogPrefix(request, name); + break; + case Goog4Disposition::Consume: + request.DeleteHeader(name.c_str()); + break; + case Goog4Disposition::Reject: + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "GOOG4 authentication: header '{}' announces aws-chunked body framing, which the GCS " + "XML API cannot parse. Dropping it would misdescribe the body already on the wire.", + name); + } + } +} + +void applyGcsConditionalDialectToResponse(const Poco::Net::HTTPResponse & poco_response, Aws::Http::HttpResponse & sdk_response) +{ + /// Each value is passed as a NAMED LVALUE on purpose, because the two `AddHeader` overloads of + /// `Aws::Http::Standard::StandardHttpResponse` are not equivalent: the `const Aws::String &` one + /// assigns through `operator[]` and replaces an existing header, while the `Aws::String &&` one + /// calls `emplace` and silently keeps the existing value. The caller's copy loop has already + /// installed the server's own `etag` and every `x-amz-meta-*`, so passing a temporary here would + /// no-op and leave the response unadapted. + if (poco_response.has("x-goog-generation")) + { + const std::string quoted_generation = "\"" + poco_response.get("x-goog-generation") + "\""; + sdk_response.AddHeader("ETag", quoted_generation); + } + + for (const auto & [name, value] : poco_response) + { + const std::string lower_name = toLower(name); + if (!lower_name.starts_with(GOOG_META_PREFIX)) + continue; + + const std::string amz_name = std::string{AMZ_META_PREFIX} + lower_name.substr(GOOG_META_PREFIX.size()); + if (poco_response.has(amz_name) && poco_response.get(amz_name) != value) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "GCS native-conditional response: '{}' and '{}' carry different values, so the object's " + "attributes are ambiguous", name, amz_name); + const std::string mapped_value = value; + sdk_response.AddHeader(amz_name, mapped_value); + } +} + +} + +#endif diff --git a/src/IO/S3/GCSConditionalDialect.h b/src/IO/S3/GCSConditionalDialect.h new file mode 100644 index 000000000000..6cbc8a833a92 --- /dev/null +++ b/src/IO/S3/GCSConditionalDialect.h @@ -0,0 +1,55 @@ +#pragma once +#include "config.h" +#if USE_AWS_S3 + +namespace Aws::Http { class HttpRequest; class HttpResponse; } +namespace Poco::Net { class HTTPResponse; } + +namespace DB::S3 +{ + +/// The GCS native-conditional adapter, request side. Applied at the wire boundary by the GCS-mode +/// Poco HTTP clients ONLY for a request marked `NativeConditional`, so ordinary traffic through the +/// same client keeps upstream AWS semantics. Translations: +/// - `If-None-Match: *` becomes `x-goog-if-generation-match: 0`; +/// - `If-Match: ""` (quotes optional) becomes `x-goog-if-generation-match: `; +/// - `x-amz-meta-*` becomes `x-goog-meta-*`, the prefix GCS documents for object metadata. +/// Fail-close guards, the request never leaves the process: +/// - `If-None-Match` with any value other than `*` (LOGICAL_ERROR: CAS only ever sends `*`); +/// - a non-numeric `If-Match` (CORRUPTED_DATA: a persisted token, or a storage response the +/// generation kind was stamped onto, that is not a generation number); +/// - the same metadata key under both prefixes with different values (BAD_ARGUMENTS); +/// - a CONDITIONAL CompleteMultipartUpload (POST with `uploadId` and no `partNumber`): GCS +/// silently ignores preconditions there (measured live 2026-07-03) — silent data loss. +void applyGcsConditionalDialectToRequest(Aws::Http::HttpRequest & request); + +/// Authentication preparation for the native OAuth path, run only for a `NativeConditional` request: +/// drop the stale AWS signing artifacts so the Bearer token is the only credential on the wire. +/// Every other `x-amz-*` header passes through unchanged, matching the ordinary OAuth path — there is +/// deliberately no GOOG4-style allowlist here. +void prepareGcsRequestForOAuthAuthentication(Aws::Http::HttpRequest & request); + +/// Authentication preparation for the GOOG4-HMAC path, run for EVERY request that client sends. +/// This path normalises prefixes deliberately: it signs with Google's native scheme, so every +/// `x-amz-*` header must have a decided fate before signing — dropped as an AWS signing artifact, +/// renamed to its `x-goog-` counterpart, or consumed because GCS has no counterpart. An `x-amz-*` +/// header with no rule raises BAD_ARGUMENTS rather than being guessed at or sent as-is. Whether GCS +/// would in fact reject a mixed-prefix request has not been measured, and the thrown message says so +/// too: the refusal is fail-closed under that uncertainty, not a consequence of a known rejection. No +/// request shape ClickHouse constructs on a normal bucket produces one. +void prepareGcsRequestForGoog4Authentication(Aws::Http::HttpRequest & request); + +/// The adapter, response side, applied only for a `NativeConditional` request: copies the header +/// changes the AWS SDK parser needs from `poco_response` onto `sdk_response`. The generation IS the +/// incarnation token on GCS, so it is installed QUOTED as `ETag` and rides the entire existing +/// ETag/token plumbing unchanged; `x-goog-meta-*` is presented as `x-amz-meta-*`. The same metadata +/// key arriving under both prefixes with different values raises CORRUPTED_DATA. A `Default` +/// response is never passed here and so keeps its upstream ETag and headers byte-for-byte. +/// Consequence: CAS object attributes are legible only through a marked read. The AWS SDK parses only +/// `x-amz-meta-*` into its metadata map and this function holds the only reverse mapping, so a +/// `Default` read of a CAS object's attributes yields a silently empty map rather than an error. +void applyGcsConditionalDialectToResponse(const Poco::Net::HTTPResponse & poco_response, Aws::Http::HttpResponse & sdk_response); + +} + +#endif diff --git a/src/IO/S3/GOOG4Signer.cpp b/src/IO/S3/GOOG4Signer.cpp new file mode 100644 index 000000000000..740b65d85c44 --- /dev/null +++ b/src/IO/S3/GOOG4Signer.cpp @@ -0,0 +1,149 @@ +#include + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include + +#include +#include + +#include +#include + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; +} + +namespace DB::S3 +{ + +namespace +{ + +constexpr auto UNSIGNED_PAYLOAD = "UNSIGNED-PAYLOAD"; + +std::string hmacSHA256(const std::string & key, const std::string & message) +{ + unsigned char out[SHA256_DIGEST_LENGTH]; + unsigned int out_len = 0; + HMAC(EVP_sha256(), + key.data(), static_cast(key.size()), + reinterpret_cast(message.data()), message.size(), + out, &out_len); + return std::string(reinterpret_cast(out), out_len); +} + +std::string sha256Hex(const std::string & data) +{ + unsigned char out[SHA256_DIGEST_LENGTH]; + SHA256(reinterpret_cast(data.data()), data.size(), out); + return hexString(out, SHA256_DIGEST_LENGTH); +} + +} + +void signRequestGOOG4( + Aws::Http::HttpRequest & request, + const Aws::Auth::AWSCredentials & credentials, + std::chrono::system_clock::time_point now) +{ + const std::time_t now_t = std::chrono::system_clock::to_time_t(now); + std::tm tm_utc{}; + gmtime_r(&now_t, &tm_utc); + const std::string timestamp = fmt::format( + "{:04}{:02}{:02}T{:02}{:02}{:02}Z", + tm_utc.tm_year + 1900, tm_utc.tm_mon + 1, tm_utc.tm_mday, + tm_utc.tm_hour, tm_utc.tm_min, tm_utc.tm_sec); + const std::string datestamp = timestamp.substr(0, 8); + + request.SetHeaderValue("x-goog-date", timestamp); + request.SetHeaderValue("x-goog-content-sha256", UNSIGNED_PAYLOAD); + + /// Canonical headers: `host` + every x-goog-* header, lowercase names, sorted. + /// std::map keeps them sorted for us. + std::map signed_headers_map; + for (const auto & [name, value] : request.GetHeaders()) + { + std::string lower = Aws::Utils::StringUtils::ToLower(name.c_str()); + if (lower == "host" || lower.starts_with("x-goog-")) + signed_headers_map.emplace(std::move(lower), value); + } + if (!signed_headers_map.contains("host")) + throw Exception(ErrorCodes::LOGICAL_ERROR, "GOOG4 signing requires a Host header on the request"); + + std::string canonical_headers; + std::string signed_headers; + for (const auto & [name, value] : signed_headers_map) + { + canonical_headers += name + ":" + value + "\n"; + if (!signed_headers.empty()) + signed_headers += ";"; + signed_headers += name; + } + + /// Canonical query string: URL-encoded key=value pairs sorted by key; a parameter without a + /// value still gets a trailing `=` (e.g. `versioning=`). + /// + /// `Aws::Http::URI` has no ready-made helper for this: `CanonicalizeQueryString` only rewrites + /// the query string when it already contains an `=`, so a bare flag like `?versioning` (no `=`) + /// passes through unsorted and unencoded. `GetQueryStringParameters` doesn't help either — for + /// a valueless flag it has no `=` to split on, so it treats the whole `key` as the `value` too + /// (`versioning` becomes `versioning=versioning`, not `versioning=`). Parse the raw query string + /// by hand instead, splitting each `key[=value]` pair on the first `=` with an empty value when + /// absent, then URL-encode and join sorted `key=value` pairs with `&`. + std::map query_params; + { + const std::string raw_query = request.GetUri().GetQueryString(); + size_t pos = raw_query.empty() ? std::string::npos : 1; /// skip leading '?' + while (pos != std::string::npos && pos < raw_query.size()) + { + const size_t amp = raw_query.find('&', pos); + const std::string pair = raw_query.substr(pos, amp == std::string::npos ? std::string::npos : amp - pos); + const size_t eq = pair.find('='); + std::string key = eq == std::string::npos ? pair : pair.substr(0, eq); + std::string value = eq == std::string::npos ? std::string() : pair.substr(eq + 1); + query_params.emplace( + Aws::Utils::StringUtils::URLDecode(key.c_str()), + Aws::Utils::StringUtils::URLDecode(value.c_str())); + pos = amp == std::string::npos ? std::string::npos : amp + 1; + } + } + std::string canonical_query; + for (const auto & [key, value] : query_params) + { + if (!canonical_query.empty()) + canonical_query += "&"; + canonical_query += Aws::Utils::StringUtils::URLEncode(key.c_str()) + "=" + Aws::Utils::StringUtils::URLEncode(value.c_str()); + } + const std::string canonical_uri = request.GetUri().GetURLEncodedPath(); + + const std::string method = Aws::Http::HttpMethodMapper::GetNameForHttpMethod(request.GetMethod()); + + const std::string canonical_request = fmt::format( + "{}\n{}\n{}\n{}\n{}\n{}", + method, canonical_uri, canonical_query, canonical_headers, signed_headers, UNSIGNED_PAYLOAD); + + const std::string scope = fmt::format("{}/auto/storage/goog4_request", datestamp); + const std::string string_to_sign = fmt::format( + "GOOG4-HMAC-SHA256\n{}\n{}\n{}", timestamp, scope, sha256Hex(canonical_request)); + + std::string key = hmacSHA256("GOOG4" + credentials.GetAWSSecretKey(), datestamp); + key = hmacSHA256(key, "auto"); + key = hmacSHA256(key, "storage"); + key = hmacSHA256(key, "goog4_request"); + const std::string signature = hexString(hmacSHA256(key, string_to_sign).data(), SHA256_DIGEST_LENGTH); + + request.SetHeaderValue("authorization", fmt::format( + "GOOG4-HMAC-SHA256 Credential={}/{}, SignedHeaders={}, Signature={}", + credentials.GetAWSAccessKeyId(), scope, signed_headers, signature)); +} + +} + +#endif diff --git a/src/IO/S3/GOOG4Signer.h b/src/IO/S3/GOOG4Signer.h new file mode 100644 index 000000000000..4b1f4c1b89c0 --- /dev/null +++ b/src/IO/S3/GOOG4Signer.h @@ -0,0 +1,30 @@ +#pragma once +#include "config.h" +#if USE_AWS_S3 + +#include + +namespace Aws::Http { class HttpRequest; } +namespace Aws::Auth { class AWSCredentials; } + +namespace DB::S3 +{ + +/// Sign `request` in place with GOOG4-HMAC-SHA256 — Google Cloud Storage's native V4 HMAC scheme +/// for the XML API. Structurally sigv4 with renamed constants: key prefix `GOOG4`, scope +/// terminator `goog4_request`, headers `x-goog-date` / `x-goog-content-sha256`. Bodies are never +/// hashed (`UNSIGNED-PAYLOAD`), so streaming uploads sign in O(1). +/// +/// Signs the `host` header plus EVERY `x-goog-*` header present on the request (GCS requires all +/// x-goog headers to be signed); other headers ride unsigned. `now` is injected so unit tests can +/// pin the timestamp to fixed vectors. +/// +/// Live-validated against GCS 2026-07-03 (see `utils/ca-soak/scripts/gcs_goog4_probe.py`, 12/12). +void signRequestGOOG4( + Aws::Http::HttpRequest & request, + const Aws::Auth::AWSCredentials & credentials, + std::chrono::system_clock::time_point now); + +} + +#endif diff --git a/src/IO/S3/PocoHTTPClient.cpp b/src/IO/S3/PocoHTTPClient.cpp index 5e3f11aec973..1b258935687d 100644 --- a/src/IO/S3/PocoHTTPClient.cpp +++ b/src/IO/S3/PocoHTTPClient.cpp @@ -7,6 +7,9 @@ #if USE_AWS_S3 #include +#include +#include +#include #include #include @@ -24,6 +27,7 @@ #include #include +#include #include #include #include @@ -89,6 +93,7 @@ namespace DB::ErrorCodes extern const int DNS_ERROR; extern const int AUTHENTICATION_FAILED; extern const int BAD_ARGUMENTS; + extern const int LOGICAL_ERROR; } namespace HistogramMetrics @@ -221,6 +226,7 @@ PocoHTTPClient::PocoHTTPClient(const PocoHTTPClientConfiguration & client_config , remote_host_filter(client_configuration.remote_host_filter) , s3_max_redirects(client_configuration.s3_max_redirects) , s3_use_adaptive_timeouts(client_configuration.s3_use_adaptive_timeouts) + , expect_continue_min_bytes(client_configuration.expect_continue_min_bytes) , http_max_fields(client_configuration.http_max_fields) , http_max_field_name_size(client_configuration.http_max_field_name_size) , http_max_field_value_size(client_configuration.http_max_field_value_size) @@ -466,6 +472,12 @@ void PocoHTTPClient::makeRequestInternalImpl( Aws::Utils::RateLimits::RateLimiterInterface *, Aws::Utils::RateLimits::RateLimiterInterface *) const { + /// Every request reaching this common HTTP boundary was built by `PocoHTTPClientFactory`, the + /// sole process-wide `Aws::Http::HttpClientFactory`, and so must be an `ExtendedHttpRequest`. + /// A foreign request would still read safely as `Default` via `isNativeConditionalRequest`, so + /// this is a construction-invariant check, not the read path itself. + chassert(dynamic_cast(&request) != nullptr); + LoggerPtr log = getLogger("AWSClient"); auto uri = request.GetUri().GetURIString(); @@ -618,6 +630,43 @@ void PocoHTTPClient::makeRequestInternalImpl( Stopwatch watch; + /// A conditional write (`If-None-Match` / `If-Match`) that loses its precondition can waste + /// a LARGE body: streaming multi-MB into a request the server has already decided to reject + /// makes some stores (e.g. RustFS) close mid-upload or answer a retryable 500, which the SDK + /// then RETRIES up to `s3_retry_attempts` (500) times — a ~40-min stall that hangs CA INSERTs + /// (see B118). `Expect: 100-continue` lets the server reject (e.g. 412) BEFORE the body, so we + /// skip the doomed upload. + /// + /// `expect_continue_min_bytes` is the negotiation gate: `0` (the default, carried by every + /// non-CAS S3 client) DISABLES it entirely — non-CAS conditional PUTs keep upstream wire + /// behaviour and this whole block, INCLUDING the body-size probe, is skipped. A positive value + /// negotiates Expect for a conditional PUT whose body is at least that many bytes; only a CAS + /// conditional-write client raises it (the single-attempt client built in `ObjectStorageBackend`), + /// so the scope is exactly CAS-owned conditional writes. `x-goog-if-generation-match` is what a + /// native-conditional GCS request carries instead: the GCS-mode clients translate If-None-Match / + /// If-Match before delegating here, so all three forms are visible at this point. + bool conditional_write = false; + if (expect_continue_min_bytes > 0 + && method == Poco::Net::HTTPRequest::HTTP_PUT + && (poco_request.has("if-none-match") || poco_request.has("if-match") + || poco_request.has("x-goog-if-generation-match"))) + { + size_t content_body_size = 0; + if (const auto & content_body = request.GetContentBody()) + { + content_body->clear(); + content_body->seekg(0, std::ios_base::end); + const auto end_pos = content_body->tellg(); + content_body->clear(); + content_body->seekg(0, std::ios_base::beg); + if (end_pos > 0) + content_body_size = static_cast(end_pos); + } + conditional_write = content_body_size >= expect_continue_min_bytes; + } + if (conditional_write) + poco_request.setExpectContinue(true); + auto & request_body_stream = session->sendRequest(poco_request, &connect_time, &first_byte_time); /// We record connect time here and not earlier, so that if an exception occurs while sending a request, /// we won't record the same latency twice. @@ -625,7 +674,20 @@ void PocoHTTPClient::makeRequestInternalImpl( observeLatency(request, first_byte_latency_type, static_cast(first_byte_time)); latency_recorded = true; - if (request.GetContentBody()) + /// With `Expect: 100-continue`, peek the interim response after the headers. `true` means + /// the server sent `100 Continue` (proceed with the body); `false` means it already sent a + /// FINAL response (now in `poco_response`) and the body must NOT be sent. `receiveResponse` + /// below is still called in both cases (Poco contract) and skips re-reading the headers. + bool skip_body = false; + if (conditional_write) + { + setTimeouts(*session, getTimeouts(method, first_attempt, /*first_byte*/ true)); + skip_body = !session->peekResponse(poco_response); + if (enable_s3_requests_logging) + LOG_TEST(log, "Expect: 100-continue peek -> {}", skip_body ? "final response, skipping body" : "100 Continue"); + } + + if (request.GetContentBody() && !skip_body) { if (enable_s3_requests_logging) LOG_TEST(log, "Writing request body."); @@ -694,6 +756,12 @@ void PocoHTTPClient::makeRequestInternalImpl( response->SetResponseCode(static_cast(status_code)); response->SetContentType(poco_response.getContentType()); + auto apply_gcs_native_response_adaptation = [&] + { + if (isNativeConditionalRequest(request)) + applyGcsConditionalDialectToResponse(poco_response, *response); + }; + if (enable_s3_requests_logging) { WriteBufferFromOwnString headers_ss; @@ -702,12 +770,14 @@ void PocoHTTPClient::makeRequestInternalImpl( response->AddHeader(header_name, header_value); headers_ss << header_name << ": " << header_value << "; "; } + apply_gcs_native_response_adaptation(); LOG_TEST(log, "Received headers: {}", headers_ss.str()); } else { for (const auto & [header_name, header_value] : poco_response) response->AddHeader(header_name, header_value); + apply_gcs_native_response_adaptation(); } /// Request is successful but for some special requests we can have actual error message in body @@ -855,6 +925,15 @@ void PocoHTTPClientGCPOAuth::makeRequestInternal( Aws::Utils::RateLimits::RateLimiterInterface * readLimiter, Aws::Utils::RateLimits::RateLimiterInterface * writeLimiter) const { + /// A `Default` request keeps pre-CAS upstream behaviour: the Bearer token replaces `Authorization` + /// and every other SDK header is left alone. Only a `NativeConditional` request acquires + /// generation semantics and has its stale AWS signing artifacts removed. + if (isNativeConditionalRequest(request)) + { + applyGcsConditionalDialectToRequest(request); + prepareGcsRequestForOAuthAuthentication(request); + } + { std::lock_guard lock(mutex); if (!bearer_token || std::chrono::system_clock::now() > bearer_token->is_valid_to) @@ -943,6 +1022,30 @@ PocoHTTPClientGCPOAuth::BearerToken PocoHTTPClientGCPOAuth::requestBearerTokenFr }; } +PocoHTTPClientGCSHMAC::PocoHTTPClientGCSHMAC(const PocoHTTPClientConfiguration & client_configuration) + : PocoHTTPClient(client_configuration) + , credentials_provider(client_configuration.gcs_hmac_credentials_provider) +{ + if (!credentials_provider) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "PocoHTTPClientGCSHMAC requires a credentials provider (http_client = gcs_hmac wiring bug)"); +} + +void PocoHTTPClientGCSHMAC::makeRequestInternal( + Aws::Http::HttpRequest & request, + std::shared_ptr & response, + Aws::Utils::RateLimits::RateLimiterInterface * readLimiter, + Aws::Utils::RateLimits::RateLimiterInterface * writeLimiter) const +{ + /// Generation semantics only for a marked request; GOOG4 authentication for every request, + /// because this client always signs with Google's native scheme. + if (isNativeConditionalRequest(request)) + applyGcsConditionalDialectToRequest(request); + prepareGcsRequestForGoog4Authentication(request); + signRequestGOOG4(request, credentials_provider->GetAWSCredentials(), std::chrono::system_clock::now()); + PocoHTTPClient::makeRequestInternal(request, response, readLimiter, writeLimiter); +} + } #endif diff --git a/src/IO/S3/PocoHTTPClient.h b/src/IO/S3/PocoHTTPClient.h index 0cda72dc47f2..2614b8daa57f 100644 --- a/src/IO/S3/PocoHTTPClient.h +++ b/src/IO/S3/PocoHTTPClient.h @@ -30,6 +30,11 @@ namespace Aws::Http::Standard class StandardHttpResponse; } +namespace Aws::Auth +{ +class AWSCredentialsProvider; +} + namespace DB { class Context; @@ -78,6 +83,9 @@ struct PocoHTTPClientConfiguration : public Aws::Client::ClientConfiguration HTTPHeaderEntries extra_headers; String http_client; + /// Credentials for the GOOG4-HMAC signer (http_client = gcs_hmac only): the same provider + /// chain the AWS path builds (inline keys, use_environment_credentials, ...). + std::shared_ptr gcs_hmac_credentials_provider; String service_account; String metadata_service; String request_token_path; @@ -87,6 +95,10 @@ struct PocoHTTPClientConfiguration : public Aws::Client::ClientConfiguration /// See PoolBase::BehaviourOnLimit bool s3_use_adaptive_timeouts = true; + /// Conditional PUT (If-None-Match / If-Match) bodies >= this negotiate Expect: 100-continue (B118). + /// `0` (the default) disables it, so non-CAS S3 clients keep upstream behaviour; only a CAS + /// conditional-write client raises it (see the single-attempt client in `ObjectStorageBackend`). + size_t expect_continue_min_bytes = DEFAULT_EXPECT_CONTINUE_MIN_BYTES; size_t http_keep_alive_timeout = DEFAULT_HTTP_KEEP_ALIVE_TIMEOUT; size_t http_keep_alive_max_requests = DEFAULT_HTTP_KEEP_ALIVE_MAX_REQUEST; @@ -229,6 +241,7 @@ class PocoHTTPClient : public Aws::Http::HttpClient const RemoteHostFilter & remote_host_filter; unsigned int s3_max_redirects = DEFAULT_MAX_REDIRECTS; bool s3_use_adaptive_timeouts = true; + size_t expect_continue_min_bytes = DEFAULT_EXPECT_CONTINUE_MIN_BYTES; const UInt64 http_max_fields = 1000000; const UInt64 http_max_field_name_size = 128 * 1024; const UInt64 http_max_field_value_size = 128 * 1024; @@ -273,6 +286,26 @@ class PocoHTTPClientGCPOAuth : public PocoHTTPClient BearerToken requestBearerTokenFromADC() const; }; +/// GCS with HMAC credentials over the XML API, signed with Google's native GOOG4-HMAC-SHA256 — +/// the ONLY way HMAC credentials get enforced conditional semantics on GCS (the S3-compatible +/// sigv4 surface silently ignores If-None-Match / If-Match; measured 2026-07-03). Adapts a +/// `NativeConditional` request, prepares every request for GOOG4 authentication, then signs. +/// Selected by `http_client = gcs_hmac`. +class PocoHTTPClientGCSHMAC : public PocoHTTPClient +{ +public: + explicit PocoHTTPClientGCSHMAC(const PocoHTTPClientConfiguration & client_configuration); + +private: + void makeRequestInternal( + Aws::Http::HttpRequest & request, + std::shared_ptr & response, + Aws::Utils::RateLimits::RateLimiterInterface * readLimiter, + Aws::Utils::RateLimits::RateLimiterInterface * writeLimiter) const override; + + std::shared_ptr credentials_provider; +}; + } #endif diff --git a/src/IO/S3/PocoHTTPClientFactory.cpp b/src/IO/S3/PocoHTTPClientFactory.cpp index 0fb1cf40d93c..0a3a4f94fec4 100644 --- a/src/IO/S3/PocoHTTPClientFactory.cpp +++ b/src/IO/S3/PocoHTTPClientFactory.cpp @@ -14,6 +14,13 @@ namespace DB::S3 { + +bool isNativeConditionalRequest(const Aws::Http::HttpRequest & request) noexcept +{ + const auto * extended_request = dynamic_cast(&request); + return extended_request != nullptr && extended_request->isNativeConditional(); +} + std::shared_ptr PocoHTTPClientFactory::CreateHttpClient(const Aws::Client::ClientConfiguration & client_configuration) const { @@ -23,6 +30,9 @@ PocoHTTPClientFactory::CreateHttpClient(const Aws::Client::ClientConfiguration & if (Poco::toLower(poco_client_configuration.http_client) == "gcp_oauth") return std::make_shared(poco_client_configuration); + if (Poco::toLower(poco_client_configuration.http_client) == "gcs_hmac") + return std::make_shared(poco_client_configuration); + return std::make_shared(poco_client_configuration); } @@ -39,7 +49,7 @@ std::shared_ptr PocoHTTPClientFactory::CreateHttpRequest std::shared_ptr PocoHTTPClientFactory::CreateHttpRequest( const Aws::Http::URI & uri, Aws::Http::HttpMethod method, const Aws::IOStreamFactory &) const { - auto request = Aws::MakeShared("PocoHTTPClientFactory", uri, method); + auto request = Aws::MakeShared("PocoHTTPClientFactory", uri, method); /// Don't create default response stream. Actual response stream will be set later in PocoHTTPClient. request->SetResponseStreamFactory(null_factory); diff --git a/src/IO/S3/PocoHTTPClientFactory.h b/src/IO/S3/PocoHTTPClientFactory.h index 60704332e7b1..1bc2ffc26a25 100644 --- a/src/IO/S3/PocoHTTPClientFactory.h +++ b/src/IO/S3/PocoHTTPClientFactory.h @@ -1,6 +1,7 @@ #pragma once #include +#include namespace Aws::Http { @@ -10,6 +11,27 @@ class HttpRequest; namespace DB::S3 { + +/// The typed HTTP request every `PocoHTTPClientFactory::CreateHttpRequest` overload constructs, +/// carrying the `NativeConditional` bit `Client::BuildHttpRequest` derives on every SDK attempt from +/// the operation wrapper's `RequestWithNativeConditionalMode::isNativeConditional`. A request +/// reaching `PocoHTTPClient` through any other path (e.g. built directly in a test) is foreign and +/// reads as `Default` via `isNativeConditionalRequest`. +class ExtendedHttpRequest final : public Aws::Http::Standard::StandardHttpRequest +{ +public: + using StandardHttpRequest::StandardHttpRequest; + + void setNativeConditional(bool value = true) { native_conditional = value; } + bool isNativeConditional() const { return native_conditional; } + +private: + bool native_conditional = false; +}; + +/// False for a foreign `Aws::Http::HttpRequest` that isn't an `ExtendedHttpRequest`. +bool isNativeConditionalRequest(const Aws::Http::HttpRequest & request) noexcept; + class PocoHTTPClientFactory : public Aws::Http::HttpClientFactory { public: diff --git a/src/IO/S3/Requests.h b/src/IO/S3/Requests.h index 90134daea238..d4d2f3daca85 100644 --- a/src/IO/S3/Requests.h +++ b/src/IO/S3/Requests.h @@ -13,6 +13,7 @@ #include #include #include +#include #include #include #include @@ -61,8 +62,23 @@ inline void setChecksumAlgorithm(R & request) } }; +/// Non-template interface so callers that only see the SDK's `Aws::AmazonWebServiceRequest` base +/// (e.g. `Client::BuildHttpRequest`) can still ask whether a request opted into the typed +/// `NativeConditional` request mode, without knowing which `ExtendedRequest` it is. +class RequestWithNativeConditionalMode +{ +public: + RequestWithNativeConditionalMode() = default; + RequestWithNativeConditionalMode(const RequestWithNativeConditionalMode &) = default; + RequestWithNativeConditionalMode & operator=(const RequestWithNativeConditionalMode &) = default; + RequestWithNativeConditionalMode(RequestWithNativeConditionalMode &&) = default; + RequestWithNativeConditionalMode & operator=(RequestWithNativeConditionalMode &&) = default; + virtual ~RequestWithNativeConditionalMode() = default; + virtual bool isNativeConditional() const = 0; +}; + template -class ExtendedRequest : public BaseRequest +class ExtendedRequest : public BaseRequest, public RequestWithNativeConditionalMode { public: Aws::Endpoint::EndpointParameters GetEndpointContextParams() const override @@ -134,12 +150,20 @@ class ExtendedRequest : public BaseRequest RequestChecksum::setChecksumAlgorithm(*this); } + /// Marks this request as eligible for the typed `NativeConditional` HTTP mode (see + /// `WriteSettings::object_storage_request_mode`). `Client::BuildHttpRequest` re-derives the + /// resulting HTTP bit from this on every SDK attempt, so setting it once here is enough to + /// survive retries and redirects, which each rebuild the HTTP request from scratch. + void setNativeConditional(bool value = true) const { native_conditional = value; } + bool isNativeConditional() const override { return native_conditional; } + protected: mutable std::string region_override; mutable std::optional uri_override; mutable ApiMode api_mode{ApiMode::AWS}; mutable bool checksum = true; bool is_s3express_bucket = false; + mutable bool native_conditional = false; }; class CopyObjectRequest : public ExtendedRequest @@ -158,6 +182,7 @@ using ListObjectsV2Request = ExtendedRequest; using ListObjectsRequest = ExtendedRequest; using GetObjectRequest = ExtendedRequest; using GetObjectTaggingRequest = ExtendedRequest; +using GetBucketVersioningRequest = ExtendedRequest; class UploadPartRequest : public ExtendedRequest { diff --git a/src/IO/S3/copyS3File.cpp b/src/IO/S3/copyS3File.cpp index 73143fbaf52e..794be89d6c42 100644 --- a/src/IO/S3/copyS3File.cpp +++ b/src/IO/S3/copyS3File.cpp @@ -53,6 +53,7 @@ namespace ErrorCodes extern const int S3_ERROR; extern const int INVALID_CONFIG_PARAMETER; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; } namespace S3RequestSetting @@ -235,9 +236,10 @@ namespace } ProfileEvents::increment(ProfileEvents::WriteBufferFromS3RequestsErrors, 1); throw S3Exception( + PreformattedMessage::create("Message: {}, Key: {}, Bucket: {}, Tags: {}", + outcome.GetError().GetMessage(), dest_key, dest_bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")), outcome.GetError().GetErrorType(), - "Message: {}, Key: {}, Bucket: {}, Tags: {}", - outcome.GetError().GetMessage(), dest_key, dest_bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")); + outcome.GetError().GetExceptionName()); } } @@ -626,7 +628,11 @@ namespace ThreadPoolCallbackRunnerUnsafe schedule_, BlobStorageLogWriterPtr blob_storage_log_, std::function fallback_method_, +<<<<<<< HEAD bool is_ranged_copy_) +======= + bool allow_fallback_ = true) +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) : UploadHelper( client_ptr_, dest_bucket_, @@ -645,6 +651,7 @@ namespace , is_ranged_copy(is_ranged_copy_) , read_settings(read_settings_) , fallback_method(std::move(fallback_method_)) + , allow_fallback(allow_fallback_) { } @@ -690,6 +697,7 @@ namespace bool is_ranged_copy; const ReadSettings read_settings; std::function fallback_method; + const bool allow_fallback; void performSingleOperationCopy() { @@ -718,6 +726,7 @@ namespace request.SetContentType("binary/octet-stream"); client_ptr->setKMSHeaders(request); + } void processCopyRequest(S3::CopyObjectRequest & request) @@ -749,6 +758,12 @@ namespace { if (!supports_multipart_copy || outcome.GetError().GetExceptionName() == "AccessDenied") { + if (!allow_fallback) + throw S3Exception( + outcome.GetError().GetMessage(), + outcome.GetError().GetErrorType(), + outcome.GetError().GetExceptionName()); + LOG_INFO( log, "Multipart upload using copy is not supported, will try regular upload for Bucket: {}, Key: {}, Object size: " @@ -788,12 +803,13 @@ namespace } throw S3Exception( + PreformattedMessage::create("Message: {}, Key: {}, Bucket: {}, Object size: {}", + outcome.GetError().GetMessage(), + dest_key, + dest_bucket, + size), outcome.GetError().GetErrorType(), - "Message: {}, Key: {}, Bucket: {}, Object size: {}", - outcome.GetError().GetMessage(), - dest_key, - dest_bucket, - size); + outcome.GetError().GetExceptionName()); } } @@ -808,6 +824,9 @@ namespace if (e.getS3ErrorCode() != Aws::S3::S3Errors::ACCESS_DENIED) throw; + if (!allow_fallback) + throw; + tryLogCurrentException(log, "Multi part copy failed, trying with regular upload"); fallback_method(); } @@ -972,11 +991,50 @@ void copyS3File( const ReadSettings & read_settings, BlobStorageLogWriterPtr blob_storage_log, ThreadPoolCallbackRunnerUnsafe schedule, +<<<<<<< HEAD const CreateReadBuffer & fallback_file_reader, const std::optional & object_metadata) { copyS3FileImpl( std::move(src_s3_client), +======= + const CreateReadBuffer& fallback_file_reader, + const std::optional & object_metadata, + ObjectStorageCopyMode copy_mode) +{ + if (!dest_s3_client) + dest_s3_client = src_s3_client; + + std::function fallback_method = [&] mutable + { + copyDataToS3File( + fallback_file_reader, + src_offset, + src_size, + dest_s3_client, + dest_bucket, + dest_key, + settings, + blob_storage_log, + schedule, + object_metadata); + }; + + if (!settings[S3RequestSetting::allow_native_copy]) + { + if (copy_mode == ObjectStorageCopyMode::NativeOnly) + throw Exception( + ErrorCodes::NOT_IMPLEMENTED, + "Native-only S3 object copy is unavailable because allow_native_copy is disabled"); + + LOG_TRACE(getLogger("copyS3File"), "Native copy is disable for {}", src_key); + fallback_method(); + return; + } + + CopyFileHelper helper{ + src_s3_client, +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) src_bucket, src_key, /* src_offset= */ 0, @@ -991,6 +1049,7 @@ void copyS3File( std::move(schedule), fallback_file_reader, object_metadata, +<<<<<<< HEAD /* is_ranged_copy= */ false); } @@ -1028,6 +1087,13 @@ void copyS3FileRange( fallback_file_reader, object_metadata, /* is_ranged_copy= */ true); +======= + schedule, + blob_storage_log, + std::move(fallback_method), + /*allow_fallback=*/copy_mode == ObjectStorageCopyMode::Default}; + helper.performCopy(); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/IO/S3/copyS3File.h b/src/IO/S3/copyS3File.h index 9997f637c2a8..3854fe508e73 100644 --- a/src/IO/S3/copyS3File.h +++ b/src/IO/S3/copyS3File.h @@ -5,6 +5,7 @@ #if USE_AWS_S3 #include +#include #include #include #include @@ -38,6 +39,9 @@ std::unique_ptr createS3UploadBody( /// (copyDataToS3File()). /// /// read_settings - is used for throttling in case of native copy is not possible +/// +/// `copy_mode = NativeOnly` forbids the client-side read-write fallback. If native copy is disabled +/// or cannot complete, the failure is propagated instead. void copyS3File( std::shared_ptr src_s3_client, const String & src_bucket, @@ -50,6 +54,7 @@ void copyS3File( const ReadSettings & read_settings, BlobStorageLogWriterPtr blob_storage_log, ThreadPoolCallbackRunnerUnsafe schedule, +<<<<<<< HEAD const CreateReadBuffer & fallback_file_reader, const std::optional & object_metadata = std::nullopt); @@ -75,6 +80,11 @@ void copyS3FileRange( ThreadPoolCallbackRunnerUnsafe schedule, const CreateReadBuffer & fallback_file_reader, const std::optional & object_metadata = std::nullopt); +======= + const CreateReadBuffer& fallback_file_reader, + const std::optional & object_metadata = std::nullopt, + ObjectStorageCopyMode copy_mode = ObjectStorageCopyMode::Default); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// Copies data from any seekable source to S3. /// The same functionality can be done by using the function copyData() and the class WriteBufferFromS3 diff --git a/src/IO/S3/getObjectInfo.cpp b/src/IO/S3/getObjectInfo.cpp index deec76d3fdc6..63aeb736960c 100644 --- a/src/IO/S3/getObjectInfo.cpp +++ b/src/IO/S3/getObjectInfo.cpp @@ -24,7 +24,8 @@ namespace const S3::Client & client, const String & bucket, const String & key, - const String & version_id) + const String & version_id, + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default) { ProfileEvents::increment(ProfileEvents::S3HeadObject); if (client.isClientForDisk()) @@ -38,6 +39,8 @@ namespace if (!version_id.empty()) req.SetVersionId(version_id); + req.setNativeConditional(request_mode == ObjectStorageRequestMode::NativeConditional); + return client.HeadObject(req); } @@ -67,9 +70,10 @@ namespace const String & key, const String & version_id, bool with_metadata, - bool with_tags) + bool with_tags, + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default) { - auto outcome = headObject(client, bucket, key, version_id); + auto outcome = headObject(client, bucket, key, version_id, request_mode); if (!outcome.IsSuccess()) return {std::nullopt, outcome.GetError()}; @@ -141,11 +145,12 @@ ObjectInfo getObjectInfoIfExists( const String & key, const String & version_id, bool with_metadata, - bool with_tags) + bool with_tags, + ObjectStorageRequestMode request_mode) { Expect404ResponseScope scope; // 404 is not an error - auto [object_info, error] = tryGetObjectInfo(client, bucket, key, version_id, with_metadata, with_tags); + auto [object_info, error] = tryGetObjectInfo(client, bucket, key, version_id, with_metadata, with_tags, request_mode); if (object_info) return *object_info; diff --git a/src/IO/S3/getObjectInfo.h b/src/IO/S3/getObjectInfo.h index 314d81c6daa5..60683cbf3abf 100644 --- a/src/IO/S3/getObjectInfo.h +++ b/src/IO/S3/getObjectInfo.h @@ -6,6 +6,7 @@ #include #include #include +#include namespace DB::S3 { @@ -22,13 +23,16 @@ struct ObjectInfo }; /// Ignore if object does not exist +/// `request_mode` marks the HEAD wrapper as eligible for the typed NativeConditional request mode +/// (see ObjectStorageRequestMode); the client's HTTP layer decides whether it actually takes effect. ObjectInfo getObjectInfoIfExists( const S3::Client & client, const String & bucket, const String & key, const String & version_id = {}, bool with_metadata = false, - bool with_tags = false); + bool with_tags = false, + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default); ObjectInfo getObjectInfo( const S3::Client & client, diff --git a/src/IO/S3/tests/gtest_aws_s3_client.cpp b/src/IO/S3/tests/gtest_aws_s3_client.cpp index ea476825dafa..61edec9e1ceb 100644 --- a/src/IO/S3/tests/gtest_aws_s3_client.cpp +++ b/src/IO/S3/tests/gtest_aws_s3_client.cpp @@ -6,8 +6,11 @@ #if USE_AWS_S3 +#include #include +#include #include +#include #include #include @@ -20,16 +23,23 @@ #include #include #include +#include #include +#include +#include +#include +#include #include #include #include #include #include +#include #include #include #include +#include #include #include #include @@ -42,6 +52,11 @@ namespace DB::S3RequestSetting extern const S3RequestSettingsUInt64 max_unexpected_write_error_retries; } +namespace ProfileEvents +{ + extern const Event S3SingleAttemptRetryConsultations; +} + /* * When all tests are executed together, `Context::getGlobalContextInstance()` is not null. Global context is used by * ProxyResolvers to get proxy configuration (used by S3 clients). If global context does not have a valid ConfigRef, it relies on @@ -198,6 +213,202 @@ static void testServerSideEncryption( EXPECT_EQ(content, expected_headers); } +TEST(IOTestAwsS3Client, DoesNotRetryPreconditionFailed) +{ + /// B166: a 412 Precondition Failed (conditional CAS/dedup writes of the content-addressed + /// backend) must NOT be retried, even when the SDK marks it retryable because an S3-compatible + /// server (e.g. RustFS) returned a body whose ExceptionName it could not parse. Retrying it is a + /// storm that stalls the write path. + DB::S3::Client::RetryStrategy strategy(DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 10}); + + Aws::Client::AWSError precondition(Aws::Client::CoreErrors::UNKNOWN, /*isRetryable=*/true); + precondition.SetResponseCode(Aws::Http::HttpResponseCode::PRECONDITION_FAILED); + EXPECT_FALSE(strategy.ShouldRetry(precondition, /*attemptedRetries=*/0)); + EXPECT_TRUE(DB::S3::isPreconditionFailedError(precondition)); // one policy: agrees via response code + + /// A genuinely transient error is still retried (the guard is specific to 412). + Aws::Client::AWSError unavailable(Aws::Client::CoreErrors::SLOW_DOWN, /*isRetryable=*/true); + unavailable.SetResponseCode(Aws::Http::HttpResponseCode::SERVICE_UNAVAILABLE); + EXPECT_TRUE(strategy.ShouldRetry(unavailable, /*attemptedRetries=*/0)); + EXPECT_FALSE(DB::S3::isPreconditionFailedError(unavailable)); + + /// The one 412 policy also matches on the canonical name / raw body (the two CA conditional + /// ops see an error whose ExceptionName the SDK DID parse, or whose body carries the token). + Aws::Client::AWSError named(Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed", "precondition failed", false); + EXPECT_TRUE(DB::S3::isPreconditionFailedError(named)); + + /// Typed-exception surface consumed by S3 request finalization: name and message. + EXPECT_TRUE(DB::S3Exception("boom", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed").isPreconditionFailed()); + EXPECT_FALSE(DB::S3Exception("boom", Aws::S3::S3Errors::NO_SUCH_KEY, "NoSuchKey").isPreconditionFailed()); +} + +/// Every consultation is counted, not just the first: simulating two retryable 5xx decisions in a row +/// proves the counter tracks each SDK consultation rather than being fixed/clamped at 1, which is what +/// makes it a live tripwire ("SDK-level retries must remain zero for conditional writes") rather than a +/// value nothing ever touches. +TEST(IOTestAwsS3Client, SingleAttemptRetryStrategyRefusesAndCounts) +{ + using ProfileEvents::global_counters; + const auto before = global_counters[ProfileEvents::S3SingleAttemptRetryConsultations].load(); + DB::S3::SingleAttemptRetryStrategy strategy; + const Aws::Client::AWSError retryable_5xx( + Aws::Client::CoreErrors::INTERNAL_FAILURE, /*isRetryable=*/true); + EXPECT_FALSE(strategy.ShouldRetry(retryable_5xx, /*attempted=*/0)); + EXPECT_FALSE(strategy.ShouldRetry(retryable_5xx, /*attempted=*/1)); + EXPECT_EQ(strategy.GetMaxAttempts(), 1); + EXPECT_EQ(global_counters[ProfileEvents::S3SingleAttemptRetryConsultations].load() - before, 2u); +} + +struct ConditionalPutWireObservation +{ + bool negotiated_expect_continue = false; + bool has_if_none_match = false; + bool has_generation_match = false; + std::string generation_match; +}; + +/// Drive a single-part conditional PUT (`If-None-Match: *`) with `body_size` bytes through a real +/// S3 client whose `expect_continue_min_bytes` gate is `threshold`, against the mock HTTP server, and +/// report what the request that reached the wire carried. `http_client` selects the GCS-mode client +/// to exercise; `request_mode` decides whether that client sees the write as native-conditional. +static ConditionalPutWireObservation observeConditionalPut( + uint64_t threshold, + size_t body_size, + const std::string & http_client = "", + DB::ObjectStorageRequestMode request_mode = DB::ObjectStorageRequestMode::Default) +{ + TestPocoHTTPServer http; + + DB::RemoteHostFilter remote_host_filter; + DB::S3::URI uri(http.getUrl() + "/IOTestAwsS3ClientExpectContinue/test.txt"); + + DB::S3::PocoHTTPClientConfiguration client_configuration = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ true, + /* s3_slow_all_threads_after_retryable_error = */ true, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ false, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}, + uri.uri.getScheme()); + + client_configuration.endpointOverride = uri.endpoint; + client_configuration.expect_continue_min_bytes = threshold; + client_configuration.http_client = http_client; + + DB::S3::ClientSettings client_settings{ + .use_virtual_addressing = uri.is_virtual_hosted_style, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; + + std::shared_ptr client = DB::S3::ClientFactory::instance().create( + client_configuration, + client_settings, + /* access_key_id = */ "ACCESS_KEY_ID", + /* secret_access_key = */ "SECRET_ACCESS_KEY", + /* server_side_encryption_customer_key_base64 = */ "", + /* sse_kms_config = */ {}, + /* headers = */ {}, + DB::S3::CredentialsConfiguration{ + .use_environment_credentials = false, + .use_insecure_imds_request = false, + }); + + DB::S3::S3RequestSettings request_settings; + request_settings[DB::S3RequestSetting::max_unexpected_write_error_retries] = 1; + + DB::WriteSettings write_settings; + write_settings.object_storage_write_if_none_match = "*"; + write_settings.object_storage_request_mode = request_mode; + + DB::WriteBufferFromS3 write_buffer( + client, + uri.bucket, + uri.key, + DB::DBMS_DEFAULT_BUFFER_SIZE, + request_settings, + /* blob_log = */ nullptr, + /* object_metadata = */ std::nullopt, + /* schedule = */ {}, + write_settings); + + const std::string body(body_size, 'x'); + write_buffer.write(body.data(), body.size()); + write_buffer.finalize(); + + const auto & header = http.getLastRequestHeader(); + ConditionalPutWireObservation observed; + observed.negotiated_expect_continue = header.has("Expect"); + observed.has_if_none_match = header.has("if-none-match"); + observed.has_generation_match = header.has("x-goog-if-generation-match"); + if (observed.has_generation_match) + observed.generation_match = header.get("x-goog-if-generation-match"); + return observed; +} + +static bool conditionalPutNegotiatesExpectContinue(uint64_t threshold, size_t body_size) +{ + return observeConditionalPut(threshold, body_size).negotiated_expect_continue; +} + +TEST(IOTestAwsS3Client, ExpectContinueOnlyWhenThresholdPositive) +{ + /// RExpect: `Expect: 100-continue` (B118) is scoped to CAS-owned conditional writes. A non-CAS S3 + /// client carries the default threshold 0 (disabled) and must NOT negotiate Expect on a conditional + /// PUT — that is the upstream wire behaviour a non-CAS disk (e.g. Iceberg's If-None-Match commits) + /// must keep. A CAS conditional-write client raises the threshold (see the single-attempt client in + /// ObjectStorageBackend) and DOES negotiate it for a body at least that large. + EXPECT_FALSE(conditionalPutNegotiatesExpectContinue(/*threshold=*/0, /*body_size=*/64)); + EXPECT_TRUE(conditionalPutNegotiatesExpectContinue(/*threshold=*/8, /*body_size=*/64)); + /// A positive threshold still excludes a body below it (only large bodies warrant the round-trip). + EXPECT_FALSE(conditionalPutNegotiatesExpectContinue(/*threshold=*/128, /*body_size=*/64)); +} + +/// The GCS-mode clients translate conditions before delegating to the common HTTP boundary, so the +/// `Expect: 100-continue` gate — which lives at that boundary and recognises +/// `x-goog-if-generation-match` alongside the standard headers — still sees the condition either way. +/// Both requests below use the same endpoint, a mock server with no `storage.googleapis.com` in its +/// hostname, so nothing here is endpoint-sniffed. +TEST(IOTestAwsS3Client, GcsHmacTranslatesConditionsOnlyWhenMarkedAndKeepsExpectGate) +{ + const auto native = observeConditionalPut( + /*threshold=*/8, /*body_size=*/64, "gcs_hmac", DB::ObjectStorageRequestMode::NativeConditional); + EXPECT_TRUE(native.has_generation_match); + EXPECT_EQ(native.generation_match, "0"); + EXPECT_FALSE(native.has_if_none_match); + EXPECT_TRUE(native.negotiated_expect_continue); + + /// A Default request through the very same client keeps the standard ETag precondition, and the + /// threshold semantics are unchanged by which form the condition took. + const auto standard = observeConditionalPut( + /*threshold=*/8, /*body_size=*/64, "gcs_hmac", DB::ObjectStorageRequestMode::Default); + EXPECT_TRUE(standard.has_if_none_match); + EXPECT_FALSE(standard.has_generation_match); + EXPECT_TRUE(standard.negotiated_expect_continue); + + /// The pre-existing body-size gate still applies to both forms. + EXPECT_FALSE(observeConditionalPut( + /*threshold=*/128, /*body_size=*/64, "gcs_hmac", DB::ObjectStorageRequestMode::NativeConditional) + .negotiated_expect_continue); + EXPECT_FALSE(observeConditionalPut( + /*threshold=*/128, /*body_size=*/64, "gcs_hmac", DB::ObjectStorageRequestMode::Default) + .negotiated_expect_continue); +} + +/// A `Default` PUT through the GOOG4 client must survive the authentication allowlist: whatever +/// `x-amz-*` headers the SDK puts on an ordinary write have to be translated or consumed, never +/// rejected. This is the ordinary-traffic regression the allowlist could break. +TEST(IOTestAwsS3Client, GcsHmacDefaultPutPassesTheAuthenticationAllowlist) +{ + EXPECT_NO_THROW(observeConditionalPut( + /*threshold=*/0, /*body_size=*/64, "gcs_hmac", DB::ObjectStorageRequestMode::Default)); +} + TEST(IOTestAwsS3Client, AppendExtraSSECHeadersRead) { /// See https://github.com/ClickHouse/ClickHouse/pull/19748 @@ -769,6 +980,717 @@ TEST(IOTestAwsS3Client, WebIdentityConfiguredFromKmsRoleOverrideAndTokenFile) "arn:aws:iam::123456789012:role/from_kms_role_arn_override")); } +namespace +{ + +/// Builds a real `DB::S3::Client` with the given `http_client` value, wired the same way +/// `ClientFactory::create` wires disk configuration, but never sent over the wire: these tests only +/// exercise `Client::BuildHttpRequest`, which does no I/O. +std::unique_ptr makeClientWithHttpClient(const std::string & http_client) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::URI uri("https://storage.googleapis.com/bucket/key"); + + DB::S3::PocoHTTPClientConfiguration client_configuration = DB::S3::ClientFactory::instance().createClientConfiguration( + /*force_region=*/"us-east-1", + remote_host_filter, + /*s3_max_redirects=*/100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /*s3_slow_all_threads_after_network_error=*/true, + /*s3_slow_all_threads_after_retryable_error=*/true, + /*enable_s3_requests_logging=*/false, + /*for_disk_s3=*/false, + /*opt_disk_name=*/{}, + /*request_throttler=*/{}, + uri.uri.getScheme()); + + client_configuration.endpointOverride = uri.endpoint; + client_configuration.http_client = http_client; + + DB::S3::ClientSettings client_settings{ + .use_virtual_addressing = uri.is_virtual_hosted_style, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; + + return DB::S3::ClientFactory::instance().create( + client_configuration, + client_settings, + /*access_key_id=*/"ACCESS_KEY_ID", + /*secret_access_key=*/"SECRET_ACCESS_KEY", + /*server_side_encryption_customer_key_base64=*/"", + /*sse_kms_config=*/{}, + /*headers=*/{}, + DB::S3::CredentialsConfiguration{ + .use_environment_credentials = false, + .use_insecure_imds_request = false, + }); +} + +/// A minimal scripted HTTP server: the Nth request it receives is answered with +/// `responses[min(N, responses.size() - 1)]`, so a short script (e.g. one retryable response then one +/// success) naturally "then always succeeds" once it runs out. Lets a test drive one genuine SDK-level +/// retry through a real `DB::S3::Client`, rather than standing in for the SDK's own per-attempt +/// behaviour by calling the same functions twice by hand. +struct ScriptedResponse +{ + Poco::Net::HTTPResponse::HTTPStatus status; + std::vector> headers; +}; + +/// One real request as it reached the wire: method plus every header, captured before the scripted +/// response is sent. Lets a test drive several real SDK calls (e.g. CreateMultipartUpload, UploadPart, +/// CompleteMultipartUpload) against one server and inspect what each one actually carried, rather than +/// only the single most-recent request `TestPocoHTTPServer` keeps. +struct CapturedRequest +{ + std::string method; + Poco::Net::MessageHeader headers; +}; + +class ScriptedResponseServer +{ +public: + explicit ScriptedResponseServer(std::vector responses_) + : responses(std::move(responses_)) + , server_socket(std::make_unique(0)) + , handler_factory(new Factory(*this)) + , server_params(new Poco::Net::HTTPServerParams()) + , server(std::make_unique(handler_factory, *server_socket, server_params)) + { + server->start(); + } + + /// `server_socket->address()` is the wildcard bind address (`0.0.0.0:PORT`), which is not a usable + /// connection target and could silently conflate distinct servers under the same host string. Build + /// the URL from an explicit loopback address plus the bound port instead. + std::string getUrl() const { return "http://127.0.0.1:" + std::to_string(server_socket->address().port()); } + + /// Requests in arrival order. The tests using this all drive their calls sequentially against a + /// single-threaded client, so no concurrent capture ever races with a concurrent read here. + const std::vector & getCapturedRequests() const { return captured_requests; } + +private: + class Handler : public Poco::Net::HTTPRequestHandler + { + public: + explicit Handler(ScriptedResponseServer & owner_) : owner(owner_) { } + + void handleRequest(Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) override + { + owner.captured_requests.push_back(CapturedRequest{request.getMethod(), request}); + + /// The connection is kept alive across requests (the SDK reuses it for the multipart/batch + /// sequences this server now handles), so an unread request body left in the socket buffer + /// corrupts the next request's parse -- its bytes prepend the following request line. Every + /// request body must be drained here even though nothing needs its content. + request.stream().ignore(std::numeric_limits::max()); + + const size_t index = owner.request_count.fetch_add(1); + const auto & scripted = owner.responses[std::min(index, owner.responses.size() - 1)]; + response.setStatus(scripted.status); + for (const auto & [name, value] : scripted.headers) + response.set(name, value); + response.setContentLength(0); + response.send(); + } + + private: + ScriptedResponseServer & owner; + }; + + class Factory : public Poco::Net::HTTPRequestHandlerFactory + { + public: + explicit Factory(ScriptedResponseServer & owner_) : owner(owner_) { } + Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override { return new Handler(owner); } + + private: + ScriptedResponseServer & owner; + }; + + std::vector responses; + std::atomic request_count{0}; + std::vector captured_requests; + std::unique_ptr server_socket; + Poco::SharedPtr handler_factory; + Poco::AutoPtr server_params; + std::unique_ptr server; +}; + +/// `Client::BuildHttpRequest` is not `final`, and the protected constructor `Client` exposes is +/// commented "visible for testing" — this subclass uses exactly that seam to observe every real +/// `BuildHttpRequest` call the vendored SDK makes for a genuine attempt, without adding any +/// observability to production code (the mode has no wire representation by design, so there is no +/// other way to see it from outside the process). +class RecordingClient : public DB::S3::Client +{ +public: + RecordingClient( + size_t max_redirects_, + DB::S3::ServerSideEncryptionKMSConfig sse_kms_config_, + const std::shared_ptr & credentials_provider_, + const DB::S3::PocoHTTPClientConfiguration & client_configuration_, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy sign_payloads_, + const DB::S3::ClientSettings & client_settings_) + : DB::S3::Client(max_redirects_, std::move(sse_kms_config_), credentials_provider_, client_configuration_, sign_payloads_, client_settings_) + { + } + + /// One entry per real `BuildHttpRequest` call, i.e. one per genuine SDK attempt. + mutable std::vector observed_native_conditional; + + void BuildHttpRequest(const Aws::AmazonWebServiceRequest & request, const std::shared_ptr & httpRequest) const override + { + DB::S3::Client::BuildHttpRequest(request, httpRequest); + observed_native_conditional.push_back(DB::S3::isNativeConditionalRequest(*httpRequest)); + } +}; + +std::unique_ptr makeRecordingClient( + const std::string & endpoint, unsigned int max_retries, unsigned int max_redirects, const std::string & http_client = "gcs_hmac") +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::URI uri(endpoint + "/bucket"); + + DB::S3::PocoHTTPClientConfiguration client_configuration = DB::S3::ClientFactory::instance().createClientConfiguration( + /*force_region=*/"us-east-1", + remote_host_filter, + max_redirects, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = max_retries}, + /*s3_slow_all_threads_after_network_error=*/false, + /*s3_slow_all_threads_after_retryable_error=*/false, + /*enable_s3_requests_logging=*/false, + /*for_disk_s3=*/false, + /*opt_disk_name=*/{}, + /*request_throttler=*/{}, + uri.uri.getScheme()); + + client_configuration.endpointOverride = uri.endpoint; + /// The default `gcs_hmac`, not `gcp_oauth`: both make `supportsGcsNativeConditionalRequests()` + /// true, but `gcp_oauth` fetches a bearer token from the GCE metadata server on every real request + /// (`PocoHTTPClientGCPOAuth::requestBearerToken`) -- a real network call this test cannot make. + /// `gcs_hmac` signs locally from the credentials handed to it below, no token fetch involved. An + /// empty `http_client` selects the ordinary (non-GCS) HMAC path instead, wired the same way as a + /// plain S3-compatible disk -- it never invokes either GCS client class. + client_configuration.http_client = http_client; + /// `ClientFactory::create` would clamp this to 1 when `s3_slow_all_threads_after_retryable_error` + /// is set (external retry coordination); here we want the SDK's own retry loop to actually run. + client_configuration.retryStrategy = std::make_shared(client_configuration.retry_strategy); + + DB::S3::ClientSettings client_settings{ + .use_virtual_addressing = uri.is_virtual_hosted_style, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; + + Aws::Auth::AWSCredentials credentials("ACCESS_KEY_ID", "SECRET_ACCESS_KEY"); + auto credentials_provider = DB::S3::getCredentialsProvider( + client_configuration, + credentials, + DB::S3::CredentialsConfiguration{.use_environment_credentials = false, .use_insecure_imds_request = false}); + /// `PocoHTTPClientGCSHMAC`'s constructor throws `LOGICAL_ERROR` without this -- `ClientFactory::create` + /// wires it the same way for the real `gcs_hmac` path. + client_configuration.gcs_hmac_credentials_provider = credentials_provider; + + return std::make_unique( + max_redirects, + DB::S3::ServerSideEncryptionKMSConfig{}, + credentials_provider, + client_configuration, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + client_settings); +} + +} + +TEST(IOTestAwsS3Client, RequestModeDefaultsToDefault) +{ + DB::WriteSettings settings; + EXPECT_EQ(settings.object_storage_request_mode, DB::ObjectStorageRequestMode::Default); +} + +TEST(IOTestAwsS3Client, FactoryAlwaysCreatesExtendedHttpRequest) +{ + DB::S3::PocoHTTPClientFactory factory; + const Aws::IOStreamFactory stream_factory = [] { return nullptr; }; + + auto from_string_uri = factory.CreateHttpRequest( + Aws::String("http://localhost/bucket/key"), Aws::Http::HttpMethod::HTTP_GET, stream_factory); + ASSERT_TRUE(from_string_uri); + EXPECT_TRUE(dynamic_cast(from_string_uri.get())); + + auto from_uri = factory.CreateHttpRequest( + Aws::Http::URI("http://localhost/bucket/key"), Aws::Http::HttpMethod::HTTP_PUT, stream_factory); + ASSERT_TRUE(from_uri); + EXPECT_TRUE(dynamic_cast(from_uri.get())); +} + +TEST(IOTestAwsS3Client, NativeConditionalModeRequiresExplicitGcsHttpClient) +{ + EXPECT_TRUE(makeClientWithHttpClient("gcp_oauth")->supportsGcsNativeConditionalRequests()); + EXPECT_TRUE(makeClientWithHttpClient("gcs_hmac")->supportsGcsNativeConditionalRequests()); + /// The comparison is case-insensitive, matching how `ClientFactory::create` already lower-cases + /// this same field before dispatching on it. + EXPECT_TRUE(makeClientWithHttpClient("GCS_HMAC")->supportsGcsNativeConditionalRequests()); + EXPECT_FALSE(makeClientWithHttpClient("")->supportsGcsNativeConditionalRequests()); + EXPECT_FALSE(makeClientWithHttpClient("some_other_client")->supportsGcsNativeConditionalRequests()); +} + +TEST(IOTestAwsS3Client, ForeignHttpRequestReadsAsDefault) +{ + Aws::Http::Standard::StandardHttpRequest foreign_request(Aws::Http::URI("http://localhost/x"), Aws::Http::HttpMethod::HTTP_GET); + EXPECT_FALSE(DB::S3::isNativeConditionalRequest(foreign_request)); +} + +TEST(IOTestAwsS3Client, NativeConditionalStaysFalseThroughBuildHttpRequestOnNonGcsClient) +{ + /// Closes a coverage gap: nothing else in this file drives `Client::BuildHttpRequest` with a + /// native-marked request against a non-GCS `http_client`. Without this, dropping or inverting the + /// `&& supportsGcsNativeConditionalRequests()` conjunct would not fail any test here, even though + /// that conjunct is exactly what keeps the HTTP bit false for AWS-compatible CAS requests. + auto client = makeClientWithHttpClient("some_other_client"); + ASSERT_FALSE(client->supportsGcsNativeConditionalRequests()); + + DB::S3::PutObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + request.setNativeConditional(true); + + DB::S3::PocoHTTPClientFactory factory; + const Aws::IOStreamFactory stream_factory = [] { return nullptr; }; + auto http_request = factory.CreateHttpRequest( + Aws::Http::URI("https://s3.amazonaws.com/bucket/key"), Aws::Http::HttpMethod::HTTP_PUT, stream_factory); + client->BuildHttpRequest(request, http_request); + EXPECT_FALSE(DB::S3::isNativeConditionalRequest(*http_request)); +} + +TEST(IOTestAwsS3Client, NativeConditionalModeIsRederivedOnEverySdkAttempt) +{ + /// Drives real `DB::S3::Client::GetBucketVersioning` calls through a `RecordingClient`, which + /// records `isNativeConditionalRequest` on every real `Client::BuildHttpRequest` call -- i.e. once + /// per genuine SDK attempt, including the extra attempt a real SDK-level retry triggers. + /// + /// This does not separately drive a real 301 redirect: in the vendored SDK, `AttemptExhaustively` + /// recreates the HTTP request unconditionally at the retry tail regardless of cause + /// (`contrib/aws/src/aws-cpp-sdk-core/source/client/AWSClient.cpp:405`), and `BuildHttpRequest` runs + /// at the top of the next `AttemptOneRequest` exactly as in the retry case (`AWSClient.cpp:564`) -- + /// a redirect only changes the URI passed into that same recreation, it is not a separate mechanism. + /// `Client::doRequest`'s own manual redirect loop is even less in doubt: it re-enters `MakeRequest` + /// wholesale, which calls `BuildHttpRequest` fresh by construction. So the retry case below already + /// exercises the machinery a redirect would use. + auto runOnce = [](const std::string & endpoint, bool native_conditional) -> std::vector + { + /// Note: `ASSERT_*` cannot be used in this lambda -- it returns `std::vector`, not + /// `void`, and the macro expands to a bare `return;` on failure. `EXPECT_*` only records. + auto client = makeRecordingClient(endpoint, /*max_retries=*/2, /*max_redirects=*/2); + EXPECT_TRUE(client->supportsGcsNativeConditionalRequests()); + + DB::S3::GetBucketVersioningRequest request; + request.SetBucket("bucket"); + request.setNativeConditional(native_conditional); + + auto outcome = client->GetBucketVersioning(request); + EXPECT_TRUE(outcome.IsSuccess()); + return client->observed_native_conditional; + }; + + { + SCOPED_TRACE("ordinary request: one successful attempt, mode stays false"); + ScriptedResponseServer server({{Poco::Net::HTTPResponse::HTTP_OK, {}}}); + const auto observed = runOnce(server.getUrl(), /*native_conditional=*/false); + ASSERT_EQ(observed.size(), 1u); + EXPECT_FALSE(observed[0]); + } + + { + SCOPED_TRACE("native request through a genuine SDK-level retry: both attempts see the mode"); + ScriptedResponseServer server({ + {Poco::Net::HTTPResponse::HTTP_INTERNAL_SERVER_ERROR, {}}, + {Poco::Net::HTTPResponse::HTTP_OK, {}}, + }); + const auto observed = runOnce(server.getUrl(), /*native_conditional=*/true); + /// The size assertion is load-bearing, not cosmetic: a 500 that was silently not retried (or a + /// retry that reused a stale HTTP request) would leave a one-element vector, and an + /// all-elements-true assertion alone would not catch that. + ASSERT_EQ(observed.size(), 2u); + EXPECT_TRUE(observed[0]); + EXPECT_TRUE(observed[1]); + } + + { + SCOPED_TRACE("ordinary again: the mode does not leak from a previous native call"); + ScriptedResponseServer server({{Poco::Net::HTTPResponse::HTTP_OK, {}}}); + const auto observed = runOnce(server.getUrl(), /*native_conditional=*/false); + ASSERT_EQ(observed.size(), 1u); + EXPECT_FALSE(observed[0]); + } +} + +/// Response adaptation is gated on the same typed bit as the request side. Both HEADs below get an +/// identical response — a GCS-style one carrying both an ETag and a generation — so the only thing +/// that can produce different results is the mode. +TEST(IOTestAwsS3Client, ResponseGenerationAndMetadataAdaptedOnlyWhenMarked) +{ + const std::vector script{{Poco::Net::HTTPResponse::HTTP_OK, { + {"ETag", "\"6654c734ccab8f440ff0825eb443dc7f\""}, + {"x-goog-generation", "1783078552147137"}, + {"x-goog-meta-cas-envelope", "v1"}, + }}}; + + auto headOnce = [&script](bool native_conditional) + { + ScriptedResponseServer server(script); + auto client = makeRecordingClient(server.getUrl(), /*max_retries=*/0, /*max_redirects=*/0); + + DB::S3::HeadObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + request.setNativeConditional(native_conditional); + return client->HeadObject(request); + }; + + { + SCOPED_TRACE("marked: the generation becomes the SDK-visible ETag and the metadata crosses over"); + auto outcome = headOnce(/*native_conditional=*/true); + ASSERT_TRUE(outcome.IsSuccess()); + EXPECT_EQ(outcome.GetResult().GetETag(), "\"1783078552147137\""); + const auto & metadata = outcome.GetResult().GetMetadata(); + ASSERT_TRUE(metadata.contains("cas-envelope")); + EXPECT_EQ(metadata.at("cas-envelope"), "v1"); + } + + { + SCOPED_TRACE("Default: the upstream ETag survives even though a generation is present"); + auto outcome = headOnce(/*native_conditional=*/false); + ASSERT_TRUE(outcome.IsSuccess()); + EXPECT_EQ(outcome.GetResult().GetETag(), "\"6654c734ccab8f440ff0825eb443dc7f\""); + EXPECT_FALSE(outcome.GetResult().GetMetadata().contains("cas-envelope")); + } +} + +/// Pins the ordinary S3-interoperability HMAC path (`http_client` left empty, exactly as configured for +/// a plain S3-compatible disk): it never becomes a `PocoHTTPClientGCPOAuth` or `PocoHTTPClientGCSHMAC`, +/// so none of the GCS request/response adaptation in `GCSConditionalDialect.cpp` is even reachable from +/// it, CAS or no CAS. `native_conditional=true` is still passed on the HEAD below to prove that even a +/// caller that mismarks a request cannot make this client honour it -- `supportsGcsNativeConditionalRequests` +/// already gates the request-side bit off (see `NativeConditionalModeRequiresExplicitGcsHttpClient` / +/// `NativeConditionalStaysFalseThroughBuildHttpRequestOnNonGcsClient`), and this closes the matching gap +/// on the response side: this would fail if `applyGcsConditionalDialectToResponse` were ever hoisted out +/// of the two GCS subclasses into the shared `PocoHTTPClient::makeRequestInternal`. +TEST(IOTestAwsS3Client, OrdinaryHmacClientNeverAppliesGcsAdaptation) +{ + const std::vector script{{Poco::Net::HTTPResponse::HTTP_OK, { + {"ETag", "\"deadbeefcafebabe0000000000000001\""}, + {"x-goog-generation", "1234567890123456"}, + {"x-goog-meta-cas-envelope", "v1"}, + }}}; + ScriptedResponseServer server(script); + auto client = makeRecordingClient(server.getUrl(), /*max_retries=*/0, /*max_redirects=*/0, /*http_client=*/""); + + DB::S3::HeadObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + request.setNativeConditional(true); + auto outcome = client->HeadObject(request); + + ASSERT_TRUE(outcome.IsSuccess()); + EXPECT_EQ(outcome.GetResult().GetETag(), "\"deadbeefcafebabe0000000000000001\""); + EXPECT_FALSE(outcome.GetResult().GetMetadata().contains("cas-envelope")); + + ASSERT_EQ(client->observed_native_conditional.size(), 1u); + EXPECT_FALSE(client->observed_native_conditional[0]); +} + +/// Drives real `PutObject`, `CopyObject`, `DeleteObject`, and batch `DeleteObjects` requests through the +/// same ordinary (non-GCS) HMAC client and inspects the literal wire headers. Every assertion here is +/// falsifiable by a concrete regression: `EXPECT_TRUE(... has ...)` on an `x-amz-*` name fails if that +/// header were ever renamed or dropped (e.g. by widening the GOOG4 allowlist's reach, or applying +/// `renameToGoogPrefix` outside the two GCS clients), and the SigV4 `EXPECT_TRUE(starts_with(...))` +/// checks fail if Bearer or GOOG4 authentication ever became reachable from a client with no +/// `http_client` configured. +TEST(IOTestAwsS3Client, OrdinaryHmacRequestsKeepUpstreamHeadersAndAuth) +{ + ScriptedResponseServer server({{Poco::Net::HTTPResponse::HTTP_OK, {}}}); + auto client = makeRecordingClient(server.getUrl(), /*max_retries=*/0, /*max_redirects=*/0, /*http_client=*/""); + const auto & captured = server.getCapturedRequests(); + + { + SCOPED_TRACE("PUT with x-amz-meta-*"); + DB::S3::PutObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + request.AddMetadata("cas-envelope", "v1"); + /// `SetContentLength` explicitly, matching every production caller (e.g. `copyS3File.cpp`'s + /// `fillPutRequest`) for stylistic consistency -- the SDK computes it from the body itself when + /// omitted (`ExtendedRequest::IsStreaming` is always `false`, so the chunked path never engages), + /// so this call is not load-bearing for the keep-alive corruption below. That corruption's sole + /// cause is `ScriptedResponseServer` not draining the request body before responding; see the + /// fix in `ScriptedResponseServer::Handler::handleRequest`. + request.SetContentLength(7); + request.SetBody(Aws::MakeShared("gtest", "payload")); + client->PutObject(request); + + ASSERT_EQ(captured.size(), 1u); + EXPECT_TRUE(captured[0].headers.has("x-amz-meta-cas-envelope")); + EXPECT_FALSE(captured[0].headers.has("x-goog-meta-cas-envelope")); + EXPECT_TRUE(captured[0].headers.get("authorization", "").starts_with("AWS4-HMAC-SHA256")); + /// An earlier version of this assertion claimed the SDK's default checksum is always present; + /// a real run showed that is wrong. The conclusion below is right, but the reason is NOT that an + /// unset algorithm leaves nothing to compute: a bare `PutObjectRequest` does report a default + /// algorithm name from `GetChecksumAlgorithmName()`. What actually suppresses the header is that + /// `PocoHTTPClientConfiguration` sets `requestChecksumCalculation` to `WHEN_REQUIRED` + /// unconditionally, which leaves the SDK's checksum interceptor gating purely on + /// `RequestChecksumRequired()` -- and this fork overrides that to `is_s3express_bucket`. + /// Independently, `setChecksumAlgorithm` is only ever reached from `setIsS3ExpressBucket`. + /// So the fact pinned here is that an ordinary HMAC client injects no checksum, and it is NOT a + /// test of `WriteBufferFromS3`'s S3Express-only checksum policy: this test never goes through + /// `WriteBufferFromS3` at all, so widening that policy tomorrow would not be caught here. + EXPECT_FALSE(captured[0].headers.has("x-amz-checksum-crc32")); + EXPECT_FALSE(captured[0].headers.has("x-amz-sdk-checksum-algorithm")); + } + + { + SCOPED_TRACE("If-None-Match with a non-star value: passes through unmolested"); + /// A non-star `If-None-Match` reaching `applyGcsConditionalDialectToRequest` aborts the process + /// with `LOGICAL_ERROR` (see ops notes) -- but that function is never called for this client at + /// all, so this is not the reachable case the death-test split exists for. `EXPECT_NO_THROW` is + /// the correct assertion here precisely because the guard is structurally unreachable, which is + /// exactly what this test is pinning. + DB::S3::PutObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + request.SetIfNoneMatch("some-non-star-value"); + /// `SetContentLength` explicitly, matching every production caller (e.g. `copyS3File.cpp`'s + /// `fillPutRequest`) for stylistic consistency -- the SDK computes it from the body itself when + /// omitted (`ExtendedRequest::IsStreaming` is always `false`, so the chunked path never engages), + /// so this call is not load-bearing for the keep-alive corruption below. That corruption's sole + /// cause is `ScriptedResponseServer` not draining the request body before responding; see the + /// fix in `ScriptedResponseServer::Handler::handleRequest`. + request.SetContentLength(7); + request.SetBody(Aws::MakeShared("gtest", "payload")); + EXPECT_NO_THROW(client->PutObject(request)); + + ASSERT_EQ(captured.size(), 2u); + EXPECT_EQ(captured[1].headers.get("if-none-match", ""), "some-non-star-value"); + EXPECT_FALSE(captured[1].headers.has("x-goog-if-generation-match")); + } + + { + SCOPED_TRACE("CopyObject: existing targeted mappings are the AWS ones, no goog- rename"); + /// Also a negative control for `CopyObjectRequestGetRequestSpecificHeadersRenamesOnlyUnderGcsApiMode` + /// below: this client's `api_mode` never becomes GCS (no `gcs_hmac`, no GCS-shaped endpoint, real + /// credentials), so `CopyObjectRequest::GetRequestSpecificHeaders` must leave these headers alone. + DB::S3::CopyObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("dest-key"); + request.SetCopySource("bucket/src-key"); + request.SetMetadataDirective(Aws::S3::Model::MetadataDirective::REPLACE); + request.SetStorageClass(Aws::S3::Model::StorageClass::STANDARD); + request.AddMetadata("cas-envelope", "v1"); + client->CopyObject(request); + + ASSERT_EQ(captured.size(), 3u); + EXPECT_TRUE(captured[2].headers.has("x-amz-copy-source")); + EXPECT_TRUE(captured[2].headers.has("x-amz-metadata-directive")); + EXPECT_TRUE(captured[2].headers.has("x-amz-storage-class")); + EXPECT_TRUE(captured[2].headers.has("x-amz-meta-cas-envelope")); + EXPECT_FALSE(captured[2].headers.has("x-goog-copy-source")); + EXPECT_FALSE(captured[2].headers.has("x-goog-metadata-directive")); + EXPECT_FALSE(captured[2].headers.has("x-goog-storage-class")); + } + + { + SCOPED_TRACE("DELETE: single object"); + DB::S3::DeleteObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("key"); + client->DeleteObject(request); + + ASSERT_EQ(captured.size(), 4u); + EXPECT_EQ(captured[3].method, "DELETE"); + EXPECT_TRUE(captured[3].headers.get("authorization", "").starts_with("AWS4-HMAC-SHA256")); + } + + { + SCOPED_TRACE("batch DeleteObjects"); + DB::S3::DeleteObjectsRequest request; + request.SetBucket("bucket"); + Aws::S3::Model::ObjectIdentifier obj1; + obj1.SetKey("key1"); + Aws::S3::Model::ObjectIdentifier obj2; + obj2.SetKey("key2"); + std::vector objects{obj1, obj2}; // STYLE_CHECK_ALLOW_STD_CONTAINERS + Aws::S3::Model::Delete del; + del.SetObjects(objects); + del.SetQuiet(true); + request.SetDelete(del); + client->DeleteObjects(request); + + ASSERT_EQ(captured.size(), 5u); + EXPECT_EQ(captured[4].method, "POST"); + EXPECT_TRUE(captured[4].headers.get("authorization", "").starts_with("AWS4-HMAC-SHA256")); + } +} + +/// The deferred allowlist gap from Task 4: `GcsHmacDefaultPutPassesTheAuthenticationAllowlist` exercised +/// only a small single-part PUT. Multipart is a distinct request shape family (`CreateMultipartUpload`, +/// `UploadPart`, `CompleteMultipartUpload`), each with its own header set, and none of them were driven +/// through the GOOG4 preparation before. Each `EXPECT_NO_THROW` below fails if the allowlist regresses +/// to reject a header this shape actually carries (`BAD_ARGUMENTS` from `prepareGcsRequestForGoog4Authentication`); +/// the header assertions fail if a `Rename` mapping stops firing and a stale `x-amz-*` header reaches the wire. +/// `Default` mode is used throughout, so `applyGcsConditionalDialectToRequest` (with its `LOGICAL_ERROR` +/// guards) is never invoked here -- no death-test split is needed for this test. +TEST(IOTestAwsS3Client, GcsHmacDefaultMultipartPassesTheAuthenticationAllowlist) +{ + ScriptedResponseServer server({{Poco::Net::HTTPResponse::HTTP_OK, {}}}); + auto client = makeRecordingClient(server.getUrl(), /*max_retries=*/0, /*max_redirects=*/0, /*http_client=*/"gcs_hmac"); + const auto & captured = server.getCapturedRequests(); + + { + DB::S3::CreateMultipartUploadRequest create_request; + create_request.SetBucket("bucket"); + create_request.SetKey("key"); + create_request.SetStorageClass(Aws::S3::Model::StorageClass::STANDARD); + create_request.AddMetadata("cas-envelope", "v1"); + EXPECT_NO_THROW(client->CreateMultipartUpload(create_request)); + } + + { + DB::S3::UploadPartRequest upload_part_request; + upload_part_request.SetBucket("bucket"); + upload_part_request.SetKey("key"); + upload_part_request.SetUploadId("test-upload-id"); + upload_part_request.SetPartNumber(1); + /// `SetContentLength` explicitly, matching `copyS3File.cpp`'s `makeUploadPartRequest` for + /// stylistic consistency -- not load-bearing here (see the comment on the `PutObjectRequest` + /// above). The real failure this test once hit -- `captured[2].method` reading back as + /// `"part-bodyPOST"`, the following CompleteMultipartUpload's parse corrupted by this request's + /// unread body on the shared keep-alive connection -- was caused solely by + /// `ScriptedResponseServer` not draining the request body before responding. + upload_part_request.SetContentLength(9); + upload_part_request.SetBody(Aws::MakeShared("gtest", "part-body")); + EXPECT_NO_THROW(client->UploadPart(upload_part_request)); + } + + { + DB::S3::CompleteMultipartUploadRequest complete_request; + complete_request.SetBucket("bucket"); + complete_request.SetKey("key"); + complete_request.SetUploadId("test-upload-id"); + Aws::S3::Model::CompletedMultipartUpload completed; + Aws::S3::Model::CompletedPart part; + part.WithPartNumber(1).WithETag("\"etag1\""); + completed.AddParts(part); + complete_request.SetMultipartUpload(completed); + /// Deliberately not marked NativeConditional and no If-Match/If-None-Match is set, so this POST + /// (uploadId, no partNumber) never reaches `applyGcsConditionalDialectToRequest`'s conditional + /// CompleteMultipartUpload guard -- see the file-level comment above. + EXPECT_NO_THROW(client->CompleteMultipartUpload(complete_request)); + } + + ASSERT_EQ(captured.size(), 3u); + + SCOPED_TRACE("CreateMultipartUpload: storage class and metadata renamed, nothing x-amz- left"); + EXPECT_TRUE(captured[0].headers.has("x-goog-storage-class")); + EXPECT_TRUE(captured[0].headers.has("x-goog-meta-cas-envelope")); + EXPECT_FALSE(captured[0].headers.has("x-amz-storage-class")); + EXPECT_FALSE(captured[0].headers.has("x-amz-meta-cas-envelope")); + + EXPECT_EQ(captured[2].method, "POST"); +} + +/// The second deferred shape: CopyObject through the GOOG4 preparation (`prepareGcsRequestForGoog4Authentication` +/// in `GCSConditionalDialect.cpp`), not previously exercised at all. This is a DIFFERENT mechanism from +/// the pre-existing, non-CAS `CopyObjectRequest::GetRequestSpecificHeaders` rename in `Requests.cpp`, +/// which is gated on the request's `api_mode` field, not on `http_client`. The mock endpoint here +/// (`127.0.0.1:PORT`) has no GCS-recognisable substring, so `Client`'s constructor never sets +/// `api_mode` to `GCS` even for this `gcs_hmac` client (that requires `provider_type == GCS` first, +/// which is endpoint-string-only) -- the `x-goog-copy-source` etc. observed below come entirely from +/// the GOOG4 preparation step, not from `Requests.cpp`. See +/// `CopyObjectRequestGetRequestSpecificHeadersRenamesOnlyUnderGcsApiMode` for that separate mechanism, +/// tested directly against the request object with no client or server involved. +TEST(IOTestAwsS3Client, GcsHmacDefaultCopyObjectPassesTheAuthenticationAllowlist) +{ + ScriptedResponseServer server({{Poco::Net::HTTPResponse::HTTP_OK, {}}}); + auto client = makeRecordingClient(server.getUrl(), /*max_retries=*/0, /*max_redirects=*/0, /*http_client=*/"gcs_hmac"); + + DB::S3::CopyObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("dest-key"); + request.SetCopySource("bucket/src-key"); + request.SetMetadataDirective(Aws::S3::Model::MetadataDirective::REPLACE); + request.SetStorageClass(Aws::S3::Model::StorageClass::STANDARD); + request.AddMetadata("cas-envelope", "v1"); + + EXPECT_NO_THROW(client->CopyObject(request)); + + const auto & captured = server.getCapturedRequests(); + ASSERT_EQ(captured.size(), 1u); + EXPECT_TRUE(captured[0].headers.has("x-goog-copy-source")); + EXPECT_TRUE(captured[0].headers.has("x-goog-metadata-directive")); + EXPECT_TRUE(captured[0].headers.has("x-goog-storage-class")); + EXPECT_TRUE(captured[0].headers.has("x-goog-meta-cas-envelope")); + EXPECT_FALSE(captured[0].headers.has("x-amz-copy-source")); + EXPECT_FALSE(captured[0].headers.has("x-amz-metadata-directive")); + EXPECT_FALSE(captured[0].headers.has("x-amz-storage-class")); + EXPECT_FALSE(captured[0].headers.has("x-amz-meta-cas-envelope")); +} + +/// The pre-existing (pre-CAS), non-GOOG4 CopyObject header mapping: `CopyObjectRequest::GetRequestSpecificHeaders` +/// in `Requests.cpp` renames `x-amz-copy-source`/`x-amz-metadata-directive`/`x-amz-storage-class`/ +/// `x-amz-meta-*` to their `x-goog-` counterparts, gated purely on the request's `api_mode` field (set +/// by `Client::doRequest` from the CLIENT's own `api_mode`, itself derived from `deduceProviderType` +/// matching the endpoint string against `storage.googleapis.com` -- see the file-level comment on +/// `GcsHmacDefaultCopyObjectPassesTheAuthenticationAllowlist` above). Driving this end-to-end through a +/// real `Client` would need a live server reachable AT a `storage.googleapis.com`-shaped hostname, which +/// this test harness cannot provide cheaply (no local DNS/network alias for that name; see the +/// integration test's own note on the same gap). `setApiMode` is public on `ExtendedRequest` for +/// exactly this reason: it lets a unit test set the one bit `GetRequestSpecificHeaders` reads without +/// needing a `Client` or any I/O at all. +TEST(IOTestAwsS3Client, CopyObjectRequestGetRequestSpecificHeadersRenamesOnlyUnderGcsApiMode) +{ + auto makeRequest = [] + { + DB::S3::CopyObjectRequest request; + request.SetBucket("bucket"); + request.SetKey("dest-key"); + request.SetCopySource("bucket/src-key"); + request.SetMetadataDirective(Aws::S3::Model::MetadataDirective::REPLACE); + request.SetStorageClass(Aws::S3::Model::StorageClass::STANDARD); + request.AddMetadata("cas-envelope", "v1"); + return request; + }; + + { + SCOPED_TRACE("api_mode left at its default (AWS): headers are untouched"); + auto request = makeRequest(); + const auto headers = request.GetRequestSpecificHeaders(); + EXPECT_TRUE(headers.contains("x-amz-copy-source")); + EXPECT_TRUE(headers.contains("x-amz-metadata-directive")); + EXPECT_TRUE(headers.contains("x-amz-storage-class")); + EXPECT_TRUE(headers.contains("x-amz-meta-cas-envelope")); + EXPECT_FALSE(headers.contains("x-goog-copy-source")); + } + + { + SCOPED_TRACE("api_mode explicitly set to GCS: every mapped header is renamed"); + auto request = makeRequest(); + request.setApiMode(DB::S3::ApiMode::GCS); + const auto headers = request.GetRequestSpecificHeaders(); + EXPECT_TRUE(headers.contains("x-goog-copy-source")); + EXPECT_TRUE(headers.contains("x-goog-metadata-directive")); + EXPECT_TRUE(headers.contains("x-goog-storage-class")); + EXPECT_TRUE(headers.contains("x-goog-meta-cas-envelope")); + EXPECT_FALSE(headers.contains("x-amz-copy-source")); + EXPECT_FALSE(headers.contains("x-amz-metadata-directive")); + EXPECT_FALSE(headers.contains("x-amz-storage-class")); + EXPECT_FALSE(headers.contains("x-amz-meta-cas-envelope")); + } +} + TEST(IOTestAwsS3Client, WrongSigningRegionBadRequest) { { diff --git a/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp b/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp new file mode 100644 index 000000000000..348ceaf005d6 --- /dev/null +++ b/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp @@ -0,0 +1,438 @@ +#include "config.h" +#if USE_AWS_S3 +#include +#include +#include +#include +#include +#include +#include +#include +#include /// DEBUG_OR_SANITIZER_BUILD +#include + +using namespace DB::S3; + +static Aws::Http::Standard::StandardHttpRequest makeRequest( + const char * url = "https://storage.googleapis.com/b/k", + Aws::Http::HttpMethod method = Aws::Http::HttpMethod::HTTP_PUT) +{ + Aws::Http::Standard::StandardHttpRequest request{Aws::Http::URI(url), method}; + request.SetHeaderValue("host", "storage.googleapis.com"); + return request; +} + +/// Installs every AWS SigV4 artifact both authentication paths must clear. +static void addAwsAuthArtifacts(Aws::Http::HttpRequest & r) +{ + r.SetHeaderValue("authorization", "AWS4-HMAC-SHA256 ..."); + r.SetHeaderValue("x-amz-date", "20260703T000000Z"); + r.SetHeaderValue("x-amz-content-sha256", "deadbeef"); + r.SetHeaderValue("x-amz-security-token", "tok"); + r.SetHeaderValue("x-amz-api-version", "2006-03-01"); +} + +static void expectNoAwsAuthArtifacts(const Aws::Http::HttpRequest & r) +{ + EXPECT_FALSE(r.HasHeader("authorization")); + EXPECT_FALSE(r.HasHeader("x-amz-date")); + EXPECT_FALSE(r.HasHeader("x-amz-content-sha256")); + EXPECT_FALSE(r.HasHeader("x-amz-security-token")); + EXPECT_FALSE(r.HasHeader("x-amz-api-version")); +} + +/// --------------------------------------------------------------------------------------------- +/// Conditions: only the native-conditional adapter translates them. +/// --------------------------------------------------------------------------------------------- + +TEST(GCSConditionalDialect, IfNoneMatchStarBecomesGenerationZero) +{ + auto r = makeRequest(); + r.SetHeaderValue("if-none-match", "*"); + applyGcsConditionalDialectToRequest(r); + EXPECT_FALSE(r.HasHeader("if-none-match")); + EXPECT_EQ(r.GetHeaderValue("x-goog-if-generation-match"), "0"); +} + +TEST(GCSConditionalDialect, IfMatchDigitsMappedQuotesStripped) +{ + auto r = makeRequest(); + r.SetHeaderValue("if-match", "\"1783078552147137\""); + applyGcsConditionalDialectToRequest(r); + EXPECT_FALSE(r.HasHeader("if-match")); + EXPECT_EQ(r.GetHeaderValue("x-goog-if-generation-match"), "1783078552147137"); +} + +TEST(GCSConditionalDialect, IfMatchUnquotedDigitsAlsoAccepted) +{ + auto r = makeRequest(); + r.SetHeaderValue("if-match", "1783078552147137"); + applyGcsConditionalDialectToRequest(r); + EXPECT_EQ(r.GetHeaderValue("x-goog-if-generation-match"), "1783078552147137"); +} + +TEST(GCSConditionalDialect, NonNumericIfMatchThrows) +{ + /// CORRUPTED_DATA, not a broken internal invariant: the value can come from a persisted manifest + /// token or from a storage HEAD whose response carried no generation, and `mintingTypeMatches` + /// upstream only compares the token KIND, never the shape of its value. + auto r = makeRequest(); + r.SetHeaderValue("if-match", "\"6654c734ccab8f440ff0825eb443dc7f\""); + EXPECT_THROW(applyGcsConditionalDialectToRequest(r), DB::Exception); +} + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(GCSConditionalDialect, NonStarIfNoneMatchThrows) +{ + /// LOGICAL_ERROR: CAS only ever sends `*`, so any other value is a wiring break, not input. + /// Under abort_on_logical_error that aborts at construction instead of being catchable, so + /// GCSConditionalDialectDeathTest.NonStarIfNoneMatchAborts proves it there. + auto r = makeRequest(); + r.SetHeaderValue("if-none-match", "\"123\""); + EXPECT_THROW(applyGcsConditionalDialectToRequest(r), DB::Exception); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(GCSConditionalDialectDeathTest, NonStarIfNoneMatchAborts) +{ + auto r = makeRequest(); + r.SetHeaderValue("if-none-match", "\"123\""); + EXPECT_DEATH({ applyGcsConditionalDialectToRequest(r); }, ""); +} +#endif + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(GCSConditionalDialect, ConditionalCompleteMultipartUploadThrows) +{ + /// GCS silently IGNORES preconditions on CompleteMultipartUpload (measured live 2026-07-03) -- + /// sending one would be silent data loss, so this fails closed client-side with a LOGICAL_ERROR: + /// every conditional non-blob write, including create-if-absent artifacts and conditional + /// replacements, forces a single PUT. Reaching here is a wiring break and aborts under + /// abort_on_logical_error; see the DeathTest below. Blob publication uses unconditional multipart. + auto r = makeRequest("https://storage.googleapis.com/b/k?uploadId=abc", Aws::Http::HttpMethod::HTTP_POST); + r.SetHeaderValue("if-none-match", "*"); + EXPECT_THROW(applyGcsConditionalDialectToRequest(r), DB::Exception); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(GCSConditionalDialectDeathTest, ConditionalCompleteMultipartUploadAborts) +{ + auto r = makeRequest("https://storage.googleapis.com/b/k?uploadId=abc", Aws::Http::HttpMethod::HTTP_POST); + r.SetHeaderValue("if-none-match", "*"); + EXPECT_DEATH({ applyGcsConditionalDialectToRequest(r); }, ""); +} +#endif + +TEST(GCSConditionalDialect, UnconditionalCompleteMultipartUploadPasses) +{ + auto r = makeRequest("https://storage.googleapis.com/b/k?uploadId=abc", Aws::Http::HttpMethod::HTTP_POST); + EXPECT_NO_THROW(applyGcsConditionalDialectToRequest(r)); +} + +TEST(GCSConditionalDialect, UploadPartIsNotComplete) +{ + /// PUT ?partNumber=N&uploadId=... is an UploadPart, not a Complete — must not trip the guard. + auto r = makeRequest("https://storage.googleapis.com/b/k?partNumber=1&uploadId=abc", Aws::Http::HttpMethod::HTTP_PUT); + EXPECT_NO_THROW(applyGcsConditionalDialectToRequest(r)); +} + +/// Neither authentication preparation may acquire condition semantics: a request that was never +/// marked native-conditional must keep its standard ETag preconditions all the way to the wire. +TEST(GCSConditionalDialect, AuthenticationPreparationLeavesConditionsAlone) +{ + auto goog4 = makeRequest(); + goog4.SetHeaderValue("if-match", "\"6654c734ccab8f440ff0825eb443dc7f\""); + goog4.SetHeaderValue("if-none-match", "*"); + prepareGcsRequestForGoog4Authentication(goog4); + EXPECT_EQ(goog4.GetHeaderValue("if-match"), "\"6654c734ccab8f440ff0825eb443dc7f\""); + EXPECT_EQ(goog4.GetHeaderValue("if-none-match"), "*"); + EXPECT_FALSE(goog4.HasHeader("x-goog-if-generation-match")); + + auto oauth = makeRequest(); + oauth.SetHeaderValue("if-match", "\"6654c734ccab8f440ff0825eb443dc7f\""); + prepareGcsRequestForOAuthAuthentication(oauth); + EXPECT_EQ(oauth.GetHeaderValue("if-match"), "\"6654c734ccab8f440ff0825eb443dc7f\""); + EXPECT_FALSE(oauth.HasHeader("x-goog-if-generation-match")); +} + +/// --------------------------------------------------------------------------------------------- +/// Request metadata: the one targeted prefix mapping the adapter owns. +/// --------------------------------------------------------------------------------------------- + +TEST(GCSConditionalDialect, RequestMetadataPrefixIsMapped) +{ + auto r = makeRequest(); + r.SetHeaderValue("x-amz-meta-foo", "bar"); + r.SetHeaderValue("x-goog-meta-already", "kept"); + applyGcsConditionalDialectToRequest(r); + EXPECT_FALSE(r.HasHeader("x-amz-meta-foo")); + EXPECT_EQ(r.GetHeaderValue("x-goog-meta-foo"), "bar"); + EXPECT_EQ(r.GetHeaderValue("x-goog-meta-already"), "kept"); +} + +TEST(GCSConditionalDialect, RequestMetadataDoesNotTouchOtherAmzHeaders) +{ + /// The adapter is not a blanket rewrite: only conditions and `x-amz-meta-*` are its business. + /// Whatever else the SDK put on the request is the authentication preparation's problem. + auto r = makeRequest(); + addAwsAuthArtifacts(r); + r.SetHeaderValue("x-amz-storage-class", "STANDARD"); + r.SetHeaderValue("x-amz-tagging", "a=b"); + applyGcsConditionalDialectToRequest(r); + EXPECT_EQ(r.GetHeaderValue("authorization"), "AWS4-HMAC-SHA256 ..."); + EXPECT_EQ(r.GetHeaderValue("x-amz-date"), "20260703T000000Z"); + EXPECT_EQ(r.GetHeaderValue("x-amz-storage-class"), "STANDARD"); + EXPECT_EQ(r.GetHeaderValue("x-amz-tagging"), "a=b"); + EXPECT_FALSE(r.HasHeader("x-goog-storage-class")); +} + +TEST(GCSConditionalDialect, ConflictingRequestMetadataRejected) +{ + auto r = makeRequest(); + r.SetHeaderValue("x-amz-meta-foo", "one"); + r.SetHeaderValue("x-goog-meta-foo", "two"); + EXPECT_THROW(applyGcsConditionalDialectToRequest(r), DB::Exception); +} + +TEST(GCSConditionalDialect, AgreeingRequestMetadataAccepted) +{ + auto r = makeRequest(); + r.SetHeaderValue("x-amz-meta-foo", "same"); + r.SetHeaderValue("x-goog-meta-foo", "same"); + EXPECT_NO_THROW(applyGcsConditionalDialectToRequest(r)); + EXPECT_FALSE(r.HasHeader("x-amz-meta-foo")); + EXPECT_EQ(r.GetHeaderValue("x-goog-meta-foo"), "same"); +} + +/// --------------------------------------------------------------------------------------------- +/// Native OAuth authentication preparation. +/// --------------------------------------------------------------------------------------------- + +TEST(GCSConditionalDialect, OAuthPreparationDropsAwsAuthArtifacts) +{ + auto r = makeRequest(); + addAwsAuthArtifacts(r); + prepareGcsRequestForOAuthAuthentication(r); + expectNoAwsAuthArtifacts(r); +} + +TEST(GCSConditionalDialect, OAuthPreparationPassesRemainingAmzHeadersThrough) +{ + /// Native OAuth has no GOOG4-style allowlist by design: after the signing artifacts are gone it + /// matches the ordinary OAuth path, which leaves SDK headers untouched. Pinning this stops a + /// later adapter change from silently broadening OAuth rewriting. + auto r = makeRequest(); + r.SetHeaderValue("x-amz-sdk-checksum-algorithm", "CRC32"); + r.SetHeaderValue("x-amz-checksum-crc32", "abcd=="); + r.SetHeaderValue("x-amz-trailer", "x-amz-checksum-crc32"); + r.SetHeaderValue("x-amz-decoded-content-length", "1024"); + r.SetHeaderValue("x-amz-storage-class", "STANDARD"); + r.SetHeaderValue("x-amz-tagging", "a=b"); + EXPECT_NO_THROW(prepareGcsRequestForOAuthAuthentication(r)); + EXPECT_EQ(r.GetHeaderValue("x-amz-sdk-checksum-algorithm"), "CRC32"); + EXPECT_EQ(r.GetHeaderValue("x-amz-checksum-crc32"), "abcd=="); + EXPECT_EQ(r.GetHeaderValue("x-amz-trailer"), "x-amz-checksum-crc32"); + EXPECT_EQ(r.GetHeaderValue("x-amz-decoded-content-length"), "1024"); + EXPECT_EQ(r.GetHeaderValue("x-amz-storage-class"), "STANDARD"); + EXPECT_EQ(r.GetHeaderValue("x-amz-tagging"), "a=b"); +} + +/// --------------------------------------------------------------------------------------------- +/// GOOG4 authentication preparation: every x-amz-* header has a decided fate. +/// --------------------------------------------------------------------------------------------- + +TEST(GCSConditionalDialect, Goog4PreparationDropsAwsAuthArtifacts) +{ + auto r = makeRequest(); + addAwsAuthArtifacts(r); + prepareGcsRequestForGoog4Authentication(r); + expectNoAwsAuthArtifacts(r); + /// Dropped, not renamed: a `x-goog-`-prefixed copy of a SigV4 artifact would be signed as part of + /// the GOOG4 canonical request. + EXPECT_FALSE(r.HasHeader("x-goog-date")); + EXPECT_FALSE(r.HasHeader("x-goog-content-sha256")); + EXPECT_FALSE(r.HasHeader("x-goog-security-token")); + EXPECT_FALSE(r.HasHeader("x-goog-api-version")); +} + +TEST(GCSConditionalDialect, Goog4PreparationRenamesTargetedStorageAndCopyHeaders) +{ + auto r = makeRequest(); + r.SetHeaderValue("x-amz-meta-foo", "bar"); + r.SetHeaderValue("x-amz-storage-class", "STANDARD"); + r.SetHeaderValue("x-amz-copy-source", "b/src"); + r.SetHeaderValue("x-amz-copy-source-range", "bytes=0-9"); + r.SetHeaderValue("x-amz-metadata-directive", "REPLACE"); + prepareGcsRequestForGoog4Authentication(r); + EXPECT_EQ(r.GetHeaderValue("x-goog-meta-foo"), "bar"); + EXPECT_EQ(r.GetHeaderValue("x-goog-storage-class"), "STANDARD"); + EXPECT_EQ(r.GetHeaderValue("x-goog-copy-source"), "b/src"); + EXPECT_EQ(r.GetHeaderValue("x-goog-copy-source-range"), "bytes=0-9"); + EXPECT_EQ(r.GetHeaderValue("x-goog-metadata-directive"), "REPLACE"); + EXPECT_FALSE(r.HasHeader("x-amz-meta-foo")); + EXPECT_FALSE(r.HasHeader("x-amz-storage-class")); + EXPECT_FALSE(r.HasHeader("x-amz-copy-source")); + EXPECT_FALSE(r.HasHeader("x-amz-copy-source-range")); + EXPECT_FALSE(r.HasHeader("x-amz-metadata-directive")); +} + +TEST(GCSConditionalDialect, Goog4PreparationRenameIsIdempotentAfterTheAdapter) +{ + /// A marked request runs the adapter first, which already moved `x-amz-meta-*` across. The + /// preparation must then find nothing to do rather than tripping its own conflict check. + auto r = makeRequest(); + r.SetHeaderValue("x-amz-meta-foo", "bar"); + applyGcsConditionalDialectToRequest(r); + EXPECT_NO_THROW(prepareGcsRequestForGoog4Authentication(r)); + EXPECT_EQ(r.GetHeaderValue("x-goog-meta-foo"), "bar"); +} + +TEST(GCSConditionalDialect, Goog4PreparationConsumesSdkChecksumHeaders) +{ + /// Flexible checksums are an S3 protocol feature with no GCS XML API counterpart; the body they + /// describe goes out unchanged, so consuming them is lossless on the wire. + auto r = makeRequest(); + r.SetHeaderValue("x-amz-sdk-checksum-algorithm", "CRC32"); + r.SetHeaderValue("x-amz-checksum-crc32", "abcd=="); + r.SetHeaderValue("x-amz-checksum-sha256", "efgh=="); + prepareGcsRequestForGoog4Authentication(r); + EXPECT_FALSE(r.HasHeader("x-amz-sdk-checksum-algorithm")); + EXPECT_FALSE(r.HasHeader("x-amz-checksum-crc32")); + EXPECT_FALSE(r.HasHeader("x-amz-checksum-sha256")); + /// Consumed, not renamed — GCS would not understand them under the other prefix either. + EXPECT_FALSE(r.HasHeader("x-goog-sdk-checksum-algorithm")); + EXPECT_FALSE(r.HasHeader("x-goog-checksum-crc32")); +} + +TEST(GCSConditionalDialect, Goog4PreparationRejectsAwsChunkedFraming) +{ + /// BAD_ARGUMENTS: these announce a body framing GCS cannot parse, and consuming them would + /// misdescribe a body already on the wire, so refuse rather than guess. + auto trailer = makeRequest(); + trailer.SetHeaderValue("x-amz-trailer", "x-amz-checksum-crc32"); + EXPECT_THROW(prepareGcsRequestForGoog4Authentication(trailer), DB::Exception); + + auto decoded = makeRequest(); + decoded.SetHeaderValue("x-amz-decoded-content-length", "1024"); + EXPECT_THROW(prepareGcsRequestForGoog4Authentication(decoded), DB::Exception); +} + +TEST(GCSConditionalDialect, Goog4PreparationRejectsUnknownAmzExtension) +{ + /// BAD_ARGUMENTS before any network I/O: GCS rejects a mixed-prefix request, so an unmapped + /// header can be neither translated nor sent. + for (const char * header : {"x-amz-tagging", "x-amz-acl", "x-amz-server-side-encryption", + "x-amz-server-side-encryption-customer-key", "x-amz-website-redirect-location"}) + { + auto r = makeRequest(); + r.SetHeaderValue(header, "whatever"); + EXPECT_THROW(prepareGcsRequestForGoog4Authentication(r), DB::Exception) << header; + } +} + +TEST(GCSConditionalDialect, Goog4PreparationLeavesNonAmzHeadersAlone) +{ + /// `amz-sdk-invocation-id` and `amz-sdk-request` do not carry the `x-amz-` prefix and are not + /// part of any canonical request, so they pass through untouched. + auto r = makeRequest(); + r.SetHeaderValue("amz-sdk-invocation-id", "id"); + r.SetHeaderValue("amz-sdk-request", "attempt=1"); + r.SetHeaderValue("content-type", "binary/octet-stream"); + prepareGcsRequestForGoog4Authentication(r); + EXPECT_EQ(r.GetHeaderValue("amz-sdk-invocation-id"), "id"); + EXPECT_EQ(r.GetHeaderValue("amz-sdk-request"), "attempt=1"); + EXPECT_EQ(r.GetHeaderValue("content-type"), "binary/octet-stream"); + EXPECT_EQ(r.GetHeaderValue("host"), "storage.googleapis.com"); +} + +/// --------------------------------------------------------------------------------------------- +/// Response adaptation. +/// --------------------------------------------------------------------------------------------- + +namespace +{ + +/// A real SDK response object, so these tests exercise the type `PocoHTTPClient` actually fills. +struct ResponseFixture +{ + /// `StandardHttpResponse`'s constructor builds its body stream by CALLING the originating + /// request's response-stream factory, so the request must carry one or the response cannot be + /// constructed at all. + static std::shared_ptr makeOriginatingRequest() + { + auto request = std::make_shared( + Aws::Http::URI("https://storage.googleapis.com/b/k"), Aws::Http::HttpMethod::HTTP_HEAD); + request->SetResponseStreamFactory([] { return Aws::New("gtest", ""); }); + return request; + } + + std::shared_ptr request = makeOriginatingRequest(); + Aws::Http::Standard::StandardHttpResponse sdk{request}; + Poco::Net::HTTPResponse poco; + + /// Mirrors PocoHTTPClient: every response header is copied onto the SDK response first, and the + /// adaptation runs on top of that. + void copyThenAdapt() + { + for (const auto & [name, value] : poco) + sdk.AddHeader(name, value); + applyGcsConditionalDialectToResponse(poco, sdk); + } +}; + +} + +/// This test and `ResponseMetadataPrefixIsMapped` look redundant and are not: only this one can +/// catch an install that fails to REPLACE. The copy loop above has already put the server's `etag` on +/// the response, so a wrong `AddHeader` overload — the `Aws::String &&` one emplaces instead of +/// assigning — leaves the server value standing and the substitution silently does nothing. No +/// `x-amz-meta-*` key is pre-occupied, so the metadata test would insert successfully either way and +/// stay green. Do not delete this as a duplicate. +TEST(GCSConditionalDialect, ResponseGenerationOverridesETag) +{ + ResponseFixture f; + f.poco.set("ETag", "\"6654c734ccab8f440ff0825eb443dc7f\""); + f.poco.set("x-goog-generation", "1783078552147137"); + f.copyThenAdapt(); + EXPECT_EQ(f.sdk.GetHeader("etag"), "\"1783078552147137\""); +} + +TEST(GCSConditionalDialect, ResponseWithoutGenerationKeepsETag) +{ + ResponseFixture f; + f.poco.set("ETag", "\"abc\""); + f.copyThenAdapt(); + EXPECT_EQ(f.sdk.GetHeader("etag"), "\"abc\""); +} + +/// This one stays despite `IOTestAwsS3Client.ResponseGenerationAndMetadataAdaptedOnlyWhenMarked` +/// covering the same mapping end to end: that test drives a whole client, so it can only report that +/// the mapping is absent, while this one localises the absence to the response adapter itself. +TEST(GCSConditionalDialect, ResponseMetadataPrefixIsMapped) +{ + ResponseFixture f; + f.poco.set("x-goog-generation", "42"); + f.poco.set("x-goog-meta-cas-envelope", "v1"); + f.copyThenAdapt(); + EXPECT_EQ(f.sdk.GetHeader("x-amz-meta-cas-envelope"), "v1"); +} + +TEST(GCSConditionalDialect, ConflictingResponseMetadataRejected) +{ + ResponseFixture f; + f.poco.set("x-goog-meta-cas-envelope", "v1"); + f.poco.set("x-amz-meta-cas-envelope", "v2"); + EXPECT_THROW(f.copyThenAdapt(), DB::Exception); +} + +TEST(GCSConditionalDialect, AgreeingResponseMetadataAccepted) +{ + ResponseFixture f; + f.poco.set("x-goog-meta-cas-envelope", "v1"); + f.poco.set("x-amz-meta-cas-envelope", "v1"); + EXPECT_NO_THROW(f.copyThenAdapt()); + EXPECT_EQ(f.sdk.GetHeader("x-amz-meta-cas-envelope"), "v1"); +} +#endif diff --git a/src/IO/S3/tests/gtest_goog4_signer.cpp b/src/IO/S3/tests/gtest_goog4_signer.cpp new file mode 100644 index 000000000000..7ef9b9e6c684 --- /dev/null +++ b/src/IO/S3/tests/gtest_goog4_signer.cpp @@ -0,0 +1,151 @@ +#include "config.h" +#if USE_AWS_S3 +#include +#include +#include +#include +#include + +using namespace DB::S3; + +static std::chrono::system_clock::time_point fixedNow() +{ + /// 2026-07-03 00:00:00 UTC + return std::chrono::system_clock::from_time_t(1783036800); +} + +TEST(GOOG4Signer, PutWithGenerationPrecondition) +{ + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dir/obj.txt"), Aws::Http::HttpMethod::HTTP_PUT); + request.SetHeaderValue("host", "storage.googleapis.com"); + request.SetHeaderValue("x-goog-if-generation-match", "0"); + + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + EXPECT_EQ(request.GetHeaderValue("x-goog-date"), "20260703T000000Z"); + EXPECT_EQ(request.GetHeaderValue("x-goog-content-sha256"), "UNSIGNED-PAYLOAD"); + EXPECT_EQ(request.GetHeaderValue("authorization"), + "GOOG4-HMAC-SHA256 Credential=GOOGTESTACCESSKEY/20260703/auto/storage/goog4_request, " + "SignedHeaders=host;x-goog-content-sha256;x-goog-date;x-goog-if-generation-match, " + "Signature=4f82e49c69753329afd4768ccf1db6b472dbbd86d082a08b5b9f9fe368fb6ef6"); +} + +TEST(GOOG4Signer, GetWithQueryString) +{ + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/?versioning"), Aws::Http::HttpMethod::HTTP_GET); + request.SetHeaderValue("host", "storage.googleapis.com"); + + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + EXPECT_EQ(request.GetHeaderValue("authorization"), + "GOOG4-HMAC-SHA256 Credential=GOOGTESTACCESSKEY/20260703/auto/storage/goog4_request, " + "SignedHeaders=host;x-goog-content-sha256;x-goog-date, " + "Signature=28a981c32acff334738b9ea1a0f82c28c9a1ccff5b6dc8fb92a2e6622c8db73f"); +} + +TEST(GOOG4Signer, NonGoogHeadersAreNotSigned) +{ + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dir/obj.txt"), Aws::Http::HttpMethod::HTTP_PUT); + request.SetHeaderValue("host", "storage.googleapis.com"); + request.SetHeaderValue("x-goog-if-generation-match", "0"); + request.SetHeaderValue("content-type", "binary/octet-stream"); + request.SetHeaderValue("amz-sdk-invocation-id", "whatever"); + + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + /// Unsigned headers must not perturb the signature: same vector as PutWithGenerationPrecondition. + EXPECT_NE(request.GetHeaderValue("authorization").find( + "Signature=4f82e49c69753329afd4768ccf1db6b472dbbd86d082a08b5b9f9fe368fb6ef6"), std::string::npos); +} + +TEST(GOOG4Signer, NothingAmzPrefixedSurvivesIntoTheSignature) +{ + /// The composition the GOOG4 client performs: authentication preparation first, then signing. + /// GCS rejects a request that mixes the prefixes, so after preparation the canonical request must + /// contain no `x-amz-*` header at all — and the request itself must carry none either. + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dir/obj.txt"), Aws::Http::HttpMethod::HTTP_PUT); + request.SetHeaderValue("host", "storage.googleapis.com"); + request.SetHeaderValue("x-goog-if-generation-match", "0"); + request.SetHeaderValue("authorization", "AWS4-HMAC-SHA256 ..."); + request.SetHeaderValue("x-amz-date", "20260703T000000Z"); + request.SetHeaderValue("x-amz-content-sha256", "deadbeef"); + request.SetHeaderValue("x-amz-meta-foo", "bar"); + request.SetHeaderValue("x-amz-storage-class", "STANDARD"); + request.SetHeaderValue("x-amz-sdk-checksum-algorithm", "CRC32"); + + prepareGcsRequestForGoog4Authentication(request); + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + for (const auto & [name, value] : request.GetHeaders()) + EXPECT_FALSE(name.starts_with("x-amz-")) << name; + + const auto authorization = request.GetHeaderValue("authorization"); + EXPECT_EQ(authorization.find("x-amz-"), std::string::npos) << authorization; + /// The surviving x-goog- headers ARE signed, so the preparation did not simply drop everything. + EXPECT_NE(authorization.find("x-goog-meta-foo"), std::string::npos) << authorization; + EXPECT_NE(authorization.find("x-goog-storage-class"), std::string::npos) << authorization; +} + +TEST(GOOG4Signer, DefaultPutHasNoGenerationPreconditionInTheSignature) +{ + /// The `Default`-mode counterpart of `PutWithGenerationPrecondition`: an ordinary (non-CAS) write + /// through `gcs_hmac` never acquires `x-goog-if-generation-match` at all, so GOOG4 authentication + /// must sign it the same way it would sign any other GOOG4 PUT, with that header simply absent from + /// `SignedHeaders` -- not replaced by some other precondition, not rejected. + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dir/obj.txt"), Aws::Http::HttpMethod::HTTP_PUT); + request.SetHeaderValue("host", "storage.googleapis.com"); + + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + /// Compare against a request signed the same way but WITH the precondition set: identical + /// SignedHeaders/Credential scope apart from the one header, proving the precondition's absence (not + /// some other divergence) is what changes between the two -- the same fixed key/time/path as + /// `PutWithGenerationPrecondition` isolates that one variable. + Aws::Http::Standard::StandardHttpRequest with_precondition( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dir/obj.txt"), Aws::Http::HttpMethod::HTTP_PUT); + with_precondition.SetHeaderValue("host", "storage.googleapis.com"); + with_precondition.SetHeaderValue("x-goog-if-generation-match", "0"); + signRequestGOOG4(with_precondition, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + const auto authorization = request.GetHeaderValue("authorization"); + EXPECT_EQ(authorization.find("x-goog-if-generation-match"), std::string::npos) << authorization; + EXPECT_NE(authorization.find("SignedHeaders=host;x-goog-content-sha256;x-goog-date,"), std::string::npos) << authorization; + EXPECT_NE(authorization, with_precondition.GetHeaderValue("authorization")); +} + +TEST(GOOG4Signer, CopyObjectHeadersAreRenamedAndSignedAsGoogPrefixed) +{ + /// The other deferred shape from Task 4/5: a CopyObject request, prepared for GOOG4 the same way + /// `PocoHTTPClientGCSHMAC::makeRequestInternal` does it before signing. `x-amz-copy-source` and + /// `x-amz-metadata-directive` have their own `Rename` rule in `GOOG4_HEADER_RULES` distinct from the + /// storage-class/meta-* one `NothingAmzPrefixedSurvivesIntoTheSignature` already covers, so nothing + /// existing exercised them until now. + Aws::Http::Standard::StandardHttpRequest request( + Aws::Http::URI("https://storage.googleapis.com/test-bucket/dest.txt"), Aws::Http::HttpMethod::HTTP_PUT); + request.SetHeaderValue("host", "storage.googleapis.com"); + request.SetHeaderValue("x-amz-copy-source", "test-bucket/src.txt"); + request.SetHeaderValue("x-amz-metadata-directive", "REPLACE"); + request.SetHeaderValue("x-amz-meta-cas-envelope", "v1"); + + prepareGcsRequestForGoog4Authentication(request); + signRequestGOOG4(request, Aws::Auth::AWSCredentials("GOOGTESTACCESSKEY", "testsecretkey"), fixedNow()); + + for (const auto & [name, value] : request.GetHeaders()) + EXPECT_FALSE(name.starts_with("x-amz-")) << name; + + EXPECT_EQ(request.GetHeaderValue("x-goog-copy-source"), "test-bucket/src.txt"); + EXPECT_EQ(request.GetHeaderValue("x-goog-metadata-directive"), "REPLACE"); + EXPECT_EQ(request.GetHeaderValue("x-goog-meta-cas-envelope"), "v1"); + + const auto authorization = request.GetHeaderValue("authorization"); + EXPECT_EQ(authorization.find("x-amz-"), std::string::npos) << authorization; + EXPECT_NE(authorization.find("x-goog-copy-source"), std::string::npos) << authorization; + EXPECT_NE(authorization.find("x-goog-metadata-directive"), std::string::npos) << authorization; + EXPECT_NE(authorization.find("x-goog-meta-cas-envelope"), std::string::npos) << authorization; +} +#endif diff --git a/src/IO/S3AuthSettings.cpp b/src/IO/S3AuthSettings.cpp index 3265edba656c..830630a42f1e 100644 --- a/src/IO/S3AuthSettings.cpp +++ b/src/IO/S3AuthSettings.cpp @@ -25,9 +25,11 @@ namespace DB DECLARE(Bool, no_sign_request, S3::DEFAULT_NO_SIGN_REQUEST, "", 0) \ DECLARE(Bool, use_insecure_imds_request, false, "", 0) \ DECLARE(Bool, use_adaptive_timeouts, S3::DEFAULT_USE_ADAPTIVE_TIMEOUTS, "", 0) \ + DECLARE(UInt64, expect_continue_min_bytes, S3::DEFAULT_EXPECT_CONTINUE_MIN_BYTES, "", 0) \ DECLARE(Bool, is_virtual_hosted_style, false, "", 0) \ DECLARE(Bool, disable_checksum, S3::DEFAULT_DISABLE_CHECKSUM, "", 0) \ DECLARE(Bool, gcs_issue_compose_request, false, "", 0) \ + DECLARE(UInt64, gcs_max_conditional_put_bytes, S3::DEFAULT_GCS_MAX_CONDITIONAL_PUT_BYTES, "", 0) \ DECLARE(S3UriStyle, uri_style, S3UriStyle::AUTO, "", 0) #define AUTH_SETTINGS(DECLARE, ALIAS) \ diff --git a/src/IO/S3Common.cpp b/src/IO/S3Common.cpp index 7db3a9cb4379..9f986d89d126 100644 --- a/src/IO/S3Common.cpp +++ b/src/IO/S3Common.cpp @@ -43,11 +43,57 @@ bool S3Exception::isAccessTokenExpiredError() const return code == Aws::S3::S3Errors::INVALID_ACCESS_KEY_ID || code == Aws::S3::S3Errors::ACCESS_DENIED || code == Aws::S3::S3Errors::INVALID_SIGNATURE || code == Aws::S3::S3Errors::UNKNOWN; } +<<<<<<< HEAD bool isTransientCompleteMultipartUploadError(const Aws::S3::S3Error & error) { return error.GetErrorType() == Aws::S3::S3Errors::NO_SUCH_KEY || error.GetExceptionName() == "InvalidPart" || error.GetExceptionName() == "InvalidPartOrder"; +======= +bool S3Exception::isPreconditionFailed() const +{ + /// See `S3::isPreconditionFailedError`. The thrown exception no longer carries the HTTP status, so + /// only the name and raw message are available here — fail-safe: matching too broadly maps a hard + /// error to a retryable re-validate, never a false success. + return exception_name == "PreconditionFailed" + || message().find("PreconditionFailed") != std::string::npos; +} + +namespace S3 +{ + +/// A synchronous rejection PROVING the request was never applied — matched by the canonical S3 error +/// code STRING (many of these are UNKNOWN in the SDK's modeled enum, mirroring +/// ObjectStorageBackend::finalizeConditionalWrite's own name-first matching) plus the modeled enum +/// value where one exists, belt-and-suspenders. +bool isMalformedRequestError(const S3Exception & e) +{ + const String & name = e.getExceptionName(); + return name == "MalformedXML" || name == "MalformedPOSTRequest" || name == "InvalidArgument" + || name == "InvalidRequest" || name == "InvalidBucketName" || name == "KeyTooLongError" + || e.getS3ErrorCode() == Aws::S3::S3Errors::INVALID_PARAMETER_VALUE + || e.getS3ErrorCode() == Aws::S3::S3Errors::INVALID_REQUEST + || e.getS3ErrorCode() == Aws::S3::S3Errors::VALIDATION; +} + +bool isEntityTooLargeError(const S3Exception & e) +{ + /// No modeled enum value for this error — name-only match, same as PreconditionFailed elsewhere. + return e.getExceptionName() == "EntityTooLarge"; +} + +bool isAccessDeniedError(const S3Exception & e) +{ + const String & name = e.getExceptionName(); + return name == "AccessDenied" || name == "InvalidAccessKeyId" || name == "SignatureDoesNotMatch" + || name == "InvalidToken" || name == "ExpiredToken" || name == "AccountProblem" + || e.getS3ErrorCode() == Aws::S3::S3Errors::ACCESS_DENIED + || e.getS3ErrorCode() == Aws::S3::S3Errors::INVALID_ACCESS_KEY_ID + || e.getS3ErrorCode() == Aws::S3::S3Errors::SIGNATURE_DOES_NOT_MATCH + || e.getS3ErrorCode() == Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID; +} + +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/IO/S3Common.h b/src/IO/S3Common.h index 0604767f8e4b..7604e5028e90 100644 --- a/src/IO/S3Common.h +++ b/src/IO/S3Common.h @@ -46,9 +46,18 @@ class S3Exception : public Exception { } - S3Exception(const std::string & msg, Aws::S3::S3Errors code_) + S3Exception(const std::string & msg, Aws::S3::S3Errors code_, String exception_name_ = {}) : Exception(msg, ErrorCodes::S3_ERROR) , code(code_) + , exception_name(std::move(exception_name_)) + {} + + /// Preserves the static format string (system.text_log / system.errors grouping) while also + /// carrying the canonical S3 error name — build msg with PreformattedMessage::create. + S3Exception(PreformattedMessage && msg, Aws::S3::S3Errors code_, String exception_name_) + : Exception(std::move(msg), ErrorCodes::S3_ERROR) + , code(code_) + , exception_name(std::move(exception_name_)) {} Aws::S3::S3Errors getS3ErrorCode() const @@ -56,15 +65,57 @@ class S3Exception : public Exception return code; } + /// The canonical S3 error code string from the response XML `` (e.g. "PreconditionFailed", + /// "NoSuchKey") as reported by `Aws::Client::AWSError::GetExceptionName`. Errors unmodeled by the + /// SDK (a conditional-PUT 412 is one) have `getS3ErrorCode` == UNKNOWN, so this name is the only + /// machine-readable discriminator. Empty when the throw site did not attach it. + /// Not `Exception::name`; this is the AWS `` string. + const String & getExceptionName() const + { + return exception_name; + } + bool isRetryableError() const; bool isAccessTokenExpiredError() const; + /// True for a conditional-request 412 (a lost `If-Match`/`If-None-Match`). The thrown exception + /// discards the HTTP status, so it matches on the canonical `` name and the raw message — + /// see `S3::isPreconditionFailedError` for the full (response-code-aware) policy. + bool isPreconditionFailed() const; + S3Exception * clone() const override { return new S3Exception(*this); } void rethrow() const override { throw *this; } /// NOLINT(bugprone-exception-copy-constructor-throws,cert-err60-cpp) private: Aws::S3::S3Errors code; + String exception_name; }; + +namespace S3 +{ + +/// One policy for "is this error a conditional-request 412 (`PreconditionFailed`)?", shared by the +/// retry strategy and the CA conditional delete/copy paths. The HTTP status is authoritative — a +/// non-AWS body (e.g. RustFS) leaves the SDK-parsed `ExceptionName` empty — with the canonical `` +/// name and the raw message as fallbacks. Fail-safe by direction: over-matching only forces a caller +/// re-validate, never a false success. +template +inline bool isPreconditionFailedError(const Aws::Client::AWSError & error) +{ + return error.GetResponseCode() == Aws::Http::HttpResponseCode::PRECONDITION_FAILED + || error.GetExceptionName() == "PreconditionFailed" + || error.GetMessage().find("PreconditionFailed") != std::string::npos; +} + +/// Error-name classifiers factored out of the CAS conditional-write outcome mapping +/// (`CasRequestControl.cpp`), so the name lists live next to the other S3 error classifiers here +/// and are available for reuse. +bool isMalformedRequestError(const S3Exception & e); +bool isEntityTooLargeError(const S3Exception & e); +bool isAccessDeniedError(const S3Exception & e); + +} + } #endif diff --git a/src/IO/S3Defines.h b/src/IO/S3Defines.h index 7826b0488847..9a825efb75b9 100644 --- a/src/IO/S3Defines.h +++ b/src/IO/S3Defines.h @@ -24,6 +24,7 @@ inline static constexpr bool DEFAULT_USE_ADAPTIVE_TIMEOUTS = true; inline static constexpr uint64_t DEFAULT_MIN_UPLOAD_PART_SIZE = 16 * 1024 * 1024; inline static constexpr uint64_t DEFAULT_MAX_UPLOAD_PART_SIZE = 5ull * 1024 * 1024 * 1024; inline static constexpr uint64_t DEFAULT_MAX_SINGLE_PART_UPLOAD_SIZE = 32 * 1024 * 1024; +inline static constexpr uint64_t DEFAULT_GCS_MAX_CONDITIONAL_PUT_BYTES = 1ULL << 30; inline static constexpr uint64_t DEFAULT_STRICT_UPLOAD_PART_SIZE = 0; inline static constexpr uint64_t DEFAULT_UPLOAD_PART_SIZE_MULTIPLY_FACTOR = 2; inline static constexpr uint64_t DEFAULT_UPLOAD_PART_SIZE_MULTIPLY_PARTS_COUNT_THRESHOLD = 500; @@ -36,6 +37,13 @@ inline static constexpr uint64_t DEFAULT_LIST_OBJECT_KEYS_SIZE = 1000; inline static constexpr uint64_t DEFAULT_MAX_SINGLE_READ_TRIES = 4; inline static constexpr uint64_t DEFAULT_MAX_UNEXPECTED_WRITE_ERROR_RETRIES = 4; inline static constexpr uint64_t DEFAULT_MAX_REDIRECTS = 10; +/// Gate for the `Expect: 100-continue` negotiation on a conditional write (If-None-Match / If-Match): +/// `0` = disabled (never negotiate Expect); a positive `N` negotiates Expect for a conditional `PUT` +/// whose body is at least `N` bytes, so the server can reject (e.g. 412) BEFORE the body is streamed +/// (B118). The default is DISABLED: only a CAS conditional-write client raises this (see the +/// single-attempt client built in `ObjectStorageBackend`), so non-CAS S3 traffic keeps upstream +/// behaviour instead of negotiating Expect on large conditional PUTs it never negotiated before. +inline static constexpr uint64_t DEFAULT_EXPECT_CONTINUE_MIN_BYTES = 0; inline static constexpr uint64_t DEFAULT_RETRY_ATTEMPTS = 500; inline static constexpr uint64_t DEFAULT_RETRY_INITIAL_DELAY_MS = 25; inline static constexpr uint64_t DEFAULT_RETRY_MAX_DELAY_MS = 5000; diff --git a/src/IO/WriteBufferFromFileBase.h b/src/IO/WriteBufferFromFileBase.h index 47dd4f5ed7ae..b60e951d8edf 100644 --- a/src/IO/WriteBufferFromFileBase.h +++ b/src/IO/WriteBufferFromFileBase.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include @@ -15,6 +16,12 @@ class WriteBufferFromFileBase : public BufferWithOwnMemory void sync() override = 0; virtual std::string getFileName() const = 0; + + /// The object-storage ETag/token the write produced, if any (e.g. the S3 PutObject / + /// CompleteMultipartUpload response ETag). Empty for backends that do not return a write-time + /// ETag (local files, etc.). Valid only after a successful finalize(). Lets content-addressed + /// callers record the just-written incarnation's token WITHOUT a follow-up HEAD. + virtual std::optional getResultObjectETag() const { return {}; } }; } diff --git a/src/IO/WriteBufferFromFileDecorator.h b/src/IO/WriteBufferFromFileDecorator.h index 07f843986bb0..cc05743642f5 100644 --- a/src/IO/WriteBufferFromFileDecorator.h +++ b/src/IO/WriteBufferFromFileDecorator.h @@ -19,6 +19,15 @@ class WriteBufferFromFileDecorator : public WriteBufferFromFileBase void preFinalize() override; + /// Forward the wrapped buffer's write-time ETag (if it is a file buffer that produced one), so a + /// decorated S3 buffer still lets content-addressed callers skip the post-write HEAD. + std::optional getResultObjectETag() const override + { + if (const auto * file_buf = dynamic_cast(impl.get())) + return file_buf->getResultObjectETag(); + return {}; + } + const WriteBuffer & getImpl() const { return *impl; } protected: diff --git a/src/IO/WriteBufferFromS3.cpp b/src/IO/WriteBufferFromS3.cpp index 2dbfc8d0f9fd..69fea4b686e5 100644 --- a/src/IO/WriteBufferFromS3.cpp +++ b/src/IO/WriteBufferFromS3.cpp @@ -64,6 +64,7 @@ namespace ErrorCodes extern const int S3_ERROR; extern const int INVALID_CONFIG_PARAMETER; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; } struct WriteBufferFromS3::PartData @@ -405,6 +406,15 @@ void WriteBufferFromS3::writeMultipartUpload() void WriteBufferFromS3::createMultipartUpload() { + if (write_settings.s3_force_single_part_upload) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "A conditional write would start a MULTIPART upload, but the target store enforces " + "no preconditions on CompleteMultipartUpload (GCS, measured 2026-07-03) — refusing " + "(silent-data-loss risk). The single-PUT budget is governed by the disk's " + "`gcs_max_conditional_put_bytes` S3 setting; the production-grade path for bigger conditional writes " + "(unconditional multipart to a temp key + conditional Compose) is not implemented yet. {}", + getShortLogDetails()); + LOG_TEST(limited_log, "Create multipart upload. {}", getShortLogDetails()); S3::CreateMultipartUploadRequest req; @@ -646,6 +656,12 @@ bool WriteBufferFromS3::completeMultipartUpload() if (!write_settings.object_storage_write_if_match.empty()) req.SetIfMatch(write_settings.object_storage_write_if_match); + /// Defense in depth only: a conditional write on a generation-token store never reaches this + /// request in the first place (WriteSettings forces a single PUT below the cap, so + /// createMultipartUpload throws first). Marking it anyway lets Task 4's native adapter reject a + /// conditional CompleteMultipartUpload outright if that invariant is ever violated. + req.setNativeConditional(write_settings.object_storage_request_mode == ObjectStorageRequestMode::NativeConditional); + Aws::S3::Model::CompletedMultipartUpload multipart_upload; for (size_t i = 0; i < multipart_tags.size(); ++i) { @@ -680,6 +696,7 @@ bool WriteBufferFromS3::completeMultipartUpload() if (outcome.IsSuccess()) { + object_etag = outcome.GetResult().GetETag(); LOG_TRACE(limited_log, "Multipart upload has completed. {}, Parts: {}", getShortLogDetails(), multipart_tags.size()); return true; } @@ -696,10 +713,19 @@ bool WriteBufferFromS3::completeMultipartUpload() } else { + /// Pass the canonical S3 error name: a conditional-write 412 is UNMODELED for the SDK + /// (the error type is UNKNOWN), so the name is the caller's only typed signal. throw S3Exception( +<<<<<<< HEAD error.GetErrorType(), "Message: {}, Key: {}, Bucket: {}, Tags: {}", error.GetMessage(), key, bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")); +======= + PreformattedMessage::create("Message: {}, Key: {}, Bucket: {}, Tags: {}", + outcome.GetError().GetMessage(), key, bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")), + outcome.GetError().GetErrorType(), + outcome.GetError().GetExceptionName()); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } @@ -735,6 +761,10 @@ S3::PutObjectRequest WriteBufferFromS3::getPutRequest(PartData & data) client_ptr->setKMSHeaders(req); + /// The actual PUT that produces a CAS incarnation token: eligible for the typed NativeConditional + /// HTTP mode when the caller marked this write as such (see WriteSettings::object_storage_request_mode). + req.setNativeConditional(write_settings.object_storage_request_mode == ObjectStorageRequestMode::NativeConditional); + return req; } @@ -774,6 +804,7 @@ void WriteBufferFromS3::makeSinglepartUpload(WriteBufferFromS3::PartData && data if (outcome.IsSuccess()) { + object_etag = outcome.GetResult().GetETag(); LOG_TRACE(limited_log, "Single part upload has completed. {}, size {}", getShortLogDetails(), content_length); return; } @@ -789,17 +820,20 @@ void WriteBufferFromS3::makeSinglepartUpload(WriteBufferFromS3::PartData && data else { /// PreconditionFailed is an expected response for conditional writes (e.g. If-None-Match: *), - /// not a genuine error — the caller handles it. - if (outcome.GetError().GetExceptionName() == "PreconditionFailed") + /// not a genuine error — the caller handles it (see `S3::isPreconditionFailedError`). + if (S3::isPreconditionFailedError(outcome.GetError())) LOG_INFO(log, "S3Exception name {}, Message: {}, bucket {}, key {}, object size {}", outcome.GetError().GetExceptionName(), outcome.GetError().GetMessage(), bucket, key, content_length); else LOG_ERROR(log, "S3Exception name {}, Message: {}, bucket {}, key {}, object size {}", outcome.GetError().GetExceptionName(), outcome.GetError().GetMessage(), bucket, key, content_length); + /// Pass the canonical S3 error name: a conditional-write 412 is UNMODELED for the SDK + /// (the error type is UNKNOWN), so the name is the caller's only typed signal. throw S3Exception( + PreformattedMessage::create("Message: {}, bucket {}, key {}, object size {}", + outcome.GetError().GetMessage(), bucket, key, content_length), outcome.GetError().GetErrorType(), - "Message: {}, bucket {}, key {}, object size {}", - outcome.GetError().GetMessage(), bucket, key, content_length); + outcome.GetError().GetExceptionName()); } } diff --git a/src/IO/WriteBufferFromS3.h b/src/IO/WriteBufferFromS3.h index 6a5e88875f27..fbce538c0a43 100644 --- a/src/IO/WriteBufferFromS3.h +++ b/src/IO/WriteBufferFromS3.h @@ -51,6 +51,10 @@ class WriteBufferFromS3 final : public WriteBufferFromFileBase void preFinalize() override; std::string getFileName() const override { return key; } void sync() override { next(); } + /// The object ETag from the PutObject / CompleteMultipartUpload response, captured on a + /// successful upload. Lets content-addressed callers record the written incarnation's token + /// without a follow-up HEAD. Valid only after a successful finalize(). + std::optional getResultObjectETag() const override { return object_etag; } private: /// Receives response from the server after sending all data. @@ -88,6 +92,10 @@ class WriteBufferFromS3 final : public WriteBufferFromFileBase const WriteSettings write_settings; const std::shared_ptr client_ptr; const std::optional object_metadata; + /// Set from the PutObject / CompleteMultipartUpload response ETag on a successful upload; read + /// by getResultObjectETag() after finalize(). Written by the upload worker, read after the + /// finalize barrier (happens-before), so no extra synchronization is needed. + std::optional object_etag; LoggerPtr log = getLogger("WriteBufferFromS3"); LogSeriesLimiterPtr limited_log = std::make_shared(log, 1, 5); diff --git a/src/IO/WriteSettings.h b/src/IO/WriteSettings.h index 4ca0180194e0..19aae46676f3 100644 --- a/src/IO/WriteSettings.h +++ b/src/IO/WriteSettings.h @@ -7,9 +7,40 @@ #include #endif +#include + namespace DB { +/// Per-write retry-behavior selector, resolved by the object storage that executes the write. +/// SingleAttempt: exactly one HTTP attempt, no SDK-transparent retries — for conditional writes +/// whose retry loop lives above the storage client (it must resolve an uncertain PUT before +/// reissuing). Backends without a SingleAttempt implementation report it via +/// IObjectStorage::supportsRetryProfile; writers must fail closed rather than fall through. +enum class ObjectStorageRetryProfile : uint8_t +{ + Default, + SingleAttempt, +}; + +/// Per-copy transport requirement, resolved by the object storage that executes the copy. +/// `NativeOnly` requires a provider-native same-store copy and forbids a client-side fallback. +enum class ObjectStorageCopyMode : uint8_t +{ + Default, + NativeOnly, +}; + +/// Per-request GCS conditional-dialect opt-in, carried alongside the write itself so it survives +/// into the object storage request that ends up on the wire (see `RequestWithNativeConditionalMode`). +/// NativeConditional: this write is content-addressed-storage-owned and may use GCS generation +/// tokens instead of the AWS-style ETag plumbing, when the client's HTTP layer supports it. +enum class ObjectStorageRequestMode : uint8_t +{ + Default, + NativeConditional, +}; + /// Settings to be passed to IDisk::writeFile() struct WriteSettings { @@ -26,6 +57,12 @@ struct WriteSettings size_t filesystem_cache_reserve_space_wait_lock_timeout_milliseconds = 1000; bool s3_allow_parallel_part_upload = true; + /// Overrides S3RequestSetting::check_objects_after_upload for this write (nullopt = no + /// override). Writers of CAS-MUTABLE keys (content-addressed shard manifests) set `false`: + /// such a key is legitimately replaced by a concurrent conditional PUT between this upload and + /// the check's HEAD, so the size comparison false-positives ("it's a bug in S3") under normal + /// contention. Integrity for those keys is the conditional PUT outcome + token, not a recheck. + std::optional s3_check_objects_after_upload_override; bool azure_allow_parallel_part_upload = true; bool use_adaptive_write_buffer = false; @@ -47,6 +84,29 @@ struct WriteSettings /// 0 disables. Honored only by metadata storages that support inline data. size_t inline_file_max_bytes = 0; + /// A conditional write on a generation-token store (GCS) must never take the multipart path: + /// GCS enforces no preconditions on CompleteMultipartUpload (measured 2026-07-03). The size + /// ceiling for this write comes from the object storage's own `gcs_max_conditional_put_bytes`. + bool s3_force_single_part_upload = false; + + /// Overrides S3RequestSetting::max_unexpected_write_error_retries (default 4) for this write. + /// WriteBufferFromS3::makeSinglepartUpload/completeMultipartUpload run their OWN retry loop above + /// the S3 client that reissues the identical request (WITH its If-None-Match/If-Match condition) + /// on a NO_SUCH_KEY response — a second retry-affecting layer a client-level override + /// (a client-level profile override) does not reach. A CAS conditional write sets this to 1 for + /// exactly one attempt at this layer too (RFC cas-s3-timeout-retry-control). 0 = no override. + size_t s3_max_unexpected_write_error_retries_override = 0; + + /// Selects the retry profile the object storage should execute this write under; see + /// ObjectStorageRetryProfile. + ObjectStorageRetryProfile object_storage_retry_profile = ObjectStorageRetryProfile::Default; + + /// Selects the transport requirement for an object storage copy; see `ObjectStorageCopyMode`. + ObjectStorageCopyMode object_storage_copy_mode = ObjectStorageCopyMode::Default; + + /// Selects the object storage request mode this write should carry; see ObjectStorageRequestMode. + ObjectStorageRequestMode object_storage_request_mode = ObjectStorageRequestMode::Default; + bool operator==(const WriteSettings & other) const = default; }; diff --git a/src/IO/tests/gtest_read_buffer_from_file_view.cpp b/src/IO/tests/gtest_read_buffer_from_file_view.cpp new file mode 100644 index 000000000000..b4a433361856 --- /dev/null +++ b/src/IO/tests/gtest_read_buffer_from_file_view.cpp @@ -0,0 +1,280 @@ +#include + +#include +#include + +#include + +using namespace DB; + +namespace +{ + +/// How the inner buffer reacts to setReadUntilPosition - the axis that broke B115. +enum class InnerMode : uint8_t +{ + /// Like local file descriptors: setReadUntilPosition is a no-op, the buffer is kept. + FileLike, + /// Like ReadBufferFromS3: a range change rebases the offset to the CONSUMER position and + /// DISCARDS the working buffer (the next nextImpl re-fetches from the consumer position). + RemoteLike, +}; + +/// A seekable ReadBufferFromFileBase over a string, reading at most `chunk` bytes per nextImpl, +/// with selectable setReadUntilPosition semantics. Mirrors the state conventions of real +/// implementations: `file_offset` is the absolute offset of working_buffer.end(). +class FakeInnerBuffer : public ReadBufferFromFileBase +{ +public: + FakeInnerBuffer(String data_, size_t chunk, InnerMode mode_) + : ReadBufferFromFileBase(chunk, nullptr, 0) + , data(std::move(data_)) + , mode(mode_) + { + } + + String getFileName() const override { return "fake_inner"; } + std::optional tryGetFileSize() override { return data.size(); } + size_t getFileOffsetOfBufferEnd() const override { return file_offset; } + off_t getPosition() override { return file_offset - available(); } + + off_t seek(off_t off, int whence) override + { + EXPECT_EQ(whence, SEEK_SET); + const size_t target = static_cast(off); + /// In-buffer seek (both real local and S3 buffers do this). + if (!working_buffer.empty() && target + working_buffer.size() >= file_offset && target < file_offset) + { + pos = working_buffer.end() - (file_offset - target); + return off; + } + resetWorkingBuffer(); + file_offset = target; + return off; + } + + void setReadUntilPosition(size_t position) override + { + if (read_until && *read_until == position) + return; + if (mode == InnerMode::RemoteLike) + { + /// ReadBufferFromS3: offset = getPosition(); resetWorkingBuffer(); impl.reset(); + file_offset = getPosition(); + resetWorkingBuffer(); + } + read_until = position; + } + + void setReadUntilEnd() override { setReadUntilPosition(data.size()); } + +private: + bool nextImpl() override + { + const size_t limit = read_until ? std::min(*read_until, data.size()) : data.size(); + if (file_offset >= limit) + return false; + const size_t to_read = std::min(limit - file_offset, internal_buffer.size()); + memcpy(internal_buffer.begin(), data.data() + file_offset, to_read); + working_buffer = Buffer(internal_buffer.begin(), internal_buffer.begin() + to_read); + file_offset += to_read; + return true; + } + + String data; + InnerMode mode; + size_t file_offset = 0; + std::optional read_until; +}; + +constexpr size_t kHeader = 256; /// the view's left bound (the CHCA envelope size in production) + +String makePayload(size_t size) +{ + String s(size, 0); + for (size_t i = 0; i < size; ++i) + s[i] = static_cast((i * 131 + 7) % 251); + return s; +} + +std::unique_ptr makeView(const String & payload, size_t chunk, InnerMode mode) +{ + String object = String(kHeader, '\xee') + payload; + auto inner = std::make_unique(std::move(object), chunk, mode); + return std::make_unique(std::move(inner), "viewed", kHeader, kHeader + payload.size()); +} + +String readExact(ReadBuffer & buf, size_t n) +{ + String out(n, 0); + buf.readStrict(out.data(), n); + return out; +} + +struct Case +{ + size_t chunk; + InnerMode mode; +}; + +class ReadBufferFromFileViewTest : public ::testing::TestWithParam +{ +}; + +} + +TEST_P(ReadBufferFromFileViewTest, SequentialReadWholeView) +{ + const auto [chunk, mode] = GetParam(); + const auto payload = makePayload(1000); + auto view = makeView(payload, chunk, mode); + + EXPECT_EQ(readExact(*view, payload.size()), payload); + EXPECT_TRUE(view->eof()); + EXPECT_EQ(view->getPosition(), static_cast(payload.size())); +} + +TEST_P(ReadBufferFromFileViewTest, SeekAndRead) +{ + const auto [chunk, mode] = GetParam(); + const auto payload = makePayload(1000); + auto view = makeView(payload, chunk, mode); + + for (size_t target : {size_t(0), size_t(700), size_t(20), size_t(21), size_t(999), size_t(5)}) + { + EXPECT_EQ(view->seek(target, SEEK_SET), static_cast(target)); + EXPECT_EQ(view->getPosition(), static_cast(target)); + EXPECT_EQ(readExact(*view, 1), payload.substr(target, 1)); + EXPECT_EQ(view->getPosition(), static_cast(target + 1)); + } +} + +/// B115 regression. The in-order MergeTree reader adjusts the right mark (setReadUntilPosition) +/// while the consumer is mid-buffer. A remote-like inner buffer legitimately discards its working +/// buffer on the range change; the view MUST keep reporting the consumer's position - before the +/// fix it teleported forward by the discarded bytes, so the next seek was treated as "already +/// there" and a stale block was re-served (duplicated + missing granules at the SQL level). +TEST_P(ReadBufferFromFileViewTest, SetReadUntilPositionMidBufferKeepsPosition) +{ + const auto [chunk, mode] = GetParam(); + const auto payload = makePayload(1000); + auto view = makeView(payload, chunk, mode); + + EXPECT_EQ(readExact(*view, 36), payload.substr(0, 36)); + EXPECT_EQ(view->getPosition(), 36); + + view->setReadUntilPosition(72); + EXPECT_EQ(view->getPosition(), 36) << "position must survive a right-bound change"; + + /// The consumer's next seek to its current position must be a no-op... + EXPECT_EQ(view->seek(36, SEEK_SET), 36); + /// ...and the bytes must continue from 36, not from a stale buffer. + EXPECT_EQ(readExact(*view, 36), payload.substr(36, 36)); +} + +/// Truncate-then-extend: the right bound shrinks below already-buffered data, the consumer reads +/// up to it, the bound is extended again. The continuation must produce the file's real bytes +/// (before the fix the view's incremental buffer-end accounting drifted from the inner buffer's). +TEST_P(ReadBufferFromFileViewTest, SetReadUntilTruncateThenExtend) +{ + const auto [chunk, mode] = GetParam(); + const auto payload = makePayload(1000); + auto view = makeView(payload, chunk, mode); + + EXPECT_EQ(readExact(*view, 10), payload.substr(0, 10)); + + view->setReadUntilPosition(30); + EXPECT_EQ(view->getPosition(), 10); + EXPECT_EQ(readExact(*view, 20), payload.substr(10, 20)); + EXPECT_TRUE(view->eof()); + EXPECT_EQ(view->getPosition(), 30); + + view->setReadUntilPosition(500); + EXPECT_EQ(view->getPosition(), 30); + EXPECT_EQ(readExact(*view, 100), payload.substr(30, 100)); + + view->setReadUntilEnd(); + EXPECT_EQ(readExact(*view, payload.size() - 130), payload.substr(130)); + EXPECT_TRUE(view->eof()); +} + +/// The exact shape of the failing compact-part in-order read: per granule, adjust the right +/// mark, seek to the granule's block, read it. Every block must contain its own bytes. +TEST_P(ReadBufferFromFileViewTest, GranulePatternRegression) +{ + const auto [chunk, mode] = GetParam(); + constexpr size_t block = 36; + constexpr size_t blocks = 20; + const auto payload = makePayload(block * blocks); + auto view = makeView(payload, chunk, mode); + + for (size_t g = 0; g < blocks; ++g) + { + view->setReadUntilPosition(std::min((g + 2) * block, payload.size())); + EXPECT_EQ(view->seek(g * block, SEEK_SET), static_cast(g * block)); + EXPECT_EQ(readExact(*view, block), payload.substr(g * block, block)) << "block " << g; + } +} + +/// Randomized conformance battery against a golden model. +TEST_P(ReadBufferFromFileViewTest, RandomizedOps) +{ + const auto [chunk, mode] = GetParam(); + const auto payload = makePayload(2000); + + for (unsigned seed = 1; seed <= 5; ++seed) + { + auto view = makeView(payload, chunk, mode); + size_t model_pos = 0; + size_t model_until = payload.size(); + unsigned rng = seed; + auto next_rand = [&rng] { rng = rng * 1103515245 + 12345; return (rng >> 8) % 1000; }; + + for (int step = 0; step < 300; ++step) + { + switch (next_rand() % 3) + { + case 0: /// read up to the current until-bound + { + const size_t want = next_rand() % 64; + const size_t n = std::min(want, model_until - model_pos); + if (n) + { + ASSERT_EQ(readExact(*view, n), payload.substr(model_pos, n)) << "seed " << seed << " step " << step; + model_pos += n; + } + break; + } + case 1: /// seek (never beyond the current until-bound - the consumer contract: + /// the right mark always covers the ranges being read) + { + const size_t target = next_rand() % (model_until + 1); + ASSERT_EQ(view->seek(target, SEEK_SET), static_cast(target)); + model_pos = target; + break; + } + case 2: /// move the right bound (never below the consumer position) + { + const size_t until = model_pos + next_rand() % (payload.size() - model_pos + 1); + view->setReadUntilPosition(until); + model_until = until; + break; + } + default: + UNREACHABLE(); + } + ASSERT_EQ(view->getPosition(), static_cast(model_pos)) << "seed " << seed << " step " << step; + } + } +} + +INSTANTIATE_TEST_SUITE_P( + ChunksAndModes, + ReadBufferFromFileViewTest, + ::testing::Values( + Case{7, InnerMode::FileLike}, + Case{7, InnerMode::RemoteLike}, + Case{108, InnerMode::FileLike}, + Case{108, InnerMode::RemoteLike}, + Case{1 << 20, InnerMode::FileLike}, + Case{1 << 20, InnerMode::RemoteLike})); diff --git a/src/IO/tests/gtest_read_buffer_from_memory.cpp b/src/IO/tests/gtest_read_buffer_from_memory.cpp new file mode 100644 index 000000000000..b7955f816e79 --- /dev/null +++ b/src/IO/tests/gtest_read_buffer_from_memory.cpp @@ -0,0 +1,19 @@ +#include + +#include + +#include + +using namespace DB; + +/// An empty file materialized into an OWNED in-memory buffer must construct without undefined +/// behaviour: std::memcpy's pointer arguments are __attribute__((nonnull)), so memcpy(dst, nullptr, 0) +/// -- which an empty std::string_view (data() == nullptr) produces -- is UB that the asan_ubsan lane +/// aborts on (STID 5930-5afa, PR #2073). The buffer must construct and be immediately at EOF. +TEST(ReadBufferFromMemoryFileBase, EmptyOwnedBufferConstructsWithoutUB) +{ + /// ReadBufferFromMemoryFileBase's constructor is protected; ReadBufferFromOwnMemoryFile is the + /// public concrete class that always passes owns_memory=true, exercising the guarded memcpy path. + ReadBufferFromOwnMemoryFile buf("empty", std::string_view{}); + EXPECT_TRUE(buf.eof()); +} diff --git a/src/IO/tests/gtest_s3_auth_settings.cpp b/src/IO/tests/gtest_s3_auth_settings.cpp new file mode 100644 index 000000000000..d3bad451d438 --- /dev/null +++ b/src/IO/tests/gtest_s3_auth_settings.cpp @@ -0,0 +1,40 @@ +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; + +namespace DB::S3AuthSetting +{ + extern const S3AuthSettingsUInt64 gcs_max_conditional_put_bytes; +} + +namespace +{ +Poco::AutoPtr makeDiskConfig(const std::string & inner) +{ + std::istringstream iss("" + inner + ""); + return new Poco::Util::XMLConfiguration(iss); +} +} + +/// The cap is a property of the GCS conditional-write dialect, so it is read from the disk block +/// unprefixed, exactly like `gcs_issue_compose_request` beside it. +TEST(S3AuthSettingsConfig, GcsConditionalPutCapParsesFromDiskBlock) +{ + Settings query_settings; + + auto with_override = makeDiskConfig( + "4096"); + S3::S3AuthSettings overridden(*with_override, query_settings, "disk"); + EXPECT_EQ(overridden[S3AuthSetting::gcs_max_conditional_put_bytes].value, 4096u); + + auto without = makeDiskConfig("http://x/y"); + S3::S3AuthSettings defaulted(*without, query_settings, "disk"); + EXPECT_EQ(defaulted[S3AuthSetting::gcs_max_conditional_put_bytes].value, + S3::DEFAULT_GCS_MAX_CONDITIONAL_PUT_BYTES); +} diff --git a/src/IO/tests/gtest_writebuffer_s3.cpp b/src/IO/tests/gtest_writebuffer_s3.cpp index ed90522ffa3f..93ac4a74acff 100644 --- a/src/IO/tests/gtest_writebuffer_s3.cpp +++ b/src/IO/tests/gtest_writebuffer_s3.cpp @@ -19,7 +19,12 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +#include +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include @@ -33,14 +38,23 @@ #include #include #include +<<<<<<< HEAD +======= +#include +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include #include +#include +#include #include +#include #include +#include + namespace DB { @@ -56,10 +70,16 @@ namespace Setting extern const SettingsUInt64 s3_upload_part_size_multiply_parts_count_threshold; } +namespace S3RequestSetting +{ + extern const S3RequestSettingsBool allow_native_copy; +} + namespace ErrorCodes { extern const int LOGICAL_ERROR; extern const int S3_ERROR; + extern const int NOT_IMPLEMENTED; } } @@ -191,6 +211,9 @@ struct EventCounts size_t copyObject = 0; size_t uploadPartCopy = 0; size_t writtenSize = 0; + size_t copyObject = 0; + size_t deleteObject = 0; + size_t getBucketVersioning = 0; size_t totalRequestsCount() const { @@ -237,6 +260,9 @@ struct InjectionModel DeclareInjectCall(CompleteMultipartUpload) DeclareInjectCall(AbortMultipartUpload) DeclareInjectCall(UploadPart) + DeclareInjectCall(CopyObject) + DeclareInjectCall(DeleteObject) + DeclareInjectCall(GetBucketVersioning) #undef DeclareInjectCall }; @@ -296,6 +322,9 @@ struct Client : DB::S3::Client { ++counters.putObject; + if (const auto * wrapper = dynamic_cast(&request)) + last_put_object_native_conditional = wrapper->isNativeConditional(); + if (injections) { if (auto opt_val = injections->call(request)) @@ -311,6 +340,7 @@ struct Client : DB::S3::Client Aws::S3::Model::PutObjectOutcome outcome; Aws::S3::Model::PutObjectResult result(outcome.GetResultWithOwnership()); + result.SetETag("etag-singlepart-" + request.GetKey()); return result; } @@ -345,6 +375,13 @@ struct Client : DB::S3::Client { ++counters.headObject; + /// The request's DYNAMIC type is still the production `DB::S3::HeadObjectRequest` wrapper -- + /// this override only sees it through the SDK base-class reference. Mirrors the dynamic_cast + /// `Client::BuildHttpRequest` itself does, so a test can observe the mark this mock never + /// forwards through an HTTP layer. + if (const auto * wrapper = dynamic_cast(&request)) + last_head_object_native_conditional = wrapper->isNativeConditional(); + if (injections) { if (auto opt_val = injections->call(request)) @@ -365,6 +402,9 @@ struct Client : DB::S3::Client { ++counters.multiUploadCreate; + if (const auto * wrapper = dynamic_cast(&request)) + last_create_multipart_native_conditional = wrapper->isNativeConditional(); + if (injections) { if (auto opt_val = injections->call(request)) @@ -385,6 +425,9 @@ struct Client : DB::S3::Client { ++counters.uploadParts; + if (const auto * wrapper = dynamic_cast(&request)) + last_upload_part_native_conditional = wrapper->isNativeConditional(); + if (injections) { if (auto opt_val = injections->call(request)) @@ -408,6 +451,9 @@ struct Client : DB::S3::Client { ++counters.multiUploadComplete; + if (const auto * wrapper = dynamic_cast(&request)) + last_complete_multipart_native_conditional = wrapper->isNativeConditional(); + if (injections) { if (auto opt_val = injections->call(request)) @@ -425,6 +471,7 @@ struct Client : DB::S3::Client bStore.CompleteMPU(request.GetKey(), request.GetUploadId(), etags); Aws::S3::Model::CompleteMultipartUploadResult result; + result.SetETag("etag-multipart-" + request.GetKey()); return Aws::S3::Model::CompleteMultipartUploadOutcome(result); } @@ -447,12 +494,16 @@ struct Client : DB::S3::Client return Aws::S3::Model::AbortMultipartUploadOutcome(result); } +<<<<<<< HEAD /// Whole-object server-side copy. A CopyObject request carries no byte range, so it always copies the /// entire source object -- modelling the real S3 behaviour that makes it unsafe for a partial range. +======= +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) Aws::S3::Model::CopyObjectOutcome CopyObject(const Aws::S3::Model::CopyObjectRequest & request) const override { ++counters.copyObject; +<<<<<<< HEAD const auto [src_bucket, src_key] = splitCopySource(request.GetCopySource()); const String & src_data = store->GetBucketStore(src_bucket).objects[src_key]; store->GetBucketStore(request.GetBucket()).PutObject(request.GetKey(), src_data); @@ -487,11 +538,88 @@ struct Client : DB::S3::Client Aws::S3::Model::UploadPartCopyResult result; result.SetCopyPartResult(copy_part_result); return Aws::S3::Model::UploadPartCopyOutcome(result); +======= + if (const auto * wrapper = dynamic_cast(&request)) + last_copy_object_native_conditional = wrapper->isNativeConditional(); + + last_copy_object_if_match = request.IfMatchHasBeenSet(); + last_copy_object_if_none_match = request.IfNoneMatchHasBeenSet(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + /// CopySource is "/"; parse it back apart to look the source object up + /// (both source and destination live in the same S3MemStrore in these tests). + const std::string & copy_source = request.GetCopySource(); + const size_t sep = copy_source.find('/'); + chassert(sep != std::string::npos); + const std::string src_bucket_name = copy_source.substr(0, sep); + const std::string src_key = copy_source.substr(sep + 1); + + auto & src_store = store->GetBucketStore(src_bucket_name); + const std::string data = src_store.objects.at(src_key); + + auto & dst_store = store->GetBucketStore(request.GetBucket()); + dst_store.PutObject(request.GetKey(), data); + + Aws::S3::Model::CopyObjectResult result; + Aws::S3::Model::CopyObjectResultDetails details; + details.SetETag("etag-copy-" + request.GetKey()); + result.SetCopyObjectResultDetails(details); + return Aws::S3::Model::CopyObjectOutcome(result); + } + + Aws::S3::Model::DeleteObjectOutcome DeleteObject(const Aws::S3::Model::DeleteObjectRequest & request) const override + { + ++counters.deleteObject; + + if (const auto * wrapper = dynamic_cast(&request)) + last_delete_object_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + bStore.objects.erase(request.GetKey()); + + Aws::S3::Model::DeleteObjectResult result; + return Aws::S3::Model::DeleteObjectOutcome(result); + } + + Aws::S3::Model::GetBucketVersioningOutcome GetBucketVersioning(const Aws::S3::Model::GetBucketVersioningRequest & request) const override + { + ++counters.getBucketVersioning; + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + Aws::S3::Model::GetBucketVersioningResult result; + result.SetStatus(Aws::S3::Model::BucketVersioningStatus::Enabled); + return Aws::S3::Model::GetBucketVersioningOutcome(result); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::shared_ptr store; mutable EventCounts counters; mutable std::shared_ptr injections; + mutable bool last_head_object_native_conditional = false; + mutable bool last_delete_object_native_conditional = false; + mutable bool last_put_object_native_conditional = false; + mutable bool last_create_multipart_native_conditional = false; + mutable bool last_upload_part_native_conditional = false; + mutable bool last_complete_multipart_native_conditional = false; + mutable bool last_copy_object_native_conditional = false; + mutable bool last_copy_object_if_match = false; + mutable bool last_copy_object_if_none_match = false; void resetCounters() const { counters = {}; } }; @@ -535,6 +663,7 @@ struct UploadPartFailIngection: InjectionModel } }; +<<<<<<< HEAD /// Fails the first `fail_times` CompleteMultipartUpload calls with the un-typed MinIO `InvalidPart` /// eventual-consistency error, then lets the real mock store handle the rest. The AWS SDK cannot map /// InvalidPart to a typed model error, so it produces UNKNOWN as the error type and keeps @@ -558,6 +687,37 @@ struct CompleteMPUInvalidPartOnceIngection : InjectionModel size_t fail_times; size_t calls = 0; +======= +/// Injects an arbitrary AWSError on DeleteObject -- used to drive the conditional-remove +/// (`removeObjectIfTokenMatches`) outcome mapping: a 412-shaped error (exception name "PreconditionFailed", +/// matched by `S3::isPreconditionFailedError`) must map to `ConditionalRemoveOutcome::TokenMismatch`, and a +/// 404-shaped error (a `NO_SUCH_KEY`/`RESOURCE_NOT_FOUND`/`NO_SUCH_BUCKET` error type, matched by +/// `S3::isNotFoundError`) must map to `ConditionalRemoveOutcome::NotFound`. +struct DeleteObjectErrorInjection: InjectionModel +{ + explicit DeleteObjectErrorInjection(Aws::Client::AWSError error_) : error(std::move(error_)) {} + + std::optional call(const Aws::S3::Model::DeleteObjectRequest & /*request*/) override + { + return error; + } + + Aws::Client::AWSError error; +}; + +/// Injects an arbitrary `CopyObject` error to exercise ordinary-copy fallback and native-only +/// fail-close behavior. +struct CopyObjectErrorInjection: InjectionModel +{ + explicit CopyObjectErrorInjection(Aws::Client::AWSError error_) : error(std::move(error_)) {} + + std::optional call(const Aws::S3::Model::CopyObjectRequest & /*request*/) override + { + return error; + } + + Aws::Client::AWSError error; +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) }; struct BaseSyncPolicy @@ -655,7 +815,7 @@ class WBS3Test : public ::testing::Test return *async_policy; } - std::unique_ptr getWriteBuffer(String file_name = "file") + std::unique_ptr getWriteBuffer(String file_name = "file", const WriteSettings & write_settings = {}) { S3::S3RequestSettings request_settings; request_settings.updateFromSettings(settings, /* if_changed */true, /* validate_settings */false); @@ -672,7 +832,8 @@ class WBS3Test : public ::testing::Test request_settings, nullptr, std::nullopt, - getAsyncPolicy().getScheduler()); + getAsyncPolicy().getScheduler(), + write_settings); } void setInjectionModel(std::shared_ptr injections_) @@ -1234,6 +1395,33 @@ TEST_F(WBS3Test, PrefinalizeCalledMultipleTimes) { #endif } +// The object ETag from the PutObject / CompleteMultipartUpload response is surfaced via +// getResultObjectETag() after a successful finalize() — lets content-addressed callers record the +// just-written incarnation's token WITHOUT a follow-up HEAD (CA head-after-put elimination). +TEST_F(WBS3Test, ResultObjectETagIsCaptured) { + // Singlepart upload: the PutObject response ETag. + { + auto buffer = getWriteBuffer("singlepart-file"); + writeAsOneBlock(*buffer, 10); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + ASSERT_TRUE(buffer->getResultObjectETag().has_value()); + ASSERT_EQ(*buffer->getResultObjectETag(), "etag-singlepart-singlepart-file"); + } + + // Multipart upload: the final object ETag comes from CompleteMultipartUpload, NOT a per-part tag. + { + getSettings()[Setting::s3_max_single_part_upload_size] = 0; // no single part — force multipart + getSettings()[Setting::s3_min_upload_part_size] = 1; + auto buffer = getWriteBuffer("multipart-file"); + writeAsOneBlock(*buffer, 10); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + ASSERT_TRUE(buffer->getResultObjectETag().has_value()); + ASSERT_EQ(*buffer->getResultObjectETag(), "etag-multipart-multipart-file"); + } +} + TEST_P(SyncAsync, EmptyFile) { getSettings()[Setting::s3_check_objects_after_upload] = true; @@ -1468,6 +1656,402 @@ TEST_P(SyncAsync, StrictUploadPartSize) { } } +/// Task 3: the actual PutObject request a single-part upload issues must carry the typed +/// NativeConditional mode exactly when the caller's WriteSettings asked for it -- the old blanket GCS +/// dialect stays authoritative over the wire until a later task; this only proves the mark reaches +/// the production request object (mirrors the HEAD/DELETE marking tests in +/// S3ObjectStorageConditionalOpsTest below). +TEST_F(WBS3Test, PutObjectNativeConditionalModePropagates) +{ + WriteSettings ws; + ws.object_storage_request_mode = ObjectStorageRequestMode::NativeConditional; + + auto buffer = getWriteBuffer("native_conditional_put", ws); + buffer->write('A'); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + + EXPECT_EQ(client->counters.putObject, 1); + EXPECT_TRUE(client->last_put_object_native_conditional); +} + +/// The control: an ordinary (Default-mode) single-part upload must NOT pick up the mark. +TEST_F(WBS3Test, PutObjectOrdinaryWriteRemainsDefault) +{ + auto buffer = getWriteBuffer("ordinary_put"); + buffer->write('A'); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + + EXPECT_EQ(client->counters.putObject, 1); + EXPECT_FALSE(client->last_put_object_native_conditional); +} + +/// A multipart upload's CompleteMultipartUpload request must carry the mode too (Task 4's native +/// adapter consumes it as a defense-in-depth guard against a conditional multipart completion), while +/// CreateMultipartUpload and UploadPart -- which no consumer needs marked -- must NOT. +TEST_F(WBS3Test, CompleteMultipartUploadNativeConditionalModePropagatesButCreateAndUploadPartDoNot) +{ + getSettings()[Setting::s3_max_single_part_upload_size] = 0; // force multipart + getSettings()[Setting::s3_min_upload_part_size] = 1; + + WriteSettings ws; + ws.object_storage_request_mode = ObjectStorageRequestMode::NativeConditional; + + auto buffer = getWriteBuffer("native_conditional_multipart", ws); + buffer->write('A'); + buffer->next(); + buffer->write('A'); + + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + + EXPECT_EQ(client->counters.multiUploadComplete, 1); + EXPECT_TRUE(client->last_complete_multipart_native_conditional); + EXPECT_FALSE(client->last_create_multipart_native_conditional); + EXPECT_FALSE(client->last_upload_part_native_conditional); +} + +/// Mock-S3 coverage for token-exact removal plus ordinary and native-only copy modes. +class S3ObjectStorageConditionalOpsTest : public ::testing::Test +{ +public: + const String bucket = "cond-ops-bucket"; + const String disk_name = "cond-ops-disk"; + + std::shared_ptr object_storage; + MockS3::Client * mock_client = nullptr; + std::shared_ptr store; + +protected: + std::shared_ptr createObjectStorage( + const String & storage_bucket, + const String & storage_disk_name, + bool allow_native_copy, + MockS3::Client *& client_out) + { + auto owned_client = std::make_unique(store); + client_out = owned_client.get(); + + auto settings = std::make_unique(); + settings->request_settings[S3RequestSetting::allow_native_copy] = allow_native_copy; + + S3::URI uri; + uri.bucket = storage_bucket; + S3Capabilities capabilities; + ObjectStorageKeyGeneratorPtr key_generator; + + return std::make_shared( + std::move(owned_client), + std::move(settings), + std::move(uri), + capabilities, + key_generator, + storage_disk_name); + } + + void resetObjectStorage(bool allow_native_copy = true) + { + object_storage = createObjectStorage(bucket, disk_name, allow_native_copy, mock_client); + } + + void SetUp() override + { + /// `removeObjectIfTokenMatches` and `copyObject` call `BlobStorageLogWriter::create`, which + /// falls back to `Context::getGlobalContextInstance` + /// when there is no query context. Force that global context to exist (harmless -- blob + /// storage logging stays off by default) regardless of which other gtest TU ran first. + (void)getContext(); + + store = std::make_shared(); + store->CreateBucket(bucket); + resetObjectStorage(); + } + + void TearDown() override + { + object_storage.reset(); + mock_client = nullptr; + store.reset(); + } +}; + +TEST_F(S3ObjectStorageConditionalOpsTest, DefaultCopyObjectMayFallback) +{ + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + mock_client->setInjectionModel(std::make_shared( + Aws::Client::AWSError(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied", "access denied", false))); + + object_storage->copyObject( + StoredObject("src-key"), StoredObject("dst-key"), ReadSettings{}, WriteSettings{}, std::nullopt); + + EXPECT_TRUE(object_storage->supportsCopyMode(ObjectStorageCopyMode::Default)); + EXPECT_TRUE(store->GetBucketStore(bucket).objects.contains("dst-key")); + EXPECT_EQ(mock_client->counters.copyObject, 1); + EXPECT_EQ(mock_client->counters.putObject, 1); + EXPECT_FALSE(mock_client->last_copy_object_native_conditional); + EXPECT_FALSE(mock_client->last_copy_object_if_match); + EXPECT_FALSE(mock_client->last_copy_object_if_none_match); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCopyObjectUsesNativeTransport) +{ + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + object_storage->copyObject( + StoredObject("src-key"), StoredObject("dst-key"), ReadSettings{}, write_settings, std::nullopt); + + EXPECT_TRUE(object_storage->supportsCopyMode(ObjectStorageCopyMode::NativeOnly)); + EXPECT_EQ(store->GetBucketStore(bucket).objects.at("dst-key"), "hello-world"); + EXPECT_EQ(mock_client->counters.copyObject, 1); + EXPECT_EQ(mock_client->counters.putObject, 0); + EXPECT_FALSE(mock_client->last_copy_object_native_conditional); + EXPECT_FALSE(mock_client->last_copy_object_if_match); + EXPECT_FALSE(mock_client->last_copy_object_if_none_match); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCopyObjectNeverFallsBack) +{ + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + mock_client->setInjectionModel(std::make_shared( + Aws::Client::AWSError(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied", "access denied", false))); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + EXPECT_THROW( + object_storage->copyObject( + StoredObject("src-key"), StoredObject("dst-key"), ReadSettings{}, write_settings, std::nullopt), + DB::S3Exception); + + EXPECT_EQ(mock_client->counters.copyObject, 1); + EXPECT_EQ(mock_client->counters.putObject, 0); + EXPECT_FALSE(store->GetBucketStore(bucket).objects.contains("dst-key")); + + resetObjectStorage(/*allow_native_copy=*/false); + EXPECT_TRUE(object_storage->supportsCopyMode(ObjectStorageCopyMode::Default)); + EXPECT_FALSE(object_storage->supportsCopyMode(ObjectStorageCopyMode::NativeOnly)); + EXPECT_THROW({ + try + { + object_storage->copyObject( + StoredObject("src-key"), StoredObject("disabled-dst-key"), ReadSettings{}, write_settings, std::nullopt); + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::NOT_IMPLEMENTED); + throw; + } + }, DB::Exception); + EXPECT_EQ(mock_client->counters.copyObject, 0); + EXPECT_EQ(mock_client->counters.putObject, 0); + EXPECT_FALSE(store->GetBucketStore(bucket).objects.contains("disabled-dst-key")); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCrossStorageCopyUsesNativeTransport) +{ + const String destination_bucket = "cond-ops-destination-bucket"; + store->CreateBucket(destination_bucket); + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + + MockS3::Client * destination_client = nullptr; + auto destination_storage = createObjectStorage( + destination_bucket, "cond-ops-destination-disk", /*allow_native_copy=*/true, destination_client); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + object_storage->copyObjectToAnotherObjectStorage( + StoredObject("src-key"), + StoredObject("dst-key"), + ReadSettings{}, + write_settings, + *destination_storage, + std::nullopt); + + EXPECT_EQ(store->GetBucketStore(destination_bucket).objects.at("dst-key"), "hello-world"); + EXPECT_EQ(destination_client->counters.copyObject, 1); + EXPECT_EQ(destination_client->counters.putObject, 0); + EXPECT_EQ(mock_client->counters.getObject, 0); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCrossStorageCopyNeverFallsBackAfterAccessDenied) +{ + const String destination_bucket = "cond-ops-destination-bucket"; + store->CreateBucket(destination_bucket); + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + + MockS3::Client * destination_client = nullptr; + auto destination_storage = createObjectStorage( + destination_bucket, "cond-ops-destination-disk", /*allow_native_copy=*/true, destination_client); + destination_client->setInjectionModel(std::make_shared( + Aws::Client::AWSError(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied", "access denied", false))); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + EXPECT_THROW( + object_storage->copyObjectToAnotherObjectStorage( + StoredObject("src-key"), + StoredObject("dst-key"), + ReadSettings{}, + write_settings, + *destination_storage, + std::nullopt), + DB::S3Exception); + + EXPECT_EQ(destination_client->counters.copyObject, 1); + EXPECT_EQ(destination_client->counters.putObject, 0); + EXPECT_EQ(mock_client->counters.getObject, 0); + EXPECT_FALSE(store->GetBucketStore(destination_bucket).objects.contains("dst-key")); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCrossStorageCopyNeverFallsBackWhenNativeCopyIsDisabled) +{ + const String destination_bucket = "cond-ops-destination-bucket"; + store->CreateBucket(destination_bucket); + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + + resetObjectStorage(/*allow_native_copy=*/false); + MockS3::Client * destination_client = nullptr; + auto destination_storage = createObjectStorage( + destination_bucket, "cond-ops-destination-disk", /*allow_native_copy=*/true, destination_client); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + EXPECT_THROW({ + try + { + object_storage->copyObjectToAnotherObjectStorage( + StoredObject("src-key"), + StoredObject("dst-key"), + ReadSettings{}, + write_settings, + *destination_storage, + std::nullopt); + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::NOT_IMPLEMENTED); + throw; + } + }, DB::Exception); + + EXPECT_EQ(destination_client->counters.copyObject, 0); + EXPECT_EQ(destination_client->counters.putObject, 0); + EXPECT_EQ(mock_client->counters.getObject, 0); + EXPECT_FALSE(store->GetBucketStore(destination_bucket).objects.contains("dst-key")); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, NativeOnlyCopyToNonS3StorageFailsClosed) +{ + store->GetBucketStore(bucket).PutObject("src-key", "hello-world"); + + Poco::TemporaryFile destination_directory; + destination_directory.createDirectories(); + LocalObjectStorage destination_storage(LocalObjectStorageSettings( + "cond-ops-local-destination", destination_directory.path(), /*read_only_=*/false)); + + WriteSettings write_settings; + write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; + EXPECT_THROW({ + try + { + object_storage->copyObjectToAnotherObjectStorage( + StoredObject("src-key"), + StoredObject("dst-key"), + ReadSettings{}, + write_settings, + destination_storage, + std::nullopt); + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::NOT_IMPLEMENTED); + throw; + } + }, DB::Exception); + + EXPECT_EQ(mock_client->counters.getObject, 0); + EXPECT_FALSE(destination_storage.exists(StoredObject("dst-key"))); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, RemoveObjectIfTokenMatchesSuccess) +{ + store->GetBucketStore(bucket).PutObject("key1", "data"); + + auto result = object_storage->removeObjectIfTokenMatches(StoredObject("key1"), "etag-1"); + + ASSERT_EQ(result.outcome, ConditionalRemoveOutcome::Removed); + ASSERT_EQ(mock_client->counters.deleteObject, 1); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, RemoveObjectIfTokenMatchesPreconditionFailedIsTokenMismatch) +{ + mock_client->setInjectionModel(std::make_shared( + Aws::Client::AWSError(Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed", "precondition failed", false))); + + auto result = object_storage->removeObjectIfTokenMatches(StoredObject("key1"), "stale-etag"); + + ASSERT_EQ(result.outcome, ConditionalRemoveOutcome::TokenMismatch); +} + +TEST_F(S3ObjectStorageConditionalOpsTest, RemoveObjectIfTokenMatchesNotFoundIsNotFound) +{ + mock_client->setInjectionModel(std::make_shared( + Aws::Client::AWSError(Aws::S3::S3Errors::NO_SUCH_KEY, "NoSuchKey", "not found", false))); + + auto result = object_storage->removeObjectIfTokenMatches(StoredObject("missing-key"), "any-etag"); + + ASSERT_EQ(result.outcome, ConditionalRemoveOutcome::NotFound); +} + +/// `tryGetObjectMetadataWithNativeToken` must mark its HEAD wrapper eligible for the typed +/// NativeConditional mode — the mark is what makes a GCS-mode client apply generation semantics to +/// this HEAD — and it must keep tryGetObjectMetadata's existing missing-object contract of returning +/// nullopt. +TEST_F(S3ObjectStorageConditionalOpsTest, NativeTokenHeadIsMarkedAndMissingIsNullopt) +{ + store->GetBucketStore(bucket).PutObject("existing-key", "some-body"); + + auto found = object_storage->tryGetObjectMetadataWithNativeToken("existing-key", /*with_tags=*/false); + ASSERT_TRUE(found.has_value()); + EXPECT_EQ(found->size_bytes, 9u); + EXPECT_TRUE(mock_client->last_head_object_native_conditional); + + auto missing = object_storage->tryGetObjectMetadataWithNativeToken("missing-key", /*with_tags=*/false); + EXPECT_FALSE(missing.has_value()); + EXPECT_TRUE(mock_client->last_head_object_native_conditional); +} + +/// The token-exact DELETE removeObjectIfTokenMatches issues (CAS's `If-Match` reclaim) must be marked +/// eligible for the typed NativeConditional mode -- it is the exact-delete path a GCS generation token +/// belongs on. +TEST_F(S3ObjectStorageConditionalOpsTest, GenerationDeleteUsesNativeConditionalMode) +{ + store->GetBucketStore(bucket).PutObject("key1", "data"); + + auto result = object_storage->removeObjectIfTokenMatches(StoredObject("key1"), "etag-1"); + + ASSERT_EQ(result.outcome, ConditionalRemoveOutcome::Removed); + EXPECT_TRUE(mock_client->last_delete_object_native_conditional); +} + +/// An ordinary (non-conditional) delete must NOT pick up the native mark -- only the exact-token +/// delete path is content-addressed-storage-owned. +TEST_F(S3ObjectStorageConditionalOpsTest, OrdinaryDeleteRemainsDefault) +{ + store->GetBucketStore(bucket).PutObject("key1", "data"); + + object_storage->removeObjectIfExists(StoredObject("key1")); + + /// Pin that the ordinary delete actually reached the singular DeleteObject hook this test reads -- + /// otherwise a future refactor onto the batch DeleteObjects path would silently stop exercising + /// this assertion (the field would sit unwritten at its `false` initializer) and this test would + /// keep passing while proving nothing. + ASSERT_EQ(mock_client->counters.deleteObject, 1); + EXPECT_FALSE(mock_client->last_delete_object_native_conditional); +} + [[maybe_unused]] static String fillStringWithPattern(String pattern, int n) { String data; diff --git a/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp b/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp new file mode 100644 index 000000000000..fde269c4c999 --- /dev/null +++ b/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp @@ -0,0 +1,114 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +ColumnsDescription ContentAddressedGarbageCollectionLogElement::getColumnsDescription() +{ + auto type_enum = std::make_shared(DataTypeEnum8::Values{ + {"Start", static_cast(START)}, {"Finish", static_cast(FINISH)}, + {"Phase", static_cast(PHASE)}}); + auto outcome_enum = std::make_shared(DataTypeEnum8::Values{ + {"Unknown", static_cast(UNKNOWN)}, {"Success", static_cast(SUCCESS)}, + {"NotALeader", static_cast(NOT_A_LEADER)}, {"Error", static_cast(FAILED)}, + {"Deferred", static_cast(DEFERRED)}}); + auto trigger_enum = std::make_shared(DataTypeEnum8::Values{ + {"Scheduled", static_cast(SCHEDULED)}, {"Manual", static_cast(MANUAL)}}); + auto lc_string = std::make_shared(std::make_shared()); + + return ColumnsDescription + { + {"hostname", lc_string, "Host name of the server executing the round."}, + {"event_date", std::make_shared(), "Event date."}, + {"event_time", std::make_shared(), "Event time."}, + {"event_time_microseconds", std::make_shared(6), "Event time with microseconds."}, + {"event_type", type_enum, "Start or Finish of a GC round, or one Phase of it."}, + {"disk_name", lc_string, "Content-addressed disk the round ran on."}, + {"server_root_id", lc_string, "Identifies the mount whose GC scheduler ran this round. Distinguishes concurrent mounters of the same shared pool; join on this column when correlating rounds against `system.cas_mounts`."}, + {"gc_id", std::make_shared(), "GC scheduler instance id (which mounter)."}, + {"trigger", trigger_enum, "Scheduled (background tick) or Manual (SYSTEM command)."}, + {"round", std::make_shared(), "GC round number (0 on Start)."}, + {"outcome", outcome_enum, "Unknown (Start) / Success (led, folded, and completed) / NotALeader (another replica holds the GC lease) / Deferred (led but took the skip-unchanged fast path -- no fold ran) / Error (the round threw)."}, + {"candidates_marked", std::make_shared(), "Objects retired (marked) this round."}, + {"objects_deleted", std::make_shared(), "Objects physically deleted this round."}, + {"objects_absent", std::make_shared(), "Retire candidates found already absent."}, + {"objects_replaced", std::make_shared(), "412-saves (a resurrection won the race)."}, + {"objects_spared", std::make_shared(), "Candidates spared (in-degree > 0 at recheck)."}, + {"manifests_deleted", std::make_shared(), "Owner-removed manifest bodies physically deleted this round (counted separately from blob deletes, B11)."}, + {"entries_condemned", std::make_shared(), "Retired entries newly condemned this round (retired-cursor pipeline stage 1)."}, + {"entries_graduated", std::make_shared(), "Retired entries newly floor-passed and republished delete_pending this round (stage 2; deleted the NEXT round)."}, + {"entries_redeleted", std::make_shared(), "Pending exact-token blob deletes executed this round (stage 3)."}, + {"fence_outs", std::make_shared(), "Expired mounts fenced out by this round's heartbeat floor."}, + {"anomalies", std::make_shared(), "Fold clamps surfaced (and survived) this round; steady >0 warrants a look at the round log details."}, + {"duration_ms", std::make_shared(), "Round wall-clock duration (Finish)."}, + {"error", std::make_shared(), "Exception text when outcome = Error."}, + {"ProfileEvents", std::make_shared(lc_string, std::make_shared()), + "On a Start/Finish row: the per-round ProfileEvents delta (the Cas* counters and S3 events for this round). On a Phase row: THAT PHASE's delta, so `GROUP BY phase` over `ProfileEvents['S3ListObjects']` attributes the round's LIST budget to the phase that spent it. Empty on the `meta_pool_wait` row by construction — that phase's work runs on other threads (read its `phase_metrics` instead)."}, + {"round_id", std::make_shared(), + "Correlator for every row of one round attempt (its Start, each Phase, and its Finish). Minted per attempt; unlike `round` it exists even for a round that never committed and for a round that never led. Group by this column to reconstruct one round."}, + {"phase", lc_string, + "The GC phase this row describes (empty on Start/Finish), in execution order: lease, pre_fold_ref_drain, heartbeat_floor, defer_decision, parent_seal_read, fold_ref_group, fold_seal_read, fold_ref_intake, fold_reduce, fold_seal_write, pending_deletes, meta_pool_wait, round_commit, handoff_reclaim, manifest_deletes, namespace_cleanup, ref_object_cleanup, orphan_sweep. A round that defers, or that never acquires the lease, emits only the phases it reached."}, + {"phase_duration_microseconds", std::make_shared(), + "Wall-clock duration of this phase in microseconds (Phase rows only). Microseconds because several phases are routinely sub-millisecond and the point is to see when they are not. Phase durations do not sum to the round's `duration_ms`: the round also does untimed bookkeeping between phases."}, + {"phase_metrics", std::make_shared(lc_string, std::make_shared()), + "Phase-specific semantic counts a phase computes for itself and no ProfileEvent can supply (Phase rows only) — for example `changed_shards` on defer_decision, `logs_accounted`/`logs_applied` on fold_ref_intake, `transactions_unapplied` on fold_reduce, `jobs_scheduled`/`jobs_completed` on meta_pool_wait. The verb counts ride the `ProfileEvents` column of the same row."}, + }; +} + +void ContentAddressedGarbageCollectionLogElement::appendToBlock(MutableColumns & columns) const +{ + size_t i = 0; + columns[i++]->insert(getFQDNOrHostName()); + columns[i++]->insert(DateLUT::instance().toDayNum(event_time).toUnderType()); + columns[i++]->insert(event_time); + columns[i++]->insert(event_time_microseconds); + columns[i++]->insert(static_cast(event_type)); + columns[i++]->insert(disk_name); + columns[i++]->insert(srid); + columns[i++]->insert(gc_id); + columns[i++]->insert(static_cast(trigger)); + columns[i++]->insert(round); + columns[i++]->insert(static_cast(outcome)); + columns[i++]->insert(candidates_marked); + columns[i++]->insert(objects_deleted); + columns[i++]->insert(objects_absent); + columns[i++]->insert(objects_replaced); + columns[i++]->insert(objects_spared); + columns[i++]->insert(manifests_deleted); + columns[i++]->insert(entries_condemned); + columns[i++]->insert(entries_graduated); + columns[i++]->insert(entries_redeleted); + columns[i++]->insert(fence_outs); + columns[i++]->insert(anomalies); + columns[i++]->insert(duration_ms); + columns[i++]->insert(error); + { + Map map; + map.reserve(profile_events.size()); + for (const auto & [k, v] : profile_events) + map.push_back(Tuple{k, v}); + columns[i++]->insert(map); + } + columns[i++]->insert(round_id); + columns[i++]->insert(phase); + columns[i++]->insert(phase_duration_microseconds); + { + Map map; + map.reserve(phase_metrics.size()); + for (const auto & [k, v] : phase_metrics) + map.push_back(Tuple{k, v}); + columns[i++]->insert(map); + } +} + +} diff --git a/src/Interpreters/ContentAddressedGarbageCollectionLog.h b/src/Interpreters/ContentAddressedGarbageCollectionLog.h new file mode 100644 index 000000000000..9cbdbd3525f6 --- /dev/null +++ b/src/Interpreters/ContentAddressedGarbageCollectionLog.h @@ -0,0 +1,63 @@ +#pragma once +#include +#include +#include +#include + +namespace DB +{ + +struct ContentAddressedGarbageCollectionLogElement +{ + /// `PHASE`: one row per GC phase, emitted between the round's `START` and `FINISH` and correlated + /// with them by `round_id`. + enum EventType : int8_t { START = 1, FINISH = 2, PHASE = 3 }; + /// `DEFERRED`: the round acquired the GC lease and took the skip-unchanged fast path -- no fold, no + /// pre-CAS deletes, no `gc/state` CAS. Kept distinct from `SUCCESS` so a query against this table can + /// tell a round that genuinely folded and found nothing apart from one that never folded at all. + enum Outcome : int8_t { UNKNOWN = 1, SUCCESS = 2, NOT_A_LEADER = 3, FAILED = 4, DEFERRED = 5 }; + enum Trigger : int8_t { SCHEDULED = 1, MANUAL = 2 }; + + time_t event_time = 0; + Decimal64 event_time_microseconds = 0; + + EventType event_type = START; + String disk_name; + String srid; /// server_root_id of the mount whose GC scheduler ran this round + String gc_id; + Trigger trigger = SCHEDULED; + + UInt64 round = 0; + Outcome outcome = UNKNOWN; /// UNKNOWN on START; set to SUCCESS/NOT_A_LEADER/FAILED on FINISH + UInt64 candidates_marked = 0; + UInt64 objects_deleted = 0; + UInt64 objects_absent = 0; + UInt64 objects_replaced = 0; + UInt64 objects_spared = 0; + UInt64 manifests_deleted = 0; /// owner-removed manifest bodies deleted (B11 — distinct from blob deletes) + UInt64 entries_condemned = 0; /// retired-cursor pipeline: entries newly condemned this round + UInt64 entries_graduated = 0; /// retired-cursor pipeline: entries newly round-passed (delete_pending) this round + UInt64 entries_redeleted = 0; /// retired-cursor pipeline: pending exact-token blob deletes executed this round + UInt64 fence_outs = 0; /// expired mounts fenced-out by the round's heartbeat floor + UInt64 anomalies = 0; /// fold clamps surfaced this round + UInt64 duration_ms = 0; + String error; + std::map profile_events; /// per-round delta (FINISH); per-phase delta (PHASE) + + String round_id; /// correlator for every row of one round attempt + String phase; /// empty on START/FINISH + UInt64 phase_duration_microseconds = 0; /// PHASE rows only + std::map phase_metrics; /// PHASE rows only + + static std::string name() { return "ContentAddressedGarbageCollectionLog"; } + static ColumnsDescription getColumnsDescription(); + static NamesAndAliases getNamesAndAliases() { return {}; } + void appendToBlock(MutableColumns & columns) const; +}; + +class ContentAddressedGarbageCollectionLog : public SystemLog +{ + using SystemLog::SystemLog; +}; + +} diff --git a/src/Interpreters/ContentAddressedLog.cpp b/src/Interpreters/ContentAddressedLog.cpp new file mode 100644 index 000000000000..9ae6506ea326 --- /dev/null +++ b/src/Interpreters/ContentAddressedLog.cpp @@ -0,0 +1,73 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +ColumnsDescription ContentAddressedLogElement::getColumnsDescription() +{ + auto lc_string = std::make_shared(std::make_shared()); + return ColumnsDescription + { + {"hostname", lc_string, "Host name of the server that emitted the event."}, + {"event_date", std::make_shared(), "Event date."}, + {"event_time", std::make_shared(), "Event time."}, + {"event_time_microseconds", std::make_shared(6), "Event time with microseconds."}, + {"event_type", lc_string, "The CA decision/event (blob_put, blob_reuse_adopt, root_remove, indegree_zero, gc_retire_decision, gc_recheck_verdict, blob_delete, dangling_access, corrupt_dangle, ...)."}, + {"disk_name", lc_string, "Content-addressed disk / pool the event belongs to."}, + {"namespace", std::make_shared(), "roots/ (server/table), empty if N/A."}, + {"ref_name", std::make_shared(), "Part name / ref the event concerns, empty if N/A."}, + {"object_kind", lc_string, "none/blob/manifest/root/snapshot."}, + {"object_hash", std::make_shared(), "Content hash (lowercase hex) of the object, empty if N/A."}, + {"token", std::make_shared(), "Incarnation token (ETag) involved, empty if N/A."}, + {"round", std::make_shared(), "GC round (0 if N/A)."}, + {"generation", std::make_shared(), "GC snapshot generation (0 if N/A)."}, + {"at_version", std::make_shared(), "Manifest shard_version of the driving journal record (0 if N/A)."}, + {"outcome", lc_string, "Decision outcome (ok/adopt/resurrect/deleted/replaced/spared/absent/zeroed/skipped/...)."}, + {"reason", lc_string, "Human-readable WHY of the decision (the rationale) -- templated across rows, so LowCardinality."}, + {"thread_id", std::make_shared(), "OS thread that emitted the event."}, + {"query_id", std::make_shared(), "Query id for correlation with system.query_log (empty if N/A)."}, + {"detail", std::make_shared(lc_string, std::make_shared()), + "Structured event-specific facts (e.g. condemn_round, superseded_token, code, site)."}, + }; +} + +void ContentAddressedLogElement::appendToBlock(MutableColumns & columns) const +{ + size_t i = 0; + columns[i++]->insert(getFQDNOrHostName()); + columns[i++]->insert(DateLUT::instance().toDayNum(event_time).toUnderType()); + columns[i++]->insert(event_time); + columns[i++]->insert(event_time_microseconds); + columns[i++]->insert(event_type); + columns[i++]->insert(disk_name); + columns[i++]->insert(namespace_); + columns[i++]->insert(ref_name); + columns[i++]->insert(object_kind); + columns[i++]->insert(object_hash); + columns[i++]->insert(token); + columns[i++]->insert(round); + columns[i++]->insert(gen); + columns[i++]->insert(at_version); + columns[i++]->insert(outcome); + columns[i++]->insert(reason); + columns[i++]->insert(thread_id); + columns[i++]->insert(query_id); + { + Map map; + map.reserve(detail.size()); + for (const auto & [k, v] : detail) + map.push_back(Tuple{k, v}); + columns[i++]->insert(map); + } +} + +} diff --git a/src/Interpreters/ContentAddressedLog.h b/src/Interpreters/ContentAddressedLog.h new file mode 100644 index 000000000000..84c7ae3c251c --- /dev/null +++ b/src/Interpreters/ContentAddressedLog.h @@ -0,0 +1,47 @@ +#pragma once +#include +#include +#include +#include +#include + +namespace DB +{ + +/// One row per content-addressed (CA) decision/event (B170). The decoupled Core POD `Cas::CasEvent` +/// is mapped to this element by `ContentAddressedMetadataStorage::makeCasEventSink` and forwarded to +/// the SystemLog. Optional (off by default); enabled for soak/CI. The set is exhaustive enough to +/// reconstruct an entity's whole lifetime; `reason`/`detail` carry each decision's rationale. +struct ContentAddressedLogElement +{ + time_t event_time = 0; + Decimal64 event_time_microseconds = 0; + + String event_type; /// Cas::CasEventType name (snake_case), LowCardinality in the table + String disk_name; + String namespace_; + String ref_name; + String object_kind; /// none/blob/manifest/root/snap + String object_hash; + String token; + UInt64 round = 0; + UInt64 gen = 0; + UInt64 at_version = 0; + String outcome; + String reason; + UInt64 thread_id = 0; + String query_id; + std::map detail; + + static std::string name() { return "ContentAddressedLog"; } + static ColumnsDescription getColumnsDescription(); + static NamesAndAliases getNamesAndAliases() { return {}; } + void appendToBlock(MutableColumns & columns) const; +}; + +class ContentAddressedLog : public SystemLog +{ + using SystemLog::SystemLog; +}; + +} diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index 03884fccbad1..9036cd7dc8ab 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -7027,6 +7027,30 @@ std::shared_ptr Context::getPartLog() const return shared->system_logs->part_log; } +std::shared_ptr Context::getContentAddressedGarbageCollectionLog() const +{ + std::lock_guard lock(mutex_shared_context); + if (!shared) + return {}; + + SharedLockGuard lock2(shared->mutex); + if (!shared->system_logs) + return {}; + return shared->system_logs->cas_gc_log; +} + +std::shared_ptr Context::getContentAddressedLog() const +{ + std::lock_guard lock(mutex_shared_context); + if (!shared) + return {}; + + SharedLockGuard lock2(shared->mutex); + if (!shared->system_logs) + return {}; + return shared->system_logs->cas_log; +} + std::shared_ptr Context::getBackgroundSchedulePoolLog() const { SharedLockGuard lock(shared->mutex); diff --git a/src/Interpreters/Context.h b/src/Interpreters/Context.h index e0fc54ccb600..2caa4d5ead16 100644 --- a/src/Interpreters/Context.h +++ b/src/Interpreters/Context.h @@ -133,6 +133,8 @@ class QueryMetricLog; class QueryThreadLog; class QueryViewsLog; class PartLog; +class ContentAddressedGarbageCollectionLog; +class ContentAddressedLog; class BackgroundSchedulePoolLog; class TextLog; class TraceLog; @@ -1778,6 +1780,8 @@ class Context: public ContextData, public std::enable_shared_from_this /// Returns an object used to log operations with parts if it possible. /// Provide table name to make required checks. std::shared_ptr getPartLog() const; + std::shared_ptr getContentAddressedGarbageCollectionLog() const; + std::shared_ptr getContentAddressedLog() const; std::shared_ptr getBackgroundSchedulePoolLog() const; diff --git a/src/Interpreters/InterpreterSystemQuery.cpp b/src/Interpreters/InterpreterSystemQuery.cpp index e070cd3aafcf..7a5e22180ae0 100644 --- a/src/Interpreters/InterpreterSystemQuery.cpp +++ b/src/Interpreters/InterpreterSystemQuery.cpp @@ -21,6 +21,14 @@ #include #include #include +#include +#include +/// Direct, though `ContentAddressedMetadataStorage.h` above would also pull it in: this TU renders +/// `FsckReport`'s hard findings into the SQL row, and `CasFsck.h`'s `kFsckHardFindings` tripwire is what +/// breaks THIS build when a finding is added. Depending on another header's include list for that would +/// make the coverage silently removable. +#include +#include #include #include #include @@ -77,6 +85,7 @@ #include #include #include +#include #include #include #include @@ -103,6 +112,7 @@ #include #include #include +#include #include "config.h" @@ -280,6 +290,18 @@ AccessType getRequiredAccessType(StorageActionBlockType action_type) constexpr std::string_view table_is_not_replicated = "Table {} is not replicated"; +/// A table in a database created with `lazy_load_tables = 1` stays wrapped in a `StorageTableProxy` +/// until its first access, so a `dynamic_cast` to the real engine (e.g. `StorageReplicatedMergeTree`) +/// fails and a `SYSTEM` verb that targets one specific named table misreports it as not replicated. +/// Materialize the proxy before such a cast; generic query paths already materialize on read by +/// design and must not go through this helper. +StoragePtr unwrapTableProxy(const StoragePtr & storage) +{ + if (const auto * proxy = dynamic_cast(storage.get())) + return proxy->getNested(); + return storage; +} + } /// Implements SYSTEM [START|STOP] @@ -1133,6 +1155,100 @@ BlockIO InterpreterSystemQuery::execute() break; } + case Type::CAS_GC_RUN: + { + /// A manual GC RUN executes REGARDLESS of SYSTEM CAS GC STOP: STOP pauses only the + /// background PACER, not the GC engine, so an explicit operator round still runs (explicit intent + /// wins). A round that acquires the lease sets the disk's in-process is_leader=true, which its + /// introspection can surface transiently even while the background scheduler stays stopped — + /// until a peer mounter steals the lease or GC START resumes pacing. This is truthful (the round + /// DID lead) and harmless (no background thread acts on it while stopped). + getContext()->checkAccess(AccessType::SYSTEM_CAS_GC_RUN); + result = runContentAddressedGcRun(query.disk); + break; + } + case Type::CAS_GC_REBUILD: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_GC_REBUILD); + result = runContentAddressedGcRebuild(query.disk, query.cas_gc_rebuild_force); + break; + } + case Type::CAS_FSCK: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_FSCK); + result = runContentAddressedFsck(query.disk); + break; + } + case Type::CAS_FORGET: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_FORGET); + contentAddressedForget(query.disk); + break; + } + case Type::CAS_GC_STOP: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_GC_STOP); + contentAddressedGcStop(query.disk); + break; + } + case Type::CAS_GC_START: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_GC_START); + contentAddressedGcStart(query.disk); + break; + } + case Type::CAS_DROP_POOL_MEMBER: + { + getContext()->checkAccess(AccessType::SYSTEM_CAS_DROP_POOL_MEMBER); + + auto disk = getContext()->getDisk(query.disk); + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "SYSTEM CAS DROP POOL MEMBER: disk '{}' is not a content-addressed disk", query.disk); + ca->checkNotReadOnly("SYSTEM CAS DROP POOL MEMBER"); + + const auto & host_store = ca->store(); + const auto report = Cas::decommissionPoolMember( + host_store->poolBackendPtr(), host_store->poolConfig(), query.replica, {}, + [ca] { ca->requestGcRoundSoon(); }); + + /// One-row summary result set (precedent: SYNC_FILESYSTEM_CACHE's MutableColumns/ + /// SourceFromSingleChunk construction above). + ColumnsDescription columns{NamesAndTypesList{ + {"server_root_id", std::make_shared()}, + {"namespaces_removed", std::make_shared()}, + {"namespaces_already_removed", std::make_shared()}, + {"committed_refs_removed", std::make_shared()}, + {"precommits_removed", std::make_shared()}, + {"manifest_debris_removed", std::make_shared()}, + {"staging_objects_removed", std::make_shared()}, + {"mountpoint_objects_removed", std::make_shared()}, + {"slot_removed", std::make_shared()}, + {"warnings", std::make_shared()}, + }}; + Block sample_block; + for (const auto & column : columns) + sample_block.insert({column.type->createColumn(), column.type, column.name}); + + MutableColumns res_columns = sample_block.cloneEmptyColumns(); + size_t i = 0; + res_columns[i++]->insert(report.srid); + res_columns[i++]->insert(report.namespaces_removed); + res_columns[i++]->insert(report.namespaces_already_removed); + res_columns[i++]->insert(report.committed_refs_removed); + res_columns[i++]->insert(report.precommits_removed); + res_columns[i++]->insert(report.manifest_debris_removed); + res_columns[i++]->insert(report.staging_objects_removed); + res_columns[i++]->insert(report.mountpoint_objects_removed); + res_columns[i++]->insert(static_cast(report.slot_removed)); + res_columns[i++]->insert(fmt::format("{}", fmt::join(report.warnings, "; "))); + + size_t num_rows = res_columns[0]->size(); + auto source = std::make_shared(std::make_shared(std::move(sample_block)), Chunk(std::move(res_columns), num_rows)); + result.pipeline = QueryPipeline(std::move(source)); + break; + } case Type::RESTART_DISK: { restartDisk(query.disk); @@ -1396,7 +1512,7 @@ void InterpreterSystemQuery::restoreReplica() { getContext()->checkAccess(AccessType::SYSTEM_RESTORE_REPLICA, table_id); - const StoragePtr table_ptr = DatabaseCatalog::instance().getTable(table_id, getContext()); + const StoragePtr table_ptr = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); auto * const table_replicated_ptr = dynamic_cast(table_ptr.get()); @@ -1471,7 +1587,9 @@ StoragePtr InterpreterSystemQuery::doRestartReplica(const StorageID & replica, C return nullptr; } - if (!dynamic_cast(table.get())) + /// Only the type check needs the materialized (unwrapped) storage; the possibly-still-proxied + /// `table` is what actually stays registered in `database` and is what gets locked/detached below. + if (!dynamic_cast(unwrapTableProxy(table).get())) { if (throw_on_error) throw Exception(ErrorCodes::BAD_ARGUMENTS, table_is_not_replicated.data(), replica.getNameForLogs()); @@ -1685,7 +1803,7 @@ void InterpreterSystemQuery::dropReplica(ASTSystemQuery & query) if (!table_id.empty()) { getContext()->checkAccess(AccessType::SYSTEM_DROP_REPLICA, table_id); - StoragePtr table = DatabaseCatalog::instance().getTable(table_id, getContext()); + StoragePtr table = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); if (!dropStorageReplica(query.replica, table)) throw Exception(ErrorCodes::BAD_ARGUMENTS, table_is_not_replicated.data(), table_id.getNameForLogs()); @@ -2213,7 +2331,7 @@ bool InterpreterSystemQuery::trySyncReplica(StoragePtr table, SyncReplicaMode sy break; } - if (auto * storage_replicated = dynamic_cast(table.get())) + if (auto * storage_replicated = dynamic_cast(unwrapTableProxy(table).get())) { auto log = getLogger("InterpreterSystemQuery"); LOG_TRACE(log, "Synchronizing entries in replica's queue with table's log and waiting for current last entry to be processed"); @@ -2259,7 +2377,7 @@ void InterpreterSystemQuery::syncReplica(ASTSystemQuery & query) void InterpreterSystemQuery::waitLoadingParts() { getContext()->checkAccess(AccessType::SYSTEM_WAIT_LOADING_PARTS, table_id); - StoragePtr table = DatabaseCatalog::instance().getTable(table_id, getContext()); + StoragePtr table = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); if (auto * merge_tree = dynamic_cast(table.get())) { @@ -2343,11 +2461,12 @@ namespace MergeTreeData & getMergeTreeWithManualSelector(const StoragePtr & table, const StorageID & table_id, const char * action) { - auto * merge_tree = dynamic_cast(table.get()); + const StoragePtr unwrapped = unwrapTableProxy(table); + auto * merge_tree = dynamic_cast(unwrapped.get()); if (!merge_tree) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Command {} is supported only for MergeTree-family tables, but got: {}", - action, table->getName()); + action, unwrapped->getName()); const auto algorithm = (*merge_tree->getSettings())[MergeTreeSetting::merge_selector_algorithm].value; if (algorithm != MergeSelectorAlgorithm::MANUAL) @@ -2420,6 +2539,363 @@ void InterpreterSystemQuery::syncMerges() throw DB::Exception(DB::ErrorCodes::TIMEOUT_EXCEEDED, "SYNC MERGES {}: command timed out. See the 'max_execution_time' setting", table_id.getNameForLogs()); } +namespace +{ + +/// One-row-per-disk result-set builders for the CAS GC verbs, mirroring the SYSTEM CAS +/// DROP POOL MEMBER precedent (ColumnsDescription + MutableColumns + SourceFromSingleChunk; see also +/// SYNC_FILESYSTEM_CACHE above). +ColumnsDescription contentAddressedGcRoundColumns() +{ + return ColumnsDescription{NamesAndTypesList{ + {"disk", std::make_shared()}, + {"acquired_lease", std::make_shared()}, + {"deferred", std::make_shared()}, + {"round", std::make_shared()}, + {"candidates_marked", std::make_shared()}, + {"objects_deleted", std::make_shared()}, + {"objects_absent", std::make_shared()}, + {"objects_replaced", std::make_shared()}, + {"objects_spared", std::make_shared()}, + {"manifests_deleted", std::make_shared()}, + {"entries_condemned", std::make_shared()}, + {"entries_graduated", std::make_shared()}, + {"entries_redeleted", std::make_shared()}, + {"fence_outs", std::make_shared()}, + {"anomalies", std::make_shared()}, + /// Task 7: the retire pipeline's REMAINING (not this-round-delta) sizes, read from the gc/state + /// this round's single CAS just published -- see `Cas::RoundReport`'s field comments. Zero on a + /// non-authoritative row (!acquired_lease or deferred), same as every other counter above. + {"pending_candidates", std::make_shared()}, + {"pending_condemned", std::make_shared()}, + {"pending_retired", std::make_shared()}, + }}; +} + +void appendContentAddressedGcRoundRow(MutableColumns & res_columns, const String & disk_name, const Cas::RoundReport & rep) +{ + size_t i = 0; + res_columns[i++]->insert(disk_name); + res_columns[i++]->insert(static_cast(rep.acquired_lease)); + res_columns[i++]->insert(static_cast(rep.deferred)); + res_columns[i++]->insert(rep.round); + res_columns[i++]->insert(rep.candidates); + res_columns[i++]->insert(rep.deleted); + res_columns[i++]->insert(rep.absent); + res_columns[i++]->insert(rep.replaced); + res_columns[i++]->insert(rep.spared); + res_columns[i++]->insert(rep.manifests_deleted); + res_columns[i++]->insert(rep.condemned); + res_columns[i++]->insert(rep.graduated); + res_columns[i++]->insert(rep.redeleted); + res_columns[i++]->insert(rep.fence_outs); + res_columns[i++]->insert(rep.anomalies.size()); + res_columns[i++]->insert(rep.pending_candidates); + res_columns[i++]->insert(rep.pending_condemned); + res_columns[i++]->insert(rep.pending_retired); +} + +ColumnsDescription contentAddressedGcRebuildColumns() +{ + return ColumnsDescription{NamesAndTypesList{ + {"disk", std::make_shared()}, + {"performed", std::make_shared()}, + {"round", std::make_shared()}, + {"generation", std::make_shared()}, + {"namespaces", std::make_shared()}, + {"shards", std::make_shared()}, + {"committed_refs", std::make_shared()}, + {"live_precommits", std::make_shared()}, + {"unowned_alive_manifests", std::make_shared()}, + {"edges", std::make_shared()}, + {"clamped_shards", std::make_shared()}, + /// 1 => the rebuild found no fold seal at all and carried NO durable hold forward, having + /// concluded from enumeration alone that the pool never sealed a baseline. On a pool that has + /// ever completed a GC round this means the object listing lied. + {"virgin_by_enumeration", std::make_shared()}, + /// Which generation's fold seal the rebuild carried holds from; 0 when it carried none. + {"adopted_seal_generation", std::make_shared()}, + }}; +} + +void appendContentAddressedGcRebuildRow(MutableColumns & res_columns, const String & disk_name, const Cas::RebuildReport & rep) +{ + size_t i = 0; + res_columns[i++]->insert(disk_name); + res_columns[i++]->insert(static_cast(rep.performed)); + res_columns[i++]->insert(rep.round); + res_columns[i++]->insert(rep.generation); + res_columns[i++]->insert(rep.namespaces); + res_columns[i++]->insert(rep.shards); + res_columns[i++]->insert(rep.committed_refs); + res_columns[i++]->insert(rep.live_precommits); + res_columns[i++]->insert(rep.unowned_alive_manifests); + res_columns[i++]->insert(rep.edges); + res_columns[i++]->insert(rep.clamped_shards); + res_columns[i++]->insert(static_cast(rep.virgin_by_enumeration)); + res_columns[i++]->insert(rep.adopted_seal_generation); +} + +/// SYSTEM CAS FSCK's one-row-per-disk summary. Named UInt64 columns only, no DETAIL +/// keyword (YAGNI -- the offline `clickhouse-disks cas-fsck --detail` applet already covers per-object +/// listing). Field order/names mirror `Cas::FsckReport`; the row was a deliberate SUBSET of it until +/// 2026-07-29, and is no longer one where findings are concerned -- see the rule stated at +/// `stale_edge` below. `CommandFsck.cpp`'s `formatFsckSummary` line carries the same fields. +ColumnsDescription contentAddressedFsckColumns() +{ + return ColumnsDescription{NamesAndTypesList{ + {"disk", std::make_shared()}, + {"reachable", std::make_shared()}, + {"dangling", std::make_shared()}, + {"unreachable", std::make_shared()}, + {"pending_gc", std::make_shared()}, + {"awaiting_gc", std::make_shared()}, + {"unaccounted", std::make_shared()}, + /// EVERY TERM OF `FsckReport::clean` APPEARS HERE. This row is the only view of a report a SQL + /// consumer ever gets, so a hard finding the row omits is a finding no query can see — the same + /// shape that hid `corrupted_runs` from the text summary for months. The row was a deliberate + /// subset until 2026-07-29 and `stale_edge`/`corrupted_runs` were invisible from SQL while + /// `clickhouse-disks cas-fsck` surfaced them; then + /// `lifeless_keys` was added to `clean` in 2026-07-30 and missed here too. Every time, the rule + /// was written in prose, and every time the prose did not hold. + /// + /// So it no longer lives only in prose: `kFsckHardFindings` (`CasFsck.h`) is the list `clean` is + /// computed from, and the `static_assert` beside it breaks the build in THIS translation unit when + /// a term is added. Read that assert's message before bumping its count -- it names this site as + /// one of the three that owes an update, and says that two of the three have no test that can fail + /// on their behalf. WHICH two is in the comment above the assert, not in the message; this site is + /// one of them. + /// + /// `stale_edge` is nonzero only in `detail` mode and this row is built from a summary scan, so it + /// reads 0 here always — present because "absent" and "zero" are different facts to a consumer, + /// and a column that appears the day the scan gains detail is a schema change nobody asked for. + {"stale_edge", std::make_shared()}, + {"corrupted_runs", std::make_shared()}, + /// The ref-stream verdicts (spec §7). `chain_broken` is a HARD finding — it belongs on the row + /// for the same reason `dangling` does. `unchecked` is its honest companion: namespaces the audit + /// could not prove either way, so a zero here is what makes the other zeros mean something. + {"chain_broken", std::make_shared()}, + {"unchecked", std::make_shared()}, + /// A malformed/non-canonical namespace-tree key, or an ambiguous/unreadable catalog + /// incarnation — a term of `clean`. + {"lifeless_keys", std::make_shared()}, + /// A COMPLETE, canonical namespace-life key whose life is absent from a catalog cut taken after + /// the listing: janitor-pending debris, NOT a term of `clean` (see `FsckClass::JanitorPending`). + {"namespace_janitor_pending", std::make_shared()}, + {"namespace_janitor_pending_bytes", std::make_shared()}, + {"namespace_janitor_pending_lives", std::make_shared()}, + {"ref_records_walked", std::make_shared()}, + {"physical_bytes", std::make_shared()}, + {"referenced_logical_bytes", std::make_shared()}, + {"distinct_blobs", std::make_shared()}, + {"total_blob_refs", std::make_shared()}, + }}; +} + +void appendContentAddressedFsckRow(MutableColumns & res_columns, const String & disk_name, const Cas::FsckReport & rep) +{ + size_t i = 0; + res_columns[i++]->insert(disk_name); + res_columns[i++]->insert(rep.reachable); + res_columns[i++]->insert(rep.dangling); + res_columns[i++]->insert(rep.unreachable); + res_columns[i++]->insert(rep.pending_gc); + res_columns[i++]->insert(rep.awaiting_gc); + res_columns[i++]->insert(rep.unaccounted); + res_columns[i++]->insert(rep.stale_edge); + res_columns[i++]->insert(rep.corrupted_runs); + res_columns[i++]->insert(rep.chain_broken); + res_columns[i++]->insert(rep.unchecked); + res_columns[i++]->insert(rep.lifeless_keys); + res_columns[i++]->insert(rep.namespace_janitor_pending); + res_columns[i++]->insert(rep.namespace_janitor_pending_bytes); + res_columns[i++]->insert(rep.namespace_janitor_pending_lives); + res_columns[i++]->insert(rep.ref_records_walked); + res_columns[i++]->insert(rep.physical_bytes); + res_columns[i++]->insert(rep.referenced_logical_bytes); + res_columns[i++]->insert(rep.distinct_blobs); + res_columns[i++]->insert(rep.total_blob_refs); +} + +} + +BlockIO InterpreterSystemQuery::runContentAddressedGcRun(const String & disk_name) +{ + ColumnsDescription columns = contentAddressedGcRoundColumns(); + Block sample_block; + for (const auto & column : columns) + sample_block.insert({column.type->createColumn(), column.type, column.name}); + MutableColumns res_columns = sample_block.cloneEmptyColumns(); + + if (!disk_name.empty()) + { + auto disk = getContext()->getDisk(disk_name); + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + appendContentAddressedGcRoundRow(res_columns, disk_name, ca->runGarbageCollectionRoundNow()); /// synchronous, one round + } + else + { + size_t ran = 0; + for (const auto & [name, disk] : getContext()->getDisksMap()) + { + if (auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk)) + { + appendContentAddressedGcRoundRow(res_columns, name, ca->runGarbageCollectionRoundNow()); /// synchronous, one round + ++ran; + } + } + if (ran == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "No content-addressed disks are configured on this node"); + } + + size_t num_rows = res_columns[0]->size(); + auto source = std::make_shared(std::make_shared(std::move(sample_block)), Chunk(std::move(res_columns), num_rows)); + BlockIO result; + result.pipeline = QueryPipeline(std::move(source)); + return result; +} + +BlockIO InterpreterSystemQuery::runContentAddressedGcRebuild(const String & disk_name, bool force) +{ + /// REBUILD requires an EXPLICIT disk (E1): the destructive baseline rebuild must never fan out + /// across every content-addressed disk on the node. The parser enforces this syntactically; this is + /// the fail-closed backstop for a directly-constructed AST. + if (disk_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "SYSTEM CAS GC REBUILD requires an explicit disk name"); + + auto disk = getContext()->getDisk(disk_name); + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + + Cas::RebuildReport rep = ca->runGcRebuildNow(force); /// synchronous, one rebuild + if (!rep.performed) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "CAS GC rebuild refused: {}", rep.refusal); + LOG_INFO(log, + "CAS GC rebuild on disk '{}' completed: round={} generation={} namespaces={} shards={} " + "committed_refs={} live_precommits={} unowned_alive_manifests={} edges={} clamped_shards={} " + "virgin_by_enumeration={} adopted_seal_generation={}", + disk_name, rep.round, rep.generation, rep.namespaces, rep.shards, rep.committed_refs, + rep.live_precommits, rep.unowned_alive_manifests, rep.edges, rep.clamped_shards, + rep.virgin_by_enumeration, rep.adopted_seal_generation); + + ColumnsDescription columns = contentAddressedGcRebuildColumns(); + Block sample_block; + for (const auto & column : columns) + sample_block.insert({column.type->createColumn(), column.type, column.name}); + MutableColumns res_columns = sample_block.cloneEmptyColumns(); + appendContentAddressedGcRebuildRow(res_columns, disk_name, rep); + + size_t num_rows = res_columns[0]->size(); + auto source = std::make_shared(std::make_shared(std::move(sample_block)), Chunk(std::move(res_columns), num_rows)); + BlockIO result; + result.pipeline = QueryPipeline(std::move(source)); + return result; +} + +BlockIO InterpreterSystemQuery::runContentAddressedFsck(const String & disk_name) +{ + /// FSCK runs on a RUNNING disk (rev.8): the scan is read-only and revalidates every ref-walk finding + /// against a fresh authoritative read, so it needs no quiesce. The disk is REQUIRED, enforced by the + /// parser -- no fan-out form. + if (disk_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "SYSTEM CAS FSCK requires an explicit disk name"); + + auto disk = getContext()->getDisk(disk_name); + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + + const Cas::FsckReport rep = ca->runFsckNow(/* detail= */ false); /// summary only (no DETAIL keyword yet) + + ColumnsDescription columns = contentAddressedFsckColumns(); + Block sample_block; + for (const auto & column : columns) + sample_block.insert({column.type->createColumn(), column.type, column.name}); + MutableColumns res_columns = sample_block.cloneEmptyColumns(); + appendContentAddressedFsckRow(res_columns, disk_name, rep); + + size_t num_rows = res_columns[0]->size(); + auto source = std::make_shared(std::make_shared(std::move(sample_block)), Chunk(std::move(res_columns), num_rows)); + BlockIO result; + result.pipeline = QueryPipeline(std::move(source)); + return result; +} + +void InterpreterSystemQuery::contentAddressedForget(const String & disk_name) +{ + /// The operator "fire-marshal" verb (spec §5): a force-Vanish that decommissions a content-addressed + /// disk NODE-LOCALLY. Unlike the store()-class verbs, FORGET must work on a disk that is NOT live -- + /// that is its whole purpose (a stuck transient/IdentityLost pool, an operator-asserted decommission) -- + /// so it does NOT go through `checkOpAdmitted`/`store()` (which refuse a not-live disk). It is a + /// lifecycle verb like the Factory class: it reaches the pool directly and drives it to + /// `Vanished(forgotten)`. FORGET is an operator ASSERTION, not an erasure proof; the resulting [D5] + /// error message says so. + if (disk_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "SYSTEM CAS FORGET requires an explicit disk name"); + + auto disk = getContext()->getDisk(disk_name); /// UNKNOWN_DISK on a bad name + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + + ca->forgetDisk(); + LOG_WARNING(log, + "SYSTEM CAS FORGET decommissioned content-addressed disk '{}' (node-local; erasure " + "NOT verified). The disk stays registered and answers store-class access with a typed error; a " + "server restart re-registers the name.", + disk_name); +} + +void InterpreterSystemQuery::contentAddressedGcStop(const String & disk_name) +{ + /// SYSTEM CAS GC STOP (spec §6): stop ONLY the background GC scheduler on this disk. The + /// disk stays fully usable -- reads and writes are unaffected; this is granular operator control of GC + /// alone (e.g. to pause reclamation during an incident), not a lifecycle transition. STOP-IN-PLACE: the + /// scheduler object is retained so a later GC START restarts the SAME instance (its gc_id and lease + /// observation history preserved). Idempotent; works even on a not-live/Vanished disk (stopping GC on a + /// sick disk is legitimate). The disk is REQUIRED -- there is no fan-out form. + if (disk_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "SYSTEM CAS GC STOP requires an explicit disk name"); + + auto disk = getContext()->getDisk(disk_name); /// UNKNOWN_DISK on a bad name + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + + ca->gcStop(); + LOG_INFO(log, + "SYSTEM CAS GC STOP: stopped the background garbage-collection scheduler on " + "content-addressed disk '{}' (the disk stays fully usable; SYSTEM CAS GC START " + "resumes it).", + disk_name); +} + +void InterpreterSystemQuery::contentAddressedGcStart(const String & disk_name) +{ + /// SYSTEM CAS GC START (spec §6): restart the background GC scheduler stopped by GC STOP. + /// It re-enters the SAME scheduler instance; leadership is NOT auto-restored -- the scheduler re-acquires + /// the durable `gc/state` lease through the next round's normal acquisition. Idempotent (a no-op on a + /// running scheduler). Refuses on a decommissioned/uncertain pool (typed error) -- restarting GC there + /// would only spin failing rounds. The disk is REQUIRED -- there is no fan-out form. + if (disk_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "SYSTEM CAS GC START requires an explicit disk name"); + + auto disk = getContext()->getDisk(disk_name); /// UNKNOWN_DISK on a bad name + auto * ca = ContentAddressedMetadataStorage::tryFromDisk(disk); + if (!ca) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Disk '{}' is not a content-addressed disk", disk_name); + + ca->gcStart(); + LOG_INFO(log, + "SYSTEM CAS GC START: resumed the background garbage-collection scheduler on " + "content-addressed disk '{}'.", + disk_name); +} + void InterpreterSystemQuery::loadPrimaryKeys() { loadOrUnloadPrimaryKeysImpl(true); @@ -2435,7 +2911,7 @@ void InterpreterSystemQuery::loadOrUnloadPrimaryKeysImpl(bool load) if (!table_id.empty()) { getContext()->checkAccess(load ? AccessType::SYSTEM_LOAD_PRIMARY_KEY : AccessType::SYSTEM_UNLOAD_PRIMARY_KEY, table_id.database_name, table_id.table_name); - StoragePtr table = DatabaseCatalog::instance().getTable(table_id, getContext()); + StoragePtr table = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); if (auto * merge_tree = dynamic_cast(table.get())) { @@ -2581,12 +3057,16 @@ void InterpreterSystemQuery::flushDistributed(ASTSystemQuery & query) if (query.query_settings) settings_changes = query.query_settings->as()->changes; +<<<<<<< HEAD /// Keep the StoragePtr alive for the whole flush: the table holds no other owning /// reference here (DROP on an Atomic database does not take the exclusive drop_lock, /// and the flush does not hold an async-insert lock), so a concurrent DROP could /// otherwise destroy the table while flushClusterNodesAllData is still running. auto table = DatabaseCatalog::instance().getTable(table_id, getContext()); if (auto * storage_distributed = dynamic_cast(table.get())) +======= + if (auto * storage_distributed = dynamic_cast(unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())).get())) +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) storage_distributed->flushClusterNodesAllData(getContext(), settings_changes); else throw Exception(ErrorCodes::BAD_ARGUMENTS, "Table {} is not distributed", table_id.getNameForLogs()); @@ -2600,7 +3080,7 @@ void InterpreterSystemQuery::flushObjectStorageQueue(ASTSystemQuery & query) if (query.queue_path.empty()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "PATH must be specified for SYSTEM FLUSH OBJECT STORAGE QUEUE"); - auto table = DatabaseCatalog::instance().getTable(table_id, context); + auto table = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, context)); auto * queue = dynamic_cast(table.get()); if (!queue) throw Exception(ErrorCodes::BAD_ARGUMENTS, @@ -2744,7 +3224,7 @@ void InterpreterSystemQuery::prewarmMarkCache() getContext()->checkAccess(AccessType::SYSTEM_PREWARM_MARK_CACHE, table_id); - auto table_ptr = DatabaseCatalog::instance().getTable(table_id, getContext()); + auto table_ptr = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); auto * merge_tree = dynamic_cast(table_ptr.get()); if (!merge_tree) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Command PREWARM MARK CACHE is supported only for MergeTree table, but got: {}", table_ptr->getName()); @@ -2768,7 +3248,7 @@ void InterpreterSystemQuery::prewarmPrimaryIndexCache() getContext()->checkAccess(AccessType::SYSTEM_PREWARM_PRIMARY_INDEX_CACHE, table_id); - auto table_ptr = DatabaseCatalog::instance().getTable(table_id, getContext()); + auto table_ptr = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); auto * merge_tree = dynamic_cast(table_ptr.get()); if (!merge_tree) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Command PREWARM PRIMARY INDEX CACHE is supported only for MergeTree table, but got: {}", table_ptr->getName()); @@ -3197,6 +3677,41 @@ AccessRightsElements InterpreterSystemQuery::getRequiredAccessForDDLOnCluster() required_access.emplace_back(AccessType::SYSTEM_WAIT_BLOBS_CLEANUP); break; } + case Type::CAS_GC_RUN: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_GC_RUN); + break; + } + case Type::CAS_GC_REBUILD: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_GC_REBUILD); + break; + } + case Type::CAS_DROP_POOL_MEMBER: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_DROP_POOL_MEMBER); + break; + } + case Type::CAS_FSCK: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_FSCK); + break; + } + case Type::CAS_FORGET: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_FORGET); + break; + } + case Type::CAS_GC_STOP: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_GC_STOP); + break; + } + case Type::CAS_GC_START: + { + required_access.emplace_back(AccessType::SYSTEM_CAS_GC_START); + break; + } case Type::UNFREEZE: { required_access.emplace_back(AccessType::SYSTEM_UNFREEZE); diff --git a/src/Interpreters/InterpreterSystemQuery.h b/src/Interpreters/InterpreterSystemQuery.h index 123bbf0458bb..c4af357f8d4a 100644 --- a/src/Interpreters/InterpreterSystemQuery.h +++ b/src/Interpreters/InterpreterSystemQuery.h @@ -75,6 +75,14 @@ class InterpreterSystemQuery : public IInterpreter, WithMutableContext void scheduleMerge(ASTSystemQuery & query); void syncMerges(); + BlockIO runContentAddressedGcRun(const String & disk_name); + BlockIO runContentAddressedGcRebuild(const String & disk_name, bool force); + + BlockIO runContentAddressedFsck(const String & disk_name); + void contentAddressedForget(const String & disk_name); + void contentAddressedGcStop(const String & disk_name); + void contentAddressedGcStart(const String & disk_name); + void loadPrimaryKeys(); void unloadPrimaryKeys(); void loadOrUnloadPrimaryKeysImpl(bool load); diff --git a/src/Interpreters/MergeTreeTransaction/VersionMetadataOnDisk.cpp b/src/Interpreters/MergeTreeTransaction/VersionMetadataOnDisk.cpp index ac43dbc38b72..ae05bc4b4bb8 100644 --- a/src/Interpreters/MergeTreeTransaction/VersionMetadataOnDisk.cpp +++ b/src/Interpreters/MergeTreeTransaction/VersionMetadataOnDisk.cpp @@ -343,6 +343,18 @@ void VersionMetadataOnDisk::storeInfoToDataPartStorage( static constexpr auto filename = TXN_VERSION_METADATA_FILE_NAME; static constexpr auto tmp_filename = TMP_TXN_VERSION_METADATA_FILE_NAME; + if (data_part_storage.supportsAtomicFileWrites()) + { + /// Single atomic write: storages that publish file writes atomically do not need + /// the tmp+replace dance (which exists only for partial-local-write crash safety). + auto write_settings = storage.getContext()->getWriteSettings(); + auto buf = data_part_storage.writeFile(filename, 256, write_settings); + new_info.writeToBuffer(*buf, /*one_line=*/false); + buf->finalize(); + buf->sync(); + return; + } + try { { diff --git a/src/Interpreters/ServerAsynchronousMetrics.cpp b/src/Interpreters/ServerAsynchronousMetrics.cpp index 3ca6fc71beb9..aa0896a3c0e7 100644 --- a/src/Interpreters/ServerAsynchronousMetrics.cpp +++ b/src/Interpreters/ServerAsynchronousMetrics.cpp @@ -15,7 +15,11 @@ #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include @@ -385,6 +389,7 @@ void ServerAsynchronousMetrics::updateImpl(TimePoint update_time, TimePoint curr } #endif +<<<<<<< HEAD if (auto object_storage_disk = std::dynamic_pointer_cast(disk)) { dead_blobs_queue_estimate[name] = static_cast(object_storage_disk->getDeadBlobsQueueEstimate()); @@ -432,6 +437,38 @@ void ServerAsynchronousMetrics::updateImpl(TimePoint update_time, TimePoint curr "Estimated number of blobs enqueued for removal from the disk object storage (the blob manager dead queue), keyed by the disk name. Disks without blob replication report 0." }; new_values["MissingBlobsQueueEstimate"] = { "disk", std::move(missing_blobs_queue_estimate), "Estimated number of blobs awaiting replication to other locations of the disk (the blob manager missing queue), keyed by the disk name. Disks without blob replication report 0." }; +======= + /// Per-disk CAS GC health, for Prometheus scraping. `tryFromDisk` returns nullptr for a + /// disk whose metadata storage is not content-addressed (the common case); `gcHealth()` + /// returns nullopt for a content-addressed disk whose GC scheduler has not started yet + /// (still opening, read-only, or GC disabled by configuration) -- both are skipped + /// silently, same as the DiskUsed_/DiskTotal_ metrics above skip disks that don't report + /// space. This runs on every asynchronous-metrics tick for every configured disk and must + /// never throw. + try + { + if (auto * ca_storage = ContentAddressedMetadataStorage::tryFromDisk(disk)) + { + if (auto health = ca_storage->gcHealth()) + { + new_values[fmt::format("CASGCIsLeader_{}", name)] = { health->is_leader ? 1 : 0, + "Whether this server currently holds the content-addressed garbage-collection lease for the disk (1) or not (0, e.g. another replica is leading)." }; + new_values[fmt::format("CASGCPendingReclaim_{}", name)] = { health->pending_reclaim, + "Cumulative content-addressed objects condemned minus objects physically deleted by this process while it has held the GC lease on the disk. A persistently growing value indicates GC is not keeping up with reclaim." }; + new_values[fmt::format("CASGCLastSuccessAgeSeconds_{}", name)] = { health->last_success_age_seconds, + "Seconds since this process last completed a successful content-addressed GC round as leader on the disk (0 if it has never led one)." }; + new_values[fmt::format("CASGCWedgedNamespaces_{}", name)] = { health->wedged_namespace_count, + "Number of content-addressed namespaces on the disk currently stuck behind a wedged reference lane, unable to make GC progress." }; + } + } + } + catch (...) // NOLINT(bugprone-empty-catch) + { + /// Sampled on every server tick for every disk; a transient failure here (e.g. a + /// store health query hiccup) must never break the rest of asynchronous-metrics + /// collection. + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/Interpreters/SystemLog.cpp b/src/Interpreters/SystemLog.cpp index 778e7b087fa8..e935c6a1d0f7 100644 --- a/src/Interpreters/SystemLog.cpp +++ b/src/Interpreters/SystemLog.cpp @@ -34,6 +34,8 @@ #include #include #include +#include +#include #include #include #include diff --git a/src/Interpreters/SystemLog.h b/src/Interpreters/SystemLog.h index fbd498ea97d8..abea7b712ac6 100644 --- a/src/Interpreters/SystemLog.h +++ b/src/Interpreters/SystemLog.h @@ -17,6 +17,8 @@ M(QueryLog, query_log, "Contains information about executed queries, for example, start time, duration of processing, error messages.") \ M(QueryThreadLog, query_thread_log, "Contains information about threads that execute queries, for example, thread name, thread start time, duration of query processing.") \ M(PartLog, part_log, "This table contains information about events that occurred with data parts in the MergeTree family tables, such as adding or merging data.") \ + M(ContentAddressedGarbageCollectionLog, cas_gc_log, "Per-round records of the content-addressed (CA) MergeTree garbage collector: a Start and a Finish row per GC round, with counts of objects marked/deleted, duration, outcome, and per-round ProfileEvents.") \ + M(ContentAddressedLog, cas_log, "Per-event content-addressed (CA) MergeTree audit log: one row per blob/ref/GC decision (put, reuse, retire, delete, root add/remove, in-degree-zero, fence, lease, ...) plus errors (dangling access, fail-closed). Enabled by default while the CA disk feature is experimental (see config.xml); it is the primary forensic instrument for a CA issue and costs nothing when no CA disk is configured.") \ M(BackgroundSchedulePoolLog, background_schedule_pool_log, "Contains history of background schedule pool task executions.") \ M(TraceLog, trace_log, "Contains stack traces collected by the sampling query profiler.") \ M(CrashLog, crash_log, "Contains information about stack traces for fatal errors. The table does not exist in the database by default, it is created only when fatal errors occur.") \ diff --git a/src/Interpreters/ThreadStatusExt.cpp b/src/Interpreters/ThreadStatusExt.cpp index 36b18da61316..c5efed73daf9 100644 --- a/src/Interpreters/ThreadStatusExt.cpp +++ b/src/Interpreters/ThreadStatusExt.cpp @@ -138,6 +138,8 @@ ThreadGroup::ThreadGroup(ThreadGroupPtr parent_thread_group) , global_context(parent->global_context) , fatal_error_callback(parent->fatal_error_callback) , os_threads_nice_value(parent->os_threads_nice_value) + /// Keep the parent group alive: this child parents its trackers at the parent's via raw pointers (B90). + , parent_thread_group(parent) , memory_spill_scheduler(parent->memory_spill_scheduler) , performance_counters(VariableContext::Process, &parent->performance_counters) , memory_tracker(&parent->memory_tracker, VariableContext::Process, /*log_peak_memory_usage_in_destructor*/ false) @@ -152,6 +154,8 @@ ThreadGroup::ThreadGroup(ContextPtr query_context_, ThreadGroupPtr parent_thread , global_context(query_context_->getGlobalContext()) , fatal_error_callback(parent->fatal_error_callback) , os_threads_nice_value(parent->os_threads_nice_value) + /// Keep the parent group alive: this child parents its trackers at the parent's via raw pointers (B90). + , parent_thread_group(parent) , memory_spill_scheduler(parent->memory_spill_scheduler) , performance_counters(VariableContext::Process, &parent->performance_counters) , memory_tracker(&parent->memory_tracker, VariableContext::Process, /*log_peak_memory_usage_in_destructor*/ false) diff --git a/src/Parsers/ASTSystemQuery.cpp b/src/Parsers/ASTSystemQuery.cpp index bf7cd7f177dc..c5da6d8dffdf 100644 --- a/src/Parsers/ASTSystemQuery.cpp +++ b/src/Parsers/ASTSystemQuery.cpp @@ -169,6 +169,9 @@ void ASTSystemQuery::formatImpl(WriteBuffer & ostr, const FormatSettings & setti Type::CLEAR_DISTRIBUTED_CACHE, Type::SYNC_FILESYSTEM_CACHE, Type::CLEAR_QUERY_CACHE, + /// The grammar parses ` FROM DISK ` before `ON CLUSTER` (ParserSystemQuery.cpp), + /// so the round-trip format must print it last too. + Type::CAS_DROP_POOL_MEMBER, }; if (!queries_with_on_cluster_at_end.contains(type) && !cluster.empty()) @@ -276,14 +279,51 @@ void ASTSystemQuery::formatImpl(WriteBuffer & ostr, const FormatSettings & setti break; } + case Type::CAS_GC_REBUILD: + { + /// FORCE precedes the required disk name: SYSTEM CAS GC REBUILD + /// [FORCE] . + if (cas_gc_rebuild_force) + print_keyword(" FORCE"); + if (!disk.empty()) + { + ostr << ' '; + print_identifier(disk); + } + break; + } + case Type::CAS_DROP_POOL_MEMBER: + { + /// SYSTEM CAS DROP POOL MEMBER FROM DISK -- both required, both + /// quoted string literals (unlike the sibling CAS_* commands' bare identifier + /// disk target: an srid is an opaque server-root path, not necessarily identifier-shaped). + ostr << ' ' << quoteString(replica); + print_keyword(" FROM DISK ") << quoteString(disk); + break; + } + case Type::CAS_FSCK: + case Type::CAS_FORGET: + case Type::CAS_GC_STOP: + case Type::CAS_GC_START: + { + /// SYSTEM CAS FSCK/FORGET/GC STOP/GC START -- the disk is REQUIRED + /// (unlike GC RUN's optional disk): each scan/decommission/scheduler-control verb targets + /// exactly one disk, never a fan-out. + ostr << ' '; + print_identifier(disk); + break; + } case Type::RELOAD_DICTIONARY: case Type::UNLOAD_DICTIONARY: case Type::RELOAD_MODEL: case Type::RELOAD_FUNCTION: + case Type::CAS_GC_RUN: case Type::RESTART_DISK: case Type::WAIT_BLOBS_CLEANUP: case Type::CLEAR_DISK_METADATA_CACHE: { + /// RELOAD DICTIONARY prints its database/table target, RELOAD MODEL/FUNCTION their + /// identifier target; CAS GC RUN's disk is optional. if (table) { ostr << ' '; diff --git a/src/Parsers/ASTSystemQuery.h b/src/Parsers/ASTSystemQuery.h index 1439e9c1743e..2ee18aa8bb5e 100644 --- a/src/Parsers/ASTSystemQuery.h +++ b/src/Parsers/ASTSystemQuery.h @@ -159,6 +159,7 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster INSTRUMENT_ADD, INSTRUMENT_REMOVE, RESET_DDL_WORKER, +<<<<<<< HEAD STOP_ALL_BACKGROUND, START_ALL_BACKGROUND, PAUSE_ALL_BACKGROUND, @@ -169,6 +170,15 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster PAUSE, CANCEL, REFRESH, +======= + CAS_GC_RUN, + CAS_GC_REBUILD, + CAS_DROP_POOL_MEMBER, + CAS_FSCK, + CAS_FORGET, + CAS_GC_STOP, + CAS_GC_START, +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) END }; @@ -199,6 +209,9 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster String storage_policy; String volume; String disk; + /// SYSTEM CAS GC REBUILD FORCE [] — the raw baseline-rebuild disaster + /// recovery command's optional FORCE keyword (bypass the "healthy state" refusal). + bool cas_gc_rebuild_force = false; UInt64 seconds{}; UInt64 untracked_memory_size{}; diff --git a/src/Parsers/ParserSystemQuery.cpp b/src/Parsers/ParserSystemQuery.cpp index 4ac8beaf354b..d750b1d94a61 100644 --- a/src/Parsers/ParserSystemQuery.cpp +++ b/src/Parsers/ParserSystemQuery.cpp @@ -514,6 +514,70 @@ bool ParserSystemQuery::parseImpl(IParser::Pos & pos, ASTPtr & node, Expected & return false; break; } + case Type::CAS_GC_RUN: + { + /// SYSTEM CAS GC RUN [] [ON CLUSTER cluster]. The disk is OPTIONAL. + /// When omitted, the empty disk means "all content-addressed disks on this node". + /// First try the full target form (which also handles ON CLUSTER); if no disk follows, + /// fall back to parsing just the optional ON CLUSTER clause and leave the disk empty. + auto saved_pos = pos; + Expected target_expected = expected; + if (!parseQueryWithOnClusterAndTarget(res, pos, target_expected, SystemQueryTargetType::Disk)) + { + pos = saved_pos; + res->disk.clear(); + if (!parseQueryWithOnCluster(res, pos, expected)) + return false; + } + break; + } + case Type::CAS_GC_REBUILD: + { + /// SYSTEM CAS GC REBUILD [FORCE] [ON CLUSTER cluster]. Unlike the + /// per-round GC RUN command, REBUILD requires an EXPLICIT disk: the destructive + /// baseline rebuild must never fan out across every content-addressed disk from a bare + /// command. parseQueryWithOnClusterAndTarget requires the target, so omitting the disk is a + /// syntax error. + res->cas_gc_rebuild_force = ParserKeyword{Keyword::FORCE}.ignore(pos, expected); + if (!parseQueryWithOnClusterAndTarget(res, pos, expected, SystemQueryTargetType::Disk)) + return false; + break; + } + case Type::CAS_FSCK: + case Type::CAS_FORGET: + case Type::CAS_GC_STOP: + case Type::CAS_GC_START: + { + /// SYSTEM CAS FSCK/FORGET/GC STOP/GC START [ON CLUSTER cluster]. + /// Unlike GC RUN, the disk is REQUIRED -- mirrors CAS_GC_REBUILD (minus the FORCE + /// keyword): parseQueryWithOnClusterAndTarget requires the target, so omitting the disk is a + /// syntax error rather than a silent fan-out across every content-addressed disk. + if (!parseQueryWithOnClusterAndTarget(res, pos, expected, SystemQueryTargetType::Disk)) + return false; + break; + } + case Type::CAS_DROP_POOL_MEMBER: + { + /// SYSTEM CAS DROP POOL MEMBER FROM DISK [ON CLUSTER cluster]. + /// Both the srid and the disk name are REQUIRED quoted string literals -- an srid is an + /// opaque server-root path (may contain '/'), not the bare identifier the sibling + /// CAS_* commands' disk TARGET accepts, so this does not go through + /// parseQueryWithOnClusterAndTarget. + ASTPtr ast; + if (!ParserStringLiteral{}.parse(pos, ast, expected)) + return false; + res->replica = ast->as().value.safeGet(); + if (!ParserKeyword{Keyword::FROM}.ignore(pos, expected)) + return false; + if (!ParserKeyword{Keyword::DISK}.ignore(pos, expected)) + return false; + if (!ParserStringLiteral{}.parse(pos, ast, expected)) + return false; + res->disk = ast->as().value.safeGet(); + if (!parseQueryWithOnCluster(res, pos, expected)) + return false; + break; + } /// FLUSH DISTRIBUTED requires table /// START/STOP DISTRIBUTED SENDS does not require table case Type::STOP_DISTRIBUTED_SENDS: diff --git a/src/Parsers/tests/gtest_Parser.cpp b/src/Parsers/tests/gtest_Parser.cpp index 588067e1751f..144ad7d08ade 100644 --- a/src/Parsers/tests/gtest_Parser.cpp +++ b/src/Parsers/tests/gtest_Parser.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -844,6 +845,7 @@ INSTANTIATE_TEST_SUITE_P(ParserRenameQuery, ParserTest, } }))); +<<<<<<< HEAD #ifdef DEBUG_OR_SANITIZER_BUILD /// Regression test for the UBSan "member call on null pointer of type DB::IAST" at /// ASTRenameQuery::formatQueryImpl (RENAME DATABASE branch). A RENAME DATABASE node always @@ -879,6 +881,50 @@ TEST(ParserRenameQueryDeathTest, FormatNullToDatabaseAborts) EXPECT_DEATH(ast->formatWithSecretsOneLine(), "elements.at\\(0\\).to.database"); } #endif +======= +// SYSTEM CAS DROP POOL MEMBER: srid and disk are both required quoted string literals +// (an srid is an opaque server-root path, not identifier-shaped); ON CLUSTER round-trips as a bare +// identifier (ASTQueryWithOnCluster::formatOnCluster uses backQuoteIfNeed, no quoting needed for a +// plain name), even though the parser also accepts a quoted string literal for it on input. +INSTANTIATE_TEST_SUITE_P(ParserSystemQuery, ParserTest, + ::testing::Combine( + ::testing::Values(std::make_shared()), + ::testing::ValuesIn(std::initializer_list{ + { + "SYSTEM CAS DROP POOL MEMBER 'srv1' FROM DISK 'disk1'", + "SYSTEM CAS DROP POOL MEMBER 'srv1' FROM DISK 'disk1'" + }, + { + "SYSTEM CAS DROP POOL MEMBER 'srv1' FROM DISK 'disk1' ON CLUSTER my_cluster", + "SYSTEM CAS DROP POOL MEMBER 'srv1' FROM DISK 'disk1' ON CLUSTER my_cluster" + }, + { + "SYSTEM CAS DROP POOL MEMBER 'srv1'", // missing FROM DISK + nullptr + }, + { + "SYSTEM CAS DROP POOL MEMBER FROM DISK 'disk1'", // missing srid + nullptr + }, + { + "SYSTEM CAS GC RUN", + "SYSTEM CAS GC RUN" + }, + { + "SYSTEM CAS GC RUN disk1", + "SYSTEM CAS GC RUN disk1" + }, + { + /// CAS_GC_RUN goes through the shared parseQueryWithOnClusterAndTarget + /// helper (like RESTART_DISK / WAIT_BLOBS_CLEANUP / CLEAR_DISK_METADATA_CACHE), whose + /// round-trip format always normalizes to "ON CLUSTER cluster target" -- unlike + /// CAS_DROP_POOL_MEMBER, which has its own dedicated grammar and prints + /// ON CLUSTER last. + "SYSTEM CAS GC RUN disk1 ON CLUSTER my_cluster", + "SYSTEM CAS GC RUN ON CLUSTER my_cluster disk1" + }, +}))); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) static constexpr size_t kDummyMaxQuerySize = 256 * 1024; static constexpr size_t kDummyMaxParserDepth = 256; diff --git a/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp b/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp index 37a9ad04aa93..a6bcc7bf66be 100644 --- a/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp +++ b/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -16,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -40,6 +42,7 @@ namespace ErrorCodes extern const int LOGICAL_ERROR; extern const int FILE_DOESNT_EXIST; extern const int CORRUPTED_DATA; + extern const int SUPPORT_IS_DISABLED; } namespace @@ -305,6 +308,16 @@ bool DataPartStorageOnDiskBase::isStoredOnRemoteDisk() const return volume->getDisk()->isRemote(); } +bool DataPartStorageOnDiskBase::isContentAddressed() const +{ + return volume->getDisk()->isContentAddressed(); +} + +bool DataPartStorageOnDiskBase::supportsAtomicFileWrites() const +{ + return volume->getDisk()->supportsAtomicFileWrites(); +} + std::optional DataPartStorageOnDiskBase::getCacheName() const { if (volume->getDisk()->supportsCache()) @@ -436,6 +449,18 @@ void DataPartStorageOnDiskBase::backup( auto disk = volume->getDisk(); + /// B34: the temporary-hard-link BACKUP path (used for Ordinary, non-UUID databases) calls + /// disk->createHardLink with a non-part-shaped temp path, which on a CAS disk + /// would otherwise surface as a raw LOGICAL_ERROR. Fail closed with a clear message instead. + /// The pointer-holding path (make_temporary_hard_links=false, used by Atomic/UUID databases) + /// uses getStorageObjects and round-trips on a CAS disk, so it is left untouched. + if (make_temporary_hard_links && disk->isContentAddressed()) + throw Exception( + ErrorCodes::SUPPORT_IS_DISABLED, + "BACKUP via temporary hard links is not supported on a CAS disk yet (B16/B34); " + "use an Atomic database (which backs up via pointer-holding) instead; disk '{}'", + disk->getName()); + fs::path temp_part_dir; std::shared_ptr temp_dir_owner; if (make_temporary_hard_links) @@ -541,8 +566,20 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( const ClonePartParams & params) const { auto disk = volume->getDisk(); - if (params.external_transaction) - params.external_transaction->createDirectories(to); + + /// A CAS disk models a part as one atomic unit (N files -> one manifest -> one ref). + /// The per-file createHardLink autocommit Backup uses with no enclosing transaction would publish a + /// one-file ref per file and overwrite the destination, leaving the clone with only its last file + /// (the B21 corruption mode — seen as system.detached_parts listing metadata_version.txt instead of + /// the detached part dir, B36). When the caller did not supply a transaction, run the whole clone + /// through ONE self-created disk transaction so all files land in a single content-addressed part. + DiskTransactionPtr owned_transaction; + if (!params.external_transaction && disk->isContentAddressed()) + owned_transaction = disk->createTransaction(); + const DiskTransactionPtr & clone_transaction = params.external_transaction ? params.external_transaction : owned_transaction; + + if (clone_transaction) + clone_transaction->createDirectories(to); else disk->createDirectories(to); @@ -557,11 +594,12 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( /* max_level= */ {}, params.copy_instead_of_hardlink, params.files_to_copy_instead_of_hardlinks, - params.external_transaction); + clone_transaction); if (save_metadata_callback) save_metadata_callback(disk); +<<<<<<< HEAD /// Also remove any leftover `txn_version.txt.tmp`: leaving it without the main file makes the /// cloned/frozen part load as a rolled-back transaction (see `VersionMetadataOnDisk::loadMetadata`) /// and get discarded as `Outdated`. Remove the temporary file before the main file so the cleanup @@ -575,6 +613,35 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( if (!params.keep_metadata_version) params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*params.external_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); +======= + if (clone_transaction) + { + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); + if (!params.keep_metadata_version) + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); + + /// When the caller wants a fresh metadata version written into the clone (the Replicated queue + /// clone path — `executeReplaceRange`/`replacePartitionFrom`/`movePartitionToTable` set + /// `metadata_version_to_write`), write `metadata_version.txt` INSIDE the clone transaction so it + /// is part of the single whole-part commit. On a content-addressed disk the part is published + /// atomically at `commit`; a separate post-clone autocommit `writeFile` of this part file (what + /// `cloneAndLoadDataPart` does for non-CA disks) would hit the per-file-autocommit guard (B21). + /// `cloneAndLoadDataPart`'s own post-clone write now runs unconditionally (the freeze special + /// case was dropped with all-tree Task 10): identical bytes land as a byte-equal repoint + /// no-op, differing bytes as a legal repoint. + if (params.metadata_version_to_write.has_value()) + { + chassert(!params.keep_metadata_version); + auto out_metadata = clone_transaction->writeFile( + fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME, + 4096, + WriteMode::Rewrite, + write_settings); + writeText(*params.metadata_version_to_write, *out_metadata); + out_metadata->finalize(); + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } else { @@ -586,12 +653,20 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*disk, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); } +<<<<<<< HEAD /// Make the hardlink clone durable (the Backup loop above fsyncs nothing). This runs /// synchronously before freeze returns, so a caller that afterwards makes a destructive change /// (e.g. DETACH commits a covering empty part and drops the source) sees the clone already on /// disk. See the commit message / #111382 for the full rationale. if (params.fsync_part_directory && !params.external_transaction && !disk->isRemote()) fsyncFrozenCloneTree(*disk, fs::path(to) / dir_path); +======= + /// Commit the self-created transaction (the whole-part clone commit point for CA). An external + /// transaction is committed by its owner, as before. Before the arena scope below, so the commit's + /// own allocations are not attributed to the MergeTree arena. + if (owned_transaction) + owned_transaction->commit(); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// The SingleDiskVolume and the DataPartStorageOnDiskFull built by `create` are stored on the /// frozen part for its whole lifetime; route them into the dedicated MergeTree arena, like the @@ -606,6 +681,52 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( return frozen_storage; } +namespace +{ + +/// Recursively copy every file under `source_path` on `src_disk` into `destination_path` through +/// `dst_transaction`'s NON-autocommit `writeFile` (IDiskTransaction::writeFile, NOT +/// writeFileWithAutoCommit) -- the same primitive `freeze` already uses for a single file (the +/// metadata_version.txt write, DataPartStorageOnDiskBase::freeze). Cross-disk, so it cannot reuse +/// Backup()/BackupImpl: that helper's transactional branch calls transaction->copyFile, which is +/// SAME-disk only (DiskObjectStorageTransaction::copyFile throws NOT_IMPLEMENTED across disks on +/// CA), and its non-transactional branch always autocommits per file via IDisk::copyFile / +/// copyDirectoryContent. Sequential, not the parallel copyThroughBuffers thread pool: a +/// content-addressed transaction batches every file into ONE eventual manifest, and its staging +/// map is not mutex-guarded. Its callers are a background move and a user-issued cross-disk attach, +/// so parallelizing this remains a deferred optimization whose cost is now visible to a waiting +/// statement rather than only to a background operation. +void copyDirectoryContentIntoTransaction( + IDisk & src_disk, + const String & source_path, + IDiskTransaction & dst_transaction, + const String & destination_path, + const ReadSettings & read_settings, + const WriteSettings & write_settings, + const std::function & cancellation_hook) +{ + dst_transaction.createDirectories(destination_path); + for (auto it = src_disk.iterateDirectory(source_path); it->isValid(); it->next()) + { + auto source = it->path(); + auto destination = fs::path(destination_path) / it->name(); + + if (src_disk.existsDirectory(source)) + { + copyDirectoryContentIntoTransaction( + src_disk, source, dst_transaction, destination, read_settings, write_settings, cancellation_hook); + continue; + } + + auto in = src_disk.readFile(source, read_settings); + auto out = dst_transaction.writeFile(destination, DBMS_DEFAULT_BUFFER_SIZE, WriteMode::Rewrite, write_settings); + copyData(*in, *out, cancellation_hook); + out->finalize(); + } +} + +} + MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( const std::string & to, const std::string & dir_path, @@ -616,30 +737,61 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( const ClonePartParams & params) const { auto src_disk = volume->getDisk(); - if (params.external_transaction) - params.external_transaction->createDirectories(to); - else - dst_disk->createDirectories(to); - /// freezeRemote() using copy instead of hardlinks for all files - /// In this case, files_to_copy_intead_of_hardlinks is set by empty - Backup( - src_disk, - dst_disk, - getRelativePath(), - fs::path(to) / dir_path, - read_settings, - write_settings, - params.make_source_readonly, - /* max_level= */ {}, - true, - /* files_to_copy_intead_of_hardlinks= */ {}, - params.external_transaction); + /// A content-addressed destination models a part as ONE atomic unit: N files become one manifest + /// and one ref. The generic path below fans the files onto a thread pool, and each becomes an + /// independent autocommit transaction against that same ref -- two of them resolve it as absent, + /// both publish a one-file manifest, and the loser is refused. So when the caller supplied no + /// transaction of its own, run the whole clone through ONE self-created transaction, the same + /// shape `freeze` uses. `Backup` cannot serve this path: its transactional branch calls + /// `copyFile` on the transaction, which is same-disk only and refuses a cross-disk + /// content-addressed copy. + DiskTransactionPtr owned_transaction; + if (!params.external_transaction && dst_disk->isContentAddressed()) + owned_transaction = dst_disk->createTransaction(); + + if (owned_transaction) + { + try + { + copyDirectoryContentIntoTransaction( + *src_disk, getRelativePath(), *owned_transaction, fs::path(to) / dir_path, + read_settings, write_settings, /* cancellation_hook= */ {}); + } + catch (...) + { + owned_transaction->undo(); + throw; + } + } + else + { + if (params.external_transaction) + params.external_transaction->createDirectories(to); + else + dst_disk->createDirectories(to); + + /// `freezeRemote` using copy instead of hardlinks for all files + /// In this case, files_to_copy_intead_of_hardlinks is set by empty + Backup( + src_disk, + dst_disk, + getRelativePath(), + fs::path(to) / dir_path, + read_settings, + write_settings, + params.make_source_readonly, + /* max_level= */ {}, + true, + /* files_to_copy_intead_of_hardlinks= */ {}, + params.external_transaction); + } /// The save_metadata_callback function acts on the target dist. if (save_metadata_callback) save_metadata_callback(dst_disk); +<<<<<<< HEAD /// Also remove any leftover `txn_version.txt.tmp`: leaving it without the main file makes the /// cloned/frozen part load as a rolled-back transaction (see `VersionMetadataOnDisk::loadMetadata`) /// and get discarded as `Outdated`. Remove the temporary file before the main file so the cleanup @@ -653,6 +805,28 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( if (!params.keep_metadata_version) params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*params.external_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); +======= + /// These removals belong to the clone. On the content-addressed arm they MUST go through the same + /// transaction: sent straight to the disk they would autocommit, which is exactly the + /// one-publish-per-file behaviour the single transaction above exists to prevent. + if (const DiskTransactionPtr & clone_transaction = owned_transaction ? owned_transaction : params.external_transaction) + { + try + { + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); + if (!params.keep_metadata_version) + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); + if (owned_transaction) + owned_transaction->commit(); + } + catch (...) + { + if (owned_transaction) + owned_transaction->undo(); + throw; + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } else { @@ -695,18 +869,46 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::clonePart( dir_path, getRelativePath(), path_to_clone, fullPath(dst_disk, path_to_clone)); } - try + if (dst_disk->isContentAddressed()) { - dst_disk->createDirectories(to); - src_disk->copyDirectoryContent(getRelativePath(), dst_disk, path_to_clone, read_settings, write_settings, cancellation_hook); + /// L2 (MOVE-to-CA fix): a content-addressed disk models a part as ONE atomic unit (N + /// files -> one manifest -> one ref). The generic per-file autocommit path below would + /// publish a separate one-file ref per file -- colliding on the shared "moving" ref + /// before L1, and throwing NOT_IMPLEMENTED on a non-first content file even after L1 + /// ("Autocommit writes are not supported for content part files"). Run the whole clone + /// through ONE self-created disk transaction instead, mirroring freeze's + /// owned_transaction shape -- but streaming cross-disk bytes, since freeze's Backup() is + /// same-disk hardlink/copyFile (throws NOT_IMPLEMENTED for CA cross-disk). + auto clone_transaction = dst_disk->createTransaction(); + try + { + copyDirectoryContentIntoTransaction( + *src_disk, getRelativePath(), *clone_transaction, path_to_clone, + read_settings, write_settings, cancellation_hook); + clone_transaction->commit(); + } + catch (...) + { + LOG_WARNING(log, "Rolling back transaction after failed attempt to move a data part to {}", path_to_clone); + clone_transaction->undo(); + throw; + } } - catch (...) + else { - /// It's safe to remove it recursively (even with zero-copy-replication) - /// because we've just did full copy through copyDirectoryContent - LOG_WARNING(log, "Removing directory {} after failed attempt to move a data part", path_to_clone); - dst_disk->removeRecursive(path_to_clone); - throw; + try + { + dst_disk->createDirectories(to); + src_disk->copyDirectoryContent(getRelativePath(), dst_disk, path_to_clone, read_settings, write_settings, cancellation_hook); + } + catch (...) + { + /// It's safe to remove it recursively (even with zero-copy-replication) + /// because we've just did full copy through copyDirectoryContent + LOG_WARNING(log, "Removing directory {} after failed attempt to move a data part", path_to_clone); + dst_disk->removeRecursive(path_to_clone); + throw; + } } /// The SingleDiskVolume and the DataPartStorageOnDiskFull built by `create` are stored on the diff --git a/src/Storages/MergeTree/DataPartStorageOnDiskBase.h b/src/Storages/MergeTree/DataPartStorageOnDiskBase.h index bceb343c342d..dbe0f3915987 100644 --- a/src/Storages/MergeTree/DataPartStorageOnDiskBase.h +++ b/src/Storages/MergeTree/DataPartStorageOnDiskBase.h @@ -40,6 +40,8 @@ class DataPartStorageOnDiskBase : public IDataPartStorage std::string getDiskName() const override; std::string getDiskType() const override; bool isStoredOnRemoteDisk() const override; + bool isContentAddressed() const override; + bool supportsAtomicFileWrites() const override; std::optional getCacheName() const override; bool supportZeroCopyReplication() const override; bool supportParallelWrite() const override; diff --git a/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp b/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp index dba2684113ac..74ddaa16b49e 100644 --- a/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp +++ b/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -9,14 +10,24 @@ #include #include #include +#include #include +#include +#include + namespace DB { +namespace FailPoints +{ + extern const char part_storage_fail_commit_transaction[]; +} + namespace ErrorCodes { extern const int LOGICAL_ERROR; + extern const int FAULT_INJECTED; } DataPartStorageOnDiskFull::DataPartStorageOnDiskFull(VolumePtr volume_, std::string root_path_, std::string part_dir_) @@ -51,17 +62,43 @@ DataPartStoragePtr DataPartStorageOnDiskFull::getProjection(const std::string & bool DataPartStorageOnDiskFull::exists() const { - return volume->getDisk()->existsDirectory(fs::path(root_path) / part_dir); + auto path = fs::path(root_path) / part_dir; + /// CA read-your-writes: a part dir being assembled by this transaction (e.g. a carried-forward + /// projection dir staged into the open whole-part txn) is not on committed metadata yet. Mirrors + /// existsDirectory at directory granularity for the part's OWN directory. + if (transaction && transaction->hasInFlightDirectory(path)) + return true; + return volume->getDisk()->existsDirectory(path); } bool DataPartStorageOnDiskFull::existsFileImpl(const std::string & name) const { +<<<<<<< HEAD return volume->getDisk()->existsFile(fs::path(root_path) / part_dir / name); +======= + auto path = fs::path(root_path) / part_dir / name; + /// B59: a part still being assembled by this transaction can have staged-but-uncommitted files + /// (e.g. projection temp blocks on a content-addressed disk). Consult the held transaction first. + if (transaction && transaction->tryGetInFlightFileSize(path).has_value()) + return true; + if (looksLikePackedSkipIndexFile(name)) + { + if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) + return true; + } + return volume->getDisk()->existsFile(path); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } bool DataPartStorageOnDiskFull::existsDirectory(const std::string & name) const { - return volume->getDisk()->existsDirectory(fs::path(root_path) / part_dir / name); + auto path = fs::path(root_path) / part_dir / name; + /// CA read-your-writes: a part still being assembled by this transaction can have a staged-but-uncommitted + /// directory (e.g. a carried-forward projection hardlinked into the open whole-part txn) that committed + /// metadata cannot see yet. Mirrors existsFile (B59) at directory granularity. + if (transaction && transaction->hasInFlightDirectory(path)) + return true; + return volume->getDisk()->existsDirectory(path); } class DataPartStorageIteratorOnDisk final : public IDataPartStorageIterator @@ -83,11 +120,52 @@ class DataPartStorageIteratorOnDisk final : public IDataPartStorageIterator DirectoryIteratorPtr it; }; +/// CA read-your-writes directory enumeration: a merged view of the committed disk entries PLUS the +/// immediate children this transaction has STAGED under the part dir (deduplicated). Used so +/// loadProjections' withPartFormatFromDisk can iterate a staged-but-uncommitted projection directory and +/// find its mark file. Mirrors existsFile/existsDirectory (B59) at the enumeration level; the committed +/// entries dominate (a name present both on disk and staged appears once). +class DataPartStorageMergedIterator final : public IDataPartStorageIterator +{ +public: + DataPartStorageMergedIterator(DiskPtr disk_, std::string dir_path_, std::vector names_) + : disk(std::move(disk_)), dir_path(std::move(dir_path_)), names(std::move(names_)) + { + } + + void next() override { ++pos; } + bool isValid() const override { return pos < names.size(); } + std::string name() const override { return names[pos]; } + std::string path() const override { return fs::path(dir_path) / names[pos]; } + bool isFile() const override { return isValid() && disk->existsFile(path()); } + +private: + DiskPtr disk; + std::string dir_path; + std::vector names; + size_t pos = 0; +}; + DataPartStorageIteratorPtr DataPartStorageOnDiskFull::iterate() const { + auto dir_path = fs::path(root_path) / part_dir; + if (transaction) + { + if (auto staged = transaction->listInFlightDirectory(dir_path); !staged.empty()) + { + /// Union the committed entries with the staged children (set semantics, committed dominates). + std::set names(staged.begin(), staged.end()); + if (volume->getDisk()->existsDirectory(dir_path)) + for (auto it = volume->getDisk()->iterateDirectory(dir_path); it->isValid(); it->next()) + names.insert(it->name()); + return std::make_unique( + volume->getDisk(), dir_path, std::vector(names.begin(), names.end())); + } + } + return std::make_unique( volume->getDisk(), - volume->getDisk()->iterateDirectory(fs::path(root_path) / part_dir)); + volume->getDisk()->iterateDirectory(dir_path)); } Poco::Timestamp DataPartStorageOnDiskFull::getFileLastModified(const String & file_name) const @@ -102,10 +180,21 @@ size_t DataPartStorageOnDiskFull::getFileSizeImpl(const String & file_name) cons std::optional DataPartStorageOnDiskFull::getPackedFileUncompressedSize(const std::string & file_name) const { + auto path = fs::path(root_path) / part_dir / file_name; + /// B59: see existsFile — the merge stats the staged temp files before reading them back. + if (transaction) + if (auto size = transaction->tryGetInFlightFileSize(path)) + return *size; if (looksLikePackedSkipIndexFile(file_name)) if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(file_name)) +<<<<<<< HEAD return reader->getFileUncompressedSize(file_name); return {}; +======= + return reader->getFileSize(file_name); + } + return volume->getDisk()->getFileSize(path); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } UInt32 DataPartStorageOnDiskFull::getRefCount(const String & file_name) const @@ -116,7 +205,17 @@ UInt32 DataPartStorageOnDiskFull::getRefCount(const String & file_name) const std::vector DataPartStorageOnDiskFull::getRemotePaths(const std::string & file_name) const { const std::string path = fs::path(root_path) / part_dir / file_name; - auto objects = volume->getDisk()->getStorageObjects(path); + + /// B59: a file staged by this transaction resolves to its already-uploaded blob object(s) before commit. + /// A mutable per-part file intentionally does NOT resolve here (tryGetInFlightStorageObjects returns + /// nullopt → falls through): it has no blob object and must be read via tryReadFileInFlight. The merge + /// reads projection column blocks (blob-backed) through this path, not mutable files. + StoredObjects objects; + if (transaction) + if (auto inflight = transaction->tryGetInFlightStorageObjects(path)) + objects = std::move(*inflight); + if (objects.empty()) + objects = volume->getDisk()->getStorageObjects(path); std::vector remote_paths; remote_paths.reserve(objects.size()); @@ -142,7 +241,68 @@ void DataPartStorageOnDiskFull::prepareReadImpl( std::optional read_hint, ReadPipeline & pipeline) const { +<<<<<<< HEAD volume->getDisk()->prepareRead(fs::path(root_path) / part_dir / name, settings, read_hint, pipeline); +======= + auto path = fs::path(root_path) / part_dir / name; + + /// B59: read-your-writes for a part still being assembled by this transaction. A projection + /// spill-and-merge reads its own temp blocks back before the parent part's single commit; on a + /// content-addressed disk those files are staged in the transaction (blob uploaded, no ref yet), + /// so the committed metadata path can't see them. If the held transaction resolves the file + /// in-flight, serve it via a custom pipeline source that reads through the transaction. Gated on + /// `transaction != nullptr` so committed-part reads (no open transaction) are unchanged. + if (transaction) + { + StoredObjects inflight_objects; + if (auto objs = transaction->tryGetInFlightStorageObjects(path)) + inflight_objects = std::move(*objs); + else if (auto size = transaction->tryGetInFlightFileSize(path)) + /// Mutable per-part file staged inline (no blob object); synthesize a placeholder so the + /// single-object pipeline is satisfied — the custom creator below ignores it and reads the + /// inline bytes through the transaction. + inflight_objects = StoredObjects{StoredObject(path, path, *size)}; + + if (!inflight_objects.empty()) + { + /// Safe to capture the raw transaction pointer: no cache/gather/async stage is added on this + /// branch, so the custom source is consumed synchronously inside build() during this read and + /// the pointer is never retained past it. + auto * tx = transaction.get(); + pipeline.setSource( + [tx, path](const StoredObject &, const ReadSettings & read_settings, bool /*use_external_buffer*/, bool /*restrict_seek*/) + { + return tx->tryReadFileInFlight(path, read_settings, std::nullopt); + }, + std::move(inflight_objects), + settings); + return; + } + } + + if (looksLikePackedSkipIndexFile(name)) + { + if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) + { + /// Packed substreams skip the disk's normal pipeline (filesystem cache, + /// async prefetch, etc.) and read through PackedFilesReader::readFile, which + /// opens the archive via the underlying disk and wraps the result with + /// ReadBufferFromFileView at the right offset. The archive's current location is + /// captured here and passed in, so the reader holds no path of its own. + auto disk = volume->getDisk(); + String archive_path = fs::path(root_path) / part_dir / String(SKIP_INDICES_PACKED_FILENAME); + ReadPipeline::BufferCreator creator = + [reader, disk, archive_path, name, read_hint](const StoredObject &, const ReadSettings & s, bool, bool) + { + return reader->readFile(disk, archive_path, name, s, read_hint); + }; + pipeline.setSource(std::move(creator), StoredObjects{StoredObject{}}, settings); + return; + } + } + + volume->getDisk()->prepareRead(path, settings, read_hint, pipeline); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr DataPartStorageOnDiskFull::readFileIfExistsImpl( @@ -150,7 +310,26 @@ std::unique_ptr DataPartStorageOnDiskFull::readFileIfExi const ReadSettings & settings, std::optional read_hint) const { +<<<<<<< HEAD return volume->getDisk()->readFileIfExists(fs::path(root_path) / part_dir / name, settings, read_hint); +======= + auto path = fs::path(root_path) / part_dir / name; + /// B59: serve a file staged by this transaction (uploaded blob or inline mutable bytes) before commit. + /// This direct delegate bypasses prepareRead, so the in-flight guard must be repeated here; it is the + /// only path that reaches the inline-mutable case via a returned buffer. + if (transaction) + if (auto rb = transaction->tryReadFileInFlight(path, settings, read_hint)) + return rb; + if (looksLikePackedSkipIndexFile(name)) + { + if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) + return reader->readFile( + volume->getDisk(), + fs::path(root_path) / part_dir / String(SKIP_INDICES_PACKED_FILENAME), + name, settings, read_hint); + } + return volume->getDisk()->readFileIfExists(path, settings, read_hint); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr DataPartStorageOnDiskFull::writeFile( @@ -239,20 +418,38 @@ void DataPartStorageOnDiskFull::createProjection(const std::string & name) void DataPartStorageOnDiskFull::beginTransaction() { + /// A borrowed projection sub-part shares the PARENT part's whole-part transaction (on a + /// content-addressed disk a part is one atomic unit: one manifest + one ref). It must not open its + /// own — riding the parent transaction is the point (B58) — so begin is a no-op here. This + /// centralizes the rule the 6 merge/mutate call sites used to duplicate as + /// `if (!isContentAddressed()) beginTransaction()`. + if (has_shared_transaction) + return; + if (transaction) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "Uncommitted{}transaction already exists", has_shared_transaction ? " shared " : " "); + throw Exception(ErrorCodes::LOGICAL_ERROR, "Uncommitted transaction already exists"); transaction = volume->getDisk()->createTransaction(); } void DataPartStorageOnDiskFull::commitTransaction() { + /// The mirror of beginTransaction: a borrowed projection sub-part rides the parent's transaction and + /// is published by the parent's single commit. Committing here would be committing someone else's + /// transaction, so it is a no-op. + if (has_shared_transaction) + return; + if (!transaction) throw Exception(ErrorCodes::LOGICAL_ERROR, "There is no uncommitted transaction"); - if (has_shared_transaction) - throw Exception(ErrorCodes::LOGICAL_ERROR, "Cannot commit shared transaction"); + /// Regression gate for the part-durability-before-Keeper-commit invariant: lets a test fail the + /// close of the PART's deferred disk transaction specifically (autocommit one-shot disk ops are + /// not affected, unlike disk_object_storage_fail_commit_metadata_transaction). + fiu_do_on(FailPoints::part_storage_fail_commit_transaction, + { + throw Exception(ErrorCodes::FAULT_INJECTED, "part_storage_fail_commit_transaction"); + }); transaction->commit(); transaction.reset(); diff --git a/src/Storages/MergeTree/DataPartsExchange.cpp b/src/Storages/MergeTree/DataPartsExchange.cpp index 519cead7f015..7f832485abf0 100644 --- a/src/Storages/MergeTree/DataPartsExchange.cpp +++ b/src/Storages/MergeTree/DataPartsExchange.cpp @@ -5,6 +5,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -45,6 +48,15 @@ namespace CurrentMetrics namespace DB { +namespace FailPoints +{ + /// CAS fetch-by-relink, receiver side. Both exist because the two exits they drive are properties of + /// the sender/receiver PAIR and of the interval between the receiver's publish and its confirm — and + /// neither is reachable from configuration, so an integration test cannot produce them any other way. + extern const char cas_relink_receiver_force_mechanism_failure[]; + extern const char cas_relink_receiver_pause_before_confirm[]; +} + namespace MergeTreeSetting { extern const MergeTreeSettingsBool allow_remote_fs_zero_copy_replication; @@ -61,6 +73,7 @@ namespace ErrorCodes extern const int CHECKSUM_DOESNT_MATCH; extern const int INSECURE_PATH; extern const int LOGICAL_ERROR; + extern const int NETWORK_ERROR; extern const int S3_ERROR; extern const int ZERO_COPY_REPLICATION_ERROR; } @@ -80,13 +93,82 @@ constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_PARTS_ZERO_COPY = 6; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_PARTS_PROJECTION = 7; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_METADATA_VERSION = 8; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_COLUMNS_SUBSTREAMS = 9; +<<<<<<< HEAD constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS = 10; +======= +/// CAS replication 2b: fetch-by-relink. The receiver advertises its content-addressed pool identity +/// (`cas_pool_uuid`) and, if it matches the sender's own pool, the sender sends only the +/// part's content id (`part_id`) + the mutable header — no file bytes — and the receiver "fetches" by +/// publishing its own ref to the blobs already present in the shared pool (the CA analogue of the +/// zero-copy metadata-only fetch). Everything is gated behind a matching pool_uuid, so a non-CA fetch +/// is byte-for-byte unchanged. +/// Kept although nothing gates on it any more: the offer gate moved to `..._WITH_CA_CONFIRM` below, but +/// 10 is a version peers still advertise, and deleting the record of what it meant would leave the next +/// reader unable to tell what an incoming 10 promises (a relink it will NOT confirm). +[[maybe_unused]] constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_CA_RELINK = 10; +/// CAS replication, publish-then-confirm (spec §wire-protocol). A relink offer is now accompanied by a +/// source token, and the endpoint answers a second, part-less request that asks whether that token is +/// still exactly what the sender's ref names. A server advertising this version serves the confirm +/// action; a receiver advertising it must confirm before it promotes. +constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM = 11; +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) std::string getEndpointId(const std::string & node_id) { return "DataPartsExchange:" + node_id; } +/// CAS replication 2b. The receiver advertises its target pool's identity under this request param so +/// the sender can decide whether a fetch-by-relink (same pool) is possible. +constexpr auto CA_POOL_UUID_PARAM = "cas_pool_uuid"; +/// Set on the response when the sender chose the relink path; the receiver then reads the relink payload +/// (the opaque encoded PartManifest body — self-contained, see part_manifest_v2 below) instead of the +/// byte stream. +constexpr auto CA_RELINK_COOKIE = "cas_relink"; +/// All-tree task 7: the manifest is now self-contained (uuid.txt/metadata_version.txt are ordinary +/// manifest entries, task 6), so the wire payload dropped its trailing metadata_version field (the +/// manifest bytes are now the ONLY field). Bumped from `part_manifest_v1` so a mixed-build pair (old +/// sender, new receiver) does not try to parse the old two-field payload under the new one-field shape +/// — the receiver rejects a cookie value it does not recognize and falls back to a byte fetch instead +/// of desyncing on the wire format. +constexpr auto CA_RELINK_COOKIE_VALUE = "part_manifest_v2"; + +/// CAS fetch-by-relink, publish-then-confirm (spec §wire-protocol). Three names make up the second +/// request of the handshake. +/// +/// The request parameter both selects the confirm action and carries its only argument: the opaque +/// source token the sender minted for the offer. There is no separate action flag, because an action +/// without its token is not a question anyone can answer, and a token without the action would have to +/// be ignored — one name cannot be half-present. +constexpr auto CA_CONFIRM_ACTION_PARAM = "cas_confirm"; +/// Response cookie on the relink offer: the token, opaque to the receiver, echoed back verbatim. +constexpr auto CA_CONFIRM_TOKEN_COOKIE = "cas_source_token"; +/// Response cookie on the confirm: the answer. +constexpr auto CA_CONFIRM_ANSWER_COOKIE = "cas_confirm_answer"; +/// The ONLY value that authorizes the receiver to promote. +constexpr auto CA_CONFIRM_ANSWER_PROVEN = "yes"; +/// Everything else: the source did not prove the binding. The wire vocabulary is deliberately BINARY +/// even though `CasConfirmAnswer` has three values. `No` and `Unknown` are one outcome for every caller +/// (see `CasConfirmAnswer`): gate 1 evaluates the mount fence LAST, so a mount that has already lost +/// its fence — and can no longer speak for the namespace at all — still answers `No` for a token that +/// does not match its last-known row. Putting `no` on the wire as a distinct value would invite a +/// receiver to act on it as knowledge, and it is not knowledge. The distinction is diagnostic only, so +/// it is logged on the sender, where the gate that produced it can be named, and never transmitted. +/// An ABSENT cookie reads as unproven too, which is what makes an older peer and a failed request the +/// same safe outcome as a refusal. +constexpr auto CA_CONFIRM_ANSWER_UNPROVEN = "unproven"; + +/// Resolve a disk to the content-addressed exchange facade, or nullptr if the disk is not CA. The +/// cast targets the purpose-built INTERFACE (IContentAddressedExchange), never the concrete +/// metadata-storage class (M-W design section 4). Used by both the relink sender (the part's +/// disk) and the relink receiver (the target disk). +IContentAddressedExchange * tryGetContentAddressedExchange(const DiskPtr & disk) +{ + if (!disk || !disk->isContentAddressed()) + return nullptr; + return dynamic_cast(disk->getMetadataStorage().get()); +} + /// Simple functor for tracking fetch progress in system.replicated_fetches table. struct ReplicatedFetchReadCallback { @@ -133,8 +215,108 @@ std::string Service::getId(const std::string & node_id) const return getEndpointId(node_id); } +CasConfirmAnswer Service::resolveContentAddressedConfirm( + const String & pool_uuid, + const String & server_root_id, + const String & root_namespace, + const String & ref_name, + const String & part_name, + const String & manifest_ref_text) const +{ + /// CAS fetch-by-relink, publish-then-confirm (spec §confirm-primitive). The receiver's own `+1` is + /// already durable when this runs; a `Yes` is what authorizes it to promote a part whose blobs are + /// protected only by THIS server's committed binding of that exact manifest. Every field below comes + /// from a remote peer, so nothing here is trusted beyond being used as a lookup key. + if (pool_uuid.empty() || server_root_id.empty() || root_namespace.empty() || ref_name.empty() || part_name.empty()) + return CasConfirmAnswer::Unknown; + + /// Routing. A pool UUID identifies the shared pool, not the mount: every server root writing into it + /// reports the same one, so the namespace's owner decides which instance may answer. EXACTLY one + /// match is required — zero means this table has no such disk, several mean the question is + /// ambiguous, and both are `Unknown` rather than a guess. + const IContentAddressedExchange * matched = nullptr; + DiskPtr matched_disk; + for (const auto & disk : data.getDisks()) + { + const auto * ca_meta = tryGetContentAddressedExchange(disk); + if (!ca_meta || ca_meta->getPoolUUID() != pool_uuid || !ca_meta->ownsNamespace(server_root_id, root_namespace)) + continue; + if (matched) + return CasConfirmAnswer::Unknown; + matched = ca_meta; + matched_disk = disk; + } + if (!matched) + return CasConfirmAnswer::Unknown; + + /// Gate 0 — the part-anchored fast filter. It is an AVAILABILITY filter and never a proof (spec + /// §confirm-primitive, demoted in rev.5): `rollbackDeletingParts` puts a part back to `Outdated` + /// after a failed filesystem removal, and the in-memory part path is deliberately not updated by a + /// `delete_tmp_*` rename, so an `Active`/`Outdated` part object authorizes nothing. What it buys is + /// a cheap `No` that costs no ledger work; every `Yes` is earned by gate 1 alone. + /// + /// `Deleting` is excluded by the state filter, an unknown name yields no part at all, and a part of + /// this name living on ANOTHER disk is rejected explicitly — `MOVE ... TO DISK` leaves a same-name + /// `Active` part behind on the destination disk, and only the instance the token routed to may be + /// the one the confirm is about. The parts set is read under its own lock, which + /// `getPartIfExists` takes and releases, and the part reference is dropped before any ledger lock. + { + const auto part_info = MergeTreePartInfo::tryParsePartName(part_name, data.format_version); + if (!part_info) + return CasConfirmAnswer::Unknown; + const auto part = data.getPartIfExists( + *part_info, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); + if (!part || part->getDataPartStorage().getDiskName() != matched_disk->getName()) + return CasConfirmAnswer::No; + } + + /// Gate 1 — authoritative, and the only source of a `Yes`. + return matched->confirmExactRef(root_namespace, ref_name, manifest_ref_text); +} + +void Service::answerContentAddressedConfirm(const String & token_text, HTTPServerResponse & response) const +{ + /// The confirm action's whole handler. It reads no part parameter, sends no body, and touches no + /// send metric: the request asks a question about a binding, it does not transfer anything. + const auto token = decodeCasRelinkSourceToken(token_text); + if (!token) + { + /// The raw text is NOT logged: it is unvalidated peer bytes, and a decoded token is the only + /// form this server has established is free of the control characters that forge log lines. + LOG_DEBUG(log, "Relink confirm is unproven: the source token ({} bytes) is not one this server minted", + token_text.size()); + response.addCookie({CA_CONFIRM_ANSWER_COOKIE, CA_CONFIRM_ANSWER_UNPROVEN}); + return; + } + + const CasConfirmAnswer answer = resolveContentAddressedConfirm( + token->pool_uuid, token->server_root_id, token->root_namespace, + token->ref_name, token->part_name, token->manifest_ref_text); + + /// The `No`/`Unknown` distinction stays here, on the node that computed it and can name the binding + /// that produced it. It is triage information, not an authorization, and the wire carries only the + /// authorization (`CA_CONFIRM_ANSWER_UNPROVEN`). + if (answer != CasConfirmAnswer::Yes) + LOG_DEBUG(log, "Relink confirm is unproven ({}) for ref '{}' (part {}, manifest {}) in namespace '{}'", + answer == CasConfirmAnswer::No ? "no" : "unknown", + token->ref_name, token->part_name, token->manifest_ref_text, token->root_namespace); + + response.addCookie({CA_CONFIRM_ANSWER_COOKIE, + answer == CasConfirmAnswer::Yes ? CA_CONFIRM_ANSWER_PROVEN : CA_CONFIRM_ANSWER_UNPROVEN}); +} + void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuffer & out, HTTPServerResponse & response) { + /// CAS fetch-by-relink, publish-then-confirm (spec §wire-protocol): the second request of the + /// handshake, dispatched before `part` is required because a confirm carries none — the part name + /// is inside the token. Authentication parity with the fetch is inherent: the shared handler + /// authenticates before it dispatches to any endpoint. + if (const String confirm_token = params.get(CA_CONFIRM_ACTION_PARAM, ""); !confirm_token.empty()) + { + answerContentAddressedConfirm(confirm_token, response); + return; + } + // nothing to read from body body.reset(); @@ -148,7 +330,11 @@ void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuf MergeTreePartInfo::fromPartName(part_name, data.format_version); /// We pretend to work as older server version, to be sure that client will correctly process our version +<<<<<<< HEAD response.addCookie({"server_protocol_version", toString(std::min(client_protocol_version, REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS))}); +======= + response.addCookie({"server_protocol_version", toString(std::min(client_protocol_version, REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM))}); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) LOG_TRACE(log, "Sending part {}", part_name); @@ -212,6 +398,51 @@ void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuf writeBinary(projections.size(), out); } + /// CAS replication 2b — fetch-by-relink (spec §4). If the part is on a content-addressed disk and + /// the receiver advertised a `cas_pool_uuid` equal to THIS server's own pool_uuid + /// (same shared pool), send only the part's content id + the mutable header — no file bytes — so + /// the receiver can "fetch" by publishing its own ref to the blobs already in the shared pool. + /// Strictly gated on a matching pool_uuid: a non-CA part, a CA part on a different pool, or a + /// receiver without the capability all fall through to the unchanged byte path below. + /// + /// The gate is `..._WITH_CA_CONFIRM`, not `..._WITH_CA_RELINK`: a receiver is offered a relink + /// only once it advertises that it will confirm the offer before promoting it. A receiver that + /// still advertises `..._WITH_CA_RELINK` gets the bytes — mixed versions degrade to bytes, never + /// to an unconfirmed relink. This gate and the version the client advertises + /// (`fetchSelectedPart`) are one change in two places; separated in either order they either + /// disable relink outright or hand an unconfirmed relink to a receiver that claimed it confirms. + if (client_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM + && part->getDataPartStorage().isContentAddressed()) + { + const String receiver_pool_uuid = parse(params.get(CA_POOL_UUID_PARAM, "")); + DiskPtr part_disk = data.getStoragePolicy()->tryGetDiskByName(part->getDataPartStorage().getDiskName()); + auto * ca_meta = tryGetContentAddressedExchange(part_disk); + if (ca_meta && !receiver_pool_uuid.empty() && receiver_pool_uuid == ca_meta->getPoolUUID()) + { + auto offer = ca_meta->getRelinkOffer(part->getDataPartStorage().getRelativePath()); + if (offer) + { + LOG_DEBUG(log, "Sending part {} by relink (content-addressed, shared pool {}), manifest payload {} bytes", + part_name, receiver_pool_uuid, offer->manifest_bytes.size()); + response.addCookie({CA_RELINK_COOKIE, CA_RELINK_COOKIE_VALUE}); + /// The source token for the confirm request the receiver makes before it promotes + /// (spec §wire-protocol). It always accompanies the offer, and its ABSENCE is what + /// tells a confirm-capable receiver that this sender predates the handshake. + response.addCookie({CA_CONFIRM_TOKEN_COOKIE, offer->confirm_token}); + /// The relink payload (B7 part_manifest_v2, all-tree task 7): the opaque encoded + /// PartManifest body (the receiver decodes it, ignores the sender identity, and + /// stages its OWN local manifest over the shared-pool blobs; the legacy part_id wire + /// field carries it). Self-contained: uuid.txt/metadata_version.txt are ordinary + /// manifest entries now (task 6), so no separate mutable-header field is sent. + writeStringBinary(offer->manifest_bytes, out); + data.addLastSentPart(part->info); + return; + } + /// No offer (no committed ref for this part here, or no mintable token) — fall through + /// to the byte path. + } + } + if ((*data_settings)[MergeTreeSetting::allow_remote_fs_zero_copy_replication] && client_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_PARTS_ZERO_COPY) { @@ -428,7 +659,8 @@ std::pair Fetcher::fetchSelected const String & tmp_prefix_, std::optional * tagger_ptr, bool try_zero_copy, - DiskPtr disk) + DiskPtr disk, + bool allow_ca_relink) { if (blocker.isCancelled()) throw Exception(ErrorCodes::ABORTED, "Fetching of part was cancelled"); @@ -461,13 +693,55 @@ std::pair Fetcher::fetchSelected { {"endpoint", endpoint_id}, {"part", part_name}, +<<<<<<< HEAD {"client_protocol_version", toString(REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS)}, +======= + /// Advertising `..._WITH_CA_CONFIRM` is a PROMISE, not a capability list: this receiver will + /// confirm a relink offer against its source before it promotes (`relinkPartToDisk`). It is the + /// pair of the sender-side offer gate on the same constant, and the two cannot be separated — + /// see the comment there. + {"client_protocol_version", toString(REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM)}, +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) {"compress", "false"} }); if (disk) LOG_TRACE(log, "Will fetch to disk {} with type {}", disk->getName(), disk->getDataSourceDescription().toString()); + /// CAS replication 2b — fetch-by-relink (spec §4). Advertise this replica's target content-addressed + /// pool identity so a same-pool sender can relink instead of streaming bytes. The target disk is the + /// provided one if it is CA, else the first CA disk among the table's disks. A non-CA fetch adds + /// nothing here and is byte-for-byte unchanged. + /// Gated on `allow_ca_relink` alone (B66b). That flag is the RECURSION BRAKE and nothing else: not + /// advertising is what makes the sender stream bytes, so every same-sender byte re-request below + /// clears it, and a persistent relink-mechanism failure therefore costs exactly one relink attempt. + /// The gate used to be `try_zero_copy && !to_detached`, and BOTH halves were accidents of that same + /// brake — `try_zero_copy` because the fallback re-requests with it false, and `!to_detached` + /// because the relink path staged at the ACTIVE part path and ignored `to_detached`. `to_detached` + /// is now a parameter of `relinkPartToDisk` (it stages under the `detached/` parent), and + /// `try_zero_copy` goes back to meaning real zero-copy only. + String advertised_pool_uuid; + if (allow_ca_relink) + { + if (auto * ca_meta = tryGetContentAddressedExchange(disk)) + { + advertised_pool_uuid = ca_meta->getPoolUUID(); + uri.addQueryParameter(CA_POOL_UUID_PARAM, advertised_pool_uuid); + } + else if (!disk) + { + for (const auto & data_disk : data.getDisks()) + { + if (auto * ca_disk_meta = tryGetContentAddressedExchange(data_disk)) + { + advertised_pool_uuid = ca_disk_meta->getPoolUUID(); + uri.addQueryParameter(CA_POOL_UUID_PARAM, advertised_pool_uuid); + break; + } + } + } + } + Strings capability; if (try_zero_copy && (*data_settings)[MergeTreeSetting::allow_remote_fs_zero_copy_replication]) { @@ -622,6 +896,78 @@ std::pair Fetcher::fetchSelected if (server_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_PARTS_PROJECTION) readBinary(projections, *in); + /// CAS replication 2b — fetch-by-relink (spec §4; B7 part_manifest_v2, all-tree task 7). The sender + /// chose to relink: it sent only the part's encoded PartManifest body, no file bytes. Build the part + /// by staging this server's OWN local manifest over the blobs already in the shared pool (adopt-by-hash + /// -> revalidate -> promote inside adoptPartFromManifest) — self-contained since task 6 routed + /// uuid.txt/metadata_version.txt through the content path, so there is no separate mutable header to + /// reconstruct. If the relink is not possible (blob missing/condemned — a transient or a + /// genuinely-different pool the cheap pre-filter let through, or a mixed-build pair offering an + /// unrecognized cookie value), fall back to a normal byte fetch by re-requesting WITHOUT relink. + String ca_relink = parse(in->getResponseCookie(CA_RELINK_COOKIE, "")); + if (!ca_relink.empty()) + { + /// Re-request without the relink capability: pass the SAME (CA) disk but disable zero-copy/relink + /// so the sender streams bytes; on CA the downloaded files content-address and dedup. + /// + /// THE RECURSION BRAKE (B66b). `allow_ca_relink=false` is what bounds this: the re-request does + /// not advertise the pool identity, so the sender cannot offer relink again, so this lambda + /// cannot be reached a second time for the same fetch. Before relink had its own capability the + /// brake was implicit in `try_zero_copy=false`; with the two decoupled it has to be spelled out, + /// and it must be spelled out at EVERY same-sender fallback — a relink failure that is a + /// property of the pair reproduces on every attempt, so without the brake the fallback re-offers + /// and recurses without bound. The failures it actually bounds are the ones that leave the CA + /// disk resolved and matching: a mixed build offering an unrecognized cookie value, a sender that + /// predates the confirm handshake, an undecodable manifest, a local ref conflict. (The + /// reservation-outside-the-pool exit below is bounded twice over — it re-requests with the + /// non-CA disk it resolved, which cannot advertise anything either way — so do not read that one + /// as evidence that the brake is redundant.) + auto fall_back_to_byte_fetch = [&] + { + temporary_directory_lock = {}; + return fetchSelectedPart( + metadata_snapshot, context, part_name, zookeeper_name, replica_path, host, port, timeouts, + user, password, interserver_scheme, throttler, to_detached, tmp_prefix, nullptr, false, disk, + /*allow_ca_relink=*/ false); + }; + + if (ca_relink != CA_RELINK_COOKIE_VALUE) + { + /// Mixed-build cluster (rolling upgrade): this receiver build does not recognize the sender's + /// relink wire format. Bail out before reading anything else off the stream rather than + /// misparsing an incompatible payload shape. + LOG_INFO(log, "Part {} was offered by relink with cookie '{}' (this build expects '{}'); " + "falling back to a byte fetch", part_name, ca_relink, CA_RELINK_COOKIE_VALUE); + return fall_back_to_byte_fetch(); + } + + auto * chosen_ca = tryGetContentAddressedExchange(disk); + if (!chosen_ca || chosen_ca->getPoolUUID() != advertised_pool_uuid) + { + LOG_INFO(log, "Part {} was offered by relink for content-addressed pool '{}', but reservation landed " + "outside the advertised pool on disk {} (chosen pool: '{}'); falling back to a byte fetch", + part_name, advertised_pool_uuid, disk->getName(), chosen_ca ? chosen_ca->getPoolUUID() : ""); + return fall_back_to_byte_fetch(); + } + + String sender_manifest_bytes; + readStringBinary(sender_manifest_bytes, *in); + assertEOF(*in); + + /// Publish-then-confirm (spec §core-idea) happens inside `relinkPartToDisk`, including the second + /// interserver request; the token cookie is the sender's offer identity and is opaque here. A + /// `nullptr` means the mechanism cannot work but the sender still has the part, so the byte + /// re-request below is sound; a THROW means the source did not prove the binding, and the whole + /// point of it being a throw is that this fallback must NOT run for it. + auto relinked = relinkPartToDisk(part_name, tmp_prefix, disk, to_detached, sender_manifest_bytes, + in->getResponseCookie(CA_CONFIRM_TOKEN_COOKIE, ""), uri, creds, timeouts, read_settings); + if (relinked) + return std::make_pair(std::move(relinked), std::move(temporary_directory_lock)); + + LOG_INFO(log, "Relink of part {} is not possible on this pair; falling back to a byte fetch", part_name); + return fall_back_to_byte_fetch(); + } + if (!remote_fs_metadata.empty()) { if (!try_zero_copy) @@ -663,7 +1009,10 @@ std::pair Fetcher::fetchSelected temporary_directory_lock = {}; - /// Try again but without zero-copy + /// Try again but without zero-copy. `allow_ca_relink=false` for the same reason as the relink + /// branch's fallback above: this is a same-sender byte re-request, and it must not re-open a + /// capability the failed attempt is not evidence about. It also preserves the behaviour this + /// call had while relink rode on `try_zero_copy` — the flag it already passes as false. return fetchSelectedPart( metadata_snapshot, context, @@ -673,7 +1022,8 @@ std::pair Fetcher::fetchSelected host, port, timeouts, - user, password, interserver_scheme, throttler, to_detached, tmp_prefix, nullptr, false, disk); + user, password, interserver_scheme, throttler, to_detached, tmp_prefix, nullptr, false, disk, + /*allow_ca_relink=*/ false); } } @@ -990,6 +1340,312 @@ MergeTreeData::MutableDataPartPtr Fetcher::downloadPartToDisk( return new_data_part; } +/// The receiver's half of publish-then-confirm, and its complete failure taxonomy (spec +/// §failure-taxonomy). Every exit of `relinkPartToDisk` is one of these seven rows; the last two columns +/// are the questions a reviewer has to be able to answer without reading the control flow, because a +/// part-exchange path that gets them wrong either loses a part or commits it twice. +/// +/// 1. THE SOURCE SENT NO TOKEN (a peer that predates the handshake). +/// `+1`: never staged. Action: return `nullptr`, the caller byte-fetches from the same sender. +/// Lose a part? No -- the sender still has it and streams it. +/// Double-promote? No -- nothing was staged, so there is nothing to promote. +/// +/// 2. `prepareAdoptFromManifest` -> `MechanismFallbackAllowed` (manifest decode failure, or the +/// retryable staging class: body-absent precommit / precommit no longer the live owner / ref +/// conflict). +/// `+1`: NOTHING IS PUBLISHED, and that -- not "never staged" -- is what makes the byte fallback +/// sound here. A precommit whose ref-log append came back `Unresolved` may in fact be durable, so +/// `prepareEntries`' own `abandon` queues the exact removal for it (`PartWriteTxn::precommitAdd` +/// records the intent BEFORE the append precisely so that removal is never skipped) and leaves the +/// manifest body for GC rather than deleting it. A precommit is not a committed ref: a later byte +/// fetch publishes the same ref name over it without conflict, and a removal that could not be +/// appended at all leaks retained blobs -- it never double-publishes. +/// Action: return `nullptr`, the caller byte-fetches. Lose a part? No, as row 1. +/// Double-promote? No -- no handle exists, and nothing was committed. +/// +/// 3. THE CONFIRM DID NOT PROVE THE SOURCE: an `unproven` answer, an absent answer cookie, a transport +/// failure, a timeout. All one outcome, deliberately (`CasConfirmAnswer`: only `yes` authorizes). +/// `+1`: durable, then released by `abort`. Action: THROW a locally generated retry-later +/// `NETWORK_ERROR` naming the source and the part -- never `nullptr`, because a byte re-request goes +/// back to the very source whose state is in doubt. +/// Lose a part? No -- the queue stores the exception, backs off, and re-executes the entry, which +/// recomputes the source and the covering-part discovery. The fetch is postponed, not dropped. +/// Double-promote? No -- `abort` appends the exact precommit removal and no committed ref exists. +/// +/// 4. CONFIRM `yes`, `promote` -> `Committed`. +/// `+1`: committed. Action: return the relinked part; the usual `tmp-fetch_` re-key follows. +/// Lose a part? No. Double-promote? No -- `promote` is the handle's single terminal operation, the +/// handle is released immediately after it, and a second call is rejected rather than re-driving a +/// finished transaction. +/// +/// 5. CONFIRM `yes`, `promote` -> `MechanismFallbackAllowed` (a local ref conflict; the source proved +/// its side, this receiver could not commit its own). The promote was rejected BEFORE its ref-log +/// append, so "nothing was committed" is proven, not assumed -- see row 5b for the case where it is +/// not. +/// `+1`: released -- a failed `promote` abandons its build on the way out. +/// Action: return `nullptr`, the caller byte-fetches. Lose a part? No, as row 1. +/// Double-promote? No -- the byte fetch starts from a clean slate. +/// +/// 5b. CONFIRM `yes`, `promote` -> `Unresolved` (the promotion's ref-log append was attempted and came +/// back without a verdict; the receiver's ref MAY be committed). +/// `+1`: still owed -- the handle attempts its abandon, which is REJECTED by the state machine if +/// the promote in fact landed (a promoted binding is no longer a precommit), so no committed ref is +/// ever undone here. +/// Action: THROW the retry-later `NETWORK_ERROR`, as row 3 -- returning `nullptr` is the one thing +/// that must not happen, because a byte fetch would publish the part a SECOND time over a relink +/// that may already be committed. +/// Lose a part? No -- retry-later, as row 3. Double-promote? No -- nothing is published on this exit. +/// +/// 6. ANY OTHER EXCEPTION (an unclassified local error, or a `promote` failure outside the known +/// retryable class). +/// `+1`: durable if one was staged, then released by the scope guard. Action: propagate. +/// Lose a part? No -- retry-later, exactly as row 3. Double-promote? No -- the scope guard runs +/// `abort` before the exception leaves the function, and the handle's own destructor is the backstop +/// if that abort's append fails. +/// +/// The asymmetry between rows 2/5 and row 3 is the entire point of the typed boundary. A byte +/// re-request goes back to the SAME sender, so it is a sound recovery exactly when the doubt is about +/// the MECHANISM and the sender is known to still hold the part -- and never when the doubt is about +/// the source itself. `adoptPartFromManifest` used to collapse the two by catching every `Exception` +/// and returning `false`. +/// +/// B66b — WHAT CHANGES WHEN THE TARGET IS `detached/`. Every row above still holds, and the two columns +/// that matter are unchanged in every one of them, but two rows hold for a DIFFERENT reason and that +/// difference is worth stating rather than rediscovering: +/// +/// - Row 3 (and row 6, which recovers the same way) argues "no part is lost" from the replication queue: +/// it stores the exception, backs off, and re-executes the entry. Two of the three detached callers +/// have no queue entry -- `FETCH PARTITION`/`FETCH PART ... FROM` are user DDL -- so the retry-later +/// error surfaces to the user, who re-issues the statement. Nothing is lost either way, and for a +/// stronger reason than in the active case: a detached fetch is not replication, so no replicated +/// state was ever expecting the part. (The third, `executeClonePartFromShard`, IS a queue entry and +/// recovers exactly as the active path does.) +/// - Row 4's "no double-promote" is about the relink's own terminal operation and is unaffected. What +/// the CALLER then does with the part differs: `renameTo(detached/, true)` rather than +/// `renameTempPartAndReplace`. Both are ref repoints within one namespace on a content-addressed disk +/// (`detached/` is a ref-name prefix, not a namespace), and the detached one keeps its existing +/// collision behaviour -- an existing `detached/` is displaced. That is the pre-existing +/// semantic of a detached BYTE fetch, deliberately left alone: relink must not change what a fetch +/// into `detached/` means, only how the bytes get there. +/// +/// The staged ref itself is `detached/tmp-fetch_` rather than `tmp-fetch_`, which is what +/// keeps a failed detached relink from ever being visible as a live part: the abandoned precommit and +/// the abandoned staging directory both live in the detached ref space. +/// +/// What a `yes` does NOT prove: `CaRelinkConfirmCore.tla` config `_sab_holeylist` shows that with every +/// confirm rule intact and one incomplete listing page permitted, `ConfirmedRelinkNeverDangles` still +/// breaks (BACKLOG `{#list-as-journal-dataloss-2026-07-25}`). A confirmed relink is therefore NOT proven +/// dangle-free; a `yes` means only "the source still holds exactly this manifest right now", which is +/// what closes the codex-6 handoff window and nothing more. +MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( + const String & part_name, + const String & tmp_prefix, + DiskPtr disk, + bool to_detached, + const String & sender_manifest_bytes, + const String & source_token, + const Poco::URI & fetch_uri, + const Poco::Net::HTTPBasicCredentials & credentials, + const ConnectionTimeouts & timeouts, + const ReadSettings & read_settings) +{ + auto * ca_meta = tryGetContentAddressedExchange(disk); + if (!ca_meta) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "relinkPartToDisk called for a non-content-addressed disk {}", disk->getName()); + + if (tmp_prefix.empty() + || part_name.empty() + || std::string::npos != tmp_prefix.find_first_of("/.") + || std::string::npos != part_name.find_first_of("/.")) + throw Exception(ErrorCodes::LOGICAL_ERROR, "`tmp_prefix` and `part_name` cannot be empty or contain '.' or '/' characters."); + + /// Taxonomy row 1 — the capability gate, and it comes FIRST so a pre-confirm sender costs nothing. + /// An absent token is how such a sender identifies itself: it offers relink exactly as before and + /// simply has no token cookie to attach. There is no version number to consult here and that is + /// deliberate — the sender's advertised version says what it can serve, while the token's presence + /// says what it actually did for THIS offer, and only the latter can be confirmed. A relink that + /// cannot be confirmed is never promoted, so the bytes are fetched instead. + if (source_token.empty()) + { + LOG_INFO(log, "Part {} was offered by relink without a source token, so the offer cannot be confirmed " + "(the sender predates the publish-then-confirm handshake); falling back to a byte fetch", part_name); + return nullptr; + } + + /// Test-only. Forces the "mechanism failed, the sender still has the part" exit (the ACTION of + /// taxonomy rows 2 and 5) on EVERY attempt, which is precisely the shape the recursion brake has to + /// bound: a persistent property of this sender/receiver pair, so the byte re-request re-offers and + /// re-fails unless it clears `allow_ca_relink`. It fires AFTER the token gate and BEFORE + /// `prepareAdoptFromManifest`, so nothing is staged and no `+1` has to be released — the failpoint + /// injects the exit, never a half-finished transaction. + fiu_do_on(FailPoints::cas_relink_receiver_force_mechanism_failure, + { + LOG_INFO(log, "Failpoint cas_relink_receiver_force_mechanism_failure: abandoning the relink of part {} " + "before anything is staged", part_name); + return nullptr; + }); + + /// Stage under the tmp-fetch dir OF THE TARGET PARENT — the table dir, or `TABLE/detached` when + /// the caller asked for a detached fetch (B66b). The parent is composed exactly as + /// `downloadPartToDisk` composes it, so the two fetch paths put a part in the same place and the + /// caller's finalization is unchanged: `renameTempPartAndReplace`'s moveDirectory(tmp-fetch_ + /// -> ) for the active path, `renameTo(detached/)` for the detached one. Both are ref + /// repoints within one namespace on a content-addressed disk (`detached/` is a ref-name prefix, not + /// a namespace), so a relinked part re-keys exactly as a byte-fetched one does. + /// + /// The ref name is NOT built here: the disk-relative path is handed to the CA exchange whole and its + /// router folds `TABLE/detached/DIR` onto the `detached/DIR` ref, the same routing every other + /// read and write of a detached part goes through. This side has no business knowing that prefix, + /// and the sender's half of the offer (`getRelinkOffer`) is already addressed by path too. + const String part_dir = tmp_prefix + part_name; + const String part_relative_path + = data.getRelativeDataPath() + String(to_detached ? MergeTreeData::DETACHED_DIR_NAME : ""); + const String part_path = fs::path(part_relative_path) / part_dir; + + LOG_DEBUG(log, "Relinking part {} (staged as {}) onto content-addressed disk {} from a {}-byte transferred manifest.", + part_name, part_path, disk->getName(), sender_manifest_bytes.size()); + + /// T1 — PUBLISH. Adopt-from-manifest and precommit, stopping short of the promote (B7 + /// part_manifest_v2, all-tree task 7): the receiver decodes the transferred body and stages its OWN + /// local manifest over the shared-pool blobs (adopt-by-hash). Self-contained: + /// uuid.txt/metadata_version.txt are ordinary entries in the transferred manifest (task 6), so there + /// is no sidecar to reconstruct. Trust boundary is the interserver channel, as for a normal part + /// fetch — see `prepareAdoptFromManifest`. + /// + /// The order is the whole protocol. This `+1` must be DURABLE before the source is asked anything, + /// because the question "do you still hold it?" only excludes a later removal if the receiver's own + /// reference is already in the ref log when that removal is appended (spec §correctness). Asking + /// first and publishing after would prove nothing about the interval in between. What it does NOT + /// establish is that every subsequent GC fold OBSERVES that reference -- see "What a `yes` does NOT + /// prove" above; ordering is necessary here, not sufficient. + std::unique_ptr prepared; + if (ca_meta->prepareAdoptFromManifest(part_path, sender_manifest_bytes, prepared) + == CaRelinkPrepare::MechanismFallbackAllowed) + return nullptr; /// taxonomy row 2 + if (!prepared) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "Relink of part {} reported a prepared write but produced no handle", part_name); + + /// Belt-and-braces over the handle's own destructor: the durable `+1` is released on EVERY exit that + /// is not a completed promote, including an exception. `abort` is the non-throwing form by contract — + /// this runs while the retry-later error of row 3 is already in flight. + SCOPE_EXIT({ + if (prepared) + prepared->abort(); + }); + + /// Test-only, and this is the ONE seam worth injecting on the whole path: it opens the window the + /// protocol exists to make safe. The receiver's `+1` is durable and its release is armed, and the + /// source has not been asked anything yet, so a test that holds the fetch here can do to the source + /// exactly what codex-6 described — merge the part away, run GC to fixpoint — and then observe both + /// halves of the contract: the source's blobs survive the round (this receiver's binding protects + /// them) and the confirm that follows refuses to authorize a promote (the binding it named is gone). + FailPointInjection::pauseFailPoint(FailPoints::cas_relink_receiver_pause_before_confirm); + + /// T2 — CONFIRM. One read-only interserver question, aimed at the endpoint copied out of the fetch + /// URI so it reaches exactly the table and replica that made the offer. Only the literal + /// `CA_CONFIRM_ANSWER_PROVEN` cookie authorizes a promote: an `unproven` answer, an absent cookie + /// (any peer that does not implement the action) and a failed request are ONE outcome, and it is not + /// knowledge about the source — see `CasConfirmAnswer` on why `no` is never put on the wire. + bool source_proved_the_binding = false; + try + { + Poco::URI confirm_uri; + confirm_uri.setScheme(fetch_uri.getScheme()); + confirm_uri.setHost(fetch_uri.getHost()); + confirm_uri.setPort(fetch_uri.getPort()); + Poco::URI::QueryParameters confirm_params; + for (const auto & fetch_param : fetch_uri.getQueryParameters()) + if (fetch_param.first == "endpoint") + confirm_params.push_back(fetch_param); + confirm_params.emplace_back(CA_CONFIRM_ACTION_PARAM, source_token); + confirm_params.emplace_back("compress", "false"); + confirm_uri.setQueryParameters(confirm_params); + + /// `read_settings` is the caller's, which already caps HTTP retries at one: the queue owns the + /// retry policy for a fetch, and a silently retried confirm would widen the window it measures. + auto confirm_in = BuilderRWBufferFromHTTP(confirm_uri) + .withConnectionGroup(HTTPConnectionGroupType::HTTP) + .withBypassProxy(true) + .withMethod(Poco::Net::HTTPRequest::HTTP_POST) + .withTimeouts(timeouts) + .withSettings(read_settings) + .withDelayInit(false) + .create(credentials); + /// The confirm answer is a cookie and the response body is empty by construction. Requiring EOF + /// before reading the answer means a response carrying anything at all — a misrouted reply, a + /// desynchronized peer — is unproven rather than half-parsed. + assertEOF(*confirm_in); + source_proved_the_binding + = confirm_in->getResponseCookie(CA_CONFIRM_ANSWER_COOKIE, "") == CA_CONFIRM_ANSWER_PROVEN; + } + catch (...) + { + /// Not a fallback: a confirm that could not be delivered is the same "not proven" as a refusal, + /// and it takes the same path out. Logged rather than propagated so the error the caller sees is + /// the one that names the relink — but logged in full, because the reason (refused, timed out, + /// 500) exists nowhere else. `information`, not `error`: a peer restarting mid-fetch is ordinary, + /// and the throw below is what makes the failure loud. + tryLogCurrentException(log, fmt::format("while confirming the relink offer for part {} with {}", + part_name, fetch_uri.getHost()), LogsLevel::information); + source_proved_the_binding = false; + } + + if (!source_proved_the_binding) + { + /// Taxonomy row 3. Locally generated on purpose — nothing here is the source's error to report — + /// and thrown rather than returned, because the one recovery that is NOT sound after this is a + /// byte re-request to the same source. `NETWORK_ERROR` puts it in the retry-later class, so the + /// queue stores it, backs off, and re-selects on re-execution. + throw Exception(ErrorCodes::NETWORK_ERROR, + "Source {} did not prove it still holds the manifest it offered for part {} by relink; " + "the relink is abandoned and the fetch will be retried later", + fetch_uri.getHost(), part_name); + } + + /// T3 — PROMOTE. Only now, and only because the source proved the binding at T2 > T1. + switch (prepared->promote()) + { + case CaRelinkPromote::Committed: + break; + case CaRelinkPromote::MechanismFallbackAllowed: + return nullptr; /// taxonomy row 5 + case CaRelinkPromote::Unresolved: + /// The promotion append may have landed, so this is the ONE promote outcome that is not row + /// 5: returning `nullptr` would send the caller to fetch the bytes and publish the part a + /// second time over a relink that may already be committed. Thrown in the retry-later class + /// instead, exactly as an unproven confirm is (row 3) -- the queue stores it, backs off, and + /// re-executes, by which time the ref lane has resolved the ambiguity one way or the other. + throw Exception(ErrorCodes::NETWORK_ERROR, + "Relink of part {} from {} could not be resolved: the promotion may or may not have " + "committed, so the bytes must NOT be fetched; the fetch will be retried later", + part_name, fetch_uri.getHost()); + } + /// The single terminal operation is done, so the handle owes nothing; releasing it here also disarms + /// the scope guard for the part-building code below. + prepared.reset(); + + auto volume = std::make_shared("volume_" + part_name, disk); + + MergeTreeData::MutableDataPartPtr new_data_part; + MergeTreeDataPartBuilder builder(data, part_name, volume, part_relative_path, part_dir, getReadSettings()); + /// Read the part format from the now-published manifest (type + storage type), exactly as the byte + /// fetch does — authoritative over the transferred `part_type` header (kept for protocol symmetry). + new_data_part = builder.withPartFormatFromDisk().build(); + + new_data_part->version->setAndStoreCreationTID(Tx::NonTransactionalTID, nullptr); + new_data_part->is_temp = true; + /// The blobs are shared in the pool; a discarded temporary relink part must NOT reclaim them (another + /// replica's ref keeps them alive). Same policy a zero-copy-fetched temporary part uses. + new_data_part->remove_tmp_policy = IMergeTreeDataPart::BlobsRemovalPolicyForTemporaryParts::PRESERVE_BLOBS; + new_data_part->modification_time = time(nullptr); + new_data_part->loadColumnsChecksumsIndexes(true, false); + + LOG_DEBUG(log, "Relink of part {} onto disk {} finished (no bytes transferred).", part_name, disk->getName()); + return new_data_part; +} + } } diff --git a/src/Storages/MergeTree/DataPartsExchange.h b/src/Storages/MergeTree/DataPartsExchange.h index 6e79d67d5714..5bec506eac21 100644 --- a/src/Storages/MergeTree/DataPartsExchange.h +++ b/src/Storages/MergeTree/DataPartsExchange.h @@ -18,11 +18,22 @@ namespace zkutil using ZooKeeperPtr = std::shared_ptr; } +/// Only the content-addressed relink's confirm request needs these, and only as parameter types, so +/// they are declared rather than included — this header is pulled in by the whole replication tree. +namespace Poco { class URI; } +namespace Poco::Net { class HTTPBasicCredentials; } + namespace DB { class StorageReplicatedMergeTree; class ReadWriteBufferFromHTTP; +struct ReadSettings; + +/// Declared by `ContentAddressedExchange.h` (the narrow content-addressed seam). Opaque-enum-declared +/// here so this header stays free of content-addressed includes; the definition must keep the same +/// underlying type. +enum class CasConfirmAnswer : uint8_t; namespace DataPartsExchange { @@ -41,6 +52,27 @@ class Service final : public InterserverIOEndpoint void processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuffer & out, HTTPServerResponse & response) override; private: + /// CAS fetch-by-relink, publish-then-confirm: answer one relink confirm token — "is `manifest_ref_text` + /// still exactly what `ref_name` names here?" — for a receiver that has already made its own `+1` + /// durable and may promote only on `Yes`. Everything content-addressed is behind + /// `IContentAddressedExchange`; what has to live here is what only the storage can see: which of this + /// table's disks is entitled to answer (`ownsNamespace` under a matching pool UUID, exactly one match + /// or `Unknown`), and gate 0, the part-anchored filter over this table's parts set. Never throws, and + /// `No` is not knowledge — see `CasConfirmAnswer`. + /// The confirm action's handler: decode the peer's token, resolve it, and set the answer cookie. + /// Exactly two answers cross the wire — proven, and not proven — because only `Yes` authorizes + /// anything and `No` is not knowledge (see `CasConfirmAnswer`). Never throws: an unparsable token + /// is one more unproven answer, not an error the receiver would have to classify. + void answerContentAddressedConfirm(const String & token_text, HTTPServerResponse & response) const; + + CasConfirmAnswer resolveContentAddressedConfirm( + const String & pool_uuid, + const String & server_root_id, + const String & root_namespace, + const String & ref_name, + const String & part_name, + const String & manifest_ref_text) const; + MergeTreeData::DataPartPtr findPart(const String & name); MergeTreeData::DataPart::Checksums sendPartFromDisk( @@ -81,7 +113,20 @@ class Fetcher final : private boost::noncopyable const String & tmp_prefix_ = "", std::optional * tagger_ptr = nullptr, bool try_zero_copy = true, - DiskPtr dest_disk = nullptr); + DiskPtr dest_disk = nullptr, + /// CAS fetch-by-relink (spec §B66b): may this request advertise its content-addressed pool + /// identity, i.e. may the sender answer with a relink offer instead of the part's bytes? + /// + /// It is a capability of its own rather than a rider on `try_zero_copy`, and it carries the + /// RECURSION BRAKE. Relink used to be gated on `try_zero_copy` purely because the byte-fetch + /// fallback re-requests with `try_zero_copy=false`, so the brake came for free; with the two + /// decoupled, every same-sender byte re-request must clear THIS flag explicitly or a + /// persistent relink-mechanism failure re-offers, re-fails and re-requests without bound. + /// + /// It defaults to `true`, and that default is what makes a manual `FETCH PARTITION`/`FETCH + /// PART` (which passes `try_fetch_shared=false`, so `try_zero_copy` is already false) relink, + /// into `detached/` as well as into the active part path. + bool allow_ca_relink = true); /// You need to stop the data transfer. ActionBlocker blocker; @@ -111,6 +156,36 @@ class Fetcher final : private boost::noncopyable ThrottlerPtr throttler, bool sync); + /// CAS replication 2b — fetch-by-relink (spec §4), publish-then-confirm (spec §core-idea). Build a + /// part WITHOUT downloading any bytes by publishing this server's own ref to the blobs already in the + /// shared content-addressed pool. Stages the ref under the tmp-fetch dir of the target parent — the + /// table dir, or `detached/` when `to_detached` (B66b) — so the caller's finalization re-keys it to + /// the final part name, exactly as for a byte-fetched part: `renameTempPartAndReplace` for the + /// active path, `renameTo(detached/)` for the detached one. Then it ASKS THE SOURCE whether it + /// still holds exactly the manifest it offered, and only then promotes and loads the part. + /// Self-contained (all-tree task 7): the transferred manifest alone is enough to rebuild the part — + /// no separate uuid/metadata_version wire fields to reconstruct as a sidecar. + /// + /// The whole failure taxonomy lives at the definition; the two outcomes a CALLER must distinguish: + /// `nullptr` means relink cannot work here and the source still has the part, so a byte re-request to + /// the SAME source is sound; a THROW means the source could not prove it still holds the manifest, + /// and the one recovery that is not sound is asking that same source for the bytes. + /// + /// `source_token`, `fetch_uri` and the connection parameters are what the confirm request is built + /// from: the token is the sender's opaque offer identity, and the request is aimed at the endpoint + /// COPIED out of the fetch URI so it cannot reach a different table or replica than the offer did. + MergeTreeData::MutableDataPartPtr relinkPartToDisk( + const String & part_name, + const String & tmp_prefix, + DiskPtr disk, + bool to_detached, + const String & sender_manifest_bytes, + const String & source_token, + const Poco::URI & fetch_uri, + const Poco::Net::HTTPBasicCredentials & credentials, + const ConnectionTimeouts & timeouts, + const ReadSettings & read_settings); + MergeTreeData::MutableDataPartPtr downloadPartToDiskRemoteMeta( const String & part_name, const String & replica_path, diff --git a/src/Storages/MergeTree/IDataPartStorage.h b/src/Storages/MergeTree/IDataPartStorage.h index 0a6b26737fe0..e338c62551f0 100644 --- a/src/Storages/MergeTree/IDataPartStorage.h +++ b/src/Storages/MergeTree/IDataPartStorage.h @@ -199,6 +199,15 @@ class IDataPartStorage : public boost::noncopyable virtual std::string getDiskName() const = 0; virtual std::string getDiskType() const = 0; virtual bool isStoredOnRemoteDisk() const { return false; } + /// True when the underlying disk stores a part as one atomic content-addressed unit (one manifest + /// + one ref). On such disks a projection sub-part must be written through the PARENT part's + /// whole-part transaction rather than its own sub-transaction (otherwise the projection is lost + /// from the committed manifest — B58). + virtual bool isContentAddressed() const { return false; } + /// True when the underlying disk publishes a file write atomically in one shot (no partial + /// content ever becomes visible under the file's final name). Such disks do not need the + /// tmp-file + `replaceFile` crash-safety dance that plain local writes require. + virtual bool supportsAtomicFileWrites() const { return false; } virtual std::optional getCacheName() const { return std::nullopt; } virtual bool supportZeroCopyReplication() const { return false; } virtual bool supportParallelWrite() const = 0; diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index dd0357ed1503..6bbe96cb79a6 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -1614,15 +1614,25 @@ MergeTreeDataPartBuilder IMergeTreeDataPart::getProjectionPartBuilder( const String & projection_name, ProjectionDescriptionRawPtr projection, PartDirIntent intent, bool is_temp_projection) { const char * projection_extension = is_temp_projection ? ".tmp_proj" : ".proj"; + /// On a content-addressed disk a part is one atomic unit, so a temp projection sub-part (written + /// during a merge/mutate rebuild under `.tmp_proj`) must share the PARENT part's whole-part + /// transaction -- its files are re-keyed into the parent manifest when `.tmp_proj` is renamed + /// to `.proj`. On any other disk a temp projection keeps its own sub-transaction, as before. + const bool use_parent_transaction = !is_temp_projection || getDataPartStorage().isContentAddressed(); + /// The projection storage is stored on the resulting projection part for its lifetime, so create /// it in the dedicated arena (this is the part-lifetime projection-storage creation site). /// `CreateFresh` takes the non-initializing variant, so nothing is seeded from a nested leftover. MutableDataPartStoragePtr projection_storage; { ScopedJemallocThreadArena mergetree_arena_scope(JemallocMergeTreeArena::getArenaIndex()); +<<<<<<< HEAD projection_storage = intent == PartDirIntent::CreateFresh ? getDataPartStorage().getProjectionNoInitialize(projection_name + projection_extension, !is_temp_projection) : getDataPartStorage().getProjection(projection_name + projection_extension, !is_temp_projection); +======= + projection_storage = getDataPartStorage().getProjection(projection_name + projection_extension, use_parent_transaction); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } if (intent == PartDirIntent::CreateFresh && projection_storage->exists()) { diff --git a/src/Storages/MergeTree/MergeProjectionPartsTask.cpp b/src/Storages/MergeTree/MergeProjectionPartsTask.cpp index c68c8cbde3e8..12c67e6c0c43 100644 --- a/src/Storages/MergeTree/MergeProjectionPartsTask.cpp +++ b/src/Storages/MergeTree/MergeProjectionPartsTask.cpp @@ -128,6 +128,10 @@ bool MergeProjectionPartsTask::executeStep() /// FIXME (alesapin) we should use some temporary storage for this, /// not commit each subprojection part + /// + /// A borrowed (CA) recursively-merged projection sub-part shares the parent part's whole-part + /// transaction (the nested MergeTask skipped its own begin), so it is committed by the parent's + /// single commit; the storage makes commitTransaction a no-op there, so this is unconditional (B58). next_level_parts.back()->getDataPartStorage().commitTransaction(); next_level_parts.back()->is_temp = true; next_level_parts.back()->temp_projection_block_number = block_num; diff --git a/src/Storages/MergeTree/MergeTask.cpp b/src/Storages/MergeTree/MergeTask.cpp index 74f4d50043e7..90ed28f6e841 100644 --- a/src/Storages/MergeTree/MergeTask.cpp +++ b/src/Storages/MergeTree/MergeTask.cpp @@ -594,9 +594,21 @@ bool MergeTask::ExecuteAndFinalizeHorizontalPart::prepare() const std::optional builder; if (global_ctx->parent_part) { +<<<<<<< HEAD /// Non-initializing, so nothing is seeded from an existing directory. auto data_part_storage = global_ctx->parent_part->getDataPartStorage().getProjectionNoInitialize(local_tmp_part_basename, /* use parent transaction */ false); builder.emplace(*global_ctx->data, global_ctx->future_part->name, data_part_storage, getReadSettings(), PartDirIntent::CreateFresh); +======= + /// On a content-addressed disk a part is one atomic unit (one manifest + one ref), so the + /// projection sub-part must be written through the PARENT part's whole-part transaction -- + /// mirroring the INSERT path -- for its files to land in the parent manifest and survive a + /// reload. On any other disk the projection keeps its own sub-transaction, as before. + global_ctx->projection_uses_parent_transaction + = global_ctx->parent_part->getDataPartStorage().isContentAddressed(); + auto data_part_storage = global_ctx->parent_part->getDataPartStorage().getProjection( + local_tmp_part_basename, /* use_parent_transaction */ global_ctx->projection_uses_parent_transaction); + builder.emplace(*global_ctx->data, global_ctx->future_part->name, data_part_storage, getReadSettings()); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) builder->withParentPart(global_ctx->parent_part); } else @@ -618,6 +630,8 @@ bool MergeTask::ExecuteAndFinalizeHorizontalPart::prepare() const if (global_ctx->parent_part && data_part_storage->exists()) throw Exception(ErrorCodes::LOGICAL_ERROR, "Projection merge directory {} already exists", data_part_storage->getFullPath()); + /// A borrowed projection sub-part shares the parent's already-open transaction; the storage makes + /// beginTransaction a no-op in that case, so this can be called unconditionally. data_part_storage->beginTransaction(); global_ctx->storage_snapshot = std::make_shared(*global_ctx->data, global_ctx->metadata_snapshot); @@ -1532,6 +1546,9 @@ void MergeTask::ExecuteAndFinalizeHorizontalPart::calculateProjectionForBlock( *global_ctx->data, result, projection, global_ctx->new_data_part.get(), ++ctx->projection_block_num, global_ctx->context); tmp_part->finalize(); + /// A borrowed (CA) temp projection sub-part rides the parent's whole-part transaction and is + /// committed by the parent's single commit; the storage makes commitTransaction a no-op there, + /// so this can be called unconditionally (B58). tmp_part->part->getDataPartStorage().commitTransaction(); ctx->projection_parts[projection.name].emplace_back(std::move(tmp_part->part)); } @@ -1574,6 +1591,8 @@ void MergeTask::ExecuteAndFinalizeHorizontalPart::finalizeProjections() const *global_ctx->data, result, projection, global_ctx->new_data_part.get(), ++ctx->projection_block_num, global_ctx->context); temp_part->finalize(); + /// See the matching note above: a borrowed (CA) temp projection sub-part rides the parent + /// transaction, so commitTransaction is a no-op and can be called unconditionally. temp_part->part->getDataPartStorage().commitTransaction(); ctx->projection_parts[projection.name].emplace_back(std::move(temp_part->part)); } diff --git a/src/Storages/MergeTree/MergeTask.h b/src/Storages/MergeTree/MergeTask.h index 0ffa0f7fcd60..1212e3f785cd 100644 --- a/src/Storages/MergeTree/MergeTask.h +++ b/src/Storages/MergeTree/MergeTask.h @@ -216,6 +216,10 @@ class MergeTask ProjectionDescriptionRawPtr projection{nullptr}; /// This will be either nullptr or new_data_part, so raw pointer is ok. IMergeTreeDataPart * parent_part{nullptr}; + /// True only when this MergeTask builds a projection sub-part (`parent_part != nullptr`) whose + /// parent lives on a content-addressed disk: the sub-part then shares the parent's whole-part + /// transaction and must NOT begin/commit its own (B58). False for non-CA disks and top-level parts. + bool projection_uses_parent_transaction{false}; MergedPartOffsetsPtr merged_part_offsets; ContextPtr context{nullptr}; time_t time_of_merge{0}; diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index df436b80aa27..27a68c831acf 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -51,9 +51,11 @@ #include #include #include +#include #include #include #include +#include #include #include #include @@ -7032,6 +7034,32 @@ MergeTreeData::PartsToRemoveFromZooKeeper MergeTreeData::removePartsInRangeFromW new_data_part->getDataPartStorage().commitTransaction(); rollback_tx_guard.reset(); +<<<<<<< HEAD +======= + /// On a content-addressed disk a part directory becomes durable only when its disk-storage + /// transaction is committed (the ref to its manifest is published at commit, not at rename). + /// The flow below rolls back the in-memory MergeTreeData transaction (to keep the empty part + /// Outdated, not Active), which never calls commitTransaction on the disk storage — so on a CA + /// disk the empty covering part would leave NO on-disk ref and vanish on restart/reattach, + /// defeating its sole purpose (it exists only to cover the dropped parts on disk so a restart + /// does not treat them as uncovered unexpected parts and trip TOO_MANY_UNEXPECTED_DATA_PARTS). + /// On a plain disk the rename in renameTempPartAndAdd is already durable, so this is a no-op + /// there. Commit the disk storage transaction here (CA only) so the ref is published before the + /// in-memory rollback; the part still ends up Outdated, exactly as on a plain disk. + /// + /// [TXN-ONE-PIPELINE] (`2026-07-16-cas-txn-one-pipeline-design.md`, Audit 7 / Tension 2): this + /// hand-placed `commitTransaction()` is NOT made redundant by moving publication into `commit` + /// — it is the direct consequence of that design. There is no `precommit` phase under the + /// one-pipeline model, and this rollback path (by construction, to keep the part Outdated) never + /// reaches `MergeTreeData::Transaction::commit`, the only other place a disk transaction is + /// committed. So this call remains the ONLY thing that publishes the empty cover's ref. Keep it. + if (new_data_part->getDataPartStorage().isContentAddressed() + && new_data_part->getDataPartStorage().hasActiveTransaction()) + new_data_part->getDataPartStorage().commitTransaction(); + + /// It will add the empty part to the set of Outdated parts without making it Active (exactly what we need) + transaction.rollback(&lock); +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) new_data_part->remove_time.store(0, std::memory_order_relaxed); /// Such parts are always local, they don't participate in replication, they don't have shared blobs. /// So we don't have locks for shared data in zk for them, and can just remove blobs (this avoids leaving garbage in S3) @@ -8114,6 +8142,56 @@ void MergeTreeData::checkAlterPartitionIsPossible( can_execute_alter_on_disk = std::ranges::contains(supported_commands, command.type); break; } + case MetadataStorageType::CAS: + { + /// On a CAS disk whole-part publication is transactional. Same-disk clones use + /// `DataPartStorageOnDiskBase::freeze`; cross-disk clones use + /// `DataPartStorageOnDiskBase::freezeRemote`, which streams source bytes into ONE CA + /// transaction. When both disks share a pool, the publish dedup-resolves existing blobs + /// and becomes a ref repoint, although the sequential source read still happens. + /// `moveDirectory` re-keys the detached-staging → active rename into a complete active + /// ref — so these are SUPPORTED and verified (read back identical data, survive restart): + /// `ATTACH PARTITION`/`ATTACH PART` (re-clone of the table's own + /// detached parts), `REPLACE PARTITION`/`ATTACH PARTITION ... FROM` (parses to + /// `REPLACE_PARTITION`), and `MOVE PARTITION ... TO TABLE`. The pointer-unlink commands + /// `DROP PARTITION` / `DETACH PARTITION` / `DROP DETACHED PARTITION` are also fine. + /// `FETCH PARTITION`/`FETCH PART` is also SUPPORTED — it is a `ReplicatedMergeTree` op + /// (now supported on CA), and a `to_detached` fetch takes the byte-fetch path: the + /// downloaded files content-address into the `detached/` namespace (relink-into-detached + /// is deferred, see backlog). `ALTER ... FETCH PART` parses to the same `FETCH_PARTITION` + /// command type (with `part=true`), so this entry covers both. + /// `FREEZE PARTITION`/`FREEZE ALL` and `UNFREEZE PARTITION`/`UNFREEZE ALL` are now SUPPORTED: + /// a freeze publishes each part as its own ref in the `shadow/` namespace (a GC root sharing + /// the live blobs zero-copy — no byte copy); UNFREEZE removes the backup's refs. + /// `FORGET PARTITION` is SUPPORTED on CA — it only manipulates ZooKeeper partition metadata + /// (removes block-number nodes from ZooKeeper) and does not write, clone, or touch any part + /// files on disk, so it is safe on a content-addressed disk. + /// NOTE: `MOVE_PARTITION` also admits cross-disk + /// `MOVE ... TO DISK/VOLUME` (this check cannot distinguish the destination); that uses + /// the byte-copy `clonePart` path (NOT the corrupting per-file hardlink), but only + /// same-disk `MOVE ... TO TABLE` is verified here — cross-disk is a follow-up to verify. + const static auto supported_commands = { + PartitionCommand::DROP_PARTITION, + PartitionCommand::DROP_DETACHED_PARTITION, + PartitionCommand::FORGET_PARTITION, + PartitionCommand::ATTACH_PARTITION, + PartitionCommand::REPLACE_PARTITION, + PartitionCommand::MOVE_PARTITION, + PartitionCommand::FETCH_PARTITION, + PartitionCommand::FREEZE_PARTITION, + PartitionCommand::FREEZE_ALL_PARTITIONS, + PartitionCommand::UNFREEZE_PARTITION, + PartitionCommand::UNFREEZE_ALL_PARTITIONS, + }; + + if (!std::ranges::contains(supported_commands, command.type)) + throw Exception( + ErrorCodes::SUPPORT_IS_DISABLED, + "Partition operation ALTER TABLE {} is not supported on a CAS disk yet " + "(it clones parts file-by-file with no transaction, which would corrupt the clone); disk '{}'", + command.typeToString(), disk->getName()); + break; + } case MetadataStorageType::StaticWeb: case MetadataStorageType::WebIndex: { @@ -9155,6 +9233,14 @@ void MergeTreeData::restorePartFromBackup(std::shared_ptr r /// Copy files from the backup to the directory `tmp_part_dir`. disk->createDirectories(temp_part_dir); + /// A content-addressed disk publishes a part as ONE manifest (N files -> one ref) atomically, so the + /// per-file copyFileToDisk autocommit below is rejected for content part files. Route the restore + /// through one whole-part transaction (mirrors DataPartStorageOnDiskBase::freeze's owned_transaction): + /// all files land in a single content-addressed part at tmp_restore_, published by tx->commit(). + DiskTransactionPtr restore_tx; + if (disk->isContentAddressed()) + restore_tx = disk->createTransaction(); + for (const String & filename : filenames) { /// Needs to create subdirectories before copying the files. Subdirectories are used to represent projections. @@ -9178,10 +9264,29 @@ void MergeTreeData::restorePartFromBackup(std::shared_ptr r continue; } +<<<<<<< HEAD size_t file_size = backup->copyFileToDisk(part_path_in_backup_fs / filename, disk, temp_part_dir / filename, WriteMode::Rewrite, fsync_files); reservation->update(reservation->getSize() - file_size); +======= + if (restore_tx) + { + auto in = backup->readFile(part_path_in_backup_fs / filename); + auto out = restore_tx->writeFile(temp_part_dir / filename, DBMS_DEFAULT_BUFFER_SIZE, WriteMode::Rewrite, getContext()->getWriteSettings()); + copyData(*in, *out); + out->finalize(); + reservation->update(reservation->getSize() - backup->getFileSize(part_path_in_backup_fs / filename)); + } + else + { + size_t file_size = backup->copyFileToDisk(part_path_in_backup_fs / filename, disk, temp_part_dir / filename, WriteMode::Rewrite); + reservation->update(reservation->getSize() - file_size); + } +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } + if (restore_tx) + restore_tx->commit(); + if (auto part = loadPartRestoredFromBackup(part_name, disk, temp_part_dir, detach_if_broken)) restored_parts_holder->addPart(part); else @@ -10636,12 +10741,31 @@ void MergeTreeData::Transaction::clear() void MergeTreeData::Transaction::renameParts() { + /// Materialize every part of this transaction: perform the deferred tmp->final renames, then + /// close each part's disk-storage transaction, making the parts DURABLE on their disks. + /// + /// Contract: after renameParts returns, every part of this transaction is durable at its + /// final name. commit only flips in-memory visibility (its commitTransaction loop remains as + /// a safety net for paths that do not come through here); rollback compensates with new + /// operations over committed disk state (removing a rolled-back part reclaims its disk data; + /// on a content-addressed disk that drops the published ref). + /// + /// Ordering is load-bearing: every call site invokes renameParts BEFORE its external Keeper + /// commit decision. A part must be durable before its block_id/part-znode is registered, + /// otherwise a fault between the Keeper commit and the disk commit leaves a phantom part whose + /// surviving block_id silently dedups a byte-identical client retry (acked data loss). This + /// also keeps the disk commit (network I/O on object storages) off the data_parts lock, which + /// Transaction::commit holds. for (const auto & part_need_rename : precommitted_parts_need_rename) { LOG_TEST(data.log, "Renaming part to {}", part_need_rename->name); part_need_rename->renameTo(part_need_rename->name, true); } precommitted_parts_need_rename.clear(); + + for (const auto & part : precommitted_parts) + if (part->getDataPartStorage().hasActiveTransaction()) + part->getDataPartStorage().commitTransaction(); } MergeTreeData::DataPartsVector MergeTreeData::Transaction::commit() diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index 3caa4b0f4c19..d5e83eb87475 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -374,9 +374,17 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr DataPartsVector commit(); DataPartsVector commit(DataPartsLock & lock); - /// Rename should be done explicitly, before calling commit(), to - /// guarantee that no lock held during rename (since rename is IO - /// bound, while data parts lock is the bottleneck) + /// Renames should be done explicitly, before calling commit, to + /// guarantee that no lock is held during the rename and the disk + /// commit (both are IO bound, while the data parts lock is the + /// bottleneck). Contract: after renameParts every part of this + /// transaction is durable on its disk at its final name; commit only + /// flips in-memory visibility, and rollback compensates via new disk + /// operations (part removal). Every caller runs this BEFORE its + /// external Keeper commit decision: a part must be durable before its + /// block_id/part-znode is registered in Keeper, otherwise a fault between the two commits + /// leaves a phantom part whose surviving block_id silently dedups a byte-identical client + /// retry (acked data loss). void renameParts(); void addPart(MutableDataPartPtr & part, bool need_rename); diff --git a/src/Storages/MergeTree/MergeTreeDataWriter.cpp b/src/Storages/MergeTree/MergeTreeDataWriter.cpp index 0df5d547c2a0..30626a63e9fe 100644 --- a/src/Storages/MergeTree/MergeTreeDataWriter.cpp +++ b/src/Storages/MergeTree/MergeTreeDataWriter.cpp @@ -1126,6 +1126,8 @@ MergeTreeTemporaryPartPtr MergeTreeDataWriter::writeProjectionPartImpl( auto projection_part_storage = new_data_part->getDataPartStoragePtr(); auto data_settings = data.getSettings(&projection.settings_changes); + /// A temp projection sub-part opens a transaction only if it owns one; a borrowed (CA) projection + /// storage makes beginTransaction a no-op, so the `isContentAddressed()` branch is no longer needed. if (is_temp) projection_part_storage->beginTransaction(); diff --git a/src/Storages/MergeTree/MergeTreeDeduplicationLog.cpp b/src/Storages/MergeTree/MergeTreeDeduplicationLog.cpp index 9987e466c53b..a4ff5691bf4a 100644 --- a/src/Storages/MergeTree/MergeTreeDeduplicationLog.cpp +++ b/src/Storages/MergeTree/MergeTreeDeduplicationLog.cpp @@ -20,6 +20,7 @@ namespace DB namespace ErrorCodes { extern const int ABORTED; + extern const int LOGICAL_ERROR; } namespace @@ -103,8 +104,14 @@ void MergeTreeDeduplicationLog::load() { if (auto * object_storage = dynamic_cast(disk.get())) { - // MetadataStorageType::Plain does not have directory concept. When checking `logs_dir` existence, it might return false. - if (object_storage->getMetadataStorage()->getType() != MetadataStorageType::Plain) + // Plain and ContentAddressed object storages do not materialize empty directories, so a + // missing logs_dir is normal for a fresh table: fall through so the current_writer is still + // created (an INSERT must have a writer, else addPart fails closed). For these types a + // missing dir is NOT evidence of nothing to do; iterateDirectory below finds any logs that + // already exist, and rotate() creates the writer when there are none. Any other object + // storage returns here: a missing dir means there is genuinely nothing and nowhere to write. + const auto type = object_storage->getMetadataStorage()->getType(); + if (type != MetadataStorageType::Plain && type != MetadataStorageType::CAS) return; } } @@ -268,7 +275,15 @@ std::vector MergeTreeDeduplicationLog: throw Exception(ErrorCodes::ABORTED, "Storage has been shutdown when we add this part."); } - chassert(current_writer != nullptr); + /// A disk that cannot host the append-mode log leaves current_writer null; the release-build + /// chassert above is a no-op, so dereferencing it would segfault. Fail closed with a clear + /// exception instead of crashing the server (B37). + if (!current_writer) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "MergeTree deduplication log has no writer (the disk does not support the on-disk " + "deduplication log); cannot add part {}", + part_info.getPartNameAndCheckFormat(format_version)); for (const auto & block_id : block_ids) { @@ -306,7 +321,13 @@ void MergeTreeDeduplicationLog::dropPart(const MergeTreePartInfo & drop_part_inf throw Exception(ErrorCodes::ABORTED, "Storage has been shutdown when we drop this part."); } - chassert(current_writer != nullptr); + /// As in addPart: a null writer must produce a clear exception, never a segfault (B37). + if (!current_writer) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "MergeTree deduplication log has no writer (the disk does not support the on-disk " + "deduplication log); cannot drop part {}", + drop_part_info.getPartNameAndCheckFormat(format_version)); for (auto itr = deduplication_map.begin(); itr != deduplication_map.end(); /* no increment here, we erasing from map */) { diff --git a/src/Storages/MergeTree/MutateTask.cpp b/src/Storages/MergeTree/MutateTask.cpp index 0e5d228f1508..524a0758f9d1 100644 --- a/src/Storages/MergeTree/MutateTask.cpp +++ b/src/Storages/MergeTree/MutateTask.cpp @@ -2194,6 +2194,9 @@ void PartMergerWriter::writeTempProjectionPart(size_t projection_idx, Chunk chun ctx->context); tmp_part->finalize(); + /// A borrowed (CA) temp projection sub-part shares the new (parent) part's whole-part transaction + /// (see `IMergeTreeDataPart::getProjectionPartBuilder`) and is committed by the parent's single + /// commit; the storage makes commitTransaction a no-op there, so this is called unconditionally (B58). tmp_part->part->getDataPartStorage().commitTransaction(); projection_parts[projection.name].emplace_back(std::move(tmp_part->part)); } diff --git a/src/Storages/MergeTree/tests/gtest_deduplication_log_null_writer.cpp b/src/Storages/MergeTree/tests/gtest_deduplication_log_null_writer.cpp new file mode 100644 index 000000000000..7a2f72313a19 --- /dev/null +++ b/src/Storages/MergeTree/tests/gtest_deduplication_log_null_writer.cpp @@ -0,0 +1,139 @@ +#include + +#include +#include +#include +#include +#include /// DEBUG_OR_SANITIZER_BUILD + +#include +#include +#include +#include + +using namespace DB; + +namespace DB::ErrorCodes +{ + extern const int LOGICAL_ERROR; +} + +namespace +{ +constexpr auto FORMAT_VERSION = MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING; + +/// B37 regression: a `MergeTreeDeduplicationLog` whose `current_writer` is null (the disk could not +/// host the append-mode log -- see `MergeTreeDeduplicationLog::load()`'s early-return path for a +/// `DiskObjectStorage` whose metadata storage type is neither `Plain` nor `ContentAddressed`) used to +/// be dereferenced unconditionally by `addPart`/`dropPart`: a release-build `chassert` is a no-op, so +/// this was a null-pointer dereference (segfault) rather than a handled error. The fix makes both +/// throw a `LOGICAL_ERROR` `DB::Exception` instead. +/// +/// There is no way to drive this from a stateless SQL test: every disk type that reaches production +/// either materializes `logs_dir` (so `load()` takes the normal `rotate()` path and sets a writer) or +/// is one of the two types (`Plain`, `ContentAddressed`) `load()` explicitly special-cases to still get +/// a writer. So this test constructs the log directly and never calls `load()` -- `current_writer` +/// simply stays at its default-constructed null value, which is the exact precondition the guard in +/// `addPart`/`dropPart` exists for. +struct DeduplicationLogNullWriterFixture : public ::testing::Test +{ + std::filesystem::path base_path; + DiskPtr disk; + std::unique_ptr log; + + void SetUp() override + { + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(reinterpret_cast(this)); + base_path = std::filesystem::temp_directory_path() / ("dedup_log_null_writer_gtest_" + unique); + std::filesystem::create_directories(base_path); + disk = std::make_shared("test_disk_" + unique, base_path.string()); + + /// deduplication_window != 0 so addPart/dropPart don't bail out on the "deduplication is off" + /// fast path before ever reaching the null-writer guard. `load()` is deliberately NOT called: + /// that is what leaves `current_writer` null. + log = std::make_unique("deduplication_logs", /*deduplication_window_=*/4, FORMAT_VERSION, disk); + } + + void TearDown() override + { + log.reset(); + std::error_code ec; + std::filesystem::remove_all(base_path, ec); + } +}; + +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +/// gtest runs *DeathTest suites before others; reuse the same fixture via an alias so the death arm +/// gets the same null-writer precondition. +using DeduplicationLogNullWriterDeathTest = DeduplicationLogNullWriterFixture; +#endif + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST_F(DeduplicationLogNullWriterFixture, AddPartThrowsLogicalErrorInsteadOfCrashing) +{ + /// LOGICAL_ERROR "no writer" is a broken-invariant guard (addPart on a null current_writer). Under + /// abort_on_logical_error it aborts at construction instead of being catchable -- the DeathTest + /// below proves the abort in those builds. + auto part_info = MergeTreePartInfo::fromPartName("all_0_0_0", FORMAT_VERSION); + + EXPECT_THROW( + { + try + { + log->addPart({"block-1"}, part_info); + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::LOGICAL_ERROR); + EXPECT_NE(e.message().find("no writer"), std::string::npos); + throw; + } + }, + Exception); + + /// The object stays alive and usable after the guard fires: it isn't left half-corrupted by the + /// failed call, and repeating the same call (still no writer) throws again, cleanly, rather than + /// crashing or behaving differently the second time. + EXPECT_THROW(log->addPart({"block-1"}, part_info), Exception); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST_F(DeduplicationLogNullWriterDeathTest, AddPartAborts) +{ + auto part_info = MergeTreePartInfo::fromPartName("all_0_0_0", FORMAT_VERSION); + EXPECT_DEATH({ log->addPart({"block-1"}, part_info); }, "no writer"); +} +#endif + +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST_F(DeduplicationLogNullWriterFixture, DropPartThrowsLogicalErrorInsteadOfCrashing) +{ + auto part_info = MergeTreePartInfo::fromPartName("all_0_0_0", FORMAT_VERSION); + + EXPECT_THROW( + { + try + { + log->dropPart(part_info); + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::LOGICAL_ERROR); + EXPECT_NE(e.message().find("no writer"), std::string::npos); + throw; + } + }, + Exception); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST_F(DeduplicationLogNullWriterDeathTest, DropPartAborts) +{ + auto part_info = MergeTreePartInfo::fromPartName("all_0_0_0", FORMAT_VERSION); + EXPECT_DEATH({ log->dropPart(part_info); }, "no writer"); +} +#endif diff --git a/src/Storages/MergeTree/tests/gtest_projection_borrowed_transaction.cpp b/src/Storages/MergeTree/tests/gtest_projection_borrowed_transaction.cpp new file mode 100644 index 000000000000..ae079a61450f --- /dev/null +++ b/src/Storages/MergeTree/tests/gtest_projection_borrowed_transaction.cpp @@ -0,0 +1,86 @@ +#include + +#include +#include +#include + +#include +#include +#include +#include + +using namespace DB; + +namespace +{ + /// A DiskLocal-backed parent part storage. `DiskLocal::createTransaction` yields a real + /// transaction object, which is all `beginTransaction` needs to hand a NON-NULL transaction to a + /// borrowed projection sub-part (the `has_shared_transaction == true` case). + struct ParentStorageFixture + { + std::filesystem::path base_path; + DiskPtr disk; + VolumePtr volume; + MutableDataPartStoragePtr parent; + + ParentStorageFixture() + { + const auto unique = std::to_string(::getpid()) + "_" + + std::to_string(reinterpret_cast(this)); + base_path = std::filesystem::temp_directory_path() / ("proj_txn_gtest_" + unique); + std::filesystem::create_directories(base_path / "all_1_1_0"); + disk = std::make_shared("test_disk_" + unique, base_path.string()); + volume = std::make_shared("test_volume", disk); + parent = std::make_shared(volume, /*root_path=*/"", "all_1_1_0"); + } + + ~ParentStorageFixture() + { + std::error_code ec; + std::filesystem::remove_all(base_path, ec); + } + }; +} + +/// A projection sub-part that BORROWS the parent's whole-part transaction (the CA-disk shape: +/// getProjection(..., use_parent_transaction = true)) must let begin/commit be NO-OPS — it rides the +/// parent's single commit. Before the encapsulation this threw "Uncommitted shared transaction already +/// exists" / "Cannot commit shared transaction", forcing every caller to branch on isContentAddressed(). +TEST(ProjectionBorrowedTransaction, BorrowedStorageBeginCommitAreNoOps) +{ + ParentStorageFixture fx; + + /// Parent opens the whole-part transaction (as MergeTask/writer do for a CA part). + fx.parent->beginTransaction(); + ASSERT_TRUE(fx.parent->hasActiveTransaction()); + + /// Borrowed projection sub-part: shares the parent transaction (has_shared_transaction == true). + auto proj = fx.parent->getProjection("p.proj", /*use_parent_transaction=*/true); + EXPECT_TRUE(proj->hasActiveTransaction()); + + /// The encapsulated rule: begin/commit on the borrowed storage are silent no-ops (they must NOT + /// open a second transaction, nor commit the parent's). + EXPECT_NO_THROW(proj->beginTransaction()); + EXPECT_NO_THROW(proj->commitTransaction()); + + /// The parent's transaction is untouched by the projection's no-ops and still commits cleanly. + EXPECT_TRUE(fx.parent->hasActiveTransaction()); + EXPECT_NO_THROW(fx.parent->commitTransaction()); + EXPECT_FALSE(fx.parent->hasActiveTransaction()); +} + +/// The non-CA temp-projection shape (use_parent_transaction = false) is unchanged: the sub-part OWNS +/// its transaction, so begin creates it and commit commits it (has_shared_transaction == false, so the +/// no-op path never triggers). +TEST(ProjectionBorrowedTransaction, OwnedProjectionStorageStillBeginsAndCommits) +{ + ParentStorageFixture fx; + + auto proj = fx.parent->getProjection("q.proj", /*use_parent_transaction=*/false); + EXPECT_FALSE(proj->hasActiveTransaction()); + + EXPECT_NO_THROW(proj->beginTransaction()); + EXPECT_TRUE(proj->hasActiveTransaction()); + EXPECT_NO_THROW(proj->commitTransaction()); + EXPECT_FALSE(proj->hasActiveTransaction()); +} diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index 66650b8aeb23..62e30fdd16d7 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -14,6 +14,7 @@ #include #include #include +#include #include #include #include @@ -190,11 +191,15 @@ static bool supportTransaction(const Disks & disks, LoggerPtr log) { for (const auto & disk : disks) { - if (!supportWritingWithAppend(disk)) - { - LOG_DEBUG(log, "Disk {} does not support writing with append", disk->getName()); - return false; - } + if (supportWritingWithAppend(disk)) + continue; + /// A content-addressed disk does not support append, but persists the per-part mutable + /// transaction file (txn_version.txt) via its per-ref sidecar, which is all MVCC needs. + if (auto * obj = dynamic_cast(disk.get()); + obj && obj->getMetadataStorage()->supportsTransactionalMutableFiles()) + continue; + LOG_DEBUG(log, "Disk {} does not support transactions", disk->getName()); + return false; } return true; } diff --git a/src/Storages/StorageProxy.h b/src/Storages/StorageProxy.h index bccc4ffeae5e..7301bb94f3a3 100644 --- a/src/Storages/StorageProxy.h +++ b/src/Storages/StorageProxy.h @@ -163,6 +163,15 @@ class StorageProxy : public IStorage void mutate(const MutationCommands & commands, ContextPtr context) override { getNested()->mutate(commands, context); } + /// Must forward alongside `mutate`: `IStorage`'s default throws NOT_IMPLEMENTED ("doesn't + /// support mutations"), so a non-forwarding proxy rejects every mutation on a wrapped table + /// even though the nested engine supports them (found via `ALTER TABLE ... MATERIALIZE TTL` + /// on a `lazy_load_tables = 1` table wrapped in `StorageTableProxy`). + void checkMutationIsPossible(const MutationCommands & commands, const Settings & settings) const override + { + getNested()->checkMutationIsPossible(commands, settings); + } + CancellationCode killMutation(const String & mutation_id) override { return getNested()->killMutation(mutation_id); } void startup() override { getNested()->startup(); } diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 17c5cd176f43..6eefda9c3e95 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -531,6 +531,18 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( { if (disk->getDataSourceDescription().metadata_type == MetadataStorageType::Keeper) throw Exception(ErrorCodes::BAD_ARGUMENTS, "ReplicatedMergeTree doesn't work with 's3_with_keeper' disk type"); + + /// B33 (lifted, CAS replication 2b + Phase 3.2): ReplicatedMergeTree on a content-addressed disk + /// is allowed. INSERT/SELECT/merge/mutation and fetch-by-relink (the CA analogue of zero-copy + /// replication) route through the working whole-part CA transaction / the relink path. The + /// replication-queue CLONE paths (queue-driven REPLACE/MOVE/ATTACH PARTITION FROM, the + /// cloneAndLoadDataPart-on-the-queue path) were audited in Phase 3.2: they reach the SAME + /// whole-part ContentAddressedTransaction the non-replicated stack uses (see + /// `MergeTreeData::checkAlterPartitionIsPossible`, reached here by dynamic dispatch — the + /// Phase 3.2 fail-closed override in this class was a pure delegation and was deleted by the + /// tail de-patch), NOT the per-file-autocommit B21 mode, so they are now permitted. The + /// zero-copy lockSharedData/unlockSharedData calls these reach are safe no-ops on CA (they + /// early-return on !supportZeroCopyReplication, which CA is). } initializeDirectoriesAndFormatVersion(relative_data_path_, LoadingStrictnessLevel::ATTACH <= mode, date_column_name); diff --git a/src/Storages/StorageTableProxy.h b/src/Storages/StorageTableProxy.h index 26f69e01992a..8420c0010dae 100644 --- a/src/Storages/StorageTableProxy.h +++ b/src/Storages/StorageTableProxy.h @@ -72,6 +72,14 @@ class StorageTableProxy final : public StorageProxy StoragePolicyPtr getStoragePolicy() const override { return nullptr; } bool isView() const override { return false; } + /// NOTE: this proxy deliberately does NOT forward `checkTableCanBeRenamed` to the nested engine. + /// Doing so would materialize the lazy table (`getNested`) while `DatabaseAtomic` holds its + /// non-recursive database mutex, and a schema-inferred lazy `Buffer` resolves its destination via + /// `DatabaseCatalog::getTable` in its constructor -- re-entering the same database and self- + /// deadlocking. Bypassing the nested engine's rename restriction for a lazy (never-accessed) table + /// is a pre-existing gap tracked in docs/superpowers/cas/BACKLOG.md; the correct fix is to + /// materialize before the database mutex is taken, at the interpreter level. + /// /// Startup is deferred until first access via `getNested`. void startup() override { } diff --git a/src/Storages/System/StorageSystemContentAddressedMounts.cpp b/src/Storages/System/StorageSystemContentAddressedMounts.cpp new file mode 100644 index 000000000000..2a1b82a01ee7 --- /dev/null +++ b/src/Storages/System/StorageSystemContentAddressedMounts.cpp @@ -0,0 +1,273 @@ +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int INVALID_STATE; +} + +StorageSystemContentAddressedMounts::StorageSystemContentAddressedMounts(const StorageID & table_id_) + : StorageWithCommonVirtualColumns(table_id_) +{ + StorageInMemoryMetadata storage_metadata; + storage_metadata.setColumns(ColumnsDescription( + { + {"disk", std::make_shared(), "Name of the content-addressed disk."}, + {"server_root_id", std::make_shared(), "Server root id owning the mount slot."}, + {"server_uuid", std::make_shared(), "UUID of the server incarnation holding the lease."}, + {"hostname", std::make_shared(), "Hostname recorded in the lease body."}, + {"process_id", std::make_shared(), "Process id recorded in the lease body."}, + {"writer_epoch", std::make_shared(), "Fenced writer epoch of the incarnation."}, + {"renewal_sequence", std::make_shared(), "Lease renewal sequence number."}, + {"started_at", std::make_shared(3), "Time when the lease started."}, + {"expires_at", std::make_shared(3), "Time when the lease expires."}, + {"min_active_build_sequence", std::make_shared(), "Oldest in-flight build sequence (UINT64_MAX means the mount said farewell)."}, + {"gc_fenced", std::make_shared(), "1 if GC fenced this slot out (terminal)."}, + {"state", std::make_shared(), "Mount slot state: live, expired, terminated, fenced or corrupt."}, + {"is_leader", std::make_shared(std::make_shared()), "1 if this server's GC scheduler holds this disk's leadership lease. NULL on rows describing other servers' mounts."}, + {"pending_reclaim", std::make_shared(std::make_shared()), "Cumulative condemned-minus-deleted backlog observed by this process's GC on this disk. NULL on rows describing other servers' mounts."}, + {"last_success_age_seconds", std::make_shared(std::make_shared()), "Seconds since this disk's GC last led a round (0 if it never led). NULL on rows describing other servers' mounts."}, + {"wedged_namespace_count", std::make_shared(std::make_shared()), "Ref-append lanes currently wedged on this disk. NULL on rows describing other servers' mounts."}, + {"lifecycle", std::make_shared(), "This server's content-addressed pool lifecycle for the disk (non-gated snapshot, always populated so a not-live disk stays visible): live, not_live, identity_lost, vanished, constructing (never started) or shutdown (torn down)."}, + {"lifecycle_reason", std::make_shared(), "The enum-clean sub-state word for a vanished disk: replaced or forgotten. Empty for every other lifecycle (so lifecycle || '(' || lifecycle_reason || ')' reads e.g. vanished(forgotten))."}, + {"lifecycle_detail", std::make_shared(), "The full typed reason text naming the actual cause when not live: the vanish diagnosis (data root replaced by a foreign pool / decommissioned by SYSTEM CAS FORGET at ", + "object_storages3" + "http://fakegcs:8080/plainhmacbucket/plain/" + "gcs_hmacGOOG1EFAKEACCESSKEYID" + "fake-goog4-hmac-secret", + ) + node.replace_in_config( + CONFIG_IN_CONTAINER, + "", + "
plain_gcs_hmac
" + "
", + ) + node.replace_in_config( + CONFIG_IN_CONTAINER, + "", + "" + "http://fakegcs:8080/plainhmacbucket/ordinary/" + "gcs_hmac" + "GOOG1EFAKEACCESSKEYID" + "fake-goog4-hmac-secret" + "", + ) + node.restart_clickhouse() + + for disk in CAS_DISKS: + _create_and_fill(node, disk) + _create_and_fill(node, PLAIN_DISK) + _create_and_fill(node, PLAIN_HMAC_DISK) + yield cluster + finally: + cluster.shutdown() + + +def _create_and_fill(node, disk): + table = "t_" + disk + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query( + """ + CREATE TABLE {} (id Int64, data String) + ENGINE = MergeTree() ORDER BY id + SETTINGS storage_policy = '{}' + """.format( + table, disk + ) + ) + node.query( + "INSERT INTO {} SELECT number, toString(number) FROM numbers({})".format( + table, NUM_ROWS + ) + ) + + +def _control(path): + container = cluster.get_container_id(GCS_HOST) + raw = cluster.exec_in_container( + container, ["curl", "-sS", "http://localhost:{}{}".format(GCS_PORT, path)] + ) + return json.loads(raw) + + +def _control_post(path): + container = cluster.get_container_id(GCS_HOST) + raw = cluster.exec_in_container( + container, + ["curl", "-sS", "-X", "POST", "http://localhost:{}{}".format(GCS_PORT, path)], + ) + return json.loads(raw) if raw.strip().startswith(("{", "[")) else raw + + +def _counters(): + return _control("/_control/counters") + + +def _set_if_match_mode(mode): + return _control_post("/_control/mode?if_match=" + mode) + + +def _set_omit_generation(enabled): + return _control_post("/_control/mode?omit_generation=" + ("1" if enabled else "0")) + + +def _condemn_blob(bucket, blob_key): + return _control_post( + "/_control/condemn?bucket={}&key={}".format( + urllib.parse.quote(bucket, safe=""), urllib.parse.quote(blob_key, safe="") + ) + ) + + +def _raw(method, path, headers=()): + """Issue one request to the fake from inside its own container and return its status code. + + Used only to drive request shapes production never sends, so that the fake's own discriminating + power can be asserted rather than assumed. + + HEAD goes through `--head` rather than `-X HEAD`: with `-X HEAD` curl sends the request but still + waits for a response body, and a HEAD reply never has one, so it blocks until something kills it. + `--max-time` is here for the same class of mistake — a fixture hang should cost seconds, not the + module's whole budget. + """ + container = cluster.get_container_id(GCS_HOST) + command = ["curl", "-sS", "--max-time", "30", "-o", "/dev/null", "-w", "%{http_code}"] + command += ["--head"] if method == "HEAD" else ["-X", method] + for name, value in headers: + command += ["-H", "{}: {}".format(name, value)] + command.append("http://localhost:{}{}".format(GCS_PORT, path)) + return int(cluster.exec_in_container(container, command).strip()) + + +def _token_fetches(): + container = cluster.get_container_id(METADATA_HOST) + raw = cluster.exec_in_container( + container, + ["curl", "-sS", "http://localhost:{}/_control/tokens".format(METADATA_PORT)], + ) + return json.loads(raw)["fetches"] + + +def _reset_token_fetches(): + container = cluster.get_container_id(METADATA_HOST) + cluster.exec_in_container( + container, + ["curl", "-sS", "http://localhost:{}/_control/tokens/reset".format(METADATA_PORT)], + ) + + +def _captured(bucket=None): + records = _control("/_control/requests") + if bucket is None: + return records + return [r for r in records if r["bucket"] == bucket] + + +def _minted(): + return _control("/_control/minted") + + +def _next_seq(): + """The `seq` the fake's next captured request will carry, so a later slice can start here.""" + return len(_control("/_control/requests")) + + +def _captured_since(seq, bucket=None): + return [r for r in _captured(bucket) if r["seq"] >= seq] + + +def _unquote(value): + return value.strip().strip('"') + + +def _generation_preconditions(records): + return [ + r["headers"]["x-goog-if-generation-match"] + for r in records + if "x-goog-if-generation-match" in r["headers"] + ] + + +def _has_goog_metadata(record): + return any(name.startswith("x-goog-meta-") for name in record["headers"]) + + +def _is_translated(record): + return "x-goog-if-generation-match" in record["headers"] or _has_goog_metadata(record) + + +def _blob_publications(records, key=None): + publications = [ + record + for record in records + if record["operation"] in ("blob_put", "staged_copy", "blob_multipart_complete") + ] + if key is not None: + publications = [record for record in publications if record["key"] == key] + return publications + + +def _meta_requests(records, blob_key): + return [record for record in records if record["key"] == blob_key + ".meta"] + + +def _assert_default_blob_publication(record): + assert record["request_class"] == "blob_body", record + assert "x-goog-if-generation-match" not in record["headers"], record + assert "if-match" not in record["headers"], record + assert "if-none-match" not in record["headers"], record + assert not _has_goog_metadata(record), record + if record["bucket"] == CAS_DISKS["cas_gcs_oauth"]: + assert record["headers"].get("authorization", "").startswith("Bearer "), record + if record["bucket"] == CAS_DISKS["cas_gcs_hmac"]: + assert record["headers"].get("authorization", "").startswith("GOOG4-HMAC-SHA256 "), record + + +# `AWS_HEADERS_CLEARED_BEFORE_GCS_AUTHENTICATION` in GCSConditionalDialect.cpp lists +# `x-amz-api-version`, and `prepareGcsRequestForOAuthAuthentication` — which runs ONLY for a marked +# request on the OAuth client — deletes every header in it. So on a `gcp_oauth` client the header's +# PRESENCE means the request was `Default` and its ABSENCE means the request was marked. That makes +# marking observable on the wire for the request kinds the SDK stamps with it, which measurement says +# are GET, HEAD and DELETE but not PUT. +# +# This does NOT hold on `gcs_hmac`: `prepareGcsRequestForGoog4Authentication` runs for every request +# that client sends, marked or not, so the header is always absent there and says nothing about mode. +# +# One other thing deletes the same header, and understanding why it does not fire here is what makes +# the discriminator trustworthy. `Client::BuildHttpRequest` also drops `x-amz-api-version` when +# `api_mode == ApiMode::GCS`, for every request and before the marking logic runs. `api_mode` becomes +# GCS only inside a block gated on `provider_type == ProviderType::GCS`, and `deduceProviderType` is +# pure endpoint-substring matching: GCS requires `storage.googleapis.com` in the URL. This fixture's +# endpoint deliberately contains no such substring, so `provider_type` is UNKNOWN, that block never +# runs, and the header survives to become a marking signal. +# +# Note what this does NOT mean. It is not that these disks have credentials: `gcp_oauth` deliberately +# builds an EMPTY credentials provider chain ("we don't provide any credentials to avoid signing" in +# Credentials.cpp), which is also why no SigV4 artifact such as `x-amz-content-sha256` ever appears. +# Against a real `storage.googleapis.com` endpoint `provider_type` WOULD be GCS, those empty +# credentials would select `ApiMode::GCS`, and the header would be stripped from every request. So +# this discriminator is an artifact of the fixture's non-GCS hostname and could not be reproduced +# against production GCS. Marking itself is unaffected — it depends on `http_client`, not the endpoint +# — so the test still fences the behaviour; only the ability to OBSERVE it is endpoint-dependent. +# +# Consequence for a future reader: if these assertions ever start failing uniformly rather than for +# one request, suspect that the endpoint or `provider_type` changed and the discriminator is gone, +# before suspecting that marking broke. +_MARKING_OBSERVABLE_METHODS = ("GET", "HEAD", "DELETE") + + +def _looks_default_on_oauth(record): + return "x-amz-api-version" in record["headers"] + + +@pytest.mark.parametrize("request_class", ("blob_meta", "cas_control")) +@pytest.mark.parametrize( + "method,query,headers,expected", + ( + ("POST", {"uploads": [""]}, {}, "blob_multipart_create"), + ( + "PUT", + {"partNumber": ["1"], "uploadId": ["upload-1"]}, + {"x-goog-if-generation-match": "7"}, + "blob_multipart_part", + ), + ("POST", {"uploadId": ["upload-1"]}, {}, "blob_multipart_complete"), + ), +) +def test_mock_classifies_multipart_before_object_role( + request_class, method, query, headers, expected +): + """A forbidden mutable multipart request must remain visible to confinement assertions.""" + assert ( + GCS_MOCK_NAMESPACE["_request_operation"]( + "oauthbucket", request_class, method, query, headers + ) + == expected + ) + + +def test_data_is_readable_on_every_disk(): + """Mount, write and read back on both `http_client` values and on the ordinary disk. + + A CAS mount runs `runCapabilityProbe` against the store and refuses the mount unless conditional + create, conditional overwrite, wrong-token delete rejection and correct-token delete all behave. + Reaching a correct SELECT therefore proves the whole battery passed over GCS generation + semantics. Would fail if: the request-mode plumbing stopped marking any CAS operation, since the + fake rejects a generation sent as an ETag and an ETag sent as a generation. + """ + node = cluster.instances["node"] + expected_sum = (NUM_ROWS - 1) * NUM_ROWS // 2 + for disk in list(CAS_DISKS) + [PLAIN_DISK, PLAIN_HMAC_DISK]: + table = "t_" + disk + assert int(node.query("SELECT count() FROM {}".format(table))) == NUM_ROWS + assert int(node.query("SELECT sum(id) FROM {}".format(table))) == expected_sum + + +def test_fake_service_keeps_the_two_token_domains_disjoint(): + """The fixture's own invariant, asserted rather than assumed. + + Negative control: nothing in ClickHouse can flip this — it is a property of the fake. It is + asserted anyway because every assertion below is only meaningful while it holds. + """ + minted = _minted() + assert minted["generations"], "the fake minted no generation, so nothing below is meaningful" + assert minted["etags"], "the fake minted no ETag, so nothing below is meaningful" + for generation in minted["generations"]: + assert re.fullmatch(r"[0-9]{16}", generation), generation + for etag in minted["etags"]: + assert not _unquote(etag).isdigit(), etag + assert not (set(minted["generations"]) & {_unquote(e) for e in minted["etags"]}) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_blob_publication_request_budget_and_default_mode(disk): + """Pin the fresh and cold-duplicate request shapes for one real blob key per CAS disk. + + The OAuth disk publishes from local staging with an unconditional body PUT. The GOOG4 disk uses + explicit S3 staging and publishes with a native-only, unconditional copy. Repeating byte-identical + data then selects a blob that both inserts touched and proves the cold path uses one body `HEAD`, + one metadata GET, and no publication. The mock proves syntax, routing, and count isolation only; + live GCS acceptance belongs to the credential-gated Task 10 lane. + + Would fail if: the mandatory blob `HEAD` were skipped or duplicated, fresh publication read meta + before writing, a fresh body regained a conditional request mode, `Clean` metadata stopped being + created, or a cold duplicate issued another body PUT/copy. + """ + node = cluster.instances["node"] + bucket = CAS_DISKS[disk] + table = "task9_budget_" + disk + insert = ( + "INSERT INTO {} SELECT number, concat('task9-{}-', toString(number), repeat('q', 2048)) " + "FROM numbers(32)".format(table, disk) + ) + + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query( + "CREATE TABLE {} (id UInt64, payload String) ENGINE = MergeTree ORDER BY id " + "SETTINGS storage_policy = '{}'".format(table, disk) + ) + + fresh_seq = _next_seq() + node.query(insert) + fresh = _captured_since(fresh_seq, bucket) + publications = _blob_publications(fresh) + assert publications, "the fresh insert published no classified blob body" + + for publication in publications: + key = publication["key"] + heads = [r for r in fresh if r["key"] == key and r["operation"] == "native_token_head"] + assert len(heads) == 1, (key, heads) + assert heads[0]["status"] == 404, heads[0] + if disk == "cas_gcs_oauth": + assert not _looks_default_on_oauth(heads[0]), heads[0] + assert heads[0]["headers"].get("authorization", "").startswith("Bearer "), heads[0] + else: + assert heads[0]["headers"].get("authorization", "").startswith( + "GOOG4-HMAC-SHA256 " + ), heads[0] + assert len(_blob_publications(fresh, key)) == 1, (key, _blob_publications(fresh, key)) + _assert_default_blob_publication(publication) + + meta = _meta_requests(fresh, key) + publication_seq = publication["seq"] + assert not [ + r for r in meta if r["method"] == "GET" and r["seq"] < publication_seq + ], (key, meta) + creates = [ + r + for r in meta + if r["method"] == "PUT" + and r["headers"].get("x-goog-if-generation-match") == "0" + and '"st":"clean"' in r["request_body"] + ] + assert len(creates) == 1, (key, meta) + + duplicate_seq = _next_seq() + node.query(insert) + duplicate = _captured_since(duplicate_seq, bucket) + reusable = [] + for key in {record["key"] for record in publications}: + heads = [ + r for r in duplicate if r["key"] == key and r["operation"] == "native_token_head" + ] + meta_gets = [r for r in _meta_requests(duplicate, key) if r["method"] == "GET"] + if heads and meta_gets: + assert len(heads) == 1, (key, heads) + assert heads[0]["status"] == 200, heads[0] + assert len(meta_gets) == 1, (key, meta_gets) + assert not _blob_publications(duplicate, key), (key, _blob_publications(duplicate, key)) + reusable.append(key) + assert reusable, "no fresh blob was observed as a cold duplicate on the second insert" + + if disk == "cas_gcs_hmac": + target = sorted(reusable)[0] + condemned = _condemn_blob(bucket, target) + assert condemned["state"] == "condemned", condemned + + retry_seq = _next_seq() + node.query(insert) + retry = _captured_since(retry_seq, bucket) + + target_heads = [ + r for r in retry if r["key"] == target and r["operation"] == "native_token_head" + ] + target_meta_gets = [r for r in _meta_requests(retry, target) if r["method"] == "GET"] + staging_gets = [ + r for r in retry if r["request_class"] == "staging" and r["method"] == "GET" + ] + retagged_puts = [ + r for r in retry if r["key"] == target and r["operation"] == "blob_put" + ] + conditional_copies = [r for r in retry if r["operation"] == "conditional_copy"] + + assert len(target_heads) == 1, target_heads + assert len(target_meta_gets) == 1, target_meta_gets + assert len(staging_gets) == 1, staging_gets + assert len(retagged_puts) == 1, retagged_puts + assert not conditional_copies, conditional_copies + _assert_default_blob_publication(retagged_puts[0]) + assert retagged_puts[0]["response_generation"] != target_heads[0]["response_generation"] + + clean_cas = [ + r + for r in _meta_requests(retry, target) + if r["method"] == "PUT" + and r["headers"].get("x-goog-if-generation-match", "0") != "0" + and '"st":"clean"' in r["request_body"] + ] + assert len(clean_cas) == 1, clean_cas + + node.query("DROP TABLE {} SYNC".format(table)) + + +def test_default_blob_multipart_is_allowed_but_mutable_cas_stays_single_part(): + """A large OAuth blob may use multipart because its publication is unconditional and Default. + + Mutable CAS metadata/control PUTs still carry `NativeConditional` generation preconditions and + never fragment. Would fail if the old generation-wide single-part restriction survived, or if the + new multipart permission leaked from blob bodies into mutable coordination objects. + """ + node = cluster.instances["node"] + disk = "cas_gcs_oauth" + bucket = CAS_DISKS[disk] + table = "task9_blob_multipart" + + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query( + "CREATE TABLE {} (id UInt64, payload String) ENGINE = MergeTree ORDER BY id " + "SETTINGS storage_policy = '{}'".format(table, disk) + ) + + first_seq = _next_seq() + node.query( + "INSERT INTO {} SELECT 1, arrayStringConcat(arrayMap(x -> hex(cityHash64(x + 987654321)), " + "range(160000)))".format(table), + settings={"s3_max_single_part_upload_size": 0, "s3_min_upload_part_size": 65536}, + ) + records = _captured_since(first_seq, bucket) + multipart = [ + r + for r in records + if r["operation"] + in ("blob_multipart_create", "blob_multipart_part", "blob_multipart_complete") + ] + assert [r for r in multipart if r["operation"] == "blob_multipart_create"], multipart + assert [r for r in multipart if r["operation"] == "blob_multipart_part"], multipart + assert [r for r in multipart if r["operation"] == "blob_multipart_complete"], multipart + assert all(r["request_class"] == "blob_body" for r in multipart), multipart + assert all(not _is_translated(r) for r in multipart), multipart + + conditional_puts = [ + r + for r in records + if r["operation"] == "conditional_put" + and r["request_class"] in ("blob_meta", "cas_control") + ] + assert conditional_puts, "the insert issued no classified mutable conditional PUT" + assert all("uploads" not in r["query"] and "uploadId" not in r["query"] for r in conditional_puts) + + node.query("DROP TABLE {} SYNC".format(table)) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_cas_conditional_ops_use_generation_preconditions(disk): + """Create-if-absent and compare-and-set overwrite both travel as generation preconditions. + + Would fail if: `Client::BuildHttpRequest` stopped copying the mode, or CAS stopped marking its + conditional writes — the preconditions would then arrive as ETag-valued `If-Match` / + `If-None-Match` instead, and both value-domain assertions would break. + + The accepted-versus-rejected split matters. An earlier version of this test required EVERY + precondition to name a minted generation, and it failed against correct behaviour: the capability + battery fabricates known-wrong tokens on purpose, so a precondition the service never minted is + expected as long as the service refused it. + """ + records = _captured(CAS_DISKS[disk]) + assert records, "no request reached the fake for disk {}".format(disk) + + conditional = [r for r in records if "x-goog-if-generation-match" in r["headers"]] + preconditions = [r["headers"]["x-goog-if-generation-match"] for r in conditional] + assert "0" in preconditions, "no create-if-absent precondition was sent" + assert [p for p in preconditions if p != "0"], "no compare-and-set precondition was sent" + + minted = _minted() + known = set(minted["generations"]) | {"0"} + etag_values = {_unquote(e) for e in minted["etags"]} + + for record in conditional: + value = record["headers"]["x-goog-if-generation-match"] + # Always: the value lives in the generation domain and never in the ETag domain. This is the + # cross-domain check the whole fixture exists for. + assert value.isdigit(), "a non-numeric value reached the generation domain: {!r}".format(value) + assert value not in etag_values, "an ETag was sent as a generation: {}".format(value) + + # The capability battery deliberately fabricates wrong tokens (`900000000000000001` and + # friends in `CasProbe`) to prove the store enforces preconditions, so a precondition the + # service never minted is expected — but ONLY if the service rejected it. An ACCEPTED + # precondition must name a real generation, which is the half that would break if the token + # plumbing regressed. + if record["status"] < 300: + assert value in known, "the service accepted a precondition it never minted: {}".format(value) + else: + assert record["status"] == 412, ( + "a conditional request failed with {} rather than a precondition failure".format( + record["status"] + ) + ) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_no_cas_request_sends_a_generation_as_an_etag(disk): + """The exact-delete safety invariant, stated over the whole captured run. + + A numeric generation placed in an ETag-valued `If-Match` is the failure mode the design calls + safety-critical: the design does not assume whether GCS would reject, compare or ignore it. + Would fail if: any CAS conditional operation lost its mode, since CAS token values are + generations here. + """ + generations = set(_minted()["generations"]) + for record in _captured(CAS_DISKS[disk]): + for header in ("if-match", "if-none-match"): + value = record["headers"].get(header) + if value is None: + continue + assert _unquote(value) not in generations, ( + "{} {} sent a generation in {}: {}".format( + record["method"], record["key"], header, value + ) + ) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_stale_exact_delete_preserves_the_object_and_a_matching_one_removes_it(disk): + """The capability battery's delete pair, read off the wire. + + `runCapabilityProbe` deletes with a known-wrong generation, requires the object to survive, then + deletes with the correct one. Both halves must be visible, on one key, in that order. + Would fail if: the wrong-token DELETE were honoured (no 412 would appear), or the correct-token + DELETE were refused (mount would fail before this test ran). + """ + records = _captured(CAS_DISKS[disk]) + deletes = [ + r + for r in records + if r["operation"] == "exact_delete" + ] + assert deletes, "no generation-conditioned DELETE was sent" + + rejected = [r for r in deletes if r["status"] == 412] + accepted = [r for r in deletes if r["status"] == 204] + assert rejected, "no DELETE was rejected on a stale generation" + assert accepted, "no DELETE was accepted on a matching generation" + + keys_with_both = set(r["key"] for r in rejected) & set(r["key"] for r in accepted) + assert keys_with_both, "no single key saw both a rejected and an accepted exact DELETE" + + key = sorted(keys_with_both)[0] + first_rejection = min(r["seq"] for r in rejected if r["key"] == key) + later_success = min(r["seq"] for r in accepted if r["key"] == key) + assert first_rejection < later_success + + # Between the rejection and the successful delete the object must still be readable: that is the + # half of the battery that proves the store did not honour the stale token. + survived = [ + r + for r in records + if r["key"] == key + and r["method"] in ("GET", "HEAD") + and r["status"] == 200 + and first_rejection < r["seq"] < later_success + ] + assert survived, "the object was not observed alive between the stale and the matching DELETE" + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_list_stays_unmarked_and_its_etag_never_becomes_a_cas_token(disk): + """LIST keeps upstream ETag semantics on a CAS disk. + + Would fail if: LIST acquired the request mode — it would carry a translated header — or if a + LIST-derived ETag were ever accepted as a generation, which the value-domain check catches. + """ + records = _captured(CAS_DISKS[disk]) + lists = [r for r in records if r["method"] == "GET" and not r["key"]] + assert lists, "no LIST reached the fake" + for record in lists: + assert not _is_translated(record), "a LIST carried a translated header: {}".format( + record["query"] + ) + + etag_values = {_unquote(e) for e in _minted()["etags"]} + for value in _generation_preconditions(records): + assert value not in etag_values + + +def test_ordinary_gcp_oauth_traffic_keeps_upstream_semantics(): + """The upgrade regression this change exists to remove, checked on a non-CAS disk. + + The two absence assertions alone would be vacuous, and that is worth spelling out: a generation + precondition is only ever emitted for a request that already carried `If-Match`/`If-None-Match`, + and `x-goog-meta-*` only for one that carried `x-amz-meta-*`. An ordinary disk sends none of + those, so marking every request on the client — the exact regression this plan removes — would + leave those two assertions green. They are kept because they are cheap and true, not because they + fence anything. + + The assertion that DOES fence it is the last one. Marking a request runs + `prepareGcsRequestForOAuthAuthentication`, which deletes `x-amz-api-version`, so an ordinary + request must still carry it. Would fail if: `Client::BuildHttpRequest` marked requests it should + not — the header would vanish from this bucket. + """ + records = _captured(PLAIN_BUCKET) + assert records, "the ordinary disk sent no request, so this test would be vacuous" + for record in records: + assert record["headers"].get("authorization", "").startswith("Bearer "), record + assert "x-goog-if-generation-match" not in record["headers"], record["query"] + assert "if-match" not in record["headers"], record["query"] + assert "if-none-match" not in record["headers"], record["query"] + assert "x-amz-copy-source" not in record["headers"], record["query"] + assert "x-goog-copy-source" not in record["headers"], record["query"] + assert not _has_goog_metadata(record), record["query"] + + observable = [r for r in records if r["method"] in _MARKING_OBSERVABLE_METHODS] + assert observable, "no GET/HEAD/DELETE on the ordinary disk, so the check below would be vacuous" + for record in observable: + assert _looks_default_on_oauth(record), ( + "an ordinary {} on {} lost x-amz-api-version, so it was marked".format( + record["method"], record["key"] or "(list)" + ) + ) + + +def test_ordinary_goog4_traffic_keeps_upstream_semantics(): + """Exercise ordinary GOOG4 read/write/list/delete/multipart forms independently of CAS.""" + node = cluster.instances["node"] + assert ( + node.query( + "SELECT count() FROM system.disks WHERE name = '{}'".format(PLAIN_HMAC_DISK) + ).strip() + == "1" + ), "the ordinary GOOG4 disk is absent" + + table = "task9_plain_gcs_hmac_s3" + multipart_table = "task9_plain_gcs_hmac_multipart" + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query("DROP TABLE IF EXISTS {} SYNC".format(multipart_table)) + first_seq = _next_seq() + node.query( + "CREATE TABLE {} (line String) ENGINE = S3(plain_gcs_hmac_conn, " + "filename='ordinary.txt', format='LineAsString')".format(table) + ) + node.query("INSERT INTO {} VALUES ('goog4')".format(table)) + assert node.query("SELECT * FROM {}".format(table)) == "goog4\n" + ordinary_etag = node.query( + "SELECT _etag FROM s3(plain_gcs_hmac_conn, filename='ordinary.txt', " + "format='LineAsString') LIMIT 1" + ).strip() + assert ordinary_etag and not ordinary_etag.isdigit(), ordinary_etag + assert ( + node.query( + "SELECT * FROM s3(plain_gcs_hmac_conn, filename='ordinary*.txt', " + "format='LineAsString')" + ) + == "goog4\n" + ) + node.query("TRUNCATE TABLE {}".format(table)) + + node.query( + "CREATE TABLE {} (line String) ENGINE = S3(plain_gcs_hmac_conn, " + "filename='multipart.txt', format='LineAsString')".format(multipart_table) + ) + node.query( + "INSERT INTO {} SELECT repeat('m', 512 * 1024)".format(multipart_table), + settings={"s3_max_single_part_upload_size": 0, "s3_min_upload_part_size": 65536}, + ) + node.query("TRUNCATE TABLE {}".format(multipart_table)) + node.query("DROP TABLE {} SYNC".format(table)) + node.query("DROP TABLE {} SYNC".format(multipart_table)) + + records = _captured_since(first_seq, PLAIN_HMAC_BUCKET) + assert records, "the ordinary GOOG4 workload sent no request" + for record in records: + headers = record["headers"] + assert record["request_class"] == "ordinary_non_cas", record + assert headers.get("authorization", "").startswith("GOOG4-HMAC-SHA256 "), record + assert "x-goog-if-generation-match" not in headers, record + assert "if-match" not in headers, record + assert "if-none-match" not in headers, record + assert "x-amz-copy-source" not in headers, record + assert "x-goog-copy-source" not in headers, record + assert not _has_goog_metadata(record), record + + assert [ + r + for r in records + if r["method"] == "HEAD" and r["key"] == "ordinary/ordinary.txt" + ], "ordinary GOOG4 issued no HEAD for ordinary.txt" + assert [r for r in records if r["method"] == "GET" and r["key"]], ( + "ordinary GOOG4 issued no object GET" + ) + assert [r for r in records if r["method"] == "GET" and "list-type=2" in r["query"]], ( + "ordinary GOOG4 issued no ListObjectsV2" + ) + assert [r for r in records if r["method"] == "PUT"], "ordinary GOOG4 issued no PUT" + assert [r for r in records if r["method"] == "DELETE"], "ordinary GOOG4 issued no DELETE" + multipart = [ + r + for r in records + if r["operation"] + in ("blob_multipart_create", "blob_multipart_part", "blob_multipart_complete") + ] + assert [r for r in multipart if r["operation"] == "blob_multipart_create"], multipart + assert [r for r in multipart if r["operation"] == "blob_multipart_part"], multipart + assert [r for r in multipart if r["operation"] == "blob_multipart_complete"], multipart + assert all(not _is_translated(r) for r in multipart), multipart + + +def test_marked_and_default_heads_coexist_on_one_oauth_client(): + """Per-request marking, observed on ONE client, for ONE method, in ONE bucket. + + The CAS `gcp_oauth` disk owns a single S3 client and issues HEADs of both kinds: CAS metadata + reads are marked, while `probeSentinelRaw` deliberately goes through the ordinary throwing + `getObjectMetadata` because it must tell no-such-key from no-such-bucket from a transient failure, + and it discards the metadata anyway. Marking deletes `x-amz-api-version`, so the two kinds are + distinguishable on the wire even though they are the same verb on the same key space. + + This is the assertion I earlier reported the fixture could not make. I was wrong for a specific + reason worth keeping: marking adds no header, which is true, but it REMOVES one, and an absence is + just as observable as a presence. + + Would fail if: every request were marked (the `Default` HEAD would lose the header) or none were + (all the marked HEADs would keep it). Both directions fire, which is what makes it a partition + rather than a one-sided check. + + Only the OAuth disk can support this. On `gcs_hmac`, + `prepareGcsRequestForGoog4Authentication` runs for every request the client sends, so the header + is absent regardless of mode and carries no information. + """ + heads = [r for r in _captured(CAS_DISKS["cas_gcs_oauth"]) if r["method"] == "HEAD"] + assert heads, "no HEAD reached the fake, so this test would be vacuous" + + default_heads = [r for r in heads if _looks_default_on_oauth(r)] + marked_heads = [r for r in heads if not _looks_default_on_oauth(r)] + + assert marked_heads, "no HEAD was marked — CAS metadata reads lost their request mode" + assert default_heads, ( + "every HEAD was marked — the sentinel probe's ordinary metadata read was marked too, " + "which is the whole-client marking regression this plan removes" + ) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_translated_requests_are_confined_to_conditional_operations(disk): + """Translated headers appear only where a precondition or custom metadata was actually sent. + + Weaker than the test above and deliberately kept for both disks, since it is the only isolation + statement available on `gcs_hmac`: a request carrying no CAS precondition must carry no + `x-goog-if-generation-match`, so a blanket translation would show up here as a generation + precondition on a plain read. + + Would fail if: the dialect began emitting generation preconditions for requests that carried no + ETag precondition — for instance by defaulting a missing precondition to `0`. + """ + records = _captured(CAS_DISKS[disk]) + translated = [r for r in records if _is_translated(r)] + assert translated, "nothing was translated at all, so this test would be vacuous" + + for record in records: + if record["method"] == "GET" and not record["key"]: + assert not _is_translated(record), "a LIST carried a translated header" + + # Every translated request is a mutable conditional PUT or an exact DELETE. Blob publication, + # including native staged copy, is now deliberately Default and absent from this set. + for record in translated: + assert record["method"] in ("PUT", "DELETE", "POST"), ( + "a {} on {} carried a translated header but is not a mutation".format( + record["method"], record["key"] or "(list)" + ) + ) + + +@pytest.mark.parametrize("disk", sorted(CAS_DISKS)) +def test_the_fake_refused_nothing_it_had_to_serve(disk): + """A `501 NotImplemented` from the fake means the mount needed an operation the fake refuses. + + That is a fixture gap, not a product bug, and it must not hide behind a passing suite. Would fail + if: a CAS path started using versioning or another operation the deterministic fixture does not + model. Multipart blob publication is modelled and classified explicitly. + """ + refused = [r for r in _captured(CAS_DISKS[disk]) if r["status"] == 501] + assert not refused, "the fake refused operations it was asked for: {}".format( + [(r["method"], r["key"], r["query"]) for r in refused] + ) + + +# --------------------------------------------------------------------------------------------------- +# Adversarial coverage. Everything below runs after the tests above on purpose: some of these restart +# the server or drive the fake into a mode no correct client provokes, and the assertions above read +# the whole capture log. +# --------------------------------------------------------------------------------------------------- + +# A bucket no disk is configured against, so requests this file issues by hand are invisible to every +# per-bucket assertion above. +PROBE_BUCKET = "probebucket" + + +def test_the_fake_refuses_a_keyless_write_and_a_bucket_level_object_subresource(): + """Two request shapes that must not be served half-way. + + A keyless `PUT /bucket` is `CreateBucket`, which this fake does not model; served as a generic + object write it would mint a phantom object at the empty key that then appears in every later + listing of the bucket. `GET /bucket?tagging` is a bucket-level address for an object-level + subresource; answered by the bare-listing shortcut it would return a full object listing to a + caller that asked for a tag set. + + Negative control on the fixture, named as such: no production change flips this. It is asserted + because both shapes would corrupt the capture log the tests above read, silently and in a way that + reads as a ClickHouse bug. + """ + assert _raw("PUT", "/{}".format(PROBE_BUCKET)) == 501 + assert _raw("GET", "/{}?tagging".format(PROBE_BUCKET)) == 501 + assert _raw("DELETE", "/{}".format(PROBE_BUCKET)) == 501 + + listing = [ + r + for r in _captured(PROBE_BUCKET) + if r["method"] == "GET" and not r["key"] and r["status"] == 200 + ] + assert not listing, "a bucket-level subresource was served as a listing" + + +def test_a_generation_in_the_etag_domain_is_caught_by_the_fake_but_only_in_its_strict_mode(): + """What each kind of real service would do with the request shape the design calls unsafe. + + The design refuses to assume whether GCS rejects, compares or ignores a numeric generation placed + in an ETag-valued `If-Match`. This drives that shape by hand, under both of the fake's modes, and + reads the two answers off the wire: + + - `reject` (the default): `400`, and the object survives. The mistake is loud. + - `ignore`: `204`, and the object is GONE. The caller is told its exact delete succeeded when + nothing was ever compared, which is data loss with no error anywhere. + + The conclusion is what makes this worth having, so state it rather than leave it implied: because + a permissive service answers the unsafe shape with success, the fixture's safety CANNOT rest on + the service's answer. `test_no_cas_request_sends_a_generation_as_an_etag` — which inspects the + header CAS actually sent, whatever the service did with it — is the load-bearing fence, and this + test is why. + + Negative control on the fixture: nothing in ClickHouse flips it. Production never sends this + shape, which is exactly why it has to be driven by hand to be observed at all. + """ + key = "cas-token-in-etag-domain" + path = "/{}/{}".format(PROBE_BUCKET, key) + + try: + for mode, expected_status, expected_after in ( + ("reject", 400, 200), + ("ignore", 204, 404), + ): + assert _set_if_match_mode(mode)["if_match"] == mode + assert _raw("PUT", path) == 200 + generation = _captured(PROBE_BUCKET)[-1]["response_generation"] + assert generation and generation.isdigit(), generation + + assert ( + _raw("DELETE", path, [("If-Match", generation)]) == expected_status + ), "mode {} answered the unsafe shape unexpectedly".format(mode) + assert _raw("HEAD", path) == expected_after, ( + "mode {}: the object's survival does not match the delete's answer".format(mode) + ) + finally: + assert _set_if_match_mode("reject")["if_match"] == "reject" + + +def test_a_permissive_service_does_not_change_what_cas_puts_on_the_wire(): + """Remount the whole node against the permissive service and re-read every disk. + + The point is that correctness here is a property of the client, not of the store: under `ignore` + the fake compares nothing when a generation arrives in the ETag domain, so a client that had lost + its native mark would sail through the capability battery and mount successfully. The mount below + still passes for the opposite reason — CAS never sends that shape at all — and the delta assertion + is what says so. + + Would fail if: any CAS conditional operation lost its request mode. The precondition would move + to `If-Match`, the permissive fake would swallow it, and `numeric_if_match` would be non-zero for + a run in which every table still read back correctly. That is precisely the regression a strict + fake would have masked as a loud mount failure and this one catches as a silent one. + """ + node = cluster.instances["node"] + tables = ["t_" + disk for disk in list(CAS_DISKS) + [PLAIN_DISK, PLAIN_HMAC_DISK]] + # Read the counts before the remount rather than comparing against NUM_ROWS: later tests in this + # file insert more rows, and a constant here would make this test's correctness depend on where it + # sits in the file. + before_counts = {t: int(node.query("SELECT count() FROM {}".format(t))) for t in tables} + before_sums = {t: int(node.query("SELECT sum(id) FROM {}".format(t))) for t in tables} + assert all(count > 0 for count in before_counts.values()), before_counts + + before_numeric = _counters().get("numeric_if_match", 0) + first_new_seq = _next_seq() + try: + assert _set_if_match_mode("ignore")["if_match"] == "ignore" + node.restart_clickhouse() + + for table in tables: + assert int(node.query("SELECT count() FROM {}".format(table))) == before_counts[table] + assert int(node.query("SELECT sum(id) FROM {}".format(table))) == before_sums[table] + finally: + assert _set_if_match_mode("reject")["if_match"] == "reject" + node.restart_clickhouse() + + # The remount must actually have reached the store, or every assertion below is vacuous. A fresh + # mount runs the capability battery, so its generation preconditions are the strongest available + # evidence that this is a new mount's traffic and not a replay of the log read above. + remounted = _captured_since(first_new_seq) + assert remounted, "the restart produced no request at all" + for bucket in CAS_DISKS.values(): + fresh = _captured_since(first_new_seq, bucket) + assert _generation_preconditions(fresh), ( + "no generation precondition after the remount of {}, so the capability battery did not " + "run and this test proves nothing".format(bucket) + ) + + assert _counters().get("numeric_if_match", 0) == before_numeric, ( + "a request put a numeric value in the ETag domain while the fake was permissive enough to " + "accept it" + ) + + +def test_native_conditional_writes_seen_so_far_are_single_part(): + """Multipart permission is confined to Default blob-body publication. + + This prefix check localises an accidental multipart mutable write before the adversarial restart + tests. The run-wide version remains last in the module. + """ + multipart = [ + r + for r in _captured() + if r["bucket"] in CAS_DISKS.values() + if r["operation"] + in ("blob_multipart_create", "blob_multipart_part", "blob_multipart_complete") + ] + assert all(r["request_class"] == "blob_body" for r in multipart), multipart + assert all(not _is_translated(r) for r in multipart), multipart + + +def test_interleaved_ordinary_and_cas_operations_do_not_leak_mode_or_build_a_client(): + """Two statements over one interleaved workload on the OAuth clients. + + Mode isolation: the ordinary disk's requests must still carry `x-amz-api-version` (marking deletes + it) and the CAS disk's traffic must still be marked, in a slice of the log where the two disks' + traffic is interleaved rather than separated by phase. The contribution here is the INTERLEAVING; + that a single OAuth client carries both marked and unmarked requests is established separately by + `test_marked_and_default_heads_coexist_on_one_oauth_client`, which asserts both halves non-empty in + one bucket. This test asserts only the marked half, deliberately -- duplicating the partition would + add a second place to keep in step and no new fencing power. Would fail if: the mode became a + property of the client rather than of the request. + + Client count: the metadata server hands out a token when a client's cache is first populated, and + answers a 24-hour expiry, so within this slice a new token fetch means a new client with a new + token cache. Zero new fetches says the request mode built neither. Would fail if: selecting the + request mode constructed a third client — the base and single-attempt clients already exist by + this point, having been built during the mount and the first conditional write. + + The `> 0` preconditions matter three times here. Without the store-traffic check the mode + assertions would hold over an empty slice; without the per-method check they would hold over a + slice containing only writes; and without the lifetime-total token check the `== 0` would hold on + a fixture whose metadata server was never reached at all. + + The read has to be one that MUST reach object storage, and the first version of this test got that + wrong: it used `SELECT count()`, which is answered from part metadata and never fetched a column, + so the ordinary disk's slice held nothing but PUTs and the per-method precondition below caught it. + A column read of the rows just inserted, with the mark and uncompressed caches dropped first, is + what actually issues a GET. + """ + node = cluster.instances["node"] + assert _token_fetches() > 0, ( + "the metadata server was never asked for a token, so counting new fetches proves nothing" + ) + + first_new_seq = _next_seq() + _reset_token_fetches() + + for round_index in range(3): + base = 10000 + round_index * 100 + for disk in (PLAIN_DISK, "cas_gcs_oauth"): + table = "t_" + disk + node.query( + "INSERT INTO {} SELECT number, toString(number) FROM numbers({}, 10)".format( + table, base + ) + ) + # Drop the caches that would otherwise answer the read from memory, then read a COLUMN of + # the rows just written rather than a count. + node.query("SYSTEM DROP MARK CACHE") + node.query("SYSTEM DROP UNCOMPRESSED CACHE") + assert ( + int(node.query("SELECT sum(id) FROM {} WHERE id >= {}".format(table, base))) + >= base + ) + + cas_bucket = CAS_DISKS["cas_gcs_oauth"] + plain = _captured_since(first_new_seq, PLAIN_BUCKET) + cas = _captured_since(first_new_seq, cas_bucket) + assert plain, "the ordinary disk sent nothing in this slice" + assert cas, "the CAS disk sent nothing in this slice" + + # Named explicitly rather than folded into the `observable_plain` check, so that a workload which + # stops reaching the store says WHICH method vanished instead of just going quiet. Only GET is + # required: marking is a per-request property, so one observable request is enough to fence it, and + # forcing a DELETE would mean waiting on part-removal timing. + assert [r for r in plain if r["method"] == "GET" and r["key"]], ( + "the ordinary disk issued no object GET in this slice, so the read never reached the store" + ) + + observable_plain = [r for r in plain if r["method"] in _MARKING_OBSERVABLE_METHODS] + assert observable_plain, "no GET/HEAD/DELETE on the ordinary disk in this slice" + for record in observable_plain: + assert _looks_default_on_oauth(record), ( + "an interleaved ordinary {} on {} was marked".format( + record["method"], record["key"] or "(list)" + ) + ) + + observable_cas = [r for r in cas if r["method"] in _MARKING_OBSERVABLE_METHODS] + assert [r for r in observable_cas if not _looks_default_on_oauth(r)], ( + "no CAS request in this slice was marked" + ) + + new_fetches = _token_fetches() + assert new_fetches == 0, ( + "an interleaved workload fetched {} new metadata token(s), so it built a client with a new " + "token cache".format(new_fetches) + ) + + +def test_a_write_whose_response_carries_no_generation_is_refused(): + """The one input that can reach the "no valid generation" refusal. + + A real GCS always answers a successful object write with `x-goog-generation`, and the response + adapter turns that into the SDK's `ETag`. When it is absent the SDK sees the store's real ETag + instead, which is not a generation, so `tokenFromWriteResult` must refuse to attribute the write to + an incarnation rather than patching the missing token over with a fresh HEAD — a HEAD returns + whatever incarnation happens to be current, which on a lost race is somebody else's. + + The error text is the whole discriminator, and it is tight: had the code HEADed and adopted the + current incarnation instead of refusing, the INSERT would have SUCCEEDED. It failed, naming the + missing generation. So a regression that replaced the strict branch with a HEAD-and-adopt fallback + turns the error assertion red on its own. + + Do NOT add an assertion here about which requests follow that write. The remaining conditional + metadata/control lane may classify an unattributed attempt as unresolved and call + `resolveByExactGet`, while the globally enabled injection can be consumed by more than one object + kind. Blob-body publication is no longer part of this test: it is unconditional, consumes no + response generation, and therefore cannot be the source of this refusal. + + What fences the behaviour is the error text above, and nothing else here needs to. + + Would fail if: the strict Generation branch in `tokenFromWriteResult` were replaced by, or fell + back to, the ETag dialect's HEAD path. + + The mode is global while it is on, so a background CAS operation on the other disk can fail during + the window too. That is logged, not fatal, and the restored-mode INSERT at the end is what says + the disk is healthy again. + """ + node = cluster.instances["node"] + table = "t_cas_gcs_oauth" + cas_bucket = CAS_DISKS["cas_gcs_oauth"] + + first_new_seq = _next_seq() + try: + assert _set_omit_generation(True)["omit_generation"] is True + error = node.query_and_get_error( + "INSERT INTO {} SELECT number, toString(number) FROM numbers(20000, 50)".format(table) + ) + finally: + assert _set_omit_generation(False)["omit_generation"] is False + + assert "carried no valid generation" in error, error + + # Positive proof that the fake actually produced the condition under test: a successful object + # write really did answer without a generation. Without this the error assertion above could be + # satisfied by an INSERT that failed for some entirely unrelated reason, and a mode switch that + # silently stopped working would look like a pass. + ungenerated = [ + r + for r in _captured_since(first_new_seq, cas_bucket) + if r["method"] == "PUT" and r["status"] == 200 and r["response_generation"] is None + ] + assert ungenerated, "the mode was on but no successful PUT answered without a generation" + + # Restoring the mode must restore the disk, or the failure above was something other than the + # missing generation. + node.query("INSERT INTO {} SELECT number, toString(number) FROM numbers(30000, 50)".format(table)) + assert int(node.query("SELECT count() FROM {} WHERE id >= 30000".format(table))) == 50 + + +# --------------------------------------------------------------------------------------------------- +def test_a_reload_that_would_flip_the_token_dialect_is_refused(): + """A live CAS mount must keep the incarnation-token dialect it was opened with. + + The pool derives persistent state from that dialect -- how a token value is normalised, whether a + listing may supply one at all, and which preconditions the mount had to satisfy -- so a reload that + swapped the client for one minting the other kind would leave persisted tokens uncomparable. + + WHAT THIS TEST DOES NOT PROVE, stated because the obvious reading is wrong. It flips the DISK-LEVEL + `http_client`, which a guard reading only the disk section would also have refused. So it does not + discriminate where the check lives; it only shows that a flip is refused and that the old client + survives. The placement is what actually matters -- the effective value is merged from the storage's + current settings, any endpoint-level block and the disk section, so a disk-section-only check misses + a flip arriving from an endpoint block and falsely refuses a no-op reload whenever the effective + value comes from elsewhere -- and that property is covered by reading the code, not by this test. + Writing the discriminating version needs a CAS mount pinned to ETag, which this fixture cannot host: + the fake mints numeric ETags, so an ETag-dialect mount sends a numeric `If-Match` and the fake's own + domain check rejects it as a generation reaching the ETag domain. Teaching it a second ETag shape is + the prerequisite, and is deliberately not done here. + + Asserting the refusal is not enough on its own, because "the reload was refused" and "the reload was + refused AND the old client survived" are different claims and only the second is the guarantee. So + the test also shows the mount still speaks generation afterwards: a fresh conditional write still + carries a numeric precondition, which only a generation-dialect client sends. + + Would fail if: the pin were not installed at startup, or were checked after the client had already + been replaced. It would NOT fail if the pin read a single config section, which is the gap above. + """ + node = cluster.instances["node"] + table = "t_cas_gcs_oauth" + cas_bucket = CAS_DISKS["cas_gcs_oauth"] + + + try: + # An explicit non-GCS value, not a removed key. Settings merge through `updateIfChanged`, which + # applies only values the incoming config actually SET, so deleting `http_client` leaves the old + # one in force and flips nothing -- the first version of this test deleted it and the guard + # correctly stayed silent. No validation rejects an unrecognised value; it simply selects the + # ordinary client, which is an ETag store. + node.replace_in_config( + CONFIG_IN_CONTAINER, + "gcp_oauth", + "none", + ) + try: + reload_error = node.query_and_get_error_with_retry( + "SYSTEM RELOAD CONFIG", retry_count=1, sleep_time=0 + ) + except Exception: + reload_error = "" + + # Whether the refusal reaches the client or only the log depends on how config reload reports a + # failing disk, so accept either -- but require one of them, and require the specific reason rather + # than any failure. + logged = node.grep_in_log("cannot change its conditional-operation dialect on reload") + assert "conditional-operation dialect" in reload_error or logged, ( + "the reload was neither refused to the client nor recorded as refused in the log; " + "error was {!r}".format(reload_error) + ) + + # The guarantee: the old client survived, so this mount still speaks the generation dialect. The + # evidence is a generation PRECONDITION on the wire, which only a generation-dialect client sends. + # Not the `numeric_if_match` counter -- that one counts `If-Match` (exact-token) requests, and an + # INSERT sends `x-goog-if-generation-match` for create-if-absent instead, so the counter would have + # stayed flat here for a reason that has nothing to do with the dialect. + after_reload_seq = _next_seq() + node.query( + "INSERT INTO {} SELECT number, toString(number) FROM numbers(30000, 20)".format(table) + ) + post_reload = _captured_since(after_reload_seq, cas_bucket) + assert post_reload, "the INSERT after the refused reload reached the store not at all" + conditional = [r for r in post_reload if "x-goog-if-generation-match" in r["headers"]] + assert conditional, ( + "no generation precondition was sent after the refused reload, so the mount is no longer " + "speaking the generation dialect it was opened with" + ) + finally: + # Always restore, even on a failed assertion: leaving the disk configured for the other + # dialect would break every test that runs after this one. + node.replace_in_config( + CONFIG_IN_CONTAINER, + "none", + "gcp_oauth", + ) + node.query("SYSTEM RELOAD CONFIG") + node.query( + "INSERT INTO {} SELECT number, toString(number) FROM numbers(31000, 20)".format(table) + ) + + +# MUST STAY LAST IN THIS FILE. The fake's capture log is global and cumulative and nothing in this +# module resets it, so this assertion covers exactly the traffic that precedes it. +# Add new tests ABOVE this line. +# --------------------------------------------------------------------------------------------------- + + +def test_multipart_remained_confined_to_default_blob_publication_during_the_whole_run(): + """Run-wide classification fence for multipart and conditional isolation. + + At least one blob completion must exist, while every multipart request must name a blob body and + carry no translated conditional header. This replaces the stale run-wide prohibition from the + conditional-blob design. + """ + records = [r for r in _captured() if r["bucket"] in CAS_DISKS.values()] + multipart = [ + r + for r in records + if r["operation"] + in ("blob_multipart_create", "blob_multipart_part", "blob_multipart_complete") + ] + assert [r for r in multipart if r["operation"] == "blob_multipart_complete"], multipart + assert all(r["request_class"] == "blob_body" for r in multipart), multipart + assert all(not _is_translated(r) for r in multipart), multipart diff --git a/tests/integration/test_cas_insert_fault_recovery/__init__.py b/tests/integration/test_cas_insert_fault_recovery/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node1.xml b/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node1.xml new file mode 100644 index 000000000000..304a1cf626a7 --- /dev/null +++ b/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node1.xml @@ -0,0 +1,10 @@ + + + + + + node1 + + + + diff --git a/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node2.xml b/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node2.xml new file mode 100644 index 000000000000..574cfa176cc1 --- /dev/null +++ b/tests/integration/test_cas_insert_fault_recovery/configs/server_root_id_node2.xml @@ -0,0 +1,10 @@ + + + + + + node2 + + + + diff --git a/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml b/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml new file mode 100644 index 000000000000..0149d398aa1c --- /dev/null +++ b/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml @@ -0,0 +1,27 @@ + + + + + object_storage + s3 + cas + + http://rustfs1:11121/test/shared_pool/ + clickhouse + clickhouse + + + + + + +
+ disk_cas_shared +
+
+
+
+
+
diff --git a/tests/integration/test_cas_insert_fault_recovery/test.py b/tests/integration/test_cas_insert_fault_recovery/test.py new file mode 100644 index 000000000000..e1f3ff4ee23a --- /dev/null +++ b/tests/integration/test_cas_insert_fault_recovery/test.py @@ -0,0 +1,163 @@ +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +# Two replicas of one ReplicatedMergeTree on a SHARED content-addressed pool. +STORAGE_POLICY = "cas_shared" + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node1", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node1.xml"], + macros={"replica": "node1"}, + with_rustfs=True, + with_zookeeper=True, + stay_alive=True, + ) + cluster.add_instance( + "node2", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node2.xml"], + macros={"replica": "node2"}, + with_rustfs=True, + with_zookeeper=True, + stay_alive=True, + ) + + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def _wait_until(predicate, timeout=180, interval=2, desc=""): + # Condition-based wait (systematic-debugging): the ordinary lost-part recovery is asynchronous + # (part-check retry/backoff), so gating on a fixed-timeout `SYSTEM SYNC REPLICA` is inherently flaky — + # that call blocks on the very recovery we are waiting for. Poll the actual OUTCOME instead, with a + # generous cap. Transient errors while node1 is mid-restart are swallowed and retried. + deadline = time.time() + timeout + last = None + while time.time() < deadline: + try: + last = predicate() + except Exception as e: # node briefly unavailable during restart, etc. + last = e + if last is True: + return + time.sleep(interval) + raise AssertionError("timed out after {}s waiting for: {} (last={!r})".format(timeout, desc, last)) + + +def test_post_multi_termination_uses_ordinary_lost_part_recovery(start_cluster): + # HISTORY: this test was authored (2026-07-16) against the OLD commit ordering, where the disk + # commit ran AFTER the Keeper multi — the failpoint then left a phantom ZK part entry and the + # assertion was "ordinary lost-part recovery runs (ReplicatedDataLoss bumps, empty cover)". + # One day later the R3 acked-data-loss fix (`77484196b0d`) deliberately REVERSED that order: + # `renameParts` closes the part's disk-storage transaction BEFORE the Keeper multi, so a part + # must be durable before its block_id/part znode is registered. Under the new ordering the + # failpoint (`disk_object_storage_fail_commit_metadata_transaction`, fired from inside + # `renameParts`) aborts the INSERT BEFORE anything reaches ZK — there is no phantom part, no + # lost part, and NOTHING to recover. The old predicate waited forever (600s timeouts on all + # three sanitizer CI lanes of PR#2073 and on a local release build). + # + # The test now asserts the NEW invariant, which is strictly stronger for the user: + # 1. the failed INSERT leaves NO trace: no ZK part entry, no replication-queue debris, + # count() stays 0 on both replicas after a node1 restart, and `ReplicatedDataLoss` does + # NOT bump (nothing was ever lost); + # 2. THE R3 GUARD: retrying the SAME insert (same bytes => same block_id) actually lands — + # a phantom block_id surviving the failed attempt would silently dedup the retry away + # (the acked-data-loss class the reordering exists to prevent); + # 3. no CA-specific wedge: no LOGICAL_ERROR in either server's log, queues drained. + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + + node1.query("DROP TABLE IF EXISTS t SYNC") + node2.query("DROP TABLE IF EXISTS t SYNC") + + create = ( + "CREATE TABLE t (a UInt64) ENGINE = ReplicatedMergeTree('/clickhouse/tables/t', '{{replica}}') " + "ORDER BY a SETTINGS storage_policy = '{policy}'" + ).format(policy=STORAGE_POLICY) + node1.query(create) + node2.query(create) + + def loss_count(): + return int( + node1.query( + "SELECT sum(value) FROM system.events WHERE event = 'ReplicatedDataLoss'" + ) + or 0 + ) + + loss_before = loss_count() + + # Force the disk commit to throw. Under the R3 ordering this fires inside `renameParts`, + # BEFORE the Keeper multi — the INSERT fails with nothing registered anywhere (ONCE failpoint). + node1.query("SYSTEM ENABLE FAILPOINT disk_object_storage_fail_commit_metadata_transaction") + node1.query_and_get_error("INSERT INTO t VALUES (1)") + + # No phantom state may exist even across a restart: ZK has no part entry, so startup's + # `checkPartsImpl` has nothing to reconcile and no recovery runs. + node1.restart_clickhouse() + + def node1_clean(): + # The failed INSERT left no trace: nothing to recover (ReplicatedDataLoss unchanged), + # no rows, no replication-queue debris. + cnt = node1.query("SELECT count() FROM t").strip() + queue = node1.query( + "SELECT count() FROM system.replication_queue WHERE table = 't'" + ).strip() + return loss_count() == loss_before and cnt == "0" and queue == "0" + + _wait_until( + node1_clean, + timeout=120, + desc="node1 restarts clean: no phantom part, no recovery triggered, queue empty", + ) + + # THE R3 GUARD (acked-data-loss class): retrying the SAME insert (same bytes => same block_id) + # must genuinely land. If the failed attempt had leaked its block_id into ZK, dedup would + # silently swallow this retry and count() would stay 0 — exactly the silent loss the + # renameParts-before-Keeper ordering exists to prevent. + node1.query("INSERT INTO t VALUES (1)") + + def retry_landed_everywhere(): + return ( + node1.query("SELECT count() FROM t").strip() == "1" + and node2.query("SELECT count() FROM t").strip() == "1" + ) + + _wait_until( + retry_landed_everywhere, + timeout=120, + desc="the retried identical INSERT lands and replicates (no phantom-block_id dedup)", + ) + + # The regression guard: no CA-specific exception / LOGICAL_ERROR left either server wedged. The + # expected `FILE_DOESNT_EXIST` interserver miss is tolerated (it is not a LOGICAL_ERROR). + for node in (node1, node2): + assert not node.contains_in_log( + "LOGICAL_ERROR" + ), "unexpected LOGICAL_ERROR in {}'s log — a CA-specific failure, not ordinary lost-part recovery".format( + node.name + ) + + # Server is healthy (no wedge): a fresh, different INSERT also succeeds end to end. + node1.query("INSERT INTO t VALUES (2)") + + def replicated_two_rows(): + return ( + node1.query("SELECT count() FROM t").strip() == "2" + and node2.query("SELECT count() FROM t").strip() == "2" + ) + + _wait_until(replicated_two_rows, timeout=120, desc="fresh INSERT replicates to both replicas") + + node1.query("DROP TABLE IF EXISTS t SYNC") + node2.query("DROP TABLE IF EXISTS t SYNC") diff --git a/tests/integration/test_cas_lazy_load_recovery/__init__.py b/tests/integration/test_cas_lazy_load_recovery/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_lazy_load_recovery/configs/server_root_id_node1.xml b/tests/integration/test_cas_lazy_load_recovery/configs/server_root_id_node1.xml new file mode 100644 index 000000000000..304a1cf626a7 --- /dev/null +++ b/tests/integration/test_cas_lazy_load_recovery/configs/server_root_id_node1.xml @@ -0,0 +1,10 @@ + + + + + + node1 + + + + diff --git a/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml b/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml new file mode 100644 index 000000000000..0149d398aa1c --- /dev/null +++ b/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml @@ -0,0 +1,27 @@ + + + + + object_storage + s3 + cas + + http://rustfs1:11121/test/shared_pool/ + clickhouse + clickhouse + + + + + + +
+ disk_cas_shared +
+
+
+
+
+
diff --git a/tests/integration/test_cas_lazy_load_recovery/test.py b/tests/integration/test_cas_lazy_load_recovery/test.py new file mode 100644 index 000000000000..3eeae3debc7d --- /dev/null +++ b/tests/integration/test_cas_lazy_load_recovery/test.py @@ -0,0 +1,88 @@ +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +STORAGE_POLICY = "cas_shared" + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node1", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node1.xml"], + macros={"replica": "node1"}, + with_rustfs=True, + with_zookeeper=True, + stay_alive=True, + ) + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def _create(node): + # lazy_load_tables=1: the CAS table attaches as a proxy and its real storage is built on first + # access. A transient object-store outage during that build is ridden out / retried on a later + # access instead of being cached as a permanently-FAILED AsyncLoader job (which, for a non-lazy + # database, would strand the table until a full server restart). + node.query("CREATE DATABASE IF NOT EXISTS lazy_db ENGINE = Atomic SETTINGS lazy_load_tables = 1") + node.query( + "CREATE TABLE IF NOT EXISTS lazy_db.t (k UInt64, v UInt64) " + "ENGINE = ReplicatedMergeTree('/clickhouse/tables/lazy_t', '{replica}') " + "ORDER BY k SETTINGS storage_policy = '%s', min_bytes_for_wide_part = 0" % STORAGE_POLICY + ) + + +def test_lazy_cas_table_self_heals_after_s3_recovery(start_cluster): + node = cluster.instances["node1"] + _create(node) + node.query("INSERT INTO lazy_db.t SELECT number, number FROM numbers(100)") + assert node.query("SELECT count() FROM lazy_db.t").strip() == "100" + + # Restart so the table re-attaches as a lazy proxy (its real storage is not yet constructed; the + # disk mounts at startup while S3 is up, the storage is built only on first access below). + node.restart_clickhouse() + + # Touch the table while S3 is unreachable: the lazy first-access build (its CAS ref-recovery LIST + # over the object store) cannot complete, so the client query fails within its bounded timeout. + # Note: the build does NOT fail fast server-side -- it blocks on the object store's own retry until + # S3 returns (see the BACKLOG "block-until-recovered" note); the client-side timeout is what makes + # this probe short. We assert the probe DID hit the outage (raised): the build needs several object- + # store round-trips, so the freezer (effective within milliseconds of `pause_container` returning) + # reliably catches it -- if this ever flakes, the pause raced a sub-millisecond full build, not a + # real self-heal regression. + with cluster.pause_container("rustfs1", wait_for_paused=False): + probe_raised = False + try: + node.query("SELECT count() FROM lazy_db.t", timeout=30) + except Exception: + probe_raised = True # expected while the object store is unreachable + assert probe_raised, "the probe should have failed while S3 was unreachable (did the pause race the build?)" + + # S3 is back (context exit unpaused rustfs). WITHOUT a server restart and WITHOUT any DETACH, a + # later access must make the table usable again. This proves the key Layer 2 property: a transient + # object-store outage during a lazy CAS table's first-access build leaves NO permanently-cached + # AsyncLoader FAILED state (a non-lazy table whose load failed would stay FAILED until a full server + # restart). What actually recovers here is the original in-flight build completing once S3 returns + # (the block-until-recovered path), which is sufficient for "usable again without restart"; this + # test does not (and, given block-until-recovered, cannot) assert a proxy retry of a THROWN build. + deadline = time.time() + 180 + last = None + while time.time() < deadline: + try: + last = node.query("SELECT count() FROM lazy_db.t").strip() + except Exception as e: + last = "err: " + str(e) + if last == "100": + break + time.sleep(3) + assert last == "100", ( + "lazy CAS table must become usable again on a later access after S3 returns, with no server " + "restart (last=%r)" % last + ) diff --git a/tests/integration/test_cas_mount_renewal_retry/__init__.py b/tests/integration/test_cas_mount_renewal_retry/__init__.py new file mode 100644 index 000000000000..8b137891791f --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/__init__.py @@ -0,0 +1 @@ + diff --git a/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml b/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml new file mode 100644 index 000000000000..997e9a217631 --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml @@ -0,0 +1,34 @@ + + + system +
cas_log
+ 100 + + + + + + object_storage + s3 + cas + itest-cas-renewal + + http://s3proxy:11121/test/cas_mount_renewal/ + clickhouse + clickhouse + false + + + + + +
+ disk_cas_renewal +
+
+
+
+
+ diff --git a/tests/integration/test_cas_mount_renewal_retry/docker_compose_proxy.yml b/tests/integration/test_cas_mount_renewal_retry/docker_compose_proxy.yml new file mode 100644 index 000000000000..67f39af36922 --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/docker_compose_proxy.yml @@ -0,0 +1,30 @@ +services: + s3proxy: + image: python:3.12-slim + command: ["python3", "/proxy/s3_fault_proxy.py"] + environment: + RUSTFS_UPSTREAM: "rustfs1:11121" + S3_PROXY_PORT: "11121" + S3_PROXY_CTL_PORT: "8474" + depends_on: + rustfs1: + condition: service_started + volumes: + # Compose resolves relative binds against the generated first compose file in + # `/_instances-*/node`, not against this appended file, so `../..` is the + # directory holding this file. + - ../../s3_fault_proxy.py:/proxy/s3_fault_proxy.py:ro + expose: + - "11121" + ports: + - "127.0.0.1::8474" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8474/healthz', timeout=2)"] + interval: 1s + timeout: 3s + retries: 30 + + node: + depends_on: + s3proxy: + condition: service_healthy diff --git a/tests/integration/test_cas_mount_renewal_retry/s3_fault_proxy.py b/tests/integration/test_cas_mount_renewal_retry/s3_fault_proxy.py new file mode 100644 index 000000000000..cf3102611a19 --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/s3_fault_proxy.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""S3 fault-injection / list-anomaly proxy for the CA scenario suite (S22, S27). + +Sits between ClickHouse and RustFS: ClickHouse's `ca` disk endpoint points at this proxy, which +forwards every request verbatim to the real RustFS upstream. SigV4 signs the `host` header, so the +proxy MUST preserve the client's Host header on the forwarded request; RustFS validates the signature +against the received Host, so forwarding it unchanged keeps auth valid while the TCP connection goes +to the upstream container. + +Two fault families, both DISARMED by default (rate 0) so cluster bring-up + the CA capability probe +are never disturbed. The scenario ARMS faults for the workload window via the control port, then +disarms before the checkpoint. + +- S22 (fault injection): with probability `rate`, a matched request gets a bounded transient fault: + `503 SlowDown` / `429 SlowDown` (S3-style retryable body), artificial latency (`slow`), or a + mid-response connection close (`reset`). Applied to GET/PUT/HEAD/POST/LIST per `methods`. +- S27 (list anomaly): for `LIST` (GET with `list-type=2`) whose `prefix` matches `list_prefix`, + rewrite the returned XML to inject a duplicate key, or drop the continuation token — so the CA GC + discovery/token-diff path must treat the page as ambiguous and re-read (never skip a fold). + +Focused tests can additionally restrict S22 faults to a `path_substring` and an atomic +`remaining_faults` budget. The `drop_after_forward` mode records the upstream result and request +body digest, then closes the downstream connection so retry recovery can prove response-loss cases. + +Control plane (separate port): POST /config {json}, GET /stats, GET /healthz. Deterministic: fault +decisions are driven by a seeded PRNG keyed per-request-index, so a given (seed, rate) is reproducible. +""" + +import hashlib +import http.client +import http.server +import json +import os +import random +import socket +import socketserver +import sys +import threading +import time + +UPSTREAM = os.environ.get("RUSTFS_UPSTREAM", "rustfs1:11121") # host:port of the real store +S3_PORT = int(os.environ.get("S3_PROXY_PORT", "11121")) +CTL_PORT = int(os.environ.get("S3_PROXY_CTL_PORT", "8474")) + +# Runtime-mutable fault config (guarded by _cfg_lock). rate=0 => pure pass-through. +_cfg_lock = threading.Lock() +_DEFAULT_CFG = { + "rate": 0.0, # fraction of matched requests that get a fault + "modes": ["503"], # subset of {503,429,slow,reset,drop_after_forward} + "methods": ["GET", "PUT", "HEAD", "POST"], # HTTP methods eligible for S22 faults + "slow_ms": 1500, # latency for the "slow" mode + "seed": 1, + "path_substring": None, # optional request-path scope + "remaining_faults": None, # optional exact finite fault budget; None = legacy unlimited + # S27 list-anomaly config (independent of the S22 fault rate): + "list_anomaly": None, # None | "duplicate" | "drop_token" + "list_prefix": "roots/", # only LIST calls whose prefix contains this are perturbed +} +_DEFAULT_STATS = { + "forwarded": 0, + "faults": 0, + "list_perturbed": 0, + "by_mode": {}, + "drop_after_forward": [], +} +_cfg = dict(_DEFAULT_CFG) +_stats = dict(_DEFAULT_STATS) +_stats["by_mode"] = {} +_stats["drop_after_forward"] = [] +_req_index = [0] +_idx_lock = threading.Lock() + + +def _get_cfg(): + with _cfg_lock: + return dict(_cfg) + + +def _next_index(): + with _idx_lock: + _req_index[0] += 1 + return _req_index[0] + + +def _bump(stat, key=None): + with _cfg_lock: + if key is None: + _stats[stat] = _stats.get(stat, 0) + 1 + else: + _stats[stat][key] = _stats[stat].get(key, 0) + 1 + + +def _record_drop_after_forward(method, path, body, upstream_status, upstream_etag): + record = { + "method": method, + "path": path, + "request_body_sha256": hashlib.sha256(body).hexdigest(), + "upstream_status": upstream_status, + "upstream_etag": upstream_etag, + } + with _cfg_lock: + records = _stats["drop_after_forward"] + records.append(record) + del records[:-64] + + +def _reset(): + with _cfg_lock: + _cfg.clear() + _cfg.update(_DEFAULT_CFG) + _stats.clear() + _stats.update(_DEFAULT_STATS) + _stats["by_mode"] = {} + _stats["drop_after_forward"] = [] + with _idx_lock: + _req_index[0] = 0 + + +_SLOWDOWN_BODY = (b'' + b'SlowDownPlease reduce your request rate.' + b'/fault-proxy') + + +class Handler(http.server.BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, *a): + pass # quiet + + # --- fault decision ----------------------------------------------------- + def _should_fault(self, cfg): + # Keep the original unscoped decision path byte-for-byte: existing scenario configs omit + # both new fields and therefore retain their request-index/seed behavior. + if cfg.get("path_substring") is None and cfg.get("remaining_faults") is None: + if cfg["rate"] <= 0 or self.command not in cfg["methods"]: + return None + idx = _next_index() + rng = random.Random(f"{cfg['seed']}:{idx}") + if rng.random() < cfg["rate"]: + return rng.choice(cfg["modes"]) if cfg["modes"] else None + return None + + # A scoped finite rule is one atomic decision. In particular, concurrent matching requests + # cannot all observe the same positive remaining count and over-consume the configured budget. + with _cfg_lock: + current = _cfg + if current["rate"] <= 0 or self.command not in current["methods"]: + return None + path_substring = current.get("path_substring") + if path_substring is not None and path_substring not in (self.path or ""): + return None + remaining = current.get("remaining_faults") + if remaining is not None and remaining <= 0: + return None + idx = _next_index() + rng = random.Random(f"{current['seed']}:{idx}") + if rng.random() >= current["rate"]: + return None + mode = rng.choice(current["modes"]) if current["modes"] else None + if mode is not None and remaining is not None: + current["remaining_faults"] = remaining - 1 + return mode + + def _emit_fault(self, mode): + _bump("faults") + _bump("by_mode", mode) + if mode in ("503", "429"): + code = 503 if mode == "503" else 429 + self.send_response(code) + self.send_header("Content-Type", "application/xml") + self.send_header("Content-Length", str(len(_SLOWDOWN_BODY))) + self.send_header("Connection", "keep-alive") + self.end_headers() + self.wfile.write(_SLOWDOWN_BODY) + elif mode == "slow": + time.sleep(_get_cfg()["slow_ms"] / 1000.0) + self._forward() # after the delay, serve the real response + elif mode == "reset": + # Abruptly close the connection with no valid response -> client sees a transport error. + try: + self.close_connection = True + self.connection.close() + except Exception: + pass + elif mode == "drop_after_forward": + self._forward(drop_after_forward=True) + + # --- request body ------------------------------------------------------- + def _read_body(self): + length = self.headers.get("Content-Length") + if length is not None: + return self.rfile.read(int(length)) + if self.headers.get("Transfer-Encoding", "").lower() == "chunked": + # De-chunk into a flat body (dev-scale payloads are small). + data = bytearray() + while True: + line = self.rfile.readline().strip() + if not line: + continue + size = int(line.split(b";")[0], 16) + if size == 0: + self.rfile.readline() # trailing CRLF + break + data += self.rfile.read(size) + self.rfile.readline() + return bytes(data) + return b"" + + # --- forward to upstream ------------------------------------------------ + def _forward(self, drop_after_forward=False): + body = getattr(self, "_cached_body", None) + if body is None: + body = self._read_body() + cfg = _get_cfg() + conn = http.client.HTTPConnection(UPSTREAM, timeout=60) + # Preserve headers verbatim (incl. Host, so SigV4 stays valid); strip hop-by-hop + Expect. + # Expect: 100-continue MUST be dropped: http.client sends the body immediately (no 100 wait), + # so relaying Expect makes the upstream reply with an interim 100 that getresponse() would + # misread — corrupting the upload (observed: size-0 blobs on >=64 KiB PUTs). We already + # buffered the full body and send it directly, so no 100-continue negotiation is needed. + _HOP = {"transfer-encoding", "expect", "connection", "keep-alive", "proxy-connection", + "te", "trailer", "upgrade"} + fwd_headers = {} + for k, v in self.headers.items(): + if k.lower() in _HOP: + continue + fwd_headers[k] = v + fwd_headers["Content-Length"] = str(len(body)) + try: + conn.request(self.command, self.path, body=body, headers=fwd_headers) + resp = conn.getresponse() + data = resp.read() + except Exception as e: + self.send_response(502) + msg = f"proxy upstream error: {e}".encode() + self.send_header("Content-Length", str(len(msg))) + self.end_headers() + self.wfile.write(msg) + conn.close() + return + # S27: perturb LIST XML if configured and this is a matching list call. + if (cfg.get("list_anomaly") and self.command == "GET" + and "list-type=2" in (self.path or "") and cfg["list_prefix"] in _decode_prefix(self.path)): + perturbed = _perturb_list_xml(data, cfg["list_anomaly"]) + if perturbed is not None: + data = perturbed + _bump("list_perturbed") + _bump("forwarded") + if os.environ.get("S3_PROXY_DEBUG") and self.command in ("HEAD", "POST"): + print(f"[dbg-resp] {self.command} {self.path[:50]} -> {resp.status} " + f"upstreamCL={resp.getheader('Content-Length')} bodylen={len(data)}", flush=True) + if drop_after_forward: + _record_drop_after_forward( + self.command, + self.path, + body, + resp.status, + resp.getheader("ETag") or "", + ) + conn.close() + self.close_connection = True + try: + self.connection.shutdown(socket.SHUT_RDWR) + except OSError: + pass + self.connection.close() + return + self.send_response(resp.status) + is_head = (self.command == "HEAD") + for k, v in resp.getheaders(): + # For HEAD, rustfs returns the object's real Content-Length with an EMPTY body — preserve + # it verbatim (the CA dedup probe reads it). For methods with a body, we resend the actual + # byte count below. Always drop hop-by-hop framing headers. + if k.lower() in ("transfer-encoding", "connection"): + continue + if k.lower() == "content-length" and not is_head: + continue + self.send_header(k, v) + if not is_head: + self.send_header("Content-Length", str(len(data))) + self.end_headers() + if not is_head: + self.wfile.write(data) + conn.close() + + def _handle(self): + cfg = _get_cfg() + # Cache body once (fault paths + forward both may need it). + try: + self._cached_body = self._read_body() + except Exception: + self._cached_body = b"" + if os.environ.get("S3_PROXY_DEBUG"): + print(f"[dbg] {self.command} {self.path[:60]} CL={self.headers.get('Content-Length')} " + f"TE={self.headers.get('Transfer-Encoding')} CE={self.headers.get('Content-Encoding')} " + f"Expect={self.headers.get('Expect')} read={len(self._cached_body)}", flush=True) + mode = self._should_fault(cfg) + if mode is None: + self._forward() + else: + self._emit_fault(mode) + + do_GET = _handle + do_PUT = _handle + do_POST = _handle + do_HEAD = _handle + do_DELETE = _handle + + +def _decode_prefix(path): + # extract the `prefix=` query value (URL-encoded); good enough to match "roots/" + import urllib.parse + q = urllib.parse.urlparse(path).query + params = urllib.parse.parse_qs(q) + return urllib.parse.unquote(params.get("prefix", [""])[0]) + + +def _perturb_list_xml(xml_bytes, anomaly): + """Inject a LIST-page anomaly. 'duplicate' repeats the first key; 'drop_token' removes + the continuation token so the client cannot prove it saw the whole listing. Returns perturbed + bytes, or None if there was nothing to perturb (caller keeps the original).""" + try: + s = xml_bytes.decode("utf-8", "replace") + except Exception: + return None + if anomaly == "duplicate": + i = s.find("") + j = s.find("") + if i == -1 or j == -1: + return None + block = s[i:j + len("")] + return (s[:j + len("")] + block + s[j + len(""):]).encode() + if anomaly == "drop_token": + import re + out = re.sub(r".*?", "", s) + out = re.sub(r"true", "false", out) + return out.encode() + return None + + +class ThreadingHTTPServer(socketserver.ThreadingMixIn, http.server.HTTPServer): + daemon_threads = True + allow_reuse_address = True + + +class CtlHandler(http.server.BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, *a): + pass + + def _json(self, code, obj): + body = json.dumps(obj).encode() + self.send_response(code) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_GET(self): + if self.path.startswith("/healthz"): + return self._json(200, {"ok": True, "upstream": UPSTREAM}) + if self.path.startswith("/stats"): + with _cfg_lock: + snapshot = dict(_stats) + snapshot["by_mode"] = dict(_stats["by_mode"]) + snapshot["drop_after_forward"] = list(_stats["drop_after_forward"]) + return self._json(200, snapshot) + self._json(404, {"error": "not found"}) + + def do_POST(self): + if not self.path.startswith("/config"): + return self._json(404, {"error": "not found"}) + length = int(self.headers.get("Content-Length", "0")) + try: + patch = json.loads(self.rfile.read(length) or b"{}") + except Exception as e: + return self._json(400, {"error": f"bad json: {e}"}) + reset = bool(patch.pop("reset", False)) + if reset: + _reset() + with _cfg_lock: + _cfg.update(patch) + snap = dict(_cfg) + self._json(200, {"ok": True, "config": snap}) + + +def main(): + s3 = ThreadingHTTPServer(("0.0.0.0", S3_PORT), Handler) + ctl = ThreadingHTTPServer(("0.0.0.0", CTL_PORT), CtlHandler) + print(f"[s3_fault_proxy] S3 :{S3_PORT} -> {UPSTREAM}; control :{CTL_PORT}", flush=True) + threading.Thread(target=ctl.serve_forever, daemon=True).start() + s3.serve_forever() + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/integration/test_cas_mount_renewal_retry/test.py b/tests/integration/test_cas_mount_renewal_retry/test.py new file mode 100644 index 000000000000..31fa1529c0c3 --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/test.py @@ -0,0 +1,333 @@ +import hashlib +import json +import os +import subprocess +import time +import urllib.request + +import pytest + +from helpers.cluster import ClickHouseCluster + + +cluster = ClickHouseCluster(__file__) + +DISK = "disk_cas_renewal" +SERVER_ROOT_ID = "itest-cas-renewal" +STORAGE_POLICY = "cas_mount_renewal" +MOUNT_OBJECT_KEY = "cas_mount_renewal/gc/server-roots/{}/mount".format(SERVER_ROOT_ID) +MOUNT_REQUEST_PATH = "/test/{}".format(MOUNT_OBJECT_KEY) +RENEWAL_EVENTS = ( + "CASMountRenewalAttempts", + "CASMountRenewalRetries", + "CASMountRenewalResolved", + "CASMountRenewalRecovered", + "CASMountRenewalDeadlineExceeded", + "CASRemountAttempts", + "CASRemountSucceeded", + "CASRemountFailed", +) + + +def _control(base_url, path, patch=None): + if patch is None: + request = urllib.request.Request("{}{}".format(base_url, path)) + else: + request = urllib.request.Request( + "{}{}".format(base_url, path), + data=json.dumps(patch).encode(), + headers={"Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=10) as response: + return json.loads(response.read().decode()) + + +def _wait_until(probe, timeout=40): + deadline = time.monotonic() + timeout + last = None + while time.monotonic() < deadline: + last = probe() + if last: + return last + time.sleep(0.2) + raise AssertionError("condition did not become true within {}s; last={!r}".format(timeout, last)) + + +def _profile_events(node): + rows = node.query( + "SELECT event, value FROM system.events WHERE event IN ({}) FORMAT TSV".format( + ", ".join("'{}'".format(event) for event in RENEWAL_EVENTS) + ) + ) + values = {event: 0 for event in RENEWAL_EVENTS} + for row in rows.splitlines(): + event, value = row.split("\t") + values[event] = int(value) + return values + + +def _event_delta(before, after): + return {event: after[event] - before[event] for event in RENEWAL_EVENTS} + + +def _mount_snapshot(node): + row = node.query( + "SELECT renewal_sequence, state, lifecycle, gc_fenced " + "FROM system.cas_mounts " + "WHERE disk = '{}' AND server_root_id = '{}' LIMIT 1 FORMAT TSV".format( + DISK, SERVER_ROOT_ID + ) + ).strip() + assert row, "the local CAS mount row must be visible" + sequence, state, lifecycle, gc_fenced = row.split("\t") + return { + "sequence": int(sequence), + "state": state, + "lifecycle": lifecycle, + "gc_fenced": int(gc_fenced), + } + + +def _read_mount_object(): + response = cluster.rustfs_client.get_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) + try: + body = response.read() + finally: + response.close() + response.release_conn() + stat = cluster.rustfs_client.stat_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) + return body, stat.etag.strip('"') + + +def _decode_mount(body): + lines = body.decode().splitlines() + assert len(lines) == 2, lines + header = json.loads(lines[0]) + assert header["type"] == "cas_mount_lease" and int(header["v"]) > 0, header + return json.loads(lines[1]) + + +def _renewal_log_rows(node, since, sequence): + node.query("SYSTEM FLUSH LOGS") + rows = node.query( + "SELECT outcome, detail['seq'], detail['write_attempt_id'], " + "detail['attempts_sent'], detail['classification'] " + "FROM system.cas_log " + "WHERE event_type = 'watermark_renew' AND disk_name = '{}' " + "AND detail['server_root_id'] = '{}' " + "AND event_time_microseconds >= toDateTime64('{}', 6) " + "AND detail['seq'] = '{}' " + "ORDER BY event_time_microseconds FORMAT TSV".format( + DISK, SERVER_ROOT_ID, since, sequence + ) + ) + return [tuple(row.split("\t")) for row in rows.splitlines() if row] + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node", + main_configs=["configs/storage_conf.xml"], + with_rustfs=True, + stay_alive=True, + ) + cluster.base_cmd.extend( + ["--file", os.path.join(os.path.dirname(__file__), "docker_compose_proxy.yml")] + ) + + control_url = None + try: + cluster.start() + binding = subprocess.check_output( + cluster.base_cmd + ["port", "s3proxy", "8474"], text=True + ).strip() + control_url = "http://{}".format(binding) + _wait_until(lambda: _control(control_url, "/healthz"), timeout=30) + _control(control_url, "/config", {"reset": True}) + + node = cluster.instances["node"] + node.query( + "CREATE TABLE renewal_probe (id UInt64, payload String) " + "ENGINE = MergeTree ORDER BY id SETTINGS storage_policy = '{}'".format( + STORAGE_POLICY + ) + ) + node.query("INSERT INTO renewal_probe VALUES (0, 'before')") + yield {"node": node, "control_url": control_url} + finally: + if control_url is not None: + try: + _control(control_url, "/config", {"reset": True}) + except Exception: + pass + cluster.shutdown() + + +def test_transient_mount_renewal_retries_without_remount(start_cluster): + node = start_cluster["node"] + control_url = start_cluster["control_url"] + _control(control_url, "/config", {"reset": True}) + mount_before = _mount_snapshot(node) + _, token_before = _read_mount_object() + counters_before = _profile_events(node) + since = node.query("SELECT toString(now64(6))").strip() + + _control( + control_url, + "/config", + { + "rate": 1.0, + "modes": ["503"], + "methods": ["PUT"], + "path_substring": MOUNT_REQUEST_PATH, + "remaining_faults": 1, + "seed": 801, + }, + ) + + def recovered_snapshot(): + mount = _mount_snapshot(node) + counters = _profile_events(node) + if ( + mount["sequence"] > mount_before["sequence"] + and counters["CASMountRenewalRecovered"] + > counters_before["CASMountRenewalRecovered"] + ): + return mount, counters + return None + + mount_after, counters_after = _wait_until(recovered_snapshot) + _control(control_url, "/config", {"rate": 0.0}) + stats = _control(control_url, "/stats") + body_after, token_after = _read_mount_object() + mount_body = _decode_mount(body_after) + delta = _event_delta(counters_before, counters_after) + sequence = mount_after["sequence"] + rows = _wait_until( + lambda: ( + found + if {row[0] for row in found} >= {"retrying", "recovered"} + else None + ) + if (found := _renewal_log_rows(node, since, sequence)) + else None, + timeout=20, + ) + + assert delta["CASMountRenewalAttempts"] > 1, delta + assert delta["CASMountRenewalRetries"] > 0, delta + assert delta["CASMountRenewalRecovered"] > 0, delta + assert delta["CASMountRenewalDeadlineExceeded"] == 0, delta + assert delta["CASRemountAttempts"] == 0, delta + assert delta["CASRemountSucceeded"] == 0, delta + assert delta["CASRemountFailed"] == 0, delta + assert mount_after["state"] == "live", mount_after + assert mount_after["lifecycle"] == "live", mount_after + assert mount_after["gc_fenced"] == 0, mount_after + assert int(mount_body["seq"]) == sequence + assert token_after != token_before + assert stats["faults"] == 1, stats + assert stats["by_mode"].get("503") == 1, stats + print("targeted request count (transient renewal): {}".format(stats["faults"]), flush=True) + + retrying = next(row for row in rows if row[0] == "retrying") + recovered = next(row for row in rows if row[0] == "recovered") + assert retrying[1] == recovered[1] == str(sequence), rows + assert retrying[2] == recovered[2], rows + assert int(recovered[3]) > 1, rows + assert recovered[4] == "committed_after_retry", rows + + node.query( + "ALTER TABLE renewal_probe UPDATE payload = 'after-retry' WHERE id = 0 " + "SETTINGS mutations_sync = 2" + ) + assert node.query("SELECT payload FROM renewal_probe WHERE id = 0").strip() == "after-retry" + + +def test_landed_response_lost_adopts_exact_mount_write(start_cluster): + node = start_cluster["node"] + control_url = start_cluster["control_url"] + _control(control_url, "/config", {"reset": True}) + mount_before = _mount_snapshot(node) + body_before, token_before = _read_mount_object() + counters_before = _profile_events(node) + since = node.query("SELECT toString(now64(6))").strip() + + _control( + control_url, + "/config", + { + "rate": 1.0, + "modes": ["drop_after_forward"], + "methods": ["PUT"], + "path_substring": MOUNT_REQUEST_PATH, + "remaining_faults": 1, + "seed": 802, + }, + ) + + def resolved_snapshot(): + mount = _mount_snapshot(node) + counters = _profile_events(node) + if ( + mount["sequence"] > mount_before["sequence"] + and counters["CASMountRenewalResolved"] + > counters_before["CASMountRenewalResolved"] + and counters["CASMountRenewalRecovered"] + > counters_before["CASMountRenewalRecovered"] + ): + return mount, counters + return None + + mount_after, counters_after = _wait_until(resolved_snapshot) + _control(control_url, "/config", {"rate": 0.0}) + stats = _control(control_url, "/stats") + body_after, token_after = _read_mount_object() + mount_body = _decode_mount(body_after) + delta = _event_delta(counters_before, counters_after) + sequence = mount_after["sequence"] + rows = _wait_until( + lambda: ( + found + if any(row[0] == "recovered" and row[4] == "committed_by_get" for row in found) + else None + ) + if (found := _renewal_log_rows(node, since, sequence)) + else None, + timeout=20, + ) + + records = stats["drop_after_forward"] + assert stats["faults"] == 1, stats + assert stats["by_mode"].get("drop_after_forward") == 1, stats + assert len(records) == 1, records + record = records[0] + assert record["method"] == "PUT", record + assert record["path"].split("?", 1)[0] == MOUNT_REQUEST_PATH, record + assert 200 <= record["upstream_status"] < 300, record + assert record["request_body_sha256"] == hashlib.sha256(body_after).hexdigest(), record + assert record["upstream_etag"].strip('"') == token_after, record + assert body_after != body_before + assert token_after != token_before + assert int(mount_body["seq"]) == sequence + + assert delta["CASMountRenewalAttempts"] == 1, delta + assert delta["CASMountRenewalRetries"] == 0, delta + assert delta["CASMountRenewalResolved"] == 1, delta + assert delta["CASMountRenewalRecovered"] == 1, delta + assert delta["CASMountRenewalDeadlineExceeded"] == 0, delta + assert delta["CASRemountAttempts"] == 0, delta + assert delta["CASRemountSucceeded"] == 0, delta + assert delta["CASRemountFailed"] == 0, delta + assert mount_after["state"] == "live", mount_after + assert mount_after["lifecycle"] == "live", mount_after + assert mount_after["gc_fenced"] == 0, mount_after + + recovered = next(row for row in rows if row[0] == "recovered") + assert recovered[1] == str(sequence), rows + assert recovered[2] and mount_body["write_attempt_id"].startswith(recovered[2]), rows + assert recovered[3] == "1", rows + assert recovered[4] == "committed_by_get", rows + print("targeted request count (landed response lost): {}".format(stats["faults"]), flush=True) diff --git a/tests/integration/test_cas_ref_snaplog/__init__.py b/tests/integration/test_cas_ref_snaplog/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml b/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml new file mode 100644 index 000000000000..74a0ba8de29c --- /dev/null +++ b/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml @@ -0,0 +1,43 @@ + + + + + + object_storage + s3 + cas + itest-ref-snaplog + http://rustfs1:11121/test/cas_snaplog_data/ + clickhouse + clickhouse + 1 + 1 + + + + object_storage + s3 + cas + itest-ref-snaplog + http://rustfs1:11121/test/cas_snaplog_data/ + clickhouse + clickhouse + true + 0 + + + + + +
+ disk_ca +
+
+
+
+
+
diff --git a/tests/integration/test_cas_ref_snaplog/test.py b/tests/integration/test_cas_ref_snaplog/test.py new file mode 100644 index 000000000000..0a9628673a26 --- /dev/null +++ b/tests/integration/test_cas_ref_snaplog/test.py @@ -0,0 +1,170 @@ +import shlex +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +STORAGE_POLICY = "ref_snaplog" +RO_DISK = "disk_ca_ro" + +# Endpoint is http://rustfs1:11121/test/cas_snaplog_data/, so the pool lives under bucket `test`, +# prefix `cas_snaplog_data/`. The snapshot+log ref protocol keeps a table's immutable transaction logs +# and snapshots under cas/ns/stream/, part manifests under cas/manifests/, and content blobs under blobs/. +POOL = "cas_snaplog_data" +BLOBS_PREFIX = POOL + "/blobs/" +REFS_PREFIX = POOL + "/cas/ns/stream/" +MANIFESTS_PREFIX = POOL + "/cas/manifests/" + +NUM_ROWS = 20000 +NUM_INSERTS = 8 + +# Background GC runs every 1s with a 2s grace. After DROP TABLE ... SYNC the dropped namespace's content +# (blobs) and part manifests become GC fodder; we poll until they drain. Bounded wait on a known +# background process, not a race hack. +RECLAIM_RETRIES = 120 +RECLAIM_SLEEP = 1.0 # total bound ~= 120s + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node", + main_configs=["configs/storage_conf.xml"], + with_rustfs=True, + stay_alive=True, + ) + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def _count(prefix): + return len( + list( + cluster.rustfs_client.list_objects( + cluster.rustfs_bucket, prefix, recursive=True + ) + ) + ) + + +def _content_objects(): + # Content blobs + part manifests: the objects the GC fold + condemn/delete pipeline and the + # namespace-cleanup item reclaim after a namespace is removed. + return _count(BLOBS_PREFIX) + _count(MANIFESTS_PREFIX) + + +def _disks(node, query): + # Run a clickhouse-disks command against the read-only CA window over the same pool. + return node.exec_in_container( + [ + "bash", + "-c", + "/usr/bin/clickhouse disks -C /etc/clickhouse-server/config.xml " + "--disk {} --save-logs --query {}".format(RO_DISK, shlex.quote(query)), + ] + ) + + +def test_ref_snaplog_lifecycle_reclaims_and_fsck_clean(): + node = cluster.instances["node"] + + for t in ("ref_t1", "ref_t1_renamed", "ref_t2"): + node.query("DROP TABLE IF EXISTS {} SYNC".format(t)) + + content_baseline = _content_objects() + + # (1) Two tables on the CA/rustfs policy. + for t in ("ref_t1", "ref_t2"): + node.query( + "CREATE TABLE {} (id Int64, data String) ENGINE = MergeTree() ORDER BY id " + "SETTINGS storage_policy = '{}'".format(t, STORAGE_POLICY) + ) + + # (2) Several inserts each -> distinct content blobs + one immutable ref-log transaction per insert. + for t in ("ref_t1", "ref_t2"): + for i in range(NUM_INSERTS): + node.query( + "INSERT INTO {} SELECT number + {off}, toString(number + {off}) " + "FROM numbers({rows})".format(t, off=i * NUM_ROWS, rows=NUM_ROWS) + ) + + assert int(node.query("SELECT count() FROM ref_t1")) == NUM_INSERTS * NUM_ROWS + assert int(node.query("SELECT count() FROM ref_t2")) == NUM_INSERTS * NUM_ROWS + + # The snapshot+log ref format is actually in use: immutable ref objects exist under cas/ns/stream/. + assert _count(REFS_PREFIX) > 0, "expected ref log/snapshot objects under cas/ns/stream/" + assert ( + _content_objects() > content_baseline + ), "expected content objects to rise above baseline after inserts" + + # Read-only fsck agrees while data is present: no authoritative ref names a missing object. + live_fsck = _disks(node, "cas-fsck") + assert "dangling=0" in live_fsck, live_fsck + + # (3) Rename one table: data must survive (its ref namespace and its logs/snapshots are unaffected). + node.query("RENAME TABLE ref_t1 TO ref_t1_renamed") + assert ( + int(node.query("SELECT count() FROM ref_t1_renamed")) == NUM_INSERTS * NUM_ROWS + ) + + # (4) Drop both: the writer appends remove_namespace; background GC folds the -1 edges, condemns and + # deletes the now-unreferenced blobs, and runs the namespace-cleanup item that reclaims the + # removed namespace's physical @cas@ prefixes (part manifests + verbatim files). + node.query("DROP TABLE ref_t1_renamed SYNC") + node.query("DROP TABLE ref_t2 SYNC") + # The content count at the moment the refs are unlinked, so the reclamation below is measured + # against what was actually there to reclaim. + after_drop_content = _content_objects() + + # (5) THE RECLAMATION: the pool's CONTENT (blobs + part manifests) drains back to baseline. Polled + # with an early exit, then cross-checked against GC's own bookkeeping — a pool that shrank for + # some other reason must not pass for a round that reclaimed it. + final = _content_objects() + for _ in range(RECLAIM_RETRIES): + if final <= content_baseline: + break + time.sleep(RECLAIM_SLEEP) + final = _content_objects() + + assert final <= content_baseline, ( + "the dropped namespaces' content was not reclaimed: baseline={}, " + "at_drop={}, final={} (blobs={}, manifests={})".format( + content_baseline, + after_drop_content, + final, + _count(BLOBS_PREFIX), + _count(MANIFESTS_PREFIX), + ) + ) + + node.query("SYSTEM FLUSH LOGS") + rounds = int( + node.query( + "SELECT count() FROM system.cas_gc_log " + "WHERE event_type = 'Finish' AND outcome = 'Success'" + ).strip() + ) + assert rounds > 0, "no successful GC round ran at all" + + deleted = int( + node.query( + "SELECT sum(objects_deleted + manifests_deleted) " + "FROM system.cas_gc_log WHERE event_type = 'Finish'" + ).strip() + or 0 + ) + assert deleted > 0, "the pool's content drained but GC's own bookkeeping reports no deletion" + + # (6) Read-only consumers on the DRAINED pool. + final_fsck = _disks(node, "cas-fsck") + assert "dangling=0" in final_fsck, final_fsck + + # `cas-gc-dryrun` on a fully drained pool has nothing left to preview. + dryrun = _disks(node, "cas-gc-dryrun") + assert "preview_deletes=0" in dryrun, dryrun diff --git a/tests/integration/test_cas_replicated_relink/__init__.py b/tests/integration/test_cas_replicated_relink/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_replicated_relink/configs/server_root_id_node1.xml b/tests/integration/test_cas_replicated_relink/configs/server_root_id_node1.xml new file mode 100644 index 000000000000..3104d621390d --- /dev/null +++ b/tests/integration/test_cas_replicated_relink/configs/server_root_id_node1.xml @@ -0,0 +1,12 @@ + + + + + + node1 + + + + diff --git a/tests/integration/test_cas_replicated_relink/configs/server_root_id_node2.xml b/tests/integration/test_cas_replicated_relink/configs/server_root_id_node2.xml new file mode 100644 index 000000000000..960bd079825a --- /dev/null +++ b/tests/integration/test_cas_replicated_relink/configs/server_root_id_node2.xml @@ -0,0 +1,12 @@ + + + + + + node2 + + + + diff --git a/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml b/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml new file mode 100644 index 000000000000..4513d345a2fb --- /dev/null +++ b/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml @@ -0,0 +1,36 @@ + + + + + object_storage + s3 + cas + + http://rustfs1:11121/test/shared_pool/ + clickhouse + clickhouse + + + 1 + 1 + + + + + +
+ disk_cas_shared +
+
+
+
+
+
diff --git a/tests/integration/test_cas_replicated_relink/configs/storage_conf_other_pool.xml b/tests/integration/test_cas_replicated_relink/configs/storage_conf_other_pool.xml new file mode 100644 index 000000000000..b8900813e0a3 --- /dev/null +++ b/tests/integration/test_cas_replicated_relink/configs/storage_conf_other_pool.xml @@ -0,0 +1,31 @@ + + + + + + object_storage + s3 + cas + http://rustfs1:11121/test/other_pool/ + clickhouse + clickhouse + node2_other + 0 + + + + + +
+ disk_cas_other +
+
+
+
+
+
diff --git a/tests/integration/test_cas_replicated_relink/test.py b/tests/integration/test_cas_replicated_relink/test.py new file mode 100644 index 000000000000..83d8d239ba92 --- /dev/null +++ b/tests/integration/test_cas_replicated_relink/test.py @@ -0,0 +1,905 @@ +import re +import shlex +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +# Both replicas mount the SAME content-addressed pool (endpoint .../root/shared_pool/). A +# ReplicatedMergeTree part written on one replica is therefore ALREADY present (as content blobs + +# manifest) in the pool when the other replica needs it — so the "fetch" is a fetch-by-relink: the +# fetching replica publishes its own ref to the existing blobs instead of downloading any bytes (the CA +# analogue of zero-copy replication, spec §4). +STORAGE_POLICY = "cas_shared" +CA_DISK = "disk_cas_shared" + +# A second, independent pool mounted by node2 only (configs/storage_conf_other_pool.xml). Used for the +# cross-pool leg of B66b: relink is gated on both sides naming the same pool, so a fetch into this one +# must degrade to bytes. +OTHER_STORAGE_POLICY = "cas_other" +OTHER_CA_DISK = "disk_cas_other" + +# The shared pool's blob prefix inside the `test` RustFS bucket. The relink proof is that the fetch does +# NOT create new objects under here: relink publishes a ref (per-server, under store/), never a blob. +BLOBS_PREFIX = "shared_pool/blobs/" + +NUM_ROWS = 10000 + +# ---------------------------------------------------------------------------------------------------- +# WHY EVERY RELINK ASSERTION BELOW IS A POSITIVE ONE +# +# "The fetch created no new blobs" is NOT by itself evidence that a relink happened. On a +# content-addressed disk a BYTE fetch writes the very same content, which deduplicates against the +# blobs already in the pool, so its blob-count delta is zero too. A test that only counts blobs is +# therefore green whether the protocol worked or silently fell back — the single easiest worthless test +# on this path. +# +# So each relink test asserts a signal that is reachable ONLY through the intended path: +# +# RELINK RAN -> the receiver's `Relink of part

onto disk finished (no bytes transferred).` +# That line is the last statement of `Fetcher::relinkPartToDisk` and is reachable only +# after the confirm answered `yes` AND `promote()` returned `Committed` (taxonomy +# row 4). Every other row returns or throws before it. +# BYTES RAN -> the receiver's `Download of part

onto disk finished.` from +# `downloadPartToDisk`, plus the specific line naming WHY relink was declined. +# +# The blob-count / `CASBlobPut == 0` checks are kept as corroboration, never as the proof. +# ---------------------------------------------------------------------------------------------------- + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node1", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node1.xml"], + macros={"replica": "node1"}, + with_rustfs=True, + with_zookeeper=True, + stay_alive=True, + ) + cluster.add_instance( + "node2", + main_configs=[ + "configs/storage_conf.xml", + "configs/server_root_id_node2.xml", + "configs/storage_conf_other_pool.xml", + ], + macros={"replica": "node2"}, + with_rustfs=True, + with_zookeeper=True, + stay_alive=True, + ) + + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def blob_keys(): + """Every object key under the shared pool's blob prefix, as a set.""" + objects = cluster.rustfs_client.list_objects( + cluster.rustfs_bucket, BLOBS_PREFIX, recursive=True + ) + return {obj.object_name for obj in objects} + + +def count_blobs(): + return len(blob_keys()) + + +def log_lines(node, pattern): + """Server-log lines matching an extended regular expression. + + Deliberately NOT `instance.grep_in_log`: that one globs `clickhouse-server.log*`, which includes + `clickhouse-server.err.log`, so any warning-or-above line is counted twice. Several assertions here + are exact counts, and a doubled count is indistinguishable from a real second attempt. + """ + out = node.exec_in_container( + [ + "bash", + "-c", + "grep -a -E {} /var/log/clickhouse-server/clickhouse-server.log || true".format( + shlex.quote(pattern) + ), + ] + ) + return [line for line in out.splitlines() if line.strip()] + + +def wait_for_log_lines(node, pattern, timeout=60): + """Poll until at least one line matches, then return the matches. Fails loudly on timeout.""" + deadline = time.time() + timeout + while True: + found = log_lines(node, pattern) + if found: + return found + assert time.time() < deadline, "timed out waiting for log lines matching {!r} on {}".format( + pattern, node.name + ) + time.sleep(0.5) + + +def relink_finished_pattern(table, part, disk=CA_DISK): + """The receiver-side proof that the publish→confirm→promote path completed for this exact part.""" + return r"default\.{} .*Relink of part {} onto disk {} finished \(no bytes transferred\)".format( + table, re.escape(part), disk + ) + + +def download_finished_pattern(table, part, disk=CA_DISK): + """The receiver-side proof that the BYTE path completed for this exact part.""" + return r"default\.{} .*Download of part {} onto disk {} finished".format( + table, re.escape(part), disk + ) + + +def relink_offer_pattern(table, part): + """The SENDER-side line, one per relink offer actually made. The attempt counter.""" + return r"default\.{} .*Sending part {} by relink".format(table, re.escape(part)) + + +def assert_relinked(node, table, part, disk=CA_DISK, timeout=60): + wait_for_log_lines(node, relink_finished_pattern(table, part, disk), timeout=timeout) + assert not log_lines(node, download_finished_pattern(table, part, disk)), ( + "part {} of {} was relinked AND byte-downloaded on {} — the relink proof is not exclusive".format( + part, table, node.name + ) + ) + + +def assert_byte_downloaded(node, table, part, disk=CA_DISK, timeout=60): + wait_for_log_lines(node, download_finished_pattern(table, part, disk), timeout=timeout) + assert not log_lines(node, relink_finished_pattern(table, part, disk)), ( + "part {} of {} was expected to arrive as bytes but a relink completed on {}".format( + part, table, node.name + ) + ) + + +def assert_no_new_blobs(before_keys): + """Corroboration for a relink: the fetch added no object under the pool's blob prefix. + + Phrased as "no NEW key" rather than "the same count" on purpose — background GC may reclaim + unrelated debris at any moment on this fixture (`gc_interval_sec` is 1), and a count that went DOWN + says nothing about whether the fetch wrote anything. + """ + new_keys = sorted(blob_keys() - before_keys) + assert not new_keys, "the fetch wrote {} new blob(s), e.g. {}".format(len(new_keys), new_keys[:5]) + + +def cas_blob_puts(node): + return int(node.query("SELECT sum(value) FROM system.events WHERE event = 'CASBlobPut'") or 0) + + +def active_part_names(node, table): + return node.query( + "SELECT name FROM system.parts WHERE database = 'default' AND table = '{}' AND active " + "ORDER BY name".format(table) + ).split() + + +def any_state_part_count(node, table, part): + return int( + node.query( + "SELECT count() FROM system.parts WHERE database = 'default' AND table = '{}' " + "AND name = '{}'".format(table, part) + ) + ) + + +def wait_until(predicate, timeout, what): + deadline = time.time() + timeout + while True: + if predicate(): + return + assert time.time() < deadline, "timed out waiting for {}".format(what) + time.sleep(0.5) + + +def fsck(node, disk=CA_DISK): + """`SYSTEM CAS FSCK ` as a dict of column -> value. + + Driven through `clickhouse-client --format` rather than a trailing `FORMAT` clause: `ASTSystemQuery` + is not an `ASTQueryWithOutput`, so `SYSTEM ... FORMAT TSVWithNames` is a syntax error. Reading the + header is what keeps this from depending on the column ORDER of the summary. + """ + out = node.exec_in_container( + [ + "bash", + "-c", + "clickhouse client --format TSVWithNames --query {}".format( + shlex.quote("SYSTEM CAS FSCK '{}'".format(disk)) + ), + ] + ).splitlines() + header, row = out[0].split("\t"), out[1].split("\t") + summary = dict(zip(header, row)) + assert "dangling" in summary, "unexpected FSCK summary shape: {}".format(out) + return summary + + +def gc_round(node, disk=CA_DISK): + node.query("SYSTEM CAS GC RUN '{}'".format(disk)) + + +def drop_everywhere(table): + for node in (cluster.instances["node1"], cluster.instances["node2"]): + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + + +def create_replicated(node, table, policy=STORAGE_POLICY, zk_path=None, extra_settings=""): + node.query( + "CREATE TABLE {table} (id Int64, v UInt64, s String) " + "ENGINE = ReplicatedMergeTree('{zk}', '{{replica}}') ORDER BY id " + "SETTINGS storage_policy = '{policy}'{extra}".format( + table=table, + zk=zk_path or "/clickhouse/tables/" + table, + policy=policy, + extra=(", " + extra_settings) if extra_settings else "", + ) + ) + + +def insert_rows(node, table, start, rows=NUM_ROWS): + node.query( + "INSERT INTO {table} SELECT number, number * 10, toString(number) " + "FROM numbers({start}, {rows})".format(table=table, start=start, rows=rows) + ) + + +def test_replicated_fetch_by_relink(): + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + + node1.query("DROP TABLE IF EXISTS r SYNC") + node2.query("DROP TABLE IF EXISTS r SYNC") + + # Two replicas of ONE ReplicatedMergeTree table on the shared CA pool. Lifting B33 is what makes this + # CREATE succeed at all; the shared-pool mount is what makes the second replica start. + create_tpl = ( + "CREATE TABLE r (id Int64, v UInt64, s String) " + "ENGINE = ReplicatedMergeTree('/clickhouse/tables/r', '{{replica}}') " + "ORDER BY id SETTINGS storage_policy = '{policy}'" + ) + node1.query(create_tpl.format(policy=STORAGE_POLICY)) + node2.query(create_tpl.format(policy=STORAGE_POLICY)) + + # (1) INSERT on replica node1. node2 must replicate the part. + node1.query( + "INSERT INTO r SELECT number, number * 10, toString(number) FROM numbers({rows})".format( + rows=NUM_ROWS + ) + ) + + # Blob count after the insert, BEFORE node2 fetches. This is the relink baseline. + blobs_after_insert = count_blobs() + assert blobs_after_insert > 0, "insert must have written content blobs to the shared pool" + + # (2) node2 fetches the part. SYNC REPLICA blocks until the queue (the fetch) drains. + node2.query("SYSTEM SYNC REPLICA r", timeout=60) + + # (3) node2 reads the SAME rows back. + expected_sum_id = (NUM_ROWS - 1) * NUM_ROWS // 2 + assert int(node2.query("SELECT count() FROM r")) == NUM_ROWS + assert int(node2.query("SELECT sum(id) FROM r")) == expected_sum_id + assert int(node2.query("SELECT sum(v) FROM r")) == expected_sum_id * 10 + + # (4) THE RELINK PROOF: the fetch created NO new blob objects. node2 published a ref to the blobs + # node1 already wrote — it did not download/re-write them. (Relink, not byte download.) + blobs_after_fetch = count_blobs() + assert blobs_after_fetch == blobs_after_insert, ( + "fetch-by-relink must not create new blob objects: had {} after insert, {} after node2 fetched " + "(a byte download would have re-written blobs)".format( + blobs_after_insert, blobs_after_fetch + ) + ) + + # (5) A merge on node1 fetched-by-relink by node2: insert a second part on node1, OPTIMIZE to merge, + # and confirm node2 picks up the merged part with still no new blobs beyond the merge's own. + node1.query( + "INSERT INTO r SELECT number, number * 10, toString(number) FROM numbers({a}, {rows})".format( + a=NUM_ROWS, rows=NUM_ROWS + ) + ) + node2.query("SYSTEM SYNC REPLICA r", timeout=60) + blobs_before_merge = count_blobs() + + node1.query("OPTIMIZE TABLE r FINAL") + node1.query("SYSTEM SYNC REPLICA r", timeout=60) + blobs_after_merge_on_node1 = count_blobs() + + # node2 fetches the merged part. The merge itself may write new blobs on node1 (the merged content), + # but node2's FETCH of that merged part must add NOTHING further (relink). + node2.query("SYSTEM SYNC REPLICA r", timeout=60) + blobs_after_merge_fetch = count_blobs() + assert blobs_after_merge_fetch == blobs_after_merge_on_node1, ( + "fetch-by-relink of the merged part must not create new blobs: {} after node1 merged, {} after " + "node2 fetched".format(blobs_after_merge_on_node1, blobs_after_merge_fetch) + ) + + assert int(node2.query("SELECT count() FROM r")) == 2 * NUM_ROWS + assert int(node1.query("SELECT count() FROM r")) == 2 * NUM_ROWS + + node1.query("DROP TABLE IF EXISTS r SYNC") + node2.query("DROP TABLE IF EXISTS r SYNC") + + +def test_relink_happy_path_proof(): + """Task 16 step 2 — the happy path, proved POSITIVELY. + + Taxonomy row 4 (confirm `yes` -> `promote` -> `Committed`). The proof is the receiver's + `... finished (no bytes transferred)` line, which no other row can reach; `CASBlobPut == 0` and the + flat blob count are corroboration only (see the note at the top of this file). + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "relink_happy" + drop_everywhere(table) + + create_replicated(node1, table) + create_replicated(node2, table) + + node2.query("SYSTEM STOP FETCHES {}".format(table)) + insert_rows(node1, table, 0) + part = active_part_names(node1, table)[0] + + blobs_before = blob_keys() + puts_before = cas_blob_puts(node2) + + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=60) + + # THE PROOF: reachable only after a confirm `yes` and a committed promote. + assert_relinked(node2, table, part) + + # Corroboration, in the plan's own terms: the receiver issued no blob PUT at all, and the pool's + # blob set is byte-identical to what the sender's insert left behind. + assert cas_blob_puts(node2) == puts_before + assert_no_new_blobs(blobs_before) + + assert int(node2.query("SELECT count() FROM {}".format(table))) == NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(table))) == int( + node1.query("SELECT sum(v) FROM {}".format(table)) + ) + + drop_everywhere(table) + + +def test_fetch_part_into_detached_relinks(): + """Task 16 step 5 — B66b, manual caller #1: `ALTER TABLE ... FETCH PART ... FROM`. + + Taxonomy row 4 with `to_detached=true`: the staged ref is `detached/tmp-fetch_` and the + finalization is `renameTo(detached/)`. Before B66b the relink capability was gated on + `!to_detached`, so this fetch could only ever be bytes. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "b66b_part_src", "b66b_part_dst" + drop_everywhere(src) + drop_everywhere(dst) + + create_replicated(node1, src) + create_replicated(node2, dst) + insert_rows(node1, src, 0) + part = active_part_names(node1, src)[0] + + blobs_before = blob_keys() + puts_before = cas_blob_puts(node2) + + node2.query( + "ALTER TABLE {dst} FETCH PART '{part}' FROM '/clickhouse/tables/{src}'".format( + dst=dst, part=part, src=src + ) + ) + + assert_relinked(node2, dst, part) + assert cas_blob_puts(node2) == puts_before + assert_no_new_blobs(blobs_before) + + # ... and the detached part is a real, readable part once attached. + node2.query("ALTER TABLE {} ATTACH PART '{}'".format(dst, part)) + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(dst))) == int( + node1.query("SELECT sum(v) FROM {}".format(src)) + ) + + drop_everywhere(src) + drop_everywhere(dst) + + +def test_fetch_partition_into_detached_relinks(): + """Task 16 step 5 — B66b, manual caller #2: `ALTER TABLE ... FETCH PARTITION ... FROM`. + + Same taxonomy row as the FETCH PART leg; a separate test because it is a separate call site (it + fetches a whole partition through its own thread pool) and Task 15 changed both. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "b66b_partition_src", "b66b_partition_dst" + drop_everywhere(src) + drop_everywhere(dst) + + create_replicated(node1, src) + create_replicated(node2, dst) + insert_rows(node1, src, 0) + part = active_part_names(node1, src)[0] + + blobs_before = blob_keys() + puts_before = cas_blob_puts(node2) + + node2.query( + "ALTER TABLE {dst} FETCH PARTITION ID 'all' FROM '/clickhouse/tables/{src}'".format( + dst=dst, src=src + ) + ) + + assert_relinked(node2, dst, part) + assert cas_blob_puts(node2) == puts_before + assert_no_new_blobs(blobs_before) + + node2.query("ALTER TABLE {} ATTACH PARTITION ID 'all'".format(dst)) + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + + drop_everywhere(src) + drop_everywhere(dst) + + +def test_detached_fetch_cross_pool_falls_back_to_bytes(): + """Task 16 step 5 — the cross-pool leg: relink is gated on ONE pool, so this must be bytes. + + Not a taxonomy row at all: the sender's pre-filter (`receiver_pool_uuid == getPoolUUID()`) declines + to make an offer, so the receiver never enters `relinkPartToDisk`. The positive signal is therefore + the byte path's own completion line plus the ABSENCE of any relink offer for this part. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "xpool_src", "xpool_dst" + drop_everywhere(src) + drop_everywhere(dst) + + create_replicated(node1, src) + create_replicated(node2, dst, policy=OTHER_STORAGE_POLICY) + insert_rows(node1, src, 0) + part = active_part_names(node1, src)[0] + + node2.query( + "ALTER TABLE {dst} FETCH PART '{part}' FROM '/clickhouse/tables/{src}'".format( + dst=dst, part=part, src=src + ) + ) + + # The bytes really moved: the receiver ran `downloadPartToDisk` onto the OTHER pool's disk. + assert_byte_downloaded(node2, dst, part, disk=OTHER_CA_DISK) + # ... and the sender never offered a relink for it, which is what makes the byte path the *intended* + # outcome here rather than an accident of some later failure. + assert not log_lines(node1, relink_offer_pattern(src, part)) + + node2.query("ALTER TABLE {} ATTACH PART '{}'".format(dst, part)) + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(dst))) == int( + node1.query("SELECT sum(v) FROM {}".format(src)) + ) + + drop_everywhere(src) + drop_everywhere(dst) + + +def test_attach_partition_from_relinks_on_queue_fetch(): + """Task 16 step 6 (RPL-5) — `ATTACH PARTITION ... FROM` replicates as `REPLACE_RANGE`. + + The source table exists only on node1, so node2 cannot clone locally and its queue entry falls + through to `executeReplaceRange`'s `fetchSelectedPart` — a THIRD fetch call site, with its own + `tmp_replace_from_fetch_` prefix. Taxonomy row 4; the proof is the same relink-finished line. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "rpl5_attach_src", "rpl5_attach_dst" + drop_everywhere(src) + drop_everywhere(dst) + + node1.query( + "CREATE TABLE {src} (id Int64, v UInt64, s String) ENGINE = MergeTree ORDER BY id " + "SETTINGS storage_policy = '{policy}'".format(src=src, policy=STORAGE_POLICY) + ) + create_replicated(node1, dst) + create_replicated(node2, dst) + insert_rows(node1, src, 0) + + node1.query("ALTER TABLE {dst} ATTACH PARTITION tuple() FROM {src}".format(dst=dst, src=src)) + part = active_part_names(node1, dst)[0] + + blobs_before = blob_keys() + puts_before = cas_blob_puts(node2) + + node2.query("SYSTEM SYNC REPLICA {}".format(dst), timeout=90) + + assert_relinked(node2, dst, part) + assert cas_blob_puts(node2) == puts_before + assert_no_new_blobs(blobs_before) + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + + node1.query("DROP TABLE IF EXISTS {} SYNC".format(src)) + drop_everywhere(dst) + + +def test_replace_partition_relinks_on_queue_fetch(): + """Task 16 step 6 (RPL-5) — `REPLACE PARTITION`, i.e. the same entry with a drop range attached. + + Separate from the ATTACH leg because the destination is non-empty: node2 must drop its own covering + part AND fetch the replacement, so the relink runs against a partition that already had a ref. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "rpl5_replace_src", "rpl5_replace_dst" + drop_everywhere(src) + drop_everywhere(dst) + + node1.query( + "CREATE TABLE {src} (id Int64, v UInt64, s String) ENGINE = MergeTree ORDER BY id " + "SETTINGS storage_policy = '{policy}'".format(src=src, policy=STORAGE_POLICY) + ) + create_replicated(node1, dst) + create_replicated(node2, dst) + + # Destination starts non-empty and replicated, so REPLACE really replaces something. + insert_rows(node1, dst, 0) + node2.query("SYSTEM SYNC REPLICA {}".format(dst), timeout=60) + + insert_rows(node1, src, 5 * NUM_ROWS) + node1.query("ALTER TABLE {dst} REPLACE PARTITION tuple() FROM {src}".format(dst=dst, src=src)) + part = active_part_names(node1, dst)[0] + + node2.query("SYSTEM SYNC REPLICA {}".format(dst), timeout=90) + + assert_relinked(node2, dst, part) + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(dst))) == int( + node1.query("SELECT sum(v) FROM {}".format(dst)) + ) + + node1.query("DROP TABLE IF EXISTS {} SYNC".format(src)) + drop_everywhere(dst) + + +def interserver_request(node, target_host, params): + """One raw interserver request, straight at the sender's `DataPartsExchange` endpoint. + + The version-mix behaviour lives on the wire and nowhere else: which protocol version the peer + advertises is not configurable, so the only way to exercise a NON-confirm-capable peer against this + build's sender is to be that peer. Returns (headers, body_size). + """ + query = "&".join("{}={}".format(k, v) for k, v in params) + out = node.exec_in_container( + [ + "bash", + "-c", + "curl -sS -o /tmp/ca_ism_body -D /tmp/ca_ism_hdr -w '%{{http_code}} %{{size_download}}' " + "{url} >/tmp/ca_ism_stat; cat /tmp/ca_ism_hdr; echo '--STAT--'; cat /tmp/ca_ism_stat".format( + url=shlex.quote("http://{}:9009/?{}".format(target_host, query)) + ), + ] + ) + headers, stat = out.split("--STAT--") + http_code, size = stat.split() + assert http_code == "200", "interserver request failed: {}\n{}".format(stat, headers) + return headers, int(size) + + +def test_version_mix_legacy_peer_gets_bytes(): + """Task 16 step 7 — version mix: a peer that does not promise to confirm is served BYTES. + + This is the sender-side half of the mixed-build gate, and it is the half that is reachable without a + second binary: the offer is gated on `client_protocol_version >= 11` (`..._WITH_CA_CONFIRM`), so a + peer advertising 10 — a build that would relink WITHOUT confirming — must get the byte stream. + Degrading to bytes, never to an unconfirmed relink, is the whole point of moving the gate to 11. + + The control request (identical, but advertising 11) is what makes the negative meaningful: it proves + the request is otherwise perfectly relinkable, so the absence of an offer in the v10 case is the + version gate and not a malformed request. + + The receiver-side row-1 branch — a genuinely OLD sender that offers a relink with NO source token — + is NOT covered here; see the report accompanying this task. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "vermix" + drop_everywhere(table) + + create_replicated(node1, table) + create_replicated(node2, table) + insert_rows(node1, table, 0) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=60) + part = active_part_names(node1, table)[0] + + # The pool identity as the SERVER reports it: taken from the sender's own offer line rather than + # re-derived from the pool metadata, so the value fed back in is exactly what `getPoolUUID` returns. + offers = wait_for_log_lines(node1, relink_offer_pattern(table, part)) + pool_uuid = re.search(r"shared pool ([0-9a-f]+)\)", offers[-1]).group(1) + + endpoint = "DataPartsExchange:/clickhouse/tables/{}/replicas/node1".format(table) + base = [ + ("endpoint", endpoint), + ("part", part), + ("compress", "false"), + ("cas_pool_uuid", pool_uuid), + ] + + # CONTROL — a confirm-capable peer: an offer, with a token, and a tiny manifest-only body. + headers_v11, size_v11 = interserver_request( + node2, "node1", base + [("client_protocol_version", "11")] + ) + assert "cas_relink=part_manifest_v2" in headers_v11, headers_v11 + assert "cas_source_token=" in headers_v11, headers_v11 + + # THE CASE UNDER TEST — a peer advertising the pre-confirm version: no offer, and the part's bytes. + headers_v10, size_v10 = interserver_request( + node2, "node1", base + [("client_protocol_version", "10")] + ) + assert "cas_relink" not in headers_v10, headers_v10 + assert "cas_source_token" not in headers_v10, headers_v10 + assert "server_protocol_version=10" in headers_v10, headers_v10 + + # Positive proof that bytes ACTUALLY moved rather than the request merely succeeding: the v10 + # response carries the whole part, orders of magnitude more than the manifest-only relink payload. + assert size_v10 > 20 * size_v11, ( + "the v10 peer should have received the part's bytes, got {} bytes against the relink offer's " + "{}".format(size_v10, size_v11) + ) + + drop_everywhere(table) + + +def test_recursion_brake_bounds_relink_to_one_attempt(): + """Task 16 step 4 — the `allow_ca_relink` recursion brake. + + A mechanism failure that is a property of the sender/receiver PAIR reproduces on every attempt, so + without the brake the byte-fetch fallback re-advertises the pool, is re-offered a relink, fails + again, and recurses until the stack is gone. The failpoint injects exactly that class of failure + (taxonomy rows 2 and 5 share this ACTION), because no configuration can produce one. + + The assertion is a COUNT, not termination: exactly ONE relink offer is made for this part, and then + the bytes arrive. Termination alone would also hold for a brake that merely bounded the recursion at + some larger depth, and it would hold vacuously if the relink path were never entered at all. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "brake" + drop_everywhere(table) + + create_replicated(node1, table) + create_replicated(node2, table) + + node2.query("SYSTEM STOP FETCHES {}".format(table)) + insert_rows(node1, table, 0) + part = active_part_names(node1, table)[0] + + node2.query("SYSTEM ENABLE FAILPOINT cas_relink_receiver_force_mechanism_failure") + try: + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=90) + + # The receiver hit the injected failure exactly once... + hits = log_lines( + node2, + r"Failpoint cas_relink_receiver_force_mechanism_failure: abandoning the relink of part {}".format( + re.escape(part) + ), + ) + assert len(hits) == 1, "expected exactly one relink attempt, got {}:\n{}".format( + len(hits), "\n".join(hits) + ) + + # ... and the SENDER, independently, made exactly one offer. This is the sharper of the two: the + # re-request is what would re-open the capability, and the sender is the only party that can say + # whether it did. + offers = log_lines(node1, relink_offer_pattern(table, part)) + assert len(offers) == 1, "expected exactly one relink offer, got {}:\n{}".format( + len(offers), "\n".join(offers) + ) + + # And the fetch still succeeded, over the byte path. + assert_byte_downloaded(node2, table, part) + assert int(node2.query("SELECT count() FROM {}".format(table))) == NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(table))) == int( + node1.query("SELECT sum(v) FROM {}".format(table)) + ) + finally: + node2.query("SYSTEM DISABLE FAILPOINT cas_relink_receiver_force_mechanism_failure") + + drop_everywhere(table) + + +# Settings that make node1 drop an outdated part — and with it the CA ref the confirm asks about — +# within a few seconds instead of the default eight minutes. +FAST_OLD_PART_REMOVAL = ( + "old_parts_lifetime = 1, cleanup_delay_period = 1, cleanup_delay_period_random_add = 1, " + "max_cleanup_delay_period = 1" +) + + +def open_publish_confirm_window(node1, node2, table, base): + """Drive a relink up to the paused point BETWEEN the receiver's durable `+1` and the confirm. + + Returns `(part, part_blobs)`: the name of the part whose relink is now stalled, and the blob keys + that its insert ADDED to the pool. The delta matters — debris from earlier tests in this module may + still be sitting in the pool and may legitimately be reclaimed while the window is open, so only the + keys this part created can be asserted about. The caller MUST resume the failpoint. + + `base` shifts the generated rows so this table's column data is unlike any other table's in this + module. Without it the content-addressed store deduplicates the insert against an earlier test's + identical blobs and the delta is EMPTY — which would make every blob assertion below vacuous. + """ + create_replicated(node1, table, extra_settings=FAST_OLD_PART_REMOVAL) + create_replicated(node2, table, extra_settings=FAST_OLD_PART_REMOVAL) + + node2.query("SYSTEM STOP FETCHES {}".format(table)) + before_insert = blob_keys() + insert_rows(node1, table, base) + part = active_part_names(node1, table)[0] + part_blobs = blob_keys() - before_insert + assert part_blobs, "the insert wrote no new blob into the shared pool" + + node2.query("SYSTEM ENABLE FAILPOINT cas_relink_receiver_pause_before_confirm") + node2.query("SYSTEM START FETCHES {}".format(table)) + # Blocks until the fetch thread is parked inside `relinkPartToDisk`, after `prepareAdoptFromManifest` + # made the receiver's `+1` durable and before the confirm request is built. + node2.query("SYSTEM WAIT FAILPOINT cas_relink_receiver_pause_before_confirm PAUSE", timeout=120) + return part, part_blobs + + +def merge_the_source_part_away(node1, table, part, base): + """While the receiver is parked: make the sender stop holding the exact binding it offered.""" + insert_rows(node1, table, base + NUM_ROWS) + node1.query("OPTIMIZE TABLE {} FINAL".format(table)) + # The confirm is answered from the sender's live state, so the test is only meaningful once the old + # part — and the ref naming its manifest — is really gone, not merely Outdated. + wait_until( + lambda: any_state_part_count(node1, table, part) == 0, + timeout=120, + what="node1 to drop the outdated part {}".format(part), + ) + + +def test_confirm_refuses_when_source_dropped_in_window(): + """Task 16 step 1 — the race the confirm exists to lose safely. + + Taxonomy row 3: the source cannot prove it still holds the offered manifest, so the receiver aborts + its durable `+1` and throws a retry-later `NETWORK_ERROR` INSTEAD of falling back to bytes. The two + assertions that matter are (a) the queue recovers by re-selecting — here, onto the covering part — + and (b) NO byte re-request ever went to the source whose state was in doubt. (b) is the entire + reason row 3 throws where rows 2 and 5 return `nullptr`. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "race_confirm" + drop_everywhere(table) + + try: + part, _ = open_publish_confirm_window(node1, node2, table, base=1_000_000) + merge_the_source_part_away(node1, table, part, base=1_000_000) + finally: + node2.query("SYSTEM DISABLE FAILPOINT cas_relink_receiver_pause_before_confirm") + + # POSITIVE SIGNAL for row 3: the locally generated refusal, naming the source and the part. + wait_for_log_lines( + node2, + r"Source .* did not prove it still holds the manifest it offered for part {}".format( + re.escape(part) + ), + timeout=120, + ) + + # (a) the queue re-selects rather than losing the data. + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=180) + assert int(node2.query("SELECT count() FROM {}".format(table))) == 2 * NUM_ROWS + assert int(node2.query("SELECT sum(v) FROM {}".format(table))) == int( + node1.query("SELECT sum(v) FROM {}".format(table)) + ) + assert active_part_names(node2, table) == active_part_names(node1, table) + + # (b) the abandoned part was never re-requested as bytes from the same source, and it was never + # promoted either — both would be a violation of the row-3 contract. + assert not log_lines(node2, download_finished_pattern(table, part)), ( + "row 3 must not fall back to a byte re-request against the source it could not confirm" + ) + assert not log_lines(node2, relink_finished_pattern(table, part)) + assert any_state_part_count(node2, table, part) == 0 + + drop_everywhere(table) + + +def test_stalled_publish_protects_source_blobs_and_commits_nothing(): + """Task 16 step 3 — the codex-6 regression, which is why publish-then-confirm exists at all. + + The receiver's `+1` is durable while the fetch is stalled. Across the stall the sender merges the + part away and GC runs to a fixpoint several times over: the offered manifest's blobs MUST survive, + because the stalled receiver's own binding protects them — that is what makes the later confirm a + meaningful question rather than a race against a sweep. And when the confirm finally answers + `unproven`, the stalled attempt must leave NOTHING committed. + + The soundness guard is the last assertion: once the attempt is abandoned and the sender no longer + holds the part, GC DOES reclaim those same blobs. Without it, "the blobs survived" would also be + satisfied by a GC that never deletes anything. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "codex6_stall" + drop_everywhere(table) + + try: + # `part_blobs` is exactly what this part's insert added to the pool — see the helper for why it + # has to be the delta and not everything under the prefix. + part, part_blobs = open_publish_confirm_window(node1, node2, table, base=2_000_000) + + merge_the_source_part_away(node1, table, part, base=2_000_000) + + # Four full GC rounds on both mounters, spread well past the pool's 3-second condemn grace, so + # a blob that was NOT protected would have been condemned, aged out and deleted in the window. + for _ in range(4): + gc_round(node1) + gc_round(node2) + time.sleep(1.5) + + missing = sorted(part_blobs - blob_keys()) + assert not missing, ( + "the stalled receiver's durable +1 must protect the offered manifest's blobs across GC; " + "{} of {} were reclaimed, e.g. {}".format(len(missing), len(part_blobs), missing[:5]) + ) + finally: + node2.query("SYSTEM DISABLE FAILPOINT cas_relink_receiver_pause_before_confirm") + + wait_for_log_lines( + node2, + r"Source .* did not prove it still holds the manifest it offered for part {}".format( + re.escape(part) + ), + timeout=120, + ) + + # Nothing was committed by the stalled attempt. + assert any_state_part_count(node2, table, part) == 0 + assert not log_lines(node2, relink_finished_pattern(table, part)) + + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=180) + assert int(node2.query("SELECT count() FROM {}".format(table))) == 2 * NUM_ROWS + + # No dangling reference anywhere in the pool, from either mounter's point of view. + for node in (node1, node2): + summary = fsck(node) + assert summary["dangling"] == "0", "{} fsck: {}".format(node.name, summary) + + # THE SOUNDNESS GUARD, and it is what makes the survival asserted earlier mean anything: with the + # part gone from both replicas and the stalled attempt abandoned, its unique blobs are unreachable, + # so GC reclaiming them proves their survival DURING the stall was the relink pin and not GC + # inactivity. Without this, "the blobs were still there" would also be what a GC that never ran + # produces. + reclaimed = set() + for _ in range(8): + gc_round(node1) + gc_round(node2) + reclaimed = part_blobs - blob_keys() + if reclaimed == part_blobs: + break + # Pool-wide: the GC lease is held by ONE server and it need not be node1. + rounds = 0 + for n in (node1, node2): + n.query("SYSTEM FLUSH LOGS") + rounds += int( + n.query( + "SELECT count() FROM system.cas_gc_log " + "WHERE event_type = 'Finish' AND outcome = 'Success'" + ).strip() + or 0 + ) + assert rounds > 0, "no successful GC round ran at all" + assert reclaimed, ( + "none of the abandoned attempt's {} blob(s) were reclaimed, so their survival during the " + "stall does not distinguish the relink pin from an inactive GC".format(len(part_blobs)) + ) + + drop_everywhere(table) diff --git a/tests/integration/test_cas_s3/__init__.py b/tests/integration/test_cas_s3/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_s3/configs/storage_conf.xml b/tests/integration/test_cas_s3/configs/storage_conf.xml new file mode 100644 index 000000000000..55e2eedfc6fb --- /dev/null +++ b/tests/integration/test_cas_s3/configs/storage_conf.xml @@ -0,0 +1,36 @@ + + + + + object_storage + s3 + cas + + itest-content-addressed-s3 + + http://rustfs1:11121/test/cas_data/ + clickhouse + clickhouse + + 60 + 100 + 5000 + 7 + 3 +

X-Cas-Test: 1
+ +
+ + + +
+ disk_cas_s3 +
+
+
+
+ + diff --git a/tests/integration/test_cas_s3/test.py b/tests/integration/test_cas_s3/test.py new file mode 100644 index 000000000000..d8173d0cdf10 --- /dev/null +++ b/tests/integration/test_cas_s3/test.py @@ -0,0 +1,223 @@ +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +STORAGE_POLICY = "cas_s3" +NUM_ROWS = 1000 +CAS_PUBLICATION_EVENTS = ( + "CASBlobBodyPutAvoided", + "CASBlobHead", + "CASBlobHeadMiss", + "CASBlobPut", + "CASBlobUploadFanoutTasks", + "CASMetaCreateClean", +) + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node", + main_configs=["configs/storage_conf.xml"], + with_rustfs=True, + stay_alive=True, + ) + + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def cas_publication_events(node): + """Return process-wide CAS publication counters for an isolated before/after budget.""" + rows = node.query( + "SELECT event, value FROM system.events WHERE event IN ({}) FORMAT TSV".format( + ", ".join("'{}'".format(event) for event in CAS_PUBLICATION_EVENTS) + ) + ) + values = {event: 0 for event in CAS_PUBLICATION_EVENTS} + for row in rows.splitlines(): + event, value = row.split("\t") + values[event] = int(value) + return values + + +def event_delta(before, after): + return {event: after[event] - before[event] for event in CAS_PUBLICATION_EVENTS} + + +def test_disk_accepts_backend_settings_that_used_to_be_rejected(): + """The CAS disk block carries settings of its underlying object storage. + + Before the `cas_` namespace, the CAS settings scanned the whole disk element and rejected every + key they did not recognise, so `http_keep_alive_timeout` -- the mitigation suggested in #2243 -- + failed server startup. The server having started with the config this module installs is most of + the proof; this test states it, and checks the disk is actually usable rather than merely + present. + """ + node = cluster.instances["node"] + assert node.query( + "SELECT count() FROM system.disks WHERE name = 'disk_cas_s3'" + ).strip() == "1" + node.query("DROP TABLE IF EXISTS t_foreign_settings SYNC") + node.query( + "CREATE TABLE t_foreign_settings (a UInt64) ENGINE = MergeTree ORDER BY a " + "SETTINGS storage_policy = '{}'".format(STORAGE_POLICY) + ) + node.query("INSERT INTO t_foreign_settings SELECT number FROM numbers(100)") + assert node.query("SELECT sum(a) FROM t_foreign_settings").strip() == "4950" + node.query("DROP TABLE t_foreign_settings SYNC") + + +def test_cas_s3(): + node = cluster.instances["node"] + + node.query("DROP TABLE IF EXISTS cas_test SYNC") + node.query( + """ + CREATE TABLE cas_test ( + id Int64, + data String + ) ENGINE = MergeTree() + ORDER BY id + SETTINGS storage_policy = '{}' + """.format( + STORAGE_POLICY + ) + ) + + # First insert of NUM_ROWS deterministic rows. The RustFS lane has no concurrent query writer, + # so process-wide ProfileEvents form an exact request budget for this operation. + before_fresh = cas_publication_events(node) + node.query( + "INSERT INTO cas_test SELECT number, toString(number) FROM numbers({})".format( + NUM_ROWS + ) + ) + fresh = event_delta(before_fresh, cas_publication_events(node)) + assert fresh["CASBlobUploadFanoutTasks"] > 0, fresh + assert ( + fresh["CASBlobHead"] + fresh["CASBlobHeadMiss"] + == fresh["CASBlobUploadFanoutTasks"] + ), fresh + assert fresh["CASBlobHeadMiss"] == fresh["CASBlobUploadFanoutTasks"], fresh + assert fresh["CASBlobBodyPutAvoided"] == 0, fresh + assert fresh["CASMetaCreateClean"] == fresh["CASBlobUploadFanoutTasks"], fresh + # `CASBlobPut` is namespace/path instrumentation: both the body and its `.meta` sibling live + # below `/blobs/`, so a fresh publication contributes exactly those two physical PUTs. + assert ( + fresh["CASBlobPut"] + == fresh["CASBlobUploadFanoutTasks"] + fresh["CASMetaCreateClean"] + ), fresh + + expected_sum = (NUM_ROWS - 1) * NUM_ROWS // 2 + assert int(node.query("SELECT count() FROM cas_test")) == NUM_ROWS + assert int(node.query("SELECT sum(id) FROM cas_test")) == expected_sum + + # A second identical insert: the row count doubles. Each part's content is identical, so the + # content-addressed disk deduplicates the blobs, but the logical row count must still double. + before_duplicate = cas_publication_events(node) + node.query( + "INSERT INTO cas_test SELECT number, toString(number) FROM numbers({})".format( + NUM_ROWS + ) + ) + duplicate = event_delta(before_duplicate, cas_publication_events(node)) + assert duplicate["CASBlobUploadFanoutTasks"] > 0, duplicate + assert ( + duplicate["CASBlobHead"] + duplicate["CASBlobHeadMiss"] + == duplicate["CASBlobUploadFanoutTasks"] + ), duplicate + assert duplicate["CASBlobHead"] == duplicate["CASBlobUploadFanoutTasks"], duplicate + assert duplicate["CASBlobHeadMiss"] == 0, duplicate + assert duplicate["CASBlobPut"] == 0, duplicate + assert duplicate["CASBlobBodyPutAvoided"] == duplicate["CASBlobHead"], duplicate + assert duplicate["CASMetaCreateClean"] == 0, duplicate + assert int(node.query("SELECT count() FROM cas_test")) == 2 * NUM_ROWS + assert int(node.query("SELECT sum(id) FROM cas_test")) == 2 * expected_sum + + # Merge the two parts together. + node.query("OPTIMIZE TABLE cas_test FINAL") + assert int(node.query("SELECT count() FROM cas_test")) == 2 * NUM_ROWS + assert int(node.query("SELECT sum(id) FROM cas_test")) == 2 * expected_sum + + # Persistence: after a restart the refs/footers/blobs in S3 must still resolve the data. + node.restart_clickhouse() + + assert int(node.query("SELECT count() FROM cas_test")) == 2 * NUM_ROWS + assert int(node.query("SELECT sum(id) FROM cas_test")) == 2 * expected_sum + + # Drop must complete without error (ref unlink + deferred GC). + node.query("DROP TABLE cas_test SYNC") + assert ( + node.query( + "SELECT count() FROM system.tables WHERE database = currentDatabase() AND name = 'cas_test'" + ).strip() + == "0" + ) + + +def test_mutations_and_patch_parts_survive_restart(): + # A mutated part and a patch part are ordinary content-addressed parts published as refs. After a + # restart the active set must be rediscovered from the refs in S3, so the post-mutation / + # post-lightweight-delete state must survive (CAS M7). + node = cluster.instances["node"] + + node.query("DROP TABLE IF EXISTS cas_mut SYNC") + node.query( + """ + CREATE TABLE cas_mut ( + id Int64, + v UInt64, + s String + ) ENGINE = MergeTree() + ORDER BY id + SETTINGS storage_policy = '{}', enable_block_number_column = 1, enable_block_offset_column = 1 + """.format( + STORAGE_POLICY + ) + ) + + node.query( + "INSERT INTO cas_mut SELECT number, number * 10, toString(number) FROM numbers({})".format( + NUM_ROWS + ) + ) + + # Heavy mutation: UPDATE one column (id/s carry forward by reference on the content-addressed disk). + node.query( + "ALTER TABLE cas_mut UPDATE v = v + 1 WHERE id % 2 = 0 SETTINGS mutations_sync = 2" + ) + # Heavy mutation: DELETE. + node.query("ALTER TABLE cas_mut DELETE WHERE id % 100 = 0 SETTINGS mutations_sync = 2") + # Data-ALTER (column type change). Via a storage policy there is no inline-disk CustomType in + # settings_changes, so this works on the content-addressed disk (see backlog B53). + node.query("ALTER TABLE cas_mut MODIFY COLUMN v Int64 SETTINGS mutations_sync = 2") + # Patch part: a forced lightweight-update DELETE (throws if unsupported, so success == patch path). + node.query( + "DELETE FROM cas_mut WHERE id % 7 = 0 " + "SETTINGS enable_lightweight_update = 1, lightweight_delete_mode = 'lightweight_update_force', lightweight_deletes_sync = 2" + ) + + count_before = int(node.query("SELECT count() FROM cas_mut")) + sum_before = int(node.query("SELECT sum(v) FROM cas_mut")) + digest_before = node.query("SELECT sum(cityHash64(id, v, s)) FROM cas_mut").strip() + + # Persistence: rediscover the active set (incl. the mutated and patch parts) from S3 refs. + node.restart_clickhouse() + + assert int(node.query("SELECT count() FROM cas_mut")) == count_before + assert int(node.query("SELECT sum(v) FROM cas_mut")) == sum_before + assert node.query("SELECT sum(cityHash64(id, v, s)) FROM cas_mut").strip() == digest_before + + node.query("DROP TABLE cas_mut SYNC") + assert ( + node.query( + "SELECT count() FROM system.tables WHERE database = currentDatabase() AND name = 'cas_mut'" + ).strip() + == "0" + ) diff --git a/tests/integration/test_cas_shared_pool/__init__.py b/tests/integration/test_cas_shared_pool/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_shared_pool/configs/server_root_id_node1.xml b/tests/integration/test_cas_shared_pool/configs/server_root_id_node1.xml new file mode 100644 index 000000000000..3104d621390d --- /dev/null +++ b/tests/integration/test_cas_shared_pool/configs/server_root_id_node1.xml @@ -0,0 +1,12 @@ + + + + + + node1 + + + + diff --git a/tests/integration/test_cas_shared_pool/configs/server_root_id_node2.xml b/tests/integration/test_cas_shared_pool/configs/server_root_id_node2.xml new file mode 100644 index 000000000000..960bd079825a --- /dev/null +++ b/tests/integration/test_cas_shared_pool/configs/server_root_id_node2.xml @@ -0,0 +1,12 @@ + + + + + + node2 + + + + diff --git a/tests/integration/test_cas_shared_pool/configs/storage_conf.xml b/tests/integration/test_cas_shared_pool/configs/storage_conf.xml new file mode 100644 index 000000000000..a29a03741ef9 --- /dev/null +++ b/tests/integration/test_cas_shared_pool/configs/storage_conf.xml @@ -0,0 +1,36 @@ + + + + + object_storage + s3 + cas + + + http://rustfs1:11121/test/shared_pool/ + clickhouse + clickhouse + + + 1 + 1 + + + + + +
+ disk_cas_shared +
+
+
+
+
+
diff --git a/tests/integration/test_cas_shared_pool/test.py b/tests/integration/test_cas_shared_pool/test.py new file mode 100644 index 000000000000..5becc8dc049a --- /dev/null +++ b/tests/integration/test_cas_shared_pool/test.py @@ -0,0 +1,347 @@ +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +# Both servers mount the SAME content-addressed pool (endpoint .../root/shared_pool/). The blob pool +# (blobs/ + parts/) is shared across servers; refs are per-server under store//..., so the +# two servers dedup identical content while keeping independent ref roots. +STORAGE_POLICY = "cas_shared" + +# blobs/ holds content blobs, parts/ holds part footers. These are the shared pool's object prefixes +# inside the `root` MinIO bucket. "No leftovers" means BOTH drain back to baseline. +BLOBS_PREFIX = "shared_pool/blobs/" +PARTS_PREFIX = "shared_pool/parts/" + +# Deterministic data. Identical rows on both nodes => identical content blobs => cross-server dedup. +NUM_ROWS = 100000 + +# Background GC: grace=3s, interval=1s. After both DROP ... SYNC the pool's objects become +# unreferenced and a sweep (run by either server) reclaims them after grace. Bounded poll: this waits +# on a known background process, it is not papering over a race. +RECLAIM_RETRIES = 60 +RECLAIM_SLEEP = 1.0 # seconds; total bound ~= 60s + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + # RustFS (not MinIO) backs the pool: the CA mount capability probe requires enforced + # conditional-DELETE semantics, which MinIO OSS lacks — the fail-closed probe aborted server + # startup there (PR#2073 CI triage). Both instances reach the shared rustfs1 and load the + # identical storage_conf.xml, so both mount the SAME shared pool. + cluster.add_instance( + "node1", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node1.xml"], + with_rustfs=True, + stay_alive=True, + ) + cluster.add_instance( + "node2", + main_configs=["configs/storage_conf.xml", "configs/server_root_id_node2.xml"], + with_rustfs=True, + stay_alive=True, + ) + + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def count_prefix(prefix): + objects = cluster.rustfs_client.list_objects( + cluster.rustfs_bucket, prefix, recursive=True + ) + return len(list(objects)) + + + +def _gc_bookkeeping(*nodes): + """Pool-wide (successful rounds, objects deleted) from the CA GC log. The GC lease is held by ONE + server per pool and which one is not fixed, so both must be asked.""" + rounds = deleted = 0 + for n in nodes: + n.query("SYSTEM FLUSH LOGS") + rounds += int( + n.query( + "SELECT count() FROM system.cas_gc_log " + "WHERE event_type = 'Finish' AND outcome = 'Success'" + ).strip() + or 0 + ) + deleted += int( + n.query( + "SELECT sum(objects_deleted + manifests_deleted + entries_redeleted) " + "FROM system.cas_gc_log WHERE event_type = 'Finish'" + ).strip() + or 0 + ) + return rounds, deleted + +def count_pool_objects(): + # The shared pool is empty only when BOTH content blobs and part footers are gone. + return count_prefix(BLOBS_PREFIX) + count_prefix(PARTS_PREFIX) + + +def test_two_servers_share_one_pool(): + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + + node1.query("DROP TABLE IF EXISTS t1 SYNC") + node2.query("DROP TABLE IF EXISTS t2 SYNC") + + # (0) Baseline pool object count before either table exists. + baseline = count_pool_objects() + + # (1) Each server creates its OWN MergeTree table on the shared pool. Distinct names => distinct + # table UUIDs => independent per-server refs, but the SAME shared blob pool. + create_tpl = ( + "CREATE TABLE {tbl} (id Int64, v UInt64, s String) " + "ENGINE = MergeTree() ORDER BY id " + "SETTINGS storage_policy = '{policy}'" + ) + node1.query(create_tpl.format(tbl="t1", policy=STORAGE_POLICY)) + node2.query(create_tpl.format(tbl="t2", policy=STORAGE_POLICY)) + + # (2) INSERT IDENTICAL deterministic data into both. The content blobs are byte-identical, so the + # shared pool dedups them across the two servers. Logical reads must still be correct on each. + insert_tpl = ( + "INSERT INTO {tbl} " + "SELECT number, number * 10, toString(number) FROM numbers({rows})" + ) + node1.query(insert_tpl.format(tbl="t1", rows=NUM_ROWS)) + node2.query(insert_tpl.format(tbl="t2", rows=NUM_ROWS)) + + expected_sum_id = (NUM_ROWS - 1) * NUM_ROWS // 2 + assert int(node1.query("SELECT count() FROM t1")) == NUM_ROWS + assert int(node2.query("SELECT count() FROM t2")) == NUM_ROWS + assert int(node1.query("SELECT sum(id) FROM t1")) == expected_sum_id + assert int(node2.query("SELECT sum(id) FROM t2")) == expected_sum_id + + # Cross-server dedup sanity: the two identical single-part inserts must NOT have doubled the pool's + # blob count. With dedup the blob count after both inserts is well below twice the per-server count. + after_insert = count_pool_objects() + assert after_insert > baseline, ( + "expected pool object count to rise above baseline {} after inserts, got {}".format( + baseline, after_insert + ) + ) + + # (3) Heavy mutations / merges on EACH server, in parallel ownership of the shared pool. + # UPDATE (id/s carry forward by reference), DELETE, then OPTIMIZE FINAL. + node1.query("ALTER TABLE t1 UPDATE v = v + 1 WHERE id % 2 = 0 SETTINGS mutations_sync = 2") + node2.query("ALTER TABLE t2 UPDATE v = v + 1 WHERE id % 2 = 0 SETTINGS mutations_sync = 2") + + node1.query("ALTER TABLE t1 DELETE WHERE id % 100 = 0 SETTINGS mutations_sync = 2") + node2.query("ALTER TABLE t2 DELETE WHERE id % 100 = 0 SETTINGS mutations_sync = 2") + + node1.query("OPTIMIZE TABLE t1 FINAL") + node2.query("OPTIMIZE TABLE t2 FINAL") + + # Post-mutation expected aggregates (identical recipe on both, so both must match). + count_after_mut = int(node1.query("SELECT count() FROM t1")) + sum_after_mut = int(node1.query("SELECT sum(v) FROM t1")) + digest_after_mut = node1.query("SELECT sum(cityHash64(id, v, s)) FROM t1").strip() + + assert int(node2.query("SELECT count() FROM t2")) == count_after_mut + assert int(node2.query("SELECT sum(v) FROM t2")) == sum_after_mut + assert node2.query("SELECT sum(cityHash64(id, v, s)) FROM t2").strip() == digest_after_mut + + # (4) Let the background GC (enabled on BOTH servers, short grace) run several sweep cycles while + # both tables are still live. The cross-server safety property: a sweep run by either server + # must NOT reclaim a blob that the OTHER server's live part references (deduped/shared blob). + # Sleeping here is waiting on the known background sweep cadence, not a race workaround. + time.sleep(3 * RECLAIM_SLEEP + 3) # > grace(3s) + a few interval(1s) cycles + + # Re-read on BOTH servers: no data lost to the other server's GC. + assert int(node1.query("SELECT count() FROM t1")) == count_after_mut + assert int(node1.query("SELECT sum(v) FROM t1")) == sum_after_mut + assert node1.query("SELECT sum(cityHash64(id, v, s)) FROM t1").strip() == digest_after_mut + + assert int(node2.query("SELECT count() FROM t2")) == count_after_mut + assert int(node2.query("SELECT sum(v) FROM t2")) == sum_after_mut + assert node2.query("SELECT sum(cityHash64(id, v, s)) FROM t2").strip() == digest_after_mut + + # (5) Both servers drop their tables. Refs are unlinked synchronously; the shared pool's blobs and + # footers become unreferenced GC fodder. Then poll until the pool drains back to baseline. + node1.query("DROP TABLE t1 SYNC") + node2.query("DROP TABLE t2 SYNC") + + at_drop = count_pool_objects() + + # THE RECLAMATION: both servers' content goes. Polled with an early exit, then cross-checked + # against GC's own bookkeeping so a pool that shrank for some other reason cannot pass for a round + # that reclaimed it. + final = count_pool_objects() + for _ in range(RECLAIM_RETRIES): + if final <= baseline: + break + time.sleep(RECLAIM_SLEEP) + final = count_pool_objects() + + assert final <= baseline, ( + "the shared pool did not drain after both servers dropped: " + "baseline={}, after_insert={}, at_drop={}, final={} (blobs={}, parts={})".format( + baseline, + after_insert, + at_drop, + final, + count_prefix(BLOBS_PREFIX), + count_prefix(PARTS_PREFIX), + ) + ) + + # Counted POOL-WIDE: exactly one server holds the GC lease for a shared pool, and it need not be + # node1 — asking only node1 yields 0 rounds whenever node2 is the leader, which is how this + # assertion first failed. + rounds, deleted = _gc_bookkeeping(node1, node2) + assert rounds > 0, "no successful GC round ran at all" + assert deleted > 0, "the shared pool drained but GC's own bookkeeping reports no deletion" + + +# Crash-resilience uses a SMALLER, DISTINCT dataset per node. Distinct content => node1's blobs are +# NOT deduped with node2's, so "node2's GC must not reclaim node1's blobs while node1 is down" is a +# real, observable invariant on the pool object count (node1's blobs cannot hide behind node2's). +CRASH_ROWS = 50000 + + +def test_pool_survives_node_crash(): + # Proves the bucket is self-describing and the pool survives a hard node crash: + # (a) the surviving node keeps running with background GC on and loses no data; + # (b) the hard-killed node recovers its data on restart (refs are durable in the bucket); + # (c) any orphaned write-session the crash left behind is eventually reclaimed (its lease + # expires; the pool drains to baseline after DROP). + # Lock-fencing safety (paused GC leader fenced by a peer's higher fence token) is covered by the + # gtest SweepStopsWhenLeadershipLost; this test focuses on crash-resilience. + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + + node1.query("DROP TABLE IF EXISTS crash1 SYNC") + node2.query("DROP TABLE IF EXISTS crash2 SYNC") + + # (0) Baseline pool object count before either table exists. + baseline = count_pool_objects() + + create_tpl = ( + "CREATE TABLE {tbl} (id Int64, v UInt64, s String) " + "ENGINE = MergeTree() ORDER BY id " + "SETTINGS storage_policy = '{policy}'" + ) + node1.query(create_tpl.format(tbl="crash1", policy=STORAGE_POLICY)) + node2.query(create_tpl.format(tbl="crash2", policy=STORAGE_POLICY)) + + # (1) DISTINCT deterministic data per node (different `v` recipe => different content blobs, so + # node1's blobs are NOT shared with node2's and cannot be hidden behind dedup). + node1.query( + "INSERT INTO crash1 SELECT number, number * 10, toString(number) " + "FROM numbers({rows})".format(rows=CRASH_ROWS) + ) + node2.query( + "INSERT INTO crash2 SELECT number, number * 7, concat('n2_', toString(number)) " + "FROM numbers({rows})".format(rows=CRASH_ROWS) + ) + + # Capture node1's authoritative aggregates BEFORE the crash; recovery must reproduce them exactly. + n1_count = int(node1.query("SELECT count() FROM crash1")) + n1_sum_id = int(node1.query("SELECT sum(id) FROM crash1")) + n1_digest = node1.query("SELECT sum(cityHash64(id, v, s)) FROM crash1").strip() + assert n1_count == CRASH_ROWS + + n2_count = int(node2.query("SELECT count() FROM crash2")) + n2_digest = node2.query("SELECT sum(cityHash64(id, v, s)) FROM crash2").strip() + assert n2_count == CRASH_ROWS + + # The pool now holds BOTH nodes' (distinct) blobs. Remember this high-water mark: after node1 is + # killed, node2's GC must NOT shrink the pool below the level needed to hold node1's blobs. + after_both_inserts = count_pool_objects() + assert after_both_inserts > baseline + + # (2) HARD-KILL node1 (SIGKILL via pkill -9 => simulated crash). node2 stays up. A crash mid-flight + # can leave node1 holding an unreleased write-session lease (an orphaned pin) on the pool. + node1.stop_clickhouse(kill=True) + + # (3) With node1 down, keep node2 working AND let node2's background GC run several sweep cycles. + # Two invariants: + # - node2 reads its OWN data correctly (no loss while it owns the pool alone); + # - node2's GC does NOT reclaim node1's blobs: node1's refs are durable roots in the bucket + # even though node1's process is gone. Sleeping here waits on the known sweep cadence. + node2.query( + "INSERT INTO crash2 SELECT number, number * 7, concat('n2b_', toString(number)) " + "FROM numbers({rows})".format(rows=CRASH_ROWS) + ) + node2.query("OPTIMIZE TABLE crash2 FINAL") + n2_count_after = int(node2.query("SELECT count() FROM crash2")) + assert n2_count_after == 2 * CRASH_ROWS + + time.sleep(3 * RECLAIM_SLEEP + 3) # > grace(3s) + a few interval(1s) cycles of node2's GC + + # node2 lost nothing. + assert int(node2.query("SELECT count() FROM crash2")) == n2_count_after + # node1's blobs were NOT swept by node2's GC: the pool still holds at least node1's portion. node1 + # contributed (after_both_inserts - baseline) objects on top of the empty baseline, so even if + # node2 had reclaimed every one of its own blobs the pool could not have dropped below that. + node1_contribution = after_both_inserts - baseline + pool_with_node1_down = count_pool_objects() + assert pool_with_node1_down >= baseline + node1_contribution, ( + "node2's GC appears to have reclaimed node1's durable refs while node1 was down: " + "baseline={}, after_both_inserts={}, node1_contribution={}, pool_now={}".format( + baseline, after_both_inserts, node1_contribution, pool_with_node1_down + ) + ) + + # (4) RESTART node1. The bucket is self-describing: node1 rebuilds its active set from the durable + # refs and must re-read its table with the EXACT pre-crash count/sum/digest. After a hard kill + # the harness reconnects on start_clickhouse via wait_start; use the instance object fresh. + # 150s, not the 60s default: a post-SIGKILL restart legitimately pays the unclean-reclaim + # cost before serving — the stale-token observation window over its own unexpired lease + # (~TTL + 5% + renew_period/2 ≈ 36.5s with defaults) plus the materialization grace + # (30s default) plus the lease re-write; ~71s observed end-to-end. A bounded wait on a + # known, by-design recovery protocol — not a race hack. + node1.start_clickhouse(start_wait_sec=150) + + assert int(node1.query("SELECT count() FROM crash1")) == n1_count + assert int(node1.query("SELECT sum(id) FROM crash1")) == n1_sum_id + assert node1.query("SELECT sum(cityHash64(id, v, s)) FROM crash1").strip() == n1_digest + + # node2 still consistent after node1 rejoined. + assert int(node2.query("SELECT count() FROM crash2")) == n2_count_after + + # (5) DROP both tables. Refs are unlinked synchronously; the orphaned write-session that node1's + # crash left behind no longer pins anything once its lease expires, so GC (run by either + # server) reclaims the lot. Bounded-poll the pool until it drains back to baseline. + node1.query("DROP TABLE crash1 SYNC") + node2.query("DROP TABLE crash2 SYNC") + + at_drop = count_pool_objects() + + # THE RECLAMATION, and the point of this test: the hard kill left nothing behind that survives the + # drop. Once both tables are gone and the orphaned write-session's lease expires, nothing pins the + # content and the pool returns to baseline. + final = count_pool_objects() + for _ in range(RECLAIM_RETRIES): + if final <= baseline: + break + time.sleep(RECLAIM_SLEEP) + final = count_pool_objects() + + assert final <= baseline, ( + "the shared pool did not drain after the crash + both DROPs: " + "baseline={}, after_both_inserts={}, at_drop={}, " + "final={} (blobs={}, parts={})".format( + baseline, + after_both_inserts, + at_drop, + final, + count_prefix(BLOBS_PREFIX), + count_prefix(PARTS_PREFIX), + ) + ) + + # Counted POOL-WIDE, for the same leader-may-be-either-node reason as the first test in this file. + rounds, deleted = _gc_bookkeeping(node1, node2) + assert rounds > 0, "no successful GC round ran at all" + assert deleted > 0, "the shared pool drained but GC's own bookkeeping reports no deletion" diff --git a/tests/integration/test_disks_app_func/test.py b/tests/integration/test_disks_app_func/test.py index 87ab2e24b9f3..b635f6aad682 100755 --- a/tests/integration/test_disks_app_func/test.py +++ b/tests/integration/test_disks_app_func/test.py @@ -178,7 +178,7 @@ def init_data_s3_rm_rec(source): write(source, "test3", "a/b/d") write(source, "test3", "a/b/e") - write(source, "test3", "d/a") + write(source, "test3", "a/d/a") def test_disks_app_func_ld(started_cluster): @@ -330,22 +330,22 @@ def test_disks_app_func_rm_shared_recursive(started_cluster): out = ls(source, "test3", ". --recursive") assert ( out - == ".:\na\n\n./a:\na\nb\nc\nd\n\n./a/a:\na\nb\nc\n\n./a/b:\na\nb\nc\nd\ne\n\n./a/c:\n\n./a/d:\n\n" + == ".:\na\n\n./a:\na\nb\nc\nd\n\n./a/a:\na\nb\nc\n\n./a/b:\na\nb\nc\nd\ne\n\n./a/c:\n\n./a/d:\na\n\n" ) remove(source, "test3", "a/a --recursive") out = ls(source, "test3", ". --recursive") assert ( - out == ".:\na\n\n./a:\nb\nc\nd\n\n./a/b:\na\nb\nc\nd\ne\n\n./a/c:\n\n./a/d:\n\n" + out == ".:\na\n\n./a:\nb\nc\nd\n\n./a/b:\na\nb\nc\nd\ne\n\n./a/c:\n\n./a/d:\na\n\n" ) remove(source, "test3", "a/b --recursive") out = ls(source, "test3", ". --recursive") - assert out == ".:\na\n\n./a:\nc\nd\n\n./a/c:\n\n./a/d:\n\n" + assert out == ".:\na\n\n./a:\nc\nd\n\n./a/c:\n\n./a/d:\na\n\n" remove(source, "test3", "a/c --recursive") out = ls(source, "test3", ". --recursive") - assert out == ".:\na\n\n./a:\nd\n\n./a/d:\n\n" + assert out == ".:\na\n\n./a:\nd\n\n./a/d:\na\n\n" remove(source, "test3", "a --recursive") out = ls(source, "test3", ". --recursive") diff --git a/tests/integration/test_gcs_live/.gitignore b/tests/integration/test_gcs_live/.gitignore new file mode 100644 index 000000000000..ea1db076ef2d --- /dev/null +++ b/tests/integration/test_gcs_live/.gitignore @@ -0,0 +1,3 @@ +# Generated per run from the GCS_LIVE_* environment variables. It embeds the live HMAC secret in +# plain text, so it must never be committed. +configs/live_gcs_generated.xml diff --git a/tests/integration/test_gcs_live/__init__.py b/tests/integration/test_gcs_live/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_gcs_live/test.py b/tests/integration/test_gcs_live/test.py new file mode 100644 index 000000000000..f2fdd5c83ef6 --- /dev/null +++ b/tests/integration/test_gcs_live/test.py @@ -0,0 +1,1532 @@ +"""The live-GCS characterization gate for the two native GCS HTTP clients. + +## Why this suite exists, and what nothing else can replace + +The unit tests prove which headers a request carries, and `test_cas_gcs` proves that a real CAS mount +still works once generation semantics stop being applied to every request on the client. Neither can +prove that *Google* accepts the resulting authenticated requests. A fake models whatever we assumed +when we wrote it, so a green `test_cas_gcs` is not evidence for anything below. + +Two things in particular can only be settled here: + + - `deduceProviderType` is pure endpoint-substring matching, and the whole `ApiMode` block in + `Client::BuildHttpRequest` is nested under `provider_type == ProviderType::GCS`. `test_cas_gcs` + deliberately uses a hostname containing no `storage.googleapis.com`, so `provider_type` is UNKNOWN + there and the api-mode transformations never run. Against a real endpoint they DO, and they run + underneath the request-mode logic. + - Whether the GOOG4 signed-header allowlist produces a signature Google actually accepts. + +## Gating + +Every live group is opt-in through environment variables and skips cleanly when they are absent. +The two credential-free helper regressions run by default, but they use only local files and a +synthetic query result. No test that touches a real bucket or issues billable requests runs by default. + + - `GCS_LIVE_BUCKET` — required for any group. A bucket the caller is willing to have + objects created and deleted in. + - `GCS_LIVE_PREFIX` — optional key prefix, default `clickhouse-gcs-live-gate`. A random + per-run suffix is always appended, so two concurrent runs cannot + share a prefix. + - `GCS_LIVE_HMAC_ACCESS_KEY_ID` — a GOOG4 HMAC key pair. Enables the ordinary and CAS GOOG4 + scenarios. + - `GCS_LIVE_HMAC_SECRET_ACCESS_KEY` + - `GCS_LIVE_OAUTH_FROM_METADATA=1` — declares that the HOST running this suite can reach the GCE + metadata server and that its service account may write to the + bucket. Enables the ordinary and CAS OAuth scenarios on GCE. + - `GCS_LIVE_OAUTH_ADC_CLIENT_ID` — Application Default Credentials, the alternative that enables + - `GCS_LIVE_OAUTH_ADC_CLIENT_SECRET` OAuth scenarios from anywhere, not just on GCE. A CAS disk accepts + - `GCS_LIVE_OAUTH_ADC_REFRESH_TOKEN` these because CAS consumes only its `cas_` namespace and leaves + `metadata_service`, `request_token_path`, `service_account`, and the + ADC triple to the underlying object storage. Either source is enough; + the ADC one exists because requiring a GCE host is what would keep + this gate from ever being run. + - `GCS_LIVE_AMBIGUITY_PROXY_URI` — URI of the operator-controlled TLS fault proxy. Together with + - `GCS_LIVE_AMBIGUITY_CONTROL_URL` its control URL and public CA file, enables the fault arms. + - `GCS_LIVE_AMBIGUITY_CA_FILE` The terminating proxy's public CA bundle. ClickHouse retains + strict certificate verification and trusts this file in addition + to the image's default CA roots. + The proxy must meet the phase contract documented on + `test_live_cas_ambiguous_staged_copy_absent_retry_uses_retagged_replacement`. + Both URLs must be credential-free endpoints. Authentication + material remains entirely inside the operator's proxy and the + ClickHouse credential variables already listed above. The two + control contracts are documented on the queued-delete and + staged-copy-ambiguity tests. + +Only disks whose gates are satisfied are written into the configuration. That is deliberate: a CAS +disk mounts and runs its capability battery at server startup with no fallback, so an unusable CAS +disk in the config would stop the server and take the other groups down with it. + +## What this gate asserts, and what it deliberately does not + +It asserts what a client can observe: that each operation SUCCEEDS against Google, that ordinary +non-CAS requests retain their ETag-based contract, that CAS records generations rather than ETags, +and that each named body-publication action was actually selected. The Task 10 cases use a statement +query id in `system.cas_log` and `system.query_log`, so unrelated background work cannot satisfy them. +The older ordinary characterization still uses process-wide `system.events`; its limitations remain +spelled out below rather than being silently hidden. + +## OPEN QUESTION FOR WHOEVER FIRST RUNS THIS WITH CREDENTIALS + +`system.events` counters are PROCESS-WIDE, and this configuration also holds several CAS disks whose +control writers issue object-storage requests of their own. Their GC schedulers are stopped before the +tests, but mount leases and other control work still exist. So every ordinary counter delta asserted here +is only as sound as the assumption that no CAS activity moved that counter inside the measured window. +Where that assumption fails, the assertion still passes — for a reason that has nothing to do with the +statement it names. + +This is an open question, not a known defect: which counters CAS can actually move during these +windows is not determinable without a real run. It is written here rather than in a tracked item +because the first run is when it matters and this docstring is what its reader will have in front of +them. **On that run, check each counter individually instead of trusting a pass** — for any counter CAS +can move, a green assertion is not evidence that the statement under test issued the operation. + +One test is EXEMPT, and the reason is the template for clearing the others: +`test_default_gcs_client_parquet_metadata_cache_keys_on_the_ordinary_etag` uses +`ParquetMetadataCacheMisses` and `ParquetMetadataCacheHits`, which only a Parquet read moves. No CAS +disk can touch either, so those two deltas mean exactly what they say. Clearing a counter means showing +that same thing about it — not observing it pass. + +One instance is already settled and serves as the pattern for the other direction. `S3ListObjects` was +asserted here and has been removed: an ordinary MergeTree lifecycle on a local-metadata disk never lists, so it could not +have been satisfied by this workload at all — but the CAS disks in this same configuration DO list, so +a background collection round inside the window could have satisfied it anyway. That is exactly the +failure mode above, and it is why "make something list somehow" would have produced a test passing for +the wrong reason rather than a working one. + +It does NOT assert the outbound header set — that `x-goog-if-generation-match` appears on the wire, +that `x-amz-date` / `x-amz-content-sha256` / `x-amz-security-token` / `x-amz-api-version` are absent, +or which headers the GOOG4 signature covers. That is a scope decision, not an impossibility, and the +alternatives considered were each worse than the gap: + + - `PocoHTTPClient` logs RESPONSE headers under `enable_s3_requests_logging` and never logs the + request headers, so the server log cannot supply them. + - A plain forward proxy would have to be named as the endpoint, which makes `deduceProviderType` + report UNKNOWN and switches off the very `ApiMode` behaviour this suite exists to exercise. + - Downgrading to plain HTTP so a proxy can read the headers puts live credentials in clear text on + the wire. + - A TLS-TERMINATING proxy does work and is the honest option: `endpoint` stays + `storage.googleapis.com`, so `provider_type` is still GCS and `ApiMode::GCS` stays active, while + the proxy observes plaintext request headers inside a process the test operator already controls — + the same trust boundary as the container that already holds the plaintext HMAC secret in its + config. It is not built here because it needs a proxy container, a generated CA distributed into + the server's trust store, and per-disk proxy configuration: real infrastructure for a property the + unit tests already establish by inspecting the request object directly, with no network at all. + +So the outbound header set stays with the unit tests. What is left for this gate is acceptance — and +acceptance is the part a unit test structurally cannot reach. +""" + +import hashlib +import html +import json +import os +import random +import string +import threading +import time +import urllib.parse +import urllib.request +from concurrent.futures import ThreadPoolExecutor + +import pytest + +from helpers.cluster import ClickHouseCluster, ClickHouseInstance + +BUCKET = os.environ.get("GCS_LIVE_BUCKET", "") +BASE_PREFIX = os.environ.get("GCS_LIVE_PREFIX", "clickhouse-gcs-live-gate") +HMAC_KEY_ID = os.environ.get("GCS_LIVE_HMAC_ACCESS_KEY_ID", "") +HMAC_SECRET = os.environ.get("GCS_LIVE_HMAC_SECRET_ACCESS_KEY", "") +OAUTH_FROM_METADATA = os.environ.get("GCS_LIVE_OAUTH_FROM_METADATA", "") == "1" +ADC_CLIENT_ID = os.environ.get("GCS_LIVE_OAUTH_ADC_CLIENT_ID", "") +ADC_CLIENT_SECRET = os.environ.get("GCS_LIVE_OAUTH_ADC_CLIENT_SECRET", "") +ADC_REFRESH_TOKEN = os.environ.get("GCS_LIVE_OAUTH_ADC_REFRESH_TOKEN", "") +ADC_AVAILABLE = bool(ADC_CLIENT_ID and ADC_CLIENT_SECRET and ADC_REFRESH_TOKEN) + +# The ambiguity arm needs infrastructure that can terminate the TLS connection after Google accepts +# a native copy, exact-delete that landed generation, and then let the writer retry. The proxy URI is +# consumed by the global HTTPS client configuration; the control URL arms the one-shot fault and +# exposes a credential-free phase report. +# None of these values carries credentials: the third is a public trust anchor. Merely having +# ordinary GCS credentials is not enough to make an ambiguous outcome controllable, so this remains a +# separate release gate rather than a best-effort timing race. +AMBIGUITY_PROXY_URI = os.environ.get("GCS_LIVE_AMBIGUITY_PROXY_URI", "") +AMBIGUITY_CONTROL_URL = os.environ.get("GCS_LIVE_AMBIGUITY_CONTROL_URL", "") +AMBIGUITY_CA_FILE = os.environ.get("GCS_LIVE_AMBIGUITY_CA_FILE", "") + + +def _credential_free_url(value): + parsed = urllib.parse.urlsplit(value) + return bool(parsed.scheme in ("http", "https") and parsed.netloc and parsed.username is None and parsed.password is None and not parsed.query and not parsed.fragment) + + +AMBIGUITY_DRIVER_AVAILABLE = bool( + AMBIGUITY_PROXY_URI + and AMBIGUITY_CONTROL_URL + and AMBIGUITY_CA_FILE + and os.path.isfile(AMBIGUITY_CA_FILE) + and _credential_free_url(AMBIGUITY_PROXY_URI) + and _credential_free_url(AMBIGUITY_CONTROL_URL) +) + +HMAC_AVAILABLE = bool(BUCKET and HMAC_KEY_ID and HMAC_SECRET) +# Either token source satisfies OAuth — the GCE metadata server, or Application Default Credentials. +OAUTH_AVAILABLE = bool(BUCKET and (OAUTH_FROM_METADATA or ADC_AVAILABLE)) + +# The endpoint must be spelled with `storage.googleapis.com`, not a regional or private alias: that +# substring is the whole of `deduceProviderType`, and the api-mode transformations this gate exists to +# exercise are nested under the provider it deduces. +GCS_ENDPOINT = "https://storage.googleapis.com" + +RUN_ID = "".join(random.choice(string.ascii_lowercase + string.digits) for _ in range(12)) +PREFIX = "{}/{}".format(BASE_PREFIX.strip("/"), RUN_ID) + +HMAC_PLAIN_DISK = "live_hmac_plain" +# A second ordinary disk exists only so a partition can be MOVED between two volumes of one policy. +# That is the one SQL statement that reaches a server-side `CopyObject`: `FREEZE` and +# `REPLACE PARTITION` hardlink the LOCAL metadata files and issue no object-storage copy at all, so a +# test built on them would leave `S3CopyObject` at zero and prove nothing about GCS accepting a copy. +HMAC_PLAIN_DISK_2 = "live_hmac_plain_cold" +HMAC_TWO_VOLUME_POLICY = "live_hmac_two_volume" +OAUTH_PLAIN_DISK = "live_oauth_plain" +OAUTH_PLAIN_DISK_2 = "live_oauth_plain_cold" +OAUTH_TWO_VOLUME_POLICY = "live_oauth_two_volume" +# An ordinary `gcs_hmac` disk pointed at a bucket that does not exist, so a refused request can be +# observed ON THE GOOG4 PATH. A disk rather than `s3(...)` because it reuses the configuration surface +# the rest of this group already exercises — NOT because the table function cannot select the client: +# `StorageS3Configuration::fromNamedCollection` does read `http_client`, which is what +# `PARQUET_NAMED_COLLECTION` below relies on. What has no spelling for it is the POSITIONAL argument +# form (`fromAST` sets `http_client` only through the BigLake ADC path, which forces `gcp_oauth`), so a +# bare `s3('url', 'key', 'secret')` would sign with ordinary AWS SigV4 and say nothing about GOOG4. +HMAC_ABSENT_BUCKET_DISK = "live_hmac_absent_bucket" +# Carries `http_client=gcs_hmac` into the object-storage TABLE ENGINE path, which is the only way to +# reach the Parquet metadata cache: that cache is consumed in `StorageObjectStorageSource` and +# `ParquetV3BlockInputFormat`, never by a MergeTree disk, so no statement on the disks above can touch +# it. +PARQUET_NAMED_COLLECTION = "live_gcs_hmac_parquet" +OAUTH_PARQUET_NAMED_COLLECTION = "live_gcs_oauth_parquet" +CAS_OAUTH_DISK = "live_cas_oauth" +CAS_HMAC_DISK = "live_cas_hmac" +CAS_OAUTH_STAGED_DISK = "live_cas_oauth_staged" +CAS_HMAC_STAGED_DISK = "live_cas_hmac_staged" +CAS_OAUTH_AMBIGUITY_DISK = "live_cas_oauth_ambiguity" +CAS_HMAC_AMBIGUITY_DISK = "live_cas_hmac_ambiguity" +AMBIGUITY_SUBPREFIX = { + "gcs_hmac": "cas-hmac-ambiguity", + "gcp_oauth": "cas-oauth-ambiguity", +} + +# Lowering the genuine-conditional ceiling makes a modest live payload prove that blob publication +# no longer inherits the former GCS-only size cliff. The blob body now uses Default mode and may be +# either a normal one-shot PUT or ordinary multipart; mutable CAS objects keep this ceiling. +FORMER_CONDITIONAL_PUT_CAP = 5 * 1024 * 1024 +LIVE_LARGE_VALUE_ITEMS = 400000 +BATCH_DELETE_LOG_PATTERN = r"Objects with paths \[" + +cluster = ClickHouseCluster(__file__) + + +def _xml(value): + """Escape a runtime value before placing it in the generated XML configuration.""" + return html.escape(str(value), quote=True) + + +def _disk_xml( + name, + subprefix, + cas, + client, + bucket=None, + skip_access_check=False, + staging_backend="local", +): + lines = [ + " <{}>".format(name), + " object_storage", + " s3", + ] + if skip_access_check: + # Only for the deliberately-unreachable-bucket disk, and only because that disk is NOT a CAS + # mount. `IDisk::startup` calls `checkAccess` and rethrows, and `DiskSelector::initialize` is a + # function-try-block with a single `catch (...)` around the whole construction loop — there is + # no per-disk isolation, so one unreachable disk aborts the entire selector build and every + # other disk in this file dies with it. `Server.cpp` hardcodes + # `registerDisks(global_skip_access_check=false)`, so the per-disk key is the only way out. + # + # NEVER put this on a `cas=True` disk. A writable generation-token CAS mount must refuse it, so + # that `runCapabilityProbe` cannot be bypassed — that battery is the only thing proving a + # token-exact DELETE really carries its generation precondition. + lines.append(" true") + if cas: + lines += [ + " cas", + " {}".format(name), + " 3600", + " {}".format(FORMER_CONDITIONAL_PUT_CAP), + " {}".format(staging_backend), + ] + lines += [ + " {}/{}/{}/{}/".format(GCS_ENDPOINT, _xml(bucket or BUCKET), _xml(PREFIX), _xml(subprefix)), + " {}".format(client), + ] + if client == "gcs_hmac": + lines += [ + " {}".format(_xml(HMAC_KEY_ID)), + " {}".format(_xml(HMAC_SECRET)), + ] + if client == "gcp_oauth" and ADC_AVAILABLE: + # `requestBearerToken` picks between the GCE metadata server and these; a CAS disk accepts them + # because CAS consumes only its `cas_` namespace. Present only when supplied, so a GCE run keeps + # using metadata. + lines += [ + " {}".format(_xml(ADC_CLIENT_ID)), + " {}".format(_xml(ADC_CLIENT_SECRET)), + " {}".format(_xml(ADC_REFRESH_TOKEN)), + ] + lines.append(" ".format(name)) + return "\n".join(lines) + + +def _policy_xml(name): + return " <{name}>\n
{name}
\n ".format(name=name) + + +def _two_volume_policy_xml(name, hot, cold): + return ( + " <{policy}>\n" + " \n" + " {hot}\n" + " {cold}\n" + " \n" + " ".format(policy=name, hot=hot, cold=cold) + ) + + +def _named_collection_xml(name, client): + lines = [ + " <{}>".format(name), + " {}/{}/{}/parquet/".format(GCS_ENDPOINT, _xml(BUCKET), _xml(PREFIX)), + " {}".format(client), + ] + if client == "gcs_hmac": + lines += [ + " {}".format(_xml(HMAC_KEY_ID)), + " {}".format(_xml(HMAC_SECRET)), + ] + elif ADC_AVAILABLE: + lines += [ + " {}".format(_xml(ADC_CLIENT_ID)), + " {}".format(_xml(ADC_CLIENT_SECRET)), + " {}".format(_xml(ADC_REFRESH_TOKEN)), + ] + lines.append(" ".format(name)) + return "\n".join(lines) + + +def _ambiguity_endpoint_settings_xml(): + entries = [] + for name, subprefix in ( + ("task10_hmac_ambiguity", AMBIGUITY_SUBPREFIX["gcs_hmac"]), + ("task10_oauth_ambiguity", AMBIGUITY_SUBPREFIX["gcp_oauth"]), + ): + endpoint = "{}/{}/{}/{}/".format(GCS_ENDPOINT, BUCKET, PREFIX, subprefix) + entries.append(" <{name}>\n {endpoint}\n 0\n ".format(name=name, endpoint=_xml(endpoint))) + return " \n{}\n \n".format("\n".join(entries)) + + +def _write_config(path): + """Build a storage configuration holding only the disks whose environment gates are satisfied.""" + disks = [] + policies = [] + if HMAC_AVAILABLE: + disks += [ + _disk_xml(HMAC_PLAIN_DISK, "plain", cas=False, client="gcs_hmac"), + _disk_xml(HMAC_PLAIN_DISK_2, "plain-cold", cas=False, client="gcs_hmac"), + _disk_xml( + HMAC_ABSENT_BUCKET_DISK, + "absent", + cas=False, + client="gcs_hmac", + bucket="clickhouse-gcs-live-gate-bucket-that-does-not-exist", + skip_access_check=True, + ), + _disk_xml(CAS_HMAC_DISK, "cas-hmac", cas=True, client="gcs_hmac"), + _disk_xml( + CAS_HMAC_STAGED_DISK, + "cas-hmac-staged", + cas=True, + client="gcs_hmac", + staging_backend="s3", + ), + ] + if AMBIGUITY_DRIVER_AVAILABLE: + disks.append( + _disk_xml( + CAS_HMAC_AMBIGUITY_DISK, + "cas-hmac-ambiguity", + cas=True, + client="gcs_hmac", + staging_backend="s3", + ) + ) + policies += [ + _policy_xml(HMAC_ABSENT_BUCKET_DISK), + _policy_xml(CAS_HMAC_DISK), + _policy_xml(CAS_HMAC_STAGED_DISK), + _two_volume_policy_xml(HMAC_TWO_VOLUME_POLICY, HMAC_PLAIN_DISK, HMAC_PLAIN_DISK_2), + ] + if AMBIGUITY_DRIVER_AVAILABLE: + policies.append(_policy_xml(CAS_HMAC_AMBIGUITY_DISK)) + if OAUTH_AVAILABLE: + disks += [ + _disk_xml(OAUTH_PLAIN_DISK, "plain-oauth", cas=False, client="gcp_oauth"), + _disk_xml(OAUTH_PLAIN_DISK_2, "plain-oauth-cold", cas=False, client="gcp_oauth"), + _disk_xml(CAS_OAUTH_DISK, "cas-oauth", cas=True, client="gcp_oauth"), + _disk_xml( + CAS_OAUTH_STAGED_DISK, + "cas-oauth-staged", + cas=True, + client="gcp_oauth", + staging_backend="s3", + ), + ] + if AMBIGUITY_DRIVER_AVAILABLE: + disks.append( + _disk_xml( + CAS_OAUTH_AMBIGUITY_DISK, + "cas-oauth-ambiguity", + cas=True, + client="gcp_oauth", + staging_backend="s3", + ) + ) + policies += [ + _policy_xml(CAS_OAUTH_DISK), + _policy_xml(CAS_OAUTH_STAGED_DISK), + _two_volume_policy_xml(OAUTH_TWO_VOLUME_POLICY, OAUTH_PLAIN_DISK, OAUTH_PLAIN_DISK_2), + ] + if AMBIGUITY_DRIVER_AVAILABLE: + policies.append(_policy_xml(CAS_OAUTH_AMBIGUITY_DISK)) + + named_collection_entries = [] + if HMAC_AVAILABLE: + named_collection_entries.append(_named_collection_xml(PARQUET_NAMED_COLLECTION, "gcs_hmac")) + if OAUTH_AVAILABLE: + named_collection_entries.append(_named_collection_xml(OAUTH_PARQUET_NAMED_COLLECTION, "gcp_oauth")) + named_collections = "" + if named_collection_entries: + named_collections = " \n{}\n \n".format("\n".join(named_collection_entries)) + + with open(path, "w", encoding="utf-8") as out: + out.write("\n") + if AMBIGUITY_DRIVER_AVAILABLE: + # Global HTTPS proxy configuration is required because a nested disk `` key would + # be rejected by `ContentAddressedSettings`. The external driver stays transparent until + # armed for the dedicated ambiguity prefix. + out.write(" {}\n".format(_xml(AMBIGUITY_PROXY_URI))) + out.write( + " /etc/clickhouse-server/extra_conf.d/{}" + "truestrict" + "\n".format(_xml(os.path.basename(AMBIGUITY_CA_FILE))) + ) + # Disable SDK-internal retries only on the two dedicated prefixes. The response-loss + # scenario must return control to `PartWriteTxn` after the first ambiguous copy; ordinary + # disks retain their pre-change retry profile. + out.write(_ambiguity_endpoint_settings_xml()) + out.write(" \n \n") + out.write("\n".join(disks)) + out.write("\n \n \n") + out.write("\n".join(policies)) + out.write("\n \n \n") + out.write(named_collections) + out.write("\n") + + +def _configured_cas_disks(): + disks = [] + if HMAC_AVAILABLE: + disks += [CAS_HMAC_DISK, CAS_HMAC_STAGED_DISK] + if AMBIGUITY_DRIVER_AVAILABLE: + disks.append(CAS_HMAC_AMBIGUITY_DISK) + if OAUTH_AVAILABLE: + disks += [CAS_OAUTH_DISK, CAS_OAUTH_STAGED_DISK] + if AMBIGUITY_DRIVER_AVAILABLE: + disks.append(CAS_OAUTH_AMBIGUITY_DISK) + return disks + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + if not (HMAC_AVAILABLE or OAUTH_AVAILABLE): + yield cluster + return + + configs_dir = os.path.join(os.path.dirname(__file__), "configs") + os.makedirs(configs_dir, exist_ok=True) + config_path = os.path.join(configs_dir, "live_gcs_generated.xml") + _write_config(config_path) + + cluster.add_instance( + "node", + main_configs=[config_path], + extra_configs=[AMBIGUITY_CA_FILE] if AMBIGUITY_DRIVER_AVAILABLE else [], + stay_alive=True, + ) + try: + cluster.start() + node = cluster.instances["node"] + # Manual rounds are part of the GC scenarios. Stop each background scheduler first so an + # uncorrelated round cannot consume a transition between the event assertions that bracket it. + for disk in _configured_cas_disks(): + node.query("SYSTEM CAS GC STOP '{}'".format(disk)) + yield cluster + finally: + # Everything this run wrote lives under `PREFIX`, which carries a per-run random suffix, so no + # two runs and nothing pre-existing can collide. The DROPs below let each CAS pool retire its + # own metadata rather than deleting a live pool's keys from underneath it. + # + # They do NOT leave the bucket exactly as found: the Parquet test writes a plain object through + # a table function, and there is no SQL verb that deletes an object. `PREFIX/parquet/` therefore + # survives the run. Delete the whole `PREFIX` afterwards, or give the bucket a lifecycle rule — + # this suite runs against a real, billable bucket and cannot clean those keys itself. + node = cluster.instances.get("node") + if node is not None: + try: + tables = node.query( + "SELECT name FROM system.tables WHERE database = currentDatabase() AND (startsWith(name, 'task10_') OR startsWith(name, 't_live_') OR startsWith(name, 'src_live_')) FORMAT TSV" + ).split() + for table in tables: + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + for disk in (HMAC_PLAIN_DISK, CAS_HMAC_DISK, CAS_OAUTH_DISK): + node.query("DROP TABLE IF EXISTS t_{} SYNC".format(disk)) + node.query("DROP TABLE IF EXISTS src_{} SYNC".format(disk)) + except Exception: # noqa: BLE001 - teardown must not mask a test failure + pass + cluster.shutdown() + + +def _events(node, names): + """Current values of the named `system.events` counters, zero-filled for absent ones.""" + rows = node.query("SELECT event, value FROM system.events WHERE event IN ({}) FORMAT TSV".format(", ".join("'{}'".format(n) for n in names))) + seen = {} + for line in rows.strip().splitlines(): + event, value = line.split("\t") + seen[event] = int(value) + return {name: seen.get(name, 0) for name in names} + + +def _create(node, disk, table=None): + table = table or "t_" + disk + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query( + """ + CREATE TABLE {} (id Int64, data String) + ENGINE = MergeTree() ORDER BY id + SETTINGS storage_policy = '{}' + """.format(table, disk) + ) + return table + + +def _opaque_generation_evidence(node, query): + """Keep raw generations in a hidden frame and return only domain evidence plus one-way digests.""" + __tracebackhide__ = True + node.query("SYSTEM FLUSH LOGS") + raw = node.query(query) + generations = [line.strip().strip('"') for line in raw.strip().splitlines() if line.strip()] + return ( + len(generations), + bool(generations) and all(generation.isdigit() for generation in generations), + tuple(hashlib.sha256(generation.encode("utf-8")).hexdigest() for generation in generations), + ) + + +def _cas_generation_domain(node, disk): + """Whether this disk recorded generations and every recorded value belongs to the numeric domain.""" + __tracebackhide__ = True + count, all_numeric, _digests = _opaque_generation_evidence( + node, + "SELECT DISTINCT token FROM system.cas_log WHERE disk_name = '{}' AND token != '' FORMAT TSV".format(disk), + ) + return count > 0, all_numeric + + +def _cas_event_generation_evidence(node, disk, object_hash, event_type, outcome=""): + """Return count, numeric-domain evidence, and an opaque digest for one CAS event generation.""" + __tracebackhide__ = True + clauses = [ + "disk_name = '{}'".format(disk), + "object_hash = '{}'".format(object_hash), + "event_type = '{}'".format(event_type), + "token != ''", + ] + if outcome: + clauses.append("outcome = '{}'".format(outcome)) + count, all_numeric, digests = _opaque_generation_evidence( + node, + "SELECT token FROM system.cas_log WHERE {} ORDER BY event_time_microseconds FORMAT TSV".format(" AND ".join(clauses)), + ) + return count, all_numeric, digests[0] if count == 1 else "" + + +def _query_id(scenario, auth_mode): + return "task10_{}_{}_{}".format(scenario, auth_mode, RUN_ID) + + +def _cas_events(node, disk, query_id="", event_types=(), object_hash=""): + """Return attributable CAS events without ever reading configuration or authentication data.""" + __tracebackhide__ = True + node.query("SYSTEM FLUSH LOGS") + clauses = ["disk_name = '{}'".format(disk)] + if query_id: + clauses.append("query_id = '{}'".format(query_id)) + if event_types: + clauses.append("event_type IN ({})".format(", ".join("'{}'".format(event_type) for event_type in event_types))) + if object_hash: + clauses.append("object_hash = '{}'".format(object_hash)) + raw = node.query("SELECT event_type, object_hash, outcome, detail FROM system.cas_log WHERE {} ORDER BY event_time_microseconds FORMAT JSONEachRow".format(" AND ".join(clauses))) + events = [json.loads(line) for line in raw.splitlines() if line] + for event in events: + event.pop("token", None) + return events + + +def _ordinary_etag_domain(node, probe): + """Keep raw ETags hidden and report only whether observed values remain ordinary and non-empty.""" + __tracebackhide__ = True + lines = node.grep_in_log("{} |".format(probe)) + values = [line.rsplit("|", 1)[1].strip().strip('"') for line in lines.splitlines() if "|" in line] + return bool(lines), bool(values), bool(values) and all(value and not value.isdigit() for value in values) + + +def _query_profile_events(node, query_id, names): + """Read named ProfileEvents from the successful query-log row for one statement.""" + node.query("SYSTEM FLUSH LOGS") + expressions = ", ".join("toUInt64(ProfileEvents['{0}']) AS {0}".format(name) for name in names) + raw = node.query( + "SELECT {} FROM system.query_log WHERE query_id = '{}' AND type = 'QueryFinish' ORDER BY event_time_microseconds DESC LIMIT 1 FORMAT JSONEachRow".format(expressions, query_id) + ).strip() + assert raw, "query log has no successful row for {}".format(query_id) + return json.loads(raw) + + +def _assert_one_head_per_blob_task(node, query_id): + profile = _query_profile_events( + node, + query_id, + ("CASBlobUploadFanoutTasks", "CASBlobHead", "CASBlobHeadMiss"), + ) + tasks = profile["CASBlobUploadFanoutTasks"] + assert tasks > 0, profile + assert profile["CASBlobHead"] + profile["CASBlobHeadMiss"] == tasks, profile + return profile + + +def _largest_blob_put(events): + puts = [event for event in events if event["event_type"] == "blob_put"] + assert puts, "the statement emitted no attributable `blob_put` event" + return max(puts, key=lambda event: int(event["detail"].get("size", "0"))) + + +def _create_payload_table(node, table, disk): + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query("CREATE TABLE {} (id UInt64, payload String CODEC(NONE)) ENGINE = MergeTree ORDER BY id SETTINGS storage_policy = '{}'".format(table, disk)) + + +def _small_payload_insert(table, auth_mode, scenario): + return "INSERT INTO {} SELECT number, concat('{}-{}-', toString(number), repeat('p', 4096)) FROM numbers(64)".format(table, scenario, auth_mode) + + +def _large_payload_insert(table, salt): + # `hex(cityHash64(...))` is deterministic but not compressible enough for `CODEC(NONE)` to hide + # the size threshold. One row keeps the target blob easy to identify by its logged logical size. + return "INSERT INTO {} SELECT 1, arrayStringConcat(arrayMap(x -> hex(cityHash64(x + {})), range({})))".format(table, salt, LIVE_LARGE_VALUE_ITEMS) + + +def _gc_until(node, disk, object_hash, event_type, outcome="", max_rounds=20): + """Run bounded synchronous GC rounds until one target transition becomes durable in the log.""" + for _ in range(max_rounds): + node.query("SYSTEM CAS GC RUN '{}'".format(disk)) + rows = _cas_events(node, disk, event_types=(event_type,), object_hash=object_hash) + if outcome: + rows = [row for row in rows if row["outcome"] == outcome] + if rows: + return rows[-1] + assert False, "{} did not emit {} outcome={!r} in {} manual rounds".format(object_hash, event_type, outcome, max_rounds) + + +def _ambiguity_driver_request(path, payload): + """Call the opt-in fault driver; callers assert only phase booleans, never auth material.""" + request = urllib.request.Request( + AMBIGUITY_CONTROL_URL.rstrip("/") + path, + data=json.dumps(payload).encode("utf-8"), + headers={"Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=30) as response: + return json.loads(response.read().decode("utf-8")) + + +def _wait_for_driver_phase(path, scenario_id, phase, timeout=60): + deadline = time.monotonic() + timeout + while True: + report = _ambiguity_driver_request(path, {"scenario_id": scenario_id}) + if report.get(phase) is True: + return report + assert time.monotonic() < deadline, "fault driver did not reach phase {}".format(phase) + time.sleep(0.1) + + +# --------------------------------------------------------------------------------------------------- +# Group 1: Default requests on `gcs_hmac` and `gcp_oauth`. Nothing here is content-addressed; the +# question is whether GCS accepts every ordinary ClickHouse object-storage operation while its ETag +# response contract remains independent from CAS generation tokens. +# --------------------------------------------------------------------------------------------------- + +requires_hmac = pytest.mark.skipif( + not HMAC_AVAILABLE, + reason="set GCS_LIVE_HMAC_ACCESS_KEY_ID and GCS_LIVE_HMAC_SECRET_ACCESS_KEY", +) +requires_oauth = pytest.mark.skipif( + not OAUTH_AVAILABLE, + reason="set GCS_LIVE_OAUTH_FROM_METADATA=1 on a GCE host, or the GCS_LIVE_OAUTH_ADC_* triple anywhere else", +) + +ORDINARY_DISK_CASES = ( + pytest.param( + "gcs_hmac", + HMAC_TWO_VOLUME_POLICY, + "/plain", + marks=requires_hmac, + id="gcs_hmac", + ), + pytest.param( + "gcp_oauth", + OAUTH_TWO_VOLUME_POLICY, + "/plain-oauth", + marks=requires_oauth, + id="gcp_oauth", + ), +) + +NAMED_COLLECTION_CASES = ( + pytest.param("gcs_hmac", PARQUET_NAMED_COLLECTION, marks=requires_hmac, id="gcs_hmac"), + pytest.param("gcp_oauth", OAUTH_PARQUET_NAMED_COLLECTION, marks=requires_oauth, id="gcp_oauth"), +) + +CAS_STREAM_CASES = ( + pytest.param( + "gcs_hmac", + CAS_HMAC_DISK, + marks=requires_hmac, + id="gcs_hmac", + ), + pytest.param( + "gcp_oauth", + CAS_OAUTH_DISK, + marks=requires_oauth, + id="gcp_oauth", + ), +) + +CAS_STAGED_CASES = ( + pytest.param("gcs_hmac", CAS_HMAC_STAGED_DISK, marks=requires_hmac, id="gcs_hmac"), + pytest.param("gcp_oauth", CAS_OAUTH_STAGED_DISK, marks=requires_oauth, id="gcp_oauth"), +) + +CAS_AMBIGUITY_CASES = ( + pytest.param("gcs_hmac", CAS_HMAC_AMBIGUITY_DISK, marks=requires_hmac, id="gcs_hmac"), + pytest.param("gcp_oauth", CAS_OAUTH_AMBIGUITY_DISK, marks=requires_oauth, id="gcp_oauth"), +) + +requires_ambiguity_driver = pytest.mark.skipif( + not AMBIGUITY_DRIVER_AVAILABLE, + reason="set credential-free GCS_LIVE_AMBIGUITY_PROXY_URI and GCS_LIVE_AMBIGUITY_CONTROL_URL " + "endpoints plus public GCS_LIVE_AMBIGUITY_CA_FILE; required response-loss and held-delete " + "scenarios cannot be synthesized by ordinary GCS credentials", +) + + +@pytest.mark.parametrize("auth_mode,policy,path_fragment", ORDINARY_DISK_CASES) +def test_default_gcs_client_accepts_the_ordinary_object_storage_operation_set(auth_mode, policy, path_fragment): + """Every S3 operation an ordinary disk issues is accepted under either GCS client. + + The `system.events` deltas are what make this non-vacuous: each named operation must have been + issued at least once, so a statement that quietly stopped reaching object storage — because a + default changed, or because a part stayed in memory — cannot leave the assertion true. + + The statement-to-operation mapping is deliberately NOT pinned. Which statement produces a batch + delete rather than singular ones is a ClickHouse implementation detail that moves between versions; + whether GCS accepts a batch delete is what this gate is asking. Pinning the mapping would make this + test fail on refactors that say nothing about GCS. + + Object LISTING is not covered by THIS test — an ordinary MergeTree lifecycle on a local-metadata + disk never issues one, see the comment on `counters` below. + `test_default_gcs_client_accepts_an_object_listing` covers it on the same authenticated client + through the table-engine path, which is a lister. + + Would fail if: bearer-token or GOOG4 authentication produced a request Google rejects for some + operation, or either configured credential source stopped being accepted by its client selector + (the disk would not resolve and no statement below would run). + """ + node = cluster.instances["node"] + # `S3ListObjects` is deliberately NOT in this set, and must not be re-added. Both of its increment + # sites live in `S3IteratorAsync::getBatchAndCheckNext` and `S3ObjectStorage::listObjects`, which + # are reached through `IObjectStorage::iterate`/`listObjects` — called by the object-storage table + # engines, the data lakes, `ObjectStorageQueue`, the plain/plain_rewritable metadata storages and + # CAS, none of which is in play here. These disks set no `metadata_type`, so they use local + # metadata: MergeTree's own `iterate` calls go through `IDisk::iterateDirectory` over the LOCAL + # metadata directory and issue no S3 listing at all. An ordinary lifecycle on a local-metadata disk + # never lists. + # + # Worse than merely unsatisfiable, it would be unsound: `system.events` is process-wide, and the + # CAS disks in this same configuration DO list, so a background GC round landing inside the delta + # window could satisfy it for a reason that has nothing to do with this test's workload. + counters = [ + "S3PutObject", + "S3GetObject", + "S3HeadObject", + "S3CopyObject", + "S3DeleteObjects", + "S3CreateMultipartUpload", + "S3UploadPart", + "S3CompleteMultipartUpload", + ] + before = _events(node, counters) + + table = _create(node, policy, "task10_plain_{}".format(auth_mode)) + + # A single-part PUT with custom metadata, then the HEAD that `s3_check_objects_after_upload` + # issues to verify it. + node.query( + "INSERT INTO {} SELECT number, toString(number) FROM numbers(500)".format(table), + settings={"s3_check_objects_after_upload": 1}, + ) + # A multipart upload: a tiny single-part ceiling rather than a large body, so the run does not + # depend on how large a default part happens to be. + node.query( + "INSERT INTO {} SELECT number, repeat('x', 4096) FROM numbers(500, 4000)".format(table), + settings={ + "s3_min_upload_part_size": 5 * 1024 * 1024, + "s3_max_single_part_upload_size": 1024, + }, + ) + assert int(node.query("SELECT count() FROM {}".format(table))) == 4500 + assert int(node.query("SELECT sum(id) FROM {}".format(table))) > 0 + + # A server-side copy: moving a partition between the two volumes of one policy copies each object + # and then deletes the source. This is the only statement here that reaches `CopyObject`. + node.query("ALTER TABLE {} MOVE PARTITION tuple() TO VOLUME 'cold'".format(table)) + assert int(node.query("SELECT count() FROM {}".format(table))) == 4500 + + # A merge (more reads and writes), then the deletes. + node.query("OPTIMIZE TABLE {} FINAL".format(table)) + node.query("ALTER TABLE {} DROP PARTITION tuple()".format(table)) + assert int(node.query("SELECT count() FROM {}".format(table))) == 0 + + after = _events(node, counters) + for name in counters: + assert after[name] > before[name], "{} was never issued ({} -> {}), so GCS acceptance of it is unproven".format(name, before[name], after[name]) + + # `S3DeleteObjects` counts the singular and batch forms together, so the counter alone cannot say + # the batch form was accepted. The two paths log differently, which separates them: + # `deleteFileFromS3` logs "Object with path was removed from S3" and `deleteFilesFromS3` logs + # "Objects with paths [,...] were removed from S3". + # + # The load-bearing one is the third line. When GCS refuses a batch `DeleteObjects`, + # `deleteFilesFromS3` logs "DeleteObjects is not supported", calls + # `s3_capabilities.setIsBatchDeleteSupported(false)` and silently retries with plain + # `DeleteObject` — so the batch form failing looks EXACTLY like success at both the counter and the + # data level. Asserting that line is absent while the plural line is present is the only way to say + # GCS accepted the batch shape rather than the fallback having covered for it. + # + # LOG LEVEL, because this assertion's ability to FAIL depends on it. The two lines sit at different + # levels: the plural "Objects with paths [...]" is `LOG_DEBUG`, the fallback notice is `LOG_TRACE`. + # A server that did not admit TRACE would make the absence check pass unconditionally — a test that + # cannot fail. It is admitted here because `add_instance` copies + # `helpers/0_common_instance_config.xml` unconditionally and that sets `test`, and + # `Poco::Message` orders `PRIO_TEST` BELOW `PRIO_TRACE`, so `test` admits trace messages. (The + # `with_installed_binary` path rewrites it to `trace`, which also admits them.) If this suite ever + # sets `copy_common_configs=False` or overrides the logger level, re-check this before trusting the + # absence half. + # Filtered to this test's own key prefix: the log carries every disk's traffic, and the CAS disks + # in this configuration delete objects too, so an unfiltered match would be the same + # someone-else's-traffic confound the module docstring warns about for counters. + batch_lines = [line for line in node.grep_in_log(BATCH_DELETE_LOG_PATTERN).splitlines() if PREFIX in line and path_fragment in line] + assert batch_lines, "no batch delete was logged for this run's own keys, so GCS acceptance of the batch DeleteObjects shape is unproven" + assert not node.grep_in_log("DeleteObjects is not supported"), ( + "GCS refused the batch DeleteObjects shape and ClickHouse fell back to singular deletes; the counter and the row counts cannot see this, which is why it is asserted here" + ) + + # The singular shape needs its own evidence. `S3DeleteObjects` aggregates both, so the counter + # moving says nothing about which of the two GCS accepted, and the assertions above speak only for + # the batch one -- a build that never issued a singular DeleteObject at all would satisfy them. + # Same prefix filter and the same reason for it. + single_lines = [line for line in node.grep_in_log("Object with path ").splitlines() if PREFIX in line and path_fragment in line] + assert single_lines, "no singular delete was logged for this run's own keys, so GCS acceptance of the singular DeleteObject shape is unproven -- only the batch shape is" + + +@requires_hmac +def test_default_gcs_hmac_reports_a_typed_error_for_a_refused_request(): + """A refused request must arrive as a typed S3 error, not an unparsed body. + + GCS answers the XML API with an `` document, and the whole point of keeping the + request on the S3 XML path is that the SDK parses it. Would fail if: the GOOG4 path returned a + response the error parser cannot read, which would surface as a generic transport failure with the + real cause only in the body. + + A disk rather than a positional `s3('url', 'key', 'secret')`: that argument form has no spelling + for `http_client` and would sign with ordinary AWS SigV4, passing or failing for a reason unrelated + to GOOG4. A NAMED COLLECTION would work — see the Parquet test below — but the disk is what the + rest of this group already exercises. + """ + node = cluster.instances["node"] + table = _create(node, HMAC_ABSENT_BUCKET_DISK) + error = node.query_and_get_error("INSERT INTO {} SELECT number, toString(number) FROM numbers(10)".format(table)) + # A parsed S3 error names the bucket problem. An unparsed one surfaces as a bare transport or + # timeout failure, which is what must not appear. + assert ("NoSuchBucket" in error) or ("S3_ERROR" in error) or ("ACCESS_DENIED" in error), error + + +@pytest.mark.parametrize("auth_mode,named_collection", NAMED_COLLECTION_CASES) +def test_default_gcs_client_accepts_an_object_listing(auth_mode, named_collection): + """GCS accepts a LIST under either authentication mode through a named collection. + + The disks above cannot produce one: they use local metadata, so MergeTree's directory iteration + reads the local metadata directory and `IObjectStorage::iterate` is never called. The table-engine + path IS a lister — `StorageObjectStorageSource` calls `object_storage->iterate` to expand a glob — + and the named collection puts that on the same `gcs_hmac` client, so the listing is signed the same + way as everything else in this group. + + The reachability proof is the DATA, not a counter, and that is deliberate: `S3ListObjects` is + exactly the counter the module's OPEN QUESTION section warns about, since the CAS disks in this + configuration list too. Reading rows that came from two separate objects through one glob cannot be + satisfied by anyone else's traffic — the listing must have enumerated both to return their union. + + Would fail if: GCS rejected a GOOG4-signed `ListObjectsV2`, or returned a body the SDK cannot parse + into keys — the glob would resolve to fewer objects and the union would be short. + """ + node = cluster.instances["node"] + probe = "listing-probe-{}".format(auth_mode) + for part in (1, 2): + node.query( + "INSERT INTO FUNCTION s3({}, filename='{}-{}.parquet', format='Parquet') SELECT {} AS part, number AS id FROM numbers(10)".format(named_collection, probe, part, part), + settings={"s3_truncate_on_insert": 1}, + ) + + glob = "s3({}, filename='{}-*.parquet', format='Parquet')".format(named_collection, probe) + assert int(node.query("SELECT count() FROM {}".format(glob))) == 20 + # Both objects, through one glob: the listing enumerated them rather than a single key being read. + assert node.query("SELECT DISTINCT part FROM {} ORDER BY part FORMAT TSV".format(glob)).split() == ["1", "2"] + + +@pytest.mark.parametrize("auth_mode,named_collection", NAMED_COLLECTION_CASES) +def test_default_gcs_client_parquet_metadata_cache_keys_on_the_ordinary_etag(auth_mode, named_collection): + """The Parquet metadata cache keys off the object's ordinary ETag, never a generation. + + Three cache consumers — the filesystem cache, the page cache and this one — key off ONE value, the + `etag` on the object metadata; only their formulas differ, and each formula is pinned by a unit + test. So this is the end-to-end arm for the shared VALUE, and what it has to establish on a live + endpoint is that the value arriving here is an ordinary ETag and not a numeric generation. A + generation reaching a cache key is the concrete bug the request-mode isolation exists to prevent: + the same object would acquire different keys depending on whether its metadata came from LIST or + from HEAD. + + It needs the object-storage TABLE ENGINE, not a disk — `ParquetV3BlockInputFormat` builds the key + and only `StorageObjectStorageSource` reaches it, so no MergeTree statement can. Selecting + `gcs_hmac` there requires a NAMED COLLECTION: `StorageS3Configuration::fromNamedCollection` reads + `http_client`, while the positional argument form does not. + + Would fail if: a generation reached the ETag field on this path — the digit check breaks; or the + key stopped being stable across two reads of one unchanged object — the hit count stays zero. + + NOT subject to the counter hazard in the module docstring, and that is deliberate rather than + lucky. Its reachability preconditions are `ParquetMetadataCacheMisses` and + `ParquetMetadataCacheHits`, which only a Parquet read moves. The CAS disks in this configuration + cannot touch either, so unlike the S3 counters in the group above these two mean what they say. + """ + node = cluster.instances["node"] + events = ["ParquetMetadataCacheMisses", "ParquetMetadataCacheHits"] + probe = "cache-key-probe-{}.parquet".format(auth_mode) + table_function = "s3({}, filename='{}', format='Parquet')".format(named_collection, probe) + + node.query( + "INSERT INTO FUNCTION {} SELECT number AS id, toString(number) AS data FROM numbers(1000)".format(table_function), + settings={"s3_truncate_on_insert": 1}, + ) + + before = _events(node, events) + # First read: cold, so the metadata is fetched from the object and the key is minted. + assert int(node.query("SELECT count() FROM {}".format(table_function))) == 1000 + after_cold = _events(node, events) + assert after_cold["ParquetMetadataCacheMisses"] > before["ParquetMetadataCacheMisses"], "no Parquet metadata cache miss, so the read never reached the object and nothing below is meaningful" + + # Second read of the same unchanged object: the key must be rebuilt identically and hit. + assert int(node.query("SELECT count() FROM {}".format(table_function))) == 1000 + after_warm = _events(node, events) + assert after_warm["ParquetMetadataCacheHits"] > after_cold["ParquetMetadataCacheHits"], "the second read of an unchanged object missed the cache, so the key is not stable" + + # The key's own ETag component comes from `cache miss | `. Raw provider values stay + # inside a traceback-hidden helper so a domain failure cannot disclose one through `--showlocals`. + cache_key_logged, etag_observed, ordinary_etag_domain = _ordinary_etag_domain(node, probe) + assert cache_key_logged, "the cache logged no key for this object, so the ETag domain cannot be checked" + assert etag_observed, "the cache key carried no observable ETag" + assert ordinary_etag_domain, "the cache key ETag was empty or entered the numeric generation domain" + + +# --------------------------------------------------------------------------------------------------- +# Groups 2 and 3: NativeConditional requests, on `gcp_oauth` and on `gcs_hmac`. Reaching a readable +# table is the strongest single assertion available: a CAS mount runs `runCapabilityProbe`, which +# requires conditional create, conditional overwrite, a REFUSED delete on a wrong token and an +# accepted delete on the right one — all against live GCS, all before the mount is allowed to +# complete. A mounted disk means Google accepted every one of them. +# --------------------------------------------------------------------------------------------------- + + +def _run_cas_group(disk): + node = cluster.instances["node"] + table = _create(node, disk) + + node.query("INSERT INTO {} SELECT number, toString(number) FROM numbers(300)".format(table)) + node.query("INSERT INTO {} SELECT number, toString(number) FROM numbers(300, 300)".format(table)) + assert int(node.query("SELECT count() FROM {}".format(table))) == 600 + assert int(node.query("SELECT uniqExact(data) FROM {}".format(table))) == 600 + + # A merge rewrites part metadata through the same conditional-write path, and dropping a partition + # drives the exact, token-carrying DELETE. + node.query("OPTIMIZE TABLE {} FINAL".format(table)) + node.query("ALTER TABLE {} DROP PARTITION tuple()".format(table)) + assert int(node.query("SELECT count() FROM {}".format(table))) == 0 + + generations_seen, numeric_generation_domain = _cas_generation_domain(node, disk) + assert generations_seen, "no incarnation generation was recorded, so the domain assertion is vacuous" + assert numeric_generation_domain, "a CAS incarnation left the numeric GCS generation domain" + + +@requires_oauth +def test_native_conditional_gcp_oauth_mounts_and_keeps_generation_tokens(): + """Group 2. Conditional PUT, native-token HEAD and exact DELETE under bearer-token auth. + + Would fail if: GCS refused a conditional create carrying `x-goog-if-generation-match`, refused an + exact DELETE, or answered a token-producing write without a generation — the token recorded would + then be an ETag and the digit assertion would break. It would also fail if the OAuth cleanup left + a stale AWS signing artifact on the request that Google rejects, which is one of the two things + only a live endpoint can settle: against `storage.googleapis.com` the `ApiMode::GCS` block in + `Client::BuildHttpRequest` becomes active, and `test_cas_gcs` cannot reach it. + + Two shapes the plan asks of this group are NOT here, because no configuration this suite can hold + produces them. Both enumerations are written out rather than asserted, since "nothing can drive + this" is a claim about a set: + + A CHECKSUM-BEARING or CHUNKED/FRAMED PUT. The only producer of `x-amz-checksum-*` is + `RequestChecksumRequired`, which returns `is_s3express_bucket`, and `setChecksumAlgorithm` has + exactly one caller, `setIsS3ExpressBucket`. `is_s3express_bucket` has one source, + `S3::isS3ExpressEndpoint(url.endpoint)`, which is `endpoint.contains("s3express")`. This gate + REQUIRES the endpoint to be `storage.googleapis.com` — that substring is what makes + `deduceProviderType` report GCS, which is the property the gate exists to exercise. `disable_checksum` + only suppresses `Content-MD5`; it never turns checksum headers on. So against a real GCS endpoint + the aws-chunked framing headers cannot appear, and the allowlist's `Consume` rule for + `x-amz-checksum-` and `Reject` rules for `x-amz-trailer` / `x-amz-decoded-content-length` are + reachable only from the dialect unit tests. They guard a future SDK change, not a current config. + + An ATTRIBUTE ROUND TRIP. No production path fills object attributes on any object storage: every + `writeObject` caller outside the object-storage layer passes `/* attributes= */ {}`, no + `ObjectAttributes{...}` is constructed outside tests, and every CAS `putIfAbsent` / + `nativeConditionalPut` site forwards a `meta` parameter without ever building a non-empty one. So + there is no SQL statement that writes custom metadata, and nothing to read back. + """ + _run_cas_group(CAS_OAUTH_DISK) + + +@requires_hmac +def test_native_conditional_gcs_hmac_mounts_and_keeps_generation_tokens(): + """Group 3. The same three operations under GOOG4 signing. + + Would fail if: the GOOG4 signed-header allowlist produced a signature Google rejects for a + conditional request — the conditional headers are exactly the ones an allowlist bug would drop or + fail to cover, and no unit test can tell a signature Google accepts from one it does not. + """ + _run_cas_group(CAS_HMAC_DISK) + + +# --------------------------------------------------------------------------------------------------- +# Group 4: the unconditional blob-publication protocol against real GCS. Each case is run once on a +# bearer-token client and once on a GOOG4 client when that credential source is available. Assertions +# use the statement's own query id, so background work or a sibling authentication mode cannot make a +# scenario green. +# --------------------------------------------------------------------------------------------------- + + +@pytest.mark.parametrize("auth_mode,stream_disk", CAS_STREAM_CASES) +def test_live_cas_fresh_streaming_then_duplicate_adoption(auth_mode, stream_disk): + """A fresh body streams after a miss; byte-identical reuse performs no second publication.""" + node = cluster.instances["node"] + table = "task10_fresh_{}".format(auth_mode) + _create_payload_table(node, table, stream_disk) + insert = _small_payload_insert(table, auth_mode, "fresh-duplicate") + + fresh_query_id = _query_id("fresh", auth_mode) + node.query(insert, query_id=fresh_query_id) + _assert_one_head_per_blob_task(node, fresh_query_id) + fresh_events = _cas_events( + node, + stream_disk, + query_id=fresh_query_id, + event_types=("blob_put", "blob_reuse_adopt"), + ) + target = _largest_blob_put(fresh_events) + target_hash = target["object_hash"] + assert target["detail"].get("publication_reason") == "absent" + assert target["detail"].get("transport") == "streaming" + + duplicate_query_id = _query_id("duplicate", auth_mode) + node.query(insert, query_id=duplicate_query_id) + duplicate_profile = _assert_one_head_per_blob_task(node, duplicate_query_id) + duplicate_events = _cas_events( + node, + stream_disk, + query_id=duplicate_query_id, + event_types=("blob_put", "blob_reuse_adopt"), + object_hash=target_hash, + ) + assert [event for event in duplicate_events if event["event_type"] == "blob_reuse_adopt"] + assert not [event for event in duplicate_events if event["event_type"] == "blob_put"] + assert duplicate_profile["CASBlobHead"] > 0, duplicate_profile + assert int(node.query("SELECT count() FROM {}".format(table))) == 128 + node.query("DROP TABLE {} SYNC".format(table)) + + +@pytest.mark.parametrize("auth_mode,stream_disk", CAS_STREAM_CASES) +def test_live_cas_concurrent_equivalent_publishers(auth_mode, stream_disk): + """Two equivalent writers may race unconditionally, but both publish one readable value.""" + node = cluster.instances["node"] + tables = [ + "task10_concurrent_{}_a".format(auth_mode), + "task10_concurrent_{}_b".format(auth_mode), + ] + for table in tables: + _create_payload_table(node, table, stream_disk) + + query_ids = [ + _query_id("concurrent_a", auth_mode), + _query_id("concurrent_b", auth_mode), + ] + barrier = threading.Barrier(2) + + def publish(table, query_id): + barrier.wait() + node.query(_small_payload_insert(table, auth_mode, "concurrent"), query_id=query_id) + + with ThreadPoolExecutor(max_workers=2) as pool: + futures = [pool.submit(publish, table, query_id) for table, query_id in zip(tables, query_ids)] + for future in futures: + future.result() + + event_sets = [] + for query_id in query_ids: + _assert_one_head_per_blob_task(node, query_id) + event_sets.append( + _cas_events( + node, + stream_disk, + query_id=query_id, + event_types=("blob_put", "blob_reuse_adopt"), + ) + ) + common = set(event["object_hash"] for event in event_sets[0]) & set(event["object_hash"] for event in event_sets[1]) + assert common, "the equivalent writers touched no common content hash" + target_hash = max( + common, + key=lambda object_hash: max(int(event["detail"].get("size", "0")) for events in event_sets for event in events if event["object_hash"] == object_hash), + ) + target_events = [event for events in event_sets for event in events if event["object_hash"] == target_hash] + publications = [event for event in target_events if event["event_type"] == "blob_put"] + assert publications, "neither concurrent writer published the shared target" + for publication in publications: + assert publication["detail"].get("publication_reason") == "absent" + assert publication["detail"].get("transport") == "streaming" + for table in tables: + assert int(node.query("SELECT count() FROM {}".format(table))) == 64 + assert int(node.query("SELECT uniqExact(payload) FROM {}".format(table))) == 64 + node.query("DROP TABLE {} SYNC".format(table)) + + +@pytest.mark.parametrize("auth_mode,stream_disk", CAS_STREAM_CASES) +def test_live_cas_streaming_blob_above_the_former_conditional_cap(auth_mode, stream_disk): + """A Default single-part blob PUT succeeds above the genuine-conditional GCS ceiling.""" + node = cluster.instances["node"] + table = "task10_former_cap_{}".format(auth_mode) + _create_payload_table(node, table, stream_disk) + query_id = _query_id("former_cap", auth_mode) + node.query( + _large_payload_insert(table, 1100000 if auth_mode == "gcs_hmac" else 1200000), + query_id=query_id, + settings={"s3_max_single_part_upload_size": 64 * 1024 * 1024}, + ) + _assert_one_head_per_blob_task(node, query_id) + target = _largest_blob_put(_cas_events(node, stream_disk, query_id=query_id, event_types=("blob_put",))) + assert int(target["detail"].get("size", "0")) > FORMER_CONDITIONAL_PUT_CAP + assert target["detail"].get("publication_reason") == "absent" + assert target["detail"].get("transport") == "streaming" + multipart = _query_profile_events(node, query_id, ("S3CreateMultipartUpload",)) + assert multipart["S3CreateMultipartUpload"] == 0, multipart + assert int(node.query("SELECT length(payload) FROM {}".format(table))) > FORMER_CONDITIONAL_PUT_CAP + node.query("DROP TABLE {} SYNC".format(table)) + + +@pytest.mark.parametrize("auth_mode,stream_disk", CAS_STREAM_CASES) +def test_live_cas_default_blob_publication_uses_multipart(auth_mode, stream_disk): + """A large Default blob body reaches Google's multipart create/part/complete protocol.""" + node = cluster.instances["node"] + table = "task10_multipart_{}".format(auth_mode) + _create_payload_table(node, table, stream_disk) + query_id = _query_id("multipart", auth_mode) + node.query( + _large_payload_insert(table, 2100000 if auth_mode == "gcs_hmac" else 2200000), + query_id=query_id, + settings={ + "s3_max_single_part_upload_size": 0, + "s3_min_upload_part_size": FORMER_CONDITIONAL_PUT_CAP, + }, + ) + _assert_one_head_per_blob_task(node, query_id) + target = _largest_blob_put(_cas_events(node, stream_disk, query_id=query_id, event_types=("blob_put",))) + assert int(target["detail"].get("size", "0")) > FORMER_CONDITIONAL_PUT_CAP + assert target["detail"].get("transport") == "streaming" + multipart = _query_profile_events( + node, + query_id, + ("S3CreateMultipartUpload", "S3UploadPart", "S3CompleteMultipartUpload"), + ) + for event in multipart.values(): + assert event > 0, multipart + assert int(node.query("SELECT count() FROM {}".format(table))) == 1 + node.query("DROP TABLE {} SYNC".format(table)) + + +@pytest.mark.parametrize("auth_mode,staged_disk", CAS_STAGED_CASES) +def test_live_cas_native_staged_copy_is_first_absent_publication(auth_mode, staged_disk): + """An S3-staged source uses Google's native copy exactly on first-plus-absent.""" + node = cluster.instances["node"] + table = "task10_staged_{}".format(auth_mode) + _create_payload_table(node, table, staged_disk) + query_id = _query_id("staged", auth_mode) + node.query(_small_payload_insert(table, auth_mode, "native-staged"), query_id=query_id) + _assert_one_head_per_blob_task(node, query_id) + put_events = _cas_events(node, staged_disk, query_id=query_id, event_types=("blob_put",)) + target = _largest_blob_put(put_events) + assert target["detail"].get("publication_reason") == "absent" + assert target["detail"].get("transport") == "server_side_copy" + copy_profile = _query_profile_events(node, query_id, ("S3CopyObject",)) + copy_publications = [event for event in put_events if event["detail"].get("transport") == "server_side_copy"] + assert copy_publications, "the staged statement emitted no successful native publication" + assert copy_profile["S3CopyObject"] == len(copy_publications), ( + copy_profile, + len(copy_publications), + ) + assert int(node.query("SELECT count() FROM {}".format(table))) == 64 + node.query("DROP TABLE {} SYNC".format(table)) + + +def test_batch_delete_log_pattern_matches_literal_prefix_via_grep_in_log(tmp_path): + """The ordinary batch-delete matcher must survive `grep_in_log`'s regex-mode `zgrep`.""" + representative = "Objects with paths [/bucket/prefix/plain/a] were removed from S3" + absent = "Object with path /bucket/prefix/plain/a was removed from S3" + (tmp_path / "batch.log").write_text(representative + "\n", encoding="utf-8") + (tmp_path / "absent.log").write_text(absent + "\n", encoding="utf-8") + + instance = object.__new__(ClickHouseInstance) + instance.logs_dir = str(tmp_path) + + matched = instance.grep_in_log( + BATCH_DELETE_LOG_PATTERN, + from_host=True, + filename="batch.log", + only_latest=True, + ) + missing = instance.grep_in_log( + BATCH_DELETE_LOG_PATTERN, + from_host=True, + filename="absent.log", + only_latest=True, + ) + assert matched.strip() == representative + assert missing == "" + + +def test_generation_evidence_does_not_cross_the_test_frame_boundary(): + """CAS generations stay inside traceback-hidden helpers; callers receive redacted evidence.""" + + class GenerationEvidenceProbeNode: + def query(self, query): + if "FORMAT JSONEachRow" in query: + return '{"event_type":"blob_retire","object_hash":"probe-hash","token":"0","outcome":"pending","detail":{}}\n' + return '"0"\n' + + node = GenerationEvidenceProbeNode() + events = _cas_events( + node, + "probe-disk", + event_types=("blob_retire",), + object_hash="probe-hash", + ) + assert events == [ + { + "event_type": "blob_retire", + "object_hash": "probe-hash", + "outcome": "pending", + "detail": {}, + } + ] + + seen, all_numeric = _cas_generation_domain(node, "probe-disk") + assert seen is True + assert all_numeric is True + + count, numeric, generation_digest = _cas_event_generation_evidence( + node, + "probe-disk", + "probe-hash", + "blob_retire", + outcome="pending", + ) + assert count == 1 + assert numeric is True + assert len(generation_digest) == 64 + assert all(character in string.hexdigits for character in generation_digest) + assert not any(value == "0" or '"token":"0"' in repr(value) for value in locals().values()), "a raw generation reached the focused test frame" + + +@pytest.mark.parametrize("auth_mode,staged_disk", CAS_STAGED_CASES) +def test_live_cas_condemned_staged_source_retags_by_streaming(auth_mode, staged_disk): + """A staged payload observed as `Condemned` gets a new streaming envelope.""" + node = cluster.instances["node"] + first_table = "task10_condemned_{}_first".format(auth_mode) + second_table = "task10_condemned_{}_second".format(auth_mode) + insert_scenario = "condemned-retag" + + _create_payload_table(node, first_table, staged_disk) + first_query_id = _query_id("condemned_seed", auth_mode) + node.query( + _small_payload_insert(first_table, auth_mode, insert_scenario), + query_id=first_query_id, + ) + target = _largest_blob_put(_cas_events(node, staged_disk, query_id=first_query_id, event_types=("blob_put",))) + target_hash = target["object_hash"] + assert target["detail"].get("transport") == "server_side_copy" + node.query("DROP TABLE {} SYNC".format(first_table)) + + _gc_until(node, staged_disk, target_hash, "blob_retire") + retired_count, retired_numeric, retired_generation_digest = _cas_event_generation_evidence( + node, + staged_disk, + target_hash, + "blob_retire", + ) + assert retired_count == 1, "the target did not record exactly one retirement generation" + assert retired_numeric, "the retired incarnation left the numeric GCS generation domain" + assert retired_generation_digest, "the retired incarnation produced no opaque generation evidence" + + _create_payload_table(node, second_table, staged_disk) + retag_query_id = _query_id("condemned_retag", auth_mode) + node.query( + _small_payload_insert(second_table, auth_mode, insert_scenario), + query_id=retag_query_id, + ) + target_events = _cas_events( + node, + staged_disk, + query_id=retag_query_id, + event_types=("blob_put", "blob_reuse_adopt"), + object_hash=target_hash, + ) + assert len(target_events) == 1, "the condemned target did not have one publication decision" + retag = target_events[0] + assert retag["event_type"] == "blob_put" + assert retag["detail"].get("publication_reason") == "condemned" + assert retag["detail"].get("transport") == "streaming" + assert int(node.query("SELECT count() FROM {}".format(second_table))) == 64 + node.query("DROP TABLE {} SYNC".format(second_table)) + + +@requires_ambiguity_driver +@pytest.mark.parametrize("auth_mode,ambiguity_disk", CAS_AMBIGUITY_CASES) +def test_live_cas_queued_old_token_delete_misses_retagged_replacement(auth_mode, ambiguity_disk): + """Hold a queued old exact DELETE across retagging, then require a provider mismatch. + + `POST /v1/queued-old-delete/arm` accepts the scenario id, non-secret prefix, and target content + hash. After the next GC cut it holds the already-authenticated exact DELETE without changing it. + `POST /v1/queued-old-delete/status` reports only `old_delete_held`; release forwards that same + request after the writer has replaced the generation. The final result must report a provider + precondition mismatch and a surviving replacement. The driver never returns or records the + signed request, authorization material, or generation value. + """ + node = cluster.instances["node"] + first_table = "task10_queued_delete_{}_first".format(auth_mode) + replacement_table = "task10_queued_delete_{}_replacement".format(auth_mode) + insert_scenario = "queued-old-delete" + + _create_payload_table(node, first_table, ambiguity_disk) + first_query_id = _query_id("queued_delete_seed", auth_mode) + node.query( + _small_payload_insert(first_table, auth_mode, insert_scenario), + query_id=first_query_id, + ) + target = _largest_blob_put( + _cas_events( + node, + ambiguity_disk, + query_id=first_query_id, + event_types=("blob_put",), + ) + ) + target_hash = target["object_hash"] + assert target["detail"].get("transport") == "server_side_copy" + node.query("DROP TABLE {} SYNC".format(first_table)) + + _gc_until(node, ambiguity_disk, target_hash, "blob_retire") + retired_count, retired_numeric, retired_generation_digest = _cas_event_generation_evidence( + node, + ambiguity_disk, + target_hash, + "blob_retire", + ) + assert retired_count == 1, "the queued target did not record exactly one retirement generation" + assert retired_numeric, "the queued incarnation left the numeric GCS generation domain" + assert retired_generation_digest, "the queued incarnation produced no opaque generation evidence" + _gc_until( + node, + ambiguity_disk, + target_hash, + "gc_recheck_verdict", + outcome="pending", + ) + _create_payload_table(node, replacement_table, ambiguity_disk) + + scenario_id = _query_id("queued_delete_driver", auth_mode) + armed = _ambiguity_driver_request( + "/v1/queued-old-delete/arm", + { + "scenario_id": scenario_id, + "object_prefix": "{}/{}/".format(PREFIX, AMBIGUITY_SUBPREFIX[auth_mode]), + "target_object_hash": target_hash, + }, + ) + assert armed.get("armed") is True, "the queued-delete driver did not arm" + + replacement_query_id = _query_id("queued_delete_retag", auth_mode) + with ThreadPoolExecutor(max_workers=1) as pool: + gc_future = pool.submit(node.query, "SYSTEM CAS GC RUN '{}'".format(ambiguity_disk)) + try: + _wait_for_driver_phase("/v1/queued-old-delete/status", scenario_id, "old_delete_held") + node.query( + _small_payload_insert(replacement_table, auth_mode, insert_scenario), + query_id=replacement_query_id, + ) + finally: + _ambiguity_driver_request("/v1/queued-old-delete/release", {"scenario_id": scenario_id}) + gc_future.result() + + retag_events = _cas_events( + node, + ambiguity_disk, + query_id=replacement_query_id, + event_types=("blob_put", "blob_reuse_adopt"), + object_hash=target_hash, + ) + assert len(retag_events) == 1, "the held-delete target had no single writer decision" + assert retag_events[0]["event_type"] == "blob_put" + assert retag_events[0]["detail"].get("publication_reason") == "condemned" + assert retag_events[0]["detail"].get("transport") == "streaming" + + delete_events = _cas_events( + node, + ambiguity_disk, + event_types=("blob_delete",), + object_hash=target_hash, + ) + replaced = [event for event in delete_events if event["outcome"] == "replaced"] + assert len(replaced) == 1, "the released old exact delete did not miss the replacement" + replaced_count, replaced_numeric, replaced_generation_digest = _cas_event_generation_evidence( + node, + ambiguity_disk, + target_hash, + "blob_delete", + outcome="replaced", + ) + assert replaced_count == 1, "the replaced delete did not record exactly one generation" + assert replaced_numeric, "the replaced delete left the numeric GCS generation domain" + assert replaced_generation_digest == retired_generation_digest, "the released delete did not carry the generation captured at retirement" + + result = _ambiguity_driver_request("/v1/queued-old-delete/result", {"scenario_id": scenario_id}) + for phase in ( + "old_delete_forwarded", + "provider_precondition_mismatch", + "replacement_present_after_old_delete", + ): + assert result.get(phase) is True, "queued-delete phase {} is unproven".format(phase) + assert int(node.query("SELECT count() FROM {}".format(replacement_table))) == 64 + node.query("DROP TABLE {} SYNC".format(replacement_table)) + + +@requires_ambiguity_driver +@pytest.mark.parametrize("auth_mode,ambiguity_disk", CAS_AMBIGUITY_CASES) +def test_live_cas_ambiguous_staged_copy_absent_retry_uses_retagged_replacement(auth_mode, ambiguity_disk): + """A landed copy loses its response, is exact-deleted, then retries absent by streaming. + + The driver contract is intentionally narrow. `POST /v1/staged-copy-ambiguity/arm` accepts a + scenario id, authentication-mode label, and non-secret object prefix. It transparently forwards + all other traffic. For the first native copy below that prefix it must: let Google accept the + copy; suppress the response; exact-delete the landed generation before the retry `HEAD`; let the + retry and retagged PUT complete; retry the OLD exact delete after replacement; and `POST + /v1/staged-copy-ambiguity/result` returns only phase booleans plus the target content hash. The + proxy owns whatever provider credentials its exact-delete control plane needs; this test never + sends, reads, logs, or records them. + """ + node = cluster.instances["node"] + table = "task10_ambiguity_{}".format(auth_mode) + _create_payload_table(node, table, ambiguity_disk) + scenario_id = _query_id("ambiguity_driver", auth_mode) + armed = _ambiguity_driver_request( + "/v1/staged-copy-ambiguity/arm", + { + "scenario_id": scenario_id, + "auth_mode": auth_mode, + "object_prefix": "{}/{}/".format(PREFIX, AMBIGUITY_SUBPREFIX[auth_mode]), + }, + ) + assert armed.get("armed") is True, "the ambiguity driver did not arm" + + query_id = _query_id("ambiguity", auth_mode) + node.query( + _small_payload_insert(table, auth_mode, "ambiguity-absent-retag"), + query_id=query_id, + ) + report = _ambiguity_driver_request("/v1/staged-copy-ambiguity/result", {"scenario_id": scenario_id}) + for phase in ( + "first_native_copy_landed", + "first_response_lost", + "old_incarnation_exact_deleted", + "retry_head_observed_absent", + "retagged_replacement_landed", + "queued_old_delete_missed_replacement", + "replacement_present_after_old_delete", + ): + assert report.get(phase) is True, "ambiguity driver phase {} is unproven".format(phase) + target_hash = report.get("target_object_hash", "") + assert target_hash, "the ambiguity driver did not identify its target hash" + + events = _cas_events( + node, + ambiguity_disk, + query_id=query_id, + event_types=("blob_put",), + object_hash=target_hash, + ) + assert len(events) == 1, "the retried target did not emit one successful publication" + assert events[0]["detail"].get("publication_reason") == "absent" + assert events[0]["detail"].get("transport") == "streaming" + profile = _query_profile_events(node, query_id, ("S3CopyObject", "S3PutObject")) + assert profile["S3CopyObject"] > 0, profile + assert profile["S3PutObject"] > 0, profile + assert int(node.query("SELECT count() FROM {}".format(table))) == 64 diff --git a/tests/integration/test_replicated_database/test.py b/tests/integration/test_replicated_database/test.py index e3d4e5230101..ad37d3bf5378 100644 --- a/tests/integration/test_replicated_database/test.py +++ b/tests/integration/test_replicated_database/test.py @@ -1398,7 +1398,19 @@ def test_replicated_table_structure_alter(started_cluster): ) competing_node.query("SYSTEM SYNC DATABASE REPLICA table_structure") +<<<<<<< HEAD competing_node.query("DETACH DATABASE table_structure SYNC") +======= + + # `system.tables` only lists an attached database, so the metadata path of `mem` must be read + # before the DETACH below; afterwards the SELECT returns nothing. + metadata_path = competing_node.query( + "SELECT metadata_path FROM system.tables WHERE database='table_structure' AND name='mem'" + ).strip() + assert metadata_path, "metadata_path of table_structure.mem is empty" + + competing_node.query("DETACH DATABASE table_structure") +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) main_node.query( "ALTER TABLE table_structure.rmt ADD COLUMN m int", settings=settings @@ -1408,9 +1420,6 @@ def test_replicated_table_structure_alter(started_cluster): ) main_node.query("INSERT INTO table_structure.rmt VALUES (1, 2, 3)") - metadata_path = competing_node.query( - "SELECT metadata_path FROM system.tables WHERE database='table_structure' AND name='mem'" - ).strip() db_disk_name = get_database_disk_name(competing_node) competing_node.exec_in_container( [ diff --git a/tests/integration/test_storage_gcp_auth/configs/filesystem_caches.xml b/tests/integration/test_storage_gcp_auth/configs/filesystem_caches.xml new file mode 100644 index 000000000000..0f903b8c8258 --- /dev/null +++ b/tests/integration/test_storage_gcp_auth/configs/filesystem_caches.xml @@ -0,0 +1,8 @@ + + + + 1Gi + /tmp/gcp_oauth_cache1 + + + diff --git a/tests/integration/test_storage_gcp_auth/configs/named_collections.xml b/tests/integration/test_storage_gcp_auth/configs/named_collections.xml index 319802fa1cbb..543911bef336 100644 --- a/tests/integration/test_storage_gcp_auth/configs/named_collections.xml +++ b/tests/integration/test_storage_gcp_auth/configs/named_collections.xml @@ -12,5 +12,16 @@ non-existing-account resolver + + + http://resolver:22234/test/ + gcp_oauth + my-account + resolver + 1048576 + 1 + diff --git a/tests/integration/test_storage_gcp_auth/configs/page_cache.xml b/tests/integration/test_storage_gcp_auth/configs/page_cache.xml new file mode 100644 index 000000000000..93a567c5cc74 --- /dev/null +++ b/tests/integration/test_storage_gcp_auth/configs/page_cache.xml @@ -0,0 +1,4 @@ + + 1000000000 + 1000000000 + diff --git a/tests/integration/test_storage_gcp_auth/gcs_mocks/echo.py b/tests/integration/test_storage_gcp_auth/gcs_mocks/echo.py index ff073b4322c4..f96ab335283f 100644 --- a/tests/integration/test_storage_gcp_auth/gcs_mocks/echo.py +++ b/tests/integration/test_storage_gcp_auth/gcs_mocks/echo.py @@ -1,14 +1,80 @@ -import http.server +import json import sys +import urllib.parse +from http import server as http_server counter = 0 expected_path = "/test/test.txt" +# --- Ordinary-contract characterization additions --- +# +# Everything below `expected_path`/`counter` is the original hard-coded single-object mock used by +# test_gcp_auth: unchanged, so that test's token-refresh-count contract stays exactly as it was. +# +# The ordinary-contract test needs more request shapes (PUT, DELETE, LIST, multipart) against +# freely-named objects, plus the ability to inspect what actually reached the wire. `objects` is an +# in-memory bucket keyed by request path; `captured_requests` records every request (method, path, +# lower-cased headers) for the test to fetch and reset independently of the OAuth token counter. +BUCKET_ROOT = "/test/" +objects = {} +generations = {} +multipart_uploads = {} +_next_upload_id = [1] +_next_generation = [1700000000000000] +captured_requests = [] + + +def stable_etag(path): + """A fixed, path-derived ETag distinct from x-goog-generation, so a test can tell whether the + response ETag or the generation reached the SDK's ETag field.""" + return "etag-" + path.strip("/").replace("/", "-") + + +def bump_generation(path): + _next_generation[0] += 1 + generations[path] = _next_generation[0] + return generations[path] + + +class RequestHandler(http_server.BaseHTTPRequestHandler): + def capture(self): + captured_requests.append( + { + "method": self.command, + "path": self.path, + "headers": {name.lower(): value for name, value in self.headers.items()}, + } + ) + + def is_authorized(self): + current_auth = f"Bearer my-secret-token-{counter}" + auth = self.headers.get("Authorization") + return bool(auth) and auth == current_auth + + def read_body(self): + length = int(self.headers.get("Content-Length", 0) or 0) + return self.rfile.read(length) if length else b"" + + def send_plain(self, status, body=b""): + self.send_response(status) + self.send_header("Content-Type", "text/plain") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + if body: + self.wfile.write(body) + + def send_xml(self, status, xml, extra_headers=None): + encoded = xml.encode() + self.send_response(status) + self.send_header("Content-Type", "application/xml") + self.send_header("Content-Length", str(len(encoded))) + for name, value in (extra_headers or {}).items(): + self.send_header(name, value) + self.end_headers() + self.wfile.write(encoded) -class RequestHandler(http.server.BaseHTTPRequestHandler): def process_head(self): global counter - global expected_path current_auth = f"Bearer my-secret-token-{counter}" auth = self.headers.get("Authorization") @@ -29,43 +95,228 @@ def process_head(self): self.end_headers() + def is_original_hardcoded_path(self): + path_only = self.path.split("?")[0] + return self.path.endswith("/ping") or path_only == expected_path + def do_HEAD(self): global counter - self.process_head() + self.capture() + + if self.is_original_hardcoded_path(): + self.process_head() + counter += 1 + return + + path_only = self.path.split("?")[0] + if not self.is_authorized(): + self.send_plain(403) + return + if path_only in objects: + self.send_response(200) + self.send_header("ETag", f'"{stable_etag(path_only)}"') + self.send_header("x-goog-generation", str(generations.get(path_only) or bump_generation(path_only))) + self.send_header("Content-Type", "text/plain") + self.send_header("Content-Length", str(len(objects[path_only]))) + self.end_headers() + else: + self.send_plain(404) counter += 1 def do_GET(self): global counter - global expected_path if self.path.endswith("/reset"): + # Deliberately does NOT capture(): resetting must stay invisible to /captured, or a test + # calling reset-then-fetch would see the reset call itself. counter = 0 + self.send_plain(200, b"OK") + return + + if self.path.endswith("/reset_captured"): + captured_requests.clear() + self.send_plain(200, b"OK") + return + + if self.path.endswith("/captured"): self.send_response(200) - self.send_header("Content-Type", "text/plain") + body = json.dumps(captured_requests).encode() + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) self.end_headers() - self.wfile.write(b"OK") + self.wfile.write(body) return + self.capture() - self.process_head() if self.path.endswith("/ping"): + self.send_plain(200, b"OK") + return + + if not self.is_authorized(): + self.send_plain(403, b"Not authorized") + return + + parsed = urllib.parse.urlsplit(self.path) + path_only = parsed.path + query = urllib.parse.parse_qs(parsed.query, keep_blank_values=True) + + if "list-type" in query: + self.handle_list(query) + counter += 1 + return + + if path_only == expected_path: + self.send_response(200) + self.send_header("ETag", f'"{stable_etag(path_only)}"') + self.send_header("x-goog-generation", str(generations.get(path_only) or bump_generation(path_only))) + self.send_header("Content-Type", "text/plain") + self.send_header("Content-Length", "2") + self.end_headers() self.wfile.write(b"OK") + counter += 1 return - current_auth = f"Bearer my-secret-token-{counter}" - auth = self.headers.get("Authorization") + if path_only in objects: + body = objects[path_only] + self.send_response(200) + self.send_header("ETag", f'"{stable_etag(path_only)}"') + self.send_header("x-goog-generation", str(generations.get(path_only) or bump_generation(path_only))) + self.send_header("Content-Type", "text/plain") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + counter += 1 + return + + self.send_plain(404, b"Not found") + + def handle_list(self, query): + prefix = query.get("prefix", [""])[0] + full_prefix = BUCKET_ROOT + prefix + keys = sorted(k for k in objects if k.startswith(full_prefix)) + contents = "".join( + "" + f"{key[len(BUCKET_ROOT):]}" + f""{stable_etag(key)}"" + f"{len(objects[key])}" + "" + for key in keys + ) + xml = ( + '' + "" + "test" + f"{prefix}" + f"{len(keys)}" + "1000" + "false" + f"{contents}" + "" + ) + self.send_xml(200, xml) + + def do_PUT(self): + self.capture() + + if not self.is_authorized(): + self.send_plain(403) + self.read_body() + return - if not auth or auth != current_auth: - self.wfile.write(b"Not authorized") + parsed = urllib.parse.urlsplit(self.path) + path_only = parsed.path + query = urllib.parse.parse_qs(parsed.query, keep_blank_values=True) + body = self.read_body() + + if "partNumber" in query and "uploadId" in query: + upload_id = query["uploadId"][0] + part_number = int(query["partNumber"][0]) + multipart_uploads.setdefault(upload_id, {"path": path_only, "parts": {}}) + multipart_uploads[upload_id]["parts"][part_number] = body + self.send_response(200) + self.send_header("ETag", f'"{stable_etag(path_only)}-part-{part_number}"') + self.send_header("Content-Length", "0") + self.end_headers() + else: + objects[path_only] = body + generation = bump_generation(path_only) + self.send_response(200) + self.send_header("ETag", f'"{stable_etag(path_only)}"') + self.send_header("x-goog-generation", str(generation)) + self.send_header("Content-Length", "0") + self.end_headers() + counter_bump() + + def do_DELETE(self): + self.capture() + + if not self.is_authorized(): + self.send_plain(403) return - if not self.path.endswith(expected_path): - self.wfile.write(b"Not found") + path_only = self.path.split("?")[0] + objects.pop(path_only, None) + generations.pop(path_only, None) + self.send_response(204) + self.send_header("Content-Length", "0") + self.end_headers() + counter_bump() + + def do_POST(self): + self.capture() + + if not self.is_authorized(): + self.send_plain(403) + self.read_body() return - self.wfile.write(b"OK") - counter += 1 + parsed = urllib.parse.urlsplit(self.path) + path_only = parsed.path + # `keep_blank_values=True` matters here: CreateMultipartUpload's real wire query is the bare + # flag `?uploads`, with no `=value` -- `parse_qs`'s default drops a key with no value entirely, + # which silently turned every CreateMultipartUpload into an unmatched 404 until this was traced. + query = urllib.parse.parse_qs(parsed.query, keep_blank_values=True) + self.read_body() + + if "uploads" in query: + upload_id = f"upload-{_next_upload_id[0]}" + _next_upload_id[0] += 1 + multipart_uploads[upload_id] = {"path": path_only, "parts": {}} + xml = ( + '' + "" + "test" + f"{path_only[len(BUCKET_ROOT):]}" + f"{upload_id}" + "" + ) + self.send_xml(200, xml) + elif "uploadId" in query: + upload_id = query["uploadId"][0] + info = multipart_uploads.pop(upload_id, {"path": path_only, "parts": {}}) + full_body = b"".join(info["parts"][part] for part in sorted(info["parts"])) + objects[path_only] = full_body + generation = bump_generation(path_only) + xml = ( + '' + "" + "test" + f"{path_only[len(BUCKET_ROOT):]}" + f""{stable_etag(path_only)}"" + "" + ) + self.send_xml(200, xml, extra_headers={"x-goog-generation": str(generation)}) + else: + self.send_plain(404) + return + counter_bump() + + +def counter_bump(): + global counter + counter += 1 -httpd = http.server.HTTPServer(("0.0.0.0", int(sys.argv[1])), RequestHandler) +httpd = http_server.HTTPServer(("0.0.0.0", int(sys.argv[1])), RequestHandler) httpd.serve_forever() diff --git a/tests/integration/test_storage_gcp_auth/test.py b/tests/integration/test_storage_gcp_auth/test.py index 01eea16d032b..b5be090d9569 100644 --- a/tests/integration/test_storage_gcp_auth/test.py +++ b/tests/integration/test_storage_gcp_auth/test.py @@ -1,5 +1,7 @@ +import json import logging import os +import re import time import pytest @@ -16,7 +18,11 @@ def started_cluster(): cluster = ClickHouseCluster(__file__) cluster.add_instance( "node", - main_configs=["configs/named_collections.xml"], + main_configs=[ + "configs/named_collections.xml", + "configs/filesystem_caches.xml", + "configs/page_cache.xml", + ], user_configs=["configs/users.xml"], with_minio=True, ) @@ -83,6 +89,14 @@ def run_gcs_mocks(cluster): def test_gcp_auth(started_cluster): + """`gcs_conn`'s URL (`http://resolver:22234/test/`) has no `storage.googleapis.com` in it, so + `Client` never deduces `ProviderType::GCS` for this connection and `api_mode` stays `AWS` for the + whole test -- this characterizes `gcp_oauth` against a proxied/private GCS endpoint (a real, + spec-supported shape), NOT the common case of a user pointing `gcp_oauth` directly at + `storage.googleapis.com`, where `api_mode` can become `GCS` and the `ApiMode::GCS`-gated header + mappings in `Requests.cpp` (see `CopyObjectRequestGetRequestSpecificHeadersRenamesOnlyUnderGcsApiMode` + in `gtest_aws_s3_client.cpp`) would actually fire. + """ node = started_cluster.instances["node"] # Reset mock counters so the test is repeatable @@ -131,3 +145,292 @@ def get_num_requests(): ) assert "AUTHENTICATION_FAILED" in ei.value.stderr + + +def test_gcp_auth_ordinary_contract(started_cluster): + """Pins Default-mode `gcp_oauth` behaviour against the same claims Task 4/5 make in the unit + tests, but for the requests ClickHouse actually issues end-to-end: Bearer authentication still + works, the response ETag is the mock's ordinary one (never the independent x-goog-generation + also present on every response), and no request ever carries `x-goog-if-generation-match` -- + this is the entire point of `Default` mode staying free of CAS's GCS generation dialect. + + `PUT` with `x-amz-meta-*` is unreachable from ordinary SQL on purpose, not by oversight: nothing + in the plain `S3(gcs_conn, ...)` write path fills the object-attributes parameter that would put + `x-amz-meta-*` on the wire -- that plumbing only exists for the CAS envelope. `CopyObject` is + likewise unreachable here: it only happens for a same-object-storage `MergeTree` part move on a + `Disk`, which this named collection does not configure. Both are covered instead, and more + precisely, by direct SDK request construction in `gtest_aws_s3_client.cpp` / `gtest_goog4_signer.cpp`. + + This test also inherits `test_gcp_auth`'s fidelity gap: `gcs_conn`'s endpoint has no + `storage.googleapis.com` substring, so it runs with `api_mode` staying `AWS`, never `GCS` -- the + `ApiMode::GCS`-gated header mappings in `Requests.cpp` do not fire on this path either. See the + `test_gcp_auth` docstring and `CopyObjectRequestGetRequestSpecificHeadersRenamesOnlyUnderGcsApiMode` + in `gtest_aws_s3_client.cpp` for where that mechanism actually gets exercised. + """ + node = started_cluster.instances["node"] + resolver_id = started_cluster.get_container_id("resolver") + + def reset(): + for port in [80, 22234]: + started_cluster.exec_in_container( + resolver_id, ["curl", "-s", f"http://localhost:{port}/reset"], nothrow=True + ) + started_cluster.exec_in_container( + resolver_id, + ["curl", "-s", "http://localhost:22234/reset_captured"], + nothrow=True, + ) + + def get_num_requests(): + count_response = started_cluster.exec_in_container( + resolver_id, ["curl", "-s", "http://localhost/counter"], nothrow=True + ) + return int(count_response) + + def get_captured(): + raw = started_cluster.exec_in_container( + resolver_id, ["curl", "-s", "http://localhost:22234/captured"], nothrow=True + ) + return json.loads(raw) + + def assert_default_oauth(requests): + for request in requests: + headers = request["headers"] + assert headers.get("authorization", "").startswith("Bearer "), request + for name in ( + "x-goog-if-generation-match", + "if-match", + "if-none-match", + "x-amz-copy-source", + "x-goog-copy-source", + ): + assert name not in headers, request + assert not [name for name in headers if name.startswith("x-goog-meta-")], request + + reset() + + node.query("DROP TABLE IF EXISTS s3_ordinary_write") + node.query( + "CREATE TABLE s3_ordinary_write (line String) ENGINE = S3(gcs_conn, filename='ordinary.txt', format='LineAsString')" + ) + + # PUT: bearer authentication drives a real write; the token-refresh count moving at all proves + # the request went through the same OAuth path as the pre-existing test_gcp_auth PUT/GET traffic. + before_write = get_num_requests() + node.query("INSERT INTO s3_ordinary_write VALUES ('hello')") + assert get_num_requests() > before_write + + put_requests = [r for r in get_captured() if r["method"] == "PUT"] + assert put_requests, "expected the INSERT to issue a PUT" + assert all(r["path"].split("?", 1)[0] == "/test/ordinary.txt" for r in put_requests) + assert_default_oauth(put_requests) + + # GET/HEAD: the response carries both a stable ETag and an independent x-goog-generation (set by + # the PUT above); a Default read must come back as the ordinary content, not fail or reinterpret + # the generation as the object's identity. + reset() + assert node.query("SELECT * FROM s3_ordinary_write") == "hello\n" + read_requests = get_captured() + assert any( + r["method"] == "HEAD" and r["path"].split("?", 1)[0] == "/test/ordinary.txt" + for r in read_requests + ), "expected the direct-key read to issue HEAD for ordinary.txt" + assert any( + r["method"] == "GET" and r["path"].split("?", 1)[0] == "/test/ordinary.txt" + for r in read_requests + ), "expected the direct-key read to issue GET for ordinary.txt" + assert_default_oauth(read_requests) + + # LIST: a glob forces a real ListObjectsV2 call (`list-type=2`), independent of the single-key + # GET/HEAD path above. + reset() + assert ( + node.query( + "SELECT * FROM s3(gcs_conn, filename='ordinary*.txt', format='LineAsString')" + ) + == "hello\n" + ) + list_requests = [ + r for r in get_captured() if r["method"] == "GET" and "list-type=2" in r["path"] + ] + assert list_requests, "expected the glob read to issue a ListObjectsV2 request" + assert_default_oauth(list_requests) + + # DELETE: TRUNCATE on the S3 engine removes the underlying object. + reset() + node.query("TRUNCATE TABLE s3_ordinary_write") + delete_requests = [r for r in get_captured() if r["method"] == "DELETE"] + assert delete_requests, "expected TRUNCATE to issue a DELETE" + assert all(r["path"] == "/test/ordinary.txt" for r in delete_requests) + assert_default_oauth(delete_requests) + + # Multipart-sized write: `gcs_conn_multipart` lowers the part-size thresholds so even a small + # INSERT forces CreateMultipartUpload / UploadPart / CompleteMultipartUpload. + reset() + node.query("DROP TABLE IF EXISTS s3_multipart_write") + node.query( + "CREATE TABLE s3_multipart_write (line String) ENGINE = S3(gcs_conn_multipart, filename='multipart.txt', format='LineAsString')" + ) + payload = "x" * (2 * 1024 * 1024) + node.query(f"INSERT INTO s3_multipart_write VALUES ('{payload}')") + + multipart_requests = get_captured() + assert any( + r["method"] == "POST" and "uploads" in r["path"] for r in multipart_requests + ), "expected CreateMultipartUpload" + assert any( + r["method"] == "PUT" and "partNumber" in r["path"] for r in multipart_requests + ), "expected UploadPart" + assert any( + r["method"] == "POST" and "uploadId=" in r["path"] and "uploads" not in r["path"] + for r in multipart_requests + ), "expected CompleteMultipartUpload" + assert_default_oauth(multipart_requests) + + node.query("DROP TABLE s3_ordinary_write") + node.query("DROP TABLE s3_multipart_write") + + +def test_gcp_auth_etag_and_cache_isolation(started_cluster): + """Regression fence for the user-visible half of the isolation plan: a `Default`-mode `gcp_oauth` + response must expose the mock's stable ETag as `_etag`, never `x-goog-generation` (which the mock + also sets, as an unrelated large counter, on every HEAD/GET/PUT/CompleteMultipartUpload response), + regardless of whether the metadata reached ClickHouse through a LIST (the XML body's ``) or a + HEAD/GET (the `ETag` header). Because `_etag` feeds the filesystem-cache key and the page-cache key + (`StorageObjectStorageSource.cpp`), a blanket generation substitution on only some response kinds + would split those caches by read path for the identical object. This is exercised end to end + rather than at the unit level, because the failure mode is in which responses the substitution + reaches, not in the cache-key hashing itself (deterministic regardless of its input, so it cannot + by itself catch a wrong-but-consistent input). + + A third `_etag` consumer named by the plan, the Parquet metadata cache + (`ParquetMetadataCache::createKey`), is deliberately NOT covered here. Proving it end to end + requires a genuinely cold metadata-cache read with both body caches off, which forces a real + ranged HTTP GET for the row-group `OffsetIndex` -- `gcs_mocks/echo.py` ignores the `Range` header + entirely and always returns the full object, so that read gets the wrong bytes and Parquet's + thrift parser rejects them (`TProtocolException: Invalid data`) regardless of ETag isolation. + Every other read in this file tolerates that gap because it goes through a body cache that, once + warm, serves sub-ranges from its own local copy rather than issuing a new ranged request to the + mock. Extending the mock to serve real `Range` responses is separate work; until then, the + Parquet-metadata-cache consumer is covered only by the same-shaped unit-level proof for the other + two consumers (Tasks 4-5) plus the shared reasoning that all three key off the identical + `object_info.metadata->etag` value validated by the assertions above. + """ + node = started_cluster.instances["node"] + resolver_id = started_cluster.get_container_id("resolver") + + def reset(): + for port in [80, 22234]: + started_cluster.exec_in_container( + resolver_id, ["curl", "-s", f"http://localhost:{port}/reset"], nothrow=True + ) + + reset() + + object_name = "cache_isolation.parquet" + node.query( + f"INSERT INTO FUNCTION s3(gcs_conn, filename='{object_name}', format='Parquet') " + f"SELECT number FROM numbers(2000) SETTINGS s3_truncate_on_insert=1" + ) + + # HEAD/GET path: a direct key, no glob. + etag_head = node.query( + f"SELECT _etag FROM s3(gcs_conn, filename='{object_name}', format='Parquet') LIMIT 1" + ).strip() + + # LIST path: a glob forces ListObjectsV2, whose XML body carries its own , independent of + # the HEAD/GET header path above. + etag_list = node.query( + f"SELECT _etag FROM s3(gcs_conn, filename='cache_isolation*.parquet', format='Parquet') LIMIT 1" + ).strip() + + assert etag_head == etag_list, (etag_head, etag_list) + # The mock's generation counter is a large, purely-numeric string (see `_next_generation` / + # `bump_generation` in `gcs_mocks/echo.py`); its ETag never is (`stable_etag` prefixes with + # "etag-"). A blanket generation-for-ETag substitution -- the bug this plan fixes -- would make + # `_etag` numeric here; this assertion is fireable because the two formats cannot collide. + assert not re.fullmatch(r"\d+", etag_head), etag_head + + # --- Filesystem cache: the first read is LIST-sourced (glob), the second is HEAD/GET-sourced (a + # direct key). The cache key is `SipHash(path, etag)` + # (`StorageObjectStorageSource.cpp`), so the second read can only be served from cache -- with no + # further `GetObject` call -- if the two read paths agree on `etag` for the identical object. A + # generation leaking into only one of the two response kinds would make this a cache miss. + fs_settings = "filesystem_cache_name='gcp_oauth_cache1', enable_filesystem_cache=1, use_page_cache_for_object_storage=0" + fs_query_id = f"fs-{object_name}-1" + node.query( + f"SELECT sum(ignore(*)) FROM s3(gcs_conn, filename='cache_isolation*.parquet', format='Parquet') SETTINGS {fs_settings}", + query_id=fs_query_id, + ) + node.query("SYSTEM FLUSH LOGS") + write_bytes = int( + node.query( + f"SELECT ProfileEvents['CachedReadBufferCacheWriteBytes'] FROM system.query_log " + f"WHERE query_id='{fs_query_id}' AND type='QueryFinish'" + ) + ) + assert write_bytes > 0 + + node.query("SYSTEM CLEAR SCHEMA CACHE") + fs_query_id_2 = f"fs-{object_name}-2" + node.query( + f"SELECT sum(ignore(*)) FROM s3(gcs_conn, filename='{object_name}', format='Parquet') SETTINGS {fs_settings}", + query_id=fs_query_id_2, + ) + node.query("SYSTEM FLUSH LOGS") + read_bytes, misses, gets = node.query( + f"SELECT ProfileEvents['CachedReadBufferReadFromCacheBytes'], " + f"ProfileEvents['CachedReadBufferReadFromCacheMisses'], ProfileEvents['S3GetObject'] " + f"FROM system.query_log WHERE query_id='{fs_query_id_2}' AND type='QueryFinish'" + ).split("\t") + # Not `read_bytes == write_bytes`: `CachedReadBufferCacheWriteBytes` counts one physical + # population of the cache, while `CachedReadBufferReadFromCacheBytes` sums every buffer instance + # that reads through the cache in that query (schema resolution, prefetch, and the execution read + # each open their own `CachedOnDiskReadBufferFromFile` and each re-reads the small cached object in + # full) -- for this object that was observed to be exactly 3x on a clean second read, so the two + # counters are not comparable quantities even when nothing is wrong. What isolation actually + # requires is that every one of those reads is a hit: zero cache misses, and no `GetObject` at all. + assert int(read_bytes) > 0 + assert int(misses) == 0 + assert int(gets) == 0 + + # --- Page cache: same cross-path shape as the filesystem cache above (LIST-sourced warm read, + # then a HEAD/GET-sourced read that must hit), over the independent page-cache key + # `"etag:" + etag` (`StorageObjectStorageSource.cpp`). Mutually exclusive with the filesystem + # cache in the read pipeline (`use_page_cache` in `StorageObjectStorageSource::createReadBuffer` + # requires `!use_filesystem_cache`), so this is its own query with the filesystem cache off. --- + node.query("SYSTEM CLEAR SCHEMA CACHE") + pc_settings = "enable_filesystem_cache=0, use_page_cache_for_object_storage=1" + pc_query_id = f"pc-{object_name}-1" + node.query( + f"SELECT sum(ignore(*)) FROM s3(gcs_conn, filename='cache_isolation*.parquet', format='Parquet') SETTINGS {pc_settings}", + query_id=pc_query_id, + ) + node.query("SYSTEM FLUSH LOGS") + misses = int( + node.query( + f"SELECT ProfileEvents['PageCacheMisses'] FROM system.query_log " + f"WHERE query_id='{pc_query_id}' AND type='QueryFinish'" + ) + ) + assert misses > 0 + + node.query("SYSTEM CLEAR SCHEMA CACHE") + pc_query_id_2 = f"pc-{object_name}-2" + node.query( + f"SELECT sum(ignore(*)) FROM s3(gcs_conn, filename='{object_name}', format='Parquet') " + f"SETTINGS {pc_settings}, read_from_page_cache_if_exists_otherwise_bypass_cache=1", + query_id=pc_query_id_2, + ) + node.query("SYSTEM FLUSH LOGS") + hits, misses_2, gets = node.query( + f"SELECT ProfileEvents['PageCacheHits'], ProfileEvents['PageCacheMisses'], " + f"ProfileEvents['S3GetObject'] FROM system.query_log " + f"WHERE query_id='{pc_query_id_2}' AND type='QueryFinish'" + ).split("\t") + assert int(hits) > 0 + assert int(misses_2) == 0 + assert int(gets) == 0 + + # Parquet metadata cache is not exercised here -- see the function docstring: proving it cold + # requires a real ranged GET that this mock cannot serve correctly. diff --git a/tests/queries/0_stateless/01271_show_privileges.reference b/tests/queries/0_stateless/01271_show_privileges.reference index 5f16f7ce7a47..30e02d7c9873 100644 --- a/tests/queries/0_stateless/01271_show_privileges.reference +++ b/tests/queries/0_stateless/01271_show_privileges.reference @@ -171,6 +171,13 @@ SYSTEM RELOAD ASYNCHRONOUS METRICS ['RELOAD ASYNCHRONOUS METRICS'] GLOBAL SYSTEM SYSTEM RECONNECT ZOOKEEPER ['SYSTEM RECONNECT ZOOKEEPER','RECONNECT ZOOKEEPER'] GLOBAL SYSTEM SYSTEM RELOAD [] \N SYSTEM SYSTEM RESTART DISK ['SYSTEM RESTART DISK'] GLOBAL SYSTEM +SYSTEM CAS GC RUN ['SYSTEM CAS GC RUN'] GLOBAL SYSTEM +SYSTEM CAS GC REBUILD ['SYSTEM CAS GC REBUILD'] GLOBAL SYSTEM +SYSTEM CAS DROP POOL MEMBER ['SYSTEM CAS DROP POOL MEMBER'] GLOBAL SYSTEM +SYSTEM CAS FSCK ['SYSTEM CAS FSCK'] GLOBAL SYSTEM +SYSTEM CAS FORGET ['SYSTEM CAS FORGET'] GLOBAL SYSTEM +SYSTEM CAS GC STOP ['SYSTEM CAS GC STOP'] GLOBAL SYSTEM +SYSTEM CAS GC START ['SYSTEM CAS GC START'] GLOBAL SYSTEM SYSTEM WAIT BLOBS CLEANUP ['SYSTEM WAIT BLOBS CLEANUP'] GLOBAL SYSTEM SYSTEM MERGES ['SYSTEM STOP MERGES','SYSTEM START MERGES','STOP MERGES','START MERGES'] TABLE SYSTEM SYSTEM TTL MERGES ['SYSTEM STOP TTL MERGES','SYSTEM START TTL MERGES','STOP TTL MERGES','START TTL MERGES'] TABLE SYSTEM diff --git a/tests/queries/0_stateless/02253_empty_part_checksums.sh b/tests/queries/0_stateless/02253_empty_part_checksums.sh new file mode 100755 index 000000000000..af4e1d896658 --- /dev/null +++ b/tests/queries/0_stateless/02253_empty_part_checksums.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Tags: zookeeper, no-replicated-database, no-shared-merge-tree, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) +# no-replicated-database because it adds extra replicas +# no-shared-merge-tree do something with parts on local fs +# add_minmax_index_for_numeric_columns=0: Adds extra files, which changes the hashes + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt sync;" +$CLICKHOUSE_CLIENT -q "CREATE TABLE rmt (a UInt8, b Int16, c Float32, d String, e Array(UInt8), f Nullable(UUID), g Tuple(UInt8, UInt16)) +ENGINE = ReplicatedMergeTree('/test/02253/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmtt', '1') ORDER BY a PARTITION BY b % 10 +SETTINGS old_parts_lifetime = 1, cleanup_delay_period = 0, cleanup_delay_period_random_add = 0, compress_marks=1, compress_primary_key=1, serialization_info_version = 'basic', +cleanup_thread_preferred_points_per_iteration=0, min_bytes_for_wide_part=0, remove_empty_parts=0, replace_long_file_name_to_hash=0, add_minmax_index_for_numeric_columns=0" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "INSERT INTO rmt SELECT rand(1), 0, 1 / rand(3), toString(rand(4)), [rand(5), rand(6)], rand(7) % 2 ? NULL : generateUUIDv4(), (rand(8), rand(9)) FROM numbers(1000);" + +$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" +$CLICKHOUSE_CLIENT -q "select count() from rmt" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt' and name='0_0_0_0'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -rf "$path" + +# detach the broken part, replace it with empty one +$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" 2>/dev/null +$CLICKHOUSE_CLIENT -q "select count() from rmt" + +$CLICKHOUSE_CLIENT --receive_timeout=60 -q "system sync replica rmt" + +# the empty part should pass the check +$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" +$CLICKHOUSE_CLIENT -q "select count() from rmt" + +$CLICKHOUSE_CLIENT -q "select name, part_type, hash_of_all_files, hash_of_uncompressed_files, uncompressed_hash_of_compressed_files from system.parts where database=currentDatabase()" + +$CLICKHOUSE_CLIENT -q "drop table rmt sync;" diff --git a/tests/queries/0_stateless/02254_projection_broken_part.sh b/tests/queries/0_stateless/02254_projection_broken_part.sh new file mode 100755 index 000000000000..84f4ef8bfe53 --- /dev/null +++ b/tests/queries/0_stateless/02254_projection_broken_part.sh @@ -0,0 +1,45 @@ +#!/usr/bin/env bash +# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" + +$CLICKHOUSE_CLIENT -q "create table projection_broken_parts_1 (a int, b int, projection ab (select a, sum(b) group by a)) + engine = ReplicatedMergeTree('/test/02254/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r1') + order by a settings index_granularity = 1;" + +$CLICKHOUSE_CLIENT -q "create table projection_broken_parts_2 (a int, b int, projection ab (select a, sum(b) group by a)) + engine = ReplicatedMergeTree('/test/02254/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r2') + order by a settings index_granularity = 1;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into projection_broken_parts_1 values (1, 1), (1, 2), (1, 3);" +$CLICKHOUSE_CLIENT -q "system sync replica projection_broken_parts_2;" +$CLICKHOUSE_CLIENT -q "select 1, *, _part from projection_broken_parts_2 order by b;" +$CLICKHOUSE_CLIENT -q "select 2, sum(b) from projection_broken_parts_2 group by a;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='projection_broken_parts_1' and name='all_0_0_0'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -f "$path/ab.proj/data.bin" + +$CLICKHOUSE_CLIENT -q "select 3, sum(b) from projection_broken_parts_1 group by a format Null;" 2>/dev/null + +num_tries=0 +while ! $CLICKHOUSE_CLIENT -q "select 4, sum(b) from projection_broken_parts_1 group by a format Null;" 2>/dev/null; do + sleep 1; + num_tries=$((num_tries+1)) + if [ $num_tries -eq 60 ]; then + break + fi +done + +$CLICKHOUSE_CLIENT -q "system sync replica projection_broken_parts_1;" +$CLICKHOUSE_CLIENT -q "select 5, sum(b) from projection_broken_parts_1 group by a;" + +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" diff --git a/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh b/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh new file mode 100755 index 000000000000..de16ba1a0bff --- /dev/null +++ b/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" + +$CLICKHOUSE_CLIENT -q "create table rmt1 (a int, b int) + engine = ReplicatedMergeTree('/test/02255/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r1') order by a settings old_parts_lifetime=100500;" + +$CLICKHOUSE_CLIENT -q "create table rmt2 (a int, b int) + engine = ReplicatedMergeTree('/test/02255/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r2') order by a settings old_parts_lifetime=100500;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1, 1), (1, 2), (1, 3);" +$CLICKHOUSE_CLIENT -q "alter table rmt1 update b = b*10 where 1 settings mutations_sync=1" +$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" +$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt2 order by b;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_0_0'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -f "$path/data.bin" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_0_0_1'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -f "$path/data.bin" + +$CLICKHOUSE_CLIENT -q "detach table rmt1 sync" +$CLICKHOUSE_CLIENT -q "attach table rmt1" 2>/dev/null + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by b;" + +$CLICKHOUSE_CLIENT -q "truncate table rmt1" + +$CLICKHOUSE_CLIENT -q "SELECT table, lost_part_count FROM system.replicas WHERE database=currentDatabase() AND lost_part_count!=0"; + +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" diff --git a/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh b/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh new file mode 100755 index 000000000000..cc4a3b53b957 --- /dev/null +++ b/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Tags: zookeeper, no-shared-merge-tree, long, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) +# no-shared-merge-tree: depend on local fs + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" + +$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') order by n;" +$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') order by n;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" +$CLICKHOUSE_CLIENT -q "system stop merges rmt2;" +$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" + +$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by n;" +$CLICKHOUSE_CLIENT -q "select 2, *, _part from rmt2 order by n;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_1_1'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -rf $path + +$CLICKHOUSE_CLIENT -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR +$CLICKHOUSE_CLIENT --min_bytes_to_use_direct_io=1 --local_filesystem_read_method=pread_threadpool -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR + +$CLICKHOUSE_CLIENT -q "detach table rmt1;" +$CLICKHOUSE_CLIENT -q "attach table rmt1;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" +$CLICKHOUSE_CLIENT -q "system start merges rmt2;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" +$CLICKHOUSE_CLIENT -q "select 3, *, _part from rmt1 order by n;" +$CLICKHOUSE_CLIENT -q "select 4, *, _part from rmt2 order by n;" + +$CLICKHOUSE_CLIENT -q "detach table rmt1;" +$CLICKHOUSE_CLIENT -q "attach table rmt1;" + +$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh b/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh new file mode 100755 index 000000000000..72a749f0ec8e --- /dev/null +++ b/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh @@ -0,0 +1,58 @@ +#!/usr/bin/env bash +# Tags: long, zookeeper, no-shared-merge-tree, no-parallel, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) +# no-shared-merge-tree: depend on local fs (remove parts) + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" + +$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') order by n + settings cleanup_delay_period=0, cleanup_delay_period_random_add=0, cleanup_thread_preferred_points_per_iteration=0, old_parts_lifetime=0" +$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') order by n" + +$CLICKHOUSE_CLIENT -q "system stop replicated sends rmt2" +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt2 values (0);" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1 pull;" + +# There's a stupid effect from "zero copy replication": +# MERGE_PARTS all_1_2_1 can be executed by rmt2 even if it was assigned by rmt1 +# After that, rmt2 will not be able to execute that merge and will only try to fetch the part from rmt2 +# But sends are stopped on rmt2... + +(sleep 5 && $CLICKHOUSE_CLIENT -q "system start replicated sends rmt2") & + +$CLICKHOUSE_CLIENT --optimize_throw_if_noop=1 -q "optimize table rmt1;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" + +$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by n;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_1_2_1'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -rf $path + +$CLICKHOUSE_CLIENT -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR +$CLICKHOUSE_CLIENT --min_bytes_to_use_direct_io=1 --local_filesystem_read_method=pread_threadpool -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR + +$CLICKHOUSE_CLIENT -q "select sleep(0.1) from numbers($(($RANDOM % 30))) settings max_block_size=1 format Null" + +$CLICKHOUSE_CLIENT -q "detach table rmt1;" +$CLICKHOUSE_CLIENT -q "attach table rmt1;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" +$CLICKHOUSE_CLIENT -q "system sync replica rmt1 pull;" +$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "select 3, *, _part from rmt1 order by n;" + +$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh b/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh new file mode 100755 index 000000000000..2f3f38cfba4c --- /dev/null +++ b/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt sync;" +$CLICKHOUSE_CLIENT -q "create table rmt (n int) engine=ReplicatedMergeTree('/test/02444/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', '1') order by n settings old_parts_lifetime=600" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt values (1);" +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt values (2);" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt pull;" +$CLICKHOUSE_CLIENT --optimize_throw_if_noop=1 -q "optimize table rmt final" +$CLICKHOUSE_CLIENT -q "system sync replica rmt;" +$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt order by n;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt' and name='all_1_1_0'") +# ensure that path is absolute before removing +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit +rm -f "$path/*.bin" + +$CLICKHOUSE_CLIENT -q "detach table rmt sync;" +$CLICKHOUSE_CLIENT -q "attach table rmt;" +$CLICKHOUSE_CLIENT -q "select 2, *, _part from rmt order by n;" + +$CLICKHOUSE_CLIENT -q "truncate table rmt;" + +$CLICKHOUSE_CLIENT -q "detach table rmt sync;" +$CLICKHOUSE_CLIENT -q "attach table rmt;" + +$CLICKHOUSE_CLIENT -q "SELECT table, lost_part_count FROM system.replicas WHERE database=currentDatabase() AND lost_part_count!=0"; + +$CLICKHOUSE_CLIENT -q "drop table rmt sync;" diff --git a/tests/queries/0_stateless/02486_truncate_and_unexpected_parts.sql b/tests/queries/0_stateless/02486_truncate_and_unexpected_parts.sql index 29946e315544..c9cec021a6ec 100644 --- a/tests/queries/0_stateless/02486_truncate_and_unexpected_parts.sql +++ b/tests/queries/0_stateless/02486_truncate_and_unexpected_parts.sql @@ -1,4 +1,3 @@ - create table rmt (n int) engine=ReplicatedMergeTree('/test/02468/{database}', '1') order by tuple() partition by n % 2 settings replicated_max_ratio_of_wrong_parts=0, max_suspicious_broken_parts=0, max_suspicious_broken_parts_bytes=0; create table rmt1 (n int) engine=ReplicatedMergeTree('/test/02468/{database}', '2') order by tuple() partition by n % 2 settings replicated_max_ratio_of_wrong_parts=0, max_suspicious_broken_parts=0, max_suspicious_broken_parts_bytes=0; diff --git a/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_MergeTree.sh b/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_MergeTree.sh index 5648db32d189..fe3dbd37e4d2 100755 --- a/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_MergeTree.sh +++ b/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_MergeTree.sh @@ -1,5 +1,8 @@ #!/usr/bin/env bash -# Tags: no-fasttest, no-random-settings, no-random-merge-tree-settings, no-encrypted-storage +# Tags: no-fasttest, no-random-settings, no-random-merge-tree-settings, no-encrypted-storage, no-cas-storage +# Tag no-cas-storage: the test uses an Ordinary database, whose BACKUP path goes via +# temporary hard links - not supported on a cas disk (Code 344 SUPPORT_IS_DISABLED; +# BACKUP/RESTORE, B16/B34). Re-checked on the T13 CA-S3 lane (2026-06-12): still fails for this reason. # Tag no-fasttest: requires S3 # Tag no-random-settings, no-random-merge-tree-settings: to avoid creating extra files like serialization.json, this test too exocit anyway diff --git a/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_ReplicatedMergeTree.sh b/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_ReplicatedMergeTree.sh index 865a43d91ef3..023e6fd8c529 100755 --- a/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_ReplicatedMergeTree.sh +++ b/tests/queries/0_stateless/02980_s3_plain_DROP_TABLE_ReplicatedMergeTree.sh @@ -1,5 +1,6 @@ #!/usr/bin/env bash -# Tags: no-fasttest, no-random-settings, no-random-merge-tree-settings, no-shared-merge-tree, no-encrypted-storage +# Tags: no-fasttest, no-random-settings, no-random-merge-tree-settings, no-shared-merge-tree, no-encrypted-storage, no-cas-storage +# no-cas-storage: BACKUP via temporary hard links is not supported on a cas disk (Code 344 SUPPORT_IS_DISABLED; BACKUP/RESTORE, B16/B34) # Tag no-fasttest: requires S3 # Tag no-random-settings, no-random-merge-tree-settings: to avoid creating extra files like serialization.json, this test too exocit anyway # Tag no-shared-merge-tree: use database ordinary diff --git a/tests/queries/0_stateless/03350_alter_table_fetch_partition_thread_pool.sql b/tests/queries/0_stateless/03350_alter_table_fetch_partition_thread_pool.sql index f5cbd809eef2..dacad1a30788 100644 --- a/tests/queries/0_stateless/03350_alter_table_fetch_partition_thread_pool.sql +++ b/tests/queries/0_stateless/03350_alter_table_fetch_partition_thread_pool.sql @@ -1,4 +1,5 @@ -- Tags: no-parallel, no-replicated-database, no-shared-merge-tree +-- no-cas-storage: FETCH PARTITION is supported on a cas disk (the gate is lifted, byte-fetch lands into detached/, see 05002), but this test fetches a 100-part partition CONCURRENTLY via the FETCH thread pool. Those parallel fetches all read-modify-write the SHARED "detached" ref object; the read side of that hot pointer object is not serialized against the truncating in-place rewrite of the local object storage, so a concurrent reader can see a torn ref/manifest (CANNOT_READ_ALL_DATA / NO_FILE_IN_DATA_PART). The atomic pointer-object publish needed to make concurrent fan-out safe is a deferred backlog item (B66a); single-part FETCH works (01650 + 05002). -- Tag: no-parallel - to avoid polluting FETCH PARTITION thread pool with other fetches -- Tag: no-replicated-database - replica_path is different diff --git a/tests/queries/0_stateless/03352_allow_suspicious_ttl.sql b/tests/queries/0_stateless/03352_allow_suspicious_ttl.sql index 5fd2bb3bf3a4..2a23baffe13c 100644 --- a/tests/queries/0_stateless/03352_allow_suspicious_ttl.sql +++ b/tests/queries/0_stateless/03352_allow_suspicious_ttl.sql @@ -1,4 +1,4 @@ - -- Tags: long, zookeeper +-- Tags: long, zookeeper -- Replicated diff --git a/tests/queries/0_stateless/03541_rename_column_start.sql b/tests/queries/0_stateless/03541_rename_column_start.sql index b5d4fa03f18a..0fa8af8b26b8 100644 --- a/tests/queries/0_stateless/03541_rename_column_start.sql +++ b/tests/queries/0_stateless/03541_rename_column_start.sql @@ -1,4 +1,4 @@ - -- Tags: zookeeper +-- Tags: zookeeper CREATE TABLE rmt (a UInt64, b UInt64) ENGINE=ReplicatedMergeTree('/clickhouse/tables/{database}/rmt', '1') diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh index fc5df9b541da..5909b66ac40b 100755 --- a/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Tags: no-fasttest +# Tags: no-fasttest, no-cas-storage # Tag no-fasttest: requires s3 storage CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh index dff7332662d0..4e5f27b909c2 100755 --- a/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Tags: no-fasttest +# Tags: no-fasttest, no-cas-storage # Tag no-fasttest: requires s3 storage CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh index 0164dd70c4e0..5be58890283d 100755 --- a/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Tags: no-fasttest +# Tags: no-fasttest, no-cas-storage # Tag no-fasttest: requires s3 storage CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) diff --git a/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh index 12b47f4f2664..a57f4159cc98 100755 --- a/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh +++ b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Tags: no-fasttest +# Tags: no-fasttest, no-cas-storage # Tag no-fasttest: requires s3 storage CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) diff --git a/tests/queries/0_stateless/03829_insert_deduplication_info_memory.sql b/tests/queries/0_stateless/03829_insert_deduplication_info_memory.sql index 2dc33bfd11c1..f03853a88cbe 100644 --- a/tests/queries/0_stateless/03829_insert_deduplication_info_memory.sql +++ b/tests/queries/0_stateless/03829_insert_deduplication_info_memory.sql @@ -8,9 +8,13 @@ DROP TABLE IF EXISTS t_dedup_memory; CREATE TABLE t_dedup_memory (x UInt32, fat FixedString(10000)) ENGINE = MergeTree ORDER BY x; -- 10 000 rows * 10 000 bytes FixedString ≈ 100 MB of column data. --- With the bug, original_block doubles this to ~200 MB, exceeding the limit. --- Without the bug, only the data columns are held, fitting within the limit. -SET max_memory_usage = '150M'; +-- With the bug, original_block doubles this to ~200 MB, which must exceed the limit. +-- Without the bug, only the data columns are held (peak ≈ 143 MB), which must fit under it. +-- The limit sits between those two peaks with enough headroom to absorb small, fixed write-path +-- buffering overhead (e.g. a content-addressed disk's insert path adds ~0.5 MB on top of the ~143 MB +-- baseline) so the test still runs unchanged on such a storage backend, while remaining well below the +-- ~200 MB doubling-bug peak it is meant to catch. +SET max_memory_usage = '170M'; INSERT INTO t_dedup_memory SELECT number, toString(number) FROM numbers(10000) SETTINGS max_insert_threads = 1, min_insert_block_size_rows = 0, min_insert_block_size_bytes = 0; diff --git a/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh b/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh new file mode 100755 index 000000000000..7c1a2df1c8fb --- /dev/null +++ b/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage +# no-shared-merge-tree: depends on local fs +# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" + +zk_path="/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/$CLICKHOUSE_DATABASE/replicas/1/parts" + +$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) + engine=ReplicatedMergeTree('/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') + order by n + settings old_parts_lifetime=100500;" + +$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) + engine=ReplicatedMergeTree('/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') + order by n + settings old_parts_lifetime=100500;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" + +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" +$CLICKHOUSE_CLIENT -q "system stop merges rmt2;" +$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" + +path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_1_1'") +$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path') format Null" || exit +rm -rf "$path" + +if $CLICKHOUSE_CLIENT -q "select * from rmt1;" >/dev/null 2>&1; then + echo "Expected read from removed part to fail" + exit 1 +fi + +$CLICKHOUSE_CLIENT -q "detach table rmt1 sync;" +$CLICKHOUSE_CLIENT -q "attach table rmt1;" + +$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" +$CLICKHOUSE_CLIENT -q "system start merges rmt2;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" +$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" +$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" + +$CLICKHOUSE_CLIENT -q "select throwIf(count() = 0, 'Missing all_0_1_1 in ZooKeeper') from system.zookeeper where path='$zk_path' and name='all_0_1_1' format Null" + +$CLICKHOUSE_CLIENT -q "detach table rmt1 sync;" +$CLICKHOUSE_CLIENT -q "attach table rmt1;" + +$CLICKHOUSE_CLIENT -q "select count(), sum(n) from rmt1;" + +$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" +$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/04278_cas_disk.reference b/tests/queries/0_stateless/04278_cas_disk.reference new file mode 100644 index 000000000000..b054d30e9cc7 --- /dev/null +++ b/tests/queries/0_stateless/04278_cas_disk.reference @@ -0,0 +1,9 @@ +basic 1000 499500 7 2020-01-01 2022-09-26 +oracle_full_match 1 +after_second_insert 2000 +after_merge 2000 999000 +active_parts 1 +slice 0 0 2 +slice 500 3 2 +slice 999 5 2 +dropped_ok diff --git a/tests/queries/0_stateless/04278_cas_disk.sql b/tests/queries/0_stateless/04278_cas_disk.sql new file mode 100644 index 000000000000..e169ad53ff43 --- /dev/null +++ b/tests/queries/0_stateless/04278_cas_disk.sql @@ -0,0 +1,49 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- Natural black-box oracle: a table on a `cas` disk must behave +-- identically to a normal MergeTree table for the same data. We compare the two +-- directly so the test is deterministic regardless of environment, and we also +-- exercise INSERT (content-addressed write), SELECT (ref->part_id->footer->blob +-- resolution), blob-level dedup of identical inserts, a merge, and DROP (removal). + +DROP TABLE IF EXISTS t_cas; +DROP TABLE IF EXISTS t_ref; + +CREATE TABLE t_cas (a UInt64, s String, d Date) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04278', + name = '04278_cas', + path = '04278_cas_pool/'); + +CREATE TABLE t_ref (a UInt64, s String, d Date) +ENGINE = MergeTree ORDER BY a; + +INSERT INTO t_cas SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000); +INSERT INTO t_ref SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000); + +SELECT 'basic', count(), sum(a), uniqExact(s), min(d), max(d) FROM t_cas; + +SELECT 'oracle_full_match', + (SELECT groupArray((a, s, d)) FROM (SELECT * FROM t_cas ORDER BY a)) + = (SELECT groupArray((a, s, d)) FROM (SELECT * FROM t_ref ORDER BY a)); + +-- Second identical insert: rows double (plain MergeTree does not dedup rows); +-- the content blobs are deduplicated internally by content-addressing. +INSERT INTO t_cas SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000); +SELECT 'after_second_insert', count() FROM t_cas; + +OPTIMIZE TABLE t_cas FINAL; +SELECT 'after_merge', count(), sum(a) FROM t_cas; +SELECT 'active_parts', count() FROM system.parts +WHERE database = currentDatabase() AND table = 't_cas' AND active; + +SELECT 'slice', a, s, count() FROM t_cas WHERE a IN (0, 500, 999) GROUP BY a, s ORDER BY a; + +DROP TABLE t_cas; +DROP TABLE t_ref; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04279_cas_gc.reference b/tests/queries/0_stateless/04279_cas_gc.reference new file mode 100644 index 000000000000..d9fadb0b0b4d --- /dev/null +++ b/tests/queries/0_stateless/04279_cas_gc.reference @@ -0,0 +1,6 @@ +count_match 1 +sum_match 1 +content_match 1 +point_match 1 +range_match 1 +dropped_ok diff --git a/tests/queries/0_stateless/04279_cas_gc.sql b/tests/queries/0_stateless/04279_cas_gc.sql new file mode 100644 index 000000000000..563fb2442313 --- /dev/null +++ b/tests/queries/0_stateless/04279_cas_gc.sql @@ -0,0 +1,66 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- Correctness-under-active-GC oracle. The `cas` disk below opts in to the background +-- reachability GC (`gc_enabled=1`, default OFF) and runs it aggressively +-- and `old_parts_lifetime=1` drops the merged-away source parts quickly so their footers/blobs become +-- unreferenced and turn into genuine GC fodder *during* the test. We assert the CA table stays +-- byte-for-byte identical to a normal MergeTree table on the same data: if a concurrent sweep ever +-- dropped a live blob, the oracle below would diverge. The test is fully deterministic — it never +-- waits on GC and never asserts that GC has run by a deadline; correctness must hold regardless of +-- whether (and how often) the background sweep fired. + +DROP TABLE IF EXISTS t_cas_gc; +DROP TABLE IF EXISTS t_ref_gc; + +CREATE TABLE t_cas_gc (a UInt64, s String, d Date) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04279', + name = '04279_cas_gc', + path = '04279_cas_gc_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1), + old_parts_lifetime = 1; + +CREATE TABLE t_ref_gc (a UInt64, s String, d Date) +ENGINE = MergeTree ORDER BY a +SETTINGS old_parts_lifetime = 1; + +-- First insert: rows 0..999. +INSERT INTO t_cas_gc SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000); +INSERT INTO t_ref_gc SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000); + +-- Second distinct insert: rows 1000..1999 (different data => different blobs, not deduped away). +INSERT INTO t_cas_gc SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000, 1000); +INSERT INTO t_ref_gc SELECT number, toString(number % 7), toDate('2020-01-01') + number FROM numbers(1000, 1000); + +-- Merge both tables: on the CA table this leaves the source parts outdated; with old_parts_lifetime=1 +-- they are removed shortly, so their footers/blobs become unreferenced and the active GC may sweep them. +OPTIMIZE TABLE t_cas_gc FINAL; +OPTIMIZE TABLE t_ref_gc FINAL; + +-- Oracle: aggregates must match the normal table exactly. +SELECT 'count_match', (SELECT count() FROM t_cas_gc) = (SELECT count() FROM t_ref_gc); +SELECT 'sum_match', (SELECT sum(a) FROM t_cas_gc) = (SELECT sum(a) FROM t_ref_gc); + +-- Oracle: full ordered content must match exactly — proves no live blob was dropped by a concurrent sweep. +SELECT 'content_match', + (SELECT groupArray((a, s, d)) FROM (SELECT * FROM t_cas_gc ORDER BY a, s, d)) + = (SELECT groupArray((a, s, d)) FROM (SELECT * FROM t_ref_gc ORDER BY a, s, d)); + +-- A few point/range reads (each resolves ref -> part_id -> footer -> blob) must also match the oracle. +SELECT 'point_match', + (SELECT groupArray((a, s)) FROM (SELECT a, s FROM t_cas_gc WHERE a IN (0, 999, 1000, 1999) ORDER BY a, s)) + = (SELECT groupArray((a, s)) FROM (SELECT a, s FROM t_ref_gc WHERE a IN (0, 999, 1000, 1999) ORDER BY a, s)); + +SELECT 'range_match', + (SELECT groupArray((a, s)) FROM (SELECT a, s FROM t_cas_gc WHERE a BETWEEN 500 AND 1500 ORDER BY a, s)) + = (SELECT groupArray((a, s)) FROM (SELECT a, s FROM t_ref_gc WHERE a BETWEEN 500 AND 1500 ORDER BY a, s)); + +DROP TABLE t_cas_gc; +DROP TABLE t_ref_gc; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04280_cas_clone_partition_works.reference b/tests/queries/0_stateless/04280_cas_clone_partition_works.reference new file mode 100644 index 000000000000..13197e78dee8 --- /dev/null +++ b/tests/queries/0_stateless/04280_cas_clone_partition_works.reference @@ -0,0 +1,8 @@ +after_replace_dst_p1 100 4950 +after_attach_from_dst_p2 50 1225 +after_move_dst_p3 30 435 +after_move_src_p3 0 +after_detach_src 50 +after_reattach_src 100 4950 +after_drop_src_p2 0 +dropped_ok diff --git a/tests/queries/0_stateless/04280_cas_clone_partition_works.sql b/tests/queries/0_stateless/04280_cas_clone_partition_works.sql new file mode 100644 index 000000000000..82e18998cb59 --- /dev/null +++ b/tests/queries/0_stateless/04280_cas_clone_partition_works.sql @@ -0,0 +1,54 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- Positive test for the part-cloning partition commands on a cas disk. +-- +-- History: these commands (MOVE PARTITION ... TO TABLE, REPLACE PARTITION, ATTACH PARTITION ... FROM, +-- plain ATTACH PARTITION of a table's own detached parts) USED to be rejected with SUPPORT_IS_DISABLED +-- on CA, because the file-by-file `createHardLink` clone path had no enclosing transaction and would +-- corrupt the clone. CAS M9 W2 made that path transactional: `DataPartStorageOnDiskBase::freeze` runs +-- the whole clone through ONE CA transaction and `moveDirectory` re-keys the detached-staging → active +-- rename into a complete active ref. So the commands now SUCCEED and read back identical data. This test +-- locks that they work and produce the correct rows (the gate at `checkAlterPartitionIsPossible` for the +-- ContentAddressed metadata type now lists them as supported). +-- +-- The tables use the DEFAULT MergeTree storage so this exercises whatever default disk the job installs: +-- on the local-CA job that is a local cas disk, on the cas-over-S3 job it is +-- a CA disk backed by minio. Both are the supported same-disk clone path. + +DROP TABLE IF EXISTS t_cas_clone_src; +DROP TABLE IF EXISTS t_cas_clone_dst; + +CREATE TABLE t_cas_clone_src (a UInt64, p UInt8) ENGINE = MergeTree PARTITION BY p ORDER BY a; +CREATE TABLE t_cas_clone_dst (a UInt64, p UInt8) ENGINE = MergeTree PARTITION BY p ORDER BY a; + +INSERT INTO t_cas_clone_src SELECT number, 1 FROM numbers(100); +INSERT INTO t_cas_clone_src SELECT number, 2 FROM numbers(50); +INSERT INTO t_cas_clone_src SELECT number, 3 FROM numbers(30); + +-- REPLACE PARTITION clones parts from another table into the destination. +ALTER TABLE t_cas_clone_dst REPLACE PARTITION 1 FROM t_cas_clone_src; +SELECT 'after_replace_dst_p1', count(), sum(a) FROM t_cas_clone_dst WHERE p = 1; + +-- ATTACH PARTITION ... FROM clones parts from another table (parses to REPLACE_PARTITION, replace=false). +ALTER TABLE t_cas_clone_dst ATTACH PARTITION 2 FROM t_cas_clone_src; +SELECT 'after_attach_from_dst_p2', count(), sum(a) FROM t_cas_clone_dst WHERE p = 2; + +-- MOVE PARTITION ... TO TABLE clones a partition to the destination and drops it from the source. +ALTER TABLE t_cas_clone_src MOVE PARTITION 3 TO TABLE t_cas_clone_dst; +SELECT 'after_move_dst_p3', count(), sum(a) FROM t_cas_clone_dst WHERE p = 3; +SELECT 'after_move_src_p3', count() FROM t_cas_clone_src WHERE p = 3; + +-- Plain ATTACH PARTITION of the table's own detached part re-clones it back. +ALTER TABLE t_cas_clone_src DETACH PARTITION 1; +SELECT 'after_detach_src', count() FROM t_cas_clone_src; +ALTER TABLE t_cas_clone_src ATTACH PARTITION 1; +SELECT 'after_reattach_src', count(), sum(a) FROM t_cas_clone_src WHERE p = 1; + +-- The pointer-unlink command DROP PARTITION still works. +ALTER TABLE t_cas_clone_src DROP PARTITION 2; +SELECT 'after_drop_src_p2', count() FROM t_cas_clone_src WHERE p = 2; + +DROP TABLE t_cas_clone_src; +DROP TABLE t_cas_clone_dst; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04282_cas_mutable_state.reference b/tests/queries/0_stateless/04282_cas_mutable_state.reference new file mode 100644 index 000000000000..6291270c33ba --- /dev/null +++ b/tests/queries/0_stateless/04282_cas_mutable_state.reference @@ -0,0 +1,7 @@ +oracle_full_match 1 +counts 1000 249500 +cas_active_parts 2 +cas_distinct_uuids 2 +cas_no_zero_uuid 0 +ref_distinct_uuids 2 +dropped_ok diff --git a/tests/queries/0_stateless/04282_cas_mutable_state.sql b/tests/queries/0_stateless/04282_cas_mutable_state.sql new file mode 100644 index 000000000000..daffbafb24e8 --- /dev/null +++ b/tests/queries/0_stateless/04282_cas_mutable_state.sql @@ -0,0 +1,57 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- B23 mutable-per-part-state oracle: with assign_part_uuids=1, two INSERTs of IDENTICAL data produce +-- two parts whose column content is byte-identical but whose per-part uuid.txt differs. On a +-- cas disk the two parts dedup to ONE shared manifest, while their mutable per-part +-- files (uuid.txt / txn_version.txt / metadata_version.txt) live in a per-ref sidecar and are +-- overlaid on read. Before B23 the second part read the FIRST part's uuid (the shared manifest +-- embedded one part's mutable files), so the two uuids collided. This is a natural black-box oracle: +-- the cas table must behave exactly like a normal MergeTree table, and the two parts +-- must carry two DISTINCT uuids. + +DROP TABLE IF EXISTS t_cas_mut; +DROP TABLE IF EXISTS t_ref_mut; + +CREATE TABLE t_cas_mut (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS assign_part_uuids = 1, disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04282', + name = '04282_cas', + path = '04282_cas_pool/'); + +CREATE TABLE t_ref_mut (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS assign_part_uuids = 1; + +-- Two IDENTICAL inserts → two parts with identical content but distinct per-part uuids. +INSERT INTO t_cas_mut SELECT number, toString(number % 5) FROM numbers(500); +INSERT INTO t_cas_mut SELECT number, toString(number % 5) FROM numbers(500); +INSERT INTO t_ref_mut SELECT number, toString(number % 5) FROM numbers(500); +INSERT INTO t_ref_mut SELECT number, toString(number % 5) FROM numbers(500); + +-- Data oracle: the cas table matches the normal table exactly. +SELECT 'oracle_full_match', + (SELECT groupArray((a, s)) FROM (SELECT * FROM t_cas_mut ORDER BY a, s)) + = (SELECT groupArray((a, s)) FROM (SELECT * FROM t_ref_mut ORDER BY a, s)); + +SELECT 'counts', count(), sum(a) FROM t_cas_mut; + +-- Two active parts, each with its OWN non-empty, DISTINCT uuid (the B23 regression: no collision). +SELECT 'cas_active_parts', count() FROM system.parts +WHERE database = currentDatabase() AND table = 't_cas_mut' AND active; +SELECT 'cas_distinct_uuids', uniqExact(uuid) FROM system.parts +WHERE database = currentDatabase() AND table = 't_cas_mut' AND active; +SELECT 'cas_no_zero_uuid', countIf(uuid = toUUID('00000000-0000-0000-0000-000000000000')) FROM system.parts +WHERE database = currentDatabase() AND table = 't_cas_mut' AND active; + +-- The same property holds on the normal table (oracle for the uuid behaviour). +SELECT 'ref_distinct_uuids', uniqExact(uuid) FROM system.parts +WHERE database = currentDatabase() AND table = 't_ref_mut' AND active; + +DROP TABLE t_cas_mut; +DROP TABLE t_ref_mut; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04283_cas_replicated_rejected.reference b/tests/queries/0_stateless/04283_cas_replicated_rejected.reference new file mode 100644 index 000000000000..241bd6aee164 --- /dev/null +++ b/tests/queries/0_stateless/04283_cas_replicated_rejected.reference @@ -0,0 +1,5 @@ +repl_count 2 +repl_sum 30 +plain_count 100 +plain_sum 9900 +dropped_ok diff --git a/tests/queries/0_stateless/04283_cas_replicated_rejected.sql b/tests/queries/0_stateless/04283_cas_replicated_rejected.sql new file mode 100644 index 000000000000..aed167652fcd --- /dev/null +++ b/tests/queries/0_stateless/04283_cas_replicated_rejected.sql @@ -0,0 +1,47 @@ +-- Tags: no-fasttest, no-shared-merge-tree +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. +-- no-shared-merge-tree: this exercises the open-source ReplicatedMergeTree path. + +-- B33 (lifted): ReplicatedMergeTree on a cas disk is now SUPPORTED. The earlier +-- SUPPORT_IS_DISABLED gate in StorageReplicatedMergeTree was removed once replication-internal clones +-- stopped corrupting content-addressed parts, so creating a ReplicatedMergeTree table on a +-- cas disk now succeeds and works end-to-end. A plain (non-replicated) MergeTree on the +-- same kind of disk must also still work. + +DROP TABLE IF EXISTS t_cas_repl; +DROP TABLE IF EXISTS t_cas_plain; + +-- (1) A ReplicatedMergeTree table on a cas disk is now supported end-to-end. +CREATE TABLE t_cas_repl (a UInt64, b UInt64) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/t_cas_repl', 'r1') +ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04283', + name = '04283_cas_repl', + path = '04283_cas_repl_pool/'); + +INSERT INTO t_cas_repl VALUES (1, 10), (2, 20); +SELECT 'repl_count', count() FROM t_cas_repl; +SELECT 'repl_sum', sum(b) FROM t_cas_repl; +DROP TABLE t_cas_repl; + +-- (2) A plain (non-replicated) MergeTree on a cas disk still works end-to-end. +CREATE TABLE t_cas_plain (a UInt64, b UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04283', + name = '04283_cas_plain', + path = '04283_cas_plain_pool/'); + +INSERT INTO t_cas_plain SELECT number, number * 2 FROM numbers(100); +SELECT 'plain_count', count() FROM t_cas_plain; +SELECT 'plain_sum', sum(b) FROM t_cas_plain; + +DROP TABLE t_cas_plain; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04284_cas_backup_pointer_holding.reference b/tests/queries/0_stateless/04284_cas_backup_pointer_holding.reference new file mode 100644 index 000000000000..d1ea139d5213 --- /dev/null +++ b/tests/queries/0_stateless/04284_cas_backup_pointer_holding.reference @@ -0,0 +1,5 @@ +before 1000 499500 7 +BACKUP_CREATED +RESTORED +after 1000 499500 7 +dropped_ok diff --git a/tests/queries/0_stateless/04284_cas_backup_pointer_holding.sh b/tests/queries/0_stateless/04284_cas_backup_pointer_holding.sh new file mode 100755 index 000000000000..f855f043c4c6 --- /dev/null +++ b/tests/queries/0_stateless/04284_cas_backup_pointer_holding.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# B34: BACKUP / RESTORE of a cas table in an Atomic (UUID) database — the default in +# the stateless suite. Two behaviors: +# 1. BACKUP uses the pointer-holding path (make_temporary_hard_links=false): it resolves the +# part's objects via getStorageObjects and never calls disk->createHardLink, so it succeeds on +# a cas disk. (The temporary-hard-link BACKUP path, used only by the deprecated +# Ordinary database engine, is fail-closed with a clear SUPPORT_IS_DISABLED message in +# DataPartStorageOnDiskBase::backup — see B34/B16.) +# 2. RESTORE onto a cas disk now succeeds end-to-end: the part files are written back +# through the disk's write path and the restored table reads back identical to the original. +# (This used to fail closed with NOT_IMPLEMENTED until the whole-part write contract, B30, +# landed; restore-onto-CA is no longer an M1 gap.) + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +backup_name="Disk('backups', '${CLICKHOUSE_TEST_UNIQUE_NAME}.zip')" + +${CLICKHOUSE_CLIENT} --multiquery < 0 on a plain MergeTree keeps an on-disk deduplication log +-- (deduplication_logs/deduplication_log_N.txt) at the table root. On a cas disk that log +-- works the same way it does on a plain s3 disk: the disk cannot host append writes, so the log +-- rewrites a fresh rotated log object per record, stored verbatim in the table's files/ namespace. This +-- test uses an INLINE cas disk, so it exercises the CA path on any test config. + +DROP TABLE IF EXISTS t_cas_deduplication; + +CREATE TABLE t_cas_deduplication (a UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS non_replicated_deduplication_window = 100, disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04285', + name = '04285_cas_deduplication', + path = '04285_cas_deduplication_pool/'); + +-- Identical inserts are deduplicated; each insert also writes a record to the on-disk log (the write +-- that used to fail closed with a null writer on a cas disk). +INSERT INTO t_cas_deduplication VALUES (1); +INSERT INTO t_cas_deduplication VALUES (1); +SELECT 'after-two-identical', count() FROM t_cas_deduplication; + +-- A distinct block is accepted. +INSERT INTO t_cas_deduplication VALUES (2); +SELECT 'after-new-block', count() FROM t_cas_deduplication; + +-- Reload the table: the deduplication log is re-read from the cas disk. +DETACH TABLE t_cas_deduplication; +ATTACH TABLE t_cas_deduplication; + +-- The first block is still deduplicated — its record was reloaded from the on-disk log, not memory. +INSERT INTO t_cas_deduplication VALUES (1); +SELECT 'after-reload-same-block', count() FROM t_cas_deduplication; + +-- A new distinct block is still accepted after the reload. +INSERT INTO t_cas_deduplication VALUES (3); +SELECT 'after-reload-new-block', count() FROM t_cas_deduplication; + +DROP TABLE t_cas_deduplication; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04286_cas_remote_data_paths.reference b/tests/queries/0_stateless/04286_cas_remote_data_paths.reference new file mode 100644 index 000000000000..c3fdc13518a0 --- /dev/null +++ b/tests/queries/0_stateless/04286_cas_remote_data_paths.reference @@ -0,0 +1,2 @@ +1 +dropped_ok diff --git a/tests/queries/0_stateless/04286_cas_remote_data_paths.sql b/tests/queries/0_stateless/04286_cas_remote_data_paths.sql new file mode 100644 index 000000000000..58359183a7ef --- /dev/null +++ b/tests/queries/0_stateless/04286_cas_remote_data_paths.sql @@ -0,0 +1,41 @@ +-- Tags: no-fasttest, no-cas-storage +-- no-fasttest: cas is an object-storage metadata type; keep it off the minimal +-- fasttest image. +-- no-cas-storage: the coverage lives in the INLINE content-addressed disk created +-- below, so the test is meaningful on every ordinary lane. On lanes where the DEFAULT MergeTree +-- storage is itself content-addressed, `system.remote_data_paths` (whose `disk_name` filter is +-- applied only after the traversal) also walks the huge shared default pool holding the whole +-- run's data, and on the S3 (RustFS) variant that walk does not fit the 600s test timeout. + +-- B38: querying system.remote_data_paths traverses the disk and probes existsFile on pool sub-dirs +-- (e.g. the "store" directory). Such a path resolves to a directory object key; existsFile must treat +-- a directory as not-a-file and not let the raw filesystem "Is a directory" error escape. So a +-- system.remote_data_paths query over a cas table must succeed (return the parts' +-- remote paths) instead of throwing. + +DROP TABLE IF EXISTS t_cas_rdp; + +CREATE TABLE t_cas_rdp (a UInt64, b UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04286', + name = '04286_cas_rdp', + path = '04286_cas_rdp_pool/'); + +INSERT INTO t_cas_rdp SELECT number, number * 2 FROM numbers(100); + +-- The traversal (with shadow paths) must be QUERYABLE: it used to throw `Is a directory` (Code 1001) +-- when it probed the CA pool sub-dir (e.g. "store") via existsFile. After the B38 fix it returns a +-- result without raising. We assert the query succeeds (count() is a non-negative number) rather than +-- a specific row count: the CA disk's object-storage directory model determines how many rows the +-- traversal yields, which is orthogonal to the not-throwing contract this test pins. +SELECT count() >= 0 +FROM system.remote_data_paths +WHERE disk_name = '04286_cas_rdp' +SETTINGS traverse_shadow_remote_data_paths = 1; + +DROP TABLE t_cas_rdp; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04287_cas_detach_partition_listing.reference b/tests/queries/0_stateless/04287_cas_detach_partition_listing.reference new file mode 100644 index 000000000000..a4f7c658a48c --- /dev/null +++ b/tests/queries/0_stateless/04287_cas_detach_partition_listing.reference @@ -0,0 +1,4 @@ +count_before 100 +count_after 0 +detached all_1_2_1 +dropped_ok diff --git a/tests/queries/0_stateless/04287_cas_detach_partition_listing.sql b/tests/queries/0_stateless/04287_cas_detach_partition_listing.sql new file mode 100644 index 000000000000..fc500ba3d39d --- /dev/null +++ b/tests/queries/0_stateless/04287_cas_detach_partition_listing.sql @@ -0,0 +1,38 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- B36: after DETACH PARTITION on a cas disk, system.detached_parts must list the +-- detached part DIRECTORY name (e.g. all_1_2_1), not a sidecar / mutable file (metadata_version.txt). +-- The detached namespace is a container of detached part directories; the CA disk listing of the +-- "detached" path must yield the part directory names, not the files inside them. + +DROP TABLE IF EXISTS t_cas_detach; + +CREATE TABLE t_cas_detach (a UInt64, b UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04287', + name = '04287_cas_detach', + path = '04287_cas_detach_pool/'); + +INSERT INTO t_cas_detach SELECT number, number * 2 FROM numbers(50); +INSERT INTO t_cas_detach SELECT number, number * 2 FROM numbers(50, 50); +OPTIMIZE TABLE t_cas_detach FINAL; + +SELECT 'count_before', count() FROM t_cas_detach; + +ALTER TABLE t_cas_detach DETACH PARTITION tuple(); + +SELECT 'count_after', count() FROM t_cas_detach; + +-- The detached parts listing must show the part directory name, not metadata_version.txt. +SELECT 'detached', name +FROM system.detached_parts +WHERE database = currentDatabase() AND table = 't_cas_detach' +ORDER BY name; + +DROP TABLE t_cas_detach; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04288_cas_detached_part_modification_time.reference b/tests/queries/0_stateless/04288_cas_detached_part_modification_time.reference new file mode 100644 index 000000000000..180d4e299167 --- /dev/null +++ b/tests/queries/0_stateless/04288_cas_detached_part_modification_time.reference @@ -0,0 +1,4 @@ +count_before 50 +count_after 0 +detached_mtime_readable all_1_1_0 1 +dropped_ok diff --git a/tests/queries/0_stateless/04288_cas_detached_part_modification_time.sql b/tests/queries/0_stateless/04288_cas_detached_part_modification_time.sql new file mode 100644 index 000000000000..6ec35913f58f --- /dev/null +++ b/tests/queries/0_stateless/04288_cas_detached_part_modification_time.sql @@ -0,0 +1,40 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- B44: reading system.detached_parts.modification_time of a part detached on a cas +-- disk must NOT throw. The detached part is stored under a ref named "detached" whose manifest keys +-- are shaped /; system.detached_parts reads the modification time by calling +-- IDisk::getLastModified on the detached part DIRECTORY (/detached/). Before the +-- fix, parsePartFilePath reported part_name="detached" + a non-empty file equal to the detached part +-- directory name, so getLastModified fell through to the part-file manifest lookup and threw +-- "ContentAddressed: file not in manifest". getLastModified now recognises the detached +-- part directory and reports the "detached" ref manifest object's mtime. + +DROP TABLE IF EXISTS t_cas_detach_mtime; + +CREATE TABLE t_cas_detach_mtime (a UInt64, b UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04288', + name = '04288_cas_detach_mtime', + path = '04288_cas_detach_mtime_pool/'); + +INSERT INTO t_cas_detach_mtime SELECT number, number * 2 FROM numbers(50); + +SELECT 'count_before', count() FROM t_cas_detach_mtime; + +ALTER TABLE t_cas_detach_mtime DETACH PARTITION tuple(); + +SELECT 'count_after', count() FROM t_cas_detach_mtime; + +-- The modification_time read must succeed (be non-NULL) instead of throwing FILE_DOESNT_EXIST. +SELECT 'detached_mtime_readable', name, modification_time IS NOT NULL +FROM system.detached_parts +WHERE database = currentDatabase() AND table = 't_cas_detach_mtime' +ORDER BY name; + +DROP TABLE t_cas_detach_mtime; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04289_cas_multi_detach_drop.reference b/tests/queries/0_stateless/04289_cas_multi_detach_drop.reference new file mode 100644 index 000000000000..f87bae700d97 --- /dev/null +++ b/tests/queries/0_stateless/04289_cas_multi_detach_drop.reference @@ -0,0 +1,7 @@ +active_parts 3 +detached_after 3 +detached_names 1_1_1_0 +detached_names 2_2_2_0 +detached_names 3_3_3_0 +detached_after_drop 0 +dropped_ok diff --git a/tests/queries/0_stateless/04289_cas_multi_detach_drop.sql b/tests/queries/0_stateless/04289_cas_multi_detach_drop.sql new file mode 100644 index 000000000000..597a0c52c3d6 --- /dev/null +++ b/tests/queries/0_stateless/04289_cas_multi_detach_drop.sql @@ -0,0 +1,43 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- B46/B47: multiple partitions detached on a cas disk must COEXIST under the one +-- shared "detached" ref, and DROP DETACHED PARTITION ALL must remove them. +-- B46: each DETACH PARTITION clones one part into detached// via a fresh CA commit; +-- the commit used to REWRITE the shared "detached" ref, so each detach overwrote the previous +-- one and only the last detached part was listed. commit now MERGES into the existing detached +-- ref's manifest + sidecar, so all detached parts coexist. +-- B47: DROP DETACHED PARTITION first renames the detached part to "deleting_" +-- (PartsTemporaryRename) then removes it; CA moveDirectory ignored a detached->detached rename +-- (the rename was a no-op, so removeRecursive on the renamed dir found nothing). moveDirectory +-- now re-keys the detached part dir within the shared detached ref, and removeRecursive handles a +-- detached part directory by removing only that part's keys from the shared ref. + +DROP TABLE IF EXISTS t_cas_multi_detach; + +CREATE TABLE t_cas_multi_detach (p UInt64, v UInt64) +ENGINE = MergeTree PARTITION BY p ORDER BY v +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04289', + name = '04289_cas_multi_detach', + path = '04289_cas_multi_detach_pool/'); + +INSERT INTO t_cas_multi_detach VALUES (1, 1), (2, 2), (3, 3); +SELECT 'active_parts', count() FROM system.parts WHERE database = currentDatabase() AND table = 't_cas_multi_detach' AND active; + +ALTER TABLE t_cas_multi_detach DETACH PARTITION ALL; + +-- All three partitions must be listed as detached parts (not just the last one detached). +SELECT 'detached_after', count() FROM system.detached_parts WHERE database = currentDatabase() AND table = 't_cas_multi_detach'; +SELECT 'detached_names', name FROM system.detached_parts WHERE database = currentDatabase() AND table = 't_cas_multi_detach' ORDER BY name; + +ALTER TABLE t_cas_multi_detach DROP DETACHED PARTITION ALL SETTINGS allow_drop_detached = 1; + +-- DROP DETACHED PARTITION ALL must remove every detached part. +SELECT 'detached_after_drop', count() FROM system.detached_parts WHERE database = currentDatabase() AND table = 't_cas_multi_detach'; + +DROP TABLE t_cas_multi_detach; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04290_cas_no_leftovers.reference b/tests/queries/0_stateless/04290_cas_no_leftovers.reference new file mode 100644 index 000000000000..ea1449c10eba --- /dev/null +++ b/tests/queries/0_stateless/04290_cas_no_leftovers.reference @@ -0,0 +1,5 @@ +rows 600000 +grew_above_baseline 1 +fsck_unreachable 0 +fsck_dangling 0 +pool_meta_present 1 diff --git a/tests/queries/0_stateless/04290_cas_no_leftovers.sh b/tests/queries/0_stateless/04290_cas_no_leftovers.sh new file mode 100755 index 000000000000..9b05005d7bb9 --- /dev/null +++ b/tests/queries/0_stateless/04290_cas_no_leftovers.sh @@ -0,0 +1,132 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-parallel +# ^ cas is an object-storage metadata type (keep it off the minimal fasttest image); +# no-parallel because we inspect a known on-disk pool directory from the shell and must not race +# another test sharing the same path. + +# North-star "no S3 leftovers" oracle for the content-addressed pool, exercised over a `local` +# object_storage backend so the pool is a plain directory the test shell can inspect directly. +# +# We put the pool under CLICKHOUSE_USER_FILES_UNIQUE (an absolute path both the server and this +# shell can see on a local run) and enable the background reachability GC aggressively +# (gc_enabled=1, grace=2s, interval=1s). We then: +# (1) record the baseline blobs+parts object count (~0), +# (2) CREATE a MergeTree on the CA disk and INSERT several distinct batches to make many blobs, +# (3) assert the count rose above baseline, +# (4) DROP TABLE ... SYNC so the refs are unlinked and the blobs/footers become GC fodder, +# (5) drain the retire pipeline deterministically via `SYSTEM CAS GC RUN` (bounded +# loop on the `pending_*` gauges, NOT a fixed sleep), then run `FSCK` directly on the running +# disk (T13): a clean reachability audit reading back zero `unreachable`/`dangling` is a +# strictly stronger no-leftovers oracle than polling the pool directory ever was. +# `_pool_meta` (durable single-owner marker) and the `store/` metadata tree are expected to remain. +# Teardown is fail-closed (spec rev.8 §5/§9): `SYSTEM CAS FORGET` the disk (force-Vanish, +# node-local), verify it reads `vanished(forgotten)` in system.cas_mounts, and only then +# `rm -rf` — FORGET stopped and joined every CAS background thread for this disk. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +POOL_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}_04290_${RANDOM}" + +# Fresh pool dir for this run. +rm -rf "${POOL_DIR:?}" +mkdir -p "${POOL_DIR}" + +# Count regular files (objects) currently living under blobs/ and parts/ in the pool. +count_pool_objects() { + local n_blobs n_parts + n_blobs=$(find "${POOL_DIR}/ca/blobs" "${POOL_DIR}/ca/packs" -type f 2>/dev/null | wc -l) + n_parts=$(find "${POOL_DIR}/ca/trees" -type f 2>/dev/null | wc -l) + echo $(( n_blobs + n_parts )) +} + +DISK_NAME="ca_04290_${CLICKHOUSE_TEST_UNIQUE_NAME}_${RANDOM}" +DISK_DEF="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04290', + name = '${DISK_NAME}', + path = '${POOL_DIR}/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1)" + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_cas_leftovers SYNC" + +# (1) Baseline. +BASELINE=$(count_pool_objects) + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_cas_leftovers (a UInt64, s String, d Date) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = ${DISK_DEF}" + +# (2) Several distinct inserts -> several distinct parts/blobs (distinct data => no dedup-away). +for i in 0 1 2 3 4 5; do + $CLICKHOUSE_CLIENT --query " + INSERT INTO t_cas_leftovers + SELECT number + ${i} * 100000, toString(number + ${i} * 100000), toDate('2020-01-01') + (number % 1000) + FROM numbers(100000)" +done + +$CLICKHOUSE_CLIENT --query "SELECT 'rows', count() FROM t_cas_leftovers" + +# (3) Pool must have grown above baseline. +AFTER_INSERT=$(count_pool_objects) +if [ "$AFTER_INSERT" -gt "$BASELINE" ]; then + echo "grew_above_baseline 1" +else + echo "grew_above_baseline 0 (baseline=${BASELINE} after_insert=${AFTER_INSERT})" +fi + +# (4) Drop: refs unlinked synchronously, blobs/footers become unreferenced GC fodder. +$CLICKHOUSE_CLIENT --query "DROP TABLE t_cas_leftovers SYNC" + +# (5) Drain GC deterministically: loop `SYSTEM CAS GC RUN` rounds until the retire +# pipeline's `pending_*` gauges (Task 7) read back to empty. Bounded (~60 rounds, half-second +# spacing), not a fixed sleep; column values are looked up BY HEADER NAME (not position) so the +# loop keeps working if the result set gains columns. +PENDING=1 +for _ in $(seq 1 60); do + PENDING=$($CLICKHOUSE_CLIENT --query "SYSTEM CAS GC RUN '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print $col["pending_condemned"] } # already candidates+retired per its doc in Gc/CasGc.h; summing all three double-counts') + [ "${PENDING}" = "0" ] && break + sleep 0.5 +done + +if [ "${PENDING}" != "0" ]; then + echo "FAIL: GC did not drain the retire pipeline within the bounded loop (pending=${PENDING})" >&2 + exit 1 +fi + +# (6) FSCK runs directly on the running disk (T13): a reachability audit that must read back zero +# unreachable/dangling objects. This is a strictly stronger no-leftovers oracle than the old +# dir-poll. +$CLICKHOUSE_CLIENT --query "SYSTEM CAS FSCK '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print "fsck_unreachable", $col["unreachable"]; print "fsck_dangling", $col["dangling"] }' + +# _pool_meta must still be present (durable single-owner marker is never GC'd). +if [ -f "${POOL_DIR}/ca/_pool_meta" ]; then + echo "pool_meta_present 1" +else + echo "pool_meta_present 0" +fi + +# (7) Fail-closed teardown (spec rev.8 §5/§9): FORGET the disk (force-Vanish, node-local; the table is +# already dropped above), verify it reads exactly `vanished(forgotten)` in the mounts table, and +# only then rm. A failed FORGET or an unexpected lifecycle aborts with the pool dir left in place. +# FORGET logs an operator WARNING; the harness runs the client at --send_logs_level=warning, so that +# expected warning would stream to stderr and be flagged as a failure -- suppress it for this call. +$CLICKHOUSE_CLIENT --allow_repeated_settings --send_logs_level=fatal \ + --query "SYSTEM CAS FORGET '${DISK_NAME}'" || { + echo "FORGET failed — leaving pool dir in place (fail-closed)"; exit 1; } +LIFECYCLE=$($CLICKHOUSE_CLIENT --query " + SELECT lifecycle || '(' || lifecycle_reason || ')' FROM system.cas_mounts + WHERE disk = '${DISK_NAME}'") +[ "${LIFECYCLE}" = "vanished(forgotten)" ] || { + echo "unexpected lifecycle after FORGET: ${LIFECYCLE}"; exit 1; } + +rm -rf "${POOL_DIR:?}" # safe: FORGET stopped and joined every CAS thread for this disk diff --git a/tests/queries/0_stateless/04292_cas_mutations.reference b/tests/queries/0_stateless/04292_cas_mutations.reference new file mode 100644 index 000000000000..98e15e398c73 --- /dev/null +++ b/tests/queries/0_stateless/04292_cas_mutations.reference @@ -0,0 +1,7 @@ +after_update_v: match +after_delete: match +after_update_s: match +after_update_v_doubled: match +after_multi_update: match +final_rows_match 1 +final_data_match 1 diff --git a/tests/queries/0_stateless/04292_cas_mutations.sh b/tests/queries/0_stateless/04292_cas_mutations.sh new file mode 100755 index 000000000000..bfbb0bb7d981 --- /dev/null +++ b/tests/queries/0_stateless/04292_cas_mutations.sh @@ -0,0 +1,114 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# Correctness oracle for mutations on content-addressed disks (CAS M7). +# After supportsHardLinks() was flipped to true, mutations are enabled on +# content-addressed disks. A mutation builds the new part through a +# whole-part transaction: unchanged columns are carried forward by reference +# (same blob) and changed columns are written fresh. +# +# Strategy: both tables receive identical data and identical mutations; +# after each mutation we assert the full ordered contents are equal +# (CA vs plain MergeTree). Every assertion is a self-checking CA-vs-plain +# equality so the reference file is trivially correct (no hand-computed +# arithmetic needed). + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK_CA="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04292', + name = '04292_cas_mut', + path = '04292_cas_mut_pool/')" + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_plain SYNC" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_ca (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id +SETTINGS disk = ${DISK_CA}" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_plain (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id" + +# Seed both tables with identical deterministic data. +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_ca SELECT number, number * 10, toString(number) FROM numbers(100)" +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_plain SELECT number, number * 10, toString(number) FROM numbers(100)" + +# Helper: compare full ordered contents. +CMP_QUERY="SELECT if( + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id)), + 'match', 'DIFF')" + +# --- Mutation 1: UPDATE one column (id/s carry forward by reference on CA) --- +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_ca UPDATE v = v + 1 WHERE id % 3 = 0 SETTINGS mutations_sync = 2" +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_plain UPDATE v = v + 1 WHERE id % 3 = 0 SETTINGS mutations_sync = 2" + +echo -n 'after_update_v: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Mutation 2: DELETE --- +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_ca DELETE WHERE id % 7 = 0 SETTINGS mutations_sync = 2" +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_plain DELETE WHERE id % 7 = 0 SETTINGS mutations_sync = 2" + +echo -n 'after_delete: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Mutation 3: UPDATE string column for a range of rows --- +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_ca UPDATE s = concat(s, '_x') WHERE id > 50 SETTINGS mutations_sync = 2" +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_plain UPDATE s = concat(s, '_x') WHERE id > 50 SETTINGS mutations_sync = 2" + +echo -n 'after_update_s: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# NOTE: a column-type change (`MODIFY COLUMN v Int64`) is deliberately NOT exercised here. It is a +# data-`ALTER` that runs `checkAlterIsPossible`, which on a table created with an inline +# `disk = disk(...)` setting trips a PRE-EXISTING, engine-agnostic bug: the `disk` value is stored as a +# `CustomType` in `settings_changes` and several ALTER sub-checks read it as a `String` (`BAD_GET`). +# That is orthogonal to content-addressing (it reproduces on any inline-disk table). `MODIFY COLUMN` on +# a content-addressed disk is covered through the storage-policy path by the CA-default suite run. + +# --- Mutation 4: UPDATE the numeric column again (compounding the carry-forward) --- +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_ca UPDATE v = v * 2 WHERE id % 2 = 0 SETTINGS mutations_sync = 2" +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_plain UPDATE v = v * 2 WHERE id % 2 = 0 SETTINGS mutations_sync = 2" + +echo -n 'after_update_v_doubled: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Mutation 5: multi-column UPDATE in one mutation (both data columns rewritten together) --- +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_ca UPDATE v = v + id, s = concat('p_', s) WHERE id < 40 SETTINGS mutations_sync = 2" +$CLICKHOUSE_CLIENT --query " +ALTER TABLE t_plain UPDATE v = v + id, s = concat('p_', s) WHERE id < 40 SETTINGS mutations_sync = 2" + +echo -n 'after_multi_update: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Final sanity: row count and data equality --- +$CLICKHOUSE_CLIENT --query " +SELECT 'final_rows_match', count() = (SELECT count() FROM t_plain) FROM t_ca" +$CLICKHOUSE_CLIENT --query " +SELECT 'final_data_match', + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id))" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE t_plain SYNC" diff --git a/tests/queries/0_stateless/04293_cas_lightweight_delete.reference b/tests/queries/0_stateless/04293_cas_lightweight_delete.reference new file mode 100644 index 000000000000..18c1efab4a4a --- /dev/null +++ b/tests/queries/0_stateless/04293_cas_lightweight_delete.reference @@ -0,0 +1,6 @@ +after_delete_mod5: match +after_delete_like: match +after_delete_range: match +after_optimize: match +final_rows_match 1 +final_data_match 1 diff --git a/tests/queries/0_stateless/04293_cas_lightweight_delete.sh b/tests/queries/0_stateless/04293_cas_lightweight_delete.sh new file mode 100755 index 000000000000..ba3b5e049670 --- /dev/null +++ b/tests/queries/0_stateless/04293_cas_lightweight_delete.sh @@ -0,0 +1,101 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# Correctness oracle for lightweight DELETE on content-addressed disks (CAS M7). +# After the supportsHardLinks() gate was lifted, lightweight DELETE is enabled on +# content-addressed disks. Unlike heavy mutations, lightweight DELETE uses row- +# existence bitmaps stored alongside each part and is applied physically during +# the next OPTIMIZE/merge. +# +# Strategy: both tables receive identical data and identical lightweight DELETEs; +# after each DELETE (and after OPTIMIZE FINAL) we assert the full ordered contents +# are equal (CA vs plain MergeTree). Every assertion is a self-checking CA-vs-plain +# equality so the reference file is trivially correct (no hand-computed arithmetic +# needed). + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK_CA="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04293', + name = '04293_cas_lwd', + path = '04293_cas_lwd_pool/')" + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_plain SYNC" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_ca (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id +SETTINGS disk = ${DISK_CA}" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_plain (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id" + +# Seed both tables with identical deterministic data spread across two parts +# (lightweight DELETEs across multiple parts are more meaningful than single-part). +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_ca SELECT number, number * 10, toString(number) FROM numbers(100)" +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_ca SELECT number, number * 10, toString(number) FROM numbers(100, 100)" +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_plain SELECT number, number * 10, toString(number) FROM numbers(100)" +$CLICKHOUSE_CLIENT --query " +INSERT INTO t_plain SELECT number, number * 10, toString(number) FROM numbers(100, 100)" + +# Helper: compare full ordered contents of both tables. +CMP_QUERY="SELECT if( + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id)), + 'match', 'DIFF')" + +# --- Lightweight DELETE 1: every 5th row --- +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_ca WHERE id % 5 = 0 SETTINGS lightweight_deletes_sync = 2" +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_plain WHERE id % 5 = 0 SETTINGS lightweight_deletes_sync = 2" + +echo -n 'after_delete_mod5: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Lightweight DELETE 2: rows whose string starts with '1' (overlaps first delete) --- +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_ca WHERE s LIKE '1%' SETTINGS lightweight_deletes_sync = 2" +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_plain WHERE s LIKE '1%' SETTINGS lightweight_deletes_sync = 2" + +echo -n 'after_delete_like: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Lightweight DELETE 3: rows with v > 1500 --- +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_ca WHERE v > 1500 SETTINGS lightweight_deletes_sync = 2" +$CLICKHOUSE_CLIENT --query " +DELETE FROM t_plain WHERE v > 1500 SETTINGS lightweight_deletes_sync = 2" + +echo -n 'after_delete_range: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- OPTIMIZE FINAL: force merge so lightweight deletes are physically applied --- +$CLICKHOUSE_CLIENT --query "OPTIMIZE TABLE t_ca FINAL" +$CLICKHOUSE_CLIENT --query "OPTIMIZE TABLE t_plain FINAL" + +echo -n 'after_optimize: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Final sanity: row count and data equality --- +$CLICKHOUSE_CLIENT --query " +SELECT 'final_rows_match', count() = (SELECT count() FROM t_plain) FROM t_ca" +$CLICKHOUSE_CLIENT --query " +SELECT 'final_data_match', + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id))" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE t_plain SYNC" diff --git a/tests/queries/0_stateless/04294_cas_patch_parts.reference b/tests/queries/0_stateless/04294_cas_patch_parts.reference new file mode 100644 index 000000000000..0f54f244519c --- /dev/null +++ b/tests/queries/0_stateless/04294_cas_patch_parts.reference @@ -0,0 +1,6 @@ +after_patch_delete_1: match +after_patch_delete_2: match +ca_has_patch_part: 1 +after_optimize: match +final_rows_match 1 +final_data_match 1 diff --git a/tests/queries/0_stateless/04294_cas_patch_parts.sh b/tests/queries/0_stateless/04294_cas_patch_parts.sh new file mode 100755 index 000000000000..deee3f7b9946 --- /dev/null +++ b/tests/queries/0_stateless/04294_cas_patch_parts.sh @@ -0,0 +1,88 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# Correctness oracle for PATCH PARTS (the native lightweight-update model, B5) on content-addressed +# disks (CAS M7). The default lightweight DELETE mode is `alter_update` (a heavy mutation); this test +# forces the lightweight-update path with `lightweight_delete_mode = 'lightweight_update_force'`, which +# produces a PATCH PART (an `UPDATE _row_exists = 0`). `_force` THROWS if the table cannot do a +# lightweight update, so a successful run is itself proof the patch-part path was exercised — on a +# content-addressed disk the patch part is written through the same whole-part transaction as any part. +# +# Lightweight updates require materialized `_block_number` / `_block_offset` columns +# (enable_block_number_column / enable_block_offset_column) and a non-UNIQUE-KEY custom-partitioned +# table. Both tables get identical settings, data, and operations; every assertion is a self-checking +# CA-vs-plain equality so the reference file is trivially correct. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK_CA="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04294', + name = '04294_cas_patch', + path = '04294_cas_patch_pool/')" + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_plain SYNC" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_ca (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id +SETTINGS disk = ${DISK_CA}, enable_block_number_column = 1, enable_block_offset_column = 1" + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_plain (id UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY id +SETTINGS enable_block_number_column = 1, enable_block_offset_column = 1" + +# Seed both identically across two parts. +$CLICKHOUSE_CLIENT --query "INSERT INTO t_ca SELECT number, number * 10, toString(number) FROM numbers(100)" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_ca SELECT number, number * 10, toString(number) FROM numbers(100, 100)" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_plain SELECT number, number * 10, toString(number) FROM numbers(100)" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_plain SELECT number, number * 10, toString(number) FROM numbers(100, 100)" + +CMP_QUERY="SELECT if( + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id)), + 'match', 'DIFF')" + +# Force the patch-part (lightweight-update) path. `_force` throws if unsupported, so success == patch path. +LWU_SETTINGS="SETTINGS enable_lightweight_update = 1, lightweight_delete_mode = 'lightweight_update_force', lightweight_deletes_sync = 2" + +# --- Patch-part DELETE 1 --- +$CLICKHOUSE_CLIENT --query "DELETE FROM t_ca WHERE id % 5 = 0 ${LWU_SETTINGS}" +$CLICKHOUSE_CLIENT --query "DELETE FROM t_plain WHERE id % 5 = 0 ${LWU_SETTINGS}" +echo -n 'after_patch_delete_1: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Patch-part DELETE 2 (overlaps the first) --- +$CLICKHOUSE_CLIENT --query "DELETE FROM t_ca WHERE v > 1500 ${LWU_SETTINGS}" +$CLICKHOUSE_CLIENT --query "DELETE FROM t_plain WHERE v > 1500 ${LWU_SETTINGS}" +echo -n 'after_patch_delete_2: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# Prove a patch part really exists on the content-addressed table before it is merged away. +echo -n 'ca_has_patch_part: ' +$CLICKHOUSE_CLIENT --query " +SELECT count() > 0 FROM system.parts +WHERE database = currentDatabase() AND table = 't_ca' AND active AND startsWith(name, 'patch')" + +# --- OPTIMIZE FINAL applies the patch parts during merge --- +$CLICKHOUSE_CLIENT --query "OPTIMIZE TABLE t_ca FINAL" +$CLICKHOUSE_CLIENT --query "OPTIMIZE TABLE t_plain FINAL" +echo -n 'after_optimize: ' +$CLICKHOUSE_CLIENT --query "$CMP_QUERY" + +# --- Final equality --- +$CLICKHOUSE_CLIENT --query "SELECT 'final_rows_match', count() = (SELECT count() FROM t_plain) FROM t_ca" +$CLICKHOUSE_CLIENT --query " +SELECT 'final_data_match', + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_ca ORDER BY id)) = + (SELECT groupArray((id, v, s)) FROM (SELECT id, v, s FROM t_plain ORDER BY id))" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_ca SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE t_plain SYNC" diff --git a/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.reference b/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.reference new file mode 100644 index 000000000000..52b50e1048c9 --- /dev/null +++ b/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.reference @@ -0,0 +1,5 @@ +grew_above_baseline 1 +rows_after_mutations_correct 1 +fsck_unreachable 0 +fsck_dangling 0 +pool_meta_present 1 diff --git a/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.sh b/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.sh new file mode 100755 index 000000000000..ee8655a88f86 --- /dev/null +++ b/tests/queries/0_stateless/04295_cas_mutation_no_leftovers.sh @@ -0,0 +1,131 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-parallel +# ^ cas is an object-storage metadata type (keep it off the minimal fasttest image); +# no-parallel because we inspect a known on-disk pool directory from the shell and must not race +# another test sharing the same path. + +# No-leftovers oracle for MUTATIONS + lightweight DELETE (patch parts) on the content-addressed pool +# (CAS M7), exercised over a `local` object_storage backend so the pool is a plain directory the test +# shell can inspect directly. Mirrors 04290 but adds heavy mutations and a patch-part lightweight +# DELETE before the drop: a mutation supersedes the source part (its uniquely-owned blobs become +# unreachable) and writes a new part; carried-forward columns stay referenced. We assert that after +# DROP, draining the retire pipeline via `SYSTEM CAS GC RUN` then running `FSCK` on the +# running disk (T13) reads back zero `unreachable`/`dangling` objects (no mutated-away or patch-part +# blobs left behind), and that `_pool_meta` survives. Teardown is fail-closed (spec rev.8 §5/§9): FORGET +# the disk, verify `vanished(forgotten)`, then rm. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +POOL_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}_04295_${RANDOM}" + +rm -rf "${POOL_DIR:?}" +mkdir -p "${POOL_DIR}" + +count_pool_objects() { + local n_blobs n_parts + n_blobs=$(find "${POOL_DIR}/ca/blobs" "${POOL_DIR}/ca/packs" -type f 2>/dev/null | wc -l) + n_parts=$(find "${POOL_DIR}/ca/trees" -type f 2>/dev/null | wc -l) + echo $(( n_blobs + n_parts )) +} + +DISK_NAME="ca_04295_${CLICKHOUSE_TEST_UNIQUE_NAME}_${RANDOM}" +DISK_DEF="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04295', + name = '${DISK_NAME}', + path = '${POOL_DIR}/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1)" + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_cas_mut_leftovers SYNC" + +BASELINE=$(count_pool_objects) + +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_cas_mut_leftovers (a UInt64, v UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = ${DISK_DEF}, enable_block_number_column = 1, enable_block_offset_column = 1" + +# Several distinct inserts -> several distinct parts/blobs. +for i in 0 1 2 3; do + $CLICKHOUSE_CLIENT --query " + INSERT INTO t_cas_mut_leftovers + SELECT number + ${i} * 100000, (number + ${i} * 100000) * 10, toString(number + ${i} * 100000) + FROM numbers(100000)" +done + +AFTER_INSERT=$(count_pool_objects) +if [ "$AFTER_INSERT" -gt "$BASELINE" ]; then + echo "grew_above_baseline 1" +else + echo "grew_above_baseline 0 (baseline=${BASELINE} after_insert=${AFTER_INSERT})" +fi + +# Heavy mutation (rewrites the v column; a/s carry forward by reference -> shared blobs). +$CLICKHOUSE_CLIENT --query "ALTER TABLE t_cas_mut_leftovers UPDATE v = v + 1 WHERE a % 2 = 0 SETTINGS mutations_sync = 2" +# Heavy mutation: delete part of the data. +$CLICKHOUSE_CLIENT --query "ALTER TABLE t_cas_mut_leftovers DELETE WHERE a % 5 = 0 SETTINGS mutations_sync = 2" +# Patch part: forced lightweight-update DELETE (throws if unsupported, so success == patch path). +$CLICKHOUSE_CLIENT --query " + DELETE FROM t_cas_mut_leftovers WHERE a % 7 = 0 + SETTINGS enable_lightweight_update = 1, lightweight_delete_mode = 'lightweight_update_force', lightweight_deletes_sync = 2" + +# Self-checking row count: a ranges over [0, 400000); the two deletes drop a%5=0 and a%7=0 +# (the UPDATE does not change the row count), so the survivors are exactly a%5!=0 AND a%7!=0. +$CLICKHOUSE_CLIENT --query " +SELECT 'rows_after_mutations_correct', + count() = (SELECT count() FROM numbers(400000) WHERE number % 5 != 0 AND number % 7 != 0) +FROM t_cas_mut_leftovers" + +# Drop: every ref (original, mutated, and patch parts) is unlinked; all blobs/footers become GC fodder. +$CLICKHOUSE_CLIENT --query "DROP TABLE t_cas_mut_leftovers SYNC" + +# Drain GC deterministically: loop `SYSTEM CAS GC RUN` rounds until the retire +# pipeline's `pending_*` gauges (Task 7) read back to empty. Bounded (~60 rounds, half-second +# spacing), not a fixed sleep; column values are looked up BY HEADER NAME (not position) so the +# loop keeps working if the result set gains columns. +PENDING=1 +for _ in $(seq 1 60); do + PENDING=$($CLICKHOUSE_CLIENT --query "SYSTEM CAS GC RUN '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print $col["pending_condemned"] } # already candidates+retired per its doc in Gc/CasGc.h; summing all three double-counts') + [ "${PENDING}" = "0" ] && break + sleep 0.5 +done + +if [ "${PENDING}" != "0" ]; then + echo "FAIL: GC did not drain the retire pipeline within the bounded loop (pending=${PENDING})" >&2 + exit 1 +fi + +# FSCK runs directly on the running disk (T13): a reachability audit that must read back zero +# unreachable/dangling objects. This is a strictly stronger no-leftovers oracle than the old dir-poll. +$CLICKHOUSE_CLIENT --query "SYSTEM CAS FSCK '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print "fsck_unreachable", $col["unreachable"]; print "fsck_dangling", $col["dangling"] }' + +if [ -f "${POOL_DIR}/ca/_pool_meta" ]; then + echo "pool_meta_present 1" +else + echo "pool_meta_present 0" +fi + +# Fail-closed teardown (spec rev.8 §5/§9): FORGET the disk (force-Vanish, node-local; the table is +# already dropped above), verify it reads exactly `vanished(forgotten)` in the mounts table, and only +# then rm. A failed FORGET or an unexpected lifecycle aborts with the pool dir left in place. FORGET logs +# an operator WARNING; the harness runs the client at --send_logs_level=warning, so that expected warning +# would stream to stderr and be flagged as a failure -- suppress it for this call. +$CLICKHOUSE_CLIENT --allow_repeated_settings --send_logs_level=fatal \ + --query "SYSTEM CAS FORGET '${DISK_NAME}'" || { + echo "FORGET failed — leaving pool dir in place (fail-closed)"; exit 1; } +LIFECYCLE=$($CLICKHOUSE_CLIENT --query " + SELECT lifecycle || '(' || lifecycle_reason || ')' FROM system.cas_mounts + WHERE disk = '${DISK_NAME}'") +[ "${LIFECYCLE}" = "vanished(forgotten)" ] || { + echo "unexpected lifecycle after FORGET: ${LIFECYCLE}"; exit 1; } + +rm -rf "${POOL_DIR:?}" # safe: FORGET stopped and joined every CAS thread for this disk diff --git a/tests/queries/0_stateless/04299_cas_projection_inline_disk.reference b/tests/queries/0_stateless/04299_cas_projection_inline_disk.reference new file mode 100644 index 000000000000..f45aa2ce0497 --- /dev/null +++ b/tests/queries/0_stateless/04299_cas_projection_inline_disk.reference @@ -0,0 +1,60 @@ +count 2000 +sum_b 9000 +by_b 0 200 +by_b 1 200 +by_b 2 200 +by_b 3 200 +by_b 4 200 +by_b 5 200 +by_b 6 200 +by_b 7 200 +by_b 8 200 +by_b 9 200 +after_merge_count 2000 +after_merge_by_b 0 200 +after_merge_by_b 1 200 +after_merge_by_b 2 200 +after_merge_by_b 3 200 +after_merge_by_b 4 200 +after_merge_by_b 5 200 +after_merge_by_b 6 200 +after_merge_by_b 7 200 +after_merge_by_b 8 200 +after_merge_by_b 9 200 +has_projection 1 +uses_projection 1 +after_merge_reload_projection 1 +after_merge_reload_uses_projection 1 +after_add_projection_count 2000 +projections_after_add p_by_b 1 +projections_after_add p_sum 1 +uses_p_sum 1 +projections_after_materialize_reload p_by_b 1 +projections_after_materialize_reload p_sum 1 +after_materialize_reload_uses_p_sum 1 +projections_after_update_reload p_by_b 1 +projections_after_update_reload p_sum 1 +after_update_reload_uses_p_sum 1 +after_drop_projection_count 2000 +projections_after_drop p_sum 1 +after_reload_by_b 0 200 +after_reload_by_b 1 200 +after_reload_by_b 2 200 +after_reload_by_b 3 200 +after_reload_by_b 4 200 +after_reload_by_b 5 200 +after_reload_by_b 6 200 +after_reload_by_b 7 200 +after_reload_by_b 8 200 +after_reload_by_b 9 200 +after_reload_sum_b 0 199000 +after_reload_sum_b 1 199200 +after_reload_sum_b 2 199400 +after_reload_sum_b 3 199600 +after_reload_sum_b 4 199800 +after_reload_sum_b 5 200000 +after_reload_sum_b 6 200200 +after_reload_sum_b 7 200400 +after_reload_sum_b 8 200600 +after_reload_sum_b 9 200800 +dropped_ok diff --git a/tests/queries/0_stateless/04299_cas_projection_inline_disk.sql b/tests/queries/0_stateless/04299_cas_projection_inline_disk.sql new file mode 100644 index 000000000000..ac3d2e8fe296 --- /dev/null +++ b/tests/queries/0_stateless/04299_cas_projection_inline_disk.sql @@ -0,0 +1,125 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- Projections on a cas disk: the projection's files are stored as nested keys +-- (.proj/) in the parent part's manifest. Verify INSERT writes a projection, a +-- projection-optimized SELECT returns correct results, and a merge (OPTIMIZE FINAL) rebuilds it. + +DROP TABLE IF EXISTS t_proj_cas; + +CREATE TABLE t_proj_cas (a UInt64, b UInt64, PROJECTION p_by_b (SELECT a, b ORDER BY b)) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '04299', + name = '04299_cas_projection', + path = '04299_cas_projection_pool/'); + +INSERT INTO t_proj_cas SELECT number, number % 10 FROM numbers(1000); +INSERT INTO t_proj_cas SELECT number, number % 10 FROM numbers(1000, 1000); + +SELECT 'count', count() FROM t_proj_cas; +SELECT 'sum_b', sum(b) FROM t_proj_cas; +SELECT 'by_b', b, count() FROM t_proj_cas GROUP BY b ORDER BY b; + +OPTIMIZE TABLE t_proj_cas FINAL; +SELECT 'after_merge_count', count() FROM t_proj_cas; +SELECT 'after_merge_by_b', b, count() FROM t_proj_cas GROUP BY b ORDER BY b; + +SELECT 'has_projection', countDistinct(name) FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas' AND active; + +-- Prove the projection is actually selected by the optimizer (not a silent base-table fallback). +SET optimize_use_projections = 1, force_optimize_projection = 1; +SELECT 'uses_projection', countIf(explain LIKE '%p_by_b%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, count() FROM t_proj_cas GROUP BY b); + +DROP TABLE t_proj_cas; + +-- ALTER ADD/DROP/MATERIALIZE PROJECTION + DETACH/ATTACH durability on the cas disk. We use +-- the server's default cas storage policy here rather than an inline `disk = disk(...)` +-- definition: an ALTER runs `checkColumnFilenamesForCollision`, which re-applies the table's raw +-- `settings_changes` AST through the generic settings path, and the inline `disk(...)` function value +-- is a CustomType that cannot be assigned to the String `disk` setting there (BAD_GET). That is a +-- pre-existing, metadata-type-independent inline-disk-vs-ALTER issue, unrelated to content addressing; +-- the projection ALTER mechanics on the CA disk are identical with the default-disk table. On the +-- cas-default test job this plain table lands on a CA disk; on the normal job it lands on +-- the local disk. The expected values below are the same on both (the oracle) — that equivalence is the +-- whole point of B58: a merge/mutate-rebuilt projection must survive a reload on CA exactly as on a +-- normal disk. +DROP TABLE IF EXISTS t_proj_cas_alter; + +CREATE TABLE t_proj_cas_alter (a UInt64, b UInt64, PROJECTION p_by_b (SELECT a, b ORDER BY b)) +ENGINE = MergeTree ORDER BY a; + +INSERT INTO t_proj_cas_alter SELECT number, number % 10 FROM numbers(1000); +INSERT INTO t_proj_cas_alter SELECT number, number % 10 FROM numbers(1000, 1000); +OPTIMIZE TABLE t_proj_cas_alter FINAL; + +-- B58 DURABILITY (merge): the merge-rebuilt projection must survive a DETACH/ATTACH — it must live in the +-- committed manifest, not only in memory. Reload and assert the projection is still active and usable. +DETACH TABLE t_proj_cas_alter; +ATTACH TABLE t_proj_cas_alter; +SELECT 'after_merge_reload_projection', countDistinct(name) FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas_alter' AND active; +SET optimize_use_projections = 1, force_optimize_projection = 1; +SELECT 'after_merge_reload_uses_projection', countIf(explain LIKE '%p_by_b%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, count() FROM t_proj_cas_alter GROUP BY b); +SET force_optimize_projection = 0; + +-- ALTER ADD PROJECTION on an existing table, then MATERIALIZE it on existing parts (rebuild path). +-- This exercises the temp-projection (.tmp_proj -> .proj) flow inside the mutated part on +-- the CA disk. +ALTER TABLE t_proj_cas_alter ADD PROJECTION p_sum (SELECT b, sum(a) GROUP BY b); +ALTER TABLE t_proj_cas_alter MATERIALIZE PROJECTION p_sum SETTINGS mutations_sync = 2; +SELECT 'after_add_projection_count', count() FROM t_proj_cas_alter; +-- After MATERIALIZE both the pre-existing p_by_b and the freshly built p_sum must be active. B58: the +-- mutation must carry p_by_b forward and persist p_sum into the manifest of the rebuilt part. +SELECT 'projections_after_add', name, count() FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas_alter' AND active GROUP BY name ORDER BY name; + +-- The newly materialized projection must actually be selected by the optimizer. +SELECT 'uses_p_sum', countIf(explain LIKE '%p_sum%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, sum(a) FROM t_proj_cas_alter GROUP BY b); + +-- B58 DURABILITY (materialize): both projections must survive a DETACH/ATTACH after MATERIALIZE. +DETACH TABLE t_proj_cas_alter; +ATTACH TABLE t_proj_cas_alter; +SELECT 'projections_after_materialize_reload', name, count() FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas_alter' AND active GROUP BY name ORDER BY name; +SET optimize_use_projections = 1, force_optimize_projection = 1; +SELECT 'after_materialize_reload_uses_p_sum', countIf(explain LIKE '%p_sum%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, sum(a) FROM t_proj_cas_alter GROUP BY b); +SET force_optimize_projection = 0; + +-- B58 DURABILITY (data mutation): a mutation rebuilds the part; the surviving projections must be carried +-- into the mutated part's manifest and stay usable after a reload. We use a DELETE that matches no rows so +-- the row data (and therefore every expected value below) is unchanged across CA and non-CA — the part is +-- still fully rewritten, exercising the mutation projection path. +ALTER TABLE t_proj_cas_alter DELETE WHERE b = 999 SETTINGS mutations_sync = 2; +DETACH TABLE t_proj_cas_alter; +ATTACH TABLE t_proj_cas_alter; +SELECT 'projections_after_update_reload', name, count() FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas_alter' AND active GROUP BY name ORDER BY name; +SET optimize_use_projections = 1, force_optimize_projection = 1; +SELECT 'after_update_reload_uses_p_sum', countIf(explain LIKE '%p_sum%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, sum(a) FROM t_proj_cas_alter GROUP BY b); +SET force_optimize_projection = 0; + +-- DROP a projection: results unchanged, the projection's nested keys leave the new part version. +ALTER TABLE t_proj_cas_alter DROP PROJECTION p_by_b SETTINGS mutations_sync = 2; +SELECT 'after_drop_projection_count', count() FROM t_proj_cas_alter; +SELECT 'projections_after_drop', name, count() FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cas_alter' AND active GROUP BY name ORDER BY name; + +-- Persistence: reload from the disk and re-read. `p_by_b` is gone, so the count() query falls back to the +-- base table; the surviving `p_sum` still serves the sum(a) aggregation after the reload. +DETACH TABLE t_proj_cas_alter; +ATTACH TABLE t_proj_cas_alter; +SELECT 'after_reload_by_b', b, count() FROM t_proj_cas_alter GROUP BY b ORDER BY b; +SELECT 'after_reload_sum_b', b, sum(a) FROM t_proj_cas_alter GROUP BY b ORDER BY b; + +DROP TABLE t_proj_cas_alter; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04300_cas_projection_multiblock.reference b/tests/queries/0_stateless/04300_cas_projection_multiblock.reference new file mode 100644 index 000000000000..c16767ba74c9 --- /dev/null +++ b/tests/queries/0_stateless/04300_cas_projection_multiblock.reference @@ -0,0 +1,14 @@ +count 2600000 +by_b_top 1299999 3899998 +by_b_top 1299998 3899996 +by_b_top 1299997 3899994 +after_merge_by_b_top 1299999 3899998 +after_merge_by_b_top 1299998 3899996 +after_merge_by_b_top 1299997 3899994 +after_materialize_count 2600000 +projection_active 1 +uses_projection 1 +after_reload_by_b_top 1299999 3899998 +after_reload_by_b_top 1299998 3899996 +after_reload_by_b_top 1299997 3899994 +dropped_ok diff --git a/tests/queries/0_stateless/04300_cas_projection_multiblock.sql b/tests/queries/0_stateless/04300_cas_projection_multiblock.sql new file mode 100644 index 000000000000..bdeab3af8848 --- /dev/null +++ b/tests/queries/0_stateless/04300_cas_projection_multiblock.sql @@ -0,0 +1,53 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- A projection built across MULTIPLE temp projection blocks (spill-and-merge) must read its own staged +-- temp blocks back on a content-addressed disk (B59). MergeProjectionPartsTask only EXERCISES the +-- read-back path when it has >1 temp projection part to merge (selected_parts.size() > 1); with a single +-- temp part it just renames it. The temp-part flush threshold is min_insert_block_size_rows, and the +-- background merge/mutation runs in the server's background context (NOT the client query settings), so +-- the threshold is the server default (DEFAULT_INSERT_BLOCK_SIZE = 1048449). We therefore make the +-- projection emit MORE rows than that: a high-cardinality GROUP BY key (1.3M distinct groups) forces >=2 +-- temp projection parts for BOTH an OPTIMIZE merge and an ALTER ... MATERIALIZE PROJECTION rebuild. + +DROP TABLE IF EXISTS t_pmb; +CREATE TABLE t_pmb (a UInt64, b UInt64, PROJECTION p_by_b (SELECT b, sum(a) GROUP BY b)) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk(type = object_storage, object_storage_type = local, metadata_type = cas, + name = '04300_pmb', cas_server_root_id = '04300', path = '04300_pmb_pool/'), + -- The final check asserts the optimizer SELECTS the projection, which holds only while the + -- projection reads fewer marks than the base table. Randomized granularity (tiny + -- index_granularity_bytes with enable_block_offset_column widening base rows) can bring the two + -- within one mark of each other and flip the choice, so pin the defaults. + index_granularity = 8192, index_granularity_bytes = 10485760; + +-- 1.3M distinct b values, each appearing twice: a = number and a = number + 1300000, so for group b the +-- two rows are b and b + 1300000 -> sum(a) = 2*b + 1300000. The projection emits 1.3M rows > 1048449 -> +-- >= 2 temp projection parts on rebuild. +INSERT INTO t_pmb SELECT number, number FROM numbers(1300000); +INSERT INTO t_pmb SELECT number + 1300000, number FROM numbers(1300000); + +SELECT 'count', count() FROM t_pmb; +SELECT 'by_b_top', b, sum(a) AS s FROM t_pmb GROUP BY b ORDER BY s DESC, b LIMIT 3; + +-- MERGE the parts: the projection rebuild merges >1 temp projection part (multi-block read-back). +OPTIMIZE TABLE t_pmb FINAL; +SELECT 'after_merge_by_b_top', b, sum(a) AS s FROM t_pmb GROUP BY b ORDER BY s DESC, b LIMIT 3; + +-- MUTATION that rebuilds the projection across >1 temp projection block: +ALTER TABLE t_pmb MATERIALIZE PROJECTION p_by_b SETTINGS mutations_sync = 2; +SELECT 'after_materialize_count', count() FROM t_pmb; +SELECT 'projection_active', countDistinct(name) FROM system.projection_parts WHERE database = currentDatabase() AND table = 't_pmb' AND active; + +-- Prove the projection is actually selected by the optimizer (not a silent base-table fallback). +SET optimize_use_projections = 1, force_optimize_projection = 1; +SELECT 'uses_projection', countIf(explain LIKE '%p_by_b%') > 0 +FROM (EXPLAIN actions = 1 SELECT b, sum(a) FROM t_pmb GROUP BY b); +SET force_optimize_projection = 0; + +-- survives reload: +DETACH TABLE t_pmb; ATTACH TABLE t_pmb; +SELECT 'after_reload_by_b_top', b, sum(a) AS s FROM t_pmb GROUP BY b ORDER BY s DESC, b LIMIT 3; + +DROP TABLE t_pmb; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/04316_reader_executor_basic.sql b/tests/queries/0_stateless/04316_reader_executor_basic.sql index 950db07f0003..f063513d9199 100644 --- a/tests/queries/0_stateless/04316_reader_executor_basic.sql +++ b/tests/queries/0_stateless/04316_reader_executor_basic.sql @@ -1,9 +1,12 @@ --- Tags: no-distributed-cache, no-encrypted-storage +-- Tags: no-distributed-cache, no-encrypted-storage, no-cas-storage -- The executor does not implement the distributed cache or decryption, so it -- falls back on those storage configs and the activation check below would not -- hold. Those stages can't be turned off from the test (unlike async prefetch -- and the filesystem cache), so skip them; the test still runs on local disk and --- plain object storage where the executor engages. +-- plain object storage where the executor engages. Content-addressed storage +-- always adds a `file_view` stage (the payload is a byte window inside a shared +-- blob), which the executor falls back on the same way -- see +-- `ReadPipeline::tryBuildReaderExecutor` -- so it never engages there either. -- -- Smoke test for the experimental ReaderExecutor read path. Reads a MergeTree -- table with `use_reader_executor = 1`, checks the data comes back correct (full diff --git a/tests/queries/0_stateless/04327_reader_executor_metrics.sql b/tests/queries/0_stateless/04327_reader_executor_metrics.sql index 6c805e545511..65cfb6b88da7 100644 --- a/tests/queries/0_stateless/04327_reader_executor_metrics.sql +++ b/tests/queries/0_stateless/04327_reader_executor_metrics.sql @@ -1,3 +1,4 @@ +<<<<<<< HEAD -- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas -- Like 04316, the executor falls back on the distributed cache and decryption -- (which can't be disabled from the test), so its metrics would not be emitted on @@ -5,6 +6,15 @@ -- object storage where the executor engages. -- no-parallel-replicas: the counters are incremented on whichever replica reads the -- mark, so the initiator's `query_log` row does not carry them. +======= +-- Tags: no-distributed-cache, no-encrypted-storage, no-cas-storage +-- Like 04316, the executor falls back on the distributed cache and decryption +-- (which can't be disabled from the test), so its metrics would not be emitted on +-- those storage configs. Skip them; the test still runs on local disk and plain +-- object storage where the executor engages. Content-addressed storage always +-- adds a `file_view` stage (byte window inside a shared blob), which the +-- executor falls back on the same way -- see `ReadPipeline::tryBuildReaderExecutor`. +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) -- -- Checks that the experimental ReaderExecutor emits its observability metrics. -- Reads a MergeTree table with `use_reader_executor = 1` and verifies, via the diff --git a/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql b/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql index 418b6281a9fe..5c9b82c57bcc 100644 --- a/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql +++ b/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql @@ -1,9 +1,18 @@ +<<<<<<< HEAD -- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas -- The executor falls back on the distributed cache and decryption (which can't be -- disabled from the test), so its metrics would not be emitted there; skip those -- configs (as in 04316 / 04327). -- no-parallel-replicas: the counters are incremented on whichever replica reads the -- mark, so the initiator's `query_log` row does not carry them. +======= +-- Tags: no-distributed-cache, no-encrypted-storage, no-cas-storage +-- The executor falls back on the distributed cache and decryption (which can't be +-- disabled from the test), so its metrics would not be emitted there; skip those +-- configs (as in 04316 / 04327). Content-addressed storage always adds a +-- `file_view` stage (byte window inside a shared blob), which the executor +-- falls back on the same way -- see `ReadPipeline::tryBuildReaderExecutor`. +>>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) -- -- End-to-end check that the modeled-cost KPI asynchronous metric -- `ReaderExecutorModeledCostMsPerRequestedMiB` moves when the executor does work. diff --git a/tests/queries/0_stateless/05000_cas_projection_carry_forward.reference b/tests/queries/0_stateless/05000_cas_projection_carry_forward.reference new file mode 100644 index 000000000000..42430b9414c0 --- /dev/null +++ b/tests/queries/0_stateless/05000_cas_projection_carry_forward.reference @@ -0,0 +1,10 @@ +count 100000 +no_projection 1 0 0 1249950000 +no_projection 1 0 2 1250000000 +no_projection 1 1 1 1249975000 +no_projection 1 1 3 1250025000 +with_projection 1 0 0 1249950000 +with_projection 1 0 2 1250000000 +with_projection 1 1 1 1249975000 +with_projection 1 1 3 1250025000 +projection_parts 2 4 diff --git a/tests/queries/0_stateless/05000_cas_projection_carry_forward.sql b/tests/queries/0_stateless/05000_cas_projection_carry_forward.sql new file mode 100644 index 000000000000..b36959270616 --- /dev/null +++ b/tests/queries/0_stateless/05000_cas_projection_carry_forward.sql @@ -0,0 +1,45 @@ +-- Tags: no-random-settings, no-random-merge-tree-settings + +-- B63: MATERIALIZE PROJECTION over a table with HETEROGENEOUS projection coverage. The first part +-- predates ADD PROJECTION (it must BUILD the projection); a later part already has it (the mutation +-- CARRIES IT FORWARD). On a content-addressed disk the carried-forward projection part was registered +-- in-memory without its rows_count / index granularity (the hardlinked files are not yet committed, so +-- it cannot reload them from disk), so a projection-served SELECT read back NOTHING from that part and +-- silently dropped its rows from the aggregate. The fix copies the source projection part's already-loaded +-- read-time state. This oracle compares the projection-served aggregate against the non-projection one in +-- the same run, so it is correct on both a plain and a content-addressed default disk. + +DROP TABLE IF EXISTS t_proj_cf; + +CREATE TABLE t_proj_cf (k1 UInt32, k2 UInt32, k3 UInt32, value UInt32) +ENGINE = MergeTree ORDER BY tuple(); + +-- First part: NO projection yet. +INSERT INTO t_proj_cf SELECT 1, number % 2, number % 4, number FROM numbers(50000); + +SYSTEM STOP MERGES t_proj_cf; + +ALTER TABLE t_proj_cf ADD PROJECTION aaaa (SELECT k1, k2, k3, sum(value) GROUP BY k1, k2, k3); + +-- Second part: built WITH the projection (INSERT after ADD PROJECTION). +INSERT INTO t_proj_cf SELECT 1, number % 2, number % 4, number FROM numbers(100000) LIMIT 50000, 100000; + +SYSTEM START MERGES t_proj_cf; + +ALTER TABLE t_proj_cf MATERIALIZE PROJECTION aaaa SETTINGS mutations_sync = 2; + +SELECT 'count', count() FROM t_proj_cf; + +SELECT 'no_projection', k1, k2, k3, sum(value) v +FROM t_proj_cf GROUP BY k1, k2, k3 ORDER BY k1, k2, k3 +SETTINGS optimize_use_projections = 0; + +SELECT 'with_projection', k1, k2, k3, sum(value) v +FROM t_proj_cf GROUP BY k1, k2, k3 ORDER BY k1, k2, k3; + +-- Every active part must carry a non-empty projection part after MATERIALIZE. +SELECT 'projection_parts', countDistinct(parent_name), min(rows) +FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_proj_cf' AND active; + +DROP TABLE t_proj_cf; diff --git a/tests/queries/0_stateless/05001_cas_attach_partition_projection.reference b/tests/queries/0_stateless/05001_cas_attach_partition_projection.reference new file mode 100644 index 000000000000..1d143c7cd0d1 --- /dev/null +++ b/tests/queries/0_stateless/05001_cas_attach_partition_projection.reference @@ -0,0 +1,11 @@ +before_attach_rows 7 +after_attach_rows 7 +data 7 21 21 +projection_served 0 0 +projection_served 1 1 +projection_served 2 2 +projection_served 3 3 +projection_served 4 4 +projection_served 5 5 +projection_served 6 6 +1 diff --git a/tests/queries/0_stateless/05001_cas_attach_partition_projection.sql b/tests/queries/0_stateless/05001_cas_attach_partition_projection.sql new file mode 100644 index 000000000000..5808b8d05efb --- /dev/null +++ b/tests/queries/0_stateless/05001_cas_attach_partition_projection.sql @@ -0,0 +1,46 @@ +-- Tags: no-random-settings, no-random-merge-tree-settings + +-- B64: DETACH PARTITION + ATTACH PARTITION of a part that has a projection. On a content-addressed +-- disk the part is re-attached from its detached STAGING directory (detached/attaching_/), so +-- the projection sub-directory is read as the NESTED path detached/attaching_/.proj. The +-- CA metadata storage recognized a projection directory only as a DIRECT child of a part +-- (/.proj), so the nested staging shape was missed: existsDirectory(".proj") returned +-- false during the attach-time load, and IMergeTreeDataPart::loadProjections registered the surviving +-- projection part with EMPTY columns and rows_count == 0 — making it unusable (PROJECTION_NOT_USED) and +-- causing CHECK TABLE to throw BROKEN_PROJECTION (in-memory columns empty vs on-disk columns), even +-- though the on-disk projection data was intact. Same projection-on-CA family as B58/B63, on the +-- ATTACH-clone path. This oracle exercises DETACH+ATTACH PARTITION (no projection drop) and asserts the +-- surviving projection re-attaches with the correct rows, is usable, and CHECK TABLE passes. It is +-- correct on both a plain and a content-addressed default disk. + +DROP TABLE IF EXISTS t_attach_proj; + +CREATE TABLE t_attach_proj (x Int32, y Int32, PROJECTION p (SELECT x, y ORDER BY x)) +ENGINE = MergeTree() PARTITION BY intDiv(y, 100) ORDER BY y; + +INSERT INTO t_attach_proj SELECT number, number FROM numbers(7); + +SELECT 'before_attach_rows', min(rows) +FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_attach_proj' AND active; + +ALTER TABLE t_attach_proj DETACH PARTITION 0; +ALTER TABLE t_attach_proj ATTACH PARTITION 0; + +-- The surviving projection must re-attach with the correct row count (rows > 0), not empty. +SELECT 'after_attach_rows', min(rows) +FROM system.projection_parts +WHERE database = currentDatabase() AND table = 't_attach_proj' AND active; + +-- Base data must be intact. +SELECT 'data', count(), sum(x), sum(y) FROM t_attach_proj; + +-- The projection must be usable: force_optimize_projection requires a projection to serve the query, +-- so this throws if the projection is broken/empty. +SELECT 'projection_served', x, y FROM t_attach_proj ORDER BY x +SETTINGS optimize_use_projections = 1, force_optimize_projection = 1; + +-- CHECK TABLE must pass (the projection's in-memory columns must match the on-disk columns). +CHECK TABLE t_attach_proj SETTINGS check_query_single_value_result = 1; + +DROP TABLE t_attach_proj; diff --git a/tests/queries/0_stateless/05002_cas_fetch_partition.reference b/tests/queries/0_stateless/05002_cas_fetch_partition.reference new file mode 100644 index 000000000000..b883895433c7 --- /dev/null +++ b/tests/queries/0_stateless/05002_cas_fetch_partition.reference @@ -0,0 +1,8 @@ +src_parts 1 +detached_parts 1 +detached_after_attach 0 +attached_rows 3 +data_readback 0 a +data_readback 2 b +data_readback 4 c +dropped_ok diff --git a/tests/queries/0_stateless/05002_cas_fetch_partition.sql b/tests/queries/0_stateless/05002_cas_fetch_partition.sql new file mode 100644 index 000000000000..cbb05b2cafac --- /dev/null +++ b/tests/queries/0_stateless/05002_cas_fetch_partition.sql @@ -0,0 +1,58 @@ +-- Tags: no-fasttest, no-shared-merge-tree, no-replicated-database +-- ^ no-fasttest: cas is an object-storage metadata type; keep it off the minimal image. +-- no-shared-merge-tree: this exercises open-source ReplicatedMergeTree on a cas disk. +-- no-replicated-database: the source replica_path is hard-coded per the database, not per the replica. + +-- ALTER TABLE ... FETCH PARTITION ... FROM '' on a cas disk: the gate is lifted +-- and a to_detached fetch takes the byte-fetch path (the downloaded files content-address into the +-- detached/ namespace; relink-into-detached is deferred). The fetched part must land usably in the CA +-- detached/ namespace: system.detached_parts lists it, ATTACH publishes an active part out of it, and a +-- SELECT reads back the exact source data. Both tables share one inline CA pool (a single server fetches +-- from its own zk path, as 03350 does), so this also exercises the cross-table detached landing. + +DROP TABLE IF EXISTS t_cas_fetch_src; +DROP TABLE IF EXISTS t_cas_fetch_dst; + +CREATE TABLE t_cas_fetch_src (key Int, s String) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/t_cas_fetch_src', 'r1') +PARTITION BY (key % 2) ORDER BY key +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05002', + name = '05002_cas_fetch', + path = '05002_cas_fetch_pool/'); + +CREATE TABLE t_cas_fetch_dst (key Int, s String) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/t_cas_fetch_dst', 'r1') +PARTITION BY (key % 2) ORDER BY key +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05002', + name = '05002_cas_fetch', + path = '05002_cas_fetch_pool/'); + +INSERT INTO t_cas_fetch_src VALUES (0, 'a'), (2, 'b'), (4, 'c'); +SELECT 'src_parts', count() FROM system.parts WHERE database = currentDatabase() AND table = 't_cas_fetch_src' AND active AND partition = '0'; + +-- Fetch the single part of partition 0 into the destination's detached/ namespace. +ALTER TABLE t_cas_fetch_dst FETCH PARTITION 0 FROM '/clickhouse/tables/{database}/t_cas_fetch_src' + SETTINGS insert_keeper_fault_injection_probability = 0; + +-- The fetched part must be present as a detached part. +SELECT 'detached_parts', count() FROM system.detached_parts WHERE database = currentDatabase() AND table = 't_cas_fetch_dst'; + +-- ATTACH publishes an active part out of the detached landing; SELECT must read back the exact data. +ALTER TABLE t_cas_fetch_dst ATTACH PARTITION 0 + SETTINGS insert_keeper_fault_injection_probability = 0; + +SELECT 'detached_after_attach', count() FROM system.detached_parts WHERE database = currentDatabase() AND table = 't_cas_fetch_dst'; +SELECT 'attached_rows', count() FROM t_cas_fetch_dst; +SELECT 'data_readback', key, s FROM t_cas_fetch_dst ORDER BY key; + +DROP TABLE t_cas_fetch_src; +DROP TABLE t_cas_fetch_dst; +SELECT 'dropped_ok'; diff --git a/tests/queries/0_stateless/05003_cas_freeze.reference b/tests/queries/0_stateless/05003_cas_freeze.reference new file mode 100644 index 000000000000..1740ea879482 --- /dev/null +++ b/tests/queries/0_stateless/05003_cas_freeze.reference @@ -0,0 +1,7 @@ +live_before_freeze 1 3 +live_before_freeze 2 2 +is_frozen 1 +live_after_drop 2 2 +command_type partition_id part_name backup_name +SYSTEM UNFREEZE 1 1_1_1_0 backup_05003 +dropped_ok diff --git a/tests/queries/0_stateless/05003_cas_freeze.sh b/tests/queries/0_stateless/05003_cas_freeze.sh new file mode 100755 index 000000000000..db4021963aa9 --- /dev/null +++ b/tests/queries/0_stateless/05003_cas_freeze.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# CA durability oracle: a FREEZE PARTITION snapshot is an independent GC root — it survives +# ALTER TABLE ... DROP PARTITION on the same partition and remains independently recoverable. +# This property is CA-specific: on a plain disk FREEZE makes a hard-link snapshot in shadow/ which +# is independent by construction; on a CA disk the frozen part must be written as a separate shadow +# ref (not merely an alias of the live part ref) so DROP PARTITION cannot destroy it. + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +UNFREEZE_STRUCTURE='command_type String, partition_id String, part_name String, backup_name String, backup_path String, part_backup_path String' + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_freeze;" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE t_cas_freeze (k UInt32, v String) +ENGINE = MergeTree ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05003', + name = '05003_cas_freeze', + path = '05003_cas_freeze_pool/');" + +# Two partitions: k=1 (will be frozen then dropped) and k=2 (must survive untouched). +${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MERGES t_cas_freeze;" +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_freeze VALUES (1, 'a'), (1, 'b'), (1, 'c');" +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_freeze VALUES (2, 'x'), (2, 'y');" +${CLICKHOUSE_CLIENT} --query "SYSTEM START MERGES t_cas_freeze;" + +${CLICKHOUSE_CLIENT} --query "SELECT 'live_before_freeze', k, count() FROM t_cas_freeze GROUP BY k ORDER BY k;" + +# Freeze only partition 1. The shadow ref becomes an independent GC root on the CA disk. +${CLICKHOUSE_CLIENT} --query "ALTER TABLE t_cas_freeze FREEZE PARTITION 1 WITH NAME 'backup_05003';" + +${CLICKHOUSE_CLIENT} --query " +SELECT 'is_frozen', count() FROM system.parts +WHERE database = currentDatabase() AND table = 't_cas_freeze' + AND partition_id = '1' AND is_frozen AND active;" + +# Drop the live partition 1. On a CA disk this must NOT remove the shadow ref. +${CLICKHOUSE_CLIENT} --query "ALTER TABLE t_cas_freeze DROP PARTITION 1;" + +# Live k=1 is gone; k=2 is untouched. +${CLICKHOUSE_CLIENT} --query "SELECT 'live_after_drop', k, count() FROM t_cas_freeze GROUP BY k ORDER BY k;" + +# THE KEY ASSERTION: SYSTEM UNFREEZE finds and removes the frozen snapshot of partition 1, +# proving it survived the DROP PARTITION as an independent shadow ref. +# SYSTEM UNFREEZE does not accept a FORMAT clause; default output is TSV, piped through +# clickhouse-local to filter to deterministic columns (backup_path/part_backup_path are +# absolute paths; command_type/partition_id/part_name/backup_name are stable). +${CLICKHOUSE_CLIENT} --query "SYSTEM UNFREEZE WITH NAME 'backup_05003';" \ + | ${CLICKHOUSE_LOCAL} --structure "$UNFREEZE_STRUCTURE" \ + --query "SELECT command_type, partition_id, part_name, backup_name FROM table ORDER BY partition_id FORMAT TSVWithNames" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_freeze;" +${CLICKHOUSE_CLIENT} --query "SELECT 'dropped_ok';" diff --git a/tests/queries/0_stateless/05004_cas_transactions.reference b/tests/queries/0_stateless/05004_cas_transactions.reference new file mode 100644 index 000000000000..5c0d3efe9fd2 --- /dev/null +++ b/tests/queries/0_stateless/05004_cas_transactions.reference @@ -0,0 +1,7 @@ +base 1 +in_txn 2 +after_commit 2 +in_txn2 3 +after_rollback 2 +rolled_back_absent 0 +done diff --git a/tests/queries/0_stateless/05004_cas_transactions.sh b/tests/queries/0_stateless/05004_cas_transactions.sh new file mode 100755 index 000000000000..6a99162587cb --- /dev/null +++ b/tests/queries/0_stateless/05004_cas_transactions.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-ordinary-database +# no-fasttest: cas is an object-storage metadata type; not available on the minimal +# fasttest image. +# no-ordinary-database: transactions require DatabaseAtomic (or similar); they are not supported +# on DatabaseOrdinary. + +# CA transactions oracle: proves that transactional INSERT/COMMIT/ROLLBACK works correctly on a +# content-addressed (CA) disk. Three scenarios are verified: +# 1. A committed transaction's rows become visible after COMMIT. +# 2. A rolled-back transaction's rows are absent after ROLLBACK; prior data is intact. +# 3. Counts are deterministic: base=1, after commit=2, after rollback=2, rolled-back row absent. +# +# MERGES ARE STOPPED immediately after CREATE to prevent any background merge from firing on +# transactional parts during the test (transactional multi-part merges are not yet implemented on +# CA disks — B53 in the backlog). + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_txn;" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE t_cas_txn (k UInt32, v String) +ENGINE = MergeTree ORDER BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05004', + name = '05004_cas_transactions', + path = '05004_cas_transactions_pool/');" + +${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MERGES t_cas_txn;" + +# ── Step 1: base row (outside any transaction) ────────────────────────────── +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_txn VALUES (1, 'a');" +${CLICKHOUSE_CLIENT} --query "SELECT 'base', count() FROM t_cas_txn;" + +# ── Step 2: committed transaction ─────────────────────────────────────────── +# BEGIN … COMMIT must share one client connection (one --query / multiquery block). +${CLICKHOUSE_CLIENT} --query " +BEGIN TRANSACTION; +INSERT INTO t_cas_txn VALUES (2, 'b'); +SELECT 'in_txn', count() FROM t_cas_txn; +COMMIT;" + +${CLICKHOUSE_CLIENT} --query "SELECT 'after_commit', count() FROM t_cas_txn;" + +# ── Step 3: rolled-back transaction ───────────────────────────────────────── +${CLICKHOUSE_CLIENT} --query " +BEGIN TRANSACTION; +INSERT INTO t_cas_txn VALUES (3, 'c'); +SELECT 'in_txn2', count() FROM t_cas_txn; +ROLLBACK;" + +${CLICKHOUSE_CLIENT} --query "SELECT 'after_rollback', count() FROM t_cas_txn;" +${CLICKHOUSE_CLIENT} --query "SELECT 'rolled_back_absent', count() FROM t_cas_txn WHERE k = 3;" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_txn;" +${CLICKHOUSE_CLIENT} --query "SELECT 'done';" diff --git a/tests/queries/0_stateless/05005_cas_backup_restore.reference b/tests/queries/0_stateless/05005_cas_backup_restore.reference new file mode 100644 index 000000000000..36da2c7b5808 --- /dev/null +++ b/tests/queries/0_stateless/05005_cas_backup_restore.reference @@ -0,0 +1,6 @@ +source 5 9 ['a','b','c','d','e'] +restored 5 9 ['a','b','c','d','e'] +projection 1 2 +projection 2 2 +projection 3 1 +done diff --git a/tests/queries/0_stateless/05005_cas_backup_restore.sh b/tests/queries/0_stateless/05005_cas_backup_restore.sh new file mode 100755 index 000000000000..4c361705368d --- /dev/null +++ b/tests/queries/0_stateless/05005_cas_backup_restore.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# no-fasttest: cas is an object-storage metadata type; not available on the minimal +# fasttest image. + +# CA BACKUP/RESTORE round-trip oracle: proves a table on a content-addressed (CA) disk survives a +# full BACKUP -> DROP -> RESTORE cycle with byte-for-byte data equality, including a PROJECTION. +# RESTORE materializes each part through one whole-part ContentAddressedTransaction +# (restorePartFromBackup, commit d384298602b); BACKUP-read already worked. This is the inline-CA +# oracle for B16/B34. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +backup_name="Disk('backups', '${CLICKHOUSE_TEST_UNIQUE_NAME}')" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_br;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_br_restored;" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE t_cas_br (k UInt32, v String, PROJECTION p (SELECT k, count() GROUP BY k)) +ENGINE = MergeTree ORDER BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05005', + name = '05005_cas_backup_restore', + path = '05005_cas_backup_restore_pool/');" + +# Two inserts -> two parts; deterministic rows. +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_br VALUES (1, 'a'), (2, 'b'), (1, 'c');" +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_br VALUES (3, 'd'), (2, 'e');" + +${CLICKHOUSE_CLIENT} --query "SELECT 'source', count(), sum(k), arraySort(groupArray(v)) FROM t_cas_br;" + +${CLICKHOUSE_CLIENT} --query "BACKUP TABLE t_cas_br TO ${backup_name} FORMAT Null;" + +${CLICKHOUSE_CLIENT} --query "RESTORE TABLE t_cas_br AS t_cas_br_restored FROM ${backup_name} FORMAT Null;" + +# Round-trip data equality on the restored table. +${CLICKHOUSE_CLIENT} --query "SELECT 'restored', count(), sum(k), arraySort(groupArray(v)) FROM t_cas_br_restored;" + +# Projection-served query on the restored table (proves the projection round-tripped). +${CLICKHOUSE_CLIENT} --query "SELECT 'projection', k, count() FROM t_cas_br_restored GROUP BY k ORDER BY k SETTINGS force_optimize_projection = 1;" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_br;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_br_restored;" +${CLICKHOUSE_CLIENT} --query "SELECT 'done';" diff --git a/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.reference b/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.reference new file mode 100644 index 000000000000..78bf4c7a8ea9 --- /dev/null +++ b/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.reference @@ -0,0 +1 @@ +1000000 499999500000 499999500000 1000000 diff --git a/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.sql b/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.sql new file mode 100644 index 000000000000..e7e82ea3bf40 --- /dev/null +++ b/tests/queries/0_stateless/05006_cas_deduplication_blob_insert.sql @@ -0,0 +1,18 @@ +-- Tags: long +-- On a content-addressed S3 disk (the cas_s3 test lane), byte-identical column +-- blobs deduplicate to a single object, so the second column's conditional PUT (If-None-Match: *) loses +-- its precondition. Before the `Expect: 100-continue` fix the rejected large body triggered a +-- 500/broken-pipe retry storm in the S3 client and this INSERT hung for tens of minutes (B118). +-- Regression: the INSERT must complete and the data must round-trip. On non-CA storage this is a +-- trivial fast insert. + +DROP TABLE IF EXISTS t_cas_deduplicated_blob; + +CREATE TABLE t_cas_deduplicated_blob (x UInt64, y UInt64) ENGINE = MergeTree ORDER BY x; + +-- x and y are byte-identical -> same content hash -> the second blob's conditional PUT 412s. +INSERT INTO t_cas_deduplicated_blob SELECT number, number FROM numbers(1000000); + +SELECT count(), sum(x), sum(y), sum(x = y) FROM t_cas_deduplicated_blob; + +DROP TABLE t_cas_deduplicated_blob; diff --git a/tests/queries/0_stateless/05007_cas_gc_introspection.reference b/tests/queries/0_stateless/05007_cas_gc_introspection.reference new file mode 100644 index 000000000000..486914465921 --- /dev/null +++ b/tests/queries/0_stateless/05007_cas_gc_introspection.reference @@ -0,0 +1,6 @@ +1 1 1 +1 +1 1 1 +1 1 1 +1 +ok diff --git a/tests/queries/0_stateless/05007_cas_gc_introspection.sh b/tests/queries/0_stateless/05007_cas_gc_introspection.sh new file mode 100755 index 000000000000..cf287a343845 --- /dev/null +++ b/tests/queries/0_stateless/05007_cas_gc_introspection.sh @@ -0,0 +1,109 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# Introspection coverage for the content-addressed (CA) garbage collector: the +# `SYSTEM CAS GC RUN ` command runs one GC round synchronously and +# the round is recorded in `system.cas_gc_log` (a Start + Finish row +# per round, like `part_log`). We build a CA disk inline (named, so the SYSTEM command can target it), +# create garbage by inserting then truncating, run the round a few times, flush the log, and assert +# the rows are there with the right shape — including a non-empty per-round `ProfileEvents` delta +# (the Manual round runs on the query thread, which always has an attached ThreadStatus that captures +# ProfileEvents). +# +# This is a .sh test (not .sql) because `SYSTEM CAS GC RUN` now returns a +# one-row-per-disk result set (UX pass); the three synchronous rounds below only care about their +# side effects on the log, so their own output is redirected to /dev/null. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP TABLE IF EXISTS t_cas_gc_introspection; + +-- A named inline CA disk: the \`name\` is what \`SYSTEM CAS GC RUN \` +-- targets and what lands in the log's \`disk_name\` column. +CREATE TABLE t_cas_gc_introspection (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05007', + name = '05007_cas_gc_introspection', + path = '05007_cas_gc_introspection_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1), + old_parts_lifetime = 1; + +-- Two distinct inserts => distinct blobs (not deduped away), then TRUNCATE drops every ref so the +-- blobs/trees become unreferenced GC fodder. +INSERT INTO t_cas_gc_introspection SELECT number, toString(number) FROM numbers(1000); +INSERT INTO t_cas_gc_introspection SELECT number, toString(number) FROM numbers(1000, 1000); +TRUNCATE TABLE t_cas_gc_introspection; +""" + +# Run several synchronous rounds: the first rounds mark the retired candidates, later rounds delete +# them once the durable watermark floor advances past the builds (the background renewer does this). +${CLICKHOUSE_CLIENT} -q "SYSTEM CAS GC RUN '05007_cas_gc_introspection'" > /dev/null +${CLICKHOUSE_CLIENT} -q "SYSTEM CAS GC RUN '05007_cas_gc_introspection'" > /dev/null +${CLICKHOUSE_CLIENT} -q "SYSTEM CAS GC RUN '05007_cas_gc_introspection'" > /dev/null + +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM FLUSH LOGS cas_gc_log; + +-- A Start, a Finish, and a Manual-triggered row were all recorded for this disk. +SELECT + countIf(event_type = 'Start') > 0, + countIf(event_type = 'Finish') > 0, + countIf(trigger = 'Manual') > 0 +FROM system.cas_gc_log +WHERE disk_name LIKE '%05007_cas_gc_introspection%'; + +-- A synchronous Manual Finish captured a non-empty per-round ProfileEvents delta (the round touches +-- the object storage, so Cas*/Disk*/S3* counters are non-zero). The query thread is always attached, +-- so capture is active for the Manual path. +SELECT any(length(ProfileEvents)) > 0 +FROM system.cas_gc_log +WHERE disk_name LIKE '%05007_cas_gc_introspection%' + AND event_type = 'Finish' + AND trigger = 'Manual'; + +-- Per-phase rows: a folding round emits one Phase row per phase it reached (19 of them), so a run +-- that folded at least once must show most of the phase vocabulary, including the fold's own +-- ref-prefix enumeration and the round-commit CAS. +SELECT countDistinct(phase) >= 10, + countIf(phase = 'fold_ref_group') > 0, + countIf(phase = 'round_commit') > 0 +FROM system.cas_gc_log +WHERE disk_name LIKE '%05007_cas_gc_introspection%' + AND event_type = 'Phase'; + +-- The correlator: every row of the most recent round of this disk -- its Start, each Phase, and its +-- Finish -- shares one \`round_id\`, and that round emitted at least one Phase row between them. +SELECT countIf(event_type = 'Start') = 1, + countIf(event_type = 'Finish') = 1, + countIf(event_type = 'Phase') > 0 +FROM system.cas_gc_log +WHERE round_id = ( + SELECT round_id FROM system.cas_gc_log + WHERE disk_name LIKE '%05007_cas_gc_introspection%' AND event_type = 'Finish' + ORDER BY event_time_microseconds DESC LIMIT 1); + +-- A Phase row's \`ProfileEvents\` is that phase's own delta, not the whole round's: the phase that +-- enumerates the ref prefix must show fewer events than the round summary it is a part of. That +-- enumeration lives in \`defer_decision\`, which owns the \`cas/refs/\` LIST. It is deliberately NOT +-- \`fold_ref_group\`: that phase is I/O-free by construction -- the keys are already in hand -- so its +-- own delta is empty and this assertion would answer 0 there no matter how healthy the round was. +SELECT max(length(ProfileEvents)) > 0 +FROM system.cas_gc_log +WHERE disk_name LIKE '%05007_cas_gc_introspection%' + AND event_type = 'Phase' AND phase = 'defer_decision'; + +-- The error path: a non-CA disk (the always-present local \`default\`) is rejected. +SYSTEM CAS GC RUN 'default'; -- { serverError BAD_ARGUMENTS } + +DROP TABLE t_cas_gc_introspection; +SELECT 'ok'; +""" diff --git a/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.reference b/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.reference new file mode 100644 index 000000000000..9972842f9827 --- /dev/null +++ b/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.reference @@ -0,0 +1 @@ +1 1 diff --git a/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.sh b/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.sh new file mode 100755 index 000000000000..8cb535ea7bd4 --- /dev/null +++ b/tests/queries/0_stateless/05008_cas_gc_snapshot_prune.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# End-to-end: a content-addressed (CA) GC round physically deletes unreferenced objects ONLY through +# the ack-floor retired-cursor pipeline — a blob is condemned (stage 1), floor-passed / republished as +# `delete_pending` (stage 2), then exact-token deleted (stage 3), surfaced as `entries_redeleted` / +# `objects_deleted` in `system.cas_gc_log`. We build a named inline CA +# disk, create garbage (INSERT then TRUNCATE), then run synchronous GC rounds in a retry loop until a +# round reports a physical delete — graduation (and so the physical delete) is round-paced, gated on +# `condemn_round < current_round` in `settleEntry` (`renewWatermarkOnce` exists but does not gate it), +# so we poll for a round boundary to pass rather than assume a fixed round count. Once a delete is +# observed we assert every physical delete went through stage 3 (`entries_redeleted >= objects_deleted`): +# the redelete loop is the SOLE content-delete site and counts one redelete per attempt +# (Deleted/Absent/Replaced), so this inequality is a structural identity of the pipeline and an ad-hoc +# delete that bypassed the graduate->redelete path would break it. This proves deletion fires +# end-to-end through the real SystemLog path. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK="05008_cas_gc_snapshot_prune" + +# CA-over-LOCAL object storage emits a one-time about emulated conditional operations on +# mount; the .sh harness fails on ANY client stderr, so send only error+ logs to the client (real +# errors still surface and fail the test; the expected mount warning does not). +CLIENT="$CLICKHOUSE_CLIENT --send_logs_level=error" + +$CLIENT -q "DROP TABLE IF EXISTS t_ca_p9" + +$CLIENT -q " +CREATE TABLE t_ca_p9 (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '${DISK}', + name = '${DISK}', + path = '${DISK}_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1), + old_parts_lifetime = 1" + +# Two distinct inserts => distinct blobs (not deduplicated away); TRUNCATE drops every ref so the +# blobs/trees become unreferenced GC fodder. +$CLIENT -q "INSERT INTO t_ca_p9 SELECT number, toString(number) FROM numbers(1000)" +$CLIENT -q "INSERT INTO t_ca_p9 SELECT number, toString(number) FROM numbers(1000, 1000)" +$CLIENT -q "TRUNCATE TABLE t_ca_p9" + +# Run synchronous rounds until a physical delete is observed (bounded retries; graduation is +# round-paced, so we poll for a round boundary to pass rather than assume a fixed round count). +deleted=0 +for _ in $(seq 1 40); do + $CLIENT -q "SYSTEM CAS GC RUN '${DISK}'" > /dev/null + $CLIENT -q "SYSTEM FLUSH LOGS cas_gc_log" + deleted=$($CLIENT -q " + SELECT sum(objects_deleted) + FROM system.cas_gc_log + WHERE disk_name LIKE '%${DISK}%' AND event_type = 'Finish'") + if [ "${deleted:-0}" -gt 0 ]; then break; fi + sleep 0.5 +done + +# A physical delete happened, and every physical delete went through stage 3 of the retired-cursor +# pipeline (`entries_redeleted >= objects_deleted`, a structural identity: the redelete loop is the sole +# content-delete site and increments `redeleted` once per attempt). Expect: "deleted>0 redeleted>=deleted" +# => 1 1. +$CLIENT -q " +SELECT + sum(objects_deleted) > 0, + sum(entries_redeleted) >= sum(objects_deleted) +FROM system.cas_gc_log +WHERE disk_name LIKE '%${DISK}%' AND event_type = 'Finish'" + +$CLIENT -q "DROP TABLE t_ca_p9" diff --git a/tests/queries/0_stateless/05009_cas_event_log.reference b/tests/queries/0_stateless/05009_cas_event_log.reference new file mode 100644 index 000000000000..1ee92adcf718 --- /dev/null +++ b/tests/queries/0_stateless/05009_cas_event_log.reference @@ -0,0 +1,4 @@ +rows 2000 +1 +has_blob_put 1 +ok diff --git a/tests/queries/0_stateless/05009_cas_event_log.sql b/tests/queries/0_stateless/05009_cas_event_log.sql new file mode 100644 index 000000000000..5e001ba651e8 --- /dev/null +++ b/tests/queries/0_stateless/05009_cas_event_log.sql @@ -0,0 +1,44 @@ +-- Tags: no-fasttest +-- ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +-- Default-ON contract for `system.cas_log`: the per-event content-addressed audit log is +-- enabled by default. `programs/server/config.xml` ships a `` section because the +-- CAS disk feature is experimental and this audit log is its primary forensic instrument (it costs +-- nothing when no CAS disk is configured — events are emitted only by content-addressed disks). After we +-- exercise a content-addressed disk end-to-end (INSERT, OPTIMIZE), the table exists and carries this +-- disk's write-path events. + +DROP TABLE IF EXISTS t_cas_event_log; + +CREATE TABLE t_cas_event_log (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05009', + name = '05009_cas_event_log', + path = '05009_cas_event_log_pool/'); + +-- Exercise the content-addressed write/merge path: this is exactly the work that emits put/ref events. +INSERT INTO t_cas_event_log SELECT number, toString(number % 7) FROM numbers(1000); +INSERT INTO t_cas_event_log SELECT number, toString(number % 7) FROM numbers(1000, 1000); +OPTIMIZE TABLE t_cas_event_log FINAL; + +SELECT 'rows', count() FROM t_cas_event_log; + +-- Make the buffered events durable before we read them back. +SYSTEM FLUSH LOGS cas_log; + +-- Default-on assertion #1: the table exists (the config ships the section). +EXISTS TABLE system.cas_log; + +-- Default-on assertion #2: our disk's write path emitted at least one `blob_put` event. Filter by +-- disk_name so parallel tests sharing this system table (e.g. the lane's own cas_s3 disk) +-- cannot perturb the result. +SELECT 'has_blob_put', count() > 0 +FROM system.cas_log +WHERE disk_name = '05009_cas_event_log' AND event_type = 'blob_put'; + +DROP TABLE t_cas_event_log; +SELECT 'ok'; diff --git a/tests/queries/0_stateless/05010_cas_mounts_gc_health.reference b/tests/queries/0_stateless/05010_cas_mounts_gc_health.reference new file mode 100644 index 000000000000..92f6b78e6c78 --- /dev/null +++ b/tests/queries/0_stateless/05010_cas_mounts_gc_health.reference @@ -0,0 +1,3 @@ +1 0 +1 1 +ok diff --git a/tests/queries/0_stateless/05010_cas_mounts_gc_health.sh b/tests/queries/0_stateless/05010_cas_mounts_gc_health.sh new file mode 100755 index 000000000000..047ee5b0631e --- /dev/null +++ b/tests/queries/0_stateless/05010_cas_mounts_gc_health.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# B3: system.cas_mounts exposes per-disk GC health (is_leader / pending_reclaim / +# last_success_age_seconds / wedged_namespace_count), replacing the retired process-global +# CasGcIsLeader / CasGcPendingReclaimEntries CurrentMetrics gauges (clobbered with >= 2 CAS disks). +# Build one named inline CA disk, run a synchronous GC round so this process has led at least once, +# then assert the column shapes on the healthy single-disk fixture. +# +# This is a .sh test (not .sql) because `SYSTEM CAS GC RUN` now returns a +# one-row-per-disk result set (UX pass); the round below only cares about its side effect (leading +# once), so its own output is redirected to /dev/null. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP TABLE IF EXISTS t_cas_mounts_gc_health; + +CREATE TABLE t_cas_mounts_gc_health (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05010', + name = '05010_cas_mounts_gc_health', + path = '05010_cas_mounts_gc_health_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 1), + old_parts_lifetime = 1; + +INSERT INTO t_cas_mounts_gc_health SELECT number, toString(number) FROM numbers(100); +TRUNCATE TABLE t_cas_mounts_gc_health; +""" + +${CLICKHOUSE_CLIENT} -q "SYSTEM CAS GC RUN '05010_cas_mounts_gc_health'" > /dev/null + +${CLICKHOUSE_CLIENT} --multiline -q """ +SELECT is_leader, wedged_namespace_count +FROM system.cas_mounts +WHERE disk LIKE '%05010_cas_mounts_gc_health%'; + +SELECT pending_reclaim >= 0, last_success_age_seconds < 60 +FROM system.cas_mounts +WHERE disk LIKE '%05010_cas_mounts_gc_health%'; + +DROP TABLE t_cas_mounts_gc_health; +SELECT 'ok'; +""" diff --git a/tests/queries/0_stateless/05011_cas_gc_rebuild_access.reference b/tests/queries/0_stateless/05011_cas_gc_rebuild_access.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/05011_cas_gc_rebuild_access.sh b/tests/queries/0_stateless/05011_cas_gc_rebuild_access.sh new file mode 100755 index 000000000000..1c530430bed3 --- /dev/null +++ b/tests/queries/0_stateless/05011_cas_gc_rebuild_access.sh @@ -0,0 +1,66 @@ +#!/usr/bin/env bash +# Tags: no-parallel + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Intent (E1): +# 1) A role granted only "SYSTEM CAS GC RUN" is REFUSED (ACCESS_DENIED) +# when it runs "SYSTEM CAS GC REBUILD ", but ALLOWED to run the per-round +# "SYSTEM CAS GC RUN". Granting the new +# "SYSTEM CAS GC REBUILD" right then permits REBUILD. +# 2) "SYSTEM CAS GC REBUILD" with NO disk is a SYNTAX_ERROR (required disk); +# naming a non-content-addressed disk yields BAD_ARGUMENTS (not a silent all-disks fan-out). +# 3) A user with ZERO grants gets ACCESS_DENIED on the plain +# "SYSTEM CAS GC RUN 'no_such_disk'" -- the privilege check runs +# before disk resolution, so denial fires even though the named disk does not exist (it would +# otherwise be UNKNOWN_DISK). +# (No CA disk needs to exist: the privilege check and the grammar/required-disk check both fire +# before any disk I/O; assert on the specific error codes. The `default` disk always exists and is +# never content-addressed, so it deterministically yields BAD_ARGUMENTS once a check is passed.) + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05011; +CREATE USER user_test_05011 IDENTIFIED WITH plaintext_password BY 'user_test_05011'; +REVOKE ALL ON *.* FROM user_test_05011; +GRANT SYSTEM CAS GC RUN ON *.* TO user_test_05011; +""" + +# GC-only role: REBUILD is refused; the per-round GC is allowed (fails later, on the disk-type check). +${CLICKHOUSE_CLIENT} --multiline --user user_test_05011 --password user_test_05011 -q """ +SYSTEM CAS GC REBUILD default; -- { serverError ACCESS_DENIED } +SYSTEM CAS GC RUN default; -- { serverError BAD_ARGUMENTS } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +GRANT SYSTEM CAS GC REBUILD ON *.* TO user_test_05011; +""" + +# Granting the new right permits REBUILD (fails later, on the disk-type check). +${CLICKHOUSE_CLIENT} --multiline --user user_test_05011 --password user_test_05011 -q """ +SYSTEM CAS GC REBUILD default; -- { serverError BAD_ARGUMENTS } +""" + +# REBUILD requires an explicit disk (syntax error), and never silently fans out across all disks. +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM CAS GC REBUILD; -- { clientError SYNTAX_ERROR } +SYSTEM CAS GC REBUILD default; -- { serverError BAD_ARGUMENTS } +""" + +# A zero-grant user is denied before the disk is even resolved: naming a disk that does not exist +# still yields ACCESS_DENIED, not UNKNOWN_DISK. +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05011_zero_grants; +CREATE USER user_test_05011_zero_grants IDENTIFIED WITH plaintext_password BY 'user_test_05011_zero_grants'; +REVOKE ALL ON *.* FROM user_test_05011_zero_grants; +""" + +${CLICKHOUSE_CLIENT} --multiline --user user_test_05011_zero_grants --password user_test_05011_zero_grants -q """ +SYSTEM CAS GC RUN 'no_such_disk'; -- { serverError ACCESS_DENIED } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05011; +DROP USER IF EXISTS user_test_05011_zero_grants; +""" diff --git a/tests/queries/0_stateless/05012_cas_mounts_typed_columns.reference b/tests/queries/0_stateless/05012_cas_mounts_typed_columns.reference new file mode 100644 index 000000000000..87a5f014d9e5 --- /dev/null +++ b/tests/queries/0_stateless/05012_cas_mounts_typed_columns.reference @@ -0,0 +1,3 @@ +DateTime64(3) +UUID +DateTime64(3) diff --git a/tests/queries/0_stateless/05012_cas_mounts_typed_columns.sql b/tests/queries/0_stateless/05012_cas_mounts_typed_columns.sql new file mode 100644 index 000000000000..5b050154c846 --- /dev/null +++ b/tests/queries/0_stateless/05012_cas_mounts_typed_columns.sql @@ -0,0 +1,8 @@ +-- E3: system.cas_mounts exposes typed columns for the lease identity/timing fields +-- (server_uuid as UUID, started_at/expires_at as DateTime64(3)) instead of raw String/UInt64. +-- No CA disk needs to be mounted -- the table's ColumnsDescription is static. + +SELECT type FROM system.columns +WHERE database = 'system' AND table = 'cas_mounts' + AND name IN ('server_uuid', 'started_at', 'expires_at') +ORDER BY name; diff --git a/tests/queries/0_stateless/05013_system_cas_drop_pool_member.reference b/tests/queries/0_stateless/05013_system_cas_drop_pool_member.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/05013_system_cas_drop_pool_member.sql b/tests/queries/0_stateless/05013_system_cas_drop_pool_member.sql new file mode 100644 index 000000000000..e4732c06a050 --- /dev/null +++ b/tests/queries/0_stateless/05013_system_cas_drop_pool_member.sql @@ -0,0 +1,2 @@ +-- Grammar + dispatch only: execution needs a CA disk (covered by the integration test). +SYSTEM CAS DROP POOL MEMBER 'srv1' FROM DISK 'no_such_disk'; -- { serverError UNKNOWN_DISK } diff --git a/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.reference b/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.reference new file mode 100644 index 000000000000..d00491fd7e5b --- /dev/null +++ b/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.reference @@ -0,0 +1 @@ +1 diff --git a/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.sql b/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.sql new file mode 100644 index 000000000000..8188ac59c42a --- /dev/null +++ b/tests/queries/0_stateless/05014_insert_dedup_disk_commit_failpoint.sql @@ -0,0 +1,28 @@ +-- Tags: zookeeper, no-fasttest, no-parallel +-- no-fasttest: needs an object-storage disk (storage_policy 's3_cache'). +-- no-parallel: enables a server-global failpoint on the part disk-transaction commit. + +DROP TABLE IF EXISTS t_dedup_disk_commit SYNC; + +CREATE TABLE t_dedup_disk_commit (k UInt64, v String) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/t_dedup_disk_commit', 'r1') +ORDER BY k +SETTINGS storage_policy = 's3_cache'; + +SYSTEM ENABLE FAILPOINT part_storage_fail_commit_transaction; + +-- The disk-storage commit of the inserted part fails. The part must NOT be registered in Keeper: +-- before the fix the disk commit ran only in MergeTreeData::Transaction::commit, AFTER the Keeper +-- multi had durably created the block_id dedup znode, so this failure left a phantom dedup token. +INSERT INTO t_dedup_disk_commit SETTINGS insert_deduplicate = 1, insert_keeper_fault_injection_probability = 0 VALUES (1, 'x'); -- { serverError FAULT_INJECTED } + +SYSTEM DISABLE FAILPOINT part_storage_fail_commit_transaction; + +-- Byte-identical retry of the failed INSERT: it must really insert. Before the fix it silently +-- deduplicated against the phantom block_id ("already exists ... ignoring it") and was acked with +-- zero rows written — the acked-then-lost data loss. +INSERT INTO t_dedup_disk_commit SETTINGS insert_deduplicate = 1, insert_keeper_fault_injection_probability = 0 VALUES (1, 'x'); + +SELECT count() FROM t_dedup_disk_commit; + +DROP TABLE t_dedup_disk_commit SYNC; diff --git a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference new file mode 100644 index 000000000000..b261da18d51a --- /dev/null +++ b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference @@ -0,0 +1,2 @@ +1 +0 diff --git a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh new file mode 100755 index 000000000000..7496a8519800 --- /dev/null +++ b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# An explicit `use_fake_transaction=1` on a `cas` disk would silently break the +# atomic manifest/ref publish (per-file autocommit, no commit point for the transaction). The disk +# factory must reject it at CREATE TABLE time with BAD_ARGUMENTS instead of silently corrupting +# writes later -- mirrors the existing missing-`server_root_id` fail-close handling. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +${CLICKHOUSE_CLIENT} -q " +DROP TABLE IF EXISTS t_cas_reject_fake_transaction; +CREATE TABLE t_cas_reject_fake_transaction (a UInt64, s String) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05015', + name = '05015_cas_reject_fake_transaction', + path = '05015_cas_reject_fake_transaction_pool/', + use_fake_transaction = 1); +" 2>&1 | grep -cm1 "use_fake_transaction. cannot be enabled for metadata type" + +${CLICKHOUSE_CLIENT} -q "SELECT count() FROM system.tables WHERE name = 't_cas_reject_fake_transaction'" diff --git a/tests/queries/0_stateless/05016_cas_drop_pool_member_access.reference b/tests/queries/0_stateless/05016_cas_drop_pool_member_access.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/05016_cas_drop_pool_member_access.sh b/tests/queries/0_stateless/05016_cas_drop_pool_member_access.sh new file mode 100755 index 000000000000..d5855c7dba48 --- /dev/null +++ b/tests/queries/0_stateless/05016_cas_drop_pool_member_access.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +# Tags: no-parallel + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Intent: `SYSTEM CAS DROP POOL MEMBER` checks access BEFORE resolving the disk (same +# pattern as the GC/GC REBUILD verbs covered by 05011_cas_gc_rebuild_access.sh), so this needs no +# CA disk at all: +# 1) A user with ZERO grants is refused with ACCESS_DENIED, even though the named disk does not +# exist (it would otherwise be UNKNOWN_DISK once past the access check). +# 2) After granting "SYSTEM CAS DROP POOL MEMBER", the same query passes the access +# check and fails later with UNKNOWN_DISK -- proving that grant, and only that grant, is what +# unlocks the verb. + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05016; +CREATE USER user_test_05016 IDENTIFIED WITH plaintext_password BY 'user_test_05016'; +REVOKE ALL ON *.* FROM user_test_05016; +""" + +${CLICKHOUSE_CLIENT} --multiline --user user_test_05016 --password user_test_05016 -q """ +SYSTEM CAS DROP POOL MEMBER 'x' FROM DISK 'y'; -- { serverError ACCESS_DENIED } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +GRANT SYSTEM CAS DROP POOL MEMBER ON *.* TO user_test_05016; +""" + +${CLICKHOUSE_CLIENT} --multiline --user user_test_05016 --password user_test_05016 -q """ +SYSTEM CAS DROP POOL MEMBER 'x' FROM DISK 'y'; -- { serverError UNKNOWN_DISK } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05016; +""" diff --git a/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.reference b/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.reference new file mode 100644 index 000000000000..29b63ba2b7be --- /dev/null +++ b/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.reference @@ -0,0 +1,2 @@ +TableProxy +1 diff --git a/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.sh b/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.sh new file mode 100755 index 000000000000..e60e86d28f41 --- /dev/null +++ b/tests/queries/0_stateless/05017_lazy_load_tables_sync_replica.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Tags: zookeeper, no-replicated-database +# no-replicated-database: the test creates its own Atomic database with `lazy_load_tables = 1` +# and an explicit ReplicatedMergeTree ZooKeeper path, which would conflict with the DDL +# replication mechanism of DatabaseReplicated. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Regression test: a table in a database created with `lazy_load_tables = 1` stays wrapped in a +# `StorageTableProxy` until first access. Before this fix, `SYSTEM SYNC REPLICA` on such a table +# failed with `BAD_ARGUMENTS: Table ... is not replicated`, because the interpreter cast the proxy +# directly to `StorageReplicatedMergeTree` instead of materializing it first. + +LAZY_DB="${CLICKHOUSE_DATABASE}_lazy" + +${CLICKHOUSE_CLIENT} -q "DROP DATABASE IF EXISTS ${LAZY_DB}" +${CLICKHOUSE_CLIENT} -q "CREATE DATABASE ${LAZY_DB} ENGINE = Atomic SETTINGS lazy_load_tables = 1" + +${CLICKHOUSE_CLIENT} -q " + CREATE TABLE ${LAZY_DB}.t (a UInt64) + ENGINE = ReplicatedMergeTree('/clickhouse/tables/${CLICKHOUSE_TEST_ZOOKEEPER_PREFIX}/lazy_sync_replica', 'r1') + ORDER BY a +" +${CLICKHOUSE_CLIENT} -q "INSERT INTO ${LAZY_DB}.t VALUES (1)" + +${CLICKHOUSE_CLIENT} -q "DETACH DATABASE ${LAZY_DB}" +${CLICKHOUSE_CLIENT} -q "ATTACH DATABASE ${LAZY_DB}" + +# Confirm the table is still an unmaterialized proxy at this point. +${CLICKHOUSE_CLIENT} -q "SELECT engine FROM system.tables WHERE database = '${LAZY_DB}' AND name = 't'" + +# This must succeed without first touching the table, i.e. without materializing the proxy +# through any other path. +${CLICKHOUSE_CLIENT} -q "SYSTEM SYNC REPLICA ${LAZY_DB}.t" + +${CLICKHOUSE_CLIENT} -q "SELECT count() FROM ${LAZY_DB}.t" + +${CLICKHOUSE_CLIENT} -q "DROP DATABASE ${LAZY_DB}" diff --git a/tests/queries/0_stateless/05019_cas_fsck_access.reference b/tests/queries/0_stateless/05019_cas_fsck_access.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/05019_cas_fsck_access.sh b/tests/queries/0_stateless/05019_cas_fsck_access.sh new file mode 100755 index 000000000000..1f3a08a69dcb --- /dev/null +++ b/tests/queries/0_stateless/05019_cas_fsck_access.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Tags: no-parallel + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Access control for `SYSTEM CAS FSCK` (mirrors 05011_cas_gc_rebuild_access.sh): +# 1) A zero-grant user is denied before the disk is even resolved -- naming a disk that does not +# exist still yields ACCESS_DENIED, not UNKNOWN_DISK. +# 2) Granting "SYSTEM CAS FSCK" permits the verb; it then fails later, on the +# disk-type check (the `default` disk always exists and is never content-addressed, so the +# query deterministically fails with BAD_ARGUMENTS instead). +# (The UNMOUNT/MOUNT siblings this file once covered were removed with the Dormant lifecycle, +# spec rev.8 §9; FORGET / GC STOP / GC START access coverage is tracked for the acceptance task.) + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05019; +CREATE USER user_test_05019 IDENTIFIED WITH plaintext_password BY 'user_test_05019'; +REVOKE ALL ON *.* FROM user_test_05019; +""" + +# Zero grants: denied before the disk is resolved. +${CLICKHOUSE_CLIENT} --multiline --user user_test_05019 --password user_test_05019 -q """ +SYSTEM CAS FSCK 'no_such_disk'; -- { serverError ACCESS_DENIED } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +GRANT SYSTEM CAS FSCK ON *.* TO user_test_05019; +""" + +# Granting the FSCK right permits it (fails later, on the disk-type check). +${CLICKHOUSE_CLIENT} --multiline --user user_test_05019 --password user_test_05019 -q """ +SYSTEM CAS FSCK default; -- { serverError BAD_ARGUMENTS } +""" + +# The verb requires an explicit disk (syntax error). +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM CAS FSCK; -- { clientError SYNTAX_ERROR } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS user_test_05019; +""" diff --git a/tests/queries/0_stateless/05020_cas_fsck.reference b/tests/queries/0_stateless/05020_cas_fsck.reference new file mode 100644 index 000000000000..2f34f4455804 --- /dev/null +++ b/tests/queries/0_stateless/05020_cas_fsck.reference @@ -0,0 +1,7 @@ +3 +0 0 0 +disk reachable dangling unreachable pending_gc awaiting_gc unaccounted stale_edge corrupted_runs chain_broken unchecked lifeless_keys namespace_janitor_pending namespace_janitor_pending_bytes namespace_janitor_pending_lives ref_records_walked physical_bytes referenced_logical_bytes distinct_blobs total_blob_refs + 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 +fsck_non_ca_disk_rejected: 1 +fsck_requires_disk: 1 +second_forget_idempotent: vanished(forgotten) diff --git a/tests/queries/0_stateless/05020_cas_fsck.sh b/tests/queries/0_stateless/05020_cas_fsck.sh new file mode 100755 index 000000000000..a10ca4e33762 --- /dev/null +++ b/tests/queries/0_stateless/05020_cas_fsck.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# `SYSTEM CAS FSCK ` (runs on a RUNNING disk, T13) + GC RUN's `pending_*` drain +# columns + the fail-closed FORGET teardown (spec rev.8 §5/§9). FSCK is a read-only reachability audit +# that now runs directly on the mounted, live disk and prints a clean one-row summary. The GC RUN result +# set carries the retire pipeline's REMAINING (not this-round-delta) `pending_*` columns; on a disk with +# nothing outstanding to reclaim they read 0. Teardown is fail-closed: DROP the table, `SYSTEM CONTENT +# ADDRESSED FORGET` the disk (force-Vanish, node-local), verify via system.cas_mounts that +# it reads exactly `vanished(forgotten)`, and only THEN `rm -rf` the pool dir (FORGET stopped and joined +# every CAS background thread for this disk). A failed FORGET or an unexpected lifecycle aborts the test +# with the pool dir left in place (the scripts have no `set -e`, so the checks are explicit). + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK_NAME="ca_fsck_${CLICKHOUSE_TEST_UNIQUE_NAME}_${RANDOM}" +POOL_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}_fsck_${RANDOM}" +rm -rf "${POOL_DIR:?}" +mkdir -p "${POOL_DIR}" +DISK_CA="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05020', + name = '${DISK_NAME}', + path = '${POOL_DIR}/')" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_fsck SYNC" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE t_fsck (id UInt64) ENGINE = MergeTree ORDER BY id +SETTINGS disk = ${DISK_CA}" + +# --- GC RUN's result set carries the new pending_* columns while the disk is mounted, and they read 0 +# on this fresh pool (nothing was ever written, so nothing was ever condemned) --- +${CLICKHOUSE_CLIENT} --format TSVWithNames --query "SYSTEM CAS GC RUN '${DISK_NAME}'" \ + | tr '\t' '\n' | grep -c "pending_candidates\|pending_condemned\|pending_retired" +${CLICKHOUSE_CLIENT} --format TSV --query "SYSTEM CAS GC RUN '${DISK_NAME}'" \ + | awk -F'\t' '{print $(NF-2), $(NF-1), $NF}' + +# --- FSCK on the RUNNING, healthy pool (T13: FSCK runs on a mounted disk): a clean one-row summary, +# no dangling/unreachable --- +${CLICKHOUSE_CLIENT} --format TSVWithNames --query "SYSTEM CAS FSCK '${DISK_NAME}'" \ + | sed "s/${DISK_NAME}//" + +# --- A non-CA disk is rejected (the always-present local \`default\`) --- +echo -n 'fsck_non_ca_disk_rejected: ' +${CLICKHOUSE_CLIENT} --query "SYSTEM CAS FSCK default" 2>&1 \ + | grep -cm1 "is not a content-addressed disk" + +# --- FSCK requires an explicit disk (syntax error) --- +echo -n 'fsck_requires_disk: ' +${CLICKHOUSE_CLIENT} --query "SYSTEM CAS FSCK" 2>&1 \ + | grep -cm1 "Syntax error" + +# --- Fail-closed teardown (spec rev.8 §5/§9): DROP the table, FORGET the disk (force-Vanish, node-local), +# verify it reads exactly `vanished(forgotten)`, and only then rm. A failed FORGET or an unexpected +# lifecycle aborts here, leaving the pool dir in place. --- +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_fsck SYNC" +# FORGET logs an operator WARNING (the decommission is deliberately prominent in the server log); the +# clickhouse-test harness runs the client at --send_logs_level=warning, which would stream that expected +# warning to stderr and be flagged as a failure. Suppress it on the client for the FORGET call only. +${CLICKHOUSE_CLIENT} --allow_repeated_settings --send_logs_level=fatal \ + --query "SYSTEM CAS FORGET '${DISK_NAME}'" || { + echo "FORGET failed — leaving pool dir in place (fail-closed)"; exit 1; } +LIFECYCLE=$(${CLICKHOUSE_CLIENT} --query " + SELECT lifecycle || '(' || lifecycle_reason || ')' FROM system.cas_mounts + WHERE disk = '${DISK_NAME}'") +[ "${LIFECYCLE}" = "vanished(forgotten)" ] || { + echo "unexpected lifecycle after FORGET: ${LIFECYCLE}"; exit 1; } + +# --- A second FORGET is idempotent: it succeeds and the disk stays `vanished(forgotten)` (an already +# terminal Vanished pool is the terminal truth — nothing to force, nothing to double-retire). --- +${CLICKHOUSE_CLIENT} --allow_repeated_settings --send_logs_level=fatal \ + --query "SYSTEM CAS FORGET '${DISK_NAME}'" || { + echo "second FORGET failed — leaving pool dir in place (fail-closed)"; exit 1; } +LIFECYCLE_AGAIN=$(${CLICKHOUSE_CLIENT} --query " + SELECT lifecycle || '(' || lifecycle_reason || ')' FROM system.cas_mounts + WHERE disk = '${DISK_NAME}'") +echo "second_forget_idempotent: ${LIFECYCLE_AGAIN}" + +rm -rf "${POOL_DIR:?}" # safe: FORGET stopped and joined every CAS thread for this disk diff --git a/tests/queries/0_stateless/05021_lazy_load_tables_mutations.reference b/tests/queries/0_stateless/05021_lazy_load_tables_mutations.reference new file mode 100644 index 000000000000..8482a9714e8f --- /dev/null +++ b/tests/queries/0_stateless/05021_lazy_load_tables_mutations.reference @@ -0,0 +1,2 @@ +1 11 +2 20 diff --git a/tests/queries/0_stateless/05021_lazy_load_tables_mutations.sh b/tests/queries/0_stateless/05021_lazy_load_tables_mutations.sh new file mode 100755 index 000000000000..c43684319c66 --- /dev/null +++ b/tests/queries/0_stateless/05021_lazy_load_tables_mutations.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Tags: zookeeper, no-replicated-database +# no-replicated-database: the test creates its own Atomic database with `lazy_load_tables = 1` +# and an explicit ReplicatedMergeTree ZooKeeper path, which would conflict with the DDL +# replication mechanism of DatabaseReplicated. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Regression test: a table in a database created with `lazy_load_tables = 1` stays wrapped in a +# `StorageTableProxy` until first access. Before this fix, any mutation (`ALTER ... UPDATE`, +# `MATERIALIZE TTL`, ...) on such a table failed with NOT_IMPLEMENTED "Table engine +# ReplicatedMergeTree doesn't support mutations": `StorageProxy` forwarded `mutate` but not +# `checkMutationIsPossible`, so `IStorage`'s throwing default fired with the nested engine's name. + +LAZY_DB="${CLICKHOUSE_DATABASE}_lazy" + +${CLICKHOUSE_CLIENT} -q "DROP DATABASE IF EXISTS ${LAZY_DB}" +${CLICKHOUSE_CLIENT} -q "CREATE DATABASE ${LAZY_DB} ENGINE = Atomic SETTINGS lazy_load_tables = 1" + +${CLICKHOUSE_CLIENT} -q " + CREATE TABLE ${LAZY_DB}.t (a UInt64, b UInt64) + ENGINE = ReplicatedMergeTree('/clickhouse/tables/${CLICKHOUSE_TEST_ZOOKEEPER_PREFIX}/lazy_mutations', 'r1') + ORDER BY a +" +${CLICKHOUSE_CLIENT} -q "INSERT INTO ${LAZY_DB}.t (a, b) VALUES (1, 10), (2, 20)" + +# DETACH + ATTACH the database so the table goes back to an unmaterialized proxy: the INSERT above +# has already materialized it once, and the point is to mutate through the fresh proxy. +${CLICKHOUSE_CLIENT} -q "DETACH DATABASE ${LAZY_DB}" +${CLICKHOUSE_CLIENT} -q "ATTACH DATABASE ${LAZY_DB}" + +${CLICKHOUSE_CLIENT} -q "ALTER TABLE ${LAZY_DB}.t UPDATE b = b + 1 WHERE a = 1 SETTINGS mutations_sync = 2" +# NOTE: MATERIALIZE TTL through a lazy proxy is still broken differently (the proxy's cached +# in-memory metadata carries columns only, no TTL -- see the StorageProxy forwarding audit report); +# this test deliberately pins only what the checkMutationIsPossible forward fixes. + +${CLICKHOUSE_CLIENT} -q "SELECT a, b FROM ${LAZY_DB}.t ORDER BY a" + +${CLICKHOUSE_CLIENT} -q "DROP DATABASE ${LAZY_DB}" diff --git a/tests/queries/0_stateless/05022_cas_verb_access.reference b/tests/queries/0_stateless/05022_cas_verb_access.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/05022_cas_verb_access.sh b/tests/queries/0_stateless/05022_cas_verb_access.sh new file mode 100755 index 000000000000..a1a750c724fe --- /dev/null +++ b/tests/queries/0_stateless/05022_cas_verb_access.sh @@ -0,0 +1,69 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Access control for the three lifecycle verbs added in the rev.8 disk-lifecycle round -- SYSTEM CONTENT +# ADDRESSED FORGET / GC STOP / GC START -- mirroring 05019_cas_fsck_access.sh. For each verb: +# 1) A zero-grant user is denied BEFORE the disk is resolved -- naming a disk that does not exist still +# yields ACCESS_DENIED (the access check runs ahead of getDisk), not UNKNOWN_DISK. +# 2) Granting the matching right permits the verb; it then fails later on the disk-type check (the always +# -present `default` disk exists and is never content-addressed, so it deterministically fails with +# BAD_ARGUMENTS without any lifecycle side effect). +# 3) The verb requires an explicit disk (all three route through the target-required parser like FSCK, so +# omitting the disk is a client-side SYNTAX_ERROR, not a silent fan-out). +# A unique user name keeps this parallel-safe (no global fixed-name object), and every verb run targets only +# `no_such_disk`/`default`, so nothing is ever actually decommissioned or reconfigured. + +USER="user_test_${CLICKHOUSE_TEST_UNIQUE_NAME}" + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS ${USER}; +CREATE USER ${USER} IDENTIFIED WITH plaintext_password BY 'pw'; +REVOKE ALL ON *.* FROM ${USER}; +""" + +# (1) Zero grants: each verb is denied before the disk is resolved. +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS FORGET 'no_such_disk'; -- { serverError ACCESS_DENIED } +""" +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS GC STOP 'no_such_disk'; -- { serverError ACCESS_DENIED } +""" +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS GC START 'no_such_disk'; -- { serverError ACCESS_DENIED } +""" + +# Grant each verb its matching right. +${CLICKHOUSE_CLIENT} --multiline -q """ +GRANT SYSTEM CAS FORGET ON *.* TO ${USER}; +GRANT SYSTEM CAS GC STOP ON *.* TO ${USER}; +GRANT SYSTEM CAS GC START ON *.* TO ${USER}; +""" + +# (2) Granted: the verb is permitted, then fails later on the disk-type check against `default`. +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS FORGET default; -- { serverError BAD_ARGUMENTS } +""" +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS GC STOP default; -- { serverError BAD_ARGUMENTS } +""" +${CLICKHOUSE_CLIENT} --multiline --user "${USER}" --password pw -q """ +SYSTEM CAS GC START default; -- { serverError BAD_ARGUMENTS } +""" + +# (3) Each verb requires an explicit disk (syntax error). +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM CAS FORGET; -- { clientError SYNTAX_ERROR } +""" +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM CAS GC STOP; -- { clientError SYNTAX_ERROR } +""" +${CLICKHOUSE_CLIENT} --multiline -q """ +SYSTEM CAS GC START; -- { clientError SYNTAX_ERROR } +""" + +${CLICKHOUSE_CLIENT} --multiline -q """ +DROP USER IF EXISTS ${USER}; +""" diff --git a/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.reference b/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.reference new file mode 100644 index 000000000000..acfbe0e078ed --- /dev/null +++ b/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.reference @@ -0,0 +1,14 @@ +empty_table_state_before_drop live +empty_table_has_format_version 1 +empty_table_has_no_ref_stream_before_drop 1 +empty_table_state_after_sync_drop removing +empty_table_terminal_stream_record_exists 1 +one_part_table_state_before_drop live +one_part_table_state_after_sync_drop removing +negative_control_state_after_truncate live +negative_control_state_after_reinsert live +negative_control_rows 1 +cycle_leaks_after_sync_drop 0 +captured_rows_absent_after_gc_fixpoint 1 +fsck_unreachable 0 +fsck_dangling 0 diff --git a/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.sh b/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.sh new file mode 100755 index 000000000000..ef5e2377cc21 --- /dev/null +++ b/tests/queries/0_stateless/05023_cas_dropns_leaked_namespace.sh @@ -0,0 +1,196 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type (keep it off the minimal fasttest image); this test uses its +# own unique local-object-storage pool and a per-run CAS disk name, so unlike 04290_cas_no_leftovers +# it does not need no-parallel. + +# FINDING #2 regression test: `DROP TABLE ... SYNC` on a content-addressed MergeTree used to leave the +# table's CAS ref-catalog row `live` forever whenever `DirShape::TableDir`'s `existsDirectory` observed +# zero committed refs -- an empty table, or one whose last part was just removed. `dropAllData`'s own +# `existsDirectory` precheck skipped `removeRecursive`/`dropNamespace` entirely in that shape, so the +# SQL-level drop completed normally while the CAS catalog row leaked, one per create/drop cycle. +# +# The primary oracle is the pool's OWN plain-text `cas/ref_catalog` object, read directly off disk: the +# exact `st` (lifecycle) field recorded for the table's logical namespace. `SYSTEM CAS FSCK`'s +# unreachable/dangling counts are a secondary check only -- fsck correctly regards a `live` leak as +# CONSISTENT (nothing is unreachable; the row simply never dies), so it cannot detect this defect on its +# own; `04290_cas_no_leftovers.sh`'s fsck-only oracle is exactly why FINDING #2 shipped unnoticed. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +POOL_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}_05023_${RANDOM}" +DISK_NAME="ca_05023_${CLICKHOUSE_TEST_UNIQUE_NAME}_${RANDOM}" +SERVER_ROOT_ID="dropns05023" + +rm -rf "${POOL_DIR:?}" +mkdir -p "${POOL_DIR}" + +CATALOG_FILE="${POOL_DIR}/ca/cas/ref_catalog" + +# The pool's own plain-text catalog line for namespace $1, or empty if the namespace has no row at all. +catalog_line() { + grep -F "\"ns\":\"$1\"" "${CATALOG_FILE}" 2>/dev/null || true +} + +# The `st` (lifecycle) word recorded for namespace $1: "live"/"creating"/"removing", or "absent" if the +# namespace has no catalog row (matches `04290`'s field-by-name discipline: never assume a position). +catalog_state() { + local line + line=$(catalog_line "$1") + if [ -z "${line}" ]; then + echo "absent" + return + fi + echo "${line}" | grep -o '"st":"[a-z]*"' | head -1 | sed -E 's/"st":"([a-z]*)"/\1/' +} + +# ClickHouse's own store// fanout with the CAS archive boundary marker, exactly as +# `Cas::mirroredArchiveNamespace` builds it -- see PartPathParser.cpp. `$1` is the table's UUID. +namespace_of() { + local uuid="$1" + echo "${SERVER_ROOT_ID}/store/${uuid:0:3}/${uuid}@cas@" +} + +DISK_DEF="disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '${SERVER_ROOT_ID}', + name = '${DISK_NAME}', + path = '${POOL_DIR}/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 100000)" +# ^ gc_enabled=1 so `SYSTEM CAS GC RUN` is available; the interval is long enough that no background +# round can fire during the test's own window, so the post-drop catalog state read directly below is +# stable -- only the manual `GC RUN` loop at the end may advance it. + +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_dropns_empty SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_dropns_one_part SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_dropns_negative_control SYNC" +$CLICKHOUSE_CLIENT --query "DROP TABLE IF EXISTS t_dropns_cycle SYNC" + +# ---- (1) an EMPTY table: zero parts, zero namespace files beyond format_version.txt ---- +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_dropns_empty (a UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = ${DISK_DEF}" + +EMPTY_UUID=$($CLICKHOUSE_CLIENT --query "SELECT uuid FROM system.tables WHERE database = currentDatabase() AND name = 't_dropns_empty'") +EMPTY_NS=$(namespace_of "${EMPTY_UUID}") + +echo "empty_table_state_before_drop $(catalog_state "${EMPTY_NS}")" + +# Its only payload is the namespace-level format_version.txt: no ref stream object anywhere yet (a +# files-only life never touched by a ref op has no `_log`/`_snap` at all). +FORMAT_VERSION_HITS=$(find "${POOL_DIR}/ca/cas/ns/state" -path '*_files/format_version.txt' 2>/dev/null | wc -l) +STREAM_HITS_BEFORE=$(find "${POOL_DIR}/ca/cas/ns/stream" -type f 2>/dev/null | wc -l) +echo "empty_table_has_format_version $([ "${FORMAT_VERSION_HITS}" -ge 1 ] && echo 1 || echo 0)" +echo "empty_table_has_no_ref_stream_before_drop $([ "${STREAM_HITS_BEFORE}" -eq 0 ] && echo 1 || echo 0)" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_dropns_empty SYNC" + +# The current branch fails here by leaving st:"live"; the fix must show "removing" (a terminal stream +# record now exists but the catalog row itself is not deleted until GC folds and reclaims it). +echo "empty_table_state_after_sync_drop $(catalog_state "${EMPTY_NS}")" +STREAM_HITS_AFTER=$(find "${POOL_DIR}/ca/cas/ns/stream" -type f 2>/dev/null | wc -l) +echo "empty_table_terminal_stream_record_exists $([ "${STREAM_HITS_AFTER}" -ge 1 ] && echo 1 || echo 0)" + +# ---- (2) same shape, but with one committed part: "parts removed first, files remain" is a separate +# path from the zero-part path above; pin it too. ---- +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_dropns_one_part (a UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = ${DISK_DEF}" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_dropns_one_part VALUES (1)" + +ONE_PART_UUID=$($CLICKHOUSE_CLIENT --query "SELECT uuid FROM system.tables WHERE database = currentDatabase() AND name = 't_dropns_one_part'") +ONE_PART_NS=$(namespace_of "${ONE_PART_UUID}") +echo "one_part_table_state_before_drop $(catalog_state "${ONE_PART_NS}")" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_dropns_one_part SYNC" +echo "one_part_table_state_after_sync_drop $(catalog_state "${ONE_PART_NS}")" + +# ---- (3) negative control: removing the ONLY part while keeping the table must never admit removal. ---- +$CLICKHOUSE_CLIENT --query " +CREATE TABLE t_dropns_negative_control (a UInt64) +ENGINE = MergeTree ORDER BY a +SETTINGS disk = ${DISK_DEF}" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_dropns_negative_control VALUES (1)" + +NEGATIVE_UUID=$($CLICKHOUSE_CLIENT --query "SELECT uuid FROM system.tables WHERE database = currentDatabase() AND name = 't_dropns_negative_control'") +NEGATIVE_NS=$(namespace_of "${NEGATIVE_UUID}") + +$CLICKHOUSE_CLIENT --query "TRUNCATE TABLE t_dropns_negative_control" +echo "negative_control_state_after_truncate $(catalog_state "${NEGATIVE_NS}")" +$CLICKHOUSE_CLIENT --query "INSERT INTO t_dropns_negative_control VALUES (2)" +echo "negative_control_state_after_reinsert $(catalog_state "${NEGATIVE_NS}")" +echo "negative_control_rows $($CLICKHOUSE_CLIENT --query "SELECT count() FROM t_dropns_negative_control")" + +$CLICKHOUSE_CLIENT --query "DROP TABLE t_dropns_negative_control SYNC" + +# ---- (4) several same-SQL-name create/drop cycles: every fresh Atomic UUID gets its OWN namespace, so +# a stale predecessor cannot mask a fresh leak, and a same-name CREATE must never wait on the old +# UUID's GC. ---- +CYCLE_NS_LIST=() +CYCLE_LEAK_COUNT=0 +for i in 1 2 3; do + $CLICKHOUSE_CLIENT --query " + CREATE TABLE t_dropns_cycle (a UInt64) + ENGINE = MergeTree ORDER BY a + SETTINGS disk = ${DISK_DEF}" + CYCLE_UUID=$($CLICKHOUSE_CLIENT --query "SELECT uuid FROM system.tables WHERE database = currentDatabase() AND name = 't_dropns_cycle'") + CYCLE_NS=$(namespace_of "${CYCLE_UUID}") + CYCLE_NS_LIST+=("${CYCLE_NS}") + + $CLICKHOUSE_CLIENT --query "DROP TABLE t_dropns_cycle SYNC" + if [ "$(catalog_state "${CYCLE_NS}")" = "live" ]; then + CYCLE_LEAK_COUNT=$((CYCLE_LEAK_COUNT + 1)) + fi +done +echo "cycle_leaks_after_sync_drop ${CYCLE_LEAK_COUNT}" + +# ---- (5) drive manual GC to a bounded fixpoint (Task 7 `pending_*` gauges, not a fixed sleep), then +# assert every captured namespace's catalog row is gone. ---- +ALL_NS=("${EMPTY_NS}" "${ONE_PART_NS}" "${CYCLE_NS_LIST[@]}") + +PENDING=1 +for _ in $(seq 1 60); do + PENDING=$($CLICKHOUSE_CLIENT --query "SYSTEM CAS GC RUN '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print $col["pending_condemned"] }') + [ "${PENDING}" = "0" ] && break + sleep 0.5 +done +if [ "${PENDING}" != "0" ]; then + echo "FAIL: GC did not drain within the bounded loop (pending=${PENDING})" >&2 + exit 1 +fi + +ROWS_STILL_PRESENT=0 +for ns in "${ALL_NS[@]}"; do + if [ "$(catalog_state "${ns}")" != "absent" ]; then + ROWS_STILL_PRESENT=$((ROWS_STILL_PRESENT + 1)) + fi +done +echo "captured_rows_absent_after_gc_fixpoint $([ "${ROWS_STILL_PRESENT}" -eq 0 ] && echo 1 || echo 0)" + +# ---- (6) SYSTEM CAS FSCK: secondary check only -- a `live` leak alone would read as CONSISTENT here, +# which is exactly why the fsck-only oracle in 04290_cas_no_leftovers.sh did not catch this defect. ---- +$CLICKHOUSE_CLIENT --query "SYSTEM CAS FSCK '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print "fsck_unreachable", $col["unreachable"]; print "fsck_dangling", $col["dangling"] }' + +# ---- (7) fail-closed teardown (spec rev.8 §5/§9): FORGET the disk (all tables already dropped above), +# verify it, only then rm. ---- +$CLICKHOUSE_CLIENT --allow_repeated_settings --send_logs_level=fatal \ + --query "SYSTEM CAS FORGET '${DISK_NAME}'" || { + echo "FORGET failed — leaving pool dir in place (fail-closed)"; exit 1; } +LIFECYCLE=$($CLICKHOUSE_CLIENT --query " + SELECT lifecycle || '(' || lifecycle_reason || ')' FROM system.cas_mounts + WHERE disk = '${DISK_NAME}'") +[ "${LIFECYCLE}" = "vanished(forgotten)" ] || { + echo "unexpected lifecycle after FORGET: ${LIFECYCLE}"; exit 1; } + +rm -rf "${POOL_DIR:?}" # safe: FORGET stopped and joined every CAS thread for this disk diff --git a/tests/queries/0_stateless/05024_cas_freeze_two_roots.reference b/tests/queries/0_stateless/05024_cas_freeze_two_roots.reference new file mode 100644 index 000000000000..0abb40bd8cc4 --- /dev/null +++ b/tests/queries/0_stateless/05024_cas_freeze_two_roots.reference @@ -0,0 +1,9 @@ +foreign_unfreeze +command_type partition_id backup_name +unfreeze_b +command_type partition_id backup_name +UNFREEZE ALL 1 own_b_05024 +unfreeze_a +command_type partition_id backup_name +UNFREEZE ALL 1 shared_05024 +dropped_ok diff --git a/tests/queries/0_stateless/05024_cas_freeze_two_roots.sh b/tests/queries/0_stateless/05024_cas_freeze_two_roots.sh new file mode 100755 index 000000000000..62f82baa28da --- /dev/null +++ b/tests/queries/0_stateless/05024_cas_freeze_two_roots.sh @@ -0,0 +1,124 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +set -euo pipefail + +# A `FREEZE` snapshot on a content-addressed disk belongs to the server root that made it. `UNFREEZE` +# is local and destructive, so releasing one root's freeze must not touch another root's. +# +# Two server roots sharing one pool is how two replicas of one table look from the pool's side. The +# destructive lookup needs ONE table path reachable from both roots, and the shadow path embeds the +# table UUID in an Atomic database -- so the UUID is reused sequentially rather than creating two +# tables, which would have two UUIDs, two namespaces, and no cross-root lookup to test. +# +# The UUID is generated per run, not a fixed literal: a repeated run in the same server would +# otherwise collide with the previous run's asynchronous cleanup (the Atomic database still holds +# the UUID mapping until the deferred drop completes, and the CAS namespace stays in its removal +# state until a terminal fold -- which never comes here, since the test keeps GC off). + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +TABLE_UUID=$(${CLICKHOUSE_CLIENT} --query "SELECT generateUUIDv4()") +SHARED_BACKUP='shared_05024' +B_BACKUP='own_b_05024' +DISK_B='05024_cas_freeze_b' +UNFREEZE_STRUCTURE='command_type String, partition_id String, part_name String, backup_name String, backup_path String, part_backup_path String' + +# `ALTER ... UNFREEZE` returns rows only under `alter_partition_verbose_result=1`; the default is off. +# `backup_path` and `part_backup_path` are absolute, so only the stable columns are printed. +unfreeze_and_print() { + ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $1 UNFREEZE WITH NAME '$2' SETTINGS alter_partition_verbose_result = 1;" \ + | ${CLICKHOUSE_LOCAL} --structure "$UNFREEZE_STRUCTURE" \ + --query "SELECT command_type, partition_id, backup_name FROM table ORDER BY partition_id FORMAT TSVWithNames" +} + +create_on_root() { + # $1 = table name, $2 = `server_root_id`, $3 = disk name. One pool, two roots. + ${CLICKHOUSE_CLIENT} --query " + CREATE TABLE $1 UUID '${TABLE_UUID}' (k UInt32, v String) + ENGINE = MergeTree ORDER BY k PARTITION BY k + SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '$2', + name = '$3', + path = '05024_cas_freeze_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 100000);" +} + +drain_gc() { + local pending=1 + for _ in $(seq 1 60); do + pending=$(${CLICKHOUSE_CLIENT} --query "SYSTEM CAS GC RUN '${DISK_B}'" --format TSVWithNames \ + | awk -F'\t' 'NR == 1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print $col["pending_condemned"] }') + [ "$pending" = "0" ] && return + sleep 0.5 + done + echo "FAIL: GC did not drain within the bounded loop (pending=${pending})" >&2 + return 1 +} + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_freeze_a;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_freeze_b;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_cas_freeze_anchor;" + +# Keep root B's disk alive for every collection round. This avoids relying on an inline disk remaining +# registered after its only table is dropped. +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE t_cas_freeze_anchor (k UInt32) +ENGINE = MergeTree ORDER BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05024_root_b', + name = '${DISK_B}', + path = '05024_cas_freeze_pool/', + cas_gc_enabled = 1, + cas_gc_interval_sec = 100000);" + +# Root A freezes, then releases the UUID. Its freeze must outlive both the table and a collection round. +create_on_root t_cas_freeze_a 05024_root_a 05024_cas_freeze_a +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_freeze_a VALUES (1, 'a');" +${CLICKHOUSE_CLIENT} --query "ALTER TABLE t_cas_freeze_a FREEZE PARTITION 1 WITH NAME '${SHARED_BACKUP}';" +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_freeze_a;" +drain_gc + +# Root B reaches the SAME table path by reusing the UUID. Its own backup uses a distinct name: making +# both roots publish the same ref would mix two independent CAS writer lifecycles before `UNFREEZE` +# gets a chance to exercise the destructive lookup under test. +create_on_root t_cas_freeze_b 05024_root_b "${DISK_B}" +${CLICKHOUSE_CLIENT} --query "INSERT INTO t_cas_freeze_b VALUES (1, 'b');" +${CLICKHOUSE_CLIENT} --query "ALTER TABLE t_cas_freeze_b FREEZE PARTITION 1 WITH NAME '${B_BACKUP}';" + +# (1) B has no `shared_05024` freeze. Its foreign `UNFREEZE` must be a no-op; pre-fix it finds and +# drops A's pool-global shadow namespace and prints A's row here. +echo 'foreign_unfreeze' +unfreeze_and_print t_cas_freeze_b "${SHARED_BACKUP}" + +# (2) B releases its OWN freeze. This catches a fix that scopes publication but leaves bulk lookup on +# the old unprefixed subtree. +echo 'unfreeze_b' +unfreeze_and_print t_cas_freeze_b "${B_BACKUP}" + +# The foreign no-op must remain harmless after the retire pipeline reaches a fixpoint. +drain_gc + +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_freeze_b;" + +# (3) A's freeze must still be there. Recreate A's table on root A with the same UUID -- the freeze is +# addressed by path, so the recreated table reaches its predecessor's snapshot -- and release it. +# Pre-fix this prints nothing, because B's foreign unfreeze above already dropped the shared namespace. +create_on_root t_cas_freeze_a 05024_root_a 05024_cas_freeze_a +echo 'unfreeze_a' +unfreeze_and_print t_cas_freeze_a "${SHARED_BACKUP}" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_freeze_a;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE t_cas_freeze_anchor;" +${CLICKHOUSE_CLIENT} --query "SELECT 'dropped_ok';" diff --git a/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.reference b/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.reference new file mode 100644 index 000000000000..113baaf7bfc4 --- /dev/null +++ b/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.reference @@ -0,0 +1,6 @@ +leg1 32 32 32 +leg1_roundtrip 32 32 +leg2 32 32 32 +leg3 32 32 32 +leg3_roundtrip 32 32 +dropped_ok diff --git a/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.sh b/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.sh new file mode 100755 index 000000000000..d10d57faceaf --- /dev/null +++ b/tests/queries/0_stateless/05025_cas_attach_partition_cross_disk.sh @@ -0,0 +1,142 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# `ATTACH PARTITION FROM` across disks clones the part through `freezeRemote`, and a +# content-addressed destination models a part as ONE atomic unit: N files, one manifest, one ref. +# Without a single transaction every file autocommits as its own one-file manifest against the same +# ref, so two of them resolve the ref as absent and the loser hits the unique-ref guard -- the very +# first attach fails. +# +# The two tables must be on DIFFERENT disks: the clone path is chosen by `on_same_disk`, and a +# same-disk attach goes through `freeze`, which already has the transaction branch. +# +# Leg 2 is the same-pool content-addressed case. Its row-level result verifies that the destination +# resolves the shared content correctly without relying on a particular physical-publication branch. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Every table this test creates, including leg 3's, so a run interrupted mid-script can be repeated. +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS src_plain;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS dst_cas;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS src_cas;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS dst_cas_same_pool;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS src_plain_repl SYNC;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS dst_cas_repl SYNC;" + +# ---------------------------------------------------------------- leg 1: local -> content-addressed + +# The local disks embed the database in their NAME, not only in the path: custom disks live for the +# whole server lifetime, and a repeated run gets a fresh database -- a fixed name with a +# database-dependent path would be a redefinition of an existing disk, which is rejected. + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE src_plain (k UInt32, v String) +ENGINE = MergeTree ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = local, + name = '${CLICKHOUSE_DATABASE}_05025_plain', + path = '${CLICKHOUSE_DISKS_FILES}/${CLICKHOUSE_DATABASE}_05025_plain/');" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE dst_cas (k UInt32, v String) +ENGINE = MergeTree ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05025_dst', + name = '05025_cas_dst', + path = '05025_cas_dst_pool/');" + +${CLICKHOUSE_CLIENT} --query "INSERT INTO src_plain SELECT number % 2, toString(number) FROM numbers(64);" + +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas ATTACH PARTITION 1 FROM src_plain;" + +# Completeness, not mere presence: a partially published part would satisfy a bare count(). +${CLICKHOUSE_CLIENT} --query "SELECT 'leg1', count(), sum(k), uniqExact(v) FROM dst_cas;" + +# A detach/attach round trip reads the part back from its manifest rather than from whatever the +# writing session still had warm. +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas DETACH PARTITION 1;" +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas ATTACH PARTITION 1;" +${CLICKHOUSE_CLIENT} --query "SELECT 'leg1_roundtrip', count(), sum(k) FROM dst_cas;" + +# ------------------------------------------- leg 2: content-addressed -> content-addressed, one pool + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE src_cas (k UInt32, v String) +ENGINE = MergeTree ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05025_shared_a', + name = '05025_cas_shared_a', + path = '05025_cas_shared_pool/');" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE dst_cas_same_pool (k UInt32, v String) +ENGINE = MergeTree ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05025_shared_b', + name = '05025_cas_shared_b', + path = '05025_cas_shared_pool/');" + +${CLICKHOUSE_CLIENT} --query "INSERT INTO src_cas SELECT number % 2, toString(number) FROM numbers(64);" + +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas_same_pool ATTACH PARTITION 1 FROM src_cas;" + +${CLICKHOUSE_CLIENT} --query "SELECT 'leg2', count(), sum(k), uniqExact(v) FROM dst_cas_same_pool;" + +# ------------------------------------------------ leg 3: replicated, local -> content-addressed + +# The replicated ATTACH is a DIFFERENT shape and reaches the same function: of the replicated clone +# sites only the ATTACH branch passes `must_on_same_disk=false`, and its clone params set +# `metadata_version_to_write`, so after the transaction commits the caller writes +# `metadata_version.txt` separately -- a repoint of an already-published part rather than a file +# inside the clone. REPLACE on a replicated table is same-disk only and cannot reach this path. + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE src_plain_repl (k UInt32, v String) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/05025_src_repl', 'r1') +ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = local, + name = '${CLICKHOUSE_DATABASE}_05025_plain_repl', + path = '${CLICKHOUSE_DISKS_FILES}/${CLICKHOUSE_DATABASE}_05025_plain_repl/');" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE dst_cas_repl (k UInt32, v String) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/05025_dst_repl', 'r1') +ORDER BY k PARTITION BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05025_dst_repl', + name = '05025_cas_dst_repl', + path = '05025_cas_dst_repl_pool/');" + +${CLICKHOUSE_CLIENT} --query "INSERT INTO src_plain_repl SELECT number % 2, toString(number) FROM numbers(64);" +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas_repl ATTACH PARTITION 1 FROM src_plain_repl;" +${CLICKHOUSE_CLIENT} --query "SELECT 'leg3', count(), sum(k), uniqExact(v) FROM dst_cas_repl;" + +# The metadata-version repoint lands on a committed part, so the part must still read after a +# detach/attach round trip -- that is what proves the repoint did not corrupt the published ref. +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas_repl DETACH PARTITION 1;" +${CLICKHOUSE_CLIENT} --query "ALTER TABLE dst_cas_repl ATTACH PARTITION 1;" +${CLICKHOUSE_CLIENT} --query "SELECT 'leg3_roundtrip', count(), sum(k) FROM dst_cas_repl;" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE src_plain;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE dst_cas;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE src_cas;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE dst_cas_same_pool;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE src_plain_repl SYNC;" +${CLICKHOUSE_CLIENT} --query "DROP TABLE dst_cas_repl SYNC;" +${CLICKHOUSE_CLIENT} --query "SELECT 'dropped_ok';" diff --git a/tests/queries/0_stateless/05026_cas_manifest_path_newline.reference b/tests/queries/0_stateless/05026_cas_manifest_path_newline.reference new file mode 100644 index 000000000000..a9f6068a6403 --- /dev/null +++ b/tests/queries/0_stateless/05026_cas_manifest_path_newline.reference @@ -0,0 +1,11 @@ +insert_ok +1 +rows +30 +projection_result +v0 10 +v1 10 +v2 10 +unreachable 0 +dangling 0 +dropped_ok diff --git a/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh b/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh new file mode 100755 index 000000000000..da67e3fbd980 --- /dev/null +++ b/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh @@ -0,0 +1,54 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. + +# A projection name is used verbatim as a part-relative directory, so a projection named with a newline +# puts a newline in a part-file path. `MergeTree` allows that, and a content-addressed disk has to carry it: +# the manifest writes each path twice -- escaped in its record line, and in an `Inline` entry's payload-zone +# banner -- and both spellings have to agree or the writer cannot read back what it just wrote. +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +DISK_NAME="ca_05026_${CLICKHOUSE_TEST_UNIQUE_NAME}_${RANDOM}" +POOL_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}_05026_${RANDOM}" +TABLE="t_05026_${CLICKHOUSE_TEST_UNIQUE_NAME}" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS ${TABLE};" + +${CLICKHOUSE_CLIENT} --query " +CREATE TABLE ${TABLE} (k UInt32, v String, PROJECTION \`p +q\` (SELECT v, count() GROUP BY v)) +ENGINE = MergeTree ORDER BY k +SETTINGS disk = disk( + type = object_storage, + object_storage_type = local, + metadata_type = cas, + cas_server_root_id = '05026', + name = '${DISK_NAME}', + path = '${POOL_DIR}/');" + +echo 'insert_ok' +${CLICKHOUSE_CLIENT} --query "INSERT INTO ${TABLE} SELECT number, 'v' || (number % 3) FROM numbers(30);" \ + && echo 1 + +# The part is read back through the same manifest that was just written -- on the broken tree the `INSERT` +# never got this far, because the manifest it wrote could not be decoded. +echo 'rows' +${CLICKHOUSE_CLIENT} --query "SELECT count() FROM ${TABLE};" + +# And the projection itself is usable, which is what a newline in its name must not prevent. The settings +# `optimize_use_projections` and `force_optimize_projection` are the assertion: without the latter, this +# same result comes from the main part and the query proves nothing about the projection at all. +echo 'projection_result' +${CLICKHOUSE_CLIENT} --query "SELECT v, count() FROM ${TABLE} GROUP BY v ORDER BY v + SETTINGS optimize_use_projections = 1, force_optimize_projection = 1;" + +# No manifest body without a committed owner: the successful `INSERT` left nothing orphaned, and on the +# broken tree the failed one did -- that object wedged every later collection round. +${CLICKHOUSE_CLIENT} --query "SYSTEM CAS FSCK '${DISK_NAME}'" --format TSVWithNames \ + | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } + { print "unreachable", $col["unreachable"]; print "dangling", $col["dangling"] }' + +${CLICKHOUSE_CLIENT} --query "DROP TABLE ${TABLE};" +${CLICKHOUSE_CLIENT} --query "SELECT 'dropped_ok';" From 553d07b5503d3154a0d4eeb30aa79d1defcaa3ed Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:40:25 +0200 Subject: [PATCH 02/15] Resolve conflicts in cherry-pick of #2159 Kept the antalya-26.8 shape everywhere and layered the CAS changes on top: - `DiskObjectStorageTransaction`: base already routes every metadata effect through `addOperation` (eager vs queued, driven by `appliesOperationsEagerly`), which is the same mechanism as the PR's `dispatch`. Kept `addOperation` at the call sites, made it also honour `transactionIsStagingOverlay`, and dropped the duplicate `dispatch` template. - `copyS3File`: threaded `copy_mode` through the new `copyS3FileImpl`; `copyS3FileRange` passes `ObjectStorageCopyMode::Default`. - `DataPartStorageOnDiskFull`: moved the in-flight read-your-writes checks into the new `*Impl` methods (packed skip-index handling now lives in the base wrappers) and restored `getPackedFileUncompressedSize`, which the auto-merge had mixed up. - `DataPartStorageOnDiskBase::freeze`/`freezeRemote`: clone-transaction cleanup keeps the base `txn_version.txt.tmp` removal and `writeInvalidatedSystemColumnsFile`. - `LocalObjectStorage`: added the directory check to the relocated `tryGetObjectMetadata`. - `ThreadStatus`: base already keeps the parent `ThreadGroup` alive via `ThreadGroup::parent`, so the PR's duplicate `parent_thread_group` member is not added. - `RegisterDiskObjectStorage`: `use_fake_transaction` no longer exists on antalya-26.8 (object storage disks always use real transactions), so the PR's guard is not needed. - `MergeTreeData::removePartsInRangeFromWorkingSet...`: base already commits the empty covering part's disk transaction unconditionally. - `DataPartsExchange`: kept base `..._WITH_INVALIDATED_SYSTEM_COLUMNS = 10` and the PR's `..._WITH_CA_CONFIRM = 11` as the advertised maximum. - CI: added `cas_functional_tests_jobs` to the 26.8 `FUNCTIONAL_TESTS_JOBS` lists, `start_rustfs` into the 26.8 `start` sequence, removed a duplicated `cas_functional_tests_jobs` block from `altinity_jobs.py`, and hand-merged the generated workflow YAMLs (only CAS jobs added). - Removed duplicated `copyObject` counter and merged both `CopyObject` mocks in `gtest_writebuffer_s3.cpp`. - Stateless tests deleted on antalya-26.8 (converted to integration tests) stay deleted. Source-PR: #2159 (https://github.com/Altinity/ClickHouse/pull/2159) --- .github/workflows/fast_builds.yml | 18 +- .github/workflows/master.yml | 26 +- .github/workflows/pull_request.yml | 1537 +++-------------- .github/workflows/pull_request_community.yml | 178 +- ci/defs/altinity_jobs.py | 50 - ci/jobs/functional_tests.py | 36 +- ci/jobs/scripts/clickhouse_proc.py | 10 - ci/jobs/scripts/clickhouse_service.py | 62 - ci/workflows/pull_request.py | 18 +- ci/workflows/pull_request_community.py | 9 +- ci/workflows/release_branches.py | 4 - .../server-config/storing-data.mdx | 7 +- programs/disks/DisksApp.cpp | 3 - programs/disks/ICommand.h | 3 - programs/server/Server.cpp | 4 - src/Common/FailPoint.cpp | 5 +- src/Common/ThreadStatus.h | 12 - src/Core/ServerSettings.cpp | 5 - .../DiskObjectStorage/DiskObjectStorage.cpp | 8 - .../DiskObjectStorage/DiskObjectStorage.h | 8 +- .../DiskObjectStorageTransaction.cpp | 115 +- .../DiskObjectStorageTransaction.h | 11 - .../MetadataStorages/IMetadataStorage.h | 4 +- .../MetadataStorageFactory.cpp | 5 +- .../ObjectStorages/IObjectStorage.h | 4 +- .../Local/LocalObjectStorage.cpp | 30 +- .../ObjectStorages/S3/S3ObjectStorage.cpp | 4 +- .../RegisterDiskObjectStorage.cpp | 25 - src/Disks/IDiskTransaction.h | 4 +- src/IO/ReadPipeline.cpp | 27 +- src/IO/ReadPipeline.h | 7 +- src/IO/S3/copyS3File.cpp | 72 +- src/IO/S3/copyS3File.h | 9 +- src/IO/S3Common.cpp | 5 +- src/IO/WriteBufferFromS3.cpp | 6 - src/IO/tests/gtest_writebuffer_s3.cpp | 80 +- src/Interpreters/InterpreterSystemQuery.cpp | 6 +- .../ServerAsynchronousMetrics.cpp | 67 +- src/Interpreters/ThreadStatusExt.cpp | 4 - src/Parsers/ASTSystemQuery.h | 3 - src/Parsers/tests/gtest_Parser.cpp | 4 +- .../MergeTree/DataPartStorageOnDiskBase.cpp | 30 +- .../MergeTree/DataPartStorageOnDiskFull.cpp | 62 +- src/Storages/MergeTree/DataPartsExchange.cpp | 11 - src/Storages/MergeTree/IMergeTreeDataPart.cpp | 8 +- src/Storages/MergeTree/MergeTask.cpp | 11 +- src/Storages/MergeTree/MergeTreeData.cpp | 33 +- .../test_replicated_database/test.py | 6 +- .../0_stateless/02253_empty_part_checksums.sh | 40 - .../02254_projection_broken_part.sh | 45 - .../02255_broken_parts_chain_on_start.sh | 44 - .../02369_lost_part_intersecting_merges.sh | 52 - .../02370_lost_part_intersecting_merges.sh | 58 - ...2444_async_broken_outdated_part_loading.sh | 36 - ...eplicated_missing_covered_part_on_start.sh | 59 - .../04327_reader_executor_metrics.sql | 14 +- ...04328_reader_executor_kpi_async_metric.sql | 13 +- 57 files changed, 590 insertions(+), 2427 deletions(-) delete mode 100755 tests/queries/0_stateless/02253_empty_part_checksums.sh delete mode 100755 tests/queries/0_stateless/02254_projection_broken_part.sh delete mode 100755 tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh delete mode 100755 tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh delete mode 100755 tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh delete mode 100755 tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh delete mode 100755 tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh diff --git a/.github/workflows/fast_builds.yml b/.github/workflows/fast_builds.yml index 99429e92916e..ae6f4362ccec 100644 --- a/.github/workflows/fast_builds.yml +++ b/.github/workflows/fast_builds.yml @@ -1325,7 +1325,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -1346,7 +1346,7 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "Release Builds" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "Fast Builds" --ci --timestamp stateless_tests_arm_binary_cas_s3_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] @@ -1376,7 +1376,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -1397,7 +1397,7 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "Release Builds" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "Fast Builds" --ci --timestamp stateless_tests_amd_binary_cas_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] @@ -1427,7 +1427,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -1448,15 +1448,11 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "Release Builds" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "Fast Builds" --ci --timestamp finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD:.github/workflows/fast_builds.yml - needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, sign_release_amd_release, sign_release_arm_release, source_upload, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] -======= - needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, sign_release_amd_release, sign_release_arm_release, source_upload, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS):.github/workflows/release_builds.yml + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, sign_release_amd_release, sign_release_arm_release, source_upload, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: diff --git a/.github/workflows/master.yml b/.github/workflows/master.yml index 50fe3da00779..3aa3611508ec 100644 --- a/.github/workflows/master.yml +++ b/.github/workflows/master.yml @@ -3319,7 +3319,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3370,7 +3370,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3421,7 +3421,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3472,7 +3472,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3523,7 +3523,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3574,7 +3574,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3625,7 +3625,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3676,7 +3676,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3727,7 +3727,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -3778,7 +3778,7 @@ jobs: cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' export PYTHONPATH=./ci:.: cat > ./ci/tmp/workflow_inputs.json << 'EOF' - ${{ toJson(github.event.inputs) }} + ${{ toJson(inputs) }} EOF cat > ./ci/tmp/workflow_job.json << 'EOF' ${{ toJson(job) }} @@ -7263,11 +7263,7 @@ jobs: finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD - needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_llvm_coverage_per_test, build_amd_msan, build_amd_release, build_amd_release_pr_cache_warmup, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_release_pr_cache_warmup, build_arm_tsan, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, sign_release_amd_release, sign_release_arm_release, source_upload, sqllogic_test, sqlstorm_test, sqltest, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_3_3, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_4_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_5_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_6_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_7_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_8_8, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_parallel_1_4, stateless_tests_amd_tsan_parallel_2_4, stateless_tests_amd_tsan_parallel_3_4, stateless_tests_amd_tsan_parallel_4_4, stateless_tests_amd_tsan_s3_storage_parallel_1_3, stateless_tests_amd_tsan_s3_storage_parallel_2_3, stateless_tests_amd_tsan_s3_storage_parallel_3_3, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_azure_amd_msan, stress_test_azure_amd_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_llvm_coverage_per_test, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, clickbench_amd_release, clickbench_arm_release, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_6, integration_tests_amd_tsan_2_6, integration_tests_amd_tsan_3_6, integration_tests_amd_tsan_4_6, integration_tests_amd_tsan_5_6, integration_tests_amd_tsan_6_6, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, sign_release_amd_release, sign_release_arm_release, source_upload, sqllogic_test, sqlstorm_test, sqltest, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_4_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_5_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_6_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_7_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_8_8, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_4, stateless_tests_amd_msan_wasmedge_parallel_2_4, stateless_tests_amd_msan_wasmedge_parallel_3_4, stateless_tests_amd_msan_wasmedge_parallel_4_4, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_amd_tsan_s3_storage_parallel_1_2, stateless_tests_amd_tsan_s3_storage_parallel_2_2, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_arm_ubsan, stress_test_azure_amd_msan, stress_test_azure_amd_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_llvm_coverage_per_test, build_amd_msan, build_amd_release, build_amd_release_pr_cache_warmup, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_release_pr_cache_warmup, build_arm_tsan, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, sign_release_amd_release, sign_release_arm_release, source_upload, sqllogic_test, sqlstorm_test, sqltest, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_3, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_3_3, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_1_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_2_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_3_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_4_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_5_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_6_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_7_8, stateless_tests_amd_llvm_coverage_per_test_per_test_coverage_8_8, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_1_4, stateless_tests_amd_tsan_parallel_2_4, stateless_tests_amd_tsan_parallel_3_4, stateless_tests_amd_tsan_parallel_4_4, stateless_tests_amd_tsan_s3_storage_parallel_1_3, stateless_tests_amd_tsan_s3_storage_parallel_2_3, stateless_tests_amd_tsan_s3_storage_parallel_3_3, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_azure_amd_msan, stress_test_azure_amd_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: diff --git a/.github/workflows/pull_request.yml b/.github/workflows/pull_request.yml index f91d1f0e9f73..da1e917b9527 100644 --- a/.github/workflows/pull_request.yml +++ b/.github/workflows/pull_request.yml @@ -576,63 +576,6 @@ jobs: if: steps.upload_CH_ARM_ASAN_UBSAN_GH.outcome == 'failure' run: echo "::warning title=GH artifact upload failed::Failed to upload [CH_ARM_ASAN_UBSAN_GH] to GitHub artifacts (e.g. quota/rate limit). Downstream consumers will fall back to S3." -<<<<<<< HEAD -======= - build_arm_ubsan: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFybV91YnNhbik=') }} - name: "Build (arm_ubsan)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Build (arm_ubsan)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Build (arm_ubsan)' --workflow "PR" --ci --timestamp - - - name: Upload artifact CH_ARM_UBSAN_GH - id: upload_CH_ARM_UBSAN_GH - uses: actions/upload-artifact@v7 - continue-on-error: true - with: - name: CH_ARM_UBSAN_GH - path: ci/tmp/build/programs/self-extracting/clickhouse - retention-days: 1 - - name: Warn on failed upload of CH_ARM_UBSAN_GH - if: steps.upload_CH_ARM_UBSAN_GH.outcome == 'failure' - run: echo "::warning title=GH artifact upload failed::Failed to upload [CH_ARM_UBSAN_GH] to GitHub artifacts (e.g. quota/rate limit). Downstream consumers will fall back to S3." - ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) build_arm_binary: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] @@ -905,11 +848,7 @@ jobs: build_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-builder] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFtZF9yZWxlYXNlKQ==') }} name: "Build (amd_release)" outputs: @@ -951,11 +890,7 @@ jobs: build_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-builder] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVpbGQgKGFybV9yZWxlYXNlKQ==') }} name: "Build (arm_release)" outputs: @@ -1282,19 +1217,11 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'AST fuzzer (amd_debug, targeted, old_compatibility)' --workflow "PR" --ci --timestamp -<<<<<<< HEAD bugfix_validation_unit_tests: runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnVnZml4IHZhbGlkYXRpb24gKHVuaXQgdGVzdHMp') }} name: "Bugfix validation (unit tests)" -======= - stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIDEvMik=') }} - name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 1/2)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1330,16 +1257,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh -<<<<<<< HEAD PYTHONUNBUFFERED=1 python3 -m praktika run 'Bugfix validation (unit tests)' --workflow "PR" --ci --timestamp -======= - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2: + stateless_tests_amd_debug_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIDIvMik=') }} - name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)" + needs: [build_amd_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHBhcmFsbGVsKQ==') }} + name: "Stateless tests (amd_debug, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1354,7 +1278,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)" + test_name: "Stateless tests (amd_debug, parallel)" - name: Prepare env script run: | @@ -1371,24 +1295,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_ASAN_UBSAN_GH + - name: Download artifact CH_AMD_DEBUG_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_ASAN_UBSAN_GH + name: CH_AMD_DEBUG_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgMS8yKQ==') }} - name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)" + stateless_tests_amd_debug_sequential: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHNlcXVlbnRpYWwp') }} + name: "Stateless tests (amd_debug, sequential)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1403,7 +1327,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)" + test_name: "Stateless tests (amd_debug, sequential)" - name: Prepare env script run: | @@ -1420,24 +1344,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_ASAN_UBSAN_GH + - name: Download artifact CH_AMD_DEBUG_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_ASAN_UBSAN_GH + name: CH_AMD_DEBUG_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, sequential)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgMi8yKQ==') }} - name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)" + stateless_tests_amd_msan_wasmedge_parallel_1_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1452,7 +1376,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" - name: Prepare env script run: | @@ -1469,25 +1393,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_ASAN_UBSAN_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_ASAN_UBSAN_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, 2/2)' --workflow "PR" --ci --timestamp ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 1/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHBhcmFsbGVsKQ==') }} - name: "Stateless tests (amd_debug, parallel)" + stateless_tests_amd_msan_wasmedge_parallel_2_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1502,7 +1425,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, parallel)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" - name: Prepare env script run: | @@ -1519,28 +1442,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 2/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_debug_sequential: + stateless_tests_amd_msan_wasmedge_parallel_3_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIHNlcXVlbnRpYWwp') }} - name: "Stateless tests (amd_debug, sequential)" + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1555,7 +1474,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_debug, sequential)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" - name: Prepare env script run: | @@ -1572,31 +1491,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_DEBUG_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_DEBUG_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, sequential)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 3/8)' --workflow "PR" --ci --timestamp -<<<<<<< HEAD - stateless_tests_amd_msan_wasmedge_parallel_1_8: + stateless_tests_amd_msan_wasmedge_parallel_4_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" -======= - stateless_tests_amd_tsan_parallel_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] - needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIDEvMik=') }} - name: "Stateless tests (amd_tsan, parallel, 1/2)" + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1611,7 +1523,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, parallel, 1/2)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" - name: Prepare env script run: | @@ -1628,24 +1540,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_TSAN_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 4/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_parallel_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] - needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIDIvMik=') }} - name: "Stateless tests (amd_tsan, parallel, 2/2)" + stateless_tests_amd_msan_wasmedge_parallel_5_8: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA1Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1660,7 +1572,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, parallel, 2/2)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" - name: Prepare env script run: | @@ -1677,25 +1589,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_TSAN_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 5/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_sequential_1_2: + stateless_tests_amd_msan_wasmedge_parallel_6_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgMS8yKQ==') }} - name: "Stateless tests (amd_tsan, sequential, 1/2)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA2Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1710,10 +1621,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: -<<<<<<< HEAD - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/8)" -======= - test_name: "Stateless tests (amd_tsan, sequential, 1/2)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" - name: Prepare env script run: | @@ -1730,24 +1638,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_TSAN_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 6/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_sequential_2_2: + stateless_tests_amd_msan_wasmedge_parallel_7_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgMi8yKQ==') }} - name: "Stateless tests (amd_tsan, sequential, 2/2)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA3Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1762,7 +1670,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, sequential, 2/2)" + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" - name: Prepare env script run: | @@ -1779,24 +1687,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH + - name: Download artifact CH_AMD_MSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_TSAN_GH + name: CH_AMD_MSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 7/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_1_4: + stateless_tests_amd_msan_wasmedge_parallel_8_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAxLzQp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/4)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA4Lzgp') }} + name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1811,8 +1719,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 1/4)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" - name: Prepare env script run: | @@ -1840,19 +1747,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 1/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 8/8)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_2_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD + stateless_tests_amd_msan_wasmedge_sequential_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAyLzQp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/4)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDEvMik=') }} + name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1867,7 +1768,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 2/8)" + test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" - name: Prepare env script run: | @@ -1895,19 +1796,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 2/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_3_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD + stateless_tests_amd_msan_wasmedge_sequential_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCAzLzQp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/4)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDIvMik=') }} + name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1922,7 +1817,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 3/8)" + test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" - name: Prepare env script run: | @@ -1950,19 +1845,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 3/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_4_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD + stateless_tests_amd_debug_distributed_plan_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA0LzQp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/4)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHBhcmFsbGVsKQ==') }} + name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -1977,7 +1866,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 4/8)" + test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" - name: Prepare env script run: | @@ -1994,24 +1883,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_DEBUG_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_DEBUG_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 4/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_5_8: + stateless_tests_amd_debug_distributed_plan_s3_storage_sequential: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA1Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHNlcXVlbnRpYWwp') }} + name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2026,7 +1915,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 5/8)" + test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" - name: Prepare env script run: | @@ -2043,24 +1932,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_DEBUG_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_DEBUG_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 5/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, sequential)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_6_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA2Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + stateless_tests_arm_binary_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBwYXJhbGxlbCk=') }} + name: "Stateless tests (arm_binary, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2075,7 +1964,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 6/8)" + test_name: "Stateless tests (arm_binary, parallel)" - name: Prepare env script run: | @@ -2092,24 +1981,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_ARM_BIN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_ARM_BIN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 6/8)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_wasmedge_parallel_7_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA3Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + stateless_tests_arm_binary_sequential: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] + needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBzZXF1ZW50aWFsKQ==') }} + name: "Stateless tests (arm_binary, sequential)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2124,7 +2013,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 7/8)" + test_name: "Stateless tests (arm_binary, sequential)" - name: Prepare env script run: | @@ -2141,569 +2030,7 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_MSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 7/8)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_msan_wasmedge_parallel_8_8: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHBhcmFsbGVsLCA4Lzgp') }} - name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_msan, WasmEdge, parallel, 8/8)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_MSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_MSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, parallel, 8/8)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_msan_wasmedge_sequential_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] -<<<<<<< HEAD - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDEvMik=') }} - name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 1/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_MSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_MSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 1/2)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_msan_wasmedge_sequential_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] -<<<<<<< HEAD - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgV2FzbUVkZ2UsIHNlcXVlbnRpYWwsIDIvMik=') }} - name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_msan, WasmEdge, sequential, 2/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_MSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_MSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, WasmEdge, sequential, 2/2)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_debug_distributed_plan_s3_storage_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHBhcmFsbGVsKQ==') }} - name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, parallel)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_DEBUG_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_DEBUG_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, parallel)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_debug_distributed_plan_s3_storage_sequential: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfZGVidWcsIGRpc3RyaWJ1dGVkIHBsYW4sIHMzIHN0b3JhZ2UsIHNlcXVlbnRpYWwp') }} - name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_debug, distributed plan, s3 storage, sequential)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_DEBUG_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_DEBUG_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_debug, distributed plan, s3 storage, sequential)' --workflow "PR" --ci --timestamp - -<<<<<<< HEAD -======= - stateless_tests_amd_tsan_s3_storage_parallel_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIDEvMik=') }} - name: "Stateless tests (amd_tsan, s3 storage, parallel, 1/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_tsan, s3 storage, parallel, 1/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_tsan_s3_storage_parallel_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIDIvMik=') }} - name: "Stateless tests (amd_tsan, s3 storage, parallel, 2/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_tsan, s3 storage, parallel, 2/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_tsan_s3_storage_sequential_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgMS8yKQ==') }} - name: "Stateless tests (amd_tsan, s3 storage, sequential, 1/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_tsan, s3 storage, sequential, 1/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, 1/2)' --workflow "PR" --ci --timestamp - - stateless_tests_amd_tsan_s3_storage_sequential_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgMi8yKQ==') }} - name: "Stateless tests (amd_tsan, s3 storage, sequential, 2/2)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (amd_tsan, s3 storage, sequential, 2/2)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, 2/2)' --workflow "PR" --ci --timestamp - ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - stateless_tests_arm_binary_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] - needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBwYXJhbGxlbCk=') }} - name: "Stateless tests (arm_binary, parallel)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (arm_binary, parallel)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_ARM_BIN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_ARM_BIN_GH - path: ./ci/tmp - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, parallel)' --workflow "PR" --ci --timestamp - - stateless_tests_arm_binary_sequential: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD - needs: [build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBzZXF1ZW50aWFsKQ==') }} - name: "Stateless tests (arm_binary, sequential)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stateless tests (arm_binary, sequential)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Download artifact CH_ARM_BIN_GH + - name: Download artifact CH_ARM_BIN_GH uses: actions/download-artifact@v8 continue-on-error: true with: @@ -2716,18 +2043,11 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, sequential)' --workflow "PR" --ci --timestamp -<<<<<<< HEAD stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests: runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGlzdHJpYnV0ZWQgcGxhbiwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)" -======= - stateless_tests_amd_binary_cas_s3_storage_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} - name: "Stateless tests (amd_binary, cas s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2742,7 +2062,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" + test_name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)" - name: Prepare env script run: | @@ -2759,24 +2079,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_BINARY_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_BINARY_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} - name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} + name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2791,7 +2104,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" + test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)" - name: Prepare env script run: | @@ -2808,24 +2121,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_ASAN_UBSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_ASAN_UBSAN_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} - name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + stateless_tests_amd_tsan_parallel_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] + needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} + name: "Stateless tests (amd_tsan, parallel, selected tests)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2840,7 +2146,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" + test_name: "Stateless tests (amd_tsan, parallel, selected tests)" - name: Prepare env script run: | @@ -2857,24 +2163,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_ASAN_UBSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_ASAN_UBSAN_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} - name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" + stateless_tests_amd_tsan_sequential_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} + name: "Stateless tests (amd_tsan, sequential, selected tests)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2889,7 +2188,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" + test_name: "Stateless tests (amd_tsan, sequential, selected tests)" - name: Prepare env script run: | @@ -2906,24 +2205,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: + stateless_tests_amd_tsan_s3_storage_parallel_selected_tests: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} - name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} + name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2938,7 +2230,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" + test_name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" - name: Prepare env script run: | @@ -2955,24 +2247,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_TSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_TSAN_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} - name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + stateless_tests_amd_tsan_s3_storage_sequential_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} + name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2987,7 +2272,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" + test_name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" - name: Prepare env script run: | @@ -3004,24 +2289,17 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH - uses: actions/download-artifact@v8 - continue-on-error: true - with: - name: CH_AMD_MSAN_GH - path: ./ci/tmp - - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, selected tests)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} - name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + stateless_tests_amd_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3036,7 +2314,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" + test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" - name: Prepare env script run: | @@ -3053,24 +2331,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_BINARY_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_BINARY_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} - name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3085,7 +2363,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" - name: Prepare env script run: | @@ -3102,24 +2380,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_MSAN_GH + - name: Download artifact CH_AMD_ASAN_UBSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_MSAN_GH + name: CH_AMD_ASAN_UBSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_arm_binary_cas_s3_storage_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} - name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3134,7 +2412,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" + test_name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" - name: Prepare env script run: | @@ -3151,24 +2429,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_ARM_BIN_GH + - name: Download artifact CH_AMD_ASAN_UBSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_ARM_BIN_GH + name: CH_AMD_ASAN_UBSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_binary_cas_storage_parallel: + stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} - name: "Stateless tests (amd_binary, cas storage, parallel)" + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3183,7 +2461,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_binary, cas storage, parallel)" + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" - name: Prepare env script run: | @@ -3200,25 +2478,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF - - name: Download artifact CH_AMD_BINARY_GH + - name: Download artifact CH_AMD_TSAN_GH uses: actions/download-artifact@v8 continue-on-error: true with: - name: CH_AMD_BINARY_GH + name: CH_AMD_TSAN_GH path: ./ci/tmp - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "PR" --ci --timestamp - stateless_tests_arm_asan_ubsan_azure_parallel: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHBhcmFsbGVsKQ==') }} - name: "Stateless tests (arm_asan_ubsan, azure, parallel)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} + name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3233,7 +2510,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)" + test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" - name: Prepare env script run: | @@ -3250,17 +2527,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_TSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_TSAN_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, distributed plan, parallel, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "PR" --ci --timestamp - stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgZGIgZGlzaywgZGlzdHJpYnV0ZWQgcGxhbiwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} - name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)" + stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3275,7 +2559,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" - name: Prepare env script run: | @@ -3292,17 +2576,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_asan_ubsan, db disk, distributed plan, sequential, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_parallel_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] - needs: [build_amd_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} - name: "Stateless tests (amd_tsan, parallel, selected tests)" + stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3317,7 +2608,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, parallel, selected tests)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" - name: Prepare env script run: | @@ -3334,17 +2625,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, parallel, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_sequential_selected_tests: + stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} - name: "Stateless tests (amd_tsan, sequential, selected tests)" + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} + name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3359,7 +2657,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, sequential, selected tests)" + test_name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" - name: Prepare env script run: | @@ -3376,17 +2674,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_MSAN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_MSAN_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, sequential, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_s3_storage_parallel_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + stateless_tests_arm_binary_cas_s3_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} - name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (arm_binary, cas s3 storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3401,7 +2706,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" + test_name: "Stateless tests (arm_binary, cas s3 storage, parallel)" - name: Prepare env script run: | @@ -3418,17 +2723,24 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_ARM_BIN_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_ARM_BIN_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (arm_binary, cas s3 storage, parallel)' --workflow "PR" --ci --timestamp - stateless_tests_amd_tsan_s3_storage_sequential_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} - name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" + stateless_tests_amd_binary_cas_storage_parallel: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} + name: "Stateless tests (amd_binary, cas storage, parallel)" outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -3443,7 +2755,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: - test_name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" + test_name: "Stateless tests (amd_binary, cas storage, parallel)" - name: Prepare env script run: | @@ -3460,11 +2772,18 @@ jobs: EOF ENV_SETUP_SCRIPT_EOF + - name: Download artifact CH_AMD_BINARY_GH + uses: actions/download-artifact@v8 + continue-on-error: true + with: + name: CH_AMD_BINARY_GH + path: ./ci/tmp + - name: Run id: run run: | . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, selected tests)' --workflow "PR" --ci --timestamp + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_binary, cas storage, parallel)' --workflow "PR" --ci --timestamp stateless_tests_arm_asan_ubsan_azure_parallel_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] @@ -3860,11 +3179,7 @@ jobs: stateless_tests_arm_asan_ubsan_azure_sequential_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 32g] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHNlcXVlbnRpYWwsIDEvMik=') }} name: "Stateless tests (arm_asan_ubsan, azure, sequential, 1/2)" outputs: @@ -3913,11 +3228,7 @@ jobs: stateless_tests_arm_asan_ubsan_azure_sequential_2_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 32g] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYXNhbl91YnNhbiwgYXp1cmUsIHNlcXVlbnRpYWwsIDIvMik=') }} name: "Stateless tests (arm_asan_ubsan, azure, sequential, 2/2)" outputs: @@ -3966,11 +3277,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDEvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 1/8)" outputs: @@ -4019,11 +3326,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDIvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 2/8)" outputs: @@ -4072,11 +3375,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDMvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 3/8)" outputs: @@ -4125,11 +3424,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDQvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 4/8)" outputs: @@ -4178,11 +3473,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDUvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 5/8)" outputs: @@ -4231,11 +3522,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDYvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 6/8)" outputs: @@ -4284,11 +3571,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDcvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 7/8)" outputs: @@ -4337,11 +3620,7 @@ jobs: integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9hc2FuX3Vic2FuLCBkYiBkaXNrLCBvbGQgYW5hbHl6ZXIsIDgvOCk=') }} name: "Integration tests (amd_asan_ubsan, db disk, old analyzer, 8/8)" outputs: @@ -4390,11 +3669,7 @@ jobs: integration_tests_arm_binary_distributed_plan_1_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDEvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 1/4)" outputs: @@ -4443,11 +3718,7 @@ jobs: integration_tests_arm_binary_distributed_plan_2_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDIvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 2/4)" outputs: @@ -4496,11 +3767,7 @@ jobs: integration_tests_arm_binary_distributed_plan_3_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDMvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 3/4)" outputs: @@ -4549,11 +3816,7 @@ jobs: integration_tests_arm_binary_distributed_plan_4_4: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFybV9iaW5hcnksIGRpc3RyaWJ1dGVkIHBsYW4sIDQvNCk=') }} name: "Integration tests (arm_binary, distributed plan, 4/4)" outputs: @@ -4602,15 +3865,9 @@ jobs: integration_tests_amd_tsan_1_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAxLzgp') }} name: "Integration tests (amd_tsan, 1/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAxLzYp') }} - name: "Integration tests (amd_tsan, 1/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -4657,15 +3914,9 @@ jobs: integration_tests_amd_tsan_2_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAyLzgp') }} name: "Integration tests (amd_tsan, 2/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAyLzYp') }} - name: "Integration tests (amd_tsan, 2/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -4712,15 +3963,9 @@ jobs: integration_tests_amd_tsan_3_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAzLzgp') }} name: "Integration tests (amd_tsan, 3/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCAzLzYp') }} - name: "Integration tests (amd_tsan, 3/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -4767,15 +4012,9 @@ jobs: integration_tests_amd_tsan_4_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA0Lzgp') }} name: "Integration tests (amd_tsan, 4/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA0LzYp') }} - name: "Integration tests (amd_tsan, 4/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -4822,15 +4061,9 @@ jobs: integration_tests_amd_tsan_5_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA1Lzgp') }} name: "Integration tests (amd_tsan, 5/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA1LzYp') }} - name: "Integration tests (amd_tsan, 5/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -4877,15 +4110,9 @@ jobs: integration_tests_amd_tsan_6_8: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA2Lzgp') }} name: "Integration tests (amd_tsan, 6/8)" -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF90c2FuLCA2LzYp') }} - name: "Integration tests (amd_tsan, 6/6)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -5030,11 +4257,7 @@ jobs: integration_tests_amd_msan_1_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAxLzEwKQ==') }} name: "Integration tests (amd_msan, 1/10)" outputs: @@ -5083,11 +4306,7 @@ jobs: integration_tests_amd_msan_2_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAyLzEwKQ==') }} name: "Integration tests (amd_msan, 2/10)" outputs: @@ -5136,11 +4355,7 @@ jobs: integration_tests_amd_msan_3_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAzLzEwKQ==') }} name: "Integration tests (amd_msan, 3/10)" outputs: @@ -5189,11 +4404,7 @@ jobs: integration_tests_amd_msan_4_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA0LzEwKQ==') }} name: "Integration tests (amd_msan, 4/10)" outputs: @@ -5242,11 +4453,7 @@ jobs: integration_tests_amd_msan_5_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA1LzEwKQ==') }} name: "Integration tests (amd_msan, 5/10)" outputs: @@ -5295,11 +4502,7 @@ jobs: integration_tests_amd_msan_6_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA2LzEwKQ==') }} name: "Integration tests (amd_msan, 6/10)" outputs: @@ -5348,11 +4551,7 @@ jobs: integration_tests_amd_msan_7_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA3LzEwKQ==') }} name: "Integration tests (amd_msan, 7/10)" outputs: @@ -5401,11 +4600,7 @@ jobs: integration_tests_amd_msan_8_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA4LzEwKQ==') }} name: "Integration tests (amd_msan, 8/10)" outputs: @@ -5454,11 +4649,7 @@ jobs: integration_tests_amd_msan_9_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCA5LzEwKQ==') }} name: "Integration tests (amd_msan, 9/10)" outputs: @@ -5507,11 +4698,7 @@ jobs: integration_tests_amd_msan_10_10: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW50ZWdyYXRpb24gdGVzdHMgKGFtZF9tc2FuLCAxMC8xMCk=') }} name: "Integration tests (amd_msan, 10/10)" outputs: @@ -5812,11 +4999,7 @@ jobs: docker_server_image: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_release, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'RG9ja2VyIHNlcnZlciBpbWFnZQ==') }} name: "Docker server image" outputs: @@ -5858,11 +5041,7 @@ jobs: docker_keeper_image: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] -======= - needs: [build_amd_release, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'RG9ja2VyIGtlZXBlciBpbWFnZQ==') }} name: "Docker keeper image" outputs: @@ -5904,11 +5083,7 @@ jobs: install_packages_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_release, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW5zdGFsbCBwYWNrYWdlcyAoYW1kX3JlbGVhc2Up') }} name: "Install packages (amd_release)" outputs: @@ -5950,11 +5125,7 @@ jobs: install_packages_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'SW5zdGFsbCBwYWNrYWdlcyAoYXJtX3JlbGVhc2Up') }} name: "Install packages (arm_release)" outputs: @@ -5996,11 +5167,7 @@ jobs: compatibility_check_amd_release: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_release, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'Q29tcGF0aWJpbGl0eSBjaGVjayAoYW1kX3JlbGVhc2Up') }} name: "Compatibility check (amd_release)" outputs: @@ -6042,11 +5209,7 @@ jobs: compatibility_check_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'Q29tcGF0aWJpbGl0eSBjaGVjayAoYXJtX3JlbGVhc2Up') }} name: "Compatibility check (arm_release)" outputs: @@ -6088,11 +5251,7 @@ jobs: stress_test_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9kZWJ1Zyk=') }} name: "Stress test (amd_debug)" outputs: @@ -6134,11 +5293,7 @@ jobs: stress_test_amd_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9hc2FuX3Vic2FuKQ==') }} name: "Stress test (amd_asan_ubsan)" outputs: @@ -6180,11 +5335,7 @@ jobs: stress_test_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF90c2FuKQ==') }} name: "Stress test (amd_tsan)" outputs: @@ -6226,11 +5377,7 @@ jobs: stress_test_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFtZF9tc2FuKQ==') }} name: "Stress test (amd_msan)" outputs: @@ -6272,11 +5419,7 @@ jobs: stress_test_arm_release: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9yZWxlYXNlKQ==') }} name: "Stress test (arm_release)" outputs: @@ -6318,11 +5461,7 @@ jobs: stress_test_arm_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_debug, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9kZWJ1Zyk=') }} name: "Stress test (arm_debug)" outputs: @@ -6364,11 +5503,7 @@ jobs: stress_test_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9hc2FuX3Vic2FuKQ==') }} name: "Stress test (arm_asan_ubsan)" outputs: @@ -6410,11 +5545,7 @@ jobs: stress_test_arm_asan_ubsan_s3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9hc2FuX3Vic2FuLCBzMyk=') }} name: "Stress test (arm_asan_ubsan, s3)" outputs: @@ -6456,11 +5587,7 @@ jobs: stress_test_arm_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV90c2FuKQ==') }} name: "Stress test (arm_tsan)" outputs: @@ -6502,11 +5629,7 @@ jobs: stress_test_arm_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, build_arm_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_msan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV9tc2FuKQ==') }} name: "Stress test (arm_msan)" outputs: @@ -6546,57 +5669,9 @@ jobs: . ./ci/tmp/praktika_setup_env.sh PYTHONUNBUFFERED=1 python3 -m praktika run 'Stress test (arm_msan)' --workflow "PR" --ci --timestamp -<<<<<<< HEAD ast_fuzzer_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - stress_test_arm_ubsan: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_ubsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RyZXNzIHRlc3QgKGFybV91YnNhbik=') }} - name: "Stress test (arm_ubsan)" - outputs: - data: ${{ steps.run.outputs.DATA }} - pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} - steps: - - name: Checkout code - uses: actions/checkout@v6 - with: - ref: ${{ env.CHECKOUT_REF }} - - - name: Setup - uses: ./.github/actions/runner_setup - - name: Docker setup - uses: ./.github/actions/docker_setup - with: - test_name: "Stress test (arm_ubsan)" - - - name: Prepare env script - run: | - rm -rf ./ci/tmp - mkdir -p ./ci/tmp - cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' - export PYTHONPATH=./ci:.: - - cat > ./ci/tmp/workflow_job.json << 'EOF' - ${{ toJson(job) }} - EOF - cat > ./ci/tmp/workflow_status.json << 'EOF' - ${{ toJson(needs) }} - EOF - ENV_SETUP_SCRIPT_EOF - - - name: Run - id: run - run: | - . ./ci/tmp/praktika_setup_env.sh - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stress test (arm_ubsan)' --workflow "PR" --ci --timestamp - - ast_fuzzer_amd_debug: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX2RlYnVnKQ==') }} name: "AST fuzzer (amd_debug)" outputs: @@ -6645,11 +5720,7 @@ jobs: ast_fuzzer_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYXJtX2FzYW5fdWJzYW4p') }} name: "AST fuzzer (arm_asan_ubsan)" outputs: @@ -6698,11 +5769,7 @@ jobs: ast_fuzzer_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX3RzYW4p') }} name: "AST fuzzer (amd_tsan)" outputs: @@ -6751,11 +5818,7 @@ jobs: ast_fuzzer_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QVNUIGZ1enplciAoYW1kX21zYW4p') }} name: "AST fuzzer (amd_msan)" outputs: @@ -6804,11 +5867,7 @@ jobs: buzzhouse_amd_debug: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfZGVidWcp') }} name: "BuzzHouse (amd_debug)" outputs: @@ -6857,11 +5916,7 @@ jobs: buzzhouse_arm_asan_ubsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhcm1fYXNhbl91YnNhbik=') }} name: "BuzzHouse (arm_asan_ubsan)" outputs: @@ -6910,11 +5965,7 @@ jobs: buzzhouse_amd_tsan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfdHNhbik=') }} name: "BuzzHouse (amd_tsan)" outputs: @@ -6963,11 +6014,7 @@ jobs: buzzhouse_amd_msan: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'QnV6ekhvdXNlIChhbWRfbXNhbik=') }} name: "BuzzHouse (amd_msan)" outputs: @@ -7100,11 +6147,7 @@ jobs: sqllogic_test: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U1FMTG9naWMgdGVzdA==') }} name: "SQLLogic test" outputs: @@ -7146,11 +6189,7 @@ jobs: sqlstorm_test: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64] -<<<<<<< HEAD needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, build_arm_release, config_workflow, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_arm_binary_parallel] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U1FMU3Rvcm0gdGVzdA==') }} name: "SQLStorm test" outputs: @@ -7369,11 +6408,7 @@ jobs: finish_workflow: runs-on: [self-hosted, altinity-on-demand, altinity-style-checker] -<<<<<<< HEAD - needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_debug_targeted, ast_fuzzer_amd_debug_targeted_old_compatibility, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, bugfix_validation_unit_tests, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_toolchain_pgo_bolt_aarch64, build_toolchain_pgo_bolt_amd64, build_wasm_parser, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, ci_tests, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_asan_ubsan_targeted, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, keeper_stress_tests_pr, parser_memory_check, promql_compliance, quick_functional_tests, source_upload, sqllogic_test, sqlstorm_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_sequential_selected_tests, stateless_tests_amd_tsan_sequential_selected_tests, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_asan_ubsan_targeted, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] -======= - needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_debug_targeted, ast_fuzzer_amd_debug_targeted_old_compatibility, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_arm_ubsan, build_toolchain_pgo_bolt_aarch64, build_toolchain_pgo_bolt_amd64, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, ci_tests, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_asan_ubsan_targeted, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_6, integration_tests_amd_tsan_2_6, integration_tests_amd_tsan_3_6, integration_tests_amd_tsan_4_6, integration_tests_amd_tsan_5_6, integration_tests_amd_tsan_6_6, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, keeper_stress_tests_pr, quick_functional_tests, source_upload, sqllogic_test, sqlstorm_test, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_1_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_2_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_1_2, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_2_2, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_4, stateless_tests_amd_msan_wasmedge_parallel_2_4, stateless_tests_amd_msan_wasmedge_parallel_3_4, stateless_tests_amd_msan_wasmedge_parallel_4_4, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_1_2, stateless_tests_amd_tsan_parallel_2_2, stateless_tests_amd_tsan_s3_storage_parallel_1_2, stateless_tests_amd_tsan_s3_storage_parallel_2_2, stateless_tests_amd_tsan_s3_storage_sequential_1_2, stateless_tests_amd_tsan_s3_storage_sequential_2_2, stateless_tests_amd_tsan_sequential_1_2, stateless_tests_amd_tsan_sequential_2_2, stateless_tests_arm_asan_ubsan_azure_parallel, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_asan_ubsan_targeted, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, stress_test_arm_ubsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + needs: [ast_fuzzer_amd_debug, ast_fuzzer_amd_debug_targeted, ast_fuzzer_amd_debug_targeted_old_compatibility, ast_fuzzer_amd_msan, ast_fuzzer_amd_tsan, ast_fuzzer_arm_asan_ubsan, bugfix_validation_unit_tests, build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_msan, build_amd_release, build_amd_tsan, build_arm_asan_ubsan, build_arm_binary, build_arm_debug, build_arm_msan, build_arm_release, build_arm_tsan, build_toolchain_pgo_bolt_aarch64, build_toolchain_pgo_bolt_amd64, build_wasm_parser, buzzhouse_amd_debug, buzzhouse_amd_msan, buzzhouse_amd_tsan, buzzhouse_arm_asan_ubsan, ci_tests, compatibility_check_amd_release, compatibility_check_arm_release, config_workflow, docker_keeper_image, docker_server_image, dockers_build_amd, dockers_build_arm, dockers_build_multiplatform_manifest, fast_test, install_packages_amd_release, install_packages_arm_release, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_2_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_3_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_4_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_5_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_6_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_7_8, integration_tests_amd_asan_ubsan_db_disk_old_analyzer_8_8, integration_tests_amd_asan_ubsan_targeted, integration_tests_amd_msan_10_10, integration_tests_amd_msan_1_10, integration_tests_amd_msan_2_10, integration_tests_amd_msan_3_10, integration_tests_amd_msan_4_10, integration_tests_amd_msan_5_10, integration_tests_amd_msan_6_10, integration_tests_amd_msan_7_10, integration_tests_amd_msan_8_10, integration_tests_amd_msan_9_10, integration_tests_amd_tsan_1_8, integration_tests_amd_tsan_2_8, integration_tests_amd_tsan_3_8, integration_tests_amd_tsan_4_8, integration_tests_amd_tsan_5_8, integration_tests_amd_tsan_6_8, integration_tests_amd_tsan_7_8, integration_tests_amd_tsan_8_8, integration_tests_arm_binary_distributed_plan_1_4, integration_tests_arm_binary_distributed_plan_2_4, integration_tests_arm_binary_distributed_plan_3_4, integration_tests_arm_binary_distributed_plan_4_4, keeper_stress_tests_pr, parser_memory_check, promql_compliance, quick_functional_tests, source_upload, sqllogic_test, sqlstorm_test, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_binary_cas_s3_storage_parallel, stateless_tests_amd_binary_cas_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_parallel, stateless_tests_amd_debug_distributed_plan_s3_storage_sequential, stateless_tests_amd_debug_parallel, stateless_tests_amd_debug_sequential, stateless_tests_amd_msan_cas_s3_storage_parallel_1_3, stateless_tests_amd_msan_cas_s3_storage_parallel_2_3, stateless_tests_amd_msan_cas_s3_storage_parallel_3_3, stateless_tests_amd_msan_wasmedge_parallel_1_8, stateless_tests_amd_msan_wasmedge_parallel_2_8, stateless_tests_amd_msan_wasmedge_parallel_3_8, stateless_tests_amd_msan_wasmedge_parallel_4_8, stateless_tests_amd_msan_wasmedge_parallel_5_8, stateless_tests_amd_msan_wasmedge_parallel_6_8, stateless_tests_amd_msan_wasmedge_parallel_7_8, stateless_tests_amd_msan_wasmedge_parallel_8_8, stateless_tests_amd_msan_wasmedge_sequential_1_2, stateless_tests_amd_msan_wasmedge_sequential_2_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2, stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_parallel_selected_tests, stateless_tests_amd_tsan_s3_storage_sequential_selected_tests, stateless_tests_amd_tsan_sequential_selected_tests, stateless_tests_arm_asan_ubsan_azure_parallel_1_8, stateless_tests_arm_asan_ubsan_azure_parallel_2_8, stateless_tests_arm_asan_ubsan_azure_parallel_3_8, stateless_tests_arm_asan_ubsan_azure_parallel_4_8, stateless_tests_arm_asan_ubsan_azure_parallel_5_8, stateless_tests_arm_asan_ubsan_azure_parallel_6_8, stateless_tests_arm_asan_ubsan_azure_parallel_7_8, stateless_tests_arm_asan_ubsan_azure_parallel_8_8, stateless_tests_arm_asan_ubsan_azure_sequential_1_2, stateless_tests_arm_asan_ubsan_azure_sequential_2_2, stateless_tests_arm_asan_ubsan_targeted, stateless_tests_arm_binary_cas_s3_storage_parallel, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential, stress_test_amd_asan_ubsan, stress_test_amd_debug, stress_test_amd_msan, stress_test_amd_tsan, stress_test_arm_asan_ubsan, stress_test_arm_asan_ubsan_s3, stress_test_arm_debug, stress_test_arm_msan, stress_test_arm_release, stress_test_arm_tsan, unit_tests_asan_ubsan, unit_tests_asan_ubsan_function_prop_fuzzer, unit_tests_msan, unit_tests_msan_function_prop_fuzzer, unit_tests_tsan, unit_tests_tsan_function_prop_fuzzer] if: ${{ !cancelled() && needs.config_workflow.outputs.pipeline_status != '' }} name: "Finish Workflow" outputs: @@ -7508,22 +6543,12 @@ jobs: - stateless_tests_amd_debug_distributed_plan_s3_storage_sequential - stateless_tests_arm_binary_parallel - stateless_tests_arm_binary_sequential -<<<<<<< HEAD - stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests - stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests - stateless_tests_amd_tsan_parallel_selected_tests - stateless_tests_amd_tsan_sequential_selected_tests - stateless_tests_amd_tsan_s3_storage_parallel_selected_tests - stateless_tests_amd_tsan_s3_storage_sequential_selected_tests - - stateless_tests_arm_asan_ubsan_azure_parallel_1_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_2_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_3_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_4_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_5_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_6_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_7_8 - - stateless_tests_arm_asan_ubsan_azure_parallel_8_8 -======= - stateless_tests_amd_binary_cas_s3_storage_parallel - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2 - stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2 @@ -7534,8 +6559,14 @@ jobs: - stateless_tests_amd_msan_cas_s3_storage_parallel_3_3 - stateless_tests_arm_binary_cas_s3_storage_parallel - stateless_tests_amd_binary_cas_storage_parallel - - stateless_tests_arm_asan_ubsan_azure_parallel ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + - stateless_tests_arm_asan_ubsan_azure_parallel_1_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_2_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_3_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_4_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_5_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_6_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_7_8 + - stateless_tests_arm_asan_ubsan_azure_parallel_8_8 - stateless_tests_arm_asan_ubsan_azure_sequential_1_2 - stateless_tests_arm_asan_ubsan_azure_sequential_2_2 - integration_tests_amd_asan_ubsan_db_disk_old_analyzer_1_8 diff --git a/.github/workflows/pull_request_community.yml b/.github/workflows/pull_request_community.yml index 25854d5fba10..15bde4690dc0 100644 --- a/.github/workflows/pull_request_community.yml +++ b/.github/workflows/pull_request_community.yml @@ -2088,7 +2088,6 @@ jobs: if-no-files-found: ignore retention-days: 14 -<<<<<<< HEAD stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests: runs-on: [self-hosted, altinity-on-demand, altinity-builder, 64g] needs: [build_amd_asan_ubsan, config_workflow, fast_test] @@ -2350,13 +2349,134 @@ jobs: needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgcGFyYWxsZWwsIHNlbGVjdGVkIHRlc3RzKQ==') }} name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" -======= + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_TSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, selected tests)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_tsan_s3_storage_parallel_selected_tests + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + + stateless_tests_amd_tsan_s3_storage_sequential_selected_tests: + runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] + if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} + name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" + outputs: + data: ${{ steps.run.outputs.DATA }} + pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} + steps: + - name: Checkout code + uses: actions/checkout@v6 + with: + ref: ${{ env.CHECKOUT_REF }} + + - name: Setup + uses: ./.github/actions/runner_setup + - name: Docker setup + uses: ./.github/actions/docker_setup + with: + test_name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" + + - name: Prepare env script + run: | + rm -rf ./ci/tmp + mkdir -p ./ci/tmp + cat > ./ci/tmp/praktika_setup_env.sh << 'ENV_SETUP_SCRIPT_EOF' + export PYTHONPATH=./ci:.: + + cat > ./ci/tmp/workflow_job.json << 'EOF' + ${{ toJson(job) }} + EOF + cat > ./ci/tmp/workflow_status.json << 'EOF' + ${{ toJson(needs) }} + EOF + ENV_SETUP_SCRIPT_EOF + + - name: Download artifact CH_AMD_TSAN + uses: actions/download-artifact@v8 + with: + name: CH_AMD_TSAN + path: ./ci/tmp + + - name: Run + id: run + run: | + . ./ci/tmp/praktika_setup_env.sh + PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, selected tests)' --workflow "Community PR" --ci --timestamp + + - name: Upload failure report artifact + if: failure() + uses: actions/upload-artifact@v7 + continue-on-error: true + with: + name: failure-stateless_tests_amd_tsan_s3_storage_sequential_selected_tests + path: | + ci/tmp/result_*.json + ci/tmp/test_result.txt + ci/tmp/pytest*.jsonl + ci/tmp/gtest.json + ci/tmp/logs.tar.gz + ci/tmp/configs.tar.gz + if-no-files-found: ignore + retention-days: 14 + stateless_tests_amd_binary_cas_s3_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_binary, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} name: "Stateless tests (amd_binary, cas s3 storage, parallel)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2371,9 +2491,6 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: -<<<<<<< HEAD - test_name: "Stateless tests (amd_tsan, s3 storage, parallel, selected tests)" -======= test_name: "Stateless tests (amd_binary, cas s3 storage, parallel)" - name: Prepare env script @@ -2421,7 +2538,7 @@ jobs: stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 1/2)" outputs: @@ -2485,7 +2602,7 @@ jobs: stateless_tests_amd_asan_ubsan_cas_s3_storage_parallel_2_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_asan_ubsan, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYXNhbl91YnNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} name: "Stateless tests (amd_asan_ubsan, cas s3 storage, parallel, 2/2)" outputs: @@ -2549,7 +2666,7 @@ jobs: stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzIp') }} name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" outputs: @@ -2567,7 +2684,6 @@ jobs: uses: ./.github/actions/docker_setup with: test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Prepare env script run: | @@ -2594,22 +2710,14 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh -<<<<<<< HEAD - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, parallel, selected tests)' --workflow "Community PR" --ci --timestamp -======= PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 1/2)' --workflow "Community PR" --ci --timestamp ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Upload failure report artifact if: failure() uses: actions/upload-artifact@v7 continue-on-error: true with: -<<<<<<< HEAD - name: failure-stateless_tests_amd_tsan_s3_storage_parallel_selected_tests -======= name: failure-stateless_tests_amd_tsan_cas_s3_storage_parallel_1_2 ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) path: | ci/tmp/result_*.json ci/tmp/test_result.txt @@ -2620,19 +2728,11 @@ jobs: if-no-files-found: ignore retention-days: 14 -<<<<<<< HEAD - stateless_tests_amd_tsan_s3_storage_sequential_selected_tests: - runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 32g] - needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] - if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgczMgc3RvcmFnZSwgc2VxdWVudGlhbCwgc2VsZWN0ZWQgdGVzdHMp') }} - name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" -======= stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfdHNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzIp') }} name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) outputs: data: ${{ steps.run.outputs.DATA }} pipeline_status: ${{ steps.run.outputs.pipeline_status || 'undefined' }} @@ -2647,11 +2747,7 @@ jobs: - name: Docker setup uses: ./.github/actions/docker_setup with: -<<<<<<< HEAD - test_name: "Stateless tests (amd_tsan, s3 storage, sequential, selected tests)" -======= test_name: "Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)" ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Prepare env script run: | @@ -2678,20 +2774,13 @@ jobs: id: run run: | . ./ci/tmp/praktika_setup_env.sh -<<<<<<< HEAD - PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, s3 storage, sequential, selected tests)' --workflow "Community PR" --ci --timestamp -======= PYTHONUNBUFFERED=1 python3 -m praktika run 'Stateless tests (amd_tsan, cas s3 storage, parallel, 2/2)' --workflow "Community PR" --ci --timestamp ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) - name: Upload failure report artifact if: failure() uses: actions/upload-artifact@v7 continue-on-error: true with: -<<<<<<< HEAD - name: failure-stateless_tests_amd_tsan_s3_storage_sequential_selected_tests -======= name: failure-stateless_tests_amd_tsan_cas_s3_storage_parallel_2_2 path: | ci/tmp/result_*.json @@ -2705,7 +2794,7 @@ jobs: stateless_tests_amd_msan_cas_s3_storage_parallel_1_3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAxLzMp') }} name: "Stateless tests (amd_msan, cas s3 storage, parallel, 1/3)" outputs: @@ -2769,7 +2858,7 @@ jobs: stateless_tests_amd_msan_cas_s3_storage_parallel_2_3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAyLzMp') }} name: "Stateless tests (amd_msan, cas s3 storage, parallel, 2/3)" outputs: @@ -2833,7 +2922,7 @@ jobs: stateless_tests_amd_msan_cas_s3_storage_parallel_3_3: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester] - needs: [build_amd_debug, build_amd_msan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_msan, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfbXNhbiwgY2FzIHMzIHN0b3JhZ2UsIHBhcmFsbGVsLCAzLzMp') }} name: "Stateless tests (amd_msan, cas s3 storage, parallel, 3/3)" outputs: @@ -2897,7 +2986,7 @@ jobs: stateless_tests_arm_binary_cas_s3_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester-aarch64, 16c] - needs: [build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhcm1fYmluYXJ5LCBjYXMgczMgc3RvcmFnZSwgcGFyYWxsZWwp') }} name: "Stateless tests (arm_binary, cas s3 storage, parallel)" outputs: @@ -2961,7 +3050,7 @@ jobs: stateless_tests_amd_binary_cas_storage_parallel: runs-on: [self-hosted, altinity-on-demand, altinity-func-tester, 16c] - needs: [build_amd_binary, build_amd_debug, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_debug_parallel, stateless_tests_arm_binary_parallel] + needs: [build_amd_asan_ubsan, build_amd_binary, build_amd_debug, build_amd_tsan, build_arm_binary, config_workflow, fast_test, stateless_tests_amd_asan_ubsan_db_disk_distributed_plan_sequential_selected_tests, stateless_tests_amd_asan_ubsan_distributed_plan_parallel_selected_tests, stateless_tests_amd_debug_parallel, stateless_tests_amd_tsan_parallel_selected_tests, stateless_tests_arm_binary_parallel, stateless_tests_arm_binary_sequential] if: ${{ !cancelled() && !contains(needs.*.outputs.pipeline_status, 'failure') && !contains(needs.*.outputs.pipeline_status, 'undefined') && !contains(fromJson(needs.config_workflow.outputs.data).workflow_config.cache_success_base64, 'U3RhdGVsZXNzIHRlc3RzIChhbWRfYmluYXJ5LCBjYXMgc3RvcmFnZSwgcGFyYWxsZWwp') }} name: "Stateless tests (amd_binary, cas storage, parallel)" outputs: @@ -3013,7 +3102,6 @@ jobs: continue-on-error: true with: name: failure-stateless_tests_amd_binary_cas_storage_parallel ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) path: | ci/tmp/result_*.json ci/tmp/test_result.txt diff --git a/ci/defs/altinity_jobs.py b/ci/defs/altinity_jobs.py index b10dd792a307..5c8f832bdb03 100644 --- a/ci/defs/altinity_jobs.py +++ b/ci/defs/altinity_jobs.py @@ -121,53 +121,3 @@ class AltinityJobConfigs: requires=[ArtifactNames.CH_AMD_BINARY_GH], ), ) - # Stateless tests with a content-addressed disk as the default MergeTree storage. - cas_functional_tests_jobs = common_ft_job_config.parametrize( - # CAS over S3: RustFS, not MinIO OSS, because the incarnation pool needs - # enforced conditional deletes. - Job.ParamSet( - parameter="amd_binary, cas s3 storage, parallel", - runs_on=RunnerLabels.AMD_MEDIUM_CPU, - requires=[ArtifactNames.CH_AMD_BINARY_GH], - ), - # The sanitizer lanes are sharded because an unsharded one exceeds the 6h - # GitHub job timeout and is killed before it uploads any results. - *[ - Job.ParamSet( - parameter=f"amd_asan_ubsan, cas s3 storage, parallel, {batch}/{total_batches}", - runs_on=RunnerLabels.AMD_MEDIUM_CPU, - requires=[ArtifactNames.CH_AMD_ASAN_UBSAN_GH], - ) - for total_batches in (2,) - for batch in range(1, total_batches + 1) - ], - *[ - Job.ParamSet( - parameter=f"amd_tsan, cas s3 storage, parallel, {batch}/{total_batches}", - runs_on=RunnerLabels.AMD_MEDIUM, - requires=[ArtifactNames.CH_AMD_TSAN_GH], - ) - for total_batches in (2,) - for batch in range(1, total_batches + 1) - ], - *[ - Job.ParamSet( - parameter=f"amd_msan, cas s3 storage, parallel, {batch}/{total_batches}", - runs_on=RunnerLabels.FUNC_TESTER_AMD, - requires=[ArtifactNames.CH_AMD_MSAN_GH], - ) - for total_batches in (3,) - for batch in range(1, total_batches + 1) - ], - Job.ParamSet( - parameter="arm_binary, cas s3 storage, parallel", - runs_on=RunnerLabels.ARM_MEDIUM_CPU, - requires=[ArtifactNames.CH_ARM_BINARY_GH], - ), - # CAS over local object storage. - Job.ParamSet( - parameter="amd_binary, cas storage, parallel", - runs_on=RunnerLabels.AMD_MEDIUM_CPU, - requires=[ArtifactNames.CH_AMD_BINARY_GH], - ), - ) diff --git a/ci/jobs/functional_tests.py b/ci/jobs/functional_tests.py index 28e471327708..daa9750bc88a 100644 --- a/ci/jobs/functional_tests.py +++ b/ci/jobs/functional_tests.py @@ -158,13 +158,9 @@ def run_tests( "old analyzer": "--analyzer", "WasmEdge": "--wasm-engine wasmedge", "s3 storage": "--s3-storage", -<<<<<<< HEAD - "DBReplicated": "--db-replicated", -======= "cas storage": "--cas-storage", "cas s3 storage": "--cas-s3-storage", - "DatabaseReplicated": "--db-replicated", ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + "DBReplicated": "--db-replicated", "DatabaseOrdinary": "--db-ordinary", "wide parts enabled": "--wide-parts", "ParallelReplicas": "--parallel-rep", @@ -995,7 +991,6 @@ def main(): setup_notes = [] def start(): -<<<<<<< HEAD # `from_commands_run` captures this closure's stdout into the step # Result.info (hence CIDB test_context_raw) only when it returns a # failing value. Print a concise "SETUP FAILURE: " marker @@ -1005,6 +1000,15 @@ def start(): if not (CH.start_seaweedfs(test_type="stateless") and CH.start_azurite()): print("SETUP FAILURE: seaweedfs/azurite did not start") return False + if is_cas_s3: + # The CA-over-S3 pool lives on RustFS (M-W D-W8): the incarnation pool + # needs ENFORCED conditional deletes, which MinIO OSS lacks (the + # fail-closed capability probe rejects it). start_rustfs wipes its data + # dir per run, so no pool state bleeds between runs (the local-CA + # analogue is the per-run server-store wipe). MinIO keeps the non-CA + # s3 disks. + if not CH.start_rustfs(): + return False if not CH.start(): print("SETUP FAILURE: clickhouse-server process did not start") return False @@ -1013,26 +1017,6 @@ def start(): # timeout; the marker just names the sub-step for triage. print("SETUP FAILURE: clickhouse-server not ready (wait_ready)") return False -======= - res = CH.start_minio(test_type="stateless") and CH.start_azurite() - if res and is_cas_s3: - # The CA-over-S3 pool lives on RustFS (M-W D-W8): the incarnation pool - # needs ENFORCED conditional deletes, which MinIO OSS lacks (the - # fail-closed capability probe rejects it). start_rustfs wipes its data - # dir per run, so no pool state bleeds between runs (the local-CA - # analogue is the per-run server-store wipe). MinIO keeps the non-CA - # s3 disks. - res = CH.start_rustfs() - res = res and CH.start() - res = res and CH.wait_ready() - if res: - if not CH.start_kafka(): - info.add_workflow_warning("Failed to start Kafka") - print("Failed to start Kafka") - # Fail fast on infra setup errors so we don't burn time - # triaging Kafka/Avro test failures caused by a broken setup. - return False ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if not CH.start_kafka(): info.add_workflow_warning("Failed to start Kafka") diff --git a/ci/jobs/scripts/clickhouse_proc.py b/ci/jobs/scripts/clickhouse_proc.py index 2023daef88f5..4aa0ff66b6bd 100644 --- a/ci/jobs/scripts/clickhouse_proc.py +++ b/ci/jobs/scripts/clickhouse_proc.py @@ -1396,21 +1396,11 @@ def dump_system_tables(self): self.restore_system_metadata_files_from_remote_database_disk() -<<<<<<< HEAD # Caches created via the disk() function live one level deeper, under # disks//status. cache_status_files = glob.glob( f"{self.ch_var_lib_dir}/filesystem_caches/*/status" ) + glob.glob(f"{self.ch_var_lib_dir}/filesystem_caches/disks/*/status") -======= - # `**`, not `*`: dynamic cache disks created by tests nest their path, e.g. - # `filesystem_caches/disks/cache_03517/status` — a one-level glob missed exactly that file, - # and the scrape died on its flock (`StatusFile.cpp` "Another server instance ... is already - # running") when the server had not released it. - cache_status_files = glob.glob( - f"{self.ch_var_lib_dir}/filesystem_caches/**/status", recursive=True - ) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) if cache_status_files: print( f"WARNING: Server died? Removing cache status files: {cache_status_files}" diff --git a/ci/jobs/scripts/clickhouse_service.py b/ci/jobs/scripts/clickhouse_service.py index 6dc7ddaaa86f..448c7b7d4168 100644 --- a/ci/jobs/scripts/clickhouse_service.py +++ b/ci/jobs/scripts/clickhouse_service.py @@ -61,68 +61,6 @@ def install_base(config_dir, var_lib_dir) -> None: ) def __enter__(self): -<<<<<<< HEAD -======= - Utils.add_to_PATH(temp_dir) - - # Download binary if absent - clickhouse_bin = Path(temp_dir) / "clickhouse" - if not clickhouse_bin.exists(): - self._download_binary() - - # Create symlinks if absent - for link_name in ("clickhouse-server", "clickhouse-client", "clickhouse-local"): - link_path = Path(temp_dir) / link_name - if not link_path.exists(): - Utils.link(clickhouse_bin, link_path) - - # Copy server config files if absent - config_dir = Path(self.ch_config_dir) - if not (config_dir / "config.xml").exists(): - config_dir.mkdir(parents=True, exist_ok=True) - src_dir = Path("./programs/server") - for name in ("config.xml", "users.xml"): - shutil.copy(src_dir / name, config_dir / name) - shutil.copytree( - src_dir / "config.d", - config_dir / "config.d", - symlinks=False, - dirs_exist_ok=True, - ) - - # Recreate data directory so it is owned by the current process user. - # If the directory was created on the host by a different UID (e.g. 501 - # on macOS) and the server runs as root inside Docker, ClickHouse raises - # MISMATCHING_USERS_FOR_PROCESS_AND_DATA and refuses to start. - if Path(self.run_path).exists(): - shutil.rmtree(self.run_path) - Path(self.run_path).mkdir(parents=True, exist_ok=True) - Path(self.log_dir).mkdir(parents=True, exist_ok=True) - Path(self.pid_file).unlink(missing_ok=True) - - argv = [ - str(Path(temp_dir) / "clickhouse-server"), - "--config-file", self.config_file, - "--pid-file", self.pid_file, - "--", - "--path", self.run_path, - "--user_files_path", self.user_files_path, - "--top_level_domains_path", f"{self.ch_config_dir}/top_level_domains", - "--logger.stderr", f"{self.log_dir}/stderr.log", - # NOTE (strtgbb): master binary rejects unknown cas_log keys (ErrorCodes 137) - "--skip_check_for_incorrect_settings", "1", - ] - print(f"Starting ClickHouse server: {shlex.join(argv)}") - with open(f"{self.log_dir}/clickhouse-server.log", "w") as log_fd: - self._proc = subprocess.Popen( - argv, - stderr=subprocess.STDOUT, - stdout=log_fd, - start_new_session=True, - cwd=self.run_path, - ) - ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) try: Utils.add_to_PATH(temp_dir) diff --git a/ci/workflows/pull_request.py b/ci/workflows/pull_request.py index 217bef7a4a56..41ea2b41bba3 100644 --- a/ci/workflows/pull_request.py +++ b/ci/workflows/pull_request.py @@ -14,7 +14,6 @@ from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job from ci.jobs.scripts.workflow_hooks.trusted import can_be_tested -<<<<<<< HEAD # Functional tests with sanitizers are trimmed down in pull requests: instead of # the full suite, their `selected tests` counterparts run only the tests selected # for the change. The full suite still runs here in the debug and plain binary @@ -31,21 +30,11 @@ # discovery cannot provide a representative WasmEdge smoke test. Keep the # established full-suite MSan/WasmEdge lanes until that coverage exists. or "amd_msan, WasmEdge" in job.name -] + JobConfigs.stateless_tests_selected_pr_jobs +] + JobConfigs.stateless_tests_selected_pr_jobs + AltinityJobConfigs.cas_functional_tests_jobs ALL_FUNCTIONAL_TESTS = [job.name for job in FUNCTIONAL_TESTS_JOBS] CORE_BLOCKING_JOB_NAMES = [ -======= -FUNCTIONAL_TESTS_JOBS = [ - *JobConfigs.functional_tests_jobs, - *AltinityJobConfigs.cas_functional_tests_jobs, -] - -ALL_FUNCTIONAL_TESTS = [job.name for job in FUNCTIONAL_TESTS_JOBS] - -FUNCTIONAL_TESTS_PARALLEL_BLOCKING_JOB_NAMES = [ ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) job.name for job in FUNCTIONAL_TESTS_JOBS if any( @@ -71,11 +60,6 @@ STYLE_AND_FAST_TESTS = [ # JobNames.STYLE_CHECK, JobNames.FAST_TEST, -<<<<<<< HEAD -======= - # NOTE (strtgbb): CI_TESTS temporarily not gating builds (allow_failure + 137 OOM during setup) - # JobNames.CI_TESTS, ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) # *[j.name for j in JobConfigs.tidy_build_arm_jobs], ] diff --git a/ci/workflows/pull_request_community.py b/ci/workflows/pull_request_community.py index 552eb43fcf5d..047900ff1570 100644 --- a/ci/workflows/pull_request_community.py +++ b/ci/workflows/pull_request_community.py @@ -6,7 +6,6 @@ from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job -<<<<<<< HEAD # Functional tests with sanitizers are trimmed down in pull requests: instead of # the full suite, their `selected tests` counterparts run only the tests selected # for the change. Keep this aligned with `ci/workflows/pull_request.py`. @@ -21,13 +20,7 @@ # discovery cannot provide a representative WasmEdge smoke test. Keep the # established full-suite MSan/WasmEdge lanes until that coverage exists. or "amd_msan, WasmEdge" in job.name -] + JobConfigs.stateless_tests_selected_pr_jobs -======= -FUNCTIONAL_TESTS_JOBS = [ - *JobConfigs.functional_tests_jobs, - *AltinityJobConfigs.cas_functional_tests_jobs, -] ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) +] + JobConfigs.stateless_tests_selected_pr_jobs + AltinityJobConfigs.cas_functional_tests_jobs FUNCTIONAL_TESTS_PARALLEL_BLOCKING_JOB_NAMES = [ job.name diff --git a/ci/workflows/release_branches.py b/ci/workflows/release_branches.py index fb030a1eb0b4..9bd2a3448731 100644 --- a/ci/workflows/release_branches.py +++ b/ci/workflows/release_branches.py @@ -1,6 +1,5 @@ from praktika import Workflow -<<<<<<< HEAD from ci.defs.defs import ( BINARIES_WITH_LONG_RETENTION, DOCKERS, @@ -8,10 +7,7 @@ SECRETS, ArtifactConfigs, ) -======= -from ci.defs.defs import BINARIES_WITH_LONG_RETENTION, DOCKERS, SECRETS, ArtifactConfigs from ci.defs.altinity_jobs import AltinityJobConfigs ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) from ci.defs.job_configs import JobConfigs from ci.jobs.scripts.workflow_hooks.filter_job import should_skip_job diff --git a/docs/concepts/features/configuration/server-config/storing-data.mdx b/docs/concepts/features/configuration/server-config/storing-data.mdx index 080525b89666..ae854dd3a88a 100644 --- a/docs/concepts/features/configuration/server-config/storing-data.mdx +++ b/docs/concepts/features/configuration/server-config/storing-data.mdx @@ -49,13 +49,8 @@ It requires specifying:
-<<<<<<< HEAD:docs/concepts/features/configuration/server-config/storing-data.mdx -Optionally, `metadata_type` can be specified (it is equal to `local` by default), but it can also be set to `plain`, `web` and, starting from `24.4`, `plain_rewritable`. -Usage of `plain` metadata type is described in [plain storage section](/concepts/features/configuration/server-config/storing-data#plain-storage), `web` metadata type can be used only with `web` object storage type, `local` metadata type stores metadata files locally (each metadata files contains mapping to files in object storage and some additional meta information about them). -======= Optionally, `metadata_type` can be specified (it is equal to `local` by default), but it can also be set to `plain`, `web`, `plain_rewritable` (starting from `24.4`) and `cas`. -Usage of `plain` metadata type is described in [plain storage section](/operations/storing-data#plain-storage), `web` metadata type can be used only with `web` object storage type, `local` metadata type stores metadata files locally (each metadata files contains mapping to files in object storage and some additional meta information about them). ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS):docs/en/operations/storing-data.md +Usage of `plain` metadata type is described in [plain storage section](/concepts/features/configuration/server-config/storing-data#plain-storage), `web` metadata type can be used only with `web` object storage type, `local` metadata type stores metadata files locally (each metadata files contains mapping to files in object storage and some additional meta information about them). For example: diff --git a/programs/disks/DisksApp.cpp b/programs/disks/DisksApp.cpp index 66e1e4dbef6b..39f15a9a6f7e 100644 --- a/programs/disks/DisksApp.cpp +++ b/programs/disks/DisksApp.cpp @@ -342,16 +342,13 @@ void DisksApp::registerCommands() command_descriptions.emplace("switch-disk", makeCommandSwitchDisk()); command_descriptions.emplace("current_disk_with_path", makeCommandGetCurrentDiskAndPath()); command_descriptions.emplace("touch", makeCommandTouch()); -<<<<<<< HEAD command_descriptions.emplace("du", makeCommandDiskUsage()); command_descriptions.emplace("wc", makeCommandWordCount()); -======= command_descriptions.emplace("cas-fsck", makeCommandFsck()); command_descriptions.emplace("cas-gc-dryrun", makeCommandCaGcDryRun()); command_descriptions.emplace("cas-gc-rebuild", makeCommandCaGcRebuild()); command_descriptions.emplace("cas-inspect", makeCommandCaInspect()); command_descriptions.emplace("cas-drop-member", makeCommandCaDropMember()); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) command_descriptions.emplace("read-checksums", makeCommandReadChecksums()); command_descriptions.emplace("help", makeCommandHelp(*this)); command_descriptions.emplace("packed-io", makeCommandPackedIO()); diff --git a/programs/disks/ICommand.h b/programs/disks/ICommand.h index 0bfb8a3d2ae9..ad840494e8cb 100644 --- a/programs/disks/ICommand.h +++ b/programs/disks/ICommand.h @@ -133,16 +133,13 @@ DB::CommandPtr makeCommandSwitchDisk(); DB::CommandPtr makeCommandGetCurrentDiskAndPath(); DB::CommandPtr makeCommandHelp(const DisksApp & disks_app); DB::CommandPtr makeCommandTouch(); -<<<<<<< HEAD DB::CommandPtr makeCommandDiskUsage(); DB::CommandPtr makeCommandWordCount(); -======= DB::CommandPtr makeCommandFsck(); DB::CommandPtr makeCommandCaGcDryRun(); DB::CommandPtr makeCommandCaGcRebuild(); DB::CommandPtr makeCommandCaInspect(); DB::CommandPtr makeCommandCaDropMember(); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) DB::CommandPtr makeCommandReadChecksums(); DB::CommandPtr makeCommandPackedIO(); } diff --git a/programs/server/Server.cpp b/programs/server/Server.cpp index 792313e8e354..9a071177a008 100644 --- a/programs/server/Server.cpp +++ b/programs/server/Server.cpp @@ -113,11 +113,7 @@ #include #include #include -<<<<<<< HEAD -======= #include -#include ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include #include diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 276389051430..02b669dced58 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -349,7 +349,6 @@ static struct InitFiu PAUSEABLE_ONCE(iceberg_compaction_pause_before_metadata_commit) \ REGULAR(tcp_handler_fail_connection_setup) \ REGULAR(distributed_plan_status_check_reenqueue_fault) \ -<<<<<<< HEAD PAUSEABLE(keeper_changelog_read_plan_resolved) \ PAUSEABLE(keeper_changelog_removed_from_disk_set) \ PAUSEABLE(keeper_changelog_readahead_fill_wedge) \ @@ -370,11 +369,9 @@ static struct InitFiu PAUSEABLE_ONCE(limit_by_transform_mid_loop_pause) \ PAUSEABLE_ONCE(aggregating_in_order_transform_mid_loop_pause) \ REGULAR(smt_force_takeover_predicate_true) \ - REGULAR(smt_takeover_fake_hardware_error_after_set) -======= + REGULAR(smt_takeover_fake_hardware_error_after_set) \ REGULAR(cas_relink_receiver_force_mechanism_failure) \ PAUSEABLE_ONCE(cas_relink_receiver_pause_before_confirm) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) namespace FailPoints { diff --git a/src/Common/ThreadStatus.h b/src/Common/ThreadStatus.h index 35c4aa5b9457..c8f60bcd18d5 100644 --- a/src/Common/ThreadStatus.h +++ b/src/Common/ThreadStatus.h @@ -96,19 +96,7 @@ class ThreadGroup const Int32 os_threads_nice_value; -<<<<<<< HEAD MemorySpillSchedulerPtr memory_spill_scheduler; -======= - /// A borrowed child group (materialized view / async-insert flush) parents its `memory_tracker` - /// and `performance_counters` at the parent group's trackers via RAW pointers. Retain a shared_ptr - /// to the parent so those trackers cannot be freed while any thread is still attached to this child - /// group — otherwise a detached task (e.g. an S3 upload scheduled via `threadPoolCallbackRunnerUnsafe`) - /// that attaches the child group can walk a freed parent tracker chain (use-after-free, B90). Null for - /// a top-level query/background group, whose parent is a process-lifetime tracker (user/total/background). - ThreadGroupPtr parent_thread_group; - - MemorySpillScheduler::Ptr memory_spill_scheduler; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) ProfileEvents::Counters performance_counters{VariableContext::Process}; MemoryTracker memory_tracker{VariableContext::Process}; diff --git a/src/Core/ServerSettings.cpp b/src/Core/ServerSettings.cpp index 871bd4533fa9..8136f3c869a1 100644 --- a/src/Core/ServerSettings.cpp +++ b/src/Core/ServerSettings.cpp @@ -182,18 +182,13 @@ A value of `0` means unlimited. ::: )", 0) \ DECLARE(UInt64, max_format_parsing_thread_pool_size, 100, R"( -<<<<<<< HEAD Maximum total number of threads to use for parsing input. )", 0) \ -======= - Maximum total number of threads to use for parsing input. - )", 0) \ DECLARE(UInt64, cas_blob_upload_pool_size, 16, R"( ClickHouse uses threads from this dedicated server-wide pool to upload blobs in parallel when committing a content-addressed (CAS) part. `cas_blob_upload_pool_size` limits the maximum number of threads in the pool. Zero is rejected: the pool must have at least one thread. )", 0) \ ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) DECLARE(UInt64, max_format_parsing_thread_pool_free_size, 0, R"( Maximum number of idle standby threads to keep in the thread pool for parsing input. )", 0) \ diff --git a/src/Disks/DiskObjectStorage/DiskObjectStorage.cpp b/src/Disks/DiskObjectStorage/DiskObjectStorage.cpp index 822a8f350bf9..1050fb30434e 100644 --- a/src/Disks/DiskObjectStorage/DiskObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/DiskObjectStorage.cpp @@ -805,8 +805,6 @@ bool DiskObjectStorage::supportsHardLinks() const return !metadata_storage->isWriteOnce() && !metadata_storage->isPlain(); } -<<<<<<< HEAD -======= bool DiskObjectStorage::isContentAddressed() const { return metadata_storage->isContentAddressed(); @@ -817,8 +815,6 @@ bool DiskObjectStorage::supportsAtomicFileWrites() const return metadata_storage->supportsAtomicFileWrites(); } - ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) String DiskObjectStorage::getReadResourceName() const { std::unique_lock lock(resource_mutex); @@ -853,7 +849,6 @@ void DiskObjectStorage::prepareRead( std::optional read_hint, ReadPipeline & pipeline) const { -<<<<<<< HEAD /// A small file may be stored inline in its metadata; serve the content directly. if (metadata_storage->supportsInlineData()) { @@ -874,8 +869,6 @@ void DiskObjectStorage::prepareRead( } } - const StoredObjects storage_objects = metadata_storage->getStorageObjects(path); -======= /// Content-addressed reads: in-manifest bytes come from memory (no object exists); a /// blob-backed part file translates to its physical blob object + a payload window, which /// rides the STANDARD pipeline below (gather/caches/async prefetch — same chain as plain @@ -896,7 +889,6 @@ void DiskObjectStorage::prepareRead( const auto storage_objects = ca_blob_view ? StoredObjects{ca_blob_view->object} : metadata_storage->getStorageObjects(path); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) auto read_settings = updateIOSchedulingSettings(settings, getReadResourceName(), getWriteResourceName()); auto global_context = Context::getGlobalContextInstance(); diff --git a/src/Disks/DiskObjectStorage/DiskObjectStorage.h b/src/Disks/DiskObjectStorage/DiskObjectStorage.h index c5c015cfdc11..7ff33b2c08f2 100644 --- a/src/Disks/DiskObjectStorage/DiskObjectStorage.h +++ b/src/Disks/DiskObjectStorage/DiskObjectStorage.h @@ -51,22 +51,16 @@ friend class DiskObjectStorageReservation; DataSourceDescription getDataSourceDescription() const override { return data_source_description; } -<<<<<<< HEAD /// Keeper metadata replicates itself; in-memory metadata is transient and has no local /// metadata files zero-copy could ship (see `getReplicatedFilesDescriptionForRemoteDisk`). - bool supportZeroCopyReplication() const override - { - return metadata_storage->getType() != MetadataStorageType::Keeper - && metadata_storage->getType() != MetadataStorageType::Memory; -======= /// A content-addressed pool deduplicates blobs across parts and has no per-replica unique blob /// ids; the zero-copy subsystem (B1) is explicitly out of scope for M1, so advertise it as /// unsupported (honest capability — B31). Other object-storage metadata types keep the old rule. bool supportZeroCopyReplication() const override { return metadata_storage->getType() != MetadataStorageType::Keeper + && metadata_storage->getType() != MetadataStorageType::Memory && metadata_storage->getType() != MetadataStorageType::CAS; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } bool supportParallelWrite() const override { return object_storages->takePointingTo(cluster->getLocalLocation())->supportParallelWrite(); } diff --git a/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.cpp b/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.cpp index eab4a61ff899..c20e854fe980 100644 --- a/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.cpp +++ b/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.cpp @@ -110,7 +110,7 @@ MultipleDisksObjectStorageTransaction::MultipleDisksObjectStorageTransaction( void DiskObjectStorageTransaction::addOperation(std::function op) { - if (metadata_storage->appliesOperationsEagerly()) + if (metadata_storage->appliesOperationsEagerly() || metadata_storage->transactionIsStagingOverlay()) op(metadata_transaction); else operations_to_execute.push_back(std::move(op)); @@ -118,11 +118,7 @@ void DiskObjectStorageTransaction::addOperation(std::function>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->createDirectory(path); }); @@ -130,11 +126,7 @@ void DiskObjectStorageTransaction::createDirectory(const std::string & path) void DiskObjectStorageTransaction::createDirectories(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->createDirectoryRecursive(path); }); @@ -142,11 +134,7 @@ void DiskObjectStorageTransaction::createDirectories(const std::string & path) void DiskObjectStorageTransaction::moveDirectory(const std::string & from_path, const std::string & to_path) { -<<<<<<< HEAD addOperation([from_path, to_path](MetadataTransactionPtr tx) -======= - dispatch([from_path, to_path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->moveDirectory(from_path, to_path); }); @@ -154,11 +142,7 @@ void DiskObjectStorageTransaction::moveDirectory(const std::string & from_path, void DiskObjectStorageTransaction::moveFile(const String & from_path, const String & to_path) { -<<<<<<< HEAD addOperation([from_path, to_path](MetadataTransactionPtr tx) -======= - dispatch([from_path, to_path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->moveFile(from_path, to_path); }); @@ -166,11 +150,7 @@ void DiskObjectStorageTransaction::moveFile(const String & from_path, const Stri void DiskObjectStorageTransaction::truncateFile(const String & path, size_t size) { -<<<<<<< HEAD addOperation([path, size](MetadataTransactionPtr tx) -======= - dispatch([path, size](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->truncateFile(path, size); }); @@ -194,11 +174,7 @@ void DiskObjectStorageTransaction::decrementBlobRefCount(const std::string & blo void DiskObjectStorageTransaction::replaceFile(const std::string & from_path, const std::string & to_path) { -<<<<<<< HEAD addOperation([from_path, to_path](MetadataTransactionPtr tx) -======= - dispatch([from_path, to_path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->replaceFile(from_path, to_path); }); @@ -206,11 +182,7 @@ void DiskObjectStorageTransaction::replaceFile(const std::string & from_path, co void DiskObjectStorageTransaction::removeFile(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->unlinkFile(path, /*if_exists=*/false, /*should_remove_objects=*/true); }); @@ -218,11 +190,7 @@ void DiskObjectStorageTransaction::removeFile(const std::string & path) void DiskObjectStorageTransaction::removeSharedFile(const std::string & path, bool keep_shared_data) { -<<<<<<< HEAD addOperation([path, keep_shared_data](MetadataTransactionPtr tx) -======= - dispatch([path, keep_shared_data](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->unlinkFile(path, /*if_exists=*/false, /*should_remove_objects=*/!keep_shared_data); }); @@ -233,22 +201,14 @@ void DiskObjectStorageTransaction::removeSharedRecursive( { if (!keep_all_shared_data && file_names_remove_metadata_only.empty()) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->removeRecursive(path, /*should_remove_objects=*/nullptr); }); } else { -<<<<<<< HEAD addOperation([path, keep_all_shared_data, file_names_remove_metadata_only](MetadataTransactionPtr tx) -======= - dispatch([path, keep_all_shared_data, file_names_remove_metadata_only](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->removeRecursive(path, /*should_remove_objects=*/[keep_all_shared_data, file_names_remove_metadata_only](const std::string & relative_path) { @@ -260,11 +220,7 @@ void DiskObjectStorageTransaction::removeSharedRecursive( void DiskObjectStorageTransaction::removeSharedFileIfExists(const std::string & path, bool keep_shared_data) { -<<<<<<< HEAD addOperation([path, keep_shared_data](MetadataTransactionPtr tx) -======= - dispatch([path, keep_shared_data](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->unlinkFile(path, /*if_exists=*/true, /*should_remove_objects=*/!keep_shared_data); }); @@ -272,11 +228,7 @@ void DiskObjectStorageTransaction::removeSharedFileIfExists(const std::string & void DiskObjectStorageTransaction::removeDirectory(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->removeDirectory(path); }); @@ -284,11 +236,7 @@ void DiskObjectStorageTransaction::removeDirectory(const std::string & path) void DiskObjectStorageTransaction::removeRecursive(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->removeRecursive(path, /*should_remove_objects=*/nullptr); }); @@ -296,11 +244,7 @@ void DiskObjectStorageTransaction::removeRecursive(const std::string & path) void DiskObjectStorageTransaction::removeFileIfExists(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->unlinkFile(path, /*if_exists=*/true, /*should_remove_objects=*/true); }); @@ -311,11 +255,7 @@ void DiskObjectStorageTransaction::removeSharedFiles(const RemoveBatchRequest & for (const auto & [path, if_exists] : files) { const bool should_remove_objects = !keep_all_batch_data && !file_names_remove_metadata_only.contains(fs::path(path).filename()); -<<<<<<< HEAD addOperation([path, if_exists, should_remove_objects](MetadataTransactionPtr tx) -======= - dispatch([path, if_exists, should_remove_objects](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->unlinkFile(path, if_exists, should_remove_objects); }); @@ -528,14 +468,10 @@ void DiskObjectStorageTransaction::writeFileUsingBlobWritingFunction( /// We always use mode Rewrite because we simulate append using metadata and different files object.bytes_size = std::move(write_blob_function)(blob_path, WriteMode::Rewrite, /*object_attributes=*/std::nullopt); -<<<<<<< HEAD - addOperation([object, mode](MetadataTransactionPtr tx) -======= - /// [TXN-ONE-PIPELINE] Routed through dispatch for uniformity. Unreachable on CA (Audit 6): + /// [TXN-ONE-PIPELINE] Routed through addOperation for uniformity. Unreachable on CA (Audit 6): /// generateObjectKeyForPath above throws NOT_IMPLEMENTED first, so CA never reaches this metadata - /// effect and never queues. On ordinary storage dispatch queues exactly as before. - dispatch([object, mode](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + /// effect and never queues. On ordinary storage addOperation queues exactly as before. + addOperation([object, mode](MetadataTransactionPtr tx) { if (mode == WriteMode::Rewrite) { @@ -553,17 +489,13 @@ void DiskObjectStorageTransaction::writeFileUsingBlobWritingFunction( void DiskObjectStorageTransaction::createHardLink(const std::string & src_path, const std::string & dst_path) { -<<<<<<< HEAD - addOperation([src_path, dst_path](MetadataTransactionPtr tx) -======= - /// For CA `dispatch` runs eagerly (call-time), which is load-bearing for read-your-writes: a + /// For CA `addOperation` runs eagerly (call-time), which is load-bearing for read-your-writes: a /// carried-forward projection hardlinked into the open whole-part transaction during a mutation must /// be visible to `loadProjections` (same finalize, before commit) via the directory overlay. Deferring /// it to commit replay would hide it until after `loadProjections` ran (B58/B63). The metadata-level /// `createHardLink` is an idempotent map assignment, so eager staging is equivalent to the queued /// replay — commit publishes the manifest from the staging. - dispatch([src_path, dst_path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + addOperation([src_path, dst_path](MetadataTransactionPtr tx) { tx->createHardLink(src_path, dst_path); }); @@ -597,11 +529,7 @@ std::vector DiskObjectStorageTransaction::listInFlightDirectory(con void DiskObjectStorageTransaction::setReadOnly(const std::string & path) { -<<<<<<< HEAD addOperation([path](MetadataTransactionPtr tx) -======= - dispatch([path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->setReadOnly(path); }); @@ -609,11 +537,7 @@ void DiskObjectStorageTransaction::setReadOnly(const std::string & path) void DiskObjectStorageTransaction::setLastModified(const std::string & path, const Poco::Timestamp & timestamp) { -<<<<<<< HEAD addOperation([path, timestamp](MetadataTransactionPtr tx) -======= - dispatch([path, timestamp](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->setLastModified(path, timestamp); }); @@ -621,11 +545,7 @@ void DiskObjectStorageTransaction::setLastModified(const std::string & path, con void DiskObjectStorageTransaction::chmod(const String & path, mode_t mode) { -<<<<<<< HEAD addOperation([path, mode](MetadataTransactionPtr tx) -======= - dispatch([path, mode](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) { tx->chmod(path, mode); }); @@ -723,15 +643,11 @@ void DiskObjectStorageTransaction::copyFileImpl( return; } -<<<<<<< HEAD - addOperation([blobs_to_create, missing_locations, to_file_path](MetadataTransactionPtr tx) -======= - /// [TXN-ONE-PIPELINE] Routed through dispatch for uniformity. Unreachable on CA (Audit 6): + /// [TXN-ONE-PIPELINE] Routed through addOperation for uniformity. Unreachable on CA (Audit 6): /// copyFileImpl calls generateObjectKeyForPath above, which throws NOT_IMPLEMENTED on CA before this /// point (and the empty-source case returns via the real writeFile above), so CA never queues here. - /// On ordinary storage dispatch queues exactly as before. - dispatch([blobs_to_create, missing_locations, to_file_path](MetadataTransactionPtr tx) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + /// On ordinary storage addOperation queues exactly as before. + addOperation([blobs_to_create, missing_locations, to_file_path](MetadataTransactionPtr tx) { for (const auto & blob : blobs_to_create) tx->recordBlobsReplication(blob, missing_locations); @@ -754,12 +670,12 @@ void DiskObjectStorageTransaction::commit() { /// [TXN-ONE-PIPELINE] An eager staging-overlay transaction (e.g. CA) must route every mutating method /// straight to the metadata transaction at call time and keep this queue empty. A non-empty queue here - /// means a mutating method bypassed `dispatch` — which would re-introduce the two-timeline split this + /// means a mutating method bypassed `addOperation` — which would re-introduce the two-timeline split this /// design eliminates. Fail closed with a real throw (NOT chassert, which is a no-op in release builds). if (metadata_storage->transactionIsStagingOverlay() && !operations_to_execute.empty()) throw Exception(ErrorCodes::LOGICAL_ERROR, "An eager staging-overlay transaction must not queue deferred operations " - "(a mutating method bypassed dispatch): {} queued", operations_to_execute.size()); + "(a mutating method bypassed addOperation): {} queued", operations_to_execute.size()); auto component_guard = Coordination::setCurrentComponent("DiskObjectStorageTransaction::commit"); chassert(operations_to_execute.empty() || !metadata_storage->appliesOperationsEagerly()); @@ -806,17 +722,14 @@ void DiskObjectStorageTransaction::commit() TransactionCommitOutcomeVariant DiskObjectStorageTransaction::tryCommit(const TransactionCommitOptionsVariant & options) { -<<<<<<< HEAD - chassert(operations_to_execute.empty() || !metadata_storage->appliesOperationsEagerly()); -======= /// [TXN-ONE-PIPELINE] See commit(): an eager staging-overlay transaction must never queue deferred - /// operations. Fail closed (real throw, not chassert) if a mutating method bypassed `dispatch`. + /// operations. Fail closed (real throw, not chassert) if a mutating method bypassed `addOperation`. if (metadata_storage->transactionIsStagingOverlay() && !operations_to_execute.empty()) throw Exception(ErrorCodes::LOGICAL_ERROR, "An eager staging-overlay transaction must not queue deferred operations " - "(a mutating method bypassed dispatch): {} queued", operations_to_execute.size()); + "(a mutating method bypassed addOperation): {} queued", operations_to_execute.size()); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + chassert(operations_to_execute.empty() || !metadata_storage->appliesOperationsEagerly()); for (size_t i = 0; i < operations_to_execute.size(); ++i) { try diff --git a/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.h b/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.h index e524b35231cd..c720c011fde1 100644 --- a/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.h +++ b/src/Disks/DiskObjectStorage/DiskObjectStorageTransaction.h @@ -144,17 +144,6 @@ struct DiskObjectStorageTransaction : public IDiskTransaction, public std::enabl const ReadSettings & read_settings, const WriteSettings & write_settings); - /// [TXN-ONE-PIPELINE] Route one metadata effect either into the FIFO replay queue (ordinary object - /// storage) or straight to the metadata transaction at call time (eager staging overlay, e.g. CA). - template - void dispatch(Operation && operation) - { - if (metadata_storage->transactionIsStagingOverlay()) - operation(metadata_transaction); - else - operations_to_execute.emplace_back(std::forward(operation)); - } - private: std::unique_ptr writeFileImpl( /// NOLINT bool autocommit, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h b/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h index d22a0ed0f0f7..27dfec46900a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/IMetadataStorage.h @@ -153,7 +153,6 @@ class IMetadataTransaction : private boost::noncopyable throwNotImplemented(); } -<<<<<<< HEAD /// Increment the reference count of a data blob shared between metadata files. virtual void incrementBlobRefCount(const std::string & /* blob */) { @@ -171,7 +170,7 @@ class IMetadataTransaction : private boost::noncopyable { throwNotImplemented(); } -======= + /// In-flight read-your-writes for a part being assembled by THIS transaction (B59). A CA part-build /// transaction stages blobs (uploaded) + mutable bytes before the single commit; these let a reader /// that holds the transaction resolve those staged files before they are committed. Default: no @@ -187,7 +186,6 @@ class IMetadataTransaction : private boost::noncopyable /// Immediate-child names staged directly under `path` (one level). Used so loadProjections' /// withPartFormatFromDisk can iterate a staged projection dir to find its mark file. Default: empty. virtual std::vector listInFlightDirectory(const std::string & /*path*/) const { return {}; } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) virtual ~IMetadataTransaction() = default; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp index 0dbe2048e81e..e1fd3da946a3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/MetadataStorageFactory.cpp @@ -7,12 +7,9 @@ #endif #include #include -<<<<<<< HEAD -#include -======= #include #include ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) +#include #include #include #include diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index 42b02114fe14..ef2aac33c2be 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -489,7 +489,6 @@ class IObjectStorage virtual bool supportParallelWrite() const { return false; } -<<<<<<< HEAD /// Whether a fetched `ObjectMetadata` is guaranteed to carry at least one comparable generation /// token — a non-empty `etag`, a known size, or a known modification time — so that two fetches /// of the same path can prove the object was not overwritten in between. Web origins may @@ -497,7 +496,7 @@ class IObjectStorage /// that must reread the same generation of an object (e.g. lazy materialization) have to skip /// such storages instead of failing close at read time. virtual bool supportsObjectGenerationComparison() const { return true; } -======= + /// True when the incarnation tokens this storage returns from writes/HEADs are GCS generation /// numbers riding the ETag plumbing (http_client = gcs_hmac or gcp_oauth). /// Consumers (the CAS backend) stamp TokenType::Generation and route conditional writes @@ -531,7 +530,6 @@ class IObjectStorage { return mode == ObjectStorageCopyMode::Default; } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) virtual ReadSettings patchSettings(const ReadSettings & read_settings) const; diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp index afc0ae4c65f1..bb15df679c15 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/Local/LocalObjectStorage.cpp @@ -725,6 +725,13 @@ std::optional LocalObjectStorage::tryGetObjectMetadata(const std auto resolved_path = resolvePathRelativelyToKeyPrefix(path); LOG_TEST(log, "Getting metadata for path: {}", resolved_path); + /// A directory is not an object: fs::file_size would throw "Is a directory". Treat it as a + /// missing object (nullopt) so callers probing whether a path is a readable object do not get + /// a raw filesystem error (B38: system.remote_data_paths traversal on a CAS pool). + std::error_code error; + if (fs::is_directory(resolved_path, error)) + return {}; + return tryStatResolvedPath(resolved_path); } @@ -788,30 +795,7 @@ ObjectMetadata LocalObjectStorage::getObjectMetadata(const std::string & path, b throw fs::filesystem_error( "Got unexpected error while getting file metadata", resolved_path, std::error_code(errno, std::generic_category())); -<<<<<<< HEAD return makeObjectMetadata(file_stat); -======= - object_metadata.size_bytes = fs::file_size(resolved_path); - object_metadata.etag = std::to_string(std::chrono::duration_cast(time.time_since_epoch()).count()); - object_metadata.last_modified = Poco::Timestamp::fromEpochTime( - std::chrono::duration_cast(time.time_since_epoch()).count()); - return object_metadata; -} - -std::optional LocalObjectStorage::tryGetObjectMetadata(const std::string & path, bool) const -{ - auto resolved_path = resolvePathRelativelyToKeyPrefix(path); - LOG_TEST(log, "Getting metadata for path: {}", resolved_path); - - /// A directory is not an object: fs::file_size would throw "Is a directory". Treat it as a - /// missing object (nullopt) so callers probing whether a path is a readable object do not get - /// a raw filesystem error (B38: system.remote_data_paths traversal on a CAS pool). - std::error_code error; - if (fs::is_directory(resolved_path, error)) - return {}; - - return tryStatResolvedPath(resolved_path); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } void LocalObjectStorage::listObjects(const std::string & path, RelativePathsWithMetadata & children, size_t/* max_keys */) const diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index 65fb7753ca69..bfe6b7a7b988 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -906,7 +906,6 @@ void S3ObjectStorage::applyNewSettings( modified_settings->request_settings.proxy_resolver = DB::ProxyConfigurationResolverProvider::getFromOldSettingsFormat( ProxyConfiguration::protocolFromString(uri.uri.getScheme()), config_prefix, config); -<<<<<<< HEAD /// The effective credentials of a non-disk S3 storage depend on the accessing session's restriction mode /// (`s3_allow_server_credentials_in_user_queries`), not only on the stored settings. Rebuild the client when /// that mode differs from the one the current client was built under, so an opt-in session cannot leave a @@ -915,7 +914,7 @@ void S3ObjectStorage::applyNewSettings( /// constant and this adds no rebuilds. const bool restricts_now = !for_disk_s3 && context->shouldRestrictUserQueryS3Credentials(); const bool restriction_mode_changed = client_restricts_server_credentials != restricts_now; -======= + /// A caller that derived persistent state from the conditional-ops dialect pinned it (see /// `IObjectStorage::pinConditionalOpsGenerationDialect`). Refuse before the client is replaced, so a /// rejected reload leaves the working client and its dialect in place. @@ -940,7 +939,6 @@ void S3ObjectStorage::applyNewSettings( modified_settings->auth_settings[S3AuthSetting::http_client].value, would_be_generation ? "generation" : "ETag"); } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) auto current_settings = s3_settings.get(); /// A change in the accessing session's restriction mode forces a client rebuild even for an otherwise static diff --git a/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp b/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp index f40cb2fc2eaf..c0fc0689065b 100644 --- a/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/RegisterDiskObjectStorage.cpp @@ -6,19 +6,12 @@ #include #include #include -#include -#include #include namespace DB { -namespace ErrorCodes -{ - extern const int BAD_ARGUMENTS; -} - void registerObjectStorages(); void registerMetadataStorages(); void registerDiskObjectStorage(DiskFactory & factory, bool global_skip_access_check); @@ -84,24 +77,6 @@ void registerDiskObjectStorage(DiskFactory & factory, bool global_skip_access_ch LOG_DEBUG(getLogger("registerDiskObjectStorage"), "Metadata type hint: {}", compatibility_metadata_type_hint); auto metadata_storage = MetadataStorageFactory::instance().create(name, config, config_prefix, cluster, object_storages, compatibility_metadata_type_hint); -<<<<<<< HEAD -======= - /// Content-addressed metadata (like Keeper) requires real, deferred disk transactions: a part's - /// file->blob mappings are accumulated across the whole part write and the manifest + ref are - /// published atomically when the transaction commits. A fake (per-file autocommit) transaction - /// would write each file independently with no commit point for the manifest/ref publish. - const auto metadata_type = metadata_storage->getType(); - const bool needs_real_transaction = metadata_type == MetadataStorageType::Keeper - || metadata_type == MetadataStorageType::CAS; - /// An explicit `use_fake_transaction=true` on a metadata type that requires deferred - /// transactions would silently break the atomic manifest/ref publish (per-file autocommit, - /// no commit point). Reject it instead of honoring it. - if (needs_real_transaction && config.getBool(config_prefix + ".use_fake_transaction", false)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Disk '{}': `use_fake_transaction` cannot be enabled for metadata type '{}'", - name, magic_enum::enum_name(metadata_type)); - bool use_fake_transaction = config.getBool(config_prefix + ".use_fake_transaction", !needs_real_transaction); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) DiskPtr disk = std::make_shared( name, std::move(cluster), diff --git a/src/Disks/IDiskTransaction.h b/src/Disks/IDiskTransaction.h index 3c57c9f63fe4..0a8a9ee3470e 100644 --- a/src/Disks/IDiskTransaction.h +++ b/src/Disks/IDiskTransaction.h @@ -139,7 +139,6 @@ struct IDiskTransaction : private boost::noncopyable /// Truncate file to the target size. virtual void truncateFile(const std::string & src_path, size_t size) = 0; -<<<<<<< HEAD /// Increment the reference count of a data blob shared between metadata files. virtual void incrementBlobRefCount(const std::string & /* blob */) { @@ -151,7 +150,7 @@ struct IDiskTransaction : private boost::noncopyable { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Blob reference counting is not implemented for this disk transaction"); } -======= + /// In-flight read-your-writes for a part being assembled by THIS transaction (B59). Forwarded to the /// metadata transaction by object-storage disk transactions; default (e.g. local disk) is no in-flight /// visibility, so a reader falls through to the committed path. @@ -168,7 +167,6 @@ struct IDiskTransaction : private boost::noncopyable /// STAGED directly under `path` (one level, the directory prefix stripped). Forwarded to the metadata /// transaction; default (e.g. local disk) is empty. virtual std::vector listInFlightDirectory(const std::string & /*path*/) const { return {}; } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) }; using DiskTransactionPtr = std::shared_ptr; diff --git a/src/IO/ReadPipeline.cpp b/src/IO/ReadPipeline.cpp index 0ffe16612b18..9687bc0e31ec 100644 --- a/src/IO/ReadPipeline.cpp +++ b/src/IO/ReadPipeline.cpp @@ -165,11 +165,11 @@ void ReadPipeline::needDecryption(String path, size_t buffer_size, KeyFinderFunc .key_finder = std::move(key_finder)}); } -<<<<<<< HEAD void ReadPipeline::needLongConnectionLimit(std::shared_ptr limit) { long_connection_limit = std::move(limit); -======= +} + void ReadPipeline::needFileView(String file_name, size_t left_bound, size_t right_bound) { if (right_bound < left_bound) @@ -179,7 +179,6 @@ void ReadPipeline::needFileView(String file_name, size_t left_bound, size_t righ .file_name = std::move(file_name), .left_bound = left_bound, .right_bound = right_bound}; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr ReadPipeline::build() const @@ -219,26 +218,14 @@ std::unique_ptr ReadPipeline::tryBuildReaderExecutor() c if (!settings.reader_executor.enabled) return nullptr; -<<<<<<< HEAD - /// The executor implements neither async prefetch nor the distributed cache, so fall back rather - /// than silently drop those stages. Decryption, the filesystem cache, and the page (memory) cache - /// ARE supported (fed below). - if (distributed_cache || async_prefetch) - { - LOG_DEBUG(log, - "use_reader_executor: falling back to the legacy read path " - "(distributed cache or async prefetch not supported by the executor)"); -======= - /// The executor does not implement caches, decryption, async prefetch, the - /// distributed cache, or a file_view byte window, so fall back rather than - /// silently drop a configured stage. - if (distributed_cache || memory_cache || !filesystem_caches.empty() - || !decryption_stages.empty() || async_prefetch || file_view) + /// The executor implements neither async prefetch, the distributed cache, nor a file_view byte + /// window, so fall back rather than silently drop those stages. Decryption, the filesystem cache, + /// and the page (memory) cache ARE supported (fed below). + if (distributed_cache || async_prefetch || file_view) { LOG_DEBUG(log, "use_reader_executor: falling back to the legacy read path " - "(caches/decryption/file_view not yet supported by the executor)"); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + "(distributed cache, async prefetch or file_view not supported by the executor)"); return nullptr; } diff --git a/src/IO/ReadPipeline.h b/src/IO/ReadPipeline.h index 0526237d0bc3..11ebe2ffa865 100644 --- a/src/IO/ReadPipeline.h +++ b/src/IO/ReadPipeline.h @@ -161,7 +161,6 @@ class ReadPipeline /// read from the encryption header. It must return the decryption key. void needDecryption(String path, size_t buffer_size, KeyFinderFunc key_finder); -<<<<<<< HEAD /// Let the `ReaderExecutor` path reuse held source connections, bounded by this limit. When it is /// not set, the executor uses the stateless one-shot path. void needLongConnectionLimit(std::shared_ptr limit); @@ -170,7 +169,7 @@ class ReadPipeline /// disks on random-object-key backends (see `DiskEncrypted::prepareRead`). Deterministic-path /// backends and url or external reads leave it null, so a reused key cannot serve a stale header. void needEncryptionHeaderCache(std::shared_ptr cache) { encryption_header_cache = std::move(cache); } -======= + /// -- File view stage -- /// Exposes ONLY the byte window [left_bound, right_bound) of the underlying chain as a /// standalone file named `file_name` (ReadBufferFromFileView). Used by content-addressed @@ -179,7 +178,6 @@ class ReadPipeline /// seeks and right bounds are translated and forwarded down the standard chain — but /// inside decryption, which operates on logical-file bytes. void needFileView(String file_name, size_t left_bound, size_t right_bound); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// -- Build the final ReadBuffer chain -- /// Uses the ReadSettings stored in the source stage. @@ -254,12 +252,9 @@ class ReadPipeline std::optional distributed_cache; std::optional async_prefetch; VectorWithMemoryTracking decryption_stages; -<<<<<<< HEAD /// Global encryption-header cache for the executor; null unless a random-object-key disk set it. std::shared_ptr encryption_header_cache; -======= std::optional file_view; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) LoggerPtr log = getLogger("ReadPipeline"); diff --git a/src/IO/S3/copyS3File.cpp b/src/IO/S3/copyS3File.cpp index 794be89d6c42..812c5c78e97f 100644 --- a/src/IO/S3/copyS3File.cpp +++ b/src/IO/S3/copyS3File.cpp @@ -628,11 +628,8 @@ namespace ThreadPoolCallbackRunnerUnsafe schedule_, BlobStorageLogWriterPtr blob_storage_log_, std::function fallback_method_, -<<<<<<< HEAD - bool is_ranged_copy_) -======= + bool is_ranged_copy_, bool allow_fallback_ = true) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) : UploadHelper( client_ptr_, dest_bucket_, @@ -932,7 +929,8 @@ namespace ThreadPoolCallbackRunnerUnsafe schedule, const CreateReadBuffer & fallback_file_reader, const std::optional & object_metadata, - bool is_ranged_copy) + bool is_ranged_copy, + ObjectStorageCopyMode copy_mode) { if (!dest_s3_client) dest_s3_client = src_s3_client; @@ -954,6 +952,11 @@ namespace if (!settings[S3RequestSetting::allow_native_copy]) { + if (copy_mode == ObjectStorageCopyMode::NativeOnly) + throw Exception( + ErrorCodes::NOT_IMPLEMENTED, + "Native-only S3 object copy is unavailable because allow_native_copy is disabled"); + LOG_TRACE(getLogger("copyS3File"), "Native copy is disable for {}", src_key); fallback_method(); return; @@ -974,7 +977,8 @@ namespace schedule, blob_storage_log, std::move(fallback_method), - is_ranged_copy}; + is_ranged_copy, + /*allow_fallback=*/copy_mode == ObjectStorageCopyMode::Default}; helper.performCopy(); } } @@ -991,50 +995,12 @@ void copyS3File( const ReadSettings & read_settings, BlobStorageLogWriterPtr blob_storage_log, ThreadPoolCallbackRunnerUnsafe schedule, -<<<<<<< HEAD const CreateReadBuffer & fallback_file_reader, - const std::optional & object_metadata) -{ - copyS3FileImpl( - std::move(src_s3_client), -======= - const CreateReadBuffer& fallback_file_reader, const std::optional & object_metadata, ObjectStorageCopyMode copy_mode) { - if (!dest_s3_client) - dest_s3_client = src_s3_client; - - std::function fallback_method = [&] mutable - { - copyDataToS3File( - fallback_file_reader, - src_offset, - src_size, - dest_s3_client, - dest_bucket, - dest_key, - settings, - blob_storage_log, - schedule, - object_metadata); - }; - - if (!settings[S3RequestSetting::allow_native_copy]) - { - if (copy_mode == ObjectStorageCopyMode::NativeOnly) - throw Exception( - ErrorCodes::NOT_IMPLEMENTED, - "Native-only S3 object copy is unavailable because allow_native_copy is disabled"); - - LOG_TRACE(getLogger("copyS3File"), "Native copy is disable for {}", src_key); - fallback_method(); - return; - } - - CopyFileHelper helper{ - src_s3_client, ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + copyS3FileImpl( + std::move(src_s3_client), src_bucket, src_key, /* src_offset= */ 0, @@ -1049,8 +1015,8 @@ void copyS3File( std::move(schedule), fallback_file_reader, object_metadata, -<<<<<<< HEAD - /* is_ranged_copy= */ false); + /* is_ranged_copy= */ false, + copy_mode); } void copyS3FileRange( @@ -1086,14 +1052,8 @@ void copyS3FileRange( std::move(schedule), fallback_file_reader, object_metadata, - /* is_ranged_copy= */ true); -======= - schedule, - blob_storage_log, - std::move(fallback_method), - /*allow_fallback=*/copy_mode == ObjectStorageCopyMode::Default}; - helper.performCopy(); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + /* is_ranged_copy= */ true, + ObjectStorageCopyMode::Default); } } diff --git a/src/IO/S3/copyS3File.h b/src/IO/S3/copyS3File.h index 3854fe508e73..5847889e933b 100644 --- a/src/IO/S3/copyS3File.h +++ b/src/IO/S3/copyS3File.h @@ -54,9 +54,9 @@ void copyS3File( const ReadSettings & read_settings, BlobStorageLogWriterPtr blob_storage_log, ThreadPoolCallbackRunnerUnsafe schedule, -<<<<<<< HEAD const CreateReadBuffer & fallback_file_reader, - const std::optional & object_metadata = std::nullopt); + const std::optional & object_metadata = std::nullopt, + ObjectStorageCopyMode copy_mode = ObjectStorageCopyMode::Default); /// Copies exactly `[src_offset, src_offset + src_size)` of a LARGER source object of size `src_object_size`. /// @@ -80,11 +80,6 @@ void copyS3FileRange( ThreadPoolCallbackRunnerUnsafe schedule, const CreateReadBuffer & fallback_file_reader, const std::optional & object_metadata = std::nullopt); -======= - const CreateReadBuffer& fallback_file_reader, - const std::optional & object_metadata = std::nullopt, - ObjectStorageCopyMode copy_mode = ObjectStorageCopyMode::Default); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// Copies data from any seekable source to S3. /// The same functionality can be done by using the function copyData() and the class WriteBufferFromS3 diff --git a/src/IO/S3Common.cpp b/src/IO/S3Common.cpp index 9f986d89d126..7e37d55e6724 100644 --- a/src/IO/S3Common.cpp +++ b/src/IO/S3Common.cpp @@ -43,13 +43,13 @@ bool S3Exception::isAccessTokenExpiredError() const return code == Aws::S3::S3Errors::INVALID_ACCESS_KEY_ID || code == Aws::S3::S3Errors::ACCESS_DENIED || code == Aws::S3::S3Errors::INVALID_SIGNATURE || code == Aws::S3::S3Errors::UNKNOWN; } -<<<<<<< HEAD bool isTransientCompleteMultipartUploadError(const Aws::S3::S3Error & error) { return error.GetErrorType() == Aws::S3::S3Errors::NO_SUCH_KEY || error.GetExceptionName() == "InvalidPart" || error.GetExceptionName() == "InvalidPartOrder"; -======= +} + bool S3Exception::isPreconditionFailed() const { /// See `S3::isPreconditionFailedError`. The thrown exception no longer carries the HTTP status, so @@ -93,7 +93,6 @@ bool isAccessDeniedError(const S3Exception & e) || e.getS3ErrorCode() == Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID; } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/IO/WriteBufferFromS3.cpp b/src/IO/WriteBufferFromS3.cpp index 69fea4b686e5..31fbe6172098 100644 --- a/src/IO/WriteBufferFromS3.cpp +++ b/src/IO/WriteBufferFromS3.cpp @@ -716,16 +716,10 @@ bool WriteBufferFromS3::completeMultipartUpload() /// Pass the canonical S3 error name: a conditional-write 412 is UNMODELED for the SDK /// (the error type is UNKNOWN), so the name is the caller's only typed signal. throw S3Exception( -<<<<<<< HEAD - error.GetErrorType(), - "Message: {}, Key: {}, Bucket: {}, Tags: {}", - error.GetMessage(), key, bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")); -======= PreformattedMessage::create("Message: {}, Key: {}, Bucket: {}, Tags: {}", outcome.GetError().GetMessage(), key, bucket, fmt::join(multipart_tags.begin(), multipart_tags.end(), " ")), outcome.GetError().GetErrorType(), outcome.GetError().GetExceptionName()); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/IO/tests/gtest_writebuffer_s3.cpp b/src/IO/tests/gtest_writebuffer_s3.cpp index 93ac4a74acff..443144189548 100644 --- a/src/IO/tests/gtest_writebuffer_s3.cpp +++ b/src/IO/tests/gtest_writebuffer_s3.cpp @@ -19,12 +19,9 @@ #include #include #include -<<<<<<< HEAD #include -======= #include #include ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include @@ -38,10 +35,7 @@ #include #include #include -<<<<<<< HEAD -======= #include ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include @@ -211,7 +205,6 @@ struct EventCounts size_t copyObject = 0; size_t uploadPartCopy = 0; size_t writtenSize = 0; - size_t copyObject = 0; size_t deleteObject = 0; size_t getBucketVersioning = 0; @@ -494,51 +487,12 @@ struct Client : DB::S3::Client return Aws::S3::Model::AbortMultipartUploadOutcome(result); } -<<<<<<< HEAD /// Whole-object server-side copy. A CopyObject request carries no byte range, so it always copies the /// entire source object -- modelling the real S3 behaviour that makes it unsafe for a partial range. -======= ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) Aws::S3::Model::CopyObjectOutcome CopyObject(const Aws::S3::Model::CopyObjectRequest & request) const override { ++counters.copyObject; -<<<<<<< HEAD - const auto [src_bucket, src_key] = splitCopySource(request.GetCopySource()); - const String & src_data = store->GetBucketStore(src_bucket).objects[src_key]; - store->GetBucketStore(request.GetBucket()).PutObject(request.GetKey(), src_data); - - Aws::S3::Model::CopyObjectResult result; - return Aws::S3::Model::CopyObjectOutcome(result); - } - - /// Ranged server-side copy of one multipart part. Honours the `CopySourceRange` so only the requested - /// bytes are copied -- this is the path a partial-range copy must take. - Aws::S3::Model::UploadPartCopyOutcome UploadPartCopy(const Aws::S3::Model::UploadPartCopyRequest & request) const override - { - ++counters.uploadPartCopy; - - const auto [src_bucket, src_key] = splitCopySource(request.GetCopySource()); - const String & src_data = store->GetBucketStore(src_bucket).objects[src_key]; - - size_t begin = 0; - size_t end = src_data.size() - 1; - const String & range = request.GetCopySourceRange(); - if (const String prefix = "bytes="; range.starts_with(prefix)) - { - int ret = sscanf(range.c_str(), "bytes=%zu-%zu", &begin, &end); /// NOLINT - chassert(ret == 2); - } - - auto & dstStore = store->GetBucketStore(request.GetBucket()); - auto etag = dstStore.UploadPart(request.GetUploadId(), src_data.substr(begin, end - begin + 1)); - - Aws::S3::Model::CopyPartResult copy_part_result; - copy_part_result.SetETag(etag); - Aws::S3::Model::UploadPartCopyResult result; - result.SetCopyPartResult(copy_part_result); - return Aws::S3::Model::UploadPartCopyOutcome(result); -======= if (const auto * wrapper = dynamic_cast(&request)) last_copy_object_native_conditional = wrapper->isNativeConditional(); @@ -572,6 +526,34 @@ struct Client : DB::S3::Client return Aws::S3::Model::CopyObjectOutcome(result); } + /// Ranged server-side copy of one multipart part. Honours the `CopySourceRange` so only the requested + /// bytes are copied -- this is the path a partial-range copy must take. + Aws::S3::Model::UploadPartCopyOutcome UploadPartCopy(const Aws::S3::Model::UploadPartCopyRequest & request) const override + { + ++counters.uploadPartCopy; + + const auto [src_bucket, src_key] = splitCopySource(request.GetCopySource()); + const String & src_data = store->GetBucketStore(src_bucket).objects[src_key]; + + size_t begin = 0; + size_t end = src_data.size() - 1; + const String & range = request.GetCopySourceRange(); + if (const String prefix = "bytes="; range.starts_with(prefix)) + { + int ret = sscanf(range.c_str(), "bytes=%zu-%zu", &begin, &end); /// NOLINT + chassert(ret == 2); + } + + auto & dstStore = store->GetBucketStore(request.GetBucket()); + auto etag = dstStore.UploadPart(request.GetUploadId(), src_data.substr(begin, end - begin + 1)); + + Aws::S3::Model::CopyPartResult copy_part_result; + copy_part_result.SetETag(etag); + Aws::S3::Model::UploadPartCopyResult result; + result.SetCopyPartResult(copy_part_result); + return Aws::S3::Model::UploadPartCopyOutcome(result); + } + Aws::S3::Model::DeleteObjectOutcome DeleteObject(const Aws::S3::Model::DeleteObjectRequest & request) const override { ++counters.deleteObject; @@ -605,7 +587,6 @@ struct Client : DB::S3::Client Aws::S3::Model::GetBucketVersioningResult result; result.SetStatus(Aws::S3::Model::BucketVersioningStatus::Enabled); return Aws::S3::Model::GetBucketVersioningOutcome(result); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::shared_ptr store; @@ -663,7 +644,6 @@ struct UploadPartFailIngection: InjectionModel } }; -<<<<<<< HEAD /// Fails the first `fail_times` CompleteMultipartUpload calls with the un-typed MinIO `InvalidPart` /// eventual-consistency error, then lets the real mock store handle the rest. The AWS SDK cannot map /// InvalidPart to a typed model error, so it produces UNKNOWN as the error type and keeps @@ -687,7 +667,8 @@ struct CompleteMPUInvalidPartOnceIngection : InjectionModel size_t fail_times; size_t calls = 0; -======= +}; + /// Injects an arbitrary AWSError on DeleteObject -- used to drive the conditional-remove /// (`removeObjectIfTokenMatches`) outcome mapping: a 412-shaped error (exception name "PreconditionFailed", /// matched by `S3::isPreconditionFailedError`) must map to `ConditionalRemoveOutcome::TokenMismatch`, and a @@ -717,7 +698,6 @@ struct CopyObjectErrorInjection: InjectionModel } Aws::Client::AWSError error; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) }; struct BaseSyncPolicy diff --git a/src/Interpreters/InterpreterSystemQuery.cpp b/src/Interpreters/InterpreterSystemQuery.cpp index 7a5e22180ae0..096035d391c8 100644 --- a/src/Interpreters/InterpreterSystemQuery.cpp +++ b/src/Interpreters/InterpreterSystemQuery.cpp @@ -3057,16 +3057,12 @@ void InterpreterSystemQuery::flushDistributed(ASTSystemQuery & query) if (query.query_settings) settings_changes = query.query_settings->as()->changes; -<<<<<<< HEAD /// Keep the StoragePtr alive for the whole flush: the table holds no other owning /// reference here (DROP on an Atomic database does not take the exclusive drop_lock, /// and the flush does not hold an async-insert lock), so a concurrent DROP could /// otherwise destroy the table while flushClusterNodesAllData is still running. - auto table = DatabaseCatalog::instance().getTable(table_id, getContext()); + auto table = unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())); if (auto * storage_distributed = dynamic_cast(table.get())) -======= - if (auto * storage_distributed = dynamic_cast(unwrapTableProxy(DatabaseCatalog::instance().getTable(table_id, getContext())).get())) ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) storage_distributed->flushClusterNodesAllData(getContext(), settings_changes); else throw Exception(ErrorCodes::BAD_ARGUMENTS, "Table {} is not distributed", table_id.getNameForLogs()); diff --git a/src/Interpreters/ServerAsynchronousMetrics.cpp b/src/Interpreters/ServerAsynchronousMetrics.cpp index aa0896a3c0e7..60ba83d68020 100644 --- a/src/Interpreters/ServerAsynchronousMetrics.cpp +++ b/src/Interpreters/ServerAsynchronousMetrics.cpp @@ -15,11 +15,8 @@ #include -<<<<<<< HEAD #include -======= #include ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) #include #include @@ -389,12 +386,42 @@ void ServerAsynchronousMetrics::updateImpl(TimePoint update_time, TimePoint curr } #endif -<<<<<<< HEAD if (auto object_storage_disk = std::dynamic_pointer_cast(disk)) { dead_blobs_queue_estimate[name] = static_cast(object_storage_disk->getDeadBlobsQueueEstimate()); missing_blobs_queue_estimate[name] = static_cast(object_storage_disk->getMissingBlobsQueueEstimate()); } + + /// Per-disk CAS GC health, for Prometheus scraping. `tryFromDisk` returns nullptr for a + /// disk whose metadata storage is not content-addressed (the common case); `gcHealth()` + /// returns nullopt for a content-addressed disk whose GC scheduler has not started yet + /// (still opening, read-only, or GC disabled by configuration) -- both are skipped + /// silently, same as the DiskUsed_/DiskTotal_ metrics above skip disks that don't report + /// space. This runs on every asynchronous-metrics tick for every configured disk and must + /// never throw. + try + { + if (auto * ca_storage = ContentAddressedMetadataStorage::tryFromDisk(disk)) + { + if (auto health = ca_storage->gcHealth()) + { + new_values[fmt::format("CASGCIsLeader_{}", name)] = { health->is_leader ? 1 : 0, + "Whether this server currently holds the content-addressed garbage-collection lease for the disk (1) or not (0, e.g. another replica is leading)." }; + new_values[fmt::format("CASGCPendingReclaim_{}", name)] = { health->pending_reclaim, + "Cumulative content-addressed objects condemned minus objects physically deleted by this process while it has held the GC lease on the disk. A persistently growing value indicates GC is not keeping up with reclaim." }; + new_values[fmt::format("CASGCLastSuccessAgeSeconds_{}", name)] = { health->last_success_age_seconds, + "Seconds since this process last completed a successful content-addressed GC round as leader on the disk (0 if it has never led one)." }; + new_values[fmt::format("CASGCWedgedNamespaces_{}", name)] = { health->wedged_namespace_count, + "Number of content-addressed namespaces on the disk currently stuck behind a wedged reference lane, unable to make GC progress." }; + } + } + } + catch (...) // NOLINT(bugprone-empty-catch) + { + /// Sampled on every server tick for every disk; a transient failure here (e.g. a + /// store health query hiccup) must never break the rest of asynchronous-metrics + /// collection. + } } if (!disk_total.empty()) @@ -437,38 +464,6 @@ void ServerAsynchronousMetrics::updateImpl(TimePoint update_time, TimePoint curr "Estimated number of blobs enqueued for removal from the disk object storage (the blob manager dead queue), keyed by the disk name. Disks without blob replication report 0." }; new_values["MissingBlobsQueueEstimate"] = { "disk", std::move(missing_blobs_queue_estimate), "Estimated number of blobs awaiting replication to other locations of the disk (the blob manager missing queue), keyed by the disk name. Disks without blob replication report 0." }; -======= - /// Per-disk CAS GC health, for Prometheus scraping. `tryFromDisk` returns nullptr for a - /// disk whose metadata storage is not content-addressed (the common case); `gcHealth()` - /// returns nullopt for a content-addressed disk whose GC scheduler has not started yet - /// (still opening, read-only, or GC disabled by configuration) -- both are skipped - /// silently, same as the DiskUsed_/DiskTotal_ metrics above skip disks that don't report - /// space. This runs on every asynchronous-metrics tick for every configured disk and must - /// never throw. - try - { - if (auto * ca_storage = ContentAddressedMetadataStorage::tryFromDisk(disk)) - { - if (auto health = ca_storage->gcHealth()) - { - new_values[fmt::format("CASGCIsLeader_{}", name)] = { health->is_leader ? 1 : 0, - "Whether this server currently holds the content-addressed garbage-collection lease for the disk (1) or not (0, e.g. another replica is leading)." }; - new_values[fmt::format("CASGCPendingReclaim_{}", name)] = { health->pending_reclaim, - "Cumulative content-addressed objects condemned minus objects physically deleted by this process while it has held the GC lease on the disk. A persistently growing value indicates GC is not keeping up with reclaim." }; - new_values[fmt::format("CASGCLastSuccessAgeSeconds_{}", name)] = { health->last_success_age_seconds, - "Seconds since this process last completed a successful content-addressed GC round as leader on the disk (0 if it has never led one)." }; - new_values[fmt::format("CASGCWedgedNamespaces_{}", name)] = { health->wedged_namespace_count, - "Number of content-addressed namespaces on the disk currently stuck behind a wedged reference lane, unable to make GC progress." }; - } - } - } - catch (...) // NOLINT(bugprone-empty-catch) - { - /// Sampled on every server tick for every disk; a transient failure here (e.g. a - /// store health query hiccup) must never break the rest of asynchronous-metrics - /// collection. - } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } } diff --git a/src/Interpreters/ThreadStatusExt.cpp b/src/Interpreters/ThreadStatusExt.cpp index c5efed73daf9..36b18da61316 100644 --- a/src/Interpreters/ThreadStatusExt.cpp +++ b/src/Interpreters/ThreadStatusExt.cpp @@ -138,8 +138,6 @@ ThreadGroup::ThreadGroup(ThreadGroupPtr parent_thread_group) , global_context(parent->global_context) , fatal_error_callback(parent->fatal_error_callback) , os_threads_nice_value(parent->os_threads_nice_value) - /// Keep the parent group alive: this child parents its trackers at the parent's via raw pointers (B90). - , parent_thread_group(parent) , memory_spill_scheduler(parent->memory_spill_scheduler) , performance_counters(VariableContext::Process, &parent->performance_counters) , memory_tracker(&parent->memory_tracker, VariableContext::Process, /*log_peak_memory_usage_in_destructor*/ false) @@ -154,8 +152,6 @@ ThreadGroup::ThreadGroup(ContextPtr query_context_, ThreadGroupPtr parent_thread , global_context(query_context_->getGlobalContext()) , fatal_error_callback(parent->fatal_error_callback) , os_threads_nice_value(parent->os_threads_nice_value) - /// Keep the parent group alive: this child parents its trackers at the parent's via raw pointers (B90). - , parent_thread_group(parent) , memory_spill_scheduler(parent->memory_spill_scheduler) , performance_counters(VariableContext::Process, &parent->performance_counters) , memory_tracker(&parent->memory_tracker, VariableContext::Process, /*log_peak_memory_usage_in_destructor*/ false) diff --git a/src/Parsers/ASTSystemQuery.h b/src/Parsers/ASTSystemQuery.h index 2ee18aa8bb5e..d6cb01ba5f90 100644 --- a/src/Parsers/ASTSystemQuery.h +++ b/src/Parsers/ASTSystemQuery.h @@ -159,7 +159,6 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster INSTRUMENT_ADD, INSTRUMENT_REMOVE, RESET_DDL_WORKER, -<<<<<<< HEAD STOP_ALL_BACKGROUND, START_ALL_BACKGROUND, PAUSE_ALL_BACKGROUND, @@ -170,7 +169,6 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster PAUSE, CANCEL, REFRESH, -======= CAS_GC_RUN, CAS_GC_REBUILD, CAS_DROP_POOL_MEMBER, @@ -178,7 +176,6 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster CAS_FORGET, CAS_GC_STOP, CAS_GC_START, ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) END }; diff --git a/src/Parsers/tests/gtest_Parser.cpp b/src/Parsers/tests/gtest_Parser.cpp index 144ad7d08ade..2898d2122ec8 100644 --- a/src/Parsers/tests/gtest_Parser.cpp +++ b/src/Parsers/tests/gtest_Parser.cpp @@ -845,7 +845,6 @@ INSTANTIATE_TEST_SUITE_P(ParserRenameQuery, ParserTest, } }))); -<<<<<<< HEAD #ifdef DEBUG_OR_SANITIZER_BUILD /// Regression test for the UBSan "member call on null pointer of type DB::IAST" at /// ASTRenameQuery::formatQueryImpl (RENAME DATABASE branch). A RENAME DATABASE node always @@ -881,7 +880,7 @@ TEST(ParserRenameQueryDeathTest, FormatNullToDatabaseAborts) EXPECT_DEATH(ast->formatWithSecretsOneLine(), "elements.at\\(0\\).to.database"); } #endif -======= + // SYSTEM CAS DROP POOL MEMBER: srid and disk are both required quoted string literals // (an srid is an opaque server-root path, not identifier-shaped); ON CLUSTER round-trips as a bare // identifier (ASTQueryWithOnCluster::formatOnCluster uses backQuoteIfNeed, no quoting needed for a @@ -924,7 +923,6 @@ INSTANTIATE_TEST_SUITE_P(ParserSystemQuery, ParserTest, "SYSTEM CAS GC RUN ON CLUSTER my_cluster disk1" }, }))); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) static constexpr size_t kDummyMaxQuerySize = 256 * 1024; static constexpr size_t kDummyMaxParserDepth = 256; diff --git a/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp b/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp index a6bcc7bf66be..3ff5c51b25ee 100644 --- a/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp +++ b/src/Storages/MergeTree/DataPartStorageOnDiskBase.cpp @@ -599,27 +599,19 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( if (save_metadata_callback) save_metadata_callback(disk); -<<<<<<< HEAD /// Also remove any leftover `txn_version.txt.tmp`: leaving it without the main file makes the /// cloned/frozen part load as a rolled-back transaction (see `VersionMetadataOnDisk::loadMetadata`) /// and get discarded as `Outdated`. Remove the temporary file before the main file so the cleanup /// is fail-closed: a failure between the two removals leaves a valid `txn_version.txt` rather than /// the dangerous tmp-only state. - if (params.external_transaction) - { - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TMP_TXN_VERSION_METADATA_FILE_NAME); - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); - if (!params.keep_metadata_version) - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); - IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*params.external_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); -======= if (clone_transaction) { clone_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TMP_TXN_VERSION_METADATA_FILE_NAME); clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); if (!params.keep_metadata_version) clone_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); + IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*clone_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); /// When the caller wants a fresh metadata version written into the clone (the Replicated queue /// clone path — `executeReplaceRange`/`replacePartitionFrom`/`movePartitionToTable` set @@ -641,7 +633,6 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( writeText(*params.metadata_version_to_write, *out_metadata); out_metadata->finalize(); } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } else { @@ -653,20 +644,18 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freeze( IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*disk, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); } -<<<<<<< HEAD /// Make the hardlink clone durable (the Backup loop above fsyncs nothing). This runs /// synchronously before freeze returns, so a caller that afterwards makes a destructive change /// (e.g. DETACH commits a covering empty part and drops the source) sees the clone already on /// disk. See the commit message / #111382 for the full rationale. if (params.fsync_part_directory && !params.external_transaction && !disk->isRemote()) fsyncFrozenCloneTree(*disk, fs::path(to) / dir_path); -======= + /// Commit the self-created transaction (the whole-part clone commit point for CA). An external /// transaction is committed by its owner, as before. Before the arena scope below, so the commit's /// own allocations are not attributed to the MergeTree arena. if (owned_transaction) owned_transaction->commit(); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) /// The SingleDiskVolume and the DataPartStorageOnDiskFull built by `create` are stored on the /// frozen part for its whole lifetime; route them into the dedicated MergeTree arena, like the @@ -791,21 +780,11 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( if (save_metadata_callback) save_metadata_callback(dst_disk); -<<<<<<< HEAD /// Also remove any leftover `txn_version.txt.tmp`: leaving it without the main file makes the /// cloned/frozen part load as a rolled-back transaction (see `VersionMetadataOnDisk::loadMetadata`) /// and get discarded as `Outdated`. Remove the temporary file before the main file so the cleanup /// is fail-closed: a failure between the two removals leaves a valid `txn_version.txt` rather than /// the dangerous tmp-only state. - if (params.external_transaction) - { - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TMP_TXN_VERSION_METADATA_FILE_NAME); - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); - if (!params.keep_metadata_version) - params.external_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); - IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*params.external_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); -======= /// These removals belong to the clone. On the content-addressed arm they MUST go through the same /// transaction: sent straight to the disk they would autocommit, which is exactly the /// one-publish-per-file behaviour the single transaction above exists to prevent. @@ -814,9 +793,11 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( try { clone_transaction->removeFileIfExists(fs::path(to) / dir_path / "delete-on-destroy.txt"); + clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TMP_TXN_VERSION_METADATA_FILE_NAME); clone_transaction->removeFileIfExists(fs::path(to) / dir_path / VersionMetadata::TXN_VERSION_METADATA_FILE_NAME); if (!params.keep_metadata_version) clone_transaction->removeFileIfExists(fs::path(to) / dir_path / IMergeTreeDataPart::METADATA_VERSION_FILE_NAME); + IMergeTreeDataPart::writeInvalidatedSystemColumnsFile(*clone_transaction, fs::path(to) / dir_path, params.invalidated_columns_to_write, write_settings); if (owned_transaction) owned_transaction->commit(); } @@ -826,7 +807,6 @@ MutableDataPartStoragePtr DataPartStorageOnDiskBase::freezeRemote( owned_transaction->undo(); throw; } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } else { diff --git a/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp b/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp index 74ddaa16b49e..a32ef79cd324 100644 --- a/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp +++ b/src/Storages/MergeTree/DataPartStorageOnDiskFull.cpp @@ -73,21 +73,12 @@ bool DataPartStorageOnDiskFull::exists() const bool DataPartStorageOnDiskFull::existsFileImpl(const std::string & name) const { -<<<<<<< HEAD - return volume->getDisk()->existsFile(fs::path(root_path) / part_dir / name); -======= auto path = fs::path(root_path) / part_dir / name; /// B59: a part still being assembled by this transaction can have staged-but-uncommitted files /// (e.g. projection temp blocks on a content-addressed disk). Consult the held transaction first. if (transaction && transaction->tryGetInFlightFileSize(path).has_value()) return true; - if (looksLikePackedSkipIndexFile(name)) - { - if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) - return true; - } return volume->getDisk()->existsFile(path); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } bool DataPartStorageOnDiskFull::existsDirectory(const std::string & name) const @@ -174,27 +165,21 @@ Poco::Timestamp DataPartStorageOnDiskFull::getFileLastModified(const String & fi } size_t DataPartStorageOnDiskFull::getFileSizeImpl(const String & file_name) const -{ - return volume->getDisk()->getFileSize(fs::path(root_path) / part_dir / file_name); -} - -std::optional DataPartStorageOnDiskFull::getPackedFileUncompressedSize(const std::string & file_name) const { auto path = fs::path(root_path) / part_dir / file_name; /// B59: see existsFile — the merge stats the staged temp files before reading them back. if (transaction) if (auto size = transaction->tryGetInFlightFileSize(path)) return *size; + return volume->getDisk()->getFileSize(path); +} + +std::optional DataPartStorageOnDiskFull::getPackedFileUncompressedSize(const std::string & file_name) const +{ if (looksLikePackedSkipIndexFile(file_name)) if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(file_name)) -<<<<<<< HEAD return reader->getFileUncompressedSize(file_name); return {}; -======= - return reader->getFileSize(file_name); - } - return volume->getDisk()->getFileSize(path); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } UInt32 DataPartStorageOnDiskFull::getRefCount(const String & file_name) const @@ -241,9 +226,6 @@ void DataPartStorageOnDiskFull::prepareReadImpl( std::optional read_hint, ReadPipeline & pipeline) const { -<<<<<<< HEAD - volume->getDisk()->prepareRead(fs::path(root_path) / part_dir / name, settings, read_hint, pipeline); -======= auto path = fs::path(root_path) / part_dir / name; /// B59: read-your-writes for a part still being assembled by this transaction. A projection @@ -280,29 +262,7 @@ void DataPartStorageOnDiskFull::prepareReadImpl( } } - if (looksLikePackedSkipIndexFile(name)) - { - if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) - { - /// Packed substreams skip the disk's normal pipeline (filesystem cache, - /// async prefetch, etc.) and read through PackedFilesReader::readFile, which - /// opens the archive via the underlying disk and wraps the result with - /// ReadBufferFromFileView at the right offset. The archive's current location is - /// captured here and passed in, so the reader holds no path of its own. - auto disk = volume->getDisk(); - String archive_path = fs::path(root_path) / part_dir / String(SKIP_INDICES_PACKED_FILENAME); - ReadPipeline::BufferCreator creator = - [reader, disk, archive_path, name, read_hint](const StoredObject &, const ReadSettings & s, bool, bool) - { - return reader->readFile(disk, archive_path, name, s, read_hint); - }; - pipeline.setSource(std::move(creator), StoredObjects{StoredObject{}}, settings); - return; - } - } - volume->getDisk()->prepareRead(path, settings, read_hint, pipeline); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr DataPartStorageOnDiskFull::readFileIfExistsImpl( @@ -310,9 +270,6 @@ std::unique_ptr DataPartStorageOnDiskFull::readFileIfExi const ReadSettings & settings, std::optional read_hint) const { -<<<<<<< HEAD - return volume->getDisk()->readFileIfExists(fs::path(root_path) / part_dir / name, settings, read_hint); -======= auto path = fs::path(root_path) / part_dir / name; /// B59: serve a file staged by this transaction (uploaded blob or inline mutable bytes) before commit. /// This direct delegate bypasses prepareRead, so the in-flight guard must be repeated here; it is the @@ -320,16 +277,7 @@ std::unique_ptr DataPartStorageOnDiskFull::readFileIfExi if (transaction) if (auto rb = transaction->tryReadFileInFlight(path, settings, read_hint)) return rb; - if (looksLikePackedSkipIndexFile(name)) - { - if (auto reader = getSkipIndicesPackedReader(); reader && reader->exists(name)) - return reader->readFile( - volume->getDisk(), - fs::path(root_path) / part_dir / String(SKIP_INDICES_PACKED_FILENAME), - name, settings, read_hint); - } return volume->getDisk()->readFileIfExists(path, settings, read_hint); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } std::unique_ptr DataPartStorageOnDiskFull::writeFile( diff --git a/src/Storages/MergeTree/DataPartsExchange.cpp b/src/Storages/MergeTree/DataPartsExchange.cpp index 7f832485abf0..a3155848eaf1 100644 --- a/src/Storages/MergeTree/DataPartsExchange.cpp +++ b/src/Storages/MergeTree/DataPartsExchange.cpp @@ -93,9 +93,7 @@ constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_PARTS_ZERO_COPY = 6; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_PARTS_PROJECTION = 7; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_METADATA_VERSION = 8; constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_COLUMNS_SUBSTREAMS = 9; -<<<<<<< HEAD constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS = 10; -======= /// CAS replication 2b: fetch-by-relink. The receiver advertises its content-addressed pool identity /// (`cas_pool_uuid`) and, if it matches the sender's own pool, the sender sends only the /// part's content id (`part_id`) + the mutable header — no file bytes — and the receiver "fetches" by @@ -111,7 +109,6 @@ constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS = 10 /// still exactly what the sender's ref names. A server advertising this version serves the confirm /// action; a receiver advertising it must confirm before it promotes. constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM = 11; ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) std::string getEndpointId(const std::string & node_id) { @@ -330,11 +327,7 @@ void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuf MergeTreePartInfo::fromPartName(part_name, data.format_version); /// We pretend to work as older server version, to be sure that client will correctly process our version -<<<<<<< HEAD - response.addCookie({"server_protocol_version", toString(std::min(client_protocol_version, REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS))}); -======= response.addCookie({"server_protocol_version", toString(std::min(client_protocol_version, REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM))}); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) LOG_TRACE(log, "Sending part {}", part_name); @@ -693,15 +686,11 @@ std::pair Fetcher::fetchSelected { {"endpoint", endpoint_id}, {"part", part_name}, -<<<<<<< HEAD - {"client_protocol_version", toString(REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS)}, -======= /// Advertising `..._WITH_CA_CONFIRM` is a PROMISE, not a capability list: this receiver will /// confirm a relink offer against its source before it promotes (`relinkPartToDisk`). It is the /// pair of the sender-side offer gate on the same constant, and the two cannot be separated — /// see the comment there. {"client_protocol_version", toString(REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM)}, ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) {"compress", "false"} }); diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 6bbe96cb79a6..d5618943754e 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -1626,13 +1626,9 @@ MergeTreeDataPartBuilder IMergeTreeDataPart::getProjectionPartBuilder( MutableDataPartStoragePtr projection_storage; { ScopedJemallocThreadArena mergetree_arena_scope(JemallocMergeTreeArena::getArenaIndex()); -<<<<<<< HEAD projection_storage = intent == PartDirIntent::CreateFresh - ? getDataPartStorage().getProjectionNoInitialize(projection_name + projection_extension, !is_temp_projection) - : getDataPartStorage().getProjection(projection_name + projection_extension, !is_temp_projection); -======= - projection_storage = getDataPartStorage().getProjection(projection_name + projection_extension, use_parent_transaction); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + ? getDataPartStorage().getProjectionNoInitialize(projection_name + projection_extension, use_parent_transaction) + : getDataPartStorage().getProjection(projection_name + projection_extension, use_parent_transaction); } if (intent == PartDirIntent::CreateFresh && projection_storage->exists()) { diff --git a/src/Storages/MergeTree/MergeTask.cpp b/src/Storages/MergeTree/MergeTask.cpp index 90ed28f6e841..35f1c288c169 100644 --- a/src/Storages/MergeTree/MergeTask.cpp +++ b/src/Storages/MergeTree/MergeTask.cpp @@ -594,21 +594,16 @@ bool MergeTask::ExecuteAndFinalizeHorizontalPart::prepare() const std::optional builder; if (global_ctx->parent_part) { -<<<<<<< HEAD - /// Non-initializing, so nothing is seeded from an existing directory. - auto data_part_storage = global_ctx->parent_part->getDataPartStorage().getProjectionNoInitialize(local_tmp_part_basename, /* use parent transaction */ false); - builder.emplace(*global_ctx->data, global_ctx->future_part->name, data_part_storage, getReadSettings(), PartDirIntent::CreateFresh); -======= /// On a content-addressed disk a part is one atomic unit (one manifest + one ref), so the /// projection sub-part must be written through the PARENT part's whole-part transaction -- /// mirroring the INSERT path -- for its files to land in the parent manifest and survive a /// reload. On any other disk the projection keeps its own sub-transaction, as before. global_ctx->projection_uses_parent_transaction = global_ctx->parent_part->getDataPartStorage().isContentAddressed(); - auto data_part_storage = global_ctx->parent_part->getDataPartStorage().getProjection( + /// Non-initializing, so nothing is seeded from an existing directory. + auto data_part_storage = global_ctx->parent_part->getDataPartStorage().getProjectionNoInitialize( local_tmp_part_basename, /* use_parent_transaction */ global_ctx->projection_uses_parent_transaction); - builder.emplace(*global_ctx->data, global_ctx->future_part->name, data_part_storage, getReadSettings()); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + builder.emplace(*global_ctx->data, global_ctx->future_part->name, data_part_storage, getReadSettings(), PartDirIntent::CreateFresh); builder->withParentPart(global_ctx->parent_part); } else diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 27a68c831acf..fc3d728db6a8 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -7034,32 +7034,6 @@ MergeTreeData::PartsToRemoveFromZooKeeper MergeTreeData::removePartsInRangeFromW new_data_part->getDataPartStorage().commitTransaction(); rollback_tx_guard.reset(); -<<<<<<< HEAD -======= - /// On a content-addressed disk a part directory becomes durable only when its disk-storage - /// transaction is committed (the ref to its manifest is published at commit, not at rename). - /// The flow below rolls back the in-memory MergeTreeData transaction (to keep the empty part - /// Outdated, not Active), which never calls commitTransaction on the disk storage — so on a CA - /// disk the empty covering part would leave NO on-disk ref and vanish on restart/reattach, - /// defeating its sole purpose (it exists only to cover the dropped parts on disk so a restart - /// does not treat them as uncovered unexpected parts and trip TOO_MANY_UNEXPECTED_DATA_PARTS). - /// On a plain disk the rename in renameTempPartAndAdd is already durable, so this is a no-op - /// there. Commit the disk storage transaction here (CA only) so the ref is published before the - /// in-memory rollback; the part still ends up Outdated, exactly as on a plain disk. - /// - /// [TXN-ONE-PIPELINE] (`2026-07-16-cas-txn-one-pipeline-design.md`, Audit 7 / Tension 2): this - /// hand-placed `commitTransaction()` is NOT made redundant by moving publication into `commit` - /// — it is the direct consequence of that design. There is no `precommit` phase under the - /// one-pipeline model, and this rollback path (by construction, to keep the part Outdated) never - /// reaches `MergeTreeData::Transaction::commit`, the only other place a disk transaction is - /// committed. So this call remains the ONLY thing that publishes the empty cover's ref. Keep it. - if (new_data_part->getDataPartStorage().isContentAddressed() - && new_data_part->getDataPartStorage().hasActiveTransaction()) - new_data_part->getDataPartStorage().commitTransaction(); - - /// It will add the empty part to the set of Outdated parts without making it Active (exactly what we need) - transaction.rollback(&lock); ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) new_data_part->remove_time.store(0, std::memory_order_relaxed); /// Such parts are always local, they don't participate in replication, they don't have shared blobs. /// So we don't have locks for shared data in zk for them, and can just remove blobs (this avoids leaving garbage in S3) @@ -9264,10 +9238,6 @@ void MergeTreeData::restorePartFromBackup(std::shared_ptr r continue; } -<<<<<<< HEAD - size_t file_size = backup->copyFileToDisk(part_path_in_backup_fs / filename, disk, temp_part_dir / filename, WriteMode::Rewrite, fsync_files); - reservation->update(reservation->getSize() - file_size); -======= if (restore_tx) { auto in = backup->readFile(part_path_in_backup_fs / filename); @@ -9278,10 +9248,9 @@ void MergeTreeData::restorePartFromBackup(std::shared_ptr r } else { - size_t file_size = backup->copyFileToDisk(part_path_in_backup_fs / filename, disk, temp_part_dir / filename, WriteMode::Rewrite); + size_t file_size = backup->copyFileToDisk(part_path_in_backup_fs / filename, disk, temp_part_dir / filename, WriteMode::Rewrite, fsync_files); reservation->update(reservation->getSize() - file_size); } ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) } if (restore_tx) diff --git a/tests/integration/test_replicated_database/test.py b/tests/integration/test_replicated_database/test.py index ad37d3bf5378..4a55f4758145 100644 --- a/tests/integration/test_replicated_database/test.py +++ b/tests/integration/test_replicated_database/test.py @@ -1398,9 +1398,6 @@ def test_replicated_table_structure_alter(started_cluster): ) competing_node.query("SYSTEM SYNC DATABASE REPLICA table_structure") -<<<<<<< HEAD - competing_node.query("DETACH DATABASE table_structure SYNC") -======= # `system.tables` only lists an attached database, so the metadata path of `mem` must be read # before the DETACH below; afterwards the SELECT returns nothing. @@ -1409,8 +1406,7 @@ def test_replicated_table_structure_alter(started_cluster): ).strip() assert metadata_path, "metadata_path of table_structure.mem is empty" - competing_node.query("DETACH DATABASE table_structure") ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) + competing_node.query("DETACH DATABASE table_structure SYNC") main_node.query( "ALTER TABLE table_structure.rmt ADD COLUMN m int", settings=settings diff --git a/tests/queries/0_stateless/02253_empty_part_checksums.sh b/tests/queries/0_stateless/02253_empty_part_checksums.sh deleted file mode 100755 index af4e1d896658..000000000000 --- a/tests/queries/0_stateless/02253_empty_part_checksums.sh +++ /dev/null @@ -1,40 +0,0 @@ -#!/usr/bin/env bash -# Tags: zookeeper, no-replicated-database, no-shared-merge-tree, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) -# no-replicated-database because it adds extra replicas -# no-shared-merge-tree do something with parts on local fs -# add_minmax_index_for_numeric_columns=0: Adds extra files, which changes the hashes - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt sync;" -$CLICKHOUSE_CLIENT -q "CREATE TABLE rmt (a UInt8, b Int16, c Float32, d String, e Array(UInt8), f Nullable(UUID), g Tuple(UInt8, UInt16)) -ENGINE = ReplicatedMergeTree('/test/02253/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmtt', '1') ORDER BY a PARTITION BY b % 10 -SETTINGS old_parts_lifetime = 1, cleanup_delay_period = 0, cleanup_delay_period_random_add = 0, compress_marks=1, compress_primary_key=1, serialization_info_version = 'basic', -cleanup_thread_preferred_points_per_iteration=0, min_bytes_for_wide_part=0, remove_empty_parts=0, replace_long_file_name_to_hash=0, add_minmax_index_for_numeric_columns=0" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "INSERT INTO rmt SELECT rand(1), 0, 1 / rand(3), toString(rand(4)), [rand(5), rand(6)], rand(7) % 2 ? NULL : generateUUIDv4(), (rand(8), rand(9)) FROM numbers(1000);" - -$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" -$CLICKHOUSE_CLIENT -q "select count() from rmt" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt' and name='0_0_0_0'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -rf "$path" - -# detach the broken part, replace it with empty one -$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" 2>/dev/null -$CLICKHOUSE_CLIENT -q "select count() from rmt" - -$CLICKHOUSE_CLIENT --receive_timeout=60 -q "system sync replica rmt" - -# the empty part should pass the check -$CLICKHOUSE_CLIENT -q "check table rmt settings check_query_single_value_result = 1" -$CLICKHOUSE_CLIENT -q "select count() from rmt" - -$CLICKHOUSE_CLIENT -q "select name, part_type, hash_of_all_files, hash_of_uncompressed_files, uncompressed_hash_of_compressed_files from system.parts where database=currentDatabase()" - -$CLICKHOUSE_CLIENT -q "drop table rmt sync;" diff --git a/tests/queries/0_stateless/02254_projection_broken_part.sh b/tests/queries/0_stateless/02254_projection_broken_part.sh deleted file mode 100755 index 84f4ef8bfe53..000000000000 --- a/tests/queries/0_stateless/02254_projection_broken_part.sh +++ /dev/null @@ -1,45 +0,0 @@ -#!/usr/bin/env bash -# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" - -$CLICKHOUSE_CLIENT -q "create table projection_broken_parts_1 (a int, b int, projection ab (select a, sum(b) group by a)) - engine = ReplicatedMergeTree('/test/02254/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r1') - order by a settings index_granularity = 1;" - -$CLICKHOUSE_CLIENT -q "create table projection_broken_parts_2 (a int, b int, projection ab (select a, sum(b) group by a)) - engine = ReplicatedMergeTree('/test/02254/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r2') - order by a settings index_granularity = 1;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into projection_broken_parts_1 values (1, 1), (1, 2), (1, 3);" -$CLICKHOUSE_CLIENT -q "system sync replica projection_broken_parts_2;" -$CLICKHOUSE_CLIENT -q "select 1, *, _part from projection_broken_parts_2 order by b;" -$CLICKHOUSE_CLIENT -q "select 2, sum(b) from projection_broken_parts_2 group by a;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='projection_broken_parts_1' and name='all_0_0_0'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -f "$path/ab.proj/data.bin" - -$CLICKHOUSE_CLIENT -q "select 3, sum(b) from projection_broken_parts_1 group by a format Null;" 2>/dev/null - -num_tries=0 -while ! $CLICKHOUSE_CLIENT -q "select 4, sum(b) from projection_broken_parts_1 group by a format Null;" 2>/dev/null; do - sleep 1; - num_tries=$((num_tries+1)) - if [ $num_tries -eq 60 ]; then - break - fi -done - -$CLICKHOUSE_CLIENT -q "system sync replica projection_broken_parts_1;" -$CLICKHOUSE_CLIENT -q "select 5, sum(b) from projection_broken_parts_1 group by a;" - -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" diff --git a/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh b/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh deleted file mode 100755 index de16ba1a0bff..000000000000 --- a/tests/queries/0_stateless/02255_broken_parts_chain_on_start.sh +++ /dev/null @@ -1,44 +0,0 @@ -#!/usr/bin/env bash -# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" - -$CLICKHOUSE_CLIENT -q "create table rmt1 (a int, b int) - engine = ReplicatedMergeTree('/test/02255/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r1') order by a settings old_parts_lifetime=100500;" - -$CLICKHOUSE_CLIENT -q "create table rmt2 (a int, b int) - engine = ReplicatedMergeTree('/test/02255/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', 'r2') order by a settings old_parts_lifetime=100500;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1, 1), (1, 2), (1, 3);" -$CLICKHOUSE_CLIENT -q "alter table rmt1 update b = b*10 where 1 settings mutations_sync=1" -$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" -$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt2 order by b;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_0_0'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -f "$path/data.bin" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_0_0_1'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -f "$path/data.bin" - -$CLICKHOUSE_CLIENT -q "detach table rmt1 sync" -$CLICKHOUSE_CLIENT -q "attach table rmt1" 2>/dev/null - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by b;" - -$CLICKHOUSE_CLIENT -q "truncate table rmt1" - -$CLICKHOUSE_CLIENT -q "SELECT table, lost_part_count FROM system.replicas WHERE database=currentDatabase() AND lost_part_count!=0"; - -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists projection_broken_parts_1 sync;" diff --git a/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh b/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh deleted file mode 100755 index cc4a3b53b957..000000000000 --- a/tests/queries/0_stateless/02369_lost_part_intersecting_merges.sh +++ /dev/null @@ -1,52 +0,0 @@ -#!/usr/bin/env bash -# Tags: zookeeper, no-shared-merge-tree, long, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) -# no-shared-merge-tree: depend on local fs - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" - -$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') order by n;" -$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') order by n;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" -$CLICKHOUSE_CLIENT -q "system stop merges rmt2;" -$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" - -$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by n;" -$CLICKHOUSE_CLIENT -q "select 2, *, _part from rmt2 order by n;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_1_1'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -rf $path - -$CLICKHOUSE_CLIENT -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR -$CLICKHOUSE_CLIENT --min_bytes_to_use_direct_io=1 --local_filesystem_read_method=pread_threadpool -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR - -$CLICKHOUSE_CLIENT -q "detach table rmt1;" -$CLICKHOUSE_CLIENT -q "attach table rmt1;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" -$CLICKHOUSE_CLIENT -q "system start merges rmt2;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" -$CLICKHOUSE_CLIENT -q "select 3, *, _part from rmt1 order by n;" -$CLICKHOUSE_CLIENT -q "select 4, *, _part from rmt2 order by n;" - -$CLICKHOUSE_CLIENT -q "detach table rmt1;" -$CLICKHOUSE_CLIENT -q "attach table rmt1;" - -$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh b/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh deleted file mode 100755 index 72a749f0ec8e..000000000000 --- a/tests/queries/0_stateless/02370_lost_part_intersecting_merges.sh +++ /dev/null @@ -1,58 +0,0 @@ -#!/usr/bin/env bash -# Tags: long, zookeeper, no-shared-merge-tree, no-parallel, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) -# no-shared-merge-tree: depend on local fs (remove parts) - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" - -$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') order by n - settings cleanup_delay_period=0, cleanup_delay_period_random_add=0, cleanup_thread_preferred_points_per_iteration=0, old_parts_lifetime=0" -$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) engine=ReplicatedMergeTree('/test/02369/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') order by n" - -$CLICKHOUSE_CLIENT -q "system stop replicated sends rmt2" -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt2 values (0);" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1 pull;" - -# There's a stupid effect from "zero copy replication": -# MERGE_PARTS all_1_2_1 can be executed by rmt2 even if it was assigned by rmt1 -# After that, rmt2 will not be able to execute that merge and will only try to fetch the part from rmt2 -# But sends are stopped on rmt2... - -(sleep 5 && $CLICKHOUSE_CLIENT -q "system start replicated sends rmt2") & - -$CLICKHOUSE_CLIENT --optimize_throw_if_noop=1 -q "optimize table rmt1;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" - -$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt1 order by n;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_1_2_1'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -rf $path - -$CLICKHOUSE_CLIENT -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR -$CLICKHOUSE_CLIENT --min_bytes_to_use_direct_io=1 --local_filesystem_read_method=pread_threadpool -q "select * from rmt1;" 2>&1 | grep LOGICAL_ERROR - -$CLICKHOUSE_CLIENT -q "select sleep(0.1) from numbers($(($RANDOM % 30))) settings max_block_size=1 format Null" - -$CLICKHOUSE_CLIENT -q "detach table rmt1;" -$CLICKHOUSE_CLIENT -q "attach table rmt1;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" -$CLICKHOUSE_CLIENT -q "system sync replica rmt1 pull;" -$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "select 3, *, _part from rmt1 order by n;" - -$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh b/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh deleted file mode 100755 index 2f3f38cfba4c..000000000000 --- a/tests/queries/0_stateless/02444_async_broken_outdated_part_loading.sh +++ /dev/null @@ -1,36 +0,0 @@ -#!/usr/bin/env bash -# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) - -CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CURDIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt sync;" -$CLICKHOUSE_CLIENT -q "create table rmt (n int) engine=ReplicatedMergeTree('/test/02444/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/rmt', '1') order by n settings old_parts_lifetime=600" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt values (1);" -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt values (2);" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt pull;" -$CLICKHOUSE_CLIENT --optimize_throw_if_noop=1 -q "optimize table rmt final" -$CLICKHOUSE_CLIENT -q "system sync replica rmt;" -$CLICKHOUSE_CLIENT -q "select 1, *, _part from rmt order by n;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt' and name='all_1_1_0'") -# ensure that path is absolute before removing -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path')" || exit -rm -f "$path/*.bin" - -$CLICKHOUSE_CLIENT -q "detach table rmt sync;" -$CLICKHOUSE_CLIENT -q "attach table rmt;" -$CLICKHOUSE_CLIENT -q "select 2, *, _part from rmt order by n;" - -$CLICKHOUSE_CLIENT -q "truncate table rmt;" - -$CLICKHOUSE_CLIENT -q "detach table rmt sync;" -$CLICKHOUSE_CLIENT -q "attach table rmt;" - -$CLICKHOUSE_CLIENT -q "SELECT table, lost_part_count FROM system.replicas WHERE database=currentDatabase() AND lost_part_count!=0"; - -$CLICKHOUSE_CLIENT -q "drop table rmt sync;" diff --git a/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh b/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh deleted file mode 100755 index 7c1a2df1c8fb..000000000000 --- a/tests/queries/0_stateless/04215_replicated_missing_covered_part_on_start.sh +++ /dev/null @@ -1,59 +0,0 @@ -#!/usr/bin/env bash -# Tags: long, zookeeper, no-shared-merge-tree, no-cas-storage -# no-shared-merge-tree: depends on local fs -# no-cas-storage: test asserts system.parts.path is an absolute local FS path; on a cas disk the path is a relative object-storage key (orthogonal part-file path-shape) - -CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CUR_DIR"/../shell_config.sh - -$CLICKHOUSE_CLIENT -q "drop table if exists rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table if exists rmt2 sync;" - -zk_path="/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/$CLICKHOUSE_DATABASE/replicas/1/parts" - -$CLICKHOUSE_CLIENT -q "create table rmt1 (n int) - engine=ReplicatedMergeTree('/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '1') - order by n - settings old_parts_lifetime=100500;" - -$CLICKHOUSE_CLIENT -q "create table rmt2 (n int) - engine=ReplicatedMergeTree('/test/04215/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/{database}', '2') - order by n - settings old_parts_lifetime=100500;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (1);" -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (2);" - -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt2;" -$CLICKHOUSE_CLIENT -q "system stop merges rmt2;" -$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" - -path=$($CLICKHOUSE_CLIENT -q "select path from system.parts where database='$CLICKHOUSE_DATABASE' and table='rmt1' and name='all_0_1_1'") -$CLICKHOUSE_CLIENT -q "select throwIf(substring('$path', 1, 1) != '/', 'Path is relative: $path') format Null" || exit -rm -rf "$path" - -if $CLICKHOUSE_CLIENT -q "select * from rmt1;" >/dev/null 2>&1; then - echo "Expected read from removed part to fail" - exit 1 -fi - -$CLICKHOUSE_CLIENT -q "detach table rmt1 sync;" -$CLICKHOUSE_CLIENT -q "attach table rmt1;" - -$CLICKHOUSE_CLIENT --insert_keeper_fault_injection_probability=0 -q "insert into rmt1 values (3);" -$CLICKHOUSE_CLIENT -q "system start merges rmt2;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" -$CLICKHOUSE_CLIENT -q "optimize table rmt1 final;" -$CLICKHOUSE_CLIENT -q "system sync replica rmt1;" - -$CLICKHOUSE_CLIENT -q "select throwIf(count() = 0, 'Missing all_0_1_1 in ZooKeeper') from system.zookeeper where path='$zk_path' and name='all_0_1_1' format Null" - -$CLICKHOUSE_CLIENT -q "detach table rmt1 sync;" -$CLICKHOUSE_CLIENT -q "attach table rmt1;" - -$CLICKHOUSE_CLIENT -q "select count(), sum(n) from rmt1;" - -$CLICKHOUSE_CLIENT -q "drop table rmt1 sync;" -$CLICKHOUSE_CLIENT -q "drop table rmt2 sync;" diff --git a/tests/queries/0_stateless/04327_reader_executor_metrics.sql b/tests/queries/0_stateless/04327_reader_executor_metrics.sql index 65cfb6b88da7..776478a029b3 100644 --- a/tests/queries/0_stateless/04327_reader_executor_metrics.sql +++ b/tests/queries/0_stateless/04327_reader_executor_metrics.sql @@ -1,20 +1,12 @@ -<<<<<<< HEAD --- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas --- Like 04316, the executor falls back on the distributed cache and decryption --- (which can't be disabled from the test), so its metrics would not be emitted on --- those storage configs. Skip them; the test still runs on local disk and plain --- object storage where the executor engages. --- no-parallel-replicas: the counters are incremented on whichever replica reads the --- mark, so the initiator's `query_log` row does not carry them. -======= --- Tags: no-distributed-cache, no-encrypted-storage, no-cas-storage +-- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas, no-cas-storage -- Like 04316, the executor falls back on the distributed cache and decryption -- (which can't be disabled from the test), so its metrics would not be emitted on -- those storage configs. Skip them; the test still runs on local disk and plain -- object storage where the executor engages. Content-addressed storage always -- adds a `file_view` stage (byte window inside a shared blob), which the -- executor falls back on the same way -- see `ReadPipeline::tryBuildReaderExecutor`. ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) +-- no-parallel-replicas: the counters are incremented on whichever replica reads the +-- mark, so the initiator's `query_log` row does not carry them. -- -- Checks that the experimental ReaderExecutor emits its observability metrics. -- Reads a MergeTree table with `use_reader_executor = 1` and verifies, via the diff --git a/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql b/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql index 5c9b82c57bcc..5cd702a203f5 100644 --- a/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql +++ b/tests/queries/0_stateless/04328_reader_executor_kpi_async_metric.sql @@ -1,18 +1,11 @@ -<<<<<<< HEAD --- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas --- The executor falls back on the distributed cache and decryption (which can't be --- disabled from the test), so its metrics would not be emitted there; skip those --- configs (as in 04316 / 04327). --- no-parallel-replicas: the counters are incremented on whichever replica reads the --- mark, so the initiator's `query_log` row does not carry them. -======= --- Tags: no-distributed-cache, no-encrypted-storage, no-cas-storage +-- Tags: no-distributed-cache, no-encrypted-storage, no-parallel-replicas, no-cas-storage -- The executor falls back on the distributed cache and decryption (which can't be -- disabled from the test), so its metrics would not be emitted there; skip those -- configs (as in 04316 / 04327). Content-addressed storage always adds a -- `file_view` stage (byte window inside a shared blob), which the executor -- falls back on the same way -- see `ReadPipeline::tryBuildReaderExecutor`. ->>>>>>> a49d9ed16df (Merge pull request #2159 from Altinity/feature/antalya-26.6/CAS) +-- no-parallel-replicas: the counters are incremented on whichever replica reads the +-- mark, so the initiator's `query_log` row does not carry them. -- -- End-to-end check that the modeled-cost KPI asynchronous metric -- `ReaderExecutorModeledCostMsPerRequestedMiB` moves when the executor does work. From d225334d9e5d083bf8744296dff0fcc9d70783d5 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 10 Sep 2026 13:51:51 +0200 Subject: [PATCH 03/15] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/2300 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements CAS improvements # Conflicts: # src/Common/ErrorCodes.cpp # src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp # src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h # src/IO/ReadBufferFromS3.cpp # src/IO/ReadSettings.h # src/IO/S3/tests/TestPocoHTTPServer.h # src/IO/WriteSettings.h --- .../server-config/storing-data.mdx | 14 +- docs/en/antalya/cas/architecture/backend.md | 14 +- .../cas/architecture/garbage-collection.md | 1 + .../cas/architecture/manifests-and-refs.md | 2 +- .../cas/architecture/mounts-and-leases.md | 78 +- docs/en/antalya/cas/architecture/read-path.md | 39 +- .../antalya/cas/architecture/replication.md | 43 +- .../cas/architecture/storage-layout.md | 2 +- docs/en/antalya/cas/bucket-requirements.md | 74 +- docs/en/antalya/cas/configuration.md | 72 +- docs/en/antalya/cas/index.md | 8 + docs/en/antalya/cas/operations/debugging.md | 38 +- docs/en/antalya/cas/operations/migration.md | 6 +- docs/en/antalya/cas/operations/monitoring.md | 15 +- .../antalya/cas/operations/troubleshooting.md | 36 +- docs/en/antalya/cas/quick-start.md | 6 +- .../en/operations/system-tables/cas_gc_log.md | 7 +- docs/en/operations/system-tables/cas_log.md | 6 +- programs/disks/CommandCaInspect.cpp | 12 +- src/Common/CurrentMetrics.cpp | 2 + src/Common/ErrorCodes.cpp | 20 + src/Common/FailPoint.cpp | 4 +- src/Common/ProfileEvents.cpp | 25 +- src/Common/setThreadName.h | 2 +- .../ContentAddressed/Backend/CasBackend.h | 438 +-- .../ContentAddressed/Backend/CasEtag.cpp | 34 + .../ContentAddressed/Backend/CasEtag.h | 90 + .../ContentAddressed/Backend/CasFence.h | 35 + .../ContentAddressed/Backend/CasHotKeys.cpp | 360 ++ .../ContentAddressed/Backend/CasHotKeys.h | 124 + .../Backend/CasInMemoryBackend.cpp | 452 ++- .../Backend/CasInMemoryBackend.h | 241 +- .../Backend/CasInstrumentedBackend.cpp | 18 +- .../Backend/CasInstrumentedBackend.h | 136 +- .../Backend/CasObjectStorageBackend.cpp | 940 +++-- .../Backend/CasObjectStorageBackend.h | 283 +- .../ContentAddressed/Backend/CasProbe.cpp | 302 +- .../ContentAddressed/Backend/CasProbe.h | 33 +- .../Backend/CasRequestBudget.cpp | 71 + .../Backend/CasRequestBudget.h | 66 + .../Backend/CasRequestControl.cpp | 912 ----- .../Backend/CasRequestControl.h | 635 ---- .../ContentAddressed/Backend/CasRequests.cpp | 1220 +++++++ .../ContentAddressed/Backend/CasRequests.h | 527 +++ .../ContentAddressed/Backend/CasRetry.cpp | 32 + .../ContentAddressed/Backend/CasRetry.h | 80 + .../Backend/CasSentinelProbe.cpp | 79 +- .../Backend/CasSentinelProbe.h | 26 +- .../Backend/CasThrottlingBackend.h | 161 + .../Backend/CasTransportAccess.h | 25 + .../ContentAddressed/Backend/CasWriteResult.h | 130 + .../ContentAddressedMetadataStorage.cpp | 146 +- .../ContentAddressedMetadataStorage.h | 58 +- .../ContentAddressedSettings.cpp | 49 +- .../ContentAddressedSettings.h | 19 +- .../ContentAddressedTransaction.cpp | 33 +- .../ContentAddressedTransaction.h | 10 +- .../Formats/CasBlobEnvelopeFormat.cpp | 163 +- .../Formats/CasBlobEnvelopeFormat.h | 39 +- .../Formats/CasBlobMetaFormat.cpp | 52 +- .../Formats/CasBlobMetaFormat.h | 7 + .../Formats/CasEnvelopeLimits.h | 16 + .../Formats/CasFoldSealFormat.cpp | 315 +- .../Formats/CasFoldSealFormat.h | 76 +- .../ContentAddressed/Formats/CasFormat.cpp | 87 +- .../ContentAddressed/Formats/CasFormat.h | 109 +- .../Formats/CasGcMaintenanceStateFormat.cpp | 12 +- .../Formats/CasGcOutcomesFormat.cpp | 89 +- .../Formats/CasGcOutcomesFormat.h | 13 +- .../Formats/CasGcStateFormat.cpp | 76 +- .../Formats/CasGcStateFormat.h | 4 +- .../ContentAddressed/Formats/CasLayout.cpp | 19 +- .../ContentAddressed/Formats/CasLayout.h | 20 + .../Formats/CasPartManifestFormat.cpp | 151 +- .../Formats/CasPartManifestFormat.h | 16 +- .../Formats/CasPoolMetaFormat.cpp | 132 +- .../Formats/CasPoolMetaFormat.h | 28 +- .../Formats/CasRecordStreamFormat.cpp | 164 +- .../Formats/CasRecordStreamFormat.h | 73 +- .../Formats/CasRefCatalogFormat.cpp | 113 +- .../Formats/CasRefCatalogFormat.h | 12 +- .../Formats/CasRefCkptFormat.cpp | 77 +- .../Formats/CasRefLogFormat.cpp | 278 +- .../Formats/CasRefLogFormat.h | 29 +- .../Formats/CasRefSnapshotFormat.cpp | 161 +- .../Formats/CasRefSnapshotFormat.h | 4 +- .../Formats/CasRefWireVocab.cpp | 42 +- .../Formats/CasRefWireVocab.h | 13 +- .../Formats/CasServerRootFormats.cpp | 107 +- .../Formats/CasServerRootFormats.h | 16 +- .../Formats/CasTextFormat.cpp | 159 +- .../ContentAddressed/Formats/CasTextFormat.h | 122 +- .../ContentAddressed/Formats/CasWireVocab.cpp | 109 +- .../ContentAddressed/Formats/CasWireVocab.h | 168 +- .../ContentAddressed/Formats/README.md | 78 +- .../ContentAddressed/Gc/CasBlobInDegree.cpp | 142 +- .../ContentAddressed/Gc/CasBlobInDegree.h | 82 +- .../ContentAddressed/Gc/CasGc.cpp | 1040 ++++-- .../ContentAddressed/Gc/CasGc.h | 84 +- .../ContentAddressed/Gc/CasGcKeyReader.h | 25 + .../Gc/CasGcMaintenanceState.cpp | 22 +- .../Gc/CasGcMaintenanceState.h | 21 +- .../ContentAddressed/Gc/CasGcMetaWriter.cpp | 62 +- .../ContentAddressed/Gc/CasGcMetaWriter.h | 26 +- .../ContentAddressed/Gc/CasGcReadAhead.cpp | 139 + .../ContentAddressed/Gc/CasGcReadAhead.h | 103 + .../ContentAddressed/Gc/CasGcScheduler.cpp | 111 +- .../ContentAddressed/Gc/CasGcScheduler.h | 19 +- .../ContentAddressed/Gc/CasGcShardPlan.cpp | 8 +- .../ContentAddressed/Gc/CasGcShardPlan.h | 15 +- .../Gc/CasNamespaceJanitor.cpp | 62 +- .../ContentAddressed/Gc/CasNamespaceJanitor.h | 18 +- .../Gc/CasOrphanManifestSweep.cpp | 307 +- .../Gc/CasOrphanManifestSweep.h | 54 +- .../Gc/CatalogLifecycleReconciler.cpp | 25 +- .../Gc/CatalogLifecycleReconciler.h | 14 +- .../Parts/PartFolderAccess.cpp | 56 +- .../ContentAddressed/Parts/PartFolderAccess.h | 53 +- .../ContentAddressed/Pool/CasBlobMeta.cpp | 27 +- .../ContentAddressed/Pool/CasBlobMeta.h | 79 +- .../ContentAddressed/Pool/CasDetachedWork.cpp | 3 +- .../ContentAddressed/Pool/CasDetachedWork.h | 7 +- .../ContentAddressed/Pool/CasKeyReader.cpp | 32 + .../ContentAddressed/Pool/CasKeyReader.h | 55 + .../Pool/CasManifestReader.cpp | 44 +- .../ContentAddressed/Pool/CasManifestReader.h | 55 +- .../ContentAddressed/Pool/CasMountRuntime.cpp | 228 +- .../ContentAddressed/Pool/CasMountRuntime.h | 138 +- .../ContentAddressed/Pool/CasPartWriteTxn.cpp | 275 +- .../ContentAddressed/Pool/CasPartWriteTxn.h | 11 +- .../ContentAddressed/Pool/CasPlainObjects.cpp | 112 +- .../ContentAddressed/Pool/CasPlainObjects.h | 67 +- .../ContentAddressed/Pool/CasPool.cpp | 548 +-- .../ContentAddressed/Pool/CasPool.h | 177 +- .../ContentAddressed/Pool/CasPoolMeta.cpp | 109 +- .../ContentAddressed/Pool/CasRefCatalog.cpp | 515 +-- .../ContentAddressed/Pool/CasRefCatalog.h | 233 +- .../ContentAddressed/Pool/CasRefCkpt.cpp | 208 +- .../ContentAddressed/Pool/CasRefCkpt.h | 103 +- .../ContentAddressed/Pool/CasRefLedger.cpp | 1505 ++++---- .../ContentAddressed/Pool/CasRefLedger.h | 188 +- .../ContentAddressed/Pool/CasRefProtocol.cpp | 50 +- .../ContentAddressed/Pool/CasRefProtocol.h | 43 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 1479 ++++---- .../ContentAddressed/Pool/CasServerRoot.h | 275 +- .../Primitives/CasBlobDigest.cpp | 14 +- .../Primitives/CasBlobDigest.h | 16 +- .../Primitives/CasEnumWireTable.h | 73 + .../Primitives/CasEnumWireTableAsserts.h | 38 + .../ContentAddressed/Primitives/CasEvent.h | 11 +- .../ContentAddressed/Primitives/CasTypes.h | 15 +- .../Primitives/CasWriteOnceKey.h | 26 + .../ContentAddressed/README.md | 17 +- .../Tools/CasDecommission.cpp | 237 +- .../ContentAddressed/Tools/CasDecommission.h | 17 +- .../ContentAddressed/Tools/CasFsck.cpp | 118 +- .../ContentAddressed/Tools/CasInspect.cpp | 143 +- .../ContentAddressed/Tools/CasInspect.h | 4 +- .../benchmarks/benchmark_cas_ref_protocol.cpp | 528 ++- .../AzureBlobStorage/AzureObjectStorage.h | 2 + .../ObjectStorages/IObjectStorage.cpp | 35 + .../ObjectStorages/IObjectStorage.h | 33 +- .../ObjectStorages/S3/S3ObjectStorage.cpp | 359 +- .../ObjectStorages/S3/S3ObjectStorage.h | 92 +- src/Disks/tests/cas_format_test_battery.h | 33 +- src/Disks/tests/cas_sweep_test_support.h | 11 +- src/Disks/tests/cas_test_helpers.h | 1350 +++++--- src/Disks/tests/gtest_ca_transaction.cpp | 36 +- src/Disks/tests/gtest_ca_wiring.cpp | 177 +- src/Disks/tests/gtest_cas_b140_dangle.cpp | 7 +- src/Disks/tests/gtest_cas_backend.cpp | 1290 +++---- .../tests/gtest_cas_backend_contract.cpp | 199 +- .../tests/gtest_cas_backend_generation.cpp | 290 +- src/Disks/tests/gtest_cas_backend_listing.cpp | 35 +- src/Disks/tests/gtest_cas_blob_digest.cpp | 18 +- .../tests/gtest_cas_blob_envelope_format.cpp | 180 +- src/Disks/tests/gtest_cas_blob_indegree.cpp | 451 ++- src/Disks/tests/gtest_cas_blob_meta.cpp | 110 +- .../tests/gtest_cas_blob_meta_format.cpp | 56 +- .../tests/gtest_cas_bootstrap_ordering.cpp | 233 +- .../tests/gtest_cas_bulk_delete_backend.cpp | 197 ++ .../tests/gtest_cas_bulk_delete_engine.cpp | 91 + .../tests/gtest_cas_confirm_exact_ref.cpp | 699 +++- src/Disks/tests/gtest_cas_decommission.cpp | 751 ++-- .../gtest_cas_decommission_catalog_duties.cpp | 103 +- src/Disks/tests/gtest_cas_detached_work.cpp | 351 +- src/Disks/tests/gtest_cas_empty_proof.cpp | 2 +- src/Disks/tests/gtest_cas_encoding_pins.cpp | 323 +- src/Disks/tests/gtest_cas_enum_wire_table.cpp | 131 + .../tests/gtest_cas_event_dispatcher.cpp | 37 +- src/Disks/tests/gtest_cas_event_log.cpp | 507 +-- .../tests/gtest_cas_fence_generation.cpp | 85 +- src/Disks/tests/gtest_cas_fold_seal_codec.cpp | 2 +- .../tests/gtest_cas_fold_seal_format.cpp | 123 +- src/Disks/tests/gtest_cas_forget.cpp | 79 +- src/Disks/tests/gtest_cas_format.cpp | 135 +- src/Disks/tests/gtest_cas_format_battery.cpp | 43 +- src/Disks/tests/gtest_cas_fsck.cpp | 227 +- src/Disks/tests/gtest_cas_gc_ack_floor.cpp | 559 ++- .../tests/gtest_cas_gc_arithmetic_intake.cpp | 49 +- src/Disks/tests/gtest_cas_gc_attempt.cpp | 52 +- src/Disks/tests/gtest_cas_gc_bounded_walk.cpp | 89 +- .../gtest_cas_gc_bulk_delete_fallback.cpp | 184 + src/Disks/tests/gtest_cas_gc_fold.cpp | 79 +- .../tests/gtest_cas_gc_frontier_gate.cpp | 873 +++-- src/Disks/tests/gtest_cas_gc_hold_grammar.cpp | 357 +- src/Disks/tests/gtest_cas_gc_key_reader.cpp | 134 + src/Disks/tests/gtest_cas_gc_leak.cpp | 39 +- src/Disks/tests/gtest_cas_gc_log.cpp | 355 +- .../gtest_cas_gc_maintenance_state_format.cpp | 201 +- .../gtest_cas_gc_manifest_bulk_delete.cpp | 196 ++ src/Disks/tests/gtest_cas_gc_meta_writer.cpp | 22 +- .../tests/gtest_cas_gc_outcomes_format.cpp | 89 +- src/Disks/tests/gtest_cas_gc_read_ahead.cpp | 553 +++ src/Disks/tests/gtest_cas_gc_rebuild.cpp | 159 +- src/Disks/tests/gtest_cas_gc_resume.cpp | 52 +- src/Disks/tests/gtest_cas_gc_round.cpp | 563 ++- src/Disks/tests/gtest_cas_gc_round_defer.cpp | 86 +- .../tests/gtest_cas_gc_shard_incarnation.cpp | 130 +- src/Disks/tests/gtest_cas_gc_shard_plan.cpp | 31 +- src/Disks/tests/gtest_cas_gc_state_format.cpp | 42 +- src/Disks/tests/gtest_cas_gc_stop_start.cpp | 7 +- .../tests/gtest_cas_gc_teardown_stop.cpp | 484 +++ .../tests/gtest_cas_gc_undercount_repro.cpp | 40 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 955 ++++-- .../tests/gtest_cas_holey_list_detector.cpp | 38 +- src/Disks/tests/gtest_cas_hot_keys.cpp | 771 +++++ src/Disks/tests/gtest_cas_ids.cpp | 15 +- src/Disks/tests/gtest_cas_inspect.cpp | 127 +- .../gtest_cas_iobjectstorage_defaults.cpp | 161 + src/Disks/tests/gtest_cas_json_writer.cpp | 42 +- .../tests/gtest_cas_lifecycle_condition.cpp | 93 +- .../tests/gtest_cas_lifecycle_snapshot.cpp | 5 +- .../tests/gtest_cas_list_liar_end_to_end.cpp | 34 +- src/Disks/tests/gtest_cas_manifest_reader.cpp | 49 + src/Disks/tests/gtest_cas_mount.cpp | 1410 +++++--- .../tests/gtest_cas_mount_claim_conflicts.cpp | 223 +- src/Disks/tests/gtest_cas_mount_runtime.cpp | 176 + ...est_cas_namespace_file_request_profile.cpp | 31 +- .../tests/gtest_cas_namespace_janitor.cpp | 571 ++-- .../tests/gtest_cas_namespace_life_id.cpp | 15 +- .../tests/gtest_cas_ns_creation_lifecycle.cpp | 350 +- .../tests/gtest_cas_ns_file_incarnation.cpp | 98 +- .../tests/gtest_cas_ns_file_read_contract.cpp | 41 +- src/Disks/tests/gtest_cas_observability.cpp | 188 +- src/Disks/tests/gtest_cas_operation_gate.cpp | 9 +- .../tests/gtest_cas_orphan_manifest_sweep.cpp | 162 +- .../tests/gtest_cas_orphan_nomination.cpp | 110 +- .../tests/gtest_cas_orphan_sweep_requests.cpp | 390 +++ src/Disks/tests/gtest_cas_parallel_commit.cpp | 2 +- .../tests/gtest_cas_part_folder_access.cpp | 413 +-- .../tests/gtest_cas_part_folder_view.cpp | 3 +- .../tests/gtest_cas_part_manifest_format.cpp | 125 +- src/Disks/tests/gtest_cas_part_write.cpp | 911 +++-- .../gtest_cas_part_write_root_dangle.cpp | 29 +- src/Disks/tests/gtest_cas_plain_objects.cpp | 88 + src/Disks/tests/gtest_cas_pluggable_hash.cpp | 168 +- src/Disks/tests/gtest_cas_pool.cpp | 1958 ++++++++--- src/Disks/tests/gtest_cas_pool_meta.cpp | 78 + src/Disks/tests/gtest_cas_probe.cpp | 343 +- .../tests/gtest_cas_protocol_scenarios.cpp | 93 +- .../gtest_cas_rebuild_condemn_nothing.cpp | 93 +- .../tests/gtest_cas_record_stream_format.cpp | 182 +- .../tests/gtest_cas_recovery_grounding.cpp | 146 +- .../tests/gtest_cas_recovery_streaming.cpp | 51 +- src/Disks/tests/gtest_cas_ref_carve.cpp | 12 +- src/Disks/tests/gtest_cas_ref_catalog.cpp | 1794 ++++++++-- .../gtest_cas_ref_catalog_birth_wiring.cpp | 444 ++- .../tests/gtest_cas_ref_chunked_flush.cpp | 306 +- src/Disks/tests/gtest_cas_ref_ckpt.cpp | 867 +++-- src/Disks/tests/gtest_cas_ref_ckpt_join.cpp | 174 +- .../tests/gtest_cas_ref_contiguous_alloc.cpp | 179 +- .../tests/gtest_cas_ref_epoch_seal_format.cpp | 48 +- src/Disks/tests/gtest_cas_ref_gc.cpp | 594 +++- .../tests/gtest_cas_ref_install_safety.cpp | 348 +- src/Disks/tests/gtest_cas_ref_log_format.cpp | 108 +- src/Disks/tests/gtest_cas_ref_protocol.cpp | 51 + .../tests/gtest_cas_ref_read_contract.cpp | 54 +- .../tests/gtest_cas_ref_recovery_cas_walk.cpp | 686 ++-- .../tests/gtest_cas_ref_snapshot_format.cpp | 55 +- ...test_cas_ref_snapshot_publish_ordering.cpp | 185 +- .../gtest_cas_ref_wedge_every_attempt.cpp | 877 +++-- src/Disks/tests/gtest_cas_ref_writer.cpp | 1281 ++++--- src/Disks/tests/gtest_cas_repoint.cpp | 2 +- src/Disks/tests/gtest_cas_request_control.cpp | 1494 -------- src/Disks/tests/gtest_cas_requests.cpp | 3033 +++++++++++++++++ .../tests/gtest_cas_retirement_sweep.cpp | 125 +- .../gtest_cas_s3_bulk_delete_fallback.cpp | 402 +++ .../gtest_cas_s3_single_attempt_client.cpp | 731 ++++ src/Disks/tests/gtest_cas_s3_staging.cpp | 300 +- src/Disks/tests/gtest_cas_sentinel_probe.cpp | 197 +- .../tests/gtest_cas_server_root_format.cpp | 49 +- src/Disks/tests/gtest_cas_settings.cpp | 77 +- .../tests/gtest_cas_shutdown_context.cpp | 5 +- src/Disks/tests/gtest_cas_slot_occupy.cpp | 446 ++- .../gtest_cas_sweep_deletion_premise.cpp | 48 +- src/Disks/tests/gtest_cas_text_format.cpp | 56 +- src/Disks/tests/gtest_cas_throttling_gate.cpp | 129 + .../tests/gtest_cas_truncate_reclaim.cpp | 8 +- .../tests/gtest_cas_txn_apply_ledger.cpp | 7 +- src/Disks/tests/gtest_cas_upload_detached.cpp | 88 +- src/Disks/tests/gtest_cas_upload_fanout.cpp | 80 +- src/Disks/tests/gtest_cas_upstream_slice.cpp | 777 +++++ src/Disks/tests/gtest_cas_wire_vocab.cpp | 313 +- src/Disks/tests/gtest_cas_write_once_key.cpp | 32 + src/Disks/tests/gtest_cas_writer_duties.cpp | 171 +- src/IO/ObjectStorageRequestMode.h | 18 + src/IO/ObjectStorageRequestProfile.h | 31 + src/IO/ReadBufferFromS3.cpp | 43 +- src/IO/ReadBufferFromS3.h | 28 + src/IO/ReadSettings.h | 22 + src/IO/S3/Client.cpp | 12 +- src/IO/S3/Client.h | 9 + src/IO/S3/Requests.h | 7 + src/IO/S3/deleteFileFromS3.cpp | 5 +- src/IO/S3/deleteFileFromS3.h | 3 +- src/IO/S3/getObjectInfo.cpp | 16 +- src/IO/S3/getObjectInfo.h | 5 +- src/IO/S3/tests/TestPocoHTTPServer.h | 54 +- src/IO/S3/tests/gtest_aws_s3_client.cpp | 20 +- src/IO/S3/tests/gtest_cas_aws_s3_client.cpp | 481 +++ .../tests/gtest_gcs_conditional_dialect.cpp | 4 +- src/IO/WriteBufferFromS3.cpp | 14 +- src/IO/WriteSettings.h | 53 +- src/IO/tests/gtest_cas_readbuffer_s3.cpp | 546 +++ src/IO/tests/gtest_cas_writebuffer_s3.cpp | 1165 +++++++ .../ContentAddressedGarbageCollectionLog.cpp | 11 +- .../ContentAddressedGarbageCollectionLog.h | 10 +- src/Storages/MergeTree/DataPartsExchange.cpp | 257 +- src/Storages/MergeTree/DataPartsExchange.h | 12 +- .../MergeTree/DataPartsExchangeCasRouting.cpp | 147 + .../MergeTree/DataPartsExchangeCasRouting.h | 103 + src/Storages/MergeTree/MergeTreeData.cpp | 9 + .../StorageSystemContentAddressedMounts.cpp | 8 +- .../tests/gtest_cas_relink_pool_routing.cpp | 158 + ...orage_policy_for_merge_tree_by_default.xml | 8 +- ...orage_policy_for_merge_tree_by_default.xml | 2 +- .../configs/storage_conf.xml | 4 + .../configs/storage_conf.xml | 2 + .../test_cas_gc_bulk_delete/__init__.py | 0 .../configs/storage_conf.xml | 38 + .../test_cas_gc_bulk_delete/test.py | 130 + .../test_cas_gc_s3/configs/storage_conf.xml | 2 + .../configs/storage_conf.xml | 2 + tests/integration/test_cas_gc_sharded/test.py | 16 +- .../test_cas_gcs/configs/config.xml | 4 + .../test_cas_gcs/gcs_mocks/server.py | 259 +- tests/integration/test_cas_gcs/test.py | 443 ++- .../test_cas_gcs_relink_liveness/__init__.py | 0 .../configs/storage_conf.xml | 27 + .../test_cas_gcs_relink_liveness/test.py | 306 ++ .../configs/storage_conf.xml | 2 + .../configs/storage_conf.xml | 2 + .../configs/request_budget_disks.xml | 51 + .../configs/storage_conf.xml | 14 + .../configs/unsafe_remount.xml | 14 + .../test_cas_mount_renewal_retry/test.py | 356 +- .../configs/storage_conf.xml | 4 + .../configs/storage_conf.xml | 2 + .../configs/storage_conf_other_pool.xml | 2 + .../configs/storage_conf_tiered.xml | 32 + .../test_cas_replicated_relink/test.py | 314 +- .../test_cas_s3/configs/storage_conf.xml | 6 +- .../configs/storage_conf.xml | 2 + tests/integration/test_gcs_live/test.py | 162 +- .../{04278_cas_disk.sql => 04278_cas_disk.sh} | 33 +- .../{04279_cas_gc.sql => 04279_cas_gc.sh} | 38 +- ...e_state.sql => 04282_cas_mutable_state.sh} | 40 +- .../04283_cas_replicated_rejected.sh | 60 + .../04283_cas_replicated_rejected.sql | 47 - .../04284_cas_backup_pointer_holding.sh | 13 +- ...5_cas_deduplication_window_inline_disk.sh} | 32 +- .../04286_cas_remote_data_paths.sh | 53 + .../04286_cas_remote_data_paths.sql | 41 - .../04287_cas_detach_partition_listing.sh | 50 + .../04287_cas_detach_partition_listing.sql | 38 - ...288_cas_detached_part_modification_time.sh | 52 + ...88_cas_detached_part_modification_time.sql | 40 - .../04289_cas_multi_detach_drop.sh | 55 + .../04289_cas_multi_detach_drop.sql | 43 - .../0_stateless/04290_cas_no_leftovers.sh | 2 +- .../0_stateless/04292_cas_mutations.sh | 12 +- .../04293_cas_lightweight_delete.sh | 12 +- .../0_stateless/04294_cas_patch_parts.sh | 12 +- .../04295_cas_mutation_no_leftovers.sh | 2 +- ...ql => 04299_cas_projection_inline_disk.sh} | 51 +- ...sql => 04300_cas_projection_multiblock.sh} | 36 +- .../0_stateless/05002_cas_fetch_partition.sh | 71 + .../0_stateless/05002_cas_fetch_partition.sql | 58 - .../0_stateless/05003_cas_freeze.reference | 2 +- tests/queries/0_stateless/05003_cas_freeze.sh | 21 +- .../0_stateless/05004_cas_transactions.sh | 12 +- .../0_stateless/05005_cas_backup_restore.sh | 14 +- .../0_stateless/05007_cas_gc_introspection.sh | 28 +- .../05008_cas_gc_snapshot_prune.sh | 6 +- .../0_stateless/05009_cas_event_log.sh | 56 + .../0_stateless/05009_cas_event_log.sql | 44 - .../0_stateless/05010_cas_mounts_gc_health.sh | 18 +- .../05015_cas_reject_fake_transaction.sh | 6 +- tests/queries/0_stateless/05020_cas_fsck.sh | 2 +- .../05023_cas_dropns_leaked_namespace.sh | 28 +- .../0_stateless/05024_cas_freeze_two_roots.sh | 24 +- .../05025_cas_attach_partition_cross_disk.sh | 34 +- .../05026_cas_manifest_path_newline.sh | 8 +- 404 files changed, 49265 insertions(+), 20825 deletions(-) create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasFence.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h delete mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.cpp delete mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasTransportAccess.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasWriteResult.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasEnvelopeLimits.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcKeyReader.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.cpp create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTable.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTableAsserts.h create mode 100644 src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasWriteOnceKey.h create mode 100644 src/Disks/tests/gtest_cas_bulk_delete_backend.cpp create mode 100644 src/Disks/tests/gtest_cas_bulk_delete_engine.cpp create mode 100644 src/Disks/tests/gtest_cas_enum_wire_table.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_key_reader.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_read_ahead.cpp create mode 100644 src/Disks/tests/gtest_cas_gc_teardown_stop.cpp create mode 100644 src/Disks/tests/gtest_cas_hot_keys.cpp create mode 100644 src/Disks/tests/gtest_cas_iobjectstorage_defaults.cpp create mode 100644 src/Disks/tests/gtest_cas_manifest_reader.cpp create mode 100644 src/Disks/tests/gtest_cas_mount_runtime.cpp create mode 100644 src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp create mode 100644 src/Disks/tests/gtest_cas_plain_objects.cpp create mode 100644 src/Disks/tests/gtest_cas_pool_meta.cpp create mode 100644 src/Disks/tests/gtest_cas_ref_protocol.cpp delete mode 100644 src/Disks/tests/gtest_cas_request_control.cpp create mode 100644 src/Disks/tests/gtest_cas_requests.cpp create mode 100644 src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp create mode 100644 src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp create mode 100644 src/Disks/tests/gtest_cas_throttling_gate.cpp create mode 100644 src/Disks/tests/gtest_cas_upstream_slice.cpp create mode 100644 src/Disks/tests/gtest_cas_write_once_key.cpp create mode 100644 src/IO/ObjectStorageRequestMode.h create mode 100644 src/IO/ObjectStorageRequestProfile.h create mode 100644 src/IO/S3/tests/gtest_cas_aws_s3_client.cpp create mode 100644 src/IO/tests/gtest_cas_readbuffer_s3.cpp create mode 100644 src/IO/tests/gtest_cas_writebuffer_s3.cpp create mode 100644 src/Storages/MergeTree/DataPartsExchangeCasRouting.cpp create mode 100644 src/Storages/MergeTree/DataPartsExchangeCasRouting.h create mode 100644 src/Storages/tests/gtest_cas_relink_pool_routing.cpp create mode 100644 tests/integration/test_cas_gc_bulk_delete/__init__.py create mode 100644 tests/integration/test_cas_gc_bulk_delete/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_gc_bulk_delete/test.py create mode 100644 tests/integration/test_cas_gcs_relink_liveness/__init__.py create mode 100644 tests/integration/test_cas_gcs_relink_liveness/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_gcs_relink_liveness/test.py create mode 100644 tests/integration/test_cas_mount_renewal_retry/configs/request_budget_disks.xml create mode 100644 tests/integration/test_cas_mount_renewal_retry/configs/unsafe_remount.xml create mode 100644 tests/integration/test_cas_replicated_relink/configs/storage_conf_tiered.xml rename tests/queries/0_stateless/{04278_cas_disk.sql => 04278_cas_disk.sh} (54%) mode change 100644 => 100755 rename tests/queries/0_stateless/{04279_cas_gc.sql => 04279_cas_gc.sh} (63%) mode change 100644 => 100755 rename tests/queries/0_stateless/{04282_cas_mutable_state.sql => 04282_cas_mutable_state.sh} (56%) mode change 100644 => 100755 create mode 100755 tests/queries/0_stateless/04283_cas_replicated_rejected.sh delete mode 100644 tests/queries/0_stateless/04283_cas_replicated_rejected.sql rename tests/queries/0_stateless/{04285_cas_deduplication_window_inline_disk.sql => 04285_cas_deduplication_window_inline_disk.sh} (50%) mode change 100644 => 100755 create mode 100755 tests/queries/0_stateless/04286_cas_remote_data_paths.sh delete mode 100644 tests/queries/0_stateless/04286_cas_remote_data_paths.sql create mode 100755 tests/queries/0_stateless/04287_cas_detach_partition_listing.sh delete mode 100644 tests/queries/0_stateless/04287_cas_detach_partition_listing.sql create mode 100755 tests/queries/0_stateless/04288_cas_detached_part_modification_time.sh delete mode 100644 tests/queries/0_stateless/04288_cas_detached_part_modification_time.sql create mode 100755 tests/queries/0_stateless/04289_cas_multi_detach_drop.sh delete mode 100644 tests/queries/0_stateless/04289_cas_multi_detach_drop.sql rename tests/queries/0_stateless/{04299_cas_projection_inline_disk.sql => 04299_cas_projection_inline_disk.sh} (72%) mode change 100644 => 100755 rename tests/queries/0_stateless/{04300_cas_projection_multiblock.sql => 04300_cas_projection_multiblock.sh} (59%) mode change 100644 => 100755 create mode 100755 tests/queries/0_stateless/05002_cas_fetch_partition.sh delete mode 100644 tests/queries/0_stateless/05002_cas_fetch_partition.sql create mode 100755 tests/queries/0_stateless/05009_cas_event_log.sh delete mode 100644 tests/queries/0_stateless/05009_cas_event_log.sql diff --git a/docs/concepts/features/configuration/server-config/storing-data.mdx b/docs/concepts/features/configuration/server-config/storing-data.mdx index ae854dd3a88a..45b720de326d 100644 --- a/docs/concepts/features/configuration/server-config/storing-data.mdx +++ b/docs/concepts/features/configuration/server-config/storing-data.mdx @@ -464,7 +464,9 @@ and the [`system.cas_gc_log`](/operations/system-tables/cas_gc_log), [content-addressed storage documentation](/antalya/cas) for the architecture, operations runbooks, and a live-validated quick start. -Configuration: +Configuration: `http_keep_alive_timeout` and `http_keep_alive_max_requests` are set here for the +reason explained under +[recommended keep-alive settings](/antalya/cas/configuration#recommended-keep-alive-settings). ```xml @@ -473,6 +475,8 @@ Configuration: cas https://s3.eu-west-1.amazonaws.com/clickhouse-eu-west-1.clickhouse.com/data/ 1 + 30 + 10000 server-{replica} disks/s3_cas/cas_scratch/ @@ -482,7 +486,6 @@ Configuration: 60 1 67108864 - always ``` @@ -539,15 +542,14 @@ disk-level and server-level settings surface. view cache. - `cas_part_folder_cache_max_entry_bytes` — `16` MiB by default. Maximum size of a single cached part-folder view entry. -- `cas_part_folder_validate` — `always` (default), `never`, or `age `. Controls how often a - `ForceFresh` read re-proves a cached manifest body via a `HEAD` request: `always` re-proves every - time (the original, pre-optimization behavior), `never` trusts the cache without re-proving, and - `age ` re-proves only once the cached entry is older than the given number of seconds. - `cas_manifest_decode_cache_bytes` — `128` MiB by default. Byte bound for the decoded-manifest cache. `0` disables decode caching entirely (a diagnostic mode). - `cas_gc_meta_pool_size` — `16` by default. Bounded thread-pool size for the GC's per-hash freshness-meta writes (condemn/spare/delete), so a mass `DROP` condemning millions of blobs does not run fully sequentially. +- `cas_gc_read_concurrency` — `16` by default. Bounded thread-pool size for the GC fold's read-ahead of + checkpoints, ref logs, manifest bodies and zero-candidate `HEAD`s. The fold's decisions stay on the + round thread in their original order; only the fetches overlap. `1` disables read-ahead. - `skip_access_check` — `false` by default. Skips the disk's `CAS` capability probe ("start now, fix later"). The server-level `skip_access_check` flag skips the generic disk access check; this disk key governs the `CAS` capability probe. diff --git a/docs/en/antalya/cas/architecture/backend.md b/docs/en/antalya/cas/architecture/backend.md index b1d88843471e..cc0740ea73da 100644 --- a/docs/en/antalya/cas/architecture/backend.md +++ b/docs/en/antalya/cas/architecture/backend.md @@ -108,15 +108,17 @@ storage, and GC would silently stop reclaiming. `runCapabilityProbe` (`Backend/CasProbe.cpp`) runs a throwaway-key battery against every writable mount, described in full on the [bucket requirements](/antalya/cas/bucket-requirements) page. It is fail-closed: any check that does not pass throws `NOT_IMPLEMENTED` naming the specific failure, and -the mount refuses to become writable. Two further gates run as the battery's opening steps, and one +the mount refuses to become writable. The one tolerated exception is a versioning probe that cannot +answer at all, described in the first bullet below. Two further gates run as the battery's opening steps, and one sits genuinely alongside it. The distinction matters: because the versioning check runs *inside* the battery, skipping the battery used to skip it too, which is exactly why the third gate exists. -- `checkPoolPreconditions` — inside the battery. On the `GCS`-dialect combination only, requires bucket versioning to be - *verifiably* off. A confirmed `Enabled` and an inconclusive probe both throw: `CAS` cannot assume - the safe answer here, because what it would do on a versioned bucket is delete objects it believes - it reclaimed. A probe is inconclusive when the credential may not read the bucket's versioning - configuration, or when the backend cannot answer at all. +- `checkPoolPreconditions` — inside the battery. On the `GCS`-dialect combination only, checks that + bucket versioning is off. A confirmed `Enabled` throws: what `CAS` would do on a versioned bucket is + delete objects it believes it reclaimed. An inconclusive probe — the credential may not read the + bucket's versioning configuration, or the backend cannot answer at all — logs a warning and lets + the mount proceed, since it is not evidence of a versioned bucket; verifying it then falls to the + operator, as it already does for soft delete. - `checkSkipAccessCheckSupport` — alongside the battery, in the skip branch of `Pool::open`, since it is the gate that decides whether the battery may be skipped at all. It asks whether the backend may serve a writable mount that skips the battery at all. The `GCS`-dialect combination refuses, so `skip_access_check = true` cannot reach a diff --git a/docs/en/antalya/cas/architecture/garbage-collection.md b/docs/en/antalya/cas/architecture/garbage-collection.md index c918a0a35f14..a604104c7796 100644 --- a/docs/en/antalya/cas/architecture/garbage-collection.md +++ b/docs/en/antalya/cas/architecture/garbage-collection.md @@ -226,6 +226,7 @@ the user-facing configuration surface. | Setting | Default | Bounds | |---|---|---| | `cas_gc_meta_pool_size` | 16 | bounded pool for condemn-marker writes | +| `cas_gc_read_concurrency` | 16 | bounded pool for the fold's read-ahead; `1` disables | ## Observability {#observability} diff --git a/docs/en/antalya/cas/architecture/manifests-and-refs.md b/docs/en/antalya/cas/architecture/manifests-and-refs.md index df86d82e575d..ba476cffe88a 100644 --- a/docs/en/antalya/cas/architecture/manifests-and-refs.md +++ b/docs/en/antalya/cas/architecture/manifests-and-refs.md @@ -101,7 +101,7 @@ swept for that root. flowchart TD A["LIST one page of cas/manifests/
freeze candidates with exact GET"] --> B{"build-prefix eligible?
durable watermark fact only"} B -->|"epoch less than lease epoch"| ELIG["eligible, old-epoch debris"] - B -->|"same epoch, min_active clears build_seq"| ELIG + B -->|"same epoch, min_active_build_sequence clears build_seq"| ELIG B -->|"no lease, or epoch ahead, or build may be live"| SKIP["skip"] ELIG --> C["protection view: committed manifests
plus live precommits
plus manifests with an unfolded minus-one"] C -->|"key protected"| SKIP2["skip"] diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md index d778756ce323..eb852b070da0 100644 --- a/docs/en/antalya/cas/architecture/mounts-and-leases.md +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -9,7 +9,7 @@ doc_type: 'reference' Page 4 of 4 in the CAS architecture set. Covers server identity, the mount lease that fences writers, and the server-scoped control-plane objects. No external coordinator is involved: there -is no ZooKeeper/Keeper client anywhere in this protocol — `MountLeaseKeeper` is a local lease +is no ZooKeeper/Keeper client anywhere in this protocol — `MountLeaseRenewer` is a local lease *renewer*, not a Keeper client. ## `cas_server_root_id` — the identity {#server-root-id} @@ -61,32 +61,36 @@ Two failure modes this closes: over, regardless of lease expiry. - A **same-uuid live twin** (two processes sharing one uuid file and `server_root_id`) is caught separately, by the mount claim's token-stability observation, and aborts with an operator-facing message rather - than corrupting the pool. + than corrupting the pool — this is the default behavior, with `cas_unsafe_remount_no_delay` off. + With it on, a same-uuid claim over such a slot reclaims at once instead of observing (see + `cas_unsafe_remount_no_delay` in the configuration reference). ## The mount lease {#mount-lease} One object, `gc/server-roots//mount`, carries **both** the liveness lease and the build watermark — there is no separate watermark object. `MountLease` fields: `server_uuid`, `writer_epoch`, `write_attempt_id`, `hostname`, `pid`, `started_at_ms`, renewal `seq`, -`expires_at_ms`, `min_active` (the build-watermark floor), and `gc_fenced`. +`expires_at_ms`, `min_active_build_sequence` (the build-watermark floor), and `gc_fenced`. - **Logical renewal identity.** Each holder-originated body has a fresh nonzero `write_attempt_id`. One logical renewal fixes one immutable `(key, bytes, expected token, write_attempt_id)` tuple before I/O. Every physical retry repeats it byte-for-byte; a later GC fence preserves the observed ID, while reclaim and successor bodies mint new IDs. - **Resolve before retry.** A transient or ambiguous conditional `PUT` is followed by one exact - `GET`. The keeper adopts the result only when the complete body, including `write_attempt_id`, + `GET`, except that an attempt whose transport error names a failed connection is reissued first + after a flat pause and settled by the reissue's own answer (a 2xx) or by the exact `GET` that + follows its `412`. The renewer adopts the result only when the complete body, including `write_attempt_id`, equals its immutable request. If the predecessor token is still current, another identical `PUT` may follow bounded backoff. A same-pair twin, GC-fenced body, successor, foreign holder, or absent body is never treated as this renewal. - **Absolute deadline.** Renewal uses `CLOCK_BOOTTIME`, not `CLOCK_MONOTONIC`, so a VM resumed from suspend correctly observes itself expired. Its absolute deadline is the minimum of the existing request-operation budget and the last confirmed lease deadline minus the safety margin. The - controller checks that one configured attempt still fits before each backend `PUT` or resolving + controller checks that one attempt envelope still fits before each backend `PUT` or resolving `GET`, after each interruptible backoff, and before accepting success. A retry, `GET`, response timestamp, or wall-clock step never extends authority. -- **Cadence.** The runtime normally starts a logical renewal every `mount_renew_period` (default - 10 s), with TTL `mount_lease_ttl_ms` (default 30 s, TTL/3 renewal ratio). The next beat is anchored +- **Cadence.** The runtime normally starts a logical renewal every `cas_mount_renew_period_ms` (default + 10 s), with TTL `cas_mount_lease_ttl_ms` (default 30 s, TTL/3 renewal ratio). The next beat is anchored at the committed body's pre-I/O BOOTTIME start. A slow recovery therefore causes an immediate catch-up beat when the nominal cadence has elapsed; it does not wait a fresh full period after the response. @@ -94,10 +98,10 @@ watermark — there is no separate watermark object. `MountLease` fields: `serve and rechecks it immediately before the object-store call and on every conditional retry. Reads are not gated. - **Request-budget admission.** `refAppendFenceOk` refuses to *start* a ref-log attempt unless - `attempt_timeout + safety_margin` fits inside the remaining lease, rejecting with - `BAD_ARGUMENTS` at request-admission time rather than mid-flight. + `2 × envelope + safety_margin` fits inside the remaining lease (a write and its settlement read), + rejecting with `BAD_ARGUMENTS` at request-admission time rather than mid-flight. -**Losing the lease is neither read-only mode nor a process abort.** `MountLeaseKeeper` is a +**Losing the lease is neither read-only mode nor a process abort.** `MountLeaseRenewer` is a synchronous durable-slot state machine. A committed result advances its token, sequence, confirmed BOOTTIME deadline, and cadence anchor. Any admitted deterministic failure, confirmed conflict, or ambiguity left at the deadline/attempt limit moves it to `RenewalTerminal`; it cannot mint another @@ -105,7 +109,7 @@ body or publish a clean farewell. Owner cancellation before any request is the o `NotAttempted` result and leaves clean release possible. Cancellation after a request was sent is terminal because that request may still land. -After the keeper call returns, `CasMountRuntime` consumes the result. A terminal result trips the +After the renewer call returns, `CasMountRuntime` consumes the result. A terminal result trips the local fence (latches `lost`, bumps the fence generation, moves the in-process runtime to `TransientNotLive`) and latches one self-remount generation. A confirmed foreign/successor or same-pair conflict remains a typed fail-closed error; it is never adopted. A real fence still costs @@ -115,9 +119,25 @@ inside authority already proved by the last confirmed lease. GC's own view of a dead server is symmetric and clock-skew-immune: a slot becomes fence-eligible only after the leader observes the *same* renewal token hold stable, on its own monotonic clock, -for `TTL + TTL/20 + cadence` — the identical formula a re-mounting server uses to wait out a -predecessor. The stamped `expires_at_ms` never participates in that decision; wall-clock `now` is -audit-only. +for `TTL + floor(TTL/20) + period` — close to, but not identical to, the threshold a re-mounting +server uses to wait out a predecessor, which observes `TTL + floor(TTL/20) + max(1, +floor(period/2))`. Both thresholds are evaluated purely on the observer's own clock and its own +configured `TTL`/`period`; nothing about the writer's timing travels on the wire. The stamped +`expires_at_ms` never participates in either decision — it is a writer-stamped diagnostic used by +`system.cas_mounts` and by the non-authoritative decommission epoch-recovery precheck, never an +authorization; local fencing is derived instead from the confirmed request's pre-I/O `BOOTTIME` +anchor plus the TTL, and wall-clock `now` stays audit-only. + +Every server sharing a pool must therefore run the identical `cas_mount_lease_ttl_ms` and +`cas_mount_renew_period_ms`: a member or GC leader configured with a shorter threshold than its +peers can fence out a healthy peer whose token-update gap merely exceeds that shorter threshold — +a peer renewing frequently stays live, one that missed a renewal does not. Change these values only +with every member of the pool stopped; a graceful restart removes only that member's own startup +observation and does not make mixed thresholds safe. With the defaults (TTL 30 s, period 10 s, +margin 2 s), `TTL − margin − period − 2 × envelope = 4 s` is the scheduling-lateness budget before +the first renewal attempt of a period can begin, where `envelope = attempt_timeout + 2 × cap` and +`cap` is `attempt_timeout` when the disk's `connect_timeout_ms` is `0`, else +`min(connect_timeout_ms, attempt_timeout)` (7 s with defaults). ## The two monotone counters {#counters} @@ -133,9 +153,9 @@ into "not found". Global build ordering is the **pair** `(writer_epoch, build_seq)` compared lexicographically — the exact comparison GC uses for eligibility. The durable authority for both is the mount object -itself: no mount means no deletion authority means nothing is swept. `min_active`, the oldest +itself: no mount means no deletion authority means nothing is swept. `min_active_build_sequence`, the oldest in-flight `build_seq`, rides in the same mount object as the watermark floor; `UINT64_MAX` in -`min_active` is the farewell/retired sentinel, not a real build. +`min_active_build_sequence` is the farewell/retired sentinel, not a real build. ## Mount claim outcomes {#claim-outcomes} @@ -153,9 +173,10 @@ a `MountClaimResult::Kind` together with a `MountPriorState` describing which ce | `MountPriorState` | Certificate that justified the reclaim | |---|---| | `None` | no reclaim needed (fresh claim or same-epoch refresh) | -| `Clean` | the predecessor's own graceful farewell (`min_active == UINT64_MAX`) | +| `Clean` | the predecessor's own graceful farewell (`min_active_build_sequence == UINT64_MAX`) | | `Fenced` | GC's own threshold-gated fence-out (`gc_fenced`) | | `UncleanObserved` | this claimant's own token-stability observation held for the full `TTL + drift` window | +| `UncleanUnsafe` | the operator's explicit `cas_unsafe_remount_no_delay` authorization carried the slot's exact token — not a certificate of death | ## Behavioral mount-slot model {#mount-state-machines} @@ -166,12 +187,13 @@ the claim outcomes above and is shown here as behavior, not as a type in the cod stateDiagram-v2 [*] --> Absent Absent --> Live: claimMount putIfAbsent, seq=1 - Live --> Live: keeper beat, putOverwrite seq+1 + Live --> Live: renewer beat, putOverwrite seq+1 Live --> Fenced: GC observes a stable token past threshold, gc_fenced=1, body preserved - Live --> Terminated: certified drain, terminal farewell (expires_at=now, min_active=MAX) + Live --> Terminated: certified drain, terminal farewell (expires_at=now, min_active_build_sequence=MAX) Fenced --> Live: same-uuid claim with a fresh writer_epoch, instant reclaim Terminated --> Live: same-uuid claim with a fresh writer_epoch, instant reclaim Live --> Live: same-uuid claim, proven-dead token via UncleanObserved + Live --> Live: same-uuid claim under cas_unsafe_remount_no_delay, no observation Fenced --> Fenced: same uuid and epoch claim, FencedSelf, no write Live --> Absent: decommission tail, mount then epoch then owner tombstone Terminated --> [*] @@ -202,24 +224,24 @@ under a live mount is an operator-level event. **Writable open** runs in a strict order: bootstrap-residual proof, capability probe under a random per-mount prefix, pool-meta create-or-validate, `validateServerRootId`, owner claim, -`allocateWriterEpoch`, mount claim and synchronous keeper start, materialization grace if the -predecessor was unclean (default 30 s), arm the fence, then create and release the runtime-owned -renewal and remount workers before the writable pool becomes externally visible. If the grace period -consumed the TTL, one fresh synchronous renewal re-anchors the deadline before the fence is armed. +`allocateWriterEpoch`, mount claim and synchronous renewer start, arm the fence, then create and +release the runtime-owned renewal and remount workers before the writable pool becomes externally +visible. If the claim consumed the TTL, one fresh synchronous renewal re-anchors the deadline +before the fence is armed. Failure to construct either worker joins the partial pair, closes the fence, and fails the writable open. No incident path constructs a thread. The renewal and remount workers are separate and long-lived under one stable `CasMountRuntime`. `scheduleRemount` increments a requested-generation latch and wakes the persistent remount worker, -including while an older generation is active. Before keeper replacement, remount requests -`ParkRequested` and waits for the renewal driver to report `Parked`, which proves that no keeper call +including while an older generation is active. Before renewer replacement, remount requests +`ParkRequested` and waits for the renewal driver to report `Parked`, which proves that no renewer call is in flight. A successful remount handles only its snapshotted generation; a newer request is processed before renewal resumes. **Clean unmount:** request stop and join both persistent workers, drain the ref lanes, and only if -the drain *certified* quiescence call `MountLeaseKeeper::release` on an `Active` keeper to write the -terminal farewell (`expires_at_ms` already expired, `min_active = UINT64_MAX`). That sentinel is what -lets a successor reclaim instantly. A `RenewalTerminal` keeper, an unresolved ref write, or a sent +the drain *certified* quiescence call `MountLeaseRenewer::release` on an `Active` renewer to write the +terminal farewell (`expires_at_ms` already expired, `min_active_build_sequence = UINT64_MAX`). That sentinel is what +lets a successor reclaim instantly. A `RenewalTerminal` renewer, an unresolved ref write, or a sent renewal ambiguity writes no farewell — an unearned farewell would let a successor start mutating while a stale conditional request from the predecessor is still in flight. diff --git a/docs/en/antalya/cas/architecture/read-path.md b/docs/en/antalya/cas/architecture/read-path.md index beb2d87d99fb..49a6864a0109 100644 --- a/docs/en/antalya/cas/architecture/read-path.md +++ b/docs/en/antalya/cas/architecture/read-path.md @@ -36,30 +36,33 @@ decoding the whole body is cheaper than any partial-read machinery would be. | Cache | Keyed by | Setting | Default | What still hits the network | |---|---|---|---|---| -| Manifest decode cache | `(ManifestId, Token)` | `cas_manifest_decode_cache_bytes` | 128 MiB | A mandatory `HEAD` on **every** access, cache hit or miss | -| Part-folder view cache (`Cas::CachedPartFolderAccess`, `Parts/PartFolderAccess.h`) | Part ref key | `cas_part_folder_cache_bytes`, `cas_part_folder_cache_max_entries`, `cas_part_folder_cache_max_entry_bytes` | 64 MiB / 10 000 entries / 16 MiB | Its `ForceFresh` policy re-proves the manifest body via that same mandatory `HEAD`, paced by `cas_part_folder_validate` (`always` \| `never` \| `age `) | +| Manifest decode cache | `ManifestId` | `cas_manifest_decode_cache_bytes` | 128 MiB | Nothing on a hit; one `GET` on a miss | +| Part-folder view cache (`Cas::CachedPartFolderAccess`, `Parts/PartFolderAccess.h`) | Part ref key | `cas_part_folder_cache_bytes`, `cas_part_folder_cache_max_entries`, `cas_part_folder_cache_max_entry_bytes` | 64 MiB / 10 000 entries / 16 MiB | Nothing on a validated hit; a `ForceFresh` access bypasses the retained view and rebuilds from the manifest decode cache | -**The `HEAD` is mandatory even on a cache hit** — the page's most counter-intuitive fact, because it -means a cache hit still costs one object-store round trip: +**A cache hit costs no request.** A manifest id is minted once and its body is written once, so one +id names one content forever and a cached decode can be served without asking the object store: ```mermaid flowchart TD - A["readManifestShared(ManifestId)"] --> B["HEAD the manifest key"] - B -->|"absent"| C["throw FILE_DOESNT_EXIST --
a live ref must never name a missing object"] - B -->|"present, token t"| D{"cache lookup (ManifestId, t)"} - D -->|hit| E["return the cached decode -- no GET"] - D -->|miss| F["GET the body"] - F --> G{"body's own ref and namespace
match the key?"} - G -->|no| H["throw CORRUPTED_DATA"] - G -->|yes| I["decode, insert into cache keyed by (ManifestId, t), return"] + A["readManifestShared(ManifestId)"] --> B{"decode cache lookup by ManifestId"} + B -->|hit| C["return the cached decode -- no request"] + B -->|miss| D["GET the body"] + D -->|"absent"| E["throw FILE_DOESNT_EXIST --
a live ref must never name a missing object"] + D -->|"present"| F{"body's own ref and namespace
match the key?"} + F -->|no| G["throw CORRUPTED_DATA"] + F -->|yes| H["decode, insert into the cache keyed by ManifestId, return"] ``` -The `HEAD` is what proves the live ref still names an existing object — the no-dangle invariant — -and it supplies the token that keys the cache; only then is the decode cache consulted. On a miss, -the `GET` is followed by the two identity checks in the diagram, each `CORRUPTED_DATA` on failure. -Only a fully validated decode enters the cache. Setting either cache's byte budget to `0` disables -retention while leaving the `HEAD`-and-validate sequence intact — a cache is purely an -optimization, never a trust boundary. +On a miss, the `GET` is followed by the two identity checks in the diagram, each `CORRUPTED_DATA` on +failure, and only a fully validated decode enters the cache. A live ref that names a missing body is +detected on a miss, by the garbage collector before it deletes a manifest, and by `fsck`; a reader +holding a cached decode for a manifest the collector has since removed sees a snapshot-consistent +manifest and fails with a typed error when it reads a blob that is gone. Write paths that carry +entries forward from a committed part (hardlinks, renames, single-file rewrites, relink) adopt the +source blobs on the strength of the source ref's live edge, which the collector honours; deleting +objects out of band, behind the collector's back, is outside that contract and is what `fsck` +reports. Setting either cache's byte budget to `0` disables retention while leaving the +`GET`-and-validate sequence intact — a cache is purely an optimization, never a trust boundary. The part-folder view cache is invalidated on every promote and repoint, and is single-flight on a cold build: concurrent readers of the same not-yet-cached view coalesce into one build rather than diff --git a/docs/en/antalya/cas/architecture/replication.md b/docs/en/antalya/cas/architecture/replication.md index 371ea74b39f7..a9f3221badf7 100644 --- a/docs/en/antalya/cas/architecture/replication.md +++ b/docs/en/antalya/cas/architecture/replication.md @@ -31,11 +31,11 @@ sequenceDiagram participant Snd as Sender participant S3 as Shared pool - R->>Snd: GET part, cas_pool_uuid = R's pool uuid, client_protocol_version = 11 + R->>Snd: GET part, cas_pool_uuid = every pool of R's policy, client_protocol_version = 11 Note over R: advertising 11 is a promise to confirm before promoting Snd->>Snd: same disk pool uuid? identity, never endpoint plus prefix Snd->>S3: resolve the offer once -- manifest bytes and confirm token from the SAME view - Snd-->>R: cookie cas_relink = part_manifest_v2, cookie cas_source_token = ..., body = manifest bytes + Snd-->>R: cookies cas_relink = part_manifest_v2, cas_source_token = ..., cas_pool_uuid = the matched pool -- body = manifest bytes Note over Snd: sender is fire-and-forget -- it releases the part here rect rgba(120,160,255,0.12) @@ -63,7 +63,7 @@ sequenceDiagram | # | Gate | What it enforces | |---|---|---| -| 1 | Pool identity | The receiver advertises `cas_pool_uuid`; the sender offers relink only if its own disk's pool uuid is **equal**. Matching by endpoint and prefix was tried and rejected — a minted pool uuid is the identity | +| 1 | Pool identity | The receiver advertises `cas_pool_uuid` — the pool uuids of every content-addressed disk of its storage policy that is not read-only, as one list — and the sender offers relink only if its own disk's pool uuid is **in** it, naming that uuid in a `cas_pool_uuid` response cookie. Matching by endpoint and prefix was tried and rejected — a minted pool uuid is the identity | | 2 | Protocol version 11 | On the receiver side, advertising it is a promise to run the confirm round trip before promoting | | 3 | One resolution for two outputs | The manifest bytes and the confirm token come from the **same** view. Two separate calls would allow a repoint in between and hand the receiver a token naming a manifest whose entries it never adopted | | 4 | The receiver trusts nothing from the wire but the entry list | The sender's manifest id, namespace and payload digest are ignored; the target namespace and ref come from the receiver's own router, and manifest path hygiene is validated at decode | @@ -77,6 +77,40 @@ cannot be entered twice for one fetch. Byte-fetched files content-address and de anyway, so falling back never loses the dedup property, only the zero-byte-move property for that one fetch. +## Where a relinked part lands {#relink-placement} + +The offer decides the disk. Once the sender has named the pool, the receiver places the part on the +first disk of its storage policy that belongs to that pool, and reserves space there directly — ahead +of everything the policy would otherwise consult: volume order, JBOD balancing, +`max_data_part_size_bytes`, and `TTL ... TO DISK|VOLUME` move rules. A part that is already in the +pool never travels as bytes merely because the policy would have put it somewhere else. + +A TTL rule is not ignored, it is deferred: the background mover sees a part that is not in its TTL +destination and moves it there afterwards. The bytes then travel once, as a read from the pool on the +receiver, and the sender is never loaded. + +Two things do not bend to the offer. A disk the caller supplied (zero-copy `MOVE` re-fetching a shared +part onto the move's destination) is never overridden — a content-addressed disk cannot reach that path +at all, since it does not support zero-copy replication. And a read-only disk is never a candidate: its +pool is advertised only if some other disk of that pool in the policy is writable, and when none is, the +sender streams bytes and the ordinary placement applies. + +A pool disk that is not live — its mount lease lost, its identity lost, or the storage shut down — is +still the target. The relink's own write gate refuses it and the fetch fails; a replication-queue fetch +is retried by the queue, while a manual `FETCH PART` or `FETCH PARTITION` reports the error to the user. +The part is never quietly placed on another disk instead. This is the behaviour a single-disk +content-addressed policy always had, and a mixed policy now shares it. + +The byte-fetch fallback after a relink that failed for a mechanism reason (a corrupted manifest, a +body-absent precommit, a ref conflict) re-requests the bytes on the same pool disk, where they +content-address and deduplicate against the pool — the placement outlives the relink. A manifest of a +newer format generation is not degraded to bytes today (a tracked gap, `[relink-fallback-unknown-format-version]` +in the backlog). + +During a rolling upgrade a sender that predates the pool-set advertise compares the whole `cas_pool_uuid` +value with its own pool id, so a receiver whose policy holds several pools gets bytes from such a sender +until it is upgraded; a receiver with one pool is unaffected, its advertise is byte-for-byte the old one. + ## What actually seals "commit before release" {#relink-seal} The receiver's `+1` — its precommit binding — is durable **before** the sender is asked anything, @@ -109,7 +143,8 @@ content, and that root can confirm its exact refs. `DETACH`, `ATTACH`, `delete_tmp_` cleanup, and merge-result renames all reduce to the same two moves: re-key any *staged* source into the destination, then `republishRef(src → dst)` for any -*committed* source. `republishRef` re-reads the source manifest freshly, publishes an +*committed* source. `republishRef` resolves the source ref freshly and reads its manifest through +the manifest cache, publishes an equivalent-entry manifest under the destination ref — a **new** manifest id, with blobs untouched and adopted by evidence — then drops the source ref. A destination that already exists with identical entries just drops the source, an idempotent re-drive; one with different entries diff --git a/docs/en/antalya/cas/architecture/storage-layout.md b/docs/en/antalya/cas/architecture/storage-layout.md index e4d20725836b..b136552acf37 100644 --- a/docs/en/antalya/cas/architecture/storage-layout.md +++ b/docs/en/antalya/cas/architecture/storage-layout.md @@ -45,7 +45,7 @@ namespace's shape and never interprets its contents. | `gc/gen//attempt//outcomes//.zst` | GC outcome log | `cas_gc_outcomes` | GC | | `gc/server-roots//owner` | server-root owner singleton | `cas_owner` | mount | | `gc/server-roots//epoch` | server-root epoch singleton | `cas_epoch` | mount | -| `gc/server-roots//mount` | mount lease (incl. `min_active` watermark) | `cas_mount_lease` | mount | +| `gc/server-roots//mount` | mount lease (incl. `min_active_build_sequence` watermark) | `cas_mount_lease` | mount | | `roots/` | loose mountpoint object, verbatim | — (never interpreted) | upper layers | | `staging//…` | S3-native upload staging scratch | — | writer, own mount only | diff --git a/docs/en/antalya/cas/bucket-requirements.md b/docs/en/antalya/cas/bucket-requirements.md index 67a3795ab682..700d0a94d371 100644 --- a/docs/en/antalya/cas/bucket-requirements.md +++ b/docs/en/antalya/cas/bucket-requirements.md @@ -32,11 +32,13 @@ Bucket **versioning is not required** — in fact it must be **disabled** on the dialect (see below), because a token-exact delete on a versioned bucket archives a noncurrent generation instead of reclaiming storage, silently stopping GC reclamation. -On the generation-token dialect that requirement is checked, and checked strictly: a writable mount -proceeds only when the probe *confirms* versioning is disabled. A bucket reported as versioned and a -probe that could not answer — the credential may not read the bucket's versioning configuration, or -the backend cannot report it — both refuse the mount. `CAS` does not assume the safe answer, because -the failure it would be assuming away is `GC` deleting objects it believes it reclaimed. +On the generation-token dialect that requirement is checked at mount. A bucket reported as versioned +refuses the mount. A probe that could not answer — the credential may not read the bucket's versioning +configuration (`storage.buckets.get` on GCS), or the backend cannot report it — does not: the mount +proceeds and logs a warning naming what it could not verify, because an unreadable configuration is +not evidence of a versioned bucket, and refusing on it would turn a missing IAM grant into an outage. +In that case confirming that versioning is disabled is your responsibility, exactly as soft delete is +below; grant the permission if you want the mount to verify it for you. Because that check is part of the mount battery, `skip_access_check = true` is refused on a writable generation-token disk. Mount the disk read-only if you need to start before the access check can @@ -54,6 +56,68 @@ Soft delete does not leave the deleted generation live, so it does not break exa the way versioning does. What it does is delay physical reclamation until the retention period expires: `GC` reports space as reclaimed while the bill still reflects it. +## Request rate, and the limit that is not the one you expect {#request-rate} + +Google Cloud Storage publishes two kinds of ceiling, and the one that constrains `CAS` is the +smaller and less-known of them. + +A bucket starts at roughly **1000 object writes per second** — uploads, updates and deletes — and +roughly **5000 object reads per second**, counting listings and metadata reads as reads. Those +ceilings are not fixed: Cloud Storage raises them by splitting the index range behind the bucket, +which it says takes "on the order of minutes" to detect and act on. Buckets with a hierarchical +namespace start up to eight times higher. + +Separately, Cloud Storage applies **a much smaller limit to repeated writes to the same object +name**. Google documents that this limit exists but does not publish its value. Measured against a +live bucket from this codebase, it begins to bite at approximately one mutation per second on a +single key, and it does not participate in the auto-scaling above — splitting an index range cannot +help a single name. + +That second limit is the one `CAS` meets first, because two of its objects are single fixed names +written on a hot path: + +| Object | One per | Written on | +|---|---|---| +| `cas/ns/state//_ckpt` | table | every durable ref-log transaction, plus namespace birth, epoch seal and snapshot | +| `cas/ref_catalog` | pool | twice per `CREATE TABLE` and twice per `DROP TABLE` | + +Blob bodies and their metadata sidecars are named by content hash and are therefore spread the way +Google's own guidance asks for: it recommends "completely random object names" for the best load +distribution, and a hashed prefix where names would otherwise be sequential. Ref-log transactions are +sequential within a namespace but are written under a per-namespace prefix, so they scale with the +number of tables rather than sharing one index range. + +### What this means for a deployment {#rate-consequences} + +- **A single table commits at about one transaction per second** on Google Cloud Storage. Inserts, + merges and mutations on that table queue behind the checkpoint write; they do not fail, but the + lane's throughput is capped and each flush's tail takes longer than it would on a store without + the per-name limit. +- **A pool performs about one table lifecycle transition per second.** Concurrent `CREATE TABLE` or + `DROP TABLE` beyond that rate contends on the pool-wide catalog. Test suites that create hundreds + of tables in parallel are the case that provokes this; ordinary production DDL is not. +- **Ramping up gradually is Google's documented expectation.** Its guidance is to increase the + request rate "no faster than doubling the rate over a period of 20 minutes", and to pause or + reduce the rate when latency or error rates rise. A pool that goes from idle to full write load in + one step will see throttling before the bucket has redistributed the load. + +### Throttling is a retryable condition, not a failure {#rate-errors} + +Cloud Storage signals a rate it will not serve with HTTP `429`, `408`, or a `5xx` status, and its +retry guidance names all three, together with socket timeouts and TCP disconnects, as retryable with +exponential backoff and jitter. Every mutable-object write `CAS` issues carries a generation +precondition, which places it in Google's *conditionally idempotent* class — a retry either applies +exactly once or fails the precondition, never applies twice. Retrying them is therefore safe by +Google's own rule, not merely by ours. + +### Reads over a wide-area link want a cache disk {#rate-reads} + +The read ceiling is high enough that `CAS` does not approach it, but latency is a separate matter. +A cacheless `CAS` disk pays a round trip per column file per part: measured against a bucket in +another region, a `SELECT` issued about 725 ranged reads and took 3.6 seconds at the median and 15.7 +seconds at the ninety-ninth percentile. Put a `cache` disk in front of the `CAS` disk for any +deployment where the bucket is not local to the server. + ## Platform support {#platform-support} The deterministic request-construction coverage is green, but the diff --git a/docs/en/antalya/cas/configuration.md b/docs/en/antalya/cas/configuration.md index d7ee12130993..bf0fe17f7dd3 100644 --- a/docs/en/antalya/cas/configuration.md +++ b/docs/en/antalya/cas/configuration.md @@ -15,7 +15,9 @@ A `CAS` disk is an `object_storage` disk with `metadata_type` set to `cas` and a `cas_server_root_id`. The recommended shape layers a `type=cache` disk in front of it — the local filesystem cache absorbs repeated reads of the same blob, while the `CAS` disk underneath stays the single source of truth the pool's other members and GC also read from. The storage policy references -the **cached** disk, not the raw `CAS` disk directly: +the **cached** disk, not the raw `CAS` disk directly. `http_keep_alive_timeout` and +`http_keep_alive_max_requests` are set here for the reason explained under +[recommended keep-alive settings](#recommended-keep-alive-settings): ```xml @@ -29,6 +31,8 @@ the **cached** disk, not the raw `CAS` disk directly: https://bucket.s3.amazonaws.com/cas/ ... ... + 30 + 10000 cache @@ -92,17 +96,80 @@ entirely before release. Treat this table as a snapshot of the current build, no | `cas_blob_hash` | `cityhash128` | Pool blob content-hash function (`cityhash128` \| `xxh3-128` \| `sha256`). Recorded in the pool at creation; a mismatching config is refused at mount | | `cas_blob_hash_allow_new` | `false` | Explicit opt-in to admit a new hash algorithm into an existing pool. One-way: once admitted, the pool carries both algorithms permanently | | `skip_access_check` | `false` | Skip the boot-time capability probe (start now, fix later). Only the preflight probe is skipped — the conditional-write correctness check still runs on every writable mount. **Not available on a writable generation-token (GCS) disk**, which refuses to mount with it: there, the probe battery is the only proof that a token-exact delete carries its generation precondition. Mount such a disk read-only if you need to defer the check | +| `cas_mount_lease_ttl_ms` | `30000` | Milliseconds for which a mount lease remains valid after a successful claim or renewal (≥ 1). Lower values shorten stale-mount recovery but reduce tolerance for object-storage and scheduling delays | +| `cas_mount_renew_period_ms` | `10000` | Milliseconds between background mount-lease renewals (≥ 1). It must leave enough time for two attempt envelopes (a renewal write and its settlement read) and the lease safety margin before the TTL expires: `period + 2 × envelope + margin < TTL` | | `cas_gc_snapshot_generations_to_keep` | `3` | GC snapshot generations retained | | `cas_gc_shards` | `1` | Blob-hash-prefix reducer shards (≥ 1). Recorded in the pool at creation; a mismatching config is refused at mount | | `gcs_max_conditional_put_bytes` | 1 GiB | Largest conditional non-blob `PUT` on a generation-token store, including create-if-absent metadata/control artifacts and conditional replacements. Blob publication is unconditional, uses ordinary multipart, and is not subject to this cap | | `cas_part_folder_cache_bytes` | 64 MiB | Part-folder view cache byte budget (`0` disables retention) | | `cas_part_folder_cache_max_entries` | `10000` | Part-folder view cache entry cap | | `cas_part_folder_cache_max_entry_bytes` | 16 MiB | Oversized part-folder views bypass retention above this size | -| `cas_part_folder_validate` | `always` | Cache body re-proof policy (`always` \| `never` \| `age `). **Leave at `always`**: the other modes trade the fail-closed body-existence check for an optimization — this is a trust decision about unverified data, not a performance knob | | `cas_manifest_decode_cache_bytes` | 128 MiB | Manifest decode cache byte budget (`0` disables) | | `cas_gc_meta_pool_size` | `16` | Bounded pool size for GC per-hash freshness-meta writes | +| `cas_gc_read_concurrency` | `16` | Bounded pool size for the GC fold's read-ahead of checkpoints, ref logs, manifests and zero-candidate HEADs; `1` disables | +| `cas_attempt_timeout_ms` | `5000` | Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. Together with the connect cap it forms the attempt envelope (`cas_attempt_timeout_ms + 2 × cap`; the cap is `cas_attempt_timeout_ms` itself when the disk's `connect_timeout_ms` is `0`, else `min(connect_timeout_ms, cas_attempt_timeout_ms)`) that the lease arithmetic reserves: one TCP connect and one TLS handshake under the cap each, send/receive bounded per socket operation by `cas_attempt_timeout_ms`. With background renewal the cadence check requires `cas_mount_renew_period_ms + 2 × envelope + cas_lease_safety_margin_ms < cas_mount_lease_ttl_ms`, which puts an effective ceiling on the frozen connect cap: under the defaults (TTL 30000, period 10000, margin 2000) the envelope must stay under 9000, so a disk `connect_timeout_ms` of 2000 ms or more refuses to open writable — lower the connect timeout or raise the TTL if you hit this | +| `cas_lease_safety_margin_ms` | `2000` | Startup-only margin validated against the mount lease TTL: the attempt envelope + `cas_lease_safety_margin_ms` must be strictly less than the mount lease TTL, and `cas_mount_renew_period_ms` + 2 × envelope + `cas_lease_safety_margin_ms` too, or the disk refuses to open writable | +| `cas_unsafe_remount_no_delay` | `0` | Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same `server_uuid` (a copied uuid file, a stalled predecessor). After such a reclaim the predecessor can still start conditional writes until its own cutoff (`confirmed deadline − cas_lease_safety_margin_ms − 2 × envelope`) or until its next renewal meets the token guard, and a request it already sent may still materialize later. That is not a data hazard: ref-log keys carry `(writer_epoch, sequence)` and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles any straggler (recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions). The exposure is availability, not data. Intended for test stands and deployments that guarantee one process per uuid | | `cas_staging_backend` | `local` | Blob staging backend (`local` \| `s3`); `s3` is opt-in and requires native same-store copy on writable mount | +All servers sharing a pool must run the same `cas_mount_lease_ttl_ms` and `cas_mount_renew_period_ms`. +Startup reclaim and GC's fence-out both judge liveness by the mount slot's write token holding stable +on the observer's own `CLOCK_BOOTTIME`, using the observer's own threshold — nothing about a writer's +timing travels on the wire. Startup observes `cas_mount_lease_ttl_ms + floor(cas_mount_lease_ttl_ms / +20) + max(1, floor(cas_mount_renew_period_ms / 2))`; GC observes `cas_mount_lease_ttl_ms + +floor(cas_mount_lease_ttl_ms / 20) + cas_mount_renew_period_ms`. A pool member or GC leader +configured with a shorter threshold than its peers can therefore fence out a healthy peer whose +token-update gap exceeds that shorter threshold — a peer renewing frequently stays live, one that +missed a renewal does not. Change these values only with every member of the pool stopped: a +graceful restart removes only that member's own startup observation and does not make mixed +thresholds safe. + +A shorter TTL reduces the tolerance for object-storage delays; a shorter renewal period increases it +(renewal starts earlier) at the cost of more background traffic. With the defaults, +`cas_mount_lease_ttl_ms − cas_lease_safety_margin_ms − cas_mount_renew_period_ms − 2 × envelope = +4000` ms is the scheduling-lateness budget before the first renewal attempt of a period can begin, +where `envelope = cas_attempt_timeout_ms + 2 × cap` (7000 ms with defaults) and `cap` is +`cas_attempt_timeout_ms` when the disk's `connect_timeout_ms` is `0`, else +`min(connect_timeout_ms, cas_attempt_timeout_ms)` (1000 ms with defaults); the renewal then keeps +retrying until `confirmed deadline − cas_lease_safety_margin_ms`. + +The `expires_at_ms` stamped into the mount object is a writer-stamped diagnostic used by +`system.cas_mounts` and by the non-authoritative decommission epoch-recovery precheck; it never +authorizes a reclaim or a GC fence-out. Local fencing is derived instead from the confirmed +request's pre-I/O `CLOCK_BOOTTIME` anchor plus the TTL. + +## Recommended keep-alive settings {#recommended-keep-alive-settings} + +On a `CAS` disk, set `http_keep_alive_timeout` to `30` and `http_keep_alive_max_requests` to `10000`, +alongside the disk's other settings: + +```xml + + + + + object_storage + s3 + cas + {replica} + https://example-bucket.s3.amazonaws.com/cas/ + ... + ... + 30 + 10000 + + + + +``` + +The generic S3 default, `http_keep_alive_max_requests = 100`, is the whole lifetime of a +connection under `CAS`'s control-plane request rate rather than a headroom margin: every ~100 +requests, a connection is torn down and recreated, and its local port then cycles through +`TIME_WAIT`. Under sustained load this churn exhausts the ephemeral port range +(`EADDRNOTAVAIL`) and starves the mount-lease renewal request. Raising the two settings above +removes that churn, with no measured cost. + ## Advanced GC pacing settings {#advanced-gc-pacing-settings} These settings bound individual phases of a `GC` round. The first two accept any `UInt64` value; @@ -112,6 +179,7 @@ for the remaining caps, `0` means unbounded. |---|---|---|---| | `cas_manifest_sweep_list_budget_keys` | `1000` | `UInt64` | Orphan-manifest sweep `LIST` budget per round | | `cas_manifest_sweep_delete_budget_keys` | `100` | `UInt64` | Orphan-manifest sweep `DELETE` budget per round | +| `cas_gc_bulk_delete_chunk_keys` | `1000` | `1`–`1000` | Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots) | | `cas_gc_round_graduation_budget` | `5000` | `0` = unbounded | Blob-graduation (`condemned` → `delete_pending`) cohort cap per round | | `cas_gc_round_redelete_budget` | `5000` | `0` = unbounded | Exact-token re-delete cohort cap for prior `delete_pending` rows per round | | `cas_gc_round_sweep_namespace_budget` | `20` | `0` = unbounded | Distinct namespaces per orphan-manifest sweep page whose protection view may be built | diff --git a/docs/en/antalya/cas/index.md b/docs/en/antalya/cas/index.md index 2bc71046494b..cb1d563eebf0 100644 --- a/docs/en/antalya/cas/index.md +++ b/docs/en/antalya/cas/index.md @@ -64,6 +64,14 @@ Two consequences for planning: Each prefix is a fully independent pool (its own refs, leases, and `GC`), so rounds stay short regardless of the total fleet size. +:::tip +For replicated tables on `CAS`, enable +[`execute_merges_on_single_replica_time_threshold`](/operations/settings/merge-tree-settings#execute_merges_on_single_replica_time_threshold). +This lets one replica perform each merge while the others wait for and fetch the resulting part, +avoiding redundant merge work across replicas. Set the threshold higher than the usual merge +duration for your workload. +::: + ## Status {#status} `CAS` is **experimental**. It ships in Altinity Antalya builds. Experimental means the on-disk diff --git a/docs/en/antalya/cas/operations/debugging.md b/docs/en/antalya/cas/operations/debugging.md index cbc8b0c693f0..cb2e48f4f3b3 100644 --- a/docs/en/antalya/cas/operations/debugging.md +++ b/docs/en/antalya/cas/operations/debugging.md @@ -112,8 +112,6 @@ SELECT event_time_microseconds, event_type, outcome, reason, detail['write_attempt_id'] AS write_attempt_id, detail['attempts_sent'] AS attempts_sent, detail['classification'] AS classification, - detail['deadline_source'] AS deadline_source, - detail['stop_cause'] AS stop_cause, detail['attempt_no'] AS remount_attempt, detail['step'] AS remount_step, detail['error'] AS error @@ -123,15 +121,33 @@ WHERE disk_name = 'cas' ORDER BY event_time_microseconds; ``` -Interpret the sequence as follows: - -- `retrying -> recovered` with the same `write_attempt_id` means an in-budget blip recovered in the - existing epoch; `classification = 'committed_by_get'` means exact `GET` proved a landed request, - while `committed_after_retry` means a later identical physical `PUT` completed. -- A `failed` renewal carries the decisive `unresolved_reason`, `deadline_source`, `stop_cause`, and - `classification`. `external_lease_deadline`, `cancelled`, `conflict`, - `fence_or_lifecycle_lost`, and `attempts_exhausted` are different operator diagnoses; do not - collapse them into a generic timeout. +A `watermark_renew` row now carries only two detail keys beyond the identifying ones: +`attempts_sent` (the number of physical HTTP attempts the whole logical renewal made) and +`classification`. There is no per-attempt `retrying` row any more — a renewal that recovers after +one or more physical attempts produces exactly one `recovered` row when it settles, not a `retrying` +row followed by a `recovered` one — and the older `unresolved_reason`, `deadline_source`, and +`stop_cause` keys are gone; everything they used to distinguish is now named directly by +`classification`. Interpret the sequence as follows: + +- `outcome = 'recovered'` means an in-budget renewal landed, in the same epoch. `classification` + says how: `committed_by_read` means an exact `GET` proved a landed request; `committed_after_retry` + means a later identical physical `PUT` completed and the response itself proved it. +- `outcome = 'failed'` carries the decisive `classification`: `external_lease_deadline` (the + confirmed lease's own safety margin, not the request policy, ran out first — check object-store + latency or `BOOTTIME` advancement before anything else), `request_deadline` (the ninety-second + request policy exhausted first), `unresolved` (every attempt was ambiguous and never settled by + the time the operation gave up), `conflict` (an exact resolve read found another body — a + same-pair twin, a GC-fenced body, a successor epoch, or a foreign holder), `cancelled` (a + renewal in flight was cancelled by shutdown or a remount park request; expected during graceful + shutdown), `fence_or_lifecycle_lost` (another local fence loss or a terminal lifecycle transition + closed admission while the operation was active), `deterministic_failure` (the store's own + answer proved the write never applied), and `vanished` (an exact resolve read proved the mount + slot absent — the pool directory was removed or renamed out of band, or a decommission raced the + renewal). `terminal_unclassified` means the renewal terminated through a path that assigned no + classification; that is a defect to report together with the surrounding rows, not an operator + condition. Do not collapse these into a generic timeout — the action + differs by classification, and only `external_lease_deadline` and `request_deadline` are about a + deadline at all. - A following `mount_remount` row names the whole-chain `attempt_no` and final `step`. An `ok` row restored `Live` under the reported fresh `writer_epoch`; a `failed` row's `step` and optional `error` identify where that whole-chain attempt stopped. diff --git a/docs/en/antalya/cas/operations/migration.md b/docs/en/antalya/cas/operations/migration.md index df51e79144a3..e8258b96d61d 100644 --- a/docs/en/antalya/cas/operations/migration.md +++ b/docs/en/antalya/cas/operations/migration.md @@ -19,7 +19,9 @@ disk and its data are untouched until a partition is explicitly moved. A storage policy can carry both an ordinary disk and a `CAS` disk as separate volumes. `ALTER TABLE ... MOVE PARTITION ... TO DISK` then moves data between them without an `INSERT`/`DROP` cycle. As on the [configuration](/antalya/cas/configuration#disk-config) page, the recommended shape layers a -`type=cache` disk over the `CAS` disk, and the policy's volume references the **cached** disk name: +`type=cache` disk over the `CAS` disk, and the policy's volume references the **cached** disk name. +`http_keep_alive_timeout` and `http_keep_alive_max_requests` are set here for the reason explained +under [recommended keep-alive settings](/antalya/cas/configuration#recommended-keep-alive-settings): ```xml @@ -37,6 +39,8 @@ the [configuration](/antalya/cas/configuration#disk-config) page, the recommende https://bucket.s3.amazonaws.com/cas/ ... ... + 30 + 10000 cache diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md index 58600e8e058c..8bde03328575 100644 --- a/docs/en/antalya/cas/operations/monitoring.md +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -82,9 +82,12 @@ SETTINGS system_events_show_zero_values = 1; ``` `system.cas_log` records only nontrivial logical renewals. A `watermark_renew` row has outcome -`retrying`, `recovered`, or `failed`, with detail keys `server_root_id`, `writer_epoch`, `seq`, a -shortened `write_attempt_id`, `attempts_sent`, `elapsed_ms`, `remaining_confirmed_budget_ms`, -`unresolved_reason`, `deadline_source`, `stop_cause`, and `classification`. Ordinary first-attempt +`recovered` or `failed` — there is no per-attempt `retrying` row; the terminal event is the whole +story — with detail keys `server_root_id`, `writer_epoch`, `seq`, a shortened `write_attempt_id`, +`attempts_sent`, `elapsed_ms`, `remaining_confirmed_budget_ms`, and `classification`. The older +`unresolved_reason`, `deadline_source`, and `stop_cause` keys no longer exist; `classification` +carries what they used to say between them (see [debugging](/antalya/cas/operations/debugging#trace-renewal-remount) +for the full value list). Ordinary first-attempt success produces no row. Every `mount_remount` attempt produces one final row with outcome `ok` or `failed` and details `attempt_no`, `step`, `server_root_id`, optional `writer_epoch`, and optional `error`. @@ -121,6 +124,12 @@ changed shard needing a fold and no graduation was due — a cheap round, not a `Finish` row is worth a steady watch: it is fold clamps surfaced and survived, so a non-zero value that persists across rounds is more interesting than an isolated one. +A dashboard alert that filters on `outcome = 'Error'` alone misses `Aborted` and `Stopped` rows too +— see the [`outcome` column](/operations/system-tables/cas_gc_log#columns) for what each one means. +A round that is recurring `Aborted` rather than `Error` still deserves attention: it keeps retrying, +but the underlying transient condition (backend unavailability, a lost lease, a competing leader) +has not gone away. + Which phase dominates round duration or the `LIST` budget — reproduced from the [per-phase rows](/operations/system-tables/cas_gc_log#per-phase-rows) reference: diff --git a/docs/en/antalya/cas/operations/troubleshooting.md b/docs/en/antalya/cas/operations/troubleshooting.md index be541add39f8..1113ec1855dd 100644 --- a/docs/en/antalya/cas/operations/troubleshooting.md +++ b/docs/en/antalya/cas/operations/troubleshooting.md @@ -16,16 +16,16 @@ tools. | Symptom | Diagnosis | Action | |---|---|---| -| A server keeps losing its mount lease and self-remounting | Check `system.cas_mounts` for the server's own `state`/`expires_at`, then correlate `watermark_renew` and `mount_remount` in `system.cas_log`; losing the lease trips a local fence and latches a remount generation | Read `classification`, `deadline_source`, and `stop_cause` before changing anything. Look for object-store latency consuming the confirmed lease or BOOTTIME advancement; see [the decision flow](#mount-renewal-remount-flow) and [the mount lease](/antalya/cas/architecture/mounts-and-leases#mount-lease) | -| Writes slow down or stall under load, with no exception reaching the client | S3 `SlowDown`/`ServiceUnavailable`/`RequestTimeout`/`InternalError` (5xx) responses are not on `CasRequestController`'s definite-failure whitelist (only malformed-request, entity-too-large, and access-denied are), so they classify as `Unresolved` and are retried automatically. Confirm with `sum(ProfileEvents['CASConditionalWriteUnresolved'])` rising alongside `sum(ProfileEvents['CASConditionalWriteAttempts'])` over `system.query_log` for the affected window (or `ProfileEvent_CASConditionalWriteUnresolved` in `system.metric_log` for a cumulative view across queries), and check `system.blob_storage_log` for `disk_name = ''` rows with a nonzero `error_code` around the same window | Nothing to configure per-request: the controller retries the same `(key, bytes)` with capped-exponential backoff (200ms initial, capped at 5s) for up to 16 attempts inside a 90-second operation deadline, and the mount-lease renewer keeps extending the fence across the disruption — this is the "blips, throttling, partial outages" case the write path is built to survive. Confirm the mount lease itself is still renewing (`system.cas_mounts.expires_at` moving forward, `last_success_age_seconds` not climbing) — if it is, this is expected and self-resolving. If `SlowDown` responses are sustained rather than transient, check the bucket's request-rate limits against the pool's actual PUT/GET rate (see [bucket requirements](/antalya/cas/bucket-requirements)) and consider lowering `cas_blob_upload_pool_size` to reduce concurrent upload traffic; a write only surfaces a client-visible `NETWORK_ERROR` if the 90-second deadline is exhausted before the store recovers, and that error is retried by the ordinary merge/insert backoff, not silently dropped | +| A server keeps losing its mount lease and self-remounting | Check `system.cas_mounts` for the server's own `state`/`expires_at`, then correlate `watermark_renew` and `mount_remount` in `system.cas_log`; losing the lease trips a local fence and latches a remount generation | Read the failed renewal's `classification` before changing anything — it alone now says why (see [the decision flow](#mount-renewal-remount-flow)). Look for object-store latency consuming the confirmed lease or BOOTTIME advancement; see [the mount lease](/antalya/cas/architecture/mounts-and-leases#mount-lease) | +| Writes slow down or stall under load, with no exception reaching the client | S3 `SlowDown`/`ServiceUnavailable`/`RequestTimeout`/`InternalError` (5xx) responses are not on the request engine's `isDefinitelyRefusedWrite` definite-failure list (only malformed-request, entity-too-large, and access-denied that no credential refresh can fix are), so they classify as ambiguous and are retried automatically. Confirm with `sum(ProfileEvents['CASConditionalWriteUnresolved'])` rising alongside `sum(ProfileEvents['CASConditionalWriteAttempts'])` over `system.query_log` for the affected window (or `ProfileEvent_CASConditionalWriteUnresolved` in `system.metric_log` for a cumulative view across queries), and check `system.blob_storage_log` for `disk_name = ''` rows with a nonzero `error_code` around the same window | Nothing to configure per-request: the request engine retries the same `(key, bytes)` with capped-exponential backoff (200ms initial, capped at 5s, full jitter) until the 90-second operation deadline — there is no separate attempts ceiling, only the deadline — and the mount-lease renewer keeps extending the fence across the disruption — this is the "blips, throttling, partial outages" case the write path is built to survive. Confirm the mount lease itself is still renewing (`system.cas_mounts.expires_at` moving forward, `last_success_age_seconds` not climbing) — if it is, this is expected and self-resolving. If `SlowDown` responses are sustained rather than transient, check the bucket's request-rate limits against the pool's actual PUT/GET rate (see [bucket requirements](/antalya/cas/bucket-requirements)) and consider lowering `cas_blob_upload_pool_size` to reduce concurrent upload traffic; a write only surfaces a client-visible `NETWORK_ERROR` if the 90-second deadline is exhausted before the store recovers, and that error is retried by the ordinary merge/insert backoff, not silently dropped | | `GC` never seems to reclaim space after tables are dropped | `SELECT * FROM system.cas_gc_log WHERE event_type='Finish' ORDER BY event_time DESC LIMIT 5` — check `outcome`; also `SELECT is_leader FROM system.cas_mounts` on this node | If `outcome != 'Success'`/`'Deferred'`, see [reading GC health](/antalya/cas/operations/monitoring#gc-health); if this node is not the leader (`is_leader = 0`), it never reclaims for this disk — check the peer holding leadership. Reclamation also needs at least two full rounds past condemnation by design (the grace period is rounds, not acks) — a single manual `SYSTEM CAS GC RUN` will not finish it | | A dangling-access exception or `CORRUPTED_DATA` on read | Run `clickhouse-disks cas-fsck --detail` and check `dangling` specifically — it is the one class that means data loss, distinct from `unreachable`/`awaiting-gc`, which are just waiting for graduation | A nonzero `dangling` count is a real incident: collect the `--detail` output (see [what to collect before filing a bug](/antalya/cas/operations/debugging#filing-a-bug)) before taking any destructive action | | `SYSTEM CAS FSCK` or `clickhouse-disks cas-fsck` times out on a large pool | The scan is bounded by `--timeout` (default 600s / the `SYSTEM` form has no override); a large `roots/` prefix can make the scan slow | Retry with `--partial` to see the counts accumulated so far instead of aborting empty-handed, or `--namespace ` to scope the scan to a subset of namespaces | | `SYSTEM CAS DROP POOL MEMBER` returns a non-empty `warnings` column | A per-object drain step could not confirm emptiness; the mount slot is left terminated but not fully drained, as a resume anchor | Rerun the same command — it is resumable and skips namespaces already marked removed, reporting them under `namespaces_already_removed` | | Writes or `ALTER`s on a `CAS` disk fail with a `READONLY`-class error | The disk's metadata storage rejects every mutating entry point; this is deliberate for a disk opened with `true`, used by every offline `clickhouse-disks` tool | Confirm whether the disk was intentionally configured read-only (offline inspection, `cas-fsck`, `cas-gc-dryrun`, `cas-gc-rebuild`, `cas-drop-member` all require it); a production disk serving writes must not carry `true` | | A table stays unavailable after a transient network error during startup | `AsyncLoader` has no retry/requeue path for a failed table load job: a transient S3 `NETWORK_ERROR` during `CAS` ref-table startup recovery can leave the job permanently `FAILED` | Restart the server, or issue a fresh load for the table; this is a one-shot job design, not a `CAS`-specific bug | -| A mounted pool directory was removed or renamed out of band | Renewal observes an absent, foreign, successor, or otherwise conflicting mount body and terminates the keeper with a typed fail-closed exception; the runtime closes the local write fence and requests remount rather than adopting the body | Never remove or rename a live pool's storage path. To retire a member permanently use [`SYSTEM CAS DROP POOL MEMBER`](/antalya/cas/operations/migration#decommission) instead of raw filesystem operations; collect the `watermark_renew` classification and subsequent `mount_remount` step | -| Stale-looking part metadata after an out-of-band change to the pool | The part-folder view cache may be serving a retained (not re-validated) view | Set the disk-level `cas_part_folder_cache_bytes = 0` as a diagnostic kill switch to disable retention, and run `fsck`/integrity checks with `cas_part_folder_validate = always` so every read re-proves the body | +| A mounted pool directory was removed or renamed out of band | Renewal observes an absent, foreign, successor, or otherwise conflicting mount body and terminates the keeper with a typed fail-closed exception; the runtime closes the local write fence and requests remount rather than adopting the body | Never remove or rename a live pool's storage path. To retire a member permanently use [`SYSTEM CAS DROP POOL MEMBER`](/antalya/cas/operations/migration#decommission) instead of raw filesystem operations; collect the `watermark_renew` classification (`vanished` for an absent body, `conflict` for a foreign, successor, or otherwise conflicting one) and the subsequent `mount_remount` step | +| Stale-looking part metadata after an out-of-band change to the pool | The part-folder view cache may be serving a retained (not re-validated) view | Set the disk-level `cas_part_folder_cache_bytes = 0` to disable view retention and `cas_manifest_decode_cache_bytes = 0` to make every manifest read fetch the body, then run `fsck`; both are diagnostic kill switches, not steady-state settings | | A wide merge (many thousands of columns) fails with a port-exhaustion error from the network layer | Each column in a wide part can cost a separate object-store operation in one merge, and a very wide part can issue on the order of the column count in requests, exhausting local ephemeral TCP ports under load | Reduce concurrent merge parallelism on that table, or increase the host's ephemeral port range; this is a general high-fan-out-merge limit, not specific to content addressing | ## Mount renewal and remount decision flow {#mount-renewal-remount-flow} @@ -33,28 +33,30 @@ tools. Start with the `watermark_renew` timeline described in [debugging](/antalya/cas/operations/debugging#trace-renewal-remount), then follow the matching case: -1. **Recovered blip.** `retrying` is followed by `recovered` for the same shortened - `write_attempt_id`; `CASMountRenewalRecovered` rises while `CASMountLeaseLost` and all remount - counters stay flat. No intervention is needed unless the rate is sustained; investigate backend - throttling/latency before the blips consume the lease budget. +1. **Recovered blip.** A single `outcome = 'recovered'` row (there is no separate `retrying` row to + look for) with `classification` of `committed_by_read` or `committed_after_retry`; + `CASMountRenewalRecovered` rises while `CASMountLeaseLost` and all remount counters stay flat. No + intervention is needed unless the rate is sustained; investigate backend throttling/latency before + the blips consume the lease budget. 2. **External lease-safety exhaustion.** The failed row has - `classification = 'external_lease_deadline'` and - `deadline_source = 'external_lease_safety'`; `CASMountRenewalDeadlineExceeded` and + `classification = 'external_lease_deadline'`; `CASMountRenewalDeadlineExceeded` and `CASMountLeaseLost` rise. The runtime correctly refused to manufacture authority beyond the last confirmed lease. Check object-store latency and BOOTTIME/suspend history, then follow the ensuing - remount. -3. **Cancellation.** `stop_cause = 'cancelled'` after a sent request is terminal and suppresses a - clean farewell because the request may still land. Cancellation before any request is - `NotAttempted`, remains `Active`, and emits no failed aggregate row; during graceful shutdown that - is the expected clean-release path. + remount. `classification = 'request_deadline'` is the sibling case: the ninety-second request + policy exhausted first rather than the lease's own safety margin. +3. **Cancellation.** `classification = 'cancelled'` after a sent request is terminal and suppresses a + clean farewell because the request may still land. Cancellation before any request remains + `Active` and emits no failed aggregate row; during graceful shutdown that is the expected + clean-release path. 4. **Confirmed conflict.** `classification = 'conflict'` means exact resolution found another body; inspect `server_root_id`, `writer_epoch`, `seq`, and `write_attempt_id`. Same-pair twins, GC-fenced bodies, successor epochs, and foreign holders all remain fail closed. Do not delete or rewrite the mount key by hand. -5. **Fence or lifecycle loss.** `stop_cause = 'fence_or_lifecycle_lost'` means another local loss, +5. **Fence or lifecycle loss.** `classification = 'fence_or_lifecycle_lost'` means another local loss, remount park request, or terminal lifecycle closed admission while the operation was active. A parked result reuses the already-requested recovery generation and must not double-count - `CASMountLeaseLost`. + `CASMountLeaseLost`. `classification = 'unresolved'` is a related but distinct case: every attempt + stayed ambiguous and the operation gave up without ever settling one way or the other. 6. **Whole-chain remount failure.** Read the following `mount_remount` row. Its `attempt_no`, `step`, and optional `error` identify the failed owner/catalog/epoch/claim/install/quiescence/fence step. The current protocol retries the whole chain with bounded backoff; it does not preserve per-step diff --git a/docs/en/antalya/cas/quick-start.md b/docs/en/antalya/cas/quick-start.md index a54d9345fbb0..0491f6972e8b 100644 --- a/docs/en/antalya/cas/quick-start.md +++ b/docs/en/antalya/cas/quick-start.md @@ -56,7 +56,9 @@ literal string, as above, is enough; on a replicated cluster where every replica disk's `endpoint` already uses, giving each replica a distinct subtree from one template. **S3 endpoint variant.** Swap `object_storage_type` to `s3` and add the usual object-storage -connection keys; nothing else in this config changes: +connection keys, plus `http_keep_alive_timeout` and `http_keep_alive_max_requests` — see +[recommended keep-alive settings](/antalya/cas/configuration#recommended-keep-alive-settings) for +why; nothing else in this config changes: ```xml @@ -67,6 +69,8 @@ connection keys; nothing else in this config changes: https://bucket.s3.amazonaws.com/cas/ ... ... + 30 + 10000 ``` diff --git a/docs/en/operations/system-tables/cas_gc_log.md b/docs/en/operations/system-tables/cas_gc_log.md index 5fd04b4fb11d..e92c2a6e3bd6 100644 --- a/docs/en/operations/system-tables/cas_gc_log.md +++ b/docs/en/operations/system-tables/cas_gc_log.md @@ -38,20 +38,21 @@ specified (it is enabled by default in the shipped `config.xml`). - `gc_id` ([String](/sql-reference/data-types/string)) — The GC scheduler instance id (which mounter ran the round). - `trigger` ([Enum8](/sql-reference/data-types/enum)) — `Scheduled` (background tick) or `Manual` (`SYSTEM` command). - `round` ([UInt64](/sql-reference/data-types/int-uint)) — The GC round number (`0` on a `Start` row). -- `outcome` ([Enum8](/sql-reference/data-types/enum)) — `Unknown` (on a `Start` row), `Success` (led, folded, and completed), `NotALeader` (another replica holds the GC lease), `Deferred` (led but took the skip-unchanged fast path — no fold ran, because no changed shard reached the fold threshold and no graduation was due), or `Error` (the round threw). +- `outcome` ([Enum8](/sql-reference/data-types/enum)) — `Unknown` (on a `Start` row), `Success` (led, folded, and completed), `NotALeader` (another replica holds the GC lease), `Deferred` (led but took the skip-unchanged fast path — no fold ran, because no changed shard reached the fold threshold and no graduation was due), `Aborted` (the round threw a transient error — backend unavailability, a lost lease, a concurrent leader; the next scheduled round retries it), `Stopped` (a transient error observed after the disk began shutting down: the round was cut short so the shutdown need not wait for it, and the next start re-derives its work), or `Error` (the round threw a non-transient error — investigate). - `candidates_marked` ([UInt64](/sql-reference/data-types/int-uint)) — Objects retired (marked) this round. - `objects_deleted` ([UInt64](/sql-reference/data-types/int-uint)) — Objects physically deleted this round. - `objects_absent` ([UInt64](/sql-reference/data-types/int-uint)) — Retire candidates found already absent. - `objects_replaced` ([UInt64](/sql-reference/data-types/int-uint)) — `412`-saves (a resurrection won the race against the delete). - `objects_spared` ([UInt64](/sql-reference/data-types/int-uint)) — Candidates spared because their in-degree was greater than zero at recheck. -- `manifests_deleted` ([UInt64](/sql-reference/data-types/int-uint)) — Owner-removed manifest bodies physically deleted this round, counted separately from blob deletes. +- `manifests_deleted` ([UInt64](/sql-reference/data-types/int-uint)) — Owner-removed manifest bodies deleted or found already absent this round (a batch delete of write-once keys cannot tell the two apart), counted separately from blob deletes. - `entries_condemned` ([UInt64](/sql-reference/data-types/int-uint)) — Retired entries newly condemned this round (retired-cursor pipeline stage 1). - `entries_graduated` ([UInt64](/sql-reference/data-types/int-uint)) — Retired entries newly floor-passed and republished `delete_pending` this round (pipeline stage 2; deleted the next round). - `entries_redeleted` ([UInt64](/sql-reference/data-types/int-uint)) — Pending exact-token blob deletes executed this round (pipeline stage 3). - `fence_outs` ([UInt64](/sql-reference/data-types/int-uint)) — Expired mounts fenced out by this round's heartbeat floor. - `anomalies` ([UInt64](/sql-reference/data-types/int-uint)) — Fold clamps surfaced (and survived) this round. A steady non-zero value warrants a look at the round log details. - `duration_ms` ([UInt64](/sql-reference/data-types/int-uint)) — The round wall-clock duration (on a `Finish` row). -- `error` ([String](/sql-reference/data-types/string)) — The exception text when `outcome = 'Error'`. +- `error` ([String](/sql-reference/data-types/string)) — The exception text when `outcome = 'Aborted'`, `'Stopped'` or `'Error'`. On a `Stopped` row it names the request the shutdown refused, not the shutdown itself. +- `error_code` ([Int32](/sql-reference/data-types/int-uint)) — The exception code when `outcome = 'Aborted'`, `'Stopped'` or `'Error'`; `0` otherwise. Key monitoring on this column rather than on the `error` text. On an `Aborted`, `Stopped` or `Error` row the counters still report everything the round completed before it threw, and `round != 0` on such a row means the round's closing compare-and-swap committed and the failure hit only post-commit cleanup. - `ProfileEvents` ([Map(LowCardinality(String), UInt64)](/sql-reference/data-types/map)) — On a `Start`/`Finish` row, the per-round `ProfileEvents` delta (the `CAS*` counters and S3/disk events for this round). On a `Phase` row, **that phase's** delta, so `GROUP BY phase` over `ProfileEvents['S3ListObjects']` attributes the round's `LIST` budget to the phase that spent it. - `round_id` ([String](/sql-reference/data-types/string)) — The correlator for every row of one round attempt: its `Start`, each of its `Phase` rows, and its `Finish`. Minted per attempt, so unlike `round` it exists even for a round that never committed and for a round that never led. Group by this column to reconstruct one round. - `phase` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The GC phase this row describes; empty on `Start`/`Finish`. See [Per-phase rows](#per-phase-rows) for the phase list. diff --git a/docs/en/operations/system-tables/cas_log.md b/docs/en/operations/system-tables/cas_log.md index 17616cd9b329..d4742cda279e 100644 --- a/docs/en/operations/system-tables/cas_log.md +++ b/docs/en/operations/system-tables/cas_log.md @@ -25,13 +25,13 @@ specified (it is enabled by default in the shipped `config.xml`). - `event_date` ([Date](/sql-reference/data-types/date)) — Event date. - `event_time` ([DateTime](/sql-reference/data-types/datetime)) — Event time. - `event_time_microseconds` ([DateTime64(6)](/sql-reference/data-types/datetime64)) — Event time with microseconds precision. -- `event_type` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The CAS decision/event, e.g. `blob_put`, `blob_reuse_adopt`, `root_remove`, `indegree_zero`, `gc_retire_decision`, `gc_recheck_verdict`, `blob_delete`, `dangling_access`, `corrupt_dangle`. +- `event_type` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The CAS decision/event, e.g. `blob_put`, `blob_reuse_adopt`, `root_remove`, `indegree_zero`, `gc_retire_decision`, `gc_recheck_verdict`, `blob_delete`, `dangling_access`, `corrupt_dangle`, `watermark_renew`, `mount_remount`. - `disk_name` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — The content-addressed disk / pool the event belongs to. - `namespace` ([String](/sql-reference/data-types/string)) — `roots/` (server/table); empty if not applicable. - `ref_name` ([String](/sql-reference/data-types/string)) — Part name / ref the event concerns; empty if not applicable. - `object_kind` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — One of `none`, `blob`, `manifest`, `root`, `snapshot`. - `object_hash` ([String](/sql-reference/data-types/string)) — Content hash (lowercase hex) of the object; empty if not applicable. -- `token` ([String](/sql-reference/data-types/string)) — Incarnation token (`ETag`) involved; empty if not applicable. +- `token` ([String](/sql-reference/data-types/string)) — On events about a stored object, the incarnation involved, rendered uniformly as `:` (e.g. `etag:"a1b2c3"` on S3-compatible stores, `generation:1234` on GCS); the part-build lifecycle events reuse the column for the 128-bit build id in hex; empty if not applicable. - `round` ([UInt64](/sql-reference/data-types/int-uint)) — GC round (`0` if not applicable). - `generation` ([UInt64](/sql-reference/data-types/int-uint)) — GC snapshot generation (`0` if not applicable). - `at_version` ([UInt64](/sql-reference/data-types/int-uint)) — Manifest `shard_version` of the driving journal record (`0` if not applicable). @@ -39,7 +39,7 @@ specified (it is enabled by default in the shipped `config.xml`). - `reason` ([LowCardinality(String)](/sql-reference/data-types/lowcardinality)) — Human-readable rationale for the decision. Templated across rows, so it is `LowCardinality`. - `thread_id` ([UInt64](/sql-reference/data-types/int-uint)) — OS thread that emitted the event. - `query_id` ([String](/sql-reference/data-types/string)) — Query id for correlation with [`system.query_log`](/operations/system-tables/query_log); empty if not applicable. -- `detail` ([Map(LowCardinality(String), String)](/sql-reference/data-types/map)) — Structured event-specific facts, e.g. `condemn_round`, `superseded_token`, `code`, `site`. +- `detail` ([Map(LowCardinality(String), String)](/sql-reference/data-types/map)) — Structured event-specific facts, e.g. `condemn_round`, `superseded_token`, `code`, `site`, or — on `watermark_renew` — `attempts_sent` and `classification`; see [debugging](/antalya/cas/operations/debugging#trace-renewal-remount) for the mount-renewal detail keys. ## Example {#example} diff --git a/programs/disks/CommandCaInspect.cpp b/programs/disks/CommandCaInspect.cpp index b4a3a24dbb89..32df388f2cdb 100644 --- a/programs/disks/CommandCaInspect.cpp +++ b/programs/disks/CommandCaInspect.cpp @@ -48,11 +48,17 @@ class CommandCaInspect final : public ICommand if (!ca->isReadOnly()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: open the CA disk read-only"); - const auto got = ca->store()->backend().get(key); + /// `store()` hands back a snapshot of the pool pointer; `openRequests()`/`layout()` return + /// references into that Pool object. Keeping the shared_ptr alive for the whole operation, + /// rather than letting each `store()` call's temporary expire, is what keeps those references + /// valid and pins both calls to the SAME pool if a concurrent remount swaps it out from under `ca`. + const Cas::PoolPtr pool = ca->store(); + Cas::CasOperation op = pool->openRequests().admit(); + const auto got = op.read(key, Cas::Retry::standard()); if (!got) throw Exception(ErrorCodes::BAD_ARGUMENTS, "cas-inspect: key '{}' does not exist", key); - const Cas::Layout & layout = ca->store()->layout(); + const Cas::Layout & layout = pool->layout(); std::optional resolved_life; std::optional life_id; if (const auto parsed = layout.parseRefObjectKey(key)) @@ -61,7 +67,7 @@ class CommandCaInspect final : public ICommand life_id = *parsed_ckpt; if (life_id) { - const Cas::CasRefCatalog::Snapshot cut = Cas::CasRefCatalog::read(ca->store()->backend(), layout); + const Cas::CasRefCatalog::Snapshot cut = Cas::CasRefCatalog::read(op, layout); resolved_life = cut.life_index.resolve(*life_id); } diff --git a/src/Common/CurrentMetrics.cpp b/src/Common/CurrentMetrics.cpp index 289f08919b70..200f3c6001af 100644 --- a/src/Common/CurrentMetrics.cpp +++ b/src/Common/CurrentMetrics.cpp @@ -250,6 +250,8 @@ M(CASPartFolderCacheEntries, "Entries retained by the CA part-folder view cache") \ M(CASManifestDecodeCacheBytes, "Bytes retained by the CA manifest decode cache") \ M(CASManifestDecodeCacheEntries, "Entries retained by the CA manifest decode cache") \ + M(CASHotKeyCacheBytes, "Bytes retained by the CA hot-key lane's cache of last known objects") \ + M(CASHotKeyCacheEntries, "Entries retained by the CA hot-key lane's cache of last known objects") \ M(CASBlobUploadPoolThreads, "Number of threads in the CA blob upload thread pool.") \ M(CASBlobUploadPoolThreadsActive, "Number of threads in the CA blob upload thread pool running a task.") \ M(CASBlobUploadPoolThreadsScheduled, "Number of queued or active jobs in the CA blob upload thread pool.") \ diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index 408d299ad89d..5b2ffdf0b757 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -681,6 +681,7 @@ M(1006, INVALID_CURSOR_LOOKUP) \ M(1007, ILLEGAL_STREAM) \ M(1008, TEMPORARY_DATA_NOT_IN_CACHE) \ +<<<<<<< HEAD M(1009, S3_OBJECT_CHANGED_DURING_READ) \ M(1010, UNIQUE_KEY_DENSE_INDEX_UNREADABLE) \ M(1011, HANDLER_ALREADY_EXISTS) \ @@ -691,6 +692,21 @@ M(1016, PENDING_MUTATIONS_NOT_ALLOWED) \ M(1017, EXPORT_PARTITION_ALREADY_EXPORTED) \ M(1018, PARTITION_EXPORT_FAILED) \ +======= + M(1009, PENDING_MUTATIONS_NOT_ALLOWED) \ + /* 1010 and 1011 predate the fork's error-code range policy stated below, and are kept as-is \ + * rather than renumbered: they currently collide with upstream ClickHouse's own 1010 \ + * (UNIQUE_KEY_DENSE_INDEX_UNREADABLE) and 1011 (HANDLER_ALREADY_EXISTS). */ \ + M(1010, EXPORT_PARTITION_ALREADY_EXPORTED) \ + M(1011, PARTITION_EXPORT_FAILED) \ + /* 1012 and 1013 are intentionally skipped: they collide with upstream ClickHouse's \ + * HANDLER_DOESNT_EXIST and AMBIGUOUS_HANDLER. Fork-specific error codes live in the 1030-1099 \ + * range, chosen to sit well above upstream's maximum error code (1017 at the time this range \ + * was reserved) so upstream can keep adding codes below it without colliding with the fork's. \ + * A new fork error code goes in this range, not below 1030. CAS codes occupy 1037-1038. */ \ + M(1037, CAS_WRITE_UNATTRIBUTED) \ + M(1038, CAS_DELETE_MARKER) \ +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) /* See END */ #ifdef APPLY_FOR_EXTERNAL_ERROR_CODES @@ -707,6 +723,7 @@ namespace ErrorCodes APPLY_FOR_ERROR_CODES(M) #undef M +<<<<<<< HEAD constexpr ErrorCode END = 1018; #if !defined(CLICKHOUSE_PARSER_MINIMAL_BUILD) @@ -715,6 +732,9 @@ namespace ErrorCodes * around 150 KB of data for the whole table, which the server wants for `system.errors` but a * standalone build of the parser has nothing to do with: the names above are all it needs. */ +======= + constexpr ErrorCode END = 1038; +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) ErrorPairHolder values[END + 1]{}; #endif diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 02b669dced58..569b13e59d46 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -371,7 +371,9 @@ static struct InitFiu REGULAR(smt_force_takeover_predicate_true) \ REGULAR(smt_takeover_fake_hardware_error_after_set) \ REGULAR(cas_relink_receiver_force_mechanism_failure) \ - PAUSEABLE_ONCE(cas_relink_receiver_pause_before_confirm) + PAUSEABLE_ONCE(cas_relink_receiver_pause_before_confirm) \ + REGULAR(cas_relink_sender_omit_pool_cookie) \ + REGULAR(cas_relink_receiver_drop_forced_disk) namespace FailPoints { diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index da1faded7ff0..71bd02e91797 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -835,6 +835,10 @@ The server successfully detected this situation and will download merged part fr M(CASRefBatchedMutations, "Number of CAS ref mutations committed through the per-namespace batching queue. Growth indicates reference-write activity.", ValueType::Number) \ M(CASRefBatchScopeCuts, "Number of CAS ref batches cut short by scope limits. Growing values indicate smaller batches and more write overhead.", ValueType::Number) \ M(CASRefQueueWaitMicroseconds, "Total time CAS ref writers spent queued, in microseconds. A rising value indicates ref-write contention or backend latency.", ValueType::Microseconds) \ + M(CASHotKeyQueueWaitMicroseconds, "Total time CAS writers of a shared key spent queued in the hot-key lane before holding it or leaving, in microseconds. A rising value with a flat write rate means the holder is slow, not the store.", ValueType::Microseconds) \ + M(CASHotKeyCacheStarts, "Number of hot-key lane holds that started from the pool's last known object instead of a read.", ValueType::Number) \ + M(CASHotKeyReadStarts, "Number of hot-key lane holds that started from a read of the key.", ValueType::Number) \ + M(CASHotKeyCacheVerdictsReread, "Number of verdicts (a refusal or a decline) a hot-key lane decide rendered on a cached object and that were re-rendered on a fresh read instead of delivered.", ValueType::Number) \ M(CASRefRecoveryRestarts, "Number of CAS ref-table recovery retries after a snapshot or log vanished during reading. A non-zero value indicates concurrent cleanup or backend inconsistency.", ValueType::Number) \ M(CASRefRecoveryRetries, "Number of CAS ref-table recovery attempts retried after a transient object-store error before the table's load fails. A non-zero value indicates transient object-store disruption during table startup.", ValueType::Number) \ M(CASRefAppendWedged, "Number of CAS ref-log append lanes that exhausted retries after an uncertain PUT. A non-zero value indicates ref-log progress may be stalled.", ValueType::Number) \ @@ -843,6 +847,11 @@ The server successfully detected this situation and will download merged part fr M(CASRefAppendDefiniteFailure, "Number of CAS ref-log appends rejected with certainty. A non-zero value indicates invalid requests or backend rejection requiring investigation.", ValueType::Number) \ M(CASRefAppendSealRejected, "Number of CAS ref-log transactions conclusively rejected by a successor's epoch seal occupying the id they derived. This is the protocol working -- the writer was deposed and its operation was never acknowledged -- but a lane that keeps counting here is a writer that has lost its mount and does not yet know it.", ValueType::Number) \ M(CASRefAppendOccupantUnreadable, "Number of CAS ref-log appends that met a DIFFERENT object at the id they derived and could not read it to tell a successor's epoch seal from a breach of mount write-exclusivity. The decision is deferred to the next attempt, which re-derives the same id. Sustained growth means a real breach may be going unreported: the loud interference path is only reached once the occupant can be read.", ValueType::Number) \ + M(CASRelinkConfirmRefusedRefMutationInFlight, "Number of CAS fetch-by-relink confirms this server answered Unknown because a queued or in-flight ref-lane mutation names the asked-about ref or the whole namespace. Expected under write load; the receiver retries the fetch.", ValueType::Number) \ + M(CASRelinkConfirmRefusedLaneWedged, "Number of CAS fetch-by-relink confirms answered Unknown because the namespace's ref lane holds an unresolved append (a wedge). Lasts until the next flush or a remount resolves it.", ValueType::Number) \ + M(CASRelinkConfirmRefusedLaneBroken, "Number of CAS fetch-by-relink confirms answered Unknown because the namespace's ref lane is in NeedsRecovery, Closed or Faulted state, or is Writing with nothing carved. A growing value outside induced faults is a lane defect, not load.", ValueType::Number) \ + M(CASRelinkConfirmRefusedStateLockBusy, "Number of CAS fetch-by-relink confirms answered Unknown because the ref table's state lock was held. Under write load the usual holder is the table's own append leader, arming or installing a chunk; otherwise a recovery, a listing or a snapshot publish. The confirm never waits for it.", ValueType::Number) \ + M(CASRelinkConfirmRefusedMountCannotSpeak, "Number of CAS fetch-by-relink confirms answered Unknown because this mount cannot speak for the namespace: its ref table is unrecovered or mid-recovery, its catalog life was invalidated, its runtime was superseded by a remount, or its mount fence is no longer held. Neither a lane defect nor write load. A growing value means this writer is losing, or has already lost, its claim to the namespace.", ValueType::Number) \ M(CASRefNeedsRecovery, "Number of CAS ref append lanes moved to `NeedsRecovery` because a known-durable transaction could not be installed. Such a lane refuses writes, snapshots, and confirmation until durable replay completes.", ValueType::Number) \ M(CASRefSweepDeferred, "Number of stale-precommit sweeps deferred after a read-only failure. A non-zero value indicates cleanup is waiting for a later trigger.", ValueType::Number) \ M(CASRefSweepRearmed, "Number of failed or partial stale-precommit sweeps scheduled for retry. Growing values indicate persistent cleanup or backend errors.", ValueType::Number) \ @@ -853,7 +862,7 @@ The server successfully detected this situation and will download merged part fr M(CASRefLogBodyGets, "Number of CAS ref-log bodies read and decoded during GC. Growth indicates more reference history to process.", ValueType::Number) \ M(CASRefManifestBodyFoldGets, "Number of manifest bodies read while GC follows reference edges. High values indicate cache misses or many referenced manifests.", ValueType::Number) \ M(CASRefEmittedEdges, "Number of reachability edges emitted while GC folds CAS reference history. Growth indicates more reference relationships to process.", ValueType::Number) \ - M(CASRefCleanupObjectsDeleted, "Number of old CAS ref logs and snapshots deleted after safe coverage was confirmed. Growth indicates cleanup progress.", ValueType::Number) \ + M(CASRefCleanupObjectsDeleted, "Number of old CAS ref logs and snapshots deleted after safe coverage was confirmed. Includes keys that were already absent, since a batch delete of write-once keys cannot tell the two apart. Growth indicates cleanup progress.", ValueType::Number) \ M(CASRefSnapshotPutBytes, "Total bytes written to CAS ref-table snapshots. A high value indicates frequent or large snapshot publication.", ValueType::Bytes) \ M(CASRefSnapshotTailLogs, "Number of CAS ref-log entries compacted into published snapshots. Growth indicates snapshot maintenance work.", ValueType::Number) \ M(CASRefSnapshotPublishDispatched, "Number of background CAS ref-table snapshot publications started. High values indicate frequent threshold or read-triggered publishing.", ValueType::Number) \ @@ -904,6 +913,9 @@ The server successfully detected this situation and will download merged part fr M(CASGCGetStream, "Number of streaming CAS GC GET requests. Grows with large collection or recovery reads.", ValueType::Number) \ M(CASGCDelete, "Number of CAS GC DELETE requests. Grows with successful cleanup attempts.", ValueType::Number) \ M(CASGCList, "Number of CAS GC LIST requests. Growing values indicate more collection enumeration.", ValueType::Number) \ + M(CASGCReadAheadHit, "Number of CAS GC fold reads and HEADs answered by the fold's read-ahead. Growth means the round's small-object round trips overlapped instead of serializing.", ValueType::Number) \ + M(CASGCReadAheadMiss, "Number of CAS GC fold reads and HEADs performed inline because nothing was hinted for the key. A large value against hits means a hint set is narrower than the walk.", ValueType::Number) \ + M(CASGCReadAheadWasted, "Number of CAS GC read-ahead results fetched and never taken: a namespace held below its lookahead, or a HEAD candidate that kept an edge. Bounded by the read-ahead window per namespace.", ValueType::Number) \ M(CASServerPut, "Number of CAS server-object PUT requests. Grows with server metadata writes.", ValueType::Number) \ M(CASServerPutDeduplicated, "Number of deduplicating CAS server-object PUT requests. Growth indicates reused server objects.", ValueType::Number) \ M(CASServerOverwrite,"Number of CAS server-object overwrite requests. Growing values indicate repeated replacement writes.", ValueType::Number) \ @@ -943,6 +955,7 @@ The server successfully detected this situation and will download merged part fr M(CASMetaResurrectClean, "Number of condemned-body replacement paths that entered Clean metadata reconciliation. Counts the reason entry, not a guaranteed metadata reset.", ValueType::Number) \ M(CASGCMetaOps, "Number of per-hash metadata operations executed by CAS GC. Growing values indicate more GC candidates or metadata work.", ValueType::Number) \ M(CASGCEnumerationPages, "Number of CAS GC LIST pages fetched while enumerating the object universe. Growing values indicate a larger universe or more frequent scans.", ValueType::Number) \ + M(CASBulkDeleteRequests, "Number of CAS batch delete requests: one DeleteObjects carrying up to 1000 write-once keys (manifest bodies, ref logs, ref snapshots). The per-key class counters (CASManifestDelete, CASRootDelete) say how many keys each request carried.", ValueType::Number) \ M(CASGCRefWalkPlansBuilt, "Number of complete catalog-authoritative CAS ref walk plans constructed by ordinary GC and rebuild. A regular or rebuilding invocation that reaches the post-LIST catalog cut increments this exactly once, including a round that later defers.", ValueType::Number) \ M(CASGCUnmatchedAdoptedParentLives, "Number of adopted-parent CAS ref-life rows dropped because the post-LIST catalog cut has no matching physical life. Each occurrence is inert for planning and suppression and is logged with its exact physical life id; a persistent nonzero rate indicates old generation state is outliving catalog removal.", ValueType::Number) \ M(CASGCStuckRemovals, "Number of adopted CAS GC rounds that observed a Removing namespace at or beyond the diagnostic age threshold without terminal cleanup evidence. Incremented and warned every such round; diagnostic only, with no effect on folding, suppression, appends, or deletion.", ValueType::Number) \ @@ -962,6 +975,15 @@ The server successfully detected this situation and will download merged part fr M(CASConditionalWriteDefiniteFailure, "Number of CAS conditional writes rejected with certainty before applying. A non-zero value indicates invalid requests, oversized entities, or access denial.", ValueType::Number) \ M(CASConditionalWriteUnresolved, "Number of CAS conditional writes with an unknown outcome after conflict, timeout, connection loss, or server error. A non-zero value indicates backend instability or state requiring resolution.", ValueType::Number) \ M(CASConditionalWriteFenceLostPostWrite, "Number of CAS writes that succeeded but lost the final mount-fence check. A non-zero value indicates late responses after the mount lifecycle changed.", ValueType::Number) \ + M(CASRequestAttempt, "Number of physical requests the CAS request contract started. Each one was admitted by the mount fence and reserved against the call's deadline before it was sent.", ValueType::Number) \ + M(CASRequestReissue, "Number of CAS requests re-sent: after a jittered backoff for an ordinary failure, after a flat pause for a connect-failure hint, or at once with no pause at all for a first-attempt fuse. Growth means the object store is throttling, failing, or contended.", ValueType::Number) \ + M(CASRequestConflictPause, "Number of clean lost races the CAS request contract repaid after a flat jitter instead of a growing backoff: the resolve read had settled the conflict and no transport fault preceded it.", ValueType::Number) \ + M(CASRequestResolveRead, "Number of requests the CAS request contract made to settle a refused precondition or an ambiguous write: a body read, or a HEAD where the caller needs only presence. A connect-hinted attempt reissues without one.", ValueType::Number) \ + M(CASRequestGaveUp, "Number of CAS writes that ended without a proven outcome, at a deadline, on a lost mount fence, or unresolved. A non-zero value means callers are being asked to retry later.", ValueType::Number) \ + M(CASRequestRefused, "Number of CAS writes the store itself refused, proving they never applied: a malformed request, an entity too large, or an access or credential denial that no credential refresh was performed for, either because the disk has no refresh mechanism or because this write had already spent its one refresh.", ValueType::Number) \ + M(CASRequestFenceLostPostWrite, "Number of CAS writes that were proven durable but lost the mount fence before the call could claim them. A non-zero value indicates late responses after the mount lifecycle changed.", ValueType::Number) \ + M(CASRequestConnectFailureHint, "Number of CAS write attempts whose transport error named a failed connection (no free local port, refused or unreachable peer, connect timeout). Under a reissuing policy the engine reissues them after a flat pause without a settle read, when the deadline and the fence admit it. Growth means the server cannot open connections to the object store.", ValueType::Number) \ + M(CASRequestFirstAttemptFuse, "Number of CAS control requests whose first HTTP attempt matched the adaptive first-attempt timeout; the engine reissues them at once as attempt 2 when the policy and the gates permit. Growth means the object store does not answer a fresh connection within the first-attempt timeout.", ValueType::Number) \ M(CASMountRenewalAttempts, "Number of physical conditional renewal PUTs sent for CAS mount leases. This counts transport attempts, not logical renewals.", ValueType::Number) \ M(CASMountRenewalRetries, "Number of physical conditional renewal PUTs sent after the first attempt of one logical CAS mount-lease renewal.", ValueType::Number) \ M(CASMountRenewalResolved, "Number of CAS mount-lease renewals whose committed outcome was proved by an exact resolving GET.", ValueType::Number) \ @@ -981,7 +1003,6 @@ The server successfully detected this situation and will download merged part fr M(CASRefRecoveryStragglerAdopted, "Number of straggler ref-log transactions a recovery compare-and-swap walk met at the slot it tried to seal and adopted, re-sealing at the new T+1. Non-zero means writes from a dying epoch were still materializing when recovery ran.", ValueType::Number) \ M(CASRefRecoveryCancelled, "Number of CAS ref-table recovery attempts abandoned because a self-remount requested cancellation before re-arming the mount fence. Non-zero means remounts are overlapping recoveries; nothing is written or installed on this path.", ValueType::Number) \ M(CASRefRecoveryStreamHole, "Number of times CAS ref-table recovery found a 404 BELOW a durable same-epoch witness -- a hole in a stream INV-1 makes dense. Restarted while the restart budget lasts (a racing cleanup is the innocent explanation), then reported as corruption. Any sustained non-zero value is data loss, not noise.", ValueType::Number) \ - M(CASPartFolderValidateSkipped, "Number of CAS part-folder validation HEADs skipped by policy or a fresh retained view. High values reduce reads but can delay detecting external changes.", ValueType::Number) \ M(CASBlobAdoptTrusted, "Number of CAS blob adoptions trusted through a durable manifest edge without per-file probes. Growth indicates manifest-based relinking.", ValueType::Number) \ M(S3GetObjectTagging, "Number of S3 API GetObjectTagging calls.", ValueType::Number) \ M(S3HeadObjectMicroseconds, "Time of S3 API HeadObject execution.", ValueType::Microseconds) \ diff --git a/src/Common/setThreadName.h b/src/Common/setThreadName.h index e3380d5eaf7a..09396b073877 100644 --- a/src/Common/setThreadName.h +++ b/src/Common/setThreadName.h @@ -39,7 +39,7 @@ namespace DB M(CAS_ANOMALY_DIAG, "CasAnomalyDiag") \ M(CAS_GC_HEARTBEAT, "CasGcHeartbeat") \ M(CAS_GC_SCHEDULER, "CasGcSched") \ - M(CAS_LEASE_KEEPER, "CasLeaseKeeper") \ + M(CAS_LEASE_RENEWER, "CasLeaseRenewer") \ M(CAS_REF_SNAPSHOT_PUBLISH, "CasRefSnapPub") \ M(CAS_REMOUNT, "CasRemount") \ M(CGROUP_MEMORY_OBSERVER, "CgrpMemUsgObsr") \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h index e40521134377..96911defc15f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h @@ -1,11 +1,18 @@ #pragma once +#include +#include #include +#include #include #include +#include +#include #include #include +#include +#include +#include #include -#include #include #include #include @@ -16,119 +23,14 @@ namespace DB::Cas { -/// User metadata carried alongside an object (S3 x-amz-meta-*). The CA store uses exactly one entry, -/// "cas_owner" = "::" — the owner triple the GC watermark reads. -using ObjectMeta = std::map; - -/// A byte window requested from an object. An absent length means that the window extends to EOF. -/// Backends use the same semantics for materialized and forward-only reads: the offset is exact, -/// while a backend may expose an advisory end when its underlying read buffer cannot enforce one. -struct Range -{ - uint64_t offset = 0; - std::optional length; /// nullopt => to the end - bool whole() const { return offset == 0 && !length; } -}; - -/// Materialized object bytes together with the incarnation and user metadata observed by the read. -/// The token identifies the exact object version whose bytes are in `bytes`; callers may use it to -/// validate a subsequent token-conditional mutation. -struct GetResult -{ - String bytes; - Token token; /// token of the incarnation the bytes came from - ObjectMeta attributes; -}; - -/// A forward-only read of a WRITE-ONCE object (runs, seals): nothing is materialized by the seam. -/// MUTABLE objects (root shards, gc/state, mounts) MUST keep using `get` — their bytes may change -/// under an open stream. `token` identifies the incarnation the stream reads, same as `get`. -struct GetStreamResult -{ - std::unique_ptr stream; - Token token; -}; - -/// Metadata returned by `Backend::head`. For an absent key, `exists` is false and the other fields -/// retain their defaults; for a present key, `size`, `token`, and `attributes` describe one current -/// incarnation as observed by the backend. -struct HeadResult -{ - bool exists = false; - uint64_t size = 0; - Token token; - ObjectMeta attributes; -}; - -/// Outcome of a write-once create or a token-conditional overwrite. A precondition failure means -/// that the backend preserved the existing object; it is an expected result, not an exception. -enum class PutOutcome : uint8_t -{ - Done, /// object written; the returned PutResult.token is the new incarnation's token - PreconditionFailed, /// If-None-Match hit an existing key / If-Match mismatched — nothing changed -}; - -/// Outcome of a compare-and-set write. `Conflict` means that the expected token (or expected -/// absence) did not match and that the backend left the object unchanged. -enum class CasOutcome : uint8_t -{ - Committed, - Conflict, /// expected token (or absence) did not match — nothing changed -}; - -/// Result of a backend write: the outcome plus the resulting object token (previously a `Token * out_token` -/// out-parameter). `token` is set ONLY when the write actually landed an incarnation (a `Done`/`Committed` -/// outcome); on `PreconditionFailed`/`Conflict` nothing was written and `token` is left default-constructed, -/// exactly mirroring the old contract where callers only read `*out_token` on success. -template -struct WriteResultT -{ - Outcome outcome; - Token token; -}; - -using PutResult = WriteResultT; -using CasResult = WriteResultT; - -/// Result of deleting one exact incarnation. `TokenMismatch` and `NotFound` are deliberately -/// distinct: the former proves that another incarnation is now current, while the latter means -/// there is no object to remove. `created_delete_marker` exposes a storage-versioning behavior -/// that is incompatible with current-object reclamation. -struct DeleteOutcome -{ - enum class Kind : uint8_t { Deleted, TokenMismatch, NotFound } kind = Kind::NotFound; - /// TRUE if the backend reported a delete marker was created because versioning is enabled. The - /// capability probe rejects this for the current-object storage model: exact deletion must reclaim - /// the current object rather than archive a noncurrent version. - bool created_delete_marker = false; -}; - -/// A key returned by `Backend::list`. The `token` field is populated ONLY when the backend -/// returns TRUE from `supportsListTokens` — it identifies the key's current incarnation, matching -/// what `head` would return for the same key at that instant. Callers that do not need the token -/// (e.g. GC fence sweep, orphan sweep) ignore the field; GC discover uses it to skip unchanged -/// root shards. -struct ListedKey -{ - String key; - uint64_t size = 0; - std::optional token; /// present iff supportsListTokens() == true -}; -/// One page returned by `Backend::list`. `keys` contains only the requested prefix and the cursor -/// resumes strictly after the last returned key; an empty cursor marks the end of the enumeration. -struct ListPage -{ - std::vector keys; - String next_cursor; /// Last returned key; empty => no more pages. -}; - -/// Typed erasure evidence for one key or one prefix. `head`/ -/// `get` deliberately flatten every kind of miss (a clean absence, a missing bucket/container, a -/// permission failure, a transport fault) into one "not found" result, which is exactly right for -/// their callers (a plain read) but wrong for lifecycle recovery, which must never treat a -/// transport/permission failure as proof that data is gone. `ProbeOutcome` keeps the four cases -/// distinct: only a backend's OWN authoritative "not found" evidence earns `KeyAbsent` — a timeout, -/// a 5xx, or an unclassifiable error is ALWAYS `Indeterminate`, never promoted to absence. +/// Typed erasure evidence for one key or one prefix. Ordinary `head`/`read` answer only two things +/// for a plain caller: the object, or nullopt for the store's own authoritative "key not found" (see +/// `isObjectNotFound`) -- every other fault (a missing bucket/container, a permission failure, a +/// transport fault) propagates as an exception instead of being flattened into absence. That +/// present/absent/throw shape is exactly right for a plain read, but wrong for lifecycle recovery, +/// which must never treat a fault it cannot classify as proof that data is gone. `ProbeOutcome` keeps +/// the four cases distinct: only a backend's OWN authoritative "not found" evidence earns `KeyAbsent` +/// -- a timeout, a 5xx, or an unclassifiable error is ALWAYS `Indeterminate`, never promoted to absence. enum class ProbeOutcome : uint8_t { Present, /// the key (or, for a prefix probe, at least one object under it) exists @@ -215,110 +117,156 @@ inline BlobPayloadCopyResult copyBlobPayloadBounded(ReadBuffer & from, WriteBuff } -/// Token-aware storage seam used by the content-addressed pool. TOKEN SEMANTICS ARE THE CONTRACT: -/// - every present key has exactly one current incarnation identified by an opaque Token; -/// - putOverwrite/casPut succeed only against the expected current token (or expected absence); -/// - deleteExact removes ONLY the incarnation whose token matches — wrong token MUST be a -/// TokenMismatch with the object untouched (backends that silently ignore the condition are -/// rejected by `Cas::Probe`); -/// - conditional PUTs are protocol hygiene; casPut and deleteExact are SAFETY-critical. +/// Etag-aware storage seam used by the content-addressed pool. INCARNATION SEMANTICS ARE THE +/// CONTRACT: +/// - every present key has exactly one current incarnation identified by an opaque backend value; +/// - `write` with an `expected_value` succeeds only against that exact current value (or expected +/// absence); +/// - `remove` removes ONLY the incarnation whose value matches — a mismatch MUST report +/// `RawRemoval::Mismatch` with the object untouched (backends that silently ignore the condition +/// are rejected by `Cas::Probe`). /// -/// TOKEN ⟹ CONTENT PRECONDITION (read-path caches depend on this): a token must uniquely identify -/// the byte-content of the incarnation it labels — i.e. `head(k).token == prior get(k).token` MUST -/// imply the bytes are unchanged. The protocol's SAFETY only needs the contrapositive (changed -/// bytes ⟹ a new token, so a stale CAS/delete is rejected), but `Cas::Pool`'s read-path decode -/// cache (`readShardDecoded`) skips a re-`get`+decode on a token match, so a backend whose token -/// could REPEAT across different content would make it serve stale manifests (wrong results). Holds -/// for every backend in use: S3 ETag is content-derived; the emulated/in-memory backends mint a -/// strictly-monotonic sequence that is never reused. A backend with a weak/recycled token must NOT -/// be used as a Cas pool. The capability probe currently verifies conditional-operation behavior but -/// does not test token non-reuse across different contents, so this invariant remains a requirement -/// of every backend implementation. +/// VALUE ⟹ CONTENT PRECONDITION (read-path caches depend on this): a value must uniquely identify the +/// byte-content of the incarnation it labels — i.e. `head(k)`'s value equalling a prior `read(k)`'s +/// value MUST imply the bytes are unchanged. The protocol's SAFETY only needs the contrapositive +/// (changed bytes ⟹ a new value, so a stale conditional write/delete is rejected), but `Cas::Pool`'s +/// read-path decode cache (`readShardDecoded`) skips a re-`read`+decode on a value match, so a backend +/// whose value could REPEAT across different content would make it serve stale manifests (wrong +/// results). Holds for every backend in use: S3 ETag is content-derived; the emulated/in-memory +/// backends mint a strictly-monotonic sequence that is never reused. A backend with a weak/recycled +/// value must NOT be used as a Cas pool. The capability probe currently verifies conditional-operation +/// behavior but does not test value non-reuse across different contents, so this invariant remains a +/// requirement of every backend implementation. /// /// Most ops take/return whole `String` bodies — sufficient for manifests, trees, and probe/GC -/// objects. Large content blobs use the transport-only `publishBlob` seam; reads stay String-based -/// because blob payload reads go through the wiring's read stack, not this seam. +/// objects. Large content blobs use the transport-only `publish` seam; reads stay String-based because +/// blob payload reads go through the wiring's read stack, not this seam. class Backend { public: + Backend(); virtual ~Backend() = default; - /// Reads the selected bytes and their token, or returns nullopt when the key is absent. For a - /// mutable object, callers must use this materialized form so the body is fixed before parsing. - virtual std::optional get(const String & key, Range range) = 0; - std::optional get(const String & key) { return get(key, {}); } - - /// Forward-only stream over the object's `range` (default: whole object) for WRITE-ONCE objects - /// (runs, seals). The returned `stream` yields exactly the window's bytes and nothing is - /// materialized whole by the seam — the caller reads at its own pace. MUTABLE objects (root - /// shards, gc/state, mounts) MUST keep using `get`: their bytes can change under an open stream. - /// CAVEAT: the window END is advisory on storages where `setReadUntilPosition` is a hint - /// (LocalObjectStorage) — the stream may yield bytes past the window; consumers MUST bound their - /// own consumption (RunFileReader bounds to its data_end). The window START is always exact. - virtual std::optional getStream(const String & key, Range range) = 0; - std::optional getStream(const String & key) { return getStream(key, {}); } - - /// Returns the current incarnation's existence, size, token, and metadata without reading its - /// body. The result describes one point-in-time observation; a later operation must use the - /// returned token when it needs to protect against replacement. - virtual HeadResult head(const String & key) = 0; - - /// Creates `key` only when it is absent. `PreconditionFailed` leaves the existing object intact; - /// storage failures are reported as exceptions. On success, the returned token identifies the - /// newly created incarnation. - virtual PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) = 0; - PutResult putIfAbsent(const String & key, const String & bytes) { return putIfAbsent(key, bytes, {}); } - - /// Unconditionally publishes one complete blob body. The caller owns every lifecycle decision; - /// this method only executes the selected streaming or native-copy transport and returns after - /// the complete destination becomes visible. - virtual void publishBlob(const BlobPublishRequest & request) = 0; - - /// Replaces the current object only when its token equals `expected`. A mismatch leaves the - /// object unchanged and returns `PreconditionFailed`; the returned token is meaningful only on - /// `Done`. - virtual PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, - const ObjectMeta & meta) = 0; - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected) - { - return putOverwrite(key, bytes, expected, {}); - } - /// expected == nullopt => create-if-absent CAS (the first write of a root manifest). - /// A non-null expected token conditionally replaces that exact current incarnation. Conflicts - /// leave the object unchanged and are returned as an outcome rather than an exception. - virtual CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) = 0; - CasResult casPut(const String & key, const String & bytes, const std::optional & expected) + /// ---- The transport primitives ---- + /// + /// These are the ONLY methods that reach the store. Each takes a `TransportAccess`, which nothing + /// outside `CasRequests` can construct, so no caller can reach the store without the request + /// contract's retry, deadline and fence rules. They deal in the store's own strings: an + /// incarnation VALUE means no more here than "what the store answered", and every grammar, + /// key-binding and dialect check on it belongs to `CasRequests`. + + /// An object's bytes together with the value naming the incarnation they were read from. + struct Raw { String bytes; String value; }; + /// One object's size and incarnation value, without its body. + struct RawMeta { uint64_t size; String value; }; + /// One listed key; `value` is present only on a backend that surfaces per-key incarnations + /// through LIST -- see `supportsListTokens`. + struct RawListedKey { String key; uint64_t size; std::optional value; }; + struct RawListPage { std::vector keys; String next_cursor; }; + /// The store refused the write's precondition. Nothing was written; what the key holds now is + /// whatever a read finds. An expected outcome, never an error. + struct RawConflict {}; + /// `DeleteMarker` is a removal that did NOT reclaim: a versioned bucket archived a noncurrent + /// version instead. Distinct from `Removed` because reclaiming the storage is the point. + enum class RawRemoval : uint8_t { Removed, Gone, Mismatch, DeleteMarker }; + + /// Reads the whole object, or nullopt when the key is absent. + virtual std::optional read (const String & key, TransportAccess &) = 0; + /// One point-in-time observation of the current incarnation's size and value; nullopt when absent. + virtual std::optional head (const String & key, TransportAccess &) = 0; + /// One page of keys under `prefix`, resuming strictly after `cursor`; an empty `next_cursor` + /// marks the end of the enumeration. + virtual RawListPage list (const String & prefix, const String & cursor, size_t limit, TransportAccess &) = 0; + /// Removes ONLY the incarnation whose value equals `expected_value`; a mismatch must leave the + /// object untouched. + virtual RawRemoval remove(const String & key, const String & expected_value, TransportAccess &) = 0; + /// Creates the key (`expected_value == nullopt`) or replaces exactly the incarnation named by + /// `expected_value`. Returns the store's own value for what it just wrote, UNVALIDATED: a value + /// that fails the dialect grammar means the write may still have landed, which the caller settles + /// by reading the key back, and which this seam must never report as corruption. + virtual std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess &) = 0; + /// A forward-only read of a WRITE-ONCE object (runs, seals): nothing is materialized by the seam. + /// MUTABLE objects (root shards, gc/state, mounts) MUST use `read` -- their bytes may change + /// under an open stream. Null when the key is absent. + virtual std::unique_ptr stream(const String & key, TransportAccess &) = 0; + /// Executes one unconditional blob publication and returns once the complete destination is + /// visible. Transport only: it observes no destination state and produces no incarnation. + virtual void publish(const BlobPublishRequest & request, TransportAccess &) = 0; + + /// Removes up to 1000 WRITE-ONCE keys in one request with no per-key precondition; an absent key + /// is success. The caller proves the keys are write-once by minting them as `WriteOnceKey`, so a + /// backend never sees a mutable control key through this verb. Throws on any failure; the whole + /// chunk is reissued by the engine, which is sound because a key already deleted is absent, and + /// absence is success. + virtual void removeManyWriteOnce(const std::vector & keys, TransportAccess &) = 0; + + /// Authoritative, cache-bypassing probe of one key -- see `ProbeOutcome`. DEFAULT (used by every + /// backend without sharper raw-error evidence, e.g. `InMemoryBackend`): derived from `head`/`read` + /// alone, so it can only distinguish `Present` from `KeyAbsent`, and ANY exception from either + /// call is `Indeterminate` -- never promoted to `KeyAbsent`. A backend able to surface real + /// container/permission evidence (the S3-native and Local paths of `ObjectStorageBackend`) + /// overrides this to sharpen the classification. + virtual SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) { - return casPut(key, bytes, expected, {}); + try + { + const auto meta = head(key, access); + if (!meta) + return {ProbeOutcome::KeyAbsent, std::nullopt}; + auto raw = read(key, access); + /// Vanished between the two: still a clean, authoritative miss, not an error. + if (!raw) + return {ProbeOutcome::KeyAbsent, std::nullopt}; + return {ProbeOutcome::Present, std::move(raw->bytes)}; + } + catch (...) + { + return {ProbeOutcome::Indeterminate, std::nullopt}; + } } - /// Deletes only the current incarnation identified by `token`. A token mismatch must leave the - /// object untouched; the result distinguishes that case from an already absent key. - virtual DeleteOutcome deleteExact(const String & key, const Token & token) = 0; + /// The dialect this backend mints its incarnation values in. + virtual Dialect dialect() const = 0; + + /// Identifies this backend INSTANCE, so an incarnation observed elsewhere can be refused rather + /// than used as a precondition here. Assigned at construction and never reused in this process. + uint64_t backendId() const { return backend_id; } + + /// The budget for one HTTP attempt in milliseconds; 0 when this backend has no such notion. The + /// request contract reserves it before every attempt it starts. + virtual uint64_t attemptTimeoutMs() const { return 0; } - /// Lists one page of keys under `prefix`, starting after `cursor` and returning at most `limit` - /// entries. `ListPage::next_cursor` is the only supported continuation state. - virtual ListPage list(const String & prefix, const String & cursor, size_t limit) = 0; + /// What one attempt may cost end to end, connect included; the contract reserves THIS. A backend + /// with no connect notion answers its attempt timeout. A decorator that forwards `attemptTimeoutMs` + /// to an inner backend must forward THIS too -- the default falls back to `attemptTimeoutMs`, which + /// would silently drop the inner backend's connect contribution. + virtual uint64_t attemptEnvelopeMs() const { return attemptTimeoutMs(); } + + /// Asks the storage to re-acquire credentials. TRUE when fresh ones were installed, so the + /// caller's reissue can sign with them; FALSE when this backend has no refresh mechanism, which + /// makes an expired-credential failure terminal for the caller's policy rather than retryable. + virtual bool refreshCredentials() { return false; } /// Capability fact about the LIST seam: TRUE iff this backend can surface a per-key incarnation - /// token through `list` (i.e. each `ListedKey` carries a token that uniquely identifies the + /// value through `list` (i.e. each `RawListedKey` carries a value that uniquely identifies the /// current incarnation of that key, matching what `head` would return). /// /// Why this matters: S3 ETags are content-derived and are returned in list responses; the - /// in-memory backend mints a monotonic token it can also surface through `list`. A backend that - /// cannot surface per-key tokens through `list` MUST return FALSE. + /// in-memory backend mints a monotonic value it can also surface through `list`. A backend that + /// cannot surface per-key values through `list` MUST return FALSE. /// - /// FALSE ⇒ GC `discover` must read every root-shard body to learn the current token (fail closed). - /// TRUE ⇒ `discover` may skip an unchanged root-shard body read when the listed token equals - /// the persisted folded token, saving a GET per unchanged shard. + /// FALSE ⇒ GC `discover` must read every root-shard body to learn the current incarnation (fail + /// closed). TRUE ⇒ `discover` may skip an unchanged root-shard body read when the listed value + /// equals the persisted folded one, saving a GET per unchanged shard. virtual bool supportsListTokens() const = 0; /// Pool-level preconditions beyond per-op conditional semantics — checked by the capability - /// probe BEFORE the op battery. Default: nothing to check. The S3 backend fails closed here - /// unless a generation-dialect (GCS) bucket is VERIFIABLY free of object versioning: a - /// token-exact DELETE against a versioned bucket archives a noncurrent generation instead of - /// reclaiming storage, so GC "reclaim" would silently stop reclaiming. + /// probe BEFORE the op battery. Default: nothing to check. The S3 backend refuses here when a + /// generation-dialect (GCS) bucket is verified to have object versioning enabled: a token-exact + /// DELETE against a versioned bucket archives a noncurrent generation instead of reclaiming + /// storage, so GC "reclaim" would silently stop reclaiming. A probe that cannot answer is not + /// evidence of that, so it warns and the mount proceeds. virtual void checkPoolPreconditions() {} /// Fail-closed precondition: may this backend serve a WRITABLE mount that skips the access-check @@ -330,98 +278,24 @@ class Backend /// Fail-closed precondition: a Native-mode backend MUST have a /// working single-attempt conditional-write path before it coordinates a WRITABLE pool — silently /// running CAS conditional writes under the disk's default (~500-attempt) transparent retry policy - /// is exactly the hazard this seam forbids. Checked by the capability probe alongside - /// checkPoolPreconditions. Default: nothing to check (EmulatedSingleProcess and non-S3 backends + /// is exactly the hazard this seam forbids. Asked by `Pool::open` through + /// `backendForCapabilityPredicates()`, alongside `checkPoolPreconditions`, before the probe + /// operation is admitted. Default: nothing to check (EmulatedSingleProcess and non-S3 backends /// are not gated here — see ObjectStorageBackend's override for the one backend that is). virtual void checkConditionalWriteSingleAttemptSupport() {} - /// Authoritative, cache-bypassing probe of one key — see `ProbeOutcome`. DEFAULT (used by every - /// backend without sharper raw-error evidence, e.g. `InMemoryBackend`): derived from `head`/`get` - /// alone, so it can only distinguish `Present` from `KeyAbsent`, and ANY exception from either - /// call is `Indeterminate` — never promoted to `KeyAbsent`. A backend able to surface real - /// container/permission evidence (the S3-native and Local paths of `ObjectStorageBackend`) - /// overrides this to sharpen the classification. - virtual SentinelProbeResult probeSentinelRaw(const String & key) - { - try - { - const HeadResult hr = head(key); - if (!hr.exists) - return {ProbeOutcome::KeyAbsent, std::nullopt}; - auto g = get(key); - /// Vanished between head and get: still a clean, authoritative miss, not an error. - if (!g) - return {ProbeOutcome::KeyAbsent, std::nullopt}; - return {ProbeOutcome::Present, std::move(g->bytes)}; - } - catch (...) - { - return {ProbeOutcome::Indeterminate, std::nullopt}; - } - } - +private: + uint64_t backend_id; }; -using BackendPtr = std::shared_ptr; - -/// Walk every key under `prefix` exactly once, resuming by the backend's explicit last-returned-key -/// cursor (`ListPage::next_cursor`, empty => done). This centralizes the pagination contract shared by -/// GC, fsck, and cleanup sweeps: each returned key is delivered once, and the backend's cursor is the -/// only state used to request the next page. -/// -/// `on_page_fetched`, if set, fires exactly once per physical `backend.list` call (including an -/// empty/undersized final page) — a GC-owned caller's hook for a page-level ProfileEvents counter, -/// without misattributing a non-GC caller (e.g. fsck) that leaves it unset. Trails `page_limit` -/// (rather than sitting before it) so the two existing callers that override `page_limit` -/// (`Gc::fold`, `CasFsck.cpp`'s `listAll`) can override `page_limit` without changing callback order. -inline void forEachListedKey(Backend & backend, const String & prefix, - const std::function & cb, - size_t page_limit = 1000, - const std::function & on_page_fetched = {}) -{ - String cursor; - for (;;) - { - const ListPage page = backend.list(prefix, cursor, page_limit); - if (on_page_fetched) - on_page_fetched(); - for (const ListedKey & k : page.keys) - cb(k); - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } -} - -/// The normalized verdict of a token-exact delete, unifying the DeleteOutcome::Kind three-way that GC -/// (blob + manifest delete) and the orphan-manifest sweep each mapped by hand. -enum class DeleteClass : uint8_t { Deleted, Absent, Replaced }; - -/// Converts a backend-specific delete outcome into the three states used by cleanup callers. The -/// default branch is fail-safe: an unknown value is treated as `Replaced`, so cleanup never reports -/// an unverified deletion as successful. -inline DeleteClass classifyDeleteOutcome(const DeleteOutcome & d) +inline Backend::Backend() { - switch (d.kind) - { - case DeleteOutcome::Kind::Deleted: return DeleteClass::Deleted; - case DeleteOutcome::Kind::NotFound: return DeleteClass::Absent; - case DeleteOutcome::Kind::TokenMismatch: return DeleteClass::Replaced; - } - return DeleteClass::Replaced; /// unreachable; fail-safe toward "leave it" (never a false Deleted) + /// Per instance, monotonic, never reused: an incarnation carries the id of the backend that + /// observed it, so it can be refused anywhere else. Starts at 1, leaving 0 naming no backend. + static std::atomic next_backend_id{1}; + backend_id = next_backend_id.fetch_add(1, std::memory_order_relaxed); } -/// Returns the stable lowercase label used when reporting a normalized delete result. Unknown enum -/// values are labeled `replaced`, matching `classifyDeleteOutcome`'s fail-safe behavior. -inline std::string_view deleteClassName(DeleteClass c) -{ - switch (c) - { - case DeleteClass::Deleted: return "deleted"; - case DeleteClass::Absent: return "absent"; - case DeleteClass::Replaced: return "replaced"; - } - return "replaced"; -} +using BackendPtr = std::shared_ptr; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.cpp new file mode 100644 index 000000000000..f029480e6fe8 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.cpp @@ -0,0 +1,34 @@ +#include + +#include + +#include + +namespace DB::Cas +{ + +bool isIncarnationValue(Dialect dialect, const String & value) +{ + switch (dialect) + { + case Dialect::Generation: + { + if (value.empty() || value == "0") + return false; + if (value.size() > 1 && value.front() == '0') + return false; + return std::all_of(value.begin(), value.end(), [](char c) { return c >= '0' && c <= '9'; }); + } + case Dialect::ETag: + { + String trimmed = value; + boost::algorithm::trim(trimmed); + return !trimmed.empty() && trimmed != "*" && trimmed.find(',') == String::npos; + } + case Dialect::Emulated: + return !value.empty(); + } + return false; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.h new file mode 100644 index 000000000000..cce53b29fc21 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasEtag.h @@ -0,0 +1,90 @@ +#pragma once +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// The per-dialect grammar a response value must meet to be an incarnation. Generation: canonical +/// positive decimal (no leading zero, not "0" -- zero is the dialect's absence sentinel). ETag: +/// non-empty, not "*" after trimming whitespace, no comma (a list matches any member). Emulated: +/// non-empty. `ObjectStorageBackend::isValidTokenValue` forwards here. +bool isIncarnationValue(Dialect dialect, const String & value); + +/// One backend-observed incarnation of an object: the transport's own value naming that incarnation, +/// together with the backend and key it was observed against. "Etag" names the ROLE this value plays +/// -- whatever the backend hands back as the object's identity for a conditional write -- not the wire +/// field: it is a literal ETag on S3-compatible stores, a generation number on GCS's JSON dialect, and +/// a minted sequence value on the emulated/in-memory backends. Not default-constructible, not +/// constructible from a bare `String`, and minted ONLY by `CasRequests` -- a caller can hold one only +/// by way of an admitted read or write, so an `Etag` is always traceable to the request that produced +/// it. +class Etag +{ +public: + Etag() = delete; + bool operator==(const Etag &) const = default; + + /// "etag:" | "generation:" | "emulated:" + String render() const; + const String & key() const { return key_; } + Dialect dialect() const { return dialect_; } + uint64_t backendId() const { return backend_id_; } + +private: + friend class CasRequests; + + Etag(uint64_t backend_id, String key, Dialect dialect, String value) + : backend_id_(backend_id), key_(std::move(key)), dialect_(dialect), value_(std::move(value)) + { + } + + /// The transport's own text; CasRequests reads it to build the next conditional request. + const String & value() const { return value_; } + + uint64_t backend_id_; + String key_; + Dialect dialect_; + String value_; +}; + +inline String Etag::render() const +{ + switch (dialect_) + { + case Dialect::ETag: return "etag:" + value_; + case Dialect::Generation: return "generation:" + value_; + case Dialect::Emulated: return "emulated:" + value_; + } + UNREACHABLE(); +} + +/// An incarnation as recorded in a persisted manifest/ref: the dialect word and value, without any +/// live backend to check them against. Forward-only: a `PersistedEtag` is captured FROM a live +/// `Etag`, never the reverse -- a persisted record must never be trusted to mint a live one. +/// `matches` re-derives the same rendering the live incarnation would produce and compares it +/// textually, so the two representations can never drift apart. +struct PersistedEtag +{ + String dialect; /// "etag" | "generation" | "emulated" + String value; + + static PersistedEtag capture(const Etag & live); + bool matches(const Etag & live) const; +}; + +inline PersistedEtag PersistedEtag::capture(const Etag & live) +{ + const String rendered = live.render(); + const auto colon = rendered.find(':'); + return PersistedEtag{rendered.substr(0, colon), rendered.substr(colon + 1)}; +} + +inline bool PersistedEtag::matches(const Etag & live) const +{ + return live.render() == dialect + ":" + value; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasFence.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasFence.h new file mode 100644 index 000000000000..dd3e23da718a --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasFence.h @@ -0,0 +1,35 @@ +#pragma once +#include +#include + +namespace DB::Cas +{ + +/// The mount fence a write is admitted under, expressed as three closures so a caller (or a test) +/// can swap in a real mount's fence or a fixed, always-open one without a virtual base class. +struct Fence +{ + enum class Admit : uint8_t { Ok, LostOrRearmed, NoBudget }; + + /// The fence's current generation. + std::function generation; + /// Whether a write admitted under `admitted_generation`, expected to still be running + /// `needed_ms` from now, may proceed. + std::function admit; + /// Throw if `admitted_generation` is no longer the fence's live generation. + std::function check_or_throw; + + /// A fence that never trips: generation 0 forever, `admit` always `Ok`, `check_or_throw` never + /// throws. For backends with no mount lease to enforce (in-memory, tests). + static Fence open(); +}; + +inline Fence Fence::open() +{ + return Fence{ + []() -> uint64_t { return 0; }, + [](uint64_t, uint64_t) { return Fence::Admit::Ok; }, + [](uint64_t) {}}; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.cpp new file mode 100644 index 000000000000..0f8c9363d144 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.cpp @@ -0,0 +1,360 @@ +#include +#include + +#include +#include +#include +#include +#include + +#include +#include + +namespace ProfileEvents +{ + extern const Event CASHotKeyQueueWaitMicroseconds; + extern const Event CASHotKeyCacheStarts; + extern const Event CASHotKeyReadStarts; + extern const Event CASHotKeyCacheVerdictsReread; +} + +namespace CurrentMetrics +{ + extern const Metric CASHotKeyCacheBytes; + extern const Metric CASHotKeyCacheEntries; +} + +namespace DB::Cas +{ + +namespace +{ + +GaveUp::Source sourceFor(const Retry::Bound & bound) +{ + return bound.lease_bound ? GaveUp::Source::Lease : GaveUp::Source::Policy; +} + +/// The wait slice: how late, at most, a waiter notices its own fence or deadline. A handover wakes it +/// at once through the lane's condition variable. +constexpr auto kWaitSlice = std::chrono::milliseconds(200); + +} + +CasHotKeys::CasHotKeys(uint64_t cache_budget_bytes_) + : cache_budget_bytes(cache_budget_bytes_) + , cache(cache_budget_bytes_ == 0 + ? nullptr + /// No count cap: the weight above is never zero, so the byte budget alone already bounds + /// how many entries the cache can hold. `size_ratio` is unused by the LRU policy. + : std::make_unique("LRU", CurrentMetrics::CASHotKeyCacheBytes, CurrentMetrics::CASHotKeyCacheEntries, + cache_budget_bytes_, Cache::NO_MAX_COUNT, Cache::DEFAULT_SIZE_RATIO)) +{ +} + +CasHotKeys::~CasHotKeys() +{ + /// A lane surviving the pool that owns it means some caller's stack still points into it: a + /// lifetime error, not a state this destructor could ever see in a correct program. + chassert(lanes.empty()); +} + +WriteResult CasHotKeys::submit(const String & key, CasOperation & op, const Retry & policy, const Decide & decide) +{ + /// Frozen here so that the queue wait and the write share one deadline; a policy already frozen + /// by the caller's loop is returned unchanged. + const Retry frozen = op.freeze(policy); + const Retry::Bound bound = frozen.bind(op.owner.now_ms()); + const uint64_t entered_ms = op.owner.now_ms(); + + Item item{}; + { + std::lock_guard lock(mutex); + auto [it, inserted] = lanes.try_emplace(key); + try + { + if (enter_after_lane_hook_for_test) + enter_after_lane_hook_for_test(); + item.ticket = ++next_ticket; + it->second.queue.push_back(&item); + } + catch (...) + { + /// A lane that holds nothing is nobody's; erasing it puts the map back as it was. + if (inserted) + lanes.erase(it); + throw; + } + } + + /// From here the item is in the queue, and this guard is the only thing that removes it: on every + /// exit, normal or by unwinding, it runs the leave step, which allocates nothing and cannot throw. + bool entered_hold = false; + struct Leave + { + CasHotKeys & owner; + const String & key; + Item & item; + const bool & entered_hold; + uint64_t entered_ms; + CasOperation & op; + ~Leave() noexcept { owner.leave(key, item, entered_hold, entered_ms, op); } + } guard{*this, key, item, entered_hold, entered_ms, op}; + + for (;;) + { + /// Outside the mutex: the engine's own admission in the engine's order (the fence generation, + /// the lease budget, then the caller's liveness, which the engine reports as a lost fence), + /// then the caller's bound. A lost fence is therefore never reported as a policy deadline, and + /// an exhausted lease is reported as the lease. + std::optional leaving; + CasOperation::WriteState nothing_sent; + switch (op.gate(0)) + { + case CasOperation::Gate::FenceLost: + leaving = op.gaveUp(GaveUp::Why::FenceLost, sourceFor(bound), nothing_sent); + break; + case CasOperation::Gate::NoBudget: + leaving = op.gaveUp(GaveUp::Why::Deadline, GaveUp::Source::Lease, nothing_sent); + break; + case CasOperation::Gate::Ok: + break; + } + if (!leaving && !op.fits(0, bound)) + leaving = op.gaveUp(GaveUp::Why::Deadline, sourceFor(bound), nothing_sent); + + /// Read outside the mutex, as `leave` already does: the mutex is a leaf that calls nothing, + /// and `now_ms` is a closure the pool injects, not a fixed clock. + const uint64_t now = op.owner.now_ms(); + std::unique_lock lock(mutex); + if (leaving) + return std::move(*leaving); /// the lock is released before the guard erases the item + Lane & lane = lanes.at(key); + if (lane.queue.front() == &item) + { + lane.holder_since_ms = now; + entered_hold = true; + break; + } + lane.cv.wait_for(lock, kWaitSlice); + } + ProfileEvents::increment(ProfileEvents::CASHotKeyQueueWaitMicroseconds, (op.owner.now_ms() - entered_ms) * 1000); + return hold(key, op, frozen, bound, decide); +} + +void CasHotKeys::leave(const String & key, Item & item, bool entered_hold, uint64_t entered_ms, CasOperation & op) noexcept +{ + /// The front item's ticket and how long it has held, when this caller left at its own bound behind + /// it: what makes a stuck holder visible while it is stuck. + std::optional> stuck_behind; + { + std::lock_guard lock(mutex); + auto it = lanes.find(key); + chassert(it != lanes.end()); + Lane & lane = it->second; + auto found = std::find(lane.queue.begin(), lane.queue.end(), &item); + chassert(found != lane.queue.end()); + lane.queue.erase(found); + if (entered_hold) + lane.holder_since_ms.reset(); + else if (!lane.queue.empty() && lane.holder_since_ms) + { + /// `now_ms` is the pool's injected clock and can throw; the fallback is no snapshot, never + /// a throw out of this locked section, whose sole duty -- removing the item -- must complete + /// regardless. + try + { + const uint64_t now = op.owner.now_ms(); + stuck_behind = std::pair{lane.queue.front()->ticket, now - *lane.holder_since_ms}; + } + catch (...) + { + } + } + if (lane.queue.empty()) + lanes.erase(it); + else + lane.cv.notify_all(); + } + try + { + if (!entered_hold) + { + const uint64_t now = op.owner.now_ms(); + ProfileEvents::increment(ProfileEvents::CASHotKeyQueueWaitMicroseconds, (now - entered_ms) * 1000); + } + if (stuck_behind) + LOG_WARNING(getLogger("CasHotKeys"), + "hot key '{}': a writer left at its own bound while ticket {} has held the key for {} ms", + key, stuck_behind->first, stuck_behind->second); + } + catch (...) + { + tryLogCurrentException("CasHotKeys"); + } +} + +WriteResult CasHotKeys::hold(const String & key, CasOperation & op, const Retry & policy, const Retry::Bound & bound, + const Decide & decide) +{ + CasOperation::WriteState state; + std::optional base; + bool from_cache = false; + /// A single-attempt submission never starts from a hint: its one attempt is on fresh state, as + /// the engine's own verb reads before it. + if (!policy.single_attempt) + { + if (auto remembered = cached(key)) + { + base = std::move(remembered); + from_cache = true; + ProfileEvents::increment(ProfileEvents::CASHotKeyCacheStarts); + } + } + const auto read_base = [&]() -> std::optional + { + ProfileEvents::increment(ProfileEvents::CASHotKeyReadStarts); + CasOperation::Resolved resolved = op.observe(key, policy, bound); + state.last_seen = resolved.seen; + base.reset(); + if (const auto * object = std::get_if(&state.last_seen)) + base = *object; + else if (!std::holds_alternative(state.last_seen)) + return op.gaveUpAfterFailedObservation(resolved.stop, state, bound); + return std::nullopt; + }; + if (!from_cache) + if (auto refused = read_base()) + return *refused; + + std::optional candidate; + bool verdict_on_hint = false; + try + { + candidate = decide(base); + verdict_on_hint = from_cache && !candidate; + } + catch (...) + { + if (!from_cache) + throw; + verdict_on_hint = true; + } + if (verdict_on_hint) + { + /// A verdict rendered on a hint proves only that a hint is not a proof: the entry is dropped, + /// the key is read, and what the `decide` does on the read is the result. A read the caller's + /// own gate refuses is that give-up, as for any caller that cannot read. + forget(key); + ProfileEvents::increment(ProfileEvents::CASHotKeyCacheVerdictsReread); + if (auto refused = read_base()) + return *refused; + from_cache = false; + candidate = decide(base); + } + if (!candidate) + { + if (base) + remember(key, *base); + return Declined{state.last_seen}; + } + + WriteResult result = [&] + { + try + { + return base ? op.replace(key, *candidate, base->etag, policy) : op.create(key, *candidate, policy); + } + catch (...) + { + forget(key); + throw; + } + }(); + std::visit(detail::Overload{ + [&](const Committed & committed) { remember(key, Object{*candidate, committed.etag}); }, + [&](const Conflict & conflict) + { + if (const auto * object = std::get_if(&conflict.seen)) + remember(key, *object); + else + forget(key); + }, + [&](const Refused &) { forget(key); }, + [&](const GaveUp & gave_up) { if (gave_up.sent_any) forget(key); }, + [&](const Declined &) {}}, result); + return result; +} + +std::optional CasHotKeys::cached(const String & key) const +{ + if (!cache) + return std::nullopt; + if (auto hit = cache->get(key)) + return hit->object; + return std::nullopt; +} + +void CasHotKeys::remember(const String & key, Object object) noexcept +{ + if (!cache) + return; + try + { + if (cache_fill_hook_for_test) + cache_fill_hook_for_test(); + /// Computed here, where a throw (allocation, `Etag::render`) is still just a failed fill: never + /// zero, since the key, the incarnation and the containers weigh something even when the object + /// is empty, so the byte budget bounds the entry count as well as the bytes. + const size_t weight = key.size() + object.bytes.size() + object.etag.render().size() + 64; + /// An object above the budget is not a hint worth evicting everything else for. + if (weight > cache_budget_bytes) + { + forget(key); + return; + } + auto remembered = std::make_shared(Remembered{std::move(object), weight}); + cache->set(key, remembered); + } + catch (...) + { + /// A hint that could not be stored is no hint: the next hold reads. The result that led here + /// stands; a failed fill must never replace it. + tryLogCurrentException("CasHotKeys"); + forget(key); + } +} + +void CasHotKeys::forget(const String & key) noexcept +{ + if (!cache) + return; + try + { + cache->remove(key); + } + catch (...) + { + tryLogCurrentException("CasHotKeys"); + } +} + +size_t CasHotKeys::queueDepthForTest(const String & key) const +{ + std::lock_guard lock(mutex); + const auto it = lanes.find(key); + return it == lanes.end() ? 0 : it->second.queue.size(); +} + +size_t CasHotKeys::laneCountForTest() const +{ + std::lock_guard lock(mutex); + return lanes.size(); +} + +size_t CasHotKeys::cacheEntriesForTest() const +{ + return cache ? cache->count() : 0; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.h new file mode 100644 index 000000000000..d41ec95e1652 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasHotKeys.h @@ -0,0 +1,124 @@ +#pragma once +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::Cas +{ + +class CasOperation; + +/// One conditional write in flight per key from the operations that write through here, in arrival +/// order, plus the last object this pool knows per key so the next write needs no read. +/// +/// Compare-and-swap is needed only against other servers; inside one server every writer of a shared +/// key used to race every other, and each lost race cost a read, a refused write, a resolve read and a +/// growing sleep. `submit` takes a FIFO ticket for the key, waits its turn re-checking the caller's +/// own fence, lease and deadline, obtains a base (the cache's object, else the engine's own read), +/// runs the caller's `decide` on it, lands the candidate through the engine's `replace` or `create` +/// on the caller's own operation and thread, and returns the engine's result unchanged. The caller +/// keeps the retry loop: a `Conflict` is a lost race against another server, and the caller submits +/// again after `Retry::conflictBackoff`. +/// +/// The cache is a hint and never a source of truth: every write against it is conditional on its +/// etag, so a stale entry costs one 412 and one resolve read; and a verdict a `decide` renders on it +/// (a refusal by exception, or "nothing to write") is never delivered, because a refusal without a +/// write is the one thing a 412 cannot correct -- the entry is dropped, the key is read, and the +/// `decide` runs again on the read. The store's answer to a write decided on a hint is delivered +/// whatever it is: it is a fact about the store, and the caller's ordinary retry learns the rest. +/// +/// One instance per pool, shared by its three request planes; a `CasRequests` built without one owns +/// a private instance with no cache. Every callback (`decide`, a `Liveness` closure, a backend hook) +/// runs with `mutex` released: the mutex is a leaf that calls nothing. +class CasHotKeys +{ +public: + /// `cache_budget_bytes` bounds the remembered objects; 0 disables the cache, and every hold reads. + explicit CasHotKeys(uint64_t cache_budget_bytes); + ~CasHotKeys(); + CasHotKeys(const CasHotKeys &) = delete; + CasHotKeys & operator=(const CasHotKeys &) = delete; + + /// The caller's mutation of `key`, the engine's own decide shape: the candidate bytes to write over + /// `base`, nothing (`Declined`), or a refusal by exception. `base` is absent when the key does not + /// exist; a caller that refuses to bootstrap throws there. A `decide` may be run twice, and a + /// later run on a fresh read is the decision that counts; it issues no write through this lane; + /// its reads go through `op`; it reads `base->bytes` and never `base->etag`. + using Decide = std::function(const std::optional &)>; + + /// One hold on `key`: wait for the turn, obtain a base, run `decide`, one engine write, remember. + /// `policy` is frozen at entry, so time in the queue spends the caller's window. Returns the + /// engine's own result for that write, in class and content, and propagates a `decide`'s + /// exception as `readModifyWrite` does, never from a cached base. + WriteResult submit(const String & key, CasOperation & op, const Retry & policy, const Decide & decide); + + /// Items in the key's queue, the holder included; 0 for a key with no lane. + size_t queueDepthForTest(const String & key) const; + /// Lanes in existence: a lane lives while its queue holds an item. + size_t laneCountForTest() const; + /// Entries the cache holds. + size_t cacheEntriesForTest() const; + + /// TEST SEAM: runs under `mutex` after the key's lane was found or created and before the item + /// is queued; a throw here is the enqueue's allocation failing. + std::function enter_after_lane_hook_for_test; + /// TEST SEAM: runs inside `remember` before the cache is filled; a throw here is the fill failing. + std::function cache_fill_hook_for_test; + +private: + /// One queued submission, on its caller's stack: the caller's own guard is the only thing that + /// removes it, so nothing here outlives the stack it lives on. + struct Item + { + uint64_t ticket; + }; + /// One key. Created on first use, erased when its queue empties; referenced only by threads whose + /// item is in its queue. + struct Lane + { + std::deque queue; /// guarded by `mutex`; the front is the holder + std::optional holder_since_ms; /// guarded by `mutex`; for the log line + std::condition_variable cv; + }; + /// A remembered object together with its precomputed weight: `Etag::render` allocates and can + /// throw, and the cache's own accounting calls the weight after it has already subtracted the + /// entry being replaced, so the weight itself must never throw. + struct Remembered + { + Object object; + size_t weight; + }; + struct RememberedWeight + { + size_t operator()(const Remembered & remembered) const noexcept { return remembered.weight; } + }; + using Cache = CacheBase, RememberedWeight>; + + WriteResult hold(const String & key, CasOperation & op, const Retry & policy, const Retry::Bound & bound, + const Decide & decide); + void leave(const String & key, Item & item, bool entered_hold, uint64_t entered_ms, CasOperation & op) noexcept; + + std::optional cached(const String & key) const; + void remember(const String & key, Object object) noexcept; + void forget(const String & key) noexcept; + + mutable std::mutex mutex; /// `mutable` for the const test seams + uint64_t next_ticket = 0; /// guarded by `mutex`; the holder's identity in the log line + std::unordered_map lanes; /// guarded by `mutex` + const uint64_t cache_budget_bytes; + /// The last known object per key. Its synchronization is its own; it is read and written with + /// `mutex` released. Null when the budget is 0. + std::unique_ptr cache; +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.cpp index fef7cdaafad8..b6bf981a40f1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.cpp @@ -2,8 +2,9 @@ #include #include #include +#include +#include #include -#include namespace DB { @@ -11,6 +12,7 @@ namespace ErrorCodes { extern const int CORRUPTED_DATA; extern const int FILE_DOESNT_EXIST; + extern const int LOGICAL_ERROR; } } @@ -20,105 +22,176 @@ namespace DB::Cas namespace { -/// The windowed slice of `data` for `range`, with the same clamping `get` documents: an offset at or -/// past EOF yields an empty result; an open-ended length runs to EOF. Shared by `get` and `getStream` -/// so the two stay in lockstep. -String sliceWindow(const String & data, Range range) +/// A CALLER bug, refused before it ever reaches the store: an empty, wildcard or list value would +/// turn a conditional mutation into an unconditional one. Stricter than +/// `isIncarnationValue(Dialect::Emulated, ...)` (non-empty only): this backend is a test double +/// reused across the whole CAS gtest suite, so it also refuses `*` and a comma even though no value +/// it currently mints can contain either. +bool isValidEmulatedTokenValue(const String & value) { - const size_t offset = static_cast(range.offset); - if (offset >= data.size()) - return {}; - if (range.length.has_value()) - return data.substr(offset, static_cast(*range.length)); - return data.substr(offset); + return !value.empty() && value != "*" && value.find(',') == String::npos; } +/// Every conditional mutation refuses a malformed expected value unconditionally, matching the +/// production backend's own guard: this is the test backend for that contract, and must not accept +/// anything production refuses. A caller holding only a HEAD-derived value for a key it has not +/// confirmed exists must gate on presence itself before calling, exactly as every production call +/// site already does. +void checkExpectedValue(const String & key, const String & value) +{ + if (!isValidEmulatedTokenValue(value)) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "InMemoryBackend: refusing a conditional mutation of '{}' with a malformed token '{}': " + "an empty, wildcard or list token would turn the precondition into an unconditional write", + key, value); } -Token InMemoryBackend::mintToken() +} + +String InMemoryBackend::mintValue() { - Token t; - t.value = std::to_string(++token_seq_); - t.type = TokenType::Emulated; - return t; + return std::to_string(++token_seq_); } -std::optional InMemoryBackend::get(const String & key, Range range) +std::exception_ptr InMemoryBackend::takeArmedFailure(ArmedFailures & armed, const String & key) { + std::lock_guard lock(mutex_); + const auto it = armed.find(key); + if (it == armed.end() || it->second.empty()) + return nullptr; + std::exception_ptr error = it->second.front(); + it->second.erase(it->second.begin()); + return error; +} + +std::function InMemoryBackend::hookFor(const Hooks & hooks, const String & key) const +{ + std::lock_guard lock(mutex_); + const auto it = hooks.find(key); + return it == hooks.end() ? std::function{} : it->second; +} + +std::optional InMemoryBackend::read(const String & key, TransportAccess &) +{ + if (auto armed = takeArmedFailure(read_failures_, key)) + std::rethrow_exception(armed); + std::lock_guard lock(mutex_); auto it = store_.find(key); if (it == store_.end()) return std::nullopt; - GetResult gr; - gr.bytes = sliceWindow(it->second.bytes, range); - gr.token = it->second.token; - gr.attributes = it->second.meta; - return gr; + return Raw{it->second.bytes, it->second.value}; } -std::optional InMemoryBackend::getStream(const String & key, Range range) +std::unique_ptr InMemoryBackend::stream(const String & key, TransportAccess &) { std::lock_guard lock(mutex_); auto it = store_.find(key); if (it == store_.end()) - return std::nullopt; + return nullptr; - /// Copy the windowed bytes into an owning buffer — the in-memory backend has no separate storage - /// to stream from, so the "stream" reads from a private copy of exactly the requested window. - GetStreamResult sr; - sr.stream = std::make_unique(sliceWindow(it->second.bytes, range)); - sr.token = it->second.token; - return sr; + /// Copies the bytes into an owning buffer — the in-memory backend has no separate storage to + /// stream from, so the "stream" reads from a private copy taken while the lock is held. + return std::make_unique(it->second.bytes); } -HeadResult InMemoryBackend::head(const String & key) +std::optional InMemoryBackend::head(const String & key, TransportAccess &) { + if (auto armed = takeArmedFailure(head_failures_, key)) + std::rethrow_exception(armed); + std::lock_guard lock(mutex_); auto it = store_.find(key); if (it == store_.end()) - return HeadResult{}; + return std::nullopt; + + return RawMeta{static_cast(it->second.bytes.size()), it->second.value}; +} - HeadResult hr; - hr.exists = true; - hr.size = static_cast(it->second.bytes.size()); - hr.token = it->second.token; - hr.attributes = it->second.meta; - return hr; +std::expected InMemoryBackend::write( + const String & key, const String & bytes, const std::optional & expected_value, TransportAccess &) +{ + return applyWrite(key, bytes, expected_value); } -PutResult InMemoryBackend::putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) +std::expected InMemoryBackend::applyWrite( + const String & key, const String & bytes, const std::optional & expected_value) +{ + if (expected_value) + checkExpectedValue(key, *expected_value); + + if (auto armed = takeArmedFailure(write_failures_, key)) + std::rethrow_exception(armed); + + /// Both hooks run with NO lock held: a hook exists to read and write this backend from inside a + /// write, and `mutex_` is not recursive. + if (auto hook = hookFor(before_write_hooks_, key)) + hook(); + + auto result = writeUnderLock(key, bytes, expected_value); + if (!result.has_value()) + return result; + + if (auto hook = hookFor(write_committed_hooks_, key)) + hook(); + + /// Last, so the object is durable and every observer has run before the response goes missing. + if (takeAmbiguousLandedWrite(key)) + throw Poco::TimeoutException("InMemoryBackend: the write of '" + key + "' landed and its response was lost"); + + return result; +} + +std::expected InMemoryBackend::writeUnderLock( + const String & key, const String & bytes, const std::optional & expected_value) { std::lock_guard lock(mutex_); - // One-shot injected ambiguous outcome: throw WITHOUT touching the store, modeling a request whose - // own attempt outcome never reached the caller (see the header doc for the classification this - // must produce). std::runtime_error, not DB::Exception, is deliberate: it dodges BOTH - // classification paths in BOTH build configurations -- dynamic_cast fails (so - // isDeterministicLocalFailure is never consulted), and classifyConditionalWriteResult falls through - // to its Unresolved default because it isn't an S3Exception. A DB::Exception would have been - // fragile: picking a code outside isDeterministicLocalFailure's set is a landmine for the next - // person who extends that set. - auto ambiguous_it = ambiguous_put_keys_.find(key); - if (ambiguous_it != ambiguous_put_keys_.end()) + // One-shot injected ambiguous outcome: throw WITHOUT touching the store, modeling a request + // whose own attempt outcome never reached the caller. Poco::TimeoutException, not + // DB::Exception, is deliberate: a client-side timeout is the real shape of this failure, and + // its class is what every caller classifies by -- ambiguous, in both build configurations. + auto ambiguous_it = ambiguous_write_keys_.find(key); + if (ambiguous_it != ambiguous_write_keys_.end()) + { + ambiguous_write_keys_.erase(ambiguous_it); + throw Poco::TimeoutException("InMemoryBackend: injected ambiguous write outcome for '" + key + "'"); + } + + auto refuse_it = refuse_next_write_keys_.find(key); + if (refuse_it != refuse_next_write_keys_.end()) + { + refuse_next_write_keys_.erase(refuse_it); + return std::unexpected(RawConflict{}); + } + + if (!expected_value) { - ambiguous_put_keys_.erase(ambiguous_it); - throw std::runtime_error("InMemoryBackend: injected ambiguous putIfAbsent outcome for '" + key + "'"); + if (store_.contains(key)) + return std::unexpected(RawConflict{}); + + String v = mintValue(); + Object obj; + obj.bytes = bytes; + obj.value = v; + store_[key] = std::move(obj); + return v; } - if (store_.contains(key)) - return {PutOutcome::PreconditionFailed, {}}; + auto it = store_.find(key); + if (it == store_.end()) + return std::unexpected(RawConflict{}); + if (enforce_tokens_ && it->second.value != *expected_value) + return std::unexpected(RawConflict{}); - Token t = mintToken(); - Object obj; - obj.bytes = bytes; - obj.token = t; - obj.meta = meta; - store_[key] = std::move(obj); - return {PutOutcome::Done, t}; + String v = mintValue(); + it->second.bytes = bytes; + it->second.value = v; + return v; } -void InMemoryBackend::publishBlob(const BlobPublishRequest & request) +void InMemoryBackend::publish(const BlobPublishRequest & request, TransportAccess &) { if (const auto * streaming = std::get_if(&request.publication)) { @@ -126,7 +199,7 @@ void InMemoryBackend::publishBlob(const BlobPublishRequest & request) if (!payload) throw Exception( ErrorCodes::CORRUPTED_DATA, - "InMemoryBackend::publishBlob: payload source for {} returned no reader", + "InMemoryBackend::publish: payload source for {} returned no reader", request.destination_key); /// Drain before taking the store lock: the source may itself read another object from this @@ -145,7 +218,7 @@ void InMemoryBackend::publishBlob(const BlobPublishRequest & request) if (!copy_result.exact(streaming->payload_size)) throw Exception( ErrorCodes::CORRUPTED_DATA, - "InMemoryBackend::publishBlob: source yielded {}{} payload bytes for {}, declared {} -- nothing was published", + "InMemoryBackend::publish: source yielded {}{} payload bytes for {}, declared {} -- nothing was published", copy_result.has_excess ? "more than " : "", copy_result.copied, request.destination_key, @@ -154,7 +227,7 @@ void InMemoryBackend::publishBlob(const BlobPublishRequest & request) std::lock_guard lock(mutex_); Object object; object.bytes = std::move(body); - object.token = mintToken(); + object.value = mintValue(); store_[request.destination_key] = std::move(object); return; } @@ -165,141 +238,119 @@ void InMemoryBackend::publishBlob(const BlobPublishRequest & request) if (source == store_.end()) throw Exception( ErrorCodes::FILE_DOESNT_EXIST, - "InMemoryBackend::publishBlob: staging object {} is absent", + "InMemoryBackend::publish: staging object {} is absent", staged.object_key); Object object; object.bytes = source->second.bytes; - object.token = mintToken(); + object.value = mintValue(); store_[request.destination_key] = std::move(object); } -PutResult InMemoryBackend::putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) +Backend::RawRemoval InMemoryBackend::applyDelete(const String & key, const String & expected_value) { - std::lock_guard lock(mutex_); + // Caller holds the mutex. auto it = store_.find(key); if (it == store_.end()) - return {PutOutcome::PreconditionFailed, {}}; + return RawRemoval::Gone; - if (enforce_tokens_ && it->second.token != expected) - return {PutOutcome::PreconditionFailed, {}}; + if (enforce_tokens_ && it->second.value != expected_value) + return RawRemoval::Mismatch; - Token t = mintToken(); - it->second.bytes = bytes; - it->second.token = t; - it->second.meta = meta; - return {PutOutcome::Done, t}; + store_.erase(it); + return simulate_delete_markers_ ? RawRemoval::DeleteMarker : RawRemoval::Removed; } -CasResult InMemoryBackend::casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) +Backend::RawRemoval InMemoryBackend::remove(const String & key, const String & expected_value, TransportAccess &) { + /// See `checkExpectedValue`: a malformed value is refused as a caller bug, unconditionally -- + /// covering both the immediate delete below and the hold_deletes_ enqueue path, so a queued + /// PendingDelete can never carry a malformed value either. + checkExpectedValue(key, expected_value); + std::lock_guard lock(mutex_); - // One-shot injected conflict - auto fail_it = fail_next_cas_.find(key); - if (fail_it != fail_next_cas_.end()) + if (hold_deletes_) { - fail_next_cas_.erase(fail_it); - return {CasOutcome::Conflict, {}}; + // Validate the key exists (and the value matches if enforcing) before queuing, + // but don't remove yet — just enqueue. + auto it = store_.find(key); + if (it == store_.end()) + return RawRemoval::Gone; + if (enforce_tokens_ && it->second.value != expected_value) + return RawRemoval::Mismatch; + PendingDelete pd; + pd.key = key; + pd.value = expected_value; + pending_deletes_.push_back(std::move(pd)); + return simulate_delete_markers_ ? RawRemoval::DeleteMarker : RawRemoval::Removed; } - auto it = store_.find(key); - bool exists = (it != store_.end()); - - if (!expected.has_value()) - { - // create-if-absent CAS - if (exists) - return {CasOutcome::Conflict, {}}; - Token t = mintToken(); - Object obj; - obj.bytes = bytes; - obj.token = t; - obj.meta = meta; - store_[key] = std::move(obj); - return {CasOutcome::Committed, t}; - } - else - { - // swap-if-current CAS - if (!exists) - return {CasOutcome::Conflict, {}}; - if (enforce_tokens_ && it->second.token != *expected) - return {CasOutcome::Conflict, {}}; - Token t = mintToken(); - it->second.bytes = bytes; - it->second.token = t; - it->second.meta = meta; - return {CasOutcome::Committed, t}; - } + return applyDelete(key, expected_value); } -DeleteOutcome InMemoryBackend::applyDelete(const String & key, const Token & token) +void InMemoryBackend::removeManyWriteOnce(const std::vector & keys, TransportAccess &) { - // Caller holds the mutex. - auto it = store_.find(key); - if (it == store_.end()) + std::exception_ptr armed; + std::function hook; { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::NotFound; - return d; + std::lock_guard lock(mutex_); + ++bulk_remove_calls_; + if (!armed_bulk_remove_failures_.empty()) + { + armed = armed_bulk_remove_failures_.front(); + armed_bulk_remove_failures_.erase(armed_bulk_remove_failures_.begin()); + } + hook = before_bulk_remove_hook_; } + if (armed) + std::rethrow_exception(armed); + if (hook) + hook(); - if (enforce_tokens_ && it->second.token != token) + std::lock_guard lock(mutex_); + for (const WriteOnceKey & key : keys) { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; + auto it = store_.find(key.str()); + if (it == store_.end()) + continue; /// absent is success + if (hold_deletes_) + { + PendingDelete pd; + pd.key = key.str(); + pd.value = it->second.value; + pending_deletes_.push_back(std::move(pd)); + continue; + } + store_.erase(it); } - - store_.erase(it); - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::Deleted; - d.created_delete_marker = simulate_delete_markers_; - return d; } -DeleteOutcome InMemoryBackend::deleteExact(const String & key, const Token & token) +void InMemoryBackend::failNextBulkRemoveWith(std::exception_ptr error) { std::lock_guard lock(mutex_); + armed_bulk_remove_failures_.push_back(std::move(error)); +} - if (hold_deletes_) - { - // Validate the key exists (and token matches if enforcing) before queuing, - // but don't remove yet — just enqueue. - auto it = store_.find(key); - if (it == store_.end()) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::NotFound; - return d; - } - if (enforce_tokens_ && it->second.token != token) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; - } - PendingDelete pd; - pd.key = key; - pd.token = token; - pending_deletes_.push_back(std::move(pd)); - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::Deleted; - d.created_delete_marker = simulate_delete_markers_; - return d; - } +void InMemoryBackend::onBeforeBulkRemove(std::function hook) +{ + std::lock_guard lock(mutex_); + before_bulk_remove_hook_ = std::move(hook); +} - return applyDelete(key, token); +size_t InMemoryBackend::bulkRemoveCalls() const +{ + std::lock_guard lock(mutex_); + return bulk_remove_calls_; } -ListPage InMemoryBackend::list(const String & prefix, const String & cursor, size_t limit) +Backend::RawListPage InMemoryBackend::list(const String & prefix, const String & cursor, size_t limit, TransportAccess &) { if (limit == 0) return {}; std::lock_guard lock(mutex_); - ListPage page; + RawListPage page; // Cursor is the last key returned by the previous page. auto it = cursor.empty() ? store_.lower_bound(prefix) : store_.upper_bound(cursor); @@ -310,10 +361,10 @@ ListPage InMemoryBackend::list(const String & prefix, const String & cursor, siz if (!it->first.starts_with(prefix)) break; - ListedKey lk; + RawListedKey lk; lk.key = it->first; lk.size = static_cast(it->second.bytes.size()); - lk.token = it->second.token; /// in-memory backend always surfaces the token (supportsListTokens == true) + lk.value = it->second.value; /// in-memory backend always surfaces it (supportsListTokens == true) page.keys.push_back(std::move(lk)); ++count; ++it; @@ -326,6 +377,49 @@ ListPage InMemoryBackend::list(const String & prefix, const String & cursor, siz return page; } +bool InMemoryBackend::refreshCredentials() +{ + std::lock_guard lock(mutex_); + ++refresh_credentials_calls_; + return refresh_credentials_result_; +} + +size_t InMemoryBackend::refreshCredentialsCalls() const +{ + std::lock_guard lock(mutex_); + return refresh_credentials_calls_; +} + +void InMemoryBackend::failNextWriteWith(const String & key, std::exception_ptr error) +{ + std::lock_guard lock(mutex_); + write_failures_[key].push_back(std::move(error)); +} + +void InMemoryBackend::failNextReadWith(const String & key, std::exception_ptr error) +{ + std::lock_guard lock(mutex_); + read_failures_[key].push_back(std::move(error)); +} + +void InMemoryBackend::failNextHeadWith(const String & key, std::exception_ptr error) +{ + std::lock_guard lock(mutex_); + head_failures_[key].push_back(std::move(error)); +} + +void InMemoryBackend::onBeforeWrite(const String & key, std::function hook) +{ + std::lock_guard lock(mutex_); + before_write_hooks_[key] = std::move(hook); +} + +void InMemoryBackend::onWriteCommitted(const String & key, std::function hook) +{ + std::lock_guard lock(mutex_); + write_committed_hooks_[key] = std::move(hook); +} + void InMemoryBackend::setHoldDeletes(bool hold) { std::lock_guard lock(mutex_); @@ -338,33 +432,41 @@ size_t InMemoryBackend::pendingDeletes() const return pending_deletes_.size(); } -DeleteOutcome InMemoryBackend::landPendingDelete(size_t i) +Backend::RawRemoval InMemoryBackend::landPendingDelete(size_t i) { std::lock_guard lock(mutex_); if (i >= pending_deletes_.size()) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::NotFound; - return d; - } + return RawRemoval::Gone; PendingDelete pd = pending_deletes_[i]; pending_deletes_.erase(pending_deletes_.begin() + static_cast(i)); - // Apply the token check at LAND time — the object may have been modified since the delete was enqueued. - return applyDelete(pd.key, pd.token); + // Apply the value check at LAND time — the object may have been modified since the delete was enqueued. + return applyDelete(pd.key, pd.value); +} + +void InMemoryBackend::refuseNextWrite(const String & key) +{ + std::lock_guard lock(mutex_); + refuse_next_write_keys_.insert(key); +} + +void InMemoryBackend::injectAmbiguousWrite(const String & key) +{ + std::lock_guard lock(mutex_); + ambiguous_write_keys_.insert(key); } -void InMemoryBackend::failNextCasPut(const String & key) +void InMemoryBackend::injectAmbiguousLandedWrite(const String & key) { std::lock_guard lock(mutex_); - fail_next_cas_.insert(key); + ambiguous_landed_keys_.insert(key); } -void InMemoryBackend::injectAmbiguousPutIfAbsent(const String & key) +bool InMemoryBackend::takeAmbiguousLandedWrite(const String & key) { std::lock_guard lock(mutex_); - ambiguous_put_keys_.insert(key); + return ambiguous_landed_keys_.erase(key) != 0; } void InMemoryBackend::setEnforceTokens(bool enforce) @@ -379,4 +481,10 @@ void InMemoryBackend::setSimulateDeleteMarkers(bool simulate) simulate_delete_markers_ = simulate; } +void InMemoryBackend::setRefreshCredentialsResult(bool result) +{ + std::lock_guard lock(mutex_); + refresh_credentials_result_ = result; +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h index 5e67aca61f04..2074750e3ef6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h @@ -1,5 +1,7 @@ #pragma once #include +#include +#include #include #include #include @@ -8,14 +10,15 @@ namespace DB::Cas { -/// Thread-safe, token-enforcing in-memory `Backend` implementation used by CAS tests. +/// Thread-safe, value-enforcing in-memory `Backend` implementation used by CAS tests. Mints its +/// values in the `Dialect::Emulated` dialect. /// -/// All successful writes mint a monotonically increasing token (`TokenType::Emulated`). -/// Tokens NEVER repeat across the lifetime of a backend instance. +/// All successful writes mint a monotonically increasing value. Values NEVER repeat across the +/// lifetime of a backend instance. /// /// The backend also exposes fault-injection controls for probe tests and CAS correctness tests: /// - `setHoldDeletes` / `landPendingDelete`: simulate async/delayed conditional deletes -/// - `failNextCasPut`: inject a one-shot conflict +/// - `refuseNextWrite`: inject a one-shot conflict /// - `setEnforceTokens(false)`: mimic a "dumb" backend that ignores token checks /// - `setSimulateDeleteMarkers`: mimic S3 versioning-enabled buckets /// @@ -26,132 +29,200 @@ class InMemoryBackend : public Backend public: InMemoryBackend() = default; - /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the - /// overrides below would otherwise shadow them for callers holding a concrete backend type. - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - // ---- Backend interface ---- - /// Returns the requested byte window, current token, and metadata, or `nullopt` when the key is absent. - std::optional get(const String & key, Range range) override; - - /// Returns a forward-only stream over the requested byte window, or `nullopt` when the key is absent. - /// The in-memory implementation copies the window into an owning read buffer while holding the - /// backend lock, so the returned stream remains independent of later backend mutations. - std::optional getStream(const String & key, Range range) override; - - /// Returns the current existence, size, token, and metadata without materializing the body. - HeadResult head(const String & key) override; - - /// The in-memory backend mints a monotonic token it surfaces through `list` — TRUE. - bool supportsListTokens() const override { return true; } - - /// Creates `key` only when it is absent. On success stores `bytes` and `meta` under a new token; - /// on a precondition failure leaves the existing object untouched. - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override; + /// Returns the stored bytes and the key's current incarnation value, or `nullopt` when absent. + std::optional read(const String & key, TransportAccess & access) override; + + /// Returns the current size and incarnation value without materializing the body. + std::optional head(const String & key, TransportAccess & access) override; + + /// Lists up to `limit` keys under `prefix` in map order. `cursor` is the last key from the + /// previous page; `next_cursor` is set only when more matching keys remain. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override; + + /// Removes exactly the incarnation named by `expected_value`, or queues that check for a later + /// `landPendingDelete` when delete holding is enabled. A queued delete is reported as removed, + /// but its expected value is rechecked when it is landed. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override; + + /// Deletes every present key with no precondition; an absent key is success. Honours + /// `hold_deletes_` exactly as `remove` does: a held delete is queued and lands only on + /// `landPendingDelete`. + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override; + /// The next `removeManyWriteOnce` throws `error` instead of deleting anything; one-shot, like + /// `failNextWriteWith`. + void failNextBulkRemoveWith(std::exception_ptr error); + /// Runs before a `removeManyWriteOnce` applies, with no backend lock held, on the attempt that + /// will delete (an armed failure fires first and skips the hook). + void onBeforeBulkRemove(std::function hook); + /// How many `removeManyWriteOnce` calls reached the store, armed failures included. + size_t bulkRemoveCalls() const; + + /// Creates the key when `expected_value` is empty, or replaces the incarnation it names. A + /// refused precondition leaves the store unchanged. Value enforcement can be disabled with + /// `setEnforceTokens` to model a backend that incorrectly ignores the condition. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override; + + /// A forward-only reader over a private copy of the bytes; null when the key is absent. + std::unique_ptr stream(const String & key, TransportAccess & access) override; /// Publishes either `[fresh_envelope][payload]` or the complete staged bytes as one atomic /// in-memory replacement. Streaming sources are fully validated before the destination changes. - void publishBlob(const BlobPublishRequest & request) override; + void publish(const BlobPublishRequest & request, TransportAccess & access) override; - /// Replaces the existing object only when `expected` is its current token. Token enforcement can - /// be disabled with `setEnforceTokens` to model a backend that incorrectly ignores this condition. - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, - const ObjectMeta & meta) override; + /// This backend mints its own emulated values. + Dialect dialect() const override { return Dialect::Emulated; } - /// Performs create-if-absent when `expected` is empty, or replace-if-current-token otherwise. - /// Conflicts leave the store unchanged and are returned as an outcome rather than an exception. - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override; + /// Zero unless a fixture sets one. The engine reserves this before every attempt it starts, so a + /// pool fixture that configures `CasRequestBudget::attempt_timeout_ms` must set the SAME value + /// here: production pairs the two (`ContentAddressedMetadataStorage` builds its backend from the + /// pool's budget), and a fixture that sets only the budget leaves the engine reserving nothing. + uint64_t attemptTimeoutMs() const override { return attempt_timeout_ms; } + void setAttemptTimeoutMs(uint64_t ms) { attempt_timeout_ms = ms; } - /// Removes exactly the incarnation named by `token`, or queues that token check for a later - /// `landPendingDelete` when delete holding is enabled. A queued delete is reported as accepted, - /// but its token is rechecked when it is landed. - DeleteOutcome deleteExact(const String & key, const Token & token) override; + /// The in-memory backend mints a monotonic value it surfaces through `list` — TRUE. + bool supportsListTokens() const override { return true; } - /// Lists up to `limit` keys under `prefix` in map order. `cursor` is the last key from the previous - /// page; returned tokens identify the listed incarnations and `next_cursor` is set only when more - /// matching keys remain. - ListPage list(const String & prefix, const String & cursor, size_t limit) override; + /// Whatever `setRefreshCredentialsResult` last configured; FALSE by default, so a test that has + /// not opted in models a backend with no refresh mechanism. + bool refreshCredentials() override; // ---- Fault-injection controls ---- - /// When true, `deleteExact` validates and enqueues deletes rather than applying them immediately. - /// The caller sees `Deleted` (the send was accepted), but the object remains until - /// `landPendingDelete`, where the token is checked again. + /// When true, `remove` validates and enqueues deletes rather than applying them immediately. + /// The caller sees `Removed` (the send was accepted), but the object remains until + /// `landPendingDelete`, where the expected value is checked again. void setHoldDeletes(bool hold); /// Returns the number of currently held deletes. size_t pendingDeletes() const; - /// Applies and removes the held delete at index `i`. The token is evaluated against the current - /// object at land time; the queue entry is removed whether the result is `TokenMismatch` or - /// `Deleted`. An invalid index returns `NotFound`. - DeleteOutcome landPendingDelete(size_t i); - - /// Injects a one-shot artificial `Conflict` on the next `casPut` for `key`. - void failNextCasPut(const String & key); - - /// Injects a one-shot AMBIGUOUS outcome on the next `putIfAbsent` for `key`: instead of attempting - /// the write, that call throws a plain (non-`DB::Exception`) exception -- classified `Unresolved`, - /// never `DefiniteFailure`, by `classifyConditionalWriteResult` regardless of build flags -- and the - /// store is left exactly as it was. Models a request whose own HTTP attempt outcome is lost (a - /// timeout, a dropped connection) rather than a clean `PreconditionFailed`, for tests of controlled - /// ops (`CasRequestController::slotOccupy` and its callers) that must exercise the "ambiguous - /// attempt, resolve before deciding" path without a live network. One-shot, mirroring - /// `failNextCasPut`'s contract: consumed by the first matching `putIfAbsent` call, whether the key - /// was already present or not. - void injectAmbiguousPutIfAbsent(const String & key); - - /// Enables or disables token checks for delete, overwrite, and CAS operations. Disabling checks - /// models a backend that reports every expected token as matching. + /// Applies and removes the held delete at index `i`. The expected value is evaluated against the + /// current object at land time; the queue entry is removed whether the result is `Mismatch` or + /// `Removed`. An invalid index returns `NotFound`. + RawRemoval landPendingDelete(size_t i); + + /// Refuses the next write of `key` once, as a clean precondition failure that leaves the store + /// unchanged -- whatever the write's shape and whichever surface issued it. + void refuseNextWrite(const String & key); + + /// Injects a one-shot AMBIGUOUS outcome on the next write of `key`: instead of attempting it, that + /// call throws `Poco::TimeoutException` and the store is left exactly as it was. Models a request + /// whose own HTTP attempt outcome is lost (a timeout, a dropped connection) rather than a clean + /// refusal, for tests that must exercise the "ambiguous attempt, resolve before deciding" path + /// without a live network. + void injectAmbiguousWrite(const String & key); + + /// The other ambiguity, and the only one that can prove a resolve read settles a commit: the next + /// write of `key` IS APPLIED and then throws `Poco::TimeoutException`, so the object is durable and + /// its incarnation was never returned. One-shot. + void injectAmbiguousLandedWrite(const String & key); + + /// Enables or disables value checks for remove and replace. Disabling checks models a backend + /// that reports every expected value as matching. void setEnforceTokens(bool enforce); - /// When true, successful deletes report `created_delete_marker = true`, modelling a versioned S3 - /// bucket whose delete creates a marker instead of reclaiming the current object. + /// When true, a successful `remove` answers `DeleteMarker`, modelling a versioned S3 bucket whose + /// delete creates a marker instead of reclaiming the current object. void setSimulateDeleteMarkers(bool simulate); + /// What `refreshCredentials` answers: TRUE models a storage that installed fresh credentials. + void setRefreshCredentialsResult(bool result); + + /// How many times `refreshCredentials` has been called on this backend. + size_t refreshCredentialsCalls() const; + + /// The next `write` naming `key` throws `error` instead of applying it, and the store is left + /// exactly as it was. Each arming is consumed by one write, so arming twice fails two consecutive + /// attempts of the same call. An `exception_ptr` rather than a concrete type because a caller + /// classifies a failed attempt by its exception CLASS, and the classes worth exercising span + /// `S3Exception`, `DB::Exception`, `Poco::Exception` and plain `std::exception`. + void failNextWriteWith(const String & key, std::exception_ptr error); + /// The read-side siblings, for the read loop's own classification. `read` and `head` are armed + /// separately because the two resolve loops differ in exactly which of them they issue: a + /// presence-only caller must be able to fail its HEAD without a body read stealing the arming. + void failNextReadWith(const String & key, std::exception_ptr error); + void failNextHeadWith(const String & key, std::exception_ptr error); + + /// Runs before a write of `key` is applied, with no backend lock held -- so a hook may itself read + /// and write this backend, which is what it exists for: a hook that replaces `key` models a + /// permanently hot key whose incarnation moves under every attempt. A hook that writes the same + /// key re-enters this callback, so a hook must guard its own recursion. + void onBeforeWrite(const String & key, std::function hook); + /// Runs after a write of `key` is durable and BEFORE its value is returned, with no backend lock + /// held -- the point at which a fact outside the store can change while a write is in flight. + void onWriteCommitted(const String & key, std::function hook); + private: /// Complete in-memory incarnation state for one key. All fields are read or modified while - /// `mutex_` is held; replacing `token` marks a new incarnation even when the bytes are unchanged. + /// `mutex_` is held; replacing `value` marks a new incarnation even when the bytes are unchanged. struct Object { String bytes; - Token token; - ObjectMeta meta; + String value; }; - /// Token captured when a held delete is queued. It is intentionally checked again at land time so - /// a replacement between send and land produces `TokenMismatch` rather than deleting the new object. + /// Value captured when a held delete is queued. It is intentionally checked again at land time so + /// a replacement between send and land produces `Mismatch` rather than deleting the new object. struct PendingDelete { String key; - Token token; + String value; }; - /// Mints the next process-local token. Tokens are strictly increasing and never reused by this - /// backend instance, which also makes token equality a safe content-cache identity check in tests. - Token mintToken(); + /// Mints the next process-local incarnation value. Strictly increasing and never reused by this + /// backend instance, which also makes value equality a safe content-cache identity check in tests. + String mintValue(); - /// Applies an exact-token delete while `mutex_` is already held. Used by immediate deletes and by + /// Applies an exact-value delete while `mutex_` is already held. Used by immediate deletes and by /// `landPendingDelete` after its queue entry has been removed. - DeleteOutcome applyDelete(const String & key, const Token & token); + RawRemoval applyDelete(const String & key, const String & expected_value); + + using ArmedFailures = std::map>; + using Hooks = std::map>; + + /// Consumes and returns the next failure armed for `key`, or null when none is. + std::exception_ptr takeArmedFailure(ArmedFailures & armed, const String & key); + /// Consumes the landed-then-lost arming for `key`, if there is one. + bool takeAmbiguousLandedWrite(const String & key); + /// A copy of the hook registered for `key`, taken under the lock so the caller can run it without + /// one. + std::function hookFor(const Hooks & hooks, const String & key) const; + /// One write, whichever verb asked for it: armed failure, hooks, the store mutation and the knobs. + std::expected applyWrite(const String & key, const String & bytes, + const std::optional & expected_value); + /// The part of `applyWrite` that touches the store, run with `mutex_` held. + std::expected writeUnderLock(const String & key, const String & bytes, + const std::optional & expected_value); mutable std::mutex mutex_; std::map store_; uint64_t token_seq_ = 0; + /// Set before any operation runs, and read without the lock for the same reason the engine reads + /// it once at construction: it belongs to setup, not to a request. + uint64_t attempt_timeout_ms = 0; + // Fault-injection state. These fields are protected by `mutex_` just like `store_`. bool hold_deletes_ = false; std::vector pending_deletes_; - std::set fail_next_cas_; - std::set ambiguous_put_keys_; + std::set refuse_next_write_keys_; + std::set ambiguous_write_keys_; + std::set ambiguous_landed_keys_; bool enforce_tokens_ = true; bool simulate_delete_markers_ = false; + bool refresh_credentials_result_ = false; + size_t refresh_credentials_calls_ = 0; + ArmedFailures write_failures_; + ArmedFailures read_failures_; + ArmedFailures head_failures_; + Hooks before_write_hooks_; + Hooks write_committed_hooks_; + std::vector armed_bulk_remove_failures_; + std::function before_bulk_remove_hook_; + size_t bulk_remove_calls_ = 0; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.cpp index 4749d0c8ca35..4782bd3f1011 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.cpp @@ -15,6 +15,8 @@ extern const Event CASBlobGetStream; extern const Event CASBlobDelete; extern const Event CASBlobList; +extern const Event CASBulkDeleteRequests; + extern const Event CASManifestPut; extern const Event CASManifestPutDeduplicated; extern const Event CASManifestOverwrite; @@ -82,6 +84,10 @@ namespace DB::Cas /// Maps `(CasNs, CasOp)` to the corresponding `ProfileEvents::Event`. The table is row-major: the /// outer index is the namespace and the inner index is the operation. Its rows and columns must stay /// in lockstep with the `CasNs` and `CasOp` enum orderings. +/// +/// `CasOp::Read` deliberately keeps the `CAS*Get` event names: renaming a user-visible ProfileEvent +/// is its own change, made when the events are retired, and nothing outside this table would have +/// been improved by doing it here. static const ProfileEvents::Event cas_event_table[CAS_NS_COUNT][CAS_OP_COUNT] = { /* Blob */ {ProfileEvents::CASBlobPut, ProfileEvents::CASBlobPutDeduplicated, ProfileEvents::CASBlobOverwrite, @@ -134,10 +140,18 @@ void incrementCasEvent(CasNs ns, CasOp op) ProfileEvents::increment(cas_event_table[static_cast(ns)][static_cast(op)]); } -void InstrumentedBackend::publishBlob(const BlobPublishRequest & request) +void InstrumentedBackend::publish(const BlobPublishRequest & request, TransportAccess & access) { - inner->publishBlob(request); + inner->publish(request, access); incrementCasEvent(classifyCasNs(request.destination_key), CasOp::Put); } +void InstrumentedBackend::removeManyWriteOnce(const std::vector & keys, TransportAccess & access) +{ + inner->removeManyWriteOnce(keys, access); + ProfileEvents::increment(ProfileEvents::CASBulkDeleteRequests); + for (const WriteOnceKey & key : keys) + incrementCasEvent(classifyCasNs(key.str()), CasOp::Delete); +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h index 5df949883e62..ed0c25dab861 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h @@ -35,15 +35,19 @@ enum class CasNs : uint8_t }; static constexpr size_t CAS_NS_COUNT = 6; -/// Operation + outcome class (11 classes), mapped from the `Backend` method and its return value. -/// putIfAbsent → Done ⇒ Put ; PreconditionFailed ⇒ PutDeduplicated -/// putOverwrite → Done ⇒ Overwrite ; PreconditionFailed ⇒ CasConflict -/// casPut → Committed ⇒ Cas ; Conflict ⇒ CasConflict -/// head → exists ⇒ Head ; !exists ⇒ HeadMiss (the 404 signal) -/// get → Get (all calls, hit or miss) -/// getStream → GetStream (all calls, hit or miss) -/// deleteExact → Delete (all outcomes) -/// list → List +/// Operation + outcome class, mapped from the `Backend` primitive and its result. +/// write, no expected value → a value ⇒ Put ; RawConflict ⇒ PutDeduplicated +/// write, an expected value → a value ⇒ Overwrite ; RawConflict ⇒ CasConflict +/// head → present ⇒ Head ; absent ⇒ HeadMiss (the 404 signal) +/// read → Read (all calls, hit or miss) +/// stream → GetStream +/// remove → Delete (all outcomes) +/// list → List +/// publish → Put +/// +/// `Cas` has no current producer: the primitive `write` cannot tell a compare-and-set from any other +/// conditional replacement, so every conditional replace counts as `Overwrite`/`CasConflict`. Kept +/// for the `CAS*CompareSwap` events it still backs. enum class CasOp : uint8_t { Put = 0, @@ -53,7 +57,7 @@ enum class CasOp : uint8_t CasConflict, Head, HeadMiss, - Get, + Read, GetStream, Delete, List, @@ -74,14 +78,6 @@ void incrementCasEvent(CasNs ns, CasOp op); class InstrumentedBackend final : public Backend { public: - /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the - /// overrides below would otherwise shadow them for callers holding a concrete backend type. - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - explicit InstrumentedBackend(BackendPtr inner_) : inner(std::move(inner_)) {} /// Capability checks are deliberately uninstrumented: they do not represent storage operations. @@ -90,88 +86,82 @@ class InstrumentedBackend final : public Backend void checkConditionalWriteSingleAttemptSupport() override { inner->checkConditionalWriteSingleAttemptSupport(); } /// The typed sentinel probe is a diagnostic/authoritative read, not a routine storage operation — - /// deliberately uninstrumented (no ProfileEvent), like the capability checks above. MUST still be - /// forwarded explicitly: `Backend::probeSentinelRaw`'s generic default derives its classification from - /// THIS object's own `head`/`get` (virtual dispatch would otherwise resolve back to - /// `InstrumentedBackend`'s plain, non-typed overrides above), silently discarding whatever sharper - /// container/permission evidence the wrapped `inner` backend (e.g. `ObjectStorageBackend`'s S3/Local - /// classification) is able to provide. - SentinelProbeResult probeSentinelRaw(const String & key) override { return inner->probeSentinelRaw(key); } + /// deliberately uninstrumented (no ProfileEvent), like the capability checks above. + SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) override + { + return inner->probeSentinelRaw(key, access); + } /// Delegate the read and count it after the inner call succeeds or returns absent. Exceptions /// propagate unchanged and therefore do not produce a separate outcome event. - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - auto result = inner->get(key, range); - incrementCasEvent(classifyCasNs(key), CasOp::Get); + auto result = inner->read(key, access); + incrementCasEvent(classifyCasNs(key), CasOp::Read); return result; } - /// Delegate a forward-only read stream and count the request after the stream is acquired. - std::optional getStream(const String & key, Range range) override + /// Count `Head` or `HeadMiss` from the returned presence after delegating to the backend. + std::optional head(const String & key, TransportAccess & access) override { - auto result = inner->getStream(key, range); - incrementCasEvent(classifyCasNs(key), CasOp::GetStream); + auto result = inner->head(key, access); + incrementCasEvent(classifyCasNs(key), result ? CasOp::Head : CasOp::HeadMiss); return result; } - /// Count `Head` or `HeadMiss` from the returned presence flag after delegating to the backend. - HeadResult head(const String & key) override + /// Delegate one paginated listing and classify the prefix used for the request. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - HeadResult result = inner->head(key); - incrementCasEvent(classifyCasNs(key), result.exists ? CasOp::Head : CasOp::HeadMiss); - return result; + auto page = inner->list(prefix, cursor, limit, access); + incrementCasEvent(classifyCasNs(prefix), CasOp::List); + return page; } - /// Count a successful create as `Put` and an existing-key precondition result as `PutDeduplicated`. - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + /// Delegate the conditional removal and count every returned outcome as `Delete`. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - PutResult result = inner->putIfAbsent(key, bytes, meta); - incrementCasEvent(classifyCasNs(key), result.outcome == PutOutcome::Done ? CasOp::Put : CasOp::PutDeduplicated); - return result; + auto outcome = inner->remove(key, expected_value, access); + incrementCasEvent(classifyCasNs(key), CasOp::Delete); + return outcome; } - /// Count one successful physical blob publication after delegating exactly once. The backend has - /// no lifecycle reason to classify here; decision diagnostics remain with the writer. - void publishBlob(const BlobPublishRequest & request) override; - - /// Count a successful token-conditional overwrite as `Overwrite`; a precondition conflict is - /// counted as `CasConflict`. - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, - const ObjectMeta & meta) override + /// Delegate the batch removal, count the request once, and count each key it named as a `Delete` + /// in its own namespace class -- the per-key counters say how many keys of each class one request + /// carried, the request counter says how many requests it took. Out of line, like `publish`, so + /// the header need not declare the `CASBulkDeleteRequests` extern. + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override; + + /// Count a create and a replacement separately, and each of them separately from its refusal: + /// they cost the same one request, but a pool whose creates are mostly refused and one whose + /// replacements mostly conflict are different problems. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - PutResult result = inner->putOverwrite(key, bytes, expected, meta); - incrementCasEvent(classifyCasNs(key), result.outcome == PutOutcome::Done ? CasOp::Overwrite : CasOp::CasConflict); + auto result = inner->write(key, bytes, expected_value, access); + const CasOp op = expected_value ? (result ? CasOp::Overwrite : CasOp::CasConflict) + : (result ? CasOp::Put : CasOp::PutDeduplicated); + incrementCasEvent(classifyCasNs(key), op); return result; } - /// Count a committed compare-and-swap as `Cas`; conflicts are counted as `CasConflict`. - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// Delegate a forward-only read stream and count the request after the stream is acquired. + std::unique_ptr stream(const String & key, TransportAccess & access) override { - CasResult result = inner->casPut(key, bytes, expected, meta); - incrementCasEvent(classifyCasNs(key), result.outcome == CasOutcome::Committed ? CasOp::Cas : CasOp::CasConflict); + auto result = inner->stream(key, access); + incrementCasEvent(classifyCasNs(key), CasOp::GetStream); return result; } - /// Delegate token-exact deletion and count every returned deletion outcome as `Delete`. - DeleteOutcome deleteExact(const String & key, const Token & token) override - { - DeleteOutcome outcome = inner->deleteExact(key, token); - incrementCasEvent(classifyCasNs(key), CasOp::Delete); - return outcome; - } - - /// Delegate one paginated listing and classify the prefix used for the request. - ListPage list(const String & prefix, const String & cursor, size_t limit) override - { - ListPage page = inner->list(prefix, cursor, limit); - incrementCasEvent(classifyCasNs(prefix), CasOp::List); - return page; - } + /// Count one successful physical blob publication after delegating exactly once. The backend has + /// no lifecycle reason to classify here; decision diagnostics remain with the writer. + void publish(const BlobPublishRequest & request, TransportAccess & access) override; - /// This capability is a property of the wrapped backend, not an operation to count. + /// These are properties of the wrapped backend, not operations to count. + Dialect dialect() const override { return inner->dialect(); } bool supportsListTokens() const override { return inner->supportsListTokens(); } + uint64_t attemptTimeoutMs() const override { return inner->attemptTimeoutMs(); } + uint64_t attemptEnvelopeMs() const override { return inner->attemptEnvelopeMs(); } + bool refreshCredentials() override { return inner->refreshCredentials(); } private: BackendPtr inner; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp index b7b04fdb401b..0589929d2912 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp @@ -1,6 +1,7 @@ #include -#include +#include +#include #include #include #include @@ -10,12 +11,15 @@ #include #include #include +#include #include #include #include #include #include +#include +#include #include "config.h" @@ -27,6 +31,14 @@ #include #include +namespace ProfileEvents +{ + extern const Event CASConditionalWriteAttempts; + extern const Event CASConditionalWriteCommitted; + extern const Event CASConditionalWriteDefiniteFailure; + extern const Event CASConditionalWriteUnresolved; +} + namespace DB { namespace ErrorCodes @@ -34,47 +46,122 @@ namespace ErrorCodes extern const int CORRUPTED_DATA; extern const int FILE_DOESNT_EXIST; extern const int NOT_IMPLEMENTED; + extern const int LOGICAL_ERROR; + extern const int CAS_WRITE_UNATTRIBUTED; } } namespace DB::Cas { -ObjectStorageBackend::ObjectStorageBackend(ObjectStoragePtr object_storage_, Mode mode_) +namespace +{ + +/// Outcome of ONE HTTP attempt at a CAS conditional write, for the ProfileEvents below only -- +/// distinct from `detail::ConditionalWriteOutcome`, which the caller (`nativeConditionalPut`) acts on. +/// - Committed: the attempt's own request completed successfully (2xx). +/// - DefiniteFailure: a synchronous rejection that PROVES the request was never applied server-side +/// -- a WHITELISTED malformed-request / entity-too-large / access-denied error ONLY. +/// - Unresolved: everything else -- a lost precondition, a client-side timeout, a connection loss, a +/// 5xx, or any error this classifier does not recognize. +enum class CasWriteOutcome : uint8_t +{ + Committed, + DefiniteFailure, + Unresolved, +}; + +/// The exception path: classify what `buf.finalize()` threw for ONE CAS conditional-write HTTP +/// attempt. Never rethrows, never touches counters. +CasWriteOutcome classifyConditionalWriteResult([[maybe_unused]] const std::exception & e) +{ +#if USE_AWS_S3 + /// `PreconditionFailed`/`NoSuchKey` (a lost If-None-Match/If-Match), any 5xx + /// (InternalError/ServiceUnavailable/SlowDown/RequestTimeout), and any S3 error this function does + /// not recognize all fall through to the fail-safe default below: Unresolved. Only the WHITELIST + /// below proves the request was never applied. + if (const auto * s3e = dynamic_cast(&e)) + { + if (S3::isMalformedRequestError(*s3e) || S3::isEntityTooLargeError(*s3e) || S3::isAccessDeniedError(*s3e)) + return CasWriteOutcome::DefiniteFailure; + } +#endif + /// Poco::Net::NetException (connection loss) / Poco::TimeoutException (client-side timeout) and + /// every other error type: the request's fate is unproven -- fail toward "resolve before + /// reissuing", never toward a false DefiniteFailure. + return CasWriteOutcome::Unresolved; +} + +/// The success path: `buf.finalize()` returned without throwing. Always Committed -- kept as a named, +/// counted entry point so both paths of a classify-then-record call site read the same way. +constexpr CasWriteOutcome classifyConditionalWriteResult() +{ + return CasWriteOutcome::Committed; +} + +/// Records the start of one HTTP attempt for a CAS conditional write (the attempts counter). +void recordConditionalWriteAttemptStarted() +{ + ProfileEvents::increment(ProfileEvents::CASConditionalWriteAttempts); +} + +/// Records one attempt's terminal outcome (the per-class outcome counters). +void recordConditionalWriteOutcome(CasWriteOutcome outcome) +{ + switch (outcome) + { + case CasWriteOutcome::Committed: + ProfileEvents::increment(ProfileEvents::CASConditionalWriteCommitted); + return; + case CasWriteOutcome::DefiniteFailure: + ProfileEvents::increment(ProfileEvents::CASConditionalWriteDefiniteFailure); + return; + case CasWriteOutcome::Unresolved: + ProfileEvents::increment(ProfileEvents::CASConditionalWriteUnresolved); + return; + } +} + +} + +ObjectStorageBackend::ObjectStorageBackend(ObjectStoragePtr object_storage_, Mode mode_, + bool single_attempt_control_plane_, uint64_t attempt_timeout_ms_, + uint64_t connect_timeout_cap_ms_) : object_storage(std::move(object_storage_)) , mode(mode_) + , single_attempt_control_plane(single_attempt_control_plane_) + , attempt_timeout_ms(attempt_timeout_ms_) + , connect_timeout_cap_ms(connect_timeout_cap_ms_) , emu_root(object_storage->getCommonKeyPrefix()) { if (mode == Mode::Native && object_storage->conditionalOpsUseGenerationTokens()) - native_token_type = TokenType::Generation; + native_token_type = Dialect::Generation; } /// See Backend::checkPoolPreconditions. Only the Native, generation-dialect (GCS) combination has /// anything to check: a token-exact DELETE on a versioned bucket archives a noncurrent generation -/// instead of reclaiming storage, so GC "reclaim" would silently stop reclaiming. Both an enabled -/// bucket and an unverifiable probe refuse the mount. +/// instead of reclaiming storage, so GC "reclaim" would silently stop reclaiming. A bucket VERIFIED +/// to have versioning enabled refuses the mount. A probe that cannot answer does not: it is not +/// evidence of a versioned bucket, its usual cause is a credential without permission to read the +/// bucket configuration, and refusing on it would turn a missing IAM grant into a hard outage. The +/// mount proceeds with a warning that names what was not verified and how the operator can verify it. void ObjectStorageBackend::checkPoolPreconditions() { - if (mode != Mode::Native || native_token_type != TokenType::Generation) + if (mode != Mode::Native || native_token_type != Dialect::Generation) return; const auto versioned = object_storage->isBucketVersioningEnabled(); if (!versioned.has_value()) { - /// An unverifiable probe fails the mount, exactly like a confirmed Enabled below. Proceeding - /// on the ASSUMPTION that versioning is off was the earlier behaviour and it is not - /// defensible: what GC does on a versioned bucket is delete objects it believes it reclaimed, - /// so the assumption is silently wrong in precisely the case that matters, and it is wrong - /// without bound (a warning at mount does not stop the next round). The operator can prove - /// the bucket's state with one call and grant the permission the probe needs. - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + LOG_WARNING(getLogger("CasObjectStorageBackend"), "CAS on GCS: could not VERIFY the bucket-versioning precondition (the versioning check " - "request failed — e.g. the credential lacks permission to read it — or this backend " - "cannot answer it) — refusing to mount writable. CAS cannot assume versioning is off: if " - "it is actually enabled, token-exact DELETEs archive noncurrent generations instead of " - "reclaiming storage and GC silently stops reclaiming space. Grant the credential " - "permission to read the bucket's versioning configuration, confirm versioning is " - "disabled, and retry the mount."); + "request failed, e.g. the credential lacks permission to read the bucket configuration, " + "or this backend cannot answer it). Mounting anyway. If versioning IS enabled on this " + "bucket, token-exact DELETEs archive noncurrent generations instead of reclaiming storage " + "and GC silently stops reclaiming space. Confirm by hand that versioning is disabled, or " + "grant the credential permission to read the bucket's versioning configuration " + "(storage.buckets.get on GCS) so the next mount can verify it."); + return; } if (*versioned) @@ -91,7 +178,7 @@ void ObjectStorageBackend::checkPoolPreconditions() /// DELETE, and nothing else in the mount path proves it. void ObjectStorageBackend::checkSkipAccessCheckSupport() { - if (mode != Mode::Native || native_token_type != TokenType::Generation) + if (mode != Mode::Native || native_token_type != Dialect::Generation) return; throw Exception(ErrorCodes::NOT_IMPLEMENTED, @@ -131,32 +218,20 @@ void ObjectStorageBackend::checkConditionalWriteSingleAttemptSupport() /// Native helpers /// ========================================================================================= -bool ObjectStorageBackend::isValidGenerationTokenValue(const String & value) +bool ObjectStorageBackend::isValidTokenValue(Dialect type, const String & value) { - return !value.empty() && std::all_of(value.begin(), value.end(), [](char c) { return c >= '0' && c <= '9'; }); + return isIncarnationValue(type, value); } -std::optional ObjectStorageBackend::nativeHead(const String & key) +std::optional ObjectStorageBackend::nativeHead(const String & key, const ObjectStorageControlRequest & request) { - auto metadata = object_storage->tryGetObjectMetadataWithNativeToken(key, /*with_tags=*/false); + auto metadata = object_storage->tryGetObjectMetadataWithNativeToken(key, /*with_tags=*/false, request); if (!metadata) return std::nullopt; - HeadResult hr; - hr.exists = true; - hr.size = metadata->size_bytes; - hr.token = tokenForHead(metadata->etag); - /// A generation-token store guarantees a numeric x-goog-generation on every successful HEAD; - /// a missing or non-numeric value (a proxy dropping the header, a service regression) means the - /// ordinary ETag fell through unmapped. There is no follow-up HEAD to patch this over, so surface - /// the failure here rather than minting a token that would poison the first conditional operation - /// that trusts it -- exactly the contract tokenFromWriteResult already enforces on the write path. - if (native_token_type == TokenType::Generation && !isValidGenerationTokenValue(hr.token.value)) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS on GCS: a HEAD of {} succeeded but its response carried no valid generation ({})", - key, metadata->etag); - hr.attributes = ObjectMeta(metadata->attributes.begin(), metadata->attributes.end()); - return hr; + /// Normalized (a generation arrives quoted through the SDK's ETag field) and otherwise as the + /// store gave it. Whether it IS an incarnation is judged where the answer can be acted on. + return RawMeta{metadata->size_bytes, normalizeTokenValue(metadata->etag)}; } /// Finalize a conditional write (the condition rode on the buffer's WriteSettings) and map a @@ -177,7 +252,7 @@ std::optional ObjectStorageBackend::nativeHead(const String & key) /// coverage. Unit tests cover the emulated semantics, the typed exception path, and this classifier /// through the test-only `detail` declaration. #if USE_AWS_S3 -PutOutcome detail::finalizeConditionalWrite(WriteBuffer & buf) +detail::ConditionalWriteOutcome detail::finalizeConditionalWrite(WriteBuffer & buf) { try { @@ -188,38 +263,38 @@ PutOutcome detail::finalizeConditionalWrite(WriteBuffer & buf) if (e.isPreconditionFailed() || e.getExceptionName() == "NoSuchKey" || e.getS3ErrorCode() == Aws::S3::S3Errors::NO_SUCH_KEY) - return PutOutcome::PreconditionFailed; + return ConditionalWriteOutcome::PreconditionLost; throw; } - return PutOutcome::Done; + return ConditionalWriteOutcome::Applied; } #endif /// Build-dispatching shim for the write paths below: without the AWS SDK there is no S3Exception /// to classify, so the errors of finalize simply propagate. -static PutOutcome finalizeConditionalWrite(WriteBuffer & buf) +static detail::ConditionalWriteOutcome finalizeConditionalWrite(WriteBuffer & buf) { #if USE_AWS_S3 return detail::finalizeConditionalWrite(buf); #else buf.finalize(); - return PutOutcome::Done; + return detail::ConditionalWriteOutcome::Applied; #endif } /// Instrument the same single `finalize` call used by both Native write paths without changing their -/// `Done`/`PreconditionFailed`-or-rethrow contract. A classified precondition loss is `Unresolved`, -/// not `Committed` or a definite exception, because the response does not prove who created or -/// replaced the object; the higher-level request controller may then resolve it with exact-key state. -static PutOutcome finalizeConditionalWriteInstrumented(WriteBuffer & buf) +/// Applied/PreconditionLost-or-rethrow contract. A classified precondition loss is `Unresolved`, not +/// `Committed` or a definite exception, because the response does not prove who created or replaced +/// the object -- the caller's own retry loop resolves it with exact-key state. +static detail::ConditionalWriteOutcome finalizeConditionalWriteInstrumented(WriteBuffer & buf) { recordConditionalWriteAttemptStarted(); try { - const PutOutcome legacy = finalizeConditionalWrite(buf); + const detail::ConditionalWriteOutcome outcome = finalizeConditionalWrite(buf); recordConditionalWriteOutcome( - legacy == PutOutcome::Done ? classifyConditionalWriteResult() : CasWriteOutcome::Unresolved); - return legacy; + outcome == detail::ConditionalWriteOutcome::Applied ? classifyConditionalWriteResult() : CasWriteOutcome::Unresolved); + return outcome; } catch (const std::exception & e) { @@ -231,50 +306,42 @@ static PutOutcome finalizeConditionalWriteInstrumented(WriteBuffer & buf) /// Issue a conditional PUT (the condition rides on `ws`) and map a precondition loss — see /// finalizeConditionalWrite. The condition is checked by the backend when the object is completed, /// so the precondition loss always surfaces from the buffer's finalize, never from write. -PutResult ObjectStorageBackend::nativeConditionalPut(const String & key, const String & bytes, const WriteSettings & ws, const ObjectMeta & meta) +std::expected ObjectStorageBackend::nativeConditionalPut( + const String & key, const String & bytes, const WriteSettings & ws) { - std::optional attrs; - if (!meta.empty()) - attrs.emplace(meta.begin(), meta.end()); /// ObjectMeta is the same map type as ObjectAttributes auto buf = object_storage->writeObject( - StoredObject(key), WriteMode::Rewrite, attrs, DBMS_DEFAULT_BUFFER_SIZE, ws); + StoredObject(key), WriteMode::Rewrite, /*attributes=*/std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, ws); buf->write(bytes.data(), bytes.size()); - if (finalizeConditionalWriteInstrumented(*buf) == PutOutcome::PreconditionFailed) - return {PutOutcome::PreconditionFailed, {}}; - - /// Attribute the token of the incarnation WE just wrote (model WCreate) -- see - /// tokenFromWriteResult for the exact generation-vs-ETag policy. The S3 write returns its object - /// ETag/generation in the PutObject/CompleteMultipartUpload response, so no follow-up HEAD is - /// needed for most backends — this is ~73% of the CA backend's HEADs. - return {PutOutcome::Done, tokenFromWriteResult(key, buf->getResultObjectETag())}; -} - -namespace -{ - -/// Keep the emulated backend's publication memory bound to one materialized body at a time. -std::mutex & emulatedBlobPublicationMutex() -{ - static std::mutex mutex; - return mutex; -} - + if (finalizeConditionalWriteInstrumented(*buf) == detail::ConditionalWriteOutcome::PreconditionLost) + return std::unexpected(RawConflict{}); + + /// The response's own value for what it just wrote, normalized and otherwise untouched. An S3 + /// write carries its object ETag/generation in the PutObject/CompleteMultipartUpload response, so + /// no follow-up HEAD is needed -- and when it carries none, an empty value is the honest answer: + /// the write may have landed, which only the caller can resolve by reading the key back. + return normalizeTokenValue(buf->getResultObjectETag().value_or(String{})); } -/// True when an exception from `IObjectStorage::readObject` means "the object is simply not there". +/// True when an exception from a read means "the KEY is simply not there". /// Two surfaces: -/// 1. S3/RustFS: `S3Exception` with `S3Errors::NO_SUCH_KEY` (the modeled enum — the primary -/// signal) or `getExceptionName() == "NoSuchKey"` (the canonical XML `` string, present -/// when the SDK was able to parse it; mirrors `finalizeConditionalWrite`'s detection). +/// 1. S3/RustFS: `S3Exception` with `S3Errors::NO_SUCH_KEY` (the modeled enum — the primary +/// signal), `getExceptionName() == "NoSuchKey"` (the canonical XML `` string, present when +/// the SDK was able to parse it; mirrors `finalizeConditionalWrite`'s detection), or +/// `RESOURCE_NOT_FOUND`, the generic code the SDK derives from a 404 whose body it could not +/// parse into a name. /// 2. Local / emulated: `DB::Exception` with `ErrorCodes::FILE_DOESNT_EXIST` (from /// `ReadBufferFromFile` when `open(2)` returns ENOENT). /// +/// `NO_SUCH_BUCKET` is deliberately NOT here even though it is the third member of the store's own +/// 404 family: a vanished CONTAINER is not an absent key, and answering "absent" for it would let a +/// caller read an empty pool out of an outage. It propagates, and `probeSentinelRaw` classifies it. /// Any other error (network, auth, throttle, corruption) propagates unchanged — fail-closed. static bool isObjectNotFound(const std::exception & e) { #if USE_AWS_S3 if (const auto * s3e = dynamic_cast(&e)) return s3e->getS3ErrorCode() == Aws::S3::S3Errors::NO_SUCH_KEY + || s3e->getS3ErrorCode() == Aws::S3::S3Errors::RESOURCE_NOT_FOUND || s3e->getExceptionName() == "NoSuchKey"; #endif if (const auto * dbe = dynamic_cast(&e)) @@ -282,81 +349,17 @@ static bool isObjectNotFound(const std::exception & e) return false; } -/// Read `range` of the object at `path` as a TRUE ranged read: seek to the offset and bound the -/// read window. Seek the storage buffer to the requested offset and bound the returned bytes instead -/// of reading a whole snapshot run and slicing it afterward; snapshot runs can be gigabytes at scale, -/// while the caller's memory budget is O(block). -static String readObjectRanged(IObjectStorage & object_storage, const String & path, Range range, - uint64_t known_size = 0) +/// Read the whole object at `path`. A caller that already knows the size passes it so the read +/// buffer is sized to the body instead of the storage's ~1 MiB default. +static String readWholeObject(IObjectStorage & object_storage, const String & path, uint64_t known_size = 0) { auto buf = object_storage.readObject( StoredObject(path), casSizedReadSettings(getReadSettings(), known_size), /*read_hint=*/std::nullopt); String content; - if (range.whole()) - { - readStringUntilEOF(content, *buf); - return content; - } - - /// An offset at or past EOF yields an empty result, matching the range contract of the previous - /// whole-read implementation. - /// `seek` past the object size may throw depending on the storage, so fail-close the window - /// against the known size before touching the buffer position. - /// Native callers already HEAD the key, so passing its size avoids another metadata round trip. - /// A zero size means the caller does not know it and metadata must be fetched here. - const uint64_t object_size = known_size != 0 ? known_size - : object_storage.getObjectMetadata(path, /*with_tags=*/false).size_bytes; - if (range.offset >= object_size) - return {}; - - /// The readable window, clamped to EOF. `setReadUntilPosition` is only a hint (not every object - /// storage honors it — LocalObjectStorage does not), so the exact byte count below is what bounds - /// the read; the hint lets storages that DO honor it avoid over-fetching. - const uint64_t available = object_size - range.offset; - const uint64_t to_read = range.length.has_value() ? std::min(*range.length, available) : available; - - if (range.length.has_value()) - buf->setReadUntilPosition(range.offset + *range.length); - buf->seek(static_cast(range.offset), SEEK_SET); - - content.resize(to_read); - const size_t got = buf->read(content.data(), to_read); - content.resize(got); + readStringUntilEOF(content, *buf); return content; } -/// Open a forward-only stream over `range` of the object at `path`, positioned at the window's first -/// byte and bounded to its last. Mirrors -/// `readObjectRanged`'s seek + bound, but RETURNS the buffer instead of draining it — the caller reads -/// at its own pace, so nothing is materialized whole. Returns nullptr when the offset is at or past EOF -/// (the empty-window clamp), matching the ranged-get contract. -static std::unique_ptr openObjectRangedStream(IObjectStorage & object_storage, const String & path, Range range, - uint64_t known_size = 0) -{ - auto buf = object_storage.readObject( - StoredObject(path), casSizedReadSettings(getReadSettings(), known_size), /*read_hint=*/std::nullopt); - if (range.whole()) - return buf; - - /// Clamp exactly like `readObjectRanged`: an offset at or past EOF yields an empty stream, and - /// `seek` past the object size may throw depending on the storage, so fail-close against the known - /// size before touching the buffer position. - /// As in `readObjectRanged`, a caller-supplied size avoids another metadata round trip; zero means - /// that the size is unknown and must be fetched. - const uint64_t object_size = known_size != 0 ? known_size - : object_storage.getObjectMetadata(path, /*with_tags=*/false).size_bytes; - if (range.offset >= object_size) - return std::make_unique(std::string_view{}); - - /// `setReadUntilPosition` is only a hint (LocalObjectStorage does not honor it), but for a returned - /// stream it is the only bound available — the caller drains to EOF, so a storage that DOES honor - /// the hint stops at the window end, and one that does not over-reads only the trailing bytes. - if (range.length.has_value()) - buf->setReadUntilPosition(range.offset + *range.length); - buf->seek(static_cast(range.offset), SEEK_SET); - return buf; -} - ReadSettings casSizedReadSettings(const ReadSettings & base, uint64_t known_size) { if (known_size == 0) @@ -467,17 +470,14 @@ bool ObjectStorageBackend::emuExists(const String & key) const return object_storage->exists(StoredObject(emuPath(key))); } -String ObjectStorageBackend::emuRead(const String & key, Range range) const +String ObjectStorageBackend::emuRead(const String & key) const { - return readObjectRanged(*object_storage, emuPath(key), range); + return readWholeObject(*object_storage, emuPath(key)); } -Token ObjectStorageBackend::emuWrite(const String & key, const String & bytes, const ObjectMeta & meta) +String ObjectStorageBackend::emuWrite(const String & key, const String & bytes) { - std::optional attrs; - if (!meta.empty()) - attrs.emplace(meta.begin(), meta.end()); /// ObjectMeta is the same map type as ObjectAttributes - auto buf = object_storage->writeObject(StoredObject(emuPath(key)), WriteMode::Rewrite, attrs); + auto buf = object_storage->writeObject(StoredObject(emuPath(key)), WriteMode::Rewrite); buf->write(bytes.data(), bytes.size()); buf->finalize(); @@ -485,26 +485,42 @@ Token ObjectStorageBackend::emuWrite(const String & key, const String & bytes, c return emuMintToken(key, metadata ? metadata->etag : String{}, /*just_wrote=*/true); } -void ObjectStorageBackend::emuPublishBlobAtomically(const String & key, const String & bytes) +void ObjectStorageBackend::emuPublishBlobAtomically(const String & key, const String & envelope, ReadBuffer & payload, uint64_t payload_size) { if (object_storage->getType() != ObjectStorageType::Local) throw Exception( ErrorCodes::NOT_IMPLEMENTED, - "ObjectStorageBackend::publishBlob: atomic emulated publication requires local object storage"); + "ObjectStorageBackend::publish: atomic emulated publication requires local object storage"); const String destination_object = emuPath(key); const String temporary_object = destination_object + ".publish-" + toString(UUIDHelpers::generateV4()) + ".tmp"; const String root = object_storage->getCommonKeyPrefix(); const String destination_path = resolvePathRelativelyToBase(destination_object, root); const String temporary_path = resolvePathRelativelyToBase(temporary_object, root); - const auto existing_token_state = emu_token_state.find(key); + /// The body is STREAMED into the temporary file -- envelope, then a bounded copy of the payload -- + /// never materialized in memory. (An earlier revision accumulated envelope+payload in one String, + /// whose growth doubling made the peak allocation up to 2x the payload, and serialized every + /// publication behind a dedicated mutex just to bound that peak to one body at a time; streaming + /// removes both.) The destination stays untouched until the byte count has been validated: a short + /// or long source aborts on the temporary file, which is then removed. try { auto out = object_storage->writeObject(StoredObject(temporary_object), WriteMode::Rewrite); - out->write(bytes.data(), bytes.size()); + out->write(envelope.data(), envelope.size()); + const auto copy_result = blob_publication_detail::copyBlobPayloadBounded(payload, *out, payload_size); + if (!copy_result.exact(payload_size)) + { + out->cancel(); + throw Exception( + ErrorCodes::CORRUPTED_DATA, + "ObjectStorageBackend::publish: source yielded {}{} payload bytes for {}, declared {} -- nothing was published", + copy_result.has_excess ? "more than " : "", + copy_result.copied, + key, + payload_size); + } out->finalize(); - std::filesystem::rename(temporary_path, destination_path); } catch (...) { @@ -517,26 +533,49 @@ void ObjectStorageBackend::emuPublishBlobAtomically(const String & key, const St /// existing disambiguator is sufficient: if the next observation sees the same ETag, it returns /// a token distinct from the old incarnation; if the ETag changed, emuMintToken resets the state /// to that new ETag. With no existing state, this backend has issued no same-process stale token - /// that needs fencing. The post-rename increment cannot allocate or throw. + /// that needs fencing. The post-rename increment cannot allocate or throw. `emu_mutex` spans the + /// rename and the bump so a concurrent emulated observation never sees the new incarnation with + /// the old disambiguator. + std::lock_guard lock(emu_mutex); + const auto existing_token_state = emu_token_state.find(key); + try + { + std::filesystem::rename(temporary_path, destination_path); + } + catch (...) + { + std::error_code cleanup_error; + std::filesystem::remove(temporary_path, cleanup_error); + throw; + } if (existing_token_state != emu_token_state.end()) ++existing_token_state->second.second; } -Token ObjectStorageBackend::emuObserveToken(const String & key) +String ObjectStorageBackend::emuObserveToken(const String & key) { const auto metadata = object_storage->tryGetObjectMetadata(emuPath(key), /*with_tags=*/false); return emuMintToken(key, metadata ? metadata->etag : String{}, /*just_wrote=*/false); } -Token ObjectStorageBackend::emuMintToken(const String & key, const String & etag, bool just_wrote) +String ObjectStorageBackend::emuMintToken(const String & key, const String & etag, bool just_wrote) { emuPruneTokenState(emuNowNs()); - /// Anomalous: the object storage reported no etag at all (LocalObjectStorage always does; this - /// guards a hypothetical future/test double). Mint a fresh, UNPERSISTED value — never worse than - /// the old counter for this case, but never masquerading as a real etag-derived identity. + /// The object storage identified the object with nothing at all (LocalObjectStorage always + /// reports its mtime; this is the anomaly). There is no identity to hand out and none to invent: + /// a minted nonce would name an incarnation the store cannot recognise on the next conditional + /// request. A just-completed write is therefore unattributed -- it may have landed, and the + /// caller resolves that by reading back -- and an observation simply has nothing to report. if (etag.empty()) - return Token{std::to_string(++emu_seq), TokenType::Emulated}; + { + if (just_wrote) + throw Exception(ErrorCodes::CAS_WRITE_UNATTRIBUTED, + "CAS backend: the emulated store accepted a write of '{}' but reports no etag for it; " + "the write may have committed and must be resolved by reading back", key); + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS backend: the emulated store reports no etag for '{}', so it names no incarnation", key); + } auto it = emu_token_state.find(key); if (it != emu_token_state.end() && it->second.first == etag) @@ -548,57 +587,63 @@ Token ObjectStorageBackend::emuMintToken(const String & key, const String & etag /// tokens, so bump a small per-key disambiguator (mtime-quantum guard, triage §3.18 19c step 4). if (just_wrote) ++it->second.second; - const String value = it->second.second == 0 ? etag : etag + "#" + std::to_string(it->second.second); - return Token{value, TokenType::Emulated}; + return it->second.second == 0 ? etag : etag + "#" + std::to_string(it->second.second); } /// The etag advanced (or this key is seen for the first time): the bare etag is the token, and any /// previous disambiguator is dropped — a genuinely new incarnation starts clean. emu_token_state[key] = {etag, 0}; - return Token{etag, TokenType::Emulated}; + return etag; } /// ========================================================================================= /// Backend interface /// ========================================================================================= -std::optional ObjectStorageBackend::get(const String & key, Range range) +ReadSettings ObjectStorageBackend::readSettingsFor(const ObjectStorageControlRequest & request) const +{ + ReadSettings rs = getReadSettings(); + /// Mark the request for the store's native conditional dialect, so a GCS read is answered with a + /// generation rather than an MD5-shaped ETag. + rs.object_storage_request_mode = ObjectStorageRequestMode::NativeConditional; + rs.object_storage_retry_profile = request.profile; + rs.object_storage_attempt_timeout_ms = request.attempt_timeout_ms; + rs.object_storage_connect_timeout_cap_ms = request.connect_timeout_cap_ms; + rs.object_storage_attempt_number = request.attempt_number; + return rs; +} + +std::optional ObjectStorageBackend::read(const String & key, TransportAccess & access) +{ + return readUnder(key, controlRequest(access.attemptNo())); +} + +std::optional ObjectStorageBackend::readUnder(const String & key, const ObjectStorageControlRequest & request) { if (mode == Mode::Native) { - auto hr = nativeHead(key); - if (!hr) - return std::nullopt; - - /// The object may be deleted between the HEAD above and the GET below (a GC or concurrent - /// writer racing the read window). Catch the not-found signal and honor the `optional` - /// contract — callers such as `Pool::loadShardDecoded` already handle a nullopt return and - /// treat it as "raced a deletion, absent". Any other error (network, auth, corruption) - /// propagates unchanged — fail-closed by construction. - /// - /// A REPLACEMENT racing the same window (HEAD observes token A, GET reads the bytes of a - /// subsequently-written incarnation B) is likewise not a hazard: HEAD strictly precedes GET, so - /// the returned token is never NEWER than the returned bytes — a mixed pair is always - /// (bytes_newer, token_older), never the reverse. Every consumer of this token uses it as a - /// conditional precondition (`casPut`/`putOverwrite`/`deleteExact`), which fails closed EXACTLY - /// in the mixed case, so a stale token costs a retry, never lets a caller act on a - /// bytes/token pair that never coexisted. This also covers `known_size`: content-addressed blob - /// bodies are byte-identical across incarnations (a "replacement" only rotates envelope/token), - /// mutable control objects are read-modify-CAS loops that re-validate on conflict, and write-once - /// objects self-validate their contents on decode. - GetResult gr; + /// ONE request: an S3 GET answers with the incarnation of the bytes it returned, so no HEAD + /// is needed to name them and no HEAD-to-GET window exists in which the two could disagree. + /// The value is returned UNVALIDATED -- `CasRequests` is where a response value becomes an + /// incarnation, and it is the one place that can decide what a malformed one means. try { - gr.bytes = readObjectRanged(*object_storage, key, range, hr->size); + /// An absent key is an ordinary answer here, exactly as for the HEAD helpers this GET + /// replaced: without the scope the HTTP client logs the 404 at Error, and a stateless + /// test whose stderr is checked fails on the log line alone. + Expect404ResponseScope scope; + auto got = object_storage->readSmallObjectAndGetObjectMetadata( + StoredObject(key), readSettingsFor(request), casMaxStoredObjectBytes()); + return Raw{std::move(got.data), normalizeTokenValue(got.metadata.etag)}; } catch (const std::exception & e) { + /// The object is simply not there; every other error (network, auth, corruption) + /// propagates unchanged -- fail-closed by construction. if (isObjectNotFound(e)) return std::nullopt; throw; } - gr.token = hr->token; - return gr; } std::lock_guard lock(emu_mutex); @@ -609,10 +654,10 @@ std::optional ObjectStorageBackend::get(const String & key, Range ran /// caller in this process can delete the file in between. External deletion (e.g. a test teardown /// racing a read) is still handled: convert FILE_DOESNT_EXIST to nullopt rather than letting it /// escape as an unexplained exception. - GetResult gr; + Raw raw; try { - gr.bytes = emuRead(key, range); + raw.bytes = emuRead(key); } catch (const std::exception & e) { @@ -620,120 +665,101 @@ std::optional ObjectStorageBackend::get(const String & key, Range ran return std::nullopt; throw; } - gr.token = emuObserveToken(key); - return gr; + raw.value = emuObserveToken(key); + return raw; } -std::optional ObjectStorageBackend::getStream(const String & key, Range range) +std::unique_ptr ObjectStorageBackend::stream(const String & key, TransportAccess &) { - if (mode == Mode::Native) + /// No HEAD: this is ONE request, and the caller reserved one. Opening an object-storage buffer + /// issues nothing by itself, so the first GET is forced HERE -- otherwise the request that finds + /// the object absent, or fails, would happen after the call returned and be accounted to no + /// attempt at all. A present but empty object still yields a buffer; only a not-found is null. + /// + /// The buffer carries the storage's ORDINARY read settings, not this mount's control-plane + /// profile. `ReadBufferFromS3::nextImpl` re-reads `max_single_read_retries` on every `next()`, so + /// opening under SingleAttempt would strip the SDK's retries from the whole BODY -- which the + /// caller reads at its own pace, long after this attempt returned -- and stretch a single + /// attempt's timeout across the entire transfer. Only the open is the caller's to bound. Nor is + /// the request marked NativeConditional: a stream observes no incarnation to answer with. + try { - auto hr = nativeHead(key); - if (!hr) - return std::nullopt; - - /// Same HEAD-then-read race as `get`: the object may be deleted between the HEAD above and the - /// stream open below. Honor the `optional` contract on a not-found signal; any other error - /// (network, auth, corruption) propagates unchanged — fail-closed by construction. - GetStreamResult sr; - try - { - sr.stream = openObjectRangedStream(*object_storage, key, range, hr->size); - } - catch (const std::exception & e) + /// Same reason as `readUnder`: a not-found is an ordinary answer, not an error to log. + Expect404ResponseScope scope; + std::unique_ptr buf; + if (mode == Mode::Native) + buf = object_storage->readObject(StoredObject(key), getReadSettings(), /*read_hint=*/std::nullopt); + else { - if (isObjectNotFound(e)) - return std::nullopt; - throw; + /// Same lock the other emulated paths take, held across the open alone: there is no token + /// to observe here, so nothing else needs to be serialized with it. + std::lock_guard lock(emu_mutex); + buf = object_storage->readObject(StoredObject(emuPath(key)), getReadSettings(), /*read_hint=*/std::nullopt); } - sr.token = hr->token; - return sr; - } - - std::lock_guard lock(emu_mutex); - if (!emuExists(key)) - return std::nullopt; - - /// The emulated path holds emu_mutex across the exists-check and the stream open, matching `get`. - /// External deletion still converts to nullopt rather than escaping as an unexplained exception. - GetStreamResult sr; - try - { - sr.stream = openObjectRangedStream(*object_storage, emuPath(key), range); + buf->nextIfAtEnd(); + return buf; } catch (const std::exception & e) { if (isObjectNotFound(e)) - return std::nullopt; + return nullptr; throw; } - sr.token = emuObserveToken(key); - return sr; } -HeadResult ObjectStorageBackend::head(const String & key) +std::optional ObjectStorageBackend::head(const String & key, TransportAccess & access) +{ + return headUnder(key, controlRequest(access.attemptNo())); +} + +std::optional ObjectStorageBackend::headUnder(const String & key, const ObjectStorageControlRequest & request) { if (mode == Mode::Native) - { - auto hr = nativeHead(key); - return hr ? *hr : HeadResult{}; - } + return nativeHead(key, request); std::lock_guard lock(emu_mutex); if (!emuExists(key)) - return HeadResult{}; + return std::nullopt; auto metadata = object_storage->tryGetObjectMetadata(emuPath(key), /*with_tags=*/false); /// A path that exists on the Local filesystem but yields no object metadata is a directory, not /// an object (`tryGetObjectMetadata` returns nullopt for a directory). HEAD must report it as - /// not-an-object (exists=false) — otherwise existsFile/getStorageObjects treat a pool sub-dir (e.g. - /// `store`, traversed by system.remote_data_paths) as a file and a later body read throws EISDIR. + /// not-an-object — otherwise existsFile/getStorageObjects treat a pool sub-dir (e.g. `store`, + /// traversed by system.remote_data_paths) as a file and a later body read throws EISDIR. if (!metadata) - return HeadResult{}; - HeadResult hr; - hr.exists = true; - hr.size = metadata->size_bytes; - hr.attributes = ObjectMeta(metadata->attributes.begin(), metadata->attributes.end()); - hr.token = emuObserveToken(key); - return hr; + return std::nullopt; + return RawMeta{metadata->size_bytes, emuObserveToken(key)}; } /// See Backend::probeSentinelRaw / CasBackend.h's ProbeOutcome for the semantics this classifies. -SentinelProbeResult ObjectStorageBackend::probeSentinelRaw(const String & key) +SentinelProbeResult ObjectStorageBackend::probeSentinelRaw(const String & key, TransportAccess & access) +{ + return probeSentinelUnder(key, controlRequest(access.attemptNo())); +} + +SentinelProbeResult ObjectStorageBackend::probeSentinelUnder(const String & key, const ObjectStorageControlRequest & request) { if (mode == Mode::Native) { try { - /// `getObjectMetadata` (unlike `tryGetObjectMetadata`/`nativeHead`) is the THROWING raw-HEAD - /// primitive — it does NOT collapse NO_SUCH_KEY/NO_SUCH_BUCKET/RESOURCE_NOT_FOUND into one - /// `nullopt` before we get a chance to classify the S3 error. Its result is discarded here; - /// only whether (and how) it throws matters — the body comes from `get` below. - object_storage->getObjectMetadata(key, /*with_tags=*/false); - - /// The raw HEAD proved the key present. Delegate the body read to the existing `get`, which - /// already HEADs again and reads — an extra round trip this authoritative, low-rate probe can - /// afford, in exchange for reusing its already-correct HEAD→GET race handling. Kept INSIDE - /// this try: a transient failure here must also classify Indeterminate, never escape unclassified. - auto g = get(key); - if (!g) - return {ProbeOutcome::KeyAbsent, std::nullopt}; /// raced a deletion right after the raw HEAD - return {ProbeOutcome::Present, std::move(g->bytes)}; + /// One `read`: unlike a bodyless HEAD 404, a GET 404 carries a response body, so the + /// SDK can parse its `` and a missing key and a missing bucket arrive as different + /// errors -- which is the whole distinction this probe exists to make. + auto raw = readUnder(key, request); + if (!raw) + return {ProbeOutcome::KeyAbsent, std::nullopt}; + return {ProbeOutcome::Present, std::move(raw->bytes)}; } #if USE_AWS_S3 catch (const S3Exception & e) { + /// `read` already answers the key-absent half of the store's 404 family (see + /// `isObjectNotFound`), so what reaches here is what it deliberately does not flatten. + /// The limit of the one-read shape: every case below classifies an error a GET raised, so + /// a store whose HEAD and GET answer differently for the same key is classified by its GET. switch (e.getS3ErrorCode()) { - case Aws::S3::S3Errors::NO_SUCH_KEY: - return {ProbeOutcome::KeyAbsent, std::nullopt}; - case Aws::S3::S3Errors::RESOURCE_NOT_FOUND: - /// A HEAD response carries no body, so the SDK cannot parse a `NoSuchKey` `` - /// and instead derives this generic code straight from the HTTP 404 status (see - /// `isNotFoundError`, `src/IO/S3/getObjectInfo.cpp`) — this is what a REAL S3 HEAD - /// on an absent key actually throws. The container/key distinction is deliberately - /// NOT attempted here (a bodyless 404 cannot carry it). - return {ProbeOutcome::KeyAbsent, std::nullopt}; case Aws::S3::S3Errors::NO_SUCH_BUCKET: return {ProbeOutcome::ContainerAbsent, std::nullopt}; case Aws::S3::S3Errors::ACCESS_DENIED: @@ -751,7 +777,7 @@ SentinelProbeResult ObjectStorageBackend::probeSentinelRaw(const String & key) } } - /// EmulatedSingleProcess (Local): stat the configured container directory FIRST — `emuExists`/`get` + /// EmulatedSingleProcess (Local): stat the configured container directory FIRST — `emuExists`/`read` /// alone cannot distinguish "this key is absent" from "the whole pool directory is gone" (Local /// listing is best-effort and silently reports zero either way, see LocalObjectStorage::listObjects). try @@ -759,10 +785,10 @@ SentinelProbeResult ObjectStorageBackend::probeSentinelRaw(const String & key) if (!object_storage->existsOrHasAnyChild(emu_root)) return {ProbeOutcome::ContainerAbsent, std::nullopt}; - auto g = get(key); - if (!g) + auto raw = readUnder(key, request); + if (!raw) return {ProbeOutcome::KeyAbsent, std::nullopt}; - return {ProbeOutcome::Present, std::move(g->bytes)}; + return {ProbeOutcome::Present, std::move(raw->bytes)}; } catch (...) { @@ -776,11 +802,11 @@ SentinelProbeResult ObjectStorageBackend::probeSentinelRaw(const String & key) /// unconditional multipart-capable write. CAS-mutable keys (shard manifests, gc/state, the registry) /// also skip the racy post-upload existence/size check; a publish's manifest CAS was observed racing /// the GC fence there. -WriteSettings ObjectStorageBackend::conditionalWriteSettings() const +WriteSettings ObjectStorageBackend::conditionalWriteSettings(size_t attempt_no) const { WriteSettings ws; ws.object_storage_request_mode = ObjectStorageRequestMode::NativeConditional; - if (native_token_type == TokenType::Generation) + if (native_token_type == Dialect::Generation) ws.s3_force_single_part_upload = true; ws.s3_check_objects_after_upload_override = false; /// Exactly one attempt at the WriteBufferFromS3 layer too: makeSinglepartUpload/ @@ -792,74 +818,56 @@ WriteSettings ObjectStorageBackend::conditionalWriteSettings() const /// profile to its own single-attempt client. A backend that cannot honor it is rejected for /// writable Native mounts by checkConditionalWriteSingleAttemptSupport (fail closed). ws.object_storage_retry_profile = ObjectStorageRetryProfile::SingleAttempt; + /// And that attempt is bounded by the same budget the caller reserved for it. Without this the + /// write would run under the storage's own timeout while its caller waited on a shorter one. + ws.object_storage_attempt_timeout_ms = attempt_timeout_ms; + /// And that attempt's connect is bounded by the same frozen cap every other control request carries. + ws.object_storage_connect_timeout_cap_ms = connect_timeout_cap_ms; + /// The caller's own physical-attempt count, so the HTTP client sees a reissue as attempt >= 2. + ws.object_storage_attempt_number = attempt_no; return ws; } -/// See the declaration in the header for the policy. Centralizes the generation-vs-ETag attribution -/// decision for all successful conditional non-blob writes, including create-if-absent artifacts -/// and conditional replacements. -/// -/// The strict Generation-dialect check below is gated on `etag.has_value()`, not merely on -/// `native_token_type`: `WriteBufferFromS3` unconditionally assigns `object_etag = outcome.GetResult().GetETag()` -/// on BOTH of its success paths -- `makeSinglepartUpload` (WriteBufferFromS3.cpp) and -/// `completeMultipartUpload` (WriteBufferFromS3.cpp) -- so a successful S3 write always leaves -/// `getResultObjectETag()` holding a value, empty string included; `has_value()` is exactly "this was -/// a real S3-style write response", the only case Step 7's "a missing x-goog-generation is an -/// exception" rule is ABOUT. `S3ObjectStorage::writeObject` returns that `WriteBufferFromS3` directly, -/// undecorated, so this holds for the whole CAS-over-S3 write path with no wrapping in between. A -/// backend with no write-time-token concept at all (local files, or a non-S3 `IObjectStorage` -/// exercising Generation dialect purely for a unit test, see -/// `CASBackendGeneration.StampedTokenTypeFollowsNativeKind`) reports `nullopt` structurally, not a -/// broken response, and keeps falling back to a fresh HEAD exactly like the ETag dialect. A future -/// change that wraps the returned write buffer in a decorator would need to re-derive or preserve this -/// chain -- `WriteBufferFromFileDecorator::getResultObjectETag` returns `nullopt` for a wrapped impl -/// that is not itself a `WriteBufferFromFileBase`, which would silently turn a hard failure back into -/// a HEAD fallback. -/// -Token ObjectStorageBackend::tokenFromWriteResult(const String & key, const std::optional & etag) +std::expected ObjectStorageBackend::write( + const String & key, const String & bytes, const std::optional & expected_value, TransportAccess & access) { - if (native_token_type == TokenType::Generation && etag.has_value()) - { - /// Validate the MINTED value, not the raw one: the HTTP boundary presents the generation - /// through the SDK's ETag field and therefore quotes it, and `tokenForHead` is what strips - /// that transport syntax. Validating before the strip would reject every real GCS write. - /// The message still reports the raw arrival, since that is what needs diagnosing. - const Token token = tokenForHead(*etag); - if (!isValidGenerationTokenValue(token.value)) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS on GCS: a conditional write to {} succeeded but its response carried no " - "valid generation ({}) -- there is no follow-up HEAD to patch this over, so the write " - "cannot be attributed to an incarnation", - key, *etag); - return token; - } - - /// ETag dialect (and any backend with no write-time token at all, e.g. local files): unchanged - /// pre-existing behavior -- an absent/empty value falls back to a fresh HEAD of `key`. - if (etag && !etag->empty()) - return tokenForHead(*etag); + /// An empty, wildcard or list value would turn the precondition into an unconditional write -- + /// refuse it as a caller bug before anything else runs. + if (expected_value && !isValidTokenValue(dialect(), *expected_value)) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS backend: refusing a conditional mutation of '{}' with a malformed token '{}' (dialect {}): " + "an empty, wildcard or list token would turn the precondition into an unconditional write", + key, *expected_value, static_cast(dialect())); - auto hr = nativeHead(key); - return hr ? hr->token : Token{}; -} - -PutResult ObjectStorageBackend::putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) -{ if (mode == Mode::Native) { - WriteSettings ws = conditionalWriteSettings(); - ws.object_storage_write_if_none_match = "*"; - return nativeConditionalPut(key, bytes, ws, meta); + WriteSettings ws = conditionalWriteSettings(access.attemptNo()); + if (expected_value) + ws.object_storage_write_if_match = *expected_value; + else + ws.object_storage_write_if_none_match = "*"; + return nativeConditionalPut(key, bytes, ws); } std::lock_guard lock(emu_mutex); - if (emuExists(key)) - return {PutOutcome::PreconditionFailed, {}}; + const bool exists = emuExists(key); + if (!expected_value) + { + if (exists) + return std::unexpected(RawConflict{}); + } + else + { + if (!exists) + return std::unexpected(RawConflict{}); + if (emuObserveToken(key) != *expected_value) + return std::unexpected(RawConflict{}); + } - return {PutOutcome::Done, emuWrite(key, bytes, meta)}; + return emuWrite(key, bytes); } -void ObjectStorageBackend::publishBlob(const BlobPublishRequest & request) +void ObjectStorageBackend::publish(const BlobPublishRequest & request, TransportAccess &) { if (const auto * streaming = std::get_if(&request.publication)) { @@ -867,37 +875,14 @@ void ObjectStorageBackend::publishBlob(const BlobPublishRequest & request) if (!payload) throw Exception( ErrorCodes::CORRUPTED_DATA, - "ObjectStorageBackend::publishBlob: payload source for {} returned no reader", + "ObjectStorageBackend::publish: payload source for {} returned no reader", request.destination_key); if (mode != Mode::Native) { - /// The emulated adapter's writes are whole-body operations. Serialize materialization so - /// concurrent publications retain the existing one-body peak-memory bound. - std::lock_guard publish_lock(emulatedBlobPublicationMutex()); - - String body = streaming->fresh_envelope; - blob_publication_detail::BlobPayloadCopyResult copy_result; - { - WriteBufferFromString out(body, AppendModeTag{}); - copy_result = blob_publication_detail::copyBlobPayloadBounded(*payload, out, streaming->payload_size); - if (copy_result.exact(streaming->payload_size)) - out.finalize(); - else - out.cancel(); - } - - if (!copy_result.exact(streaming->payload_size)) - throw Exception( - ErrorCodes::CORRUPTED_DATA, - "ObjectStorageBackend::publishBlob: source yielded {}{} payload bytes for {}, declared {} -- nothing was published", - copy_result.has_excess ? "more than " : "", - copy_result.copied, - request.destination_key, - streaming->payload_size); - - std::lock_guard lock(emu_mutex); - emuPublishBlobAtomically(request.destination_key, body); + /// Streams straight into the temporary file and renames -- see emuPublishBlobAtomically. + emuPublishBlobAtomically( + request.destination_key, streaming->fresh_envelope, *payload, streaming->payload_size); return; } @@ -925,7 +910,7 @@ void ObjectStorageBackend::publishBlob(const BlobPublishRequest & request) out->cancel(); throw Exception( ErrorCodes::CORRUPTED_DATA, - "ObjectStorageBackend::publishBlob: source yielded {}{} payload bytes for {}, declared {} -- upload aborted, nothing published", + "ObjectStorageBackend::publish: source yielded {}{} payload bytes for {}, declared {} -- upload aborted, nothing published", copy_result.has_excess ? "more than " : "", copy_result.copied, request.destination_key, @@ -939,14 +924,14 @@ void ObjectStorageBackend::publishBlob(const BlobPublishRequest & request) if (mode != Mode::Native) throw Exception( ErrorCodes::NOT_IMPLEMENTED, - "ObjectStorageBackend::publishBlob: verbatim staged publication requires Native mode"); + "ObjectStorageBackend::publish: verbatim staged publication requires Native mode"); WriteSettings write_settings; write_settings.object_storage_copy_mode = ObjectStorageCopyMode::NativeOnly; if (!object_storage->supportsCopyMode(write_settings.object_storage_copy_mode)) throw Exception( ErrorCodes::NOT_IMPLEMENTED, - "ObjectStorageBackend::publishBlob: object storage {} does not support native-only same-store copy", + "ObjectStorageBackend::publish: object storage {} does not support native-only same-store copy", object_storage->getName()); object_storage->copyObject( @@ -956,123 +941,58 @@ void ObjectStorageBackend::publishBlob(const BlobPublishRequest & request) write_settings); } -PutResult ObjectStorageBackend::putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -{ - /// §3.18 №19: reject a wrong-dialect expected token before it ever reaches the wire (Native) or - /// the emu compare (Emulated) — see mintingTypeMatches. - if (!mintingTypeMatches(expected.type)) - return {PutOutcome::PreconditionFailed, {}}; - - if (mode == Mode::Native) - { - WriteSettings ws = conditionalWriteSettings(); - ws.object_storage_write_if_match = expected.value; - return nativeConditionalPut(key, bytes, ws, meta); - } - - std::lock_guard lock(emu_mutex); - if (!emuExists(key)) - return {PutOutcome::PreconditionFailed, {}}; - if (!tokenMatches(emuObserveToken(key), expected)) - return {PutOutcome::PreconditionFailed, {}}; - - return {PutOutcome::Done, emuWrite(key, bytes, meta)}; -} - -CasResult ObjectStorageBackend::casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) +Backend::RawRemoval ObjectStorageBackend::remove(const String & key, const String & expected_value, TransportAccess & access) { - /// §3.18 №19: a create-if-absent CAS (expected == nullopt) has no token to validate; only the - /// swap form carries one, and it must match this backend's own minting dialect before anything - /// else runs. - if (expected.has_value() && !mintingTypeMatches(expected->type)) - return {CasOutcome::Conflict, {}}; - - if (mode == Mode::Native) - { - WriteSettings ws = conditionalWriteSettings(); - if (expected.has_value()) - ws.object_storage_write_if_match = expected->value; - else - ws.object_storage_write_if_none_match = "*"; - - /// The PUT-side outcomes (Done / PreconditionFailed) collapse onto CAS outcomes 1:1: a lost - /// condition — whether a mismatched If-Match or a 404 on an If-Match PUT — is a Conflict. - PutResult put = nativeConditionalPut(key, bytes, ws, meta); - return put.outcome == PutOutcome::Done - ? CasResult{CasOutcome::Committed, put.token} - : CasResult{CasOutcome::Conflict, {}}; - } - - std::lock_guard lock(emu_mutex); - const bool exists = emuExists(key); - - if (!expected.has_value()) - { - if (exists) - return {CasOutcome::Conflict, {}}; - } - else - { - if (!exists) - return {CasOutcome::Conflict, {}}; - if (!tokenMatches(emuObserveToken(key), *expected)) - return {CasOutcome::Conflict, {}}; - } - - return {CasOutcome::Committed, emuWrite(key, bytes, meta)}; + return removeUnder(key, expected_value, controlRequest(access.attemptNo())); } -DeleteOutcome ObjectStorageBackend::deleteExact(const String & key, const Token & token) +Backend::RawRemoval ObjectStorageBackend::removeUnder( + const String & key, const String & expected_value, const ObjectStorageControlRequest & request) { - /// §3.18 №19: same local dialect guard as putOverwrite/casPut — never forward a foreign-dialect - /// value as the removeObjectIfTokenMatches argument. - if (!mintingTypeMatches(token.type)) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; - } + /// Same grammar guard as `write`, and for the same reason: an empty, wildcard or list value would + /// turn the condition into an unconditional delete. + if (!isValidTokenValue(dialect(), expected_value)) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS backend: refusing a conditional mutation of '{}' with a malformed token '{}' (dialect {}): " + "an empty, wildcard or list token would turn the precondition into an unconditional write", + key, expected_value, static_cast(dialect())); if (mode == Mode::Native) { - /// `removeObjectIfTokenMatches` maps onto `DeleteOutcome` one-to-one. `NOT_IMPLEMENTED` from a - /// backend that does not enforce conditional removal propagates — fail-closed by construction. - auto result = object_storage->removeObjectIfTokenMatches(StoredObject(key), token.value); - DeleteOutcome d; - d.created_delete_marker = result.created_delete_marker; + /// `NOT_IMPLEMENTED` from a storage that does not enforce conditional removal propagates — + /// fail-closed by construction. + auto result = object_storage->removeObjectIfTokenMatches(StoredObject(key), expected_value, request); switch (result.outcome) { case ConditionalRemoveOutcome::Removed: - d.kind = DeleteOutcome::Kind::Deleted; - break; + /// A delete marker means the storage archived a noncurrent version instead of + /// reclaiming the current object -- a removal that did not reclaim. + return result.created_delete_marker ? RawRemoval::DeleteMarker : RawRemoval::Removed; case ConditionalRemoveOutcome::TokenMismatch: - d.kind = DeleteOutcome::Kind::TokenMismatch; - break; + return RawRemoval::Mismatch; case ConditionalRemoveOutcome::NotFound: - d.kind = DeleteOutcome::Kind::NotFound; - break; + return RawRemoval::Gone; } - return d; + UNREACHABLE(); } std::lock_guard lock(emu_mutex); - DeleteOutcome d; if (!emuExists(key)) - { - d.kind = DeleteOutcome::Kind::NotFound; - return d; - } - if (!tokenMatches(emuObserveToken(key), token)) - { - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; - } + return RawRemoval::Gone; + if (emuObserveToken(key) != expected_value) + return RawRemoval::Mismatch; object_storage->removeObjectIfExists(StoredObject(emuPath(key))); + emuForgetDeletedToken(key); + return RawRemoval::Removed; +} + +void ObjectStorageBackend::emuForgetDeletedToken(const String & key) +{ /// Keep the deleted incarnation's last-minted etag around ONLY while a same-mtime-quantum /// collision with an immediate recreate is still possible (emuMintToken) — once it is /// comfortably old, erase it so `emu_token_state` does not grow for the lifetime of the backend - /// instance (codex-review-triage §3.18, Important #1). + /// instance. if (auto it = emu_token_state.find(key); it != emu_token_state.end()) { const uint64_t now_ns = emuNowNs(); @@ -1081,11 +1001,42 @@ DeleteOutcome ObjectStorageBackend::deleteExact(const String & key, const Token else emu_token_expiry.push_back(EmuTokenExpiry{now_ns, key, it->second}); } - d.kind = DeleteOutcome::Kind::Deleted; - return d; } -ListPage ObjectStorageBackend::list(const String & prefix, const String & cursor, size_t limit) +void ObjectStorageBackend::removeManyWriteOnce(const std::vector & keys, TransportAccess & access) +{ + if (keys.empty()) + return; + if (mode == Mode::Native) + { + StoredObjects objects; + objects.reserve(keys.size()); + for (const WriteOnceKey & key : keys) + objects.emplace_back(key.str()); + /// `NOT_IMPLEMENTED` from a storage without a batch delete propagates -- this layer never + /// substitutes a per-key loop of its own (that would run under a single admission for up to + /// 1000 keys). The caller in CasGc.cpp catches it and retries one key per admitted request. + object_storage->removeObjectsIfExistUnderProfile(objects, controlRequest(access.attemptNo())); + return; + } + + std::lock_guard lock(emu_mutex); + for (const WriteOnceKey & key : keys) + { + if (!emuExists(key.str())) + continue; + object_storage->removeObjectIfExists(StoredObject(emuPath(key.str()))); + emuForgetDeletedToken(key.str()); + } +} + +Backend::RawListPage ObjectStorageBackend::list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) +{ + return listUnder(prefix, cursor, limit, controlRequest(access.attemptNo())); +} + +Backend::RawListPage ObjectStorageBackend::listUnder( + const String & prefix, const String & cursor, size_t limit, const ObjectStorageControlRequest & request) { /// Use the lazy object-storage iterator instead of `listObjects(..., max_keys=0)`: the latter /// materialized the whole prefix, then sliced client-side, so a paginated walk re-fetched the full @@ -1104,32 +1055,29 @@ ListPage ObjectStorageBackend::list(const String & prefix, const String & cursor object_storage->listObjects(physical_prefix, children, /*max_keys=*/0); /// Hold emu_mutex across the whole scan: emuMintToken below reads/updates emu_token_state, the - /// same per-key state get/head/put*/delete* mutate under this lock (see the "caller holds + /// same per-key state read/head/write/remove mutate under this lock (see the "caller holds /// emu_mutex" contract on the private emu* helpers). std::lock_guard lock(emu_mutex); - std::vector all; + std::vector all; all.reserve(children.size()); for (const auto & child : children) { if (!child->relative_path.starts_with(physical_prefix)) continue; - ListedKey lk; + RawListedKey lk; lk.key = child->relative_path.substr(strip.size()); lk.size = child->metadata ? child->metadata->size_bytes : 0; - /// §3.18 №18: mint DIRECTLY as TokenType::Emulated — do NOT call tokenForList, which always - /// stamps native_token_type (ETag/Generation) regardless of mode and would surface a token - /// of the wrong dialect for every Emulated consumer (head/get mint Emulated). if (child->metadata) - lk.token = emuMintToken(lk.key, child->metadata->etag, /*just_wrote=*/false); + lk.value = emuMintToken(lk.key, child->metadata->etag, /*just_wrote=*/false); all.push_back(std::move(lk)); } - std::sort(all.begin(), all.end(), [](const ListedKey & a, const ListedKey & b) { return a.key < b.key; }); + std::sort(all.begin(), all.end(), [](const RawListedKey & a, const RawListedKey & b) { return a.key < b.key; }); - ListPage page; + RawListPage page; auto all_it = cursor.empty() - ? std::lower_bound(all.begin(), all.end(), prefix, [](const ListedKey & a, const String & s) { return a.key < s; }) - : std::upper_bound(all.begin(), all.end(), cursor, [](const String & s, const ListedKey & a) { return s < a.key; }); + ? std::lower_bound(all.begin(), all.end(), prefix, [](const RawListedKey & a, const String & s) { return a.key < s; }) + : std::upper_bound(all.begin(), all.end(), cursor, [](const String & s, const RawListedKey & a) { return s < a.key; }); while (all_it != all.end() && page.keys.size() < limit) { page.keys.push_back(*all_it); @@ -1144,26 +1092,39 @@ ListPage ObjectStorageBackend::list(const String & prefix, const String & cursor ? std::nullopt : std::optional(cursor); - ListPage page; - auto it = object_storage->iterate(physical_prefix, /*max_keys=*/0, /*with_tags=*/false, start_after); + /// The FIRST page the caller asked for is the page the store is asked for: a request that + /// enumerates the storage's default page (a thousand keys) to answer a caller that wanted a + /// handful spends the caller's whole attempt on keys it will never read -- and on a large prefix + /// that enumeration is exactly what a slow store cannot deliver within an attempt. One key MORE + /// than the page, because not every storage pages: the fallback iterator lists `max_keys` keys + /// once and ends, and a page that ends exactly at the limit would then be indistinguishable from + /// the end of the prefix -- the extra key is what proves there is more, whichever iterator + /// answers. A RESUMED page is not bounded: a storage that ignores `start_after` would answer it + /// with the first keys of the prefix again, and a bound there would drop every key past it. The + /// first page is the one the emptiness probe needs; a walk keeps the storage's own page size. + static constexpr size_t max_store_page = 1'000'000; + const size_t store_page = cursor.empty() ? std::min(limit, max_store_page) + 1 : 0; + RawListPage page; + auto it = object_storage->iterate(physical_prefix, /*max_keys=*/store_page, /*with_tags=*/false, start_after, request); for (; it->isValid(); it->next()) { const auto child = it->current(); if (!child->relative_path.starts_with(physical_prefix)) continue; - ListedKey lk; + RawListedKey lk; lk.key = child->relative_path.substr(strip.size()); if (!cursor.empty() && lk.key <= cursor) continue; lk.size = child->metadata ? child->metadata->size_bytes : 0; - /// Surface the per-key incarnation token (matching what `head` would return, see above) so the - /// `supportsListTokens() == true` capability is honest. A listing without an etag leaves the - /// token unset, which GC discover treats as Read (fail closed). The supportsListTokens()+ - /// empty-etag gate now lives in tokenForList. + /// Surface the per-key incarnation value (matching what `head` would return) so the + /// `supportsListTokens() == true` capability is honest. A listing without an etag leaves it + /// unset, which GC discover treats as Read (fail closed). The supportsListTokens()+ + /// empty-etag gate lives in tokenForList; whether the value it passes IS an incarnation is + /// judged where the answer can be acted on, not here. if (child->metadata) - lk.token = tokenForList(child->metadata->etag); + lk.value = tokenForList(child->metadata->etag); if (page.keys.size() == limit) { @@ -1176,4 +1137,21 @@ ListPage ObjectStorageBackend::list(const String & prefix, const String & cursor return page; } +void ensureBackendMatchesBudget(const ObjectStorageBackend & backend, const CasRequestBudget & budget) +{ + const uint64_t budget_connect_cap = budget.connect_timeout_cap_ms.value_or(0); + if (backend.connectTimeoutCapMs() != budget_connect_cap) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS: backend connect timeout cap ({} ms) does not match the pool's request budget ({} ms); " + "the mount's lease arithmetic was validated against the budget alone, so a backend built with " + "another cap could silently outlive it", + backend.connectTimeoutCapMs(), budget_connect_cap); + if (backend.attemptTimeoutMs() != budget.attempt_timeout_ms) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS: backend attempt timeout ({} ms) does not match the pool's request budget ({} ms); " + "the mount's lease arithmetic was validated against the budget alone, so a backend built with " + "another timeout could silently outlive it", + backend.attemptTimeoutMs(), budget.attempt_timeout_ms); +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h index bd369d4e3603..f9f51396efbb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h @@ -1,5 +1,6 @@ #pragma once #include +#include #include #include #include @@ -18,16 +19,22 @@ namespace DB::Cas constexpr uint64_t CAS_FOLD_READ_SLACK_BYTES = 4096; ReadSettings casSizedReadSettings(const ReadSettings & base, uint64_t known_size); -#if USE_AWS_S3 namespace detail { +/// Outcome of ONE conditional-write attempt at the transport level: did the store apply it, or did it +/// refuse the precondition (If-None-Match/If-Match)? Distinct from `Backend::RawConflict` because a +/// caller inside this TU needs to know WHICH of "committed" or "precondition lost" happened before it +/// decides what to return, whereas `RawConflict` only ever means the latter. +enum class ConditionalWriteOutcome : uint8_t { Applied, PreconditionLost }; + +#if USE_AWS_S3 /// Finalize a conditional write (the condition rode on the buffer's WriteSettings) and map a /// precondition loss to an OUTCOME — anything else propagates. This is the classifier for the /// typed `S3Exception` signal; exposed here for unit tests only — production callers go through /// `ObjectStorageBackend`. See the definition for the exact matching rules. -PutOutcome finalizeConditionalWrite(WriteBuffer & buf); -} +ConditionalWriteOutcome finalizeConditionalWrite(WriteBuffer & buf); #endif +} /// Production Backend over IObjectStorage. /// @@ -48,31 +55,70 @@ PutOutcome finalizeConditionalWrite(WriteBuffer & buf); class ObjectStorageBackend final : public Backend { public: - /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the - /// overrides below would otherwise shadow them for callers holding a concrete backend type. - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - enum class Mode { Native, EmulatedSingleProcess }; /// Construct a backend over `object_storage`. Native mode uses the storage's conditional /// operations and native token dialect; `EmulatedSingleProcess` serializes operations locally for /// tests and local development. A Native generation-token store must use a single PUT because /// its multipart completion path does not enforce the precondition. - ObjectStorageBackend(ObjectStoragePtr object_storage_, Mode mode_); + /// + /// `single_attempt_control_plane` selects the SingleAttempt retry profile for the READ-class + /// requests below: a writable Native mount owns its own retry policy and a transparently retried + /// request would outlive the caller's deadline, while a read-only mount has no such deadline and + /// keeps the storage's default. `attempt_timeout_ms` bounds ONE attempt of those requests; 0 + /// leaves the storage's own timeout in place. `connect_timeout_cap_ms` caps the connect portion of + /// that same attempt (0 = no cap), frozen by the mount at open. All three are supplied by the mount + /// that opens the pool; the defaults are what a narrow unit test constructing a bare backend gets. + ObjectStorageBackend(ObjectStoragePtr object_storage_, Mode mode_, + bool single_attempt_control_plane_ = false, uint64_t attempt_timeout_ms_ = 0, + uint64_t connect_timeout_cap_ms_ = 0); - /// Read an object or return `nullopt` if it is absent. Native mode HEADs first so the returned - /// token identifies the incarnation whose bytes are read; a not-found race is also reported as + /// Read the whole object, or return `nullopt` if it is absent. Native mode reads the incarnation + /// value out of the GET response itself, so no HEAD precedes it; a not-found race is reported as /// `nullopt`, while unrelated storage errors propagate. - std::optional get(const String & key, Range range) override; - /// Open a forward-only ranged stream for a write-once object. The stream is not materialized in - /// memory; mutable objects must use `get` because their contents may change while it is open. - std::optional getStream(const String & key, Range range) override; - /// Return the current size, attributes, and incarnation token, or an absent `HeadResult`. - HeadResult head(const String & key) override; + std::optional read(const String & key, TransportAccess & access) override; + /// Return the current size and incarnation value, or `nullopt` when the key is absent. + std::optional head(const String & key, TransportAccess & access) override; + /// Return a page after `cursor`; the next cursor is the last returned key and is empty at the end. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override; + /// Remove only the incarnation named by `expected_value`, preserving the object on a mismatch and + /// reporting a versioned bucket's delete marker as the non-reclaiming removal it is. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override; + /// Create the key (`expected_value == nullopt`) or replace exactly the incarnation it names. The + /// returned value is the write response's own; a response that carries none at all is + /// `CAS_WRITE_UNATTRIBUTED`, never patched over by a follow-up read. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override; + /// Open a forward-only whole-object stream for a write-once object, or null when the key is + /// absent. Nothing is materialized; mutable objects must use `read` because their contents may + /// change while the stream is open. + std::unique_ptr stream(const String & key, TransportAccess & access) override; + /// Execute the selected unconditional blob transport without observing destination state or + /// returning a write-response value. Streaming uses ordinary write settings; staged bytes require + /// a native same-store copy. + void publish(const BlobPublishRequest & request, TransportAccess & access) override; + /// Removes up to 1000 write-once keys in one request. Native mode issues one `DeleteObjects` + /// under the control-plane profile; `EmulatedSingleProcess` deletes each present key under the + /// emulation lock with the same token bookkeeping as the single-key delete. + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override; + /// Native mints its store's own dialect (ETag or GCS generation); the emulated adapter mints its + /// own values. + Dialect dialect() const override { return mode == Mode::Native ? native_token_type : Dialect::Emulated; } + /// The budget for one attempt of a read-class request, as configured by the mount. + uint64_t attemptTimeoutMs() const override { return attempt_timeout_ms; } + /// What one attempt may cost end to end, connect included: the attempt timeout plus two connect + /// caps (TCP, then TLS), saturating. Delegates to `CasRequestBudget::attemptEnvelopeMs` (the single + /// definition of this formula) rather than re-deriving it here. + uint64_t attemptEnvelopeMs() const override + { + return CasRequestBudget{.attempt_timeout_ms = attempt_timeout_ms, .connect_timeout_cap_ms = connect_timeout_cap_ms} + .attemptEnvelopeMs(); + } + /// The frozen connect cap this backend was constructed with; see the constructor. + uint64_t connectTimeoutCapMs() const { return connect_timeout_cap_ms; } + /// Ask the storage to re-acquire credentials through its refresh callback. + bool refreshCredentials() override { return object_storage->tryRefreshCredentialsViaCallback(); } + /// S3 ETags are content-derived and surfaced in list responses — TRUE for ETag-token Native /// and EmulatedSingleProcess modes. FALSE on a generation-token store (GCS): the XML LIST /// surfaces MD5-style ETags in the response BODY, which the header-level response adaptation @@ -80,31 +126,11 @@ class ObjectStorageBackend final : public Backend /// `If-Match` token; generation stores deliberately omit it and make GC re-read each shard. /// Consumers already treat absent list tokens as Read/fail-closed (GC discover re-reads every /// shard — a cost, not a correctness change). - bool supportsListTokens() const override { return native_token_type != TokenType::Generation; } - - /// Create `key` only if it is absent. On a precondition failure the object is untouched and the - /// result has no token; on success the token identifies the newly written incarnation. - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override; + bool supportsListTokens() const override { return native_token_type != Dialect::Generation; } - /// Execute the selected unconditional blob transport without observing destination state or - /// returning a write-response token. Streaming uses ordinary write settings; staged bytes require - /// a native same-store copy. - void publishBlob(const BlobPublishRequest & request) override; - /// Replace `key` only when its current token exactly equals `expected`; a mismatch leaves the - /// existing incarnation untouched. Storage exceptions propagate instead of being reported as a - /// successful or failed precondition. - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override; - /// Perform a compare-and-set: `expected == nullopt` means create-if-absent. A conflict leaves the - /// object untouched; a committed result carries the new incarnation token. - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override; - /// Remove only the incarnation matching `token`, preserving the object on a mismatch and exposing - /// whether the storage created a delete marker. - DeleteOutcome deleteExact(const String & key, const Token & token) override; - /// Return a page after `cursor`; the next cursor is the last returned key and is empty at the end. - ListPage list(const String & prefix, const String & cursor, size_t limit) override; - - /// Pool-level precondition: on a Native, generation-dialect (GCS) backend, reject the pool unless - /// object versioning is VERIFIABLY disabled — see Backend::checkPoolPreconditions. + /// Pool-level precondition: on a Native, generation-dialect (GCS) backend, reject the pool when + /// object versioning is verified ENABLED; warn and continue when the probe cannot answer — see + /// Backend::checkPoolPreconditions. void checkPoolPreconditions() override; /// Fail-closed precondition: a Native, generation-dialect (GCS) backend refuses a writable mount @@ -119,19 +145,20 @@ class ObjectStorageBackend final : public Backend /// `EmulatedSingleProcess`. void checkConditionalWriteSingleAttemptSupport() override; - /// See Backend::probeSentinelRaw. Native: a raw HEAD via `IObjectStorage::getObjectMetadata` (the - /// THROWING variant — unlike `tryGetObjectMetadata`/`nativeHead`, it never swallows the S3 error), - /// classified by S3 error code. EmulatedSingleProcess (Local): stats the configured container - /// directory (`emu_root`) first — `ContainerAbsent` if it is gone — then the key. - SentinelProbeResult probeSentinelRaw(const String & key) override; + /// See Backend::probeSentinelRaw. Native: ONE `read`, classified by the S3 error it throws. A GET + /// 404 carries a response body, so the SDK can parse its `` and tell `NoSuchKey` from + /// `NoSuchBucket` -- a distinction a bodyless HEAD 404 cannot make. + /// EmulatedSingleProcess (Local): stats the configured container directory (`emu_root`) first -- + /// `ContainerAbsent` if it is gone -- then the key. + SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) override; - /// The token kind this backend's object storage mints: TokenType::ETag for AWS-compatible - /// stores, TokenType::Generation when the storage mints GCS generations (the + /// The token kind this backend's object storage mints: Dialect::ETag for AWS-compatible + /// stores, Dialect::Generation when the storage mints GCS generations (the /// generation rides the ETag plumbing; the VALUE stays opaque either way). - TokenType nativeTokenType() const { return native_token_type; } - void setNativeTokenTypeForTest(TokenType t) { native_token_type = t; } + Dialect nativeTokenType() const { return native_token_type; } + void setNativeTokenTypeForTest(Dialect t) { native_token_type = t; } - /// ---- Token policy (single source of truth; see the .cpp) ---- + /// ---- Etag value policy (single source of truth; see the .cpp) ---- /// A GCS generation reaches this layer through the AWS SDK's ETag field, which the HTTP boundary /// fills with an ETag-shaped — that is, quoted — value. A generation is a number, and quotes are /// transport syntax that must not enter CAS protocol state, where token values are compared for @@ -143,55 +170,38 @@ class ObjectStorageBackend final : public Backend /// corrupt the AWS-compatible path. String normalizeTokenValue(const String & etag) const { - if (native_token_type != TokenType::Generation) + if (native_token_type != Dialect::Generation) return etag; if (etag.size() >= 2 && etag.front() == '"' && etag.back() == '"') return etag.substr(1, etag.size() - 2); return etag; } - /// Mint the incarnation token for a key we just HEAD'd or wrote: the object ETag/generation - /// string carried under this backend's native dialect (native_token_type). - /// - /// This is the ONLY site that mints a Generation token: `tokenForList` is the sole other - /// `native_token_type` mint, and `supportsListTokens` above returns false for Generation, so it - /// cannot produce one. - Token tokenForHead(const String & etag) const - { - return Token{normalizeTokenValue(etag), native_token_type}; - } - - /// The token to surface for a LISTED key: present iff this backend surfaces per-key list tokens - /// (supportsListTokens — FALSE on a generation store, where a list-derived token is a poisoned - /// If-Match) AND the listing carried a non-empty etag. Matches what tokenForHead would return. - std::optional tokenForList(const String & etag) const + /// The normalized incarnation value to surface for a LISTED key: present iff this backend surfaces + /// per-key list values (supportsListTokens — FALSE on a generation store, where a list-derived + /// value is a poisoned If-Match) AND the listing carried a non-empty etag. Matches what + /// `normalizeTokenValue` would return for the same etag. + std::optional tokenForList(const String & etag) const { if (!supportsListTokens() || etag.empty()) return std::nullopt; - return Token{etag, native_token_type}; + return normalizeTokenValue(etag); } - /// Whether an observed incarnation token satisfies an expected one: exact identity (value AND - /// type). Every conditional compare in this backend goes through here. - static bool tokenMatches(const Token & observed, const Token & expected) - { - return observed == expected; - } + /// The per-dialect grammar a response value must meet to be an incarnation. Generation: canonical + /// positive decimal AFTER the SDK ETag-field quote strip (no leading zero, not "0" — zero is the + /// dialect's absence sentinel). ETag: non-empty, not "*" after trimming whitespace, no comma (a + /// list matches any member). Emulated: non-empty. + static bool isValidTokenValue(Dialect type, const String & value); /// Settings for a Native COMPARE/CREATE write (create-if-absent, compare-and-set): mark the request /// conditional, make exactly one attempt at every retry layer, skip the racy post-upload /// existence/size check, and force a single PUT on generation stores because GCS does not - /// enforce the condition on multipart completion. - WriteSettings conditionalWriteSettings() const; - WriteSettings conditionalWriteSettingsForTest() const { return conditionalWriteSettings(); } - /// Convert a successful write/copy response's incarnation-identifying string into this backend's - /// token -- the ONE place that decides how strictly to trust it ("Exact successful-write token"). - /// Generation dialect (GCS): the response MUST carry a non-empty, purely numeric generation; a - /// missing or non-numeric value is an exception -- there is no follow-up HEAD, so a broken or - /// lying response can never be silently patched over by a later, unrelated read. Every other - /// dialect (ETag, and any backend with no write-time token at all, e.g. local files) keeps the - /// pre-existing behavior: an absent value falls back to a fresh HEAD of `key`. - Token tokenFromWriteResult(const String & key, const std::optional & etag); + /// enforce the condition on multipart completion. `attempt_no` is the engine's own 1-based + /// physical-attempt count (see `TransportAccess::attemptNo`), carried into + /// `object_storage_attempt_number` so the HTTP client sees a reissue as attempt >= 2. + WriteSettings conditionalWriteSettings(size_t attempt_no) const; + WriteSettings conditionalWriteSettingsForTest() const { return conditionalWriteSettings(/*attempt_no=*/1); } /// Override the emulated backend's wall clock for deterministic expiry tests. void setEmuNowNsForTest(uint64_t now_ns); /// Return the guarded per-key token-state size for expiry tests. @@ -200,7 +210,38 @@ class ObjectStorageBackend final : public Backend private: const ObjectStoragePtr object_storage; const Mode mode; - TokenType native_token_type = TokenType::ETag; + Dialect native_token_type = Dialect::ETag; + /// See the constructor: what the READ-class requests (read, head, list, remove) carry. + const bool single_attempt_control_plane; + const uint64_t attempt_timeout_ms; + const uint64_t connect_timeout_cap_ms; + ObjectStorageRetryProfile controlPlaneProfile() const + { + return single_attempt_control_plane ? ObjectStorageRetryProfile::SingleAttempt : ObjectStorageRetryProfile::Default; + } + /// The control-request context every read-class primitive builds from `access.attemptNo()`: this + /// backend's own retry profile, attempt timeout and frozen connect cap, plus the caller's attempt + /// number -- see `ObjectStorageControlRequest`. + ObjectStorageControlRequest controlRequest(size_t attempt_no) const + { + return ObjectStorageControlRequest{ + .profile = controlPlaneProfile(), + .attempt_timeout_ms = attempt_timeout_ms, + .connect_timeout_cap_ms = connect_timeout_cap_ms, + .attempt_number = attempt_no}; + } + /// The read settings a request carries: the native conditional dialect, plus the control-request + /// context (retry profile, per-attempt bound and connect cap, attempt number) its caller built. + ReadSettings readSettingsFor(const ObjectStorageControlRequest & request) const; + + /// The keyed primitive's body, taking the control-request context its caller built + /// (`controlRequest(access.attemptNo())`) rather than deriving it again here. + std::optional readUnder(const String & key, const ObjectStorageControlRequest & request); + std::optional headUnder(const String & key, const ObjectStorageControlRequest & request); + RawListPage listUnder(const String & prefix, const String & cursor, size_t limit, + const ObjectStorageControlRequest & request); + RawRemoval removeUnder(const String & key, const String & expected_value, const ObjectStorageControlRequest & request); + SentinelProbeResult probeSentinelUnder(const String & key, const ObjectStorageControlRequest & request); /// EmulatedSingleProcess state: per-key {etag, disambiguator} — see emuMintToken. A successfully /// deleted entry is retained only while its etag is recent enough that an immediate recreate could /// land in the same mtime quantum. `deleteExact` erases already-old entries immediately and queues @@ -217,34 +258,17 @@ class ObjectStorageBackend final : public Backend }; std::deque emu_token_expiry; uint64_t emu_now_ns_for_test = 0; - /// Fallback nonce for the (anomalous) case where the object storage reports an EMPTY etag: mints a - /// fresh, unpersisted value each time — never worse than the old counter for that case, but never - /// masquerading as a real etag-derived identity either. - uint64_t emu_seq = 0; - - /// Look up Native metadata and convert the storage ETag or generation to this backend's token. On - /// a generation-token store, the minted token is validated exactly like a write result (see - /// isValidGenerationTokenValue) before this returns it: a missing/malformed x-goog-generation on an - /// otherwise-successful HEAD would otherwise mint an invalid token here with no check at all, one - /// layer before tokenFromWriteResult's own check on the write path. - std::optional nativeHead(const String & key); - /// True iff `value` is a well-formed generation: non-empty and every character an ASCII digit. - /// Shared by nativeHead and tokenFromWriteResult so the two places that mint a Generation token - /// from a remote response cannot drift apart on what "valid" means. Deliberately NOT folded into - /// tokenForHead, which stays a pure minter with no opinion on the value it is handed. - static bool isValidGenerationTokenValue(const String & value); - /// Write a body with the condition already encoded in `ws`, finalize it, classify a lost - /// precondition, and return the new token when the write succeeds. - PutResult nativeConditionalPut(const String & key, const String & bytes, const WriteSettings & ws, const ObjectMeta & meta); + /// Look up Native metadata and normalize the storage ETag or generation into an incarnation + /// value. The value is returned as the store gave it: whether it IS an incarnation is judged by + /// whoever can act on the answer, never here. + std::optional nativeHead(const String & key, const ObjectStorageControlRequest & request); - /// §3.18 №19 hardening: whether `t` is the dialect this backend itself mints (native_token_type - /// for Native mode, always TokenType::Emulated for EmulatedSingleProcess). Every conditional - /// mutation checks this BEFORE touching the wire (Native forwards only Token::value as the - /// If-Match/removeObjectIfTokenMatches argument, blind to Token::type) or comparing values - /// (Emulated) — a foreign-dialect token is rejected locally rather than trusted to the remote - /// backend, or to a value-space that was never designed to discriminate it. - bool mintingTypeMatches(TokenType t) const { return t == (mode == Mode::Native ? native_token_type : TokenType::Emulated); } + /// Write a body with the condition already encoded in `ws`, finalize it, map a lost precondition + /// onto `RawConflict`, and return the write response's own value on success -- normalized, and + /// otherwise untouched: an empty one means the response named no incarnation, which is the + /// caller's to resolve, not this seam's to refuse. + std::expected nativeConditionalPut(const String & key, const String & bytes, const WriteSettings & ws); /// ---- Emulated helpers (caller holds emu_mutex) ---- /// @@ -257,26 +281,41 @@ class ObjectStorageBackend final : public Backend /// The caller holds `emu_mutex` for all five helpers below, preserving the exists/read and /// observe/write checks as one process-local operation. bool emuExists(const String & key) const; - String emuRead(const String & key, Range range) const; + String emuRead(const String & key) const; /// Write a body as the new incarnation of `key` and return its freshly minted token (the /// object's own post-write etag — see emuMintToken). - Token emuWrite(const String & key, const String & bytes, const ObjectMeta & meta); + String emuWrite(const String & key, const String & bytes); /// Write a complete blob body to a sibling temporary local object, then atomically replace `key` /// and advance any existing same-ETag disambiguator. A failure before the rename leaves the old /// destination and its token state untouched and cleans the temporary. - void emuPublishBlobAtomically(const String & key, const String & bytes); + /// Streams `envelope` + exactly `payload_size` bytes of `payload` into a temporary sibling of + /// `key`, then renames it into place -- nothing is visible at the destination until the byte count + /// has been validated, and the rename keeps publication atomic. Takes `emu_mutex` itself (for the + /// rename + token-state bump only); the caller must NOT hold it. + void emuPublishBlobAtomically(const String & key, const String & envelope, ReadBuffer & payload, uint64_t payload_size); + /// Caller holds emu_mutex: the token bookkeeping after a delete at `key`. + void emuForgetDeletedToken(const String & key); /// Return the current emulated token for a key we just read/HEAD'd, reflecting its on-disk etag — /// does NOT advance the same-etag disambiguator (that only applies to a just-completed write). - Token emuObserveToken(const String & key); + String emuObserveToken(const String & key); uint64_t emuNowNs() const; /// Examine a fixed number of oldest deleted-state records, expiring only an exact current match. void emuPruneTokenState(uint64_t now_ns); /// Single source of truth for minting an emulated token from an observed `etag`: the wire value IS /// the etag while it is the first thing minted for `key` at that etag, or `etag#N` once a SAME-etag /// rewrite forces a disambiguator (`just_wrote` — see the mtime-quantum note in emu_token_state's - /// declaration and codex-review-triage §3.18 19c step 4). An empty `etag` (the storage could not - /// report one) falls back to a fresh, UNPERSISTED monotonic value from emu_seq. - Token emuMintToken(const String & key, const String & etag, bool just_wrote); + /// declaration). An empty `etag` means the storage could not identify the object at all, and + /// there is nothing to invent from: a just-completed write cannot be attributed + /// (`CAS_WRITE_UNATTRIBUTED`) and an observation has no incarnation to report (`CORRUPTED_DATA`). + String emuMintToken(const String & key, const String & etag, bool just_wrote); }; +/// Fail-closed programmer-error guard: a mount opens exactly one backend and one `CasRequestBudget` +/// together (`ContentAddressedMetadataStorage::openPoolView`), and the pool's lease arithmetic +/// (`validateCasRequestBudget`, `CasMountRuntime::admit`) is validated against the budget alone -- +/// never against the backend it hands to the request layer. If the two ever disagree, the backend +/// would silently outlive (or underlive) the envelope the lease math was checked against. Throws +/// `LOGICAL_ERROR` naming both values; called once at open, before `Pool::open`. +void ensureBackendMatchesBudget(const ObjectStorageBackend & backend, const CasRequestBudget & budget); + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.cpp index bb653e90fcea..625c04af8b41 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.cpp @@ -5,6 +5,7 @@ namespace DB { namespace ErrorCodes { + extern const int CAS_DELETE_MARKER; extern const int NOT_IMPLEMENTED; } } @@ -12,238 +13,187 @@ namespace ErrorCodes namespace DB::Cas { -void runCapabilityProbe(Backend & backend, const String & probe_prefix) +namespace +{ + +/// `op.remove`'s own delete-marker signal (`CAS_DELETE_MARKER`, thrown outside the attempt loop because +/// a versioned bucket answers this way every time) is re-signalled here as the operator-facing message +/// the probe has always thrown for it — everything else propagates unchanged. Every `remove` the battery +/// issues against a live incarnation goes through this: a store that ignores the delete precondition AND +/// mints delete markers must still be reported with this message, not the engine's terse one. +Removal removeOrReportDeleteMarker(CasOperation & op, const String & key, const Etag & seen) +{ + try + { + return op.remove(key, seen, Retry::standard()); + } + catch (const DB::Exception & e) + { + if (e.code() != DB::ErrorCodes::CAS_DELETE_MARKER) + throw; + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "CasProbe: remove succeeded but created a versioning delete marker — the bucket has " + "object VERSIONING enabled, and a content-addressed pool cannot run on a versioned bucket: " + "every GC delete would archive a noncurrent version instead of reclaiming storage (the bucket " + "grows forever), and the constantly-rewritten ref objects would pile up versions on every " + "commit. This is NOT ignorable and has no override. Use a bucket where versioning was NEVER " + "enabled — note that merely SUSPENDING versioning is not enough (deletes on a " + "versioning-suspended bucket still mint delete markers, so this probe will refuse again)"); + } +} + +} + +void runCapabilityProbe(CasOperation & op, const String & probe_prefix) { - // Probe key used for the primary battery steps. // Sub-directory style ("probe_prefix/token") ensures that list(probe_prefix, …) works for both the // in-memory backend (prefix match) and the LocalObjectStorage backend (directory listing). const String key = probe_prefix + "/token"; - // Probe key used for the casPut chain. - const String cas_key = probe_prefix + "/cas"; // Best-effort cleanup — runs at function exit regardless of outcome. - // We capture the keys we need to clean up. auto cleanup = [&]() noexcept { - // Skip the delete when HEAD says the key is already gone (the happy path: step 8 deleted - // it). A deleteExact with the absent HeadResult's EMPTY token is a malformed conditional - // op — AWS S3 answers 400 InvalidArgument ("If-Match cannot be empty"), which lands as a - // scary AWSClient log line on every mount even though the catch swallows it. - for (const auto & k : {key, cas_key}) + // Skip the remove when HEAD says the key is already gone (the happy path: the battery's own + // delete already ran). `Etag` can only be minted from an actual HEAD/read observation, so + // an unconditional "delete with whatever precondition" this backend never saw is not + // constructible here — the gate below is the only way to reach `remove` at all. + try { - try - { - const auto h = backend.head(k); - if (h.exists) - backend.deleteExact(k, h.token); - } - catch (...) {} /// NOLINT(bugprone-empty-catch) + const auto h = op.head(key, Retry::standard()); + if (h) + op.remove(key, h->etag, Retry::standard()); } + catch (...) {} /// NOLINT(bugprone-empty-catch) }; try { - // ---- Step 0: store-level preconditions (backend-specific; throws = mount refused). ---- - backend.checkPoolPreconditions(); - - // ---- Step 0b: conditional writes must use one underlying HTTP attempt. Transparent SDK - // retries can outlive the writer's mount lease and hide whether a conditional operation - // committed; CAS retries must instead be explicit and state-aware. Throws = mount refused. - // Keep this separate from Step 0 so each precondition remains independently unit-testable. ---- - backend.checkConditionalWriteSingleAttemptSupport(); - - // ---- Step 1: putIfAbsent fresh → Done; read-after-write returns the bytes. ---- - Token t1; + // ---- Step 1: create fresh -> Committed; read-after-write returns the bytes. ---- + Etag t1 = [&] { - const auto res = backend.putIfAbsent(key, "probe-v1"); - t1 = res.token; - if (res.outcome != PutOutcome::Done) + WriteResult r = op.create(key, "probe-v1", Retry::standard()); + if (!std::holds_alternative(r)) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putIfAbsent on a fresh key returned PreconditionFailed — backend is unexpectedly occupied or broken"); - } + "CasProbe: create on a fresh key did not commit — backend is unexpectedly occupied or broken"); + return std::get(r).etag; + }(); { - const auto g = backend.get(key); - if (!g.has_value() || g->bytes != "probe-v1") + const auto g = op.read(key, Retry::standard()); + if (!g || g->bytes != "probe-v1") throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: read-after-write failed — putIfAbsent succeeded but the object is not readable"); + "CasProbe: read-after-write failed — create succeeded but the object is not readable"); } - // ---- Step 2: putIfAbsent same key → PreconditionFailed; bytes intact. ---- + // ---- Step 2: create the same key again -> Conflict; bytes intact. ---- { - const auto outcome = backend.putIfAbsent(key, "should-not-land").outcome; - if (outcome != PutOutcome::PreconditionFailed) + WriteResult r = op.create(key, "should-not-land", Retry::standard()); + const auto * conflict = std::get_if(&r); + if (!conflict) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putIfAbsent on an existing key was not rejected (PreconditionFailed expected) — " + "CasProbe: create on an existing key was not rejected (a conflict was expected) — " "backend does not enforce conditional create"); - const auto g = backend.get(key); - if (!g.has_value() || g->bytes != "probe-v1") + const auto * seen = std::get_if(&conflict->seen); + if (!seen || seen->bytes != "probe-v1") throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putIfAbsent conflict was 'reported' but the original bytes were clobbered — " + "CasProbe: create's conflict was reported but the original bytes were clobbered — " "backend does not enforce conditional create"); } - // ---- Step 3: putOverwrite wrong token → PreconditionFailed; bytes intact. ---- + // ---- Step 3: replace against the CURRENT incarnation (t1) -> Committed; incarnation changed; + // bytes replaced. Every "wrong incarnation" step below reuses THIS key's own prior + // incarnations (never a synthetic value) — an `Etag` is minted only from an + // actual backend observation, so there is no other way to name one that is + // guaranteed wrong yet dialect-valid. ---- + Etag t2 = [&] { - /// Wrong-token values are NUMERIC on purpose: a generation-dialect backend (GCS) - /// validates the If-Match FORMAT client-side and throws on a non-numeric value (an - /// ETag-kind token leaking into a generation dialect) — the probe's synthetic wrong - /// tokens must be format-valid for EVERY token kind, merely guaranteed-wrong. A huge - /// numeric is a wrong ETag on AWS (412), a wrong generation on GCS (412), and a wrong - /// sequence on the emulated backends (TokenMismatch). - /// - /// The TYPE must be the LIVE dialect (t1.type, just observed from this same backend), - /// never a hardcoded TokenType::Emulated: a backend that mints a different dialect - /// (e.g. Native/ETag) rejects a foreign-dialect token locally, before the wrong VALUE - /// ever reaches the wire — which would make this check pass vacuously against a - /// non-enforcing store instead of proving enforcement (codex-review-triage §3.18, - /// Critical: the №19 local dialect guard must not defeat this probe). - Token wrong_token{"900000000000000001", t1.type}; - const auto outcome = backend.putOverwrite(key, "clobbered", wrong_token).outcome; - if (outcome != PutOutcome::PreconditionFailed) + WriteResult r = op.replace(key, "probe-v2", t1, Retry::standard()); + if (!std::holds_alternative(r)) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putOverwrite with a wrong token was not rejected (PreconditionFailed expected) — " - "backend does not enforce conditional overwrite"); - const auto g = backend.get(key); - if (!g.has_value() || g->bytes != "probe-v1") + "CasProbe: replace with the correct incarnation was rejected — backend does not accept a valid overwrite"); + Etag next = std::get(r).etag; + if (next == t1) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putOverwrite with wrong token was 'rejected' but the original bytes were clobbered"); - } - - // ---- Step 4: putOverwrite correct token → Done; bytes replaced; token changed. ---- - Token t2; + "CasProbe: replace succeeded but did not mint a new incarnation — an incarnation must " + "change on every write"); + return next; + }(); { - const auto res = backend.putOverwrite(key, "probe-v2", t1); - t2 = res.token; - if (res.outcome != PutOutcome::Done) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putOverwrite with the correct token was rejected — backend does not accept valid overwrite"); - if (t2 == t1) + const auto g = op.read(key, Retry::standard()); + if (!g || g->bytes != "probe-v2") throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putOverwrite succeeded but did not mint a new token — tokens must change on every write"); - const auto g = backend.get(key); - if (!g.has_value() || g->bytes != "probe-v2") - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: putOverwrite succeeded but the new bytes are not readable"); + "CasProbe: replace succeeded but the new bytes are not readable"); } - // ---- Step 5: casPut chain. ---- - // 5a: create-if-absent (nullopt expected). - Token ct1; - { - const auto res = backend.casPut(cas_key, "cas-s1", std::nullopt); - ct1 = res.token; - if (res.outcome != CasOutcome::Committed) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut create-if-absent (nullopt expected) was not committed — " - "backend does not support CAS create-if-absent"); - } - // 5b: conflict on existing (nullopt expected, but key exists). - { - const auto outcome = backend.casPut(cas_key, "cas-s1x", std::nullopt).outcome; - if (outcome != CasOutcome::Conflict) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut with nullopt expected against an existing key was not Conflict — " - "backend does not enforce create-if-absent semantics on casPut"); - } - // 5c: conflict on stale token. - { - Token stale{"900000000000000002", ct1.type}; /// numeric + live dialect: see step 3 - const auto outcome = backend.casPut(cas_key, "cas-s1y", stale).outcome; - if (outcome != CasOutcome::Conflict) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut with a stale token was not Conflict — " - "backend does not enforce token-exact CAS"); - } - // Bytes must still be the original. + // ---- Step 4: replace against t1, now STALE (the key committed to t2 in step 3) -> Conflict; + // bytes intact. ---- { - const auto g = backend.get(cas_key); - if (!g.has_value() || g->bytes != "cas-s1") + WriteResult r = op.replace(key, "clobbered", t1, Retry::standard()); + const auto * conflict = std::get_if(&r); + if (!conflict) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut conflicts were reported but the original bytes were altered"); - } - // 5d: commit on current token. - { - const auto res = backend.casPut(cas_key, "cas-s2", ct1); - if (res.outcome != CasOutcome::Committed) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut with the current token was not committed — " - "backend does not honor casPut with matching token"); - const auto g = backend.get(cas_key); - if (!g.has_value() || g->bytes != "cas-s2") + "CasProbe: replace with a stale incarnation was not rejected (a conflict was expected) — " + "backend does not enforce conditional overwrite"); + const auto * seen = std::get_if(&conflict->seen); + if (!seen || seen->bytes != "probe-v2") throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: casPut committed but new bytes are not readable"); + "CasProbe: replace with a stale incarnation was 'rejected' but the original bytes were clobbered"); } - // ---- Step 6: deleteExact wrong token → TokenMismatch AND the object still readable. ---- + // ---- Step 5: remove with a STALE incarnation (t1) -> Mismatch; the object survives. A store + // that ignores this precondition removes the object here, so this path can reach a + // delete marker exactly like step 7's — route it through the same reporter. ---- { - Token wrong_token{"900000000000000003", t2.type}; /// numeric + live dialect: see step 3 - const auto d = backend.deleteExact(key, wrong_token); - if (d.kind != DeleteOutcome::Kind::TokenMismatch) + const Removal d = removeOrReportDeleteMarker(op, key, t1); + if (d != Removal::Mismatch) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: deleteExact with a wrong token was not TokenMismatch — " - "delete with mismatching token was honored — backend does not enforce conditional deletes"); - const auto g = backend.get(key); - if (!g.has_value()) + "CasProbe: remove with a stale incarnation was not rejected (a mismatch was expected) — " + "backend does not enforce conditional deletes"); + const auto g = op.read(key, Retry::standard()); + if (!g) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: deleteExact with a wrong token was rejected (correctly) but the object was deleted anyway — " + "CasProbe: remove with a stale incarnation was rejected (correctly) but the object was deleted anyway — " "backend does not enforce conditional deletes"); } - // ---- Step 7: list(probe_prefix) contains the probe key (list-after-write). ---- + // ---- Step 6: list(probe_prefix) contains the probe key (list-after-write). ---- { - const auto page = backend.list(probe_prefix, "", 100); bool found = false; - for (const auto & listed : page.keys) + op.forEachListedKey(probe_prefix, [&](const ListedKey & listed) -> bool { - if (listed.key == key) - { - found = true; - break; - } - } + if (listed.key != key) + return true; + found = true; + return false; + }, Retry::standard()); if (!found) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "CasProbe: list-after-write failed — the probe key '{}' is not visible in the listing under prefix '{}'", key, probe_prefix); } - // ---- Step 8: deleteExact correct token → Deleted; object gone; no delete marker; - // list no longer contains the key. ---- + // ---- Step 7: remove with the CORRECT incarnation (t2) -> Removed; no delete marker; object + // gone; list no longer contains the key. ---- { - const auto d = backend.deleteExact(key, t2); - if (d.kind != DeleteOutcome::Kind::Deleted) + const Removal d = removeOrReportDeleteMarker(op, key, t2); + if (d != Removal::Removed) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: deleteExact with the correct token was not Deleted — backend rejected a valid token-exact delete"); - if (d.created_delete_marker) + "CasProbe: remove with the correct incarnation was not Removed — backend rejected a valid incarnation-exact delete"); + const auto g = op.read(key, Retry::standard()); + if (g) throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: deleteExact succeeded but created a versioning delete marker — the bucket has " - "object VERSIONING enabled, and a content-addressed pool cannot run on a versioned bucket: " - "every GC delete would archive a noncurrent version instead of reclaiming storage (the bucket " - "grows forever), and the constantly-rewritten ref objects would pile up versions on every " - "commit. This is NOT ignorable and has no override. Use a bucket where versioning was NEVER " - "enabled — note that merely SUSPENDING versioning is not enough (deletes on a " - "versioning-suspended bucket still mint delete markers, so this probe will refuse again)"); - const auto g = backend.get(key); - if (g.has_value()) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: deleteExact succeeded (Deleted) but the object is still readable — backend delete is not effective"); - // List-after-delete. - const auto page = backend.list(probe_prefix, "", 100); - for (const auto & listed : page.keys) + "CasProbe: remove succeeded (Removed) but the object is still readable — backend delete is not effective"); + bool still_listed = false; + op.forEachListedKey(probe_prefix, [&](const ListedKey & listed) -> bool { - if (listed.key == key) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "CasProbe: list-after-delete failed — the deleted probe key '{}' is still visible in the listing under prefix '{}'", - key, probe_prefix); - } - } - - // ---- Step 9: cleanup (best-effort; also deletes cas_key). ---- - // cas_key is still alive — clean it up via its current token. - { - const auto h = backend.head(cas_key); - if (h.exists) - backend.deleteExact(cas_key, h.token); + if (listed.key != key) + return true; + still_listed = true; + return false; + }, Retry::standard()); + if (still_listed) + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "CasProbe: list-after-delete failed — the deleted probe key '{}' is still visible in the listing under prefix '{}'", + key, probe_prefix); } } catch (...) @@ -253,8 +203,8 @@ void runCapabilityProbe(Backend & backend, const String & probe_prefix) throw; } - // Normal-exit cleanup (cas_key was cleaned inside the try; key was deleted in step 8). - // Call cleanup anyway to handle any partial state edge cases — it is a no-op if keys are gone. + // Normal-exit cleanup (the key was already deleted in step 7). Call cleanup anyway to handle any + // partial state edge cases — it is a no-op if the key is gone. cleanup(); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.h index 9ac92529f70b..99ec36581af2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasProbe.h @@ -1,23 +1,26 @@ #pragma once -#include +#include #include namespace DB::Cas { -/// Run the capability battery against `backend`, using throwaway keys under `probe_prefix`. +/// Run the capability battery against `op`, using throwaway keys under `probe_prefix`. /// -/// The probe validates the backend preconditions required by a writable content-addressed pool: -/// 1. Store-level safety checks pass, including the requirement that conditional writes use one -/// underlying HTTP attempt. Hidden SDK retries can outlive the writer's mount lease and obscure -/// whether a conditional operation committed; retries must therefore be explicit CAS state-machine -/// transitions rather than transparent client behavior. -/// 2. Conditional-create and conditional-overwrite are enforced (`putIfAbsent` prevents overwrites, -/// and `putOverwrite` rejects a wrong-token update). -/// 3. `casPut` supports create-if-absent, conflict-on-existing, conflict-on-stale, and commit-on-current. -/// 4. Conditional-delete is enforced (`deleteExact` with a wrong token is rejected and the object survives). -/// 5. Listing reflects both creation and deletion of a probe object. -/// 6. Successful deletion does not create a versioning delete marker. A content-addressed pool cannot +/// `op` must already be admitted. The two store-level precondition hooks — +/// `Backend::checkPoolPreconditions` and `Backend::checkConditionalWriteSingleAttemptSupport` — are the +/// CALLER's responsibility, run through `CasRequests::backendForCapabilityPredicates()` before admitting +/// `op` and calling this: an admitted `CasOperation` carries no route back to the raw backend, by design, +/// so this function cannot reach them itself. +/// +/// The battery validates the backend preconditions required by a writable content-addressed pool: +/// 1. `create` and `replace` are conditional and exact: `create` refuses an occupied key +/// (create-if-absent, conflict-on-existing) and `replace` refuses a stale incarnation +/// (conflict-on-stale), while a matching incarnation commits (commit-on-current). +/// 2. Conditional-delete is enforced (`remove` with a stale incarnation is rejected and the object +/// survives). +/// 3. Listing reflects both creation and deletion of a probe object. +/// 4. Successful deletion does not create a versioning delete marker. A content-addressed pool cannot /// reclaim storage correctly from a versioned bucket: garbage-collection deletes would archive old /// versions instead of removing objects, and repeated ref updates would accumulate versions. /// @@ -25,9 +28,9 @@ namespace DB::Cas /// specific failed check. This is fail-closed: a backend that does not pass the battery MUST NOT be /// used to coordinate a content-addressed pool. /// -/// Cleanup of probe keys is best-effort and runs unconditionally: after the battery completes, or on +/// Cleanup of the probe key is best-effort and runs unconditionally: after the battery completes, or on /// the failure path immediately before the check-failing exception is rethrown. Cleanup itself suppresses /// exceptions so that it cannot hide the capability-check failure. -void runCapabilityProbe(Backend & backend, const String & probe_prefix); +void runCapabilityProbe(CasOperation & op, const String & probe_prefix); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp new file mode 100644 index 000000000000..ca00928ef9e6 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp @@ -0,0 +1,71 @@ +#include + +#include +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} +} + +namespace DB::Cas +{ + +uint64_t CasRequestBudget::attemptEnvelopeMs() const +{ + /// The TCP connect and the TLS handshake each get one connect interval from Poco, so an HTTPS + /// attempt may spend two caps before any request I/O. Scheme-agnostic on purpose: conservative for + /// plain HTTP, exact for HTTPS. + const uint64_t cap = connect_timeout_cap_ms.value_or(0); + const uint64_t connects = cap > std::numeric_limits::max() / 2 ? std::numeric_limits::max() : 2 * cap; + return attempt_timeout_ms > std::numeric_limits::max() - connects + ? std::numeric_limits::max() + : attempt_timeout_ms + connects; +} + +void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, + uint64_t mount_renew_period_ms, bool background_renewal) +{ + if (budget.attempt_timeout_ms == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "CAS request budget rejected: attempt_timeout_ms must be at least 1; a zero would reserve nothing " + "while the request keeps the storage's own timeout"); + const uint64_t envelope = budget.attemptEnvelopeMs(); + /// Subtractions against the unsigned TTL: the sums could wrap for absurd values and read as small. + const bool one_envelope_fits = envelope < mount_lease_ttl_ms + && budget.lease_safety_margin_ms < mount_lease_ttl_ms - envelope; + if (!one_envelope_fits) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "CAS request budget rejected: the attempt envelope ({} ms = attempt_timeout_ms {} + connect cap {}) " + "plus lease_safety_margin_ms ({}) must be strictly less than the mount lease TTL ({} ms). " + "A writable mount refuses to open with this budget.", + envelope, budget.attempt_timeout_ms, budget.connect_timeout_cap_ms.value_or(0), budget.lease_safety_margin_ms, mount_lease_ttl_ms); + if (background_renewal) + { + /// A renewal is a write: two envelopes (the attempt and its settlement read) after one period. + /// Saturating doubling first (matching the production horizon checks' own arithmetic), then + /// subtraction-based comparisons against the TTL -- no truncating division, so this enforces + /// exactly the inequality the exception message states, not an off-by-one-tighter one. + const uint64_t two_envelope = envelope > std::numeric_limits::max() / 2 + ? std::numeric_limits::max() : 2 * envelope; + const bool cadence_fits = mount_renew_period_ms < mount_lease_ttl_ms + && two_envelope < mount_lease_ttl_ms - mount_renew_period_ms + && budget.lease_safety_margin_ms < mount_lease_ttl_ms - mount_renew_period_ms - two_envelope; + if (!cadence_fits) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "CAS mount renewal cadence rejected: mount_renew_period_ms ({}) + 2 × attempt envelope ({} ms) + " + "lease_safety_margin_ms ({}) must be strictly less than the mount lease TTL ({} ms)", + mount_renew_period_ms, envelope, budget.lease_safety_margin_ms, mount_lease_ttl_ms); + } + LOG_INFO(getLogger("CasRequestBudget"), + "CAS request budget in effect: attempt_timeout_ms={} connect_timeout_cap_ms={} envelope_ms={} lease_safety_margin_ms={} " + "(mount_lease_ttl_ms={} mount_renew_period_ms={})", + budget.attempt_timeout_ms, budget.connect_timeout_cap_ms.value_or(0), envelope, budget.lease_safety_margin_ms, + mount_lease_ttl_ms, mount_renew_period_ms); +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h new file mode 100644 index 000000000000..1d96e41b199d --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h @@ -0,0 +1,66 @@ +#pragma once +#include +#include + +namespace DB::Cas +{ + +/// The limits a writable mount is configured with. `CasMountRuntime::admit` measures a request against +/// `lease_safety_margin_ms` plus a caller-supplied need expressed in attempt envelopes (one for an +/// ordinary attempt, TWO for a ref-log append's write-plus-settlement-read -- see +/// `CasMountRuntime::refAppendFenceOk`), and the three `recovery_retry_*` fields bound a whole +/// ref-table recovery; see `validateCasRequestBudget` for the relationship a writable mount enforces +/// at startup. +struct CasRequestBudget +{ + /// Maximum client wait budgeted for one HTTP attempt. The request contract reserves this before + /// every attempt it starts; the actual socket-level wait is configured on the object storage's + /// client (the object storage backend's single-attempt client), not by this struct. + uint64_t attempt_timeout_ms = 5000; + /// Startup-only margin folded into `validateCasRequestBudget`'s inequality against the mount lease + /// TTL. Not consulted at runtime by the engine itself -- the caller's fence (backed by the local + /// write fence's own deadline) is what actually gates lease-relative timing per attempt. + uint64_t lease_safety_margin_ms = 2000; + + /// The cap the single-attempt client puts on one TCP connect and again on one TLS handshake, + /// frozen when the pool opens as `min(disk connect_timeout_ms, attempt_timeout_ms)` (a configured + /// zero means unbounded and is normalized to the attempt timeout) so a later client reload cannot + /// widen the envelope. Empty when the storage has no S3 client: the envelope is the attempt alone. + std::optional connect_timeout_cap_ms = 1000; + + /// What ONE physical attempt is allowed to cost end to end: two connect caps (TCP, then TLS) plus + /// the attempt timeout, saturating. The request contract reserves this, not the bare attempt timeout, before + /// every attempt it starts. + uint64_t attemptEnvelopeMs() const; + + /// Recovery-level retry (`CasRefLedger::ensureRefTableRecovered`): a whole ref-table recovery + /// attempt (LIST + snapshot/log GETs + seal PUT) that fails with a transient NETWORK_ERROR is + /// retried, with capped-exponential backoff, until this total wall-clock budget is spent — then the + /// error propagates and the table's load fails for this touch (the `lazy_load_tables` database + /// setting makes the NEXT touch retry). This sits ON TOP of each write's own `Retry` policy window + /// (`Retry::standard()`'s 90s): one recovery attempt may itself burn ~90s inside a single seal PUT. + /// Independent of the mount-lease invariants validated in `validateCasRequestBudget` — not part of + /// that inequality set. + uint64_t recovery_retry_budget_ms = 120000; + uint64_t recovery_retry_initial_backoff_ms = 1000; + uint64_t recovery_retry_max_backoff_ms = 30000; +}; + +/// Startup validation: a writable mount refuses to open with an inconsistent budget rather than +/// silently falling back to an unbounded or unsafe retry policy. Throws +/// `BAD_ARGUMENTS` unless: +/// attemptEnvelopeMs() + lease_safety_margin_ms < mount_lease_ttl_ms +/// and, when `background_renewal` is true (the mount runs a background renewer): +/// mount_renew_period_ms + 2 × attemptEnvelopeMs() + lease_safety_margin_ms < mount_lease_ttl_ms +/// (a renewal is a write: two envelopes for the attempt and its settlement read). +/// +/// A successor mounting over an unclean predecessor waits at least one lease TTL, plus its +/// materialization grace period, before trusting recovery listings. This is long enough for any +/// conditional PUT still in flight at the predecessor to either land or be abandoned by its own +/// exhausted retry budget. The predecessor's budget is constrained by +/// `attemptEnvelopeMs() + lease_safety_margin_ms < mount_lease_ttl_ms`, so no additional handover +/// check is needed here. +void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, uint64_t mount_renew_period_ms, + bool background_renewal); + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.cpp deleted file mode 100644 index 0e81b0e7415c..000000000000 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.cpp +++ /dev/null @@ -1,912 +0,0 @@ -#include - -#include -#include -#include -#include - -#include "config.h" - -#if USE_AWS_S3 -#include -#endif - -#include -#include -#include -#include -#include - -namespace ProfileEvents -{ - extern const Event CASConditionalWriteAttempts; - extern const Event CASConditionalWriteCommitted; - extern const Event CASConditionalWriteDefiniteFailure; - extern const Event CASConditionalWriteUnresolved; - extern const Event CASConditionalWriteFenceLostPostWrite; -} - -namespace DB -{ -namespace ErrorCodes -{ - extern const int BAD_ARGUMENTS; - extern const int CORRUPTED_DATA; - extern const int LOGICAL_ERROR; - extern const int NETWORK_ERROR; - extern const int NOT_IMPLEMENTED; -} -} - -namespace DB::Cas -{ - -CasWriteOutcome classifyConditionalWriteResult([[maybe_unused]] const std::exception & e) -{ -#if USE_AWS_S3 - /// `PreconditionFailed`/`NoSuchKey` (a lost If-None-Match/If-Match — see - /// ObjectStorageBackend::finalizeConditionalWrite for the exact matching), any 5xx - /// (InternalError/ServiceUnavailable/SlowDown/RequestTimeout), and any S3 error this function does - /// not recognize all fall through to the fail-safe default below: Unresolved. Only the WHITELIST - /// below proves the request was never applied. - if (const auto * s3e = dynamic_cast(&e)) - { - if (S3::isMalformedRequestError(*s3e) || S3::isEntityTooLargeError(*s3e) || S3::isAccessDeniedError(*s3e)) - return CasWriteOutcome::DefiniteFailure; - } -#endif - /// Poco::Net::NetException (connection loss) / Poco::TimeoutException (client-side timeout) and - /// every other error type: the request's fate is unproven — fail toward "resolve before - /// reissuing, never toward a false - /// DefiniteFailure. - return CasWriteOutcome::Unresolved; -} - -void recordConditionalWriteAttemptStarted() -{ - ProfileEvents::increment(ProfileEvents::CASConditionalWriteAttempts); -} - -void recordConditionalWriteOutcome(CasWriteOutcome outcome) -{ - switch (outcome) - { - case CasWriteOutcome::Committed: - ProfileEvents::increment(ProfileEvents::CASConditionalWriteCommitted); - return; - case CasWriteOutcome::DefiniteFailure: - ProfileEvents::increment(ProfileEvents::CASConditionalWriteDefiniteFailure); - return; - case CasWriteOutcome::Unresolved: - ProfileEvents::increment(ProfileEvents::CASConditionalWriteUnresolved); - return; - } -} - -namespace -{ - -uint64_t steadyClockNowMs() -{ - return static_cast(std::chrono::duration_cast( - std::chrono::steady_clock::now().time_since_epoch()).count()); -} - -/// The default inter-attempt backoff sleep. NOT a race-fix sleep: it is deliberate, bounded, -/// fence-gated pacing of reissues toward a recovering object store, and it is injectable so tests -/// never wait on it. -void threadSleepMs(uint64_t ms) -{ - std::this_thread::sleep_for(std::chrono::milliseconds(ms)); -} - -/// Deterministic caller/local bugs a mutable conditional retry loop must surface immediately: -/// reissuing only replays the same failure and buries the root cause behind a retryable exception. -/// The set: -/// LOGICAL_ERROR — a local invariant violation -/// NOT_IMPLEMENTED — a deterministic mode or capability guard -/// BAD_ARGUMENTS — a deterministic encode/argument rejection (e.g. BAD_ARGUMENTS escaping -/// buildHeader's second, intended_ref-less encode) -/// CORRUPTED_DATA — integrity failure; retrying re-reads/re-streams the same bad bytes (the same -/// fail-fast rule the driver-side correctness markers enforce) -/// Fail-safe either way: a propagated exception is never a false Committed. -bool isDeterministicLocalFailure(int code) -{ - return code == ErrorCodes::LOGICAL_ERROR || code == ErrorCodes::NOT_IMPLEMENTED - || code == ErrorCodes::BAD_ARGUMENTS || code == ErrorCodes::CORRUPTED_DATA; -} - -enum class OverwriteGateClosure : uint8_t -{ - Open, - FenceOrLifecycleLost, - Cancelled, - Deadline, -}; - -struct OverwriteGateSample -{ - OverwriteGateClosure closure; - CasOverwriteStopCause stop_cause; -}; - -/// Sample the operation stop cause and clock exactly once. The returned closure has already applied -/// the protocol precedence; callers only map its position to `CasUnresolvedReason`. -OverwriteGateSample sampleOverwriteGate( - const CasOverwriteOperationContext & context, - const std::function & now_ms, - uint64_t required_time_ms) -{ - const CasOverwriteStopCause stop_cause = context.stop_cause(); - const uint64_t now = now_ms(); - const bool deadline_closed - = now >= context.absolute_deadline_ms || required_time_ms > context.absolute_deadline_ms - now; - - if (stop_cause == CasOverwriteStopCause::FenceOrLifecycleLost) - return {OverwriteGateClosure::FenceOrLifecycleLost, stop_cause}; - if (stop_cause == CasOverwriteStopCause::Cancelled) - return {OverwriteGateClosure::Cancelled, stop_cause}; - if (deadline_closed) - return {OverwriteGateClosure::Deadline, stop_cause}; - return {OverwriteGateClosure::Open, stop_cause}; -} - -uint64_t saturatingAdd(uint64_t lhs, uint64_t rhs) -{ - if (rhs > std::numeric_limits::max() - lhs) - return std::numeric_limits::max(); - return lhs + rhs; -} - -} - -void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, uint64_t mount_renew_period_ms) -{ - if (budget.max_attempts < 1) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "CAS request budget rejected: max_attempts must be at least 1 (got {}) — zero would let " - "putIfAbsentControlled return Unresolved without ever sending an attempt.", - budget.max_attempts); - - /// Overflow-safe: `attempt_timeout_ms + lease_safety_margin_ms` could wrap uint64 for absurd config - /// values, which would make the sum spuriously small and the inequality below pass when it should - /// fail closed. Compare via subtraction against the (unsigned, so already non-negative) TTL instead - /// of computing the sum directly. - if (!(budget.attempt_timeout_ms < mount_lease_ttl_ms - && budget.lease_safety_margin_ms < mount_lease_ttl_ms - budget.attempt_timeout_ms)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "CAS request budget rejected: attempt_timeout_ms ({}) + lease_safety_margin_ms ({}) must be " - "strictly less than the mount lease TTL ({} ms). A writable mount refuses to open with " - "this budget.", - budget.attempt_timeout_ms, budget.lease_safety_margin_ms, mount_lease_ttl_ms); - /// STRICTLY less, and the strictness is the load-bearing half. `attempt_timeout_ms > - /// operation_deadline_ms` is the obvious error — a single attempt cannot outlast the logical - /// operation it belongs to. EQUALITY is the subtle one, and it is worse than useless: the deadline - /// is captured as `now + operation_deadline_ms` and every pre-send gate below asks - /// `now + attempt_timeout_ms > deadline_ms`, so equal values collapse that to `now_2 > now_1` and - /// ONE elapsed millisecond between the two clock reads refuses the operation having sent NOTHING. - /// The resulting behaviour is "mostly works, occasionally refuses with nothing sent", decided by - /// the scheduler rather than by the budget — exactly the flakiness this validation exists to catch, - /// and observed three times in tests before it was forbidden. A caller that wants one attempt says - /// `max_attempts = 1`; the equality adds only the race. - if (!(budget.attempt_timeout_ms < budget.operation_deadline_ms)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "CAS request budget rejected: attempt_timeout_ms ({}) must be strictly less than " - "operation_deadline_ms ({}) — equality turns the pre-send gate into a wall-clock race that " - "refuses after a single elapsed tick, having sent nothing. Use max_attempts to bound the " - "number of attempts.", - budget.attempt_timeout_ms, budget.operation_deadline_ms); - if (!(budget.retry_initial_backoff_ms <= budget.retry_max_backoff_ms)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "CAS request budget rejected: retry_initial_backoff_ms ({}) must not exceed " - "retry_max_backoff_ms ({}) — the capped-exponential backoff cap cannot sit below its own " - "starting value. Set both to 0 to disable inter-attempt backoff.", - budget.retry_initial_backoff_ms, budget.retry_max_backoff_ms); - - LOG_INFO(getLogger("CasRequestControl"), - "CAS request budget in effect: attempt_timeout_ms={} operation_deadline_ms={} max_attempts={} " - "lease_safety_margin_ms={} retry_initial_backoff_ms={} retry_max_backoff_ms={} " - "(mount_lease_ttl_ms={} mount_renew_period_ms={})", - budget.attempt_timeout_ms, budget.operation_deadline_ms, budget.max_attempts, - budget.lease_safety_margin_ms, budget.retry_initial_backoff_ms, budget.retry_max_backoff_ms, - mount_lease_ttl_ms, mount_renew_period_ms); -} - -namespace -{ -/// Shared by both public entry points below so the log line and the exception's message text can -/// never drift apart. Rate-limited (not per-distinct-`why` -- `LogSeriesLimiter` keys on the LOGGER -/// NAME only, so under a sustained outage where `why` keeps changing slightly, only the first message -/// in each window prints; this is the intended throttle, not a bug). Warning-level visibility is -/// intentional: this condition is expected to self-heal -/// (the caller retries), but an operator watching CAS logs directly should see it without having to -/// know to look at system.replication_queue. -void logCasWriteRetryLater(const String & why) -{ - LogSeriesLimiter log(getLogger("CasWriteRetryLater"), /*allowed_count=*/1, /*interval_s=*/30); - LOG_WARNING(log, "CAS write could not be committed ({}); retrying later", why); -} -} - -[[noreturn]] void throwCasWriteRetryLater(const String & why) -{ - logCasWriteRetryLater(why); - throw Exception(ErrorCodes::NETWORK_ERROR, "CAS write could not be committed ({}); retrying later", why); -} - -std::exception_ptr makeCasWriteRetryLaterExceptionPtr(const String & why) -{ - logCasWriteRetryLater(why); - return std::make_exception_ptr( - Exception(ErrorCodes::NETWORK_ERROR, "CAS write could not be committed ({}); retrying later", why)); -} - -[[noreturn]] void throwCasTransientUnavailable(const String & subject, const String & condition) -{ - /// The code is coarse (it shares a `system.errors` row with socket failures), so the MESSAGE must - /// carry the whole truth: which CA condition refused, and that the refusal is a state rather than - /// damage. Consumers key on the code; operators read this line. - /// - /// The shared suffix carries ONLY the classification, because that is the one claim true at every - /// site: retry-later is right even where the condition may turn out terminal, since the next attempt - /// re-decides against fresh state. Any promise about HOW the condition clears belongs in `condition`, - /// where the site that can actually prove it makes it -- `checkFenceOrThrow` provably cannot. - throw Exception(ErrorCodes::NETWORK_ERROR, - "{} -- {}; TRANSIENT unavailability, not damage", subject, condition); -} - -CasRequestController::CasRequestController(BackendPtr backend_, CasRequestBudget budget_, std::function now_ms_, - std::function sleep_ms_) - : backend(std::move(backend_)) - , budget(budget_) - , now_ms(now_ms_ ? std::move(now_ms_) : std::function(steadyClockNowMs)) - , sleep_ms(sleep_ms_ ? std::move(sleep_ms_) : std::function(threadSleepMs)) -{ -} - -void CasRequestController::setSleepFnForTest(std::function sleep_ms_) -{ - sleep_ms = sleep_ms_ ? std::move(sleep_ms_) : std::function(threadSleepMs); -} - -uint64_t CasRequestController::backoffBeforeAttempt(uint32_t next_attempt) const -{ - const uint64_t initial = budget.retry_initial_backoff_ms; - const uint64_t cap = budget.retry_max_backoff_ms; - if (initial == 0 || next_attempt < 2) - return 0; - /// Saturating `initial << doublings`: `initial > cap >> doublings` implies the unshifted product - /// already exceeds the cap, so return the cap without ever computing an overflowing shift. - const uint32_t doublings = next_attempt - 2; - if (doublings >= 63 || initial > (cap >> doublings)) - return cap; - return std::min(initial << doublings, cap); -} - -bool CasRequestController::pauseBeforeReissue(uint32_t completed_attempt, uint64_t deadline_ms, - const std::function & fence_ok, CasUnresolvedReason * out_reason) -{ - /// Fence BEFORE the sleep (the pre-attempt fence rule applies to the whole loop, not just the - /// attempt): a fence lost mid-backoff aborts the operation instantly — sleeping first would keep a - /// fenced writer alive for up to a full backoff cap after it lost its right to write. - if (!fence_ok()) - { - if (out_reason) - *out_reason = CasUnresolvedReason::FenceLostMidWay; - return false; - } - const uint64_t backoff = backoffBeforeAttempt(completed_attempt + 1); - if (backoff == 0) - return true; - /// Never serve a sleep the operation cannot afford: if the backoff plus one more attempt would - /// cross the operation deadline, give up NOW instead of sleeping into a guaranteed Unresolved. - if (now_ms() + backoff + budget.attempt_timeout_ms > deadline_ms) - { - if (out_reason) - *out_reason = CasUnresolvedReason::DeadlineMidWay; - return false; - } - sleep_ms(backoff); - return true; -} - -CasWriteOutcome CasRequestController::resolveByExactGet(std::string_view key, std::string_view expected_bytes, - Token * out_token) -{ - const String key_s{key}; - std::optional got; - try - { - got = backend->get(key_s); - } - catch (const std::exception &) - { - /// The GET itself failed (network, auth, ...): the object's identity cannot be proven either - /// way — an unresolved read leaves this Unresolved, exactly like an absent read. - return CasWriteOutcome::Unresolved; - } - - if (!got) - return CasWriteOutcome::Unresolved; /// absent -> another attempt may still be legal - - if (got->bytes == expected_bytes) - { - if (out_token) - *out_token = got->token; - return CasWriteOutcome::Committed; /// identical deterministic bytes -> the earlier attempt DID commit - } - - /// A DIFFERENT valid object at the exact key this create intended: a real conflict, not a retryable - /// ambiguity. Fail closed rather than silently treating it as Unresolved/DefiniteFailure. - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CasRequestController: exact-key resolution at '{}' observed a DIFFERENT object than the one " - "this attempt intended to create — a real conflict, not a retryable ambiguity", key_s); -} - -CasWriteOutcome CasRequestController::putIfAbsentControlled( - std::string_view key, std::string_view bytes, const std::function & fence_ok, Token * out_token, - CasUnresolvedReason * out_reason) -{ - const String key_s{key}; - const String bytes_s{bytes}; - const uint64_t deadline_ms = now_ms() + budget.operation_deadline_ms; - /// Diagnostic bookkeeping only -- nothing below branches on it (finding #37 defect 3). - uint32_t attempts_sent = 0; - /// Does an EARLIER attempt of THIS call remain unresolved -- sent, and neither proven applied nor - /// proven refused? Set on the one path that produces exactly that state: an attempt whose outcome - /// was ambiguous and whose exact-key resolve came back absent or unreadable. The request may have - /// been received; an absent read now proves nothing about what materializes later, which is the - /// whole reason the reissue loop exists. This is NOT diagnostic: it decides the CALL's verdict at - /// the DefiniteFailure arm below. A pre-attempt gate refusal never sets it -- those return without - /// sending, so they leave nothing that could land. - bool earlier_attempt_unresolved = false; - const auto unresolved = [&](CasUnresolvedReason reason) - { - if (out_reason) - *out_reason = reason; - return CasWriteOutcome::Unresolved; - }; - if (out_reason) - *out_reason = CasUnresolvedReason::NotUnresolved; - - for (uint32_t attempt = 1; attempt <= budget.max_attempts; ++attempt) - { - /// Gate BEFORE every attempt: the - /// local mount fence must still hold, and there must be enough of the operation's own deadline - /// left for one more attempt to plausibly complete. Neither check sends anything to the backend. - if (!fence_ok()) - return unresolved(attempts_sent == 0 ? CasUnresolvedReason::NoAttemptSent - : CasUnresolvedReason::FenceLostMidWay); - if (now_ms() + budget.attempt_timeout_ms > deadline_ms) - return unresolved(attempts_sent == 0 ? CasUnresolvedReason::NoAttemptSent - : CasUnresolvedReason::DeadlineMidWay); - ++attempts_sent; - - /// The committed incarnation's token, filled by whichever leg proves Committed below. - Token committed_token; - CasWriteOutcome attempt_outcome{}; - try - { - const PutResult put = backend->putIfAbsent(key_s, bytes_s); - /// PreconditionFailed here means only "the key already exists" — it does NOT prove who - /// created it (possibly OUR earlier unresolved attempt). Collapse it onto Unresolved so it - /// goes through the SAME resolve-before-reissue path as an ambiguous exception, never a - /// false DefiniteFailure/Committed. - attempt_outcome = put.outcome == PutOutcome::Done ? CasWriteOutcome::Committed : CasWriteOutcome::Unresolved; - if (put.outcome == PutOutcome::Done) - committed_token = put.token; - } - catch (const std::exception & e) - { - attempt_outcome = classifyConditionalWriteResult(e); - } - - if (attempt_outcome == CasWriteOutcome::DefiniteFailure) - { - /// THIS attempt is proven never applied — but the verdict belongs to the CALL. An earlier - /// attempt that is still unresolved may yet materialize at the key, and a caller reading - /// `DefiniteFailure` acts on "the key is unwritten": `CasRefLedger::commitRefChunk` clears - /// its apply-pending marker and reports the txn id never used, so the next append re-derives - /// that id and a late-landing predecessor becomes an acked-then-lost transaction. Ambiguity - /// dominates a definite refusal that came after it; the caller wedges and resolves the key - /// instead. No resolve and no retry either way — this attempt has nothing left to settle. - if (earlier_attempt_unresolved) - return unresolved(CasUnresolvedReason::DefiniteFailureAfterAmbiguity); - return CasWriteOutcome::DefiniteFailure; /// every attempt of this call was proven never applied - } - - if (attempt_outcome == CasWriteOutcome::Unresolved) - { - /// Resolve-before-reissue. May throw CORRUPTED_DATA (a real - /// conflict) straight out of this call — that is never a retry signal. - attempt_outcome = resolveByExactGet(key_s, bytes_s, &committed_token); - if (attempt_outcome == CasWriteOutcome::Unresolved) - { - /// This attempt is now one that may still land: it was sent, and the resolve settled - /// nothing. Recorded BEFORE the exhaustion checks below so it is set no matter which of - /// them ends the loop, and read by the DefiniteFailure arm of every later attempt. - earlier_attempt_unresolved = true; - /// Absent/unreadable: another attempt of the SAME (key, bytes) may be legal — after the - /// fence-gated capped-exponential backoff (pauseBeforeReissue). No pause after the LAST - /// attempt: the budget is spent, sleeping would only delay the Unresolved verdict. - /// - /// Both refusals report through `unresolved`, never a bare `return Unresolved`: this is - /// the ordinary way a busy lane exhausts itself, so leaving `out_reason` at its initial - /// `NotUnresolved` here made the ref lane's wedge message read "is UNCERTAIN (not - /// unresolved)" for the single most common wedge there is. - if (attempt == budget.max_attempts) - return unresolved(CasUnresolvedReason::AttemptsExhausted); - CasUnresolvedReason pause_reason = CasUnresolvedReason::AttemptsExhausted; - if (!pauseBeforeReissue(attempt, deadline_ms, fence_ok, &pause_reason)) - return unresolved(pause_reason); - continue; - } - } - - /// attempt_outcome == Committed here (either the attempt's own 2xx, or resolution found - /// identical bytes). Final fence check before reporting success: a fence lost here means the - /// write may have landed but this call must never claim it did. Count this "response observed - /// after the local fence" leg separately from the generic Unresolved classifier so a cross-epoch - /// fence loss is visible rather than folded into ordinary retry-budget exhaustion. - if (!fence_ok()) - { - ProfileEvents::increment(ProfileEvents::CASConditionalWriteFenceLostPostWrite); - return unresolved(CasUnresolvedReason::FenceLostPostWrite); - } - if (out_token) - *out_token = committed_token; - return CasWriteOutcome::Committed; - } - - return unresolved(CasUnresolvedReason::AttemptsExhausted); /// budget exhausted, no definite outcome -} - -CasOverwriteResult CasRequestController::putOverwriteControlled( - std::string_view key, std::string_view bytes, const Token & expected, const std::function & fence_ok) -{ - const uint64_t operation_start_ms = now_ms(); - const uint64_t deadline_ms = saturatingAdd(operation_start_ms, budget.operation_deadline_ms); - const CasOverwriteOperationContext context{ - .absolute_deadline_ms = deadline_ms, - .deadline_source = CasOverwriteDeadlineSource::RequestBudget, - .stop_cause = [&fence_ok] - { - return fence_ok() ? CasOverwriteStopCause::Continue : CasOverwriteStopCause::FenceOrLifecycleLost; - }, - .wait_before_retry = [this](uint64_t wait_ms) - { - sleep_ms(wait_ms); - return true; - }, - .observe = [](const CasOverwriteProgress &) {}, - }; - return putOverwriteControlledImpl(key, bytes, expected, context, /*preserve_legacy_gates=*/true); -} - -CasOverwriteResult CasRequestController::putOverwriteControlled( - std::string_view key, - std::string_view bytes, - const Token & expected, - const CasOverwriteOperationContext & context) -{ - return putOverwriteControlledImpl(key, bytes, expected, context, /*preserve_legacy_gates=*/false); -} - -CasOverwriteResult CasRequestController::putOverwriteControlledImpl( - std::string_view key, - std::string_view bytes, - const Token & expected, - const CasOverwriteOperationContext & context, - bool preserve_legacy_gates) -{ - const String key_s{key}; - const String bytes_s{bytes}; - CasOverwriteDiagnostics diagnostics; - diagnostics.deadline_source = context.deadline_source; - bool ambiguity_observed = false; - bool earlier_attempt_unresolved = false; - bool observer_failure_reported = false; - - const auto observe = [&](CasOverwriteProgressKind kind, uint32_t attempt_no) noexcept - { - try - { - context.observe(CasOverwriteProgress{kind, attempt_no}); - } - catch (...) - { - if (!observer_failure_reported) - { - observer_failure_reported = true; - try - { - LOG_DEBUG(getLogger("CasRequestControl"), - "CAS overwrite progress observer threw; suppressing this and further observer exceptions"); - } - catch (...) - { - } - } - } - }; - - const auto unresolved = [&](CasUnresolvedReason reason, CasOverwriteStopCause stop_cause) - { - diagnostics.unresolved_reason = reason; - diagnostics.stop_cause = stop_cause; - return CasOverwriteResult{CasOverwriteOutcome::Unresolved, {}, diagnostics}; - }; - - const auto resultForGate = [&](const OverwriteGateSample & sample, bool commit_proved) - -> std::optional - { - if (sample.closure == OverwriteGateClosure::Open) - return std::nullopt; - - if (sample.closure == OverwriteGateClosure::Deadline) - { - return unresolved( - diagnostics.attempts_sent == 0 ? CasUnresolvedReason::NoAttemptSent : CasUnresolvedReason::DeadlineMidWay, - CasOverwriteStopCause::Continue); - } - - const CasUnresolvedReason reason = diagnostics.attempts_sent == 0 - ? CasUnresolvedReason::NoAttemptSent - : (commit_proved ? CasUnresolvedReason::FenceLostPostWrite : CasUnresolvedReason::FenceLostMidWay); - if (commit_proved) - ProfileEvents::increment(ProfileEvents::CASConditionalWriteFenceLostPostWrite); - return unresolved(reason, sample.stop_cause); - }; - - const auto gate = [&](uint64_t required_time_ms, bool commit_proved = false) - -> std::optional - { - return resultForGate(sampleOverwriteGate(context, now_ms, required_time_ms), commit_proved); - }; - - /// The existing overload historically checked its fence before each `PUT`/sleep and after a - /// proven commit, but did not add clock/fence samples around the resolving `GET`. Keep that exact - /// schedule while adapting its callbacks into the context representation; otherwise an injected - /// clock that advances per sample loses a physical attempt solely because the adapter observed it. - const auto legacyGate = [&](uint64_t required_time_ms, bool commit_proved, bool check_deadline) - -> std::optional - { - const CasOverwriteStopCause stop_cause = context.stop_cause(); - if (stop_cause != CasOverwriteStopCause::Continue) - { - const OverwriteGateClosure closure = stop_cause == CasOverwriteStopCause::FenceOrLifecycleLost - ? OverwriteGateClosure::FenceOrLifecycleLost - : OverwriteGateClosure::Cancelled; - return resultForGate({closure, stop_cause}, commit_proved); - } - if (!check_deadline) - return std::nullopt; - - const uint64_t now = now_ms(); - if (now > context.absolute_deadline_ms || required_time_ms > context.absolute_deadline_ms - now) - return resultForGate({OverwriteGateClosure::Deadline, CasOverwriteStopCause::Continue}, commit_proved); - return std::nullopt; - }; - - while (true) - { - const auto pre_put_refusal = preserve_legacy_gates - ? legacyGate(budget.attempt_timeout_ms, /*commit_proved=*/false, /*check_deadline=*/true) - : gate(budget.attempt_timeout_ms); - if (pre_put_refusal) - return *pre_put_refusal; - if (diagnostics.attempts_sent >= budget.max_attempts) - return unresolved(CasUnresolvedReason::AttemptsExhausted, CasOverwriteStopCause::Continue); - - const uint32_t attempt_no = diagnostics.attempts_sent + 1; - if (attempt_no > 1) - observe(CasOverwriteProgressKind::RetryStarted, attempt_no); - ++diagnostics.attempts_sent; - observe(CasOverwriteProgressKind::PutStarted, attempt_no); - - std::optional put; - bool attempt_may_still_land = false; - try - { - put = backend->putOverwrite(key_s, bytes_s, expected); - } - catch (const std::exception & e) - { - /// A deterministic local bug or a whitelisted synchronous rejection proves no retry can - /// help. It surfaces unchanged unless an earlier request from this logical operation is - /// still ambiguous; that earlier request dominates the call-wide result because it may - /// still land after this later attempt was refused. - const auto * db_e = dynamic_cast(&e); - const bool definite_failure = (db_e && isDeterministicLocalFailure(db_e->code())) - || classifyConditionalWriteResult(e) == CasWriteOutcome::DefiniteFailure; - if (definite_failure) - { - if (!preserve_legacy_gates) - { - if (auto refusal = gate(/*required_time_ms=*/0)) - return *refusal; - if (earlier_attempt_unresolved) - return unresolved( - CasUnresolvedReason::DefiniteFailureAfterAmbiguity, - CasOverwriteStopCause::Continue); - } - throw; - } - attempt_may_still_land = true; - /// Else ambiguous -- fall through to resolve below. - } - - if (put && put->outcome == PutOutcome::Done) - { - const auto post_write_refusal = preserve_legacy_gates - ? legacyGate(/*required_time_ms=*/0, /*commit_proved=*/true, /*check_deadline=*/false) - : gate(/*required_time_ms=*/0, /*commit_proved=*/true); - if (post_write_refusal) - return *post_write_refusal; - return {CasOverwriteOutcome::Committed, put->token, diagnostics}; - } - - /// Ambiguous: either a caught transient exception, or PreconditionFailed (which alone does - /// NOT prove a real conflict -- it may be our own earlier attempt's write landing under a - /// concurrent resolve). Resolve with one GET. - if (!ambiguity_observed) - { - ambiguity_observed = true; - observe(CasOverwriteProgressKind::BecameAmbiguous, attempt_no); - } - if (!preserve_legacy_gates) - { - if (auto refusal = gate(budget.attempt_timeout_ms)) - return *refusal; - } - - observe(CasOverwriteProgressKind::ResolveStarted, attempt_no); - std::optional got; - try - { - got = backend->get(key_s); - diagnostics.resolve_observation_completed = true; - diagnostics.observed_bytes = got ? std::optional{got->bytes} : std::nullopt; - } - catch (const std::exception &) - { - diagnostics.resolve_observation_completed = false; - diagnostics.observed_bytes.reset(); - got.reset(); /// GET failed: still ambiguous, fall through to retry below. - } - - if (got && got->token != expected && got->bytes == bytes_s) - { - diagnostics.resolved_by_get = true; - observe(CasOverwriteProgressKind::ResolvedByGet, attempt_no); - const auto post_write_refusal = preserve_legacy_gates - ? legacyGate(/*required_time_ms=*/0, /*commit_proved=*/true, /*check_deadline=*/false) - : gate(/*required_time_ms=*/0, /*commit_proved=*/true); - if (post_write_refusal) - return *post_write_refusal; - return {CasOverwriteOutcome::Committed, got->token, diagnostics}; - } - - /// Apply stop/deadline precedence to the completed resolve before accepting a conflict, - /// waiting, or letting attempt exhaustion decide whether another `PUT` is legal. - if (!preserve_legacy_gates) - { - if (auto refusal = gate(/*required_time_ms=*/0)) - return *refusal; - } - - if (got && got->token == expected) - { - /// The token we CAS'd against is STILL current: our attempt never applied. Fall through - /// to the pause-and-reissue gate below (same key, bytes, expected). - } - else if (got) - { - /// A DIFFERENT token AND different bytes: a genuine competing write. Real conflict -- - /// never collapsed into Unresolved/DefiniteFailure, never thrown. - return {CasOverwriteOutcome::Conflict, {}, diagnostics}; - } - /// else: the GET itself failed or the key vanished -- still ambiguous, fall through to retry. - - if (attempt_may_still_land) - earlier_attempt_unresolved = true; - - /// Attempt exhaustion participates only when another physical `PUT` would be sent. It is - /// deliberately evaluated after the final attempt's resolving `GET` and after the gate above. - if (diagnostics.attempts_sent >= budget.max_attempts) - { - if (!preserve_legacy_gates) - { - if (auto refusal = gate(budget.attempt_timeout_ms)) - return *refusal; - } - return unresolved(CasUnresolvedReason::AttemptsExhausted, CasOverwriteStopCause::Continue); - } - - const uint64_t backoff_ms = backoffBeforeAttempt(attempt_no + 1); - if (backoff_ms == 0) - continue; - - const uint64_t retry_reservation_ms = saturatingAdd(backoff_ms, budget.attempt_timeout_ms); - const auto pre_wait_refusal = preserve_legacy_gates - ? legacyGate(retry_reservation_ms, /*commit_proved=*/false, /*check_deadline=*/true) - : gate(retry_reservation_ms); - if (pre_wait_refusal) - return *pre_wait_refusal; - - const bool wait_completed = context.wait_before_retry(backoff_ms); - if (preserve_legacy_gates) - continue; - - const OverwriteGateSample after_wait = sampleOverwriteGate(context, now_ms, budget.attempt_timeout_ms); - if (!wait_completed && after_wait.stop_cause == CasOverwriteStopCause::Continue) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CasRequestController: wait_before_retry returned false while stop_cause remained Continue"); - if (auto refusal = resultForGate(after_wait, /*commit_proved=*/false)) - return *refusal; - } -} - -CasOverwriteResult CasRequestController::putIfAbsentControlledMutable( - std::string_view key, std::string_view bytes, const std::function & fence_ok) -{ - const String key_s{key}; - const String bytes_s{bytes}; - const uint64_t deadline_ms = now_ms() + budget.operation_deadline_ms; - - for (uint32_t attempt_no = 1; attempt_no <= budget.max_attempts; ++attempt_no) - { - if (!fence_ok()) - return {CasOverwriteOutcome::Unresolved, {}, {}}; - if (now_ms() + budget.attempt_timeout_ms > deadline_ms) - return {CasOverwriteOutcome::Unresolved, {}, {}}; - - std::optional put; - try - { - put = backend->putIfAbsent(key_s, bytes_s); - } - catch (const std::exception & e) - { - /// Same rethrow convention as `putOverwriteControlled`. - if (const auto * db_e = dynamic_cast(&e); db_e && isDeterministicLocalFailure(db_e->code())) - throw; - if (classifyConditionalWriteResult(e) == CasWriteOutcome::DefiniteFailure) - throw; - /// Else ambiguous -- fall through to resolve below. - } - - if (put && put->outcome == PutOutcome::Done) - { - if (!fence_ok()) - { - ProfileEvents::increment(ProfileEvents::CASConditionalWriteFenceLostPostWrite); - return {CasOverwriteOutcome::Unresolved, {}, {}}; - } - return {CasOverwriteOutcome::Committed, put->token, {}}; - } - - /// Ambiguous: either a caught transient exception, or PreconditionFailed (which alone does - /// NOT prove a real conflict -- it may be our own earlier attempt's write landing under a - /// concurrent resolve, or a racing writer creating the identical value). Resolve with one GET. - std::optional got; - try - { - got = backend->get(key_s); - } - catch (const std::exception &) - { - got.reset(); /// GET failed: still ambiguous, fall through to retry below. - } - - if (!got) - { - /// Still absent: our attempt never applied. Fall through to the pause-and-reissue gate - /// below (same key, bytes). - } - else if (got->bytes == bytes_s) - { - if (!fence_ok()) - { - ProfileEvents::increment(ProfileEvents::CASConditionalWriteFenceLostPostWrite); - return {CasOverwriteOutcome::Unresolved, {}, {}}; - } - return {CasOverwriteOutcome::Committed, got->token, {}}; - } - else - { - /// Present with DIFFERENT bytes: something else already occupies the key with a - /// different value. For a MUTABLE marker this is a normal outcome, not corruption -- - /// return it as a value, never thrown. - return {CasOverwriteOutcome::Conflict, {}, {}}; - } - - if (attempt_no == budget.max_attempts || !pauseBeforeReissue(attempt_no, deadline_ms, fence_ok)) - return {CasOverwriteOutcome::Unresolved, {}, {}}; - } - - return {CasOverwriteOutcome::Unresolved, {}, {}}; /// attempt budget exhausted without a definite outcome -} - -SlotOccupyResult CasRequestController::slotOccupy( - std::string_view key, std::string_view bytes, const std::function & fence_ok) -{ - const String key_s{key}; - const String bytes_s{bytes}; - const uint64_t deadline_ms = now_ms() + budget.operation_deadline_ms; - - /// Pre-attempt gate -- the same two checks every controlled op runs before its first (here, only) - /// attempt: the mount fence must still hold, and there must be enough of the operation's own - /// deadline left for one attempt to plausibly complete. Neither check sends anything to the - /// backend, so a refusal here PROVES the key is untouched by this call. UNLIKE every sibling - /// controlled op, there is no post-write result recheck below: the stronger post-I/O consistency - /// check (fence generation together with wedge/txn identity) is the CALLER's contract (Task 4/6's - /// re-acquire-lock-and-checkFenceOrThrow step), not this raw primitive's. The admission predicate is - /// checked again only if this attempt needs a second backend request to resolve its result; a - /// `Created` result still performs no post-write recheck. - if (!fence_ok() || now_ms() + budget.attempt_timeout_ms > deadline_ms) - return {.kind = SlotOccupyResult::Kind::Unresolved, .occupant_bytes = {}, .occupant_token = {}, - .unresolved_reason = CasUnresolvedReason::NoAttemptSent}; - - std::optional put; - try - { - put = backend->putIfAbsent(key_s, bytes_s); - } - catch (const std::exception & e) - { - /// Same rethrow convention as putOverwriteControlled/putIfAbsentControlledMutable: a - /// deterministic local bug, or a whitelisted synchronous rejection that PROVES the request was - /// never applied, surfaces unchanged -- SlotOccupyResult::Kind has no DefiniteFailure member to - /// carry either one. Anything else is ambiguous: fall through to the raw resolve GET below, - /// exactly like a clean PreconditionFailed -- this primitive cannot and does not distinguish - /// the two. - if (const auto * db_e = dynamic_cast(&e); db_e && isDeterministicLocalFailure(db_e->code())) - throw; - if (classifyConditionalWriteResult(e) == CasWriteOutcome::DefiniteFailure) - throw; - } - - if (put && put->outcome == PutOutcome::Done) - return {.kind = SlotOccupyResult::Kind::Created, .occupant_bytes = {}, .occupant_token = {}, - .unresolved_reason = CasUnresolvedReason::NotUnresolved}; - - /// Ambiguous attempt or a clean conflict: resolve with exactly ONE raw exact GET -- no byte-compare, - /// no throw on a different occupant [codex finding 3: this is a DEDICATED slot operation, not - /// putIfAbsentControlled (which retries the same (key, bytes) internally) or resolveByExactGet - /// (which compares against an expected body and throws CORRUPTED_DATA on a mismatch) composed - /// together]. Adjudicating whether the occupant is "mine" is entirely the CALLER's job (the - /// CaCasMountCore `mine` contract), never this primitive's. - /// - /// Whole-object resolution is safe here because `slotOccupy` is scoped by its callers (Task 4/6, - /// spec INV-2) to small, write-once control slots -- ref-log transactions and epoch seals -- whose - /// size is bounded by their own format's registry cap (the strict-grammar object caps - /// CasRefLogFormat/CasRefCkptFormat enforce on decode), never a data blob. slotOccupy itself stays - /// format-agnostic (it takes a raw key/bytes pair, per the "Interface handed to Stage B" contract in - /// the plan) and does not encode any format's cap here -- the size bound is a property of what - /// callers are allowed to pass it, enforced where the returned bytes are decoded, not by this seam. - if (!fence_ok()) - return {.kind = SlotOccupyResult::Kind::Unresolved, .occupant_bytes = {}, .occupant_token = {}, - .unresolved_reason = CasUnresolvedReason::AttemptsExhausted}; - - std::optional got; - try - { - got = backend->get(key_s); - } - catch (const std::exception &) - { - got.reset(); /// the GET itself failed: still unresolved -- a one-shot primitive never retries - } - - if (!got) - /// The occupant that caused the conflict vanished before this GET (or the GET itself failed): - /// the outcome is unknowable right now -- NEVER a fabricated Created. - return {.kind = SlotOccupyResult::Kind::Unresolved, .occupant_bytes = {}, .occupant_token = {}, - .unresolved_reason = CasUnresolvedReason::AttemptsExhausted}; - - return {.kind = SlotOccupyResult::Kind::Occupied, .occupant_bytes = std::move(got->bytes), - .occupant_token = got->token, .unresolved_reason = CasUnresolvedReason::NotUnresolved}; -} - -} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.h deleted file mode 100644 index e01534d48a2f..000000000000 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestControl.h +++ /dev/null @@ -1,635 +0,0 @@ -#pragma once -#include -#include -#include -#include -#include -#include -#include - -namespace DB::Cas -{ - -/// Outcome of ONE HTTP attempt for a CAS conditional write (`If-None-Match`/`If-Match`), issued with -/// the generic S3 client's transparent retries disabled for that attempt. This is the seam the -/// `CasRequestController` is built on: it decides whether another attempt is legal and how an -/// uncertain result is resolved. -/// - Committed: the attempt's own request completed successfully (2xx) — the object is durable. -/// - DefiniteFailure: a synchronous rejection that PROVES the request was never applied server-side -/// — a WHITELISTED malformed-request / entity-too-large / access-denied error ONLY. Never -/// `PreconditionFailed`: a lost precondition means the key exists, not that the request failed. -/// - Unresolved: everything else — `PreconditionFailed`/`NoSuchKey`, a client-side timeout, a -/// connection loss, a 5xx, or any error this classifier does not recognize. The caller resolves -/// the exact key before deciding whether another attempt is legal; ambiguity always -/// resolves toward Unresolved, never toward a false DefiniteFailure or a false Committed. -enum class CasWriteOutcome : uint8_t -{ - Committed, - DefiniteFailure, - Unresolved, -}; - -/// WHY a controlled write came back `Unresolved`. It exists because `Unresolved` covers two materially -/// different states, and telling them apart is the difference between a five-minute triage and an hour -/// of it — and, since finding #37 defect 3, between a table that keeps its write availability and one -/// that loses it until remount. -/// -/// `NoAttemptSent` is the one that carries real information: both pre-attempt gates (the mount fence -/// and the operation deadline) reject BEFORE anything reaches the backend, so on the very first -/// iteration the key is PROVABLY unwritten — there is no ambiguity to resolve, only a lost right to -/// write. Every other reason leaves an object that may or may not be durable, which is what the -/// wedge/resolve machinery exists for. -/// -/// NOT purely diagnostic any more: `unresolvedProvesNothingWasSent` below turns this into the fact the -/// ref append lane acts on (`CasRefLedger::commitRefChunk`'s `Unresolved` arm), so ADDING A MEMBER HERE -/// IS A PROTOCOL DECISION — read that predicate before you do. -enum class CasUnresolvedReason : uint8_t -{ - NotUnresolved, /// the call did not return Unresolved - NoAttemptSent, /// a pre-attempt gate rejected on the FIRST iteration: nothing was ever sent - FenceLostMidWay, /// >= 1 attempt was sent, then the mount fence dropped - DeadlineMidWay, /// >= 1 attempt was sent, then the operation deadline left no room for another - FenceLostPostWrite,/// an attempt COMMITTED but the fence had dropped by the time it returned - AttemptsExhausted, /// the genuine case the "retry budget exhausted" wording describes - /// A LATER attempt was definitively refused while an EARLIER one of the same call is still - /// unresolved. The refusal proves only its own attempt never applied; the earlier one may still - /// materialize at the key, so the CALL cannot report `DefiniteFailure` (see - /// `putIfAbsentControlled`). Reported instead of the definite verdict, never alongside it. - DefiniteFailureAfterAmbiguity, -}; - -/// Does this `Unresolved` PROVE that no attempt ever reached the network — i.e. that the key is -/// unwritten and there is nothing for an exact-key resolution to settle? -/// -/// True for exactly ONE value, and that is the whole design: `NoAttemptSent` is reported only when a -/// pre-attempt gate rejected while `attempts_sent == 0`, so `backend->putIfAbsent` was never called -/// (see `putIfAbsentControlled`). Every other value — including `NotUnresolved`, which a caller can -/// still observe if some path returns `Unresolved` without recording a reason — leaves an object that -/// MAY be durable, and callers that protect themselves against that (the ref lane's append wedge) must -/// keep doing so. -/// -/// Written as an allow-list: a switch with no `default` and a trailing `return false`, so a member -/// added to `CasUnresolvedReason` later fails BOTH ways safely. The missing case is a `-Wswitch` build -/// error, which forces the contributor to classify it deliberately; and if that diagnostic is ever -/// silenced, the runtime answer for the unclassified member is "no, this does not prove anything", -/// which is the conservative side. Never turn this into a deny-list — a new reason must not be able to -/// claim "nothing was sent" by omission. -constexpr bool unresolvedProvesNothingWasSent(CasUnresolvedReason reason) -{ - switch (reason) - { - case CasUnresolvedReason::NoAttemptSent: - return true; - case CasUnresolvedReason::NotUnresolved: - case CasUnresolvedReason::FenceLostMidWay: - case CasUnresolvedReason::DeadlineMidWay: - case CasUnresolvedReason::FenceLostPostWrite: - case CasUnresolvedReason::AttemptsExhausted: - /// The whole point of this value is that an earlier attempt WAS sent and may still land. - case CasUnresolvedReason::DefiniteFailureAfterAmbiguity: - return false; - } - return false; -} - -/// Human-readable tail for an exception or log line, so the two states above stop reading alike. -constexpr std::string_view describeUnresolvedReason(CasUnresolvedReason reason) -{ - switch (reason) - { - case CasUnresolvedReason::NotUnresolved: return "not unresolved"; - case CasUnresolvedReason::NoAttemptSent: return "no attempt was sent (the mount fence or the " - "operation deadline rejected before the first " - "request) — the key is provably unwritten"; - case CasUnresolvedReason::FenceLostMidWay: return "the mount fence dropped after at least one " - "attempt had been sent"; - case CasUnresolvedReason::DeadlineMidWay: return "the operation deadline ran out after at least " - "one attempt had been sent"; - case CasUnresolvedReason::FenceLostPostWrite: return "an attempt committed but the mount fence had " - "dropped before it returned"; - case CasUnresolvedReason::AttemptsExhausted: return "the attempt budget was exhausted without a " - "definite outcome"; - case CasUnresolvedReason::DefiniteFailureAfterAmbiguity: - return "a later attempt was definitively refused, but " - "an earlier attempt of the same call is still " - "unresolved and may yet land"; - } - return "unspecified"; -} - -/// The success path: `buf.finalize()` returned without throwing. Always Committed — kept as a named, -/// counted entry point so both paths of a classify-then-record call site read the same way (see the -/// exception overload below). -constexpr CasWriteOutcome classifyConditionalWriteResult() -{ - return CasWriteOutcome::Committed; -} - -/// The exception path: classify what `buf.finalize()` threw for ONE CAS conditional-write HTTP -/// attempt, according to the CAS conditional-write operation classes. Pure — never rethrows, never -/// touches counters; see recordConditionalWriteOutcome for the counters hookup. -CasWriteOutcome classifyConditionalWriteResult(const std::exception & e); - -/// Records the start of one HTTP attempt for a CAS conditional write (the attempts counter). -void recordConditionalWriteAttemptStarted(); - -/// Records one attempt's terminal outcome (the per-class outcome counters). Callers pass the result of -/// whichever classifyConditionalWriteResult overload applies, or an outcome already known by -/// construction (e.g. the legacy `PutOutcome::PreconditionFailed` path, which today resolves without -/// throwing — see ObjectStorageBackend::nativeConditionalPut). -void recordConditionalWriteOutcome(CasWriteOutcome outcome); - -/// The three separate limits a CAS-owned retry controller enforces for ONE logical conditional-write -/// operation. Never represented by a single `request_timeout_ms` value — see `validateCasRequestBudget` -/// for the relationship a writable mount enforces at startup, and `CasRequestController` for the -/// runtime use. -struct CasRequestBudget -{ - /// Maximum client wait budgeted for one HTTP attempt. `CasRequestController` uses this ONLY as a - /// per-attempt scheduling check (an attempt is not started unless it could still finish inside the - /// operation deadline) — the actual socket-level wait is configured on the object storage's client - /// (the object storage backend's single-attempt client), not by this struct. - uint64_t attempt_timeout_ms = 5000; - /// Maximum wall-clock time for the COMPLETE logical operation — every attempt, every exact-key - /// resolution, and every inter-attempt backoff sleep — counted from the first call to - /// `putIfAbsentControlled`. A DURATION, not an absolute deadline: each call establishes its own - /// `now + operation_deadline_ms` bound. - /// - /// This deadline is the authoritative bound on how long a CAS conditional write keeps riding an S3 - /// disruption server-side before the caller sees an abort. 90s absorbs a ~60s object-store outage - /// with margin (see the arithmetic on `max_attempts` below) — PROVIDED the mount fence stays alive. - /// The fence, not this deadline, is - /// what binds under a TOTAL outage: lease renewals are conditional writes against the same store, - /// so when everything is unreachable the fence deadline freezes at `last_renew + mount_lease_ttl` - /// and `fence_ok` stops the loop ≈ TTL−attempt_timeout−margin (~23s) after the last successful - /// renewal — the required fail-closed behavior (never an attempt past the lease), not a - /// budget limitation. While renewals DO land (blips, throttling, partial outages — the runtime-owned - /// renewal worker keeps extending the fence deadline), the op is NOT bounded by - /// the lease TTL and rides the full deadline here. - uint64_t operation_deadline_ms = 90000; - /// Maximum number of controlled attempts for one logical operation (the first attempt counts as 1). - /// Sized so the operation deadline above — never this count — is what binds under the observed - /// failure shape (~3s adaptive first-attempt PUT timeout per failed attempt + capped-exponential - /// backoff): 16 attempts × ~3s + Σ backoff (0.2+0.4+0.8+1.6+3.2 + 10×5 = 56.2s) ≈ 104s > 90s. - uint32_t max_attempts = 16; - /// Startup-only margin folded into `validateCasRequestBudget`'s inequality against the mount lease - /// TTL. Not consulted at runtime by the controller itself — the caller's `fence_ok` callback (backed - /// by the local write fence's own deadline) is what actually gates lease-relative timing per attempt. - uint64_t lease_safety_margin_ms = 2000; - /// Inter-attempt backoff (`cas_s3_retry_initial_backoff_ms` / - /// `cas_s3_retry_max_backoff_ms`): the sleep before reissuing - /// after an ambiguous attempt whose resolve observed the key absent, capped exponential — - /// `initial · 2^(reissues-1)`, never above `retry_max_backoff_ms`. 0 disables backoff (immediate - /// reissue — the pre-backoff behavior, and what most exhaustion-path unit tests configure). The - /// controller checks the fence BEFORE every sleep and never sleeps past the operation deadline. - uint64_t retry_initial_backoff_ms = 200; - uint64_t retry_max_backoff_ms = 5000; - - /// Recovery-level retry (`CasRefLedger::ensureRefTableRecovered`): a whole ref-table recovery - /// attempt (LIST + snapshot/log GETs + seal PUT) that fails with a transient NETWORK_ERROR is - /// retried, with capped-exponential backoff, until this total wall-clock budget is spent — then the - /// error propagates and the table's load fails for this touch (the `lazy_load_tables` database - /// setting makes the NEXT touch retry). This sits ON TOP of the per-request `operation_deadline_ms` - /// envelope above: one recovery attempt may itself burn ~90s inside a single seal PUT. Independent - /// of the mount-lease invariants validated in `validateCasRequestBudget` — not part of that - /// inequality set. - uint64_t recovery_retry_budget_ms = 120000; - uint64_t recovery_retry_initial_backoff_ms = 1000; - uint64_t recovery_retry_max_backoff_ms = 30000; -}; - -/// Startup validation: a writable mount refuses to open with an inconsistent budget rather than -/// silently falling back to an unbounded or unsafe retry policy. Throws -/// `BAD_ARGUMENTS` unless ALL hold: -/// attempt_timeout_ms + lease_safety_margin_ms < mount_lease_ttl_ms -/// attempt_timeout_ms < operation_deadline_ms (STRICTLY — see below) -/// retry_initial_backoff_ms <= retry_max_backoff_ms -/// -/// The middle one is strict on purpose. Equality does not mean "one attempt's worth of budget": the -/// deadline is captured as `now + operation_deadline_ms` and each pre-send gate asks -/// `now + attempt_timeout_ms > deadline_ms`, so equal values reduce it to `now_2 > now_1` and a single -/// elapsed millisecond refuses the operation having sent NOTHING. Bound the attempt COUNT with -/// `max_attempts`, never by starving the deadline. -/// `mount_renew_period_ms` takes no part in the inequality (the renewer keeps the fence deadline -/// refreshed well ahead of the TTL by construction) — it is accepted only so the effective-values log -/// line records the full picture in one place. -/// -/// A successor mounting over an unclean predecessor waits at least one lease TTL, plus its -/// materialization grace period, before trusting recovery listings. This is long enough for any -/// conditional PUT still in flight at the predecessor to either land or be abandoned by its own -/// exhausted retry budget. The predecessor's budget is constrained by -/// `attempt_timeout_ms + lease_safety_margin_ms < mount_lease_ttl_ms`, so no additional handover -/// check is needed here. -void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, uint64_t mount_renew_period_ms); - -/// Throw the recoverable "CAS write could not be committed, retry later" condition. -/// -/// WHY NETWORK_ERROR (this replaces an earlier ABORTED throw): -/// A content-addressed write can fail for a reason that is neither the caller's fault -/// nor permanent: the mount-lease / write fence was lost (e.g. a renewal PUT timed out -/// against a slow or throttling object store), or a conditional PUT exhausted its retry -/// budget mid-outage. The right response is "abandon this attempt, try again later" -- -/// which is precisely what a transient error means. -/// -/// It previously threw `ABORTED`, which was actively harmful to background merges: -/// `ReplicatedMergeMutateTaskBase` treats `ABORTED` as "merge deliberately cancelled -/// (shutdown / `DROP` / merges-blocker), not an error", so it neither records -/// `last_exception_time_ms` nor lets `ReplicatedMergeTreeQueue`'s exponential backoff -/// engage. Under a sustained store outage the queue re-executed the merge roughly every -/// 2 seconds, recomputing the whole (possibly multi-GiB) output part every time for the -/// entire outage -- hundreds of full recomputes, and invisible in system.replication_queue. -/// -/// `NETWORK_ERROR` is the best-fitting EXISTING code: -/// - it is NOT in the merge "retry silently, no backoff" exemption set (only `ABORTED` -/// and `PART_IS_TEMPORARILY_LOCKED` are), so the existing backoff -- capped by -/// `max_postpone_time_for_failed_replicated_merges_ms` -- engages automatically; -/// - it is already in ClickHouse's transient/retryable taxonomy -/// (`checkDataPart::isRetryableException` lists it beside `ABORTED`), so a part under -/// verification is not misread as corrupted; -/// - nothing on the merge / insert / replication commit path special-cases it in a way -/// that would misfire (ZooKeeper retriability keys on `Coordination::Exception`, a -/// different type), and it is not caught specially on the CAS write path. -/// -/// Honest caveat: `NETWORK_ERROR` is coarser than the true condition. For the -/// throttled-store / timed-out / lost-lease cases it is accurate; for a purely logical -/// fence loss (e.g. the namespace is being dropped) it slightly overstates "network". -/// The precise cause is always in the exception MESSAGE, never inferred from the code. -/// -/// If that imprecision ever matters -- operator confusion, or a future upstream change -/// that attaches merge-path handling to `NETWORK_ERROR` and reintroduces a collision -- -/// switch to a dedicated code (e.g. CAS_WRITE_RETRY_LATER) by changing the -/// single throw below. A dedicated code is honest and collision-proof by construction -/// (backoff still engages, since only `ABORTED` / `PART_IS_TEMPORARILY_LOCKED` are exempt); -/// the only extra work is one appended line in `ErrorCodes.cpp` and, optionally, adding it -/// to `checkDataPart::isRetryableException` and an HTTP-status mapping for the foreground -/// `INSERT` client. We deliberately kept `NETWORK_ERROR` for now to add zero new coupling to -/// generic ClickHouse code, consistent with the rest of the CAS layer. -/// -/// SCOPE: only the ESCAPING retry-later throws route here (fence lost or a controlled write outcome -/// remaining uncertain). Startup/decommission and generic live-lock-brake `ABORTED` values keep -/// their meaning and are not rerouted here. -[[noreturn]] void throwCasWriteRetryLater(const String & why); - -/// Same classification as `throwCasWriteRetryLater`, but returns the exception as a -/// `std::exception_ptr` for call sites that fail a pending future/promise (`CasRefLedger`'s -/// `complete_error`) rather than throw directly. Both entry points route through the SAME -/// construction internally, so the error code / message shape has exactly one place that decides it. -std::exception_ptr makeCasWriteRetryLaterExceptionPtr(const String & why); - -/// Throw the recoverable "this content-addressed disk cannot serve the request right now" condition. -/// Sibling of `throwCasWriteRetryLater`, same class for the same reasons (see the long rationale above), -/// differing only in what it describes: that one names a WRITE whose commit did not land, this one names -/// a DISK STATE that refused the request before it started -- on either plane. -/// -/// The class is load-bearing beyond CAS. `ReplicatedMergeTreePartCheckThread::checkPartImpl` rethrows -/// (leaving the part queued for a later check) exactly when `checkDataPart::isRetryableException` -/// recognises the error, and otherwise declares the part broken -- detach and re-fetch. -/// `INVALID_STATE` is absent from that classifier, so a lease blip used to read as part corruption -/// (BACKLOG `{#lease-blip-part-check-collapse}`). Re-coding the CA transients -/// was chosen over widening the upstream classifier because `INVALID_STATE` is broad: widening it would -/// also reclassify 18 unrelated TERMINAL sites, CA and non-CA alike. -/// -/// SCOPE, narrow by design: a refusal routes here when it either names an AUTO-RECOVERING disk condition, -/// or CANNOT ESTABLISH that its condition is terminal. `checkFenceOrThrow` is the second kind -- one guard -/// trips for a lease blip and for a FORGET decommission alike and it cannot tell them apart -- and it is in -/// scope for write-plane uniformity: its 32 sibling write-transient sites already mint this class, and an -/// unproven condition must be retried rather than consumed as damage. What is NEVER in scope is a refusal -/// whose condition is PROVEN terminal: `IdentityLost`, both `Vanished` flavours, a storage that is not -/// started, an unbootstrappable prefix, a proven-absent pool identity, a closed writer epoch -- all keep -/// `INVALID_STATE`. A proven-terminal state that read as retryable would make every consumer retry forever -/// against a disk that is never coming back. -/// -/// `subject` names the refusing disk or pool (e.g. "content-addressed disk 'ca'") and `condition` states -/// the CA condition truthfully, INCLUDING any promise about how it clears -- only the site knows whether -/// it can make one. What is appended HERE is the classification alone, so it cannot drift between call -/// sites. Unlike `throwCasWriteRetryLater` this deliberately does not log: these sites fire once per -/// refused operation (tens of thousands within a single observed lease gap) and every caller already -/// reports the exception it receives. -[[noreturn]] void throwCasTransientUnavailable(const String & subject, const String & condition); - -/// Outcome of a controlled MUTABLE conditional overwrite (`putOverwriteControlled`) -- an If-Match -/// replace whose caller can, unlike a content-addressed create, supply the intended bytes for -/// GET-based resolution, because the payload here is deterministic (a pure function of the -/// caller's record), not freshly minted per attempt. -/// - Committed: an attempt's own request completed (2xx) and the final fence check held, or -/// resolution proved the intended bytes are already what's currently stored -- `token` names -/// that incarnation. -/// - Conflict: resolution proved the key's CURRENT token AND bytes both differ from what this -/// call intended -- a genuine competing write. Returned as a value, never thrown, never -/// collapsed into Unresolved/DefiniteFailure -- mirrors the existing uncontrolled -/// casMeta/CasResult contract (a conflict lets the caller reload and decide). -/// - Unresolved: budget exhausted, fence lost, or the current token still equals `expected` (the -/// attempt provably never applied) with the resolve unable to prove either outcome yet -- -/// caller must not ACK. -enum class CasOverwriteOutcome : uint8_t -{ - Committed, - Conflict, - Unresolved, -}; - -enum class CasOverwriteDeadlineSource : uint8_t -{ - RequestBudget, - ExternalLeaseSafety, -}; - -enum class CasOverwriteStopCause : uint8_t -{ - Continue, - Cancelled, - FenceOrLifecycleLost, -}; - -enum class CasOverwriteProgressKind : uint8_t -{ - PutStarted, - BecameAmbiguous, - ResolveStarted, - RetryStarted, - ResolvedByGet, -}; - -struct CasOverwriteProgress -{ - CasOverwriteProgressKind kind; - uint32_t attempt_no; -}; - -/// Per-operation gates for a controlled mutable overwrite. `absolute_deadline_ms` uses the same -/// clock as the controller's injected `now_ms`; it is fixed by the caller before controller entry -/// and therefore cannot be re-anchored after preemption. `wait_before_retry` is interruptible and -/// must return false only after publishing a non-`Continue` stop cause. `observe` is diagnostic only: -/// an exception from it is contained and cannot affect the protocol result. -struct CasOverwriteOperationContext -{ - uint64_t absolute_deadline_ms; - CasOverwriteDeadlineSource deadline_source; - std::function stop_cause; - std::function wait_before_retry; - std::function observe; -}; - -struct CasOverwriteDiagnostics -{ - uint32_t attempts_sent = 0; - bool resolved_by_get = false; - CasUnresolvedReason unresolved_reason = CasUnresolvedReason::NotUnresolved; - CasOverwriteDeadlineSource deadline_source = CasOverwriteDeadlineSource::RequestBudget; - CasOverwriteStopCause stop_cause = CasOverwriteStopCause::Continue; - /// The last exact resolving GET completed by this controller. `resolve_observation_completed` - /// distinguishes a confirmed absence (`observed_bytes == nullopt`) from a failed/not-run read. - /// Terminal protocol owners use this snapshot instead of starting diagnostic I/O after the - /// controller has closed its deadline/cancellation gate. - bool resolve_observation_completed = false; - std::optional observed_bytes; -}; - -/// Result of one `CasRequestController::putOverwriteControlled` operation. `token` is meaningful -/// only when `outcome` is `Committed`. -struct CasOverwriteResult -{ - CasOverwriteOutcome outcome = CasOverwriteOutcome::Unresolved; - Token token; /// set ONLY on Committed - CasOverwriteDiagnostics diagnostics; -}; - -/// Result of one `CasRequestController::slotOccupy` operation — a WRITE-ONCE conditional create whose -/// body is content-addressed or otherwise not byte-comparable across separate CALLS the way -/// `putOverwriteControlled`'s deterministic marker is (each caller of `slotOccupy` — an epoch seal, a -/// wedge retry — mints its own attempt and decides for itself, from `Occupied`'s bytes, whether the -/// occupant is its own earlier write or something else entirely; see the adjudication note below). -/// - Created: this call's OWN conditional create committed — the key held nothing before it. -/// - Occupied: the key already holds an object, observed by ONE raw exact `GET` after the create -/// conflicted — `occupant_bytes`/`occupant_token` name exactly what is there NOW. The primitive -/// never compares these bytes against what this call attempted and never throws on a mismatch: -/// unlike `resolveByExactGet` (whose caller supplies ONE expected body across every retry of the -/// SAME logical attempt), `slotOccupy` never retries, so there is no "our earlier attempt" to -/// distinguish from a genuine foreign occupant — that adjudication (the `CaCasMountCore` `mine` -/// contract: an occupant is this caller's write only if the BYTES match, never a generation/shape -/// match alone) is entirely the CALLER's job. -/// - Unresolved: the outcome is unknowable right now — a pre-attempt gate refused (fence lost / -/// deadline exhausted, `unresolved_reason == NoAttemptSent`, nothing was sent — `unresolvedProvesNothingWasSent` -/// is TRUE only for this case), admission was lost after the create but before its resolution GET, -/// or the conditional create was itself ambiguous (a transient exception) and the follow-up resolve -/// GET found nothing (the occupant that caused the conflict vanished before the GET, or the GET -/// itself failed). Both post-create cases report `unresolved_reason == AttemptsExhausted`, for which -/// `unresolvedProvesNothingWasSent` is FALSE — NEVER fabricated into -/// a false `Created`. CALLERS: do not log a bare `describeUnresolvedReason(AttemptsExhausted)` for -/// this case — it reads "the attempt budget was exhausted", which is misleading for a primitive -/// with no retry budget, and it silently folds admission loss together with "the resolve GET found -/// nothing" and "the occupant that caused the conflict was DELETED under a live epoch" (a -/// GC-invariant alarm, not routine contention) into the same generic wording. `SlotOccupyResult` -/// carries no discriminator between these sub-cases; the day a caller NEEDS the split is the trigger -/// for adding a dedicated `CasUnresolvedReason` value (a gated protocol decision, not a drive-by). -struct SlotOccupyResult -{ - enum class Kind : uint8_t { Created, Occupied, Unresolved }; - Kind kind = Kind::Unresolved; - /// Occupied only: the occupant, fetched by exact GET after the conditional create conflicted. - String occupant_bytes; - Token occupant_token; - /// Unresolved only: why the attempt outcome is unknowable right now. - CasUnresolvedReason unresolved_reason{}; -}; - -/// CAS-owned retry controller: the only place that decides whether a conditional-write attempt may be -/// reissued. It does not touch a writer cache or return ACK. Callers update their cache and acknowledge -/// the operation only after this controller has resolved the outcome and performed its final fence -/// check, using the returned `CasWriteOutcome`. -class CasRequestController -{ -public: - /// `now_ms_`: monotonic-ish clock, defaulting to `std::chrono::steady_clock`; tests inject a fake - /// one to drive deadline behavior deterministically (no sleeps). - /// `sleep_ms_`: the inter-attempt backoff sleep, defaulting to a real `std::this_thread::sleep_for`; - /// tests inject a recorder/no-op to assert the backoff schedule without wall-clock waits. The - /// controller only ever sleeps BETWEEN attempts of one logical operation, on the calling thread, - /// with no Pool mutex held (every call site — the ref append lane's leader, `stageManifest`, and - /// snapshot publishes — invokes the controller outside its locks; the append lane's - /// LEADERSHIP is deliberately held across the sleep: same-table appends must queue behind an - /// unresolved predecessor PUT anyway, preserving the writer's per-table ordering. - CasRequestController(BackendPtr backend_, CasRequestBudget budget_, std::function now_ms_ = {}, - std::function sleep_ms_ = {}); - - /// Controlled `putIfAbsent` with resolve-before-reissue. Performs at - /// most `budget.max_attempts` attempts of the exact SAME (key, bytes) — never a different key, never - /// a different body — bounded by `budget.operation_deadline_ms` measured from this call's own start, - /// with capped-exponential inter-attempt backoff (`retry_initial_backoff_ms`/`retry_max_backoff_ms`). - /// `fence_ok` is consulted before EVERY attempt (a false answer sends no further attempt), before - /// EVERY backoff sleep (a fence lost mid-loop aborts instantly, never after a pointless sleep), and - /// once more before a `Committed` return (a false answer there means the write may have landed but - /// this call reports `Unresolved`, never a false `Committed`). A sleep is - /// never entered when it (plus one more attempt) could not fit the operation deadline. An uncertain - /// attempt is resolved via `resolveByExactGet` before deciding whether to reissue. - /// Throws `CORRUPTED_DATA` if resolution ever observes DIFFERENT valid bytes at `key` — a real - /// conflict, never collapsed into `Unresolved`/`DefiniteFailure`. Returns `Unresolved` (never - /// throws) when the fence is lost or the budget is exhausted before a definite outcome is reached. - /// - /// THE VERDICT IS THE CALL'S, NOT THE LAST ATTEMPT'S. `DefiniteFailure` is returned only when EVERY - /// attempt this call sent was itself proven never applied. One attempt's whitelisted rejection - /// proves nothing about an EARLIER attempt of the same call that went ambiguous: that request may - /// have been received and may still materialize at `key` (an absent resolve GET is not evidence — - /// `unresolvedProvesNothingWasSent`). Any such attempt therefore dominates the result, which becomes - /// `Unresolved`/`DefiniteFailureAfterAmbiguity` — the wedge path — because a caller acting on - /// `DefiniteFailure` declares the key unwritten and reuses the id (`CasRefLedger::commitRefChunk`), - /// which an ambiguous predecessor can turn into an acked-then-lost transaction. Attempts a - /// pre-attempt gate refused never reach the backend, so they never make this call ambiguous. - /// `out_token` (optional): set ONLY on a `Committed` return, to the committed incarnation's token — - /// the attempt's own `PutResult` token, or the token the resolve GET observed when it proved an - /// earlier ambiguous attempt landed. Lets audit emitters (e.g. `PartWriteTxn::stageManifest`'s - /// `ManifestPut` event) keep the token without a follow-up HEAD. Untouched on any other return. - /// `out_reason` (optional): WHY an `Unresolved` was returned. Diagnostic only — the returned - /// outcome is unchanged, so no caller's decision depends on it. It exists because `Unresolved` - /// currently conflates two very different situations, and the resulting message - /// ("retry budget exhausted") is printed even where NOTHING was ever sent: finding #37 defect 3, - /// whose own note records that the opacity "plausibly fed 3 prior wrong analyses" — and it did so - /// again on 2026-07-24, when a sanitizer-slow unit test fenced itself and the text sent the CI - /// triage looking for a retry problem that did not exist. `NoAttemptSent` is the load-bearing - /// distinction: it means the key was provably never written, whereas the other reasons leave a - /// possibly-durable object behind. - CasWriteOutcome putIfAbsentControlled(std::string_view key, std::string_view bytes, - const std::function & fence_ok, Token * out_token = nullptr, - CasUnresolvedReason * out_reason = nullptr); - - /// One-shot exact-key resolution of an uncertain immutable create: - /// - identical bytes observed at `key` -> Committed (the earlier attempt DID commit) - /// - DIFFERENT bytes observed at `key` -> throws CORRUPTED_DATA (a real conflict, not a retry - /// signal — never silently treated as ambiguous) - /// - absent, or the GET itself fails -> Unresolved (another attempt may still be legal) - /// NEVER returns DefiniteFailure: an absent or unreadable key proves nothing about whether the - /// original request will eventually be provably non-applied, so resolution alone can never produce - /// that verdict. `out_token` (optional): set ONLY on `Committed`, to the observed incarnation's token. - CasWriteOutcome resolveByExactGet(std::string_view key, std::string_view expected_bytes, - Token * out_token = nullptr); - - /// Controlled If-Match overwrite with resolve-before-reissue, for a MUTABLE marker whose bytes - /// are deterministic so GET-based resolution can compare them (unlike a content-addressed - /// create's freshly-minted-per-attempt body). Performs at most `budget.max_attempts` attempts of - /// the exact SAME (key, bytes, expected token), bounded by `budget.operation_deadline_ms`, with - /// the same fence/backoff/deadline gates as `putIfAbsentControlled`. An ambiguous attempt - /// (`PreconditionFailed`, or a transient exception classified `Unresolved`) is resolved with ONE - /// GET at `key`: - /// - the current token still equals `expected` -> the attempt provably never applied; another - /// attempt of the SAME (key, bytes, expected) is legal (fence/backoff/deadline-gated) - /// - the current bytes equal `bytes` -> Committed (an earlier ambiguous attempt of - /// THIS call already landed); `token` is the observed incarnation - /// - neither -> Conflict: a genuine competing write - /// landed; returned as a value, never thrown - /// - the GET itself fails -> still ambiguous; reissue is safe - /// A whitelisted `DefiniteFailure` classification, or a deterministic local failure - /// (`isDeterministicLocalFailure`), rethrows the original exception rather than collapsing it - /// into an outcome. - CasOverwriteResult putOverwriteControlled(std::string_view key, std::string_view bytes, - const Token & expected, const std::function & fence_ok); - - /// Controlled overwrite with a caller-owned absolute deadline, cancellation/lifecycle cause, - /// interruptible retry wait, and contained diagnostic observer. Stop and deadline gates run - /// before every backend request, before and after every wait, and before accepting a proven - /// commit. The physical-attempt limit is considered only when another `PUT` would be sent, so the - /// exact resolving `GET` for the final sent attempt is never suppressed. - CasOverwriteResult putOverwriteControlled( - std::string_view key, - std::string_view bytes, - const Token & expected, - const CasOverwriteOperationContext & context); - - /// Controlled put-if-absent for a MUTABLE marker whose bytes are deterministic, where an - /// EXISTING DIFFERENT value at the key is a normal outcome (Conflict), not corruption. This is - /// the create-side sibling of `putOverwriteControlled` and deliberately does NOT reuse - /// `putIfAbsentControlled`: that method's resolve (`resolveByExactGet`) throws `CORRUPTED_DATA` - /// on any different bytes at the key, which is correct for the ref-log lane's immutable, - /// content-addressed keys (a different value there truly is impossible-by-construction) but - /// wrong for a mutable state marker (e.g. a blob's freshness-meta sidecar), where a - /// pre-existing DIFFERENT value is an expected, non-corrupt state a racing writer or GC pass - /// left behind. Performs at most `budget.max_attempts` attempts of the exact SAME (key, bytes), - /// bounded by `budget.operation_deadline_ms`, with the same fence/backoff/deadline gates as - /// `putIfAbsentControlled`. An ambiguous attempt (`PreconditionFailed`, or a transient exception - /// classified `Unresolved`) is resolved with ONE GET at `key`: - /// - absent -> the attempt provably never applied; another attempt of - /// the SAME (key, bytes) is legal (fence/backoff/deadline-gated) - /// - present, bytes equal `bytes` -> Committed (an earlier ambiguous attempt of THIS call, - /// or a racing writer creating the identical value, already landed); `token` is the observed - /// incarnation - /// - present, bytes differ -> Conflict: something else already occupies the key with - /// a different value; returned as a value, never thrown - /// - the GET itself fails -> still ambiguous; reissue is safe - /// Same DefiniteFailure/deterministic-local-failure rethrow convention as `putOverwriteControlled`. - CasOverwriteResult putIfAbsentControlledMutable(std::string_view key, std::string_view bytes, - const std::function & fence_ok); - - /// A DEDICATED RAW slot-occupy primitive [codex finding 3]: exactly ONE fence/deadline-gated - /// conditional create of `bytes` at `key`; on conflict, exactly ONE raw exact `GET` of the - /// occupant. NEVER retries internally, NEVER lists, and NEVER composes `putIfAbsentControlled` - /// (which retries the SAME (key, bytes) internally) or `resolveByExactGet` (which compares against - /// an expected body and throws `CORRUPTED_DATA` on a mismatch) — both contradict "one conditional - /// create" and `Occupied(bytes, token)` respectively. This is the primitive every seal writer and - /// wedge retry uses (spec INV-2): each CALL is one bounded attempt, and a caller that wants to keep - /// trying calls this again later, under its OWN fence/deadline/backoff discipline. - /// - /// `fence_ok` and the operation deadline are checked before the (only) create attempt — a refusal - /// there sends nothing and reports `Unresolved`/`NoAttemptSent`, exactly like every other - /// controlled op's first iteration. If the create conflicts or is ambiguous, `fence_ok` is checked - /// once more immediately before its resolution GET; refusal starts no GET and reports - /// `Unresolved`/`AttemptsExhausted`, because the create was already sent. There is deliberately no - /// post-I/O fence recheck after either request: verifying that a `Created`/`Occupied` result is still - /// relevant remains the caller's contract (Task 4/6's recheck under its own state lock). - /// - /// CONSEQUENCE, stated bluntly because it is the OPPOSITE of every sibling op's behavior: a - /// `Created` or `Occupied` returned here may come from a call whose fence was lost WHILE the PUT or - /// GET was in flight — this primitive does not know and does not check. Acting on either result - /// (adopting, acknowledging, installing) without the caller's OWN post-I/O - /// `checkFenceOrThrow(admitted_generation)` under its own lock is a correctness bug, not a missed - /// diagnostic — see Task 4's `resolveWedgeOnce` and Task 6's recovery CAS-walk in the plan for the - /// exact recheck shape. - /// - /// A whitelisted synchronous rejection (`classifyConditionalWriteResult`'s `DefiniteFailure`) or a - /// deterministic local failure (`isDeterministicLocalFailure`) RETHROWS the original exception - /// unchanged — the same convention as `putOverwriteControlled`/ - /// `putIfAbsentControlledMutable` (`SlotOccupyResult::Kind` has no `DefiniteFailure` member to carry - /// it). Any other exception, or a clean `PreconditionFailed`, is ambiguous and falls through to the - /// resolve GET identically — this primitive cannot and does not distinguish the two. - /// - /// Op-count contract (asserted by every `gtest_cas_slot_occupy.cpp` test): `Created` costs exactly - /// one backend op (the create); `Occupied` costs exactly two (the create, then the resolve GET); - /// `Unresolved` costs at most two (zero when a pre-attempt gate refuses, one when admission is lost - /// before resolution, otherwise the create plus a resolve GET that came up empty or failed). - SlotOccupyResult slotOccupy(std::string_view key, std::string_view bytes, - const std::function & fence_ok); - - /// Test-only: replace the inter-attempt backoff sleep (e.g. with a no-op) on an already-constructed - /// controller — for tests that reach the controller only through a fully-wired Pool/disk and cannot - /// pass the ctor parameter (see `Pool::setCasRetrySleepForTest`). Passing an empty function restores - /// the real sleep. Not thread-safe: call before driving any traffic through the controller. - void setSleepFnForTest(std::function sleep_ms_); - -private: - CasOverwriteResult putOverwriteControlledImpl( - std::string_view key, - std::string_view bytes, - const Token & expected, - const CasOverwriteOperationContext & context, - bool preserve_legacy_gates); - - /// The gate between a completed ambiguous attempt and its reissue: fence check FIRST (a fence lost - /// mid-loop must abort before any sleep), then the capped-exponential backoff sleep — skipped - /// entirely (returning false, no sleep served) when the sleep plus one more attempt could not fit - /// the operation deadline. Returns true when the loop may proceed to the next attempt; the loop - /// top's own pre-attempt fence/deadline checks re-run AFTER the sleep. - /// - /// `out_reason` (optional) receives WHICH of the two refusals returned false, so a caller reporting - /// an `Unresolved` from here does not have to guess between them. Both are mid-way by construction: - /// this gate is only reached once an attempt has been sent. - bool pauseBeforeReissue(uint32_t completed_attempt, uint64_t deadline_ms, const std::function & fence_ok, - CasUnresolvedReason * out_reason = nullptr); - /// The backoff scheduled before attempt `next_attempt` (attempt 2 sleeps `retry_initial_backoff_ms`, - /// doubling per reissue), saturating at `retry_max_backoff_ms`. 0 when backoff is disabled. - uint64_t backoffBeforeAttempt(uint32_t next_attempt) const; - - BackendPtr backend; - CasRequestBudget budget; - std::function now_ms; - std::function sleep_ms; -}; - -} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp new file mode 100644 index 000000000000..898c0c8c1504 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp @@ -0,0 +1,1220 @@ +#include + +#include +#include +#include +#include +#include +#include +#include + +#include "config.h" + +#if USE_AWS_S3 +#include +#endif + +#include + +#include + +#include +#include +#include +#include + +namespace ProfileEvents +{ + extern const Event CASRequestAttempt; + extern const Event CASRequestReissue; + extern const Event CASRequestConflictPause; + extern const Event CASRequestResolveRead; + extern const Event CASRequestGaveUp; + extern const Event CASRequestRefused; + extern const Event CASRequestFenceLostPostWrite; + extern const Event CASRequestConnectFailureHint; + extern const Event CASRequestFirstAttemptFuse; +} + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; + extern const int CAS_DELETE_MARKER; + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; + extern const int NETWORK_ERROR; + extern const int NOT_IMPLEMENTED; +} + +namespace DB::Cas +{ + +namespace detail +{ + +void recordAttempt() +{ + ProfileEvents::increment(ProfileEvents::CASRequestAttempt); +} + +void recordReissue() +{ + ProfileEvents::increment(ProfileEvents::CASRequestReissue); +} + +void recordConflictPause() +{ + ProfileEvents::increment(ProfileEvents::CASRequestConflictPause); +} + +void recordFirstAttemptFuse() +{ + ProfileEvents::increment(ProfileEvents::CASRequestFirstAttemptFuse); +} + +} + +namespace +{ +/// Shared by the two entry points below so the log line and the exception's message text can never +/// drift apart. Rate-limited (not per-distinct-`why` -- `LogSeriesLimiter` keys on the LOGGER NAME +/// only, so under a sustained outage where `why` keeps changing slightly, only the first message in +/// each window prints; this is the intended throttle, not a bug). Warning-level visibility is +/// intentional: this condition is expected to self-heal (the caller retries), but an operator watching +/// CAS logs directly should see it without having to know to look at system.replication_queue. +void logCasWriteRetryLater(const String & why) +{ + LogSeriesLimiter log(getLogger("CasWriteRetryLater"), /*allowed_count=*/1, /*interval_s=*/30); + LOG_WARNING(log, "CAS write could not be committed ({}); retrying later", why); +} +} + +[[noreturn]] void throwCasWriteRetryLater(const String & why) +{ + logCasWriteRetryLater(why); + throw Exception(ErrorCodes::NETWORK_ERROR, "CAS write could not be committed ({}); retrying later", why); +} + +std::exception_ptr makeCasWriteRetryLaterExceptionPtr(const String & why) +{ + logCasWriteRetryLater(why); + return std::make_exception_ptr( + Exception(ErrorCodes::NETWORK_ERROR, "CAS write could not be committed ({}); retrying later", why)); +} + +[[noreturn]] void throwCasTransientUnavailable(const String & subject, const String & condition) +{ + /// The code is coarse (it shares a `system.errors` row with socket failures), so the MESSAGE must + /// carry the whole truth: which CA condition refused, and that the refusal is a state rather than + /// damage. Consumers key on the code; operators read this line. + /// + /// The shared suffix carries ONLY the classification, because that is the one claim true at every + /// site: retry-later is right even where the condition may turn out terminal, since the next attempt + /// re-decides against fresh state. Any promise about HOW the condition clears belongs in `condition`, + /// where the site that can actually prove it makes it -- `checkFenceOrThrow` provably cannot. + throw Exception(ErrorCodes::NETWORK_ERROR, + "{} -- {}; TRANSIENT unavailability, not damage", subject, condition); +} + +namespace +{ + +/// `CLOCK_BOOTTIME` milliseconds -- the clock a mount lease deadline is expressed on, reproduced here +/// rather than shared because `Backend/` does not depend on the mount plane. +uint64_t bootClockMs() +{ + return clock_gettime_ns(CLOCK_BOOTTIME) / 1000000; +} + +uint64_t saturatingAdd(uint64_t lhs, uint64_t rhs) +{ + return lhs > std::numeric_limits::max() - rhs ? std::numeric_limits::max() : lhs + rhs; +} + +} + +/// The failure class a fresh credential could fix, named here rather than taken from +/// `S3Exception::isAccessTokenExpiredError`, which also fires on `S3Errors::UNKNOWN` -- the SDK's code +/// for EVERY error it does not model. Borrowing it would put throttling codes an S3-compatible store +/// reports under a non-AWS name into the credential class, and would carve a real access denial out of +/// `isDefinitelyRefusedWrite` so the write wedges its caller instead of being refused. `UNKNOWN` is +/// therefore never matched by code alone; a store that spells the error out by name still matches. +/// +/// Declared in the header: a caller whose OWN loop makes the next physical attempt has to tell this +/// class apart from the rest of `isDefinitelyRefusedWrite`. +bool isRefreshableCredentialError([[maybe_unused]] const std::exception & e) +{ +#if USE_AWS_S3 + const auto * s3 = dynamic_cast(&e); + if (!s3) + return false; + const Aws::S3::S3Errors code = s3->getS3ErrorCode(); + if (code == Aws::S3::S3Errors::INVALID_ACCESS_KEY_ID || code == Aws::S3::S3Errors::ACCESS_DENIED + || code == Aws::S3::S3Errors::INVALID_SIGNATURE || code == Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID) + return true; + const String & name = s3->getExceptionName(); + return name == "ExpiredToken" || name == "InvalidToken" || name == "InvalidAccessKeyId" + || name == "SignatureDoesNotMatch" || name == "AccessDenied" || name == "AccountProblem"; +#else + return false; +#endif +} + +namespace +{ + +/// The store's authoritative answer that there is nothing at the key -- a fact about the OBJECT, which +/// reissuing only replays. Deliberately narrower than "not retryable": a missing bucket or an +/// unmodeled name is an answer about reaching the store, which an S3-compatible store gives +/// transiently when it misroutes a request, so it stays in the ambiguous class and is reissued until +/// the deadline. The credential codes never reach here -- `refreshAndClassifyReadFault` consumes them +/// first. +bool isAuthoritativeAbsence([[maybe_unused]] const std::exception & e) +{ +#if USE_AWS_S3 + if (const auto * s3 = dynamic_cast(&e)) + { + const Aws::S3::S3Errors code = s3->getS3ErrorCode(); + return code == Aws::S3::S3Errors::NO_SUCH_KEY || code == Aws::S3::S3Errors::NO_SUCH_UPLOAD; + } +#endif + return false; +} + +GaveUp::Source sourceFor(const Retry::Bound & bound) +{ + return bound.lease_bound ? GaveUp::Source::Lease : GaveUp::Source::Policy; +} + +/// Could the precondition this write was built with still be met by what the resolve read saw? A +/// create needs the key absent; a replace needs the incarnation it named to still be current. +bool preconditionStillSatisfiable(const Observation & seen, const std::optional & expected) +{ + return std::visit(detail::Overload{ + /// The read itself failed, so it proved nothing either way and an ambiguous attempt may still + /// be alive. Reporting a conflict on it would name an occupant nobody observed. + [](const NotObserved &) { return true; }, + [&](const ProvenAbsent &) { return !expected.has_value(); }, + [&](const Meta & m) { return expected.has_value() && m.etag == *expected; }, + [&](const Object & o) { return expected.has_value() && o.etag == *expected; }}, + seen); +} + +/// Drop the body from an observation the write engine had to fetch to prove whose bytes were at the +/// key. The presence-only loop is defined by what it reports, so the demotion happens on its results +/// rather than being trusted to every branch that builds one. +Observation withoutBody(Observation seen) +{ + if (const auto * obj = std::get_if(&seen)) + return Meta{obj->bytes.size(), obj->etag}; + return seen; +} + +WriteResult withoutBody(WriteResult result) +{ + if (auto * conflict = std::get_if(&result)) + conflict->seen = withoutBody(std::move(conflict->seen)); + else if (auto * declined = std::get_if(&result)) + declined->seen = withoutBody(std::move(declined->seen)); + else if (auto * gave_up = std::get_if(&result)) + gave_up->last_seen = withoutBody(std::move(gave_up->last_seen)); + return result; +} + +} + +bool isDeterministicLocalFailure(int code) +{ + return code == ErrorCodes::LOGICAL_ERROR || code == ErrorCodes::NOT_IMPLEMENTED + || code == ErrorCodes::BAD_ARGUMENTS || code == ErrorCodes::CORRUPTED_DATA; +} + +bool isDefinitelyRefusedWrite([[maybe_unused]] const std::exception & e) +{ +#if USE_AWS_S3 + if (const auto * s3 = dynamic_cast(&e)) + /// The refresh class is included rather than left to overlap: the two name lists agree, but + /// `InvalidSignature` carries a code `isAccessDeniedError` does not match, and an error that is + /// refreshable without being refusable would spend the whole deadline whenever no refresh is + /// available -- which is every CAS disk today. + return S3::isMalformedRequestError(*s3) || S3::isEntityTooLargeError(*s3) + || S3::isAccessDeniedError(*s3) || isRefreshableCredentialError(e); +#endif + return false; +} + +bool isConnectFailureHint([[maybe_unused]] const std::exception & e) +{ +#if USE_AWS_S3 + const auto * s3 = dynamic_cast(&e); + if (!s3 || s3->getS3ErrorCode() != Aws::S3::S3Errors::NETWORK_CONNECTION) + return false; + /// This repository's Poco (`SocketImpl::error`, `SocketImpl::connect`) is the source of every text. + static constexpr std::array texts{ + "Cannot assign requested address", "Connection refused", "No route to host", + "Network is unreachable", "connect timed out"}; + const std::string_view message = s3->message(); + for (std::string_view text : texts) + if (message.find(text) != std::string_view::npos) + return true; +#endif + return false; +} + +bool isFirstAttemptFuseTimeout([[maybe_unused]] const std::exception & e, [[maybe_unused]] size_t attempt_no) +{ +#if USE_AWS_S3 + /// The connect-failure hint is checked FIRST: `connect timed out` belongs to that classifier, and + /// an attempt whose text matches both stays a hint, reissued without a preceding settle read. + if (attempt_no != 1 || isConnectFailureHint(e)) + return false; + const auto * s3 = dynamic_cast(&e); + if (!s3 || s3->getS3ErrorCode() != Aws::S3::S3Errors::NETWORK_CONNECTION) + return false; + const std::string_view message = s3->message(); + return message.find("Timeout") != std::string_view::npos; +#else + return false; +#endif +} + +CasRequests::CasRequests(BackendPtr backend_, Fence fence_, + std::function now_ms_, std::function sleep_ms_, + CasHotKeys * hot_keys_) + : backend(std::move(backend_)) + , fence(std::move(fence_)) + , now_ms(now_ms_ ? std::move(now_ms_) : std::function(bootClockMs)) + , sleep_ms(sleep_ms_ ? std::move(sleep_ms_) : std::function(sleepForMilliseconds)) + , attempt_reservation_ms(backend->attemptEnvelopeMs()) + , own_hot_keys(hot_keys_ ? nullptr : std::make_unique(0)) + , hot_keys(hot_keys_ ? hot_keys_ : own_hot_keys.get()) +{ +} + +void CasRequests::setNowFnForTest(std::function now_ms_) +{ + now_ms = now_ms_ ? std::move(now_ms_) : std::function(bootClockMs); +} + +void CasRequests::setSleepFnForTest(std::function sleep_ms_) +{ + sleep_ms = sleep_ms_ ? std::move(sleep_ms_) : std::function(sleepForMilliseconds); +} + +CasOperation CasRequests::admit(Liveness liveness) +{ + return CasOperation(*this, fence.generation(), std::move(liveness)); +} + +CasOperation CasRequests::resume(uint64_t admitted_generation, Liveness liveness) +{ + return CasOperation(*this, admitted_generation, std::move(liveness)); +} + +std::optional CasRequests::tryMint(const String & key, String value) const +{ + if (!isIncarnationValue(backend->dialect(), value)) + return std::nullopt; + return Etag(backend->backendId(), key, backend->dialect(), std::move(value)); +} + +Etag CasRequests::mint(const String & key, String value) const +{ + if (auto minted = tryMint(key, value)) + return std::move(*minted); + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS: the store answered for '{}' with a value '{}' that is not a valid incarnation", key, value); +} + +const String & CasRequests::valueFor(const String & key, const Etag & inc) const +{ + if (inc.key() != key || inc.backendId() != backend->backendId()) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS: an incarnation of '{}' observed on backend {} cannot be the precondition for '{}' on backend {}", + inc.key(), inc.backendId(), key, backend->backendId()); + return inc.value(); +} + +namespace +{ + +/// The one mapping from a refused admission to the exception a read-class caller sees. The request +/// gate and the streamed body below both refuse through it, so a body refused mid-transfer reads +/// exactly like an open refused before it: `NoBudget` is the retry-later class, everything else the +/// tripped-fence class. Free, not a member: the body's copy must not reference an operation. +[[noreturn]] void throwReadRefused(Fence::Admit admit, std::string_view verb, const String & subject, + std::string_view when) +{ + if (admit == Fence::Admit::NoBudget) + throwCasWriteRetryLater(fmt::format("{} of '{}': no lease budget {}", verb, subject, when)); + throwCasTransientUnavailable(fmt::format("CAS {} of '{}'", verb, subject), + fmt::format("mount fence tripped {}", when)); +} + +/// The body of a streamed object, re-admitted at every refill. `Backend::stream` bounds only the +/// open; the SDK reads the body at the consumer's pace, under the storage's ordinary settings, long +/// after the attempt that opened it returned -- so a fold parked in a multi-gigabyte run body would +/// outlive the fence that refused every other request of its operation. The check is the operation's +/// own admission, asked once per SDK buffer, and its refusal is the exception a refused open produces. +/// +/// The predicate is held BY VALUE -- the fence's `admit` closure, the admitted generation, the +/// caller's liveness -- never through the operation: `CasOperation::stream` returns a buffer that can +/// outlive the operation object (S3 staging opens its stream under a local mount-plane operation and +/// hands the buffer to the backend). Modelled on `LimitReadBuffer`: no byte is copied, the SDK's +/// window is exposed as this buffer's own. +class AdmittedBodyReadBuffer : public ReadBuffer +{ +public: + AdmittedBodyReadBuffer(std::unique_ptr in_, String key_, + std::function admit_, + uint64_t admitted_generation_, Liveness liveness_) + /// The open already loaded a window (`Backend::stream` forces the first GET): adopt it, so + /// the first refill this buffer asks for is the SECOND window and the SDK buffer is never + /// asked to advance over pending data. + : ReadBuffer(in_->position(), in_->available(), 0) + , in(std::move(in_)) + , key(std::move(key_)) + , admit(std::move(admit_)) + , admitted_generation(admitted_generation_) + , liveness(std::move(liveness_)) + { + } + +private: + bool nextImpl() override + { + /// Let the SDK buffer account the bytes the consumer took from the shared window. + in->position() = position(); + + Fence::Admit verdict = admit(admitted_generation, 0); + if (verdict == Fence::Admit::Ok && liveness && !liveness()) + verdict = Fence::Admit::LostOrRearmed; + if (verdict != Fence::Admit::Ok) + throwReadRefused(verdict, "stream body", key, "mid-body"); + + if (!in->next()) + { + BufferBase::set(in->position(), 0, 0); + return false; + } + BufferBase::set(in->position(), in->available(), 0); + return true; + } + + std::unique_ptr in; + String key; + std::function admit; + uint64_t admitted_generation; + Liveness liveness; +}; + +} + +CasOperation::Gate CasOperation::gate(uint64_t needed_ms) const +{ + switch (owner.fence.admit(admitted_generation, needed_ms)) + { + case Fence::Admit::LostOrRearmed: return Gate::FenceLost; + case Fence::Admit::NoBudget: return Gate::NoBudget; + case Fence::Admit::Ok: break; + } + /// The caller's own facts are the second half of admission, and the engine does not need to know + /// which of the two refused: a stopping task and a lost lease end the operation the same way. + if (liveness && !liveness()) + return Gate::FenceLost; + return Gate::Ok; +} + +uint64_t CasOperation::reservedFor(uint64_t sleep_ms, uint32_t envelopes) const +{ + uint64_t total = sleep_ms; + for (uint32_t i = 0; i < envelopes; ++i) + total = saturatingAdd(total, owner.attempt_reservation_ms); + return total; +} + +Retry CasOperation::freeze(const Retry & policy) const +{ + if (policy.policy_deadline_ms) + return policy; + Retry frozen = policy; + frozen.policy_deadline_ms = saturatingAdd(owner.now_ms(), policy.window_ms); + return frozen; +} + +bool CasOperation::fits(uint64_t needed_ms, const Retry::Bound & bound) const +{ + /// Strict at the boundary. A backend with no attempt timeout reserves 0 and full jitter can draw a + /// 0 sleep, so a `needed <= remaining` test would keep issuing requests at and past the deadline -- + /// the one thing "no request after the boundary" promises never happens. + const uint64_t now = owner.now_ms(); + if (now >= bound.deadline_ms) + return false; + return needed_ms <= bound.deadline_ms - now; +} + +bool CasOperation::refreshAndClassifyReadFault(const std::exception & e, bool & refresh_attempted, bool & refreshed) +{ + if (const auto * db_e = dynamic_cast(&e); db_e && isDeterministicLocalFailure(db_e->code())) + return true; + /// A local failure -- a bad allocation, a logic error raised inside the attempt -- is not a + /// transport fault, and reissuing it would spend the whole deadline replaying the same bug. Every + /// exception the transport raises is a `Poco::Exception`, `DB::Exception` included. + if (!dynamic_cast(&e)) + return true; + /// A credential failure gets ONE refresh per call -- the storage hands back a fresh client every + /// time it is asked, so refreshing per attempt would reissue a permanent denial to the deadline. + /// Without new credentials nothing would sign differently, so a read propagates rather than + /// spending its policy on a request that cannot start succeeding. + if (isRefreshableCredentialError(e)) + { + if (refresh_attempted) + return true; + refresh_attempted = true; + refreshed = owner.backend->refreshCredentials(); + return !refreshed; + } + /// The store's own answer decides. A refusal it proved never applied, and an authoritative absence, + /// both replay identically; everything else -- a throttle, a 5xx, a missing bucket, an unmodeled + /// name an S3-compatible store reports -- may still be transient. + return isDefinitelyRefusedWrite(e) || isAuthoritativeAbsence(e); +} + +void CasOperation::giveUpReadFenceLost(std::string_view verb, const String & subject, std::string_view when) +{ + last_read_stop = ReadStop::FenceLost; + throwReadRefused(Fence::Admit::LostOrRearmed, verb, subject, when); +} + +void CasOperation::giveUpReadNoBudget(std::string_view verb, const String & subject, std::string_view what) +{ + last_read_stop = ReadStop::NoBudgetLease; + throwReadRefused(Fence::Admit::NoBudget, verb, subject, what); +} + +void CasOperation::giveUpReadDeadline(std::string_view verb, const String & subject, + const Retry::Bound & bound, uint32_t attempts_made) +{ + last_read_stop = ReadStop::PolicyExhausted; + throwCasWriteRetryLater(fmt::format("{} of '{}': gave up at the {} deadline after {} attempt(s)", + verb, subject, bound.lease_bound ? "lease" : "policy", attempts_made)); +} + +std::optional CasOperation::readUnder(const String & key, const Retry & policy, const Retry::Bound & bound) +{ + return readLoop("read", key, policy, bound, [&](auto & access) -> std::optional + { + auto raw = owner.backend->read(key, access); + if (!raw) + return std::nullopt; + return Object{std::move(raw->bytes), owner.mint(key, std::move(raw->value))}; + }); +} + +std::optional CasOperation::headUnder(const String & key, const Retry & policy, const Retry::Bound & bound) +{ + return readLoop("head", key, policy, bound, [&](auto & access) -> std::optional + { + auto raw = owner.backend->head(key, access); + if (!raw) + return std::nullopt; + return Meta{raw->size, owner.mint(key, std::move(raw->value))}; + }); +} + +ListPage CasOperation::listUnder(const String & prefix, const String & cursor, size_t limit, + const Retry & policy, const Retry::Bound & bound) +{ + return readLoop("list", prefix, policy, bound, [&](auto & access) + { + Backend::RawListPage raw = owner.backend->list(prefix, cursor, limit, access); + ListPage page; + page.next_cursor = std::move(raw.next_cursor); + page.keys.reserve(raw.keys.size()); + for (auto & listed : raw.keys) + { + ListedKey entry{std::move(listed.key), listed.size, std::nullopt}; + if (listed.value) + entry.etag = owner.mint(entry.key, std::move(*listed.value)); + page.keys.push_back(std::move(entry)); + } + return page; + }); +} + +Removal CasOperation::removeUnder(const String & key, const String & expected_value, + const Retry & policy, const Retry::Bound & bound) +{ + const Backend::RawRemoval raw = readLoop("remove", key, policy, bound, [&](auto & access) + { + return owner.backend->remove(key, expected_value, access); + }); + switch (raw) + { + case Backend::RawRemoval::Removed: return Removal::Removed; + case Backend::RawRemoval::Gone: return Removal::Gone; + case Backend::RawRemoval::Mismatch: return Removal::Mismatch; + case Backend::RawRemoval::DeleteMarker: break; + } + /// Thrown outside the attempt loop: a versioned bucket answers this way every time, so reissuing + /// would spend the whole deadline to be told the same thing. + throw Exception(ErrorCodes::CAS_DELETE_MARKER, + "CAS remove of '{}' archived a noncurrent version instead of reclaiming the object: " + "the bucket has object versioning enabled", key); +} + +std::optional CasOperation::read(const String & key, const Retry & policy) +{ + return readUnder(key, policy, policy.bind(owner.now_ms())); +} + +std::optional CasOperation::head(const String & key, const Retry & policy) +{ + return headUnder(key, policy, policy.bind(owner.now_ms())); +} + +ListPage CasOperation::list(const String & prefix, const String & cursor, size_t limit, const Retry & policy) +{ + return listUnder(prefix, cursor, limit, policy, policy.bind(owner.now_ms())); +} + +void CasOperation::forEachListedKey(const String & prefix, const ListedKeyFn & fn, const Retry & per_page, + size_t page_limit, const std::function & on_page_fetched) +{ + String cursor; + for (;;) + { + ListPage page = list(prefix, cursor, page_limit, per_page); + if (on_page_fetched) + on_page_fetched(); + for (const ListedKey & entry : page.keys) + if (!fn(entry)) + return; + if (page.next_cursor.empty()) + return; + cursor = std::move(page.next_cursor); + } +} + +void CasOperation::removeManyWriteOnce(const std::vector & keys, const Retry & policy) +{ + if (keys.size() > kBulkDeleteMaxKeys) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS removeManyWriteOnce: {} keys in one chunk, the limit is {}; the consumer chunks its input", + keys.size(), kBulkDeleteMaxKeys); + if (keys.empty()) + return; + const Retry::Bound bound = policy.bind(owner.now_ms()); + const String subject = fmt::format("{} (+{} keys)", keys.front().str(), keys.size() - 1); + readLoop("removeManyWriteOnce", subject, policy, bound, [&](auto & access) + { + owner.backend->removeManyWriteOnce(keys, access); + return true; + }); +} + +Removal CasOperation::remove(const String & key, const Etag & seen, const Retry & policy) +{ + return removeUnder(key, owner.valueFor(key, seen), policy, policy.bind(owner.now_ms())); +} + +Removal CasOperation::removeCurrent(const String & key, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + for (uint32_t attempt = 1;; ++attempt) + { + const std::optional seen = headUnder(key, policy, bound); + if (!seen) + return Removal::Gone; + const Removal removed = removeUnder(key, owner.valueFor(key, seen->etag), policy, bound); + if (removed != Removal::Mismatch) + return removed; + + /// A `Mismatch` is a VALUE, so it never reaches the fault check inside the two loops above: this + /// verb has to honour `once` itself, and its contract is that it never hands a `Mismatch` back. + /// The message says what actually stopped it, which is the policy and not the deadline. + if (policy.single_attempt) + throwCasWriteRetryLater(fmt::format( + "removeCurrent of '{}': the observed incarnation was replaced and the policy allows no reissue", key)); + + /// Another incarnation became current between the observation and the delete. Re-observe, paced + /// like every other reissue: the reservation covers the next `head` and the `remove` after it. + const uint64_t pause_ms = Retry::backoff(attempt); + const uint64_t needed = reservedFor(pause_ms, 2); + switch (gate(needed)) + { + case Gate::FenceLost: giveUpReadFenceLost("removeCurrent", key, "before the reissue"); + case Gate::NoBudget: giveUpReadNoBudget("removeCurrent", key, "for the reissue"); + case Gate::Ok: break; + } + if (!fits(needed, bound)) + giveUpReadDeadline("removeCurrent", key, bound, attempt); + detail::recordReissue(); + owner.sleep_ms(pause_ms); + } +} + +SentinelProbeResult CasOperation::probeSentinel(const String & key, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + bool refresh_attempted = false; + for (uint32_t attempt = 1;; ++attempt) + { + const uint64_t reservation = reservedFor(0, 1); + switch (gate(reservation)) + { + case Gate::FenceLost: giveUpReadFenceLost("probeSentinel", key, "before the request"); + case Gate::NoBudget: giveUpReadNoBudget("probeSentinel", key, "for one more request"); + case Gate::Ok: break; + } + /// Before the first attempt nothing has been probed, so the caller is told the request never + /// happened rather than being handed an outcome no request produced. + if (!fits(reservation, bound)) + giveUpReadDeadline("probeSentinel", key, bound, attempt - 1); + + detail::recordAttempt(); + SentinelProbeResult result{ProbeOutcome::Indeterminate, std::nullopt}; + try + { + result = owner.withTransportAccess(attempt, [&](auto & access) + { + return owner.backend->probeSentinelRaw(key, access); + }); + } + catch (const std::exception & e) + { + /// The probe never counts a fuse, so whether this fault was a credential reissue is not + /// interesting here -- only the write and read loops guard a counter with it. + bool refreshed = false; + if (refreshAndClassifyReadFault(e, refresh_attempted, refreshed)) + throw; + /// The probe reports every transport failure as `Indeterminate` rather than by throwing, so + /// a decorator that does throw is folded onto the same inconclusive outcome. + } + + /// The reissue is driven by the OUTCOME: this is the one primitive whose failures never reach a + /// `catch`, and `Indeterminate` means "inconclusive", which is exactly what a retry resolves. + /// The other four outcomes are authoritative answers and return on the first attempt. + if (result.outcome != ProbeOutcome::Indeterminate || policy.single_attempt) + return result; + + const uint64_t pause_ms = Retry::backoff(attempt); + const uint64_t needed = reservedFor(pause_ms, 1); + /// A lost fence and a spent lease budget are not things the store said, so they are reported the + /// way every other read verb reports them rather than being dressed up as an outcome. + switch (gate(needed)) + { + case Gate::FenceLost: giveUpReadFenceLost("probeSentinel", key, "before the reissue"); + case Gate::NoBudget: giveUpReadNoBudget("probeSentinel", key, "for the reissue"); + case Gate::Ok: break; + } + /// Only the policy's own bound ends the loop with a value: a probe DID run, and what it saw is + /// more than a give-up exception could say. Every consumer treats `Indeterminate` fail-closed. + if (!fits(needed, bound)) + return result; + detail::recordReissue(); + owner.sleep_ms(pause_ms); + } +} + +std::unique_ptr CasOperation::stream(const String & key, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + std::unique_ptr body = readLoop("stream", key, policy, bound, [&](auto & access) + { + return owner.backend->stream(key, access); + }); + if (!body) + return nullptr; /// absent: the open already answered + /// By value, deliberately: nothing the buffer holds may reference this operation or its owner. + return std::make_unique(std::move(body), key, owner.fence.admit, + admitted_generation, liveness); +} + +void CasOperation::publish(const BlobPublishRequest & request, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + readLoop("publish", request.destination_key, policy, bound, [&](auto & access) + { + owner.backend->publish(request, access); + }); +} + +CasOperation::Resolved CasOperation::observe(const String & key, const Retry & policy, const Retry::Bound & bound) +{ + last_read_stop.reset(); + try + { + auto got = readUnder(key, policy, bound); + if (!got) + return {ProvenAbsent{}, std::nullopt}; + return {std::move(*got), std::nullopt}; + } + catch (const Exception & e) + { + /// A local bug replays identically on every reissue, so it is never swallowed into "nothing + /// observed". Anything else settled nothing -- and `last_read_stop` says whether a bound + /// refused the read or the transport itself failed, which the exception cannot. + if (isDeterministicLocalFailure(e.code())) + throw; + return {NotObserved{}, last_read_stop}; + } + catch (const std::exception & e) + { + /// A failure that is not the transport's is not an observation: it is the same local bug the + /// read loop refuses to reissue, and swallowing it here would hide it just as thoroughly. + if (!dynamic_cast(&e)) + throw; + return {NotObserved{}, last_read_stop}; + } +} + +CasOperation::Resolved CasOperation::observePresence(const String & key, const Retry & policy, const Retry::Bound & bound) +{ + last_read_stop.reset(); + try + { + auto got = headUnder(key, policy, bound); + if (!got) + return {ProvenAbsent{}, std::nullopt}; + return {std::move(*got), std::nullopt}; + } + catch (const Exception & e) + { + if (isDeterministicLocalFailure(e.code())) + throw; + return {NotObserved{}, last_read_stop}; + } + catch (const std::exception & e) + { + if (!dynamic_cast(&e)) + throw; + return {NotObserved{}, last_read_stop}; + } +} + +WriteResult CasOperation::gaveUp(GaveUp::Why why, GaveUp::Source source, WriteState & state) const +{ + ProfileEvents::increment(ProfileEvents::CASRequestGaveUp); + return GaveUp{why, source, state.sent_any, state.last_seen, state.attempts_sent}; +} + +WriteResult CasOperation::gaveUpForReadStop(ReadStop stop, WriteState & state, const Retry::Bound & bound) const +{ + switch (stop) + { + case ReadStop::FenceLost: + return gaveUp(GaveUp::Why::FenceLost, sourceFor(bound), state); + case ReadStop::NoBudgetLease: + /// The fence's budget IS the mount lease, so the lease is the bound that ended this call. + return gaveUp(GaveUp::Why::Deadline, GaveUp::Source::Lease, state); + case ReadStop::PolicyExhausted: + return gaveUp(GaveUp::Why::Deadline, sourceFor(bound), state); + } + UNREACHABLE(); +} + +WriteResult CasOperation::gaveUpAfterFailedObservation(std::optional stop, WriteState & state, + const Retry::Bound & bound) const +{ + if (stop) + return gaveUpForReadStop(*stop, state, bound); + /// No bound refused; the read itself failed. That is what `Unresolved` names, and it is the honest + /// answer -- claiming a deadline the clock never reached would misreport which bound to widen. + return gaveUp(GaveUp::Why::Unresolved, sourceFor(bound), state); +} + +WriteResult CasOperation::postCommit(Etag inc, bool resolved_by_read, WriteState & state, const Retry::Bound & bound) +{ + /// Admission once more, now that the write is proven durable: a fence lost here means the object + /// may well exist, but this call must never claim it -- the caller has to resolve the key instead. + switch (gate(0)) + { + case Gate::FenceLost: + ProfileEvents::increment(ProfileEvents::CASRequestFenceLostPostWrite); + return gaveUp(GaveUp::Why::FenceLost, sourceFor(bound), state); + case Gate::NoBudget: + /// The fence's budget IS the mount lease, so its refusal names the lease as the bound that + /// ended this call, whichever bound produced the policy's own deadline. + return gaveUp(GaveUp::Why::Deadline, GaveUp::Source::Lease, state); + case Gate::Ok: break; + } + return Committed{std::move(inc), state.attempts_sent, resolved_by_read}; +} + +std::optional CasOperation::gatedPause(uint64_t pause_ms, uint32_t envelopes, WriteState & state, + const Retry::Bound & bound, void (*record)(), bool should_sleep) +{ + const uint64_t needed = reservedFor(pause_ms, envelopes); + switch (gate(needed)) + { + case Gate::FenceLost: return gaveUp(GaveUp::Why::FenceLost, sourceFor(bound), state); + case Gate::NoBudget: return gaveUp(GaveUp::Why::Deadline, GaveUp::Source::Lease, state); + case Gate::Ok: break; + } + if (!fits(needed, bound)) + return gaveUp(GaveUp::Why::Deadline, sourceFor(bound), state); + record(); + /// `should_sleep` is false ONLY for the fuse's zero-pause reissue: that pause is not "zero + /// milliseconds", it is NO PAUSE AT ALL, so it must not call `sleep_ms` even with a zero argument. + /// The three ordinary callers always sleep, even when a drawn jitter happens to be exactly zero -- + /// `Retry::backoff`'s full jitter includes zero -- so their call to `sleep_ms(0)` is kept + /// unconditional here to leave their observable behaviour exactly as it was before this collapse. + if (should_sleep) + owner.sleep_ms(pause_ms); + return std::nullopt; +} + +std::optional CasOperation::pauseAndReissue(WriteState & state, const Retry::Bound & bound) +{ + return gatedPause(Retry::backoff(++state.reissues), 2, state, bound, detail::recordReissue, /*should_sleep=*/true); +} + +std::optional CasOperation::pauseForConflict(WriteState & state, const Retry::Bound & bound) +{ + return gatedPause(Retry::conflictBackoff(), 2, state, bound, detail::recordConflictPause, /*should_sleep=*/true); +} + +/// A flat pause before reissuing an attempt whose failure text named a failed connection. +static constexpr uint64_t kConnectHintPauseMs = 50; + +std::optional CasOperation::pauseFlat(WriteState & state, const Retry::Bound & bound) +{ + return gatedPause(kConnectHintPauseMs, 2, state, bound, detail::recordReissue, /*should_sleep=*/true); +} + +std::optional CasOperation::reissueAtOnce(WriteState & state, const Retry::Bound & bound) +{ + return gatedPause(0, 2, state, bound, detail::recordReissue, /*should_sleep=*/false); +} + +WriteResult CasOperation::writeLoop(const String & key, const String & bytes, const std::optional & expected, + const Retry & policy, const Retry::Bound & bound, WriteState & state, + ResolveWith resolve_refusal_with) +{ + std::optional expected_value; + if (expected) + expected_value = owner.valueFor(key, *expected); + + /// This inner write's own bytes are what an ambiguity of it could have landed. A previous inner + /// write of the same call sent DIFFERENT bytes and ended in a conflict, which proved its attempts + /// dead -- carrying its ambiguity forward is how a competitor's identical object gets claimed. + state.any_ambiguous = false; + + for (;;) + { + /// A write reserves TWO envelopes: the attempt, and the exact read that settles it. That is + /// what keeps "every conflict is settled by one read" true at the deadline edge. + const uint64_t reservation = reservedFor(0, 2); + switch (gate(reservation)) + { + case Gate::FenceLost: return gaveUp(GaveUp::Why::FenceLost, sourceFor(bound), state); + case Gate::NoBudget: return gaveUp(GaveUp::Why::Deadline, GaveUp::Source::Lease, state); + case Gate::Ok: break; + } + if (!fits(reservation, bound)) + return gaveUp(GaveUp::Why::Deadline, sourceFor(bound), state); + + detail::recordAttempt(); + ++state.attempts_sent; + state.sent_any = true; + + /// Disengaged means the attempt threw: its fate is unproven, and nothing may be read out of it. + std::optional> outcome; + /// A credential answer is given BEFORE the store applies anything, so the attempt provably did + /// not land. It is the one failure that is neither a commit nor an ambiguity, and keeping it out + /// of `any_ambiguous` is what lets a second credential failure of this inner write be refused + /// instead of resolved by a read and reissued to the deadline. + bool credential_answer = false; + bool refreshed = false; + /// Set once `refreshed` is known: true only when the credential-owned reissue below (the one + /// guarded by this exact expression) is what will actually resend this attempt. An earlier + /// ambiguity of this inner write routes the reissue through the ordinary hint/fuse/backoff + /// mechanisms instead -- the credential answer never gets to skip their read or their pacing -- + /// so a hint or fuse counter must still count in that case even though credentials were refreshed. + bool refresh_owns_reissue = false; + bool connect_hint = false; + bool fuse = false; + try + { + outcome = owner.withTransportAccess(state.attempts_sent, [&](auto & access) + { + return owner.backend->write(key, bytes, expected_value, access); + }); + } + catch (const Exception & e) + { + if (isDeterministicLocalFailure(e.code())) + throw; + /// ONE refresh per call, only for the class a credential could explain, and only when a + /// reissue could sign with what it installs. So an oversized entity never triggers a + /// re-acquisition, a denial fresh credentials do not fix is refused on the second look + /// rather than reissued to the deadline (the storage hands back a new client every time it + /// is asked), and under a single-attempt policy the refusal below is literally "no refresh + /// installed credentials and no earlier ambiguity". + credential_answer = isRefreshableCredentialError(e); + if (credential_answer && !state.refresh_attempted && !policy.single_attempt) + { + state.refresh_attempted = true; + refreshed = owner.backend->refreshCredentials(); + } + /// Mirrors the condition guarding the credential-owned reissue below exactly, so the two + /// can never drift: `state.any_ambiguous` here is still this attempt's INCOMING value, + /// because the update below only fires when `!credential_answer`, which `refreshed` implies + /// false for. `!policy.single_attempt` is redundant today -- `refreshed` can only be set + /// above under `!policy.single_attempt` already -- but it is kept so this stays an exact + /// copy of the reissue branch's condition rather than a hand-simplified one that could + /// silently stop matching it. + refresh_owns_reissue = refreshed && !policy.single_attempt && !state.any_ambiguous; + /// A refusal that FOLLOWS an ambiguous attempt of this inner write proves nothing about that + /// attempt, so it is settled by the read below instead of ending the call here. + const bool definitely_refused = !refreshed && isDefinitelyRefusedWrite(e); + if (definitely_refused && !state.any_ambiguous) + { + ProfileEvents::increment(ProfileEvents::CASRequestRefused); + return Refused{e.code(), e.message(), state.attempts_sent}; + } + /// A refusal-class exception is never a hint, even when an earlier ambiguity of this inner + /// write kept it from ending the call above: that earlier attempt's fate is what the read + /// below must settle, and a hint reissue would skip it. Both counters below are recorded + /// here, at classification, regardless of `policy.single_attempt` (`Retry::once` never acts + /// on either, but the attempt's transport error still named what it named) -- except when + /// the credential answer OWNS the reissue: a credential answer whose text happens to also + /// match a hint or fuse text is a credential reissue, not a hint or fuse one, and must not + /// inflate these counts. When an earlier ambiguity of this inner write keeps the credential + /// answer from owning the reissue, the hint/fuse mechanism reissues it instead, exactly as + /// if credentials had never been refreshed, so the counter must still count it. + connect_hint = !definitely_refused && isConnectFailureHint(e); + if (connect_hint && !refresh_owns_reissue) + ProfileEvents::increment(ProfileEvents::CASRequestConnectFailureHint); + /// Checked AFTER the hint, so a hinted attempt stays hinted (reissued before its read); a + /// fuse timeout is reissued after the settle read runs below. + fuse = isFirstAttemptFuseTimeout(e, state.attempts_sent); + if (fuse && !refresh_owns_reissue) + ProfileEvents::increment(ProfileEvents::CASRequestFirstAttemptFuse); + } + catch (const std::exception & e) + { + /// Only the transport can have landed anything, and every exception it raises is a + /// `Poco::Exception`. A local fault -- a bad allocation, a logic error raised inside the + /// attempt -- is not a store answer, and settling it by a read would bury the bug behind an + /// outcome the store never gave. What is left is an unmodeled transport failure that may + /// still have landed, so `outcome` stays disengaged and the read below settles it. + if (!dynamic_cast(&e)) + throw; + } + + if (outcome && outcome->has_value()) + { + if (auto inc = owner.tryMint(key, std::move(**outcome))) + return postCommit(std::move(*inc), /*resolved_by_read=*/false, state, bound); + /// A 2xx carrying a value no grammar accepts: the write may well have landed, so this is an + /// ambiguity to settle by reading, never a corruption verdict about the object. + outcome.reset(); + } + if (!outcome && !credential_answer) + state.any_ambiguous = true; + + /// Nothing for a read to settle: this attempt did not apply, and no EARLIER attempt of this + /// inner write is unresolved either. Re-send it under the credentials the refresh installed. + if (refresh_owns_reissue) + { + if (auto given_up = pauseAndReissue(state, bound)) + return *given_up; + continue; + } + + /// The failure text named a failed CONNECTION -- counted above, at classification, whether or + /// not this policy or the deadline/fence gates below let the reissue actually happen. A read now + /// would meet the same broken condition, so the reissue itself is the cheaper probe: the attempt + /// stays ambiguous (`any_ambiguous` is set above), and if the reissue meets a refused precondition + /// the read below settles it. + if (connect_hint && !policy.single_attempt) + { + if (auto given_up = pauseFlat(state, bound)) + return *given_up; + continue; + } + + /// Every refused precondition, and every ambiguous attempt that was not reissued on a + /// connect-failure hint, is settled by ONE exact read, under every policy: a refused + /// precondition does not say WHO holds the key, and a 404 and a 412 reach here as the same + /// answer. A hinted attempt reaches this read only when its reissue meets a refused + /// precondition. A refused precondition needs only to know WHAT is there, so a presence-only + /// caller settles it with a HEAD; proving an ambiguous attempt landed needs the bytes, and there + /// the body read is unavoidable. + ProfileEvents::increment(ProfileEvents::CASRequestResolveRead); + const Resolved resolved = resolve_refusal_with == ResolveWith::Presence && !state.any_ambiguous + ? observePresence(key, policy, bound) + : observe(key, policy, bound); + state.last_seen = resolved.seen; + /// A bound refused the resolve, so say WHICH. Erasing it here is what let a lost fence be + /// reported as an ordinary conflict and a lease refusal as a policy deadline. + if (resolved.stop) + return gaveUpForReadStop(*resolved.stop, state, bound); + + /// No attempt of this inner write is unresolved, so the refused precondition IS the answer. + /// Identical bytes here are somebody else's object, and the caller that owns the key's meaning + /// decides what that means. + if (!state.any_ambiguous) + return Conflict{state.last_seen, state.attempts_sent, state.any_ambiguous}; + + if (!preconditionStillSatisfiable(state.last_seen, expected)) + { + /// The precondition has MOVED, and THAT is what makes byte equality a proof: an attempt + /// that never applied leaves the key exactly as it found it, so under an incarnation that + /// did not move our own bytes cannot be told from the bytes already there. A create reaches + /// here for every object it sees -- an occupied key never satisfies its precondition. + if (const auto * obj = std::get_if(&state.last_seen); obj && obj->bytes == bytes) + return postCommit(obj->etag, /*resolved_by_read=*/true, state, bound); + /// Nothing of this inner write's is at the key, and a reissue would be refused too. + return Conflict{state.last_seen, state.attempts_sent, state.any_ambiguous}; + } + + /// Unresolved but repeatable: the precondition would still be met -- or nothing was observed at + /// all, which proves neither way and leaves this write's ambiguity alive. A reissue is what + /// settles it, and re-sending the same bytes under the same precondition is safe: it ends at + /// the store's own answer. A policy with no reissue has to say it settled nothing. + if (policy.single_attempt) + return gaveUp(GaveUp::Why::Unresolved, sourceFor(bound), state); + /// The first attempt met the adaptive first-attempt timeout: a connection-quality answer, not a + /// store fault. The read above settled nothing new about it, so re-send at once as attempt 2 + /// under the full attempt budget; the backoff index is untouched because no store fault was + /// seen yet. + if (fuse) + { + if (auto given_up = reissueAtOnce(state, bound)) + return *given_up; + continue; + } + if (auto given_up = pauseAndReissue(state, bound)) + return *given_up; + } +} + +WriteResult CasOperation::create(const String & key, const String & bytes, const Retry & policy) +{ + WriteState state; + return writeLoop(key, bytes, std::nullopt, policy, policy.bind(owner.now_ms()), state, ResolveWith::Body); +} + +WriteResult CasOperation::replace(const String & key, const String & bytes, const Etag & seen, const Retry & policy) +{ + WriteState state; + return writeLoop(key, bytes, seen, policy, policy.bind(owner.now_ms()), state, ResolveWith::Body); +} + +WriteResult CasOperation::readModifyWrite(const String & key, const DecideOnObject & decide, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + WriteState state; + + Resolved resolved = observe(key, policy, bound); + state.last_seen = resolved.seen; + std::optional current; + if (const auto * obj = std::get_if(&state.last_seen)) + current = *obj; + else if (!std::holds_alternative(state.last_seen)) + return gaveUpAfterFailedObservation(resolved.stop, state, bound); + + for (;;) + { + /// Outside every classification: `decide` is the caller's own control flow, and an exception + /// from it is theirs to see unchanged. + const std::optional next = decide(current); + if (!next) + return Declined{state.last_seen}; + + WriteResult result = writeLoop(key, *next, + current ? std::optional(current->etag) : std::nullopt, policy, bound, state, + ResolveWith::Body); + if (!std::holds_alternative(result)) + return result; + + /// The write's own resolve read IS this iteration's read: a hot key never costs a second GET + /// per conflict. + if (const auto * obj = std::get_if(&state.last_seen)) + current = *obj; + else if (std::holds_alternative(state.last_seen)) + current.reset(); + + if (policy.single_attempt) + return result; + /// A clean lost race is settled: the resolve read holds the fresh object and the next + /// iteration decides on it. Only a conflict that settled a transport fault is paced by the + /// growing schedule. + if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, bound) : pauseForConflict(state, bound)) + return *given_up; + + /// Only when the resolve settled nothing is a fresh read owed; otherwise `current` already is + /// what the store held. + if (std::holds_alternative(state.last_seen)) + { + resolved = observe(key, policy, bound); + state.last_seen = resolved.seen; + if (const auto * obj = std::get_if(&state.last_seen)) + current = *obj; + else if (std::holds_alternative(state.last_seen)) + current.reset(); + else + return gaveUpAfterFailedObservation(resolved.stop, state, bound); + } + } +} + +WriteResult CasOperation::readModifyWriteOnPresence(const String & key, const DecideOnMeta & decide, const Retry & policy) +{ + const Retry::Bound bound = policy.bind(owner.now_ms()); + WriteState state; + + Resolved resolved = observePresence(key, policy, bound); + state.last_seen = resolved.seen; + std::optional current; + if (const auto * meta = std::get_if(&state.last_seen)) + current = *meta; + else if (!std::holds_alternative(state.last_seen)) + return gaveUpAfterFailedObservation(resolved.stop, state, bound); + + for (;;) + { + const std::optional next = decide(current); + if (!next) + return Declined{state.last_seen}; + + WriteResult result = writeLoop(key, *next, + current ? std::optional(current->etag) : std::nullopt, policy, bound, state, + ResolveWith::Presence); + /// A refused precondition was settled by a HEAD, but proving an ambiguous attempt landed needs + /// the bytes; this loop is presence-only by contract, so that body stops here. + state.last_seen = withoutBody(std::move(state.last_seen)); + if (!std::holds_alternative(result)) + return withoutBody(std::move(result)); + + if (const auto * meta = std::get_if(&state.last_seen)) + current = *meta; + else if (std::holds_alternative(state.last_seen)) + current.reset(); + + if (policy.single_attempt) + return Conflict{state.last_seen, state.attempts_sent, state.any_ambiguous}; + /// A clean lost race is settled: the resolve read holds the fresh object and the next + /// iteration decides on it. Only a conflict that settled a transport fault is paced by the + /// growing schedule. + if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, bound) : pauseForConflict(state, bound)) + return *given_up; + + if (std::holds_alternative(state.last_seen)) + { + resolved = observePresence(key, policy, bound); + state.last_seen = resolved.seen; + if (const auto * meta = std::get_if(&state.last_seen)) + current = *meta; + else if (std::holds_alternative(state.last_seen)) + current.reset(); + else + return gaveUpAfterFailedObservation(resolved.stop, state, bound); + } + } +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h new file mode 100644 index 000000000000..dd79f1a9db4b --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h @@ -0,0 +1,527 @@ +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// TRUE when the store's own answer proves this write never applied: a malformed request, an entity +/// too large, an access denial, or a credential failure. Whether a STALE CREDENTIAL explains it is not +/// asked here -- the engine asks the backend for fresh credentials first, and refuses only when no +/// refresh HELPED: the error was outside the credential class, the one refresh this call is allowed +/// installed nothing, or there is no reissue left to sign with what it did install. A refresh that +/// helps re-sends the attempt, which is known not to have applied. A non-S3 exception is never a +/// refusal: an unmodeled error may have landed. +bool isDefinitelyRefusedWrite(const std::exception & e); + +/// TRUE when a transport failure's text says the CONNECTION itself failed: no free local port, a +/// refused or unreachable peer, or the connect poll's own timeout. A hint, not a verdict: the same +/// errno can be reported after `send` or `recv`, so a hinted attempt keeps every property of an +/// ambiguous one; what the hint changes is only that the engine reissues before spending a read. +bool isConnectFailureHint(const std::exception & e); + +/// TRUE when a FIRST physical attempt (`attempt_no == 1`) failed with the adaptive first-attempt +/// timeout: an `S3Exception` naming `NETWORK_CONNECTION` whose text is the generic transport-timeout +/// one, not a connect-failure hint (checked first, so a hinted attempt stays hinted -- the failed +/// connection it names is a different condition from the fuse, and must not be claimed by it). A +/// connection-quality answer about a fresh connection, not a store fault -- attempt 2 runs under the +/// full attempt budget, so the right response is to re-send at once rather than pace it like a fault. +bool isFirstAttemptFuseTimeout(const std::exception & e, size_t attempt_no); + +/// Deterministic caller/local bugs, surfaced unchanged by every loop here: reissuing only replays the +/// same failure and buries the root cause behind a retryable exception. The set is `LOGICAL_ERROR`, +/// `NOT_IMPLEMENTED`, `BAD_ARGUMENTS` and `CORRUPTED_DATA`. +bool isDeterministicLocalFailure(int code); + +/// Throw the recoverable "CAS write could not be committed, retry later" condition. +/// +/// WHY NETWORK_ERROR (this replaces an earlier ABORTED throw): +/// A content-addressed write can fail for a reason that is neither the caller's fault +/// nor permanent: the mount-lease / write fence was lost (e.g. a renewal PUT timed out +/// against a slow or throttling object store), or a conditional PUT exhausted its retry +/// budget mid-outage. The right response is "abandon this attempt, try again later" -- +/// which is precisely what a transient error means. +/// +/// It previously threw `ABORTED`, which was actively harmful to background merges: +/// `ReplicatedMergeMutateTaskBase` treats `ABORTED` as "merge deliberately cancelled +/// (shutdown / `DROP` / merges-blocker), not an error", so it neither records +/// `last_exception_time_ms` nor lets `ReplicatedMergeTreeQueue`'s exponential backoff +/// engage. Under a sustained store outage the queue re-executed the merge roughly every +/// 2 seconds, recomputing the whole (possibly multi-GiB) output part every time for the +/// entire outage -- hundreds of full recomputes, and invisible in system.replication_queue. +/// +/// `NETWORK_ERROR` is the best-fitting EXISTING code: +/// - it is NOT in the merge "retry silently, no backoff" exemption set (only `ABORTED` +/// and `PART_IS_TEMPORARILY_LOCKED` are), so the existing backoff -- capped by +/// `max_postpone_time_for_failed_replicated_merges_ms` -- engages automatically; +/// - it is already in ClickHouse's transient/retryable taxonomy +/// (`checkDataPart::isRetryableException` lists it beside `ABORTED`), so a part under +/// verification is not misread as corrupted; +/// - nothing on the merge / insert / replication commit path special-cases it in a way +/// that would misfire (ZooKeeper retriability keys on `Coordination::Exception`, a +/// different type), and it is not caught specially on the CAS write path. +/// +/// SCOPE: only the ESCAPING retry-later throws route here (fence lost or a controlled write outcome +/// remaining uncertain). Startup/decommission and generic live-lock-brake `ABORTED` values keep +/// their meaning and are not rerouted here. +[[noreturn]] void throwCasWriteRetryLater(const String & why); + +/// Same classification as `throwCasWriteRetryLater`, but returns the exception as a +/// `std::exception_ptr` for call sites that fail a pending future/promise rather than throw directly. +/// Both entry points route through the SAME construction internally, so the error code / message shape +/// has exactly one place that decides it. +std::exception_ptr makeCasWriteRetryLaterExceptionPtr(const String & why); + +/// Throw the recoverable "this content-addressed disk cannot serve the request right now" condition. +/// Sibling of `throwCasWriteRetryLater`, same class for the same reasons, differing only in what it +/// describes: that one names a WRITE whose commit did not land, this one names a DISK STATE that +/// refused the request before it started -- on either plane. +/// +/// `subject` names the refusing disk or pool (e.g. "content-addressed disk 'ca'") and `condition` states +/// the CA condition truthfully, INCLUDING any promise about how it clears -- only the site knows whether +/// it can make one. What is appended HERE is the classification alone, so it cannot drift between call +/// sites. Unlike `throwCasWriteRetryLater` this deliberately does not log: these sites can fire +/// repeatedly per refused operation and every caller already reports the exception it receives. +[[noreturn]] void throwCasTransientUnavailable(const String & subject, const String & condition); + +/// The failure class a FRESH CREDENTIAL could fix -- a subset of `isDefinitelyRefusedWrite`, exposed +/// because a caller whose own loop makes the next physical attempt must not treat it as terminal: the +/// engine refreshes once before it gives the answer, and the caller's next attempt signs with what the +/// refresh installed. +bool isRefreshableCredentialError(const std::exception & e); + +/// Facts the fence cannot see, sampled by the caller. Non-throwing; FALSE ends the operation exactly +/// like a lost fence, because the engine does not need to know which of the two refused. +using Liveness = std::function; +/// `decide` sees the current object (or absence) and returns the bytes to write, or nullopt for +/// "nothing to do". It may throw: the exception is the caller's control flow and propagates unchanged. +using DecideOnObject = std::function(const std::optional &)>; +using DecideOnMeta = std::function(const std::optional &)>; + +/// One key returned by `list`. `etag` is present only on a backend that surfaces per-key +/// incarnations through LIST -- see `Backend::supportsListTokens`. +struct ListedKey +{ + String key; + uint64_t size; + std::optional etag; +}; +/// One page of an enumeration. `next_cursor` resumes strictly after the last returned key; empty +/// marks the end. +struct ListPage +{ + std::vector keys; + String next_cursor; +}; +/// The walk's callback: FALSE stops the walk. +using ListedKeyFn = std::function; + +namespace detail +{ +/// The engine's per-attempt counters, behind functions so the header need not declare the events. +void recordAttempt(); +void recordReissue(); +void recordConflictPause(); +void recordFirstAttemptFuse(); +} + +class CasOperation; + +/// The only caller of `Backend`. It owns the three things a physical request must be measured +/// against -- the transport, the mount fence, and the clock -- and it is the sole minter of +/// `Etag`, so a caller can hold one only by way of a request this class admitted. +/// +/// It is constructed with a fence because a fence is a property of whoever holds the lease, not of a +/// call: the mount plane passes the mount fence, the GC plane and the offline tools an open one. A +/// verb is never called here; verbs live on `CasOperation`, which carries the generation the caller +/// was admitted under. +class CasRequests +{ +public: + /// `now_ms` defaults to `CLOCK_BOOTTIME` milliseconds -- the same clock a mount lease deadline is + /// expressed on, so `Retry::untilLeaseSafe` and this engine compare like with like. `sleep_ms` + /// defaults to a real sleep. `attempt_reservation_ms` is taken from the backend's own attempt + /// envelope (attempt timeout plus its connect caps): it is what the engine reserves before it + /// starts anything. `hot_keys` is the pool's + /// write lane, shared by its planes; without one this object owns a private lane with no cache, + /// so a write through it costs today's read and write. + CasRequests(BackendPtr backend_, Fence fence_, + std::function now_ms_ = {}, std::function sleep_ms_ = {}, + CasHotKeys * hot_keys_ = nullptr); + + /// Both may be called concurrently on one `CasRequests`: neither writes a member, and the only + /// state either reads is the backend and the fence -- whose closures must therefore be thread-safe + /// too. The test setters below DO write members, and belong to setup, before any operation runs. + /// + /// Admitted now, under the fence's current generation. + CasOperation admit(Liveness liveness = {}); + /// Admitted earlier: the generation came from a persisted runtime record, and an operation that + /// resumes under a generation the fence has since moved past gives up rather than writing. + CasOperation resume(uint64_t admitted_generation, Liveness liveness = {}); + + /// The capability predicates and `dialect()`. + Backend & backendForCapabilityPredicates() { return *backend; } + + /// A caller's own inter-iteration wait, paced through the same clock the engine's own sleeps use -- + /// so a test that replaces the sleep sees no real time pass in either. `CasOperation` carries the + /// same call for the loops that hold an operation rather than the plane it was admitted on. + void pause(uint64_t ms) { sleep_ms(ms); } + + void setNowFnForTest(std::function now_ms_); + void setSleepFnForTest(std::function sleep_ms_); + void setAttemptReservationForTest(uint64_t ms) { attempt_reservation_ms = ms; } + + /// What every write on this plane reserves before it starts an attempt (the backend's own attempt + /// envelope). A caller that derives its OWN policy window from a write's cost -- rather than from a + /// constant that predates this reservation -- reads it here instead of duplicating the backend call. + uint64_t attemptReservationMs() const { return attempt_reservation_ms; } + +private: + friend class CasOperation; + /// `CasOperation::owner` is a `CasRequests &`: `CasOperation`'s own friendship with `CasHotKeys` + /// does not extend to what that reference points at, so the lane needs its own grant to reach the + /// clock and sleep it reads and paces through `op.owner`. + friend class CasHotKeys; + + /// The one place a transport key is created. Every verb reaches the store through this, so no + /// engine code -- and nothing outside it -- can name the key's type, let alone construct one. + /// `attempt_no` is the caller's own 1-based physical-attempt count, threaded to the transport + /// through `TransportAccess::attemptNo()` so a reissue is seen as attempt >= 2. + template + auto withTransportAccess(size_t attempt_no, Fn && fn) + { + TransportAccess access(attempt_no); + return std::forward(fn)(access); + } + + /// The store's answer for `key`, as an incarnation. Throws `CORRUPTED_DATA` naming the key when + /// the value fails this backend's dialect grammar. + Etag mint(const String & key, String value) const; + /// `mint` without the verdict, for the one caller that must treat a malformed value as an + /// ambiguity to settle by reading rather than as corruption: a write's own 2xx response. + std::optional tryMint(const String & key, String value) const; + /// The transport value to send as a precondition. Throws `LOGICAL_ERROR` when the incarnation + /// names another key or another backend -- a precondition built from it would silently mean + /// something else. + const String & valueFor(const String & key, const Etag & inc) const; + + BackendPtr backend; + Fence fence; + std::function now_ms; + std::function sleep_ms; + uint64_t attempt_reservation_ms; + /// The private lane of a `CasRequests` built without a pool; null when `hot_keys` is the pool's. + std::unique_ptr own_hot_keys; + CasHotKeys * hot_keys; +}; + +/// The cap on one `removeManyWriteOnce` chunk -- also the ceiling a batch-delete request can carry. +inline constexpr size_t kBulkDeleteMaxKeys = 1000; + +/// One admitted operation: the unit a policy, a fence generation and a liveness predicate apply to. +/// Move-only, and every request it makes re-checks its admission -- before each attempt, before each +/// sleep, and once more after a proven commit, so a write whose fence was lost while it was in flight +/// is never reported as committed. +/// +/// SINGLE-THREADED: it carries mutable per-call state, so one operation belongs to one task. A caller +/// that fans work out gives each task its own, built from `generation()` through `CasRequests::resume`. +class CasOperation +{ +public: + CasOperation(CasOperation &&) = default; + CasOperation(const CasOperation &) = delete; + CasOperation & operator=(const CasOperation &) = delete; + + uint64_t generation() const { return admitted_generation; } + /// A caller's own inter-iteration wait, paced through the same clock the engine's own sleeps use -- + /// so a test that replaces the sleep sees no real time pass in either. For the hand-written loops + /// that reissue something the engine must not reissue for them. + void pause(uint64_t ms) { owner.sleep_ms(ms); } + /// The verdict point: is this operation still admitted? For the sites that guard a decision rather + /// than a request. + bool admitted() const { return gate(0) == Gate::Ok; } + + /// The write lane for keys several writers of this pool share. + CasHotKeys & hotKeys() const { return *owner.hot_keys; } + + /// `policy` with its window turned into an absolute deadline on this operation's clock, taken NOW. + /// A hand-written loop freezes its policy once before it starts and passes the frozen value to + /// every call it makes, so the loop ends when the window it was given ends -- rather than granting + /// each verb of each iteration a fresh one, which is how a bounded document promise became hours + /// of paced retrying. A policy that already carries a deadline is returned unchanged, so freezing + /// twice cannot extend it. + Retry freeze(const Retry & policy) const; + + std::optional read(const String & key, const Retry & policy); + std::optional head(const String & key, const Retry & policy); + ListPage list(const String & prefix, const String & cursor, size_t limit, const Retry & policy); + /// Walks every key under `prefix` exactly once. The policy governs EACH PAGE, not the walk: a walk + /// is an unbounded number of requests, and a silently truncated enumeration is the error a + /// coverage record exists to prevent. `on_page_fetched` fires once per page DELIVERED; a page that + /// took several reissues still fires once, and `CASRequestAttempt` is the physical count. + void forEachListedKey(const String & prefix, const ListedKeyFn & fn, const Retry & per_page, + size_t page_limit = 1000, const std::function & on_page_fetched = {}); + Removal remove(const String & key, const Etag & seen, const Retry & policy); + /// `head` then `remove` of what it saw, repeating on `Mismatch`. `Gone` when the key is already + /// absent; never returns `Mismatch` -- under `once`, where there is no reissue to resolve one, a + /// `Mismatch` is the retry-later throw the read verbs use when their policy is exhausted. + Removal removeCurrent(const String & key, const Retry & policy); + /// Deletes ONE chunk of up to `kBulkDeleteMaxKeys` write-once keys as one request under the + /// policy: admission, fence, budget, deadline, backoff and reissue exactly as `remove`. A reissue + /// resends the whole chunk; a key the failed attempt already deleted is absent, and absence is + /// success. Throws when the policy is exhausted. More keys than the cap is a caller bug: the + /// consumer chunks, so that every chunk that succeeded is recorded before a later one can fail. + void removeManyWriteOnce(const std::vector & keys, const Retry & policy); + /// The one primitive that reports failure as a value, so the policy reissues on the OUTCOME: + /// `Indeterminate` is retried, the four authoritative outcomes return at once, and an + /// `Indeterminate` that outlives the bound is returned rather than thrown. Admission refused before + /// the first attempt still throws -- nothing was probed. + SentinelProbeResult probeSentinel(const String & key, const Retry & policy); + /// The OPEN is under the policy; the body is the SDK's. + std::unique_ptr stream(const String & key, const Retry & policy); + /// The INITIATION is under the policy; the transfer is the SDK's. + void publish(const BlobPublishRequest & request, const Retry & policy); + + WriteResult create(const String & key, const String & bytes, const Retry & policy); + WriteResult replace(const String & key, const String & bytes, const Etag & seen, const Retry & policy); + /// Read, decide, write, and re-decide on conflict against what the write's own resolve read + /// already observed. `decide` returning nullopt is `Declined`. + WriteResult readModifyWrite(const String & key, const DecideOnObject & decide, const Retry & policy); + /// The same loop over `head`, settling a refused precondition with a `head` too. Proving that an + /// ambiguous attempt landed needs the bytes, so that one path does read a body; either way the verb + /// reports a `Meta` and never an `Object`. + WriteResult readModifyWriteOnPresence(const String & key, const DecideOnMeta & decide, const Retry & policy); + +private: + friend class CasRequests; + /// The lane waits on this operation's own admission and reads and writes through its verbs; it + /// needs the gate, the reservation, the resolve read and the give-up helpers, never the transport. + friend class CasHotKeys; + + CasOperation(CasRequests & owner_, uint64_t admitted_generation_, Liveness liveness_) + : owner(owner_), admitted_generation(admitted_generation_), liveness(std::move(liveness_)) + { + } + + enum class Gate : uint8_t { Ok, FenceLost, NoBudget }; + /// The admission point: the fence for `needed_ms` from now, then the caller's own facts. + Gate gate(uint64_t needed_ms) const; + + /// Everything ONE logical write call accumulates. `attempts_sent`, `sent_any`, `last_seen`, + /// `reissues` and `refresh_attempted` outlive each attempt and, for `readModifyWrite`, each inner + /// write, so a `GaveUp` reports what the whole call did rather than what its last attempt did. + struct WriteState + { + uint32_t attempts_sent = 0; + bool sent_any = false; + /// The one field that belongs to the INNER write instead: did any of ITS attempts end without + /// proof of whether it applied? Every attempt of one inner write sends the same bytes, which is + /// what makes "the resolve read found our bytes" a statement about an attempt of ours; an inner + /// write that ended in `Conflict` saw the precondition move, which proves its ambiguous + /// attempts dead. A credential answer never sets it: the store gives one before applying + /// anything. + bool any_ambiguous = false; + Observation last_seen = NotObserved{}; + uint32_t reissues = 0; + bool refresh_attempted = false; + }; + + /// Why a read-class request stopped without an answer. Every give-up below throws the same + /// `NETWORK_ERROR`, so a caller that SWALLOWS the exception -- only the resolve read does -- cannot + /// recover from it which bound refused, and reporting a lease refusal as a policy deadline is the + /// confusion `GaveUp::Source` exists to prevent. `PolicyExhausted` names the `Retry` bound, whose + /// own source is `Bound::lease_bound`. + enum class ReadStop : uint8_t { FenceLost, NoBudgetLease, PolicyExhausted }; + + /// What the resolve read saw, and why it stopped when it saw nothing. `stop` is set ONLY when a + /// bound refused; a read that failed at the transport leaves it empty, and that is the one case + /// `NotObserved` is still the whole story. + struct Resolved + { + Observation seen; + std::optional stop; + }; + + /// One read-class request under the policy: admission, attempt, classification, jittered reissue. + /// Returns whatever `once` returns, or throws -- the read surface reports failure by exception. + template + auto readLoop(std::string_view verb, const String & subject, const Retry & policy, + const Retry::Bound & bound, Fn && once); + + std::optional readUnder(const String & key, const Retry & policy, const Retry::Bound & bound); + std::optional headUnder(const String & key, const Retry & policy, const Retry::Bound & bound); + ListPage listUnder(const String & prefix, const String & cursor, size_t limit, + const Retry & policy, const Retry::Bound & bound); + Removal removeUnder(const String & key, const String & expected_value, + const Retry & policy, const Retry::Bound & bound); + + /// How a write settles a REFUSED PRECONDITION, which needs only to know what is at the key. An + /// ambiguous attempt always reads the body, whichever this says, because only the bytes can prove + /// the attempt landed. + enum class ResolveWith : uint8_t { Body, Presence }; + + /// The write engine: one call, any policy. `Committed` and `Conflict` are proven by an exact read + /// or by the reissue's own 2xx before they are reported; an attempt whose transport error named a + /// failed connection is reissued before its read. `Refused`, `Declined` and `GaveUp` report what + /// the store or the bounds said. + WriteResult writeLoop(const String & key, const String & bytes, const std::optional & expected, + const Retry & policy, const Retry::Bound & bound, WriteState & state, + ResolveWith resolve_refusal_with); + /// The resolve read: an exact read under the same policy and deadline, reporting what it saw and, + /// when it saw nothing, which bound stopped it. + Resolved observe(const String & key, const Retry & policy, const Retry::Bound & bound); + /// The presence-only sibling, for the one loop that must not fetch a body. + Resolved observePresence(const String & key, const Retry & policy, const Retry::Bound & bound); + + WriteResult postCommit(Etag inc, bool resolved_by_read, WriteState & state, const Retry::Bound & bound); + WriteResult gaveUp(GaveUp::Why why, GaveUp::Source source, WriteState & state) const; + /// The bound that refused the resolve read, reported as the outcome it actually is. + WriteResult gaveUpForReadStop(ReadStop stop, WriteState & state, const Retry::Bound & bound) const; + /// A resolve read that produced nothing. The fence is NOT resampled: a second sample reports a + /// state the read never saw, which is how a lease refusal used to be reported as a policy deadline. + WriteResult gaveUpAfterFailedObservation(std::optional stop, WriteState & state, + const Retry::Bound & bound) const; + /// The shared shape behind every gated pause below: admission for `envelopes` attempt reservations + /// plus `pause_ms`, the deadline check, the counter this pause records itself under, then the sleep + /// -- called even with a zero `pause_ms` UNLESS `should_sleep` is false, which is reserved for the + /// fuse's zero-pause reissue (a pace of "no pause", not "a zero-length one"). A value means the call + /// ended during it; nullopt means the caller may send another attempt. + std::optional gatedPause(uint64_t pause_ms, uint32_t envelopes, WriteState & state, + const Retry::Bound & bound, void (*record)(), bool should_sleep); + /// Admission, then the jittered sleep. A value means the call ended during it; nullopt means the + /// caller may send another attempt. + std::optional pauseAndReissue(WriteState & state, const Retry::Bound & bound); + /// The sibling for a clean lost race: the same admission and the same reservation, a flat + /// `Retry::conflictBackoff` sleep, and `state.reissues` untouched, so a transport fault that follows + /// starts its own schedule at the beginning. + std::optional pauseForConflict(WriteState & state, const Retry::Bound & bound); + /// The sibling for a failure text that named a failed connection. The same admission and the same + /// reservation, a flat `kConnectHintPauseMs` sleep, and `state.reissues` untouched. + std::optional pauseFlat(WriteState & state, const Retry::Bound & bound); + /// The sibling for a first-attempt fuse timeout: the same admission and the same reservation, NO + /// sleep at all, and `state.reissues` untouched -- the fuse is a connection-quality answer about a + /// fresh connection, not a store fault, so nothing here is paced against it. + std::optional reissueAtOnce(WriteState & state, const Retry::Bound & bound); + + /// `sleep_ms` plus `envelopes` attempt reservations, saturating. + uint64_t reservedFor(uint64_t sleep_ms, uint32_t envelopes) const; + /// Is there room to START something needing `needed_ms` before the bound? The guarantee is on the + /// start side: nothing is begun that could not finish inside it. + bool fits(uint64_t needed_ms, const Retry::Bound & bound) const; + + /// One failed read-class attempt, classified. A credential failure is refreshed HERE so the reissue + /// signs with the new client, at most once per call -- `refresh_attempted` is the caller's, and a + /// second credential failure under the same call is classified as if no refresh were available. + /// `refreshed` is set true only when THIS call installed new credentials, mirroring the write + /// loop's own local of the same name, so the caller can keep a credential reissue from also + /// inflating a counter whose text it happens to match. TRUE means the failure must surface unchanged. + bool refreshAndClassifyReadFault(const std::exception & e, bool & refresh_attempted, bool & refreshed); + + /// Each records its cause in `last_read_stop` before throwing, so the resolve read can report it. + [[noreturn]] void giveUpReadFenceLost(std::string_view verb, const String & subject, std::string_view when); + [[noreturn]] void giveUpReadNoBudget(std::string_view verb, const String & subject, std::string_view what); + [[noreturn]] void giveUpReadDeadline(std::string_view verb, const String & subject, + const Retry::Bound & bound, uint32_t attempts_made); + + CasRequests & owner; + uint64_t admitted_generation; + Liveness liveness; + /// Written immediately before a read-class give-up throws, cleared and read only by the resolve + /// read that swallows it. Every other caller lets the exception carry the verdict. + std::optional last_read_stop; +}; + +template +auto CasOperation::readLoop(std::string_view verb, const String & subject, const Retry & policy, + const Retry::Bound & bound, Fn && once) +{ + bool refresh_attempted = false; + /// Two counters, deliberately kept separate: `attempt_no` is the PHYSICAL attempt count handed to + /// the transport (so a reissue is seen as attempt >= 2); `ordinary_reissues` is the + /// exponential-backoff index. They advance together on an ordinary failure, but the first-attempt + /// fuse below advances `attempt_no` alone (via `continue`, skipping the backoff pause) so a + /// following ordinary failure's backoff still starts from `backoff(1)`, undisturbed. + for (uint32_t attempt_no = 1, ordinary_reissues = 0;; ++attempt_no) + { + const uint64_t reservation = reservedFor(0, 1); + switch (gate(reservation)) + { + case Gate::FenceLost: giveUpReadFenceLost(verb, subject, "before the request"); + case Gate::NoBudget: giveUpReadNoBudget(verb, subject, "for one more request"); + case Gate::Ok: break; + } + if (!fits(reservation, bound)) + giveUpReadDeadline(verb, subject, bound, attempt_no - 1); + + detail::recordAttempt(); + try + { + return owner.withTransportAccess(attempt_no, [&](auto & access) { return once(access); }); + } + catch (const std::exception & e) + { + bool refreshed = false; + if (refreshAndClassifyReadFault(e, refresh_attempted, refreshed)) + throw; + /// Classified -- and counted -- before the single-attempt check below: `Retry::once` forbids + /// the REISSUE, not the observation that this attempt hit the fuse, and the write path + /// already counts at classification the same way -- except when `refreshed` is also true: a + /// credential answer whose text happens to also match the fuse text is a credential reissue, + /// not a fuse one, and must not inflate this count. + const bool fuse = isFirstAttemptFuseTimeout(e, attempt_no); + if (fuse && !refreshed) + detail::recordFirstAttemptFuse(); + if (policy.single_attempt) + throw; + /// A first-attempt fuse is a connection-quality answer, not a store fault: re-check + /// admission for the reissue alone and send it at once. `ordinary_reissues` stays put, so a + /// following ordinary failure's backoff starts at `backoff(1)`, exactly as if this attempt + /// had never happened. + if (fuse) + { + const uint64_t needed = reservedFor(0, 1); + switch (gate(needed)) + { + case Gate::FenceLost: giveUpReadFenceLost(verb, subject, "before the reissue"); + case Gate::NoBudget: giveUpReadNoBudget(verb, subject, "for the reissue"); + case Gate::Ok: break; + } + if (!fits(needed, bound)) + giveUpReadDeadline(verb, subject, bound, attempt_no); + detail::recordReissue(); + continue; + } + } + + const uint64_t pause_ms = Retry::backoff(++ordinary_reissues); + const uint64_t needed = reservedFor(pause_ms, 1); + switch (gate(needed)) + { + case Gate::FenceLost: giveUpReadFenceLost(verb, subject, "before the reissue"); + case Gate::NoBudget: giveUpReadNoBudget(verb, subject, "for the reissue"); + case Gate::Ok: break; + } + if (!fits(needed, bound)) + giveUpReadDeadline(verb, subject, bound, attempt_no); + detail::recordReissue(); + owner.sleep_ms(pause_ms); + } +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp new file mode 100644 index 000000000000..c7b4c27215b5 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp @@ -0,0 +1,32 @@ +#include + +#include + +#include +#include + +namespace DB::Cas +{ + +uint64_t Retry::backoff(uint32_t attempt) +{ + if (attempt == 0) + return 0; + const uint32_t doublings = std::min(attempt - 1, 20); + const uint64_t ceiling = std::min(5000, 200ull << doublings); + return thread_local_rng() % (ceiling + 1); /// full jitter: uniform(0, ceiling) +} + +Retry::Bound Retry::bind(uint64_t now_ms) const +{ + const uint64_t own_deadline_ms = policy_deadline_ms + ? *policy_deadline_ms + : (now_ms > std::numeric_limits::max() - window_ms + ? std::numeric_limits::max() + : now_ms + window_ms); + if (lease_deadline_ms && *lease_deadline_ms < own_deadline_ms) + return {*lease_deadline_ms, true}; + return {own_deadline_ms, false}; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h new file mode 100644 index 000000000000..de235e8fd3fa --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h @@ -0,0 +1,80 @@ +#pragma once +#include +#include + +namespace DB::Cas +{ + +/// A retry policy for one logical CAS write, expressed WITHOUT touching a clock: `window_ms` is the +/// policy's own budget measured from the call's start, and `lease_deadline_ms` (already reduced by +/// the caller's safety margin) is an absolute bound on whatever clock the caller's mount lease is +/// tracked against. Binding a policy to an absolute deadline is deferred to `bind`, which takes the +/// caller's own `now_ms` -- `CasRequests` runs on an injected clock in tests, so a `Retry` built at +/// construction time from the real clock could never be exercised deterministically, and `GaveUp` +/// could not tell a lease-caused deadline from a policy-caused one without recording which bound won. +struct Retry +{ + uint64_t window_ms; + std::optional lease_deadline_ms; + bool single_attempt; + /// An ABSOLUTE bound on the caller's own clock, replacing `window_ms` in `bind`. A hand-written + /// loop freezes one before it starts (`CasOperation::freeze`) and shares it across every call it + /// makes, so the loop as a whole ends when that deadline does instead of granting each of its + /// iterations a fresh window. Empty for a single verb, which gets its window from where it is + /// called. + std::optional policy_deadline_ms = std::nullopt; + + /// Full jitter: uniform(0, min(5000, 200 << (attempt-1))) milliseconds. `attempt` is 1-based; + /// `attempt == 0` returns 0. + static uint64_t backoff(uint32_t attempt); + + /// The flat pause after a clean lost race: `backoff(1)`, uniform over [0, 200] ms. It does not grow + /// with the writer's loss count, because a settled conflict has nothing to wait for but + /// desynchronisation from its competitors, and growing it with the loss count made the oldest + /// loser the slowest and the likeliest to lose again. A conflict that settled a transport fault + /// keeps `backoff(attempt)`: the fault is what must pace the loop. + static uint64_t conflictBackoff() { return backoff(1); } + + /// A policy with `ms` milliseconds of its own budget and no lease bound. + static Retry within(uint64_t ms) { return {.window_ms = ms, .lease_deadline_ms = std::nullopt, .single_attempt = false}; } + /// `within(90'000)` -- the default write policy. + static Retry standard() { return within(90'000); } + /// A policy bound by the mount lease: `lease_deadline_ms` minus `margin`, clamped at 0 -- never + /// risk a write landing after this node's fence may already be gone. `window_ms` defaults to the + /// standard 90 s write budget; a caller whose own budget is deliberately much smaller (the + /// graceful-shutdown farewell, whose window is derived from what ONE write costs, not from the + /// standard policy) passes its own window explicitly, and `bind` still takes whichever of the two + /// bounds is smaller. + static Retry untilLeaseSafe(uint64_t lease_deadline_ms, uint64_t margin, uint64_t window_ms = 90'000) + { + return {.window_ms = window_ms, + .lease_deadline_ms = lease_deadline_ms > margin ? lease_deadline_ms - margin : 0, + .single_attempt = false}; + } + /// The standard policy, but at most one attempt is ever sent. + static Retry once() { return {.window_ms = 90'000, .lease_deadline_ms = std::nullopt, .single_attempt = true}; } + + /// This policy, made single-attempt. A frozen loop policy keeps its absolute deadline through it, + /// which is what lets a loop send an unrepeatable request under the same bound as the rest. + Retry asSingleAttempt() const + { + Retry copy = *this; + copy.single_attempt = true; + return copy; + } + + /// A policy bound to an absolute deadline on the caller's own clock, plus which bound produced + /// it -- the caller's own budget, or the (smaller) lease bound. + struct Bound + { + uint64_t deadline_ms; + bool lease_bound; + }; + /// Bind this policy to `now_ms`: `deadline_ms = min(policy_deadline_ms ?: now_ms + window_ms, + /// lease_deadline_ms)`, `lease_bound` true exactly when the lease bound was the smaller of the + /// two. Called once at call entry with the owner's `now_ms()`. A frozen policy therefore keeps + /// the lease bound honest: the smaller of the two still wins, and still says so. + Bound bind(uint64_t now_ms) const; +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.cpp index af52f2854bfd..d53812acac21 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.cpp @@ -6,9 +6,9 @@ namespace DB::Cas { -SentinelProbeResult probeSentinel(Backend & backend, const String & key) +SentinelProbeResult probeSentinel(CasOperation & op, const String & key, const Retry & policy) { - return backend.probeSentinelRaw(key); + return op.probeSentinel(key, policy); } namespace @@ -29,42 +29,67 @@ bool isProbeSubtreeDebris(const String & probe_root, const String & key) } -BootstrapResidual probePoolBootstrapResidual(Backend & backend, const Layout & layout) +BootstrapResidual probePoolBootstrapResidual(CasOperation & op, const Layout & layout) { const String pool_meta_key = layout.poolMetaKey(); const String catalog_key = layout.refCatalogKey(); const String prefix = layout.poolPrefix() + "/"; const String probe_root = layout.poolPrefix() + "/_probe/"; - /// Classification is order-independent for correctness: every listed key is examined, and finding - /// `_pool_meta` anywhere is decisive. It relies on lexicographic LIST order only for COST — `_pool_meta` - /// sorts first under `/`, so a healthy pool short-circuits on the first page rather than - /// enumerating its whole content on every open. + /// `_pool_meta` present is decisive on its own, and one exact read of that key answers it without + /// enumerating anything: a LIST page over a large pool is the most expensive request a store + /// answers (it must enumerate and sort the prefix), and a store that is merely slow to LIST would + /// otherwise refuse to reopen a pool it can read perfectly well. So the existing-pool case is + /// settled by the read, and the LIST below runs only when the key is absent -- which is exactly + /// the case that needs the absence-of-residue proof. A read that could not settle presence (a + /// refusal, a deadline) is not a verdict; it falls through to the LIST, which may still see the + /// key. + try + { + if (op.read(pool_meta_key, Retry::standard())) + return BootstrapResidual::PoolMetaPresent; + } + catch (...) + { + LOG_WARNING(getLogger("CasBootstrap"), + "Pool prefix '{}': the exact read of '{}' could not settle whether the pool exists; " + "falling back to the residual LIST: {}", + prefix, pool_meta_key, getCurrentExceptionMessage(/*with_stacktrace=*/false)); + } + + /// The LIST answers one question: is there anything under the prefix besides the ignorable + /// battery debris (and, as the one retryable exception, the canonical empty catalog)? The first + /// residual key settles it, so the walk stops there. A probe-only or catalog-only prefix is walked + /// to completion -- debris is normally a few keys but every interrupted open leaves one more, so + /// it may span pages -- and the page is kept small because the enumeration cost of a large prefix + /// is what a slow store cannot deliver within an attempt. `_pool_meta` is still recognised if the + /// walk meets it (the exact read above may have been refused): it sorts before every key family + /// CAS itself writes under `/`, so on a lexicographic listing it is met before anything + /// that could have stopped the walk. Foreign residue that sorts earlier, or a listing that is not + /// lexicographic, can only make this refuse an existing pool, never bootstrap over one. + constexpr size_t page_limit = 32; + bool has_pool_meta = false; bool has_residual = false; bool has_catalog = false; try { - String cursor; - for (;;) + op.forEachListedKey(prefix, [&](const ListedKey & listed) -> bool { - const ListPage page = backend.list(prefix, cursor, 1000); - for (const ListedKey & listed : page.keys) + if (listed.key == pool_meta_key) + { + has_pool_meta = true; + return false; /// decisive — the pool is authoritative; stop the walk + } + if (isProbeSubtreeDebris(probe_root, listed.key)) + return true; /// crash leftover / concurrent opener's battery — ignore ([D2]) + if (listed.key == catalog_key) { - if (listed.key == pool_meta_key) - return BootstrapResidual::PoolMetaPresent; /// decisive — the pool is authoritative - if (isProbeSubtreeDebris(probe_root, listed.key)) - continue; /// crash leftover / concurrent opener's battery — ignore ([D2]) - if (listed.key == catalog_key) - { - has_catalog = true; - continue; - } - has_residual = true; /// a non-`_probe` object, and no `_pool_meta` seen (so far) + has_catalog = true; + return true; } - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } + has_residual = true; /// a non-`_probe` object, and no `_pool_meta` seen — decisive too + return false; + }, Retry::standard(), page_limit); } catch (...) { @@ -77,6 +102,8 @@ BootstrapResidual probePoolBootstrapResidual(Backend & backend, const Layout & l prefix, getCurrentExceptionMessage(/*with_stacktrace=*/false)); return BootstrapResidual::Indeterminate; } + if (has_pool_meta) + return BootstrapResidual::PoolMetaPresent; if (has_residual) return BootstrapResidual::ResidualWithoutMeta; if (!has_catalog) @@ -88,7 +115,7 @@ BootstrapResidual probePoolBootstrapResidual(Backend & backend, const Layout & l /// license to mint `_pool_meta`. try { - const auto got = backend.get(catalog_key); + const auto got = op.read(catalog_key, Retry::standard()); if (!got) return BootstrapResidual::ResidualWithoutMeta; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.h index c0d926f1c31d..54862280726f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasSentinelProbe.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include namespace DB::Cas @@ -9,11 +9,17 @@ namespace DB::Cas /// timeouts / 5xx / connection errors => Indeterminate; permission errors => AccessDenied; /// missing container/bucket/prefix-parent => ContainerAbsent; a clean authoritative miss => KeyAbsent. /// -/// Free-function entry point (spec §2) — a thin dispatch to the backend's own typed-evidence -/// classification (`Backend::probeSentinelRaw`; see there for the per-backend semantics: the -/// S3-native raw HEAD error, the Local container-directory stat, or the generic head/get-based -/// default for a backend without sharper evidence). -SentinelProbeResult probeSentinel(Backend & backend, const String & key); +/// Free-function entry point — a thin dispatch to `op`'s own typed-evidence classification +/// (`CasOperation::probeSentinel`, which in turn reaches `Backend::probeSentinelRaw`; see there for the +/// per-backend semantics: the S3-native raw HEAD error, the Local container-directory stat, or the +/// generic head/get-based default for a backend without sharper evidence). +/// +/// `policy` is required and has no default, because `Indeterminate` is the one outcome the request +/// contract REISSUES on and the right number of reissues is the caller's question, not this function's. +/// A caller that owns an outer retry loop -- the lifecycle gate, whose inconclusive verdict IS +/// `StayTransient` for the recovery loop to retry -- passes `Retry::once()`, or it pays its own loop's +/// interval inside every probe. A caller with no loop of its own passes `Retry::standard()`. +SentinelProbeResult probeSentinel(CasOperation & op, const String & key, const Retry & policy); /// Verdict of the zero-write startup bootstrap residual check ("Startup ordered vs the capability /// probe"). Before a writable `Pool::open` runs @@ -42,14 +48,16 @@ enum class BootstrapResidual : uint8_t Indeterminate, }; -/// Zero-write authoritative classification of a pool prefix for the startup bootstrap decision. A single -/// paginated LIST of `layout.poolPrefix()`; each listed object is classified as the `_pool_meta` +/// Zero-write authoritative classification of a pool prefix for the startup bootstrap decision. One exact +/// read of `_pool_meta` first: present is decisive, and it costs no enumeration, so an existing pool +/// reopens without a LIST at all. Only when the key is absent (or the read could not settle it) does a +/// single paginated LIST of `layout.poolPrefix()` run; each listed object is classified as the `_pool_meta` /// sentinel, capability-battery debris under the reserved `/_probe/` subtree (a crash-mid-battery /// leftover OR a concurrent fresh opener's in-flight battery — [D2]), or genuine residual CAS state. It /// NEVER writes, and it IGNORES probe debris exactly so a normal restart after a crash-mid-battery still /// bootstraps cleanly. On non-strong-LIST backends this is the single best-effort authoritative check the /// weaker guarantee allows — still fail-closed on any residual object found. Used by `Pool::open` BEFORE /// the capability battery so that no probe write ever precedes the emptiness proof. -BootstrapResidual probePoolBootstrapResidual(Backend & backend, const Layout & layout); +BootstrapResidual probePoolBootstrapResidual(CasOperation & op, const Layout & layout); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h new file mode 100644 index 000000000000..e620195a31c3 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h @@ -0,0 +1,161 @@ +#pragma once +#include + +#include "config.h" + +#if USE_AWS_S3 + +#include + +#include +#include + +namespace DB::ErrorCodes +{ + extern const int BAD_ARGUMENTS; +} + +namespace DB::Cas +{ + +/// A `Backend` decorator that refuses a chosen share of the requests passing through it, so a test +/// can drive the request contract's reissue, resolve and give-up paths against a real backend with +/// no network in the picture. +/// +/// The refusal is an `S3Exception` carrying `SLOW_DOWN` (HTTP 429) or `SERVICE_UNAVAILABLE` (503). +/// Both are RETRYABLE by `S3Exception::isRetryableError` -- its unretryable set holds neither -- and +/// that is the property being modelled: a retryable store refusal leaves the attempt AMBIGUOUS, so +/// the caller must resolve it by reading rather than treat it as a definite failure. +class ThrottlingBackend final : public Backend +{ +public: + /// `FirstPerKey` refuses the first request naming each key and forwards every later one; + /// `EveryNth` refuses every n-th request across all keys. + enum class Mode : uint8_t { FirstPerKey, EveryNth }; + + ThrottlingBackend(BackendPtr inner_, Mode mode_, size_t n_, int status) + : inner(std::move(inner_)) + , mode(mode_) + , every_nth(n_) + , error(status == 429 ? Aws::S3::S3Errors::SLOW_DOWN : Aws::S3::S3Errors::SERVICE_UNAVAILABLE) + { + if (status != 429 && status != 503) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "ThrottlingBackend: refusal status must be 429 or 503, got {}", status); + if (mode == Mode::EveryNth && every_nth == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "ThrottlingBackend: EveryNth needs a period of at least 1"); + } + + /// How many requests naming `key` this backend has refused. A `list` counts under its prefix and + /// a `publish` under its destination key. + size_t refusals(const String & key) const + { + std::lock_guard lock(mutex); + const auto it = refusal_counts.find(key); + return it == refusal_counts.end() ? 0 : it->second; + } + + /// Every key (or list prefix / publish destination) this backend has ever decided a request for, in + /// `FirstPerKey` mode every one of them by construction (`refused_keys` records the key whether or + /// not this particular call refused it), sorted. Lets a coverage test enumerate what it must check + /// without carrying its own duplicate list of keys. + std::vector decidedKeys() const + { + std::lock_guard lock(mutex); + return std::vector(refused_keys.begin(), refused_keys.end()); + } + + std::optional read(const String & key, TransportAccess & access) override + { + refuseOrPass(key); + return inner->read(key, access); + } + + std::optional head(const String & key, TransportAccess & access) override + { + refuseOrPass(key); + return inner->head(key, access); + } + + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + refuseOrPass(prefix); + return inner->list(prefix, cursor, limit, access); + } + + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override + { + refuseOrPass(key); + return inner->remove(key, expected_value, access); + } + + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override + { + for (const WriteOnceKey & key : keys) + refuseOrPass(key.str()); + inner->removeManyWriteOnce(keys, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override + { + refuseOrPass(key); + return inner->write(key, bytes, expected_value, access); + } + + std::unique_ptr stream(const String & key, TransportAccess & access) override + { + refuseOrPass(key); + return inner->stream(key, access); + } + + void publish(const BlobPublishRequest & request, TransportAccess & access) override + { + refuseOrPass(request.destination_key); + inner->publish(request, access); + } + + SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) override + { + refuseOrPass(key); + return inner->probeSentinelRaw(key, access); + } + + /// Facts about the wrapped backend, not requests to refuse. + Dialect dialect() const override { return inner->dialect(); } + bool supportsListTokens() const override { return inner->supportsListTokens(); } + uint64_t attemptTimeoutMs() const override { return inner->attemptTimeoutMs(); } + uint64_t attemptEnvelopeMs() const override { return inner->attemptEnvelopeMs(); } + bool refreshCredentials() override { return inner->refreshCredentials(); } + void checkPoolPreconditions() override { inner->checkPoolPreconditions(); } + void checkSkipAccessCheckSupport() override { inner->checkSkipAccessCheckSupport(); } + void checkConditionalWriteSingleAttemptSupport() override { inner->checkConditionalWriteSingleAttemptSupport(); } + +private: + /// Decide this request and record a refusal, then throw outside the lock. + void refuseOrPass(const String & key) + { + bool refusing = false; + { + std::lock_guard lock(mutex); + refusing = mode == Mode::FirstPerKey ? refused_keys.insert(key).second : (++requests % every_nth) == 0; + if (refusing) + ++refusal_counts[key]; + } + if (refusing) + throw S3Exception(error, "throttled by ThrottlingBackend: {}", key); + } + + const BackendPtr inner; + const Mode mode; + const size_t every_nth; + const Aws::S3::S3Errors error; + + mutable std::mutex mutex; + std::set refused_keys; + std::map refusal_counts; + size_t requests = 0; +}; + +} + +#endif diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasTransportAccess.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasTransportAccess.h new file mode 100644 index 000000000000..49a3bf419874 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasTransportAccess.h @@ -0,0 +1,25 @@ +#pragma once +#include + +namespace DB::Cas +{ + +/// A capability token: holding one proves the holder is `CasRequests`. Not copyable, not +/// constructible outside that one friend, and carries no data besides the engine's own attempt +/// count -- its main job is to gate access at compile time to the backend entry points that must +/// not be called except through the contract. +class TransportAccess +{ + friend class CasRequests; + explicit TransportAccess(size_t attempt_no_) : attempt_no(attempt_no_) {} + size_t attempt_no; + +public: + TransportAccess(const TransportAccess &) = delete; + TransportAccess & operator=(const TransportAccess &) = delete; + /// The engine's 1-based count of physical attempts of the logical call this request belongs to, + /// for the transport to number its request with. Only "1 versus more than 1" is relied upon. + size_t attemptNo() const { return attempt_no; } +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasWriteResult.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasWriteResult.h new file mode 100644 index 000000000000..72679c54b64a --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasWriteResult.h @@ -0,0 +1,130 @@ +#pragma once +#include +#include +#include + +#include + +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int ABORTED; +} + +namespace DB::Cas +{ + +/// Declared (and defined) in `CasRequests.h`/`.cpp`. Forward-declared narrowly here, rather than +/// including that header, because `CasRequests.h` itself includes this one for `WriteResult` -- +/// including it back would be circular. +[[noreturn]] void throwCasWriteRetryLater(const String & why); +[[noreturn]] void throwCasTransientUnavailable(const String & subject, const String & condition); + +struct Object { String bytes; Etag etag; }; +struct Meta { uint64_t size; Etag etag; }; +enum class Removal : uint8_t { Removed, Gone, Mismatch }; + +/// What a write attempt observed of the key's current state before giving up, so a caller (or the +/// message built by `orThrow`) can report exactly what was seen instead of just that something failed. +struct NotObserved {}; +struct ProvenAbsent {}; +using Observation = std::variant; + +/// A durable write landed: `etag` names the incarnation it created (or, for a retried write +/// resolved by a read, the incarnation already present), `attempts_sent` counts the HTTP attempts this +/// call made, and `resolved_by_read` is true when the commit was proven by a read rather than by the +/// attempt's own response. +struct Committed { Etag etag; uint32_t attempts_sent; bool resolved_by_read; }; +/// The write was never attempted or never needed -- e.g. `putIfAbsent` finding the key already +/// present under the caller's intended content. `seen` is whatever the resolve read observed. +struct Declined { Observation seen; }; +/// A competing write won: the key's current state does not match what this call expected. +/// `attempts_sent` counts the HTTP attempts this call made, the same count `Committed` and `GaveUp` +/// carry: an operator's attempt counters sum over ALL the endings of a write, and losing the key is +/// one an operator wants counted rather than dropped. `any_ambiguous` is true when an attempt of the +/// inner write ended without proof of whether it applied before the resolve read settled the race: a +/// caller outside the engine paces such a conflict on the growing schedule, as the engine's own +/// loops do, and a clean lost race on the flat one. `attempts_sent` cannot stand in for it: a +/// throttled first attempt whose resolve read finds the key moved is one attempt, ambiguous. +struct Conflict { Observation seen; uint32_t attempts_sent = 0; bool any_ambiguous = false; }; +/// The store itself refused the request (not a lost precondition) -- `store_error` is a ClickHouse +/// error code and `message` explains it. `attempts_sent` is the same count `Conflict` carries, for +/// the same reason. +struct Refused { int store_error; String message; uint32_t attempts_sent = 0; }; +/// No attempt landed and none can be proven safe to keep making. +struct GaveUp +{ + enum class Why : uint8_t { Deadline, FenceLost, Unresolved }; + enum class Source : uint8_t { Policy, Lease }; + Why why; Source deadline_source; bool sent_any; Observation last_seen; + /// The HTTP attempts this call made, the same count `Committed` carries. Operator counters -- the + /// mount renewal's attempt and retry counters among them -- have to count the attempts of a write + /// that GAVE UP as well as of one that committed, and `sent_any` cannot say how many. It stays + /// beside this because the readers that only branch on "was anything sent" branch on it by name. + uint32_t attempts_sent = 0; +}; +using WriteResult = std::variant; + +namespace detail +{ + +/// Helper for std::visit with multiple lambdas; no shared one exists in the tree yet. +template +struct Overload : Ts... +{ + using Ts::operator()...; +}; +template +Overload(Ts...) -> Overload; + +inline String renderObservation(const Observation & seen) +{ + return std::visit(Overload{ + [](const NotObserved &) -> String { return "nothing observed"; }, + [](const ProvenAbsent &) -> String { return "absent"; }, + [](const Meta &) -> String { return "present (meta)"; }, + [](const Object & o) -> String { return "present (" + o.etag.render() + ")"; }}, seen); +} + +} + +/// Collapse a `WriteResult` into the incarnation a caller can act on: `nullopt` for a declined write +/// (nothing changed, nothing to report), the committed incarnation otherwise -- or throw, mapping +/// every non-success alternative to the error class its meaning already implies. `what` names the +/// call for the thrown message. +inline std::optional orThrow(WriteResult && result, std::string_view what) +{ + using detail::Overload; + using detail::renderObservation; + return std::visit(Overload{ + [](Committed & c) -> std::optional { return std::move(c.etag); }, + [](Declined &) -> std::optional { return std::nullopt; }, + [&](Conflict & c) -> std::optional + { + throw Exception(ErrorCodes::ABORTED, "{}: conflict, observed {}", what, renderObservation(c.seen)); + }, + [&](Refused & r) -> std::optional + { + throw Exception(r.store_error, "{}: the store refused the write: {}", what, r.message); + }, + [&](GaveUp & g) -> std::optional + { + switch (g.why) + { + case GaveUp::Why::FenceLost: + throwCasTransientUnavailable(String(what), "mount fence tripped: the durable write is refused because this node no longer holds the mount incarnation it was admitted under"); + case GaveUp::Why::Deadline: + throwCasWriteRetryLater(fmt::format("{}: gave up at the {} deadline after {} attempt(s)", what, g.deadline_source == GaveUp::Source::Lease ? "lease" : "policy", g.sent_any ? "one or more" : "zero")); + case GaveUp::Why::Unresolved: + throwCasWriteRetryLater(fmt::format("{}: the write is unresolved (sent, resolve read found {})", what, renderObservation(g.last_seen))); + } + UNREACHABLE(); + }}, result); +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp index c9cf2b166389..78666e0ad297 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp @@ -1,11 +1,12 @@ #include +#include "config.h" #include #include #include #include #include #include -#include +#include #include #include #include @@ -29,7 +30,6 @@ #include #include #include -#include #include #include #include @@ -77,11 +77,18 @@ namespace ContentAddressedSetting extern const ContentAddressedSettingsUInt64 gc_round_prefix_wholesale_budget; extern const ContentAddressedSettingsUInt64 gc_round_handoff_prefix_wholesale_budget; extern const ContentAddressedSettingsUInt64 gc_round_outcome_entry_budget; + extern const ContentAddressedSettingsUInt64 mount_lease_ttl_ms; + extern const ContentAddressedSettingsUInt64 mount_renew_period_ms; + extern const ContentAddressedSettingsBool unsafe_remount_no_delay; extern const ContentAddressedSettingsUInt64 part_folder_cache_bytes; extern const ContentAddressedSettingsUInt64 part_folder_cache_max_entries; extern const ContentAddressedSettingsUInt64 part_folder_cache_max_entry_bytes; extern const ContentAddressedSettingsUInt64 manifest_decode_cache_bytes; extern const ContentAddressedSettingsUInt64 gc_meta_pool_size; + extern const ContentAddressedSettingsUInt64 gc_read_concurrency; + extern const ContentAddressedSettingsUInt64 gc_bulk_delete_chunk_keys; + extern const ContentAddressedSettingsUInt64 attempt_timeout_ms; + extern const ContentAddressedSettingsUInt64 lease_safety_margin_ms; extern const ContentAddressedSettingsBool blob_hash_allow_new; } @@ -137,7 +144,7 @@ const char * casLifecycleReasonWord(Cas::PoolLifecycle lc) /// serverRootId, scratchPath, stagingBackend, objectStorage, gcHealth, /// lifecycleSnapshot (both non-store()-gated introspection reads for system.cas_mounts -- /// readable in EVERY lifecycle state including a not-live/vanished/null pool, spec §7), -/// parseStagingBackend/parsePartFolderValidate/ +/// parseStagingBackend/ /// tryFromDisk (static), checkNotReadOnly, the *ForTest seams, serverPrefix/liveNamespace/ /// shadowNamespace/shadowScope/route/classifyDirectory (pure path computation, no pool I/O), /// ownsNamespace (the relink-confirm routing predicate -- a string comparison against @@ -294,16 +301,22 @@ ContentAddressedMetadataStorage::ContentAddressedMetadataStorage( , gc_round_prefix_wholesale_budget(settings_[ContentAddressedSetting::gc_round_prefix_wholesale_budget].value) , gc_round_handoff_prefix_wholesale_budget(settings_[ContentAddressedSetting::gc_round_handoff_prefix_wholesale_budget].value) , gc_round_outcome_entry_budget(settings_[ContentAddressedSetting::gc_round_outcome_entry_budget].value) + , mount_lease_ttl(std::chrono::milliseconds(settings_[ContentAddressedSetting::mount_lease_ttl_ms].value)) + , mount_renew_period(std::chrono::milliseconds(settings_[ContentAddressedSetting::mount_renew_period_ms].value)) + , cas_unsafe_remount_no_delay(settings_[ContentAddressedSetting::unsafe_remount_no_delay].value) , cas_part_folder_cache_bytes(settings_[ContentAddressedSetting::part_folder_cache_bytes].value) , cas_part_folder_cache_max_entries(settings_[ContentAddressedSetting::part_folder_cache_max_entries].value) , cas_part_folder_cache_max_entry_bytes(settings_[ContentAddressedSetting::part_folder_cache_max_entry_bytes].value) , manifest_decode_cache_bytes(settings_[ContentAddressedSetting::manifest_decode_cache_bytes].value) , gc_meta_pool_size(settings_[ContentAddressedSetting::gc_meta_pool_size].value) + , gc_read_concurrency(settings_[ContentAddressedSetting::gc_read_concurrency].value) + , gc_bulk_delete_chunk_keys(settings_[ContentAddressedSetting::gc_bulk_delete_chunk_keys].value) + , cas_attempt_timeout_ms(settings_[ContentAddressedSetting::attempt_timeout_ms].value) + , cas_lease_safety_margin_ms(settings_[ContentAddressedSetting::lease_safety_margin_ms].value) , staging_backend(settings_.stagingBackend()) , blob_hash_algo(settings_.blobHashAlgo()) , blob_hash_allow_new(settings_[ContentAddressedSetting::blob_hash_allow_new].value) , skip_access_check(settings_.skipAccessCheck()) - , part_folder_validate(settings_.partFolderValidate()) { } @@ -328,35 +341,6 @@ Cas::StagingBackend ContentAddressedMetadataStorage::parseStagingBackend( return parseStagingBackend(config.getString(config_prefix + ".staging_backend", "local")); } -Cas::PartFolderValidate ContentAddressedMetadataStorage::parsePartFolderValidate(const std::string & value) -{ - using PartFolderValidate = Cas::PartFolderValidate; - if (value == "always") - return {PartFolderValidate::Mode::Always, 0}; - if (value == "never") - return {PartFolderValidate::Mode::Never, 0}; - if (value.starts_with("age ")) - { - /// `std::from_chars` against an UNSIGNED type never accepts a leading '-' (unlike - /// `std::stoull`, which silently negates modulo 2^64) -- a malformed/negative/non-digit/empty - /// suffix falls through to the terminal throw below instead of wrapping into an astronomical - /// age_seconds that behaves as skip-forever. - const std::string age_str = value.substr(4); - uint64_t age_seconds = 0; - const auto [ptr, ec] = std::from_chars(age_str.data(), age_str.data() + age_str.size(), age_seconds); - if (ec == std::errc{} && ptr == age_str.data() + age_str.size()) - return {PartFolderValidate::Mode::Age, age_seconds}; - } - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Unknown cas_part_folder_validate value '{}' (expected 'always', 'never', or 'age ')", value); -} - -Cas::PartFolderValidate ContentAddressedMetadataStorage::parsePartFolderValidate( - const Poco::Util::AbstractConfiguration & config, const std::string & config_prefix) -{ - return parsePartFolderValidate(config.getString(config_prefix + ".part_folder_validate", "always")); -} - ContentAddressedMetadataStorage * ContentAddressedMetadataStorage::tryFromDisk(const DiskPtr & disk) { /// The cheap predicate FIRST, never an exception probe: for every non-object-storage disk @@ -490,15 +474,19 @@ CasLifecycleSnapshot ContentAddressedMetadataStorage::lifecycleSnapshot() const Cas::GcRoundLogger ContentAddressedMetadataStorage::makeGcRoundLogger() const { - /// Unit tests pass a null context (no system logs); the scheduler then runs without a sink. + /// Unit tests pass a null context (no system logs); the scheduler then runs with the test hook + /// as its only sink, or without a sink. + const std::function hook = gc_round_row_hook_for_test; if (!context) - return {}; + return hook ? Cas::GcRoundLogger(hook) : Cas::GcRoundLogger{}; const ContextWeakPtr weak_context = *context; /// The configured disk name (threaded from the metadata-storage factory); falls back to /// storage_path_prefix for callers that don't supply one (e.g. unit tests). const String disk = disk_name; - return [weak_context, disk](const Cas::GcRoundLogRecord & r) + return [weak_context, disk, hook](const Cas::GcRoundLogRecord & r) { + if (hook) + hook(r); auto ctx = weak_context.lock(); if (!ctx) { @@ -547,6 +535,12 @@ Cas::GcRoundLogger ContentAddressedMetadataStorage::makeGcRoundLogger() const case Cas::GcRoundLogRecord::Outcome::Deferred: e.outcome = ContentAddressedGarbageCollectionLogElement::DEFERRED; break; + case Cas::GcRoundLogRecord::Outcome::Aborted: + e.outcome = ContentAddressedGarbageCollectionLogElement::ABORTED; + break; + case Cas::GcRoundLogRecord::Outcome::Stopped: + e.outcome = ContentAddressedGarbageCollectionLogElement::STOPPED; + break; } e.round = r.round; e.candidates_marked = r.candidates_marked; @@ -562,6 +556,7 @@ Cas::GcRoundLogger ContentAddressedMetadataStorage::makeGcRoundLogger() const e.anomalies = r.anomalies; e.duration_ms = r.duration_ms; e.error = r.error; + e.error_code = r.error_code; e.profile_events = r.profile_events; e.round_id = r.round_id; e.phase = r.phase; @@ -701,6 +696,23 @@ Cas::RebuildReport ContentAddressedMetadataStorage::runGcRebuildNow(bool force) return gc.rebuildBaseline(force); } +std::optional ContentAddressedMetadataStorage::freezeConnectTimeoutCapMs( + [[maybe_unused]] const ObjectStoragePtr & object_storage, [[maybe_unused]] uint64_t cas_attempt_timeout_ms) +{ +#if USE_AWS_S3 + const auto s3_client = object_storage->tryGetS3StorageClient(); + if (!s3_client) + return std::nullopt; + /// A configured zero is "unbounded" to Poco: the cap is then the attempt timeout itself. + const auto configured = static_cast(std::max(0, s3_client->getClientConfiguration().connectTimeoutMs)); + return configured == 0 ? cas_attempt_timeout_ms : std::min(configured, cas_attempt_timeout_ms); +#else + /// Without the AWS S3 client compiled in there is no S3 storage to read a connect timeout from, + /// so nothing is frozen: the same answer an S3-less object storage gets above. + return std::nullopt; +#endif +} + ContentAddressedMetadataStorage::PoolView ContentAddressedMetadataStorage::openPoolView(bool context_available) const { /// Native mode rides real conditional ops (probed fail-closed by Pool::open); Local object @@ -708,8 +720,6 @@ ContentAddressedMetadataStorage::PoolView ContentAddressedMetadataStorage::openP const auto mode = object_storage->getType() == ObjectStorageType::Local ? Cas::ObjectStorageBackend::Mode::EmulatedSingleProcess : Cas::ObjectStorageBackend::Mode::Native; - auto backend = std::make_shared(object_storage, mode); - const Cas::TokenType backend_token_type = backend->nativeTokenType(); /// EmulatedSingleProcess emulates the conditional-op / exact-token semantics in-process (local /// object storage has none). That emulation is per-process: two servers pointed at the SAME local @@ -790,12 +800,38 @@ ContentAddressedMetadataStorage::PoolView ContentAddressedMetadataStorage::openP pool_config.gc_round_handoff_prefix_wholesale_budget = gc_round_handoff_prefix_wholesale_budget; pool_config.gc_round_outcome_entry_budget = gc_round_outcome_entry_budget; pool_config.gc_meta_pool_size = gc_meta_pool_size; + pool_config.gc_read_concurrency = gc_read_concurrency; + pool_config.gc_bulk_delete_chunk_keys = gc_bulk_delete_chunk_keys; + pool_config.cas_request_budget.attempt_timeout_ms = cas_attempt_timeout_ms; + pool_config.cas_request_budget.lease_safety_margin_ms = cas_lease_safety_margin_ms; + /// Frozen here, once: see `freezeConnectTimeoutCapMs`. A later reload that widens the disk's + /// connect timeout cannot widen the envelope the lease arithmetic was validated against. + pool_config.cas_request_budget.connect_timeout_cap_ms = freezeConnectTimeoutCapMs(object_storage, cas_attempt_timeout_ms); + pool_config.mount_lease_ttl_ms = mount_lease_ttl; + pool_config.mount_renew_period = mount_renew_period; + pool_config.unsafe_remount_no_delay = cas_unsafe_remount_no_delay; pool_config.event_sink = makeCasEventSink(); + /// Built here rather than above so it carries the budget the pool was configured with. Only a + /// WRITABLE Native mount takes the single-attempt profile for its control-plane requests: it owns + /// its own deadline and retry policy, and a transparently retried request would outlive them. A + /// read-only mount has no such deadline, so it keeps the storage's default. + auto backend = std::make_shared( + object_storage, mode, + /*single_attempt_control_plane_=*/!read_only && mode == Cas::ObjectStorageBackend::Mode::Native, + pool_config.cas_request_budget.attempt_timeout_ms, + pool_config.cas_request_budget.connect_timeout_cap_ms.value_or(0)); + + /// A programmer error, not an input one: the handoff above is code, not configuration. The pool's + /// lease arithmetic was validated against `pool_config.cas_request_budget` alone (never against the + /// backend), so a backend built with a different cap or timeout would silently outlive it. Fail + /// closed before `Pool::open` rather than trust the two stayed in sync. + Cas::ensureBackendMatchesBudget(*backend, pool_config.cas_request_budget); + PoolView view; view.physical_key_prefix = physical_key_prefix_local; view.pool_prefix = pool_prefix; - view.native_token_type = backend_token_type; + view.native_token_type = backend->nativeTokenType(); view.pool = Cas::Pool::open(std::move(backend), std::move(pool_config)); return view; } @@ -845,8 +881,7 @@ void ContentAddressedMetadataStorage::startup() Cas::CachedPartFolderAccess::CacheParams{ .cache_bytes = cas_part_folder_cache_bytes, .max_entries = cas_part_folder_cache_max_entries, - .max_entry_bytes = cas_part_folder_cache_max_entry_bytes, - .validate = part_folder_validate}); + .max_entry_bytes = cas_part_folder_cache_max_entry_bytes}); /// Reclaim this mount's leaked `staging//` debris after an explicit, writable S3 /// staging mount has passed the native-copy check above. The prefix is keyed by this mount's own @@ -905,13 +940,22 @@ void ContentAddressedMetadataStorage::startup() /// that swapped the client under that state would leave persisted tokens uncomparable. The refusal /// has to happen in the object storage: only there is the effective `http_client` known, merged from /// the storage's current settings, any endpoint-level block and the disk's own section. - object_storage->pinConditionalOpsGenerationDialect(native_token_type == Cas::TokenType::Generation); + object_storage->pinConditionalOpsGenerationDialect(native_token_type == Cas::Dialect::Generation); } void ContentAddressedMetadataStorage::shutdown() { - /// Wait for any in-flight synchronous round to finish cleanly first (gc_scheduler_mutex is held - /// for a round's whole duration) -- unchanged priority: clean GC completion over fast shutdown. + /// Arm the pool BEFORE waiting for `gc_scheduler_mutex`: a synchronous round holds that mutex + /// for its whole duration and releases it only once its next request is refused. The arm frees, + /// nulls and swaps nothing -- every pointer swap still happens below, under the same locks as + /// before -- so taking `pointer_mutex` alone here, before the outer lock, inverts no order. + Cas::PoolPtr pool; + { + std::lock_guard ptr_lock(pointer_mutex); + pool = cas_store; + } + if (pool) + pool->beginTeardown(); std::lock_guard round_lock(gc_scheduler_mutex); shutdown_called = true; stopAndDrainForTeardown(); @@ -959,6 +1003,10 @@ void ContentAddressedMetadataStorage::stopAndDrainForTeardown() noexcept } }; + /// The destructor reaches here without `shutdown`'s arm: arm now, before the join, so a + /// background round is refused at its next request rather than joined at its end. + if (old_pool) + old_pool->beginTeardown(); guarded([&] { if (old_scheduler) old_scheduler->stop(); }, "CAS storage teardown: stopping GC"); guarded([&] { old_part_access.reset(); }, "CAS storage teardown: releasing part access"); guarded([&] @@ -966,7 +1014,7 @@ void ContentAddressedMetadataStorage::stopAndDrainForTeardown() noexcept if (!old_pool) return; const auto & budget = old_pool->poolConfig().cas_request_budget; - const uint64_t deadline_ms = budget.attempt_timeout_ms + budget.lease_safety_margin_ms; + const uint64_t deadline_ms = budget.attemptEnvelopeMs() + budget.lease_safety_margin_ms; if (old_pool->stopAndDrainDetachedWork(deadline_ms)) return; ProfileEvents::increment(ProfileEvents::CASDetachedWorkDrainTimeouts); @@ -1256,9 +1304,15 @@ void ContentAddressedMetadataStorage::confirmPoolIdentityForEmptyEnumeration(con const Cas::PoolPtr pool = store(); /// Live here (past the op gate's Live, non-terminal admission). ++empty_proof_probe_count_for_test; + /// The open plane: this probe is what authorizes an empty answer, and a mount whose lease has + /// blipped must still be able to ask it. + Cas::CasOperation probe_op = pool->openRequests().admit(); const Cas::SentinelProbeResult probe = empty_proof_probe_override_for_test ? empty_proof_probe_override_for_test() - : Cas::probeSentinel(pool->backend(), pool->layout().poolMetaKey()); + /// `once`: this probe is the gate that authorises an empty answer, and an inconclusive one is a + /// refusal the caller retries -- reissuing here would stall a directory listing for the whole + /// retry window instead. + : Cas::probeSentinel(probe_op, pool->layout().poolMetaKey(), Cas::Retry::once()); switch (probe.outcome) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h index 043ce8f9280b..bffb608688cc 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h @@ -167,15 +167,6 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// thin wrapper around the string-taking overload for callers that still hold a config reference. static Cas::StagingBackend parseStagingBackend(const Poco::Util::AbstractConfiguration & config, const std::string & config_prefix); - /// Parses a `part_folder_validate` value (`always` | `never` | `age `). The `age` form - /// accepts only a non-negative integer number of seconds; malformed input and unknown modes throw - /// `BAD_ARGUMENTS` instead of silently selecting a policy. - static Cas::PartFolderValidate parsePartFolderValidate(const std::string & value); - - /// Reads `part_folder_validate` from `config`, defaulting to `always`, and parses it. Kept only as - /// a thin wrapper around the string-taking overload for callers that still hold a config reference. - static Cas::PartFolderValidate parsePartFolderValidate(const Poco::Util::AbstractConfiguration & config, const std::string & config_prefix); - /// Returns the content-addressed metadata storage backing `disk`, or nullptr if `disk` is not /// content-addressed. Plain (non-object-storage) disks do not implement `getMetadataStorage` at /// all and throw `NOT_IMPLEMENTED`; that is treated as "not content-addressed" rather than @@ -192,6 +183,17 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// observe the detached-work stop latch without going through the lifecycle gate. Cas::PoolPtr poolForTest() const; + /// What `openPoolView` freezes into `pool_config.cas_request_budget.connect_timeout_cap_ms`: the + /// connect cap every control request and every single-attempt clone will carry for the life of + /// this mount. `nullopt` when `object_storage` has no S3 client (the envelope is then the attempt + /// alone); otherwise `min(disk connect_timeout_ms, cas_attempt_timeout_ms)`, with a configured zero + /// (Poco's "unbounded") normalized to `cas_attempt_timeout_ms` rather than treated as no limit. + /// Exposed here (not test-only) because it is a pure read of already-public state -- freezing it in + /// one place, unit-testable without opening a pool, is what keeps a later client reload from + /// widening the envelope the lease arithmetic was validated against. + static std::optional freezeConnectTimeoutCapMs( + const ObjectStoragePtr & object_storage, uint64_t cas_attempt_timeout_ms); + /// Runs one synchronous GC round on the caller's thread and emits Start and Finish rows to /// `system.cas_gc_log`. Throws `BAD_ARGUMENTS` when GC is disabled /// by read-only mode or configuration. @@ -558,6 +560,15 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// default; production installs none. void setGcVerbAdmitWindowHookForTest(std::function fn) { gc_verb_admit_window_hook_for_test = std::move(fn); } + /// Test-only: sees every round-log row the storage's own scheduler emits (Start, Phase, Finish), + /// before and independently of the system-log path -- which a unit-test storage (null `Context`) + /// does not have at all. Lets a test park a synchronous round on a phase row while it holds + /// `gc_scheduler_mutex`, and read the round's outcome afterwards. Set before the first round. + void setGcRoundRowHookForTest(std::function fn) + { + gc_round_row_hook_for_test = std::move(fn); + } + /// Test-only fault-injection/hook seam for `ContentAddressedTransaction::publishStaging`'s /// promote/repoint call, keyed by the full `(ns, ref)` routed identity via `PartRefKey::cacheKey()` /// (mirrors `CasRefLedger::setRefPreCarveHookForTest`'s no-op-in-production shape) -- a bare ref @@ -611,6 +622,11 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC const uint64_t gc_round_prefix_wholesale_budget; const uint64_t gc_round_handoff_prefix_wholesale_budget; const uint64_t gc_round_outcome_entry_budget; + const std::chrono::milliseconds mount_lease_ttl; + const std::chrono::milliseconds mount_renew_period; + /// See `PoolConfig::unsafe_remount_no_delay` -- the operator's explicit acceptance of an + /// unobserved same-uuid reclaim. + const bool cas_unsafe_remount_no_delay; /// Part-folder view cache settings. `cas_part_folder_cache_bytes == 0` disables retention. const uint64_t cas_part_folder_cache_bytes; const uint64_t cas_part_folder_cache_max_entries; @@ -619,6 +635,17 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC const uint64_t manifest_decode_cache_bytes; /// Bounded pool size for GC's per-hash freshness-metadata writes. const uint64_t gc_meta_pool_size; + /// Bounded pool size for the GC fold's read-ahead; 1 disables it. + const uint64_t gc_read_concurrency; + /// Keys per batch delete request for the write-once families. + const uint64_t gc_bulk_delete_chunk_keys; + /// The budget for one HTTP attempt of a writable Native mount's control-plane requests; feeds + /// `Cas::PoolConfig::cas_request_budget.attempt_timeout_ms` and the backend's own + /// `attemptTimeoutMs()`. + const uint64_t cas_attempt_timeout_ms; + /// Startup-only margin validated against the mount lease TTL; feeds + /// `Cas::PoolConfig::cas_request_budget.lease_safety_margin_ms`. + const uint64_t cas_lease_safety_margin_ms; /// Configured staging backend; `Local` preserves the existing write path. const Cas::StagingBackend staging_backend; /// Blob content-hash function passed to `Cas::PoolConfig`. @@ -627,8 +654,6 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC const bool blob_hash_allow_new; /// Per-disk `` policy passed to `Cas::PoolConfig`. const bool skip_access_check; - /// Policy controlling when retained part-folder views revalidate their manifest body. - const Cas::PartFolderValidate part_folder_validate; /// A single coherent snapshot of the pool and its cached part-folder facade, taken under ONE /// `pointer_mutex` acquisition (see `poolAccess()`) so no caller can observe `pool` from one mount /// generation and `part_access` from another -- the two used to be fetched by two separate calls @@ -653,7 +678,7 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// native_token_type`) -- immutable afterwards. `startup` also hands it to the object storage as a /// pin, which is what refuses a reload that would flip the dialect under this live pool; the check /// belongs there because only the object storage knows the effective `http_client`. - Cas::TokenType native_token_type = Cas::TokenType::ETag; + Cas::Dialect native_token_type = Cas::Dialect::ETag; /// shared_ptr so `runGarbageCollectionRoundNow`/`runOneGcRoundForTest` can take a snapshot under /// `pointer_mutex`, release it, and run the (long) round via the snapshot -- never holding /// `pointer_mutex` itself for the round's duration, so `gcHealth`/`store`/`partAccess` never @@ -663,9 +688,9 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// `gcStop`/`gcStart`. Serializes them against each other. Lock order when nested locks are needed: /// `lifecycle_mutex` -> `gc_scheduler_mutex` -> `pointer_mutex`, never the reverse. mutable std::mutex lifecycle_mutex; - /// Serializes ONE synchronous GC round at a time and makes `shutdown` wait for an in-flight round - /// to finish cleanly (clean GC completion has priority over fast shutdown) -- held for the WHOLE - /// round. Deliberately NOT the same mutex as `pointer_mutex` below: this one can be held for a + /// Serializes ONE synchronous GC round at a time. `shutdown` still takes it, but arms the pool + /// first, so a round in flight is refused at its next request once the pool is armed and the wait + /// is one request long -- held for the WHOLE round. Deliberately NOT the same mutex as `pointer_mutex` below: this one can be held for a /// long time, so nothing that only needs a brief pointer snapshot may share it. mutable std::mutex gc_scheduler_mutex; bool shutdown_called TSA_GUARDED_BY(gc_scheduler_mutex) = false; @@ -750,7 +775,7 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// captured while the concrete backend is still in scope. Reading it back through `pool` /// would mean unwrapping the instrumentation decorator `Pool::open` adds, so it is returned /// here instead. - Cas::TokenType native_token_type = Cas::TokenType::ETag; + Cas::Dialect native_token_type = Cas::Dialect::ETag; }; /// Builds the backend + `Cas::PoolConfig` and opens a pool exactly as `startup()` does. A read-only @@ -793,6 +818,7 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// TOCTOU tests). Empty in production; a `const` GC verb reads it and calls the const-qualified /// `std::function::operator()`, so it needs no `mutable`. std::function gc_verb_admit_window_hook_for_test; + std::function gc_round_row_hook_for_test; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp index bf6ab3b90e33..b2a3651b2472 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp @@ -61,7 +61,7 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, gc_snapshot_generations_to_keep, 3, "GC snapshot generations retained", 0) \ DECLARE(UInt64, gc_shards, 1, "Blob-hash-prefix reducer shards (>= 1); creation-time only", 0) \ DECLARE(UInt64, manifest_sweep_list_budget_keys, 1000, "Orphan-manifest sweep LIST budget per round", 0) \ - DECLARE(UInt64, manifest_sweep_delete_budget_keys, 100, "Orphan-manifest sweep DELETE budget per round", 0) \ + DECLARE(UInt64, manifest_sweep_delete_budget_keys, 100, "Orphan-manifest sweep candidate budget per round: keys that pass every retain check, whose bodies are read and decided", 0) \ DECLARE(UInt64, gc_round_graduation_budget, 5000, "Blob graduation (condemned -> delete_pending) cohort cap per round (0 = unbounded)", 0) \ DECLARE(UInt64, gc_round_redelete_budget, 5000, "Blob redelete (exact-token delete of a prior delete_pending row) cohort cap per round (0 = unbounded)", 0) \ DECLARE(UInt64, gc_round_sweep_namespace_budget, 20, "Orphan-manifest sweep: distinct namespaces per page whose protection view may be built (0 = unbounded)", 0) \ @@ -70,13 +70,19 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, gc_round_prefix_wholesale_budget, 20000, "Generation-prefix wholesale delete (prune only) object cap per round (0 = unbounded)", 0) \ DECLARE(UInt64, gc_round_handoff_prefix_wholesale_budget, 5000, "Post-CAS hand-off generation-prefix reclaim object cap per round, reserved separately from gc_round_prefix_wholesale_budget so a prune-heavy round cannot starve the one-shot hand-off (0 = unbounded)", 0) \ DECLARE(UInt64, gc_round_outcome_entry_budget, 5000, "GcOutcomes per-round entry cap across the redelete/spared audit log (0 = unbounded)", 0) \ + DECLARE(UInt64, mount_lease_ttl_ms, 30000, "Mount lease validity after a successful claim or renewal, in milliseconds", 0) \ + DECLARE(UInt64, mount_renew_period_ms, 10000, "Interval between background mount lease renewals, in milliseconds", 0) \ + DECLARE(Bool, unsafe_remount_no_delay, false, "Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same server_uuid (a copied uuid file, a stalled predecessor): after such a reclaim the predecessor can still start conditional writes until its own cutoff (confirmed deadline − margin − 2 × envelope) or until its next renewal meets the token guard, and a request it already sent may materialize later. Ref-log keys carry (writer_epoch, sequence) and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles stragglers -- the exposure is availability, not data: recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions. Intended for test stands and deployments that guarantee one process per uuid", 0) \ DECLARE(String, server_root_id, "", "REQUIRED explicit layout subtree identity; macros expand as in the s3 endpoint", 0) \ DECLARE(UInt64, part_folder_cache_bytes, 64ULL << 20, "Part-folder view cache byte budget (0 disables retention)", 0) \ DECLARE(UInt64, part_folder_cache_max_entries, 10000, "Part-folder view cache entry cap", 0) \ DECLARE(UInt64, part_folder_cache_max_entry_bytes, 16ULL << 20, "Oversized part-folder views bypass retention above this size", 0) \ - DECLARE(String, part_folder_validate, "always", "ForceFresh body re-proof policy (always | never | age )", 0) \ DECLARE(UInt64, manifest_decode_cache_bytes, 128ULL << 20, "Manifest DECODE cache byte budget (0 disables)", 0) \ DECLARE(UInt64, gc_meta_pool_size, 16, "Bounded pool size for GC per-hash freshness-meta writes", 0) \ + DECLARE(UInt64, gc_read_concurrency, 16, "Bounded pool size for the GC fold's read-ahead of checkpoints, ref logs, manifest bodies and zero-candidate HEADs; 1 disables read-ahead", 0) \ + DECLARE(UInt64, gc_bulk_delete_chunk_keys, 1000, "Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots); 1 to 1000", 0) \ + DECLARE(UInt64, attempt_timeout_ms, 5000, "Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. With the connect cap it forms the attempt envelope the lease arithmetic reserves", 0) \ + DECLARE(UInt64, lease_safety_margin_ms, 2000, "Startup-only margin validated against the mount lease TTL: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ DECLARE(String, staging_backend, "local", "Blob staging backend (local | s3); s3 is opt-in", 0) \ DECLARE_SETTINGS_TRAITS(ContentAddressedSettingsTraits, LIST_OF_CONTENT_ADDRESSED_SETTINGS, CONTENT_ADDRESSED_SETTINGS_SUPPORTED_TYPES) @@ -85,11 +91,10 @@ struct ContentAddressedSettingsImpl : public BaseSettings= 1 (got {}, {})", - settings[ContentAddressedSetting::gc_interval_sec].value, settings[ContentAddressedSetting::gc_shards].value); + "content_addressed disk: cas_gc_interval_sec, cas_gc_shards and cas_gc_read_concurrency must be >= 1 " + "(got {}, {}, {})", + settings[ContentAddressedSetting::gc_interval_sec].value, settings[ContentAddressedSetting::gc_shards].value, + settings[ContentAddressedSetting::gc_read_concurrency].value); + + if (settings[ContentAddressedSetting::gc_bulk_delete_chunk_keys] == 0 + || settings[ContentAddressedSetting::gc_bulk_delete_chunk_keys] > 1000) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: gc_bulk_delete_chunk_keys must be between 1 and 1000 (got {})", + settings[ContentAddressedSetting::gc_bulk_delete_chunk_keys].value); + + if (settings[ContentAddressedSetting::attempt_timeout_ms] == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: cas_attempt_timeout_ms must be >= 1 (got {})", + settings[ContentAddressedSetting::attempt_timeout_ms].value); + + if (settings[ContentAddressedSetting::mount_lease_ttl_ms] == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: cas_mount_lease_ttl_ms must be >= 1 (got {})", + settings[ContentAddressedSetting::mount_lease_ttl_ms].value); + + if (settings[ContentAddressedSetting::mount_renew_period_ms] == 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: cas_mount_renew_period_ms must be >= 1 (got {})", + settings[ContentAddressedSetting::mount_renew_period_ms].value); /// The layout subtree identity is explicit and REQUIRED — no default, so an ABSENT key throws a /// typed `NO_ELEMENTS_IN_CONFIG` (mirroring the `metadata_type` check in `MetadataStorageFactory`), @@ -241,7 +270,6 @@ void ContentAddressedSettings::validate() impl->blob_hash_algo_cached = Cas::parseBlobHashAlgo(settings[ContentAddressedSetting::blob_hash].value); impl->staging_backend_cached = ContentAddressedMetadataStorage::parseStagingBackend(settings[ContentAddressedSetting::staging_backend].value); - impl->part_folder_validate_cached = ContentAddressedMetadataStorage::parsePartFolderValidate(settings[ContentAddressedSetting::part_folder_validate].value); } Cas::BlobHashAlgo ContentAddressedSettings::blobHashAlgo() const @@ -259,9 +287,4 @@ Cas::StagingBackend ContentAddressedSettings::stagingBackend() const return impl->staging_backend_cached; } -Cas::PartFolderValidate ContentAddressedSettings::partFolderValidate() const -{ - return impl->part_folder_validate_cached; -} - } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.h index f9619ee93e75..47a9943421c6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.h @@ -11,15 +11,11 @@ namespace Poco { namespace Util { class AbstractConfiguration; } } // NOLINT(cpp namespace DB::Cas { -/// Forward declared to keep this header light: the full definitions live in -/// `ContentAddressedMetadataStorage.h` (`StagingBackend`) and `Parts/PartFolderAccess.h` -/// (`PartFolderValidate`), which are heavy and — in `ContentAddressedMetadataStorage.h`'s -/// case — will itself include this header once the metadata storage is rewired onto it. -/// Both are legal opaque declarations: `StagingBackend` fixes no explicit underlying type -/// (matching its definition, which leaves it as the implicit `int`), and `PartFolderValidate` -/// is only ever used here as an incomplete-type function return, never stored by value. +/// Forward declared to keep this header light: the full definition lives in +/// `ContentAddressedMetadataStorage.h`, which is heavy and will itself include this header once the +/// metadata storage is rewired onto it. A legal opaque declaration: `StagingBackend` fixes no +/// explicit underlying type, matching its definition. enum class StagingBackend; -struct PartFolderValidate; } namespace DB @@ -72,8 +68,8 @@ struct ContentAddressedSettings /// Fail-closed checks: `gc_interval_sec` and `gc_shards` must both be >= 1; `server_root_id` must /// be present (an ABSENT key throws a typed `NO_ELEMENTS_IN_CONFIG`, distinct from a /// PRESENT-but-invalid value, which throws `Cas::validateServerRootId`'s `BAD_ARGUMENTS`); and the - /// three enum-valued string settings (`blob_hash`, `staging_backend`, `part_folder_validate`) must - /// parse. The parsed enum values are cached for the typed accessors below. + /// two enum-valued string settings (`blob_hash`, `staging_backend`) must parse. The parsed enum + /// values are cached for the typed accessors below. void validate(); /// Typed accessors for the enum-valued string settings, parsed and cached by `validate`. @@ -83,11 +79,10 @@ struct ContentAddressedSettings /// only; the two scopes are deliberately distinct. bool skipAccessCheck() const; Cas::StagingBackend stagingBackend() const; - Cas::PartFolderValidate partFolderValidate() const; private: /// The parsed enum values live inside `impl` (defined in the .cpp, where the forward-declared - /// `Cas::StagingBackend` / `Cas::PartFolderValidate` types are complete), not as members here. + /// `Cas::StagingBackend` type is complete), not as members here. std::unique_ptr impl; }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp index 1216906194b4..3a78e4a5eb3f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp @@ -113,7 +113,7 @@ ContentAddressedTransaction::~ContentAddressedTransaction() /// backstop for aborted/exception-unwound transactions whose publishStaging never ran. cleanupPendingTempFiles(); - /// An uncommitted transaction's uploads become min_active-spared debris: abandon every + /// An uncommitted transaction's uploads become min_active_build_sequence-spared debris: abandon every /// still-open PartWriteTxn so its build_seq is retired. This replaces the former pin machinery. if (committed) return; @@ -289,7 +289,8 @@ void ContentAddressedTransaction::uploadPendingBlobs(PartStaging & st) const uint64_t payload_size = pb.size; source.open = [store, staging_key, header_len, payload_size]() -> std::unique_ptr { - auto staged = store->backend().getStream(staging_key); + Cas::CasOperation op = store->mountRequests().admit(); + auto staged = op.stream(staging_key, Cas::Retry::standard()); if (!staged) throw Exception( ErrorCodes::FILE_DOESNT_EXIST, @@ -297,7 +298,7 @@ void ContentAddressedTransaction::uploadPendingBlobs(PartStaging & st) staging_key); String encoded_header(header_len, '\0'); - staged->stream->readStrict(encoded_header.data(), encoded_header.size()); + staged->readStrict(encoded_header.data(), encoded_header.size()); const Cas::EnvelopeHeader decoded = Cas::decodeEnvelopeHeader( encoded_header, header_len + payload_size, @@ -309,7 +310,7 @@ void ContentAddressedTransaction::uploadPendingBlobs(PartStaging & st) staging_key, decoded.header_len, header_len); - return std::move(staged->stream); + return staged; }; } else @@ -759,8 +760,8 @@ std::string ContentAddressedTransaction::buildS3StagingBlobHeader( header.kind = Cas::ObjectKind::Blob; header.incarnation_tag = (static_cast(thread_local_rng()) << 64) | thread_local_rng(); header.build_id = 0; /// not known at stream time; diagnostic-only (not read by GC/read paths) - /// ch = the real ClickHouse VERSION_INTEGER (diagnostic-only; consistent with `PartWriteTxn::buildHeader`). - /// The v3 envelope drops hash_algo/domain_id/writer_version, so forensics ride on ch + bld. + /// `chver` = the real ClickHouse VERSION_INTEGER (diagnostic-only; consistent with `PartWriteTxn::buildHeader`). + /// The envelope drops hash_algo/domain_id/writer_version, so forensics ride on `chver` + `build`. header.provenance = Cas::Provenance{ /*created_at_ms*/ 0, cfg.server_id, VERSION_INTEGER, Cas::ProvenanceOp::Other}; header.intended_ref = route.ns.string() + "/" + route.ref; @@ -906,10 +907,13 @@ std::unique_ptr ContentAddressedTransaction::writeFile( /// buffer writes this header first, UNHASHED and excluded from the reported size, so the /// content key stays the pool's hash of `payload` and `blob_size` stays the payload size. std::string envelope_header = buildS3StagingBlobHeader(*r); - /// rev.7 [C2]: capture the fence generation now, re-checked immediately before the durable - /// `sink->finalize()` in `finalizeImpl` (the streaming upload becomes durable there). + /// The staged upload becomes durable in `sink->finalize()`, outside this request contract by + /// design, so its admission is re-checked there through the callback below, against the + /// generation captured HERE. `admit().generation()` reads a plain value off a temporary + /// `CasOperation` that does not outlive this statement -- it names the fence generation at + /// THIS instant, not a live operation the two calls share. const Cas::PoolPtr pool = metadata_storage.store(); - const uint64_t admitted_generation = pool->fenceGeneration(); + const uint64_t admitted_generation = pool->mountRequests().admit().generation(); return std::make_unique( std::move(object_sink), staging_key, @@ -1208,8 +1212,9 @@ void ContentAddressedTransaction::createHardLink(const std::string & path_from, /// Carry forward from the COMMITTED source part: read the source manifest, find the named entry, /// record a TOKENLESS W-EVIDENCE dep for its blob (no HEAD before precommit; promote re-proves it). - /// ForceFresh getView == resolveRef(allow_stale=false) + readManifestShared, so this is the same - /// request pattern as before, now instrumented via the facade. + /// ForceFresh getView == resolveRef(allow_stale=false) + readManifestShared; the decode is served + /// from the manifest cache when warm, so a burst of hardlinks from one source part costs no + /// manifest request after the first. auto view = metadata_storage.partAccess()->getView(src->refKey(), Cas::Freshness::ForceFresh); if (!view) throw Exception(ErrorCodes::FILE_DOESNT_EXIST, @@ -1615,9 +1620,9 @@ void ContentAddressedTransaction::unlinkFile(const std::string & path, bool if_e std::erase_if(st.entries, [&](const Cas::ManifestEntry & e) { return e.path == r->file; }); if (!staged_here) { - /// One mandatory body-HEAD per (transaction, ref), not per file: the MergeTree fast-removal - /// path unlinks every file of the part through THIS transaction right before removeDirectory. - /// The first unlink re-proves the body ForceFresh; the rest of the burst reuses that proof. + /// One fresh resolve per (transaction, ref), not per file: the MergeTree fast-removal path + /// unlinks every file of the part through THIS transaction right before removeDirectory. + /// The first unlink resolves ForceFresh; the rest of the burst reads the retained view. const String memo_key = r->refKey().cacheKey(); const bool already_proven = force_fresh_validated_refs.contains(memo_key); const auto view = metadata_storage.partAccess()->getView( diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.h index a9ac41cfaac7..60e473ede583 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.h @@ -158,11 +158,11 @@ class ContentAddressedTransaction : public IMetadataTransaction bool committed = false; bool failed = false; - /// Memoizes, per (this transaction, ref), whether `unlinkFile` has already re-proven a committed - /// ref's manifest body `ForceFresh`. The MergeTree fast-removal path unlinks every file of a part - /// through ONE transaction right before `removeDirectory`: the first unlink's `ForceFresh` view - /// proves the body once; the rest of the burst reuse that proof (`Cas::Freshness::CachedForLoad`) - /// instead of paying one manifest-body HEAD per file. Cleared in `commit()`'s epilogue. + /// Memoizes, per (this transaction, ref), whether `unlinkFile` has already resolved a committed + /// ref `ForceFresh`. The MergeTree fast-removal path unlinks every file of a part through ONE + /// transaction right before `removeDirectory`: the first unlink resolves fresh and bypasses the + /// retained view; the rest of the burst reads the retained view (`Cas::Freshness::CachedForLoad`). + /// Cleared in `commit()`'s epilogue. std::unordered_set force_fresh_validated_refs; /// Stage a content part file as a blob without recording a dependency proof, and add/replace its diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.cpp index 65176572e896..523912c519c4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.cpp @@ -1,7 +1,11 @@ #include +#include #include +#include #include #include +#include +#include namespace DB { @@ -20,31 +24,86 @@ namespace constexpr std::string_view kBlobType = "cas_blob"; -std::string_view opToWord(ProvenanceOp op) +namespace EnvelopeWire { - switch (op) - { - case ProvenanceOp::Other: return "other"; - case ProvenanceOp::Insert: return "insert"; - case ProvenanceOp::Merge: return "merge"; - case ProvenanceOp::Mutation: return "mutation"; - case ProvenanceOp::Attach: return "attach"; - case ProvenanceOp::Repack: return "repack"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob envelope: unknown ProvenanceOp {}", static_cast(op)); + constexpr WireKey type{"type"}; + constexpr WireKey version{"v"}; + constexpr WireKey tag{"tag"}; + constexpr WireKey build{"build"}; + constexpr WireKey time_ms{"time_ms"}; + constexpr WireKey creator{"creator"}; + constexpr WireKey op{"op"}; + constexpr WireKey chver{"chver"}; + constexpr WireKey ref{"ref"}; + /// Not a field this build understands: written only to exercise the reader's `!`-key policy. + constexpr WireKey unknown_critical{"!x"}; } -ProvenanceOp opFromWord(std::string_view w) +constexpr EnumWireTable kProvenanceOpWords{{{ + {ProvenanceOp::Other, "other"}, + {ProvenanceOp::Insert, "insert"}, + {ProvenanceOp::Merge, "merge"}, + {ProvenanceOp::Mutation, "mutation"}, + {ProvenanceOp::Attach, "attach"}, + {ProvenanceOp::Repack, "repack"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + +/// Byte cost of one JSON key as written by `CasJsonWriter::key`: the leading `{`/`,` separator (1) +/// plus the opening quote (1), the key text, and the closing quote and colon (2). +constexpr size_t keyCost(WireKey key) { - if (w == "other") return ProvenanceOp::Other; - if (w == "insert") return ProvenanceOp::Insert; - if (w == "merge") return ProvenanceOp::Merge; - if (w == "mutation") return ProvenanceOp::Mutation; - if (w == "attach") return ProvenanceOp::Attach; - if (w == "repack") return ProvenanceOp::Repack; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob envelope: unknown op '{}'", w); + return 4 + key.text.size(); } +/// A quoted `hex128Value` is always exactly this wide -- 2 quote bytes plus 2 hex digits per +/// `UInt128` byte -- because `writeHexUIntLowercase` zero-pads; there is no smaller or larger case. +constexpr size_t kQuotedHex128Len = 2 + sizeof(UInt128) * 2; + +/// Maximum decimal digits an unquoted `writeIntText` value can produce for each integer width the +/// envelope persists, taken from the type itself rather than re-typed as a literal. +constexpr size_t kMaxU64DecimalLen = std::numeric_limits::digits10 + 1; +constexpr size_t kMaxU32DecimalLen = std::numeric_limits::digits10 + 1; + +/// The longest persisted `op` word, found by walking the table rather than hardcoding one -- the +/// worst case must track `kProvenanceOpWords` even if a future entry outgrows "mutation". +constexpr size_t maxProvenanceOpWordLen() +{ + size_t max_len = 0; + for (const auto & entry : kProvenanceOpWords.entries) + max_len = std::max(max_len, entry.word.size()); + return max_len; +} + +/// Mandatory (always-written whenever `provenance` is set) non-`ref` fields at type maxima, in the +/// exact field order `encodeEnvelopeHeader` writes them. `CasPoolMetaFormat.cpp` records why 240 was +/// chosen as the floor above this bound. +constexpr size_t kMandatoryNonRefWorstCase = + keyCost(EnvelopeWire::type) + 2 + kBlobType.size() + + keyCost(EnvelopeWire::version) + kMaxU32DecimalLen + + keyCost(EnvelopeWire::tag) + kQuotedHex128Len + + keyCost(EnvelopeWire::build) + kQuotedHex128Len + + keyCost(EnvelopeWire::time_ms) + kMaxU64DecimalLen + + keyCost(EnvelopeWire::creator) + kQuotedHex128Len + + keyCost(EnvelopeWire::op) + 2 + maxProvenanceOpWordLen() + + keyCost(EnvelopeWire::chver) + kMaxU32DecimalLen; + +/// The encoder always frames `ref`, even when empty: the key (`,"ref":`), the empty quotes, the +/// closing `}`, and the trailing '\n' reserved at byte `blob_header_len - 1`. +constexpr size_t kRefFramingAndTerminator = keyCost(EnvelopeWire::ref) + 2 + 1 + 1; + +/// The worst-case byte count `encodeEnvelopeHeader` can ever produce before the diagnostic `ref` +/// gets any budget at all. Proven, not merely documented: the static_assert below fails the BUILD if +/// a future key or type change ever closes the margin under `kMinBlobHeaderLen`. +constexpr size_t kMandatoryDescriptorWorstCase = kMandatoryNonRefWorstCase + kRefFramingAndTerminator; + +static_assert(kMandatoryDescriptorWorstCase <= kMinBlobHeaderLen - 1, + "the mandatory blob-envelope fields plus the empty-ref framing must fit under kMinBlobHeaderLen " + "(the trailing '\\n' is already counted above, so the spare byte is the diagnostic ref's floor " + "budget, not the newline); if a field grew, either shrink it back or " + "raise kMinBlobHeaderLen (CasEnvelopeLimits.h) to match"); + /// The escaped byte-length of one raw ref char under the frozen envelope alphabet (see writeEnvelopeRefField). size_t escapedLen(char c) { @@ -96,7 +155,20 @@ void writeEnvelopeRefField(String & json, size_t budget, std::string_view raw_re } -String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len) +const size_t mandatory_descriptor_worst_case = kMandatoryDescriptorWorstCase; + +std::string_view provenanceOpToWireWord(ProvenanceOp op) +{ + return kProvenanceOpWords.toWord(op, "CAS blob envelope"); +} + +ProvenanceOp provenanceOpFromWireWord(std::string_view w) +{ + return kProvenanceOpWords.fromWord(w, "CAS blob envelope"); +} + +String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len, + std::optional version_override) { if (header.kind != ObjectKind::Blob) throw Exception(ErrorCodes::LOGICAL_ERROR, @@ -108,23 +180,21 @@ String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len) { CasJsonWriter buf(256); bool first = true; - writeKey(buf, "type", first); writeStringValue(buf, kBlobType); - writeKey(buf, "v", first); writeIntText(currentCompatibilityVersion(), buf); - writeKey(buf, "tag", first); writeHex128Value(buf, header.incarnation_tag); - writeKey(buf, "bld", first); writeHex128Value(buf, header.build_id); + writeStringField(buf, EnvelopeWire::type, kBlobType, first); + writeNumberField(buf, EnvelopeWire::version, version_override.value_or(currentCompatibilityVersion()), first); + writeHex128Field(buf, EnvelopeWire::tag, header.incarnation_tag, first); + writeHex128Field(buf, EnvelopeWire::build, header.build_id, first); if (header.provenance) { - writeKey(buf, "ts", first); writeIntText(header.provenance->created_at_ms, buf); - writeKey(buf, "by", first); writeHex128Value(buf, header.provenance->creator_server_id); - writeKey(buf, "op", first); writeStringValue(buf, opToWord(header.provenance->op)); - writeKey(buf, "ch", first); writeIntText(header.provenance->ch_version, buf); + writeNumberField(buf, EnvelopeWire::time_ms, header.provenance->created_at_ms, first); + writeHex128Field(buf, EnvelopeWire::creator, header.provenance->creator_server_id, first); + writeStringField(buf, EnvelopeWire::op, provenanceOpToWireWord(header.provenance->op), first); + writeNumberField(buf, EnvelopeWire::chver, header.provenance->ch_version, first); } /// Test-only critical extension: an unknown `!`-key BEFORE `ref`. if (header.emit_unknown_critical_key) - { - writeKey(buf, "!x", first); writeStringValue(buf, "1"); - } - json = std::move(buf).take(); /// e.g. {"type":"cas_blob","v":3,...,"ch":26006001 (no ref, no closing brace) + writeStringField(buf, EnvelopeWire::unknown_critical, "1", first); + json = std::move(buf).take(); /// e.g. {"type":"cas_blob","v":1,...,"chver":26006001 (no ref, no closing brace) } /// Optional `ref`, truncated to the exact remaining budget. Layout after this block: @@ -132,15 +202,18 @@ String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len) /// (byte blob_header_len-1 is reserved for '\n'; the pad zone fills the gap with spaces). if (header.intended_ref) { - static constexpr std::string_view ref_key = ",\"ref\":"; + /// 4 = the `,"` before and `":` after the key text — the `,"ref":` framing minus the key itself. + constexpr size_t ref_key_size = 4 + EnvelopeWire::ref.text.size(); /// +3 = opening quote + closing quote + closing brace. - const size_t fixed = json.size() + ref_key.size() + 3; + const size_t fixed = json.size() + ref_key_size + 3; if (blob_header_len < 1 || fixed > static_cast(blob_header_len) - 1) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS blob envelope: non-ref fields ({} bytes) do not fit blob_header_len {} before the ref", fixed, blob_header_len); const size_t budget = (static_cast(blob_header_len) - 1) - fixed; - json += ref_key; + json += ",\""; + json += EnvelopeWire::ref.text; + json += "\":"; writeEnvelopeRefField(json, budget, *header.intended_ref); } json += '}'; @@ -173,7 +246,7 @@ EnvelopeHeader decodeEnvelopeHeader(std::string_view head_bytes, uint64_t /*obje String key; while (r.nextKey(key)) { - if (key == "type") + if (key == EnvelopeWire::type) { const String t = r.readString(); if (t != kBlobType) @@ -181,37 +254,37 @@ EnvelopeHeader decodeEnvelopeHeader(std::string_view head_bytes, uint64_t /*obje "CAS blob envelope: object is a '{}', not a '{}'", t, kBlobType); saw_type = true; } - else if (key == "v") + else if (key == EnvelopeWire::version) { h.compatibility_version = r.readU32Number(); checkCompatibility(h.compatibility_version, "blob envelope"); saw_v = true; } - else if (key == "tag") + else if (key == EnvelopeWire::tag) h.incarnation_tag = r.readHex128(); - else if (key == "bld") + else if (key == EnvelopeWire::build) h.build_id = r.readHex128(); - else if (key == "ts") + else if (key == EnvelopeWire::time_ms) { prov.created_at_ms = r.readU64Number(); have_prov = true; } - else if (key == "by") + else if (key == EnvelopeWire::creator) { prov.creator_server_id = r.readHex128(); have_prov = true; } - else if (key == "op") + else if (key == EnvelopeWire::op) { - prov.op = opFromWord(r.readString()); + prov.op = provenanceOpFromWireWord(r.readString()); have_prov = true; } - else if (key == "ch") + else if (key == EnvelopeWire::chver) { prov.ch_version = static_cast(r.readU64Number()); have_prov = true; } - else if (key == "ref") + else if (key == EnvelopeWire::ref) h.intended_ref = r.readString(); else r.skipUnknown(key); /// `!`-key -> UNKNOWN_FORMAT_VERSION; unknown plain key -> skipped (tolerant) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.h index 19250fe69ddd..e968bbec5dbc 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobEnvelopeFormat.h @@ -1,4 +1,5 @@ #pragma once +#include #include #include #include @@ -32,6 +33,20 @@ enum class ProvenanceOp : uint8_t Repack = 5, }; +/// Returns the persisted wire word for a validated provenance operation. +/// The largest descriptor `encodeEnvelopeHeader` can ever produce before the diagnostic `ref` gets any +/// budget: every mandatory field at its type maximum, the longest provenance word, the `ref` framing +/// with empty quotes, the closing brace and the trailing newline. A `static_assert` beside its +/// definition proves it fits under `kMinBlobHeaderLen`; this declaration exists so the boundary test +/// can confirm the SAME number against bytes the encoder actually produced, which is the half a +/// compile-time proof cannot do — an understated formula satisfies the assert quite happily. +extern const size_t mandatory_descriptor_worst_case; + +std::string_view provenanceOpToWireWord(ProvenanceOp op); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +ProvenanceOp provenanceOpFromWireWord(std::string_view w); + /// Optional diagnostic metadata recorded with an envelope. The fields identify when and where the /// incarnation was created, the ClickHouse build that wrote it, and the operation that produced it; /// none of them participates in object identity or a protocol decision. @@ -55,19 +70,20 @@ struct Provenance /// algorithm and digest are already present in the object key and manifest reference, `domain_id` had /// no validating consumer, and `header_hash` had no consumer once the CityHash64 check left the /// envelope. Writer forensics are represented -/// by `ch` and `bld`, so a separate `writer_version` is unnecessary. The `v` field is the sole format -/// compatibility gate; a reader rejects a version it does not understand before interpreting the body. +/// by `chver` and `build`, so a separate `writer_version` is unnecessary. The `v` field is the sole +/// format compatibility gate; a reader rejects a version it does not understand before interpreting +/// the body. struct EnvelopeHeader { ObjectKind kind = ObjectKind::Blob; /// Set by decode from the header `v`; encode stamps `currentCompatibilityVersion`. A reader /// fails closed (UNKNOWN_FORMAT_VERSION) when `v` exceeds what this build understands. uint32_t compatibility_version = 0; - UInt128 incarnation_tag{}; /// `tag` - UInt128 build_id{}; /// `bld` - std::optional provenance; /// `ts` / `by` / `op` / `ch` - std::optional intended_ref; /// `ref` (diagnostic; truncated on encode to fit the header) - uint32_t header_len = 0; /// filled by encode/decode = blob_header_len (payload offset) + UInt128 incarnation_tag{}; /// `tag` + UInt128 build_id{}; /// `build` + std::optional provenance; /// `time_ms` / `creator` / `op` / `chver` + std::optional intended_ref; /// `ref` (diagnostic; truncated on encode to fit the header) + uint32_t header_len = 0; /// filled by encode/decode = blob_header_len (payload offset) /// Test-only knob: emit an unknown `!`-critical key. Decoding the resulting header must fail /// closed with `UNKNOWN_FORMAT_VERSION`, exercising the compatibility rule for critical extensions. bool emit_unknown_critical_key = false; @@ -78,7 +94,14 @@ struct EnvelopeHeader /// diagnostic `ref` is the only truncatable field and is shortened, never dropped, when necessary to /// preserve the fixed layout. The header is built without payload bytes, so an upload can stage the /// header before the payload is streamed. -String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len); +/// `version_override` exists for one caller: the boundary test that has to see what the descriptor +/// costs at the WIDEST version the budget reserves room for. The budget is sized for a ten-digit +/// version; production has only ever written a one-digit one, so a test that encodes at the current +/// version and adds the missing digits arithmetically never sends the boundary through the encoder +/// at all -- it re-derives the formula it is supposed to be checking. Production passes nothing and +/// gets `currentCompatibilityVersion()`. +String encodeEnvelopeHeader(EnvelopeHeader & header, uint32_t blob_header_len, + std::optional version_override = {}); /// Parses and validates the JSON descriptor, its expected `type`, and its compatibility version. /// Derives `header_len` from the terminating '\n' and requires every preceding byte in the pad zone to diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.cpp index b62fd3b82424..2b40584e0199 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include @@ -17,26 +18,30 @@ namespace DB::Cas namespace { -std::string_view metaStateToWord(MetaState s) +namespace BlobMetaWire { - switch (s) - { - case MetaState::Clean: return "clean"; - case MetaState::Condemned: return "condemned"; - } - // The enum is persisted as a closed vocabulary. Do not silently invent a spelling for a value - // added without a corresponding format decision: that would make the writer emit data older - // readers cannot classify. - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob meta: unknown MetaState {}", static_cast(s)); + constexpr WireKey state{"state"}; + constexpr WireKey condemn_round{"condemn_round"}; + constexpr WireKey size{"size"}; } -MetaState metaStateFromWord(std::string_view w) +constexpr EnumWireTable kMetaStateWords{{{ + {MetaState::Clean, "clean"}, + {MetaState::Condemned, "condemned"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + +} + +std::string_view metaStateToWireWord(MetaState state) { - if (w == "clean") return MetaState::Clean; - if (w == "condemned") return MetaState::Condemned; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob meta: unknown state '{}'", w); + return kMetaStateWords.toWord(state, "CAS blob meta"); } +MetaState metaStateFromWireWord(std::string_view w) +{ + return kMetaStateWords.fromWord(w, "CAS blob meta"); } String encodeBlobMeta(const BlobMeta & meta) @@ -46,12 +51,9 @@ String encodeBlobMeta(const BlobMeta & meta) // `version` is represented by the header line. The JSON body contains only fields that describe // the current marker and its accounting data. bool first = true; - writeKey(out, "st", first); - writeStringValue(out, metaStateToWord(meta.state)); - writeKey(out, "cr", first); - writeU64StringValue(out, meta.condemn_round); - writeKey(out, "sz", first); - writeU64StringValue(out, meta.size); + writeWordField(out, BlobMetaWire::state, metaStateToWireWord(meta.state), first); + writeU64StringField(out, BlobMetaWire::condemn_round, meta.condemn_round, first); + writeU64StringField(out, BlobMetaWire::size, meta.size, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -72,20 +74,20 @@ BlobMeta decodeBlobMeta(std::string_view bytes) String key; while (r.nextKey(key)) { - if (key == "st") + if (key == BlobMetaWire::state) { - m.state = metaStateFromWord(r.readString()); + m.state = metaStateFromWireWord(r.readString()); saw_state = true; } - else if (key == "cr") + else if (key == BlobMetaWire::condemn_round) m.condemn_round = r.readU64String(); - else if (key == "sz") + else if (key == BlobMetaWire::size) m.size = r.readU64String(); else r.skipUnknown(key); } if (!saw_state) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob meta: missing st"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob meta: missing state"); if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS blob meta: trailing bytes"); return m; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.h index 6694fbafeb33..9176dc641203 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasBlobMetaFormat.h @@ -19,6 +19,13 @@ enum class MetaState : uint8_t /// so a writer may republish it by replacing the body and updating this marker. }; +/// Convert a meta-state discriminator to its canonical wire word. Throws `LOGICAL_ERROR` for an +/// out-of-range enum value. +std::string_view metaStateToWireWord(MetaState state); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +MetaState metaStateFromWireWord(std::string_view w); + /// The durable per-hash meta record. Its text representation consists of a format header followed by /// one JSON object with the state word, the GC condemnation round, and the raw body size. `size` is /// retained for introspection, fsck, and GC accounting; reads of the blob never consult the meta. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasEnvelopeLimits.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasEnvelopeLimits.h new file mode 100644 index 000000000000..92e6b252ab1b --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasEnvelopeLimits.h @@ -0,0 +1,16 @@ +#pragma once + +#include + +namespace DB::Cas +{ + +/// The pool-wide floor for `blob_header_len`. One compile-time owner, read by BOTH +/// `validatePoolBlobHeaderLen` (pool creation / decode) and the blob-envelope codec, so the +/// mandatory-descriptor worst-case proof and the enforced floor can never guard different numbers. +/// The byte-for-byte worst-case derivation (`kMandatoryDescriptorWorstCase`) lives beside the +/// envelope key constants in `CasBlobEnvelopeFormat.cpp`; `CasPoolMetaFormat.cpp` records why 240 +/// (rather than the bare worst case) was chosen as the floor. +inline constexpr uint64_t kMinBlobHeaderLen = 240; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.cpp index b4bba2bff08f..b2c0bf1f9756 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.cpp @@ -2,8 +2,10 @@ #include #include #include +#include #include #include +#include #include #include @@ -20,43 +22,57 @@ namespace ErrorCodes namespace DB::Cas { -std::string_view holdReasonToWord(HoldReason r) -{ - switch (r) - { - case HoldReason::GapBelowWitness: return "gap_below_witness"; - case HoldReason::UnconsumedSealCrossing: return "unconsumed_seal_crossing"; - case HoldReason::WitnessDisappeared: return "witness_disappeared"; - case HoldReason::BodyUndecodable: return "body_undecodable"; - case HoldReason::ManifestBodyMissing: return "manifest_body_missing"; - case HoldReason::CheckpointUndecodable: return "checkpoint_undecodable"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown hold reason {}", static_cast(r)); -} - namespace { -HoldReason holdReasonFromWord(std::string_view w) +namespace FoldSealWire { - if (w == "gap_below_witness") return HoldReason::GapBelowWitness; - if (w == "unconsumed_seal_crossing") return HoldReason::UnconsumedSealCrossing; - if (w == "witness_disappeared") return HoldReason::WitnessDisappeared; - if (w == "body_undecodable") return HoldReason::BodyUndecodable; - if (w == "manifest_body_missing") return HoldReason::ManifestBodyMissing; - if (w == "checkpoint_undecodable") return HoldReason::CheckpointUndecodable; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown hold reason '{}'", w); + constexpr WireKey generation{"generation"}; + constexpr WireKey parent_generation{"parent_generation"}; + constexpr WireKey kind{"kind"}; + constexpr WireKey run_key{"key"}; + constexpr WireKey checksum{"checksum"}; + constexpr WireKey shard{"shard"}; + constexpr WireKey key_generation{"key_generation"}; + constexpr WireKey life{"life"}; + constexpr WireKey classification{"class"}; + constexpr WireKey fold_epoch{"fold_epoch"}; + constexpr WireKey fold_seq{"fold_seq"}; + constexpr WireKey hold_reason{"hold_reason"}; + constexpr WireKey hold_epoch{"hold_epoch"}; + constexpr WireKey hold_seq{"hold_seq"}; + constexpr WireKey retries{"retries"}; + constexpr WireKey retry_round{"retry_round"}; + constexpr WireKey remove_epoch{"remove_epoch"}; + constexpr WireKey remove_seq{"remove_seq"}; + constexpr WireKey condemned_total{"condemned"}; + constexpr WireKey pending_total{"pending"}; + constexpr WireKey oldest_round{"oldest_round"}; } -/// The classification set is CLOSED. Every consumer of a coverage row branches on exact values — the -/// sweep's §6 deletion premise refuses a row by testing `== 4` and then `== 0` — so a value outside the -/// set is not an unknown variant to be tolerated forward: it is a row that passes every refusal written -/// in terms of the set and reaches the irreversible delete. One predicate, used by both directions, so -/// the writer's self-check and the reader's fail-close can never name different sets. -bool isKnownClassification(uint64_t classification) -{ - return classification == 0 || classification == 1 || classification == 2 || classification == 4; -} +constexpr std::string_view kRefLifeTag = "ref_life"; +constexpr std::string_view kBlobRunTag = "blob_run"; +constexpr std::string_view kCondemnedTag = "condemned"; + +constexpr EnumWireTable kHoldReasonWords{{{ + {HoldReason::GapBelowWitness, "gap_below_witness"}, + {HoldReason::UnconsumedSealCrossing, "unconsumed_seal_crossing"}, + {HoldReason::WitnessDisappeared, "witness_disappeared"}, + {HoldReason::BodyUndecodable, "body_undecodable"}, + {HoldReason::ManifestBodyMissing, "manifest_body_missing"}, + {HoldReason::CheckpointUndecodable, "checkpoint_undecodable"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + +constexpr EnumWireTable kCoverageClassWords{{{ + {CoverageClass::Absent, "absent"}, + {CoverageClass::Unchanged, "unchanged"}, + {CoverageClass::Folded, "folded"}, + {CoverageClass::Clamped, "clamped"}, +}}}; + +static_assert(casEnumTableCoversEnum()); /// A hold names a position the fold must resolve, and both components of that id are nonzero (the /// canonical `RefTxnId` rule `renderRefTxnId` enforces for every id that becomes a key). A zero @@ -83,16 +99,16 @@ void insertRecordOnce(Map & map, const Key & key, Value && value, std::string_vi what, key); } -/// Emit one run record (`k` = "btr") WITHOUT its line terminator; the caller closes (and measures) the +/// Emit one run record (`kind` = `blob_run`) WITHOUT its line terminator; the caller closes (and measures) the /// line, and sorts the vector by key first. void writeRun(CasJsonWriter & out, std::string_view kind, const RunRef & r) { bool first = true; - writeKey(out, "k", first); writeStringValue(out, kind); - writeKey(out, "key", first); writeStringValue(out, r.key); - writeKey(out, "ck", first); writeHex128Value(out, r.checksum); - writeKey(out, "shard", first); writeIntText(r.shard, out); - writeKey(out, "gen", first); writeU64StringValue(out, r.generation); + writeWordField(out, FoldSealWire::kind, kind, first); + writeStringField(out, FoldSealWire::run_key, r.key, first); + writeHex128Field(out, FoldSealWire::checksum, r.checksum, first); + writeNumberField(out, FoldSealWire::shard, r.shard, first); + writeU64StringField(out, FoldSealWire::key_generation, r.key_generation, first); closeObject(out, first); } @@ -106,7 +122,7 @@ void validateFoldSealStructure( std::vector run_seen(gc_shards, false); for (const RunRef & run : seal.blob_target_runs) { - if (run.key.empty() || run.generation == 0) + if (run.key.empty() || run.key_generation == 0) throw Exception(error_code, "CAS fold seal {}: blob-target run requires a nonempty key and nonzero physical generation", source); @@ -121,10 +137,10 @@ void validateFoldSealStructure( run_seen[run.shard] = true; const auto parsed = layout.parseBlobTargetRunKey(run.key); - if (!parsed || parsed->generation != run.generation || parsed->shard != run.shard || parsed->seq != 0) + if (!parsed || parsed->generation != run.key_generation || parsed->shard != run.shard || parsed->seq != 0) throw Exception(error_code, "CAS fold seal {}: blob-target run key '{}' is not canonical for generation {}, shard {}, sequence 0", - source, run.key, run.generation, run.shard); + source, run.key, run.key_generation, run.shard); } if (seal.condemned_summary.size() != gc_shards) @@ -154,6 +170,41 @@ void validateFoldSealStructure( } +std::string_view holdReasonToWord(HoldReason r) +{ + return kHoldReasonWords.toWord(r, "CAS fold seal hold reason"); +} + +HoldReason holdReasonFromWord(std::string_view w) +{ + return kHoldReasonWords.fromWord(w, "CAS fold seal hold reason"); +} + +std::string_view coverageClassToWord(CoverageClass c) +{ + return kCoverageClassWords.toWord(c, "CAS fold seal classification"); +} + +CoverageClass coverageClassFromWord(std::string_view w) +{ + return kCoverageClassWords.fromWord(w, "CAS fold seal classification"); +} + +namespace +{ +/// A seal can carry thousands of `ref_life` rows, so a rejected word names the row that carries it -- +/// "unknown word" alone leaves an operator scanning the object by hand. +CoverageClass coverageClassInRow(std::string_view w, std::string_view life_hex) +{ + return kCoverageClassWords.fromWord(w, fmt::format("CAS fold seal: ref_life '{}' classification", life_hex)); +} + +HoldReason holdReasonInRow(std::string_view w, std::string_view life_hex) +{ + return kHoldReasonWords.fromWord(w, fmt::format("CAS fold seal: ref_life '{}' hold_reason", life_hex)); +} +} + FoldSealCaps foldSealCaps() { const FormatTraits & t = traitsFor(FormatId::FoldSeal); @@ -205,8 +256,8 @@ String encodeFoldSeal(const CasFoldSeal & seal) /// meta line { bool first = true; - writeKey(out, "g", first); writeU64StringValue(out, seal.generation); - writeKey(out, "pg", first); writeU64StringValue(out, seal.parent_generation); + writeU64StringField(out, FoldSealWire::generation, seal.generation, first); + writeU64StringField(out, FoldSealWire::parent_generation, seal.parent_generation, first); closeObject(out, first); closeLine("meta"); } @@ -227,22 +278,19 @@ String encodeFoldSeal(const CasFoldSeal & seal) /// process, not corruption arriving from a store — and none of these shapes is repairable once /// durable, so none is ever written. /// - /// A classification outside the closed set first, because the two checks after it are stated in - /// terms of the set and a row they cannot classify makes their answers meaningless. - if (!isKnownClassification(cov.classification)) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CAS fold seal: coverage '{}' has classification {}, which is not one of the four the " - "fold grammar defines (0 absent, 1 unchanged, 2 folded, 4 clamped) — every consumer " - "branches on those exact values, so this row would pass refusals meant to stop it", - life_hex, cov.classification); - /// A classification-4 row whose hold was dropped is indistinguishable, once durable, from a - /// namespace that stopped for no reason — and a hold on any other classification claims a stop - /// that did not happen. - if ((cov.classification == 4) != cov.hold.has_value()) + /// A classification outside the four named values first, because the two checks after it are + /// stated in terms of those names and a row they cannot classify makes their answers meaningless. + /// `coverageClassToWord` IS that range check (it throws `LOGICAL_ERROR` for a value the table + /// does not index), so capturing its result here also gives the record its wire value below. + const std::string_view classification_word = coverageClassToWord(cov.classification); + /// A clamped row whose hold was dropped is indistinguishable, once durable, from a namespace that + /// stopped for no reason — and a hold on any other classification claims a stop that did not + /// happen. + if ((cov.classification == CoverageClass::Clamped) != cov.hold.has_value()) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS fold seal: coverage '{}' has classification {} and {} hold — the hold fields are " - "required for classification 4 and forbidden otherwise", - life_hex, cov.classification, cov.hold ? "a" : "no"); + "required for classification clamped and forbidden otherwise", + life_hex, classification_word, cov.hold ? "a" : "no"); /// A hold that names no position resolves itself on the next round (nothing sorts below /// `{0, 0}`) and cannot be rendered where the sweep reports why it retained a manifest. if (cov.hold && !isCanonicalHoldPosition(cov.hold->offending_position)) @@ -263,28 +311,26 @@ String encodeFoldSeal(const CasFoldSeal & seal) life_state.cleanup_evidence->remove_txn_id.ref_sequence); bool first = true; - writeKey(out, "k", first); writeStringValue(out, "rfl"); - writeKey(out, "life", first); writeHex128Value(out, life_id); - writeKey(out, "cls", first); writeIntText(static_cast(cov.classification), out); - writeKey(out, "lfe", first); writeU64StringValue(out, cov.last_folded_ref_id.writer_epoch); - writeKey(out, "lfs", first); writeU64StringValue(out, cov.last_folded_ref_id.ref_sequence); + writeWordField(out, FoldSealWire::kind, kRefLifeTag, first); + writeHex128Field(out, FoldSealWire::life, life_id, first); + writeWordField(out, FoldSealWire::classification, classification_word, first); + writeU64StringField(out, FoldSealWire::fold_epoch, cov.last_folded_ref_id.writer_epoch, first); + writeU64StringField(out, FoldSealWire::fold_seq, cov.last_folded_ref_id.ref_sequence, first); if (cov.hold) { - writeKey(out, "hr", first); writeStringValue(out, holdReasonToWord(cov.hold->reason)); - writeKey(out, "hpe", first); writeU64StringValue(out, cov.hold->offending_position.writer_epoch); - writeKey(out, "hps", first); writeU64StringValue(out, cov.hold->offending_position.ref_sequence); - writeKey(out, "hrc", first); writeIntText(cov.hold->retry_count, out); - writeKey(out, "hnr", first); writeU64StringValue(out, cov.hold->next_retry_round); + writeWordField(out, FoldSealWire::hold_reason, holdReasonToWord(cov.hold->reason), first); + writeU64StringField(out, FoldSealWire::hold_epoch, cov.hold->offending_position.writer_epoch, first); + writeU64StringField(out, FoldSealWire::hold_seq, cov.hold->offending_position.ref_sequence, first); + writeNumberField(out, FoldSealWire::retries, cov.hold->retry_count, first); + writeU64StringField(out, FoldSealWire::retry_round, cov.hold->next_retry_round, first); } if (life_state.cleanup_evidence) { - writeKey(out, "rte", first); - writeU64StringValue(out, life_state.cleanup_evidence->remove_txn_id.writer_epoch); - writeKey(out, "rts", first); - writeU64StringValue(out, life_state.cleanup_evidence->remove_txn_id.ref_sequence); + writeU64StringField(out, FoldSealWire::remove_epoch, life_state.cleanup_evidence->remove_txn_id.writer_epoch, first); + writeU64StringField(out, FoldSealWire::remove_seq, life_state.cleanup_evidence->remove_txn_id.ref_sequence, first); } closeObject(out, first); - closeLine("rfl"); + closeLine(kRefLifeTag); ++n; } @@ -293,8 +339,8 @@ String encodeFoldSeal(const CasFoldSeal & seal) std::sort(runs.begin(), runs.end(), [](const RunRef & a, const RunRef & b) { return a.key < b.key; }); for (const RunRef & r : runs) { - writeRun(out, "btr", r); - closeLine("btr"); + writeRun(out, kBlobRunTag, r); + closeLine(kBlobRunTag); } } n += seal.blob_target_runs.size(); @@ -303,13 +349,13 @@ String encodeFoldSeal(const CasFoldSeal & seal) for (const auto & [shard, s] : seal.condemned_summary) { bool first = true; - writeKey(out, "k", first); writeStringValue(out, "cnd"); - writeKey(out, "shard", first); writeIntText(shard, out); - writeKey(out, "ct", first); writeIntText(s.condemned_total, out); - writeKey(out, "pt", first); writeIntText(s.pending_total, out); - writeKey(out, "ocr", first); writeU64StringValue(out, s.oldest_nonpending_condemn_round); + writeWordField(out, FoldSealWire::kind, kCondemnedTag, first); + writeNumberField(out, FoldSealWire::shard, shard, first); + writeNumberField(out, FoldSealWire::condemned_total, s.condemned_total, first); + writeNumberField(out, FoldSealWire::pending_total, s.pending_total, first); + writeU64StringField(out, FoldSealWire::oldest_round, s.oldest_nonpending_condemn_round, first); closeObject(out, first); - closeLine("cnd"); + closeLine(kCondemnedTag); ++n; } @@ -338,18 +384,24 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect String key; while (r.nextKey(key)) { - if (key == "g") seal.generation = r.readU64String(); - else if (key == "pg") seal.parent_generation = r.readU64String(); + if (key == FoldSealWire::generation) seal.generation = r.readU64String(); + else if (key == FoldSealWire::parent_generation) seal.parent_generation = r.readU64String(); else r.skipUnknown(key); /// Strict => any unknown key is CORRUPTED_DATA } } uint64_t seen = 0; + /// One line scratch and one reader for the whole loop: a decoder that rebuilds them per + /// row pays an allocation per row for the seen-key store and the line, which profiling put + /// at about a fifth of the instructions executed inside a row. + String row_line; + JsonObjectReader row_reader; while (true) { - const String line = readLine(in, line_cap, "fold seal"); - ReadBufferFromMemory l(line.data(), line.size()); - JsonObjectReader r(l, KeyStrictness::Strict, "fold seal"); + readLineInto(in, row_line, line_cap, "fold seal"); + ReadBufferFromMemory l(row_line.data(), row_line.size()); + row_reader.reset(l, KeyStrictness::Strict, "fold seal"); + JsonObjectReader & r = row_reader; String key; if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: empty line"); @@ -370,22 +422,19 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect seal.generation, *expected_generation); return seal; } - if (key != "k") - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: record must start with \"k\""); + if (key != FoldSealWire::kind) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: record must start with \"kind\""); const String kind = r.readString(); - if (kind == "rfl") + if (kind == kRefLifeTag) { std::optional life_id; RefCoverage cov; - /// Read WIDE and validated before it is narrowed to the persisted byte. `cls` is the field - /// every consumer branches on, and a plain `static_cast` maps 258 onto 2 ("all - /// records through the cursor were folded") and 256 onto 0 — a forged or damaged seal would - /// buy full coverage with an integer no reader ever sees. - std::optional classification; /// The hold fields are read individually so the grammar can be checked on WHICH of them /// arrived, not merely on how many. `JsonObjectReader` already rejects a duplicate key, so - /// a second `hr` can never quietly rewrite the reason. + /// a second `hold_reason` can never quietly rewrite the reason. + std::optional classification_word; + std::optional hold_reason_word; std::optional hold_reason; std::optional hold_epoch; std::optional hold_sequence; @@ -395,18 +444,21 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect std::optional remove_txn_sequence; while (r.nextKey(key)) { - if (key == "life") life_id = r.readHex128(); - else if (key == "cls") classification = r.readU64Number(); - else if (key == "lfe") cov.last_folded_ref_id.writer_epoch = r.readU64String(); - else if (key == "lfs") cov.last_folded_ref_id.ref_sequence = r.readU64String(); - else if (key == "hr") hold_reason = holdReasonFromWord(r.readString()); - else if (key == "hpe") hold_epoch = r.readU64String(); - else if (key == "hps") hold_sequence = r.readU64String(); - else if (key == "hrc") hold_retry_count = r.readU32Number(); - else if (key == "hnr") hold_next_retry_round = r.readU64String(); - else if (key == "rte") remove_txn_epoch = r.readU64String(); - else if (key == "rts") remove_txn_sequence = r.readU64String(); - else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown rfl key '{}'", key); + if (key == FoldSealWire::life) life_id = r.readHex128(); + /// The two word-valued fields are collected as words and converted BELOW, once the row's + /// life id is known: a seal can carry thousands of rows, so an unknown word has to say + /// WHICH row carries it. + else if (key == FoldSealWire::classification) classification_word = r.readString(); + else if (key == FoldSealWire::fold_epoch) cov.last_folded_ref_id.writer_epoch = r.readU64String(); + else if (key == FoldSealWire::fold_seq) cov.last_folded_ref_id.ref_sequence = r.readU64String(); + else if (key == FoldSealWire::hold_reason) hold_reason_word = r.readString(); + else if (key == FoldSealWire::hold_epoch) hold_epoch = r.readU64String(); + else if (key == FoldSealWire::hold_seq) hold_sequence = r.readU64String(); + else if (key == FoldSealWire::retries) hold_retry_count = r.readU32Number(); + else if (key == FoldSealWire::retry_round) hold_next_retry_round = r.readU64String(); + else if (key == FoldSealWire::remove_epoch) remove_txn_epoch = r.readU64String(); + else if (key == FoldSealWire::remove_seq) remove_txn_sequence = r.readU64String(); + else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown ref_life key '{}'", key); } if (!life_id || *life_id == 0) @@ -414,16 +466,13 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect "CAS fold seal: a ref-life row is missing a nonzero opaque life id"); const String life_hex = u128ToHex(*life_id); - /// `cls` is required, not defaulted: an absent one would read as 0 ("no round folded this - /// namespace"), which is a claim about a fold, not the absence of one. - if (!classification) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: rfl '{}' missing cls", life_hex); - if (!isKnownClassification(*classification)) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS fold seal: coverage '{}' has classification {}, which is not one of the four " - "the fold grammar defines (0 absent, 1 unchanged, 2 folded, 4 clamped)", - life_hex, *classification); - cov.classification = static_cast(*classification); /// in range, so narrowing is exact + /// `class` is required, not defaulted: an absent one would read as `absent` ("no round + /// folded this namespace"), which is a claim about a fold, not the absence of one. + if (!classification_word) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: ref_life '{}' missing class", life_hex); + if (hold_reason_word) + hold_reason = holdReasonInRow(*hold_reason_word, life_hex); + cov.classification = coverageClassInRow(*classification_word, life_hex); /// The same strict grammar the encoder enforces, applied to bytes we did not write. A /// PARTIAL hold is corruption, never a hold with defaults: a hold whose offending position @@ -432,11 +481,11 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect || hold_retry_count || hold_next_retry_round; const bool every_hold_field = hold_reason && hold_epoch && hold_sequence && hold_retry_count && hold_next_retry_round; - if (cov.classification == 4) + if (cov.classification == CoverageClass::Clamped) { if (!every_hold_field) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS fold seal: coverage '{}' is held (classification 4) but its hold is " + "CAS fold seal: coverage '{}' is held (classification clamped) but its hold is " "incomplete — reason, offending position, retry count and next retry round are " "all required", life_hex); /// PRESENT is not enough: the position must be one a fold can actually retry. `{0, 0}` @@ -456,8 +505,8 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect else if (any_hold_field) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: coverage '{}' carries hold fields at classification {} — they are " - "forbidden on anything but a held (classification 4) row", - life_hex, cov.classification); + "forbidden on anything but a held (classification clamped) row", + life_hex, coverageClassToWord(cov.classification)); const bool any_cleanup_field = remove_txn_epoch || remove_txn_sequence; const bool every_cleanup_field = remove_txn_epoch && remove_txn_sequence; @@ -478,7 +527,7 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect "CAS fold seal: a second ref-life record for '{}' -- a life id appears at most once", life_hex); } - else if (kind == "btr") + else if (kind == kBlobRunTag) { std::optional run_key; std::optional checksum; @@ -486,19 +535,19 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect std::optional generation; while (r.nextKey(key)) { - if (key == "key") run_key = r.readString(); - else if (key == "ck") checksum = r.readHex128(); - else if (key == "shard") shard = r.readU64Number(); - else if (key == "gen") generation = r.readU64String(); + if (key == FoldSealWire::run_key) run_key = r.readString(); + else if (key == FoldSealWire::checksum) checksum = r.readHex128(); + else if (key == FoldSealWire::shard) shard = r.readU64Number(); + else if (key == FoldSealWire::key_generation) generation = r.readU64String(); else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown run key '{}'", key); } if (!run_key || !checksum || !shard || !generation) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS fold seal: btr requires key, ck, shard, and gen"); + "CAS fold seal: blob_run requires key, checksum, shard, and key_generation"); seal.blob_target_runs.push_back(RunRef{ - .key = std::move(*run_key), .checksum = *checksum, .shard = *shard, .generation = *generation}); + .key = std::move(*run_key), .checksum = *checksum, .shard = *shard, .key_generation = *generation}); } - else if (kind == "cnd") + else if (kind == kCondemnedTag) { std::optional shard; std::optional condemned_total; @@ -506,15 +555,15 @@ CasFoldSeal decodeFoldSeal(std::string_view data, std::optional expect std::optional oldest_nonpending_condemn_round; while (r.nextKey(key)) { - if (key == "shard") shard = r.readU64Number(); - else if (key == "ct") condemned_total = r.readU64Number(); - else if (key == "pt") pending_total = r.readU64Number(); - else if (key == "ocr") oldest_nonpending_condemn_round = r.readU64String(); - else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown cnd key '{}'", key); + if (key == FoldSealWire::shard) shard = r.readU64Number(); + else if (key == FoldSealWire::condemned_total) condemned_total = r.readU64Number(); + else if (key == FoldSealWire::pending_total) pending_total = r.readU64Number(); + else if (key == FoldSealWire::oldest_round) oldest_nonpending_condemn_round = r.readU64String(); + else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS fold seal: unknown condemned key '{}'", key); } if (!shard || !condemned_total || !pending_total || !oldest_nonpending_condemn_round) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS fold seal: cnd requires shard, ct, pt, and ocr"); + "CAS fold seal: condemned requires shard, condemned, pending, and oldest_round"); insertRecordOnce(seal.condemned_summary, *shard, CondemnedSummary{ .condemned_total = *condemned_total, .pending_total = *pending_total, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.h index f6d00a6e3b50..02a043e9b2b2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFoldSealFormat.h @@ -27,7 +27,7 @@ struct RunRef String key; UInt128 checksum{}; uint64_t shard = 0; /// gc-shard this run belongs to (REQUIRED for blob_target_runs) - uint64_t generation = 0; /// generation whose key namespace physically holds the object (for retention) + uint64_t key_generation = 0; /// generation whose key namespace physically holds the object (for retention) bool operator==(const RunRef &) const = default; }; @@ -36,10 +36,12 @@ struct RunRef /// correlating logs. Persisted as a word, so an unknown word is `CORRUPTED_DATA` rather than a silently /// reinterpreted integer. /// -/// THESE ARE WIRE VALUES, AND THEY ARE APPEND-ONLY. A durable seal written by one build is read by -/// another, so a renumbered value or a reused word makes an older seal describe a hold that is not the -/// one it recorded — and a hold's whole job is to say truthfully what stopped a namespace and where. -/// Add new reasons at the end; never renumber, never repurpose a retired word. +/// THE WORDS ARE THE WIRE, AND THE WORD VOCABULARY IS APPEND-ONLY. A durable seal written by one build +/// is read by another, so a reused or repurposed word makes an older seal describe a hold that is not +/// the one it recorded — and a hold's whole job is to say truthfully what stopped a namespace and +/// where. The enumerator NUMBERS never leave memory; they are constrained only by the wire table's +/// density-and-order proof, so inserting a value in the middle is a compile-time question, not a +/// durability one. Add new reasons freely; never reuse or repurpose a retired word. enum class HoldReason : uint8_t { GapBelowWitness = 1, /// 404 at the expected id with a durable witness above it, same epoch @@ -55,6 +57,34 @@ enum class HoldReason : uint8_t /// namespace — and a second rendering of these words elsewhere would be a second place for them to drift. std::string_view holdReasonToWord(HoldReason r); +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. Paired with the renderer so a caller +/// that needs to prove the vocabulary round-trips does not have to reach for the table itself. +HoldReason holdReasonFromWord(std::string_view w); + +/// What the current round did for one life-keyed `CasFoldSeal::ref_lives` row. A BOUNDED enum: the type +/// itself is the closed set, so a producer cannot construct a fifth shape without an explicit cast, and +/// the decoder's wire-word lookup refuses anything else as `CORRUPTED_DATA` rather than silently +/// reinterpreting an integer. +/// +/// THE WORDS ARE THE WIRE, AND THE WORD VOCABULARY IS APPEND-ONLY, for the same reason `HoldReason`'s +/// is: a durable seal written by one build is read by another. The enumerator numbers never leave +/// memory. `Clamped` sits at 3, not the 4 a retired byte-valued wire used for it: nothing outside this +/// JSON ever persisted the raw byte, and a dense range is what makes the wire table's lookup a direct +/// index rather than a search. +enum class CoverageClass : uint8_t +{ + Absent = 0, /// no round has folded a ref cursor for this namespace + Unchanged = 1, /// folded, but nothing moved this round + Folded = 2, /// every record through the observed cursor was folded + Clamped = 3, /// folding stopped below the ref-log cursor; must be read again next round +}; + +/// The wire word one `CoverageClass` is persisted as; `fromWord` rejects anything else as +/// `CORRUPTED_DATA`. Exported for the same reason `holdReasonToWord` is: the classification is rendered +/// outside the codec too (`cas-inspect`, the sweep's retention messages). +std::string_view coverageClassToWord(CoverageClass c); +CoverageClass coverageClassFromWord(std::string_view w); + /// The durable hold on one namespace. It rides `RefCoverage` across rounds and across `REBUILD`, and /// clears ONLY by folding through `offending_position` and adopting the result in `gc/state` — never by /// observing another absent, because an absent is exactly the observation a lying store produces. @@ -87,21 +117,17 @@ struct RefHold bool operator==(const RefHold &) const = default; }; -/// Records what the current round did for one life-keyed `CasFoldSeal::ref_lives` row. -/// `classification` is a persisted byte: -/// 0 means absent, 1 means unchanged, 2 means all records through the observed cursor were folded, and 4 -/// means folding was clamped below the ref-log cursor. A clamped entry must be read again in the next -/// round, because an unfolded event may become foldable by then. +/// Records what the current round did for one life-keyed `CasFoldSeal::ref_lives` row. See +/// `CoverageClass` for what each value means. /// -/// THE SET {0, 1, 2, 4} IS CLOSED, and both codecs enforce it (decode `CORRUPTED_DATA`, encode -/// `LOGICAL_ERROR`). Every consumer branches on exact values — the sweep's §6 deletion premise refuses a -/// row by testing `== 4` and `== 0` — so an unrecognized byte is not a variant to tolerate: it passes -/// every refusal stated in terms of the set and reaches the delete. The decoder also validates BEFORE -/// narrowing to the byte, because a wide integer on the wire (258, say) truncates into the set and would -/// otherwise claim a coverage the fold never proved. +/// Every consumer branches on exact values — the sweep's §6 deletion premise refuses a row by testing +/// `== Clamped` and `== Absent` — so the CLOSED set matters beyond the codec, and it is now the type +/// itself: `CoverageClass` names exactly the four shapes, both codecs go through the shared wire table +/// (decode `CORRUPTED_DATA`, encode `LOGICAL_ERROR` on the one path that still reaches an out-of-range +/// value, an explicit cast), and an unrecognized wire word is refused before it ever reaches a consumer. struct RefCoverage { - uint8_t classification = 0; + CoverageClass classification = CoverageClass::Absent; /// The greatest `RefTxnId` whose owner changes have contributed their manifest-edge deltas. There is /// one ref-log stream per namespace life, so this cursor is stored in that life-keyed row. @@ -109,11 +135,11 @@ struct RefCoverage /// offending transaction so the complete transaction is retried rather than partially applied. RefTxnId last_folded_ref_id{}; - /// STRICT GRAMMAR: present if and only if `classification == 4`. Both directions enforce it — the - /// encoder refuses to write a classification-4 row without a hold (a clamp whose reason was lost is - /// indistinguishable from a clean cursor once it is durable) and refuses to write a hold on any - /// other classification (`LOGICAL_ERROR`); the decoder rejects both shapes as `CORRUPTED_DATA`. The - /// pairing lives in the type, not only in the codec, so no producer can construct the forbidden + /// STRICT GRAMMAR: present if and only if `classification == CoverageClass::Clamped`. Both directions + /// enforce it — the encoder refuses to write a clamped row without a hold (a clamp whose reason was + /// lost is indistinguishable from a clean cursor once it is durable) and refuses to write a hold on + /// any other classification (`LOGICAL_ERROR`); the decoder rejects both shapes as `CORRUPTED_DATA`. + /// The pairing lives in the type, not only in the codec, so no producer can construct the forbidden /// combination by forgetting a field. std::optional hold = std::nullopt; @@ -147,7 +173,7 @@ struct RefLifeFoldState /// must not be interpreted as zero. struct CondemnedSummary { - uint64_t condemned_total = 0; /// count of `kCondemned` rows in this shard's sealed run + uint64_t condemned_total = 0; /// count of `RunMarker::Condemned` rows in this shard's sealed run uint64_t pending_total = 0; /// how many of those are `delete_pending` (a graduation is due) uint64_t oldest_nonpending_condemn_round = UINT64_MAX; /// min condemn_round over non-pending; UINT64_MAX = none bool operator==(const CondemnedSummary &) const = default; @@ -189,10 +215,10 @@ FoldSealCaps foldSealCaps(); void checkFoldSealObjectBytes(uint64_t encoded_bytes); /// Encodes a fold seal as a strict, raw text control object. The header and meta lines are followed by -/// tagged records in the fixed `rfl`/`btr`/`cnd` order and a record-count trailer. Map iteration and +/// tagged records in the fixed `ref_life`/`blob_run`/`condemned` order and a record-count trailer. Map iteration and /// run references are sorted so retries produce byte-identical output for write-once adoption. /// -/// Enforces the whole coverage grammar — the closed classification set, the classification-4 hold +/// Enforces the whole coverage grammar — the closed classification set, the clamped-classification hold /// pairing, and the hold's canonical offending position — and BOTH byte caps: every emitted line against /// `line_cap` — header, meta, records and trailer alike, with no exception — and the whole object /// against `object_cap`. Both PUT sites go through this function, so the gate cannot be bypassed by diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.cpp index 362306691473..4ec24eeae61e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.cpp @@ -1,6 +1,11 @@ #include #include +#include + +#include +#include + namespace DB { namespace ErrorCodes @@ -16,57 +21,13 @@ namespace DB::Cas namespace { -/// Generation-1 baseline for every class. A future format change appends to that class's array and +/// Generation-1 baseline for every class. A future format change appends to that class's own array and /// bumps `G_BUILD`: additive changes use the previous reader floor, while breaking changes use the -/// new generation as the floor. Existing entries are immutable history. +/// new generation as the floor. Existing entries are immutable history. Every class currently shares +/// this baseline; a class that outgrows it gets its own named array again, the way the pre-reset +/// history once had. constexpr FormatChangePoint BASELINE[] = {{1, 1}}; -/// The two ref classes changed at generation 4 (INV-1, per-namespace contiguous ids) AND AGAIN at -/// generation 5 (Stage B's recreate-only "format bump B": the ref layer re-keyed under -/// `//`). Both changes are BREAKING even though not one byte of the encoding moved -/// either time -- a generation-3 stream's ids came from a pool-wide counter and legitimately skip, -/// which a generation-4 reader reports as corruption, and a generation-4 key names no incarnation at -/// all, which a generation-5 reader also reports as corruption (`Layout::parseRefObjectKey`). Each -/// floor is the change generation itself. -constexpr FormatChangePoint REF_STREAM[] = { - {1, 1}, - {kContiguousRefStreamsGeneration, kContiguousRefStreamsGeneration}, - {kNamespaceLifeKeyedGeneration, kNamespaceLifeKeyedGeneration}, - {kOpaqueNamespaceLifeLayoutGeneration, kOpaqueNamespaceLifeLayoutGeneration}, -}; - -/// `cas_ref_ckpt` is BORN at generation 4, so it has no generation-1 baseline to inherit: there is no -/// such thing as a generation-1 `_ckpt` object, and claiming one would say a generation-1 reader could -/// read it. Generation 5 re-keys it under `//` exactly like `REF_STREAM` above, for -/// the same reason and with the same floor. -constexpr FormatChangePoint REF_CKPT[] = { - {kContiguousRefStreamsGeneration, kContiguousRefStreamsGeneration}, - {kNamespaceLifeKeyedGeneration, kNamespaceLifeKeyedGeneration}, - {kOpaqueNamespaceLifeLayoutGeneration, kOpaqueNamespaceLifeLayoutGeneration}, - {kCommittedRefFrontierGeneration, kCommittedRefFrontierGeneration}, -}; - -/// `cas_ref_catalog` is BORN at generation 4, one generation BEFORE the bump that makes namespace -/// existence catalog-authoritative (Stage B's Task 4, "format bump B" -- `kNamespaceLifeKeyedGeneration`): -/// Task 2 introduced the catalog OBJECT while `G_BUILD` was still the value -/// `kContiguousRefStreamsGeneration` names, and Task 4 is the later change that actually wires -/// discovery to read it and bumps the floor. The catalog's own encoding is unaffected by that bump (it -/// reuses `kContiguousRefStreamsGeneration` as its birth generation, not a second constant named after -/// itself, for the same reason `REF_CKPT` originally did), so it carries no second change point here. -constexpr FormatChangePoint REF_CATALOG[] = {{kContiguousRefStreamsGeneration, kContiguousRefStreamsGeneration}}; -constexpr FormatChangePoint GC_MAINTENANCE_STATE[] = {{kUnifiedRefLifeFoldGeneration, kUnifiedRefLifeFoldGeneration}}; -constexpr FormatChangePoint POOL_META[] = { - {1, 1}, - {kPoolGcShardsGeneration, kPoolGcShardsGeneration}, - {kCommittedRefFrontierGeneration, kCommittedRefFrontierGeneration}, - {kMountWriteAttemptIdGeneration, kMountWriteAttemptIdGeneration}, -}; - -constexpr FormatChangePoint MOUNT_LEASE[] = { - {1, 1}, - {kMountWriteAttemptIdGeneration, kMountWriteAttemptIdGeneration}, -}; - } std::span changePoints(FormatId id) @@ -75,17 +36,11 @@ std::span changePoints(FormatId id) { case FormatId::RefLog: case FormatId::RefSnapshot: - return REF_STREAM; case FormatId::RefCkpt: - return REF_CKPT; case FormatId::RefCatalog: - return REF_CATALOG; case FormatId::GcMaintenanceState: - return GC_MAINTENANCE_STATE; case FormatId::PoolMeta: - return POOL_META; case FormatId::MountLease: - return MOUNT_LEASE; case FormatId::Blob: case FormatId::GcState: case FormatId::Roster: @@ -181,6 +136,18 @@ constexpr FormatTraits TRAITS[] = }; } +uint64_t casMaxStoredObjectBytes() +{ + static const uint64_t bound = [] + { + uint64_t largest_uncompressed = 0; + for (const FormatTraits & t : TRAITS) + largest_uncompressed = std::max(largest_uncompressed, t.object_cap); + return static_cast(ZSTD_compressBound(largest_uncompressed)); + }(); + return bound; +} + const FormatTraits & traitsFor(FormatId id) { for (const FormatTraits & t : TRAITS) @@ -197,6 +164,18 @@ const FormatTraits * traitsForType(std::string_view type) return nullptr; } +std::span allRegisteredFormatIds() +{ + static const auto ids = [] + { + std::array out{}; + for (size_t i = 0; i < std::size(TRAITS); ++i) + out[i] = TRAITS[i].id; + return out; + }(); + return ids; +} + std::string_view storedSuffix(FormatId id) { return traitsFor(id).compression == CompressionPolicy::Always ? ".zst" : ""; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.h index 1acc2b3925de..bfcb8cf652fb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasFormat.h @@ -15,87 +15,10 @@ namespace DB::Cas /// compatibility_version <= G_BUILD. Bump this (and append a change-point in CasFormat.cpp) when a new /// format generation is introduced. /// -/// Generation 2 is the first generation that understands mixed-algorithm pools: the schema-3 -/// source-edge settlement key includes the algorithm prefix, so a generation-1 reader can open the -/// pool but cannot decode its GC state. Pool admission CAS-raises `min_reader_generation` to this -/// build's own floor (`G_BUILD`), and a persisted floor above `G_BUILD` fails closed. -/// -/// Generation 3 replaced mutable ref-shard objects with immutable `_log` and `_snap` objects. -/// -/// Generation 4 makes each namespace's ref-log ids per-namespace and CONTIGUOUS within a writer epoch -/// (INV-1). The bytes of a `_log`/`_snap` object did not change, but their MEANING did: a generation-3 -/// pool's ids were drawn from a pool-wide counter and are full of legitimate holes, which this build -/// reads as a truncated -- i.e. corrupt -- stream. The per-object forward gate cannot reject such a -/// pool (its version is not in the future), so pool-meta decoding applies -/// `kContiguousRefStreamsGeneration` as a backward floor. Pools below the floor must be recreated; -/// there is no migration path in the pre-release format. -/// -/// Generation 5 (Stage B's own recreate-only bump, the plan's "format bump B") re-keys the ref layer -/// under `//` (spec INV-3: the whole-pool namespace catalog mints the incarnation). -/// Again the bytes of `_log`/`_snap`/`_ckpt` objects did not change, but the KEY SHAPE they live under -/// did: a generation-4 key named a namespace directly (`cas/refs//_log/`), while this -/// generation's reader recognizes only the incarnation-qualified shape -/// (`cas/refs///_log/`) -- `Layout::parseRefObjectKey`/`parseRefCkptKey` already -/// refuse the un-incarnated shape with `CORRUPTED_DATA` (Stage B Tasks 1/1c landed that refusal ahead -/// of this bump, deliberately: the pre-release format carries zero persisted data and zero compat -/// obligation, so the key shapes and the bump that makes them the ONLY readable shape need not land in -/// the same commit). `kNamespaceLifeKeyedGeneration` is the backward floor for this change, applied the -/// same way `kContiguousRefStreamsGeneration` is. -/// -/// Generation 6 replaces that namespace-bearing grammar with opaque pool-wide life identifiers and -/// splits hot ref streams from point-read state: `cas/ns/stream//...` contains `_log`, `_snap` -/// while `cas/ns/state//...` contains `_ckpt` and `_files`. A generation-5 -/// pool must be recreated; no dual parser or copy-forward path exists. -/// -/// Generation 7 replaces the fold seal's independent namespace-keyed coverage and cleanup -/// collections with one opaque-life-keyed row and removes the retired terminal-marker object class. A generation-6 -/// pool must be recreated; there is no dual reader for the split grammar. -/// -/// Generation 8 persists the creation-time `gc_shards` authority in `_pool_meta`. Generation-7 pools -/// must be recreated because namespace admission can precede creation of `gc/state`; accepting a -/// metadata object without this field would leave different openers charging different seal bounds. -/// Generation 9 adds `_ckpt.committed_through`, the exact recovery frontier. Generation-8 pools -/// must be recreated: the absence of this field has the incompatible meaning that no transaction has -/// entered durable logical history. Generation 10 adds the required `write_attempt_id` to mount -/// leases. Generation-9 pools must be recreated because a missing attempt identity makes ambiguous -/// mount writes impossible to distinguish from a different body under the same writer incarnation. -constexpr uint32_t G_BUILD = 10; - -/// The pool-format generation at which ref-log ids became per-namespace and contiguous. Pool metadata -/// below this value cannot be opened, because its ref streams carry holes this build reports as -/// corruption; the backward-floor check is applied by `decodePoolMeta`. Named separately from `G_BUILD` -/// so a later generation that CAN still read a generation-4 pool does not silently move the floor with -/// it. -constexpr uint32_t kContiguousRefStreamsGeneration = 4; - -/// The pool-format generation at which the ref layer (and, per Stage B's Task 4b, namespace files) -/// became incarnation-scoped under `//`. Pool metadata below this value cannot be -/// opened: its ref-object keys carry no incarnation segment, which this build's parsers refuse as -/// corruption rather than read as a compatibility case (see the `G_BUILD` doc above). The backward- -/// floor check is applied by `decodePoolMeta`, exactly mirroring `kContiguousRefStreamsGeneration`; -/// named separately for the same reason that one is -- so a later generation that can still read a -/// generation-5 pool does not silently move this floor with it. Pools below the floor must be -/// recreated; there is no migration path in the pre-release format. -constexpr uint32_t kNamespaceLifeKeyedGeneration = 5; - -/// The recreate-only generation at which namespace text disappeared from physical life keys and hot -/// ref streams were separated from point-read namespace state. -constexpr uint32_t kOpaqueNamespaceLifeLayoutGeneration = 6; - -/// The recreate-only generation at which one unified ref-life row replaced the split coverage and -/// namespace-cleanup grammar. -constexpr uint32_t kUnifiedRefLifeFoldGeneration = 7; - -/// The recreate-only generation at which `_pool_meta` became the authority for `gc_shards`. -constexpr uint32_t kPoolGcShardsGeneration = 8; - -/// The recreate-only generation at which `_ckpt` gained its exact committed-transaction frontier. -constexpr uint32_t kCommittedRefFrontierGeneration = 9; - -/// The recreate-only generation at which mount leases gained their required durable holder-write -/// identity. The pool-level reader floor rejects every older pool before it can interpret a mount -/// body without this field. -constexpr uint32_t kMountWriteAttemptIdGeneration = 10; +/// The generation history was reset to this {1, 1} baseline: CAS is pre-release, carries no persisted +/// data, and so pays no compatibility cost for starting the count over. Every class's `changePoints` +/// begins at generation 1 until a future change appends a real entry. +constexpr uint32_t G_BUILD = 1; /// Stable identifiers for every self-describing persisted object class. The text registry uses the /// corresponding `type` string as the on-disk identity. Numeric values are part of the format history: @@ -151,20 +74,20 @@ void checkCompatibility(uint32_t compatibility_version, std::string_view what); /// One append-only entry in a class's format history. At `generation`, the class's ENCODING or the /// MEANING of what it encodes changed, and a reader must understand at least `min_reader` to read an /// object written at that generation. Additive changes retain the previous reader floor; breaking -/// changes set the floor to the change generation itself. Generation 4's ref-stream entry is the -/// worked example of the second kind: not one byte of `cas_ref_log` moved, but its ids became dense, -/// so an older stream is unreadable to this build and the floor is the change generation. +/// changes set the floor to the change generation itself — even when not one byte of the encoding +/// moves: a change that makes ids dense, for example, leaves the bytes readable but their MEANING +/// unreadable to an older build, so the floor is the change generation. struct FormatChangePoint { uint16_t generation; uint16_t min_reader; }; -/// Returns the append-only change-point history for `id`, oldest first. A class's history begins at -/// the generation it was BORN in, not at 1: the classes that existed from the start carry the frozen -/// `{1, 1}` baseline, while `RefCkpt` — introduced at generation 4 — begins at `{4, 4}`, because there -/// is no such thing as a generation-1 `_ckpt` and claiming one would say a generation-1 reader could -/// read it. Future changes append entries without editing old ones. +/// Returns the append-only change-point history for `id`, oldest first. After the pre-release +/// generation reset every class carries the shared `{1, 1}` baseline. A class born LATER than the +/// current baseline must begin its history at its birth generation, not at 1 — claiming an earlier +/// entry would say an older reader could read an object kind that did not yet exist. Future changes +/// append entries without editing old ones. std::span changePoints(FormatId id); /// The text-format registry has one row per decodable persisted object. Each row is the single source @@ -202,6 +125,14 @@ const FormatTraits & traitsFor(FormatId id); /// Looks up a header-line `type` string. Returns nullptr for an unregistered type; it does not throw /// because callers use this result to classify the input before decoding it. const FormatTraits * traitsForType(std::string_view type); +std::span allRegisteredFormatIds(); + +/// The largest number of STORED bytes any materialized object may occupy: the largest whole-object +/// cap in the registry, expanded by zstd's worst-case bound because the `Always` policy stores those +/// objects compressed. A materialized read refuses anything larger rather than allocating it. A +/// format with no whole-object cap is streamed rather than materialized, so it does not raise this +/// bound. +uint64_t casMaxStoredObjectBytes(); /// Returns the storage-key suffix for `id`: `.zst` for `Always`, and an empty suffix otherwise. /// Key builders use this policy directly so a point lookup never has to inspect the object body or /// try multiple keys. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.cpp index c5dda3286ad6..cc13021b020d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcMaintenanceStateFormat.cpp @@ -13,6 +13,11 @@ namespace DB::ErrorCodes namespace DB::Cas { +namespace GcMaintenanceWire +{ + constexpr WireKey janitor_cursor{"janitor_cursor"}; +} + String encodeGcMaintenanceState(const GcMaintenanceState & state) { if (state.janitor_cursor.size() > kMaxGcMaintenanceCursorBytes) @@ -22,8 +27,7 @@ String encodeGcMaintenanceState(const GcMaintenanceState & state) CasJsonWriter out; writeHeaderLine(out, FormatId::GcMaintenanceState); bool first = true; - writeKey(out, "cur", first); - writeStringValue(out, state.janitor_cursor); + writeStringField(out, GcMaintenanceWire::janitor_cursor, state.janitor_cursor, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -46,7 +50,7 @@ GcMaintenanceState decodeGcMaintenanceState(std::string_view data) String key; while (reader.nextKey(key)) { - if (key == "cur") + if (key == GcMaintenanceWire::janitor_cursor) { result.janitor_cursor = reader.readString(); has_cursor = true; @@ -55,7 +59,7 @@ GcMaintenanceState decodeGcMaintenanceState(std::string_view data) reader.skipUnknown(key); } if (!has_cursor) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc maintenance state: missing cur"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc maintenance state: missing janitor_cursor"); if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc maintenance state: trailing bytes"); if (result.janitor_cursor.size() > kMaxGcMaintenanceCursorBytes) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.cpp index e69ea98a799e..2f9aa0e141d2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include @@ -18,27 +19,31 @@ namespace DB::Cas namespace { -std::string_view outcomeKindToWord(OutcomeKind o) +namespace GcOutcomesWire { - switch (o) - { - case OutcomeKind::Deleted: return "deleted"; - case OutcomeKind::Absent: return "absent"; - case OutcomeKind::Replaced: return "replaced"; - case OutcomeKind::Spared: return "spared"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS outcome log: unknown OutcomeKind {}", static_cast(o)); + constexpr WireKey kind{"kind"}; + constexpr WireKey outcome{"outcome"}; } -OutcomeKind outcomeKindFromWord(std::string_view w) +constexpr EnumWireTable kOutcomeKindWords{{{ + {OutcomeKind::Deleted, "deleted"}, + {OutcomeKind::Absent, "absent"}, + {OutcomeKind::Replaced, "replaced"}, + {OutcomeKind::Spared, "spared"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + +} + +std::string_view outcomeKindToWireWord(OutcomeKind outcome) { - if (w == "deleted") return OutcomeKind::Deleted; - if (w == "absent") return OutcomeKind::Absent; - if (w == "replaced") return OutcomeKind::Replaced; - if (w == "spared") return OutcomeKind::Spared; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS outcome log: unknown outcome '{}'", w); + return kOutcomeKindWords.toWord(outcome, "CAS outcome log outcome kind"); } +OutcomeKind outcomeKindFromWireWord(std::string_view w) +{ + return kOutcomeKindWords.fromWord(w, "CAS outcome log outcome kind"); } String encodeOutcomeLog(const OutcomeLog & log) @@ -48,12 +53,10 @@ String encodeOutcomeLog(const OutcomeLog & log) for (const OutcomeEntry & e : log.entries) { bool first = true; - writeKey(out, "k", first); - writeStringValue(out, objectKindToWord(e.kind)); - writeBlobRefFields(out, first, e.ref); /// ha + h - writeTokenFields(out, first, e.token); /// tt + tv - writeKey(out, "oc", first); - writeStringValue(out, outcomeKindToWord(e.outcome)); + writeWordField(out, GcOutcomesWire::kind, objectKindToWord(e.kind), first); + writeBlobRefFields(out, first, e.ref); /// algo + digest + writeTokenFields(out, first, e.token); /// token_type + token + writeWordField(out, GcOutcomesWire::outcome, outcomeKindToWireWord(e.outcome), first); closeObject(out, first); writeChar('\n', out); } @@ -68,14 +71,19 @@ OutcomeLog decodeOutcomeLog(std::string_view data) const uint64_t line_cap = traitsFor(FormatId::GcOutcomes).line_cap; OutcomeLog log; + /// One line scratch and one reader for the whole loop, as the other row decoders do: + /// rebuilding them per row costs an allocation per row for the seen-key store and the line. + String row_line; + JsonObjectReader row_reader; while (true) { - const String line = readLine(in, line_cap, "outcome log"); - ReadBufferFromMemory line_in(line.data(), line.size()); - JsonObjectReader r(line_in, KeyStrictness::Tolerant, "outcome log"); + readLineInto(in, row_line, line_cap, "outcome log"); + ReadBufferFromMemory line_in(row_line.data(), row_line.size()); + row_reader.reset(line_in, KeyStrictness::Tolerant, "outcome log"); + JsonObjectReader & r = row_reader; String key; - /// The first key distinguishes a trailer ("n") from a record ("k"). + /// The first key distinguishes a trailer (`n`) from a record (`kind`). if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS outcome log: empty line"); if (key == "n") @@ -92,34 +100,19 @@ OutcomeLog decodeOutcomeLog(std::string_view data) } OutcomeEntry e; - String ha; - String hhex; - String tv; - bool have_ha = false; - bool have_h = false; - bool have_tt = false; - TokenType tt{}; + BlobRefFields blob_ref_fields; + TokenFields token_fields; do { - if (key == "k") e.kind = objectKindFromWord(r.readString(), "outcome log"); - else if (key == "ha") { ha = r.readString(); have_ha = true; } - else if (key == "h") { hhex = r.readString(); have_h = true; } - else if (key == "tt") { tt = tokenTypeFromWord(r.readString(), "outcome log"); have_tt = true; } - else if (key == "tv") tv = r.readString(); - else if (key == "oc") e.outcome = outcomeKindFromWord(r.readString()); + if (key == GcOutcomesWire::kind) e.kind = objectKindFromWord(r.readString(), "outcome log"); + else if (matchBlobRefFields(key, r, blob_ref_fields)) {} + else if (matchTokenFields(key, r, token_fields)) {} + else if (key == GcOutcomesWire::outcome) e.outcome = outcomeKindFromWireWord(r.readString()); else r.skipUnknown(key); } while (r.nextKey(key)); - if (!have_ha || !have_h || !have_tt) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS outcome log: record missing ha/h/tt"); - const BlobHashAlgo algo = blobHashAlgoFromWord(ha, "outcome log"); - /// Validate the digest width before `fromHex`: a width mismatch must surface as the - /// CORRUPTED_DATA required for malformed serialized input, not fromHex's BAD_ARGUMENTS. - if (hhex.size() != blobHashLenFor(algo) * 2) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS outcome log: digest width {} does not match algo '{}'", hhex.size(), ha); - e.ref = BlobRef{algo, codecFor(algo).fromHex(hhex)}; - e.token = Token{tv, tt}; + e.ref = blob_ref_fields.build("outcome log"); + e.token = token_fields.build("outcome log"); if (!line_in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS outcome log: junk after record"); log.entries.push_back(std::move(e)); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.h index 09a850ee66ff..adc7bad57305 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcOutcomesFormat.h @@ -1,4 +1,5 @@ #pragma once +#include #include #include #include @@ -15,8 +16,8 @@ namespace DB::Cas /// `gc/gen/{g}/attempt/{a}/outcomes/{round}/{shard}`. It contains the results of exact-token deletes /// for entries that were already published as `delete_pending`, as well as candidates spared when /// the one-pass merge found a live in-degree. The log is written before the round's single state CAS; -/// `putIfAbsent` adopts an existing durable log on replay rather than treating a byte difference as -/// an error. The uncompressed payload is a header line, one flat JSON record per entry in insertion +/// An existing durable log is adopted on replay rather than treated as an error on a byte +/// difference. The uncompressed payload is a header line, one flat JSON record per entry in insertion /// order, and an `{"n":count}` trailer. `FormatId::GcOutcomes` stores the sealed payload in one zstd /// frame, so its object-storage key has the `.zst` suffix. enum class OutcomeKind : uint8_t @@ -27,6 +28,12 @@ enum class OutcomeKind : uint8_t Spared = 4, /// The merge found a positive in-degree, so the candidate was kept alive. }; +/// Canonical wire word for one `OutcomeKind`. +std::string_view outcomeKindToWireWord(OutcomeKind outcome); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +OutcomeKind outcomeKindFromWireWord(std::string_view w); + /// One observation about a blob incarnation considered by GC. `token` identifies the exact /// incarnation that GC examined, while `ref` identifies the content address; retaining both lets /// replay and inspection distinguish an absent object from a replacement that won a race with GC. @@ -34,7 +41,7 @@ struct OutcomeEntry { ObjectKind kind = ObjectKind::Blob; BlobRef ref{}; - Token token; + PersistedEtag token; OutcomeKind outcome = OutcomeKind::Spared; }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.cpp index 7012c6787f70..5cc268a4d9c3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.cpp @@ -16,6 +16,24 @@ namespace ErrorCodes namespace DB::Cas { +namespace GcStateWire +{ + constexpr WireKey round{"round"}; + constexpr WireKey gc_shards{"gc_shards"}; + constexpr WireKey snap_generation{"snap_generation"}; + constexpr WireKey snap_pruned_through{"snap_pruned_through"}; + constexpr WireKey snap_attempt{"snap_attempt"}; + constexpr WireKey manifest_sweep_cursor{"manifest_sweep_cursor"}; + constexpr WireKey lease_owner{"lease_owner"}; + constexpr WireKey lease_seq{"lease_seq"}; +} + +namespace GcHeartbeatWire +{ + constexpr WireKey owner{"owner"}; + constexpr WireKey hb_seq{"hb_seq"}; +} + String encodeGcState(const GcState & state) { if (state.gc_shards < 1) @@ -23,14 +41,14 @@ String encodeGcState(const GcState & state) CasJsonWriter out(256); writeHeaderLine(out, FormatId::GcState); bool first = true; - writeKey(out, "rnd", first); writeU64StringValue(out, state.round); - writeKey(out, "gcs", first); writeIntText(state.gc_shards, out); - writeKey(out, "sg", first); writeU64StringValue(out, state.snap_generation); - writeKey(out, "spt", first); writeU64StringValue(out, state.snap_pruned_through); - writeKey(out, "sa", first); writeU64StringValue(out, state.snap_attempt); - writeKey(out, "msc", first); writeStringValue(out, state.manifest_sweep_cursor); - writeKey(out, "lo", first); writeHex128Value(out, state.lease.owner); - writeKey(out, "ls", first); writeU64StringValue(out, state.lease.seq); + writeU64StringField(out, GcStateWire::round, state.round, first); + writeNumberField(out, GcStateWire::gc_shards, state.gc_shards, first); + writeU64StringField(out, GcStateWire::snap_generation, state.snap_generation, first); + writeU64StringField(out, GcStateWire::snap_pruned_through, state.snap_pruned_through, first); + writeU64StringField(out, GcStateWire::snap_attempt, state.snap_attempt, first); + writeStringField(out, GcStateWire::manifest_sweep_cursor, state.manifest_sweep_cursor, first); + writeHex128Field(out, GcStateWire::lease_owner, state.lease.owner, first); + writeU64StringField(out, GcStateWire::lease_seq, state.lease.seq, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -49,20 +67,32 @@ GcState decodeGcState(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "rnd") state.round = r.readU64String(); - else if (key == "gcs") { state.gc_shards = r.readU64Number(); saw_gcs = true; } - else if (key == "sg") state.snap_generation = r.readU64String(); - else if (key == "spt") state.snap_pruned_through = r.readU64String(); - else if (key == "sa") state.snap_attempt = r.readU64String(); - else if (key == "msc") state.manifest_sweep_cursor = r.readString(); - else if (key == "lo") state.lease.owner = r.readHex128(); - else if (key == "ls") state.lease.seq = r.readU64String(); - else r.skipUnknown(key); + if (key == GcStateWire::round) + state.round = r.readU64String(); + else if (key == GcStateWire::gc_shards) + { + state.gc_shards = r.readU64Number(); + saw_gcs = true; + } + else if (key == GcStateWire::snap_generation) + state.snap_generation = r.readU64String(); + else if (key == GcStateWire::snap_pruned_through) + state.snap_pruned_through = r.readU64String(); + else if (key == GcStateWire::snap_attempt) + state.snap_attempt = r.readU64String(); + else if (key == GcStateWire::manifest_sweep_cursor) + state.manifest_sweep_cursor = r.readString(); + else if (key == GcStateWire::lease_owner) + state.lease.owner = r.readHex128(); + else if (key == GcStateWire::lease_seq) + state.lease.seq = r.readU64String(); + else + r.skipUnknown(key); } - /// Fail closed on an absent gcs: the writer always emits it, so a missing key means a corrupt object. + /// Fail closed on an absent gc_shards: the writer always emits it, so a missing key means a corrupt object. /// Do NOT silently keep the struct default (1) — that would hide corruption (no-fallback principle). if (!saw_gcs) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc/state: missing gcs"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc/state: missing gc_shards"); if (state.gc_shards == 0) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc/state: gc_shards must be >= 1"); if (!body_in.eof() || !in.eof()) @@ -75,8 +105,8 @@ String encodeGcHeartbeat(const GcHeartbeat & hb) CasJsonWriter out(256); writeHeaderLine(out, FormatId::GcHeartbeat); bool first = true; - writeKey(out, "by", first); writeHex128Value(out, hb.owner); - writeKey(out, "seq", first); writeU64StringValue(out, hb.hb_seq); + writeHex128Field(out, GcHeartbeatWire::owner, hb.owner, first); + writeU64StringField(out, GcHeartbeatWire::hb_seq, hb.hb_seq, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -96,12 +126,12 @@ GcHeartbeat decodeGcHeartbeat(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "by") + if (key == GcHeartbeatWire::owner) { hb.owner = r.readHex128(); saw_by = true; } - else if (key == "seq") + else if (key == GcHeartbeatWire::hb_seq) { hb.hb_seq = r.readU64String(); saw_seq = true; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.h index 88bce088ed13..ca38e649b82c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasGcStateFormat.h @@ -45,7 +45,7 @@ String encodeGcState(const GcState & state); /// Decode a complete `cas_gc_state` text object. The header and size limits are checked before the /// body is parsed; unknown non-reserved fields are tolerated for forward evolution, but malformed -/// input, trailing bytes, a missing `gcs`, or a zero shard count raises `CORRUPTED_DATA` rather than +/// input, trailing bytes, a missing `gc_shards`, or a zero shard count raises `CORRUPTED_DATA` rather than /// falling back to a default state. GcState decodeGcState(std::string_view data); @@ -53,7 +53,7 @@ GcState decodeGcState(std::string_view data); /// independently of round progress, because its lease renewal counter can remain unchanged during a /// long fold. A follower that observes the heartbeat advance backs off from stealing the lease; this /// prevents mistaking an alive, mid-round leader for a stalled one. The value is persisted as the -/// versioned `cas_gc_hb` text object, whose body contains `by` and `seq` string values, replacing the +/// versioned `cas_gc_hb` text object, whose body contains `owner` and `hb_seq` string values, replacing the /// former unversioned 24-byte record. struct GcHeartbeat { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.cpp index 5bd928ec01a3..d3c92e767c12 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.cpp @@ -1,4 +1,5 @@ #include +#include #include #include #include @@ -67,14 +68,16 @@ std::optional Layout::parseBlobKey(std::string_view key) const if (shard.size() != 2 || hex.size() < 2 || shard != hex.substr(0, 2)) return std::nullopt; /// malformed shard/hex shape -- not ours - /// `` -> `BlobHashAlgo`: the small enum-value set makes a linear scan against - /// `blobHashAlgoName` (the ONE name authority) cheaper and safer than a second name table that - /// could drift from it. + /// `` -> `BlobHashAlgo` through the wire table itself, whose coverage is proven against + /// the enum at compile time. A hand-written candidate list here would be a second enumeration that + /// a new algorithm could silently outgrow: the parser would reject a segment the writer emits. + /// This path answers "is this key ours?", so an unknown segment is `nullopt` -- debris, not + /// corruption -- which is why it scans rather than calling the throwing `fromWord`. std::optional algo; - for (BlobHashAlgo candidate : {BlobHashAlgo::CityHash128, BlobHashAlgo::XXH3_128, BlobHashAlgo::Sha256}) - if (algo_name == blobHashAlgoName(candidate)) + for (const auto & entry : kBlobHashAlgoWords.entries) + if (algo_name == entry.word) { - algo = candidate; + algo = entry.value; break; } if (!algo) @@ -200,8 +203,8 @@ NamespaceLifePhysicalId Layout::namespaceLifePhysicalIdOf(std::string_view key, if (!incarnation) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "CasLayout: object '{}' names no life: '{}' is not 32 lower-case hex digits of a nonzero " - "life id. Generation-5 namespace-bearing pools are rejected by the pool-metadata format " - "gate before this generation-6 physical-key parser is reached", + "life id. Pools whose keys predate the opaque-life layout are rejected by the " + "pool-metadata format gate before this physical-key parser is reached", key, segment); return *incarnation; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h index b1c556c600a5..5ab57bb7854b 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -168,6 +169,18 @@ class Layout return namespaceStreamPrefix(ns_id) + "_snap/" + renderRefTxnId(id) + String(storedSuffix(FormatId::RefSnapshot)); } + /// The write-once forms of `refLogKey` and `refSnapshotKey`: the same strings, typed as keys a + /// precondition-free delete may take. Both objects are published with a create-only write at a + /// life-qualified key. + WriteOnceKey writeOnceRefLogKey(const NamespaceLifeId & ns_id, const RefTxnId & id) const + { + return WriteOnceKey(refLogKey(ns_id, id)); + } + WriteOnceKey writeOnceRefSnapshotKey(const NamespaceLifeId & ns_id, const RefTxnId & id) const + { + return WriteOnceKey(refSnapshotKey(ns_id, id)); + } + /// The life's checkpoint object (spec INV-4) at `/cas/ns/state//_ckpt`. Unlike /// immutable stream objects it is mutable (token-CAS), carries no transaction id, and therefore lives /// in the point/path-addressed state tree rather than a `_log`/`_snap` directory -- @@ -273,6 +286,13 @@ class Layout + manifestOrdinalFileName(id.ref.manifest_ordinal); } + /// The write-once form of `manifestKey`: a manifest is published with a create-only write at a key + /// whose epoch, build sequence and ordinal never repeat under one server root. + WriteOnceKey writeOnceManifestKey(const ManifestId & id) const + { + return WriteOnceKey(manifestKey(id)); + } + /// Inverse of `manifestKey`: parses `/cas/manifests//-/.zst`. /// Strict: rejects the old decimal directory shape (it is not two fixed-width hex fields joined by /// '-'), a missing namespace/build/ordinal segment, trailing garbage, a file not ending in the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.cpp index 5e7f9ff5ffbf..e0722add9bdc 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -22,41 +23,37 @@ namespace DB::Cas namespace { -std::string_view placementToWord(EntryPlacement p) +namespace PartManifestWire { - switch (p) - { - case EntryPlacement::Inline: return "inline"; - case EntryPlacement::Blob: return "blob"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: unknown placement {}", static_cast(p)); + constexpr WireKey ns{"namespace"}; + constexpr WireKey payload_digest{"payload_digest"}; + constexpr WireKey path{"path"}; + constexpr WireKey place{"place"}; + constexpr WireKey size{"size"}; } -EntryPlacement placementFromWord(std::string_view w) -{ - if (w == "inline") return EntryPlacement::Inline; - if (w == "blob") return EntryPlacement::Blob; - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: unknown placement '{}'", w); -} +constexpr EnumWireTable kEntryPlacementWords{{{ + {EntryPlacement::Inline, "inline"}, + {EntryPlacement::Blob, "blob"}, +}}}; + +static_assert(casEnumTableCoversEnum()); -/// One entry-record line: {"p","pm", then either the Blob's "ha"/"h"/"sz" or the Inline's "il"}. +/// One entry-record line: `path`/`place`, followed by either `algo`/`digest`/`size` for a Blob or +/// `size` for Inline bytes. void writeEntryRecord(CasJsonWriter & out, const ManifestEntry & e) { bool first = true; - writeKey(out, "p", first); - writeStringValue(out, e.path); - writeKey(out, "pm", first); - writeStringValue(out, placementToWord(e.placement)); + writeStringField(out, PartManifestWire::path, e.path, first); + writeWordField(out, PartManifestWire::place, entryPlacementToWireWord(e.placement), first); if (e.placement == EntryPlacement::Blob) { - writeBlobRefFields(out, first, e.ref); /// ha + h - writeKey(out, "sz", first); - writeIntText(e.blob_size, out); + writeBlobRefFields(out, first, e.ref); /// algo + digest + writeNumberField(out, PartManifestWire::size, e.blob_size, first); } else { - writeKey(out, "il", first); - writeIntText(e.inline_bytes.size(), out); + writeNumberField(out, PartManifestWire::size, e.inline_bytes.size(), first); } closeObject(out, first); writeChar('\n', out); @@ -72,7 +69,7 @@ String bannerFor(std::string_view path, uint64_t n) CasJsonWriter w(path.size() + 32); w.append("==> "); w.stringValue(path); - w.append(" il="); + w.append(" size="); w.u64Number(n); w.append(" <=="); return std::move(w).take(); @@ -80,6 +77,16 @@ String bannerFor(std::string_view path, uint64_t n) } +std::string_view entryPlacementToWireWord(EntryPlacement placement) +{ + return kEntryPlacementWords.toWord(placement, "PartManifest: EntryPlacement"); +} + +EntryPlacement entryPlacementFromWireWord(std::string_view w) +{ + return kEntryPlacementWords.fromWord(w, "PartManifest: EntryPlacement"); +} + String encodePartManifest(const PartManifest & m) { /// Canonical path order plus duplicate-path rejection makes the encoded record sequence @@ -97,15 +104,13 @@ String encodePartManifest(const PartManifest & m) CasJsonWriter out(256); writeHeaderLine(out, FormatId::PartManifest); - /// descriptor meta line: ManifestRef (me/mb/mo, shared rendering with refsnaplog) + root + /// descriptor meta line: ManifestRef (epoch/build/ord, shared rendering with refsnaplog) + root /// namespace + payload digest. { bool first = true; - writeManifestRefFields(out, first, "", m.ref); - writeKey(out, "ns", first); - writeStringValue(out, m.root_namespace_id.string()); - writeKey(out, "pd", first); - writeHex128Value(out, m.payload_digest); + writeManifestRefFields(out, first, kBareManifestRefKeys, m.ref); + writeStringField(out, PartManifestWire::ns, m.root_namespace_id.string(), first); + writeHex128Field(out, PartManifestWire::payload_digest, m.payload_digest, first); closeObject(out, first); writeChar('\n', out); } @@ -144,43 +149,44 @@ PartManifest decodePartManifest(std::string_view data) const String meta = readLine(in, line_cap, "cas_part_manifest"); ReadBufferFromMemory mm(meta.data(), meta.size()); JsonObjectReader r(mm, KeyStrictness::Tolerant, "cas_part_manifest"); - std::optional me; - std::optional mb; - std::optional mo; + ManifestRefFields fields; std::optional ns; std::optional pd; String key; while (r.nextKey(key)) { - if (key == "me") me = r.readU64String(); - else if (key == "mb") mb = r.readU64String(); - else if (key == "mo") mo = r.readU64Number(); - else if (key == "ns") ns = r.readString(); - else if (key == "pd") pd = r.readHex128(); + if (matchManifestRefFields(key, r, kBareManifestRefKeys, fields)) {} + else if (key == PartManifestWire::ns) ns = r.readString(); + else if (key == PartManifestWire::payload_digest) pd = r.readHex128(); else r.skipUnknown(key); } - if (!me || !mb || !mo) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: descriptor missing me/mb/mo"); if (!ns) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: descriptor missing ns"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: descriptor missing namespace"); if (!pd) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: descriptor missing pd"); - m.ref = manifestRefFromFields(*me, *mb, *mo, "PartManifest", "descriptor"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: descriptor missing payload_digest"); + m.ref = fields.buildRef("PartManifest", "descriptor"); m.root_namespace_id = RootNamespace(*ns); m.payload_digest = *pd; if (!mm.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: junk after descriptor line"); } - /// entry record lines, until the trailer. Inline entries remember their declared `il` length so + /// entry record lines, until the trailer. Inline entries remember their declared `size` so /// the payload zone below can read exactly that many raw bytes back into `inline_bytes`. /// Index-aligned with `m.entries` (Blob entries push an unused 0 placeholder). std::vector inline_lens; + String blob_ref_what; /// reused across Blob entries so the error context does not allocate per row + /// One line scratch and one reader for the whole loop: a decoder that rebuilds them per + /// row pays an allocation per row for the seen-key store and the line, which profiling put + /// at about a fifth of the instructions executed inside a row. + String row_line; + JsonObjectReader row_reader; while (true) { - const String line = readLine(in, line_cap, "cas_part_manifest"); - ReadBufferFromMemory l(line.data(), line.size()); - JsonObjectReader r(l, KeyStrictness::Tolerant, "cas_part_manifest"); + readLineInto(in, row_line, line_cap, "cas_part_manifest"); + ReadBufferFromMemory l(row_line.data(), row_line.size()); + row_reader.reset(l, KeyStrictness::Tolerant, "cas_part_manifest"); + JsonObjectReader & r = row_reader; String key; if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: empty line"); @@ -198,8 +204,8 @@ PartManifest decodePartManifest(std::string_view data) break; } - if (key != "p") - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: record must start with \"p\""); + if (key != PartManifestWire::path) + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: record must start with \"path\""); ManifestEntry e; e.path = r.readString(); @@ -218,48 +224,37 @@ PartManifest decodePartManifest(std::string_view data) } std::optional pm; - std::optional ha; - std::optional h; - std::optional sz; - std::optional il; + BlobRefFields blob_ref; + std::optional size; while (r.nextKey(key)) { - if (key == "pm") pm = r.readString(); - else if (key == "ha") ha = r.readString(); - else if (key == "h") h = r.readString(); - else if (key == "sz") sz = r.readU64Number(); - else if (key == "il") il = r.readU64Number(); + if (key == PartManifestWire::place) pm = r.readString(); + else if (matchBlobRefFields(key, r, blob_ref)) {} + else if (key == PartManifestWire::size) size = r.readU64Number(); else r.skipUnknown(key); } if (!l.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: junk after record"); if (!pm) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: entry '{}' missing pm", e.path); - e.placement = placementFromWord(*pm); + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: entry '{}' missing place", e.path); + e.placement = entryPlacementFromWireWord(*pm); if (e.placement == EntryPlacement::Blob) { - if (!ha || !h || !sz) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: blob entry '{}' missing ha/h/sz", e.path); - const BlobHashAlgo algo = blobHashAlgoFromWord(*ha, "PartManifest entry"); - /// Validate the digest width before calling `fromHex`. A width mismatch otherwise - /// produces `BAD_ARGUMENTS` instead of the `CORRUPTED_DATA` required for malformed - /// serialized input, allowing an invalid manifest to escape the decoder's fail-closed - /// error contract. - const uint64_t expected_hex_len = blobHashLenFor(algo) * 2; - if (h->size() != expected_hex_len) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "PartManifest: entry '{}' digest hex width {} does not match algo width {}", - e.path, h->size(), expected_hex_len); - e.ref = BlobRef{algo, codecFor(algo).fromHex(*h)}; - e.blob_size = *sz; + if (!size) + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: blob entry '{}' missing size", e.path); + blob_ref_what.assign("PartManifest entry '"); + blob_ref_what += e.path; + blob_ref_what += '\''; + e.ref = blob_ref.build(blob_ref_what); + e.blob_size = *size; inline_lens.push_back(0); /// unused for Blob; keeps inline_lens index-aligned with entries } else { - if (!il) - throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: inline entry '{}' missing il", e.path); - inline_lens.push_back(*il); /// bytes filled from the payload zone below + if (!size) + throw Exception(ErrorCodes::CORRUPTED_DATA, "PartManifest: inline entry '{}' missing size", e.path); + inline_lens.push_back(*size); /// bytes filled from the payload zone below } /// Canonical ascending-order and no-duplicate-path enforcement: compare only against the @@ -277,7 +272,7 @@ PartManifest decodePartManifest(std::string_view data) } /// payload zone: for each Inline entry, in the same order it appeared above, a banner line then - /// exactly `il` raw bytes then a terminating '\n'. + /// exactly `size` raw bytes then a terminating '\n'. for (size_t i = 0; i < m.entries.size(); ++i) { if (m.entries[i].placement != EntryPlacement::Inline) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.h index f2a416d15743..a26597e1bebe 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPartManifestFormat.h @@ -15,14 +15,14 @@ namespace DB::Cas /// stable for the surrounding CAS protocol. /// /// header line {"type":"cas_part_manifest","v":N} -/// descriptor meta line {"me","mb","mo"} (the ManifestRef, shared rendering with -/// refsnaplog, `CasWireVocab.h`) + "ns" (root namespace) + "pd" +/// descriptor meta line {"epoch","build","ord"} (the ManifestRef, shared rendering with +/// refsnaplog, `CasWireVocab.h`) + `root_namespace` + `payload_digest` /// (payload digest, 32 lowercase hex) -/// one entry-record line each {"p":path,"pm":placement-word, then either the Blob's -/// {"ha","h","sz"} or the Inline's {"il"}}, in canonical path order +/// one entry-record line each {"path":path,"place":placement-word, then either the Blob's +/// {"algo","digest","size"} or the Inline's {"size"}}, in canonical path order /// trailer line {"n":entry-count} /// PAYLOAD ZONE (raw, follows the trailer): for each Inline entry, in path order, a -/// `head -v`-style banner line `==> "" il= <==\n`, then +/// `head -v`-style banner line `==> "" size= <==\n`, then /// exactly `n` raw bytes, then `\n`. The path uses the same writer as /// the entry-record line, so decode can rebuild the banner byte-wise. /// Blob entries carry no @@ -42,6 +42,12 @@ enum class EntryPlacement : uint8_t Blob = 2, /// bytes stored as a content-addressed blob at `blobKey` }; +/// Canonical wire word for one manifest entry placement. +std::string_view entryPlacementToWireWord(EntryPlacement placement); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +EntryPlacement entryPlacementFromWireWord(std::string_view w); + /// One file entry inside a part manifest. `ref` is meaningful only for `Blob`; `inline_bytes` only /// for `Inline`. `blob_size` is the raw `Blob` byte count (0 for `Inline` — decode never fills it for /// an inline entry, since the wire format carries no redundant size for inline bytes). Use `size()` diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.cpp index 50e9843e9254..80f9572c4547 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.cpp @@ -1,8 +1,10 @@ +#include #include #include #include #include #include +#include namespace DB { @@ -16,29 +18,28 @@ namespace ErrorCodes namespace DB::Cas { -/// Minimum `blob_header_len` that provably fits the v3 `cas_blob` JSON envelope's mandatory (always- -/// written) non-ref fields, computed at type maxima from `encodeEnvelopeHeader` (CasBlobEnvelopeFormat.cpp): -/// {"type":"cas_blob" 18 -/// ,"v": 5 + 10 (currentCompatibilityVersion) 15 -/// ,"tag":"<32 hex>" 7 + 34 41 -/// ,"bld":"<32 hex>" 7 + 34 41 -/// ,"ts": 6 + 20 (created_at_ms) 26 -/// ,"by":"<32 hex>" 7 + 34 41 -/// ,"op":"" 6 + 10 (longest op word "mutation") 16 -/// ,"ch": 6 + 10 (VERSION_INTEGER) 16 -/// non-ref JSON = 214 bytes -/// The encoder then always frames the ref: `,"ref":` (7) + `""` (2) + `}` (1), and reserves byte -/// blob_header_len-1 for '\n' (1) = 11 bytes. So the mandatory content needs 214 + 11 = 225 bytes; -/// below that, encodeEnvelopeHeader throws LOGICAL_ERROR on the FIRST blob write (the old drop-and-retry -/// that used to mask this is gone). We floor at 240 (a multiple of 8 comfortably above 225, leaving -/// >= 15 bytes for the diagnostic ref even at type maxima, and well under the 256 default) so a -/// misconfigured pool fails at CREATION with BAD_ARGUMENTS, not at first write with LOGICAL_ERROR. -static constexpr uint64_t kMinBlobHeaderLen = 240; +namespace PoolMetaWire +{ + constexpr WireKey pool_id{"pool_id"}; + constexpr WireKey blob_header_len{"blob_header_len"}; + constexpr WireKey gc_shards{"gc_shards"}; + constexpr WireKey min_reader_generation{"min_reader_generation"}; + constexpr WireKey algos_used{"algos_used"}; +} + +/// Minimum `blob_header_len` that provably fits the `cas_blob` JSON envelope's mandatory-descriptor +/// worst case. The byte-for-byte derivation (`kMandatoryDescriptorWorstCase`, currently 239 bytes) lives +/// beside the envelope key constants in `CasBlobEnvelopeFormat.cpp`, next to the compile-time proof that +/// it fits under this floor; below that bound, `encodeEnvelopeHeader` throws `LOGICAL_ERROR` on the +/// FIRST blob write (the old drop-and-retry that used to mask this is gone). We floor at 240 (a +/// multiple of 8 comfortably above the worst case, leaving at least one byte for the diagnostic `ref` +/// even at type maxima, and well under the 256 default) so a misconfigured pool fails at CREATION with +/// `BAD_ARGUMENTS`, not at first write with `LOGICAL_ERROR`. void validatePoolBlobHeaderLen(uint64_t blob_header_len, int error_code, std::string_view what) { if (blob_header_len < kMinBlobHeaderLen) - throw Exception(error_code, "CAS {}: blob_header_len must be >= {} (v3 envelope minimum), got {}", + throw Exception(error_code, "CAS {}: blob_header_len must be >= {} (blob envelope minimum), got {}", what, kMinBlobHeaderLen, blob_header_len); if (blob_header_len % 8 != 0) throw Exception(error_code, "CAS {}: blob_header_len must be a multiple of 8, got {}", what, blob_header_len); @@ -52,14 +53,15 @@ void validatePoolAlgosUsed(const std::vector & algos_used, int error_co throw Exception(error_code, "CAS {}: algos_used must be non-empty", what); for (size_t i = 0; i < algos_used.size(); ++i) { - try - { - blobHashAlgoName(static_cast(algos_used[i])); - } - catch (const Exception &) - { + /// A direct membership scan, not `blobHashAlgoName`: that throws `LOGICAL_ERROR`, which + /// aborts at construction under a sanitizer/debug build before any catch can run, but + /// this function validates a raw byte vector, so it must reject cleanly rather than abort. + bool known = false; + for (const auto & entry : kBlobHashAlgoWords.entries) + if (static_cast(entry.value) == algos_used[i]) + known = true; + if (!known) throw Exception(error_code, "CAS {}: algos_used contains an unknown algo {}", what, algos_used[i]); - } if (i > 0 && algos_used[i] <= algos_used[i - 1]) throw Exception(error_code, "CAS {}: algos_used must be strictly sorted with no duplicates, got {} at index {} not after {}", @@ -69,30 +71,23 @@ void validatePoolAlgosUsed(const std::vector & algos_used, int error_co String encodePoolMeta(const PoolMeta & pm) { + validatePoolAlgosUsed(pm.algos_used, ErrorCodes::CORRUPTED_DATA, "pool meta"); + CasJsonWriter out(256); writeHeaderLine(out, FormatId::PoolMeta); bool first = true; - writeKey(out, "pid", first); - writeHex128Value(out, pm.pool_id); - writeKey(out, "hln", first); - writeIntText(pm.blob_header_len, out); - writeKey(out, "gcs", first); - writeIntText(pm.gc_shards, out); - writeKey(out, "mrg", first); - writeIntText(pm.min_reader_generation, out); - writeKey(out, "alg", first); - { - /// Comma-joined algo words (tiny list, <=3): "ch128" or "ch128,sha256". - String joined; - for (size_t i = 0; i < pm.algos_used.size(); ++i) - { - if (i != 0) - joined += ','; - joined += blobHashAlgoName(static_cast(pm.algos_used[i])); - } - writeStringValue(out, joined); - } + writeHex128Field(out, PoolMetaWire::pool_id, pm.pool_id, first); + writeNumberField(out, PoolMetaWire::blob_header_len, pm.blob_header_len, first); + writeNumberField(out, PoolMetaWire::gc_shards, pm.gc_shards, first); + writeNumberField(out, PoolMetaWire::min_reader_generation, pm.min_reader_generation, first); + /// Sized by the whole algo vocabulary and safe to index by `algos_used`: the validation above + /// admits only known algo bytes in strictly increasing order, so the vector cannot be longer + /// than the table. Relaxing that check to non-strict ordering would overrun this array. + std::array algo_words; + for (size_t i = 0; i < pm.algos_used.size(); ++i) + algo_words[i] = kBlobHashAlgoWords.toWord(static_cast(pm.algos_used[i]), "CAS pool meta"); + writeWordArrayField(out, PoolMetaWire::algos_used, std::span{algo_words}.first(pm.algos_used.size()), first); closeObject(out, first); writeChar('\n', out); @@ -104,18 +99,14 @@ PoolMeta decodePoolMeta(std::string_view data) ReadBufferFromMemory in(data.data(), data.size()); const TextHeader header = expectHeaderLine(in, FormatId::PoolMeta); - /// An older pool predates a breaking ref-layer change this build cannot reconcile, so - /// reject it before reading the metadata body. Writers always emit the current generation, while - /// `expectHeaderLine` separately rejects a future generation that this build cannot understand. - /// Generation 10 is the latest recreate-only authority floor and rejects old pools before any - /// mount lease body lacking its durable write-attempt identity can be interpreted. - if (header.v < kMountWriteAttemptIdGeneration) + /// The format-generation baseline is 1; a header below it cannot have been written by any build + /// this codec understands. `expectHeaderLine` above already rejects the symmetric FUTURE case + /// (`v > G_BUILD`); reject the backward case here, before the metadata body is read. + if (header.v < 1) throw Exception(ErrorCodes::UNKNOWN_FORMAT_VERSION, - "CAS pool format {} predates generation-10 mount-attempt-identity floor; recreate the pool. " - "This build requires the durable mount write attempt identity " - "in the generation-10 format " - "(generation {}+), and CAS is pre-release: there is no in-place migration.", - header.v, kMountWriteAttemptIdGeneration); + "CAS pool format {} predates the format-generation baseline; recreate the pool " + "(CAS is pre-release, so there is no in-place migration)", + header.v); const String body = readLine(in, traitsFor(FormatId::PoolMeta).line_cap, "pool meta"); ReadBufferFromMemory body_in(body.data(), body.size()); @@ -127,43 +118,32 @@ PoolMeta decodePoolMeta(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "pid") + if (key == PoolMetaWire::pool_id) { pm.pool_id = r.readHex128(); saw_pid = true; } - else if (key == "hln") + else if (key == PoolMetaWire::blob_header_len) pm.blob_header_len = r.readU64Number(); - else if (key == "gcs") + else if (key == PoolMetaWire::gc_shards) { pm.gc_shards = r.readU64Number(); saw_gc_shards = true; } - else if (key == "mrg") + else if (key == PoolMetaWire::min_reader_generation) pm.min_reader_generation = r.readU64Number(); - else if (key == "alg") + else if (key == PoolMetaWire::algos_used) { - const String joined = r.readString(); - size_t start = 0; - while (start <= joined.size()) - { - const size_t comma = joined.find(',', start); - const String word = joined.substr(start, comma == String::npos ? String::npos : comma - start); - if (word.empty()) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: empty algo word in '{}'", joined); + for (const String & word : r.readStringArray()) pm.algos_used.push_back(static_cast(blobHashAlgoFromWord(word, "pool meta algo"))); - if (comma == String::npos) - break; - start = comma + 1; - } } else r.skipUnknown(key); } if (!saw_pid) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: missing pid"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: missing pool_id"); if (!saw_gc_shards || pm.gc_shards == 0) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: missing or zero gcs"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: missing or zero gc_shards"); if (!body_in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS pool meta: junk after body object"); if (!in.eof()) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.h index 2ca0894d2f01..80acffecde30 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasPoolMetaFormat.h @@ -10,12 +10,13 @@ namespace DB::Cas { -class Backend; +class CasOperation; class Layout; /// `_pool_meta` — the pool identity and the pool-wide constants that every reader and writer must -/// agree on. The v3 text representation is a header line followed by one JSON body object: -/// {"pid":"<32hex>","hln":,"mrg":,"alg":""}. +/// agree on. The text representation is a header line followed by one JSON body object: +/// {"pool_id":"<32hex>","blob_header_len":,"gc_shards":, +/// "min_reader_generation":,"algos_used":["",...]}. /// /// The persisted object is authoritative after creation. On reopen, `createOrValidate` uses its /// `blob_header_len` and reader-generation floor rather than replacing them with local configuration; @@ -37,7 +38,7 @@ struct PoolMeta /// because changing it would move the blob payload offset for existing objects; a new hash algorithm /// is rejected unless `allow_new` is set, and concurrent admission is retried from fresh metadata. /// - /// `allow_mint` (spec §2 [C4][D2]) gates the create-if-absent path: minting a fresh `_pool_meta` is a + /// `allow_mint` gates the create-if-absent path: minting a fresh `_pool_meta` is a /// consequential write that establishes a brand-new pool identity, so it is permitted ONLY on the /// writable startup path that has just passed the zero-write residual proof (`Pool::open`). Every /// non-bootstrap caller — a read-only/observe open, `openForDecommission` — passes `false`; an absent @@ -49,25 +50,25 @@ struct PoolMeta /// pool-lifecycle entry point cannot silently re-arm the observe-mint footgun by omission. The two /// production callers pass it explicitly; only test minting sites opt in with `allow_mint=true`. static PoolMeta createOrValidate( - Backend &, const Layout &, uint64_t blob_header_len, uint64_t gc_shards, + CasOperation &, const Layout &, uint64_t blob_header_len, uint64_t gc_shards, BlobHashAlgo blob_hash_algo = BlobHashAlgo::CityHash128, bool allow_new = false, bool allow_mint = false); /// Convenience for single-shard callers. Production pool opening passes the configured value to /// the explicit overload above; this preserves compact single-shard codec/unit fixtures. static PoolMeta createOrValidate( - Backend & backend, const Layout & layout, uint64_t blob_header_len, + CasOperation & op, const Layout & layout, uint64_t blob_header_len, BlobHashAlgo blob_hash_algo = BlobHashAlgo::CityHash128, bool allow_new = false, bool allow_mint = false) { return createOrValidate( - backend, layout, blob_header_len, /*gc_shards=*/1, blob_hash_algo, allow_new, allow_mint); + op, layout, blob_header_len, /*gc_shards=*/1, blob_hash_algo, allow_new, allow_mint); } }; /// Serializes valid pool metadata as the versioned `_pool_meta` text object. The output includes the /// format header, one JSON body line, and its terminating newline; it is suitable for a conditional -/// backend write and preserves the sorted algorithm set as comma-separated vocabulary words. +/// backend write and preserves the sorted algorithm set as a JSON array of vocabulary words. String encodePoolMeta(const PoolMeta &); /// Parses and validates a persisted `_pool_meta` object. Unknown JSON keys are tolerated for additive @@ -76,11 +77,12 @@ String encodePoolMeta(const PoolMeta &); /// corruption or compatibility error code. PoolMeta decodePoolMeta(std::string_view); -/// Checks the fixed blob-envelope size invariant. The length must be 8-byte aligned, at most 16 KiB, -/// and at least 240 bytes: v3's mandatory envelope fields, framing, and newline consume 225 bytes at -/// type maxima, while 240 leaves room for a diagnostic `ref`. The caller supplies the error code so -/// persisted violations can be reported as `CORRUPTED_DATA` and bad creation arguments as -/// `BAD_ARGUMENTS`. +/// Checks the fixed blob-envelope size invariant: 8-byte aligned, at most 16 KiB, and at least +/// `kMinBlobHeaderLen`. That floor and the worst case it must clear are derived once beside the +/// envelope encoder, which also proves the relation at compile time — no number is restated here, +/// because a second copy is exactly what a single owner exists to prevent. The caller supplies the +/// error code so persisted violations can be reported as `CORRUPTED_DATA` and bad creation arguments +/// as `BAD_ARGUMENTS`. void validatePoolBlobHeaderLen(uint64_t blob_header_len, int error_code, std::string_view what); /// Checks that every admitted hash algorithm is known, that the set is non-empty, and that its numeric diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.cpp index b21458aaf6e2..78dedf30a4b2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include @@ -19,6 +20,32 @@ namespace DB::Cas namespace { +namespace RunWire +{ + constexpr WireKey ref{"ref"}; + constexpr WireKey src{"src"}; + constexpr WireKey mark{"mark"}; + constexpr WireKey pending{"pending"}; + constexpr WireKey size{"size"}; + constexpr WireKey condemn_round{"condemn_round"}; + constexpr WireKey confirmed{"confirmed"}; +} + +namespace RunHeaderWire +{ + constexpr WireKey type{"type"}; + constexpr WireKey version{"v"}; + constexpr WireKey kind{"kind"}; +} + +constexpr EnumWireTable kRunMarkerWords{{{ + {RunMarker::Zero, "zero"}, + {RunMarker::Edge, "edge"}, + {RunMarker::Condemned, "condemned"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + UInt128 toWideChecksum(CityHash_v1_0_2::uint128 h) { /// Keep the high and low halves in the same order for the write-side helper and the streaming @@ -34,20 +61,21 @@ int hexNibble(char c) return -1; } +/// The run `ref` carries the algorithm as a raw leading byte, so this is the byte-side counterpart of +/// the word table -- and it walks that same table rather than listing the enumerators again. A second +/// list is how the writer and the reader come to disagree about which algorithms exist: `renderB` +/// writes whatever the enum holds, and a hand-written switch here would reject exactly what a new +/// enumerator adds. BlobHashAlgo algoFromByte(uint8_t b, std::string_view what) { - switch (b) - { - case static_cast(BlobHashAlgo::CityHash128): return BlobHashAlgo::CityHash128; - case static_cast(BlobHashAlgo::XXH3_128): return BlobHashAlgo::XXH3_128; - case static_cast(BlobHashAlgo::Sha256): return BlobHashAlgo::Sha256; - default: - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown algo byte {} in record key", what, b); - } + for (const auto & entry : kBlobHashAlgoWords.entries) + if (static_cast(entry.value) == b) + return entry.value; + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown algo byte {} in record key", what, b); } -/// `b` = the algo byte as two lowercase hex chars, then the digest hex at the algo's width. The algo -/// byte leads so that string-sorting `b` reproduces the binary (algo, digest) byte order. +/// `ref` = the algo byte as two lowercase hex chars, then the digest hex at the algo's width. The +/// algo byte leads so that string-sorting `ref` reproduces the binary (algo, digest) byte order. String renderB(const BlobRef & ref) { static constexpr char H[] = "0123456789abcdef"; @@ -72,32 +100,25 @@ BlobRef parseB(std::string_view b) if (digest_hex.size() != static_cast(blobHashLenFor(algo)) * 2) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: digest hex width {} does not match algo width {}", digest_hex.size(), blobHashLenFor(algo) * 2); + for (const char c : digest_hex) + if (!isLowercaseHexChar(c)) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: non-lowercase-hex digest in record key"); BlobRef ref; ref.algo = algo; ref.digest = codecFor(algo).fromHex(String(digest_hex)); return ref; } -std::string_view markerToWord(char m) -{ - switch (m) - { - case kEdgeActive: return "edge"; - case kZeroMarker: return "zero"; - case kCondemned: return "condemned"; - default: - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: unknown row marker 0x{:02x}", static_cast(m)); - } } -char markerFromWord(std::string_view w) +std::string_view runMarkerToWireWord(RunMarker marker) { - if (w == "edge") return kEdgeActive; - if (w == "zero") return kZeroMarker; - if (w == "condemned") return kCondemned; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: unknown row marker '{}'", w); + return kRunMarkerWords.toWord(marker, "CAS cas_run: RunMarker"); } +RunMarker runMarkerFromWireWord(std::string_view w) +{ + return kRunMarkerWords.fromWord(w, "CAS cas_run: RunMarker"); } void writeRunHeaderLine(WriteBuffer & out, std::string_view kind) @@ -105,12 +126,9 @@ void writeRunHeaderLine(WriteBuffer & out, std::string_view kind) const FormatTraits & t = traitsFor(FormatId::RunFile); CasJsonWriter line(64); bool first = true; - writeKey(line, "type", first); - writeStringValue(line, t.type); - writeKey(line, "v", first); - writeIntText(currentCompatibilityVersion(), line); - writeKey(line, "kind", first); - writeStringValue(line, kind); + writeStringField(line, RunHeaderWire::type, t.type, first); + writeNumberField(line, RunHeaderWire::version, currentCompatibilityVersion(), first); + writeStringField(line, RunHeaderWire::kind, kind, first); closeObject(line, first); writeChar('\n', line); const std::string_view line_view = line.view(); @@ -174,23 +192,16 @@ void SourceEdgeRunWriter::append(const SourceEdgeRecord & rec) scratch.clear(); bool first = true; - writeKey(scratch, "b", first); - writeStringValue(scratch, renderB(rec.ref)); - writeKey(scratch, "s", first); - writeHex128Value(scratch, rec.source_id); - writeKey(scratch, "m", first); - writeStringValue(scratch, markerToWord(rec.marker)); - if (rec.marker == kCondemned) + writeStringField(scratch, RunWire::ref, renderB(rec.ref), first); + writeHex128Field(scratch, RunWire::src, rec.source_id, first); + writeWordField(scratch, RunWire::mark, runMarkerToWireWord(rec.marker), first); + if (rec.marker == RunMarker::Condemned) { - writeKey(scratch, "pend", first); - writeBoolValue(scratch, rec.delete_pending); - writeTokenFields(scratch, first, rec.token); /// tt + tv - writeKey(scratch, "sz", first); - writeIntText(rec.size, scratch); - writeKey(scratch, "cr", first); - writeU64StringValue(scratch, rec.condemn_round); - writeKey(scratch, "mc", first); - writeBoolValue(scratch, rec.marker_confirmed); + writeBoolField(scratch, RunWire::pending, rec.delete_pending, first); + writeTokenFields(scratch, first, rec.token); /// token_type + token + writeNumberField(scratch, RunWire::size, rec.size, first); + writeU64StringField(scratch, RunWire::condemn_round, rec.condemn_round, first); + writeBoolField(scratch, RunWire::confirmed, rec.marker_confirmed, first); } closeObject(scratch, first); writeChar('\n', scratch); @@ -235,9 +246,12 @@ bool SourceEdgeRunReader::next(SourceEdgeRecord & rec) if (done) return false; - const String line = readLine(hashing, traitsFor(FormatId::RunFile).line_cap, "cas_run"); - ReadBufferFromMemory line_in(line.data(), line.size()); - JsonObjectReader r(line_in, KeyStrictness::Strict, "cas_run"); + readLineInto(hashing, scratch, traitsFor(FormatId::RunFile).line_cap, "cas_run"); + ReadBufferFromMemory line_in(scratch.data(), scratch.size()); + /// Re-point the reader rather than building one per row: a fresh reader re-allocates its + /// seen-key store and value scratch every row, and this loop runs once per record. + reader.reset(line_in, KeyStrictness::Strict, "cas_run"); + JsonObjectReader & r = reader; String key; if (!r.nextKey(key)) @@ -263,41 +277,37 @@ bool SourceEdgeRunReader::next(SourceEdgeRecord & rec) SourceEdgeRecord out; String b; - String tv; - bool have_b = false; - bool have_s = false; - bool have_m = false; - bool have_pend = false; - bool have_tt = false; - bool have_tv = false; - bool have_sz = false; - bool have_cr = false; - bool have_mc = false; - TokenType tt{}; + TokenFields token_fields; + bool have_ref = false; + bool have_src = false; + bool have_mark = false; + bool have_pending = false; + bool have_size = false; + bool have_condemn_round = false; + bool have_confirmed = false; do { - if (key == "b") { b = r.readString(); have_b = true; } - else if (key == "s") { out.source_id = r.readHex128(); have_s = true; } - else if (key == "m") { out.marker = markerFromWord(r.readString()); have_m = true; } - else if (key == "pend") { out.delete_pending = r.readBool(); have_pend = true; } - else if (key == "tt") { tt = tokenTypeFromWord(r.readString(), "cas_run"); have_tt = true; } - else if (key == "tv") { tv = r.readString(); have_tv = true; } - else if (key == "sz") { out.size = r.readU64Number(); have_sz = true; } - else if (key == "cr") { out.condemn_round = r.readU64String(); have_cr = true; } - else if (key == "mc") { out.marker_confirmed = r.readBool(); have_mc = true; } + if (key == RunWire::ref) { b = r.readString(); have_ref = true; } + else if (key == RunWire::src) { out.source_id = r.readHex128(); have_src = true; } + else if (key == RunWire::mark) { out.marker = runMarkerFromWireWord(r.readString()); have_mark = true; } + else if (key == RunWire::pending) { out.delete_pending = r.readBool(); have_pending = true; } + else if (matchTokenFields(key, r, token_fields)) {} + else if (key == RunWire::size) { out.size = r.readU64Number(); have_size = true; } + else if (key == RunWire::condemn_round) { out.condemn_round = r.readU64String(); have_condemn_round = true; } + else if (key == RunWire::confirmed) { out.marker_confirmed = r.readBool(); have_confirmed = true; } else r.skipUnknown(key); /// Strict => any unknown key is CORRUPTED_DATA } while (r.nextKey(key)); - if (!have_b || !have_s || !have_m) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: record missing b/s/m"); + if (!have_ref || !have_src || !have_mark) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: record missing ref/src/mark"); out.ref = parseB(b); - if (out.marker == kCondemned) + if (out.marker == RunMarker::Condemned) { - if (!have_pend || !have_tt || !have_tv || !have_sz || !have_cr || !have_mc) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: condemned record missing pend/tt/tv/sz/cr/mc"); - out.token = Token{tv, tt}; + if (!have_pending || !have_size || !have_condemn_round || !have_confirmed) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: condemned record missing pending/size/condemn_round/confirmed"); + out.token = token_fields.build("cas_run"); } - else if (have_pend || have_tt || have_tv || have_sz || have_cr || have_mc) + else if (have_pending || token_fields.type_word || token_fields.value || have_size || have_condemn_round || have_confirmed) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_run: non-condemned record carries condemned fields"); if (!line_in.eof()) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.h index d5f9a4801caf..5741a61fc9b7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRecordStreamFormat.h @@ -1,7 +1,10 @@ #pragma once +#include #include #include +#include #include +#include #include #include #include @@ -11,6 +14,11 @@ #include #include +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + namespace DB::Cas { @@ -18,14 +26,30 @@ namespace DB::Cas /// format, shared by this codec and the GC fold that interprets the rows. /// /// Source-edge rows use `source_id == 0` as a sentinel key. A real active edge must never use that key; -/// both sentinel tags are restricted to it. `kZeroMarker` describes a zero transition for the current -/// generation and is dropped when the row is carried forward. `kCondemned` carries the condemned +/// both sentinel tags are restricted to it. `RunMarker::Zero` describes a zero transition for the current +/// generation and is dropped when the row is carried forward. `RunMarker::Condemned` carries the condemned /// incarnation at the sentinel key across generations until settlement; its payload contains the full /// deletion token and other condemned-row state. A condemned row subsumes the zero marker for that /// generation. -constexpr char kEdgeActive = 0x01; -constexpr char kZeroMarker = 0x00; -constexpr char kCondemned = 0x02; +enum class RunMarker : char +{ + Zero = 0x00, + Edge = 0x01, + Condemned = 0x02, +}; + +constexpr char runMarkerByte(RunMarker marker) +{ + return static_cast(marker); +} + +inline RunMarker runMarkerFromByte(char byte, std::string_view what) +{ + if (byte != runMarkerByte(RunMarker::Zero) && byte != runMarkerByte(RunMarker::Edge) + && byte != runMarkerByte(RunMarker::Condemned)) + throw Exception(ErrorCodes::CORRUPTED_DATA, "{}: unknown marker byte {}", what, static_cast(byte)); + return static_cast(byte); +} /// The `cas_run` codec represents the GC source-edge in-degree data plane as sorted NDJSON. This is /// the `RecordStream` family @@ -33,36 +57,37 @@ constexpr char kCondemned = 0x02; /// whole — streamed one line at a time over a `ReadBuffer`), `line_cap = 4 KiB`, `PinnedRaw` (no /// compression) + `Strict` (byte-deterministic for `putDeterministicArtifact` adoption). /// -/// This file is backend-free: it accepts caller-owned `ReadBuffer`/`WriteBuffer` objects and never -/// includes backend or GC subsystem headers. The GC layer owns the stream lifetime and the bridge to +/// This file is backend-free: it accepts caller-owned `ReadBuffer`/`WriteBuffer` objects and reaches +/// no backend or GC machinery -- `PersistedEtag` is a value type with no live backend behind +/// it, which is exactly why a persisted row may hold one. The GC layer owns the stream lifetime and the bridge to /// packed keys and condemned rows; this codec owns only the durable text representation and its /// identifier-layer types. Keeping that boundary physical prevents storage or GC dependencies from /// leaking into the format implementation. /// /// File shape: -/// {"type":"cas_run","v":3,"kind":"source_edge"} header line (type + v + kind gate) -/// {"b":"01","s":"<32hex>","m":"edge"} an active-edge / zero-marker row -/// {"b":"01","s":"00000000000000000000000000000000","m":"condemned","pend":false,"tt":"etag","tv":"...","sz":123,"cr":"7","mc":false} +/// {"type":"cas_run","v":1,"kind":"source_edge"} header line (type + v + kind gate) +/// {"ref":"01","src":"<32hex>","mark":"edge"} an active-edge / zero-marker row +/// {"ref":"01","src":"00000000000000000000000000000000","mark":"condemned","pending":false,"token_type":"etag","token":"...","size":123,"condemn_round":"7","confirmed":false} /// {"n":184267} trailer: record count /// -/// The record key `b` is the algo BYTE as two lowercase hex chars followed by the digest hex at the -/// algo's width; `s` is the 32-hex source id. String-sorting records by (b, s) reproduces the current +/// The record key `ref` is the algo BYTE as two lowercase hex chars followed by the digest hex at the +/// algo's width; `src` is the 32-hex source id. String-sorting records by (`ref`, `src`) reproduces the current /// `(algorithm, digest, source_id)` byte order (lowercase hex preserves unsigned byte order and the /// algorithm byte is emitted first) — the invariant the fold's two-cursor merge depends on. The row-tag word -/// `m` maps to the `kEdgeActive`/`kZeroMarker`/`kCondemned` bytes; a `condemned` row additionally -/// carries the retired incarnation (`pend`/`tt`/`tv`/`sz`/`cr`) and the durable condemn-marker -/// confirmation bit (`mc`). +/// `mark` maps to the `RunMarker` bytes; a `condemned` row additionally +/// carries the retired incarnation (`pending`/`token_type`/`token`/`size`/`condemn_round`) and the durable condemn-marker +/// confirmation bit (`confirmed`). /// One decoded source-edge row. All fields are identifier-layer types so the codec stays backend-free. /// The condemned-only fields (`delete_pending`/`token`/`size`/`condemn_round`/`marker_confirmed`) are -/// meaningful only when `marker == kCondemned`. +/// meaningful only when `marker == RunMarker::Condemned`. struct SourceEdgeRecord { BlobRef ref{}; UInt128 source_id{}; - char marker = kEdgeActive; + RunMarker marker = RunMarker::Edge; bool delete_pending = false; - Token token{}; + PersistedEtag token{}; uint64_t size = 0; uint64_t condemn_round = 0; bool marker_confirmed = false; /// durable Condemned meta confirmed for this entry (graduation gate) @@ -71,6 +96,12 @@ struct SourceEdgeRecord /// The header-line `kind` word for the only live `cas_run` kind. inline constexpr std::string_view kSourceEdgeKindWord = "source_edge"; +/// Canonical wire word for one source-edge run marker. +std::string_view runMarkerToWireWord(RunMarker marker); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +RunMarker runMarkerFromWireWord(std::string_view w); + /// Write the typed header line `{"type":"cas_run","v":G_BUILD,"kind":""}\n` with a fixed key /// order for byte-determinism. The `kind` field distinguishes the record schema within the run /// family, so a reader can reject a valid run of the wrong kind before interpreting any records. @@ -159,6 +190,12 @@ class SourceEdgeRunReader HashingReadBuffer hashing; uint64_t seen = 0; bool done = false; + /// Reused line scratch, mirroring the writer's: `readLineInto` clears it without releasing its + /// buffer, so a run of any length allocates only up to the longest line it has actually seen. + String scratch; + /// Reused object reader, for the same reason: its per-object buffers then cost one allocation + /// for the whole run rather than one per row. It starts unbound and every row re-points it. + JsonObjectReader reader; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.cpp index c5b119ba44ca..6cf44651508e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.cpp @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -20,30 +21,30 @@ namespace ErrorCodes namespace DB::Cas { -std::string_view nsStateToWord(NsState s) +namespace { - switch (s) - { - case NsState::Creating: return "creating"; - case NsState::Live: return "live"; - case NsState::Removing: return "removing"; - } - /// Every value reaching here came from a live `NsState` or from `nsStateFromWord`, which already - /// validated it on decode -- so this is a bug in THIS process, not corruption arriving from a - /// store. - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS ref catalog: unknown ns state {}", static_cast(s)); -} -NsState nsStateFromWord(std::string_view w) +namespace RefCatalogWire { - if (w == "creating") return NsState::Creating; - if (w == "live") return NsState::Live; - if (w == "removing") return NsState::Removing; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: unknown ns state '{}'", w); + constexpr WireKey kind{"kind"}; + constexpr WireKey ns{"ns"}; + constexpr WireKey state{"state"}; + constexpr WireKey life{"life"}; + constexpr WireKey remove_round{"remove_round"}; + constexpr WireKey creator{"creator"}; + constexpr WireKey creator_epoch{"creator_epoch"}; + constexpr WireKey creator_fence{"creator_fence"}; } -namespace -{ +constexpr std::string_view kEntryTag = "entry"; + +constexpr EnumWireTable kNsStateWords{{{ + {NsState::Creating, "creating"}, + {NsState::Live, "live"}, + {NsState::Removing, "removing"}, +}}}; + +static_assert(casEnumTableCoversEnum()); /// `creator` is required iff `state == Creating`, forbidden otherwise -- one predicate, used by both /// directions of the codec, so the writer's self-check and the reader's fail-close can never disagree. @@ -70,6 +71,16 @@ bool isCanonicalCatalogOrder(const std::vector & entries) } +std::string_view nsStateToWord(NsState s) +{ + return kNsStateWords.toWord(s, "CAS ref catalog"); +} + +NsState nsStateFromWord(std::string_view w) +{ + return kNsStateWords.fromWord(w, "CAS ref catalog ns state"); +} + String encodeRefCatalog(const RefCatalog & catalog) { const uint64_t line_cap = traitsFor(FormatId::RefCatalog).line_cap; @@ -136,22 +147,20 @@ String encodeRefCatalog(const RefCatalog & catalog) e.ns.string(), nsStateToWord(e.state), e.removal_started_round ? "carries" : "lacks"); bool first = true; - writeKey(out, "k", first); writeStringValue(out, "ent"); - writeKey(out, "ns", first); writeStringValue(out, e.ns.string()); - writeKey(out, "st", first); writeStringValue(out, nsStateToWord(e.state)); - writeKey(out, "inc", first); writeHex128Value(out, e.incarnation); + writeStringField(out, RefCatalogWire::kind, kEntryTag, first); + writeStringField(out, RefCatalogWire::ns, e.ns.string(), first); + writeStringField(out, RefCatalogWire::state, nsStateToWord(e.state), first); + writeHex128Field(out, RefCatalogWire::life, e.incarnation, first); if (e.removal_started_round) - { - writeKey(out, "rsr", first); writeU64StringValue(out, *e.removal_started_round); - } + writeU64StringField(out, RefCatalogWire::remove_round, *e.removal_started_round, first); if (e.creator) { - writeKey(out, "csr", first); writeStringValue(out, e.creator->server_root_id); - writeKey(out, "cwe", first); writeU64StringValue(out, e.creator->writer_epoch); - writeKey(out, "cfg", first); writeU64StringValue(out, e.creator->fence_generation); + writeStringField(out, RefCatalogWire::creator, e.creator->server_root_id, first); + writeU64StringField(out, RefCatalogWire::creator_epoch, e.creator->writer_epoch, first); + writeU64StringField(out, RefCatalogWire::creator_fence, e.creator->fence_generation, first); } closeObject(out, first); - closeLine("ent"); + closeLine("entry"); } const size_t trailer_start = out.size(); @@ -169,11 +178,17 @@ RefCatalog decodeRefCatalog(std::string_view data) RefCatalog catalog; uint64_t seen = 0; + /// One line scratch and one reader for the whole loop: a decoder that rebuilds them per + /// row pays an allocation per row for the seen-key store and the line, which profiling put + /// at about a fifth of the instructions executed inside a row. + String row_line; + JsonObjectReader row_reader; for (;;) { - const String line = readLine(in, line_cap, "ref catalog"); - ReadBufferFromMemory l(line.data(), line.size()); - JsonObjectReader r(l, KeyStrictness::Strict, "ref catalog"); + readLineInto(in, row_line, line_cap, "ref catalog"); + ReadBufferFromMemory l(row_line.data(), row_line.size()); + row_reader.reset(l, KeyStrictness::Strict, "ref catalog"); + JsonObjectReader & r = row_reader; String key; if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: empty line"); @@ -190,10 +205,10 @@ RefCatalog decodeRefCatalog(std::string_view data) "CAS ref catalog: trailer count {} != {} records", n, seen); return catalog; } - if (key != "k") - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: record must start with \"k\""); + if (key != RefCatalogWire::kind) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: record must start with \"kind\""); const String kind = r.readString(); - if (kind != "ent") + if (kind != kEntryTag) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: unknown record kind '{}'", kind); String ns_str; @@ -205,20 +220,20 @@ RefCatalog decodeRefCatalog(std::string_view data) std::optional removal_started_round; while (r.nextKey(key)) { - if (key == "ns") ns_str = r.readString(); - else if (key == "st") st_word = r.readString(); - else if (key == "inc") inc = r.readHex128(); - else if (key == "csr") csr = r.readString(); - else if (key == "cwe") cwe = r.readU64String(); - else if (key == "cfg") cfg = r.readU64String(); - else if (key == "rsr") removal_started_round = r.readU64String(); - else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: unknown ent key '{}'", key); + if (key == RefCatalogWire::ns) ns_str = r.readString(); + else if (key == RefCatalogWire::state) st_word = r.readString(); + else if (key == RefCatalogWire::life) inc = r.readHex128(); + else if (key == RefCatalogWire::creator) csr = r.readString(); + else if (key == RefCatalogWire::creator_epoch) cwe = r.readU64String(); + else if (key == RefCatalogWire::creator_fence) cfg = r.readU64String(); + else if (key == RefCatalogWire::remove_round) removal_started_round = r.readU64String(); + else throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: unknown entry key '{}'", key); } if (!l.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: junk after record"); if (!st_word) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: entry '{}' missing st", ns_str); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: entry '{}' missing state", ns_str); const NsState state = nsStateFromWord(*st_word); /// throws CORRUPTED_DATA on an unknown word /// A missing "ns" key reads as the same empty string a present-but-empty one would, and both @@ -234,7 +249,7 @@ RefCatalog decodeRefCatalog(std::string_view data) ns_str, ns_str.size(), kMaxNamespaceBytes); if (!inc) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: entry '{}' missing inc", ns_str); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: entry '{}' missing life", ns_str); if (*inc == 0) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref catalog: namespace '{}' has a zero incarnation -- 0 never names a life", ns_str); @@ -315,7 +330,7 @@ uint64_t worstCaseEntryFoldReservationBytes() /// coverage record plus terminal cleanup evidence, all numeric fields at maximum width. seal.ref_lives[std::numeric_limits::max()] = RefLifeFoldState{ .coverage = RefCoverage{ - .classification = 4, + .classification = CoverageClass::Clamped, .last_folded_ref_id = RefTxnId{kU64Max, kU64Max}, .hold = RefHold{.reason = HoldReason::UnconsumedSealCrossing, .offending_position = RefTxnId{kU64Max, kU64Max}, @@ -340,7 +355,7 @@ uint64_t widestBlobTargetRunReservationBytes(const Layout & layout, uint64_t gc_ .key = layout.blobTargetRunKey(max, max, gc_shards - 1, 0), .checksum = std::numeric_limits::max(), .shard = gc_shards - 1, - .generation = max}); + .key_generation = max}); return encodeFoldSeal(seal).size() - encodeFoldSeal(CasFoldSeal{}).size(); } @@ -368,7 +383,7 @@ void checkFoldSealReservation( /// wrap to a remainder far smaller than the true reservation, which would answer "fits" for an /// `entry_count` that plainly does not. const uint64_t ref_lives = mulByteBudget(entry_count, worstCaseEntryFoldReservationBytes()); - /// `validateFoldSealStructure` permits at most one canonical seq-0 `btr` per shard, so charging + /// `validateFoldSealStructure` permits at most one canonical seq-0 `blob_run` per shard, so charging /// one widest row for every shard covers the full legal run domain without per-entry arithmetic. const uint64_t blob_target_runs = mulByteBudget( gc_shards, widestBlobTargetRunReservationBytes(layout, gc_shards)); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.h index ca1fbe5a6ddc..0e93845731f4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCatalogFormat.h @@ -16,7 +16,7 @@ class Layout; /// The byte bound every namespace name admitted into `ref_catalog` must satisfy (spec INV-3: /// "namespace names get a byte bound"). It keeps the catalog's operator-visible row and line grammar /// bounded, and both directions of the codec enforce it. Logical namespace bytes do NOT enter -/// predicate (2): fold-seal `rfl` rows are keyed only by the fixed-width opaque life id. +/// predicate (2): fold-seal `ref_life` rows are keyed only by the fixed-width opaque life id. constexpr size_t kMaxNamespaceBytes = 512; /// One namespace's catalog lifecycle state (spec INV-3, §3). `Creating` blocks publication and @@ -44,7 +44,7 @@ std::string_view nsStateToWord(NsState s); /// Inverse of `nsStateToWord`; throws `CORRUPTED_DATA` for anything but the three registered words. NsState nsStateFromWord(std::string_view w); -/// The fence identity of the mounted writer CREATING one namespace (spec §3): the server root plus +/// The fence identity of the mounted writer CREATING one namespace: the server root plus /// the writer epoch and admission fence generation captured at the moment `Creating` was minted. It /// is what a reconciler compares against `CasServerRoot`'s liveness/fence machinery before a stalled /// `Creating` entry may be CAS-reconciled away (INV-3: "stalled creators occupy entries until @@ -92,7 +92,7 @@ struct RefCatalog bool operator==(const RefCatalog &) const = default; }; -/// Encodes `catalog` as the canonical `cas_ref_catalog` text object: a header line, one "ent" record +/// Encodes `catalog` as the canonical `cas_ref_catalog` text object: a header line, one "entry" record /// per entry in canonical (ns-sorted) order, and a record-count trailer -- the same tagged-record /// container `encodeFoldSeal` uses. Enforces the FULL strict grammar on the way out: canonical order /// and no duplicate namespace, a non-empty namespace within the `kMaxNamespaceBytes` bound, nonzero @@ -103,8 +103,8 @@ struct RefCatalog /// instead) -- but deliberately does NOT enforce the whole-object cap itself: that predicate must /// name the namespace under admission, which only a caller of `checkCatalogAdmission` knows. /// -/// These bytes go to and come from the backend DIRECTLY, exactly like `cas_ref_ckpt`: the Pool-side -/// `CasRefCatalog::read`/`casUpdateImpl` (`Pool/CasRefCatalog.cpp`) bypass `sealObject`/`openObject`, +/// These bytes go to and come from the backend DIRECTLY: the catalog read and update paths bypass +/// `sealObject`/`openObject`, /// which are the identity under this class's `CompressionPolicy::Never` and would add nothing. A /// policy flip to `Always` therefore breaks this silently -- and is caught, because `storedSuffix` /// would stop being empty and the registry test asserting `storedSuffix(FormatId::RefCatalog) == ""` @@ -144,7 +144,7 @@ uint64_t widestCondemnedSummaryReservationBytes(uint64_t gc_shards); /// PRE-PUT GATE, predicate (2) of INV-3's additive admission. Reserves the widest fixed frame, one /// widest ref-life row per candidate catalog entry, and one widest blob-target plus condemned-summary -/// row per authoritative GC shard. The `btr` multiplier follows the authoritative fold-seal grammar: +/// row per authoritative GC shard. The `blob_run` multiplier follows the authoritative fold-seal grammar: /// at most one canonical sequence-0 run is legal for each shard. Equality is accepted; refuses /// (`LIMIT_EXCEEDED`, naming `ns`) one entry over. Every multiplication and addition saturates, so an /// unreachable-in-practice count can never wrap into something that reads as "fits". diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.cpp index 6ff7fa5dda43..4034f5c244c1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefCkptFormat.cpp @@ -15,6 +15,22 @@ namespace ErrorCodes namespace DB::Cas { +namespace +{ + +namespace RefCkptWire +{ + constexpr WireKey life_epoch{"life_epoch"}; + constexpr WireKey committed_epoch{"committed_epoch"}; + constexpr WireKey committed_seq{"committed_seq"}; + constexpr WireKey snapshot_epoch{"snapshot_epoch"}; + constexpr WireKey snapshot_seq{"snapshot_seq"}; + constexpr WireKey seal_epoch{"seal_epoch"}; + constexpr WireKey seal_seq{"seal_seq"}; +} + +} + void checkRefCkptInvariants(const RefCkpt & ckpt, std::string_view what) { /// PRESENT means REAL. `life_epoch` may be absent (no writer of this object knew the namespace's @@ -89,16 +105,13 @@ String encodeRefCkpt(const RefCkpt & ckpt) /// written by the one shared `RefTxnId` writer the `_log` and `_snap` formats also use, so the /// three ref formats cannot disagree on the encoding. if (ckpt.life_epoch) - { - writeKey(out, "le", first); - writeU64StringValue(out, *ckpt.life_epoch); - } + writeU64StringField(out, RefCkptWire::life_epoch, *ckpt.life_epoch, first); if (ckpt.committed_through) - writeRefTxnIdFields(out, first, "cte", "cts", *ckpt.committed_through); + writeRefTxnIdFields(out, first, RefCkptWire::committed_epoch, RefCkptWire::committed_seq, *ckpt.committed_through); if (ckpt.checkpoint_snapshot_id) - writeRefTxnIdFields(out, first, "cse", "css", *ckpt.checkpoint_snapshot_id); + writeRefTxnIdFields(out, first, RefCkptWire::snapshot_epoch, RefCkptWire::snapshot_seq, *ckpt.checkpoint_snapshot_id); if (ckpt.last_epoch_seal) - writeRefTxnIdFields(out, first, "lse", "lss", *ckpt.last_epoch_seal); + writeRefTxnIdFields(out, first, RefCkptWire::seal_epoch, RefCkptWire::seal_seq, *ckpt.last_epoch_seal); closeObject(out, first); writeChar('\n', out); @@ -127,22 +140,22 @@ RefCkpt decodeRefCkpt(std::string_view data) JsonObjectReader r(body_in, KeyStrictness::Strict, "cas_ref_ckpt"); RefCkpt ckpt; - std::optional cse; - std::optional css; - std::optional lse; - std::optional lss; - std::optional cte; - std::optional cts; + std::optional snapshot_epoch; + std::optional snapshot_seq; + std::optional seal_epoch; + std::optional seal_seq; + std::optional committed_epoch; + std::optional committed_seq; String key; while (r.nextKey(key)) { - if (key == "le") ckpt.life_epoch = r.readU64String(); - else if (key == "cte") cte = r.readU64String(); - else if (key == "cts") cts = r.readU64String(); - else if (key == "cse") cse = r.readU64String(); - else if (key == "css") css = r.readU64String(); - else if (key == "lse") lse = r.readU64String(); - else if (key == "lss") lss = r.readU64String(); + if (key == RefCkptWire::life_epoch) ckpt.life_epoch = r.readU64String(); + else if (key == RefCkptWire::committed_epoch) committed_epoch = r.readU64String(); + else if (key == RefCkptWire::committed_seq) committed_seq = r.readU64String(); + else if (key == RefCkptWire::snapshot_epoch) snapshot_epoch = r.readU64String(); + else if (key == RefCkptWire::snapshot_seq) snapshot_seq = r.readU64String(); + else if (key == RefCkptWire::seal_epoch) seal_epoch = r.readU64String(); + else if (key == RefCkptWire::seal_seq) seal_seq = r.readU64String(); else r.skipUnknown(key); } @@ -151,23 +164,23 @@ RefCkpt decodeRefCkpt(std::string_view data) /// deletable" today and as "recovery has no base" tomorrow -- both of which a reader would trust. /// Fail closed instead. (A missing whole field is a legitimate absence, not truncation: every field /// of this object is optional, so there is nothing to miss.) - if (cse || css) + if (snapshot_epoch || snapshot_seq) { - if (!cse || !css) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: checkpoint_snapshot_id needs both cse and css"); - ckpt.checkpoint_snapshot_id = RefTxnId{*cse, *css}; + if (!snapshot_epoch || !snapshot_seq) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: checkpoint_snapshot_id needs both snapshot_epoch and snapshot_seq"); + ckpt.checkpoint_snapshot_id = RefTxnId{*snapshot_epoch, *snapshot_seq}; } - if (cte || cts) + if (committed_epoch || committed_seq) { - if (!cte || !cts) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: committed_through needs both cte and cts"); - ckpt.committed_through = RefTxnId{*cte, *cts}; + if (!committed_epoch || !committed_seq) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: committed_through needs both committed_epoch and committed_seq"); + ckpt.committed_through = RefTxnId{*committed_epoch, *committed_seq}; } - if (lse || lss) + if (seal_epoch || seal_seq) { - if (!lse || !lss) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: last_epoch_seal needs both lse and lss"); - ckpt.last_epoch_seal = RefTxnId{*lse, *lss}; + if (!seal_epoch || !seal_seq) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: last_epoch_seal needs both seal_epoch and seal_seq"); + ckpt.last_epoch_seal = RefTxnId{*seal_epoch, *seal_seq}; } if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS cas_ref_ckpt: trailing bytes"); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.cpp index be7ee5567575..8c9d010ee678 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -20,28 +21,27 @@ namespace DB::Cas namespace { -std::string_view opKindToWord(RefOpKind k) +namespace RefLogWire { - switch (k) - { - case RefOpKind::NamespaceBirth: return "namespace_birth"; - case RefOpKind::OwnerTransition: return "owner_transition"; - case RefOpKind::SetPublishedAt: return "set_published_at"; - case RefOpKind::RemoveNamespace: return "remove_namespace"; - case RefOpKind::EpochSeal: return "epoch_seal"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: unknown op kind {}", static_cast(k)); + constexpr WireKey ns{"namespace"}; + constexpr WireKey txn_epoch{"txn_epoch"}; + constexpr WireKey txn_seq{"txn_seq"}; + constexpr WireKey prev_epoch{"!prev_epoch"}; + constexpr WireKey prev_seq{"!prev_seq"}; + constexpr WireKey op{"op"}; + constexpr WireKey ref{"ref"}; + constexpr WireKey published_ms{"published_ms"}; } -RefOpKind opKindFromWord(std::string_view w) -{ - if (w == "namespace_birth") return RefOpKind::NamespaceBirth; - if (w == "owner_transition") return RefOpKind::OwnerTransition; - if (w == "set_published_at") return RefOpKind::SetPublishedAt; - if (w == "remove_namespace") return RefOpKind::RemoveNamespace; - if (w == "epoch_seal") return RefOpKind::EpochSeal; - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: unknown op kind '{}'", w); -} +constexpr EnumWireTable kRefOpWords{{{ + {RefOpKind::NamespaceBirth, "namespace_birth"}, + {RefOpKind::OwnerTransition, "owner_transition"}, + {RefOpKind::SetPublishedAt, "set_published_at"}, + {RefOpKind::RemoveNamespace, "remove_namespace"}, + {RefOpKind::EpochSeal, "epoch_seal"}, +}}}; + +static_assert(casEnumTableCoversEnum()); /// Byte budget over the encoded text. A removal-class transaction uses the larger complete-table /// budget and has neither an op-count nor a per-op cap; normal transactions are bounded by @@ -69,22 +69,10 @@ void checkBudget(const std::vector & ops, size_t encoded_bytes) } } -void writeBindingFields(CasJsonWriter & out, bool & first, std::string_view prefix, const RefOwnerBinding & b) -{ - checkCanonicalRefName(b.ref_name, "RefLogTxn", "owner binding ref_name"); - checkManifestRef(b.manifest_ref, "RefLogTxn", "owner binding manifest_ref"); - out.key(prefix, "bk", first); - writeStringValue(out, refOwnerKindToWord(b.kind)); - out.key(prefix, "rn", first); - writeStringValue(out, b.ref_name); - writeManifestRefFields(out, first, prefix, b.manifest_ref); -} - void writeOp(CasJsonWriter & out, const RefOp & op) { bool first = true; - writeKey(out, "op", first); - writeStringValue(out, opKindToWord(op.kind)); + writeWordField(out, RefLogWire::op, refOpKindToWireWord(op.kind), first); switch (op.kind) { case RefOpKind::NamespaceBirth: @@ -93,78 +81,59 @@ void writeOp(CasJsonWriter & out, const RefOp & op) break; case RefOpKind::OwnerTransition: if (op.old_binding) - writeBindingFields(out, first, "o", *op.old_binding); + writeBindingFields(out, first, kOldBindingKeys, *op.old_binding); if (op.new_binding) - writeBindingFields(out, first, "n", *op.new_binding); + writeBindingFields(out, first, kNewBindingKeys, *op.new_binding); break; case RefOpKind::SetPublishedAt: checkCanonicalRefName(op.ref_name, "RefLogTxn", "set_published_at ref_name"); checkManifestRef(op.expected_manifest_ref, "RefLogTxn", "set_published_at manifest_ref"); - writeKey(out, "rn", first); - writeStringValue(out, op.ref_name); - writeManifestRefFields(out, first, "", op.expected_manifest_ref); - writeKey(out, "ts", first); - writeIntText(op.published_at_ms, out); + writeStringField(out, RefLogWire::ref, op.ref_name, first); + writeManifestRefFields(out, first, kBareManifestRefKeys, op.expected_manifest_ref); + writeNumberField(out, RefLogWire::published_ms, op.published_at_ms, first); break; } closeObject(out, first); writeChar('\n', out); } -/// Collector for a ManifestRef's three flat fields under an optional prefix. -struct ManifestFields -{ - std::optional me; - std::optional mb; - std::optional mo; - - bool any() const { return me || mb || mo; } - ManifestRef build(std::string_view what) const - { - if (!me || !mb || !mo) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: {} manifest_ref missing me/mb/mo", what); - return manifestRefFromFields(*me, *mb, *mo, "RefLogTxn", what); - } -}; - /// Collector for one binding (old/new) under a prefix. struct BindingFields { - std::optional bk; - std::optional rn; - ManifestFields mf; + std::optional kind; + std::optional ref; + ManifestRefFields manifest_fields; - bool any() const { return bk || rn || mf.any(); } + bool any() const { return kind || ref || manifest_fields.any(); } RefOwnerBinding build(std::string_view what) const { - if (!bk || !rn) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: {} binding missing bk/rn", what); + if (!kind || !ref) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: {} binding missing kind/ref", what); RefOwnerBinding b; - b.kind = refOwnerKindFromWord(*bk, "RefLogTxn owner binding"); - b.ref_name = *rn; + b.kind = refOwnerKindFromWord(*kind, "RefLogTxn owner binding"); + b.ref_name = *ref; checkCanonicalRefName(b.ref_name, "RefLogTxn", "owner binding ref_name"); - b.manifest_ref = mf.build(what); + b.manifest_ref = manifest_fields.buildRef("RefLogTxn", what); return b; } }; -/// The log transaction's header-object meta line (ns + txn_id + the optional `prev_epoch_seal` +/// The log transaction's header-object meta line (`namespace` + txn_id + the optional `prev_epoch_seal` /// chain). Shared by `encodeRefLogTxn` and `removalFramingSize` so the two never disagree by a byte; /// `removalFramingSize` always passes `std::nullopt` -- a removal transaction is never a sequence-1 -/// epoch-transition record. Additive: the `"!pse"`/`"!pss"` pair is emitted only when +/// epoch-transition record. Additive: the `"!prev_epoch"`/`"!prev_seq"` pair is emitted only when /// `prev_epoch_seal` is set, so a body without it is byte-identical to the pre-EpochSeal wire shape. /// `!`-prefixed: `prev_epoch_seal` is INV-2 chain evidence, not cosmetic metadata -- a decoder that /// doesn't understand it must refuse the object rather than silently drop the chain link while /// otherwise passing the structural grammar (`JsonObjectReader::skipUnknown` rejects any unrecognized -/// `!`-key with `UNKNOWN_FORMAT_VERSION`, tolerant or not; see task-1 review finding M4). +/// `!`-key with `UNKNOWN_FORMAT_VERSION`, tolerant or not). void writeLogMeta(CasJsonWriter & out, const String & ns, const RefTxnId & txn_id, const std::optional & prev_epoch_seal) { bool first = true; - writeKey(out, "ns", first); - writeStringValue(out, ns); - writeRefTxnIdFields(out, first, "we", "rs", txn_id); + writeStringField(out, RefLogWire::ns, ns, first); + writeRefTxnIdFields(out, first, RefLogWire::txn_epoch, RefLogWire::txn_seq, txn_id); if (prev_epoch_seal) - writeRefTxnIdFields(out, first, "!pse", "!pss", *prev_epoch_seal); + writeRefTxnIdFields(out, first, RefLogWire::prev_epoch, RefLogWire::prev_seq, *prev_epoch_seal); closeObject(out, first); writeChar('\n', out); } @@ -175,9 +144,9 @@ RefOp readOpRecord(JsonObjectReader & r, RefOpKind kind) op.kind = kind; /// set_published_at fields - std::optional sp_rn; - ManifestFields sp_mf; - std::optional sp_ts; + std::optional sp_ref; + ManifestRefFields sp_manifest_fields; + std::optional sp_published_ms; /// owner_transition bindings BindingFields ob; BindingFields nb; @@ -185,24 +154,30 @@ RefOp readOpRecord(JsonObjectReader & r, RefOpKind kind) String key; while (r.nextKey(key)) { - if (key == "rn") sp_rn = r.readString(); - else if (key == "me") sp_mf.me = r.readU64String(); - else if (key == "mb") sp_mf.mb = r.readU64String(); - else if (key == "mo") sp_mf.mo = r.readU64Number(); - else if (key == "ts") sp_ts = r.readU64Number(); - else if (key == "obk") ob.bk = r.readString(); - else if (key == "orn") ob.rn = r.readString(); - else if (key == "ome") ob.mf.me = r.readU64String(); - else if (key == "omb") ob.mf.mb = r.readU64String(); - else if (key == "omo") ob.mf.mo = r.readU64Number(); - else if (key == "nbk") nb.bk = r.readString(); - else if (key == "nrn") nb.rn = r.readString(); - else if (key == "nme") nb.mf.me = r.readU64String(); - else if (key == "nmb") nb.mf.mb = r.readU64String(); - else if (key == "nmo") nb.mf.mo = r.readU64Number(); + if (key == RefLogWire::ref) + sp_ref = r.readString(); + else if (matchManifestRefFields(key, r, kBareManifestRefKeys, sp_manifest_fields)) + { + } + else if (key == RefLogWire::published_ms) + sp_published_ms = r.readU64Number(); + else if (key == kOldBindingKeys.kind) + ob.kind = r.readString(); + else if (key == kOldBindingKeys.ref) + ob.ref = r.readString(); + else if (matchManifestRefFields(key, r, kOldBindingKeys.manifest, ob.manifest_fields)) + { + } + else if (key == kNewBindingKeys.kind) + nb.kind = r.readString(); + else if (key == kNewBindingKeys.ref) + nb.ref = r.readString(); + else if (matchManifestRefFields(key, r, kNewBindingKeys.manifest, nb.manifest_fields)) + { + } else if (key == "pl") - /// `"pl"` (payload) was removed from the op wire in stage-1 T12 (the `set_payload` op became - /// `set_published_at`). The retired op WORD is already rejected by `opKindFromWord`, but this + /// `"pl"` (payload) was removed from the op wire when the `set_payload` op became + /// `set_published_at`. The retired op WORD is already rejected by `refOpKindFromWireWord`, but this /// generic reader reads field keys before switching on kind, so a `"pl"` field paired with a /// still-recognized op word would otherwise be `skipUnknown`'d. It is a KNOWN-removed field, /// not a genuinely-unknown one -- reject it explicitly rather than silently discard it. @@ -224,12 +199,12 @@ RefOp readOpRecord(JsonObjectReader & r, RefOpKind kind) op.new_binding = nb.build("new"); break; case RefOpKind::SetPublishedAt: - if (!sp_rn || !sp_ts) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: set_published_at missing rn/ts"); - op.ref_name = *sp_rn; + if (!sp_ref || !sp_published_ms) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: set_published_at missing ref/published_ms"); + op.ref_name = *sp_ref; checkCanonicalRefName(op.ref_name, "RefLogTxn", "set_published_at ref_name"); - op.expected_manifest_ref = sp_mf.build("set_published_at manifest_ref"); - op.published_at_ms = *sp_ts; + op.expected_manifest_ref = sp_manifest_fields.buildRef("RefLogTxn", "set_published_at"); + op.published_at_ms = *sp_published_ms; break; } return op; @@ -237,6 +212,16 @@ RefOp readOpRecord(JsonObjectReader & r, RefOpKind kind) } +std::string_view refOpKindToWireWord(RefOpKind kind) +{ + return kRefOpWords.toWord(kind, "RefLogTxn"); +} + +RefOpKind refOpKindFromWireWord(std::string_view w) +{ + return kRefOpWords.fromWord(w, "RefLogTxn"); +} + bool refLogTxnIsEpochSeal(const RefLogTxn & txn) { return txn.ops.size() == 1 && txn.ops.front().kind == RefOpKind::EpochSeal; @@ -325,29 +310,44 @@ RefLogTxn decodeRefLogTxn(std::string_view data, const String & expected_ns, con ReadBufferFromMemory m(line.data(), line.size()); JsonObjectReader r(m, KeyStrictness::Tolerant, "cas_ref_log"); bool saw_ns = false; - bool saw_we = false; - bool saw_rs = false; - std::optional pse; - std::optional pss; + bool saw_txn_epoch = false; + bool saw_txn_seq = false; + std::optional prev_epoch; + std::optional prev_seq; String key; while (r.nextKey(key)) { - if (key == "ns") { txn.ns = r.readString(); saw_ns = true; } - else if (key == "we") { txn.txn_id.writer_epoch = r.readU64String(); saw_we = true; } - else if (key == "rs") { txn.txn_id.ref_sequence = r.readU64String(); saw_rs = true; } - else if (key == "!pse") pse = r.readU64String(); - else if (key == "!pss") pss = r.readU64String(); - else r.skipUnknown(key); + if (key == RefLogWire::ns) + { + txn.ns = r.readString(); + saw_ns = true; + } + else if (key == RefLogWire::txn_epoch) + { + txn.txn_id.writer_epoch = r.readU64String(); + saw_txn_epoch = true; + } + else if (key == RefLogWire::txn_seq) + { + txn.txn_id.ref_sequence = r.readU64String(); + saw_txn_seq = true; + } + else if (key == RefLogWire::prev_epoch) + prev_epoch = r.readU64String(); + else if (key == RefLogWire::prev_seq) + prev_seq = r.readU64String(); + else + r.skipUnknown(key); } - if (!saw_ns || !saw_we || !saw_rs) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: meta line missing ns/we/rs"); - /// Both-or-neither: `nextKey` already rejects a repeated "!pse"/"!pss" (duplicate-key check), so + if (!saw_ns || !saw_txn_epoch || !saw_txn_seq) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: meta line missing namespace/txn_epoch/txn_seq"); + /// Both-or-neither: `nextKey` already rejects a repeated "!prev_epoch"/"!prev_seq" (duplicate-key check), so /// this only guards against a body carrying exactly one of the pair. - if (pse || pss) + if (prev_epoch || prev_seq) { - if (!pse || !pss) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: prev_epoch_seal needs both !pse and !pss"); - txn.prev_epoch_seal = RefTxnId{*pse, *pss}; + if (!prev_epoch || !prev_seq) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: prev_epoch_seal needs both !prev_epoch and !prev_seq"); + txn.prev_epoch_seal = RefTxnId{*prev_epoch, *prev_seq}; } if (!m.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: junk after meta line"); @@ -365,11 +365,16 @@ RefLogTxn decodeRefLogTxn(std::string_view data, const String & expected_ns, con expected_ns, expected_txn_id.writer_epoch, expected_txn_id.ref_sequence); /// op record lines, until the trailer + /// One line scratch and one reader for the whole loop, as the other row decoders do: + /// rebuilding them per row costs an allocation per row for the seen-key store and the line. + String row_line; + JsonObjectReader row_reader; while (true) { - const String line = readLine(in, line_cap, "cas_ref_log"); - ReadBufferFromMemory l(line.data(), line.size()); - JsonObjectReader r(l, KeyStrictness::Tolerant, "cas_ref_log"); + readLineInto(in, row_line, line_cap, "cas_ref_log"); + ReadBufferFromMemory l(row_line.data(), row_line.size()); + row_reader.reset(l, KeyStrictness::Tolerant, "cas_ref_log"); + JsonObjectReader & r = row_reader; String key; if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: empty line"); @@ -385,9 +390,9 @@ RefLogTxn decodeRefLogTxn(std::string_view data, const String & expected_ns, con "RefLogTxn: trailer count {} != {} ops", n, txn.ops.size()); break; } - if (key != "op") + if (key != RefLogWire::op) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: record must start with \"op\""); - const RefOpKind kind = opKindFromWord(r.readString()); + const RefOpKind kind = refOpKindFromWireWord(r.readString()); txn.ops.push_back(readOpRecord(r, kind)); if (!l.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefLogTxn: junk after op record"); @@ -440,4 +445,41 @@ size_t removalFramingSize(const String & ns, const RefTxnId & txn_id, uint64_t o return out.size(); } +std::optional peekRefLogMeta(const String & sealed_bytes) +{ + try + { + const String text = openObject(FormatId::RefLog, sealed_bytes); + ReadBufferFromMemory in(text.data(), text.size()); + const uint64_t line_cap = traitsFor(FormatId::RefLog).line_cap; + readLine(in, line_cap, "cas_ref_log"); /// header line -- skipped, the version is not judged here + const String meta = readLine(in, line_cap, "cas_ref_log"); + ReadBufferFromMemory m(meta.data(), meta.size()); + JsonObjectReader r(m, KeyStrictness::Tolerant, "cas_ref_log"); + RefLogMetaPeek peek; + bool saw_ns = false; + bool saw_epoch = false; + bool saw_seq = false; + String key; + while (r.nextKey(key)) + { + if (key == RefLogWire::ns) { peek.ns = r.readString(); saw_ns = true; } + else if (key == RefLogWire::txn_epoch) { peek.writer_epoch = r.readU64String(); saw_epoch = true; } + else if (key == RefLogWire::txn_seq) { peek.ref_sequence = r.readU64String(); saw_seq = true; } + else r.skipUnknown(key); + } + if (!saw_ns || !saw_epoch || !saw_seq) + return std::nullopt; + return peek; + } + catch (...) + { + /// Deliberately total: a diagnostic that throws while explaining an anomaly replaces the + /// anomaly's report with its own. A seal-linked txn reaches here too -- its `!`-prefixed chain + /// keys make the tolerant reader refuse the line -- and answering `nullopt` is correct: this + /// peek identifies a writer, it does not certify an object. + return std::nullopt; + } +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.h index 34347fb97aaf..e4949cf130b8 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefLogFormat.h @@ -13,7 +13,7 @@ namespace DB::Cas /// Text codec for `cas_ref_log`, the immutable object stored at `_log/`. Each object contains /// exactly one committed transaction: its namespace, transaction id, and the batch of `RefOp`s applied -/// by that commit. The body has a header, a meta line `{"ns","we","rs",["!pse","!pss"]}`, one JSON +/// by that commit. The body has a header, a meta line `{"namespace","txn_epoch","txn_seq",["!prev_epoch","!prev_seq"]}`, one JSON /// record per op, and a `{"n":count}` trailer. Records are emitted in the transaction's stored order /// and contain no codec-generated timestamps, so encoding the same value is byte-identical. This /// determinism is a property of the representation, not an adoption gate: ref commits use @@ -21,7 +21,7 @@ namespace DB::Cas /// returned text. /// /// `RefOpKind::EpochSeal` closes an epoch transition in-band (spec INV-2): a seal transaction contains -/// exactly that one op, and the meta line's optional `prev_epoch_seal` (wire fields `!pse`/`!pss`, +/// exactly that one op, and the meta line's optional `prev_epoch_seal` (wire fields `!prev_epoch`/`!prev_seq`, /// CRITICAL -- an unrecognized `!`-key fails closed with `UNKNOWN_FORMAT_VERSION` rather than being /// silently skipped, since dropping it would lose INV-2's chain evidence while still passing the /// structural grammar) chains to the transaction id of the seal that closed the PRECEDING epoch, and @@ -41,6 +41,13 @@ enum class RefOpKind : uint8_t EpochSeal = 5, }; +/// Convert a ref-log operation discriminator to its canonical wire word. Throws `LOGICAL_ERROR` if +/// `kind` is not represented by this format. +std::string_view refOpKindToWireWord(RefOpKind kind); + +/// Its fail-closed inverse: an unknown word is `CORRUPTED_DATA`. +RefOpKind refOpKindFromWireWord(std::string_view w); + /// One operation inside a `RefLogTxn`. Only the fields documented next to `kind` are meaningful for /// that kind, and the codec never reads or writes the others. `OwnerTransition` optionally removes /// `old_binding` and/or installs `new_binding`; `SetPublishedAt` carries the expected manifest and the @@ -174,4 +181,22 @@ void validateEpochSealGrammarStructural(const RefLogTxn & txn); /// mint a sequence-1 transaction. Throws CORRUPTED_DATA on violation. void validateEpochSealGrammarContextual(const RefLogTxn & txn, uint64_t life_epoch); +/// The three identity fields of a `cas_ref_log` meta line, read WITHOUT trusting the object: this is +/// the anomaly diagnostic's view of an object found at a key it should not occupy, so the body is not +/// expected to match that key's identity. +struct RefLogMetaPeek +{ + String ns; + uint64_t writer_epoch = 0; + uint64_t ref_sequence = 0; +}; + +/// Best-effort identification of a sealed `cas_ref_log` object: opens it, skips the header line, and +/// reads the meta line's three identity fields. Never validates the header version, never reads past +/// the meta line, and answers `nullopt` for anything it cannot read -- truncation, garbage, a +/// different format, or a meta line missing one of the three. It lives HERE, beside the key +/// constants, because a caller that spelled those keys itself would silently stop matching the first +/// time they are renamed, and this reader has no output an ordinary test would miss. +std::optional peekRefLogMeta(const String & sealed_bytes); + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.cpp index f31e9fed4ca2..8d8b9e6f8244 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.cpp @@ -20,6 +20,19 @@ namespace DB::Cas namespace { +namespace RefSnapWire +{ + constexpr WireKey ns{"namespace"}; + constexpr WireKey snapshot_epoch{"snapshot_epoch"}; + constexpr WireKey snapshot_seq{"snapshot_seq"}; + constexpr WireKey lifecycle{"lifecycle"}; + constexpr WireKey kind{"kind"}; + constexpr WireKey ref{"ref"}; + constexpr WireKey published_ms{"published_ms"}; +} + +constexpr std::string_view kLiveLifecycleWord = "live"; + void checkCommittedSorted(const std::vector & rows) { for (size_t i = 1; i < rows.size(); ++i) @@ -59,13 +72,10 @@ void writeCommittedRow(CasJsonWriter & out, const RefCommittedRow & row) checkCanonicalRefName(row.ref_name, "RefTableSnapshot", "committed ref_name"); checkManifestRef(row.manifest_ref, "RefTableSnapshot", "committed"); bool first = true; - writeKey(out, "k", first); - writeStringValue(out, "c"); - writeKey(out, "rn", first); - writeStringValue(out, row.ref_name); - writeManifestRefFields(out, first, "", row.manifest_ref); - writeKey(out, "ts", first); - writeIntText(row.published_at_ms, out); + writeWordField(out, RefSnapWire::kind, refOwnerKindToWord(RefOwnerKind::Committed), first); + writeStringField(out, RefSnapWire::ref, row.ref_name, first); + writeManifestRefFields(out, first, kBareManifestRefKeys, row.manifest_ref); + writeNumberField(out, RefSnapWire::published_ms, row.published_at_ms, first); closeObject(out, first); writeChar('\n', out); } @@ -79,49 +89,30 @@ void writePrecommitRow(CasJsonWriter & out, const RefOwnerBinding & row) checkCanonicalRefName(row.ref_name, "RefTableSnapshot", "precommit ref_name"); checkManifestRef(row.manifest_ref, "RefTableSnapshot", "precommit"); bool first = true; - writeKey(out, "k", first); - writeStringValue(out, "p"); - writeKey(out, "rn", first); - writeStringValue(out, row.ref_name); - writeManifestRefFields(out, first, "", row.manifest_ref); + writeWordField(out, RefSnapWire::kind, refOwnerKindToWord(RefOwnerKind::Precommit), first); + writeStringField(out, RefSnapWire::ref, row.ref_name, first); + writeManifestRefFields(out, first, kBareManifestRefKeys, row.manifest_ref); closeObject(out, first); writeChar('\n', out); } -/// The snapshot's header-object meta line (`ns`, `snapshot_id`, and the required generation-8 -/// `lc:"live"` constant). Shared by +/// The snapshot's header-object meta line (`namespace`, `snapshot_id`, and the required +/// `lifecycle:"live"` constant). Shared by /// `encodeRefTableSnapshot` and `snapshotFramingSize` so the two never disagree by a /// byte. Assumes the caller has already validated the snapshot (or is measuring framing only). void writeSnapshotMeta(CasJsonWriter & out, const RefTableSnapshot & snapshot) { bool first = true; - writeKey(out, "ns", first); - writeStringValue(out, snapshot.ns); - writeRefTxnIdFields(out, first, "we", "rs", snapshot.snapshot_id); - writeKey(out, "lc", first); - writeStringValue(out, "live"); + writeStringField(out, RefSnapWire::ns, snapshot.ns, first); + writeRefTxnIdFields(out, first, RefSnapWire::snapshot_epoch, RefSnapWire::snapshot_seq, snapshot.snapshot_id); + /// A snapshot object exists only for a live namespace -- `RefLifecycle::Removed` has no snapshot + /// representation -- so the wire carries exactly one lifecycle word. The reader keeps the + /// fail-closed half: any other word, or none, is rejected there. + writeWordField(out, RefSnapWire::lifecycle, kLiveLifecycleWord, first); closeObject(out, first); writeChar('\n', out); } -/// Collector for a ManifestRef's three flat fields (bare "me"/"mb"/"mo"). -struct ManifestFields -{ - std::optional me; - std::optional mb; - std::optional mo; - - /// Reconstruct a manifest reference after the tolerant reader has collected all three flat - /// fields. Missing fields are malformed input; `manifestRefFromFields` performs the remaining - /// range checks and reports the same corruption context as the row decoder. - ManifestRef build(std::string_view what) const - { - if (!me || !mb || !mo) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: {} manifest_ref missing me/mb/mo", what); - return manifestRefFromFields(*me, *mb, *mo, "RefTableSnapshot", what); - } -}; - } String encodeRefTableSnapshot(const RefTableSnapshot & snapshot) @@ -160,40 +151,46 @@ RefTableSnapshot decodeRefTableSnapshot( ReadBufferFromMemory meta_buf(line.data(), line.size()); JsonObjectReader r(meta_buf, KeyStrictness::Tolerant, "cas_ref_snap"); bool saw_ns = false; - bool saw_we = false; - bool saw_rs = false; - bool saw_lc = false; + bool saw_snapshot_epoch = false; + bool saw_snapshot_seq = false; + RefLifecycle lifecycle = RefLifecycle::Removed; String key; while (r.nextKey(key)) { - if (key == "ns") { snapshot.ns = r.readString(); saw_ns = true; } - else if (key == "we") { snapshot.snapshot_id.writer_epoch = r.readU64String(); saw_we = true; } - else if (key == "rs") { snapshot.snapshot_id.ref_sequence = r.readU64String(); saw_rs = true; } - else if (key == "lc") + if (key == RefSnapWire::ns) { snapshot.ns = r.readString(); saw_ns = true; } + else if (key == RefSnapWire::snapshot_epoch) { snapshot.snapshot_id.writer_epoch = r.readU64String(); saw_snapshot_epoch = true; } + else if (key == RefSnapWire::snapshot_seq) { snapshot.snapshot_id.ref_sequence = r.readU64String(); saw_snapshot_seq = true; } + else if (key == RefSnapWire::lifecycle) { - const String lifecycle = r.readString(); - if (lifecycle != "live") + const String lifecycle_word = r.readString(); + if (lifecycle_word != kLiveLifecycleWord) throw Exception(ErrorCodes::CORRUPTED_DATA, - "RefTableSnapshot: lifecycle must be exactly 'live', got '{}'", lifecycle); - saw_lc = true; + "RefTableSnapshot: lifecycle must be exactly '{}', got '{}'", kLiveLifecycleWord, lifecycle_word); + lifecycle = RefLifecycle::Live; } else if (key == "rte" || key == "rts") throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: meta carries retired terminal field '{}'", key); else r.skipUnknown(key); } - if (!saw_ns || !saw_we || !saw_rs || !saw_lc) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: meta line missing ns/we/rs/lc"); + if (!saw_ns || !saw_snapshot_epoch || !saw_snapshot_seq || lifecycle != RefLifecycle::Live) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: meta line missing namespace/snapshot_epoch/snapshot_seq/lifecycle"); if (!meta_buf.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: junk after meta line"); } /// record lines (committed then precommit), until the trailer + /// One line scratch and one reader for the whole loop: a decoder that rebuilds them per + /// row pays an allocation per row for the seen-key store and the line, which profiling put + /// at about a fifth of the instructions executed inside a row. + String row_line; + JsonObjectReader row_reader; while (true) { - const String line = readLine(in, line_cap, "cas_ref_snap"); - ReadBufferFromMemory l(line.data(), line.size()); - JsonObjectReader r(l, KeyStrictness::Tolerant, "cas_ref_snap"); + readLineInto(in, row_line, line_cap, "cas_ref_snap"); + ReadBufferFromMemory l(row_line.data(), row_line.size()); + row_reader.reset(l, KeyStrictness::Tolerant, "cas_ref_snap"); + JsonObjectReader & r = row_reader; String key; if (!r.nextKey(key)) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: empty line"); @@ -210,22 +207,22 @@ RefTableSnapshot decodeRefTableSnapshot( "RefTableSnapshot: trailer count {} != {} rows", n, snapshot.committed.size() + snapshot.precommits.size()); break; } - if (key != "k") - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: record must start with \"k\""); - const String k = r.readString(); + if (key != RefSnapWire::kind) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: record must start with \"kind\""); + const RefOwnerKind kind = refOwnerKindFromWord(r.readString(), "RefTableSnapshot row kind"); - std::optional rn; - ManifestFields mf; - std::optional ts; + std::optional ref; + ManifestRefFields mf; + std::optional published_ms; while (r.nextKey(key)) { - if (key == "rn") rn = r.readString(); - else if (key == "me") mf.me = r.readU64String(); - else if (key == "mb") mf.mb = r.readU64String(); - else if (key == "mo") mf.mo = r.readU64Number(); - else if (key == "ts") ts = r.readU64Number(); + if (key == RefSnapWire::ref) ref = r.readString(); + else if (matchManifestRefFields(key, r, kBareManifestRefKeys, mf)) + { + } + else if (key == RefSnapWire::published_ms) published_ms = r.readU64Number(); else if (key == "pl") - /// `"pl"` (payload) was removed from the row wire in stage-1 T12. It is a KNOWN-removed + /// `"pl"` (payload) was removed from the row wire. It is a KNOWN-removed /// field, not a genuinely-unknown future one the tolerant reader may skip -- silently /// discarding a persisted payload would lose data -- so reject it explicitly. throw Exception(ErrorCodes::CORRUPTED_DATA, @@ -235,30 +232,36 @@ RefTableSnapshot decodeRefTableSnapshot( if (!l.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: junk after record"); - if (k == "c") + /// A `switch` rather than an if/else-if chain: the row kinds partition the enum, and a future + /// enumerator must not be able to arrive here, pass the word lookup, and then fall out of the + /// chain as a silently dropped row. With no default arm, adding one is a build error. + switch (kind) + { + case RefOwnerKind::Committed: { - if (!rn || !ts) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: committed row missing rn/ts"); + if (!ref || !published_ms) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: committed row missing ref/published_ms"); RefCommittedRow row; - row.ref_name = *rn; + row.ref_name = *ref; checkCanonicalRefName(row.ref_name, "RefTableSnapshot", "committed ref_name"); - row.manifest_ref = mf.build("committed"); - row.published_at_ms = *ts; + row.manifest_ref = mf.buildRef("RefTableSnapshot", "committed"); + row.published_at_ms = *published_ms; snapshot.committed.push_back(std::move(row)); + break; } - else if (k == "p") + case RefOwnerKind::Precommit: { - if (!rn) - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: precommit row missing rn"); + if (!ref) + throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: precommit row missing ref"); RefOwnerBinding row; - row.kind = RefOwnerKind::Precommit; - row.ref_name = *rn; + row.kind = kind; + row.ref_name = *ref; checkCanonicalRefName(row.ref_name, "RefTableSnapshot", "precommit ref_name"); - row.manifest_ref = mf.build("precommit"); + row.manifest_ref = mf.buildRef("RefTableSnapshot", "precommit"); snapshot.precommits.push_back(std::move(row)); + break; + } } - else - throw Exception(ErrorCodes::CORRUPTED_DATA, "RefTableSnapshot: unknown row kind '{}'", k); } /// The object key is supplied separately from the body. Check the binding before accepting any diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.h index 07972a0d47ab..79573cf4767a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefSnapshotFormat.h @@ -26,7 +26,7 @@ namespace DB::Cas /// published through the ordinary single-owner `putIfAbsentControlled` path, not a /// `putDeterministicArtifact` byte-adoption gate. -/// In-memory ref-table lifecycle. Only `Live` is serializable as a generation-8 snapshot; terminal +/// In-memory ref-table lifecycle. Only `Live` is serializable as a snapshot; terminal /// state lives in the removal log and fold evidence and has no snapshot DTO representation. enum class RefLifecycle : uint8_t { @@ -47,7 +47,7 @@ struct RefCommittedRow /// The complete state of one namespace's ref table in one canonical snapshot object. `precommits` /// reuses `RefOwnerBinding` from `CasRefWireVocab.h`; every entry's `kind` must be `Precommit`. -/// Generation 8 serializes only `Live` snapshots. Both row vectors must already be strictly sorted by +/// A snapshot serializes only `Live` namespaces. Both row vectors must already be strictly sorted by /// their documented keys, because the codec /// validates and emits the caller-provided order rather than sorting it. struct RefTableSnapshot diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.cpp index bf1a5445e4df..c184f99d4f4c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.cpp @@ -1,4 +1,6 @@ #include +#include +#include #include namespace DB @@ -12,21 +14,26 @@ namespace ErrorCodes namespace DB::Cas { +namespace +{ + +constexpr EnumWireTable kRefOwnerKindWords{{{ + {RefOwnerKind::Committed, "committed"}, + {RefOwnerKind::Precommit, "precommit"}, +}}}; + +static_assert(casEnumTableCoversEnum()); + +} + std::string_view refOwnerKindToWord(RefOwnerKind k) { - switch (k) - { - case RefOwnerKind::Committed: return "committed"; - case RefOwnerKind::Precommit: return "precommit"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref wire: unknown RefOwnerKind {}", static_cast(k)); + return kRefOwnerKindWords.toWord(k, "CAS ref wire: RefOwnerKind"); } RefOwnerKind refOwnerKindFromWord(std::string_view w, std::string_view what) { - if (w == "committed") return RefOwnerKind::Committed; - if (w == "precommit") return RefOwnerKind::Precommit; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown owner kind '{}'", what, w); + return kRefOwnerKindWords.fromWord(w, what); } void checkRefTxnIdNonzero(const RefTxnId & id, std::string_view format, std::string_view field) @@ -36,12 +43,19 @@ void checkRefTxnIdNonzero(const RefTxnId & id, std::string_view format, std::str "{}: {} fields must both be nonzero, got {}-{}", format, field, id.writer_epoch, id.ref_sequence); } -void writeRefTxnIdFields(CasJsonWriter & out, bool & first, std::string_view epoch_key, std::string_view seq_key, const RefTxnId & id) +void writeRefTxnIdFields(CasJsonWriter & out, bool & first, WireKey epoch_key, WireKey seq_key, const RefTxnId & id) +{ + writeU64StringField(out, epoch_key, id.writer_epoch, first); + writeU64StringField(out, seq_key, id.ref_sequence, first); +} + +void writeBindingFields(CasJsonWriter & out, bool & first, const BindingWireKeys & keys, const RefOwnerBinding & binding) { - writeKey(out, epoch_key, first); - writeU64StringValue(out, id.writer_epoch); - writeKey(out, seq_key, first); - writeU64StringValue(out, id.ref_sequence); + checkCanonicalRefName(binding.ref_name, "RefLogTxn", "owner binding ref_name"); + checkManifestRef(binding.manifest_ref, "RefLogTxn", "owner binding manifest_ref"); + writeWordField(out, keys.kind, refOwnerKindToWord(binding.kind), first); + writeStringField(out, keys.ref, binding.ref_name, first); + writeManifestRefFields(out, first, keys.manifest, binding.manifest_ref); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.h index fdccef8f7fdb..0feacf7352d3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasRefWireVocab.h @@ -35,9 +35,8 @@ struct RefOwnerBinding bool operator==(const RefOwnerBinding &) const = default; }; -/// Convert an owner-kind discriminator to its canonical text word. Throws `CORRUPTED_DATA` for a -/// value not represented by this format; accepting an unknown value would produce an ambiguous wire -/// record. +/// Convert an owner-kind discriminator to its canonical text word. Throws `LOGICAL_ERROR` for an +/// out-of-range enum value. std::string_view refOwnerKindToWord(RefOwnerKind k); /// Parse a canonical owner-kind word. `what` identifies the containing field in the @@ -48,9 +47,13 @@ RefOwnerKind refOwnerKindFromWord(std::string_view w, std::string_view what); /// in-progress JSON object, both as decimal STRINGS -- the representation is width-independent, so no /// consumer has to care how large a `ref_sequence` can get. `epoch_key`/`seq_key` name the two fields, /// letting each format distinguish its primary id from any secondary id it embeds (for example, -/// `cas_ref_log`'s `we`/`rs` versus its `prev_epoch_seal` pair) while sharing one writer so the +/// `cas_ref_log`'s `txn_epoch`/`txn_seq` versus its `prev_epoch_seal` pair) while sharing one writer so the /// formats can never disagree on the representation. -void writeRefTxnIdFields(CasJsonWriter & out, bool & first, std::string_view epoch_key, std::string_view seq_key, const RefTxnId & id); +void writeRefTxnIdFields(CasJsonWriter & out, bool & first, WireKey epoch_key, WireKey seq_key, const RefTxnId & id); + +/// Append one owner binding's flat fields named by `keys` to a ref-log `owner_transition` object. +/// The binding's ref name and manifest reference are validated before writing. +void writeBindingFields(CasJsonWriter & out, bool & first, const BindingWireKeys & keys, const RefOwnerBinding & binding); /// `RefTxnId`'s validity rule applied to ONE field of a decoded or about-to-be-encoded record: both /// components nonzero. `renderRefTxnId` refuses to build a key from anything else, so a half-zero id diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.cpp index f523553271b4..a01a62ed30f6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.cpp @@ -14,6 +14,31 @@ namespace ErrorCodes namespace DB::Cas { +namespace OwnerWire +{ + constexpr WireKey server_uuid{"server_uuid"}; + constexpr WireKey retired_at_ms{"retired_at_ms"}; +} + +namespace ServerEpochWire +{ + constexpr WireKey next_writer_epoch{"next_writer_epoch"}; +} + +namespace MountLeaseWire +{ + constexpr WireKey server_uuid{"server_uuid"}; + constexpr WireKey writer_epoch{"writer_epoch"}; + constexpr WireKey hostname{"hostname"}; + constexpr WireKey pid{"pid"}; + constexpr WireKey started_at_ms{"started_at_ms"}; + constexpr WireKey seq{"seq"}; + constexpr WireKey expires_at_ms{"expires_at_ms"}; + constexpr WireKey min_active_build_sequence{"min_active_build_sequence"}; + constexpr WireKey gc_fenced{"gc_fenced"}; + constexpr WireKey write_attempt_id{"write_attempt_id"}; +} + namespace { @@ -32,13 +57,9 @@ String encodeOwner(const OwnerObject & o) CasJsonWriter out(256); writeHeaderLine(out, FormatId::Owner); bool first = true; - writeKey(out, "su", first); - writeHex128Value(out, o.server_uuid); + writeHex128Field(out, OwnerWire::server_uuid, o.server_uuid, first); if (o.retired_at_ms) - { - writeKey(out, "rt", first); - writeIntText(*o.retired_at_ms, out); - } + writeNumberField(out, OwnerWire::retired_at_ms, *o.retired_at_ms, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -58,19 +79,19 @@ OwnerObject decodeOwner(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "su") + if (key == OwnerWire::server_uuid) { o.server_uuid = r.readHex128(); saw = true; } - else if (key == "rt") + else if (key == OwnerWire::retired_at_ms) rt = r.readU64Number(); else r.skipUnknown(key); } o.retired_at_ms = rt; if (!saw) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS owner: missing su"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS owner: missing server_uuid"); if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS owner: trailing bytes"); return o; @@ -81,8 +102,7 @@ String encodeServerEpoch(const ServerEpoch & e) CasJsonWriter out(256); writeHeaderLine(out, FormatId::ServerEpoch); bool first = true; - writeKey(out, "nwe", first); - writeU64StringValue(out, e.next_writer_epoch); + writeU64StringField(out, ServerEpochWire::next_writer_epoch, e.next_writer_epoch, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -101,7 +121,7 @@ ServerEpoch decodeServerEpoch(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "nwe") + if (key == ServerEpochWire::next_writer_epoch) { e.next_writer_epoch = r.readU64String(); saw = true; @@ -110,7 +130,7 @@ ServerEpoch decodeServerEpoch(std::string_view data) r.skipUnknown(key); } if (!saw) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-epoch: missing nwe"); + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-epoch: missing next_writer_epoch"); if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-epoch: trailing bytes"); return e; @@ -121,16 +141,16 @@ String encodeMountLease(const MountLease & m) CasJsonWriter out(256); writeHeaderLine(out, FormatId::MountLease); bool first = true; - writeKey(out, "su", first); writeHex128Value(out, m.server_uuid); - writeKey(out, "we", first); writeU64StringValue(out, m.writer_epoch); - writeKey(out, "hn", first); writeStringValue(out, m.hostname); - writeKey(out, "pid", first); writeIntText(m.pid, out); - writeKey(out, "sat", first); writeIntText(m.started_at_ms, out); - writeKey(out, "seq", first); writeU64StringValue(out, m.seq); - writeKey(out, "eat", first); writeIntText(m.expires_at_ms, out); - writeKey(out, "ma", first); writeU64StringValue(out, m.min_active); - writeKey(out, "fen", first); writeBoolValue(out, m.gc_fenced); - writeKey(out, "write_attempt_id", first); writeHex128Value(out, m.write_attempt_id); + writeHex128Field(out, MountLeaseWire::server_uuid, m.server_uuid, first); + writeU64StringField(out, MountLeaseWire::writer_epoch, m.writer_epoch, first); + writeStringField(out, MountLeaseWire::hostname, m.hostname, first); + writeNumberField(out, MountLeaseWire::pid, m.pid, first); + writeNumberField(out, MountLeaseWire::started_at_ms, m.started_at_ms, first); + writeU64StringField(out, MountLeaseWire::seq, m.seq, first); + writeNumberField(out, MountLeaseWire::expires_at_ms, m.expires_at_ms, first); + writeU64StringField(out, MountLeaseWire::min_active_build_sequence, m.min_active_build_sequence, first); + writeBoolField(out, MountLeaseWire::gc_fenced, m.gc_fenced, first); + writeHex128Field(out, MountLeaseWire::write_attempt_id, m.write_attempt_id, first); closeObject(out, first); writeChar('\n', out); return std::move(out).take(); @@ -151,32 +171,47 @@ MountLease decodeMountLease(std::string_view data) String key; while (r.nextKey(key)) { - if (key == "su") + if (key == MountLeaseWire::server_uuid) { m.server_uuid = r.readHex128(); saw_su = true; } - else if (key == "we") + else if (key == MountLeaseWire::writer_epoch) { m.writer_epoch = r.readU64String(); saw_we = true; } - else if (key == "hn") m.hostname = r.readString(); - else if (key == "pid") m.pid = r.readU64Number(); - else if (key == "sat") m.started_at_ms = r.readU64Number(); - else if (key == "seq") m.seq = r.readU64String(); - else if (key == "eat") m.expires_at_ms = r.readU64Number(); - else if (key == "ma") m.min_active = r.readU64String(); - else if (key == "fen") m.gc_fenced = r.readBool(); - else if (key == "write_attempt_id") + else if (key == MountLeaseWire::hostname) + m.hostname = r.readString(); + else if (key == MountLeaseWire::pid) + m.pid = r.readU64Number(); + else if (key == MountLeaseWire::started_at_ms) + m.started_at_ms = r.readU64Number(); + else if (key == MountLeaseWire::seq) + m.seq = r.readU64String(); + else if (key == MountLeaseWire::expires_at_ms) + m.expires_at_ms = r.readU64Number(); + else if (key == MountLeaseWire::min_active_build_sequence) + m.min_active_build_sequence = r.readU64String(); + else if (key == MountLeaseWire::gc_fenced) + m.gc_fenced = r.readBool(); + else if (key == MountLeaseWire::write_attempt_id) { m.write_attempt_id = r.readHex128(); saw_write_attempt_id = true; } - else r.skipUnknown(key); + else + r.skipUnknown(key); } - if (!saw_su || !saw_we || !saw_write_attempt_id || m.write_attempt_id == UInt128{}) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS mount-lease: missing or zero identity field"); + /// Named one by one rather than as a single condition: these are three separate identities, and a + /// shared message cannot tell an operator which of them the object is missing -- nor let a test + /// prove that each is actually required. + if (!saw_su) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS mount-lease: missing server_uuid"); + if (!saw_we) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS mount-lease: missing writer_epoch"); + if (!saw_write_attempt_id || m.write_attempt_id == UInt128{}) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS mount-lease: missing or zero write_attempt_id"); if (!body_in.eof() || !in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS mount-lease: trailing bytes"); return m; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.h index 0c73e8c5a6e6..8e0bd8118a1e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasServerRootFormats.h @@ -17,7 +17,7 @@ namespace DB::Cas /// The owner object permanently binds a configured `server_root_id` to one server UUID. The epoch /// object stores the next writer epoch and is CAS-bumped so an epoch is never reused after a /// restart or supersession. The mount object is the expirable liveness lease for one writer -/// incarnation; its `min_active` value carries the GC acknowledgement floor, while `gc_fenced` is a +/// incarnation; its `min_active_build_sequence` value carries the GC acknowledgement floor, while `gc_fenced` is a /// terminal fence-out marker for that incarnation. /// Permanent identity anchor for one configured server root. It is created with put-if-absent, never @@ -42,7 +42,7 @@ struct ServerEpoch /// The current liveness lease for one `(server_uuid, writer_epoch)` writer incarnation. The pool /// layer renews and replaces this object with CAS/overwrite operations, and GC may fence an expired -/// lease by setting `gc_fenced`; a fenced incarnation must not resume writing. `min_active` is the +/// lease by setting `gc_fenced`; a fenced incarnation must not resume writing. `min_active_build_sequence` is the /// merged GC acknowledgement floor, with `UINT64_MAX` marking a clean farewell (retired lease). struct MountLease { @@ -53,7 +53,7 @@ struct MountLease uint64_t started_at_ms = 0; uint64_t seq = 0; uint64_t expires_at_ms = 0; - uint64_t min_active = 0; /// UINT64_MAX = retired (farewell) + uint64_t min_active_build_sequence = 0; /// UINT64_MAX = retired (farewell) bool gc_fenced = false; /// GC fence-out of an expired lease; terminal /// One holder-originated body identity. Every physical retry of that one logical write reuses /// it; a GC fence copies the observed value while every successor holder body mints a new one. @@ -66,9 +66,9 @@ struct MountLease /// decisions belong to the caller that coordinates the server-root object. String encodeOwner(const OwnerObject & o); -/// Decode an owner anchor, requiring its `su` field, tolerating an absent optional `rt` retirement +/// Decode an owner anchor, requiring its `server_uuid` field, tolerating an absent optional `retired_at_ms` retirement /// timestamp, and rejecting bytes after the body line. Unknown JSON fields are skipped for -/// forward-compatible reads; malformed input, a missing `su`, and trailing data throw +/// forward-compatible reads; malformed input, a missing `server_uuid`, and trailing data throw /// `CORRUPTED_DATA`. OwnerObject decodeOwner(std::string_view data); @@ -77,13 +77,13 @@ OwnerObject decodeOwner(std::string_view data); /// codec. String encodeServerEpoch(const ServerEpoch & e); -/// Decode the epoch counter, requiring its `nwe` field and rejecting bytes after the body line. -/// Unknown JSON fields are skipped for forward-compatible reads; malformed input, a missing `nwe`, +/// Decode the epoch counter, requiring its `next_writer_epoch` field and rejecting bytes after the body line. +/// Unknown JSON fields are skipped for forward-compatible reads; malformed input, a missing `next_writer_epoch`, /// and trailing data throw `CORRUPTED_DATA`. ServerEpoch decodeServerEpoch(std::string_view data); /// Encode the complete mount-lease body as canonical text with the `cas_mount_lease` header and a -/// final newline. This preserves full-range `uint64_t` values such as `min_active` as decimal JSON +/// final newline. This preserves full-range `uint64_t` values such as `min_active_build_sequence` as decimal JSON /// strings and writes `gc_fenced` as a JSON boolean; lease renewal, fencing, and token checks remain /// in the caller. String encodeMountLease(const MountLease & m); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.cpp index 9814d5d14811..aefa25c871c6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.cpp @@ -4,6 +4,7 @@ #include #include #include +#include #include #include #include @@ -109,6 +110,33 @@ void CasJsonWriter::stringValue(std::string_view s) appendChar('"'); } +void CasJsonWriter::wordValue(std::string_view word) +{ + /// The contract, checked where it is cheap to check: a vocabulary word carries no byte the JSON + /// escaper would rewrite. Violating it would emit malformed JSON rather than a mis-escaped + /// string, so this is a programming error and belongs in the debug build, not a runtime branch on + /// the encode hot path. + chassert(std::none_of(word.begin(), word.end(), + [](char c) { return isSpecialJsonByte(static_cast(c)); })); + appendChar('"'); + buf.append(word.data(), word.size()); + appendChar('"'); +} + +void CasJsonWriter::wordArray(std::span words) +{ + appendChar('['); + bool first = true; + for (const std::string_view word : words) + { + if (!first) + appendChar(','); + first = false; + wordValue(word); + } + appendChar(']'); +} + /// ---- read-side pull cursor ---- /// A canonical-text parse failure is CORRUPTED_DATA regardless of which ReadHelpers primitive @@ -139,13 +167,36 @@ auto JsonObjectReader::guarded(F && f) } JsonObjectReader::JsonObjectReader(ReadBuffer & in_, KeyStrictness strictness_, std::string_view what_) - : in(in_), strictness(strictness_), what(what_) + : in(&in_), strictness(strictness_), what(what_) { - guarded([&] { assertChar('{', in); }); + guarded([&] { assertChar('{', *in); }); +} + +void JsonObjectReader::reset(ReadBuffer & in_, KeyStrictness strictness_, std::string_view what_) +{ + in = &in_; + strictness = strictness_; + what = what_; + /// `clear` on both keeps their buffers: that is the whole point of reusing the reader. + seen_keys.clear(); + scratch.clear(); + first = true; + done = false; + guarded([&] { assertChar('{', *in); }); +} + +std::string_view JsonObjectReader::readStringIntoScratch() +{ + scratch.clear(); + readJSONStringInto(scratch, *in, jsonReadSettings()); + return scratch; } bool JsonObjectReader::nextKey(String & key) { + /// A default-constructed reader is unbound until `reset`; using one is a programming error, so + /// this belongs in the debug build rather than as a branch on the decode hot path. + chassert(in != nullptr); return guarded([&]() -> bool { if (done) @@ -153,7 +204,7 @@ bool JsonObjectReader::nextKey(String & key) if (first) { first = false; - if (checkChar('}', in)) + if (checkChar('}', *in)) { done = true; return false; @@ -161,15 +212,15 @@ bool JsonObjectReader::nextKey(String & key) } else { - if (checkChar('}', in)) + if (checkChar('}', *in)) { done = true; return false; } - assertChar(',', in); + assertChar(',', *in); } - readJSONString(key, in, jsonReadSettings()); - assertChar(':', in); + readJSONString(key, *in, jsonReadSettings()); + assertChar(':', *in); if (std::find(seen_keys.begin(), seen_keys.end(), key) != seen_keys.end()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: duplicate key '{}'", what, key); seen_keys.push_back(key); @@ -182,18 +233,38 @@ String JsonObjectReader::readString() return guarded([&] { String s; - readJSONString(s, in, jsonReadSettings()); + readJSONString(s, *in, jsonReadSettings()); return s; }); } +std::vector JsonObjectReader::readStringArray() +{ + return guarded([&] + { + std::vector words; + assertChar('[', *in); + if (checkChar(']', *in)) + return words; + + while (true) + { + String word; + readJSONString(word, *in, jsonReadSettings()); + words.push_back(std::move(word)); + if (checkChar(']', *in)) + return words; + assertChar(',', *in); + } + }); +} + UInt128 JsonObjectReader::readHex128() { return guarded([&] { - const String hex = readString(); - if (hex.size() != 32 - || std::any_of(hex.begin(), hex.end(), [](char c) { return unhex(c) == 0xff || (c >= 'A' && c <= 'F'); })) + const std::string_view hex = readStringIntoScratch(); + if (hex.size() != 32 || std::any_of(hex.begin(), hex.end(), [](char c) { return !isLowercaseHexChar(c); })) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: expected 32 lowercase hex chars, got '{}'", what, hex); return unhexUInt(hex.data()); }); @@ -203,7 +274,7 @@ uint64_t JsonObjectReader::readU64String() { return guarded([&] { - const String s = readString(); + const std::string_view s = readStringIntoScratch(); ReadBufferFromMemory buf(s.data(), s.size()); uint64_t v = 0; readIntText(v, buf); @@ -218,7 +289,7 @@ uint64_t JsonObjectReader::readU64Number() return guarded([&] { uint64_t v = 0; - readIntText(v, in); + readIntText(v, *in); return v; }); } @@ -235,9 +306,9 @@ bool JsonObjectReader::readBool() { return guarded([&] { - if (checkString("true", in)) + if (checkString("true", *in)) return true; - assertString("false", in); + assertString("false", *in); return false; }); } @@ -251,20 +322,28 @@ void JsonObjectReader::skipUnknown(const String & key) "CAS {}: critical key '{}' is not understood by this build", what, key); if (strictness == KeyStrictness::Strict) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown key '{}' in a strict format", what, key); - skipJSONField(in, key, jsonReadSettings()); + skipJSONField(*in, key, jsonReadSettings()); }); } /// ---- header line / trailer line / raw line access ---- +namespace +{ +namespace ContainerWire +{ + constexpr WireKey type{"type"}; + constexpr WireKey version{"v"}; + constexpr WireKey count{"n"}; +} +} + void writeHeaderLine(CasJsonWriter & out, FormatId id) { const FormatTraits & t = traitsFor(id); bool first = true; - writeKey(out, "type", first); - writeStringValue(out, t.type); - writeKey(out, "v", first); - writeIntText(currentCompatibilityVersion(), out); + writeStringField(out, ContainerWire::type, t.type, first); + writeNumberField(out, ContainerWire::version, currentCompatibilityVersion(), first); closeObject(out, first); writeChar('\n', out); } @@ -272,29 +351,49 @@ void writeHeaderLine(CasJsonWriter & out, FormatId id) void writeTrailerLine(CasJsonWriter & out, uint64_t n) { bool first = true; - writeKey(out, "n", first); - writeIntText(n, out); + writeNumberField(out, ContainerWire::count, n, first); closeObject(out, first); writeChar('\n', out); } -String readLine(ReadBuffer & in, uint64_t line_cap, std::string_view what) +void readLineInto(ReadBuffer & in, String & line, uint64_t line_cap, std::string_view what) { - String line; + /// `clear` keeps the capacity, so a caller that reuses one scratch across a stream's rows stops + /// allocating after the longest line it has seen -- the same bound the writer's scratch has. + line.clear(); while (true) { if (in.eof()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: truncated object (line without terminator)", what); - const char c = *in.position(); - ++in.position(); - if (c == '\n') - return line; - line.push_back(c); - if (line.size() > line_cap) + + /// Take the whole run up to the terminator in one append rather than a byte at a time: a + /// per-character `push_back` also re-checks the cap on every character, and a stream row is + /// hundreds of characters long. + const char * const from = in.position(); + const char * const found = find_first_symbols<'\n'>(from, in.buffer().end()); + const size_t taken = static_cast(found - from); + if (line.size() + taken > line_cap) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: line exceeds the {}-byte cap", what, line_cap); + line.append(from, taken); + in.position() += taken; + + /// `find_first_symbols` stops either at the terminator or at the end of the buffered window. + /// Only the first case ends the line; the second needs the next window. + if (found != in.buffer().end()) + { + ++in.position(); + return; + } } } +String readLine(ReadBuffer & in, uint64_t line_cap, std::string_view what) +{ + String line; + readLineInto(in, line, line_cap, what); + return line; +} + namespace { TextHeader parseHeaderObject(std::string_view line, std::string_view what) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.h index 50edd33bd4ba..8bbbf58a4efd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasTextFormat.h @@ -6,6 +6,7 @@ #include #include #include +#include #include #include @@ -59,21 +60,19 @@ class CasJsonWriter append("\":"); } - /// Same, for the prefixed key vocabulary ("o"/"n" + "me"/"mb"/"mo"/"bk"/"rn") — the - /// prefix and name are appended back to back, no composed temporary. - void key(std::string_view prefix, std::string_view name, bool & first) - { - appendChar(first ? '{' : ','); - first = false; - appendChar('"'); - append(prefix); - append(name); - append("\":"); - } - /// Quoted JSON string with full escaping (bulk-run scan). Defined in CasTextFormat.cpp. void stringValue(std::string_view s); + /// A value from a wire VOCABULARY -- an enum table's word or a record tag. Those are drawn from + /// `[a-z0-9_]` by construction, so this writes the bytes as they are instead of running them + /// through the escaper's byte scan and state machine. Output is identical to `stringValue` for + /// every input the contract admits; a caller that passes an arbitrary string is the bug this + /// asserts against, and `writeWordField` is the only intended way in. + void wordValue(std::string_view word); + + /// JSON array of canonical word strings, emitted without intermediate storage. + void wordArray(std::span words); + void u64Number(uint64_t v) { char digits[24]; @@ -154,6 +153,75 @@ inline void writeIntText(uint64_t v, CasJsonWriter & out) { out.u64Number(v); } void writeHeaderLine(CasJsonWriter & out, FormatId id); void writeTrailerLine(CasJsonWriter & out, uint64_t n); +/// A wire-key carrier for migrated writer call sites. Its explicit constructor makes a codec pass a +/// named constant, while an inline `WireKey{"..."}` is deliberately loud. `WireKey` borrows its +/// `string_view`; the referenced text must outlive the key, as with string literals and static constants. +struct WireKey +{ + std::string_view text; + + explicit constexpr WireKey(std::string_view text_) : text(text_) {} + + friend constexpr bool operator==(std::string_view s, const WireKey & k) { return s == k.text; } +}; + +inline void writeKey(CasJsonWriter & out, WireKey key, bool & first) +{ + writeKey(out, key.text, first); +} + +/// For a value that comes from a wire vocabulary (an enum table's word, a record tag). It is NOT +/// interchangeable with `writeStringField`: this one promises its value needs no JSON escaping and +/// skips the escaper accordingly, which is why an open string must never be routed through it. +inline void writeWordField(CasJsonWriter & out, WireKey key, std::string_view word, bool & first) +{ + writeKey(out, key, first); + out.wordValue(word); +} + +inline void writeWordArrayField(CasJsonWriter & out, WireKey key, std::span words, bool & first) +{ + writeKey(out, key, first); + out.wordArray(words); +} + +inline void writeStringField(CasJsonWriter & out, WireKey key, std::string_view value, bool & first) +{ + writeKey(out, key, first); + writeStringValue(out, value); +} + +inline void writeU64StringField(CasJsonWriter & out, WireKey key, uint64_t value, bool & first) +{ + writeKey(out, key, first); + writeU64StringValue(out, value); +} + +inline void writeNumberField(CasJsonWriter & out, WireKey key, uint64_t value, bool & first) +{ + writeKey(out, key, first); + out.u64Number(value); +} + +inline void writeHex128Field(CasJsonWriter & out, WireKey key, const UInt128 & value, bool & first) +{ + writeKey(out, key, first); + writeHex128Value(out, value); +} + +inline void writeBoolField(CasJsonWriter & out, WireKey key, bool value, bool & first) +{ + writeKey(out, key, first); + writeBoolValue(out, value); +} + +/// True iff `c` is one lowercase hexadecimal digit. Persisted CAS digests deliberately reject +/// uppercase spellings so each digest has one canonical textual representation. +constexpr bool isLowercaseHexChar(char c) +{ + return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f'); +} + /// Pull cursor over one canonical JSON object. /// /// The reader borrows the input buffer and records the object name for exception messages. It @@ -167,12 +235,27 @@ class JsonObjectReader public: /// Consumes the opening `{`; throws `CORRUPTED_DATA` when the object does not start there. JsonObjectReader(ReadBuffer & in_, KeyStrictness strictness_, std::string_view what_); + + /// An unbound reader, for a decoder that wants one reader outside its row loop and re-points it + /// per row. `reset` must be called before any read; nothing else is valid on it. + JsonObjectReader() = default; + + /// Re-point an existing reader at another object, as the constructor would, but WITHOUT + /// releasing the buffers it has already grown. A stream decoder reads one object per row, and a + /// reader built fresh each time re-allocates its seen-key store and its value scratch on every + /// row; measured on the `cas_run` decode path, allocation accounting is about a fifth of all + /// instructions executed inside the decoder. Reusing one reader amortises that away. The + /// object-level state -- the key set and the position in the object -- is reset in full, so a + /// reused reader accepts and rejects exactly what a fresh one would. + void reset(ReadBuffer & in_, KeyStrictness strictness_, std::string_view what_); /// Advances to the next key; false when the closing '}' was consumed. The caller must /// consume the value (one read* / skipUnknown) before the next call. Duplicate keys are /// rejected with `CORRUPTED_DATA`. bool nextKey(String & key); /// Reads the value for the key returned by `nextKey` as a JSON string. String readString(); + /// Reads the value for the key returned by `nextKey` as an array of JSON strings. + std::vector readStringArray(); /// Reads a quoted 32-character lowercase hexadecimal string as a `UInt128`. UInt128 readHex128(); /// Reads a quoted decimal u64 string and rejects empty, trailing, or non-decimal text. @@ -193,10 +276,16 @@ class JsonObjectReader template auto guarded(F && f); - ReadBuffer & in; - KeyStrictness strictness; + /// Reads one JSON string into `scratch` and returns a view of it, so a value that is parsed and + /// discarded -- a hex digest, a decimal counter -- costs no allocation once the scratch has + /// grown. The view is valid until the next read on this reader. + std::string_view readStringIntoScratch(); + + ReadBuffer * in = nullptr; + KeyStrictness strictness = KeyStrictness::Strict; String what; std::vector seen_keys; + String scratch; bool first = true; bool done = false; }; @@ -218,6 +307,11 @@ TextHeader expectHeaderLine(ReadBuffer & in, FormatId id); std::optional sniffHeaderLine(std::string_view bytes); /// Reads one line (excluding the '\n' terminator); CORRUPTED_DATA on missing terminator or a line /// longer than `line_cap`. +/// Read one terminator-delimited line into `line`, replacing its contents and KEEPING its capacity, +/// so a caller streaming many rows can reuse one scratch and stop allocating after the longest line. +void readLineInto(ReadBuffer & in, String & line, uint64_t line_cap, std::string_view what); + +/// Allocating form, for callers that read a single line. String readLine(ReadBuffer & in, uint64_t line_cap, std::string_view what); /// Position of the next byte `stringValue` treats specially (control byte, '"', '\\', or the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.cpp index 31d44f260121..ab5fe8344082 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.cpp @@ -1,7 +1,10 @@ #include #include #include +#include #include +#include +#include namespace DB { @@ -14,74 +17,62 @@ namespace ErrorCodes namespace DB::Cas { -std::string_view tokenTypeToWord(TokenType t) +static_assert(casEnumTableCoversEnum()); +static_assert(casEnumTableCoversEnum()); + +std::string_view dialectWordFromString(std::string_view w, std::string_view what) +{ + /// Parse then re-render, so what a record carries is the table's own spelling rather than the + /// caller's copy of it. + return kTokenTypeWords.toWord(kTokenTypeWords.fromWord(w, what), what); +} + +uint8_t dialectByteFromWord(std::string_view w, std::string_view what) { - switch (t) - { - case TokenType::ETag: return "etag"; - case TokenType::Generation: return "generation"; - case TokenType::Emulated: return "emulated"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS wire: unknown TokenType {}", static_cast(t)); + return static_cast(kTokenTypeWords.fromWord(w, what)); } -TokenType tokenTypeFromWord(std::string_view w, std::string_view what) +std::string_view dialectWordFromByte(uint8_t byte, std::string_view what) { - if (w == "etag") return TokenType::ETag; - if (w == "generation") return TokenType::Generation; - if (w == "emulated") return TokenType::Emulated; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown token type '{}'", what, w); + for (const auto & entry : kTokenTypeWords.entries) + if (static_cast(entry.value) == byte) + return entry.word; + /// The byte comes off persisted media, so an unknown one is malformed data, not a caller bug. + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown dialect byte {}", what, byte); } BlobHashAlgo blobHashAlgoFromWord(std::string_view w, std::string_view what) { - if (w == "ch128") return BlobHashAlgo::CityHash128; - if (w == "xxh3") return BlobHashAlgo::XXH3_128; - if (w == "sha256") return BlobHashAlgo::Sha256; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown blob hash algo '{}'", what, w); + return kBlobHashAlgoWords.fromWord(w, what); } std::string_view objectKindToWord(ObjectKind k) { - switch (k) - { - case ObjectKind::Blob: return "blob"; - } - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS wire: unknown ObjectKind {}", static_cast(k)); + return kObjectKindWords.toWord(k, "CAS wire: ObjectKind"); } ObjectKind objectKindFromWord(std::string_view w, std::string_view what) { - if (w == "blob") return ObjectKind::Blob; - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: unknown object kind '{}'", what, w); + return kObjectKindWords.fromWord(w, what); } -void writeTokenFields(CasJsonWriter & out, bool & first, const Token & t) +void writeTokenFields(CasJsonWriter & out, bool & first, const PersistedEtag & inc) { - writeKey(out, "tt", first); - writeStringValue(out, tokenTypeToWord(t.type)); - writeKey(out, "tv", first); - writeStringValue(out, t.value); + writeStringField(out, SharedWire::token_type, dialectWordFromString(inc.dialect, "wire: dialect"), first); + writeStringField(out, SharedWire::token, inc.value, first); } void writeBlobRefFields(CasJsonWriter & out, bool & first, const BlobRef & r) { - writeKey(out, "ha", first); - writeStringValue(out, blobHashAlgoName(r.algo)); - writeKey(out, "h", first); - writeStringValue(out, codecFor(r.algo).toHex(r.digest)); + writeStringField(out, SharedWire::algo, blobHashAlgoName(r.algo), first); + writeStringField(out, SharedWire::digest, codecFor(r.algo).toHex(r.digest), first); } -void writeManifestRefFields(CasJsonWriter & out, bool & first, std::string_view prefix, const ManifestRef & r) +void writeManifestRefFields(CasJsonWriter & out, bool & first, const ManifestRefWireKeys & keys, const ManifestRef & r) { - /// Unlike the WriteBuffer overload, the two-part key() form appends the prefix and name back - /// to back with no composed String(prefix) + "..." temporary. - out.key(prefix, "me", first); - out.u64StringValue(r.writer_epoch); - out.key(prefix, "mb", first); - out.u64StringValue(r.build_sequence); - out.key(prefix, "mo", first); - out.u64Number(r.manifest_ordinal); + writeU64StringField(out, keys.epoch, r.writer_epoch, first); + writeU64StringField(out, keys.build, r.build_sequence, first); + writeNumberField(out, keys.ord, r.manifest_ordinal, first); } ManifestRef manifestRefFromFields(uint64_t writer_epoch, uint64_t build_sequence, uint64_t manifest_ordinal, @@ -100,4 +91,38 @@ ManifestRef manifestRefFromFields(uint64_t writer_epoch, uint64_t build_sequence return r; } +ManifestRef ManifestRefFields::buildRef(std::string_view what, std::string_view context) const +{ + if (!epoch || !build || !ord) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: {} manifest_ref missing epoch/build/ord", what, context); + return manifestRefFromFields(*epoch, *build, *ord, what, context); +} + +BlobRef BlobRefFields::build(std::string_view what) const +{ + if (!algo_word || !digest_hex) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: blob ref missing algo/digest", what); + const BlobHashAlgo algo = blobHashAlgoFromWord(*algo_word, what); + /// Validate the digest width before calling `fromHex`. A width mismatch otherwise produces + /// `BAD_ARGUMENTS` instead of the `CORRUPTED_DATA` required for malformed serialized input, + /// allowing an invalid record to escape the decoder's fail-closed error contract. + const uint64_t expected_hex_len = blobHashLenFor(algo) * 2; + if (digest_hex->size() != expected_hex_len) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS {}: digest hex width {} does not match algo width {}", what, digest_hex->size(), expected_hex_len); + /// Same fence, same reason as the width check above: a right-width but non-lowercase-hex digest + /// must also surface as CORRUPTED_DATA rather than `DigestCodec::fromHex`'s BAD_ARGUMENTS. Mirrors + /// `JsonObjectReader::readHex128`'s lowercase-hex predicate. + if (std::any_of(digest_hex->begin(), digest_hex->end(), [](char c) { return !isLowercaseHexChar(c); })) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: digest is not lowercase hex, got '{}'", what, *digest_hex); + return BlobRef{algo, codecFor(algo).fromHex(*digest_hex)}; +} + +PersistedEtag TokenFields::build(std::string_view what) const +{ + if (!type_word || !value) + throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS {}: token missing token_type/token", what); + return PersistedEtag{String(dialectWordFromString(*type_word, what)), *value}; +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.h index 8727e6e219b6..9741b258e785 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasWireVocab.h @@ -1,9 +1,12 @@ #pragma once +#include +#include #include #include #include #include #include +#include #include namespace DB::Cas @@ -15,41 +18,99 @@ namespace DB::Cas /// unrecognized value with `CORRUPTED_DATA`; silently choosing a default would turn malformed /// persisted data into a different valid-looking record. -/// Convert a token discriminator to its canonical wire word. Throws `CORRUPTED_DATA` if `t` is not -/// one of the token types understood by this build. -std::string_view tokenTypeToWord(TokenType t); +/// The incarnation-dialect wire vocabulary; coverage is proven in `CasWireVocab.cpp`. It has two +/// persisted encodings -- the word, used by every JSON codec here, and the one byte the condemned-row +/// payload stores -- and both go through this one table so they can never name different sets. +inline constexpr EnumWireTable kTokenTypeWords{{{ + {Dialect::ETag, "etag"}, + {Dialect::Generation, "generation"}, + {Dialect::Emulated, "emulated"}, +}}}; -/// Parse a canonical token-type word. `what` identifies the containing codec or field in the -/// `CORRUPTED_DATA` exception; unknown words are rejected rather than treated as a default type. -TokenType tokenTypeFromWord(std::string_view w, std::string_view what); +/// The `ObjectKind` wire vocabulary; coverage is proven in `CasWireVocab.cpp`. +inline constexpr EnumWireTable kObjectKindWords{{{ + {ObjectKind::Blob, "blob"}, +}}}; + +/// Validate a persisted dialect word and return its canonical spelling. `what` identifies the +/// containing codec or field in the `CORRUPTED_DATA` exception; an unrecognized word is rejected +/// rather than carried into a record no reader could decode. +std::string_view dialectWordFromString(std::string_view w, std::string_view what); + +/// The same vocabulary in the one-byte form the condemned-row payload stores. Both directions are +/// fail-closed: an unrecognized word or byte is `CORRUPTED_DATA`. +uint8_t dialectByteFromWord(std::string_view w, std::string_view what); +std::string_view dialectWordFromByte(uint8_t byte, std::string_view what); /// Parse a canonical blob-hash algorithm word. The write side uses `blobHashAlgoName` directly, so /// this is its fail-closed inverse. `what` identifies the containing codec or field in the /// `CORRUPTED_DATA` exception. BlobHashAlgo blobHashAlgoFromWord(std::string_view w, std::string_view what); -/// Convert an envelope object-kind discriminator to its canonical wire word. Throws -/// `CORRUPTED_DATA` if `k` is not represented by this format. +/// Convert an envelope object-kind discriminator to its canonical wire word. Throws `LOGICAL_ERROR` +/// for an out-of-range enum value. std::string_view objectKindToWord(ObjectKind k); /// Parse a canonical envelope object-kind word. `what` identifies the containing codec or field in /// the `CORRUPTED_DATA` exception; unknown words are rejected rather than treated as a default kind. ObjectKind objectKindFromWord(std::string_view w, std::string_view what); -/// Append the sibling fields `tt` and `tv` to an in-progress JSON object. The caller owns `first`, -/// which must describe the fields already written to that object; the token value is JSON-escaped. -void writeTokenFields(CasJsonWriter & out, bool & first, const Token & t); +/// Append the sibling fields `token_type` and `token` to an in-progress JSON object. The caller owns `first`, +/// which must describe the fields already written to that object; the value is JSON-escaped. The +/// dialect is validated on the way out, so a record can never persist a word its reader would reject. +void writeTokenFields(CasJsonWriter & out, bool & first, const PersistedEtag & inc); -/// Append the sibling fields `ha` and `h` to an in-progress JSON object. The algorithm word and +/// Append the sibling fields `algo` and `digest` to an in-progress JSON object. The algorithm word and /// lowercase digest are canonical, and the digest is rendered at the width required by `r.algo`. void writeBlobRefFields(CasJsonWriter & out, bool & first, const BlobRef & r); -/// Append the three flat `ManifestRef` fields `me`, `mb`, and `mo` to an in-progress JSON object. -/// `prefix` is prepended to each key, allowing the ref codecs to distinguish old and new owner -/// bindings (`ome`/`omb`/`omo` and `nme`/`nmb`/`nmo`) while part manifests and ordinary rows use an -/// empty prefix. The two unbounded `uint64_t` values are decimal JSON strings; the bounded ordinal -/// is a JSON number. All consumers use this exact spelling and representation. -void writeManifestRefFields(CasJsonWriter & out, bool & first, std::string_view prefix, const ManifestRef & r); +/// The `algo`/`digest` and `token_type`/`token` key spellings, named once so `writeBlobRefFields`/`writeTokenFields` +/// and the `match*Fields` collectors below can never drift apart on the literal. +namespace SharedWire +{ + inline constexpr WireKey algo{"algo"}; + inline constexpr WireKey digest{"digest"}; + inline constexpr WireKey token_type{"token_type"}; + inline constexpr WireKey token{"token"}; +} + +/// One `ManifestRef`'s three flat key names. Every bundle spells the SAME wire representation +/// (two decimal-string `uint64_t`s and one JSON-number ordinal); only the key names vary per +/// binding role. Member names carry the semantic role the ref plays (`epoch`/`build`/`ord`); the +/// bundle constants below carry the CURRENT wire spelling for each role. +struct ManifestRefWireKeys +{ + WireKey epoch; + WireKey build; + WireKey ord; +}; + +/// The unprefixed `epoch`/`build`/`ord` spelling used by part manifests, snapshot rows, and the +/// `set_published_at` ref-log op. +inline constexpr ManifestRefWireKeys kBareManifestRefKeys{WireKey{"epoch"}, WireKey{"build"}, WireKey{"ord"}}; +/// The `old_epoch`/`old_build`/`old_ord` spelling for a ref-log owner_transition's OLD binding. +inline constexpr ManifestRefWireKeys kOldManifestRefKeys{WireKey{"old_epoch"}, WireKey{"old_build"}, WireKey{"old_ord"}}; +/// The `new_epoch`/`new_build`/`new_ord` spelling for a ref-log owner_transition's NEW binding. +inline constexpr ManifestRefWireKeys kNewManifestRefKeys{WireKey{"new_epoch"}, WireKey{"new_build"}, WireKey{"new_ord"}}; + +/// One owner binding's key names: the owner-kind word, the ref name, and its nested `ManifestRef` +/// bundle. Only the ref-log owner_transition op uses this bundle (old/new binding sides). +struct BindingWireKeys +{ + WireKey kind; + WireKey ref; + ManifestRefWireKeys manifest; +}; + +/// The `old_kind`/`old_ref`/`old_epoch`/`old_build`/`old_ord` spelling for the OLD binding side. +inline constexpr BindingWireKeys kOldBindingKeys{WireKey{"old_kind"}, WireKey{"old_ref"}, kOldManifestRefKeys}; +/// The `new_kind`/`new_ref`/`new_epoch`/`new_build`/`new_ord` spelling for the NEW binding side. +inline constexpr BindingWireKeys kNewBindingKeys{WireKey{"new_kind"}, WireKey{"new_ref"}, kNewManifestRefKeys}; + +/// Append the three flat `ManifestRef` fields named by `keys` to an in-progress JSON object. The +/// two unbounded `uint64_t` values are decimal JSON strings; the bounded ordinal is a JSON number. +/// All consumers use this exact representation; only the key spelling varies by `keys`. +void writeManifestRefFields(CasJsonWriter & out, bool & first, const ManifestRefWireKeys & keys, const ManifestRef & r); /// Construct a `ManifestRef` from decoded field values and validate the complete domain range: /// nonzero `writer_epoch` and `build_sequence`, and `manifest_ordinal` in @@ -59,4 +120,75 @@ void writeManifestRefFields(CasJsonWriter & out, bool & first, std::string_view ManifestRef manifestRefFromFields(uint64_t writer_epoch, uint64_t build_sequence, uint64_t manifest_ordinal, std::string_view caller, std::string_view what); +/// Collector for one `ManifestRef`'s three flat fields, filled in by repeated calls to +/// `matchManifestRefFields` as a tolerant reader walks an object's keys. `buildRef` checks that the +/// group is all-or-nothing complete, then delegates the completed group to `manifestRefFromFields`, +/// which performs the nonzero and range checks. +struct ManifestRefFields +{ + std::optional epoch; + std::optional build; + std::optional ord; + + bool any() const { return epoch || build || ord; } + + /// `what` names the codec (passed through to `manifestRefFromFields` as its `caller`); `context` + /// names the field being reconstructed (e.g. "descriptor", "committed"). Throws `CORRUPTED_DATA` + /// if the group is not all-or-nothing complete. + ManifestRef buildRef(std::string_view what, std::string_view context) const; +}; + +/// Collector for one `BlobRef`'s two flat fields (`algo`/`digest`), filled in by `matchBlobRefFields`. +struct BlobRefFields +{ + std::optional algo_word; + std::optional digest_hex; + + /// Requires both fields, parses the algorithm word, and checks the digest hex width against the + /// algorithm's width BEFORE calling `fromHex` -- a width mismatch must surface as `CORRUPTED_DATA` + /// (malformed persisted input), not `DigestCodec::fromHex`'s `BAD_ARGUMENTS` (a caller-contract + /// violation). `what` identifies the field in the exception. + BlobRef build(std::string_view what) const; +}; + +/// Collector for one persisted incarnation's two flat fields (`token_type`/`token`), filled in by +/// `matchTokenFields`. +struct TokenFields +{ + std::optional type_word; + std::optional value; + + /// Requires both fields and validates the dialect word. `what` identifies the enclosing codec + /// in `CORRUPTED_DATA` exceptions. + PersistedEtag build(std::string_view what) const; +}; + +/// Each `match*Fields` helper tests `key` against the one or two field names it owns, consumes the +/// value on a match via `r`, and reports whether it recognized the key. None of them loop over an +/// object's keys or validate a completed group -- that is the caller's (tolerant-reader loop) and +/// the collector's `build`/`buildRef` job respectively. Defined inline: a decoder's per-key dispatch +/// is a hot path, so the helpers are header-defined for the per-key dispatch to inline them. + +inline bool matchManifestRefFields(std::string_view key, JsonObjectReader & r, const ManifestRefWireKeys & keys, ManifestRefFields & fields) +{ + if (key == keys.epoch) { fields.epoch = r.readU64String(); return true; } + if (key == keys.build) { fields.build = r.readU64String(); return true; } + if (key == keys.ord) { fields.ord = r.readU64Number(); return true; } + return false; +} + +inline bool matchBlobRefFields(std::string_view key, JsonObjectReader & r, BlobRefFields & fields) +{ + if (key == SharedWire::algo) { fields.algo_word = r.readString(); return true; } + if (key == SharedWire::digest) { fields.digest_hex = r.readString(); return true; } + return false; +} + +inline bool matchTokenFields(std::string_view key, JsonObjectReader & r, TokenFields & fields) +{ + if (key == SharedWire::token_type) { fields.type_word = r.readString(); return true; } + if (key == SharedWire::token) { fields.value = r.readString(); return true; } + return false; +} + } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/README.md b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/README.md index 7c4c99aa36fd..50fe520b649e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/README.md +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/README.md @@ -18,30 +18,53 @@ trailer, followed by a banner-framed raw payload zone for inline file bytes. | Key (under the pool prefix) | Object | Codec | Writer | |---|---|---|---| -| `_pool_meta` | pool identity + floors | `CasPoolMetaFormat` | pool create/admit | -| `cas/ns/stream//_log/…​.zst` | ref transaction log | `CasRefLogFormat` (`.zst`) | writer commit path | -| `cas/ns/stream//_snap/…​.zst` | complete ref table | `CasRefSnapshotFormat` (`.zst`) | writer/GC fold | -| `cas/ns/state//_ckpt` | mutable life checkpoint | `CasRefCkptFormat` | writer/GC fold | +| `_pool_meta` | pool identity + floors (`pool_id`, `blob_header_len`, `gc_shards`, `min_reader_generation`, `algos_used` array) | `CasPoolMetaFormat` | pool create/admit | +| `cas/ns/stream//_log/…​.zst` | ref transaction log (`namespace`, `txn_epoch`/`txn_seq`, optional critical `!prev_epoch`/`!prev_seq`; `set_published_at` uses `ref`/`published_ms`) | `CasRefLogFormat` (`.zst`) | writer commit path | +| `cas/ns/stream//_snap/…​.zst` | complete live ref table (`namespace`, `snapshot_epoch`/`snapshot_seq`, `lifecycle:"live"`; `kind` is `committed`/`precommit`, with `ref` and committed-only `published_ms`) | `CasRefSnapshotFormat` (`.zst`) | writer/GC fold | +| `cas/ns/state//_ckpt` | mutable life checkpoint (`life_epoch`, `committed_epoch`/`committed_seq`, `snapshot_epoch`/`snapshot_seq`, `seal_epoch`/`seal_seq`) | `CasRefCkptFormat` | writer/GC fold | | `cas/ns/state//_files/…​` | namespace-owned raw files | — | upper layers | -| `cas/manifests//-/.zst` | part manifest | `CasPartManifestFormat` | part build | -| blob keys (`CasLayout::blobKey`) | blob envelope + payload | `CasBlobEnvelopeFormat` | uploads | -| blob-meta keys (`CasLayout::blobMetaKey`) | freshness sidecar | `CasBlobMetaFormat` | dedup/GC | -| `gc/state`, `gc/hb` | GC state / leader heartbeat | `CasGcStateFormat` | GC | -| `gc/maintenance_state` | leak-only namespace-janitor cursor | `CasGcMaintenanceStateFormat` | future janitor | -| `gc/gen//attempt//outcomes/…​.zst` | outcome log | `CasGcOutcomesFormat` (`.zst`) | GC | -| `gc/gen//attempt//fold_seal` | fold seal (deterministic) | `CasFoldSealFormat` | GC | -| `gc/gen//…​/runs` | GC source-edge record-stream runs | `CasRecordStreamFormat` | GC | -| `gc/server-roots//{owner,epoch,mount}` | server-root singletons | `CasServerRootFormats` | mount | +| `cas/ref_catalog` | namespace lifecycle catalog (`kind:"entry"`, `ns`, `state`, `life`, `remove_round`, `creator`, `creator_epoch`, `creator_fence`) | `CasRefCatalogFormat` | namespace admission/removal | +| `cas/manifests//-/.zst` | part manifest (`namespace`, `payload_digest`; entry `path`, `place`, `size`) | `CasPartManifestFormat` | part build | +| blob keys (`CasLayout::blobKey`) | blob envelope (`type`, `v`, `tag`, `build`, `time_ms`, `creator`, `op`, `chver`, `ref`) + payload | `CasBlobEnvelopeFormat` | uploads | +| blob-meta keys (`CasLayout::blobMetaKey`) | freshness sidecar (`state`, `condemn_round`, `size`) | `CasBlobMetaFormat` | dedup/GC | +| `gc/state`, `gc/hb` | GC state (`round`, `gc_shards`, `snap_generation`, `snap_pruned_through`, `snap_attempt`, `manifest_sweep_cursor`, `lease_owner`, `lease_seq`) / heartbeat (`owner`, `hb_seq`) | `CasGcStateFormat` | GC | +| `gc/maintenance_state` | leak-only namespace-janitor cursor (`janitor_cursor`) | `CasGcMaintenanceStateFormat` | future janitor | +| `gc/gen//attempt//outcomes/…​.zst` | outcome log (`kind`, `outcome`) | `CasGcOutcomesFormat` (`.zst`) | GC | +| `gc/gen//attempt//fold_seal` | fold seal (deterministic; `generation`/`parent_generation`, `kind`: `ref_life`/`blob_run`/`condemned`) | `CasFoldSealFormat` | GC | +| `gc/gen//…​/runs` | GC source-edge record-stream runs (`ref`, `src`, `mark`; condemned: `pending`, `size`, `condemn_round`, `confirmed`) | `CasRecordStreamFormat` | GC | +| `gc/server-roots//{owner,epoch,mount}` | server-root singletons (`server_uuid`, optional `retired_at_ms`; `next_writer_epoch`; `server_uuid`, `writer_epoch`, `hostname`, `pid`, `started_at_ms`, `seq`, `expires_at_ms`, `min_active_build_sequence`, `gc_fenced`, `write_attempt_id`) | `CasServerRootFormats` | mount | | `roots/…` | raw passthrough (verbatim) | — (never interpreted) | upper layers | ## Codec table Authoritative per-format traits (type string, family, strictness, compression policy, caps) live -in `CasFormat.cpp` (`TRAITS`), asserted complete by `gtest_cas_text_format.cpp`. Key naming: keys -2–5 chars; fixed-width `UInt128` identities = 32-char lowercase hex strings; blob digests = -algo-width hex (two chars per digest byte), rendered with their algo name (`sha256:ab12…`) wherever -a bare hex would be ambiguous; unbounded u64 = decimal strings; bounded counts/lengths/ms-timestamps -= numbers; units documented here per object as codecs land. +in `CasFormat.cpp` (`TRAITS`), asserted complete by `gtest_cas_text_format.cpp`. + +Key naming follows a deliberate split between metadata written once per object and fields repeated +once per record, not a flat character-count budget: + +- metadata written once per object (`namespace`, `writer_epoch`, `blob_header_len`, …) uses + descriptive names; +- fields repeated once per record (`ref`, `mark`, `op`, `class`, `place`, …) use short, semantic + words whose meaning is clear in the record rather than the C++ member name verbatim; +- the fixed `cas_blob` descriptor uses its own separately budgeted compact vocabulary (`tag`, + `build`, `chver`, …), because it must fit before the pool-wide fixed payload offset; +- common framing stays `type`, `v`, and `n`; +- `!` stays the must-understand prefix for critical fields; +- C++ member names obey an asymmetric rule: a member may be fuller than its wire key, never more + cryptic than it. + +Exact full C++ member names everywhere were deliberately rejected — see +`docs/superpowers/specs/2026-08-28-cas-semantic-wire-keys-design.md` ("Rejected alternatives"). +Fixed-width `UInt128` identities render as 32-char lowercase hex strings; blob digests render as +algo-width hex (two chars per digest byte), with their algo name (`sha256:ab12…`) wherever a bare +hex would be ambiguous; unbounded u64 = decimal strings; bounded counts/lengths/ms-timestamps = +numbers; units documented here per object as codecs land. + +`CasWireVocab.{h,cpp}` owns repeated value fields: `BlobRef` uses `algo`/`digest`, a persisted `Etag` +uses the jointly required `token_type`/`token` (the wire key spellings predate, and are independent +of, the C++ type's own name), `ManifestRef` uses `epoch`/`build`/`ord`, and owner-transition bindings +use the corresponding `old_*` and `new_*` key bundles. ## Evolution rules (one screen) @@ -62,18 +85,9 @@ a bare hex would be ambiguous; unbounded u64 = decimal strings; bounded counts/l enforcement. In practice the mismatch never arises: `Always` objects are read via a constructed `.zst`-suffixed key, so a raw body is not GETtable at that key. -## Generation 10 mount-attempt identity {#generation-10-mount-attempt-identity} - -Generation 10 is a breaking, recreate-only change for the unreleased CAS format: - -- `MountLease` adds the required full-word key `write_attempt_id`, encoded as a nonzero 32-character - lowercase `UInt128` hex value. It identifies one holder-originated logical write; all physical - retries reuse the exact body and ID. A GC fence preserves the observed ID, while reclaim and - successor bodies mint a new one. The decoder rejects a missing or zero value. -- `FormatId::MountLease` has a breaking generation-10 change point because its canonical body gained - that required field. -- `FormatId::PoolMeta` has the matching generation-10 change point and reader floor. `decodePoolMeta` - rejects a generation-9 pool before any old mount body can be interpreted without attempt identity. +## Generation history {#generation-history} -There is no generation-9 decoder, compatibility alias, or migration. CAS is pre-release; recreate a -generation-9 pool with a generation-10 writer. +The format's generation history was reset to a flat `{1, 1}` baseline (`G_BUILD == 1`): CAS is +pre-release, carries no persisted data, and pays no compatibility cost for starting the count over. +Every class's `changePoints` begins at generation 1; a future breaking change appends a real entry to +that class's own array and bumps `G_BUILD`, the same way it always has. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.cpp index d018228750da..3170b197968d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasBlobInDegree.cpp @@ -1,4 +1,5 @@ #include +#include #include #include #include @@ -40,11 +41,11 @@ const UInt128 kZeroSourceId{0}; /// Streams a shard's prior source-edge run at O(one block) resident memory: chains the run SEGMENTS the /// caller resolved from the parent seal (`blob_target_runs` filtered to one shard) and exposes a one-row /// lookahead for the fold merge. The prior run carries -/// BOTH surviving edges (`kEdgeActive`) AND the retired `kCondemned` sentinel rows at the zero source id, +/// BOTH surviving edges (`RunMarker::Edge`) AND the retired `RunMarker::Condemned` sentinel rows at the zero source id, /// so the cursor stops at edges AND at condemned rows (exposing the type via `rowType`), while zero-marker /// sentinels are dropped on carry (per-generation, never carried forward). Row/key invariants are enforced -/// while streaming: `kEdgeActive` never at `source_id = 0`; sentinel rows (`kZeroMarker` / -/// `kCondemned`) ONLY at `source_id = 0`; at most one sentinel per blob; an unknown value byte or an empty +/// while streaming: `RunMarker::Edge` never at `source_id = 0`; sentinel rows (`RunMarker::Zero` / +/// `RunMarker::Condemned`) ONLY at `source_id = 0`; at most one sentinel per blob; an unknown value byte or an empty /// payload is `CORRUPTED_DATA`. Resolution uses the exact object references supplied by the caller, so a run /// sealed for generation G that physically lives under an older generation's key is reached /// without key construction. An empty `segments` is the fresh-pool / empty baseline. The row stream is @@ -53,18 +54,18 @@ class PriorEdgeCursor { public: /// The key codec is stateless and self-describing, so a run may freely mix supported hash algorithms. - PriorEdgeCursor(Backend & backend_, const std::vector & segments_) - : backend(backend_), segments(segments_) + PriorEdgeCursor(CasOperation & op_, const std::vector & segments_) + : op(op_), segments(segments_) { advance(); } bool valid() const { return has_current; } const String & key() const { return current_key; } - /// The value byte of the current row: `kEdgeActive` (a surviving edge) or `kCondemned` (a retired + /// The value byte of the current row: `RunMarker::Edge` (a surviving edge) or `RunMarker::Condemned` (a retired /// sentinel row). Zero markers are never surfaced (dropped on carry). - char rowType() const { return current_type; } - /// The decoded retired sentinel for the current row (only valid when `rowType() == kCondemned`). + RunMarker rowType() const { return current_type; } + /// The decoded retired sentinel for the current row (only valid when `rowType() == RunMarker::Condemned`). const CondemnedRow & condemnedRow() const { return current_condemned; } /// Advance to the next surviving edge OR retired sentinel, dropping zero markers, enforcing the @@ -87,40 +88,37 @@ class PriorEdgeCursor SourceEdgeKeyCodec::parse(k, bh, sid); if (p.empty()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS source-edge run: empty row payload"); - const char v = p[0]; + const RunMarker v = runMarkerFromByte(p[0], "CAS source-edge run"); const bool sentinel_key = (sid == kZeroSourceId); if (sentinel_key) { /// A sentinel key carries exactly one row per blob and never an edge. - if (v == kEdgeActive) + if (v == RunMarker::Edge) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS source-edge run: active edge at the reserved sentinel source_id 0"); - if (v != kZeroMarker && v != kCondemned) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS source-edge run: unknown sentinel row type 0x{:02x}", static_cast(v)); if (have_sentinel_blob && sentinel_blob == bh) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS source-edge run: duplicate sentinel row for one blob"); have_sentinel_blob = true; sentinel_blob = bh; - if (v == kZeroMarker) + if (v == RunMarker::Zero) continue; // A zero marker is per-generation and is dropped on carry. /// A retired sentinel: decode and surface it (settled at close-out, not an edge). current_condemned = decodeCondemnedRow(p); current_key = k; - current_type = kCondemned; + current_type = RunMarker::Condemned; has_current = true; return; } /// A non-sentinel key must carry a surviving edge and nothing else. - if (v != kEdgeActive) + if (v != RunMarker::Edge) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS source-edge run: sentinel row type 0x{:02x} at a non-sentinel key", static_cast(v)); current_key = k; - current_type = kEdgeActive; + current_type = RunMarker::Edge; has_current = true; return; } @@ -140,18 +138,18 @@ class PriorEdgeCursor } /// Typed open validates the NDJSON header before any row is consumed. Each row carries its /// own algorithm byte, so no separate width gate is needed. - reader = openSourceEdgeRun(backend, segments[seg_idx].key); + reader = openSourceEdgeRun(op, segments[seg_idx].key); } } private: - Backend & backend; + CasOperation & op; const std::vector & segments; size_t seg_idx = 0; std::optional reader; String current_key; - char current_type = kEdgeActive; + RunMarker current_type = RunMarker::Edge; CondemnedRow current_condemned; bool has_current = false; @@ -191,9 +189,9 @@ void assertValidSourceEdgeId(const UInt128 & source_id) String encodeCondemnedRow(const CondemnedRow & row) { String out; - out.push_back(kCondemned); + out.push_back(runMarkerByte(RunMarker::Condemned)); out.push_back(static_cast((row.delete_pending ? 1 : 0) | (row.marker_confirmed ? 2 : 0))); - out.push_back(static_cast(row.token.type)); + out.push_back(static_cast(dialectByteFromWord(row.token.dialect, "condemned row"))); auto beU64 = [&](uint64_t v) { for (int i = 7; i >= 0; --i) out += static_cast((v >> (8 * i)) & 0xFF); }; beU64(row.condemn_round); beU64(row.size); @@ -209,7 +207,7 @@ CondemnedRow decodeCondemnedRow(std::string_view p) { /// [0]=0x02 [1]=flags [2]=token_type [3..10]=round [11..18]=size [19..20]=len [21..]=value constexpr size_t kFixed = 21; - if (p.size() < kFixed || p[0] != kCondemned) + if (p.size() < kFixed || runMarkerFromByte(p[0], "CAS condemned row") != RunMarker::Condemned) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS condemned row: malformed header"); CondemnedRow row; const uint8_t flags = static_cast(p[1]); @@ -217,10 +215,7 @@ CondemnedRow decodeCondemnedRow(std::string_view p) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS condemned row: unknown flags 0x{:02x}", flags); row.delete_pending = flags & 1; row.marker_confirmed = flags & 2; - const uint8_t type = static_cast(p[2]); - if (type < 1 || type > 3) - throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS condemned row: unknown token_type {}", type); - row.token.type = static_cast(type); + row.token.dialect = String(dialectWordFromByte(static_cast(p[2]), "condemned row")); auto beU64 = [&](size_t off) { uint64_t v = 0; for (int i = 0; i < 8; ++i) v = (v << 8) | static_cast(p[off + i]); return v; }; row.condemn_round = beU64(3); row.size = beU64(11); @@ -248,19 +243,16 @@ bool SourceEdgeRunView::next(String & key, String & payload) key = SourceEdgeKeyCodec::key(rec.ref, rec.source_id); switch (rec.marker) { - case kEdgeActive: - case kZeroMarker: - payload = String(1, rec.marker); + case RunMarker::Edge: + case RunMarker::Zero: + payload = String(1, runMarkerByte(rec.marker)); break; - case kCondemned: + case RunMarker::Condemned: payload = encodeCondemnedRow(CondemnedRow{.delete_pending = rec.delete_pending, .token = rec.token, .size = rec.size, .condemn_round = rec.condemn_round, .marker_confirmed = rec.marker_confirmed}); break; - default: - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS source-edge run: unknown row marker 0x{:02x}", static_cast(rec.marker)); } return true; } @@ -280,14 +272,14 @@ SourceEdgeRunView openSourceEdgeRun(std::string_view bytes) return SourceEdgeRunView(std::make_unique(bytes.data(), bytes.size())); } -SourceEdgeRunView openSourceEdgeRun(Backend & backend, const String & key) +SourceEdgeRunView openSourceEdgeRun(CasOperation & op, const String & key) { - /// Streaming: `getStream` is a forward-only read of the write-once run — nothing is - /// materialized whole (cas_run is object_cap = 0). Absent object => fail-closed. - auto sr = backend.getStream(key); - if (!sr) + /// Streaming: a forward-only read of the write-once run — nothing is materialized whole + /// (cas_run is object_cap = 0). Absent object => fail-closed. + auto stream = op.stream(key, Retry::standard()); + if (!stream) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS source-edge run: object {} is absent", key); - return SourceEdgeRunView(std::move(sr->stream)); + return SourceEdgeRunView(std::move(stream)); } namespace @@ -338,27 +330,39 @@ void SourceEdgeKeyCodec::parse(std::string_view key, BlobRef & ref, UInt128 & so source_id = u128FromBytesBE(String(key.substr(1 + digest_len, 16)), "src-edge run key source_id"); } -void putDeterministicArtifact(Backend & backend, const String & key, const String & bytes) +void putDeterministicArtifact(CasOperation & op, const String & key, const String & bytes) { - if (backend.putIfAbsent(key, bytes).outcome == PutOutcome::PreconditionFailed) + WriteResult result = op.create(key, bytes, Retry::standard()); + if (const auto * conflict = std::get_if(&result)) { - const auto existing = backend.get(key); - if (!existing || existing->bytes != bytes) + /// Only something the resolve read actually OBSERVED can support a corruption verdict. + if (const auto * occupant = std::get_if(&conflict->seen)) + { + if (occupant->bytes == bytes) + return; /// our own deterministic replay; adopt (no-op). throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc: deterministic artifact at {} occupied by divergent bytes (impossible under " "correct operation; refusing to proceed)", key); - /// byte-equal => our own deterministic replay; adopt (no-op). + } + if (std::holds_alternative(conflict->seen)) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS gc: deterministic artifact at {} refused the write but reads as absent (impossible " + "under correct operation; refusing to proceed)", key); + /// Nothing was observed, so the key's state is unknown. Reporting that as corruption would be a + /// deterministic local failure, which nothing above this retries -- it falls through instead. } + const String what = "CAS gc: deterministic artifact at " + key; + orThrow(std::move(result), what); } -void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, +void foldDeltasIntoGeneration(CasOperation & op, const Layout & layout, const std::vector & prior_runs, uint64_t new_generation, uint64_t attempt, uint64_t shard, std::vector scattered, std::vector & out_runs, uint64_t current_round, uint64_t condemn_round, - const std::function(const BlobRef &)> & head_blob, - const std::function(const BlobRef &)> & peek_head, + const BlobHeadFn & head_blob, + const BlobHeadFn & peek_head, const std::function & confirm_condemned_marker, RetiredMergeResult * out_retired, bool suppress_destructive, @@ -389,12 +393,12 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, return a.source_id < b.source_id; }); - PriorEdgeCursor cursor(backend, prior_runs); + PriorEdgeCursor cursor(op, prior_runs); DB::WriteBufferFromOwnString out; SourceEdgeRunWriter writer(out); // sorted NDJSON; byte-deterministic for write-once adoption - // Streaming two-cursor merge over the prior run (surviving edges AND retired kCondemned + // Streaming two-cursor merge over the prior run (surviving edges AND retired RunMarker::Condemned // sentinel rows at the zero source id) and this round's edge deltas (by (blob_hash, source_id)). All // rows for one blob are adjacent in both inputs; the sentinel key (source_id 0) sorts first. We resolve // final presence per edge locally (idempotent: prior present + activate => present; any remove => @@ -427,7 +431,7 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, /// 1. the round-paced floor: a blob condemned in round R cannot graduate before R+1, so a `+1` /// that lands in the same round as the condemnation is always folded before any delete; /// 2. the exact-token delete: a writer that resurrected the blob replaced its incarnation, so a - /// stale token's delete finds a TokenMismatch and removes nothing; + /// stale entry's delete matches nothing and removes nothing; /// 3. THIS: the entry is settled against `indeg` recomputed by the merge that just ran, so an edge /// folded after the condemnation but before the delete pass spares the blob outright -- /// `indeg > 0` wins over `delete_pending`, unconditionally and past the floor. @@ -537,12 +541,12 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, if (cur_edges == 0 && cur_touched && peek_head) { if (const auto hr = peek_head(cur_blob); - hr && hr->exists && hr->token != stale.token) + hr && !stale.token.matches(hr->etag)) { RetiredEntry fresh; fresh.kind = ObjectKind::Blob; fresh.ref = cur_blob; - fresh.token = hr->token; + fresh.token = PersistedEtag::capture(hr->etag); fresh.size = hr->size; fresh.condemn_round = condemn_round; ReplacedEntry re; @@ -560,35 +564,35 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, /// incarnation token for the later exact-token delete; an absent object needs no entry. else if (cur_edges == 0 && cur_touched && head_blob) { - if (const auto hr = head_blob(cur_blob); hr && hr->exists) + if (const auto hr = head_blob(cur_blob); hr) { RetiredEntry fresh; fresh.kind = ObjectKind::Blob; fresh.ref = cur_blob; - fresh.token = hr->token; + fresh.token = PersistedEtag::capture(hr->etag); fresh.size = hr->size; fresh.condemn_round = condemn_round; rmr.still_retired.push_back(std::move(fresh)); } } - /// Emit at most one sentinel row per blob: the `kCondemned` row when the + /// Emit at most one sentinel row per blob: the `RunMarker::Condemned` row when the /// blob is condemned/carried/graduated this pass (still_retired grew for it), else a per-generation - /// `kZeroMarker` when it transitioned to zero this pass but was not condemned (redelete-dropped or + /// `RunMarker::Zero` when it transitioned to zero this pass but was not condemned (redelete-dropped or /// absent-at-condemn). A blob with surviving edges (cur_edges > 0) emits neither — its edge rows /// were appended inline, and a condemned/zeroed blob has NO surviving edges, so appending the /// sentinel now (its key sorts first for the blob, and no edge rows precede it) keeps the run - /// sorted. `still_retired` therefore mirrors exactly the emitted `kCondemned` rows, in order. + /// sorted. `still_retired` therefore mirrors exactly the emitted `RunMarker::Condemned` rows, in order. if (rmr.still_retired.size() > retired_before) { const RetiredEntry & e = rmr.still_retired.back(); - writer.append(SourceEdgeRecord{.ref = cur_blob, .source_id = kZeroSourceId, .marker = kCondemned, + writer.append(SourceEdgeRecord{.ref = cur_blob, .source_id = kZeroSourceId, .marker = RunMarker::Condemned, .delete_pending = e.delete_pending, .token = e.token, .size = e.size, .condemn_round = e.condemn_round, .marker_confirmed = e.marker_confirmed}); } else if (cur_edges == 0 && cur_touched) - writer.append(SourceEdgeRecord{.ref = cur_blob, .source_id = kZeroSourceId, .marker = kZeroMarker}); + writer.append(SourceEdgeRecord{.ref = cur_blob, .source_id = kZeroSourceId, .marker = RunMarker::Zero}); }; auto openBlobIfNeeded = [&](const BlobRef & b) { @@ -622,9 +626,9 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, openBlobIfNeeded(blob_ref); /// A retired sentinel row from the prior run: stash it for close-out settlement. It is not an edge - /// and NEVER a touch — a carried kCondemned row must not force a zero-marker or a peek_head HEAD + /// and NEVER a touch — a carried RunMarker::Condemned row must not force a zero-marker or a peek_head HEAD /// and never a touch. Deltas never key the zero source id, so no delta merges at this key. - if (from_prior && cursor.rowType() == kCondemned) + if (from_prior && cursor.rowType() == RunMarker::Condemned) { cur_condemned = cursor.condemnedRow(); cursor.advance(); @@ -673,7 +677,7 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, if (present) { - writer.append(SourceEdgeRecord{.ref = blob_ref, .source_id = source_id, .marker = kEdgeActive}); + writer.append(SourceEdgeRecord{.ref = blob_ref, .source_id = source_id, .marker = RunMarker::Edge}); ++cur_edges; } } @@ -687,12 +691,12 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, /// seal's RunRef.checksum and verified before any consumer acts on the run. const UInt128 run_checksum = sourceEdgeRunChecksum(run_bytes); const String run_key = layout.blobTargetRunKey(new_generation, attempt, shard, 0); - putDeterministicArtifact(backend, run_key, run_bytes); + putDeterministicArtifact(op, run_key, run_bytes); out_runs.push_back(RunRef{.key = run_key, .checksum = run_checksum, - .shard = shard, .generation = new_generation}); + .shard = shard, .key_generation = new_generation}); } -std::vector zeroInDegree(Backend & backend, const std::vector & runs) +std::vector zeroInDegree(CasOperation & op, const std::vector & runs) { std::vector result; for (const RunRef & run : runs) @@ -700,12 +704,12 @@ std::vector zeroInDegree(Backend & backend, const std::vector +#include #include #include #include @@ -20,13 +20,13 @@ namespace DB::Cas /// In-memory description of a blob incarnation condemned by the in-degree merge. The exact token and /// size are captured from the blob HEAD so the GC caller can issue an exact-token deletion later; -/// `condemn_round` controls round-paced graduation. Entries are decoded from `kCondemned` rows and are +/// `condemn_round` controls round-paced graduation. Entries are decoded from `RunMarker::Condemned` rows and are /// returned through `RetiredMergeResult`; the type itself has no serialized representation. struct RetiredEntry { ObjectKind kind = ObjectKind::Blob; BlobRef ref{}; - Token token; /// the exact incarnation token GC observed (exact-token delete) + PersistedEtag token; /// the exact incarnation GC observed; the delete re-heads and compares it uint64_t size = 0; uint64_t condemn_round = 0; /// the GC round that condemned this incarnation (round-paced /// graduation: an entry graduates only once condemn_round < the @@ -44,6 +44,10 @@ struct RetiredEntry /// graduation (delete_pending rows always carry it). }; +/// The fold's HEAD hooks (`head_blob`, `peek_head`): the caller issues the request on its own admitted +/// operation and reports what it saw, or nothing when the object is absent. +using BlobHeadFn = std::function(const BlobRef &)>; + /// Backend-independent codec for source-edge keys. A key is `algo` (u8), the digest at that algorithm's /// native width, and `source_id` (16 bytes, big-endian). The packed byte order is exactly /// `(BlobRef, source_id)` order, which lets the fold merge compare keys directly. The leading algorithm @@ -67,24 +71,34 @@ UInt128 sourceEdgeId(const ManifestId & id, const String & path); /// producers of real source edges must fail closed on a hash collision with it. void assertValidSourceEdgeId(const UInt128 & source_id); -/// Serialized payload of a condemned source-edge sentinel. The payload retains the full incarnation -/// token, including its type, because deletion must remain exact-token guarded. Its fixed prefix is -/// `[0x02][flags][token_type][round BE64][size BE64][token_len BE16]`, followed by token bytes. +/// Serialized payload of a condemned source-edge sentinel. The payload retains the full incarnation, +/// dialect included, because deletion must remain exact-token guarded. Its fixed prefix is +/// `[0x02][flags][token_type][round BE64][size BE64][token_len BE16]`, followed by token bytes; +/// `token_type` is the dialect byte of `CasWireVocab`'s vocabulary. /// `flags` bit 0 is `delete_pending`, bit 1 is `marker_confirmed`. struct CondemnedRow { bool delete_pending = false; - Token token; // {value, type} — the full token required by exact-token deletion + PersistedEtag token; // the incarnation the exact-token delete must re-observe uint64_t size = 0; uint64_t condemn_round = 0; bool marker_confirmed = false; // durable Condemned meta confirmed (graduation gate) - bool operator==(const CondemnedRow &) const = default; + /// Spelled out rather than defaulted because `PersistedEtag` carries no equality of its + /// own. A field added above belongs here too. + bool operator==(const CondemnedRow & o) const + { + return delete_pending == o.delete_pending + && token.dialect == o.token.dialect && token.value == o.token.value + && size == o.size && condemn_round == o.condemn_round + && marker_confirmed == o.marker_confirmed; + } }; -/// Encode a condemned-row payload. Throws `CORRUPTED_DATA` if the token cannot fit in its u16 length. +/// Encode a condemned-row payload. Throws `CORRUPTED_DATA` if the token cannot fit in its u16 length +/// or the dialect is not one this vocabulary knows. String encodeCondemnedRow(const CondemnedRow & row); -/// Decode and validate a condemned-row payload. Unknown flags, token types, or inconsistent lengths +/// Decode and validate a condemned-row payload. Unknown flags, dialect bytes, or inconsistent lengths /// throw `CORRUPTED_DATA`. CondemnedRow decodeCondemnedRow(std::string_view payload); @@ -112,7 +126,7 @@ class SourceEdgeRunView private: friend SourceEdgeRunView openSourceEdgeRun(std::string_view bytes); - friend SourceEdgeRunView openSourceEdgeRun(Backend & backend, const String & key); + friend SourceEdgeRunView openSourceEdgeRun(CasOperation & op, const String & key); /// Keep the underlying stream alive for the reader, which borrows it rather than owning it. explicit SourceEdgeRunView(std::unique_ptr stream_); @@ -122,19 +136,21 @@ class SourceEdgeRunView /// Open a typed source-edge run. The NDJSON header must identify a `cas_run` of kind `source_edge`; /// otherwise opening fails closed. The memory overload borrows caller-owned bytes. The backend overload -/// streams the write-once object through `getStream`, retaining only one record-sized buffer. +/// streams the write-once object: the reader itself holds one record-sized buffer, but the open stream +/// underneath it buffers on its own terms, unbounded and unmeasured here. SourceEdgeRunView openSourceEdgeRun(std::string_view bytes); -SourceEdgeRunView openSourceEdgeRun(Backend & backend, const String & key); +SourceEdgeRunView openSourceEdgeRun(CasOperation & op, const String & key); /// Store a deterministic write-once artifact (same inputs => byte-identical bytes): the blob in-degree -/// runs and fold seals. `putIfAbsent`; on a `PreconditionFailed` the key is already -/// occupied — `get` it and compare bytes: byte-equal means our own deterministic replay (adopt, no-op), -/// divergent bytes are impossible under correct operation and we fail closed with `CORRUPTED_DATA` -/// rather than let a divergent artifact disagree with the adopted snapshot. Deterministic artifacts are -/// therefore byte-equal-or-`CORRUPTED_DATA`. It is +/// runs and fold seals. `create`; a `Conflict` means the write was refused, and only what its resolve +/// read OBSERVED decides what that means: byte-equal bytes are our own deterministic replay (adopt, +/// no-op), divergent bytes or a key that reads as absent are impossible under correct operation and +/// fail closed with `CORRUPTED_DATA` rather than let a divergent artifact disagree with the adopted +/// snapshot, and an unobserved key proves nothing and is reported as the ordinary refusal it is -- +/// a corruption verdict there would be a deterministic local failure no caller retries. It is /// NOT for observation-bearing artifacts (outcome logs) — those carry HEAD-observed -/// tokens that two observers may legitimately differ on and keep first-durable-write-wins semantics. -void putDeterministicArtifact(Backend & backend, const String & key, const String & bytes); +/// incarnations that two observers may legitimately differ on and keep first-durable-write-wins semantics. +void putDeterministicArtifact(CasOperation & op, const String & key, const String & bytes); /// One source-edge update before merging: the edge `(ref, source_id)`, and whether it is an activation /// (+edge) or a removal (−edge). Idempotent under re-fold at the merge (set membership, not a counter). @@ -193,13 +209,13 @@ struct BlobCandidate /// seal's exact reference is authoritative and key construction is not used. `new_generation`, `attempt`, /// and `shard` name only the output run's key namespace. /// The fresh entry that re-condemns the current -/// token, paired with the STALE entry's token it superseded. Kept as its own struct (rather than a +/// incarnation, paired with the STALE entry's it superseded. Kept as its own struct (rather than a /// field bolted onto `RetiredEntry`) so the common merge element stays slim — only replaced entries -/// carry the extra superseded token. +/// carry the extra superseded incarnation. struct ReplacedEntry { - RetiredEntry fresh; /// the freshly condemned CURRENT token (also pushed into still_retired byte-identically) - Token old_token; /// the superseded (stale) entry's token — what republication replaced + RetiredEntry fresh; /// the freshly condemned CURRENT incarnation (also pushed into still_retired byte-identically) + PersistedEtag old_token; /// the superseded (stale) entry's — what republication replaced }; /// One example of an unmatched-remove delta, kept for the caller's single once-per-round WARNING @@ -218,7 +234,7 @@ struct RetiredMergeResult std::vector still_retired; /// carried + newly-condemned + newly-PENDING entries (the next list) std::vector graduated; /// newly floor-passed this pass — published pending, deleted NEXT pass std::vector spared; /// in-degree recovered — entry dropped - std::vector redelete; /// pending in the PRIOR list — execute deleteExact pre-CAS, drop + std::vector redelete; /// pending in the PRIOR list — execute the exact-incarnation delete pre-CAS, drop std::vector replaced; /// re-condemned CURRENT tokens that superseded a stale entry after republication; caller emits blob_retire_replaced /// Count of `remove == true` deltas that matched no presence for their `(BlobRef, source_id)` key — @@ -324,10 +340,10 @@ struct GcRoundWorkBudget bool outcomeEntryAvailable() const { return max_outcome_entries == 0 || outcome_entries_used < max_outcome_entries; } }; -/// Merge the prior generation's source-edge run with new deltas. The prior run's `kCondemned` rows RIDE +/// Merge the prior generation's source-edge run with new deltas. The prior run's `RunMarker::Condemned` rows RIDE /// the source-edge run itself at the zero-sentinel key (`source_id = 0`), so there is no separate /// `prior_retired` cursor — the prior run IS the retired input. `PriorEdgeCursor` decodes each sentinel -/// `kCondemned` row and hands it to the per-blob close-out (in ascending hash order, exactly the order +/// `RunMarker::Condemned` row and hands it to the per-blob close-out (in ascending hash order, exactly the order /// the old sorted vector had). Settlement rules, in order, per condemned row for blob `h` with post-merge /// in-degree `d`: /// delete_pending (prior pass) -> redelete if d = 0 (the caller executes the exact-token @@ -342,9 +358,9 @@ struct GcRoundWorkBudget /// `confirm_condemned_marker` below): an unconfirmed /// entry is carried unchanged instead; /// d = 0 otherwise -> still_retired, carried byte-unchanged. -/// A carried `kCondemned` row is SETTLEMENT-ONLY: it never sets the blob's `cur_touched` bit, so a +/// A carried `RunMarker::Condemned` row is SETTLEMENT-ONLY: it never sets the blob's `cur_touched` bit, so a /// generation that only carries the row emits no zero-marker and pays no `peek_head` HEAD. The surviving -/// `still_retired` entries are re-emitted as `kCondemned` sentinel rows into the OUTPUT run (one sentinel +/// `still_retired` entries are re-emitted as `RunMarker::Condemned` sentinel rows into the OUTPUT run (one sentinel /// per blob, emitted before the blob's edges since the sentinel key sorts first), so the next generation /// reads them back — `still_retired` mirrors exactly those rows, in the same order. /// When the pass is clamped on any shard, landed-before-cut events may remain unfolded behind the clamp, @@ -381,14 +397,14 @@ struct GcRoundWorkBudget /// for merge-mechanics unit tests only; the real GC round always passes the gate. /// The merge comparator is exactly `(ref.algo, ref.digest, source_id)` (that is, `BlobRef::operator<` /// followed by `source_id`), which is also the raw key order produced by `SourceEdgeKeyCodec`. -void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, +void foldDeltasIntoGeneration(CasOperation & op, const Layout & layout, const std::vector & prior_runs, uint64_t new_generation, uint64_t attempt, uint64_t shard, std::vector scattered, std::vector & out_runs, uint64_t current_round = 0, uint64_t condemn_round = 0, - const std::function(const BlobRef &)> & head_blob = {}, - const std::function(const BlobRef &)> & peek_head = {}, + const BlobHeadFn & head_blob = {}, + const BlobHeadFn & peek_head = {}, const std::function & confirm_condemned_marker = {}, RetiredMergeResult * out_retired = nullptr, bool suppress_destructive = false, @@ -408,6 +424,6 @@ void foldDeltasIntoGeneration(Backend & backend, const Layout & layout, /// shard) and return every blob written at in-degree 0 (the candidates that transitioned to zero). An /// empty `runs` is an empty baseline. Each `RunRef` supplies the exact object key, so resolution never /// reconstructs a key from generation metadata. -std::vector zeroInDegree(Backend & backend, const std::vector & runs); +std::vector zeroInDegree(CasOperation & op, const std::vector & runs); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp index 0a0e7cf2725c..336ab21fbefa 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -26,6 +27,14 @@ #include #include #include +#include + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} namespace ProfileEvents { @@ -62,6 +71,7 @@ namespace ErrorCodes extern const int BAD_ARGUMENTS; extern const int CORRUPTED_DATA; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; } } @@ -81,9 +91,28 @@ void onGcEnumerationPage() /// Defined below; forward-declared so the post-CAS hand-off delete in `runRegularRound` can /// reach the same wholesale LIST-delete helper the retention prune uses. -uint64_t deletePrefixWholesale(Backend & backend, const String & prefix, uint64_t bounded_remaining, +uint64_t deletePrefixWholesale(CasOperation & op, const String & prefix, uint64_t bounded_remaining, bool * out_fully_drained = nullptr); +/// The text an incarnation carries into the event log. Persisted and live incarnations render the +/// same way, so the column speaks one vocabulary whichever half of the pipeline wrote the row. +String renderIncarnation(const PersistedEtag & token) +{ + return token.dialect + ":" + token.value; +} + +/// The label one removal carries into the event log and the outcomes audit rows. +std::string_view removalName(Removal removal) +{ + switch (removal) + { + case Removal::Removed: return "deleted"; + case Removal::Gone: return "absent"; + case Removal::Mismatch: return "replaced"; + } + UNREACHABLE(); +} + } std::set RefPlan::lifeIds() const @@ -218,7 +247,7 @@ RefPlan buildRefWalkPlan(RoundInput && round_input) ++plan.dropped_holds; continue; } - it->second.fold_state.coverage.classification = 4; + it->second.fold_state.coverage.classification = CoverageClass::Clamped; it->second.fold_state.coverage.hold = hold; } for (const auto & [life_id, checkpoint] : ref_scan.checkpoint_observations) @@ -311,6 +340,14 @@ Gc::Gc(PoolPtr store_, UInt128 gc_id_, std::function now_ms_fn_, /// `store->poolConfig()` AFTER the null check above. meta_writer = std::make_unique( store, logger, static_cast(store->poolConfig().gc_meta_pool_size)); + /// The fold's read-ahead pool, built here for the same reason. The queue is UNBOUNDED because the + /// hinting sites throttle themselves against `GcReadAhead::window`; a bounded queue would only + /// move the throttle into `scheduleOrThrowOnError`, blocking the round thread instead of the + /// hint loop that already knows how much it wants in flight. + const size_t read_concurrency = std::max(1, store->poolConfig().gc_read_concurrency); + read_pool = std::make_unique( + CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, CurrentMetrics::LocalThreadScheduled, + /*max_threads*/ read_concurrency, /*max_free_threads*/ read_concurrency, /*queue_size*/ 0); } void Gc::runNamespaceJanitorPage( @@ -321,18 +358,13 @@ void Gc::runNamespaceJanitorPage( NamespaceJanitorResult janitor_result; try { - Backend & backend = store->backend(); + CasRequests & requests = store->openRequests(); const Layout & layout = store->layout(); - NamespaceJanitor janitor(backend, layout, 1000); - const uint64_t admitted_generation = leased_state.lease.seq; - janitor_result = janitor.runOnePage(suppress_destructive, [&] - { - const auto got = backend.get(layout.gcStateKey()); - if (!got) - return false; - const GcState current = decodeGcState(got->bytes); - return current.lease.owner == gc_id && current.lease.seq == admitted_generation; - }); + NamespaceJanitor janitor(requests, layout, 1000); + /// ONE authority read per page, made here rather than from the predicate: the janitor's + /// operation samples its liveness before every request, and a page walks up to a thousand keys. + refreshAuthority(leased_state.lease.seq); + janitor_result = janitor.runOnePage(suppress_destructive, [this] { return authority_held; }); for (const String & anomaly : janitor_result.anomalies) LOG_WARNING(logger, "CAS namespace janitor: {}", anomaly); if (janitor_result.leaked) @@ -348,11 +380,32 @@ void Gc::runNamespaceJanitorPage( t.metric("leaked", janitor_result.leaked); } -RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool allow_steal, UniversePolicy policy) +uint64_t removeChunkWriteOnceOrOneByOne(CasOperation & op, const std::vector & chunk, const Retry & policy) +{ + try + { + op.removeManyWriteOnce(chunk, policy); + return 1; + } + catch (const Exception & e) + { + if (e.code() != ErrorCodes::NOT_IMPLEMENTED) + throw; + for (const WriteOnceKey & key : chunk) + op.removeManyWriteOnce({key}, policy); + /// +1: the failed bulk attempt above is itself a call this helper made. + return 1 + chunk.size(); + } +} + +RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool allow_steal, UniversePolicy policy, + RoundReport * progress) { - RoundReport report; + RoundReport local_report; + RoundReport & report = progress ? *progress : local_report; + report = RoundReport{}; GcState state; - Token state_token; + std::optional state_etag; /// Every exit path waits for this round's meta jobs. The throwing `meta_pool_wait` phase below is /// a protocol barrier -- this round's condemns must be durable no later than the ledger they are @@ -367,7 +420,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// are correlated by `round_id` and not by the round number a follower never learns. { GcPhaseTimer t(phase_sink, "lease"); - report.acquired_lease = acquireOrRenewLease(state, state_token, allow_steal); + report.acquired_lease = acquireOrRenewLease(state, state_etag, allow_steal); t.metric("acquired", report.acquired_lease ? 1 : 0); t.metric("steal_allowed", allow_steal ? 1 : 0); } @@ -393,7 +446,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// attempt is idempotent. const Layout & layout = store->layout(); - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const uint64_t new_round = state.round + 1; /// ONE budget instance for the WHOLE round, threaded into every destructive-or-observability-write @@ -427,12 +480,13 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al t.metric("deleted", drain_result.deleted); } - /// Token-guarded fence-out of dead mounts (liveness only — graduation itself paces on GC + /// Etag-guarded fence-out of dead mounts (liveness only — graduation itself paces on GC /// rounds via `new_round`, not on heartbeat acks). Fencing no longer trusts a predecessor's stamped /// `expires_at_ms` against our wall clock — it /// fences ONLY once `mount_obs` has watched the mount's write-token hold unchanged for the full - /// threshold on THIS leader's own monotonic clock (mirrors `claimMountAwaitingExpiry`'s identical - /// `TTL + Drift` threshold for a mount's own reopen). + /// threshold on THIS leader's own monotonic clock (shares `claimMountAwaitingExpiry`'s + /// `TTL + Drift` formula for a mount's own reopen wait, but with the full renewal period as the + /// cadence term below instead of half of it, so the two thresholds are close, not identical). const uint64_t ttl_ms = static_cast(store->poolConfig().mount_lease_ttl_ms.count()); /// The formula is shared with `claimMountAwaitingExpiry` via /// `mountObservationThresholdMs` -- see its doc comment (CasServerRoot.h). @@ -443,7 +497,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// PUT per newly-fenced mount. { GcPhaseTimer t(phase_sink, "heartbeat_floor"); - const HeartbeatFloor floor = computeHeartbeatFloor(backend, layout, now_ms_fn(), mono_ms_fn(), + const HeartbeatFloor floor = computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn(), stable_threshold_ms, mount_obs); report.fence_outs = floor.fenced_now; if (floor.fenced_now > 0) @@ -539,7 +593,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al { ++rounds_since_last_fold_; report.deferred = true; - /// A DEFER round mints no new round -- unlike the fold path below (CasGc.cpp:642), which sets + /// A DEFER round mints no new round -- unlike the fold path below, which sets /// `report.round = state.round` only AFTER the round's single `gc/state` CAS has committed /// `next.round = new_round` and `state` was reassigned to that committed `next` (so on that /// path `state.round` reads the FRESH round number). Here the round CAS never runs, so `state` @@ -588,7 +642,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al }); /// Capture the PARENT seal's run refs BEFORE fold mutates - /// `state.snap_generation`/`snap_attempt` in-memory (CasGc.cpp:838). We compare these against the + /// `state.snap_generation`/`snap_attempt` in-memory. We compare these against the /// NEW seal's refs post-CAS to detect a ref that moved OFF an already-pruned generation (the /// wholesale prune skipped it while it was still referenced and its cursor advanced past it), and /// hand-off delete that generation's now-unreferenced leftover. Absent parent seal => empty. @@ -607,7 +661,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// The pass performs discovery, windowing, and the three-cursor merge (spare / graduate / condemn). /// It emits phases 5..10 of its own. - FoldResult folded = fold(state, state_token, report, new_round, *walk_plan, policy, round_work_budget); + FoldResult folded = fold(state, state_etag, report, new_round, *walk_plan, policy, round_work_budget); /// THE ROUND'S DESTRUCTIVE GATE, read once, here, and consulted at EVERY destructive site below. /// It is available this early because `fold` computes it (see `FoldResult::suppress_destructive`), @@ -665,52 +719,37 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al suppress_destructive ? kNothingToDelete : merge.redelete; for (const RetiredEntry & entry : redelete_now) { - DeleteOutcome del = backend.deleteExact(layout.blobKey(entry.ref), entry.token); - if (del.created_delete_marker) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CAS gc: delete of blob {} created a delete marker — versioning is enabled " - "on the pool (mis-provisioned; the capability probe must reject this)", blobIdOf(entry.ref)); - - /// A RustFS quirk: a conditional delete (`If-Match`) against an ABSENT - /// object can answer HTTP 412 (precondition failed) instead of 404 — we map that 412 to - /// TokenMismatch. Backend-agnostically disambiguate here: a genuine TokenMismatch means the - /// object exists under a different (fresh) token; if a follow-up HEAD shows the object is - /// gone, the "mismatch" was actually the object being absent — treat it as Absent (NotFound) - /// end-to-end so the `.meta` cleanup below still runs. - bool absent_on_mismatch_quirk = false; - if (del.kind == DeleteOutcome::Kind::TokenMismatch) - { - const HeadResult head = backend.head(layout.blobKey(entry.ref)); - if (!head.exists) - { - del.kind = DeleteOutcome::Kind::NotFound; - absent_on_mismatch_quirk = true; - } - } - - const DeleteClass del_class = classifyDeleteOutcome(del); - const OutcomeKind outcome_kind = del_class == DeleteClass::Deleted ? OutcomeKind::Deleted - : del_class == DeleteClass::Absent ? OutcomeKind::Absent - : OutcomeKind::Replaced; + /// The condemned incarnation is a PERSISTED pair and cannot itself be a precondition, so + /// the round observes the blob and compares the two renderings. Observing first also + /// settles the absent case without spending a conditional delete against a key that is + /// already gone. + const String blob_key = layout.blobKey(entry.ref); + const std::optional observed = op.head(blob_key, Retry::standard()); + Removal del = Removal::Gone; + if (observed) + del = entry.token.matches(observed->etag) + ? op.remove(blob_key, observed->etag, Retry::standard()) + : Removal::Mismatch; + + const OutcomeKind outcome_kind = del == Removal::Removed ? OutcomeKind::Deleted + : del == Removal::Gone ? OutcomeKind::Absent + : OutcomeKind::Replaced; OutcomeEntry outcome{.kind = entry.kind, .ref = entry.ref, .token = entry.token, .outcome = outcome_kind}; - const String del_outcome{deleteClassName(del_class)}; - /// The single content-delete site is attributable per row. TokenMismatch (a writer - /// recreated the incarnation) is terminal-OK: the fresh incarnation is a live object. + const String del_outcome{removalName(del)}; + /// The single content-delete site is attributable per row. A mismatch (a writer recreated + /// the incarnation) is terminal-OK: the fresh incarnation is a live object. EventEmitter{*store}.emit([&](CasEvent & e) { e.type = CasEventType::BlobDelete; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(entry.ref); - e.token = entry.token.value; + e.token = renderIncarnation(entry.token); e.round = new_round; e.gen = generation; e.outcome = del_outcome; - e.reason = absent_on_mismatch_quirk - ? "delete_pending published by a prior pass; exact-token delete (pre-CAS) " - "(delete returned token-mismatch but the object is absent — backend 412-on-absent quirk)" - : "delete_pending published by a prior pass; exact-token delete (pre-CAS)"; + e.reason = "delete_pending published by a prior pass; exact-incarnation delete (pre-CAS)"; e.detail = {{"condemn_round", std::to_string(entry.condemn_round)}, - {"key", layout.blobKey(entry.ref)}}; + {"key", blob_key}}; }); /// The audit row is observability only -- the delete above already executed regardless of /// this cap. Skipping it here bounds the per-shard `GcOutcomes` body without skipping or @@ -722,12 +761,12 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al } ++report.redeleted; ProfileEvents::increment(ProfileEvents::CASGCRetiredRedeleted); - /// Drop the per-hash meta only on Deleted/NotFound — a Replaced (TokenMismatch) outcome - /// means a writer already resurrected a fresh incarnation at this hash (INV-1), and that - /// writer's own republication path already flipped the meta back to Clean; blindly deleting here - /// would race that legitimate Clean write for no reason (the meta is advisory, but there is no - /// reason to touch it on that path at all). - if (del_class == DeleteClass::Deleted || del_class == DeleteClass::Absent) + /// Drop the per-hash meta only on a removal or a proven absence — a mismatch means a + /// writer already resurrected a fresh incarnation at this hash, and that writer's + /// own republication path already flipped the meta back to Clean; blindly deleting here + /// would race that legitimate Clean write for no reason (the meta is advisory, but there is + /// no reason to touch it on that path at all). + if (del == Removal::Removed || del == Removal::Gone) { meta_writer->scheduleConfirmedMetaDelete(entry.ref); } @@ -750,7 +789,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al e.type = CasEventType::GcRecheckVerdict; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(entry.ref); - e.token = entry.token.value; + e.token = renderIncarnation(entry.token); e.round = new_round; e.gen = generation; e.outcome = "spared"; @@ -790,7 +829,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al e.type = CasEventType::GcRecheckVerdict; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(entry.ref); - e.token = entry.token.value; + e.token = renderIncarnation(entry.token); e.round = new_round; e.gen = generation; e.outcome = "pending"; @@ -812,13 +851,13 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al e.type = CasEventType::BlobRetireReplaced; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(entry.ref); - e.token = entry.token.value; + e.token = renderIncarnation(entry.token); e.round = new_round; e.gen = generation; e.outcome = "replaced"; e.reason = "current object token differs from the retired entry — republication replaced the " "incarnation; superseded the stale entry and re-condemned the current token"; - e.detail = {{"superseded_token", replaced.old_token.value}}; + e.detail = {{"superseded_token", renderIncarnation(replaced.old_token)}}; }); /// The supersede is ALSO a blob entering the retired set fresh (a re-condemn of the /// CURRENT token) — write the meta Condemned exactly like a fresh `head_blob` condemn would, @@ -837,12 +876,21 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al { const String key = layout.outcomesKey(generation, attempt, new_round, shard); const String body = sealObject(FormatId::GcOutcomes, encodeOutcomeLog(log)); - if (backend.putIfAbsent(key, body).outcome == PutOutcome::PreconditionFailed) + WriteResult written = op.create(key, body, Retry::standard()); + if (const auto * conflict = std::get_if(&written)) { - const auto existing = backend.get(key); + /// The conflict's observation IS the read that used to follow the refused create. Only + /// something that read actually OBSERVED can support a verdict about the key; a resolve + /// read that settled nothing says nothing about whether the object is there. + if (std::holds_alternative(conflict->seen)) + throw Exception(ErrorCodes::ABORTED, + "CAS gc: the create of the outcome log at {} was refused and its resolve read " + "observed nothing, so what the key holds is unknown", key); + const auto * existing = std::get_if(&conflict->seen); if (!existing) throw Exception(ErrorCodes::ABORTED, - "CAS gc: outcome log at {} vanished between putIfAbsent and read", key); + "CAS gc: outcome log at {} refused the create and its resolve read observed {}", + key, detail::renderObservation(conflict->seen)); if (existing->bytes != body) { try { log = decodeOutcomeLog(openObject(FormatId::GcOutcomes, existing->bytes)); } @@ -853,6 +901,8 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al } } } + else + orThrow(std::move(written), fmt::format("CAS gc: outcome log at {}", key)); for (const OutcomeEntry & o : log.entries) { switch (o.outcome) @@ -898,7 +948,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al } /// Retired-in-snapshot — there is NO separate retired-list object to publish anymore. The - /// round's surviving condemned entries were already sealed as `kCondemned` rows inside the fold's + /// round's surviving condemned entries were already sealed as `RunMarker::Condemned` rows inside the fold's /// `blob_target_runs` (durable before this CAS, via `putDeterministicArtifact`), and the per-shard /// `condemned_summary` the seal carries makes the next round's graduation/carry decisions zero-I/O. @@ -920,13 +970,13 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// current shard's run back at an older generation's key). Retention must never reclaim these. std::set referenced_generations; for (const RunRef & r : folded.fold_seal.blob_target_runs) - referenced_generations.insert(r.generation); + referenced_generations.insert(r.key_generation); /// ALSO protect every generation the PARENT (currently-adopted, pre-fold) seal references /// (`parent_seal_runs`, captured above): this prune runs BEFORE the round's own gc/state CAS below, so /// a losing leader must not destroy what the winning leader's already-adopted seal still points at — /// pre-CAS destructive actions may only rely on PREVIOUSLY PUBLISHED state (triage #5). for (const RunRef & r : parent_seal_runs) - referenced_generations.insert(r.generation); + referenced_generations.insert(r.key_generation); /// Retention floor uses THIS round's (post-fold) `generation`, so `gc_snapshot_generations_to_keep` /// keeps exactly that many generations back from the current one. If this round's `gc/state` CAS /// then LOSES, the prune reclaimed one generation deeper than the durably-adopted generation would @@ -938,12 +988,22 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al round_commit_timer->metric("generations_visited", next.snap_pruned_through - pruned_through_before); round_commit_timer->metric("pruned_through", next.snap_pruned_through); round_commit_timer->metric("generations_referenced", referenced_generations.size()); - const CasResult res = backend.casPut(layout.gcStateKey(), encodeGcState(next), state_token); - if (res.outcome != CasOutcome::Committed) + WriteResult commit = op.replace(layout.gcStateKey(), encodeGcState(next), *state_etag, + Retry::standard()); + if (const auto * conflict = std::get_if(&commit)) + { + /// A refused precondition whose resolve read settled nothing proves only that this commit did + /// not apply -- naming a competing leader would assert something nobody observed. + if (std::holds_alternative(conflict->seen)) + throw Exception(ErrorCodes::ABORTED, + "CAS gc round: the gc/state commit was refused and its resolve read observed nothing; " + "retry next round"); throw Exception(ErrorCodes::ABORTED, - "CAS gc round: gc/state moved during the round (another leader advanced it); retry next round"); + "CAS gc round: gc/state moved during the round (another leader advanced it, observed {}); " + "retry next round", detail::renderObservation(conflict->seen)); + } + state_etag = orThrow(std::move(commit), "CAS gc round commit"); state = std::move(next); - state_token = res.token; report.round = state.round; round_commit_timer->metric("round", report.round); round_commit_timer->metric("generation", generation); @@ -961,7 +1021,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// Post-CAS reference-parent HAND-OFF DELETE. `pruneSupersededGenerations` SKIPS a /// generation the live seal still references AND advances `snap_pruned_through` PAST it - /// (CasGc.cpp:1066 computes the cursor as `g - 1` after the loop increments `g` over every skipped + /// (`pruneSupersededGenerations` computes the cursor as `g - 1` after the loop increments `g` over every skipped /// generation). So once a skipped generation is behind the cursor, the wholesale prune NEVER revisits /// it — a ref that later moves off it would strand that generation's WHOLE prefix (fold seal, retired/ /// outcomes sets, all shards' runs), not just the single carried run object. Reclaim it HERE, now that @@ -977,7 +1037,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al uint64_t objects_reclaimed = 0; std::set new_referenced_generations; for (const RunRef & r : folded.fold_seal.blob_target_runs) - new_referenced_generations.insert(r.generation); + new_referenced_generations.insert(r.key_generation); std::set handed_off; /// dedupe: multiple parent refs can share one generation /// GATED like every other destructive site, and it is also the FIRST destructive site of the @@ -1000,11 +1060,11 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al for (const RunRef & old_ref : handoff_candidates) { /// Only generations the wholesale prune already passed AND that no live ref still pins. - if (old_ref.generation > state.snap_pruned_through) + if (old_ref.key_generation > state.snap_pruned_through) continue; /// not yet pruned-through: the normal prune will reclaim it when it ages out - if (new_referenced_generations.contains(old_ref.generation)) + if (new_referenced_generations.contains(old_ref.key_generation)) continue; /// still referenced by a (possibly different-shard) live ref: keep it - if (!handed_off.insert(old_ref.generation).second) + if (!handed_off.insert(old_ref.key_generation).second) continue; /// already reclaimed this round via another shard's ref /// `bounded_remaining` draws from the hand-off's OWN reserve, never `UINT64_MAX` and never /// `pruneSupersededGenerations`' shared remainder: this hand-off is a ONE-SHOT event (see the @@ -1018,13 +1078,13 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al if (remaining == 0) break; const uint64_t reclaimed = deletePrefixWholesale( - backend, layout.gcGenPrefix(old_ref.generation), remaining); + op, layout.gcGenPrefix(old_ref.key_generation), remaining); round_work_budget.handoff_prefix_wholesale_objects_used += reclaimed; objects_reclaimed += reclaimed; LOG_TRACE(logger, "CAS GC hand-off: generation {} moved out of the live seal below the retention cursor " "({} objects) — post-CAS wholesale reclaim (the prune had skipped it while referenced)", - old_ref.generation, reclaimed); + old_ref.key_generation, reclaimed); } t.metric("generations_reclaimed", handed_off.size()); t.metric("objects_reclaimed", objects_reclaimed); @@ -1039,6 +1099,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// never re-derived by this pipeline, converting a bounded burst into a permanent leak. It drains /// the whole of `folded.mf_cleanup` every round it runs; only a crash (or the destructive-suppression /// gate below) leaves an entry for the orphan-manifest sweep to reclaim later. + /// The bodies go in batch requests of write-once keys; see the block. /// /// PHASE 15/18 `manifest_deletes`. { @@ -1048,32 +1109,61 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// sealed AND taken on a round that could prove its frontier -- an unprovable round's `-1` may /// itself be the observation that is missing an owner elsewhere, so deleting the body on it is /// exactly the irreversible step the gate exists to withhold. - static const std::map kNoManifestCleanup; - const std::map & mf_cleanup_now = + static const std::map kNoManifestCleanup; + const std::map & mf_cleanup_now = suppress_destructive ? kNoManifestCleanup : folded.mf_cleanup; + + /// Chunks of write-once keys, one request each, with no per-key precondition: a manifest key is + /// never written twice, so the body at it is the one the fold observed or nothing. The engine + /// reissues a failed chunk whole; a chunk that exhausts its policy throws here, and the chunks + /// before it are already recorded below. A key that one of the exhausted chunk's own attempts + /// did delete is not recorded either: deletion and recording are all-or-nothing per request, + /// never per key, so the next round's fold sees that key as already gone. The etag the fold + /// observed rides the event as information only. + const size_t chunk_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); uint64_t attempted = 0; - for (const auto & [id, token] : mf_cleanup_now) + uint64_t requests = 0; + std::vector chunk; + std::vector *> chunk_entries; + const auto flush = [&] { - ++attempted; - const DeleteOutcome mdel = backend.deleteExact(layout.manifestKey(id), token); /// NotFound/TokenMismatch tolerated - const DeleteClass mdel_class = classifyDeleteOutcome(mdel); - if (mdel_class == DeleteClass::Deleted) - ++report.manifests_deleted; - EventEmitter{*store}.emit([&](CasEvent & e) + if (chunk.empty()) + return; + /// A backend without `DeleteObjects` (GCS) falls back to one admitted delete per key here; + /// `chunk_entries`' per-key bookkeeping below is unaffected either way -- it counts objects + /// that are gone after this call returns, not how many requests it took to get them there. + requests += removeChunkWriteOnceOrOneByOne(op, chunk, Retry::standard()); + for (const auto * entry : chunk_entries) { - e.type = CasEventType::ManifestDelete; - e.namespace_ = id.root_namespace.string(); - e.object_kind = CasEventObjectKind::Manifest; - e.object_hash = manifestRefDebugString(id.ref); - e.token = token.value; - e.round = new_round; - e.gen = generation; - e.outcome = String{deleteClassName(mdel_class)}; - e.reason = "owner-removed manifest body; exact-token delete after decrements adopted"; - }); + ++report.manifests_deleted; + EventEmitter{*store}.emit([&](CasEvent & e) + { + e.type = CasEventType::ManifestDelete; + e.namespace_ = entry->first.root_namespace.string(); + e.object_kind = CasEventObjectKind::Manifest; + e.object_hash = manifestRefDebugString(entry->first.ref); + e.token = entry->second.render(); + e.round = new_round; + e.gen = generation; + e.outcome = "deleted_or_absent"; + e.reason = "owner-removed manifest body; batch delete of a write-once key after decrements adopted"; + }); + } + chunk.clear(); + chunk_entries.clear(); + }; + for (const auto & entry : mf_cleanup_now) + { + ++attempted; + chunk.push_back(layout.writeOnceManifestKey(entry.first)); + chunk_entries.push_back(&entry); + if (chunk.size() >= chunk_keys) + flush(); } + flush(); t.metric("attempted", attempted); - t.metric("deleted", report.manifests_deleted - manifests_deleted_before); + t.metric("accepted", report.manifests_deleted - manifests_deleted_before); + t.metric("requests", requests); t.metric("suppressed", suppress_destructive ? 1 : 0); } @@ -1105,25 +1195,31 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al ManifestSweepResult & sweep = folded.orphan_sweep; for (const ManifestSweepResult::Nomination & nomination : sweep.nominations) { - const DeleteOutcome outcome = backend.deleteExact(nomination.key, nomination.token); - const DeleteClass outcome_class = classifyDeleteOutcome(outcome); + /// The nominated incarnation is persisted, so the body is observed and the two renderings + /// compared before the removal names a precondition. + const std::optional observed = op.head(nomination.key, Retry::standard()); + Removal outcome = Removal::Gone; + if (observed) + outcome = nomination.token.matches(observed->etag) + ? op.remove(nomination.key, observed->etag, Retry::standard()) + : Removal::Mismatch; EventEmitter{*store}.emit([&](CasEvent & e) { e.type = CasEventType::ManifestDelete; e.namespace_ = nomination.id.root_namespace.string(); e.object_kind = CasEventObjectKind::Manifest; e.object_hash = nomination.key; - e.token = nomination.token.value; + e.token = renderIncarnation(nomination.token); e.round = new_round; e.gen = generation; - e.outcome = String{deleteClassName(outcome_class)}; - e.reason = "orphan-manifest sweep: source edges retired and adopted before exact-token delete"; + e.outcome = String{removalName(outcome)}; + e.reason = "orphan-manifest sweep: source edges retired and adopted before exact-incarnation delete"; }); - if (outcome.kind == DeleteOutcome::Kind::TokenMismatch) + if (outcome == Removal::Mismatch) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS orphan sweep: manifest key {} changed token after exact GET; immutable manifest " - "identity suffered illegal ABA, retained replacement", nomination.key); - if (outcome_class == DeleteClass::Deleted) + "CAS orphan sweep: manifest key {} changed incarnation after the exact read; immutable " + "manifest identity suffered illegal ABA, retained replacement", nomination.key); + if (outcome == Removal::Removed) ++sweep.deleted; else ++sweep.skipped; @@ -1133,6 +1229,8 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al t.metric("list_budget_keys", store->poolConfig().manifest_sweep_list_budget_keys); t.metric("suppressed", suppress_destructive ? 1 : 0); t.metric("listed", sweep.listed); + t.metric("floor_lookups", sweep.floor_lookups); + t.metric("floor_reads", sweep.floor_reads); t.metric("deleted", sweep.deleted); t.metric("skipped", sweep.skipped); t.metric("undecodable", sweep.undecodable); @@ -1164,10 +1262,9 @@ void Gc::reportStuckRemovals(const RefPlan & plan, uint64_t current_round) } } -bool Gc::foldManifestEdges(const ManifestId & id, int sign, std::vector & deltas, - std::map & mf_cleanup, uint32_t txn_ordinal) +bool Gc::foldManifestEdges(GcReadAhead & reads, const ManifestId & id, int sign, std::vector & deltas, + std::map & mf_cleanup, uint32_t txn_ordinal) { - Backend & backend = store->backend(); const Layout & layout = store->layout(); const String key = layout.manifestKey(id); @@ -1177,7 +1274,11 @@ bool Gc::foldManifestEdges(const ManifestId & id, int sign, std::vector fail closed) ProfileEvents::increment(ProfileEvents::CASRefManifestBodyFoldGets); /// one body GET per manifest fold @@ -1191,8 +1292,8 @@ bool Gc::foldManifestEdges(const ManifestId & id, int sign, std::vectortoken); /// owner removed: defer exact-token body delete to recheck + mf_cleanup.emplace(id, got->etag); /// owner removed: defer the exact body delete to recheck return true; } -Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(const std::map & ref_tables, +Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(GcReadAhead & reads, + const std::map & ref_tables, const CasRefCatalog::Snapshot & catalog_cut) { /// Read the checkpoint of every namespace in the round's catalog cut, every namespace `ref_tables` @@ -1259,7 +1361,12 @@ Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(const std::mapbackend(); + /// + /// READS UNDER THE CALLER'S ADMISSION, not one of its own. This function used to admit a fresh + /// operation, which meant that a fence moving mid-round would hand it a NEWER generation than the + /// round holds and let it read on regardless; taking the caller's read-ahead makes the same fence + /// movement fail this read the way it fails every other read of the round. Strictly the + /// fail-closed direction, and it is why there is no `admit` here any more. const Layout & layout = store->layout(); std::set witness_namespaces; @@ -1269,7 +1376,16 @@ Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(const std::map witness_keys; + witness_keys.reserve(witness_namespaces.size()); for (const String & ns_str : witness_namespaces) { const RootNamespace ns{ns_str}; @@ -1282,13 +1398,30 @@ Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(const std::mapns != ns || (entry_it->state != NsState::Live && entry_it->state != NsState::Removing)) continue; - const String ckpt_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(entry_it->ns, entry_it->incarnation)); + witness_keys.push_back( + {ns_str, layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(entry_it->ns, entry_it->incarnation))}); + } + + size_t next_hint = 0; + const auto topUpWitnessHints = [&] + { + while (next_hint < witness_keys.size() && reads.pending() < reads.window()) + reads.hintRead(witness_keys[next_hint++].ckpt_key); + }; + + CheckpointWitnesses out; + for (const WitnessKey & witness_key : witness_keys) + { + const String & ns_str = witness_key.ns; + const String & ckpt_key = witness_key.ckpt_key; + topUpWitnessHints(); /// THE GET AND THE DECODE ARE SPLIT HERE, rather than taken together through `readCkpt`, so the /// catch below can scope to the DECODE ALONE. Wrapping the read too would turn a transport /// failure -- which says nothing about this object and everything about the round's ability to /// read anything -- into a per-namespace hold, silently narrowing a pool-wide outage to one - /// namespace. A backend throw still propagates and fails the round, exactly as it always did. - const std::optional got = backend.get(ckpt_key); + /// namespace. A backend throw still propagates and fails the round, exactly as it always did -- + /// a read-ahead worker's failure is rethrown by the take below, at this same site. + const std::optional got = reads.takeRead(ckpt_key); /// ABSENT IS NORMAL AND IS NOT A WITNESS: a namespace has no `_ckpt` until its first snapshot /// publication commits, and one that 404s mid-round is a namespace being reclaimed. Neither says /// anything about which ids exist, so neither may hold the walk -- and neither may throw @@ -1329,7 +1462,7 @@ Gc::CheckpointWitnesses Gc::readCheckpointWitnesses(const std::map> Gc::newestFoldSealRef() { - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); const String gen_prefix = layout.gcGenPrefix(0); const String top = gen_prefix.substr(0, gen_prefix.size() - 2); /// ".../gc/gen/" @@ -1341,13 +1474,13 @@ std::optional> Gc::newestFoldSealRef() std::set listed_generations; bool listed_anything = false; std::optional> newest; - forEachListedKey(backend, top, [&](const ListedKey & k) + op.forEachListedKey(top, [&](const ListedKey & k) { listed_anything = true; const size_t from = top.size(); const size_t gen_end = k.key.find('/', from); if (gen_end == String::npos) - return; + return true; uint64_t generation = 0; try { @@ -1355,10 +1488,11 @@ std::optional> Gc::newestFoldSealRef() } catch (...) // NOLINT(bugprone-empty-catch) { - return; /// foreign key shape under `gc/gen` is debris, not a generation number + return true; /// foreign key shape under `gc/gen` is debris, not a generation number } listed_generations.insert(generation); - }, 1000, onGcEnumerationPage); + return true; + }, Retry::standard(), 1000, onGcEnumerationPage); const uint64_t listed_max_generation = listed_generations.empty() ? 0 : *listed_generations.rbegin(); /// STEP DOWN THROUGH THE GENERATIONS THE LISTING ITSELF REPORTED until one carries a seal. The @@ -1460,11 +1594,11 @@ std::optional> Gc::newestFoldSealRef() Gc::GenerationSealProbe Gc::probeGenerationForSeal(uint64_t generation) { - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); GenerationSealProbe probe; - forEachListedKey(backend, layout.gcGenPrefix(generation), [&](const ListedKey & k) + op.forEachListedKey(layout.gcGenPrefix(generation), [&](const ListedKey & k) { probe.generation_exists = true; /// ANY object proves this generation was minted /// Parse a candidate attempt out of the path and then PROVE it by rebuilding the key: only a @@ -1474,11 +1608,11 @@ Gc::GenerationSealProbe Gc::probeGenerationForSeal(uint64_t generation) static constexpr std::string_view kAttempt = "/attempt/"; const size_t a_begin = k.key.find(kAttempt); if (a_begin == String::npos) - return; + return true; const size_t a_from = a_begin + kAttempt.size(); const size_t a_end = k.key.find('/', a_from); if (a_end == String::npos) - return; + return true; uint64_t attempt = 0; try { @@ -1486,13 +1620,14 @@ Gc::GenerationSealProbe Gc::probeGenerationForSeal(uint64_t generation) } catch (...) // NOLINT(bugprone-empty-catch) { - return; /// foreign key shape is debris, not an attempt + return true; /// foreign key shape is debris, not an attempt } if (layout.foldSealKey(generation, attempt) != k.key) - return; + return true; if (!probe.seal_attempt || *probe.seal_attempt < attempt) probe.seal_attempt = attempt; - }, 1000, onGcEnumerationPage); + return true; + }, Retry::standard(), 1000, onGcEnumerationPage); return probe; } @@ -1536,12 +1671,18 @@ void Gc::FoldResult::FrontierDeficit::count(FrontierUnproven reason) } } -Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & report, +Gc::FoldResult Gc::fold(GcState & state, std::optional & /*state_etag*/, + RoundReport & report, uint64_t current_round, const RefPlan & walk_plan, UniversePolicy policy, GcRoundWorkBudget & work_budget) { - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); + /// The fold's read-ahead. It fetches through `op`'s own admitted generation and hands every result + /// back at the site that would otherwise have read inline, so the walk's order, its counters, its + /// holds and its events are what they were; only the moment of the fetch moves. At + /// `gc_read_concurrency` 1 it hints nothing and every take IS the original inline read. + GcReadAhead reads(op, store->openRequests(), *read_pool, store->poolConfig().gc_read_concurrency); FoldResult result; /// 1. Group the round's one enumeration of `cas/ns/stream/` (taken before the defer decision) into @@ -1583,13 +1724,13 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// `cleanupRefObjects` and terminal-evidence attribution -- retains both the chosen incarnation and /// the lifecycle/absence distinction instead of re-reading or reducing the catalog independently. result.catalog_cut = catalog_snapshot; - /// THE POSITIVE EMPTY-UNIVERSE PROOF (see the destructive gate below). `token` is guaranteed by + /// THE POSITIVE EMPTY-UNIVERSE PROOF (see the destructive gate below). `etag` is guaranteed by /// `CasRefCatalog::read` on every operational path -- absence there is `CORRUPTED_DATA`, never an /// empty snapshot -- but the check stays here so this fails closed if a bootstrap/test snapshot /// ever reaches this line. `entries` (not `live_incarnation`, which drops `Creating`) is the right /// source: a catalog holding only `Creating` rows must NOT read as an empty universe, and `entries` /// is the one view that still carries those rows. - result.catalog_cut_proved_empty = catalog_snapshot.token.has_value() && catalog_snapshot.catalog.entries.empty(); + result.catalog_cut_proved_empty = catalog_snapshot.etag.has_value() && catalog_snapshot.catalog.entries.empty(); /// A malformed ref-object key or namespace aborts ref folding for the whole round: the /// round produces no ref delta, advances no cursor, and authorizes no destructive work -- recorded as @@ -1662,17 +1803,41 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & result.fold_seal.ref_lives = walk_plan.successorFoldStates(); /// Retired-in-snapshot: the prior generation's condemned entries RIDE the source-edge run as - /// `kCondemned` sentinel rows, so the round no longer reads any separate retired-list object — + /// `RunMarker::Condemned` sentinel rows, so the round no longer reads any separate retired-list object — /// the parent seal's `blob_target_runs` ARE the retired input. The per-gc-shard `condemned_summary` /// the seal carries below is distilled from the `still_retired` rows each shard re-emits, making the /// next round's `graduationDue` / pure-carry decisions zero-I/O. const uint64_t condemn_round = state.round + 1; result.retired_merge.resize(state.gc_shards); + /// HEAD READ-AHEAD FOR THE REDUCE PHASE. `head_candidates[shard]` is filled in that phase with the + /// blobs the merge can bring to in-degree zero, in the merge's own ascending key order; `head_blob` + /// below tops the hints up a window deep before each take. It is EMPTY everywhere else, the whole of + /// intake included, so every take outside that phase is the plain inline HEAD. + /// + /// Hints are issued from INSIDE the lambda rather than in one burst at phase start, so the requests + /// this can ever add are bounded by one window past the last candidate the merge actually reaches. + /// A superset that overshoots badly therefore costs a window, not its own size -- which matters + /// because nothing bounds a round's condemnation count. + std::vector> head_candidates(state.gc_shards); + size_t head_hint_shard = 0; + size_t next_head_hint = 0; + const auto topUpHeadHints = [&] + { + const std::vector & shard_candidates = head_candidates[head_hint_shard]; + while (next_head_hint < shard_candidates.size() && reads.pending() < reads.window()) + reads.hintHead(layout.blobKey(shard_candidates[next_head_hint++])); + }; + /// Condemn-time observation: ONE HEAD per new zero-transition captures the exact incarnation token /// the eventual delete carries (absent => a prior landed delete => nothing to condemn). Emits the /// Candidate trail (IndegZero / GcRetireObserve / BlobRetire) exactly where the decision is made. - const auto head_blob = [&](const BlobRef & ref) -> std::optional + /// + /// THE READ-AHEAD NEVER RUNS THIS LAMBDA, only feeds it. Everything below the HEAD is + /// side-effecting -- the trail, the counters, the condemn-marker write -- and running it over a + /// SUPERSET would stamp `Condemned` on blobs this round never condemns, forcing a live writer to + /// republish each one. The prefetch is a bare HEAD; the decision stays here. + const auto head_blob = [&](const BlobRef & ref) -> std::optional { EventEmitter{*store}.emit([&](CasEvent & e) { @@ -1683,19 +1848,20 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & e.gen = state.snap_generation + 1; e.reason = "last folded owner edge dropped; in-degree reached 0"; }); - const HeadResult observed = backend.head(layout.blobKey(ref)); + topUpHeadHints(); + const std::optional observed = reads.takeHead(layout.blobKey(ref)); EventEmitter{*store}.emit([&](CasEvent & e) { e.type = CasEventType::GcRetireObserve; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(ref); - e.token = observed.exists ? observed.token.value : ""; + e.token = observed ? observed->etag.render() : ""; e.round = condemn_round; e.gen = state.snap_generation + 1; - e.outcome = observed.exists ? "present" : "absent"; - e.reason = "zero-in-degree candidate; HEAD-observe the current token"; + e.outcome = observed ? "present" : "absent"; + e.reason = "zero-in-degree candidate; observe the current incarnation"; }); - if (!observed.exists) + if (!observed) return std::nullopt; ++report.candidates; ++report.condemned; @@ -1705,21 +1871,22 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & e.type = CasEventType::BlobRetire; e.object_kind = CasEventObjectKind::Blob; e.object_hash = blobIdOf(ref); - e.token = observed.token.value; + e.token = observed->etag.render(); e.round = condemn_round; e.gen = state.snap_generation + 1; e.outcome = "retired"; e.reason = "condemned zero-in-degree candidate; entering the current retired list"; }); - HeadResult adjusted = observed; - adjusted.size = retiredLogicalSize(ObjectKind::Blob, observed.size, store->poolMeta().blob_header_len); + Meta adjusted = *observed; + adjusted.size = retiredLogicalSize(ObjectKind::Blob, observed->size, store->poolMeta().blob_header_len); /// This candidate unconditionally becomes a fresh `RetiredEntry` in `closeBlob` (the ONLY /// caller of `head_blob`) whenever this lambda returns a value — so this is exactly the round's /// side-effecting condemn site. Write the meta Condemned so the writer's point-read gate /// sees it; a successful write records the in-process (hash, token) confirmation the graduation /// gate consumes (`scheduleCondemnMarkerWrite` captures everything BY VALUE — never by reference /// to `cur_blob`, which the fold's tight streaming loop mutates while the job is queued). - meta_writer->scheduleCondemnMarkerWrite(ref, observed.token, condemn_round, adjusted.size); + meta_writer->scheduleCondemnMarkerWrite(ref, PersistedEtag::capture(observed->etag), + condemn_round, adjusted.size); return adjusted; }; @@ -1729,12 +1896,21 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// hook is `head_blob` above, reserved for a genuinely NEW zero-in-degree candidate. A supersede's /// own event is `blob_retire_replaced`, emitted once below from `merge.replaced`. Plain HEAD, no /// events, no counters. - const auto peek_head = [&](const BlobRef & ref) -> std::optional + /// + /// AND IT IS NOT READ AHEAD, deliberately, unlike `head_blob`'s. This HEAD is not data handed to a + /// decision made elsewhere -- it IS the decision: `hr && !stale.token.matches(hr->etag)` is the + /// supersede branch itself, so observing earlier narrows the window in which a republication can be + /// seen and would genuinely change which entries supersede. The consequence of a missed supersede is + /// benign (the stale entry graduates and its exact-token delete mismatches, so reclamation is + /// delayed, never wrong), but "only the moment of the fetch moves, never a decision" is the property + /// this whole read-ahead is worth trusting for, and it is not worth spending on the rare blob that + /// carries a condemned row AND is touched again in the same round. + const auto peek_head = [&](const BlobRef & ref) -> std::optional { - HeadResult hr = backend.head(layout.blobKey(ref)); - if (!hr.exists) + std::optional hr = op.head(layout.blobKey(ref), Retry::standard()); + if (!hr) return std::nullopt; - hr.size = retiredLogicalSize(ObjectKind::Blob, hr.size, store->poolMeta().blob_header_len); + hr->size = retiredLogicalSize(ObjectKind::Blob, hr->size, store->poolMeta().blob_header_len); return hr; }; @@ -1754,7 +1930,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & return true; try { - if (const auto lm = loadMeta(backend, layout, entry.ref); lm && lm->meta.state == MetaState::Condemned) + if (const auto lm = loadMeta(op, layout, entry.ref); lm && lm->meta.state == MetaState::Condemned) { meta_writer->noteCondemnMarkerDurable(entry.ref, entry.token); /// memoize for a round-CAS-abort replay return true; @@ -1772,7 +1948,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// resurrected this exact hash under a FRESH token between the original swallowed write and this /// retry, the retry stamps `Condemned` over that writer's live, uncondemned incarnation. This is /// never destructive -- the eventual exact-token delete is a no-op against the fresh token - /// (`DeleteOutcome::TokenMismatch`/`NotFound`) -- worst case the resurrecting writer's later + /// (it finds a different incarnation, or none) -- worst case the resurrecting writer's later /// same-token adopter sees stale `Condemned` metadata and republishes once unnecessarily. meta_writer->scheduleCondemnMarkerWrite(entry.ref, entry.token, entry.condemn_round, entry.size); return false; @@ -1853,7 +2029,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// absent, no listed id above it => this namespace's frontier this round (normal end); /// absent, a listed id above it => impossible under contiguity, so the store is lying or a /// durable record was lost: HOLD the namespace at - /// classification 4 with its cursor unmoved. + /// classification `Clamped` with its cursor unmoved. /// /// Epochs are crossed only over a consumed `EpochSeal` (INV-2): the seal folds as an applied table /// no-op (probe B2 `produced=false`) and the next epoch's start is `{E', 1}`, reached through the @@ -1873,6 +2049,15 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// It also carries probe B1's two numbers -- reported on EVERY healthy round, so /// "logs_accounted always equals logs_applied" becomes an observable property of the table rather than /// a claim in a comment. + /// + /// THE REQUESTS ARE THE SAME ONES; ONLY THEIR TIMING MOVED. Four sites below hand their keys to the + /// round's `GcReadAhead` before the walk reaches them -- the checkpoints, each namespace's first walk + /// position, this epoch's next positions, and a decoded log's manifest edges -- so the phase's round + /// trips overlap instead of running strictly one after another. Every take happens where the inline + /// read happened, in the same order, and increments the same counters, which is why this row's + /// semantic metrics are identical at any `gc_read_concurrency`. Its S3 VERB counts are not: a request + /// a worker performed lands on that worker's ProfileEvents, the same gap `meta_pool_wait` has always + /// had. Read `CASGCReadAheadHit`/`Miss`/`Wasted` on this row for the read-ahead's own behaviour. std::optional intake_timer; intake_timer.emplace(phase_sink, "fold_ref_intake"); uint64_t intake_tables_changed = 0; @@ -1891,7 +2076,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// The round's SECOND witness source, independent of the listing -- see `readCheckpointWitnesses` /// for what it decides and why a listing alone cannot decide it. Its `undecodable` half names the /// namespaces whose `_ckpt` is present and unreadable; each of those is HELD below, and only those. - const CheckpointWitnesses checkpoints = readCheckpointWitnesses(ref_tables, catalog_snapshot); + const CheckpointWitnesses checkpoints = readCheckpointWitnesses(reads, ref_tables, catalog_snapshot); const std::map & checkpoint_witness = checkpoints.witnesses; /// WHICH NAMESPACES THIS ROUND WALKS -- i.e. THE ROUND'S UNIVERSE, the set the destructive gate owes @@ -2012,12 +2197,75 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & ++intake_tails_below_cursor; } + /// READ-AHEAD OF EACH NAMESPACE'S FIRST WALK POSITION, in walk order, kept a window deep. On a wide + /// pool this is the phase's shape: many namespaces, one read each, previously taken strictly one + /// after another. + /// + /// The key is recomputed from exactly the inputs the walk below uses -- the sealed cursor, the + /// catalog entry and the checkpoint grounding -- and ONLY where the walk will actually read it. A + /// namespace whose first position sits above `committed_through` is refused by the ceiling test + /// before any read (this is the ordinary QUIET namespace: its frontier is proved by the ceiling, + /// and it costs no request at all today), so hinting it would ADD a request the round never makes. + /// A namespace whose checkpoint is unusable is held without reading, and is skipped here for the + /// same reason. + std::vector first_walk_keys; + first_walk_keys.reserve(walk_targets.size()); + for (const WalkTarget & target : walk_targets) + { + if (checkpoints.undecodable.contains(target.ns)) + continue; + const RootNamespace target_ns{target.ns}; + const auto target_entry_it = std::lower_bound( + catalog_snapshot.catalog.entries.begin(), catalog_snapshot.catalog.entries.end(), target_ns, + [](const CatalogEntry & entry, const RootNamespace & needle) { return entry.ns < needle; }); + if (target_entry_it == catalog_snapshot.catalog.entries.end() || target_entry_it->ns != target_ns) + continue; + + std::optional target_checkpoint; + if (const auto it = checkpoints.recovery_checkpoints.find(target.ns); + it != checkpoints.recovery_checkpoints.end()) + target_checkpoint = it->second; + + std::optional target_grounding; + try + { + target_grounding = chooseRecoveryGrounding(std::optional{*target_entry_it}, target_checkpoint); + } + catch (const Exception &) + { + continue; /// the walk below holds this namespace without reading; so does the hint pass + } + if (!target_grounding->committed_through) + continue; + + const auto target_cursor_it = parent_ref_lives.find(target.life_id); + const RefTxnId target_cursor = target_cursor_it != parent_ref_lives.end() + ? target_cursor_it->second.coverage.last_folded_ref_id : RefTxnId{}; + std::optional target_expected; + if (target_cursor != RefTxnId{}) + target_expected = RefTxnId{target_cursor.writer_epoch, target_cursor.ref_sequence + 1}; + else if (target_checkpoint && target_checkpoint->life_epoch) + target_expected = RefTxnId{*target_checkpoint->life_epoch, 1}; + if (!target_expected || *target_grounding->committed_through < *target_expected) + continue; + + first_walk_keys.push_back(layout.refLogKey( + NamespaceLifeId::fromCatalogEntry(target_ns, target.life_id), *target_expected)); + } + size_t next_first_walk_hint = 0; + const auto topUpFirstWalkHints = [&] + { + while (next_first_walk_hint < first_walk_keys.size() && reads.pending() < reads.window()) + reads.hintRead(first_walk_keys[next_first_walk_hint++]); + }; + for (const WalkTarget & target : walk_targets) { const String & ns_str = target.ns; const RefTableListing & listing = *target.listing; if (ref_folding_aborted) break; + topUpFirstWalkHints(); const RootNamespace ns{ns_str}; /// Every walk target came out of this round's own catalog read, so its incarnation is the REAL /// one and the life below is minted from a catalog entry rather than guessed from a key. @@ -2062,7 +2310,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & cursor_it != parent_ref_lives.end() ? cursor_it->second.coverage.hold : std::nullopt; RefCoverage cov; - cov.classification = 0; + cov.classification = CoverageClass::Absent; bool table_changed = false; /// THE FRONTIER PROOF for this namespace, and there is exactly one thing that establishes it: /// the walk read the expected-next position by exact key, found it ABSENT, and no witness put @@ -2193,7 +2441,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & const RefTxnId & witness) -> std::optional { const EpochCrossResult crossing = - crossEpochFromSeal(backend, layout, ns, from_seal, seal_proven, witness, life); + crossEpochFromSeal(op, layout, ns, from_seal, seal_proven, witness, life); intake_absent_probes += crossing.absent_probes; /// a failed crossing pays its reads too ProfileEvents::increment(ProfileEvents::CASRefLogBodyGets, crossing.body_gets); if (crossing.outcome == EpochCrossOutcome::StartInvalid) @@ -2300,7 +2548,21 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// GET + decode the expected record. Absence is the decision point of the whole walk, and /// an invalid body is a per-namespace hold: the key belongs to exactly one namespace, so it /// can never be grounds for discarding another namespace's fold. - const auto got = backend.get(layout.refLogKey(life, *expected)); + /// + /// LOOKAHEAD over this epoch's next arithmetic positions, and never past the ceiling the + /// test above enforces: `committed_through` was snapshotted before the walk and bounds + /// what this round may read at all, so every position hinted here is one this walk goes on + /// to read unless something stops it first. A hold or an epoch crossing stops it, leaving + /// at most a window's worth of bodies fetched and untaken -- bounded, counted, and never a + /// read the sequential walk would not have made. + for (uint64_t ahead_k = 1; ahead_k <= reads.window(); ++ahead_k) + { + const RefTxnId ahead{expected->writer_epoch, expected->ref_sequence + ahead_k}; + if (*grounding->committed_through < ahead) + break; + reads.hintRead(layout.refLogKey(life, ahead)); + } + const auto got = reads.takeRead(layout.refLogKey(life, *expected)); if (!got) { ++intake_absent_probes; @@ -2391,12 +2653,21 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// re-fold would then clamp on that missing body forever. A missing manifest body is a per-table /// CLAMP (barrier), never a round abort: keep the cursor below THIS log and re-read it next /// round. A removed precommit whose body never existed emitted no edge -- skip, no clamp. + /// + /// The decode above named every manifest key this log folds, so they are fetched together + /// here and taken one at a time in edge order below. A log with a single edge gains + /// nothing; a merge or a mutation log with dozens turns dozens of serial round trips into + /// one. A clamp mid-log leaves the rest of this log's bodies fetched and untaken, which is + /// the bounded waste the read-ahead counts. + for (const RefManifestEdge & edge : edges) + reads.hintRead(layout.manifestKey(edge.manifest_id)); + std::vector log_deltas; - std::map log_mf_cleanup; + std::map log_mf_cleanup; for (const RefManifestEdge & edge : edges) { ProfileEvents::increment(ProfileEvents::CASRefEmittedEdges); /// one manifest-edge event - if (foldManifestEdges(edge.manifest_id, edge.change, log_deltas, log_mf_cleanup, + if (foldManifestEdges(reads, edge.manifest_id, edge.change, log_deltas, log_mf_cleanup, txn_ordinal)) continue; @@ -2574,7 +2845,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & : (same_position ? UINT32_MAX : 0); effective->next_retry_round = current_round + 1; cov.hold = effective; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; ++intake_tables_held; /// A held namespace is unproven BY DEFINITION -- the hold names a position the walk could /// not resolve, so everything at or above it is unaccounted. Stated here rather than left to @@ -2584,7 +2855,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & unproven_reason = FoldResult::FrontierUnproven::Held; } else - cov.classification = table_changed ? 2 : 1; + cov.classification = table_changed ? CoverageClass::Folded : CoverageClass::Unchanged; result.fold_seal.ref_lives.at(target.life_id).coverage = cov; ++result.frontier_namespaces; @@ -2643,7 +2914,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & for (const WalkTarget & target : walk_targets) { RefCoverage cov; - cov.classification = 1; + cov.classification = CoverageClass::Unchanged; if (const auto pit = parent_ref_lives.find(target.life_id); pit != parent_ref_lives.end()) { cov.last_folded_ref_id = pit->second.coverage.last_folded_ref_id; @@ -2653,7 +2924,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & if (pit->second.coverage.hold) { cov.hold = pit->second.coverage.hold; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; } } RefLifeFoldState & ref_life_state = result.fold_seal.ref_lives.at(target.life_id); @@ -2671,7 +2942,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// It counts the CUT ARITHMETICALLY, not by listed ids. Under arithmetic intake a listed-id count is /// not even the right question: a hint hole means a round legitimately applies records the listing /// never mentioned, so the old recomputation would report fewer logs than folded and fail every - /// healthy round on a lying store -- it would have made this task's own fix unshippable. + /// healthy round on a lying store. /// /// BE HONEST ABOUT WHAT IS LEFT. The old formula could disagree with reality because it was derived /// from a different source (the listing) than the counter. This one is derived from the runs the @@ -2818,7 +3089,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// is deterministic (same refs for the same inputs), so seal determinism / crash-replay adoption hold. /// An empty delta with a NON-EMPTY retired list still runs the merge: settlement must happen every /// pass (carried/graduated/redeleted entries), and that pass reads the run to recompute in-degrees. - /// Distill one shard's `condemned_summary` entry from the `kCondemned` rows it re-emitted this pass + /// Distill one shard's `condemned_summary` entry from the `RunMarker::Condemned` rows it re-emitted this pass /// (`still_retired` mirrors those rows exactly). Folding shards call this; it makes the next /// round's `graduationDue` and pure-carry decisions read only the seal, never a run. auto summarize = [](const std::vector & still) -> CondemnedSummary @@ -2996,7 +3267,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// cut and `_ckpt` frontier the round's own universe came from -- which is exactly what an /// authoritative universe means, and is why this is the gate's term and not a separate one. universe_authoritative, - &work_budget); + &work_budget, read_pool.get(), store->poolConfig().gc_read_concurrency); for (const ManifestSweepResult::Nomination & nomination : result.orphan_sweep.nominations) orphan_source_retirements.insert( orphan_source_retirements.end(), @@ -3004,6 +3275,39 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & nomination.source_retirements.end()); } + /// WHICH BLOBS THE MERGE CAN CLOSE AT ZERO. `closeBlob` HEADs under `cur_edges == 0 && cur_touched`: + /// no key of the blob ended present, and at least one was touched. Every prior-run edge that survives + /// increments `cur_edges`, so reaching zero requires each of them to be killed at its own key by a + /// `-1` delta or a source retirement -- which puts a removal-last verdict in the map below -- while + /// any activation-last verdict leaves an edge standing and is the exclusion clause. Retirements are + /// folded in AFTER the deltas because the merge applies them that way, unconditionally: a key whose + /// deltas end in an activation but which a retirement then clears is a candidate too. + /// + /// IT IS A SUPERSET, AND NOTHING MAY COME TO DEPEND ON IT BEING EXACT. A blob named here that keeps + /// an untouched prior edge costs one HEAD the merge never takes; a candidate this set misses is + /// HEADed inline, which is simply the behaviour with no read-ahead at all. Both are counted, and + /// neither is asserted. + /// + /// PLACED HERE, not at the top of the phase: `orphan_source_retirements` is decided just above by + /// the sweep, and a retirement is a removal like any other. The round's cut is frozen well before + /// this point -- intake has finished and `deltas` has taken its final form -- so no HEAD is issued + /// before the round knows what it folded. + { + std::map, bool> last_verdict_is_remove; + for (const BlobDelta & delta : deltas) + last_verdict_is_remove[{delta.ref, delta.source_id}] = delta.remove; + for (const BlobSourceRetirement & retirement : orphan_source_retirements) + last_verdict_is_remove[{retirement.ref, retirement.source_id}] = true; + + std::set removed; + std::set surviving_add; + for (const auto & [edge, is_remove] : last_verdict_is_remove) + (is_remove ? removed : surviving_add).insert(edge.first); + for (const BlobRef & ref : removed) + if (!surviving_add.contains(ref)) + head_candidates[blobShard(ref, state.gc_shards)].push_back(ref); + } + if (state.gc_shards == 1) { /// SINGLE-SHARD PATH (gc_shards == 1). Every blob routes to shard 0, so the entire delta stream @@ -3018,8 +3322,10 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & else { /// Either a real delta or a non-empty retired input: run the merge (empty deltas still settle - /// the kCondemned rows riding the parent run). The prior runs are the parent seal's shard-0 refs. - foldDeltasIntoGeneration(backend, layout, priorRunsFor(0), + /// the RunMarker::Condemned rows riding the parent run). The prior runs are the parent seal's shard-0 refs. + head_hint_shard = 0; + next_head_hint = 0; + foldDeltasIntoGeneration(op, layout, priorRunsFor(0), new_generation, attempt, /*shard*/0, std::move(deltas), result.fold_seal.blob_target_runs, current_round, condemn_round, head_blob, peek_head, @@ -3061,8 +3367,10 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & /// A reducer owns exactly one disjoint shard. Two replicas may run reducers for DIFFERENT /// shards concurrently (CasGcScheduler ownership); their run-key namespaces never collide. std::vector shard_runs; + head_hint_shard = shard; + next_head_hint = 0; foldDeltasIntoGeneration( - backend, layout, priorRunsFor(shard), new_generation, attempt, shard, + op, layout, priorRunsFor(shard), new_generation, attempt, shard, std::move(buckets[shard]), shard_runs, current_round, condemn_round, head_blob, peek_head, confirm_condemned_marker, @@ -3189,7 +3497,7 @@ Gc::FoldResult Gc::fold(GcState & state, Token & /*state_token*/, RoundReport & t.metric("seal_cleanup_evidence", std::count_if( result.fold_seal.ref_lives.begin(), result.fold_seal.ref_lives.end(), [](const auto & item) { return item.second.cleanup_evidence.has_value(); })); - putDeterministicArtifact(backend, layout.foldSealKey(new_generation, attempt), seal_body); + putDeterministicArtifact(op, layout.foldSealKey(new_generation, attempt), seal_body); } /// One-pass round: the fold NO LONGER CASes gc/state. (new_generation, attempt) are adopted @@ -3246,7 +3554,7 @@ void Gc::cleanupRefObjects( if (suppress_destructive) return; - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); if (!folded.catalog_cut) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS GC ref cleanup: fold result carries no catalog cut"); @@ -3267,43 +3575,42 @@ void Gc::cleanupRefObjects( const CatalogEntry & observed_entry = *entry_it; const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry_it->ns, entry_it->incarnation); - /// Current-life ref cleanup is not the dead-life janitor: every irreversible key delete must - /// still be licensed by the SAME complete catalog observation and GC lease that adopted the - /// fold. Re-read both after the target HEAD and immediately before `deleteExact`. A moved token, - /// changed row/life, missing or unreadable authority object, or changed owner/sequence stops the - /// whole cleanup pass. Continuing with another row/key would turn a refusal into a fallback. - const auto deleteRefObject = [&](const String & key) + /// Current-life ref cleanup is not the dead-life janitor: every irreversible delete is still + /// licensed by the SAME complete catalog observation and GC lease that adopted the fold, + /// re-read immediately before the deletes it licenses. The unit + /// of licence is one chunk of write-once keys: a `_log` or `_snap` key is published once at a + /// life-qualified key and never rewritten, so there is nothing to HEAD, and the plan below is + /// derived from durable state a successor leader derives too. A moved incarnation, changed + /// row/life, missing or unreadable authority object, or changed owner/sequence stops the whole + /// cleanup pass; continuing with another chunk would turn a refusal into a fallback. + const auto authorityHolds = [&](const String & first_key) -> bool { - const HeadResult h = backend.head(key); - if (!h.exists) - return true; - try { - const CasRefCatalog::Snapshot current_catalog = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot current_catalog = CasRefCatalog::read(op, layout); current_catalog.life_index.throwIfAmbiguous("CAS GC ref cleanup revalidation"); const auto current_entry_it = std::lower_bound( current_catalog.catalog.entries.begin(), current_catalog.catalog.entries.end(), ns, [](const CatalogEntry & entry, const RootNamespace & needle) { return entry.ns < needle; }); const std::optional current_life = current_catalog.life_index.resolve(life.incarnation); - if (current_catalog.token != folded.catalog_cut->token + if (current_catalog.etag != folded.catalog_cut->etag || current_entry_it == current_catalog.catalog.entries.end() || current_entry_it->ns != ns || *current_entry_it != observed_entry || !current_life || *current_life != life) { LOG_DEBUG(logger, - "CAS GC ref cleanup stopped before deleting '{}': catalog observation/life moved", - key); + "CAS GC ref cleanup stopped before the chunk starting at '{}': catalog observation/life moved", + first_key); return false; } - const auto current_state_object = backend.get(layout.gcStateKey()); + const auto current_state_object = op.read(layout.gcStateKey(), Retry::standard()); if (!current_state_object) { LOG_WARNING(logger, - "CAS GC ref cleanup stopped before deleting '{}': mandatory gc/state is absent", - key); + "CAS GC ref cleanup stopped before the chunk starting at '{}': mandatory gc/state is absent", + first_key); return false; } const GcState current_state = decodeGcState(current_state_object->bytes); @@ -3311,21 +3618,17 @@ void Gc::cleanupRefObjects( || current_state.lease.seq != adopted_lease.seq) { LOG_DEBUG(logger, - "CAS GC ref cleanup stopped before deleting '{}': GC fence moved", - key); + "CAS GC ref cleanup stopped before the chunk starting at '{}': GC fence moved", first_key); return false; } } catch (const std::exception & e) { LOG_WARNING(logger, - "CAS GC ref cleanup stopped before deleting '{}': authority revalidation failed: {}", - key, e.what()); + "CAS GC ref cleanup stopped before the chunk starting at '{}': authority revalidation failed: {}", + first_key, e.what()); return false; } - - backend.deleteExact(key, h.token); - ProfileEvents::increment(ProfileEvents::CASRefCleanupObjectsDeleted); /// cleanup object deletion return true; }; @@ -3348,7 +3651,7 @@ void Gc::cleanupRefObjects( std::optional retained_log_proof; try { - retained_log_proof = readCheckpointSnapshotBase(backend, layout, life, *checkpoint).predecessor_seal_id; + retained_log_proof = readCheckpointSnapshotBase(op, layout, life, *checkpoint).predecessor_seal_id; } catch (const Exception & e) { @@ -3360,30 +3663,42 @@ void Gc::cleanupRefObjects( const RefCleanupPlan plan = planRefCleanup( listing, durable_cursor, checkpoint_snapshot_id, retained_log_proof); + std::vector cohort; + cohort.reserve(plan.deletable_logs.size() + plan.deletable_snapshots.size()); for (const RefTxnId & log_id : plan.deletable_logs) - { - /// Cumulative per-round cap, never amortized against the per-key fail-close - /// validation `deleteRefObject` performs (HEAD + catalog re-read + gc/state re-read before - /// every exact delete stays exactly as expensive per key as before). Exhaustion simply stops - /// the round's cleanup pass here; `planRefCleanup` recomputes the SAME remaining candidates - /// from durable state next round, so nothing here needs its own cursor. - if (!work_budget.refCleanupAvailable()) - return; - if (!deleteRefObject(layout.refLogKey(life, log_id))) - return; - ++work_budget.ref_cleanup_objects_used; - } + cohort.push_back(layout.writeOnceRefLogKey(life, log_id)); for (const RefTxnId & snap_id : plan.deletable_snapshots) { - /// Task 5's rule, asserted where it is acted on rather than only where it is computed: the - /// snapshot the checkpoint names is the one a recovering reader will sample, so it must + /// The snapshot the checkpoint names is the one a recovering reader will sample, so it must /// survive every cleanup that the same checkpoint authorized. chassert(snap_id < checkpoint_snapshot_id); + cohort.push_back(layout.writeOnceRefSnapshotKey(life, snap_id)); + } + + const size_t chunk_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); + for (size_t begin = 0; begin < cohort.size(); ) + { + /// Cumulative per-round cap in KEYS, exactly as before; a chunk is cut to what remains. The + /// plan recomputes the same remaining candidates from durable state next round, so nothing + /// here needs its own cursor. if (!work_budget.refCleanupAvailable()) return; - if (!deleteRefObject(layout.refSnapshotKey(life, snap_id))) + size_t end = std::min(cohort.size(), begin + chunk_keys); + if (work_budget.max_ref_cleanup_objects != 0) + end = std::min(end, begin + (work_budget.max_ref_cleanup_objects - work_budget.ref_cleanup_objects_used)); + std::vector chunk(cohort.begin() + begin, cohort.begin() + end); + if (!authorityHolds(chunk.front().str())) return; - ++work_budget.ref_cleanup_objects_used; + /// A backend without `DeleteObjects` (GCS) falls back to one admitted delete per key here. + /// The budget and the profile event below count OBJECTS in `chunk`, which is the same + /// `chunk.size()` whichever way `removeChunkWriteOnceOrOneByOne` actually sent them. + removeChunkWriteOnceOrOneByOne(op, chunk, Retry::standard()); + work_budget.ref_cleanup_objects_used += chunk.size(); + ProfileEvents::increment(ProfileEvents::CASRefCleanupObjectsDeleted, chunk.size()); /// cleanup object deletion + /// Advance by what was actually sent, not the nominal chunk size: the budget cap above can + /// truncate a chunk short of `chunk_keys`, and advancing by the full stride would skip the + /// untried remainder instead of retrying it next iteration. + begin = end; } } } @@ -3393,12 +3708,12 @@ namespace /// GC-metadata wholesale delete of every object under `prefix`. Returns the number of objects deleted. /// `bounded_remaining` caps how many objects this call may delete (0 => stop immediately, deleting none). /// -/// Token source: the in-memory and S3 backends surface a per-key token through `list` -/// (`supportsListTokens()`), so `deleteExact` straight from the listed token; otherwise HEAD first. +/// Precondition source: the in-memory and S3 backends surface a per-key incarnation through `list` +/// (`supportsListTokens()`), so the removal goes straight from the listed one; otherwise observe first. /// -/// 404 / NotFound is FAIL-OPEN: an object that vanished between LIST and delete (a concurrent crashed +/// An absence is FAIL-OPEN: an object that vanished between LIST and delete (a concurrent crashed /// attempt, or a racing prune) is already reclaimed — never throw on a benign missing GC-internal object -/// during a prune (it would only wedge GC). A genuine TokenMismatch is +/// during a prune (it would only wedge GC). A genuine mismatch is /// likewise tolerated here: the object was rewritten under us (another attempt is live at this key) — the /// safe direction during a best-effort prune is to leave it for a later round, never to force-delete. /// `out_fully_drained`, when set, reports whether the WHOLE prefix was exhausted (every listed key @@ -3407,7 +3722,7 @@ namespace /// remain, and the cursor must stay put so a later round's fresh budget can finish the same prefix /// instead of stranding the remainder permanently. `bounded_remaining == 0` conservatively reports /// `false` (nothing was even examined, so completeness cannot be claimed). -uint64_t deletePrefixWholesale(Backend & backend, const String & prefix, uint64_t bounded_remaining, +uint64_t deletePrefixWholesale(CasOperation & op, const String & prefix, uint64_t bounded_remaining, bool * out_fully_drained) { if (out_fully_drained) @@ -3417,22 +3732,22 @@ uint64_t deletePrefixWholesale(Backend & backend, const String & prefix, uint64_ String cursor; while (deleted < bounded_remaining) { - ListPage page = backend.list(prefix, cursor, kListPageLimit); + ListPage page = op.list(prefix, cursor, kListPageLimit, Retry::standard()); /// One page fetched, not one increment per listed key below. ProfileEvents::increment(ProfileEvents::CASGCEnumerationPages); for (const auto & listed : page.keys) { if (deleted >= bounded_remaining) return deleted; - if (listed.token.has_value()) + if (listed.etag.has_value()) { - /// deleteExact tolerates NotFound (returns Kind::NotFound) and TokenMismatch — both are - /// benign here (already gone / rewritten by a live attempt); do not throw. - backend.deleteExact(listed.key, *listed.token); + /// `Gone` and `Mismatch` are both benign here (already gone / rewritten by a live + /// attempt); do not throw. + op.remove(listed.key, *listed.etag, Retry::standard()); } - else if (const auto head = backend.head(listed.key); head.exists) + else if (const auto head = op.head(listed.key, Retry::standard())) { - backend.deleteExact(listed.key, head.token); + op.remove(listed.key, head->etag, Retry::standard()); } ++deleted; } @@ -3463,7 +3778,7 @@ void Gc::pruneSupersededGenerations(uint64_t adopted_generation, uint64_t attemp if (keep == 0) return; /// keep ALL (debug/forensics — replay GC's in-degree view as-of a past round) - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); static constexpr uint64_t kMaxPrunePerRound = 64; /// bound the per-round prune burst @@ -3520,7 +3835,7 @@ void Gc::pruneSupersededGenerations(uint64_t adopted_generation, uint64_t attemp break; bool fully_drained = false; const uint64_t reclaimed = deletePrefixWholesale( - backend, layout.gcGenPrefix(g), remaining, &fully_drained); + op, layout.gcGenPrefix(g), remaining, &fully_drained); work_budget.prefix_wholesale_objects_used += reclaimed; if (!fully_drained) break; @@ -3547,7 +3862,8 @@ void Gc::pruneSupersededGenerations(uint64_t adopted_generation, uint64_t attemp std::optional Gc::readFoldSeal(uint64_t generation, uint64_t attempt) { - if (const auto got = store->backend().get(store->layout().foldSealKey(generation, attempt))) + CasOperation op = store->openRequests().admit(); + if (const auto got = op.read(store->layout().foldSealKey(generation, attempt), Retry::standard())) return decodeFoldSeal( got->bytes, store->layout(), store->poolConfig().gc_shards, generation); return std::nullopt; @@ -3596,14 +3912,15 @@ std::vector Gc::discoverUniverse() /// The filter itself lives in `CasRefCatalog::liveUniverse` (review Important C) -- fsck's own /// reachability walk needed the identical catalog-authoritative set and is not this class, so the /// filter moved to where both can share it rather than grow a second copy that could disagree. - return CasRefCatalog::liveUniverse(store->backend(), store->layout()); + CasOperation op = store->openRequests().admit(); + return CasRefCatalog::liveUniverse(op, store->layout()); } bool Gc::graduationDue(const GcState & state, uint64_t current_round) { /// Retired-in-snapshot: the graduation signal is read from the adopted fold seal's per-shard /// `condemned_summary` — ZERO backend I/O beyond the single seal read. A summary distilled from this - /// generation's `kCondemned` rows says, per shard, how many entries are `delete_pending` (a graduation + /// generation's `RunMarker::Condemned` rows says, per shard, how many entries are `delete_pending` (a graduation /// is already published) and the oldest non-pending condemn round (one crosses the floor once /// `condemn_round < current_round`). if (state.snap_generation == 0) @@ -3643,12 +3960,12 @@ RefScanSummary Gc::enumerateRefPrefix() /// name is absorbed per key by `parseRefObjectKeyForEnumeration`, which is what keeps this /// enumeration -- which runs before the fold, outside its catch -- unable to wedge the round. const Layout & layout = store->layout(); - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); RefScanSummary scan; static constexpr size_t kListPageLimit = 1000; size_t count_in_page = 0; - forEachListedKey(backend, layout.casRefsPrefix(), [&](const ListedKey & lk) + op.forEachListedKey(layout.casRefsPrefix(), [&](const ListedKey & lk) { scan.keys.push_back(lk.key); const auto parsed = parseRefObjectKeyForEnumeration(layout, lk.key); @@ -3668,8 +3985,9 @@ RefScanSummary Gc::enumerateRefPrefix() count_in_page = 0; ProfileEvents::increment(ProfileEvents::CASRefGlobalListPages); } - }, kListPageLimit, onGcEnumerationPage); - /// The walk's `backend.list` lands at least once even for an empty/undersized final page -- + return true; + }, Retry::standard(), kListPageLimit, onGcEnumerationPage); + /// The walk's list lands at least once even for an empty/undersized final page -- /// count it (one increment per physical LIST call). if (count_in_page > 0 || scan.keys.empty()) ProfileEvents::increment(ProfileEvents::CASRefGlobalListPages); @@ -3683,7 +4001,8 @@ RoundInput Gc::listRefPrefix(const GcState & state) /// plan. A listed id absent from the later cut is dead, inert debris: it contributes no work and /// cannot force DEFER. RefScanSummary scan = enumerateRefPrefix(); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(store->backend(), store->layout()); + CasOperation op = store->openRequests().admit(); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, store->layout()); /// TEST SEAM: see `setPostHotScanCatalogReadHookForTest`. Moved into a local before invoking (the /// same reason `create_namespace_step1_pre_read_hook_for_test` is swapped rather than called /// directly): a hook that reassigns the member from inside its own body would otherwise reassign @@ -3721,8 +4040,19 @@ RebuildReport Gc::rebuildBaseline(bool force) /// Writes ONLY the GC plane; namespace streams/state, manifests, and blobs are read-only inputs; /// the rebuild never deletes them. RebuildReport rep; - Backend & backend = store->backend(); + CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); + /// The rebuild reads through the same seam the fold does, so the two share one implementation of + /// "read this key" rather than growing a second. + /// + /// WHAT THAT MEANS HERE, exactly: `readCheckpointWitnesses` hints its own keys and does not ask who + /// called it, so the rebuild's checkpoint reads ARE fetched ahead, on the same pool and by the same + /// rule as the fold's. That is a gain and not an accident -- a rebuild reads every namespace's + /// checkpoint too. Nothing on THIS path hints a manifest key, so `foldManifestEdges` below takes + /// them one at a time, exactly as it did before: a rebuild walks a plan it already holds rather + /// than discovering its next key from the body it just read, so a lookahead would have nothing to + /// hide behind. + GcReadAhead reads(op, store->openRequests(), *read_pool, store->poolConfig().gc_read_concurrency); /// Read bookkeeping health before the lease (the lease acquire on an absent state CREATES a /// bootstrap body, which must not make scenario (а) look healthy). A generation-0 ref-baseline @@ -3733,7 +4063,7 @@ RebuildReport Gc::rebuildBaseline(bool force) bool healthy = false; bool validate_generation_zero_ref_baseline = false; { - const auto got = backend.get(layout.gcStateKey()); + const auto got = op.read(layout.gcStateKey(), Retry::standard()); /// The state's own decode stays inside its own try: an undecodable `gc/state` IS scenario (а), /// the disaster this command exists for. The prior-seal refusal below must NOT be swallowed by /// that catch, so the seal is read outside it. @@ -3794,7 +4124,7 @@ RebuildReport Gc::rebuildBaseline(bool force) "to rebuild; this pool must be recreated.", st.snap_generation, st.snap_attempt); for (const RunRef & r : seal->blob_target_runs) - if (!backend.head(r.key).exists) + if (!op.head(r.key, Retry::standard())) healthy = false; prior_seal = std::move(seal); rep.adopted_seal_generation = st.snap_generation; @@ -3872,8 +4202,8 @@ RebuildReport Gc::rebuildBaseline(bool force) /// has_observation==false always takes the non-steal branch on its one and only call), pass it /// explicitly rather than rely on that invariant. GcState state; - Token state_token; - if (!acquireOrRenewLease(state, state_token, /*allow_steal=*/false)) + std::optional state_etag; + if (!acquireOrRenewLease(state, state_etag, /*allow_steal=*/false)) { rep.refusal = "another GC leader holds the lease"; return rep; @@ -3890,7 +4220,7 @@ RebuildReport Gc::rebuildBaseline(bool force) || drain_result.catalog_resolution != CatalogResolution::DrainComplete) throwCasWriteRetryLater("CAS GC rebuild lost authority before the catalog settled"); const RefScanSummary rebuild_ref_scan = enumerateRefPrefix(); - const CasRefCatalog::Snapshot rebuild_work_catalog_cut = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot rebuild_work_catalog_cut = CasRefCatalog::read(op, layout); RefScanSummary rebuild_round_scan = rebuild_ref_scan; if (prior_seal) @@ -3902,7 +4232,7 @@ RebuildReport Gc::rebuildBaseline(bool force) /// universe. `recoverRefTableDetailedFromAuthority` deliberately has no internal catalog or /// checkpoint read: a later cut could admit a different life or frontier than the one every other /// part of this rebuild is using. - const CheckpointWitnesses rebuild_checkpoints = readCheckpointWitnesses({}, rebuild_walk_plan.catalogCut()); + const CheckpointWitnesses rebuild_checkpoints = readCheckpointWitnesses(reads, {}, rebuild_walk_plan.catalogCut()); if (validate_generation_zero_ref_baseline) { @@ -3912,8 +4242,9 @@ RebuildReport Gc::rebuildBaseline(bool force) for (const NamespaceLifeId & life : rebuild_walk_universe) { std::vector table_keys; - forEachListedKey(backend, layout.namespaceStreamPrefix(life), - [&](const ListedKey & lk) { table_keys.push_back(lk.key); }, 1000, onGcEnumerationPage); + op.forEachListedKey(layout.namespaceStreamPrefix(life), + [&](const ListedKey & lk) { table_keys.push_back(lk.key); return true; }, + Retry::standard(), 1000, onGcEnumerationPage); std::map grouped; try { @@ -3949,12 +4280,12 @@ RebuildReport Gc::rebuildBaseline(bool force) { const String gen_prefix = layout.gcGenPrefix(0); const String top = gen_prefix.substr(0, gen_prefix.size() - 2); /// ".../gc/gen/" - forEachListedKey(backend, top, [&](const ListedKey & k) + op.forEachListedKey(top, [&](const ListedKey & k) { const size_t from = top.size(); const size_t slash = k.key.find('/', from); if (slash == String::npos) - return; + return true; try { max_gen = std::max(max_gen, static_cast(std::stoull(k.key.substr(from, slash - from)))); @@ -3963,7 +4294,8 @@ RebuildReport Gc::rebuildBaseline(bool force) { /// Foreign key shape under `gc/gen` is debris, not a numbering input. } - }, 1000, onGcEnumerationPage); + return true; + }, Retry::standard(), 1000, onGcEnumerationPage); } const uint64_t generation = max_gen + 1; const uint64_t budget = rebuild_edge_budget_override ? rebuild_edge_budget_override @@ -3977,14 +4309,14 @@ RebuildReport Gc::rebuildBaseline(bool force) std::vector attempt_of(gc_shards, 0); /// The fold is EDGE-ONLY here: a rebuild condemns nothing (spec §7, and the deletion below), so no /// condemn round is stamped and no head source is supplied. `current_round` 0 graduates nothing and - /// `condemn_round` 0 with an empty `head_blob` mints no `kCondemned` row -- this call is + /// `condemn_round` 0 with an empty `head_blob` mints no `RunMarker::Condemned` row -- this call is /// `foldDeltasIntoGeneration`'s pure edge form. auto flush_shard = [&](uint64_t shard) { if (buckets[shard].empty()) return; std::vector out; - foldDeltasIntoGeneration(backend, layout, prior_runs[shard], generation, ++attempt_of[shard], + foldDeltasIntoGeneration(op, layout, prior_runs[shard], generation, ++attempt_of[shard], shard, std::move(buckets[shard]), out, /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, @@ -4023,7 +4355,7 @@ RebuildReport Gc::rebuildBaseline(bool force) /// round once it is known, and nothing else is touched. std::set minted_hold_lives; uint64_t max_fence_round = 0; - std::map mf_cleanup_unused; + std::map mf_cleanup_unused; for (const NamespaceLifeId & life : rebuild_walk_universe) { @@ -4050,11 +4382,11 @@ RebuildReport Gc::rebuildBaseline(bool force) checkpoint_it != rebuild_checkpoints.recovery_checkpoints.end()) checkpoint = checkpoint_it->second; const RecoveredRefTable recovered = recoverRefTableDetailedFromAuthority( - backend, layout, *entry_it, checkpoint); + op, layout, *entry_it, checkpoint); const RefTableState & st = recovered.state; RefCoverage cov; - cov.classification = 2; /// Folded (full coverage) unless a bodiless precommit clamps + cov.classification = CoverageClass::Folded; /// unless a bodiless precommit clamps it below cov.last_folded_ref_id = st.getGreatestApplied(); /// Whether the hold on this row was minted BY THIS REBUILD (and so still owes a retry round) /// rather than carried from the prior seal. Tracked explicitly instead of by looking for a @@ -4070,7 +4402,7 @@ RebuildReport Gc::rebuildBaseline(bool force) { const ManifestId id{ns, row.manifest_ref}; owned_manifest_keys.insert(layout.manifestKey(id)); - if (!foldManifestEdges(id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) + if (!foldManifestEdges(reads, id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) { rep.refusal = "committed ref '" + ns.string() + "/" + ref_name + "' names a missing or invalid part manifest — that is DATA LOSS the rebuild " @@ -4086,7 +4418,7 @@ RebuildReport Gc::rebuildBaseline(bool force) { const ManifestId id{ns, manifest_ref}; owned_manifest_keys.insert(layout.manifestKey(id)); - if (foldManifestEdges(id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) + if (foldManifestEdges(reads, id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) ++rep.live_precommits; else { @@ -4100,7 +4432,7 @@ RebuildReport Gc::rebuildBaseline(bool force) /// still missing -- and clears once the namespace makes durable progress. /// RESIDUAL, named rather than hidden: progress unrelated to this precommit also clears /// it, and the precommit's edges stay missing until another rebuild. - cov.classification = 4; /// Clamped + cov.classification = CoverageClass::Clamped; cov.hold = RefHold{.reason = HoldReason::ManifestBodyMissing, .offending_position = RefTxnId{cov.last_folded_ref_id.writer_epoch, cov.last_folded_ref_id.ref_sequence + 1}, @@ -4123,7 +4455,7 @@ RebuildReport Gc::rebuildBaseline(bool force) const auto pit = prior_seal->ref_lives.find(life.incarnation); if (pit != prior_seal->ref_lives.end() && pit->second.coverage.hold) { - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = pit->second.coverage.hold; minted_here = false; /// a carried hold rides VERBATIM; its retry fields are not ours } @@ -4145,26 +4477,27 @@ RebuildReport Gc::rebuildBaseline(bool force) { const RootNamespace ns{ns_str}; std::vector deltas; - forEachListedKey(backend, layout.manifestNamespacePrefix(ns), [&](const ListedKey & k) + op.forEachListedKey(layout.manifestNamespacePrefix(ns), [&](const ListedKey & k) { if (owned_manifest_keys.contains(k.key)) - return; + return true; /// The one shared manifest-path parser for the canonical hexadecimal manifest identifier, /// also used by fsck's parseBuildPrefix and the orphan sweep's parseListedManifestObject. const auto parsed = layout.parseManifestKey(k.key); if (!parsed) - return; /// foreign key shape — debris + return true; /// foreign key shape — debris const ManifestRef & mref = parsed->ref; if (prefixEligible(*store, ns, BuildPrefix{mref.writer_epoch, mref.build_sequence})) - return; /// provably dead — the orphan sweep's territory, never an edge + return true; /// provably dead — the orphan sweep's territory, never an edge const ManifestId id{ns, mref}; - if (foldManifestEdges(id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) + if (foldManifestEdges(reads, id, +1, deltas, mf_cleanup_unused, /*txn_ordinal=*/0)) { ++rep.unowned_alive_manifests; route_deltas(deltas); } /// A missing/invalid UNOWNED body is debris (no owner claims it) — skip, never refuse. - }, 1000, onGcEnumerationPage); + return true; + }, Retry::standard(), 1000, onGcEnumerationPage); } /// A REBUILD CONDEMNS NOTHING (spec §7). @@ -4213,7 +4546,7 @@ RebuildReport Gc::rebuildBaseline(bool force) /// `mountObservationThresholdMs` -- see its doc comment (CasServerRoot.h). const uint64_t stable_threshold_ms = mountObservationThresholdMs( ttl_ms, static_cast(store->poolConfig().mount_renew_period.count())); - computeHeartbeatFloor(backend, layout, now_ms_fn(), mono_ms_fn(), stable_threshold_ms, mount_obs); + computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn(), stable_threshold_ms, mount_obs); /// Retired-in-snapshot: the rebuilt seal's `condemned_summary` must be TOTAL over gc_shards so a /// subsequent regular round reads graduation/carry decisions zero-I/O off it (and its `carryParentRefs` @@ -4229,7 +4562,7 @@ RebuildReport Gc::rebuildBaseline(bool force) for (uint64_t a : attempt_of) seal_attempt = std::max(seal_attempt, a); validateFoldSealForWrite(seal, layout, gc_shards); - putDeterministicArtifact(backend, layout.foldSealKey(generation, seal_attempt), encodeFoldSeal(seal)); + putDeterministicArtifact(op, layout.foldSealKey(generation, seal_attempt), encodeFoldSeal(seal)); GcState next = state; next.round = round; @@ -4240,12 +4573,14 @@ RebuildReport Gc::rebuildBaseline(bool force) /// family independently of that, and the two reasons are stated apart on purpose — a future reader /// must not take this line as evidence that REBUILD still produces condemnations somewhere. next.manifest_sweep_cursor = ""; - const CasResult res = backend.casPut(layout.gcStateKey(), encodeGcState(next), state_token); - if (res.outcome != CasOutcome::Committed) + WriteResult commit = op.replace(layout.gcStateKey(), encodeGcState(next), *state_etag, + Retry::standard()); + if (std::holds_alternative(commit)) { rep.refusal = "gc/state changed under the rebuild (a competing writer) — re-run"; return rep; } + orThrow(std::move(commit), "CAS gc rebuild: gc/state commit"); rep.performed = true; rep.round = round; @@ -4274,13 +4609,13 @@ std::vector Gc::previewDeletes() { std::vector out; - const auto state_bytes = store->backend().get(store->layout().gcStateKey()); + CasOperation op = store->openRequests().admit(); + const auto state_bytes = op.read(store->layout().gcStateKey(), Retry::standard()); if (!state_bytes) return out; const GcState state = decodeGcState(state_bytes->bytes); const Layout & layout = store->layout(); - Backend & backend = store->backend(); /// Resolve the run objects THROUGH the adopted seal's refs, never by /// `blobTargetRunKey` construction: with reference-parent carry a shard's current run may physically @@ -4298,32 +4633,32 @@ std::vector Gc::previewDeletes() const auto it = runs_by_shard.find(shard); static const std::vector kEmptyRuns; const std::vector & shard_runs = it != runs_by_shard.end() ? it->second : kEmptyRuns; - for (const BlobCandidate & cand : zeroInDegree(backend, shard_runs)) + for (const BlobCandidate & cand : zeroInDegree(op, shard_runs)) { - const HeadResult observed = backend.head(layout.blobKey(cand.ref)); - if (!observed.exists) + const std::optional observed = op.head(layout.blobKey(cand.ref), Retry::standard()); + if (!observed) continue; PreviewEntry e; e.kind = ObjectKind::Blob; e.ref = cand.ref; e.key = layout.blobKey(cand.ref); - e.size = observed.size; + e.size = observed->size; e.reason = "unreachable"; out.push_back(std::move(e)); } - /// Retired-in-snapshot: stream the SAME adopted seal runs and emit every `kCondemned` - /// sentinel row. The stored token IS the authority — NO HEAD here (a HEAD would defeat the point - /// and cost I/O). `delete_pending` rows are deleted next fold; the rest await graduation. Preview + /// Retired-in-snapshot: stream the SAME adopted seal runs and emit every `RunMarker::Condemned` + /// sentinel row. The stored incarnation IS the authority — no observation here (one would defeat + /// the point and cost I/O). `delete_pending` rows are deleted next fold; the rest await graduation. Preview /// stays WRITE-FREE (`openSourceEdgeRun` is a pure reader). Output is a superset of the above. for (const RunRef & run : shard_runs) { - SourceEdgeRunView reader = openSourceEdgeRun(backend, run.key); + SourceEdgeRunView reader = openSourceEdgeRun(op, run.key); String key; String payload; while (reader.next(key, payload)) { - if (payload.empty() || payload[0] != kCondemned) + if (payload.empty() || runMarkerFromByte(payload[0], "CAS source-edge run") != RunMarker::Condemned) continue; BlobRef ref; UInt128 source_id; @@ -4354,31 +4689,59 @@ void Gc::rememberObservation(const GcLease & lease) last_seen_seq = lease.seq; } +void Gc::refreshAuthority(uint64_t admitted_generation) +{ + /// Fail-closed first, so every early exit below leaves this leader deposed. + authority_held = false; + try + { + CasOperation op = store->openRequests().admit(); + const auto got = op.read(store->layout().gcStateKey(), Retry::standard()); + if (!got) + return; + const GcState current = decodeGcState(got->bytes); + authority_held = current.lease.owner == gc_id && current.lease.seq == admitted_generation; + } + catch (...) + { + tryLogCurrentException(logger, + "CAS gc: the leader-authority probe failed; this round's destructive operations treat the " + "lease as lost"); + } +} + void Gc::pulseHeartbeat(Pool & store, UInt128 gc_id) { + CasOperation op = store.openRequests().admit(); const String key = store.layout().gcHbKey(); - const auto got = store.backend().get(key); + const auto got = op.read(key, Retry::standard()); GcHeartbeat hb; - std::optional expected; if (got) - { hb = decodeGcHeartbeat(got->bytes); - expected = got->token; - } hb.owner = gc_id; ++hb.hb_seq; - store.backend().casPut(key, encodeGcHeartbeat(hb), expected); + /// ONE attempt, and the outcome is discarded: a pulse that loses its race is replaced by the next + /// one on cadence, so a deposed leader must never spend a whole retry budget fighting for this key. + const String body = encodeGcHeartbeat(hb); + if (got) + op.replace(key, body, got->etag, Retry::once()); + else + op.create(key, body, Retry::once()); } -bool Gc::acquireOrRenewLease(GcState & state, Token & state_token, bool allow_steal) +bool Gc::acquireOrRenewLease(GcState & state, std::optional & state_etag, bool allow_steal) { + CasOperation op = store->openRequests().admit(); const String key = store->layout().gcStateKey(); - for (int attempt = 0; attempt < 2; ++attempt) + /// What the decision that actually landed wrote. `readModifyWrite` re-decides on every conflict + /// against what the losing write's own resolve read observed, so the last decision is the committed + /// one, and a competing leader's write makes the next decision see a moved lease tuple. + GcState decided; + const std::optional committed = orThrow(op.readModifyWrite(key, + [&](const std::optional & current_object) -> std::optional { - const auto got = store->backend().get(key); - - if (!got) + if (!current_object) { if (has_observation) throw Exception(ErrorCodes::CORRUPTED_DATA, @@ -4391,18 +4754,11 @@ bool Gc::acquireOrRenewLease(GcState & state, Token & state_token, bool allow_st /// the authoritative value from the persisted GcState (pool is authoritative on reopen). /// PoolConfig carries the configured value from the disk XML. fresh.gc_shards = store->poolConfig().gc_shards; - const CasResult acquire_res = store->backend().casPut(key, encodeGcState(fresh), std::nullopt); - if (acquire_res.outcome == CasOutcome::Committed) - { - rememberObservation(fresh.lease); - state = std::move(fresh); - state_token = acquire_res.token; - return true; - } - continue; + decided = fresh; + return encodeGcState(fresh); } - GcState current = decodeGcState(got->bytes); + const GcState current = decodeGcState(current_object->bytes); if (current.gc_shards != store->poolConfig().gc_shards) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS gc/state gc_shards {} disagrees with the pool-authoritative _pool_meta value {}", @@ -4412,19 +4768,12 @@ bool Gc::acquireOrRenewLease(GcState & state, Token & state_token, bool allow_st { GcState next = current; ++next.lease.seq; - const CasResult renew_res = store->backend().casPut(key, encodeGcState(next), got->token); - if (renew_res.outcome == CasOutcome::Committed) - { - rememberObservation(next.lease); - state = std::move(next); - state_token = renew_res.token; - return true; - } - continue; + decided = next; + return encodeGcState(next); } GcHeartbeat hb; - if (const auto hb_got = store->backend().get(store->layout().gcHbKey())) + if (const auto hb_got = op.read(store->layout().gcHbKey(), Retry::standard())) hb = decodeGcHeartbeat(hb_got->bytes); /// Observation-based heartbeat liveness, symmetric with the frozen-lease-tuple check below: /// ANY movement of the observed (owner, hb_seq) pair between this contender's two ticks is @@ -4460,27 +4809,23 @@ bool Gc::acquireOrRenewLease(GcState & state, Token & state_token, bool allow_st last_seen_hb_owner = hb.owner; last_seen_hb_seq = hb.hb_seq; } - return false; + return std::nullopt; } GcState next = current; next.lease.owner = gc_id; ++next.lease.seq; - const CasResult steal_res = store->backend().casPut(key, encodeGcState(next), got->token); - if (steal_res.outcome == CasOutcome::Committed) - { - rememberObservation(next.lease); - state = std::move(next); - state_token = steal_res.token; - return true; - } + decided = next; + return encodeGcState(next); + }, Retry::standard()), "CAS gc lease"); - if (const auto reread = store->backend().get(key)) - rememberObservation(decodeGcState(reread->bytes).lease); - return false; - } + if (!committed) + return false; /// declined: a live incumbent holds the lease - return false; + rememberObservation(decided.lease); + state = std::move(decided); + state_etag = committed; + return true; } CatalogLifecycleReconcileResult Gc::drainCompletedRemoving(const GcState & leased_state) @@ -4500,28 +4845,17 @@ CatalogLifecycleReconcileResult Gc::drainCompletedRemoving(const GcState & lease "CAS GC pre-fold drain: adopted parent seal (generation {}, attempt {}) is missing", leased_state.snap_generation, leased_state.snap_attempt); - Backend & backend = store->backend(); - const Layout & layout = store->layout(); + /// The drain erases catalog rows, so its operation carries this leader's authority as its + /// liveness: `CatalogLifecycleReconciler` and `deleteCompletedRemovingAtSnapshot` decide + /// `FencedOut` from `op.admitted()`, and the GC plane's fence is open, so without this the verdict + /// would be a constant TRUE and a deposed leader would keep erasing. The refresh below runs at the + /// top of every erase attempt, so a leader deposed between two erases of one drain stops before the + /// second -- the single reading taken here would otherwise authorise all of them. const uint64_t admitted_generation = leased_state.lease.seq; - const auto check_fence = [&](uint64_t expected_generation) - { - if (expected_generation != admitted_generation) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CAS GC pre-fold drain: internal leader generation mismatch (expected {}, admitted {})", - expected_generation, admitted_generation); - const auto got = backend.get(layout.gcStateKey()); - if (!got) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS GC pre-fold drain: gc/state vanished while checking leader generation {}", - admitted_generation); - const GcState current = decodeGcState(got->bytes); - if (current.lease.owner != gc_id || current.lease.seq != admitted_generation) - return CasRefCatalog::LeaderFenceStatus::Moved; - return CasRefCatalog::LeaderFenceStatus::Held; - }; - - return CatalogLifecycleReconciler( - backend, layout, *parent, admitted_generation, check_fence).reconcile(); + refreshAuthority(admitted_generation); + CasOperation op = store->openRequests().admit([this] { return authority_held; }); + return CatalogLifecycleReconciler(op, store->layout(), *parent) + .reconcile([this, admitted_generation] { refreshAuthority(admitted_generation); }); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h index 2754fe433c19..bd0d65e2e573 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h @@ -1,7 +1,9 @@ #pragma once +#include #include #include #include +#include #include #include #include @@ -69,6 +71,28 @@ enum class UniversePolicy : uint8_t /// decision ever reads them. uint64_t retiredLogicalSize(ObjectKind kind, uint64_t object_size, uint64_t blob_header_len); +/// Deletes `chunk` as one bulk `removeManyWriteOnce` request, falling back to one admitted request per +/// key when the object storage answers with `NOT_IMPLEMENTED` -- the signal +/// `S3ObjectStorage::removeObjectsIfExistImpl` gives (without sending anything else itself) once +/// `DeleteObjects` is known unsupported (a configured GCS backend, or one that just failed a batch +/// attempt this same call). The fallback is not merely "the same deletes issued more slowly": each +/// `op.removeManyWriteOnce({key}, policy)` is its OWN admission (fence, budget, deadline checked +/// afresh), which one bulk call covering up to `kBulkDeleteMaxKeys` physical deletes under a SINGLE +/// admission cannot be -- exactly the gap a storage-side per-key loop would have left open. Every other +/// failure propagates unchanged: retry/reissue for it is the engine's own policy, applied to each +/// admitted attempt -- bulk or single -- the same way it always was. +/// +/// Returns the number of `op.removeManyWriteOnce` calls THIS HELPER issued: 1 for the bulk path, or +/// 1 + `chunk.size()` for the fallback -- the failed bulk attempt counted alongside the one call per key +/// that followed it, since that attempt is a call this helper made whether or not it reached the network +/// (there is no signal available here to tell "sent and rejected" apart from "refused locally, unsent"; +/// `S3ObjectStorage::removeObjectsIfExistImpl` reports both as the same NOT_IMPLEMENTED). This is call +/// COUNT, not a distinct network-request count -- the same granularity `CountingBackend::bulkRemoveCalls` +/// and the `CASBulkDeleteRequests` profile event already use elsewhere for "request". +/// Declared here (not file-local) so a unit test can drive it directly against a scripted backend, +/// rather than only through a full GC round. +uint64_t removeChunkWriteOnceOrOneByOne(CasOperation & op, const std::vector & chunk, const Retry & policy); + /// Pure skip-unchanged decision. Returns true iff the current round may be /// DEFERRED (re-adopt the sealed generation, no fold/delete). A round MUST fold when: enough shards /// changed (>= fold_threshold), OR a destructive decision is due (graduation_due), OR the defer bound @@ -429,8 +453,15 @@ class Gc /// `policy` is the destructive gate's universe seam — see `UniversePolicy`. Production passes /// nothing; a test whose subject is the suppressed gate passes `StageA_Suppressed` here, which is /// the only way to reach that posture. + /// `progress` (optional) is the caller's window into a round that THROWS: the round accumulates its + /// report directly in `*progress` (reset at entry) as each phase completes, so on an exception the + /// caller still sees everything the round durably did before it died -- `round` is stamped only + /// after the round's single `gc/state` CAS commits, so `progress->round != 0` on a failed round + /// proves the round committed and died in the post-CAS tail. On the success path `*progress` equals + /// the returned report. RoundReport runRegularRound(std::function on_lease_acquired = {}, bool allow_steal = true, - UniversePolicy policy = UniversePolicy::kDefault); + UniversePolicy policy = UniversePolicy::kDefault, + RoundReport * progress = nullptr); /// Advisory heartbeat: bump /gc/hb to {gc_id, hb_seq+1}. Best-effort (a lost CAS is /// harmless — the next pulse retries). Touches NO Gc instance state. Static by design. @@ -446,7 +477,7 @@ class Gc String key; uint64_t size = 0; String reason; /// "unreachable" | "delete_pending" | "awaiting_graduation" - Token token; /// stored condemn-time token (empty for "unreachable") + PersistedEtag token; /// stored condemn-time incarnation (empty for "unreachable") uint64_t condemn_round = 0; }; @@ -502,10 +533,10 @@ class Gc private: /// Lease acquire/renew/steal per the documented observation protocol. On success `state` holds the - /// committed gc/state (with our lease) and `state_token` its backend token. `allow_steal=false` - /// suppresses only the steal CAS (see runRegularRound's doc comment) — acquiring a free lease and - /// renewing our own are unaffected. - bool acquireOrRenewLease(GcState & state, Token & state_token, bool allow_steal); + /// committed gc/state (with our lease) and `state_etag` the etag that write created. + /// `allow_steal=false` suppresses only the steal (see runRegularRound's doc comment) — acquiring a + /// free lease and renewing our own are unaffected. + bool acquireOrRenewLease(GcState & state, std::optional & state_etag, bool allow_steal); /// Catalog-only helping barrier run immediately after lease acquisition. It validates the adopted /// parent and delegates deterministic `Removing`-row settlement to `CatalogLifecycleReconciler`. @@ -524,13 +555,13 @@ class Gc /// What one fold produced. The blob deltas are sealed /// into a write-once generation; `fold_seal` is the durable index of WHAT WAS FOLDED (a CasFoldSeal), /// `root_shards` the discovered universe, `mf_cleanup` the part-manifest cleanup work keyed by - /// ManifestId (owner-removed bodies whose exact-token delete is deferred until their decrements are - /// sealed), and `retired_merge` the per-gc-shard ack-floor retired-cursor outcome. + /// ManifestId (owner-removed bodies whose exact-incarnation delete is deferred until their + /// decrements are sealed), and `retired_merge` the per-gc-shard ack-floor retired-cursor outcome. struct FoldResult { CasFoldSeal fold_seal; std::vector> root_shards; - std::map mf_cleanup; + std::map mf_cleanup; /// Bounded orphan candidates exact-read before reduce. Their source retirements ride this /// fold's runs; their manifest tokens become deletable only after the round CAS adopts them. ManifestSweepResult orphan_sweep; @@ -681,8 +712,6 @@ class Gc /// missing body or a true-removal old body missing at removal-fold => fail-closed FOR THAT DECISION /// (clamp the shard's last_folded_ref_id below it, record the anomaly, stop folding THIS shard) — /// never guess a delta and never wedge the round on a missing body. - /// On success `state` carries the committed snap_generation and `state_token` the committed gc/state - /// token. The committed pair is THREADED into retire, never re-read (zombie-steal protection). /// Round-paced graduation: `current_round` (= state.round + 1, the SAME basis condemn_round is /// stamped at) is the threshold the fold's two-cursor merge graduates/condemns against — an entry /// graduates once `condemn_round < current_round`, i.e. it survived at least one full round after @@ -690,7 +719,8 @@ class Gc /// in-memory; the SINGLE round CAS commits them. /// `walk_plan` owns the round's one enumeration of `cas/ns/stream/` (see `RefScanSummary`) and /// its catalog cut; the fold regroups those keys strictly rather than listing the prefix again. - FoldResult fold(GcState & state, Token & state_token, RoundReport & report, uint64_t current_round, + FoldResult fold(GcState & state, std::optional & state_etag, + RoundReport & report, uint64_t current_round, const RefPlan & walk_plan, UniversePolicy policy, /// One instance for the WHOLE round, owned by `runRegularRound` and threaded through /// every destructive-work family the round touches — see `GcRoundWorkBudget`. @@ -738,7 +768,8 @@ class Gc /// (`HoldReason::CheckpointUndecodable`) and folds every other namespace normally. std::map undecodable; }; - CheckpointWitnesses readCheckpointWitnesses(const std::map & ref_tables, + CheckpointWitnesses readCheckpointWitnesses(GcReadAhead & reads, + const std::map & ref_tables, const CasRefCatalog::Snapshot & catalog_cut); /// What ONE generation's prefix says about itself: whether the generation exists at all, and the @@ -773,13 +804,13 @@ class Gc std::optional> newestFoldSealRef(); /// Read ONE part manifest named by `id`, validate it, and append sign*(+1) blob deltas for each - /// blob entry to `deltas`. On sign<0 queue (id -> token) into mf_cleanup. Returns whether a body was + /// blob entry to `deltas`. On sign<0 queue (id -> incarnation) into mf_cleanup. Returns whether a body was /// read+validated: false => ABSENT body (404; the caller decides per the 404 rule). A body that is /// PRESENT but fails refMatchesBody / manifestNamespaceMatches throws CORRUPTED_DATA. /// `txn_ordinal` stamps every delta this call pushes with the round-local ordinal of the ref /// transaction that emitted it (probe B2 — see `TxnApplyLedger`). - bool foldManifestEdges(const ManifestId & id, int sign, std::vector & deltas, - std::map & mf_cleanup, uint32_t txn_ordinal); + bool foldManifestEdges(GcReadAhead & reads, const ManifestId & id, int sign, std::vector & deltas, + std::map & mf_cleanup, uint32_t txn_ordinal); @@ -873,6 +904,10 @@ class Gc /// Update the remembered observation (steal protocol step 3/4). void rememberObservation(const GcLease & lease); + /// Re-read `gc/state` and record whether this leader still holds the lease it was admitted under. + /// Fail-closed: an absent, unreadable or undecodable state reads as deposed. + void refreshAuthority(uint64_t admitted_generation); + PoolPtr store; /// Where `GcPhaseTimer` sends one record per GC phase. Empty unless a `CasGcScheduler` installed one /// for the current round, in which case every phase of that round emits a row. @@ -896,6 +931,18 @@ class Gc /// a round folds; incremented on every DEFER. Bounds batching via `gc_fold_max_defer_rounds`. uint64_t rounds_since_last_fold_ = 0; + /// THIS LEADER'S OWN AUTHORITY VERDICT, and the reason it is a cached bool rather than a probe. + /// + /// It is what the `Liveness` predicates of the round's destructive operations sample -- the pre-fold + /// catalog drain and the namespace janitor page, both of which erase objects a deposed leader must + /// not touch. The engine samples a `Liveness` before EVERY request and before every sleep, so a + /// predicate that read `gc/state` itself would put one `GET` on the hot path of every listed key. + /// The read that sets this flag is therefore made by the round, at the granularity the old + /// hand-written fence check had (once per drain, once per janitor page), never from inside the + /// predicate. The staleness that buys is bounded by that granularity and stated where each caller + /// refreshes it. + bool authority_held = false; + /// the contender's observation window (steal protocol) bool has_observation = false; UInt128 last_seen_owner{}; @@ -929,6 +976,11 @@ class Gc /// initialized before that check. std::unique_ptr meta_writer; + /// The fold's read-ahead pool, sized by `gc_read_concurrency`. A `unique_ptr` for the same reason + /// as `meta_writer`: the size comes from `store->poolConfig()`, which may only be read after the + /// constructor body has validated `store`. + std::unique_ptr read_pool; + /// Probe B1's two numbers for the round: the ref-log POSITIONS the sealed coverage declares covered /// (counted arithmetically over each namespace's cut -- not by listed ids, which under arithmetic /// intake say nothing about what was applied), and the ref logs that actually folded. They are EQUAL diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcKeyReader.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcKeyReader.h new file mode 100644 index 000000000000..3d4d6ca82ffb --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcKeyReader.h @@ -0,0 +1,25 @@ +#pragma once +#include +#include + +namespace DB::Cas +{ + +/// The reader over the GC's read-ahead: a hint is a worker request, a take is the worker's result +/// (or an inline read for a key nobody hinted), and a discard drops a hinted key and counts it as +/// wasted at once. +class ReadAheadKeyReader final : public KeyReader +{ +public: + explicit ReadAheadKeyReader(GcReadAhead & reads_) : reads(reads_) {} + void hint(const String & key) override { reads.hintRead(key); } + std::optional take(const String & key) override { return reads.takeRead(key); } + void discard(const String & key) override { reads.discardRead(key); } + size_t pending() const override { return reads.pending(); } + size_t window() const override { return reads.window(); } + +private: + GcReadAhead & reads; +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.cpp index 19ff16bb956c..b082c3820203 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.cpp @@ -9,32 +9,32 @@ namespace DB::ErrorCodes namespace DB::Cas { -GcMaintenanceReadResult readGcMaintenanceState(Backend & backend, const Layout & layout) +GcMaintenanceReadResult readGcMaintenanceState(CasOperation & op, const Layout & layout) { - const auto got = backend.get(layout.gcMaintenanceStateKey()); + const auto got = op.read(layout.gcMaintenanceStateKey(), Retry::standard()); if (!got) - return {.status = GcMaintenanceReadStatus::Absent, .state = std::nullopt, .token = std::nullopt, .diagnostic = {}}; + return {.status = GcMaintenanceReadStatus::Absent, .state = std::nullopt, .etag = std::nullopt, .diagnostic = {}}; try { return {.status = GcMaintenanceReadStatus::Valid, .state = decodeGcMaintenanceState(got->bytes), - .token = got->token, .diagnostic = {}}; + .etag = got->etag, .diagnostic = {}}; } catch (const DB::Exception & e) { if (e.code() != ErrorCodes::CORRUPTED_DATA) throw; return {.status = GcMaintenanceReadStatus::Corrupt, .state = std::nullopt, - .token = got->token, .diagnostic = e.message()}; + .etag = got->etag, .diagnostic = e.message()}; } } -GcMaintenanceCasResult casGcMaintenanceState( - Backend & backend, const Layout & layout, const std::optional & expected, const GcMaintenanceState & next) +WriteResult casGcMaintenanceState( + CasOperation & op, const Layout & layout, const std::optional & expected, + const GcMaintenanceState & next, const Retry & policy) { - const CasResult result = backend.casPut(layout.gcMaintenanceStateKey(), encodeGcMaintenanceState(next), expected); - if (result.outcome == CasOutcome::Committed) - return {.outcome = GcMaintenanceCasOutcome::Committed, .token = result.token}; - return {.outcome = GcMaintenanceCasOutcome::Conflict, .token = {}}; + const String key = layout.gcMaintenanceStateKey(); + const String bytes = encodeGcMaintenanceState(next); + return expected ? op.replace(key, bytes, *expected, policy) : op.create(key, bytes, policy); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.h index 6d941177d19f..5c12272fc03a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMaintenanceState.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -12,18 +12,17 @@ struct GcMaintenanceReadResult { GcMaintenanceReadStatus status; std::optional state; - std::optional token; + std::optional etag; String diagnostic; }; -enum class GcMaintenanceCasOutcome : uint8_t { Committed, Conflict }; -struct GcMaintenanceCasResult -{ - GcMaintenanceCasOutcome outcome = GcMaintenanceCasOutcome::Conflict; - Token token; -}; -GcMaintenanceReadResult readGcMaintenanceState(Backend & backend, const Layout & layout); -GcMaintenanceCasResult casGcMaintenanceState( - Backend & backend, const Layout & layout, const std::optional & expected, const GcMaintenanceState & next); +GcMaintenanceReadResult readGcMaintenanceState(CasOperation & op, const Layout & layout); +/// `create`s the maintenance-state key on absence, `replace`s it when `expected` names the +/// incarnation last observed. `policy` is named by the caller because a write issued from inside an +/// already-failed step (a reset after a failed enumeration) must send at most one attempt rather than +/// spend the round's remaining time retrying a write nothing downstream is waiting on. +WriteResult casGcMaintenanceState( + CasOperation & op, const Layout & layout, const std::optional & expected, + const GcMaintenanceState & next, const Retry & policy); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.cpp index b60df45da424..cd36d866f776 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.cpp @@ -1,9 +1,11 @@ #include +#include #include #include #include +#include namespace ProfileEvents { @@ -24,6 +26,13 @@ namespace DB::Cas namespace { +/// The registry key for one condemned incarnation: the persisted pair rendered the way a live +/// `Etag` renders itself, so two dialects can never collide on a shared value. +String condemnMarkerKey(const PersistedEtag & token) +{ + return token.dialect + ":" + token.value; +} + /// The per-hash freshness-meta operations GC schedules on the bounded pool are /// best-effort/idempotent by design. The meta is only a point-read freshness marker for the writer/ /// promote gate; the ledger retired-set + the exact-token body delete remain the actual safety @@ -46,8 +55,8 @@ namespace /// a writer reading `Clean` would reuse the exact condemned token, which a stale pre-CAS exact-token /// redelete then deletes -- live-blob data loss (INV_NO_LOSS). Removing the clear restores the exact-token /// delete argument in full: once a hash is `Condemned`, observing `Clean` means EITHER the condemned body -/// is absent OR a writer already changed its incarnation token, so every stale `deleteExact(t1)` finds the -/// body absent or `TokenMismatch`. +/// is absent OR a writer already changed its incarnation, so every stale exact-incarnation delete of the +/// condemned incarnation finds the body absent or holding a different one. /// Write the per-hash meta to Condemned: a blob newly entering the retired set this round (either the /// fresh zero-in-degree condemn, or a republication-supersede re-condemn of the current token). Absent meta @@ -55,17 +64,18 @@ namespace /// alone rather than clobbering a possibly-newer condemn_round. /// /// Returns whether durable Condemned evidence exists after the call: the conditional write committed, -/// or an already-Condemned meta was observed. A lost CAS reports false and writes nothing further (the -/// loser re-reads next time); a thrown backend error propagates (the scheduling wrapper swallows it) — -/// either way the entry stays UNCONFIRMED and the graduation gate carries it. -bool writeCondemnedMeta(Pool & pool, const BlobRef & ref, uint64_t condemn_round, uint64_t size) +/// or an already-Condemned meta was observed. Any other write outcome reports false and writes nothing +/// further (the loser re-reads next time); a thrown backend error propagates (the scheduling wrapper +/// swallows it) — either way the entry stays UNCONFIRMED and the graduation gate carries it. +bool writeCondemnedMeta(CasOperation & op, const Layout & layout, const BlobRef & ref, + uint64_t condemn_round, uint64_t size) { - const auto lm = loadMeta(pool.backend(), pool.layout(), ref); + const auto lm = loadMeta(op, layout, ref); const BlobMeta desired{.state = MetaState::Condemned, .condemn_round = condemn_round, .size = size}; if (!lm) - return putMetaIfAbsent(pool, ref, desired).outcome == CasOverwriteOutcome::Committed; + return std::holds_alternative(putMetaIfAbsent(op, layout, ref, desired)); if (lm->meta.state != MetaState::Condemned) - return casMeta(pool, ref, lm->etag, desired).outcome == CasOverwriteOutcome::Committed; + return std::holds_alternative(casMeta(op, layout, ref, lm->etag, desired)); return true; } @@ -73,12 +83,12 @@ bool writeCondemnedMeta(Pool & pool, const BlobRef & ref, uint64_t condemn_round /// exact-token delete. NO tombstone -- an absent meta reads exactly like a Clean one (absent /// means not condemned"). Idempotent: an already-absent meta, or one a racing writer/GC pass already /// moved, is a silent no-op. -void deleteConfirmedMeta(Backend & backend, const Layout & layout, const BlobRef & ref) +void deleteConfirmedMeta(CasOperation & op, const Layout & layout, const BlobRef & ref) { - const auto lm = loadMeta(backend, layout, ref); + const auto lm = loadMeta(op, layout, ref); if (!lm) return; - deleteMetaExact(backend, layout, ref, lm->etag); + deleteMetaExact(op, layout, ref, lm->etag); } } @@ -134,12 +144,15 @@ void GcMetaWriter::submit(std::function op) } } -void GcMetaWriter::scheduleCondemnMarkerWrite(const BlobRef & ref, const Token & token, +void GcMetaWriter::scheduleCondemnMarkerWrite(const BlobRef & ref, const PersistedEtag & token, uint64_t condemn_round, uint64_t size) { + /// The job admits its OWN operation: a `CasOperation` carries per-call state and belongs to one + /// task, while several of these run concurrently on the pool. submit([st = state, ref, token, condemn_round, size]() { - if (writeCondemnedMeta(*st->store, ref, condemn_round, size)) + CasOperation op = st->store->openRequests().admit(); + if (writeCondemnedMeta(op, st->store->layout(), ref, condemn_round, size)) st->noteCondemnMarkerDurable(ref, token); }); } @@ -148,7 +161,8 @@ void GcMetaWriter::scheduleConfirmedMetaDelete(const BlobRef & ref) { submit([st = state, ref]() { - deleteConfirmedMeta(st->store->backend(), st->store->layout(), ref); + CasOperation op = st->store->openRequests().admit(); + deleteConfirmedMeta(op, st->store->layout(), ref); }); } @@ -187,35 +201,35 @@ uint64_t GcMetaWriter::completed() const return state->completed.load(std::memory_order_relaxed); } -void GcMetaWriter::State::noteCondemnMarkerDurable(const BlobRef & ref, const Token & token) +void GcMetaWriter::State::noteCondemnMarkerDurable(const BlobRef & ref, const PersistedEtag & token) { std::lock_guard lock(condemn_marker_mutex); - condemn_markers_confirmed.emplace(ref, token.value); + condemn_markers_confirmed.emplace(ref, condemnMarkerKey(token)); } -bool GcMetaWriter::State::condemnMarkerConfirmedInProcess(const BlobRef & ref, const Token & token) +bool GcMetaWriter::State::condemnMarkerConfirmedInProcess(const BlobRef & ref, const PersistedEtag & token) { std::lock_guard lock(condemn_marker_mutex); - return condemn_markers_confirmed.contains({ref, token.value}); + return condemn_markers_confirmed.contains({ref, condemnMarkerKey(token)}); } -void GcMetaWriter::State::forgetCondemnMarker(const BlobRef & ref, const Token & token) +void GcMetaWriter::State::forgetCondemnMarker(const BlobRef & ref, const PersistedEtag & token) { std::lock_guard lock(condemn_marker_mutex); - condemn_markers_confirmed.erase({ref, token.value}); + condemn_markers_confirmed.erase({ref, condemnMarkerKey(token)}); } -void GcMetaWriter::noteCondemnMarkerDurable(const BlobRef & ref, const Token & token) +void GcMetaWriter::noteCondemnMarkerDurable(const BlobRef & ref, const PersistedEtag & token) { state->noteCondemnMarkerDurable(ref, token); } -bool GcMetaWriter::condemnMarkerConfirmedInProcess(const BlobRef & ref, const Token & token) +bool GcMetaWriter::condemnMarkerConfirmedInProcess(const BlobRef & ref, const PersistedEtag & token) { return state->condemnMarkerConfirmedInProcess(ref, token); } -void GcMetaWriter::forgetCondemnMarker(const BlobRef & ref, const Token & token) +void GcMetaWriter::forgetCondemnMarker(const BlobRef & ref, const PersistedEtag & token) { state->forgetCondemnMarker(ref, token); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.h index d1332854f416..9392da61c9e6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcMetaWriter.h @@ -1,7 +1,7 @@ #pragma once +#include #include -#include #include #include @@ -29,11 +29,11 @@ class GcMetaWriter GcMetaWriter(const GcMetaWriter &) = delete; GcMetaWriter & operator=(const GcMetaWriter &) = delete; - /// Publish durable Condemned evidence for one (blob, exact incarnation-token) pair. On success the + /// Publish durable Condemned evidence for one (blob, exact incarnation) pair. On success the /// pair is recorded in the in-process confirmation registry, which the graduation gate reads. A - /// lost CAS or a thrown error leaves the pair UNCONFIRMED: the gate then carries the entry and a - /// later round retries the write. - void scheduleCondemnMarkerWrite(const BlobRef & ref, const Token & token, + /// refused write or a thrown error leaves the pair UNCONFIRMED: the gate then carries the entry and + /// a later round retries the write. + void scheduleCondemnMarkerWrite(const BlobRef & ref, const PersistedEtag & token, uint64_t condemn_round, uint64_t size); /// Drop the freshness meta of a blob whose body is confirmed deleted or absent. @@ -51,12 +51,14 @@ class GcMetaWriter uint64_t scheduled() const; uint64_t completed() const; - /// The in-process condemn-marker confirmation registry, keyed (blob, exact token value). Pool + /// The in-process condemn-marker confirmation registry, keyed (blob, rendered incarnation). It is + /// keyed by the PERSISTED pair because every entry that consults it arrives from a durable + /// condemned row; a live observation enters through `PersistedEtag::capture`. Pool /// completions insert concurrently with the round thread's reads, and the round thread also /// inserts directly when it re-checks a marker synchronously. - void noteCondemnMarkerDurable(const BlobRef & ref, const Token & token); - bool condemnMarkerConfirmedInProcess(const BlobRef & ref, const Token & token); - void forgetCondemnMarker(const BlobRef & ref, const Token & token); + void noteCondemnMarkerDurable(const BlobRef & ref, const PersistedEtag & token); + bool condemnMarkerConfirmedInProcess(const BlobRef & ref, const PersistedEtag & token); + void forgetCondemnMarker(const BlobRef & ref, const PersistedEtag & token); private: /// Everything a job reaches. Held by `shared_ptr` and captured by value into every job. @@ -69,9 +71,9 @@ class GcMetaWriter std::mutex condemn_marker_mutex; std::set> condemn_markers_confirmed; - void noteCondemnMarkerDurable(const BlobRef & ref, const Token & token); - bool condemnMarkerConfirmedInProcess(const BlobRef & ref, const Token & token); - void forgetCondemnMarker(const BlobRef & ref, const Token & token); + void noteCondemnMarkerDurable(const BlobRef & ref, const PersistedEtag & token); + bool condemnMarkerConfirmedInProcess(const BlobRef & ref, const PersistedEtag & token); + void forgetCondemnMarker(const BlobRef & ref, const PersistedEtag & token); }; /// Catch each meta-operation exception, count the job, and put it on the pool -- running it diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.cpp new file mode 100644 index 000000000000..fe6eecedd72a --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.cpp @@ -0,0 +1,139 @@ +#include +#include + +#include + +namespace ProfileEvents +{ + extern const Event CASGCReadAheadHit; + extern const Event CASGCReadAheadMiss; + extern const Event CASGCReadAheadWasted; +} + +namespace DB::Cas +{ + +GcReadAhead::GcReadAhead(CasOperation & op_, CasRequests & requests_, ThreadPool & pool_, size_t concurrency_) + : op(op_), requests(requests_), pool(pool_), concurrency(concurrency_), generation(op_.generation()) +{ +} + +GcReadAhead::~GcReadAhead() +{ + /// A worker holds its own slot and touches `requests`, which outlives this object; waiting here is + /// what keeps every worker inside the round that issued it. `wait`, not `get`: an exception nobody + /// took is dropped with the result it belongs to, and a destructor may not throw. + size_t wasted = 0; + for (auto & [key, slot] : reads) + { + slot->future.wait(); + ++wasted; + } + for (auto & [key, slot] : heads) + { + slot->future.wait(); + ++wasted; + } + if (wasted != 0) + ProfileEvents::increment(ProfileEvents::CASGCReadAheadWasted, wasted); +} + +template +void GcReadAhead::hint(Slots & slots, const String & key, Request request) +{ + if (concurrency <= 1 || slots.contains(key)) + return; + + auto slot = std::make_shared>(); + slots.emplace(key, slot); + try + { + pool.scheduleOrThrowOnError([slot, key, requests_ptr = &requests, gen = generation, request] + { + try + { + CasOperation worker = requests_ptr->resume(gen); + slot->promise.set_value(request(worker, key)); + } + catch (...) + { + slot->promise.set_exception(std::current_exception()); + } + }); + } + catch (...) + { + /// Nothing will ever satisfy this slot's promise, so a later take would wait on it forever. + /// Drop it and let the take read inline; the scheduling failure itself propagates to the + /// hinting site, which is a round-thread site like any other. + slots.erase(key); + throw; + } +} + +template +std::optional GcReadAhead::take(Slots & slots, const String & key, Inline inline_request) +{ + const auto it = slots.find(key); + if (it == slots.end()) + { + ProfileEvents::increment(ProfileEvents::CASGCReadAheadMiss); + return inline_request(op, key); + } + + std::shared_ptr> slot = std::move(it->second); + slots.erase(it); + ProfileEvents::increment(ProfileEvents::CASGCReadAheadHit); + /// Rethrows the worker's exception at the site that would otherwise have read inline, so a + /// transport failure fails the round from the same place and with the same type it always did. + return slot->future.get(); +} + +void GcReadAhead::hintRead(const String & key) +{ + hint(reads, key, + [](CasOperation & worker, const String & k) { return worker.read(k, Retry::standard()); }); +} + +void GcReadAhead::hintHead(const String & key) +{ + hint(heads, key, + [](CasOperation & worker, const String & k) { return worker.head(k, Retry::standard()); }); +} + +template +void GcReadAhead::discard(Slots & slots, const String & key) +{ + const auto it = slots.find(key); + if (it == slots.end()) + return; + std::shared_ptr> slot = std::move(it->second); + slots.erase(it); + /// `wait`, not `get`: the result and any exception belong to a request nobody wanted. + slot->future.wait(); + ProfileEvents::increment(ProfileEvents::CASGCReadAheadWasted); +} + +void GcReadAhead::discardRead(const String & key) +{ + discard(reads, key); +} + +void GcReadAhead::discardHead(const String & key) +{ + discard(heads, key); +} + +std::optional GcReadAhead::takeRead(const String & key) +{ + return take(reads, key, + [](CasOperation & inline_op, const String & k) { return inline_op.read(k, Retry::standard()); }); +} + +std::optional GcReadAhead::takeHead(const String & key) +{ + return take(heads, key, + [](CasOperation & inline_op, const String & k) { return inline_op.head(k, Retry::standard()); }); +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.h new file mode 100644 index 000000000000..19b9df5a24c8 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcReadAhead.h @@ -0,0 +1,103 @@ +#pragma once + +#include +#include +#include + +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// Read-ahead in front of ONE admitted operation. A caller HINTS keys the sequential code will read +/// next; workers fetch them on `pool`, each through an operation resumed under the SAME admitted +/// generation as the caller's (no liveness -- exactly the fold's own admission); a TAKE returns the +/// fetched result, rethrows the worker's exception, or -- for a key nobody hinted -- performs the +/// request inline. This is a cache of RESULTS, never of decisions: every decode, counter and event +/// stays at the take site, so a round at concurrency 1 (nothing is ever hinted) and a round at 16 +/// read the same keys in the same order and decide the same way; only WHEN the bytes were fetched +/// moves. +/// +/// Why a result may be fetched early: the objects the fold reads are write-once (a present body is +/// the same body later), an absent position at or below a checkpoint's `committed_through` was +/// durable before the round began (it is a gap whenever it is read), and a manifest body still being +/// uploaded when the early read lands yields the same hold a slightly earlier sequential read yields +/// today. Nothing is hinted above `committed_through`, so no request is issued that the sequential +/// walk would not issue -- with one bounded exception: a ref-log hinting site does not yet know where +/// an epoch's closing seal is (it learns that only by decoding the log at that position), so it may +/// hint ids past the seal, inside the SAME epoch, that turn out not to exist. Those are discarded +/// rather than taken, overshooting by at most one window per epoch crossing, and counted wasted. +/// +/// Memory is the CALLER's to bound: `pending` counts hinted-but-untaken slots and `window` is how +/// many a hinting site keeps in flight. A key hinted twice is one request. Results never taken are +/// awaited by the destructor and counted as wasted, or discarded explicitly by `discardRead`/ +/// `discardHead`, which counts them at once. Only the owning thread touches the maps; a worker +/// touches only its own slot. +class GcReadAhead +{ +public: + GcReadAhead(CasOperation & op_, CasRequests & requests_, ThreadPool & pool_, size_t concurrency_); + ~GcReadAhead(); + + GcReadAhead(const GcReadAhead &) = delete; + GcReadAhead & operator=(const GcReadAhead &) = delete; + + void hintRead(const String & key); + void hintHead(const String & key); + + std::optional takeRead(const String & key); + std::optional takeHead(const String & key); + + /// Drops a hinted key the caller will never take. The request is already in flight or done; the + /// wait is bounded by that one request, its result and its exception are dropped, and the slot is + /// counted as wasted now rather than in the destructor. The exception is dropped because no + /// sequential walk would have issued this request, so none could have failed on it; the fence and + /// the budget are re-observed by the very next request anyway. A key nobody hinted is a no-op. + void discardRead(const String & key); + void discardHead(const String & key); + + /// Hinted but not yet taken, both verbs together: what a hinting site throttles itself against. + size_t pending() const { return reads.size() + heads.size(); } + + /// How many requests a hinting site should keep in flight. Zero at concurrency 1, which is what + /// makes every `while (pending() < window())` loop hint nothing at all on the sequential setting. + size_t window() const { return concurrency <= 1 ? 0 : 4 * concurrency; } + +private: + template + struct Slot + { + std::promise> promise; + std::future> future; + Slot() : future(promise.get_future()) {} + }; + + template + using Slots = std::unordered_map>>; + + template + void hint(Slots & slots, const String & key, Request request); + + template + std::optional take(Slots & slots, const String & key, Inline inline_request); + + template + void discard(Slots & slots, const String & key); + + CasOperation & op; + CasRequests & requests; + ThreadPool & pool; + const size_t concurrency; + /// The generation the caller's operation was admitted under. A worker resumes under exactly this + /// one, so a fence that moves under the round fails a worker's request the way it fails the + /// round's own -- never with a fresher admission the round itself would not have had. + const uint64_t generation; + + Slots reads; + Slots heads; +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp index f265444576a8..610d13879779 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp @@ -12,9 +12,35 @@ #include #include +namespace DB::ErrorCodes +{ + extern const int S3_ERROR; + extern const int NETWORK_ERROR; + extern const int ABORTED; + extern const int TIMEOUT_EXCEEDED; + extern const int SOCKET_TIMEOUT; + extern const int MEMORY_LIMIT_EXCEEDED; +} + namespace DB::Cas { +bool isTransientGcRoundError(int code) +{ + /// Codes that name a condition which clears without intervention: the backend refused or timed out + /// (`S3_ERROR`, `NETWORK_ERROR`, `TIMEOUT_EXCEEDED`, `SOCKET_TIMEOUT`), another actor legitimately + /// moved shared state (`ABORTED` -- the round CAS's own "another leader advanced it"), or memory + /// pressure hit a manual round running on a budgeted query thread (`MEMORY_LIMIT_EXCEEDED`). + /// Everything else -- notably `LOGICAL_ERROR`, `CORRUPTED_DATA`, `BAD_ARGUMENTS` -- stays + /// non-transient BY OMISSION: an unrecognised code must read as a real failure, never as noise. + return code == ErrorCodes::S3_ERROR + || code == ErrorCodes::NETWORK_ERROR + || code == ErrorCodes::ABORTED + || code == ErrorCodes::TIMEOUT_EXCEEDED + || code == ErrorCodes::SOCKET_TIMEOUT + || code == ErrorCodes::MEMORY_LIMIT_EXCEEDED; +} + namespace { /// Non-zero events of a per-round snapshot, keyed by event name. The snapshot is already a @@ -192,9 +218,29 @@ Cas::RoundReport CasGcScheduler::runRoundLogged(Cas::Gc & round_gc, GcRoundLogRe Rec fin = start; fin.event_type = Rec::EventType::Finish; + /// Lives OUTSIDE the try and is filled progressively by the round (the `progress` out-parameter of + /// `runRegularRound`), so the Finish row of a THROWING round still carries everything the round + /// durably did before it died -- `round != 0` on such a row proves the round's `gc/state` CAS + /// committed and the failure hit the post-CAS tail. + Cas::RoundReport rep; + const auto fill_counters = [&fin](const Cas::RoundReport & r) + { + fin.round = r.round; + fin.candidates_marked = r.candidates; + fin.objects_deleted = r.deleted; + fin.objects_absent = r.absent; + fin.objects_replaced = r.replaced; + fin.objects_spared = r.spared; + fin.manifests_deleted = r.manifests_deleted; + fin.entries_condemned = r.condemned; + fin.entries_graduated = r.graduated; + fin.entries_redeleted = r.redeleted; + fin.fence_outs = r.fence_outs; + fin.anomalies = r.anomalies.size(); + }; try { - const Cas::RoundReport rep = round_gc.runRegularRound(std::move(on_lease_acquired), allow_steal); + (void)round_gc.runRegularRound(std::move(on_lease_acquired), allow_steal, Cas::UniversePolicy::kDefault, &rep); if (rep.acquired_lease) { /// Keep health state per scheduler. Process-global gauges cannot distinguish multiple @@ -210,18 +256,7 @@ Cas::RoundReport CasGcScheduler::runRoundLogged(Cas::Gc & round_gc, GcRoundLogRe fin.outcome = !rep.acquired_lease ? Rec::Outcome::NotALeader : rep.deferred ? Rec::Outcome::Deferred : Rec::Outcome::Success; - fin.round = rep.round; - fin.candidates_marked = rep.candidates; - fin.objects_deleted = rep.deleted; - fin.objects_absent = rep.absent; - fin.objects_replaced = rep.replaced; - fin.objects_spared = rep.spared; - fin.manifests_deleted = rep.manifests_deleted; - fin.entries_condemned = rep.condemned; - fin.entries_graduated = rep.graduated; - fin.entries_redeleted = rep.redeleted; - fin.fence_outs = rep.fence_outs; - fin.anomalies = rep.anomalies.size(); + fill_counters(rep); fin.duration_ms = std::chrono::duration_cast( std::chrono::steady_clock::now() - t0).count(); fin.profile_events = collect_profile_events(); @@ -230,8 +265,14 @@ Cas::RoundReport CasGcScheduler::runRoundLogged(Cas::Gc & round_gc, GcRoundLogRe } catch (...) { - fin.outcome = Rec::Outcome::Failed; + fin.error_code = getCurrentExceptionCode(); + /// Non-transient first, so a bug that coincides with a restart is never masked; then the + /// teardown flag, the only witness of a refused teardown fence (see `Outcome::Stopped`). + fin.outcome = !isTransientGcRoundError(fin.error_code) ? Rec::Outcome::Failed + : store->teardownBegun() ? Rec::Outcome::Stopped + : Rec::Outcome::Aborted; fin.error = getCurrentExceptionMessage(false); + fill_counters(rep); fin.duration_ms = std::chrono::duration_cast( std::chrono::steady_clock::now() - t0).count(); fin.profile_events = collect_profile_events(); @@ -297,6 +338,13 @@ void CasGcScheduler::loop() /// correctness issue. std::lock_guard round_lock(gc_round_mutex); + /// A round that starts after the pool's teardown began would emit a Start row and be + /// refused at its first lease request -- a row that says nothing. Checked here, under the + /// round mutex, so it also covers the tick queued behind a manual round; the extra round + /// on a plain `stop` (above) is a different race and stays as described. + if (store->teardownBegun()) + return; + /// runRoundLogged emits the Start + Finish table rows (incl. the per-round /// ProfileEvents delta) and rethrows on a round exception (after an Aborted Finish). /// on_lease_acquired (onLeaseAcquired, shared with runOneRoundNow) fires the instant the @@ -341,9 +389,29 @@ void CasGcScheduler::loop() } catch (...) { + if (store->teardownBegun() && isTransientGcRoundError(getCurrentExceptionCode())) + { + /// The disk is being torn down and the round was cut at its next request: expected, + /// recorded as `Stopped` by `runRoundLogged`, not an error to raise. + LOG_INFO(log, "CA GC round stopped by the disk's teardown: {}", getCurrentExceptionMessage(false)); + continue; + } /// Idempotent round - the next tick retries; failures must never kill the pacing thread. - /// runRoundLogged already emitted the Aborted Finish row before rethrowing. - i_am_leader.store(false, std::memory_order_relaxed); + /// runRoundLogged already emitted the classified (Aborted/Failed) Finish row before rethrowing. + /// + /// Leadership is dropped only on a NON-transient failure. Dropping it on every failure + /// silenced the advisory heartbeat for a whole interval (`heartbeatLoop` gates its pulses on + /// `i_am_leader`), and when the failure was itself a backend outage the durable lease + /// `(owner, seq)` was frozen too -- together exactly the two-of-two dead-leader signature + /// `acquireOrRenewLease` steals on. A live leader blocked on a flaky store was then deposed, + /// and every handover forces the successor into a full fold: more single-attempt conditional + /// writes against the same flaky backend, a self-reinforcing loop. Keeping the flag keeps the + /// pulses; the lease protocol stays authoritative -- a mounter that really died stops pulsing + /// with or without this flag. A non-transient failure still clears it: a logic-broken leader + /// must stay depositable, and with the flag held its heartbeat would keep beating and no + /// follower could ever steal a lease whose holder cannot complete a round. + if (!isTransientGcRoundError(getCurrentExceptionCode())) + i_am_leader.store(false, std::memory_order_relaxed); tryLogCurrentException(log, "CA GC round failed (will retry next tick)"); } } @@ -384,7 +452,16 @@ void CasGcScheduler::heartbeatLoop() } catch (...) { - tryLogCurrentException(log, "CA GC heartbeat pulse failed (advisory; will retry)"); + /// A pulse refused by the open plane during teardown is the expected end of this loop and + /// not a failure to report; `stop` joins it moments later. Only a TRANSIENT failure is + /// silent, for the same fail-closed reason the round classifier refuses to relabel a + /// non-transient one: a corrupt heartbeat that happens to coincide with a restart is an + /// incident, and swallowing it would be the one place this teardown path hides a defect. + const bool tearing_down = store->teardownBegun(); + if (!tearing_down || !isTransientGcRoundError(getCurrentExceptionCode())) + tryLogCurrentException(log, "CA GC heartbeat pulse failed (advisory; will retry)"); + if (tearing_down) + return; } } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h index 2cce48399b66..80b50f2fc448 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h @@ -27,7 +27,15 @@ struct GcRoundLogRecord /// -- no fold, no pre-CAS deletes, no `gc/state` CAS. Distinct from `Success` so a reader of /// `system.cas_gc_log` (or this scheduler's own log line) can tell a round /// that genuinely folded and found nothing apart from one that never folded at all. - enum class Outcome { Unknown, Success, NotALeader, Failed, Deferred }; + /// `Aborted`: the round threw an exception whose code names a transient condition (backend + /// unavailability, a lost lease, a concurrent leader) -- the next scheduled round retries it and + /// nothing durable is wrong. `Stopped`: the same transient class, observed after the pool's + /// teardown began -- a correlation, not a cause: the engine reports a refused teardown fence + /// exactly like a lost mount fence, so the flag is the only witness, and the row says so honestly + /// rather than reading a clean restart as a backend incident. `Failed` is reserved for everything + /// else (a logic error, corrupted data, an unclassified code): fail-closed, an unrecognised + /// failure reads as real -- during a teardown too. + enum class Outcome { Unknown, Success, NotALeader, Failed, Deferred, Aborted, Stopped }; enum class Trigger { Scheduled, Manual }; EventType event_type = EventType::Start; @@ -51,6 +59,9 @@ struct GcRoundLogRecord UInt64 anomalies = 0; /// fold clamps surfaced (never wedging) this round UInt64 duration_ms = 0; String error; + /// `getCurrentExceptionCode()` of the failure on an `Aborted`/`Failed` Finish row; 0 otherwise. + /// The structured twin of `error`: oracles and operators key on this, never on message wording. + Int32 error_code = 0; /// On a `Start`/`Finish` row: the whole round's delta. On a `Phase` row: THAT PHASE's delta. std::map profile_events; @@ -75,6 +86,12 @@ struct GcRoundLogRecord using GcRoundLogger = std::function; +/// True when an exception code names a condition that clears by itself -- the backend was unreachable +/// or slow, or another actor legitimately moved shared state -- so the next scheduled round is the +/// retry. False for everything else, deliberately including any code not on the list: an unrecognised +/// failure must read as a real one. +bool isTransientGcRoundError(int code); + /// Paces regular content-addressed garbage-collection rounds for one pool. The scheduler does not /// implement the GC protocol: `Cas::Gc` owns lease acquisition, work deduplication, and the /// split-brain-safe round operations, so schedulers on different mounters may run independently. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.cpp index c3b44569d074..7f102a2b86d3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.cpp @@ -40,13 +40,13 @@ bool ShardReducer::owns(const BlobRef & ref) const return blobShard(ref, gc_shards) == shard; } -std::vector ShardReducer::reduce(Backend & backend, const Layout & layout, +std::vector ShardReducer::reduce(CasOperation & op, const Layout & layout, const std::vector & prior_runs, uint64_t new_generation, uint64_t attempt, std::vector shard_deltas, uint64_t current_round, uint64_t condemn_round, - const std::function(const BlobRef &)> & head_blob, - const std::function(const BlobRef &)> & peek_head, + const BlobHeadFn & head_blob, + const BlobHeadFn & peek_head, const std::function & confirm_condemned_marker, RetiredMergeResult * out_retired, bool suppress_destructive, @@ -54,7 +54,7 @@ std::vector ShardReducer::reduce(Backend & backend, const Layout & layou GcRoundWorkBudget * work_budget) const { std::vector out_runs; - foldDeltasIntoGeneration(backend, layout, prior_runs, new_generation, attempt, shard, + foldDeltasIntoGeneration(op, layout, prior_runs, new_generation, attempt, shard, std::move(shard_deltas), out_runs, current_round, condemn_round, head_blob, peek_head, confirm_condemned_marker, out_retired, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.h index 20edc54dfb88..86e698dc6a63 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcShardPlan.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -72,7 +72,7 @@ uint64_t manifestCleanupShard(const ManifestId & id, uint64_t gc_shards); /// uses with `shard == 0`), so `gc_shards == 1` with `shard == 0` reproduces the non-sharded fold /// byte-for-byte. This keeps the one-shard configuration compatible with the original fold path. /// -/// NOTE on durable writes: `reduce` writes the per-shard in-degree run directly via `backend` +/// NOTE on durable writes: `reduce` writes the per-shard in-degree run directly through `op` /// (under `blobTargetRunKey(new_generation, shard, 0)`), exactly as `foldDeltasIntoGeneration` /// does. Returning the durable write here (rather than an in-memory map) keeps the round driver /// stateless: it simply constructs a `ShardReducer` per shard, calls `reduce`, and the sealed @@ -90,8 +90,9 @@ class ShardReducer /// Merge `shard_deltas` (the caller's per-shard `BlobDelta` slice produced by `foldManifestEdges` /// and bucketed by `blobShard`) into a new in-degree generation for this shard. Writes the sealed - /// run under `blobTargetRunKey(new_generation, shard, 0)` via `backend`, appends its `RunRef` to - /// `out_runs`, and returns the `RunRef`. The call is idempotent (write-once via `putIfAbsent`). + /// run under `blobTargetRunKey(new_generation, shard, 0)` through `op`, appends its `RunRef` to + /// `out_runs`, and returns the `RunRef`. The call is idempotent: the run is written once, and a + /// second call finds the identical object already there. /// /// `prior_runs` are the parent generation's run segments for this shard, resolved BY THE CALLER from /// the parent fold seal's `blob_target_runs` filtered to `shard`. An empty vector is the @@ -100,13 +101,13 @@ class ShardReducer /// PRECONDITION: every `BlobDelta` in `shard_deltas` must be owned by this reducer /// (`blobShard(d.ref, gc_shards) == shard`). This is a caller contract; there is no /// underflow throw backstopping it — pass a misbucketed delta and the fold silently misroutes it. - std::vector reduce(Backend & backend, const Layout & layout, + std::vector reduce(CasOperation & op, const Layout & layout, const std::vector & prior_runs, uint64_t new_generation, uint64_t attempt, std::vector shard_deltas, uint64_t current_round = 0, uint64_t condemn_round = 0, - const std::function(const BlobRef &)> & head_blob = {}, - const std::function(const BlobRef &)> & peek_head = {}, + const BlobHeadFn & head_blob = {}, + const BlobHeadFn & peek_head = {}, const std::function & confirm_condemned_marker = {}, RetiredMergeResult * out_retired = nullptr, bool suppress_destructive = false, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp index e2eb6fc761e4..f8c153ed0f20 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp @@ -6,15 +6,32 @@ namespace DB::Cas { -NamespaceJanitorResult NamespaceJanitor::runOnePage( - bool suppress_deletes, const std::function & fence_held) +namespace +{ + +/// The legacy `casPut` this write replaces reported a definite conflict as a value (never a failure to +/// this caller) and reported a store failure -- a refusal, an exhausted policy -- by throwing. Only +/// `Refused`/`GaveUp` are the alternatives a thrown exception used to carry, so only those propagate; +/// `Committed`/`Declined`/`Conflict` stay silent exactly as they did before. +void throwOnRefusedOrGaveUp(WriteResult && result, std::string_view what) +{ + if (std::holds_alternative(result) || std::holds_alternative(result)) + (void)orThrow(std::move(result), what); +} + +} + +NamespaceJanitorResult NamespaceJanitor::runOnePage(bool suppress_deletes, Liveness liveness) { NamespaceJanitorResult result; - const GcMaintenanceReadResult progress = readGcMaintenanceState(backend, layout); + CasOperation op = requests.admit(std::move(liveness)); + const GcMaintenanceReadResult progress = readGcMaintenanceState(op, layout); if (progress.status == GcMaintenanceReadStatus::Corrupt) { result.anomalies.push_back(progress.diagnostic); - (void)casGcMaintenanceState(backend, layout, progress.token, GcMaintenanceState{}); + throwOnRefusedOrGaveUp( + casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::standard()), + "CAS namespace janitor: corrupt maintenance-state reset"); return result; } @@ -22,17 +39,17 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( ListPage page; try { - page = backend.list(layout.namespaceRootPrefix(), cursor, page_budget); + page = op.list(layout.namespaceRootPrefix(), cursor, page_budget, Retry::standard()); } catch (...) { - (void)casGcMaintenanceState(backend, layout, progress.token, GcMaintenanceState{}); + (void)casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::once()); throw; } result.pages = 1; result.keys = page.keys.size(); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); bool ambiguous = false; try { @@ -45,10 +62,13 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( } /// A valid page is complete only when the round had deletion authority for every dead-life /// candidate on it. Advancing while the global gate is closed can phase-lock a dead page onto - /// every suppressed round and a different page onto every bounded forced fold. Ambiguous cuts and - /// observed fence loss have the same shape: retain the old cursor so an authoritative round - /// retries the exact page. Malformed keys, absent objects and token mismatches are final per-key - /// outcomes and therefore do not by themselves prevent progress. + /// every suppressed round and a different page onto every bounded forced fold. An ambiguous cut + /// retains the old cursor so an authoritative round retries the exact page; a lost liveness sample + /// only reaches this retained-cursor path when it is caught between the two `op.admitted()` checks + /// below -- a sample lost earlier throws out of a read verb (the maintenance read, the list, or a + /// HEAD) before this line is ever reached, ending the page by exception instead. Malformed keys, + /// absent objects and token mismatches are final per-key outcomes and therefore do not by + /// themselves prevent progress. bool page_decided = !ambiguous && !suppress_deletes; for (const ListedKey & listed : page.keys) @@ -83,15 +103,15 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( if (ambiguous || suppress_deletes || catalog_cut.life_index.resolve(*life_id)) continue; - std::optional token = listed.token; - if (!token) + std::optional etag = listed.etag; + if (!etag) { try { - const HeadResult current = backend.head(listed.key); - if (!current.exists) + const std::optional current = op.head(listed.key, Retry::standard()); + if (!current) continue; - token = current.token; + etag = current->etag; } catch (const std::exception & e) { @@ -101,14 +121,14 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( continue; } } - if (!fence_held()) + if (!op.admitted()) { page_decided = false; break; } try { - if (backend.deleteExact(listed.key, *token).kind == DeleteOutcome::Kind::Deleted) + if (op.remove(listed.key, *etag, Retry::standard()) == Removal::Removed) ++result.deleted; } catch (const std::exception & e) @@ -122,7 +142,7 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( /// Recheck even when the page had no dead candidate. A tenure that observes fence loss after LIST /// or after the last exact delete must not publish progress. Loss after this check may still race /// with the leak-only maintenance CAS; already completed exact deletes remain safe to repeat. - if (page_decided && !fence_held()) + if (page_decided && !op.admitted()) page_decided = false; if (page_decided) @@ -130,7 +150,9 @@ NamespaceJanitorResult NamespaceJanitor::runOnePage( const GcMaintenanceState next{.janitor_cursor = page.next_cursor}; try { - (void)casGcMaintenanceState(backend, layout, progress.token, next); + const WriteResult published = casGcMaintenanceState(op, layout, progress.etag, next, Retry::standard()); + if (std::holds_alternative(published) || std::holds_alternative(published)) + result.anomalies.push_back("cursor publication did not commit"); } catch (const std::exception & e) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h index e60f87e2e6a0..d23d21402c73 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -20,13 +20,21 @@ struct NamespaceJanitorResult class NamespaceJanitor { public: - NamespaceJanitor(Backend & backend_, const Layout & layout_, size_t page_budget_) - : backend(backend_), layout(layout_), page_budget(page_budget_) {} + NamespaceJanitor(CasRequests & requests_, const Layout & layout_, size_t page_budget_) + : requests(requests_), layout(layout_), page_budget(page_budget_) {} - NamespaceJanitorResult runOnePage(bool suppress_deletes, const std::function & fence_held); + /// `liveness` is admitted once for the whole page (one `CasOperation` covers the read, the list, + /// every delete and the cursor publication): a fact the fence cannot see, such as "this tenure + /// still holds the GC round's own lease" -- see `CasRequests::admit`. It is SAMPLED BEFORE EVERY + /// REQUEST the page makes (and before every reissue of one), not just at the two points this + /// function itself checks `op.admitted()` -- so it must be cheap and must never throw. A sample + /// that returns false ends whichever request was about to be sent: a read verb (the maintenance + /// read, the list, a HEAD) throws out of this call, and a write verb (a delete, the cursor + /// publication) reports it as `GaveUp` rather than sending anything. + NamespaceJanitorResult runOnePage(bool suppress_deletes, Liveness liveness); private: - Backend & backend; + CasRequests & requests; const Layout & layout; size_t page_budget; }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.cpp index 17ba6337d1e4..28a471b673ff 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.cpp @@ -1,5 +1,5 @@ #include -#include +#include #include #include #include @@ -10,11 +10,14 @@ #include #include #include +#include #include #include +#include #include #include #include +#include #include #include @@ -40,9 +43,10 @@ void onGcEnumerationPage() /// namespace is rooted by `server_root_id`, but that id is a clean relative path and can contain slashes. /// Try namespace prefixes from longest to shortest and accept the first durable mount body. Without a /// mount there is no deletion authority, so the caller must leave the prefix untouched. The mount's -/// `writer_epoch` and `min_active` are the single durable epoch/floor pair used for eligibility, including +/// `writer_epoch` and `min_active_build_sequence` are the single durable epoch/floor pair used for eligibility, including /// across process replacement and the retired sentinel. -std::optional floorForNamespace(Pool & store, const RootNamespace & ns) +std::optional floorForNamespace(CasOperation & op, const Layout & layout, const RootNamespace & ns, + uint64_t * reads = nullptr) { const String & value = ns.string(); size_t pos = value.size(); @@ -55,7 +59,9 @@ std::optional floorForNamespace(Pool & store, const RootNamespace & const String server_root_id = value.substr(0, pos); if (!server_root_id.empty()) { - if (const auto got = store.backend().get(store.layout().mountKey(server_root_id))) + if (reads) + ++*reads; + if (const auto got = op.read(layout.mountKey(server_root_id), Retry::standard())) return decodeMountLease(got->bytes); } if (pos == 0) @@ -91,14 +97,13 @@ std::optional parseListedManifestObject(const Layout & lay /// The fold seal `gc/state` currently adopts, or `nullopt` when the pool has no `gc/state` or no seal /// at `(snap_generation, snap_attempt)` — a pool whose GC has never completed a round. It is read ONCE /// per sweep pass; every namespace the pass touches takes its coverage row out of the same object. -std::optional readAdoptedFoldSeal(Pool & store) +std::optional readAdoptedFoldSeal(CasOperation & op, const Layout & layout) { - const Layout & layout = store.layout(); - const auto state_got = store.backend().get(layout.gcStateKey()); + const auto state_got = op.read(layout.gcStateKey(), Retry::standard()); if (!state_got) return std::nullopt; const GcState state = decodeGcState(state_got->bytes); - const auto got = store.backend().get(layout.foldSealKey(state.snap_generation, state.snap_attempt)); + const auto got = op.read(layout.foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard()); if (!got) return std::nullopt; return decodeFoldSeal(got->bytes, state.snap_generation); @@ -171,9 +176,13 @@ struct NamespaceProtection /// invalid transaction (via the authority-grounded recovery / `decodeRefLogTxn`); the caller SKIPS the /// namespace's deletions on such a throw rather than substituting an empty owner set. /// -/// `work_budget`, when set, bounds the committed-tail walk below: each ref-log GET the walk issues -/// consumes one unit of `GcRoundWorkBudget::sweep_recovery_op_budget`, shared with every other -/// namespace this round touches. Exhaustion sets `NamespaceProtection::recovery_incomplete` and stops +/// `work_budget`, when set, bounds the committed-tail walk below: each ref-log the walk TAKES (decodes +/// and applies) consumes one unit of `GcRoundWorkBudget::sweep_recovery_op_budget`, shared with every +/// other namespace this round touches. A hinting reader may prefetch up to one window of logs beyond +/// the charged position before the walk reaches them -- those prefetches are not charged, since the +/// budget bounds decoded work, not requests in flight -- and a hint issued past an epoch's seal is +/// discarded rather than taken, bounding the waste of one crossing to one window, counted in +/// `CASGCReadAheadWasted`. Exhaustion sets `NamespaceProtection::recovery_incomplete` and stops /// the walk — it is deliberately NOT plumbed into `recoverRefTableDetailedFromAuthority` itself: that /// function is a shared recovery primitive also used by `fsck` (which needs a COMPLETE table to audit) /// and the GC rebuild path (which needs a complete table to reconstruct in-degree from scratch), so @@ -181,8 +190,8 @@ struct NamespaceProtection /// Reaching the budget before even calling `recoverRefTableDetailedFromAuthority` (already spent by an /// earlier namespace) skips that call entirely and reports incomplete immediately. NamespaceProtection activeManifestKeys( - Pool & store, const CatalogEntry & catalog_entry, const RefCkpt & ckpt, - const std::optional & coverage, GcRoundWorkBudget * work_budget = nullptr) + CasOperation & op, KeyReader & reader, const Layout & layout, const CatalogEntry & catalog_entry, + const RefCkpt & ckpt, const std::optional & coverage, GcRoundWorkBudget * work_budget = nullptr) { NamespaceProtection protection; if (work_budget && !work_budget->sweepRecoveryOpAvailable()) @@ -191,8 +200,6 @@ NamespaceProtection activeManifestKeys( return protection; } std::set & active = protection.active; - const Layout & layout = store.layout(); - Backend & backend = store.backend(); const RootNamespace & ns = catalog_entry.ns; const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(catalog_entry.ns, catalog_entry.incarnation); @@ -200,7 +207,7 @@ NamespaceProtection activeManifestKeys( /// The exact row and `_ckpt` come from the caller's frozen catalog cut. Do not resolve `ns` here: /// a later catalog cut can name a reborn life and turn this old life into an apparent orphan. const RecoveredRefTable recovered = recoverRefTableDetailedFromAuthority( - backend, layout, catalog_entry, ckpt); + op, layout, catalog_entry, ckpt, &reader); if (work_budget) ++work_budget->sweep_recovery_ops_used; /// one coarse unit for the snapshot+tail recovery itself const RefTableState & state = recovered.state; @@ -228,7 +235,7 @@ NamespaceProtection activeManifestKeys( renderRefTxnId(from_cursor)); const RefTxnId exact_next_epoch{from_cursor.writer_epoch + 1, 1}; const EpochCrossResult crossing = crossEpochFromSeal( - backend, layout, ns, from_cursor, std::nullopt, exact_next_epoch, life); + op, layout, ns, from_cursor, std::nullopt, exact_next_epoch, life); if (!crossing.proved()) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS orphan sweep: exact next-epoch record {} does not prove that folded cursor {} was its seal " @@ -252,7 +259,7 @@ NamespaceProtection activeManifestKeys( /// cursor, first try its ordinary same-epoch successor; only a 404 there can ask the shared /// chain proof to establish a cross with kind unknown. That preserves tails after a cleaned /// cursor without guessing that a missing cursor was a seal. - const auto cursor_got = backend.get(layout.refLogKey(life, cursor)); + const auto cursor_got = op.read(layout.refLogKey(life, cursor), Retry::standard()); if (cursor_got) { const RefLogTxn cursor_txn = decodeRefLogTxn( @@ -288,6 +295,37 @@ NamespaceProtection activeManifestKeys( } } + /// Guards the walk's most recently hinted-but-not-yet-taken range (from `arm`'s `from` up to one + /// window) so that an early exit from the loop below -- the work-budget `break`, the corrupted-tail + /// `throw`, or the missing-cursor epoch cross -- frees it instead of leaving it pinned against the + /// SAME reader for whatever this page reads next (later candidates, later namespaces). An ordinary + /// same-epoch advance re-arms it on the new position without discarding: those hints are still + /// wanted. A seal crossing discards explicitly, right where the crossing happens, and disarms so + /// this guard's own destructor does not repeat it. + struct OutstandingHintGuard + { + KeyReader & reader; + const Layout & layout; + const NamespaceLifeId & life; + RefTxnId from{}; + RefTxnId committed_through{}; + bool armed = false; + + void arm(const RefTxnId & from_, const RefTxnId & committed_through_) + { + from = from_; + committed_through = committed_through_; + armed = true; + } + void discardNow() + { + if (armed) + discardRefLogHintsOfEpoch(reader, layout, life, from, committed_through); + armed = false; + } + ~OutstandingHintGuard() { discardNow(); } + } outstanding_hints{reader, layout, life}; + while (id <= *ckpt.committed_through) { /// UNCERTAINTY, work-budget arm: the committed-tail walk is a finite but potentially huge range @@ -295,13 +333,19 @@ NamespaceProtection activeManifestKeys( /// namespace it touches. Stopping HERE — before the next GET — leaves `active`/ /// `tail_removal_targets` genuinely partial, so the caller must treat the whole namespace as /// undecided this page (fail-closed retain), never authorize a deletion from what was collected - /// so far. + /// so far. `outstanding_hints`'s destructor frees whatever the walk hinted ahead of this point. if (work_budget && !work_budget->sweepRecoveryOpAvailable()) { protection.recovery_incomplete = true; break; } - const auto got = backend.get(layout.refLogKey(life, id)); + if (id.ref_sequence < std::numeric_limits::max()) + { + const RefTxnId hint_from{id.writer_epoch, id.ref_sequence + 1}; + hintRefLogsWithinEpoch(reader, layout, life, hint_from, *ckpt.committed_through); + outstanding_hints.arm(hint_from, *ckpt.committed_through); + } + const auto got = reader.take(layout.refLogKey(life, id)); if (work_budget) ++work_budget->sweep_recovery_ops_used; if (!got) @@ -313,6 +357,9 @@ NamespaceProtection activeManifestKeys( throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS orphan sweep: committed tail log {} is absent under the supplied _ckpt frontier", renderRefTxnId(id)); + /// The old epoch's hints past this missing cursor do not exist either; discard them before + /// crossing, exactly as the seal path below does. + outstanding_hints.discardNow(); id = cross_from_missing_cursor(*prior); prior.reset(); prior_is_seal.reset(); @@ -333,6 +380,7 @@ NamespaceProtection activeManifestKeys( { if (is_seal) { + outstanding_hints.discardNow(); prior.reset(); prior_is_seal.reset(); } @@ -353,10 +401,11 @@ NamespaceProtection activeManifestKeys( NamespaceFoldView namespaceFoldView(Pool & store, const RootNamespace & ns) { + CasOperation op = store.openRequests().admit(); NamespaceFoldView view; - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(store.backend(), store.layout()); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, store.layout()); catalog_cut.life_index.throwIfAmbiguous("CAS orphan manifest sweep"); - view.coverage = coverageOf(readAdoptedFoldSeal(store), catalog_cut, ns); + view.coverage = coverageOf(readAdoptedFoldSeal(op, store.layout()), catalog_cut, ns); return view; } @@ -420,21 +469,21 @@ bool manifestDeletionPremise(const NamespaceFoldView & view, const ManifestKey & /// UNCERTAINTY, hold arm. A hold names the exact position the fold could not resolve, and everything /// at or above it is unaccounted -- including, for all this predicate can tell, the record that - /// grants or removes this very manifest. `classification == 4` is tested separately from the hold - /// even though the seal's strict grammar pairs them: the thing standing between a clamped namespace - /// and an irreversible delete must not be a codec invariant enforced somewhere else. + /// grants or removes this very manifest. `classification == Clamped` is tested separately from the + /// hold even though the seal's strict grammar pairs them: the thing standing between a clamped + /// namespace and an irreversible delete must not be a codec invariant enforced somewhere else. if (cov.hold) return retain(SweepRetainClass::Hold, "namespace held at " + renderRefTxnId(cov.hold->offending_position) + " (" + String{holdReasonToWord(cov.hold->reason)} + ", retried " + std::to_string(cov.hold->retry_count) + " round(s)): every record at or above " "that position is unaccounted for"); - if (cov.classification == 4) + if (cov.classification == CoverageClass::Clamped) return retain(SweepRetainClass::Hold, - "namespace coverage is classified clamped (4) with no hold recorded: whatever " + "namespace coverage is classified clamped with no hold recorded: whatever " "stopped the fold was not carried, so nothing above its cursor is accounted for"); - if (cov.classification == 0) + if (cov.classification == CoverageClass::Absent) return retain(SweepRetainClass::NoCoverage, - "namespace coverage is classified absent (0): no round folded it, so its cursor " + "namespace coverage is classified absent: no round folded it, so its cursor " "is not the result of any walk"); /// RULE 1 (spec §6). Grants do not cross epochs, so every `+1` that could name an epoch-`E` build @@ -473,13 +522,22 @@ bool manifestDeletionPremise(const NamespaceFoldView & view, const ManifestKey & return true; } -bool prefixEligible(Pool & store, const RootNamespace & ns, const BuildPrefix & prefix) +namespace +{ + +/// Eligibility comes only from the durable mount-lease floor. A missing floor means NOT eligible; +/// do not replace that authority check with a frozen-sequence or judged-dead guess. Compare +/// `writer_epoch` first, then `build_sequence`, so old-epoch +/// debris drains after a process restart even when its build_sequence is above the current min_active_build_sequence. +bool prefixEligibleOn(CasOperation & op, const Layout & layout, const RootNamespace & ns, const BuildPrefix & prefix) +{ + return prefixEligibleUnder(floorForNamespace(op, layout, ns), prefix); +} + +} + +bool prefixEligibleUnder(const std::optional & floor, const BuildPrefix & prefix) { - /// Eligibility comes only from the durable mount-lease floor. A missing floor means NOT eligible; - /// do not replace that authority check with a frozen-sequence or judged-dead guess. Compare - /// `writer_epoch` first, then `build_sequence`, so old-epoch - /// debris drains after a process restart even when its build_sequence is above the current min_active. - const auto floor = floorForNamespace(store, ns); if (!floor) return false; @@ -488,19 +546,24 @@ bool prefixEligible(Pool & store, const RootNamespace & ns, const BuildPrefix & return true; if (prefix.writer_epoch > w.writer_epoch) return false; - if (w.min_active == std::numeric_limits::max()) + if (w.min_active_build_sequence == std::numeric_limits::max()) return true; /// farewell/retired sentinel: every seq is retired - return w.min_active > prefix.build_sequence; + return w.min_active_build_sequence > prefix.build_sequence; +} + +bool prefixEligible(Pool & store, const RootNamespace & ns, const BuildPrefix & prefix) +{ + CasOperation op = store.openRequests().admit(); + return prefixEligibleOn(op, store.layout(), ns, prefix); } uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefix & prefix, std::vector * warnings) { - if (!prefixEligible(store, ns, prefix)) - return 0; /// not eligible by the durable watermark fact — delete nothing (controls #8/#9) - const Layout & layout = store.layout(); - Backend & backend = store.backend(); + CasOperation op = store.openRequests().admit(); + if (!prefixEligibleOn(op, layout, ns, prefix)) + return 0; /// not eligible by the durable watermark fact — delete nothing (controls #8/#9) /// Build the protection view. A missing snapshot body, an invalid transaction, or an incomplete /// ordered view throws, causing the sweep to skip deletion and surface the error; it never substitutes @@ -512,9 +575,9 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi /// the same way the periodic sweep always has: skip and retry next round. /// The §6 premise's durable half, read before the protection view so both share one seal read. NamespaceFoldView view; - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(store.backend(), store.layout()); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); catalog_cut.life_index.throwIfAmbiguous("CAS orphan manifest sweep"); - view.coverage = coverageOf(readAdoptedFoldSeal(store), catalog_cut, ns); + view.coverage = coverageOf(readAdoptedFoldSeal(op, layout), catalog_cut, ns); const CatalogEntry * catalog_entry = catalogEntryOf(catalog_cut, ns); if (!catalog_entry || catalog_entry->state == NsState::Creating) return 0; /// absent/Creating names have no recovery authority and therefore no deletion authority @@ -523,7 +586,7 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi try { const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(catalog_entry->ns, catalog_entry->incarnation); - const std::optional ckpt = readCkpt(backend, layout, life); + const std::optional ckpt = readCkpt(op, layout, life); if (!ckpt) { const String warning = "CAS orphan sweep: namespace " + ns.string() @@ -533,7 +596,8 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi warnings->push_back(warning); return 0; } - protection = activeManifestKeys(store, *catalog_entry, ckpt->ckpt, view.coverage); + InlineKeyReader reader(op); + protection = activeManifestKeys(op, reader, layout, *catalog_entry, ckpt->ckpt, view.coverage); view.tail_removal_targets = protection.tail_removal_targets; } @@ -554,10 +618,10 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi + renderRefTxnId(RefTxnId{prefix.writer_epoch, prefix.build_sequence}) + "/"; uint64_t deleted = 0; - forEachListedKey(backend, prefix_key, [&](const ListedKey & listed) + op.forEachListedKey(prefix_key, [&](const ListedKey & listed) { if (protection.active.contains(listed.key)) - return; /// owned by a committed or precommit owner — never sweep + return true; /// owned by a committed or precommit owner — never sweep /// THE §6 SAFETY FLOOR, under the watermark eligibility already established above. The /// watermark says the build is retired; the premise says the ref stream can be shown not to @@ -569,22 +633,23 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi "CAS orphan sweep: retaining {} -- {}", listed.key, retain_reason); if (warnings) warnings->push_back("CAS orphan sweep: retained " + listed.key + " -- " + retain_reason); - return; + return true; } - /// Exact-token delete: HEAD for the current token, then deleteExact. A 404 between HEAD and - /// delete (or a TokenMismatch — a fresh owner reclaimed it) is tolerated (record-and-continue), - /// same as always, regardless of `warnings` -- that is the normal "someone else already reclaimed - /// it" race, not a failure. A THROWN exception (a transient backend hiccup) is the one thing - /// `warnings` changes: opted-in (non-null), it is recorded and the sweep moves to the next key; - /// opted-out (nullptr, every pre-existing caller), it propagates exactly as before (fail-close). + /// Exact-token delete: HEAD for the current incarnation, then remove exactly it -- never + /// `removeCurrent`, whose re-head would delete whatever a fresh owner put there instead. A + /// `Gone` (a 404 between the two) or a `Mismatch` (a fresh owner reclaimed the key) is + /// tolerated, same as always, regardless of `warnings` -- that is the normal "someone else + /// already reclaimed it" race, not a failure. A THROWN exception (a transient backend hiccup) + /// is the one thing `warnings` changes: opted-in (non-null), it is recorded and the sweep moves + /// to the next key; opted-out (nullptr, every pre-existing caller), it propagates exactly as + /// before (fail-close). try { - const HeadResult head = backend.head(listed.key); - if (!head.exists) - return; - const DeleteOutcome outcome = backend.deleteExact(listed.key, head.token); /// NotFound/TokenMismatch spared - if (classifyDeleteOutcome(outcome) == DeleteClass::Deleted) + const std::optional head = op.head(listed.key, Retry::standard()); + if (!head) + return true; + if (op.remove(listed.key, head->etag, Retry::standard()) == Removal::Removed) ++deleted; } catch (...) @@ -594,7 +659,8 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi warnings->push_back("CAS orphan sweep: " + listed.key + " delete failed: " + getCurrentExceptionMessage(/*with_stacktrace=*/false)); } - }, 1000, onGcEnumerationPage); + return true; + }, Retry::standard(), 1000, onGcEnumerationPage); return deleted; } @@ -604,54 +670,44 @@ ManifestSweepResult planManifestCursorPage( uint64_t list_budget, uint64_t nomination_budget, bool catalog_recovery_authoritative, - GcRoundWorkBudget * work_budget) + GcRoundWorkBudget * work_budget, + ThreadPool * read_pool, + size_t read_concurrency) { ManifestSweepResult result; result.next_cursor = cursor; if (list_budget == 0) return result; - Backend & backend = store.backend(); const Layout & layout = store.layout(); - const ListPage page = backend.list(layout.casManifestsPrefix(), cursor, list_budget); + CasOperation op = store.openRequests().admit(); + /// The page's reader. With a pool and concurrency above one the candidates' bodies and the two + /// ref-stream walks overlap their round trips; otherwise every read is inline and the page is the + /// sequential one, request for request. + std::optional read_ahead; + std::unique_ptr reader; + if (read_pool && read_concurrency > 1) + { + read_ahead.emplace(op, store.openRequests(), *read_pool, read_concurrency); + reader = std::make_unique(*read_ahead); + } + else + reader = std::make_unique(op); + const ListPage page = op.list(layout.casManifestsPrefix(), cursor, list_budget, Retry::standard()); /// This pass fetches exactly one page per round (the cursor advances across rounds, not within this /// call), so the metric increments once per call, not once per listed key. ProfileEvents::increment(ProfileEvents::CASGCEnumerationPages); - /// Freeze every possible destructive candidate BEFORE the later catalog cut. A same-name rebirth can - /// replace this logical manifest key between the observations; classifying the old bytes against the - /// later lifecycle cut is safe only when deletion retains the old exact token, so the replacement - /// loses `deleteExact`. Do not take a fresh GET after the catalog read: that would splice new-life - /// bytes into old candidate selection and authorize their deletion with the new token. - /// - /// Bounded to `nomination_budget` well-formed keys — never the whole `list_budget`-sized - /// page — since `nomination_budget` is the hard ceiling on how many of them this call can ever - /// nominate. A well-formed key beyond this cap has no frozen body; it is retained where its absence - /// is discovered below, in the exact same "budget exhausted, cursor does not step over it" shape the - /// nomination-count exhaustion already uses. - std::map> observed_candidates; - if (nomination_budget > 0) - { - uint64_t frozen = 0; - for (const ListedKey & listed : page.keys) - { - if (frozen >= nomination_budget) - break; - if (parseListedManifestObject(layout, listed.key)) - { - observed_candidates.emplace(listed.key, backend.get(listed.key)); - ++frozen; - } - } - } - /// One seal and one later catalog cut for the whole page; every namespace joins through those same /// immutable observations. - const std::optional adopted_seal = readAdoptedFoldSeal(store); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const std::optional adopted_seal = readAdoptedFoldSeal(op, layout); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); catalog_cut.life_index.throwIfAmbiguous("CAS orphan manifest sweep"); - std::map eligible_by_prefix; + /// One floor observation per namespace for the whole page. Every build of a namespace is judged + /// against the same mount body; reading it per build prefix would cost three requests per listed + /// key on a pool where every INSERT is its own build. + std::map> floor_by_ns; std::map view_by_ns; std::map> active_by_ns; std::set errored_namespaces; /// protection view unavailable => skip, never delete @@ -662,6 +718,11 @@ ManifestSweepResult planManifestCursorPage( String decided_through; bool budget_exhausted = false; + /// Keys that passed every retain check. Their bodies are read AFTER the loop, and after the + /// catalog cut: a manifest key is write-once, so its bytes do not depend on when they are read, + /// and the only thing a later read can observe differently is absence, which retains. + std::vector candidates; + for (const ListedKey & listed : page.keys) { ++result.listed; @@ -674,7 +735,7 @@ ManifestSweepResult planManifestCursorPage( /// A budget of ZERO is not exhaustion but a list-only pass: nothing is ever deletable, so /// freezing the cursor on it would make the sweep spin on one page forever. That pass keeps /// its pre-existing behaviour and advances. - if (budget_exhausted || (nomination_budget > 0 && result.nominations.size() >= nomination_budget)) + if (budget_exhausted || (nomination_budget > 0 && candidates.size() >= nomination_budget)) { budget_exhausted = true; ++result.skipped; @@ -700,13 +761,13 @@ ManifestSweepResult planManifestCursorPage( const CatalogEntry * catalog_entry = catalogEntryOf(catalog_cut, parsed->ns); if (catalog_entry) { - const String eligibility_key = parsed->ns.string() + "\n" - + std::to_string(parsed->prefix.writer_epoch) + "\n" - + std::to_string(parsed->prefix.build_sequence); - auto [eligible_it, eligible_inserted] = eligible_by_prefix.emplace(eligibility_key, false); - if (eligible_inserted) - eligible_it->second = prefixEligible(store, parsed->ns, parsed->prefix); - if (!eligible_it->second) + auto [floor_it, floor_inserted] = floor_by_ns.emplace(parsed->ns.string(), std::nullopt); + if (floor_inserted) + { + ++result.floor_lookups; + floor_it->second = floorForNamespace(op, layout, parsed->ns, &result.floor_reads); + } + if (!prefixEligibleUnder(floor_it->second, parsed->prefix)) { ++result.skipped; decided_through = listed.key; @@ -767,7 +828,7 @@ ManifestSweepResult planManifestCursorPage( { const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry( catalog_entry->ns, catalog_entry->incarnation); - const std::optional ckpt = readCkpt(backend, layout, life); + const std::optional ckpt = readCkpt(op, layout, life); if (!ckpt) { LOG_WARNING(getLogger("CasOrphanManifestSweep"), @@ -778,7 +839,7 @@ ManifestSweepResult planManifestCursorPage( else { NamespaceProtection protection = activeManifestKeys( - store, *catalog_entry, ckpt->ckpt, view_it->second.coverage, work_budget); + op, *reader, layout, *catalog_entry, ckpt->ckpt, view_it->second.coverage, work_budget); if (protection.recovery_incomplete) { /// The committed-tail walk stopped early: `active`/`tail_removal_targets` @@ -854,25 +915,21 @@ ManifestSweepResult planManifestCursorPage( } } - /// This exact token and bytes were captured before the catalog cut (see above). A missing body - /// has no deletion authority; a later replacement loses the old token at `deleteExact`. - /// - /// A well-formed key can legitimately be ABSENT here: the freeze loop above caps - /// fan-out at `nomination_budget` candidates, so a key beyond that cap was never frozen. Treat - /// it exactly like nomination-count exhaustion -- retain, and do NOT advance the cursor past - /// it, so the very next page/round examines it with a fresh budget instead of losing it. - const auto observed_it = observed_candidates.find(parsed->key); - if (observed_it == observed_candidates.end()) - { - budget_exhausted = true; - ++result.skipped; - continue; - } - const std::optional & got = observed_it->second; + candidates.push_back(*parsed); + decided_through = listed.key; + } + + size_t next_body_hint = 0; + for (const ListedManifestObject & candidate : candidates) + { + while (next_body_hint < candidates.size() && reader->pending() < reader->window()) + reader->hint(candidates[next_body_hint++].key); + const std::optional got = reader->take(candidate.key); if (!got) { + /// Gone since the LIST: a fresh writer never reuses the key, so there is nothing to + /// nominate and nothing to retain. The key was decided above; the cursor stands. ++result.skipped; - decided_through = listed.key; continue; } std::optional body; @@ -889,22 +946,21 @@ ManifestSweepResult planManifestCursorPage( /// reclamation for the whole pool rather than for this one key. LOG_ERROR(getLogger("CasOrphanManifestSweep"), "CAS orphan sweep: manifest at {} cannot be decoded and was retained; run cas-fsck to " - "enumerate such objects", parsed->key); + "enumerate such objects", candidate.key); ++result.undecodable; ++result.skipped; - decided_through = listed.key; continue; } - const ManifestId id{parsed->ns, parsed->ref}; + const ManifestId id{candidate.ns, candidate.ref}; if (!refMatchesBody(id.ref, *body) || !manifestNamespaceMatches(id.root_namespace, *body)) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS orphan sweep: manifest identity mismatch at {} while deriving exact source edges", - parsed->key); + candidate.key); ManifestSweepResult::Nomination nomination{ .id = id, - .key = parsed->key, - .token = got->token, + .key = candidate.key, + .token = PersistedEtag::capture(got->etag), .source_retirements = {}}; for (const ManifestEntry & entry : body->entries) if (entry.placement == EntryPlacement::Blob) @@ -912,7 +968,6 @@ ManifestSweepResult planManifestCursorPage( .ref = entry.ref, .source_id = sourceEdgeId(id, entry.path)}); result.nominations.push_back(std::move(nomination)); - decided_through = listed.key; } if (budget_exhausted) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.h index d2f44b205cb7..d4da7e38f5a5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasOrphanManifestSweep.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -39,7 +40,7 @@ struct ManifestKey enum class SweepRetainClass : uint8_t { None = 0, /// the premise admitted the deletion; no retention happened - NoCoverage, /// no sealed coverage row for the namespace (a classification-0 row counts here) + NoCoverage, /// no sealed coverage row for the namespace (a classification-`Absent` row counts here) Hold, /// the namespace is held, or is classified clamped UnconsumedSeal, /// rule (1): the cursor has not consumed the build epoch's closing seal TailRemoval, /// rule (2): an unconsumed tail record names this manifest as a removal target @@ -131,13 +132,21 @@ struct ManifestSweepResult uint64_t retained_tail_removal = 0; uint64_t retained_work_budget = 0; + /// The floor a namespace's builds are judged against is one mount body per server root. A page + /// resolves it once per namespace (`floor_lookups`); each lookup reads the mount key of every + /// `/`-prefix of the namespace until one answers (`floor_reads`), so an absent mount costs the + /// whole chain once. + uint64_t floor_lookups = 0; + uint64_t floor_reads = 0; + /// Exact-GET/decode candidates. The reducer must adopt every `source_retirements` entry before the - /// caller may exact-token-delete `key` with `token`. + /// caller may delete `key`, and only after re-observing `token` at it: a key whose incarnation + /// moved on belongs to a fresh owner and must be left alone. struct Nomination { ManifestId id; String key; - Token token; + PersistedEtag token; std::vector source_retirements; }; std::vector nominations; @@ -153,7 +162,7 @@ struct ManifestSweepResult /// manifest bodies written before `PrecommitAdd` and never named by any live owner, scoped to ONE /// namespace + ONE build prefix. Rules: /// - eligibility from the durable watermark fact only: the retired sentinel -/// (`min_active == UINT64_MAX`), or `min_active > build_sequence`, or a replaced incarnation — +/// (`min_active_build_sequence == UINT64_MAX`), or `min_active_build_sequence > build_sequence`, or a replaced incarnation — /// NEVER a frozen-seq / judged-dead heuristic alone (a missing watermark => not eligible); /// - the active `ManifestId` set comes from the namespace's committed + live-precommit owner view; /// - delete only bodies whose `ManifestId` is ABSENT from the active set, by exact token; @@ -161,19 +170,19 @@ struct ManifestSweepResult /// - a 404 between listing and deletion is record-and-continue, never a throw; /// - never GETs a condemned body to revive it — eligibility + /// exact-token delete only. -/// Returns the number of bodies actually deleted (a `DeleteClass::Deleted`-classified exact-token -/// delete only, never a spared `NotFound`/`TokenMismatch`) — the decommission manifest-debris drain +/// Returns the number of bodies actually deleted (a `Removal::Removed` exact-token +/// delete only, never a spared `Gone`/`Mismatch`) — the decommission manifest-debris drain /// (`Core/CasDecommission.cpp`) sums this across every eligible build prefix into /// `DecommissionReport::manifest_debris_removed`. /// /// `warnings`, when non-null, opts in to the decommission drain's tolerate-and-continue contract: a -/// per-key transient failure (a thrown backend exception on `head`/`deleteExact`) +/// per-key transient failure (a thrown backend exception on `head`/`remove`) /// is pushed onto `*warnings` and the sweep continues with the next key, instead of throwing out of /// this call; likewise a protection-view-unavailable namespace (the pre-existing corrupt-snapshot skip /// below) also pushes a "cannot confirm emptiness" warning, not just a `LOG_WARNING`. `warnings == /// nullptr` (the default, every pre-existing caller) preserves the original behaviour exactly: a /// per-key failure propagates as an exception (fail-close default), and the protection-view skip is -/// log-only. `NotFound`/`TokenMismatch` delete outcomes stay silently spared either way — those are the +/// log-only. `Gone`/`Mismatch` delete outcomes stay silently spared either way — those are the /// normal "a fresh owner reclaimed it" race the periodic sweep expects, not a failure to warn about. /// This direct decommission path relies on the caller's held server-root claim/fence: while that claim /// is held, a same-server-root rebirth cannot become live between its catalog cut and exact-token delete. @@ -186,24 +195,41 @@ uint64_t sweepNamespace(Pool & store, const RootNamespace & ns, const BuildPrefi /// judged-dead heuristic. A missing lease provides no deletion authority, so the prefix is not eligible. bool prefixEligible(Pool & store, const RootNamespace & ns, const BuildPrefix & prefix); -/// Plan one cursor page without deleting. Every candidate is exact-GET, decoded and identity-validated; -/// its exact manifest-source edges are returned for accounting-neutral retirement in the next fold. +/// The pure half of `prefixEligible`: whether `prefix` is retired under one observation of the mount +/// floor. `nullopt` (no mount body under any prefix of the namespace) admits nothing. Retirement is +/// permanent -- the epoch and the acknowledgement floor only grow and the farewell is terminal -- so +/// an admission derived from any observation stays true afterwards, which is what lets a page judge +/// every build of a namespace against one read. +bool prefixEligibleUnder(const std::optional & floor, const BuildPrefix & prefix); + +/// Plan one cursor page without deleting. A key is decided from key-derived facts first; only a +/// candidate's body is read, after the catalog cut, then decoded and identity-validated; its exact +/// manifest-source edges are returned for accounting-neutral retirement in the next fold. /// Catalog-named namespaces are retain-only unless the caller explicitly authorizes recovery from its /// frozen catalog cut and the exact `_ckpt` frontier of the life named there. /// -/// `work_budget`, when set, bounds the body-GET/retention fan-out to `nomination_budget` well-formed -/// candidates (never the whole `list_budget`-sized page), caps how many DISTINCT namespaces this page -/// may build a fresh protection view for, and caps the committed-tail recovery walk's ref-log GET +/// `nomination_budget` is a candidate budget: the page stops deciding once it has that many +/// candidates, and reads exactly that many bodies at most. `work_budget`, when set, additionally +/// caps how many DISTINCT namespaces this page may build a fresh protection view for, and caps the +/// committed-tail recovery walk's ref-log GET /// count cumulatively across the round (shared with every other destructive-work family via the same /// `GcRoundWorkBudget` instance). Exhausting either cap retains every remaining candidate belonging to /// the affected namespace on THIS page rather than deciding it without a complete protection view; /// `nullptr` (the default) reproduces the pre-budget unbounded behavior. +/// +/// `read_pool` and `read_concurrency` (default `nullptr`/1, i.e. every read inline on the caller's +/// operation) drive the page's read-ahead: with a pool and a concurrency above one, the candidates' +/// bodies and the catalog-recovery/committed-tail ref-log walks overlap their round trips on `read_pool` +/// instead of serializing request by request. Every decision this function makes is unchanged by that +/// choice; see `GcReadAhead` for what may be fetched early and why. ManifestSweepResult planManifestCursorPage( Pool & store, const String & cursor, uint64_t list_budget, uint64_t nomination_budget, bool catalog_recovery_authoritative, - GcRoundWorkBudget * work_budget = nullptr); + GcRoundWorkBudget * work_budget = nullptr, + ThreadPool * read_pool = nullptr, + size_t read_concurrency = 1); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.cpp index 899f93504fde..56244a5d39bf 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.cpp @@ -1,5 +1,5 @@ #include -#include +#include #include #include @@ -18,14 +18,10 @@ namespace DB::Cas { CatalogLifecycleReconciler::CatalogLifecycleReconciler( - Backend & backend_, const Layout & layout_, const CasFoldSeal & adopted_parent_, - uint64_t admitted_generation_, - std::function check_fence_) - : backend(backend_) + CasOperation & op_, const Layout & layout_, const CasFoldSeal & adopted_parent_) + : op(op_) , layout(layout_) , adopted_parent(adopted_parent_) - , admitted_generation(admitted_generation_) - , check_fence(std::move(check_fence_)) { } @@ -62,7 +58,8 @@ CatalogResolution CatalogLifecycleReconciler::resolveExactRow( return CatalogResolution::ExactRowStillPresent; } -CatalogLifecycleReconcileResult CatalogLifecycleReconciler::reconcile() +CatalogLifecycleReconcileResult CatalogLifecycleReconciler::reconcile( + const std::function & refresh_authority) { CatalogLifecycleReconcileResult result{ .authority_status = AuthorityStatus::Authoritative, @@ -70,14 +67,19 @@ CatalogLifecycleReconcileResult CatalogLifecycleReconciler::reconcile() .retired_lives = {}, .final_catalog_cut = std::nullopt, .deleted = 0}; - CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); for (;;) { const std::optional eligible = selectEligible(catalog); if (!eligible) { - if (check_fence(admitted_generation) == CasRefCatalog::LeaderFenceStatus::Moved) + /// The verdict gets its own refresh, not just each erase: a deposition landing after the + /// last erase is invisible to the reading that erase took, so without this the drain hands + /// a deposed leader `Authoritative` and the round only learns better at its `gc/state` + /// commit -- after a ref walk and a fold seal it never had the authority to build. + refresh_authority(); + if (!op.admitted()) { result.authority_status = AuthorityStatus::FencedOut; return result; @@ -89,8 +91,7 @@ CatalogLifecycleReconcileResult CatalogLifecycleReconciler::reconcile() CasRefCatalog::CompletedRemovingDeleteResult delete_result = CasRefCatalog::deleteCompletedRemovingAtSnapshot( - backend, layout, std::move(catalog), *eligible, adopted_parent, - admitted_generation, check_fence); + op, layout, std::move(catalog), *eligible, adopted_parent, refresh_authority); if (!delete_result.catalog_snapshot) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS catalog lifecycle reconciliation returned no catalog resolution snapshot"); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.h index fd34011c64ef..1069bc9d63f5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CatalogLifecycleReconciler.h @@ -43,22 +43,22 @@ class CatalogLifecycleReconciler { public: CatalogLifecycleReconciler( - Backend & backend_, const Layout & layout_, const CasFoldSeal & adopted_parent_, - uint64_t admitted_generation_, - std::function check_fence_); + CasOperation & op_, const Layout & layout_, const CasFoldSeal & adopted_parent_); - CatalogLifecycleReconcileResult reconcile(); + /// `refresh_authority` is forwarded to each erase, which runs it at the top of every attempt, so + /// every erase is authorised by a reading taken in its own attempt, and it is run once more before + /// the drain-complete verdict, which would otherwise report from the reading its last erase left. + /// It is a refresh, not a verdict: the verdict stays `op.admitted()`. + CatalogLifecycleReconcileResult reconcile(const std::function & refresh_authority); private: std::optional selectEligible(const CasRefCatalog::Snapshot & catalog) const; static CatalogResolution resolveExactRow( const CasRefCatalog::Snapshot & catalog, const CatalogEntry & observed); - Backend & backend; + CasOperation & op; const Layout & layout; const CasFoldSeal & adopted_parent; - uint64_t admitted_generation; - std::function check_fence; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.cpp index bce2e9423704..b29fe64b872d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.cpp @@ -29,7 +29,6 @@ namespace ProfileEvents extern const Event CASPartFolderViewOversizedBypasses; extern const Event CASPartFolderViewInvalidations; extern const Event CASRefRollbackBestEffortDropFailed; - extern const Event CASPartFolderValidateSkipped; extern const Event CASRefRepoint; } @@ -43,12 +42,11 @@ namespace DB::Cas { PartFolderView::PartFolderView(PartRefKey key_, Cas::ManifestId manifest_id_, uint64_t manifest_size_, - std::shared_ptr manifest_, uint64_t validated_at_ms_) + std::shared_ptr manifest_) : key(std::move(key_)) , manifest_id(std::move(manifest_id_)) , manifest_size(manifest_size_) , manifest_body(std::move(manifest_)) - , validated_at_ms(validated_at_ms_) { chassert(manifest_body); /// The binary-search contract: entries must be strictly ascending by `path` (sorted and unique) — @@ -62,12 +60,10 @@ PartFolderView::PartFolderView(PartRefKey key_, Cas::ManifestId manifest_id_, ui } std::shared_ptr PartFolderView::make( - PartRefKey key, const Cas::Resolved & resolved, std::shared_ptr manifest, - uint64_t validated_at_ms) + PartRefKey key, const Cas::Resolved & resolved, std::shared_ptr manifest) { return std::make_shared( - std::move(key), resolved.manifest_id, resolved.manifest_size, - std::move(manifest), validated_at_ms); + std::move(key), resolved.manifest_id, resolved.manifest_size, std::move(manifest)); } std::optional PartFolderView::projectionDirPrefix(const std::string & file) @@ -145,11 +141,9 @@ CachedPartFolderAccess::CachedPartFolderAccess(Cas::PoolPtr store_) { } -CachedPartFolderAccess::CachedPartFolderAccess(Cas::PoolPtr store_, CacheParams params_, std::function now_ms_fn_) - : store(std::move(store_)), params(params_), now_ms_fn(std::move(now_ms_fn_)) +CachedPartFolderAccess::CachedPartFolderAccess(Cas::PoolPtr store_, CacheParams params_) + : store(std::move(store_)), params(params_) { - if (!now_ms_fn) - now_ms_fn = []() -> uint64_t { return timeInMilliseconds(std::chrono::system_clock::now()); }; if (params.cache_bytes > 0) view_cache = std::make_unique( "LRU", CurrentMetrics::CASPartFolderCacheBytes, CurrentMetrics::CASPartFolderCacheEntries, @@ -171,8 +165,7 @@ CachedPartFolderAccess::getView(const PartRefKey & key, Freshness freshness) con const String cache_key = key.cacheKey(); /// Retained views serve `CachedForLoad` directly only after their manifest ID matches the fresh - /// resolve. `ForceFresh` must re-prove the manifest body unless the configured validation policy - /// explicitly permits a recent retained view; a fresh ref resolve proves ref currency, not body existence. + /// resolve. `ForceFresh` and `StrictValidate` always rebuild from the pool's manifest cache. if (freshness == Freshness::CachedForLoad && view_cache) { if (auto cached = view_cache->get(cache_key)) @@ -190,28 +183,6 @@ CachedPartFolderAccess::getView(const PartRefKey & key, Freshness freshness) con } } - /// With a non-`Always` validation policy, `ForceFresh` may serve a retained view without another - /// body HEAD when its manifest ID still matches and its validation timestamp is within the age - /// policy. `StrictValidate` bypasses retention. A manifest-ID mismatch always rebuilds, because all - /// part content is represented by the manifest. - if (freshness == Freshness::ForceFresh && view_cache && params.validate.mode != PartFolderValidate::Mode::Always) - { - if (auto cached = view_cache->get(cache_key); - cached && cached->manifestId() == resolved->manifest_id) - { - const bool fresh_enough = params.validate.mode == PartFolderValidate::Mode::Never - || (now_ms_fn() - cached->validatedAtMs()) < params.validate.age_seconds * 1000ULL; - if (fresh_enough) - { - ProfileEvents::increment(ProfileEvents::CASPartFolderViewHits); - ProfileEvents::increment(ProfileEvents::CASPartFolderValidateSkipped); - recordDecision(cache_key, LastDecision::Hit, cached.get(), /*retained=*/true); - emitResolveEvent(key, *resolved); - return cached; - } - } - } - auto view = buildView(key, *resolved, freshness); /// Retain eligible views. `StrictValidate` never populates the cache, and oversized views are @@ -264,10 +235,10 @@ void CachedPartFolderAccess::emitResolveEvent(const PartRefKey & key, const Cas: std::shared_ptr CachedPartFolderAccess::buildView( const PartRefKey & key, const Cas::Resolved & resolved, Freshness freshness) const { - /// Fresh modes do not coalesce: each `ForceFresh`/`StrictValidate` call owns its mandatory HEAD. - /// Only cold `CachedForLoad` builds use single-flight. + /// Fresh modes do not coalesce: each `ForceFresh`/`StrictValidate` call owns its own read (a cache + /// hit costs no request). Only cold `CachedForLoad` builds use single-flight. if (freshness != Freshness::CachedForLoad) - return PartFolderView::make(key, resolved, store->readManifestShared(resolved.manifest_id), now_ms_fn()); + return PartFolderView::make(key, resolved, store->readManifestShared(resolved.manifest_id)); std::promise> promise; std::shared_future> future; @@ -292,7 +263,7 @@ std::shared_ptr CachedPartFolderAccess::buildView( }); try { - auto view = PartFolderView::make(key, resolved, store->readManifestShared(resolved.manifest_id), now_ms_fn()); + auto view = PartFolderView::make(key, resolved, store->readManifestShared(resolved.manifest_id)); promise.set_value(view); return view; } @@ -505,9 +476,10 @@ Cas::CommitOutcome CachedPartFolderAccess::publishEntries(const PartRefKey & dst bool CachedPartFolderAccess::republishRef(const PartRefKey & src, const PartRefKey & dst) { - /// Content addressing has no rename, so move a committed ref by reading the source body freshly, - /// publishing equivalent entries at the destination, and then dropping the source. The source - /// body is re-proved and is never taken from a retained view. + /// Content addressing has no rename, so move a committed ref by reading the source manifest + /// through the pool's manifest cache after a fresh ref resolve, publishing equivalent entries at + /// the destination, and then dropping the source. The source blobs are adopted by evidence of the + /// live source edge, never re-probed; the decode is never taken from a retained view. auto resolved = store->resolveRef(src.ns, src.ref); if (!resolved) return false; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.h index cfa81f83662a..a9f4f3c40665 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Parts/PartFolderAccess.h @@ -52,11 +52,10 @@ struct CommitOutcome bool created = false; }; -/// Read-freshness policy at the part-folder access boundary. The -/// mutable-read-vs-write-evidence distinction is carried by the METHOD, not a fourth value: -/// mutable per-part reads call `resolve` (no manifest involved); write-path source reads call -/// `getView`, which under ForceFresh always re-proves the manifest body (mandatory HEAD in -/// `readManifestShared` — a fresh ref resolve alone proves ref currency, NOT body existence). +/// Read-freshness policy at the part-folder access boundary. The mutable-read-vs-write-evidence +/// distinction is carried by the METHOD, not a fourth value: mutable per-part reads call `resolve` +/// (no manifest involved); write-path source reads call `getView`, which under ForceFresh always +/// resolves fresh and bypasses the retained view. enum class Freshness { CachedForLoad, /// repeated load-window reads; stale-tolerant resolve (allow_stale=true) @@ -79,15 +78,12 @@ class PartFolderView /// must be non-null and its entries must be strictly ascending by canonical path; this is the /// ordering required by the binary-search and range-scan helpers. PartFolderView(PartRefKey key_, Cas::ManifestId manifest_id_, uint64_t manifest_size_, - std::shared_ptr manifest_, uint64_t validated_at_ms_); + std::shared_ptr manifest_); - /// Joins a fresh `Resolved` with its validated shared decode. `validated_at_ms` is supplied by - /// the caller after `readManifestShared` has proven the manifest body with a HEAD. Keeping the - /// timestamp outside this helper lets `CachedPartFolderAccess` use one injectable clock for both - /// the stamp and its age-window comparison. + /// Joins a fresh `Resolved` with its validated shared decode. static std::shared_ptr make( PartRefKey key, const Cas::Resolved & resolved, - std::shared_ptr manifest, uint64_t validated_at_ms); + std::shared_ptr manifest); /// Recognizes a projection directory by its last path component, `.proj` or `.tmp_proj`, and /// returns the corresponding in-tree prefix. The input is the routed file path; unrelated paths @@ -97,10 +93,6 @@ class PartFolderView const PartRefKey & refKey() const { return key; } const Cas::ManifestId & manifestId() const { return manifest_id; } const std::shared_ptr & manifest() const { return manifest_body; } - /// The wall-clock ms at which this view's manifest body was last proven live by a HEAD. A - /// refresh that changes only ref metadata carries the original stamp forward because it did not - /// re-prove the body. - uint64_t validatedAtMs() const { return validated_at_ms; } /// Finds an entry by canonical path using the manifest's sorted-entry invariant. const Cas::ManifestEntry * findFile(const String & path) const; @@ -122,7 +114,6 @@ class PartFolderView Cas::ManifestId manifest_id; uint64_t manifest_size = 0; std::shared_ptr manifest_body; - uint64_t validated_at_ms = 0; }; } @@ -132,17 +123,6 @@ namespace DB::Cas { class PartWriteTxn; } namespace DB::Cas { -/// Controls whether `ForceFresh` must re-prove the manifest body on every access. `Always` (the default) -/// preserves the fail-closed body check; `Age` and `Never` may serve a retained view after a fresh ref -/// resolve when its manifest ID matches. A ref resolve proves ref currency, but not that the manifest -/// body still exists, so these modes trade that additional check for a bounded performance optimization. -struct PartFolderValidate -{ - enum class Mode : uint8_t { Always, Age, Never }; - Mode mode = Mode::Always; - uint64_t age_seconds = 0; /// only meaningful for Mode::Age -}; - class CachedPartFolderAccess; /// A part write that has been staged and PRECOMMITTED but not yet promoted -- the durable-but- @@ -237,8 +217,6 @@ class CachedPartFolderAccess /// path takes a per-disk global mutex and allocates on EVERY read. Off by default so the read /// hit path never pays for it; the disk factory / tests turn it on when they consult `explain`. bool explain_enabled = false; - /// The `ForceFresh` manifest-body re-proof policy. `Always` is the fail-closed default. - PartFolderValidate validate; }; /// `CacheParams params_ = {}` cannot be a default argument here — Clang's complete-class- @@ -247,16 +225,12 @@ class CachedPartFolderAccess /// argument written inside the class body is evaluated too early. Two overloads sidestep it; the /// single-arg form default-constructs `CacheParams` (retention disabled) out-of-line. explicit CachedPartFolderAccess(Cas::PoolPtr store_); - /// `now_ms_fn_`: wall-clock ms, injected (tests) for the age-window comparison AND the - /// retained view's `validated_at_ms` stamp -- the SAME function drives both, so a test controls - /// each side of the comparison exactly. Defaults to `std::chrono::system_clock` (mirrors - /// `Cas::Gc`'s `now_ms_fn` convention) when empty. - CachedPartFolderAccess(Cas::PoolPtr store_, CacheParams params_, std::function now_ms_fn_ = {}); + CachedPartFolderAccess(Cas::PoolPtr store_, CacheParams params_); /// Resolves the ref and, when present, reads and validates its manifest into an immutable view. - /// `nullptr` means the ref is absent. Strict validation and the default `ForceFresh` policy reach - /// `readManifestShared`'s mandatory HEAD because a fresh ref resolve alone does not prove that the - /// manifest body still exists. + /// `nullptr` means the ref is absent. `ForceFresh` and `StrictValidate` bypass the retained view + /// and read through the pool's manifest cache; a manifest is immutable per id, so a retained view + /// can be stale only by naming a different manifest id, which the fresh resolve detects. std::shared_ptr getView(const PartRefKey & key, Freshness freshness) const; /// Ref-only resolution (per-part reads, part-dir existence, publish stamps): no @@ -364,9 +338,6 @@ class CachedPartFolderAccess private: Cas::PoolPtr store; CacheParams params; - /// Wall-clock milliseconds; see the constructor comment. `std::function::operator` is const, so this is - /// callable from const methods (`getView`, `buildView`) without a `mutable` qualifier. - std::function now_ms_fn; /// Supplies the conservative encoded-manifest weight used by `CacheBase` for eviction decisions. struct ViewWeight @@ -384,7 +355,7 @@ class CachedPartFolderAccess mutable std::unordered_map>> inflight; /// Reads a manifest and constructs a view. Cold `CachedForLoad` builds are single-flight per key; - /// fresh modes perform their own read so each call retains its validation guarantee. + /// fresh modes perform their own read. std::shared_ptr buildView( const PartRefKey & key, const Cas::Resolved & resolved, Freshness freshness) const; /// Removes a retained view and records the invalidation for diagnostics. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.cpp index d1c44d2b168f..a6ded95e363c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.cpp @@ -1,5 +1,4 @@ #include -#include #include @@ -13,34 +12,34 @@ namespace ProfileEvents namespace DB::Cas { -std::optional loadMeta(Backend & backend, const Layout & layout, const BlobRef & ref) +std::optional loadMeta(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Retry & policy) { - const String key = layout.blobMetaKey(ref); - auto got = backend.get(key); + auto got = op.read(layout.blobMetaKey(ref), policy); if (!got) return std::nullopt; - return LoadedMeta{.meta = decodeBlobMeta(got->bytes), .etag = got->token}; + return LoadedMeta{.meta = decodeBlobMeta(got->bytes), .etag = std::move(got->etag)}; } -CasOverwriteResult putMetaIfAbsent(Pool & pool, const BlobRef & ref, const BlobMeta & meta) +WriteResult putMetaIfAbsent(CasOperation & op, const Layout & layout, const BlobRef & ref, + const BlobMeta & meta, const Retry & policy) { ProfileEvents::increment(ProfileEvents::CASMetaPut); - const String key = pool.layout().blobMetaKey(ref); - return pool.stagingPutIfAbsentMutable(key, encodeBlobMeta(meta)); + return op.create(layout.blobMetaKey(ref), encodeBlobMeta(meta), policy); } -CasOverwriteResult casMeta(Pool & pool, const BlobRef & ref, const Token & expected, const BlobMeta & meta) +WriteResult casMeta(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Etag & expected, const BlobMeta & meta) { ProfileEvents::increment(ProfileEvents::CASMetaCompareSwap); - const String key = pool.layout().blobMetaKey(ref); - return pool.stagingConditionalOverwrite(key, encodeBlobMeta(meta), expected); + return op.replace(layout.blobMetaKey(ref), encodeBlobMeta(meta), expected, Retry::standard()); } -DeleteOutcome deleteMetaExact(Backend & backend, const Layout & layout, const BlobRef & ref, const Token & expected) +Removal deleteMetaExact(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Etag & expected) { ProfileEvents::increment(ProfileEvents::CASMetaDelete); - const String key = layout.blobMetaKey(ref); - return backend.deleteExact(key, expected); + return op.remove(layout.blobMetaKey(ref), expected, Retry::standard()); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.h index c6faa5021519..df84aee4d9f2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasBlobMeta.h @@ -1,7 +1,6 @@ #pragma once -#include -#include +#include #include #include #include @@ -11,55 +10,53 @@ namespace DB::Cas { -class Pool; - -/// A decoded blob meta record together with the backend token observed for the same incarnation. -/// The token is returned with the decoded record because the next conditional update or exact delete -/// must be guarded by the version that was actually read; comparing encoded meta bytes would not -/// provide that protection. +/// A decoded blob meta record together with the incarnation the same read observed. The incarnation +/// travels with the record because the next conditional update or exact delete must be guarded by the +/// version that was actually read; comparing encoded meta bytes would not provide that protection. struct LoadedMeta { BlobMeta meta; - Token etag; + Etag etag; }; /// Shared lifecycle operations for the blob freshness marker used by the writer and GC. The key is /// built from the complete `BlobRef`, so each algorithm uses its own digest representation and no /// pool-wide digest width is threaded through these functions. The marker is a point-read hint rather -/// than the blob lifetime's linearization point: the blob body's incarnation tag and exact-token body -/// deletion provide the safety guarantee, while a stale marker can at most make a writer re-upload. +/// than the blob lifetime's linearization point: the blob body's incarnation tag and exact-incarnation +/// body deletion provide the safety guarantee, while a stale marker can at most make a writer +/// re-upload. /// /// `loadMeta` is used in the adopt path, so its backend must provide strong read-after-write -/// consistency: after a successful meta write, the one subsequent GET must observe that write. -/// Conditional updates and deletion use the backend token, not the encoded meta bytes. +/// consistency: after a successful meta write, the one subsequent read must observe that write. +/// Conditional updates and deletion use the observed incarnation, not the encoded meta bytes. /// -/// Returns the current decoded marker and its conditional token, or nullopt when the meta key is -/// absent. Decoding errors propagate as exceptions. -std::optional loadMeta(Backend & backend, const Layout & layout, const BlobRef & ref); - -/// Creates the marker only when its key is absent, controlled: a SlowDown/429/5xx on the attempt is -/// resolved-and-reissued within budget rather than escaping as a raw client error (triage: S22 RCA). -/// A precondition failure (another -/// writer already created the marker -- possibly with a DIFFERENT record, e.g. a stale `Condemned` -/// marker still present when a vanished body is freshly re-uploaded) is reported as -/// `CasOverwriteOutcome::Conflict`, never thrown -- this uses `putIfAbsentControlledMutable`, NOT the -/// ref-log lane's `putIfAbsentControlled` (that method's resolve throws `CORRUPTED_DATA` on any -/// different bytes at the key, which is correct for the ref-log's immutable content-addressed keys -/// but wrong for this mutable marker, where a pre-existing different value is an expected, non-corrupt -/// outcome). -CasOverwriteResult putMetaIfAbsent(Pool & pool, const BlobRef & ref, const BlobMeta & meta); - -/// Replaces the marker only when its current backend token equals `expected`, controlled (same -/// budgeted resolve-and-reissue as putMetaIfAbsent). A genuine conflict (current token AND bytes both -/// differ from what this call intended) is reported as `CasOverwriteOutcome::Conflict`, never thrown -- -/// exactly like the previous uncontrolled `CasResult` contract -- so the caller's existing -/// reload-and-retry metadata reconciliation in `PartWriteTxn::ensureBlobPresent` keeps working unchanged. -CasOverwriteResult casMeta(Pool & pool, const BlobRef & ref, const Token & expected, const BlobMeta & meta); - -/// Deletes only the marker incarnation identified by `expected`. A token mismatch leaves the current -/// marker untouched; `NotFound` is distinct from that case so callers can tell absence from a raced -/// replacement. The backend's complete `DeleteOutcome` is returned, including any storage-specific -/// delete-marker status. -DeleteOutcome deleteMetaExact(Backend & backend, const Layout & layout, const BlobRef & ref, const Token & expected); +/// Returns the current decoded marker and the incarnation to guard the next write with, or nullopt +/// when the meta key is absent. Decoding errors propagate as exceptions. +/// +/// `policy` lets a hand-written loop pass the bound it froze at entry, so this read ends with the rest +/// of the loop instead of starting a fresh window. Same for `putMetaIfAbsent` below. +std::optional loadMeta(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Retry & policy = Retry::standard()); + +/// Creates the marker only when its key is absent, on the plane `op` belongs to -- like its siblings, +/// so one caller's decision cannot end up split across two fences. Anything at the key that this call +/// did not itself write -- a stale `Condemned` marker still present when a vanished body is freshly +/// re-uploaded, or a racing writer's byte-identical marker -- comes back as `Conflict` carrying what +/// was observed, never as a throw: this marker is mutable, so a pre-existing different value is an +/// expected outcome rather than corruption. +WriteResult putMetaIfAbsent(CasOperation & op, const Layout & layout, const BlobRef & ref, + const BlobMeta & meta, const Retry & policy = Retry::standard()); + +/// Replaces the marker only when its current incarnation is `expected`, on the plane `op` belongs to. +/// A competing write is reported as `Conflict` carrying what the resolve read observed, never thrown, +/// so the caller's own reload-and-retry reconciliation decides what to do about it. +WriteResult casMeta(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Etag & expected, const BlobMeta & meta); + +/// Deletes only the marker incarnation named by `expected`. `Mismatch` leaves the current marker +/// untouched and is distinct from `Gone`, so callers can tell absence from a raced replacement. A +/// versioned bucket that archives instead of reclaiming raises `CAS_DELETE_MARKER`. +Removal deleteMetaExact(CasOperation & op, const Layout & layout, const BlobRef & ref, + const Etag & expected); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.cpp index c836f9932d79..e61145e03ce8 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.cpp @@ -6,8 +6,7 @@ namespace DB::Cas bool DetachedStopToken::stopping() const { - std::lock_guard lock(state->mutex); - return state->stopping; + return state->stopping.load(std::memory_order_acquire); } struct DetachedTaskLease::Completion diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.h index 5e317e429071..1190dc1df939 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasDetachedWork.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include @@ -19,7 +20,11 @@ struct DetachedRegistryState std::mutex mutex; std::condition_variable cv; uint64_t in_flight = 0; - bool stopping = false; + /// Written under `mutex`, so the dispatch-side check-and-count and the drain's `in_flight == 0` + /// wait stay serialized against the stop; read WITHOUT it by the open request plane's fence before + /// every attempt and every sleep of every request -- a pool-wide mutex on that path is not + /// acceptable, and the readers need only the flag's current truth. + std::atomic stopping{false}; }; /// Read-only view of the registry, handed to every task. The ONLY way a task asks whether teardown has diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.cpp new file mode 100644 index 000000000000..5844902dd239 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.cpp @@ -0,0 +1,32 @@ +#include + +#include + +namespace DB::Cas +{ + +void hintRefLogsWithinEpoch(KeyReader & reader, const Layout & layout, const NamespaceLifeId & life, + RefTxnId first, const RefTxnId & committed_through) +{ + while (reader.pending() < reader.window() && first <= committed_through) + { + reader.hint(layout.refLogKey(life, first)); + if (first.ref_sequence == std::numeric_limits::max()) + return; + ++first.ref_sequence; + } +} + +void discardRefLogHintsOfEpoch(KeyReader & reader, const Layout & layout, const NamespaceLifeId & life, + RefTxnId first, const RefTxnId & committed_through) +{ + for (size_t n = 0; n < reader.window() && first <= committed_through; ++n) + { + reader.discard(layout.refLogKey(life, first)); + if (first.ref_sequence == std::numeric_limits::max()) + return; + ++first.ref_sequence; + } +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.h new file mode 100644 index 000000000000..a4bb79c91f28 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasKeyReader.h @@ -0,0 +1,55 @@ +#pragma once +#include +#include +#include +#include + +namespace DB::Cas +{ + +/// What a sequential walk needs from whoever fetches its objects. `take` returns the object (or +/// nullopt when absent) and is the only call that decides anything; `hint` may start fetching a key +/// the walk will take later; `discard` drops a hint the walk will never take. A walk that hints +/// nothing and takes everything in order behaves exactly like one that reads inline: the reader is a +/// cache of results, never of decisions. +class KeyReader +{ +public: + virtual ~KeyReader() = default; + virtual void hint(const String & key) = 0; + virtual std::optional take(const String & key) = 0; + virtual void discard(const String & key) = 0; + /// Hinted and not yet taken. + virtual size_t pending() const = 0; + /// How many hints a walk keeps outstanding; 0 means "do not hint". + virtual size_t window() const = 0; +}; + +/// The sequential reader: every take is one inline read on the caller's operation. +class InlineKeyReader final : public KeyReader +{ +public: + explicit InlineKeyReader(CasOperation & op_) : op(op_) {} + void hint(const String &) override {} + std::optional take(const String & key) override { return op.read(key, Retry::standard()); } + void discard(const String &) override {} + size_t pending() const override { return 0; } + size_t window() const override { return 0; } + +private: + CasOperation & op; +}; + +/// Hints the ref-log ids of `first`'s epoch, from `first` upward, while the reader has window and the +/// id is within the committed frontier. Only this epoch: past its seal the ids do not exist, and a +/// walk learns where the seal is only by decoding it. +void hintRefLogsWithinEpoch(KeyReader & reader, const Layout & layout, const NamespaceLifeId & life, + RefTxnId first, const RefTxnId & committed_through); + +/// The other half of the rule above, called when a walk crosses an epoch: every hint of the old epoch +/// from `first` up to one window is dropped, so the window is free for the new epoch. Discarding an +/// unhinted key is a no-op, so over-asking by a window is harmless. +void discardRefLogHintsOfEpoch(KeyReader & reader, const Layout & layout, const NamespaceLifeId & life, + RefTxnId first, const RefTxnId & committed_through); + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.cpp index 778814997ca1..dc43ce62f347 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.cpp @@ -30,9 +30,9 @@ namespace DB::Cas { CasManifestReader::CasManifestReader( - Backend & backend_, const Layout & layout_, const PoolMeta & meta_, + CasRequests & requests_, const Layout & layout_, const PoolMeta & meta_, const CasEventSink & event_sink_, size_t manifest_decode_cache_bytes) - : backend(backend_), layout(layout_), meta(meta_), event_sink(event_sink_) + : requests(requests_), layout(layout_), meta(meta_), event_sink(event_sink_) { if (manifest_decode_cache_bytes > 0) manifest_cache = std::make_unique( @@ -40,30 +40,21 @@ CasManifestReader::CasManifestReader( manifest_decode_cache_bytes, /*max_count=*/16384, ManifestDecodeCache::DEFAULT_SIZE_RATIO); } -size_t CasManifestReader::ManifestCacheKeyHash::operator()(const ManifestCacheKey & k) const -{ - /// Combine the manifest-id hash with the token's bytes + type. The token is part of the key so a - /// re-incarnation under the same id misses (the immutable bytes changed identity). - const size_t h1 = std::hash{}(k.manifest_id); - const size_t h2 = std::hash{}(k.token.value); - const size_t h3 = std::hash{}(static_cast(k.token.type)); - size_t h = h1; - h ^= h2 + 0x9e3779b97f4a7c15ULL + (h << 6) + (h >> 2); - h ^= h3 + 0x9e3779b97f4a7c15ULL + (h << 6) + (h >> 2); - return h; -} - std::shared_ptr CasManifestReader::readManifestShared(const ManifestId & id) { + /// One id names one content forever (minted once, written once, only ever deleted), so a cached + /// decode is served without any request. + if (manifest_cache) + if (auto cached = manifest_cache->get(id)) + return cached; + /// A live reference naming a missing manifest body is a dangling-reference violation /// (`INV-NO-DANGLE`). Never substitute an empty manifest: callers must observe the missing object - /// as an exception. + /// as an exception. The `GET` alone carries the absence signal, so no `HEAD` precedes it. const String key = layout.manifestKey(id); - - /// `HEAD` is mandatory even on a cache hit. It proves that the live reference still names an - /// existing object and supplies the token that identifies the immutable bytes being reused. - const HeadResult head = backend.head(key); - if (!head.exists) + CasOperation op = requests.admit(); + std::optional object = op.read(key, Retry::standard()); + if (!object) { if (event_sink) { @@ -79,15 +70,6 @@ std::shared_ptr CasManifestReader::readManifestShared(const throw Exception(ErrorCodes::FILE_DOESNT_EXIST, "live ref names manifest at {} but its object is missing — INV-NO-DANGLE", key); } - - if (manifest_cache) - if (auto cached = manifest_cache->get(ManifestCacheKey{.manifest_id = id, .token = head.token})) - return cached; - - std::optional object = backend.get(key); - if (!object) - throw Exception(ErrorCodes::FILE_DOESNT_EXIST, - "manifest at {} vanished between head and get — INV-NO-DANGLE", key); ProfileEvents::increment(ProfileEvents::CASPartFolderManifestGets); PartManifest body = decodePartManifest(openObject(FormatId::PartManifest, object->bytes)); @@ -132,7 +114,7 @@ std::shared_ptr CasManifestReader::readManifestShared(const auto decoded = std::make_shared(std::move(body)); if (manifest_cache) - manifest_cache->set(ManifestCacheKey{.manifest_id = id, .token = head.token}, decoded); + manifest_cache->set(id, decoded); return decoded; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.h index af5d31776857..67a124a16889 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasManifestReader.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -21,29 +21,30 @@ struct BlobLocation uint64_t length = 0; }; -/// Reads and validates part manifests, caches immutable decodes, and translates blob entries into -/// ranged object reads. A read first obtains the object's current backend token, then reuses a -/// decode only for the matching `(ManifestId, Token)` pair; a cache miss performs a `GET` and -/// validates both the manifest reference and owning namespace before publication into the cache. -/// Missing or changing objects and failed identity checks are surfaced as exceptions, never as an -/// empty or partially trusted manifest. +/// Reads and validates part manifests, caches immutable decodes by `ManifestId`, and translates blob +/// entries into ranged object reads. A manifest id is minted once and its body is written once, so +/// one id names one content forever: a cache hit is served without any request, and a miss performs +/// one `GET` and validates both the manifest reference and the owning namespace before publication +/// into the cache. A missing body, a decode failure or a failed identity check is surfaced as an +/// exception, never as an empty or partially trusted manifest. /// -/// The reader receives its backend, immutable layout and pool metadata, and event sink by reference; -/// it has no `Pool` back-reference and owns no `Pool`-level mutex. The decode cache is a -/// byte-weighted `CacheBase` LRU whose synchronization is internal to `CacheBase`; a null cache -/// means caching is disabled (`manifest_decode_cache_bytes == 0`). +/// The reader receives its `CasRequests`, immutable layout and pool metadata, and event sink by +/// reference; it has no `Pool` back-reference and owns no `Pool`-level mutex. Each cache-miss read +/// admits its own operation, so a lost mount lease refuses a miss the same way any other read does. +/// The decode cache is a byte-weighted `CacheBase` LRU whose synchronization is internal to +/// `CacheBase`; a null cache means caching is disabled (`manifest_decode_cache_bytes == 0`). class CasManifestReader { public: /// Binds the reader to the pool environment. A positive cache budget creates the byte-weighted - /// LRU; zero disables caching while leaving the mandatory `HEAD` and validation sequence intact. + /// LRU; zero disables caching while leaving the one-`GET`-and-validate sequence intact. CasManifestReader( - Backend & backend_, const Layout & layout_, const PoolMeta & meta_, + CasRequests & requests_, const Layout & layout_, const PoolMeta & meta_, const CasEventSink & event_sink_, size_t manifest_decode_cache_bytes); /// Reads a manifest by value using the fail-closed sequence described above. A missing body, - /// disappearance between `HEAD` and `GET`, decode failure, or either identity mismatch throws; - /// only a fully validated decode can enter the cache. + /// decode failure, or either identity mismatch throws; only a fully validated decode can enter + /// the cache. PartManifest readManifest(const ManifestId & id); /// Reads a manifest like `readManifest` but returns the immutable shared decode. This preserves @@ -59,25 +60,9 @@ class CasManifestReader size_t manifestDecodeCacheBytes() const { return manifest_cache ? manifest_cache->sizeInBytes() : 0; } private: - /// The cache must include the backend token: a reused manifest identifier can refer to a new - /// object incarnation, and its immutable decoded bytes must not be reused across incarnations. - struct ManifestCacheKey - { - ManifestId manifest_id; - Token token; - bool operator==(const ManifestCacheKey &) const = default; - }; - - /// Hashes both identity components and the token type so cache lookup uses the same complete - /// identity as `ManifestCacheKey::operator==`. - struct ManifestCacheKeyHash - { - size_t operator()(const ManifestCacheKey & k) const; - }; - /// Estimates retained decode memory from fixed object overhead plus entry path and inline-byte - /// storage. Weighting by bytes gives a server reading many parts an honest memory ceiling instead - /// of a count-only bound; the cache key still provides the fail-closed token semantics. + /// storage. Weighting by bytes gives a server reading many parts an honest memory ceiling + /// instead of a count-only bound. struct PartManifestWeight { /// Returns the approximate bytes retained for one decoded manifest by the cache. @@ -89,9 +74,9 @@ class CasManifestReader return bytes; } }; - using ManifestDecodeCache = CacheBase; + using ManifestDecodeCache = CacheBase, PartManifestWeight>; - Backend & backend; + CasRequests & requests; const Layout & layout; const PoolMeta & meta; const CasEventSink & event_sink; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 1b6a57d331dd..e3f52fd7957a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -34,7 +34,6 @@ namespace ProfileEvents namespace DB::Cas { -void reportMountRenewProgress(const CasOverwriteProgress & progress) noexcept; void reportMountRenewCompletion(const MountRenewResult & result) noexcept; void configureMountRenewObservability( const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; @@ -53,6 +52,8 @@ int64_t wallClockNowSeconds() CasMountRuntime::CasMountRuntime( BackendPtr backend_ptr_, + CasRequests & mount_requests_, + CasRequests & farewell_requests_, const Layout & layout_, MountConfig config_, String server_root_id_, @@ -60,6 +61,8 @@ CasMountRuntime::CasMountRuntime( CasRequestBudget cas_request_budget_, std::function remount_attempt_) : backend_ptr(std::move(backend_ptr_)) + , mount_requests(mount_requests_) + , farewell_requests(farewell_requests_) , layout(layout_) , config(std::move(config_)) , server_root_id(std::move(server_root_id_)) @@ -132,18 +135,31 @@ void CasMountRuntime::checkFenceOrThrow(uint64_t admitted_generation) const "system.cas_mounts for the disk's lifecycle before retrying"); } -bool CasMountRuntime::refAppendFenceOk() const +Fence::Admit CasMountRuntime::admit(uint64_t admitted_generation, uint64_t needed_ms) const { - /// `mayMutate` checks the latch and deadline. The additional budget check prevents starting a - /// controlled request that cannot plausibly finish, including its safety margin, before expiry. - if (mount_fence.lost.load(std::memory_order_acquire)) - return false; + if (mount_fence.lost.load(std::memory_order_acquire) || fenceGeneration() != admitted_generation) + return Fence::Admit::LostOrRearmed; const uint64_t now = bootMsNow(); const uint64_t deadline = mount_fence.deadline_boot_ms.load(std::memory_order_acquire); if (now >= deadline) - return false; - const uint64_t margin = cas_request_budget.attempt_timeout_ms + cas_request_budget.lease_safety_margin_ms; - return margin < deadline - now; + return Fence::Admit::NoBudget; + /// Compared by subtraction rather than as the sum `needed_ms + margin`, which can wrap for an + /// absurd configuration and then read as if there were room. + const uint64_t remaining = deadline - now; + if (needed_ms >= remaining || cas_request_budget.lease_safety_margin_ms >= remaining - needed_ms) + return Fence::Admit::NoBudget; + return Fence::Admit::Ok; +} + +bool CasMountRuntime::refAppendFenceOk() const +{ + /// Two envelopes' worth of room under the live generation -- a write and its settlement read, which + /// is what `writeLoop` reserves -- so a ref-log attempt is not started when it cannot plausibly + /// finish, safety margin included, before the lease expires. + const uint64_t envelope_ms = cas_request_budget.attemptEnvelopeMs(); + const uint64_t needed_ms = envelope_ms > std::numeric_limits::max() / 2 + ? std::numeric_limits::max() : 2 * envelope_ms; + return admit(fenceGeneration(), needed_ms) == Fence::Admit::Ok; } void CasMountRuntime::setMountDeadline(uint64_t deadline_boot_ms) @@ -181,8 +197,8 @@ uint64_t CasMountRuntime::peekNextBuildSeq() void CasMountRuntime::renewWatermarkOnce() { - auto call = admitKeeperCall(RenewalDriverState::Dormant, RenewalDriverState::DirectCall); - (void)renewKeeperOnce( + auto call = admitRenewerCall(RenewalDriverState::Dormant, RenewalDriverState::DirectCall); + (void)renewRenewerOnce( std::move(call), RenewalDriverState::DirectCall, /*propagate_failure=*/true, @@ -259,8 +275,8 @@ CasMountRuntime::DriverLease::DriverLease(CasMountRuntime & runtime_, RenewalDri bool CasMountRuntime::renewalWorkerMayRenew() const { return renewal_driver_state == RenewalDriverState::WorkerIdle - && mount_keeper - && mount_keeper->state() == MountLeaseKeeperState::Active; + && mount_renewer + && mount_renewer->state() == MountLeaseRenewerState::Active; } CasMountRuntime::DriverLease::~DriverLease() @@ -321,7 +337,7 @@ RenewalDriverState CasMountRuntime::DriverLease::finish( return runtime.renewal_driver_state; } -CasMountRuntime::AdmittedKeeperCall CasMountRuntime::admitKeeperCall( +CasMountRuntime::AdmittedRenewerCall CasMountRuntime::admitRenewerCall( RenewalDriverState required, RenewalDriverState active) { @@ -334,24 +350,24 @@ CasMountRuntime::AdmittedKeeperCall CasMountRuntime::admitKeeperCall( throw Exception( ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal driver is not admitted from the required state"); - if (!mount_keeper) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a keeper"); - if (mount_keeper->state() != MountLeaseKeeperState::Active) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active keeper"); - MountLeaseKeeper * keeper = mount_keeper.get(); + if (!mount_renewer) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a renewer"); + if (mount_renewer->state() != MountLeaseRenewerState::Active) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active renewer"); + MountLeaseRenewer * renewer = mount_renewer.get(); auto lease = std::make_unique(*this, active); renewal_driver_state = active; driver_cv.notify_all(); - return AdmittedKeeperCall{std::move(lease), keeper}; + return AdmittedRenewerCall{std::move(lease), renewer}; } -void CasMountRuntime::installKeeper( +void CasMountRuntime::installRenewer( UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms) { - auto replacement = std::make_unique( - backend_ptr, layout, server_root_id, our_uuid, writer_epoch, + auto replacement = std::make_unique( + mount_requests, farewell_requests, layout, server_root_id, our_uuid, writer_epoch, config.mount_lease_ttl_ms, now_ms, [this] { return minActive(); }, [this](CasEvent e) { emitEvent(std::move(e)); }, @@ -363,21 +379,21 @@ void CasMountRuntime::installKeeper( && renewal_driver_state != RenewalDriverState::Parked) throw Exception( ErrorCodes::LOGICAL_ERROR, - "CAS mount runtime: keeper replacement requires Dormant or Parked renewal ownership"); - mount_keeper = std::move(replacement); + "CAS mount runtime: renewer replacement requires Dormant or Parked renewal ownership"); + mount_renewer = std::move(replacement); } -uint64_t CasMountRuntime::startKeeper() +uint64_t CasMountRuntime::startRenewer() { RenewalDriverState active; RenewalDriverState destination; - MountLeaseKeeper * keeper; + MountLeaseRenewer * renewer; { std::lock_guard lock(driver_mutex); - if (!mount_keeper) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startKeeper without a keeper"); - if (mount_keeper->state() != MountLeaseKeeperState::New) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startKeeper requires a New keeper"); + if (!mount_renewer) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer without a renewer"); + if (mount_renewer->state() != MountLeaseRenewerState::New) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer requires a New renewer"); if (renewal_driver_state == RenewalDriverState::Dormant) { active = RenewalDriverState::StartupCall; @@ -390,15 +406,15 @@ uint64_t CasMountRuntime::startKeeper() } else { - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startKeeper is not admitted in the current state"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer is not admitted in the current state"); } - keeper = mount_keeper.get(); + renewer = mount_renewer.get(); renewal_driver_state = active; driver_cv.notify_all(); } DriverLease lease(*this, active); - const uint64_t anchor = keeper->start(); + const uint64_t anchor = renewer->start([this] { return !renewalCancelled(); }); (void)lease.finish(destination); return anchor; } @@ -407,70 +423,36 @@ MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(bool worker_c { return MountRenewOperationEnvironment{ .boot_ms = [this] { return bootMsNow(); }, - .stop_cause = [this, worker_call] + .live = [this, worker_call] { - return config.renewal_stop_cause_for_test - ? config.renewal_stop_cause_for_test() - : renewalStopCause(worker_call); - }, - .wait_before_retry = [this, worker_call](uint64_t wait_ms) { return waitForRetry(wait_ms, worker_call); }, - .observe = [](const CasOverwriteProgress & progress) - { - switch (progress.kind) - { - case CasOverwriteProgressKind::PutStarted: - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts); - break; - case CasOverwriteProgressKind::RetryStarted: - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries); - break; - case CasOverwriteProgressKind::ResolvedByGet: - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalResolved); - break; - case CasOverwriteProgressKind::BecameAmbiguous: - case CasOverwriteProgressKind::ResolveStarted: - break; - } - reportMountRenewProgress(progress); + return config.renewal_live_for_test ? config.renewal_live_for_test() : renewalLive(worker_call); }, + .cancelled = [this] { return renewalCancelled(); }, }; } -CasOverwriteStopCause CasMountRuntime::renewalStopCause(bool worker_call) const +bool CasMountRuntime::renewalLive(bool worker_call) const { std::lock_guard lock(driver_mutex); if (workers_stop_requested) - return CasOverwriteStopCause::Cancelled; - if (worker_call + return false; + return !(worker_call && (renewal_driver_state == RenewalDriverState::ParkRequested || renewal_driver_state == RenewalDriverState::Parked || lifecycle() != PoolLifecycle::Live - || mount_fence.lost.load(std::memory_order_acquire))) - return CasOverwriteStopCause::FenceOrLifecycleLost; - return CasOverwriteStopCause::Continue; + || mount_fence.lost.load(std::memory_order_acquire))); +} + +bool CasMountRuntime::renewalCancelled() const +{ + std::lock_guard lock(driver_mutex); + return workers_stop_requested; } -bool CasMountRuntime::waitForRetry(uint64_t wait_ms, bool worker_call) +void CasMountRuntime::sleepInterruptibly(uint64_t ms) { std::unique_lock lock(driver_mutex); - driver_cv.wait_for(lock, std::chrono::milliseconds(wait_ms), [this, worker_call] - { - return workers_stop_requested - || (worker_call - && (renewal_driver_state == RenewalDriverState::ParkRequested - || renewal_driver_state == RenewalDriverState::Parked - || lifecycle() != PoolLifecycle::Live - || mount_fence.lost.load(std::memory_order_acquire))); - }); - if (workers_stop_requested) - return false; - if (worker_call - && (renewal_driver_state == RenewalDriverState::ParkRequested - || renewal_driver_state == RenewalDriverState::Parked - || lifecycle() != PoolLifecycle::Live - || mount_fence.lost.load(std::memory_order_acquire))) - return false; - return true; + driver_cv.wait_for(lock, std::chrono::milliseconds(ms), [this] { return workers_stop_requested; }); } void CasMountRuntime::consumeRenewResult( @@ -480,15 +462,21 @@ void CasMountRuntime::consumeRenewResult( bool propagate_failure) { /// Driver ownership has already been restored by `DriverLease::finish`; this is the single logical - /// consumption boundary and it runs without `driver_mutex` or keeper access. - if (result.outcome == MountRenewOutcome::Committed - && (result.diagnostics.attempts_sent > 1 || result.diagnostics.resolved_by_get)) + /// consumption boundary and it runs without `driver_mutex` or renewer access. + /// The physical counters come off the result rather than off a per-attempt callback, so they count + /// the same on every ending: a renewal that gave up still sent what it sent. + if (result.attempts_sent > 0) + { + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts, result.attempts_sent); + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries, result.attempts_sent - 1); + } + if (result.resolved_by_read) + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalResolved); + + if (result.outcome == MountRenewOutcome::Committed && (result.attempts_sent > 1 || result.resolved_by_read)) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRecovered); if (result.outcome == MountRenewOutcome::Terminal - && result.diagnostics.deadline_source == CasOverwriteDeadlineSource::ExternalLeaseSafety - && result.diagnostics.stop_cause == CasOverwriteStopCause::Continue - && (result.diagnostics.unresolved_reason == CasUnresolvedReason::NoAttemptSent - || result.diagnostics.unresolved_reason == CasUnresolvedReason::DeadlineMidWay)) + && result.deadline_source == GaveUp::Source::Lease) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalDeadlineExceeded); if (result.outcome == MountRenewOutcome::Committed) @@ -531,8 +519,8 @@ void CasMountRuntime::consumeRenewResult( std::rethrow_exception(result.failure); } -uint64_t CasMountRuntime::renewKeeperOnce( - AdmittedKeeperCall call, +uint64_t CasMountRuntime::renewRenewerOnce( + AdmittedRenewerCall call, RenewalDriverState active, bool propagate_failure, bool worker_call) @@ -541,7 +529,13 @@ uint64_t CasMountRuntime::renewKeeperOnce( /// whole-chain finalizer to deliver after `remount_mutex` is released. configureMountRenewObservability( &server_root_id, &event_sink, active == RenewalDriverState::RemountCall); - const MountRenewResult result = call.keeper->renew(cas_request_budget, renewalEnvironment(worker_call)); + /// The remount redo re-anchors the lease BEFORE `armMountFence`, with the fence still latched lost, + /// so it renews on the renewer's open plane: admitted under the mount fence it could only ever give + /// up, and every remount would fail at this step. `RemountCall` is reached from + /// `renewRenewerForRemountOnce` alone. + const MountRenewResult result = active == RenewalDriverState::RemountCall + ? call.renewer->renewForRemount(renewalEnvironment(worker_call)) + : call.renewer->renew(renewalEnvironment(worker_call)); const RenewalDriverState destination = active == RenewalDriverState::WorkerCall ? RenewalDriverState::WorkerIdle : (active == RenewalDriverState::RemountCall ? RenewalDriverState::Parked : RenewalDriverState::Dormant); @@ -552,33 +546,33 @@ uint64_t CasMountRuntime::renewKeeperOnce( return result.attempt_start_boot_ms; } -uint64_t CasMountRuntime::renewKeeperForStartupOnce() +uint64_t CasMountRuntime::renewRenewerForStartupOnce() { - auto call = admitKeeperCall(RenewalDriverState::Dormant, RenewalDriverState::StartupCall); - return renewKeeperOnce( + auto call = admitRenewerCall(RenewalDriverState::Dormant, RenewalDriverState::StartupCall); + return renewRenewerOnce( std::move(call), RenewalDriverState::StartupCall, /*propagate_failure=*/true, /*worker_call=*/false); } -uint64_t CasMountRuntime::renewKeeperForRemountOnce() +uint64_t CasMountRuntime::renewRenewerForRemountOnce() { - auto call = admitKeeperCall(RenewalDriverState::Parked, RenewalDriverState::RemountCall); - return renewKeeperOnce( + auto call = admitRenewerCall(RenewalDriverState::Parked, RenewalDriverState::RemountCall); + return renewRenewerOnce( std::move(call), RenewalDriverState::RemountCall, /*propagate_failure=*/true, /*worker_call=*/false); } -void CasMountRuntime::keeperReset() +void CasMountRuntime::renewerReset() { std::lock_guard lock(driver_mutex); if (renewal_driver_state != RenewalDriverState::Dormant && renewal_driver_state != RenewalDriverState::Parked) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: keeper reset while renewal is active"); - mount_keeper.reset(); + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewer reset while renewal is active"); + mount_renewer.reset(); } ThreadFromGlobalPool CasMountRuntime::makeWorker(std::function body) @@ -596,8 +590,8 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) || workers_starting || workers_started || renewal_worker.joinable() || remount_worker.joinable()) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: background workers cannot start in the current state"); - if (!mount_keeper || mount_keeper->state() != MountLeaseKeeperState::Active) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: background workers require an Active keeper"); + if (!mount_renewer || mount_renewer->state() != MountLeaseRenewerState::Active) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: background workers require an Active renewer"); workers_starting = true; workers_stop_requested = false; worker_loops_released = false; @@ -648,7 +642,7 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) void CasMountRuntime::renewalLoop() { - setThreadName(ThreadName::CAS_LEASE_KEEPER); + setThreadName(ThreadName::CAS_LEASE_RENEWER); { std::unique_lock lock(driver_mutex); driver_cv.wait(lock, [this] { return worker_loops_released; }); @@ -661,7 +655,7 @@ void CasMountRuntime::renewalLoop() if (config.renewal_before_driver_lock_hook_for_test) config.renewal_before_driver_lock_hook_for_test(); - AdmittedKeeperCall call; + AdmittedRenewerCall call; { std::unique_lock lock(driver_mutex); if (workers_stop_requested || remountTerminal()) @@ -688,7 +682,7 @@ void CasMountRuntime::renewalLoop() continue; } - const uint64_t last_anchor = mount_keeper->lastCommittedAttemptStartBootMs(); + const uint64_t last_anchor = mount_renewer->lastCommittedAttemptStartBootMs(); const uint64_t period_ms = static_cast(std::max(0, renewal_period.count())); const uint64_t due = last_anchor > std::numeric_limits::max() - period_ms ? std::numeric_limits::max() @@ -698,23 +692,23 @@ void CasMountRuntime::renewalLoop() { /// Every runtime notification can change the cadence decision: park/resume may happen /// entirely while this worker is idle, and a remount may publish an already-overdue - /// keeper anchor. Re-sample state and BOOTTIME after any wake instead of retaining the + /// renewer anchor. Re-sample state and BOOTTIME after any wake instead of retaining the /// old relative wait until its wall-clock timeout. driver_cv.wait_for(lock, std::chrono::milliseconds(due - now)); continue; } - MountLeaseKeeper * keeper = mount_keeper.get(); + MountLeaseRenewer * renewer = mount_renewer.get(); auto lease = std::make_unique(*this, RenewalDriverState::WorkerCall); renewal_driver_state = RenewalDriverState::WorkerCall; driver_cv.notify_all(); - call = AdmittedKeeperCall{std::move(lease), keeper}; + call = AdmittedRenewerCall{std::move(lease), renewer}; } if (config.renewal_admitted_hook_for_test) config.renewal_admitted_hook_for_test(); try { - (void)renewKeeperOnce( + (void)renewRenewerOnce( std::move(call), RenewalDriverState::WorkerCall, /*propagate_failure=*/false, @@ -821,8 +815,8 @@ void CasMountRuntime::remountLoop() if (remount_requested_generation > remount_handled_generation) continue; if (lifecycle() == PoolLifecycle::Live - && mount_keeper - && mount_keeper->state() == MountLeaseKeeperState::Active) + && mount_renewer + && mount_renewer->state() == MountLeaseRenewerState::Active) { renewal_driver_state = RenewalDriverState::WorkerIdle; driver_cv.notify_all(); @@ -945,7 +939,7 @@ void CasMountRuntime::enterIdentityLost() { /// `TransientNotLive -> IdentityLost`, one way. The compare-exchange FROM `TransientNotLive` gives /// the brief's "from TransientNotLive only" precondition, idempotency (a second call finds the state - /// already `IdentityLost` and its exchange fails), and safety against a concurrent keeper + /// already `IdentityLost` and its exchange fails), and safety against a concurrent renewer /// `noteLeaseLost` (which only ever moves `Live -> TransientNotLive`, never away from it). It does NOT /// set `vanished_intent` (that latch is reserved for the `Vanished*` idempotency/FORGET protocol); /// rev.8 makes `IdentityLost` a fail-loud TERMINAL state through `remountTerminal`, which folds it @@ -1130,13 +1124,13 @@ void CasMountRuntime::finishTeardown(bool drained) { stopBackgroundWorkers(); - if (!mount_keeper) + if (!mount_renewer) return; - if (drained && mount_keeper->state() == MountLeaseKeeperState::Active) + if (drained && mount_renewer->state() == MountLeaseRenewerState::Active) { try { - mount_keeper->release(); + mount_renewer->release(); } catch (...) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 8a6c2cea2f8f..7f3ff1d76f36 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -1,6 +1,7 @@ #pragma once #include -#include +#include +#include #include #include #include @@ -75,18 +76,22 @@ struct MountConfig RuntimeWorkerFactory worker_factory = {}; /// Deterministic test interposition after the remount worker has confirmed renewal is parked, /// immediately before it releases `driver_mutex` and begins the real remount callback. + /// Runs with `driver_mutex` held: it must issue no backend request and never wait on the pool's + /// hot-key lane, whose holders sleep under that mutex, or the test deadlocks itself. std::function remount_parked_hook_for_test = {}; /// Deterministic test interposition at the top of the renewal loop, before it acquires /// `driver_mutex` to inspect cadence or parking state. std::function renewal_before_driver_lock_hook_for_test = {}; /// Deterministic test interposition after a due worker has atomically reserved renewal ownership - /// and captured its keeper, but before keeper/backend I/O starts. + /// and captured its renewer, but before renewer/backend I/O starts. std::function renewal_admitted_hook_for_test = {}; - /// Deterministic test interposition after terminal ownership has been deposited and the keeper is + /// Deterministic test interposition after terminal ownership has been deposited and the renewer is /// no longer reachable by the completed call. std::function renewal_terminal_deposited_hook_for_test = {}; /// Deterministic test interposition after the parked renewal predicate has sampled terminal false, /// immediately before the condition-variable wait atomically releases `driver_mutex`. + /// Runs with `driver_mutex` held: it must issue no backend request and never wait on the pool's + /// hot-key lane, whose holders sleep under that mutex, or the test deadlocks itself. std::function renewal_parked_predicate_false_hook_for_test = {}; /// Deterministic test interposition immediately before a terminal publisher attempts to acquire /// `driver_mutex`. @@ -95,18 +100,23 @@ struct MountConfig /// contention, but before it blocks acquiring the mutex. std::function terminal_publication_driver_lock_contended_hook_for_test = {}; /// Deterministic test interposition immediately after a terminal publisher acquires `driver_mutex`. + /// Runs with `driver_mutex` held: it must issue no backend request and never wait on the pool's + /// hot-key lane, whose holders sleep under that mutex, or the test deadlocks itself. std::function terminal_publication_driver_lock_acquired_hook_for_test = {}; /// Deterministic failure injection at the vanished-reason preparation boundary. std::function vanished_reason_prepare_hook_for_test = {}; - /// Test-only override for exact pre/post-send controller gate interleavings. - std::function renewal_stop_cause_for_test = {}; + /// Test-only override of the renewal's liveness predicate, for exact pre/post-send gate + /// interleavings. FALSE ends the renewal exactly as a lost fence does. + std::function renewal_live_for_test = {}; }; /// Local, in-memory write fence. It is deliberately not checked by reading the object store for every -/// write: the `MountLeaseKeeper` is the sole lease reader/renewer. A successful renewal translates the -/// durable `expires_at_ms` into `deadline_boot_ms`; a foreign owner, newer `writer_epoch`, or failed -/// renewal latches `lost`. Mutable operations are allowed only while the latch is clear and the local -/// deadline has not passed. The `writer_epoch` is the durable fencing token. +/// write: the `MountLeaseRenewer` is the sole lease reader/renewer. A successful renewal computes +/// `deadline_boot_ms` from its own confirmed request's pre-I/O `CLOCK_BOOTTIME` anchor plus the lease +/// TTL, never from the durable `expires_at_ms` stamp, which is a writer-stamped diagnostic only; a +/// foreign owner, newer `writer_epoch`, or failed renewal latches `lost`. Mutable operations are +/// allowed only while the latch is clear and the local deadline has not passed. The `writer_epoch` is +/// the durable fencing token. /// /// The fence uses `CLOCK_BOOTTIME`, not `CLOCK_MONOTONIC`: monotonic time does not advance while a VM is /// suspended, so a resumed sleeper would compute the same "not yet expired" verdict it had before the nap @@ -125,7 +135,7 @@ struct MountFence }; /// Owns the live writer-incarnation mechanics shared by the pool's mount and recovery orchestration: -/// the `MountLeaseKeeper`, local `MountFence`, build watermark and in-flight build registry, +/// the `MountLeaseRenewer`, local `MountFence`, build watermark and in-flight build registry, /// `live_writer_epoch`, unclean-boundary marker, and both persistent workers. `Pool` retains the higher-level /// claim/recovery sequence and its `remount_mutex`; in particular, the runtime does not acquire or own /// the ref-ledger locks. The runtime receives its backend, layout, configuration, event sink, request @@ -136,6 +146,10 @@ class CasMountRuntime public: CasMountRuntime( BackendPtr backend_ptr_, + /// The two planes the `MountLeaseRenewer` runs on: renewals under the mount fence, the farewell + /// on an open one. Owned by `Pool` and outliving this runtime. + CasRequests & mount_requests_, + CasRequests & farewell_requests_, const Layout & layout_, MountConfig config_, String server_root_id_, @@ -156,7 +170,7 @@ class CasMountRuntime /// Test/assertion accessor for the next-to-allocate build_seq under the lock. uint64_t peekNextBuildSeq(); /// Renew the merged mount heartbeat once, including its build-watermark floor. A read-only runtime - /// has no keeper and fails with a logical exception rather than fabricating a heartbeat. + /// has no renewer and fails with a logical exception rather than fabricating a heartbeat. void renewWatermarkOnce(); /// ---- local write fence ---- @@ -209,7 +223,7 @@ class CasMountRuntime /// `enterVanished`, OR EARLY (spec §5 step 1) by FORGET's `publishVanishedIntent`, and NEVER by the /// non-absorbing `IdentityLost` ([C1]). This is the EARLIEST terminal signal: it can already be true /// while the state is still pre-terminal (mid-FORGET). Consulted alongside `isVanished()` by every - /// background worker that must self-exit the moment the pool is (being driven) terminal — the keeper + /// background worker that must self-exit the moment the pool is (being driven) terminal — the renewer /// callback (`scheduleRemount`), the remount loop, and the GC scheduler. bool vanishedIntentPublished() const { return vanished_intent.load(std::memory_order_acquire); } @@ -295,6 +309,32 @@ class CasMountRuntime /// it cannot plausibly finish before the fence expires. bool refAppendFenceOk() const; + /// TRUE once the pool has reached — or is being driven toward — a state on which the self-remount + /// worker must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by + /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since + /// it is published before the state store) OR `IdentityLost` (a fail-loud TERMINAL state — no + /// demoted observer; recovery is restart or FORGET). Consulted by `scheduleRemount` before arming and by + /// the remount loop at every step boundary. (The GC scheduler applies the same three-way test through + /// `Pool`.) + bool remountTerminal() const + { + return vanished_intent.load(std::memory_order_acquire) + || lifecycle() == PoolLifecycle::IdentityLost; + } + + /// The inter-attempt sleep the mount plane runs on. A plain sleep would hold a parked or stopping + /// renewal for the whole capped backoff; this one wakes on the same stop signal the workers watch. + /// It shortens a stop, not a fence loss: the fence cannot see a stop request, so a woken operation + /// still reissues unless its own liveness predicate refuses. + void sleepInterruptibly(uint64_t ms); + + /// The mount fence's admission verdict, as `Fence::admit` expects it: may a request admitted under + /// `admitted_generation`, still expected to be running `needed_ms` from now, proceed? + /// `LostOrRearmed` when the fence is latched lost or a fresh lease incarnation replaced the one the + /// caller was admitted under; `NoBudget` when the live lease has no room left for `needed_ms` plus + /// the safety margin, so nothing is begun that could land after this node's fence may be gone. + Fence::Admit admit(uint64_t admitted_generation, uint64_t needed_ms) const; + /// The `writer_epoch` of the live mount incarnation. Bumped by `tryRemountOnce` (self-remount after a /// GC fence-out) — a `PartWriteTxn` minted under an older epoch fails closed on its next step. uint64_t liveWriterEpoch() const { return live_writer_epoch.load(std::memory_order_acquire); } @@ -321,12 +361,12 @@ class CasMountRuntime /// Publish the live-incarnation `live_writer_epoch` with release ordering. void setLiveWriterEpoch(uint64_t v); - /// ---- mount-lease keeper and persistent workers ---- - void installKeeper(UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms); - uint64_t startKeeper(); - uint64_t renewKeeperForStartupOnce(); - uint64_t renewKeeperForRemountOnce(); - void keeperReset(); + /// ---- mount-lease renewer and persistent workers ---- + void installRenewer(UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms); + uint64_t startRenewer(); + uint64_t renewRenewerForStartupOnce(); + uint64_t renewRenewerForRemountOnce(); + void renewerReset(); void startBackgroundWorkers(std::chrono::milliseconds period); void stopBackgroundWorkers(); /// Latch a recovery generation. Persistent remount ownership means this never constructs a thread. @@ -334,7 +374,7 @@ class CasMountRuntime bool scheduleRemountForTest(); void beginShutdownForTest(); /// Return how many times `scheduleRemount` was entered, including calls refused by the background - /// setting. This is useful for testing the keeper's loss callback without starting a real recovery. + /// setting. This is useful for testing the renewer's loss callback without starting a real recovery. uint64_t scheduleRemountCallCountForTest() const { return schedule_remount_calls_for_test.load(std::memory_order_relaxed); @@ -345,14 +385,21 @@ class CasMountRuntime bool workersRunningForTest() const; uint64_t remountRequestedGenerationForTest() const; - /// Join both persistent workers before an `Active` keeper may write its clean farewell. + /// Join both persistent workers before an `Active` renewer may write its clean farewell. void finishTeardown(bool drained); /// Sleep through the injected test hook when present; otherwise use the production thread sleep. /// `Pool` claim observation and materialization grace waits share this seam so tests control both. void waitSleep(uint64_t ms) const; - - /// Forward keeper events to the injected sink. The sink is held by reference so it observes the + /// Swap the wait hook after construction -- a test that must change what a wait DOES partway + /// through a scenario (e.g. driving a second incarnation's renewal from inside the observed + /// incarnation's own poll) cannot express that through `PoolConfig::wait_sleep_fn` alone, since + /// that value is fixed at open time. Unsynchronized against `waitSleep`'s `const` read of the same + /// field: safe only called from the test's own thread before any worker is running (no persistent + /// renewal/remount worker reads `config.wait_sleep_fn` concurrently with this write). + void setWaitSleepForTest(std::function fn) { config.wait_sleep_fn = std::move(fn); } + + /// Forward renewer events to the injected sink. The sink is held by reference so it observes the /// owning pool's current event routing for the runtime's entire lifetime. void emitEvent(CasEvent && e) const { if (event_sink) event_sink(std::move(e)); } @@ -370,22 +417,22 @@ class CasMountRuntime bool finished = false; }; - struct AdmittedKeeperCall + struct AdmittedRenewerCall { std::unique_ptr lease; - MountLeaseKeeper * keeper = nullptr; + MountLeaseRenewer * renewer = nullptr; }; - /// The renewal worker may drive a renewal only while it exclusively owns the driver and the keeper - /// is Active. Requires `driver_mutex`. `admitKeeperCall` enforces the same three conditions for every + /// The renewal worker may drive a renewal only while it exclusively owns the driver and the renewer + /// is Active. Requires `driver_mutex`. `admitRenewerCall` enforces the same three conditions for every /// other driver; the worker loop must park rather than throw when they do not hold, so it needs the /// predicate separately. Both the park test and the wake predicate use this one definition, so they /// cannot drift apart. bool renewalWorkerMayRenew() const; - AdmittedKeeperCall admitKeeperCall(RenewalDriverState required, RenewalDriverState active); - uint64_t renewKeeperOnce( - AdmittedKeeperCall call, + AdmittedRenewerCall admitRenewerCall(RenewalDriverState required, RenewalDriverState active); + uint64_t renewRenewerOnce( + AdmittedRenewerCall call, RenewalDriverState active, bool propagate_failure, bool worker_call); @@ -398,26 +445,19 @@ class CasMountRuntime void renewalLoop(); void remountLoop(); ThreadFromGlobalPool makeWorker(std::function body); - CasOverwriteStopCause renewalStopCause(bool worker_call) const; - bool waitForRetry(uint64_t wait_ms, bool worker_call); + /// The renewal's liveness: facts the mount fence cannot see -- a shutdown request, a parked or + /// park-requested driver, a pool that left `Live`. FALSE ends the renewal. + bool renewalLive(bool worker_call) const; + /// Whether this node has already been asked to stop. Sampled ONCE, before the write, so a refusal + /// caused by the stop cannot be mistaken for one that preceded it. + bool renewalCancelled() const; void tripFenceWithoutOperationalLoss(); std::unique_lock lockTerminalPublication(); - /// TRUE once the pool has reached — or is being driven toward — a state on which the self-remount - /// worker must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by - /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since - /// it is published before the state store) OR `IdentityLost` (rev.8: a fail-loud TERMINAL state — no - /// demoted observer; recovery is restart or FORGET). Consulted by `scheduleRemount` before arming and by - /// the remount loop at every step boundary. (The GC scheduler applies the same three-way test through - /// `Pool`, spec §9 rev.8 item 8.) - bool remountTerminal() const - { - return vanished_intent.load(std::memory_order_acquire) - || lifecycle() == PoolLifecycle::IdentityLost; - } - /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- BackendPtr backend_ptr; + CasRequests & mount_requests; + CasRequests & farewell_requests; const Layout & layout; MountConfig config; String server_root_id; @@ -430,7 +470,7 @@ class CasMountRuntime /// epoch is from a dead incarnation), never for ordering. next_build_seq is a strictly-increasing /// per-process counter (monotonicity is load-bearing — a seq is never reused or lowered); /// active_build_seqs holds the seqs of in-flight builds, so `minActive` yields the GC floor. The floor - /// is published by the merged `mount_keeper` + /// is published by the merged `mount_renewer` /// beat (there is no standalone watermark object anymore). ATOMIC because a self-remount re-stamps it /// (kept equal to `live_writer_epoch`) from the runtime-owned remount worker while `epoch`/`writerEpoch` /// may observe it; the ref-lane hot readers were moved to `liveWriterEpoch`, so this now backs only @@ -447,15 +487,15 @@ class CasMountRuntime /// Synchronous mount-lease protocol state. Constructed and started on a writable open after the /// owner/epoch/mount startup protocol; the runtime-owned renewal worker is its sole background /// driver and publishes successful anchors or terminal loss into the local fence. After both - /// workers join, teardown releases an `Active` keeper so a same-server reopen can reclaim + /// workers join, teardown releases an `Active` renewer so a same-server reopen can reclaim /// immediately. Null on a read-only open. - std::unique_ptr mount_keeper; + std::unique_ptr mount_renewer; std::atomic live_writer_epoch{0}; /// One mutex/condition pair owns driver admission, worker lifecycle, cadence, and the remount /// generation latch, and terminal predicates paired with `driver_cv`. It is never held across - /// keeper/backend calls, remount callbacks, logging, or joins. + /// renewer/backend calls, remount callbacks, logging, or joins. mutable std::mutex driver_mutex; mutable std::condition_variable driver_cv; RenewalDriverState renewal_driver_state = RenewalDriverState::Dormant; @@ -472,7 +512,7 @@ class CasMountRuntime std::atomic schedule_remount_calls_for_test{0}; /// Local write fence. The unarmed default (`deadline_boot_ms = UINT64_MAX`, `lost = false`) permits - /// mutation until a keeper supplies a real lease deadline or reports that the lease was lost. This + /// mutation until a renewer supplies a real lease deadline or reports that the lease was lost. This /// is the gate at the ref-append mutation chokepoint. MountFence mount_fence; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.cpp index 92c502cde3d0..516debe79ef7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.cpp @@ -4,7 +4,7 @@ #include #include #include -#include +#include #include #include #include @@ -22,6 +22,8 @@ namespace ProfileEvents { extern const Event CASBlobBodyPutAvoided; extern const Event CASBlobAdoptTrusted; + extern const Event CASMetaPut; + extern const Event CASMetaCompareSwap; extern const Event CASMetaCreateClean; extern const Event CASMetaAdoptBackfill; extern const Event CASMetaResurrectClean; @@ -72,7 +74,10 @@ uint64_t nowMs() bool isDeterministicBlobPublicationFailure(const std::exception & error) { - if (classifyConditionalWriteResult(error) == CasWriteOutcome::DefiniteFailure) + /// A refused write, EXCEPT the class a fresh credential fixes: the engine refreshes once before it + /// hands the failure back, so this loop's next physical attempt signs with what the refresh + /// installed, and `max_publication_attempts` is what bounds it if the refresh did not help. + if (isDefinitelyRefusedWrite(error) && !isRefreshableCredentialError(error)) return true; if (const auto * db_error = dynamic_cast(&error)) @@ -125,6 +130,7 @@ BlobSource BlobSource::fromString(String bytes) PartWriteTxn::PartWriteTxn(PoolPtr store_, UInt128 build_id_, uint64_t build_seq_, uint64_t epoch_, PartWriteInfo info_) : store(std::move(store_)) + , txn_generation(store->mountRequests().admit().generation()) , build_id(build_id_) , build_seq(build_seq_) , epoch(epoch_) @@ -259,17 +265,46 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) ErrorCodes::LOGICAL_ERROR, "PartWriteTxn::ensureBlobPresent: durable precommit required before materializing {}", blobIdOf(req.ref)); - /// This generation belongs to the operation, not to one observation/publication attempt. In - /// particular, an outer retry after ambiguous I/O must not adopt a re-armed incarnation, and a - /// trip-and-rearm hidden inside the mandatory `HEAD` must still invalidate the original writer. - const uint64_t admitted_generation = store->fenceGeneration(); + /// One operation per upload task, never shared: `fanOutBlobUploads` runs these concurrently and the + /// handle carries per-call state. It RESUMES on the generation the build was admitted under rather + /// than sampling a fresh one, so an outer retry after ambiguous I/O cannot adopt a re-armed + /// incarnation, a trip-and-rearm hidden inside the mandatory `HEAD` still invalidates the original + /// writer, and a re-arm between the precommit and this upload refuses the upload instead of proving + /// a dependency under an incarnation the precommit never saw. The build's own facts -- cancellation + /// and a superseded writer epoch -- stay in `requireAlive`, where each states which one refused. + CasOperation op = store->mountRequests().resume(txn_generation); + /// ONE bound for the whole publication loop, frozen before it starts: every HEAD and marker write + /// below shares this deadline, so a body whose publication keeps coming back ambiguous is refused + /// as retry-later inside one standard window instead of spending a fresh window per verb across + /// eight iterations. The paced retry is a bare sleep that does not consult the deadline, so the + /// loop can sleep one backoff (at most 5 s) past it before the next verb refuses to start. The + /// attempt cap below is the secondary bound. + const Retry policy = op.freeze(Retry::standard()); + /// The unrepeatable publication, under the SAME bound: the engine may never reissue an envelope + /// (see the publication call below), but the loop's deadline still governs whether one may start. + const Retry publication_policy = policy.asSingleAttempt(); const BlobRef & ref = req.ref; const BlobSource & source = req.source; const String key = store->layout().blobKey(ref); + const String meta_key = store->layout().blobMetaKey(ref); const PoolMeta & pool_meta = store->poolMeta(); const PoolConfig & pool_config = store->poolConfig(); + /// The verdict points. A decision that produces durable metadata or dependency readiness is refused + /// once the operation is no longer admitted, even where the mount has already re-armed and is + /// writable again by the time the request that crossed the boundary returned. + auto requireAdmitted = [&](std::string_view verdict) + { + if (!op.admitted()) + throwCasTransientUnavailable( + fmt::format("PartWriteTxn::ensureBlobPresent of '{}'", key), + fmt::format("the mount no longer admits this build {} -- either a lease loss the disk " + "auto-recovers from, or a FORGET decommission / lost identity that does NOT " + "recover; consult system.cas_mounts for the disk's lifecycle before retrying", + verdict)); + }; + auto buildHeader = [&]() { EnvelopeHeader header; @@ -281,18 +316,26 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) return encodeEnvelopeHeader(header, static_cast(pool_meta.blob_header_len)); }; - auto validateMetaSize = [&](const LoadedMeta & loaded) + auto validateMetaSize = [&](const BlobMeta & observed) { - if (loaded.meta.size != source.size) + if (observed.size != source.size) throw Exception( ErrorCodes::CORRUPTED_DATA, "PartWriteTxn::ensureBlobPresent: metadata for {} declares logical size {}, expected {}", key, - loaded.meta.size, + observed.size, source.size); }; - auto reconcileMetaClean = [&](std::optional loaded, BlobPublicationReason reason) + /// Bring the freshness marker to `Clean`. A publication that followed an ABSENT observation has + /// nothing at the marker key to decide from, so its create IS the whole reconciliation; routing it + /// through a read-decide-write would spend a GET on every insert to learn what the create settles + /// for itself. A resurrect already READ the stale `Condemned` marker before it published, and the + /// incarnation that read observed is the precondition its compare-swap needs -- so it spends no + /// second GET either. Only a write that loses -- a racing writer's marker, a marker that moved + /// under the resurrect -- needs the read, and there the engine's own loop is what bounds the + /// retries at the policy's deadline instead of a fixed count of unpaced attempts. + auto reconcileMetaClean = [&](const std::optional & loaded, BlobPublicationReason reason) { if (reason == BlobPublicationReason::Absent) ProfileEvents::increment(ProfileEvents::CASMetaCreateClean); @@ -300,48 +343,76 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) ProfileEvents::increment(ProfileEvents::CASMetaResurrectClean); const BlobMeta clean{.state = MetaState::Clean, .condemn_round = 0, .size = source.size}; - constexpr int max_meta_attempts = 8; - for (int attempt = 0; attempt < max_meta_attempts; ++attempt) + const String what = fmt::format( + "PartWriteTxn::ensureBlobPresent: reconciling the freshness metadata of '{}' to `Clean` " + "after blob publication", key); + + std::optional first; + if (reason == BlobPublicationReason::Absent) { - if (loaded) - { - validateMetaSize(*loaded); - if (loaded->meta.state == MetaState::Clean) - return; - if (casMeta(*store, ref, loaded->etag, clean).outcome == CasOverwriteOutcome::Committed) - return; - } - else if (putMetaIfAbsent(*store, ref, clean).outcome == CasOverwriteOutcome::Committed) - { - return; - } - loaded = loadMeta(store->backend(), store->layout(), ref); + ProfileEvents::increment(ProfileEvents::CASMetaPut); + first = op.create(meta_key, encodeBlobMeta(clean), policy); } - throwCasWriteRetryLater(fmt::format( - "PartWriteTxn::ensureBlobPresent: freshness metadata for {} did not reconcile to `Clean` " - "within {} attempts after blob publication", - key, - max_meta_attempts)); + else if (loaded) + { + ProfileEvents::increment(ProfileEvents::CASMetaCompareSwap); + first = op.replace(meta_key, encodeBlobMeta(clean), loaded->etag, policy); + } + /// Anything but a lost race is this call's answer, and `orThrow` maps it exactly as it maps + /// the read-decide-write's own result. + if (first && !std::holds_alternative(*first)) + { + orThrow(std::move(*first), what); + return; + } + + orThrow( + op.readModifyWrite( + meta_key, + [&](const std::optional & current) -> std::optional + { + if (current) + { + const BlobMeta observed = decodeBlobMeta(current->bytes); + validateMetaSize(observed); + if (observed.state == MetaState::Clean) + return std::nullopt; + /// The same two choke points the standalone marker writes count on, so a + /// reconciliation stays visible as a marker create or a marker compare-swap. + ProfileEvents::increment(ProfileEvents::CASMetaCompareSwap); + } + else + ProfileEvents::increment(ProfileEvents::CASMetaPut); + return encodeBlobMeta(clean); + }, + policy), + what); }; constexpr int max_publication_attempts = 8; for (int attempt = 0; attempt < max_publication_attempts; ++attempt) { requireAlive(); - const HeadResult head = store->backend().head(key); - std::optional loaded; + /// Pace the reissues the way the request engine paces its own, and through the engine's own + /// clock: an ambiguous publication is most often a store under load, and a large body + /// republished eight times back to back is what makes that worse. After `requireAlive`, so a + /// cancelled or superseded build fails closed instead of spending a backoff first. + if (attempt > 0) + op.pause(Retry::backoff(attempt)); + const std::optional present = op.head(key, policy); BlobPublicationReason reason = BlobPublicationReason::Absent; + std::optional loaded; - if (head.exists) + if (present) { - if (head.size < pool_meta.blob_header_len) + if (present->size < pool_meta.blob_header_len) throw Exception( ErrorCodes::CORRUPTED_DATA, "PartWriteTxn::ensureBlobPresent: blob {} size {} is below envelope length {}", key, - head.size, + present->size, pool_meta.blob_header_len); - const uint64_t logical_size = head.size - pool_meta.blob_header_len; + const uint64_t logical_size = present->size - pool_meta.blob_header_len; if (logical_size != source.size) throw Exception( ErrorCodes::CORRUPTED_DATA, @@ -350,37 +421,39 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) logical_size, source.size); - loaded = loadMeta(store->backend(), store->layout(), ref); + loaded = loadMeta(op, store->layout(), ref, policy); if (loaded) - validateMetaSize(*loaded); + validateMetaSize(loaded->meta); if (!loaded || loaded->meta.state == MetaState::Clean) { - /// Observation can produce durable metadata and dependency readiness too. Refuse both - /// when the mandatory `HEAD` crossed a fence generation, even if the mount has already - /// re-armed and is writable again by the time it returns. - store->checkFenceOrThrow(admitted_generation); + /// Observation can produce durable metadata and dependency readiness too. + requireAdmitted("after the mandatory `HEAD`"); if (!loaded) { + /// A backfilled marker is a point-read hint for the next observer, so a competing + /// writer that got there first settles the same question: its outcome is not read. ProfileEvents::increment(ProfileEvents::CASMetaAdoptBackfill); putMetaIfAbsent( - *store, + op, + store->layout(), ref, - BlobMeta{.state = MetaState::Clean, .condemn_round = 0, .size = logical_size}); + BlobMeta{.state = MetaState::Clean, .condemn_round = 0, .size = logical_size}, + policy); } - store->checkFenceOrThrow(admitted_generation); + requireAdmitted("before the body-put-avoided observation is recorded"); ProfileEvents::increment(ProfileEvents::CASBlobBodyPutAvoided); EventEmitter{*store}.emit([&](CasEvent & event) { event.type = CasEventType::BlobReuseAdopt; event.object_kind = CasEventObjectKind::Blob; event.object_hash = blobIdOf(ref); - event.token = head.token.value; + event.token = present->etag.render(); event.outcome = "observed"; event.reason = "a present non-condemned blob was observed after mandatory `HEAD`"; event.detail = {{"action", "observed"}, {"size", std::to_string(source.size)}}; }); - store->checkFenceOrThrow(admitted_generation); + requireAdmitted("before the observed dependency proof is returned"); return BlobUploadResult{ ref, BlobDepRecord{ObjectKind::Blob, BlobDependencyProof::Materialized, source.size}, @@ -389,7 +462,6 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) reason = BlobPublicationReason::Condemned; } - store->checkFenceOrThrow(admitted_generation); const bool first_publication = source.beginPublication(); BlobPublicationTransport transport; @@ -414,11 +486,19 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) try { - store->backend().publishBlob(BlobPublishRequest{key, std::move(publication)}); + /// A single physical publication, which the engine may never reissue: a re-sent envelope + /// re-publishes the same `incarnation_tag`, so on a content-derived-ETag dialect the + /// republished body carries the incarnation GC condemned and the exact-incarnation delete + /// would remove a live body. Every physical publication therefore mints its own envelope -- + /// each iteration of this loop builds one. The verbatim staged copy is additionally a + /// once-only privilege `beginPublication` spends. + op.publish(BlobPublishRequest{key, std::move(publication)}, publication_policy); } catch (const std::exception & error) { - if (isDeterministicBlobPublicationFailure(error)) + /// A publication this build is no longer admitted to make is not an ambiguity to retry: + /// every further request of this operation refuses the same way. + if (isDeterministicBlobPublicationFailure(error) || !op.admitted()) throw; if (attempt + 1 == max_publication_attempts) throwCasWriteRetryLater(fmt::format( @@ -440,11 +520,10 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) continue; } + reconcileMetaClean(loaded, reason); /// A publication may land just as this mount loses its fence. The bytes are harmless debris, /// but they cannot become dependency proof for the fenced transaction. - store->checkFenceOrThrow(admitted_generation); - reconcileMetaClean(loaded, reason); - store->checkFenceOrThrow(admitted_generation); + requireAdmitted("before the publication is recorded"); EventEmitter{*store}.emit([&](CasEvent & event) { event.type = CasEventType::BlobPut; @@ -461,7 +540,7 @@ BlobUploadResult PartWriteTxn::ensureBlobPresent(const BlobUploadRequest & req) {"size", std::to_string(source.size)}, {"build_id", u128ToHex(build_id)}}; }); - store->checkFenceOrThrow(admitted_generation); + requireAdmitted("before the published dependency proof is returned"); return BlobUploadResult{ ref, BlobDepRecord{ObjectKind::Blob, BlobDependencyProof::Materialized, source.size}, @@ -562,37 +641,48 @@ ManifestId PartWriteTxn::stageManifest(std::vector entries) const ManifestId id{owning_ns, ref}; const String key = store->layout().manifestKey(id); - /// Body PUT through the Pool's shared request controller: - /// budgeted attempts + resolve-before-reissue, replacing the old bare single-attempt write whose + /// Body PUT on the Pool's staging plane, under the mount fence a `precommitAdd` would fail anyway: + /// budgeted attempts with resolve-before-reissue, replacing the old bare single-attempt write whose /// whole S3-blip tolerance was ONE ~3s adaptive-timeout attempt (a 19s object-store pause killed an /// INSERT through it while every plain read/write path survived — v3 soak evidence). Reissuing this - /// conditional PUT is sound: the body bytes are fixed for the whole operation (`encoded` is built - /// once; `encodePartManifest` is canonical/deterministic), so `resolveByExactGet` can prove whether - /// an ambiguous attempt landed. Still NO preliminary HEAD. A DIFFERENT object at this key is a - /// ManifestId collision — the controller's resolve raises CORRUPTED_DATA (a proven conflict, - /// fail-closed before any owner transition can name this id), subsuming the old - /// PreconditionFailed->LOGICAL_ERROR mapping. - /// - /// fence_ok is the ref lane's own mount predicate (`refAppendFenceOk`: fence not lost + enough - /// lease left for one more attempt): staging runs on this writable Pool under that same mount - /// lease, and a fenced writer must not keep PUTting bodies ahead of a precommitAdd that would fail - /// the same fence anyway. There is no ref-table runtime here, so the lane's extra - /// `superseded_by_remount` term does not apply. - Token manifest_token; - const CasWriteOutcome put_outcome = store->stagingPutIfAbsent(key, encoded, &manifest_token); - if (put_outcome == CasWriteOutcome::DefiniteFailure) - throwCasWriteRetryLater(fmt::format( - "stageManifest: part-manifest PUT at '{}' definitively failed (non-retryable rejection); " - "nothing was named — the caller re-stages with a fresh ManifestId", key)); - /// Unresolved = budget exhausted (or fence lost) without a definite outcome. Unlike the ref-log - /// lane there is nothing to wedge: this id was never named by any owner transition - /// (`next_manifest_ordinal` is already past it, so no re-stage ever reuses the key), and a - /// late-landing body is inert unreferenced debris for the orphan-manifest sweep. NETWORK_ERROR = - /// the same retryable abort class the ref lane's exhausted budget maps to. - if (put_outcome == CasWriteOutcome::Unresolved) - throwCasWriteRetryLater(fmt::format( - "stageManifest: part-manifest PUT at '{}' is UNCERTAIN (retry budget exhausted) — " - "nothing conclusive was named; the caller re-stages with a fresh ManifestId", key)); + /// conditional PUT is sound: the body bytes are fixed for the whole call (`encoded` is built once; + /// `encodePartManifest` is canonical/deterministic), so the engine's resolve read can prove whether + /// an ambiguous attempt landed. Still NO preliminary HEAD. + WriteResult staged = store->stagingPutIfAbsent(key, encoded); + const Etag manifest_incarnation = std::visit(detail::Overload{ + [](Committed & committed) -> Etag { return std::move(committed.etag); }, + [&](Conflict & conflict) -> Etag + { + /// Our own bytes under our own `ManifestId` name this same body, whoever wrote them; a + /// DIFFERENT object under an id this build minted is a ManifestId collision, fail-closed + /// before any owner transition can name it. + if (const auto * object = std::get_if(&conflict.seen); object && object->bytes == encoded) + return object->etag; + throw Exception(ErrorCodes::CORRUPTED_DATA, + "stageManifest: part-manifest key '{}' already holds {} that is not this manifest's body " + "-- a ManifestId collision", key, detail::renderObservation(conflict.seen)); + }, + [&](Declined &) -> Etag + { + throw Exception(ErrorCodes::LOGICAL_ERROR, + "stageManifest: the part-manifest create at '{}' declined; a create has nothing to decline", key); + }, + [&](Refused & refused) -> Etag + { + throwCasWriteRetryLater(fmt::format( + "stageManifest: part-manifest PUT at '{}' definitively failed ({}); " + "nothing was named — the caller re-stages with a fresh ManifestId", key, refused.message)); + }, + /// Unlike the ref-log lane there is nothing to wedge: this id was never named by any owner + /// transition (`next_manifest_ordinal` is already past it, so no re-stage ever reuses the key), + /// and a late-landing body is inert unreferenced debris for the orphan-manifest sweep. + [&](GaveUp &) -> Etag + { + throwCasWriteRetryLater(fmt::format( + "stageManifest: part-manifest PUT at '{}' is UNCERTAIN (retry budget exhausted) — " + "nothing conclusive was named; the caller re-stages with a fresh ManifestId", key)); + }}, + staged); EventEmitter{*store}.emit([&](CasEvent & e) { @@ -600,7 +690,7 @@ ManifestId PartWriteTxn::stageManifest(std::vector entries) e.namespace_ = owning_ns.string(); e.object_kind = CasEventObjectKind::Manifest; e.object_hash = manifestRefDebugString(id.ref); - e.token = manifest_token.value; + e.token = manifest_incarnation.render(); e.reason = "stageManifest: part-manifest body written"; }); @@ -734,7 +824,8 @@ bool PartWriteTxn::promote(const RootNamespace & target_ns, const String & final /// Read + validate the manifest body ONCE (O(manifest entries), one streaming read). Absent or /// invalid ⇒ fail closed: a committed ref must never name a missing/mismatched manifest. const String manifest_key = store->layout().manifestKey(id); - const auto body_got = store->backend().get(manifest_key); + CasOperation op = store->mountRequests().admit(); + const auto body_got = op.read(manifest_key, Retry::standard()); if (!body_got) throwCasWriteRetryLater(fmt::format( "promote: manifest body absent at {} — failing closed (retry with a fresh ManifestId)", manifest_key)); @@ -1083,10 +1174,10 @@ void PartWriteTxn::abandon() /// `Uncertain` tolerance above, no-ops) -- it never corrupts. alive = false; - /// No longer in-flight: retire the seq so the per-server active-build floor (`min_active`) can advance + /// No longer in-flight: retire the seq so the per-server active-build floor (`min_active_build_sequence`) can advance /// (idempotent). This runs AFTER the precommit removal above (mirrors `PartWriteTxn::promote`, which retires /// after its commit) so the build stays active until its precommit binding's removal is durable: - /// retiring first would advance `min_active` past a build whose precommit binding is still live in the + /// retiring first would advance `min_active_build_sequence` past a build whose precommit binding is still live in the /// ref log, letting a freshness-window consumer judge the manifest build-dead while an un-removed /// precommit still names it. Ordering removal-before-retire keeps that happens-before clean. store->retireBuildSeq(build_seq); @@ -1117,8 +1208,8 @@ void PartWriteTxn::abandon() void PartWriteTxn::cleanupStagedManifestDebrisBestEffort() { /// Best-effort writer cleanup of THIS build's pre-precommit/staged `_manifests` debris. The common case - /// is writer cleanup; a missed object is benign — the namespace-scoped orphan sweep reclaims it. Exact-token delete only; never - /// throws. SKIP the manifest that became a live precommit owner: its body is a live precommit input + /// is writer cleanup; a missed object is benign — the namespace-scoped orphan sweep reclaims it. Exact-incarnation + /// delete only; never throws. SKIP the manifest that became a live precommit owner: its body is a live precommit input /// whose deletion is GC's job after the sealed decrement (never writer-delete it). /// /// "Became a live precommit owner" is decided from the ATTEMPT, not from a confirmed append: any @@ -1127,6 +1218,7 @@ void PartWriteTxn::cleanupStagedManifestDebrisBestEffort() /// precommit that turns out to be live and whose body is gone clamps GC's fold barrier forever), /// while keeping it is not: an unreferenced body is ordinary orphan-sweep debris. const bool precommit_attempted = precommit_state != PrecommitState::NotAttempted; + CasOperation op = store->mountRequests().admit(); for (const ManifestId & id : staged_manifests) { if (precommit_attempted && id.ref == precommit_manifest && id.root_namespace == precommit_target_ns) @@ -1134,9 +1226,8 @@ void PartWriteTxn::cleanupStagedManifestDebrisBestEffort() try { const String key = store->layout().manifestKey(id); - const HeadResult hr = store->backend().head(key); - if (hr.exists) - store->backend().deleteExact(key, hr.token); + if (const auto observed = op.head(key, Retry::standard())) + op.remove(key, observed->etag, Retry::standard()); } catch (...) // NOLINT(bugprone-empty-catch) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h index f7a8026bec01..953059a76ebc 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h @@ -31,8 +31,10 @@ struct BlobSource std::shared_ptr> publication_attempted = std::make_shared>(false); - /// Atomically consume the logical source's first-publication privilege. Called after the final - /// fence check and immediately before backend publication I/O. + /// Atomically consume the logical source's first-publication privilege -- the verbatim staged copy + /// is available to exactly one physical publication of this source, whichever one gets here first. + /// A fenced-out writer may spend it before the engine refuses its publication; the only effect is + /// that a later publication streams instead of copying. bool beginPublication() const { return !publication_attempted->exchange(true, std::memory_order_acq_rel); @@ -349,6 +351,11 @@ class PartWriteTxn void cleanupStagedManifestDebrisBestEffort(); PoolPtr store; + /// The mount incarnation this BUILD was admitted under, sampled once when it began. Every upload + /// task resumes on it rather than admitting afresh, so a fence re-armed mid-transaction makes the + /// upload give up instead of minting a dependency proof under an incarnation the precommit never + /// saw -- a re-arm that keeps the writer epoch is invisible to `requireAlive`. + uint64_t txn_generation{}; UInt128 build_id{}; uint64_t build_seq{}; /// per-process monotone sequence uint64_t epoch{}; /// owning Pool's process_epoch diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.cpp index 511ff5cc2516..0c87b37758fb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.cpp @@ -1,64 +1,32 @@ #include #include #include - -namespace DB -{ -namespace ErrorCodes -{ - extern const int ABORTED; -} -} +#include namespace DB::Cas { -namespace -{ - constexpr size_t MAX_CAS_ATTEMPTS = 100; -} - void CasPlainObjects::casPutObject(const String & full_key, const String & bytes) { - /// The read determines whether this is a conditional create or replacement. The token is only - /// valid for the incarnation returned by that head, so a precondition failure means another - /// writer won the race and the loop must observe the new incarnation before trying again. - /// - /// SINGLE-APPENDER INVARIANT: `bytes` is frozen by the caller before this loop starts (see the - /// append-base note at `ContentAddressedTransaction::writeFile`'s Append branch); the loop only - /// re-reads the TOKEN on conflict, never the base content. This is correct only while nothing - /// concurrently appends to the same key — a losing retry would overwrite the winner's bytes with a - /// stale, pre-conflict payload (a lost update). Implement a real `casAppendObject` (re-reading the - /// base content, not just the token, inside the loop) before adding any concurrent appender. - /// - /// rev.7 [C2]: the fence generation captured at admission is re-checked immediately before EVERY - /// durable PUT below, not just the first attempt. A mismatch (the mount lease was lost, or re-armed - /// under a fresh incarnation, since admission) aborts with the typed transient error before the backend - /// is ever touched. - const uint64_t admitted_generation = fence_generation_fn(); - - for (size_t attempt = 0; attempt < MAX_CAS_ATTEMPTS; ++attempt) - { - HeadResult head = backend.head(full_key); - check_fence_or_throw_fn(admitted_generation); - if (!head.exists) - { - if (backend.putIfAbsent(full_key, bytes).outcome == PutOutcome::Done) - return; - } - else - { - if (backend.putOverwrite(full_key, bytes, head.token).outcome == PutOutcome::Done) - return; - } - /// `PreconditionFailed` means the observed state changed under us; re-head and retry. - } - throw Exception(ErrorCodes::ABORTED, "object CAS contention on '{}'", full_key); + /// SINGLE-APPENDER INVARIANT: `bytes` is frozen by the caller before this call (see the + /// append-base note at `ContentAddressedTransaction::writeFile`'s Append branch); `decide` below + /// always returns the same frozen bytes regardless of what it observes at the key. This is correct + /// only while nothing concurrently appends to the same key -- a losing retry would overwrite the + /// winner's bytes with a stale, pre-conflict payload (a lost update). Implement a real + /// `casAppendObject` (deciding from the current body, not just presence) before adding any + /// concurrent appender. + CasOperation op = requests.admit(); + WriteResult result = op.readModifyWriteOnPresence( + full_key, + [&](const std::optional &) -> std::optional { return bytes; }, + Retry::standard()); + orThrow(std::move(result), fmt::format("object CAS write on '{}'", full_key)); } std::optional CasPlainObjects::casGetObject(const String & full_key) { - std::optional result = backend.get(full_key); + CasOperation op = requests.admit(); + std::optional result = op.read(full_key, Retry::standard()); if (!result) return std::nullopt; return result->bytes; @@ -66,26 +34,11 @@ std::optional CasPlainObjects::casGetObject(const String & full_key) void CasPlainObjects::casRemoveObject(const String & full_key) { - /// Delete only the incarnation observed by the preceding head. A token mismatch leaves the - /// replacement untouched and is retried against a fresh observation; absence is a successful - /// no-op. - /// - /// rev.7 [C2]: same fence-generation admission as `casPutObject` -- the admitted generation is - /// re-checked immediately before every durable delete. - const uint64_t admitted_generation = fence_generation_fn(); - - for (size_t attempt = 0; attempt < MAX_CAS_ATTEMPTS; ++attempt) - { - const HeadResult head = backend.head(full_key); - if (!head.exists) - return; - check_fence_or_throw_fn(admitted_generation); - const DeleteOutcome outcome = backend.deleteExact(full_key, head.token); - if (outcome.kind == DeleteOutcome::Kind::Deleted || outcome.kind == DeleteOutcome::Kind::NotFound) - return; - /// `TokenMismatch` means a concurrent rewrite; re-head and retry. - } - throw Exception(ErrorCodes::ABORTED, "object CAS contention on '{}' (runaway live-lock brake)", full_key); + /// `removeCurrent` re-heads and retries against a concurrent replacement itself, and reports + /// absence (its own no-op) the same way whether the key was never there or just vanished, so the + /// result carries nothing this caller acts on differently. + CasOperation op = requests.admit(); + op.removeCurrent(full_key, Retry::standard()); } void CasPlainObjects::putNamespaceFile(const NamespaceLifeId & life, const String & name, const String & bytes) @@ -102,20 +55,14 @@ std::vector CasPlainObjects::listNamespaceFiles(const NamespaceLifeId & { const String prefix = layout.namespaceFilesPrefix(life); std::vector names; - String cursor; - while (true) + CasOperation op = requests.admit(); + op.forEachListedKey(prefix, [&](const ListedKey & entry) { - ListPage page = backend.list(prefix, cursor, /*limit*/ 1000); - for (const ListedKey & listed : page.keys) - { - /// Strip the storage prefix so callers receive the bare flat file name. - if (listed.key.starts_with(prefix)) - names.push_back(listed.key.substr(prefix.size())); - } - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } + /// Strip the storage prefix so callers receive the bare flat file name. + if (entry.key.starts_with(prefix)) + names.push_back(entry.key.substr(prefix.size())); + return true; + }, Retry::standard()); /// Backends are not required to return pages in the same order, so make the public result /// deterministic instead of relying on `InMemoryBackend` ordering. std::sort(names.begin(), names.end()); @@ -143,7 +90,8 @@ bool CasPlainObjects::mountpointObjectExists(const String & key) /// the `store` pool subdirectory traversed by `system.remote_data_paths`. The local backend /// treats a directory as not an object, so this returns false instead of attempting a body read /// that would raise a filesystem exception for a directory. - return backend.head(layout.mountpointObjectKey(key)).exists; + CasOperation op = requests.admit(); + return op.head(layout.mountpointObjectKey(key), Retry::standard()).has_value(); } void CasPlainObjects::removeMountpointObject(const String & key) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.h index d1eae78d6e57..76af7681a3eb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPlainObjects.h @@ -1,9 +1,7 @@ #pragma once -#include +#include #include #include -#include -#include #include #include #include @@ -16,35 +14,23 @@ namespace DB::Cas /// by the namespace LIFE, never by its bare name -- and mountpoint objects mirrored by path. The object /// bodies are raw passthrough bytes; this component does not decode them as CAS metadata. /// -/// The component holds references to the shared `Backend` and `Layout` only. It owns no pool mutex +/// The component holds references to the shared `CasRequests` and `Layout` only. It owns no pool mutex /// and has no pool back-reference, allowing `Pool` to retain thin forwarding methods with the same /// external interface. The private helpers implement the shared head-plus-conditional-write and -/// head-plus-exact-delete protocols used by both object families. A conditional outcome means that -/// the observed incarnation changed, so the helper re-reads the head and retries; the fixed bound -/// prevents an unexpected continuous conflict from becoming an unbounded operation and reports -/// `ABORTED` when it is reached. -/// -/// Every durable write/delete on this surface is fence-generation-gated (rev.7 [C2]): `Pool` injects -/// two callbacks that reach its `mount_runtime` (declared AFTER this member, hence constructed -/// after it -- these callbacks capture `Pool` itself and are invoked only at runtime, post- -/// construction, exactly like `ref_ledger`'s callbacks in `CasPool.cpp`, so referencing a -/// not-yet-constructed sibling member through them is safe). +/// head-plus-exact-delete protocols used by both object families, over an operation admitted fresh for +/// each call: every attempt inside it re-checks admission before touching the store, so a mount lease +/// lost mid-call is refused rather than written through. class CasPlainObjects { public: - CasPlainObjects( - Backend & backend_, const Layout & layout_, - std::function fence_generation_fn_, - std::function check_fence_or_throw_fn_) - : backend(backend_), layout(layout_) - , fence_generation_fn(std::move(fence_generation_fn_)) - , check_fence_or_throw_fn(std::move(check_fence_or_throw_fn_)) + CasPlainObjects(CasRequests & requests_, const Layout & layout_) + : requests(requests_), layout(layout_) { } /// Stores the raw bytes under ONE LIFE's `_files/` prefix. Existing files are replaced - /// conditionally using the object incarnation observed by `Backend::head`; a storage failure or an - /// exhausted conflict-retry bound is propagated as an exception. + /// conditionally using the object incarnation observed by the admitted operation; a storage failure + /// or an exhausted conflict-retry bound is propagated as an exception. /// /// `life` is supplied by the caller and never re-derived here, so this surface issues no catalog /// request of its own. A stale writer therefore targets its own old incarnation's key and cannot @@ -53,7 +39,7 @@ class CasPlainObjects /// Reads a namespace file of ONE LIFE without interpreting its body. Returns `nullopt` when the /// object is absent and propagates backend read failures. A stale reader may see stale bytes or - /// `NotFound`, never a newer incarnation's data: its key names the life it was given. + /// absence, never a newer incarnation's data: its key names the life it was given. std::optional getNamespaceFile(const NamespaceLifeId & life, const String & name); /// Enumerates the file names directly below ONE LIFE's `_files/` prefix. Fetches all paginated @@ -61,9 +47,9 @@ class CasPlainObjects /// listing order. std::vector listNamespaceFiles(const NamespaceLifeId & life); - /// Removes the current OBJECT incarnation of one of a life's files, if any (the object token, not - /// the namespace incarnation, which `life` fixes). A concurrent replacement is never removed - /// accidentally: the exact-delete helper re-reads and retries with the new token. + /// Removes the current OBJECT incarnation of one of a life's files, if any (the object incarnation, + /// not the namespace incarnation, which `life` fixes). A concurrent replacement is never removed + /// accidentally: the underlying `removeCurrent` re-heads and retries against the new incarnation. void removeNamespaceFile(const NamespaceLifeId & life, const String & name); /// Stores raw bytes for a loose mountpoint file at the path-derived object key. The key is @@ -80,32 +66,27 @@ class CasPlainObjects /// attempting to read a directory as an object. bool mountpointObjectExists(const String & key); - /// Removes the current path-mirrored mountpoint-object incarnation, if present, using exact-token - /// deletion so a concurrent rewrite remains intact. + /// Removes the current path-mirrored mountpoint-object incarnation, if present, using + /// `removeCurrent`'s re-head-and-retry so a concurrent rewrite remains intact. void removeMountpointObject(const String & key); private: - /// Creates or conditionally replaces one raw object. The method re-heads after a conditional - /// conflict and throws `ABORTED` after the bounded retry loop cannot establish a stable token. - /// Fence-generation-gated (rev.7 [C2]): captures the fence generation at admission for the call's - /// whole retry loop; every iteration re-checks it immediately before its durable PUT. + /// Creates or conditionally replaces one raw object. The write always sends `bytes` regardless of + /// what is currently there, settling a refused precondition with a HEAD; only an ambiguous attempt + /// reads the body, because only the bytes can prove it landed. Retries a lost precondition under + /// the engine's own bound; an exhausted retry or a store refusal is propagated as an exception. void casPutObject(const String & full_key, const String & bytes); - /// Reads one raw object by its complete backend key and returns `nullopt` when it is absent. A read, - /// not a durable-effect operation -- NOT fence-gated (rev.7 [C2] scopes the gate to durable writes). + /// Reads one raw object by its complete backend key and returns `nullopt` when it is absent. std::optional casGetObject(const String & full_key); - /// Removes one raw object by exact token. Absence is a successful no-op; a token mismatch causes - /// a fresh head and retry, while a bounded retry failure throws `ABORTED`. Fence-generation-gated - /// the same way as `casPutObject`. + /// Removes one raw object at its current incarnation. Absence is a successful no-op; a concurrent + /// replacement is retried against the freshly observed incarnation, and an exhausted retry is + /// propagated as an exception. void casRemoveObject(const String & full_key); - Backend & backend; + CasRequests & requests; const Layout & layout; - - /// ---- fence-generation admission (injected by `Pool`; see the class doc comment) ---- - std::function fence_generation_fn; - std::function check_fence_or_throw_fn; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 1ea5c2d567cb..8ea206618492 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -87,21 +88,7 @@ void validateWritableMountTiming(const PoolConfig & config) "CAS mount timing rejected: lease TTL must be positive and renewal period non-negative"); const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); const uint64_t period_ms = static_cast(config.mount_renew_period.count()); - validateCasRequestBudget(config.cas_request_budget, ttl_ms, period_ms); - if (!config.background_watermark) - return; - - const uint64_t safety_ms = config.cas_request_budget.lease_safety_margin_ms; - const uint64_t attempt_ms = config.cas_request_budget.attempt_timeout_ms; - const bool cadence_fits = safety_ms <= ttl_ms - && period_ms <= ttl_ms - safety_ms - && attempt_ms <= ttl_ms - safety_ms - period_ms; - if (!cadence_fits) - throw Exception( - ErrorCodes::BAD_ARGUMENTS, - "CAS mount renewal cadence rejected: period ({} ms) + attempt timeout ({} ms) must be " - "at most TTL ({} ms) - safety margin ({} ms) when background renewal is enabled", - period_ms, attempt_ms, ttl_ms, safety_ms); + validateCasRequestBudget(config.cas_request_budget, ttl_ms, period_ms, config.background_watermark); } /// The verdict of the pool-lifecycle identity gate (step 0 of `tryRemountOnce`, spec §2). Exactly one @@ -127,10 +114,13 @@ struct LifecycleGate /// `min_reader_generation` are legally mutable and are deliberately not compared (the format gate is the /// decode itself succeeding). LifecycleGate probePoolLifecycleGate( - Backend & backend, const Layout & layout, const String & srid, + CasOperation & op, const Layout & layout, const String & srid, UInt128 expected_pool_id, uint64_t expected_blob_header_len) { - const SentinelProbeResult meta_probe = probeSentinel(backend, layout.poolMetaKey()); + /// `once` on both probes: an inconclusive answer IS this gate's verdict (`StayTransient`), and the + /// recovery loop that called it is what retries. Reissuing here would spend the loop's whole + /// interval inside a single probe and make a terminal transition wait for it. + const SentinelProbeResult meta_probe = probeSentinel(op, layout.poolMetaKey(), Retry::once()); switch (meta_probe.outcome) { case ProbeOutcome::Present: @@ -163,7 +153,7 @@ LifecycleGate probePoolLifecycleGate( /// authoritatively absent ⇒ `IdentityLost` (a fail-loud terminal state), regardless of whatever /// else remains under the prefix. Erasure is never PROVEN by the system — only asserted by the /// operator's `FORGET` — so there is no prefix-emptiness leg here. - const SentinelProbeResult owner_probe = probeSentinel(backend, layout.ownerKey(srid)); + const SentinelProbeResult owner_probe = probeSentinel(op, layout.ownerKey(srid), Retry::once()); if (owner_probe.outcome != ProbeOutcome::KeyAbsent) return {LifecycleGateVerdict::StayTransient, "_pool_meta absent but the owner sentinel was not conclusively absent"}; @@ -181,6 +171,36 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) : pool_backend(std::move(backend_)) , config(std::move(config_)) , meta(std::move(meta_)) + , hot_keys(config.hot_key_cache_bytes) + /// The mount plane's fence reaches `mount_runtime`, declared far below: the closures capture + /// `this` and run only after construction, exactly like `ref_ledger`'s callbacks. All three planes + /// take the fence's own clock, so a policy bound to a mount-lease deadline and the fence that + /// enforces it are read from the same source. + , mount_requests(pool_backend, Fence{ + [this] { return mount_runtime.fenceGeneration(); }, + [this](uint64_t g, uint64_t needed) { return mount_runtime.admit(g, needed); }, + [this](uint64_t g) { mount_runtime.checkFenceOrThrow(g); }}, + config.boot_ms_fn, + config.retry_sleep_fn ? config.retry_sleep_fn : mountPlaneSleepFn(), + &hot_keys) + , farewell_requests(pool_backend, Fence::open(), config.boot_ms_fn, config.retry_sleep_fn, &hot_keys) + /// The open plane's fence is the pool's teardown flag: generation 0 forever, exactly like + /// `Fence::open`, but `admit` refuses once `beginTeardown` ran. A GC round, an FSCK or a probe in + /// flight is then refused at its next request instead of running to completion under a disk that + /// is being torn down. The ref ledger and the farewell live on the other two planes, so + /// teardown's own I/O never meets this fence. A write already proven durable is admitted ONCE + /// MORE (`postCommit`), so an armed teardown can turn a landed `gc/state` into a give-up rather + /// than a commit. That is safe and not merely tolerable: the round is one-pass, so the next round + /// reads the state this one committed and re-derives the rest, exactly as after a crash at that + /// instant -- and every step of the tail the give-up skipped is admitted on THIS plane, so it + /// would have been refused anyway. What it costs is the round number on the round's own row. + , gc_requests(pool_backend, Fence{ + [] { return uint64_t{0}; }, + [this](uint64_t, uint64_t) { return teardownBegun() ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}, + config.boot_ms_fn, + config.retry_sleep_fn ? config.retry_sleep_fn : openPlaneSleepFn(), + &hot_keys) /// Seed the monotone admitted-algo cache from the pool state `createOrValidate` already /// established (fresh create, steady-state member, or a just-completed admission union) -- /// register-before-first-write means this Pool's own `writeAlgo()` is ALWAYS a @@ -189,36 +209,30 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) /// `Layout` no longer captures a pool algo -- every blob key is built from a /// `BlobRef` (algo + digest) directly, so the constructor takes only the pool prefix. , pool_layout(config.pool_prefix) - /// Plain-object surface component: binds to this Pool's own backend + layout (declared after - /// both, so this reference-holding member is constructed last and destroyed first) plus two - /// fence-generation callbacks reaching `mount_runtime` (declared AFTER `plain_objects`, hence - /// constructed after it -- these callbacks capture `this` and are invoked only at runtime, - /// post-construction, exactly like `ref_ledger`'s callbacks below, so referencing a - /// not-yet-constructed sibling member through them is safe). - , plain_objects( - *pool_backend, pool_layout, - [this] { return mount_runtime.fenceGeneration(); }, - [this] (uint64_t gen) { mount_runtime.checkFenceOrThrow(gen); }) - /// Manifest reader component: backend/layout/meta by reference + the event-sink reference. The - /// sink is installed by the factory before writable mounting starts. Owns the decode cache, - /// built from the same config bytes the Pool ctor used before. - , manifest_reader(*pool_backend, pool_layout, meta, event_sink_, config.manifest_decode_cache_bytes) - /// Ref-log / ref-table subsystem. Injected with backend/layout + the - /// RefLedgerConfig slice + the event-sink reference + the pool `cas_request_budget` + the RAW mount - /// `boot_ms_fn` (for its retry controller), plus callbacks into the mount/watermark state that lives + /// Plain-object surface component, on the MOUNT plane: the namespace-file and mountpoint writes it + /// offers are durable mutations whose right to land is the mount lease. + , plain_objects(mount_requests, pool_layout) + /// Manifest reader component, on the MOUNT plane: a content read on a mount whose lease is gone is + /// refused, which is what the reader's own contract promises and what the metadata storage's op gate + /// already does for every content read. A reader for work that outlives the lease (GC) is a separate + /// reader over that plane, never this one shared across two. The event sink is installed by the + /// factory before writable mounting starts. + , manifest_reader(mount_requests, pool_layout, meta, event_sink_, config.manifest_decode_cache_bytes) + /// Ref-log / ref-table subsystem, on the MOUNT plane: a ref-lane write and a mount-lease renewal + /// are then measured against the same fence and the same clock. Injected with the + /// RefLedgerConfig slice + the event-sink reference + the pool `cas_request_budget`, plus callbacks + /// into the mount/watermark state that lives /// on `mount_runtime` (reached through Pool delegates). The callbacks capture `this`; they are /// invoked only at runtime (post-construction), so referencing `mount_runtime` (declared AFTER /// `ref_ledger`, hence constructed after it) is safe -- exactly as the pre-3.5 layout referenced the /// mount raw-members that also followed `ref_ledger`. Declared/constructed BEFORE `mount_runtime`, /// preserving the original member order verbatim (see the header note). , ref_ledger( - pool_backend, pool_layout, config.refLedgerConfig(), event_sink_, config.cas_request_budget, + mount_requests, pool_layout, config.refLedgerConfig(), event_sink_, config.cas_request_budget, config.server_root_id, - config.boot_ms_fn, [this] { return liveWriterEpoch(); }, [this] { return refAppendFenceOk(); }, [this] { return mount_runtime.fenceGeneration(); }, - [this] (uint64_t gen) { mount_runtime.checkFenceOrThrow(gen); }, [this] { return bootMsNow(); }, [this] { return mayMutate(); }, [this] (const String & key, const String & reason, const std::optional & offending_ns) @@ -228,13 +242,14 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) [this] (const RootNamespace & ns) { cancelInflightBuildsForNamespace(ns); }, config.recovery_pre_first_request_hook_for_test) /// Mount / write-fence / build-watermark / self-remount runtime. Injected with - /// backend/layout + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool + /// backend/layout + the mount and farewell planes + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool /// `cas_request_budget` + the `remount_attempt` callback (== `Pool::tryRemountOnce`, whose claim/ /// recovery ORCHESTRATION stays on Pool). The callback captures `this`; it is invoked only at runtime /// (post-construction). Declared/constructed AFTER `ref_ledger`, preserving the original member order /// verbatim (mount destroyed first, ledger last; both orders proven safe -- see the header note). , mount_runtime( - pool_backend, pool_layout, config.mountConfig(), config.server_root_id, event_sink_, + pool_backend, mount_requests, farewell_requests, + pool_layout, config.mountConfig(), config.server_root_id, event_sink_, config.cas_request_budget, [this] { return tryRemountOnce(); }) { @@ -252,7 +267,8 @@ std::vector Pool::refreshAdmittedAlgos() /// A direct GET+decode of `_pool_meta`, not a re-run of `createOrValidate`'s admission logic -- /// this Pool's OWN algo is already admitted, so all this /// needs is the CURRENT authoritative `algos_used`, unioned into the monotone cache. - const auto existing = pool_backend->get(pool_layout.poolMetaKey()); + CasOperation op = gc_requests.admit(); + const auto existing = op.read(pool_layout.poolMetaKey(), Retry::standard()); std::lock_guard lock(admitted_algos_mutex); if (existing) @@ -268,7 +284,7 @@ std::vector Pool::refreshAdmittedAlgos() return admitted_algos; } -/// ==== mount-runtime delegates ==== The mount lease keeper, the local write +/// ==== mount-runtime delegates ==== The mount lease renewer, the local write /// fence, the per-server build watermark, the live-incarnation epoch, and the self-remount recovery /// thread live in the `mount_runtime` member (Pool/CasMountRuntime.h); Pool keeps these thin public /// forwarders so the wiring, PartWriteTxn, Gc, the ref-ledger callbacks, and every test call site are unchanged. @@ -341,8 +357,9 @@ String Pool::lifecycleReasonDetail(PoolLifecycle lc) const void Pool::throwIfLifecycleTerminal() const { /// The typed error carries the sub-state in its message so a wrong diagnosis is impossible from the - /// first error line (spec §1 [D5]). `Live`/`TransientNotLive` proceed here — the transient class is - /// still gated only by the write fence in this task (the full six-class gate is Task 8). + /// first error line. `Live`/`TransientNotLive` proceed here — the transient class is + /// still gated only by the write fence; the destructive gate additionally requires the other + /// lifecycle proofs. const PoolLifecycle lc = mount_runtime.lifecycle(); if (lc == PoolLifecycle::Live || lc == PoolLifecycle::TransientNotLive) return; @@ -376,6 +393,9 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// FAIL-CLOSED: the capability probe throws NOT_IMPLEMENTED on any failed check, and /// PoolMeta::createOrValidate is pool-authoritative — the config constants apply only at creation. Layout layout(config.pool_prefix); + /// The whole bootstrap runs on an OPEN fence: there is no mount lease yet, and the claim below is + /// what establishes one. + CasRequests bootstrap_requests(backend, Fence::open(), config.boot_ms_fn); bool initialize_empty_catalog = false; /// The probe writes and deletes throwaway keys to verify conditional-op enforcement. A read-only /// open must never mutate the pool it inspects; fsck only reads, so skip it. (Pool meta below is @@ -390,7 +410,8 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// crash-mid-battery still bootstraps cleanly. This closes the "restart poisons a /// partially-erased pool" hole: a missing `_pool_meta` over residual data now fails startup loud /// with zero writes, instead of minting a fresh identity on top of the old objects. - switch (probePoolBootstrapResidual(*backend, layout)) + CasOperation residual_op = bootstrap_requests.admit(); + switch (probePoolBootstrapResidual(residual_op, layout)) { case BootstrapResidual::PoolMetaPresent: break; /// authoritative existing pool; its catalog is mandatory below. @@ -414,7 +435,7 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// relied on to catch a straggler afterwards. Clearing the prefix also destroys the /// durable writer-epoch counter, so a recreation by the SAME server uuid is handed the /// very `(uuid, epoch)` the survivor still holds -- and the two are then indistinguishable - /// to the lease protocol, which reads the survivor's renewal as its own keeper adopting a + /// to the lease protocol, which reads the survivor's renewal as its own renewer adopting a /// refreshed body. The fence only bites when the recreating mount is DISTINGUISHABLE (a /// different server uuid, or a surviving epoch counter): then the survivor's next renewal /// finds a slot it cannot hold and its local fence latches shut. @@ -428,7 +449,8 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// Only on this arm: `EmptyOrProbeOnly` proves there is no slot object to read (a mount /// lease is itself residual), and `PoolMetaPresent` is not a recreation at all -- neither /// pays for the scan. - const std::vector held = probeNonTerminalMountSlots(*backend, layout); + CasOperation slots_op = bootstrap_requests.admit(); + const std::vector held = probeNonTerminalMountSlots(slots_op, layout); if (!held.empty()) { String detail; @@ -458,6 +480,15 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) if (!config.skip_access_check) { + /// The two STORE-LEVEL gates run before the battery writes anything, so a backend this build + /// must refuse is refused without leaving `_probe/` debris behind. They ask the backend + /// directly because an admitted operation deliberately has no route back to it; the predicates + /// come from the same engine the probe operation is admitted from, so they can never be asked + /// of a different backend than the one probed. + Backend & probe_backend = bootstrap_requests.backendForCapabilityPredicates(); + probe_backend.checkPoolPreconditions(); + probe_backend.checkConditionalWriteSingleAttemptSupport(); + /// Give each mount a PER-MOUNT UNIQUE probe key prefix so two servers mounting the SAME /// shared pool concurrently never collide on the (formerly fixed) `/_probe/token` / /// `/_probe/cas` keys. Without this, the loser of the `putIfAbsent` race aborts startup @@ -466,13 +497,14 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// independently. A crashed mount leaves harmless `_probe//...` debris under the `_probe/` /// namespace only (never the content planes) — acceptable. const UInt128 probe_uid = (static_cast(thread_local_rng()) << 64) | thread_local_rng(); - runCapabilityProbe(*backend, config.pool_prefix + "/_probe/" + u128ToHex(probe_uid)); + CasOperation probe_op = bootstrap_requests.admit(); + runCapabilityProbe(probe_op, config.pool_prefix + "/_probe/" + u128ToHex(probe_uid)); } else { - /// skip_access_check: skip the access-check-class probe I/O (store preconditions + the - /// `_probe/` round trip, both folded into runCapabilityProbe above) but NOT the two - /// fail-closed gates below — see `PoolConfig::skip_access_check`. + /// skip_access_check: skip the access-check-class probe I/O (the store-precondition gate and + /// the `_probe/` round trip above) but NOT the two fail-closed gates below — see + /// `PoolConfig::skip_access_check`. /// /// First, whether this backend may skip the battery AT ALL. A generation-dialect (GCS) /// backend may not: the battery is the only thing that proves a token-exact DELETE @@ -492,14 +524,18 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// the narrowly-defined retry path when this opener (or a concurrent opener) completed this step /// but did not reach the pool-meta create. if (initialize_empty_catalog) - CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + { + CasOperation catalog_op = bootstrap_requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(catalog_op, layout); + } /// `allow_mint` = writable open only: a writable `Pool::open` reaches here having just passed the /// zero-write residual proof above, so minting a missing `_pool_meta` is safe. A read-only/observe /// open never ran that proof (and there is no truly-read-only backend — `openPoolView` opens the same /// writable object storage and only sets `read_only`), so it must NEVER mint: an absent meta fails /// closed instead (spec §2 [C4][D2]). + CasOperation meta_op = bootstrap_requests.admit(); PoolMeta meta = PoolMeta::createOrValidate( - *backend, layout, config.blob_header_len, config.gc_shards, config.blob_hash_algo, config.blob_hash_allow_new, + meta_op, layout, config.blob_header_len, config.gc_shards, config.blob_hash_algo, config.blob_hash_allow_new, /*allow_mint=*/!config.read_only); config.gc_shards = meta.gc_shards; const BlobHashAlgo write_algo = config.blob_hash_algo; /// `config` is moved-from just below @@ -549,7 +585,8 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol const ObserveRefCatalog observe_catalog = [s = store.get()]() { - CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*s->pool_backend, s->pool_layout); + CasOperation op = s->gc_requests.admit(); + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, s->pool_layout); snapshot.life_index.throwIfAmbiguous("CAS server-root mount safety"); return snapshot.catalog; }; @@ -560,7 +597,11 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// 2. Owner anchor — IDENTITY (clock-free). A foreign uuid fails closed; an absent owner over a /// non-empty subtree is CORRUPTED_DATA; a fresh empty root is claimed. - claimOwnerOrThrow(*store->pool_backend, store->pool_layout, srid, our_uuid, observe_catalog); + /// Every step of this protocol runs on the OPEN plane. These are bootstrap-control writes: they + /// establish the very right to write, and the mount fence they would otherwise be gated on is + /// either unarmed (a first open) or latched lost (a self-remount, which could then never reclaim). + CasOperation owner_op = store->gc_requests.admit(); + claimOwnerOrThrow(owner_op, store->pool_layout, srid, our_uuid, observe_catalog); /// Wall-clock `now_ms`, hoisted above the writer_epoch allocation below: the absent-epoch /// branch's `DecommissionRecovery` policy needs it to judge a surviving mount's liveness before @@ -585,8 +626,9 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol const EpochMintPolicy epoch_policy = (policy == MountClaimPolicy::NoWait) ? EpochMintPolicy::DecommissionRecovery : EpochMintPolicy::NormalMount; + CasOperation epoch_op = store->gc_requests.admit(); uint64_t writer_epoch = allocateWriterEpoch( - *store->pool_backend, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); + epoch_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); /// 4. Mount lease — LIVENESS. Decide over the current mount object using the wall-clock `now_ms` @@ -622,7 +664,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// Mount-slot writer audit (the "foreign writer" instrument): route every mount-slot /// write/conflict event through the Pool's own sink. The factory installs the configured sink /// before this mount protocol starts, including before either runtime worker can emit. - /// `s` outlives the lambda: the runtime-owned keeper and workers are stopped before `Pool` + /// `s` outlives the lambda: the runtime-owned renewer and workers are stopped before `Pool` /// destruction reaches the event dispatcher. const auto emit_mount_event = [s = store.get()](CasEvent e) { s->emitEvent(std::move(e)); }; @@ -640,7 +682,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// (the lease expired mid-open — e.g. a slow first beat — and a GC round fenced it), that is a /// RECOVERABLE state, not a wedge: a fence costs an epoch, so allocate a fresh writer_epoch and /// re-claim. Bounded so a pathological fence storm still fails closed. The fence can surface two - /// ways: `claimMount` observes an already-fenced own slot (`FencedSelf`), or the keeper's adopt + /// ways: `claimMount` observes an already-fenced own slot (`FencedSelf`), or the renewer's adopt /// races a fence between its GET and CAS (`MountFencedException` from `start()`). /// which certificate of death (if any) justified the reclaim FINALLY adopted below /// (the last iteration's `claim` before `break` -- `claim` itself is loop-scoped). Read after the @@ -656,18 +698,42 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol MountClaimResult claim; if (policy == MountClaimPolicy::WaitForExpiry) { - claim = claimMountAwaitingExpiry( - *store->pool_backend, store->pool_layout, srid, our_uuid, writer_epoch, - [&now_ms]() { return now_ms(); }, [raw] { return raw->bootMsNow(); }, - ttl_ms, poll_interval_ms, sleep_ms, on_wait_start, emit_mount_event); + CasOperation claim_op = store->gc_requests.admit(); + const bool unsafe = store->config.unsafe_remount_no_delay; + if (unsafe) + { + /// One bare attempt first: a slot held by OUR uuid under another epoch is reclaimed at + /// once under the operator's authorization, carrying the exact token this read saw so a + /// slot that moves in between is refused. Every outcome but a claim (an absent slot + /// freshly minted, or a same-epoch refresh) falls through to the ordinary observed path + /// below. + claim = claimMount(claim_op, store->pool_layout, srid, our_uuid, writer_epoch, now_ms(), ttl_ms, + /*proven_dead_incarnation=*/{}, emit_mount_event); + if (claim.kind == MountClaimResult::LiveDoubleStart && claim.etag + && claim.body && claim.body->server_uuid == our_uuid) + claim = claimMount(claim_op, store->pool_layout, srid, our_uuid, writer_epoch, now_ms(), ttl_ms, + {}, emit_mount_event, /*unsafe_reclaim_authorization=*/claim.etag); + } + /// A `FencedSelf`, a foreign-uuid `LiveDoubleStart`/`ForeignOwner`, or a raced + /// `LiveDoubleStart` the authorization above did not cover falls through here and re-runs + /// the bare `claimMount` a second time inside `claimMountAwaitingExpiry`'s own loop; under + /// the knob that means the same conflict is recorded twice in the mount audit stream for + /// one open. The outcome this open ends in is unaffected -- only the audit stream gains a + /// duplicate row, and only when the knob is set. + if (!unsafe || claim.kind != MountClaimResult::Claimed) + claim = claimMountAwaitingExpiry( + claim_op, store->pool_layout, srid, our_uuid, writer_epoch, + [&now_ms]() { return now_ms(); }, [raw] { return raw->bootMsNow(); }, + ttl_ms, poll_interval_ms, sleep_ms, on_wait_start, emit_mount_event); } else { /// NoWait (decommission gate): a single unobserved attempt -- no bounded wait-and-retry /// for a stale-looking lease to lapse. Anything but Claimed/FencedSelf below is refused /// immediately. - claim = claimMount(*store->pool_backend, store->pool_layout, srid, our_uuid, writer_epoch, - now_ms(), ttl_ms, /*proven_dead_token=*/{}, emit_mount_event); + CasOperation claim_op = store->gc_requests.admit(); + claim = claimMount(claim_op, store->pool_layout, srid, our_uuid, writer_epoch, + now_ms(), ttl_ms, /*proven_dead_incarnation=*/{}, emit_mount_event); } if (claim.kind == MountClaimResult::FencedSelf) { @@ -677,46 +743,56 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol "({} recoveries exhausted) — a fresh writer_epoch kept being fenced before we " "could adopt it. This should not persist; investigate GC fence-out timing.", srid, max_fence_recoveries); + CasOperation reallocate_op = store->gc_requests.admit(); writer_epoch = allocateWriterEpoch( - *store->pool_backend, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); + reallocate_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); continue; } if (claim.kind != MountClaimResult::Claimed) { if (policy == MountClaimPolicy::NoWait) + { /// No FORCE variant, no wait-and-observe: the decommission gate treats any live-looking - /// or foreign-owner lease as an immediate refusal. + /// or foreign-owner lease as an immediate refusal. A raced write that observed nobody + /// has no holder to describe, and the lease this server merely proposed is not one -- + /// naming it would send an operator after this very process. + const String holder = claim.body + ? fmt::format("mount lease held by uuid={} epoch={} pid={} hostname={} (expires_at_ms={})", + u128ToHex(claim.body->server_uuid), claim.body->writer_epoch, + claim.body->pid, claim.body->hostname, claim.body->expires_at_ms) + : String("the conditional write that lost the mount slot observed nothing at the key, " + "so the holder is unknown to this server"); throw Exception(ErrorCodes::ABORTED, - "CAS decommission '{}': pool member is alive or contended — mount lease held by " - "uuid={} epoch={} pid={} hostname={} (expires_at_ms={}). Refusing (no FORCE variant " - "exists; stop the server or wait for its lease to lapse).", - srid, u128ToHex(claim.body.server_uuid), claim.body.writer_epoch, claim.body.pid, - claim.body.hostname, claim.body.expires_at_ms); + "CAS decommission '{}': pool member is alive or contended — {}. Refusing (no FORCE " + "variant exists; stop the server or wait for its lease to lapse).", + srid, holder); + } /// LiveDoubleStart (waited out the bound → a live twin) or ForeignOwner → fail closed /// with the actionable, multi-line startup error. throw Exception(ErrorCodes::ABORTED, "{}", mountDoubleStartMessage(srid, claim.body)); } claimed_prior = claim.prior; - /// The mount object now holds OUR live (uuid, epoch) body. `installKeeper` constructs the keeper + /// The mount object now holds OUR live (uuid, epoch) body. `installRenewer` constructs the renewer /// -- which ADOPTS that very (uuid, epoch) slot rather than self-tripping the double-start guard -- /// AND wires its `minActive` build-watermark reader, its event sink, and the fence-coupling /// runtime-owned synchronous renewal driver, all captured on `mount_runtime`. - store->mount_runtime.installKeeper(our_uuid, writer_epoch, now_ms); + store->mount_runtime.installRenewer(our_uuid, writer_epoch, now_ms); try { - claim_anchor_boot_ms = store->mount_runtime.startKeeper(); + claim_anchor_boot_ms = store->mount_runtime.startRenewer(); } catch (const MountFencedException &) { - /// The GC fenced our fresh lease between the keeper's adopt GET and CAS. Recoverable: - /// drop this keeper, take a fresh epoch, and re-claim. + /// The GC fenced our fresh lease between the renewer's adopt GET and CAS. Recoverable: + /// drop this renewer, take a fresh epoch, and re-claim. if (fence_recovery >= max_fence_recoveries) throw; - store->mount_runtime.keeperReset(); + store->mount_runtime.renewerReset(); + CasOperation refenced_epoch_op = store->gc_requests.admit(); writer_epoch = allocateWriterEpoch( - *store->pool_backend, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); + refenced_epoch_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); continue; } @@ -724,9 +800,11 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol } /// A reclaim over a predecessor whose death was NOT proven clean may still have a conditional PUT - /// from that predecessor in flight -- `Fenced` and `UncleanObserved` are exactly the two - /// `MountPriorState`s with no such proof (`Clean`, drained farewell, and `None`, a fresh mount / - /// same-epoch refresh with nothing to hand over, are the proven ones). + /// from that predecessor in flight -- `Fenced`, `UncleanObserved`, and `UncleanUnsafe` are exactly + /// the three `MountPriorState`s with no such proof (`UncleanUnsafe` has no proof at all, not merely + /// no proof of a CLEAN death: it is the operator's explicit `cas_unsafe_remount_no_delay` + /// acceptance of that risk, with no observation behind it). `Clean`, drained farewell, and `None`, a + /// fresh mount / same-epoch refresh with nothing to hand over, are the proven ones. /// An EXHAUSTIVE switch, not a positive allowlist -- a future `MountPriorState` /// enumerator with no proof of clean death must fail the BUILD (a missing `-Wswitch` case), never /// silently fall through to "clean". @@ -748,6 +826,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol break; case MountPriorState::Fenced: case MountPriorState::UncleanObserved: + case MountPriorState::UncleanUnsafe: unclean_reclaim = true; break; } @@ -756,7 +835,10 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol LOG_INFO(getLogger("CasPool"), "Content-addressed mount {} follows a predecessor whose death was not proven clean " "(writer_epoch {}). Opening without a grace period: a still-in-flight conditional PUT from " - "that predecessor is fenced by the recovery seal, whenever it arrives.", srid, writer_epoch); + "that predecessor is fenced by the recovery seal, whenever it arrives.{}", srid, writer_epoch, + claimed_prior == MountPriorState::UncleanUnsafe + ? " (reclaimed without observation under cas_unsafe_remount_no_delay)" + : ""); } /// Arm the local write fence: cache (uuid, epoch) and set the boottime deadline at the claim @@ -770,14 +852,20 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol : claim_anchor_boot_ms + ttl_ms_u - safety_ms; const uint64_t now_boot_ms = store->bootMsNow(); const uint64_t period_ms = static_cast(store->config.mount_renew_period.count()); + const uint64_t envelope_ms = store->config.cas_request_budget.attemptEnvelopeMs(); + const uint64_t two_envelopes_ms = envelope_ms > std::numeric_limits::max() / 2 + ? std::numeric_limits::max() : 2 * envelope_ms; const uint64_t renewal_window_ms = store->config.background_watermark - ? period_ms + store->config.cas_request_budget.attempt_timeout_ms - : store->config.cas_request_budget.attempt_timeout_ms; - /// Preserve one ordinary cadence followed by one physical renewal attempt inside the safe lease - /// window. If that publication horizon was consumed, re-anchor synchronously before opening the - /// fence; the keeper independently retains its per-request deadline checks. + ? (two_envelopes_ms > std::numeric_limits::max() - period_ms + ? std::numeric_limits::max() : period_ms + two_envelopes_ms) + : two_envelopes_ms; + /// Preserve one ordinary cadence followed by one physical renewal attempt (a write and its + /// settlement read: two envelopes) inside the safe lease window. If that publication horizon was + /// consumed, re-anchor synchronously before opening the fence; the renewer independently retains + /// its per-request deadline checks. STRICT, like `CasMountRuntime::admit`: a horizon that fits + /// exactly still starts a renewal the fence would then refuse. const bool renewal_window_fits = now_boot_ms <= safe_deadline - && renewal_window_ms <= safe_deadline - now_boot_ms; + && renewal_window_ms < safe_deadline - now_boot_ms; if (!renewal_window_fits) { /// The claim path outlived the lease TTL: its anchor can no longer authorize an armed fence (a @@ -794,7 +882,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol LOG_WARNING(getLogger("CasPool"), "Content-addressed mount {}: the mount claim consumed the lease TTL ({} ms) before the write " "fence could be armed; re-writing the lease first", srid, ttl_ms_u); - claim_anchor_boot_ms = store->mount_runtime.renewKeeperForStartupOnce(); + claim_anchor_boot_ms = store->mount_runtime.renewRenewerForStartupOnce(); } store->mount_runtime.setLiveWriterEpoch(writer_epoch); store->armMountFence( @@ -806,7 +894,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// Gate the two persistent runtime workers with `background_watermark`: they run only in production /// (`background_watermark` = context != nullptr && !read_only), never in unit tests — which /// drive `renewWatermarkOnce` explicitly and rely on the armed sub-TTL deadline, never on a loop. - /// The synchronous keeper is still started above (it must adopt the mount and arm the fence on + /// The synchronous renewer is still started above (it must adopt the mount and arm the fence on /// every writable open); only the worker pair is conditional. The merged /// heartbeat renews at `mount_renew_period` — one beat now renews the lease and the floor. if (store->config.background_watermark) @@ -839,10 +927,17 @@ PoolPtr Pool::openForDecommission(BackendPtr backend, PoolConfig config, const S /// fenced/terminated/clean-farewell lease reclaims; a live lease refuses immediately (no bounded /// observation wait -- see `mountWritable`). Owner anchor absent + mount absent = nothing to /// decommission. - std::optional victim_uuid = readOwnerUuid(*backend, layout, victim_srid); + /// The open plane: this factory impersonates the victim to take its mount, so there is no lease of + /// ours to be gated on until the claim below establishes one. `config.retry_sleep_fn` must travel + /// with `config.boot_ms_fn`: a retry loop bound to a frozen test clock that only a fake sleep + /// advances would otherwise retry forever against this engine's default REAL sleep, which never + /// calls it -- the deadline it measures against would never appear to elapse. + CasRequests bootstrap_requests(backend, Fence::open(), config.boot_ms_fn, config.retry_sleep_fn); + CasOperation owner_op = bootstrap_requests.admit(); + std::optional victim_uuid = readOwnerUuid(owner_op, layout, victim_srid); if (!victim_uuid) { - if (const auto mount = backend->get(layout.mountKey(victim_srid))) + if (const auto mount = owner_op.read(layout.mountKey(victim_srid), Retry::standard())) victim_uuid = decodeMountLease(mount->bytes).server_uuid; /// partial hand-cleanup: adopt from the lease else throw Exception(ErrorCodes::BAD_ARGUMENTS, @@ -857,8 +952,9 @@ PoolPtr Pool::openForDecommission(BackendPtr backend, PoolConfig config, const S /// `_pool_meta` must already be present. It never bootstraps: `allow_mint=false` so an absent meta /// (a partially-erased pool whose owner anchor survives) fails closed with INVALID_STATE rather than /// minting a fresh identity here (spec §2 [C4][D2]). + CasOperation meta_op = bootstrap_requests.admit(); PoolMeta meta = PoolMeta::createOrValidate( - *backend, layout, config.blob_header_len, config.gc_shards, config.blob_hash_algo, config.blob_hash_allow_new, + meta_op, layout, config.blob_header_len, config.gc_shards, config.blob_hash_algo, config.blob_hash_allow_new, /*allow_mint=*/false); config.gc_shards = meta.gc_shards; const BlobHashAlgo write_algo = config.blob_hash_algo; /// `config` is moved-from just below @@ -904,7 +1000,7 @@ Pool::~Pool() } }; - /// 1. Stop and join both persistent mount-runtime workers before draining or releasing the keeper. + /// 1. Stop and join both persistent mount-runtime workers before draining or releasing the renewer. guarded([this] { if (config.teardown_phase1_throw_for_test) @@ -912,7 +1008,7 @@ Pool::~Pool() mount_runtime.stopBackgroundWorkers(); }, "CAS pool teardown: stopping background workers"); - /// 2. The farewell marker the keeper's `release` writes is a certificate that no in-flight ref-log + /// 2. The farewell marker the renewer's `release` writes is a certificate that no in-flight ref-log /// conditional PUT from this incarnation can land after it. A successor treats it as proof of a /// clean death (`MountPriorState::Clean`, no observation wait needed). Writing it without an actual /// drain would be a protocol-safety bug: an uncertain PUT this incarnation is still resolving could @@ -928,7 +1024,7 @@ Pool::~Pool() if (config.teardown_phase2_throw_for_test) config.teardown_phase2_throw_for_test(); const bool ref_lanes_drained = ref_ledger.drainRefLanesForShutdown( - config.cas_request_budget.attempt_timeout_ms + config.cas_request_budget.lease_safety_margin_ms); + config.cas_request_budget.attemptEnvelopeMs() + config.cas_request_budget.lease_safety_margin_ms); drained = ref_lanes_drained && !writerCleanupDutiesPending(); }, "CAS pool teardown: draining ref lanes"); @@ -988,14 +1084,23 @@ bool Pool::tryDispatchDetached(std::function task) return true; } -bool Pool::stopAndDrainDetachedWork(uint64_t deadline_ms) +void Pool::beginTeardown() noexcept { { std::lock_guard lock(detached_work->mutex); - detached_work->stopping = true; + detached_work->stopping.store(true, std::memory_order_release); } detached_work->cv.notify_all(); +} +bool Pool::teardownBegun() const noexcept +{ + return detached_work->stopping.load(std::memory_order_acquire); +} + +bool Pool::stopAndDrainDetachedWork(uint64_t deadline_ms) +{ + beginTeardown(); std::unique_lock lock(detached_work->mutex); return detached_work->cv.wait_for(lock, std::chrono::milliseconds(deadline_ms), [this] { return detached_work->in_flight == 0; }); @@ -1009,14 +1114,12 @@ uint64_t Pool::detachedWorkInFlight() const bool Pool::detachedWorkStoppingForTest() const { - std::lock_guard lock(detached_work->mutex); - return detached_work->stopping; + return teardownBegun(); } -void Pool::setDetachedDrainDeadlineBudgetForTest(uint64_t attempt_timeout_ms, uint64_t lease_safety_margin_ms) +void Pool::setDetachedDrainDeadlineBudgetForTest(const CasRequestBudget & budget) { - config.cas_request_budget.attempt_timeout_ms = attempt_timeout_ms; - config.cas_request_budget.lease_safety_margin_ms = lease_safety_margin_ms; + config.cas_request_budget = budget; } void Pool::forgetDisk(const std::function & stop_and_join_gc, const String & reason) @@ -1032,13 +1135,19 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri /// Idempotent: an already-terminal `Vanished` pool (a second FORGET, or a pool that naturally vanished /// as replaced) is already the terminal truth — nothing to force, and re-running the teardown - /// would double-retire the keeper. `IdentityLost`/`TransientNotLive`/`Live` all proceed (FORGET is + /// would double-retire the renewer. `IdentityLost`/`TransientNotLive`/`Live` all proceed (FORGET is /// their escape hatch). Reading `isVanished()` here without the lock is safe: only a terminal transition /// sets it, terminal states are absorbing, and a natural transition that wins concurrently below merely /// makes our own `enterVanished` a no-op (first terminal transition wins). if (mount_runtime.isVanished()) return; + /// This protocol deliberately does NOT arm the open plane. The GC join at (3+4) would be bounded + /// by it, but an already-latched self-remount completes its current step before the loop bails at + /// (5a), and that step's pool-identity probe is admitted on the open plane -- an arm here refuses + /// it, so the reclaim `finishTeardown` is written to override could never happen. Server shutdown + /// arms instead: it joins the same scheduler with no remount step to preserve. + /// /// (1) Publish the terminal-intent latch FIRST (spec §5). The runtime stops latching remounts and /// the remount loop bails at its next step boundary, so every join below is bounded to one step + one /// backend timeout. @@ -1064,13 +1173,13 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri /// that window re-arms the local fence (`lost = false`). Now that the remount worker is JOINED and can /// never run again, re-latch the fence so the terminal `mayMutate() == false` holds regardless of any /// such raced reclaim. Idempotent; the durable mount lease the reclaim wrote is retired by the - /// `finishTeardown` below (it operates on whatever keeper is current — the reclaimed one). + /// `finishTeardown` below (it operates on whatever renewer is current — the reclaimed one). mount_runtime.tripMountLost(); /// (5b) Drain the ref lanes (bounded by one attempt's budget + safety margin) to learn whether a clean /// farewell is EARNED — exactly the `~Pool` rule. const bool ref_lanes_drained = ref_ledger.drainRefLanesForShutdown( - config.cas_request_budget.attempt_timeout_ms + config.cas_request_budget.lease_safety_margin_ms); + config.cas_request_budget.attemptEnvelopeMs() + config.cas_request_budget.lease_safety_margin_ms); const bool drained = ref_lanes_drained && !writerCleanupDutiesPending(); /// (3+5c) Retire the merged heartbeat: a clean-release farewell ONLY if the lanes provably drained, @@ -1079,10 +1188,10 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri mount_runtime.finishTeardown(drained); /// The pool object OUTLIVES this FORGET (it stays registered, `Vanished(forgotten)`, until DROP/restart), - /// so `~Pool` will re-run the same teardown. Drop the keeper now so that later teardown finds none and - /// skips it: `MountLeaseKeeper::release` is admitted only from `Active`, so a keeper already released - /// here must not be released again. `keeperReset` is safe now: both keeper-driving workers are joined. - mount_runtime.keeperReset(); + /// so `~Pool` will re-run the same teardown. Drop the renewer now so that later teardown finds none and + /// skips it: `MountLeaseRenewer::release` is admitted only from `Active`, so a renewer already released + /// here must not be released again. `renewerReset` is safe now: both renewer-driving workers are joined. + mount_runtime.renewerReset(); /// (6) Publish the terminal state + WARN, under remount serialization — matching the natural-transition /// contract. Every pool thread is already joined, so taking `remount_mutex` here cannot self-deadlock. @@ -1223,8 +1332,37 @@ bool Pool::tryRemountOnce() return false; { step = "pool_identity_probe"; - const LifecycleGate gate = probePoolLifecycleGate( - *pool_backend, pool_layout, config.server_root_id, meta.pool_id, meta.blob_header_len); + /// The open plane: this runs with the mount fence latched lost -- that is what a remount is + /// recovering from -- so an operation admitted under the fence could never issue the probe. + /// + /// Admitted with NO liveness predicate. Refusing the probe once the pool is already terminal + /// would be circular -- this probe is what establishes terminality -- and it would make the + /// verdicts below unreachable in exactly the states they exist for, the mid-FORGET `Replaced` + /// bail among them. A predicate is also the wrong instrument for ending it: the engine samples + /// one before the first request and reports a refusal as a THROWN fence loss, which is not how + /// a remount step reports anything. Both sentinel reads are `once`, so the probe is at most two + /// physical requests and cannot outlast a shutdown. + CasOperation probe_op = gc_requests.admit(); + LifecycleGate gate{LifecycleGateVerdict::StayTransient, {}}; + try + { + gate = probePoolLifecycleGate( + probe_op, pool_layout, config.server_root_id, meta.pool_id, meta.blob_header_len); + } + catch (...) + { + /// The same contract as the startup-protocol steps below: a remount attempt reports failure + /// by returning false, never by throwing at whoever called it. A probe that could not reach + /// the store proved nothing, so the pool stays where it was and the loop retries. + try + { + error = getCurrentExceptionMessage(/*with_stacktrace*/ false); + } + catch (...) // NOLINT(bugprone-empty-catch) + { + } + return false; + } switch (gate.verdict) { case LifecycleGateVerdict::Recover: @@ -1275,13 +1413,14 @@ bool Pool::tryRemountOnce() } /// The same startup protocol as Pool::open steps 2-4, as a FRESH incarnation (the old one is - /// dead by the fence-out contract and its keeper never re-mints). Open THROWS on any failure + /// dead by the fence-out contract and its renewer never re-mints). Open THROWS on any failure /// (startup is fail-closed); the remount RETURNS false instead — the recovery loop retries. try { const ObserveRefCatalog observe_catalog = [this]() { - CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*pool_backend, pool_layout); + CasOperation op = gc_requests.admit(); + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, pool_layout); snapshot.life_index.throwIfAmbiguous("CAS server-root remount safety"); return snapshot.catalog; }; @@ -1290,20 +1429,30 @@ bool Pool::tryRemountOnce() step = "ref_catalog_observe"; (void)observe_catalog(); step = "owner_claim"; - claimOwnerOrThrow(*pool_backend, pool_layout, srid, our_uuid, observe_catalog); + /// The open plane throughout: the mount fence is latched lost here (starting the renewer does + /// not clear it), so an operation admitted under the fence could never make the claim that + /// re-establishes it. + CasOperation owner_op = gc_requests.admit(); + claimOwnerOrThrow(owner_op, pool_layout, srid, our_uuid, observe_catalog); step = "writer_epoch_allocate"; + CasOperation epoch_op = gc_requests.admit(); const uint64_t writer_epoch = allocateWriterEpoch( - *pool_backend, pool_layout, srid, EpochMintPolicy::NormalMount, 0, observe_catalog); + epoch_op, pool_layout, srid, EpochMintPolicy::NormalMount, 0, observe_catalog); result_writer_epoch = writer_epoch; /// Mount-slot writer audit: `this` is already fully open (setEventSink ran long ago), so /// unlike the initial `open`, every event fired below reaches the real sink immediately. const auto emit_mount_event = [this](CasEvent e) { emitEvent(std::move(e)); }; - const auto sleep_ms = [](uint64_t ms) { std::this_thread::sleep_for(std::chrono::milliseconds(ms)); }; + /// Routes through `mount_runtime.waitSleep` (which itself routes through `config.wait_sleep_fn` + /// when a test injected one) rather than a bare `sleep_for` directly, so a test intercepting + /// `wait_sleep_fn` observes every wait a self-remount can block on, exactly like `Pool::open`'s + /// own observation poll above. + const auto sleep_ms = [this](uint64_t ms) { mount_runtime.waitSleep(ms); }; step = "mount_claim"; + CasOperation claim_op = gc_requests.admit(); const MountClaimResult claim = claimMountAwaitingExpiry( - *pool_backend, pool_layout, srid, our_uuid, writer_epoch, + claim_op, pool_layout, srid, our_uuid, writer_epoch, now_ms, [this] { return bootMsNow(); }, ttl_ms, poll_interval_ms, sleep_ms, [&srid](const MountLease & held, uint64_t threshold_ms) { @@ -1338,15 +1487,15 @@ bool Pool::tryRemountOnce() /// build's own tests the moment someone reuses an epoch across a remount. chassert(writer_epoch > mount_runtime.liveWriterEpoch()); - /// The persistent renewal worker is parked before this callback is entered, so keeper + /// The persistent renewal worker is parked before this callback is entered, so renewer /// replacement cannot race any synchronous lease operation. - step = "keeper_install"; - mount_runtime.installKeeper(our_uuid, writer_epoch, now_ms); - step = "keeper_start"; - uint64_t remount_anchor_boot_ms = mount_runtime.startKeeper(); + step = "renewer_install"; + mount_runtime.installRenewer(our_uuid, writer_epoch, now_ms); + step = "renewer_start"; + uint64_t remount_anchor_boot_ms = mount_runtime.startRenewer(); /// Re-establish the ref-protocol incarnation BEFORE re-arming the fence. Order is load-bearing: - /// Starting the keeper does NOT clear `lost`, so the fence stays closed here and no append/publish can race the + /// Starting the renewer does NOT clear `lost`, so the fence stays closed here and no append/publish can race the /// swap. /// 1. Bump the live epoch so every subsequent `allocateRefTxnId` sorts strictly above any older /// (dead-incarnation or twin) durable log. Do this BEFORE `armMountFence` so there is no window @@ -1380,18 +1529,24 @@ bool Pool::tryRemountOnce() : remount_anchor_boot_ms + ttl_ms - safety_ms; const uint64_t now_boot_ms = mount_runtime.bootMsNow(); const uint64_t period_ms = static_cast(config.mount_renew_period.count()); + const uint64_t envelope_ms = config.cas_request_budget.attemptEnvelopeMs(); + const uint64_t two_envelopes_ms = envelope_ms > std::numeric_limits::max() / 2 + ? std::numeric_limits::max() : 2 * envelope_ms; const uint64_t renewal_window_ms = config.background_watermark - ? period_ms + config.cas_request_budget.attempt_timeout_ms - : config.cas_request_budget.attempt_timeout_ms; + ? (two_envelopes_ms > std::numeric_limits::max() - period_ms + ? std::numeric_limits::max() : period_ms + two_envelopes_ms) + : two_envelopes_ms; + /// STRICT, like the open path and `CasMountRuntime::admit`: a horizon that fits exactly still + /// starts a renewal the fence would then refuse. const bool renewal_window_fits = now_boot_ms <= safe_deadline - && renewal_window_ms <= safe_deadline - now_boot_ms; + && renewal_window_ms < safe_deadline - now_boot_ms; if (!renewal_window_fits) { - step = "keeper_redo"; - remount_anchor_boot_ms = mount_runtime.renewKeeperForRemountOnce(); + step = "renewer_redo"; + remount_anchor_boot_ms = mount_runtime.renewRenewerForRemountOnce(); } - /// No-throw commit section: publish the fence and lifecycle only after epoch, keeper, recovery + /// No-throw commit section: publish the fence and lifecycle only after epoch, renewer, recovery /// cancellation, and ref-runtime quiescence are complete. step = "arm_fence"; mount_runtime.armMountFence( @@ -1458,7 +1613,7 @@ void Pool::enqueueWriterCleanupDuty( } catch (...) { - /// The build deliberately remains active. Advancing `min_active` after losing the only cleanup + /// The build deliberately remains active. Advancing `min_active_build_sequence` after losing the only cleanup /// duty would make an uncertain owner grant look dead; pinning the floor until process exit is /// the safe failure direction, and successor recovery handles the durable remnant. writer_cleanup_queue_failed.store(true, std::memory_order_release); @@ -1600,56 +1755,6 @@ BlobLocation Pool::locate(const ManifestEntry & entry) const return manifest_reader.locate(entry); } -namespace -{ -/// a tolerant, read-only peek at the -/// `cas_ref_log` TEXT object (codecs-v3 phase 3) WITHOUT `decodeRefLogTxn`'s expected-value cross-check -/// -- the whole point of this diagnostic is that the body is NOT expected to match this key's identity. -/// It `openObject`s the stored `.zst`, skips the header line, and reads `ns`/`we`/`rs` off the meta -/// line (`we`/`rs` are decimal u64 strings). Never validates the header `v`, never reads past the meta -/// line (the ops are irrelevant to identifying the writer), and swallows any truncation/garbage: this -/// is a background diagnostic only, never a decode anything else depends on. -struct ForeignRefLogHeaderPeek -{ - String ns; - uint64_t writer_epoch = 0; - uint64_t ref_sequence = 0; -}; - -std::optional peekForeignRefLogHeader(const String & bytes) -{ - try - { - const String text = openObject(FormatId::RefLog, bytes); - ReadBufferFromMemory in(text.data(), text.size()); - const uint64_t line_cap = traitsFor(FormatId::RefLog).line_cap; - readLine(in, line_cap, "cas_ref_log"); /// header line -- skip - const String meta = readLine(in, line_cap, "cas_ref_log"); - ReadBufferFromMemory m(meta.data(), meta.size()); - JsonObjectReader r(m, KeyStrictness::Tolerant, "cas_ref_log"); - ForeignRefLogHeaderPeek peek; - bool saw_ns = false; - bool saw_we = false; - bool saw_rs = false; - String key; - while (r.nextKey(key)) - { - if (key == "ns") { peek.ns = r.readString(); saw_ns = true; } - else if (key == "we") { peek.writer_epoch = r.readU64String(); saw_we = true; } - else if (key == "rs") { peek.ref_sequence = r.readU64String(); saw_rs = true; } - else r.skipUnknown(key); - } - if (!saw_ns || !saw_we || !saw_rs) - return std::nullopt; - return peek; - } - catch (...) - { - return std::nullopt; - } -} -} - void Pool::reportImpossibleInterference(const String & key, const String & reason, const std::optional & offending_ns) { @@ -1676,7 +1781,7 @@ void Pool::reportImpossibleInterference(const String & key, const String & reaso /// requests -- never the caller's thread, and never blocking this call's own return. try { - /// The lease owns the pool reference for this task's lifetime; capturing one here as well would + /// The lease owns the pool reference until it releases it; capturing one here as well would /// put it outside the lease's release ordering. const bool dispatched = tryDispatchDetached([this, key](DetachedStopToken token) { @@ -1685,7 +1790,10 @@ void Pool::reportImpossibleInterference(const String & key, const String & reaso return; try { - const auto got = pool_backend->get(key); + /// The open plane: this diagnostic runs after the fence was deliberately tripped a few + /// lines above, and a stop request ends it through the operation's own liveness. + CasOperation op = gc_requests.admit([token] { return !token.stopping(); }); + const auto got = op.read(key, Retry::standard()); if (!got) { LOG_ERROR(getLogger("CasPool"), @@ -1693,7 +1801,7 @@ void Pool::reportImpossibleInterference(const String & key, const String & reaso "time the background diagnostic GET ran", key); return; } - if (const auto peek = peekForeignRefLogHeader(got->bytes)) + if (const auto peek = peekRefLogMeta(got->bytes)) LOG_ERROR(getLogger("CasPool"), "CAS anomaly diagnostics: offending object at '{}' ({} bytes) decodes as a ref-log " "header: namespace='{}', writer_epoch={}, ref_sequence={}", @@ -1732,7 +1840,8 @@ uint64_t Pool::currentGcRound() const /// Read `gc/state` once (no CAS loop — a point-in-time read is sufficient; a concurrent /// GC advance only makes the returned round larger, which is strictly more conservative for the /// `precommitAdd` self-floor). Returns 0 when absent (pool never GC'd — no round to floor to). - const auto state_bytes = pool_backend->get(pool_layout.gcStateKey()); + CasOperation op = gc_requests.admit(); + const auto state_bytes = op.read(pool_layout.gcStateKey(), Retry::standard()); if (!state_bytes) return 0; return decodeGcState(state_bytes->bytes).round; @@ -1769,7 +1878,10 @@ NamespaceListing Pool::listNamespaces(const String & prefix) /// opaque life id and therefore cannot mint a namespace during discovery. std::unordered_set found; std::vector skipped; - const CasRefCatalog::Snapshot cut = CasRefCatalog::read(*pool_backend, pool_layout); + /// The mount plane, like every other content read: enumerating a namespace on a mount whose lease + /// is gone must be refused, not answered. + CasOperation catalog_op = mount_requests.admit(); + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(catalog_op, pool_layout); for (const CatalogEntry & entry : cut.catalog.entries) { try @@ -1795,7 +1907,9 @@ std::vector Pool::listMirroredChildren(const String & prefix) /// Namespace children come from the catalog. `roots/` is still listed for loose mountpoint files, /// whose logical paths retain path identity. std::unordered_set children; - const CasRefCatalog::Snapshot cut = CasRefCatalog::read(*pool_backend, pool_layout); + /// The mount plane: this is a content enumeration, the same class as `listNamespaces` above. + CasOperation catalog_op = mount_requests.admit(); + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(catalog_op, pool_layout); for (const CatalogEntry & entry : cut.catalog.entries) { if (!entry.ns.string().starts_with(prefix)) @@ -1808,27 +1922,20 @@ std::vector Pool::listMirroredChildren(const String & prefix) } const String roots_full = pool_layout.rootsPrefix() + prefix; + CasOperation roots_op = mount_requests.admit(); + roots_op.forEachListedKey(roots_full, [&](const ListedKey & listed) { - String cursor; - while (true) + const String & key = listed.key; + if (key.starts_with(roots_full)) { - ListPage page = pool_backend->list(roots_full, cursor, /*limit*/ 1000); - for (const ListedKey & listed : page.keys) - { - const String & key = listed.key; - if (!key.starts_with(roots_full)) - continue; - const std::string_view rest(key.data() + roots_full.size(), key.size() - roots_full.size()); - const size_t slash = rest.find('/'); - const std::string_view seg = slash == std::string_view::npos ? rest : rest.substr(0, slash); - if (!seg.empty()) - children.emplace(seg); - } - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; + const std::string_view rest(key.data() + roots_full.size(), key.size() - roots_full.size()); + const size_t slash = rest.find('/'); + const std::string_view seg = slash == std::string_view::npos ? rest : rest.substr(0, slash); + if (!seg.empty()) + children.emplace(seg); } - } + return true; + }, Retry::standard()); return {children.begin(), children.end()}; } @@ -1839,7 +1946,22 @@ std::vector Pool::listMirroredChildren(const String & prefix) void Pool::setCasRetrySleepForTest(std::function sleep_fn) { - ref_ledger.setCasRetrySleepForTest(std::move(sleep_fn)); + /// All three planes, not just the ledger's: a test that replaces the retry sleep must not be left + /// with a real one on the plane the site under test happens to use. + farewell_requests.setSleepFnForTest(sleep_fn); + ref_ledger.setCasRetrySleepForTest(sleep_fn); + /// `CasRequests` falls back to the engine's plain sleep for an empty argument -- which is neither + /// the mount plane's nor the open plane's default. Re-install both, so clearing the seam cannot + /// leave a parked renewal held for a whole capped backoff, or the open plane deaf to a teardown. + gc_requests.setSleepFnForTest(sleep_fn ? sleep_fn : openPlaneSleepFn()); + mount_requests.setSleepFnForTest(sleep_fn ? std::move(sleep_fn) : mountPlaneSleepFn()); +} + +void Pool::setCasRequestNowFnForTest(std::function now_fn) +{ + mount_requests.setNowFnForTest(now_fn); + farewell_requests.setNowFnForTest(now_fn); + gc_requests.setNowFnForTest(std::move(now_fn)); } void Pool::setRefRecoveryRetrySleepForTest( @@ -1946,19 +2068,9 @@ size_t Pool::wedgedRefLaneCount() return ref_ledger.wedgedRefLaneCount(); } -CasWriteOutcome Pool::stagingPutIfAbsent(std::string_view key, std::string_view bytes, Token * out_token) -{ - return ref_ledger.stagingPutIfAbsent(key, bytes, out_token); -} - -CasOverwriteResult Pool::stagingConditionalOverwrite(std::string_view key, std::string_view bytes, const Token & expected) -{ - return ref_ledger.stagingConditionalOverwrite(key, bytes, expected); -} - -CasOverwriteResult Pool::stagingPutIfAbsentMutable(std::string_view key, std::string_view bytes) +WriteResult Pool::stagingPutIfAbsent(const String & key, const String & bytes) { - return ref_ledger.stagingPutIfAbsentMutable(key, bytes); + return ref_ledger.stagingPutIfAbsent(key, bytes); } void Pool::cancelInflightBuildsForNamespace(const RootNamespace & ns) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index 69eba5061fd3..cbde11caa87c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -9,7 +9,7 @@ #include #include #include -#include +#include #include #include #include @@ -75,6 +75,10 @@ struct PoolConfig /// was count-bounded only (16384 entries) — decoded manifests carry inline bytes, so the worst /// case was multi-GB. 0 disables decode caching (every read decodes fresh — diagnostic mode). uint64_t manifest_decode_cache_bytes = 128ULL << 20; + /// Byte bound for the hot-key lane's cache of last known objects (the catalog today). 0 disables + /// the cache and every catalog write reads first. 16 MiB: the catalog is under 1 MiB, and the + /// bound exists so a later opt-in of per-namespace keys has one. + uint64_t hot_key_cache_bytes = 16ULL << 20; /// How many superseded snapshot generations to retain. After committing /// generation G, generations <= G - this are pruned (bounded per round). 0 = keep ALL /// (debug/forensics — replay GC's in-degree view as-of a past round). Default 3 = the safety @@ -88,6 +92,8 @@ struct PoolConfig /// per completed GC round; the delete budget separately bounds exact-token destructive work. uint64_t manifest_sweep_list_budget_keys = 1000; uint64_t manifest_sweep_delete_budget_keys = 100; + /// Keys per batch delete request for the write-once families; tests lower it to exercise chunk boundaries. + uint64_t gc_bulk_delete_chunk_keys = 1000; /// Per-round blob-deletion work envelope: caps how many entries the fold's graduation /// (condemned -> delete_pending) and redelete (exact-token delete of a prior delete_pending row) /// arms move out of the durable retired pipeline in one round. Excess entries are carried @@ -168,6 +174,11 @@ struct PoolConfig /// feedback_ca_gc_never_throw_on_404) and `Gc::runRegularRound` waits for the round's whole batch /// before the round's single gc/state CAS, so the meta writes are durable before that CAS commits. uint64_t gc_meta_pool_size = 16; + /// Bounded pool size for the fold's read-ahead of checkpoints, ref logs, manifest bodies and + /// zero-candidate HEADs. Every decision stays on the round thread, in the order it always ran; + /// only the fetch overlaps. `1` issues no read-ahead at all and is the sequential round, request + /// for request. + uint64_t gc_read_concurrency = 16; /// Tests drive `renewWatermarkOnce` explicitly; gates both persistent runtime workers. bool background_watermark = false; /// Installed on the pool before a writable mount can start its runtime-owned workers. @@ -205,6 +216,11 @@ struct PoolConfig /// `mount_renew_period` (default ttl/3) so a healthy mount renews well before expiry. std::chrono::milliseconds mount_lease_ttl_ms{30000}; std::chrono::milliseconds mount_renew_period{10000}; /// = ttl/3 by default + + /// `cas_unsafe_remount_no_delay`: reclaim a same-uuid, different-epoch, uncertified mount slot at + /// once, with no token-stability observation at all. Unsafe whenever two processes can hold the + /// same server_uuid; see the setting's own description for the exact risk. + bool unsafe_remount_no_delay = false; bool read_only = false; /// observe-only open: skip the mutating capability probe; reads only /// Boot-time "start now, fix later": skip the access-check-class part of the capability probe @@ -234,6 +250,15 @@ struct PoolConfig /// boot clock (`Pool::bootMs`); injected by tests to drive the fence deadline deterministically. std::function boot_ms_fn = {}; + /// The inter-attempt sleep for the mount, farewell and GC request planes (`mount_requests`, + /// `farewell_requests`, `gc_requests`), installed at their CONSTRUCTION -- before this `Pool` has + /// claimed or read anything. Empty = each plane's own production default (an interruptible real + /// sleep for the mount and GC planes, `CasRequests`'s own real sleep for the farewell plane). A test + /// that also freezes `boot_ms_fn` must supply a matching sleep here: a retry loop bound to a clock + /// that only moves when this function is called would otherwise retry forever against a REAL sleep + /// that never calls it, because the deadline it measures against never appears to elapse. + std::function retry_sleep_fn = {}; + /// Test hook for open/remount waits: `Pool::waitSleep` -- the mount-claim observation loop's poll -- /// routes through this function when set instead of a real `std::this_thread::sleep_for`, so a test /// observes every wait without actually blocking. Empty (the production default) sleeps for real. @@ -405,13 +430,22 @@ class Pool : public std::enable_shared_from_this ~Pool(); bool tryDispatchDetached(std::function task); + /// Marks this pool as being torn down. The open request plane refuses every further admission and + /// wakes a retry sleep on it, and no new detached task is accepted. Idempotent, and it frees, + /// nulls and swaps nothing: it can be called before any lock a teardown takes, so a GC round + /// holding such a lock is refused at its next request instead of being waited out. + void beginTeardown() noexcept; + bool teardownBegun() const noexcept; bool stopAndDrainDetachedWork(uint64_t deadline_ms); uint64_t detachedWorkInFlight() const; uint64_t detachedWorkInFlightForTest() const { return detachedWorkInFlight(); } bool detachedWorkStoppingForTest() const; /// Test-only: shortens the metadata-storage teardown deadline without changing any request path. /// Call before dispatching detached work; production configuration remains immutable after open. - void setDetachedDrainDeadlineBudgetForTest(uint64_t attempt_timeout_ms, uint64_t lease_safety_margin_ms); + /// Replaces the whole budget (not just `attempt_timeout_ms`) because the drain deadline is read + /// from `attemptEnvelopeMs()`, which also folds in `connect_timeout_cap_ms` -- a caller that only + /// overrode the attempt timeout would silently keep whatever connect cap the pool froze at open. + void setDetachedDrainDeadlineBudgetForTest(const CasRequestBudget & budget); /// ---- per-server watermark surface ---- /// process_epoch: random nonzero per Pool (process). GC checks epoch EQUALITY, never ordering. @@ -427,7 +461,7 @@ class Pool : public std::enable_shared_from_this uint64_t minActive(); /// Test/assertion accessor for the next-to-allocate build_seq under the lock. uint64_t peekNextBuildSeq(); - /// Renew the merged heartbeat once (bump seq, refresh min_active from the live callback, stamp a + /// Renew the merged heartbeat once (bump seq, refresh min_active_build_sequence from the live callback, stamp a /// fresh expires_at_ms). The build-watermark floor rides this beat. In production this is driven by /// the background renewer (background_watermark). void renewWatermarkOnce(); @@ -441,7 +475,7 @@ class Pool : public std::enable_shared_from_this /// or foreign observation; the gated mutate chokepoints then fail closed. void tripMountLost(); /// Refresh the write-fence deadline (a CLOCK_BOOTTIME-milliseconds instant; release). - /// keeper renew calls this on success. + /// renewer renew calls this on success. void setMountDeadline(uint64_t deadline_boot_ms); /// Arm the fence at startup: set (uuid, epoch, deadline), clear `lost`. void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); @@ -449,6 +483,11 @@ class Pool : public std::enable_shared_from_this { mount_runtime.setArmMountFenceInterpositionHookForTest(std::move(hook)); } + /// Swap the observation-wait hook after open -- see `CasMountRuntime::setWaitSleepForTest`. + void setWaitSleepForTest(std::function fn) + { + mount_runtime.setWaitSleepForTest(std::move(fn)); + } /// The fence clock: CLOCK_BOOTTIME in milliseconds (includes VM-suspend time, unlike /// CLOCK_MONOTONIC — see `MountFence`). Consults the injected `config.boot_ms_fn` if set (tests), /// otherwise `bootMs`. @@ -508,7 +547,7 @@ class Pool : public std::enable_shared_from_this /// next step boundary, bounding the joins below); (2) trip the local fence (the deliberate /// decommission act, allowed on a live disk); (3+4) stop the GC scheduler via `stop_and_join_gc` — /// injected because the scheduler is owned above the Pool, a no-op in contexts that run none — and stop - /// + join both persistent workers; (5) drain the ref lanes (bounded) and retire the keeper WITHOUT an + /// + join both persistent workers; (5) drain the ref lanes (bounded) and retire the renewer WITHOUT an /// unearned clean farewell (the lease expires by observation unless the lanes provably drained); then /// (6) publish `Vanished(forgotten)` carrying `reason` (the [D5] message with the operator's decommission /// timestamp). Idempotent: an already-`Vanished` pool returns immediately (first terminal transition @@ -555,15 +594,16 @@ class Pool : public std::enable_shared_from_this { return ref_ledger.confirmExactRef(ns, ref_name, manifest_ref); } - /// Read the single immutable part manifest named by `id`. Derives the key via CasLayout::manifestKey, - /// decodes the body, and fails CLOSED: a committed ref naming a missing body throws FILE_DOESNT_EXIST - /// (INV-NO-DANGLE surfaced on the read path); a body whose `ref` ≠ id.ref (refMatchesBody) or whose + /// Read the single immutable part manifest named by `id`. Serves the cached decode when the id + /// is cached (no request); otherwise derives the key via CasLayout::manifestKey, GETs and decodes + /// the body, and fails CLOSED: an absent body throws FILE_DOESNT_EXIST (a committed ref naming a + /// missing body is a dangling reference); a body whose `ref` ≠ id.ref (refMatchesBody) or whose /// `root_namespace_id` ≠ id.root_namespace (manifestNamespaceMatches) throws CORRUPTED_DATA — the - /// ref is addressing the wrong object, or a cross-namespace dangle. Token-gated decode cache below. + /// ref is addressing the wrong object, or a cross-namespace dangle. Id-keyed decode cache below. PartManifest readManifest(const ManifestId & id); - /// Identical to `readManifest` (same mandatory HEAD, same fail-closed validation, same decode - /// cache) but returns the SHARED immutable decode the manifest cache holds — no per-call copy. - /// The wiring read path uses this variant. + /// Identical to `readManifest` (same fail-closed validation, same decode cache) but returns the + /// SHARED immutable decode the manifest cache holds — no per-call copy. The wiring read path + /// uses this variant. std::shared_ptr readManifestShared(const ManifestId & id); BlobLocation locate(const ManifestEntry & entry) const; /// Blob placement only std::map listRefs(const RootNamespace & ns); @@ -707,23 +747,32 @@ class Pool : public std::enable_shared_from_this const PoolConfig & poolConfig() const { return config; } const PoolMeta & poolMeta() const { return meta; } const Layout & layout() const { return pool_layout; } - Backend & backend() { return *pool_backend; } + + /// ---- the three request planes ---- + /// The mount plane: the durable writes whose right to land IS this node's mount lease. An + /// operation admitted here is refused the moment the fence trips, is re-armed under a fresh lease + /// incarnation, or runs out of room before the lease expires. + CasRequests & mountRequests() { return mount_requests; } + /// The farewell plane, on an open fence -- shared with the mount-lease renewer's own claim/adopt, + /// not only its release: a self-remount claims with the fence already latched lost, so gating the + /// claim on the fence could never reclaim, and refusing the farewell because the mount fence has + /// already run down would leave the slot looking live until GC fences it out. + CasRequests & farewellRequests() { return farewell_requests; } + /// The open-fence plane: GC, the offline tools, this pool's own reads, and the bootstrap-control + /// claims. None of them hold a mount lease -- the claims are what ESTABLISHES one, so gating them + /// on the fence would make a self-remount, which runs with the fence latched lost, unable ever to + /// reclaim. + CasRequests & openRequests() { return gc_requests; } /// The owning `BackendPtr` itself (not just a reference into it): the decommission slot-retirement /// decommission step (`CasDecommission.cpp`) must keep the backend alive across `admin.reset()` -- the graceful /// close that stamps the mount's farewell -- to physically delete the control objects afterward. A /// bare `Backend &` from `backend()` would dangle the instant the owning `Pool` is destroyed. BackendPtr poolBackendPtr() const { return pool_backend; } - /// Staging PUT surface for `PartWriteTxn`: the methods wrap the ref-ledger's retry controller - /// AND the ref-lane fence predicate, so `PartWriteTxn` reaches neither directly (the `friend` is gone). - /// Behavior-identical to the previously-inlined controller+fence at CasPartWriteTxn.cpp stageManifest / - /// mutable marker writes; thin delegates to `ref_ledger`. - CasWriteOutcome stagingPutIfAbsent(std::string_view key, std::string_view bytes, Token * out_token = nullptr); - /// Same retry/fence policy as `stagingPutIfAbsent`, for a mutable If-Match overwrite. - CasOverwriteResult stagingConditionalOverwrite(std::string_view key, std::string_view bytes, const Token & expected); - /// Same retry/fence policy as `stagingPutIfAbsent`, for a mutable marker where an existing - /// DIFFERENT value at the key is a normal Conflict outcome, not corruption. - CasOverwriteResult stagingPutIfAbsentMutable(std::string_view key, std::string_view bytes); + /// Staging write surface for `PartWriteTxn`: thin delegate onto the ref ledger, so a staging write + /// is admitted on the same plane and under the same policy as a ref-lane write and `PartWriteTxn` + /// reaches it neither directly. + WriteResult stagingPutIfAbsent(const String & key, const String & bytes); /// CAS mixed-algo pools: /// the NODE-LOCAL algo this Pool mints NEW content with (`PoolConfig::blob_hash_algo` -- never @@ -758,10 +807,10 @@ class Pool : public std::enable_shared_from_this void setLiveWriterEpochForTest(uint64_t writer_epoch) { mount_runtime.setLiveWriterEpoch(writer_epoch); } /// Self-remount after a GC fence-out (liveness counterpart of the fence-out safety rule): the - /// OLD incarnation may never write again (the keeper never re-mints), but a FRESH incarnation — + /// OLD incarnation may never write again (the renewer never re-mints), but a FRESH incarnation — /// durable writer_epoch bump + mount reclaim + re-armed write fence — is exactly what a server /// restart would create, so a live server may create it in place. Runs the same claim machinery as - /// `Pool::open`. Orchestration stays here; the owned mount primitives it drives (keeper swap, + /// `Pool::open`. Orchestration stays here; the owned mount primitives it drives (renewer swap, /// epoch bump, fence re-arm) live on `mount_runtime`. Returns false (and changes nothing durable /// beyond the epoch bump) when the /// mount cannot be claimed (foreign owner / a genuinely live twin) — the caller retries. Safe to @@ -775,7 +824,7 @@ class Pool : public std::enable_shared_from_this /// Test seam: how many times `scheduleRemount` has been ENTERED, counted /// unconditionally as its very first statement. This increments even under the default /// `background_watermark = false` (no worker exists; a - /// test never pays for a real self-remount attempt racing this Pool's own still-live keeper, which + /// test never pays for a real self-remount attempt racing this Pool's own still-live renewer, which /// -- confirmed while building this seam -- reliably takes 30+ seconds per call and is not something /// a fast unit test should be driving). Positively pins that a production call site (e.g. /// `reportImpossibleInterference`) actually invoked `scheduleRemount`, as opposed to merely observing @@ -835,7 +884,7 @@ class Pool : public std::enable_shared_from_this }; /// The writable-mount startup tail shared by `open` and `openForDecommission`: owner claim → - /// writer_epoch → mount claim (+fence-recovery loop) → `MountLeaseKeeper` start → watermark + /// writer_epoch → mount claim (+fence-recovery loop) → `MountLeaseRenewer` start → watermark /// anchor. `our_uuid` is the identity to mount as -- `config.server_id` for a normal open, the /// victim's owner uuid for decommission (impersonation). `policy` changes only what happens when /// the mount claim does not resolve `Claimed`/`FencedSelf`: `WaitForExpiry` observes a stale- @@ -864,10 +913,9 @@ class Pool : public std::enable_shared_from_this /// `cancel_inflight_builds` callback. void cancelInflightBuildsForNamespace(const RootNamespace & ns); - /// Delegate to `mount_runtime`: the write fence moved there. pre-attempt fence check: extends - /// `mayMutate` with the REMAINING budget check -- an attempt is not even started unless there is - /// enough of the mount lease left for one more attempt_timeout plus the lease safety margin. Passed - /// as `fence_ok` to every `CasRequestController` call the ref-log writer path makes. + /// Delegate to `mount_runtime`: the write fence moved there. Extends `mayMutate` with the REMAINING + /// budget check -- work is not started unless there is enough of the mount lease left for TWO full + /// attempt envelopes (a write and its settlement read) plus the safety margin. bool refAppendFenceOk() const; /// incidental-detection reaction for a foreign-interference @@ -1018,13 +1066,24 @@ class Pool : public std::enable_shared_from_this ref_ledger.setSnapshotBeforeCkptCasHookForTest(std::move(hook)); } - /// Test-only: replace the request controller's inter-attempt backoff sleep (e.g. with a no-op) — - /// for tests that drive a persistent conditional-write fault to budget exhaustion through a fully - /// wired Pool/disk and must not serve the production capped-exponential sleeps for real (see - /// `CasRequestController::setSleepFnForTest`). Call before driving traffic; empty restores the - /// real sleep. + /// Test-only: replace the inter-attempt backoff sleep (e.g. with a clock-advancing no-op) on all + /// three request planes and on ref-table recovery, for tests that drive a persistent write fault to + /// exhaustion through a fully wired Pool/disk and must not serve the production sleeps for real. + /// Call before driving traffic. On the three request planes an empty function restores each plane's + /// own default, the mount plane's interruptible sleep included. + /// + /// It does NOT bound a reissue the engine refuses to start: the engine's inter-attempt backoff is + /// jittered and drawn before the sleep, and admission compares that drawn duration against the + /// lease. A test that needs a reissue admitted, or refused, deterministically has to arrange the + /// clock, not the sleep. void setCasRetrySleepForTest(std::function sleep_fn); + /// Test-only: replace the request engine's clock on all three planes. A test driving a PERSISTENT + /// transient fault must run the retry window on a clock it advances; the sleep seam alone cannot + /// bound it, because a read the engine keeps reissuing is bounded by the policy deadline and the + /// deadline is read from this clock. + void setCasRequestNowFnForTest(std::function now_fn); + /// Test-only: replace only ref-table recovery's token-aware retry delay seam. void setRefRecoveryRetrySleepForTest( std::function &)> sleep_fn); @@ -1037,6 +1096,16 @@ class Pool : public std::enable_shared_from_this /// `refQueuePendingForTest`; used to assert the baton is not stranded on a pre-tenure fault. bool refLeaderActiveForTest(const RootNamespace & ns) { return ref_ledger.refLeaderActiveForTest(ns); } + /// Test-only: the carved-item mirror's size for `ns` (see `CasRefLedger::refCarvedForTest`). + size_t refCarvedForTest(const RootNamespace & ns) { return ref_ledger.refCarvedForTest(ns); } + + /// Test-only: whether the carved item named `ref_name` is already completed (see + /// `CasRefLedger::refCarvedItemDoneForTest`). + bool refCarvedItemDoneForTest(const RootNamespace & ns, const String & ref_name) + { + return ref_ledger.refCarvedItemDoneForTest(ns, ref_name); + } + /// Test seam: how many concurrent `ensureRefTableRecovered` callers for `ns` are /// PARKED right now waiting on the leader's in-flight recovery (see `RefTableRuntime:: /// recovery_waiters_for_test`) -- lets a test `yield()`-poll for "a second caller actually reached @@ -1097,10 +1166,44 @@ class Pool : public std::enable_shared_from_this return std::forward(mutation)(); } + /// The mount plane's inter-attempt sleep: interruptible, so a parked or stopping renewal is not + /// held for a whole capped backoff. Named rather than inlined because the test seam has to be able + /// to put it back. + std::function mountPlaneSleepFn() + { + return [this](uint64_t ms) { mount_runtime.sleepInterruptibly(ms); }; + } + + /// The open plane's inter-attempt sleep: woken by `beginTeardown`, so a retry backing off on the + /// GC plane cannot hold a teardown for a whole capped backoff. A predicate wait, so the detached + /// tasks' own completions -- which notify the same variable -- do not cut a sleep short. Named + /// for the same reason as `mountPlaneSleepFn`: the test seam has to be able to put it back. + std::function openPlaneSleepFn() + { + return [this](uint64_t ms) + { + std::unique_lock lock(detached_work->mutex); + detached_work->cv.wait_for(lock, std::chrono::milliseconds(ms), + [this] { return detached_work->stopping.load(std::memory_order_acquire); }); + }; + } + BackendPtr pool_backend; PoolConfig config; PoolMeta meta; + /// The pool's write lane for keys several of its writers share, declared before the three planes + /// that carry a pointer to it, so it outlives every operation they admit. `mutable` for the same + /// reason the planes are. + mutable CasHotKeys hot_keys; + + /// The three planes' engines, declared before every component that is handed one and after the + /// config they take their clock from. `mutable` because issuing a request is not a change to the + /// pool: a `const` observer still has to read the store. + mutable CasRequests mount_requests; + mutable CasRequests farewell_requests; + mutable CasRequests gc_requests; + std::shared_ptr detached_work = std::make_shared(); mutable std::mutex writer_cleanup_mutex; @@ -1141,7 +1244,7 @@ class Pool : public std::enable_shared_from_this /// `mount_runtime.finishTeardown` exactly as before. CasRefLedger ref_ledger; /// The mount / write-fence / build-watermark / self-remount runtime, extracted - /// from Pool. Owns the `MountLeaseKeeper`, the local `MountFence`, the per-server + /// from Pool. Owns the `MountLeaseRenewer`, the local `MountFence`, the per-server /// build watermark (`process_epoch` + the `builds_mutex`-guarded seq/registry) and its in-flight-build /// map, the live-incarnation `live_writer_epoch`, the unclean-epoch high-water-mark, and the /// persistent renewal and remount workers (with one driver mutex/condition pair). Injected with backend/layout diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPoolMeta.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPoolMeta.cpp index 388d2a110521..fddb27103311 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPoolMeta.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPoolMeta.cpp @@ -1,9 +1,10 @@ #include -#include +#include #include #include #include #include +#include namespace DB { @@ -62,69 +63,71 @@ String joinAlgoNames(const std::vector & algos_used) } /// The relaxed admission check (replaces an earlier fail-close that required the single pool algo to match): -/// `pm`/`token` are the most-recently-read `_pool_meta` state (present, decoded, valid). Already a -/// member of `algos_used` => OK, no write (steady state). Not a member and `!allow_new` => -/// `BAD_ARGUMENTS` (the pool is never touched). Not a member and `allow_new` => CAS-union `config_algo` -/// into `algos_used` (recomputed from the FRESH value on every retry -- union-only, so there is no -/// ABA) and raises `min_reader_generation` to THIS build's own floor (`G_BUILD`, `CasFormat.h`) in -/// the SAME write (first registration of a schema-3-bearing algo also raises -/// `min_reader_generation` -- a build that cannot decode schema-3 settlement state has an OLDER -/// `G_BUILD` and is correctly refused by the startup gate once a future generation bump lands here). -/// On a CAS conflict, re-read and retry the whole decision (a concurrent admitter may have unioned a -/// DIFFERENT algo, or the very one we wanted, in the meantime). -PoolMeta admitOrValidate( - Backend & backend, const String & key, PoolMeta pm, Token token, - BlobHashAlgo config_algo, bool allow_new) +/// re-reads `key` itself (via `readModifyWrite`'s own observe) rather than trusting a snapshot the +/// caller already holds, so a caller that only knows the key is present -- never a stale decode -- +/// can ask this to settle admission. Already a member of `algos_used` => OK, no write (steady state, +/// `decide` declines). Not a member and `!allow_new` => `BAD_ARGUMENTS` (the pool is never touched). +/// Not a member and `allow_new` => CAS-union `config_algo` into `algos_used` (recomputed from the +/// FRESH value on every retry -- union-only, so there is no ABA) and raises `min_reader_generation` to +/// THIS build's own floor (`G_BUILD`, `CasFormat.h`) in the SAME write (first registration of a +/// schema-3-bearing algo also raises `min_reader_generation` -- a build that cannot decode schema-3 +/// settlement state has an OLDER `G_BUILD` and is correctly refused by the startup gate once a future +/// generation bump lands here). A concurrent admitter's own union is folded in the same way, since +/// `decide` runs again against whatever `readModifyWrite` observes on retry. +PoolMeta admitOrValidate(CasOperation & op, const String & key, BlobHashAlgo config_algo, bool allow_new) { - for (;;) - { - if (isAlgoAdmittedIn(pm, config_algo)) - return pm; - - if (!allow_new) - throwNotAdmitted(pm, config_algo); - - PoolMeta next = pm; - next.algos_used.push_back(static_cast(config_algo)); - std::sort(next.algos_used.begin(), next.algos_used.end()); - next.min_reader_generation = G_BUILD; - - const CasResult res = backend.casPut(key, encodePoolMeta(next), token); - if (res.outcome == CasOutcome::Committed) - return next; - - auto fresh = backend.get(key); - if (!fresh) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CAS pool meta: '{}' vanished mid-admission (conflicting write then a concurrent delete)", key); - pm = decodePoolMeta(fresh->bytes); - token = fresh->token; - /// loop: re-evaluate membership against the FRESH pm (never re-encode the stale `next`) - } + PoolMeta observed; + WriteResult result = op.readModifyWrite( + key, + [&](const std::optional & current) -> std::optional + { + if (!current) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CAS pool meta: '{}' vanished mid-admission (conflicting write then a concurrent delete)", key); + observed = decodePoolMeta(current->bytes); + + if (isAlgoAdmittedIn(observed, config_algo)) + return std::nullopt; + if (!allow_new) + throwNotAdmitted(observed, config_algo); + + observed.algos_used.push_back(static_cast(config_algo)); + std::sort(observed.algos_used.begin(), observed.algos_used.end()); + observed.min_reader_generation = G_BUILD; + return encodePoolMeta(observed); + }, + Retry::standard()); + orThrow(std::move(result), fmt::format("CAS pool meta admission on '{}'", key)); + return observed; } } PoolMeta PoolMeta::createOrValidate( - Backend & backend, const Layout & layout, uint64_t blob_header_len, uint64_t gc_shards, + CasOperation & op, const Layout & layout, uint64_t blob_header_len, uint64_t gc_shards, BlobHashAlgo blob_hash_algo, bool allow_new, bool allow_mint) { /// The passed config is the caller's responsibility — reject bad values before any I/O. validatePoolBlobHeaderLen(blob_header_len, ErrorCodes::BAD_ARGUMENTS, "pool meta"); if (gc_shards == 0) throw Exception(ErrorCodes::BAD_ARGUMENTS, "CAS pool meta: gc_shards must be >= 1"); - /// Defense against a garbage `static_cast` past the caller's own boundary: `blobHashAlgoName` - /// throws BAD_ARGUMENTS for anything `BlobHashAlgo` does not actually admit. + /// `blobHashAlgoName` rejects an out-of-range `BlobHashAlgo` with `LOGICAL_ERROR`: a programming + /// error that aborts debug and sanitizer builds, not an input-validation fence. blobHashAlgoName(blob_hash_algo); const String key = layout.poolMetaKey(); /// Present => the pool is authoritative; ignore the passed config's blob_header_len and run the - /// flag-gated admission check rather than the old single-value fail-close. - if (auto existing = backend.get(key)) + /// flag-gated admission check rather than the old single-value fail-close. The steady state (the + /// configured algo is already admitted) is decided from THIS read, so the common open costs one + /// GET; only a union or a `!allow_new` refusal falls through to `admitOrValidate`, which re-reads + /// the key itself as part of its own conditional write. + if (auto existing = op.read(key, Retry::standard())) { PoolMeta pm = decodePoolMeta(existing->bytes); - return admitOrValidate(backend, key, std::move(pm), existing->token, blob_hash_algo, allow_new); + if (isAlgoAdmittedIn(pm, blob_hash_algo)) + return pm; + return admitOrValidate(op, key, blob_hash_algo, allow_new); } /// Absent => mint a pool id and try to create the object with `algos_used = {blob_hash_algo}`. @@ -132,7 +135,7 @@ PoolMeta PoolMeta::createOrValidate( /// build at all), so the reader-generation floor is stamped at THIS build's `G_BUILD` at /// creation, not left at 0. /// - /// BOOTSTRAP GATE (spec §2 [C4][D2]): minting is permitted ONLY on the verified bootstrap path. A + /// BOOTSTRAP GATE: minting is permitted ONLY on the verified bootstrap path. A /// non-bootstrap caller (a read-only/observe open, `openForDecommission`) passes `allow_mint=false` /// and fails closed here — never minting a fresh identity outside that path (an observe scan that /// minted would poison the next writable mount's residual check). The residual EMPTINESS proof itself @@ -152,17 +155,17 @@ PoolMeta PoolMeta::createOrValidate( pm.min_reader_generation = G_BUILD; pm.algos_used = {static_cast(blob_hash_algo)}; - if (backend.casPut(key, encodePoolMeta(pm), /*expected*/ std::nullopt).outcome == CasOutcome::Committed) + WriteResult result = op.create(key, encodePoolMeta(pm), Retry::standard()); + if (std::holds_alternative(result)) return pm; /// Lost the race: the winner's object MUST be present now. The loser UNIONS its algo via the SAME /// flag-gated admission path as a reopen, instead of the old unconditional fail-close. - auto winner = backend.get(key); - if (!winner) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CAS pool meta: create-if-absent reported Conflict but '{}' is absent on re-read", key); - PoolMeta winner_pm = decodePoolMeta(winner->bytes); - return admitOrValidate(backend, key, std::move(winner_pm), winner->token, blob_hash_algo, allow_new); + if (std::holds_alternative(result)) + return admitOrValidate(op, key, blob_hash_algo, allow_new); + + orThrow(std::move(result), fmt::format("CAS pool meta creation on '{}'", key)); + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS pool meta: create-if-absent on '{}' neither committed, conflicted, nor threw", key); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.cpp index b9b952a7321a..d4e12fc65c97 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.cpp @@ -1,10 +1,13 @@ #include -#include +#include +#include #include #include +#include #include #include #include +#include namespace DB { @@ -22,70 +25,91 @@ namespace { -CasRefCatalog::Snapshot readOptionalForBootstrap(Backend & backend, const Layout & layout) +/// The one verdict for an absent mandatory catalog, so the read entry point and the mutation's own +/// `decide` cannot drift apart on what absence means. +[[noreturn]] void throwMandatoryCatalogAbsent(const String & key) { - const auto got = backend.get(layout.refCatalogKey()); + throw Exception(ErrorCodes::CORRUPTED_DATA, + "Mandatory CAS ref catalog '{}' is absent -- refusing to interpret opaque life " + "objects as an empty ownership universe", + key); +} + +/// Every non-committed alternative of a catalog write, as the exception its meaning already implies. +/// `Declined` cannot reach here: `create` never declines, and no `decide` in this file returns +/// nothing to write. +[[noreturn]] void throwCatalogWriteFailure(WriteResult result, const String & what) +{ + orThrow(std::move(result), what); + throw Exception(ErrorCodes::LOGICAL_ERROR, "{}: the write was declined, which this call cannot produce", what); +} + +CasRefCatalog::Snapshot readOptionalForBootstrap(CasOperation & op, const Layout & layout, + const Retry & policy = Retry::standard()) +{ + const std::optional got = op.read(layout.refCatalogKey(), policy); if (!got) { RefCatalog empty; return CasRefCatalog::Snapshot{ - .catalog = empty, .token = std::nullopt, .life_index = CatalogLifeIndex(empty)}; + .catalog = empty, .etag = std::nullopt, .life_index = CatalogLifeIndex(empty)}; } RefCatalog catalog = decodeRefCatalog(got->bytes); return CasRefCatalog::Snapshot{ - .catalog = catalog, .token = got->token, .life_index = CatalogLifeIndex(catalog)}; + .catalog = catalog, .etag = got->etag, .life_index = CatalogLifeIndex(catalog)}; } } -CasRefCatalog::Snapshot CasRefCatalog::read(Backend & backend, const Layout & layout) +CasRefCatalog::Snapshot CasRefCatalog::read(CasOperation & op, const Layout & layout, const Retry & policy) { - Snapshot snapshot = readOptionalForBootstrap(backend, layout); - if (!snapshot.token) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "Mandatory CAS ref catalog '{}' is absent -- refusing to interpret opaque life " - "objects as an empty ownership universe", - layout.refCatalogKey()); + Snapshot snapshot = readOptionalForBootstrap(op, layout, policy); + if (!snapshot.etag) + throwMandatoryCatalogAbsent(layout.refCatalogKey()); return snapshot; } -CasRefCatalog::Snapshot CasRefCatalog::initializeEmptyForNewPool(Backend & backend, const Layout & layout) +CasRefCatalog::Snapshot CasRefCatalog::initializeEmptyForNewPool(CasOperation & op, const Layout & layout) { + const String key = layout.refCatalogKey(); RefCatalog empty; const String canonical_empty = encodeRefCatalog(empty); - const PutResult put = backend.putIfAbsent(layout.refCatalogKey(), canonical_empty); - if (put.outcome == PutOutcome::Done) - return Snapshot{.catalog = empty, .token = put.token, .life_index = CatalogLifeIndex(empty)}; - - /// A second opener can win after both proved the prefix empty. Decode its exact object before - /// accepting the race; conflict is never a license to continue with an assumed empty catalog or - /// arbitrary decoded body. - const auto got = backend.get(layout.refCatalogKey()); - if (!got) + WriteResult result = op.create(key, canonical_empty, Retry::standard()); + if (const auto * committed = std::get_if(&result)) + return Snapshot{.catalog = empty, .etag = committed->etag, .life_index = CatalogLifeIndex(empty)}; + + /// A second opener can win after both proved the prefix empty. The refused precondition was + /// settled by an exact read, so the winner's object is decoded from what that read observed; + /// a conflict is never a license to continue with an assumed empty catalog or an arbitrary body. + const auto * conflict = std::get_if(&result); + if (!conflict) + throwCatalogWriteFailure(std::move(result), fmt::format("CAS ref catalog '{}' bootstrap create", key)); + + const auto * occupant = std::get_if(&conflict->seen); + if (!occupant) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS ref catalog '{}' disappeared after bootstrap create conflict", - layout.refCatalogKey()); - RefCatalog catalog = decodeRefCatalog(got->bytes); - if (!catalog.entries.empty() || got->bytes != canonical_empty) + "CAS ref catalog '{}' disappeared after bootstrap create conflict", key); + RefCatalog catalog = decodeRefCatalog(occupant->bytes); + if (!catalog.entries.empty() || occupant->bytes != canonical_empty) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS ref catalog '{}' conflicts with bootstrap's required canonical empty catalog", - layout.refCatalogKey()); - return Snapshot{.catalog = std::move(catalog), .token = got->token, .life_index = CatalogLifeIndex(empty)}; + "CAS ref catalog '{}' conflicts with bootstrap's required canonical empty catalog", key); + return Snapshot{.catalog = std::move(catalog), .etag = occupant->etag, + .life_index = CatalogLifeIndex(empty)}; } std::optional CasRefCatalog::lifeIfCataloged( - Backend & backend, const Layout & layout, const RootNamespace & ns) + CasOperation & op, const Layout & layout, const RootNamespace & ns) { - const Snapshot snap = read(backend, layout); + const Snapshot snap = read(op, layout); for (const CatalogEntry & entry : snap.catalog.entries) if (entry.ns.string() == ns.string() && entry.state != NsState::Creating) return snap.life_index.resolve(entry.incarnation); return std::nullopt; } -std::vector CasRefCatalog::liveUniverse(Backend & backend, const Layout & layout) +std::vector CasRefCatalog::liveUniverse(CasOperation & op, const Layout & layout) { - const Snapshot snap = read(backend, layout); + const Snapshot snap = read(op, layout); snap.life_index.throwIfAmbiguous("CAS live namespace discovery"); std::vector universe; universe.reserve(snap.catalog.entries.size()); @@ -101,52 +125,108 @@ std::vector CasRefCatalog::liveUniverse(Backend & backend, cons namespace { -/// Live-lock brake, the same shape and for the same reason as `publishCkpt`'s/`allocateWriterEpoch`'s -/// on their own contended token-CAS singletons: the catalog is ONE object mutated by every lifecycle -/// transition of every namespace in the pool, so persistent contention is a real, not theoretical, -/// exit condition to plan for. +/// Thrown from inside a `casUpdate` `mutate` closure to signal a refusal that must STOP the attempt +/// rather than be treated as a refused precondition to retry: `casUpdateImpl` propagates whatever +/// `mutate` throws straight out, uncaught, which is exactly the behavior these three need. Retrying +/// any of them against a freshly re-read catalog would just re-decide against an entry that is, by +/// definition, no longer `observed` -- token-exactness means the FIRST mismatch is final, not a reason +/// to loop. Each is caught by its own exact type right where it is thrown; deriving from +/// `std::exception` is only so the throw itself is well-formed, never so a caller catches these by +/// base class. +struct CatalogFenceMovedMarker : std::exception {}; +struct CatalogEntryMismatchMarker : std::exception {}; +struct CatalogCreatorStillLiveMarker : std::exception {}; + +/// Live-lock brake for the ONE loop below that is written by hand rather than driven by the engine: +/// the catalog is a single object mutated by every lifecycle transition of every namespace in the +/// pool, so persistent contention is a real, not theoretical, exit condition to plan for. +/// +/// It bounds ATTEMPTS, and is the SECONDARY bound: the loop freezes one `Retry` before it starts and +/// every verb of every iteration shares that absolute deadline, so wall-clock time is already bounded +/// by one standard window. This cap exists so a call that somehow converges on neither still ends. constexpr size_t kMaxCatalogCasAttempts = 100; /// Shared body of `casUpdate`/`casAdmitEntry`. `encode` turns a freshly `mutate`d candidate into the /// bytes to write: the plain path just grammar-checks (`encodeRefCatalog`), the admitting path also -/// runs both admission predicates (`checkCatalogAdmission`) first. Retries on `Conflict` against a -/// FRESH read, exactly like `PoolMeta::admitOrValidate` -- never re-encoding the stale candidate. +/// runs both admission predicates (`checkCatalogAdmission`) first. A refused precondition re-runs +/// `mutate` against the resolve read's fresh body -- the one the pool's hot-key lane remembers and +/// the next hold starts from -- never re-encoding the stale candidate. RefCatalog casUpdateImpl( - Backend & backend, const Layout & layout, + CasOperation & op, const Layout & layout, const std::function & mutate, - const std::function & encode) + const std::function & encode, + const Retry & policy = Retry::standard()) { const String key = layout.refCatalogKey(); - CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + /// The candidate the LAST `decide` produced, which is the one the engine wrote: every earlier one + /// belongs to an attempt whose precondition was refused. + std::optional written; - for (size_t attempt = 0; attempt < kMaxCatalogCasAttempts; ++attempt) + const auto decide = [&](const std::optional & current) -> std::optional { - snap.life_index.throwIfAmbiguous("CAS ref catalog mutation"); - RefCatalog candidate = mutate(snap.catalog); - const String bytes = encode(candidate); - const CasResult res = backend.casPut(key, bytes, snap.token); - if (res.outcome == CasOutcome::Committed) - return candidate; - - snap = CasRefCatalog::read(backend, layout); - /// `read` treats authoritative absence after a conflict as corruption. Therefore no retry - /// can turn a vanished mandatory catalog into a one-update replacement authority. - } + /// Absence of the mandatory catalog is corruption, not a fresh bootstrap. Refusing here is + /// what stops any mutation from replacing every other namespace with a one-update catalog. + if (!current) + throwMandatoryCatalogAbsent(key); + const RefCatalog durable = decodeRefCatalog(current->bytes); + CatalogLifeIndex(durable).throwIfAmbiguous("CAS ref catalog mutation"); + RefCatalog candidate = mutate(durable); + String bytes = encode(candidate); + written = std::move(candidate); + return bytes; + }; - throwCasWriteRetryLater(fmt::format( - "CAS ref catalog '{}' did not converge after {} attempts", key, kMaxCatalogCasAttempts)); + /// One hold at a time per pool on this key, from the pool's last known catalog when the lane holds + /// one. The loop is this function's: a `Conflict` is a lost race against another server (the lane + /// never conflicts with itself), repaid after the flat jitter, or after the growing schedule when + /// the conflict settled a transport fault. Under a single-attempt policy the first `Conflict` is + /// the answer, as the engine's own verb answers it. + const Retry frozen = op.freeze(policy); + uint32_t settled_faults = 0; + for (;;) + { + WriteResult result = op.hotKeys().submit(key, op, frozen, decide); + if (const auto * conflict = std::get_if(&result); conflict && !frozen.single_attempt) + { + op.pause(conflict->any_ambiguous ? Retry::backoff(++settled_faults) : Retry::conflictBackoff()); + continue; + } + /// The fence can be lost in two places and both mean the same to a lifecycle caller: inside + /// `decide`, which throws the marker itself, and between two attempts, where the engine + /// notices it first and no further `decide` runs. Normalising the second onto the first is + /// what keeps "the fence moved" a returned outcome rather than an exception. + if (const auto * gave_up = std::get_if(&result); gave_up && gave_up->why == GaveUp::Why::FenceLost) + throw CatalogFenceMovedMarker{}; + if (!std::holds_alternative(result)) + throwCatalogWriteFailure(std::move(result), fmt::format("CAS ref catalog '{}' update", key)); + return std::move(*written); + } } -/// Thrown from inside a `casUpdate` `mutate` closure to signal a refusal that must STOP the attempt -/// rather than be treated as a `Conflict` to retry: `casUpdateImpl` propagates whatever `mutate` -/// throws straight out, uncaught, which is exactly the behavior these three need. Retrying any of them -/// against a freshly re-read catalog would just re-decide against an entry that is, by definition, no -/// longer `observed` -- token-exactness means the FIRST mismatch is final, not a reason to loop. -/// Each is caught by its own exact type right where it is thrown; deriving from `std::exception` -/// is only so the throw itself is well-formed, never so a caller catches these by base class. -struct CatalogFenceMovedMarker : std::exception {}; -struct CatalogEntryMismatchMarker : std::exception {}; -struct CatalogCreatorStillLiveMarker : std::exception {}; +/// `casUpdate`'s own guard, shared with the two lifecycle callers that need to catch the fence marker +/// the public entry point translates away. +std::function identityPreserving( + const std::function & mutate) +{ + return [&mutate](const RefCatalog & current) -> RefCatalog + { + RefCatalog candidate = mutate(current); + if (candidate.entries.size() != current.entries.size()) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CasRefCatalog::casUpdate cannot add or delete catalog entries -- use casAdmitEntry, " + "deleteCompletedRemoving, or cancelStalledCreating"); + for (size_t i = 0; i < current.entries.size(); ++i) + { + if (candidate.entries[i].ns != current.entries[i].ns + || candidate.entries[i].incarnation != current.entries[i].incarnation) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "CasRefCatalog::casUpdate cannot replace catalog identity at row {} -- namespace " + "and incarnation are immutable outside the narrow admission/deletion APIs", + i); + } + return candidate; + }; +} /// Two `thread_local_rng` draws composed into a `UInt128`, the same pattern already used throughout /// this tree to mint build ids and incarnation tags (`CasPartWriteTxn.cpp`'s `mintU128`, @@ -190,6 +270,13 @@ struct CatalogEntryAlreadyPresentMarker : std::exception {}; /// Empty (no-op) in production, mirroring every other `*_hook_for_test` in this tree. std::function create_namespace_step1_pre_read_hook_for_test; +/// Fires once, synchronously, right before `createNamespace`'s own pre-check read -- the window in +/// which a sibling opener of the SAME namespace can complete an entire birth (or begin a removal) that +/// this call's pre-check then observes. Lets a test land that interleaving deterministically instead of +/// relying on real thread scheduling. Empty (no-op) in production, mirroring every other +/// `*_hook_for_test` in this tree. +std::function create_namespace_pre_check_hook_for_test; + /// Step 1 of `createNamespace`, split out so it can recheck presence on EVERY catalog read this loop /// performs (the first one, and any `Conflict` retry's re-read), not only the snapshot-in-time read /// `createNamespace` itself already did before calling in. That single upfront read cannot see a @@ -198,7 +285,8 @@ std::function create_namespace_step1_pre_read_hook_for_test; /// canonical-order/no-duplicate grammar check abort the process with `LOGICAL_ERROR` for what is, at /// this call site only, an ordinary race outcome. RefCatalog createNamespaceStep1( - Backend & backend, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry) + CasOperation & op, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry, + const Retry & policy) { /// Moved into a local before invoking, not called on the global directly: a hook that reassigns /// `create_namespace_step1_pre_read_hook_for_test` from inside its own body (a test driving a @@ -222,42 +310,36 @@ RefCatalog createNamespaceStep1( next.entries.insert(it, entry); return next; }; - return casUpdateImpl(backend, layout, mutate, + return casUpdateImpl(op, layout, mutate, [&entry, gc_shards, &layout](const RefCatalog & c) { return checkCatalogAdmission(c, gc_shards, layout, entry.ns); - }); + }, + policy); } } RefCatalog CasRefCatalog::casUpdate( - Backend & backend, const Layout & layout, const std::function & mutate) + CasOperation & op, const Layout & layout, const std::function & mutate) { - const auto identity_preserving_mutate = [&](const RefCatalog & current) -> RefCatalog + try { - RefCatalog candidate = mutate(current); - if (candidate.entries.size() != current.entries.size()) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CasRefCatalog::casUpdate cannot add or delete catalog entries -- use casAdmitEntry, " - "deleteCompletedRemoving, or cancelStalledCreating"); - for (size_t i = 0; i < current.entries.size(); ++i) - { - if (candidate.entries[i].ns != current.entries[i].ns - || candidate.entries[i].incarnation != current.entries[i].incarnation) - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CasRefCatalog::casUpdate cannot replace catalog identity at row {} -- namespace " - "and incarnation are immutable outside the narrow admission/deletion APIs", - i); - } - return candidate; - }; - return casUpdateImpl( - backend, layout, identity_preserving_mutate, [](const RefCatalog & c) { return encodeRefCatalog(c); }); + return casUpdateImpl( + op, layout, identityPreserving(mutate), [](const RefCatalog & c) { return encodeRefCatalog(c); }); + } + catch (const CatalogFenceMovedMarker &) + { + /// The marker is this file's private signal; a caller outside it gets the exception class every + /// other admission refusal raises. + throwCasTransientUnavailable( + fmt::format("CAS ref catalog '{}' update", layout.refCatalogKey()), + "mount fence tripped: the update was admitted under an incarnation this node no longer holds"); + } } RefCatalog CasRefCatalog::casAdmitEntry( - Backend & backend, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry) + CasOperation & op, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry) { if (entry.state == NsState::Removing) throw Exception(ErrorCodes::LOGICAL_ERROR, @@ -279,17 +361,27 @@ RefCatalog CasRefCatalog::casAdmitEntry( next.entries.insert(it, entry); return next; }; - return casUpdateImpl(backend, layout, mutate, - [&entry, gc_shards, &layout](const RefCatalog & c) - { - return checkCatalogAdmission(c, gc_shards, layout, entry.ns); - }); + try + { + return casUpdateImpl(op, layout, mutate, + [&entry, gc_shards, &layout](const RefCatalog & c) + { + return checkCatalogAdmission(c, gc_shards, layout, entry.ns); + }); + } + catch (const CatalogFenceMovedMarker &) + { + /// The marker is this file's private signal; a caller outside it gets the exception class every + /// other admission refusal raises. + throwCasTransientUnavailable( + fmt::format("CAS ref catalog '{}' update", layout.refCatalogKey()), + "mount fence tripped: the update was admitted under an incarnation this node no longer holds"); + } } CasRefCatalog::BeginRemovingOutcome CasRefCatalog::beginRemoving( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - uint64_t removal_started_round, uint64_t admitted_generation, - const std::function & check_fence_or_throw) + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + uint64_t removal_started_round) { if (observed.state != NsState::Live || observed.creator || observed.removal_started_round) throw Exception(ErrorCodes::LOGICAL_ERROR, @@ -298,8 +390,8 @@ CasRefCatalog::BeginRemovingOutcome CasRefCatalog::beginRemoving( const auto mutate = [&](const RefCatalog & cur) -> RefCatalog { - try { check_fence_or_throw(admitted_generation); } - catch (...) { throw CatalogFenceMovedMarker{}; } + if (!op.admitted()) + throw CatalogFenceMovedMarker{}; const auto it = findEntry(cur, observed.ns); if (it == cur.entries.end() || *it != observed) @@ -314,7 +406,7 @@ CasRefCatalog::BeginRemovingOutcome CasRefCatalog::beginRemoving( try { - casUpdateImpl(backend, layout, mutate, [](const RefCatalog & c) { return encodeRefCatalog(c); }); + casUpdateImpl(op, layout, mutate, [](const RefCatalog & c) { return encodeRefCatalog(c); }); } catch (const CatalogFenceMovedMarker &) { @@ -322,7 +414,7 @@ CasRefCatalog::BeginRemovingOutcome CasRefCatalog::beginRemoving( } catch (const CatalogEntryMismatchMarker &) { - const Snapshot current = read(backend, layout); + const Snapshot current = read(op, layout); const auto it = findEntry(current.catalog, observed.ns); if (it != current.catalog.entries.end() && it->incarnation == observed.incarnation @@ -334,9 +426,9 @@ CasRefCatalog::BeginRemovingOutcome CasRefCatalog::beginRemoving( } CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemoving( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - const CasFoldSeal & authoritative_parent, uint64_t admitted_generation, - const std::function & check_fence) + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + const CasFoldSeal & authoritative_parent, + const std::function & refresh_authority) { if (observed.state != NsState::Removing || !observed.removal_started_round) return { @@ -354,15 +446,13 @@ CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemov .catalog_snapshot = std::nullopt}; return deleteCompletedRemovingAtSnapshot( - backend, layout, read(backend, layout), observed, authoritative_parent, - admitted_generation, check_fence); + op, layout, read(op, layout), observed, authoritative_parent, refresh_authority); } CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemovingAtSnapshot( - Backend & backend, const Layout & layout, Snapshot catalog_snapshot, + CasOperation & op, const Layout & layout, Snapshot catalog_snapshot, const CatalogEntry & observed, const CasFoldSeal & authoritative_parent, - uint64_t admitted_generation, - const std::function & check_fence) + const std::function & refresh_authority) { if (observed.state != NsState::Removing || !observed.removal_started_round) return { @@ -394,42 +484,49 @@ CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemov .catalog_snapshot = std::move(catalog_snapshot)}; }; + /// ONE bound for the whole loop, frozen before the first iteration: every erase and every + /// resolution read below shares this deadline, so a permanently contended catalog gives up + /// retry-later within one standard window rather than spending a fresh window per verb per + /// iteration. The paced retry at the end of the loop is a bare sleep that does not consult the + /// deadline, so the loop can sleep one backoff (at most 5 s) past it before the next erase refuses + /// to start. + const Retry policy = op.freeze(Retry::standard()); + for (size_t attempt = 0; attempt < kMaxCatalogCasAttempts; ++attempt) { + /// So the admission consulted below is not one reading taken before the first attempt. + if (refresh_authority) + refresh_authority(); + catalog_snapshot.life_index.throwIfAmbiguous("CAS completed-removal deletion"); + /// A caller-supplied cut without an incarnation cannot state a precondition, and an erase that + /// fell back to an unconditional write would delete whatever a concurrent writer had put there. + if (!catalog_snapshot.etag) + throwMandatoryCatalogAbsent(layout.refCatalogKey()); const auto observed_it = findEntry(catalog_snapshot.catalog, observed.ns); if (observed_it == catalog_snapshot.catalog.entries.end() || *observed_it != observed) return resolved_result(CompletedRemovingDeleteOutcome::EntryChanged); - bool fence_lost = check_fence(admitted_generation) == LeaderFenceStatus::Moved; - - std::optional cas_result; - std::exception_ptr attempt_failure; - if (!fence_lost) - { - RefCatalog candidate = catalog_snapshot.catalog; - candidate.entries.erase(candidate.entries.begin() + (observed_it - catalog_snapshot.catalog.entries.begin())); - try - { - cas_result = backend.casPut( - layout.refCatalogKey(), encodeRefCatalog(candidate), catalog_snapshot.token); - } - catch (...) - { - attempt_failure = std::current_exception(); - } - } + /// Nothing is attempted without admission, and a refusal here has sent nothing. + if (!op.admitted()) + return resolved_result(CompletedRemovingDeleteOutcome::FencedOut); - /// The response to a conditional erase is not authority for what became durable. Resolve - /// every attempted erase, and a pre-CAS fence refusal, through one complete catalog read. - /// This snapshot is also the next retry/selection cut, so no second read separates them. - catalog_snapshot = read(backend, layout); + RefCatalog candidate = catalog_snapshot.catalog; + candidate.entries.erase(candidate.entries.begin() + (observed_it - catalog_snapshot.catalog.entries.begin())); + WriteResult erase = op.replace(layout.refCatalogKey(), encodeRefCatalog(candidate), + *catalog_snapshot.etag, policy); - if (!fence_lost) - fence_lost = check_fence(admitted_generation) == LeaderFenceStatus::Moved; - if (fence_lost) + /// An operation whose admission is gone cannot issue the resolution read either, so the call + /// ends HERE and reports the cut it was given rather than a fresh one. There is nothing further + /// this actor may learn, and nothing further it may do. + if (!op.admitted()) return resolved_result(CompletedRemovingDeleteOutcome::FencedOut); + /// The response to a conditional erase is not authority for what became durable. Resolve every + /// attempted erase through one complete catalog read. This snapshot is also the next + /// retry/selection cut, so no second read separates them. + catalog_snapshot = read(op, layout, policy); + const auto current_it = findEntry(catalog_snapshot.catalog, observed.ns); const bool old_life_still_cataloged = current_it != catalog_snapshot.catalog.entries.end() && current_it->incarnation == observed.incarnation; @@ -438,15 +535,25 @@ CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemov ? CompletedRemovingDeleteOutcome::Deleted : CompletedRemovingDeleteOutcome::EntryChanged); - if (attempt_failure) - std::rethrow_exception(attempt_failure); - if (cas_result && cas_result->outcome == CasOutcome::Committed) - throwCasWriteRetryLater(fmt::format( - "CAS ref catalog erase for namespace '{}' reported committed, but a complete resolution read " - "still observed incarnation {}", - observed.ns.string(), u128ToHex(observed.incarnation))); - /// A token conflict that leaves the exact old row present retries from this mandatory - /// resolution snapshot. The fence is checked again immediately before the next CAS. + /// The row survived the attempt. Only a refused precondition may be tried again against the + /// mandatory resolution snapshot above; every other alternative is terminal for this call, and + /// a commit the resolution read contradicts is reported rather than believed. + if (!std::holds_alternative(erase)) + { + if (std::holds_alternative(erase)) + throwCasWriteRetryLater(fmt::format( + "CAS ref catalog erase for namespace '{}' reported committed, but a complete resolution read " + "still observed incarnation {}", + observed.ns.string(), u128ToHex(observed.incarnation))); + throwCatalogWriteFailure(std::move(erase), fmt::format( + "CAS ref catalog erase for namespace '{}'", observed.ns.string())); + } + + /// Only a refused precondition reaches here, so this pause paces one contended key's retries. + /// The argument is the number of reissues so far, which is one more than the zero-based + /// iteration: `backoff(0)` is no wait at all, and the first retry is the one most likely to + /// collide with the writer that just won. + op.pause(Retry::backoff(static_cast(attempt) + 1)); } throwCasWriteRetryLater(fmt::format( @@ -455,9 +562,8 @@ CasRefCatalog::CompletedRemovingDeleteResult CasRefCatalog::deleteCompletedRemov } CasRefCatalog::StalledCreatingCancelOutcome CasRefCatalog::cancelStalledCreating( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - const std::function & is_creator_fence_terminal, - uint64_t admitted_generation, const std::function & check_fence_or_throw) + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + const std::function & is_creator_fence_terminal) { if (observed.state != NsState::Creating || !observed.creator) throw Exception(ErrorCodes::LOGICAL_ERROR, @@ -467,8 +573,8 @@ CasRefCatalog::StalledCreatingCancelOutcome CasRefCatalog::cancelStalledCreating const auto mutate = [&](const RefCatalog & cur) -> RefCatalog { - try { check_fence_or_throw(admitted_generation); } - catch (...) { throw CatalogFenceMovedMarker{}; } + if (!op.admitted()) + throw CatalogFenceMovedMarker{}; const auto it = findEntry(cur, observed.ns); if (it == cur.entries.end() || *it != observed) @@ -483,7 +589,7 @@ CasRefCatalog::StalledCreatingCancelOutcome CasRefCatalog::cancelStalledCreating try { - casUpdateImpl(backend, layout, mutate, [](const RefCatalog & c) { return encodeRefCatalog(c); }); + casUpdateImpl(op, layout, mutate, [](const RefCatalog & c) { return encodeRefCatalog(c); }); } catch (const CatalogFenceMovedMarker &) { return StalledCreatingCancelOutcome::FencedOut; } catch (const CatalogEntryMismatchMarker &) { return StalledCreatingCancelOutcome::EntryChanged; } @@ -492,9 +598,7 @@ CasRefCatalog::StalledCreatingCancelOutcome CasRefCatalog::cancelStalledCreating } CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::completeCreation( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - uint64_t admitted_generation, const std::function & check_fence_or_throw, - const CkptDeadline & deadline) + CasOperation & op, const Layout & layout, const CatalogEntry & observed, const Retry & policy) { if (observed.state != NsState::Creating || !observed.creator) throw Exception(ErrorCodes::LOGICAL_ERROR, @@ -503,13 +607,14 @@ CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::completeCreation( /// Step 2 (spec §3): INV-4's first `_ckpt` writer for this incarnation, and the only writer that /// will ever know its genesis epoch -- see `Pool/CasRefCkpt.h`'s `publishCkpt` doc for the merge - /// discipline this rides on unchanged. `FencedOut` here ends the attempt: nothing durable changed. + /// discipline this rides on unchanged. `FencedOut` here ends the attempt without step 3; the + /// `_ckpt` itself may or may not have become durable, which is why the entry is left `Creating` + /// for whichever actor next reconciles it rather than cleaned up here. const RefCkpt contribution{.life_epoch = std::optional{observed.creator->writer_epoch}, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - if (publishCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(observed.ns, observed.incarnation), - contribution, admitted_generation, check_fence_or_throw, - deadline) == CkptPublishOutcome::FencedOut) + if (publishCkpt(op, layout, NamespaceLifeId::fromCatalogEntry(observed.ns, observed.incarnation), + contribution, policy) == CkptPublishOutcome::FencedOut) return NamespaceCreationOutcome::FencedOut; /// Step 3. `mutate` is the fence re-check point `casUpdate`'s header doc names -- checked FIRST, @@ -518,8 +623,10 @@ CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::completeCreation( /// (both are truthful refusals of a CAS that was never sent; this is only which one speaks first). const auto mutate = [&](const RefCatalog & cur) -> RefCatalog { - try { check_fence_or_throw(admitted_generation); } - catch (...) { throw CatalogFenceMovedMarker{}; } /// typed, not propagated -- publishCkpt's own precedent + /// Typed, not propagated: the caller asked "did this land", and "the fence moved, so nothing + /// was sent" is an answer, not a failure of the operation. + if (!op.admitted()) + throw CatalogFenceMovedMarker{}; const auto it = findEntry(cur, observed.ns); if (it == cur.entries.end() || *it != observed) @@ -533,7 +640,8 @@ CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::completeCreation( try { - casUpdate(backend, layout, mutate); + casUpdateImpl(op, layout, identityPreserving(mutate), + [](const RefCatalog & c) { return encodeRefCatalog(c); }, policy); } catch (const CatalogFenceMovedMarker &) { return NamespaceCreationOutcome::FencedOut; } catch (const CatalogEntryMismatchMarker &) { return NamespaceCreationOutcome::Superseded; } @@ -541,39 +649,39 @@ CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::completeCreation( } CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::createNamespace( - Backend & backend, const Layout & layout, uint64_t gc_shards, - const RootNamespace & ns, const CreatorFence & creator, - uint64_t admitted_generation, const std::function & check_fence_or_throw, - const CkptDeadline & deadline) + CasOperation & op, const Layout & layout, uint64_t gc_shards, + const RootNamespace & ns, const CreatorFence & creator, const Retry & policy) { /// Read-first, per the Task 2 review's own note on `casAdmitEntry`: a namespace that already - /// carries an entry is THIS function's job to reject with a clear message, not `casAdmitEntry`'s - /// duplicate-namespace grammar refusal (which would report a `LOGICAL_ERROR` about canonical order - /// -- true, but useless to a caller trying to understand why its create failed). A concurrent + /// carries an entry is THIS function's job to notice and report `Superseded` for, not + /// `casAdmitEntry`'s duplicate-namespace grammar refusal (which would report a `LOGICAL_ERROR` about + /// canonical order -- true, but useless to a caller whose create merely lost a race). A concurrent /// insert of the SAME namespace between this read and step 1 is still caught -- `casAdmitEntry`'s /// own grammar check is the backstop, not the only check. - const Snapshot snap = read(backend, layout); + if (create_namespace_pre_check_hook_for_test) + { + std::function hook_to_run; + std::swap(hook_to_run, create_namespace_pre_check_hook_for_test); + hook_to_run(); + } + const Snapshot snap = read(op, layout, policy); const auto existing = findEntry(snap.catalog, ns); if (existing != snap.catalog.entries.end()) { - /// `Creating` is not this function's problem to solve (the class-level doc above says so) -- - /// it is exactly the race `resolveNamespaceLife`'s own loop is built to absorb: sibling openers - /// of the SAME namespace (e.g. concurrent per-part freeze threads of one query, which share one - /// mount's fence) can all observe "no entry" before any of them lands step 1, then race into - /// this call. Reporting `Superseded` sends the loser back through the loop, where it re-reads - /// and takes the documented resume path (its own fence: `completeCreation`; a foreign one: - /// `reconcileStaleCreator`) instead of aborting the server for an outcome the design already - /// names and handles. `Live`/`Removing` stay a `LOGICAL_ERROR`: `namespaceLife`'s caller filters - /// `Live` before ever reaching here and refuses `Removing` outright, so seeing either here means - /// a caller bypassed that dispatch, not a race. - if (existing->state == NsState::Creating) - return NamespaceCreationOutcome::Superseded; - throw Exception(ErrorCodes::LOGICAL_ERROR, - "CasRefCatalog::createNamespace: namespace '{}' already carries a catalog entry (state " - "'{}') -- a stalled Creating entry is resumed through reconcileStaleCreator + " - "completeCreation, never a fresh createNamespace call; an existing Live or Removing " - "namespace must complete its current lifecycle before a fresh creation can be admitted", - ns.string(), nsStateToWord(existing->state)); + /// This read is a snapshot taken AFTER the caller's own "no entry" read (`resolveNamespaceLife`'s + /// loop, or any other dispatcher that only reaches `createNamespace` once it has seen nothing to + /// adopt). A sibling opener of the SAME namespace -- concurrent threads of one server: parallel + /// background movers, inserts, or per-part `FREEZE` -- can land anywhere in its own three-step + /// sequence in the gap between those two reads, so EVERY state observed here is a race outcome, + /// never a caller bug: `Creating` (a sibling landed step 1 only), `Live` (a sibling completed all + /// three steps and already won birth), and `Removing` (a concurrent drop) are all reported + /// `Superseded`, sending the loser back through its own resume loop rather than aborting the + /// server for an outcome the design already names and handles. There the loop's fresh re-read + /// tells the loser what actually happened: `Live` is adopted directly, `Removing` is refused by + /// the loop's own `Removing` branch, and a still-`Creating` entry resumes through + /// `reconcileStaleCreator` + `completeCreation` (or, if it is this caller's own fence, straight + /// through `completeCreation`). + return NamespaceCreationOutcome::Superseded; } const CatalogEntry entry{.ns = ns, .state = NsState::Creating, @@ -587,13 +695,19 @@ CasRefCatalog::NamespaceCreationOutcome CasRefCatalog::createNamespace( /// above) and reports the race as `Superseded` instead. try { - createNamespaceStep1(backend, layout, gc_shards, entry); /// step 1 + createNamespaceStep1(op, layout, gc_shards, entry, policy); /// step 1 } catch (const CatalogEntryAlreadyPresentMarker &) { return NamespaceCreationOutcome::Superseded; } - return completeCreation(backend, layout, entry, admitted_generation, check_fence_or_throw, deadline); + catch (const CatalogFenceMovedMarker &) + { + /// The creator's own admission moved while step 1 waited its turn or wrote: an answer, not a + /// failure, and the same one the two later steps already give. + return NamespaceCreationOutcome::FencedOut; + } + return completeCreation(op, layout, entry, policy); } void CasRefCatalog::setCreateNamespaceStep1PreReadHookForTest(std::function hook) @@ -601,25 +715,31 @@ void CasRefCatalog::setCreateNamespaceStep1PreReadHookForTest(std::function hook) +{ + create_namespace_pre_check_hook_for_test = std::move(hook); +} + CasRefCatalog::ReconcileCreatorOutcome CasRefCatalog::reconcileStaleCreator( - Backend & backend, const Layout & layout, const CatalogEntry & observed, const CreatorFence & new_creator, - const std::function & is_creator_fence_terminal, - uint64_t admitted_generation, const std::function & check_fence_or_throw) + CasOperation & op, const Layout & layout, const CatalogEntry & observed, const CreatorFence & new_creator, + const std::function & is_creator_fence_terminal, const Retry & policy) { if (observed.state != NsState::Creating || !observed.creator) throw Exception(ErrorCodes::LOGICAL_ERROR, "CasRefCatalog::reconcileStaleCreator: namespace '{}' is not a Creating entry with a " "creator fence -- nothing to reconcile", observed.ns.string()); - /// Review I6: the fence re-check is checked FIRST, on every fresh read this CAS retries -- the same - /// placement `completeCreation` uses for exactly the same reason (see that function's own doc). - /// Token-exactness (the catalog's own entry, by full value) comes next: it is the cheaper, purely + /// The admission check comes FIRST, on every fresh read this retries -- the same placement + /// `completeCreation` uses for exactly the same reason (see that function's own doc). + /// Entry-exactness (the catalog's own entry, by full value) comes next: it is the cheaper, purely /// local comparison, and a mismatch here means the question "is the OLD creator's fence terminal" is /// moot -- `observed` no longer describes anything live to reconcile. const auto mutate = [&](const RefCatalog & cur) -> RefCatalog { - try { check_fence_or_throw(admitted_generation); } - catch (...) { throw CatalogFenceMovedMarker{}; } /// typed, not propagated -- completeCreation's own precedent + /// Typed, not propagated: the caller asked "did this land", and "the fence moved, so nothing + /// was sent" is an answer, not a failure of the operation. + if (!op.admitted()) + throw CatalogFenceMovedMarker{}; const auto it = findEntry(cur, observed.ns); if (it == cur.entries.end() || *it != observed) @@ -634,7 +754,8 @@ CasRefCatalog::ReconcileCreatorOutcome CasRefCatalog::reconcileStaleCreator( try { - casUpdate(backend, layout, mutate); + casUpdateImpl(op, layout, identityPreserving(mutate), + [](const RefCatalog & c) { return encodeRefCatalog(c); }, policy); } catch (const CatalogFenceMovedMarker &) { return ReconcileCreatorOutcome::FencedOut; } catch (const CatalogEntryMismatchMarker &) { return ReconcileCreatorOutcome::EntryChanged; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.h index 6eca5f4c985f..b1ed539def66 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCatalog.h @@ -1,6 +1,6 @@ #pragma once #include -#include +#include #include #include #include @@ -13,32 +13,34 @@ namespace DB::Cas { /// The `cas/ref_catalog` object (spec INV-3) as seen from the pool side: reading the current -/// catalog, and the generic token-CAS retry primitive every lifecycle transition rides. This class +/// catalog, and the generic conditional-update primitive every lifecycle transition rides. This class /// builds ONLY that primitive -- the actual lifecycle steps (the three-conditional-write creation -/// sequence, the removal terminal-record-then-entry-delete sequence) are later tasks' job, built ON -/// TOP of `casUpdate`/`casAdmitEntry`. +/// sequence, the removal terminal-record-then-entry-delete sequence) are built ON TOP of +/// `casUpdate`/`casAdmitEntry`, further down in this same class. class CasRefCatalog { public: - /// The catalog snapshot as read from the backend: the decoded object plus the token an update - /// must present to `casPut`. Operational reads always return a token because the catalog is a - /// mandatory control object after pool bootstrap. + /// The catalog snapshot as read from the backend: the decoded object plus the incarnation an + /// update must present as its precondition. Operational reads always carry one because the catalog + /// is a mandatory control object after pool bootstrap. struct Snapshot { RefCatalog catalog; - std::optional token; + std::optional etag; CatalogLifeIndex life_index; }; /// Reads and decodes the mandatory current catalog. Absence is corruption, never an empty /// authority set: without the catalog, opaque life keys cannot prove ownership. - static Snapshot read(Backend & backend, const Layout & layout); + /// `policy` lets a hand-written loop pass the bound it froze at entry, so its resolution reads end + /// with the rest of the loop instead of each starting a fresh window. + static Snapshot read(CasOperation & op, const Layout & layout, const Retry & policy = Retry::standard()); /// Materializes the explicit empty catalog for a prefix already proven new by /// `probePoolBootstrapResidual`. This is the only absence-tolerant catalog operation: no /// existing-pool caller can accidentally turn authoritative absence into an empty catalog. A /// concurrent bootstrap winner is accepted only after its object is read and decoded. - static Snapshot initializeEmptyForNewPool(Backend & backend, const Layout & layout); + static Snapshot initializeEmptyForNewPool(CasOperation & op, const Layout & layout); /// The catalog's life for `ns` if a `Live`/`Removing` entry names it, else `nullopt` -- ONE catalog /// read and, crucially, NO WRITE OF ANY KIND. This is the resolution a READ or a REMOVAL uses: it @@ -52,50 +54,43 @@ class CasRefCatalog /// `Creating` is excluded for the same reason `liveUniverse` excludes it: no publication can exist /// under an entry still being created, so there is nothing to resolve to and nothing to read. static std::optional lifeIfCataloged( - Backend & backend, const Layout & layout, const RootNamespace & ns); + CasOperation & op, const Layout & layout, const RootNamespace & ns); /// Every `Live`/`Removing` life the catalog currently names, from this call's own catalog `GET`. /// This helper is for independent readers such as `CasFsck`; a GC fold instead keeps the immutable /// post-LIST snapshot attached to its scan and reuses that exact cut throughout the round. /// `Creating` is excluded: spec §3, no publication can exist yet. - static std::vector liveUniverse(Backend & backend, const Layout & layout); + static std::vector liveUniverse(CasOperation & op, const Layout & layout); - /// The generic token-CAS retry loop shared by every catalog mutation, mirroring - /// `PoolMeta::admitOrValidate`'s loop: read the current snapshot, apply `mutate` to obtain the - /// CANDIDATE next catalog, `casPut` it against the mandatory object's observed token, and on - /// `Conflict` re-read and re-apply `mutate` to the - /// FRESH snapshot -- never re-encoding the stale candidate. `mutate` must return a canonically - /// ordered, grammar-valid candidate; `encodeRefCatalog` (called internally) enforces that. + /// The generic read-modify-write shared by every catalog mutation: `mutate` turns the durable + /// catalog into the CANDIDATE next one, which is encoded and written against the incarnation the + /// same call read. A refused precondition re-runs `mutate` against the FRESH body -- never + /// re-encoding the stale candidate. `mutate` must return a canonically ordered, grammar-valid + /// candidate; `encodeRefCatalog` (called internally) enforces that. /// - /// Bounded (the same live-lock brake `publishCkpt`/`allocateWriterEpoch` use on their own - /// contended token-CAS singletons): after 100 conflicting attempts it gives up and raises the - /// typed retryable error `throwCasWriteRetryLater`, naming the key and the attempt count, rather - /// than spinning forever against a pathologically busy catalog. + /// The engine's own policy is the bound: persistent conflict ends at the write deadline with the + /// typed retryable error `throwCasWriteRetryLater`, rather than spinning forever against a + /// pathologically busy catalog. /// - /// A re-read that finds the object genuinely ABSENT after it was previously observed present is - /// corruption, not a fresh bootstrap. The required `read` throws before another CAS attempt, so - /// no mutation can replace every other namespace with a one-update catalog. + /// A read that finds the object ABSENT is corruption, not a fresh bootstrap -- `mutate` is never + /// offered an absent catalog to build a replacement authority from, so no mutation can replace + /// every other namespace with a one-update catalog. /// /// This primitive runs NO admission check: Constraint 13 (removal is never refused) means /// whether a candidate must clear the additive predicate is the CALLER's decision, not this /// loop's. A caller mutating an entry's state without growing the catalog (a removal transition) /// uses this directly. /// - /// THE FENCE OBLIGATION (Task 3 carry-over from the Task 2 review): this loop has no fence - /// parameter and performs no fence check of its own -- `publishCkpt`'s "AFTER the read, BEFORE the - /// CAS, on every attempt" discipline has no equivalent built in here. The seam a fenced caller + /// THE FENCE OBLIGATION: this loop performs no fence check of its own. The seam a fenced caller /// MUST use is `mutate` itself: it runs, fresh, after EVERY read this loop performs (the very - /// first one and every one after a `Conflict`), immediately before the candidate it returns is - /// encoded and `casPut`. A caller that needs its own write fenced (any catalog mutation minted - /// under a mount incarnation -- which is every one Task 3 onward adds) MUST throw from inside - /// `mutate`, checking on EVERY invocation, not once before calling `casUpdate`: checking once - /// before the call fences against the read this loop is *about* to perform, not the one it just - /// did, and a `Conflict` retry performs an entirely new read `mutate` is never told about except - /// by being called again. `completeCreation`'s own `mutate` (below) is the first production - /// caller to ride this seam, and does so by wrapping its `check_fence_or_throw` call at the top of - /// its `mutate`, exactly where `publishCkpt` places the identical check. + /// first one and every one after a refused precondition), immediately before the candidate it + /// returns is encoded and written. A caller that needs its own write fenced (any catalog mutation + /// minted under a mount incarnation) MUST refuse from inside `mutate`, checking on EVERY + /// invocation, not once before calling `casUpdate`: checking once before the call fences against + /// the read this loop is *about* to perform, not the one it just did, and a retry performs an + /// entirely new read `mutate` is never told about except by being called again. static RefCatalog casUpdate( - Backend & backend, const Layout & layout, + CasOperation & op, const Layout & layout, const std::function & mutate); /// Admits exactly ONE new namespace into the catalog under INV-3's two-predicate gate, inserting @@ -104,11 +99,11 @@ class CasRefCatalog /// entry point that accepted a free-form candidate could be handed a REMOVAL by a future caller /// that reads as correct, silently reopening Constraint 13 (removal is never refused) behind a /// name that says "admitting". A namespace `entry.ns` already carries an entry is a bug in the - /// caller (Task 3's creation lifecycle owns checking that first) and surfaces as + /// caller (`createNamespace` below owns checking that first) and surfaces as /// `encodeRefCatalog`'s own canonical-order/no-duplicate grammar check, inside /// `checkCatalogAdmission`. static RefCatalog casAdmitEntry( - Backend & backend, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry); + CasOperation & op, const Layout & layout, uint64_t gc_shards, const CatalogEntry & entry); enum class BeginRemovingOutcome : uint8_t { @@ -119,13 +114,12 @@ class CasRefCatalog }; /// Exact `Live -> Removing` transition. The immutable observed row is compared by full value on - /// every catalog retry, and the mount fence is checked after every fresh read and before its CAS. - /// A row already `Removing` under the same namespace/life resolves an ambiguous or concurrent - /// transition positively; no caller may change its recorded start round afterward. + /// every catalog retry, and `op`'s admission is checked after every fresh read and before its + /// write. A row already `Removing` under the same namespace/life resolves an ambiguous or + /// concurrent transition positively; no caller may change its recorded start round afterward. static BeginRemovingOutcome beginRemoving( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - uint64_t removal_started_round, uint64_t admitted_generation, - const std::function & check_fence_or_throw); + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + uint64_t removal_started_round); /// Outcome of the only fold-authorized catalog deletion. A refusal never writes the catalog. enum class CompletedRemovingDeleteOutcome : uint8_t @@ -136,23 +130,17 @@ class CasRefCatalog FencedOut, }; - /// Authority result for completed-removal erases. Only an explicit `Moved` is a fence outcome; - /// exceptions mean authority could not be evaluated and propagate to the caller. - enum class LeaderFenceStatus : uint8_t - { - Held, - Moved, - }; - struct CompletedRemovingDeleteResult { CompletedRemovingDeleteOutcome outcome; /// Present only when a mandatory fresh catalog read proves that the exact observed life is no /// longer cataloged, whether this actor's erase committed or another actor removed/replaced it. std::optional invalidated_life; - /// The complete mandatory resolution snapshot after an attempted erase. The GC drain feeds - /// this directly into its next deterministic selection; proof refusal performs no read and - /// leaves it absent. + /// The catalog cut this call ends on. After an attempted erase it is the mandatory resolution + /// read's snapshot, which the GC drain feeds directly into its next deterministic selection. + /// On `FencedOut` it is instead the cut the call was GIVEN: an operation whose admission is + /// gone cannot issue the resolution read, so the erase's own effect is left unreported. Proof + /// refusal performs no read at all and leaves this absent. std::optional catalog_snapshot; bool operator==(CompletedRemovingDeleteOutcome expected) const { return outcome == expected; } @@ -161,21 +149,27 @@ class CasRefCatalog /// Exact-CAS-deletes `observed` only when it is a complete `Removing` row and the authoritative /// adopted parent carries cleanup evidence, but no hold, in the row keyed by the same opaque life /// id. The whole parent seal is consumed so a caller cannot separate the life id from its proof or - /// reduce the proof to a caller-computed boolean. The leader fence is checked after every fresh - /// catalog read and before every attempted CAS. + /// reduce the proof to a caller-computed boolean. `op`'s admission is checked after every fresh + /// catalog read and before every attempted erase. + /// + /// `refresh_authority` runs at the top of every attempt, before the checks that decide whether to + /// send one. It is not a verdict -- the verdict stays `op.admitted()` -- but that liveness may be + /// a cached flag its holder refreshes from a fact this loop cannot see, and one reading taken + /// before the first erase must not authorise the rest. Mandatory: a caller whose liveness needs no + /// refresh passes a no-op and says so, instead of erasing unrefreshed because the argument was + /// easy to omit. static CompletedRemovingDeleteResult deleteCompletedRemoving( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - const CasFoldSeal & authoritative_parent, uint64_t admitted_generation, - const std::function & check_fence); + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + const CasFoldSeal & authoritative_parent, + const std::function & refresh_authority); - /// Same exact deletion, using the caller's complete selected catalog snapshot and token for the - /// one CAS attempt. Its mandatory resolution snapshot is returned in the result so a catalog-only - /// drain can select the next row without an intervening read. + /// Same exact deletion, using the caller's complete selected catalog snapshot and its incarnation + /// for the one erase attempt. Its mandatory resolution snapshot is returned in the result so a + /// catalog-only drain can select the next row without an intervening read. static CompletedRemovingDeleteResult deleteCompletedRemovingAtSnapshot( - Backend & backend, const Layout & layout, Snapshot catalog_snapshot, + CasOperation & op, const Layout & layout, Snapshot catalog_snapshot, const CatalogEntry & observed, const CasFoldSeal & authoritative_parent, - uint64_t admitted_generation, - const std::function & check_fence); + const std::function & refresh_authority); /// Outcome of exact stalled-creation cancellation, the only other exported deletion shape. enum class StalledCreatingCancelOutcome : uint8_t @@ -189,11 +183,10 @@ class CasRefCatalog /// Exact-CAS-deletes one observed `Creating` row only after its complete creator fence is proven /// terminal. This performs no `_ckpt` or other physical cleanup; debris belongs to the janitor. static StalledCreatingCancelOutcome cancelStalledCreating( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - const std::function & is_creator_fence_terminal, - uint64_t admitted_generation, const std::function & check_fence_or_throw); + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + const std::function & is_creator_fence_terminal); - /// === Task 3: the §3 creation lifecycle, built on the two primitives above === + /// === The creation lifecycle, built on the two primitives above === /// Outcome of the two-step tail every creation attempt ends in (`_ckpt` publish + `Creating -> /// Live` CAS) -- shared by a fresh `createNamespace` and a reconciler that just adopted a stalled @@ -203,9 +196,11 @@ class CasRefCatalog { Live, /// the entry reached `Live`; `_ckpt` is durable with this creator's `writer_epoch` /// as `life_epoch` (spec INV-4: the genesis epoch, recorded nowhere else). - FencedOut, /// this caller's OWN admitted generation moved before the `_ckpt` publish or the - /// `Creating -> Live` CAS. Nothing more was written; the caller's own mount - /// incarnation is gone, so it cannot be the one to retry. + FencedOut, /// this caller's OWN admission was lost, at the `_ckpt` publish or at the + /// `Creating -> Live` write. Whether the step it was running landed is + /// UNRESOLVED -- admission is reported lost both before an attempt is sent and + /// after one is proven durable, so a resumer re-reads rather than assumes. The + /// caller's own mount incarnation is gone, so it cannot be the one to retry. Superseded, /// the catalog entry no longer equals what this caller observed -- a concurrent /// reconciler stole it, or a race already carried it to `Live`/`Removing`. Nothing /// was written; a DIFFERENT actor now owns whatever happens to this namespace next. @@ -226,34 +221,32 @@ class CasRefCatalog EntryChanged, /// the catalog's current entry for `observed.ns` no longer equals /// `observed` -- token-exactness failed. Not written; the caller must /// re-read the catalog before trying again. - FencedOut, /// review I6: this caller's OWN admitted generation moved before the CAS - /// -- nothing was written, and the caller's own mount incarnation is gone, - /// so it cannot be the one to retry. Mirrors `NamespaceCreationOutcome:: - /// FencedOut`; without this check a deposed mount could still steal a - /// `Creating` entry onto its own dead fence before the following - /// `completeCreation` refuses it -- the catalog would be mutated by an - /// actor this subsystem otherwise never lets touch it. + FencedOut, /// this caller's OWN admission was lost, and whether its write landed is + /// unresolved; its mount incarnation is gone either way, so it cannot be + /// the one to retry. Mirrors `NamespaceCreationOutcome::FencedOut`; + /// without this check a deposed mount could still steal a `Creating` + /// entry onto its own dead fence before the following `completeCreation` + /// refuses it -- the catalog would be mutated by an actor this subsystem + /// otherwise never lets touch it. }; /// The full, fresh §3 sequence for a namespace that carries NO catalog entry yet: mints a random /// nonzero incarnation (spec: "fresh_random_128"), runs step 1 (`casAdmitEntry` inserting `{ns, /// Creating, incarnation, creator}`), then steps 2+3 via `completeCreation` below. /// - /// Per the Task 2 review's own note on `casAdmitEntry` ("a namespace `entry.ns` already carries an - /// entry is a bug in the caller -- Task 3's creation lifecycle owns checking that first"): this - /// function reads the catalog FIRST rather than handing `casAdmitEntry` a doomed insert and letting - /// its own grammar check report a confusing duplicate-namespace message. A namespace already - /// `Creating` is not this function's problem to solve -- that is exactly what `reconcileStaleCreator` - /// + `completeCreation` are for, so this reports `Superseded` (never `LOGICAL_ERROR`) and sends the - /// caller back through its own resume loop: sibling openers of the same namespace that all observed - /// "no entry" before any of them landed step 1 race in here exactly this way. A namespace already - /// `Live`/`Removing` IS a caller bug (recreating an existing name is removal's business, not - /// creation's) and still throws `LOGICAL_ERROR` naming the observed state. + /// This function reads the catalog FIRST rather than handing `casAdmitEntry` a doomed insert and + /// letting its own grammar check report a confusing duplicate-namespace message. That read is a + /// snapshot taken AFTER the caller's own "no entry" read, so a sibling opener of the SAME namespace + /// (concurrent threads of one server: parallel background movers, inserts, `FREEZE`) can have landed + /// anywhere in its own three-step sequence in between -- ANY state observed here (`Creating`, `Live`, + /// `Removing`) is that race, never a caller bug, and is reported `Superseded` uniformly, sending the + /// caller back through its own resume loop (`resolveNamespaceLife`'s loop re-reads and dispatches: + /// `Live` is adopted, `Removing` is refused there, a still-`Creating` entry resumes through + /// `reconcileStaleCreator` + `completeCreation`). static NamespaceCreationOutcome createNamespace( - Backend & backend, const Layout & layout, uint64_t gc_shards, + CasOperation & op, const Layout & layout, uint64_t gc_shards, const RootNamespace & ns, const CreatorFence & creator, - uint64_t admitted_generation, const std::function & check_fence_or_throw, - const CkptDeadline & deadline); + const Retry & policy = Retry::standard()); /// Fires once, synchronously, right after `createNamespace`'s own pre-check read observed no /// entry and right before its step 1 performs its own (first) catalog read -- the exact window a @@ -263,6 +256,13 @@ class CasRefCatalog /// `CasRefCatalog` itself carries no state. static void setCreateNamespaceStep1PreReadHookForTest(std::function hook); + /// Fires once, synchronously, right before `createNamespace`'s own pre-check read -- the exact + /// window a sibling opener of the same namespace can complete an entire birth (or begin a removal) + /// in, for a test to drive that interleaving deterministically instead of relying on real thread + /// scheduling. Empty (no-op) hook in production; a stateless class-scope hook (rather than an + /// instance member) because `CasRefCatalog` itself carries no state. + static void setCreateNamespacePreCheckHookForTest(std::function hook); + /// Steps 2 (`_ckpt` publish) + 3 (`Creating -> Live` CAS) alone, given an entry the caller already /// owns as `observed` -- either the entry `createNamespace`'s own step 1 just inserted, or one a /// caller just reconciled onto itself via `reconcileStaleCreator`. Exposed separately (rather than @@ -275,9 +275,8 @@ class CasRefCatalog /// this module's own bug, not a race. A `FencedOut` from `publishCkpt` ends the attempt here. /// /// Step 3: `CasRefCatalog::casUpdate`'s `mutate` is the fence re-check point (see the class-level - /// note below) -- `check_fence_or_throw(admitted_generation)` runs FIRST, on every fresh read this - /// retry loop performs, exactly like `publishCkpt`'s own re-check; a throw from it is caught and - /// reported as `FencedOut`, nothing else. ONLY THEN is the fresh entry for `observed.ns` compared + /// note below) -- `op.admitted()` is consulted FIRST, on every fresh read this retry loop performs; + /// a refusal is reported as `FencedOut`, nothing else. ONLY THEN is the fresh entry for `observed.ns` compared /// against `observed` by FULL VALUE equality (`CatalogEntry::operator==`) -- the value-CAS that /// plays the role `publishCkpt`'s object token plays for `_ckpt`, since one catalog object holds /// every namespace's entry and there is no separate per-entry token to CAS against. A mismatch @@ -288,13 +287,12 @@ class CasRefCatalog /// `publishCkpt`); a caller that manages to make BOTH stale sees `FencedOut`, not `Superseded` -- /// both are truthful refusals of a CAS that was never sent. static NamespaceCreationOutcome completeCreation( - Backend & backend, const Layout & layout, const CatalogEntry & observed, - uint64_t admitted_generation, const std::function & check_fence_or_throw, - const CkptDeadline & deadline); + CasOperation & op, const Layout & layout, const CatalogEntry & observed, + const Retry & policy = Retry::standard()); /// Stale-`Creating` reconciliation (spec INV-3: "stalled creators occupy entries until - /// fence-terminal reconciliation"; TLA Task 3 obligation 1: "the call-site is where - /// token-exactness is enforced"). `observed` must be a `Creating` entry this caller read a moment + /// fence-terminal reconciliation"; token-exactness is enforced right here, at this call site). + /// `observed` must be a `Creating` entry this caller read a moment /// ago (`LOGICAL_ERROR` otherwise -- a caller mistake, not a race). Refuses, WITHOUT writing /// anything, unless BOTH hold against a FRESH catalog read: /// - `is_creator_fence_terminal(*observed.creator)` -- injected rather than reaching into @@ -305,8 +303,8 @@ class CasRefCatalog /// `CreatorFence`, so the mount layer stays independent of the ref-catalog format), built from /// `writer_epoch` plus the mount-terminality certificates /// `probeNonTerminalMountSlots`/`computeHeartbeatFloor` already use -- NEVER from - /// `CreatorFence::fence_generation`. That field IS persisted (Task 2 serializes it into the - /// catalog entry), so it reaches the object store fine; what it is NOT is comparable across + /// `CreatorFence::fence_generation`. That field IS persisted (the catalog entry's own encoding + /// carries it), so it reaches the object store fine; what it is NOT is comparable across /// actors: it mirrors `CasMountRuntime::fence_generation`, an in-process atomic that each mount /// bumps from its OWN zero on every open, so a different actor's counter (or the SAME actor's /// after a restart) starts over at the same values and answers a different question than "is @@ -315,28 +313,29 @@ class CasRefCatalog /// (token-exactness: a concurrent reconciler, or the original creator finishing on its own, /// invalidates this immediately). /// On success, CASes `creator` to `new_creator` -- `state` and `incarnation` are UNCHANGED, so the - /// caller resumes with `completeCreation(backend, layout, {..., .creator = new_creator}, ...)` over - /// the SAME incarnation, never a fresh one (rebirth under a fresh incarnation is Task 5/removal's - /// business, not a live reconciliation's). + /// caller resumes with `completeCreation(op, layout, {..., .creator = new_creator})` over the SAME + /// incarnation, never a fresh one (rebirth under a fresh incarnation is removal's business, not a + /// live reconciliation's). /// - /// `admitted_generation`/`check_fence_or_throw` (review I6): re-checked FIRST on every fresh read - /// this CAS retries, exactly like `completeCreation`'s own placement -- a caller whose OWN mount - /// fence has already moved must not be the one to steal a `Creating` entry onto its own (now dead) - /// fence, even though the following `completeCreation` would go on to refuse it as `FencedOut` - /// anyway: by then the catalog would already have been mutated by a deposed actor, the one posture - /// this subsystem otherwise refuses everywhere else. + /// `op.admitted()` is consulted FIRST on every fresh read this retries, exactly like + /// `completeCreation`'s own placement -- a caller whose OWN mount fence has already moved must not + /// be the one to steal a `Creating` entry onto its own (now dead) fence, even though the following + /// `completeCreation` would go on to refuse it as `FencedOut` anyway: by then the catalog would + /// already have been mutated by a deposed actor, the one posture this subsystem otherwise refuses + /// everywhere else. static ReconcileCreatorOutcome reconcileStaleCreator( - Backend & backend, const Layout & layout, const CatalogEntry & observed, + CasOperation & op, const Layout & layout, const CatalogEntry & observed, const CreatorFence & new_creator, const std::function & is_creator_fence_terminal, - uint64_t admitted_generation, const std::function & check_fence_or_throw); + const Retry & policy = Retry::standard()); /// Spec §3: "`Creating` forbids publication -- no ref writes admitted while the entry is /// Creating." Throws `throwCasWriteRetryLater`'s class (transient: `Creating` resolves once the /// creator finishes or is reconciled away) if `catalog`'s entry for `ns` is `Creating`; a no-op for /// every other case -- no entry, `Live`, or `Removing` -- since this is ONLY the birth-lifecycle /// gate on the catalog's own `Creating` state, never a general existence/removal check (that role - /// moves onto the catalog in Task 4/Task 6). Takes an already-read `RefCatalog` rather than + /// belongs to the catalog-governed append path described just below). Takes an already-read + /// `RefCatalog` rather than /// `Backend`/`Layout`, so a caller that is about to append anyway (and so already holds a fresh /// read for its OWN purposes) pays no second GET here. /// diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.cpp index a29b1d5faafe..d7a30304eb68 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.cpp @@ -1,8 +1,10 @@ #include -#include +#include #include +#include #include #include +#include namespace DB { @@ -19,10 +21,6 @@ namespace DB::Cas namespace { -/// Live-lock brake, the same shape and for the same reason as `CasPlainObjects`': the deadline is the -/// real bound, and this only stops an unexpected continuous conflict from spinning until it elapses. -constexpr size_t MAX_CKPT_CAS_ATTEMPTS = 100; - /// The per-field semantic maximum for an OPTIONAL field: a present value beats an absent one (an /// absence is "this writer knew nothing", never "this writer says none"), and two present values /// resolve by the field's own order -- for `RefTxnId` that is writer_epoch then ref_sequence, the @@ -184,54 +182,41 @@ RecoveryGrounding chooseRecoveryGrounding(const std::optional & ca return result; } -std::optional readCkpt(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +std::optional readCkpt(CasOperation & op, const Layout & layout, const NamespaceLifeId & life) { - std::optional got = backend.get(layout.refCkptKey(life)); + const std::optional got = op.read(layout.refCkptKey(life), Retry::standard()); if (!got) return std::nullopt; /// Materialized read, then decode: the object is MUTABLE, so the body must be fixed before it is - /// parsed, and the token must be the one that labels exactly these bytes. - return CkptSample{decodeRefCkpt(got->bytes), got->token}; + /// parsed, and the incarnation must be the one that labels exactly these bytes. + return CkptSample{decodeRefCkpt(got->bytes), got->etag}; } -CkptPublishOutcome publishCkpt(Backend & backend, const Layout & layout, const NamespaceLifeId & life, - const RefCkpt & contribution, uint64_t admitted_generation, - const std::function & check_fence_or_throw, - const CkptDeadline & deadline, - const std::function & admit_request) +CkptPublishOutcome publishCkpt(CasOperation & op, const Layout & layout, const NamespaceLifeId & life, + const RefCkpt & contribution, const Retry & policy) { const String key = layout.refCkptKey(life); - std::optional current; - bool have_current = false; - const auto request_is_admitted = [&] + + /// Why `decide` had nothing to write. Both answers are declines, and only the fence tells them + /// apart, so the reason is recorded where it is decided instead of re-derived from the outcome. + enum class Decline : uint8_t { - try - { - if (admit_request) - admit_request(); - return true; - } - catch (...) - { - return false; - } + Identical, + Fenced, }; + std::optional decline; + /// A decline on any decision AFTER the first can only follow a refused attempt of this call, and an + /// attempt is only ever refused after it was sent. That distinction is what keeps `IdenticalSkip`'s + /// promise -- no write was issued -- true wherever it is reported. + size_t decisions = 0; - for (size_t attempt = 0; attempt < MAX_CKPT_CAS_ATTEMPTS; ++attempt) + const auto decide = [&](const std::optional & current) -> std::optional { - if (deadline.now_ms() >= deadline.deadline_ms) - break; - - /// Read the WHOLE body every attempt. A retry after a conflict must merge against what is - /// there NOW: reusing the previous attempt's reading is precisely the read-modify-write with - /// the merge left out, one round later. - if (!have_current) - { - if (!request_is_admitted()) - return CkptPublishOutcome::FencedOut; - current = readCkpt(backend, layout, life); - have_current = true; - } + ++decisions; + decline.reset(); + std::optional durable; + if (current) + durable = decodeRefCkpt(current->bytes); /// The one rule the commutative merge cannot state, and it has to be decided HERE, before the /// merge: the semantic maximum turns a decrease into a body identical to the stored one, which @@ -243,125 +228,66 @@ CkptPublishOutcome publishCkpt(Backend & backend, const Layout & layout, const N /// transient control signal every other refusal in this function returns rather than throws. A /// writer that is still admitted and yet contributing a superseded epoch is the fence violation /// this detects, and that one is corruption. - if (current && lifeEpochWouldDecrease(current->ckpt, contribution)) + if (durable && lifeEpochWouldDecrease(*durable, contribution)) { - try - { - check_fence_or_throw(admitted_generation); - } - catch (...) + if (!op.admitted()) { - return CkptPublishOutcome::FencedOut; + decline = Decline::Fenced; + return std::nullopt; } - throwLifeEpochDecrease(current->ckpt, contribution, key); + throwLifeEpochDecrease(*durable, contribution, key); } /// ANY writer may create the object; none of them may invent a field. An absent `_ckpt` is /// created from the contribution as it stands, so a publisher that knows only the checkpoint /// creates one that knows only the checkpoint, and the birth transaction's `life_epoch` merges /// into it whenever it arrives -- in either order, because the merge is a per-field maximum. - const RefCkpt merged = current ? mergeCkpt(current->ckpt, contribution) : contribution; + const RefCkpt merged = durable ? mergeCkpt(*durable, contribution) : contribution; - /// Nothing new: return WITHOUT a CAS. This is not an optimization -- both writers publish on - /// every snapshot and every seal, and most of those carry a checkpoint the object already has, - /// so issuing the write anyway would mint a fresh token per no-op and turn every other writer's - /// in-flight CAS into a conflict, for a body byte-identical to the one already stored. - if (current && merged == current->ckpt) + /// Nothing new: write NOTHING. This is not an optimization -- both writers publish on every + /// snapshot and every seal, and most of those carry a checkpoint the object already has, so + /// issuing the write anyway would mint a fresh incarnation per no-op and turn every other + /// writer's in-flight write into a conflict, for a body byte-identical to the one stored. + if (durable && merged == *durable) { - try - { - check_fence_or_throw(admitted_generation); - } - catch (...) - { - return CkptPublishOutcome::FencedOut; - } - return CkptPublishOutcome::IdenticalSkip; + decline = op.admitted() ? Decline::Identical : Decline::Fenced; + return std::nullopt; } + return encodeRefCkpt(merged); + }; - /// AFTER the read, BEFORE the CAS, on EVERY attempt (spec §3). A generation that moved since - /// admission means this writer's lease incarnation is gone and the body it just merged is - /// stale, so the CAS must never be sent -- and because the check precedes it, nothing was. - try - { - check_fence_or_throw(admitted_generation); - } - catch (...) - { - /// Typed, not propagated: the caller asked "did this land", and "the fence moved, so - /// nothing was sent" is an answer, not a failure of the operation. Only the fence check is - /// wrapped, so nothing else can be mistaken for it. + WriteResult result = op.readModifyWrite(key, decide, policy); + if (std::holds_alternative(result)) + return CkptPublishOutcome::Published; + if (std::holds_alternative(result)) + { + if (decline == Decline::Fenced) return CkptPublishOutcome::FencedOut; - } - - const std::optional expected = - current ? std::optional{current->token} : std::nullopt; - /// Encode before entering the ambiguity catch. Allocation or invariant failures happen before - /// any request is sent and must propagate as themselves, not trigger a needless resolution GET. - const String merged_bytes = encodeRefCkpt(merged); - if (!request_is_admitted()) + /// The durable body already carries this contribution and an attempt of this call was sent to + /// get there, so the write is resolved rather than skipped. + return decisions > 1 ? CkptPublishOutcome::Published : CkptPublishOutcome::IdenticalSkip; + } + if (const auto * gave_up = std::get_if(&result)) + { + /// A lost fence is an expected, transient control signal, so it is a value rather than a + /// throw. It says nothing about the object: the engine reports it both before an attempt is + /// sent and after one is proven durable, so the caller must re-read rather than assume. + /// Every other give-up leaves the contribution unpublished, and the caller must be told that + /// rather than left to assume it landed. + if (gave_up->why == GaveUp::Why::FenceLost) return CkptPublishOutcome::FencedOut; - try - { - if (backend.casPut(key, merged_bytes, expected).outcome == CasOutcome::Committed) - return CkptPublishOutcome::Published; - } - catch (...) - { - /// A thrown CAS response does not say whether the object changed. Never retry its bytes - /// from memory: first point-read the exact mutable object, including its fresh token. If - /// that observation includes this contribution under the semantic join, the write is - /// resolved durable; otherwise that exact observation is the only valid base for a retry. - if (!request_is_admitted()) - return CkptPublishOutcome::FencedOut; - try - { - current = readCkpt(backend, layout, life); - have_current = true; - } - catch (...) - { - throwCasWriteRetryLater("CAS _ckpt for namespace '" + life.ns.string() - + "': a CAS response was ambiguous and the mandatory exact-read resolution failed (" - + getCurrentExceptionMessage(/*with_stacktrace*/ false) + ")"); - } - - try - { - check_fence_or_throw(admitted_generation); - } - catch (...) - { - return CkptPublishOutcome::FencedOut; - } - - if (current && lifeEpochWouldDecrease(current->ckpt, contribution)) - throwLifeEpochDecrease(current->ckpt, contribution, key); - - const RefCkpt resolved_merge = current ? mergeCkpt(current->ckpt, contribution) : contribution; - if (current && resolved_merge == current->ckpt) - return CkptPublishOutcome::Published; - - /// `current` is the exact observation made after the ambiguous response. The next loop - /// iteration retries the SAME contribution against its token (or expected absence), with - /// no blind CAS and no redundant intervening GET. - continue; - } - /// `Conflict`: the incarnation we read is no longer current, so another writer's merge landed - /// first. Nothing of ours was written; re-read and merge against the winner. - current.reset(); - have_current = false; + throwCasWriteRetryLater("CAS _ckpt for namespace '" + life.ns.string() + + "': persistent CAS contention, the checkpoint contribution was not published"); } - - /// Fail closed. Every attempt was all-or-nothing, so there is no partial state -- only an - /// unpublished contribution, which the caller must be told about rather than left to assume. - throwCasWriteRetryLater("CAS _ckpt for namespace '" + life.ns.string() - + "': persistent CAS contention, the checkpoint contribution was not published"); + /// Only `Conflict` and `Refused` are left, and both are an exception. + const String what = "CAS _ckpt for namespace '" + life.ns.string() + "'"; + orThrow(std::move(result), what); + UNREACHABLE(); } -MissingBaseVerdict classifyMissingSampledBase(const Token & sampled_token, const std::optional & current_token) +MissingBaseVerdict classifyMissingSampledBase(const Etag & sampled, const std::optional & current) { - if (current_token && !(*current_token == sampled_token)) + if (current && !(*current == sampled)) return MissingBaseVerdict::RestartRecovery; return MissingBaseVerdict::Corrupted; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.h index 40b3771611f8..99ff787ee81c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefCkpt.h @@ -1,13 +1,10 @@ #pragma once -#include +#include #include #include #include -#include #include -#include #include -#include namespace DB::Cas { @@ -46,96 +43,66 @@ RecoveryGrounding chooseRecoveryGrounding(const std::optional & ca /// Compatible contributions still merge commutatively, but the committed frontier is deliberately not /// an unconstrained CRDT maximum: a cross-epoch pair must be numerically adjacent and carry its seal /// evidence. That makes arbitrary regrouping of a corrupt historical set invalid, while the actual -/// publish protocol remains simple: each token-CAS merges one contribution with the one durable body it +/// publish protocol remains simple: each write merges one contribution with the one durable body it /// just read. /// /// It is therefore NOT where `life_epoch`'s may-not-decrease rule lives, and that is a placement /// decision rather than an omission: a commutative function does not know which of its arguments is the /// durable one, so it cannot tell a decrease from an increase. That rule belongs to `publishCkpt`, which -/// does know (see `checkLifeEpochDoesNotDecrease` in the `.cpp`). +/// does know. RefCkpt mergeCkpt(const RefCkpt & a, const RefCkpt & b); /// What one `publishCkpt` call did. enum class CkptPublishOutcome : uint8_t { - Published, /// the merged body is durable -- this call's CAS committed it + Published, /// the contribution is durable, and this call sent at least one write before it + /// was; which write made it durable -- this one or a competitor's -- is not claimed IdenticalSkip, /// the contribution added nothing to what was already there; NO write was issued - FencedOut, /// the admitted fence generation moved before the CAS; NOTHING was written + FencedOut, /// this actor's admission is gone. Whether its last attempt landed is UNRESOLVED: + /// admission can be lost before the write is sent, and equally after the write is + /// proven durable. Re-read before assuming either way }; -/// The retry bound for `publishCkpt`: an absolute point on a monotonic millisecond clock, plus that -/// clock. Both are required and must be the SAME clock -- the caller passes its own injectable boot -/// clock (`CasRefLedger`'s `boot_ms_fn`), so a test drives the exhaustion arm deterministically -/// instead of sleeping, and a VM suspend cannot shorten the window. -struct CkptDeadline -{ - std::function now_ms; - uint64_t deadline_ms = 0; -}; - -/// Merge `contribution` into `ns`'s `_ckpt` and make the result durable. -/// -/// One attempt is: GET the object -> decode it -> merge -> (identical? return without a CAS) -> -/// re-check the fence -> token-CAS. A CAS conflict means another writer's read-modify-write landed -/// between our GET and our CAS, so the whole attempt repeats against the NEW body -- never against the -/// one we already read, which is the point of re-reading rather than retrying the same bytes. -/// A THROWN CAS response is ambiguous rather than a conflict: exact-read the object, validate its body -/// and token, then check admission again. If the durable body semantically includes the contribution, -/// the write is resolved; otherwise retry the same contribution against that exact-read token. An -/// unreadable resolution fails retry-later, and no path issues two CAS attempts without an intervening -/// exact observation. +/// Merge `contribution` into `life`'s `_ckpt` and make the result durable, as ONE read-modify-write +/// on `op`. /// /// An ABSENT object is created from `contribution` as it stands. Every writer may create it and none /// may complete it: a publisher that knows only the checkpoint creates one that knows only the /// checkpoint, and the field a different writer knows merges in whenever it arrives, in either order. /// That is the whole reason each field is optional rather than defaulted. /// -/// FENCE DISCIPLINE (spec §3, the same value at every site of the trio): `check_fence_or_throw` is -/// re-run on EVERY attempt, AFTER that attempt's read and immediately BEFORE its CAS -- not once at -/// entry. A generation that moved means the mount lease incarnation changed since this work was -/// admitted, so this writer's body is stale even if the fence happens to be live again; the CAS must -/// not be sent. That refusal is returned as `FencedOut` rather than thrown: it is an expected, -/// transient control signal (the same class the request controller reports as `Unresolved`), and the -/// snapshot publisher that calls this sits after a durable PUT where an exception would be worse than -/// a value. NOTHING has been written when it is returned -- the check precedes the CAS. +/// Both DECLINE-TIME verdicts consult `op.admitted()` before they speak, because a writer the fence is +/// about to refuse has landed nothing AT THAT POINT: it is told `FencedOut` rather than `IdenticalSkip` +/// or a corruption verdict. `FencedOut` is returned rather than thrown because it is an expected, +/// transient control signal, and the snapshot publisher that calls this sits after a durable PUT where +/// an exception would be worse than a value. It does NOT promise the object is unchanged -- a lost +/// admission is also reported for a write already proven durable. /// /// FAILS CLOSED, never open: /// - an existing `_ckpt` that does not decode PROPAGATES `CORRUPTED_DATA` and is never overwritten. /// It is the only record of recovery's base and of what cleanup may delete; replacing it with a /// body derived from `contribution` alone would erase the base while leaving a well-formed object /// behind -- corruption laundered into something a reader would trust. -/// - a contribution whose `life_epoch` is BELOW the durable one raises `CORRUPTED_DATA`, checked after -/// that attempt's read and before its merge, so no body is built and no CAS is sent. This is the one -/// refusal that HAS to live here rather than in `mergeCkpt`: only this function knows which side is -/// durable. It is reported as corruption ONLY for a writer the fence still admits -- one the fence -/// is about to refuse gets `FencedOut` like every other refusal here, since it landed nothing. -/// - exhausting the deadline (or the live-lock brake) under persistent conflict throws the -/// retry-later class. No partial state exists to clean up: every attempt either committed the -/// complete merged body or changed nothing. -/// -/// `admitted_generation` is the fence generation the CALLER captured when its work was admitted, and -/// `check_fence_or_throw` is the callback the pool wires from `CasMountRuntime::checkFenceOrThrow` -/// (the ledger never owns a `CasMountRuntime`; it receives the pair the way `CasPlainObjects` does). -/// `admit_request` is independent of that post-read fence contract: when supplied, it is checked -/// immediately before every raw backend request, and refusal returns `FencedOut` without starting it. -CkptPublishOutcome publishCkpt(Backend & backend, const Layout & layout, const NamespaceLifeId & life, - const RefCkpt & contribution, uint64_t admitted_generation, - const std::function & check_fence_or_throw, - const CkptDeadline & deadline, - const std::function & admit_request = {}); +/// - a contribution whose `life_epoch` is BELOW the durable one raises `CORRUPTED_DATA`, decided +/// before the merge, so no body is built and no write is sent. This is the one refusal that HAS to +/// live here rather than in `mergeCkpt`: only this function knows which side is durable. +/// - exhausting the policy under persistent conflict throws the retry-later class. No partial state +/// exists to clean up: every attempt either committed the complete merged body or changed nothing. +CkptPublishOutcome publishCkpt(CasOperation & op, const Layout & layout, const NamespaceLifeId & life, + const RefCkpt & contribution, const Retry & policy = Retry::standard()); -/// One observation of a namespace's `_ckpt`: the decoded body and the incarnation TOKEN it was read -/// at. The token is what the missing-base revalidation adjudicates against, so a reader that keeps -/// only the body cannot apply the rule. +/// One observation of a namespace's `_ckpt`: the decoded body and the incarnation it was read at. The +/// incarnation is what the missing-base revalidation adjudicates against, so a reader that keeps only +/// the body cannot apply the rule. struct CkptSample { RefCkpt ckpt; - Token token; + Etag etag; }; /// Point-read of `life`'s `_ckpt`. `nullopt` means the object is absent (a namespace whose creation has /// not published one yet); a present-but-undecodable object throws `CORRUPTED_DATA`. -std::optional readCkpt(Backend & backend, const Layout & layout, const NamespaceLifeId & life); +std::optional readCkpt(CasOperation & op, const Layout & layout, const NamespaceLifeId & life); /// The verdict of INV-4's three-way revalidation, for the one leg that is not simply "it is there". enum class MissingBaseVerdict : uint8_t @@ -145,14 +112,14 @@ enum class MissingBaseVerdict : uint8_t }; /// Adjudicate a sampled recovery anchor that turned out to be unavailable, by comparing the `_ckpt` -/// token this recovery sampled against the token a fresh re-read observes. The anchor is the +/// incarnation this recovery sampled against the one a fresh re-read observes. The anchor is the /// checkpoint-named snapshot and its retained same-id non-seal log witness; the caller supplies this -/// verdict after either exact GET is absent. +/// verdict after either exact read is absent. /// -/// - token ADVANCED -> `RestartRecovery`. Cleanup legitimately advanced the checkpoint and deleted -/// the previous anchor while we were reading. Nothing is wrong; restart from the newer base -/// (bounded by the caller's own restart budget). -/// - token UNCHANGED -> `Corrupted`. The checkpoint still names an object that is not there, and +/// - incarnation ADVANCED -> `RestartRecovery`. Cleanup legitimately advanced the checkpoint and +/// deleted the previous anchor while we were reading. Nothing is wrong; restart from the newer +/// base (bounded by the caller's own restart budget). +/// - incarnation UNCHANGED -> `Corrupted`. The checkpoint still names an object that is not there, and /// the deletion gate makes that unreachable in an honest run: the named snapshot and matching log /// are both retained. Something deleted a live anchor. /// - `_ckpt` itself ABSENT on the re-read -> `Corrupted` for the same reason, and more bluntly: the @@ -161,7 +128,7 @@ enum class MissingBaseVerdict : uint8_t /// /// Pure, so it is decided the same way at every call site; the caller raises `CORRUPTED_DATA` on /// `Corrupted` with its own context. -MissingBaseVerdict classifyMissingSampledBase(const Token & sampled_token, const std::optional & current_token); +MissingBaseVerdict classifyMissingSampledBase(const Etag & sampled, const std::optional & current); /// INV-4's snapshot-deletion gate: a snapshot is deletable only STRICTLY BELOW the checkpoint. Strict /// rather than at-or-below because the checkpoint names the snapshot a recovery is entitled to fetch diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.cpp index 0f6200c90553..196984a9aa68 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.cpp @@ -1,6 +1,6 @@ #include #include -#include +#include #include #include #include @@ -57,6 +57,11 @@ namespace ProfileEvents extern const Event CASRefAppendDefiniteFailure; extern const Event CASRefAppendSealRejected; extern const Event CASRefAppendOccupantUnreadable; + extern const Event CASRelinkConfirmRefusedRefMutationInFlight; + extern const Event CASRelinkConfirmRefusedLaneWedged; + extern const Event CASRelinkConfirmRefusedLaneBroken; + extern const Event CASRelinkConfirmRefusedStateLockBusy; + extern const Event CASRelinkConfirmRefusedMountCannotSpeak; extern const Event CASRefNeedsRecovery; extern const Event CASRefSweepDeferred; extern const Event CASRefSweepRearmed; @@ -81,14 +86,22 @@ namespace DB::Cas namespace { +/// The relink confirm's logger, resolved once. `LOG_IMPL` evaluates its logger argument BEFORE testing +/// the level, so `getLogger` at the call site would take the global logger-registry lock on every +/// refusal even with tracing off -- and `confirmExactRef` refuses while holding pool-wide append +/// admission, on a path a remote peer drives. +const LoggerPtr & confirmLogger() +{ + static const LoggerPtr logger = getLogger("CasRefLedger"); + return logger; +} + /// Classifies whether an exception thrown out of a ref-table recovery attempt (checkpoint/snapshot/log -/// GETs, or the seal PUT) is a TRANSIENT object-store transport failure worth retrying, -/// vs. a terminal condition (corruption, decode failure, logic error, resource limit) that must fail -/// fast. The recovery reads call the backend directly (not through `ref_request_controller`), so a -/// transient blip surfaces as the object storage's native code -- `S3_ERROR` for the S3 backend, or a -/// socket/timeout/Poco transport code -- NOT the `NETWORK_ERROR` that only the seal PUT's controller -/// re-mints. Retrying only `NETWORK_ERROR` would leave the LIST/GET legs unprotected, which is exactly -/// the exact-read path the recovery retry boundary protects. +/// reads, or the seal write) is a TRANSIENT object-store transport failure worth retrying, vs. a +/// terminal condition (corruption, decode failure, logic error, resource limit) that must fail fast. +/// A read that exhausts its policy surfaces as `NETWORK_ERROR`, but a store's own transport failure +/// can also reach here unclassified -- `S3_ERROR` for the S3 backend, or a socket/timeout/Poco code -- +/// so the set covers both rather than only the one the engine re-mints. bool isTransientRecoveryError(int code) { return code == ErrorCodes::NETWORK_ERROR @@ -183,17 +196,15 @@ Occupant classifyRefLogOccupant(const RootNamespace & ns, const RefTxnId & id, c } CasRefLedger::CasRefLedger( - BackendPtr backend_ptr, + CasRequests & mount_requests_, const Layout & layout_, RefLedgerConfig config_, const CasEventSink & event_sink_, CasRequestBudget cas_request_budget_, String server_root_id_, - std::function controller_boot_ms_fn, std::function live_epoch_fn_, std::function fence_ok_fn_, std::function fence_generation_fn_, - std::function check_fence_or_throw_, std::function boot_ms_now_fn_, std::function may_mutate_, std::function &)> on_impossible_interference_, @@ -201,7 +212,7 @@ CasRefLedger::CasRefLedger( std::function publish_error_hook_, std::function cancel_inflight_builds_, std::function recovery_pre_first_request_hook_for_test_) - : backend(*backend_ptr) + : mount_requests(mount_requests_) , layout(layout_) , config(std::move(config_)) , event_sink(event_sink_) @@ -210,7 +221,6 @@ CasRefLedger::CasRefLedger( , live_epoch_fn(std::move(live_epoch_fn_)) , fence_ok_fn(std::move(fence_ok_fn_)) , fence_generation_fn(std::move(fence_generation_fn_)) - , check_fence_or_throw(std::move(check_fence_or_throw_)) , boot_ms_now_fn(std::move(boot_ms_now_fn_)) , may_mutate(std::move(may_mutate_)) , on_impossible_interference(std::move(on_impossible_interference_)) @@ -219,18 +229,11 @@ CasRefLedger::CasRefLedger( , cancel_inflight_builds(std::move(cancel_inflight_builds_)) , recovery_pre_first_request_hook_for_test(std::move(recovery_pre_first_request_hook_for_test_)) { - /// The ref-log writer path uses the same retry controller and clock seam as the mount's local - /// write fence, so deadline-sensitive tests exercise both paths with one monotonic clock. - /// The raw mount `boot_ms_fn` -- the SAME fake-clock seam the local write fence uses -- is reused - /// here rather than adding a second clock knob; both are monotonic-ms clocks and tests that need - /// deterministic deadline behavior already inject it. - ref_request_controller = std::make_unique(backend_ptr, cas_request_budget, controller_boot_ms_fn); - /// Default backoff sleep for the recovery retry loop (`ensureRefTableRecovered`): sleep in short /// slices and stop early if the mount fence drops (shutdown / lease loss), so teardown never waits /// out a full 30s backoff. This is deliberate, bounded backoff against external object-store I/O - /// failure -- NOT masking a race -- exactly like `CasRequestControl`'s own inter-attempt - /// `threadSleepMs`; the slice loop additionally makes it interruptible, which that one is not. + /// failure -- NOT masking a race -- exactly like the request engine's own inter-attempt sleep; the + /// slice loop additionally makes it interruptible, which that one is not. recovery_retry_sleep_fn = [this](uint64_t total_ms, const std::optional & token) { constexpr uint64_t slice_ms = 200; @@ -244,28 +247,27 @@ CasRefLedger::CasRefLedger( }; } -CasWriteOutcome CasRefLedger::stagingPutIfAbsent(std::string_view key, std::string_view bytes, Token * out_token) -{ - /// The ref lane's mount predicate (`fence_ok_fn` == `Pool::refAppendFenceOk`, with no per-table - /// runtime term) gates every attempt, matching the other staged writes. - return ref_request_controller->putIfAbsentControlled(key, bytes, fence_ok_fn, out_token); -} - -CasOverwriteResult CasRefLedger::stagingConditionalOverwrite(std::string_view key, std::string_view bytes, const Token & expected) +WriteResult CasRefLedger::stagingPutIfAbsent(const String & key, const String & bytes) { - /// The supplied write is controlled by the same retry and mount-fence policy as other staged - /// writes. - return ref_request_controller->putOverwriteControlled(key, bytes, expected, fence_ok_fn); + /// Admitted under the mount fence's CURRENT generation: a staged write belongs to the caller in + /// front of it, not to a transaction admitted earlier, so there is nothing to resume under. + CasOperation op = mount_requests.admit(); + return op.create(key, bytes, Retry::standard()); } -CasOverwriteResult CasRefLedger::stagingPutIfAbsentMutable(std::string_view key, std::string_view bytes) +void CasRefLedger::refuseUnlessAdmitted(const CasOperation & op, std::string_view what) const { - return ref_request_controller->putIfAbsentControlledMutable(key, bytes, fence_ok_fn); + if (op.admitted()) + return; + throwCasTransientUnavailable( + fmt::format("content-addressed pool '{}'", server_root_id), + fmt::format("{}: the operation is no longer admitted -- the mount lease has too little time left " + "for another request, or this table's runtime was detached", what)); } void CasRefLedger::setCasRetrySleepForTest(std::function sleep_fn) { - ref_request_controller->setSleepFnForTest(sleep_fn); + mount_requests.setSleepFnForTest(sleep_fn); recovery_retry_sleep_fn = [recovery_sleep_fn = std::move(sleep_fn)]( uint64_t total_ms, const std::optional &) { @@ -423,15 +425,17 @@ ConfirmAnswer CasRefLedger::confirmExactRef(const RootNamespace & ns, const Stri /// `sweepStalePrecommitsForRead` and `maybeScheduleSnapshotPublish`, the three maintenance calls /// `resolveRef` performs and all three of which can do I/O. /// - /// ONE snapshot across BOTH lane mutexes. `pending`/`leader_active` live under + /// ONE snapshot across BOTH lane mutexes. `pending`/`carved` live under /// `ref_queue_mutex`, the rows and the wedge under `state_mutex`, and the whole point of the /// rules is their CONJUNCTION -- read at different instants they would prove nothing. The lock /// ORDER is the one the rest of this file already establishes (`enforceRefTableCacheBudget` /// nests `state_mutex` under `ref_queue_mutex`, and nothing anywhere takes them the other way /// round). Because admission (`appendRefOps`' `pending.push_back`) happens under - /// `ref_queue_mutex`, an append is either entirely before this snapshot -- and then visible as a - /// pending item -- or entirely after it. There is no interleaving in which a removal is admitted - /// and this function still answers `Yes`. + /// `ref_queue_mutex`, an append is either entirely before this snapshot -- and then visible in + /// `pending` or, once carved, in `carved` -- or entirely after it. There is no interleaving in + /// which a mutation of the asked-about ref is admitted and this function still answers `Yes`; a + /// mutation of another ref may be admitted, and the answer is still right, because it cannot move + /// this ref's row. /// /// What a `Yes` does NOT prove, stated so nobody has to rediscover it: that this runtime's /// recovered view is a COMPLETE replay of the durable log. Completeness is recovery's contract, not @@ -443,6 +447,12 @@ ConfirmAnswer CasRefLedger::confirmExactRef(const RootNamespace & ns, const Stri /// Rule 2 (residency). Direct slot lookup, never a catalog observation or exact-runtime acquisition: /// a read-only query must not let a peer grow this writer's cache or make the next reader pay for a /// recovery it invented. A cold or evicted table is simply unknown here. + /// + /// These two arms are the ONLY refusals in this function that are deliberately not counted: a table + /// this mount has never touched, or has dropped under cache-budget pressure, is ordinary cache + /// behaviour rather than the lane, mount or load condition each counter below separates. Counting + /// it would put a number that moves with cache size next to numbers that describe this writer's + /// health. There is also no runtime here to attribute the refusal to. const auto it = ref_name_slots.find(ns.string()); if (it == ref_name_slots.end()) return ConfirmAnswer::Unknown; @@ -450,16 +460,31 @@ ConfirmAnswer CasRefLedger::confirmExactRef(const RootNamespace & ns, const Stri return ConfirmAnswer::Unknown; RefTableRuntime & rt = *it->second.current; - /// `try_to_lock`, not a blocking acquire: `ensureRefTableRecovered` holds `state_mutex` across its - /// whole exact replay, so blocking here would make a confirm WAIT on someone else's recovery -- - /// up to the full retry envelope -- while holding `ref_queue_mutex`, which is pool-wide append - /// admission. That is the zero-I/O contract broken by proxy: the query would not issue a request, - /// it would merely be paid for by one, and it would stall every table's lane meanwhile. Failing to - /// take the lock is just one more ambiguity, so it answers like every other one. (Same technique, - /// and same non-blocking rationale, as `enforceRefTableCacheBudget`'s candidate loop.) + /// Every refusal below is attributed here, on the node that computed it: `ConfirmAnswer` crosses + /// two interfaces as a three-value enum and stays that way, so the counters and this trace line are + /// the only way a live gate can tell load (`RefMutationInFlight`) from a fault (`LaneWedged`, + /// `LaneBroken`), from contention (`StateLockBusy`), or from this mount losing its claim to the + /// namespace (`MountCannotSpeak`). The invariant to keep when editing below: every + /// `return ConfirmAnswer::Unknown` past this point goes through `refuse`, and the only uncounted + /// refusals in this function are the two residency arms above, which say why. + const auto refuse = [&](ProfileEvents::Event reason, std::string_view why) + { + ProfileEvents::increment(reason); + LOG_TRACE(confirmLogger(), "Relink confirm for ref '{}' in namespace '{}' is unknown: {}", + ref_name, ns.string(), why); + return ConfirmAnswer::Unknown; + }; + + /// `try_to_lock`, not a blocking acquire: this function already holds `ref_queue_mutex`, which is + /// pool-wide append admission, so blocking here would stall EVERY table's lane for as long as + /// whoever holds `state_mutex` keeps it. That is the zero-I/O contract broken by proxy: the query + /// would not issue a request, it would merely wait on one, and it would hold up admission + /// meanwhile. Failing to take the lock is just one more ambiguity, so it answers like every other + /// one. (Same technique, and same non-blocking rationale, as `enforceRefTableCacheBudget`'s + /// candidate loop.) std::unique_lock slock(rt.state_mutex, std::try_to_lock); if (!slock.owns_lock()) - return ConfirmAnswer::Unknown; + return refuse(ProfileEvents::CASRelinkConfirmRefusedStateLockBusy, "the table's state lock is held"); /// Rule 2 (warm). An unrecovered or mid-recovery runtime has an EMPTY `state`, which would read as /// "the ref does not exist" -- knowledge it does not have. `superseded_by_remount` is the same @@ -469,16 +494,40 @@ ConfirmAnswer CasRefLedger::confirmExactRef(const RootNamespace & ns, const Stri if (!rt.recovered || rt.recovery_in_progress || rt.catalog_life_invalidated.load(std::memory_order_acquire) || rt.superseded_by_remount.load(std::memory_order_acquire)) - return ConfirmAnswer::Unknown; - - /// Rule 3 (lane quiescent). A wedge is "an object that may be durable and is not applied" -- it may - /// BE the removal being asked about. A pending item or an active leader tenure is a mutation this - /// table has already admitted; mid-tenure, a chunked flush has committed some of its transactions - /// and not others, and `leader_active` spans the whole tenure, so that partially-durable window is - /// covered too. None of the three says anything about WHICH ref is affected, so all three are - /// table-scoped refusals. - if (rt.lane_state != RefLaneState::Ready || !rt.pending.empty() || rt.leader_active) - return ConfirmAnswer::Unknown; + return refuse(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak, + "the table is unrecovered, recovering, retired or superseded by a remount"); + + /// Rule 3 (no admitted mutation of THIS ref). The hazard is a committed row that lags a transaction + /// of the asked-about ref: the leader does not hold `state_mutex` across the `PUT`, so between + /// "durable" and "installed" that ref's row is stale, and a `Yes` read off it would authorize a + /// receiver to promote over a blob the transaction may already have retired. A mutation of ANOTHER + /// ref cannot change this ref's binding or the blobs its manifest protects, so its row is exactly as + /// authoritative as on an idle lane; refusing for it is what starved two replicas of each other on a + /// slow control plane. Every admitted mutation names its scope (`MutationScope`, recorded at + /// admission under `ref_queue_mutex` and validated against its ops at flush), and it is visible in + /// `pending` from admission to carve and in `carved` from carve to the tenure's exit guard, so "a + /// change of this ref is queued or in flight" is read from those two. The lane states other than + /// `Ready`/`Writing` refuse table-wide: `Wedged` holds a transaction that may be durable, and once + /// its tenure exits nothing but the attempt and the lane state records WHICH ref it touched -- the + /// exit guard clears the carved mirror, and the chunk's items were completed with an error before + /// that -- and `NeedsRecovery`, `Closed`, `Faulted` are fences on the whole view. + /// `Writing` with nothing carved cannot happen; it fails closed. + if (rt.lane_state == RefLaneState::Wedged) + return refuse(ProfileEvents::CASRelinkConfirmRefusedLaneWedged, "the lane holds an unresolved append"); + if (rt.lane_state != RefLaneState::Ready && rt.lane_state != RefLaneState::Writing) + return refuse(ProfileEvents::CASRelinkConfirmRefusedLaneBroken, "the lane is neither Ready nor Writing"); + if (rt.lane_state == RefLaneState::Writing && rt.carved.empty()) + return refuse(ProfileEvents::CASRelinkConfirmRefusedLaneBroken, "the lane is Writing with nothing carved"); + const auto covers = [&](const MutationScope & scope) + { + return scope.kind == MutationScope::Kind::WholeShard || scope.ref_name == ref_name; + }; + for (const auto & item : rt.pending) + if (covers(item->scope)) + return refuse(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight, "a queued mutation names this ref"); + for (const auto & item : rt.carved) + if (covers(item->scope)) + return refuse(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight, "a carved mutation names this ref"); /// Rule 5 (exact row equality) -- the only rule that can answer `No` at all. On a table that passed /// rules 2-4 the committed map is this writer's view, so a missing row or a different `ManifestRef` @@ -503,7 +552,8 @@ ConfirmAnswer CasRefLedger::confirmExactRef(const RootNamespace & ns, const Stri if (!fence_ok_fn() || rt.catalog_life_invalidated.load(std::memory_order_acquire) || rt.superseded_by_remount.load(std::memory_order_acquire)) - return ConfirmAnswer::Unknown; + return refuse(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak, + "this mount no longer holds the namespace's write fence"); return ConfirmAnswer::Yes; } @@ -518,7 +568,7 @@ std::shared_ptr CasRefLedger::lookupRefTableRunti std::shared_ptr CasRefLedger::acquireRefTableRuntime( const NamespaceLifeId & life, uint64_t admitted_generation) { - check_fence_or_throw(admitted_generation); + refuseUnlessAdmitted(mount_requests.resume(admitted_generation), "ref-table runtime install"); std::shared_ptr result; bool generation_moved = false; @@ -552,7 +602,7 @@ std::shared_ptr CasRefLedger::acquireRefTableRunt } if (generation_moved) - check_fence_or_throw(admitted_generation); + refuseUnlessAdmitted(mount_requests.resume(admitted_generation), "ref-table runtime install"); if (identity_conflict) throwCasWriteRetryLater(fmt::format( "CAS namespace '{}': the cached runtime identity changed while publishing catalog life {}; " @@ -571,7 +621,7 @@ std::shared_ptr CasRefLedger::acquireReadableRefT /// runtime. if (auto current = lookupRefTableRuntime(ns)) { - check_fence_or_throw(current->admitted_fence_generation); + refuseUnlessAdmitted(mount_requests.resume(current->admitted_fence_generation), "resident readable runtime"); { std::lock_guard queue_lock(ref_queue_mutex); if (current->removal_admission_closed) @@ -586,9 +636,9 @@ std::shared_ptr CasRefLedger::acquireReadableRefT } const uint64_t admitted_generation = fence_generation_fn(); - check_fence_or_throw(admitted_generation); - const CasRefCatalog::Snapshot first_catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + CasOperation op = mount_requests.resume(admitted_generation); + const CasRefCatalog::Snapshot first_catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "cold readable runtime admission"); first_catalog.life_index.throwIfAmbiguous("CAS cold readable runtime admission"); const auto it = std::find_if(first_catalog.catalog.entries.begin(), first_catalog.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); @@ -607,8 +657,8 @@ std::shared_ptr CasRefLedger::acquireReadableRefT /// after this read are caught by `invalidateRemovedCatalogLife` exactly as before. This second GET /// is deliberately immediately before the queue-locked fence/slot recheck in /// `acquireRefTableRuntime`; the held-handle warm path above pays none. - const CasRefCatalog::Snapshot second_catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + const CasRefCatalog::Snapshot second_catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "cold readable runtime admission"); /// The first read's ambiguity validation does not cover an aliasing incarnation admitted BETWEEN /// the reads; physical life-owned keys use only the incarnation, so an ambiguous second cut must /// refuse admission even when this namespace's own row is untouched. @@ -721,8 +771,8 @@ void CasRefLedger::checkRecoveryStillAdmitted(const RootNamespace & ns, RefTable ProfileEvents::increment(ProfileEvents::CASRefRecoveryCancelled); throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}' was cancelled by a self-remount before the mount " - "fence was re-armed; nothing was written and nothing installed — the next touch recovers under " - "the fresh incarnation", ns.string())); + "fence was re-armed; the last attempt's fate is unresolved and nothing is installed — the next " + "touch recovers under the fresh incarnation", ns.string())); } if (rt.catalog_life_invalidated.load(std::memory_order_acquire)) @@ -739,16 +789,9 @@ void CasRefLedger::checkRecoveryStillAdmitted(const RootNamespace & ns, RefTable "CAS ref-table recovery for namespace '{}': this cached table was superseded by a self-remount " "mid-recovery — retry against the fresh mount incarnation", ns.string())); - /// The FENCE is deliberately NOT checked here, and the omission is the point. `checkFenceOrThrow` - /// asks two things at once -- "is the fence held right now" and "is the generation still mine" -- and - /// the first has no business gating a READ. Most of this walk is reads, and a mount that has - /// transiently lost its lease can still honestly serve them from durable data; refusing at every GET - /// would turn a lease blip into "this table cannot be read at all". - /// - /// The fence gates exactly the three sites that spend it, which is the trio: every `slotOccupy` - /// (through its own `admitted_fence_ok`), the `_ckpt` CAS (inside `publishCkpt`), and the install. - /// A walk that keeps reading after the generation moved simply wastes its own I/O and is then refused - /// at the first of those -- bounded, and strictly better than refusing the reads themselves. + /// The FENCE is deliberately NOT checked here: the walk's own `CasOperation` carries the admitted + /// generation and refuses every request under a generation the fence has moved past, so a second + /// check would only report the same fact from a different sample. } std::optional CasRefLedger::runRecoveryWalkOnce( @@ -765,6 +808,24 @@ std::optional CasRefLedger::runRecoveryWalkOnce( /// under the predecessor even if the same logical name is concurrently rebound. const NamespaceLifeId life = rt.life; + /// ONE operation for the whole walk, resumed under the generation this recovery was admitted at: + /// its reads, its seal creates and its `_ckpt` publishes are all measured against that admission, + /// and a result returning after a fence bump can install nothing. + /// + /// The liveness carries EVERY term `checkRecoveryStillAdmitted` polls except the generation, which + /// is the fence's. That makes the poll and the request gate one rule rather than two that can drift: + /// a walk cancelled between two of its own polls used to keep reading until it reached the next one. + /// The predicate is a bool where the poll throws, so the poll still runs at the boundaries -- it is + /// what turns each of these facts into the right exception, and what LATCHES a cancellation for the + /// caller's retry classification. + CasOperation op = mount_requests.resume(admitted_generation, [&rt, &token] + { + return !(token && token->stopping()) + && !rt.recovery_cancel_requested.load(std::memory_order_acquire) + && !rt.catalog_life_invalidated.load(std::memory_order_acquire) + && !rt.superseded_by_remount.load(std::memory_order_acquire); + }); + /// ---- Step 2: immutable runtime authority and checkpoint ---- /// The runtime was admitted for this exact life before entering recovery, so this walk must not take /// another catalog cut. Retirement invalidates the runtime through `catalog_life_invalidated`, which @@ -776,7 +837,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( if (recovery_pre_first_request_hook_for_test) recovery_pre_first_request_hook_for_test(); checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional sampled_ckpt = readCkpt(backend, layout, life); + const std::optional sampled_ckpt = readCkpt(op, layout, life); std::optional accepted_ckpt_sample = sampled_ckpt; checkRecoveryStillAdmitted(ns, rt, cancelled, token); @@ -795,16 +856,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( try { checkRecoveryStillAdmitted(ns, rt, cancelled, token); - std::function admit_snapshot_base_request; - if (token) - { - admit_snapshot_base_request = [this, &ns, &rt, &cancelled, &token] - { - checkRecoveryStillAdmitted(ns, rt, cancelled, token); - }; - } - CheckpointSnapshotBase base = readCheckpointSnapshotBase( - backend, layout, life, sampled_ckpt->ckpt, admit_snapshot_base_request); + CheckpointSnapshotBase base = readCheckpointSnapshotBase(op, layout, life, sampled_ckpt->ckpt); base_snapshot = std::move(base.snapshot); base_snapshot_bytes = base.bytes; } @@ -818,9 +870,9 @@ std::optional CasRefLedger::runRecoveryWalkOnce( /// unchanged checkpoint turns every helper failure (missing, malformed, or seal) into the /// fail-closed corruption it describes. checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional current = readCkpt(backend, layout, life); - if (classifyMissingSampledBase(sampled_ckpt->token, - current ? std::optional(current->token) : std::nullopt) + const std::optional current = readCkpt(op, layout, life); + if (classifyMissingSampledBase(sampled_ckpt->etag, + current ? std::optional(current->etag) : std::nullopt) == MissingBaseVerdict::RestartRecovery) return std::nullopt; throw; @@ -857,24 +909,6 @@ std::optional CasRefLedger::runRecoveryWalkOnce( builder.applyOne(std::move(txn), encoded_bytes); }; - const auto check_recovery_write_admitted = [this, &ns, &rt, &cancelled, &token](uint64_t expected_generation) - { - checkRecoveryStillAdmitted(ns, rt, cancelled, token); - check_fence_or_throw(expected_generation); - if (rt.catalog_life_invalidated.load(std::memory_order_acquire)) - throwCasWriteRetryLater(fmt::format( - "CAS ref-table recovery for namespace '{}': catalog retirement invalidated life {} " - "before its checkpoint contribution", - rt.life.ns.string(), renderIncarnation(rt.life.incarnation))); - }; - std::function admit_recovery_request; - if (token) - { - admit_recovery_request = [this, &ns, &rt, &cancelled, &token] - { - checkRecoveryStillAdmitted(ns, rt, cancelled, token); - }; - } const auto publish_recovered_frontier = [&](const RefLogTxn & txn) { const RefCkpt contribution{ @@ -884,9 +918,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( .last_epoch_seal = refLogTxnIsEpochSeal(txn) ? std::optional{txn.txn_id} : txn.prev_epoch_seal}; checkRecoveryStillAdmitted(ns, rt, cancelled, token); - if (publishCkptContribution( - life, contribution, admitted_generation, check_recovery_write_admitted, admit_recovery_request) - == CkptPublishOutcome::FencedOut) + if (publishCkptContribution(op, life, contribution) == CkptPublishOutcome::FencedOut) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved before the " "checkpoint could record recovered txn {}-{}; nothing is installed", @@ -897,7 +929,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( /// one successor between lookahead and our CAS, restart from the exact newer checkpoint so the /// installed state covers every transaction its frontier certifies. checkRecoveryStillAdmitted(ns, rt, cancelled, token); - std::optional exact = readCkpt(backend, layout, life); + std::optional exact = readCkpt(op, layout, life); if (!exact || !exact->ckpt.committed_through || *exact->ckpt.committed_through < txn.txn_id) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': exact checkpoint read after publishing " @@ -906,7 +938,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( if (*exact->ckpt.committed_through != txn.txn_id) return false; - /// This recovery itself may advance `_ckpt`. The just-read token and decoded body are the + /// This recovery itself may advance `_ckpt`. The just-read incarnation and decoded body are the /// latest authority cut the private candidate has validated, so the final install boundary /// compares against this sample rather than the original one. accepted_ckpt_sample = std::move(exact); @@ -932,7 +964,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( checkRecoveryStillAdmitted(ns, rt, cancelled, token); const RefTxnId id{epoch, sequence}; - if (const auto got = backend.get(layout.refLogKey(life, id))) + if (const auto got = op.read(layout.refLogKey(life, id), Retry::standard())) { /// `runRecoveryWalkOnce` is the writer recovery entry point even after a process /// restart, when no in-memory attempt survives. A readable birth checkpoint with no @@ -975,19 +1007,19 @@ std::optional CasRefLedger::runRecoveryWalkOnce( ? RefTxnId{id.writer_epoch + 1, 1} : RefTxnId{id.writer_epoch, id.ref_sequence + 1}; checkRecoveryStillAdmitted(ns, rt, cancelled, token); - if (backend.get(layout.refLogKey(life, following_id))) + if (op.read(layout.refLogKey(life, following_id), Retry::standard())) { checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional current = readCkpt(backend, layout, life); - if (!sampled_ckpt || !current || current->token != sampled_ckpt->token) + const std::optional current = readCkpt(op, layout, life); + if (!sampled_ckpt || !current || current->etag != sampled_ckpt->etag) return std::nullopt; const String frontier_description = sampled_frontier ? fmt::format("{}-{}", sampled_frontier->writer_epoch, sampled_frontier->ref_sequence) : "with only a life epoch"; throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref-table recovery for namespace '{}': exact checkpoint {} " - "had two durable successors through {}-{} while its token remained unchanged; " - "the append lane permits at most one unfrontiered transaction", + "had two durable successors through {}-{} while its incarnation remained " + "unchanged; the append lane permits at most one unfrontiered transaction", ns.string(), frontier_description, following_id.writer_epoch, following_id.ref_sequence); } @@ -1039,12 +1071,12 @@ std::optional CasRefLedger::runRecoveryWalkOnce( /// re-read the exact mutable checkpoint to distinguish a concurrent frontier movement /// from durable-data loss under an unchanged authority token. checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional current = readCkpt(backend, layout, life); - if (!current || current->token != sampled_ckpt->token) + const std::optional current = readCkpt(op, layout, life); + if (!current || current->etag != sampled_ckpt->etag) return std::nullopt; throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS ref-table recovery for namespace '{}': committed log id {}-{} is absent while " - "the exact checkpoint frontier {}-{} and its token remain unchanged", + "the exact checkpoint frontier {}-{} and its incarnation remain unchanged", ns.string(), id.writer_epoch, id.ref_sequence, sampled_frontier->writer_epoch, sampled_frontier->ref_sequence); } @@ -1057,7 +1089,7 @@ std::optional CasRefLedger::runRecoveryWalkOnce( if (sampled_seal_is_after_hole) { checkRecoveryStillAdmitted(ns, rt, cancelled, token); - if (backend.get(layout.refLogKey(life, *sampled_ckpt->ckpt.last_epoch_seal))) + if (op.read(layout.refLogKey(life, *sampled_ckpt->ckpt.last_epoch_seal), Retry::standard())) { ProfileEvents::increment(ProfileEvents::CASRefRecoveryStreamHole); hole_detail = fmt::format( @@ -1115,92 +1147,91 @@ std::optional CasRefLedger::runRecoveryWalkOnce( validateEpochSealGrammarContextual(seal_txn, *sampled_ckpt->ckpt.life_epoch); const String seal_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(seal_txn)); - /// Presented on EVERY attempt: the generation this recovery was admitted under, never the - /// current one. A seal written by an incarnation that no longer owns the namespace is a write - /// from a dead mount, and refusing pre-attempt leaves the slot provably untouched. - const auto admitted_fence_ok = [this, &rt, admitted_generation, &token] - { - return fence_ok_fn() - && !(token && token->stopping()) - && !rt.catalog_life_invalidated.load(std::memory_order_acquire) - && !rt.superseded_by_remount.load(std::memory_order_acquire) - && fence_generation_fn() == admitted_generation; - }; - + /// One bounded attempt under the generation this recovery was admitted at, never the + /// current one: a seal written by an incarnation that no longer owns the namespace is a + /// write from a dead mount, and refusing pre-attempt leaves the slot provably untouched. checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const SlotOccupyResult occupied = - ref_request_controller->slotOccupy( - layout.refLogKey(life, id), seal_bytes, admitted_fence_ok); + const WriteResult sealed = op.create(layout.refLogKey(life, id), seal_bytes, Retry::once()); - switch (occupied.kind) + if (const auto * conflict = std::get_if(&sealed)) { - case SlotOccupyResult::Kind::Created: + const auto * occupant_object = std::get_if(&conflict->seen); + if (!occupant_object) + /// The create lost the slot and the settling read could not say to what. Continuing + /// would expose a dead epoch that may or may not be closed. + throwCasWriteRetryLater(fmt::format( + "CAS ref-table recovery for namespace '{}': the epoch seal at {}-{} lost its slot " + "to an occupant the settling read could not observe; the table stays unrecovered " + "rather than being exposed with a dead epoch that may or may not be closed", + ns.string(), id.writer_epoch, id.ref_sequence)); + + /// Someone reached this slot first. A DECODE FAILURE here propagates: an object at a + /// key this namespace owns that is not a transaction of this namespace at this id is + /// corruption or a protocol breach, and the one thing recovery must not do is guess + /// past it. + RefLogTxn occupant = decodeRefLogTxn( + openObject(FormatId::RefLog, occupant_object->bytes), ns.string(), id); + const bool occupant_is_seal = refLogTxnIsEpochSeal(occupant); + const RefLogTxn frontier_txn = occupant; + apply_one(std::move(occupant), occupant_object->bytes.size()); + if (!publish_recovered_frontier(frontier_txn)) + return std::nullopt; + if (occupant_is_seal) { - /// The epoch is ours to close and now IS closed. Apply our own seal to the candidate: - /// it is a durable transaction of this stream like any other, and the next recovery - /// will read it back exactly where we put it. - RefLogTxn applied = seal_txn; - apply_one(std::move(applied), seal_bytes.size()); - ProfileEvents::increment(ProfileEvents::CASRefRecoveryEpochSealed); - if (!publish_recovered_frontier(seal_txn)) - return std::nullopt; + /// A concurrent recoverer closed this epoch (or our own earlier attempt did, and + /// its acknowledgment was lost). Either way the epoch is closed by a seal that is + /// as good as ours -- adopt it and continue. Contesting a peer's CORRECT write is + /// how two recoverers of the same table turn a designed race into an incident. + ProfileEvents::increment(ProfileEvents::CASRefRecoveryEpochSealAdopted); ++epoch; sequence = 1; slot_attempts_this_epoch = 0; - break; - } - case SlotOccupyResult::Kind::Occupied: - { - /// Someone reached this slot first. A DECODE FAILURE here propagates: an object at a - /// key this namespace owns that is not a transaction of this namespace at this id is - /// corruption or a protocol breach, and the one thing recovery must not do is guess - /// past it. - RefLogTxn occupant = decodeRefLogTxn( - openObject(FormatId::RefLog, occupied.occupant_bytes), ns.string(), id); - const bool occupant_is_seal = refLogTxnIsEpochSeal(occupant); - const RefLogTxn frontier_txn = occupant; - apply_one(std::move(occupant), occupied.occupant_bytes.size()); - if (!publish_recovered_frontier(frontier_txn)) - return std::nullopt; - if (occupant_is_seal) - { - /// A concurrent recoverer closed this epoch (or our own earlier attempt did, and - /// its acknowledgment was lost). Either way the epoch is closed by a seal that is - /// as good as ours -- adopt it and continue. Contesting a peer's CORRECT write is - /// how two recoverers of the same table turn a designed race into an incident. - ProfileEvents::increment(ProfileEvents::CASRefRecoveryEpochSealAdopted); - ++epoch; - sequence = 1; - slot_attempts_this_epoch = 0; - } - else - { - /// A STRAGGLER: an ordinary transaction of the dead epoch landed at `T+1` between - /// our read and our create. Adopt it, advance `T` by exactly ONE, and try the seal - /// again at the NEW `T+1`. Never mint `T+2` around it: ids are state-derived - /// (INV-1/INV-2), and writing past an occupied slot puts a hole in the durable - /// stream that no later reader can distinguish from a lost object. - ProfileEvents::increment(ProfileEvents::CASRefRecoveryStragglerAdopted); - ++sequence; - } - break; } - case SlotOccupyResult::Kind::Unresolved: + else { - /// The store will not say whether our seal landed. There is no honest way to continue: - /// exposing the table would publish a dead epoch that may or may not be closed, and - /// re-deriving the slot later needs a fresh read anyway. Fail this attempt into the - /// caller's transient-retry loop, which either succeeds on a later attempt or spends - /// its budget and leaves the table unrecovered. - throwCasWriteRetryLater(fmt::format( - "CAS ref-table recovery for namespace '{}': the epoch seal at {}-{} is UNRESOLVED " - "({}); the table stays unrecovered rather than being exposed with a dead epoch that " - "may or may not be closed", - ns.string(), id.writer_epoch, id.ref_sequence, - unresolvedProvesNothingWasSent(occupied.unresolved_reason) - ? "nothing was sent" : "the outcome of the attempt is unknown")); + /// A STRAGGLER: an ordinary transaction of the dead epoch landed at `T+1` between + /// our read and our create. Adopt it, advance `T` by exactly ONE, and try the seal + /// again at the NEW `T+1`. Never mint `T+2` around it: ids are state-derived, and + /// writing past an occupied slot puts a hole in the durable stream that no later + /// reader can distinguish from a lost object. + ProfileEvents::increment(ProfileEvents::CASRefRecoveryStragglerAdopted); + ++sequence; } + continue; } + + if (const auto * gave_up = std::get_if(&sealed)) + /// The store will not say whether our seal landed. There is no honest way to continue: + /// exposing the table would publish a dead epoch that may or may not be closed, and + /// re-deriving the slot later needs a fresh read anyway. Fail this attempt into the + /// caller's transient-retry loop, which either succeeds on a later attempt or spends + /// its budget and leaves the table unrecovered. + throwCasWriteRetryLater(fmt::format( + "CAS ref-table recovery for namespace '{}': the epoch seal at {}-{} is UNRESOLVED " + "({}); the table stays unrecovered rather than being exposed with a dead epoch that " + "may or may not be closed", + ns.string(), id.writer_epoch, id.ref_sequence, + gave_up->sent_any ? "the outcome of the attempt is unknown" : "nothing was sent")); + + if (const auto * refused = std::get_if(&sealed)) + /// The store proved this attempt never applied. The slot is untouched, so the epoch is + /// still unclosed and the table still may not be exposed. + throwCasWriteRetryLater(fmt::format( + "CAS ref-table recovery for namespace '{}': the store refused the epoch seal at " + "{}-{} ({}); the table stays unrecovered with its dead epoch still open", + ns.string(), id.writer_epoch, id.ref_sequence, refused->message)); + + /// The epoch is ours to close and now IS closed. Apply our own seal to the candidate: + /// it is a durable transaction of this stream like any other, and the next recovery + /// will read it back exactly where we put it. + RefLogTxn applied = seal_txn; + apply_one(std::move(applied), seal_bytes.size()); + ProfileEvents::increment(ProfileEvents::CASRefRecoveryEpochSealed); + if (!publish_recovered_frontier(seal_txn)) + return std::nullopt; + ++epoch; + sequence = 1; + slot_attempts_this_epoch = 0; } } @@ -1214,16 +1245,15 @@ std::optional CasRefLedger::runRecoveryWalkOnce( .committed_through = last_epoch_seal, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = last_epoch_seal}; - if (publishCkptContribution( - life, contribution, admitted_generation, check_recovery_write_admitted, admit_recovery_request) - == CkptPublishOutcome::FencedOut) + if (publishCkptContribution(op, life, contribution) == CkptPublishOutcome::FencedOut) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved before the " - "checkpoint could record the epoch seal {}-{}; nothing was written and nothing is installed", + "checkpoint could record the epoch seal {}-{}; the last attempt's fate is unresolved and " + "nothing is installed", ns.string(), last_epoch_seal->writer_epoch, last_epoch_seal->ref_sequence)); checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional exact = readCkpt(backend, layout, life); + const std::optional exact = readCkpt(op, layout, life); if (!exact || exact->ckpt.committed_through != private_frontier || exact->ckpt.last_epoch_seal != last_epoch_seal) return std::nullopt; accepted_ckpt_sample = exact; @@ -1231,12 +1261,12 @@ std::optional CasRefLedger::runRecoveryWalkOnce( /// Final authority validation is the recovery linearization point. The last exact log probe fixed /// the private cut, but another actor could have changed `_ckpt` immediately afterwards. Install - /// only when both the exact object token and its complete decoded body remain equal to the latest - /// authority sample this private candidate accepted. + /// only when both the exact object incarnation and its complete decoded body remain equal to the + /// latest authority sample this private candidate accepted. checkRecoveryStillAdmitted(ns, rt, cancelled, token); - const std::optional final_ckpt = readCkpt(backend, layout, life); + const std::optional final_ckpt = readCkpt(op, layout, life); if (!final_ckpt || !accepted_ckpt_sample - || final_ckpt->token != accepted_ckpt_sample->token + || final_ckpt->etag != accepted_ckpt_sample->etag || final_ckpt->ckpt != accepted_ckpt_sample->ckpt) return std::nullopt; checkRecoveryStillAdmitted(ns, rt, cancelled, token); @@ -1268,18 +1298,29 @@ NamespaceLifeId CasRefLedger::resolveNamespaceLife( const RootNamespace & ns, uint64_t admitted_generation, uint64_t live_epoch, bool * lifecycle_refusal) { - /// Bounded exactly like `CasRefCatalog::casUpdateImpl`'s own live-lock brake, but against THIS - /// loop's re-read cycle only -- every primitive called below already bounds its OWN retry against - /// the catalog's single contended object. A duel between two openers (one creating, one - /// reconciling a stale creator) converges in a handful of rounds; this guards only against a - /// pathologically un-converging sequence of them. + /// The SECONDARY bound. Wall-clock time is already bounded by the frozen policy below, which + /// every verb of every iteration shares; this cap exists so a sequence that somehow converges on + /// neither a resolution nor the deadline still ends. A duel between two openers (one creating, one + /// reconciling a stale creator) converges in a handful of rounds. static constexpr size_t kMaxResolveAttempts = 32; - const CkptDeadline deadline{boot_ms_now_fn, boot_ms_now_fn() + cas_request_budget.operation_deadline_ms}; const CreatorFence our_fence{server_root_id, live_epoch, admitted_generation}; + CasOperation op = mount_requests.resume(admitted_generation); + + /// ONE bound for the whole loop, frozen before the first iteration: every read and every protocol + /// call below shares this deadline, so a namespace whose catalog entry keeps moving gives up + /// retry-later within one standard window rather than spending a fresh window per verb per + /// iteration. The paced re-read below is a bare sleep that does not consult the deadline, so the + /// loop can sleep one backoff (at most 5 s) past it before the next read refuses to start. + const Retry policy = op.freeze(Retry::standard()); for (size_t attempt = 0; attempt < kMaxResolveAttempts; ++attempt) { - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + /// Paces ONLY the re-reads a competing actor forced. A `Live` outcome is this open's own + /// success and its single confirming re-read is not contention, so it is never delayed. The + /// argument is the number of collisions so far, so the first one waits the shortest draw. + const auto paceReRead = [&] { op.pause(Retry::backoff(static_cast(attempt) + 1)); }; + + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout, policy); const auto it = std::find_if(snap.catalog.entries.begin(), snap.catalog.entries.end(), [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); @@ -1291,12 +1332,14 @@ NamespaceLifeId CasRefLedger::resolveNamespaceLife( /// catalog on the next loop iteration to learn it -- one extra GET, paid once per birth, /// never per write. const auto outcome = CasRefCatalog::createNamespace( - backend, layout, config.gc_shards, ns, our_fence, - admitted_generation, check_fence_or_throw, deadline); + op, layout, config.gc_shards, ns, our_fence, policy); if (outcome == CasRefCatalog::NamespaceCreationOutcome::FencedOut) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved while " - "birthing its catalog entry; nothing was written and nothing installed", ns.string())); + "birthing its catalog entry; the last attempt's fate is unresolved and nothing is " + "installed", ns.string())); + if (outcome == CasRefCatalog::NamespaceCreationOutcome::Superseded) + paceReRead(); continue; /// Live or Superseded: re-read (Superseded means a DIFFERENT actor won birth) } @@ -1321,13 +1364,15 @@ NamespaceLifeId CasRefLedger::resolveNamespaceLife( /// not dead), so this case is checked FIRST and unconditionally, before any terminality probe. if (it->creator->server_root_id == server_root_id && it->creator->writer_epoch == live_epoch) { - const auto outcome = CasRefCatalog::completeCreation( - backend, layout, *it, admitted_generation, check_fence_or_throw, deadline); + const auto outcome = CasRefCatalog::completeCreation(op, layout, *it, policy); if (outcome == CasRefCatalog::NamespaceCreationOutcome::FencedOut) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved while " - "resuming its own stalled creation; nothing was written and nothing installed", + "resuming its own stalled creation; the last attempt's fate is unresolved and nothing " + "is installed", ns.string())); + if (outcome == CasRefCatalog::NamespaceCreationOutcome::Superseded) + paceReRead(); continue; /// Live or Superseded: re-read either way } @@ -1335,29 +1380,32 @@ NamespaceLifeId CasRefLedger::resolveNamespaceLife( /// fresh read -- never busy-loop this instant) or provably dead, in which case reconciliation /// steals it onto our own fence and this open resumes `completeCreation` itself. const auto reconcile_outcome = CasRefCatalog::reconcileStaleCreator( - backend, layout, *it, our_fence, - [this](const CreatorFence & f) { return isCreatorFenceTerminal(backend, layout, f.server_root_id, f.writer_epoch); }, - admitted_generation, check_fence_or_throw); + op, layout, *it, our_fence, + [&](const CreatorFence & f) { return isCreatorFenceTerminal(op, layout, f.server_root_id, f.writer_epoch, policy); }, + policy); switch (reconcile_outcome) { case CasRefCatalog::ReconcileCreatorOutcome::FencedOut: - /// Review I6: our OWN mount fence moved before the steal CAS -- nothing was written, and + /// Our OWN mount fence moved before the steal CAS -- the CAS's fate is unresolved, and /// this mount is the wrong actor to retry (its incarnation is gone). throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved while " - "reconciling a stalled foreign creator; nothing was written and nothing installed", + "reconciling a stalled foreign creator; the last attempt's fate is unresolved and " + "nothing is installed", ns.string())); case CasRefCatalog::ReconcileCreatorOutcome::Reconciled: { CatalogEntry resumed = *it; resumed.creator = our_fence; - const auto outcome = CasRefCatalog::completeCreation( - backend, layout, resumed, admitted_generation, check_fence_or_throw, deadline); + const auto outcome = CasRefCatalog::completeCreation(op, layout, resumed, policy); if (outcome == CasRefCatalog::NamespaceCreationOutcome::FencedOut) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': the mount incarnation moved while " - "completing a reconciled creation; nothing was written and nothing installed", + "completing a reconciled creation; the last attempt's fate is unresolved and " + "nothing is installed", ns.string())); + if (outcome == CasRefCatalog::NamespaceCreationOutcome::Superseded) + paceReRead(); continue; /// Live or Superseded: re-read either way } case CasRefCatalog::ReconcileCreatorOutcome::CreatorFenceStillLive: @@ -1367,7 +1415,9 @@ NamespaceLifeId CasRefLedger::resolveNamespaceLife( "CAS ref-table recovery for namespace '{}': its catalog entry is still Creating " "under a creator fence that is not yet provably dead; retry later", ns.string())); case CasRefCatalog::ReconcileCreatorOutcome::EntryChanged: - continue; /// token-exactness failed: someone else already moved this entry; re-read + /// Token-exactness failed: someone else already moved this entry. Pace before re-reading. + paceReRead(); + continue; } } @@ -1419,16 +1469,16 @@ void CasRefLedger::ensureRefTableRecovered( }); /// ---- Step 1: capture the admitted generation, ONCE ---- - /// The trio (spec §3, codex finding 7): this ONE value is what the walk's every `slotOccupy` and its - /// `_ckpt` CAS present, and what the install below presents one final time. One capture point, three - /// checks, no re-derivation -- a value re-read midway would let a recovery that lost the mount - /// "recover" its right to write by observing a fresh incarnation it was never admitted under. + /// This ONE value admits the walk's operation, so every + /// request it makes is measured against it, and the install below presents it one final time. One + /// capture point, no re-derivation -- a value re-read midway would let a recovery that lost the + /// mount "recover" its right to write by observing a fresh incarnation it was never admitted under. /// /// Captured for the WHOLE call, not per attempt, for the same reason: the transient-retry loop below /// exists for object-store blips, and a generation that moved is not one. The loop refuses to /// re-drive under a moved generation (below), so the budget is never burned on a doomed retry. const uint64_t admitted_generation = rt.admitted_fence_generation; - check_fence_or_throw(admitted_generation); + refuseUnlessAdmitted(mount_requests.resume(admitted_generation), "ref recovery walk admission"); /// Preserve this runtime's exact writer identity across the unlocked walk. The runtime stays in /// `NeedsRecovery` until the same lock installs a result, so no later append can replace it here. const std::optional retained_attempt = rt.append_attempt; @@ -1510,7 +1560,9 @@ void CasRefLedger::ensureRefTableRecovered( /// of time to come back, and a recovery whose window straddled a fence bump describes a /// mount incarnation that no longer owns this namespace. It must publish NOTHING: the /// table stays unrecovered and the next touch recovers it properly. - check_fence_or_throw(admitted_generation); + if (recovery_install_probe_for_test) + recovery_install_probe_for_test(); + refuseUnlessAdmitted(mount_requests.resume(admitted_generation), "ref recovery install"); if (rt.catalog_life_invalidated.load(std::memory_order_acquire)) throwCasWriteRetryLater(fmt::format( "CAS ref-table recovery for namespace '{}': catalog retirement invalidated life {} " @@ -1551,6 +1603,13 @@ void CasRefLedger::ensureRefTableRecovered( || !isTransientRecoveryError(code)) throw; /// a latched terminal case, or a non-transient failure -- fail fast + /// A cancellation now ends the walk through the operation's liveness, which refuses + /// silently -- so the exception that arrives here carries a transport code and `cancelled` + /// is not yet latched. Take the latch before the backoff rather than one whole attempt + /// later: the self-remount barrier is waiting for this recovery to stop. + if (rt.recovery_cancel_requested.load(std::memory_order_acquire)) + checkRecoveryStillAdmitted(ns, rt, cancelled, token); + const uint64_t elapsed_ms = boot_ms_now_fn() - recovery_start_ms; /// Fail closed BEFORE sleeping: budget spent, mount fence lost, this runtime superseded by a /// self-remount, or the incarnation that admitted this recovery has moved (retrying under a @@ -1563,9 +1622,9 @@ void CasRefLedger::ensureRefTableRecovered( || fence_generation_fn() != admitted_generation) throw; - /// Saturating `initial << recovery_retry_num` (mirrors `CasRequestController::backoffBefore - /// Attempt`): `initial > cap >> n` implies the unshifted product already exceeds the cap, so - /// return the cap without ever computing an overflowing/UB shift for large retry counts. + /// Saturating `initial << recovery_retry_num`: `initial > cap >> n` implies the unshifted + /// product already exceeds the cap, so return the cap without ever computing an + /// overflowing/UB shift for large retry counts. const uint64_t init_backoff = cas_request_budget.recovery_retry_initial_backoff_ms; const uint64_t cap_backoff = cas_request_budget.recovery_retry_max_backoff_ms; const uint64_t backoff_ms = (recovery_retry_num >= 63 || init_backoff > (cap_backoff >> recovery_retry_num)) @@ -2064,7 +2123,7 @@ RefTxnId CasRefLedger::appendRefOpsOnRuntime( /// Build the responsibility set (its own `item`) BEFORE publishing the baton, so becoming /// leader contains NO throwing operation once `leader_active` is set: the only allocation is /// this first `push_back`, done here while still holding `lk` and NOT yet leader. If it throws - /// (a `bad_alloc` at the pre-tenure point; codex stage-1 review, Important), the baton is never + /// (a `bad_alloc` at the pre-tenure point), the baton is never /// taken -- but `item` is already in `pending` (pushed above), so it must be un-enqueued before /// propagating, else a future leader would carve an item whose `build_ops` closure died with /// this unwinding caller (the same use-after-free the exit guard prevents post-publication). @@ -2162,6 +2221,11 @@ void CasRefLedger::completeOwnedItemsAndReleaseLeadership( /// no-op for them; it only matters for an item the leader owned but never got to carve. std::erase(rt->pending, owned); } + /// The tenure is over: every carved item is completed (above, or by its chunk's commit) and its + /// effect is either installed, or its failure is recorded by the lane state (`Wedged` for an + /// ambiguous `PUT`, `NeedsRecovery` for a durable-but-not-installed chunk), so the confirm no + /// longer needs to see it. + rt->carved.clear(); rt->leader_active = false; rt->cv.notify_all(); } @@ -2203,7 +2267,7 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrcatalog_life_invalidated.load(std::memory_order_acquire) - && !rt->superseded_by_remount.load(std::memory_order_acquire) - && fence_generation_fn() == admitted; - }; + return !rt->catalog_life_invalidated.load(std::memory_order_acquire) + && !rt->superseded_by_remount.load(std::memory_order_acquire); + }); - SlotOccupyResult occupied; + std::optional attempted; try { if (wedge_before_slot_occupy_hook_for_test) wedge_before_slot_occupy_hook_for_test(); - occupied = ref_request_controller->slotOccupy(wedge.key, wedge.bytes, admitted_fence_ok); + attempted = op.create(wedge.key, wedge.bytes, Retry::once()); } catch (...) { - /// `ambiguous-then-definite`, the model-proven control. `slotOccupy` rethrows only a definite - /// refusal of THIS attempt (a whitelisted synchronous rejection, or a deterministic local - /// failure) -- and a definite refusal of a LATER attempt proves nothing whatsoever about the - /// EARLIER ambiguous one, which may still be in flight or may already have landed. So the lane - /// stays wedged: unwedging here is exactly how an acked-then-lost transaction gets written - /// around. The id is not consumed either, so the next attempt re-derives the SAME one. + /// `ambiguous-then-definite`, the model-proven control. A deterministic local failure surfaces + /// unchanged, and it proves nothing whatsoever about the EARLIER ambiguous attempt, which may + /// still be in flight or may already have landed. So the lane stays wedged: unwedging here is + /// exactly how an acked-then-lost transaction gets written around. The id is not consumed + /// either, so the next attempt re-derives the SAME one. result.kind = WedgeResolution::StillWedged; result.survivor_error = makeCasWriteRetryLaterExceptionPtr(fmt::format( - "CAS ref-log append for namespace '{}': the bounded retry of wedged txn {}-{} was definitively " - "refused ({}), which says nothing about the earlier ambiguous attempt — the lane stays wedged", + "CAS ref-log append for namespace '{}': the bounded retry of wedged txn {}-{} failed ({}), " + "which says nothing about the earlier ambiguous attempt — the lane stays wedged", ns.string(), wedge.txn_id.writer_epoch, wedge.txn_id.ref_sequence, getCurrentExceptionMessage(/*with_stacktrace*/ false))); return result; } + /// A store refusal is the same control as the throw above: it proves only its OWN attempt never + /// applied, and the earlier ambiguous one is what the wedge is about. + if (const auto * refused = std::get_if(&*attempted)) + { + result.kind = WedgeResolution::StillWedged; + result.survivor_error = makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}': the bounded retry of wedged txn {}-{} was definitively " + "refused ({}), which says nothing about the earlier ambiguous attempt — the lane stays wedged", + ns.string(), wedge.txn_id.writer_epoch, wedge.txn_id.ref_sequence, refused->message)); + return result; + } + /// ---- Classify the occupant OFF the lock: pure, and the decode allocates ---- /// The three-way `mine | successor's seal | foreign` adjudication is the CALLER's job by - /// construction (`slotOccupy` never compares bytes), and "mine" means BYTE EQUALITY -- never a - /// shape or generation match, which is the aliasing the phase-0 model rejected. - const Occupant occupant = occupied.kind == SlotOccupyResult::Kind::Occupied - ? classifyRefLogOccupant(ns, wedge.txn_id, occupied.occupant_bytes, wedge.bytes) + /// construction (the engine never compares bytes for meaning), and "mine" means BYTE EQUALITY -- + /// never a shape or generation match, which is the aliasing the phase-0 model rejected. + const auto * conflict = std::get_if(&*attempted); + const auto * occupant_object = conflict ? std::get_if(&conflict->seen) : nullptr; + const auto * gave_up = std::get_if(&*attempted); + /// A conflict whose settling read observed nothing is as unresolved as a give-up: the key holds + /// something this call could not name, so nothing may be adopted or unwedged from it. + const bool unresolved = gave_up || (conflict && !occupant_object); + const Occupant occupant = occupant_object + ? classifyRefLogOccupant(ns, wedge.txn_id, occupant_object->bytes, wedge.bytes) : Occupant::NotOccupied; const bool exact_attempt_is_durable - = occupied.kind == SlotOccupyResult::Kind::Created || occupant == Occupant::Ours; + = std::holds_alternative(*attempted) || occupant == Occupant::Ours; /// Caller holds `state_mutex`. Keeping the identity predicate in one place is part of the safety /// rule: adding a frontier must not create yet another subtly different notion of "same attempt". const auto same_wedge_under_lock = [&] @@ -2334,26 +2413,6 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrcatalog_life_invalidated.load(std::memory_order_acquire) - || rt->superseded_by_remount.load(std::memory_order_acquire)) - throwCasWriteRetryLater(fmt::format( - "CAS namespace '{}': its captured runtime was retired before wedged-frontier publication", - rt->life.ns.string())); - - bool same_wedge = false; - { - std::lock_guard lock(rt->state_mutex); - same_wedge = same_wedge_under_lock(); - } - if (!same_wedge) - throwCasWriteRetryLater(fmt::format( - "CAS namespace '{}': the captured wedge changed before frontier publication", - rt->life.ns.string())); - }; - const RefCkpt frontier{ .life_epoch = std::nullopt, .committed_through = wedge.txn_id, @@ -2364,8 +2423,7 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrlife, frontier, wedge.admitted_fence_generation, check_wedge_admitted); + frontier_outcome = publishCkptContribution(op, rt->life, frontier); } catch (...) { @@ -2399,20 +2457,11 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrstate_mutex); - /// The generation this attempt was admitted under, presented back. `checkFenceOrThrow` reports a - /// moved incarnation by throwing; it is CAUGHT here rather than propagated, because the caller's - /// retry classification keys on the retry-later error class and a routine lease blip must not - /// reach it as a hard failure. Nothing is installed and nothing is unwedged either way, which is - /// the whole meaning of INERT here. - bool fence_moved = false; - try - { - check_fence_or_throw(wedge.admitted_fence_generation); - } - catch (...) - { - fence_moved = true; - } + /// The generation this attempt was admitted under, presented back. A moved incarnation is a + /// verdict here, not an exception: the caller's retry classification keys on the retry-later + /// error class and a routine lease blip must not reach it as a hard failure. Nothing is + /// installed and nothing is unwedged either way, which is the whole meaning of INERT here. + const bool fence_moved = !op.admitted(); /// The remount half of the same question, checked separately because the two are independent /// facts even though today's ordering makes one imply the other: `quiesceRefTablesForRemount` @@ -2442,14 +2491,14 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrsent_any ? Reason::RefusedPreAttempt : Reason::ResolveFoundNothing; result.kind = WedgeResolution::StillWedged; } @@ -2644,10 +2693,8 @@ CasRefLedger::resolveWedgeOnce(const RootNamespace & ns, const std::shared_ptrref_name != scope_ref) + return &op.old_binding->ref_name; + if (op.new_binding && op.new_binding->ref_name != scope_ref) + return &op.new_binding->ref_name; + return nullptr; + } + if (op.kind == RefOpKind::SetPublishedAt && op.ref_name != scope_ref) + return &op.ref_name; + return nullptr; +} +} + void CasRefLedger::flushRefBatch(const RootNamespace & ns, const std::shared_ptr & rt, std::vector> & owned_items) { @@ -2863,16 +2935,18 @@ void CasRefLedger::flushRefBatch(const RootNamespace & ns, const std::shared_ptr /// `seen_refs`/`batch` growth and only recorded the batch into `owned_items` afterwards, so any throw /// after the first pop stranded already-popped items -- neither in `pending` nor in `owned_items` -- /// and their waiters hung forever. Instead: - /// PLAN (may throw, mutates NOTHING): under `ref_queue_mutex`, scan `pending` WITHOUT popping and - /// build the selection count, reserving every container (`batch`, `owned_items`) that the publish - /// below grows. A throw here leaves `pending`/`owned_items` byte-for-byte unchanged, so the - /// leadership-exit guard completes only the leader's own item and the untouched followers stay - /// queued for a later leader. + /// PLAN (may throw, mutates no CONTENT): under `ref_queue_mutex`, scan `pending` WITHOUT popping + /// and build the selection count, reserving CAPACITY in every container (`batch`, `owned_items`, + /// `rt->carved`) that the publish below grows -- a capacity change, never a size or content + /// change. A throw here leaves `pending`/`owned_items`/`carved` byte-for-byte unchanged in + /// content, so the leadership-exit guard completes only the leader's own item and the untouched + /// followers stay queued for a later leader. /// PUBLISH (no-throw): still under the SAME continuous `ref_queue_mutex` hold (no TOCTOU by /// construction), pop the selected front items and append them to `batch` and `owned_items` using /// only non-throwing operations (capacity pre-reserved; `shared_ptr` copies and `deque::pop_front` - /// never throw). ProfileEvents increments are deferred past the plan so the plan is literally - /// non-mutating. + /// never throw). ProfileEvents increments are deferred past the plan so the plan performs no + /// observable mutation beyond the reserved capacity above. The same items are appended to + /// `rt->carved`, the confirm-visible mirror the exit guard clears. std::vector> batch; { std::lock_guard g(ref_queue_mutex); @@ -2916,6 +2990,9 @@ void CasRefLedger::flushRefBatch(const RootNamespace & ns, const std::shared_ptr if (carve_hook_for_test) carve_hook_for_test(CarvePhaseForTest::PlanReserveOwned); owned_items.reserve(owned_items.size() + selected); + /// Reserve the confirm-visible mirror too, for the same reason: the publish appends into it and + /// must not throw. + rt->carved.reserve(rt->carved.size() + selected); /// --- PUBLISH (no-throw) --- if (carve_hook_for_test) @@ -2924,6 +3001,7 @@ void CasRefLedger::flushRefBatch(const RootNamespace & ns, const std::shared_ptr { batch.push_back(rt->pending.front()); /// shared_ptr copy, capacity reserved owned_items.push_back(rt->pending.front()); /// same item into the responsibility set + rt->carved.push_back(rt->pending.front()); /// and into the confirm-visible mirror rt->pending.pop_front(); } @@ -3047,10 +3125,34 @@ void CasRefLedger::flushRefBatch(const RootNamespace & ns, const std::shared_ptr /// this item. `item_ops` was built in step 1 against the pre-boundary state; the carve /// deduplicates ref names within a batch, so the overflowing item operates on a ref distinct /// from the just-committed chunk's and re-validating it against the reseeded `working` is - /// consistent. + /// consistent. The scope is validated here as well, because the confirm relies on it (see + /// `confirmExactRef`, rule 3). RefTableState item_scratch = working; try { + /// Scope validation. `MutationScope` is what `confirmExactRef` reads to decide whether a + /// queued or in-flight mutation may change the ref it is asked about, so a `Ref{name}` item + /// whose ops mutate ANOTHER ref would let the confirm answer `Yes` off a row this very item is + /// about to change. Checked before anything durable and failing only this item: every + /// production caller names the exact ref its ops mutate, so a mismatch is a programming error. + if (it->scope.kind == MutationScope::Kind::Ref) + { + for (const RefOp & op : item_ops) + { + /// A namespace removal names no ref and moves every row, so it is outside every + /// `Ref` scope. The confirm's answer about every OTHER ref rests on this scope + /// check alone, so it is rejected here rather than left to any later one. + if (op.kind == RefOpKind::RemoveNamespace) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "ref mutation on namespace '{}' is scoped to ref '{}' but its {} op moves every ref", + ns.string(), it->scope.ref_name, refOpKindToWireWord(op.kind)); + if (const String * other = refNamedOutsideScope(op, it->scope.ref_name)) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "ref mutation on namespace '{}' is scoped to ref '{}' but its {} op names ref '{}'", + ns.string(), it->scope.ref_name, refOpKindToWireWord(op.kind), *other); + } + } + /// Whole-item shape validation (prerequisite to `dropNamespace`): the /// per-op loop below previews each op as its OWN single-op trial transaction, so a /// whole-transaction-shape rule like "remove_namespace must be the FINAL op" trivially @@ -3237,9 +3339,9 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt { /// Reconstructed locally so this arm has the SAME completion + fence semantics as when it lived /// inline in `flushRefBatch`: `complete_error` wakes a chunk's waiters under `ref_queue_mutex`, and - /// `fence_ok` folds `superseded_by_remount` into the append fence so a self-remount landing between a - /// leader's pre-allocate re-check and its `PUT` reports Unresolved rather than committing against a - /// stale cache. + /// `runtime_live` folds `superseded_by_remount` into the append operation's liveness so a + /// self-remount landing between a leader's pre-allocate re-check and its write ends the operation + /// rather than committing against a stale cache. auto complete_error = [&](const std::vector> & items, std::exception_ptr e) { std::lock_guard g(ref_queue_mutex); @@ -3250,10 +3352,11 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt } rt->cv.notify_all(); }; - const auto fence_ok = [this, &rt] + /// Everything the append fence cannot see. The generation term is the operation's own, so it is + /// deliberately absent here. + const auto runtime_live = [&rt] { - return fence_ok_fn() - && !rt->catalog_life_invalidated.load(std::memory_order_acquire) + return !rt->catalog_life_invalidated.load(std::memory_order_acquire) && !rt->superseded_by_remount.load(std::memory_order_acquire); }; @@ -3287,8 +3390,9 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt } const uint64_t admitted_generation = rt->admitted_fence_generation; - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + CasOperation catalog_op = mount_requests.resume(admitted_generation, runtime_live); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(catalog_op, layout); + refuseUnlessAdmitted(catalog_op, "terminal removal append"); catalog.life_index.throwIfAmbiguous("CAS terminal removal append"); const auto entry_it = std::find_if( catalog.catalog.entries.begin(), catalog.catalog.entries.end(), @@ -3317,8 +3421,9 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt try { const uint64_t admitted_generation = rt->admitted_fence_generation; - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + CasOperation catalog_op = mount_requests.resume(admitted_generation, runtime_live); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(catalog_op, layout); + refuseUnlessAdmitted(catalog_op, "removal-class append"); const NamespaceLifeId & life = rt->life; const auto entry_it = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), [&](const CatalogEntry & entry) @@ -3468,7 +3573,7 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt /// /// THE PLACEMENT IS THE CORRECTNESS ARGUMENT, and it has to hold for BOTH chunk shapes, because /// `commitRefChunk` has two different first durable effects. An ordinary chunk's is the ref-log - /// `putIfAbsentControlled` far below; a `NamespaceBirth` chunk's is the `_ckpt` publish, which is + /// create far below; a `NamespaceBirth` chunk's is the `_ckpt` publish, which is /// EARLIER. This call therefore sits above both, and every statement between here and the lock above /// is in-memory only. A "pure" preparation that published the `_ckpt` itself would be a lie, and /// moving that publish later would change fault semantics the directive says to preserve. @@ -3498,6 +3603,12 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt /// move is COW-pointer-only and happens here, while nothing is durable. std::optional candidate{std::move(prepared->candidate)}; + /// ONE operation for this chunk's whole durable phase -- the birth `_ckpt`, the log create and the + /// committed-frontier publish -- resumed under the generation the transaction was admitted at. The + /// engine refuses every one of them once that generation moves, so a chunk admitted by an + /// incarnation this mount no longer holds can make nothing durable. + CasOperation op = mount_requests.resume(admitted_fence_generation, runtime_live); + /// INV-4's FIRST `_ckpt` writer, and the ONLY writer anywhere that knows this namespace's /// `life_epoch`: it is the writer epoch of its `namespace_birth`, which is this transaction. No /// later writer can recover it (a table recovered from a snapshot never replays the birth), so if @@ -3537,8 +3648,7 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt { try { - if (publishCkptContribution(rt->life, *prepared->birth_contribution, - admitted_fence_generation, check_fence_or_throw) + if (publishCkptContribution(op, rt->life, *prepared->birth_contribution) == CkptPublishOutcome::FencedOut) { complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( @@ -3602,7 +3712,7 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt /// append for this table already advanced applied state. That other append can be the birth /// itself, whose `_ckpt` a cleanup call here would then delete out from under it -- the same harm /// class the ambiguous `Writing -> Wedged` branch is deliberately excluded to avoid, against an - /// object with no repair path (BACKLOG `{#ckpt-damage-no-repair-path}`). "`putIfAbsentControlled` + /// object that has no repair path. "the ref-log create /// was never reached" is true of THIS attempt; it says nothing about whether a DIFFERENT attempt /// for this same namespace already made the birth durable. Now that Critical B removed the other /// call sites too, `_ckpt` debris from a never-born namespace is reclaimed only by the future @@ -3615,35 +3725,31 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt } const RefAppendAttempt & active_attempt = *rt->append_attempt; - CasWriteOutcome outcome{}; - /// WHY an Unresolved came back. Two jobs (finding #37 defect 3): the wedge message stops claiming an - /// exhausted retry budget when in fact no request was ever sent, and -- see the `Unresolved` arm -- - /// the one reason that PROVES nothing was sent decides whether the lane wedges at all. - CasUnresolvedReason unresolved_reason = CasUnresolvedReason::NotUnresolved; + std::optional written; try { - outcome = ref_request_controller->putIfAbsentControlled( - active_attempt.key, active_attempt.bytes, fence_ok, /*out_token=*/nullptr, &unresolved_reason); + written = op.create(active_attempt.key, active_attempt.bytes, Retry::standard()); } catch (...) { + /// Every classified outcome comes back as a value, so an exception here is a deterministic + /// local failure raised at a point where an attempt may already have been sent. That is + /// ambiguous, and it therefore transfers ownership from `Writing` to `Wedged`; the exact + /// attempt remains installed. const std::exception_ptr write_error = std::current_exception(); - /// `putIfAbsentControlled` throws CORRUPTED_DATA when resolve-before-reissue observes a DIFFERENT - /// object already at this txn's key -- a proven different-object conflict, not an unresolved PUT. - /// Any other exception after the send boundary is ambiguous and therefore transfers ownership - /// from `Writing` to `Wedged`; the exact attempt remains installed. - if (getCurrentExceptionCode() != ErrorCodes::CORRUPTED_DATA) { - { - std::lock_guard lock(rt->state_mutex); - if (rt->lane_state == RefLaneState::Writing && rt->append_attempt - && rt->append_attempt->txn_id == id) - rt->lane_state = RefLaneState::Wedged; - } - complete_error(chunk_survivors, write_error); - return false; + std::lock_guard lock(rt->state_mutex); + if (rt->lane_state == RefLaneState::Writing && rt->append_attempt + && rt->append_attempt->txn_id == id) + rt->lane_state = RefLaneState::Wedged; } - /// THREE-WAY ADJUDICATION, the same one the wedge resolution owes [review HIGH-2]. "A different + complete_error(chunk_survivors, write_error); + return false; + } + + if (const auto * conflict = std::get_if(&*written)) + { + /// THREE-WAY ADJUDICATION, the same one the wedge resolution owes. "A different /// object at our derived key" is not one situation but two, and they call for opposite /// reactions. One of them is EXPECTED: a successor that sealed our epoch put its epoch-closing /// record at exactly the id we keep re-deriving, and INV-2 says we must keep re-deriving it @@ -3651,37 +3757,60 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt /// foreign interference would fence the mount and raise an anomaly alarm on the designed path. /// The other is a genuine breach of write-exclusivity and must be exactly as loud as before. /// - /// The occupant is read once, by exact key, here -- `putIfAbsentControlled` proved the mismatch - /// but does not hand back what it saw. One extra request on a path that is already exceptional - /// and already fatal to this attempt. + /// The occupant arrives WITH the conflict: the write's own settling read already observed the + /// key, so no second request is issued here. + const auto * occupant_object = std::get_if(&conflict->seen); + if (!occupant_object) + { + /// The settling read named NO occupant: it either failed, or it proved the key absent after + /// the store had already refused our create. Neither observation can be adjudicated, and + /// neither is terminal -- `resolveWedgeOnce` meets the identical observation and keeps the + /// lane WEDGED, and this site owes the same answer. The wedge is what makes the next flush + /// re-create at this exact key and adjudicate whatever it then finds; faulting instead would + /// spend the table's write availability until a remount on a read that may well succeed on + /// the next attempt. + /// + /// It must be COUNTED, because this arm is quiet by construction: the loud interference + /// report is only reached once the occupant can be NAMED, so a real breach whose occupant + /// keeps failing to be read would otherwise show up as nothing but a throttled log line + /// under load. Sustained growth on this counter is the signal that the loud path is starved. + ProfileEvents::increment(ProfileEvents::CASRefAppendOccupantUnreadable); + { + std::lock_guard lock(rt->state_mutex); + if (rt->lane_state == RefLaneState::Writing && rt->append_attempt + && rt->append_attempt->txn_id == id) + rt->lane_state = RefLaneState::Wedged; + } + ProfileEvents::increment(ProfileEvents::CASRefAppendWedged); + complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}': the store refused txn {}-{}'s create and the " + "settling read named no occupant (observed {}) — the append lane is wedged until the " + "SAME key resolves durable or a conclusive rejection is observed", + ns.string(), id.writer_epoch, id.ref_sequence, detail::renderObservation(conflict->seen)))); + return false; + } + Occupant occupant = Occupant::Foreign; bool classified = false; try { - if (const auto got = backend.get(active_attempt.key)) - { - occupant = classifyRefLogOccupant(ns, id, got->bytes, active_attempt.bytes); - classified = true; - } + occupant = classifyRefLogOccupant(ns, id, occupant_object->bytes, active_attempt.bytes); + classified = true; } catch (...) // NOLINT(bugprone-empty-catch) { - /// Left unclassified deliberately -- see below. The original conflict is what the survivors - /// are told about; this read's own failure is not their business. + /// Left unclassified deliberately -- see below. The conflict itself is what the survivors + /// are told about; the adjudication's own failure is not their business. } if (!classified) { - /// We could not learn WHICH of the two this is, so we decide NEITHER. Reporting foreign - /// interference would fence the mount on a guess, and reporting a conclusive rejection would - /// acknowledge a deposition we did not observe. The id is not consumed and nothing is - /// recorded, so the next append re-derives the same id, meets the same conflict, and - /// classifies again -- deferring costs one round trip and decides nothing wrongly. + /// An occupant WAS observed, and naming it raised something other than the malformed-object + /// codes `classifyRefLogOccupant` answers `Foreign` for. We could not learn WHICH of the two + /// situations this is, so we decide NEITHER: reporting foreign interference would fence the + /// mount on a guess, and reporting a conclusive rejection would acknowledge a deposition we + /// did not observe. /// - /// It must be COUNTED, because deferring is the one arm here that is quiet by construction: - /// the loud interference report is only reached once the occupant can be read, so a real - /// breach whose occupant keeps failing to read would otherwise show up as nothing but a - /// throttled log line under load. Sustained growth on this counter is the signal that the - /// loud path is being starved. + /// It must be COUNTED, for the same reason the arm above is. ProfileEvents::increment(ProfileEvents::CASRefAppendOccupantUnreadable); { std::lock_guard lock(rt->state_mutex); @@ -3691,9 +3820,9 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt complete_error(chunk_survivors, std::make_exception_ptr(Exception( ErrorCodes::CORRUPTED_DATA, "CAS ref-log append for namespace '{}': a DIFFERENT object occupies the id {}-{} this table " - "derived, and reading it to tell a successor's epoch seal from foreign interference did not " - "succeed — the lane is faulted until remount recovery adjudicates durable state", - ns.string(), id.writer_epoch, id.ref_sequence))); + "derived, and telling a successor's epoch seal from foreign interference did not succeed " + "(observed {}) — the lane is faulted until remount recovery adjudicates durable state", + ns.string(), id.writer_epoch, id.ref_sequence, detail::renderObservation(conflict->seen)))); return false; } if (occupant == Occupant::SuccessorSeal) @@ -3717,298 +3846,286 @@ bool CasRefLedger::commitRefChunk(const RootNamespace & ns, const std::shared_pt ns.string(), id.writer_epoch, id.writer_epoch, id.ref_sequence))); return false; } - /// A genuine breach. This table's appends are now BLOCKED, and that is the intended contract: - /// under mount-lease exclusivity this key is exclusively ours, so a foreign object at it is - /// corruption or a protocol breach, not a race. The id is not consumed, so the next attempt - /// derives the SAME id and hits the SAME conflict, loudly, until a remount-level recovery (a - /// fresh writer epoch is a fresh key namespace) clears it. Advancing past the occupant, which is - /// what the pool-wide allocator did, would have written this table's stream around a foreign - /// object and hidden the violation -- and produced the hole INV-1 exists to forbid. + if (occupant != Occupant::Ours) + { + /// A genuine breach. This table's appends are now BLOCKED, and that is the intended contract: + /// under mount-lease exclusivity this key is exclusively ours, so a foreign object at it is + /// corruption or a protocol breach, not a race. The id is not consumed, so the next attempt + /// derives the SAME id and hits the SAME conflict, loudly, until a remount-level recovery (a + /// fresh writer epoch is a fresh key namespace) clears it. Advancing past the occupant, which is + /// what the pool-wide allocator did, would have written this table's stream around a foreign + /// object and hidden the violation -- and produced a hole in a stream that must stay dense. + /// + /// Route it through the anomaly policy, exactly as the wedge-resolution site does for the + /// identical observation. Failing closed is right, but failing closed FOREVER is + /// not: without this the mount stays blocked on this table until somebody notices and remounts + /// by hand. One impossibility, one reaction. The report is deliberately BEFORE the survivors are + /// completed, so the fence is closed by the time any caller wakes and can retry. + const String attempt_key = active_attempt.key; + { + std::lock_guard lock(rt->state_mutex); + rt->append_attempt.reset(); + rt->lane_state = RefLaneState::Faulted; + } + const String interference_detail = fmt::format( + "ref-log append for namespace '{}' txn {}-{} observed a DIFFERENT object already at the id it " + "derived, and it is not an epoch seal of this namespace (observed {})", + ns.string(), id.writer_epoch, id.ref_sequence, detail::renderObservation(conflict->seen)); + on_impossible_interference(attempt_key, interference_detail, ns.string()); + complete_error(chunk_survivors, std::make_exception_ptr(Exception( + ErrorCodes::CORRUPTED_DATA, "CAS {}", interference_detail))); + return false; + } + /// `Occupant::Ours`: our OWN bytes at the id we derived, so an earlier attempt of this exact + /// transaction is already durable and byte equality is the proof (never a shape or generation + /// match). This lane's state machine does not produce it -- an attempt that may have landed + /// leaves the lane `Wedged`, and resolving that wedge advances the id -- but it is decided here + /// rather than left to the arm above, because reporting this table's own content as foreign + /// interference would fence the mount. The commit path below installs it, exactly as the wedge + /// site's identical observation does. + } + else if (const auto * refused = std::get_if(&*written)) + { + /// Proof that nothing became durable returns the exact attempt to `Ready`. + { + std::lock_guard lock(rt->state_mutex); + if (rt->lane_state == RefLaneState::Writing && rt->append_attempt + && rt->append_attempt->txn_id == id) + { + rt->append_attempt.reset(); + rt->lane_state = RefLaneState::Ready; + } + } + ProfileEvents::increment(ProfileEvents::CASRefAppendDefiniteFailure); + complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}' definitively failed ({}); " + "cached state is unchanged and txn id {}-{} was never used (a retry re-derives it)", + ns.string(), refused->message, id.writer_epoch, id.ref_sequence))); + return false; + } + else if (const auto * gave_up = std::get_if(&*written)) + { + /// The ONE give-up shape that must NOT wedge. The wedge exists because an unresolved write MAY + /// HAVE LANDED: the durable log may or may not contain this transaction, only a read of that + /// exact key can settle it, and until it does, minting a later id would build on a state that + /// may be missing a landed transaction. All of that presupposes an attempt was SENT. + /// + /// `sent_any` is the whole call's, not its last attempt's: it is false only when every gate + /// refused before the first request reached the network, so the key is provably unwritten, + /// there is nothing for a wedge to resolve, and wedging is pointless. /// - /// Route it through the anomaly policy, exactly as the wedge-resolution site does for the - /// identical observation [review I5]. Failing closed is right, but failing closed FOREVER is - /// not: without this the mount stays blocked on this table until somebody notices and remounts - /// by hand. One impossibility, one reaction. The report is deliberately BEFORE the survivors are - /// completed, so the fence is closed by the time any caller wakes and can retry. - const String attempt_key = active_attempt.key; + /// It is no longer HARMFUL, and the difference is worth stating because the old comment here + /// rested on it: a wedge over a never-written key used to be unclearable, because resolution + /// was a bare read and a read can only ever report absent. The every-attempt rule replaced + /// that with a conditional CREATE, so such a wedge now clears on the next caller's flush by + /// landing the transaction. What remains is that this lane would be blocked until then for no + /// reason at all -- a transient fence blip in the pre-attempt gate would cost the table its + /// write availability, and buy nothing, since there is provably nothing to resolve. + /// + /// The counterexample this argument deliberately excludes: a fence lost or a deadline reached + /// AFTER at least one attempt, and an attempt that COMMITTED but returned under a dropped + /// fence, both report `sent_any` true. Each may have left a durable object, so each keeps + /// wedging. + if (!gave_up->sent_any) + { + { + std::lock_guard lock(rt->state_mutex); + if (rt->lane_state == RefLaneState::Writing && rt->append_attempt + && rt->append_attempt->txn_id == id) + { + rt->append_attempt.reset(); + rt->lane_state = RefLaneState::Ready; + } + } + /// Count it. Before this arm existed these refusals bumped `CASRefAppendWedged`, so + /// removing the wedge also removed the only signal they were happening at all -- and a + /// soak oracle watching that counter fall could not tell "the fix works" from "nothing + /// happened". A separate event keeps both readings available: the wedge counter now means + /// only genuinely ambiguous appends, and this one means availability preserved. + ProfileEvents::increment(ProfileEvents::CASRefAppendPreAttemptRefused); + /// The id is not consumed: it was derived from `greatest_applied`, which this + /// refusal leaves exactly as it was, so the next caller on this table derives the SAME id + /// and the durable stream keeps no trace of the refusal. That is the free half of the + /// every-attempt rule -- an attempt that provably sent nothing owes nothing. + complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}' txn {}-{} was refused BEFORE any request was " + "sent — the append lane is NOT wedged (nothing can be durable, so there is " + "nothing to resolve) and the txn id is not consumed (a retry re-derives it)", + ns.string(), id.writer_epoch, id.ref_sequence))); + return false; + } { std::lock_guard lock(rt->state_mutex); - rt->append_attempt.reset(); - rt->lane_state = RefLaneState::Faulted; + if (rt->lane_state == RefLaneState::Writing && rt->append_attempt + && rt->append_attempt->txn_id == id) + rt->lane_state = RefLaneState::Wedged; } - on_impossible_interference(attempt_key, - fmt::format("ref-log append for namespace '{}' txn {}-{} observed a DIFFERENT object already at " - "the id it derived, and it is not an epoch seal of this namespace ({})", - ns.string(), id.writer_epoch, id.ref_sequence, - getCurrentExceptionMessage(/*with_stacktrace*/ false)), - ns.string()); - complete_error(chunk_survivors, write_error); + ProfileEvents::increment(ProfileEvents::CASRefAppendWedged); + complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}' txn {}-{} is UNCERTAIN (last observed {}) — " + "the append lane is wedged until the SAME key resolves durable or a conclusive rejection " + "is observed; this outcome is unproven, not failure", + ns.string(), id.writer_epoch, id.ref_sequence, detail::renderObservation(gave_up->last_seen)))); return false; } - switch (outcome) + { - case CasWriteOutcome::Committed: + /// The log object is durable -- this call's own attempt committed it, or the conflict above + /// proved an earlier attempt of this exact transaction already had. Either way it is not yet + /// admitted to logical history. Publish its exact frontier + /// under the SAME admission generation before any local consequence can make a later id + /// observable or wake a waiter. `Published` and `IdenticalSkip` both prove the contribution + /// durable. `FencedOut`, contention exhaustion, decode failure, or any other unresolved + /// publication leaves the log known durable but uninstalled, which is exactly + /// `NeedsRecovery`; recovery owns resolution of that window. + CkptPublishOutcome frontier_outcome = CkptPublishOutcome::FencedOut; + try + { + frontier_outcome = publishCkptContribution(op, rt->life, prepared->commit_contribution); + } + catch (...) { - /// A durable log object is not yet admitted to logical history. Publish its exact frontier - /// under the SAME admission generation before any local consequence can make a later id - /// observable or wake a waiter. `Published` and `IdenticalSkip` both prove the contribution - /// durable. `FencedOut`, contention exhaustion, decode failure, or any other unresolved - /// publication leaves the log known durable but uninstalled, which is exactly - /// `NeedsRecovery`; recovery owns resolution of that window. - const auto check_commit_admitted = [this, &rt](uint64_t expected_generation) + const std::exception_ptr frontier_error = std::current_exception(); { - check_fence_or_throw(expected_generation); - if (rt->catalog_life_invalidated.load(std::memory_order_acquire) - || rt->superseded_by_remount.load(std::memory_order_acquire)) - throwCasWriteRetryLater(fmt::format( - "CAS namespace '{}': its captured runtime was retired before committed-frontier publication", - rt->life.ns.string())); - }; - CkptPublishOutcome frontier_outcome = CkptPublishOutcome::FencedOut; + std::lock_guard lock(rt->state_mutex); + requireRecovery(*rt, ns, "committed-frontier publication"); + } + complete_error(chunk_survivors, frontier_error); + return false; + } + if (frontier_outcome == CkptPublishOutcome::FencedOut) + { + { + std::lock_guard lock(rt->state_mutex); + requireRecovery(*rt, ns, "committed-frontier publication fence"); + } + complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS ref-log append for namespace '{}': txn {}-{} is durable, but the mount fence " + "moved before its checkpoint frontier was published; the lane needs recovery", + ns.string(), id.writer_epoch, id.ref_sequence))); + return false; + } + + if (carve_hook_for_test) + carve_hook_for_test(CarvePhaseForTest::PostDurableInstall); + bool install_refused = false; + std::exception_ptr install_admission_error; + { + std::lock_guard lock(rt->state_mutex); + /// The checkpoint CAS can succeed and the fence can move before the state lock is + /// reached. Re-present the same admission INSIDE the install hold, immediately before + /// inspecting and swapping the candidate. A stale runtime may leave both log and + /// frontier durable, but it must neither install nor acknowledge them. try { - frontier_outcome = publishCkptContribution( - rt->life, prepared->commit_contribution, admitted_fence_generation, check_commit_admitted); + refuseUnlessAdmitted(op, "committed-frontier publication"); } catch (...) { - const std::exception_ptr frontier_error = std::current_exception(); - { - std::lock_guard lock(rt->state_mutex); - requireRecovery(*rt, ns, "committed-frontier publication"); - } - complete_error(chunk_survivors, frontier_error); - return false; + install_admission_error = std::current_exception(); + requireRecovery(*rt, ns, "post-frontier install admission"); } - if (frontier_outcome == CkptPublishOutcome::FencedOut) + /// Only this leader mutates `rt->state`, so the candidate's base snapshot is still the + /// current one: there is one append-lane leader per table at a time (the `leader_active` + /// baton), the wedge-resolution apply ran earlier in this same flush on this same thread, + /// recovery installs a state exactly once per runtime and has already completed for this + /// table, and every other consumer (readers, the snapshot publisher) only COPIES the + /// state under this mutex. Evaluated here, one statement before the install, and + /// asserted inside it: the comparison allocates nothing, and the identifier is short + /// enough that even the failure path's message is inline-buffered rather than heap + /// allocated, so no build can turn the assert itself into an allocation in the region. + const bool state_unchanged + = !install_admission_error + && rt->lane_state == RefLaneState::Writing + && rt->append_attempt + && rt->append_attempt->txn_id == id + && rt->append_attempt->bytes == active_attempt.bytes + && rt->state.getGreatestApplied() == candidate_base_id; + if (!install_admission_error && !state_unchanged) { - { - std::lock_guard lock(rt->state_mutex); - requireRecovery(*rt, ns, "committed-frontier publication fence"); - } - complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( - "CAS ref-log append for namespace '{}': txn {}-{} is durable, but the mount fence " - "moved before its checkpoint frontier was published; the lane needs recovery", - ns.string(), id.writer_epoch, id.ref_sequence))); - return false; + /// RELEASE-mode counterpart of the `chassert` inside the region below, which is a + /// no-op in a release build and therefore no guard at all for a window that spans a + /// full network round trip. Swapping the candidate in anyway would DISCARD whatever + /// advanced the table. The object is durable and this runtime cannot record it, + /// `LOGICAL_ERROR` here, where the wedge site's identical refusal reports the + /// retry-later class, and the asymmetry is deliberate: THIS one is reachable only by + /// a second writer inside one process -- a bug in this build, which a debug build + /// should abort on and shout about. The wedge site's is reachable by an ordinary + /// remount racing a slow resolution, which is a retryable fact about the world, not a + /// bug. Same refusal, different provenance, so different loudness. + requireRecovery(*rt, ns, "commitRefChunk install"); + install_refused = true; } - - if (carve_hook_for_test) - carve_hook_for_test(CarvePhaseForTest::PostDurableInstall); - bool install_refused = false; - std::exception_ptr install_admission_error; + else if (!install_admission_error) { - std::lock_guard lock(rt->state_mutex); - /// The checkpoint CAS can succeed and the fence can move before the state lock is - /// reached. Re-present the same admission INSIDE the install hold, immediately before - /// inspecting and swapping the candidate. A stale runtime may leave both log and - /// frontier durable, but it must neither install nor acknowledge them. + std::optional completed_attempt; try { - check_commit_admitted(admitted_fence_generation); + DENY_ALLOCATIONS_IN_SCOPE; + if (install_region_probe_for_test) + install_region_probe_for_test(); + chassert(state_unchanged); + rt->state.swap(*candidate); + rt->tail_count_since_snapshot.fetch_add(1, std::memory_order_relaxed); + rt->tail_bytes_since_snapshot.fetch_add(active_attempt.bytes.size(), std::memory_order_relaxed); + rt->append_attempt.swap(completed_attempt); + rt->lane_state = RefLaneState::Ready; } catch (...) { - install_admission_error = std::current_exception(); - requireRecovery(*rt, ns, "post-frontier install admission"); - } - /// Only this leader mutates `rt->state`, so the candidate's base snapshot is still the - /// current one: there is one append-lane leader per table at a time (the `leader_active` - /// baton), the wedge-resolution apply ran earlier in this same flush on this same thread, - /// recovery installs a state exactly once per runtime and has already completed for this - /// table, and every other consumer (readers, the snapshot publisher) only COPIES the - /// state under this mutex. Evaluated here, one statement before the install, and - /// asserted inside it: the comparison allocates nothing, and the identifier is short - /// enough that even the failure path's message is inline-buffered rather than heap - /// allocated, so no build can turn the assert itself into an allocation in the region. - const bool state_unchanged - = !install_admission_error - && rt->lane_state == RefLaneState::Writing - && rt->append_attempt - && rt->append_attempt->txn_id == id - && rt->append_attempt->bytes == active_attempt.bytes - && rt->state.getGreatestApplied() == candidate_base_id; - if (!install_admission_error && !state_unchanged) - { - /// RELEASE-mode counterpart of the `chassert` inside the region below, which is a - /// no-op in a release build and therefore no guard at all for a window that spans a - /// full network round trip. Swapping the candidate in anyway would DISCARD whatever - /// advanced the table. The object is durable and this runtime cannot record it, - /// `LOGICAL_ERROR` here, where the wedge site's identical refusal reports the - /// retry-later class, and the asymmetry is deliberate: THIS one is reachable only by - /// a second writer inside one process -- a bug in this build, which a debug build - /// should abort on and shout about. The wedge site's is reachable by an ordinary - /// remount racing a slow resolution, which is a retryable fact about the world, not a - /// bug. Same refusal, different provenance, so different loudness. requireRecovery(*rt, ns, "commitRefChunk install"); - install_refused = true; + throw; } - else if (!install_admission_error) + candidate.reset(); + completed_attempt.reset(); + try { - std::optional completed_attempt; - try - { - DENY_ALLOCATIONS_IN_SCOPE; - if (install_region_probe_for_test) - install_region_probe_for_test(); - chassert(state_unchanged); - rt->state.swap(*candidate); - rt->tail_count_since_snapshot.fetch_add(1, std::memory_order_relaxed); - rt->tail_bytes_since_snapshot.fetch_add(active_attempt.bytes.size(), std::memory_order_relaxed); - rt->append_attempt.swap(completed_attempt); - rt->lane_state = RefLaneState::Ready; - } - catch (...) - { - requireRecovery(*rt, ns, "commitRefChunk install"); - throw; - } - candidate.reset(); - completed_attempt.reset(); - try - { - rt->state.materializeCommitted(); - } - catch (...) - { - tryLogCurrentException(getLogger("CasPool"), fmt::format( - "CAS ref-log append for namespace '{}': committed txn {}-{} was applied durably, but " - "the post-commit overlay fold failed and was retained coherently for the next flush", - ns.string(), id.writer_epoch, id.ref_sequence)); - } + rt->state.materializeCommitted(); } - } - if (install_admission_error) - { - complete_error(chunk_survivors, install_admission_error); - return false; - } - if (install_refused) - { - complete_error(chunk_survivors, std::make_exception_ptr(Exception( - ErrorCodes::LOGICAL_ERROR, - "CAS ref-log append for namespace '{}': txn {}-{} is durable but this table changed " - "before installation; the lane needs recovery and refuses later writes until replay", - ns.string(), id.writer_epoch, id.ref_sequence))); - return false; - } - if (carve_hook_for_test) - carve_hook_for_test(CarvePhaseForTest::PostInstallPreAck); - ProfileEvents::increment(ProfileEvents::CASRefBatchFlushes); - ProfileEvents::increment(ProfileEvents::CASRefBatchedMutations, chunk_survivors.size()); - { - std::lock_guard g(ref_queue_mutex); - for (const auto & it : chunk_survivors) + catch (...) { - it->committed_id = id; - it->done = true; + tryLogCurrentException(getLogger("CasPool"), fmt::format( + "CAS ref-log append for namespace '{}': committed txn {}-{} was applied durably, but " + "the post-commit overlay fold failed and was retained coherently for the next flush", + ns.string(), id.writer_epoch, id.ref_sequence)); } - rt->cv.notify_all(); } - /// The threshold trigger -- off the lane, - /// dispatched AFTER waking every waiter above so this commit's own callers are never - /// delayed by it. Per chunk (spec §3): each committed chunk schedules its own publication, - /// and settlement coalesces the triggers so a mid-tenure publisher never suppresses a later - /// chunk (`settleSnapshotPublish`). - maybeScheduleSnapshotPublish(ns, rt); - return true; } - case CasWriteOutcome::DefiniteFailure: + if (install_admission_error) { - /// Proof that nothing became durable returns the exact attempt to `Ready`. - { - std::lock_guard lock(rt->state_mutex); - if (rt->lane_state == RefLaneState::Writing && rt->append_attempt - && rt->append_attempt->txn_id == id) - { - rt->append_attempt.reset(); - rt->lane_state = RefLaneState::Ready; - } - } - ProfileEvents::increment(ProfileEvents::CASRefAppendDefiniteFailure); - complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( - "CAS ref-log append for namespace '{}' definitively failed (non-retryable rejection); " - "cached state is unchanged and txn id {}-{} was never used (a retry re-derives it)", + complete_error(chunk_survivors, install_admission_error); + return false; + } + if (install_refused) + { + complete_error(chunk_survivors, std::make_exception_ptr(Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS ref-log append for namespace '{}': txn {}-{} is durable but this table changed " + "before installation; the lane needs recovery and refuses later writes until replay", ns.string(), id.writer_epoch, id.ref_sequence))); return false; } - case CasWriteOutcome::Unresolved: + if (carve_hook_for_test) + carve_hook_for_test(CarvePhaseForTest::PostInstallPreAck); + ProfileEvents::increment(ProfileEvents::CASRefBatchFlushes); + ProfileEvents::increment(ProfileEvents::CASRefBatchedMutations, chunk_survivors.size()); { - /// The ONE `Unresolved` shape that must NOT wedge (finding #37 defect 3). The wedge exists - /// because an `Unresolved` PUT MAY HAVE LANDED: the durable log may or may not contain this - /// transaction, only `resolveByExactGet` on that exact key can settle it, and until it does, - /// minting a later id would build on a state that may be missing a landed transaction. All of - /// that presupposes an attempt was SENT. - /// - /// `unresolvedProvesNothingWasSent` is true only for `NoAttemptSent`, which - /// `putIfAbsentControlled` reports only when a pre-attempt gate -- the mount fence or the - /// operation deadline -- rejected while `attempts_sent == 0`, i.e. strictly before the first - /// `backend->putIfAbsent`. Nothing reached the network, so the key is provably unwritten: - /// there is nothing for a wedge to resolve, and wedging is pointless. - /// - /// It is no longer HARMFUL, and the difference is worth stating because the old comment here - /// rested on it: a wedge over a never-written key used to be unclearable, because resolution - /// was a bare read and a read can only ever report absent. The every-attempt rule replaced - /// that with a conditional CREATE, so such a wedge now clears on the next caller's flush by - /// landing the transaction. What remains is that this lane would be blocked until then for no - /// reason at all -- a transient fence blip in the pre-attempt gate would cost the table its - /// write availability, and buy nothing, since there is provably nothing to resolve. - /// - /// The counterexample this argument deliberately excludes: a fence lost or a deadline reached - /// AFTER at least one attempt is `FenceLostMidWay`/`DeadlineMidWay`, and an attempt that - /// COMMITTED but returned under a dropped fence is `FenceLostPostWrite`. Each of those may - /// have left a durable object, so each keeps wedging -- as does anything a future contributor - /// adds to the enum without classifying it (see the predicate's allow-list construction). - if (unresolvedProvesNothingWasSent(unresolved_reason)) - { - { - std::lock_guard lock(rt->state_mutex); - if (rt->lane_state == RefLaneState::Writing && rt->append_attempt - && rt->append_attempt->txn_id == id) - { - rt->append_attempt.reset(); - rt->lane_state = RefLaneState::Ready; - } - } - /// Count it. Before this arm existed these refusals bumped `CASRefAppendWedged`, so - /// removing the wedge also removed the only signal they were happening at all -- and a - /// soak oracle watching that counter fall could not tell "the fix works" from "nothing - /// happened". A separate event keeps both readings available: the wedge counter now means - /// only genuinely ambiguous appends, and this one means availability preserved. - ProfileEvents::increment(ProfileEvents::CASRefAppendPreAttemptRefused); - /// The id is not consumed (INV-1): it was derived from `greatest_applied`, which this - /// refusal leaves exactly as it was, so the next caller on this table derives the SAME id - /// and the durable stream keeps no trace of the refusal. That is the free half of the - /// every-attempt rule -- an attempt that provably sent nothing owes nothing. - /// The installed attempt is retired below; no request was sent. - /// and is what makes the genuinely ambiguous path below allocation-free. - complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( - "CAS ref-log append for namespace '{}' txn {}-{} was refused BEFORE any request was " - "sent ({}) — the append lane is NOT wedged (nothing can be durable, so there is " - "nothing to resolve) and the txn id is not consumed (a retry re-derives it)", - ns.string(), id.writer_epoch, id.ref_sequence, - describeUnresolvedReason(unresolved_reason)))); - return false; - } + std::lock_guard g(ref_queue_mutex); + for (const auto & it : chunk_survivors) { - std::lock_guard lock(rt->state_mutex); - if (rt->lane_state == RefLaneState::Writing && rt->append_attempt - && rt->append_attempt->txn_id == id) - rt->lane_state = RefLaneState::Wedged; + it->committed_id = id; + it->done = true; } - ProfileEvents::increment(ProfileEvents::CASRefAppendWedged); - complete_error(chunk_survivors, makeCasWriteRetryLaterExceptionPtr(fmt::format( - "CAS ref-log append for namespace '{}' txn {}-{} is UNCERTAIN ({}) — " - "the append lane is wedged until the SAME key resolves durable or a conclusive rejection " - "is observed; this outcome is unproven, not failure", - ns.string(), id.writer_epoch, id.ref_sequence, - describeUnresolvedReason(unresolved_reason)))); - return false; + rt->cv.notify_all(); } + /// The threshold trigger -- off the lane, + /// dispatched AFTER waking every waiter above so this commit's own callers are never + /// delayed by it. Per chunk: each committed chunk schedules its own publication, + /// and settlement coalesces the triggers so a mid-tenure publisher never suppresses a later + /// chunk (`settleSnapshotPublish`). + maybeScheduleSnapshotPublish(ns, rt); + return true; } - /// Unreachable: the switch above covers every `CasWriteOutcome`. Kept explicit so the function has a - /// defined return on all control-flow paths. - return false; } bool CasRefLedger::hasStateBearingSnapshotCandidateUnderStateLock(const RefTableRuntime & rt) const @@ -4093,6 +4210,15 @@ void CasRefLedger::dispatchSnapshotPublisher(const RootNamespace & ns, const std } catch (...) { + { + /// Pace the exception exactly like an ordinary non-Committed publish. Every ordinary + /// failure arm inside the attempt arms this backoff before returning; an exception + /// thrown before any of them reaches here with the deadline unarmed, and settlement's + /// ONLY pacing gate is that deadline -- so without this the publisher redispatches at + /// full speed for as long as the fault persists. + std::lock_guard lock(rt->state_mutex); + advancePublishBackoff(*rt); + } if (publish_error_hook) publish_error_hook(); try @@ -4285,17 +4411,10 @@ void clampedCounterSub(std::atomic & counter, uint64_t amount) } -CkptPublishOutcome CasRefLedger::publishCkptContribution(const NamespaceLifeId & life, const RefCkpt & contribution, - uint64_t admitted_generation, - const std::function & check_admission, - const std::function & admit_request) +CkptPublishOutcome CasRefLedger::publishCkptContribution(CasOperation & op, const NamespaceLifeId & life, + const RefCkpt & contribution) { - /// The retry window is the SAME budget every other CAS operation of this ledger rides, measured on - /// the ledger's own injectable boot clock -- so a test drives the exhaustion arm without sleeping, - /// and a VM suspend cannot shorten it. - const CkptDeadline deadline{boot_ms_now_fn, boot_ms_now_fn() + cas_request_budget.operation_deadline_ms}; - const CkptPublishOutcome outcome = publishCkpt( - backend, layout, life, contribution, admitted_generation, check_admission, deadline, admit_request); + const CkptPublishOutcome outcome = publishCkpt(op, layout, life, contribution); if (outcome == CkptPublishOutcome::Published) ProfileEvents::increment(ProfileEvents::CASRefCheckpointPublished); else if (outcome == CkptPublishOutcome::IdenticalSkip) @@ -4330,13 +4449,12 @@ bool CasRefLedger::tryPublishSnapshotAndAdvanceCheckpointOnceOnRuntimeImpl( /// recheck discipline -- the same value at every site), so a publish admitted under an incarnation /// that has since been replaced can advance nothing. const uint64_t admitted_generation = rt->admitted_fence_generation; - const auto runtime_still_admitted = [this, &rt, admitted_generation] + CasOperation op = mount_requests.resume(admitted_generation, [&rt] { return !rt->catalog_life_invalidated.load(std::memory_order_acquire) - && !rt->superseded_by_remount.load(std::memory_order_acquire) - && fence_ok_fn() - && fence_generation_fn() == admitted_generation; - }; + && !rt->superseded_by_remount.load(std::memory_order_acquire); + }); + const auto runtime_still_admitted = [&op] { return op.admitted(); }; /// ONE copy of the live state, at a transaction boundary -- no /// replay, no per-entry retention. The tail counters are captured in the SAME critical section so @@ -4423,11 +4541,22 @@ bool CasRefLedger::tryPublishSnapshotAndAdvanceCheckpointOnceOnRuntimeImpl( return false; } const String key = layout.refSnapshotKey(rt->life, candidate_x); - const CasWriteOutcome outcome - = ref_request_controller->putIfAbsentControlled(key, bytes, runtime_still_admitted); - if (outcome != CasWriteOutcome::Committed) + const WriteResult put = op.create(key, bytes, Retry::standard()); + if (const auto * conflict = std::get_if(&put)) + { + /// A snapshot key names the exact state it encodes, so a re-run of this publish -- and only a + /// re-run -- can find its own identical body already there. DIFFERENT bytes under this mount's + /// exclusive lease are corruption, and the publisher must not paper over them with a backoff. + const auto * occupant = std::get_if(&conflict->seen); + if (!occupant || occupant->bytes != bytes) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS ref table '{}': a DIFFERENT object occupies snapshot {}-{} (observed {})", + ns.string(), candidate_x.writer_epoch, candidate_x.ref_sequence, + detail::renderObservation(conflict->seen)); + } + else if (!std::holds_alternative(put)) { - /// DefiniteFailure/Unresolved: DO NOT prune (no durable covering snapshot -- pruning the tail + /// Refused or unresolved: DO NOT prune (no durable covering snapshot -- pruning the tail /// without one is data loss). Arm the bounded per-table backoff so the read path does not /// re-dispatch this full-snapshot encode+PUT until it elapses -- the read-triggered PUT-storm /// latch breaker. A later trigger past the deadline retries. @@ -4449,33 +4578,24 @@ bool CasRefLedger::tryPublishSnapshotAndAdvanceCheckpointOnceOnRuntimeImpl( /// know has already recorded -- in either order. /// /// A checkpoint that does NOT advance leaves the attempt unadopted: the backoff is armed and this - /// returns false, so a later trigger re-runs the whole publish. The re-run's body PUT resolves to - /// `Committed` against its own identical bytes, so retrying costs one conditional PUT and not a - /// second snapshot. Adopting instead would mark this snapshot as the newest -- suppressing every - /// later publish for it -- while the checkpoint still pointed below it, leaving recovery replaying - /// from an older base with nothing scheduled to fix it. + /// returns false, so a later trigger re-runs the whole publish. The re-run's body create meets its + /// own identical bytes as a conflict, which the occupant compare above accepts, so retrying costs + /// one conditional PUT and not a second snapshot. Adopting instead would mark this snapshot as the + /// newest -- suppressing every later publish for it -- while the checkpoint still pointed below it, + /// leaving recovery replaying from an older base with nothing scheduled to fix it. bool ckpt_advanced = false; if (!runtime_still_admitted()) return false; - const auto check_runtime_admission = [this, &rt](uint64_t generation) - { - if (snapshot_before_ckpt_cas_hook_for_test) - snapshot_before_ckpt_cas_hook_for_test(); - check_fence_or_throw(generation); - if (rt->catalog_life_invalidated.load(std::memory_order_acquire) - || rt->superseded_by_remount.load(std::memory_order_acquire)) - throwCasWriteRetryLater(fmt::format( - "CAS namespace '{}': its captured runtime was retired before checkpoint publication", - rt->life.ns.string())); - }; + if (snapshot_before_ckpt_cas_hook_for_test) + snapshot_before_ckpt_cas_hook_for_test(); try { - ckpt_advanced = publishCkptContribution(rt->life, RefCkpt{.life_epoch = std::nullopt, - .committed_through = candidate_x, - .checkpoint_snapshot_id = candidate_x, - .last_epoch_seal = std::nullopt}, - admitted_generation, - check_runtime_admission) != CkptPublishOutcome::FencedOut; + ckpt_advanced = publishCkptContribution(op, rt->life, + RefCkpt{.life_epoch = std::nullopt, + .committed_through = candidate_x, + .checkpoint_snapshot_id = candidate_x, + .last_epoch_seal = std::nullopt}) + != CkptPublishOutcome::FencedOut; } catch (...) { @@ -4801,7 +4921,7 @@ NamespaceLifeId CasRefLedger::namespaceLife(const RootNamespace & ns) auto rt = lookupRefTableRuntime(ns); if (rt) { - check_fence_or_throw(rt->admitted_fence_generation); + refuseUnlessAdmitted(mount_requests.resume(rt->admitted_fence_generation), "resident namespace life"); bool removal_closed = false; { std::lock_guard queue_lock(ref_queue_mutex); @@ -4811,7 +4931,8 @@ NamespaceLifeId CasRefLedger::namespaceLife(const RootNamespace & ns) { /// A lost erase response can leave only the detached predecessor's close bit. Reconcile /// before refusing so an absent/replaced row frees the logical name without rebinding it. - reconcileCatalogCut(CasRefCatalog::read(backend, layout)); + CasOperation reconcile_op = mount_requests.resume(rt->admitted_fence_generation); + reconcileCatalogCut(CasRefCatalog::read(reconcile_op, layout)); const auto refreshed = lookupRefTableRuntime(ns); if (refreshed == rt) throwCasWriteRetryLater(fmt::format( @@ -4828,9 +4949,9 @@ NamespaceLifeId CasRefLedger::namespaceLife(const RootNamespace & ns) /// A cold mutation observes or births the durable identity before allocating any local state. const uint64_t admitted_generation = fence_generation_fn(); - check_fence_or_throw(admitted_generation); - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + CasOperation op = mount_requests.resume(admitted_generation); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "cold mutable namespace admission"); const auto entry_it = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); if (entry_it != catalog.catalog.entries.end() && entry_it->state == NsState::Removing) @@ -4842,7 +4963,7 @@ NamespaceLifeId CasRefLedger::namespaceLife(const RootNamespace & ns) = entry_it != catalog.catalog.entries.end() && entry_it->state == NsState::Live ? NamespaceLifeId::fromCatalogEntry(entry_it->ns, entry_it->incarnation) : resolveNamespaceLife(ns, admitted_generation, live_epoch_fn()); - check_fence_or_throw(admitted_generation); + refuseUnlessAdmitted(op, "cold mutable namespace admission, after life resolution"); rt = acquireRefTableRuntime(life, admitted_generation); ensureRefTableRecovered(ns, *rt); return rt->life; @@ -4874,7 +4995,7 @@ bool CasRefLedger::namespaceStillLogicallyPresent(const RootNamespace & ns) /// short of that falls through to the exact cold-path observation below. if (const auto current = lookupRefTableRuntime(ns)) { - check_fence_or_throw(current->admitted_fence_generation); + refuseUnlessAdmitted(mount_requests.resume(current->admitted_fence_generation), "resident namespace presence"); bool closed = false; { std::lock_guard queue_lock(ref_queue_mutex); @@ -4895,9 +5016,9 @@ bool CasRefLedger::namespaceStillLogicallyPresent(const RootNamespace & ns) /// the one answer that must never be manufactured by a race, so it alone is re-confirmed by a /// second read of this namespace's row before being trusted. const uint64_t admitted_generation = fence_generation_fn(); - check_fence_or_throw(admitted_generation); - const CasRefCatalog::Snapshot first_catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + CasOperation op = mount_requests.resume(admitted_generation); + const CasRefCatalog::Snapshot first_catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "namespace presence probe"); if (namespace_presence_probe_after_first_read_hook_for_test) namespace_presence_probe_after_first_read_hook_for_test(); const auto find_entry = [&ns](const CasRefCatalog::Snapshot & snap) -> const CatalogEntry * @@ -4917,8 +5038,8 @@ bool CasRefLedger::namespaceStillLogicallyPresent(const RootNamespace & ns) /// continuously) while adding nothing to this row's proof. A row that appears in between /// answers present: `true` is always the safe direction, and the caller's next poll runs the /// full state dispatch against a fresh observation. - const CasRefCatalog::Snapshot second_catalog = CasRefCatalog::read(backend, layout); - check_fence_or_throw(admitted_generation); + const CasRefCatalog::Snapshot second_catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "namespace presence probe"); if (find_entry(second_catalog)) return true; return false; /// no catalog row in two atomic observations: proven absent @@ -4952,8 +5073,8 @@ bool CasRefLedger::namespaceStillLogicallyPresent(const RootNamespace & ns) if (namespace_presence_probe_after_terminal_proven_hook_for_test) namespace_presence_probe_after_terminal_proven_hook_for_test(); - check_fence_or_throw(admitted_generation); - const CasRefCatalog::Snapshot post_terminal_catalog = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot post_terminal_catalog = CasRefCatalog::read(op, layout); + refuseUnlessAdmitted(op, "namespace presence probe"); const CatalogEntry * post_terminal_entry = find_entry(post_terminal_catalog); if (!post_terminal_entry || post_terminal_entry->incarnation == entry->incarnation) return false; /// terminal durably proven and nothing has since occupied `ns` under a new life @@ -4978,7 +5099,8 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( /// class shares the bigger complete-table byte budget (encodeRefLogTxn's own `checkBudget`, keyed /// off the presence of a `RemoveNamespace` op) and is exempt from the ordinary per-op admission /// check (it only ever shrinks state; see `flushRefBatch`'s `state_growing` filter). - const CasRefCatalog::Snapshot initial_catalog = CasRefCatalog::read(backend, layout); + CasOperation initial_op = mount_requests.admit(); + const CasRefCatalog::Snapshot initial_catalog = CasRefCatalog::read(initial_op, layout); const auto initial_it = std::find_if(initial_catalog.catalog.entries.begin(), initial_catalog.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); if (initial_it == initial_catalog.catalog.entries.end()) @@ -4991,15 +5113,14 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( if (initial_it->state == NsState::Creating) { const CatalogEntry & observed = *initial_it; - const uint64_t admitted_generation = fence_generation_fn(); + CasOperation cancel_op = mount_requests.resume(fence_generation_fn()); switch (CasRefCatalog::cancelStalledCreating( - backend, layout, observed, - [this](const CreatorFence & creator) + cancel_op, layout, observed, + [&](const CreatorFence & creator) { return isCreatorFenceTerminal( - backend, layout, creator.server_root_id, creator.writer_epoch); - }, - admitted_generation, check_fence_or_throw)) + cancel_op, layout, creator.server_root_id, creator.writer_epoch); + })) { case CasRefCatalog::StalledCreatingCancelOutcome::Cancelled: invalidateRemovedCatalogLife(NamespaceLifeId::fromCatalogEntry(observed.ns, observed.incarnation)); @@ -5048,13 +5169,14 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( } const uint64_t admitted_generation = fence_generation_fn(); + CasOperation removal_op = mount_requests.resume(admitted_generation); std::optional observed_live; if (initial_it->state == NsState::Live) observed_live = *initial_it; bool removing_durable = false; try { - const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(removal_op, layout); const auto entry_it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); if (entry_it == snapshot.catalog.entries.end()) @@ -5076,12 +5198,11 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( { observed_live = *entry_it; uint64_t removal_started_round = 0; - if (const auto got = backend.get(layout.gcStateKey())) + if (const auto got = removal_op.read(layout.gcStateKey(), Retry::standard())) removal_started_round = decodeGcState(got->bytes).round; switch (CasRefCatalog::beginRemoving( - backend, layout, *observed_live, removal_started_round, - admitted_generation, check_fence_or_throw)) + removal_op, layout, *observed_live, removal_started_round)) { case CasRefCatalog::BeginRemovingOutcome::Transitioned: case CasRefCatalog::BeginRemovingOutcome::AlreadyRemoving: @@ -5112,7 +5233,7 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( /// changed or fenced case remains closed (fail-close) and propagates the original error. try { - const CasRefCatalog::Snapshot fresh = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot fresh = CasRefCatalog::read(removal_op, layout); const auto fresh_it = std::find_if(fresh.catalog.entries.begin(), fresh.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); if (observed_live @@ -5124,7 +5245,7 @@ DropNamespaceStats CasRefLedger::dropNamespaceImpl( } else if (observed_live && fresh_it != fresh.catalog.entries.end() && *fresh_it == *observed_live) { - check_fence_or_throw(admitted_generation); + refuseUnlessAdmitted(removal_op, "reopening the removal lane"); std::lock_guard queue_lock(ref_queue_mutex); rt->removal_admission_closed = false; rt->cv.notify_all(); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h index 163956607278..73ce041c73da 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h @@ -1,6 +1,8 @@ #pragma once #include -#include +#include +#include +#include #include #include #include @@ -37,11 +39,14 @@ enum class ResolveAudit : uint8_t { Emit, Deferred }; /// there is no independent apply marker or durable-id floor whose combinations form a second, /// implicit state machine. /// -/// `Ready` is the only state that admits a new append or certifies a cached row. `Writing` owns the +/// `Ready` is the state that admits a new append. A cached row is certified (`confirmExactRef`) in +/// `Ready` and in `Writing` alike, and in both only while no queued or carved mutation names that row's +/// ref -- a `Ready` lane with such a mutation queued refuses too. `Writing` owns the /// exact attempt before its first possible send. `Wedged` owns that same attempt after an ambiguous -/// result. `NeedsRecovery` means a transaction is known durable but cannot be installed in this cache; -/// it is a hard write and certification fence until replay completes. `Closed` records a successor's -/// epoch seal, and `Faulted` records foreign or internally inconsistent durable state. +/// result and certifies nothing. `NeedsRecovery` means a transaction is known durable but cannot be +/// installed in this cache; it is a hard write and certification fence until replay completes. `Closed` +/// records a successor's epoch seal, and `Faulted` records foreign or internally inconsistent durable +/// state. enum class RefLaneState : uint8_t { Ready, @@ -57,7 +62,9 @@ enum class RefLaneState : uint8_t /// `Yes` is the only answer that AUTHORIZES anything, so it is the only one that must be earned: it is /// returned exclusively when every rule of the lane snapshot holds. `Unknown` is the catch-all for /// every ambiguity, and it is the answer this primitive is biased towards: a cold, evicted, recovering, -/// busy or non-`Ready` table answers `Unknown` rather than doing any work to find out. +/// busy, fenced-out, wedged or otherwise broken table answers `Unknown` rather than doing any work to +/// find out, and so does a table with a queued or in-flight mutation of the asked-about ref (or of the +/// whole namespace); a mutation of another ref does not refuse. /// /// `No` means "this runtime's committed row for that ref is not the manifest you asked about" -- and /// nothing more. It is NOT a proof of the negative about the durable table, because the mount fence is @@ -84,7 +91,9 @@ class CasRefLedger { public: CasRefLedger( - BackendPtr backend_ptr, + /// The mount plane. Every request this ledger makes is admitted on it, so a ref-lane write and + /// a mount-lease renewal are measured against the same fence and the same clock. + CasRequests & mount_requests_, const Layout & layout_, RefLedgerConfig config_, const CasEventSink & event_sink_, @@ -95,23 +104,19 @@ class CasRefLedger /// unlike the mount-state functions below, because it is a fixed identity for this ledger's /// whole lifetime (mirrors `CasMountRuntime`'s own by-value `server_root_id`). String server_root_id_, - /// Monotonic mount clock used by the retry controller; it may be empty when the controller's - /// default clock is appropriate. - std::function controller_boot_ms_fn, /// Callbacks into mount and watermark state owned by `Pool`, bound for this ledger's lifetime: std::function live_epoch_fn_, std::function fence_ok_fn_, - /// The two fence-GENERATION primitives (`CasMountRuntime::fenceGeneration`/`checkFenceOrThrow`), - /// injected exactly as `CasPlainObjects` takes them. `fence_ok_fn` above answers "may this mount - /// write AT ALL, right now"; these two answer the different question an append lane must ask - /// across an I/O window: "is this still the SAME mount incarnation that admitted the transaction - /// I am about to act on?" A wedge captures the generation at admission and presents it back on - /// every later retry and before every install, so a result that returns after a fence loss or a - /// re-arm is inert for the superseded runtime instead of installing a stale view (spec §3, - /// "the mount-fence generation is captured at admission and required on every slot-occupy and - /// install"). + /// The fence-GENERATION primitive (`CasMountRuntime::fenceGeneration`), injected exactly as + /// `CasPlainObjects` takes it. `fence_ok_fn` above answers "may this mount write AT ALL, right + /// now"; this answers the different question an append lane must ask across an I/O window: "is + /// this still the SAME mount incarnation that admitted the transaction I am about to act on?" A + /// wedge captures the generation at admission and presents it back -- through `mount_requests`, + /// by resuming an operation under it -- on every later retry and before every install, so a + /// result that returns after a fence loss or a re-arm is inert for the superseded runtime instead + /// of installing a stale view: the generation is captured at admission and required on every + /// slot-occupy and install. std::function fence_generation_fn_, - std::function check_fence_or_throw_, std::function boot_ms_now_fn_, std::function may_mutate_, std::function &)> on_impossible_interference_, @@ -154,7 +159,8 @@ class CasRefLedger /// receiver drives, so it must never be able to make this writer do work. /// /// The rules are evaluated as one snapshot spanning both lane mutexes, in this order: table warm - /// and resident; lane state `Ready`; exact committed-row equality; mount fence live last. Every + /// and resident; lane state `Ready` or `Writing`, with no queued or carved mutation whose + /// `MutationScope` covers the ref; exact committed-row equality; mount fence live last. Every /// ambiguity answers `Unknown` -- see `ConfirmAnswer`, and the .cpp for why the order and the /// two-mutex hold are what make a `Yes` a linearization point rather than a guess. ConfirmAnswer confirmExactRef(const RootNamespace & ns, const String & ref_name, @@ -282,26 +288,18 @@ class CasRefLedger /// mutation can appear after shutdown has taken its snapshot. bool drainRefLanesForShutdown(uint64_t wait_budget_ms); - /// Performs a staged conditional create through the ledger's retry controller and append-fence - /// predicate. Callers do not access either dependency directly, so every attempt observes the same - /// mount admission rule. - CasWriteOutcome stagingPutIfAbsent(std::string_view key, std::string_view bytes, Token * out_token); - - /// Same retry/fence policy as `stagingPutIfAbsent`, for a MUTABLE If-Match overwrite whose bytes - /// are deterministic (safe for GET-based resolution). - CasOverwriteResult stagingConditionalOverwrite(std::string_view key, std::string_view bytes, const Token & expected); - - /// Same retry/fence policy as `stagingPutIfAbsent`, for a MUTABLE marker where an existing - /// DIFFERENT value at the key is a normal Conflict outcome, not corruption (see - /// `CasRequestController::putIfAbsentControlledMutable`). - CasOverwriteResult stagingPutIfAbsentMutable(std::string_view key, std::string_view bytes); + /// A staged conditional create on the mount plane. Callers reach the store only through these, so + /// every attempt observes the same mount admission rule. `Conflict` names what occupies the key; a + /// caller whose key is content-addressed decides for itself whether a different occupant is + /// corruption. + WriteResult stagingPutIfAbsent(const String & key, const String & bytes); /// Hooks required by `EventEmitter`: events are delivered to the injected sink when one is present. bool hasEventSink() const noexcept { return static_cast(event_sink); } void emitEvent(CasEvent && e) const { if (event_sink) event_sink(std::move(e)); } - /// Replaces the retry controller's delay seam for deterministic tests; production callers leave it - /// untouched. + /// Replaces the mount plane's inter-attempt delay seam for deterministic tests; production callers + /// leave it untouched. void setCasRetrySleepForTest(std::function sleep_fn); /// Replaces only ref-table recovery's token-aware retry delay seam for deterministic tests. @@ -401,6 +399,12 @@ class CasRefLedger /// `install_region_probe_for_test`). void setInstallRegionProbeForTest(std::function probe) { install_region_probe_for_test = std::move(probe); } + /// Installs a probe at `ensureRefTableRecovered`'s step 8 -- the LAST admission recheck before a + /// materialized recovery result installs, after the final authority read and O(N) materialization + /// have already run. A test pausing here and then latching a stop token observes whether that stop + /// is honored before install (see `recovery_install_probe_for_test`). + void setRecoveryInstallProbeForTest(std::function probe) { recovery_install_probe_for_test = std::move(probe); } + /// Installs the pre-tenure fault seam (see `ref_pre_tenure_hook_for_test`). void setRefPreTenureHookForTest(std::function hook) { ref_pre_tenure_hook_for_test = std::move(hook); } @@ -481,6 +485,30 @@ class CasRefLedger return it != ref_name_slots.end() && it->second.current->leader_active; } + /// Returns the number of items carved by the current tenure and not yet released by its exit guard. + /// Under the queue mutex, like `refQueuePendingForTest`. + size_t refCarvedForTest(const RootNamespace & ns) + { + std::lock_guard g(ref_queue_mutex); + const auto it = ref_name_slots.find(ns.string()); + return it == ref_name_slots.end() ? 0 : it->second.current->carved.size(); + } + + /// Returns whether the `carved` entry for `ref_name` (if any) is already completed. Lets a test + /// PROVE an earlier chunk's item is done rather than infer it from carve-hook ordering. Under the + /// queue mutex, like `refCarvedForTest`; `done` itself is guarded by the same mutex. + bool refCarvedItemDoneForTest(const RootNamespace & ns, const String & ref_name) + { + std::lock_guard g(ref_queue_mutex); + const auto it = ref_name_slots.find(ns.string()); + if (it == ref_name_slots.end()) + return false; + for (const auto & item : it->second.current->carved) + if (item->scope.kind == MutationScope::Kind::Ref && item->scope.ref_name == ref_name) + return item->done; + return false; + } + /// Returns the number of callers currently waiting for `ns` recovery under its state mutex. uint64_t refRecoveryWaitersForTest(const RootNamespace & ns) { @@ -600,10 +628,10 @@ class CasRefLedger String bytes; /// `CasMountRuntime::fenceGeneration()` as read at this transaction's ADMISSION -- the same /// critical section that snapshotted the state and derived the id, i.e. one atomic reading of - /// "what this attempt was allowed to do". Every later `slotOccupy` retry is gated on THIS value - /// (never the current one), and every install is preceded by presenting it back through - /// `checkFenceOrThrow`: a retry admitted under a dead incarnation must send nothing, and a - /// result that returns after a fence bump/re-arm must install nothing. + /// "what this attempt was allowed to do". Every later retry of this attempt is admitted by + /// resuming an operation under THIS value (never the current one), and every install is + /// preceded by presenting it back: a retry admitted under a dead incarnation must send + /// nothing, and a result that returns after a fence bump/re-arm must install nothing. uint64_t admitted_fence_generation = 0; }; @@ -667,7 +695,7 @@ class CasRefLedger private: /// Injected storage and mount environment. The member order is part of construction/destruction /// behavior because the callbacks and references are used by the runtime owned below. - Backend & backend; + CasRequests & mount_requests; const Layout & layout; RefLedgerConfig config; const CasEventSink & event_sink; @@ -678,7 +706,6 @@ class CasRefLedger std::function live_epoch_fn; std::function fence_ok_fn; std::function fence_generation_fn; - std::function check_fence_or_throw; std::function boot_ms_now_fn; std::function may_mutate; std::function &)> on_impossible_interference; @@ -712,10 +739,10 @@ class CasRefLedger /// One coherent decoded `RefTableState` and append runtime for a namespace. It is recovered lazily /// and evicted only as a whole. `state_mutex` is separate from - /// `ref_queue_mutex` (which only ever guards `pending`/`leader_active`) so a reader (resolveRef/ + /// `ref_queue_mutex` (which only ever guards `pending`/`carved`/`leader_active`) so a reader (resolveRef/ /// listRefs) can observe `state` without contending with the flush leader's network round trip -- /// the leader only holds `state_mutex` for the brief copy-out-before-validate and the - /// apply-after-commit steps, never for the `putIfAbsentControlled` call itself. + /// apply-after-commit steps, never for the durable write itself. struct RefTableRuntime { /// An allocator can reuse an evicted predecessor's address for its successor. This monotone id @@ -844,6 +871,18 @@ class CasRefLedger uint64_t publish_backoff_ms = 0; std::deque> pending; /// guarded by ref_queue_mutex + /// The current tenure's carved items, from the carve (`flushRefBatch`'s PUBLISH phase) to the + /// tenure's exit guard (`completeOwnedItemsAndReleaseLeadership`), both under `ref_queue_mutex`. + /// The carve pops an item out of `pending`, so `pending` alone cannot show a mutation between + /// carve and install -- the window in which its transaction may be durable while the committed + /// row still lags it. The mirror makes that item visible, under the same mutex, to a reader + /// (such as `confirmExactRef`) that holds only the runtime. An item is completed by its chunk's + /// install or earlier by an error, often long before the exit guard; the mirror keeps it + /// regardless. + /// Over-inclusive on purpose: an installed item and an item that failed validation before any + /// send both stay here until the exit guard; that is one tenure of over-refusal for their refs, + /// never an under-refusal. + std::vector> carved; /// guarded by ref_queue_mutex bool leader_active = false; /// guarded by ref_queue_mutex /// Set before the exact `Live -> Removing` catalog CAS and retained until that life is deleted. /// New positive mutations check it in the same queue critical section as admission; the one @@ -958,13 +997,6 @@ class CasRefLedger return rt.state.nextTxnId(live_epoch_fn()); } - /// The CAS-owned retry controller this Pool's ref-log writer path uses for every conditional - /// log/snapshot `PUT` and uncertain-result resolution. It is also shared by the part-manifest - /// write and mutable freshness-meta writes. The controller is stateless per call (immutable - /// budget/clock/sleep — the sleep fn mutates only through the test-only seam, before traffic), so - /// concurrent lanes and builds use the one instance safely. - std::unique_ptr ref_request_controller; - /// Test-only hook called before a compatible append batch is carved; null in production. std::function ref_pre_carve_hook_for_test; @@ -988,6 +1020,8 @@ class CasRefLedger /// otherwise non-throwing regions. A test that installs a throwing probe must therefore disarm it /// after the region it targets, or every later install throws too. Null in production. std::function install_region_probe_for_test; + /// See `setRecoveryInstallProbeForTest`. Null in production. + std::function recovery_install_probe_for_test; std::function append_after_runtime_capture_hook_for_test; std::function read_before_state_lock_hook_for_test; std::function readable_catalog_after_observation_hook_for_test; @@ -1047,9 +1081,10 @@ class CasRefLedger /// Every `createNamespace`/`completeCreation`/`reconcileStaleCreator` outcome that writes nothing /// (`FencedOut`, `Superseded`, a reconciled entry, `EntryChanged`) re-reads the catalog and loops; /// `CreatorFenceStillLive` throws the retry-later class, which this function's caller (the transient - /// retry loop) or a higher one re-drives. Bounded against a pathological duel between two openers; - /// each primitive this loop calls has its OWN bounded retry against the catalog's single object, so - /// this bound is only against THIS loop's re-read cycle. + /// retry loop) or a higher one re-drives. One `Retry` is frozen before the loop and shared by every + /// read and every protocol call it makes, so the whole resolution ends within one standard window; + /// a re-read forced by a competing actor is paced by a jittered sleep, and an iteration cap is the + /// secondary bound. NamespaceLifeId resolveNamespaceLife( const RootNamespace & ns, uint64_t admitted_generation, uint64_t live_epoch, bool * lifecycle_refusal = nullptr); @@ -1060,8 +1095,8 @@ class CasRefLedger /// cleanup could account for. Everything terminal throws. /// /// `admitted_generation` is the ONE fence generation this whole recovery was admitted under: the - /// walk presents it to every `slotOccupy` and to the `_ckpt` CAS, and the caller presents the same - /// value once more immediately before installing. + /// walk resumes its operation under it, so every request the walk makes is measured against it, and + /// the caller presents the same value once more immediately before installing. /// `retained_attempt` is copied under `state_mutex` before the unlocked walk. It is evidence from /// this runtime's admitted writer, not a second recovery authority: only the exact slot it names /// is compared byte-for-byte, and a successor seal remains the existing conclusive-loss case. @@ -1079,8 +1114,7 @@ class CasRefLedger /// independently disqualify it. Throws; remount cancellation raises the retry-later class and /// LATCHES through `cancelled` so the caller's transient loop does not re-drive it. /// - /// The FENCE is deliberately absent: it gates the three sites that spend it (every `slotOccupy`, the - /// `_ckpt` CAS, the install), not every read. See the definition for why. + /// The FENCE is deliberately absent: the walk's own operation carries it. See the definition. void checkRecoveryStillAdmitted( const RootNamespace & ns, RefTableRuntime & rt, bool & cancelled, const std::optional & token = std::nullopt) const; @@ -1108,9 +1142,10 @@ class CasRefLedger /// and apply-after-commit ordering -- the LIVE state is still only ever advanced once the object is /// durable; `commitRefChunk`'s pre-`PUT` apply targets a private candidate that nothing else can /// observe. Every item it carves out of `pending` is appended to - /// `owned_items` (the leader's responsibility set) at the moment it is carved. When a batch's total - /// op count exceeds `ref_txn_max_ops`, the validation loop emits SEVERAL ref-log transactions in one - /// tenure via `commitRefChunk` -- each a complete commit boundary. + /// `owned_items` (the leader's responsibility set) and to `rt->carved` (the confirm-visible mirror, + /// see `RefTableRuntime::carved`) at the moment it is carved. When a batch's total op count exceeds + /// `ref_txn_max_ops`, the validation loop emits SEVERAL ref-log transactions in one tenure via + /// `commitRefChunk` -- each a complete commit boundary. void flushRefBatch(const RootNamespace & ns, const std::shared_ptr & rt, std::vector> & owned_items); @@ -1130,8 +1165,8 @@ class CasRefLedger }; /// ONE bounded resolution attempt for `rt`'s outstanding wedge (spec INV-1's every-attempt rule): - /// at most one `slotOccupy(wedge.key, wedge.bytes, ...)` per calling flush, gated on the wedge's - /// ORIGINAL `admitted_fence_generation` rather than the current one. There is deliberately NO + /// at most one conditional create of the wedge's exact key and bytes per calling flush, admitted by + /// resuming under the wedge's ORIGINAL `admitted_fence_generation` rather than the current one. There is deliberately NO /// background retry thread and no deadline-resetting loop: a permanently quiet wedged namespace /// waits for its next caller or for a remount, which is acceptable precisely because the wedged /// operation was never acknowledged. @@ -1143,8 +1178,8 @@ class CasRefLedger /// "absent", which is not a rejection: the earlier ambiguous attempt could still land afterwards. /// /// Post-I/O recheck: the outcome is adjudicated on an I/O result, so before ANY action follows from - /// it (adopt, acknowledge, unwedge, fail the survivors) this re-acquires `state_mutex`, presents - /// `admitted_fence_generation` back through `checkFenceOrThrow`, and compares the full wedge + /// it (adopt, acknowledge, unwedge, fail the survivors) this re-acquires `state_mutex`, compares + /// `admitted_fence_generation` against the fence's CURRENT generation, and compares the full wedge /// identity against what is still installed. A result that returns after a fence bump/re-arm, or /// after the wedge it belonged to was replaced, is INERT for this runtime. WedgeResolutionResult resolveWedgeOnce( @@ -1179,10 +1214,12 @@ class CasRefLedger /// Leadership-exit guard for `appendRefOps`: under `ref_queue_mutex`, completes every still-unfinished /// item this leader owned (with `flush_exception` when unwinding, or a fail-closed `LOGICAL_ERROR` - /// otherwise), removes each owned item from `pending` so no future leader can carve it, and releases - /// leadership (`leader_active = false` + `cv.notify_all`). On the normal path every owned item is - /// already `done`, so only the leadership release has effect. This is the single authority that - /// resets `leader_active` on any exit from the leader loop. + /// otherwise), removes each owned item from `pending` so no future leader can carve it, clears + /// `rt->carved` (the confirm-visible mirror, now that every carved item's fate -- installed or a + /// recorded lane failure -- no longer needs to be seen), and releases leadership + /// (`leader_active = false` + `cv.notify_all`). On the normal path every owned item is already + /// `done`, so only the leadership release and the mirror clear have effect. This is the single + /// authority that resets `leader_active` on any exit from the leader loop. void completeOwnedItemsAndReleaseLeadership( const RootNamespace & ns, const std::shared_ptr & rt, const std::vector> & owned_items, @@ -1212,10 +1249,15 @@ class CasRefLedger /// - `runRecoveryWalkOnce` contributes `last_epoch_seal` once its own CAS-walk minted or adopted /// one -- it is the only writer that mints seals, so it is the only writer that can record /// where the chain now ends. - CkptPublishOutcome publishCkptContribution(const NamespaceLifeId & life, const RefCkpt & contribution, - uint64_t admitted_generation, - const std::function & check_admission, - const std::function & admit_request = {}); + CkptPublishOutcome publishCkptContribution(CasOperation & op, const NamespaceLifeId & life, + const RefCkpt & contribution); + + /// The verdict points that guard a DECISION rather than a request: the engine refuses a request on + /// its own, but a result already in hand must not be acted on once `op` has stopped being admitted. + /// `op.admitted()` folds every reason together -- a moved mount incarnation, a lease with too little + /// time left, or a caller-supplied liveness term that has since gone false -- because none of them + /// leaves the caller anything more specific to act on than "this admission no longer holds". + void refuseUnlessAdmitted(const CasOperation & op, std::string_view what) const; /// Common candidate predicate for scheduler admission and execution after capture. Caller holds /// `rt.state_mutex`; an epoch seal is not state-bearing and cannot be snapshotted. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.cpp index 5f99e880c968..15ffd10e9666 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.cpp @@ -854,7 +854,7 @@ RefCleanupPlan planRefCleanup(const RefTableListing & listing, const RefTxnId & return plan; } -EpochCrossResult crossEpochFromSeal(Backend & backend, const Layout & layout, const RootNamespace & ns, +EpochCrossResult crossEpochFromSeal(CasOperation & op, const Layout & layout, const RootNamespace & ns, const RefTxnId & from_seal, std::optional seal_proven, const RefTxnId & witness, const NamespaceLifeId & life) { @@ -870,19 +870,19 @@ EpochCrossResult crossEpochFromSeal(Backend & backend, const Layout & layout, co return result; } - /// `life`: REQUIRED, not resolved here (review NEW-3) -- an internal fallback resolve was tried - /// once already (review C3, `Gc::fold`) and once more here (fsck's own independent walk defaulted - /// to `nullopt` and re-resolved), and both times a caller that had already committed to one `life` - /// for the rest of its walk could silently diverge from this function's OWN resolution if the - /// namespace is dropped and recreated between the two reads. `CasFsck.cpp`'s stream walk resolves - /// `life` once, at the top of its own function, and must pass that SAME value here rather than let - /// this function re-derive it a second time. + /// `life`: REQUIRED, not resolved here -- an internal fallback resolve was tried once already + /// (in `Gc::fold`) and once more here (fsck's own independent walk defaulted to `nullopt` and + /// re-resolved), and both times a caller that had already committed to one `life` for the rest of + /// its walk could silently diverge from this function's OWN resolution if the namespace is dropped + /// and recreated between the two reads. `CasFsck.cpp`'s stream walk resolves `life` once, at the + /// top of its own function, and must pass that SAME value here rather than let this function + /// re-derive it a second time. uint64_t target_epoch = witness.writer_epoch; while (target_epoch > from_seal.writer_epoch) { const RefTxnId start{target_epoch, 1}; result.probed = start; - const auto body = backend.get(layout.refLogKey(life, start)); + const auto body = op.read(layout.refLogKey(life, start), Retry::standard()); if (!body) { ++result.absent_probes; @@ -952,8 +952,7 @@ std::optional nextRefLogIdWithinCommittedFrontier( } CheckpointSnapshotBase readCheckpointSnapshotBase( - Backend & backend, const Layout & layout, const NamespaceLifeId & life, const RefCkpt & checkpoint, - const std::function & admit_request) + CasOperation & op, const Layout & layout, const NamespaceLifeId & life, const RefCkpt & checkpoint) { const RootNamespace & ns = life.ns; if (!checkpoint.checkpoint_snapshot_id) @@ -969,9 +968,7 @@ CheckpointSnapshotBase readCheckpointSnapshotBase( ns.string()); } const RefTxnId snapshot_id = *checkpoint.checkpoint_snapshot_id; - if (admit_request) - admit_request(); - const auto log = backend.get(layout.refLogKey(life, snapshot_id)); + const auto log = op.read(layout.refLogKey(life, snapshot_id), Retry::standard()); if (!log) { throw Exception(ErrorCodes::CORRUPTED_DATA, @@ -1007,9 +1004,7 @@ CheckpointSnapshotBase readCheckpointSnapshotBase( if (base_txn.prev_epoch_seal) { predecessor_seal_id = *base_txn.prev_epoch_seal; - if (admit_request) - admit_request(); - const auto predecessor = backend.get(layout.refLogKey(life, *predecessor_seal_id)); + const auto predecessor = op.read(layout.refLogKey(life, *predecessor_seal_id), Retry::standard()); if (!predecessor) { throw Exception(ErrorCodes::CORRUPTED_DATA, @@ -1030,9 +1025,7 @@ CheckpointSnapshotBase readCheckpointSnapshotBase( } } - if (admit_request) - admit_request(); - const auto snapshot = backend.get(layout.refSnapshotKey(life, snapshot_id)); + const auto snapshot = op.read(layout.refSnapshotKey(life, snapshot_id), Retry::standard()); if (!snapshot) { throw Exception(ErrorCodes::CORRUPTED_DATA, @@ -1047,8 +1040,8 @@ CheckpointSnapshotBase readCheckpointSnapshotBase( } RecoveredRefTable recoverRefTableDetailedFromAuthority( - Backend & backend, const Layout & layout, const std::optional & catalog_entry, - const std::optional & ckpt) + CasOperation & op, const Layout & layout, const std::optional & catalog_entry, + const std::optional & ckpt, KeyReader * reader) { /// The frozen catalog row and `_ckpt` supplied by the caller determine every recovery boundary; /// this function must not re-read either mutable object, or enumerate the stream, because that @@ -1066,7 +1059,7 @@ RecoveredRefTable recoverRefTableDetailedFromAuthority( uint64_t base_snapshot_bytes = 0; if (base_id) { - CheckpointSnapshotBase base = readCheckpointSnapshotBase(backend, layout, life, *ckpt); + CheckpointSnapshotBase base = readCheckpointSnapshotBase(op, layout, life, *ckpt); base_snapshot = std::move(base.snapshot); base_snapshot_bytes = base.bytes; } @@ -1077,7 +1070,11 @@ RecoveredRefTable recoverRefTableDetailedFromAuthority( RefTxnId id = *grounding.walk_from; while (id <= *grounding.committed_through) { - const auto got = backend.get(layout.refLogKey(life, id)); + const String key = layout.refLogKey(life, id); + if (reader && id.ref_sequence < std::numeric_limits::max()) + hintRefLogsWithinEpoch(*reader, layout, life, RefTxnId{id.writer_epoch, id.ref_sequence + 1}, + *grounding.committed_through); + const auto got = reader ? reader->take(key) : op.read(key, Retry::standard()); if (!got) { /// `NamespaceLifeId` is opaque and unique to one logical life. A later birth has a @@ -1099,7 +1096,12 @@ RecoveredRefTable recoverRefTableDetailedFromAuthority( if (const std::optional next = nextRefLogIdWithinCommittedFrontier( id, is_seal, *grounding.committed_through)) + { + if (is_seal && reader && id.ref_sequence < std::numeric_limits::max()) + discardRefLogHintsOfEpoch(*reader, layout, life, RefTxnId{id.writer_epoch, id.ref_sequence + 1}, + *grounding.committed_through); id = *next; + } else break; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.h index db061bb7ebcb..2a819e27eff8 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefProtocol.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -60,6 +61,10 @@ enum class RootMutationOrigin : uint8_t /// call touches. The flat-combining batch builder admits at most ONE mutation per ref name into a /// single flush (per-ref durable histories stay bit-identical to the unbatched protocol) and flushes /// `WholeShard` calls SOLO (dropNamespace and anything touching multiple refs wholesale). +/// +/// It is also safety-bearing: `CasRefLedger::confirmExactRef` refuses to certify a ref while a queued +/// or carved item's scope covers it, and `flushRefBatch` fails an item whose ops name a ref outside its +/// declared scope, so a caller must name exactly the ref its ops mutate. struct MutationScope { enum class Kind : uint8_t { Ref, WholeShard }; @@ -432,7 +437,7 @@ RefTableState replay(const std::optional & snapshot, std::span /// Everything a successful recovery of one ref table seeds. Produced by streaming replay /// (`RefReplayBuilder::finish`) rather than assigned field-by-field into the runtime, so the whole /// publication is one value installed atomically -- a prose field list would drift, but a struct that -/// the install copies wholesale cannot silently lose a field (Codex review round 4, spec §5). +/// the install copies wholesale cannot silently lose a field. /// /// `finish` populates the fields that are a pure function of `(base snapshot, replayed tail)`: `state`, /// `newest_snapshot_id`, `tail_count`, `tail_bytes`, and `base_snapshot_bytes`. The @@ -466,7 +471,7 @@ struct RecoveryResult std::optional last_epoch_seal; }; -/// The streaming generalisation of `replay` (spec §5): owns a PRIVATE candidate `RefTableState` and +/// The streaming generalisation of `replay`: owns a PRIVATE candidate `RefTableState` and /// applies decoded transactions into it ONE AT A TIME, in place, discarding the candidate on any throw. /// It is the memory fix for a long post-snapshot tail: `replay` takes the whole `tail` materialised in a /// vector (every decoded transaction resident at once, each up to the 20 MiB normal-class cap), whereas @@ -524,7 +529,7 @@ uint64_t decodedRefLogTxnFootprint(const RefLogTxn & txn); /// identical seam. void reportReplayMemoryDelta(int64_t delta_footprint_bytes); -/// Test-only observability for the streaming-recovery memory invariant (spec §5): while a probe is +/// Test-only observability for the streaming-recovery memory invariant: while a probe is /// installed, each recovery loop reports the resident footprint of every decoded transaction it holds, /// for exactly the span it holds it (`reportReplayMemoryDelta` + `decodedRefLogTxnFootprint`). A /// memory-bound test tracks the peak of the summed reported footprint and asserts it stays within a @@ -726,13 +731,13 @@ struct EpochCrossResult /// disagree about when an epoch boundary has been proved -- a rule that says which records a cut /// contains cannot have two implementations. /// -/// `life`: the namespace's life, REQUIRED (review NEW-3 -- a `nullopt`-resolves-internally default was -/// tried once and reintroduced the exact divergence review C3 removed from `Gc::fold`, just relocated -/// into `CasFsck.cpp`'s independent walk, which had its OWN already-resolved `life` in scope one call -/// site above and simply did not pass it). Every caller must resolve `life` itself, ONCE, and pass the -/// SAME value here that it uses for every other read in its own walk -- this function no longer -/// resolves anything on its own, so there is no second resolution left to disagree with the first. -EpochCrossResult crossEpochFromSeal(Backend & backend, const Layout & layout, const RootNamespace & ns, +/// `life`: the namespace's life, REQUIRED -- a `nullopt`-resolves-internally default was tried once +/// and let two callers resolve `life` independently and disagree, even though one of them +/// (`CasFsck.cpp`'s walk) already had its OWN resolved `life` in scope one call site above and simply +/// did not pass it in. Every caller must resolve `life` itself, ONCE, and pass the SAME value here +/// that it uses for every other read in its own walk -- this function no longer resolves anything on +/// its own, so there is no second resolution left to disagree with the first. +EpochCrossResult crossEpochFromSeal(CasOperation & op, const Layout & layout, const RootNamespace & ns, const RefTxnId & from_seal, std::optional seal_proven, const RefTxnId & witness, const NamespaceLifeId & life); @@ -763,8 +768,10 @@ struct RecoveredRefTable /// the named predecessor to be an `EpochSeal`, then read the snapshot. This order prevents a forged /// snapshot at any historical seal or contextually invalid epoch start from becoming state. Cleanup /// retains both the matching log and returned predecessor proof while the checkpoint names this base. -/// When supplied, `admit_request` runs immediately before each raw backend request so a caller may -/// refuse later requests without changing their durable order. Its default preserves read-only callers. +/// Every read below is one of `op`'s own requests, so a caller that needs to refuse a later request +/// mid-walk (a recovery attempt superseded while it runs) folds that fact into `op`'s `Liveness` +/// predicate at admission rather than passing a callback here -- the operation already re-checks it +/// before each request. struct CheckpointSnapshotBase { RefTableSnapshot snapshot; @@ -775,8 +782,7 @@ struct CheckpointSnapshotBase }; CheckpointSnapshotBase readCheckpointSnapshotBase( - Backend & backend, const Layout & layout, const NamespaceLifeId & life, const RefCkpt & checkpoint, - const std::function & admit_request = {}); + CasOperation & op, const Layout & layout, const NamespaceLifeId & life, const RefCkpt & checkpoint); /// Recover a ref table from ONE immutable lifecycle authority cut supplied by the caller. `catalog_entry` /// is either the exact row from that caller's frozen catalog cut or absence from that same cut; `ckpt` is @@ -790,8 +796,11 @@ CheckpointSnapshotBase readCheckpointSnapshotBase( /// under this immutable authority. In particular, this read-only API never probes or adopts `F+1`. There /// is deliberately no self-resolving compatibility overload: every consumer must pass the row from its /// frozen catalog cut explicitly. +/// +/// `reader`, when set, fetches the logs; it may prefetch ids of the current epoch and drops them at a +/// seal. Every decision and every error path is the same with and without it. RecoveredRefTable recoverRefTableDetailedFromAuthority( - Backend & backend, const Layout & layout, const std::optional & catalog_entry, - const std::optional & ckpt); + CasOperation & op, const Layout & layout, const std::optional & catalog_entry, + const std::optional & ckpt, KeyReader * reader = nullptr); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 93b962c6980b..10de8e758b3c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1,6 +1,6 @@ #include #include -#include +#include #include #include #include @@ -22,6 +22,7 @@ #include #include #include +#include #include namespace ProfileEvents @@ -38,14 +39,12 @@ namespace ErrorCodes extern const int CORRUPTED_DATA; extern const int FILE_DOESNT_EXIST; extern const int LOGICAL_ERROR; - extern const int NETWORK_ERROR; } } namespace DB::Cas { -void reportMountRenewProgress(const CasOverwriteProgress & progress) noexcept; void reportMountRenewCompletion(const MountRenewResult & result) noexcept; void configureMountRenewObservability( const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; @@ -57,10 +56,37 @@ void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcep namespace { -/// TRUE iff a `list(prefix, "", 1)` over `prefix` returns at least one key. -bool prefixHasAnyKey(Backend & b, const String & prefix) +/// TRUE iff a one-key listing of `prefix` returns anything. +bool prefixHasAnyKey(CasOperation & op, const String & prefix) { - return !b.list(prefix, /*cursor*/ "", /*limit*/ 1).keys.empty(); + return !op.list(prefix, /*cursor*/ "", /*limit*/ 1, Retry::standard()).keys.empty(); +} + +/// The write's own verdict on whether somebody else holds the key: for a refused precondition, what +/// the write's resolve read saw there; nothing when this write landed. Every other ending failed to +/// reach the store and must surface as itself -- reading it as a rival writer is how a transport +/// outage becomes a "double start" report. +std::optional conflictOrThrow(WriteResult && result, const String & what) +{ + if (Conflict * conflict = std::get_if(&result)) + return std::move(conflict->seen); + orThrow(std::move(result), what); + return std::nullopt; +} + +/// A raced claim, reported as the OCCUPANT the write's own resolve read observed rather than as the +/// lease this server proposed and failed to install -- that body is what the caller renders into the +/// fail-closed operator message, and naming ourselves there points an operator at the wrong process. +/// The observed incarnation names the body returned beside it, so the caller's observation loop +/// compares like with like. A conflict the resolve read could not settle to a body saw NOBODY, and +/// reports nobody: the caller's own re-read is what identifies the holder there. +MountClaimResult racedDoubleStart(const Observation & seen) +{ + if (const Object * occupant = std::get_if(&seen)) + return {.kind = MountClaimResult::LiveDoubleStart, + .body = decodeMountLease(occupant->bytes), + .etag = occupant->etag}; + return {.kind = MountClaimResult::LiveDoubleStart, .body = std::nullopt, .etag = std::nullopt}; } uint64_t defaultBootMs() @@ -70,17 +96,24 @@ uint64_t defaultBootMs() return static_cast(ts.tv_sec) * 1000 + static_cast(ts.tv_nsec) / 1000000; } +/// Why a renewal ended without a retained lease, in the vocabulary the audit event reports. Each +/// value is assigned from exactly one arm of the write's verdict, so the event never re-derives a +/// reason from state the request engine does not carry. enum class MountRenewTerminalClassification : uint8_t { - FromDiagnostics, + Unclassified, DeterministicFailure, Conflict, Vanished, + Cancelled, + FenceOrLifecycleLost, + ExternalLeaseDeadline, + RequestDeadline, + Unresolved, }; -/// Retry admission (`RetryStarted`/`PutStarted`) touches only this fixed-size state. First ambiguity -/// may deliver its one bounded warning/event before the controller's following pre-resolve gate; it -/// never runs after a pre-request gate. +/// One logical renewal's audit snapshot. Fixed-size and trivially copyable so a reentrant event sink +/// gets a distinct stack slot instead of aliasing the call that is still running. struct MountRenewObservabilityContext { bool active = false; @@ -94,15 +127,10 @@ struct MountRenewObservabilityContext uint64_t observability_start_boot_ms = 0; uint64_t confirmed_deadline_boot_ms = 0; uint64_t initial_confirmed_budget_ms = 0; - CasOverwriteDeadlineSource deadline_source = CasOverwriteDeadlineSource::RequestBudget; - CasOverwriteStopCause stop_cause = CasOverwriteStopCause::Continue; - CasUnresolvedReason unresolved_reason = CasUnresolvedReason::NotUnresolved; MountRenewOutcome outcome = MountRenewOutcome::NotAttempted; - MountRenewTerminalClassification terminal_classification = MountRenewTerminalClassification::FromDiagnostics; + MountRenewTerminalClassification terminal_classification = MountRenewTerminalClassification::Unclassified; uint32_t attempts_sent = 0; - uint32_t ambiguity_attempt_no = 0; - bool resolved_by_get = false; - bool retrying_delivered = false; + bool resolved_by_read = false; }; static_assert(std::is_trivially_copyable_v); @@ -119,7 +147,7 @@ struct MountRenewObservabilityConfiguration /// registered outer per-call snapshot stable without allocation, including while a parked redo holds /// `remount_mutex`. Overflow suppresses rich event/log delivery for the nested call rather than /// aliasing an outer call or changing protocol behavior; physical attempt truth is independently -/// retained by the stack-local observer in `MountLeaseKeeper::renew`. +/// retained by the stack-local observer in `MountLeaseRenewer::renew`. struct MountRenewObservabilityStack { static constexpr size_t capacity = 8; @@ -138,6 +166,12 @@ MountRenewObservabilityContext * currentMountRenewObservability() noexcept return &mount_renew_observability.contexts[mount_renew_observability.depth - 1]; } +void markMountRenewTermination(MountRenewTerminalClassification classification) noexcept +{ + if (MountRenewObservabilityContext * context = currentMountRenewObservability()) + context->terminal_classification = classification; +} + enum class MountRenewObservabilityRegistration : uint8_t { Stack, @@ -205,7 +239,6 @@ void initializeMountRenewObservability( UInt128 write_attempt_id, uint64_t attempt_start_boot_ms, uint64_t confirmed_deadline_boot_ms, - CasOverwriteDeadlineSource deadline_source, const CasEventSink & event_sink) noexcept { MountRenewObservabilityContext * context = currentMountRenewObservability(); @@ -228,46 +261,9 @@ void initializeMountRenewObservability( .initial_confirmed_budget_ms = confirmed_deadline_boot_ms > attempt_start_boot_ms ? confirmed_deadline_boot_ms - attempt_start_boot_ms : 0, - .deadline_source = deadline_source, }; } -constexpr std::string_view unresolvedReasonName(CasUnresolvedReason reason) -{ - switch (reason) - { - case CasUnresolvedReason::NotUnresolved: return "not_unresolved"; - case CasUnresolvedReason::NoAttemptSent: return "no_attempt_sent"; - case CasUnresolvedReason::FenceLostMidWay: return "fence_lost_mid_way"; - case CasUnresolvedReason::DeadlineMidWay: return "deadline_mid_way"; - case CasUnresolvedReason::FenceLostPostWrite: return "fence_lost_post_write"; - case CasUnresolvedReason::AttemptsExhausted: return "attempts_exhausted"; - case CasUnresolvedReason::DefiniteFailureAfterAmbiguity: return "definite_failure_after_ambiguity"; - } - return "unknown"; -} - -constexpr std::string_view deadlineSourceName(CasOverwriteDeadlineSource source) -{ - switch (source) - { - case CasOverwriteDeadlineSource::RequestBudget: return "request_budget"; - case CasOverwriteDeadlineSource::ExternalLeaseSafety: return "external_lease_safety"; - } - return "unknown"; -} - -constexpr std::string_view stopCauseName(CasOverwriteStopCause cause) -{ - switch (cause) - { - case CasOverwriteStopCause::Continue: return "continue"; - case CasOverwriteStopCause::Cancelled: return "cancelled"; - case CasOverwriteStopCause::FenceOrLifecycleLost: return "fence_or_lifecycle_lost"; - } - return "unknown"; -} - uint64_t elapsedSince(uint64_t start_boot_ms, uint64_t now_boot_ms) { return now_boot_ms >= start_boot_ms ? now_boot_ms - start_boot_ms : 0; @@ -287,9 +283,6 @@ void emitMountRenewEvent( std::string_view outcome, uint32_t attempts_sent, uint64_t now_boot_ms, - CasUnresolvedReason unresolved_reason, - CasOverwriteDeadlineSource deadline_source, - CasOverwriteStopCause stop_cause, std::string_view classification, uint64_t remount_attempt_no) noexcept { @@ -300,11 +293,9 @@ void emitMountRenewEvent( CasEvent event; event.type = CasEventType::WatermarkRenew; event.outcome = String{outcome}; - event.reason = outcome == "retrying" - ? "CAS mount renewal entered bounded retry after an ambiguous physical attempt" - : (outcome == "recovered" - ? "CAS mount renewal recovered before its confirmed lease-safety deadline" - : "CAS mount renewal ended without retained authority and fenced the mount"); + event.reason = outcome == "recovered" + ? "CAS mount renewal recovered before its confirmed lease-safety deadline" + : "CAS mount renewal ended without retained authority and fenced the mount"; event.detail = { {"server_root_id", *context.server_root_id}, {"writer_epoch", std::to_string(context.writer_epoch)}, @@ -313,9 +304,6 @@ void emitMountRenewEvent( {"attempts_sent", std::to_string(attempts_sent)}, {"elapsed_ms", std::to_string(elapsedSince(context.observability_start_boot_ms, now_boot_ms))}, {"remaining_confirmed_budget_ms", std::to_string(remainingConfirmedBudget(context, now_boot_ms))}, - {"unresolved_reason", String{unresolvedReasonName(unresolved_reason)}}, - {"deadline_source", String{deadlineSourceName(deadline_source)}}, - {"stop_cause", String{stopCauseName(stop_cause)}}, {"classification", String{classification}}, }; if (remount_attempt_no != 0) @@ -328,72 +316,19 @@ void emitMountRenewEvent( } } -void deliverMountRenewRetrying( - const MountRenewObservabilityContext & context, - const String & write_attempt_id, - uint64_t now_boot_ms, - uint64_t remount_attempt_no) noexcept -{ - /// Publish the structured event before the text logger. Either callback may consume recovery - /// budget, but this transition is followed by the controller's pre-resolve gate, so it cannot - /// start backend I/O after that budget has expired. - emitMountRenewEvent( - context, - write_attempt_id, - "retrying", - context.ambiguity_attempt_no, - now_boot_ms, - CasUnresolvedReason::NotUnresolved, - context.deadline_source, - CasOverwriteStopCause::Continue, - "ambiguous", - remount_attempt_no); - try - { - LOG_WARNING( - getLogger("CasMountLeaseKeeper"), - "CAS mount renewal '{}' entered retry after physical attempt {} (writer_epoch={}, seq={}, " - "remaining_confirmed_budget_ms={})", - *context.server_root_id, - context.ambiguity_attempt_no, - context.writer_epoch, - context.seq, - remainingConfirmedBudget(context, now_boot_ms)); - } - catch (...) - { - } -} - -constexpr std::string_view terminalClassificationName(const MountRenewObservabilityContext & context) +constexpr std::string_view terminalClassificationName(MountRenewTerminalClassification classification) { - switch (context.terminal_classification) + switch (classification) { case MountRenewTerminalClassification::DeterministicFailure: return "deterministic_failure"; case MountRenewTerminalClassification::Conflict: return "conflict"; case MountRenewTerminalClassification::Vanished: return "vanished"; - case MountRenewTerminalClassification::FromDiagnostics: break; - } - - switch (context.unresolved_reason) - { - case CasUnresolvedReason::AttemptsExhausted: return "attempts_exhausted"; - case CasUnresolvedReason::DefiniteFailureAfterAmbiguity: return "definite_failure_after_ambiguity"; - case CasUnresolvedReason::FenceLostMidWay: - case CasUnresolvedReason::FenceLostPostWrite: - return context.stop_cause == CasOverwriteStopCause::Cancelled - ? "cancelled" - : "fence_or_lifecycle_lost"; - case CasUnresolvedReason::NoAttemptSent: - case CasUnresolvedReason::DeadlineMidWay: - if (context.stop_cause == CasOverwriteStopCause::Cancelled) - return "cancelled"; - if (context.stop_cause == CasOverwriteStopCause::FenceOrLifecycleLost) - return "fence_or_lifecycle_lost"; - return context.deadline_source == CasOverwriteDeadlineSource::ExternalLeaseSafety - ? "external_lease_deadline" - : "request_deadline"; - case CasUnresolvedReason::NotUnresolved: return "terminal_unclassified"; + case MountRenewTerminalClassification::Cancelled: return "cancelled"; + case MountRenewTerminalClassification::FenceOrLifecycleLost: return "fence_or_lifecycle_lost"; + case MountRenewTerminalClassification::ExternalLeaseDeadline: return "external_lease_deadline"; + case MountRenewTerminalClassification::RequestDeadline: return "request_deadline"; + case MountRenewTerminalClassification::Unresolved: return "unresolved"; + case MountRenewTerminalClassification::Unclassified: return "terminal_unclassified"; } return "terminal_unclassified"; } @@ -409,15 +344,12 @@ void deliverMountRenewObservability( const uint64_t now_boot_ms = defaultBootMs(); const String write_attempt_id = u128ToHex(context.write_attempt_id).substr(0, 12); - if (context.ambiguity_attempt_no != 0 && !context.retrying_delivered) - deliverMountRenewRetrying(context, write_attempt_id, now_boot_ms, remount_attempt_no); - for (uint32_t attempt_no = 2; attempt_no <= context.attempts_sent; ++attempt_no) { try { LOG_DEBUG( - getLogger("CasMountLeaseKeeper"), + getLogger("CasMountLeaseRenewer"), "CAS mount renewal '{}' physical retry attempt {} (writer_epoch={}, seq={})", *context.server_root_id, attempt_no, @@ -430,11 +362,11 @@ void deliverMountRenewObservability( } const bool recovered = context.outcome == MountRenewOutcome::Committed - && (context.attempts_sent > 1 || context.resolved_by_get); + && (context.attempts_sent > 1 || context.resolved_by_read); if (recovered) { - const std::string_view classification = context.resolved_by_get - ? "committed_by_get" + const std::string_view classification = context.resolved_by_read + ? "committed_by_read" : "committed_after_retry"; emitMountRenewEvent( context, @@ -442,15 +374,12 @@ void deliverMountRenewObservability( "recovered", context.attempts_sent, now_boot_ms, - context.unresolved_reason, - context.deadline_source, - context.stop_cause, classification, remount_attempt_no); try { LOG_INFO( - getLogger("CasMountLeaseKeeper"), + getLogger("CasMountLeaseRenewer"), "CAS mount renewal '{}' recovered after {} physical attempts in {} ms " "(classification={}, confirmed_deadline_boot_ms={})", *context.server_root_id, @@ -465,22 +394,19 @@ void deliverMountRenewObservability( } else if (context.outcome == MountRenewOutcome::Terminal) { - const std::string_view classification = terminalClassificationName(context); + const std::string_view classification = terminalClassificationName(context.terminal_classification); emitMountRenewEvent( context, write_attempt_id, "failed", context.attempts_sent, now_boot_ms, - context.unresolved_reason, - context.deadline_source, - context.stop_cause, classification, remount_attempt_no); try { LOG_WARNING( - getLogger("CasMountLeaseKeeper"), + getLogger("CasMountLeaseRenewer"), "CAS mount renewal '{}' fenced after {} physical attempts in {} ms " "(classification={}, confirmed_deadline_boot_ms={})", *context.server_root_id, @@ -504,9 +430,9 @@ void deliverMountRenewObservability( /// names the current mount holder in its DecommissionRecovery live-refusal message. String describeMountHolder(const MountLease & m); -std::optional readOwnerObject(Backend & b, const Layout & l, const String & server_root_id) +std::optional readOwnerObject(CasOperation & op, const Layout & l, const String & server_root_id) { - const auto got = b.get(l.ownerKey(server_root_id)); + const auto got = op.read(l.ownerKey(server_root_id), Retry::standard()); if (!got) return std::nullopt; return decodeOwner(got->bytes); @@ -537,48 +463,6 @@ void configureMountRenewObservability( }; } -void reportMountRenewProgress(const CasOverwriteProgress & progress) noexcept -{ - MountRenewObservabilityContext * context = currentMountRenewObservability(); - if (!context || !context->active) - return; - - switch (progress.kind) - { - case CasOverwriteProgressKind::PutStarted: - context->attempts_sent = std::max(context->attempts_sent, progress.attempt_no); - break; - case CasOverwriteProgressKind::BecameAmbiguous: - if (context->ambiguity_attempt_no == 0) - { - context->ambiguity_attempt_no = progress.attempt_no; - if (!context->deferred) - { - /// Mark first, because either diagnostic callback may synchronously renew another - /// Pool. The fixed observation stack keeps this outer snapshot stable. - context->retrying_delivered = true; - try - { - const uint64_t now_boot_ms = defaultBootMs(); - const String write_attempt_id = u128ToHex(context->write_attempt_id).substr(0, 12); - deliverMountRenewRetrying( - *context, write_attempt_id, now_boot_ms, /*remount_attempt_no=*/0); - } - catch (...) - { - /// First-ambiguity observability is diagnostic-only. The controller now runs - /// its pre-resolve gate before starting any additional backend I/O. - } - } - } - break; - case CasOverwriteProgressKind::RetryStarted: - case CasOverwriteProgressKind::ResolveStarted: - case CasOverwriteProgressKind::ResolvedByGet: - break; - } -} - void reportMountRenewCompletion(const MountRenewResult & result) noexcept { if (mount_renew_observability.suppressed_depth != 0) @@ -591,11 +475,8 @@ void reportMountRenewCompletion(const MountRenewResult & result) noexcept return; context->completed = true; context->outcome = result.outcome; - context->attempts_sent = std::max(context->attempts_sent, result.diagnostics.attempts_sent); - context->resolved_by_get = result.diagnostics.resolved_by_get; - context->unresolved_reason = result.diagnostics.unresolved_reason; - context->deadline_source = result.diagnostics.deadline_source; - context->stop_cause = result.diagnostics.stop_cause; + context->attempts_sent = std::max(context->attempts_sent, result.attempts_sent); + context->resolved_by_read = result.resolved_by_read; if (context->deferred) return; @@ -617,7 +498,7 @@ void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcep } bool serverRootSubtreeEmpty( - Backend & b, const Layout & l, const String & srid, const RefCatalog & catalog_observation) + CasOperation & op, const Layout & l, const String & srid, const RefCatalog & catalog_observation) { const String owned_prefix = srid + "/"; for (const CatalogEntry & entry : catalog_observation.entries) @@ -626,23 +507,23 @@ bool serverRootSubtreeEmpty( /// Manifests and loose roots retain logical path identity. Opaque namespace stream/state debris /// alone is not evidence that this server root owns live work. - if (prefixHasAnyKey(b, l.casManifestsServerPrefix(srid))) + if (prefixHasAnyKey(op, l.casManifestsServerPrefix(srid))) return false; - if (prefixHasAnyKey(b, l.serverRootDataPrefix(srid))) + if (prefixHasAnyKey(op, l.serverRootDataPrefix(srid))) return false; return true; } -std::optional readOwnerUuid(Backend & b, const Layout & l, const String & server_root_id) +std::optional readOwnerUuid(CasOperation & op, const Layout & l, const String & server_root_id) { - const std::optional owner = readOwnerObject(b, l, server_root_id); + const std::optional owner = readOwnerObject(op, l, server_root_id); if (!owner) return std::nullopt; return owner->server_uuid; } void claimOwnerOrThrow( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, const ObserveRefCatalog & observe_catalog) { if (!observe_catalog) @@ -651,7 +532,7 @@ void claimOwnerOrThrow( /// Owner present → it is identity: equal UUID is ok, a different UUID fails closed regardless /// of any lease/clock state. - if (const std::optional owner = readOwnerObject(b, l, srid)) + if (const std::optional owner = readOwnerObject(op, l, srid)) { if (owner->server_uuid == our_uuid) { @@ -673,27 +554,33 @@ void claimOwnerOrThrow( /// Owner absent. Claiming is allowed ONLY over a provably-empty subtree; an absent owner over /// existing data means the identity was lost and must never be silently re-claimed. - if (!serverRootSubtreeEmpty(b, l, srid, observe_catalog())) + if (!serverRootSubtreeEmpty(op, l, srid, observe_catalog())) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-root '{}' has no owner anchor but its data subtree is non-empty " "(identity lost over existing data) — refusing to re-claim", srid); - const PutResult put = b.putIfAbsent(key, encodeOwner(OwnerObject{ - .server_uuid = our_uuid, - .retired_at_ms = std::nullopt, - })); - if (put.outcome == PutOutcome::Done) + const std::optional occupant = conflictOrThrow( + op.create(key, encodeOwner(OwnerObject{.server_uuid = our_uuid, .retired_at_ms = std::nullopt}), + Retry::standard()), + fmt::format("CAS server-root '{}' owner claim", srid)); + if (!occupant) return; /// The conditional create conflicted. Recompute the whole catalog + manifest + roots bundle; /// no stale emptiness result is carried across the conflict. - if (!serverRootSubtreeEmpty(b, l, srid, observe_catalog())) + if (!serverRootSubtreeEmpty(op, l, srid, observe_catalog())) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-root '{}' owner claim conflicted and newly visible owned work blocks recreation", srid); - /// Race: another process claimed between our get and our putIfAbsent. Re-read and compare. - const std::optional reread = readOwnerObject(b, l, srid); + /// Race: another process claimed between our read and our create. The write's own resolve read + /// already observed who took the key, and reading again would answer a later question than the one + /// the conflict asked. Only an observation that settled nothing still owes a read. + std::optional reread; + if (const Object * observed = std::get_if(&*occupant)) + reread = decodeOwner(observed->bytes); + else if (!std::holds_alternative(*occupant)) + reread = readOwnerObject(op, l, srid); if (!reread) throw Exception(ErrorCodes::CORRUPTED_DATA, "CAS server-root '{}' owner anchor vanished during claim", srid); @@ -709,118 +596,115 @@ void claimOwnerOrThrow( } uint64_t allocateWriterEpoch( - Backend & b, const Layout & l, const String & srid, EpochMintPolicy policy, uint64_t now_ms, + CasOperation & op, const Layout & l, const String & srid, EpochMintPolicy policy, uint64_t now_ms, const ObserveRefCatalog & observe_catalog) { if (!observe_catalog) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS server-root '{}': catalog observer is required", srid); const String key = l.epochKey(srid); - static constexpr int max_attempts = 100; - for (int attempt = 0; attempt < max_attempts; ++attempt) - { - const auto got = b.get(key); + uint64_t allocated = 0; + /// Set when the PREVIOUS decision wrote against an absent epoch. Its conflict means a winner may + /// have installed an epoch while owned work became visible, so the emptiness bundle that + /// authorized that attempt is recomputed before this decision accepts any epoch state at all. + bool previous_decision_saw_no_epoch = false; - ServerEpoch current; - std::optional expected; - if (got) - { - current = decodeServerEpoch(got->bytes); - expected = got->token; - } - else + WriteResult result = op.readModifyWrite(key, + [&](const std::optional & observed) -> std::optional { - /// A missing `epoch` over a non-empty subtree is a reset hazard (durable monotone - /// counter cannot be reconstructed) — fail closed. - if (!serverRootSubtreeEmpty(b, l, srid, observe_catalog())) + if (previous_decision_saw_no_epoch && !serverRootSubtreeEmpty(op, l, srid, observe_catalog())) throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}' has no durable epoch object but its data subtree is " - "non-empty (writer_epoch reset hazard) — refusing to proceed", + "CAS server-root '{}' writer_epoch allocation conflicted and newly visible owned " + "work blocks recreation", srid); + previous_decision_saw_no_epoch = !observed; - /// Same hazard through the CONTROL objects (spec rev.4 Phase C): an absent epoch while - /// a mount object exists means epoch state was lost under a live/recent mount — - /// re-minting epoch 1 there is how a same-(uuid, epoch) twin is born. This is a - /// lifecycle decision, so it uses the authoritative probe, never get-absence. - const SentinelProbeResult mount_probe = b.probeSentinelRaw(l.mountKey(srid)); - switch (mount_probe.outcome) + ServerEpoch current; + if (observed) { - case ProbeOutcome::KeyAbsent: - break; /// authoritative absence — fresh-root bootstrap proceeds below - case ProbeOutcome::Present: + current = decodeServerEpoch(observed->bytes); + } + else + { + /// A missing `epoch` over a non-empty subtree is a reset hazard (durable monotone + /// counter cannot be reconstructed) — fail closed. + if (!serverRootSubtreeEmpty(op, l, srid, observe_catalog())) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS server-root '{}' has no durable epoch object but its data subtree is " + "non-empty (writer_epoch reset hazard) — refusing to proceed", + srid); + + /// Same hazard through the CONTROL objects: an absent epoch while a mount object + /// exists means epoch state was lost under a live/recent mount — re-minting epoch 1 + /// there is how a same-(uuid, epoch) twin is born. This is a lifecycle decision, so it + /// uses the authoritative probe, never a read's absence (which flattens transport + /// faults into "not found"). + const SentinelProbeResult mount_probe = op.probeSentinel(l.mountKey(srid), Retry::standard()); + switch (mount_probe.outcome) { - if (policy == EpochMintPolicy::DecommissionRecovery) + case ProbeOutcome::KeyAbsent: + break; /// authoritative absence — fresh-root bootstrap proceeds below + case ProbeOutcome::Present: { - chassert(now_ms != 0); /// the decommission caller must pass its clock - const MountLease surviving = decodeMountLease(*mount_probe.body); - /// Deliberately weaker than claimMount's reclaim gate (this file, ~:370-380), - /// which never trusts a bare wall-clock comparison alone (only gc_fenced / - /// the clean-farewell min_active==UINT64_MAX marker / a caller-proven-dead - /// token justify a reclaim there, because clock skew can misjudge liveness). - /// This is still safe: (a) the mint below is DISTINCT from the survivor's - /// epoch by construction, so no same-(uuid, epoch) pair is ever representable - /// even if this liveness read is wrong; (b) claimMount right after this still - /// applies its own STRONG liveness gate and refuses a genuinely live member - /// regardless of what happens here. So a clock-skewed "terminal" misread can - /// only burn one epoch number on a doomed decommission attempt that aborts at - /// claimMount — it can never admit a claim over a live member. - const bool live = !surviving.gc_fenced && surviving.expires_at_ms > now_ms; - if (live) - throw Exception(ErrorCodes::ABORTED, - "CAS decommission '{}': epoch object missing but a LIVE mount lease " - "exists ({}) — refusing to re-mint an epoch under a live member " - "(stop the server or wait for its lease to lapse)", - srid, describeMountHolder(surviving)); - /// Terminal mount: proceed, but mint an epoch DISTINCT from the survivor's - /// by construction — the same-pair state is unrepresentable on this path. - current.next_writer_epoch = std::max(1, surviving.writer_epoch + 1); - break; + if (policy == EpochMintPolicy::DecommissionRecovery) + { + chassert(now_ms != 0); /// the decommission caller must pass its clock + const MountLease surviving = decodeMountLease(*mount_probe.body); + /// Deliberately weaker than claimMount's reclaim gate, which never trusts a + /// bare wall-clock comparison alone (only gc_fenced / the clean-farewell + /// min_active_build_sequence==UINT64_MAX marker / a caller-proven-dead + /// incarnation justify a reclaim there, because clock skew can misjudge + /// liveness). This is still safe: (a) the mint below is DISTINCT from the + /// survivor's epoch by construction, so no same-(uuid, epoch) pair is ever + /// representable even if this liveness read is wrong; (b) claimMount right + /// after this still applies its own STRONG liveness gate and refuses a + /// genuinely live member regardless of what happens here. So a clock-skewed + /// "terminal" misread can only burn one epoch number on a doomed + /// decommission attempt that aborts at claimMount — it can never admit a + /// claim over a live member. + const bool live = !surviving.gc_fenced && surviving.expires_at_ms > now_ms; + if (live) + throw Exception(ErrorCodes::ABORTED, + "CAS decommission '{}': epoch object missing but a LIVE mount lease " + "exists ({}) — refusing to re-mint an epoch under a live member " + "(stop the server or wait for its lease to lapse)", + srid, describeMountHolder(surviving)); + /// Terminal mount: proceed, but mint an epoch DISTINCT from the survivor's + /// by construction — the same-pair state is unrepresentable on this path. + current.next_writer_epoch = std::max(1, surviving.writer_epoch + 1); + break; + } + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS server-root '{}' has no durable epoch object but a mount lease exists — " + "durable epoch state was lost while a mount is live or recently live; " + "refusing to re-mint epoch 1. If no server is live on this root, " + "decommission it or manually remove the stale mount object '{}'.", + srid, l.mountKey(srid)); } - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}' has no durable epoch object but a mount lease exists — " - "durable epoch state was lost while a mount is live or recently live; " - "refusing to re-mint epoch 1. If no server is live on this root, " - "decommission it or manually remove the stale mount object '{}'.", - srid, l.mountKey(srid)); + case ProbeOutcome::ContainerAbsent: + case ProbeOutcome::AccessDenied: + case ProbeOutcome::Indeterminate: + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS server-root '{}': cannot verify mount-lease absence before re-minting " + "the writer epoch (probe outcome: {}) — absence was never proven; failing closed", + srid, magic_enum::enum_name(mount_probe.outcome)); } - case ProbeOutcome::ContainerAbsent: - case ProbeOutcome::AccessDenied: - case ProbeOutcome::Indeterminate: - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}': cannot verify mount-lease absence before re-minting " - "the writer epoch (probe outcome: {}) — absence was never proven; failing closed", - srid, magic_enum::enum_name(mount_probe.outcome)); - } - - if (current.next_writer_epoch == 0) - current.next_writer_epoch = 1; - } - - const uint64_t next = current.next_writer_epoch; - ServerEpoch new_state; - new_state.next_writer_epoch = next + 1; - const CasResult res = b.casPut(key, encodeServerEpoch(new_state), expected); - if (res.outcome == CasOutcome::Committed) - return next; - if (!got) - { - /// The absent-epoch create conflicted. A winner may have installed an epoch while owned - /// work became visible, so recompute the complete catalog + manifest + roots bundle - /// before the next iteration is allowed to accept either a present or absent epoch. - if (!serverRootSubtreeEmpty(b, l, srid, observe_catalog())) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}' writer_epoch allocation conflicted and newly visible owned " - "work blocks recreation", - srid); - } - /// Conflict: someone else allocated concurrently — retry against fresh state only after the - /// absent-epoch safety bundle above has been recomputed when required. - } + if (current.next_writer_epoch == 0) + current.next_writer_epoch = 1; + } - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}' writer_epoch allocation did not converge after {} attempts", - srid, max_attempts); + allocated = current.next_writer_epoch; + return encodeServerEpoch(ServerEpoch{.next_writer_epoch = allocated + 1}); + }, + Retry::standard()); + + /// Non-convergence used to be `CORRUPTED_DATA`, on the reasoning that a hundred lost conditional + /// writes really is evidence of something wrong. The bound is a wall-clock deadline now, and ninety + /// seconds of a throttled store is not evidence of anything, so `orThrow`'s retry-later class is + /// the honest verdict. Both fail closed at `Pool::open`. + orThrow(std::move(result), fmt::format("CAS server-root '{}' writer_epoch allocation", srid)); + return allocated; } namespace @@ -901,25 +785,25 @@ void emitMountEvent(const CasEventSink & sink, CasEventType type, const String & } MountClaimResult claimMount( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, - uint64_t now_ms, uint64_t ttl_ms, const std::optional & proven_dead_token, - const CasEventSink & sink) + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, + uint64_t now_ms, uint64_t ttl_ms, const std::optional & proven_dead_incarnation, + const CasEventSink & sink, const std::optional & unsafe_reclaim_authorization) { const String key = l.mountKey(srid); - const auto got = b.get(key); + const auto got = op.read(key, Retry::standard()); /// Absent → fresh claim. if (!got) { const MountLease body = makeMountBody(our_uuid, our_epoch, /*seq=*/ 1, now_ms, ttl_ms); - const PutResult put = b.putIfAbsent(key, encodeMountLease(body)); - if (put.outcome != PutOutcome::Done) - /// Raced with a concurrent writer between get and putIfAbsent. Treat as a live double - /// start — fail closed; never overwrite a slot that appeared under us. No re-read was - /// done, so no conflicting identity is known to attach to an event. - return {.kind = MountClaimResult::LiveDoubleStart, .body = body, .token = std::nullopt}; + if (const std::optional raced + = conflictOrThrow(op.create(key, encodeMountLease(body), Retry::standard()), + fmt::format("CAS mount slot claim of '{}'", key))) + /// Raced with a concurrent writer between the read and the create. Treat as a live double + /// start — fail closed; never overwrite a slot that appeared under us. + return racedDoubleStart(*raced); emitMountEvent(sink, CasEventType::MountClaim, srid, "mint", nullptr, "fresh mount slot minted"); - return {.kind = MountClaimResult::Claimed, .body = body, .token = std::nullopt}; + return {.kind = MountClaimResult::Claimed, .body = body, .etag = std::nullopt}; } const MountLease existing = decodeMountLease(got->bytes); @@ -930,7 +814,7 @@ MountClaimResult claimMount( { emitMountEvent(sink, CasEventType::MountConflict, srid, "foreign_owner", &existing, "mount slot is held by a foreign server_uuid — refusing to take over across identities"); - return {.kind = MountClaimResult::ForeignOwner, .body = existing, .token = std::nullopt}; + return {.kind = MountClaimResult::ForeignOwner, .body = existing, .etag = std::nullopt}; } /// Same uuid + same epoch: it is OUR OWN claim — but a FENCED body is terminal for this @@ -944,97 +828,108 @@ MountClaimResult claimMount( emitMountEvent(sink, CasEventType::MountConflict, srid, "fenced_by_gc", &existing, "own (uuid, epoch) mount slot is GC-fenced — terminal for this incarnation; " "recover with a fresh writer_epoch"); - return {.kind = MountClaimResult::FencedSelf, .body = existing, .token = std::nullopt}; + return {.kind = MountClaimResult::FencedSelf, .body = existing, .etag = std::nullopt}; } const MountLease body = makeMountBody(our_uuid, our_epoch, existing.seq + 1, now_ms, ttl_ms); - const PutResult put = b.putOverwrite(key, encodeMountLease(body), got->token); - if (put.outcome != PutOutcome::Done) - /// The mount changed under us between get and putOverwrite: `got->token` is now KNOWN - /// STALE (that mismatch is exactly why the put failed), not merely unknown -- leaving - /// `.token` unset (rather than handing back a token the caller would wrongly treat as - /// current) is deliberate, matching the identical race below. - return {.kind = MountClaimResult::LiveDoubleStart, .body = body, .token = std::nullopt}; + if (const std::optional raced + = conflictOrThrow(op.replace(key, encodeMountLease(body), got->etag, Retry::standard()), + fmt::format("CAS mount slot refresh of '{}'", key))) + /// The mount changed under us between the read and the write, so `got->etag` is KNOWN + /// STALE -- that mismatch is exactly why the write was refused. What the write's resolve + /// read observed is current, and it is that pair that is reported. + return racedDoubleStart(*raced); emitMountEvent(sink, CasEventType::MountClaim, srid, "refresh", &existing, "own claim replayed — refreshed seq + expiry"); - return {.kind = MountClaimResult::Claimed, .body = body, .token = std::nullopt}; + return {.kind = MountClaimResult::Claimed, .body = body, .etag = std::nullopt}; } /// Same uuid, DIFFERENT epoch: reclaim ONLY on a certificate of death that needs no fresh /// wall-clock trust — never by comparing `expires_at_ms` against `now_ms`: - /// - `gc_fenced` → the fence-out is terminal for that incarnation by construction (its keeper's + /// - `gc_fenced` → the fence-out is terminal for that incarnation by construction (its renewer's /// every renewal fails the token guard forever, so it can never write again) — there is no /// liveness left to wait for. This is what makes self-remount (and a fast restart after a /// fence-out) instant instead of an observation wait. - /// - the clean marker (`min_active == UINT64_MAX`) → the predecessor's OWN graceful farewell - /// (`MountLeaseKeeper::terminate`) — no observation needed either. - /// - `proven_dead_token` matches the token we just read → the CALLER (`claimMountAwaitingExpiry`) - /// already watched this exact token hold stable for the full observation threshold on its own - /// clock; re-deriving that here from a bare wall-clock comparison would be exactly the - /// cross-node trust would make a clock-skewed or delayed observer unsafe. + /// - the clean marker (`min_active_build_sequence == UINT64_MAX`) → the predecessor's OWN graceful farewell + /// (`MountLeaseRenewer::terminate`) — no observation needed either. + /// - `proven_dead_incarnation` matches the one we just read → the CALLER + /// (`claimMountAwaitingExpiry`) already watched that exact incarnation hold stable for the full + /// observation threshold on its own clock; re-deriving that here from a bare wall-clock + /// comparison is exactly the cross-node trust that makes a clock-skewed or delayed observer + /// unsafe. + /// - `unsafe_reclaim_authorization` matches the one we just read → the operator's + /// `cas_unsafe_remount_no_delay` setting explicitly authorized this reclaim with NO + /// observation at all; the caller read this exact token and accepted the availability risk. /// Anything else → `LiveDoubleStart` (do NOT write): a same-uuid, different-epoch, not fenced, not - /// clean-marked, not (yet) proven-dead lease may simply be a live twin, and `expires_at_ms` alone - /// can never distinguish that from a dead predecessor across two different clocks. - const bool clean_marker = existing.min_active == std::numeric_limits::max(); - const bool proven_dead = proven_dead_token && *proven_dead_token == got->token; - if (existing.gc_fenced || clean_marker || proven_dead) + /// clean-marked, not (yet) proven-dead, not unsafe-authorized lease may simply be a live twin, and + /// `expires_at_ms` alone can never distinguish that from a dead predecessor across two different + /// clocks. + const bool clean_marker = existing.min_active_build_sequence == std::numeric_limits::max(); + const bool proven_dead = proven_dead_incarnation && *proven_dead_incarnation == got->etag; + const bool unsafe_authorized = unsafe_reclaim_authorization && *unsafe_reclaim_authorization == got->etag; + if (existing.gc_fenced || clean_marker || proven_dead || unsafe_authorized) { const MountLease body = makeMountBody(our_uuid, our_epoch, existing.seq + 1, now_ms, ttl_ms); - const PutResult put = b.putOverwrite(key, encodeMountLease(body), got->token); - if (put.outcome != PutOutcome::Done) - /// The mount changed under us between get and putOverwrite — someone else is racing the - /// reclaim. Fail closed. `got->token` is now KNOWN STALE (that mismatch is exactly why the - /// put failed) -- leaving `.token` unset is deliberate, not an oversight. - return {.kind = MountClaimResult::LiveDoubleStart, .body = body, .token = std::nullopt}; + if (const std::optional raced + = conflictOrThrow(op.replace(key, encodeMountLease(body), got->etag, Retry::standard()), + fmt::format("CAS mount slot reclaim of '{}'", key))) + /// The mount changed under us between the read and the write — someone else is racing the + /// reclaim. Fail closed, and report what the write's resolve read observed rather than + /// `got->etag`, which that mismatch just proved stale. + return racedDoubleStart(*raced); const MountPriorState prior = existing.gc_fenced ? MountPriorState::Fenced : clean_marker ? MountPriorState::Clean - : MountPriorState::UncleanObserved; + : proven_dead ? MountPriorState::UncleanObserved + : MountPriorState::UncleanUnsafe; emitMountEvent(sink, CasEventType::MountClaim, srid, "reclaim", &existing, existing.gc_fenced ? "same server_uuid, different writer_epoch, GC-fenced — reclaimed" : clean_marker ? "same server_uuid, different writer_epoch, clean farewell — reclaimed" - : "same server_uuid, different writer_epoch, observed dead by " - "token-stability — reclaimed"); - return {.kind = MountClaimResult::Claimed, .body = body, .prior = prior, .token = std::nullopt}; + : proven_dead ? "same server_uuid, different writer_epoch, observed dead by " + "token-stability observation — reclaimed" + : "same server_uuid, different writer_epoch, reclaimed at once under " + "cas_unsafe_remount_no_delay — the operator accepted that a live " + "predecessor with this uuid may still be writing"); + return {.kind = MountClaimResult::Claimed, .body = body, .prior = prior, .etag = std::nullopt}; } emitMountEvent(sink, CasEventType::MountConflict, srid, "live_double_start", &existing, "same server_uuid, different writer_epoch, not fenced/clean/proven-dead — no wall-clock trust; " "the caller must run the token-stability observation wait before reclaiming"); - /// No write was attempted on this path -- `got->token` is exactly the CURRENT body's - /// token (what we just read is what's still there), so it is safe to hand back for the caller's - /// observation loop to compare across polls without a redundant re-GET. - return {.kind = MountClaimResult::LiveDoubleStart, .body = existing, .token = got->token}; + /// No write was attempted on this path -- `got->etag` is exactly the CURRENT body's + /// etag (what we just read is what's still there), so it is safe to hand back for the + /// caller's observation loop to compare across polls without a redundant re-read. + return {.kind = MountClaimResult::LiveDoubleStart, .body = existing, .etag = got->etag}; } -String mountDoubleStartMessage(const String & srid, const MountLease & existing) +String mountDoubleStartMessage(const String & srid, const std::optional & existing) { + const String identity = existing + ? fmt::format("server_uuid={} hostname={} pid={} last_seq={} expires_at_ms={}", + u128ToHex(existing->server_uuid), existing->hostname, existing->pid, + existing->seq, existing->expires_at_ms) + : String("could not be observed -- the conditional write that lost this slot saw nothing at " + "the key, so the holder's identity is unknown to this server"); return fmt::format( "Content-addressed disk cannot start: server_root_id '{}' is actively mounted by another LIVE server.\n" - " Existing mount: server_uuid={} hostname={} pid={} last_seq={} expires_at_ms={}\n" + " Existing mount: {}\n" "This server already waited for the mount lease to lapse, but it kept being renewed — a second\n" "server is holding the same CAS namespace. This prevents two ClickHouse servers from writing it.\n" " - If the other server is running intentionally, configure a unique for this disk.\n" " - If the other server is a stale/zombie process, stop it; this server will then reclaim the mount on restart.\n" - " - CLOCK SKEW CAVEAT: liveness is judged by comparing the lease's wall-clock expires_at_ms against\n" - " THIS server's clock, so a large clock skew between the two servers can misjudge it (a healthy holder\n" - " may look mounted here, or a dead one may look live). Verify both servers' clocks are in sync (NTP).\n" + " - LIVENESS: this wait judges the holder alive by its write token holding stable on THIS server's\n" + " own clock for the full observation threshold; the stamped expires_at_ms above never enters that\n" + " judgment on its own -- it is a writer-stamped diagnostic (also shown in system.cas_mounts), not\n" + " an authorization. Every server sharing this pool must run the SAME cas_mount_lease_ttl_ms and\n" + " cas_mount_renew_period_ms: a server configured with a shorter threshold than its peers can fence\n" + " out a healthy one.\n" " - If the local ClickHouse uuid file was regenerated, restore the old uuid file, or remove the stale\n" " owner object gc/server-roots/{}/owner only after verifying no server uses this root.\n" " - As a LAST RESORT, after verifying that NO server is writing this root, manually delete the mount\n" - " object gc/server-roots/{}/mount and restart; this server will then re-claim it.", - srid, u128ToHex(existing.server_uuid), existing.hostname, existing.pid, - existing.seq, existing.expires_at_ms, srid, srid); -} - -namespace -{ -/// Bounded number of observation restarts before giving up on a same-uuid slot whose write-token keeps -/// changing: each restart means the token changed DURING our observation window — i.e. something is -/// actively renewing it. A genuinely dead predecessor's token never changes again after its last -/// renewal, so it is observed stable well within one window; only a truly LIVE writer (a real second -/// incarnation, or the predecessor's own background renewer racing our first few polls) keeps resetting -/// the clock. Bounding this converts "wait forever for a live twin" into the same bounded-then-report -/// shape the old wall-clock wait had, without ever trusting a wall-clock deadline to get there. -constexpr size_t kMaxObservationRestarts = 3; + " object gc/server-roots/{}/mount and restart; this server will then re-claim it.\n" + " - For a test stand or a deployment that guarantees one process per server_uuid,\n" + " cas_unsafe_remount_no_delay reclaims a slot carrying this server's own uuid at once instead of\n" + " waiting -- but a live predecessor sharing this uuid may still be writing, so enable it only\n" + " under that guarantee.", + srid, identity, srid, srid); } uint64_t mountObservationThresholdMs(uint64_t ttl_ms, uint64_t cadence_ms) @@ -1043,7 +938,7 @@ uint64_t mountObservationThresholdMs(uint64_t ttl_ms, uint64_t cadence_ms) } MountClaimResult claimMountAwaitingExpiry( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, const std::function & now_ms_fn, const std::function & mono_ms_fn, uint64_t ttl_ms, uint64_t poll_interval_ms, @@ -1057,58 +952,61 @@ MountClaimResult claimMountAwaitingExpiry( /// Rate-bound observation threshold: the full lease TTL, plus a 5% allowance for clock-rate /// mismatch between the holder's and our own local clock, plus one poll interval for observation /// discreteness. It is measured only with OUR OWN clock (`mono_ms_fn`); no cross-node wall-clock - /// comparison participates in this loop. The shared helper keeps the startup and GC thresholds - /// identical. + /// comparison participates in this loop. `poll` here is half the renewal period (the caller's own + /// poll cadence), so this threshold is close to, but not identical to, GC's heartbeat fence-out + /// threshold, which passes the full renewal period into the same shared helper. const uint64_t threshold_ms = mountObservationThresholdMs(ttl_ms, poll); - std::optional observed; + std::optional observed; uint64_t observed_since = 0; size_t restarts = 0; while (true) { const bool threshold_met = observed && mono_ms_fn() - observed_since >= threshold_ms; - MountClaimResult r = claimMount(b, l, srid, our_uuid, our_epoch, now_ms_fn(), ttl_ms, - threshold_met ? observed : std::nullopt, sink); + MountClaimResult r = claimMount(op, l, srid, our_uuid, our_epoch, now_ms_fn(), ttl_ms, + threshold_met ? observed : std::nullopt, sink, /*unsafe_reclaim_authorization=*/{}); if (r.kind != MountClaimResult::LiveDoubleStart) return r; - /// `claimMount` already read the current body. Reuse `r.token` whenever `claimMount` - /// set it (the common case: no write was attempted, so what it read is still current) instead of - /// re-GETting the SAME key here. The rare stale-race branches deliberately leave `.token` unset - /// (see their own comments), so this still falls back to a fresh read exactly there. - std::optional current_token = r.token; - if (!current_token) + /// `claimMount` already read the current body, and a raced write reports whatever its own + /// resolve read observed. Reuse `r.etag` whenever it is set instead of re-reading the SAME key + /// here; only a raced write whose conflict observed nothing at all leaves it unset, and that is + /// exactly where this reads -- for the body as well as the incarnation, since a result with no + /// observation has no holder to report either. + std::optional current_etag = r.etag; + if (!current_etag) { - const auto got = b.get(l.mountKey(srid)); + const auto got = op.read(l.mountKey(srid), Retry::standard()); if (!got) { /// The slot vanished between claimMount's own GET and ours — normally self-resolving /// within one more `claimMount` call (which re-mints fresh on an absent slot), but under /// slot churn (something else concurrently removing/re-minting it) that resolution could /// keep losing the same race. Pace this like every other iteration and - /// count it toward the SAME bounded restart budget the token-churn case below uses, - /// instead of spinning `get`/`claimMount`/`put` at backend RTT with no sleep and no bound + /// count it toward the SAME bounded restart budget the incarnation-churn case below + /// uses, instead of spinning read/claim/write at backend RTT with no sleep and no bound /// — a persistently vanishing slot is exactly as "alive and contended" as a persistently - /// renewing token. + /// renewing holder. if (++restarts > kMaxObservationRestarts) return r; sleep_ms_fn(poll); continue; } - current_token = got->token; + current_etag = got->etag; + r.body = decodeMountLease(got->bytes); } - if (!observed || *observed != *current_token) + if (!observed || *observed != *current_etag) { if (observed && ++restarts > kMaxObservationRestarts) - /// The token kept changing across bounded restarts — the holder is genuinely alive + /// The incarnation kept changing across bounded restarts — the holder is genuinely alive /// (actively renewing), not a dead predecessor. Report it rather than waiting forever. return r; - observed = *current_token; + observed = *current_etag; observed_since = mono_ms_fn(); - if (on_wait_start) - on_wait_start(r.body, threshold_ms); + if (on_wait_start && r.body) + on_wait_start(*r.body, threshold_ms); LOG_INFO(getLogger("CasMountLease"), "Attempting to mount content-addressed server root {} after node change or hard " "restart; waiting ~{} ms (token-stability observation) to confirm the previous " @@ -1119,7 +1017,7 @@ MountClaimResult claimMountAwaitingExpiry( } } -HeartbeatFloor computeHeartbeatFloor(Backend & b, const Layout & l, uint64_t now_ms, +HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64_t now_ms, uint64_t mono_now_ms, uint64_t stable_threshold_ms, MountObservationMap & obs) { @@ -1127,224 +1025,198 @@ HeartbeatFloor computeHeartbeatFloor(Backend & b, const Layout & l, uint64_t now /// `obs` is keyed by every srid this leader has EVER observed, but a /// srid removed from the LIST entirely (its `/mount` key gone -- e.g. `SYSTEM CAS - /// DROP POOL MEMBER`) is never visited by the loop below again, so its entry would otherwise linger + /// DROP POOL MEMBER`) is never visited by the walk below again, so its entry would otherwise linger /// forever (~150-250 B/srid, worse on a long-lived leader across many decommissions). Track every /// srid actually seen THIS pass and prune anything else out of `obs` at the end -- disjoint from the - /// mid-loop `obs.erase(srid)` calls below (those fire for a srid seen but now terminal/fenced/gone + /// mid-walk `obs.erase(srid)` calls below (those fire for a srid seen but now terminal/fenced/gone /// this pass; this is for a srid not seen AT ALL). std::set seen_srids; const String prefix = l.serverRootsPrefix(); - String cursor; - while (true) + op.forEachListedKey(prefix, [&](const ListedKey & listed) { - const ListPage page = b.list(prefix, cursor, /*limit*/ 1000); - for (const auto & listed : page.keys) - { - /// `/owner` and `/epoch` objects share the subtree — only mount bodies gate the floor. - static constexpr std::string_view mount_suffix = "/mount"; - if (!listed.key.ends_with(mount_suffix)) - continue; - - const String & key = listed.key; - - /// The srid is the path segment between `serverRootsPrefix()` and the `/mount` suffix - /// (`/gc/server-roots//mount`). Used both for observability (fenced) and as - /// the key into `obs`. - const String srid = key.substr(prefix.size(), - key.size() - prefix.size() - mount_suffix.size()); - seen_srids.insert(srid); - - /// Fence-out on PreconditionFailed re-GETs and reclassifies from the top; bound the retries - /// so a pathologically contended holder cannot spin forever. On exhaustion the entry is - /// counted as live (conservative — never excluded without a landed fence-out). - constexpr int max_reclassify = 4; - for (int attempt = 0; ; ++attempt) + /// `/owner` and `/epoch` objects share the subtree — only mount bodies gate the floor. + static constexpr std::string_view mount_suffix = "/mount"; + if (!listed.key.ends_with(mount_suffix)) + return true; + + const String & key = listed.key; + + /// The srid is the path segment between `serverRootsPrefix()` and the `/mount` suffix + /// (`/gc/server-roots//mount`). Used both for observability (fenced) and as + /// the key into `obs`. + const String srid = key.substr(prefix.size(), key.size() - prefix.size() - mount_suffix.size()); + seen_srids.insert(srid); + + /// One decision per re-read: a refused fence-out re-enters this lambda with the body the + /// holder's own renewal installed, and the observation check below then sees the new + /// incarnation and restarts the window -- which counts the slot `live` and declines the write. + /// That is why no arm that counts a slot ever also asks for a fence-out body. + WriteResult fenced_out = op.readModifyWrite(key, + [&](const std::optional & observed) -> std::optional { - const auto got = b.get(key); - if (!got) + if (!observed) { obs.erase(srid); - break; /// Raced away (deleted) — nothing to classify. + return std::nullopt; /// raced away (deleted) — nothing to classify } - const MountLease m = decodeMountLease(got->bytes); + const MountLease m = decodeMountLease(observed->bytes); if (m.gc_fenced) { ++floor.already_fenced; obs.erase(srid); /// terminal — no further observation needed - break; + return std::nullopt; } - if (m.min_active == std::numeric_limits::max()) + if (m.min_active_build_sequence == std::numeric_limits::max()) { ++floor.terminated; obs.erase(srid); /// terminal — no further observation needed - break; + return std::nullopt; } - /// Observation-based liveness: stable ONLY if the - /// SAME token was already being watched and has now held for the full threshold on our - /// OWN monotonic clock. Anything else — no prior observation, or a changed token (a - /// live renewal, including one raced against our own fence-out attempt below) — - /// (re)starts the observation window and counts as `live` this call. + /// Observation-based liveness: stable ONLY if the SAME incarnation was already being + /// watched and has now held for the full threshold on our OWN monotonic clock. Anything + /// else — no prior observation, or a changed incarnation (a live renewal, including one + /// raced against our own fence-out attempt) — (re)starts the observation window and + /// counts as `live` this call. const auto it = obs.find(srid); - const bool stable = it != obs.end() && it->second.token == got->token + const bool stable = it != obs.end() && it->second.etag == observed->etag && mono_now_ms - it->second.first_seen_mono_ms >= stable_threshold_ms; if (!stable) { - if (it == obs.end() || it->second.token != got->token) - obs[srid] = MountTokenObservation{got->token, mono_now_ms}; + if (it == obs.end() || it->second.etag != observed->etag) + obs.insert_or_assign(srid, MountIncarnationObservation{observed->etag, mono_now_ms}); ++floor.live; - break; + return std::nullopt; } - const bool exhausted = attempt >= max_reclassify; - if (exhausted) - { - ++floor.live; /// conservative — never exclude without a landed fence-out - break; - } - - /// Stable past the threshold, not yet fenced → token-guarded fence-out preserving the - /// whole body (gc_fenced = true, seq + 1). + /// Stable past the threshold, not yet fenced → fence-out preserving the whole body + /// (gc_fenced = true, seq + 1) against the incarnation this decision observed. MountLease fenced = m; fenced.gc_fenced = true; fenced.seq = m.seq + 1; - const PutResult res = b.putOverwrite(key, encodeMountLease(fenced), got->token); - if (res.outcome == PutOutcome::Done) - { - ++floor.fenced_now; - floor.fenced_srids.push_back(srid); - obs.erase(srid); - LOG_INFO(getLogger("CasHeartbeatFloor"), - "CAS GC fenced out mount lease for content-addressed server root {} at " - "wall-clock ms {}: its write token held unchanged for >= {} ms on the GC " - "leader's own monotonic clock (token-stability observation)", - srid, now_ms, stable_threshold_ms); - break; - } - /// PreconditionFailed: the holder renewed between our GET and PUT — re-GET and - /// reclassify (the observation check above will see the new token and restart it). - } - } - - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } + return encodeMountLease(fenced); + }, + Retry::standard()); - /// Prune every `obs` entry for a srid this pass's LIST never saw at all. + if (std::holds_alternative(fenced_out)) + { + ++floor.fenced_now; + floor.fenced_srids.push_back(srid); + obs.erase(srid); + LOG_INFO(getLogger("CasHeartbeatFloor"), + "CAS GC fenced out mount lease for content-addressed server root {} at " + "wall-clock ms {}: its write incarnation held unchanged for >= {} ms on the GC " + "leader's own monotonic clock (token-stability observation)", + srid, now_ms, stable_threshold_ms); + return true; + } + /// Declined: the decision above already classified and counted this slot, and asked for no + /// write. Every remaining verdict means the store was not reached, which is not a + /// classification -- surface it rather than record a floor built on an unread slot. + if (!std::holds_alternative(fenced_out)) + orThrow(std::move(fenced_out), fmt::format("CAS mount fence-out of '{}'", key)); + return true; + }, Retry::standard()); + + /// Prune every `obs` entry for a srid this pass's walk never saw at all. for (auto it = obs.begin(); it != obs.end(); ) it = seen_srids.contains(it->first) ? std::next(it) : obs.erase(it); return floor; } -std::vector probeNonTerminalMountSlots(Backend & b, const Layout & l) +std::vector probeNonTerminalMountSlots(CasOperation & op, const Layout & l) { std::vector slots; - /// Same enumeration as `computeHeartbeatFloor`'s gate -- LIST the server-roots subtree, keep the + /// Same enumeration as `computeHeartbeatFloor`'s gate -- walk the server-roots subtree, keep the /// `/mount` bodies -- but read-only and without any observation state: this answers "is anyone /// still entitled to write here", not "may I fence them out". const String prefix = l.serverRootsPrefix(); - String cursor; - while (true) + op.forEachListedKey(prefix, [&](const ListedKey & listed) { - const ListPage page = b.list(prefix, cursor, /*limit*/ 1000); - for (const auto & listed : page.keys) - { - static constexpr std::string_view mount_suffix = "/mount"; - if (!listed.key.ends_with(mount_suffix)) - continue; /// `/owner` and `/epoch` share the subtree; only the lease says "live". - - const String srid = listed.key.substr(prefix.size(), - listed.key.size() - prefix.size() - mount_suffix.size()); + static constexpr std::string_view mount_suffix = "/mount"; + if (!listed.key.ends_with(mount_suffix)) + return true; /// `/owner` and `/epoch` share the subtree; only the lease says "live". - const auto got = b.get(listed.key); - if (!got) - continue; /// raced away between LIST and GET -- there is no slot to be held. + const String srid = listed.key.substr(prefix.size(), + listed.key.size() - prefix.size() - mount_suffix.size()); - MountLease m; - try - { - m = decodeMountLease(got->bytes); - } - catch (...) - { - /// An undecodable lease is the WORST case for a recreation, not an ignorable one: it is - /// what a slot written by a format this build does not understand looks like, and the - /// holder of that slot is exactly the writer we must not run over. - slots.push_back(NonTerminalMountSlot{srid, fmt::format( - "mount lease could not be decoded by this build ({})", - getCurrentExceptionMessage(/*with_stacktrace=*/false))}); - continue; - } - - if (m.gc_fenced || m.min_active == std::numeric_limits::max()) - continue; /// terminal: fenced out by GC, or the holder's own graceful farewell. + const auto got = op.read(listed.key, Retry::standard()); + if (!got) + return true; /// raced away between the listing and the read -- there is no slot to be held. + MountLease m; + try + { + m = decodeMountLease(got->bytes); + } + catch (...) + { + /// An undecodable lease is the WORST case for a recreation, not an ignorable one: it is + /// what a slot written by a format this build does not understand looks like, and the + /// holder of that slot is exactly the writer we must not run over. slots.push_back(NonTerminalMountSlot{srid, fmt::format( - "held by server uuid {} (writer_epoch {}, host '{}', pid {}, lease seq {}, stamped " - "expiry {} ms) with neither a graceful farewell nor a GC fence-out", - u128ToHex(m.server_uuid), m.writer_epoch, m.hostname, m.pid, m.seq, m.expires_at_ms)}); + "mount lease could not be decoded by this build ({})", + getCurrentExceptionMessage(/*with_stacktrace=*/false))}); + return true; } - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } + if (m.gc_fenced || m.min_active_build_sequence == std::numeric_limits::max()) + return true; /// terminal: fenced out by GC, or the holder's own graceful farewell. + + slots.push_back(NonTerminalMountSlot{srid, fmt::format( + "held by server uuid {} (writer_epoch {}, host '{}', pid {}, lease seq {}, stamped " + "expiry {} ms) with neither a graceful farewell nor a GC fence-out", + u128ToHex(m.server_uuid), m.writer_epoch, m.hostname, m.pid, m.seq, m.expires_at_ms)}); + return true; + }, Retry::standard()); return slots; } -std::vector listMounts(Backend & backend, const Layout & layout, uint64_t now_ms, uint64_t skew_margin_ms) +std::vector listMounts(CasOperation & op, const Layout & layout, uint64_t now_ms, uint64_t skew_margin_ms) { std::vector out; const String prefix = layout.serverRootsPrefix(); - String cursor; - while (true) + op.forEachListedKey(prefix, [&](const ListedKey & listed) { - const ListPage page = backend.list(prefix, cursor, 1000); - for (const auto & k : page.keys) + static constexpr std::string_view suffix = "/mount"; + if (!listed.key.ends_with(suffix)) + return true; + const auto got = op.read(listed.key, Retry::standard()); + if (!got) + return true; /// raced a delete — read-only view, skip the row + MountInfo info; + /// The srid is the path segment between `serverRootsPrefix()` and the `/mount` suffix — + /// may itself contain `/` (e.g. `shard-01/replica-a`), so slice by prefix length rather + /// than `rfind('/')`, matching `computeHeartbeatFloor`'s extraction. + info.srid = listed.key.substr(prefix.size(), listed.key.size() - prefix.size() - suffix.size()); + try { - static constexpr std::string_view suffix = "/mount"; - if (!k.key.ends_with(suffix)) - continue; - const auto got = backend.get(k.key); - if (!got) - continue; /// raced a delete — read-only view, skip the row - MountInfo info; - /// The srid is the path segment between `serverRootsPrefix()` and the `/mount` suffix — - /// may itself contain `/` (e.g. `shard-01/replica-a`), so slice by prefix length rather - /// than `rfind('/')`, matching `computeHeartbeatFloor`'s extraction. - info.srid = k.key.substr(prefix.size(), k.key.size() - prefix.size() - suffix.size()); - try - { - info.lease = decodeMountLease(got->bytes); - } - catch (...) - { - info.state = "corrupt"; - out.push_back(std::move(info)); - continue; - } - if (info.lease.gc_fenced) - info.state = "fenced"; - else if (info.lease.min_active == std::numeric_limits::max()) - info.state = "terminated"; - else if (now_ms <= info.lease.expires_at_ms + skew_margin_ms) - info.state = "live"; - else - info.state = "expired"; + info.lease = decodeMountLease(got->bytes); + } + catch (...) + { + info.state = "corrupt"; out.push_back(std::move(info)); + return true; } - if (page.next_cursor.empty()) - break; - cursor = page.next_cursor; - } + if (info.lease.gc_fenced) + info.state = "fenced"; + else if (info.lease.min_active_build_sequence == std::numeric_limits::max()) + info.state = "terminated"; + else if (now_ms <= info.lease.expires_at_ms + skew_margin_ms) + info.state = "live"; + else + info.state = "expired"; + out.push_back(std::move(info)); + return true; + }, Retry::standard()); return out; } @@ -1366,7 +1238,7 @@ FenceCertificate classifyFenceCertificate(const MountLease & lease, uint64_t fen { if (lease.gc_fenced) return FenceCertificate::GcFenced; - if (lease.min_active == std::numeric_limits::max()) + if (lease.min_active_build_sequence == std::numeric_limits::max()) return FenceCertificate::CleanFarewell; if (lease.writer_epoch != fence_writer_epoch) return FenceCertificate::SupersededEpoch; @@ -1375,10 +1247,10 @@ FenceCertificate classifyFenceCertificate(const MountLease & lease, uint64_t fen } -bool isCreatorFenceTerminal(Backend & backend, const Layout & layout, const String & server_root_id, - uint64_t writer_epoch) +bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const String & server_root_id, + uint64_t writer_epoch, const Retry & policy) { - const auto got = backend.get(layout.mountKey(server_root_id)); + const auto got = op.read(layout.mountKey(server_root_id), policy); if (!got) return false; /// absence proves nothing about liveness -- see the header doc @@ -1416,29 +1288,48 @@ bool isCreatorFenceTerminal(Backend & backend, const Layout & layout, const Stri return terminal; } -MountLeaseKeeper::MountLeaseKeeper( - BackendPtr backend_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, +/// The farewell's FLOOR, not its whole budget: a departing mount is holding shutdown open, so the +/// window still wants to be short, but it can never be shorter than what the farewell's own write +/// needs to send even one attempt. `terminate` below takes the larger of this and that requirement. +/// A window below the requirement is strictly worse than a slightly longer shutdown: the write is +/// refused before it tries the wire, the slot is left holding the departing incarnation, and the next +/// start pays a full incarnation-stability observation (up to the mount lease TTL) instead of +/// reclaiming instantly off a clean farewell. +constexpr uint64_t kFarewellBudgetMs = 10'000; + +/// Slack added on top of the write's bare two-envelope reservation (see `terminate`). `fits` admits a +/// write whose reservation exactly equals the remaining window, but only at the instant it is checked; +/// with zero slack the farewell would be admitted only to immediately re-fail its own deadline check +/// once the clock advances by even one millisecond. This mirrors `lease_safety_margin_ms`'s default +/// (`CasRequestBudget.h`) -- the same order of magnitude already trusted elsewhere on this path for +/// "admission-time arithmetic needs room to actually run, not just to pass at t=0". +constexpr uint64_t kFarewellSlackMs = 2'000; + +MountLeaseRenewer::MountLeaseRenewer( + CasRequests & mount_requests_, CasRequests & open_requests_, const Layout & layout_, + const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, - std::function min_active_fn_, + std::function min_active_build_sequence_fn_, CasEventSink event_sink_, std::chrono::milliseconds lease_safety_margin_, std::function boot_ms_fn_) - : backend(std::move(backend_)) + : mount_requests(mount_requests_) + , open_requests(open_requests_) , key(layout_.mountKey(srid_)) , srid(srid_) , server_uuid(server_uuid_) , writer_epoch(writer_epoch_) , ttl(ttl_) , now_ms_fn(std::move(now_ms_fn_)) - , min_active_fn(std::move(min_active_fn_)) + , min_active_build_sequence_fn(std::move(min_active_build_sequence_fn_)) , event_sink(std::move(event_sink_)) , lease_safety_margin(lease_safety_margin_) , boot_ms_fn(boot_ms_fn_ ? std::move(boot_ms_fn_) : defaultBootMs) { } -String MountLeaseKeeper::encodeBody( - uint64_t seq_, uint64_t wall_ms, uint64_t min_active, UInt128 write_attempt_id) const +String MountLeaseRenewer::encodeBody( + uint64_t seq_, uint64_t wall_ms, uint64_t min_active_build_sequence, UInt128 write_attempt_id) const { const uint64_t ttl_ms = static_cast(ttl.count()); const uint64_t expires_at_ms = wall_ms > std::numeric_limits::max() - ttl_ms @@ -1452,33 +1343,40 @@ String MountLeaseKeeper::encodeBody( .started_at_ms = wall_ms, .seq = seq_, .expires_at_ms = expires_at_ms, - .min_active = min_active, + .min_active_build_sequence = min_active_build_sequence, .write_attempt_id = write_attempt_id, }); } -Token MountLeaseKeeper::claim(const String & body) +const Etag & MountLeaseRenewer::precondition() const +{ + if (!last_etag) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS mount-lease: key '{}' has no incarnation to name as a write precondition", key); + return *last_etag; +} + +Etag MountLeaseRenewer::claim(CasOperation & op, const String & body) { - const HeadResult head = backend->head(key); - if (!head.exists) + /// One read decides the branch AND supplies the precondition, so both the mint and the adoption + /// are two requests: a separate presence probe would only re-ask what these bytes already answer. + const std::optional got = op.read(key, Retry::standard()); + if (!got) { - const PutResult result = backend->putIfAbsent(key, body); - if (result.outcome != PutOutcome::Done) + WriteResult minted = op.create(key, body, Retry::standard()); + if (std::holds_alternative(minted)) throw Exception( ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' appeared between head and putIfAbsent", key); + "CAS mount-lease: key '{}' appeared between the read and the create", key); + const std::optional etag + = orThrow(std::move(minted), fmt::format("CAS mount-lease mint of key '{}'", key)); emitMountEvent( event_sink, CasEventType::MountClaim, srid, "mint", nullptr, - "mount slot absent -- keeper minted it directly"); - return result.token; + "mount slot absent -- renewer minted it directly"); + return *etag; } - const auto got = backend->get(key); - if (!got) - throw Exception( - ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' vanished between head and get while claiming", key); - const MountLease observed = decodeMountLease(got->bytes); if (observed.server_uuid != server_uuid) { @@ -1504,19 +1402,19 @@ Token MountLeaseKeeper::claim(const String & body) { emitMountEvent( event_sink, CasEventType::MountConflict, srid, "fenced_by_gc", &observed, - "own mount slot was fenced by GC before keeper adoption"); + "own mount slot was fenced by GC before renewer adoption"); throw MountFencedException(fmt::format( - "CAS mount-lease: key '{}' was fenced by GC before keeper adoption ({})", + "CAS mount-lease: key '{}' was fenced by GC before renewer adoption ({})", key, describeMountHolder(observed))); } - const PutResult result = backend->putOverwrite(key, body, got->token); - if (result.outcome != PutOutcome::Done) + WriteResult adopted = op.replace(key, body, got->etag, Retry::standard()); + if (const Conflict * conflict = std::get_if(&adopted)) { - const auto current = backend->get(key); - if (current) + /// The write's own resolve read is the re-read: it observed what took the key from us. + if (const Object * occupant = std::get_if(&conflict->seen)) { - const MountLease lease = decodeMountLease(current->bytes); + const MountLease lease = decodeMountLease(occupant->bytes); if (lease.server_uuid == server_uuid && lease.gc_fenced) throw MountFencedException(fmt::format( "CAS mount-lease: key '{}' was fenced by GC inside the adoption window ({})", @@ -1526,110 +1424,121 @@ Token MountLeaseKeeper::claim(const String & body) "CAS mount-lease: key '{}' changed while adopting our own mount slot ({})", key, describeMountHolder(lease)); } - throw Exception( - ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' vanished while adopting our own mount slot", key); + if (std::holds_alternative(conflict->seen)) + throw Exception( + ErrorCodes::ABORTED, + "CAS mount-lease: key '{}' vanished while adopting our own mount slot", key); } + const std::optional etag + = orThrow(std::move(adopted), fmt::format("CAS mount-lease adoption of key '{}'", key)); emitMountEvent( event_sink, CasEventType::MountClaim, srid, "adopt", &observed, "adopted our own already-live mount slot"); - return result.token; + return *etag; } -uint64_t MountLeaseKeeper::start() +uint64_t MountLeaseRenewer::start(Liveness liveness) { - if (keeper_state != MountLeaseKeeperState::New) + if (renewer_state != MountLeaseRenewerState::New) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: start is allowed only in New state for key '{}'", key); const uint64_t wall_ms = now_ms_fn(); const uint64_t attempt_start_boot_ms = boot_ms_fn(); - const String body = encodeBody(/*seq_=*/1, wall_ms, min_active_fn(), newMountWriteAttemptId()); - const Token token = claim(body); + const String body = encodeBody(/*seq_=*/1, wall_ms, min_active_build_sequence_fn(), newMountWriteAttemptId()); + /// Off the mount fence: a self-remount claims with the fence already latched lost, and a claim + /// admitted under it would be refused on every request. What makes the claim safe is that every + /// write below is conditional. + CasOperation op = open_requests.admit(std::move(liveness)); + const Etag etag = claim(op, body); seq = 1; - last_token = token; + last_etag = etag; last_committed_attempt_start_boot_ms = attempt_start_boot_ms; const uint64_t ttl_ms = static_cast(ttl.count()); confirmed_deadline_boot_ms = attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms ? std::numeric_limits::max() : attempt_start_boot_ms + ttl_ms; - keeper_state = MountLeaseKeeperState::Active; + renewer_state = MountLeaseRenewerState::Active; return attempt_start_boot_ms; } -[[noreturn]] void MountLeaseKeeper::throwRenewConflict(const CasOverwriteDiagnostics & diagnostics) const +[[noreturn]] void MountLeaseRenewer::throwRenewConflict(const Observation & seen) const { - if (!diagnostics.resolve_observation_completed) - throw Exception( - ErrorCodes::NETWORK_ERROR, - "CAS mount-lease: key '{}' conflicted but the controller has no authoritative resolve observation", - key); - if (!diagnostics.observed_bytes) + if (const Object * occupant = std::get_if(&seen)) { - emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "vanished", nullptr, - "mount slot vanished while renewing -- failing closed"); - throw Exception( - ErrorCodes::FILE_DOESNT_EXIST, - "CAS mount-lease: key '{}' vanished while renewing -- failing closed", key); - } + markMountRenewTermination(MountRenewTerminalClassification::Conflict); + const MountLease current = decodeMountLease(occupant->bytes); + if (current.server_uuid == server_uuid && current.gc_fenced) + { + emitMountEvent( + event_sink, CasEventType::MountConflict, srid, "fenced_by_gc", ¤t, + "own mount slot was fenced by GC after lease expiry"); + throw MountFencedException(fmt::format( + "CAS mount-lease: key '{}' was fenced by GC after lease expiry ({})", + key, describeMountHolder(current))); + } + if (current.server_uuid == server_uuid && current.writer_epoch == writer_epoch) + { + emitMountEvent( + event_sink, CasEventType::MountConflict, srid, "same_epoch_state_uncertain", ¤t, + "own mount slot advanced past our incarnation -- state uncertain"); + throw Exception( + ErrorCodes::ABORTED, + "CAS mount-lease: key '{}' advanced under our own (uuid, epoch); state uncertain ({} vs our seq={})", + key, describeMountHolder(current), seq); + } + if (current.server_uuid == server_uuid) + { + emitMountEvent( + event_sink, CasEventType::MountConflict, srid, "superseded", ¤t, + "mount slot is held by a newer writer epoch"); + throw Exception( + ErrorCodes::ABORTED, + "CAS mount-lease: key '{}' was superseded by a newer incarnation ({})", + key, describeMountHolder(current)); + } - const MountLease current = decodeMountLease(*diagnostics.observed_bytes); - if (current.server_uuid == server_uuid && current.gc_fenced) - { + /// This decoded authoritative observation is the exact point at which this incarnation learns + /// that a foreign successor owns the slot. Terminal teardown intentionally performs no release + /// I/O, so account the skipped farewell here, once, before the renewer enters its terminal state. + /// The renewal may be parked under `remount_mutex`; keep the increment trace-free. + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountReleaseSkippedForeignOccupant); emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "fenced_by_gc", ¤t, - "own mount slot was fenced by GC after lease expiry"); - throw MountFencedException(fmt::format( - "CAS mount-lease: key '{}' was fenced by GC after lease expiry ({})", - key, describeMountHolder(current))); - } - if (current.server_uuid == server_uuid && current.writer_epoch == writer_epoch) - { - emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "same_epoch_state_uncertain", ¤t, - "own mount slot advanced past our token -- state uncertain"); + event_sink, CasEventType::MountConflict, srid, "foreign_writer", ¤t, + "mount slot is held by a foreign server -- failing closed"); throw Exception( ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' advanced under our own (uuid, epoch); state uncertain ({} vs our seq={})", - key, describeMountHolder(current), seq); + "CAS mount-lease: key '{}' is held by a foreign server ({}) -- failing closed", + key, describeMountHolder(current)); } - if (current.server_uuid == server_uuid) + + if (std::holds_alternative(seen)) { + markMountRenewTermination(MountRenewTerminalClassification::Vanished); emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "superseded", ¤t, - "mount slot is held by a newer writer epoch"); + event_sink, CasEventType::MountConflict, srid, "vanished", nullptr, + "mount slot vanished while renewing -- failing closed"); throw Exception( - ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' was superseded by a newer incarnation ({})", - key, describeMountHolder(current)); + ErrorCodes::FILE_DOESNT_EXIST, + "CAS mount-lease: key '{}' vanished while renewing -- failing closed", key); } - /// This decoded authoritative observation is the exact point at which this incarnation learns - /// that a foreign successor owns the slot. Terminal teardown intentionally performs no release - /// I/O, so account the skipped farewell here, once, before the keeper enters its terminal state. - /// The renewal may be parked under `remount_mutex`; keep the increment trace-free. - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountReleaseSkippedForeignOccupant); - emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "foreign_writer", ¤t, - "mount slot is held by a foreign server -- failing closed"); - throw Exception( - ErrorCodes::ABORTED, - "CAS mount-lease: key '{}' is held by a foreign server ({}) -- failing closed", - key, describeMountHolder(current)); + /// The precondition was refused but nothing identifiable was read back: neither the successor nor + /// an absence is established, so the only honest verdict is that this renewal settled nothing. + markMountRenewTermination(MountRenewTerminalClassification::Unresolved); + throwCasWriteRetryLater(fmt::format( + "CAS mount-lease: key '{}' refused our precondition and the resolving read established neither " + "an occupant nor an absence", key)); } -MountRenewResult MountLeaseKeeper::terminalResult( - uint64_t attempt_start_boot_ms, - CasOverwriteDiagnostics diagnostics, - std::exception_ptr failure) +MountRenewResult MountLeaseRenewer::terminalResult(MountRenewResult result) { - if (!failure) + if (!result.failure) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: terminal renewal requires a failure"); try { - std::rethrow_exception(failure); + std::rethrow_exception(result.failure); } catch (const Exception & e) { @@ -1639,72 +1548,50 @@ MountRenewResult MountLeaseKeeper::terminalResult( catch (...) { } - if (keeper_state != MountLeaseKeeperState::Active) + if (renewer_state != MountLeaseRenewerState::Active) throw Exception( ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: terminal renewal outside Active state (observed state {})", - static_cast(keeper_state)); - keeper_state = MountLeaseKeeperState::RenewalTerminal; - return MountRenewResult{ - .outcome = MountRenewOutcome::Terminal, - .attempt_start_boot_ms = attempt_start_boot_ms, - .diagnostics = diagnostics, - .failure = std::move(failure), - }; + static_cast(renewer_state)); + renewer_state = MountLeaseRenewerState::RenewalTerminal; + result.outcome = MountRenewOutcome::Terminal; + return result; } -MountRenewResult MountLeaseKeeper::renew( - const CasRequestBudget & budget, - const MountRenewOperationEnvironment & environment) +MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & environment) +{ + return renewOn(mount_requests, environment); +} + +MountRenewResult MountLeaseRenewer::renewForRemount(const MountRenewOperationEnvironment & environment) +{ + return renewOn(open_requests, environment); +} + +MountRenewResult MountLeaseRenewer::renewOn( + CasRequests & plane, const MountRenewOperationEnvironment & environment) { const MountRenewObservabilityRegistration observability_registration = beginMountRenewObservabilityCall(); const MountRenewObservabilityCallGuard observability_guard(observability_registration); - if (keeper_state != MountLeaseKeeperState::Active) + if (renewer_state != MountLeaseRenewerState::Active) throw Exception( ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: renew is allowed only in Active state for key '{}' (observed state {})", key, - static_cast(keeper_state)); + static_cast(renewer_state)); const auto boot_clock = environment.boot_ms ? environment.boot_ms : boot_ms_fn; - const auto stop_cause = environment.stop_cause - ? environment.stop_cause - : [] { return CasOverwriteStopCause::Continue; }; - const auto wait_before_retry = environment.wait_before_retry - ? environment.wait_before_retry - : [](uint64_t) { return true; }; - const auto downstream_observe = environment.observe - ? environment.observe - : [](const CasOverwriteProgress &) {}; - uint32_t physical_attempts_sent = 0; - const auto observe = [&physical_attempts_sent, &downstream_observe](const CasOverwriteProgress & progress) - { - /// This call-stack-owned value is protocol diagnostic truth even when rich observability is - /// intentionally suppressed after deeply reentrant sinks exhaust its bounded TLS slots. - if (progress.kind == CasOverwriteProgressKind::PutStarted) - physical_attempts_sent = std::max(physical_attempts_sent, progress.attempt_no); - downstream_observe(progress); - }; + /// Sampled BEFORE the write. A refused admission is reported as "never attempted" only when this + /// node had already been asked to stop, and reading the flag afterwards could not tell that apart + /// from a flag the refusal itself set. + const bool cancelled = environment.cancelled && environment.cancelled(); const uint64_t wall_ms = now_ms_fn(); const uint64_t attempt_start_boot_ms = boot_clock(); const uint64_t next_seq = seq + 1; const UInt128 write_attempt_id = newMountWriteAttemptId(); - const String body = encodeBody(next_seq, wall_ms, min_active_fn(), write_attempt_id); - const Token expected = last_token; - - const uint64_t safety_ms = static_cast(lease_safety_margin.count()); - const uint64_t lease_retry_deadline = confirmed_deadline_boot_ms > safety_ms - ? confirmed_deadline_boot_ms - safety_ms - : 0; - const uint64_t request_deadline = attempt_start_boot_ms > std::numeric_limits::max() - budget.operation_deadline_ms - ? std::numeric_limits::max() - : attempt_start_boot_ms + budget.operation_deadline_ms; - const uint64_t absolute_deadline = std::min(lease_retry_deadline, request_deadline); - const CasOverwriteDeadlineSource deadline_source = lease_retry_deadline <= request_deadline - ? CasOverwriteDeadlineSource::ExternalLeaseSafety - : CasOverwriteDeadlineSource::RequestBudget; + const String body = encodeBody(next_seq, wall_ms, min_active_build_sequence_fn(), write_attempt_id); if (observability_registration != MountRenewObservabilityRegistration::Ignored) { @@ -1715,110 +1602,121 @@ MountRenewResult MountLeaseKeeper::renew( write_attempt_id, attempt_start_boot_ms, confirmed_deadline_boot_ms, - deadline_source, event_sink); } - CasRequestController controller(backend, budget, boot_clock); - const CasOverwriteOperationContext context{ - .absolute_deadline_ms = absolute_deadline, - .deadline_source = deadline_source, - .stop_cause = stop_cause, - .wait_before_retry = wait_before_retry, - .observe = observe, - }; + MountRenewResult result; + result.attempt_start_boot_ms = attempt_start_boot_ms; - CasOverwriteResult controlled; - controlled.diagnostics.deadline_source = deadline_source; + CasOperation op = plane.admit(environment.live); + std::optional written; try { - controlled = controller.putOverwriteControlled(key, body, expected, context); + written = op.replace(key, body, precondition(), + Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count()))); } catch (...) { - /// The controller may propagate a deterministic/non-retryable exception after `PutStarted`. - /// Preserve the physical observer's already-published truth instead of returning the default - /// zero-attempt diagnostics from the result object that was never assigned. This local is not - /// coupled to the bounded rich-event stack and therefore remains truthful at arbitrary nesting. - controlled.diagnostics.attempts_sent = std::max( - controlled.diagnostics.attempts_sent, physical_attempts_sent); - controlled.diagnostics.deadline_source = deadline_source; - if (MountRenewObservabilityContext * observation = currentMountRenewObservability()) - observation->terminal_classification = MountRenewTerminalClassification::DeterministicFailure; - return terminalResult(attempt_start_boot_ms, controlled.diagnostics, std::current_exception()); + /// The engine surfaces a deterministic local failure unchanged rather than reissuing it. + markMountRenewTermination(MountRenewTerminalClassification::DeterministicFailure); + result.failure = std::current_exception(); + return terminalResult(std::move(result)); } - if (controlled.outcome == CasOverwriteOutcome::Committed) + if (Committed * committed = std::get_if(&*written)) { seq = next_seq; - last_token = controlled.token; + last_etag = std::move(committed->etag); last_committed_attempt_start_boot_ms = attempt_start_boot_ms; const uint64_t ttl_ms = static_cast(ttl.count()); confirmed_deadline_boot_ms = attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms ? std::numeric_limits::max() : attempt_start_boot_ms + ttl_ms; - return MountRenewResult{ - .outcome = MountRenewOutcome::Committed, - .attempt_start_boot_ms = attempt_start_boot_ms, - .diagnostics = controlled.diagnostics, - .failure = nullptr, - }; + result.outcome = MountRenewOutcome::Committed; + result.attempts_sent = committed->attempts_sent; + result.resolved_by_read = committed->resolved_by_read; + result.sent_any = committed->attempts_sent != 0; + return result; } - if (controlled.outcome == CasOverwriteOutcome::Conflict) + if (const Conflict * conflict = std::get_if(&*written)) { - if (MountRenewObservabilityContext * observation = currentMountRenewObservability()) - observation->terminal_classification = MountRenewTerminalClassification::Conflict; + result.sent_any = true; + result.attempts_sent = conflict->attempts_sent; try { - throwRenewConflict(controlled.diagnostics); + throwRenewConflict(conflict->seen); } catch (...) { - return terminalResult(attempt_start_boot_ms, controlled.diagnostics, std::current_exception()); + result.failure = std::current_exception(); } + return terminalResult(std::move(result)); } - if (controlled.diagnostics.attempts_sent == 0 - && controlled.diagnostics.stop_cause == CasOverwriteStopCause::Cancelled) + if (const Refused * refused = std::get_if(&*written)) { - return MountRenewResult{ - .outcome = MountRenewOutcome::NotAttempted, - .attempt_start_boot_ms = attempt_start_boot_ms, - .diagnostics = controlled.diagnostics, - .failure = nullptr, - }; + result.sent_any = true; + result.attempts_sent = refused->attempts_sent; + markMountRenewTermination(MountRenewTerminalClassification::DeterministicFailure); + result.failure = std::make_exception_ptr(Exception( + refused->store_error, + "CAS mount-lease: the store refused the renewal of key '{}': {}", key, refused->message)); + return terminalResult(std::move(result)); } - /// Preserve the typed vanished-slot outcome only when the controller itself completed an exact - /// resolving read. Never start diagnostic backend I/O after its terminal deadline/cancel gate. - if (controlled.diagnostics.resolve_observation_completed - && !controlled.diagnostics.observed_bytes) + if (const GaveUp * gave_up = std::get_if(&*written)) { - if (MountRenewObservabilityContext * observation = currentMountRenewObservability()) - observation->terminal_classification = MountRenewTerminalClassification::Vanished; - emitMountEvent( - event_sink, CasEventType::MountConflict, srid, "vanished", nullptr, - "mount slot vanished while renewing -- failing closed"); - return terminalResult( - attempt_start_boot_ms, - controlled.diagnostics, - std::make_exception_ptr(Exception( - ErrorCodes::FILE_DOESNT_EXIST, - "CAS mount-lease: key '{}' vanished while renewing -- failing closed", - key))); + result.sent_any = gave_up->sent_any; + result.attempts_sent = gave_up->attempts_sent; + if (gave_up->why == GaveUp::Why::Deadline) + result.deadline_source = gave_up->deadline_source; + + /// Nothing was sent and the node was already stopping: the lease is exactly as it was, so this + /// is a renewal that never ran, not one that lost its authority. + if (gave_up->why == GaveUp::Why::FenceLost && !gave_up->sent_any && cancelled) + { + markMountRenewTermination(MountRenewTerminalClassification::Cancelled); + result.outcome = MountRenewOutcome::NotAttempted; + return result; + } + + MountRenewTerminalClassification classification = MountRenewTerminalClassification::Unresolved; + switch (gave_up->why) + { + case GaveUp::Why::FenceLost: + classification = cancelled + ? MountRenewTerminalClassification::Cancelled + : MountRenewTerminalClassification::FenceOrLifecycleLost; + break; + case GaveUp::Why::Deadline: + classification = gave_up->deadline_source == GaveUp::Source::Lease + ? MountRenewTerminalClassification::ExternalLeaseDeadline + : MountRenewTerminalClassification::RequestDeadline; + break; + case GaveUp::Why::Unresolved: + classification = MountRenewTerminalClassification::Unresolved; + break; + } + markMountRenewTermination(classification); + result.failure = makeCasWriteRetryLaterExceptionPtr(fmt::format( + "CAS mount-lease renewal for key '{}' did not retain the lease ({}, {} attempt sent, last " + "observation: {})", + key, + terminalClassificationName(classification), + gave_up->sent_any ? "at least one" : "no", + detail::renderObservation(gave_up->last_seen))); + return terminalResult(std::move(result)); } - const String reason = fmt::format( - "CAS mount-lease renewal for key '{}' is unresolved: {}", - key, describeUnresolvedReason(controlled.diagnostics.unresolved_reason)); - return terminalResult( - attempt_start_boot_ms, - controlled.diagnostics, - makeCasWriteRetryLaterExceptionPtr(reason)); + /// The remaining alternative is `Declined`, which only a decide returning nothing produces; a + /// renewal always has bytes to write. + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS mount-lease: the renewal of key '{}' was declined, which a replace cannot report", key); } -void MountLeaseKeeper::terminate() +void MountLeaseRenewer::terminate(CasOperation & op) { const uint64_t wall_ms = now_ms_fn(); const String body = encodeMountLease(MountLease{ @@ -1829,15 +1727,53 @@ void MountLeaseKeeper::terminate() .started_at_ms = wall_ms, .seq = seq + 1, .expires_at_ms = wall_ms, - .min_active = std::numeric_limits::max(), + .min_active_build_sequence = std::numeric_limits::max(), .write_attempt_id = newMountWriteAttemptId(), }); - const PutResult result = backend->putOverwrite(key, body, last_token); - if (result.outcome != PutOutcome::Done) + /// The farewell is admitted on `open_requests` (see `release`, which calls this via `open_requests.admit()`), + /// so its own reservation -- attempt plus the read that settles it, `reservedFor(0, 2)` in + /// `CasOperation::writeLoop` -- is exactly `2 * open_requests.attemptReservationMs()`. A window + /// below that value refuses the write before its first attempt, deterministically, on every call: + /// `kFarewellBudgetMs` alone predates the attempt-envelope reservation and can no longer be trusted + /// to admit it. Saturating, like every other deadline computation on this path (see the + /// `expires_at_ms`/`confirmed_deadline_boot_ms` arithmetic above): an operator-configured envelope + /// is not bounds-checked against this doubling, and wrapping past `UINT64_MAX` would turn a too-long + /// window into a too-SHORT one -- the exact failure mode this fix exists to remove. + const uint64_t reservation_ms = open_requests.attemptReservationMs(); + const uint64_t doubled_reservation_ms = reservation_ms > std::numeric_limits::max() / 2 + ? std::numeric_limits::max() + : reservation_ms * 2; + const uint64_t two_envelope_reservation_plus_slack_ms = doubled_reservation_ms > std::numeric_limits::max() - kFarewellSlackMs + ? std::numeric_limits::max() + : doubled_reservation_ms + kFarewellSlackMs; + const uint64_t farewell_window_ms = std::max(kFarewellBudgetMs, two_envelope_reservation_plus_slack_ms); + /// The derived window alone is not enough: mount-control activity must also never run past the + /// point this node's own fence may already be gone (the same rule `renew` enforces via + /// `Retry::untilLeaseSafe` above). The precondition on this write already stops it from clobbering + /// a successor if it DOES land late, but a shutdown holding the process open to retry a write past + /// its own lease-safe deadline serves no one -- the successor's own reclaim does not wait for it. + /// `confirmed_deadline_boot_ms` is set at `start()` and kept current by every successful `renew`, + /// so it is valid here whenever `terminate` runs (only reachable from `release`, which requires + /// `Active`, which `start` alone establishes). + WriteResult written = op.replace(key, body, precondition(), + Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count()), farewell_window_ms)); + + if (Committed * committed = std::get_if(&written)) { - if (const auto got = backend->get(key)) + seq += 1; + last_etag = std::move(committed->etag); + emitMountEvent( + event_sink, CasEventType::MountRelease, srid, "farewell", nullptr, + "graceful release -- lease stamped already-expired and watermark retired"); + return; + } + + if (const Conflict * conflict = std::get_if(&written)) + { + /// The write's own resolve read is the re-read this branch used to issue for itself. + if (const Object * occupant = std::get_if(&conflict->seen)) { - const MountLease current = decodeMountLease(got->bytes); + const MountLease current = decodeMountLease(occupant->bytes); if (current.gc_fenced) return; ProfileEvents::increment(ProfileEvents::CASMountExclusivityViolation); @@ -1846,22 +1782,23 @@ void MountLeaseKeeper::terminate() "CAS mount-lease: release of key '{}' found a foreign incarnation ({}) and left it untouched", key, describeMountHolder(current)); } - return; + if (std::holds_alternative(conflict->seen)) + return; /// the slot is already gone; there is nothing left to hand back } - seq += 1; - last_token = result.token; - emitMountEvent( - event_sink, CasEventType::MountRelease, srid, "farewell", nullptr, - "graceful release -- lease stamped already-expired and watermark retired"); + orThrow(std::move(written), fmt::format("CAS mount-lease release of key '{}'", key)); } -void MountLeaseKeeper::release() +void MountLeaseRenewer::release() { - if (keeper_state != MountLeaseKeeperState::Active) + if (renewer_state != MountLeaseRenewerState::Active) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: release is allowed only in Active state for key '{}'", key); - keeper_state = MountLeaseKeeperState::Released; - terminate(); + renewer_state = MountLeaseRenewerState::Released; + /// Off the mount fence, for the same reason the claim is: a departing mount whose lease has already + /// run down still has to hand the slot back, and refusing the write there would leave the slot + /// looking live until GC fences it out. + CasOperation op = open_requests.admit(); + terminate(op); } void sweepOwnMountStaging(IObjectStorage & object_storage, const String & mount_staging_prefix) noexcept diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 5f19862e2348..1c1fd8928f97 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -1,6 +1,5 @@ #pragma once -#include -#include +#include #include #include #include @@ -32,7 +31,7 @@ namespace ErrorCodes namespace DB::Cas { -enum class MountLeaseKeeperState : uint8_t +enum class MountLeaseRenewerState : uint8_t { New, Active, @@ -51,16 +50,27 @@ struct MountRenewResult { MountRenewOutcome outcome = MountRenewOutcome::Terminal; uint64_t attempt_start_boot_ms = 0; - CasOverwriteDiagnostics diagnostics; + /// Physical attempts this renewal sent, for every terminal ending the engine can count: a commit, + /// a give-up, a conflict, or a store refusal. + uint32_t attempts_sent = 0; + bool resolved_by_read = false; + bool sent_any = false; + /// Which bound ended a renewal that ran out of time; unset for every other ending, including a + /// committed one -- no deadline ended it, so naming one would invent a fact. + std::optional deadline_source; std::exception_ptr failure; }; struct MountRenewOperationEnvironment { std::function boot_ms; - std::function stop_cause; - std::function wait_before_retry; - std::function observe; + /// Facts the mount fence cannot see (park requested, pool no longer live, shutdown). FALSE ends + /// the renewal exactly as a lost fence does; the engine does not need to know which refused. + std::function live; + /// Sampled ONCE before the write. A renewal refused before its first attempt is reported as + /// `NotAttempted` rather than terminal only when this node had already been asked to stop -- + /// sampling it afterwards would read a flag that the refusal itself may have set. + std::function cancelled; }; /// Validate a `server_root_id` — the explicit, configured identity of the content-addressed layout @@ -110,7 +120,6 @@ inline void validateServerRootId(const String & id) /// below can use `OwnerObject`, `ServerEpoch`, `MountLease`, and their `encode`/`decode` functions /// without duplicating the wire-format implementation. -class Backend; class Layout; /// Mount-safety claim logic. These are the identity and epoch-allocation steps a server @@ -121,7 +130,7 @@ class Layout; /// exact-component probes find neither `cas/manifests//` nor `roots//` work. Opaque /// `cas/ns/` debris alone does not identify a logical owner. bool serverRootSubtreeEmpty( - Backend & b, const Layout & l, const String & srid, const RefCatalog & catalog_observation); + CasOperation & op, const Layout & l, const String & srid, const RefCatalog & catalog_observation); /// Supplied by the pool layer so the low-level server-root protocol always observes the mandatory /// catalog. Every absent-control retry obtains a fresh, successfully decoded observation. @@ -131,18 +140,18 @@ using ObserveRefCatalog = std::function; /// a plain GET+decode. nullopt = anchor absent. Pool-member decommission uses this to read the /// victim UUID before mounting writable; `claimOwnerOrThrow` below is the identity-claiming /// counterpart used by normal opens and reuses this GET+decode path. -std::optional readOwnerUuid(Backend & b, const Layout & l, const String & server_root_id); +std::optional readOwnerUuid(CasOperation & op, const Layout & l, const String & server_root_id); /// Claim (or validate) the sticky owner anchor that binds `srid` to a server UUID (identity). /// - owner present, equal `our_uuid`, and not tombstoned → ok (return); /// - owner present and tombstoned → throw `CORRUPTED_DATA` (explicitly retired — fail closed); /// - owner present, different → throw `CORRUPTED_DATA` (foreign owner — fail closed); -/// - owner absent AND the subtree is provably empty → `putIfAbsent` the owner (claim); +/// - owner absent AND the subtree is provably empty → `create` the owner (claim); /// - owner absent BUT the subtree is non-empty → throw `CORRUPTED_DATA` (identity lost over /// existing data — never silently re-claim). /// The owner object is never deleted and never reassigned to a different UUID. void claimOwnerOrThrow( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, const ObserveRefCatalog & observe_catalog); /// Which mint policy governs `allocateWriterEpoch`'s absent-epoch branch (see below). @@ -169,43 +178,49 @@ enum class EpochMintPolicy : uint8_t /// distinct from the survivor's by construction (`now_ms` is required, nonzero, here); /// - anything else (`ContainerAbsent`/`AccessDenied`/`Indeterminate`) → throw /// `CORRUPTED_DATA` (absence was never proven; fail closed); -/// - otherwise read `next = current.next_writer_epoch`, `casPut` `{next + 1}` against the -/// observed token, retry on `Conflict` (bounded), and return `next`. -uint64_t allocateWriterEpoch(Backend & b, const Layout & l, const String & srid, +/// - otherwise read `next = current.next_writer_epoch`, write `{next + 1}` against the observed +/// incarnation, re-deciding on conflict, and return `next`. +uint64_t allocateWriterEpoch(CasOperation & op, const Layout & l, const String & srid, EpochMintPolicy policy, uint64_t now_ms, const ObserveRefCatalog & observe_catalog); -/// Which certificate of death justified a same-uuid, different-epoch mount reclaim. `None` when no +/// Which certificate of death, or the operator's explicit unsafe authorization, justified a +/// same-uuid, different-epoch mount reclaim. `None` when no /// reclaim of that kind happened (a fresh claim, a /// same-epoch refresh, `LiveDoubleStart`, `ForeignOwner`, `FencedSelf`). enum class MountPriorState { None, - Clean, /// the predecessor's own graceful farewell (`min_active == UINT64_MAX`) + Clean, /// the predecessor's own graceful farewell (`min_active_build_sequence == UINT64_MAX`) Fenced, /// the GC leader's own (already threshold-gated) fence-out (`gc_fenced`) - UncleanObserved, /// OUR observation watched the write-token hold stable for the full threshold + UncleanObserved, /// OUR observation watched the incarnation hold stable for the full threshold + UncleanUnsafe, /// the operator's explicit `cas_unsafe_remount_no_delay` authorization carried the slot's exact token }; /// Startup decision for the mount lease (`gc/server-roots//mount`), run AFTER the owner gate /// (so `our_uuid` is the established owner). The lease is LIVENESS, not identity — the owner object /// already settled who may write; the lease settles whether a live incarnation currently holds the /// slot. Decision over `get(mountKey)`: -/// - absent → write our body via `putIfAbsent` → `Claimed`; +/// - absent → write our body via `create` → `Claimed`; /// - same `server_uuid` AND same `writer_epoch` as (our_uuid, our_epoch) → it is OUR OWN claim -/// (a replay / the keeper adopting it): +/// (a replay / the renewer adopting it): /// - `gc_fenced` → terminal for THIS (uuid, epoch) — a fence costs an epoch, so refreshing it /// in place would reactivate a fenced incarnation → `FencedSelf` (no write); -/// - otherwise → refresh (`putOverwrite` to bump seq + fresh `expires_at_ms`) → `Claimed`; -/// - same `server_uuid`, DIFFERENT `writer_epoch` → reclaimed ONLY on a certificate of death that +/// - otherwise → refresh (`replace` to bump seq + fresh `expires_at_ms`) → `Claimed`; +/// - same `server_uuid`, DIFFERENT `writer_epoch` → reclaimed ONLY on a certificate of death, or the +/// operator's explicit unsafe authorization, that /// needs no fresh wall-clock trust (see /// `claimMountAwaitingExpiry` below for how a plain "looks expired" reading is turned into one): /// - `gc_fenced` (the GC leader already, itself, threshold-gated this incarnation dead; a fence -/// costs an epoch, so its keeper can never renew again) → reclaim, `prior = Fenced`; -/// - the clean marker (`min_active == UINT64_MAX`, the predecessor's own graceful farewell) → +/// costs an epoch, so its renewer can never renew again) → reclaim, `prior = Fenced`; +/// - the clean marker (`min_active_build_sequence == UINT64_MAX`, the predecessor's own graceful farewell) → /// reclaim, `prior = Clean`; -/// - `proven_dead_token` matches the CURRENTLY OBSERVED token (the caller itself watched this -/// exact token hold stable for the full observation threshold) → reclaim, `prior = +/// - `proven_dead_incarnation` matches the CURRENTLY OBSERVED incarnation (the caller itself +/// watched that exact incarnation hold stable for the full observation threshold) → reclaim, `prior = /// UncleanObserved`; +/// - `unsafe_reclaim_authorization` matches the CURRENTLY OBSERVED incarnation (the operator's +/// explicit `cas_unsafe_remount_no_delay` authorization, carrying the exact token read, with NO +/// observation at all) → reclaim, `prior = UncleanUnsafe`; /// - none of the above → `LiveDoubleStart` (do NOT write). In particular `expires_at_ms <= /// now_ms` ALONE is never sufficient — comparing a predecessor's stamp against OUR wall clock /// is unsafe because a clock-skewed or merely late-observing @@ -225,17 +240,24 @@ struct MountClaimResult FencedSelf, }; Kind kind = ForeignOwner; - MountLease body; - /// Which certificate of death justified a same-uuid, different-epoch `Claimed` reclaim (`None` for + /// The lease this result is ABOUT. For `Claimed` it is the body this server installed; for + /// `ForeignOwner`, `FencedSelf` and `LiveDoubleStart` it is the body that was observed at the key, + /// which is what `mountDoubleStartMessage` renders as the existing mount. EMPTY exactly when a + /// raced write's conflict settled to no observation at all -- nobody was seen, so no operator + /// message may name a holder, and the lease this server merely PROPOSED is not one. An optional + /// rather than the proposal, because a caller cannot check a convention it cannot see. + std::optional body; + /// Which certificate of death, or the operator's explicit unsafe authorization, justified a + /// same-uuid, different-epoch `Claimed` reclaim (`None` for /// every other `Kind`, and for the absent-slot / same-epoch-refresh `Claimed` cases). MountPriorState prior = MountPriorState::None; - /// The backend token of the body this result observed, for + /// The incarnation of the body this result observed, for /// `LiveDoubleStart` only (a fresh `Claimed`/`FencedSelf`/`ForeignOwner` write/observe has no - /// separate "prior body's token to remember" use). `claimMountAwaitingExpiry`'s observation loop - /// used to re-GET the mount key itself just to recover this token that `claimMount` had already - /// read one line earlier and thrown away -- one wasted GET per iteration. Empty for every other + /// separate "prior body's incarnation to remember" use). `claimMountAwaitingExpiry`'s observation + /// loop would otherwise re-read the mount key just to recover what `claimMount` had already read + /// one line earlier and thrown away -- one wasted read per iteration. Empty for every other /// `Kind` (nothing to compare against). - std::optional token; + std::optional etag; }; /// Thrown when a mount operation observes that OUR OWN (uuid, epoch) slot was `gc_fenced` by the GC @@ -251,20 +273,28 @@ class MountFencedException : public DB::Exception : DB::Exception(msg, DB::ErrorCodes::ABORTED) {} }; -/// `proven_dead_token`: the write-token of a same-uuid, different-epoch lease that the CALLER already -/// proved dead by observation (see `claimMountAwaitingExpiry`) — matching it against the CURRENTLY -/// observed token is the ONLY way (besides `gc_fenced` / the clean marker) a same-uuid different-epoch -/// lease is ever reclaimed. Absent (`{}`, the default) for a bare claim attempt with no such proof. +/// `proven_dead_incarnation`: the incarnation of a same-uuid, different-epoch lease that the CALLER +/// already proved dead by observation (see `claimMountAwaitingExpiry`) — matching it against the +/// CURRENTLY observed incarnation is the ONLY way (besides `gc_fenced` / the clean marker) a +/// same-uuid different-epoch lease is ever reclaimed. Absent (`{}`, the default) for a bare claim +/// attempt with no such proof. +/// `unsafe_reclaim_authorization`: the exact token the operator's `cas_unsafe_remount_no_delay` read +/// off a same-uuid, different-epoch slot before authorizing this reclaim, with no observation at all. +/// Never reused from `proven_dead_incarnation`: that one says the token was OBSERVED dead, this one +/// says the operator accepted the risk. Absent (`{}`, the default) when the knob is off. MountClaimResult claimMount( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, - uint64_t now_ms, uint64_t ttl_ms, const std::optional & proven_dead_token = {}, - const CasEventSink & sink = {}); + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, + uint64_t now_ms, uint64_t ttl_ms, const std::optional & proven_dead_incarnation = {}, + const CasEventSink & sink = {}, const std::optional & unsafe_reclaim_authorization = {}); /// Format the operator-actionable startup error shown when the mount lease is held by a genuinely /// live second server (the same `server_root_id` is mounted twice). Produced only AFTER this server /// has already waited for the lease to lapse (see `claimMountAwaitingExpiry`) and it did not — so the /// remediation is about a live twin, not about waiting. -String mountDoubleStartMessage(const String & srid, const MountLease & existing); +/// `existing` empty renders the identity block as "could not be observed": a raced write whose +/// conflict saw nothing has no holder to name, and naming the proposer would point an operator at this +/// very process. +String mountDoubleStartMessage(const String & srid, const std::optional & existing); /// Observation-based mount claim for restart recovery. /// Wraps `claimMount` in a loop: @@ -272,15 +302,15 @@ String mountDoubleStartMessage(const String & srid, const MountLease & existing) /// `Clean`), `ForeignOwner`, or `FencedSelf`; /// - a `LiveDoubleStart` from OUR OWN uuid (a stale lease from a prior incarnation of this server, /// OR a genuinely live twin — the two are indistinguishable from a bare read) is resolved by -/// WATCHING the lease's write-token on OUR OWN clock (`mono_ms_fn`), NEVER by comparing the -/// lease's stamped `expires_at_ms` against any clock: once the observed token has held stable for +/// WATCHING the lease's incarnation on OUR OWN clock (`mono_ms_fn`), NEVER by comparing the +/// lease's stamped `expires_at_ms` against any clock: once the observed incarnation has held stable for /// the full rate-bound threshold (`ttl_ms + ttl_ms / 20 + poll_interval_ms` — the lease TTL, a 5% /// clock-drift allowance, and one poll interval of discreteness. This rate bound ensures that a /// holder which last renewed before the observation began can no longer be within its lease. -/// that token is handed to `claimMount` as `proven_dead_token`, which then reclaims token-guarded -/// (`prior = UncleanObserved`). If the token changes DURING the wait (the holder renewed, or a -/// genuine twin is alive) the observation RESTARTS from the new token; bounded to a handful of -/// restarts before giving up and returning the last `LiveDoubleStart` (a holder whose token keeps +/// that incarnation is handed to `claimMount` as `proven_dead_incarnation`, which then reclaims +/// incarnation-guarded (`prior = UncleanObserved`). If the incarnation changes DURING the wait (the +/// holder renewed, or a genuine twin is alive) the observation RESTARTS from the new one; bounded to +/// a handful of restarts before giving up and returning the last `LiveDoubleStart` (a holder whose incarnation keeps /// changing across that many restarts is alive, not dead). /// `now_ms_fn` is WALL clock, used only for stamping the body we (may) write / diagnostics — it never /// participates in the reclaim decision. `mono_ms_fn` is the OBSERVATION clock: monotonic on this @@ -289,14 +319,24 @@ String mountDoubleStartMessage(const String & srid, const MountLease & existing) /// tests drive fake clocks with no real sleeping. `on_wait_start` (default no-op) fires once per /// observation-window start (including restarts), with the currently-observed lease and the /// threshold, for an operator-visible startup log. -/// All callers use this shared formula so a future adjustment cannot silently leave the startup -/// observation path and either GC heartbeat path with different thresholds. `cadence_ms` is the -/// caller's own poll or heartbeat interval; the additional interval accounts for observation -/// discreteness, while `ttl_ms / 20` allows for a five-percent clock-rate difference. +/// All callers share this one formula, not duplicated arithmetic, so a future fix or an added term +/// applies everywhere at once -- but the callers deliberately pass DIFFERENT `cadence_ms` values, so +/// the resulting thresholds are close, not identical: the startup reopen wait below passes half the +/// renewal period (`max(1, floor(mount_renew_period_ms / 2))`), while GC's heartbeat fence-out +/// (`Gc/CasGc.cpp`) passes the full renewal period. `cadence_ms` is the caller's own poll or +/// heartbeat interval; the additional interval accounts for observation discreteness, while +/// `ttl_ms / 20` allows for a five-percent clock-rate difference. uint64_t mountObservationThresholdMs(uint64_t ttl_ms, uint64_t cadence_ms); +/// Bounded number of observation restarts `claimMountAwaitingExpiry` allows before giving up on a +/// same-uuid slot whose write-token keeps changing (see the function comment above): each restart +/// means the token changed DURING the observation window, i.e. something is actively renewing it. +/// Exposed here (rather than kept file-local in the `.cpp`) so tests can size a fixture's poll count +/// against the exact bound the engine enforces. +constexpr size_t kMaxObservationRestarts = 3; + MountClaimResult claimMountAwaitingExpiry( - Backend & b, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, + CasOperation & op, const Layout & l, const String & srid, UInt128 our_uuid, uint64_t our_epoch, const std::function & now_ms_fn, const std::function & mono_ms_fn, uint64_t ttl_ms, uint64_t poll_interval_ms, @@ -304,44 +344,44 @@ MountClaimResult claimMountAwaitingExpiry( const std::function & on_wait_start = {}, const CasEventSink & sink = {}); -/// One `server_root_id`'s cross-round token-stability observation, +/// One `server_root_id`'s cross-round incarnation-stability observation, /// owned by the GC leader instance (`Cas::Gc::mount_obs`) and threaded through consecutive /// `computeHeartbeatFloor` calls — one GC round is one observation tick. Mirrors /// `claimMountAwaitingExpiry`'s observation loop, but at heartbeat-gate granularity rather than a /// tight poll loop. -struct MountTokenObservation +struct MountIncarnationObservation { - Token token; + Etag etag; uint64_t first_seen_mono_ms = 0; }; /// Keyed by `server_root_id`. In-memory only: a fresh leader (after a steal, or a process restart) /// starts with an empty map, which only delays fencing an already-dead mount by one extra round while /// it (re)establishes the observation — safe (never fences early), never unsafe. -using MountObservationMap = std::map; +using MountObservationMap = std::map; /// GC heartbeat gate (GC round protocol step 1). Run by the GC leader at the top of a round: LIST /// `gc/server-roots/` (O(servers), single-digit counts), GET each mount body, and classify + fence out /// dead mounts (liveness only — graduation itself paces on GC rounds, not on heartbeat acks). /// Classification per body: /// - `gc_fenced` already set → excluded (`already_fenced`); a fenced mount is terminal, no PUT; -/// - terminated (`min_active == UINT64_MAX`, the farewell sentinel stamped by -/// `MountLeaseKeeper::terminate`) → excluded (`terminated`). `expires_at_ms` alone cannot +/// - terminated (`min_active_build_sequence == UINT64_MAX`, the farewell sentinel stamped by +/// `MountLeaseRenewer::terminate`) → excluded (`terminated`). `expires_at_ms` alone cannot /// distinguish a graceful farewell from an unclean stop, so the sentinel — not the timestamps — is the /// terminated marker; /// - otherwise, observation-based liveness (the same /// principle `claimMountAwaitingExpiry` uses for a mount's OWN reopen, applied here to the GC's -/// fence-out): `obs` remembers, per srid, the write-token last seen and the leader's OWN -/// monotonic clock reading (`mono_now_ms`) at the moment it first saw that token. A body whose -/// CURRENT token differs from (or is absent from) `obs` is (re)started fresh — counted `live`, +/// fence-out): `obs` remembers, per srid, the incarnation last seen and the leader's OWN +/// monotonic clock reading (`mono_now_ms`) at the moment it first saw it. A body whose +/// CURRENT incarnation differs from (or is absent from) `obs` is (re)started fresh — counted `live`, /// never fenced this call, regardless of what its stamped `expires_at_ms` claims (a bare /// wall-clock stamp is never trusted — see `claimMount`'s "certificate of death" doc). Only once -/// the SAME token has held for `>= stable_threshold_ms` OF THE LEADER'S OWN CLOCK does the body +/// the SAME incarnation has held for `>= stable_threshold_ms` OF THE LEADER'S OWN CLOCK does the body /// become FENCE-eligible; -/// - FENCE-eligible → one token-guarded `putOverwrite` preserving the WHOLE body, setting -/// `gc_fenced = true` and `seq + 1`. On `Done` → excluded (`fenced_now`); on `PreconditionFailed` -/// (the holder renewed concurrently — a live token change) → re-GET and reclassify from the top -/// (bounded retries; the reclassify sees the new token and restarts the observation, counting it +/// - FENCE-eligible → one incarnation-guarded `replace` preserving the WHOLE body, setting +/// `gc_fenced = true` and `seq + 1`. On `Committed` → excluded (`fenced_now`); on `Conflict` +/// (the holder renewed concurrently — a live incarnation change) → re-read and reclassify from the top +/// (bounded retries; the reclassify sees the new incarnation and restarts the observation, counting it /// `live` — conservative, never exclude a heartbeat without a landed fence-out). /// /// `now_ms` is WALL clock, used only for the audit/diagnostic log line — it never participates in the @@ -352,7 +392,7 @@ using MountObservationMap = std::map; /// fencing one round, never fences early). /// /// The fence-out is BOTH safety and liveness. Safety: a sleeper's later renewal permanently fails -/// (its `putOverwrite` now mismatches the fenced token → `tripMountLost`), so it can never re-arm +/// (its `replace` now mismatches the fenced incarnation → `tripMountLost`), so it can never re-arm /// without a fresh `open`. Liveness: a dead server's stale mount slot must not linger forever. /// Preserving the body keeps restart recovery intact: a same-uuid reopen reads the current body and /// reclaims through the normal expired-our-uuid branch. @@ -366,7 +406,7 @@ struct HeartbeatFloor std::vector fenced_srids; }; -HeartbeatFloor computeHeartbeatFloor(Backend & b, const Layout & l, uint64_t now_ms, +HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64_t now_ms, uint64_t mono_now_ms, uint64_t stable_threshold_ms, MountObservationMap & obs); @@ -381,8 +421,8 @@ struct NonTerminalMountSlot /// Read-only scan of every mount slot under the pool prefix, answering ONE question: is some writer /// still entitled to this prefix? A slot counts as terminal on exactly the two clock-free certificates /// the mount protocol already recognises (`computeHeartbeatFloor`'s own classification): `gc_fenced` -/// (the GC leader fenced that incarnation out, and a fence costs an epoch, so its keeper can never -/// renew again) and `min_active == UINT64_MAX` (the holder's own graceful farewell). Everything else is +/// (the GC leader fenced that incarnation out, and a fence costs an epoch, so its renewer can never +/// renew again) and `min_active_build_sequence == UINT64_MAX` (the holder's own graceful farewell). Everything else is /// reported, INCLUDING a body this build cannot decode -- an unreadable lease of some other format /// generation is precisely the case that must block, not the one to wave through. /// @@ -394,11 +434,11 @@ struct NonTerminalMountSlot /// The caller is pool RECREATION (`Pool::open`'s bootstrap over a prefix with no authoritative /// `_pool_meta`): minting a fresh pool identity while a live writer still holds a slot would leave that /// writer appending its old-format transactions into the new pool. Writes nothing. -std::vector probeNonTerminalMountSlots(Backend & b, const Layout & l); +std::vector probeNonTerminalMountSlots(CasOperation & op, const Layout & l); /// A read-only snapshot of one server's mount slot, for introspection (`system.cas_mounts`). /// state: `live` (lease within TTL+skew), `expired` (lease ran out; the next GC round's heartbeat floor -/// will fence it), `terminated` (clean farewell: `min_active == UINT64_MAX`), `fenced` (`gc_fenced`), +/// will fence it), `terminated` (clean farewell: `min_active_build_sequence == UINT64_MAX`), `fenced` (`gc_fenced`), /// `corrupt` (body failed to decode — surfaced as a row, never an exception). struct MountInfo { @@ -409,7 +449,7 @@ struct MountInfo /// Enumerate every mount slot under `gc/server-roots/`, decoded and classified — the read-only sibling /// of `computeHeartbeatFloor`: ZERO writes (no fence-out), per-row fail-open. One LIST + one GET per slot. -std::vector listMounts(Backend & backend, const Layout & layout, uint64_t now_ms, uint64_t skew_margin_ms); +std::vector listMounts(CasOperation & op, const Layout & layout, uint64_t now_ms, uint64_t skew_margin_ms); /// Whether the mounted writer identified by `(server_root_id, writer_epoch)` (the two fields of a /// `CatalogEntry::creator` / `CreatorFence`, `ref_catalog`'s spec INV-3 §3, that this predicate actually @@ -435,8 +475,8 @@ std::vector listMounts(Backend & backend, const Layout & layout, uint /// two clock-free certificates `probeNonTerminalMountSlots`/`computeHeartbeatFloor` already use for the /// identical question at pool-prefix and GC-heartbeat granularity — /// - `gc_fenced` (the GC leader already fenced this incarnation; a fence costs an epoch, so its -/// keeper can never renew again), -/// - the clean-farewell sentinel `min_active == UINT64_MAX`, +/// renewer can never renew again), +/// - the clean-farewell sentinel `min_active_build_sequence == UINT64_MAX`, /// PLUS one more certificate available here that neither of those needs: a DIFFERENT `writer_epoch` /// currently live at that slot proves `writer_epoch`'s specific incarnation is superseded regardless of /// its OWN certificate — `allocateWriterEpoch`/`claimMount` are why an epoch, once superseded, is never @@ -447,7 +487,7 @@ std::vector listMounts(Backend & backend, const Layout & layout, uint /// must fail the BUILD (a missing `-Wswitch` case), never silently read as terminal. /// /// Deliberately conservative on the two cases that are NOT proof of death: an ABSENT mount slot -/// (`Backend::get` returning `nullopt` answers nothing about liveness — it is not proof either way) +/// (a read finding the key absent answers nothing about liveness — it is not proof either way) /// and an UNDECODABLE body (an unreadable lease of some other format generation is precisely the case /// that must block, not the one to wave through, mirroring `probeNonTerminalMountSlots`'s own stated /// discipline for that case) both return `false` — refuse reconciliation rather than guess. @@ -461,59 +501,78 @@ std::vector listMounts(Backend & backend, const Layout & layout, uint /// root's OTHER activity, about whether `server_root_id` will ever mount again, or about /// anything beyond this one slot's current body at the instant of this GET. A caller that needs a /// stronger, race-free guarantee (e.g. "and it will never come back") must build that from a WIDER -/// observation, the way `claimMountAwaitingExpiry`'s token-stability window does for its own decision -- +/// observation, the way `claimMountAwaitingExpiry`'s stability window does for its own decision -- /// this function performs no such window and answers from one point-in-time read alone. Answering /// "unknown" (`false`, refuse) is the fail-closed choice on every path already listed above; there is /// no path where this function answers `true` on evidence weaker than one of the three certificates. -bool isCreatorFenceTerminal(Backend & backend, const Layout & layout, const String & server_root_id, - uint64_t writer_epoch); +bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const String & server_root_id, + uint64_t writer_epoch, const Retry & policy = Retry::standard()); /// Synchronous owner of the durable mount lease and merged build-watermark body. The stable /// `CasMountRuntime` is the sole driver: this class never creates a thread, invokes a callback into the /// runtime, or performs a durable write from its destructor. /// /// ADOPT RULE (critical): the steady-state flow is `claimMount(...)` writes the live mount under -/// (our_uuid, our_epoch), THEN `keeper.start()`. So `start`'s `claim` hook must ADOPT a live mount +/// (our_uuid, our_epoch), THEN `renewer.start()`. So `start`'s `claim` hook must ADOPT a live mount /// that is ALREADY ours — same `server_uuid` AND same `writer_epoch` — instead of self-tripping the /// live-double-start guard. The discriminator is the (uuid, epoch) pair: -/// - same uuid + same epoch → our own just-written claim (or a replay) → adopt: `putOverwrite` -/// against the observed token to refresh seq/expiry (no fail); +/// - same uuid + same epoch → our own just-written claim (or a replay) → adopt: `replace` +/// against the observed incarnation to refresh seq/expiry (no fail); /// - same uuid + DIFFERENT live epoch → a newer incarnation superseded us → fail closed; /// - foreign uuid → fail closed; -/// - absent → `putIfAbsent`; expired-our-uuid (any epoch) → `putOverwrite` reclaim. -class MountLeaseKeeper +/// - absent → `create`; expired-our-uuid (any epoch) → `replace` reclaim. +/// +/// PLANES. Only the RENEWAL is admitted under the mount fence, because only a renewal writes under +/// authority the fence is tracking. The claim and the farewell are admitted off it: a self-remount +/// claims with the fence already latched lost, so a claim gated on the fence could never reclaim, and +/// a farewell refused because the fence has run down would leave the slot looking live until GC +/// fences it out. Neither is unguarded: a claim's safety is its own conditional write, and a caller +/// that has shutdown facts hands them over as a `Liveness`. +class MountLeaseRenewer { public: - MountLeaseKeeper( - BackendPtr backend_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, + MountLeaseRenewer( + CasRequests & mount_requests_, CasRequests & open_requests_, const Layout & layout_, + const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, - std::function min_active_fn_, + std::function min_active_build_sequence_fn_, CasEventSink event_sink_ = {}, std::chrono::milliseconds lease_safety_margin_ = std::chrono::milliseconds(2000), /// boot-domain clock for the on_renew_ok anchor; empty = real CLOCK_BOOTTIME. Injectable for - /// tests and wired by CasMountRuntime::installKeeper. + /// tests and wired by CasMountRuntime::installRenewer. std::function boot_ms_fn_ = {}); - /// Adopt the already-claimed mount. Returns the exact pre-I/O BOOTTIME anchor. - uint64_t start(); - MountRenewResult renew(const CasRequestBudget & budget, const MountRenewOperationEnvironment & environment); + /// Adopt the already-claimed mount. Returns the exact pre-I/O BOOTTIME anchor. `liveness` carries + /// the caller's shutdown terms; the mount fence is deliberately not consulted here. + uint64_t start(Liveness liveness = {}); + /// The steady-state renewal, admitted under the mount fence. + MountRenewResult renew(const MountRenewOperationEnvironment & environment); + /// The remount's re-anchor, which is bootstrap control rather than steady state: a remount renews + /// BEFORE it arms the fence for the new incarnation, so the fence is still latched lost and an + /// operation admitted under it would be refused before its first attempt. It admits on this + /// renewer's own open plane, the one the claim and the farewell use, so there is no plane for a + /// caller to get wrong. Same policy and same verdicts as `renew`. + MountRenewResult renewForRemount(const MountRenewOperationEnvironment & environment = {}); void release(); - MountLeaseKeeperState state() const { return keeper_state; } - bool canRelease() const { return keeper_state == MountLeaseKeeperState::Active; } + MountLeaseRenewerState state() const { return renewer_state; } + bool canRelease() const { return renewer_state == MountLeaseRenewerState::Active; } uint64_t lastCommittedAttemptStartBootMs() const { return last_committed_attempt_start_boot_ms; } private: - String encodeBody(uint64_t seq_, uint64_t wall_ms, uint64_t min_active, UInt128 write_attempt_id) const; - Token claim(const String & body); - [[noreturn]] void throwRenewConflict(const CasOverwriteDiagnostics & diagnostics) const; - MountRenewResult terminalResult( - uint64_t attempt_start_boot_ms, - CasOverwriteDiagnostics diagnostics, - std::exception_ptr failure); - void terminate(); - - BackendPtr backend; + String encodeBody(uint64_t seq_, uint64_t wall_ms, uint64_t min_active_build_sequence, UInt128 write_attempt_id) const; + /// The incarnation every guarded write of this slot names. Engaged for exactly the states that + /// admit such a write: `start` establishes it and each committed renewal replaces it. + const Etag & precondition() const; + /// One renewal admitted on `plane`; `renew` and `renewForRemount` differ only in which they pass. + MountRenewResult renewOn(CasRequests & plane, const MountRenewOperationEnvironment & environment); + Etag claim(CasOperation & op, const String & body); + [[noreturn]] void throwRenewConflict(const Observation & seen) const; + MountRenewResult terminalResult(MountRenewResult result); + void terminate(CasOperation & op); + + CasRequests & mount_requests; + CasRequests & open_requests; String key; String srid; @@ -521,15 +580,17 @@ class MountLeaseKeeper uint64_t writer_epoch; std::chrono::milliseconds ttl; std::function now_ms_fn; - std::function min_active_fn; + std::function min_active_build_sequence_fn; CasEventSink event_sink; std::chrono::milliseconds lease_safety_margin; /// boot-domain clock for the on_renew_ok anchor; empty = real CLOCK_BOOTTIME. Injectable for - /// tests and wired by CasMountRuntime::installKeeper. + /// tests and wired by CasMountRuntime::installRenewer. std::function boot_ms_fn; - MountLeaseKeeperState keeper_state = MountLeaseKeeperState::New; + MountLeaseRenewerState renewer_state = MountLeaseRenewerState::New; uint64_t seq = 0; - Token last_token; + /// The incarnation our last landed write created; every renewal and the farewell name it as the + /// precondition. Unset only before `start` has landed one. + std::optional last_etag; uint64_t confirmed_deadline_boot_ms = 0; uint64_t last_committed_attempt_start_boot_ms = 0; }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.cpp index ae34e530bb1c..90a695349916 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.cpp @@ -1,20 +1,14 @@ #include +#include namespace DB::Cas { +static_assert(casEnumTableCoversEnum()); + std::string_view blobHashAlgoName(BlobHashAlgo algo) { - switch (algo) - { - case BlobHashAlgo::CityHash128: - return "ch128"; - case BlobHashAlgo::XXH3_128: - return "xxh3"; - case BlobHashAlgo::Sha256: - return "sha256"; - } - throw Exception(ErrorCodes::BAD_ARGUMENTS, "blobHashAlgoName: unknown BlobHashAlgo {}", static_cast(algo)); + return kBlobHashAlgoWords.toWord(algo, "blobHashAlgoName"); } uint64_t blobHashLenFor(BlobHashAlgo algo) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.h index cc557fd753e1..70f50f4e6682 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasBlobDigest.h @@ -1,4 +1,5 @@ #pragma once +#include #include #include #include @@ -42,14 +43,23 @@ enum class BlobHashAlgo : uint8_t Sha256 = 3, }; +/// The `BlobHashAlgo` wire vocabulary (also the blob PATH SEGMENT, e.g. +/// `/blobs///`); coverage is proven in `CasBlobDigest.cpp`. +inline constexpr EnumWireTable kBlobHashAlgoWords{{{ + {BlobHashAlgo::CityHash128, "ch128"}, + {BlobHashAlgo::XXH3_128, "xxh3"}, + {BlobHashAlgo::Sha256, "sha256"}, +}}}; + /// The blob PATH SEGMENT for `algo`, e.g. `/blobs///`: `"ch128"` | `"xxh3"` | -/// `"sha256"`. Throws `BAD_ARGUMENTS` for an out-of-range enum value. +/// `"sha256"`. Throws `LOGICAL_ERROR` for an out-of-range enum value. std::string_view blobHashAlgoName(BlobHashAlgo algo); /// Returns the digest byte width for `algo`: 16 for `CityHash128` and `XXH3_128`, or 32 for /// `Sha256`. This is also the width used by `Cas::codecFor(algo)`'s `DigestCodec`; callers must -/// derive it from the algorithm rather than from pool state. Throws `BAD_ARGUMENTS` for an -/// out-of-range enum value, preserving the fail-closed contract of `blobHashAlgoName`. +/// derive it from the algorithm rather than from pool state. The functions over `BlobHashAlgo` +/// intentionally use different defensive codes: this one throws `BAD_ARGUMENTS` for an out-of-range +/// enum value, while `blobHashAlgoName` throws `LOGICAL_ERROR`. uint64_t blobHashLenFor(BlobHashAlgo algo); /// Parses the per-disk `blob_hash` CONFIG value: `"cityhash128"` | `"xxh3-128"` | `"sha256"`. Throws diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTable.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTable.h new file mode 100644 index 000000000000..31d76eee0bf8 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTable.h @@ -0,0 +1,73 @@ +#pragma once + +#include +#include + +#include +#include + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LOGICAL_ERROR; +} + +namespace DB::Cas +{ + +/// One persisted enum <-> wire-word vocabulary: the single carrier the encoder, the decoder, the +/// introspection renderer, and the tests all read. Every persisted enum is dense, so `toWord` is a +/// direct indexed lookup; `fromWord` is a linear pass over a handful of words. Coverage is proven +/// at each table's definition site by `casEnumTableCoversEnum` (CasEnumWireTableAsserts.h — .cpp +/// and tests only) together with the `denseAndOrdered`/`wordsUnique` predicates below. +template +struct EnumWireTable +{ + struct Entry + { + Enum value; + std::string_view word; + }; + + std::array entries; + + /// An empty table would make `denseAndOrdered`/`wordsUnique` vacuously true and `toWord`'s + /// index arithmetic read past the array — no wire vocabulary is empty, so reject at compile time. + static_assert(N > 0, "EnumWireTable must hold at least one entry"); + + constexpr bool denseAndOrdered() const + { + for (size_t i = 0; i < N; ++i) + if (static_cast(entries[i].value) != static_cast(entries[0].value) + i) + return false; + return true; + } + + constexpr bool wordsUnique() const + { + for (size_t i = 0; i < N; ++i) + for (size_t j = i + 1; j < N; ++j) + if (entries[i].word == entries[j].word) + return false; + return true; + } + + std::string_view toWord(Enum value, std::string_view what) const + { + const uint64_t index = static_cast(value) - static_cast(entries.front().value); + if (index >= entries.size() || entries[index].value != value) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "{}: value {} is outside the wire vocabulary", what, static_cast(value)); + return entries[index].word; + } + + Enum fromWord(std::string_view word, std::string_view what) const + { + for (const auto & entry : entries) + if (entry.word == word) + return entry.value; + throw Exception(ErrorCodes::CORRUPTED_DATA, "{}: unknown word '{}'", what, word); + } +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTableAsserts.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTableAsserts.h new file mode 100644 index 000000000000..f2cacfa32f28 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEnumWireTableAsserts.h @@ -0,0 +1,38 @@ +#pragma once + +/// Compile-time coverage proof for EnumWireTable: SET EQUALITY with the enum's declared values. +/// Size-plus-uniqueness is not enough (an invalid casted value satisfies both while an enumerator +/// goes missing). This header pulls in magic_enum and therefore MUST be included only from .cpp +/// files and tests, never from another header. The proof assumes every enumerator is in +/// magic_enum's reflectable range (by default -128..127), because `enum_values` sees only that range. + +#include + +#include + +namespace DB::Cas +{ + +template +consteval bool casEnumTableCoversEnum() +{ + /// One assert per table carries all three obligations: a table author cannot forget density + /// or word uniqueness, because coverage subsumes them. + if (!Table.denseAndOrdered() || !Table.wordsUnique()) + return false; + constexpr auto declared = magic_enum::enum_values(); + if (declared.size() != Table.entries.size()) + return false; + for (size_t i = 0; i < declared.size(); ++i) + { + bool found = false; + for (const auto & entry : Table.entries) + if (entry.value == declared[i]) + found = true; + if (!found) + return false; + } + return true; +} + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.h index 752ac5cf3f84..05b9b75a8967 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasEvent.h @@ -54,9 +54,12 @@ enum class CasEventObjectKind { None, Blob, Manifest, Root, Snap }; /// Pure-data event passed from the content-addressed core to the metadata-storage audit-log sink. /// Fields that do not apply to an event remain empty or zero. `reason` is mandatory for decisions /// and must explain why the operation took its outcome; `detail` carries structured facts needed -/// to reconstruct the event without parsing the free-form reason. Hashes are lowercase hexadecimal, -/// tokens identify object incarnations, and the numeric fields identify GC rounds, snapshot -/// generations, or the manifest journal version as applicable. +/// to reconstruct the event without parsing the free-form reason. Hashes are lowercase hexadecimal. +/// `token` identifies an object incarnation on events about a stored object (`object_kind` is Blob or +/// Manifest); the part-build lifecycle events that carry a token at all (`BuildStart`, `Precommit`, +/// `BuildPublish`, `BuildAbort`) reuse it for the 128-bit build id in hex and leave `object_kind` at +/// `None`. The numeric fields +/// identify GC rounds, snapshot generations, or the manifest journal version as applicable. struct CasEvent { CasEventType type = CasEventType::BlobPut; @@ -64,7 +67,7 @@ struct CasEvent String ref_name; /// the ref name — a mutable directory handle, git-style (empty if N/A) CasEventObjectKind object_kind = CasEventObjectKind::None; String object_hash; /// lowercase hex (empty if N/A) - String token; /// incarnation token (empty if N/A) + String token; /// incarnation token; the hex build id on build lifecycle events (empty if N/A) UInt64 round = 0; UInt64 gen = 0; UInt64 at_version = 0; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasTypes.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasTypes.h index ff25c75d7dd5..50c56d280ddf 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasTypes.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasTypes.h @@ -245,26 +245,13 @@ namespace DB::Cas { /// How a backend identifies one physical incarnation of an object key. -enum class TokenType : uint8_t +enum class Dialect : uint8_t { ETag = 1, /// S3-family / Azure Generation = 2, /// GCS (binding deferred; fail-closed until probed) Emulated = 3, /// test backends (in-memory fake, Local emulation) }; -/// A backend-native incarnation token. Opaque to every CAS caller and sent back to the backend -/// EXACTLY as held here. The backend owns the one conversion between its transport representation and -/// this value — see `ObjectStorageBackend::normalizeTokenValue`, which removes the quoting the GCS -/// generation picks up from riding the SDK's ETag field. -struct Token -{ - String value; - TokenType type = TokenType::ETag; - - bool empty() const { return value.empty(); } - bool operator==(const Token &) const = default; -}; - /// The ordered ref-transaction identifier. A successful writer mount establishes a strictly newer /// `writer_epoch`; within an epoch, one namespace's `ref_sequence` values are CONTIGUOUS from 1 -- /// derived per append from the table's own greatest applied id (`nextRefTxnId`), not drawn from any diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasWriteOnceKey.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasWriteOnceKey.h new file mode 100644 index 000000000000..058977b939a7 --- /dev/null +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Primitives/CasWriteOnceKey.h @@ -0,0 +1,26 @@ +#pragma once +#include +#include + +namespace DB::Cas +{ + +class Layout; + +/// The key of an object that is written once and never rewritten: a part manifest, a ref log, a ref +/// snapshot. Only `Layout` mints one, from the typed identity of such an object, so a verb that +/// accepts this type can delete without a precondition: whatever body the key holds is the one body +/// it ever held. A mutable control object (a checkpoint, the catalog, `gc/state`, a mount lease) has +/// no path to this type. +class WriteOnceKey +{ +public: + const String & str() const { return key; } + +private: + friend class Layout; + explicit WriteOnceKey(String key_) : key(std::move(key_)) {} + String key; +}; + +} diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/README.md b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/README.md index 305ffe1b1ff9..44430e021ddb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/README.md +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/README.md @@ -81,7 +81,7 @@ Primitives → Formats → Backend → Pool → Gc → Tools ≈ Parts → facad - **`Primitives/`** — the vocabulary, zero outward dependencies: `CasBlobDigest` (`BlobHashAlgo` + `BlobDigest` + `DigestCodec` + `BlobRef` — blob identity), - `CasTypes.h` (the other identity types: `RootNamespace`, `Token`, + `CasTypes.h` (the other identity types: `RootNamespace`, `Dialect`, `ManifestId`, `RefTxnId`), `CasNamespaceLifeId` (`NamespaceLifeId` — one LIFE of a namespace's ref layer, the pair every ref key is built from), `CasBlobHashingWriteBuffer` (streaming @@ -92,13 +92,14 @@ Primitives → Formats → Backend → Pool → Gc → Tools ≈ Parts → facad text/format files (`CasFormat`, `CasTextFormat`, `CasPartManifestFormat`, `CasRefLogFormat`, …) plus `CasLayout` (the object-key schema). See `Formats/README.md` for the format registry. -- **`Backend/`** — the token-aware storage seam: `CasBackend` (the contract: - `get`/`putIfAbsent`/`casPut`/`deleteExact` with CAS tokens, plus unconditional - transport-only `publishBlob`), +- **`Backend/`** — the request-contract seam: `CasBackend` (the transport + primitives, dealing only in the store's own raw incarnation strings: + `read`/`head`/`list`/`remove`/`write`/`stream`/`publish`), `CasObjectStorageBackend`, `CasInMemoryBackend`, `CasInstrumentedBackend`, - `CasRequestControl` (single-attempt conditional non-blob writes, including create-if-absent - artifacts and conditional replacements, with explicit - state-aware retries), `CasProbe` (mount-time capability probe). + `CasRequests` (the retrying, `Etag`-typed layer above it — `create`/ + `replace`/`readModifyWrite`/`read`/`remove`/`probeSentinel`/`stream`/ + `publish`, admitted and budgeted by `Retry` and `CasRequestBudget`), + `CasProbe` (mount-time capability probe). - **`Pool/`** — the pool engine: `CasPool` (composition root), `CasPartWriteTxn` (one-part write transaction), `CasRefLedger` + `CasRefProtocol` (ref-table log/snapshot/replay + intake), `CasServerRoot` (mount-claim protocol + @@ -112,7 +113,7 @@ Primitives → Formats → Backend → Pool → Gc → Tools ≈ Parts → facad `CasDecommission`, `CasInspect`. - **`Parts/`** — part semantics over the pool: `PartPathParser` (the ClickHouse-path classifier), `PartFolderAccess` (`PartRefKey` + `Freshness` - + `PartFolderValidate` + `PartFolderView` + `CachedPartFolderAccess`). + + `PartFolderView` + `CachedPartFolderAccess`). - **Top level (facade)** — the entry points: `ContentAddressedMetadataStorage` (the `IMetadataStorage` facade), `ContentAddressedTransaction` (the `IMetadataTransaction`, including the write buffers), `ContentAddressedExchange` diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp index 416f5c327663..eef2ba04d9bf 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include #include @@ -7,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -34,68 +36,76 @@ uint64_t nowMs() std::chrono::system_clock::now().time_since_epoch()).count()); } -/// Delete every object listed under `prefix` by its listed (or, absent a list-token backend, HEAD'd) -/// token. This backs the staging and roots drain phases below: the victim's writers are fenced by the -/// decommission claim (`Pool::openForDecommission`), so nothing should be racing these deletes, and a -/// plain exact-token delete of every listed object is race-free. +std::string_view removalName(Removal r) +{ + switch (r) + { + case Removal::Removed: return "removed"; + case Removal::Gone: return "gone"; + case Removal::Mismatch: return "mismatch"; + } + UNREACHABLE(); +} + +/// Delete every object listed under `prefix` by its listed (or, absent a list-incarnation backend, +/// HEAD'd) incarnation. This backs the staging and roots drain phases below: the victim's writers are +/// fenced by the decommission claim (`Pool::openForDecommission`), so nothing should be racing these +/// deletes, and a plain exact-incarnation delete of every listed object is race-free. /// -/// A per-object failure — a backend exception, a `TokenMismatch` or `NotFound` outcome, or an object +/// A per-object failure — a backend exception, a `Mismatch` or `Gone` outcome, or an object /// disappearing between `LIST` and `HEAD` — is recorded as a warning and does not prevent the remaining /// objects from being attempted. The caller keeps the pool slot whenever warnings are present, so the /// terminated slot remains available as a resume anchor instead of being deleted after an unconfirmed -/// drain. Returns only the objects whose exact-token delete was reported as `Deleted`. -uint64_t deleteListedPrefix(Backend & backend, const String & prefix, std::vector & warnings) +/// drain. Returns only the objects whose exact-incarnation delete was reported as `Removed`. +uint64_t deleteListedPrefix(CasOperation & op, const String & prefix, std::vector & warnings) { uint64_t deleted = 0; - forEachListedKey(backend, prefix, [&](const ListedKey & listed) + op.forEachListedKey(prefix, [&](const ListedKey & listed) { try { - Token token; - if (listed.token) - token = *listed.token; - else + std::optional etag = listed.etag; + if (!etag) { - const HeadResult head = backend.head(listed.key); - if (!head.exists) + const std::optional head = op.head(listed.key, Retry::standard()); + if (!head) { warnings.push_back("decommission drain: " + listed.key + " vanished before delete"); - return; + return true; } - token = head.token; + etag = head->etag; } - const DeleteOutcome outcome = backend.deleteExact(listed.key, token); - const DeleteClass outcome_class = classifyDeleteOutcome(outcome); - if (outcome_class == DeleteClass::Deleted) + const Removal outcome = op.remove(listed.key, *etag, Retry::standard()); + if (outcome == Removal::Removed) ++deleted; else warnings.push_back("decommission drain: " + listed.key + " delete outcome " - + String(deleteClassName(outcome_class))); + + String(removalName(outcome))); } catch (...) { warnings.push_back("decommission drain: " + listed.key + " delete failed: " + getCurrentExceptionMessage(/*with_stacktrace=*/false)); } - }); + return true; + }, Retry::standard()); return deleted; } -/// Delete one slot control object by a token captured at the protocol-defined fence point. Slot -/// retirement is fail-closed: unlike the debris drains above, any non-`Deleted` outcome or exception +/// Delete one slot control object by an incarnation captured at the protocol-defined fence point. Slot +/// retirement is fail-closed: unlike the debris drains above, any non-`Removed` outcome or exception /// stops the tail before it can touch the next control object. -bool deleteSlotObject(Backend & backend, const String & key, const Token & token, std::vector & warnings) +bool deleteSlotObject(CasOperation & op, const String & key, const Etag & etag, std::vector & warnings) { try { - const DeleteOutcome outcome = backend.deleteExact(key, token); - const DeleteClass outcome_class = classifyDeleteOutcome(outcome); - if (outcome_class == DeleteClass::Deleted) + const Removal outcome = op.remove(key, etag, Retry::standard()); + if (outcome == Removal::Removed) return true; warnings.push_back("slot delete failed: " + key + ": delete outcome " - + String(deleteClassName(outcome_class))); + + String(removalName(outcome))); } catch (...) { @@ -109,7 +119,9 @@ bool deleteSlotObject(Backend & backend, const String & key, const Token & token DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, const String & victim_srid, const CasEventSink & sink, - const std::function & request_gc_round) + const std::function & request_gc_round, + const std::function & drain_now_fn, + const std::function & drain_sleep_fn) { DecommissionReport report; report.srid = victim_srid; @@ -122,15 +134,61 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, request_gc_round(); }); + /// Decommission is administrative and non-hot-path: it has no mount-lease fence of its own -- the + /// exact-incarnation compare on every write is the safety mechanism -- so it opens its own + /// always-admitted request engine rather than one of `Pool`'s fenced planes. The pre-impersonation + /// cut below runs against the raw backend passed in, because `Pool::openForDecommission` has not + /// yet wrapped it for instrumentation and no `Pool` exists yet to route through. + CasRequests preflight_requests(backend, Fence::open()); + CasOperation preflight_op = preflight_requests.admit(); + /// Validate one required immutable ownership cut before impersonating the victim. The admin open /// performs its own fresh catalog observation for mount safety, but namespace selection below /// must reuse this exact pre-mutation decision rather than read a later authority set. const Layout catalog_layout(config.pool_prefix); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, catalog_layout); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(preflight_op, catalog_layout); catalog_cut.life_index.throwIfAmbiguous("CAS decommission"); + /// `drain_now_fn`/`drain_sleep_fn` below replace the clock and sleep the STANDALONE `requests` + /// engine (opened further down) paces its own retries on, but `config.boot_ms_fn`/ + /// `config.retry_sleep_fn` -- the seams `Pool::openForDecommission` itself constructs its + /// mount/farewell/GC planes with -- are distinct. Left unset, those planes retry on the real boot + /// clock and a real sleep, and a caller that fakes only the standalone engine's clock ends up + /// comparing it against an unrelated one at the mount lease's farewell bound (or, worse, against a + /// clock that only the standalone engine's sleep advances: a retry loop on `mount_requests`/ + /// `gc_requests` bound to that frozen clock while actually sleeping for real would never see its + /// own deadline elapse). Fold BOTH into `config` here, together, unless the caller already asked + /// for a specific clock or sleep of its own -- installing only one of the two is exactly the + /// half-fix that leaves the other seam retrying forever. + if (drain_now_fn && drain_sleep_fn) + { + if (!config.boot_ms_fn) + config.boot_ms_fn = drain_now_fn; + if (!config.retry_sleep_fn) + config.retry_sleep_fn = drain_sleep_fn; + } + config.event_sink = sink; PoolPtr admin = Pool::openForDecommission(std::move(backend), std::move(config), victim_srid); + if (drain_now_fn && drain_sleep_fn) + { + /// Re-affirms the same values `config` above already installed on `mount_requests`/ + /// `farewell_requests`/`gc_requests` at construction, and additionally wires `ref_ledger`'s own + /// retry sleep, which has no construction-time seam of its own. `sweepNamespace` below issues + /// its deletes on `admin`'s own GC plane. + admin->setCasRequestNowFnForTest(drain_now_fn); + admin->setCasRetrySleepForTest(drain_sleep_fn); + } + + /// A second engine over the pool's own (now instrumented) backend: `CasRequests` keeps its own + /// shared_ptr to it, so `op` stays usable after `admin.reset()` retires the `Pool` below. + CasRequests requests(admin->poolBackendPtr(), Fence::open()); + if (drain_now_fn && drain_sleep_fn) + { + requests.setNowFnForTest(drain_now_fn); + requests.setSleepFnForTest(drain_sleep_fn); + } + CasOperation op = requests.admit(); EventEmitter{*admin}.emit([&](CasEvent & e) { @@ -164,7 +222,7 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, /// Refuse a same-name lifecycle move that landed after the immutable selection cut. The /// exact-life overloads below also pin recovery to `life`, closing the race after this check: /// a later replacement can never redirect a removal to its new incarnation. - const CasRefCatalog::Snapshot current_catalog = CasRefCatalog::read(admin->backend(), admin->layout()); + const CasRefCatalog::Snapshot current_catalog = CasRefCatalog::read(op, admin->layout()); const auto current_entry = std::find_if( current_catalog.catalog.entries.begin(), current_catalog.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns.string() == ns_str; }); @@ -176,7 +234,7 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, if (selected_entry.state == NsState::Removing) { - if (!admin->backend().head(admin->layout().refCkptKey(life)).exists) + if (!op.head(admin->layout().refCkptKey(life), Retry::standard()).has_value()) throw Exception(ErrorCodes::CORRUPTED_DATA, "ca-decommission: namespace '{}' is Removing but its exact checkpoint is absent; " "the catalog row remains owned and the victim slot cannot be retired", @@ -219,11 +277,12 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, { const String debris_prefix = admin->layout().casManifestsServerPrefix(victim_srid); std::set> groups; /// (namespace, writer epoch, build sequence) - forEachListedKey(admin->backend(), debris_prefix, [&](const ListedKey & listed) + op.forEachListedKey(debris_prefix, [&](const ListedKey & listed) { if (const auto parsed = admin->layout().parseManifestKey(listed.key)) groups.emplace(parsed->root_namespace.string(), parsed->ref.writer_epoch, parsed->ref.build_sequence); - }); + return true; + }, Retry::standard()); for (const auto & [ns_str, writer_epoch, build_sequence] : groups) report.manifest_debris_removed += sweepNamespace( *admin, RootNamespace(ns_str), BuildPrefix{writer_epoch, build_sequence}, &report.warnings); @@ -233,23 +292,23 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, /// an `IObjectStorage`, while this command intentionally works at the `Backend` layer, so the same /// prefix is listed and deleted directly. The claim fences the victim's writers during this sweep. report.staging_objects_removed += deleteListedPrefix( - admin->backend(), admin->poolConfig().pool_prefix + "/staging/" + victim_srid + "/", report.warnings); + op, admin->poolConfig().pool_prefix + "/staging/" + victim_srid + "/", report.warnings); /// Drain the victim's mountpoint objects. These are loose, non-content-addressed files under /// `Layout::serverRootDataPrefix`; they have no writer epoch of their own, so the claim is what /// prevents a returning victim from racing this deletion. report.mountpoint_objects_removed += deleteListedPrefix( - admin->backend(), admin->layout().serverRootDataPrefix(victim_srid), report.warnings); + op, admin->layout().serverRootDataPrefix(victim_srid), report.warnings); /// The catalog, not physical debris, owns the slot-retirement decision. A terminal append only /// moves a row to `Removing`; GC must fold/prune/delete it before the member's ownership anchor can - /// disappear. Capture one exact whole-catalog cut after every drain, then revalidate its token and - /// canonical value immediately before entering the retirement tail. The administrative claim fences - /// the victim writer between those observations. + /// disappear. Capture one exact whole-catalog cut after every drain, then revalidate its incarnation + /// and canonical value immediately before entering the retirement tail. The administrative claim + /// fences the victim writer between those observations. std::optional retirement_catalog_cut; if (report.warnings.empty()) { - retirement_catalog_cut = CasRefCatalog::read(admin->backend(), admin->layout()); + retirement_catalog_cut = CasRefCatalog::read(op, admin->layout()); const uint64_t victim_owned_count = std::count_if( retirement_catalog_cut->catalog.entries.begin(), retirement_catalog_cut->catalog.entries.end(), [&](const CatalogEntry & entry) @@ -264,17 +323,15 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, "cleanup — re-run this command afterwards to retire the slot"); } - /// Retire the slot strictly last and only after a clean drain. Copy the layout and shared backend - /// before `admin.reset()`: graceful close destroys the `Pool`, while the backend must remain alive to - /// retire the slot objects afterwards. + /// Retire the slot strictly last and only after a clean drain. Copy the layout before + /// `admin.reset()`: graceful close destroys the `Pool`, while `op` (holding its own shared_ptr to + /// the backend) remains usable to retire the slot objects afterwards. const Layout layout = admin->layout(); - const BackendPtr pool_backend = admin->poolBackendPtr(); if (report.warnings.empty()) { - const CasRefCatalog::Snapshot fresh_retirement_catalog - = CasRefCatalog::read(admin->backend(), admin->layout()); + const CasRefCatalog::Snapshot fresh_retirement_catalog = CasRefCatalog::read(op, admin->layout()); if (!retirement_catalog_cut - || fresh_retirement_catalog.token != retirement_catalog_cut->token + || fresh_retirement_catalog.etag != retirement_catalog_cut->etag || fresh_retirement_catalog.catalog != retirement_catalog_cut->catalog) { report.warnings.push_back( @@ -287,13 +344,13 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, const String epoch_key = layout.epochKey(victim_srid); const String owner_key = layout.ownerKey(victim_srid); - /// Capture both the epoch value and its exact token while the decommission claim still fences - /// the victim. A successor can only bump this object after the farewell below releases the - /// claim, so this token is the epoch-side successor fence for the retirement tail. - std::optional claimed_epoch; + /// Capture both the epoch value and its exact incarnation while the decommission claim still + /// fences the victim. A successor can only bump this object after the farewell below releases + /// the claim, so this incarnation is the epoch-side successor fence for the retirement tail. + std::optional claimed_epoch; try { - claimed_epoch = pool_backend->get(epoch_key); + claimed_epoch = op.read(epoch_key, Retry::standard()); if (!claimed_epoch) report.warnings.push_back("slot capture failed: " + epoch_key + " is absent under the admin claim"); } @@ -304,18 +361,18 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, } /// Graceful close stamps an already-expired lease and the watermark farewell - /// (`min_active = UINT64_MAX`), making the slot `terminated` before its mutable control objects + /// (`min_active_build_sequence = UINT64_MAX`), making the slot `terminated` before its mutable control objects /// are removed and its owner anchor is tombstoned. admin.reset(); - /// Read the farewell immediately after `finishTeardown` wrote it. Its exact token is the - /// mount-side fence: deleting by this token can remove only THIS decommission's farewell, not - /// a successor reclaim. Validate the body against the epoch value captured under the claim so - /// a successor that completed before this GET is also recognized and left untouched. - std::optional farewell_mount; + /// Read the farewell immediately after `finishTeardown` wrote it. Its exact incarnation is the + /// mount-side fence: deleting by this incarnation can remove only THIS decommission's farewell, + /// not a successor reclaim. Validate the body against the epoch value captured under the claim + /// so a successor that completed before this read is also recognized and left untouched. + std::optional farewell_mount; try { - farewell_mount = pool_backend->get(mount_key); + farewell_mount = op.read(mount_key, Retry::standard()); if (!farewell_mount) report.warnings.push_back("slot capture failed: " + mount_key + " farewell is absent"); } @@ -334,7 +391,7 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, const MountLease mount_value = decodeMountLease(farewell_mount->bytes); captures_match = epoch_value.next_writer_epoch != 0 && mount_value.writer_epoch == epoch_value.next_writer_epoch - 1 - && mount_value.min_active == std::numeric_limits::max() + && mount_value.min_active_build_sequence == std::numeric_limits::max() && !mount_value.gc_fenced; if (!captures_match) { @@ -352,16 +409,16 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, } /// Mount first: if a successor reclaimed it after the farewell capture, the stale farewell - /// token yields `TokenMismatch` and the tail stops before touching epoch or owner. Epoch second: - /// its under-claim token similarly detects a successor allocation. Before touching owner, re-read - /// both mutable objects: a same-UUID successor can recreate them after both deletes without - /// rewriting the owner identity anchor. Mere presence proves that the slot is live again. Every - /// delete must be explicitly confirmed as `Deleted`, and the final owner tombstone rewrite must - /// succeed against the exact token read immediately before it. + /// incarnation yields `Mismatch` and the tail stops before touching epoch or owner. Epoch + /// second: its under-claim incarnation similarly detects a successor allocation. Before + /// touching owner, re-read both mutable objects: a same-UUID successor can recreate them after + /// both deletes without rewriting the owner identity anchor. Mere presence proves that the slot + /// is live again. Every delete must be explicitly confirmed as `Removed`, and the final owner + /// tombstone rewrite must succeed against the exact incarnation read immediately before it. /// /// ACCEPTED RESIDUAL WINDOW (final review, not closed by this recheck): a same-UUID successor /// can still recreate epoch/mount in the narrow gap strictly AFTER this liveness recheck but - /// BEFORE the owner CAS below reads its own token -- the successor's owner anchor (same + /// BEFORE the owner CAS below reads its own incarnation -- the successor's owner anchor (same /// server_uuid, not yet retired) then gets tombstoned by this decommission run. The successor's /// live process is not deleted (only its owner anchor is marked retired), but a LATER restart of /// that same identity would refuse to reclaim it (claimOwnerOrThrow's tombstone guard). This is @@ -369,15 +426,15 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, /// (finding #9) intentionally stopped short of making concurrent decommission-vs-recreate /// airtight to the microsecond, since that was explicitly not the priority for this fix. report.slot_removed = false; - if (captures_match && deleteSlotObject(*pool_backend, mount_key, farewell_mount->token, report.warnings) - && deleteSlotObject(*pool_backend, epoch_key, claimed_epoch->token, report.warnings)) + if (captures_match && deleteSlotObject(op, mount_key, farewell_mount->etag, report.warnings) + && deleteSlotObject(op, epoch_key, claimed_epoch->etag, report.warnings)) { - std::optional current_mount; - std::optional current_epoch; + std::optional current_mount; + std::optional current_epoch; bool liveness_recheck_succeeded = true; try { - current_mount = pool_backend->get(mount_key); + current_mount = op.read(mount_key, Retry::standard()); } catch (...) { @@ -387,7 +444,7 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, } try { - current_epoch = pool_backend->get(epoch_key); + current_epoch = op.read(epoch_key, Retry::standard()); } catch (...) { @@ -405,33 +462,35 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, { try { - if (const auto owner = pool_backend->get(owner_key)) + if (const auto owner = op.read(owner_key, Retry::standard())) { OwnerObject tombstoned = decodeOwner(owner->bytes); tombstoned.retired_at_ms = nowMs(); - /// Controlled, not a bare putOverwrite: a transient transport error here (or - /// one whose response was simply lost) must not be reported as a hard failure - /// when the write actually landed. A standalone controller (decommission is an - /// administrative, non-hot-path operation; no mount-lease fence applies to it - /// -- the exact-token CAS itself is the safety mechanism, same as the mount/ - /// epoch deletes above) resolves an ambiguous attempt with one GET: unchanged - /// token means the write never applied (legitimately retryable within budget); - /// matching bytes means this exact tombstone already landed (Committed, not a - /// failure); anything else is a genuine successor reclaim (Conflict). - CasRequestController controller(pool_backend, CasRequestBudget{}); - const CasOverwriteResult result = controller.putOverwriteControlled( - owner_key, encodeOwner(tombstoned), owner->token, [] { return true; }); - if (result.outcome == CasOverwriteOutcome::Committed) + /// `op.replace` resolves an ambiguous attempt with a resolve read on its own: + /// a transient transport error here (or one whose response was simply lost) + /// must not be reported as a hard failure when the write actually landed. + /// Unchanged incarnation means the write never applied (legitimately retryable + /// within budget); matching bytes means this exact tombstone already landed + /// (`Committed`, not a failure); a genuine successor reclaim is `Conflict`; + /// `Refused` is the store's own definite answer (a denial, a malformed + /// request, an expired credential) and carries its own code and message, + /// which is worth more here than the generic retry advice below. + WriteResult result = op.replace(owner_key, encodeOwner(tombstoned), owner->etag, Retry::standard()); + if (std::holds_alternative(result)) report.slot_removed = true; - else if (result.outcome == CasOverwriteOutcome::Conflict) + else if (std::holds_alternative(result)) report.warnings.push_back( "slot tombstone failed: " + owner_key + ": successor reclaimed the owner anchor before this decommission's tombstone write"); + else if (const Refused * refused = std::get_if(&result)) + report.warnings.push_back( + "slot tombstone failed: " + owner_key + ": the store refused the write (" + + std::to_string(refused->store_error) + "): " + refused->message); else report.warnings.push_back( "slot tombstone failed: " + owner_key + ": tombstone write outcome could not be resolved (retry budget exhausted " - "or the resolve GET itself failed) -- rerun the command to retry"); + "or the resolve read itself failed) -- rerun the command to retry"); } else report.warnings.push_back( diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h index a86edf9286c4..6832fd9cfe73 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.h @@ -48,8 +48,23 @@ struct DecommissionReport /// terminated slot as a resume anchor; other failures, including refusal to claim the member, propagate /// as exceptions. When set, `sink` receives `MemberDecommission` audit events for the run's begin, /// per-namespace, and end milestones. +/// +/// `drain_now_fn`/`drain_sleep_fn`, when both set, replace the clock the drain's own request engine +/// paces its retries on -- the engine this function opens is a standalone one over the instrumented +/// backend, not one of `Pool`'s planes, so `Pool::setCasRetrySleepForTest` cannot reach it. A test +/// driving a latched per-object fault to `Retry::standard()`'s own give-up needs this seam, or it pays +/// the real 90-second deadline. +/// +/// `drain_now_fn`/`drain_sleep_fn`, when BOTH set, also become the opened `Pool`'s own boot clock and +/// retry sleep (`PoolConfig::boot_ms_fn`/`retry_sleep_fn`) whenever the caller left those fields unset: +/// the mount lease's farewell deadline is bound to the boot clock, and `Pool::openForDecommission`'s own +/// mount/farewell/GC planes are constructed with it too, so a caller that fakes only the standalone +/// engine's clock must not end up comparing it against the real one, or -- worse -- against a plane that +/// shares the frozen clock but still sleeps for real between retries. DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, const String & victim_srid, const CasEventSink & sink = {}, - const std::function & request_gc_round = {}); + const std::function & request_gc_round = {}, + const std::function & drain_now_fn = {}, + const std::function & drain_sleep_fn = {}); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp index 7848ae591f27..8b67d00251a1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -51,13 +51,13 @@ void checkDeadline(const Deadline & deadline, std::string_view phase) "fsck: exceeded the deadline during '{}' — run against a QUIESCED pool or raise --timeout.", phase); } -void listAll(Backend & backend, const String & prefix, std::unordered_map & out, +void listAll(CasOperation & op, const String & prefix, std::unordered_map & out, const FsckProgress & on_progress, const Deadline & deadline, std::string_view phase) { static constexpr size_t kPageLimit = 1000; uint64_t pages = 0; size_t count_in_page = 0; - forEachListedKey(backend, prefix, [&](const ListedKey & k) + op.forEachListedKey(prefix, [&](const ListedKey & k) { out[k.key] = k.size; if (++count_in_page == kPageLimit) @@ -68,8 +68,9 @@ void listAll(Backend & backend, const String & prefix, std::unordered_map 0 || pages == 0) { @@ -124,12 +125,12 @@ using RecordRecoveryUnchecked = std::function recoverLateRefTable( - Backend & backend, const Layout & layout, const FsckRecoveryAuthority & authority, + CasOperation & op, const Layout & layout, const FsckRecoveryAuthority & authority, const RecordRecoveryUnchecked & record_unchecked) { try { - const std::optional sampled = readCkpt(backend, layout, authority.life); + const std::optional sampled = readCkpt(op, layout, authority.life); if (!sampled) { record_unchecked(authority.life.ns, layout.refCkptKey(authority.life), @@ -137,7 +138,7 @@ std::optional recoverLateRefTable( return std::nullopt; } return recoverRefTableDetailedFromAuthority( - backend, layout, authority.catalog_entry, sampled->ckpt).state; + op, layout, authority.catalog_entry, sampled->ckpt).state; } catch (const Exception & e) { @@ -153,7 +154,7 @@ std::optional recoverLateRefTable( } } -bool blobStillReferenced(Pool & store, const Layout & layout, +bool blobStillReferenced(CasOperation & op, Pool & store, const Layout & layout, const FsckRecoveryAuthorities & authorities, const String & bkey, const std::vector & labels, const Deadline & deadline, const RecordRecoveryUnchecked & record_unchecked) @@ -181,7 +182,7 @@ bool blobStillReferenced(Pool & store, const Layout & layout, } const RootNamespace rns{ns_part}; const std::optional table = recoverLateRefTable( - store.backend(), layout, authority_it->second, record_unchecked); + op, layout, authority_it->second, record_unchecked); if (!table) return true; const auto rit = table->getCommitted().find(ref_name); @@ -205,7 +206,7 @@ bool blobStillReferenced(Pool & store, const Layout & layout, } /// The manifest sibling of the `blobStillReferenced` recheck above. The ref-walk captures each committed -/// `(ref_name -> manifest_ref)` from a FRESH per-namespace recovery, but the `backend.get(mkey)` that +/// `(ref_name -> manifest_ref)` from a FRESH per-namespace recovery, but the read of `mkey` that /// confirms the manifest body runs LATER in the same (possibly long) namespace loop. A ref republished to /// a DIFFERENT manifest — or DROPPED — in that window, combined with a legitimate GC delete of the OLD /// manifest body, makes the stale captured row look like a committed ref over a missing manifest (a @@ -217,7 +218,7 @@ bool blobStillReferenced(Pool & store, const Layout & layout, /// but a same-life checkpoint advance must be visible. Fails CLOSED on any ambiguity (a throw, a corrupt /// table): treated as "still referenced", the original conservative verdict — the fix can only SHRINK /// false positives, never hide a real loss. -bool manifestStillReferenced(Backend & backend, const Layout & layout, const RootNamespace & ns, +bool manifestStillReferenced(CasOperation & op, const Layout & layout, const RootNamespace & ns, const FsckRecoveryAuthorities & authorities, const String & ref_name, const String & mkey, const Deadline & deadline, const RecordRecoveryUnchecked & record_unchecked) @@ -233,7 +234,7 @@ bool manifestStillReferenced(Backend & backend, const Layout & layout, const Roo return true; /// no original Live/Removing authority -- fail closed } const std::optional table = recoverLateRefTable( - backend, layout, authority_it->second, record_unchecked); + op, layout, authority_it->second, record_unchecked); if (!table) return true; const auto rit = table->getCommitted().find(ref_name); @@ -308,7 +309,7 @@ struct NsVerdicts /// `{life_epoch, 1}`) through that inclusive frontier. A missing required id is therefore a proven hole; /// no above-hole listing witness is needed. An epoch seal advances directly to the next epoch's first id, /// exactly as authoritative read-only recovery does. -void checkRefStream(Backend & backend, const Layout & layout, const NamespaceLifeId & life, +void checkRefStream(CasOperation & op, const Layout & layout, const NamespaceLifeId & life, const CatalogEntry & catalog_entry, const std::optional & checkpoint_sample, const Deadline & deadline, FsckReport & report, NsVerdicts & verdicts) { @@ -323,7 +324,7 @@ void checkRefStream(Backend & backend, const Layout & layout, const NamespaceLif { /// Even when the base IS the frontier and there is no replay tail, a checkpoint may not /// turn an `EpochSeal` into a state snapshot by naming a same-id `_snap`. - (void)readCheckpointSnapshotBase(backend, layout, life, *checkpoint); + (void)readCheckpointSnapshotBase(op, layout, life, *checkpoint); } catch (const Exception & e) { @@ -343,8 +344,8 @@ void checkRefStream(Backend & backend, const Layout & layout, const NamespaceLif checkDeadline(deadline, "checkpoint-base authority revalidation"); try { - const std::optional current = readCkpt(backend, layout, life); - if (!current || !checkpoint_sample || current->token != checkpoint_sample->token) + const std::optional current = readCkpt(op, layout, life); + if (!current || !checkpoint_sample || current->etag != checkpoint_sample->etag) { verdicts.recordUnchecked(report, ns, key, note + "; checkpoint authority changed while validating its snapshot base"); @@ -375,7 +376,7 @@ void checkRefStream(Backend & backend, const Layout & layout, const NamespaceLif while (expected <= *grounding.committed_through) { checkDeadline(deadline, "ref stream"); - const auto got = backend.get(layout.refLogKey(life, expected)); + const auto got = op.read(layout.refLogKey(life, expected), Retry::standard()); if (!got) { verdicts.recordChainBroken(report, ns, layout.refLogKey(life, expected), @@ -424,7 +425,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co const String & namespace_prefix, FsckReport & report) { const Layout & layout = store.layout(); - Backend & backend = store.backend(); + CasOperation op = store.openRequests().admit(); /// Path-derived per-object algorithm parsing: every listed blob-tree key -- across every /// admitted algo, not just the pool's node-local write algo -- is classified via /// `Layout::parseBlobKey`, which derives the `BlobRef` from the key's OWN `` path segment @@ -479,7 +480,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// attribution (its physical keys may exist) but is never recovered: only Live/Removing rows have a /// durable publication frontier. A diagnostic records duplicate ids and keeps walking unrelated /// unique lives. - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); struct FsckWalkLife { NamespaceLifeId life; @@ -525,7 +526,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co }; std::vector canonical_candidates; - forEachListedKey(backend, layout.namespaceRootPrefix(), [&](const ListedKey & listed) + op.forEachListedKey(layout.namespaceRootPrefix(), [&](const ListedKey & listed) { std::optional physical_id; try @@ -539,7 +540,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co else { recordLifelessKeys(NamespaceListing{{}, {{listed.key, "unrecognized key under the namespace ownership tree"}}}); - return; + return true; } } catch (const Exception & e) @@ -547,14 +548,15 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co if (e.code() != ErrorCodes::CORRUPTED_DATA) throw; recordLifelessKeys(NamespaceListing{{}, {{listed.key, e.message()}}}); - return; + return true; } canonical_candidates.push_back(CanonicalNamespaceKey{listed.key, listed.size, *physical_id}); - }); + return true; + }, Retry::standard()); /// The post-observation cut. All three catalog states -- `Creating`, `Live`, `Removing` -- /// protect a life for this purpose; only a life absent from every one of them is residue. - const CasRefCatalog::Snapshot post_listing_cut = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot post_listing_cut = CasRefCatalog::read(op, layout); std::unordered_set pending_lives; for (const CanonicalNamespaceKey & candidate : canonical_candidates) { @@ -615,7 +617,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// One materialized `_ckpt` body is part of this namespace's frozen audit authority. The /// recovery API receives exactly these bytes; `checkRefStream` receives the same decoded /// value, so the two legs cannot quietly choose different frontiers after a concurrent CAS. - const std::optional checkpoint_sample = readCkpt(backend, layout, life); + const std::optional checkpoint_sample = readCkpt(op, layout, life); const std::optional checkpoint = checkpoint_sample ? std::optional{checkpoint_sample->ckpt} : std::nullopt; const auto [authority_it, inserted] = recovery_authorities.emplace( @@ -627,12 +629,12 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// it (`chain-broken`) rather than the downstream `CORRUPTED_DATA` the replay below would /// raise about the same hole. checkRefStream( - backend, layout, life, walk_life.catalog_entry, checkpoint_sample, deadline, report, verdicts); + op, layout, life, walk_life.catalog_entry, checkpoint_sample, deadline, report, verdicts); /// This recovery's finite range comes from the original catalog row and exact `_ckpt`, never /// from a stream listing, a self-resolved name, or an F+1 probe. const RefTableState table = recoverRefTableDetailedFromAuthority( - backend, layout, authority_it->second.catalog_entry, authority_it->second.checkpoint).state; + op, layout, authority_it->second.catalog_entry, authority_it->second.checkpoint).state; for (const auto [ref_name, row] : table.getCommitted()) { const ManifestId id{ns, row.manifest_ref}; @@ -640,7 +642,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co owned_manifest_keys.insert(mkey); const String label = ns_str + "/" + ref_name; - const auto got = backend.get(mkey); + const auto got = op.read(mkey, Retry::standard()); if (!got) { /// A committed ref naming a missing manifest body would be an INV-NO-DANGLE violation — @@ -651,8 +653,8 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// plus a fresh checkpoint from its physical life. Count the dangle ONLY when the exact /// object is HEAD-absent AND that life still names THIS exact manifest — otherwise it is /// LIST/GET lag or a phantom stale-row, never a loss. - if (!backend.head(mkey).exists - && manifestStillReferenced(backend, layout, ns, recovery_authorities, ref_name, mkey, + if (!op.head(mkey, Retry::standard()) + && manifestStillReferenced(op, layout, ns, recovery_authorities, ref_name, mkey, deadline, record_recovery_unchecked)) { ++report.dangling; @@ -725,7 +727,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// misread as an unreferenced blob and fall into the dangling/pending/unaccounted pipeline /// below), and a body must never be misread as a `.meta`. std::unordered_map present_all; - listAll(backend, layout.blobsPrefix(), present_all, on_progress, deadline, "listing blobs"); + listAll(op, layout.blobsPrefix(), present_all, on_progress, deadline, "listing blobs"); std::unordered_map present_blobs; std::unordered_set present_meta_hashes; present_blobs.reserve(present_all.size()); @@ -751,12 +753,11 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co uint64_t size = exists ? it->second : 0; if (!exists) { - const HeadResult h = backend.head(bkey); - if (h.exists) + if (const std::optional h = op.head(bkey, Retry::standard())) { exists = true; - size = h.size; - report.physical_bytes += h.size; + size = h->size; + report.physical_bytes += h->size; } } @@ -766,7 +767,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// Before declaring a loss, re-resolve the referencing refs from the original audit /// authority. A later rebirth must not replace the old owner while this verdict is being /// decided. - const bool still_referenced = blobStillReferenced(store, layout, recovery_authorities, bkey, + const bool still_referenced = blobStillReferenced(op, store, layout, recovery_authorities, bkey, lit != blob_labels.end() ? lit->second : std::vector{}, deadline, record_recovery_unchecked); if (!still_referenced) @@ -800,7 +801,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// The NON-SENTINEL source edges the snapshot still holds on each unreferenced blob, collected in /// `detail` mode only. `in_run_hashes` alone answers "does GC still see this blob at all"; the /// stale-edge cross-check below needs the edge IDENTITIES so it can ask whether their source - /// manifests still exist. Sentinel rows (`source_id == 0` — `kZeroMarker`/`kCondemned`) are not + /// manifests still exist. Sentinel rows (`source_id == 0` — `RunMarker::Zero`/`RunMarker::Condemned`) are not /// edges and are excluded. std::unordered_map, BlobRefHash> unref_edge_sources; bool have_gc_state = false; @@ -814,15 +815,15 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co if (!unref_hashes.empty()) { - if (const auto state_got = backend.get(layout.gcStateKey())) + if (const auto state_got = op.read(layout.gcStateKey(), Retry::standard())) { have_gc_state = true; const GcState gc_state = decodeGcState(state_got->bytes); /// The adopted fold seal names the snapshot runs; resolution is by ref, never by key /// construction. Every row whose hash is in our candidate set marks "known to GC" — /// edges still counted (drop unfolded), an explicit zero-marker mid-pipeline, or a - /// `kCondemned` sentinel row that carries the condemned state (retired-in-snapshot): - /// the `kCondemned` rows feed `retired_by_hash` (the `PendingGc` classification) in the + /// `RunMarker::Condemned` sentinel row that carries the condemned state (retired-in-snapshot): + /// the `RunMarker::Condemned` rows feed `retired_by_hash` (the `PendingGc` classification) in the /// SAME pass, replacing the removed `retired_refs`/`decodeRetiredSet` loop. /// /// These sets are keyed by the full `BlobRef`, not a narrowed digest. The run's own @@ -830,7 +831,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// the full identity parsed from the listed blob key. This is required for mixed-algorithm /// pools: a 64-hex digest must not be truncated or compared as though it used the pool's /// local write algorithm, or its true GC state could be hidden as `Unaccounted`. - if (const auto seal_got = backend.get(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt))) + if (const auto seal_got = op.read(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt), Retry::standard())) { uint64_t rows = 0; for (const RunRef & run : decodeFoldSeal(seal_got->bytes, gc_state.snap_generation).blob_target_runs) @@ -839,7 +840,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// Typed open: the source-edge run reader goes through openSourceEdgeRun (the NDJSON /// header gates type == cas_run + kind == source_edge). Fsck keys off the row's hash /// (the record's own algo-prefixed key, never from pool meta). - SourceEdgeRunView reader = openSourceEdgeRun(backend, run.key); + SourceEdgeRunView reader = openSourceEdgeRun(op, run.key); String key; String payload; while (reader.next(key, payload)) @@ -852,7 +853,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co in_run_hashes.insert(ref); if (detail && source_id != UInt128{0}) unref_edge_sources[ref].push_back(source_id); - if (!payload.empty() && payload[0] == kCondemned) + if (!payload.empty() && runMarkerFromByte(payload[0], "CAS source-edge run") == RunMarker::Condemned) { const CondemnedRow row = decodeCondemnedRow(payload); RetiredEntry e; @@ -911,7 +912,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co { const RootNamespace ns{ns_str}; std::unordered_map manifest_bodies; - listAll(backend, layout.manifestNamespacePrefix(ns), manifest_bodies, on_progress, deadline, + listAll(op, layout.manifestNamespacePrefix(ns), manifest_bodies, on_progress, deadline, "listing manifests for the stale-edge check"); for (const auto & [mkey, _] : manifest_bodies) { @@ -919,9 +920,9 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co const std::optional id = layout.parseManifestKey(mkey); if (!id) continue; /// foreign/malformed key under `manifests/` — contributes no source edge - const auto got = backend.get(mkey); + const auto got = op.read(mkey, Retry::standard()); if (!got) - continue; /// gone between the LIST and the GET — genuinely not a live source + continue; /// gone between the LIST and the read — genuinely not a live source try { const PartManifest body = decodePartManifest(openObject(FormatId::PartManifest, got->bytes)); @@ -954,11 +955,14 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co FsckClass cls = FsckClass::Unaccounted; String note; - if (const auto rit = retired_by_hash.find(hash); rit != retired_by_hash.end() - && backend.head(bkey).token == rit->second.token) + const auto rit = retired_by_hash.find(hash); + /// HEAD only for a hash the snapshot actually retired, so an unretired blob still costs no request. + const std::optional retired_head + = rit != retired_by_hash.end() ? op.head(bkey, Retry::standard()) : std::nullopt; + if (retired_head && rit->second.token.matches(retired_head->etag)) { - /// The PRESENT incarnation is the condemned one — deletion is scheduled. A token - /// mismatch means the listed entry belongs to a displaced older incarnation and says + /// The PRESENT incarnation is the condemned one — deletion is scheduled. A mismatch + /// means the listed entry belongs to a displaced older incarnation and says /// nothing about this object; fall through to the snapshot check. cls = FsckClass::PendingGc; note = rit->second.delete_pending @@ -1053,13 +1057,13 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co for (const String & bkey : reachable_blobs) { checkDeadline(deadline, "head-checking scoped blobs"); - const HeadResult h = backend.head(bkey); + const std::optional h = op.head(bkey, Retry::standard()); const auto lit = blob_labels.find(bkey); - bool exists = h.exists; + const bool exists = h.has_value(); if (!exists) { /// Use the same HEAD-absent re-resolve as the global-mode loop above. - const bool still_referenced = blobStillReferenced(store, layout, recovery_authorities, bkey, + const bool still_referenced = blobStillReferenced(op, store, layout, recovery_authorities, bkey, lit != blob_labels.end() ? lit->second : std::vector{}, deadline, record_recovery_unchecked); if (!still_referenced) @@ -1068,7 +1072,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co if (exists) { ++report.reachable; - report.physical_bytes += h.size; + report.physical_bytes += h->size; } else ++report.dangling; @@ -1077,7 +1081,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co FsckObject o; o.key = bkey; o.kind = ObjectKind::Blob; - o.size = exists ? h.size : 0; + o.size = exists ? h->size : 0; o.cls = exists ? FsckClass::Reachable : FsckClass::Dangling; if (detail && lit != blob_labels.end()) o.reachable_from = lit->second; @@ -1096,7 +1100,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co const RootNamespace ns{ns_str}; const String manifests_prefix = layout.manifestNamespacePrefix(ns); std::unordered_map manifest_bodies; - listAll(backend, manifests_prefix, manifest_bodies, on_progress, deadline, "listing manifests"); + listAll(op, manifests_prefix, manifest_bodies, on_progress, deadline, "listing manifests"); for (const auto & [mkey, sz] : manifest_bodies) { if (owned_manifest_keys.contains(mkey)) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp index 1428be282f30..0d37a5e0ba9a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -8,7 +9,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -126,20 +129,10 @@ String renderRefTxnIdObj(const RefTxnId & id) .str(); } -String refOwnerKindName(RefOwnerKind k) -{ - switch (k) - { - case RefOwnerKind::Committed: return "Committed"; - case RefOwnerKind::Precommit: return "Precommit"; - } - return "Unknown"; -} - String renderRefOwnerBinding(const RefOwnerBinding & b) { return JsonObj() - .add("kind", jsonEscape(refOwnerKindName(b.kind))) + .add("kind", jsonEscape(refOwnerKindToWord(b.kind))) .add("ref_name", jsonEscape(b.ref_name)) .add("manifest_ref", renderManifestRef(b.manifest_ref)) .str(); @@ -175,7 +168,7 @@ String renderRefTableSnapshot(const RefTableSnapshot & s) .str(); } -/// The namespace's checkpoint (spec INV-4). Every field is optional and each absence means something +/// The namespace's checkpoint. Every field is optional and each absence means something /// different an operator needs to see: no `life_epoch` means no writer that knew this namespace's /// genesis epoch has written here yet, no `committed_through` means the life has no committed /// transaction, no `checkpoint_snapshot_id` means recovery has no snapshot base, and no @@ -196,23 +189,10 @@ String renderRefCkpt(const RootNamespace & ns, const RefCkpt & c) .str(); } -String refOpKindName(RefOpKind k) -{ - switch (k) - { - case RefOpKind::NamespaceBirth: return "NamespaceBirth"; - case RefOpKind::OwnerTransition: return "OwnerTransition"; - case RefOpKind::SetPublishedAt: return "SetPublishedAt"; - case RefOpKind::RemoveNamespace: return "RemoveNamespace"; - case RefOpKind::EpochSeal: return "EpochSeal"; - } - return "Unknown"; -} - String renderRefOp(const RefOp & op) { return JsonObj() - .add("kind", jsonEscape(refOpKindName(op.kind))) + .add("kind", jsonEscape(refOpKindToWireWord(op.kind))) .add("old_binding", op.old_binding ? renderRefOwnerBinding(*op.old_binding) : "null") .add("new_binding", op.new_binding ? renderRefOwnerBinding(*op.new_binding) : "null") .add("ref_name", jsonEscape(op.ref_name)) @@ -237,16 +217,6 @@ String renderRefLogTxn(const RefLogTxn & t) .str(); } -String placementName(EntryPlacement p) -{ - switch (p) - { - case EntryPlacement::Inline: return "Inline"; - case EntryPlacement::Blob: return "Blob"; - } - return "Unknown"; -} - /// `inline_bytes` renders as its LENGTH only, not its content — an inline file's bytes are payload /// data, not part-manifest identity, and may be arbitrarily large / non-UTF8. String renderManifestEntry(const ManifestEntry & e) @@ -256,7 +226,7 @@ String renderManifestEntry(const ManifestEntry & e) /// digest widths, and each entry's own `ref.algo` determines its width. return JsonObj() .add("path", jsonEscape(e.path)) - .add("placement", jsonEscape(placementName(e.placement))) + .add("placement", jsonEscape(entryPlacementToWireWord(e.placement))) .add("blob", jsonEscape(blobIdOf(e.ref))) .add("blob_size", jsonUInt(e.blob_size)) .add("inline_bytes_size", jsonUInt(e.inline_bytes.size())) @@ -288,7 +258,7 @@ String renderMountLease(const MountLease & m) .add("started_at_ms", jsonUInt(m.started_at_ms)) .add("seq", jsonUInt(m.seq)) .add("expires_at_ms", jsonUInt(m.expires_at_ms)) - .add("min_active", jsonUInt(m.min_active)) + .add("min_active_build_sequence", jsonUInt(m.min_active_build_sequence)) .add("gc_fenced", jsonBool(m.gc_fenced)) .add("write_attempt_id", jsonHex(m.write_attempt_id)) .str(); @@ -315,50 +285,31 @@ String renderGcState(const GcState & s) .str(); } -String tokenTypeName(TokenType t) -{ - switch (t) - { - case TokenType::ETag: return "ETag"; - case TokenType::Generation: return "Generation"; - case TokenType::Emulated: return "Emulated"; - } - return "Unknown"; -} - -/// `Token::value` is an opaque backend-native string (e.g. an S3 ETag) — NOT a 128-bit hash — so it -/// renders verbatim (escaped), not hex-converted; `type` names which backend family minted it. -String renderToken(const Token & t) +/// A recorded incarnation's value is an opaque backend-native string (e.g. an S3 ETag) — NOT a +/// 128-bit hash — so it renders verbatim (escaped), not hex-converted; `type` is the dialect word, +/// naming which backend family minted it. +String renderPersistedEtag(const PersistedEtag & inc) { return JsonObj() - .add("value", jsonEscape(t.value)) - .add("type", jsonEscape(tokenTypeName(t.type))) + .add("value", jsonEscape(inc.value)) + .add("type", jsonEscape(inc.dialect)) .str(); } -String objectKindName(ObjectKind k) -{ - switch (k) - { - case ObjectKind::Blob: return "Blob"; - } - return "Unknown"; -} - String renderRunRef(const RunRef & r) { return JsonObj() .add("key", jsonEscape(r.key)) .add("checksum", jsonHex(r.checksum)) .add("shard", jsonUInt(r.shard)) - .add("generation", jsonUInt(r.generation)) + .add("generation", jsonUInt(r.key_generation)) .str(); } String renderRefCoverage(const RefCoverage & c) { return JsonObj() - .add("classification", jsonUInt(c.classification)) + .add("classification", jsonEscape(coverageClassToWord(c.classification))) .add("last_folded_ref_id", renderRefTxnIdObj(c.last_folded_ref_id)) .str(); } @@ -379,7 +330,7 @@ String renderFoldSeal(const CasFoldSeal & seal) for (const auto & r : seal.blob_target_runs) blob_target_runs.push_back(renderRunRef(r)); - /// A fold seal carries per-GC-shard totals for `kCondemned` rows in its source runs. Render the + /// A fold seal carries per-GC-shard totals for `RunMarker::Condemned` rows in its source runs. Render the /// summary from the seal itself; the older separate retired-reference object is no longer part /// of the current layout. JsonObj condemned_summary; @@ -399,40 +350,16 @@ String renderFoldSeal(const CasFoldSeal & seal) .str(); } -String provenanceOpName(ProvenanceOp op) -{ - switch (op) - { - case ProvenanceOp::Other: return "Other"; - case ProvenanceOp::Insert: return "Insert"; - case ProvenanceOp::Merge: return "Merge"; - case ProvenanceOp::Mutation: return "Mutation"; - case ProvenanceOp::Attach: return "Attach"; - case ProvenanceOp::Repack: return "Repack"; - } - return "Unknown"; -} - String renderProvenance(const Provenance & p) { return JsonObj() .add("created_at_ms", jsonUInt(p.created_at_ms)) .add("creator_server_id", jsonHex(p.creator_server_id)) .add("ch_version", jsonUInt(p.ch_version)) - .add("op", jsonEscape(provenanceOpName(p.op))) + .add("op", jsonEscape(provenanceOpToWireWord(p.op))) .str(); } -String metaStateName(MetaState s) -{ - switch (s) - { - case MetaState::Clean: return "clean"; - case MetaState::Condemned: return "condemned"; - } - return "unknown"; -} - /// The per-hash `.meta` descriptor is the blob body's sibling and records its freshness state /// (`Clean` or `Condemned`), not its payload. It is rendered separately from `renderEnvelopeHeader`: /// the body remains an enveloped object, while the descriptor has its own format. @@ -441,7 +368,7 @@ String renderBlobMeta(const BlobMeta & m) return JsonObj() .add("object", jsonEscape("blob_meta")) .add("version", jsonUInt(m.version)) - .add("state", jsonEscape(metaStateName(m.state))) + .add("state", jsonEscape(metaStateToWireWord(m.state))) .add("condemn_round", jsonUInt(m.condemn_round)) .add("size", jsonUInt(m.size)) .str(); @@ -450,9 +377,9 @@ String renderBlobMeta(const BlobMeta & m) String renderEnvelopeHeader(const EnvelopeHeader & h) { return JsonObj() - .add("kind", jsonEscape(objectKindName(h.kind))) + .add("kind", jsonEscape(objectKindToWord(h.kind))) /// The blob identity is carried by the object key, so the envelope keeps only the provenance - /// fields needed for forensics (`ch` and `bld`) together with its compatibility version. + /// fields needed for forensics (`chver` and `build`) together with its compatibility version. .add("compatibility_version", jsonUInt(h.compatibility_version)) .add("incarnation_tag", jsonHex(h.incarnation_tag)) .add("build_id", jsonHex(h.build_id)) @@ -462,25 +389,19 @@ String renderEnvelopeHeader(const EnvelopeHeader & h) .str(); } -/// The word vocabulary a row's marker byte renders as, matching the `cas_run` NDJSON's own `"m"` field -/// words (`CasRecordStreamFormat.cpp`'s private `markerToWord`) so cas-inspect speaks the same vocabulary +/// The word vocabulary a row's marker byte renders as, matching the `cas_run` NDJSON's own `mark` field +/// words (`runMarkerToWireWord`) so cas-inspect speaks the same vocabulary /// as the on-disk format rather than inventing a second one. -String sourceEdgeRowKindName(char marker) +String sourceEdgeRowKindName(RunMarker marker) { - switch (marker) - { - case kEdgeActive: return "edge"; - case kZeroMarker: return "zero"; - case kCondemned: return "condemned"; - default: return "unknown"; - } + return String(runMarkerToWireWord(marker)); } String renderCondemnedRow(const CondemnedRow & r) { return JsonObj() .add("delete_pending", jsonBool(r.delete_pending)) - .add("token", renderToken(r.token)) + .add("token", renderPersistedEtag(r.token)) .add("size", jsonUInt(r.size)) .add("condemn_round", jsonUInt(r.condemn_round)) .add("marker_confirmed", jsonBool(r.marker_confirmed)) @@ -514,7 +435,7 @@ String renderBlobTargetRun(const ParsedBlobTargetRunKey & parsed, std::string_vi if (payload.empty()) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "cas-inspect: source-edge run row for blob {} has an empty payload", blobIdOf(ref)); - const char marker = payload[0]; + const RunMarker marker = runMarkerFromByte(payload[0], "cas-inspect: source-edge run row"); distinct_blobs.insert(ref); JsonObj row; @@ -527,20 +448,16 @@ String renderBlobTargetRun(const ParsedBlobTargetRunKey & parsed, std::string_vi switch (marker) { - case kEdgeActive: + case RunMarker::Edge: ++edge_count; break; - case kZeroMarker: + case RunMarker::Zero: ++zero_marker_count; break; - case kCondemned: + case RunMarker::Condemned: ++condemned_count; row.add("condemned", renderCondemnedRow(decodeCondemnedRow(payload))); // CORRUPTED_DATA on malformed (fail-closed) break; - default: - throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, - "cas-inspect: source-edge run row for blob {} has an unknown marker 0x{:02x}", - blobIdOf(ref), static_cast(marker)); } rows.push_back(row.str()); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h index 0c6bfa3e0cee..7d76f6250de5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasInspect.h @@ -17,8 +17,8 @@ namespace DB::Cas /// `cas/ns/stream/` and `cas/ns/state/` roots, `/mount` and `/fold_seal` suffixes, the /// `gc/gen/*/attempt/*/blob_target/*/*` source-edge run segments, then the pool-wide `gc/state` /// and `blobs/` prefix). u128 and hash fields render as lowercase hex strings (matching -/// `u128ToHex`), while backend-native `Token` values render as escaped strings. Neither is exposed -/// as an array of bytes or a raw struct dump. +/// `u128ToHex`), while a recorded incarnation's backend-native value renders as an escaped string +/// beside its dialect word. Neither is exposed as an array of bytes or a raw struct dump. /// /// Throws `ErrorCodes::BAD_ARGUMENTS` when `key` matches none of the recognized CA layouts. Any /// decode failure of a matched key (invalid header, corrupted bytes, future format version, ...) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp index f23f4dc06bae..34bde37517dd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/benchmarks/benchmark_cas_ref_protocol.cpp @@ -1,15 +1,38 @@ #include #include +#include +#include +#include #include +#include + #include +#include +#include #include #include +#include +#include +#include +#include +#include #include +#include +#include +#include +#include #include #include +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int LIMIT_EXCEEDED; + extern const int LOGICAL_ERROR; +} + /// Pure measurement, no pass/fail assertions -- see the cas-gc-rebuild BACKLOG.md entries /// "OPTIMIZATION OPPORTUNITY -- ref-ledger JSON encoding writes byte-by-byte" and the (now /// RESOLVED) "admits() re-encodes the WHOLE ref table once per state-growing op" entry for the @@ -38,6 +61,16 @@ /// ladder, rung 2 was NOT attempted either (it trades readability and needs a human decision); /// reported as DONE_WITH_CONCERNS. CasEncodingPins.* stayed byte-identical (green) throughout. /// +/// NOTE (2026-08, wire-key-rename campaign): the "Phase B baselines" table immediately below measures +/// a DIFFERENT investigation (the `RefTableState` encapsulation refactor) and predates the five-format, +/// both-directions wire-key-cut design entirely. It is NOT the "before" side for that campaign's +/// measurement, and a later reader must not diff against it for that purpose. The actual before side +/// is the pre-cut worktree pinned at commit `65ec8688cdb`; the recorded patch that builds this file +/// there lives under `docs/superpowers/cas/bench-wire-keys-phase3/`. The table is kept exactly as +/// written because it is real history for the investigation it belongs to, not because it answers this +/// one -- see the "Wire-key-cut instrument" section further down for the five new formats this +/// campaign added. +/// /// Phase B baselines, 2026-07-21, pre-encapsulation (this binary; `--benchmark_repetitions=3 /// --benchmark_report_aggregates_only=true`; medians reported). Recorded ahead of the /// `RefTableState` encapsulation refactor so later phases can re-run this exact suite unchanged and @@ -118,6 +151,15 @@ RefLogTxn makeSamplePromoteTxn() /// A synthetic snapshot of `n` committed rows plus one pending precommit ready to promote. /// Built as a RefTableSnapshot and materialized via the public `replay` entry point, so this /// helper keeps compiling unchanged when RefTableState's fields become private (Phase A). +/// +/// Committed-row field widths (load-bearing for the `cas_ref_snap` wire-key-cut benchmarks and byte +/// oracle, which measure the RELATIVE cost of a key rename against the encoded VALUE bytes as the +/// denominator): `published_at_ms` is a real 13-digit epoch-ms rather than the default `0`, and +/// `manifest_ref`'s `writer_epoch`/`build_sequence` are multi-digit (a pool old enough to have +/// restarted its writer decades of times, and a build counter past its 89811th commit -- the same +/// order of magnitude as the real ref-ledger key at the top of this file, `kSafeKeyLikeString`, and +/// `makeSamplePromoteTxn`'s ref name). A minimal `0`/`1`/`1` shrinks the value-byte denominator a key +/// rename is measured against and inflates the rename's apparent percentage cost. RefTableSnapshot makeSyntheticSnapshot(size_t n) { RefTableSnapshot snapshot; @@ -127,7 +169,8 @@ RefTableSnapshot makeSyntheticSnapshot(size_t n) { RefCommittedRow row; row.ref_name = "part_" + std::to_string(i) + "_20260719_0_1000_1"; - row.manifest_ref = ManifestRef{1, 1, static_cast(i + 1)}; + row.manifest_ref = ManifestRef{42, 89811 + static_cast(i), static_cast(i + 1)}; + row.published_at_ms = 1752900000000ULL + i; snapshot.committed.push_back(row); } std::sort(snapshot.committed.begin(), snapshot.committed.end(), @@ -550,4 +593,485 @@ static void BM_Materialize(benchmark::State & state) } BENCHMARK(BM_Materialize)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); -BENCHMARK_MAIN(); +/// ------------------------------------------------------------------------------------------------- +/// Wire-key-cut instrument (Task 7): encode AND decode for the five formats the campaign's wire-key +/// rename touched most (`cas_run`, `cas_ref_snap`, `cas_part_manifest`, `cas_fold_seal`, +/// `cas_ref_catalog`), plus a byte/cap oracle (below `reportFormatCaps`). This section only BUILDS the +/// instrument -- it does not take the before/after measurement itself, which is a later task run +/// against this same binary built on both sides of the cut. The "before" side is the pre-cut worktree +/// at `/home/mfilimonov/workspace/ClickHouse/cas-p2-before`, pinned at commit `65ec8688cdb`; the +/// recorded patch that adapts this file's one incompatible call site (`foldedClassification`/ +/// `clampedClassification` below) for that build lives under +/// `docs/superpowers/cas/bench-wire-keys-phase3/`. Every other line in this section is byte-identical +/// on both sides -- confirmed against the before-side headers, which differ from these only in +/// comment text (the retired terse wire spellings) and in `RefCoverage::classification`'s type. +/// ------------------------------------------------------------------------------------------------- + +namespace +{ + +/// `cas_run` fixture: `n` distinct blobs in strictly ascending digest order (`SourceEdgeRunWriter` +/// requires non-decreasing `(ref, source_id)` keys, and a monotonically increasing digest alone +/// satisfies that regardless of `source_id`). Marker mix models one healthy in-degree run: the +/// overwhelming majority of tracked blobs simply carry a live edge this generation (98% `Edge`); a +/// blob losing its LAST edge (`Zero`) or actually condemned for deletion (`Condemned`, carrying the +/// full retired-incarnation token) is comparatively rare at any one round -- 1% each here, not 0 and +/// not half. `source_id` is a synthetic per-record counter rather than a real backend id: the codec's +/// cost is driven by the DIGEST's hex width, not the id's numeric value. The condemned token mirrors a +/// real S3 ETag's width (a quoted 32-hex value) and `size` a realistic single-blob byte count (64 KiB, +/// a typical compressed column chunk). Record count ranges 100 to 100,000 (`RangeMultiplier(10)`), +/// matching every `Complexity()` benchmark already in this file. +std::vector makeSourceEdgeRecords(size_t n) +{ + std::vector records; + records.reserve(n); + for (size_t i = 0; i < n; ++i) + { + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(i + 1))}; + SourceEdgeRecord rec; + rec.ref = ref; + if (i % 100 == 0) + { + rec.source_id = UInt128(0); + rec.marker = RunMarker::Condemned; + rec.delete_pending = (i % 200 == 0); + rec.token = PersistedEtag{"etag", "\"e1b2c3d4e5f6071829300a0b0c0d0e0f\""}; + rec.size = 64 * 1024; + rec.condemn_round = 7; + } + else if (i % 100 == 50) + { + rec.source_id = UInt128(0); + rec.marker = RunMarker::Zero; + } + else + { + rec.source_id = UInt128(i + 1); + rec.marker = RunMarker::Edge; + } + records.push_back(rec); + } + return records; +} + +/// Runs the real `SourceEdgeRunWriter` over `records`, exactly as `CASRecordStream`'s own +/// `encodeRun` helper does -- so decode below always consumes real encoder output, never a +/// hand-built string. +String encodeSourceEdgeRun(const std::vector & records) +{ + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + for (const auto & r : records) + writer.append(r); + writer.finish(); + /// `str()` returns a `std::string &`, so returning it plainly would copy-construct the whole + /// encoded run on every call (no NRVO is available for a reference) -- `std::move` here moves it + /// instead, matching the four other encoders, which all end with `std::move(out).take()` and copy + /// nothing. `str()` finalizes `out` itself, so no separate `finalize()` call is needed first. + return std::move(out.str()); +} + +/// The ONE call site whose TYPE differs across the wire-key-rename cut this benchmark spans: on this +/// (AFTER) side `RefCoverage::classification` is the closed `CoverageClass` enum; at the pre-cut +/// commit it is a raw `uint8_t` whose CLAMPED value is ALSO renumbered (4 there, 3 here -- see +/// `CasFoldSealFormat.h`'s own history comment on `CoverageClass`). A bare numeric literal at the call +/// site would therefore silently measure the WRONG row shape on the before-side build, so the Step-4 +/// patch touches only this pair of one-line functions; every benchmark body in this file stays +/// byte-identical on both sides. +CoverageClass foldedClassification() { return CoverageClass::Folded; } +CoverageClass clampedClassification() { return CoverageClass::Clamped; } + +/// Not the record axis under test (`n` below is `ref_lives` row count): fixed at a representative +/// multi-shard pool size. A single-shard fixture would fold `blob_target_runs`/`condemned_summary` to +/// one degenerate entry each, understating the per-shard fan-out a real multi-shard pool carries in +/// both sections. +constexpr uint64_t kFoldSealGcShards = 4; + +/// `cas_fold_seal` fixture: `n` `ref_lives` rows keyed by ascending life id, split base/hold-bearing/ +/// cleanup-evidence 90%/5%/5%. Per the spec's byte table, a hold-bearing row adds 33 bytes and a +/// cleanup-evidence row adds 16 bytes over a base row's 22-plus-class-word bytes; at this 90/5/5 mix +/// the recovered uplift over an all-base fixture is 0.05*33 + 0.05*16 = 2.45 bytes/row, about 8% over +/// a base row's own ~30 bytes (the full one-third the spec's deltas imply is the all-clamped extreme, +/// not this mix) -- still enough that omitting the two minority shapes entirely would misstate the +/// row-average cost in the wrong direction. The 90/5/5 split models a healthy pool: most namespaces +/// fold cleanly every round (base: `Folded`, no hold, no cleanup evidence); a minority sit behind a +/// transient barrier (hold-bearing: `Clamped`, `ManifestBodyMissing`); a minority are mid-teardown +/// (cleanup evidence: `Folded` plus a terminal `remove_namespace` fold). Neither minority shape is the +/// common case, but neither is negligible either -- both recur every round in a live pool. +/// `RefTxnId` epoch/sequence pairs and the hold's `retry_count`/`next_retry_round` are multi-digit +/// (a pool old enough to have restarted its writer dozens of times and folded past its 100,000th +/// ref-log transaction; a hold retried past its first round but nowhere near abandoned) rather than +/// the single-digit illustrative values the spec's byte table uses to name the three row SHAPES -- +/// matching the shapes, not the spec table's example digits, is what keeps the value-byte denominator +/// realistic (see `makeSyntheticSnapshot`'s doc comment for why that denominator matters). Record +/// count ranges 100 to 100,000, matching every `Complexity()` benchmark in this file. +CasFoldSeal makeFoldSeal(size_t n) +{ + CasFoldSeal seal; + seal.generation = 7; + seal.parent_generation = 6; + for (size_t i = 0; i < n; ++i) + { + RefLifeFoldState row; + if (i % 20 == 0) + { + row.coverage = RefCoverage{ + .classification = clampedClassification(), + .last_folded_ref_id = RefTxnId{42, 103482}, + .hold = RefHold{ + .reason = HoldReason::ManifestBodyMissing, + .offending_position = RefTxnId{42, 103500}, + .retry_count = 14, + .next_retry_round = 1042}}; + } + else if (i % 20 == 1) + { + row.coverage = RefCoverage{.classification = foldedClassification(), .last_folded_ref_id = RefTxnId{42, 118203}}; + row.cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{42, 118190}}; + } + else + { + row.coverage = RefCoverage{.classification = foldedClassification(), .last_folded_ref_id = RefTxnId{42, 100123}}; + } + seal.ref_lives.emplace(UInt128(i + 1), std::move(row)); + } + for (uint64_t shard = 0; shard < kFoldSealGcShards; ++shard) + { + seal.blob_target_runs.push_back(RunRef{ + .key = fmt::format("p/gc/gen/7/attempt/1/blob_target/{}/0", shard), + .checksum = UInt128(0x1000 + shard), .shard = shard, .key_generation = 7}); + seal.condemned_summary[shard] = CondemnedSummary{ + .condemned_total = 1000 + shard, .pending_total = 10 + shard, .oldest_nonpending_condemn_round = 4}; + } + return seal; +} + +/// `cas_part_manifest` fixture: `n` entries in path order, 90% `Blob` (the column/mark/index files +/// that dominate a real MergeTree part) and every 10th `Inline` (small metadata files like +/// `count.txt`/`checksums.txt` that get embedded rather than stored as a separate blob). Blob sizes +/// cycle 4-64 KiB across 16 steps to resemble the spread of real column-chunk sizes rather than one +/// repeated constant; inline bytes are a fixed 48-byte payload, resembling a small metadata file. +/// `ref`/`root_namespace_id` are fixed -- they do not scale with entry count in a real manifest +/// either. `encodePartManifest` sorts entries itself, so input order need not be canonical. Record +/// count ranges 100 to 100,000, matching every `Complexity()` benchmark in this file (a real part +/// rarely reaches the top of that range; it stress-tests a pathologically wide/many-column part). +PartManifest makePartManifest(size_t n) +{ + PartManifest m; + m.ref = ManifestRef{5, 15, 1}; + m.root_namespace_id = RootNamespace("00/aa@cas@"); + m.entries.reserve(n); + for (size_t i = 0; i < n; ++i) + { + ManifestEntry e; + if (i % 10 == 9) + { + e.path = fmt::format("{:06}_meta.txt", i); + e.placement = EntryPlacement::Inline; + e.inline_bytes = String(48, 'x'); + } + else + { + e.path = fmt::format("{:06}_data.bin", i); + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(i + 1))}; + e.blob_size = 4096 * (1 + (i % 16)); + } + m.entries.push_back(std::move(e)); + } + m.payload_digest = computePayloadDigest(m); + return m; +} + +/// `cas_ref_catalog` fixture: `n` entries in ascending namespace order (a 7-digit zero-padded ordinal +/// keeps ascending lexical order across the whole 100..100,000 range, well under `kMaxNamespaceBytes`). +/// The mix resembles one whole-pool catalog snapshot: most namespaces are simply `Live` (96%), with a +/// small steady trickle of admission (`Creating`, 2%) and teardown (`Removing`, 2%) in flight at any +/// moment -- neither churn state is the common case, but neither is negligible either. Record count +/// ranges 100 to 100,000, matching every `Complexity()` benchmark in this file. +RefCatalog makeRefCatalog(size_t n) +{ + RefCatalog catalog; + catalog.entries.reserve(n); + for (size_t i = 0; i < n; ++i) + { + CatalogEntry e; + e.ns = RootNamespace(fmt::format("roots/ca_tbl_{:07}", i)); + e.incarnation = UInt128(i + 1); + if (i % 50 == 0) + { + e.state = NsState::Creating; + e.creator = CreatorFence{"srv-bench", 1, 1}; + } + else if (i % 50 == 25) + { + e.state = NsState::Removing; + e.removal_started_round = 42; + } + else + { + e.state = NsState::Live; + } + catalog.entries.push_back(std::move(e)); + } + return catalog; +} + +} + +/// `cas_run` is streamed (`object_cap == 0`; see `CasRecordStreamFormat.h`) and never materialized +/// whole in production, but the benchmark still needs one complete encoded run to time and to decode: +/// `encodeSourceEdgeRun` drives the real `SourceEdgeRunWriter`/`SourceEdgeRunReader` pair over an +/// in-memory buffer, the same pair the streaming production path uses over its own `WriteBuffer`/ +/// `ReadBuffer`. +static void BM_CasRunEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const std::vector records = makeSourceEdgeRecords(n); + for (auto _ : state) + benchmark::DoNotOptimize(encodeSourceEdgeRun(records)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRunEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasRunDecode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const String encoded = encodeSourceEdgeRun(makeSourceEdgeRecords(n)); + for (auto _ : state) + { + DB::ReadBufferFromMemory in(encoded.data(), encoded.size()); + SourceEdgeRunReader reader(in); + SourceEdgeRecord rec; + size_t count = 0; + while (reader.next(rec)) + ++count; + benchmark::DoNotOptimize(count); + } + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRunDecode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +/// `BM_SnapshotEncode` above already exists (for the E4 contiguous-scan investigation) and has no +/// decode counterpart. This pair is the one the wire-key-cut measurement uses: same fixture, but named +/// and shaped to match the other four formats' encode/decode pairs in this section. +static void BM_CasRefSnapEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableSnapshot snapshot = makeSyntheticSnapshot(n); + for (auto _ : state) + benchmark::DoNotOptimize(encodeRefTableSnapshot(snapshot)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRefSnapEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasRefSnapDecode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefTableSnapshot snapshot = makeSyntheticSnapshot(n); + const String encoded = encodeRefTableSnapshot(snapshot); + for (auto _ : state) + benchmark::DoNotOptimize(decodeRefTableSnapshot(encoded, snapshot.ns, snapshot.snapshot_id)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRefSnapDecode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasPartManifestEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const PartManifest m = makePartManifest(n); + for (auto _ : state) + benchmark::DoNotOptimize(encodePartManifest(m)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasPartManifestEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasPartManifestDecode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const PartManifest m = makePartManifest(n); + const String encoded = encodePartManifest(m); + for (auto _ : state) + benchmark::DoNotOptimize(decodePartManifest(encoded)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasPartManifestDecode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasFoldSealEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const CasFoldSeal seal = makeFoldSeal(n); + for (auto _ : state) + benchmark::DoNotOptimize(encodeFoldSeal(seal)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasFoldSealEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasFoldSealDecode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const CasFoldSeal seal = makeFoldSeal(n); + const String encoded = encodeFoldSeal(seal); + for (auto _ : state) + benchmark::DoNotOptimize(decodeFoldSeal(encoded)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasFoldSealDecode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasRefCatalogEncode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefCatalog catalog = makeRefCatalog(n); + for (auto _ : state) + benchmark::DoNotOptimize(encodeRefCatalog(catalog)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRefCatalogEncode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +static void BM_CasRefCatalogDecode(benchmark::State & state) +{ + const size_t n = static_cast(state.range(0)); + const RefCatalog catalog = makeRefCatalog(n); + const String encoded = encodeRefCatalog(catalog); + for (auto _ : state) + benchmark::DoNotOptimize(decodeRefCatalog(encoded)); + state.SetComplexityN(static_cast(n)); +} +BENCHMARK(BM_CasRefCatalogDecode)->RangeMultiplier(10)->Range(100, 100000)->Complexity(); + +namespace +{ + +/// Binary search on record count with the real encode -> `sealObject` -> `openObject` pipeline as the +/// oracle. `openObject` (`CasTextFormat.cpp`) enforces the registry's `object_cap` on BOTH the raw and +/// the zstd-frame-header path, so this one pipeline works whether or not the format compresses; a +/// format's OWN pre-put gate (e.g. fold-seal's `checkFoldSealObjectBytes`) may throw earlier, at the +/// encode step itself. Either `LIMIT_EXCEEDED` or `CORRUPTED_DATA` at this boundary means "does not +/// fit" and steers the search; any other exception is a fixture bug, not a capacity signal, and is +/// left to propagate rather than being misread as "found the cap". +/// +/// `known_fits_n`/`known_fits_bytes` seed the exponential search from a bytes-per-record estimate +/// measured at a small `n` -- NOT a hardcoded delta table -- purely to reduce how many large, +/// expensive encodes the search performs before bisecting. The estimate never becomes the answer: the +/// real encoder confirms every step of both the exponential growth and the final exact bisection. +template +uint64_t maxRecordCountUnderCap(FormatId id, uint64_t known_fits_n, uint64_t known_fits_bytes, Encode encode) +{ + auto fits = [&](uint64_t n) -> bool + { + try + { + const String stored = sealObject(id, encode(n)); + benchmark::DoNotOptimize(openObject(id, stored)); + return true; + } + catch (const DB::Exception & e) + { + if (e.code() == DB::ErrorCodes::CORRUPTED_DATA || e.code() == DB::ErrorCodes::LIMIT_EXCEEDED) + return false; + throw; + } + }; + + /// The bisection below is correct only if `fits(lo) == true`. The caller's `known_fits_n` comes + /// from an encode IT ran itself -- never through `openObject`, which is what actually enforces + /// `object_cap` (the raw-size check, or the zstd frame's declared decompressed size) -- so this + /// verifies the bound directly rather than trusting that claim. If `known_fits_n` itself is + /// already over the cap (e.g. a future, much larger report size), halve downward until a verified + /// fit is found; if even `n == 1` does not fit, that is a fixture/format bug, not a capacity + /// signal, and is raised loudly rather than silently reported as a wrong maximum. + uint64_t lo = known_fits_n; + while (lo > 1 && !fits(lo)) + lo /= 2; + if (!fits(lo)) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, + "maxRecordCountUnderCap: format {} does not fit its object cap even at n=1", static_cast(id)); + + const FormatTraits & traits = traitsFor(id); + const uint64_t per_record = std::max(1, known_fits_bytes / std::max(1, known_fits_n)); + uint64_t hi = std::max(lo * 2, traits.object_cap / per_record); + while (fits(hi)) + { + lo = hi; + hi *= 2; + } + while (hi - lo > 1) + { + const uint64_t mid = lo + (hi - lo) / 2; + (fits(mid) ? lo : hi) = mid; + } + return lo; +} + +/// One format's report line: decompressed bytes at `report_n`, stored bytes under the format's REAL +/// registered compression policy (or `n/a` for a policy that stores raw -- `Never`/`PinnedRaw` -- since +/// there is no separate compressed form to report, and a `0` there would read as a measurement rather +/// than "not applicable"), and the largest record count the real encoder admits under the format's +/// object cap. +template +void reportSealedFormat(std::string_view name, FormatId id, uint64_t report_n, Encode encode) +{ + const FormatTraits & traits = traitsFor(id); + const String decompressed = encode(report_n); + const String stored = sealObject(id, decompressed); + const bool stores_raw = traits.compression != CompressionPolicy::Always; + const uint64_t max_n = maxRecordCountUnderCap(id, report_n, decompressed.size(), encode); + + fmt::print("{:<18} decompressed={:>10} bytes (n={}) stored={} max_n_under_object_cap={}\n", + name, decompressed.size(), report_n, + stores_raw ? "n/a (stored raw, no compression)" : (std::to_string(stored.size()) + " bytes (zstd)"), + max_n); +} + +/// Step 3's byte and cap oracle: a small, main-less, flag-invoked harness (see `main` below) rather +/// than a benchmark or a gtest, so it never engages the timing loop and never needs a second `main` in +/// this binary. Reports, per format, at the stated `kReportN`: decompressed bytes, stored bytes under +/// the real compression policy (or `n/a`), and the maximum record count the real encoder admits under +/// the object cap (or `n/a` where none applies). +void reportFormatCaps() +{ + constexpr uint64_t kReportN = 1000; + fmt::print("=== cas format byte/cap report (n={}) ===\n", kReportN); + + /// `cas_run` is `object_cap == 0` (streamed, `RunFile` family): never materialized whole in + /// production, so there is no whole-object cap to search for and no compressed form to report. + { + const String encoded = encodeSourceEdgeRun(makeSourceEdgeRecords(kReportN)); + fmt::print("{:<18} decompressed={:>10} bytes (n={}) stored=n/a (PinnedRaw, never compressed) " + "max_n_under_object_cap=n/a (object_cap=0: streamed one line at a time, never materialized whole)\n", + "cas_run", encoded.size(), kReportN); + } + + reportSealedFormat("cas_ref_snap", FormatId::RefSnapshot, kReportN, + [](uint64_t n) { return encodeRefTableSnapshot(makeSyntheticSnapshot(n)); }); + reportSealedFormat("cas_part_manifest", FormatId::PartManifest, kReportN, + [](uint64_t n) { return encodePartManifest(makePartManifest(n)); }); + reportSealedFormat("cas_fold_seal", FormatId::FoldSeal, kReportN, + [](uint64_t n) { return encodeFoldSeal(makeFoldSeal(n)); }); + reportSealedFormat("cas_ref_catalog", FormatId::RefCatalog, kReportN, + [](uint64_t n) { return encodeRefCatalog(makeRefCatalog(n)); }); +} + +} + +/// Hand-written in place of `BENCHMARK_MAIN()` so `--report_format_caps` can dispatch to Step 3's +/// oracle BEFORE `benchmark::Initialize` ever sees argv -- keeping the byte/cap report in this same +/// binary without a second `main` or a separate gtest target, and without the report's args tripping +/// `ReportUnrecognizedArguments`. Absent that flag, behavior is exactly `BENCHMARK_MAIN()`'s. +int main(int argc, char ** argv) +{ + for (int i = 1; i < argc; ++i) + { + if (std::string_view(argv[i]) == "--report_format_caps") + { + reportFormatCaps(); + return 0; + } + } + + benchmark::Initialize(&argc, argv); + if (benchmark::ReportUnrecognizedArguments(argc, argv)) + return 1; + benchmark::RunSpecifiedBenchmarks(); + return 0; +} diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h index 88420b30472f..8d8525f4facb 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h @@ -46,6 +46,8 @@ class AzureObjectStorage : public IObjectStorage size_t max_keys, bool with_tags, const std::optional & start_after) const override; + /// Overriding one `iterate` overload hides the other from this class's scope; bring both back. + using IObjectStorage::iterate; std::string getName() const override { return "Azure"; } diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp index 2d4088d813b0..fba16861e8e2 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp @@ -67,6 +67,41 @@ ObjectStorageIteratorPtr IObjectStorage::iterate( return std::make_shared(std::move(files)); } +ObjectStorageIteratorPtr IObjectStorage::iterate( + const std::string & path_prefix, + size_t max_keys, + bool with_tags, + const std::optional & start_after, + const ObjectStorageControlRequest & request) const +{ + if (request.profile == ObjectStorageRetryProfile::SingleAttempt) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support single-attempt listing requests", getName()); + return iterate(path_prefix, max_keys, with_tags, start_after); +} + +std::optional IObjectStorage::tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags, const ObjectStorageControlRequest & request) const +{ + if (request.profile == ObjectStorageRetryProfile::SingleAttempt) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support single-attempt metadata requests", getName()); + return tryGetObjectMetadataWithNativeToken(path, with_tags); +} + +ConditionalRemoveResult IObjectStorage::removeObjectIfTokenMatches( + const StoredObject & object, const std::string & etag, const ObjectStorageControlRequest & request) +{ + if (request.profile == ObjectStorageRetryProfile::SingleAttempt) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support single-attempt removal requests", getName()); + return removeObjectIfTokenMatches(object, etag); +} + +void IObjectStorage::removeObjectsIfExistUnderProfile(const StoredObjects & objects, const ObjectStorageControlRequest & request) +{ + if (request.profile == ObjectStorageRetryProfile::SingleAttempt) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support batch removal under a retry profile", getName()); + removeObjectsIfExist(objects); +} + ThreadPool & IObjectStorage::getThreadPoolWriter() { auto context = Context::getGlobalContextInstance(); diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index ef2aac33c2be..ac4f1625ceb0 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -340,6 +340,17 @@ class IObjectStorage bool with_tags, const std::optional & start_after) const; + /// Same, under a chosen control-request context (retry profile, one attempt's timeout and connect + /// cap, and the caller's own attempt number -- see `ObjectStorageControlRequest`). A storage that + /// cannot execute the profile must refuse: a caller that asked for one attempt has its own + /// deadline, and a transparently retried request would outlive it. + virtual ObjectStorageIteratorPtr iterate( + const std::string & path_prefix, + size_t max_keys, + bool with_tags, + const std::optional & start_after, + const ObjectStorageControlRequest & request) const; + /// Get object metadata if supported. It should be possible to receive at least size of object virtual ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const = 0; virtual ObjectMetadata getObjectMetadata(const RelativePathWithMetadata & object, bool with_tags) const @@ -362,6 +373,10 @@ class IObjectStorage return tryGetObjectMetadata(path, with_tags); } + /// Same, under a chosen control-request context; see the note on `iterate`. + virtual std::optional tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags, const ObjectStorageControlRequest & request) const; + /// Read single object virtual std::unique_ptr readObject( /// NOLINT const StoredObject & object, @@ -428,6 +443,19 @@ class IObjectStorage "Conditional (token-exact) object removal is not implemented for {} object storage", getName()); } + /// Same, under a chosen control-request context; see the note on `iterate`. + virtual ConditionalRemoveResult removeObjectIfTokenMatches( + const StoredObject & object, const std::string & etag, const ObjectStorageControlRequest & request); + + /// Removes every object in ONE request with no per-key precondition; an absent object is success. + /// Content-addressed callers use it for write-once keys only, at most 1000 per call. Throws on a + /// request-level failure and on any per-key error other than "not found", naming the failed keys. + /// Same context note as `iterate`: the default forwards a Default-profile request to + /// `removeObjectsIfExist` and refuses a SingleAttempt one, so a backend without a real batch + /// delete still needs to override this for SingleAttempt to behave correctly under that profile. + virtual void removeObjectsIfExistUnderProfile( + const StoredObjects & objects, const ObjectStorageControlRequest & request); + /// Copy object with different attributes if required virtual void copyObject( /// NOLINT const StoredObject & object_from, @@ -516,8 +544,9 @@ class IObjectStorage virtual void pinConditionalOpsGenerationDialect(bool /*expect_generation_tokens*/) {} /// Whether the underlying bucket has object versioning enabled; nullopt when unknown or not - /// applicable. Used by the CAS capability probe to fail closed on GCS: on a versioned bucket - /// a token-exact DELETE archives a noncurrent generation instead of reclaiming storage. + /// applicable. Used by the CAS capability probe on GCS: a bucket verified as versioned refuses + /// the mount (a token-exact DELETE there archives a noncurrent generation instead of reclaiming + /// storage), an unknown answer is logged and tolerated. virtual std::optional isBucketVersioningEnabled() const { return std::nullopt; } /// True when this object storage can execute writes under the given retry profile. diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index bfe6b7a7b988..876894a755ae 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -40,6 +40,7 @@ #include #include +#include #include #include @@ -81,6 +82,7 @@ namespace S3RequestSetting extern const S3RequestSettingsUInt64 min_upload_part_size; extern const S3RequestSettingsUInt64 max_unexpected_write_error_retries; extern const S3RequestSettingsUInt64 max_single_operation_copy_size; + extern const S3RequestSettingsUInt64 max_single_read_retries; } @@ -94,6 +96,7 @@ namespace S3AuthSetting namespace ErrorCodes { extern const int BAD_ARGUMENTS; + extern const int CANNOT_READ_ALL_DATA; extern const int LOGICAL_ERROR; extern const int NOT_IMPLEMENTED; extern const int S3_ERROR; @@ -142,6 +145,23 @@ void logIfError(const Aws::Utils::Outcome & response, std::functi } } +/// Classifies a per-key error `Code` string from a `DeleteObjects` response body (data the SDK never +/// builds an `Aws::S3::S3Error` for, since the response as a whole was a success) the same way the SDK's +/// own `S3ErrorMarshaller::Marshall` classifies a whole-response error: `Aws::S3::S3ErrorMapper` first +/// (the S3-specific extension names -- `NoSuchKey`, `NoSuchBucket`, ...), falling back to +/// `Aws::Client::CoreErrorsMapper` for a name shared across every AWS service (`AccessDenied`, +/// `InternalError`, ...), which `S3ErrorMapper` alone does not recognize and would otherwise leave +/// classified as `UNKNOWN`. The two mappers' shared codes carry identical numeric values by construction +/// (see the "// From Core//" section of `Aws::S3::S3Errors`), so reinterpreting a `CoreErrors` result as +/// `S3Errors` is exactly what the SDK's own marshaller does. +Aws::S3::S3Errors classifyDeleteObjectsErrorCode(const String & code) +{ + if (const auto s3_specific = Aws::S3::S3ErrorMapper::GetErrorForName(code.c_str()).GetErrorType(); + s3_specific != Aws::Client::CoreErrors::UNKNOWN) + return static_cast(s3_specific); + return static_cast(Aws::Client::CoreErrorsMapper::GetErrorForName(code.c_str()).GetErrorType()); +} + } namespace @@ -156,7 +176,8 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync std::shared_ptr client_, size_t max_list_size, bool with_tags_, - const std::optional & start_after_) + const std::optional & start_after_, + size_t attempt_seed_ = 0) : IObjectStorageIteratorAsync( CurrentMetrics::ObjectStorageS3Threads, CurrentMetrics::ObjectStorageS3ThreadsActive, @@ -166,13 +187,19 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync , request(std::make_unique()) , with_tags(with_tags_) , start_after_set(start_after_.has_value() && !start_after_->empty()) +<<<<<<< HEAD , description(fmt::format("Bucket: {}, Prefix: {}", bucket_, path_prefix)) +======= + , attempt_seed(attempt_seed_) +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) { request->SetBucket(bucket_); request->SetPrefix(path_prefix); request->SetMaxKeys(static_cast(max_list_size)); if (start_after_set) request->SetStartAfter(*start_after_); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(*request, attempt_seed); } ~S3IteratorAsync() override @@ -212,6 +239,8 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync paginated_request->SetPrefix(request->GetPrefix()); paginated_request->SetMaxKeys(request->GetMaxKeys()); paginated_request->SetContinuationToken(next_continuation_token); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(*paginated_request, attempt_seed); request = std::move(paginated_request); start_after_set = false; } @@ -249,11 +278,34 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync std::unique_ptr request; const bool with_tags; bool start_after_set; +<<<<<<< HEAD const std::string description; +======= + const size_t attempt_seed; +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) }; } +template +auto S3ObjectStorage::refreshAndRetryOnExpiredCredentials(Fn && fn) const +{ + try + { + return fn(); + } + catch (const S3Exception & e) + { + if (!e.isAccessTokenExpiredError() || !credentials_refresh_callback) + throw; + auto new_client = credentials_refresh_callback(); + if (!new_client) + throw; + client.set(std::move(new_client)); + return fn(); + } +} + bool S3ObjectStorage::exists(const StoredObject & object) const { auto settings_ptr = s3_settings.get(); @@ -288,8 +340,37 @@ std::unique_ptr S3ObjectStorage::readObject( /// NOLINT blob_storage_log->local_path = object.local_path; } + const bool single_attempt = read_settings.object_storage_retry_profile == ObjectStorageRetryProfile::SingleAttempt; + S3CredentialsRefreshCallback refresh_callback = credentials_refresh_callback; + if (single_attempt) + { + request_settings[S3RequestSetting::max_single_read_retries] = 1; + if (credentials_refresh_callback) + { + /// Captures the client SLOT and a copy of the caller's refresh callback, never this + /// storage's `this`: the buffer this returns can outlive the storage, and a credential + /// expiry firing afterwards would otherwise install a fresh client into a destroyed + /// object. The copied callback can itself capture a shorter-lived object -- + /// `StorageS3Configuration::createObjectStorage`'s refresher captures the configuration + /// it was built from -- so the caller must keep that object alive as long as the buffer. + refresh_callback = [slot = client_slot, refresh = credentials_refresh_callback]() + -> std::unique_ptr + { + auto new_client = refresh(); + if (new_client) + slot->set(std::move(new_client)); + /// The buffer will not reissue this read, so it has no use for a client; refreshing + /// the disk's is what lets the caller's next request sign with the new credentials. + return nullptr; + }; + } + } + return std::make_unique( - client.get(), + clientForRetryProfile(ObjectStorageControlRequest{ + .profile = read_settings.object_storage_retry_profile, + .attempt_timeout_ms = read_settings.object_storage_attempt_timeout_ms, + .connect_timeout_cap_ms = read_settings.object_storage_connect_timeout_cap_ms}), uri.bucket, object.remote_path, uri.version_id, @@ -299,6 +380,7 @@ std::unique_ptr S3ObjectStorage::readObject( /// NOLINT /* offset */0, /* read_until_position */0, restrict_seek, +<<<<<<< HEAD /// `bytes_size` may be `StoredObject::UnknownSize` for an object whose size is not known /// (for example an HTTP source that omits `Content-Length`). It is a sentinel, not a real /// size, so it must map to `std::nullopt` (read to EOF) just like the legacy `0` value — @@ -307,6 +389,11 @@ std::unique_ptr S3ObjectStorage::readObject( /// NOLINT credentials_refresh_callback, std::move(blob_storage_log), object.etag); +======= + object.bytes_size ? std::optional(object.bytes_size) : std::nullopt, + refresh_callback, + std::move(blob_storage_log)); +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) } SmallObjectDataWithMetadata S3ObjectStorage::readSmallObjectAndGetObjectMetadata( /// NOLINT @@ -321,7 +408,15 @@ SmallObjectDataWithMetadata S3ObjectStorage::readSmallObjectAndGetObjectMetadata copyDataMaxBytes(*buffer, out, max_size_bytes); out.finalize(); - result.metadata = dynamic_cast(buffer.get())->getObjectMetadataFromTheLastRequest(); + auto * s3_buffer = dynamic_cast(buffer.get()); + if (s3_buffer->responseIdentityChanged()) + throw Exception( + ErrorCodes::CANNOT_READ_ALL_DATA, + "Object '{}' response identity changed between reissued GET requests; " + "the bytes read are not from one incarnation", + object.remote_path); + + result.metadata = s3_buffer->getObjectMetadataFromTheLastRequest(); return result; } @@ -380,13 +475,11 @@ std::unique_ptr S3ObjectStorage::writeObject( /// NOLIN /// The SingleAttempt profile (e.g. CAS conditional writes, RFC cas-s3-timeout-retry-control) rides /// on WriteSettings instead of changing this disk's shared client — every other write keeps using - /// client.get() and its normal retry policy unchanged. getSingleAttemptClient() is only invoked - /// when actually selected, so a plain write never pays for building/locking the clone. - std::shared_ptr used_client; - if (write_settings.object_storage_retry_profile == ObjectStorageRetryProfile::SingleAttempt) - used_client = getSingleAttemptClient(); - else - used_client = client.get(); + /// client.get() and its normal retry policy unchanged. + auto used_client = clientForRetryProfile(ObjectStorageControlRequest{ + .profile = write_settings.object_storage_retry_profile, + .attempt_timeout_ms = write_settings.object_storage_attempt_timeout_ms, + .connect_timeout_cap_ms = write_settings.object_storage_connect_timeout_cap_ms}); return std::make_unique( used_client, @@ -406,11 +499,22 @@ ObjectStorageIteratorPtr S3ObjectStorage::iterate( size_t max_keys, bool with_tags, const std::optional & start_after) const +{ + return iterate(path_prefix, max_keys, with_tags, start_after, ObjectStorageControlRequest{}); +} + +ObjectStorageIteratorPtr S3ObjectStorage::iterate( + const std::string & path_prefix, + size_t max_keys, + bool with_tags, + const std::optional & start_after, + const ObjectStorageControlRequest & request) const { auto settings_ptr = s3_settings.get(); if (!max_keys) max_keys = settings_ptr->request_settings[S3RequestSetting::list_object_keys_size]; - return std::make_shared(uri.bucket, path_prefix, client.get(), max_keys, with_tags, start_after); + return std::make_shared( + uri.bucket, path_prefix, clientForRetryProfile(request), max_keys, with_tags, start_after, request.attempt_number); } void S3ObjectStorage::listObjects(const std::string & path, RelativePathsWithMetadata & children, size_t max_keys) const @@ -523,6 +627,21 @@ void S3ObjectStorage::removeObjectsIfExist(const StoredObjects & objects) } ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatches(const StoredObject & object, const std::string & etag) +{ + return removeObjectIfTokenMatchesImpl(object, etag, client.get(), /*attempt_seed=*/0); +} + +ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatches( + const StoredObject & object, const std::string & etag, const ObjectStorageControlRequest & request) +{ + return refreshAndRetryOnExpiredCredentials( + [&] { return removeObjectIfTokenMatchesImpl( + object, etag, clientForRetryProfile(request), request.attempt_number); }); +} + +ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatchesImpl( + const StoredObject & object, const std::string & etag, const std::shared_ptr & used_client, + size_t attempt_seed) { S3::DeleteObjectRequest request; request.SetBucket(uri.bucket); @@ -531,10 +650,12 @@ ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatches(const Stored /// This is a content-addressed exact-token DELETE: mark it eligible for the typed NativeConditional /// mode, so a GCS-native client can send the generation token this etag actually encodes. request.setNativeConditional(); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(request, attempt_seed); ProfileEvents::increment(ProfileEvents::DiskS3DeleteObjects); - auto outcome = client.get()->DeleteObject(request); + auto outcome = used_client->DeleteObject(request); /// Mirror removeObjectImpl (deleteFileFromS3): every conditional delete lands in /// system.blob_storage_log too — GC reclaim was invisible there otherwise. TokenMismatch @@ -567,6 +688,116 @@ ConditionalRemoveResult S3ObjectStorage::removeObjectIfTokenMatches(const Stored err.GetMessage(), static_cast(err.GetErrorType()), err.GetExceptionName(), object.remote_path); } +void S3ObjectStorage::removeObjectsIfExistUnderProfile(const StoredObjects & objects, const ObjectStorageControlRequest & request) +{ + refreshAndRetryOnExpiredCredentials([&] + { + removeObjectsIfExistImpl(objects, clientForRetryProfile(request), request.attempt_number); + return 0; + }); +} + +void S3ObjectStorage::removeObjectsIfExistImpl( + const StoredObjects & objects, const std::shared_ptr & used_client, size_t attempt_seed) +{ + if (objects.empty()) + return; + + /// A batch of exactly one object is a plain `DeleteObject`, never `DeleteObjects` -- the same rule + /// `deleteFilesFromS3` applies to a single key. This is what makes the CAS-side per-key fallback + /// work on a backend with no `DeleteObjects` at all (GCS): that backend rejects the verb itself, not + /// a key count, so a "batch" of one object sent as `DeleteObjects` would fail there too. + if (objects.size() == 1) + { + const StoredObject & object = objects.front(); + deleteFileFromS3(used_client, uri.bucket, object.remote_path, /*if_exists=*/ true, + BlobStorageLogWriter::create(disk_name), object.local_path, object.bytes_size, + ProfileEvents::DiskS3DeleteObjects, attempt_seed); + return; + } + + /// GCS has no `DeleteObjects`: a capability the config declared false, or that an earlier batch + /// attempt on this same storage already learned false, must not be retried here. This storage never + /// loops over `objects` itself to work around it -- a CAS caller admits one request per physical + /// delete (see `ObjectStorageBackend::removeManyWriteOnce` and its own caller in CasGc.cpp), which an + /// internal loop over more than one object, running under a SINGLE admission, cannot be. Report the + /// absence of the capability instead, and let that caller decide how to retry. + if (auto support_batch_delete = s3_capabilities.isBatchDeleteSupported(); + support_batch_delete.has_value() && !support_batch_delete.value()) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support DeleteObjects", getName()); + + std::vector identifiers; // STYLE_CHECK_ALLOW_STD_CONTAINERS + identifiers.reserve(objects.size()); + for (const auto & object : objects) + { + Aws::S3::Model::ObjectIdentifier identifier; + identifier.SetKey(object.remote_path); + identifiers.push_back(std::move(identifier)); + } + Aws::S3::Model::Delete to_delete; + to_delete.SetObjects(std::move(identifiers)); + /// Quiet: only failed keys come back. A key that is gone or was never there is not a failure here + /// (`NoSuchKey` below), and the caller has no use for the per-key successes. + to_delete.SetQuiet(true); + + S3::DeleteObjectsRequest request; + request.SetBucket(uri.bucket); + request.SetDelete(std::move(to_delete)); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(request, attempt_seed); + + ProfileEvents::increment(ProfileEvents::DiskS3DeleteObjects); + auto outcome = used_client->DeleteObjects(request); + + /// Every key lands in system.blob_storage_log, as the single-key paths do; the batch's outcome is + /// stamped on each of them. This still happens when the batch turns out to be unsupported and the + /// call falls back below: the failed batch attempt is itself an event, same as in `deleteFilesFromS3`. + if (auto blob_storage_log = BlobStorageLogWriter::create(disk_name)) + { + for (const auto & object : objects) + blob_storage_log->addEvent(BlobStorageLogElement::EventType::Delete, + uri.bucket, object.remote_path, + object.local_path, object.bytes_size, + /* elapsed_microseconds */ 0, + outcome.IsSuccess() ? 0 : static_cast(outcome.GetError().GetErrorType()), + outcome.IsSuccess() ? "" : outcome.GetError().GetMessage()); + } + + if (!outcome.IsSuccess()) + { + const auto & err = outcome.GetError(); + /// Same classification `deleteFilesFromS3` uses to detect a backend that rejects `DeleteObjects` + /// itself (as opposed to a request that reached S3 and failed for an ordinary reason). + if ((err.GetExceptionName() == "InvalidRequest") || (err.GetExceptionName() == "InvalidArgument") + || (err.GetExceptionName() == "NotImplemented")) + { + LOG_TRACE(log, "DeleteObjects is not supported: {} (Code: {}). The caller must delete one object at a time.", + err.GetMessage(), static_cast(err.GetErrorType())); + s3_capabilities.setIsBatchDeleteSupported(false); + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support DeleteObjects", getName()); + } + + throw S3Exception(err.GetErrorType(), "{} (Code: {}) while removing {} objects from S3 in one request", + err.GetMessage(), static_cast(err.GetErrorType()), objects.size()); + } + + String failed_keys; + std::optional first_error_type; + for (const auto & err : outcome.GetResult().GetErrors()) + { + const auto error_type = classifyDeleteObjectsErrorCode(err.GetCode()); + if (S3::isNotFoundError(error_type)) + continue; + if (!failed_keys.empty()) + failed_keys += ", "; + failed_keys += err.GetKey() + " (" + err.GetCode() + ": " + err.GetMessage() + ")"; + if (!first_error_type) + first_error_type = error_type; + } + if (first_error_type) + throw S3Exception(*first_error_type, "batch removal left objects behind: [{}]", failed_keys); +} + bool S3ObjectStorage::conditionalOpsUseGenerationTokens() const { return client.get()->supportsGcsNativeConditionalRequests(); @@ -591,7 +822,12 @@ std::optional S3ObjectStorage::isBucketVersioningEnabled() const auto outcome = client.get()->GetBucketVersioning(request); if (!outcome.IsSuccess()) + { + /// The caller only learns "unknown"; the reason is what the operator needs to act on. + LOG_WARNING(log, "GetBucketVersioning for bucket `{}` failed: {} ({})", + uri.bucket, outcome.GetError().GetMessage(), outcome.GetError().GetExceptionName()); return std::nullopt; + } return outcome.GetResult().GetStatus() == Aws::S3::Model::BucketVersioningStatus::Enabled; } @@ -669,19 +905,39 @@ void S3ObjectStorage::tagObjects(const StoredObjects & objects, const std::strin std::optional S3ObjectStorage::tryGetObjectMetadata(const std::string & path, bool with_tags) const { - return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::Default); + return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::Default, client.get()); } std::optional S3ObjectStorage::tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const { - return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::NativeConditional); + return tryGetObjectMetadataImpl(path, with_tags, ObjectStorageRequestMode::NativeConditional, client.get()); } -std::optional S3ObjectStorage::tryGetObjectMetadataImpl(const std::string & path, bool with_tags, ObjectStorageRequestMode request_mode) const +std::optional S3ObjectStorage::tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags, const ObjectStorageControlRequest & request) const +{ + return refreshAndRetryOnExpiredCredentials( + [&] + { + return tryGetObjectMetadataImpl( + path, + with_tags, + ObjectStorageRequestMode::NativeConditional, + clientForRetryProfile(request), + request.attempt_number); + }); +} + +std::optional S3ObjectStorage::tryGetObjectMetadataImpl( + const std::string & path, + bool with_tags, + ObjectStorageRequestMode request_mode, + const std::shared_ptr & used_client, + size_t attempt_seed) const { auto settings_ptr = s3_settings.get(); auto object_info = S3::getObjectInfoIfExists( - *client.get(), uri.bucket, path, {}, /* with_metadata= */ true, with_tags, request_mode); + *used_client, uri.bucket, path, {}, /* with_metadata= */ true, with_tags, request_mode, attempt_seed); if (object_info.size == 0 && object_info.last_modification_time == 0 && object_info.metadata.empty()) return {}; @@ -852,12 +1108,33 @@ void S3ObjectStorage::shutdown() /// If S3 is healthy nothing wrong will be happened and S3 requests will be processed in a regular way without errors. /// This should significantly speed up shutdown process if S3 is unhealthy. const_cast(*client.get()).DisableRequestProcessing(); + + /// The SDK checks this flag only after an attempt has failed, right before deciding whether to + /// retry -- it neither blocks a request's initial dispatch nor interrupts one already in flight. + /// Every cached clone below runs `SingleAttemptRetryStrategy` (max_retries=0), so its retry strategy + /// never asks for a reissue; the one reissue the SDK makes on its own regardless of the strategy -- + /// an `AWS_GLOBAL` client re-signing for the region a 301/307/400/403 reply names -- is what the + /// disabled flag stops on a clone, since that check runs before the region redirect is considered. + /// What actually blocks a NEW request on the CAS engine's open plane (GC, FSCK, the probe) after + /// shutdown began is admission, refused at `Pool::teardownBegun()` (`CasPool.cpp`), which + /// `CasOperation::readLoop` (`CasRequests.h`) checks before every attempt, including the first. + /// The mount and farewell planes stay admitting through this window, since teardown's own drain + /// and farewell I/O run on them. + std::lock_guard lock(single_attempt_client_mutex); + single_attempt_clients_disabled = true; + for (const auto & [_, clone] : single_attempt_clients) + const_cast(*clone).DisableRequestProcessing(); } void S3ObjectStorage::startup() { /// Need to be enabled if it was disabled during shutdown() call. const_cast(*client.get()).EnableRequestProcessing(); + + std::lock_guard lock(single_attempt_client_mutex); + single_attempt_clients_disabled = false; + for (const auto & [_, clone] : single_attempt_clients) + const_cast(*clone).EnableRequestProcessing(); } void S3ObjectStorage::applyNewSettings( @@ -976,12 +1253,19 @@ std::shared_ptr S3ObjectStorage::tryGetS3StorageClient() return client.get(); } -std::shared_ptr S3ObjectStorage::getSingleAttemptClient() const +std::shared_ptr S3ObjectStorage::getSingleAttemptClient(uint64_t request_timeout_ms, uint64_t connect_timeout_cap_ms) const { auto base = client.get(); std::lock_guard lock(single_attempt_client_mutex); - if (single_attempt_client && single_attempt_client_base == base) - return single_attempt_client; + if (single_attempt_client_base != base) + { + single_attempt_clients.clear(); + single_attempt_client_base = base; + } + + const auto cache_key = std::make_pair(request_timeout_ms, connect_timeout_cap_ms); + if (auto it = single_attempt_clients.find(cache_key); it != single_attempt_clients.end()) + return it->second; auto cfg = base->getClientConfiguration(); cfg.retry_strategy.max_retries = 0; @@ -994,9 +1278,40 @@ std::shared_ptr S3ObjectStorage::getSingleAttemptClient() cons if (cfg.expect_continue_min_bytes == 0) cfg.expect_continue_min_bytes = fallback_expect_continue_min_bytes; - single_attempt_client = base->cloneWithConfigurationOverride(cfg); - single_attempt_client_base = base; - return single_attempt_client; + if (request_timeout_ms != 0) + cfg.requestTimeoutMs = static_cast(request_timeout_ms); + + /// One TCP/TLS connect may not cost more than the cap the mount froze at open: the engine reserves + /// attempt + 2 × cap per envelope, and a reloaded base client with a wider connect timeout must not + /// widen what a reissue can spend. + if (connect_timeout_cap_ms != 0) + cfg.connectTimeoutMs = cfg.connectTimeoutMs <= 0 ? static_cast(connect_timeout_cap_ms) + : std::min(cfg.connectTimeoutMs, static_cast(connect_timeout_cap_ms)); + + const auto & clone = single_attempt_clients.emplace(cache_key, base->cloneWithConfigurationOverride(cfg)).first->second; + + /// A fresh clone starts with request processing enabled regardless of the main client's state, so + /// one built after shutdown() must be disabled to match it (see the comment on shutdown() for what + /// that flag does and does not do). + if (single_attempt_clients_disabled) + const_cast(*clone).DisableRequestProcessing(); + + return clone; +} + +bool S3ObjectStorage::hasSingleAttemptClientForTest(uint64_t request_timeout_ms, uint64_t connect_timeout_cap_ms) const +{ + std::lock_guard lock(single_attempt_client_mutex); + return single_attempt_clients.contains(std::make_pair(request_timeout_ms, connect_timeout_cap_ms)); +} + +std::shared_ptr S3ObjectStorage::clientForRetryProfile(const ObjectStorageControlRequest & request) const +{ + /// getSingleAttemptClient is only invoked when actually selected, so an ordinary request never + /// pays for building or locking the clone. + if (request.profile == ObjectStorageRetryProfile::SingleAttempt) + return getSingleAttemptClient(request.attempt_timeout_ms, request.connect_timeout_cap_ms); + return client.get(); } bool S3ObjectStorage::tryRefreshCredentialsViaCallback() diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h index ad19801ed744..c33d55f9b836 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h @@ -5,9 +5,14 @@ #if USE_AWS_S3 #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) #include #include +#include #include #include #include @@ -45,8 +50,13 @@ class S3ObjectStorage : public IObjectStorage bool client_restricts_server_credentials_ = true) : uri(uri_) , disk_name(disk_name_) +<<<<<<< HEAD , client(std::move(client_)) , client_restricts_server_credentials(client_restricts_server_credentials_) +======= + , client_slot(std::make_shared>(std::move(client_))) + , client(*client_slot) +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) , s3_settings(std::move(s3_settings_)) , s3_capabilities(s3_capabilities_) , key_generator(std::move(key_generator_)) @@ -106,6 +116,13 @@ class S3ObjectStorage : public IObjectStorage bool with_tags, const std::optional & start_after) const override; + ObjectStorageIteratorPtr iterate( + const std::string & path_prefix, + size_t max_keys, + bool with_tags, + const std::optional & start_after, + const ObjectStorageControlRequest & request) const override; + /// Uses `DeleteObjectRequest`. void removeObjectIfExists(const StoredObject & object) override; @@ -116,6 +133,19 @@ class S3ObjectStorage : public IObjectStorage /// Uses `DeleteObjectRequest` with `If-Match` (token-exact removal for content-addressed disks). ConditionalRemoveResult removeObjectIfTokenMatches(const StoredObject & object, const std::string & etag) override; + ConditionalRemoveResult removeObjectIfTokenMatches( + const StoredObject & object, const std::string & etag, const ObjectStorageControlRequest & request) override; + + /// One `DeleteObjects` for the given objects (the caller chunks to at most 1000); absence is success. + /// Exactly one object is always a plain `DeleteObject` instead (never gated on `s3_capabilities`: a + /// single physical request per call, so there is nothing here for that capability to say no to). + /// For more than one object, throws `NOT_IMPLEMENTED` without sending anything once `DeleteObjects` + /// is known unsupported (a configured or a just-learned `S3Capabilities::isBatchDeleteSupported() == + /// false`) -- this storage never substitutes a per-key loop of its own, since the caller is the one + /// that can admit each physical delete as its own request. + void removeObjectsIfExistUnderProfile( + const StoredObjects & objects, const ObjectStorageControlRequest & request) override; + void tagObjects(const StoredObjects & objects, const std::string & tag_key, const std::string & tag_value) override; ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override; @@ -126,6 +156,9 @@ class S3ObjectStorage : public IObjectStorage /// `nativeHead` can read a GCS generation token where the client's HTTP layer supports one. std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const override; + std::optional tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags, const ObjectStorageControlRequest & request) const override; + void copyObject( /// NOLINT const StoredObject & object_from, const StoredObject & object_to, @@ -183,19 +216,49 @@ class S3ObjectStorage : public IObjectStorage /// (SingleAttemptRetryStrategy, max_retries=0, Expect:100-continue floor). Rebuilt whenever the /// disk client rotates (applyNewSettings/credentials refresh) — the cached clone is keyed by the /// base client's identity, so a stale clone can never outlive a rotation. - std::shared_ptr getSingleAttemptClient() const; + /// `request_timeout_ms` overrides the clone's send/receive inactivity bound; 0 keeps the disk's. + /// `connect_timeout_cap_ms` caps the clone's connect timeout (0 = no cap); the cache key is the + /// pair, so two callers asking for the same request timeout but different caps get distinct clones. + std::shared_ptr getSingleAttemptClient(uint64_t request_timeout_ms, uint64_t connect_timeout_cap_ms = 0) const; + + /// True iff a clone for exactly this (request timeout, connect cap) pair is already cached -- + /// never builds one. Lets a test prove dispatch used a SPECIFIC key (and no other) without ever + /// creating a clone itself and without measuring anything. + bool hasSingleAttemptClientForTest(uint64_t request_timeout_ms, uint64_t connect_timeout_cap_ms) const; + private: void removeObjectImpl(const StoredObject & object, bool if_exists); void removeObjectsImpl(const StoredObjects & objects, bool if_exists); /// Shared by tryGetObjectMetadata/tryGetObjectMetadataWithNativeToken: the only difference between /// the two public overrides is which ObjectStorageRequestMode the HEAD wrapper carries. - std::optional tryGetObjectMetadataImpl(const std::string & path, bool with_tags, ObjectStorageRequestMode request_mode) const; + std::optional tryGetObjectMetadataImpl( + const std::string & path, + bool with_tags, + ObjectStorageRequestMode request_mode, + const std::shared_ptr & used_client, + size_t attempt_seed = 0) const; + + ConditionalRemoveResult removeObjectIfTokenMatchesImpl( + const StoredObject & object, const std::string & etag, const std::shared_ptr & used_client, + size_t attempt_seed); + + void removeObjectsIfExistImpl( + const StoredObjects & objects, const std::shared_ptr & used_client, size_t attempt_seed); + + std::shared_ptr clientForRetryProfile(const ObjectStorageControlRequest & request) const; + + /// Runs `fn` and, if it failed because the vended credentials expired, refreshes this disk's + /// client and runs it once more. `fn` must re-read the client itself, so the second run signs + /// with the refreshed one. + template + auto refreshAndRetryOnExpiredCredentials(Fn && fn) const; const S3::URI uri; std::string disk_name; +<<<<<<< HEAD mutable MultiVersion client; /// The user-query credential restriction mode the current `client` was built under (initialized by the /// caller from the policy used to build the initial client -- e.g. a table created with the opt-in starts @@ -204,6 +267,18 @@ class S3ObjectStorage : public IObjectStorage /// vice versa). Defaults to restricted (the server default) for callers that do not pass an explicit value. /// Atomic: `applyNewSettings` can run concurrently on a shared storage. mutable std::atomic client_restricts_server_credentials = true; +======= + /// The slot this disk's client lives in, and the only thing a credential refresh installs into. + /// Held by `shared_ptr` so a read buffer -- which can outlive this storage -- carries the SLOT + /// rather than a pointer to the storage: a refresh that arrives late then replaces a client + /// nobody will read again, instead of writing into a destroyed object. + const std::shared_ptr> client_slot; + /// Reference into the slot above. Every ordinary call site keeps using `client.get()`/`client.set()` + /// unchanged; only code that must capture the client independently of this storage's own lifetime + /// (the credential-refresh lambda handed to a read buffer that can outlive this object) captures + /// `client_slot` directly instead. + MultiVersion & client; +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) MultiVersion s3_settings; S3Capabilities s3_capabilities; @@ -220,15 +295,22 @@ class S3ObjectStorage : public IObjectStorage std::atomic pinned_generation_dialect{-1}; /// -1 unpinned, 0 pinned ETag, 1 pinned generation mutable std::mutex single_attempt_client_mutex; - mutable std::shared_ptr single_attempt_client; - /// The base client the cached clone above was built from. Deliberately held as a shared_ptr (not + /// One clone per (requested timeout, connect cap) pair: the verbs of one operation ask for + /// different bounds, and a single slot would rebuild a whole S3 client (and lose its connection + /// pool) on every alternation between them. + mutable std::map, std::shared_ptr> single_attempt_clients; + /// The base client the cached clones above were built from. Deliberately held as a shared_ptr (not /// a raw pointer): a raw pointer would be compared for identity AFTER the object it once pointed /// to could have been freed and a new client reallocated at the same address by an unrelated /// rotation (ABA), which would false-match and serve a stale clone (e.g. built from retired /// credentials) indefinitely. Holding the shared_ptr pins at most one retired client version — - /// released as soon as the next rotation is observed and the clone is rebuilt — which is what + /// released as soon as the next rotation is observed and the clones are dropped — which is what /// makes the identity comparison in getSingleAttemptClient sound. mutable std::shared_ptr single_attempt_client_base; + /// Set for the duration of a `shutdown()` (cleared by the matching `startup()`); every clone already + /// cached when `shutdown()` runs is disabled there and then, and this flag disables any built + /// afterwards to match. See the comment on `shutdown()` for what that disabling does and does not do. + mutable bool single_attempt_clients_disabled = false; }; } diff --git a/src/Disks/tests/cas_format_test_battery.h b/src/Disks/tests/cas_format_test_battery.h index 173bfcc5c4df..efed9ce44a74 100644 --- a/src/Disks/tests/cas_format_test_battery.h +++ b/src/Disks/tests/cas_format_test_battery.h @@ -4,6 +4,7 @@ #include #include #include +#include namespace DB::ErrorCodes { @@ -28,12 +29,19 @@ struct FormatBatteryCase std::function make_future_version = {}; }; -/// Canonical object headers track the current compatibility generation. The type remains an -/// explicit test literal at every call site, so a registry/type mismatch cannot be hidden by a -/// self-derived expectation. +/// The canonical object header, spelled literally. +/// +/// The version is the LITERAL 1, not `currentCompatibilityVersion()`. Deriving it from production +/// was the defect: encoder output and expected bytes would then move together across a generation +/// bump, and a golden that tracks the code it is meant to pin cannot fail. The type was already a +/// literal at every call site for the same reason; the version had been left behind. +/// +/// A future generation bump is therefore SUPPOSED to break every test that uses this. That is the +/// point: the new bytes get read, agreed to, and written down, rather than being adopted silently. +/// `HeaderVersionIsTheLiteralThisBatteryPins` below fails first and says so. inline String currentFormatHeader(std::string_view type) { - return fmt::format("{{\"type\":\"{}\",\"v\":{}}}\n", type, DB::Cas::currentCompatibilityVersion()); + return fmt::format("{{\"type\":\"{}\",\"v\":1}}\n", type); } namespace cas_battery_detail @@ -53,6 +61,23 @@ void expectCode(int code, F && f, const String & context) } } +namespace DB::Cas::tests +{ +inline std::set & batteryCoveredIds() +{ + static std::set ids; + return ids; +} + +struct BatteryCoverageRegistrar +{ + explicit BatteryCoverageRegistrar(FormatId id) { batteryCoveredIds().insert(id); } +}; +} + +#define CAS_BATTERY_COVERS(format_id) \ + static const DB::Cas::tests::BatteryCoverageRegistrar battery_covers_##format_id{DB::Cas::FormatId::format_id} + inline void runFormatBattery(const FormatBatteryCase & c) { using namespace DB::Cas; diff --git a/src/Disks/tests/cas_sweep_test_support.h b/src/Disks/tests/cas_sweep_test_support.h index c1b5466a1cb1..6326dd7c0546 100644 --- a/src/Disks/tests/cas_sweep_test_support.h +++ b/src/Disks/tests/cas_sweep_test_support.h @@ -1,5 +1,5 @@ #pragma once -#include +#include #include #include #include @@ -22,10 +22,15 @@ inline ManifestSweepResult sweepManifestCursorPageForTest( { ManifestSweepResult result = planManifestCursorPage( store, cursor, list_budget, delete_budget, /*catalog_recovery_authoritative=*/true, work_budget); + CasOperation op = store.openRequests().admit(); for (const ManifestSweepResult::Nomination & nomination : result.nominations) { - const DeleteOutcome outcome = store.backend().deleteExact(nomination.key, nomination.token); - if (classifyDeleteOutcome(outcome) == DeleteClass::Deleted) + /// A nomination records the incarnation it was planned against, so the delete re-observes the + /// key and refuses unless what is there now is still that one: a key a fresh owner has since + /// replaced must survive. + const std::optional seen = op.head(nomination.key, Retry::standard()); + if (seen && nomination.token.matches(seen->etag) + && op.remove(nomination.key, seen->etag, Retry::standard()) == Removal::Removed) ++result.deleted; else ++result.skipped; diff --git a/src/Disks/tests/cas_test_helpers.h b/src/Disks/tests/cas_test_helpers.h index e7b5f6221c18..ae87f129fce3 100644 --- a/src/Disks/tests/cas_test_helpers.h +++ b/src/Disks/tests/cas_test_helpers.h @@ -9,12 +9,14 @@ #include #include #include +#include #include #include #include #include #include #include +#include #include #include #include @@ -34,6 +36,7 @@ #include #include #include +#include /// For `ChunkFaultBackend`'s `DefiniteFailure` mode, which needs a real S3-classified error, and for /// the ambiguity it raises otherwise. #include @@ -43,6 +46,7 @@ #include #include #include +#include #include #include #include @@ -75,6 +79,48 @@ namespace DB::ErrorCodes namespace DB::Cas::tests { +/// A `CasRequests` over an always-open fence, for a fixture that has a backend but no mounted pool. +/// Every operation admitted from it holds a reference to it, so it must be named and outlive them. +inline DB::Cas::CasRequests openRequestsForTest(DB::Cas::BackendPtr backend) +{ + return DB::Cas::CasRequests(std::move(backend), DB::Cas::Fence::open()); +} + +/// The same, for a fixture holding only a reference. The aliasing `shared_ptr` owns nothing, so the +/// caller keeps the backend alive for as long as the returned object and its operations live. +inline DB::Cas::CasRequests openRequestsForTest(DB::Cas::Backend & backend) +{ + return openRequestsForTest(DB::Cas::BackendPtr(std::shared_ptr(), &backend)); +} + +/// An open-fence operation together with the `CasRequests` it refers to, for a fixture that holds a +/// backend and needs to call a production entry point taking a `CasOperation &`. Neither copyable nor +/// movable: the operation points at the member beside it. +class OperationForTest +{ +public: + explicit OperationForTest(DB::Cas::BackendPtr backend) + : requests(std::move(backend), DB::Cas::Fence::open()), operation(requests.admit()) + { + } + + /// For a fixture holding only a reference: the aliasing `shared_ptr` owns nothing, so the caller + /// keeps the backend alive for as long as this object. + explicit OperationForTest(DB::Cas::Backend & backend) + : OperationForTest(DB::Cas::BackendPtr(std::shared_ptr(), &backend)) + { + } + + OperationForTest(const OperationForTest &) = delete; + OperationForTest & operator=(const OperationForTest &) = delete; + + DB::Cas::CasOperation & operator*() { return operation; } + +private: + DB::Cas::CasRequests requests; + DB::Cas::CasOperation operation; +}; + /// Deterministic two-phase barrier for worker-lifecycle tests. The worker calls `arriveAndWait` at /// the exact operation boundary under test; the test waits for that arrival and later calls /// `release`. The bounded waits are only hang protection -- correctness never depends on elapsed @@ -112,6 +158,66 @@ class ManualBarrier bool released = false; }; +/// Heap-owned wait/sleep log for a `wait_sleep_fn`-shaped test hook: a hook that pushed into a +/// stack-local vector would read (or write) a dead frame if a background completion outlives the test +/// -- a `Pool`'s own detached publish can hold `shared_from_this()` past the test function's return. +/// Mutex-guarded because that background call can race a foreground read. Construct via +/// `std::make_shared` and capture the shared_ptr by value into the hook, never the bare object by +/// reference. +class SharedWaitLog +{ +public: + void push(uint64_t ms) + { + std::lock_guard lock(mutex); + values.push_back(ms); + } + size_t size() const + { + std::lock_guard lock(mutex); + return values.size(); + } + bool empty() const + { + std::lock_guard lock(mutex); + return values.empty(); + } + std::vector snapshot() const + { + std::lock_guard lock(mutex); + return values; + } + +private: + mutable std::mutex mutex; + std::vector values; +}; + +/// Heap-owned event log for an `event_sink`-shaped test hook: a hook that pushed into a stack-local +/// vector would read (or write) a dead frame if a background completion outlives the test -- a `Pool`'s +/// own detached publish can hold `shared_from_this()` past the test function's return, and its farewell +/// or a background renewer can emit events from a thread the test itself never joins. Mutex-guarded +/// because that background call can race a foreground read. Construct via `std::make_shared` and capture +/// the shared_ptr by value into the sink, never the bare object by reference. +class SharedEventLog +{ +public: + void push(CasEvent event) + { + std::lock_guard lock(mutex); + values.push_back(std::move(event)); + } + std::vector snapshot() const + { + std::lock_guard lock(mutex); + return values; + } + +private: + mutable std::mutex mutex; + std::vector values; +}; + /// Bring up the server-wide blob upload pool (stage-1 §1) if it is not already up, so any test that /// drives a `ContentAddressedTransaction` commit -- whose `uploadPendingBlobs` fans out on this pool -- /// finds it initialized. ROBUST (init-if-not-initialized, NOT `call_once`): the raw-lifecycle suite in @@ -127,9 +233,9 @@ inline void ensureBlobUploadPoolForTest(size_t size = 8) /// Minimal `ContentAddressedSettings` for a direct-construction gtest fixture: sets only /// `server_root_id` and `scratch_path` (the two values every positional-ctor call site used to pass -/// explicitly) and validates, so the cached enum-valued accessors (`stagingBackend`, `blobHashAlgo`, -/// `partFolderValidate`) are populated from their (default) string settings exactly as the disk-factory -/// path would populate them. Callers that need a non-default setting (e.g. `staging_backend=s3`) apply +/// explicitly) and validates, so the cached enum-valued accessors (`stagingBackend`, `blobHashAlgo`) +/// are populated from their (default) string settings exactly as the disk-factory path would populate +/// them. Callers that need a non-default setting (e.g. `staging_backend=s3`) apply /// the override via `settings[ContentAddressedSetting::x] = value;` and re-run `settings.validate()` /// themselves before constructing. inline DB::ContentAddressedSettings makeSettingsForTest(const std::string & server_root_id, const std::filesystem::path & scratch_path) @@ -141,6 +247,37 @@ inline DB::ContentAddressedSettings makeSettingsForTest(const std::string & serv return settings; } +/// A clock that only ever moves when something sleeps on it, plus the record of every sleep it +/// served. Injected into `CasRequests` so a policy's whole 90-second deadline is exercised in a test +/// that takes no wall-clock time, and so the schedule itself -- how many pauses, how long -- becomes +/// an assertion rather than a wait. `now` is atomic and `sleeps` is mutex-guarded because a +/// decommission session's background mount-lease renewer reads this same clock (via +/// `PoolConfig::boot_ms_fn`) from its own thread while the caller's thread drives it forward through +/// `sleepFn` -- relaxed ordering is enough since nothing here needs a happens-before relationship +/// beyond the value eventually becoming visible; direct field reads from a single thread after the +/// clock stops moving (the common case in this file's other users) are unaffected. +struct FakeClock +{ + std::atomic now{1'000'000}; + std::vector sleeps; + + std::function nowFn() { return [this] { return now.load(std::memory_order_relaxed); }; } + std::function sleepFn() + { + return [this](uint64_t ms) + { + { + std::lock_guard lock(sleeps_mutex); + sleeps.push_back(ms); + } + now.fetch_add(ms, std::memory_order_relaxed); + }; + } + +private: + std::mutex sleeps_mutex; +}; + /// Run `fn`, expect a DB::Exception with EXACTLY `expected_code` (CORRUPTED_DATA-vs-NOT_IMPLEMENTED /// is part of the fail-closed contract: an unknown future format must be NOT_IMPLEMENTED, never /// misreported as corruption). @@ -158,6 +295,21 @@ void expectThrowsCode(int expected_code, F && fn) } } +/// Asserts the object is present before comparing its body: an absent key would otherwise dereference +/// an empty optional and take the whole binary down instead of failing this one case. +inline void expectBytes(DB::Cas::Backend & backend, const String & key, const String & expected) +{ + OperationForTest op(backend); + const auto got = (*op).read(key, DB::Cas::Retry::standard()); + ASSERT_TRUE(got.has_value()) << "object '" << key << "' is absent"; + EXPECT_EQ(got->bytes, expected); +} + +inline void expectBytes(const DB::Cas::BackendPtr & backend, const String & key, const String & expected) +{ + expectBytes(*backend, key, expected); +} + /// Build a `LocalObjectStorage` rooted at a fresh, unique temporary directory (one per call). /// /// Used by the unit tests that exercise the `Cas::Backend` seam against a real on-disk object storage @@ -252,7 +404,8 @@ inline DB::Cas::BlobRef writeBlobRaw( header.build_id = DB::UInt128(0x5678); const String head = DB::Cas::encodeEnvelopeHeader(header, static_cast(blob_header_len)); - backend.putIfAbsent(layout.blobKey(id), head + payload); + OperationForTest op(backend); + (*op).create(layout.blobKey(id), head + payload, DB::Cas::Retry::standard()); return id; } @@ -274,8 +427,9 @@ inline DB::Cas::ManifestId writeManifestRaw( body.root_namespace_id = ns; body.entries = entries; body.payload_digest = DB::Cas::computePayloadDigest(body); - backend.putIfAbsent(layout.manifestKey(id), - DB::Cas::sealObject(DB::Cas::FormatId::PartManifest, DB::Cas::encodePartManifest(body))); + OperationForTest op(backend); + (*op).create(layout.manifestKey(id), + DB::Cas::sealObject(DB::Cas::FormatId::PartManifest, DB::Cas::encodePartManifest(body)), DB::Cas::Retry::standard()); return id; } @@ -334,14 +488,16 @@ inline uint64_t appendRefLogSeed( /// sentinel), exactly as `writeRefLogTxnRaw` below now does -- otherwise this scan can miss a REAL /// incarnation's existing log/snap objects, wrongly conclude the table has none, and prepend a second /// `namespaceBirthOp` on top of a namespace that already has one. - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + OperationForTest operation(backend); + const NamespaceLifeId life + = CasRefCatalog::lifeIfCataloged(*operation, layout, ns).value_or(fixture::fixtureLife(ns)); const String prefix = layout.namespaceStreamPrefix(life); uint64_t greatest_seq = 0; bool any_log_or_snap = false; String cursor; while (true) { - const DB::Cas::ListPage page = backend.list(prefix, cursor, /*limit=*/1000); + const DB::Cas::ListPage page = (*operation).list(prefix, cursor, /*limit=*/1000, DB::Cas::Retry::standard()); for (const DB::Cas::ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -451,9 +607,9 @@ inline void deleteManifestBody( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::ManifestId & id) { const String key = layout.manifestKey(id); - const DB::Cas::HeadResult h = backend.head(key); - if (h.exists) - backend.deleteExact(key, h.token); + OperationForTest op(backend); + if (const auto h = (*op).head(key, DB::Cas::Retry::standard())) + (*op).remove(key, h->etag, DB::Cas::Retry::standard()); } /// Formerly wrote the namespace into `gc/registry`. Real write helpers now admit the authoritative @@ -474,75 +630,99 @@ inline String encodeMinimalGcState(uint64_t round) } /// Inject condemned bookkeeping + gc/state directly (bypassing a real GC round) so a test can seed the -/// GC ledger's condemned state at an arbitrary round. Retired-in-snapshot: the condemned entries are -/// seeded the way a real round leaves them — as `kCondemned` sentinel rows inside an adopted fold seal's -/// shard run (there is no separate retired-list object). A synthetic +edge/-edge pair nets each blob to -/// in-degree 0 and a `seed_head` replays the captured token/size so the fold mints the `kCondemned` row. -/// Also sets {round} on gc/state. Entries carry a `condemn_round` (default 0 → uses `round`); callers -/// pass fresh (non-pending) condemns. An empty `entries` set just advances {round}. +/// GC ledger's condemned state at an arbitrary round. The entries are written into the adopted seal's +/// shard run as `RunMarker::Condemned` sentinel rows at the zero source id -- the shape a real round +/// leaves, since there is no separate retired-list object. Also sets {round} on gc/state. An entry's +/// `condemn_round` defaults to `round` when left 0. An empty `entries` set just advances {round}. +/// +/// The rows are written from the caller's entries VERBATIM rather than folded out of synthetic deltas. +/// A fold mints each row's incarnation from a live HEAD of the blob, which can express neither of the +/// two shapes this fixture exists to build: an entry condemning an incarnation NO object carries (a +/// phantom, for the tests that check a condemnation aimed elsewhere spares the live object), and an +/// entry for a blob whose body is already gone. inline void injectRetire( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t round, uint64_t shard, std::vector entries) { + OperationForTest operation(backend); DB::Cas::GcState gc_state; - const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); - if (head.exists) - gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + const auto existing_state = (*operation).read(layout.gcStateKey(), DB::Cas::Retry::standard()); + if (existing_state) + gc_state = DB::Cas::decodeGcState(existing_state->bytes); gc_state.round = round; if (!entries.empty()) { const uint64_t generation = 1; const uint64_t attempt = 1; - uint64_t condemn_round = round; - std::unordered_map seeded; - std::vector synth; - synth.reserve(entries.size() * 2); - for (const DB::Cas::RetiredEntry & e : entries) + + /// The writer's contract: records arrive in non-decreasing `(ref, source_id)` order, and a + /// sentinel row is the only row a blob may have at the zero source id. + std::sort(entries.begin(), entries.end(), + [](const DB::Cas::RetiredEntry & a, const DB::Cas::RetiredEntry & b) { return a.ref < b.ref; }); + + String run_bytes; + /// `UINT64_MAX` is the summary's own "no non-pending entry" value, and round 0 is a real round, + /// so a zero initializer here would claim the oldest possible condemnation instead of none. + uint64_t oldest_nonpending = UINT64_MAX; + uint64_t pending_total = 0; { - if (e.condemn_round) - condemn_round = e.condemn_round; - seeded.emplace(e.ref, DB::Cas::HeadResult{.exists = true, .size = e.size, .token = e.token, .attributes = {}}); - synth.push_back(DB::Cas::BlobDelta{.ref = e.ref, .source_id = DB::UInt128{1}, .remove = false}); - synth.push_back(DB::Cas::BlobDelta{.ref = e.ref, .source_id = DB::UInt128{1}, .remove = true}); + DB::WriteBufferFromString out(run_bytes); + DB::Cas::SourceEdgeRunWriter writer(out); + for (const DB::Cas::RetiredEntry & e : entries) + { + const uint64_t condemn_round = e.condemn_round ? e.condemn_round : round; + if (e.delete_pending) + ++pending_total; + else + oldest_nonpending = std::min(oldest_nonpending, condemn_round); + writer.append(DB::Cas::SourceEdgeRecord{ + .ref = e.ref, + .source_id = DB::UInt128{0}, + .marker = DB::Cas::RunMarker::Condemned, + .delete_pending = e.delete_pending, + .token = e.token, + .size = e.size, + .condemn_round = condemn_round, + .marker_confirmed = e.marker_confirmed}); + } + writer.finish(); + out.finalize(); } - const auto seed_head = [&seeded](const DB::Cas::BlobRef & h) -> std::optional - { - const auto it = seeded.find(h); - return it == seeded.end() ? std::nullopt : std::optional(it->second); - }; - std::vector out; - DB::Cas::foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, generation, attempt, - shard, std::move(synth), out, /*current_round*/0, condemn_round, seed_head, - /*peek_head*/{}, /*confirm_condemned_marker*/{}, - /*out_retired*/nullptr, /*suppress_destructive*/false); + + const String run_key = layout.blobTargetRunKey(generation, attempt, shard, 0); + DB::Cas::orThrow((*operation).create(run_key, run_bytes, DB::Cas::Retry::standard()), + "seed the condemned run at " + run_key); DB::Cas::CasFoldSeal seal; seal.generation = generation; - for (DB::Cas::RunRef & r : out) - seal.blob_target_runs.push_back(std::move(r)); + seal.blob_target_runs.push_back(DB::Cas::RunRef{.key = run_key, + .checksum = DB::Cas::sourceEdgeRunChecksum(run_bytes), + .shard = shard, + .key_generation = generation}); /// Totality over gc_shards so a later real round's graduation/carry reads it zero-I/O. const uint64_t gc_shards = gc_state.gc_shards ? gc_state.gc_shards : 1; for (uint64_t s = 0; s < gc_shards; ++s) seal.condemned_summary[s] = DB::Cas::CondemnedSummary{}; DB::Cas::CondemnedSummary cs; cs.condemned_total = entries.size(); - cs.oldest_nonpending_condemn_round = condemn_round; + cs.pending_total = pending_total; + cs.oldest_nonpending_condemn_round = oldest_nonpending; seal.condemned_summary[shard] = cs; - backend.putIfAbsent(layout.foldSealKey(generation, attempt), DB::Cas::encodeFoldSeal(seal)); + (*operation).create(layout.foldSealKey(generation, attempt), DB::Cas::encodeFoldSeal(seal), DB::Cas::Retry::standard()); gc_state.snap_generation = generation; gc_state.snap_attempt = attempt; } const String state = DB::Cas::encodeGcState(gc_state); - if (!head.exists) - backend.putIfAbsent(layout.gcStateKey(), state); + if (!existing_state) + (*operation).create(layout.gcStateKey(), state, DB::Cas::Retry::standard()); else - backend.putOverwrite(layout.gcStateKey(), state, head.token); + (*operation).replace(layout.gcStateKey(), state, existing_state->etag, DB::Cas::Retry::standard()); } -/// Adopt a fold seal carrying a given per-gc-shard `condemned_summary` (retired-in-snapshot T4) and point +/// Adopt a fold seal carrying a given per-gc-shard `condemned_summary` and point /// gc/state at it (snap_generation / snap_attempt / gc_shards), bypassing a real GC round. If a seal /// already exists at (generation, attempt) it is overwritten with the new summary (its other fields are /// preserved); otherwise a fresh minimal seal is created. Read-modify-CAS on gc/state preserves the lease. @@ -552,9 +732,10 @@ inline void injectCondemnedSummarySeal( uint64_t generation, uint64_t attempt, uint64_t gc_shards, const std::map & summary) { + OperationForTest operation(backend); const String seal_key = layout.foldSealKey(generation, attempt); DB::Cas::CasFoldSeal seal; - const auto existing = backend.get(seal_key); + const auto existing = (*operation).read(seal_key, DB::Cas::Retry::standard()); if (existing) seal = DB::Cas::decodeFoldSeal(existing->bytes); else @@ -563,28 +744,30 @@ inline void injectCondemnedSummarySeal( seal.condemned_summary = summary; const String seal_bytes = DB::Cas::encodeFoldSeal(seal); if (existing) - backend.putOverwrite(seal_key, seal_bytes, existing->token); + (*operation).replace(seal_key, seal_bytes, existing->etag, DB::Cas::Retry::standard()); else - backend.putIfAbsent(seal_key, seal_bytes); + (*operation).create(seal_key, seal_bytes, DB::Cas::Retry::standard()); DB::Cas::GcState gc_state; - const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); - if (head.exists) - gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + const auto existing_state = (*operation).read(layout.gcStateKey(), DB::Cas::Retry::standard()); + if (existing_state) + gc_state = DB::Cas::decodeGcState(existing_state->bytes); gc_state.gc_shards = gc_shards; gc_state.snap_generation = generation; gc_state.snap_attempt = attempt; const String state = DB::Cas::encodeGcState(gc_state); - if (!head.exists) - backend.putIfAbsent(layout.gcStateKey(), state); + if (!existing_state) + (*operation).create(layout.gcStateKey(), state, DB::Cas::Retry::standard()); else - backend.putOverwrite(layout.gcStateKey(), state, head.token); + (*operation).replace(layout.gcStateKey(), state, existing_state->etag, DB::Cas::Retry::standard()); } /// Whether blob `hash` is absent from the backend (its exact-token content object is gone). inline bool blobAbsent(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash) { - return !backend.head(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)})).exists; + OperationForTest op(backend); + return !(*op).head(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), + DB::Cas::Retry::standard()).has_value(); } /// ONE round that is allowed to RECLAIM -- the name is the point, so that grepping for the tests whose @@ -620,21 +803,22 @@ inline bool runRoundsUntilAbsent( } /// The CURRENT condemned entries for `shard`, read from the adopted fold seal's `blob_target_runs` -/// (retired-in-snapshot T4): the round no longer writes a separate retired-list object — condemned -/// entries RIDE the source-edge run as `kCondemned` sentinel rows at the zero-sentinel key. This reads +///: the round no longer writes a separate retired-list object — condemned +/// entries RIDE the source-edge run as `RunMarker::Condemned` sentinel rows at the zero-sentinel key. This reads /// the seal at (snap_generation, snap_attempt), opens every run for `shard`, and reconstructs the /// `RetiredEntry` shape (hash from the run key, the rest from the decoded `CondemnedRow`). Empty when /// gc/state / the seal / the runs are absent. Used by ack-floor tests to assert pending/condemn state. inline std::vector currentRetiredSet( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t shard) { - const auto st = backend.get(layout.gcStateKey()); + OperationForTest operation(backend); + const auto st = (*operation).read(layout.gcStateKey(), DB::Cas::Retry::standard()); if (!st) return {}; const DB::Cas::GcState gc_state = DB::Cas::decodeGcState(st->bytes); if (gc_state.snap_generation == 0) return {}; - const auto seal_bytes = backend.get(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt)); + const auto seal_bytes = (*operation).read(layout.foldSealKey(gc_state.snap_generation, gc_state.snap_attempt), DB::Cas::Retry::standard()); if (!seal_bytes) return {}; const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(seal_bytes->bytes); @@ -644,12 +828,12 @@ inline std::vector currentRetiredSet( { if (run.shard != shard) continue; - auto r = DB::Cas::openSourceEdgeRun(backend, run.key); + auto r = DB::Cas::openSourceEdgeRun(*operation, run.key); String k; String p; while (r.next(k, p)) { - if (p.empty() || p[0] != DB::Cas::kCondemned) + if (p.empty() || DB::Cas::runMarkerFromByte(p[0], "CAS test source-edge run") != DB::Cas::RunMarker::Condemned) continue; DB::Cas::BlobRef ref; DB::UInt128 source_id{}; @@ -668,13 +852,13 @@ inline std::vector currentRetiredSet( return out; } -/// True iff ANY gc-shard's adopted-seal run still holds a `kCondemned` row — the ack-floor deletion -/// pipeline is in flight while this is true (retired-in-snapshot T4 replacement for the old -/// "iterate gc/state.retired_refs" probe). `gc_shards` is read from gc/state when 0 is passed. +/// True iff ANY gc-shard's adopted-seal run still holds a `RunMarker::Condemned` row — the ack-floor deletion +/// pipeline is in flight while this is true. `gc_shards` is read from gc/state when 0 is passed. inline bool anyCondemnedInSeal( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, uint64_t gc_shards = 0) { - const auto st = backend.get(layout.gcStateKey()); + OperationForTest operation(backend); + const auto st = (*operation).read(layout.gcStateKey(), DB::Cas::Retry::standard()); if (!st) return false; const DB::Cas::GcState gc_state = DB::Cas::decodeGcState(st->bytes); @@ -689,25 +873,32 @@ inline bool anyCondemnedInSeal( /// incarnation_tag in its envelope header (preserving header_len + payload), putOverwrite against the /// current token, and return the NEW token. Used to drive the W-REVALIDATE adopt branch (current token /// differs from the writer's stale observation). -inline DB::Cas::Token displaceObjectToken( +inline DB::Cas::Etag displaceObjectToken( DB::Cas::Backend & backend, const String & key, DB::Cas::ObjectKind kind) { - const auto got = backend.get(key); + OperationForTest operation(backend); + const std::optional got = (*operation).read(key, DB::Cas::Retry::standard()); if (!got) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "displaceObjectToken: object {} absent", key); DB::Cas::EnvelopeHeader header = DB::Cas::decodeEnvelopeHeader(got->bytes, got->bytes.size(), kind); - /// A fresh, distinct incarnation_tag forces a distinct body so the displaced token differs. + /// A fresh, distinct incarnation_tag forces a distinct body so the displaced incarnation differs. header.incarnation_tag = header.incarnation_tag + DB::UInt128(1); /// Re-encode at the SAME header length the object was decoded with (the v3 pad target). const String new_head = DB::Cas::encodeEnvelopeHeader(header, header.header_len); const String body = new_head + got->bytes.substr(header.header_len); - return backend.putOverwrite(key, body, got->token).token; + const std::optional displaced = DB::Cas::orThrow( + (*operation).replace(key, body, got->etag, DB::Cas::Retry::standard()), + "displace the object at " + key); + if (!displaced) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "displaceObjectToken: the replace of {} reported no incarnation", key); + return *displaced; } -inline DB::Cas::Token displaceBlobToken( +inline DB::Cas::Etag displaceBlobToken( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::BlobRef & id) { return displaceObjectToken(backend, layout.blobKey(id), DB::Cas::ObjectKind::Blob); @@ -723,8 +914,14 @@ inline DB::Cas::Token displaceBlobToken( /// what Phase-4 Lever A (spec 2026-07-06-cas-gc-round-skip-unchanged) is designed to skip -- passes 0 /// here to force fold-every-round (shouldDeferRound's liveness bound: rounds_since_last_fold(0) >= 0 /// is always true). -inline DB::Cas::PoolPtr openPoolForTest( - std::shared_ptr backend, uint64_t gc_fold_max_defer_rounds = 8) +/// Templated on the backend's own pointer type (an `InMemoryBackend`, one of its many test subclasses, +/// or a decorator that is not itself an `InMemoryBackend`, e.g. `ThrottlingBackend`) rather than fixed +/// to `InMemoryBackend`/`BackendPtr`: a fixed pair of non-template overloads is genuinely AMBIGUOUS for +/// a `shared_ptr` argument, since "derived-to-`InMemoryBackend`" and "derived-to-`Backend`" are +/// equally-ranked conversions with no tiebreaker; template argument deduction has none of that problem. +template +DB::Cas::PoolPtr openPoolForTest( + std::shared_ptr backend, uint64_t gc_fold_max_defer_rounds = 8) { return DB::Cas::Pool::open(std::move(backend), DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", @@ -748,13 +945,14 @@ inline void seedPoolMetaForRestart( DB::Cas::Backend & backend, const String & pool_prefix = "p", uint64_t gc_shards = 1) { const DB::Cas::Layout layout(pool_prefix); + OperationForTest operation(backend); DB::Cas::PoolMeta::createOrValidate( - backend, layout, /*blob_header_len=*/256, gc_shards, + *operation, layout, /*blob_header_len=*/256, gc_shards, DB::Cas::BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/true); - if (!backend.get(layout.refCatalogKey())) - DB::Cas::CasRefCatalog::initializeEmptyForNewPool(backend, layout); + if (!(*operation).read(layout.refCatalogKey(), DB::Cas::Retry::standard())) + DB::Cas::CasRefCatalog::initializeEmptyForNewPool(*operation, layout); else - (void)DB::Cas::CasRefCatalog::read(backend, layout); + (void)DB::Cas::CasRefCatalog::read(*operation, layout); } /// Write a blob object (envelope + payload) addressed by `hash`, so a HEAD returns a token. The bytes @@ -768,7 +966,9 @@ inline void writeBlobBody( header.incarnation_tag = DB::UInt128(0x1234); header.build_id = DB::UInt128(0x5678); const String head = DB::Cas::encodeEnvelopeHeader(header, static_cast(blob_header_len)); - backend.putIfAbsent(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), head + String("x")); + OperationForTest op(backend); + (*op).create(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), head + String("x"), + DB::Cas::Retry::standard()); } /// Write a raw blob body (payload written verbatim, no envelope) — the raw-body-refinement shape @@ -776,7 +976,9 @@ inline void writeBlobBody( inline void writeRawBlobBody(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash, const String & payload) { - backend.casPut(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), payload, std::nullopt); + OperationForTest op(backend); + (*op).create(layout.blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}), payload, + DB::Cas::Retry::standard()); } /// These `UInt128`-hash meta-op wrappers are the pre-mixed-algo 128-bit-only test convenience surface: @@ -794,8 +996,9 @@ inline void writeMetaClean(DB::Cas::Backend & backend, const DB::Cas::Layout & l const DB::UInt128 & hash, uint64_t size) { const DB::Cas::BlobRef ref = legacyMetaTestRef(hash); - backend.putIfAbsent(layout.blobMetaKey(ref), DB::Cas::encodeBlobMeta( - DB::Cas::BlobMeta{.state = DB::Cas::MetaState::Clean, .condemn_round = 0, .size = size})); + OperationForTest op(backend); + (*op).create(layout.blobMetaKey(ref), DB::Cas::encodeBlobMeta( + DB::Cas::BlobMeta{.state = DB::Cas::MetaState::Clean, .condemn_round = 0, .size = size}), DB::Cas::Retry::standard()); } /// Transition an existing meta descriptor to Condemned at `condemn_round`, via a read-modify-CAS on @@ -804,25 +1007,29 @@ inline void condemnMeta(DB::Cas::Backend & backend, const DB::Cas::Layout & layo const DB::UInt128 & hash, uint64_t condemn_round) { const DB::Cas::BlobRef ref = legacyMetaTestRef(hash); - const auto lm = DB::Cas::loadMeta(backend, layout, ref); + OperationForTest operation(backend); + const auto lm = DB::Cas::loadMeta(*operation, layout, ref); ASSERT_TRUE(lm.has_value()); DB::Cas::BlobMeta c = lm->meta; c.state = DB::Cas::MetaState::Condemned; c.condemn_round = condemn_round; - backend.putOverwrite(layout.blobMetaKey(ref), DB::Cas::encodeBlobMeta(c), lm->etag); + ASSERT_TRUE(std::holds_alternative( + DB::Cas::casMeta(*operation, layout, ref, lm->etag, c))); } /// Load the meta descriptor for `hash` via the shared ops layer (nullopt = absent). inline std::optional loadMetaForTest(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::UInt128 & hash) { - return DB::Cas::loadMeta(backend, layout, legacyMetaTestRef(hash)); + OperationForTest operation(backend); + return DB::Cas::loadMeta(*operation, layout, legacyMetaTestRef(hash)); } /// The latest GC generation (snap_generation pointer in gc/state), or 0 when absent. inline uint64_t currentGenerationOf(DB::Cas::Backend & backend, const DB::Cas::Layout & layout) { - const auto got = backend.get(layout.gcStateKey()); + OperationForTest op(backend); + const auto got = (*op).read(layout.gcStateKey(), DB::Cas::Retry::standard()); if (!got) return 0; return DB::Cas::decodeGcState(got->bytes).snap_generation; @@ -831,7 +1038,8 @@ inline uint64_t currentGenerationOf(DB::Cas::Backend & backend, const DB::Cas::L /// The adopted attempt (snap_attempt pointer in gc/state), or 0 when absent. inline uint64_t currentAttemptOf(DB::Cas::Backend & backend, const DB::Cas::Layout & layout) { - const auto got = backend.get(layout.gcStateKey()); + OperationForTest op(backend); + const auto got = (*op).read(layout.gcStateKey(), DB::Cas::Retry::standard()); if (!got) return 0; return DB::Cas::decodeGcState(got->bytes).snap_attempt; @@ -845,9 +1053,10 @@ inline std::vector runsForShard( { const uint64_t gen = currentGenerationOf(backend, layout); const uint64_t attempt = currentAttemptOf(backend, layout); + OperationForTest op(backend); for (uint64_t g = gen; ; --g) { - if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + if (const auto got = (*op).read(layout.foldSealKey(g, attempt), DB::Cas::Retry::standard())) { const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(got->bytes); std::vector out; @@ -861,7 +1070,7 @@ inline std::vector runsForShard( } } -/// Stream the sealed in-degree run segments `runs` and count the active source edges (`kEdgeActive` +/// Stream the sealed in-degree run segments `runs` and count the active source edges (`RunMarker::Edge` /// rows) for `ref`. Test-side replacement for the deleted per-blob point query `inDegreeInGeneration` /// (codecs-v3 phase 5: a `cas_run` is a sequential NDJSON stream with no random access, so a blob's /// in-degree is recomputed by a full stream-and-count rather than a seek). A condemned / zero-marker @@ -869,15 +1078,16 @@ inline std::vector runsForShard( inline int64_t inDegreeInRuns( DB::Cas::Backend & backend, const std::vector & runs, const DB::Cas::BlobRef & ref) { + OperationForTest operation(backend); int64_t degree = 0; for (const DB::Cas::RunRef & run : runs) { - auto r = DB::Cas::openSourceEdgeRun(backend, run.key); + auto r = DB::Cas::openSourceEdgeRun(*operation, run.key); String k; String p; while (r.next(k, p)) { - if (p.empty() || p[0] != DB::Cas::kEdgeActive) + if (p.empty() || DB::Cas::runMarkerFromByte(p[0], "CAS test source-edge run") != DB::Cas::RunMarker::Edge) continue; DB::Cas::BlobRef row_ref; DB::UInt128 source_id{}; @@ -930,8 +1140,9 @@ namespace fixture inline UInt128 catalogLifeIdForTest( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns) { + OperationForTest operation(backend); const std::optional life = - DB::Cas::CasRefCatalog::lifeIfCataloged(backend, layout, ns); + DB::Cas::CasRefCatalog::lifeIfCataloged(*operation, layout, ns); chassert(life.has_value()); return life->incarnation; } @@ -940,8 +1151,8 @@ inline UInt128 catalogLifeIdForTest( /// real round. This is the durable fact the sweep's §6 deletion premise reads /// (`CasOrphanManifestSweep.cpp`): `cursor` is the namespace's `last_folded_ref_id`, and a manifest of /// an epoch-`E` build is deletable only once that cursor sits in an epoch STRICTLY above `E`. -/// `hold`, when set, makes the row classification 4 — the strict grammar `encodeFoldSeal` enforces in -/// both directions, so a hold and a non-4 classification cannot be seeded together. +/// `hold`, when set, makes the row classification `Clamped` — the strict grammar `encodeFoldSeal` +/// enforces in both directions, so a hold and a non-clamped classification cannot be seeded together. /// /// SHARP EDGE, HANDLED HERE SO NO CALLER HAS TO KNOW IT: a fold seal must carry a `condemned_summary` /// entry for EVERY shard in `0..gc_shards-1`. A later real round adopts this object as its PARENT and @@ -961,8 +1172,9 @@ inline void seedFoldCursorForTest( DB::Cas::RefTxnId cursor, std::optional hold = std::nullopt, uint64_t generation = 1, uint64_t attempt = 1) { + OperationForTest operation(backend); DB::Cas::NamespaceLifeId life = fixture::fixtureLife(ns); - const DB::Cas::CasRefCatalog::Snapshot catalog_cut = DB::Cas::CasRefCatalog::read(backend, layout); + const DB::Cas::CasRefCatalog::Snapshot catalog_cut = DB::Cas::CasRefCatalog::read(*operation, layout); const auto catalog_it = std::find_if( catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), [&](const DB::Cas::CatalogEntry & entry) { return entry.ns.string() == ns.string(); }); @@ -972,7 +1184,7 @@ inline void seedFoldCursorForTest( entry.ns = ns; entry.state = DB::Cas::NsState::Live; entry.incarnation = fixture::fixtureLife(ns).incarnation; - DB::Cas::CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + DB::Cas::CasRefCatalog::casAdmitEntry(*operation, layout, 1, entry); life = DB::Cas::NamespaceLifeId::fromCatalogEntry(ns, entry.incarnation); } else @@ -982,21 +1194,21 @@ inline void seedFoldCursorForTest( const String seal_key = layout.foldSealKey(generation, attempt); DB::Cas::CasFoldSeal seal; - const auto existing = backend.get(seal_key); + const auto existing = (*operation).read(seal_key, DB::Cas::Retry::standard()); if (existing) seal = DB::Cas::decodeFoldSeal(existing->bytes); seal.generation = generation; DB::Cas::RefCoverage cov; - cov.classification = hold ? 4 : 2; + cov.classification = hold ? DB::Cas::CoverageClass::Clamped : DB::Cas::CoverageClass::Folded; cov.last_folded_ref_id = cursor; cov.hold = hold; seal.ref_lives[life.incarnation].coverage = cov; DB::Cas::GcState gc_state; - const DB::Cas::HeadResult head = backend.head(layout.gcStateKey()); - if (head.exists) - gc_state = DB::Cas::decodeGcState(backend.get(layout.gcStateKey())->bytes); + const auto existing_state = (*operation).read(layout.gcStateKey(), DB::Cas::Retry::standard()); + if (existing_state) + gc_state = DB::Cas::decodeGcState(existing_state->bytes); /// Totality over `gc_shards` — see the doc comment's SHARP EDGE note for what throws without it. const uint64_t gc_shards = gc_state.gc_shards ? gc_state.gc_shards : 1; @@ -1005,17 +1217,17 @@ inline void seedFoldCursorForTest( const String seal_bytes = DB::Cas::encodeFoldSeal(seal); if (existing) - backend.putOverwrite(seal_key, seal_bytes, existing->token); + (*operation).replace(seal_key, seal_bytes, existing->etag, DB::Cas::Retry::standard()); else - backend.putIfAbsent(seal_key, seal_bytes); + (*operation).create(seal_key, seal_bytes, DB::Cas::Retry::standard()); gc_state.snap_generation = generation; gc_state.snap_attempt = attempt; const String state = DB::Cas::encodeGcState(gc_state); - if (!head.exists) - backend.putIfAbsent(layout.gcStateKey(), state); + if (!existing_state) + (*operation).create(layout.gcStateKey(), state, DB::Cas::Retry::standard()); else - backend.putOverwrite(layout.gcStateKey(), state, head.token); + (*operation).replace(layout.gcStateKey(), state, existing_state->etag, DB::Cas::Retry::standard()); } /// The folded cursor sealed for (ns, shard) by the latest fold seal, or 0 when absent. After a COMPLETE @@ -1026,15 +1238,16 @@ inline uint64_t foldCursorOf( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, uint64_t shard) { chassert(shard == 0); + OperationForTest operation(backend); const std::optional life = - DB::Cas::CasRefCatalog::lifeIfCataloged(backend, layout, ns); + DB::Cas::CasRefCatalog::lifeIfCataloged(*operation, layout, ns); if (!life) return 0; const uint64_t gen = currentGenerationOf(backend, layout); const uint64_t attempt = currentAttemptOf(backend, layout); for (uint64_t g = gen; ; --g) { - if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + if (const auto got = (*operation).read(layout.foldSealKey(g, attempt), DB::Cas::Retry::standard())) { const DB::Cas::CasFoldSeal seal = DB::Cas::decodeFoldSeal(got->bytes); const auto it = seal.ref_lives.find(life->incarnation); @@ -1050,23 +1263,24 @@ inline uint64_t foldCursorOf( /// Set a server root's durable floor (so orphan-sweep eligibility can be driven). After the ack-floor /// merge the floor rides the mount lease body (`mountKey`), so this seeds a MountLease carrying -/// `{writer_epoch, min_active}` — exactly what `prefixEligible` reads. +/// `{writer_epoch, min_active_build_sequence}` — exactly what `prefixEligible` reads. inline void setWatermarkMinActive( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const String & server_root_id, - uint64_t writer_epoch, uint64_t min_active) + uint64_t writer_epoch, uint64_t min_active_build_sequence) { DB::Cas::MountLease m; m.server_uuid = DB::UInt128(0); m.writer_epoch = writer_epoch; - m.min_active = min_active; + m.min_active_build_sequence = min_active_build_sequence; m.seq = 1; m.write_attempt_id = DB::UInt128{1}; const String key = layout.mountKey(server_root_id); - const DB::Cas::HeadResult h = backend.head(key); - if (h.exists) - backend.putOverwrite(key, DB::Cas::encodeMountLease(m), h.token); + OperationForTest op(backend); + const auto h = (*op).head(key, DB::Cas::Retry::standard()); + if (h) + (*op).replace(key, DB::Cas::encodeMountLease(m), h->etag, DB::Cas::Retry::standard()); else - backend.putIfAbsent(key, DB::Cas::encodeMountLease(m)); + (*op).create(key, DB::Cas::encodeMountLease(m), DB::Cas::Retry::standard()); } /// ---- Task 10 ref snapshot+log raw fixtures ---- @@ -1085,9 +1299,12 @@ inline void writeRefSnapshotRaw( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RefTableSnapshot & snapshot) { const DB::Cas::RootNamespace ns{snapshot.ns}; - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + OperationForTest operation(backend); + const NamespaceLifeId life + = CasRefCatalog::lifeIfCataloged(*operation, layout, ns).value_or(fixture::fixtureLife(ns)); const String key = layout.refSnapshotKey(life, snapshot.snapshot_id); - backend.putIfAbsent(key, DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(snapshot))); + (*operation).create(key, DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(snapshot)), + DB::Cas::Retry::standard()); } /// Admits `ns` into the catalog as a `Live` entry, IDEMPOTENTLY (a no-op once `ns` already carries @@ -1112,7 +1329,8 @@ inline void writeRefSnapshotRaw( /// readable `life_epoch`. Ordinary fixtures use `casAdmitRecoverableEntry` below instead. inline void casAdmitEntry(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns) { - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + OperationForTest operation(backend); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(*operation, layout); for (const CatalogEntry & entry : snap.catalog.entries) if (entry.ns.string() == ns.string()) return; /// already admitted -- by an earlier raw write to the same namespace, or by the @@ -1121,7 +1339,7 @@ inline void casAdmitEntry(DB::Cas::Backend & backend, const DB::Cas::Layout & la entry.ns = ns; entry.state = NsState::Live; entry.incarnation = fixture::fixtureLife(ns).incarnation; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(*operation, layout, 1, entry); } namespace fixture @@ -1140,7 +1358,8 @@ inline void writeRecoverableCkptForRawFixture( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, const DB::Cas::RefCkpt & ckpt) { - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + OperationForTest operation(backend); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*operation, layout); const auto it = std::find_if( catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), [&] (const CatalogEntry & entry) { return entry.ns == ns; }); @@ -1149,8 +1368,9 @@ inline void writeRecoverableCkptForRawFixture( "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); - const PutResult put = backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(ckpt)); - if (put.outcome != PutOutcome::Done) + const WriteResult put = (*operation).create(layout.refCkptKey(life), encodeRefCkpt(ckpt), + DB::Cas::Retry::standard()); + if (!std::holds_alternative(put)) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' could not publish its checkpoint", ns.string()); } @@ -1161,7 +1381,8 @@ inline void advanceRecoverableCkptForRawFixture( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, const DB::Cas::RefTxnId & through) { - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + OperationForTest operation(backend); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*operation, layout); const auto it = std::find_if( catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), [&] (const CatalogEntry & entry) { return entry.ns == ns; }); @@ -1170,7 +1391,7 @@ inline void advanceRecoverableCkptForRawFixture( "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); - const std::optional sample = readCkpt(backend, layout, life); + const std::optional sample = readCkpt(*operation, layout, life); if (!sample) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' has no checkpoint to advance", ns.string()); @@ -1182,7 +1403,9 @@ inline void advanceRecoverableCkptForRawFixture( RefCkpt advanced = sample->ckpt; advanced.committed_through = through; - if (backend.casPut(layout.refCkptKey(life), encodeRefCkpt(advanced), sample->token).outcome != CasOutcome::Committed) + if (!std::holds_alternative( + (*operation).replace(layout.refCkptKey(life), encodeRefCkpt(advanced), sample->etag, + DB::Cas::Retry::standard()))) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' could not advance its checkpoint", ns.string()); } @@ -1195,7 +1418,8 @@ inline void replaceRecoverableCkptForRawFixture( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, const DB::Cas::RefCkpt & next) { - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(backend, layout); + OperationForTest operation(backend); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*operation, layout); const auto it = std::find_if( catalog_cut.catalog.entries.begin(), catalog_cut.catalog.entries.end(), [&] (const CatalogEntry & entry) { return entry.ns == ns; }); @@ -1204,7 +1428,7 @@ inline void replaceRecoverableCkptForRawFixture( "raw recovery fixture for namespace '{}' has no catalog entry", ns.string()); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(it->ns, it->incarnation); - const std::optional existing = readCkpt(backend, layout, life); + const std::optional existing = readCkpt(*operation, layout, life); if (!existing) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' has no checkpoint to replace", ns.string()); @@ -1219,7 +1443,9 @@ inline void replaceRecoverableCkptForRawFixture( throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' cannot regress its checkpoint frontier", ns.string()); - if (backend.casPut(layout.refCkptKey(life), encodeRefCkpt(next), existing->token).outcome != CasOutcome::Committed) + if (!std::holds_alternative( + (*operation).replace(layout.refCkptKey(life), encodeRefCkpt(next), existing->etag, + DB::Cas::Retry::standard()))) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "raw recovery fixture for namespace '{}' could not replace its checkpoint", ns.string()); } @@ -1232,20 +1458,24 @@ inline void publishRecoverableCkptForSemanticWrapper( DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const DB::Cas::RootNamespace & ns, const DB::Cas::RefTxnId & txn_id) { - const std::optional life = CasRefCatalog::lifeIfCataloged(backend, layout, ns); + OperationForTest operation(backend); + const std::optional life = CasRefCatalog::lifeIfCataloged(*operation, layout, ns); if (!life) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "semantic ref fixture for namespace '{}' was not admitted", ns.string()); - if (!readCkpt(backend, layout, *life)) - { - const PutResult put = backend.putIfAbsent(layout.refCkptKey(*life), encodeRefCkpt(RefCkpt{ - .life_epoch = 1, - .committed_through = txn_id, - .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt, - })); - if (put.outcome == PutOutcome::Done) + if (!readCkpt(*operation, layout, *life)) + { + const WriteResult put = (*operation).create( + layout.refCkptKey(*life), + encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = txn_id, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }), + DB::Cas::Retry::standard()); + if (std::holds_alternative(put)) return; } @@ -1264,12 +1494,13 @@ inline void casAdmitRecoverableEntry( { casAdmitEntry(backend, layout, ns); - const std::optional life = CasRefCatalog::lifeIfCataloged(backend, layout, ns); + OperationForTest operation(backend); + const std::optional life = CasRefCatalog::lifeIfCataloged(*operation, layout, ns); if (!life) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "recoverable raw fixture for namespace '{}' was not admitted", ns.string()); - if (backend.head(layout.refCkptKey(*life)).exists) + if ((*operation).head(layout.refCkptKey(*life), DB::Cas::Retry::standard())) return; writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ @@ -1294,15 +1525,16 @@ inline DB::Cas::RecoveredRefTable recoverRefTableDetailedAtCatalogCutForTest( if (it != catalog_cut.catalog.entries.end()) catalog_entry = *it; + OperationForTest operation(backend); std::optional ckpt; if (catalog_entry) { const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(catalog_entry->ns, catalog_entry->incarnation); - if (const std::optional sample = readCkpt(backend, layout, life)) + if (const std::optional sample = readCkpt(*operation, layout, life)) ckpt = sample->ckpt; } - return recoverRefTableDetailedFromAuthority(backend, layout, catalog_entry, ckpt); + return recoverRefTableDetailedFromAuthority(*operation, layout, catalog_entry, ckpt); } /// Writes `txn` at `_log/` (create-if-absent). Admits `txn.ns` into the catalog first @@ -1322,9 +1554,11 @@ inline void writeRefLogTxnRaw( { const DB::Cas::RootNamespace ns{txn.ns}; casAdmitEntry(backend, layout, ns); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(backend, layout, ns).value_or(fixture::fixtureLife(ns)); + OperationForTest operation(backend); + const NamespaceLifeId life + = CasRefCatalog::lifeIfCataloged(*operation, layout, ns).value_or(fixture::fixtureLife(ns)); const String key = layout.refLogKey(life, txn.txn_id); - backend.putIfAbsent(key, DB::Cas::sealObject(DB::Cas::FormatId::RefLog, DB::Cas::encodeRefLogTxn(txn))); + (*operation).create(key, DB::Cas::sealObject(DB::Cas::FormatId::RefLog, DB::Cas::encodeRefLogTxn(txn)), DB::Cas::Retry::standard()); } namespace fixture @@ -1445,140 +1679,194 @@ inline std::vector publishCommittedOps(const String & ref_name, return {add, promote}; } -/// Counts head/get/putIfAbsent per key for op-count assertions (Pillar B / A1 tests). -class CountingBackend : public DB::Cas::InMemoryBackend +/// Serves an inner stream in windows of at most `chunk` bytes, and records the largest window it ever +/// handed out. `InMemoryBackend` materializes the whole object behind its stream, so without this a +/// consumer receives every byte as one contiguous window -- it can hold the object entire and still +/// look like a streaming reader, and nothing at the seam can tell the two apart. +/// +/// What arming this proves is what the consumer then DOES: a reader that assumed one contiguous window +/// fails against a chunked source, so the test's own success is the evidence. The recorded window is +/// the bound it succeeded under, not a measurement of the reader's resident memory -- a consumer that +/// copies every window into a buffer of its own is invisible here, as it is to any `ReadBuffer`. +class ChunkedStreamForTest : public DB::ReadBuffer { public: - /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the - /// overrides below would otherwise shadow them for callers holding a concrete backend type. - using DB::Cas::Backend::get; - using DB::Cas::Backend::getStream; - using DB::Cas::Backend::putIfAbsent; - using DB::Cas::Backend::putOverwrite; - using DB::Cas::Backend::casPut; + ChunkedStreamForTest(std::unique_ptr inner_, size_t chunk, + std::shared_ptr> largest_) + : DB::ReadBuffer(nullptr, 0), inner(std::move(inner_)), storage(chunk), largest(std::move(largest_)) + { + } - DB::Cas::HeadResult head(const String & key) override +private: + bool nextImpl() override { + const size_t got = inner->read(storage.data(), storage.size()); + if (got == 0) + return false; + BufferBase::set(storage.data(), got, 0); + uint64_t seen = largest->load(); + while (seen < got && !largest->compare_exchange_weak(seen, got)) { - std::lock_guard lock(count_mutex); - ++head_counts[key]; - ++head_total; } - return InMemoryBackend::head(key); + return true; } - std::optional get(const String & key, DB::Cas::Range range) override + std::unique_ptr inner; + std::vector storage; + std::shared_ptr> largest; +}; + +/// Counts every request per key, for the op-count assertions (Pillar B / A1 tests). +class CountingBackend : public DB::Cas::InMemoryBackend +{ +public: + /// ---- The counters live on the transport primitives, so a request is counted once ---- + /// + /// Whichever surface a caller used, its request passes through one of the primitives below: a + /// legacy verb reaches them through its forwarder, and a `CasOperation` speaks them directly. + /// Counting here is therefore counting requests rather than callers. + /// + /// Each counter ticks BEFORE the request is served, so an injected failure still counts as a + /// request issued. A blob publication is not counted: it reaches the store through `publish`, + /// which no counter below observes. + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++get_counts[key]; - ++get_total; - /// Record the request-size shape per key so streaming-memory gates (Task 3/4) can assert - /// the resident-memory bound at the seam: a whole-object read (range.whole()) is a - /// violation for a run object; a ranged read tracks its MAX window length per key. - if (range.whole()) - ++whole_get_counts[key]; - else - { - const uint64_t len = range.length.has_value() ? *range.length : 0; - uint64_t & mx = max_ranged_get_len[key]; - mx = std::max(mx, len); - } - } - return InMemoryBackend::get(key, range); + tick(get_counts, get_total, key); + return InMemoryBackend::read(key, access); } - std::optional getStream(const String & key, DB::Cas::Range range) override + std::optional head(const String & key, DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++get_stream_counts[key]; - ++get_stream_total; - } - return InMemoryBackend::getStream(key, range); + tick(head_counts, head_total, key); + return InMemoryBackend::head(key, access); } - DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++list_counts[prefix]; - ++list_total; - } - return InMemoryBackend::list(prefix, cursor, limit); + tick(list_counts, list_total, prefix); + return InMemoryBackend::list(prefix, cursor, limit, access); } - DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + /// Create-shaped and replace-shaped writes are counted apart as well as together: whether a write + /// carried a precondition is the only thing about it the transport can still see, and the + /// namespace-file request-profile goldens read the create path off exactly that. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { { std::lock_guard lock(count_mutex); - ++put_counts[key]; - ++put_total; + ++write_counts[key]; + ++write_total; + if (expected_value) + { + ++put_overwrite_counts[key]; + ++put_overwrite_total; + } + else + { + ++put_counts[key]; + ++put_total; + } } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } + /// Every ATTEMPTED delete is counted, whatever the backend answers. The destructive gate's tests + /// assert that a suppressed round issues NONE, and an attempt that came back `Gone` is still an + /// attempt -- counting only successful ones would let a gate that leaks deletes over already-absent + /// keys read as green. + DB::Cas::Backend::RawRemoval remove(const String & key, const String & expected_value, + DB::Cas::TransportAccess & access) override + { + tick(delete_counts, delete_total, key); + return InMemoryBackend::remove(key, expected_value, access); + } - /// Counted separately from `putIfAbsent` and `casPut`, for the same reason those two are separate: a - /// replacement conditioned on an expected token is its own op with its own cost. The namespace-file - /// request-profile goldens tell the create path from the replace path on exactly this counter. - DB::Cas::PutResult putOverwrite(const String & key, const String & bytes, const DB::Cas::Token & expected, - const DB::Cas::ObjectMeta & meta) override + /// One `removeManyWriteOnce` call is one bulk-delete request that names every key it carried; the + /// per-key `delete_counts` grow by one for each key named, exactly as a single-key `remove` would, + /// so `deleteCount(key)` reads the same whichever verb deleted it. + void removeManyWriteOnce(const std::vector & keys, DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++put_overwrite_counts[key]; - ++put_overwrite_total; - } - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + for (const DB::Cas::WriteOnceKey & key : keys) + tick(delete_counts, delete_total, key.str()); + InMemoryBackend::removeManyWriteOnce(keys, access); } - /// Counted separately from `putIfAbsent`: a token-CAS is a DIFFERENT op with a different cost, and - /// the `_ckpt` no-op contract ("identical merged body issues no write") is asserted on exactly this - /// counter -- a create-if-absent count would not see the replace path at all. - DB::Cas::CasResult casPut(const String & key, const String & bytes, - const std::optional & expected, const DB::Cas::ObjectMeta & meta) override + /// A blob publication reaches the store through `publish`, not `write` -- a "zero backend requests" + /// assertion built only from the primitives above would miss one landing. + void publish(const DB::Cas::BlobPublishRequest & request, DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++cas_put_counts[key]; - ++cas_put_total; - } - return InMemoryBackend::casPut(key, bytes, expected, meta); + tick(publish_counts, publish_total, request.destination_key); + InMemoryBackend::publish(request, access); } - /// Every ATTEMPTED delete is counted, whatever the backend answers. The destructive gate's tests - /// assert that a suppressed round issues NONE, and an attempt that came back `NotFound` is still an - /// attempt -- counting only successful ones would let a gate that leaks deletes over already-absent - /// keys read as green. - DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + + std::unique_ptr stream(const String & key, DB::Cas::TransportAccess & access) override { - { - std::lock_guard lock(count_mutex); - ++delete_counts[key]; - ++delete_total; - } - return InMemoryBackend::deleteExact(key, token); + tick(get_stream_counts, get_stream_total, key); + std::unique_ptr opened = InMemoryBackend::stream(key, access); + const size_t chunk = stream_chunk.load(); + if (!opened || chunk == 0) + return opened; + auto chunked = std::make_unique(std::move(opened), chunk, largestChunkSlot(key)); + /// Hand back a buffer whose FIRST window is already loaded, as a network-backed store does: + /// `ObjectStorageBackend::stream` forces that GET so the open's own attempt is what pays for + /// it. A fixture that returned an empty buffer would let a consumer which drops the preloaded + /// window -- and so silently loses the head of every streamed body -- pass its tests. + chunked->nextIfAtEnd(); + return chunked; + } + + /// Serve every stream opened from now on in windows of at most `bytes`, as a network-backed store + /// does. Zero (the default) hands the consumer the whole object at once, which is what this + /// backend's own materialization makes of any stream. A mode rather than a count: `resetCounts` + /// leaves it alone. + void setStreamChunkForTest(size_t bytes) { stream_chunk.store(bytes); } + /// The largest contiguous window any consumer of `key`'s stream was handed. Zero when the key was + /// never streamed. + uint64_t largestStreamChunk(const String & key) const + { + std::lock_guard lock(count_mutex); + const auto it = largest_stream_chunk.find(key); + return it == largest_stream_chunk.end() ? 0 : it->second->load(); } - uint64_t headCount(const String & key) const { return lookup(head_counts, key); } - uint64_t casPutCount(const String & key) const { return lookup(cas_put_counts, key); } - uint64_t putOverwriteCount(const String & key) const { return lookup(put_overwrite_counts, key); } uint64_t getCount(const String & key) const { return lookup(get_counts, key); } + uint64_t headCount(const String & key) const { return lookup(head_counts, key); } + uint64_t listCount(const String & prefix) const { return lookup(list_counts, prefix); } + uint64_t writeCount(const String & key) const { return lookup(write_counts, key); } uint64_t putCount(const String & key) const { return lookup(put_counts, key); } + uint64_t putOverwriteCount(const String & key) const { return lookup(put_overwrite_counts, key); } uint64_t deleteCount(const String & key) const { return lookup(delete_counts, key); } + uint64_t getStreamCount(const String & key) const { return lookup(get_stream_counts, key); } + uint64_t publishCount(const String & key) const { return lookup(publish_counts, key); } + + uint64_t getTotal() const { std::lock_guard lock(count_mutex); return get_total; } + uint64_t headTotal() const { std::lock_guard lock(count_mutex); return head_total; } + uint64_t listTotal() const { std::lock_guard lock(count_mutex); return list_total; } + uint64_t writeTotal() const { std::lock_guard lock(count_mutex); return write_total; } + uint64_t putTotal() const { std::lock_guard lock(count_mutex); return put_total; } + uint64_t putOverwriteTotal() const { std::lock_guard lock(count_mutex); return put_overwrite_total; } uint64_t deleteTotal() const { std::lock_guard lock(count_mutex); return delete_total; } + uint64_t getStreamTotal() const { std::lock_guard lock(count_mutex); return get_stream_total; } + uint64_t publishTotal() const { std::lock_guard lock(count_mutex); return publish_total; } + /// Attempted deletes against any key whose path CONTAINS `substr` — the per-site assertion the /// destructive-gate tests make ("the generation prune deleted nothing", "the sweep deleted nothing"). uint64_t deleteCountForKeysContaining(const String & substr) const { - std::lock_guard lock(count_mutex); - uint64_t total = 0; - for (const auto & [key, n] : delete_counts) - if (key.find(substr) != String::npos) - total += n; - return total; + return sumForKeysContaining({&delete_counts}, substr); + } + + /// The total number of read + stream + create-shaped write requests against any key whose path + /// CONTAINS `substr` (T0 idle-round gate: zero run I/O touches every `.../blob_target/...` key). + uint64_t ioCountForKeysContaining(const String & substr) const + { + return sumForKeysContaining({&get_counts, &get_stream_counts, &put_counts}, substr); } + /// Every key this backend was ever asked to delete, in sorted order — so a failing zero-delete /// assertion names the sites that leaked instead of just reporting a count. std::vector deletedKeys() const @@ -1590,13 +1878,7 @@ class CountingBackend : public DB::Cas::InMemoryBackend keys.push_back(key); return keys; } - uint64_t getStreamCount(const String & key) const { return lookup(get_stream_counts, key); } - uint64_t listCount(const String & prefix) const { return lookup(list_counts, prefix); } - /// The max ranged-get window length observed for `key` (0 if only whole-object gets, or none). - uint64_t maxRangedGetLen(const String & key) const { return lookup(max_ranged_get_len, key); } - /// How many whole-object gets (range.whole()) hit `key` — nonzero flags a resident-memory - /// violation for a run/seal object that a streaming caller must never read whole. - uint64_t wholeGetCount(const String & key) const { return lookup(whole_get_counts, key); } + /// Every key any counted operation was issued against, plus every LIST prefix, sorted and /// de-duplicated. A request-profile gate asserts the SET, not only the totals, so a new request the /// profile does not allow names its own key in the failure instead of moving an anonymous counter. @@ -1605,8 +1887,7 @@ class CountingBackend : public DB::Cas::InMemoryBackend std::lock_guard lock(count_mutex); std::vector keys; for (const std::map * m : - {&head_counts, &get_counts, &put_counts, &put_overwrite_counts, &cas_put_counts, - &get_stream_counts, &list_counts, &delete_counts}) + {&get_counts, &head_counts, &list_counts, &write_counts, &delete_counts, &get_stream_counts}) for (const auto & [key, n] : *m) keys.push_back(key); std::sort(keys.begin(), keys.end()); @@ -1614,48 +1895,40 @@ class CountingBackend : public DB::Cas::InMemoryBackend return keys; } - uint64_t headTotal() const { std::lock_guard lock(count_mutex); return head_total; } - uint64_t getTotal() const { std::lock_guard lock(count_mutex); return get_total; } - uint64_t putTotal() const { std::lock_guard lock(count_mutex); return put_total; } - uint64_t putOverwriteTotal() const { std::lock_guard lock(count_mutex); return put_overwrite_total; } - uint64_t casPutTotal() const { std::lock_guard lock(count_mutex); return cas_put_total; } - uint64_t getStreamTotal() const { std::lock_guard lock(count_mutex); return get_stream_total; } - uint64_t listTotal() const { std::lock_guard lock(count_mutex); return list_total; } - - /// The total number of get + getStream + putIfAbsent operations against any key whose path - /// CONTAINS `substr` (T0 idle-round gate: zero run I/O touches every `.../blob_target/...` key). - uint64_t ioCountForKeysContaining(const String & substr) const - { - std::lock_guard lock(count_mutex); - uint64_t total = 0; - for (const auto & [key, n] : get_counts) - if (key.find(substr) != String::npos) total += n; - for (const auto & [key, n] : get_stream_counts) - if (key.find(substr) != String::npos) total += n; - for (const auto & [key, n] : put_counts) - if (key.find(substr) != String::npos) total += n; - return total; - } - void resetCounts() { std::lock_guard lock(count_mutex); - head_counts.clear(); get_counts.clear(); + head_counts.clear(); + list_counts.clear(); + write_counts.clear(); put_counts.clear(); put_overwrite_counts.clear(); - cas_put_counts.clear(); - get_stream_counts.clear(); - list_counts.clear(); delete_counts.clear(); - max_ranged_get_len.clear(); - whole_get_counts.clear(); - head_total = get_total = put_total = cas_put_total = get_stream_total = list_total = delete_total = 0; - put_overwrite_total = 0; - + get_stream_counts.clear(); + publish_counts.clear(); + largest_stream_chunk.clear(); + get_total = head_total = list_total = write_total = put_total = put_overwrite_total + = delete_total = get_stream_total = publish_total = 0; } private: + std::shared_ptr> largestChunkSlot(const String & key) + { + std::lock_guard lock(count_mutex); + auto & slot = largest_stream_chunk[key]; + if (!slot) + slot = std::make_shared>(0); + return slot; + } + + void tick(std::map & per_key, uint64_t & total, const String & key) + { + std::lock_guard lock(count_mutex); + ++per_key[key]; + ++total; + } + uint64_t lookup(const std::map & m, const String & key) const { std::lock_guard lock(count_mutex); @@ -1663,132 +1936,170 @@ class CountingBackend : public DB::Cas::InMemoryBackend return it == m.end() ? 0 : it->second; } + uint64_t sumForKeysContaining(std::initializer_list *> maps, + const String & substr) const + { + std::lock_guard lock(count_mutex); + uint64_t total = 0; + for (const std::map * m : maps) + for (const auto & [key, n] : *m) + if (key.find(substr) != String::npos) + total += n; + return total; + } + mutable std::mutex count_mutex; - std::map head_counts; + /// Held by `shared_ptr` so a stream outliving the map entry's rehash still records into its own slot. + std::map>> largest_stream_chunk; + std::atomic stream_chunk{0}; std::map get_counts; + std::map head_counts; + std::map list_counts; + std::map write_counts; std::map put_counts; std::map put_overwrite_counts; - std::map cas_put_counts; - std::map get_stream_counts; - std::map list_counts; std::map delete_counts; - std::map max_ranged_get_len; - std::map whole_get_counts; - uint64_t head_total = 0; + std::map get_stream_counts; + std::map publish_counts; uint64_t get_total = 0; + uint64_t head_total = 0; + uint64_t list_total = 0; + uint64_t write_total = 0; uint64_t put_total = 0; uint64_t put_overwrite_total = 0; - uint64_t cas_put_total = 0; - uint64_t get_stream_total = 0; - uint64_t list_total = 0; uint64_t delete_total = 0; + uint64_t get_stream_total = 0; + uint64_t publish_total = 0; }; -/// Records the ORDER of body-PUT / `_ckpt`-CAS operations (so a test can compare indices) and lets a -/// test inject a persistent `Conflict` on one chosen `_ckpt` key -- the same technique -/// `gtest_cas_ref_writer.cpp`'s `RefWriterTestBackend::ckpt_conflict_key`/`ckpt_conflict_count` uses to -/// drive the ledger into `NeedsRecovery`, reproduced here so this suite has no dependency on that file's -/// internal (non-exported) test type. Delegates every operation to `CountingBackend` unchanged, so the -/// per-key counters (`putCount`/`casPutCount`) remain available as the positive control. +/// Records the ORDER of writes (so a test can compare indices) and lets a test refuse or fail chosen +/// writes by key. Delegates every request to `CountingBackend` unchanged, so the per-key counters +/// remain available as the positive control. +/// +/// The journal records the KEY, not a verb: a write reaches the transport as bytes plus an optional +/// precondition, and the ordering tests it serves already distinguish their two subjects (the snapshot +/// body and the checkpoint) by key. class OrderedFaultBackend : public CountingBackend { public: - using CountingBackend::casPut; - using CountingBackend::get; - using CountingBackend::putIfAbsent; - - enum class Op : uint8_t { Put, Cas }; - struct Entry - { - Op op; - String key; - }; - - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - record(Op::Put, key); - if (fail_put_count > 0 && !fail_put_substr.empty() && key.find(fail_put_substr) != String::npos) + switch (claimFault(key)) { - --fail_put_count; - throw Poco::TimeoutException("OrderedFaultBackend: simulated PUT response lost, nothing landed"); + case Fault::Conflict: + /// A refusal (not a thrown/ambiguous response): the caller's own re-read-and-merge loop + /// treats this exactly like a concurrent writer that landed first. + return std::unexpected(RawConflict{}); + case Fault::ResponseLost: + throw Poco::TimeoutException("OrderedFaultBackend: simulated write response lost, nothing landed"); + case Fault::None: + break; } - return CountingBackend::putIfAbsent(key, bytes, meta); + return CountingBackend::write(key, bytes, expected_value, access); } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// Arms a refusal at `key` for the next `count` writes. A COUNT cannot wedge one logical write: + /// the engine reissues an unresolved or refused write until its own retry window closes, so a + /// count the reissues outlive lets the call commit in the end. Use it to bound how much contention + /// a call meets, and `armLatchedWriteConflict` when the call must not commit at all. + void armWriteConflict(const String & key, size_t count) { - record(Op::Cas, key); - if (key == fail_cas_key && fail_cas_count > 0) - { - --fail_cas_count; - /// A `Conflict` (not a thrown/ambiguous response): the caller's own re-read-and-merge loop - /// (`publishCkpt`) treats this exactly like a concurrent writer that landed first, and - /// exhausts `MAX_CKPT_CAS_ATTEMPTS` (100) without ever committing -- deterministically, with - /// no wall-clock wait, since the loop is attempt-bounded rather than only deadline-bounded. - return {CasOutcome::Conflict, {}}; - } - return CountingBackend::casPut(key, bytes, expected, meta); + std::lock_guard lock(mutex); + conflict_key = key; + conflict_count = count; } - /// Arms a persistent CAS conflict at `key` for the next `count` attempts. - void armCasConflict(const String & key, size_t count) + /// Refuses EVERY write of `key` until disarmed with an empty key, so a call meets the refusal on + /// every one of its reissues and reaches its retry deadline without committing. + void armLatchedWriteConflict(const String & key) { - fail_cas_key = key; - fail_cas_count = count; + std::lock_guard lock(mutex); + latched_conflict_key = key; } - /// Arms a persistent, never-committed PUT failure for the next `count` `putIfAbsent` calls whose key - /// contains `substr`: the object is never actually written (unlike a real ambiguous response, which - /// may or may not have landed), so the resolve-by-exact-GET a controlled `CasRequestBudget` with - /// `max_attempts = 1` performs always finds the key absent and classifies the attempt a definite, - /// non-`Committed` failure -- deterministically, with no internal retry and no wall-clock wait. - void armPutFailure(const String & substr, int count) + /// Arms a never-committed write failure for the next `count` writes whose key contains `substr`: + /// the object is never actually written (unlike a real ambiguous response, which may or may not + /// have landed), so a resolve read always finds the key absent and classifies the attempt a + /// definite, non-committed failure. The same count caveat as `armWriteConflict` applies. + void armWriteFailure(const String & substr, int count) { - fail_put_substr = substr; - fail_put_count = count; + std::lock_guard lock(mutex); + failure_substr = substr; + failure_count = count; + } + + /// Loses the response of EVERY write whose key contains `substr` until disarmed with an empty + /// substring -- the latched form of `armWriteFailure`, for a call that must never commit. + void armLatchedWriteFailure(const String & substr) + { + std::lock_guard lock(mutex); + latched_failure_substr = substr; } /// The current length of the journal -- a caller's baseline for `indicesFrom` below, so a query can /// be scoped to "since I last looked" rather than "since the pool opened" (whose earlier entries - /// belong to unrelated setup writes, e.g. the birth transaction's own checkpoint CAS). + /// belong to unrelated setup writes, e.g. the birth transaction's own checkpoint write). size_t journalSize() const { std::lock_guard lock(mutex); return journal.size(); } - /// Every index at or after `from` where `op`/`key` matches, in order. - std::vector indicesFrom(Op op, const String & key, size_t from) const + /// Every index at or after `from` where a write of `key` was issued, in order. + std::vector indicesFrom(const String & key, size_t from) const { std::lock_guard lock(mutex); std::vector result; for (size_t i = from; i < journal.size(); ++i) - if (journal[i].op == op && journal[i].key == key) + if (journal[i] == key) result.push_back(i); return result; } - /// The first index at or after `from` where `op`/`key` matches, if any. - std::optional firstIndexFrom(Op op, const String & key, size_t from) const + /// The first index at or after `from` where a write of `key` was issued, if any. + std::optional firstIndexFrom(const String & key, size_t from) const { - const auto indices = indicesFrom(op, key, from); + const auto indices = indicesFrom(key, from); return indices.empty() ? std::nullopt : std::make_optional(indices.front()); } private: - void record(Op op, const String & key) + enum class Fault : uint8_t { None, Conflict, ResponseLost }; + + /// Journals the write and consumes at most one armed fault, all under one hold: a publisher and a + /// synchronous caller write concurrently in these fixtures, and a counted fault read outside the + /// lock would be handed to both. + Fault claimFault(const String & key) { std::lock_guard lock(mutex); - journal.push_back({op, key}); + journal.push_back(key); + if (!latched_conflict_key.empty() && key == latched_conflict_key) + return Fault::Conflict; + if (key == conflict_key && conflict_count > 0) + { + --conflict_count; + return Fault::Conflict; + } + if (!latched_failure_substr.empty() && key.find(latched_failure_substr) != String::npos) + return Fault::ResponseLost; + if (failure_count > 0 && !failure_substr.empty() && key.find(failure_substr) != String::npos) + { + --failure_count; + return Fault::ResponseLost; + } + return Fault::None; } mutable std::mutex mutex; - std::vector journal; - String fail_cas_key; - size_t fail_cas_count = 0; - String fail_put_substr; - int fail_put_count = 0; + std::vector journal; + String conflict_key; + size_t conflict_count = 0; + String latched_conflict_key; + String failure_substr; + int failure_count = 0; + String latched_failure_substr; }; /// A backend whose LIST permanently omits every key under a chosen prefix while those keys stay fully @@ -1800,7 +2111,7 @@ class OrderedFaultBackend : public CountingBackend /// arithmetic walk that finds it anyway is the property under test -- these fixtures are about the walk, /// not about any one `list` call. /// -/// Erasing keys from a page cannot disturb pagination: `ListPage::next_cursor` is computed by the base +/// Erasing keys from a page cannot disturb pagination: `next_cursor` is computed by the base /// backend before the erase, so the next page still resumes strictly after the last key it returned. /// /// Templated on the base so a suite that also needs request COUNTS composes it over `CountingBackend` @@ -1810,6 +2121,9 @@ template class HintHoleBackendOn : public Base { public: + /// Unhide the legacy `list` name the primitive override below would otherwise shadow. + using Base::list; + /// Hide every key under `prefix` from LIST -- a whole namespace, including objects a later publish /// adds. void hidePrefix(const String & prefix) @@ -1830,8 +2144,8 @@ class HintHoleBackendOn : public Base /// one call, replacing whatever was hidden before. /// /// This is the RustFS defect reproduced as an interface: every one of these keys stays durable and - /// honestly served by `get` / `head` / `putIfAbsent` / `casPut` / `deleteExact`, and only - /// enumeration pretends they are not there. Stating the omission as a SET is what lets a test say + /// honestly served by every other request, and only enumeration pretends they are not there. + /// Stating the omission as a SET is what lets a test say /// the thing the defect report says -- "ids 3 and 4 are invisible while the LATER id 5 is visible" /// -- in one line, instead of assembling it from repeated single-key calls whose combined effect a /// reader has to reconstruct. @@ -1862,14 +2176,15 @@ class HintHoleBackendOn : public Base hidden_prefixes.clear(); } - DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { - DB::Cas::ListPage page = Base::list(prefix, cursor, limit); + DB::Cas::Backend::RawListPage page = Base::list(prefix, cursor, limit, access); std::lock_guard lock(hide_mutex); if (hidden_keys.empty() && hidden_prefixes.empty()) return page; const size_t before = page.keys.size(); - std::erase_if(page.keys, [&](const DB::Cas::ListedKey & k) + std::erase_if(page.keys, [&](const DB::Cas::Backend::RawListedKey & k) { if (hidden_keys.contains(k.key)) return true; @@ -1905,14 +2220,14 @@ inline void rearmMountFenceAfterAnomalyForTest(const DB::Cas::PoolPtr & store) store->armMountFence(DB::UInt128{0, 1}, store->liveWriterEpoch(), store->bootMsNow() + 600000); } -/// Delegates the FIRST matching `putIfAbsent` to `CountingBackend` -- so the write actually LANDS -- -/// and only THEN throws an ambiguous exception, modelling "our own PUT committed but its response was -/// lost". Every later call behaves normally, so a caller that retries the SAME (key, bytes) meets its -/// OWN earlier write as the occupant: the exact input the every-attempt rule's adoption arm adjudicates -/// (`slotOccupy` reports `Occupied` with bytes equal to the attempt's own). +/// Delegates the FIRST matching create-shaped write to `CountingBackend` -- so the write actually +/// LANDS -- and only THEN throws an ambiguous exception, modelling "our own PUT committed but its +/// response was lost". Every later call behaves normally, so a caller that retries the SAME (key, +/// bytes) meets its OWN earlier write as the occupant: the exact input the every-attempt rule's +/// adoption arm adjudicates (`slotOccupy` reports `Occupied` with bytes equal to the attempt's own). /// -/// `key_substr` empty means "the first putIfAbsent of any key"; set it to scope the fault to one key -/// family when the caller drives a whole Pool (whose bootstrap PUTs would otherwise consume the fault). +/// `key_substr` empty means "the first create of any key"; set it to scope the fault to one key +/// family when the caller drives a whole Pool (whose bootstrap writes would otherwise consume it). /// /// Shared rather than TU-local because two suites need exactly this shape: `gtest_cas_slot_occupy.cpp` /// pins the primitive's same-call resolve, and `gtest_cas_ref_wedge_every_attempt.cpp` drives the @@ -1920,8 +2235,6 @@ inline void rearmMountFenceAfterAnomalyForTest(const DB::Cas::PoolPtr & store) class LandedButAckLostOnceBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; - using CountingBackend::get; String key_substr; bool fired = false; /// Also lose the caller's IMMEDIATE resolve read of the same key, once. Needed only by a caller @@ -1931,46 +2244,45 @@ class LandedButAckLostOnceBackend : public CountingBackend /// retry loop -- which is why this defaults off and this file's original caller is unaffected. bool lose_resolve_read = false; - DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (!fired && (key_substr.empty() || key.find(key_substr) != String::npos)) + if (!fired && !expected_value && (key_substr.empty() || key.find(key_substr) != String::npos)) { fired = true; - CountingBackend::putIfAbsent(key, bytes, meta); /// the write LANDS + (void)CountingBackend::write(key, bytes, expected_value, access); /// the write LANDS if (lose_resolve_read) - fail_get_once_key = key; + fail_read_once_key = key; throw Poco::TimeoutException("LandedButAckLostOnceBackend: simulated lost PUT response"); } - return CountingBackend::putIfAbsent(key, bytes, meta); + return CountingBackend::write(key, bytes, expected_value, access); } - std::optional get(const String & key, DB::Cas::Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (!fail_get_once_key.empty() && key == fail_get_once_key) + if (!fail_read_once_key.empty() && key == fail_read_once_key) { - fail_get_once_key.clear(); + fail_read_once_key.clear(); throw Poco::TimeoutException( "LandedButAckLostOnceBackend: simulated lost GET (read response never arrived)"); } - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); } private: - String fail_get_once_key; + String fail_read_once_key; }; -/// A `CountingBackend` that can fault selected PUTs by key substring (skip the first `fault_skip` -/// matches, then fault the next `fault_count`), and can latch a matching PUT mid-flight. Same class of -/// seam as the wedge tests in `gtest_cas_ref_writer.cpp` use +/// A `CountingBackend` that can fault selected create-shaped writes by key substring (skip the first +/// `fault_skip` matches, then fault the next `fault_count`), and can latch a matching write mid-flight. +/// Same class of seam as the wedge tests in `gtest_cas_ref_writer.cpp` use /// (`fault_key_substr`/`corrupt_key_substr`/`armPutBlock`), narrowed to what the ref-lane tests need. /// Shared (rather than TU-local) because the chunk-boundary tests and the post-durable install-safety /// tests need exactly the same seam. class ChunkFaultBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; - using CountingBackend::get; - /// Unresolved -> a lost-response ambiguity, NOTHING landed; with a single-attempt budget this /// wedges the lane and a later resolve proves the key ABSENT. /// LandedThenLost -> our OWN exact bytes land and only the acknowledgement is lost, AND the @@ -1994,23 +2306,29 @@ class ChunkFaultBackend : public CountingBackend Mode mode = Mode::None; int fault_skip = 0; int fault_count = 0; - /// One-shot: the next `get` of exactly this key throws, then it is cleared. Armed by + /// One-shot: the next read of exactly this key throws, then it is cleared. Armed by /// `Mode::LandedThenLost` (see above); settable directly for a bare lost-read fault. - String fail_get_once_key; + String fail_read_once_key; + /// How many times a fault actually fired (a `--fault_count` write, not a skipped or non-matching + /// one), so a caller can prove the double was hit rather than infer it from an outcome that a + /// weaker policy could also produce. + int fault_hits = 0; - std::optional get(const String & key, DB::Cas::Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (!fail_get_once_key.empty() && key == fail_get_once_key) + if (!fail_read_once_key.empty() && key == fail_read_once_key) { - fail_get_once_key.clear(); + fail_read_once_key.clear(); throw Poco::TimeoutException("ChunkFaultBackend: simulated lost GET (read response never arrived)"); } - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); } - DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (mode != Mode::None && !fault_substr.empty() && key.find(fault_substr) != String::npos) + if (mode != Mode::None && !expected_value && !fault_substr.empty() && key.find(fault_substr) != String::npos) { if (fault_skip > 0) { @@ -2019,6 +2337,7 @@ class ChunkFaultBackend : public CountingBackend else if (fault_count > 0) { --fault_count; + ++fault_hits; switch (mode) { case Mode::Unresolved: @@ -2030,8 +2349,8 @@ class ChunkFaultBackend : public CountingBackend /// armed to fail ONCE for this key as well, or it would prove the object durable /// inside this very attempt and the lane would never wedge; the wedge-resolution /// GET a flush later then reads it normally. - CountingBackend::putIfAbsent(key, bytes, meta); - fail_get_once_key = key; + (void)CountingBackend::write(key, bytes, expected_value, access); + fail_read_once_key = key; throw Poco::TimeoutException("ChunkFaultBackend: object landed; response lost"); case Mode::Definite: #if USE_AWS_S3 @@ -2044,7 +2363,8 @@ class ChunkFaultBackend : public CountingBackend case Mode::ForeignConflict: /// A foreign writer lands DIFFERENT bytes at this exact key; then our response is /// lost, so resolve-before-reissue GETs foreign bytes -> CORRUPTED_DATA. - CountingBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT")); + (void)CountingBackend::write(key, bytes + String("\x01_FOREIGN_DIFFERENT"), + expected_value, access); throw Poco::TimeoutException("ChunkFaultBackend: foreign different object landed; response lost"); case Mode::None: break; @@ -2061,7 +2381,7 @@ class ChunkFaultBackend : public CountingBackend block_cv.wait_for(lk, std::chrono::seconds(20), [&] { return !block_armed; }); } } - return CountingBackend::putIfAbsent(key, bytes, meta); + return CountingBackend::write(key, bytes, expected_value, access); } void armBlock(const String & substr) @@ -2099,49 +2419,87 @@ class ChunkFaultBackend : public CountingBackend bool block_entered = false; }; -/// Fault decorator for the condemn-marker gate tests (codex-review triage 2026-07-17 §3.4): while -/// armed, every conditional-write attempt against a blob `.meta` key throws. The request controller -/// exhausts its budget and reports `Unresolved`, so `writeCondemnedMeta` returns false while the round -/// still commits the unconfirmed retired entry. Every other write passes through. Armed by default; -/// disarm (`fail_meta_writes = false`) to model the backend healing. -class MetaWriteFaultBackend : public DB::Cas::InMemoryBackend +/// `ChunkFaultBackend` COUNTS its faults, and a count can no longer make one conclusive: the write +/// engine settles every ambiguity by an exact read and then REISSUES, so a fault that runs out +/// mid-call is answered by the next attempt instead of by the call's own deadline -- which is the +/// whole difference between a wedge and a commit. This keeps the fault armed until the test clears +/// the latch, on BOTH legs: the write's, and the lost read that `Mode::LandedThenLost` arms. The read +/// leg matters just as much, because a readable key proves the commit inside the very same call. +class LatchedChunkFaultBackend : public ChunkFaultBackend { public: - /// Unhide the base convenience overloads (omitted Range/ObjectMeta/expected-token forms): the - /// overrides below would otherwise shadow them for callers holding a concrete backend type. - using DB::Cas::Backend::get; - using DB::Cas::Backend::getStream; - using DB::Cas::Backend::putIfAbsent; - using DB::Cas::Backend::putOverwrite; - using DB::Cas::Backend::casPut; + /// Set after `mode` / `fault_substr` / `fault_skip`; cleared when the scenario is over, so the + /// test's own out-of-band writes and reads are not caught by it. + bool latched = false; - DB::Cas::PutResult putIfAbsent( - const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (fail_meta_writes.load() && key.ends_with(".meta")) - throw std::runtime_error("injected fault: blob meta write lost"); - return InMemoryBackend::putIfAbsent(key, bytes, meta); + if (latched && !fail_read_once_key.empty() && key == fail_read_once_key) + throw Poco::TimeoutException("LatchedChunkFaultBackend: the lost read stays lost"); + return ChunkFaultBackend::read(key, access); } - DB::Cas::PutResult putOverwrite( - const String & key, const String & bytes, const DB::Cas::Token & expected, - const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (fail_meta_writes.load() && key.ends_with(".meta")) - throw std::runtime_error("injected fault: blob meta write lost"); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + if (latched && mode != Mode::None && fault_skip == 0 && !expected_value && !fault_substr.empty() + && key.find(fault_substr) != String::npos) + fault_count = 1; + return ChunkFaultBackend::write(key, bytes, expected_value, access); } - DB::Cas::CasResult casPut(const String & key, const String & bytes, - const std::optional & expected, - const DB::Cas::ObjectMeta & meta) override + /// Disarms completely (not just unlatches): what a caller does right after driving a call to its + /// give-up is a further mutation that must reach the store normally. + void disarm() + { + latched = false; + mode = Mode::None; + fault_count = 0; + fault_skip = 0; + fail_read_once_key.clear(); + } +}; + +/// Fault decorator for the condemn-marker gate tests: while armed, every write against a blob `.meta` +/// key throws. Every other write passes through. Armed by default; disarm +/// (`fail_meta_writes = false`) to model the backend healing. +/// +/// The fault's CLASS is chosen at arming, because the two classes model different failures and the +/// engine treats them differently. `Propagates` is a local error the write loop rethrows on the first +/// attempt, so the caller's own handler sees it at once. `Ambiguous` is a timeout the loop cannot +/// distinguish from a lost response: it resolves by a read and reissues until its policy bound, so a +/// test arming it against a PERMANENT fault must drive the operation's clock or spend the whole +/// retry window in real time. +class MetaWriteFaultBackend : public DB::Cas::InMemoryBackend +{ +public: + enum class FaultKind : uint8_t { Propagates, Ambiguous }; + + /// Fault every `.meta` write with `kind`. Construction arms `Propagates`. + void armWriteFault(FaultKind kind = FaultKind::Propagates) + { + fault_kind.store(kind); + fail_meta_writes.store(true); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (fail_meta_writes.load() && key.ends_with(".meta")) + { + if (fault_kind.load() == FaultKind::Ambiguous) + throw Poco::TimeoutException("injected fault: blob meta write response lost"); throw std::runtime_error("injected fault: blob meta write lost"); - return InMemoryBackend::casPut(key, bytes, expected, meta); + } + return InMemoryBackend::write(key, bytes, expected_value, access); } std::atomic fail_meta_writes{true}; + +private: + std::atomic fault_kind{FaultKind::Propagates}; }; /// Blocks INSIDE a blob-meta mutation until `release` is called, so a test can hold a real meta job in @@ -2153,12 +2511,6 @@ class MetaWriteFaultBackend : public DB::Cas::InMemoryBackend class MetaWriteLatchBackend : public DB::Cas::InMemoryBackend { public: - using DB::Cas::Backend::get; - using DB::Cas::Backend::getStream; - using DB::Cas::Backend::putIfAbsent; - using DB::Cas::Backend::putOverwrite; - using DB::Cas::Backend::casPut; - std::atomic entered{false}; void arm() @@ -2173,33 +2525,19 @@ class MetaWriteLatchBackend : public DB::Cas::InMemoryBackend latch_cv.notify_all(); } - DB::Cas::PutResult putIfAbsent( - const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override - { - waitIfMeta(key); - return InMemoryBackend::putIfAbsent(key, bytes, meta); - } - - DB::Cas::PutResult putOverwrite( - const String & key, const String & bytes, const DB::Cas::Token & expected, - const DB::Cas::ObjectMeta & meta) override - { - waitIfMeta(key); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); - } - - DB::Cas::CasResult casPut(const String & key, const String & bytes, - const std::optional & expected, - const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { waitIfMeta(key); - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } - DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + RawRemoval remove(const String & key, const String & expected_value, + DB::Cas::TransportAccess & access) override { waitIfMeta(key); - return InMemoryBackend::deleteExact(key, token); + return InMemoryBackend::remove(key, expected_value, access); } private: @@ -2224,24 +2562,22 @@ class MetaWriteLatchBackend : public DB::Cas::InMemoryBackend class OutcomeLogFaultBackend : public MetaWriteLatchBackend { public: - using DB::Cas::Backend::get; - using DB::Cas::Backend::putIfAbsent; - std::atomic fail_outcome_logs{false}; - DB::Cas::PutResult putIfAbsent( - const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (fail_outcome_logs.load() && key.contains("outcomes/")) - return DB::Cas::PutResult{.outcome = DB::Cas::PutOutcome::PreconditionFailed, .token = {}}; - return MetaWriteLatchBackend::putIfAbsent(key, bytes, meta); + return std::unexpected(RawConflict{}); + return MetaWriteLatchBackend::write(key, bytes, expected_value, access); } - std::optional get(const String & key, DB::Cas::Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (fail_outcome_logs.load() && key.contains("outcomes/")) return std::nullopt; - return DB::Cas::InMemoryBackend::get(key, range); + return DB::Cas::InMemoryBackend::read(key, access); } }; @@ -2259,40 +2595,27 @@ inline void awaitLatchEntered(MetaWriteLatchBackend & backend) } /// Runs a caller-supplied action ONCE, immediately before the named backend call, so a test can make -/// the mount slot change inside a window `MountLeaseKeeper::claim` holds open. Each hook clears +/// the mount slot change inside a window `MountLeaseRenewer::claim` holds open. Each hook clears /// itself after firing. class MountSlotRaceBackend : public DB::Cas::InMemoryBackend { public: - using DB::Cas::Backend::get; - using DB::Cas::Backend::getStream; - using DB::Cas::Backend::putIfAbsent; - using DB::Cas::Backend::putOverwrite; - using DB::Cas::Backend::casPut; - std::function before_put_if_absent; std::function before_get; std::function before_put_overwrite; - DB::Cas::PutResult putIfAbsent( - const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - fire(before_put_if_absent); - return InMemoryBackend::putIfAbsent(key, bytes, meta); + fire(expected_value ? before_put_overwrite : before_put_if_absent); + return InMemoryBackend::write(key, bytes, expected_value, access); } - std::optional get(const String & key, DB::Cas::Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { fire(before_get); - return InMemoryBackend::get(key, range); - } - - DB::Cas::PutResult putOverwrite( - const String & key, const String & bytes, const DB::Cas::Token & expected, - const DB::Cas::ObjectMeta & meta) override - { - fire(before_put_overwrite); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + return InMemoryBackend::read(key, access); } private: @@ -2306,6 +2629,67 @@ class MountSlotRaceBackend : public DB::Cas::InMemoryBackend } }; +/// The engine reissues an unresolved write until its OWN retry window closes, and that window is +/// measured on a clock the engine reads. Both seams here share one counter -- the sleep the engine +/// performs is what advances the clock -- so a fault that stays armed ends the call at its deadline +/// with no real time passing. Installed on the whole pool, because the ref-lane write, its settling +/// read and the recovery retry loop all pace through the same seam. The pool owns the closures and the +/// closures own the clock, so it outlives everything that can still read it. +class VirtualRetryClock +{ +public: + static std::shared_ptr installOn(const PoolPtr & store) + { + auto clock = std::make_shared(); + store->setCasRequestNowFnForTest(nowFnOf(clock)); + store->setCasRetrySleepForTest(sleepFnOf(clock)); + return clock; + } + + /// The two seams on their own, for a fixture that assembles its own `CasRequests` and ledger + /// rather than a whole `Pool`. Each closure keeps the clock alive. + static std::function nowFnOf(std::shared_ptr owned) + { + return [clock = std::move(owned)] { return clock->nowMs(); }; + } + static std::function sleepFnOf(std::shared_ptr owned) + { + return [clock = std::move(owned)](uint64_t ms) { clock->advance(ms); }; + } + + uint64_t nowMs() const + { + std::lock_guard lock(mutex); + return now_ms; + } + size_t pauseCount() const + { + std::lock_guard lock(mutex); + return pauses; + } + uint64_t longestPause() const + { + std::lock_guard lock(mutex); + return longest_pause; + } + + void advance(uint64_t ms) + { + std::lock_guard lock(mutex); + /// Plus one millisecond, because full jitter can draw a ZERO pause: a clock that does not move + /// would leave the loop reissuing for ever against a fault that never clears. + now_ms += ms + 1; + ++pauses; + longest_pause = std::max(longest_pause, ms); + } + +private: + mutable std::mutex mutex; + uint64_t now_ms = 0; + size_t pauses = 0; + uint64_t longest_pause = 0; +}; + /// Expect a DB::Exception with EXACTLY `expected_code` AND a message containing `expected_substring`. /// Needed wherever several distinct branches share one code: the code alone does not identify which /// one ran, so a test that silently takes the wrong branch would still pass. @@ -2326,3 +2710,13 @@ void expectThrowsCodeWithMessage(int expected_code, const String & expected_subs } } + +/// The mount and GC suites call these two unqualified, under `using namespace DB::Cas;`. Argument- +/// dependent lookup does not reach `DB::Cas::tests` from a `shared_ptr`, so the names +/// are re-exported here rather than moved: the qualified `DB::Cas::tests::` spelling the rest of the +/// tree uses keeps working, and there is still one definition. +namespace DB::Cas +{ +using tests::OperationForTest; +using tests::openRequestsForTest; +} diff --git a/src/Disks/tests/gtest_ca_transaction.cpp b/src/Disks/tests/gtest_ca_transaction.cpp index 324e4ad72bd0..ebf2311a8a11 100644 --- a/src/Disks/tests/gtest_ca_transaction.cpp +++ b/src/Disks/tests/gtest_ca_transaction.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -11,7 +12,6 @@ namespace ProfileEvents { extern const Event CASRefRepoint; -extern const Event CASManifestHead; } namespace DB::ErrorCodes @@ -621,13 +621,9 @@ TEST(CASTransactionRemove, UnlinkStormThenDirDropIsOneRefDrop) EXPECT_EQ(rep.dangling, 0u); } -/// Task 22 (URF plan phase 7): the MergeTree fast-removal shape's per-file ForceFresh proof is -/// memoized per (transaction, ref) in `unlinkFile` — the first unlink's `ForceFresh` `getView` re-proves -/// the manifest body with one HEAD; the rest of the burst reuses that proof via `CachedForLoad`. This -/// pins the HEAD-count side of `UnlinkStormThenDirDropIsOneRefDrop` above (which already pins the -/// repoint count): before this memoization, N unlinks of the same part paid N manifest-body HEADs; now -/// the whole storm-then-drop transaction pays exactly one. -TEST(CASTransactionRemove, UnlinkStormMemoizesOneForceFreshHead) +/// Memoizing the per-file `ForceFresh` read per (transaction, ref) saves a ref resolve and a view +/// rebuild on every file of the burst after the first. +TEST(CASTransactionRemove, UnlinkStormMemoizesOneForceFreshResolve) { auto storage = openTxStorage(); @@ -643,8 +639,16 @@ TEST(CASTransactionRemove, UnlinkStormMemoizesOneForceFreshHead) } ASSERT_TRUE(storage->existsDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0")); - const uint64_t heads_before = ProfileEvents::global_counters[ProfileEvents::CASManifestHead].load(); + /// A warm view-cache hit deliberately emits no `RefResolve`, so this counts exactly the calls that + /// did real resolve work on the part. + size_t resolves = 0; + storage->store()->setEventSink([&](DB::Cas::CasEvent e) + { + if (e.type == DB::Cas::CasEventType::RefResolve && e.ref_name == "all_1_1_0") + ++resolves; + }); + size_t resolves_after_unlinks = 0; /// 2. The MergeTree fast-removal shape: unlink every file one-by-one, THEN removeDirectory — all /// in ONE transaction (mirrors UnlinkStormThenDirDropIsOneRefDrop above). { @@ -652,17 +656,19 @@ TEST(CASTransactionRemove, UnlinkStormMemoizesOneForceFreshHead) tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/checksums.txt", /*if_exists=*/false, /*should_remove_objects=*/true); tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/data.bin", /*if_exists=*/false, /*should_remove_objects=*/true); tx->unlinkFile("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0/txn_version.txt", /*if_exists=*/false, /*should_remove_objects=*/true); + resolves_after_unlinks = resolves; tx->removeDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0"); tx->commit(DB::NoCommitOptions{}); } - /// 3. The whole part is gone, and the three-file unlink storm paid exactly ONE manifest-body HEAD - /// (the first unlink's ForceFresh proof) — not three. removeDirectory clears the staged removal - /// marks, so publishStaging's own (unmemoized) ForceFresh getView never fires for this ref either - /// (see UnlinkStormThenDirDropIsOneRefDrop's zero-repoints assertion above). + storage->store()->setEventSink(nullptr); + + /// 3. The whole part is gone, and the three-file burst resolved it exactly once: the first unlink's + /// ForceFresh read, with the other two served from the retained view it left behind. Without the + /// memo each unlink would resolve again. EXPECT_FALSE(storage->existsDirectory("b09/b09b09b0-0909-4909-8909-090909090909/all_1_1_0")); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASManifestHead].load(), heads_before + 1) - << "unlink-storm-then-dir-drop must pay exactly one ForceFresh manifest-body HEAD, not one per file"; + EXPECT_EQ(resolves_after_unlinks, 1u) + << "an unlink storm over one ref must resolve it once for the whole burst, not once per file"; } /// A create-then-remove of a new part in one transaction must discard both the manifest entries diff --git a/src/Disks/tests/gtest_ca_wiring.cpp b/src/Disks/tests/gtest_ca_wiring.cpp index 9d12f9ccb759..8fe21886f780 100644 --- a/src/Disks/tests/gtest_ca_wiring.cpp +++ b/src/Disks/tests/gtest_ca_wiring.cpp @@ -4,6 +4,24 @@ #include #include #include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include namespace DB::ErrorCodes { @@ -316,20 +334,6 @@ TEST(CASPartPathParser, SplitCacheEvictionStaysCorrect) /// the rewritten ContentAddressedMetadataStorage (real ctor over a Local object storage; the /// backend self-selects EmulatedSingleProcess token semantics). -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include using DB::Cas::tests::idOf; using DB::Cas::tests::u128Of; @@ -396,6 +400,23 @@ std::shared_ptr openWiringStorage() return storage; } +/// The current meta of a present key, read through the request engine over an open fence — the +/// sanctioned way for a fixture to observe a `Pool`'s backend without owning it (`Pool::backend()` is +/// gone; `poolBackendPtr()` is the surviving accessor). +DB::Cas::Meta headMetaOf(const DB::Cas::PoolPtr & pool, const String & key) +{ + DB::Cas::tests::OperationForTest op(*pool->poolBackendPtr()); + const auto h = (*op).head(key, DB::Cas::Retry::standard()); + if (!h.has_value()) + throw std::runtime_error("headMetaOf: key '" + key + "' is absent"); + return *h; +} + +DB::Cas::Etag headIncarnationOf(const DB::Cas::PoolPtr & pool, const String & key) +{ + return headMetaOf(pool, key).etag; +} + /// One part with a content blob, a projection file, and the small per-part files (uuid.txt, /// metadata_version.txt — ordinary Inline entries now, all-tree-part-files Task 6/9), published /// through the real PartWriteTxn into `ns` under `ref`. @@ -576,6 +597,64 @@ TEST(CASWiringRead, BlobViewPlanRidesTheStandardPipeline) } } +/// A retained view whose blob the collector has since removed still plans the read (no I/O), and the +/// read itself throws a typed exception at the first byte; it never returns an empty payload, and +/// the size it reports comes from the manifest, not from the missing object. +TEST(CASWiringRead, DeletedBlobUnderStaleViewFailsTypedNotEmpty) +{ + auto object_storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto settings = DB::Cas::tests::makeSettingsForTest( + "test", std::filesystem::temp_directory_path() / "ca_wiring_stale_view"); + auto storage = std::make_shared( + object_storage, "pool", "srv1", "", nullptr, settings); + storage->startup(); + publishWiredPart(*storage, storage->liveNamespace("a11a11a1-1111-4111-8111-111111111111"), "all_1_1_0"); + + const std::string path = "a11/a11a11a1-1111-4111-8111-111111111111/all_1_1_0/data.bin"; + auto plan_before = storage->getBlobViewPlan(path); + ASSERT_TRUE(plan_before.has_value()); /// warms the view and decode caches + + /// Remove the blob object exactly as GC would once nothing references it. + auto pool = storage->store(); + const String blob_key = pool->layout().blobKey(DB::Cas::tests::idOf("payload-A")); + { + const DB::Cas::Etag incarnation = headIncarnationOf(pool, blob_key); + DB::Cas::tests::OperationForTest op(*pool->poolBackendPtr()); + ASSERT_EQ((*op).remove(blob_key, incarnation, DB::Cas::Retry::standard()), DB::Cas::Removal::Removed); + } + + /// Planning still succeeds from the cached manifest and names the same object. + auto plan_after = storage->getBlobViewPlan(path); + ASSERT_TRUE(plan_after.has_value()); + EXPECT_EQ(plan_after->object.remote_path, plan_before->object.remote_path); + + const auto objects = storage->getStorageObjects(path); + ASSERT_EQ(objects.size(), 1u); + EXPECT_EQ(objects[0].bytes_size, String("payload-A").size()); /// size comes from the manifest + + const DB::Cas::BlobLocation location{ + .key = objects[0].remote_path, + .offset = pool->poolMeta().blob_header_len, + .length = objects[0].bytes_size}; + /// The local object storage opens the file eagerly and maps ENOENT to FILE_DOESNT_EXIST + /// (ReadBufferFromFile); on S3 the same read raises S3_ERROR at the first byte. Either way the + /// failure is a typed exception, never an empty payload. + String got; + int code = 0; + try + { + auto buf = storage->readBlobPayload(location, path, DB::ReadSettings{}); + DB::readStringUntilEOF(got, *buf); + } + catch (const DB::Exception & e) + { + code = e.code(); + } + EXPECT_EQ(code, DB::ErrorCodes::FILE_DOESNT_EXIST) + << "read of a deleted blob returned " << got.size() << " bytes (code " << code << ")"; + EXPECT_TRUE(got.empty()); +} + TEST(CASWiringRead, ProjectionDirectory) { auto storage = openWiringStorage(); @@ -1651,8 +1730,8 @@ TEST(CASWiringExchange, AdoptPartFromManifestPublishesFreshLocalManifest) /// the receiver never overwrites or re-creates the blobs — their head tokens are unchanged. const auto data_key = storage->store()->layout().blobKey(idOf("payload-A")); const auto proj_key = storage->store()->layout().blobKey(idOf("payload-B")); - const auto data_tok_before = storage->store()->backend().head(data_key).token; - const auto proj_tok_before = storage->store()->backend().head(proj_key).token; + const auto data_tok_before = headIncarnationOf(storage->store(), data_key); + const auto proj_tok_before = headIncarnationOf(storage->store(), proj_key); /// Adopt into a DIFFERENT table (a22a22a2-2222-4222-8222-222222222222). The transferred body's root_namespace_id is the sender's /// (a11a11a1-1111-4111-8111-111111111111) — the receiver must IGNORE it and use a22a22a2-2222-4222-8222-222222222222. @@ -1676,9 +1755,9 @@ TEST(CASWiringExchange, AdoptPartFromManifestPublishesFreshLocalManifest) EXPECT_FALSE(receiver_ns.string() == sender_ns.string()); /// NO blob body was uploaded: the shared blobs' incarnations are untouched by the adopt. - EXPECT_EQ(storage->store()->backend().head(data_key).token, data_tok_before) + EXPECT_EQ(headIncarnationOf(storage->store(), data_key), data_tok_before) << "adopt-from-manifest must not re-upload a blob already in the shared pool"; - EXPECT_EQ(storage->store()->backend().head(proj_key).token, proj_tok_before); + EXPECT_EQ(headIncarnationOf(storage->store(), proj_key), proj_tok_before); } /// B7 fail-closed: if a referenced blob is absent/condemned in the pool, adoptPartFromManifest must @@ -1706,10 +1785,11 @@ TEST(CASWiringExchange, AdoptFailsClosedAndFallsBackOnCondemnedBlob) /// Artificially delete a referenced pool blob — the live-sender invariant excludes this on the real /// path; §4 promote does not re-probe it, so adopt trusts the manifest edge and publishes. const auto data_key = storage->store()->layout().blobKey(idOf("payload-A")); - const auto h = storage->store()->backend().head(data_key); - ASSERT_TRUE(h.exists); - ASSERT_EQ(storage->store()->backend().deleteExact(data_key, h.token).kind, - DB::Cas::DeleteOutcome::Kind::Deleted); + { + const DB::Cas::Etag incarnation = headIncarnationOf(storage->store(), data_key); + DB::Cas::tests::OperationForTest op(*storage->store()->poolBackendPtr()); + ASSERT_EQ((*op).remove(data_key, incarnation, DB::Cas::Retry::standard()), DB::Cas::Removal::Removed); + } /// §4: promote trusts the adopted leaves — no re-probe — so adopt SUCCEEDS (returns true) and publishes. const bool ok = adoptPartFromManifestAndPromote(*exchange, kReceiverTmpFetchPath, bytes); @@ -1903,7 +1983,7 @@ TEST(CASWiringExchange, AdoptIntoADetachedTargetPublishesADetachedRefAndNoLiveRe EXPECT_TRUE(storage->existsFile(detached_tmp_path + "/p.proj/data.bin")); EXPECT_TRUE(storage->existsFile(detached_tmp_path + "/uuid.txt")); - /// Finalization, unchanged by this task: `IMergeTreeDataPart::renameTo(detached/)` is a + /// Finalization: `IMergeTreeDataPart::renameTo(detached/)` is a /// moveDirectory of the staged dir to its final detached name, which on a content-addressed disk is /// a ref repoint WITHIN the same namespace -- the same shape the active path's /// `renameTempPartAndReplace` uses, and the reason the relinked detached part needs no new @@ -1977,10 +2057,6 @@ TEST(CASWiringExchangeDeathTest, PrepareAdoptRefusesATargetThatIsNotAPartDirecto /// ==== Commit atomicity (B122): a publish failing mid-loop must not leave a PARTIAL commit ==== -#include -#include -#include -#include namespace DB::ErrorCodes { @@ -2825,21 +2901,22 @@ DB::Cas::PoolPtr openResurrectStore(std::shared_ptr & } /// Condemn (kind=Blob, hash, token) by seeding gc/state + a per-shard retired set (the durable GC ledger -/// shape — RetiredEntry, exact-token delete, unchanged by this task) AND condemning the per-hash freshness +/// shape — `RetiredEntry`, exact-token delete) AND condemning the per-hash freshness /// meta, which is what the writer's condemned decision ACTUALLY point-reads (spec §meta-protocols v3). /// Bumps the round so the retirement is a fresh one; leaves the object itself in place (condemn, NOT delete). void seedCondemnBlobToken(DB::Cas::Pool & store, const DB::UInt128 & hash, - [[maybe_unused]] const DB::Cas::Token & token, [[maybe_unused]] uint64_t size) + [[maybe_unused]] const DB::Cas::Etag & token, [[maybe_unused]] uint64_t size) { using namespace DB::Cas; - Backend & b = store.backend(); + Backend & b = *store.poolBackendPtr(); const Layout & layout = store.layout(); + DB::Cas::tests::OperationForTest op(b); GcState state; - const HeadResult head = b.head(layout.gcStateKey()); - if (head.exists) + const auto head = (*op).head(layout.gcStateKey(), Retry::standard()); + if (head.has_value()) { - const auto got = b.get(layout.gcStateKey()); + const auto got = (*op).read(layout.gcStateKey(), Retry::standard()); state = decodeGcState(got->bytes); } state.round += 1; @@ -2848,10 +2925,10 @@ void seedCondemnBlobToken(DB::Cas::Pool & store, const DB::UInt128 & hash, /// GC snapshot runs, which this writer-side edge-protection test does not exercise. The writer's /// condemned decision point-reads the per-hash freshness meta (condemned below), so bumping the round /// and condemning the meta is enough. - if (head.exists) - b.putOverwrite(layout.gcStateKey(), encodeGcState(state), head.token); + if (head.has_value()) + (void)(*op).replace(layout.gcStateKey(), encodeGcState(state), head->etag, Retry::standard()); else - b.putIfAbsent(layout.gcStateKey(), encodeGcState(state)); + (void)(*op).create(layout.gcStateKey(), encodeGcState(state), Retry::standard()); /// The writer's fresh upload (putBlob) already wrote a Clean meta for `hash` (Task 3), so this is a /// plain Clean -> Condemned CAS — exactly what GC's real condemn path does. @@ -2884,12 +2961,11 @@ TEST(CASWiringResurrect, PromoteIgnoresCondemnedMaterializedBlobEdgeProtected) /// Condemn the freshly-uploaded blob's CURRENT token (GC condemning the not-yet-folded fresh incarnation). const String blob_key = store->layout().blobKey(idOf(P)); - const HeadResult h1 = store->backend().head(blob_key); - ASSERT_TRUE(h1.exists); - const Token t0 = h1.token; + const Meta h1 = headMetaOf(store, blob_key); + const Etag t0 = h1.etag; seedCondemnBlobToken(*store, u128Of(P), t0, h1.size); { - const auto lm = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + const auto lm = DB::Cas::tests::loadMetaForTest(*store->poolBackendPtr(), store->layout(), u128Of(P)); ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned) << "precondition: the putBlob'd token must be condemned before promote"; } @@ -2900,9 +2976,7 @@ TEST(CASWiringResurrect, PromoteIgnoresCondemnedMaterializedBlobEdgeProtected) /// The ref is committed and the blob's token is unchanged — no replacement PUT ran (`Materialized` leaves are /// not re-validated: EDGE-BEFORE-OBSERVE guarantees the condemnation is doomed, not the blob). EXPECT_TRUE(store->resolveRef(ns, ref).has_value()) << "the ref must resolve after promote"; - const HeadResult h2 = store->backend().head(blob_key); - ASSERT_TRUE(h2.exists); - EXPECT_EQ(h2.token, t0) + EXPECT_EQ(headIncarnationOf(store, blob_key), t0) << "materialized leaf is edge-protected: promote must not re-upload it (token unchanged)"; } @@ -2940,18 +3014,17 @@ TEST(CASWiringResurrect, PromoteWithoutLivePrecommitAbortsWithoutResurrect) /// Physical publication through `putBlob` requires that durable edge, while this test deliberately /// needs the owner binding absent when `promote` runs. DB::Cas::tests::writeBlobRaw( - store->backend(), store->layout(), P, store->poolMeta().blob_header_len, store->poolMeta().pool_id); - DB::Cas::tests::writeMetaClean(store->backend(), store->layout(), u128Of(P), P.size()); + *store->poolBackendPtr(), store->layout(), P, store->poolMeta().blob_header_len, store->poolMeta().pool_id); + DB::Cas::tests::writeMetaClean(*store->poolBackendPtr(), store->layout(), u128Of(P), P.size()); const ManifestId id = build->stageManifest({wiringBlobEntry("data.bin", P)}); const String blob_key = store->layout().blobKey(idOf(P)); - const HeadResult h1 = store->backend().head(blob_key); - ASSERT_TRUE(h1.exists); + const Meta h1 = headMetaOf(store, blob_key); /// Condemn the leaf so that, were the blob gate reached, promote would republish it — proving the abort /// happens strictly BEFORE any blob work. - seedCondemnBlobToken(*store, u128Of(P), h1.token, h1.size); + seedCondemnBlobToken(*store, u128Of(P), h1.etag, h1.size); { - const auto lm = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + const auto lm = DB::Cas::tests::loadMetaForTest(*store->poolBackendPtr(), store->layout(), u128Of(P)); ASSERT_TRUE(lm.has_value() && lm->meta.state == MetaState::Condemned); } @@ -2968,11 +3041,9 @@ TEST(CASWiringResurrect, PromoteWithoutLivePrecommitAbortsWithoutResurrect) /// No blob work ran before the abort: the leaf's token is UNCHANGED (still the condemned one) and its /// metadata is still Condemned — the owner check aborts before any blob publication. - const HeadResult h2 = store->backend().head(blob_key); - ASSERT_TRUE(h2.exists); - EXPECT_EQ(h2.token, h1.token) + EXPECT_EQ(headIncarnationOf(store, blob_key), h1.etag) << "the aborting path must perform no PUT — the materialized leaf is untouched"; - const auto lm_after = DB::Cas::tests::loadMetaForTest(store->backend(), store->layout(), u128Of(P)); + const auto lm_after = DB::Cas::tests::loadMetaForTest(*store->poolBackendPtr(), store->layout(), u128Of(P)); EXPECT_TRUE(lm_after.has_value() && lm_after->meta.state == MetaState::Condemned) << "no republication before the owner check — the token is still the condemned one"; } diff --git a/src/Disks/tests/gtest_cas_b140_dangle.cpp b/src/Disks/tests/gtest_cas_b140_dangle.cpp index 05af3e47e1cf..9ec0cb9d6dcf 100644 --- a/src/Disks/tests/gtest_cas_b140_dangle.cpp +++ b/src/Disks/tests/gtest_cas_b140_dangle.cpp @@ -117,11 +117,14 @@ TEST(CASGCDangle, SharedBlobSurvivesDropOfOneOfTwoLiveRefs) const FsckReport rep = runFsck(*s, /*detail=*/true); + DB::Cas::tests::OperationForTest op(*b); + const bool b_present = (*op).head(s->layout().blobKey(idOf("B")), DB::Cas::Retry::once()).has_value(); + /// THE DANGLE ASSERTION: GC must NEVER delete a blob a live ref references. EXPECT_EQ(rep.dangling, 0u) << "B140-dangle: GC deleted shared blob B still referenced by the live ref rb_cur " << "after " << rounds << " rounds (dangling=" << rep.dangling << ", reachable=" << rep.reachable - << ", B_present=" << b->head(s->layout().blobKey(idOf("B"))).exists << ")."; - EXPECT_TRUE(b->head(s->layout().blobKey(idOf("B"))).exists) + << ", B_present=" << b_present << ")."; + EXPECT_TRUE(b_present) << "shared blob B must remain present while rb_cur references it"; } diff --git a/src/Disks/tests/gtest_cas_backend.cpp b/src/Disks/tests/gtest_cas_backend.cpp index 6a01821d9472..04052060d7eb 100644 --- a/src/Disks/tests/gtest_cas_backend.cpp +++ b/src/Disks/tests/gtest_cas_backend.cpp @@ -5,11 +5,17 @@ #include #include #include +#include +#include #include #include #include +#include #include #include +#include +#include +#include #include #include @@ -18,13 +24,10 @@ #include #if USE_AWS_S3 -#include #include #include -#include #include #include -#include #include #include #include @@ -34,10 +37,16 @@ using namespace DB::Cas; +using DB::Cas::tests::expectBytes; +using DB::Cas::tests::openRequestsForTest; +using DB::Cas::tests::OperationForTest; + namespace DB::ErrorCodes { +extern const int CAS_DELETE_MARKER; extern const int CORRUPTED_DATA; extern const int NOT_IMPLEMENTED; +extern const int LOGICAL_ERROR; } namespace @@ -109,10 +118,10 @@ BlobPublishRequest countedLongPublication( class PublishCountingInMemoryBackend final : public InMemoryBackend { public: - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { ++publish_calls; - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); } size_t publish_calls = 0; @@ -120,105 +129,14 @@ class PublishCountingInMemoryBackend final : public InMemoryBackend } -/// Minimal concrete implementation that overrides every pure virtual with trivial defaults. -/// Purpose: verify the interface compiles, is overridable, and result-type defaults are sane. -struct NullBackend final : Backend -{ - std::optional get(const String & /*key*/, Range /*range*/) override - { - return std::nullopt; - } - - std::optional getStream(const String & /*key*/, Range /*range*/) override - { - return std::nullopt; - } - - HeadResult head(const String & /*key*/) override - { - return HeadResult{}; - } - - PutResult putIfAbsent(const String & /*key*/, const String & /*bytes*/, const ObjectMeta & /*meta*/) override - { - return {PutOutcome::Done, {}}; - } - - void publishBlob(const BlobPublishRequest & /*request*/) override - { - } - - PutResult putOverwrite(const String & /*key*/, const String & /*bytes*/, const Token & /*expected*/, const ObjectMeta & /*meta*/) override - { - return {PutOutcome::PreconditionFailed, {}}; - } - - CasResult casPut(const String & /*key*/, const String & /*bytes*/, const std::optional & /*expected*/, const ObjectMeta & /*meta*/) override - { - return {CasOutcome::Conflict, {}}; - } - - DeleteOutcome deleteExact(const String & /*key*/, const Token & /*token*/) override - { - return DeleteOutcome{}; - } - - ListPage list(const String & /*prefix*/, const String & /*cursor*/, size_t /*limit*/) override - { - return ListPage{}; - } - - bool supportsListTokens() const override { return false; } -}; - -TEST(CASBackend, PublishBlobReturnsNoIncarnationToken) -{ - static_assert(std::is_same_v< - decltype(std::declval().publishBlob(std::declval())), - void>); -} - -TEST(CASBackend, NullBackendShapeAndDefaults) -{ - NullBackend b; - // Use the base-class reference so virtual dispatch uses base-class default args. - Backend & ref = b; - - // get returns absent - EXPECT_FALSE(ref.get("k").has_value()); - - // head returns non-existent - HeadResult h = b.head("k"); - EXPECT_FALSE(h.exists); - EXPECT_EQ(h.size, 0u); - EXPECT_TRUE(h.token.empty()); - - // putIfAbsent returns Done - EXPECT_EQ(ref.putIfAbsent("k", "v").outcome, PutOutcome::Done); - - // putOverwrite returns PreconditionFailed - EXPECT_EQ(ref.putOverwrite("k", "v", Token{}).outcome, PutOutcome::PreconditionFailed); - - // casPut returns Conflict - EXPECT_EQ(ref.casPut("k", "v", std::nullopt).outcome, CasOutcome::Conflict); - - // deleteExact default kind is NotFound - DeleteOutcome d = b.deleteExact("k", Token{}); - EXPECT_EQ(d.kind, DeleteOutcome::Kind::NotFound); - EXPECT_FALSE(d.created_delete_marker); - - // list returns empty page - ListPage page = b.list("p/", "", 10); - EXPECT_TRUE(page.keys.empty()); - EXPECT_TRUE(page.next_cursor.empty()); - - // Range::whole() helper - EXPECT_TRUE(Range{}.whole()); - Range r1; r1.offset = 1; - EXPECT_FALSE(r1.whole()); - Range r2; r2.length = 5u; - EXPECT_FALSE(r2.whole()); -} +/// `NullBackend` and its two tests (`PublishBlobReturnsNoIncarnationToken`, +/// `NullBackendShapeAndDefaults`) are deleted here: their entire subject was the shape and defaults of +/// the legacy Token-typed forwarders (get/head/putIfAbsent/putOverwrite/casPut/deleteExact/list), which +/// no longer exist -- `Backend` now declares only the primitives, all pure virtual, with no default +/// bodies to pin. The primitive surface's own shape is exercised by every concrete-backend test below +/// (`CASInMemory`, `CASObjectStorageBackend`) through `CasRequests`/`CasOperation`, and the request +/// engine's own default behaviour (a `create` finding the key occupied, a `replace` losing its +/// precondition, a `remove` of an absent key) is pinned in `gtest_cas_requests.cpp`. // ===================================================================== // Task 3: CasInMemoryBackend — enforcing token semantics @@ -227,89 +145,117 @@ TEST(CASBackend, NullBackendShapeAndDefaults) TEST(CASInMemory, PutIfAbsentAndGet) { InMemoryBackend b; - const auto put = b.putIfAbsent("k", "v1"); - const Token t1 = put.token; - EXPECT_EQ(put.outcome, PutOutcome::Done); - EXPECT_FALSE(t1.empty()); - EXPECT_EQ(b.putIfAbsent("k", "clobber").outcome, PutOutcome::PreconditionFailed); - auto g = b.get("k"); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + + const WriteResult put = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); + const Etag t1 = std::get(put).etag; + + const WriteResult clobber = op.create("k", "clobber", Retry::once()); + EXPECT_TRUE(std::holds_alternative(clobber)); + + auto g = op.read("k", Retry::once()); ASSERT_TRUE(g.has_value()); EXPECT_EQ(g->bytes, "v1"); - EXPECT_EQ(g->token, t1); - EXPECT_FALSE(b.get("absent").has_value()); + EXPECT_EQ(g->etag, t1); + EXPECT_FALSE(op.read("absent", Retry::once()).has_value()); } TEST(CASInMemory, OverwriteIsTokenExactAndMintsFreshToken) { InMemoryBackend b; - const Token t1 = b.putIfAbsent("k", "v1").token; - EXPECT_EQ(b.putOverwrite("k", "v2", Token{"wrong", TokenType::Emulated}).outcome, PutOutcome::PreconditionFailed); - EXPECT_EQ(b.get("k")->bytes, "v1"); // untouched on mismatch - const auto overwrite = b.putOverwrite("k", "v2", t1); - EXPECT_EQ(overwrite.outcome, PutOutcome::Done); - EXPECT_NE(overwrite.token, t1); // tokens never repeat - EXPECT_EQ(b.get("k")->bytes, "v2"); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + + const Etag t1 = std::get(op.create("k", "v1", Retry::once())).etag; + /// A stale precondition for the SAME key: an Etag is bound to the key it was minted for, so a + /// cross-key Etag is a caller bug (LOGICAL_ERROR), not "the wrong token" any more -- a stale + /// same-key incarnation is the real-world shape a precondition mismatch has to cover instead. + const Etag t2 = std::get(op.replace("k", "v1.5", t1, Retry::once())).etag; + EXPECT_TRUE(std::holds_alternative(op.replace("k", "v2", t1, Retry::once()))); + expectBytes(b, "k", "v1.5"); // untouched on mismatch + + const WriteResult overwrite = op.replace("k", "v2", t2, Retry::once()); + ASSERT_TRUE(std::holds_alternative(overwrite)); + EXPECT_NE(std::get(overwrite).etag, t2); // tokens never repeat + expectBytes(b, "k", "v2"); } TEST(CASInMemory, CasPutCreateAndSwap) { InMemoryBackend b; - const auto create = b.casPut("m", "s1", std::nullopt); - const Token t1 = create.token; - EXPECT_EQ(create.outcome, CasOutcome::Committed); // create-if-absent - EXPECT_EQ(b.casPut("m", "s1x", std::nullopt).outcome, CasOutcome::Conflict); // exists now - EXPECT_EQ(b.casPut("m", "s2", Token{"stale", TokenType::Emulated}).outcome, CasOutcome::Conflict); - EXPECT_EQ(b.get("m")->bytes, "s1"); - EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Committed); - EXPECT_EQ(b.get("m")->bytes, "s2"); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + + const WriteResult create = op.create("m", "s1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(create)); // create-if-absent + const Etag t1 = std::get(create).etag; + EXPECT_TRUE(std::holds_alternative(op.create("m", "s1x", Retry::once()))); // exists now + + /// A stale, same-key precondition -- see OverwriteIsTokenExactAndMintsFreshToken for why a + /// cross-key Etag can no longer stand in for "the wrong token". + const Etag t2 = std::get(op.replace("m", "s1.5", t1, Retry::once())).etag; + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t1, Retry::once()))); + EXPECT_EQ(op.read("m", Retry::once())->bytes, "s1.5"); + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t2, Retry::once()))); + EXPECT_EQ(op.read("m", Retry::once())->bytes, "s2"); } TEST(CASInMemory, DeleteExactEnforced) { InMemoryBackend b; - const Token t1 = b.putIfAbsent("k", "v1").token; - auto d1 = b.deleteExact("k", Token{"wrong", TokenType::Emulated}); - EXPECT_EQ(d1.kind, DeleteOutcome::Kind::TokenMismatch); - EXPECT_TRUE(b.get("k").has_value()); // SURVIVES wrong-token delete - auto d2 = b.deleteExact("k", t1); - EXPECT_EQ(d2.kind, DeleteOutcome::Kind::Deleted); - EXPECT_FALSE(d2.created_delete_marker); - EXPECT_FALSE(b.get("k").has_value()); - EXPECT_EQ(b.deleteExact("k", t1).kind, DeleteOutcome::Kind::NotFound); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + + const Etag t0 = std::get(op.create("k", "v1", Retry::once())).etag; + const Etag t1 = std::get(op.replace("k", "v1b", t0, Retry::once())).etag; + /// t0 is now stale for this SAME key -- see OverwriteIsTokenExactAndMintsFreshToken for why a + /// cross-key Etag can no longer stand in for "the wrong token". + EXPECT_EQ(op.remove("k", t0, Retry::once()), Removal::Mismatch); + EXPECT_TRUE(op.read("k", Retry::once()).has_value()); // SURVIVES wrong-token delete + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Removed); + EXPECT_FALSE(op.read("k", Retry::once()).has_value()); + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Gone); } -TEST(CASInMemory, RangeGetAndHeadAndList) +TEST(CASInMemory, GetAndHeadAndList) { InMemoryBackend b; - b.putIfAbsent("p/a", "0123456789"); - b.putIfAbsent("p/b", "xy"); - b.putIfAbsent("q/c", "z"); - EXPECT_EQ(b.get("p/a", Range{.offset = 2, .length = 3})->bytes, "234"); - auto h = b.head("p/a"); - EXPECT_TRUE(h.exists); - EXPECT_EQ(h.size, 10u); - auto page = b.list("p/", "", 10); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + + op.create("p/a", "0123456789", Retry::once()); + op.create("p/b", "xy", Retry::once()); + op.create("q/c", "z", Retry::once()); + EXPECT_EQ(op.read("p/a", Retry::once())->bytes, "0123456789"); + auto h = op.head("p/a", Retry::once()); + ASSERT_TRUE(h.has_value()); + EXPECT_EQ(h->size, 10u); + auto page = op.list("p/", "", 10, Retry::once()); ASSERT_EQ(page.keys.size(), 2u); // sorted, prefix-scoped EXPECT_EQ(page.keys[0].key, "p/a"); EXPECT_EQ(page.keys[1].key, "p/b"); EXPECT_TRUE(page.next_cursor.empty()); - auto page1 = b.list("p/", "", 1); // pagination + auto page1 = op.list("p/", "", 1, Retry::once()); // pagination EXPECT_EQ(page1.keys.size(), 1u); EXPECT_EQ(page1.keys[0].key, "p/a"); EXPECT_EQ(page1.next_cursor, "p/a"); EXPECT_FALSE(page1.next_cursor.empty()); - auto page2 = b.list("p/", page1.next_cursor, 1); + auto page2 = op.list("p/", page1.next_cursor, 1, Retry::once()); EXPECT_EQ(page2.keys[0].key, "p/b"); } TEST(CASInMemory, PublishBlobStreamingWritesFreshEnvelopeAndExactPayload) { InMemoryBackend backend; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const auto request = streamingPublication("blob", "fresh-envelope", "payload", 7); - backend.publishBlob(request); + op.publish(request, Retry::once()); - const auto result = backend.get("blob"); + const auto result = op.read("blob", Retry::once()); ASSERT_TRUE(result.has_value()); EXPECT_EQ(result->bytes, "fresh-envelopepayload"); } @@ -317,8 +263,10 @@ TEST(CASInMemory, PublishBlobStreamingWritesFreshEnvelopeAndExactPayload) TEST(CASInMemory, PublishBlobRejectsShortAndLongStreamingSourcesWithoutVisibility) { InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("short", "old-short").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent("long", "old-long").outcome, PutOutcome::Done); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("short", "old-short", Retry::once()))); + ASSERT_TRUE(std::holds_alternative(op.create("long", "old-long", Retry::once()))); for (const auto & [key, payload, declared_size] : std::vector>{ {"short", "abc", 4}, @@ -327,7 +275,7 @@ TEST(CASInMemory, PublishBlobRejectsShortAndLongStreamingSourcesWithoutVisibilit const auto request = streamingPublication(key, "fresh", payload, declared_size); try { - backend.publishBlob(request); + op.publish(request, Retry::once()); FAIL() << "expected a source-size mismatch for " << key; } catch (const DB::Exception & e) @@ -336,19 +284,21 @@ TEST(CASInMemory, PublishBlobRejectsShortAndLongStreamingSourcesWithoutVisibilit } } - EXPECT_EQ(backend.get("short")->bytes, "old-short"); - EXPECT_EQ(backend.get("long")->bytes, "old-long"); + EXPECT_EQ(op.read("short", Retry::once())->bytes, "old-short"); + EXPECT_EQ(op.read("long", Retry::once())->bytes, "old-long"); } TEST(CASInMemory, PublishBlobLongSourceReadsOnlyDeclaredPayloadAndOneProbeByte) { InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("long", "old-complete-body").outcome, PutOutcome::Done); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("long", "old-complete-body", Retry::once()))); auto state = std::make_shared(); try { - backend.publishBlob(countedLongPublication("long", "fresh", 3, 1024, state)); + op.publish(countedLongPublication("long", "fresh", 3, 1024, state), Retry::once()); FAIL() << "expected a long-source mismatch"; } catch (const DB::Exception & e) @@ -357,8 +307,9 @@ TEST(CASInMemory, PublishBlobLongSourceReadsOnlyDeclaredPayloadAndOneProbeByte) } EXPECT_EQ(state->bytes_exposed, 4u); - ASSERT_TRUE(backend.get("long").has_value()); - EXPECT_EQ(backend.get("long")->bytes, "old-complete-body"); + const auto still_present = op.read("long", Retry::once()); + ASSERT_TRUE(still_present.has_value()); + EXPECT_EQ(still_present->bytes, "old-complete-body"); } TEST(CASInMemory, PublishBlobKeepsThePreviousIncarnationVisibleUntilTheCompleteBodyIsReady) @@ -366,28 +317,75 @@ TEST(CASInMemory, PublishBlobKeepsThePreviousIncarnationVisibleUntilTheCompleteB using namespace std::chrono_literals; InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("blob", "old-complete-body").outcome, PutOutcome::Done); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("blob", "old-complete-body", Retry::once()))); std::promise source_opened; std::promise release_source; const std::shared_future release = release_source.get_future().share(); + /// Set when `open_payload`'s own internal wait below times out instead of observing the release. + /// The timeout alone silently lets the publisher proceed either way -- this flag is what lets the + /// test body downstream (after calling `releaseOnce`) assert that the release actually reached the + /// publisher, so a regression that leaves it unreleased for the full bound FAILS instead of quietly + /// passing because the timeout eventually let it through anyway. + std::atomic release_wait_expired{false}; const BlobPublishRequest request{ .destination_key = "blob", .publication = StreamingBlobPublication{ .payload_size = 7, .fresh_envelope = "fresh-envelope", - .open_payload = [&source_opened, release] + .open_payload = [&source_opened, release, &release_wait_expired] { source_opened.set_value(); - release.wait(); + /// Bounded, not `.wait()`: even if the release guards below somehow never fire, this + /// lambda -- and therefore `publish()`, and therefore the `std::async` task wrapping it + /// -- must still return within a bounded time, so `publication`'s blocking destructor (a + /// `std::async` future's destructor blocks until its task finishes) can never hang the + /// whole process. This is the ONLY wait `open_payload` performs, so it also bounds + /// `publish()`'s total time to this 20s plus whatever negligible in-memory work follows. + if (release.wait_for(20s) != std::future_status::ready) + release_wait_expired = true; return std::make_unique(String("payload")); }}}; - auto publication = std::async(std::launch::async, [&] { backend.publishBlob(request); }); - source_opened.get_future().wait(); - - auto observation = std::async(std::launch::async, [&] { return backend.get("blob"); }); - const auto observation_status = observation.wait_for(2s); + /// `CasOperation` carries mutable per-call state and is single-threaded by design: the publish and + /// the concurrent read below each admit their OWN operation from the shared `requests` rather than + /// racing on `op`. + CasOperation publish_op = requests.admit(); + auto publication = std::async(std::launch::async, [&] { publish_op.publish(request, Retry::once()); }); + + /// `publication` and (below) `observation` are both futures returned by `std::async`, so EACH one's + /// destructor blocks until its own task finishes. A failing ASSERT_*/EXPECT_* can unwind this + /// function while the publisher is still parked on `release`; this releases it promptly on that + /// exit rather than relying solely on the 20s bound above. Idempotent (the ordinary release near the + /// end sets `released` first) and declared right after `publication` so it protects every exit from + /// here on -- including the one right below, before `observation` exists. + bool released = false; + const auto releaseOnce = [&] + { + if (!released) + { + released = true; + release_source.set_value(); + } + }; + SCOPE_EXIT({ releaseOnce(); }); + + ASSERT_EQ(source_opened.get_future().wait_for(20s), std::future_status::ready) + << "publish() never reached open_payload"; + + CasOperation read_op = requests.admit(); + auto observation = std::async(std::launch::async, [&] { return read_op.read("blob", Retry::once()); }); + /// A SECOND copy of the same guard, declared AFTER `observation` so it tears down BEFORE + /// `observation`'s own blocking destructor on any unwind past this point. Without it, a genuine + /// visibility-lock regression (exactly what this test exists to catch) would leave `read_op.read` + /// blocked on the same lock `publish_op.publish` holds while parked on `release`, and the FIRST + /// guard above -- which, being declared earlier, tears down only AFTER `observation`'s destructor -- + /// would then release the publisher too late to ever unblock that read. + SCOPE_EXIT({ releaseOnce(); }); + + const auto observation_status = observation.wait_for(20s); EXPECT_EQ(observation_status, std::future_status::ready) << "publication must not hold the visibility lock while draining its source"; if (observation_status == std::future_status::ready) @@ -397,26 +395,38 @@ TEST(CASInMemory, PublishBlobKeepsThePreviousIncarnationVisibleUntilTheCompleteB EXPECT_EQ(visible->bytes, "old-complete-body"); } - release_source.set_value(); + releaseOnce(); + /// `open_payload` above is the ONLY wait reachable from `publish_op.publish`, and it is itself + /// bounded to 20s: a stuck publisher therefore costs at most that 20s bound (plus negligible + /// in-memory work) before `publish()` returns and `publication`'s `std::async` destructor can + /// complete -- never an unbounded hang. `EXPECT_FALSE` below turns a timeout that silently released + /// the publisher into a visible test failure instead of a pass for the wrong reason. + ASSERT_EQ(publication.wait_for(20s), std::future_status::ready) << "publish() never completed after release"; + EXPECT_FALSE(release_wait_expired.load()) + << "open_payload's internal wait timed out instead of observing the release"; EXPECT_NO_THROW(publication.get()); - ASSERT_TRUE(backend.get("blob").has_value()); - EXPECT_EQ(backend.get("blob")->bytes, "fresh-envelopepayload"); + const auto after = op.read("blob", Retry::once()); + ASSERT_TRUE(after.has_value()); + EXPECT_EQ(after->bytes, "fresh-envelopepayload"); } TEST(CASInMemory, PublishBlobCopiesStagedObjectBytesVerbatim) { InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("stage", "staged-envelopepayload").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent("blob", "old-body").outcome, PutOutcome::Done); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("stage", "staged-envelopepayload", Retry::once()))); + ASSERT_TRUE(std::holds_alternative(op.create("blob", "old-body", Retry::once()))); - backend.publishBlob(BlobPublishRequest{ + op.publish(BlobPublishRequest{ .destination_key = "blob", .publication = VerbatimStagedBlobPublication{ .object_key = "stage", - .object_size = 22}}); + .object_size = 22}}, Retry::once()); - ASSERT_TRUE(backend.get("blob").has_value()); - EXPECT_EQ(backend.get("blob")->bytes, "staged-envelopepayload"); + const auto after = op.read("blob", Retry::once()); + ASSERT_TRUE(after.has_value()); + EXPECT_EQ(after->bytes, "staged-envelopepayload"); } // ===================================================================== @@ -426,77 +436,78 @@ TEST(CASInMemory, PublishBlobCopiesStagedObjectBytesVerbatim) TEST(CASInMemoryFaults, HeldDeleteLandsLater) { InMemoryBackend b; - const Token t1 = b.putIfAbsent("k", "v1").token; + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + const Etag t1 = std::get(op.create("k", "v1", Retry::once())).etag; b.setHoldDeletes(true); - auto d = b.deleteExact("k", t1); // message "sent", not landed - EXPECT_EQ(d.kind, DeleteOutcome::Kind::Deleted); // caller sees the send accepted - EXPECT_TRUE(b.get("k").has_value()); // ... but nothing landed yet + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Removed); // message "sent", not landed + EXPECT_TRUE(op.read("k", Retry::once()).has_value()); // ... but nothing landed yet ASSERT_EQ(b.pendingDeletes(), 1u); // the object is recreated before the zombie lands: - b.putOverwrite("k", "v1'", t1); + op.replace("k", "v1'", t1, Retry::once()); auto landed = b.landPendingDelete(0); // the zombie lands NOW - EXPECT_EQ(landed.kind, DeleteOutcome::Kind::TokenMismatch); // 412 — INV-NO-RETURN in miniature - EXPECT_EQ(b.get("k")->bytes, "v1'"); + EXPECT_EQ(landed, DB::Cas::Backend::RawRemoval::Mismatch); // 412 — INV-NO-RETURN in miniature + expectBytes(b, "k", "v1'"); } TEST(CASInMemoryFaults, InjectedCasConflictFiresOnce) { InMemoryBackend b; - const Token t1 = b.casPut("m", "s1", std::nullopt).token; - b.failNextCasPut("m"); - EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Conflict); // injected - EXPECT_EQ(b.get("m")->bytes, "s1"); - EXPECT_EQ(b.casPut("m", "s2", t1).outcome, CasOutcome::Committed); // next attempt is real + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + const Etag t1 = std::get(op.create("m", "s1", Retry::once())).etag; + b.refuseNextWrite("m"); + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t1, Retry::once()))); // injected + EXPECT_EQ(op.read("m", Retry::once())->bytes, "s1"); + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t1, Retry::once()))); // next attempt is real } TEST(CASInMemoryFaults, NonEnforcingModeMimicsBadBackend) { InMemoryBackend b; + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); b.setEnforceTokens(false); // MinIO-OSS-shaped backend - b.putIfAbsent("k", "v1"); - auto d = b.deleteExact("k", Token{"totally-wrong", TokenType::Emulated}); - EXPECT_EQ(d.kind, DeleteOutcome::Kind::Deleted); // silently deletes anyway — the dangerous behavior - EXPECT_FALSE(b.get("k").has_value()); + const Etag t0 = std::get(op.create("k", "v1", Retry::once())).etag; + ASSERT_TRUE(std::holds_alternative(op.replace("k", "v2", t0, Retry::once()))); // mints a later incarnation + EXPECT_EQ(op.remove("k", t0, Retry::once()), Removal::Removed); // stale-but-same-key precondition silently deletes anyway — the dangerous behavior + EXPECT_FALSE(op.read("k", Retry::once()).has_value()); } TEST(CASInMemoryFaults, VersioningMarkerMode) { InMemoryBackend b; b.setSimulateDeleteMarkers(true); - const Token t1 = b.putIfAbsent("k", "v1").token; - EXPECT_TRUE(b.deleteExact("k", t1).created_delete_marker); // probe must reject this pool -} - -TEST(CASInMemoryBackend, RoundTripsUserMetadata) -{ - DB::Cas::InMemoryBackend backend; - const DB::Cas::ObjectMeta meta{{"cas_owner", "ab:7:42"}}; - ASSERT_EQ(backend.putIfAbsent("k/key", "body", meta).outcome, DB::Cas::PutOutcome::Done); - - const auto hr = backend.head("k/key"); - ASSERT_TRUE(hr.exists); - ASSERT_EQ(hr.attributes.at("cas_owner"), "ab:7:42"); - - const auto gr = backend.get("k/key"); - ASSERT_TRUE(gr.has_value()); - ASSERT_EQ(gr->attributes.at("cas_owner"), "ab:7:42"); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + const Etag t1 = std::get(op.create("k", "v1", Retry::once())).etag; + /// A removal that only archives (never reclaims) is not an ordinary Removed: the engine reports it + /// as CAS_DELETE_MARKER so the capability probe can reject a versioned pool. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CAS_DELETE_MARKER, [&] { op.remove("k", t1, Retry::once()); }); } // ===================================================================== -// getStream seam (forward-only reads of write-once objects) +// stream seam (forward-only reads of write-once objects) // ===================================================================== -TEST(CASBackendStream, StreamsBodyWindow) +/// The legacy getStream's byte-range window is retired along with it: the primitive `stream` takes no +/// Range, and every consumer (RunFileReader) already bounds its own consumption client-side rather than +/// relying on a server-side window. What survives here is presence: a present key opens a readable +/// stream, an absent one opens none. +TEST(CASBackendStream, StreamsWholeBodyOrNullWhenAbsent) { auto backend = std::make_shared(); - backend->putIfAbsent("k", "0123456789"); - auto got = backend->getStream("k", DB::Cas::Range{.offset = 2, .length = 5}); - ASSERT_TRUE(got.has_value()); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + op.create("k", "0123456789", Retry::once()); + + auto got = op.stream("k", Retry::once()); + ASSERT_TRUE(got != nullptr); String out; - DB::readStringUntilEOF(out, *got->stream); - EXPECT_EQ(out, "23456"); - EXPECT_FALSE(got->token.empty()); - EXPECT_FALSE(backend->getStream("absent").has_value()); + DB::readStringUntilEOF(out, *got); + EXPECT_EQ(out, "0123456789"); + + EXPECT_EQ(op.stream("absent", Retry::once()), nullptr); } // ===================================================================== @@ -509,7 +520,7 @@ extern const Event CASBlobPut; extern const Event CASBlobPutDeduplicated; extern const Event CASBlobHead; extern const Event CASBlobHeadMiss; -extern const Event CASGCCompareSwap; +extern const Event CASGCPut; } TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) @@ -534,27 +545,29 @@ TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) EXPECT_EQ(classifyCasNs("pool/cas/manifests/0/srv/store/d18/uuid@cas@/24/1/000001.proto"), CasNs::Manifest); auto inner = std::make_shared(); - InstrumentedBackend b(inner); + auto instrumented = std::make_shared(inner); + CasRequests requests = openRequestsForTest(instrumented); + CasOperation op = requests.admit(); using ProfileEvents::global_counters; const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); const auto blob_dedup_before = global_counters[ProfileEvents::CASBlobPutDeduplicated].load(); const auto blob_head_before = global_counters[ProfileEvents::CASBlobHead].load(); const auto blob_miss_before = global_counters[ProfileEvents::CASBlobHeadMiss].load(); - const auto gc_cas_before = global_counters[ProfileEvents::CASGCCompareSwap].load(); + const auto gc_put_before = global_counters[ProfileEvents::CASGCPut].load(); const String blob_key = "pool/blobs/ab/abcdef0123456789"; - /// First put of a blob ⇒ Put. - EXPECT_EQ(b.putIfAbsent(blob_key, "payload").outcome, PutOutcome::Done); - /// Second put of the same key ⇒ PutDeduplicated (content already exists). - EXPECT_EQ(b.putIfAbsent(blob_key, "payload").outcome, PutOutcome::PreconditionFailed); + /// First create of a blob ⇒ Put. + EXPECT_TRUE(std::holds_alternative(op.create(blob_key, "payload", Retry::once()))); + /// Second create of the same key ⇒ PutDeduplicated (content already exists). + EXPECT_TRUE(std::holds_alternative(op.create(blob_key, "payload", Retry::once()))); /// head of an absent blob key ⇒ HeadMiss (the 404 signal). - EXPECT_FALSE(b.head("pool/blobs/zz/absent").exists); + EXPECT_FALSE(op.head("pool/blobs/zz/absent", Retry::once()).has_value()); /// head of the present blob key ⇒ Head. - EXPECT_TRUE(b.head(blob_key).exists); - /// casPut create on a gc key ⇒ Gc Cas. - EXPECT_EQ(b.casPut("pool/gc/state", "g1", std::nullopt).outcome, CasOutcome::Committed); + EXPECT_TRUE(op.head(blob_key, Retry::once()).has_value()); + /// create on a gc key ⇒ Gc Put. + EXPECT_TRUE(std::holds_alternative(op.create("pool/gc/state", "g1", Retry::once()))); /// Under coverage builds ProfileEvents propagate into a thread-local subtree that does not reach /// `global_counters`; deltas read 0 there only (see gtest_unique_key_index_cache). #if !WITH_COVERAGE @@ -562,27 +575,32 @@ TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) EXPECT_EQ(global_counters[ProfileEvents::CASBlobPutDeduplicated].load() - blob_dedup_before, 1u); EXPECT_EQ(global_counters[ProfileEvents::CASBlobHead].load() - blob_head_before, 1u); EXPECT_EQ(global_counters[ProfileEvents::CASBlobHeadMiss].load() - blob_miss_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASGCCompareSwap].load() - gc_cas_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASGCPut].load() - gc_put_before, 1u); #else (void)blob_put_before; (void)blob_dedup_before; (void)blob_head_before; - (void)blob_miss_before; (void)gc_cas_before; + (void)blob_miss_before; (void)gc_put_before; #endif } TEST(CASInstrumentedBackend, PublishBlobDelegatesOnceAndRecordsOnePhysicalBlobWrite) { auto inner = std::make_shared(); - InstrumentedBackend backend(inner); + auto backend = std::make_shared(inner); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); using ProfileEvents::global_counters; const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); const auto request = streamingPublication("pool/blobs/ab/published", "fresh", "payload", 7); - backend.publishBlob(request); + op.publish(request, Retry::once()); EXPECT_EQ(inner->publish_calls, 1u); - ASSERT_TRUE(inner->get("pool/blobs/ab/published").has_value()); - EXPECT_EQ(inner->get("pool/blobs/ab/published")->bytes, "freshpayload"); + CasRequests inner_requests = openRequestsForTest(inner); + CasOperation inner_op = inner_requests.admit(); + const auto published = inner_op.read("pool/blobs/ab/published", Retry::once()); + ASSERT_TRUE(published.has_value()); + EXPECT_EQ(published->bytes, "freshpayload"); #if !WITH_COVERAGE EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut].load() - blob_put_before, 1u); #else @@ -594,6 +612,155 @@ TEST(CASInstrumentedBackend, PublishBlobDelegatesOnceAndRecordsOnePhysicalBlobWr // M-C2 Task 2: typed S3 precondition signal // ===================================================================== +/// The per-dialect grammar in isolation, independent of any backend fixture. +TEST(CASBackendGrammar, GenerationDialectAcceptsOnlyCanonicalPositiveDecimal) +{ + using DB::Cas::ObjectStorageBackend; + using DB::Cas::Dialect; + EXPECT_TRUE(ObjectStorageBackend::isValidTokenValue(Dialect::Generation, "123")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::Generation, "0")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::Generation, "00123")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::Generation, "\"123\"")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::Generation, "12a")); + EXPECT_TRUE(ObjectStorageBackend::isValidTokenValue(Dialect::ETag, "\"abc\"")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::ETag, " * ")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::ETag, "a,b")); + EXPECT_TRUE(ObjectStorageBackend::isValidTokenValue(Dialect::Emulated, "7")); + EXPECT_FALSE(ObjectStorageBackend::isValidTokenValue(Dialect::Emulated, "")); +} + +/// §1 (opt round-B): the fold/point GETs read tiny bodies but a default `ReadBufferFromS3` preallocates +/// ~1 MiB. `casSizedReadSettings` shrinks the buffer to the known body size + slack, capped at the +/// caller's default — never larger than before, regardless of the reported size. +TEST(CASSizedReadSettings, CapsToKnownSizePlusSlackButNeverAboveBase) +{ + DB::ReadSettings base; + base.remote_fs_settings.buffer_size = 1ULL << 20; /// 1 MiB default + base.local_fs_settings.buffer_size = 1ULL << 20; + + /// A ~3.7 KB fold body: buffer shrinks to size + slack, far below the 1 MiB default. + const auto small = DB::Cas::casSizedReadSettings(base, 3700); + EXPECT_EQ(small.remote_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); + EXPECT_EQ(small.local_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); + + /// A body larger than the default is capped AT the default (never grown). + const auto big = DB::Cas::casSizedReadSettings(base, 8ULL << 20); + EXPECT_EQ(big.remote_fs_settings.buffer_size, 1ULL << 20); + + /// Unknown size (0) = leave the base untouched (the metadata-fetch fallback path). + const auto unknown = DB::Cas::casSizedReadSettings(base, 0); + EXPECT_EQ(unknown.remote_fs_settings.buffer_size, 1ULL << 20); +} + +/// The CountingBackend recorders the streaming-memory gates consume: per-key and total stream counts. +/// A window is no longer part of the shape -- a materialized read is always whole, so `stream` no +/// longer carries one either, and it is not what the gates measure. +TEST(CASCountingBackendShape, RecordsStreamOpensPerKeyAndInTotal) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + op.create("k", String(1000, 'x'), Retry::once()); + + op.stream("k", Retry::once()); + op.stream("k", Retry::once()); + op.stream("absent", Retry::once()); + EXPECT_EQ(backend->getStreamCount("k"), 2u); + EXPECT_EQ(backend->getStreamTotal(), 3u); + + backend->resetCounts(); + EXPECT_EQ(backend->getStreamCount("k"), 0u); + EXPECT_EQ(backend->getStreamTotal(), 0u); +} + +/// Armed chunking makes this backend serve a stream the way a network-backed store does, in bounded +/// windows, instead of handing over the materialized object in one piece. The bytes a consumer reads +/// are the same either way; what changes is that a consumer which assumed one contiguous window can no +/// longer get one. +TEST(CASCountingBackendShape, AnArmedChunkBoundsTheWindowAStreamHandsOut) +{ + const String body(10'000, 'x'); + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("run", body, Retry::once()))); + + /// Unarmed: the whole object arrives as one window, which is what this backend's materialization + /// makes of any stream and exactly what the bound exists to remove. + { + auto opened = op.stream("run", Retry::once()); + ASSERT_TRUE(opened != nullptr); + String drained; + readStringUntilEOF(drained, *opened); + EXPECT_EQ(drained, body); + EXPECT_EQ(backend->largestStreamChunk("run"), 0u) << "nothing records a window while chunking is off"; + } + + backend->setStreamChunkForTest(4096); + { + auto opened = op.stream("run", Retry::once()); + ASSERT_TRUE(opened != nullptr); + String drained; + readStringUntilEOF(drained, *opened); + EXPECT_EQ(drained, body) << "chunking changes the window, never the bytes"; + EXPECT_EQ(backend->largestStreamChunk("run"), 4096u); + EXPECT_LT(backend->largestStreamChunk("run"), body.size()) + << "the consumer never held the object entire"; + } + + /// The mode outlives a counter reset, and the recorded window does not. + backend->resetCounts(); + EXPECT_EQ(backend->largestStreamChunk("run"), 0u); + auto reopened = op.stream("run", Retry::once()); + ASSERT_TRUE(reopened != nullptr); + String again; + readStringUntilEOF(again, *reopened); + EXPECT_EQ(backend->largestStreamChunk("run"), 4096u); +} + +/// What makes every request-profile gate in this tree trustworthy: a counter names a PHYSICAL request. +/// The transport primitives are the only surface left that can issue one, so this now pins that a +/// create/head/read/replace/remove issued through `CasOperation` counts exactly once each. +TEST(CASCountingBackendShape, OneRequestIsCountedOnceWhicheverSurfaceIssuedIt) +{ + auto backend = std::make_shared(); + DB::Cas::CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + DB::Cas::CasOperation op = requests.admit(); + + /// `Retry::once()` on every verb below, not `standard()`: this test pins ONE physical request per + /// call, and a healthy backend never distinguishes the two policies by outcome -- only `once()` + /// forbids a reissue by construction, so a regression that made the engine reissue speculatively + /// would still fail here instead of passing on a backend too healthy to ever need the second attempt. + EXPECT_TRUE(std::holds_alternative(op.create("k", "v", Retry::once()))); + EXPECT_TRUE(std::holds_alternative(op.create("k2", "v", Retry::once()))); + EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->putCount("k2"), 1u); + EXPECT_EQ(backend->writeTotal(), 2u); + EXPECT_EQ(backend->putOverwriteTotal(), 0u) << "neither write carried a precondition"; + + const std::optional k_meta = op.head("k", Retry::once()); + ASSERT_TRUE(k_meta); + EXPECT_EQ(backend->headCount("k"), 1u); + + expectBytes(*backend, "k", "v"); + EXPECT_TRUE(op.read("k", Retry::once())); + EXPECT_EQ(backend->getCount("k"), 2u); + + EXPECT_TRUE(std::holds_alternative(op.replace("k", "w", k_meta->etag, Retry::once()))); + EXPECT_EQ(backend->putOverwriteCount("k"), 1u) << "a write with a precondition is the replace shape"; + EXPECT_EQ(backend->writeCount("k"), 2u); + + const std::optional k2_meta = op.head("k2", Retry::once()); + ASSERT_TRUE(k2_meta); + EXPECT_EQ(op.remove("k2", k2_meta->etag, Retry::once()), Removal::Removed); + const std::optional k_meta_after = op.head("k", Retry::once()); + ASSERT_TRUE(k_meta_after); + EXPECT_EQ(op.remove("k", k_meta_after->etag, Retry::once()), Removal::Removed); + EXPECT_EQ(backend->deleteCount("k"), 1u); + EXPECT_EQ(backend->deleteCount("k2"), 1u); + EXPECT_EQ(backend->deleteTotal(), 2u); +} + #if USE_AWS_S3 namespace @@ -647,6 +814,12 @@ struct PublicationWriteBarrier std::promise opened; std::promise release; std::shared_future release_future = release.get_future().share(); + /// Set when the wait on `release_future` below times out instead of observing the release. The + /// timeout alone silently lets the write proceed either way -- this flag is what lets the test body + /// assert (after calling the release itself) that it actually reached the write, so a regression + /// that leaves it unreleased for the full bound FAILS instead of quietly passing because the + /// timeout eventually let it through anyway. + std::atomic release_wait_expired{false}; }; class PublicationRecordingLocalObjectStorage final : public DB::LocalObjectStorage @@ -674,7 +847,14 @@ class PublicationRecordingLocalObjectStorage final : public DB::LocalObjectStora if (write_barrier) { write_barrier->opened.set_value(); - write_barrier->release_future.wait(); + /// Bounded, not `.wait()`: even if a caller's release guard somehow never fires, this call + /// -- and therefore the `std::async` task wrapping the publish that reaches it -- must still + /// return within a bounded time, so that future's blocking destructor (a `std::async` + /// future's destructor blocks until its task finishes) can never hang the whole process. + /// This is the ONLY wait this write performs, so it also bounds the whole call's total time + /// to this 20s plus whatever negligible local-filesystem work follows. + if (write_barrier->release_future.wait_for(std::chrono::seconds(20)) != std::future_status::ready) + write_barrier->release_wait_expired = true; } return out; } @@ -769,12 +949,14 @@ String readStorageObject(const DB::ObjectStoragePtr & storage, const String & ke TEST(CASObjectStorageBackend, PublishBlobStreamingUsesOrdinaryDefaultWriteTransport) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - backend.setNativeTokenTypeForTest(TokenType::Generation); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + backend->setNativeTokenTypeForTest(Dialect::Generation); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/streaming"); const auto request = streamingPublication(destination, "fresh-envelope", "payload", 7); - backend.publishBlob(request); + op.publish(request, Retry::once()); ASSERT_EQ(storage->write_calls, 1u); ASSERT_TRUE(storage->last_write_mode.has_value()); @@ -796,7 +978,9 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedKeepsDestinationCompleteUntilAt using namespace std::chrono_literals; auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String key = "publish/emulated-atomic"; const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); @@ -813,15 +997,38 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedKeepsDestinationCompleteUntilAt auto opened = barrier->opened.get_future(); auto publication = std::async(std::launch::async, [&] { - backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)); + op.publish(streamingPublication(key, "fresh-envelope", "payload", 7), Retry::once()); }); - const auto opened_status = opened.wait_for(2s); + /// Same hazard as `PublishBlobKeepsThePreviousIncarnationVisibleUntilTheCompleteBodyIsReady`: an + /// exception unwinding out of this function (a failing ASSERT_*, or `readStorageObject` throwing) + /// while the write is still parked on `barrier->release_future.wait()` would deadlock `publication`'s + /// blocking `std::async` destructor against `barrier`'s own (later) teardown. Declared after + /// `publication` so it tears down FIRST, this guard releases the write unconditionally. + bool released = false; + SCOPE_EXIT({ + if (!released) + { + released = true; + barrier->release.set_value(); + } + }); + + const auto opened_status = opened.wait_for(20s); EXPECT_EQ(opened_status, std::future_status::ready); if (opened_status == std::future_status::ready) EXPECT_EQ(readStorageObject(storage, physical_key), "old-complete-body"); + released = true; barrier->release.set_value(); + /// The write inside `writeObject` is the ONLY wait reachable from `op.publish`, and it is itself + /// bounded to 20s: a stuck write therefore costs at most that 20s bound (plus negligible + /// local-filesystem work) before `publish()` returns and `publication`'s `std::async` destructor + /// can complete -- never an unbounded hang. `EXPECT_FALSE` below turns a timeout that silently + /// released the write into a visible test failure instead of a pass for the wrong reason. + ASSERT_EQ(publication.wait_for(20s), std::future_status::ready) << "publish() never completed after release"; + EXPECT_FALSE(barrier->release_wait_expired.load()) + << "the write's internal wait timed out instead of observing the release"; EXPECT_NO_THROW(publication.get()); EXPECT_EQ(storage->metadata_calls, 0u); EXPECT_EQ(readStorageObject(storage, physical_key), "fresh-envelopepayload"); @@ -830,7 +1037,9 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedKeepsDestinationCompleteUntilAt TEST(CASObjectStorageBackend, PublishBlobEmulatedWriteFailurePreservesDestinationAndCleansTemporary) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String key = "publish/emulated-failure"; const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); @@ -840,24 +1049,26 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedWriteFailurePreservesDestinatio DB::writeString(String("old-complete-body"), *out); out->finalize(); } - const Token old_token = backend.head(key).token; + const Etag old_token = op.head(key, Retry::once())->etag; storage->throw_after_open = true; EXPECT_THROW( - backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)), + op.publish(streamingPublication(key, "fresh-envelope", "payload", 7), Retry::once()), std::runtime_error); storage->throw_after_open = false; EXPECT_NE(storage->last_opened_key, physical_key); EXPECT_FALSE(storage->exists(DB::StoredObject(storage->last_opened_key))); EXPECT_EQ(readStorageObject(storage, physical_key), "old-complete-body"); - EXPECT_EQ(backend.head(key).token, old_token); + EXPECT_EQ(op.head(key, Retry::once())->etag, old_token); } TEST(CASObjectStorageBackend, PublishBlobCancelsShortAndLongStreamingSourcesBeforeVisibility) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/mismatch"); { @@ -871,7 +1082,7 @@ TEST(CASObjectStorageBackend, PublishBlobCancelsShortAndLongStreamingSourcesBefo DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - backend.publishBlob(streamingPublication(destination, "fresh", "abc", 4)); + op.publish(streamingPublication(destination, "fresh", "abc", 4), Retry::once()); }); EXPECT_EQ(storage->cancel_calls, 1u); EXPECT_EQ(storage->finalize_calls, 0u); @@ -881,7 +1092,7 @@ TEST(CASObjectStorageBackend, PublishBlobCancelsShortAndLongStreamingSourcesBefo auto state = std::make_shared(); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - backend.publishBlob(countedLongPublication(destination, "fresh", 3, 1024, state)); + op.publish(countedLongPublication(destination, "fresh", 3, 1024, state), Retry::once()); }); EXPECT_EQ(state->bytes_exposed, 4u); EXPECT_EQ(storage->cancel_calls, 2u); @@ -893,7 +1104,9 @@ TEST(CASObjectStorageBackend, PublishBlobCancelsShortAndLongStreamingSourcesBefo TEST(CASObjectStorageBackend, PublishBlobEmulatedLongSourceReadsOnlyDeclaredPayloadAndOneProbeByte) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String key = "publish/emulated-long"; const String physical_key = DB::Cas::tests::nativeKeyUnder(storage, key); @@ -907,7 +1120,7 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedLongSourceReadsOnlyDeclaredPayl auto state = std::make_shared(); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - backend.publishBlob(countedLongPublication(key, "fresh", 3, 1024, state)); + op.publish(countedLongPublication(key, "fresh", 3, 1024, state), Retry::once()); }); EXPECT_EQ(state->bytes_exposed, 4u); @@ -917,7 +1130,9 @@ TEST(CASObjectStorageBackend, PublishBlobEmulatedLongSourceReadsOnlyDeclaredPayl TEST(CASObjectStorageBackend, PublishBlobCopiesStagedBytesWithNativeOnlyDefaultRequestMode) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String staging = DB::Cas::tests::nativeKeyUnder(storage, "publish/staging"); const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/copied"); @@ -929,11 +1144,11 @@ TEST(CASObjectStorageBackend, PublishBlobCopiesStagedBytesWithNativeOnlyDefaultR } storage->resetRecording(); - backend.publishBlob(BlobPublishRequest{ + op.publish(BlobPublishRequest{ .destination_key = destination, .publication = VerbatimStagedBlobPublication{ .object_key = staging, - .object_size = 22}}); + .object_size = 22}}, Retry::once()); ASSERT_EQ(storage->copy_calls, 1u); ASSERT_TRUE(storage->last_copy_settings.has_value()); @@ -948,7 +1163,9 @@ TEST(CASObjectStorageBackend, PublishBlobCopiesStagedBytesWithNativeOnlyDefaultR TEST(CASObjectStorageBackend, PublishBlobRefusesVerbatimCopyWithoutNativeTransport) { auto storage = makePublicationRecordingStorage(); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String staging = DB::Cas::tests::nativeKeyUnder(storage, "publish/unsupported-staging"); const String destination = DB::Cas::tests::nativeKeyUnder(storage, "publish/unsupported-copy"); @@ -963,17 +1180,37 @@ TEST(CASObjectStorageBackend, PublishBlobRefusesVerbatimCopyWithoutNativeTranspo DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, [&] { - backend.publishBlob(BlobPublishRequest{ + op.publish(BlobPublishRequest{ .destination_key = destination, .publication = VerbatimStagedBlobPublication{ .object_key = staging, - .object_size = 22}}); + .object_size = 22}}, Retry::once()); }); EXPECT_EQ(storage->copy_calls, 0u); EXPECT_FALSE(storage->exists(DB::StoredObject(destination))); } +/// Every Native conditional write selects the SingleAttempt object-storage retry profile (RFC +/// cas-s3-timeout-retry-control §disable-transparent-conditional-write-retries): the CAS request +/// engine, not the object-storage client, owns retry/backoff for a conditional PUT, so a client-level +/// retry loop reissuing the identical request underneath it would double the reissue and could land a +/// write the engine itself had already given up on. Two seams prove the property without a live/fake +/// S3 endpoint: `IObjectStorage::supportsRetryProfile` is the fail-closed capability check +/// `checkConditionalWriteSingleAttemptSupport` relies on at mount time, and +/// `s3_max_unexpected_write_error_retries_override` is the SECOND retry-affecting layer above the S3 +/// client -- `WriteBufferFromS3`'s own makeSinglepartUpload/completeMultipartUpload loop, bounded +/// independently of the client-level override. +TEST(CASObjectStorageBackend, ConditionalWriteSelectsSingleAttemptAndLocalStorageDoesNotSupportIt) +{ + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); + const auto ws = backend->conditionalWriteSettingsForTest(); + EXPECT_EQ(ws.object_storage_retry_profile, DB::ObjectStorageRetryProfile::SingleAttempt); + EXPECT_EQ(ws.s3_max_unexpected_write_error_retries_override, 1u); + EXPECT_FALSE(tests::makeLocalObjectStorageForTest()->supportsRetryProfile(DB::ObjectStorageRetryProfile::SingleAttempt)); +} + /// The Native conditional-PUT path discriminates a lost precondition by the canonical S3 error code /// string ("PreconditionFailed", "NoSuchKey", ...) that `S3Exception` carries from the response XML /// `` — a 412 is UNMODELED for the AWS SDK (the enum value is UNKNOWN), so the name is the only @@ -1025,108 +1262,17 @@ TEST(CASS3Signal, FinalizeClassifierMapsPreconditionLossExactly) }; EXPECT_EQ(classify(DB::S3Exception("412", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed")), - PutOutcome::PreconditionFailed); + DB::Cas::detail::ConditionalWriteOutcome::PreconditionLost); EXPECT_EQ(classify(DB::S3Exception("404 gone under If-Match", Aws::S3::S3Errors::UNKNOWN, "NoSuchKey")), - PutOutcome::PreconditionFailed); + DB::Cas::detail::ConditionalWriteOutcome::PreconditionLost); EXPECT_EQ(classify(DB::S3Exception("retries exhausted, no name attached", Aws::S3::S3Errors::NO_SUCH_KEY)), - PutOutcome::PreconditionFailed); + DB::Cas::detail::ConditionalWriteOutcome::PreconditionLost); ThrowOnFinalizeBuffer unrelated(DB::S3Exception("503", Aws::S3::S3Errors::UNKNOWN, "SlowDown")); EXPECT_THROW(finalizeConditionalWrite(unrelated), DB::S3Exception); ThrowOnFinalizeBuffer clean; - EXPECT_EQ(finalizeConditionalWrite(clean), PutOutcome::Done); -} - -namespace -{ - -/// A `LocalObjectStorage` that round-trips user metadata in-process. The production -/// `LocalObjectStorage` deliberately drops the `attributes` argument of `writeObject` and never -/// populates `ObjectMetadata::attributes` (local files carry no `x-amz-meta-*`), so it cannot stand -/// in for S3/RustFS when verifying the metadata threading. This test-only subclass records the -/// attributes passed on write, keyed by physical path, and injects them back on metadata reads — -/// exactly what a real object store does for `x-amz-meta-*`. It exercises the `EmulatedSingleProcess` -/// `ObjectStorageBackend` threading (`putIfAbsent` → `writeObject` attributes → `head` attributes) -/// without a live S3 backend; the real S3/RustFS round trip is verified empirically out-of-band. -class AttributePreservingLocalObjectStorage final : public DB::LocalObjectStorage -{ -public: - using DB::LocalObjectStorage::LocalObjectStorage; - - std::unique_ptr writeObject( - const DB::StoredObject & object, - DB::WriteMode mode, - std::optional attributes, - size_t buf_size, - const DB::WriteSettings & write_settings) override - { - if (attributes.has_value()) - { - std::lock_guard lock(mutex); - saved_attributes[object.remote_path] = *attributes; - } - return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); - } - - std::optional tryGetObjectMetadata(const std::string & path, bool with_tags) const override - { - auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); - if (metadata) - inject(path, *metadata); - return metadata; - } - - DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override - { - auto metadata = DB::LocalObjectStorage::getObjectMetadata(path, with_tags); - inject(path, metadata); - return metadata; - } - -private: - void inject(const std::string & path, DB::ObjectMetadata & metadata) const - { - std::lock_guard lock(mutex); - if (auto it = saved_attributes.find(path); it != saved_attributes.end()) - metadata.attributes = it->second; - } - - mutable std::mutex mutex; - mutable std::map saved_attributes; -}; - -DB::ObjectStoragePtr makeAttributePreservingStorageForTest() -{ - static std::atomic counter{0}; - const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); - const auto root = (std::filesystem::temp_directory_path() / ("cas_meta_unit_" + unique)).string(); - - std::error_code ec; - std::filesystem::remove_all(root, ec); - std::filesystem::create_directories(root, ec); - - DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); - return std::make_shared(std::move(settings)); -} - -} - -/// The `EmulatedSingleProcess` `ObjectStorageBackend` must thread user metadata through to the -/// underlying object storage's `writeObject` attributes on `putIfAbsent` and read it back into -/// `HeadResult::attributes` on `head`. Verified here over an attribute-preserving object storage -/// (the production `LocalObjectStorage` drops attributes); the live S3/RustFS round trip is verified -/// empirically out-of-band. -TEST(CASObjectStorageBackend, EmulatedRoundTripsUserMetadata) -{ - ObjectStorageBackend backend(makeAttributePreservingStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); - - const DB::Cas::ObjectMeta meta{{"cas_owner", "ab:7:42"}}; - ASSERT_EQ(backend.putIfAbsent("k/key", "body", meta).outcome, DB::Cas::PutOutcome::Done); - - const auto hr = backend.head("k/key"); - ASSERT_TRUE(hr.exists); - ASSERT_EQ(hr.attributes.at("cas_owner"), "ab:7:42"); + EXPECT_EQ(finalizeConditionalWrite(clean), DB::Cas::detail::ConditionalWriteOutcome::Applied); } namespace @@ -1210,46 +1356,20 @@ TEST(CASObjectStorageBackend, NativeModeGetReturnsNulloptOnMidGetNoSuchKey) /// logical key IS the physical one the fixture wrote and armed. const auto fixture = makeThrowOnReadStorageForTest("pool/blobs/ab/abcdef0123456789abcdef0123456789"); - ObjectStorageBackend backend(fixture.storage, ObjectStorageBackend::Mode::Native); + auto backend = std::make_shared(fixture.storage, ObjectStorageBackend::Mode::Native); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); - /// `get` HEADs before it reads and answers nullopt for an absent key, so without this the nullopt - /// below would be satisfied by an object the fixture failed to place — the mid-GET race would go - /// untested and the case would still pass. - Backend & iface = backend; - ASSERT_TRUE(iface.head(fixture.key).exists); + /// `head` answers present for a key the fixture failed to place, so without this the nullopt below + /// would be satisfied vacuously — the mid-read race would go untested and the case would still pass. + ASSERT_TRUE(op.head(fixture.key, Retry::once()).has_value()); /// HEAD reports the key present; readObject then throws NO_SUCH_KEY. - /// Contract: get must return std::nullopt, not propagate the S3Exception. - /// Call through the base-class interface so the default `Range{}` arg is available. - const auto result = iface.get(fixture.key); + /// Contract: read must return std::nullopt, not propagate the S3Exception. + const auto result = op.read(fixture.key, Retry::once()); EXPECT_FALSE(result.has_value()); } -/// A ranged `get` over a real `LocalObjectStorage` returns exactly the requested window, with the -/// same clamping the old read-whole-then-substr path had: a window whose offset is at or past EOF -/// yields an empty result. The only-the-window I/O property (no whole-object read) is enforced by -/// the `readObjectRanged` rewrite and cross-checked by the request-size gate in a later task. -TEST(CASObjectStorageBackend, RangedGetReadsOnlyTheWindow) -{ - auto backend = std::make_shared( - tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); - - const String payload = String(300000, 'a') + String(300000, 'b') + String(300000, 'c'); - backend->putIfAbsent("p/obj", payload); - - const auto mid = backend->get("p/obj", DB::Cas::Range{.offset = 300000, .length = 300000}); - ASSERT_TRUE(mid.has_value()); - EXPECT_EQ(mid->bytes, String(300000, 'b')); - - const auto tail = backend->get("p/obj", DB::Cas::Range{.offset = 600000, .length = std::nullopt}); - ASSERT_TRUE(tail.has_value()); - EXPECT_EQ(tail->bytes, String(300000, 'c')); - - const auto past = backend->get("p/obj", DB::Cas::Range{.offset = 1000000, .length = 10}); - ASSERT_TRUE(past.has_value()); - EXPECT_TRUE(past->bytes.empty()); -} - /// codex-review-triage §3.18, finding 19c: the `EmulatedSingleProcess` adapter used to mint tokens /// from a plain in-process counter (`emu_seq`), NOT actually seeded from the underlying object's etag /// despite the class comment's claim. After a process restart (modeled here as a fresh @@ -1257,59 +1377,109 @@ TEST(CASObjectStorageBackend, RangedGetReadsOnlyTheWindow) /// value that TEXTUALLY collides with a token persisted before the restart (e.g. a GC condemned-delete /// token queued for replay), even though the two values name completely different incarnations of the /// key. `deleteExact` must never let a stale, pre-restart token match a freshly recreated object. +#ifndef DEBUG_OR_SANITIZER_BUILD TEST(CASObjectStorageBackend, EmuTokenSurvivesProcessRestartAcrossRecreate) { auto storage = tests::makeLocalObjectStorageForTest(); auto backend1 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests1 = openRequestsForTest(backend1); + CasOperation op1 = requests1.admit(); /// A throwaway prior mutation on a DIFFERENT key: with the old counter this advances backend1's /// process-wide op counter to 1, so "k/restart"'s own mint below lands on 2 — chosen so it collides /// with backend2's post-restart recreate mint further down (also its SECOND op; see there). - ASSERT_EQ(backend1->putIfAbsent("k/other", "junk").outcome, PutOutcome::Done); - ASSERT_EQ(backend1->putIfAbsent("k/restart", "v1").outcome, PutOutcome::Done); - const Token stale_token = backend1->head("k/restart").token; + ASSERT_TRUE(std::holds_alternative(op1.create("k/other", "junk", Retry::once()))); + const WriteResult restart_create = op1.create("k/restart", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(restart_create)); + const Etag stale_token = std::get(restart_create).etag; /// Simulate a process restart: a brand-new `ObjectStorageBackend` instance (fresh emu state) over /// the SAME underlying storage — exactly what happens when the CAS process restarts. auto backend2 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests2 = openRequestsForTest(backend2); + CasOperation op2 = requests2.admit(); /// Delete and recreate the key through the NEW instance — a fresh incarnation with a fresh mtime. /// This is backend2's first-ever op (op 1) then a delete (no mint) then the recreate (op 2) — the /// same op-index as `stale_token` above under the old counter, so the two textually collide there. - const Token current = backend2->head("k/restart").token; - ASSERT_EQ(backend2->deleteExact("k/restart", current).kind, DeleteOutcome::Kind::Deleted); - ASSERT_EQ(backend2->putIfAbsent("k/restart", "v2-after-restart").outcome, PutOutcome::Done); - - /// The pre-restart token must NEVER match the post-restart incarnation, however coincidentally a - /// process-local counter would have re-minted the identical textual value. - const auto stale_delete = backend2->deleteExact("k/restart", stale_token); - EXPECT_EQ(stale_delete.kind, DeleteOutcome::Kind::TokenMismatch); + const auto current = op2.head("k/restart", Retry::once()); + ASSERT_TRUE(current.has_value()); + ASSERT_EQ(op2.remove("k/restart", current->etag, Retry::once()), Removal::Removed); + ASSERT_TRUE(std::holds_alternative(op2.create("k/restart", "v2-after-restart", Retry::once()))); + + /// The pre-restart incarnation must NEVER be usable as a precondition against the post-restart + /// backend instance, however coincidentally a process-local counter would have re-minted the + /// identical textual value: an `Etag` carries the identity of the backend that observed it, and the + /// engine refuses one minted elsewhere before it ever reaches the store (LOGICAL_ERROR), which is a + /// STRONGER guarantee than the old bare-value comparison this test used to pin. The underlying + /// same-instance mtime-quantum disambiguation this fixture was ALSO probing is covered directly by + /// `EmuTokenDisambiguatesSameEtagRewrite`, within one backend instance where the engine's own + /// cross-backend check cannot pre-empt it. A LOGICAL_ERROR aborts under + /// DEBUG_OR_SANITIZER_BUILD before it can ever be thrown and caught here; the debug/sanitizer arm + /// of this split (below) pins the same refusal via EXPECT_DEATH instead. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] + { + op2.remove("k/restart", stale_token, Retry::once()); + }); /// The live (post-restart) incarnation must be untouched by the rejected stale delete. - EXPECT_TRUE(backend2->head("k/restart").exists); + EXPECT_TRUE(op2.head("k/restart", Retry::once()).has_value()); } +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASObjectStorageBackendDeathTest, EmuTokenSurvivesProcessRestartAcrossRecreateAborts) +{ + auto storage = tests::makeLocalObjectStorageForTest(); + + auto backend1 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests1 = openRequestsForTest(backend1); + CasOperation op1 = requests1.admit(); + ASSERT_TRUE(std::holds_alternative(op1.create("k/other", "junk", Retry::once()))); + const WriteResult restart_create = op1.create("k/restart", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(restart_create)); + const Etag stale_token = std::get(restart_create).etag; + + auto backend2 = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests2 = openRequestsForTest(backend2); + CasOperation op2 = requests2.admit(); + + const auto current = op2.head("k/restart", Retry::once()); + ASSERT_TRUE(current.has_value()); + ASSERT_EQ(op2.remove("k/restart", current->etag, Retry::once()), Removal::Removed); + ASSERT_TRUE(std::holds_alternative(op2.create("k/restart", "v2-after-restart", Retry::once()))); + + /// See EmuTokenSurvivesProcessRestartAcrossRecreate above for the property under test; a + /// LOGICAL_ERROR aborts the process under DEBUG_OR_SANITIZER_BUILD, so this arm pins the refusal + /// via EXPECT_DEATH instead of an exception. + EXPECT_DEATH( + { op2.remove("k/restart", stale_token, Retry::once()); }, + "cannot be the precondition for"); +} +#endif -/// codex-review-triage §3.18, finding №18: `list`'s `EmulatedSingleProcess` branch minted its per-key -/// token via `tokenForList`, which always stamps `native_token_type` (ETag) REGARDLESS of `mode` -- -/// while `head`/`get` mint `TokenType::Emulated`. `Token::operator==` compares type AND value, so a -/// list-derived token could never satisfy an emulated `deleteExact`/`putOverwrite` expectation: a -/// fail-safe leak (never a wrong delete), but every consumer of listed tokens (GC namespace cleanup, -/// `deletePrefixWholesale`, orphan sweep, decommission drain) always saw `TokenMismatch` against a -/// LOCAL pool. `list` must surface the SAME (type, value) as `head` for the same key. +/// `list`'s `EmulatedSingleProcess` branch must surface the SAME incarnation value `head` would for the +/// same key. An earlier defect minted the listed value under the wrong dialect regardless of `mode`, +/// so a list-derived value could never satisfy an emulated `remove`/`replace` precondition: a +/// fail-safe leak (never a wrong delete), but every consumer of listed values (GC namespace cleanup, +/// `deletePrefixWholesale`, orphan sweep, decommission drain) always saw a mismatch against a LOCAL pool. TEST(CASObjectStorageBackend, EmulatedListTokenMatchesHeadToken) { auto backend = std::make_shared( tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); - ASSERT_EQ(backend->putIfAbsent("k/listed", "body").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create("k/listed", "body", Retry::once()))); - const Token head_token = backend->head("k/listed").token; - ASSERT_EQ(head_token.type, TokenType::Emulated); + const auto head = op.head("k/listed", Retry::once()); + ASSERT_TRUE(head.has_value()); + ASSERT_EQ(head->etag.dialect(), Dialect::Emulated); - const ListPage page = backend->list("k/", "", /*limit=*/10); + const ListPage page = op.list("k/", "", /*limit=*/10, Retry::once()); ASSERT_EQ(page.keys.size(), 1u); - ASSERT_TRUE(page.keys.front().token.has_value()); - EXPECT_EQ(*page.keys.front().token, head_token); + ASSERT_TRUE(page.keys.front().etag.has_value()); + EXPECT_EQ(*page.keys.front().etag, head->etag); } namespace @@ -1355,42 +1525,50 @@ DB::ObjectStoragePtr makeFixedEtagStorageForTest() /// the second. TEST(CASObjectStorageBackend, EmuTokenDisambiguatesSameEtagRewrite) { - ObjectStorageBackend backend(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); - const auto put1 = backend.putIfAbsent("k/tick", "v1"); - ASSERT_EQ(put1.outcome, PutOutcome::Done); - const auto put2 = backend.putOverwrite("k/tick", "v2", put1.token); - ASSERT_EQ(put2.outcome, PutOutcome::Done); + const WriteResult put1 = op.create("k/tick", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put1)); + const Etag inc1 = std::get(put1).etag; + const WriteResult put2 = op.replace("k/tick", "v2", inc1, Retry::once()); + ASSERT_TRUE(std::holds_alternative(put2)); + const Etag inc2 = std::get(put2).etag; - EXPECT_NE(put1.token.value, put2.token.value); - EXPECT_EQ(put1.token.type, TokenType::Emulated); - EXPECT_EQ(put2.token.type, TokenType::Emulated); + EXPECT_NE(inc1, inc2); + EXPECT_EQ(inc1.dialect(), Dialect::Emulated); + EXPECT_EQ(inc2.dialect(), Dialect::Emulated); - /// A stale delete using the FIRST incarnation's token must not match the live (second) one. - EXPECT_EQ(backend.deleteExact("k/tick", put1.token).kind, DeleteOutcome::Kind::TokenMismatch); - EXPECT_TRUE(backend.head("k/tick").exists); + /// A stale delete using the FIRST incarnation must not match the live (second) one. + EXPECT_EQ(op.remove("k/tick", inc1, Retry::once()), Removal::Mismatch); + EXPECT_TRUE(op.head("k/tick", Retry::once()).has_value()); } TEST(CASObjectStorageBackend, PublishBlobEmulatedDisambiguatesSameEtagFromStaleDelete) { - ObjectStorageBackend backend(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(makeFixedEtagStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String key = "k/publish-tick"; - ASSERT_EQ(backend.putIfAbsent(key, "old-complete-body").outcome, PutOutcome::Done); - const Token stale_token = backend.head(key).token; + ASSERT_TRUE(std::holds_alternative(op.create(key, "old-complete-body", Retry::once()))); + const auto stale = op.head(key, Retry::once()); + ASSERT_TRUE(stale.has_value()); + const Etag stale_token = stale->etag; - backend.publishBlob(streamingPublication(key, "fresh-envelope", "payload", 7)); + op.publish(streamingPublication(key, "fresh-envelope", "payload", 7), Retry::once()); - const HeadResult published = backend.head(key); - ASSERT_TRUE(published.exists); - EXPECT_NE(published.token, stale_token); - EXPECT_EQ(published.token.type, TokenType::Emulated); - EXPECT_EQ(backend.deleteExact(key, stale_token).kind, DeleteOutcome::Kind::TokenMismatch); + const auto published = op.head(key, Retry::once()); + ASSERT_TRUE(published.has_value()); + EXPECT_NE(published->etag, stale_token); + EXPECT_EQ(published->etag.dialect(), Dialect::Emulated); + EXPECT_EQ(op.remove(key, stale_token, Retry::once()), Removal::Mismatch); - const auto live = backend.get(key); + const auto live = op.read(key, Retry::once()); ASSERT_TRUE(live.has_value()); EXPECT_EQ(live->bytes, "fresh-envelopepayload"); - EXPECT_EQ(live->token, published.token); + EXPECT_EQ(live->etag, published->etag); } namespace @@ -1482,17 +1660,20 @@ TEST(CASObjectStorageBackend, DeleteExactErasesEmuTokenStateOnlyWhenEtagIsComfor /// etag, no disambiguator) rather than a same-quantum tie with the just-consumed delete token. { const String old_etag = "1000000000000000000"; - ObjectStorageBackend backend(makeFixedNumericEtagStorageForTest(old_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); - - const auto put1 = backend.putIfAbsent("k/old", "v1"); - ASSERT_EQ(put1.outcome, PutOutcome::Done); - ASSERT_EQ(put1.token.value, old_etag); - ASSERT_EQ(backend.deleteExact("k/old", put1.token).kind, DeleteOutcome::Kind::Deleted); - - const auto put2 = backend.putIfAbsent("k/old", "v2"); - ASSERT_EQ(put2.outcome, PutOutcome::Done); - EXPECT_EQ(put2.token.value, old_etag) << "entry should have been erased on delete (etag comfortably old), " - "so the recreate mints the bare etag, not a disambiguated one"; + auto backend = std::make_shared(makeFixedNumericEtagStorageForTest(old_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + + const WriteResult put1 = op.create("k/old", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put1)); + ASSERT_EQ(PersistedEtag::capture(std::get(put1).etag).value, old_etag); + ASSERT_EQ(op.remove("k/old", std::get(put1).etag, Retry::once()), Removal::Removed); + + const WriteResult put2 = op.create("k/old", "v2", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put2)); + EXPECT_EQ(PersistedEtag::capture(std::get(put2).etag).value, old_etag) + << "entry should have been erased on delete (etag comfortably old), " + "so the recreate mints the bare etag, not a disambiguated one"; } /// An etag within the safety margin of "now": delete must RETAIN the entry, so the same @@ -1501,17 +1682,20 @@ TEST(CASObjectStorageBackend, DeleteExactErasesEmuTokenStateOnlyWhenEtagIsComfor const auto now_ns = std::chrono::duration_cast( std::chrono::system_clock::now().time_since_epoch()).count(); const String recent_etag = std::to_string(now_ns); - ObjectStorageBackend backend(makeFixedNumericEtagStorageForTest(recent_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); - - const auto put1 = backend.putIfAbsent("k/fresh", "v1"); - ASSERT_EQ(put1.outcome, PutOutcome::Done); - ASSERT_EQ(put1.token.value, recent_etag); - ASSERT_EQ(backend.deleteExact("k/fresh", put1.token).kind, DeleteOutcome::Kind::Deleted); - - const auto put2 = backend.putIfAbsent("k/fresh", "v2"); - ASSERT_EQ(put2.outcome, PutOutcome::Done); - EXPECT_EQ(put2.token.value, recent_etag + "#1") << "entry should have been RETAINED on delete (etag recent), " - "so the recreate is disambiguated against it"; + auto backend = std::make_shared(makeFixedNumericEtagStorageForTest(recent_etag), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + + const WriteResult put1 = op.create("k/fresh", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put1)); + ASSERT_EQ(PersistedEtag::capture(std::get(put1).etag).value, recent_etag); + ASSERT_EQ(op.remove("k/fresh", std::get(put1).etag, Retry::once()), Removal::Removed); + + const WriteResult put2 = op.create("k/fresh", "v2", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put2)); + EXPECT_EQ(PersistedEtag::capture(std::get(put2).etag).value, recent_etag + "#1") + << "entry should have been RETAINED on delete (etag recent), " + "so the recreate is disambiguated against it"; } } @@ -1523,157 +1707,99 @@ TEST(CASObjectStorageBackend, EmuTokenStateEventuallyPrunesDistinctShortLivedKey constexpr size_t expected_recent_key_bound = 24; auto now_ns = std::make_shared>(start_ns); - ObjectStorageBackend backend(makeClockEtagStorageForTest(now_ns), ObjectStorageBackend::Mode::EmulatedSingleProcess); + auto backend = std::make_shared(makeClockEtagStorageForTest(now_ns), ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); for (size_t i = 0; i < key_count; ++i) { const uint64_t current_ns = start_ns + i * step_ns; now_ns->store(current_ns); - backend.setEmuNowNsForTest(current_ns); + backend->setEmuNowNsForTest(current_ns); const String key = "k/short-lived-" + std::to_string(i); - const auto put = backend.putIfAbsent(key, "body"); - ASSERT_EQ(put.outcome, PutOutcome::Done); - ASSERT_EQ(backend.deleteExact(key, put.token).kind, DeleteOutcome::Kind::Deleted); + const WriteResult put = op.create(key, "body", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); + ASSERT_EQ(op.remove(key, std::get(put).etag, Retry::once()), Removal::Removed); } const uint64_t sweep_ns = start_ns + key_count * step_ns + 2'000'000'000ULL; now_ns->store(sweep_ns); - backend.setEmuNowNsForTest(sweep_ns); - const auto trigger = backend.putIfAbsent("k/sweep-trigger", "body"); - ASSERT_EQ(trigger.outcome, PutOutcome::Done); - ASSERT_EQ(backend.deleteExact("k/sweep-trigger", trigger.token).kind, DeleteOutcome::Kind::Deleted); + backend->setEmuNowNsForTest(sweep_ns); + const WriteResult trigger = op.create("k/sweep-trigger", "body", Retry::once()); + ASSERT_TRUE(std::holds_alternative(trigger)); + ASSERT_EQ(op.remove("k/sweep-trigger", std::get(trigger).etag, Retry::once()), Removal::Removed); - EXPECT_LE(backend.emuTokenStateSizeForTest(), expected_recent_key_bound) + EXPECT_LE(backend->emuTokenStateSizeForTest(), expected_recent_key_bound) << "token state should track only the bounded recent-key window, not all " << key_count << " deleted keys"; } -namespace +TEST(CASObjectStorageBackend, EnsureBackendMatchesBudgetAcceptsAMatchingHandoff) { - -/// A `LocalObjectStorage` that counts `writeObject`/`removeObjectIfTokenMatches` calls -- used to -/// prove that a wrong-dialect expected token is rejected LOCALLY, before anything reaches the wire. -class CallCountingObjectStorage final : public DB::LocalObjectStorage -{ -public: - using DB::LocalObjectStorage::LocalObjectStorage; - - std::unique_ptr writeObject( - const DB::StoredObject & object, - DB::WriteMode mode, - std::optional attributes, - size_t buf_size, - const DB::WriteSettings & write_settings) override - { - ++write_calls; - return DB::LocalObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); - } - - DB::ConditionalRemoveResult removeObjectIfTokenMatches(const DB::StoredObject & object, const std::string & etag) override - { - ++remove_if_matches_calls; - return DB::LocalObjectStorage::removeObjectIfTokenMatches(object, etag); - } - - std::atomic write_calls{0}; - std::atomic remove_if_matches_calls{0}; -}; - -DB::ObjectStoragePtr makeCallCountingStorageForTest() -{ - static std::atomic counter{0}; - const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); - const auto root = (std::filesystem::temp_directory_path() / ("cas_call_counting_unit_" + unique)).string(); - - std::error_code ec; - std::filesystem::remove_all(root, ec); - std::filesystem::create_directories(root, ec); - - DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); - return std::make_shared(std::move(settings)); -} - + CasRequestBudget budget; + budget.attempt_timeout_ms = 4000; + budget.connect_timeout_cap_ms = 900; + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess, + /*single_attempt_control_plane_=*/false, budget.attempt_timeout_ms, *budget.connect_timeout_cap_ms); + EXPECT_NO_THROW(ensureBackendMatchesBudget(*backend, budget)); } -/// codex-review-triage §3.18, finding №19: Native-mode conditional mutations forward only -/// `Token::value` to the wire (`object_storage_write_if_match` / `removeObjectIfTokenMatches`), -/// blind to `Token::type`. A wrong-dialect token whose VALUE happens to equal the live incarnation's -/// must be rejected LOCALLY -- before any wire call is made -- never merely rely on the remote -/// backend to reject a foreign-dialect value it was never designed to compare. -TEST(CASObjectStorageBackend, NativeRejectsWrongDialectTokenBeforeTouchingTheWire) +TEST(CASObjectStorageBackend, EnsureBackendMatchesBudgetRejectsAMismatchedConnectCap) { - auto storage = std::static_pointer_cast(makeCallCountingStorageForTest()); - ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - - ASSERT_EQ(backend.putIfAbsent("k/dialect", "v1").outcome, PutOutcome::Done); - const Token live = backend.head("k/dialect").token; - ASSERT_EQ(live.type, TokenType::ETag); - - storage->write_calls = 0; - storage->remove_if_matches_calls = 0; - - /// Same wire VALUE, wrong dialect TYPE (Emulated instead of this backend's native ETag dialect). - const Token wrong_type_token{live.value, TokenType::Emulated}; - - EXPECT_EQ(backend.putOverwrite("k/dialect", "v2", wrong_type_token).outcome, PutOutcome::PreconditionFailed); - EXPECT_EQ(backend.casPut("k/dialect", "v2", wrong_type_token).outcome, CasOutcome::Conflict); - EXPECT_EQ(backend.deleteExact("k/dialect", wrong_type_token).kind, DeleteOutcome::Kind::TokenMismatch); - - EXPECT_EQ(storage->write_calls.load(), 0); - EXPECT_EQ(storage->remove_if_matches_calls.load(), 0); - - /// The live incarnation must be untouched by all three rejected attempts. - EXPECT_EQ(backend.head("k/dialect").token, live); + CasRequestBudget budget; + budget.attempt_timeout_ms = 4000; + budget.connect_timeout_cap_ms = 900; + /// The backend is built with a DIFFERENT connect cap than the budget it will be paired with -- + /// exactly the handoff mistake the production check at `openPoolView` guards against. + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess, + /*single_attempt_control_plane_=*/false, budget.attempt_timeout_ms, /*connect_timeout_cap_ms_=*/1500); + /// The mismatch is a programmer error (LOGICAL_ERROR): under DEBUG_OR_SANITIZER_BUILD it aborts the + /// process before it can be caught, so that arm pins the refusal via EXPECT_DEATH. +#if defined(DEBUG_OR_SANITIZER_BUILD) + EXPECT_DEATH({ ensureBackendMatchesBudget(*backend, budget); }, "does not match the pool's request budget"); +#else + EXPECT_THROW(ensureBackendMatchesBudget(*backend, budget), DB::Exception); +#endif } -/// §1 (opt round-B): the fold/point GETs read tiny bodies but a default `ReadBufferFromS3` preallocates -/// ~1 MiB. `casSizedReadSettings` shrinks the buffer to the known body size + slack, capped at the -/// caller's default — never larger than before, regardless of the reported size. -TEST(CASSizedReadSettings, CapsToKnownSizePlusSlackButNeverAboveBase) +TEST(CASObjectStorageBackend, EnsureBackendMatchesBudgetRejectsAMismatchedAttemptTimeout) { - DB::ReadSettings base; - base.remote_fs_settings.buffer_size = 1ULL << 20; /// 1 MiB default - base.local_fs_settings.buffer_size = 1ULL << 20; - - /// A ~3.7 KB fold body: buffer shrinks to size + slack, far below the 1 MiB default. - const auto small = DB::Cas::casSizedReadSettings(base, 3700); - EXPECT_EQ(small.remote_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); - EXPECT_EQ(small.local_fs_settings.buffer_size, 3700 + DB::Cas::CAS_FOLD_READ_SLACK_BYTES); - - /// A body larger than the default is capped AT the default (never grown). - const auto big = DB::Cas::casSizedReadSettings(base, 8ULL << 20); - EXPECT_EQ(big.remote_fs_settings.buffer_size, 1ULL << 20); - - /// Unknown size (0) = leave the base untouched (the metadata-fetch fallback path). - const auto unknown = DB::Cas::casSizedReadSettings(base, 0); - EXPECT_EQ(unknown.remote_fs_settings.buffer_size, 1ULL << 20); + CasRequestBudget budget; + budget.attempt_timeout_ms = 4000; + budget.connect_timeout_cap_ms = 900; + auto backend = std::make_shared( + tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess, + /*single_attempt_control_plane_=*/false, /*attempt_timeout_ms_=*/6000, *budget.connect_timeout_cap_ms); + /// Same split as the connect-cap twin above: a LOGICAL_ERROR aborts under DEBUG_OR_SANITIZER_BUILD. +#if defined(DEBUG_OR_SANITIZER_BUILD) + EXPECT_DEATH({ ensureBackendMatchesBudget(*backend, budget); }, "does not match the pool's request budget"); +#else + EXPECT_THROW(ensureBackendMatchesBudget(*backend, budget), DB::Exception); +#endif } -/// The CountingBackend request-shape recorders that the streaming-memory gates (Task 3/4) consume: -/// per-key/total getStream counts, the max ranged-get window per key, and the whole-object get flag. -TEST(CASCountingBackendShape, RecordsGetStreamAndRangeShape) -{ - DB::Cas::tests::CountingBackend backend; - backend.putIfAbsent("k", String(1000, 'x')); - - /// A whole-object get flags the resident-memory violation; a ranged get tracks the max window. - backend.get("k"); - backend.get("k", DB::Cas::Range{.offset = 0, .length = 100}); - backend.get("k", DB::Cas::Range{.offset = 10, .length = 400}); - EXPECT_EQ(backend.wholeGetCount("k"), 1u); - EXPECT_EQ(backend.maxRangedGetLen("k"), 400u); - - /// getStream counters (per-key and total). - backend.getStream("k", DB::Cas::Range{.offset = 2, .length = 5}); - backend.getStream("k"); - backend.getStream("absent"); - EXPECT_EQ(backend.getStreamCount("k"), 2u); - EXPECT_EQ(backend.getStreamTotal(), 3u); - - backend.resetCounts(); - EXPECT_EQ(backend.wholeGetCount("k"), 0u); - EXPECT_EQ(backend.maxRangedGetLen("k"), 0u); - EXPECT_EQ(backend.getStreamTotal(), 0u); -} +/// `NativeRejectsWrongDialectTokenBeforeTouchingTheWire` is deleted here: it built a `Token{value, +/// Dialect::Emulated}` holding a NATIVE backend's live wire value under the WRONG dialect tag, to prove +/// the mismatch was caught locally rather than forwarded to the wire. `Etag` no longer admits that +/// construction -- it is minted ONLY by `CasRequests::mint`/`tryMint`, always from `backend->dialect()`, +/// so a caller can never hold an `Etag` tagged with a dialect other than the backend that observed it. +/// The property this test pinned ("a value observed under one dialect can never be mistaken for another +/// backend's incarnation") is now enforced by the type itself rather than by a runtime comparison; see +/// `CasRequests::valueFor`'s backend-identity check (also exercised, from the other side, by +/// `EmuTokenSurvivesProcessRestartAcrossRecreate` above). +/// +/// `CASBackendGrammar.RejectsEmptyStarAndListTokensOnEveryMutation` and its +/// `CASBackendGrammarDeathTest` sibling are deleted for the same reason: they built literal +/// `Token{"", ...}` / `Token{"*", ...}` / `Token{"\"a\", \"b\"", ...}` values to drive `putOverwrite`/ +/// `casPut`/`deleteExact` into the primitive's `LOGICAL_ERROR` grammar guard. `Etag::mint`/`tryMint` +/// refuse to construct an `Etag` from a malformed value in the first place (`CORRUPTED_DATA`), so no +/// caller reaching the primitives through `CasOperation` can ever hold one -- the grammar guard inside +/// `ObjectStorageBackend::write`/`removeUnder` is unreachable from the public engine surface and stays +/// as defense in depth only. The empty/`*`/quoted-list token cases the deleted tests drove are covered +/// at the `Etag::mint`/`tryMint` boundary by `CASIncarnation.GrammarRefusesTheNineWays` +/// (`gtest_cas_requests.cpp`); the grammar predicate itself remains directly pinned by +/// `CASBackendGrammar.GenerationDialectAcceptsOnlyCanonicalPositiveDecimal` above. #endif diff --git a/src/Disks/tests/gtest_cas_backend_contract.cpp b/src/Disks/tests/gtest_cas_backend_contract.cpp index bf5c2652ca2a..ee7a47fa1b89 100644 --- a/src/Disks/tests/gtest_cas_backend_contract.cpp +++ b/src/Disks/tests/gtest_cas_backend_contract.cpp @@ -2,15 +2,20 @@ #include #include #include +#include #include #include #include +#include using namespace DB::Cas; -/// Parameterized contract suite: every case creates a fresh backend from the factory, -/// then exercises the Backend seam generically (no InMemoryBackend-specific calls). -/// Fault-injection-only features are excluded — those are InMemory-specific tests. +using DB::Cas::tests::expectBytes; +using DB::Cas::tests::openRequestsForTest; + +/// Parameterized contract suite: every case creates a fresh backend from the factory, then exercises +/// the seam generically through `CasRequests`/`CasOperation` over an open fence (no InMemoryBackend- +/// specific calls). Fault-injection-only features are excluded -- those are InMemory-specific tests. class CASBackendContract : public ::testing::TestWithParam> { }; @@ -18,135 +23,173 @@ class CASBackendContract : public ::testing::TestWithParamputIfAbsent("k", "v1"); - const Token t1 = put.token; - EXPECT_EQ(put.outcome, PutOutcome::Done); - EXPECT_FALSE(t1.empty()); - EXPECT_EQ(b->putIfAbsent("k", "clobber").outcome, PutOutcome::PreconditionFailed); - auto g = b->get("k"); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto put = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); + const Etag t1 = std::get(put).etag; + EXPECT_TRUE(std::holds_alternative(op.create("k", "clobber", Retry::once()))); + auto g = op.read("k", Retry::once()); ASSERT_TRUE(g.has_value()); EXPECT_EQ(g->bytes, "v1"); - EXPECT_EQ(g->token, t1); - EXPECT_FALSE(b->get("absent").has_value()); + EXPECT_EQ(g->etag, t1); + EXPECT_FALSE(op.read("absent", Retry::once()).has_value()); } +/// A wrong-but-REAL precondition, since `Etag` has no public constructor any more: overwriting the key +/// once legitimately mints a second incarnation, which makes the FIRST one genuinely stale for this +/// same key -- a value the engine accepts as a precondition (unlike a fabricated one) but refuses as +/// the wrong one, because the object has already moved past it. TEST_P(CASBackendContract, OverwriteIsTokenExactAndMintsFreshToken) { auto b = GetParam()(); - const Token t1 = b->putIfAbsent("k", "v1").token; - EXPECT_EQ(b->putOverwrite("k", "v2", Token{"wrong", TokenType::Emulated}).outcome, PutOutcome::PreconditionFailed); - EXPECT_EQ(b->get("k")->bytes, "v1"); // untouched on mismatch - const auto overwrite = b->putOverwrite("k", "v2", t1); - EXPECT_EQ(overwrite.outcome, PutOutcome::Done); - EXPECT_NE(overwrite.token, t1); // tokens never repeat - EXPECT_EQ(b->get("k")->bytes, "v2"); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto created = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(created)); + const Etag t1 = std::get(created).etag; + const auto warmup = op.replace("k", "v1b", t1, Retry::once()); // mints a second incarnation, so t1 goes stale + ASSERT_TRUE(std::holds_alternative(warmup)); + const Etag t2 = std::get(warmup).etag; + + EXPECT_TRUE(std::holds_alternative(op.replace("k", "v2", t1, Retry::once()))); + expectBytes(b, "k", "v1b"); // untouched on mismatch + + const auto overwrite = op.replace("k", "v2", t2, Retry::once()); + ASSERT_TRUE(std::holds_alternative(overwrite)); + EXPECT_NE(std::get(overwrite).etag, t2); // etags never repeat + expectBytes(b, "k", "v2"); } TEST_P(CASBackendContract, CasPutCreateAndSwap) { auto b = GetParam()(); - const auto create = b->casPut("m", "s1", std::nullopt); - const Token t1 = create.token; - EXPECT_EQ(create.outcome, CasOutcome::Committed); // create-if-absent - EXPECT_EQ(b->casPut("m", "s1x", std::nullopt).outcome, CasOutcome::Conflict); // exists now - EXPECT_EQ(b->casPut("m", "s2", Token{"stale", TokenType::Emulated}).outcome, CasOutcome::Conflict); - EXPECT_EQ(b->get("m")->bytes, "s1"); - EXPECT_EQ(b->casPut("m", "s2", t1).outcome, CasOutcome::Committed); - EXPECT_EQ(b->get("m")->bytes, "s2"); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto create = op.create("m", "s1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(create)); // create-if-absent + const Etag t1 = std::get(create).etag; + EXPECT_TRUE(std::holds_alternative(op.create("m", "s1x", Retry::once()))); // exists now + + /// Mint a second real incarnation so `t1` becomes a genuinely stale (never fabricated) wrong swap. + const auto warmup = op.replace("m", "s1y", t1, Retry::once()); + ASSERT_TRUE(std::holds_alternative(warmup)); + const Etag t2 = std::get(warmup).etag; + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t1, Retry::once()))); + expectBytes(b, "m", "s1y"); + + EXPECT_TRUE(std::holds_alternative(op.replace("m", "s2", t2, Retry::once()))); + expectBytes(b, "m", "s2"); } TEST_P(CASBackendContract, DeleteExactnessAndSurvival) { auto b = GetParam()(); - const Token t1 = b->putIfAbsent("k", "v1").token; - auto d1 = b->deleteExact("k", Token{"wrong", TokenType::Emulated}); - EXPECT_EQ(d1.kind, DeleteOutcome::Kind::TokenMismatch); - EXPECT_TRUE(b->get("k").has_value()); // SURVIVES wrong-token delete - auto d2 = b->deleteExact("k", t1); - EXPECT_EQ(d2.kind, DeleteOutcome::Kind::Deleted); - EXPECT_FALSE(d2.created_delete_marker); - EXPECT_FALSE(b->get("k").has_value()); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto created = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(created)); + const Etag t1 = std::get(created).etag; + + /// A real but stale incarnation, minted by a legitimate overwrite (see the comment on + /// `OverwriteIsTokenExactAndMintsFreshToken`). + const auto warmup = op.replace("k", "v1b", t1, Retry::once()); + ASSERT_TRUE(std::holds_alternative(warmup)); + const Etag t2 = std::get(warmup).etag; + + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Mismatch); + EXPECT_TRUE(op.read("k", Retry::once()).has_value()); // SURVIVES wrong-incarnation delete + EXPECT_EQ(op.remove("k", t2, Retry::once()), Removal::Removed); + EXPECT_FALSE(op.read("k", Retry::once()).has_value()); } TEST_P(CASBackendContract, DeleteNotFound) { auto b = GetParam()(); - const Token t1 = b->putIfAbsent("k", "v1").token; - b->deleteExact("k", t1); - EXPECT_EQ(b->deleteExact("k", t1).kind, DeleteOutcome::Kind::NotFound); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto created = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(created)); + const Etag t1 = std::get(created).etag; + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Removed); + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Gone); } -TEST_P(CASBackendContract, RangeGet) -{ - auto b = GetParam()(); - b->putIfAbsent("k", "0123456789"); - Range r; - r.offset = 2; - r.length = 3u; - EXPECT_EQ(b->get("k", r)->bytes, "234"); -} +/// `Range` had no primitive read counterpart even before this migration -- `op.read` takes no range +/// argument at all, so a non-whole window is refused by the TYPE, not by a runtime NOT_IMPLEMENTED +/// throw. The property (the whole read still serves) is what `ReadAfterWrite` below already pins. TEST_P(CASBackendContract, Head) { auto b = GetParam()(); - b->putIfAbsent("k", "hello"); - auto h = b->head("k"); - EXPECT_TRUE(h.exists); - EXPECT_EQ(h.size, 5u); - EXPECT_FALSE(h.token.empty()); - auto h2 = b->head("missing"); - EXPECT_FALSE(h2.exists); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("k", "hello", Retry::once()))); + auto h = op.head("k", Retry::once()); + ASSERT_TRUE(h.has_value()); + EXPECT_EQ(h->size, 5u); + auto h2 = op.head("missing", Retry::once()); + EXPECT_FALSE(h2.has_value()); } TEST_P(CASBackendContract, ListPagination) { auto b = GetParam()(); - b->putIfAbsent("p/a", "0123456789"); - b->putIfAbsent("p/b", "xy"); - b->putIfAbsent("q/c", "z"); - auto page = b->list("p/", "", 10); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("p/a", "0123456789", Retry::once()))); + ASSERT_TRUE(std::holds_alternative(op.create("p/b", "xy", Retry::once()))); + ASSERT_TRUE(std::holds_alternative(op.create("q/c", "z", Retry::once()))); + auto page = op.list("p/", "", 10, Retry::once()); ASSERT_EQ(page.keys.size(), 2u); // sorted, prefix-scoped EXPECT_EQ(page.keys[0].key, "p/a"); EXPECT_EQ(page.keys[1].key, "p/b"); EXPECT_TRUE(page.next_cursor.empty()); - auto page1 = b->list("p/", "", 1); // pagination + auto page1 = op.list("p/", "", 1, Retry::once()); // pagination EXPECT_EQ(page1.keys.size(), 1u); EXPECT_EQ(page1.keys[0].key, "p/a"); EXPECT_EQ(page1.next_cursor, "p/a"); EXPECT_FALSE(page1.next_cursor.empty()); - auto page2 = b->list("p/", page1.next_cursor, 1); + auto page2 = op.list("p/", page1.next_cursor, 1, Retry::once()); EXPECT_EQ(page2.keys[0].key, "p/b"); } TEST_P(CASBackendContract, ReadAfterWrite) { auto b = GetParam()(); - const Token t1 = b->putIfAbsent("rw", "payload").token; - auto g = b->get("rw"); + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto created = op.create("rw", "payload", Retry::once()); + ASSERT_TRUE(std::holds_alternative(created)); + const Etag t1 = std::get(created).etag; + auto g = op.read("rw", Retry::once()); ASSERT_TRUE(g.has_value()); EXPECT_EQ(g->bytes, "payload"); - EXPECT_EQ(g->token, t1); - auto h = b->head("rw"); - EXPECT_TRUE(h.exists); - EXPECT_EQ(h.token, t1); + EXPECT_EQ(g->etag, t1); + auto h = op.head("rw", Retry::once()); + ASSERT_TRUE(h.has_value()); + EXPECT_EQ(h->etag, t1); } -/// After an object is created then deleted (key absent again), BOTH conditional updates against a stale -/// token must be rejected with the object still absent — a token-conditional update can never recreate a -/// missing key. For the Native S3 adapter this pins the 404-on-If-Match -> PreconditionFailed/Conflict -/// mapping; for every backend it pins that absence is not a write opportunity for a stale token. +/// After an object is created then deleted (key absent again), a conditional update against the +/// incarnation it held while alive must be rejected with the object still absent -- an +/// incarnation-conditional update can never recreate a missing key. For the Native S3 adapter this +/// pins the 404-on-If-Match -> Conflict mapping; for every backend it pins that absence is not a write +/// opportunity for a since-deleted incarnation. The legacy `putOverwrite` and `casPut(expected)` +/// verbs this test used to drive separately both reach this SAME primitive (`replace`) now. TEST_P(CASBackendContract, OverwriteAndCasOnMissingKey) { auto b = GetParam()(); - const Token t1 = b->putIfAbsent("k", "v1").token; - EXPECT_EQ(b->deleteExact("k", t1).kind, DeleteOutcome::Kind::Deleted); - ASSERT_FALSE(b->get("k").has_value()); // key is absent - - EXPECT_EQ(b->putOverwrite("k", "v2", t1).outcome, PutOutcome::PreconditionFailed); - EXPECT_FALSE(b->get("k").has_value()); // still absent - - EXPECT_EQ(b->casPut("k", "v2", t1).outcome, CasOutcome::Conflict); - EXPECT_FALSE(b->get("k").has_value()); // still absent + auto requests = openRequestsForTest(b); + auto op = requests.admit(); + const auto created = op.create("k", "v1", Retry::once()); + ASSERT_TRUE(std::holds_alternative(created)); + const Etag t1 = std::get(created).etag; + EXPECT_EQ(op.remove("k", t1, Retry::once()), Removal::Removed); + ASSERT_FALSE(op.read("k", Retry::once()).has_value()); // key is absent + + EXPECT_TRUE(std::holds_alternative(op.replace("k", "v2", t1, Retry::once()))); + EXPECT_FALSE(op.read("k", Retry::once()).has_value()); // still absent } INSTANTIATE_TEST_SUITE_P(CASInMemory, CASBackendContract, diff --git a/src/Disks/tests/gtest_cas_backend_generation.cpp b/src/Disks/tests/gtest_cas_backend_generation.cpp index b6f9502c45fa..77c3d0a2d647 100644 --- a/src/Disks/tests/gtest_cas_backend_generation.cpp +++ b/src/Disks/tests/gtest_cas_backend_generation.cpp @@ -1,5 +1,7 @@ #include #include +#include +#include #include #include @@ -24,6 +26,8 @@ #include #include #include +#include +#include #include #include @@ -123,9 +127,41 @@ std::shared_ptr makeVersioningObjectStorageForTest(std: return std::make_shared(std::move(settings), versioned); } +/// Captures what `ObjectStorageBackend` logs at WARNING and above, so a test can assert both that a +/// warning was raised and that none was. Same shape as the capture in gtest_cas_settings.cpp. +class ScopedBackendLogCapture +{ +public: + ScopedBackendLogCapture() + : logger(getLogger("CasObjectStorageBackend")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("warning"); + } + + ~ScopedBackendLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; + Poco::AutoPtr channel; + /// `shared=true` is load-bearing: `AutoPtr(ptr)` would steal a reference the fixture never owned. + Poco::AutoPtr old_channel; + int old_level; +}; + /// Every refusal reached from these mount gates is `NOT_IMPLEMENTED`, so the code alone cannot tell /// which one fired. Match a phrase unique to the intended message as well, or a test asserting the -/// unverifiable-versioning refusal would pass on the enabled-bucket refusal and vice versa. +/// enabled-versioning refusal would pass on the skip-access-check refusal and vice versa. template void expectThrowsNotImplementedSaying(const std::string & needle, F && fn) { @@ -149,80 +185,108 @@ TEST(CASBackendGeneration, NativeHeadUsesNativeTokenMetadataApi) auto storage = makeRecordingObjectStorageForTest(); auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); - ASSERT_EQ(b->putIfAbsent("p/native-head/key", "v1").outcome, PutOutcome::Done); + /// Placed through the object storage: a Native write over a local storage has no response + /// incarnation to attribute itself to. Native passes the key verbatim, so this is the object the + /// HEAD below reads -- anchored under the storage's own root, since a bare relative key would + /// resolve beside the test process. + const String key = DB::Cas::tests::nativeKeyUnder(storage, "p/native-head/key"); + { + auto out = storage->writeObject( + DB::StoredObject(key), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, DB::WriteSettings{}); + DB::writeString(String("v1"), *out); + out->finalize(); + } - /// putIfAbsent's own HEAD-fallback stamping path calls the ordinary API (untouched by this task); - /// reset the counters so only nativeHead's call, below, is observed. + /// Only nativeHead's own call may be observed. storage->ordinary_calls = 0; storage->native_calls = 0; - const auto hr = b->head("p/native-head/key"); - ASSERT_TRUE(hr.exists); + DB::Cas::tests::OperationForTest op(*b); + const auto hr = (*op).head(key, Retry::once()); + ASSERT_TRUE(hr.has_value()); EXPECT_EQ(storage->native_calls, 1); EXPECT_EQ(storage->ordinary_calls, 0); } -/// Every Token{...} the backend mints must carry native_token_type instead of a hardcoded -/// TokenType::ETag (Task 5). Mode::Native over a LocalObjectStorage has no write-time ETag, so -/// putIfAbsent's PutResult falls back to a HEAD internally — that HEAD is also a stamping site, -/// so the assertion below exercises both the direct-etag and the HEAD-fallback mint paths. +/// Every token the backend mints carries native_token_type rather than a hardcoded Dialect::ETag. +/// The HEAD mint is the site exercised here; the write-response mint has its own tests over the fake +/// S3 client below, which is the only place a Native write can produce a response incarnation. TEST(CASBackendGeneration, StampedTokenTypeFollowsNativeKind) { - auto b = std::make_shared( - DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); + b->setNativeTokenTypeForTest(Dialect::Generation); - const auto put = b->putIfAbsent("p/gen/tok", "v1"); - EXPECT_EQ(put.token.type, TokenType::Generation); + /// A local file's etag is its mtime in nanoseconds, which is also a valid generation value. + const String key = DB::Cas::tests::nativeKeyUnder(storage, "p/gen/tok"); + { + auto out = storage->writeObject(DB::StoredObject(key), DB::WriteMode::Rewrite); + DB::writeString(String("v1"), *out); + out->finalize(); + } - const auto hr = b->head("p/gen/tok"); - ASSERT_TRUE(hr.exists); - EXPECT_EQ(hr.token.type, TokenType::Generation); + DB::Cas::tests::OperationForTest op(*b); + const auto hr = (*op).head(key, Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_EQ(hr->etag.dialect(), Dialect::Generation); + EXPECT_EQ(b->dialect(), Dialect::Generation); } -/// A generation-dialect (GCS) mount needs bucket versioning to be VERIFIABLY off: a token-exact +/// A generation-dialect (GCS) mount wants bucket versioning to be verifiably off: a token-exact /// DELETE against a versioned bucket archives a noncurrent generation, so GC would delete objects it -/// believes it reclaimed. A probe that cannot answer therefore refuses the mount rather than -/// assuming the safe answer. -TEST(CASBackendGeneration, CheckPoolPreconditionsFailsClosedOnUnverifiableVersioning) +/// believes it reclaimed. A probe that cannot answer is not evidence of a versioned bucket, though: +/// the usual cause is a credential without permission to read the bucket configuration, and +/// refusing on it turns a missing IAM grant into a hard outage. So the mount proceeds, and says +/// loudly what it could not verify and how the operator can. +TEST(CASBackendGeneration, CheckPoolPreconditionsWarnsAndContinuesOnUnverifiableVersioning) { auto b = std::make_shared( makeVersioningObjectStorageForTest(std::nullopt), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); + + ScopedBackendLogCapture capture; + EXPECT_NO_THROW(b->checkPoolPreconditions()); - expectThrowsNotImplementedSaying("could not VERIFY", [&] { b->checkPoolPreconditions(); }); + const auto logged = capture.captured(); + EXPECT_NE(logged.find("could not VERIFY"), String::npos) << logged; + EXPECT_NE(logged.find("versioning"), String::npos) << logged; + EXPECT_NE(logged.find("storage.buckets.get"), String::npos) << logged; } TEST(CASBackendGeneration, CheckPoolPreconditionsRejectsEnabledVersioning) { auto b = std::make_shared( makeVersioningObjectStorageForTest(true), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); expectThrowsNotImplementedSaying("VERSIONING enabled", [&] { b->checkPoolPreconditions(); }); } -/// The one accepting case: a probe that answered, and answered "disabled". -TEST(CASBackendGeneration, CheckPoolPreconditionsAcceptsVerifiedDisabledVersioning) +/// The fully verified case: a probe that answered, and answered "disabled". Nothing to warn about. +TEST(CASBackendGeneration, CheckPoolPreconditionsAcceptsVerifiedDisabledVersioningSilently) { auto b = std::make_shared( makeVersioningObjectStorageForTest(false), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); + ScopedBackendLogCapture capture; EXPECT_NO_THROW(b->checkPoolPreconditions()); + EXPECT_TRUE(capture.captured().empty()) << capture.captured(); } /// The ETag-dialect (AWS-compatible) backend never consults bucket versioning at all — the check is -/// a silent no-op for any backend that is not Native + TokenType::Generation. Driven over a storage -/// whose probe is unverifiable, which is what a generation-dialect backend now refuses: dropping the -/// dialect guard from checkPoolPreconditions would fail this test. +/// a silent no-op for any backend that is not Native + Dialect::Generation. Driven over a storage +/// whose probe is unverifiable, which is what a generation-dialect backend warns about: dropping the +/// dialect guard from checkPoolPreconditions would fail the silence assertion. TEST(CASBackendGeneration, CheckPoolPreconditionsNoOpOnEtagDialect) { auto b = std::make_shared( makeVersioningObjectStorageForTest(std::nullopt), ObjectStorageBackend::Mode::Native); - ASSERT_EQ(b->nativeTokenType(), TokenType::ETag); + ASSERT_EQ(b->nativeTokenType(), Dialect::ETag); + ScopedBackendLogCapture capture; EXPECT_NO_THROW(b->checkPoolPreconditions()); + EXPECT_TRUE(capture.captured().empty()) << capture.captured(); } /// A writable generation-dialect (GCS) mount may not skip the mutating capability battery: that @@ -231,7 +295,7 @@ TEST(CASBackendGeneration, CheckSkipAccessCheckSupportRejectsGenerationDialect) { auto b = std::make_shared( DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); expectThrowsNotImplementedSaying("skip_access_check=true is not supported", [&] { b->checkSkipAccessCheckSupport(); }); } @@ -242,7 +306,7 @@ TEST(CASBackendGeneration, CheckSkipAccessCheckSupportAllowsEtagAndEmulatedBacke { auto etag = std::make_shared( DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - ASSERT_EQ(etag->nativeTokenType(), TokenType::ETag); + ASSERT_EQ(etag->nativeTokenType(), Dialect::ETag); EXPECT_NO_THROW(etag->checkSkipAccessCheckSupport()); auto emulated = std::make_shared( @@ -261,9 +325,9 @@ TEST(CASBackendGeneration, ListTokensDisabledOnGenerationStores) auto b = std::make_shared( DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); EXPECT_TRUE(b->supportsListTokens()); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); EXPECT_FALSE(b->supportsListTokens()); - b->setNativeTokenTypeForTest(TokenType::ETag); + b->setNativeTokenTypeForTest(Dialect::ETag); EXPECT_TRUE(b->supportsListTokens()); } @@ -271,7 +335,7 @@ TEST(CASBackendGeneration, ConditionalWriteSettingsForceSinglePutOnGenerationSto { auto b = std::make_shared( DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - b->setNativeTokenTypeForTest(TokenType::Generation); + b->setNativeTokenTypeForTest(Dialect::Generation); const auto ws = b->conditionalWriteSettingsForTest(); EXPECT_EQ(ws.object_storage_request_mode, DB::ObjectStorageRequestMode::NativeConditional); EXPECT_TRUE(ws.s3_force_single_part_upload); @@ -280,7 +344,7 @@ TEST(CASBackendGeneration, ConditionalWriteSettingsForceSinglePutOnGenerationSto ASSERT_TRUE(ws.s3_check_objects_after_upload_override.has_value()); EXPECT_FALSE(*ws.s3_check_objects_after_upload_override); - b->setNativeTokenTypeForTest(TokenType::ETag); + b->setNativeTokenTypeForTest(Dialect::ETag); const auto ws2 = b->conditionalWriteSettingsForTest(); EXPECT_EQ(ws2.object_storage_request_mode, DB::ObjectStorageRequestMode::NativeConditional); EXPECT_FALSE(ws2.s3_force_single_part_upload); @@ -290,32 +354,14 @@ TEST(CASBackendGeneration, ConditionalWriteSettingsForceSinglePutOnGenerationSto EXPECT_FALSE(*ws2.s3_check_objects_after_upload_override); } -/// C1: the three token-policy helpers are the single source of truth for how a Native-mode backend -/// mints a HEAD/PUT token, gates a LIST token, and compares tokens. Characterizes the behavior the -/// scattered call sites have today so the consolidation stays byte-for-byte behavior-preserving. -TEST(CASBackendGeneration, TokenPolicyHelpersAreConsistentWithDialect) -{ - auto b = std::make_shared( - DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - - /// ETag dialect: head/put tokens carry ETag; list surfaces the same-typed token for a non-empty etag. - ASSERT_EQ(b->nativeTokenType(), TokenType::ETag); - EXPECT_EQ(b->tokenForHead("abc").type, TokenType::ETag); - EXPECT_EQ(b->tokenForHead("abc"), (Token{"abc", TokenType::ETag})); - ASSERT_TRUE(b->tokenForList("abc").has_value()); - EXPECT_EQ(*b->tokenForList("abc"), b->tokenForHead("abc")); /// list token == head token (same etag) - EXPECT_FALSE(b->tokenForList("").has_value()); /// empty etag => no list token - - /// Generation dialect (GCS): head token flips to Generation; list tokens are disabled wholesale - /// (poisoned If-Match), so tokenForList is always nullopt regardless of the etag. - b->setNativeTokenTypeForTest(TokenType::Generation); - EXPECT_EQ(b->tokenForHead("g1").type, TokenType::Generation); - EXPECT_FALSE(b->tokenForList("g1").has_value()); - - /// tokenMatches is exact identity (value AND type) — a same-value/different-type token never matches. - EXPECT_TRUE(ObjectStorageBackend::tokenMatches(Token{"x", TokenType::ETag}, Token{"x", TokenType::ETag})); - EXPECT_FALSE(ObjectStorageBackend::tokenMatches(Token{"x", TokenType::ETag}, Token{"x", TokenType::Emulated})); -} +/// `tokenForHead` and `tokenMatches` were deleted with the `Token` type they built: minting an +/// incarnation and comparing it against a precondition are now `CasRequests::mint`/`tryMint` and +/// `CasRequests::valueFor`, always stamped with the observing backend's own dialect, so no caller can +/// construct or compare one by hand any more. The remaining piece, `tokenForList`'s ETag/Generation +/// gating, stays pinned by `CASBackendGeneration.ListTokensDisabledOnGenerationStores` above; +/// dialect-aware value validation is `CASBackendGrammar.GenerationDialectAcceptsOnlyCanonicalPositiveDecimal` +/// and the cross-backend precondition guard `CASObjectStorageBackend.EmuTokenSurvivesProcessRestartAcrossRecreate`, +/// both in gtest_cas_backend.cpp. #if USE_AWS_S3 @@ -556,7 +602,7 @@ class CASBackendGenerationS3 : public ::testing::Test /// A fresh backend, native token type forced to Generation unless overridden (the ETag dialect /// is needed to prove the generation-only quote handling does not touch it). - std::shared_ptr makeBackend(TokenType token_type = TokenType::Generation) + std::shared_ptr makeBackend(Dialect token_type = Dialect::Generation) { auto storage = makeGenerationS3ObjectStorageForTest(client); auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); @@ -572,10 +618,11 @@ TEST(CASBackendGeneration, PublishBlobAboveFormerGenerationCapUsesOrdinaryMultip auto storage = makeGenerationS3ObjectStorageForTest( client, /*force_multipart=*/true, /*conditional_put_cap=*/16); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - backend.setNativeTokenTypeForTest(TokenType::Generation); + backend.setNativeTokenTypeForTest(Dialect::Generation); const String payload(1024, 'x'); - backend.publishBlob(BlobPublishRequest{ + DB::Cas::tests::OperationForTest op(backend); + (*op).publish(BlobPublishRequest{ .destination_key = "p/gen/publish-multipart", .publication = StreamingBlobPublication{ .payload_size = payload.size(), @@ -583,7 +630,7 @@ TEST(CASBackendGeneration, PublishBlobAboveFormerGenerationCapUsesOrdinaryMultip .open_payload = [payload] { return std::make_unique(payload); - }}}); + }}}, Retry::once()); EXPECT_EQ(client->put_object_calls, 0u); EXPECT_EQ(client->create_multipart_calls, 1u); @@ -601,11 +648,12 @@ TEST(CASBackendGeneration, PublishBlobSucceedsWithoutResponseGeneration) auto storage = makeGenerationS3ObjectStorageForTest( client, /*force_multipart=*/false, /*conditional_put_cap=*/1); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - backend.setNativeTokenTypeForTest(TokenType::Generation); + backend.setNativeTokenTypeForTest(Dialect::Generation); client->put_returns_no_etag = true; const String payload = "payload"; - EXPECT_NO_THROW(backend.publishBlob(BlobPublishRequest{ + DB::Cas::tests::OperationForTest op(backend); + EXPECT_NO_THROW((*op).publish(BlobPublishRequest{ .destination_key = "p/gen/publish-no-generation", .publication = StreamingBlobPublication{ .payload_size = payload.size(), @@ -613,13 +661,41 @@ TEST(CASBackendGeneration, PublishBlobSucceedsWithoutResponseGeneration) .open_payload = [payload] { return std::make_unique(payload); - }}})); + }}}, Retry::once())); EXPECT_EQ(client->put_object_calls, 1u); EXPECT_EQ(client->head_object_calls, 0u); EXPECT_EQ(client->objects.at("p/gen/publish-no-generation"), "freshpayload"); } +/// The write-response half of the incarnation grammar: a write response that carries no ETag at all +/// must not fall back to a HEAD -- there is no HEAD that can attribute the write with certainty, since +/// the object it would read back might not even be the one this call just wrote. This is the +/// ETag-dialect sibling of PublishBlobSucceedsWithoutResponseGeneration above: a publication has no +/// incarnation to attribute in the first place, so it is unaffected by this guard. Default (ETag) +/// dialect here, deliberately NOT stamped Generation, whose own two cases are covered by +/// CASBackendGenerationS3.WriteEmptyGenerationIsUnresolvedNotThrown and WriteNonNumericGenerationIsUnresolvedNotThrown. +TEST(CASBackendGrammar, NamelessWriteResponseIsUnresolvedNotThrown) +{ + (void)getContext(); + FakeGenerationS3Client * client = nullptr; + auto storage = makeGenerationS3ObjectStorageForTest(client); + ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); + ASSERT_EQ(backend.nativeTokenType(), Dialect::ETag); + client->put_returns_no_etag = true; + + /// A 2xx write reply carrying no usable incarnation is an ambiguity the engine settles by a + /// resolve read (CasRequests::writeLoop), never an immediate corruption verdict; the mock's + /// resolve GET finds nothing at the key, so the create-if-absent precondition is still + /// satisfiable and a single-attempt policy reports GaveUp{Unresolved} rather than throwing. + DB::Cas::tests::OperationForTest op(backend); + const WriteResult result = (*op).create("p/gen/nameless-write", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_EQ(client->put_object_calls, 1u); +} + /// The moved cap, end to end: a conditional write on a generation store stays in ONE PUT up to the /// cap the OBJECT STORAGE carries, and refuses rather than silently taking the multipart path above /// it -- GCS enforces no precondition on CompleteMultipartUpload. @@ -630,13 +706,14 @@ TEST(CASBackendGeneration, ConditionalWriteHonoursTheObjectStorageConditionalPut auto storage = makeGenerationS3ObjectStorageForTest( client, /*force_multipart=*/false, /*conditional_put_cap=*/64); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - backend.setNativeTokenTypeForTest(TokenType::Generation); + backend.setNativeTokenTypeForTest(Dialect::Generation); const String small(32, 'a'); - EXPECT_NO_THROW(backend.casPut("p/gen/under-cap", small, std::nullopt, ObjectMeta{})); + DB::Cas::tests::OperationForTest op(backend); + EXPECT_NO_THROW((*op).create("p/gen/under-cap", small, Retry::once())); EXPECT_EQ(client->put_object_calls, 1u); EXPECT_EQ(client->create_multipart_calls, 0u); - const auto single_attempt_client = storage->getSingleAttemptClient(); + const auto single_attempt_client = storage->getSingleAttemptClient(/*request_timeout_ms=*/0); EXPECT_NE(dynamic_cast(single_attempt_client.get()), nullptr); EXPECT_NE( dynamic_cast( @@ -646,7 +723,7 @@ TEST(CASBackendGeneration, ConditionalWriteHonoursTheObjectStorageConditionalPut const String large(4096, 'b'); try { - backend.casPut("p/gen/over-cap", large, std::nullopt, ObjectMeta{}); + (*op).create("p/gen/over-cap", large, Retry::once()); FAIL() << "a conditional write above the cap must refuse, not go multipart"; } catch (const DB::Exception & e) @@ -665,20 +742,34 @@ TEST(CASBackendGeneration, ConditionalWriteHonoursTheObjectStorageConditionalPut /// the CAS layer receives in the shape production actually produces, which is why a mount that could /// never succeed passed every unit test. These three tests are that crossing. -TEST_F(CASBackendGenerationS3, WriteEmptyGenerationThrows) +TEST_F(CASBackendGenerationS3, WriteEmptyGenerationIsUnresolvedNotThrown) { backend = makeBackend(); - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::CORRUPTED_DATA, - [&] { backend->tokenFromWriteResult("p/gen/no-etag", String{}); }); + client->next_put_etag = ""; + /// The write may well have landed -- an empty response value says nothing about that -- so this is + /// the resolve-by-reading class, not the corrupt-response one (CasRequests::writeLoop): a 2xx + /// carrying a value no grammar accepts is an ambiguity, never an immediate corruption verdict. The + /// mock's resolve GET finds nothing at the key either, so the create-if-absent precondition is + /// still satisfiable and a single-attempt policy reports GaveUp{Unresolved} rather than throwing. + DB::Cas::tests::OperationForTest op(*backend); + const WriteResult result = (*op).create("p/gen/no-etag", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); } -TEST_F(CASBackendGenerationS3, WriteNonNumericGenerationThrows) +TEST_F(CASBackendGenerationS3, WriteNonNumericGenerationIsUnresolvedNotThrown) { backend = makeBackend(); - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::CORRUPTED_DATA, - [&] { backend->tokenFromWriteResult("p/gen/bad-etag", "\"d41d8cd98f00b204e9800998ecf8427e\""); }); + /// An MD5-shaped ETag where a generation belongs: the store answered, but not with an incarnation + /// this dialect can use; see WriteEmptyGenerationIsUnresolvedNotThrown for why this settles as an + /// ambiguity (GaveUp{Unresolved}) rather than a thrown exception. + client->next_put_etag = "\"d41d8cd98f00b204e9800998ecf8427e\""; + DB::Cas::tests::OperationForTest op(*backend); + const WriteResult result = (*op).create("p/gen/bad-etag", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); } /// A mutable conditional write whose response generation arrives quoted -- exactly what @@ -687,8 +778,13 @@ TEST_F(CASBackendGenerationS3, WriteNonNumericGenerationThrows) TEST_F(CASBackendGenerationS3, WriteGenerationTokenStripsTransportQuoting) { backend = makeBackend(); - const Token tok = backend->tokenFromWriteResult("p/gen/quoted-write", "\"1783078552147137\""); - EXPECT_EQ(tok, (Token{"1783078552147137", TokenType::Generation})); + client->next_put_etag = "\"1783078552147137\""; + DB::Cas::tests::OperationForTest op(*backend); + const WriteResult put = (*op).create("p/gen/quoted-write", "v", Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); + const Etag & minted = std::get(put).etag; + EXPECT_EQ(minted.dialect(), Dialect::Generation); + EXPECT_EQ(PersistedEtag::capture(minted).value, "1783078552147137"); } /// The same crossing on the read side: a marked HEAD whose ETag field carries a quoted generation @@ -700,9 +796,11 @@ TEST_F(CASBackendGenerationS3, HeadGenerationTokenStripsTransportQuoting) client->objects["p/gen/quoted-head"] = "body"; client->next_head_etag = "\"1783078552147137\""; - const auto hr = backend->head("p/gen/quoted-head"); - ASSERT_TRUE(hr.exists); - EXPECT_EQ(hr.token, (Token{"1783078552147137", TokenType::Generation})); + DB::Cas::tests::OperationForTest op(*backend); + const auto hr = (*op).head("p/gen/quoted-head", Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_EQ(hr->etag.dialect(), Dialect::Generation); + EXPECT_EQ(PersistedEtag::capture(hr->etag).value, "1783078552147137"); } /// The bound on that stripping. An ETag-dialect token IS the quoted ETag, and the quotes are required @@ -710,13 +808,15 @@ TEST_F(CASBackendGenerationS3, HeadGenerationTokenStripsTransportQuoting) /// This is the test that fails if the quote handling is ever made unconditional. TEST_F(CASBackendGenerationS3, EtagDialectKeepsTransportQuotingVerbatim) { - backend = makeBackend(TokenType::ETag); + backend = makeBackend(Dialect::ETag); client->objects["p/etag/quoted-head"] = "body"; client->next_head_etag = "\"d41d8cd98f00b204e9800998ecf8427e\""; - const auto hr = backend->head("p/etag/quoted-head"); - ASSERT_TRUE(hr.exists); - EXPECT_EQ(hr.token, (Token{"\"d41d8cd98f00b204e9800998ecf8427e\"", TokenType::ETag})); + DB::Cas::tests::OperationForTest op(*backend); + const auto hr = (*op).head("p/etag/quoted-head", Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_EQ(hr->etag.dialect(), Dialect::ETag); + EXPECT_EQ(PersistedEtag::capture(hr->etag).value, "\"d41d8cd98f00b204e9800998ecf8427e\""); } /// A successful HEAD on a generation-dialect backend whose response carries no ETag/generation at all must not mint a token @@ -727,8 +827,9 @@ TEST_F(CASBackendGenerationS3, HeadMissingGenerationThrows) client->objects["p/gen/no-generation-head"] = "body"; /// next_head_etag stays empty: SetETag is never called, so the response carries no ETag field. + DB::Cas::tests::OperationForTest op(*backend); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { backend->head("p/gen/no-generation-head"); }); + [&] { (*op).head("p/gen/no-generation-head", Retry::once()); }); } /// An ordinary AWS-style ETag reaching a generation-dialect backend through a successful HEAD (a proxy dropping @@ -739,8 +840,9 @@ TEST_F(CASBackendGenerationS3, HeadNonNumericGenerationThrows) client->objects["p/gen/bad-etag-head"] = "body"; client->next_head_etag = "\"d41d8cd98f00b204e9800998ecf8427e\""; + DB::Cas::tests::OperationForTest op(*backend); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { backend->head("p/gen/bad-etag-head"); }); + [&] { (*op).head("p/gen/bad-etag-head", Retry::once()); }); } #endif diff --git a/src/Disks/tests/gtest_cas_backend_listing.cpp b/src/Disks/tests/gtest_cas_backend_listing.cpp index 591dfdba65db..31c0c3d1a6e8 100644 --- a/src/Disks/tests/gtest_cas_backend_listing.cpp +++ b/src/Disks/tests/gtest_cas_backend_listing.cpp @@ -2,21 +2,29 @@ #include #include +#include + +#include "cas_test_helpers.h" #include #include using namespace DB::Cas; +using DB::Cas::tests::openRequestsForTest; + TEST(CASBackendListing, ForEachWalksEveryPageOnce) { InMemoryBackend b; + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); for (int i = 0; i < 2500; ++i) - b.putIfAbsent("p/" + std::to_string(1000000 + i), "v"); - b.putIfAbsent("q/other", "v"); /// out of prefix — must not be visited + op.create("p/" + std::to_string(1000000 + i), "v", Retry::once()); + op.create("q/other", "v", Retry::once()); /// out of prefix — must not be visited std::vector seen; - forEachListedKey(b, "p/", [&](const ListedKey & k) { seen.push_back(k.key); }, /*page_limit=*/1000); + op.forEachListedKey("p/", [&](const ListedKey & k) { seen.push_back(k.key); return true; }, + Retry::standard(), /*page_limit=*/1000); EXPECT_EQ(seen.size(), 2500u); /// paged (3 pages), no key dropped/duplicated EXPECT_TRUE(std::is_sorted(seen.begin(), seen.end())); } @@ -24,20 +32,17 @@ TEST(CASBackendListing, ForEachWalksEveryPageOnce) TEST(CASBackendListing, ForEachEmptyPrefixVisitsNothing) { InMemoryBackend b; - b.putIfAbsent("q/other", "v"); + CasRequests requests = openRequestsForTest(b); + CasOperation op = requests.admit(); + op.create("q/other", "v", Retry::once()); size_t visits = 0; - forEachListedKey(b, "p/", [&](const ListedKey &) { ++visits; }); + op.forEachListedKey("p/", [&](const ListedKey &) { ++visits; return true; }, Retry::standard()); EXPECT_EQ(visits, 0u); } -TEST(CASBackendListing, ClassifyMapsEveryDeleteKind) -{ - EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::Deleted, false}), DeleteClass::Deleted); - EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::NotFound, false}), DeleteClass::Absent); - EXPECT_EQ(classifyDeleteOutcome({DeleteOutcome::Kind::TokenMismatch, false}), DeleteClass::Replaced); - - EXPECT_EQ(deleteClassName(DeleteClass::Deleted), "deleted"); - EXPECT_EQ(deleteClassName(DeleteClass::Absent), "absent"); - EXPECT_EQ(deleteClassName(DeleteClass::Replaced), "replaced"); -} +/// `ClassifyMapsEveryDeleteKind` is deleted here: its whole subject was `classifyDeleteOutcome` and +/// `deleteClassName`, free helpers that translated the legacy `DeleteOutcome::Kind` three-value shape +/// into a `DeleteClass`. Both the legacy shape and the helpers are gone -- `CasOperation::remove` +/// already reports its outcome as the four-value `Removal` enum directly, with no separate +/// classification step to pin. diff --git a/src/Disks/tests/gtest_cas_blob_digest.cpp b/src/Disks/tests/gtest_cas_blob_digest.cpp index 9a87f08ccda2..e66ce8239d5e 100644 --- a/src/Disks/tests/gtest_cas_blob_digest.cpp +++ b/src/Disks/tests/gtest_cas_blob_digest.cpp @@ -1,7 +1,7 @@ #include -/// CAS pluggable-blob-hash Phase 2, Task 1: `BlobDigest` (the pool-scoped variable-length content digest, ADDITIVE-ONLY -- no -/// existing `UInt128 blob_hash` field is migrated in this task) + the ONE `PoolMeta`-scoped +/// `BlobDigest` is the pool-scoped variable-length content digest. It is additive: existing +/// `UInt128 blob_hash` fields retain their representation. The `PoolMeta`-scoped /// `DigestCodec` all digest<->hex/bytes conversion must route through. /// /// THE KEY GATE (`ShardOfBitIdenticalToOldHighBitsOver200RandomValues` below): `DigestCodec`'s @@ -73,7 +73,8 @@ TEST(CASBlobDigest, ShardOfViaPoolMetaConstructedCodecMatchesOldBlobShard) { auto backend = std::make_shared(); const Layout layout("p"); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); ASSERT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); const DigestCodec codec = codecFor(BlobHashAlgo::CityHash128); @@ -227,26 +228,29 @@ TEST(CASBlobDigest, PoolMetaRecordsCreatingAlgoAndWidthDerivesFromIt) { auto backend = std::make_shared(); const Layout layout("p1"); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); EXPECT_EQ(blobHashLenFor(BlobHashAlgo::CityHash128), 16u); } { auto backend = std::make_shared(); const Layout layout("p2"); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); EXPECT_EQ(blobHashLenFor(BlobHashAlgo::XXH3_128), 16u); } { auto backend = std::make_shared(); const Layout layout("p3"); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::Sha256)})); EXPECT_EQ(blobHashLenFor(BlobHashAlgo::Sha256), 32u); /// Reopen (decode path) must re-derive the same recorded algo. - const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256); + const PoolMeta reopened = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256); EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::Sha256)})); } } diff --git a/src/Disks/tests/gtest_cas_blob_envelope_format.cpp b/src/Disks/tests/gtest_cas_blob_envelope_format.cpp index 1119a5be0a64..edfa45abf828 100644 --- a/src/Disks/tests/gtest_cas_blob_envelope_format.cpp +++ b/src/Disks/tests/gtest_cas_blob_envelope_format.cpp @@ -1,7 +1,11 @@ #include "cas_format_test_battery.h" #include +#include #include +#include + +#include #include using namespace DB::Cas; @@ -22,6 +26,47 @@ EnvelopeHeader sampleHeader(const String & ref) } constexpr uint32_t L = 256; +/// The `op` word with the most bytes on the wire, found by walking the enum through the REAL +/// public encoder-facing lookup (never by hardcoding "mutation") so a future longer word is +/// automatically picked up by the boundary tests below. +ProvenanceOp longestProvenanceOp() +{ + ProvenanceOp best = ProvenanceOp::Other; + size_t best_len = 0; + for (const auto op : magic_enum::enum_values()) + { + const size_t len = provenanceOpToWireWord(op).size(); + if (len > best_len) + { + best_len = len; + best = op; + } + } + return best; +} + +/// A header whose numeric provenance fields sit at their type maxima (`created_at_ms` at the +/// `uint64_t` max, `ch_version` at the `uint32_t` max, `op` at its longest wire word), so the +/// non-`ref` JSON this produces is the largest `encodeEnvelopeHeader` can emit for real field +/// values. `v` is not settable this way -- `encodeEnvelopeHeader` always stamps +/// `currentCompatibilityVersion()` -- so this is the worst case reachable through the real encoder +/// today, not the type-level bound `kMandatoryDescriptorWorstCase` proves for a hypothetical future +/// `v` at its own `uint32_t` maximum. +EnvelopeHeader maxReachableHeader(const String & ref) +{ + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.incarnation_tag = hexToU128("0102030405060708090a0b0c0d0e0f10"); + h.build_id = hexToU128("1112131415161718191a1b1c1d1e1f20"); + h.provenance = Provenance{ + std::numeric_limits::max(), + hexToU128("2122232425262728292a2b2c2d2e2f30"), + std::numeric_limits::max(), + longestProvenanceOp()}; + h.intended_ref = ref; + return h; +} + /// The envelope has a fixed physical length. At generation 9 there is no unsupported one-digit /// version, so replacing `9` with `10` must consume one byte from the space pad rather than silently /// turning the 256-byte fixture into a different wire shape. @@ -53,8 +98,8 @@ TEST(CASBlobEnvelopeFormat, FixedLengthAndPadZone) EXPECT_EQ(head[L - 1], '\n'); /// terminator at byte 255 const String json = fmt::format(R"({{"type":"cas_blob","v":{},)", currentCompatibilityVersion()) + "\"tag\":\"0102030405060708090a0b0c0d0e0f10\"," - "\"bld\":\"1112131415161718191a1b1c1d1e1f20\",\"ts\":1752537600123," - "\"by\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"ch\":26006001," + "\"build\":\"1112131415161718191a1b1c1d1e1f20\",\"time_ms\":1752537600123," + "\"creator\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"chver\":26006001," "\"ref\":\"t-abc/all_1_2_0\"}"; ASSERT_LT(json.size(), L); EXPECT_EQ(head.substr(0, json.size()), json); /// '/' UNescaped (local escaper) @@ -93,6 +138,131 @@ TEST(CASBlobEnvelopeFormat, RefTruncatedToExactBudget) EXPECT_EQ(c, 'a'); } +TEST(CASBlobEnvelopeFormat, MandatoryWorstCaseBoundary) +{ + /// `kMandatoryDescriptorWorstCase` (239, proven at compile time against the 240 floor) assumes + /// `v` at its OWN type maximum (10 digits), because `currentCompatibilityVersion()` could grow + /// with a future generation. Nothing can make a running build emit that many digits today -- + /// `encodeEnvelopeHeader` always stamps the CURRENT `currentCompatibilityVersion()`, one digit at + /// this generation -- so the worst case reachable through the real encoder right now is 9 bytes + /// smaller: a 10-byte `ref` budget at the floor, not 1. That 9-byte gap is exactly + /// `kMaxU32DecimalLen - digit count of the current compatibility version`, so a generation that + /// reaches two digits narrows it and this expectation must be re-derived then -- the literal below + /// is deliberate, since deriving it from the version width here would restate the formula the + /// compile-time bound already owns and prove nothing about the encoder. + EnvelopeHeader h_floor = maxReachableHeader(""); + const String head_floor = encodeEnvelopeHeader(h_floor, static_cast(kMinBlobHeaderLen)); + ASSERT_EQ(head_floor.size(), kMinBlobHeaderLen); + EXPECT_EQ(head_floor[kMinBlobHeaderLen - 1], '\n'); + EXPECT_EQ(payloadOffset(decodeEnvelopeHeader(head_floor, head_floor.size(), ObjectKind::Blob)), kMinBlobHeaderLen); + const size_t json_len_floor = head_floor.find_last_not_of(' ', kMinBlobHeaderLen - 2) + 1; + const size_t budget_floor = (kMinBlobHeaderLen - 1) - json_len_floor; + EXPECT_EQ(budget_floor, 10u) << "ref budget reachable through the real encoder at the floor"; + + /// The default 256-byte header is exactly 16 bytes above the floor, so the SAME max-reachable + /// content leaves exactly 16 more bytes of `ref` budget. + EnvelopeHeader h_default = maxReachableHeader(""); + const String head_default = encodeEnvelopeHeader(h_default, L); + ASSERT_EQ(head_default.size(), L); + EXPECT_EQ(head_default[L - 1], '\n'); + EXPECT_EQ(payloadOffset(decodeEnvelopeHeader(head_default, head_default.size(), ObjectKind::Blob)), L); + const size_t json_len_default = head_default.find_last_not_of(' ', L - 2) + 1; + const size_t budget_default = (L - 1) - json_len_default; + EXPECT_EQ(budget_default, budget_floor + (L - kMinBlobHeaderLen)) + << "ref budget reachable through the real encoder at the default header length"; +} + +/// The half a `static_assert` cannot do. The compile-time bound proves the FORMULA fits under the +/// floor; it cannot notice a formula that understates the encoder — shrink any component and the +/// assert only grows happier. So this reconstructs the same number from bytes the real encoder +/// produced, and the only quantity it borrows is the version field's type width: +/// +/// what the encoder wrote at max-width values, with an empty ref +/// + the digits the version field did NOT use at this generation +/// == the mandatory worst case +/// +/// Every other field in the fixture is already at its type maximum, so nothing else is missing from +/// the measured side. An understated key cost, or a shrunken `kMaxU32DecimalLen`, moves the formula +/// without moving the encoder and lands here. +TEST(CASBlobEnvelopeFormat, WorstCaseFormulaMatchesTheEncoder) +{ + /// Drive the encoder at the WIDEST version the budget reserves room for, rather than encoding at + /// today's one-digit version and adding the missing digits by arithmetic. Doing the arithmetic + /// here would re-derive the very formula this test exists to check, and would never send the + /// ten-digit boundary through the encoder's own number formatting. + EnvelopeHeader h = maxReachableHeader(""); + const String head = encodeEnvelopeHeader(h, static_cast(kMinBlobHeaderLen), + std::numeric_limits::max()); + + /// The mandatory shape is everything up to and including the closing brace, plus the newline the + /// encoder reserves at the last byte; the padding between them is the ref budget this measures. + const size_t json_len = head.find_last_not_of(' ', kMinBlobHeaderLen - 2) + 1; + const size_t mandatory_at_max_version = json_len + 1; /// + the reserved '\n' + + EXPECT_EQ(mandatory_at_max_version, mandatory_descriptor_worst_case) + << "the formula and the encoder disagree about the mandatory descriptor at the widest " + "version: encoder wrote " << mandatory_at_max_version << " bytes, formula says " + << mandatory_descriptor_worst_case; + + /// And the whole point of the budget: even at that width one byte remains spare under the floor. + EXPECT_LE(mandatory_descriptor_worst_case, kMinBlobHeaderLen - 1); +} + +/// The version really is rendered at its full width by the encoder above, not merely accounted for. +/// Without this, an encoder that silently clamped or dropped the override would still satisfy the +/// equality it feeds. +TEST(CASBlobEnvelopeFormat, MaxWidthVersionIsActuallyRendered) +{ + EnvelopeHeader h = maxReachableHeader(""); + const String head = encodeEnvelopeHeader(h, static_cast(kMinBlobHeaderLen), + std::numeric_limits::max()); + EXPECT_NE(head.find("\"v\":4294967295"), String::npos) + << "the max-width version was not rendered; the boundary above proves nothing. Header: " + << head; +} + +TEST(CASBlobEnvelopeFormat, CriticalKeyDescriptorStillFitsAtDefaultLength) +{ + /// The test-only `!x` critical key is written BEFORE `ref`; even at max-reachable field values + /// the descriptor still fits the default 256-byte header and fails closed as + /// UNKNOWN_FORMAT_VERSION, never CORRUPTED_DATA or a LOGICAL_ERROR from encode itself. + EnvelopeHeader h = maxReachableHeader("r"); + h.emit_unknown_critical_key = true; + const String head = encodeEnvelopeHeader(h, L); + ASSERT_EQ(head.size(), L); + cas_battery_detail::expectCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, + [&] { decodeEnvelopeHeader(head, head.size(), ObjectKind::Blob); }, + "critical-key blob envelope at max-reachable field values"); +} + +/// Closed-set pin: the six `ProvenanceOp` words, walked through `magic_enum::enum_values`, which is what proves the +/// renderer and the parser consult the SAME table: a table entry missing altogether is already a +/// build error at the coverage assert, but two delegates drifting onto different tables is not. +TEST(CASBlobEnvelopeFormat, ClosedSetPinsProvenanceOpWords) +{ + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Other), "other"); + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Insert), "insert"); + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Merge), "merge"); + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Mutation), "mutation"); + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Attach), "attach"); + EXPECT_EQ(provenanceOpToWireWord(ProvenanceOp::Repack), "repack"); + for (const auto op : magic_enum::enum_values()) + EXPECT_EQ(provenanceOpFromWireWord(provenanceOpToWireWord(op)), op); +} + +TEST(CASBlobEnvelopeFormat, UnknownOpWordFailsClosed) +{ + /// `op` is written as a plain (non-critical) key, so an unrecognized word is a decode-time + /// vocabulary violation, not a missing-extension one: CORRUPTED_DATA, not UNKNOWN_FORMAT_VERSION. + EnvelopeHeader h = sampleHeader("r"); + String head = encodeEnvelopeHeader(h, L); + const size_t op_at = head.find("\"op\":\"merge\""); + ASSERT_NE(op_at, String::npos); + head.replace(op_at, String("\"op\":\"merge\"").size(), "\"op\":\"bogus\""); + cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeEnvelopeHeader(head, head.size(), ObjectKind::Blob); }, "unknown op word"); +} + TEST(CASBlobEnvelopeFormat, PadZoneSmugglingFailsClosed) { EnvelopeHeader h = sampleHeader("r"); @@ -154,6 +324,8 @@ TEST(CASBlobEnvelopeFormat, RefEscaperAlphabetPinned) << "escaper alphabet drifted: '/' must be verbatim, quote/backslash escaped, control -> \\uXXXX"; } +CAS_BATTERY_COVERS(Blob); + TEST(CASFormatBattery, BlobEnvelope) { /// The golden is CONSTRUCTED from the hand-pinned json literal (same one FixedLengthAndPadZone @@ -161,8 +333,8 @@ TEST(CASFormatBattery, BlobEnvelope) /// the encoder to itself and pin nothing. const String json = fmt::format(R"({{"type":"cas_blob","v":{},)", currentCompatibilityVersion()) + "\"tag\":\"0102030405060708090a0b0c0d0e0f10\"," - "\"bld\":\"1112131415161718191a1b1c1d1e1f20\",\"ts\":1752537600123," - "\"by\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"ch\":26006001," + "\"build\":\"1112131415161718191a1b1c1d1e1f20\",\"time_ms\":1752537600123," + "\"creator\":\"2122232425262728292a2b2c2d2e2f30\",\"op\":\"merge\",\"chver\":26006001," "\"ref\":\"t-abc/all_1_2_0\"}"; const String golden = json + String((L - 1) - json.size(), ' ') + '\n'; runFormatBattery(FormatBatteryCase{ diff --git a/src/Disks/tests/gtest_cas_blob_indegree.cpp b/src/Disks/tests/gtest_cas_blob_indegree.cpp index e47b7351f063..d1eb99b05cc2 100644 --- a/src/Disks/tests/gtest_cas_blob_indegree.cpp +++ b/src/Disks/tests/gtest_cas_blob_indegree.cpp @@ -5,11 +5,16 @@ #include #include #include +#include +#include "config.h" +#if USE_AWS_S3 +#include +#endif #include #include #include -namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int NOT_IMPLEMENTED; } +namespace DB::ErrorCodes { extern const int ABORTED; extern const int CORRUPTED_DATA; extern const int NOT_IMPLEMENTED; } using namespace DB::Cas; @@ -21,21 +26,18 @@ UInt128 s(uint64_t n) { return UInt128(n); } // source-edge id /// `BlobCandidate.ref` / `inDegreeInRuns` argument is a `BlobRef` as of Phase 3 T3. BlobRef bh(uint64_t n) { return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(n))}; } -/// Scale thresholds for the "the run genuinely spans several blocks" sanity assertions below. These are -/// NOT format constants — the SourceEdge run is a plain NDJSON stream (`CasRecordStreamFormat`) with no -/// block framing of its own — they only pin the same byte-size scale the (now-deleted, codecs-v3 phase 6) +/// Scale threshold for the "the run genuinely spans several blocks" sanity assertions below. This is +/// NOT a format constant — the SourceEdge run is a plain NDJSON stream (`CasRecordStreamFormat`) with no +/// block framing of its own — it only pins the same byte-size scale the (now-deleted, codecs-v3 phase 6) /// `CasRunFile` block codec used, so the multi-block-sized fixtures below stay meaningfully large. -/// (Previously read straight off `CasRunFile.h`'s own `kRunTargetBlockSize`/`kRunHardCapBlockSize`; this -/// file's `#include` of that header looked removable when `CasRunFile` was deleted in the phase-6 cutover, -/// but these two thresholds turned out to be the only remaining users — hence the local, explicitly-legacy -/// copies here instead of a dangling include. Values unchanged.) constexpr uint32_t kLegacyBlockSize = 256u * 1024u; -constexpr uint32_t kLegacyHardCapBlockSize = 1024u * 1024u; + } TEST(CASBlobInDegree, FoldStartsFromEmptyPriorGeneration) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Generation 1 from empty prior: two distinct edges on b1 and one on b2. @@ -46,29 +48,30 @@ TEST(CASBlobInDegree, FoldStartsFromEmptyPriorGeneration) {bh(2), s(1), false}, }; std::vector runs; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new*/1, /*attempt*/0, /*shard*/0, deltas, runs); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, /*new*/1, /*attempt*/0, /*shard*/0, deltas, runs); ASSERT_FALSE(runs.empty()); - const auto zero = zeroInDegree(backend, runs); + const auto zero = zeroInDegree(*backend_req, runs); EXPECT_TRUE(zero.empty()); /// nothing at zero yet } TEST(CASBlobInDegree, PlusMinusCancelToZeroDetectsCandidate) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Gen 1: activate edge (b1,s1) and (b2,s1). std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, {{bh(1), s(1), false}, {bh(2), s(1), false}}, runs1); /// Generation 2 merges prior gen-1 run (resolved via runs1 refs) with removal of (b1,s1): indeg(b1)=0, indeg(b2)=1. std::vector runs2; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new*/2, /*attempt*/0, 0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, /*new*/2, /*attempt*/0, 0, {{bh(1), s(1), true}}, runs2); - const auto zero = zeroInDegree(backend, runs2); + const auto zero = zeroInDegree(*backend_req, runs2); ASSERT_EQ(zero.size(), 1u); EXPECT_EQ(zero[0].ref, bh(1)); } @@ -76,17 +79,19 @@ TEST(CASBlobInDegree, PlusMinusCancelToZeroDetectsCandidate) TEST(CASBlobInDegree, RunsAreByteDeterministic) { InMemoryBackend a; + DB::Cas::tests::OperationForTest a_req(a); InMemoryBackend b2; + DB::Cas::tests::OperationForTest b2_req(b2); Layout layout{"pool"}; std::vector ra; std::vector rb; /// Same deltas in a DIFFERENT input order must produce the same sealed run bytes (sorted by key). - foldDeltasIntoGeneration(a, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + foldDeltasIntoGeneration(*a_req, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, {{bh(3), s(1), false}, {bh(1), s(1), false}, {bh(2), s(1), false}}, ra); - foldDeltasIntoGeneration(b2, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + foldDeltasIntoGeneration(*b2_req, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(3), s(1), false}}, rb); - const auto ga = a.get(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0)); - const auto gb = b2.get(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0)); + const auto ga = (*a_req).read(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0), Retry::standard()); + const auto gb = (*b2_req).read(layout.blobTargetRunKey(1, /*attempt*/0, 0, 0), Retry::standard()); ASSERT_TRUE(ga.has_value()); ASSERT_TRUE(gb.has_value()); EXPECT_EQ(ga->bytes, gb->bytes); @@ -101,82 +106,127 @@ TEST(CASBlobInDegree, SameEdgeActivatedTwiceCountsOnce) /// The source-edge set is a SET, not a counter — re-adding the same edge is a no-op. /// indeg(b1) must be 1 after both activations, not 2. InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; std::vector deltas{ {bh(1), s(1), false}, // activate (b1,s1) {bh(1), s(1), false}, // same edge again — must deduplicate }; std::vector runs; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, deltas, runs); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, deltas, runs); ASSERT_FALSE(runs.empty()); const int64_t deg = DB::Cas::tests::inDegreeInRuns(backend, runs, bh(1)); EXPECT_EQ(deg, 1); /// deduplicated, not 2 - const auto zero = zeroInDegree(backend, runs); + const auto zero = zeroInDegree(*backend_req, runs); EXPECT_TRUE(zero.empty()); /// b1 still has an active edge } TEST(CASBlobInDegree, FoldDeltaByteEqualReplayAdopts) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; std::vector deltas{{bh(1), s(1), false}}; std::vector runs1; std::vector runs2; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs1); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs1); /// Same inputs, same attempt => byte-identical run already present => adopt, no throw. - EXPECT_NO_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs2)); + EXPECT_NO_THROW(foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs2)); EXPECT_EQ(runs1, runs2); } TEST(CASBlobInDegree, FoldDeltaDivergentBytesThrowsCorrupted) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Pre-occupy the run key (attempt 7) with junk, then fold => divergent => CORRUPTED_DATA. - backend.putIfAbsent(layout.blobTargetRunKey(1, /*attempt*/7, /*shard*/0, /*seq*/0), "not-a-valid-run"); + (*backend_req).create(layout.blobTargetRunKey(1, /*attempt*/7, /*shard*/0, /*seq*/0), "not-a-valid-run", Retry::once()); std::vector deltas{{bh(1), s(1), false}}; std::vector runs; - EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs), - DB::Exception); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/7, /*shard*/0, deltas, runs); }); +} + +#if USE_AWS_S3 +namespace +{ +/// Refuses to serve one key's body. An access denial gets one credential refresh first; it surfaces on +/// the first attempt only because `InMemoryBackend::refreshCredentials` answers false by default, so the +/// write's resolve read ends having observed nothing. +class ReadRefusingBackend : public InMemoryBackend +{ +public: + std::optional read(const String & key, TransportAccess & access) override + { + if (key == refuse_key) + throw DB::S3Exception("injected access denial on the resolve read", Aws::S3::S3Errors::ACCESS_DENIED); + return InMemoryBackend::read(key, access); + } + + String refuse_key; +}; } +/// A refused write whose resolve read observed NOTHING says nothing about what is at the key, so it +/// must not be reported as pool corruption. `CORRUPTED_DATA` is a deterministic local failure: no +/// caller above reissues it, so a permission or credential blip during the resolve would wedge every +/// later round on the same artifact. The companion arm is `FoldDeltaDivergentBytesThrowsCorrupted`, +/// where the read DID observe divergent bytes and corruption is the right verdict. +TEST(CASBlobInDegree, DeterministicArtifactWhoseResolveReadObservedNothingIsNotCorruption) +{ + ReadRefusingBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); + Layout layout{"pool"}; + const String key = layout.blobTargetRunKey(1, /*attempt*/0, /*shard*/0, /*seq*/0); + + /// Occupy the key so the create's precondition is refused, THEN arm the refusal, so the failure + /// falls on the resolve read rather than on the setup. + ASSERT_TRUE(std::holds_alternative((*backend_req).create(key, "someone else's bytes", Retry::once()))); + backend.refuse_key = key; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::ABORTED, + [&] { putDeterministicArtifact(*backend_req, key, "our deterministic bytes"); }); +} +#endif + /// ==== two-cursor settlement merge (retired-in-snapshot T3, spec §2.1/§3) ==== /// -/// The retired input is no longer a separate `prior_retired` vector — the prior generation's `kCondemned` +/// The retired input is no longer a separate `prior_retired` vector — the prior generation's `RunMarker::Condemned` /// rows RIDE the source-edge run at the zero-sentinel key. These helpers build such a prior run directly /// (via the sorted-NDJSON `SourceEdgeRunWriter`, codecs-v3 phase 5) and decode a run for assertions. namespace { -/// A `kCondemned` sentinel record for `h` at the zero source_id, carrying the condemned incarnation. +/// A `RunMarker::Condemned` sentinel record for `h` at the zero source_id, carrying the condemned incarnation. SourceEdgeRecord condemnedRec(UInt128 h, const CondemnedRow & row) { return SourceEdgeRecord{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}, - .source_id = UInt128{0}, .marker = kCondemned, + .source_id = UInt128{0}, .marker = RunMarker::Condemned, .delete_pending = row.delete_pending, .token = row.token, .size = row.size, .condemn_round = row.condemn_round}; } -/// An active-edge record (`kEdgeActive`) for `h` at source `sid`. +/// An active-edge record (`RunMarker::Edge`) for `h` at source `sid`. SourceEdgeRecord edgeRec(UInt128 h, UInt128 sid) { return SourceEdgeRecord{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}, - .source_id = sid, .marker = kEdgeActive}; + .source_id = sid, .marker = RunMarker::Edge}; } -/// head_blob / peek_head stub: present with a fixed token/size. -std::function(const BlobRef &)> headPresent(const String & tok, uint64_t size) +/// `head_blob` / `peek_head` stub. Only the request engine mints an incarnation, so the stub cannot +/// fabricate one: it writes `size` bytes at the blob's own key and hands back the head of what it +/// wrote, which is the incarnation the fold then condemns. +BlobHeadFn headPresent(CasOperation & op, const Layout & layout, uint64_t size) { - return [tok, size](const BlobRef &) -> std::optional + return [&op, &layout, size](const BlobRef & ref) -> std::optional { - HeadResult hr; - hr.exists = true; - hr.size = size; - hr.token = Token{.value = tok, .type = TokenType::Emulated}; - return hr; + const String key = layout.blobKey(ref); + op.create(key, String(size, 'x'), Retry::standard()); + return op.head(key, Retry::standard()); }; } @@ -185,11 +235,11 @@ CondemnedRow condemnedRowFor(uint64_t condemn_round, const String & tok = "t", bool delete_pending = false, uint64_t size = 1) { return CondemnedRow{.delete_pending = delete_pending, - .token = Token{.value = tok, .type = TokenType::Emulated}, + .token = PersistedEtag{"emulated", tok}, .size = size, .condemn_round = condemn_round}; } -/// Build a source-edge run (`kSourceEdgeKeySchema128`) carrying the given `kCondemned` sentinel rows +/// Build a source-edge run (`kSourceEdgeKeySchema128`) carrying the given `RunMarker::Condemned` sentinel rows /// and surviving edges, write it under `blobTargetRunKey(gen, attempt, shard, 0)`, and return its /// `RunRef`. Rows are emitted in (blob_hash, source_id) order (sentinels at source_id 0 sort first /// per blob). @@ -223,8 +273,9 @@ RunRef writeSourceEdgeRun(InMemoryBackend & backend, const Layout & layout, const String bytes = out.str(); const String key = layout.blobTargetRunKey(gen, attempt, shard, 0); - backend.putIfAbsent(key, bytes); - return RunRef{.key = key, .checksum = sourceEdgeRunChecksum(bytes), .shard = shard, .generation = gen}; + DB::Cas::tests::OperationForTest op(backend); + (*op).create(key, bytes, Retry::once()); + return RunRef{.key = key, .checksum = sourceEdgeRunChecksum(bytes), .shard = shard, .key_generation = gen}; } struct DecodedRun @@ -234,10 +285,10 @@ struct DecodedRun std::vector> edges; /// (blob_hash, source_id) }; -DecodedRun decodeRun(InMemoryBackend & backend, const RunRef & run) +DecodedRun decodeRun(CasOperation & op, const RunRef & run) { DecodedRun d; - auto r = openSourceEdgeRun(backend, run.key); + auto r = openSourceEdgeRun(op, run.key); /// Every run this test helper decodes is CityHash128 (16-byte), so `.toU128()` is a /// provably-exact round trip. String k; @@ -251,11 +302,11 @@ DecodedRun decodeRun(InMemoryBackend & backend, const RunRef & run) EXPECT_FALSE(p.empty()); if (p.empty()) continue; - if (p[0] == kCondemned) + if (runMarkerFromByte(p[0], "CAS test source-edge run") == RunMarker::Condemned) d.condemned.emplace_back(bh, decodeCondemnedRow(p)); - else if (p[0] == kZeroMarker) + else if (runMarkerFromByte(p[0], "CAS test source-edge run") == RunMarker::Zero) d.zero_markers.push_back(bh); - else if (p[0] == kEdgeActive) + else if (runMarkerFromByte(p[0], "CAS test source-edge run") == RunMarker::Edge) d.edges.emplace_back(bh, sid); else ADD_FAILURE() << "unknown run row type"; @@ -272,6 +323,7 @@ DecodedRun decodeRun(InMemoryBackend & backend, const RunRef & run) TEST(CASBlobInDegree, FoldSealChecksumMismatchFailsClosed) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; const RunRef good = writeSourceEdgeRun(backend, layout, /*gen*/1, /*attempt*/0, /*shard*/0, /*condemned*/{}, /*edges*/{{b(1), s(1)}}); @@ -282,7 +334,7 @@ TEST(CASBlobInDegree, FoldSealChecksumMismatchFailsClosed) /// A delta on a DIFFERENT blob forces the two-cursor merge to stream the prior run to completion, so /// the end-of-segment verifyAgainst fires (not a row-invariant abort). EXPECT_THROW( - foldDeltasIntoGeneration(backend, layout, prior, /*new*/2, /*attempt*/0, /*shard*/0, + foldDeltasIntoGeneration(*backend_req, layout, prior, /*new*/2, /*attempt*/0, /*shard*/0, std::vector{{bh(2), s(1), false}}, out), DB::Exception); } @@ -290,21 +342,23 @@ TEST(CASBlobInDegree, FoldSealChecksumMismatchFailsClosed) TEST(CASBlobInDegree, ZeroInDegreeSealChecksumMismatchFailsClosed) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; const RunRef good = writeSourceEdgeRun(backend, layout, /*gen*/1, /*attempt*/0, /*shard*/0, /*condemned*/{}, /*edges*/{{b(1), s(1)}}); RunRef bad = good; bad.checksum = good.checksum + 1; std::vector runs{bad}; - EXPECT_THROW(zeroInDegree(backend, runs), DB::Exception); + EXPECT_THROW(zeroInDegree(*backend_req, runs), DB::Exception); } TEST(CASThreeCursorMerge, FloorBoundary) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; - /// Gen 1's run holds one unrelated surviving edge (b9) plus the carried kCondemned rows for A=b1 + /// Gen 1's run holds one unrelated surviving edge (b9) plus the carried RunMarker::Condemned rows for A=b1 /// (condemned round 2) and B=b2 (round 3); neither A nor B has any edge (in-degree 0 by definition). /// current_round = 3: strictly-below graduates, at-the-current-round stays. const RunRef gen1 = writeSourceEdgeRun(backend, layout, /*gen*/1, 0, 0, @@ -312,7 +366,7 @@ TEST(CASThreeCursorMerge, FloorBoundary) std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, /*current_round*/3, /*condemn_round*/4, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); /// Two-phase graduation: the floor-passed entry is REPUBLISHED pending (still in the list); @@ -329,8 +383,8 @@ TEST(CASThreeCursorMerge, FloorBoundary) EXPECT_TRUE(rmr.spared.empty()); EXPECT_TRUE(rmr.redelete.empty()); - /// still_retired mirrors exactly the kCondemned rows written into the output run, in order. - const DecodedRun out = decodeRun(backend, runs2[0]); + /// still_retired mirrors exactly the RunMarker::Condemned rows written into the output run, in order. + const DecodedRun out = decodeRun(*backend_req, runs2[0]); ASSERT_EQ(out.condemned.size(), 2u); EXPECT_EQ(out.condemned[0].first, b(1)); EXPECT_TRUE(out.condemned[0].second.delete_pending); @@ -342,6 +396,7 @@ TEST(CASThreeCursorMerge, FloorBoundary) TEST(CASThreeCursorMerge, PendingRedeletesAndDrops) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// A row the PRIOR pass published as delete_pending (carried on gen 1's run): this pass hands it to @@ -351,7 +406,7 @@ TEST(CASThreeCursorMerge, PendingRedeletesAndDrops) std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, /*current_round*/9, /*condemn_round*/9, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); ASSERT_EQ(rmr.redelete.size(), 1u); @@ -361,7 +416,7 @@ TEST(CASThreeCursorMerge, PendingRedeletesAndDrops) EXPECT_TRUE(rmr.spared.empty()); /// The redeleted blob leaves the run entirely (no sentinel carried, no zero marker — untouched). - const DecodedRun out = decodeRun(backend, runs2[0]); + const DecodedRun out = decodeRun(*backend_req, runs2[0]); EXPECT_TRUE(out.condemned.empty()); EXPECT_TRUE(out.zero_markers.empty()); } @@ -369,6 +424,7 @@ TEST(CASThreeCursorMerge, PendingRedeletesAndDrops) TEST(CASThreeCursorMerge, RecoverySpares) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// A (=b1) is retired at round 1 and would long since have graduated (current_round = 5) — but this @@ -377,7 +433,7 @@ TEST(CASThreeCursorMerge, RecoverySpares) std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {{bh(1), s(1), false}}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{gen1}, 2, 0, 0, {{bh(1), s(1), false}}, runs2, /*current_round*/5, /*condemn_round*/6, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); ASSERT_EQ(rmr.spared.size(), 1u); @@ -386,7 +442,7 @@ TEST(CASThreeCursorMerge, RecoverySpares) EXPECT_TRUE(rmr.still_retired.empty()); /// b1 recovered its edge: the output run carries the surviving edge and no sentinel for it. - const DecodedRun out = decodeRun(backend, runs2[0]); + const DecodedRun out = decodeRun(*backend_req, runs2[0]); EXPECT_TRUE(out.condemned.empty()); ASSERT_EQ(out.edges.size(), 1u); EXPECT_EQ(out.edges[0].first, b(1)); @@ -395,55 +451,59 @@ TEST(CASThreeCursorMerge, RecoverySpares) TEST(CASThreeCursorMerge, NewCandidateCondemned) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Gen 1: C (=b3) has one edge. Gen 2 removes it => transition to zero, not retired => /// condemned with the head-captured token at THIS pass's condemn_round. std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, - /*current_round*/0, /*condemn_round*/7, headPresent("t9", 42), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, + /*current_round*/0, /*condemn_round*/7, headPresent(*backend_req, layout, 42), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); ASSERT_EQ(rmr.still_retired.size(), 1u); EXPECT_EQ(rmr.still_retired[0].ref, bh(3)); - EXPECT_EQ(rmr.still_retired[0].token.value, "t9"); + const std::optional present = (*backend_req).head(layout.blobKey(bh(3)), Retry::standard()); + ASSERT_TRUE(present.has_value()); + EXPECT_TRUE(rmr.still_retired[0].token.matches(present->etag)); EXPECT_EQ(rmr.still_retired[0].size, 42u); EXPECT_EQ(rmr.still_retired[0].condemn_round, 7u); EXPECT_TRUE(rmr.graduated.empty()); EXPECT_TRUE(rmr.spared.empty()); - /// The fresh condemn is emitted as a kCondemned row (not a zero marker) into the output run. - const DecodedRun out = decodeRun(backend, runs2[0]); + /// The fresh condemn is emitted as a RunMarker::Condemned row (not a zero marker) into the output run. + const DecodedRun out = decodeRun(*backend_req, runs2[0]); ASSERT_EQ(out.condemned.size(), 1u); EXPECT_EQ(out.condemned[0].first, b(3)); - EXPECT_EQ(out.condemned[0].second.token.value, "t9"); + EXPECT_TRUE(out.condemned[0].second.token.matches(present->etag)); EXPECT_TRUE(out.zero_markers.empty()); } TEST(CASThreeCursorMerge, AbsentBlobNotCondemned) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Same transition-to-zero as above, but the blob object is already gone at condemn time: /// nothing to delete later, so no entry is minted — a plain zero marker is emitted instead. std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(3), s(1), false}}, runs1); std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, 2, 0, 0, {{bh(3), s(1), true}}, runs2, /*current_round*/0, /*condemn_round*/7, - [](const BlobRef &) -> std::optional { return std::nullopt; }, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + [](const BlobRef &) -> std::optional { return std::nullopt; }, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); EXPECT_TRUE(rmr.still_retired.empty()); EXPECT_TRUE(rmr.graduated.empty()); EXPECT_TRUE(rmr.spared.empty()); - const DecodedRun out = decodeRun(backend, runs2[0]); + const DecodedRun out = decodeRun(*backend_req, runs2[0]); EXPECT_TRUE(out.condemned.empty()); ASSERT_EQ(out.zero_markers.size(), 1u); EXPECT_EQ(out.zero_markers[0], b(3)); @@ -451,16 +511,18 @@ TEST(CASThreeCursorMerge, AbsentBlobNotCondemned) TEST(CASThreeCursorMerge, SnapshotEdgesUnperturbedByRetired) { - /// Retired-in-snapshot changes the byte-invariant: the retired machinery now WRITES kCondemned + /// Retired-in-snapshot changes the byte-invariant: the retired machinery now WRITES RunMarker::Condemned /// sentinel rows into the run, so a retired-engaged run is no longer byte-identical to a plain one. /// The preserved invariant (spec §2.1) is narrower: the retired machinery touches ONLY the sentinel /// namespace — the surviving EDGE rows are byte-identical to a plain fold of the same deltas. InMemoryBackend plain; + DB::Cas::tests::OperationForTest plain_req(plain); InMemoryBackend engaged; + DB::Cas::tests::OperationForTest engaged_req(engaged); Layout layout{"pool"}; std::vector r1; - foldDeltasIntoGeneration(plain, layout, /*prior_runs*/{}, 1, 0, 0, + foldDeltasIntoGeneration(*plain_req, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(2), s(2), true}}, r1); /// Engaged: the SAME deltas, but the prior run carries retired rows for b1 (which the delta re-edges @@ -469,12 +531,12 @@ TEST(CASThreeCursorMerge, SnapshotEdgesUnperturbedByRetired) {{b(1), condemnedRowFor(1)}, {b(5), condemnedRowFor(2)}}); std::vector r2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(engaged, layout, /*prior_runs*/{prior}, 2, 0, 0, + foldDeltasIntoGeneration(*engaged_req, layout, /*prior_runs*/{prior}, 2, 0, 0, {{bh(1), s(1), false}, {bh(2), s(1), false}, {bh(2), s(2), true}}, r2, - /*current_round*/9, /*condemn_round*/3, headPresent("t", 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + /*current_round*/9, /*condemn_round*/3, headPresent(*engaged_req, layout, 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); - const DecodedRun plain_run = decodeRun(plain, r1[0]); - const DecodedRun engaged_run = decodeRun(engaged, r2[0]); + const DecodedRun plain_run = decodeRun(*plain_req, r1[0]); + const DecodedRun engaged_run = decodeRun(*engaged_req, r2[0]); EXPECT_EQ(plain_run.edges, engaged_run.edges); /// edge rows byte-identical EXPECT_TRUE(plain_run.condemned.empty()); /// The engaged run carries only the retired sentinel(s) on top: b1 spared (no row), b5 graduated. @@ -485,32 +547,33 @@ TEST(CASThreeCursorMerge, SnapshotEdgesUnperturbedByRetired) TEST(CASTwoCursorMerge, CarriedSentinelIsNotATouch) { - /// Gen 1 condemns b (a real +edge/-edge net-to-zero with head_blob present) -> a kCondemned row. Gen 2 + /// Gen 1 condemns b (a real +edge/-edge net-to-zero with head_blob present) -> a RunMarker::Condemned row. Gen 2 /// has NO deltas at all: the carried row must (a) survive byte-identically, (b) emit no zero marker, /// (c) never call peek_head (a carried sentinel is not a touch). InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Gen 1: (b,s1) added then removed => net-to-zero => fresh condemn at round 5 (token "tok", size 7). std::vector runs1; RetiredMergeResult rmr1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, 0, 0, {{bh(2), s(1), false}, {bh(2), s(1), true}}, runs1, - /*current_round*/0, /*condemn_round*/5, headPresent("tok", 7), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr1); + /*current_round*/0, /*condemn_round*/5, headPresent(*backend_req, layout, 7), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr1); ASSERT_EQ(rmr1.still_retired.size(), 1u); { - const DecodedRun g1 = decodeRun(backend, runs1[0]); + const DecodedRun g1 = decodeRun(*backend_req, runs1[0]); ASSERT_EQ(g1.condemned.size(), 1u); EXPECT_EQ(g1.condemned[0].first, b(2)); - EXPECT_TRUE(g1.zero_markers.empty()); /// a condemned blob emits kCondemned, never a zero marker + EXPECT_TRUE(g1.zero_markers.empty()); /// a condemned blob emits RunMarker::Condemned, never a zero marker } /// Gen 2: empty deltas, current_round 1 (< 5 => b carries, does not graduate). peek_head must NOT fire. size_t peek_calls = 0; - auto peek = [&](const BlobRef &) -> std::optional { ++peek_calls; return {}; }; + auto peek = [&](const BlobRef &) -> std::optional { ++peek_calls; return {}; }; std::vector runs2; RetiredMergeResult rmr2; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, 0, 0, {}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, 2, 0, 0, {}, runs2, /*current_round*/1, /*condemn_round*/6, /*head_blob*/{}, peek, /*confirm_condemned_marker*/{}, &rmr2); EXPECT_EQ(peek_calls, 0u); @@ -519,10 +582,12 @@ TEST(CASTwoCursorMerge, CarriedSentinelIsNotATouch) EXPECT_EQ(rmr2.still_retired[0].condemn_round, 5u); /// carried unchanged EXPECT_TRUE(rmr2.graduated.empty()); - const DecodedRun g2 = decodeRun(backend, runs2[0]); + const DecodedRun g2 = decodeRun(*backend_req, runs2[0]); ASSERT_EQ(g2.condemned.size(), 1u); EXPECT_EQ(g2.condemned[0].first, b(2)); - EXPECT_EQ(g2.condemned[0].second.token.value, "tok"); + const std::optional present = (*backend_req).head(layout.blobKey(bh(2)), Retry::standard()); + ASSERT_TRUE(present.has_value()); + EXPECT_TRUE(g2.condemned[0].second.token.matches(present->etag)); EXPECT_EQ(g2.condemned[0].second.size, 7u); EXPECT_TRUE(g2.zero_markers.empty()); } @@ -534,6 +599,7 @@ TEST(CASTwoCursorMerge, MalformedRunFailsClosed) /// (1) An active edge at the reserved sentinel source_id 0 -> the merge cursor fails closed. { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); DB::WriteBufferFromOwnString out; SourceEdgeRunWriter writer(out); writer.append(edgeRec(1, UInt128{0})); // edge at sentinel key @@ -541,17 +607,18 @@ TEST(CASTwoCursorMerge, MalformedRunFailsClosed) out.finalize(); const String bytes = out.str(); const RunRef bad{.key = layout.blobTargetRunKey(1, 0, 0, 0), - .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .generation = 1}; - backend.putIfAbsent(bad.key, bytes); + .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .key_generation = 1}; + (*backend_req).create(bad.key, bytes, Retry::once()); std::vector runs2; - EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), + EXPECT_THROW(foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), DB::Exception); } /// (2) Two sentinel rows for one blob -> duplicate sentinel -> the merge cursor fails closed. { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); DB::WriteBufferFromOwnString out; SourceEdgeRunWriter writer(out); /// Same (b,0) key twice (equal keys are allowed by the writer) — two condemned sentinels for b1. @@ -561,26 +628,30 @@ TEST(CASTwoCursorMerge, MalformedRunFailsClosed) out.finalize(); const String bytes = out.str(); const RunRef bad{.key = layout.blobTargetRunKey(1, 0, 0, 0), - .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .generation = 1}; - backend.putIfAbsent(bad.key, bytes); + .checksum = sourceEdgeRunChecksum(bytes), .shard = 0, .key_generation = 1}; + (*backend_req).create(bad.key, bytes, Retry::once()); std::vector runs2; - EXPECT_THROW(foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), + EXPECT_THROW(foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{bad}, 2, 0, 0, {}, runs2), DB::Exception); } } -/// A prior run spanning several blocks folds correctly with the streaming prior cursor AND the backend -/// sees only block-bounded ranged/stream requests for it — never a whole-object get of the prior run -/// key. Byte-reproducibility of the merged output is the load-bearing canary (the merge logic is -/// unchanged; only the prior cursor's byte source moved from materialize-whole to stream). -TEST(CASBlobInDegree, FoldStreamsPriorRunBlockBounded) +/// A prior run several times larger than one buffer folds correctly with the streaming prior cursor +/// AND the backend is never asked to READ that run's key — the cursor reaches it only through the +/// streaming open. Byte-reproducibility of the merged output is the load-bearing canary (the merge +/// logic is unchanged; only the prior cursor's byte source moved from materialize-whole to stream). +/// What this does NOT check is how much the open stream buffers: the streaming primitive carries no +/// window, so the seam has nothing to measure. +TEST(CASBlobInDegree, FoldStreamsPriorRunWithoutReadingItWhole) { using DB::Cas::tests::CountingBackend; CountingBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); /// InMemory oracle: the SAME two folds against a plain backend must yield byte-identical runs — /// the streaming cursor changes I/O shape, not bytes. InMemoryBackend oracle; + DB::Cas::tests::OperationForTest oracle_req(oracle); Layout layout{"pool"}; /// Gen 1 from empty prior: enough edges that the SourceEdge run spills across many 256KB blocks. @@ -593,31 +664,33 @@ TEST(CASBlobInDegree, FoldStreamsPriorRunBlockBounded) std::vector runs1_c; std::vector runs1_o; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); - foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); + foldDeltasIntoGeneration(*oracle_req, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); const String gen1_run_key = layout.blobTargetRunKey(1, 0, 0, 0); - const auto gen1_run = backend.get(gen1_run_key); + const auto gen1_run = (*backend_req).read(gen1_run_key, Retry::standard()); ASSERT_TRUE(gen1_run.has_value()); const String gen1_run_bytes = gen1_run->bytes; - /// Sanity: the prior run really spans several blocks (else the block-bounded assertions are - /// vacuous). Blocks seal at kLegacyBlockSize (256KB); ~820KB is 3-4 blocks. + /// Sanity: the prior run is far larger than one read buffer, so "it was never read whole" is a + /// claim about a genuinely large object rather than one a single buffer could have swallowed. ASSERT_GT(gen1_run_bytes.size(), static_cast(kLegacyBlockSize) * 3); /// Reset counters and fold gen 2 with a small delta: remove one edge and add another. The prior - /// gen-1 run must be consumed via the streaming cursor (head + tail get + body getStream + per-seq - /// head probe), NEVER a whole-object get. + /// gen-1 run must be consumed via one streaming open per prior segment, NEVER a whole-object get. backend.resetCounts(); + /// Arm a window smaller than the run so the fold's output is proven correct under chunked delivery, + /// not merely served in one piece: unarmed, `getStream` never installs the recording wrapper at all. + backend.setStreamChunkForTest(kLegacyBlockSize / 4); std::vector gen2{{bh(0), s(1), true}, {bh(19999), s(2), false}}; std::vector runs2_c; std::vector runs2_o; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); - foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); + foldDeltasIntoGeneration(*oracle_req, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); /// Byte-reproducibility canary: streaming and materialized folds produce identical output bytes. const String gen2_run_key = layout.blobTargetRunKey(2, 0, 0, 0); - const auto gen2_c = backend.get(gen2_run_key); - const auto gen2_o = oracle.get(gen2_run_key); + const auto gen2_c = (*backend_req).read(gen2_run_key, Retry::standard()); + const auto gen2_o = (*oracle_req).read(gen2_run_key, Retry::standard()); ASSERT_TRUE(gen2_c.has_value()); ASSERT_TRUE(gen2_o.has_value()); EXPECT_EQ(gen2_c->bytes, gen2_o->bytes); @@ -625,31 +698,31 @@ TEST(CASBlobInDegree, FoldStreamsPriorRunBlockBounded) ASSERT_EQ(runs2_o.size(), 1u); EXPECT_EQ(runs2_c[0].checksum, runs2_o[0].checksum); - /// The core assertion: no whole-object get of the prior run key — every read carried a Range or a - /// stream (the resident-memory proof at the seam). - EXPECT_EQ(backend.wholeGetCount(gen1_run_key), 0u); - /// The cursor opened the prior run's segment via the streaming reader (head + tail get + getStream). + /// The core assertion: the prior run is never read whole — the cursor reaches it only through the + /// streaming open, so the seam sees no read of that key at all. + EXPECT_EQ(backend.getCount(gen1_run_key), 0u); + /// The cursor opened the prior run's segment through the streaming reader. EXPECT_GE(backend.getStreamCount(gen1_run_key), 1u); - /// Every ranged-get window on the prior run stays within one block + the footer allowance. This - /// bound is strict here because the prior run's footer fits inside the fixed tail probe (only very - /// large runs — ~13k blocks — spill the footer past the probe and add one exact-footer get; a note - /// for that regime lives in the streaming reader's open comment). - EXPECT_LE(backend.maxRangedGetLen(gen1_run_key), - static_cast(kLegacyHardCapBlockSize) + 64u * 1024u); - /// Streaming open touches the prior run's tail probe (and at most one exact-footer get); it is never - /// re-materialized whole. - EXPECT_LE(backend.getCount(gen1_run_key), 2u); -} - -/// The preview consumer `zeroInDegree` streams a multi-block run instead of materializing it whole: the -/// backend sees only block-bounded ranged/stream requests for the run key (never a whole-object get), and -/// the candidate set equals the pre-change (borrowed-mode) result. Byte-parity against an InMemory oracle -/// is the load-bearing canary — the scan logic is unchanged; only the byte source moved to the stream. -TEST(CASBlobInDegree, ZeroInDegreeStreamsBlockBounded) + /// The other half of the evidence: a nonzero value here can only come from the recording wrapper + /// `getStream` installs when armed, so this proves the arming actually took effect and the + /// byte-parity check above ran under genuinely chunked delivery, not a no-op setter. + EXPECT_GT(backend.largestStreamChunk(gen1_run_key), 0u); + /// This does NOT bound how much the stream buffers per request: the streaming primitive carries no + /// window for the seam to measure, so resident memory inside the open stream is out of its reach. +} + +/// The preview consumer `zeroInDegree` streams a large run instead of materializing it whole: the +/// backend is never asked to READ the run's key, only to open it as a stream, and the candidate set +/// equals the pre-change (borrowed-mode) result. Byte-parity against an InMemory oracle is the +/// load-bearing canary — the scan logic is unchanged; only the byte source moved to the stream. +/// As above, the buffering inside the open stream is not bounded here; nothing at the seam sees it. +TEST(CASBlobInDegree, ZeroInDegreeStreamsRunWithoutReadingItWhole) { using DB::Cas::tests::CountingBackend; CountingBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); InMemoryBackend oracle; + DB::Cas::tests::OperationForTest oracle_req(oracle); Layout layout{"pool"}; /// Gen 1 from empty prior: ~20000 active edges spill the SourceEdge run across several 256KB blocks. @@ -660,26 +733,30 @@ TEST(CASBlobInDegree, ZeroInDegreeStreamsBlockBounded) std::vector runs1_c; std::vector runs1_o; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); - foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_c); + foldDeltasIntoGeneration(*oracle_req, layout, /*prior_runs*/{}, 1, 0, 0, gen1, runs1_o); /// Gen 2 removes every edge on two of the blobs => two zero-transition markers in the gen-2 run, /// which is itself multi-block (the surviving-edge rows still span blocks). std::vector gen2{{bh(0), s(1), true}, {bh(19999), s(1), true}}; std::vector runs2_c; std::vector runs2_o; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); - foldDeltasIntoGeneration(oracle, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1_c, 2, 0, 0, gen2, runs2_c); + foldDeltasIntoGeneration(*oracle_req, layout, /*prior_runs*/runs1_o, 2, 0, 0, gen2, runs2_o); const String gen2_run_key = layout.blobTargetRunKey(2, 0, 0, 0); - const auto gen2_run = backend.get(gen2_run_key); + const auto gen2_run = (*backend_req).read(gen2_run_key, Retry::standard()); ASSERT_TRUE(gen2_run.has_value()); - /// Sanity: the run genuinely spans several blocks (else the block-bounded assertions are vacuous). + /// Sanity: the run is far larger than one read buffer, for the same reason as above. ASSERT_GT(gen2_run->bytes.size(), static_cast(kLegacyBlockSize) * 3); backend.resetCounts(); - const auto zero_c = zeroInDegree(backend, runs2_c); - const auto zero_o = zeroInDegree(oracle, runs2_o); + /// Arm a window smaller than the run so the candidate set below is proven correct under chunked + /// delivery, not merely served in one piece: unarmed, `getStream` never installs the recording + /// wrapper at all. + backend.setStreamChunkForTest(kLegacyBlockSize / 4); + const auto zero_c = zeroInDegree(*backend_req, runs2_c); + const auto zero_o = zeroInDegree(*oracle_req, runs2_o); /// Equivalence with the borrowed-mode (InMemory oracle) result: same candidates, in the same order. ASSERT_EQ(zero_c.size(), zero_o.size()); @@ -687,39 +764,79 @@ TEST(CASBlobInDegree, ZeroInDegreeStreamsBlockBounded) for (size_t i = 0; i < zero_c.size(); ++i) EXPECT_EQ(zero_c[i].ref, zero_o[i].ref); - /// The core assertion: no whole-object get of the run key — every read carried a Range or a stream. - EXPECT_EQ(backend.wholeGetCount(gen2_run_key), 0u); - /// The scan opened the run via the streaming reader (head + tail get + getStream). + /// The core assertion: the run is never read whole — the scan reaches it only through the streaming + /// open, so the seam sees no read of that key at all. + EXPECT_EQ(backend.getCount(gen2_run_key), 0u); + /// The scan opened the run through the streaming reader. EXPECT_GE(backend.getStreamCount(gen2_run_key), 1u); - /// Every ranged-get window stays within one block + the footer allowance (the seam memory bound). - EXPECT_LE(backend.maxRangedGetLen(gen2_run_key), - static_cast(kLegacyHardCapBlockSize) + 64u * 1024u); - /// Streaming open touches the tail probe (and at most one exact-footer get); never re-materialized whole. - EXPECT_LE(backend.getCount(gen2_run_key), 2u); + /// The other half of the evidence: a nonzero value here can only come from the recording wrapper + /// `getStream` installs when armed, so this proves the arming actually took effect and the + /// candidate-set check above ran under genuinely chunked delivery, not a no-op setter. + EXPECT_GT(backend.largestStreamChunk(gen2_run_key), 0u); + /// This does NOT bound how much the stream buffers per request: the streaming primitive carries no + /// window for the seam to measure, so resident memory inside the open stream is out of its reach. } -/// ==== kCondemned row codec + typed source-edge open (retired-in-snapshot T2, spec §2.1) ==== +/// ==== RunMarker::Condemned row codec + typed source-edge open (retired-in-snapshot T2, spec §2.1) ==== TEST(CASCondemnedRow, RoundTripAllTokenTypes) { - for (auto type : {DB::Cas::TokenType::ETag, DB::Cas::TokenType::Generation, DB::Cas::TokenType::Emulated}) + /// Walked over the vocabulary's own entries rather than a hand-copied list, so a dialect the + /// encoder can construct but this test forgot cannot exist. + for (const auto & entry : DB::Cas::kTokenTypeWords.entries) { DB::Cas::CondemnedRow row; - row.delete_pending = (type == DB::Cas::TokenType::Generation); - row.marker_confirmed = (type == DB::Cas::TokenType::Emulated); - row.token = DB::Cas::Token{.value = "etag-abc-123", .type = type}; + row.delete_pending = (entry.value == DB::Cas::Dialect::Generation); + row.marker_confirmed = (entry.value == DB::Cas::Dialect::Emulated); + row.token = DB::Cas::PersistedEtag{String(entry.word), "etag-abc-123"}; row.size = 4096; row.condemn_round = 7; const auto bytes = DB::Cas::encodeCondemnedRow(row); - ASSERT_EQ(bytes[0], DB::Cas::kCondemned); + ASSERT_EQ(bytes[0], DB::Cas::runMarkerByte(DB::Cas::RunMarker::Condemned)); EXPECT_EQ(DB::Cas::decodeCondemnedRow(bytes), row); } } +TEST(CASCondemnedRow, UnknownMarkerByteFailsClosedWithCorruptedData) +{ + /// This pins the condemned-row decoder's own marker validation. + DB::Cas::CondemnedRow row; + row.token = DB::Cas::PersistedEtag{"etag", "t"}; + auto bytes = DB::Cas::encodeCondemnedRow(row); + bytes[0] = 0x03; + + try + { + static_cast(DB::Cas::decodeCondemnedRow(bytes)); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } +} + +TEST(CASRecordStream, RunMarkerByteContractFailsClosed) +{ + /// This helper is defense-in-depth; upstream word validation means no input path reaches it. + for (const auto marker : {DB::Cas::RunMarker::Zero, DB::Cas::RunMarker::Edge, DB::Cas::RunMarker::Condemned}) + EXPECT_EQ(DB::Cas::runMarkerFromByte(DB::Cas::runMarkerByte(marker), "CAS test"), marker); + + try + { + static_cast(DB::Cas::runMarkerFromByte(0x03, "CAS test")); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } +} + TEST(CASCondemnedRow, UnknownFlagBitsFailClosed) { DB::Cas::CondemnedRow row; - row.token = DB::Cas::Token{.value = "t", .type = DB::Cas::TokenType::ETag}; + row.token = DB::Cas::PersistedEtag{"etag", "t"}; auto bytes = DB::Cas::encodeCondemnedRow(row); bytes[1] = 4; // flags byte: only bits 0 (delete_pending) and 1 (marker_confirmed) are defined EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); @@ -728,7 +845,7 @@ TEST(CASCondemnedRow, UnknownFlagBitsFailClosed) TEST(CASCondemnedRow, UnknownTokenTypeFailsClosed) { DB::Cas::CondemnedRow row; - row.token = DB::Cas::Token{.value = "t", .type = DB::Cas::TokenType::ETag}; + row.token = DB::Cas::PersistedEtag{"etag", "t"}; auto bytes = DB::Cas::encodeCondemnedRow(row); bytes[2] = 99; // token_type byte (offset: [0]=0x02 [1]=flags [2]=token_type) EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); @@ -737,7 +854,7 @@ TEST(CASCondemnedRow, UnknownTokenTypeFailsClosed) TEST(CASCondemnedRow, TruncatedPayloadFailsClosed) { DB::Cas::CondemnedRow row; - row.token = DB::Cas::Token{.value = "0123456789", .type = DB::Cas::TokenType::ETag}; + row.token = DB::Cas::PersistedEtag{"etag", "0123456789"}; auto bytes = DB::Cas::encodeCondemnedRow(row); bytes.resize(bytes.size() - 3); // token bytes shorter than declared token_len EXPECT_THROW(DB::Cas::decodeCondemnedRow(bytes), DB::Exception); @@ -796,6 +913,7 @@ TEST(CASBlobInDegree, TwoAlgoFoldSettlesBothInOneShardRun) /// both settle (edges present, condemn on removal works per ref), mixed rows in one run, no /// algo loop. InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; const BlobRef ch_x{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(11))}; @@ -804,20 +922,20 @@ TEST(CASBlobInDegree, TwoAlgoFoldSettlesBothInOneShardRun) const BlobRef sha_y_ref{BlobHashAlgo::Sha256, sha_y}; std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, 1, /*attempt*/0, 0, {{ch_x, s(1), false}, {sha_y_ref, s(1), false}}, runs1); ASSERT_FALSE(runs1.empty()); EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs1, ch_x), 1); EXPECT_EQ(DB::Cas::tests::inDegreeInRuns(backend, runs1, sha_y_ref), 1); - EXPECT_TRUE(zeroInDegree(backend, runs1).empty()); + EXPECT_TRUE(zeroInDegree(*backend_req, runs1).empty()); /// Remove both edges in gen 2: each transitions to zero independently, condemned per its own ref. std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, 2, /*attempt*/0, 0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, 2, /*attempt*/0, 0, {{ch_x, s(1), true}, {sha_y_ref, s(1), true}}, runs2, - /*current_round*/0, /*condemn_round*/1, headPresent("t", 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); + /*current_round*/0, /*condemn_round*/1, headPresent(*backend_req, layout, 1), /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); ASSERT_EQ(rmr.still_retired.size(), 2u); std::vector condemned_refs{rmr.still_retired[0].ref, rmr.still_retired[1].ref}; @@ -837,23 +955,24 @@ TEST(CASBlobInDegree, TwoAlgoFoldSettlesBothInOneShardRun) TEST(CASBlobInDegree, UnmatchedRemovalIsAPerKeyNoOpAndSparesSiblingEdges) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Generation 1: blob b1 is referenced by TWO distinct sources (two manifests). std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, {{bh(1), s(1), false}, {bh(1), s(2), false}}, runs1); /// Generation 2: fold a removal for a THIRD source that never had an activation folded. std::vector runs2; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, {{bh(1), s(99), true}}, runs2); /// Both original edges survive: the unmatched removal touched only its own (absent) key. - const DecodedRun out = decodeRun(backend, runs2[0]); + const DecodedRun out = decodeRun(*backend_req, runs2[0]); ASSERT_EQ(out.edges.size(), 2u) << "an unmatched removal must not strip sibling edges"; /// And the blob is NOT a deletion candidate. - const auto zero = zeroInDegree(backend, runs2); + const auto zero = zeroInDegree(*backend_req, runs2); EXPECT_TRUE(zero.empty()) << "b1 still has two live source edges"; } @@ -865,23 +984,24 @@ TEST(CASBlobInDegree, UnmatchedRemovalIsAPerKeyNoOpAndSparesSiblingEdges) TEST(CASBlobInDegree, UnmatchedRemovalIsCountedWithAnExample) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; /// Generation 1: blob b1 is referenced by TWO distinct sources (two manifests), same fixture as the /// no-op test above. std::vector runs1; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, {{bh(1), s(1), false}, {bh(1), s(2), false}}, runs1); /// Generation 2: fold a removal for a THIRD source that never had an activation folded. std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/runs1, /*new_generation*/2, /*attempt*/0, /*shard*/0, {{bh(1), s(99), true}}, runs2, /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr); /// The run is byte-identical to the no-op test's outcome for the blob's OTHER edges: both survive. - const DecodedRun out = decodeRun(backend, runs2[0]); + const DecodedRun out = decodeRun(*backend_req, runs2[0]); ASSERT_EQ(out.edges.size(), 2u) << "the counting surface must not perturb the no-op fold outcome"; EXPECT_EQ(out.edges[0].first, b(1)); EXPECT_EQ(out.edges[1].first, b(1)); @@ -915,6 +1035,7 @@ std::vector> condemnedCohort(uint64_t n, uint64 TEST(CASThreeCursorMerge, RedeleteBudgetCapsCohortAndCarriesExcess) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; const RunRef gen1 = writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, 1, /*delete_pending*/true)); @@ -923,7 +1044,7 @@ TEST(CASThreeCursorMerge, RedeleteBudgetCapsCohortAndCarriesExcess) std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, /*current_round*/9, /*condemn_round*/9, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, /*source_retirements*/{}, &budget); @@ -943,6 +1064,7 @@ TEST(CASThreeCursorMerge, RedeleteBudgetCapsCohortAndCarriesExcess) TEST(CASThreeCursorMerge, GraduationBudgetCapsCohortAndCarriesExcess) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; const RunRef gen1 = writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, /*condemn_round*/1, /*delete_pending*/false)); @@ -951,7 +1073,7 @@ TEST(CASThreeCursorMerge, GraduationBudgetCapsCohortAndCarriesExcess) std::vector runs2; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, + foldDeltasIntoGeneration(*backend_req, layout, /*prior_runs*/{gen1}, 2, 0, 0, {}, runs2, /*current_round*/5, /*condemn_round*/6, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, /*source_retirements*/{}, &budget); @@ -973,6 +1095,7 @@ TEST(CASThreeCursorMerge, GraduationBudgetCapsCohortAndCarriesExcess) TEST(CASThreeCursorMerge, RedeleteBudgetDrainsCohortToFixpointOverRounds) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest backend_req(backend); Layout layout{"pool"}; std::vector priors{writeSourceEdgeRun(backend, layout, 1, 0, 0, condemnedCohort(10, 1, /*delete_pending*/true))}; @@ -984,7 +1107,7 @@ TEST(CASThreeCursorMerge, RedeleteBudgetDrainsCohortToFixpointOverRounds) budget.max_redeletes = 3; std::vector out_runs; RetiredMergeResult rmr; - foldDeltasIntoGeneration(backend, layout, priors, 2 + rounds, 0, 0, {}, out_runs, + foldDeltasIntoGeneration(*backend_req, layout, priors, 2 + rounds, 0, 0, {}, out_runs, /*current_round*/100, /*condemn_round*/100, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &rmr, /*suppress_destructive*/false, /*out_applied_by_txn_ordinal*/nullptr, /*source_retirements*/{}, &budget); diff --git a/src/Disks/tests/gtest_cas_blob_meta.cpp b/src/Disks/tests/gtest_cas_blob_meta.cpp index a0a29bf2e8f4..7d33a8614aa6 100644 --- a/src/Disks/tests/gtest_cas_blob_meta.cpp +++ b/src/Disks/tests/gtest_cas_blob_meta.cpp @@ -23,49 +23,50 @@ TEST(CASBlobMeta, PutIfAbsentThenCasTransitions) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasOperation op = store->mountRequests().admit(); const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-a"))}; const BlobMeta clean{.state = MetaState::Clean, .size = 10}; - const CasOverwriteResult created = putMetaIfAbsent(*store, ref, clean); - EXPECT_EQ(created.outcome, CasOverwriteOutcome::Committed); + EXPECT_TRUE(std::holds_alternative(putMetaIfAbsent(op, store->layout(), ref, clean))); - const CasOverwriteResult dup = putMetaIfAbsent(*store, ref, clean); - EXPECT_EQ(dup.outcome, CasOverwriteOutcome::Committed); /// exact-byte resolution adopts the existing marker + /// A create that nothing of its own left unresolved never adopts what is already at the key, even + /// byte-identical: the marker was somebody else's write, and the conflict carries it. + EXPECT_TRUE(std::holds_alternative(putMetaIfAbsent(op, store->layout(), ref, clean))); - const auto lm = loadMeta(*backend, store->layout(), ref); + const auto lm = loadMeta(op, store->layout(), ref); ASSERT_TRUE(lm.has_value()); EXPECT_EQ(lm->meta.state, MetaState::Clean); - const CasOverwriteResult condemned = casMeta(*store, ref, lm->etag, - BlobMeta{.state = MetaState::Condemned, .condemn_round = 5, .size = 10}); - EXPECT_EQ(condemned.outcome, CasOverwriteOutcome::Committed); + EXPECT_TRUE(std::holds_alternative(casMeta(op, store->layout(), ref, lm->etag, + BlobMeta{.state = MetaState::Condemned, .condemn_round = 5, .size = 10}))); - const CasOverwriteResult stale = casMeta(*store, ref, lm->etag, /// stale token loses - BlobMeta{.state = MetaState::Clean}); - EXPECT_EQ(stale.outcome, CasOverwriteOutcome::Conflict); + /// the stale incarnation loses + EXPECT_TRUE(std::holds_alternative(casMeta(op, store->layout(), ref, lm->etag, + BlobMeta{.state = MetaState::Clean}))); } -TEST(CASBlobMeta, DeleteMetaExactMatchesEtag) +TEST(CASBlobMeta, DeleteMetaExactMatchesTheObservedIncarnation) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasOperation op = store->mountRequests().admit(); const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-b"))}; - putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Condemned}); - const auto lm = loadMeta(*backend, store->layout(), ref); + putMetaIfAbsent(op, store->layout(), ref, BlobMeta{.state = MetaState::Condemned}); + const auto lm = loadMeta(op, store->layout(), ref); ASSERT_TRUE(lm.has_value()); - EXPECT_EQ(deleteMetaExact(*backend, store->layout(), ref, lm->etag).kind, DeleteOutcome::Kind::Deleted); - EXPECT_FALSE(loadMeta(*backend, store->layout(), ref).has_value()); + EXPECT_EQ(deleteMetaExact(op, store->layout(), ref, lm->etag), Removal::Removed); + EXPECT_FALSE(loadMeta(op, store->layout(), ref).has_value()); } /// Phase 3 T3 (mixed-algo pools, was CAS pluggable-blob-hash Phase 2 Task 5 crux Test 2): the `.meta` /// API round-trips a 32-byte (`sha256`-width) `BlobRef` key — the meta object lands under a 64-hex /// key, exercising the SAME `putMetaIfAbsent`/`loadMeta`/`casMeta`/`deleteMetaExact` surface PartWriteTxn/Gc -/// use, just at a wider algo. Writes use the `Pool`'s controller; reads and exact deletion retain their -/// direct `Backend`/`Layout` surface. +/// use, just at a wider algo. TEST(CASBlobMeta, PutLoadCasDeleteRoundTripAtWidth32) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasOperation op = store->mountRequests().admit(); const Layout & layout = store->layout(); /// A distinguishable 32-byte digest (not merely a 16-byte value zero-tailed): every byte set. @@ -76,31 +77,31 @@ TEST(CASBlobMeta, PutLoadCasDeleteRoundTripAtWidth32) const String hex = codecFor(BlobHashAlgo::Sha256).toHex(h); EXPECT_EQ(hex.size(), 64u) << "a 32-byte digest renders 64 hex chars"; - const CasOverwriteResult created = putMetaIfAbsent(*store, ref, - BlobMeta{.state = MetaState::Clean, .size = 555}); - ASSERT_EQ(created.outcome, CasOverwriteOutcome::Committed); - EXPECT_TRUE(backend->head(layout.blobMetaKey(ref)).exists) + ASSERT_TRUE(std::holds_alternative( + putMetaIfAbsent(op, store->layout(), ref, BlobMeta{.state = MetaState::Clean, .size = 555}))); + EXPECT_TRUE(op.head(layout.blobMetaKey(ref), Retry::standard()).has_value()) << "the meta object must land under the 64-hex key, not a truncated 32-hex one"; - const auto lm = loadMeta(*backend, layout, ref); + const auto lm = loadMeta(op, layout, ref); ASSERT_TRUE(lm.has_value()); EXPECT_EQ(lm->meta.state, MetaState::Clean); EXPECT_EQ(lm->meta.size, 555u); - const CasOverwriteResult condemned = casMeta(*store, ref, lm->etag, - BlobMeta{.state = MetaState::Condemned, .condemn_round = 7, .size = 555}); - ASSERT_EQ(condemned.outcome, CasOverwriteOutcome::Committed); - const auto lm2 = loadMeta(*backend, layout, ref); + ASSERT_TRUE(std::holds_alternative(casMeta(op, layout, ref, lm->etag, + BlobMeta{.state = MetaState::Condemned, .condemn_round = 7, .size = 555}))); + const auto lm2 = loadMeta(op, layout, ref); ASSERT_TRUE(lm2.has_value()); EXPECT_EQ(lm2->meta.state, MetaState::Condemned); - EXPECT_EQ(deleteMetaExact(*backend, layout, ref, lm2->etag).kind, DeleteOutcome::Kind::Deleted); - EXPECT_FALSE(loadMeta(*backend, layout, ref).has_value()); + EXPECT_EQ(deleteMetaExact(op, layout, ref, lm2->etag), Removal::Removed); + EXPECT_FALSE(loadMeta(op, layout, ref).has_value()); } namespace { +/// The fault sits on the transport primitive, which is what every marker write reaches the store +/// through: a create is a `write` with no precondition, a compare-swap a `write` with one. class ControlledMetaWriteFaultBackend : public InMemoryBackend { public: @@ -109,52 +110,57 @@ class ControlledMetaWriteFaultBackend : public InMemoryBackend uint64_t create_attempts = 0; uint64_t overwrite_attempts = 0; - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - ++create_attempts; - if (throw_next_create) + if (expected_value) { - throw_next_create = false; - throw Poco::TimeoutException("scripted meta create ambiguity"); + ++overwrite_attempts; + if (throw_next_overwrite) + { + throw_next_overwrite = false; + throw Poco::TimeoutException("scripted meta overwrite ambiguity"); + } } - return InMemoryBackend::putIfAbsent(key, bytes, meta); - } - - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override - { - ++overwrite_attempts; - if (throw_next_overwrite) + else { - throw_next_overwrite = false; - throw Poco::TimeoutException("scripted meta overwrite ambiguity"); + ++create_attempts; + if (throw_next_create) + { + throw_next_create = false; + throw Poco::TimeoutException("scripted meta create ambiguity"); + } } - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; } -TEST(CASBlobMeta, WritesUsePoolRequestController) +/// An ambiguous marker write is settled by the engine's exact read and reissued, rather than escaping +/// as a raw transport error: the create's resolve proves the key still absent, the compare-swap's +/// proves the expected incarnation still current, and both are repeatable. +TEST(CASBlobMeta, AnAmbiguousMarkerWriteIsResolvedAndReissued) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); store->setCasRetrySleepForTest([](uint64_t) {}); + CasOperation op = store->mountRequests().admit(); backend->create_attempts = 0; backend->overwrite_attempts = 0; const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("hash-controlled"))}; backend->throw_next_create = true; - EXPECT_EQ( - putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Clean, .size = 10}).outcome, - CasOverwriteOutcome::Committed); + EXPECT_TRUE(std::holds_alternative( + putMetaIfAbsent(op, store->layout(), ref, BlobMeta{.state = MetaState::Clean, .size = 10}))); EXPECT_EQ(backend->create_attempts, 2u); - const auto clean = loadMeta(*backend, store->layout(), ref); + const auto clean = loadMeta(op, store->layout(), ref); ASSERT_TRUE(clean.has_value()); backend->throw_next_overwrite = true; - EXPECT_EQ( - casMeta(*store, ref, clean->etag, BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 10}).outcome, - CasOverwriteOutcome::Committed); + EXPECT_TRUE(std::holds_alternative(casMeta(op, store->layout(), ref, clean->etag, + BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 10}))); EXPECT_EQ(backend->overwrite_attempts, 2u); } diff --git a/src/Disks/tests/gtest_cas_blob_meta_format.cpp b/src/Disks/tests/gtest_cas_blob_meta_format.cpp index bb055d2cef71..581b56f1b070 100644 --- a/src/Disks/tests/gtest_cas_blob_meta_format.cpp +++ b/src/Disks/tests/gtest_cas_blob_meta_format.cpp @@ -2,10 +2,35 @@ #include #include +#include + using namespace DB::Cas; namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; } +namespace +{ +/// Same tiny inline copy as `gtest_cas_wire_vocab.cpp`'s `expectThrowsCode`: stays clear of +/// `Disks/tests/cas_test_helpers.h`'s `DB::Cas::tests::expectThrowsCode`, which would both drag +/// in the whole CAS backend/store machinery this file otherwise has no need for AND collide (same +/// namespace, same name and signature) if that header were ever included here too. +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} +} + +CAS_BATTERY_COVERS(BlobMeta); + TEST(CASFormatBattery, BlobMeta) { BlobMeta m; @@ -16,8 +41,8 @@ TEST(CASFormatBattery, BlobMeta) .id = FormatId::BlobMeta, .encode = [&] { return sealObject(FormatId::BlobMeta, encodeBlobMeta(m)); }, .decode = [](std::string_view s) { decodeBlobMeta(std::string(openObject(FormatId::BlobMeta, s))); }, - .golden = "{\"type\":\"cas_blob_meta\",\"v\":10}\n" - "{\"st\":\"clean\",\"cr\":\"0\",\"sz\":\"12345\"}\n"}); + .golden = "{\"type\":\"cas_blob_meta\",\"v\":1}\n" + "{\"state\":\"clean\",\"condemn_round\":\"0\",\"size\":\"12345\"}\n"}); } TEST(CASBlobMetaFormat, CondemnedRoundTripAllFields) @@ -31,19 +56,30 @@ TEST(CASBlobMetaFormat, CondemnedRoundTripAllFields) EXPECT_EQ(back.condemn_round, 7u); EXPECT_EQ(back.size, 4096u); EXPECT_EQ(encodeBlobMeta(m), - "{\"type\":\"cas_blob_meta\",\"v\":10}\n{\"st\":\"condemned\",\"cr\":\"7\",\"sz\":\"4096\"}\n"); + "{\"type\":\"cas_blob_meta\",\"v\":1}\n{\"state\":\"condemned\",\"condemn_round\":\"7\",\"size\":\"4096\"}\n"); +} + +/// Closed-set pin: the two `MetaState` wire words, walked through `magic_enum::enum_values` so a +/// future state a `MetaState` construction can reach but no table entry names would fail this +/// exhaustive check rather than silently pass through unspecified. +TEST(CASBlobMetaFormat, ClosedSetPinsMetaStateWords) +{ + EXPECT_EQ(metaStateToWireWord(MetaState::Clean), "clean"); + EXPECT_EQ(metaStateToWireWord(MetaState::Condemned), "condemned"); + for (const auto state : magic_enum::enum_values()) + EXPECT_EQ(metaStateFromWireWord(metaStateToWireWord(state)), state); } TEST(CASBlobMetaFormat, FailsClosedOnUnknownStateAndTruncation) { /// Unknown state word -> CORRUPTED_DATA (mirrors the old `state > Condemned` reject). - /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes - /// the header gate, which is the point — the BODY is what has to fail here. - const String bad_state = "{\"type\":\"cas_blob_meta\",\"v\":3}\n{\"st\":\"zombie\",\"cr\":\"0\",\"sz\":\"0\"}\n"; - EXPECT_THROW(decodeBlobMeta(bad_state), DB::Exception); + /// `v:1` is the baseline generation, so it always passes the header gate -- the BODY is what has + /// to fail here. + const String bad_state = "{\"type\":\"cas_blob_meta\",\"v\":1}\n{\"state\":\"zombie\",\"condemn_round\":\"0\",\"size\":\"0\"}\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeBlobMeta(bad_state); }); /// Missing state key -> CORRUPTED_DATA. - const String no_state = "{\"type\":\"cas_blob_meta\",\"v\":3}\n{\"cr\":\"0\",\"sz\":\"0\"}\n"; - EXPECT_THROW(decodeBlobMeta(no_state), DB::Exception); + const String no_state = "{\"type\":\"cas_blob_meta\",\"v\":1}\n{\"condemn_round\":\"0\",\"size\":\"0\"}\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeBlobMeta(no_state); }); /// Truncated (header only) -> CORRUPTED_DATA. - EXPECT_THROW(decodeBlobMeta("{\"type\":\"cas_blob_meta\",\"v\":3}\n"), DB::Exception); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { decodeBlobMeta("{\"type\":\"cas_blob_meta\",\"v\":1}\n"); }); } diff --git a/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp b/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp index cfb150401cf2..07b71ee96cff 100644 --- a/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp +++ b/src/Disks/tests/gtest_cas_bootstrap_ordering.cpp @@ -6,7 +6,9 @@ #include #include "cas_test_helpers.h" #include +#include +#include #include #include #include @@ -24,6 +26,7 @@ namespace DB::ErrorCodes { extern const int INVALID_STATE; +extern const int NOT_IMPLEMENTED; } using namespace DB::Cas; @@ -41,53 +44,48 @@ const String kProbeUid2 = "fedcba9876543210fedcba9876543210"; /// Records the ORDER of backend operations so a test can assert that the residual LIST precedes the first /// write, and that a fail path performs zero writes. Delegates every operation to `InMemoryBackend` /// unchanged; `Pool::open` wraps this in its `InstrumentedBackend`, which forwards every op here. -class RecordingBackend final : public InMemoryBackend +/// +/// The `write` primitive covers create, replace and conditional-put alike, so the log distinguishes +/// only writes from removals -- which is all the ordering assertions ask. +class RecordingBackend : public InMemoryBackend { public: - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; + /// Unhide the legacy `list` overloads the primitive override below would otherwise hide: the tests + /// seed and inspect this store through them. + using Backend::list; - enum class Op : uint8_t { List, PutIfAbsent, PutOverwrite, CasPut, Delete }; + enum class Op : uint8_t { List, Write, Remove }; struct Entry { Op op; String key; /// the LIST prefix, or the written key }; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Recorded at the PRIMITIVE, which every legacy forwarder reaches too, so an op is logged + /// whichever surface issued it. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { record(Op::List, prefix); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - record(Op::PutIfAbsent, key); - return InMemoryBackend::putIfAbsent(key, bytes, meta); + record(Op::Write, key); + return InMemoryBackend::write(key, bytes, expected_value, access); } - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - record(Op::PutOverwrite, key); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + record(Op::Remove, key); + return InMemoryBackend::remove(key, expected_value, access); } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override - { - record(Op::CasPut, key); - return InMemoryBackend::casPut(key, bytes, expected, meta); - } - DeleteOutcome deleteExact(const String & key, const Token & token) override - { - record(Op::Delete, key); - return InMemoryBackend::deleteExact(key, token); - } - /// The bootstrap path (battery + createOrValidate + mount protocol) issues only whole-String writes, - /// never a streaming create, so recording the four write ops above captures every write `open` can do. + /// `publish` is the one mutating primitive left unrecorded: it writes a blob, and the bootstrap + /// path (battery + `createOrValidate` + mount protocol) publishes none. static bool isWrite(Op op) { - return op == Op::PutIfAbsent || op == Op::PutOverwrite || op == Op::CasPut || op == Op::Delete; + return op == Op::Write || op == Op::Remove; } void clearLog() @@ -125,13 +123,11 @@ class RecordingBackend final : public InMemoryBackend class CatalogMissingAfterListBackend final : public InMemoryBackend { public: - using Backend::get; - - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (key == Layout{kPrefix}.refCatalogKey()) return std::nullopt; - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } }; @@ -144,6 +140,27 @@ PoolConfig makeConfig() return cfg; } +/// A one-shot `create` for seeding fixture bytes before `Pool::open` runs, asserting it committed. +void seedObject(Backend & backend, const String & key, const String & bytes) +{ + DB::Cas::tests::OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// Whether `key` has a value, through an exact read (mirrors the retired `backend->get(key).has_value()`). +bool readPresent(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).read(key, Retry::standard()).has_value(); +} + +/// Whether `key` has a value, through a HEAD (mirrors the retired `backend->head(key).exists`). +bool headPresent(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + template void expectThrowsCodeContaining(int expected_code, const String & needle, F && fn); @@ -151,9 +168,9 @@ void expectCatalogResidueRefusesWithoutPoolMeta(const String & bytes, const Stri { auto backend = std::make_shared(); const Layout layout{kPrefix}; - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), bytes).outcome, PutOutcome::Done); + seedObject(*backend, layout.refCatalogKey(), bytes); if (!extra_key.empty()) - ASSERT_EQ(backend->putIfAbsent(extra_key, "residual").outcome, PutOutcome::Done); + seedObject(*backend, extra_key, "residual"); backend->clearLog(); try @@ -166,7 +183,7 @@ void expectCatalogResidueRefusesWithoutPoolMeta(const String & bytes, const Stri EXPECT_EQ(e.code(), DB::ErrorCodes::INVALID_STATE); } EXPECT_EQ(backend->writeCount(), 0u); - EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); + EXPECT_FALSE(headPresent(*backend, layout.poolMetaKey())); } /// Index of the first op matching `pred`, if any. @@ -208,7 +225,7 @@ TEST(CASBootstrapOrdering, EmptyPrefixOpensAndListsBeforeAnyWrite) PoolPtr store = Pool::open(backend, makeConfig()); ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); - EXPECT_TRUE(backend->get(kPoolMetaKey).has_value()) << "_pool_meta must be created on a fresh empty prefix"; + EXPECT_TRUE(readPresent(*backend, kPoolMetaKey)) << "_pool_meta must be created on a fresh empty prefix"; const auto log = backend->snapshot(); const auto residual_list = firstIndex(log, [](const RecordingBackend::Entry & e) @@ -239,15 +256,70 @@ TEST(CASBootstrapOrdering, ResidualWithoutMetaFailsTypedWithZeroWrites) { auto backend = std::make_shared(); /// Seed residue an incomplete erase would have left behind (a ref-log object), with no `_pool_meta`. - ASSERT_EQ(backend->putIfAbsent(residualRefLogKey(), "x").outcome, - PutOutcome::Done); + seedObject(*backend, residualRefLogKey(), "x"); backend->clearLog(); expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", [&] { Pool::open(backend, makeConfig()); }); EXPECT_EQ(backend->writeCount(), 0u) << "the fail path must perform zero writes (battery never ran)"; - EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()) << "a fresh _pool_meta must NOT have been minted"; + EXPECT_FALSE(readPresent(*backend, kPoolMetaKey)) << "a fresh _pool_meta must NOT have been minted"; +} + +/// The engine's attempt number reaches the transport even through the bootstrap's own residual LIST. +/// A backend that fails only the FIRST attempt of every LIST +/// (as the adaptive-timeout fuse would) must still let the residual check succeed on attempt 2 -- if +/// propagation were broken every attempt would look like attempt 1 and the LIST would never succeed, +/// which the bootstrap reports as `BootstrapResidual::Indeterminate` ("could not authoritatively list"), +/// a DIFFERENT message from the one asserted below. Reuses `ResidualWithoutMetaFailsTypedWithZeroWrites`'s +/// exact seeding helper and expected error code so the assertion distinguishes "refused because listed" +/// from "refused because the LIST failed". +TEST(CASBootstrapOrdering, ResidualListSucceedsOnTheSecondAttempt) +{ + /// Every LIST whose attempt number is 1 fails as the first-attempt fuse would; attempt 2 answers. + struct FuseOnFirstList : RecordingBackend + { + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + if (access.attemptNo() == 1) + throw Poco::TimeoutException("Timeout"); + return RecordingBackend::list(prefix, cursor, limit, access); + } + }; + auto backend = std::make_shared(); + /// A healthy pool without `_pool_meta` is the shape that needs the LIST: seed one residual key. + seedObject(*backend, residualRefLogKey(), "x"); + backend->clearLog(); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", + [&] { Pool::open(backend, makeConfig()); }); + + bool listed_on_second = false; + for (const auto & e : backend->snapshot()) + listed_on_second |= (e.op == RecordingBackend::Op::List); + EXPECT_TRUE(listed_on_second) << "the residual LIST must have been answered (on attempt 2), not merely failed forever"; +} + +/// (b') The residual verdict is decided by the first residual key, not by an enumeration of the whole +/// prefix: forty residue keys and a 32-key page must cost exactly ONE list request. Enumerating a +/// large prefix is the one request a slow store cannot answer within an attempt, and a refusal +/// needs none of it. +TEST(CASBootstrapOrdering, ResidualWithoutMetaIsDecidedByTheFirstPage) +{ + auto backend = std::make_shared(); + for (uint64_t i = 1; i <= 40; ++i) + seedObject(*backend, Layout{"p"}.refLogKey(DB::Cas::tests::fixture::fixtureLife(RootNamespace{"test%2Fabcd"}), RefTxnId{1, i}), "x"); + backend->clearLog(); + + expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", + [&] { Pool::open(backend, makeConfig()); }); + + size_t root_lists = 0; + for (const auto & e : backend->snapshot()) + if (e.op == RecordingBackend::Op::List && e.key == kPrefix + "/") + ++root_lists; + EXPECT_EQ(root_lists, 1u) << "the first residual key settles the verdict; nothing past it may be enumerated"; + EXPECT_EQ(backend->writeCount(), 0u); } /// (c) A prefix containing ONLY stale, structurally-valid `_probe//…` debris (a crash-mid-battery @@ -256,27 +328,27 @@ TEST(CASBootstrapOrdering, ResidualWithoutMetaFailsTypedWithZeroWrites) TEST(CASBootstrapOrdering, StaleProbeDebrisOnlyIsTreatedAsEmpty) { auto backend = std::make_shared(); - ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/cas", "cas-s1").outcome, PutOutcome::Done); + seedObject(*backend, "p/_probe/" + kProbeUid + "/token", "probe-v1"); + seedObject(*backend, "p/_probe/" + kProbeUid + "/cas", "cas-s1"); backend->clearLog(); PoolPtr store; ASSERT_NO_THROW(store = Pool::open(backend, makeConfig())); EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); - EXPECT_TRUE(backend->get(kPoolMetaKey).has_value()) << "_pool_meta must be created over a probe-only prefix"; + EXPECT_TRUE(readPresent(*backend, kPoolMetaKey)) << "_pool_meta must be created over a probe-only prefix"; } TEST(CASBootstrapOrdering, CanonicalEmptyCatalogOnlyIsTheSoleRetryablePreMetaResidue) { auto backend = std::make_shared(); const Layout layout{kPrefix}; - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(kPrefix + "/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); + seedObject(*backend, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})); + seedObject(*backend, kPrefix + "/_probe/" + kProbeUid + "/token", "probe-v1"); backend->clearLog(); PoolPtr store; ASSERT_NO_THROW(store = Pool::open(backend, makeConfig())); - EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); + EXPECT_TRUE(headPresent(*backend, layout.poolMetaKey())); } TEST(CASBootstrapOrdering, MalformedCatalogOnlyResidueRefusesWithoutPoolMeta) @@ -316,11 +388,11 @@ TEST(CASBootstrapOrdering, ListedCatalogMissingAtExactGetRefusesWithoutPoolMeta) { auto backend = std::make_shared(); const Layout layout{kPrefix}; - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, PutOutcome::Done); + seedObject(*backend, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})); expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", [&] { Pool::open(backend, makeConfig()); }); - EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); + EXPECT_FALSE(headPresent(*backend, layout.poolMetaKey())); } /// (d) An existing healthy pool (meta present + data) → reopen is unchanged: the pool identity is @@ -342,6 +414,49 @@ TEST(CASBootstrapOrdering, HealthyPoolReopenPreservesIdentity) << "a healthy reopen must NOT re-mint _pool_meta — the pool identity must be preserved"; } +/// (d') An existing pool whose prefix the store cannot LIST at the moment (a large prefix on a store +/// that enumerates slowly, a LIST budget that expires) still reopens: `_pool_meta` present is proven by +/// ONE exact read, and the residual LIST is only the absent-key path. Before this, a pool that could be +/// read perfectly well refused to start because the enumeration that would have found the same key +/// did not return in time. +TEST(CASBootstrapOrdering, HealthyPoolReopensWhenThePrefixCannotBeListed) +{ + /// Refuses every LIST of the pool root once armed; everything else is the ordinary store. + class UnlistableRootBackend final : public RecordingBackend + { + public: + using RecordingBackend::list; + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + if (refuse_root_list && prefix == kPrefix + "/") + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "UnlistableRootBackend: the pool root cannot be enumerated right now"); + return RecordingBackend::list(prefix, cursor, limit, access); + } + std::atomic refuse_root_list{false}; + }; + + auto backend = std::make_shared(); + + UInt128 pool_id_first; + { + PoolPtr store = Pool::open(backend, makeConfig()); + pool_id_first = store->poolMeta().pool_id; + } /// clean teardown: drained farewell, so the reopen reclaims immediately + + backend->refuse_root_list = true; + backend->clearLog(); + PoolPtr store2; + ASSERT_NO_THROW(store2 = Pool::open(backend, makeConfig())) + << "an existing pool must reopen on the exact read of _pool_meta alone"; + EXPECT_EQ(store2->lifecycle(), PoolLifecycle::Live); + EXPECT_EQ(store2->poolMeta().pool_id, pool_id_first); + const auto log = backend->snapshot(); + EXPECT_FALSE(firstIndex(log, [](const RecordingBackend::Entry & e) + { return e.op == RecordingBackend::Op::List && e.key == kPrefix + "/"; }).has_value()) + << "a pool whose _pool_meta was read must not be enumerated to prove it exists"; +} + /// (e) [D2] concurrent-opener case: debris from a SECOND concurrent fresh opener's in-flight battery (a /// distinct probe uid) is skipped by the SAME structural rule as (c). Two openers racing over one shared /// pool prefix must not make each other's zero-write residual check fail. @@ -349,9 +464,9 @@ TEST(CASBootstrapOrdering, ConcurrentOpenerProbeDebrisIsAlsoSkipped) { auto backend = std::make_shared(); /// This mount's own crashed battery AND a concurrent opener's in-flight battery. - ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid + "/token", "probe-v1").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid2 + "/token", "probe-v1").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent("p/_probe/" + kProbeUid2 + "/cas", "cas-s1").outcome, PutOutcome::Done); + seedObject(*backend, "p/_probe/" + kProbeUid + "/token", "probe-v1"); + seedObject(*backend, "p/_probe/" + kProbeUid2 + "/token", "probe-v1"); + seedObject(*backend, "p/_probe/" + kProbeUid2 + "/cas", "cas-s1"); backend->clearLog(); PoolPtr store; @@ -367,13 +482,13 @@ TEST(CASBootstrapOrdering, ConcurrentOpenerProbeDebrisIsAlsoSkipped) TEST(CASBootstrapOrdering, ProbeSiblingLookalikeIsResidualNotDebris) { auto backend = std::make_shared(); - ASSERT_EQ(backend->putIfAbsent("p/_probelike/token", "x").outcome, PutOutcome::Done); + seedObject(*backend, "p/_probelike/token", "x"); backend->clearLog(); expectThrowsCodeContaining(DB::ErrorCodes::INVALID_STATE, "refusing to bootstrap over residual data", [&] { Pool::open(backend, makeConfig()); }); EXPECT_EQ(backend->writeCount(), 0u); - EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); + EXPECT_FALSE(readPresent(*backend, kPoolMetaKey)); } /// (g) An OBSERVE / read-only open over a partially-erased pool (residual data, `_pool_meta` deleted) @@ -384,8 +499,7 @@ TEST(CASBootstrapOrdering, ProbeSiblingLookalikeIsResidualNotDebris) TEST(CASBootstrapOrdering, ReadOnlyOverResidualWithoutMetaFailsClosedNoMint) { auto backend = std::make_shared(); - ASSERT_EQ(backend->putIfAbsent(residualRefLogKey(), "x").outcome, - PutOutcome::Done); + seedObject(*backend, residualRefLogKey(), "x"); backend->clearLog(); PoolConfig cfg = makeConfig(); @@ -394,7 +508,7 @@ TEST(CASBootstrapOrdering, ReadOnlyOverResidualWithoutMetaFailsClosedNoMint) [&] { Pool::open(backend, cfg); }); EXPECT_EQ(backend->writeCount(), 0u) << "an observe open must never write (least of all mint _pool_meta)"; - EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); + EXPECT_FALSE(readPresent(*backend, kPoolMetaKey)); } /// (h) An observe / read-only open over a HEALTHY pool (meta present) is unchanged: it validates the @@ -428,9 +542,10 @@ TEST(CASBootstrapOrdering, DecommissionWithAbsentMetaFailsClosedNoMint) } /// Delete only `_pool_meta`, leaving the owner anchor (and other control objects) behind. { - const auto h = backend->head(kPoolMetaKey); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->deleteExact(kPoolMetaKey, h.token).kind, DeleteOutcome::Kind::Deleted); + DB::Cas::tests::OperationForTest op(*backend); + const auto h = (*op).head(kPoolMetaKey, Retry::standard()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*op).remove(kPoolMetaKey, h->etag, Retry::once()), Removal::Removed); } backend->clearLog(); @@ -438,5 +553,5 @@ TEST(CASBootstrapOrdering, DecommissionWithAbsentMetaFailsClosedNoMint) [&] { Pool::openForDecommission(backend, makeConfig(), kSrid); }); EXPECT_EQ(backend->writeCount(), 0u) << "decommission must not mint a fresh _pool_meta"; - EXPECT_FALSE(backend->get(kPoolMetaKey).has_value()); + EXPECT_FALSE(readPresent(*backend, kPoolMetaKey)); } diff --git a/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp new file mode 100644 index 000000000000..68aeabeb9523 --- /dev/null +++ b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp @@ -0,0 +1,197 @@ +#include + +#include "config.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// `removeManyWriteOnce` deletes up to 1000 write-once keys in one request with no precondition: +/// an absent key is success, a present one is gone afterwards, and every backend honours the same +/// fault knobs the single-key delete has. + +namespace ProfileEvents +{ + extern const Event CASBulkDeleteRequests; + extern const Event CASManifestDelete; + extern const Event CASRootDelete; +} + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; +using DB::Cas::tests::openRequestsForTest; + +namespace +{ + +const Layout kLayout{"p"}; +const RootNamespace kNs{"test/aa@cas@"}; + +ManifestId manifest(uint32_t ordinal) +{ + return ManifestId{kNs, ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = ordinal}}; +} + +/// Three manifest keys with bodies and one that was never written. +struct Keys +{ + std::vector present; + WriteOnceKey absent; +}; + +Keys seed(CasOperation & op) +{ + Keys keys{.present = {}, .absent = kLayout.writeOnceManifestKey(manifest(4))}; + for (uint32_t ordinal = 1; ordinal <= 3; ++ordinal) + { + const WriteOnceKey key = kLayout.writeOnceManifestKey(manifest(ordinal)); + EXPECT_TRUE(std::holds_alternative(op.create(key.str(), "body-" + std::to_string(ordinal), Retry::once()))); + keys.present.push_back(key); + } + return keys; +} + +} + +TEST(CASBulkDeleteBackend, InMemoryDeletesPresentKeysAndTreatsAbsentAsSuccess) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + Keys keys = seed(op); + std::vector batch = keys.present; + batch.push_back(keys.absent); + + op.removeManyWriteOnce(batch, Retry::once()); + + for (const WriteOnceKey & key : batch) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); + EXPECT_EQ(backend->bulkRemoveCalls(), 1u); +} + +TEST(CASBulkDeleteBackend, InMemoryHeldDeletesLandLater) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + Keys keys = seed(op); + + backend->setHoldDeletes(true); + op.removeManyWriteOnce(keys.present, Retry::once()); + for (const WriteOnceKey & key : keys.present) + EXPECT_TRUE(op.head(key.str(), Retry::once()).has_value()) << "held, not landed: " << key.str(); + while (backend->pendingDeletes() > 0) + backend->landPendingDelete(0); + for (const WriteOnceKey & key : keys.present) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); +} + +TEST(CASBulkDeleteBackend, InMemoryArmedFailureFiresOnceAndTheHookRunsBeforeTheDeletes) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + Keys keys = seed(op); + + size_t hook_runs = 0; + backend->onBeforeBulkRemove([&] { ++hook_runs; }); + backend->failNextBulkRemoveWith(std::make_exception_ptr(Poco::TimeoutException("injected"))); + + op.removeManyWriteOnce(keys.present, Retry::standard()); /// the engine reissues the chunk + EXPECT_EQ(backend->bulkRemoveCalls(), 2u); + EXPECT_EQ(hook_runs, 1u) << "the hook runs on the attempt that deletes, not on the refused one"; + for (const WriteOnceKey & key : keys.present) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); +} + +TEST(CASBulkDeleteBackend, InstrumentedCountsOneRequestAndOneDeletePerKeyClass) +{ + auto inner = std::make_shared(); + auto backend = std::make_shared(inner); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + Keys keys = seed(op); + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(kNs, DB::UInt128(0x55)); + const WriteOnceKey log = kLayout.writeOnceRefLogKey(life, RefTxnId{1, 1}); + ASSERT_TRUE(std::holds_alternative(op.create(log.str(), "log", Retry::once()))); + + const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load(); + const auto manifest_before = ProfileEvents::global_counters[ProfileEvents::CASManifestDelete].load(); + const auto root_before = ProfileEvents::global_counters[ProfileEvents::CASRootDelete].load(); + + std::vector batch = keys.present; + batch.push_back(log); + op.removeManyWriteOnce(batch, Retry::once()); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load() - requests_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASManifestDelete].load() - manifest_before, 3u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRootDelete].load() - root_before, 1u); +} + +#if USE_AWS_S3 +TEST(CASBulkDeleteBackend, ThrottlingRefusesTheChunkOnceAndTheEngineReissuesIt) +{ + auto inner = std::make_shared(); + auto backend = std::make_shared(inner, ThrottlingBackend::Mode::FirstPerKey, 1, 429); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + /// Seeded under `Retry::standard()`: the FirstPerKey refusal on each key's own create is an + /// AMBIGUOUS attempt the engine must reissue to land at all, which only a reissuable policy grants. + std::vector present; + for (uint32_t ordinal = 1; ordinal <= 3; ++ordinal) + { + const WriteOnceKey key = kLayout.writeOnceManifestKey(manifest(ordinal)); + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "body-" + std::to_string(ordinal), Retry::standard()))); + present.push_back(key); + } + /// the FirstPerKey refusal is spent on these keys' writes above; the bulk delete's own request + /// each key names is the SECOND request naming it and passes unrefused. + op.removeManyWriteOnce(present, Retry::standard()); + for (const WriteOnceKey & key : present) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); + EXPECT_GE(backend->refusals(present.front().str()), 1u); +} +#endif + +#if USE_AWS_S3 +TEST(CASBulkDeleteBackend, EmulatedModeDeletesUnderTheEmulationLockAndForgetsTheTokens) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto backend = std::make_shared(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + Keys keys = seed(op); + std::vector batch = keys.present; + batch.push_back(keys.absent); + + op.removeManyWriteOnce(batch, Retry::once()); + + for (const WriteOnceKey & key : batch) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); + /// A recreate at a deleted key must mint a fresh incarnation, which is what the token bookkeeping + /// after a delete exists for. + EXPECT_TRUE(std::holds_alternative(op.create(keys.present.front().str(), "again", Retry::once()))); +} + +TEST(CASBulkDeleteBackend, LocalObjectStorageRefusesTheProfileOverload) +{ + auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + DB::StoredObjects objects{DB::StoredObject("p/anything")}; + expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, [&] + { + storage->removeObjectsIfExistUnderProfile(objects, DB::ObjectStorageControlRequest{ + .profile = DB::ObjectStorageRetryProfile::SingleAttempt, .attempt_timeout_ms = 1000}); + }); +} +#endif diff --git a/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp b/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp new file mode 100644 index 000000000000..fc84d67ebc61 --- /dev/null +++ b/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp @@ -0,0 +1,91 @@ +#include + +#include +#include +#include +#include +#include + +/// The engine sends one chunk of at most 1000 write-once keys as one request under the ordinary +/// attempt loop; a failed attempt reissues the whole chunk, which is sound because a key the failed +/// attempt already deleted is absent on the reissue, and absence is success. + +namespace ProfileEvents +{ + extern const Event CASRequestReissue; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::openRequestsForTest; + +namespace +{ + +const Layout kLayout{"p"}; +const RootNamespace kNs{"test/aa@cas@"}; + +std::vector manifestKeys(uint32_t count) +{ + std::vector keys; + for (uint32_t ordinal = 1; ordinal <= count; ++ordinal) + keys.push_back(kLayout.writeOnceManifestKey( + ManifestId{kNs, ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = ordinal}})); + return keys; +} + +} + +TEST(CASBulkDeleteEngine, AFailedAttemptReissuesTheWholeChunkAndAlreadyDeletedKeysAreSuccess) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const std::vector keys = manifestKeys(5); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + /// Half the chunk is gone before the failed attempt reports: the reissue must still succeed. + { + const auto h = op.head(keys[0].str(), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ(op.remove(keys[0].str(), h->etag, Retry::once()), Removal::Removed); + } + backend->failNextBulkRemoveWith(std::make_exception_ptr(Poco::TimeoutException("injected"))); + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + + op.removeManyWriteOnce(keys, Retry::standard()); + + EXPECT_EQ(backend->bulkRemoveCalls(), 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 1u); + for (const WriteOnceKey & key : keys) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); +} + +TEST(CASBulkDeleteEngine, AnEmptyChunkIsNoRequest) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + op.removeManyWriteOnce({}, Retry::once()); + EXPECT_EQ(backend->bulkRemoveCalls(), 0u); +} + +TEST(CASBulkDeleteEngine, ExactlyOneThousandKeysIsOneRequest) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + op.removeManyWriteOnce(manifestKeys(1000), Retry::once()); /// all absent: success, one request + EXPECT_EQ(backend->bulkRemoveCalls(), 1u); +} + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASBulkDeleteEngineDeathTest, MoreThanOneThousandKeysIsACallerBug) +{ + auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + EXPECT_DEATH({ op.removeManyWriteOnce(manifestKeys(1001), Retry::once()); }, "removeManyWriteOnce"); + EXPECT_EQ(backend->bulkRemoveCalls(), 0u); +} +#endif diff --git a/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp index 4b19f9ae967e..c2a40218c0a7 100644 --- a/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp +++ b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp @@ -2,6 +2,8 @@ #include "config.h" +#include + #include #include #include @@ -10,10 +12,15 @@ #include #include #include +#include #include #include #include +#include + +#include +#include #include #include #include @@ -23,6 +30,7 @@ #include #include #include +#include #include #include #include @@ -42,8 +50,10 @@ /// recover from storage to answer, and it must not even MATERIALIZE a runtime -- a read-only /// interserver query must never be able to make this writer do work. /// - The snapshot spans BOTH lane mutexes, so an append admitted concurrently is ordered strictly -/// after it: there is no window in which the confirm says `Yes` while a removal of that ref is -/// already admitted. +/// after it: there is no window in which the confirm says `Yes` while a mutation OF THAT REF is +/// already admitted. A queued or in-flight mutation of another SINGLE ref does not refuse -- rule 3 +/// reads each item's `MutationScope`, and refusing for the whole table starved two replicas of each +/// other under load on a slow control plane. A `WholeShard`-scoped mutation still refuses every ref. /// /// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. @@ -51,6 +61,15 @@ namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int MEMORY_LIMIT_EXCEEDED; +extern const int LOGICAL_ERROR; +} + +namespace ProfileEvents +{ + extern const Event CASRelinkConfirmRefusedRefMutationInFlight; + extern const Event CASRelinkConfirmRefusedLaneWedged; + extern const Event CASRelinkConfirmRefusedLaneBroken; + extern const Event CASRelinkConfirmRefusedMountCannotSpeak; } using namespace DB::Cas; @@ -65,25 +84,98 @@ namespace /// `recovery_in_progress` set). The failure is deliberately `CORRUPTED_DATA`: /// `isTransientRecoveryError` does not list it, so recovery fails fast instead of burning its retry /// budget. -class RecoveryLatchBackend : public CountingBackend +/// The engine reissues an unresolved write until its OWN retry window closes, and that window is +/// measured on a clock the engine reads. Both seams here share one counter -- the sleep the engine +/// performs is what advances the clock -- so a fault that stays armed ends the call at its deadline +/// with no real time passing. Installed on the whole pool, because the ref-lane write, its settling +/// read and the recovery retry loop all pace through the same seam. The pool owns the closures and the +/// closures own the clock, so it outlives everything that can still read it. +class VirtualRetryClock +{ +public: + static std::shared_ptr installOn(const PoolPtr & store) + { + auto clock = std::make_shared(); + store->setCasRequestNowFnForTest([clock] { return clock->nowMs(); }); + store->setCasRetrySleepForTest([clock](uint64_t ms) { clock->advance(ms); }); + return clock; + } + + uint64_t nowMs() const + { + std::lock_guard lock(mutex); + return now_ms; + } + size_t pauseCount() const + { + std::lock_guard lock(mutex); + return pauses; + } + uint64_t longestPause() const + { + std::lock_guard lock(mutex); + return longest_pause; + } + + void advance(uint64_t ms) + { + std::lock_guard lock(mutex); + /// Plus one millisecond, because full jitter can draw a ZERO pause: a clock that does not move + /// would leave the loop reissuing for ever against a fault that never clears. + now_ms += ms + 1; + ++pauses; + longest_pause = std::max(longest_pause, ms); + } + +private: + mutable std::mutex mutex; + uint64_t now_ms = 0; + size_t pauses = 0; + uint64_t longest_pause = 0; +}; + +/// `ChunkFaultBackend` COUNTS its faults, and a count can no longer make one conclusive: the write +/// engine settles every ambiguity by an exact read and then REISSUES, so a fault that runs out +/// mid-call is answered by the next attempt instead of by the call's own deadline -- which is the +/// whole difference between a wedge and a commit. +class LatchedChunkFaultBackend : public DB::Cas::tests::ChunkFaultBackend { public: - using CountingBackend::get; - using CountingBackend::getStream; - using CountingBackend::putIfAbsent; - using CountingBackend::putOverwrite; - using CountingBackend::casPut; + bool latched = false; - /// Set before the driving call; consumed by the first matching recovery GET. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + if (latched && mode != Mode::None && fault_skip == 0 && !expected_value && !fault_substr.empty() + && key.find(fault_substr) != String::npos) + fault_count = 1; + return ChunkFaultBackend::write(key, bytes, expected_value, access); + } + + void disarm() + { + latched = false; + mode = Mode::None; + fault_count = 0; + fault_skip = 0; + fail_read_once_key.clear(); + } +}; + +class RecoveryLatchBackend : public CountingBackend +{ +public: + /// Set before the driving call; consumed by the first matching recovery read. String fail_get_once_key; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (!fail_get_once_key.empty() && key == fail_get_once_key) { fail_get_once_key.clear(); throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, - "RecoveryLatchBackend: simulated non-transient exact GET failure"); + "RecoveryLatchBackend: simulated non-transient exact read failure"); } { std::unique_lock lk(m); @@ -95,7 +187,7 @@ class RecoveryLatchBackend : public CountingBackend cv.wait_for(lk, std::chrono::seconds(20), [&] { return block_key.empty(); }); } } - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); } void armBlockedGet(const String & key) @@ -174,16 +266,17 @@ struct CaseSync bool entered = false; }; -/// `num_pairs` add-then-remove precommit op pairs for distinct refs, each naming a distinct manifest. -/// Every pair is undone immediately, so the LIVE state stays ~empty and validating thousands of ops -/// stays linear -- it is the OP COUNT, not the resident state, that drives the chunk split under test. -std::vector precommitAddRemovePairs(const String & prefix, size_t num_pairs, uint64_t manifest_epoch) +/// `num_pairs` add-then-remove precommit op pairs on ONE ref, each pair naming a distinct manifest, +/// so an item scoped `MutationScope::ref(ref)` names exactly the ref its ops mutate (the flush +/// validates that). Every pair is undone immediately, so the LIVE state stays ~empty and validating +/// thousands of ops stays linear -- it is the OP COUNT, not the resident state, that drives the chunk +/// split under test. +std::vector precommitAddRemovePairs(const String & ref, size_t num_pairs, uint64_t manifest_epoch) { std::vector ops; ops.reserve(num_pairs * 2); for (size_t i = 0; i < num_pairs; ++i) { - const String ref = prefix + std::to_string(i); const ManifestRef manifest{manifest_epoch, i + 1, 1}; RefOp add; add.kind = RefOpKind::OwnerTransition; @@ -226,11 +319,25 @@ ManifestId publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const S return id; } -/// Every request class `CountingBackend` observes, summed. The zero-I/O contract is asserted against -/// this total, so a confirm that quietly grew a HEAD or a GET fails the test rather than the review. +/// Every request `CountingBackend` observes, summed: reads, heads, stream opens, writes, lists, +/// deletes and publications. The zero-I/O contract is asserted against this total, so a confirm that +/// quietly grew any one of them fails the test rather than the review. `writeTotal` and not `putTotal`: +/// a write that carried a precondition is still a write, and counting only the create-shaped ones left +/// the replace path unwatched. uint64_t backendRequests(const CountingBackend & b) { - return b.headTotal() + b.getTotal() + b.getStreamTotal() + b.putTotal() + b.listTotal(); + return b.headTotal() + b.getTotal() + b.getStreamTotal() + b.writeTotal() + b.listTotal() + + b.deleteTotal() + b.publishTotal(); +} + +/// One refusal counter's current value. `confirmExactRef` attributes every `Unknown` to exactly one of +/// these, and a live gate reads them to tell load from a fault from a lost mount -- distinctions the +/// three-value `ConfirmAnswer` cannot carry. A test that checks only the ANSWER passes just as happily +/// when two of them are swapped, so the tests that reach a refusal deterministically assert the +/// attribution as a DELTA around the confirm, never an absolute (the suite shares one process). +uint64_t refusalCount(ProfileEvents::Event event) +{ + return ProfileEvents::global_counters[event].load(); } /// One-shot throwing probe in the post-durable install region -- the only way to reach `NeedsRecovery` @@ -394,7 +501,15 @@ TEST(CASConfirmExactRef, UnrecoveredResidentTableIsUnknownWithZeroBackendRequest ASSERT_FALSE(store->refTableCachedForTest(ns)) << "that runtime must be UNRECOVERED"; backend->resetCounts(); + /// An unrecovered table is not a lane fault and not load: it is this mount being unable to speak + /// for the namespace, and the live gate must be able to tell those apart. + const uint64_t cannot_speak_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak); + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); EXPECT_EQ(store->confirmExactRef(ns, "x", ManifestRef{1, 1, 1}), ConfirmAnswer::Unknown); + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak) - cannot_speak_before, 1u) + << "an unrecovered view must be reported as this mount being unable to speak for the table"; + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken) - broken_before, 0u) + << "an unrecovered table is not a lane defect; the live gate asserts LaneBroken is zero"; EXPECT_EQ(backendRequests(*backend), 0u) << "an unrecovered table must answer Unknown without driving recovery"; EXPECT_FALSE(store->refTableCachedForTest(ns)) @@ -467,8 +582,8 @@ TEST(CASConfirmExactRef, RecoveryInProgressIsUnknownWithZeroBackendRequests) /// Rule 3, the in-flight case: an append is admitted and its leader is parked in the pre-carve window. /// Nothing is durable yet and the committed row still matches EXACTLY -- which is precisely why a /// naive implementation answers `Yes` here, and precisely why that is the TOCTOU this design closes. -/// The apply-state is still `Clean` at this point, so rule 3 is what produces the `Unknown`, not -/// rule 4. +/// The lane state is still `Ready` and the item is in `pending`, so rule 3 reading the item's scope is +/// what produces the `Unknown`. TEST(CASConfirmExactRef, InFlightAppendIsUnknown) { auto backend = std::make_shared(); @@ -498,18 +613,20 @@ TEST(CASConfirmExactRef, InFlightAppendIsUnknown) EXPECT_EQ(apply_state, RefLaneState::Ready) << "the pre-carve window is before any PUT, so rule 4 must not be what answers here"; EXPECT_EQ(while_in_flight, ConfirmAnswer::Unknown) - << "an admitted append makes the whole table's committed view provisional"; + << "an admitted mutation of THIS ref makes its committed row provisional"; EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::No); } -/// Rule 3, mid-tenure (spec §testing "mid-tenure chunked flush", `CarvePhaseForTest::ChunkReseed`): -/// one leader tenure commits MULTIPLE durable transactions, so at a chunk boundary the table is -/// PARTIALLY durable. `leader_active` covers the whole tenure, so this is already `Unknown` -- a wider -/// unknown window under load, never a hole. The confirm is issued on the leader's own thread, which is -/// safe because the boundary holds neither lane mutex. -TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) +/// Rule 3 at a chunk boundary (`CarvePhaseForTest::ChunkReseed`): one leader tenure commits MULTIPLE +/// durable transactions, so between two chunks the table is PARTIALLY durable -- for the refs those +/// chunks mutate. The seed ref is touched by neither, so its row is exactly as authoritative as on an +/// idle lane and it confirms; BOTH carved items' own refs refuse, because their transactions may be +/// durable and not installed -- the second one is what makes a rule that read only the front of the +/// mirror visible. The confirm is issued on the leader's own thread, which is safe because the +/// boundary holds neither lane mutex. +TEST(CASConfirmExactRef, UntouchedRefConfirmsMidTenure) { auto backend = std::make_shared(); auto store = openPool(backend); @@ -519,7 +636,9 @@ TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) ASSERT_EQ(store->confirmExactRef(ns, "seed", id.ref), ConfirmAnswer::Yes); std::atomic boundaries{0}; - std::atomic unknown_at_boundary{0}; + std::atomic yes_for_seed_at_boundary{0}; + std::atomic unknown_for_carved_ref_at_boundary{0}; + std::atomic unknown_for_second_carved_ref_at_boundary{0}; std::atomic requests_at_boundary{0}; store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) { @@ -527,8 +646,17 @@ TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) return; boundaries.fetch_add(1); const uint64_t before = backendRequests(*backend); - if (store->confirmExactRef(ns, "seed", id.ref) == ConfirmAnswer::Unknown) - unknown_at_boundary.fetch_add(1); + if (store->confirmExactRef(ns, "seed", id.ref) == ConfirmAnswer::Yes) + yes_for_seed_at_boundary.fetch_add(1); + /// "aaa_" has no committed row (rule 5 would say `No`), so an `Unknown` here can only come from + /// rule 3 reading the carved item's scope. + if (store->confirmExactRef(ns, "aaa_", id.ref) == ConfirmAnswer::Unknown) + unknown_for_carved_ref_at_boundary.fetch_add(1); + /// "bbb_" is the mirror's SECOND entry and likewise has no committed row, so this is the same + /// assertion made about an entry a rule that stopped at the front of `carved` would never + /// reach. + if (store->confirmExactRef(ns, "bbb_", id.ref) == ConfirmAnswer::Unknown) + unknown_for_second_carved_ref_at_boundary.fetch_add(1); requests_at_boundary.fetch_add(static_cast(backendRequests(*backend) - before)); }); @@ -549,10 +677,10 @@ TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) [&] { return store->refQueuePendingForTest(ns) >= 2; }); }); - auto append = [&store, &ns](const String & prefix, uint64_t manifest_epoch) + auto append = [&store, &ns](const String & ref, uint64_t manifest_epoch) { - std::vector item_ops = precommitAddRemovePairs(prefix, 1500, manifest_epoch); - store->appendRefOps(ns, MutationScope::ref(prefix), + std::vector item_ops = precommitAddRemovePairs(ref, 1500, manifest_epoch); + store->appendRefOps(ns, MutationScope::ref(ref), [ops = std::move(item_ops)](const RefTableState &) { return ops; }, RootMutationOrigin::Writer, RootMutationKind::Publish); }; @@ -574,11 +702,16 @@ TEST(CASConfirmExactRef, MidTenureChunkBoundaryIsUnknown) store->setCarveHookForTest(nullptr); ASSERT_GE(boundaries.load(), 1) << "the flush did not chunk -- the mid-tenure window was not exercised"; - EXPECT_EQ(unknown_at_boundary.load(), boundaries.load()) - << "a mid-tenure, partially-durable table must never confirm"; + EXPECT_EQ(yes_for_seed_at_boundary.load(), boundaries.load()) + << "a ref no carved item names must confirm mid-tenure -- that is the liveness this rule exists for"; + EXPECT_EQ(unknown_for_carved_ref_at_boundary.load(), boundaries.load()) + << "a ref a carved item names must not confirm while its transaction may be durable and not installed"; + EXPECT_EQ(unknown_for_second_carved_ref_at_boundary.load(), boundaries.load()) + << "rule 3 must scan the whole carved mirror: 'bbb_' is its second entry and has no committed " + "row, so a rule that examined only the front entry would answer No here"; EXPECT_EQ(requests_at_boundary.load(), 0) << "the mid-tenure confirm must still be I/O-free"; - /// The tenure is over: the seed ref is untouched by it and confirms again. + /// The tenure is over: the seed ref confirms as before. EXPECT_EQ(store->confirmExactRef(ns, "seed", id.ref), ConfirmAnswer::Yes); } @@ -607,6 +740,264 @@ TEST(CASConfirmExactRef, WedgedLaneIsUnknown) } +/// Rule 3, a REAL wedge: the removal of x is sent, the response is lost, the single-attempt budget is +/// exhausted, and the lane wedges. `commitRefChunk` completes the chunk's items with an error before +/// the tenure ends, so the transaction that may be durable is recorded nowhere but in the attempt and +/// the lane state -- `pending` and `carved` are both empty. Every ref refuses: x because its removal +/// may be durable, `other` because nothing but the lane state records WHICH ref the wedged transaction +/// touched. +TEST(CASConfirmExactRef, WedgedTransactionRefusesEveryRef) +{ + auto backend = std::make_shared(); + PoolConfig cfg; + /// The budget bounds the mount lease's own admission arithmetic and nothing else: a write's attempt + /// count is the `Retry` policy's. What makes the injected fault conclusive is that it stays armed + /// for the whole call while the injected clock below carries the call to its own deadline. + CasRequestBudget budget; + budget.attempt_timeout_ms = 100; + budget.lease_safety_margin_ms = 100; + cfg.cas_request_budget = budget; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); + auto store = openPoolWithConfig(backend, cfg); + auto clock = VirtualRetryClock::installOn(store); + const RootNamespace ns{"srv1/confirm_real_wedge"}; + /// Pins the namespace to the fixture life BEFORE its first real touch, so the fault key computed + /// from that same life below is the key production actually writes to. + DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns); + + const ManifestId id_x = publishEmptyPart(store, ns, "x"); + const ManifestId id_other = publishEmptyPart(store, ns, "other"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Yes); + ASSERT_EQ(store->confirmExactRef(ns, "other", id_other.ref), ConfirmAnswer::Yes); + + backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; + backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; + backend->fault_skip = 0; + backend->fault_count = 1; + backend->latched = true; + EXPECT_THROW(store->dropRef(ns, "x"), DB::Exception); + backend->disarm(); + /// The give-up was the call's OWN retry window: the fault outlasted several reissues and every one + /// of them paced through the injected sleep rather than a real one. + EXPECT_GT(clock->pauseCount(), 1u); + EXPECT_LE(clock->longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; + EXPECT_GE(clock->nowMs(), 60000u); + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_EQ(store->refQueuePendingForTest(ns), 0u); + ASSERT_EQ(store->refCarvedForTest(ns), 0u) + << "the wedged item was completed and released; only the lane state records its transaction"; + + backend->resetCounts(); + /// A wedge and a broken lane are two DIFFERENT things to a live gate -- one is an unresolved append + /// that the next flush or a remount clears, the other is a lane defect. Both refuse here, and + /// without these deltas the test would pass just as happily if the wedge branch were deleted and + /// the broken-lane branch answered for it. + const uint64_t wedged_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneWedged); + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + EXPECT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Unknown) << "x's removal may be durable"; + EXPECT_EQ(store->confirmExactRef(ns, "other", id_other.ref), ConfirmAnswer::Unknown) + << "a wedge refuses table-wide: no per-ref record of the wedged transaction survives the tenure"; + EXPECT_EQ(backendRequests(*backend), 0u) << "a wedged lane must answer without trying to resolve the wedge"; + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneWedged) - wedged_before, 2u) + << "both refusals must be reported as a wedge"; + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken) - broken_before, 0u) + << "a wedge is not a lane defect: reporting it as one would send the live gate hunting a bug"; +} + + +/// `carved` bookkeeping: a carved item leaves `pending` at the carve and is completed by its chunk's +/// install (or earlier, by an error), while the mirror is cleared only at the tenure's exit guard, so +/// the confirm reads the item from `rt.carved` from carve to tenure end. Sampled at +/// `PostDurableInstall` -- the transaction is durable, nothing is installed, `pending` is already +/// empty -- and again after the tenure: the mirror must hold exactly the carved item during, and be +/// empty after. The hook runs on the leader's own thread with neither lane mutex held, so the seams +/// (which take `ref_queue_mutex`) are safe to call from it. +TEST(CASConfirmExactRef, CarvedItemIsVisibleFromCarveToTenureEnd) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_carved"}; + publishEmptyPart(store, ns, "x"); + + std::atomic samples{0}; + std::atomic carved_during{0}; + std::atomic pending_during{0}; + store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::PostDurableInstall) + return; + samples.fetch_add(1); + carved_during.store(store->refCarvedForTest(ns)); + pending_during.store(store->refQueuePendingForTest(ns)); + }); + store->dropRef(ns, "x"); + store->setCarveHookForTest(nullptr); + + ASSERT_EQ(samples.load(), 1) << "the drop must commit exactly one chunk"; + EXPECT_EQ(carved_during.load(), 1u) + << "the carved removal must be visible while its transaction is durable but not installed"; + EXPECT_EQ(pending_during.load(), 0u) + << "the carve popped the item out of pending -- carved is the only place it can be seen"; + EXPECT_EQ(store->refCarvedForTest(ns), 0u) << "the exit guard must release the mirror"; +} + +/// Scope validation: `MutationScope` is what the confirm reads to decide whether an in-flight mutation +/// may change the ref it is asked about, so an item scoped to ref X must fail, alone, before anything +/// is durable, both when its ops mutate ref Y and when they carry a namespace removal, which names no +/// ref and moves every row. It throws `LOGICAL_ERROR`, which aborts the process in debug and +/// sanitizer builds instead of behaving like a catchable exception -- +/// `CASConfirmExactRefDeathTest.MisScopedItemAborts` below proves the abort positively in those builds. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASConfirmExactRef, MisScopedItemFailsBeforeAnythingIsDurable) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_misscoped"}; + const ManifestId seed = publishEmptyPart(store, ns, "seed"); /// the namespace is born already + + const uint64_t writes_before = backend->writeTotal(); + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "y", ManifestRef{900000003, 1, 1}}; + try + { + store->appendRefOps(ns, MutationScope::ref("x"), + [add](const RefTableState &) { return std::vector{add}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + FAIL() << "an item scoped to ref 'x' whose op binds ref 'y' must be refused"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); + } + /// A namespace removal names no ref at all, so a scope check that only compared names would let it + /// through -- and it moves every row, which is the one thing a `Ref` scope promises the confirm + /// will not happen behind its back. + RefOp remove_namespace; + remove_namespace.kind = RefOpKind::RemoveNamespace; + try + { + store->appendRefOps(ns, MutationScope::ref("x"), + [remove_namespace](const RefTableState &) { return std::vector{remove_namespace}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + FAIL() << "an item scoped to ref 'x' carrying a namespace removal must be refused"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR) + << "the scope check must be what rejects it"; + } + + /// The ref-log transaction object -- the only thing that would make this item durable -- is a + /// create, and this counts every write the backend can observe rather than that one shape, so the + /// fence still holds if the durable step ever changes shape. + EXPECT_EQ(backend->writeTotal(), writes_before) << "the refusal must happen before any object is written"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready) << "a validation failure is not a lane fault"; + EXPECT_EQ(store->confirmExactRef(ns, "seed", seed.ref), ConfirmAnswer::Yes) + << "the failed item must leave the table exactly as it was"; +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASConfirmExactRefDeathTest, MisScopedItemAborts) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_misscoped"}; + publishEmptyPart(store, ns, "seed"); + + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "y", ManifestRef{900000003, 1, 1}}; + EXPECT_DEATH({ + store->appendRefOps(ns, MutationScope::ref("x"), + [add](const RefTableState &) { return std::vector{add}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }, ""); + + RefOp remove_namespace; + remove_namespace.kind = RefOpKind::RemoveNamespace; + EXPECT_DEATH({ + store->appendRefOps(ns, MutationScope::ref("x"), + [remove_namespace](const RefTableState &) { return std::vector{remove_namespace}; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }, ""); +} +#endif + +/// The mirror must survive an item's COMPLETION, not just its carve: an item is completed by its own +/// chunk's commit, often chunks before the tenure ends. Two items whose op counts force a chunk split +/// (mirrors `UntouchedRefConfirmsMidTenure`'s co-batching) are carved together in one tenure: chunk 1 +/// = {aaa_} alone, chunk 2 = {bbb_} alone. Sampled at chunk 2's `PostDurableInstall`, the test PROVES -- +/// via `refCarvedItemDoneForTest`, not by inferring from hook order -- that aaa_ is already done, and +/// that it is still counted in `carved` alongside bbb_ until the tenure's exit guard. +TEST(CASConfirmExactRef, CarvedItemSurvivesEarlierChunkCompletion) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_carved_multi_chunk"}; + publishEmptyPart(store, ns, "seed"); + + std::atomic boundaries{0}; + std::atomic aaa_done_at_second_boundary{false}; + std::atomic carved_at_second_boundary{0}; + store->setCarveHookForTest([&](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::PostDurableInstall) + return; + if (boundaries.fetch_add(1) + 1 != 2) + return; /// only chunk 2's durable point is of interest + aaa_done_at_second_boundary.store(store->refCarvedItemDoneForTest(ns, "aaa_")); + carved_at_second_boundary.store(store->refCarvedForTest(ns)); + }); + + /// Co-batching setup identical to `UntouchedRefConfirmsMidTenure`: the pre-carve hook parks the + /// first caller until the second is queued, so both items are carved together deterministically. + auto sync = std::make_shared(); + store->setRefPreCarveHookForTest([sync, store, ns] + { + std::unique_lock lk(sync->m); + if (sync->entered) + return; + sync->entered = true; + sync->cv.notify_all(); + sync->cv.wait_for(lk, std::chrono::seconds(20), + [&] { return store->refQueuePendingForTest(ns) >= 2; }); + }); + + auto append = [&store, &ns](const String & ref, uint64_t manifest_epoch) + { + std::vector item_ops = precommitAddRemovePairs(ref, 1500, manifest_epoch); + store->appendRefOps(ns, MutationScope::ref(ref), + [ops = std::move(item_ops)](const RefTableState &) { return ops; }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }; + std::thread a([&] { append("aaa_", 900000001); }); + { + std::unique_lock lk(sync->m); + sync->cv.wait_for(lk, std::chrono::seconds(20), [&] { return sync->entered; }); + } + std::thread b([&] { append("bbb_", 900000002); }); + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(20); + while (store->refQueuePendingForTest(ns) < 2 && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); + sync->cv.notify_all(); + a.join(); + b.join(); + store->setRefPreCarveHookForTest(nullptr); + store->setCarveHookForTest(nullptr); + + ASSERT_EQ(boundaries.load(), 2) << "the flush must chunk into exactly two transactions"; + ASSERT_TRUE(aaa_done_at_second_boundary.load()) + << "chunk 1's item must already be done by the time chunk 2 goes durable"; + EXPECT_EQ(carved_at_second_boundary.load(), 2u) + << "a completed item must still be counted in the mirror until the tenure's exit guard"; + EXPECT_EQ(store->refCarvedForTest(ns), 0u) << "the exit guard must release the mirror after both chunks"; +} + + /// `NeedsRecovery` is table-scoped, so confirmation refuses even a row that still looks perfect. TEST(CASConfirmExactRef, NeedsRecoveryIsUnknown) { @@ -625,8 +1016,16 @@ TEST(CASConfirmExactRef, NeedsRecoveryIsUnknown) ASSERT_FALSE(store->refLaneWedgedForTest(ns)); ASSERT_FALSE(store->refLeaderActiveForTest(ns)); + /// The positive side of the wedge case's negative: `NeedsRecovery` is the lane defect + /// `LaneBroken` is FOR, so this is the one refusal that must be reported as one. + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + const uint64_t wedged_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneWedged); EXPECT_EQ(store->confirmExactRef(ns, "keep", keep.ref), ConfirmAnswer::Unknown) << "a table that may be missing a durable transaction cannot confirm ANY of its rows"; + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken) - broken_before, 1u) + << "a NeedsRecovery lane is exactly what LaneBroken reports"; + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneWedged) - wedged_before, 0u) + << "the lane is not wedged here, and the two counters must not be interchangeable"; } @@ -647,7 +1046,12 @@ TEST(CASConfirmExactRef, LostMountFenceIsUnknown) ASSERT_TRUE(store->refTableCachedForTest(ns)) << "the table must still be resident, so it is the FENCE that refuses, not residency"; backend->resetCounts(); + /// The other arm of the same counter: rule 6's fence check. Losing the mount is the most + /// safety-relevant refusal this function has, so it must not be the silent one. + const uint64_t cannot_speak_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak); EXPECT_EQ(store->confirmExactRef(ns, "x", id.ref), ConfirmAnswer::Unknown); + EXPECT_EQ(refusalCount(ProfileEvents::CASRelinkConfirmRefusedMountCannotSpeak) - cannot_speak_before, 1u) + << "a refusal for a lost mount fence must be counted, not invisible"; EXPECT_EQ(backendRequests(*backend), 0u); /// The fence is checked LAST, so it gates only the `Yes`: a token that does not match the committed @@ -709,6 +1113,225 @@ TEST(CASConfirmExactRef, ConcurrentAppendIsOrderedAfterTheSnapshot) } +/// The livelock shape: a mutation of ANOTHER ref is queued and its leader is parked before the carve, +/// so the lane has a pending item and an active tenure. A confirm about an untouched committed ref must +/// answer `Yes` -- the queued mutation cannot change this ref's binding or the blobs its manifest +/// protects -- while the queued ref itself answers `Unknown`. +/// +/// A SECOND queued mutation, of a third ref, sits behind the leader's own item, so the queue holds two +/// and the ref asked about last is not at its front: that is what makes a rule reading only +/// `pending`'s front item visible, which every other test in this file would pass. +TEST(CASConfirmExactRef, UntouchedRefConfirmsWhileAnotherRefIsQueued) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_liveness"}; + + const ManifestId id_x = publishEmptyPart(store, ns, "x"); + const ManifestId id_other = publishEmptyPart(store, ns, "other"); + const ManifestId id_third = publishEmptyPart(store, ns, "third"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Yes); + + LeaderLatch latch; + latch.arm(store); + std::thread dropper([&] { store->dropRef(ns, "other"); }); + latch.awaitEntered(); + /// The leader pushes its own item before it takes the baton, so the queue is [other, third] and + /// `third` is reachable only by a scan that goes past the front. + std::thread second_dropper([&] { store->dropRef(ns, "third"); }); + const auto queued_by = std::chrono::steady_clock::now() + std::chrono::seconds(20); + while (store->refQueuePendingForTest(ns) < 2 && std::chrono::steady_clock::now() < queued_by) + std::this_thread::yield(); + + /// Sampled while parked, asserted after the join (a failed assertion here would skip the release). + const bool leader_active = store->refLeaderActiveForTest(ns); + const size_t pending = store->refQueuePendingForTest(ns); + /// The refusal counters are the only way a live gate can read WHY a confirm said `Unknown`, so the + /// attribution is pinned here rather than left to the reader of the .cpp: this pair of confirms is + /// the one place where a `Yes` and a scope-driven `Unknown` are produced back to back from the same + /// lane state. + const uint64_t in_flight_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + const ConfirmAnswer untouched = store->confirmExactRef(ns, "x", id_x.ref); + const uint64_t in_flight_after_untouched + = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const ConfirmAnswer touched = store->confirmExactRef(ns, "other", id_other.ref); + const uint64_t in_flight_after_touched + = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const ConfirmAnswer touched_behind_the_front = store->confirmExactRef(ns, "third", id_third.ref); + const uint64_t in_flight_after_behind_the_front + = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const uint64_t broken_after = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + + latch.release(); + dropper.join(); + second_dropper.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_TRUE(leader_active); + EXPECT_EQ(pending, 2u) << "the second dropper must be queued behind the parked leader's own item"; + EXPECT_EQ(in_flight_after_untouched - in_flight_before, 0u) + << "a confirm that answers Yes must not be counted as a refusal"; + EXPECT_EQ(in_flight_after_touched - in_flight_after_untouched, 1u) + << "the refused confirm must be attributed to the ref-scoped mutation, which is what a live " + "gate reads to tell load from a lane fault"; + EXPECT_EQ(in_flight_after_behind_the_front - in_flight_after_touched, 1u) + << "the second queued item's ref must be refused for the same reason as the first"; + EXPECT_EQ(broken_after - broken_before, 0u) + << "the lane is Ready here: a refusal attributed to a broken lane would misreport a fault"; + EXPECT_EQ(untouched, ConfirmAnswer::Yes) + << "a queued mutation of another ref must not refuse this one"; + EXPECT_EQ(touched, ConfirmAnswer::Unknown) + << "the queued ref's own row is provisional"; + EXPECT_EQ(touched_behind_the_front, ConfirmAnswer::Unknown) + << "rule 3 must scan the whole pending queue, not only its front item"; + EXPECT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Yes); + EXPECT_EQ(store->confirmExactRef(ns, "other", id_other.ref), ConfirmAnswer::No); + EXPECT_EQ(store->confirmExactRef(ns, "third", id_third.ref), ConfirmAnswer::No); +} + + +/// Rule 3's `WholeShard` arm. A mutation that declares no ref is one that may move EVERY row, so it +/// refuses every ref for as long as it is queued or carved -- unlike a `Ref`-scoped neighbour, which +/// refuses only its own. In production `dropNamespaceImpl` and `sweepStalePrecommitsNow` are the two +/// appenders that declare `wholeShard()`. +/// +/// The two confirms of "x" differ in exactly one thing: whether the whole-shard item has been queued. +/// The first is the liveness answer this rule was narrowed to give, the second the refusal the arm +/// exists for, so deleting the arm turns the second into the first. +TEST(CASConfirmExactRef, WholeShardScopedMutationRefusesAnUntouchedRef) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_whole_shard"}; + + const ManifestId id_x = publishEmptyPart(store, ns, "x"); + publishEmptyPart(store, ns, "other"); + ASSERT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Yes); + + LeaderLatch latch; + latch.arm(store); + std::thread dropper([&] { store->dropRef(ns, "other"); }); + latch.awaitEntered(); + + /// Sampled while parked, asserted after the join (a failed assertion here would skip the release). + const ConfirmAnswer before_whole_shard = store->confirmExactRef(ns, "x", id_x.ref); + + /// Queued BEHIND the parked leader's own `Ref`-scoped item. The ops are an add/remove precommit + /// pair on a ref of its own, so the item is ordinary work that happens to declare no ref -- the + /// scope, not the ops, is what rule 3 reads. + std::thread whole_shard([&] + { + store->appendRefOps(ns, MutationScope::wholeShard(), + [](const RefTableState &) { return precommitAddRemovePairs("zzz_", 1, 900000004); }, + RootMutationOrigin::Writer, RootMutationKind::Publish); + }); + const auto queued_by = std::chrono::steady_clock::now() + std::chrono::seconds(20); + while (store->refQueuePendingForTest(ns) < 2 && std::chrono::steady_clock::now() < queued_by) + std::this_thread::yield(); + const size_t pending = store->refQueuePendingForTest(ns); + + const uint64_t in_flight_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + const ConfirmAnswer with_whole_shard = store->confirmExactRef(ns, "x", id_x.ref); + const uint64_t in_flight_after = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const uint64_t broken_after = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + + latch.release(); + dropper.join(); + whole_shard.join(); + store->setRefPreCarveHookForTest(nullptr); + + EXPECT_EQ(pending, 2u) << "the whole-shard item must be queued behind the parked leader's own item"; + EXPECT_EQ(before_whole_shard, ConfirmAnswer::Yes) + << "with only a Ref-scoped mutation of another ref queued, 'x' must still confirm"; + EXPECT_EQ(with_whole_shard, ConfirmAnswer::Unknown) + << "a queued mutation that declares no ref may move every row, so no ref may confirm"; + EXPECT_EQ(in_flight_after - in_flight_before, 1u) + << "the refusal must be attributed to an in-flight mutation, not to a lane or mount condition"; + EXPECT_EQ(broken_after - broken_before, 0u) + << "the lane is Ready here: a refusal attributed to a broken lane would misreport a fault"; + + /// The tenure is over and the whole-shard item is applied: 'x' confirms again. + EXPECT_EQ(store->confirmExactRef(ns, "x", id_x.ref), ConfirmAnswer::Yes); +} + + +/// The stale-row hazard rule 3 exists for, on the same ref: a repoint of x from m1 to m2 is DURABLE +/// and NOT installed, so the committed row still says m1. A `Yes` here would let a receiver promote a +/// manifest whose blobs the durable repoint may already have retired. The leader is parked at the +/// SECOND `PostDurableInstall` of the repointing publish (the first is its precommit), with no lane +/// mutex held; `pending` is already empty, so only the carved mirror can refuse. +TEST(CASConfirmExactRef, SameRefRepointDurableButNotInstalledIsUnknown) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/confirm_stale_row"}; + const ManifestId m1 = publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->confirmExactRef(ns, "x", m1.ref), ConfirmAnswer::Yes); + + struct Hold + { + std::mutex m; + std::condition_variable cv; + int seen = 0; + bool parked = false; + bool released = false; + }; + auto hold = std::make_shared(); + store->setCarveHookForTest([hold](CasRefLedger::CarvePhaseForTest phase) + { + if (phase != CasRefLedger::CarvePhaseForTest::PostDurableInstall) + return; + std::unique_lock lk(hold->m); + if (++hold->seen != 2) + return; /// 1 = the precommit's chunk; 2 = the promote (the repoint) -- park here + hold->parked = true; + hold->cv.notify_all(); + hold->cv.wait_for(lk, std::chrono::seconds(20), [&] { return hold->released; }); + }); + + ManifestId m2; + std::thread repointer([&] { m2 = publishEmptyPart(store, ns, "x", /*allow_repoint=*/true); }); + bool parked = false; + { + std::unique_lock lk(hold->m); + parked = hold->cv.wait_for(lk, std::chrono::seconds(20), [&] { return hold->parked; }); + } + const size_t pending_now = store->refQueuePendingForTest(ns); + const size_t carved_now = store->refCarvedForTest(ns); + const RefLaneState lane_now = store->laneStateForTest(ns); + /// `pending` is empty and the mirror holds one, so this refusal provably comes from the carved + /// loop -- the only place where its attribution can be pinned. + const uint64_t in_flight_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight); + const uint64_t broken_before = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken); + const ConfirmAnswer stale = store->confirmExactRef(ns, "x", m1.ref); + const uint64_t in_flight_delta + = refusalCount(ProfileEvents::CASRelinkConfirmRefusedRefMutationInFlight) - in_flight_before; + const uint64_t broken_delta = refusalCount(ProfileEvents::CASRelinkConfirmRefusedLaneBroken) - broken_before; + { + std::lock_guard lk(hold->m); + hold->released = true; + } + hold->cv.notify_all(); + repointer.join(); + store->setCarveHookForTest(nullptr); + + ASSERT_TRUE(parked) << "the repoint never reached its post-durable window"; + EXPECT_EQ(pending_now, 0u) << "the repoint was carved: pending cannot be what refuses"; + EXPECT_EQ(carved_now, 1u) << "the carved mirror is what the confirm must read"; + EXPECT_EQ(lane_now, RefLaneState::Writing); + EXPECT_EQ(stale, ConfirmAnswer::Unknown) + << "x's durable repoint is not installed: its row is stale and must not confirm m1"; + EXPECT_EQ(in_flight_delta, 1u) + << "a refusal read off the carved mirror is a mutation in flight, not a lane fault"; + EXPECT_EQ(broken_delta, 0u) + << "the lane is Writing with a carved item -- the healthy shape, not a broken one"; + EXPECT_EQ(store->confirmExactRef(ns, "x", m1.ref), ConfirmAnswer::No) << "installed: m1 is no longer x's binding"; + EXPECT_EQ(store->confirmExactRef(ns, "x", m2.ref), ConfirmAnswer::Yes); +} + + /// =========================================================================================== /// Task 11: the EXCHANGE-level confirm -- `IContentAddressedExchange::ownsNamespace` (routing) and /// `::confirmExactRef` (the storage forward of gate 1, plus the token text and the disk lifecycle). diff --git a/src/Disks/tests/gtest_cas_decommission.cpp b/src/Disks/tests/gtest_cas_decommission.cpp index 620dea710a3f..d590eaed7a59 100644 --- a/src/Disks/tests/gtest_cas_decommission.cpp +++ b/src/Disks/tests/gtest_cas_decommission.cpp @@ -1,5 +1,6 @@ #include "cas_test_helpers.h" #include +#include #include #include #include @@ -39,33 +40,72 @@ void drainCompletedNamespaceRemovals(const std::shared_ptr & ba ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); } -/// Fails `deleteExact` for one or two designated keys -- either by throwing (a transient backend -/// hiccup) or by returning a synthetic `TokenMismatch` (a "listed but raced" outcome) -- delegating -/// every other key to the base `InMemoryBackend` untouched. Drives the drain phases' per-object -/// fail-close path (`deleteListedPrefix`/`sweepNamespace`, `CasDecommission.cpp`/ -/// `CasOrphanManifestSweep.cpp`): a failure on one listed object must record a warning and let the rest -/// of the sweep proceed, never abort the whole phase. +/// Fails a delete for one or two designated keys -- either by throwing (a transient backend hiccup) or +/// by returning a synthetic `Mismatch` (a "listed but raced" outcome) -- delegating every other key to +/// the base `InMemoryBackend` untouched. Drives the drain phases' per-object fail-close path +/// (`deleteListedPrefix`/`sweepNamespace`, `CasDecommission.cpp`/`CasOrphanManifestSweep.cpp`): a +/// failure on one listed object must record a warning and let the rest of the sweep proceed, never +/// abort the whole phase. Injects on the `remove` PRIMITIVE, not the legacy `deleteExact`, so it +/// intercepts a caller on either surface. /// +/// The thrown fault is a `Poco::TimeoutException`, the class the request engine classifies as a +/// transport failure and reissues (`Retry::standard()`); a `std::runtime_error` propagates immediately +/// and never exercises the retry path this test's own name claims to drive. `latch` keeps it armed +/// across every reissue of the SAME logical call, so the engine reaches its own retry deadline and +/// gives up rather than recovering on a later attempt -- a one-shot throw would be outlived by the +/// reissue and the delete would simply succeed. class FailingDeleteBackend : public InMemoryBackend { public: void failWithThrow(const String & key) { throw_key = key; } void failWithTokenMismatch(const String & key) { mismatch_key = key; } + void latch() { latched = true; } /// Clears every injected failure -- the resume half of a fail-then-retry test (Task 4). - void disarm() { throw_key.clear(); mismatch_key.clear(); } + void disarm() { throw_key.clear(); mismatch_key.clear(); latched = false; } - DeleteOutcome deleteExact(const String & key, const Token & token) override + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { if (key == throw_key) - throw std::runtime_error("injected transient delete failure for " + key); + { + if (!latched) + throw_key.clear(); + throw Poco::TimeoutException("injected transient delete failure for " + key); + } if (key == mismatch_key) - return DeleteOutcome{.kind = DeleteOutcome::Kind::TokenMismatch}; - return InMemoryBackend::deleteExact(key, token); + return RawRemoval::Mismatch; + return InMemoryBackend::remove(key, expected_value, access); } private: String throw_key; String mismatch_key; + bool latched = false; +}; + +/// Fails a read of one designated key a fixed number of times with a transient `Poco::TimeoutException` +/// -- the class the request engine classifies as a transport failure and reissues -- before delegating +/// to the base `InMemoryBackend`. Drives the owner-object read `Pool::openForDecommission` itself issues +/// on the OPEN plane (`claimOwnerOrThrow` -> `readOwnerObject`, CasServerRoot.cpp) through a bounded +/// number of paced retries DURING opening, before `decommissionPoolMember` gets a chance to install +/// anything on the already-open `Pool`. +class FlakyReadBackend : public InMemoryBackend +{ +public: + void failReadNTimes(const String & key, int times) { flaky_key = key; remaining = times; } + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (key == flaky_key && remaining > 0) + { + --remaining; + throw Poco::TimeoutException("injected transient read failure for " + key); + } + return InMemoryBackend::read(key, access); + } + +private: + String flaky_key; + int remaining = 0; }; /// Replaces the durable catalog immediately after returning the first armed catalog read. This @@ -74,8 +114,6 @@ class FailingDeleteBackend : public InMemoryBackend class CatalogChangesAfterFirstReadBackend : public InMemoryBackend { public: - using Backend::get; - void armCatalogReplacement( const String & key, RefCatalog replacement_, size_t completed_reads_before_replacement = 0) { @@ -87,9 +125,11 @@ class CatalogChangesAfterFirstReadBackend : public InMemoryBackend bool fired() const { return replacement_fired; } - std::optional get(const String & key, Range range) override + /// Injects on the `read` PRIMITIVE: every caller reaches this key through `CasOperation::read`, + /// which funnels through here. + std::optional read(const String & key, TransportAccess & access) override { - auto got = InMemoryBackend::get(key, range); + auto got = InMemoryBackend::read(key, access); if (!armed || replacement_fired || key != catalog_key) return got; if (reads_to_skip > 0) @@ -101,9 +141,8 @@ class CatalogChangesAfterFirstReadBackend : public InMemoryBackend throw std::runtime_error("catalog replacement fixture: catalog is absent"); replacement_fired = true; - const PutResult put = InMemoryBackend::putOverwrite( - key, encodeRefCatalog(replacement), got->token, {}); - if (put.outcome != PutOutcome::Done) + const auto put = InMemoryBackend::write(key, encodeRefCatalog(replacement), got->value, access); + if (!put.has_value()) throw std::runtime_error("catalog replacement fixture: rewrite conflicted"); return got; } @@ -116,20 +155,22 @@ class CatalogChangesAfterFirstReadBackend : public InMemoryBackend bool replacement_fired = false; }; -std::vector> snapshotPrefixObjects( +std::vector> snapshotPrefixObjects( InMemoryBackend & backend, const String & prefix) { - std::vector> objects; + std::vector> objects; + auto requests = openRequestsForTest(backend); + auto op = requests.admit(); String cursor; while (true) { - const ListPage page = backend.list(prefix, cursor, 1000); + const ListPage page = op.list(prefix, cursor, 1000, Retry::once()); for (const ListedKey & listed : page.keys) { - const auto got = backend.get(listed.key); + const auto got = op.read(listed.key, Retry::once()); if (!got) throw std::runtime_error("prefix snapshot fixture: listed object disappeared"); - objects.emplace_back(listed.key, got->bytes, got->token); + objects.emplace_back(listed.key, got->bytes, got->etag); } if (page.next_cursor.empty()) return objects; @@ -146,43 +187,50 @@ std::vector> snapshotPrefixObjects( class SuccessorReclaimAfterFarewellBackend : public InMemoryBackend { public: - using Backend::get; - using Backend::putOverwrite; - void armForSuccessorReclaim() { armed = true; } - std::optional get(const String & key, Range range) override + /// Injects on the `read` PRIMITIVE: the retirement tail's own reads of `mount_key`/`epoch_key` + /// (`CasDecommission.cpp`) go through `CasOperation::read`, not the legacy `get`. + std::optional read(const String & key, TransportAccess & access) override { - std::optional result = InMemoryBackend::get(key, range); + std::optional result = InMemoryBackend::read(key, access); if (farewell_seen && !successor_injected && (key == mount_key || key == epoch_key)) - injectSuccessor(); + injectSuccessor(access); return result; } - PutResult putOverwrite( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + /// Injects on the `write` PRIMITIVE: `putOverwrite` is not one of the two verb-identity + /// exceptions (`putIfAbsent`/`casPut`), so both the legacy caller and `CasOperation::replace` + /// reach it here. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - const PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); - if (armed && key == mount_key && result.outcome == PutOutcome::Done) + const auto result = InMemoryBackend::write(key, bytes, expected_value, access); + if (armed && result.has_value() && key == mount_key) { const MountLease mount = decodeMountLease(bytes); - if (mount.min_active == std::numeric_limits::max()) + if (mount.min_active_build_sequence == std::numeric_limits::max()) farewell_seen = true; } return result; } bool successorInjected() const { return successor_injected; } - const Token & successorMountToken() const { return successor_mount_token; } - const Token & successorEpochToken() const { return successor_epoch_token; } + const String & successorMountValue() const { return successor_mount_value; } + const String & successorEpochValue() const { return successor_epoch_value; } const String & successorMountBytes() const { return successor_mount_bytes; } const String & successorEpochBytes() const { return successor_epoch_bytes; } private: - void injectSuccessor() + /// Every request here is issued on the PRIMITIVE `write`, using the `access` token the caller's + /// own in-flight request already holds -- no `CasRequests`/`CasOperation` exists inside this + /// reentrant hook to mint one. The captured `successor_*_value` fields are the raw wire values + /// `write` returned, which the tests compare against a real incarnation's rendered value + /// (`PersistedEtag::capture`) taken through their own `CasOperation`. + void injectSuccessor(TransportAccess & access) { - const auto epoch = InMemoryBackend::get(epoch_key, {}); - const auto mount = InMemoryBackend::get(mount_key, {}); + const auto epoch = InMemoryBackend::read(epoch_key, access); + const auto mount = InMemoryBackend::read(mount_key, access); if (!epoch || !mount) throw std::runtime_error("successor-reclaim fixture: control object disappeared before reclaim"); @@ -190,25 +238,23 @@ class SuccessorReclaimAfterFarewellBackend : public InMemoryBackend const uint64_t successor_writer_epoch = epoch_value.next_writer_epoch; ++epoch_value.next_writer_epoch; successor_epoch_bytes = encodeServerEpoch(epoch_value); - const CasResult epoch_put = InMemoryBackend::casPut( - epoch_key, successor_epoch_bytes, std::optional{epoch->token}, {}); - if (epoch_put.outcome != CasOutcome::Committed) + const auto epoch_written = InMemoryBackend::write(epoch_key, successor_epoch_bytes, epoch->value, access); + if (!epoch_written) throw std::runtime_error("successor-reclaim fixture: epoch bump conflicted"); - successor_epoch_token = epoch_put.token; + successor_epoch_value = *epoch_written; MountLease mount_value = decodeMountLease(mount->bytes); mount_value.writer_epoch = successor_writer_epoch; ++mount_value.seq; ++mount_value.started_at_ms; mount_value.expires_at_ms = mount_value.started_at_ms + 30'000; - mount_value.min_active = 0; + mount_value.min_active_build_sequence = 0; mount_value.gc_fenced = false; successor_mount_bytes = encodeMountLease(mount_value); - const PutResult mount_put = InMemoryBackend::putOverwrite( - mount_key, successor_mount_bytes, mount->token, {}); - if (mount_put.outcome != PutOutcome::Done) + const auto mount_written = InMemoryBackend::write(mount_key, successor_mount_bytes, mount->value, access); + if (!mount_written) throw std::runtime_error("successor-reclaim fixture: mount reclaim conflicted"); - successor_mount_token = mount_put.token; + successor_mount_value = *mount_written; successor_injected = true; } @@ -217,8 +263,8 @@ class SuccessorReclaimAfterFarewellBackend : public InMemoryBackend bool armed = false; bool farewell_seen = false; bool successor_injected = false; - Token successor_mount_token; - Token successor_epoch_token; + String successor_mount_value; + String successor_epoch_value; String successor_mount_bytes; String successor_epoch_bytes; }; @@ -229,45 +275,50 @@ class SuccessorReclaimAfterFarewellBackend : public InMemoryBackend class SuccessorReclaimAfterEpochDeleteBackend : public InMemoryBackend { public: - using Backend::get; - using Backend::putOverwrite; - void armForSuccessorReclaim() { armed = true; } - DeleteOutcome deleteExact(const String & key, const Token & token) override + /// Injects on the `remove` PRIMITIVE: `deleteSlotObject`'s epoch delete (`CasDecommission.cpp`) + /// goes through `CasOperation::remove`, not the legacy `deleteExact`. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - const DeleteOutcome result = InMemoryBackend::deleteExact(key, token); - if (armed && !successor_injected && key == epoch_key - && classifyDeleteOutcome(result) == DeleteClass::Deleted) - { - injectSuccessor(); - } + const RawRemoval result = InMemoryBackend::remove(key, expected_value, access); + if (armed && !successor_injected && key == epoch_key && result == RawRemoval::Removed) + injectSuccessor(access); return result; } bool successorInjected() const { return successor_injected; } uint64_t ownerRewriteAttempts() const { return owner_rewrite_attempts; } - const Token & successorMountToken() const { return successor_mount_token; } - const Token & successorEpochToken() const { return successor_epoch_token; } + const String & successorMountValue() const { return successor_mount_value; } + const String & successorEpochValue() const { return successor_epoch_value; } const String & successorMountBytes() const { return successor_mount_bytes; } const String & successorEpochBytes() const { return successor_epoch_bytes; } private: - PutResult putOverwrite( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + /// Counts on the `write` PRIMITIVE: the owner tombstone write this test asserts is never + /// attempted (`CasDecommission.cpp`'s `op.replace`) reaches the store through here, not through + /// the legacy `putOverwrite`. Guarded on `expected_value`: the primitive also sees the owner + /// anchor's own CREATE during this fixture's `openVictim`, which `putOverwrite` (a conditional + /// REPLACE only) never did -- counting it would start this counter at 1 before the interesting + /// part of the test begins. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - if (key == owner_key) + if (key == owner_key && expected_value) ++owner_rewrite_attempts; - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } - void injectSuccessor() + /// On the `write` PRIMITIVE, using the caller's own in-flight `access`: this hook is reentrant + /// (called from inside another primitive override on the same object), so it reaches the store + /// directly through `InMemoryBackend::write` rather than admitting a fresh request of its own. + void injectSuccessor(TransportAccess & access) { successor_epoch_bytes = encodeServerEpoch(ServerEpoch{.next_writer_epoch = 102}); - const PutResult epoch_put = InMemoryBackend::putIfAbsent(epoch_key, successor_epoch_bytes, {}); - if (epoch_put.outcome != PutOutcome::Done) + const auto epoch_written = InMemoryBackend::write(epoch_key, successor_epoch_bytes, std::nullopt, access); + if (!epoch_written) throw std::runtime_error("late-successor fixture: epoch recreation conflicted"); - successor_epoch_token = epoch_put.token; + successor_epoch_value = *epoch_written; successor_mount_bytes = encodeMountLease(MountLease{ .server_uuid = UInt128(0x1234), @@ -277,12 +328,12 @@ class SuccessorReclaimAfterEpochDeleteBackend : public InMemoryBackend .started_at_ms = 1'000, .seq = 1, .expires_at_ms = 31'000, - .min_active = 0, + .min_active_build_sequence = 0, }); - const PutResult mount_put = InMemoryBackend::putIfAbsent(mount_key, successor_mount_bytes, {}); - if (mount_put.outcome != PutOutcome::Done) + const auto mount_written = InMemoryBackend::write(mount_key, successor_mount_bytes, std::nullopt, access); + if (!mount_written) throw std::runtime_error("late-successor fixture: mount recreation conflicted"); - successor_mount_token = mount_put.token; + successor_mount_value = *mount_written; successor_injected = true; } @@ -292,8 +343,8 @@ class SuccessorReclaimAfterEpochDeleteBackend : public InMemoryBackend bool armed = false; bool successor_injected = false; uint64_t owner_rewrite_attempts = 0; - Token successor_mount_token; - Token successor_epoch_token; + String successor_mount_value; + String successor_epoch_value; String successor_mount_bytes; String successor_epoch_bytes; }; @@ -304,39 +355,40 @@ class SuccessorReclaimAfterEpochDeleteBackend : public InMemoryBackend class SuccessorOwnerRewriteBeforeTombstoneBackend : public InMemoryBackend { public: - using Backend::get; - void armForSuccessorRewrite() { armed = true; } - std::optional get(const String & key, Range range) override + /// Injects on the `read` PRIMITIVE: the tombstone tail's owner read (`CasDecommission.cpp`) + /// goes through `CasOperation::read`, not the legacy `get`. + std::optional read(const String & key, TransportAccess & access) override { - std::optional result = InMemoryBackend::get(key, range); + std::optional result = InMemoryBackend::read(key, access); if (armed && epoch_deleted && !successor_injected && key == owner_key && result) { successor_owner_bytes = encodeOwner(OwnerObject{ .server_uuid = decodeOwner(result->bytes).server_uuid, .retired_at_ms = std::nullopt, }); - const PutResult put = InMemoryBackend::putOverwrite( - owner_key, successor_owner_bytes, result->token, {}); - if (put.outcome != PutOutcome::Done) + const auto put = InMemoryBackend::write(owner_key, successor_owner_bytes, result->value, access); + if (!put.has_value()) throw std::runtime_error("owner-successor fixture: owner rewrite conflicted"); - successor_owner_token = put.token; + successor_owner_value = *put; successor_injected = true; } return result; } - DeleteOutcome deleteExact(const String & key, const Token & token) override + /// Injects on the `remove` PRIMITIVE: `deleteSlotObject`'s epoch delete (`CasDecommission.cpp`) + /// goes through `CasOperation::remove`, not the legacy `deleteExact`. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - const DeleteOutcome result = InMemoryBackend::deleteExact(key, token); - if (armed && key == epoch_key && classifyDeleteOutcome(result) == DeleteClass::Deleted) + const RawRemoval result = InMemoryBackend::remove(key, expected_value, access); + if (armed && key == epoch_key && result == RawRemoval::Removed) epoch_deleted = true; return result; } bool successorInjected() const { return successor_injected; } - const Token & successorOwnerToken() const { return successor_owner_token; } + const String & successorOwnerValue() const { return successor_owner_value; } const String & successorOwnerBytes() const { return successor_owner_bytes; } private: @@ -345,39 +397,10 @@ class SuccessorOwnerRewriteBeforeTombstoneBackend : public InMemoryBackend bool armed = false; bool epoch_deleted = false; bool successor_injected = false; - Token successor_owner_token; + String successor_owner_value; String successor_owner_bytes; }; -/// Models an "ambiguous success" on the final owner tombstone write: the conditional overwrite -/// actually lands (InMemoryBackend applies it), but the response is then lost (a transient -/// exception is thrown on the SAME call, exactly as a real SDK timeout after a landed write would -/// look). Before the fix, decommission caught any exception here and reported failure -/// unconditionally; the controlled overwrite must resolve this via a GET (the current bytes match -/// what was intended) and report Committed instead. -class AmbiguousOwnerTombstoneBackend : public InMemoryBackend -{ -public: - using Backend::putOverwrite; - - void armForAmbiguousTombstone() { armed = true; } - - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override - { - const PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); - if (armed && !fired && key == owner_key && result.outcome == PutOutcome::Done) - { - fired = true; - throw std::runtime_error("ambiguous-tombstone fixture: response lost after the write landed"); - } - return result; - } - -private: - inline static const String owner_key = "p/gc/server-roots/victim/owner"; - bool armed = false; - bool fired = false; -}; /// Seed one victim table with `committed` committed refs and `precommits` dangling precommit bindings, /// via the raw ref-log seeding helpers (fixture idiom of e.g. `gtest_cas_gc_fold.cpp`: `writeManifestRaw` @@ -389,13 +412,19 @@ class AmbiguousOwnerTombstoneBackend : public InMemoryBackend void makeTableWithRefs(Pool & victim, const String & ns_str, uint64_t committed, uint64_t precommits) { const RootNamespace ns(ns_str); - Backend & backend = victim.backend(); + Backend & backend = *victim.poolBackendPtr(); const Layout & layout = victim.layout(); + /// A throwaway open-fence operation, for the two `CasRefCatalog` calls below only: this fixture + /// writes everything else directly against `backend` via the raw-write helpers, unrelated to any + /// mount fence. + CasRequests requests(victim.poolBackendPtr(), Fence::open()); + CasOperation op = requests.admit(); + /// Final physical ids are pool-wide. The generic raw-write helper intentionally uses one shared /// transition sentinel, so this multi-namespace fixture admits a distinct deterministic test life /// before invoking it; the helper then resolves and preserves that existing catalog identity. - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); const auto existing = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns.string() == ns.string(); }); if (existing == catalog.catalog.entries.end()) @@ -405,7 +434,7 @@ void makeTableWithRefs(Pool & victim, const String & ns_str, uint64_t committed, entry.ns = ns; entry.state = NsState::Live; entry.incarnation = UInt128{next_test_life.fetch_add(1)}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); } uint64_t last_ref_sequence = 0; @@ -445,10 +474,11 @@ ManifestId seedOrphanManifestBody(Pool & victim, const String & ns_str) { const RootNamespace ns(ns_str); const ManifestRef ref{.writer_epoch = victim.writerEpoch(), .build_sequence = 99, .manifest_ordinal = 1}; - const ManifestId id = writeManifestRaw(victim.backend(), victim.layout(), ns, ref, {}); + const ManifestId id = writeManifestRaw(*victim.poolBackendPtr(), victim.layout(), ns, ref, {}); /// EXPECT, not ASSERT: this function returns a value now, and ASSERT_* expands to a bare `return;` /// -- invalid in a non-void function. - EXPECT_TRUE(victim.backend().head(victim.layout().manifestKey(id)).exists); + OperationForTest op(*victim.poolBackendPtr()); + EXPECT_TRUE((*op).head(victim.layout().manifestKey(id), Retry::once()).has_value()); return id; } @@ -543,31 +573,32 @@ TEST(CASDecommission, DuplicateLifeIdRefusesBeforeAnyNamespaceOrSlotMutation) .incarnation = UInt128{77}, .removal_started_round = 1}, }; - const auto empty_catalog = backend->get(layout.refCatalogKey()); + OperationForTest raw_op(*backend); + const auto empty_catalog = (*raw_op).read(layout.refCatalogKey(), Retry::once()); ASSERT_TRUE(empty_catalog); - ASSERT_EQ(backend->putOverwrite( - layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->token).outcome, PutOutcome::Done); - const auto owner_before = backend->get(layout.ownerKey("victim")); - const auto epoch_before = backend->get(layout.epochKey("victim")); - const auto mount_before = backend->get(layout.mountKey("victim")); + ASSERT_TRUE(std::holds_alternative((*raw_op).replace( + layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->etag, Retry::once()))); + const auto owner_before = (*raw_op).read(layout.ownerKey("victim"), Retry::once()); + const auto epoch_before = (*raw_op).read(layout.epochKey("victim"), Retry::once()); + const auto mount_before = (*raw_op).read(layout.mountKey("victim"), Retry::once()); ASSERT_TRUE(owner_before); ASSERT_TRUE(epoch_before); ASSERT_TRUE(mount_before); EXPECT_THROW(decommissionPoolMember( backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"), DB::Exception); - const auto owner_after = backend->get(layout.ownerKey("victim")); - const auto epoch_after = backend->get(layout.epochKey("victim")); - const auto mount_after = backend->get(layout.mountKey("victim")); + const auto owner_after = (*raw_op).read(layout.ownerKey("victim"), Retry::once()); + const auto epoch_after = (*raw_op).read(layout.epochKey("victim"), Retry::once()); + const auto mount_after = (*raw_op).read(layout.mountKey("victim"), Retry::once()); ASSERT_TRUE(owner_after); ASSERT_TRUE(epoch_after); ASSERT_TRUE(mount_after); EXPECT_EQ(owner_after->bytes, owner_before->bytes); - EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(owner_after->etag, owner_before->etag); EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); - EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(epoch_after->etag, epoch_before->etag); EXPECT_EQ(mount_after->bytes, mount_before->bytes); - EXPECT_EQ(mount_after->token, mount_before->token); + EXPECT_EQ(mount_after->etag, mount_before->etag); } TEST(CASDecommission, CatalogCutIsValidatedBeforeImpersonationAndReusedForSelection) @@ -586,9 +617,10 @@ TEST(CASDecommission, CatalogCutIsValidatedBeforeImpersonationAndReusedForSelect .removal_started_round = 1}, }; - const auto owner_before = backend->get(layout.ownerKey("victim")); - const auto epoch_before = backend->get(layout.epochKey("victim")); - const auto mount_before = backend->get(layout.mountKey("victim")); + OperationForTest raw_op(*backend); + const auto owner_before = (*raw_op).read(layout.ownerKey("victim"), Retry::once()); + const auto epoch_before = (*raw_op).read(layout.epochKey("victim"), Retry::once()); + const auto mount_before = (*raw_op).read(layout.mountKey("victim"), Retry::once()); ASSERT_TRUE(owner_before); ASSERT_TRUE(epoch_before); ASSERT_TRUE(mount_before); @@ -598,18 +630,18 @@ TEST(CASDecommission, CatalogCutIsValidatedBeforeImpersonationAndReusedForSelect backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"), DB::Exception); ASSERT_TRUE(backend->fired()); - const auto owner_after = backend->get(layout.ownerKey("victim")); - const auto epoch_after = backend->get(layout.epochKey("victim")); - const auto mount_after = backend->get(layout.mountKey("victim")); + const auto owner_after = (*raw_op).read(layout.ownerKey("victim"), Retry::once()); + const auto epoch_after = (*raw_op).read(layout.epochKey("victim"), Retry::once()); + const auto mount_after = (*raw_op).read(layout.mountKey("victim"), Retry::once()); ASSERT_TRUE(owner_after); ASSERT_TRUE(epoch_after); ASSERT_TRUE(mount_after); EXPECT_EQ(owner_after->bytes, owner_before->bytes); - EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(owner_after->etag, owner_before->etag); EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); - EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(epoch_after->etag, epoch_before->etag); EXPECT_EQ(mount_after->bytes, mount_before->bytes); - EXPECT_EQ(mount_after->token, mount_before->token); + EXPECT_EQ(mount_after->etag, mount_before->etag); } TEST(CASDecommission, NamespaceSelectionUsesThePreImpersonationCut) @@ -645,28 +677,26 @@ TEST(CASDecommission, SameNameRebirthAfterTheCutIsRefusedWithoutTouchingTheNewLi old_catalog.entries = { CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = old_life.incarnation}, }; - const auto empty_catalog = backend->get(layout.refCatalogKey()); + OperationForTest raw_op(*backend); + const auto empty_catalog = (*raw_op).read(layout.refCatalogKey(), Retry::once()); ASSERT_TRUE(empty_catalog); - ASSERT_EQ(backend->putOverwrite( - layout.refCatalogKey(), encodeRefCatalog(old_catalog), empty_catalog->token).outcome, - PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative((*raw_op).replace( + layout.refCatalogKey(), encodeRefCatalog(old_catalog), empty_catalog->etag, Retry::once()))); RefLogTxn new_birth; new_birth.ns = ns.string(); new_birth.txn_id = RefTxnId{1, 1}; new_birth.ops = {namespaceBirthOp()}; - ASSERT_EQ(backend->putIfAbsent( + ASSERT_TRUE(std::holds_alternative((*raw_op).create( layout.refLogKey(new_life, new_birth.txn_id), - sealObject(FormatId::RefLog, encodeRefLogTxn(new_birth))).outcome, - PutOutcome::Done); + sealObject(FormatId::RefLog, encodeRefLogTxn(new_birth)), Retry::once()))); RefLogTxn new_seal; new_seal.ns = ns.string(); new_seal.txn_id = RefTxnId{1, 2}; new_seal.ops = {epochSealOp()}; - ASSERT_EQ(backend->putIfAbsent( + ASSERT_TRUE(std::holds_alternative((*raw_op).create( layout.refLogKey(new_life, new_seal.txn_id), - sealObject(FormatId::RefLog, encodeRefLogTxn(new_seal))).outcome, - PutOutcome::Done); + sealObject(FormatId::RefLog, encodeRefLogTxn(new_seal)), Retry::once()))); const auto new_life_before = snapshotPrefixObjects(*backend, layout.namespaceStreamPrefix(new_life)); RefCatalog replacement; @@ -773,8 +803,8 @@ TEST(CASDecommission, CountsRealisticEpochPrecommit) /// -- a REAL build's `ManifestRef` is unique per build, and a colliding one would trip the ref /// state machine's "manifest already has a conflicting owner" guard. const ManifestRef ref{.writer_epoch = victim_epoch, .build_sequence = 2, .manifest_ordinal = 1}; - writeManifestRaw(victim->backend(), victim->layout(), ns, ref, {}); - addPrecommitTransition(victim->backend(), victim->layout(), ns, UInt128(1), "precommit_0", std::nullopt, ref); + writeManifestRaw(*victim->poolBackendPtr(), victim->layout(), ns, ref, {}); + addPrecommitTransition(*victim->poolBackendPtr(), victim->layout(), ns, UInt128(1), "precommit_0", std::nullopt, ref); } const auto report = decommissionPoolMember( @@ -829,9 +859,12 @@ TEST(CASDecommission, DrainsDebrisStagingAndRoots) /// Foreign staging + mountpoint objects, written raw (no writer machinery needed): the victim's /// writers are fenced by the claim before decommission ever gets here, so these are ordinary debris, /// not a live in-flight write. - backend->putIfAbsent("p/staging/victim/upload1.tmp", "x"); - backend->putIfAbsent("p/staging/victim/upload2.tmp", "x"); - backend->putIfAbsent("p/roots/victim/clickhouse_access_check_abc", "x"); + { + OperationForTest seed_op(*backend); + (*seed_op).create("p/staging/victim/upload1.tmp", "x", Retry::once()); + (*seed_op).create("p/staging/victim/upload2.tmp", "x", Retry::once()); + (*seed_op).create("p/roots/victim/clickhouse_access_check_abc", "x", Retry::once()); + } const auto report = decommissionPoolMember( backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); @@ -847,8 +880,9 @@ TEST(CASDecommission, DrainsDebrisStagingAndRoots) /// Nothing of the victim remains under staging/ or roots/ (scoped LISTs are empty). Those two phases /// run to completion even though the debris phase retained -- the drain is per-phase, not all-or-nothing. - EXPECT_TRUE(backend->list("p/staging/victim/", "", 10).keys.empty()); - EXPECT_TRUE(backend->list("p/roots/victim/", "", 10).keys.empty()); + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).list("p/staging/victim/", "", 10, Retry::once()).keys.empty()); + EXPECT_TRUE((*raw_op).list("p/roots/victim/", "", 10, Retry::once()).keys.empty()); } /// The §6 deletion premise applies to the decommission drain too, and this pins what that COSTS. With no @@ -878,7 +912,8 @@ TEST(CASDecommission, RetainsDebrisWhoseEpochSealIsUnconsumed) backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); EXPECT_EQ(report.manifest_debris_removed, 0u); - EXPECT_TRUE(backend->head(debris_key).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(debris_key, Retry::once()).has_value()) << "the body is retained untouched, not deleted and not corrupted"; ASSERT_FALSE(report.warnings.empty()) << "a retained manifest is a visible decision -- the operator must be able to see why the drain " @@ -904,14 +939,35 @@ TEST(CASDecommission, PerObjectFailureWarnsAndContinuesDrain) auto victim = openVictim(backend); makeTableWithRefs(*victim, "victim/db/t1", 1, 0); } - backend->putIfAbsent("p/staging/victim/upload_ok.tmp", "x"); - backend->putIfAbsent("p/staging/victim/upload_throws.tmp", "x"); - backend->putIfAbsent("p/roots/victim/clickhouse_access_check_abc", "x"); + { + OperationForTest seed_op(*backend); + (*seed_op).create("p/staging/victim/upload_ok.tmp", "x", Retry::once()); + (*seed_op).create("p/staging/victim/upload_throws.tmp", "x", Retry::once()); + (*seed_op).create("p/roots/victim/clickhouse_access_check_abc", "x", Retry::once()); + } backend->failWithThrow("p/staging/victim/upload_throws.tmp"); + backend->latch(); backend->failWithTokenMismatch("p/roots/victim/clickhouse_access_check_abc"); + /// The engine reissues an unresolved delete until its own retry window closes, measured on this + /// clock, so the latched fault reaches a genuine give-up with no real time passing. + /// Heap-owned, not a plain stack local: `decommissionPoolMember` installs this clock into the + /// Pool's `boot_ms_fn` background mount-lease renewer, which can still be running on a detached + /// thread after this function returns, so a by-reference capture of a local would dangle. Wrapped + /// (rather than passing `clock->nowFn()`/`clock->sleepFn()` directly) so the closures stored in + /// `PoolConfig` hold the shared_ptr itself, not just the raw `FakeClock*` those methods capture. + auto clock = std::make_shared(); const auto report = decommissionPoolMember( - backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", + /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); EXPECT_EQ(report.staging_objects_removed, 1u) << "the OTHER staging object must still be deleted despite the injected failure on its sibling"; @@ -919,11 +975,12 @@ TEST(CASDecommission, PerObjectFailureWarnsAndContinuesDrain) EXPECT_EQ(report.warnings.size(), 2u) << "one warning for the thrown exception, one for the TokenMismatch outcome"; - EXPECT_FALSE(backend->head("p/staging/victim/upload_ok.tmp").exists) + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head("p/staging/victim/upload_ok.tmp", Retry::once()).has_value()) << "the healthy staging object was actually deleted, not merely skipped"; - EXPECT_TRUE(backend->head("p/staging/victim/upload_throws.tmp").exists) + EXPECT_TRUE((*raw_op).head("p/staging/victim/upload_throws.tmp", Retry::once()).has_value()) << "the failing object is left behind (untouched) so a re-run can retry it"; - EXPECT_TRUE(backend->head("p/roots/victim/clickhouse_access_check_abc").exists) + EXPECT_TRUE((*raw_op).head("p/roots/victim/clickhouse_access_check_abc", Retry::once()).has_value()) << "TokenMismatch means nothing was actually deleted -- the object survives"; } @@ -939,13 +996,15 @@ TEST(CASDecommission, LifelessPhysicalKeyCannotRedirectCatalogOwnedDecommission) /// Hand-built: no helper can mint the un-incarnated shape any more. lifeless = victim->layout().casRefsPrefix() + String("victim/db/t1/_log/") + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; - ASSERT_EQ(backend->putIfAbsent(lifeless, "garbage").outcome, PutOutcome::Done); + OperationForTest seed_op(*backend); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(lifeless, "garbage", Retry::once()))); } const auto report = decommissionPoolMember( backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); EXPECT_EQ(report.namespaces_removed, 1u); - EXPECT_TRUE(backend->head(lifeless).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(lifeless, Retry::once()).has_value()) << "decommission must neither adopt nor delete an unowned physical life key"; } @@ -964,10 +1023,31 @@ TEST(CASDecommission, ManifestDebrisDeleteFailureWarnsAndContinues) debris_key = victim->layout().manifestKey(debris_id); } backend->failWithThrow(debris_key); - backend->putIfAbsent("p/staging/victim/upload_ok.tmp", "x"); + backend->latch(); + { + OperationForTest seed_op(*backend); + (*seed_op).create("p/staging/victim/upload_ok.tmp", "x", Retry::once()); + } + /// The engine reissues an unresolved delete until its own retry window closes, measured on this + /// clock, so the latched fault reaches a genuine give-up with no real time passing. + /// Heap-owned, not a plain stack local: `decommissionPoolMember` installs this clock into the + /// Pool's `boot_ms_fn` background mount-lease renewer, which can still be running on a detached + /// thread after this function returns, so a by-reference capture of a local would dangle. Wrapped + /// (rather than passing `clock->nowFn()`/`clock->sleepFn()` directly) so the closures stored in + /// `PoolConfig` hold the shared_ptr itself, not just the raw `FakeClock*` those methods capture. + auto clock = std::make_shared(); const auto report = decommissionPoolMember( - backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim", + /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); EXPECT_EQ(report.namespaces_removed, 1u) << "victim/db/t1's namespace erasure (Task 2) is untouched by either injected failure"; @@ -978,7 +1058,8 @@ TEST(CASDecommission, ManifestDebrisDeleteFailureWarnsAndContinues) << "the staging phase still ran to completion after the manifest-debris phase's failures -- " "the whole command did not abort"; - EXPECT_TRUE(backend->head(debris_key).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(debris_key, Retry::once()).has_value()) << "the failing object is left behind (untouched) so a re-run can retry it"; } @@ -1000,11 +1081,12 @@ TEST(CASDecommission, RemovesMutableSlotAndRefusesTombstonedRerun) backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin2"}, "victim"); EXPECT_TRUE(report.slot_removed); EXPECT_TRUE(report.warnings.empty()); - EXPECT_FALSE(backend->get("p/gc/server-roots/victim/mount").has_value()); - const auto owner = backend->get("p/gc/server-roots/victim/owner"); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()); + const auto owner = (*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()); ASSERT_TRUE(owner.has_value()); EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); - EXPECT_FALSE(backend->get("p/gc/server-roots/victim/epoch").has_value()); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/epoch", Retry::once()).has_value()); expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { @@ -1031,18 +1113,19 @@ TEST(CASDecommission, SuccessorReclaimFencesSlotRetirementTail) EXPECT_FALSE(report.slot_removed); ASSERT_EQ(report.warnings.size(), 1u); EXPECT_NE(report.warnings.front().find("p/gc/server-roots/victim/mount"), String::npos); - EXPECT_NE(report.warnings.front().find("replaced"), String::npos); + EXPECT_NE(report.warnings.front().find("mismatch"), String::npos); - const auto mount = backend->get("p/gc/server-roots/victim/mount"); + OperationForTest raw_op(*backend); + const auto mount = (*raw_op).read("p/gc/server-roots/victim/mount", Retry::once()); ASSERT_TRUE(mount.has_value()); - EXPECT_EQ(mount->token, backend->successorMountToken()); + EXPECT_EQ(PersistedEtag::capture(mount->etag).value, backend->successorMountValue()); EXPECT_EQ(mount->bytes, backend->successorMountBytes()); - const auto epoch = backend->get("p/gc/server-roots/victim/epoch"); + const auto epoch = (*raw_op).read("p/gc/server-roots/victim/epoch", Retry::once()); ASSERT_TRUE(epoch.has_value()); - EXPECT_EQ(epoch->token, backend->successorEpochToken()); + EXPECT_EQ(PersistedEtag::capture(epoch->etag).value, backend->successorEpochValue()); EXPECT_EQ(epoch->bytes, backend->successorEpochBytes()); - EXPECT_TRUE(backend->get("p/gc/server-roots/victim/owner").has_value()); + EXPECT_TRUE((*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()).has_value()); ASSERT_FALSE(seen.empty()); EXPECT_EQ(seen.back().outcome, "end"); @@ -1057,7 +1140,8 @@ TEST(CASDecommission, SuccessorReclaimAfterEpochDeleteKeepsOwnerAnchor) { auto victim = openVictim(backend); } const String owner_key = "p/gc/server-roots/victim/owner"; - const auto original_owner = backend->get(owner_key); + OperationForTest raw_op(*backend); + const auto original_owner = (*raw_op).read(owner_key, Retry::once()); ASSERT_TRUE(original_owner.has_value()); backend->armForSuccessorReclaim(); @@ -1069,19 +1153,19 @@ TEST(CASDecommission, SuccessorReclaimAfterEpochDeleteKeepsOwnerAnchor) EXPECT_FALSE(report.warnings.empty()); EXPECT_EQ(backend->ownerRewriteAttempts(), 0u); - const auto owner = backend->get(owner_key); + const auto owner = (*raw_op).read(owner_key, Retry::once()); ASSERT_TRUE(owner.has_value()); - EXPECT_EQ(owner->token, original_owner->token); + EXPECT_EQ(owner->etag, original_owner->etag); EXPECT_EQ(owner->bytes, original_owner->bytes); - const auto mount = backend->get("p/gc/server-roots/victim/mount"); + const auto mount = (*raw_op).read("p/gc/server-roots/victim/mount", Retry::once()); ASSERT_TRUE(mount.has_value()); - EXPECT_EQ(mount->token, backend->successorMountToken()); + EXPECT_EQ(PersistedEtag::capture(mount->etag).value, backend->successorMountValue()); EXPECT_EQ(mount->bytes, backend->successorMountBytes()); - const auto epoch = backend->get("p/gc/server-roots/victim/epoch"); + const auto epoch = (*raw_op).read("p/gc/server-roots/victim/epoch", Retry::once()); ASSERT_TRUE(epoch.has_value()); - EXPECT_EQ(epoch->token, backend->successorEpochToken()); + EXPECT_EQ(PersistedEtag::capture(epoch->etag).value, backend->successorEpochValue()); EXPECT_EQ(epoch->bytes, backend->successorEpochBytes()); } @@ -1097,9 +1181,45 @@ TEST(CASDecommission, FencedSlotRetirementTailRetiresUncontendedSlot) EXPECT_TRUE(report.warnings.empty()); EXPECT_TRUE(report.slot_removed); - EXPECT_FALSE(backend->get("p/gc/server-roots/victim/mount").has_value()); - EXPECT_FALSE(backend->get("p/gc/server-roots/victim/epoch").has_value()); - const auto owner = backend->get("p/gc/server-roots/victim/owner"); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/epoch", Retry::once()).has_value()); + const auto owner = (*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()); + ASSERT_TRUE(owner.has_value()); + EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); +} + +/// One decommission command spans several requests on the SAME open-fence engine: the +/// pre-impersonation catalog cut (before any `Pool` exists), the namespace drop's own catalog re-read, +/// and -- once the row it dropped is no longer owned -- the full retirement tail's reads, deletes and +/// final owner tombstone, all issued after `admin.reset()` destroys the `Pool`. Catalog-row deletion +/// is GC's job (`dropNamespace` only reaches `Removing`), so this is necessarily two commands: the +/// first proves the drop and its catalog read landed, the second (after GC folds the row) proves the +/// retirement tail's requests landed past the `Pool`'s own lifetime. +TEST(CASDecommission, RunsOnAnOpenFence) +{ + auto backend = std::make_shared(); + { + auto victim = openVictim(backend); + makeTableWithRefs(*victim, "victim/db/t1", 1, 0); + } + + const auto pending = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); + EXPECT_EQ(pending.namespaces_removed, 1u); + EXPECT_FALSE(pending.slot_removed); + EXPECT_FALSE(pending.warnings.empty()); + + drainCompletedNamespaceRemovals(backend); + + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin2"}, "victim"); + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/epoch", Retry::once()).has_value()); + const auto owner = (*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()); ASSERT_TRUE(owner.has_value()); EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); } @@ -1110,7 +1230,8 @@ TEST(CASDecommission, SuccessfulDecommissionLeavesTombstonedOwnerAnchor) { auto victim = openVictim(backend); } const String owner_key = "p/gc/server-roots/victim/owner"; - const auto before = backend->get(owner_key); + OperationForTest raw_op(*backend); + const auto before = (*raw_op).read(owner_key, Retry::once()); ASSERT_TRUE(before.has_value()); EXPECT_FALSE(decodeOwner(before->bytes).retired_at_ms.has_value()); @@ -1119,9 +1240,9 @@ TEST(CASDecommission, SuccessfulDecommissionLeavesTombstonedOwnerAnchor) EXPECT_TRUE(report.warnings.empty()); EXPECT_TRUE(report.slot_removed); - const auto after = backend->get(owner_key); + const auto after = (*raw_op).read(owner_key, Retry::once()); ASSERT_TRUE(after.has_value()); - EXPECT_NE(after->token, before->token); + EXPECT_NE(after->etag, before->etag); EXPECT_EQ(decodeOwner(after->bytes).server_uuid, decodeOwner(before->bytes).server_uuid); EXPECT_TRUE(decodeOwner(after->bytes).retired_at_ms.has_value()); } @@ -1140,22 +1261,27 @@ TEST(CASDecommission, SuccessorOwnerRewriteWinsBeforeTombstone) ASSERT_EQ(report.warnings.size(), 1u); EXPECT_NE(report.warnings.front().find("successor reclaimed"), String::npos); - const auto owner = backend->get("p/gc/server-roots/victim/owner"); + OperationForTest raw_op(*backend); + const auto owner = (*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()); ASSERT_TRUE(owner.has_value()); - EXPECT_EQ(owner->token, backend->successorOwnerToken()); + EXPECT_EQ(PersistedEtag::capture(owner->etag).value, backend->successorOwnerValue()); EXPECT_EQ(owner->bytes, backend->successorOwnerBytes()); EXPECT_FALSE(decodeOwner(owner->bytes).retired_at_ms.has_value()); } /// Final whole-branch review finding (Important): -/// a transient exception on the owner tombstone write must not be reported as a hard failure when the -/// write actually landed -- the controlled overwrite resolves this via GET (current bytes already -/// match the intended tombstone) instead of the old bare putOverwrite's "any exception = failure". +/// an ambiguous outcome on the owner tombstone write -- the write lands but its response is lost, the +/// same shape as a real SDK timeout after a landed write -- must not be reported as a hard failure. +/// `op.replace`'s own resolve read settles this (current bytes already match the intended tombstone) +/// and reports `Committed`. `injectAmbiguousLandedWrite` throws `Poco::TimeoutException` after +/// applying the write: the engine's write loop treats a `Poco::Exception` as a transport fault (never +/// a caller bug), which is exactly the class this scenario models -- a bare `std::runtime_error` here +/// would propagate unchanged instead of being resolved. TEST(CASDecommission, OwnerTombstoneAmbiguousSuccessResolvesToCommitted) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); { auto victim = openVictim(backend); } - backend->armForAmbiguousTombstone(); + backend->injectAmbiguousLandedWrite("p/gc/server-roots/victim/owner"); const auto report = decommissionPoolMember( backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); @@ -1163,25 +1289,21 @@ TEST(CASDecommission, OwnerTombstoneAmbiguousSuccessResolvesToCommitted) EXPECT_TRUE(report.slot_removed) << "the ambiguous write actually landed and must resolve to Committed"; EXPECT_TRUE(report.warnings.empty()); - const auto owner = backend->get("p/gc/server-roots/victim/owner"); + OperationForTest raw_op(*backend); + const auto owner = (*raw_op).read("p/gc/server-roots/victim/owner", Retry::once()); ASSERT_TRUE(owner.has_value()); EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); } -/// Delegates every op to `inner`, except `deleteExact`: while `armed`, any key starting with -/// `fail_prefix` throws an injected transient failure instead of deleting -- models a real backend -/// transiently failing to delete under one whole prefix. `disarm()` clears the failure (the resume -/// half of `FailedDrainKeepsSlotThenResumes`). Forwards every pure-virtual `Backend` member (the -/// `CasBackend.h` list) to `inner` untouched. +/// Delegates every op to `inner`, except `remove`: while `armed`, any key starting with `fail_prefix` +/// throws an injected transient failure instead of deleting -- models a real backend transiently +/// failing to delete under one whole prefix. `disarm()` clears the failure (the resume half of +/// `FailedDrainKeepsSlotThenResumes`). Forwards every pure-virtual `Backend` member (the `CasBackend.h` +/// list) to `inner` untouched. Injects on the `remove` PRIMITIVE, not the legacy `deleteExact`: +/// `deleteListedPrefix`'s mountpoint drain (`CasDecommission.cpp`) goes through `CasOperation::remove`. class FailDeletesUnderPrefixBackend : public Backend { public: - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - FailDeletesUnderPrefixBackend(std::shared_ptr inner_, String fail_prefix_) : inner(std::move(inner_)), fail_prefix(std::move(fail_prefix_)) { @@ -1189,33 +1311,32 @@ class FailDeletesUnderPrefixBackend : public Backend void disarm() { armed = false; } - std::optional get(const String & key, Range range) override { return inner->get(key, range); } - std::optional getStream(const String & key, Range range) override { return inner->getStream(key, range); } - HeadResult head(const String & key) override { return inner->head(key); } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override - { - return inner->putIfAbsent(key, bytes, meta); - } - void publishBlob(const BlobPublishRequest & request) override - { - inner->publishBlob(request); - } - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override - { - return inner->putOverwrite(key, bytes, expected, meta); - } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The transport primitives forward to `inner`, except `remove`, which is what this double + /// injects through. Declared because `Backend` declares them pure. + std::optional read(const String & key, TransportAccess & access) override { return inner->read(key, access); } + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - return inner->casPut(key, bytes, expected, meta); + if (armed && key.starts_with(fail_prefix)) + /// `Poco::TimeoutException`, the class the request engine classifies as a transport fault + /// and reissues (`Retry::standard()`): this fixture models a real backend transiently + /// failing, and `armed` stays set across every reissue of the same call, so the engine + /// reaches its own retry deadline and gives up rather than recovering on a later attempt. + throw Poco::TimeoutException("injected transient delete failure for " + key); + return inner->remove(key, expected_value, access); } - DeleteOutcome deleteExact(const String & key, const Token & token) override + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - if (armed && key.starts_with(fail_prefix)) - throw Exception(ErrorCodes::S3_ERROR, "injected transient delete failure for {}", key); - return inner->deleteExact(key, token); + return inner->write(key, bytes, expected_value, access); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override { return inner->list(prefix, cursor, limit); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + DB::Cas::Dialect dialect() const override { return inner->dialect(); } private: std::shared_ptr inner; @@ -1234,14 +1355,34 @@ TEST(CASDecommission, FailedDrainKeepsSlotThenResumes) auto victim = Pool::open(inner, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); makeTableWithRefs(*victim, "victim/db/t1", 1, 0); } - inner->putIfAbsent("p/roots/victim/loose_file", "x"); + OperationForTest raw_op(*inner); + (*raw_op).create("p/roots/victim/loose_file", "x", Retry::once()); auto failing = std::make_shared(inner, "p/roots/victim/"); + /// The engine reissues an unresolved delete until its own retry window closes, measured on this + /// clock, so the fault (armed across every reissue) reaches a genuine give-up with no real time + /// passing -- `clock` fast-forwards through the whole 90 s `Retry::standard()` window in one call. + /// `drain_now_fn` is also this session's boot clock (decommissionPoolMember unifies the two), so the + /// admin's own mount lease needs a TTL well past that fast-forward or the farewell it attempts on + /// the way out would refuse against a deadline the retry exhaustion already ran past -- a real + /// decommission's background renewer would have kept the deadline current over 90 real seconds, but + /// nothing here advances real time to let it. + auto clock = std::make_shared(); const auto first = decommissionPoolMember( - failing, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim"); + failing, + PoolConfig{.pool_prefix = "p", .server_root_id = "a1", .mount_lease_ttl_ms = std::chrono::milliseconds(300'000)}, + "victim", /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); EXPECT_FALSE(first.warnings.empty()); EXPECT_FALSE(first.slot_removed); - EXPECT_TRUE(inner->get("p/gc/server-roots/victim/mount").has_value()) + EXPECT_TRUE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()) << "slot kept -- resume anchor"; failing->disarm(); @@ -1274,15 +1415,29 @@ TEST(CASDecommission, ManifestDebrisFailureKeepsSlotThenResumes) debris_key = victim->layout().manifestKey(debris_id); } backend->failWithThrow(debris_key); + backend->latch(); + /// The engine reissues an unresolved delete until its own retry window closes, measured on this + /// clock, so the latched fault reaches a genuine give-up with no real time passing. + auto clock = std::make_shared(); const auto first = decommissionPoolMember( - backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim"); + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim", + /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); EXPECT_FALSE(first.warnings.empty()); EXPECT_FALSE(first.slot_removed); EXPECT_EQ(first.manifest_debris_removed, 0u); - EXPECT_TRUE(backend->get("p/gc/server-roots/victim/mount").has_value()) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()) << "slot kept -- resume anchor"; - EXPECT_TRUE(backend->head(debris_key).exists) + EXPECT_TRUE((*raw_op).head(debris_key, Retry::once()).has_value()) << "the failing object is left behind (untouched) so a re-run can retry it"; /// COVERAGE LOST HERE, DELIBERATELY NAMED. Before the §6 premise, clearing the injected failure let @@ -1298,11 +1453,94 @@ TEST(CASDecommission, ManifestDebrisFailureKeepsSlotThenResumes) EXPECT_EQ(second.namespaces_already_removed, 1u); EXPECT_EQ(second.manifest_debris_removed, 0u); EXPECT_FALSE(second.slot_removed); - EXPECT_TRUE(backend->head(debris_key).exists); - EXPECT_TRUE(backend->get("p/gc/server-roots/victim/mount").has_value()) + EXPECT_TRUE((*raw_op).head(debris_key, Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()) << "the slot is still the resume anchor -- nothing was retired against unreclaimed debris"; } +/// CI run 7 (PR #2300): `drain_now_fn` fakes the drain's own request clock, but the mount lease's +/// farewell deadline is bound to `PoolConfig::boot_ms_fn`, a distinct clock that -- before this test's +/// fix -- stayed on the real boot clock regardless. On a freshly booted CI host (real boot time below +/// `FakeClock`'s starting instant) the two clocks disagreed enough that the farewell's own bound looked +/// already-expired, so `admin.reset()` released nothing and the slot was never retired -- passing locally +/// only because a long-lived dev box's real uptime dwarfs the fake clock. Pin the fake clock far beyond +/// ANY real host's boot time (rather than relying on the host actually being fresh) so the mismatch -- +/// and the fix -- are exercised deterministically on every machine. +TEST(CASDecommission, DrainClockUnifiesWithTheFarewellBootClockRegardlessOfHostUptime) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } /// identity only -- no namespace, so retirement runs straight + /// to the farewell instead of stopping on an unrelated warning + + auto clock = std::make_shared(); + clock->now = 1'000'000'000'000'000ULL; /// dwarfs any real CLOCK_BOOTTIME on any host + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim", + /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head("p/gc/server-roots/victim/mount", Retry::once()).has_value()) + << "one clock for both the request engine and the mount lease -- the farewell must commit and the " + "slot must retire in a single call, with no leftover mount to resume against"; +} + +/// Review round 9r2: folding `drain_now_fn` into `config.boot_ms_fn` alone left `Pool::openForDecommission`'s +/// own mount/farewell/GC planes -- constructed and used DURING opening, before `decommissionPoolMember` +/// gets a chance to call `setCasRequestNowFnForTest`/`setCasRetrySleepForTest` on the already-open `Pool` +/// -- retrying on a clock that only a fake SLEEP advances, while those planes still slept for REAL between +/// attempts. A transient failure during opening (the owner-object read on the open plane) would then never +/// see its own `Retry::standard()` deadline elapse, because the bound is measured against a clock frozen +/// at the value it had when the retry loop started: nothing calls the fake clock's `sleepFn` from a path +/// that still sleeps for real. `PoolConfig::retry_sleep_fn` closes this by installing the matching fake +/// sleep on those same planes AT CONSTRUCTION, together with `boot_ms_fn`. +/// +/// Failing-first without risking an actual hang: this fixture flakes the owner read 5 times, so a correct +/// fix paces exactly 5 retries on the fake clock and returns in well under a second of real time; the +/// pre-fix code either never returns (the bound never elapses) or, if it did return, would show an empty +/// `clock.sleeps` (nothing ever called the fake sleep) and real wall time consumed by 5 real backoffs. +TEST(CASDecommission, OpeningRetriesPaceOnTheSameFakeClockAndSleepAsTheDrain) +{ + auto backend = std::make_shared(); + { auto victim = openVictim(backend); } /// identity only + + const Layout layout("p"); + backend->failReadNTimes(layout.ownerKey("victim"), /*times=*/5); + + auto clock = std::make_shared(); + clock->now = 1'000'000'000'000'000ULL; + + const auto started = std::chrono::steady_clock::now(); + const auto report = decommissionPoolMember( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "a1"}, "victim", + /*sink=*/{}, /*request_gc_round=*/{}, + [clock] + { + return clock->nowFn()(); + }, + [clock](uint64_t ms) + { + clock->sleepFn()(ms); + }); + const auto wall_elapsed = std::chrono::steady_clock::now() - started; + + EXPECT_TRUE(report.warnings.empty()); + EXPECT_TRUE(report.slot_removed); + EXPECT_FALSE(clock->sleeps.empty()) + << "the opening retries must have been paced on the injected clock, not a real sleep"; + EXPECT_LT(wall_elapsed, std::chrono::seconds(5)) + << "paced on the fake clock, five retries during opening should cost no real wall time at all"; +} + /// Task 5 (Task-1 carry-forward, escalated by review): preserve recovery from the legacy partial /// hand-cleanup shape where owner and epoch are absent but the mount lease remains. Triage #9 changed /// new retirements to delete `mountKey`/`epochKey` and tombstone `ownerKey`, so the current tail no @@ -1317,7 +1555,7 @@ TEST(CASDecommission, ManifestDebrisFailureKeepsSlotThenResumes) /// therefore uses a victim with NO namespaces at all: identity persisted /// (mount/owner/epoch exist from a real graceful close), data subtree genuinely empty -- the exact /// precondition the fallback is designed for. Simulate the crash directly: claim the slot once (exactly -/// `decommissionPoolMember`'s own first step), let it close gracefully (the mount-lease keeper's +/// `decommissionPoolMember`'s own first step), let it close gracefully (the mount-lease renewer's /// farewell stamp, same as a real `admin.reset()`), then manually strike `epochKey`+`ownerKey`, leaving /// `mountKey`. A `decommissionPoolMember` re-run must resolve identity via the mount-lease fallback and /// finish retiring the slot; a further re-run then sees the tombstone and refuses to resume it. @@ -1335,15 +1573,16 @@ TEST(CASDecommission, MidRetirementCrashResumesViaMountLeaseFallback) } /// Manually strike epoch + owner, leaving the mount -- the legacy partial hand-cleanup shape. + OperationForTest raw_op(*backend); for (const String & key : {layout.epochKey("victim"), layout.ownerKey("victim")}) { - const auto head = backend->head(key); - ASSERT_TRUE(head.exists); - backend->deleteExact(key, head.token); + const auto head = (*raw_op).head(key, Retry::once()); + ASSERT_TRUE(head.has_value()); + ASSERT_EQ((*raw_op).remove(key, head->etag, Retry::once()), Removal::Removed); } - ASSERT_FALSE(backend->get(layout.epochKey("victim")).has_value()); - ASSERT_FALSE(backend->get(layout.ownerKey("victim")).has_value()); - ASSERT_TRUE(backend->get(layout.mountKey("victim")).has_value()) + ASSERT_FALSE((*raw_op).head(layout.epochKey("victim"), Retry::once()).has_value()); + ASSERT_FALSE((*raw_op).head(layout.ownerKey("victim"), Retry::once()).has_value()); + ASSERT_TRUE((*raw_op).head(layout.mountKey("victim"), Retry::once()).has_value()) << "the mount lease must survive -- it is the resume anchor the fallback reads"; const auto report = decommissionPoolMember( @@ -1352,11 +1591,11 @@ TEST(CASDecommission, MidRetirementCrashResumesViaMountLeaseFallback) EXPECT_TRUE(report.warnings.empty()); EXPECT_EQ(report.namespaces_removed, 0u); EXPECT_TRUE(report.slot_removed); - EXPECT_FALSE(backend->get(layout.epochKey("victim")).has_value()); - const auto owner = backend->get(layout.ownerKey("victim")); + EXPECT_FALSE((*raw_op).head(layout.epochKey("victim"), Retry::once()).has_value()); + const auto owner = (*raw_op).read(layout.ownerKey("victim"), Retry::once()); ASSERT_TRUE(owner.has_value()); EXPECT_TRUE(decodeOwner(owner->bytes).retired_at_ms.has_value()); - EXPECT_FALSE(backend->get(layout.mountKey("victim")).has_value()); + EXPECT_FALSE((*raw_op).head(layout.mountKey("victim"), Retry::once()).has_value()); expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] { diff --git a/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp b/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp index a674f13667b6..34cb8b7fad07 100644 --- a/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp +++ b/src/Disks/tests/gtest_cas_decommission_catalog_duties.cpp @@ -24,9 +24,9 @@ PoolPtr openVictim(const std::shared_ptr & backend) return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "victim"}); } -CatalogEntry catalogEntry(Backend & backend, const Layout & layout, const RootNamespace & ns) +CatalogEntry catalogEntry(CasOperation & op, const Layout & layout, const RootNamespace & ns) { - const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + const RefCatalog catalog = CasRefCatalog::read(op, layout).catalog; const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); if (it == catalog.entries.end()) @@ -34,9 +34,9 @@ CatalogEntry catalogEntry(Backend & backend, const Layout & layout, const RootNa return *it; } -void makeRemoving(Backend & backend, const Layout & layout, const CatalogEntry & live) +void makeRemoving(CasOperation & op, const Layout & layout, const CatalogEntry & live) { - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find(next.entries.begin(), next.entries.end(), live); @@ -50,23 +50,30 @@ void makeRemoving(Backend & backend, const Layout & layout, const CatalogEntry & bool slotObjectExists(Backend & backend, const String & leaf) { - return backend.head("p/gc/server-roots/victim/" + leaf).exists; + DB::Cas::tests::OperationForTest op(backend); + return (*op).head("p/gc/server-roots/victim/" + leaf, Retry::standard()).has_value(); } class AddVictimEntryDuringRootDrainBackend final : public InMemoryBackend { public: + /// Unhide the legacy `list` overloads the primitive override below would otherwise hide. + using Backend::list; void arm() { armed = true; } bool fired() const { return added; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Intercepted at the PRIMITIVE, which every legacy forwarder reaches too, so the injection fires + /// whichever surface issued the enumeration. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); if (armed && !added && prefix == "p/roots/victim/" && cursor.empty()) { added = true; + CasRequests requests = DB::Cas::tests::openRequestsForTest(*this); + CasOperation op = requests.admit(); CasRefCatalog::casAdmitEntry( - *this, Layout("p"), 1, + op, Layout("p"), 1, CatalogEntry{ .ns = RootNamespace("victim/db/late"), .state = NsState::Live, @@ -83,24 +90,28 @@ class AddVictimEntryDuringRootDrainBackend final : public InMemoryBackend /// Admits the late catalog entry between the retirement tail's two exact catalog reads /// (`retirement_catalog_cut`, then `fresh_retirement_catalog`), never before. The mountpoint drain's /// `list("p/roots/victim/", ...)` is the last LIST call in `decommissionPoolMember` before either -/// read, so it orders the two `get("p/cas/ref_catalog")` calls that follow it: the first is +/// read, so it orders the two `read("p/cas/ref_catalog")` calls that follow it: the first is /// `retirement_catalog_cut`, the second is `fresh_retirement_catalog`. Mutating on the second call /// makes that read observe a catalog the first read did not. class MutateCatalogBetweenRetirementReadsBackend final : public InMemoryBackend { public: + /// Unhide the legacy `list` overloads the primitive override below would otherwise hide. + using Backend::list; void arm() { armed = true; } bool fired() const { return added; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Both hooks sit on the PRIMITIVES, which every legacy forwarder reaches too, so the ordering + /// they observe is the physical request order whichever surface issued each request. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); if (armed && !past_mountpoint_drain && prefix == "p/roots/victim/" && cursor.empty()) past_mountpoint_drain = true; return page; } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (armed && past_mountpoint_drain && !added && key == "p/cas/ref_catalog") { @@ -109,15 +120,17 @@ class MutateCatalogBetweenRetirementReadsBackend final : public InMemoryBackend else { added = true; + CasRequests requests = DB::Cas::tests::openRequestsForTest(*this); + CasOperation op = requests.admit(); CasRefCatalog::casAdmitEntry( - *this, Layout("p"), 1, + op, Layout("p"), 1, CatalogEntry{ .ns = RootNamespace("victim/db/late"), .state = NsState::Live, .incarnation = UInt128{707}}); } } - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } private: @@ -130,6 +143,8 @@ class MutateCatalogBetweenRetirementReadsBackend final : public InMemoryBackend TEST(CASDecommissionCatalogDuties, RemovingWithoutCheckpointIsCorruptionAndKeepsSlot) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); { auto victim = openVictim(backend); const CatalogEntry live{ @@ -137,9 +152,9 @@ TEST(CASDecommissionCatalogDuties, RemovingWithoutCheckpointIsCorruptionAndKeeps .state = NsState::Live, .incarnation = UInt128{701}}; CasRefCatalog::casAdmitEntry( - *backend, victim->layout(), victim->poolConfig().gc_shards, + catalog_op, victim->layout(), victim->poolConfig().gc_shards, live); - makeRemoving(*backend, victim->layout(), live); + makeRemoving(catalog_op, victim->layout(), live); } expectThrowsCode(ErrorCodes::CORRUPTED_DATA, [&] @@ -151,22 +166,24 @@ TEST(CASDecommissionCatalogDuties, RemovingWithoutCheckpointIsCorruptionAndKeeps EXPECT_TRUE(slotObjectExists(*backend, "owner")); EXPECT_TRUE(slotObjectExists(*backend, "epoch")); EXPECT_TRUE(slotObjectExists(*backend, "mount")); - EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/missing_ckpt")).state, + EXPECT_EQ(catalogEntry(catalog_op, Layout("p"), RootNamespace("victim/db/missing_ckpt")).state, NsState::Removing); } TEST(CASDecommissionCatalogDuties, RemovingWithCheckpointResumesTerminalAndKeepsSlotForGc) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const RootNamespace ns("victim/db/pending_terminal"); std::optional life; { auto victim = openVictim(backend); life = victim->namespaceLife(ns); - const CatalogEntry live = catalogEntry(*backend, victim->layout(), ns); - makeRemoving(*backend, victim->layout(), live); - ASSERT_TRUE(backend->head(victim->layout().refCkptKey(*life)).exists); - ASSERT_TRUE(backend->list(victim->layout().namespaceStreamPrefix(*life), "", 100).keys.empty()); + const CatalogEntry live = catalogEntry(catalog_op, victim->layout(), ns); + makeRemoving(catalog_op, victim->layout(), live); + ASSERT_TRUE(catalog_op.head(victim->layout().refCkptKey(*life), Retry::standard()).has_value()); + ASSERT_TRUE(catalog_op.list(victim->layout().namespaceStreamPrefix(*life), "", 100, Retry::standard()).keys.empty()); } std::atomic wake_requests{0}; @@ -180,11 +197,11 @@ TEST(CASDecommissionCatalogDuties, RemovingWithCheckpointResumesTerminalAndKeeps EXPECT_FALSE(report.warnings.empty()); EXPECT_TRUE(slotObjectExists(*backend, "owner")); - const ListPage stream = backend->list(Layout("p").namespaceStreamPrefix(*life), "", 100); + const ListPage stream = catalog_op.list(Layout("p").namespaceStreamPrefix(*life), "", 100, Retry::standard()); ASSERT_EQ(stream.keys.size(), 1u); const auto parsed = Layout("p").parseRefObjectKey(stream.keys.front().key); ASSERT_TRUE(parsed); - const auto body = backend->get(stream.keys.front().key); + const auto body = catalog_op.read(stream.keys.front().key, Retry::standard()); ASSERT_TRUE(body); const RefLogTxn terminal = decodeRefLogTxn( openObject(FormatId::RefLog, body->bytes), ns.string(), parsed->txn_id); @@ -196,24 +213,26 @@ TEST(CASDecommissionCatalogDuties, RemovingWithCheckpointResumesTerminalAndKeeps TEST(CASDecommissionCatalogDuties, PartialRemovalProgressStillWakesGcWhenLaterNamespaceFails) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const RootNamespace progressed_ns("victim/db/a_progressed"); const RootNamespace broken_ns("victim/db/z_missing_ckpt"); std::optional progressed_life; { auto victim = openVictim(backend); progressed_life = victim->namespaceLife(progressed_ns); - const CatalogEntry progressed_live = catalogEntry(*backend, victim->layout(), progressed_ns); - makeRemoving(*backend, victim->layout(), progressed_live); + const CatalogEntry progressed_live = catalogEntry(catalog_op, victim->layout(), progressed_ns); + makeRemoving(catalog_op, victim->layout(), progressed_live); const CatalogEntry broken_live{ .ns = broken_ns, .state = NsState::Live, .incarnation = UInt128{713}}; CasRefCatalog::casAdmitEntry( - *backend, victim->layout(), victim->poolConfig().gc_shards, broken_live); - makeRemoving(*backend, victim->layout(), broken_live); - ASSERT_FALSE(backend->head(victim->layout().refCkptKey( - NamespaceLifeId::fromCatalogEntry(broken_ns, broken_live.incarnation))).exists); + catalog_op, victim->layout(), victim->poolConfig().gc_shards, broken_live); + makeRemoving(catalog_op, victim->layout(), broken_live); + ASSERT_FALSE(catalog_op.head(victim->layout().refCkptKey( + NamespaceLifeId::fromCatalogEntry(broken_ns, broken_live.incarnation)), Retry::standard()).has_value()); } std::atomic wake_requests{0}; @@ -228,13 +247,15 @@ TEST(CASDecommissionCatalogDuties, PartialRemovalProgressStillWakesGcWhenLaterNa << "progress already made for an earlier life must wake GC even when a later life fails closed"; EXPECT_TRUE(slotObjectExists(*backend, "owner")); const ListPage progressed_stream - = backend->list(Layout("p").namespaceStreamPrefix(*progressed_life), "", 100); + = catalog_op.list(Layout("p").namespaceStreamPrefix(*progressed_life), "", 100, Retry::standard()); ASSERT_EQ(progressed_stream.keys.size(), 1u); } TEST(CASDecommissionCatalogDuties, VictimEntryAppearingBeforeTheOwnershipCutKeepsSlot) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); { auto victim = openVictim(backend); } backend->arm(); @@ -247,12 +268,14 @@ TEST(CASDecommissionCatalogDuties, VictimEntryAppearingBeforeTheOwnershipCutKeep EXPECT_NE(report.warnings.front().find("pool member decommission underway: 1 namespace(s)"), String::npos) << report.warnings.front(); EXPECT_TRUE(slotObjectExists(*backend, "owner")); - EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); + EXPECT_EQ(catalogEntry(catalog_op, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); } TEST(CASDecommissionCatalogDuties, CatalogTokenMovedBetweenOwnershipCutAndRetirementKeepsSlot) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); { auto victim = openVictim(backend); } backend->arm(); @@ -265,12 +288,14 @@ TEST(CASDecommissionCatalogDuties, CatalogTokenMovedBetweenOwnershipCutAndRetire EXPECT_NE(report.warnings.front().find("catalog changed after the victim ownership check"), String::npos) << report.warnings.front(); EXPECT_TRUE(slotObjectExists(*backend, "owner")); - EXPECT_EQ(catalogEntry(*backend, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); + EXPECT_EQ(catalogEntry(catalog_op, Layout("p"), RootNamespace("victim/db/late")).state, NsState::Live); } TEST(CASDecommissionCatalogDuties, FoldedTerminalRemainsGcOwnedAndOnlyRequestsAnotherRound) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const RootNamespace ns("victim/db/folded_terminal"); std::optional life; std::vector stream_before; @@ -287,8 +312,8 @@ TEST(CASDecommissionCatalogDuties, FoldedTerminalRemainsGcOwnedAndOnlyRequestsAn Gc gc(victim, UInt128{811}); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); - ASSERT_EQ(catalogEntry(*backend, victim->layout(), ns).state, NsState::Removing); - for (const ListedKey & key : backend->list(victim->layout().namespaceStreamPrefix(*life), "", 100).keys) + ASSERT_EQ(catalogEntry(catalog_op, victim->layout(), ns).state, NsState::Removing); + for (const ListedKey & key : catalog_op.list(victim->layout().namespaceStreamPrefix(*life), "", 100, Retry::standard()).keys) stream_before.push_back(key.key); ASSERT_FALSE(stream_before.empty()); } @@ -301,9 +326,9 @@ TEST(CASDecommissionCatalogDuties, FoldedTerminalRemainsGcOwnedAndOnlyRequestsAn EXPECT_EQ(wake_requests.load(), 1u); EXPECT_EQ(report.namespaces_already_removed, 1u); EXPECT_FALSE(report.slot_removed); - EXPECT_EQ(catalogEntry(*backend, Layout("p"), ns).state, NsState::Removing); + EXPECT_EQ(catalogEntry(catalog_op, Layout("p"), ns).state, NsState::Removing); std::vector stream_after; - for (const ListedKey & key : backend->list(Layout("p").namespaceStreamPrefix(*life), "", 100).keys) + for (const ListedKey & key : catalog_op.list(Layout("p").namespaceStreamPrefix(*life), "", 100, Retry::standard()).keys) stream_after.push_back(key.key); EXPECT_EQ(stream_after, stream_before) << "decommission must not append a second terminal or become a catalog deletion driver"; @@ -312,19 +337,21 @@ TEST(CASDecommissionCatalogDuties, FoldedTerminalRemainsGcOwnedAndOnlyRequestsAn TEST(CASDecommissionCatalogDuties, OpaqueLifeDebrisWithoutCatalogOwnershipDoesNotBlockRetirement) { auto backend = std::make_shared(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); { auto victim = openVictim(backend); } const Layout layout("p"); const NamespaceLifeId dead_life = NamespaceLifeId::fromCatalogEntry(RootNamespace("historical/name"), UInt128{709}); const String debris_key = layout.refCkptKey(dead_life); - ASSERT_EQ(backend->putIfAbsent(debris_key, "debris").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(debris_key, "debris", Retry::standard()))); const DecommissionReport report = decommissionPoolMember( backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); EXPECT_TRUE(report.warnings.empty()); EXPECT_TRUE(report.slot_removed); - EXPECT_TRUE(backend->head(debris_key).exists); + EXPECT_TRUE(catalog_op.head(debris_key, Retry::standard()).has_value()); } } diff --git a/src/Disks/tests/gtest_cas_detached_work.cpp b/src/Disks/tests/gtest_cas_detached_work.cpp index a35e3a5adf31..2b569de9eb82 100644 --- a/src/Disks/tests/gtest_cas_detached_work.cpp +++ b/src/Disks/tests/gtest_cas_detached_work.cpp @@ -6,12 +6,14 @@ #include #include +#include #include #include #include #include #include #include +#include #include using namespace DB::Cas; @@ -25,12 +27,17 @@ namespace { /// A gate a test opens explicitly, so a task can be held in flight without a sleep. +/// +/// The wait is BOUNDED and reports a failure rather than blocking for ever, and it names the gate it +/// waited on: an unbounded wait on a premise that stopped holding hung the whole binary here, which +/// hid every test that would have run after it. struct Gate { - void wait() + void wait(std::string_view name) { std::unique_lock lock(m); - cv.wait(lock, [this] { return open_; }); + if (!cv.wait_for(lock, std::chrono::seconds(60), [this] { return open_; })) + ADD_FAILURE() << "timed out waiting for '" << name << "'"; } void open() { @@ -43,13 +50,23 @@ struct Gate bool open_ = false; }; -/// Completes the watched first `GET`, then withholds its return so teardown can latch before the +/// Opens its gate on every exit from the scope, so a failing assertion cannot strand the thread +/// parked behind it: the parked thread is joined during teardown, and a gate that stayed shut turned +/// a reported failure into a whole-binary deadlock. +struct GateOpenedOnExit +{ + explicit GateOpenedOnExit(Gate & gate_) : gate(gate_) {} + GateOpenedOnExit(const GateOpenedOnExit &) = delete; + GateOpenedOnExit & operator=(const GateOpenedOnExit &) = delete; + ~GateOpenedOnExit() { gate.open(); } + Gate & gate; +}; + +/// Completes the watched first read, then withholds its return so teardown can latch before the /// helper is able to issue its next raw request. class BetweenRecoveryGetsBackend : public DB::Cas::tests::OrderedFaultBackend { public: - using DB::Cas::tests::OrderedFaultBackend::get; - void armBetweenGets(String first_key_, std::shared_ptr first_completed_, std::shared_ptr release_first_) { first_key = std::move(first_key_); @@ -58,13 +75,13 @@ class BetweenRecoveryGetsBackend : public DB::Cas::tests::OrderedFaultBackend armed.store(true); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - auto result = DB::Cas::tests::OrderedFaultBackend::get(key, range); + auto result = DB::Cas::tests::OrderedFaultBackend::read(key, access); if (key == first_key && armed.exchange(false)) { first_completed->open(); - release_first->wait(); + release_first->wait("release_first"); } return result; } @@ -77,14 +94,11 @@ class BetweenRecoveryGetsBackend : public DB::Cas::tests::OrderedFaultBackend }; /// Identifies recovery's final authority read without changing the recovery implementation: after the -/// recovered-frontier CAS, its first checkpoint `GET` verifies that contribution and its second is the +/// recovered-frontier CAS, its first checkpoint read verifies that contribution and its second is the /// final authority read immediately preceding materialization. class FinalAuthorityBackend : public DB::Cas::tests::OrderedFaultBackend { public: - using DB::Cas::tests::OrderedFaultBackend::casPut; - using DB::Cas::tests::OrderedFaultBackend::get; - void armFinalAuthorityRead(String checkpoint_key_) { checkpoint_key = std::move(checkpoint_key_); @@ -94,20 +108,22 @@ class FinalAuthorityBackend : public DB::Cas::tests::OrderedFaultBackend armed.store(true); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - auto result = DB::Cas::tests::OrderedFaultBackend::get(key, range); + auto result = DB::Cas::tests::OrderedFaultBackend::read(key, access); if (armed.load() && checkpoint_cas_committed.load() && key == checkpoint_key && gets_after_checkpoint_cas.fetch_add(1) + 1 == 2) final_authority_returned.store(true); return result; } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - CasResult result = DB::Cas::tests::OrderedFaultBackend::casPut(key, bytes, expected, meta); - if (armed.load() && key == checkpoint_key && result.outcome == CasOutcome::Committed) + auto result = DB::Cas::tests::OrderedFaultBackend::write(key, bytes, expected_value, access); + /// A value is the store's committed incarnation; a `RawConflict` is a refused precondition. + if (armed.load() && key == checkpoint_key && result.has_value()) checkpoint_cas_committed.store(true); return result; } @@ -125,23 +141,22 @@ class FinalAuthorityBackend : public DB::Cas::tests::OrderedFaultBackend CasRequestBudget oneAttemptBudget() { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; budget.lease_safety_margin_ms = 100; return budget; } -/// A ledger-level fixture keeps the real detached publisher but injects its already-public mount-fence -/// callback. That callback is the existing deterministic boundary after recovery materialized its +/// A ledger-level fixture keeps the real detached publisher but arms `CasRefLedger`'s recovery-install +/// test probe. That probe is the existing deterministic boundary after recovery materialized its /// result and before `installRecoveryResult`. class ManualDetachedLedger { public: ManualDetachedLedger() : backend(std::make_shared()) + , mount_requests(DB::Cas::tests::openRequestsForTest(backend)) , ledger( - backend, + mount_requests, layout, RefLedgerConfig{ .server_root_id = "test", @@ -153,18 +168,9 @@ class ManualDetachedLedger event_sink, oneAttemptBudget(), "test", - [] { return uint64_t{0}; }, [] { return uint64_t{1}; }, [] { return true; }, [] { return uint64_t{1}; }, - [this](uint64_t) - { - if (backend->finalAuthorityReturned() && !final_install_gate_claimed.exchange(true)) - { - final_install_reached.open(); - release_final_install.wait(); - } - }, [] { return uint64_t{0}; }, [] { return true; }, [](const String &, const String &, const std::optional &) {}, @@ -177,7 +183,31 @@ class ManualDetachedLedger {}, [](const RootNamespace &) {}) { - CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the + /// `oneAttemptBudget()` field alone; pair the two so the ledger's own admission arithmetic sees + /// what the budget claims. + backend->setAttemptTimeoutMs(oneAttemptBudget().attempt_timeout_ms); + /// The engine's own retry pauses, on a clock the engine reads: one call reaches its retry + /// deadline against a latched fault with no real time passing. `setCasRetrySleepForTest` + /// installs the sleep on both `mount_requests` and the recovery retry loop. + auto clock = std::make_shared(); + mount_requests.setNowFnForTest(DB::Cas::tests::VirtualRetryClock::nowFnOf(clock)); + ledger.setCasRetrySleepForTest(DB::Cas::tests::VirtualRetryClock::sleepFnOf(clock)); + + CasOperation op = mount_requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + + /// The deterministic boundary a test pauses on: after recovery's final authority read and O(N) + /// materialization, immediately before a materialized result installs. A no-op for every test + /// that never arms `backend`'s final-authority read. + ledger.setRecoveryInstallProbeForTest([this] + { + if (backend->finalAuthorityReturned() && !final_install_gate_claimed.exchange(true)) + { + final_install_reached.open(); + release_final_install.wait("release_final_install"); + } + }); } std::function takeDetachedTask() @@ -203,6 +233,9 @@ class ManualDetachedLedger std::shared_ptr registry = std::make_shared(); Gate final_install_reached; Gate release_final_install; + /// The mount plane the ledger admits every request on. Declared before it, and never moved: the + /// ledger holds a reference to this member. + CasRequests mount_requests; CasRefLedger ledger; private: @@ -239,8 +272,14 @@ std::shared_ptr openTestStorage(bool tiny_b storage->startup(); if (tiny_budget) { - storage->poolForTest()->setDetachedDrainDeadlineBudgetForTest( - /*attempt_timeout_ms=*/10, /*lease_safety_margin_ms=*/10); + /// `connect_timeout_cap_ms` is set explicitly (not left at whatever the pool froze at open) + /// so the drain deadline -- `attemptEnvelopeMs() + lease_safety_margin_ms` -- is really the + /// tiny 20 ms this test wants, not 10 ms of attempt timeout plus a hidden connect-cap tax. + CasRequestBudget budget; + budget.attempt_timeout_ms = 10; + budget.lease_safety_margin_ms = 10; + budget.connect_timeout_cap_ms = 0; + storage->poolForTest()->setDetachedDrainDeadlineBudgetForTest(budget); } return storage; } @@ -255,10 +294,14 @@ PoolPtr openPublishingPool(const std::shared_ptrsetAttemptTimeoutMs(config.cas_request_budget.attempt_timeout_ms); return Pool::open(backend, config); } @@ -295,8 +338,13 @@ RefTxnId publishRef(CasRefLedger & ledger, const RootNamespace & ns, const Strin } /// Leaves the runtime in `NeedsRecovery` while its first background publisher is held after capture. -/// Releasing that publisher consumes the armed snapshot failure; zero backoff then makes settlement +/// Releasing that publisher meets the latched snapshot failure; zero backoff then makes settlement /// redispatch the real token-carrying publisher, whose first action is recovery of this exact runtime. +/// +/// Both faults are LATCHED and the engine's retry clock is virtual, because a write here must reach +/// its own retry deadline without committing: the engine reissues one logical write until that +/// deadline, so a counted fault the reissues outlive would let the call commit -- the parked publisher +/// would then succeed, nothing would redispatch it, and no caller would ever reach recovery. void preparePendingRecoveryPublisher( const PoolPtr & store, const std::shared_ptr & backend, @@ -305,6 +353,8 @@ void preparePendingRecoveryPublisher( const std::shared_ptr & release_first_publisher, String & ckpt_key) { + DB::Cas::tests::VirtualRetryClock::installOn(store); + auto capture_calls = std::make_shared>(0); store->setSnapshotAfterCaptureHookForTest( [capture_calls, first_publisher_captured, release_first_publisher] @@ -312,21 +362,23 @@ void preparePendingRecoveryPublisher( if (capture_calls->fetch_add(1) != 0) return; first_publisher_captured->open(); - release_first_publisher->wait(); + release_first_publisher->wait("release_first_publisher"); }); - backend->armPutFailure("_snap/", 1); + backend->armLatchedWriteFailure("_snap/"); ASSERT_NO_THROW(publishRef(store, ns, "ref_1", 1)); - first_publisher_captured->wait(); + first_publisher_captured->wait("first_publisher_captured"); - const auto life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + const auto life = CasRefCatalog::lifeIfCataloged(catalog_op, store->layout(), ns); ASSERT_TRUE(life); ckpt_key = store->layout().refCkptKey(*life); /// The log lands before the checkpoint conflict, leaving a real unfrontiered durable transaction. - backend->armCasConflict(ckpt_key, 100); + backend->armLatchedWriteConflict(ckpt_key); EXPECT_ANY_THROW(store->dropRef(ns, "ref_1")); - backend->armCasConflict(ckpt_key, 0); + backend->armLatchedWriteConflict({}); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); } @@ -371,9 +423,9 @@ TEST(CASDetachedWork, DrainDoesNotReturnWhileWorkIsInFlight) ASSERT_TRUE(store->tryDispatchDetached([entered, release](DetachedStopToken) { entered->open(); - release->wait(); + release->wait("release"); })); - entered->wait(); + entered->wait("entered"); auto drain = std::async(std::launch::async, [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/60000); }); @@ -398,9 +450,9 @@ TEST(CASDetachedWork, ShutdownDoesNotReturnWhileWorkIsInFlight) ASSERT_TRUE(pool->tryDispatchDetached([entered, release](DetachedStopToken) { entered->open(); - release->wait(); + release->wait("release"); })); - entered->wait(); + entered->wait("entered"); auto done = std::async(std::launch::async, [&storage] { storage->shutdown(); }); awaitStopLatched(pool); @@ -423,10 +475,10 @@ TEST(CASDetachedWork, ImplicitDestructionDrains) ASSERT_TRUE(pool->tryDispatchDetached([entered, release, &finished](DetachedStopToken) { entered->open(); - release->wait(); + release->wait("release"); finished.store(true); })); - entered->wait(); + entered->wait("entered"); auto destroyed = std::async(std::launch::async, [&storage] { storage.reset(); }); awaitStopLatched(pool); @@ -462,9 +514,9 @@ TEST(CASDetachedWork, ExpiredDrainIncrementsTheTimeoutCounter) ASSERT_TRUE(pool->tryDispatchDetached([entered, release](DetachedStopToken) { entered->open(); - release->wait(); + release->wait("release"); })); - entered->wait(); + entered->wait("entered"); const auto before = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts] .load(std::memory_order_relaxed); @@ -490,10 +542,10 @@ TEST(CASDetachedWork, TaskObservesStopTokenOnceLatched) ASSERT_TRUE(store->tryDispatchDetached([entered, release, &saw_stop](DetachedStopToken token) { entered->open(); - release->wait(); + release->wait("release"); saw_stop.store(token.stopping()); })); - entered->wait(); + entered->wait("entered"); auto drain = std::async(std::launch::async, [&store] { return store->stopAndDrainDetachedWork(/*deadline_ms=*/60000); }); @@ -561,24 +613,163 @@ TEST(CASDetachedWork, FailedPublisherDispatchKeepsMutationAndClearsReservation) /// Settlement must survive a throwing error handler. Today it is a bare call after the handler, so a /// handler that throws skips it and strands the reservation for the life of the process. /// -/// The publish is FAULTED deliberately: with a healthy backend it would succeed, the handler would -/// never run, and this test would pass while exercising nothing. +/// The dispatched attempt is made to throw deliberately, via `snapshot_after_capture_hook_for_test` +/// (called inline, unguarded by any inner `catch`, so its throw reaches `dispatchSnapshotPublisher`'s +/// outer `catch (...)` that invokes `publish_error_hook_for_test`): with a healthy backend and no +/// injected throw the publish would succeed, the handler would never run, and this test would pass +/// while exercising nothing. An ordinary FAULTED WRITE does not reach the handler at all -- +/// `tryPublishSnapshotAndAdvanceCheckpointOnceOnRuntimeImpl` treats every non-`Committed` `WriteResult` +/// (including a genuine retry give-up) as an ordinary backoff-and-retry-later outcome, a plain return, +/// never a throw -- so only an exception from OUTSIDE that write (this hook stands in for one) ever +/// reaches the handler. +/// +/// The hook fires (and throws) EXACTLY ONCE, then disarms itself before throwing. An exception escaping +/// before any of the write's own failure arms now reaches the detached task's `catch (...)`, which arms +/// `advancePublishBackoff` on that exit exactly as the ordinary failure arms do, so this throw paces the +/// redispatch instead of driving it at full speed. Disarming after one throw lets the second dispatch +/// take the healthy path and settle, keeping this test's claim narrow: settlement survives ONE throwing +/// handler call. TEST(CASDetachedWork, SettlementSurvivesAThrowingErrorHandler) { auto backend = std::make_shared(); + /// Held in shared, heap-owned atomics, not plain locals: an `ASSERT_*` below can return early and + /// skip the `stopAndDrainDetachedWork` cleanup at the bottom, and even a successful drain there only + /// guarantees no NEW detached task starts -- an already-dispatched one can still be running and can + /// still invoke these hooks against a frame that has already unwound. + auto handler_ran = std::make_shared>(false); PoolConfig config; - config.publish_error_hook_for_test - = [] { throw std::runtime_error("injected: the error handler itself throws"); }; + /// No real backoff wait: the injected throw leaves the tail over-threshold, and a REAL backoff + /// sleep here would still be paid at teardown drain even though the test's own assertions never + /// wait on it directly. + config.snapshot_publish_backoff_initial_ms = 0; + config.snapshot_publish_backoff_max_ms = 0; + config.publish_error_hook_for_test = [handler_ran] + { + handler_ran->store(true); + throw std::runtime_error("injected: the error handler itself throws"); + }; auto store = openPublishingPool(backend, config); const RootNamespace ns{"srv1/handler_throws"}; - /// Arm the fault so the publisher's own PUT fails and its `catch` is entered. Use the same arming - /// call the snapshot-ordering suite uses against this backend. - backend->armPutFailure("_snap/", 1); + auto capture_hook_ran = std::make_shared>(false); + auto capture_hook_armed = std::make_shared>(true); + store->setSnapshotAfterCaptureHookForTest([capture_hook_ran, capture_hook_armed] + { + capture_hook_ran->store(true); + if (capture_hook_armed->exchange(false)) + throw std::runtime_error("injected: the dispatched attempt itself throws"); + }); ASSERT_NO_THROW(publishRef(store, ns, "ref_1", 1)); store->waitForSnapshotPublishSettleForTest(ns); + ASSERT_TRUE(capture_hook_ran->load()) << "the dispatched attempt never reached the injected throw"; + EXPECT_TRUE(handler_ran->load()) << "the injected throw must have reached the (throwing) error handler"; EXPECT_EQ(store->pendingSnapshotPublishesForTest(ns), 0); + + /// Same lifetime rule as below: the detached publisher reads the hooks' captured state, so it is + /// stopped and drained before this function returns. + ASSERT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/10000)); + store->setSnapshotAfterCaptureHookForTest(nullptr); +} + +/// A publish attempt that throws BEFORE any of the ordinary write-failure arms must pace exactly like +/// an ordinary failure: settlement's only pacing gate is the publish backoff deadline, so an exception +/// that leaves it unarmed redispatches the publisher at full speed for as long as the fault persists. +/// The clock is the injected boot clock, so the schedule is virtual and the test spends no wall time +/// waiting one out; the one real-time wait is the bounded observation window, because an unpaced +/// redispatch runs on a background thread and has to be caught in the act rather than waited out. +TEST(CASDetachedWork, ThrowingPublishAttemptIsPacedByTheBackoff) +{ + auto backend = std::make_shared(); + constexpr uint64_t step_ms = 100; + /// Held in shared, heap-owned atomics, not plain locals: `stopAndDrainDetachedWork` at the bottom + /// permanently closes admission of NEW detached tasks (via `beginTeardown`), but an `ASSERT_*` above + /// it can return early and skip that call entirely, and even a call that runs only guarantees no new + /// task starts -- an already-dispatched redispatch can still be running (or the drain can simply time + /// out) and can still invoke these hooks against a frame that has already unwound. + auto fake_boot = std::make_shared>(1000); + auto error_hook_calls = std::make_shared>(0); + PoolConfig config; + config.boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }; + /// Initial == max, so every step of the schedule is the same virtual `step_ms` and the test can + /// advance the clock by a constant. + config.snapshot_publish_backoff_initial_ms = step_ms; + config.snapshot_publish_backoff_max_ms = step_ms; + config.publish_error_hook_for_test = [error_hook_calls] + { + error_hook_calls->fetch_add(1); + }; + auto store = openPublishingPool(backend, config); + const RootNamespace ns{"srv1/throwing_publisher_pacing"}; + + auto attempts = std::make_shared>(0); + store->setSnapshotAfterCaptureHookForTest([attempts] + { + attempts->fetch_add(1); + throw std::runtime_error("injected: every publish attempt throws before its write"); + }); + + ASSERT_NO_THROW(publishRef(store, ns, "ref_1", 1)); + + /// The virtual clock does not move here, so the armed deadline is still in the future for the whole + /// window and exactly ONE attempt may have run. + const auto observe_until = std::chrono::steady_clock::now() + std::chrono::milliseconds(500); + while (std::chrono::steady_clock::now() < observe_until && attempts->load() <= 1) + std::this_thread::yield(); + EXPECT_EQ(attempts->load(), 1u) << "the throwing attempt redispatched without arming the publish backoff"; + EXPECT_GE(error_hook_calls->load(), 1u) << "the injected throw never reached the error handler"; + + /// One step of the schedule per iteration: the tail is still over threshold, so each mutation + /// re-evaluates admission, and AT MOST one attempt may pass per elapsed backoff interval. At most, + /// not exactly: a publisher dispatched by a mutation whose append has not yet returned the lane to + /// `Ready` is refused at that gate, and the refusal arms the same backoff without the attempt ever + /// reaching the hook below -- so a step can legitimately elapse with no attempt of its own. The + /// regression this test exists for is the opposite, an unpaced redispatch storm, and the ceiling + /// is what catches it; progress is asserted once after the loop. + for (uint64_t step = 1; step <= 3; ++step) + { + fake_boot->fetch_add(step_ms); + ASSERT_NO_THROW(publishRef(store, ns, "ref_" + std::to_string(step + 1), step + 1)); + /// Bounded poll rather than `waitForSnapshotPublishSettleForTest`: that call waits on a condvar + /// predicate with no deadline, and on an unpaced-redispatch regression the reservation count + /// never rests at zero long enough for the predicate to observe it, hanging the test instead of + /// failing it. + const auto settle_deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->pendingSnapshotPublishesForTest(ns) != 0) + { + ASSERT_LT(std::chrono::steady_clock::now(), settle_deadline) + << "the snapshot publish for '" << ns.string() << "' never settled: " + << "pending_snapshot_publishes stayed nonzero"; + std::this_thread::yield(); + } + EXPECT_LE(attempts->load(), 1 + step) << "more than one publish attempt ran within one backoff step"; + } + + /// Progress, on the injected clock so it is deterministic rather than a race with a worker: an + /// elapsed backoff must eventually admit a further attempt, or the pacing gate would be a wedge. + for (uint64_t extra = 0; attempts->load() < 2 && extra < 20; ++extra) + { + fake_boot->fetch_add(step_ms); + ASSERT_NO_THROW(publishRef(store, ns, "ref_progress_" + std::to_string(extra), 100 + extra)); + const auto settle = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (store->pendingSnapshotPublishesForTest(ns) != 0) + { + ASSERT_LT(std::chrono::steady_clock::now(), settle) << "a publish never settled"; + std::this_thread::yield(); + } + } + EXPECT_GE(attempts->load(), 2u) + << "no elapsed backoff ever admitted a further publish attempt: the gate is a wedge, not a pace"; + + EXPECT_EQ(error_hook_calls->load(), attempts->load()); + /// The publisher is detached work: a redispatch admitted by the last elapsed backoff can still be + /// running when this body returns, and it reads `fake_boot` through `boot_ms_fn`. Stop and drain it + /// while the locals it reads are alive. + ASSERT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/10000)); + store->setSnapshotAfterCaptureHookForTest(nullptr); } /// A publisher asleep in recovery backoff must be woken by the stop, not waited out. The injected @@ -609,9 +800,11 @@ TEST(CASDetachedWork, StopWakesRecoveryBackoffSleep) }); /// Exhaust one checkpoint publication inside recovery so the outer retry loop enters backoff. - backend->armCasConflict(ckpt_key, 100); + /// Latched: the publication must meet the refusal on every reissue, or it commits and recovery + /// never reaches its backoff. + backend->armLatchedWriteConflict(ckpt_key); release_first_publisher->open(); - sleeping->wait(); + sleeping->wait("sleeping"); const bool drained = store->stopAndDrainDetachedWork(/*deadline_ms=*/5000); release_sleep->store(true); @@ -635,7 +828,7 @@ TEST(CASDetachedWork, StopWakesAConcurrentRecoveryWaiter) if (!recovery_hook_armed->load()) return; recovery_entered->open(); - release_recovery->wait(); + release_recovery->wait("release_recovery"); }; auto store = openPublishingPool(backend, config); const RootNamespace ns{"srv1/concurrent_recovery"}; @@ -650,7 +843,10 @@ TEST(CASDetachedWork, StopWakesAConcurrentRecoveryWaiter) /// drain below waits only for the publisher parked behind it. recovery_hook_armed->store(true); auto first_recovery = std::async(std::launch::async, [&store, &ns] { store->listRefs(ns); }); - recovery_entered->wait(); + /// Declared AFTER the future, so it is destroyed BEFORE it: `first_recovery`'s destructor joins + /// the recovery thread, which cannot leave the hook until this gate is open. + GateOpenedOnExit recovery_released{*release_recovery}; + recovery_entered->wait("recovery_entered"); release_first_publisher->open(); const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); @@ -688,7 +884,7 @@ TEST(CASDetachedWork, StopIsObservedBeforeTheFirstRecoveryRequest) if (!recovery_hook_armed->load()) return; at_boundary->open(); - release->wait(); + release->wait("release"); }; auto store = openPublishingPool(backend, config); const RootNamespace ns{"srv1/pre_first_request"}; @@ -701,7 +897,7 @@ TEST(CASDetachedWork, StopIsObservedBeforeTheFirstRecoveryRequest) recovery_hook_armed->store(true); release_first_publisher->open(); - at_boundary->wait(); + at_boundary->wait("at_boundary"); const uint64_t gets_before = backend->getTotal(); auto drain = std::async(std::launch::async, @@ -769,14 +965,16 @@ TEST(CASDetachedWork, StopBetweenSnapshotBaseRequestsPreventsThePredecessorGet) preparePendingRecoveryPublisher( store, backend, ns, first_publisher_captured, release_first_publisher, ckpt_key); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(catalog_op, store->layout(), ns).value(); const String base_log_key = store->layout().refLogKey(life, base_id); const String predecessor_key = store->layout().refLogKey(life, predecessor_seal_id); auto first_get_completed = std::make_shared(); auto release_first_get = std::make_shared(); backend->armBetweenGets(base_log_key, first_get_completed, release_first_get); release_first_publisher->open(); - first_get_completed->wait(); + first_get_completed->wait("first_get_completed"); const uint64_t predecessor_gets_before = backend->getCount(predecessor_key); auto drain = std::async(std::launch::async, @@ -798,11 +996,15 @@ TEST(CASDetachedWork, StopAfterRecoveryMaterializationPreventsFinalInstall) const RootNamespace ns{"srv1/stop_before_recovery_install"}; ASSERT_NO_THROW(publishRef(fixture.ledger, ns, "ref_1", 1)); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*fixture.backend, fixture.layout, ns).value(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(fixture.backend); + CasOperation catalog_op = catalog_requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(catalog_op, fixture.layout, ns).value(); const String ckpt_key = fixture.layout.refCkptKey(life); - fixture.backend->armCasConflict(ckpt_key, 100); + /// Latched: the checkpoint write is reissued until its retry window closes, so a counted refusal + /// the reissues outlive would let this drop commit and leave the lane Ready. + fixture.backend->armLatchedWriteConflict(ckpt_key); EXPECT_ANY_THROW(fixture.ledger.dropRef(ns, "ref_1")); - fixture.backend->armCasConflict(ckpt_key, 0); + fixture.backend->armLatchedWriteConflict({}); ASSERT_EQ(fixture.ledger.laneStateForTest(ns), RefLaneState::NeedsRecovery); auto detached_publisher = fixture.takeDetachedTask(); @@ -813,8 +1015,11 @@ TEST(CASDetachedWork, StopAfterRecoveryMaterializationPreventsFinalInstall) { task(DetachedStopToken(fixture.registry)); }); + /// Released on every exit, before `running` is joined: the task parks behind this gate, and a + /// failed assertion that left it shut deadlocked the join. + GateOpenedOnExit final_install_released{fixture.release_final_install}; - fixture.final_install_reached.wait(); + fixture.final_install_reached.wait("final_install_reached"); fixture.latchStop(); fixture.release_final_install.open(); EXPECT_NO_THROW(running.get()); diff --git a/src/Disks/tests/gtest_cas_empty_proof.cpp b/src/Disks/tests/gtest_cas_empty_proof.cpp index 4754786e7624..00d43b9214d1 100644 --- a/src/Disks/tests/gtest_cas_empty_proof.cpp +++ b/src/Disks/tests/gtest_cas_empty_proof.cpp @@ -17,7 +17,7 @@ /// (Live) or READ-ONLY pool, an enumeration about to answer EMPTY at a table root must first CONFIRM the /// pool identity object (`_pool_meta`) exists with an AUTHORITATIVE, UNCACHED probe -- because "empty" at a /// table root is exactly what a silently-erased backing looks like, and a read-only pool has no -/// keeper/lease/observer to catch that erasure any other way. These tests build a real +/// renewer/lease/observer to catch that erasure any other way. These tests build a real /// `ContentAddressedMetadataStorage` over a Local object storage (the gtest_cas_operation_gate.cpp harness) /// and exercise the rule across the six cells the brief enumerates. diff --git a/src/Disks/tests/gtest_cas_encoding_pins.cpp b/src/Disks/tests/gtest_cas_encoding_pins.cpp index f4c3a913fda7..fadc15bdb62d 100644 --- a/src/Disks/tests/gtest_cas_encoding_pins.cpp +++ b/src/Disks/tests/gtest_cas_encoding_pins.cpp @@ -2,6 +2,10 @@ #include #include #include +#include +#include +#include +#include #include #include #include @@ -9,11 +13,40 @@ using namespace DB; using namespace DB::Cas; -/// These literals pin the CANONICAL BYTES of the CAS text encoders as of the commit that -/// introduced this file. The CasJsonWriter migration (2026-07-20 spec) must keep every one of -/// them green UNMODIFIED: canonical text is byte-compared on retries and deterministic adoption, -/// and the incremental ref budget counters assume these exact sizes. Never edit an expected -/// string here to make a test pass — that means the encoder's bytes drifted, which is the bug. +namespace +{ +String lineAt(const String & text, size_t index) +{ + size_t begin = 0; + for (size_t i = 0; i < index; ++i) + begin = text.find('\n', begin) + 1; + const size_t end = text.find('\n', begin); + return text.substr(begin, end - begin + 1); +} + +void expectDelta(const String & old_bytes, const String & new_bytes, size_t expected) +{ + EXPECT_EQ(new_bytes.size() - old_bytes.size(), expected) << "old: " << old_bytes << "new: " << new_bytes; +} + +CasFoldSeal oneFoldSeal() +{ + CasFoldSeal seal; + seal.generation = 5; + seal.parent_generation = 4; + return seal; +} +} + +/// The `CASEncodingPins` literals below pin the CANONICAL BYTES of the CAS text encoders: canonical +/// text is byte-compared on retries and deterministic adoption, and the incremental ref budget +/// counters assume these exact sizes. Never edit one of those expected strings to make a test pass — +/// that means the encoder's bytes drifted, which is the bug. +/// +/// The `CASWireCutDeltas` literals are the opposite kind: each is a HISTORICAL pre-cut row, kept so +/// the cost of the semantic-key rename stays measurable against what it replaced. They are +/// deliberately not the current bytes and must never be refreshed toward them — a delta measured +/// against today's encoder on both sides would always be zero. TEST(CASEncodingPins, RefLogTxnAllOpKinds) { @@ -38,7 +71,7 @@ TEST(CASEncodingPins, RefLogTxnAllOpKinds) /// not quote/newline/control bytes/U+2028, so `ref_name` -- the only free-form string `RefOp` /// still carries now that `payload` is gone -- exercises quote, newline, a bare control byte, /// and the three-byte U+2028 sequence. Backslash escaping is pinned separately, over an - /// unrestricted string, by `gtest_cas_json_writer.cpp`'s `CASJsonWriterEscaping` suite. + /// unrestricted string, by the JSON-writer escaping suite. set_published_at.ref_name = String("20260101_0_1_1_1\"c\nd") + "\x01" "e" + "\xE2\x80\xA8" "f"; set_published_at.expected_manifest_ref = ManifestRef{1, 2, 3}; set_published_at.published_at_ms = 1234; @@ -49,13 +82,13 @@ TEST(CASEncodingPins, RefLogTxnAllOpKinds) txn.ops.push_back(removal); const String expected = fmt::format("{{\"type\":\"cas_ref_log\",\"v\":{}}}\n", currentCompatibilityVersion()) + - "{\"ns\":\"roots/pin\",\"we\":\"7\",\"rs\":\"9\"}\n" + "{\"namespace\":\"roots/pin\",\"txn_epoch\":\"7\",\"txn_seq\":\"9\"}\n" "{\"op\":\"namespace_birth\"}\n" - "{\"op\":\"owner_transition\",\"obk\":\"precommit\",\"orn\":\"20260101_0_1_1_1\"," - "\"ome\":\"1\",\"omb\":\"2\",\"omo\":3,\"nbk\":\"committed\",\"nrn\":\"20260101_0_1_1_1\"," - "\"nme\":\"1\",\"nmb\":\"2\",\"nmo\":3}\n" - "{\"op\":\"set_published_at\",\"rn\":\"20260101_0_1_1_1\\\"c\\nd\\u0001e\\u2028f\"," - "\"me\":\"1\",\"mb\":\"2\",\"mo\":3,\"ts\":1234}\n" + "{\"op\":\"owner_transition\",\"old_kind\":\"precommit\",\"old_ref\":\"20260101_0_1_1_1\"," + "\"old_epoch\":\"1\",\"old_build\":\"2\",\"old_ord\":3,\"new_kind\":\"committed\",\"new_ref\":\"20260101_0_1_1_1\"," + "\"new_epoch\":\"1\",\"new_build\":\"2\",\"new_ord\":3}\n" + "{\"op\":\"set_published_at\",\"ref\":\"20260101_0_1_1_1\\\"c\\nd\\u0001e\\u2028f\"," + "\"epoch\":\"1\",\"build\":\"2\",\"ord\":3,\"published_ms\":1234}\n" "{\"op\":\"remove_namespace\"}\n" "{\"n\":4}\n"; EXPECT_EQ(encodeRefLogTxn(txn), expected); @@ -76,9 +109,9 @@ TEST(CASEncodingPins, RefSnapshotLive) snap.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "20260102_0_2_2_2", ManifestRef{4, 5, 6}}); const String expected = fmt::format("{{\"type\":\"cas_ref_snap\",\"v\":{}}}\n", currentCompatibilityVersion()) + - "{\"ns\":\"roots/pin\",\"we\":\"7\",\"rs\":\"9\",\"lc\":\"live\"}\n" - "{\"k\":\"c\",\"rn\":\"20260101_0_1_1_1\",\"me\":\"1\",\"mb\":\"2\",\"mo\":3,\"ts\":5}\n" - "{\"k\":\"p\",\"rn\":\"20260102_0_2_2_2\",\"me\":\"4\",\"mb\":\"5\",\"mo\":6}\n" + "{\"namespace\":\"roots/pin\",\"snapshot_epoch\":\"7\",\"snapshot_seq\":\"9\",\"lifecycle\":\"live\"}\n" + "{\"kind\":\"committed\",\"ref\":\"20260101_0_1_1_1\",\"epoch\":\"1\",\"build\":\"2\",\"ord\":3,\"published_ms\":5}\n" + "{\"kind\":\"precommit\",\"ref\":\"20260102_0_2_2_2\",\"epoch\":\"4\",\"build\":\"5\",\"ord\":6}\n" "{\"n\":2}\n"; EXPECT_EQ(encodeRefTableSnapshot(snap), expected); } @@ -91,20 +124,266 @@ TEST(CASEncodingPins, SourceEdgeRunLines) SourceEdgeRecord active; active.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(2))}; active.source_id = UInt128(5); - active.marker = kEdgeActive; + active.marker = RunMarker::Edge; writer.append(active); + SourceEdgeRecord condemned; + condemned.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(3))}; + condemned.source_id = UInt128(0); + condemned.marker = RunMarker::Condemned; + condemned.delete_pending = true; + condemned.token = PersistedEtag{"etag", "token"}; + condemned.size = 9; + condemned.condemn_round = 7; + condemned.marker_confirmed = true; + writer.append(condemned); + writer.finish(); out.finalize(); - /// The exact "b" rendering (algo byte + digest hex) is pinned as a whole line; the point is - /// that Task 8's line-scratch rewrite must reproduce it byte-for-byte. + /// The exact `ref` rendering (algo byte + digest hex) is pinned as a whole line; the point is + /// that the line-scratch rendering must reproduce it byte-for-byte. const String text = out.str(); const String header = fmt::format("{{\"type\":\"cas_run\",\"v\":{},\"kind\":\"source_edge\"}}\n", currentCompatibilityVersion()); const String expected_record = - "{\"b\":\"0100000000000000000000000000000002\",\"s\":\"00000000000000000000000000000005\",\"m\":\"edge\"}\n"; - const String trailer = "{\"n\":1}\n"; - /// There is exactly one record, so the whole buffer must be byte-identical to header + record + trailer. - const String expected_full = header + expected_record + trailer; + "{\"ref\":\"0100000000000000000000000000000002\",\"src\":\"00000000000000000000000000000005\",\"mark\":\"edge\"}\n"; + const String expected_condemned = + "{\"ref\":\"0100000000000000000000000000000003\",\"src\":\"00000000000000000000000000000000\",\"mark\":\"condemned\",\"pending\":true,\"token_type\":\"etag\",\"token\":\"token\",\"size\":9,\"condemn_round\":\"7\",\"confirmed\":true}\n"; + const String trailer = "{\"n\":2}\n"; + /// Both records must remain byte-identical to their canonical stored representation. + const String expected_full = header + expected_record + expected_condemned + trailer; EXPECT_EQ(text, expected_full) << text; } + +TEST(CASWireCutDeltas, ActiveCasRunRow) +{ + SourceEdgeRecord record{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(2))}, .source_id = UInt128(5), .marker = RunMarker::Edge}; + WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(record); + writer.finish(); + out.finalize(); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"b\":\"0100000000000000000000000000000002\",\"s\":\"00000000000000000000000000000005\",\"m\":\"edge\"}\n"; + expectDelta(old_bytes, lineAt(out.str(), 1), 7); +} + +TEST(CASWireCutDeltas, CondemnedCasRunRow) +{ + SourceEdgeRecord record{.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(3))}, .source_id = UInt128(0), .marker = RunMarker::Condemned, .delete_pending = true, .token = PersistedEtag{"etag", "token"}, .size = 9, .condemn_round = 7, .marker_confirmed = true}; + WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(record); + writer.finish(); + out.finalize(); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"b\":\"0100000000000000000000000000000003\",\"s\":\"00000000000000000000000000000000\",\"m\":\"condemned\",\"pend\":true,\"tt\":\"etag\",\"tv\":\"token\",\"sz\":9,\"cr\":\"7\",\"mc\":true}\n"; + expectDelta(old_bytes, lineAt(out.str(), 1), 41); +} + +TEST(CASWireCutDeltas, BlobPartManifestEntry) +{ + PartManifest manifest; + manifest.ref = ManifestRef{1, 2, 3}; + manifest.root_namespace_id = RootNamespace{"root"}; + manifest.entries = {ManifestEntry{"a", EntryPlacement::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(4))}, 9, {}}}; + const String text = encodePartManifest(manifest); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"p\":\"a\",\"pm\":\"blob\",\"ha\":\"ch128\",\"h\":\"00000000000000000000000000000004\",\"sz\":9}\n"; + expectDelta(old_bytes, lineAt(text, 2), 15); +} + +TEST(CASWireCutDeltas, InlinePartManifestEntry) +{ + PartManifest manifest; + manifest.ref = ManifestRef{1, 2, 3}; + manifest.root_namespace_id = RootNamespace{"root"}; + manifest.entries = {ManifestEntry{"a", EntryPlacement::Inline, {}, 0, "x"}}; + const String text = encodePartManifest(manifest); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"p\":\"a\",\"pm\":\"inline\",\"il\":1}\n"; + expectDelta(old_bytes, lineAt(text, 2), 8); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_banner = "==> \"a\" il=1 <==\n"; + expectDelta(old_banner, lineAt(text, 4), 2); +} + +TEST(CASWireCutDeltas, GcOutcomesRow) +{ + OutcomeLog log{{OutcomeEntry{ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(4))}, PersistedEtag{"etag", "t"}, OutcomeKind::Deleted}}}; + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00000000000000000000000000000004\",\"tt\":\"etag\",\"tv\":\"t\",\"oc\":\"deleted\"}\n"; + expectDelta(old_bytes, lineAt(encodeOutcomeLog(log), 1), 26); +} + +/// The ref-log's own op rows: the highest-cardinality record of the format and, for +/// `owner_transition`, the largest single-row cost of the whole cut -- both old-side groups and both +/// new-side groups are renamed at once. +TEST(CASWireCutDeltas, OwnerTransitionRefLogOpRow) +{ + RefLogTxn txn; + txn.ns = "root"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "r", ManifestRef{3, 4, 5}}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "r", ManifestRef{3, 4, 5}}; + txn.ops.push_back(op); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"op\":\"owner_transition\",\"obk\":\"precommit\",\"orn\":\"r\",\"ome\":\"3\",\"omb\":\"4\",\"omo\":5," + "\"nbk\":\"committed\",\"nrn\":\"r\",\"nme\":\"3\",\"nmb\":\"4\",\"nmo\":5}\n"; + expectDelta(old_bytes, lineAt(encodeRefLogTxn(txn), 2), 50); +} + +TEST(CASWireCutDeltas, SetPublishedAtRefLogOpRow) +{ + RefLogTxn txn; + txn.ns = "root"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::SetPublishedAt; + op.ref_name = "r"; + op.expected_manifest_ref = ManifestRef{3, 4, 5}; + op.published_at_ms = 6; + txn.ops.push_back(op); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"op\":\"set_published_at\",\"rn\":\"r\",\"me\":\"3\",\"mb\":\"4\",\"mo\":5,\"ts\":6}\n"; + expectDelta(old_bytes, lineAt(encodeRefLogTxn(txn), 2), 18); +} + +/// The body-less ops are the cut's only free rows: the record is the `op` key alone. This measures +/// that the ROW costs nothing extra, not that the word itself is unchanged -- it builds its old side +/// from the current word, so it cannot see a word rename. The words are pinned literally by the +/// closed-set tests; what this adds is that no framing crept in around them. +TEST(CASWireCutDeltas, BodylessRefLogOpRowsAreUnchanged) +{ + for (const RefOpKind kind : {RefOpKind::NamespaceBirth, RefOpKind::EpochSeal}) + { + RefLogTxn txn; + txn.ns = "root"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = kind; + txn.ops.push_back(op); + const String old_bytes = fmt::format("{{\"op\":\"{}\"}}\n", refOpKindToWireWord(kind)); + expectDelta(old_bytes, lineAt(encodeRefLogTxn(txn), 2), 0); + } +} + +TEST(CASWireCutDeltas, CommittedRefSnapshotRow) +{ + RefTableSnapshot snapshot; + snapshot.ns = "root"; + snapshot.snapshot_id = RefTxnId{1, 2}; + snapshot.committed.push_back(RefCommittedRow{"r", ManifestRef{3, 4, 5}, 6}); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"c\",\"rn\":\"r\",\"me\":\"3\",\"mb\":\"4\",\"mo\":5,\"ts\":6}\n"; + expectDelta(old_bytes, lineAt(encodeRefTableSnapshot(snapshot), 2), 29); +} + +TEST(CASWireCutDeltas, PrecommitRefSnapshotRow) +{ + RefTableSnapshot snapshot; + snapshot.ns = "root"; + snapshot.snapshot_id = RefTxnId{1, 2}; + snapshot.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "r", ManifestRef{3, 4, 5}}); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"p\",\"rn\":\"r\",\"me\":\"3\",\"mb\":\"4\",\"mo\":5}\n"; + expectDelta(old_bytes, lineAt(encodeRefTableSnapshot(snapshot), 2), 19); +} + +TEST(CASWireCutDeltas, BaseRefCatalogRow) +{ + RefCatalog catalog{{CatalogEntry{.ns = RootNamespace{"root"}, .state = NsState::Live, .incarnation = UInt128(7)}}}; + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"ent\",\"ns\":\"root\",\"st\":\"live\",\"inc\":\"00000000000000000000000000000007\"}\n"; + expectDelta(old_bytes, lineAt(encodeRefCatalog(catalog), 1), 9); +} + +/// The base row's delta is 22 bytes of keys and tags plus the `class` word, which costs one byte more +/// than its length (quotes, less the single numeric digit it replaces). Each word is measured +/// separately: a range over all four would accept a key rename hiding inside the spread, and a single +/// fixture would pin only one point of it. `clamped` cannot be a base row at all -- the grammar +/// requires a hold on exactly those rows -- so it is measured whole and its base part recovered by +/// subtracting the hold segment the next test pins. +TEST(CASWireCutDeltas, BaseRefLifeFoldSealRow) +{ + const auto base_delta = [](CoverageClass classification, uint8_t old_wire_value) + { + CasFoldSeal seal = oneFoldSeal(); + seal.ref_lives[UInt128(1)].coverage + = RefCoverage{.classification = classification, .last_folded_ref_id = RefTxnId{7, 11}}; + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = fmt::format( + "{{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":{},\"lfe\":\"7\",\"lfs\":\"11\"}}\n", + old_wire_value); + return lineAt(encodeFoldSeal(seal), 2).size() - old_bytes.size(); + }; + + /// The pre-cut wire numbered these 0/1/2, not the current enum's values. + EXPECT_EQ(base_delta(CoverageClass::Absent, 0), 29u); /// 22 + "absent" + EXPECT_EQ(base_delta(CoverageClass::Unchanged, 1), 32u); /// 22 + "unchanged" + EXPECT_EQ(base_delta(CoverageClass::Folded, 2), 29u); /// 22 + "folded" +} + +/// The ADDITIONS a hold contributes, isolated from the row it rides on. The with/without trick used +/// for cleanup evidence is unavailable here: the grammar requires a hold on exactly the clamped rows, +/// so a clamped row WITHOUT one cannot be encoded at all. Instead both sides are cut down to the hold +/// segment itself -- from its first key to the closing brace -- so the tag, the `class` word and the +/// fold pair are outside the comparison by construction rather than by cancellation. +TEST(CASWireCutDeltas, HoldBearingRefLifeAdditions) +{ + CasFoldSeal seal = oneFoldSeal(); + seal.ref_lives[UInt128(1)].coverage = RefCoverage{.classification = CoverageClass::Clamped, .last_folded_ref_id = RefTxnId{7, 11}, .hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{12, 13}, .retry_count = 14, .next_retry_round = 15}}; + + /// This literal is the pre-cut baseline this delta is measured against. + const String old_row = "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":4,\"lfe\":\"7\",\"lfs\":\"11\",\"hr\":\"gap_below_witness\",\"hpe\":\"12\",\"hps\":\"13\",\"hrc\":14,\"hnr\":\"15\"}\n"; + const String new_row = lineAt(encodeFoldSeal(seal), 2); + + const auto hold_segment = [](const String & row, std::string_view first_hold_key) + { + const size_t from = row.find(first_hold_key); + const size_t to = row.rfind('}'); + EXPECT_NE(from, String::npos) << "row does not carry " << first_hold_key << ": " << row; + EXPECT_NE(to, String::npos); + return to > from ? to - from : 0; + }; + + EXPECT_EQ(hold_segment(new_row, ",\"hold_reason\"") - hold_segment(old_row, ",\"hr\""), 33u); +} + +/// The cleanup-evidence pair isolated the same way: with and without, on both sides, so only the +/// two added keys remain in the difference. +TEST(CASWireCutDeltas, CleanupEvidenceRefLifeAdditions) +{ + CasFoldSeal without_evidence = oneFoldSeal(); + without_evidence.ref_lives[UInt128(1)] = RefLifeFoldState{.coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{7, 11}}}; + CasFoldSeal with_evidence = oneFoldSeal(); + with_evidence.ref_lives[UInt128(1)] = RefLifeFoldState{.coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{7, 11}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{12, 13}}}; + + /// These literals are the pre-cut baselines these deltas are measured against. + const String old_without = "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":2,\"lfe\":\"7\",\"lfs\":\"11\"}\n"; + const String old_with = "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":2,\"lfe\":\"7\",\"lfs\":\"11\",\"rte\":\"12\",\"rts\":\"13\"}\n"; + + const size_t base_delta = lineAt(encodeFoldSeal(without_evidence), 2).size() - old_without.size(); + const size_t whole_delta = lineAt(encodeFoldSeal(with_evidence), 2).size() - old_with.size(); + EXPECT_EQ(whole_delta - base_delta, 16u); +} + +TEST(CASWireCutDeltas, BlobRunFoldSealRow) +{ + CasFoldSeal seal = oneFoldSeal(); + seal.blob_target_runs.push_back(RunRef{.key = "r0", .checksum = UInt128(15), .shard = 0, .key_generation = 5}); + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"btr\",\"key\":\"r0\",\"ck\":\"0000000000000000000000000000000f\",\"shard\":0,\"gen\":\"5\"}\n"; + expectDelta(old_bytes, lineAt(encodeFoldSeal(seal), 2), 25); +} + +TEST(CASWireCutDeltas, CondemnedFoldSealSummaryRow) +{ + CasFoldSeal seal = oneFoldSeal(); + seal.condemned_summary[0] = CondemnedSummary{.condemned_total = 3, .pending_total = 1, .oldest_nonpending_condemn_round = 4}; + /// This literal is the pre-cut baseline this delta is measured against. + const String old_bytes = "{\"k\":\"cnd\",\"shard\":0,\"ct\":3,\"pt\":1,\"ocr\":\"4\"}\n"; + expectDelta(old_bytes, lineAt(encodeFoldSeal(seal), 2), 30); +} diff --git a/src/Disks/tests/gtest_cas_enum_wire_table.cpp b/src/Disks/tests/gtest_cas_enum_wire_table.cpp new file mode 100644 index 000000000000..7940c57944eb --- /dev/null +++ b/src/Disks/tests/gtest_cas_enum_wire_table.cpp @@ -0,0 +1,131 @@ +#include +#include +#include +#include +#include + +using namespace DB::Cas; + +namespace +{ + +enum class Fruit : uint8_t +{ + Apple = 0, + Pear = 1, + Plum = 2, +}; + +constexpr EnumWireTable fruits{{{ + {Fruit::Apple, "apple"}, + {Fruit::Pear, "pear"}, + {Fruit::Plum, "plum"}, +}}}; + +static_assert(fruits.denseAndOrdered()); +static_assert(fruits.wordsUnique()); +static_assert(casEnumTableCoversEnum()); + +/// A one-based dense enum exercises the index arithmetic from the first entry's value. +enum class Grade : uint8_t +{ + Low = 1, + Mid = 2, + High = 3, +}; + +constexpr EnumWireTable grades{{{ + {Grade::Low, "low"}, + {Grade::Mid, "mid"}, + {Grade::High, "high"}, +}}}; + +static_assert(grades.denseAndOrdered()); +static_assert(casEnumTableCoversEnum()); + +} + +TEST(CASEnumWireTable, RoundTripsEveryEntryBothWays) +{ + for (const auto & e : fruits.entries) + { + EXPECT_EQ(fruits.toWord(e.value, "fruits"), e.word); + EXPECT_EQ(fruits.fromWord(e.word, "fruits"), e.value); + } + for (const auto & e : grades.entries) + EXPECT_EQ(grades.fromWord(grades.toWord(e.value, "grades"), "grades"), e.value); +} + +TEST(CASEnumWireTable, FromWordFailsClosed) +{ + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { fruits.fromWord("banana", "fruits"); }); +} + +/// `LOGICAL_ERROR` aborts the process in debug/sanitizer builds (`handle_error_code`), so the +/// defensive toWord branch needs the death-test split this test directory already uses (see +/// `gtest_cas_gc_state_format.cpp`'s `RejectsZeroGcShardsOnEncode` pair) — a bare EXPECT_THROW +/// would SIGABRT the whole gate binary on those lanes. +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASEnumWireTableDeathTest, ToWordAbortsOnOutOfRangeValue) +{ + EXPECT_DEATH(fruits.toWord(static_cast(99), "fruits"), "outside the wire vocabulary"); +} +#else +TEST(CASEnumWireTable, ToWordThrowsLogicalErrorOnOutOfRangeValue) +{ + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { fruits.toWord(static_cast(99), "fruits"); }); +} +#endif + +/// The compile-time proofs must also be exercised in the direction where they can fail — a +/// predicate rewritten to `return true;` must break this file. All three are constexpr, so the +/// negative cases are plain static_asserts over deliberately bad tables: +namespace bad_tables +{ + +enum class Sparse : uint8_t { A = 0, B = 2 }; +constexpr EnumWireTable sparse{{{{Sparse::A, "a"}, {Sparse::B, "b"}}}}; +static_assert(!sparse.denseAndOrdered()); +/// ...and through the folded coverage proof, so deleting its density disjunct breaks the file +/// (this table is set-equal and word-unique — only density rejects it): +static_assert(!casEnumTableCoversEnum()); + +constexpr EnumWireTable dup_words{{{ + {Fruit::Apple, "apple"}, {Fruit::Pear, "apple"}, {Fruit::Plum, "plum"}}}}; +static_assert(!dup_words.wordsUnique()); +/// ...and through the folded proof (dense, right-sized, set-equal — only word uniqueness rejects +/// it), so deleting the uniqueness disjunct breaks the file: +static_assert(!casEnumTableCoversEnum()); + +constexpr EnumWireTable dup_value{{{ + {Fruit::Apple, "apple"}, {Fruit::Apple, "pear"}, {Fruit::Plum, "plum"}}}}; +static_assert(!casEnumTableCoversEnum()); + +constexpr EnumWireTable invalid_value{{{ + {Fruit::Apple, "apple"}, {Fruit::Pear, "pear"}, {static_cast(99), "plum"}}}}; +static_assert(!casEnumTableCoversEnum()); + +/// `dup_value` and `invalid_value` fail the folded density check before the set-equality core runs, so the +/// core needs its own failing witnesses — both dense and word-unique, so they reach it. +/// Reaches the size comparison: one enumerator short. +constexpr EnumWireTable missing_enumerator{{{ + {Fruit::Apple, "apple"}, {Fruit::Pear, "pear"}}}}; +static_assert(!casEnumTableCoversEnum()); + +/// Reaches the declared-values scan: right size, dense from Pear, an out-of-enum value present +/// and `Apple` missing — the asserts header's own motivating scenario. +constexpr EnumWireTable enumerator_missing{{{ + {Fruit::Pear, "pear"}, {Fruit::Plum, "plum"}, {static_cast(3), "quince"}}}}; +static_assert(!casEnumTableCoversEnum()); + +/// Guards the size comparison itself: every declared value present PLUS one out-of-enum entry — +/// the only miscoverage the declared-values scan cannot see (an enumerator deleted from the enum +/// while its table row survived). +constexpr EnumWireTable extra_entry{{{ + {Fruit::Apple, "apple"}, {Fruit::Pear, "pear"}, {Fruit::Plum, "plum"}, + {static_cast(3), "quince"}}}}; +static_assert(!casEnumTableCoversEnum()); + +} diff --git a/src/Disks/tests/gtest_cas_event_dispatcher.cpp b/src/Disks/tests/gtest_cas_event_dispatcher.cpp index 325c70de59f4..424194bfa22d 100644 --- a/src/Disks/tests/gtest_cas_event_dispatcher.cpp +++ b/src/Disks/tests/gtest_cas_event_dispatcher.cpp @@ -125,34 +125,40 @@ TEST(CASEventDispatcher, ReentrantSinkDoesNotDeadlock) TEST(CASEventDispatcher, LedgerEmissionOutsideLocks) { auto b = std::make_shared(); - std::vector seen; /// declared before the Pool so it outlives any late background emit - std::mutex seen_mutex; + /// Heap-owned, not plain locals: `seen`'s own declaration-before-the-Pool comment protects only + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); const RootNamespace ns{"srv1/tbl"}; const String ref = "all_0_0_0"; publishOneBlobPart(s, ns.string(), ref, "the-resolvable-payload"); - std::atomic reentered{false}; - s->setEventSink([&](CasEvent e) + auto reentered = std::make_shared>(false); + /// `s` is captured as a raw pointer (`s.get()`), not by reference and not by `shared_ptr`: a + /// `shared_ptr` capture here would make the Pool's own `event_sink` hold a permanent reference to + /// its owning Pool, a cycle that leaks it; a by-reference capture of the local `s` would dangle once + /// this frame returns. Validity is the same invariant every self-referencing hook in the production + /// code relies on (e.g. `CasPool.cpp`'s `[s = store.get()]`): the hook can only run while some other + /// `shared_ptr` keeps the Pool alive. + Pool * const s_ptr = s.get(); + s->setEventSink([seen, reentered, s_ptr, ns, ref](CasEvent e) { - { - std::lock_guard g(seen_mutex); - seen.push_back(e); - } + seen->push(e); /// Re-enter a ledger read that takes `state_mutex`, exactly once (`Deferred` => this read /// itself emits nothing, so there is no unbounded emit recursion). Under the pre-fix code the /// outer `resolveRef` still holds `state_mutex` here, so this call self-deadlocks. - if (e.type == CasEventType::RefResolve && !reentered.exchange(true)) - (void)s->resolveRef(ns, ref, false, ResolveAudit::Deferred); + if (e.type == CasEventType::RefResolve && !reentered->exchange(true)) + (void)s_ptr->resolveRef(ns, ref, false, ResolveAudit::Deferred); }); - std::promise resolve_done; - auto resolve_future = resolve_done.get_future(); + auto resolve_done = std::make_shared>(); + auto resolve_future = resolve_done->get_future(); std::thread resolver([&] { (void)s->resolveRef(ns, ref); /// ResolveAudit::Emit (default) -> emits RefResolve -> drives the sink - resolve_done.set_value(); + resolve_done->set_value(); }); /// A second thread emits upload-task-style events concurrently with the resolve, so the dispatcher's @@ -178,11 +184,10 @@ TEST(CASEventDispatcher, LedgerEmissionOutsideLocks) resolver.join(); uploader.join(); - EXPECT_TRUE(reentered.load()) << "the reentrant ledger read must have run"; - std::lock_guard g(seen_mutex); + EXPECT_TRUE(reentered->load()) << "the reentrant ledger read must have run"; size_t resolves = 0; size_t uploads = 0; - for (const auto & e : seen) + for (const auto & e : seen->snapshot()) { if (e.type == CasEventType::RefResolve) ++resolves; diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index ae65ed27274b..e55f775ef4f8 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -38,17 +38,19 @@ namespace class RenewalEventBackend final : public InMemoryBackend { public: - using InMemoryBackend::get; - using InMemoryBackend::putOverwrite; - - bool throw_before_next_overwrite = false; - bool throw_nonretryable_next_overwrite = false; - bool vanish_on_next_overwrite = false; + bool throw_before_next_write = false; + bool throw_nonretryable_next_write = false; + bool vanish_on_next_write = false; + /// Runs just before an armed fault throws. The engine draws its inter-attempt backoff randomly and + /// admits the reissue against that drawn duration, so a test that needs the ambiguity refused + /// rather than reissued has to move the injected clock here -- from inside the attempt, the only + /// point between admission and the resolve read a test can reach. + std::function before_throw; void armResolveProbe() { std::lock_guard lock(resolve_mutex); - observe_next_get = true; + observe_next_read = true; resolve_started = false; } @@ -58,40 +60,64 @@ class RenewalEventBackend final : public InMemoryBackend return resolve_started; } - std::optional get(const String & key, Range range) override + /// A plain read/replace through the primitive surface, for fixtures that need to observe or seed + /// state without going through the pool under test. + std::optional readForTest(const String & key) + { + DB::Cas::tests::OperationForTest op(*this); + return (*op).read(key, Retry::standard()); + } + + bool replaceForTest(const String & key, const String & bytes, const Etag & expected) + { + DB::Cas::tests::OperationForTest op(*this); + return std::holds_alternative((*op).replace(key, bytes, expected, Retry::standard())); + } + + /// The engine settles an ambiguous write by reading the key back, so the observation belongs on the + /// READ PRIMITIVE -- the resolve read never reaches the legacy `get`. + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { { std::lock_guard lock(resolve_mutex); - if (observe_next_get) + if (observe_next_read) { resolve_started = true; - observe_next_get = false; + observe_next_read = false; } } - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - PutResult putOverwrite( + /// The faults sit on the WRITE PRIMITIVE, and only on a CONDITIONAL write: a lease renewal is a + /// replace, so the pool's own create-if-absent writes must not consume a one-shot fault. + std::expected write( const String & key, const String & bytes, - const Token & expected, - const ObjectMeta & meta) override + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (std::exchange(vanish_on_next_overwrite, false)) + if (!expected_value) + return InMemoryBackend::write(key, bytes, expected_value, access); + if (std::exchange(vanish_on_next_write, false)) { - (void)InMemoryBackend::deleteExact(key, expected); - return {PutOutcome::PreconditionFailed, {}}; + (void)InMemoryBackend::remove(key, *expected_value, access); + return std::unexpected(RawConflict{}); } - if (std::exchange(throw_nonretryable_next_overwrite, false)) + if (std::exchange(throw_nonretryable_next_write, false)) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected deterministic renewal rejection"); - if (std::exchange(throw_before_next_overwrite, false)) + if (std::exchange(throw_before_next_write, false)) + { + if (before_throw) + before_throw(); throw Poco::TimeoutException("injected renewal timeout before commit"); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + } + return InMemoryBackend::write(key, bytes, expected_value, access); } private: std::mutex resolve_mutex; - bool observe_next_get = false; + bool observe_next_read = false; bool resolve_started = false; }; @@ -99,27 +125,33 @@ CasRequestBudget renewalEventBudget() { return CasRequestBudget{ .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = 2, .lease_safety_margin_ms = 20, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, + .connect_timeout_cap_ms = std::nullopt, }; } +/// `boot_ms` is a shared, heap-owned atomic, not a plain reference parameter: some callers mutate it +/// after the Pool exists, and the Pool can outlive this function's own call (a background publish +/// holds `shared_from_this()`), so a by-reference capture of a caller-local would dangle. PoolPtr openRenewalEventPool( const std::shared_ptr & backend, - uint64_t & boot_ms, + const std::shared_ptr> & boot_ms, CasRequestBudget budget = renewalEventBudget(), String prefix = "renewal-events", String server_root_id = "test") { + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the renewal-boundary math these tests drive matches what admits. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); return Pool::open(backend, PoolConfig{ .pool_prefix = std::move(prefix), .server_root_id = std::move(server_root_id), .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .cas_request_budget = budget, - .boot_ms_fn = [&] { return boot_ms; }, + .boot_ms_fn = [boot_ms] + { + return boot_ms->load(); + }, }); } @@ -168,114 +200,142 @@ TEST(CASEvent, ConstructAndCopyAndName) TEST(CASEvent, PoolEmitsToSink) { auto b = std::make_shared(); - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); - s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + s->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); CasEvent e; e.type = CasEventType::BlobPut; e.object_hash = "h"; s->emitEvent(std::move(e)); - ASSERT_EQ(seen.size(), 1u); - EXPECT_EQ(seen[0].type, CasEventType::BlobPut); + ASSERT_EQ(seen->snapshot().size(), 1u); + EXPECT_EQ(seen->snapshot()[0].type, CasEventType::BlobPut); /// null sink => no-op (no crash, no row); a fresh event, not the one already moved above. s->setEventSink(nullptr); CasEvent e2; e2.type = CasEventType::BlobPut; s->emitEvent(std::move(e2)); - EXPECT_EQ(seen.size(), 1u); + EXPECT_EQ(seen->snapshot().size(), 1u); } TEST(CASEvent, FirstAttemptRenewalIsSilent) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = openRenewalEventPool(backend, boot_ms); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); EXPECT_NO_THROW(store->renewWatermarkOnce()); - EXPECT_TRUE(watermarkRenewEvents(events).empty()); + EXPECT_TRUE(watermarkRenewEvents(events->snapshot()).empty()); } TEST(CASEvent, WatermarkRenewEventsAreBoundedAndComplete) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = openRenewalEventPool(backend, boot_ms); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); - backend->throw_before_next_overwrite = true; + backend->throw_before_next_write = true; EXPECT_NO_THROW(store->renewWatermarkOnce()); - const std::vector renewals = watermarkRenewEvents(events); - ASSERT_EQ(renewals.size(), 2u); - EXPECT_EQ(renewals[0].outcome, "retrying"); - EXPECT_EQ(renewals[1].outcome, "recovered"); - EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "1"); - EXPECT_EQ(renewals[1].detail.at("attempts_sent"), "2"); + const std::vector renewals = watermarkRenewEvents(events->snapshot()); + /// ONE event per logical renewal, whatever the physical attempts cost: the engine owns its own + /// reissues, and the terminal event carries their count rather than announcing each one. + ASSERT_EQ(renewals.size(), 1u); + EXPECT_EQ(renewals[0].outcome, "recovered"); + EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "2"); + EXPECT_EQ(renewals[0].detail.at("classification"), "committed_after_retry"); EXPECT_EQ(renewals[0].detail.at("server_root_id"), "test"); EXPECT_EQ(renewals[0].detail.at("writer_epoch"), std::to_string(store->writerEpoch())); EXPECT_EQ(renewals[0].detail.at("seq"), "2"); - EXPECT_EQ(renewals[0].detail.at("write_attempt_id"), renewals[1].detail.at("write_attempt_id")); EXPECT_FALSE(renewals[0].detail.at("write_attempt_id").empty()); EXPECT_LT(renewals[0].detail.at("write_attempt_id").size(), 32u); - - for (const CasEvent & event : renewals) - { - for (const String & key : { - "server_root_id", - "writer_epoch", - "seq", - "write_attempt_id", - "attempts_sent", - "elapsed_ms", - "remaining_confirmed_budget_ms", - "unresolved_reason", - "deadline_source", - "stop_cause", - "classification"}) - EXPECT_TRUE(event.detail.contains(key)) << "missing detail key " << key; - } + /// Both attempts sent the same body, so the event names the id the lease actually landed with -- + /// a reissue that minted a fresh id would leave the two disagreeing. + const MountLease landed = decodeMountLease(backend->readForTest(store->layout().mountKey("test"))->bytes); + EXPECT_EQ(renewals[0].detail.at("write_attempt_id"), u128ToHex(landed.write_attempt_id).substr(0, 12)); + + for (const String & key : { + "server_root_id", + "writer_epoch", + "seq", + "write_attempt_id", + "attempts_sent", + "elapsed_ms", + "remaining_confirmed_budget_ms", + "classification"}) + EXPECT_TRUE(renewals[0].detail.contains(key)) << "missing detail key " << key; } -TEST(CASEvent, FirstAmbiguityIsVisibleWhileResolveIsInFlight) +/// An attempt that spends the lease it was admitted under must not START the resolving read. That read +/// is the only thing that can prove the ambiguous attempt landed, and issuing it past the lease-safe +/// bound would be a request made without the authority it was admitted under -- so the renewal reports +/// the deadline that refused it instead of resolving anything. +TEST(CASEvent, AnAmbiguityPastTheLeaseBoundNeverStartsTheResolvingRead) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::atomic retrying_events{0}; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = openRenewalEventPool( backend, boot_ms, renewalEventBudget(), "renewal-inflight-ambiguity"); - store->setEventSink([&](CasEvent event) + store->setEventSink([events](CasEvent event) { - if (event.type == CasEventType::WatermarkRenew && event.outcome == "retrying") - { - retrying_events.fetch_add(1); - /// The diagnostic callback may consume the remaining recovery budget. The controller - /// must re-check its absolute deadline before starting the resolving GET. - boot_ms = 1081; - } + events->push(std::move(event)); }); - backend->throw_before_next_overwrite = true; + /// The lease was anchored at 100 with a 1000 ms TTL, so the fence expires at 1100 and holds a 20 ms + /// safety margin. At 1081 only 19 ms remain, and admission refuses the resolve read. + backend->before_throw = [boot_ms] + { + boot_ms->store(1'081); + }; + backend->throw_before_next_write = true; backend->armResolveProbe(); EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - EXPECT_EQ(retrying_events.load(), 1u) - << "first ambiguity must be externally visible before the pre-resolve deadline gate"; EXPECT_FALSE(backend->resolveStarted()) - << "a diagnostic sink that exhausts the budget must prevent the resolving GET from starting"; - EXPECT_EQ(retrying_events.load(), 1u) << "retrying delivery is bounded to the first ambiguity"; + << "an attempt that consumed the lease must not start the resolving read"; + const std::vector renewals = watermarkRenewEvents(events->snapshot()); + ASSERT_EQ(renewals.size(), 1u); + EXPECT_EQ(renewals[0].outcome, "failed"); + EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "1"); + EXPECT_EQ(renewals[0].detail.at("classification"), "external_lease_deadline"); } +/// Ten renewals nested through each other's conflict sinks, against an eight-slot observation stack. +/// The two calls beyond the stack get no rich event -- and must still report their own physical attempt +/// count, which rides the write result rather than the suppressed observation. TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) { constexpr size_t depth = 10; + constexpr size_t observation_stack_capacity = 8; std::array, depth> backends; std::array, depth> layouts; - std::array, depth> keepers; + std::array, depth> planes; + std::array, depth> renewers; std::array server_root_ids; std::array sinks; + std::array renew_events{}; uint64_t wall_ms = 100; uint64_t boot_ms = 100; std::optional deepest_result; @@ -284,16 +344,7 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) renew_at = [&](size_t index) { configureMountRenewObservability(&server_root_ids[index], &sinks[index], /*deferred=*/false); - MountRenewResult result = keepers[index]->renew( - CasRequestBudget{ - .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = 1, - .lease_safety_margin_ms = 0, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, - }, - MountRenewOperationEnvironment{}); + MountRenewResult result = renewers[index]->renew(MountRenewOperationEnvironment{}); reportMountRenewCompletion(result); return result; }; @@ -305,6 +356,8 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) server_root_ids[index] = fmt::format("deep-{}", index); sinks[index] = [&, index](CasEvent event) { + if (event.type == CasEventType::WatermarkRenew) + ++renew_events[index]; if (event.type == CasEventType::MountConflict && index + 1 < depth) { MountRenewResult child_result = renew_at(index + 1); @@ -312,8 +365,14 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) deepest_result = std::move(child_result); } }; - keepers[index] = std::make_unique( - backends[index], + /// One open-fence plane per renewer, on the same injected clock the renewer anchors its lease + /// against, and with a sleep that advances it: the deepest renewal reissues, and no unit test + /// may serve the engine's jittered backoff for real. + planes[index] = std::make_unique( + backends[index], Fence::open(), [&] { return boot_ms; }, [&](uint64_t ms) { boot_ms += ms; }); + renewers[index] = std::make_unique( + *planes[index], + *planes[index], *layouts[index], server_root_ids[index], UInt128(index + 1), @@ -324,176 +383,193 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) sinks[index], std::chrono::milliseconds(0), [&] { return boot_ms; }); - keepers[index]->start(); + renewers[index]->start(); if (index + 1 < depth) { const String key = layouts[index]->mountKey(server_root_ids[index]); - auto observed = backends[index]->get(key); + auto observed = backends[index]->readForTest(key); ASSERT_TRUE(observed.has_value()); MountLease foreign = decodeMountLease(observed->bytes); foreign.server_uuid = UInt128(100 + index); - ASSERT_EQ( - backends[index]->putOverwrite(key, encodeMountLease(foreign), observed->token).outcome, - PutOutcome::Done); + ASSERT_TRUE(backends[index]->replaceForTest(key, encodeMountLease(foreign), observed->etag)); } } - backends.back()->throw_nonretryable_next_overwrite = true; + /// The deepest slot is the only one nobody took, so its renewal can recover: the attempt is lost + /// before its answer, the resolve read finds the precondition intact, and the reissue commits. + backends.back()->throw_before_next_write = true; const MountRenewResult outer_result = renew_at(0); EXPECT_EQ(outer_result.outcome, MountRenewOutcome::Terminal); ASSERT_TRUE(deepest_result.has_value()); - EXPECT_EQ(deepest_result->outcome, MountRenewOutcome::Terminal); - EXPECT_EQ(deepest_result->diagnostics.attempts_sent, 1u) + EXPECT_EQ(deepest_result->outcome, MountRenewOutcome::Committed); + EXPECT_EQ(deepest_result->attempts_sent, 2u) << "nesting beyond the rich-event stack must not erase physical attempt truth"; + + for (size_t index = 0; index < depth; ++index) + EXPECT_EQ(renew_events[index], index < observation_stack_capacity ? 1u : 0u) + << "renewal " << index << " is " << (index < observation_stack_capacity ? "on" : "beyond") + << " the observation stack"; } TEST(CASEvent, WatermarkRenewSinkFailureCannotChangeOutcome) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + auto boot_ms = std::make_shared>(100); auto store = openRenewalEventPool(backend, boot_ms); const String mount_key = store->layout().mountKey("test"); - const uint64_t seq_before = decodeMountLease(backend->get(mount_key)->bytes).seq; + const uint64_t seq_before = decodeMountLease(backend->readForTest(mount_key)->bytes).seq; store->setEventSink([](const CasEvent & event) { if (event.type == CasEventType::WatermarkRenew) throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected renewal event sink failure"); }); - backend->throw_before_next_overwrite = true; + backend->throw_before_next_write = true; EXPECT_NO_THROW(store->renewWatermarkOnce()); - EXPECT_EQ(decodeMountLease(backend->get(mount_key)->bytes).seq, seq_before + 1); + EXPECT_EQ(decodeMountLease(backend->readForTest(mount_key)->bytes).seq, seq_before + 1); EXPECT_TRUE(store->mayMutate()); } +/// The two terminal endings a renewal reaches without ever settling its write: the store refusing it +/// outright, and the lease refusing to admit it. There is no attempt-count ending -- the engine bounds a +/// write by time, never by a number of tries -- and the deadline ending that DOES send an attempt first +/// is `AnAmbiguityPastTheLeaseBoundNeverStartsTheResolvingRead`. TEST(CASEvent, TerminalRenewalDetailsPreservePhysicalTruthAndClassification) { - const auto one_failed_event = [](const std::vector & events) -> CasEvent + const auto one_failed_event = [](const std::vector & events) -> std::optional { const std::vector renewals = watermarkRenewEvents(events); const auto failed = std::find_if(renewals.begin(), renewals.end(), [](const CasEvent & event) { return event.outcome == "failed"; }); - EXPECT_NE(failed, renewals.end()); - return failed == renewals.end() ? CasEvent{} : *failed; + if (failed == renewals.end()) + return std::nullopt; + return *failed; }; { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish + /// holds `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-deterministic-details"); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); - backend->throw_nonretryable_next_overwrite = true; - - EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - const CasEvent failed = one_failed_event(events); - EXPECT_EQ(failed.detail.at("attempts_sent"), "1"); - EXPECT_EQ(failed.detail.at("unresolved_reason"), "not_unresolved"); - EXPECT_EQ(failed.detail.at("stop_cause"), "continue"); - EXPECT_EQ(failed.detail.at("classification"), "deterministic_failure"); - } - - { - auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; - CasRequestBudget budget = renewalEventBudget(); - budget.max_attempts = 1; - auto store = openRenewalEventPool(backend, boot_ms, budget, "renewal-exhausted-details"); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); - backend->throw_before_next_overwrite = true; + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); + backend->throw_nonretryable_next_write = true; EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - const CasEvent failed = one_failed_event(events); - EXPECT_EQ(failed.detail.at("attempts_sent"), "1"); - EXPECT_EQ(failed.detail.at("unresolved_reason"), "attempts_exhausted"); - EXPECT_EQ(failed.detail.at("classification"), "attempts_exhausted"); + const std::optional failed = one_failed_event(events->snapshot()); + ASSERT_TRUE(failed.has_value()) << "the store's refusal must reach the event log"; + /// A deterministic failure reaches the renewer as the exception the engine refuses to reissue, + /// and an exception carries no attempt count -- so the classification is all this ending states. + EXPECT_EQ(failed->detail.at("classification"), "deterministic_failure"); } { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish + /// holds `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-deadline-details"); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); - boot_ms = 1071; + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); + /// The lease was anchored at 100 with a 1000 ms TTL and holds a 20 ms safety margin, so 1090 + /// leaves 10 ms of it and admission refuses the renewal before its first attempt. + boot_ms->store(1090); EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - const CasEvent failed = one_failed_event(events); - EXPECT_EQ(failed.detail.at("attempts_sent"), "0"); - EXPECT_EQ(failed.detail.at("unresolved_reason"), "no_attempt_sent"); - EXPECT_EQ(failed.detail.at("deadline_source"), "external_lease_safety"); - EXPECT_EQ(failed.detail.at("classification"), "external_lease_deadline"); + const std::optional failed = one_failed_event(events->snapshot()); + ASSERT_TRUE(failed.has_value()) << "the refused admission must reach the event log"; + EXPECT_EQ(failed->detail.at("attempts_sent"), "0"); + EXPECT_EQ(failed->detail.at("classification"), "external_lease_deadline"); } } TEST(CASEvent, ReentrantRenewalSinkPreservesOuterObservationIdentity) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; - std::vector events; + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); PoolPtr store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-reentrant-sink"); - bool reentered = false; - store->setEventSink([&](CasEvent event) + auto reentered = std::make_shared>(false); + /// `store` is captured as a raw pointer (`store.get()`), not by reference and not by `shared_ptr`: a + /// `shared_ptr` capture here would make the Pool's own `event_sink` hold a permanent reference to + /// its owning Pool, a cycle that leaks it; a by-reference capture of the local `store` would dangle + /// once this frame returns. Validity is the same invariant every self-referencing hook in the + /// production code relies on (e.g. `CasPool.cpp`'s `[s = store.get()]`): the hook can only run while + /// some other `shared_ptr` keeps the Pool alive. + Pool * const store_ptr = store.get(); + store->setEventSink([events, reentered, store_ptr](CasEvent event) { if (event.type != CasEventType::WatermarkRenew) return; - events.push_back(event); - if (event.outcome == "recovered" && !std::exchange(reentered, true)) - store->renewWatermarkOnce(); + events->push(event); + if (event.outcome == "recovered" && !reentered->exchange(true)) + store_ptr->renewWatermarkOnce(); }); - backend->throw_before_next_overwrite = true; + backend->throw_before_next_write = true; EXPECT_NO_THROW(store->renewWatermarkOnce()); - ASSERT_TRUE(reentered); - ASSERT_EQ(events.size(), 2u); - EXPECT_EQ(events[0].outcome, "retrying"); - EXPECT_EQ(events[1].outcome, "recovered"); - EXPECT_EQ(events[0].detail.at("seq"), "2"); - EXPECT_EQ(events[1].detail.at("seq"), events[0].detail.at("seq")); - EXPECT_EQ(events[1].detail.at("write_attempt_id"), events[0].detail.at("write_attempt_id")); - EXPECT_EQ(decodeMountLease(backend->get(store->layout().mountKey("test"))->bytes).seq, 3u) + ASSERT_TRUE(reentered->load()); + /// The nested renewal commits on its first attempt, which is silent, so the outer recovery is the + /// only event -- and it still names the outer renewal's own seq while the durable lease has already + /// moved past it. An observation the nested call reused would report seq 3 here. + const std::vector observed_events = events->snapshot(); + ASSERT_EQ(observed_events.size(), 1u); + EXPECT_EQ(observed_events[0].outcome, "recovered"); + EXPECT_EQ(observed_events[0].detail.at("attempts_sent"), "2"); + EXPECT_EQ(observed_events[0].detail.at("seq"), "2"); + EXPECT_EQ(decodeMountLease(backend->readForTest(store->layout().mountKey("test"))->bytes).seq, 3u) << "the nested first-attempt success must run without replacing the outer observation"; } TEST(CASEvent, PreCompletionConflictReentrancyPreservesOuterTerminalObservation) { auto inner_backend = std::make_shared(); - uint64_t inner_boot_ms = 100; + auto inner_boot_ms = std::make_shared>(100); auto inner = openRenewalEventPool( inner_backend, inner_boot_ms, renewalEventBudget(), "renewal-reentrant-inner", "inner"); auto outer_backend = std::make_shared(); - uint64_t outer_boot_ms = 100; - CasRequestBudget outer_budget = renewalEventBudget(); - outer_budget.max_attempts = 1; + auto outer_boot_ms = std::make_shared>(100); auto outer = openRenewalEventPool( - outer_backend, outer_boot_ms, outer_budget, "renewal-reentrant-outer", "outer"); - std::vector outer_events; - bool reentered = false; - outer->setEventSink([&](CasEvent event) + outer_backend, outer_boot_ms, renewalEventBudget(), "renewal-reentrant-outer", "outer"); + /// Heap-owned, not plain locals: `outer`'s `event_sink` mutates them, and the Pool can outlive this + /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture of a + /// local would dangle. `inner` (a DIFFERENT Pool from `outer`) is captured by value -- a `shared_ptr` + /// copy here is not a self-reference cycle, unlike capturing `outer` into its own sink would be. + auto outer_events = std::make_shared(); + auto reentered = std::make_shared>(false); + outer->setEventSink([outer_events, reentered, inner](CasEvent event) { - outer_events.push_back(event); - if (event.type == CasEventType::MountConflict && !std::exchange(reentered, true)) + outer_events->push(event); + if (event.type == CasEventType::MountConflict && !reentered->exchange(true)) inner->renewWatermarkOnce(); }); - outer_backend->vanish_on_next_overwrite = true; + /// The inner renewal loses its first attempt's answer and recovers on a reissue, so it has its own + /// identity and its own classification to report. Both must stay off the outer observation. + inner_backend->throw_before_next_write = true; + outer_backend->vanish_on_next_write = true; EXPECT_THROW(outer->renewWatermarkOnce(), DB::Exception); - ASSERT_TRUE(reentered); - const std::vector renewals = watermarkRenewEvents(outer_events); - ASSERT_EQ(renewals.size(), 2u); - EXPECT_EQ(renewals[0].outcome, "retrying"); - EXPECT_EQ(renewals[1].outcome, "failed"); + ASSERT_TRUE(reentered->load()); + const std::vector renewals = watermarkRenewEvents(outer_events->snapshot()); + ASSERT_EQ(renewals.size(), 1u); + EXPECT_EQ(renewals[0].outcome, "failed"); EXPECT_EQ(renewals[0].detail.at("server_root_id"), "outer"); - EXPECT_EQ(renewals[1].detail.at("server_root_id"), "outer"); - EXPECT_EQ(renewals[1].detail.at("write_attempt_id"), renewals[0].detail.at("write_attempt_id")); - EXPECT_EQ(renewals[1].detail.at("classification"), "vanished"); + EXPECT_EQ(renewals[0].detail.at("classification"), "vanished"); } /// Round-B opt §6: `emitEvent` takes the event BY VALUE (moved-through, not `const &`), so a @@ -503,21 +579,33 @@ TEST(CASEvent, PreCompletionConflictReentrancyPreservesOuterTerminalObservation) TEST(CASEvent, EmitEventMovesSourceIntoSink) { auto b = std::make_shared(); - String captured_reason; - std::map captured_detail; + /// Heap-owned, mutex-guarded, not plain locals: the Pool can outlive this stack frame (a background + /// publish holds `shared_from_this()`), so a by-reference capture of a local would dangle, and a + /// background emit could race the foreground read below. + struct Captured + { + std::mutex mutex; + String reason; + std::map detail; + }; + auto captured = std::make_shared(); auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); - s->setEventSink([&](CasEvent ev) + s->setEventSink([captured](CasEvent ev) { - captured_reason = std::move(ev.reason); - captured_detail = std::move(ev.detail); + std::lock_guard lock(captured->mutex); + captured->reason = std::move(ev.reason); + captured->detail = std::move(ev.detail); }); CasEvent e; e.type = CasEventType::BlobPut; e.reason = "sentinel-reason"; e.detail["k"] = "v"; s->emitEvent(std::move(e)); - EXPECT_EQ(captured_reason, "sentinel-reason"); - EXPECT_EQ(captured_detail.at("k"), "v"); + { + std::lock_guard lock(captured->mutex); + EXPECT_EQ(captured->reason, "sentinel-reason"); + EXPECT_EQ(captured->detail.at("k"), "v"); + } /// the source event must be MOVED-FROM after emit, not merely aliased/copied through -- reading /// `e` here is the whole point of the test, not an oversight. EXPECT_TRUE(e.reason.empty()); // NOLINT(bugprone-use-after-move, hicpp-invalid-access-moved) @@ -554,9 +642,9 @@ String publishOneBlobPart(const PoolPtr & s, const String & ns, const String & r /// Whether the CURRENT retired list (any gc-shard) still holds an entry (ack-floor pipeline in flight). bool anyRetiredPending(const PoolPtr & s) { - /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// Condemned state rides the adopted fold seal's RunMarker::Condemned rows, not a /// separate retired list — reconstruct the in-flight set from the seal. - return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); + return DB::Cas::tests::anyCondemnedInSeal(*s->poolBackendPtr(), s->layout()); } /// Drive regular GC to a fixpoint over the ACK-FLOOR round (renew the store's mount ack after each round; @@ -592,18 +680,17 @@ bool hasType(const std::vector & events, CasEventType t) TEST(CASEvent, LifecycleReconstructionFromRows) { auto b = std::make_shared(); - /// Declared BEFORE the Pool so they OUTLIVE it: the Pool's background retired-view syncer can emit - /// (e.g. a view-advance event) right up to the Pool's destructor, and a sink capturing locals that - /// die first is a use-after-scope (found by ASan 2026-07-09; the production sink captures the Context - /// shared_ptr by value and is immune). - std::vector events; - std::mutex events_mutex; + /// Heap-owned, not a plain local: the Pool's background retired-view syncer can emit (e.g. a + /// view-advance event) right up to the Pool's destructor, and a background publish can hold an + /// extra `shared_from_this()` past this frame's return regardless of declaration order relative to + /// the Pool (found by ASan 2026-07-09; the production sink captures the Context shared_ptr by value + /// and is immune) -- a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); - s->setEventSink([&](const CasEvent & e) + s->setEventSink([events](const CasEvent & e) { - std::lock_guard lock(events_mutex); - events.push_back(e); + events->push(e); }); const RootNamespace ns{"srv1/tbl"}; @@ -622,31 +709,35 @@ TEST(CASEvent, LifecycleReconstructionFromRows) runGcToFixpoint(s, gc); /// The blob must actually be gone (the delete fired). - ASSERT_FALSE(b->head(s->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))})).exists) - << "GC must have deleted the now-unreferenced blob"; + { + DB::Cas::tests::OperationForTest blob_op(b); + ASSERT_FALSE((*blob_op).head(s->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}), Retry::standard()).has_value()) + << "GC must have deleted the now-unreferenced blob"; + } /// (a) the expected taxonomy was emitted across the lifecycle (manifest model: no standalone trees). - EXPECT_TRUE(hasType(events, CasEventType::BlobPut)); - EXPECT_TRUE(hasType(events, CasEventType::RootAdd)) + const std::vector observed_events = events->snapshot(); + EXPECT_TRUE(hasType(observed_events, CasEventType::BlobPut)); + EXPECT_TRUE(hasType(observed_events, CasEventType::RootAdd)) << "a fold must have recorded the manifest owner's blob edge (+1)"; - EXPECT_TRUE(hasType(events, CasEventType::RefDrop)); - EXPECT_TRUE(hasType(events, CasEventType::IndegZero)); - EXPECT_TRUE(hasType(events, CasEventType::GcRetireObserve) - || hasType(events, CasEventType::GcRetireDecision) - || hasType(events, CasEventType::GcRecheckVerdict)) + EXPECT_TRUE(hasType(observed_events, CasEventType::RefDrop)); + EXPECT_TRUE(hasType(observed_events, CasEventType::IndegZero)); + EXPECT_TRUE(hasType(observed_events, CasEventType::GcRetireObserve) + || hasType(observed_events, CasEventType::GcRetireDecision) + || hasType(observed_events, CasEventType::GcRecheckVerdict)) << "a GC retire/recheck transition must be recorded"; - EXPECT_TRUE(hasType(events, CasEventType::BlobDelete) || hasType(events, CasEventType::ManifestDelete)) + EXPECT_TRUE(hasType(observed_events, CasEventType::BlobDelete) || hasType(observed_events, CasEventType::ManifestDelete)) << "the single content-delete site must emit a delete row"; /// (b) completeness mandate: every emitted event has a non-empty reason (the human WHY). - for (const auto & e : events) + for (const auto & e : observed_events) EXPECT_FALSE(e.reason.empty()) << "event " << toString(e.type) << " (" << e.object_hash << ") has an empty reason"; /// (c) lifecycle reconstruction: filtering by the deleted blob's object_hash yields, in time /// order, at least its in-degree-zero -> retire-observe -> delete chain — its whole story. std::vector chain; - for (const auto & e : events) + for (const auto & e : observed_events) if (e.object_hash == blob_hash) chain.push_back(e.type); diff --git a/src/Disks/tests/gtest_cas_fence_generation.cpp b/src/Disks/tests/gtest_cas_fence_generation.cpp index be621ddd9dd0..f4f060d576f9 100644 --- a/src/Disks/tests/gtest_cas_fence_generation.cpp +++ b/src/Disks/tests/gtest_cas_fence_generation.cpp @@ -43,11 +43,14 @@ namespace class TripOnHeadBackend final : public InMemoryBackend { public: - HeadResult head(const String & key) override + /// Unhide the legacy overload the primitive override below would otherwise hide. + using InMemoryBackend::head; + /// The side effect sits on the HEAD primitive, which is the only path any observation takes. + std::optional head(const String & key, TransportAccess & access) override { if (trigger) std::exchange(trigger, {})(); - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } std::function trigger; @@ -59,30 +62,35 @@ class TripOnHeadBackend final : public InMemoryBackend class TripOnSecondHeadBackend final : public InMemoryBackend { public: - using Backend::putIfAbsent; + /// Unhide the legacy overloads the primitive overrides below would otherwise hide. + using InMemoryBackend::head; - HeadResult head(const String & key) override + std::optional head(const String & key, TransportAccess & access) override { ++head_calls; if (head_calls == 2 && trigger) std::exchange(trigger, {})(); - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + /// A refused precondition in its value form: nothing was written, and the caller settles what is + /// at the key by reading -- which is the second HEAD this double trips the fence on. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { if (fail_first_put) { fail_first_put = false; - return PutResult{.outcome = PutOutcome::PreconditionFailed, .token = {}}; + return std::unexpected(RawConflict{}); } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } int head_calls = 0; - /// Default false: `Pool::open`'s own capability probe issues `putIfAbsent` calls before the test - /// gets to arm this, and those must succeed normally. The test flips this to `true` only right - /// before driving the write it actually targets. + /// Default false: `Pool::open`'s own capability probe issues writes before the test gets to arm + /// this, and those must succeed normally. The test flips this to `true` only right before driving + /// the write it actually targets. bool fail_first_put = false; std::function trigger; }; @@ -148,24 +156,28 @@ PartWriteTxnPtr precommittedBuildForBlob( class BlobPublicationFenceBackend final : public InMemoryBackend { public: + /// Unhide the legacy overload the primitive override below would otherwise hide. + using InMemoryBackend::head; enum class TripPoint : uint8_t { OnHead, AfterPublication, }; - HeadResult head(const String & key) override + /// Both seams sit on the transport primitives: a writer's mandatory HEAD and its publication both + /// reach the store through them. + std::optional head(const String & key, TransportAccess & access) override { - const HeadResult result = InMemoryBackend::head(key); + const std::optional result = InMemoryBackend::head(key, access); if (key == watched_key && trip_point == TripPoint::OnHead && trigger) std::exchange(trigger, {})(); return result; } - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { ++publish_calls; - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); if (request.destination_key == watched_key && trip_point == TripPoint::AfterPublication && trigger) std::exchange(trigger, {})(); } @@ -232,7 +244,10 @@ TEST(CASFenceGeneration, BlobPublicationFenceLossBeforeFinalCheckPublishesNothin }); EXPECT_EQ(backend->publish_calls, 0u); - EXPECT_FALSE(backend->head(backend->watched_key).exists); + { + DB::Cas::tests::OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(backend->watched_key, Retry::once()).has_value()); + } EXPECT_EQ(build->dependencyProof(ref), std::nullopt); } @@ -260,8 +275,14 @@ TEST(CASFenceGeneration, BlobPublicationHeadTripAndRearmCannotAdoptNewFenceGener }); EXPECT_EQ(backend->publish_calls, 0u); - EXPECT_FALSE(backend->head(backend->watched_key).exists); - EXPECT_EQ(loadMeta(*backend, store->layout(), ref), std::nullopt) + { + DB::Cas::tests::OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(backend->watched_key, Retry::once()).has_value()); + } + /// The mount is live again under a FRESH generation, so this read is admitted where the stale + /// operation's writes were not. + CasOperation probe = store->mountRequests().admit(); + EXPECT_FALSE(loadMeta(probe, store->layout(), ref).has_value()) << "the stale operation must not reconcile freshness metadata after trip-and-rearm"; EXPECT_EQ(build->dependencyProof(ref), std::nullopt); } @@ -283,8 +304,11 @@ TEST(CASFenceGeneration, BlobPublicationFenceLossAfterLandingReturnsNoProof) }); EXPECT_EQ(backend->publish_calls, 1u); - EXPECT_TRUE(backend->head(backend->watched_key).exists) - << "a publication that landed before fence loss is safe unreferenced debris"; + { + DB::Cas::tests::OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(backend->watched_key, Retry::once()).has_value()) + << "a publication that landed before fence loss is safe unreferenced debris"; + } EXPECT_EQ(build->dependencyProof(ref), std::nullopt); } @@ -304,8 +328,12 @@ TEST(CASFenceGeneration, PlainObjectPutAbortsWhenFenceTripsBetweenAdmissionAndDu store->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "somefile", "hello"); }); - /// No durable write ever landed -- assert via the Emulated backend listing. - EXPECT_TRUE(store->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)).empty()); + /// No durable write ever landed. Asserted through the RAW backend: every request the pool issues + /// is admitted under the mount fence, which this test has just tripped, so a read through the pool + /// would report that refusal rather than what the store holds. + DB::Cas::tests::OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).list(store->layout().namespaceFilesPrefix( + DB::Cas::tests::fixture::fixtureLife(ns)), "", 100, Retry::once()).keys.empty()); } /// `casRemoveObject`'s delete sibling, same shape: the fence trips between admission and the durable @@ -327,10 +355,13 @@ TEST(CASFenceGeneration, PlainObjectRemoveAbortsWhenFenceTripsBetweenAdmissionAn store->removeNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "victim"); }); - /// The durable delete never ran -- the object survives (reads are not fence-gated by this task). - const auto still_there = store->getNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "victim"); + /// The durable delete never ran, so the object survives -- read raw, since a read through the pool + /// is admitted under the fence this test has tripped and would report that refusal instead. + DB::Cas::tests::OperationForTest raw_op(*backend); + const auto still_there = (*raw_op).read(store->layout().namespaceFileKey( + DB::Cas::tests::fixture::fixtureLife(ns), "victim"), Retry::once()); ASSERT_TRUE(still_there.has_value()); - EXPECT_EQ(*still_there, "still here"); + EXPECT_EQ(still_there->bytes, "still here"); } /// The fence re-check must run before EVERY conditional-retry iteration, not just the first attempt @@ -354,7 +385,9 @@ TEST(CASFenceGeneration, PlainObjectPutRechecksFenceOnEveryRetryIterationNotJust }); EXPECT_EQ(backend->head_calls, 2); - EXPECT_TRUE(store->listNamespaceFiles(DB::Cas::tests::fixture::fixtureLife(ns)).empty()); + DB::Cas::tests::OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).list(store->layout().namespaceFilesPrefix( + DB::Cas::tests::fixture::fixtureLife(ns)), "", 100, Retry::once()).keys.empty()); } /// (b) The S3-native staging-buffer finalize: the fence trips AFTER the buffer is constructed diff --git a/src/Disks/tests/gtest_cas_fold_seal_codec.cpp b/src/Disks/tests/gtest_cas_fold_seal_codec.cpp index feee48a70069..dabde6cf3b82 100644 --- a/src/Disks/tests/gtest_cas_fold_seal_codec.cpp +++ b/src/Disks/tests/gtest_cas_fold_seal_codec.cpp @@ -23,7 +23,7 @@ TEST(CASFoldSealCodec, RefLifeCoverageRoundTripsLastFoldedRefId) seal.generation = 3; seal.parent_generation = 2; RefCoverage cov; - cov.classification = 1; + cov.classification = CoverageClass::Unchanged; cov.last_folded_ref_id = RefTxnId{4, 11}; constexpr UInt128 life_id{1}; seal.ref_lives[life_id].coverage = cov; diff --git a/src/Disks/tests/gtest_cas_fold_seal_format.cpp b/src/Disks/tests/gtest_cas_fold_seal_format.cpp index afe6331a4ed8..ba8c6f131be9 100644 --- a/src/Disks/tests/gtest_cas_fold_seal_format.cpp +++ b/src/Disks/tests/gtest_cas_fold_seal_format.cpp @@ -4,6 +4,8 @@ #include #include +#include + using namespace DB::Cas; namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; extern const int LOGICAL_ERROR; } @@ -15,8 +17,8 @@ CasFoldSeal sampleFoldSeal() CasFoldSeal seal; seal.generation = 7; seal.parent_generation = 6; - seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{3, 4}}; - seal.ref_lives[UInt128{2}].coverage = RefCoverage{.classification = 1}; + seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{3, 4}}; + seal.ref_lives[UInt128{2}].coverage = RefCoverage{.classification = CoverageClass::Unchanged}; seal.blob_target_runs.push_back(RunRef{.key = "gc/gen/7/blob_target/0/0", .checksum = UInt128(0xABCDEF)}); return seal; } @@ -29,23 +31,25 @@ void eraseRequiredField(String & encoded, std::string_view field) } } +CAS_BATTERY_COVERS(FoldSeal); + TEST(CASFormatBattery, FoldSeal) { CasFoldSeal seal; seal.generation = 5; seal.parent_generation = 4; - seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{7, 11}}; - seal.blob_target_runs.push_back(RunRef{.key = "r0", .checksum = UInt128(0x0f), .shard = 0, .generation = 5}); + seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{7, 11}}; + seal.blob_target_runs.push_back(RunRef{.key = "r0", .checksum = UInt128(0x0f), .shard = 0, .key_generation = 5}); seal.condemned_summary[0] = CondemnedSummary{.condemned_total = 3, .pending_total = 1, .oldest_nonpending_condemn_round = 4}; runFormatBattery({FormatId::FoldSeal, [&] { return sealObject(FormatId::FoldSeal, encodeFoldSeal(seal)); }, [](std::string_view s) { decodeFoldSeal(std::string(openObject(FormatId::FoldSeal, s))); }, currentFormatHeader("cas_fold_seal") + - "{\"g\":\"5\",\"pg\":\"4\"}\n" - "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000000001\",\"cls\":2,\"lfe\":\"7\",\"lfs\":\"11\"}\n" - "{\"k\":\"btr\",\"key\":\"r0\",\"ck\":\"0000000000000000000000000000000f\",\"shard\":0,\"gen\":\"5\"}\n" - "{\"k\":\"cnd\",\"shard\":0,\"ct\":3,\"pt\":1,\"ocr\":\"4\"}\n" + "{\"generation\":\"5\",\"parent_generation\":\"4\"}\n" + "{\"kind\":\"ref_life\",\"life\":\"00000000000000000000000000000001\",\"class\":\"folded\",\"fold_epoch\":\"7\",\"fold_seq\":\"11\"}\n" + "{\"kind\":\"blob_run\",\"key\":\"r0\",\"checksum\":\"0000000000000000000000000000000f\",\"shard\":0,\"key_generation\":\"5\"}\n" + "{\"kind\":\"condemned\",\"shard\":0,\"condemned\":3,\"pending\":1,\"oldest_round\":\"4\"}\n" "{\"n\":3}\n"}); } @@ -57,7 +61,7 @@ TEST(CASFoldSealFormat, RoundTripsAllFields) EXPECT_EQ(out.generation, in.generation); EXPECT_EQ(out.parent_generation, in.parent_generation); ASSERT_EQ(out.ref_lives.size(), in.ref_lives.size()); - EXPECT_EQ(out.ref_lives.at(UInt128{1}).coverage.classification, 2); + EXPECT_EQ(out.ref_lives.at(UInt128{1}).coverage.classification, CoverageClass::Folded); EXPECT_EQ(out.ref_lives.at(UInt128{1}).coverage.last_folded_ref_id, (RefTxnId{3, 4})); ASSERT_EQ(out.blob_target_runs.size(), 1u); EXPECT_EQ(out.blob_target_runs[0].key, "gc/gen/7/blob_target/0/0"); @@ -72,8 +76,8 @@ TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsTwoBlobTargetRunsForOneShard) seal.generation = 7; seal.parent_generation = 6; seal.blob_target_runs = { - RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, - RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .key_generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .key_generation = 7}, }; seal.condemned_summary[0] = CondemnedSummary{}; @@ -88,8 +92,8 @@ TEST(CASFoldSealFormatDeathTest, ProducerValidationRejectsMalformedSealBeforePut const Layout layout("p"); CasFoldSeal seal; seal.blob_target_runs = { - RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, - RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}}; + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .key_generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .key_generation = 7}}; seal.condemned_summary[0] = CondemnedSummary{}; EXPECT_DEATH({ validateFoldSealForWrite(seal, layout, 1); }, "duplicate blob-target shard"); } @@ -99,8 +103,8 @@ TEST(CASFoldSealFormat, ProducerValidationRejectsMalformedSealBeforePut) const Layout layout("p"); CasFoldSeal seal; seal.blob_target_runs = { - RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .generation = 7}, - RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .generation = 7}}; + RunRef{.key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, .key_generation = 7}, + RunRef{.key = layout.blobTargetRunKey(7, 2, 0, 0), .checksum = UInt128{2}, .shard = 0, .key_generation = 7}}; seal.condemned_summary[0] = CondemnedSummary{}; cas_battery_detail::expectCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { validateFoldSealForWrite(seal, layout, 1); }, "duplicate blob-target shard"); @@ -117,17 +121,17 @@ TEST(CASFoldSealFormat, AuthoritativeDecodeRequiresEveryBlobTargetAndSummaryFiel .key = layout.blobTargetRunKey(7, 1, 0, 0), .checksum = UInt128{1}, .shard = 0, - .generation = 7}); + .key_generation = 7}); seal.condemned_summary[0] = CondemnedSummary{}; const String valid = encodeFoldSeal(seal); for (const std::string_view field : { R"(,"key":"p/gc/gen/7/attempt/1/blob_target/0/0")", - R"(,"ck":"00000000000000000000000000000001")", - R"(,"gen":"7")", - ",\"ct\":0", - ",\"pt\":0", - R"(,"ocr":"18446744073709551615")"}) + R"(,"checksum":"00000000000000000000000000000001")", + R"(,"key_generation":"7")", + ",\"condemned\":0", + ",\"pending\":0", + R"(,"oldest_round":"18446744073709551615")"}) { String malformed = valid; eraseRequiredField(malformed, field); @@ -136,19 +140,19 @@ TEST(CASFoldSealFormat, AuthoritativeDecodeRequiresEveryBlobTargetAndSummaryFiel } /// `shard` occurs once on each row; remove each occurrence independently. - String missing_btr_shard = valid; - eraseRequiredField(missing_btr_shard, ",\"shard\":0"); + String missing_blob_run_shard = valid; + eraseRequiredField(missing_blob_run_shard, ",\"shard\":0"); cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(missing_btr_shard, layout, 1); }, "missing"); + [&] { decodeFoldSeal(missing_blob_run_shard, layout, 1); }, "missing"); - String missing_cnd_shard = valid; - const size_t first_shard = missing_cnd_shard.find(",\"shard\":0"); + String missing_condemned_shard = valid; + const size_t first_shard = missing_condemned_shard.find(",\"shard\":0"); ASSERT_NE(first_shard, String::npos); - const size_t second_shard = missing_cnd_shard.find(",\"shard\":0", first_shard + 1); + const size_t second_shard = missing_condemned_shard.find(",\"shard\":0", first_shard + 1); ASSERT_NE(second_shard, String::npos); - missing_cnd_shard.erase(second_shard, std::string_view(",\"shard\":0").size()); + missing_condemned_shard.erase(second_shard, std::string_view(",\"shard\":0").size()); cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(missing_cnd_shard, layout, 1); }, "missing"); + [&] { decodeFoldSeal(missing_condemned_shard, layout, 1); }, "missing"); } TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsNoncanonicalRowsAndIncompleteSummaryDomain) @@ -161,7 +165,7 @@ TEST(CASFoldSealFormat, AuthoritativeDecodeRejectsNoncanonicalRowsAndIncompleteS .key = layout.blobTargetRunKey(7, 1, 1, 0), .checksum = UInt128{1}, .shard = 1, - .generation = 7}); + .key_generation = 7}); seal.condemned_summary[0] = CondemnedSummary{}; cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, @@ -239,7 +243,7 @@ TEST(CASFoldSeal, RejectsEmptyAndBadMagic) TEST(CASFoldSeal, CoverageRecordsEveryCatalogLife) { CasFoldSeal in = sampleFoldSeal(); - in.ref_lives[UInt128{3}].coverage = RefCoverage{.classification = 0}; + in.ref_lives[UInt128{3}].coverage = RefCoverage{.classification = CoverageClass::Absent}; const CasFoldSeal out = decodeFoldSeal(encodeFoldSeal(in)); EXPECT_TRUE(out.ref_lives.contains(UInt128{3})); EXPECT_EQ(out.ref_lives.size(), 3u); @@ -252,9 +256,9 @@ TEST(CASFoldSeal, FoldSealCondemnedSummaryRoundTrips) CasFoldSeal s; s.generation = 9; s.parent_generation = 8; - s.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = 2}; + s.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = CoverageClass::Folded}; s.blob_target_runs.push_back(RunRef{.key = "gc/gen/9/blob_target/0/0", .checksum = UInt128(0x77), - .shard = 0, .generation = 9}); + .shard = 0, .key_generation = 9}); s.condemned_summary[0] = CondemnedSummary{.condemned_total = 3, .pending_total = 1, .oldest_nonpending_condemn_round = 5}; s.condemned_summary[1] = CondemnedSummary{}; /// explicit zero entry (totality over gc_shards) @@ -281,7 +285,7 @@ TEST(CASFoldSealFormat, UnifiedRefLifeRowRoundTripsCoverageHoldAndCleanupEvidenc const UInt128 life_id{0x1234}; seal.ref_lives.emplace(life_id, RefLifeFoldState{ .coverage = RefCoverage{ - .classification = 4, + .classification = CoverageClass::Clamped, .last_folded_ref_id = RefTxnId{3, 4}, .hold = RefHold{ .reason = HoldReason::ManifestBodyMissing, @@ -291,36 +295,59 @@ TEST(CASFoldSealFormat, UnifiedRefLifeRowRoundTripsCoverageHoldAndCleanupEvidenc .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{9, 10}}}); const String expected = currentFormatHeader("cas_fold_seal") + - "{\"g\":\"8\",\"pg\":\"7\"}\n" - "{\"k\":\"rfl\",\"life\":\"00000000000000000000000000001234\",\"cls\":4," - "\"lfe\":\"3\",\"lfs\":\"4\",\"hr\":\"manifest_body_missing\",\"hpe\":\"5\"," - "\"hps\":\"6\",\"hrc\":7,\"hnr\":\"8\",\"rte\":\"9\",\"rts\":\"10\"}\n" + "{\"generation\":\"8\",\"parent_generation\":\"7\"}\n" + "{\"kind\":\"ref_life\",\"life\":\"00000000000000000000000000001234\",\"class\":\"clamped\"," + "\"fold_epoch\":\"3\",\"fold_seq\":\"4\",\"hold_reason\":\"manifest_body_missing\",\"hold_epoch\":\"5\"," + "\"hold_seq\":\"6\",\"retries\":7,\"retry_round\":\"8\",\"remove_epoch\":\"9\",\"remove_seq\":\"10\"}\n" "{\"n\":1}\n"; EXPECT_EQ(encodeFoldSeal(seal), expected); EXPECT_EQ(decodeFoldSeal(expected), seal); } -/// Mutation caught: accepting the generation-6 split coverage collection would leave a second -/// namespace-keyed source of lifecycle work in a generation-7 process. +/// Closed-set pin: `CoverageClass` and `HoldReason` +/// each walked through `magic_enum::enum_values`, which is what proves the renderer and the parser +/// consult the SAME table: a table entry missing altogether is already a build error at the +/// coverage assert, but two delegates drifting onto different tables is not. +TEST(CASFoldSealFormat, ClosedSetPinsCoverageClassAndHoldReasonWords) +{ + EXPECT_EQ(coverageClassToWord(CoverageClass::Absent), "absent"); + EXPECT_EQ(coverageClassToWord(CoverageClass::Unchanged), "unchanged"); + EXPECT_EQ(coverageClassToWord(CoverageClass::Folded), "folded"); + EXPECT_EQ(coverageClassToWord(CoverageClass::Clamped), "clamped"); + for (const auto c : magic_enum::enum_values()) + EXPECT_EQ(coverageClassFromWord(coverageClassToWord(c)), c); + + EXPECT_EQ(holdReasonToWord(HoldReason::GapBelowWitness), "gap_below_witness"); + EXPECT_EQ(holdReasonToWord(HoldReason::UnconsumedSealCrossing), "unconsumed_seal_crossing"); + EXPECT_EQ(holdReasonToWord(HoldReason::WitnessDisappeared), "witness_disappeared"); + EXPECT_EQ(holdReasonToWord(HoldReason::BodyUndecodable), "body_undecodable"); + EXPECT_EQ(holdReasonToWord(HoldReason::ManifestBodyMissing), "manifest_body_missing"); + EXPECT_EQ(holdReasonToWord(HoldReason::CheckpointUndecodable), "checkpoint_undecodable"); + for (const auto r : magic_enum::enum_values()) + EXPECT_EQ(holdReasonFromWord(holdReasonToWord(r)), r); +} + +/// Mutation caught: accepting the retired split coverage-collection kind would revive a second +/// namespace-keyed source of lifecycle work alongside the unified per-life row. TEST(CASFoldSealFormat, UnifiedCodecRejectsLegacyCoverageRecord) { const String old = - "{\"type\":\"cas_fold_seal\",\"v\":7}\n" - "{\"g\":\"8\",\"pg\":\"7\"}\n" - "{\"k\":\"cov\",\"key\":\"name/0\",\"cls\":2,\"lfe\":\"3\",\"lfs\":\"4\"}\n" + "{\"type\":\"cas_fold_seal\",\"v\":1}\n" + "{\"generation\":\"8\",\"parent_generation\":\"7\"}\n" + "{\"kind\":\"cov\",\"key\":\"name/0\",\"class\":\"folded\",\"fold_epoch\":\"3\",\"fold_seq\":\"4\"}\n" "{\"n\":1}\n"; cas_battery_detail::expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(old); }, "legacy coverage"); } -/// Mutation caught: accepting the generation-6 cleanup-item state would restore the independent -/// marker-driven `Pending`/`Completed` handshake. +/// Mutation caught: accepting the retired cleanup-item kind would restore the independent +/// marker-driven `Pending`/`Completed` handshake the unified row replaced. TEST(CASFoldSealFormat, UnifiedCodecRejectsLegacyNamespaceCleanupRecord) { const String old = - "{\"type\":\"cas_fold_seal\",\"v\":7}\n" - "{\"g\":\"8\",\"pg\":\"7\"}\n" - "{\"k\":\"nsc\",\"ns\":\"name\",\"rte\":\"3\",\"rts\":\"4\",\"st\":\"completed\"}\n" + "{\"type\":\"cas_fold_seal\",\"v\":1}\n" + "{\"generation\":\"8\",\"parent_generation\":\"7\"}\n" + "{\"kind\":\"nsc\",\"ns\":\"name\",\"remove_epoch\":\"3\",\"remove_seq\":\"4\",\"st\":\"completed\"}\n" "{\"n\":1}\n"; cas_battery_detail::expectCode( DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(old); }, "legacy namespace cleanup"); diff --git a/src/Disks/tests/gtest_cas_forget.cpp b/src/Disks/tests/gtest_cas_forget.cpp index 2d8a72fc1850..d1b49b4c41a7 100644 --- a/src/Disks/tests/gtest_cas_forget.cpp +++ b/src/Disks/tests/gtest_cas_forget.cpp @@ -25,8 +25,8 @@ /// Task 10 (rev.7 spec §5): `SYSTEM CAS FORGET` — the operator force-Vanish. FORGET drives a /// content-addressed pool to `Vanished(forgotten)` with the fence-first protocol: (1) publish terminal -/// intent, (2) trip the local fence, (3+4) stop the GC scheduler, (5) join keeper/remount, drain, retire -/// the keeper WITHOUT an unearned clean farewell, (6) publish `Vanished(forgotten)` with the [D5] message +/// intent, (2) trip the local fence, (3+4) stop the GC scheduler, (5) join renewer/remount, drain, retire +/// the renewer WITHOUT an unearned clean farewell, (6) publish `Vanished(forgotten)` with the [D5] message /// carrying the decommission timestamp. These tests exercise the Pool-level protocol body (`Pool::forgetDisk`) /// and the end-to-end verb through a real `ContentAddressedMetadataStorage` (the six-class gate wired to the /// new state). Harness patterns follow gtest_cas_lifecycle_condition.cpp and gtest_cas_operation_gate.cpp. @@ -56,10 +56,11 @@ const String kForgetReason = /// gtest_cas_lifecycle_condition.cpp — used to drive a live pool into `IdentityLost`. void deleteKeyExact(DB::Cas::Backend & backend, const String & key) { - const auto got = backend.get(key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; if (got) - backend.deleteExact(key, got->token); + (*op).remove(key, got->etag, DB::Cas::Retry::once()); } /// GC's fence-out applied directly to the mount lease (preserve the body, set `gc_fenced`, bump `seq`) — @@ -67,44 +68,51 @@ void deleteKeyExact(DB::Cas::Backend & backend, const String & key) /// lease-expiry wait), reaching `armMountFence`. Mirrors gtest_cas_lifecycle_condition.cpp's helper. void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()); DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, - DB::Cas::PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(mount_key, DB::Cas::encodeMountLease(m), got->etag, DB::Cas::Retry::once()))); } -/// A Backend decorator whose head/get/list throw an untyped transport error while `fail` is armed — so a -/// self-remount attempt verdicts `StayTransient` (fast, no lease-expiry wait) and the remount loop keeps -/// spinning. Starts DISARMED so `Pool::open` succeeds. Mirrors gtest_cas_lifecycle_condition.cpp's decorator. +/// A Backend decorator whose reads, heads and lists throw an untyped transport error while `fail` is +/// armed — so a self-remount attempt verdicts `StayTransient` (fast, no lease-expiry wait) and the remount +/// loop keeps spinning. Starts DISARMED so `Pool::open` succeeds. Mirrors +/// gtest_cas_lifecycle_condition.cpp's decorator. class ToggleableTransportFaultBackend final : public DB::Cas::InMemoryBackend { public: - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - - DB::Cas::HeadResult head(const String & key) override + /// Unhide the LEGACY convenience overloads that the primitive overrides below would otherwise hide. + using Backend::head; + using Backend::list; + + /// The faults sit on the TRANSPORT PRIMITIVES, because that is where every caller reaches the store: + /// the lifecycle gate probes `_pool_meta` through `probeSentinelRaw`, which speaks only these. A + /// legacy caller still reaches the fault, through the forwarder, so arming it here covers both + /// surfaces rather than only one. + std::optional head(const String & key, DB::Cas::TransportAccess & access) override { if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::head(key); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::head(key, access); } - std::optional get(const String & key, DB::Cas::Range range) override + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::get(key, range); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::read(key, access); } - DB::Cas::ListPage list(const String & prefix, const String & cursor, size_t limit) override + + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::list(prefix, cursor, limit); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::list(prefix, cursor, limit, access); } std::atomic fail{false}; @@ -306,7 +314,7 @@ TEST(CASForget, ForgetOnIdentityLostPoolVanishesForgotten) } /// (a'') The clean-farewell is EARNED, never unconditional: on a drained pool FORGET stamps the mount lease -/// with the terminated sentinel (`min_active == UINT64_MAX`) so a same-server restart reclaims immediately, +/// with the terminated sentinel (`min_active_build_sequence == UINT64_MAX`) so a same-server restart reclaims immediately, /// but with an UNSETTLED (wedged) ref lane it must NOT — the lease is left to expire by observation. TEST(CASForget, ForgetCleanFarewellGatedOnDrain) { @@ -318,14 +326,15 @@ TEST(CASForget, ForgetCleanFarewellGatedOnDrain) auto backend = std::make_shared(); auto store = DB::Cas::tests::openPoolForTest(backend); const String mount_key = store->layout().mountKey(kSrid); - ASSERT_NE(decodeMountLease(backend->get(mount_key)->bytes).min_active, kTerminated); /// baseline + DB::Cas::tests::OperationForTest op(*backend); + ASSERT_NE(decodeMountLease((*op).read(mount_key, DB::Cas::Retry::once())->bytes).min_active_build_sequence, kTerminated); /// baseline store->forgetDisk([] {}, kForgetReason); ASSERT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); - const auto got = backend->get(mount_key); + const auto got = (*op).read(mount_key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()); - EXPECT_EQ(decodeMountLease(got->bytes).min_active, kTerminated) + EXPECT_EQ(decodeMountLease(got->bytes).min_active_build_sequence, kTerminated) << "a drained FORGET earns the clean-release farewell"; } @@ -342,9 +351,10 @@ TEST(CASForget, ForgetCleanFarewellGatedOnDrain) store->forgetDisk([] {}, kForgetReason); ASSERT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); - const auto got = backend->get(mount_key); + DB::Cas::tests::OperationForTest op(*backend); + const auto got = (*op).read(mount_key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()) << "the lease object must still be present (expiry by observation)"; - EXPECT_NE(decodeMountLease(got->bytes).min_active, kTerminated) + EXPECT_NE(decodeMountLease(got->bytes).min_active_build_sequence, kTerminated) << "an unearned clean farewell must NOT be written when the ref lanes did not drain"; } } @@ -431,12 +441,13 @@ TEST(CASForget, ForgetIntentBlocksNaturalReplacedPromotion) /// Make the identity gate verdict `Replaced`: overwrite `_pool_meta` with a FOREIGN pool_id (present, /// mismatched identity) — exactly gtest_cas_lifecycle_condition.cpp scenario (b). const String meta_key = store->layout().poolMetaKey(); - const auto got = backend->get(meta_key); + DB::Cas::tests::OperationForTest op(*backend); + const auto got = (*op).read(meta_key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()); DB::Cas::PoolMeta foreign = DB::Cas::decodePoolMeta(got->bytes); foreign.pool_id = foreign.pool_id + DB::UInt128(1); - ASSERT_EQ(backend->putOverwrite(meta_key, DB::Cas::encodePoolMeta(foreign), got->token).outcome, - DB::Cas::PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(meta_key, DB::Cas::encodePoolMeta(foreign), got->etag, DB::Cas::Retry::once()))); /// The in-flight gate (run from the GC-stop callback) reaches the `Replaced` verdict but must BAIL on the /// already-published intent rather than settle `Vanished(replaced)`. diff --git a/src/Disks/tests/gtest_cas_format.cpp b/src/Disks/tests/gtest_cas_format.cpp index 9aa93e99f76f..b3236797176c 100644 --- a/src/Disks/tests/gtest_cas_format.cpp +++ b/src/Disks/tests/gtest_cas_format.cpp @@ -1,7 +1,8 @@ #include #include -#include #include +#include +#include namespace DB::ErrorCodes { @@ -11,93 +12,52 @@ namespace DB::ErrorCodes using namespace DB::Cas; -TEST(CASFormat, ChangePointsExistForEveryClass) +/// Closed-set pin: the registry's complete set of object `type` strings. `allRegisteredFormatIds` +/// is the registry's own enumeration accessor, so this walks the SAME set the codecs and the object +/// header gate see -- a registered class with no test coverage here is a registered class this test +/// cannot see either, which is the point: a 17th, 18th, ... entry the spec's closed set does not +/// name would show up as a set-size mismatch instead of passing unnoticed. +TEST(CASFormat, RegistryTypeStringsArePinnedClosedSet) { - /// Every class that existed from the start has a non-empty, gen-1 baseline. - for (auto id : {FormatId::Blob, - FormatId::GcState, - FormatId::PoolMeta, FormatId::Roster, - FormatId::GcOutcomes, - FormatId::PartManifest, FormatId::RunFile, - FormatId::FoldSeal}) - { - auto cps = changePoints(id); - ASSERT_FALSE(cps.empty()); - EXPECT_EQ(cps.front().generation, 1u); - EXPECT_EQ(cps.front().min_reader, 1u); - } -} - -/// A class BORN after generation 1 begins its history at its birth generation, not at 1. `RefCkpt` -/// (spec INV-4) was introduced at generation 4: there is no such thing as a generation-1 `_ckpt`, and a -/// `{1, 1}` baseline would assert that a generation-1 reader could read one. Its history then gained a -/// three later breaking entries: generation 5 re-keyed it under `//`, generation 6 -/// moved it to opaque life-owned state, and generation 9 added the exact committed frontier. Neither -/// change touches the gen-1 baseline. Pinned because the decision is -/// invisible otherwise — nothing consults `changePoints` at decode time yet, so a wrong entry here -/// would sit unnoticed until the day a per-class reader floor is wired and starts admitting objects it -/// should refuse. -TEST(CASFormat, ChangePointsOfAClassBornAfterGenerationOneStartAtItsBirth) -{ - const auto cps = changePoints(FormatId::RefCkpt); - ASSERT_EQ(cps.size(), 4u); - EXPECT_EQ(cps.front().generation, kContiguousRefStreamsGeneration); - EXPECT_EQ(cps.front().min_reader, kContiguousRefStreamsGeneration); - EXPECT_GT(cps.front().generation, 1u) << "the point of this test is that it is NOT the gen-1 baseline"; - EXPECT_EQ(cps[1].generation, kNamespaceLifeKeyedGeneration); - EXPECT_EQ(cps[1].min_reader, kNamespaceLifeKeyedGeneration); - EXPECT_EQ(cps[2].generation, kOpaqueNamespaceLifeLayoutGeneration); - EXPECT_EQ(cps[2].min_reader, kOpaqueNamespaceLifeLayoutGeneration); - EXPECT_EQ(cps.back().generation, kCommittedRefFrontierGeneration); - EXPECT_EQ(cps.back().min_reader, kCommittedRefFrontierGeneration); -} + const std::set expected{ + "cas_blob", "cas_blob_meta", "cas_pool_meta", "cas_ref_log", "cas_ref_snap", + "cas_ref_ckpt", "cas_ref_catalog", "cas_gc_maintenance_state", "cas_part_manifest", + "cas_run", "cas_fold_seal", "cas_gc_state", "cas_gc_hb", "cas_gc_outcomes", + "cas_owner", "cas_epoch", "cas_mount_lease"}; + ASSERT_EQ(expected.size(), 17u); -TEST(CASFormat, PoolMetaTracksTheRecreateOnlyRecoveryFrontierGeneration) -{ - const auto cps = changePoints(FormatId::PoolMeta); - ASSERT_EQ(cps.size(), 4u); - EXPECT_EQ(cps.back().generation, kMountWriteAttemptIdGeneration); - EXPECT_EQ(cps.back().min_reader, kMountWriteAttemptIdGeneration); -} + std::set actual; + for (const auto id : allRegisteredFormatIds()) + actual.insert(traitsFor(id).type); + EXPECT_EQ(actual, expected); -TEST(CASFormat, MountAttemptIdentityIsARecreateOnlyGenerationTenChange) -{ - EXPECT_EQ(G_BUILD, 10u); - EXPECT_EQ(kMountWriteAttemptIdGeneration, 10u); - - const auto mount_points = changePoints(FormatId::MountLease); - ASSERT_EQ(mount_points.back().generation, kMountWriteAttemptIdGeneration); - EXPECT_EQ(mount_points.back().min_reader, kMountWriteAttemptIdGeneration); - - const auto pool_points = changePoints(FormatId::PoolMeta); - ASSERT_EQ(pool_points.back().generation, kMountWriteAttemptIdGeneration); - EXPECT_EQ(pool_points.back().min_reader, kMountWriteAttemptIdGeneration); + for (const auto & type : expected) + { + const FormatTraits * t = traitsForType(type); + ASSERT_NE(t, nullptr) << type; + EXPECT_EQ(t->type, type); + } } -TEST(CASPoolMeta, GenerationNinePoolIsRejectedAtReaderFloor) +/// The generation history is reset to a flat `{1, 1}` baseline for every class: CAS is pre-release and +/// carries no persisted data, so there is no compatibility cost to starting the count over. Pinned +/// because the decision is invisible otherwise — nothing consults `changePoints` at decode time yet, so +/// a wrong entry here would sit unnoticed until the day a per-class reader floor is wired and starts +/// admitting objects it should refuse. +TEST(CASFormat, EveryClassResetToTheBaselineGeneration) { - PoolMeta meta; - meta.pool_id = UInt128{1}; - meta.blob_header_len = 256; - meta.gc_shards = 1; - meta.min_reader_generation = 10; - meta.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - String encoded = encodePoolMeta(meta); - const String current = "\"v\":10"; - const size_t version = encoded.find(current); - ASSERT_NE(version, String::npos); - encoded.replace(version, current.size(), "\"v\":9"); - - try - { - decodePoolMeta(encoded); - FAIL() << "expected UNKNOWN_FORMAT_VERSION"; - } - catch (const DB::Exception & e) + for (auto id : allRegisteredFormatIds()) { - EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); - EXPECT_NE(e.message().find("generation-10 mount-attempt-identity"), String::npos); + const auto cps = changePoints(id); + ASSERT_EQ(cps.size(), 1u) << "FormatId " << static_cast(id); + EXPECT_EQ(cps.front().generation, 1u); + EXPECT_EQ(cps.front().min_reader, 1u); } + + const auto roster_cps = changePoints(FormatId::Roster); + ASSERT_EQ(roster_cps.size(), 1u); + EXPECT_EQ(roster_cps.front().generation, 1u); + EXPECT_EQ(roster_cps.front().min_reader, 1u); } TEST(CASFormat, CurrentVersionsAreGBuild) @@ -124,3 +84,18 @@ TEST(CASFormat, CheckCompatibilityFailsClosedOnFuture) EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); } } + +/// The battery's goldens spell their header version as the literal 1 rather than asking production +/// for it, so that a generation bump cannot move the expectation and the encoder output together. +/// The cost of that is a golden set which goes stale silently if nobody notices the bump; this test +/// is what notices. It fails FIRST and says what to do, so the failure that greets a generation bump +/// is one explanatory test rather than every exact-encoding golden at once. +TEST(CASFormat, HeaderVersionIsTheLiteralThisBatteryPins) +{ + EXPECT_EQ(currentCompatibilityVersion(), 1u) + << "The compatibility version has moved away from the literal 1 that " + "`currentFormatHeader` in cas_format_test_battery.h writes into every golden header. " + "That is a deliberate wire change: read the new bytes, agree to them, and update the " + "literal and the goldens together. Do NOT make the helper derive the version from " + "production again -- a golden that tracks the code it pins cannot fail."; +} diff --git a/src/Disks/tests/gtest_cas_format_battery.cpp b/src/Disks/tests/gtest_cas_format_battery.cpp index f6ba9bcdbcd0..21251204d8a5 100644 --- a/src/Disks/tests/gtest_cas_format_battery.cpp +++ b/src/Disks/tests/gtest_cas_format_battery.cpp @@ -4,20 +4,59 @@ using namespace DB::Cas; +namespace +{ +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected exception " << expected_code; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code) << e.message(); + } +} +} + /// The real cas_pool_meta case replaces the phase-1 toy proving instance. Every other control-plane /// format registers its own battery row in its own gtest_cas__format.cpp file (Tasks 3-6). +CAS_BATTERY_COVERS(PoolMeta); + TEST(CASFormatBattery, PoolMeta) { PoolMeta pm; pm.pool_id = hexToU128("00112233445566778899aabbccddeeff"); pm.blob_header_len = 256; - pm.min_reader_generation = 3; + pm.min_reader_generation = 1; pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; runFormatBattery(FormatBatteryCase{ .id = FormatId::PoolMeta, .encode = [&] { return sealObject(FormatId::PoolMeta, encodePoolMeta(pm)); }, .decode = [](std::string_view s) { decodePoolMeta(std::string(openObject(FormatId::PoolMeta, s))); }, .golden = currentFormatHeader("cas_pool_meta") + - "{\"pid\":\"00112233445566778899aabbccddeeff\",\"hln\":256,\"gcs\":1,\"mrg\":3,\"alg\":\"ch128\"}\n"}); + "{\"pool_id\":\"00112233445566778899aabbccddeeff\",\"blob_header_len\":256,\"gc_shards\":1,\"min_reader_generation\":1,\"algos_used\":[\"ch128\"]}\n"}); +} + +TEST(CASPoolMeta, RejectsInvalidAlgoArrays) +{ + const auto decode = [](std::string_view algos_used) + { + return decodePoolMeta("{\"type\":\"cas_pool_meta\",\"v\":1}\n" + "{\"pool_id\":\"00112233445566778899aabbccddeeff\",\"blob_header_len\":256,\"gc_shards\":1,\"min_reader_generation\":1,\"algos_used\":" + String(algos_used) + "}\n"); + }; + + /// The first value is the field's PREVIOUS encoding -- a comma-joined string inside one JSON + /// value. It must fail closed rather than round-trip; the rest are malformed arrays. + for (const std::string_view bad : {"\"ch128,sha256\"", "[\"ch128\",1]", "[]", "[\"sha256\",\"ch128\"]", "[\"ch128\",\"ch128\"]", "[\"unknown\"]"}) + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decode(bad); }); +} + +TEST(CASPoolMeta, ValidateAlgosUsedRejectsUnknownByte) +{ + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [] { validatePoolAlgosUsed({7}, DB::ErrorCodes::CORRUPTED_DATA, "t"); }); } diff --git a/src/Disks/tests/gtest_cas_fsck.cpp b/src/Disks/tests/gtest_cas_fsck.cpp index 71abb917eaa2..abc1ff77e505 100644 --- a/src/Disks/tests/gtest_cas_fsck.cpp +++ b/src/Disks/tests/gtest_cas_fsck.cpp @@ -9,6 +9,10 @@ #include #include #include "cas_test_helpers.h" +#include "config.h" +#if USE_AWS_S3 +#include +#endif #include #include @@ -52,7 +56,7 @@ class RepublishOnListBackend : public InMemoryBackend pending_mutation = std::move(mutation); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { std::function to_run; { @@ -65,7 +69,7 @@ class RepublishOnListBackend : public InMemoryBackend } if (to_run) to_run(); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } private: @@ -90,7 +94,7 @@ class MutateOnFirstGetBackend : public InMemoryBackend pending_mutation = std::move(mutation); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { std::function to_run; { @@ -103,7 +107,7 @@ class MutateOnFirstGetBackend : public InMemoryBackend } if (to_run) to_run(); - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } private: @@ -131,9 +135,9 @@ class FsckListingBackend : public InMemoryBackend mode = mode_; } - ListPage list(const String & listed_prefix, const String & cursor, size_t limit) override + RawListPage list(const String & listed_prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(listed_prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(listed_prefix, cursor, limit, access); if (listed_prefix != prefix) return page; if (mode == FsckListingMode::Empty) @@ -150,8 +154,12 @@ class FsckListingBackend : public InMemoryBackend FsckListingMode mode = FsckListingMode::Full; }; +#if USE_AWS_S3 /// Fail one exact GET without disturbing LIST or any other object read. This keeps the checkpoint /// authority stable while proving that fsck distinguishes a transport failure from durable corruption. +/// An access denial is the class the request engine surfaces on the first attempt instead of reissuing +/// (`InMemoryBackend::refreshCredentials` answers false by default), so the failure needs no retry +/// budget and the caller sees the injected message unchanged. class FailExactGetBackend : public InMemoryBackend { public: @@ -160,16 +168,17 @@ class FailExactGetBackend : public InMemoryBackend key = std::move(key_); } - std::optional get(const String & requested_key, Range range) override + std::optional read(const String & requested_key, TransportAccess & access) override { if (requested_key == key) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected exact GET failure"); - return InMemoryBackend::get(requested_key, range); + throw DB::S3Exception("injected access denial on exact GET", Aws::S3::S3Errors::ACCESS_DENIED); + return InMemoryBackend::read(requested_key, access); } private: String key; }; +#endif /// Publish the exact `_ckpt` authority an ordinary Live test life would have after its first committed /// record. Raw ref-log helpers deliberately do not do this: several protocol tests need malformed or @@ -177,7 +186,9 @@ class FailExactGetBackend : public InMemoryBackend /// explicit instead of accidentally borrowing the legacy LIST-only recovery rule. void writeFsckCheckpoint(Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId committed_through) { - const CasRefCatalog::Snapshot cut = CasRefCatalog::read(backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(op, layout); const auto it = std::find_if(cut.catalog.entries.begin(), cut.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); ASSERT_NE(it, cut.catalog.entries.end()); @@ -188,23 +199,25 @@ void writeFsckCheckpoint(Backend & backend, const Layout & layout, const RootNam .committed_through = committed_through, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); - const HeadResult current = backend.head(key); - const PutResult put = current.exists - ? backend.putOverwrite(key, body, current.token) - : backend.putIfAbsent(key, body); - ASSERT_EQ(put.outcome, PutOutcome::Done); + const auto current = op.head(key, Retry::once()); + const WriteResult put = current + ? op.replace(key, body, current->etag, Retry::once()) + : op.create(key, body, Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); } void writeFsckCheckpointWithBase( Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId base, std::optional last_epoch_seal = std::nullopt) { - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(backend, layout, ns); - ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = base, .checkpoint_snapshot_id = base, - .last_epoch_seal = last_epoch_seal})).outcome, PutOutcome::Done); + .last_epoch_seal = last_epoch_seal}), Retry::once()))); } void expectCheckpointBaseVerdict( @@ -231,7 +244,9 @@ void expectCheckpointBaseVerdict( /// its catalog cut, precisely the competing-cut mutation that fsck must not splice into its verdict. void replaceCatalogLife(Backend & backend, const Layout & layout, const RootNamespace & ns, UInt128 incarnation) { - CasRefCatalog::Snapshot current = CasRefCatalog::read(backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::Snapshot current = CasRefCatalog::read(op, layout); const auto it = std::find_if(current.catalog.entries.begin(), current.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); ASSERT_NE(it, current.catalog.entries.end()); @@ -239,9 +254,9 @@ void replaceCatalogLife(Backend & backend, const Layout & layout, const RootName it->state = NsState::Live; it->creator.reset(); it->removal_started_round.reset(); - ASSERT_TRUE(current.token.has_value()); - ASSERT_EQ(backend.putOverwrite(layout.refCatalogKey(), encodeRefCatalog(current.catalog), *current.token).outcome, - PutOutcome::Done); + ASSERT_TRUE(current.etag.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), encodeRefCatalog(current.catalog), *current.etag, Retry::standard()))); } FsckReport runFsckWithListingMode(FsckListingMode mode, std::string_view suffix) @@ -262,7 +277,9 @@ FsckReport runFsckWithListingMode(FsckListingMode mode, std::string_view suffix) const uint64_t frontier = publishCommittedTransition(*backend, layout, ns, "tbl", r1, r2); writeFsckCheckpoint(*backend, layout, ns, RefTxnId{1, frontier}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); backend->distort(layout.namespaceStreamPrefix(life), mode); return runFsck(*store, /*detail=*/true); } @@ -323,14 +340,16 @@ FsckReport runCheckpointBaseFsckWithListingMode( writeRefSnapshotRaw(*backend, layout, snapshotOf(base_state, ns.string())); writeFsckCheckpointWithBase(*backend, layout, ns, base); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); if (corrupt_exact_base) { const String base_snapshot_key = layout.refSnapshotKey(life, base); - const HeadResult head = backend->head(base_snapshot_key); - EXPECT_TRUE(head.exists); - if (head.exists) - EXPECT_EQ(backend->deleteExact(base_snapshot_key, head.token).kind, DeleteOutcome::Kind::Deleted); + const auto head = op.head(base_snapshot_key, Retry::once()); + EXPECT_TRUE(head.has_value()); + if (head) + EXPECT_EQ(op.remove(base_snapshot_key, head->etag, Retry::once()), Removal::Removed); } else { @@ -421,7 +440,10 @@ TEST(CASFsck, LifelessKeyIsRecordedAndTheHealthyNamespaceIsStillReported) /// Hand-built: no helper can mint the un-incarnated shape any more. const String lifeless = store->layout().casRefsPrefix() + ns.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; - ASSERT_EQ(backend->putIfAbsent(lifeless, "garbage").outcome, PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).create(lifeless, "garbage", Retry::once()))); + } FsckReport rep; ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)) @@ -464,15 +486,16 @@ TEST(CASFsck, CanonicalDeadLifeResidueIsJanitorPendingNotHardFinding) /// protocol this fixture is not driving), so inject the post-deletion catalog snapshot directly, /// mirroring `DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgresses` below. { - CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, store->layout()); + CasOperation op = store->openRequests().admit(); + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, store->layout()); const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); ASSERT_NE(it, snapshot.catalog.entries.end()); snapshot.catalog.entries.erase(it); - const auto catalog_head = backend->head(store->layout().refCatalogKey()); - ASSERT_TRUE(catalog_head.exists); - ASSERT_EQ(backend->putOverwrite(store->layout().refCatalogKey(), encodeRefCatalog(snapshot.catalog), - catalog_head.token).outcome, PutOutcome::Done); + const auto catalog_head = op.head(store->layout().refCatalogKey(), Retry::once()); + ASSERT_TRUE(catalog_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(store->layout().refCatalogKey(), encodeRefCatalog(snapshot.catalog), + catalog_head->etag, Retry::once()))); } FsckReport rep; @@ -502,13 +525,17 @@ class AdmitLifeAfterNamespaceListingBackend : public InMemoryBackend public: explicit AdmitLifeAfterNamespaceListingBackend(NamespaceLifeId life_) : protected_life(std::move(life_)) {} - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); if (!published && prefix.ends_with("/cas/ns/")) { published = true; - CasRefCatalog::casAdmitEntry(*this, Layout("p"), /*gc_shards*/1, + /// The base call above has already released the backend's lock, so admitting through this + /// same backend from here cannot deadlock. + CasRequests requests = DB::Cas::tests::openRequestsForTest(*this); + CasOperation op = requests.admit(); + CasRefCatalog::casAdmitEntry(op, Layout("p"), /*gc_shards*/1, CatalogEntry{.ns = protected_life.ns, .state = NsState::Live, .incarnation = protected_life.incarnation}); } @@ -530,8 +557,11 @@ TEST(CASFsck, LifeAdmittedBetweenNamespaceListingAndLaterCutIsNotResidue) /// The physical object exists before the listing runs, exactly as a legitimate late admission would /// leave it: written only after `casAdmitEntry` above, but here pre-seeded since the injected /// backend admits the CATALOG row, not the physical file, on the list callback. - ASSERT_EQ(backend->putIfAbsent(store->layout().namespaceFilesPrefix(life) + "format_version.txt", "1\n").outcome, - PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative( + (*op).create(store->layout().namespaceFilesPrefix(life) + "format_version.txt", "1\n", Retry::once()))); + } FsckReport rep; ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)); @@ -558,7 +588,10 @@ TEST(CASFsck, MalformedNamespaceTreeShapesStayHardFindings) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); - ASSERT_EQ(backend->putIfAbsent(c.key, "garbage").outcome, PutOutcome::Done) << c.description; + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).create(c.key, "garbage", Retry::once()))) << c.description; + } FsckReport rep; ASSERT_NO_THROW(rep = runFsck(*store, /*detail*/true)) << c.description; @@ -583,7 +616,9 @@ TEST(CASFsck, DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgr const uint64_t sequence = publishCommittedTransition(*backend, layout, unique_ns, "tbl", std::nullopt, r); writeFsckCheckpoint(*backend, layout, unique_ns, RefTxnId{1, sequence}); - CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, layout); snapshot.catalog.entries.push_back(CatalogEntry{ .ns = RootNamespace{"bad/a"}, .state = NsState::Live, .incarnation = UInt128{777}}); snapshot.catalog.entries.push_back(CatalogEntry{ @@ -593,10 +628,9 @@ TEST(CASFsck, DuplicateLifeIdIsReportedWhileAnUnrelatedUniqueNamespaceStillProgr .removal_started_round = 1}); std::sort(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); - const auto catalog_head = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(catalog_head.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head.token).outcome, - PutOutcome::Done); + const auto catalog_head = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(catalog_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head->etag, Retry::once()))); FsckReport report; ASSERT_NO_THROW(report = runFsck(*store, /*detail=*/true)); @@ -625,10 +659,11 @@ TEST(CASFsck, AmbiguousLifeUnderAPhysicalKeyIsRecordedNotAborted) writeFsckCheckpoint(*backend, layout, unique_ns, RefTxnId{1, sequence}); const NamespaceLifeId duplicated_life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"bad/a"}, UInt128{777}); - ASSERT_EQ(backend->putIfAbsent(layout.namespaceFilesPrefix(duplicated_life) + "format_version.txt", "1\n").outcome, - PutOutcome::Done); - - CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(*backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative( + op.create(layout.namespaceFilesPrefix(duplicated_life) + "format_version.txt", "1\n", Retry::once()))); + CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, layout); snapshot.catalog.entries.push_back(CatalogEntry{ .ns = RootNamespace{"bad/a"}, .state = NsState::Live, .incarnation = UInt128{777}}); snapshot.catalog.entries.push_back(CatalogEntry{ @@ -638,10 +673,9 @@ TEST(CASFsck, AmbiguousLifeUnderAPhysicalKeyIsRecordedNotAborted) .removal_started_round = 1}); std::sort(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); - const auto catalog_head = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(catalog_head.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head.token).outcome, - PutOutcome::Done); + const auto catalog_head = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(catalog_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(layout.refCatalogKey(), encodeRefCatalog(snapshot.catalog), catalog_head->etag, Retry::once()))); FsckReport report; ASSERT_NO_THROW(report = runFsck(*store, /*detail=*/true)) @@ -805,24 +839,26 @@ TEST(CASFsckAuthority, MissingBurnedEpochSealIsChainBroken) .ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = {seal}, .prev_epoch_seal = std::nullopt}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); /// The codec now rejects this skip. Deposit its old on-disk corruption shape by changing only the /// fixed-width epoch token of an otherwise encodable body, so fsck still proves that a missing /// intermediate epoch is reported rather than treated as a sparse legal transition. String skipped_bytes = encodeRefLogTxn(RefLogTxn{ .ns = ns.string(), .txn_id = RefTxnId{7, 1}, .ops = {}, .prev_epoch_seal = RefTxnId{6, 1}}); - const String old_epoch_token = R"("!pse":"6")"; + const String old_epoch_token = R"("!prev_epoch":"6")"; const auto old_epoch = skipped_bytes.find(old_epoch_token); ASSERT_NE(old_epoch, String::npos); - skipped_bytes.replace(old_epoch, old_epoch_token.size(), R"("!pse":"1")"); - ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, RefTxnId{7, 1}), - sealObject(FormatId::RefLog, skipped_bytes)).outcome, PutOutcome::Done); + skipped_bytes.replace(old_epoch, old_epoch_token.size(), R"("!prev_epoch":"1")"); + ASSERT_TRUE(std::holds_alternative(op.create(layout.refLogKey(life, RefTxnId{7, 1}), + sealObject(FormatId::RefLog, skipped_bytes), Retry::once()))); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{7, 1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = RefTxnId{6, 1}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{6, 1}}), Retry::once()))); const FsckReport report = runFsck(*store, /*detail=*/true); EXPECT_EQ(report.chain_broken, 1u); @@ -842,7 +878,9 @@ TEST(CASFsckAuthority, MissingCheckpointBaseLogIsChainBroken) fixture::admitLive(*backend, layout, ns); const RefTxnId base{1, 1}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); writeFsckCheckpointWithBase(*backend, layout, ns, base); const FsckReport report = runFsck(*store, /*detail=*/true); @@ -862,7 +900,9 @@ TEST(CASFsckAuthority, MissingCheckpointBaseSnapshotIsChainBroken) fixture::admitLive(*backend, layout, ns); const RefTxnId base{1, 1}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); writeFsckCheckpointWithBase(*backend, layout, ns, base); @@ -907,12 +947,14 @@ TEST(CASFsckAuthority, CheckpointSnapshotAtOlderEpochSealIsChainBroken) applyRefLogTxn(through_seal, seal_txn); writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{2, 1}, .checkpoint_snapshot_id = RefTxnId{1, 2}, - .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{2, 1}}), Retry::once()))); const FsckReport report = runFsck(*store, /*detail=*/true); EXPECT_EQ(report.ref_records_walked, 0u) @@ -922,8 +964,11 @@ TEST(CASFsckAuthority, CheckpointSnapshotAtOlderEpochSealIsChainBroken) "names an EpochSeal, not a snapshot base"); } -/// An unstable transport failure while exact-reading the same valid checkpoint base proves neither -/// presence nor absence. It remains the honest third answer and must not become a hard finding. +#if USE_AWS_S3 +/// A transport failure while exact-reading the same valid checkpoint base proves neither presence nor +/// absence. It remains the honest third answer and must not become a hard finding. The fault is armed +/// as an access denial so it surfaces on the read's first attempt (see `FailExactGetBackend`), which +/// keeps this test's cost at one request instead of a run through `Retry::standard()`'s deadline. TEST(CASFsckAuthority, CheckpointBaseTransportFailureIsUnchecked) { auto backend = std::make_shared(); @@ -933,7 +978,9 @@ TEST(CASFsckAuthority, CheckpointBaseTransportFailureIsUnchecked) fixture::admitLive(*backend, layout, ns); const RefTxnId base{1, 1}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); RefTableState state; @@ -946,8 +993,9 @@ TEST(CASFsckAuthority, CheckpointBaseTransportFailureIsUnchecked) const FsckReport report = runFsck(*store, /*detail=*/true); EXPECT_EQ(report.ref_records_walked, 0u); expectCheckpointBaseVerdict( - report, layout.refSnapshotKey(life, base), FsckClass::Unchecked, "injected exact GET failure"); + report, layout.refSnapshotKey(life, base), FsckClass::Unchecked, "injected access denial on exact GET"); } +#endif /// The sampled checkpoint is immutable input, but cleanup may advance `_ckpt` after that sample and /// retire its old base before fsck exact-reads it. The miss is then authority instability, not evidence @@ -962,18 +1010,21 @@ TEST(CASFsckAuthority, CheckpointBaseVanishingAfterAuthorityAdvanceIsUnchecked) fixture::admitLive(*backend, layout, ns); const RefTxnId old_base{1, 1}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); writeFsckCheckpointWithBase(*backend, layout, ns, old_base); backend->armOnFirstGet(layout.refLogKey(life, old_base), [&] { const String ckpt_key = layout.refCkptKey(life); - const HeadResult head = backend->head(ckpt_key); - ASSERT_TRUE(head.exists); - ASSERT_EQ(backend->putOverwrite(ckpt_key, encodeRefCkpt(RefCkpt{ + OperationForTest nested_op(*backend); + const auto head = (*nested_op).head(ckpt_key, Retry::once()); + ASSERT_TRUE(head.has_value()); + ASSERT_TRUE(std::holds_alternative((*nested_op).replace(ckpt_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt}), head.token).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), head->etag, Retry::once()))); }); const FsckReport report = runFsck(*store, /*detail=*/true); @@ -1347,9 +1398,10 @@ TEST(CASFsck, PhantomDanglingFromRepublishedRefIsReresolvedAway) writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, repoint_sequence}); const String old_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h1)}); - const HeadResult head = backend->head(old_key); - ASSERT_TRUE(head.exists); - backend->deleteExact(old_key, head.token); /// legitimate GC delete of the now-unreferenced blob + OperationForTest nested_op(*backend); + const auto head = (*nested_op).head(old_key, Retry::once()); + ASSERT_TRUE(head.has_value()); + (*nested_op).remove(old_key, head->etag, Retry::once()); /// legitimate GC delete of the now-unreferenced blob }); const FsckReport rep = runFsck(*store, /*detail*/true); @@ -1380,9 +1432,10 @@ TEST(CASFsck, PhantomDanglingFromDroppedRefIsReresolvedAway) writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, drop_sequence}); const String old_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h1)}); - const HeadResult head = backend->head(old_key); - ASSERT_TRUE(head.exists); - backend->deleteExact(old_key, head.token); /// legitimate GC delete after the drop folds + OperationForTest nested_op(*backend); + const auto head = (*nested_op).head(old_key, Retry::once()); + ASSERT_TRUE(head.has_value()); + (*nested_op).remove(old_key, head->etag, Retry::once()); /// legitimate GC delete after the drop folds }); const FsckReport rep = runFsck(*store, /*detail*/true); @@ -1407,9 +1460,10 @@ TEST(CASFsck, RealDanglingStillCaughtAfterReresolve) writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, sequence}); const String key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(h)}); - const HeadResult head = backend->head(key); - ASSERT_TRUE(head.exists); - backend->deleteExact(key, head.token); /// genuine loss — the ref is UNCHANGED, still names this blob + OperationForTest op(*backend); + const auto head = (*op).head(key, Retry::once()); + ASSERT_TRUE(head.has_value()); + (*op).remove(key, head->etag, Retry::once()); /// genuine loss — the ref is UNCHANGED, still names this blob const FsckReport rep = runFsck(*store, /*detail*/true); EXPECT_EQ(rep.dangling, 1u); @@ -1447,9 +1501,10 @@ TEST(CASFsck, PhantomDanglingManifestFromRepublishedRefIsReresolvedAway) *backend, store->layout(), ns, "tbl", r1, r2); /// re-publish writeFsckCheckpoint(*backend, store->layout(), ns, RefTxnId{1, repoint_sequence}); - const HeadResult head = backend->head(m1_key); - ASSERT_TRUE(head.exists); - backend->deleteExact(m1_key, head.token); /// legitimate GC delete of the superseded manifest + OperationForTest nested_op(*backend); + const auto head = (*nested_op).head(m1_key, Retry::once()); + ASSERT_TRUE(head.has_value()); + (*nested_op).remove(m1_key, head->etag, Retry::once()); /// legitimate GC delete of the superseded manifest }); const FsckReport rep = runFsck(*store, /*detail*/true); diff --git a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp index 9a6f92d3650c..ea63268a8fe3 100644 --- a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp +++ b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp @@ -1,5 +1,7 @@ #include +#include +#include #include #include #include @@ -13,6 +15,12 @@ #include #include #include "cas_test_helpers.h" +#include "config.h" + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +} namespace ProfileEvents { @@ -30,9 +38,30 @@ ManifestRef ref(const String &, uint64_t seq, uint64_t inst) { return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; } +/// A one-shot `create`, asserting it committed (mirrors the retired `backend.putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// An exact read (mirrors the retired `backend.get(key)`). +std::optional readObj(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +/// A HEAD (mirrors the retired `backend.head(key)`). +std::optional headObj(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()); +} + bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + return headObj(b, layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).has_value(); } /// Publish one physical blob through the production durable-precommit ordering. The committed fixture @@ -65,45 +94,43 @@ std::optional currentEntryFor(Backend & backend, const Layout & la return std::nullopt; } -/// Decorator reproducing the rustfs quirk (observed 2026-07-11): a conditional exact-token delete against -/// an object that is ALREADY absent can answer HTTP 412 (precondition failed), which this backend layer -/// maps to `TokenMismatch` -- not the 404-shaped `NotFound` an in-memory backend naturally returns. For -/// keys marked via `quirkOnAbsent`, `deleteExact` forces exactly that answer whenever the underlying -/// object is gone, letting a test drive the GC redelete site through the disambiguation path -/// backend-agnostically (without guessing at real rustfs HTTP mappings). -class TokenMismatchOnAbsentBackend : public InMemoryBackend +/// Counts every conditional removal sent against `watched_key` while that key is ALREADY absent. +/// A store may answer such a removal with a precondition failure rather than a clean miss (rustfs does, +/// observed 2026-07-11), so a caller that sends one cannot tell "somebody replaced it" from "it is +/// gone". The GC redelete site is not allowed to send one: it observes the blob first and compares the +/// condemned incarnation against what it saw. +class AbsentRemovalWatchBackend : public InMemoryBackend { public: - DeleteOutcome deleteExact(const String & key, const Token & token) override + void watch(const String & key) { watched_key = key; } + size_t removalsAgainstAbsent() const { return removals_against_absent; } + + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - if (quirk_keys.contains(key) && !InMemoryBackend::head(key).exists) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; - } - return InMemoryBackend::deleteExact(key, token); + if (key == watched_key && !InMemoryBackend::head(key, access)) + ++removals_against_absent; + return InMemoryBackend::remove(key, expected_value, access); } - void quirkOnAbsent(const String & key) { quirk_keys.insert(key); } - private: - std::set quirk_keys; + String watched_key; + size_t removals_against_absent = 0; }; class CkptReplacementConflictBackend : public InMemoryBackend { public: - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - if (conflict_once && key == watched_key) + /// Only the CONDITIONAL shape is refused: the fixture's own creation of the object must land. + if (conflict_once && expected_value && key == watched_key) { conflict_once = false; - return CasResult{CasOutcome::Conflict, {}}; + return std::unexpected(RawConflict{}); } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } String watched_key; @@ -115,13 +142,15 @@ TEST(CASSemanticRefFixture, WrapperCreatesInitialRecoverableCheckpoint) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/semantic-create@cas@"}; const ManifestRef manifest = ref("srv-a:1", 1, 0xAB); const uint64_t sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest); const RefTxnId expected_id{manifest.writer_epoch, sequence}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); - const auto ckpt = readCkpt(*backend, store->layout(), life); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); + const auto ckpt = readCkpt(op, store->layout(), life); ASSERT_TRUE(ckpt.has_value()); EXPECT_EQ(ckpt->ckpt.life_epoch, 1); @@ -134,24 +163,26 @@ TEST(CASSemanticRefFixture, WrapperAdvancesCheckpointWithoutDiscardingSnapshot) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/semantic-advance@cas@"}; const ManifestRef manifest = ref("srv-a:1", 1, 0xAC); const uint64_t publish_sequence = publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest); const RefTxnId publish_id{manifest.writer_epoch, publish_sequence}; writeRefSnapshotRaw(*backend, store->layout(), minimalLiveSnapshot(ns.string(), publish_id, {committedRow("tbl", manifest)})); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); - const auto before_drop = readCkpt(*backend, store->layout(), life); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); + const auto before_drop = readCkpt(op, store->layout(), life); ASSERT_TRUE(before_drop.has_value()); RefCkpt with_snapshot = before_drop->ckpt; with_snapshot.checkpoint_snapshot_id = publish_id; - ASSERT_EQ(backend->casPut( - store->layout().refCkptKey(life), encodeRefCkpt(with_snapshot), before_drop->token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(op.replace( + store->layout().refCkptKey(life), encodeRefCkpt(with_snapshot), before_drop->etag, + Retry::once()))); const uint64_t drop_sequence = dropRefTransition(*backend, store->layout(), ns, "tbl", manifest); const RefTxnId drop_id{manifest.writer_epoch, drop_sequence}; - const auto ckpt = readCkpt(*backend, store->layout(), life); + const auto ckpt = readCkpt(op, store->layout(), life); ASSERT_TRUE(ckpt.has_value()); EXPECT_EQ(ckpt->ckpt.committed_through, drop_id); @@ -163,6 +194,8 @@ TEST(CASSemanticRefFixture, CheckpointAdvanceRejectsNonMonotoneAndInvalidState) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/semantic-refusal@cas@"}; const ManifestRef manifest = ref("srv-a:1", 1, 0xAD); @@ -172,17 +205,19 @@ TEST(CASSemanticRefFixture, CheckpointAdvanceRejectsNonMonotoneAndInvalidState) const RootNamespace invalid_ns{"00/semantic-invalid@cas@"}; fixture::admitLive(*backend, store->layout(), invalid_ns); - const NamespaceLifeId invalid_life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), invalid_ns); + const NamespaceLifeId invalid_life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), invalid_ns); const String invalid_key = store->layout().refCkptKey(invalid_life); - ASSERT_EQ(backend->putIfAbsent(invalid_key, "not a checkpoint").outcome, PutOutcome::Done); + createObj(*backend, invalid_key, "not a checkpoint"); EXPECT_THROW(advanceRecoverableCkptForRawFixture(*backend, store->layout(), invalid_ns, id), DB::Exception); - EXPECT_EQ(backend->get(invalid_key)->bytes, "not a checkpoint"); + EXPECT_EQ(readObj(*backend, invalid_key)->bytes, "not a checkpoint"); } TEST(CASRawRefFixture, RawLogWriteDoesNotCreateCheckpoint) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/raw-no-ckpt@cas@"}; const RefTxnId id{1, 1}; @@ -193,14 +228,16 @@ TEST(CASRawRefFixture, RawLogWriteDoesNotCreateCheckpoint) .prev_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); - EXPECT_FALSE(readCkpt(*backend, store->layout(), life).has_value()); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); + EXPECT_FALSE(readCkpt(op, store->layout(), life).has_value()); } TEST(CASRawRefFixture, ReplaceRecoverableCheckpointWritesTheSuppliedFullState) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/replace-ckpt@cas@"}; const ManifestRef manifest = ref("srv-a:1", 1, 0xAE); const RefTxnId first_id{manifest.writer_epoch, @@ -217,8 +254,8 @@ TEST(CASRawRefFixture, ReplaceRecoverableCheckpointWritesTheSuppliedFullState) }; replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, next); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); - const auto replaced = readCkpt(*backend, store->layout(), life); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); + const auto replaced = readCkpt(op, store->layout(), life); ASSERT_TRUE(replaced.has_value()); EXPECT_EQ(replaced->ckpt.life_epoch, next.life_epoch); EXPECT_EQ(replaced->ckpt.committed_through, next.committed_through); @@ -230,12 +267,14 @@ TEST(CASRawRefFixture, ReplaceRecoverableCheckpointRejectsStaleRegressiveAndWron { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"00/replace-ckpt-refusal@cas@"}; const ManifestRef manifest = ref("srv-a:1", 1, 0xAF); const RefTxnId id{manifest.writer_epoch, publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, manifest)}; - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); - const auto existing = readCkpt(*backend, store->layout(), life); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); + const auto existing = readCkpt(op, store->layout(), life); ASSERT_TRUE(existing.has_value()); RefCkpt wrong_life = existing->ckpt; @@ -250,7 +289,7 @@ TEST(CASRawRefFixture, ReplaceRecoverableCheckpointRejectsStaleRegressiveAndWron backend->watched_key = store->layout().refCkptKey(life); backend->conflict_once = true; EXPECT_THROW(replaceRecoverableCkptForRawFixture(*backend, store->layout(), ns, existing->ckpt), DB::Exception); - EXPECT_EQ(readCkpt(*backend, store->layout(), life)->ckpt.committed_through, id); + EXPECT_EQ(readCkpt(op, store->layout(), life)->ckpt.committed_through, id); } /// The owner-removed manifest body is deleted only after a full round (its decrement is sealed — #11). @@ -264,11 +303,11 @@ TEST(CASGCRetire, ManifestBodyDeletedAfterDecrementsSealed) publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); Gc gc(store, kGc); runRegularRoundReclaiming(gc); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_TRUE(headObj(*backend, store->layout().manifestKey(ManifestId{ns, r})).has_value()); dropRefTransition(*backend, store->layout(), ns, "tbl", r); runRegularRoundReclaiming(gc); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_FALSE(headObj(*backend, store->layout().manifestKey(ManifestId{ns, r})).has_value()); } /// A publish racing the pass (in-degree restored) is SPARED, not deleted (#14). @@ -311,7 +350,7 @@ TEST(CASGCRecheck, UnreferencedBlobDeletedExactToken) dropRefTransition(*backend, store->layout(), ns, "tbl", r); // The drop's -1 condemns blob 1; the retired-cursor pipeline (condemn -> graduate -> delete) reclaims it. EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), DB::UInt128(1))); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_FALSE(headObj(*backend, store->layout().manifestKey(ManifestId{ns, r})).has_value()); } /// Task 5 (spec 2026-07-09 §raw-body-refinement, v3): GC writes the writer's freshness meta ALONGSIDE @@ -385,7 +424,7 @@ TEST(CASGCRetire, SpareLeavesMetaCondemned) publishBlobWithDurablePrecommit(store, seed_ns, "seed", id, payload); store->dropRef(seed_ns, "seed"); store->renewWatermarkOnce(); - const Token t_seed = backend->head(store->layout().blobKey(id)).token; + const Etag t_seed = headObj(*backend, store->layout().blobKey(id))->etag; const ManifestRef r1 = ref("srv-a:1", 1, 0xA1); writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", hash)}); @@ -412,7 +451,7 @@ TEST(CASGCRetire, SpareLeavesMetaCondemned) EXPECT_FALSE(currentEntryFor(*backend, store->layout(), hash).has_value()) << "the spared entry drops from the retired set"; EXPECT_TRUE(blobExists(*backend, store->layout(), hash)); - EXPECT_EQ(backend->head(store->layout().blobKey(id)).token, t_seed) + EXPECT_EQ(headObj(*backend, store->layout().blobKey(id))->etag, t_seed) << "spare does not touch the body — the incarnation token is unchanged"; /// ADD-ONLY: the spare must NOT clear the meta back to Clean (that is the deposed-leader hole). @@ -429,7 +468,7 @@ TEST(CASGCRetire, SpareLeavesMetaCondemned) const RootNamespace writer_ns{"00/spare-writer@cas@"}; auto ref_w = publishBlobWithDurablePrecommit(store, writer_ns, "writer", id, payload); EXPECT_EQ(ref_w.ref, id); - const Token t_resurrect = backend->head(store->layout().blobKey(id)).token; + const Etag t_resurrect = headObj(*backend, store->layout().blobKey(id))->etag; EXPECT_NE(t_resurrect, t_seed) << "republication displaces the body with a fresh incarnation token"; const auto lm_after = loadMetaForTest(*backend, store->layout(), hash); ASSERT_TRUE(lm_after.has_value()); @@ -438,19 +477,24 @@ TEST(CASGCRetire, SpareLeavesMetaCondemned) } /// Two-leader stale-redelete regression — the executable form of the deposed-leader spec §2. A stale -/// leader's pre-CAS exact-token redelete `deleteExact(h, t1)` must never delete a live reuse. With the -/// buggy clear-on-spare, a spare publishes `Clean`; a writer reads `Clean` and REUSES `t1`; the stale -/// `deleteExact(t1)` then deletes the LIVE body (INV_NO_LOSS). Add-only meta closes it: the spare leaves -/// `Condemned`, the writer resurrects to `t2`, and the stale `deleteExact(t1)` is a `TokenMismatch` no-op. +/// leader's pre-CAS redelete of the incarnation `t1` must never delete a live reuse. With the buggy +/// clear-on-spare, a spare publishes `Clean`; a writer reads `Clean` and REUSES `t1`; the stale redelete +/// then deletes the LIVE body (INV_NO_LOSS). Add-only meta closes it: the spare leaves `Condemned`, the +/// writer resurrects to `t2`, and the stale redelete finds a different incarnation and sends nothing. /// /// Interleaving fidelity (APPROXIMATED): the deposed leader's destructive side effect is its pre-CAS -/// exact-token `deleteExact(h, t1)`. We reproduce it deterministically by CAPTURING `t1` at condemn time -/// (exactly the token a paused leader's `delete_pending` snapshot holds) and firing that exact -/// `deleteExact` AFTER the surviving leader's spare and the writer's republication — the faithful destructive -/// op, without a mid-round CAS-interrupt seam on the delete path (which the backend does not expose). +/// redelete of `t1`. We reproduce it deterministically by CAPTURING `t1` at condemn time (exactly the +/// incarnation a paused leader's `delete_pending` snapshot holds) and replaying the round's own +/// observe-compare-remove sequence AFTER the surviving leader's spare and the writer's republication -- +/// the faithful destructive step, without a mid-round interrupt seam on the delete path. +/// +/// The replay really SENDS its removal when the comparison passes, and the removal counter below is +/// what makes the closing fsck mean something: on the clear-on-spare regression the spare publishes +/// `Clean`, the writer's dedup hit keeps `t1`, the comparison passes, the live body is removed and +/// both the counter and the fsck dangle count move. TEST(CASGCRetire, StaleRedeleteAfterSpareDoesNotDeleteLiveReuse) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPoolForTest(backend); const RootNamespace ns{"00/aa@cas@"}; @@ -472,10 +516,14 @@ TEST(CASGCRetire, StaleRedeleteAfterSpareDoesNotDeleteLiveReuse) gc.runRegularRound(); /// -1 => in-degree 0 => condemned at t1 /// The OLD leader L1's planned pre-CAS delete uses the EXACT token it observed at condemn: capture t1. + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const auto condemned_entry = currentEntryFor(*backend, store->layout(), hash); ASSERT_TRUE(condemned_entry.has_value()); - const Token t1 = condemned_entry->token; - ASSERT_EQ(backend->head(blob_key).token, t1); + const PersistedEtag t1 = condemned_entry->token; + const std::optional at_condemn = op.head(blob_key, Retry::once()); + ASSERT_TRUE(at_condemn); + ASSERT_TRUE(t1.matches(at_condemn->etag)); /// A NEW leader L2 folds a +1 that recovered h's in-degree and adopts a SPARE for h. const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); @@ -495,19 +543,29 @@ TEST(CASGCRetire, StaleRedeleteAfterSpareDoesNotDeleteLiveReuse) /// from the writer's own source — it never reuses t1. const RootNamespace writer_ns{"00/redelete-writer@cas@"}; publishBlobWithDurablePrecommit(store, writer_ns, "writer", id, payload); - const Token t2 = backend->head(blob_key).token; - EXPECT_NE(t2, t1) << "the writer resurrected to a fresh incarnation, not a reuse of t1"; - - /// L1 resumes and executes its stale pre-CAS exact-token redelete `deleteExact(h, t1)`: it must be a - /// TokenMismatch no-op (the live body is now t2), NEVER a Deleted of the live reuse. - const DeleteOutcome stale = backend->deleteExact(blob_key, t1); - EXPECT_EQ(stale.kind, DeleteOutcome::Kind::TokenMismatch) - << "the stale exact-token redelete must miss the live reuse (add-only closes INV_NO_LOSS)"; + const std::optional t2 = op.head(blob_key, Retry::once()); + ASSERT_TRUE(t2); + EXPECT_FALSE(t1.matches(t2->etag)) + << "the writer resurrected to a fresh incarnation, not a reuse of t1"; + + /// L1 resumes and replays its stale pre-CAS redelete exactly as the round performs one: observe the + /// key, compare the condemned incarnation against what is there, and remove ONLY on a match. The + /// removal is genuinely attempted on a match, so this step is destructive whenever the writer + /// reused `t1`. + const uint64_t removals_before = backend->deleteCount(blob_key); + const std::optional stale = op.head(blob_key, Retry::once()); + ASSERT_TRUE(stale); + EXPECT_FALSE(t1.matches(stale->etag)) + << "the stale redelete must miss the live reuse (add-only closes INV_NO_LOSS)"; + if (t1.matches(stale->etag)) + (void)op.remove(blob_key, stale->etag, Retry::once()); + EXPECT_EQ(backend->deleteCount(blob_key), removals_before) + << "the comparison failed, so the redelete sent no removal at all against the live body"; /// The live body under t2 survives, stays reachable via the committed r2, and fsck sees no dangle. - const HeadResult hr = backend->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_EQ(hr.token, t2); + const std::optional survivor = op.head(blob_key, Retry::once()); + ASSERT_TRUE(survivor); + EXPECT_EQ(survivor->etag, t2->etag); replaceRecoverableCkptForRawFixture( *backend, store->layout(), ns, RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, @@ -539,9 +597,11 @@ TEST(CASGCRetire, CopyForwardedBlobSurvivesWhenRepublished) /// same verified bytes under a fresh token t1, then republish a part referencing the blob (the /// promoted dst ref of a republishRef move). const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); - const Token t0 = backend->head(blob_key).token; - const auto res = backend->putOverwrite(blob_key, backend->get(blob_key)->bytes, t0); - ASSERT_EQ(res.outcome, PutOutcome::Done); + OperationForTest displace(*backend); + const Etag t0 = (*displace).head(blob_key, Retry::standard())->etag; + const WriteResult res = (*displace).replace(blob_key, readObj(*backend, blob_key)->bytes, t0, Retry::once()); + ASSERT_TRUE(std::holds_alternative(res)); + const Etag res_etag = std::get(res).etag; const ManifestRef r2 = ref("srv-a:1", 2, 0xA2); writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("a", DB::UInt128(1))}); publishCommittedTransition(*backend, store->layout(), ns, "tbl_detached", std::nullopt, r2); @@ -550,9 +610,9 @@ TEST(CASGCRetire, CopyForwardedBlobSurvivesWhenRepublished) for (int i = 0; i < 4; ++i) gc.runRegularRound(); EXPECT_FALSE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()); - const HeadResult hr = backend->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_EQ(hr.token, res.token); + const auto hr = (*displace).head(blob_key, Retry::standard()); + ASSERT_TRUE(hr.has_value()); + EXPECT_EQ(hr->etag, res_etag); } /// Copy-forward aftermath, stale-entry arm: a listed (hash, t0) entry whose incarnation was @@ -579,9 +639,11 @@ TEST(CASGCRetire, AbandonedCopyForwardDropsEntryWithoutWrongTokenDelete) ASSERT_TRUE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()); const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); - const Token t0 = backend->head(blob_key).token; - const auto res = backend->putOverwrite(blob_key, backend->get(blob_key)->bytes, t0); - ASSERT_EQ(res.outcome, PutOutcome::Done); + OperationForTest displace(*backend); + const Etag t0 = (*displace).head(blob_key, Retry::standard())->etag; + const WriteResult res = (*displace).replace(blob_key, readObj(*backend, blob_key)->bytes, t0, Retry::once()); + ASSERT_TRUE(std::holds_alternative(res)); + const Etag res_etag = std::get(res).etag; /// No events land at all (raw displacement). Drive rounds with the store's ack kept current so /// the (1, t0) entry graduates; its exact-token delete mismatches t1 and the entry drops. @@ -592,9 +654,9 @@ TEST(CASGCRetire, AbandonedCopyForwardDropsEntryWithoutWrongTokenDelete) } EXPECT_FALSE(currentEntryFor(*backend, store->layout(), DB::UInt128(1)).has_value()) << "the stale (hash, t0) entry must settle (mismatch redelete drops it), not wedge the list"; - const HeadResult hr = backend->head(blob_key); - ASSERT_TRUE(hr.exists) << "the fresh incarnation must never be deleted under the stale token"; - EXPECT_EQ(hr.token, res.token); + const auto hr = (*displace).head(blob_key, Retry::standard()); + ASSERT_TRUE(hr.has_value()) << "the fresh incarnation must never be deleted under the stale token"; + EXPECT_EQ(hr->etag, res_etag); } /// A completed round adopts the SAME attempt its fold minted (the round's single gc/state CAS commits the @@ -613,22 +675,22 @@ TEST(CASGCRecheck, CompletionInheritsFoldAttempt) Gc gc(store, kGc); gc.runRegularRound(); // round 1: one pass, single CAS commits (snap_generation, snap_attempt) - const auto after_round1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_round1 = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes); // The round adopted the attempt of THIS round's fold: snap_attempt == the lease.seq that folded it. EXPECT_EQ(after_round1.snap_attempt, after_round1.lease.seq); EXPECT_GT(after_round1.snap_generation, 0u); // The fold seal is durable under the adopted (snap_generation, snap_attempt) pair (no completion seal). - EXPECT_TRUE(backend->head(store->layout() - .foldSealKey(after_round1.snap_generation, after_round1.snap_attempt)).exists); + EXPECT_TRUE(headObj(*backend, store->layout() + .foldSealKey(after_round1.snap_generation, after_round1.snap_attempt)).has_value()); dropRefTransition(*backend, store->layout(), ns, "tbl", r); gc.runRegularRound(); // round 2: re-acquire (bump lease.seq) -> fresh attempt at its fold - const auto after_round2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_round2 = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes); EXPECT_EQ(after_round2.snap_attempt, after_round2.lease.seq); EXPECT_GT(after_round2.snap_attempt, after_round1.snap_attempt); // per-round monotone attempt EXPECT_GT(after_round2.snap_generation, after_round1.snap_generation); - EXPECT_TRUE(backend->head(store->layout() - .foldSealKey(after_round2.snap_generation, after_round2.snap_attempt)).exists); + EXPECT_TRUE(headObj(*backend, store->layout() + .foldSealKey(after_round2.snap_generation, after_round2.snap_attempt)).has_value()); } /// ---- round-paced graduation suite (spec 2026-07-02 + Task-9 amendment; re-keyed off acks in v3 Task 6) ---- @@ -650,9 +712,10 @@ TEST(CASGCAckFloor, NoOpRoundDoesNotMutateRefShards) { std::set keys; String cursor; + OperationForTest op(*backend); for (;;) { - const ListPage page = backend->list(store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*op).list(store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) keys.insert(lk.key); if (page.next_cursor.empty()) @@ -669,7 +732,7 @@ TEST(CASGCAckFloor, NoOpRoundDoesNotMutateRefShards) const std::set after = listRefKeys(); EXPECT_EQ(before, after) << "a no-op GC round must not mutate the table's ref objects"; // The registry object is gone (Task 4); the fence never existed to write it. - EXPECT_FALSE(backend->get("p/gc/registry").has_value()); + EXPECT_FALSE(readObj(*backend, "p/gc/registry").has_value()); } /// The canonical pipeline: a blob condemned at round K stays present after the condemning round; the @@ -828,6 +891,18 @@ TEST(CASGCAckFloor, PublishBeforeGraduationSpares) EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); } +namespace +{ + +/// Shared body for the fence-out timing invariant: opens a pool from `config`, then drives the exact +/// two-round scenario described below. Parameterized only by `config` so the same scenario can be run +/// against the default `PoolConfig` (`ExpiredMountFencedOutAndExcluded`) and against +/// `unsafe_remount_no_delay = true` (`CASGcFenceOut.ThresholdUnchangedByUnsafeKnob`), proving the knob +/// changes nothing about the fence-out threshold or its round count. `events` is heap-owned (not a +/// plain local) because the Pool CAN outlive the function that opened it: a background publish can hold +/// an extra `shared_from_this()` past this function's return, so a stack-local sink target -- even one +/// declared before the Pool (the fix for the 2026-07-09 ASan finding) -- is not enough. +/// /// A dead mount is fenced out by the round's heartbeat step: gc_fenced is set on its body (a /// token-guarded rewrite that bumps seq). The fence is pure liveness (re-arms the write fence so a /// resumed sleeper can never mutate again); reclaim itself no longer depends on any mount's heartbeat — @@ -838,26 +913,32 @@ TEST(CASGCAckFloor, PublishBeforeGraduationSpares) /// (`expires_at_ms`) against the GC's own clock — it fences ONLY once GC has watched a mount's write /// token hold unchanged for the full threshold on its OWN monotonic clock. That takes (at least) two /// `computeHeartbeatFloor` calls spanning the threshold, so this test drives the GC leader's own -/// (persistent) `mono_ms_fn` across two rounds: round 1 seeds the observation for both mounts; the -/// STORE's own mount is then renewed (as a live leader would) before round 2 crosses the threshold — -/// srid2, never renewed again after its one-shot claim, is the one that gets fenced. -TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) +/// (persistent) `mono_ms_fn` across three rounds: round 1 seeds the observation for both mounts; the +/// STORE's own mount is then renewed (as a live leader would); round 2, one millisecond short of the +/// threshold, discriminates the threshold's exact value (must NOT fence yet); round 3, exactly at the +/// threshold, is where srid2 — never renewed again after its one-shot claim — gets fenced. +void runExpiredMountFenceOutScenario(const PoolConfig & config) { auto backend = std::make_shared(); - std::vector events; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) - auto store = openPoolForTest(backend); + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto events = std::make_shared(); + auto store = Pool::open(backend, config); const Layout & layout = store->layout(); - // srid2's keeper claims ONE lease via `start` and is never renewed again — tests never enable + // srid2's renewer claims ONE lease via `start` and is never renewed again — tests never enable // the runtime-owned renewal worker (`background_watermark` defaults to false), so this alone models a // crashed process: a body that is live-shaped (not terminated, not fenced) but whose write token // never changes again. const String srid2 = "stale-server"; - MountLeaseKeeper srid2_keeper(backend, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, + CasRequests renewer_requests = openRequestsForTest(backend); + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + /*writer_epoch=*/1, std::chrono::milliseconds(100), [] { return 1000u; }, [] { return 0u; }, {}, std::chrono::milliseconds(0), [] { return 0u; }); - srid2_keeper.start(); - ASSERT_FALSE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); + srid2_renewer.start(); + ASSERT_FALSE(decodeMountLease(readObj(*backend, layout.mountKey(srid2))->bytes).gc_fenced); // The fence-out threshold on the GC leader's OWN monotonic clock — mirrors the production formula // in `Gc::runRegularRound` (ttl + 5% drift allowance + one round's worth of renewal slack). @@ -870,7 +951,10 @@ TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) Gc gc(store, kGc, [&] { return gc_now; }, [&] { return gc_mono; }); // Capture the emitted events so we can assert the round emits exactly one GcFenceOut row for srid2. - store->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + store->setEventSink([events](const CasEvent & e) + { + events->push(e); + }); const RootNamespace ns{"00/aa@cas@"}; const ManifestRef r = ref("srv-a:1", 1, 0xAA); @@ -885,19 +969,29 @@ TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) // The store's OWN mount renews between rounds (as a live leader would); srid2 never does. store->renewWatermarkOnce(); + gc_mono = threshold_ms - 1; + + // Round 2 (mono == threshold - 1): a discriminator for the threshold's EXACT value, not just its + // existence — one millisecond short of the full threshold, srid2's original token must NOT be fenced + // yet. Without this round, any knob-shortened positive threshold would also satisfy the fence-out + // assertion taken only at the full threshold below. + const RoundReport rep_before_threshold = gc.runRegularRound(); + EXPECT_EQ(rep_before_threshold.fence_outs, 0u); + EXPECT_FALSE(decodeMountLease(readObj(*backend, layout.mountKey(srid2))->bytes).gc_fenced); + gc_mono = threshold_ms; - // Round 2 (mono == threshold): srid2's original token has held stable for the full threshold — + // Round 3 (mono == threshold): srid2's original token has held stable for the full threshold — // fenced. The store's own (just-renewed) mount restarts its observation and stays live. const RoundReport rep = gc.runRegularRound(); EXPECT_EQ(rep.fence_outs, 1u); // exactly one dead mount fenced-out this round - const MountLease fenced = decodeMountLease(backend->get(layout.mountKey(srid2))->bytes); + const MountLease fenced = decodeMountLease(readObj(*backend, layout.mountKey(srid2))->bytes); EXPECT_TRUE(fenced.gc_fenced); // Exactly one GcFenceOut audit row was emitted, naming srid2 in its detail. size_t fence_out_rows = 0; - for (const CasEvent & e : events) + for (const CasEvent & e : events->snapshot()) if (e.type == CasEventType::GcFenceOut) { ++fence_out_rows; @@ -912,10 +1006,7 @@ TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) // srid2's writer comes back and tries to renew: its held token was invalidated by the fence rewrite, // so synchronous renewal returns a terminal failure. (It renews on its own clock; liveness is irrelevant — the token guard // trips regardless.) - const MountRenewResult renewed = srid2_keeper.renew( - CasRequestBudget{.attempt_timeout_ms = 1, .operation_deadline_ms = 10, .max_attempts = 1, - .lease_safety_margin_ms = 0, .retry_initial_backoff_ms = 0, .retry_max_backoff_ms = 0}, - MountRenewOperationEnvironment{}); + const MountRenewResult renewed = srid2_renewer.renew(MountRenewOperationEnvironment{}); ASSERT_EQ(renewed.outcome, MountRenewOutcome::Terminal); ASSERT_NE(renewed.failure, nullptr); EXPECT_THROW(std::rethrow_exception(renewed.failure), DB::Exception); @@ -926,6 +1017,25 @@ TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, blob)); } +} + +TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) +{ + runExpiredMountFenceOutScenario(PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); +} + +/// GC's fence-out threshold (`ttl + 5% drift allowance + one round's worth of renewal slack`, computed in +/// `Gc::runRegularRound`) never reads `PoolConfig::unsafe_remount_no_delay` -- that knob is consulted only +/// by `Pool::mountWritable`'s own reclaim decision, never by GC's heartbeat-floor observation. Runs the +/// IDENTICAL three-round scenario as `ExpiredMountFencedOutAndExcluded` with the knob turned on, and +/// asserts the SAME round-by-round fence-out counts -- including the one-millisecond-short discriminator +/// round -- proving the threshold's exact value and its timing are unaffected. +TEST(CASGcFenceOut, ThresholdUnchangedByUnsafeKnob) +{ + runExpiredMountFenceOutScenario( + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .unsafe_remount_no_delay = true}); +} + /// fix-round F6 (author-review: `Gc`'s own `mono_ms_fn` used to default to the RAW static `Pool:: /// bootMs()`, bypassing the Pool's own injectable `config.boot_ms_fn` -- a time-controlled test can /// desync the mount side's fake clock from the GC side's real one). This mirrors @@ -937,17 +1047,29 @@ TEST(CASGCAckFloor, ExpiredMountFencedOutAndExcluded) TEST(CASGCAckFloor, DefaultMonoClockTracksPoolsInjectedBootClockNotWallClock) { auto backend = std::make_shared(); - uint64_t fake_boot = 0; + /// Held in a shared atomic, not a plain local: this test mutates the clock below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto fake_boot = std::make_shared>(0); auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", - .boot_ms_fn = [&] { return fake_boot; }}); + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }}); const Layout & layout = store->layout(); // A stale mount, exactly as `ExpiredMountFencedOutAndExcluded`: one claim, never renewed again. const String srid2 = "stale-server"; - MountLeaseKeeper srid2_keeper(backend, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, - std::chrono::milliseconds(100), [] { return 1000u; }, [&] { return fake_boot; }); - srid2_keeper.start(); - ASSERT_FALSE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); + CasRequests renewer_requests = openRequestsForTest(backend); + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + /*writer_epoch=*/1, + std::chrono::milliseconds(100), [] { return 1000u; }, + [fake_boot] + { + return fake_boot->load(); + }); + srid2_renewer.start(); + ASSERT_FALSE(decodeMountLease(readObj(*backend, layout.mountKey(srid2))->bytes).gc_fenced); const uint64_t ttl_ms = static_cast(store->poolConfig().mount_lease_ttl_ms.count()); const uint64_t threshold_ms = ttl_ms + ttl_ms / 20 @@ -960,17 +1082,17 @@ TEST(CASGCAckFloor, DefaultMonoClockTracksPoolsInjectedBootClockNotWallClock) EXPECT_EQ(rep1.fence_outs, 0u); store->renewWatermarkOnce(); - fake_boot = threshold_ms; // advance the FAKE clock only; this test runs in well under a millisecond + fake_boot->store(threshold_ms); // advance the FAKE clock only; this test runs in well under a millisecond const RoundReport rep2 = gc.runRegularRound(); EXPECT_EQ(rep2.fence_outs, 1u) << "Gc's default mono_ms_fn must track the Pool's injected boot clock, not the real wall clock"; - EXPECT_TRUE(decodeMountLease(backend->get(layout.mountKey(srid2))->bytes).gc_fenced); + EXPECT_TRUE(decodeMountLease(readObj(*backend, layout.mountKey(srid2))->bytes).gc_fenced); } -/// deleteExact against a blob the writer RECREATED (fresh incarnation, different token) between the pending -/// publish and the deleting pass lands TokenMismatch — a terminal-OK outcome recorded as a replace: the -/// fresh incarnation is a live object and survives. report.replaced counts it. +/// A redelete of a blob the writer RECREATED (fresh incarnation) between the pending publish and the +/// deleting pass finds a different incarnation — a terminal-OK outcome recorded as a replace: the fresh +/// incarnation is a live object and survives. report.replaced counts it. TEST(CASGCAckFloor, RecreatedBlobDeleteIsTokenMismatchOk) { auto backend = std::make_shared(); @@ -1005,8 +1127,8 @@ TEST(CASGCAckFloor, RecreatedBlobDeleteIsTokenMismatchOk) // longer matches the pending entry's captured token. displaceBlobToken(*backend, store->layout(), blob_id); - // The deleting pass issues deleteExact(entry.token) → TokenMismatch → Replaced. The fresh incarnation - // survives; the entry is dropped. + // The deleting pass observes the key, finds an incarnation the entry does not name → Replaced. The + // fresh incarnation survives; the entry is dropped. const RoundReport rep = runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); EXPECT_EQ(rep.replaced, 1u); @@ -1059,27 +1181,32 @@ TEST(CASGCAckFloor, ResumeAfterCrashBetweenRetiredPutAndStateCas) // Simulate a crashed deleting pass that DID land the exact-token delete but crashed before the gc/state // CAS. The next (fresh-attempt) pass replays the delete → the object is already gone → NotFound → the // pass records Absent and completes. - ASSERT_EQ(backend->deleteExact(store->layout().blobKey(blob_id), pending_entry.token).kind, - DeleteOutcome::Kind::Deleted); - - const uint64_t round_before = decodeGcState(backend->get(store->layout().gcStateKey())->bytes).round; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const String pending_key = store->layout().blobKey(blob_id); + const std::optional doomed = op.head(pending_key, Retry::once()); + ASSERT_TRUE(doomed); + ASSERT_TRUE(pending_entry.token.matches(doomed->etag)); + ASSERT_EQ(op.remove(pending_key, doomed->etag, Retry::once()), Removal::Removed); + + const uint64_t round_before = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes).round; Gc gc2(store, kGc); const RoundReport rep = runRegularRoundReclaiming(gc2); store->renewWatermarkOnce(); EXPECT_EQ(rep.absent, 1u); // the replayed delete found the object already gone - const uint64_t round_after = decodeGcState(backend->get(store->layout().gcStateKey())->bytes).round; + const uint64_t round_after = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes).round; EXPECT_GT(round_after, round_before); // the round completed (no wedge) EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); } -/// Backend-agnostic regression for the rustfs 412-on-absent quirk: a conditional exact-token delete -/// against an object that is ALREADY absent answers `TokenMismatch`, not `NotFound`, on this backend -/// (`TokenMismatchOnAbsentBackend` reproduces it deterministically). The redelete site must disambiguate -/// via a follow-up HEAD: the object is truly gone, so the outcome must settle as Absent (never Replaced) -/// and the `.meta` cleanup (gated on Deleted/NotFound) must still run. -TEST(CASGCAckFloor, TokenMismatchOnAbsentBlobSettlesAsAbsentAndDropsMeta) +/// A blob whose body a crashed pass already deleted must settle as Absent (never Replaced), its `.meta` +/// cleanup must still run, and -- the part a store can punish -- the round must not send a conditional +/// removal against the absent key at all. A store may answer such a removal with a precondition failure +/// instead of a clean miss (rustfs does), which is indistinguishable from "somebody replaced it"; the +/// round observes first, so it never has to tell the two apart. +TEST(CASGCAckFloor, AbsentBlobSettlesAsAbsentWithoutASpeculativeConditionalRemoval) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPoolForTest(backend); const RootNamespace ns{"00/aa@cas@"}; const ManifestRef r = ref("srv-a:1", 1, 0xAA); @@ -1117,32 +1244,37 @@ TEST(CASGCAckFloor, TokenMismatchOnAbsentBlobSettlesAsAbsentAndDropsMeta) ASSERT_EQ(lm->meta.state, MetaState::Condemned); } - // The object is genuinely gone already (as if a prior crashed pass landed the delete); confirm that, - // then arm the quirk so the NEXT conditional delete against this now-absent key answers TokenMismatch - // instead of NotFound (the rustfs 412-on-absent behavior). + // The object is genuinely gone already (as if a prior crashed pass landed the delete), and from here + // every removal the round sends against this key would be sent against an absent object. + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const String blob_key = store->layout().blobKey(blob_id); - ASSERT_EQ(backend->deleteExact(blob_key, pending_entry.token).kind, DeleteOutcome::Kind::Deleted); - ASSERT_FALSE(backend->head(blob_key).exists); - backend->quirkOnAbsent(blob_key); - - // The deleting pass replays deleteExact(entry.token): the backend answers TokenMismatch (quirk), but - // the follow-up HEAD shows the object absent, so the fix disambiguates the outcome to Absent and still - // runs the `.meta` cleanup. + const std::optional doomed = op.head(blob_key, Retry::once()); + ASSERT_TRUE(doomed); + ASSERT_TRUE(pending_entry.token.matches(doomed->etag)); + ASSERT_EQ(op.remove(blob_key, doomed->etag, Retry::once()), Removal::Removed); + ASSERT_FALSE(op.head(blob_key, Retry::once())); + backend->watch(blob_key); + + // The deleting pass replays the redelete: it observes the absent key and settles Absent without a + // request the store could answer ambiguously, and the `.meta` cleanup still runs. const RoundReport rep = runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); - EXPECT_EQ(rep.absent, 1u) << "the 412-on-absent quirk must settle as Absent, not Replaced"; + EXPECT_EQ(rep.absent, 1u) << "an already-absent blob settles as Absent, not Replaced"; EXPECT_EQ(rep.replaced, 0u); + EXPECT_EQ(backend->removalsAgainstAbsent(), 0u) + << "the redelete observed the key first, so it sent no conditional removal against an absent object"; EXPECT_FALSE(currentEntryFor(*backend, store->layout(), blob).has_value()); EXPECT_FALSE(loadMetaForTest(*backend, store->layout(), blob).has_value()) - << ".meta cleanup (gated on Deleted/NotFound) must still run on the disambiguated Absent outcome"; + << ".meta cleanup (gated on a removal or a proven absence) must still run on the Absent outcome"; } /// ---- condemn-marker gate suite ---- /// /// The per-hash condemn marker is LOAD-BEARING for the delete edge: the writer's adopt gate point-reads /// the meta and an ABSENT meta reads as Clean, so a blob whose condemn-marker write was swallowed can be -/// same-token adopted by a writer landing in the [discovery-LIST, deleteExact] window — invisible to the -/// graduating fold — and the exact-token redelete then deletes a body under a live committed edge +/// same-token adopted by a writer landing in the [discovery-LIST, redelete] window — invisible to the +/// graduating fold — and the redelete then deletes a body under a live committed edge /// (dangling manifest). Graduation to `delete_pending` therefore requires CONFIRMED durable `Condemned` /// evidence for the entry; absent evidence CARRIES the entry to the next round (fail-safe delay, never a /// fail-open delete) and retries the marker so a healed backend restores liveness. @@ -1153,8 +1285,37 @@ TEST(CASGCAckFloor, TokenMismatchOnAbsentBlobSettlesAsAbsentAndDropsMeta) TEST(CASGCCondemnMarker, SwallowedMarkerWriteCarriesEntryInsteadOfDeleting) { auto backend = std::make_shared(); + /// A SWALLOWED write is the premise: the store may have applied it and said nothing, which is the + /// only shape that leaves the round committing an entry whose marker it cannot confirm. The + /// propagating kind never reaches the engine's resolve-and-reissue path at all -- the write loop + /// rethrows a non-`Poco::Exception` on its first attempt. + backend->armWriteFault(MetaWriteFaultBackend::FaultKind::Ambiguous); auto store = openPoolForTest(backend); store->setCasRetrySleepForTest([](uint64_t) {}); + + /// THE FAULT IS PERMANENT, so every condemn-marker write runs the engine's WHOLE retry window, and + /// that window has to run on a clock this test advances. A zeroed sleep alone does not do it: the + /// deadline is still measured against the real clock, so the loop would spin hot for ninety real + /// seconds per marker write. The sleep therefore moves the clock past its own pause -- plus one + /// millisecond, because full-jitter backoff may draw zero and a clock that never moves never closes + /// the window. Scoped to the GC plane, which is where `writeCondemnedMeta` runs, so the mount + /// plane's lease-bound policies keep their real clock. + /// Held in shared, heap-owned atomics, not plain locals: `store->openRequests()` is the Pool's own + /// persistent engine, and the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local -- even an already-atomic one -- + /// would dangle once the frame returns. + auto engine_now_ms = std::make_shared>(0); + auto engine_sleeps = std::make_shared>(0); + store->openRequests().setNowFnForTest([engine_now_ms] + { + return engine_now_ms->load(); + }); + store->openRequests().setSleepFnForTest([engine_now_ms, engine_sleeps](uint64_t pause_ms) + { + engine_sleeps->fetch_add(1); + engine_now_ms->fetch_add(pause_ms + 1); + }); + const RootNamespace ns{"00/aa@cas@"}; const ManifestRef r = ref("srv-a:1", 1, 0xAA); const UInt128 blob = DB::UInt128(1); @@ -1165,9 +1326,18 @@ TEST(CASGCCondemnMarker, SwallowedMarkerWriteCarriesEntryInsteadOfDeleting) runRegularRoundReclaiming(gc); // +1 folds; blob referenced dropRefTransition(*backend, store->layout(), ns, "tbl", r); - runRegularRoundReclaiming(gc); // the condemning round; the controlled marker write exhausts as Unresolved + runRegularRoundReclaiming(gc); // the condemning round; the marker write gives up without committing ASSERT_FALSE(loadMetaForTest(*backend, store->layout(), blob).has_value()) << "precondition: the injected fault must have lost the condemn-marker write"; + /// The marker write was resolved and reissued, then gave up at its own policy window. A give-up + /// needs the next jittered pause not to fit before the deadline and full jitter draws at most five + /// seconds, so it cannot happen before the clock has passed `Retry::standard()`'s window minus + /// that draw. Both assertions pin the REISSUING, which is what the ambiguous kind buys; neither + /// can tell the injected clock from the real one -- that seam bounds the reissuing in real time. + EXPECT_GT(engine_sleeps->load(), 1u) + << "an ambiguous marker write must be resolved and reissued, not surfaced on its first attempt"; + EXPECT_GT(engine_now_ms->load(), 85'000u) + << "the marker write must have spent its whole retry window before reporting failure"; ASSERT_TRUE(currentEntryFor(*backend, store->layout(), blob).has_value()) << "precondition: the retired entry must have been committed despite the lost marker"; @@ -1287,3 +1457,94 @@ TEST(CASGCCondemnMarker, LoadMetaFallbackConfirmsGraduationAfterLeaderRestart) EXPECT_TRUE(e->marker_confirmed) << "a delete_pending row confirmed via loadMeta still carries the bit"; EXPECT_TRUE(blobExists(*backend, store->layout(), blob)); } + +#if USE_AWS_S3 +/// The outcomes-log `create` meets the same shape as the round commit: a refused precondition whose +/// resolve read was itself refused. Nothing observed the key, so the round may not report that the log +/// vanished -- an absent key and an unreadable one are different answers. +/// +/// The S3 gate is the fault's, not the site's: the definitive-refusal classification that makes a +/// resolve read settle nothing rather than be reissued exists only for S3 errors. +TEST(CASGCRetire, OutcomeLogUnobservedConflictDoesNotReportItVanished) +{ + /// The fault's state is shared between the round (writes, on the test thread) and the meta + /// writer's pool (reads), so it lives under its own mutex. + class UnobservedOutcomesBackend : public InMemoryBackend + { + public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + { + std::lock_guard lock(fault_mutex); + if (arm && !expected_value && key.find("/outcomes/") != String::npos) + { + arm = false; + refused_key = key; + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + std::optional read(const String & key, TransportAccess & access) override + { + { + std::lock_guard lock(fault_mutex); + if (!refused_key.empty() && key == refused_key) + { + refused_key.clear(); + throw DB::S3Exception("UnobservedOutcomesBackend: the settling read is definitively refused", + Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); + } + } + return InMemoryBackend::read(key, access); + } + + void armOnce() + { + std::lock_guard lock(fault_mutex); + arm = true; + } + + private: + std::mutex fault_mutex; + bool arm TSA_GUARDED_BY(fault_mutex) = false; + String refused_key TSA_GUARDED_BY(fault_mutex); + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref("srv-a:1", 1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + Gc gc(store, kGc); + gc.runRegularRound(); + dropRefTransition(*backend, store->layout(), ns, "tbl", r); + + /// The condemn -> graduate -> delete pipeline needs several rounds before any round has an outcome + /// to log; the arm fires on the first one that does. + backend->armOnce(); + bool refusal_reached = false; + for (int i = 0; i < 8 && !refusal_reached; ++i) + { + try + { + runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + } + catch (const DB::Exception & e) + { + refusal_reached = true; + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED); + EXPECT_NE(e.message().find("resolve read observed nothing"), String::npos) << e.message(); + EXPECT_EQ(e.message().find("vanished"), String::npos) + << "nothing observed the key, so it may not be called vanished: " << e.message(); + } + } + EXPECT_TRUE(refusal_reached) << "no round ever wrote an outcome log, so the arm was never reached"; +} +#endif diff --git a/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp index 870bcdc20295..d583b31df256 100644 --- a/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp +++ b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp @@ -26,7 +26,7 @@ /// * absent at `expected`, no listed id above it => the namespace's frontier this round (normal end) /// * absent at `expected`, a listed id above it => IMPOSSIBLE under contiguity: the store is lying /// or a durable record was lost. Hold the namespace -/// (classification 4), cursor unmoved. +/// (classification `Clamped`), cursor unmoved. /// /// Epochs are crossed ONLY by consuming the `EpochSeal` that closes an epoch (INV-2). The seal folds as /// an applied no-op (probe B2: `produced=false`), and the next epoch's start is `{E', 1}` -- reached @@ -58,7 +58,8 @@ std::optional coverageOf(Backend & backend, const Layout & layout, const UInt128 life_id = catalogLifeIdForTest(backend, layout, ns); for (uint64_t g = gen; ; --g) { - if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + OperationForTest op(backend); + if (const auto got = (*op).read(layout.foldSealKey(g, attempt), Retry::once())) { const CasFoldSeal seal = decodeFoldSeal(got->bytes); const auto it = seal.ref_lives.find(life_id); @@ -77,10 +78,10 @@ RefTxnId cursorOf(Backend & backend, const Layout & layout, const RootNamespace return cov ? cov->last_folded_ref_id : RefTxnId{}; } -uint8_t classificationOf(Backend & backend, const Layout & layout, const RootNamespace & ns) +CoverageClass classificationOf(Backend & backend, const Layout & layout, const RootNamespace & ns) { const auto cov = coverageOf(backend, layout, ns); - return cov ? cov->classification : 0; + return cov ? cov->classification : CoverageClass::Absent; } /// The `fold_ref_intake` phase metrics of the round `sched` runs -- the only place probe B1's two @@ -135,7 +136,7 @@ TEST(CASGCArithmeticIntake, HintOmittingMiddleRecordsFoldsThroughUnnoticed) ASSERT_GT(backend->holesServed(), 0u) << "the hint hole was never actually served"; EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 5})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 2) << "a folded namespace is `changed`"; + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Folded) << "a folded namespace is `changed`"; for (uint64_t i = 1; i <= 5; ++i) EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) << "blob " << i << " lost its owner edge: its record was skipped because the hint omitted it"; @@ -166,13 +167,13 @@ TEST(CASGCArithmeticIntake, WalkEndsAtFrontierWithoutHold) ASSERT_TRUE(gc.runRegularRound().acquired_lease); EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Folded); /// A second round over an unchanged namespace pays exactly one exact GET, finds the same frontier, /// and neither advances nor holds. ASSERT_TRUE(gc.runRegularRound().acquired_lease); EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 3})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 1) << "an unchanged namespace is `carried`"; + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Unchanged) << "an unchanged namespace is `carried`"; } /// ===================== EPOCHS ARE CROSSED ONLY BY CONSUMING A SEAL ===================== @@ -211,7 +212,7 @@ TEST(CASGCArithmeticIntake, SealCrossesEpochAndIsAppliedAsNoOp) ASSERT_GT(backend->holesServed(), 0u); EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{2, 2})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Folded); for (uint64_t i = 1; i <= 4; ++i) EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(i)), 1) << "blob " << i; } @@ -291,19 +292,19 @@ TEST(CASGCArithmeticIntake, CursorRestingOnSealCrossesInALaterRound) ASSERT_TRUE(gc.runRegularRound().acquired_lease); EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{2, 1})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 2); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Folded); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); } /// ===================== IMPOSSIBLE SHAPES HOLD THE NAMESPACE ===================== /// /// `{1,3}` is genuinely absent while `{1,4}` is present AND listed. Contiguity says that cannot happen, -/// so whatever sits behind the gap may be an acked `+1`: the namespace is held at classification 4 with -/// its cursor UNMOVED, rather than sealing past the gap. +/// so whatever sits behind the gap may be an acked `+1`: the namespace is held at classification +/// `Clamped` with its cursor UNMOVED, rather than sealing past the gap. /// /// Listing-driven intake folded `{1,4}` and sealed the cursor at it -- permanently, since a record below /// the cursor is never re-read. -TEST(CASGCArithmeticIntake, GapBelowWitnessHoldsNamespaceAtClassificationFour) +TEST(CASGCArithmeticIntake, GapBelowWitnessHoldsNamespaceAtClampedClassification) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); @@ -326,7 +327,7 @@ TEST(CASGCArithmeticIntake, GapBelowWitnessHoldsNamespaceAtClassificationFour) EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) << "the cursor must not advance past a gap"; - EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Clamped); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(4)), 0) << "the record above the gap was not folded"; } @@ -357,7 +358,7 @@ TEST(CASGCArithmeticIntake, UnconsumedSealCrossingHoldsNamespace) ASSERT_TRUE(gc.runRegularRound().acquired_lease); EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 1})); - EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Clamped); const auto coverage = coverageOf(*backend, layout, ns); ASSERT_TRUE(coverage && coverage->hold.has_value()); EXPECT_EQ(coverage->hold->reason, HoldReason::UnconsumedSealCrossing); @@ -395,7 +396,7 @@ TEST(CASGCArithmeticIntake, CrossingFromANonSealRecordIsRefusedEvenWhenTheChainM EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) << "epoch 1 was never sealed, so the cursor may not leave it"; - EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Clamped); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(1)), 1) << "epoch 1's records still fold"; EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 1); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(3)), 0) @@ -415,15 +416,14 @@ TEST(CASGCArithmeticIntake, EpochStartThatAnswersOnlyEveryOtherReadHoldsInsteadO class AlternatingGetBackend : public InMemoryBackend { public: - using DB::Cas::Backend::get; String flaky; size_t reads = 0; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (key == flaky && ++reads % 2 == 0) return std::nullopt; - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } }; @@ -462,7 +462,7 @@ TEST(CASGCArithmeticIntake, EpochStartThatAnswersOnlyEveryOtherReadHoldsInsteadO EXPECT_EQ(cursorOf(*backend, layout, ns), (RefTxnId{1, 2})) << "the cursor stops on the seal it consumed and never enters the unstable epoch"; - EXPECT_EQ(classificationOf(*backend, layout, ns), 4); + EXPECT_EQ(classificationOf(*backend, layout, ns), CoverageClass::Clamped); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(2)), 0); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(3)), 0) << "nothing above the unstable position may be folded either"; @@ -483,7 +483,10 @@ TEST(CASGCArithmeticIntake, CorruptBodyClampsOneNamespaceWhileAnotherFolds) const RootNamespace ns_b{"00/bb@cas@"}; publishAt(*backend, layout, ns_a, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); - backend->putIfAbsent(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, 2}), "this is not a cas_ref_log object"); + { + OperationForTest op(*backend); + (*op).create(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, 2}), "this is not a cas_ref_log object", Retry::once()); + } writeRecoverableCkptForRawFixture(*backend, layout, ns_a, RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, @@ -505,11 +508,11 @@ TEST(CASGCArithmeticIntake, CorruptBodyClampsOneNamespaceWhileAnotherFolds) ASSERT_TRUE(gc.runRegularRound().acquired_lease); EXPECT_EQ(cursorOf(*backend, layout, ns_a), (RefTxnId{1, 1})); - EXPECT_EQ(classificationOf(*backend, layout, ns_a), 4); + EXPECT_EQ(classificationOf(*backend, layout, ns_a), CoverageClass::Clamped); EXPECT_EQ(cursorOf(*backend, layout, ns_b), (RefTxnId{1, 3})) << "a sibling namespace's corrupt body must not stop this one"; - EXPECT_EQ(classificationOf(*backend, layout, ns_b), 2); + EXPECT_EQ(classificationOf(*backend, layout, ns_b), CoverageClass::Folded); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(11)), 1); EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(12)), 1); } @@ -582,7 +585,7 @@ TEST(CASGCArithmeticIntake, WhollyOmittedNamespaceFoldsThroughAuthoritativeCheck const auto hidden_cov = coverageOf(*backend, layout, ns); ASSERT_TRUE(hidden_cov.has_value()) << "the namespace is `Live` in the catalog, so it stays in the universe even fully hidden"; - EXPECT_EQ(hidden_cov->classification, 2) << "the checkpoint's frontier is folded by exact key"; + EXPECT_EQ(hidden_cov->classification, CoverageClass::Folded) << "the checkpoint's frontier is folded by exact key"; EXPECT_EQ(hidden_cov->last_folded_ref_id, (RefTxnId{1, 3})); /// The store stops lying: the already folded namespace reappears. diff --git a/src/Disks/tests/gtest_cas_gc_attempt.cpp b/src/Disks/tests/gtest_cas_gc_attempt.cpp index b919298ae24e..546eda954a0e 100644 --- a/src/Disks/tests/gtest_cas_gc_attempt.cpp +++ b/src/Disks/tests/gtest_cas_gc_attempt.cpp @@ -47,16 +47,17 @@ ManifestRef ref(const String &, uint64_t seq, uint64_t inst) /// Whether a blob's body object is present in the backend (HEADs the object key directly). bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + OperationForTest op(b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}), Retry::once()).has_value(); } /// Whether the CURRENT retired list (any gc-shard) still holds an entry — the ack-floor deletion pipeline /// is in flight while this is true. bool anyRetiredPending(const PoolPtr & s) { - /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// Condemned state rides the adopted fold seal's RunMarker::Condemned rows, not a /// separate retired list — reconstruct the in-flight set from the seal. - return anyCondemnedInSeal(s->backend(), s->layout()); + return anyCondemnedInSeal(*s->poolBackendPtr(), s->layout()); } /// Drive regular GC to a fixpoint over the ACK-FLOOR round (advancing the store's own mount ack after each @@ -79,30 +80,36 @@ size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) return rounds; } -/// A backend that throws ONCE on the SINGLE round-commit `gc/state` CAS — the casPut that advances -/// snap_generation (the one-pass round has exactly one such CAS; the lease-acquire CAS does not advance -/// snap_generation, so "advances snap_generation" uniquely picks the round commit). +/// A backend that refuses ONCE the SINGLE round-commit `gc/state` write — the conditional write that +/// advances snap_generation (the one-pass round has exactly one such write; the lease acquire/renew does +/// not advance snap_generation, so "advances snap_generation" uniquely picks the round commit). class InterruptRoundCasBackend : public InMemoryBackend { public: explicit InterruptRoundCasBackend(String gc_state_key_) : gc_state_key(std::move(gc_state_key_)) {} - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - if (arm_interrupt && key == gc_state_key) + if (arm_interrupt && expected_value && key == gc_state_key) { - const auto stored = get(key); - const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; - const uint64_t next_gen = decodeGcState(bytes).snap_generation; - if (next_gen > stored_gen) + const auto stored = InMemoryBackend::read(key, access); + if (stored + && decodeGcState(bytes).snap_generation > decodeGcState(stored->bytes).snap_generation) { - arm_interrupt = false; /// one-shot: only depose the first round-commit CAS - throw DB::Exception(DB::ErrorCodes::ABORTED, - "test-injected: round-commit gc/state CAS denied (leader deposed mid-round; lease lost)"); + arm_interrupt = false; /// one-shot: only depose the first round-commit write + /// A REFUSAL, not a throw: a thrown transport error is an ambiguity the engine settles + /// by an exact read and then reissues while the precondition it named is unmoved, so + /// the round would commit on the reissue. A refused precondition ends the write at + /// once. The object is moved too -- the same bytes under a fresh incarnation -- because + /// a store refuses only what changed; the CONTENT is deliberately left alone, so this + /// round's own lease and cursor are exactly what a deposed round leaves behind. + (void)InMemoryBackend::write(key, stored->bytes, stored->value, access); + return std::unexpected(RawConflict{}); } } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool arm_interrupt = false; @@ -129,12 +136,13 @@ TEST(CASGCAttempt, DeposedFoldAttemptDoesNotWedge) publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); Gc gc(store, kGcA); + OperationForTest raw_op(*backend); // Round 1 (honest): fold the +1 so the blob is pinned in the in-degree generation, and adopt the // first (snap_generation, snap_attempt). runRegularRoundReclaiming(gc); EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) << "blob pinned by the committed ref"; - const auto after_fold = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_fold = decodeGcState((*raw_op).read(store->layout().gcStateKey(), Retry::once())->bytes); ASSERT_EQ(after_fold.snap_attempt, after_fold.lease.seq); ASSERT_GT(after_fold.snap_generation, 0u); @@ -149,7 +157,7 @@ TEST(CASGCAttempt, DeposedFoldAttemptDoesNotWedge) EXPECT_ANY_THROW(runRegularRoundReclaiming(gc)); // ABORTED: round-commit CAS denied backend->arm_interrupt = false; - const auto after_deposed = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_deposed = decodeGcState((*raw_op).read(store->layout().gcStateKey(), Retry::once())->bytes); EXPECT_EQ(after_deposed.snap_generation, after_fold.snap_generation) << "the denied round-commit CAS must NOT advance the adopted generation"; EXPECT_EQ(after_deposed.snap_attempt, after_fold.snap_attempt) @@ -164,9 +172,9 @@ TEST(CASGCAttempt, DeposedFoldAttemptDoesNotWedge) const uint64_t a1 = after_fold.lease.seq + 1; // round 2 renewed the lease => seq bumped once const uint64_t g_f = after_fold.snap_generation + 1; // the generation the deposed fold minted EXPECT_NE(a1, after_deposed.snap_attempt) << "the deposed attempt must differ from the adopted one"; - EXPECT_TRUE(backend->head(store->layout().foldSealKey(g_f, a1)).exists) + EXPECT_TRUE((*raw_op).head(store->layout().foldSealKey(g_f, a1), Retry::once()).has_value()) << "the deposed leader's fold seal is durable under its own (unadopted) attempt a1"; - EXPECT_FALSE(backend->head(store->layout().foldSealKey(g_f, after_deposed.snap_attempt)).exists) + EXPECT_FALSE((*raw_op).head(store->layout().foldSealKey(g_f, after_deposed.snap_attempt), Retry::once()).has_value()) << "no fold seal exists under the still-adopted attempt at the deposed fold generation (orphan is invisible)"; // An HONEST GC to a fixpoint (CAS now allowed). The KEY property: with attempt-scoping this SUCCEEDS — @@ -183,7 +191,7 @@ TEST(CASGCAttempt, DeposedFoldAttemptDoesNotWedge) // GC advanced past the deposed attempt: the adopted (snap_generation, snap_attempt) moved on, and the // adopted attempt is a fresh one (never the deposed a1). - const auto after_drain = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_drain = decodeGcState((*raw_op).read(store->layout().gcStateKey(), Retry::once())->bytes); EXPECT_GT(after_drain.snap_generation, after_fold.snap_generation) << "completion advanced the generation"; EXPECT_NE(after_drain.snap_attempt, a1) << "the drained round never adopted the deposed attempt a1"; } diff --git a/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp b/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp index e9041dd7b223..f97235ffd5aa 100644 --- a/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp +++ b/src/Disks/tests/gtest_cas_gc_bounded_walk.cpp @@ -1,5 +1,7 @@ #include +#include + #include #include #include @@ -54,12 +56,13 @@ using CountingHintHoleBackend = DB::Cas::tests::HintHoleBackendOn coverageOf(Backend & backend, const Layout & layout, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest op(backend); const uint64_t gen = currentGenerationOf(backend, layout); const uint64_t attempt = currentAttemptOf(backend, layout); const UInt128 life_id = catalogLifeIdForTest(backend, layout, ns); for (uint64_t g = gen; ; --g) { - if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + if (const auto got = (*op).read(layout.foldSealKey(g, attempt), Retry::standard())) { const CasFoldSeal seal = decodeFoldSeal(got->bytes); const auto it = seal.ref_lives.find(life_id); @@ -128,22 +131,22 @@ std::map runRoundCapturingIntake(Gc & gc, UniversePolicy policy /// A store whose writer keeps pace with the walker EXACTLY: every time the fold reads the newest record /// by exact key, one more record lands above it. /// -/// This is a mid-round appender expressed as a synchronous hook rather than as a thread, and the -/// determinism is the point. The property under test is "the round stops at the tail it froze, however -/// much arrives afterwards", and a thread can only make appends arrive at times the scheduler chooses -- -/// including, on an unlucky run, entirely after the walk has gone past. The hook reproduces the WORST -/// case (writer rate == walker rate, the rate at which the unbounded walk provably never terminates) on -/// every run, and `max_appends` bounds it so that the UNPATCHED walk still finishes and can be measured -/// rather than hanging the suite. +/// This is a mid-round appender expressed as a hook rather than as a background thread, and the +/// determinism of WHEN it fires is the point: a real thread can only make appends arrive at times the +/// scheduler chooses, including, on an unlucky run, entirely after the walk has gone past. The hook +/// reproduces the WORST case (writer rate == walker rate, the rate at which the unbounded walk provably +/// never terminates) on every run, and `max_appends` bounds it so that the UNPATCHED walk still finishes +/// and can be measured rather than hanging the suite. The GC fold read-ahead can still land `read` calls +/// for several hinted keys on different worker threads at once, so the hook's own state is mutex-guarded +/// rather than assumed single-threaded. class ChasingWriterBackend : public CountingBackend { public: - using CountingBackend::get; - /// Start appending above `published_through` (writer epoch 1) whenever the tail is read, up to /// `max_appends` further records. void arm(const Layout * layout_, const RootNamespace & ns_, uint64_t published_through, uint64_t max_appends) { + std::lock_guard lock(hook_mutex); layout = layout_; ns = ns_; published = published_through; @@ -151,29 +154,54 @@ class ChasingWriterBackend : public CountingBackend } /// Stop appending; the tail stands still from here on. - void disarm() { layout = nullptr; } + void disarm() + { + std::lock_guard lock(hook_mutex); + layout = nullptr; + } - uint64_t publishedThrough() const { return published; } + uint64_t publishedThrough() const + { + std::lock_guard lock(hook_mutex); + return published; + } - std::optional get(const String & key, DB::Cas::Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - auto result = CountingBackend::get(key, range); - if (!layout || appending || published >= limit) - return result; - if (key != layout->refLogKey(fixture::fixtureLife(ns), RefTxnId{1, published})) - return result; - - /// The walk just consumed the tail; the writer answers with the next record. Guarded against - /// re-entry because publishing issues backend calls of its own. - appending = true; - const uint64_t next = published + 1; - publishAt(*this, *layout, ns, RefTxnId{1, next}, "ref_" + std::to_string(next), next, DB::UInt128(next)); + auto result = CountingBackend::read(key, access); + + const Layout * layout_snapshot = nullptr; + RootNamespace ns_snapshot; + uint64_t next = 0; + { + std::lock_guard lock(hook_mutex); + if (!layout || appending || published >= limit) + return result; + if (key != layout->refLogKey(fixture::fixtureLife(ns), RefTxnId{1, published})) + return result; + + /// The walk just consumed the tail; the writer answers with the next record. Guarded + /// against re-entry (by another read-ahead worker, not just the same thread) because + /// publishing issues backend calls of its own. + appending = true; + layout_snapshot = layout; + ns_snapshot = ns; + next = published + 1; + } + + /// `publishAt` below must run with the mutex released: it issues backend calls of its own, and + /// holding the lock across them would either self-deadlock on a re-entrant call or serialize + /// every read-ahead worker behind this one append. + publishAt(*this, *layout_snapshot, ns_snapshot, RefTxnId{1, next}, "ref_" + std::to_string(next), next, DB::UInt128(next)); + + std::lock_guard lock(hook_mutex); published = next; appending = false; return result; } private: + mutable std::mutex hook_mutex; const Layout * layout = nullptr; RootNamespace ns{}; uint64_t published = 0; @@ -230,7 +258,7 @@ TEST(CASGCBoundedWalk, ARoundFoldsThroughItsRoundStartTailAndLeavesTheStragglers EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, planted})) << "the walk must fold through the round-start tail and no further -- it chased the writer"; EXPECT_FALSE(cov->hold.has_value()) << "reaching the committed frontier is not a hold"; - EXPECT_NE(cov->classification, 4) << "reaching the committed frontier is not a clamp"; + EXPECT_NE(cov->classification, CoverageClass::Clamped) << "reaching the committed frontier is not a clamp"; EXPECT_EQ(metric(intake, "tails_advanced"), 1u); EXPECT_EQ(metric(intake, "logs_applied"), planted) << "exactly the round-start backlog was folded"; @@ -394,7 +422,10 @@ TEST(CASGCBoundedWalk, ARawRecordBeyondTheCommittedFrontierCannotSuppressDestruc EXPECT_EQ(backend->deleteTotal(), 1u) << "the committed frontier permits the round's immediate manifest cleanup. Deleted:" << deletedKeysMessage(*backend); - EXPECT_TRUE(backend->head(layout.blobKey(legacyMetaTestRef(blob))).exists); + { + DB::Cas::tests::OperationForTest head_op(*backend); + EXPECT_TRUE((*head_op).head(layout.blobKey(legacyMetaTestRef(blob)), Retry::standard()).has_value()); + } /// The raw F+1 record remains outside the CTE; it cannot defer the normal destructive pipeline. backend->disarm(); @@ -505,7 +536,7 @@ TEST(CASGCBoundedWalk, AnAbsentManifestBodyStillHoldsWithoutAHead) ASSERT_TRUE(cov->hold.has_value()) << "an absent committed manifest body raises the fold barrier"; EXPECT_EQ(cov->hold->reason, HoldReason::ManifestBodyMissing); EXPECT_EQ(cov->hold->offending_position, (RefTxnId{1, 2})); - EXPECT_EQ(cov->classification, 4); + EXPECT_EQ(cov->classification, CoverageClass::Clamped); EXPECT_EQ(backend->headCount(layout.manifestKey(gone)), 0u) << "absence is decided by the GET, so the missing body costs no HEAD either"; } @@ -558,13 +589,13 @@ TEST(CASGCBoundedWalk, ANamespaceThatFoldedNothingKeepsItsSealedCursor) ASSERT_TRUE(after.has_value()) << "the coverage row was DROPPED -- the next round would re-fold this namespace from {0,0}"; /// The CURSOR and the HOLD are what the next round trusts, and both ride unchanged. - /// `classification` legitimately moves from 2 ("this round folded records") to 1 ("unchanged"), + /// `classification` legitimately moves from `Folded` ("this round folded records") to `Unchanged`, /// because that is what the round did — it is the one field that may differ, so it is the one field /// asserted loosely. EXPECT_EQ(after->last_folded_ref_id, before->last_folded_ref_id) << "a namespace that folded nothing must keep the cursor it had"; EXPECT_EQ(after->hold, before->hold); - EXPECT_NE(after->classification, 4) << "folding nothing is not a clamp"; + EXPECT_NE(after->classification, CoverageClass::Clamped) << "folding nothing is not a clamp"; EXPECT_EQ(metric(intake, "frontier_namespaces"), 2u) << "it stays in the round's universe, so its proof is still owed"; } diff --git a/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp b/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp new file mode 100644 index 000000000000..c490ada2c3ae --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp @@ -0,0 +1,184 @@ +#include + +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +/// `removeChunkWriteOnceOrOneByOne` (CasGc.h) is what both GC bulk-delete call sites (manifest_deletes' +/// flush() and cleanupRefObjects' chunk loop) use to survive a backend without `DeleteObjects`. Tested +/// here in isolation, directly against the engine, rather than only through the much larger machinery of +/// a full GC round. + +namespace DB::ErrorCodes +{ +extern const int CORRUPTED_DATA; +extern const int NETWORK_ERROR; +extern const int NOT_IMPLEMENTED; +} + +using namespace DB::Cas; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +const Layout kLayout{"p"}; +const RootNamespace kNs{"test/aa@cas@"}; + +WriteOnceKey manifestKey(uint32_t ordinal) +{ + return kLayout.writeOnceManifestKey( + ManifestId{kNs, ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = ordinal}}); +} + +std::vector manifestKeys(uint32_t count) +{ + std::vector keys; + for (uint32_t ordinal = 1; ordinal <= count; ++ordinal) + keys.push_back(manifestKey(ordinal)); + return keys; +} + +PoolPtr openPlainPool(const std::shared_ptr & backend) +{ + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + return Pool::open(backend, config); +} + +} + +TEST(CASGCBulkDeleteFallback, HappyPathIsOneRequest) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(3); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + + const uint64_t requests_issued = removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); + + EXPECT_EQ(requests_issued, 1u); + EXPECT_EQ(backend->bulkRemoveCalls(), 1u); + for (const WriteOnceKey & key : keys) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); +} + +TEST(CASGCBulkDeleteFallback, NotImplementedFallsBackToOneRequestPerKeyEachDeleted) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(3); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + + const uint64_t requests_issued = removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); + + EXPECT_EQ(requests_issued, 4u) << "the failed bulk attempt is itself a call, counted alongside the 3 that followed it"; + EXPECT_EQ(backend->bulkRemoveCalls(), 4u) << "1 failed bulk attempt + 3 single-key fallback requests"; + for (const WriteOnceKey & key : keys) + EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); +} + +/// A teardown begun WHILE the fallback is mid-loop stops the remainder at admission, exactly as any +/// other CAS request would be: `removeChunkWriteOnceOrOneByOne`'s per-key loop is not a special path +/// around the engine's own fence, it is ordinary calls through it. +TEST(CASGCBulkDeleteFallback, TeardownBegunBetweenTwoFallbackKeysStopsTheRemainderAtAdmission) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(4); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + + /// The hook does not run on the armed (failing) bulk attempt (it is rethrown before the hook would + /// fire), so this counts only the fallback's own per-key calls that actually reached the backend. + /// Teardown is armed once the SECOND such call has been served, so it is the THIRD key's own + /// admission -- checked at the start of its own `removeManyWriteOnce`, before this hook could run + /// again -- that is refused; the fourth key is never attempted at all. + size_t backend_calls_served = 0; + backend->onBeforeBulkRemove([&] + { + if (++backend_calls_served == 2) + store->beginTeardown(); + }); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::standard()); }); + + /// `store->beginTeardown()` is irreversible here (this test never re-opens the pool), so `op` itself + /// -- the open plane -- refuses every further request, verification reads included. Read through the + /// mount plane instead: a different fence over the SAME backend, unaffected by open-plane teardown + /// (see `CASGCTeardownStop.OpenPlaneRefusesAfterTeardownBeganAndTheMountPlaneDoesNot`). + CasOperation verify = store->mountRequests().admit(); + EXPECT_FALSE(verify.head(keys[0].str(), Retry::once()).has_value()) << "deleted before teardown began"; + EXPECT_FALSE(verify.head(keys[1].str(), Retry::once()).has_value()) << "deleted before teardown began"; + EXPECT_TRUE(verify.head(keys[2].str(), Retry::once()).has_value()) << "refused at admission, never reached the backend"; + EXPECT_TRUE(verify.head(keys[3].str(), Retry::once()).has_value()) << "never attempted"; + EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "1 failed bulk attempt + 2 single-key fallback requests that landed"; +} + +/// A REAL error on one of the fallback's per-key deletes (not "batch not supported", so not caught and +/// retried again) stops the loop exactly where it happened: the keys before it are deleted, the ones +/// from it on are never attempted, and the error itself propagates out of the helper. +TEST(CASGCBulkDeleteFallback, ARealErrorOnAFallbackKeyStopsTheRemainderAndPropagates) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(4); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + + /// The hook does not run on an armed (failing) call, so this fires only on the fallback's own + /// per-key calls that actually reached the backend -- the FIRST of which (key[0]'s own delete) arms + /// a real, non-capability failure for the call right after it, i.e. key[1]'s. + backend->onBeforeBulkRemove([&] + { + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "not a capability problem"))); + }); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); }); + + EXPECT_FALSE(op.head(keys[0].str(), Retry::once()).has_value()) << "deleted before the real error"; + EXPECT_TRUE(op.head(keys[1].str(), Retry::once()).has_value()) << "this delete is the one that failed"; + EXPECT_TRUE(op.head(keys[2].str(), Retry::once()).has_value()) << "never attempted"; + EXPECT_TRUE(op.head(keys[3].str(), Retry::once()).has_value()) << "never attempted"; + EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "1 failed bulk attempt + key[0]'s delete + key[1]'s failed attempt"; +} + +/// A failure outside the "batch delete not supported" class must propagate as-is, with no fallback: +/// the helper does not treat every `removeManyWriteOnce` failure as "try one key at a time". +TEST(CASGCBulkDeleteFallback, OtherFailureClassPropagatesWithNoFallback) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(3); + for (const WriteOnceKey & key : keys) + ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); + + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "not a capability problem"))); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); }); + + EXPECT_EQ(backend->bulkRemoveCalls(), 1u) << "no per-key fallback for a non-capability failure"; + for (const WriteOnceKey & key : keys) + EXPECT_TRUE(op.head(key.str(), Retry::once()).has_value()) << "nothing was deleted"; +} diff --git a/src/Disks/tests/gtest_cas_gc_fold.cpp b/src/Disks/tests/gtest_cas_gc_fold.cpp index c764aa0ac0b7..b8f68f3f2080 100644 --- a/src/Disks/tests/gtest_cas_gc_fold.cpp +++ b/src/Disks/tests/gtest_cas_gc_fold.cpp @@ -21,6 +21,18 @@ ManifestRef ref(const String &, uint64_t seq, uint64_t inst) { return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; } + +std::optional readOf(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +bool headExists(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} } /// Committed new_manifest => +1 per blob entry (BlobInDegreeMatchesActiveManifests). @@ -38,12 +50,12 @@ TEST(CASGCFold, FoldAdoptsAttemptEqualsLeaseSeq) Gc gc(store, kGc); gc.runRegularRound(); - const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); EXPECT_EQ(st.snap_attempt, st.lease.seq); EXPECT_GT(st.snap_generation, 0u); /// The one-pass round's fold seal is durable under (snap_generation, snap_attempt) — the adopted /// attempt locates it (a seal under any other attempt would be unadopted debris). - EXPECT_TRUE(backend->head(store->layout().foldSealKey(st.snap_generation, st.snap_attempt)).exists); + EXPECT_TRUE(headExists(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))); } TEST(CASGCFold, CommittedAddEmitsPlusOnePerBlob) @@ -142,7 +154,7 @@ TEST(CASGCFold, PromoteOfActivatedPrecommitEmitsNoDelta) promoteTransition(*backend, store->layout(), ns, DB::UInt128(7), "tbl", r); gc.runRegularRound(); EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); // unchanged, still pinned - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); // not condemned + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); // not condemned } /// Committed add naming a MISSING body (404) => clamp + anomaly, never a guessed +1, never a throw. @@ -173,7 +185,10 @@ TEST(CASGCFold, RefMismatchFailsClosed) bad.root_namespace_id = ns; bad.entries = {blobEntryFor("a", DB::UInt128(1))}; bad.payload_digest = computePayloadDigest(bad); - backend->putIfAbsent(store->layout().manifestKey(ManifestId{ns, r}), encodePartManifest(bad)); + { + OperationForTest op(*backend); + (*op).create(store->layout().manifestKey(ManifestId{ns, r}), encodePartManifest(bad), Retry::standard()); + } publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); Gc gc(store, kGc); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&]{ gc.runRegularRound(); }); @@ -228,9 +243,9 @@ TEST(CASGCFold, EmptyDeltaShardCarriesParentRunRef) Gc gc(store, kGc); gc.runRegularRound(); // round 1: folds the +1, seals the gen-1 blob_target run - const auto st1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st1 = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); const auto parent_seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); ASSERT_EQ(parent_seal.blob_target_runs.size(), 1u); const RunRef parent_ref = parent_seal.blob_target_runs.front(); @@ -240,16 +255,16 @@ TEST(CASGCFold, EmptyDeltaShardCarriesParentRunRef) EXPECT_EQ(backend->ioCountForKeysContaining("/blob_target/"), 0u) << "idle round must not GET/getStream/PUT any blob_target run object"; - const auto st2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st2 = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); EXPECT_GT(st2.snap_generation, st1.snap_generation); const auto new_seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); ASSERT_EQ(new_seal.blob_target_runs.size(), 1u); const RunRef carried = new_seal.blob_target_runs.front(); EXPECT_EQ(carried.key, parent_ref.key) << "carried ref points at the PARENT generation's run key"; EXPECT_EQ(carried.checksum, parent_ref.checksum); EXPECT_EQ(carried.shard, 0u); - EXPECT_EQ(carried.generation, st1.snap_generation) + EXPECT_EQ(carried.key_generation, st1.snap_generation) << "the carried ref names the generation whose key namespace physically holds the object"; } @@ -306,15 +321,15 @@ TEST(CASGCFold, PreviewResolvesCarriedRef) Gc gc(store, kGc); gc.runRegularRound(); // gen 1: blob referenced, in-degree 1 - const auto st1 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st1 = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); gc.runRegularRound(); // gen 2: no delta, no retired => pure ref-carry (ref points back at gen 1) - const auto st2 = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st2 = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); ASSERT_GT(st2.snap_generation, st1.snap_generation); const auto seal2 = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes); ASSERT_EQ(seal2.blob_target_runs.size(), 1u); - ASSERT_EQ(seal2.blob_target_runs.front().generation, st1.snap_generation) + ASSERT_EQ(seal2.blob_target_runs.front().key_generation, st1.snap_generation) << "the current seal's ref physically lives at the parent generation (carried, not reconstructed)"; // The preview resolves the carried ref (a gen-1 physical key) and computes in-degree 1 => blob 1 is @@ -335,13 +350,14 @@ namespace String corruptSealedRunChecksum(InMemoryBackend & backend, const Layout & layout, const GcState & st) { const String sk = layout.foldSealKey(st.snap_generation, st.snap_attempt); - const auto existing = backend.get(sk); + const auto existing = readOf(backend, sk); auto seal = decodeFoldSeal(existing->bytes); if (seal.blob_target_runs.empty()) return {}; const String run_key = seal.blob_target_runs.front().key; seal.blob_target_runs.front().checksum = seal.blob_target_runs.front().checksum + 1; - backend.putOverwrite(sk, encodeFoldSeal(seal), existing->token); + OperationForTest op(backend); + (*op).replace(sk, encodeFoldSeal(seal), existing->etag, Retry::standard()); return run_key; } } @@ -359,7 +375,7 @@ TEST(CASGCFold, PreviewDeletesSealChecksumMismatchFailsClosed) Gc gc(store, kGc); gc.runRegularRound(); // seals gen-1 with one blob_target run - const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); ASSERT_FALSE(corruptSealedRunChecksum(*backend, store->layout(), st).empty()); // A deletion preview must never be derived from an unverified run: fail closed. @@ -384,7 +400,7 @@ TEST(CASGCFold, FsckSealChecksumMismatchCataloguedAndAuditCompletes) Gc gc(store, kGc); gc.runRegularRound(); - const auto st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st = decodeGcState(readOf(*backend, store->layout().gcStateKey())->bytes); /// A present-but-unreferenced blob (written AFTER the round so GC never touches it) is what makes /// fsck enter its GC-pipeline classification path (guarded by a non-empty unreferenced set), which @@ -439,7 +455,7 @@ TEST(CASGCFold, MidLogClampPreservesEarlierRemovalBodyAndRecovers) const RoundReport clamp_report = gc.runRegularRound(); EXPECT_TRUE(clamp_report.hasAnomaly(ns, /*shard*/0)) << "the missing B body must clamp this log"; EXPECT_LT(foldCursorOf(*backend, store->layout(), ns, 0), log_seq) << "the clamp halts the cursor below the log"; - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, a})).exists) + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, a}))) << "A's body must survive the clamp round: its `-1` was staged, not merged, so no post-CAS delete " "reclaimed it -- otherwise the re-fold would clamp on A's missing body forever"; EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) << "A's `-1` was not adopted (clamp)"; @@ -464,7 +480,7 @@ TEST(CASGCFold, DeadPrecommitWithMissingBodyIsSkippedNotClampedForever) auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); /// The namespace's server-root prefix is "srv"; seed its watermark floor so build_sequence 5 is retired. const RootNamespace ns{"srv/tbl"}; - setWatermarkMinActive(*backend, store->layout(), "srv", /*writer_epoch*/1, /*min_active*/10); + setWatermarkMinActive(*backend, store->layout(), "srv", /*writer_epoch*/1, /*min_active_build_sequence*/10); /// A precommit naming a build (writer_epoch 1, build_sequence 5) whose body is never written. const ManifestRef dead = ManifestRef{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; @@ -518,7 +534,7 @@ TEST(CASGCFold, SingleAnomalySuppressesEveryDestructiveActionInTheRound) EXPECT_EQ(rep.deleted, 0u); EXPECT_EQ(rep.redeleted, 0u); EXPECT_EQ(rep.graduated, 0u); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, a})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, a}))); EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1); } @@ -533,6 +549,8 @@ TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJa { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); Gc gc(store, kGc); @@ -560,9 +578,12 @@ TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJa /// which is the physical life that owns the eventual janitor work. Spelling the sentinel here instead /// would plant debris under the wrong life and make the retention assertion vacuous. const String debris_key - = layout.namespaceFilesPrefix(CasRefCatalog::lifeIfCataloged(*backend, layout, ns_removed).value()) + = layout.namespaceFilesPrefix(CasRefCatalog::lifeIfCataloged(op, layout, ns_removed).value()) + "leftover_verbatim_file"; - backend->putIfAbsent(debris_key, "debris"); + { + OperationForTest debris_op(*backend); + (*debris_op).create(debris_key, "debris", Retry::standard()); + } const ManifestRef removed_body = ref("srv-r:1", 1, 0xEE); writeManifestRaw(*backend, layout, ns_removed, removed_body, {blobEntryFor("r", DB::UInt128(9))}); const String debris_manifest_key = layout.manifestKey(ManifestId{ns_removed, removed_body}); @@ -585,7 +606,7 @@ TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJa .last_epoch_seal = std::nullopt, }); const String covered_log_key = layout.refLogKey(fixture::fixtureLife(ns_covered), RefTxnId{1, cv1}); - ASSERT_TRUE(backend->head(covered_log_key).exists); + ASSERT_TRUE(headExists(*backend, covered_log_key)); /// Trigger the clamp in ns_clamp: drop committed A, add precommit B whose body is absent. writeManifestRaw(*backend, layout, ns_clamp, b, {blobEntryFor("b", DB::UInt128(2))}); @@ -602,13 +623,13 @@ TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJa EXPECT_EQ(rep.graduated, 0u); /// Removal folding never performs lifecycle-specific physical cleanup, with or without a clamp. - EXPECT_TRUE(backend->head(debris_manifest_key).exists) + EXPECT_TRUE(headExists(*backend, debris_manifest_key)) << "removed manifest debris remains ordinary orphan-sweep work"; - EXPECT_TRUE(backend->head(debris_key).exists) + EXPECT_TRUE(headExists(*backend, debris_key)) << "removed verbatim-file debris remains ordinary janitor work"; /// `cleanupRefObjects` must not have deleted anything anywhere this round. - EXPECT_TRUE(backend->head(covered_log_key).exists) + EXPECT_TRUE(headExists(*backend, covered_log_key)) << "a clamp anywhere in the round must suppress ref-log cleanup pool-wide, even for an unrelated live table"; /// Heal the clamp and run a clean round. Ordinary ref-log cleanup resumes, while removal debris @@ -616,9 +637,9 @@ TEST(CASGCFold, RoundSideAnomalySuppressesRefLogCleanupWhileRemovalDebrisStaysJa writeManifestRaw(*backend, layout, ns_clamp, b, {blobEntryFor("b", DB::UInt128(2))}); const RoundReport clean_rep = runRegularRoundReclaiming(gc); EXPECT_FALSE(clean_rep.hasAnomaly(ns_clamp, /*shard*/0)); - EXPECT_TRUE(backend->head(debris_manifest_key).exists) + EXPECT_TRUE(headExists(*backend, debris_manifest_key)) << "a clamp-free fold still performs no lifecycle-specific manifest deletion"; - EXPECT_TRUE(backend->head(debris_key).exists) + EXPECT_TRUE(headExists(*backend, debris_key)) << "a clamp-free fold still performs no lifecycle-specific verbatim-file deletion"; - EXPECT_FALSE(backend->head(covered_log_key).exists) << "a clamp-free round cleans the covered ref-log"; + EXPECT_FALSE(headExists(*backend, covered_log_key)) << "a clamp-free round cleans the covered ref-log"; } diff --git a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp index 13dc97ab826c..d8c90fa7dc1c 100644 --- a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp +++ b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp @@ -12,10 +12,12 @@ #include #include "cas_test_helpers.h" +#include #include #include #include +#include #include #include #include @@ -32,10 +34,9 @@ namespace DB::ErrorCodes /// /// Reachability is a property of the WHOLE POOL. A blob is unreferenced only if no namespace anywhere /// owns an edge to it, so a round that deletes one is asserting something about every namespace at -/// once -- including the ones it never looked at. Task 7 made the per-namespace half of that assertion -/// cheap and exact: one `GET` at the cursor's arithmetic successor, absent means end-of-stream. Task 8 -/// made a namespace that could NOT be walked say so durably. What neither can supply is the SET those -/// proofs have to cover, and that is what this task is about. +/// once -- including the ones it never looked at. A `GET` at the cursor's arithmetic successor makes +/// the per-namespace proof cheap and exact, and a namespace that cannot be walked reports that fact +/// durably. Neither supplies the SET those proofs have to cover. /// /// So the gate has three terms, and a round destroys only when all three are clear: /// @@ -74,6 +75,13 @@ namespace const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +/// The `CASCatalogLifecycleReconciler` suites hand their operation a liveness that reads a local bool +/// directly, and `DrainRaceBackend::afterReadOf` moves that bool at an exact request boundary -- so +/// nothing is cached behind a read and there is nothing for a refresh to re-read. Named rather than an +/// inline no-op because the argument is mandatory, precisely so that erasing without a refresh has to +/// be said out loud. +void noAuthorityRefresh() {} + /// The lying store, shared from `cas_test_helpers.h`: every key is served by exact GET while the /// selected ones are HIDDEN from every LIST. That is the only way to build the cross-namespace /// scenario -- the hidden namespace's records stay durable and readable, so a round that KNOWS to @@ -84,9 +92,8 @@ using CountingHintHoleBackend = DB::Cas::tests::HintHoleBackendOn get(const String & key, Range range) override + /// Runs after every completed read of `key`. It is where a test moves a fact the operation's + /// liveness predicate samples, so admission can be lost at an exact request boundary. + void afterReadOf(const String & key, std::function hook) { - record("get " + key); - return CountingBackend::get(key, range); + std::lock_guard lock(control_mutex); + after_read_key = key; + after_read_hook = std::move(hook); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + std::optional read(const String & key, TransportAccess & access) override { - record("list " + prefix); - return CountingBackend::list(prefix, cursor, limit); + record("get " + key); + std::optional raw = CountingBackend::read(key, access); + std::function hook; + { + std::lock_guard lock(control_mutex); + if (key == after_read_key) + hook = after_read_hook; + } + if (hook) + hook(); + return raw; } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - record("put_begin " + key); - const PutResult result = CountingBackend::putIfAbsent(key, bytes, meta); - record("put_end " + key); - return result; + record("list " + prefix); + return CountingBackend::list(prefix, cursor, limit, access); } - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - record("cas_begin " + key); + /// One primitive now carries both shapes the journal used to name separately: a write with no + /// precondition is the create, a write with one is the conditional replace. + const bool conditional = expected_value.has_value(); + record((conditional ? "cas_begin " : "put_begin ") + key); bool lose_response = false; bool force_conflict = false; { std::unique_lock lock(control_mutex); - if (key == catalog_key && block_next_catalog_cas) + if (conditional && key == catalog_key && block_next_catalog_cas) { block_next_catalog_cas = false; catalog_cas_blocked = true; control_cv.notify_all(); control_cv.wait(lock, [&] { return release_catalog_cas; }); } - if (key == catalog_key && lose_next_catalog_cas_response) + if (conditional && key == catalog_key && lose_next_catalog_cas_response) { lose_next_catalog_cas_response = false; lose_response = true; } - if (key == catalog_key && conflict_next_catalog_cas) + if (conditional && key == catalog_key && conflict_next_catalog_cas) { conflict_next_catalog_cas = false; force_conflict = true; @@ -184,14 +204,18 @@ class DrainRaceBackend final : public CountingBackend if (force_conflict) { record("cas_forced_conflict " + key); - return {.outcome = CasOutcome::Conflict, .token = {}}; + return std::unexpected(RawConflict{}); } - const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); - record("cas_end " + key); - if (lose_response && result.outcome == CasOutcome::Committed) + std::expected result = CountingBackend::write(key, bytes, expected_value, access); + record((conditional ? "cas_end " : "put_end ") + key); + if (lose_response && result.has_value()) { record("cas_response_lost " + key); - throw std::runtime_error("injected lost catalog CAS response"); + /// `Poco::TimeoutException`, because that is the class the write loop cannot distinguish + /// from a lost response: it settles the attempt by an exact read, finds these bytes under a + /// moved incarnation, and reports the write committed. A non-`Poco` exception is rethrown + /// unchanged instead, which would propagate a landed write as a failure. + throw Poco::TimeoutException("injected lost catalog CAS response"); } return result; } @@ -213,25 +237,31 @@ class DrainRaceBackend final : public CountingBackend bool release_catalog_cas = false; bool lose_next_catalog_cas_response = false; bool conflict_next_catalog_cas = false; + String after_read_key; + std::function after_read_hook; }; class PostFoldUnreadableTerminalBackend final : public CountingBackend { public: - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Unhide the names the primitive overrides below would otherwise shadow. + using CountingBackend::head; + using CountingBackend::list; + + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = CountingBackend::list(prefix, cursor, limit); + RawListPage page = CountingBackend::list(prefix, cursor, limit, access); if (prefix.ends_with("/cas/ns/")) - for (ListedKey & listed : page.keys) - listed.token.reset(); + for (RawListedKey & listed : page.keys) + listed.value.reset(); return page; } - HeadResult head(const String & key) override + std::optional head(const String & key, TransportAccess & access) override { - if (key == unreadable_key) + if (!bypass_fault && key == unreadable_key) throw std::runtime_error("injected post-fold terminal read failure for " + key); - return CountingBackend::head(key); + return CountingBackend::head(key, access); } void makeUnreadable(String key) @@ -239,13 +269,23 @@ class PostFoldUnreadableTerminalBackend final : public CountingBackend unreadable_key = std::move(key); } + /// The test's own look at the key the fault hides, taken through the same primitive with the fault + /// suspended -- there is no second door to the store. A fresh open-fence CasRequests over `this` + /// (aliasing, owns nothing): every Backend call needs a CasRequests-minted TransportAccess, and this + /// method has no caller-supplied one to reuse. bool existsIgnoringFault(const String & key) { - return CountingBackend::head(key).exists; + bypass_fault = true; + CasRequests requests(BackendPtr(std::shared_ptr(), this), Fence::open()); + CasOperation op = requests.admit(); + const bool present = op.head(key, Retry::once()).has_value(); + bypass_fault = false; + return present; } private: String unreadable_key; + bool bypass_fault = false; }; class ScopedCasGcLogCapture @@ -290,7 +330,7 @@ struct CompletedRemovingFixture }; CompletedRemovingFixture seedCompletedRemoving( - DrainRaceBackend & backend, const PoolPtr & store, const UInt128 & lease_owner) + CasOperation & op, const PoolPtr & store, const UInt128 & lease_owner) { const Layout & layout = store->layout(); CompletedRemovingFixture fixture{ @@ -298,7 +338,7 @@ CompletedRemovingFixture seedCompletedRemoving( .life_id = UInt128{177}, .checkpoint_key = {}, .checkpoint_bytes = {}}; - CasRefCatalog::casAdmitEntry(backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{ .ns = fixture.ns, .state = NsState::Live, .incarnation = fixture.life_id}); fixture.checkpoint_key = layout.refCkptKey( NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); @@ -308,9 +348,9 @@ CompletedRemovingFixture seedCompletedRemoving( .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, }); - backend.putIfAbsent(fixture.checkpoint_key, fixture.checkpoint_bytes); + op.create(fixture.checkpoint_key, fixture.checkpoint_bytes, Retry::once()); EXPECT_TRUE(store->namespaceFilesLifeIfReadable(fixture.ns)); - CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; next.entries[0].state = NsState::Removing; @@ -321,11 +361,11 @@ CompletedRemovingFixture seedCompletedRemoving( CasFoldSeal parent; parent.generation = 1; parent.ref_lives.emplace(fixture.life_id, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) parent.condemned_summary.emplace(shard, CondemnedSummary{}); - backend.putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)); + op.create(layout.foldSealKey(1, 1), encodeFoldSeal(parent), Retry::once()); GcState state; state.round = 1; @@ -333,13 +373,14 @@ CompletedRemovingFixture seedCompletedRemoving( state.snap_generation = 1; state.snap_attempt = 1; state.lease = GcLease{.owner = lease_owner, .seq = 1}; - backend.putIfAbsent(layout.gcStateKey(), encodeGcState(state)); + op.create(layout.gcStateKey(), encodeGcState(state), Retry::once()); return fixture; } void seedCompletedRemovingBatch( - DrainRaceBackend & backend, const PoolPtr & store, const UInt128 & lease_owner, size_t count) + CasOperation & op, const PoolPtr & store, const UInt128 & lease_owner, + size_t count) { const Layout & layout = store->layout(); std::vector entries; @@ -350,10 +391,10 @@ void seedCompletedRemovingBatch( .ns = RootNamespace{fmt::format("00/drain-batch-{}@cas@", i)}, .state = NsState::Live, .incarnation = UInt128{200 + i}}; - CasRefCatalog::casAdmitEntry(backend, layout, store->poolConfig().gc_shards, entry); + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, entry); entries.push_back(std::move(entry)); } - CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; for (CatalogEntry & entry : next.entries) @@ -368,12 +409,11 @@ void seedCompletedRemovingBatch( parent.generation = 1; for (const CatalogEntry & entry : entries) parent.ref_lives.emplace(entry.incarnation, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) parent.condemned_summary.emplace(shard, CondemnedSummary{}); - ASSERT_EQ(backend.putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)).outcome, - PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.foldSealKey(1, 1), encodeFoldSeal(parent), Retry::once()))); GcState state; state.round = 1; @@ -381,7 +421,7 @@ void seedCompletedRemovingBatch( state.snap_generation = 1; state.snap_attempt = 1; state.lease = GcLease{.owner = lease_owner, .seq = 1}; - ASSERT_EQ(backend.putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.gcStateKey(), encodeGcState(state), Retry::once()))); } enum class CompetingCatalogOutcome : uint8_t @@ -396,13 +436,15 @@ class CASGCCompletedRemovalFenceRace : public testing::TestWithParambytes); state.lease.owner = new_owner; ++state.lease.seq; - ASSERT_EQ(backend.casPut(layout.gcStateKey(), encodeGcState(state), got->token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.gcStateKey(), encodeGcState(state), got->etag, Retry::once()))); } size_t findJournalAfter(const std::vector & journal, const String & entry, size_t after) @@ -612,7 +654,8 @@ TEST(CASGCFrontierGate, AHiddenEdgeIsFoundByTheExactKeyProbeAndSavesTheBlobOnACo } ASSERT_TRUE(verdict.saw_fold) << "no round folded, so none published a gate verdict"; - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the blob a hidden namespace still owns must survive"; EXPECT_TRUE(verdict.frontier_complete) << "the exact-key probe reads at `cursor + 1` and a LIST hole cannot hide an exact key, so the " @@ -661,7 +704,8 @@ TEST(CASGCFrontierGate, TheSameBlobDrainsOnceHiddenGenuinelyProvesItsOwnFrontier drive(store, gc, /*rounds*/ 5, UniversePolicy::Authoritative); - EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "both namespaces genuinely proved their frontier and the blob is genuinely unreferenced -- " "the round must still be able to reclaim it"; } @@ -683,7 +727,8 @@ TEST(CASGCFrontierGate, AKnownNamespaceIsProbedByExactKeyAndItsHiddenEdgeSavesTh Gc gc(store, kGc); drive(store, gc, /*rounds*/ 5, UniversePolicy::Authoritative); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the cursor kept the namespace in the universe, so its frontier was probed and its edge folded"; } @@ -786,7 +831,8 @@ TEST(CASGCFrontierGate, EveryInventoriedDestructiveSiteIsInertUnderSuppression) << "with no universe supplied the frontier can never be complete, whatever the probes proved"; EXPECT_TRUE(verdict.suppress_destructive); expectEveryDeleteFamilyInert(*backend, "no universe supplied"); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()); /// The control: the identical pool DOES reclaim at those sites on the production path, so the zeros /// above are the gate at work and not an empty work queue -- and it is also what makes the "on that @@ -794,7 +840,7 @@ TEST(CASGCFrontierGate, EveryInventoriedDestructiveSiteIsInertUnderSuppression) drive(store, gc, /*rounds*/ 4, UniversePolicy::kDefault); EXPECT_GT(backend->deleteTotal(), 0u) << "the work queue was real -- a round with a universe drains it"; - EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists); + EXPECT_FALSE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()); } /// (1) ONE ANOMALY. A namespace whose `_ckpt` is present but undecodable records the "no usable @@ -805,6 +851,8 @@ TEST(CASGCFrontierGate, AnUndecodableCheckpointAnomalySuppressesEveryDeleteFamil { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); Gc gc(store, kGc); @@ -816,12 +864,12 @@ TEST(CASGCFrontierGate, AnUndecodableCheckpointAnomalySuppressesEveryDeleteFamil /// the very object the round's own life resolution will read, or the round folds normally and this /// test measures nothing. const std::optional damaged_life = - CasRefCatalog::lifeIfCataloged(*backend, layout, damaged); + CasRefCatalog::lifeIfCataloged(op, layout, damaged); ASSERT_TRUE(damaged_life.has_value()) << "the publish must have left a catalog entry to resolve"; - const std::optional damaged_ckpt = readCkpt(*backend, layout, *damaged_life); + const std::optional damaged_ckpt = readCkpt(op, layout, *damaged_life); ASSERT_TRUE(damaged_ckpt.has_value()) << "the publish must have left a `_ckpt` to damage"; - ASSERT_EQ(backend->casPut(layout.refCkptKey(*damaged_life), "not a checkpoint", - damaged_ckpt->token).outcome, CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(op.replace( + layout.refCkptKey(*damaged_life), "not a checkpoint", damaged_ckpt->etag, Retry::once()))); backend->resetCounts(); std::vector anomaly_counts; @@ -835,7 +883,7 @@ TEST(CASGCFrontierGate, AnUndecodableCheckpointAnomalySuppressesEveryDeleteFamil << "the undecodable `_ckpt` must be RECORDED, not silently absorbed -- a silent exit would make " "this test pass for the wrong reason"; expectEveryDeleteFamilyInert(*backend, "one anomaly"); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + EXPECT_TRUE(op.head(blobKeyOf(layout, blob), Retry::once()).has_value()); } /// (2) ONE CARRIED HOLD. The gate's second term reads the SEAL, not this round's anomaly list, so the @@ -882,7 +930,8 @@ TEST(CASGCFrontierGate, ACarriedHoldSuppressesEveryDeleteFamily) store->renewWatermarkOnce(); } expectEveryDeleteFamilyInert(*backend, "one carried hold"); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()); } /// (3c) THE PROBE BUDGET. A namespace with a sealed cursor, no `_ckpt` and no listing left can be proven @@ -891,6 +940,8 @@ TEST(CASGCFrontierGate, ACarriedHoldSuppressesEveryDeleteFamily) TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSuppressesEveryDeleteFamily) { auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolWithProbeBudget(backend, /*budget*/ 0); const Layout & layout = store->layout(); @@ -909,12 +960,12 @@ TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSuppressesEveryDeleteFamily) << "without a sealed cursor the namespace never becomes a budget-spending probe target"; const std::optional quiet_life = - CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); + CasRefCatalog::lifeIfCataloged(op, layout, quiet); ASSERT_TRUE(quiet_life.has_value()); - const std::optional quiet_ckpt = readCkpt(*backend, layout, *quiet_life); + const std::optional quiet_ckpt = readCkpt(op, layout, *quiet_life); ASSERT_TRUE(quiet_ckpt.has_value()) << "there must be a `_ckpt` to remove"; - ASSERT_EQ(backend->deleteExact(layout.refCkptKey(*quiet_life), quiet_ckpt->token).kind, - DeleteOutcome::Kind::Deleted); + ASSERT_EQ(op.remove(layout.refCkptKey(*quiet_life), quiet_ckpt->etag, Retry::once()), + Removal::Removed); backend->hidePrefix(layout.namespaceStreamPrefix(*quiet_life)); backend->resetCounts(); @@ -931,7 +982,7 @@ TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSuppressesEveryDeleteFamily) EXPECT_FALSE(verdict.frontier_complete); EXPECT_TRUE(verdict.suppress_destructive); expectEveryDeleteFamilyInert(*backend, "exhausted probe budget"); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + EXPECT_TRUE(op.head(blobKeyOf(layout, blob), Retry::once()).has_value()); } /// `frontier_proven == frontier_namespaces` is `0 == 0` -- vacuously TRUE -- on an empty universe, which @@ -944,6 +995,8 @@ TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndD { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const DB::UInt128 blob(0xbead); @@ -953,12 +1006,14 @@ TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndD /// round leaves one (`injectRetire`). writeBlobBody(*backend, layout, blob); const BlobRef blob_ref = legacyMetaTestRef(blob); - const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + const std::optional blob_observed = op.head(layout.blobKey(blob_ref), Retry::once()); + ASSERT_TRUE(blob_observed) << "the seeded blob body must be present before it is condemned"; + const PersistedEtag blob_token = PersistedEtag::capture(blob_observed->etag); injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); store->renewWatermarkOnce(); - ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()) + ASSERT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()) << "the scenario needs a genuinely, provably empty catalog, or this test measures nothing"; Gc gc(store, kGc); @@ -978,7 +1033,7 @@ TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndD /// Bound the drive at that plus one (5): enough slack for the fixture's own cadence to be measured /// without hand-counting rounds against this gate, but tight enough that a real regression in the /// confirm/retry cadence still fails loudly instead of silently absorbing into a generous loop. - ASSERT_TRUE(backend->head(layout.blobKey(blob_ref)).exists) + ASSERT_TRUE(op.head(layout.blobKey(blob_ref), Retry::once()).has_value()) << "the scenario starts with the condemned blob present, or the loop below measures nothing"; constexpr int kMaxRounds = 5; /// measured cadence (4) + 1; see the comment above @@ -991,11 +1046,11 @@ TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndD int round_blob_vanished = -1; /// the delete side GateVerdict last; int rounds_run = 0; - for (int i = 0; i < kMaxRounds && backend->head(layout.blobKey(blob_ref)).exists; ++i) + for (int i = 0; i < kMaxRounds && op.head(layout.blobKey(blob_ref), Retry::once()).has_value(); ++i) { last = runRoundCapturingGate(store, gc, UniversePolicy::Authoritative); ++rounds_run; - const bool still_present = backend->head(layout.blobKey(blob_ref)).exists; + const bool still_present = op.head(layout.blobKey(blob_ref), Retry::once()).has_value(); if (round_gate_opened_while_present < 0 && last.saw_fold && !last.suppress_destructive && still_present) round_gate_opened_while_present = rounds_run; if (round_blob_vanished < 0 && !still_present) @@ -1019,7 +1074,7 @@ TEST(CASGCFrontierGate, ADecodedTokenBearingEmptyCatalogCompletesTheFrontierAndD ASSERT_GT(round_blob_vanished, round_gate_opened_while_present) << "the delete must be a round STRICTLY LATER than the one that opened the gate, never the same " "round -- a round that both graduates and deletes in one step would hide the two-phase split"; - EXPECT_FALSE(backend->head(layout.blobKey(blob_ref)).exists) + EXPECT_FALSE(op.head(layout.blobKey(blob_ref), Retry::once()).has_value()) << "a proved-empty universe is a COMPLETE frontier, not a suppressed one -- the condemned blob " "must drain through the ordinary two-phase pipeline instead of leaking forever"; } @@ -1033,12 +1088,16 @@ TEST(CASGCFrontierGate, AZeroWalkableFrontierWithACreatingCatalogRowIsNotProvedE { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const DB::UInt128 blob(0xbead); writeBlobBody(*backend, layout, blob); const BlobRef blob_ref = legacyMetaTestRef(blob); - const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + const std::optional blob_observed = op.head(layout.blobKey(blob_ref), Retry::once()); + ASSERT_TRUE(blob_observed) << "the seeded blob body must be present before it is condemned"; + const PersistedEtag blob_token = PersistedEtag::capture(blob_observed->etag); injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); store->renewWatermarkOnce(); @@ -1050,7 +1109,7 @@ TEST(CASGCFrontierGate, AZeroWalkableFrontierWithACreatingCatalogRowIsNotProvedE entry.incarnation = hexToU128("00000000000000000000000000000042"); entry.creator = CreatorFence{ .server_root_id = "test-stalled-creator", .writer_epoch = 1, .fence_generation = 1}; - CasRefCatalog::casAdmitEntry(*backend, layout, /*gc_shards*/ 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, /*gc_shards*/ 1, entry); Gc gc(store, kGc); backend->resetCounts(); @@ -1071,7 +1130,7 @@ TEST(CASGCFrontierGate, AZeroWalkableFrontierWithACreatingCatalogRowIsNotProvedE EXPECT_TRUE(v.suppress_destructive); } expectEveryDeleteFamilyInert(*backend, "Creating-only catalog"); - EXPECT_TRUE(backend->head(layout.blobKey(blob_ref)).exists); + EXPECT_TRUE(op.head(layout.blobKey(blob_ref), Retry::once()).has_value()); } /// The bootstrap-only absent-as-empty representation (`initializeEmptyForNewPool`) must never leak into @@ -1082,8 +1141,10 @@ TEST(CASGCFrontierGate, AnAbsentCatalogNeverReadsAsAnEmptyUniverse) auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); const Layout & layout = store->layout(); - const Token catalog_token = backend->head(layout.refCatalogKey()).token; - ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog_token).kind, DeleteOutcome::Kind::Deleted); + OperationForTest raw_op(*backend); + const auto catalog_head = (*raw_op).head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(catalog_head.has_value()); + ASSERT_EQ((*raw_op).remove(layout.refCatalogKey(), catalog_head->etag, Retry::once()), Removal::Removed); Gc gc(store, kGc); backend->resetCounts(); @@ -1163,9 +1224,11 @@ TEST(CASGCFrontierGate, AMalformedCatalogNeverDecodesIntoAnEmptyProof) auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); const Layout & layout = store->layout(); - const Token bootstrap_token = backend->head(layout.refCatalogKey()).token; - ASSERT_EQ(backend->casPut(layout.refCatalogKey(), c.bytes, bootstrap_token).outcome, - CasOutcome::Committed) << c.name; + OperationForTest raw_op(*backend); + const auto bootstrap_head = (*raw_op).head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(bootstrap_head.has_value()) << c.name; + ASSERT_TRUE(std::holds_alternative( + (*raw_op).replace(layout.refCatalogKey(), c.bytes, bootstrap_head->etag, Retry::once()))) << c.name; Gc gc(store, kGc); backend->resetCounts(); @@ -1191,7 +1254,11 @@ TEST(CASGCFrontierGate, AProvedEmptyCatalogUnderStageASuppressedStaysSuppressed) writeBlobBody(*backend, layout, blob); const BlobRef blob_ref = legacyMetaTestRef(blob); - const Token blob_token = backend->head(layout.blobKey(blob_ref)).token; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const std::optional blob_observed = op.head(layout.blobKey(blob_ref), Retry::once()); + ASSERT_TRUE(blob_observed) << "the seeded blob body must be present before it is condemned"; + const PersistedEtag blob_token = PersistedEtag::capture(blob_observed->etag); injectRetire(*backend, layout, /*round*/ 1, /*shard*/ 0, {RetiredEntry{.kind = ObjectKind::Blob, .ref = blob_ref, .token = blob_token, .size = 0}}); store->renewWatermarkOnce(); @@ -1214,7 +1281,7 @@ TEST(CASGCFrontierGate, AProvedEmptyCatalogUnderStageASuppressedStaysSuppressed) EXPECT_TRUE(v.suppress_destructive); } expectEveryDeleteFamilyInert(*backend, "StageA_Suppressed over a proved-empty catalog"); - EXPECT_TRUE(backend->head(layout.blobKey(blob_ref)).exists); + EXPECT_TRUE(op.head(layout.blobKey(blob_ref), Retry::once()).has_value()); } /// THE BIRTH-AFTER-EMPTY-CUT BLOB RACE. The proved-empty exception's soundness rests on one hard fact: @@ -1234,6 +1301,8 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace doomed{"00/doomed@cas@"}; @@ -1255,7 +1324,9 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob dropRefTransition(*backend, layout, doomed, "ref_1", mref); runRegularRoundReclaiming(gc); /// condemns: durable Condemned meta store->renewWatermarkOnce(); - const Token condemned_token = backend->head(key).token; + const auto condemned_head = op.head(key, Retry::once()); + ASSERT_TRUE(condemned_head.has_value()); + const Etag condemned_token = condemned_head->etag; const auto condemned_meta = loadMetaForTest(*backend, layout, hash); ASSERT_TRUE(condemned_meta.has_value()); ASSERT_EQ(condemned_meta->meta.state, MetaState::Condemned) @@ -1270,7 +1341,7 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob remove_op.kind = RefOpKind::RemoveNamespace; const uint64_t remove_seq = appendRefLogSeed(*backend, layout, doomed, {remove_op}); publishRecoverableCkptForSemanticWrapper(*backend, layout, doomed, RefTxnId{1, remove_seq}); - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) -> RefCatalog + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) -> RefCatalog { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), @@ -1288,7 +1359,7 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob /// until round R below, rather than letting it drain the ordinary way while `doomed` is still Live. gc.runRegularRound({}, /*allow_steal*/true, UniversePolicy::StageA_Suppressed); store->renewWatermarkOnce(); - EXPECT_TRUE(backend->head(key).exists) << "the pending delete must still be carried, not yet run"; + EXPECT_TRUE(op.head(key, Retry::once()).has_value()) << "the pending delete must still be carried, not yet run"; /// Round R: its pre-fold drain (`drainCompletedRemoving`) reads the round just above's /// `cleanup_evidence` and drops `doomed`'s catalog row BEFORE this round's own hot-scan `GET` -- @@ -1296,11 +1367,11 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob /// instant that cut is taken and races a real namespace birth into the window before round R's own /// pre-CAS delete phase runs. bool hook_fired = false; - Token fresh_token{}; + std::optional fresh_token; gc.setPostHotScanCatalogReadHookForTest([&]() { hook_fired = true; - ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()) + ASSERT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()) << "the race must land inside the window where the cut itself is already empty"; const RootNamespace newborn{"00/newborn@cas@"}; @@ -1310,8 +1381,10 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob build->precommitAdd(newborn, "ref_1", new_id); /// mints `newborn` via real createNamespace const PutBlobResult uploaded = build->putBlob(id, BlobSource::fromString(payload)); EXPECT_EQ(uploaded.ref, id); - fresh_token = backend->head(key).token; - EXPECT_NE(fresh_token, condemned_token) + const auto fresh_head = op.head(key, Retry::once()); + ASSERT_TRUE(fresh_head.has_value()); + fresh_token = fresh_head->etag; + EXPECT_NE(*fresh_token, condemned_token) << "the writer must have observed Condemned and resurrected -- a fresh token, not an adopt " "of the dying incarnation"; }); @@ -1323,10 +1396,12 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob EXPECT_TRUE(verdict.frontier_complete); EXPECT_FALSE(verdict.suppress_destructive); - EXPECT_TRUE(backend->head(key).exists) + const auto surviving_head = op.head(key, Retry::once()); + EXPECT_TRUE(surviving_head.has_value()) << "the resurrected incarnation must survive round R's delete"; - EXPECT_EQ(backend->head(key).token, fresh_token) << "and it is still the writer's incarnation"; - EXPECT_EQ(backend->deleteExact(key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch) + ASSERT_TRUE(fresh_token.has_value()); + EXPECT_EQ(surviving_head->etag, *fresh_token) << "and it is still the writer's incarnation"; + EXPECT_EQ(op.remove(key, condemned_token, Retry::once()), Removal::Mismatch) << "the condemned token can never remove the fresh object (INV_NO_LOSS)"; /// A later round's own fresh catalog cut names `newborn`, folds its `+1`, and the blob's frontier is @@ -1335,7 +1410,7 @@ TEST(CASGCFrontierGate, ANamespaceBornAfterTheEmptyCutResurrectsTheCondemnedBlob ASSERT_TRUE(later.saw_fold); EXPECT_EQ(later.frontier_namespaces, 1u); EXPECT_EQ(later.frontier_proven, 1u); - EXPECT_TRUE(backend->head(key).exists) << "the newly folded owner keeps the blob alive"; + EXPECT_TRUE(op.head(key, Retry::once()).has_value()) << "the newly folded owner keeps the blob alive"; } /// The generation prune's cursor must not move on a suppressed round either. It is a monotone @@ -1355,8 +1430,9 @@ TEST(CASGCFrontierGate, ASuppressedRoundDoesNotAdvanceTheGenerationPruneCursor) runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); } + OperationForTest raw_op(*backend); const uint64_t pruned_through_before = - decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through; + decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes).snap_pruned_through; for (uint64_t i = 7; i <= 10; ++i) { @@ -1365,7 +1441,7 @@ TEST(CASGCFrontierGate, ASuppressedRoundDoesNotAdvanceTheGenerationPruneCursor) store->renewWatermarkOnce(); } - EXPECT_EQ(decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through, + EXPECT_EQ(decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes).snap_pruned_through, pruned_through_before) << "the retention cursor is a high-water mark; it may not pass a generation nothing deleted"; } @@ -1392,9 +1468,10 @@ TEST(CASGCFrontierGate, TheHandOffReclaimIsInertUnderSuppression) Gc gc(store, kGc); runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); - const uint64_t old_gen = decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_generation; + OperationForTest raw_op(*backend); + const uint64_t old_gen = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes).snap_generation; const String old_prefix = layout.gcGenPrefix(old_gen); - ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()); + ASSERT_FALSE((*raw_op).list(old_prefix, "", 1000, Retry::once()).keys.empty()); /// Idle-carry the ref until the retention cursor is strictly PAST its generation. Until then an /// ordinary prune could still reclaim it and the hand-off would not be the load-bearing path. @@ -1403,9 +1480,9 @@ TEST(CASGCFrontierGate, TheHandOffReclaimIsInertUnderSuppression) runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); } - ASSERT_GT(decodeGcState(backend->get(layout.gcStateKey())->bytes).snap_pruned_through, old_gen) + ASSERT_GT(decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes).snap_pruned_through, old_gen) << "the generation must be behind the retention cursor before the hand-off is exercised"; - ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + ASSERT_FALSE((*raw_op).list(old_prefix, "", 1000, Retry::once()).keys.empty()) << "and still retained, because a live ref pins it"; /// A real delta moves the shard's run off the old generation. This is the round the hand-off would @@ -1420,10 +1497,10 @@ TEST(CASGCFrontierGate, TheHandOffReclaimIsInertUnderSuppression) EXPECT_EQ(backend->deleteCountForKeysContaining("/gc/gen/"), 0u) << "a suppressed round hands nothing off. Deleted:" << deletedKeysMessage(*backend); - EXPECT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + EXPECT_FALSE((*raw_op).list(old_prefix, "", 1000, Retry::once()).keys.empty()) << "the superseded generation's prefix survives a suppressed round intact"; - /// AND THE OPPORTUNITY IS CONSUMED, NOT DEFERRED -- the one place in this task where the gate + /// AND THE OPPORTUNITY IS CONSUMED, NOT DEFERRED -- the gate /// costs something permanent, so it is asserted here rather than left to be discovered later. /// /// The hand-off is a one-shot DIFFERENCE: it compares the PARENT seal's runs against the new @@ -1437,7 +1514,7 @@ TEST(CASGCFrontierGate, TheHandOffReclaimIsInertUnderSuppression) /// The hand-off itself is not going untested: `CASGCRetention.HandOffDeletesSupersededRef` drives /// the same transition on an authoritative round and asserts the prefix IS reclaimed. runRegularRoundReclaiming(gc); - EXPECT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + EXPECT_FALSE((*raw_op).list(old_prefix, "", 1000, Retry::once()).keys.empty()) << "the hand-off is a one-shot difference: the suppressed round consumed it, so the prefix is " "now fsck's problem rather than a later round's"; } @@ -1480,7 +1557,7 @@ TEST(CASGCFrontierGate, TheOrphanManifestSweepAndItsCursorAreInertUnderSuppressi const ManifestRef r2{.writer_epoch = 5, .build_sequence = 0xCA02, .manifest_ordinal = 1}; writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(0xa1))}); writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(0xb2))}); - setWatermarkMinActive(*backend, layout, "test", r1.writer_epoch, /*min_active*/ 0xCA03); + setWatermarkMinActive(*backend, layout, "test", r1.writer_epoch, /*min_active_build_sequence*/ 0xCA03); /// The §6 deletion premise is a second precondition on the CONTROL arm below: a manifest of an /// epoch-`E` build is deletable only once the namespace's sealed fold cursor sits in an epoch /// strictly above `E`. Sealing that cursor here is what keeps this test about the GATE — without it @@ -1498,11 +1575,12 @@ TEST(CASGCFrontierGate, TheOrphanManifestSweepAndItsCursorAreInertUnderSuppressi store->renewWatermarkOnce(); } + OperationForTest raw_op(*backend); EXPECT_EQ(backend->deleteCountForKeysContaining("/cas/manifests/"), 0u) << "a suppressed round sweeps nothing. Deleted:" << deletedKeysMessage(*backend); - EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, r1})).exists); - EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, r2})).exists); - EXPECT_TRUE(decodeGcState(backend->get(layout.gcStateKey())->bytes).manifest_sweep_cursor.empty()) + EXPECT_TRUE((*raw_op).head(layout.manifestKey(ManifestId{ns, r1}), Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head(layout.manifestKey(ManifestId{ns, r2}), Retry::once()).has_value()); + EXPECT_TRUE(decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes).manifest_sweep_cursor.empty()) << "the sweep cursor must not advance over a range the round declined to sweep -- nothing " "revisits it"; @@ -1512,8 +1590,8 @@ TEST(CASGCFrontierGate, TheOrphanManifestSweepAndItsCursorAreInertUnderSuppressi runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); } - EXPECT_FALSE(backend->head(layout.manifestKey(ManifestId{ns, r1})).exists); - EXPECT_FALSE(backend->head(layout.manifestKey(ManifestId{ns, r2})).exists); + EXPECT_FALSE((*raw_op).head(layout.manifestKey(ManifestId{ns, r1}), Retry::once()).has_value()); + EXPECT_FALSE((*raw_op).head(layout.manifestKey(ManifestId{ns, r2}), Retry::once()).has_value()); } @@ -1528,6 +1606,8 @@ TEST(CASGCFrontierGate, TheOrphanManifestSweepAndItsCursorAreInertUnderSuppressi TEST(CASGCFrontierGate, APartialProbeBudgetPublishesATallyThatMatchesTheSealedSet) { auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolWithProbeBudget(backend, /*budget*/ 1); const Layout & layout = store->layout(); const RootNamespace a{"00/quiet_a@cas@"}; @@ -1575,8 +1655,8 @@ TEST(CASGCFrontierGate, APartialProbeBudgetPublishesATallyThatMatchesTheSealedSe { EXPECT_NE(sealedCursorOf(*backend, layout, ns), (RefTxnId{})) << "every namespace in the tally must have a sealed cursor: " << ns.string(); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - const auto checkpoint = readCkpt(*backend, layout, life); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); + const auto checkpoint = readCkpt(op, layout, life); ASSERT_TRUE(checkpoint.has_value()); EXPECT_EQ(checkpoint->ckpt.committed_through, (RefTxnId{1, 1})) << "LIST omission and the probe budget do not alter a valid CTE"; @@ -1635,6 +1715,8 @@ TEST(CASGCFrontierGate, CheckpointFrontierBehindAnInheritedCursorFailsClosed) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/checkpoint-behind-inherited-cursor@cas@"}; @@ -1652,16 +1734,16 @@ TEST(CASGCFrontierGate, CheckpointFrontierBehindAnInheritedCursorFailsClosed) ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); ASSERT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 2})); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const String checkpoint_key = layout.refCkptKey(life); - const HeadResult checkpoint_head = backend->head(checkpoint_key); - ASSERT_TRUE(checkpoint_head.exists); - ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + const auto checkpoint_head = op.head(checkpoint_key, Retry::once()); + ASSERT_TRUE(checkpoint_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(checkpoint_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - }), checkpoint_head.token).outcome, PutOutcome::Done); + }), checkpoint_head->etag, Retry::once()))); std::map intake; gc.setPhaseSink([&](const GcPhaseRecord & rec) @@ -1684,6 +1766,8 @@ TEST(CASGCFrontierGate, CheckpointFrontierCrossesAnInheritedEpochSeal) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/checkpoint-inherited-seal-crossing@cas@"}; const DB::UInt128 crossed_blob(0xfd); @@ -1704,16 +1788,16 @@ TEST(CASGCFrontierGate, CheckpointFrontierCrossesAnInheritedEpochSeal) publishAt(*backend, layout, ns, RefTxnId{2, 1}, "crossed", 2, crossed_blob, /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const String checkpoint_key = layout.refCkptKey(life); - const HeadResult checkpoint_head = backend->head(checkpoint_key); - ASSERT_TRUE(checkpoint_head.exists); - ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + const auto checkpoint_head = op.head(checkpoint_key, Retry::once()); + ASSERT_TRUE(checkpoint_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(checkpoint_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{2, 1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{1, 2}, - }), checkpoint_head.token).outcome, PutOutcome::Done); + }), checkpoint_head->etag, Retry::once()))); std::map intake; gc.setPhaseSink([&](const GcPhaseRecord & rec) @@ -1773,6 +1857,8 @@ TEST(CASGCFrontierGate, AWronglyQuietNamespaceIsWalkedTheSameRound) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace quiet{"00/quiet@cas@"}; const DB::UInt128 late_blob(0x77); @@ -1792,16 +1878,16 @@ TEST(CASGCFrontierGate, AWronglyQuietNamespaceIsWalkedTheSameRound) /// A second publish lands, and the store hides the namespace from every LIST at the same moment. publish(*backend, layout, quiet, "ref_2", 2, late_blob); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, quiet); const String checkpoint_key = layout.refCkptKey(life); - const HeadResult checkpoint_head = backend->head(checkpoint_key); - ASSERT_TRUE(checkpoint_head.exists); - ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + const auto checkpoint_head = op.head(checkpoint_key, Retry::once()); + ASSERT_TRUE(checkpoint_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(checkpoint_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - }), checkpoint_head.token).outcome, PutOutcome::Done); + }), checkpoint_head->etag, Retry::once()))); backend->hidePrefix(layout.namespaceStreamPrefix(fixture::fixtureLife(quiet))); runRegularRoundReclaiming(gc); @@ -1820,6 +1906,8 @@ TEST(CASGCFrontierGate, CheckpointFrontierBoundsOrdinaryFoldBeforeDurableSuccess { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/checkpoint-bounds-fold@cas@"}; const DB::UInt128 committed_blob(0xf1); @@ -1855,7 +1943,7 @@ TEST(CASGCFrontierGate, CheckpointFrontierBoundsOrdinaryFoldBeforeDurableSuccess ASSERT_TRUE(report.acquired_lease); gc.setPhaseSink({}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); EXPECT_EQ(inDegreeOf(*backend, layout, beyond_frontier_blob), 0) << "a durable log above `_ckpt.committed_through` is not foldable history"; @@ -1876,6 +1964,8 @@ TEST(CASGCFrontierGate, ConsumedCheckpointFrontierProvesOrdinaryLifeWithoutSucce { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/checkpoint-complete-fold@cas@"}; @@ -1900,7 +1990,7 @@ TEST(CASGCFrontierGate, ConsumedCheckpointFrontierProvesOrdinaryLifeWithoutSucce ASSERT_TRUE(report.acquired_lease); gc.setPhaseSink({}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{1, 2})), 0u) << "the checkpoint boundary proves the cut without a post-frontier 404"; @@ -1916,6 +2006,8 @@ TEST(CASGCFrontierGate, CheckpointFrontierProvesLifeWithHiddenDurableSuccessor) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/checkpoint-hidden-successor@cas@"}; const DB::UInt128 beyond_frontier_blob(0xf4); @@ -1938,7 +2030,7 @@ TEST(CASGCFrontierGate, CheckpointFrontierProvesLifeWithHiddenDurableSuccessor) .last_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); backend->hide(layout.refLogKey(life, RefTxnId{1, 2})); std::map intake; @@ -1969,6 +2061,8 @@ TEST(CASGCFrontierGate, MissingCommittedCheckpointLogHoldsInsteadOfProvingTheFro { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/missing-committed-checkpoint-log@cas@"}; @@ -1982,11 +2076,11 @@ TEST(CASGCFrontierGate, MissingCommittedCheckpointLogHoldsInsteadOfProvingTheFro .last_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const String missing_key = layout.refLogKey(life, RefTxnId{1, 2}); - const HeadResult missing_head = backend->head(missing_key); - ASSERT_TRUE(missing_head.exists); - ASSERT_EQ(backend->deleteExact(missing_key, missing_head.token).kind, DeleteOutcome::Kind::Deleted); + const auto missing_head = op.head(missing_key, Retry::once()); + ASSERT_TRUE(missing_head.has_value()); + ASSERT_EQ(op.remove(missing_key, missing_head->etag, Retry::once()), Removal::Removed); std::map intake; Gc gc(store, kGc); @@ -2011,6 +2105,8 @@ TEST(CASGCFrontierGate, HiddenCommittedCheckpointLogIsFoldedThroughTheAuthorityC { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/hidden-committed-checkpoint-log@cas@"}; const DB::UInt128 hidden_blob(0xf8); @@ -2025,7 +2121,7 @@ TEST(CASGCFrontierGate, HiddenCommittedCheckpointLogIsFoldedThroughTheAuthorityC .last_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); backend->hide(layout.refLogKey(life, RefTxnId{1, 2})); std::map intake; @@ -2080,6 +2176,8 @@ TEST(CASGCFrontierGate, EmptyCheckpointFrontierRejectsAnInheritedCursor) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/empty-checkpoint-after-cursor@cas@"}; @@ -2096,16 +2194,16 @@ TEST(CASGCFrontierGate, EmptyCheckpointFrontierRejectsAnInheritedCursor) ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); ASSERT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 1})); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const String checkpoint_key = layout.refCkptKey(life); - const HeadResult checkpoint_head = backend->head(checkpoint_key); - ASSERT_TRUE(checkpoint_head.exists); - ASSERT_EQ(backend->putOverwrite(checkpoint_key, encodeRefCkpt(RefCkpt{ + const auto checkpoint_head = op.head(checkpoint_key, Retry::once()); + ASSERT_TRUE(checkpoint_head.has_value()); + ASSERT_TRUE(std::holds_alternative(op.replace(checkpoint_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - }), checkpoint_head.token).outcome, PutOutcome::Done); + }), checkpoint_head->etag, Retry::once()))); std::map intake; gc.setPhaseSink([&](const GcPhaseRecord & rec) @@ -2128,6 +2226,8 @@ TEST(CASGCFrontierGate, CatalogLifeWithoutCheckpointDefersWithoutUsingListedFron { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/missing-checkpoint-fold@cas@"}; const DB::UInt128 blob(0xc7); @@ -2138,8 +2238,8 @@ TEST(CASGCFrontierGate, CatalogLifeWithoutCheckpointDefersWithoutUsingListedFron writeManifestRaw(*backend, layout, ns, manifest, {blobEntryFor("data.bin", blob)}); appendRefLogSeed(*backend, layout, ns, publishCommittedOps("must_remain_unfolded", manifest)); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_FALSE(readCkpt(*backend, layout, life).has_value()); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); + ASSERT_FALSE(readCkpt(op, layout, life).has_value()); std::map intake; Gc gc(store, kGc); @@ -2170,6 +2270,8 @@ TEST(CASGCFrontierGate, CatalogLifeWithoutCheckpointDefersWithoutUsingListedFron TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSealsCursorsAndDeletesNothing) { auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolWithProbeBudget(backend, /*budget*/ 0); const Layout & layout = store->layout(); const RootNamespace quiet{"00/quiet@cas@"}; @@ -2195,16 +2297,16 @@ TEST(CASGCFrontierGate, AnExhaustedProbeBudgetSealsCursorsAndDeletesNothing) EXPECT_GT(backend->deleteTotal(), 0u) << "the quiet life's checkpoint authority leaves unrelated deletion eligible"; - EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + EXPECT_FALSE(op.head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the busy life's removal remains reclaimable despite the quiet LIST omission"; EXPECT_EQ(sealedCursorOf(*backend, layout, quiet), quiet_cursor) << "the unprobed namespace's cursor rides verbatim -- it is never dropped"; - const NamespaceLifeId quiet_life = *CasRefCatalog::lifeIfCataloged(*backend, layout, quiet); - const auto quiet_checkpoint = readCkpt(*backend, layout, quiet_life); + const NamespaceLifeId quiet_life = *CasRefCatalog::lifeIfCataloged(op, layout, quiet); + const auto quiet_checkpoint = readCkpt(op, layout, quiet_life); ASSERT_TRUE(quiet_checkpoint.has_value()); EXPECT_EQ(quiet_checkpoint->ckpt.committed_through, quiet_cursor) << "the quiet life's valid CTE is unaffected by LIST omission and a zero probe budget"; - EXPECT_GT(decodeGcState(backend->get(layout.gcStateKey())->bytes).round, 1u) + EXPECT_GT(decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes).round, 1u) << "the round still commits; only its destructive half is withheld"; } @@ -2217,6 +2319,8 @@ TEST(CASGCFrontierGate, ACommittedGapIsRedetectedAndSuppressesEveryRound) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace held{"00/held@cas@"}; const RootNamespace busy{"00/busy@cas@"}; @@ -2257,8 +2361,8 @@ TEST(CASGCFrontierGate, ACommittedGapIsRedetectedAndSuppressesEveryRound) EXPECT_GT(first_intake["tables_held"], 0u); EXPECT_FALSE(first_round.anomalies.empty()); - const NamespaceLifeId held_life = *CasRefCatalog::lifeIfCataloged(*backend, layout, held); - const auto held_checkpoint = readCkpt(*backend, layout, held_life); + const NamespaceLifeId held_life = *CasRefCatalog::lifeIfCataloged(op, layout, held); + const auto held_checkpoint = readCkpt(op, layout, held_life); ASSERT_TRUE(held_checkpoint.has_value()); EXPECT_EQ(held_checkpoint->ckpt.committed_through, (RefTxnId{1, 4})); @@ -2293,10 +2397,10 @@ TEST(CASGCFrontierGate, ACommittedGapIsRedetectedAndSuppressesEveryRound) EXPECT_EQ(backend->deleteTotal(), 0u) << "the re-detected committed gap suppresses each round's destructive work. " "Deleted:" << deletedKeysMessage(*backend); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists); + EXPECT_TRUE(op.head(blobKeyOf(layout, blob), Retry::once()).has_value()); EXPECT_EQ(sealedCursorOf(*backend, layout, held), (RefTxnId{1, 2})) << "the committed gap remains unresolved and the cursor cannot advance through it"; - const auto final_checkpoint = readCkpt(*backend, layout, held_life); + const auto final_checkpoint = readCkpt(op, layout, held_life); ASSERT_TRUE(final_checkpoint.has_value()); EXPECT_EQ(final_checkpoint->ckpt.committed_through, (RefTxnId{1, 4})); } @@ -2326,7 +2430,8 @@ TEST(CASGCFrontierGate, ABlobCondemnedThisRoundIsNeverDeletedThisRound) backend->resetCounts(); runRegularRoundReclaiming(gc); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the condemning round must not also delete"; EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, blob)), 0u) << "not merely still present -- the delete was never attempted"; @@ -2366,7 +2471,8 @@ TEST(CASGCFrontierGate, ALateEdgeSparesADeletePendingBlobAtTheDeleteSite) runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the delete-site in-degree re-read spares a blob a fresh edge re-referenced"; EXPECT_GT(inDegreeOf(*backend, layout, blob), 0); } @@ -2408,7 +2514,10 @@ TEST(CASGCFrontierGate, AResurrectedIncarnationSurvivesTheDelayedStaleTokenDelet runRegularRoundReclaiming(gc); /// graduate: publishes delete_pending against THIS token store->renewWatermarkOnce(); - const Token condemned_token = backend->head(key).token; + OperationForTest raw_op(*backend); + const auto condemned_head = (*raw_op).head(key, Retry::once()); + ASSERT_TRUE(condemned_head.has_value()); + const Etag condemned_token = condemned_head->etag; const auto condemned_meta = loadMetaForTest(*backend, layout, hash); ASSERT_TRUE(condemned_meta.has_value()); ASSERT_EQ(condemned_meta->meta.state, MetaState::Condemned) @@ -2426,16 +2535,19 @@ TEST(CASGCFrontierGate, AResurrectedIncarnationSurvivesTheDelayedStaleTokenDelet const PutBlobResult uploaded = build->putBlob(id, BlobSource::fromString(payload)); EXPECT_EQ(uploaded.ref, id); build->promote(ns, "republished", build->buildId(), republished_manifest); - const Token fresh_token = backend->head(key).token; + const auto fresh_head = (*raw_op).head(key, Retry::once()); + ASSERT_TRUE(fresh_head.has_value()); + const Etag fresh_token = fresh_head->etag; ASSERT_NE(fresh_token, condemned_token) << "republication must displace the condemned incarnation"; /// GC's delayed delete still names the OLD token. It cannot touch the new object. drive(store, gc, /*rounds*/ 2, UniversePolicy::Authoritative); - ASSERT_TRUE(backend->head(key).exists) + const auto surviving_head = (*raw_op).head(key, Retry::once()); + ASSERT_TRUE(surviving_head.has_value()) << "the resurrected incarnation survives the delete published against its predecessor"; - EXPECT_EQ(backend->head(key).token, fresh_token) << "and it is still the writer's incarnation"; - EXPECT_EQ(backend->deleteExact(key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch) + EXPECT_EQ(surviving_head->etag, fresh_token) << "and it is still the writer's incarnation"; + EXPECT_EQ((*raw_op).remove(key, condemned_token, Retry::once()), Removal::Mismatch) << "the condemned token can never remove the fresh object (INV-NO-RETURN)"; } @@ -2474,7 +2586,8 @@ TEST(CASGCFrontierGate, ATokenlessRelinkMakesTheReceiverEdgeDurableBeforeTheSour dropRefTransition(*backend, layout, source, "part_1", source_ref); drive(store, gc, /*rounds*/ 4, UniversePolicy::Authoritative); - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(blobKeyOf(layout, blob), Retry::once()).has_value()) << "the source released its edge only after the receiver's was durable, so nothing may collect it"; EXPECT_EQ(inDegreeOf(*backend, layout, blob), 1) << "the receiver is the sole remaining owner"; @@ -2537,6 +2650,8 @@ TEST(CASGCFrontierGateCleanupRange, ASnapshotAtTheCheckpointSurvivesAndOnlyStric TEST(CASGCFrontierGateCleanupRange, CheckpointBaseValidatorRejectsMissingLogSnapshotAndSeal) { auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RefTxnId base{1, 1}; const RefCkpt checkpoint{ @@ -2544,36 +2659,38 @@ TEST(CASGCFrontierGateCleanupRange, CheckpointBaseValidatorRejectsMissingLogSnap .committed_through = base, .checkpoint_snapshot_id = base, .last_epoch_seal = std::nullopt}; - CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); { const RootNamespace ns{"00/cleanup-missing-base-log@cas@"}; fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, ns).value(); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base)); - EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + EXPECT_THROW((void)readCheckpointSnapshotBase(op, layout, life, checkpoint), DB::Exception); } { const RootNamespace ns{"00/cleanup-missing-base-snapshot@cas@"}; fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), .txn_id = base, .ops = {namespaceBirthOp()}, .prev_epoch_seal = std::nullopt}); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); - EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, ns).value(); + EXPECT_THROW((void)readCheckpointSnapshotBase(op, layout, life, checkpoint), DB::Exception); } { const RootNamespace ns{"00/cleanup-seal-is-not-base@cas@"}; writeSealAt(*backend, layout, ns, base); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, ns).value(); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base)); - EXPECT_THROW((void)readCheckpointSnapshotBase(*backend, layout, life, checkpoint), DB::Exception); + EXPECT_THROW((void)readCheckpointSnapshotBase(op, layout, life, checkpoint), DB::Exception); } } TEST(CASGCFrontierGateCleanupRange, LaterEpochBaseWithoutItsContextualBacklinkCannotLicenseDeletion) { auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; - CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); const RefTxnId seal_id{1, 2}; const RefTxnId base_id{2, 1}; @@ -2586,12 +2703,12 @@ TEST(CASGCFrontierGateCleanupRange, LaterEpochBaseWithoutItsContextualBacklinkCa fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), .txn_id = base_id, .ops = {}, .prev_epoch_seal = backlink}); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, ns).value(); std::optional validated_base; try { - (void)readCheckpointSnapshotBase(*backend, layout, life, RefCkpt{ + (void)readCheckpointSnapshotBase(op, layout, life, RefCkpt{ .life_epoch = 1, .committed_through = base_id, .checkpoint_snapshot_id = base_id, @@ -2620,6 +2737,8 @@ TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanito { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace removed{"00/removed@cas@"}; const RefOp birth_op = namespaceBirthOp(); @@ -2629,8 +2748,8 @@ TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanito .ns = removed.string(), .txn_id = RefTxnId{1, 1}, .ops = {birth_op}, .prev_epoch_seal = std::nullopt}); fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = removed.string(), .txn_id = RefTxnId{1, 2}, .ops = {remove_op}, .prev_epoch_seal = std::nullopt}); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, removed).value(); - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, removed).value(); + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) @@ -2643,31 +2762,31 @@ TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanito return next; }); const String ckpt_key = layout.refCkptKey(life); - backend->putIfAbsent(ckpt_key, encodeRefCkpt(RefCkpt{ + op.create(ckpt_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - })); + }), Retry::once()); /// The removal evidence must arise from a replay-valid terminal lifecycle, rather than merely /// from a raw terminal record that the recovery state machine refuses. const RecoveredRefTable recovered = recoverRefTableDetailedAtCatalogCutForTest( - *backend, layout, CasRefCatalog::read(*backend, layout), removed); + *backend, layout, CasRefCatalog::read(op, layout), removed); EXPECT_EQ(recovered.state.getLifecycle(), RefLifecycle::Removed); EXPECT_EQ(recovered.state.getRemoveTxnId(), (RefTxnId{1, 2})); Gc gc(store, kGc); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState st = decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + op.read(layout.foldSealKey(st.snap_generation, st.snap_attempt), Retry::once())->bytes); const auto row_it = seal.ref_lives.find(life.incarnation); ASSERT_NE(row_it, seal.ref_lives.end()); ASSERT_TRUE(row_it->second.cleanup_evidence.has_value()); EXPECT_EQ(row_it->second.cleanup_evidence->remove_txn_id, (RefTxnId{1, 2})); - EXPECT_TRUE(backend->head(ckpt_key).exists); + EXPECT_TRUE(op.head(ckpt_key, Retry::once()).has_value()); for (const String & key : backend->touchedKeys()) EXPECT_EQ(key.find("/_cleanup/"), String::npos) << key; @@ -2690,13 +2809,13 @@ TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanito ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); gc.setPhaseSink({}); - EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)); + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, removed)); ASSERT_FALSE(janitor_metrics.empty()) << "the namespace_cleanup phase must have run this round"; EXPECT_GE(janitor_metrics.at("janitor_deleted"), 1u) << "the janitor's OWN counter must show the delete -- now that the proved-empty gate has " "opened, not because some other site happened to remove the key"; - EXPECT_FALSE(backend->head(ckpt_key).exists); + EXPECT_FALSE(op.head(ckpt_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(ckpt_key), 1); } @@ -2707,6 +2826,8 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace removed{"00/post-fold-unreadable@cas@"}; const RootNamespace progressing{"00/post-fold-progress@cas@"}; @@ -2719,8 +2840,8 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = removed.string(), .txn_id = RefTxnId{1, 2}, .ops = {remove_op}, .prev_epoch_seal = std::nullopt}); - const NamespaceLifeId removed_life = CasRefCatalog::lifeIfCataloged(*backend, layout, removed).value(); - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + const NamespaceLifeId removed_life = CasRefCatalog::lifeIfCataloged(op, layout, removed).value(); + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) @@ -2733,12 +2854,12 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro it->removal_started_round = 1; return next; }); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(removed_life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(removed_life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - })).outcome, PutOutcome::Done); + }), Retry::once()))); const DB::UInt128 blob(0xfeed); const ManifestRef manifest = publish(*backend, layout, progressing, "victim", 1, blob); @@ -2746,9 +2867,9 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro Gc gc(store, kGc); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - const GcState folded_state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState folded_state = decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes); const CasFoldSeal folded_seal = decodeFoldSeal( - backend->get(layout.foldSealKey(folded_state.snap_generation, folded_state.snap_attempt))->bytes); + op.read(layout.foldSealKey(folded_state.snap_generation, folded_state.snap_attempt), Retry::once())->bytes); const auto folded_row = folded_seal.ref_lives.find(removed_life.incarnation); ASSERT_NE(folded_row, folded_seal.ref_lives.end()); ASSERT_TRUE(folded_row->second.cleanup_evidence.has_value()); @@ -2756,8 +2877,8 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro dropRefTransition(*backend, layout, progressing, "victim", manifest); const String terminal_key = layout.refLogKey(removed_life, RefTxnId{1, 2}); const String later_dead_residue = layout.refLogKey(removed_life, RefTxnId{1, 3}); - ASSERT_EQ(backend->putIfAbsent(later_dead_residue, "dead residue after the folded terminal").outcome, - PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + op.create(later_dead_residue, "dead residue after the folded terminal", Retry::once()))); backend->makeUnreadable(terminal_key); std::map namespace_cleanup; @@ -2773,12 +2894,12 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro gc.setPhaseSink({}); ASSERT_TRUE(report.acquired_lease); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)) + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, removed)) << "post-fold physical cleanup cannot gate catalog removal"; - EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, progressing)); + EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(op, layout, progressing)); EXPECT_EQ(report.manifests_deleted, 1u) << "the janitor leak cannot promote itself into pool-wide destructive suppression"; - EXPECT_FALSE(backend->head(layout.manifestKey(manifest_id)).exists); + EXPECT_FALSE(op.head(layout.manifestKey(manifest_id), Retry::once()).has_value()); EXPECT_TRUE(backend->existsIgnoringFault(terminal_key)); EXPECT_FALSE(backend->existsIgnoringFault(later_dead_residue)) << "one unreadable key cannot stop the perpetual janitor from deciding the rest of its page"; @@ -2803,21 +2924,21 @@ TEST(CASGCFrontierGate, UnmatchedAdoptedParentLifeDoesNotSuppressAuthoritativeDe const ManifestId manifest_id{ns, mref}; Gc gc(store, kGc); + OperationForTest raw_op(*backend); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - ASSERT_TRUE(backend->head(layout.manifestKey(manifest_id)).exists); + ASSERT_TRUE((*raw_op).head(layout.manifestKey(manifest_id), Retry::once()).has_value()); - const GcState before = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState before = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); const String parent_seal_key = layout.foldSealKey(before.snap_generation, before.snap_attempt); - const auto parent_object = backend->get(parent_seal_key); + const auto parent_object = (*raw_op).read(parent_seal_key, Retry::once()); ASSERT_TRUE(parent_object); CasFoldSeal parent = decodeFoldSeal(parent_object->bytes, before.snap_generation); const UInt128 unmatched_life = hexToU128("fedcba98765432100123456789abcdef"); ASSERT_FALSE(parent.ref_lives.contains(unmatched_life)); parent.ref_lives.emplace(unmatched_life, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{9, 9}}}); - ASSERT_EQ( - backend->putOverwrite(parent_seal_key, encodeFoldSeal(parent), parent_object->token).outcome, - PutOutcome::Done); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{9, 9}}}); + ASSERT_TRUE(std::holds_alternative( + (*raw_op).replace(parent_seal_key, encodeFoldSeal(parent), parent_object->etag, Retry::once()))); dropRefTransition(*backend, layout, ns, "victim", mref); const uint64_t events_before = @@ -2830,12 +2951,12 @@ TEST(CASGCFrontierGate, UnmatchedAdoptedParentLifeDoesNotSuppressAuthoritativeDe 1u); EXPECT_EQ(report.manifests_deleted, 1u) << "an unmatched adopted-parent row is observed and dropped, not promoted to pool-wide suppression"; - EXPECT_FALSE(backend->head(layout.manifestKey(manifest_id)).exists) + EXPECT_FALSE((*raw_op).head(layout.manifestKey(manifest_id), Retry::once()).has_value()) << "the valid manifest candidate must be physically deleted by the same authoritative round"; - const GcState after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState after = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); const CasFoldSeal successor = decodeFoldSeal( - backend->get(layout.foldSealKey(after.snap_generation, after.snap_attempt))->bytes, + (*raw_op).read(layout.foldSealKey(after.snap_generation, after.snap_attempt), Retry::once())->bytes, after.snap_generation); EXPECT_FALSE(successor.ref_lives.contains(unmatched_life)); } @@ -2844,20 +2965,14 @@ TEST(CASCatalogLifecycleReconciler, EmptyCatalogReturnsAuthoritativeCompleteCut) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - ASSERT_TRUE(CasRefCatalog::initializeEmptyForNewPool(*backend, layout).catalog.entries.empty()); + ASSERT_TRUE(CasRefCatalog::initializeEmptyForNewPool(op, layout).catalog.entries.empty()); CasFoldSeal parent; - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [](uint64_t) - { - return CasRefCatalog::LeaderFenceStatus::Held; - }); - const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + CatalogLifecycleReconciler reconciler(op, layout, parent); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(noAuthorityRefresh); EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); @@ -2871,25 +2986,19 @@ TEST(CASCatalogLifecycleReconciler, DeletesEligibleRowsFromReturnedResolutionCut { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); constexpr size_t deletes = 3; - seedCompletedRemovingBatch(*backend, store, kGc, deletes); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + seedCompletedRemovingBatch(op, store, kGc, deletes); + const auto parent_object = op.read(layout.foldSealKey(1, 1), Retry::once()); ASSERT_TRUE(parent_object); const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); backend->clearJournal(); backend->resetCounts(); - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [](uint64_t) - { - return CasRefCatalog::LeaderFenceStatus::Held; - }); - const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + CatalogLifecycleReconciler reconciler(op, layout, parent); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(noAuthorityRefresh); EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); @@ -2906,33 +3015,34 @@ TEST(CASCatalogLifecycleReconciler, ReturnsRetiredLifeWhenAuthorityMovesAfterRes { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); + const auto parent_object = op.read(layout.foldSealKey(1, 1), Retry::once()); ASSERT_TRUE(parent_object); const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); - size_t fence_checks = 0; - - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [&fence_checks](uint64_t) - { - ++fence_checks; - return fence_checks == 2 - ? CasRefCatalog::LeaderFenceStatus::Moved - : CasRefCatalog::LeaderFenceStatus::Held; - }); - const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + + /// Admission is lost after the erase has been resolved: the second catalog read of the drain is + /// the resolution cut, so the row is already gone when the loop's next verdict finds no admission. + size_t catalog_reads = 0; + bool authority_held = true; + CasOperation fenced_op = requests.admit([&] { return authority_held; }); + backend->afterReadOf(layout.refCatalogKey(), [&] + { + if (++catalog_reads == 2) + authority_held = false; + }); + + CatalogLifecycleReconciler reconciler(fenced_op, layout, parent); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(noAuthorityRefresh); EXPECT_EQ(result.authority_status, AuthorityStatus::FencedOut); EXPECT_EQ(result.catalog_resolution, CatalogResolution::ExactRowAbsent); ASSERT_EQ(result.retired_lives.size(), 1); EXPECT_EQ(result.retired_lives.front(), NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); - EXPECT_EQ(result.deleted, 0); + EXPECT_EQ(result.deleted, 1); EXPECT_FALSE(result.final_catalog_cut); } @@ -2940,33 +3050,33 @@ TEST(CASCatalogLifecycleReconciler, InitialFenceLossReportsEligibleRowStillPrese { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); + const auto parent_object = op.read(layout.foldSealKey(1, 1), Retry::once()); ASSERT_TRUE(parent_object); const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); backend->resetCounts(); - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [](uint64_t) - { - return CasRefCatalog::LeaderFenceStatus::Moved; - }); - const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + /// Admission is lost the moment the selection cut has been read, so the erase is never sent. + bool authority_held = true; + CasOperation fenced_op = requests.admit([&] { return authority_held; }); + backend->afterReadOf(layout.refCatalogKey(), [&] { authority_held = false; }); + + CatalogLifecycleReconciler reconciler(fenced_op, layout, parent); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(noAuthorityRefresh); EXPECT_EQ(result.authority_status, AuthorityStatus::FencedOut); EXPECT_EQ(result.catalog_resolution, CatalogResolution::ExactRowStillPresent); EXPECT_TRUE(result.retired_lives.empty()); EXPECT_EQ(result.deleted, 0); EXPECT_FALSE(result.final_catalog_cut); - EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 2) - << "the initial selection and mandatory erase-resolution cuts are the only catalog reads"; - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0); - EXPECT_EQ(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns), + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1) + << "an operation whose admission is gone before the erase reports the selection cut it " + "already holds and reads nothing further"; + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0); + EXPECT_EQ(CasRefCatalog::lifeIfCataloged(op, layout, fixture.ns), NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id)); } @@ -2974,127 +3084,148 @@ TEST(CASCatalogLifecycleReconciler, RetriesFromTheMandatoryConflictResolutionCut { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - seedCompletedRemoving(*backend, store, kGc); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); + seedCompletedRemoving(op, store, kGc); + const auto parent_object = op.read(layout.foldSealKey(1, 1), Retry::once()); ASSERT_TRUE(parent_object); const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); backend->clearJournal(); backend->resetCounts(); backend->conflictNextCatalogCas(layout.refCatalogKey()); - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [](uint64_t) - { - return CasRefCatalog::LeaderFenceStatus::Held; - }); - const CatalogLifecycleReconcileResult result = reconciler.reconcile(); + CatalogLifecycleReconciler reconciler(op, layout, parent); + const CatalogLifecycleReconcileResult result = reconciler.reconcile(noAuthorityRefresh); EXPECT_EQ(result.authority_status, AuthorityStatus::Authoritative); EXPECT_EQ(result.catalog_resolution, CatalogResolution::DrainComplete); EXPECT_EQ(result.deleted, 1); const std::vector journal = backend->journalSnapshot(); const String catalog_get = "get " + layout.refCatalogKey(); - EXPECT_EQ(std::count(journal.begin(), journal.end(), catalog_get), 3) - << "the token-conflict retry must reuse its mandatory resolution cut"; + EXPECT_EQ(std::count(journal.begin(), journal.end(), catalog_get), 4) + << "selection, the refused write's own resolve read, the mandatory resolution cut the retry " + "reuses, and the committed erase's resolution -- the catalog takes no cut of its own"; } -TEST(CASCatalogLifecycleReconciler, PropagatesAuthorityFailureBeforeEraseCas) +/// THE DRAIN'S AUTHORITY, end to end. `CatalogLifecycleReconciler` and +/// `deleteCompletedRemovingAtSnapshot` decide `FencedOut` from `CasOperation::admitted()`, and the GC +/// plane's fence is open -- so the only thing that can make that verdict false in production is the +/// `Liveness` the round hands its drain operation. Depose the leader in the window the round leaves +/// between acquiring its lease and the pre-fold drain, and the drain must erase nothing. Without the +/// predicate the verdict is a constant TRUE, the drain completes, and the completed-removal row is +/// gone -- which is what this test catches. +TEST(CASGCFrontierGate, ADeposedLeaderErasesNoCatalogRow) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - seedCompletedRemoving(*backend, store, kGc); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); - ASSERT_TRUE(parent_object); - const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); - size_t fence_checks = 0; - - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [&fence_checks](uint64_t) - { - if (++fence_checks == 2) - throw std::runtime_error("injected reconciler authority failure before CAS"); - return CasRefCatalog::LeaderFenceStatus::Held; - }); - try - { - (void)reconciler.reconcile(); - FAIL() << "the authority exception must propagate"; - } - catch (const std::runtime_error & e) - { - EXPECT_STREQ(e.what(), "injected reconciler authority failure before CAS"); - } -} + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); + const uint64_t catalog_writes_before = backend->putOverwriteCount(layout.refCatalogKey()); + + /// Another leader steals `gc/state` after this round's lease renewal and before its drain. + const auto depose = [&] + { + const auto got = op.read(layout.gcStateKey(), Retry::once()); + ASSERT_TRUE(got); + GcState stolen = decodeGcState(got->bytes); + stolen.lease.owner = hexToU128("00000000000000000000000000000099"); + ++stolen.lease.seq; + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.gcStateKey(), encodeGcState(stolen), got->etag, Retry::once()))); + }; -TEST(CASCatalogLifecycleReconciler, PropagatesAuthorityFailureAfterMandatoryResolution) + Gc gc(store, kGc); + EXPECT_THROW(gc.runRegularRound(depose), DB::Exception) + << "a deposed leader must give up rather than drain the catalog"; + EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(op, layout, fixture.ns)) + << "the completed-removal row survives a deposed leader's drain"; + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_writes_before) + << "and no catalog write was even attempted"; +} + +/// THE SAME AUTHORITY, now DURING the drain. A drain erases one row per iteration, and what stops a +/// leader deposed between two erases is the refresh the ERASE runs at the top of every attempt (the +/// reconciler only forwards it): the first row goes, the second is never attempted, and the drain +/// reports `FencedOut` from the cut it already holds. One reading taken before the drain would +/// authorise both erases: the erase count and the surviving-row count below are what catch that, +/// since an unrefreshed drain sends a second erase and empties the catalog under a lease this leader +/// no longer owns. +TEST(CASGCFrontierGate, ALeaderDeposedBetweenTwoErasesStopsAfterTheFirst) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); - const auto parent_object = backend->get(layout.foldSealKey(1, 1)); - ASSERT_TRUE(parent_object); - const CasFoldSeal parent = decodeFoldSeal(parent_object->bytes); - size_t fence_checks = 0; - - CatalogLifecycleReconciler reconciler( - *backend, - layout, - parent, - /*admitted_generation=*/1, - [&fence_checks](uint64_t) - { - if (++fence_checks == 3) - throw std::runtime_error("injected reconciler authority failure after resolution"); - return CasRefCatalog::LeaderFenceStatus::Held; - }); - try - { - (void)reconciler.reconcile(); - FAIL() << "the authority exception must propagate"; - } - catch (const std::runtime_error & e) - { - EXPECT_STREQ(e.what(), "injected reconciler authority failure after resolution"); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); - } + seedCompletedRemovingBatch(op, store, kGc, /*count=*/2); + const std::vector seeded{ + RootNamespace{"00/drain-batch-0@cas@"}, RootNamespace{"00/drain-batch-1@cas@"}}; + const uint64_t catalog_writes_before = backend->putOverwriteCount(layout.refCatalogKey()); + + /// The hook runs after every catalog read, and the first read it sees with an erase already behind + /// it is the resolution read that closed erase one -- exactly the window between the two erases. + /// Nothing before the drain reads or writes the catalog, so no earlier read can trip this. + bool deposed = false; + backend->afterReadOf(layout.refCatalogKey(), [&] + { + if (deposed || backend->putOverwriteCount(layout.refCatalogKey()) == catalog_writes_before) + return; + deposed = true; + const auto got = op.read(layout.gcStateKey(), Retry::once()); + ASSERT_TRUE(got); + GcState stolen = decodeGcState(got->bytes); + stolen.lease.owner = hexToU128("00000000000000000000000000000099"); + ++stolen.lease.seq; + EXPECT_TRUE(std::holds_alternative( + op.replace(layout.gcStateKey(), encodeGcState(stolen), got->etag, Retry::once()))); + }); + + Gc gc(store, kGc); + EXPECT_THROW(gc.runRegularRound(), DB::Exception) + << "a leader deposed inside its own drain must not finish the round"; + EXPECT_TRUE(deposed) << "the round must have reached a catalog read after its first erase"; + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_writes_before + 1) + << "one erase reached the store; the second was refused before it was sent"; + size_t still_cataloged = 0; + for (const RootNamespace & ns : seeded) + if (CasRefCatalog::lifeIfCataloged(op, layout, ns)) + ++still_cataloged; + EXPECT_EQ(still_cataloged, 1u) + << "one row was erased before the deposition; the other survives it"; } TEST(CASGCFrontierGate, HealthyRebuildUsesTheCatalogLifecycleReconciler) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); - const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); + const uint64_t catalog_cas_before = backend->putOverwriteCount(layout.refCatalogKey()); Gc gc(store, kGc); const RebuildReport result = gc.rebuildBaseline(/*force=*/true); EXPECT_TRUE(result.performed); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before + 1); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, fixture.ns)); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_cas_before + 1); } TEST(CASGCFrontierGate, DamagedStateRebuildDoesNotDeleteCompletedRemovingRows) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/damaged-rebuild-removing@cas@"}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{ .ns = ns, .state = NsState::Live, .incarnation = UInt128{901}}); - CasRefCatalog::casUpdate(*backend, layout, [](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; next.entries.front().state = NsState::Removing; @@ -3107,26 +3238,28 @@ TEST(CASGCFrontierGate, DamagedStateRebuildDoesNotDeleteCompletedRemovingRows) .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, }); - const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + const uint64_t catalog_cas_before = backend->putOverwriteCount(layout.refCatalogKey()); Gc gc(store, kGc); const RebuildReport result = gc.rebuildBaseline(/*force=*/false); EXPECT_TRUE(result.performed); - EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before); + EXPECT_TRUE(CasRefCatalog::lifeIfCataloged(op, layout, ns)); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_cas_before); } TEST(CASGCFrontierGate, DeferredRoundDrainsCompletedRemovingBeforeReturning) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/100); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace removed{"00/deferred-removed@cas@"}; const UInt128 life_id{77}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{ + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{ .ns = removed, .state = NsState::Live, .incarnation = life_id}); - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; next.entries[0].state = NsState::Removing; @@ -3137,31 +3270,31 @@ TEST(CASGCFrontierGate, DeferredRoundDrainsCompletedRemovingBeforeReturning) CasFoldSeal parent; parent.generation = 1; parent.ref_lives.emplace(life_id, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); for (uint64_t shard = 0; shard < store->poolConfig().gc_shards; ++shard) parent.condemned_summary.emplace(shard, CondemnedSummary{}); - ASSERT_EQ(backend->putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(parent)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.foldSealKey(1, 1), encodeFoldSeal(parent), Retry::once()))); GcState state; state.round = 1; state.gc_shards = store->poolConfig().gc_shards; state.snap_generation = 1; state.snap_attempt = 1; state.lease = GcLease{.owner = kGc, .seq = 1}; - ASSERT_EQ(backend->putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.gcStateKey(), encodeGcState(state), Retry::once()))); const String ckpt_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(removed, life_id)); - ASSERT_EQ(backend->putIfAbsent(ckpt_key, "inert checkpoint debris").outcome, PutOutcome::Done); - const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + ASSERT_TRUE(std::holds_alternative(op.create(ckpt_key, "inert checkpoint debris", Retry::once()))); + const uint64_t catalog_cas_before = backend->putOverwriteCount(layout.refCatalogKey()); Gc gc(store, kGc); const RoundReport report = runRegularRoundReclaiming(gc); ASSERT_TRUE(report.acquired_lease); EXPECT_TRUE(report.deferred); - EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, removed)); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before + 1); - EXPECT_TRUE(backend->head(ckpt_key).exists); + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, removed)); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_cas_before + 1); + EXPECT_TRUE(op.head(ckpt_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(ckpt_key), 0); } @@ -3169,9 +3302,11 @@ TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListi { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const UInt128 leader_b = hexToU128("00000000000000000000000000000002"); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); backend->clearJournal(); backend->blockNextCatalogCas(layout.refCatalogKey()); @@ -3223,7 +3358,7 @@ TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListi ASSERT_FALSE(leader_b_failure); ASSERT_TRUE(report_b.acquired_lease); ASSERT_FALSE(report_b.deferred); - ASSERT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + ASSERT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); const size_t catalog_cas_end = findJournalAfter(before_a_release, "cas_end " + layout.refCatalogKey(), 0); ASSERT_LT(catalog_cas_end, before_a_release.size()); @@ -3236,7 +3371,7 @@ TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListi const size_t fresh_catalog_cut = findJournalAfter( before_a_release, "get " + layout.refCatalogKey(), stream_list + 1); ASSERT_LT(fresh_catalog_cut, before_a_release.size()); - const GcState adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState adopted = decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes); const String successor_seal_key = layout.foldSealKey(adopted.snap_generation, adopted.snap_attempt); const size_t successor_seal_put = findJournalAfter( before_a_release, "put_end " + successor_seal_key, fresh_catalog_cut + 1); @@ -3265,7 +3400,7 @@ TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListi EXPECT_LT(successor_seal_put, successor_adoption); ASSERT_TRUE(leader_a_failure); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, fixture.ns)); /// Same discrimination as `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`: leader_b's /// round both drops `fixture.ns`'s catalog row AND, because the resulting cut is genuinely, /// provably empty, opens the destructive gate -- so the namespace janitor reclaims the checkpoint @@ -3275,7 +3410,7 @@ TEST(CASGCFrontierGate, StaleIssuedCatalogCasLosesAfterNewLeaderHelpsBeforeListi ASSERT_FALSE(janitor_metrics_b.empty()) << "the namespace_cleanup phase must have run this round"; EXPECT_GE(janitor_metrics_b.at("janitor_deleted"), 1u) << "the janitor's OWN counter must show the delete, now that the proved-empty gate has opened"; - EXPECT_FALSE(backend->get(fixture.checkpoint_key).has_value()); + EXPECT_FALSE(op.read(fixture.checkpoint_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(fixture.checkpoint_key), 1); } @@ -3283,8 +3418,10 @@ TEST(CASGCFrontierGate, LostCatalogCasResponseIsResolvedBeforeListing) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); backend->clearJournal(); backend->loseNextCatalogCasResponse(layout.refCatalogKey()); @@ -3322,7 +3459,7 @@ TEST(CASGCFrontierGate, LostCatalogCasResponseIsResolvedBeforeListing) EXPECT_LT(conclusive_rescan, stream_list); EXPECT_LT(stream_list, fresh_catalog_cut); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, fixture.ns)); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(op, layout, fixture.ns)); /// See the discrimination comment in `CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanitor`: /// attribute the delete to the janitor's own counter, never to end-state absence alone, and never /// assume survival -- both would be indistinguishable from a bug on this exact line (the old @@ -3330,7 +3467,7 @@ TEST(CASGCFrontierGate, LostCatalogCasResponseIsResolvedBeforeListing) ASSERT_FALSE(janitor_metrics.empty()) << "the namespace_cleanup phase must have run this round"; EXPECT_GE(janitor_metrics.at("janitor_deleted"), 1u) << "the janitor's OWN counter must show the delete, now that the proved-empty gate has opened"; - EXPECT_FALSE(backend->get(fixture.checkpoint_key).has_value()); + EXPECT_FALSE(op.read(fixture.checkpoint_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(fixture.checkpoint_key), 1); } @@ -3341,9 +3478,11 @@ TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrRepl { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const UInt128 leader_b = hexToU128("00000000000000000000000000000002"); - const CompletedRemovingFixture fixture = seedCompletedRemoving(*backend, store, kGc); + const CompletedRemovingFixture fixture = seedCompletedRemoving(op, store, kGc); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(fixture.ns, fixture.life_id); ASSERT_TRUE(store->refTableRecoveredForTest(fixture.ns)) @@ -3370,7 +3509,7 @@ TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrRepl backend->waitForBlockedCatalogCas(); transferGcLease(*backend, layout, leader_b); - const CasRefCatalog::Snapshot observed = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot observed = CasRefCatalog::read(op, layout); RefCatalog winner_catalog; if (GetParam() == CompetingCatalogOutcome::Replacement) { @@ -3381,16 +3520,16 @@ TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrRepl /// Mirror production's publish-then-flip order: the successor life needs a readable `_ckpt` /// before its catalog row can read `Live`, or `chooseRecoveryGrounding` rejects it. const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(fixture.ns, UInt128{178}); - backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + op.create(layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, - })); + }), Retry::once()); } - ASSERT_EQ(backend->casPut( - layout.refCatalogKey(), encodeRefCatalog(winner_catalog), observed.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(observed.etag); + ASSERT_TRUE(std::holds_alternative(op.replace( + layout.refCatalogKey(), encodeRefCatalog(winner_catalog), *observed.etag, Retry::once()))); backend->clearJournal(); const uint64_t plans_before /// NOLINT(clang-analyzer-deadcode.DeadStores) @@ -3444,9 +3583,11 @@ TEST(CASGCFrontierGate, CompletedRemovalDrainUsesNPlusOneCatalogReads) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); constexpr size_t deletes = 3; - seedCompletedRemovingBatch(*backend, store, kGc, deletes); + seedCompletedRemovingBatch(op, store, kGc, deletes); backend->clearJournal(); backend->resetCounts(); diff --git a/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp index 19a63bc6842a..3d67ac02f317 100644 --- a/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp +++ b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp @@ -24,9 +24,9 @@ /// DURABLE HOLDS (spec 2026-07-27 "ref chain complete cut" §5). /// /// A namespace whose ref-log walk meets an IMPOSSIBLE shape stops there, and that stop has to survive -/// the round. Before this task the stop was a single bit — `classification == 4` — and everything that -/// explained it (what went wrong, and exactly WHERE) lived in a log line and an in-memory anomaly, both -/// gone by the next round. That is not enough for three separate reasons: +/// the round. A classification alone cannot preserve the cause and position of the stop; without +/// durable hold evidence, that information lives only in a log line and an +/// in-memory anomaly, both gone by the next round. That is not enough for three separate reasons: /// /// * the next round could not RETRY the exact position, so a hold only survived while the round's /// hint happened to keep mentioning the namespace; @@ -36,10 +36,10 @@ /// baseline that looked proven when it was not. /// /// So the hold is now DURABLE and STRICTLY GRAMMARED: `{reason, offending_position, retry_count, -/// next_retry_round}` present if and only if `classification == 4`, rejected in both directions -/// otherwise. It rides the seal across rounds — including rounds whose hint omits the namespace -/// entirely — and across REBUILD, and it clears by exactly ONE event: the fold resolving the offending -/// position and that result being adopted in `gc/state`. +/// next_retry_round}` present if and only if `classification == CoverageClass::Clamped`, rejected in +/// both directions otherwise. It rides the seal across rounds — including rounds whose hint omits the +/// namespace entirely — and across REBUILD, and it clears by exactly ONE event: the fold resolving the +/// offending position and that result being adopted in `gc/state`. /// /// The carried hold is also a WITNESS, and a better one than the listing: it is durable proof that the /// walk once reached that position, so an absent below it is a gap rather than a frontier no matter @@ -74,6 +74,9 @@ const UInt128 kGc = hexToU128("00000000000000000000000000000001"); class HintHoleCountingBackend : public CountingBackend { public: + /// Unhide the names the primitive overrides below would otherwise shadow. + using CountingBackend::list; + void hide(const String & key) { std::lock_guard lock(m); @@ -86,14 +89,14 @@ class HintHoleCountingBackend : public CountingBackend return served; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = CountingBackend::list(prefix, cursor, limit); + RawListPage page = CountingBackend::list(prefix, cursor, limit, access); std::lock_guard lock(m); if (hidden.empty()) return page; const size_t before = page.keys.size(); - std::erase_if(page.keys, [&](const ListedKey & k) { return hidden.contains(k.key); }); + std::erase_if(page.keys, [&](const RawListedKey & k) { return hidden.contains(k.key); }); if (page.keys.size() != before) ++served; return page; @@ -138,9 +141,10 @@ std::optional newestSeal(Backend & backend, const Layout & layout) { const uint64_t gen = currentGenerationOf(backend, layout); const uint64_t attempt = currentAttemptOf(backend, layout); + OperationForTest op(backend); for (uint64_t g = gen; ; --g) { - if (const auto got = backend.get(layout.foldSealKey(g, attempt))) + if (const auto got = (*op).read(layout.foldSealKey(g, attempt), Retry::once())) return decodeFoldSeal(got->bytes); if (g == 0) return std::nullopt; @@ -177,8 +181,8 @@ RefHold holdOf(Backend & backend, const Layout & layout, const RootNamespace & n EXPECT_TRUE(cov.has_value()) << "no coverage row for " << ns.string(); if (!cov) return RefHold{}; - EXPECT_EQ(cov->classification, 4) << "a held namespace is classification 4"; - EXPECT_TRUE(cov->hold.has_value()) << "classification 4 without a hold is the forbidden shape"; + EXPECT_EQ(cov->classification, CoverageClass::Clamped) << "a held namespace is classification clamped"; + EXPECT_TRUE(cov->hold.has_value()) << "classification clamped without a hold is the forbidden shape"; return cov->hold ? *cov->hold : RefHold{}; } @@ -206,7 +210,7 @@ CasFoldSeal maximalHoldSeal(const String & map_key) seal.generation = std::numeric_limits::max(); seal.parent_generation = std::numeric_limits::max(); RefCoverage cov; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.last_folded_ref_id = RefTxnId{std::numeric_limits::max(), std::numeric_limits::max()}; cov.hold = RefHold{.reason = HoldReason::UnconsumedSealCrossing, /// the longest reason word @@ -233,7 +237,7 @@ CasFoldSeal cleanSeal(const String & map_key) seal.generation = 3; seal.parent_generation = 2; RefCoverage cov; - cov.classification = 2; + cov.classification = CoverageClass::Folded; cov.last_folded_ref_id = RefTxnId{4, 5}; fixtureCoverage(seal, map_key) = cov; return seal; @@ -245,7 +249,7 @@ CasFoldSeal heldSeal(const String & map_key) { CasFoldSeal seal = cleanSeal(map_key); RefCoverage & cov = fixtureCoverage(seal, map_key); - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{4, 6}, .retry_count = 7, .next_retry_round = 99}; return seal; @@ -271,20 +275,6 @@ String sealTextWith(const String & prototype, const std::vector & record return text + "{\"n\":" + std::to_string(records.size()) + "}\n"; } -/// Replace the coverage row's `cls` value with `raw`, VERBATIM. The point is to write integers no -/// `RefCoverage` can hold: the field is a byte in the struct, so a wide value exists only on the wire, -/// which is exactly where a reader has to catch it. `cls` is never the last field of a `cov` record, so -/// the value always ends at a comma. -String withRawClassification(const String & encoded, std::string_view raw) -{ - const size_t at = encoded.find("\"cls\":"); - EXPECT_NE(at, String::npos); - const size_t begin = at + strlen("\"cls\":"); - const size_t end = encoded.find(',', begin); - EXPECT_NE(end, String::npos); - return encoded.substr(0, begin) + String{raw} + encoded.substr(end); -} - /// Replace the FIRST occurrence of `field` with `replacement` (both are whole `"key":value` fragments), /// so a test states the exact wire shape it is feeding the decoder. String withField(const String & encoded, const String & field, const String & replacement) @@ -303,24 +293,26 @@ std::vector> illFormedSealsTheEncoderMustRe /// The pairing, both ways round. CasFoldSeal hold_on_folded = heldSeal("ns/0"); - fixtureCoverage(hold_on_folded, "ns/0").classification = 2; - out.emplace_back("a hold on a folded (2) row claims a stop that did not happen", hold_on_folded); + fixtureCoverage(hold_on_folded, "ns/0").classification = CoverageClass::Folded; + out.emplace_back("a hold on a folded row claims a stop that did not happen", hold_on_folded); CasFoldSeal clamped_without_hold = heldSeal("ns/0"); fixtureCoverage(clamped_without_hold, "ns/0").hold.reset(); - out.emplace_back("a clamped (4) row with no hold is indistinguishable from a clean cursor once " + out.emplace_back("a clamped row with no hold is indistinguishable from a clean cursor once " "durable", clamped_without_hold); - /// The closed set. 3 is the dangerous one: it passes the sweep's `== 4` and `== 0` refusals and - /// reaches the deletion premise, which is a refusal written in terms of the set. - CasFoldSeal classification_three = cleanSeal("ns/0"); - fixtureCoverage(classification_three, "ns/0").classification = 3; - out.emplace_back("classification 3 is not one of {0,1,2,4} and passes every refusal stated in terms " - "of them", classification_three); + /// The closed set is now the enum's declared values, so only an explicit cast reaches outside it. + /// 4 is the sharpest value to plant: it is outside the closed wire vocabulary. + CasFoldSeal classification_retired_wire_value = cleanSeal("ns/0"); + fixtureCoverage(classification_retired_wire_value, "ns/0").classification + = static_cast(4); + out.emplace_back("classification 4 is outside the four values the wire table declares", + classification_retired_wire_value); CasFoldSeal classification_max = cleanSeal("ns/0"); - fixtureCoverage(classification_max, "ns/0").classification = 255; - out.emplace_back("classification 255 is not one of {0,1,2,4}", classification_max); + fixtureCoverage(classification_max, "ns/0").classification = static_cast(255); + out.emplace_back("classification 255 is outside the four values the wire table declares", + classification_max); /// The self-erasing hold, and its half-zero sibling. CasFoldSeal hold_at_zero = heldSeal("ns/0"); @@ -334,6 +326,53 @@ std::vector> illFormedSealsTheEncoderMustRe return out; } +/// ---- Small raw-fixture request-engine wrappers shared by the tests below ---- +/// (`head`/`get`/`putOverwrite`/`putIfAbsent`/`deleteExact` are the legacy `Backend` verbs; every +/// caller now goes through an admitted `CasOperation`.) + +/// The durable object at `key`, or `nullopt`. +std::optional readAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::once()); +} + +/// True iff `key` exists. +bool existsAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::once()).has_value(); +} + +/// Unconditional create of a fresh key (the fixture's own corruption/injection setup, never a +/// real conflict). +void createAt(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + EXPECT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// Head, then unconditionally overwrite what was seen -- the raw-fixture corruption idiom this file's +/// tests use to replace an object's body in place. +void headThenReplace(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + const auto current = (*op).head(key, Retry::once()); + EXPECT_TRUE(current.has_value()) << "expected '" << key << "' to exist before overwrite"; + if (current) + EXPECT_TRUE(std::holds_alternative((*op).replace(key, bytes, current->etag, Retry::once()))); +} + +/// Head, then exact-delete what was seen -- the raw-fixture corruption idiom for removing an object +/// this test just observed present. +void headThenRemove(Backend & backend, const String & key) +{ + OperationForTest op(backend); + const auto current = (*op).head(key, Retry::once()); + ASSERT_TRUE(current.has_value()) << "expected '" << key << "' to exist before removal"; + ASSERT_EQ((*op).remove(key, current->etag, Retry::once()), Removal::Removed); +} + } /// ===================== THE SHARED BYTE ARITHMETIC ===================== @@ -371,7 +410,7 @@ TEST(CASGCHoldGrammarBudget, SumsSaturateInsteadOfWrapping) EXPECT_FALSE(fitsObjectCap(kMax, 2, 256 * 1024 * 1024)); } -/// ===================== THE STRICT CLASSIFICATION-4 GRAMMAR ===================== +/// ===================== THE STRICT CLAMPED-CLASSIFICATION GRAMMAR ===================== TEST(CASGCHoldGrammar, EveryHoldReasonRoundTrips) { @@ -383,7 +422,7 @@ TEST(CASGCHoldGrammar, EveryHoldReasonRoundTrips) seal.generation = 3; seal.parent_generation = 2; RefCoverage cov; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.last_folded_ref_id = RefTxnId{4, 5}; cov.hold = RefHold{.reason = reason, .offending_position = RefTxnId{4, 6}, .retry_count = 7, .next_retry_round = 99}; @@ -432,23 +471,20 @@ TEST(CASGCHoldGrammar, AHoldOnAnyOtherClassificationIsRefusedByTheDecoder) /// Bytes some other producer wrote. Built by demoting a legitimate held row's classification, so the /// hold fields are exactly the ones the encoder emits. - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, .retry_count = 0, .next_retry_round = 1}; fixtureCoverage(seal, "ns/0") = cov; - String text = encodeFoldSeal(seal); - const size_t at = text.find("\"cls\":4"); - ASSERT_NE(at, String::npos); - text[at + 6] = '2'; + const String text = withField(encodeFoldSeal(seal), R"("class":"clamped")", R"("class":"folded")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(text); }); } -TEST(CASGCHoldGrammar, ClassificationFourWithoutAHoldIsRefusedByTheDecoder) +TEST(CASGCHoldGrammar, ClampedWithoutAHoldIsRefusedByTheDecoder) { CasFoldSeal seal; seal.generation = 1; RefCoverage cov; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.last_folded_ref_id = RefTxnId{1, 1}; /// Every single hold field is REQUIRED: dropping any one of them is corruption, not a default. @@ -456,8 +492,8 @@ TEST(CASGCHoldGrammar, ClassificationFourWithoutAHoldIsRefusedByTheDecoder) .retry_count = 3, .next_retry_round = 4}; fixtureCoverage(seal, "ns/0") = cov; const String whole = encodeFoldSeal(seal); - for (const String & field : {String(R"("hr":"body_undecodable")"), String(R"("hpe":"1")"), - String(R"("hps":"2")"), String(R"("hrc":3)"), String(R"("hnr":"4")")}) + for (const String & field : {String(R"("hold_reason":"body_undecodable")"), String(R"("hold_epoch":"1")"), + String(R"("hold_seq":"2")"), String(R"("retries":3)"), String(R"("retry_round":"4")")}) { SCOPED_TRACE("without " + field); const size_t at = whole.find(field); @@ -473,18 +509,18 @@ TEST(CASGCHoldGrammar, DuplicateHoldKeyIsCorruptedData) CasFoldSeal seal; seal.generation = 1; RefCoverage cov; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, .retry_count = 0, .next_retry_round = 5}; fixtureCoverage(seal, "ns/0") = cov; const String whole = encodeFoldSeal(seal); - const String field = R"("hr":"gap_below_witness")"; + const String field = R"("hold_reason":"gap_below_witness")"; const size_t at = whole.find(field); ASSERT_NE(at, String::npos); /// The same key twice, with a DIFFERENT value: last-wins would silently rewrite the reason. String doubled = whole; - doubled.insert(at, R"("hr":"witness_disappeared",)"); + doubled.insert(at, R"("hold_reason":"witness_disappeared",)"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(doubled); }); } @@ -493,7 +529,7 @@ TEST(CASGCHoldGrammar, UnknownHoldReasonWordIsCorruptedData) CasFoldSeal seal; seal.generation = 1; RefCoverage cov; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, .retry_count = 0, .next_retry_round = 5}; fixtureCoverage(seal, "ns/0") = cov; @@ -510,48 +546,46 @@ TEST(CASGCHoldGrammar, UnknownHoldReasonWordIsCorruptedData) /// The three shapes below are one finding, and it is about what a fold seal is FOR. The hold is the only /// durable record that a namespace stopped and where; everything downstream reads the seal and nothing /// re-derives the stop. So a seal that decodes into "no hold here" is not a lossy read, it is a licence -/// to delete: the sweep's §6 refusals are stated as `classification == 4` / `== 0` / `hold.has_value()`, -/// and a row that slips past all three reaches an irreversible delete of a manifest the fold never -/// accounted for. Each shape gets past a DIFFERENT one of the decoder's checks, which is why they are -/// pinned separately rather than as one "malformed seal" case. - -/// (1) The classification the reader never sees. `cls` is narrowed to a byte, so an integer on the wire -/// is truncated first and validated (if at all) afterwards: 258 becomes 2, "everything through the -/// cursor was folded". The value has to be judged WIDE, before the narrowing, or the wire can buy -/// coverage that no fold ever performed. +/// to delete: the sweep's §6 refusals are stated as `classification == Clamped` / `== Absent` / +/// `hold.has_value()`, and a row that slips past all three reaches an irreversible delete of a manifest +/// the fold never accounted for. Each shape gets past a DIFFERENT one of the decoder's checks, which is +/// why they are pinned separately rather than as one "malformed seal" case. + +/// (1) The classification is a WORD, closed the same way `hold_reason` already is: +/// `coverageClassFromWord` refuses anything outside the four named values as `CORRUPTED_DATA` before a +/// `CoverageClass` is ever constructed, so there is no wide-integer narrowing attack left to catch here — +/// the wire carries no integer at all. TEST(CASGCHoldGrammar, AClassificationOutsideTheGrammarIsCorruptedData) { const String clean = encodeFoldSeal(cleanSeal("ns/0")); - ASSERT_EQ(fixtureCoverage(decodeFoldSeal(clean), "ns/0").classification, 2) + ASSERT_EQ(fixtureCoverage(decodeFoldSeal(clean), "ns/0").classification, CoverageClass::Folded) << "the unmodified row is the one every case below deviates from"; - /// In-range bytes that are simply not classifications. 3 is the one the sweep's refusals miss. - for (const std::string_view raw : {"3", "5", "6", "255"}) + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - SCOPED_TRACE(String{"cls="} + String{raw}); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withRawClassification(clean, raw)); }); - } + decodeFoldSeal(withField(clean, R"("class":"folded")", R"("class":"foldedx")")); + }); - /// Wide integers whose LOW BYTE lands inside the grammar: 258 -> 2 (fully folded), 256 -> 0 - /// (absent), 260 -> 4 (clamped). Each would decode as a row the fold never wrote. - for (const std::string_view raw : {"256", "258", "260", "18446744073709551615"}) + /// The bare number `4` is the classification's pre-cut wire representation — the old byte-valued + /// form. A retired spelling is legal here because this is a marked negative fixture proving the + /// decoder still refuses it now that `class` takes a word; the byte-delta pins hold the other such + /// fixtures, and both kinds are exempt from the vocabulary sweeps for the same reason. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - SCOPED_TRACE(String{"cls="} + String{raw}); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withRawClassification(clean, raw)); }); - } + decodeFoldSeal(withField(clean, R"("class":"folded")", R"("class":4)")); + }); } -/// And the field itself is required: an absent `cls` reads as 0, which is not "nothing was said about -/// this namespace" but the positive claim "no round folded it". +/// And the field itself is required: an absent `class` reads as `absent`, which is not "nothing was +/// said about this namespace" but the positive claim "no round folded it". TEST(CASGCHoldGrammar, ACoverageRowWithoutAClassificationIsCorruptedData) { const String clean = encodeFoldSeal(cleanSeal("ns/0")); - const size_t at = clean.find("\"cls\":2,"); + const String field = R"("class":"folded",)"; + const size_t at = clean.find(field); ASSERT_NE(at, String::npos); String without = clean; - without.erase(at, strlen("\"cls\":2,")); + without.erase(at, field.size()); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeFoldSeal(without); }); } @@ -567,13 +601,13 @@ TEST(CASGCHoldGrammar, AHoldWhoseOffendingPositionHasAZeroComponentIsCorruptedDa expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - decodeFoldSeal(withField(withField(held, R"("hpe":"4")", R"("hpe":"0")"), - R"("hps":"6")", R"("hps":"0")")); + decodeFoldSeal(withField(withField(held, R"("hold_epoch":"4")", R"("hold_epoch":"0")"), + R"("hold_seq":"6")", R"("hold_seq":"0")")); }); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withField(held, R"("hpe":"4")", R"("hpe":"0")")); }); + [&] { decodeFoldSeal(withField(held, R"("hold_epoch":"4")", R"("hold_epoch":"0")")); }); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withField(held, R"("hps":"6")", R"("hps":"0")")); }); + [&] { decodeFoldSeal(withField(held, R"("hold_seq":"6")", R"("hold_seq":"0")")); }); } /// (3) The duplicate row. Two `cov` records for the same (namespace, shard) — held first, clean second — @@ -608,7 +642,7 @@ TEST(CASGCHoldGrammar, ASecondCoverageRowForTheSameKeyIsCorruptedData) EXPECT_EQ(seal.ref_lives.size(), 1u); } -/// The same one-record-per-key rule applies to `cnd`: a repeated row rewrites a shard's condemned +/// The same one-record-per-key rule applies to `condemned`: a repeated row rewrites a shard's condemned /// totals, which graduation paces on. TEST(CASGCHoldGrammar, ASecondCondemnedSummaryRecordIsCorruptedData) { @@ -617,7 +651,7 @@ TEST(CASGCHoldGrammar, ASecondCondemnedSummaryRecordIsCorruptedData) .oldest_nonpending_condemn_round = 3}; const String encoded = encodeFoldSeal(seal); - /// Lines 3..4 are `rfl`, `cnd` in the encoder's fixed order. + /// Lines 3..4 are `ref_life`, `condemned` in the encoder's fixed order. std::vector lines; for (size_t begin = headerAndMetaOf(encoded).size(); begin < encoded.size();) { @@ -626,14 +660,14 @@ TEST(CASGCHoldGrammar, ASecondCondemnedSummaryRecordIsCorruptedData) lines.push_back(encoded.substr(begin, end - begin)); begin = end + 1; } - ASSERT_EQ(lines.size(), 3u) << "rfl, cnd and the trailer"; + ASSERT_EQ(lines.size(), 3u) << "ref_life, condemned and the trailer"; const String ref_life_line = lines[0]; - const String cnd_line = lines[1]; + const String condemned_line = lines[1]; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(sealTextWith(encoded, {ref_life_line, cnd_line, cnd_line})); }); + [&] { decodeFoldSeal(sealTextWith(encoded, {ref_life_line, condemned_line, condemned_line})); }); /// The unduplicated assembly is the control. - const std::vector one_of_each{ref_life_line, cnd_line}; + const std::vector one_of_each{ref_life_line, condemned_line}; EXPECT_NO_THROW(decodeFoldSeal(sealTextWith(encoded, one_of_each))); } @@ -647,12 +681,12 @@ TEST(CASGCHoldGrammar, CleanupEvidenceWithAZeroRemovalIdIsCorruptedData) const String encoded = encodeFoldSeal(seal); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withField(encoded, R"("rte":"2")", R"("rte":"0")")); }); + [&] { decodeFoldSeal(withField(encoded, R"("remove_epoch":"2")", R"("remove_epoch":"0")")); }); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withField(encoded, R"("rts":"3")", R"("rts":"0")")); }); + [&] { decodeFoldSeal(withField(encoded, R"("remove_seq":"3")", R"("remove_seq":"0")")); }); /// Omitted entirely is the same thing: the fields default to zero. expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeFoldSeal(withField(encoded, R"("rte":"2",)", "")); }); + [&] { decodeFoldSeal(withField(encoded, R"("remove_epoch":"2",)", "")); }); } /// The OBJECT cap bounds the whole seal. Nothing on the fold-seal READ path enforces it (the seal @@ -738,7 +772,10 @@ TEST(CASGCHoldGrammar, UndecodableBodyNamesTheRecordItCouldNotRead) fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); - backend->putIfAbsent(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2}), "this is not a cas_ref_log object"); + { + OperationForTest op(*backend); + (*op).create(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 2}), "this is not a cas_ref_log object", Retry::once()); + } writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 2}); Gc gc(store, kGc); @@ -786,15 +823,14 @@ TEST(CASGCHoldGrammar, AWitnessThatStopsAnsweringIsWitnessDisappeared) class AlternatingGetBackend : public InMemoryBackend { public: - using DB::Cas::Backend::get; String flaky; size_t reads = 0; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (key == flaky && ++reads % 2 == 0) return std::nullopt; - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } }; @@ -986,22 +1022,23 @@ TEST(CASGCHoldGrammar, AnUndecodableCheckpointHoldsOnlyItsOwnNamespace) /// Corrupt EXACTLY ONE OBJECT: the first namespace's `_ckpt` body. Nothing else in the pool changes, /// so everything the next round does differently is attributable to this one object. const String bad_ckpt_key = layout.refCkptKey(fixture::fixtureLife(bad)); - const HeadResult ckpt_head = backend->head(bad_ckpt_key); - ASSERT_TRUE(ckpt_head.exists); - ASSERT_EQ(backend->putOverwrite(bad_ckpt_key, "this is not a cas_ref_ckpt", ckpt_head.token).outcome, - PutOutcome::Done); + OperationForTest corrupt_op(*backend); + const auto ckpt_head = (*corrupt_op).head(bad_ckpt_key, Retry::once()); + ASSERT_TRUE(ckpt_head.has_value()); + ASSERT_TRUE(std::holds_alternative( + (*corrupt_op).replace(bad_ckpt_key, "this is not a cas_ref_ckpt", ckpt_head->etag, Retry::once()))); /// Work only a round that COMPLETES can fold. publishAt(*backend, layout, good, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(12)); const String good_ckpt_key = layout.refCkptKey(fixture::fixtureLife(good)); - const HeadResult good_ckpt_head = backend->head(good_ckpt_key); - ASSERT_TRUE(good_ckpt_head.exists); - ASSERT_EQ(backend->putOverwrite(good_ckpt_key, encodeRefCkpt(RefCkpt{ + const auto good_ckpt_head = (*corrupt_op).head(good_ckpt_key, Retry::once()); + ASSERT_TRUE(good_ckpt_head.has_value()); + ASSERT_TRUE(std::holds_alternative((*corrupt_op).replace(good_ckpt_key, encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = RefTxnId{1, 2}, .last_epoch_seal = std::nullopt, - }), good_ckpt_head.token).outcome, PutOutcome::Done); + }), good_ckpt_head->etag, Retry::once()))); ASSERT_TRUE(gc.runRegularRound().acquired_lease); @@ -1018,13 +1055,13 @@ TEST(CASGCHoldGrammar, AnUndecodableCheckpointHoldsOnlyItsOwnNamespace) ASSERT_TRUE(good_cov.has_value()); EXPECT_FALSE(good_cov->hold.has_value()) << "the corrupt object belongs to the OTHER namespace"; EXPECT_EQ(good_cov->last_folded_ref_id, (RefTxnId{1, 2})); - EXPECT_EQ(good_cov->classification, 2); + EXPECT_EQ(good_cov->classification, CoverageClass::Folded); /// And nothing was destroyed for the held namespace: a hold shuts the round's destructive gate, so /// its ref objects — including the ones a cleanup range computed WITHOUT the unreadable checkpoint /// would have widened onto — are all still there. for (const RefTxnId & id : {RefTxnId{1, 1}, RefTxnId{1, 2}}) - EXPECT_TRUE(backend->head(layout.refLogKey(fixture::fixtureLife(bad), id)).exists) + EXPECT_TRUE(existsAt(*backend, layout.refLogKey(fixture::fixtureLife(bad), id))) << "ref log " << renderRefTxnId(id) << " of the held namespace was deleted"; } @@ -1066,7 +1103,7 @@ TEST(CASGCHoldGrammar, AnUndecodableCheckpointWithNoWalkPositionRecordsAnAnomaly fixture::admitLive(*backend, layout, phantom); /// A lone `_ckpt` with an undecodable body, and NOTHING else under that namespace. - backend->putIfAbsent(layout.refCkptKey(fixture::fixtureLife(phantom)), "this is not a cas_ref_ckpt"); + createAt(*backend, layout.refCkptKey(fixture::fixtureLife(phantom)), "this is not a cas_ref_ckpt"); publishAt(*backend, layout, good, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(11), /*birth=*/true); writeCommittedCkptAt(*backend, layout, good, RefTxnId{1, 1}); @@ -1091,7 +1128,7 @@ TEST(CASGCHoldGrammar, AnUndecodableCheckpointWithNoWalkPositionRecordsAnAnomaly const auto cov = coverageOf(*backend, layout, phantom); ASSERT_TRUE(cov.has_value()); EXPECT_FALSE(cov->hold.has_value()) << "a hold here could only name a position no round ever read"; - EXPECT_EQ(cov->classification, 1) << "nothing was folded, so the row is `unchanged`"; + EXPECT_EQ(cov->classification, CoverageClass::Unchanged) << "nothing was folded, so the row is `unchanged`"; EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{})); /// Same isolation as the held arm: the pool keeps working. @@ -1102,7 +1139,7 @@ TEST(CASGCHoldGrammar, AnUndecodableCheckpointWithNoWalkPositionRecordsAnAnomaly /// The unreadable object itself is never deleted as debris — repairing it is the operator's move, /// and GC removing it would erase the only evidence of what stopped the namespace. - EXPECT_TRUE(backend->head(layout.refCkptKey(fixture::fixtureLife(phantom))).exists); + EXPECT_TRUE(existsAt(*backend, layout.refCkptKey(fixture::fixtureLife(phantom)))); } /// ===================== THE HOLD IS DURABLE ===================== @@ -1199,7 +1236,7 @@ TEST(CASGCHoldGrammar, HoldClearsOnlyByFoldingThroughTheOffendingPosition) const auto cov = coverageOf(*backend, layout, ns); ASSERT_TRUE(cov.has_value()); EXPECT_FALSE(cov->hold.has_value()) << "folding through the offending position is what clears a hold"; - EXPECT_EQ(cov->classification, 2); + EXPECT_EQ(cov->classification, CoverageClass::Folded); EXPECT_EQ(cov->last_folded_ref_id, (RefTxnId{1, 4})) << "the walk resumed past the resolved gap"; EXPECT_EQ(inDegreeOf(*backend, layout, DB::UInt128(4)), 1) << "the record above the gap finally contributed its owner edge"; @@ -1217,9 +1254,13 @@ void mutateSealAt(Backend & backend, const Layout & layout, uint64_t generation, const std::function & mutate) { const String key = layout.foldSealKey(generation, attempt); - CasFoldSeal seal = decodeFoldSeal(backend.get(key)->bytes); + OperationForTest op(backend); + CasFoldSeal seal = decodeFoldSeal((*op).read(key, Retry::once())->bytes); mutate(seal); - backend.putOverwrite(key, encodeFoldSeal(seal), backend.head(key).token); + const auto current = (*op).head(key, Retry::once()); + EXPECT_TRUE(current.has_value()); + if (current) + EXPECT_TRUE(std::holds_alternative((*op).replace(key, encodeFoldSeal(seal), current->etag, Retry::once()))); } /// Rewrite the adopted fold seal, applying `mutate` to it. Used to plant a hold that the rebuild must @@ -1227,11 +1268,15 @@ void mutateSealAt(Backend & backend, const Layout & layout, uint64_t generation, /// the carry, not about how the hold arose. void mutateAdoptedSeal(Backend & backend, const Layout & layout, const std::function & mutate) { - const GcState st = decodeGcState(backend.get(layout.gcStateKey())->bytes); + OperationForTest op(backend); + const GcState st = decodeGcState((*op).read(layout.gcStateKey(), Retry::once())->bytes); const String key = layout.foldSealKey(st.snap_generation, st.snap_attempt); - CasFoldSeal seal = decodeFoldSeal(backend.get(key)->bytes); + CasFoldSeal seal = decodeFoldSeal((*op).read(key, Retry::once())->bytes); mutate(seal); - backend.putOverwrite(key, encodeFoldSeal(seal), backend.head(key).token); + const auto current = (*op).head(key, Retry::once()); + EXPECT_TRUE(current.has_value()); + if (current) + EXPECT_TRUE(std::holds_alternative((*op).replace(key, encodeFoldSeal(seal), current->etag, Retry::once()))); } RefHold plantedHold() @@ -1261,10 +1306,10 @@ TEST(CASGCHoldGrammar, RebuildCarriesMatchingHoldAndDropsAbsentLife) mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) { RefCoverage & cov = seal.ref_lives.at(life_id).coverage; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = plantedHold(); RefCoverage gone; - gone.classification = 4; + gone.classification = CoverageClass::Clamped; gone.last_folded_ref_id = RefTxnId{2, 2}; gone.hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{2, 3}, .retry_count = 1, .next_retry_round = 2}; @@ -1278,7 +1323,7 @@ TEST(CASGCHoldGrammar, RebuildCarriesMatchingHoldAndDropsAbsentLife) ASSERT_TRUE(rebuilt.has_value()); const auto rediscovered = rebuilt->ref_lives.find(life_id); ASSERT_NE(rediscovered, rebuilt->ref_lives.end()); - EXPECT_EQ(rediscovered->second.coverage.classification, 4); + EXPECT_EQ(rediscovered->second.coverage.classification, CoverageClass::Clamped); ASSERT_TRUE(rediscovered->second.coverage.hold.has_value()); EXPECT_EQ(*rediscovered->second.coverage.hold, plantedHold()); EXPECT_FALSE(rebuilt->ref_lives.contains(absent_life_id)); @@ -1305,7 +1350,7 @@ TEST(CASGCHoldGrammar, RebuildStepsDownPastACrashedNewestGenerationToTheSealBelo writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState after_first = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState after_first = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); const uint64_t older_generation = after_first.snap_generation; const uint64_t older_attempt = after_first.snap_attempt; const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); @@ -1313,27 +1358,27 @@ TEST(CASGCHoldGrammar, RebuildStepsDownPastACrashedNewestGenerationToTheSealBelo publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, 2}); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState after_second = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState after_second = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); ASSERT_GT(after_second.snap_generation, older_generation) << "the fixture needs two generations"; /// The older generation is the one holding the pool's durable hold. mutateSealAt(*backend, layout, older_generation, older_attempt, [&](CasFoldSeal & seal) { RefCoverage & cov = seal.ref_lives.at(life_id).coverage; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = plantedHold(); }); /// THE CRASH: the newest generation's run objects are there, its seal never got written. Then /// `gc/state` is lost, which is this path's whole premise. const String newest_seal = layout.foldSealKey(after_second.snap_generation, after_second.snap_attempt); - const HeadResult seal_head = backend->head(newest_seal); - ASSERT_TRUE(seal_head.exists); - ASSERT_EQ(backend->deleteExact(newest_seal, seal_head.token).kind, DeleteOutcome::Kind::Deleted); - ASSERT_FALSE(backend->list(layout.gcGenPrefix(after_second.snap_generation), "", 1).keys.empty()) - << "the crashed generation must still hold objects, or it is not the shape being modelled"; - const HeadResult sh = backend->head(layout.gcStateKey()); - ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + headThenRemove(*backend, newest_seal); + { + OperationForTest op(*backend); + ASSERT_FALSE((*op).list(layout.gcGenPrefix(after_second.snap_generation), "", 1, Retry::once()).keys.empty()) + << "the crashed generation must still hold objects, or it is not the shape being modelled"; + } + headThenRemove(*backend, layout.gcStateKey()); Gc gc2(store, hexToU128("0000000000000000000000000000000c")); const RebuildReport rep = gc2.rebuildBaseline(/*force=*/false); @@ -1367,12 +1412,10 @@ TEST(CASGCHoldGrammar, RebuildRefusesWithAMissingPriorSeal) Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState st = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); ASSERT_GT(st.snap_generation, 0u); const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); - const HeadResult sh = backend->head(seal_key); - ASSERT_TRUE(sh.exists); - ASSERT_EQ(backend->deleteExact(seal_key, sh.token).kind, DeleteOutcome::Kind::Deleted); + headThenRemove(*backend, seal_key); /// FORCE does not buy past it either: force means "rebuild deliberately", never "drop the holds". for (const bool force : {false, true}) @@ -1381,7 +1424,7 @@ TEST(CASGCHoldGrammar, RebuildRefusesWithAMissingPriorSeal) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.rebuildBaseline(force); }); } - const GcState after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState after = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); EXPECT_EQ(after.snap_generation, st.snap_generation) << "a refused rebuild adopts nothing"; } @@ -1396,10 +1439,9 @@ TEST(CASGCHoldGrammar, RebuildRefusesWithAnUndecodablePriorSeal) Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState st = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); - backend->putOverwrite(seal_key, "{\"type\":\"cas_fold_seal\",\"v\":4}\nthis is not a seal body\n", - backend->head(seal_key).token); + headThenReplace(*backend, seal_key, "{\"type\":\"cas_fold_seal\",\"v\":1}\nthis is not a seal body\n"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.rebuildBaseline(/*force=*/true); }); } @@ -1433,14 +1475,12 @@ TEST(CASGCHoldGrammar, RebuildWithLostStateStillCarriesHoldsFromTheNewestSeal) mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) { RefCoverage & cov = seal.ref_lives.at(life_id).coverage; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = plantedHold(); }); /// The pointer vanishes; every seal object survives. - const HeadResult sh = backend->head(layout.gcStateKey()); - ASSERT_TRUE(sh.exists); - ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + headThenRemove(*backend, layout.gcStateKey()); Gc gc2(store, hexToU128("00000000000000000000000000000009")); const RebuildReport rep = gc2.rebuildBaseline(/*force=*/false); @@ -1468,12 +1508,10 @@ TEST(CASGCHoldGrammar, RebuildRefusesWhenTheNewestSealIsUnreadableAndTheStateIsL Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState st = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); const String seal_key = layout.foldSealKey(st.snap_generation, st.snap_attempt); - backend->putOverwrite(seal_key, "{\"type\":\"cas_fold_seal\",\"v\":4}\nthis is not a seal body\n", - backend->head(seal_key).token); - const HeadResult sh = backend->head(layout.gcStateKey()); - ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + headThenReplace(*backend, seal_key, "{\"type\":\"cas_fold_seal\",\"v\":1}\nthis is not a seal body\n"); + headThenRemove(*backend, layout.gcStateKey()); Gc gc2(store, hexToU128("0000000000000000000000000000000a")); for (const bool force : {false, true}) @@ -1502,18 +1540,22 @@ TEST(CASGCHoldGrammar, RebuildRefusesWhenANarrowProbeFindsASealAboveTheListingMa class BroadListHoleBackend : public InMemoryBackend { public: + /// Unhide the name the primitive override below would otherwise shadow. + using InMemoryBackend::list; + String hide_under_prefix; String hidden_key_infix; size_t holes_served = 0; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, + TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); if (prefix != hide_under_prefix) return page; const size_t before = page.keys.size(); std::erase_if(page.keys, - [&](const ListedKey & k) { return k.key.find(hidden_key_infix) != String::npos; }); + [&](const RawListedKey & k) { return k.key.find(hidden_key_infix) != String::npos; }); if (page.keys.size() != before) ++holes_served; return page; @@ -1531,13 +1573,13 @@ TEST(CASGCHoldGrammar, RebuildRefusesWhenANarrowProbeFindsASealAboveTheListingMa publishAt(*backend, layout, ns, RefTxnId{1, 2}, "ref_2", 2, DB::UInt128(2)); ASSERT_TRUE(gc.runRegularRound().acquired_lease); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState st = decodeGcState(readAt(*backend, layout.gcStateKey())->bytes); ASSERT_GT(st.snap_generation, 1u) << "the fixture needs a newer generation to hide"; const UInt128 life_id = catalogLifeIdForTest(*backend, layout, ns); mutateAdoptedSeal(*backend, layout, [&](CasFoldSeal & seal) { RefCoverage & cov = seal.ref_lives.at(life_id).coverage; - cov.classification = 4; + cov.classification = CoverageClass::Clamped; cov.hold = plantedHold(); }); @@ -1545,8 +1587,7 @@ TEST(CASGCHoldGrammar, RebuildRefusesWhenANarrowProbeFindsASealAboveTheListingMa const String gen_prefix = layout.gcGenPrefix(0); backend->hide_under_prefix = gen_prefix.substr(0, gen_prefix.size() - 2); /// ".../gc/gen/" backend->hidden_key_infix = layout.gcGenPrefix(st.snap_generation); - const HeadResult sh = backend->head(layout.gcStateKey()); - ASSERT_EQ(backend->deleteExact(layout.gcStateKey(), sh.token).kind, DeleteOutcome::Kind::Deleted); + headThenRemove(*backend, layout.gcStateKey()); Gc gc2(store, hexToU128("0000000000000000000000000000000b")); for (const bool force : {false, true}) @@ -1557,7 +1598,7 @@ TEST(CASGCHoldGrammar, RebuildRefusesWhenANarrowProbeFindsASealAboveTheListingMa ASSERT_GT(backend->holes_served, 0u) << "the broad listing never actually lied"; /// Nothing was adopted: the refusal fires before the lease, so the pool is exactly as it was. - EXPECT_FALSE(backend->head(layout.gcStateKey()).exists) + EXPECT_FALSE(existsAt(*backend, layout.gcStateKey())) << "a refused rebuild must not mint a baseline, nor a bootstrap body"; } @@ -1577,7 +1618,7 @@ TEST(CASGCHoldGrammar, RebuildProceedsOnAPoolThatNeverSealedABaselineAndCountsTh /// No round has run, so there is no `gc/state` and no seal — only owner state to rebuild from. publishAt(*backend, layout, ns, RefTxnId{1, 1}, "ref_1", 1, DB::UInt128(1), /*birth=*/true); writeCommittedCkptAt(*backend, layout, ns, RefTxnId{1, 1}); - ASSERT_FALSE(backend->head(layout.gcStateKey()).exists); + ASSERT_FALSE(existsAt(*backend, layout.gcStateKey())); using ProfileEvents::global_counters; const auto virgin_before = global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration].load(); diff --git a/src/Disks/tests/gtest_cas_gc_key_reader.cpp b/src/Disks/tests/gtest_cas_gc_key_reader.cpp new file mode 100644 index 000000000000..4f3938f01698 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_key_reader.cpp @@ -0,0 +1,134 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +/// A reader hands a sequential walk its next object and lets the walk say which keys it will want +/// (hint) and which hinted keys it will never take (discard). The inline reader ignores hints; the +/// read-ahead reader turns them into worker requests and counts a discarded one as wasted at once. + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} + +namespace ProfileEvents +{ + extern const Event CASGCReadAheadWasted; + extern const Event CASGCReadAheadHit; + extern const Event CASGCReadAheadMiss; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::openRequestsForTest; + +namespace +{ + +struct Rig +{ + std::shared_ptr backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ThreadPool pool{CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, + CurrentMetrics::LocalThreadScheduled, /*max_threads*/ 4, /*max_free_threads*/ 4, /*queue_size*/ 0}; + + void put(const String & key, const String & bytes) + { + ASSERT_TRUE(std::holds_alternative(op.create(key, bytes, Retry::once()))) << key; + } +}; + +uint64_t wasted() +{ + return ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load(); +} + +} + +TEST(CASGCKeyReader, DiscardCountsWastedAtOnceAndALaterTakeReadsInline) +{ + Rig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + ReadAheadKeyReader reader(reads); + rig.backend->resetCounts(); + + const uint64_t wasted_before = wasted(); + reader.hint("k1"); + EXPECT_EQ(reader.pending(), 1u); + reader.discard("k1"); + EXPECT_EQ(reader.pending(), 0u); + EXPECT_EQ(wasted() - wasted_before, 1u); + + const auto got = reader.take("k1"); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, "one"); + EXPECT_EQ(rig.backend->getCount("k1"), 2u) << "the discarded request and the inline one"; +} + +TEST(CASGCKeyReader, DiscardOfAnUnhintedKeyIsANoOp) +{ + Rig rig; + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + ReadAheadKeyReader reader(reads); + const uint64_t wasted_before = wasted(); + reader.discard("never-hinted"); + EXPECT_EQ(wasted() - wasted_before, 0u); + EXPECT_EQ(reader.pending(), 0u); +} + +TEST(CASGCKeyReader, DiscardSwallowsAWorkerFailureThatATakeWouldRethrow) +{ + Rig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + ReadAheadKeyReader reader(reads); + + rig.backend->failNextReadWith("k1", std::make_exception_ptr(std::runtime_error("injected worker fault"))); + reader.hint("k1"); + EXPECT_NO_THROW(reader.discard("k1")); + + rig.backend->failNextReadWith("k1", std::make_exception_ptr(std::runtime_error("injected worker fault"))); + reader.hint("k1"); + EXPECT_THROW(static_cast(reader.take("k1")), std::runtime_error); +} + +TEST(CASGCKeyReader, InlineReaderHintsNothingAndReadsOnTake) +{ + Rig rig; + rig.put("k1", "one"); + InlineKeyReader reader(rig.op); + rig.backend->resetCounts(); + EXPECT_EQ(reader.window(), 0u); + reader.hint("k1"); + EXPECT_EQ(rig.backend->getCount("k1"), 0u); + EXPECT_EQ(reader.pending(), 0u); + const auto got = reader.take("k1"); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, "one"); + EXPECT_EQ(rig.backend->getCount("k1"), 1u); + reader.discard("k1"); +} + +TEST(CASGCKeyReader, ReadAheadReaderWindowAndPendingAreTheReadAheads) +{ + Rig rig; + GcReadAhead reads(rig.op, rig.requests, rig.pool, 8); + ReadAheadKeyReader reader(reads); + EXPECT_EQ(reader.window(), reads.window()); + EXPECT_EQ(reader.window(), 32u); + rig.put("a", "1"); + reader.hint("a"); + EXPECT_EQ(reader.pending(), reads.pending()); + static_cast(reader.take("a")); +} diff --git a/src/Disks/tests/gtest_cas_gc_leak.cpp b/src/Disks/tests/gtest_cas_gc_leak.cpp index a65b225b8f9a..e6353d962261 100644 --- a/src/Disks/tests/gtest_cas_gc_leak.cpp +++ b/src/Disks/tests/gtest_cas_gc_leak.cpp @@ -47,9 +47,9 @@ PoolPtr openTestPool(std::shared_ptr & out_backend) /// (condemn -> graduate -> delete) is in flight while this is true. bool anyRetiredPending(const PoolPtr & s) { - /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// Condemned state rides the adopted fold seal's RunMarker::Condemned rows, not a /// separate retired list — reconstruct the in-flight set from the seal. - return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); + return DB::Cas::tests::anyCondemnedInSeal(*s->poolBackendPtr(), s->layout()); } /// Drive regular GC to a fixpoint. A condemned blob is not deleted in the round that folds its removal: @@ -126,13 +126,15 @@ ManifestId publishOneBlobPart( /// HEADs the object key, never the Pool's manifest decode cache). bool blobPresent(const std::shared_ptr & b, const Layout & layout, const String & payload) { - return b->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))})).exists; + DB::Cas::tests::OperationForTest op(*b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}), Retry::once()).has_value(); } /// Whether a manifest body object is present in the backend. bool manifestPresent(const std::shared_ptr & b, const Layout & layout, const ManifestId & id) { - return b->head(layout.manifestKey(id)).exists; + DB::Cas::tests::OperationForTest op(*b); + return (*op).head(layout.manifestKey(id), Retry::once()).has_value(); } /// Replace the existing ref with partB through the real durable-precommit writer sequence. The @@ -171,8 +173,11 @@ FsckReport displaceAndGc( /// Publish partB's full closure and atomically repoint the ref from partA to partB. const ManifestId part_b = publishPartBReplacement(s, ns, ref, "data-B", "mark-B"); - EXPECT_TRUE(b->head(s->layout().manifestKey(part_a)).exists) - << "partA manifest body must still be present so GC can read its -1 edges at removal-fold"; + { + DB::Cas::tests::OperationForTest op(*b); + EXPECT_TRUE((*op).head(s->layout().manifestKey(part_a), Retry::once()).has_value()) + << "partA manifest body must still be present so GC can read its -1 edges at removal-fold"; + } const auto resolved = s->resolveRef(ns, ref); EXPECT_TRUE(resolved.has_value()); @@ -314,8 +319,9 @@ TEST(CASGCLeak, ResurrectReplacedIncarnationReclaimed) /// 1. Publish ref r1 -> token A referenced; capture A. publishOneBlobPart(s, ns, "r1", P); - const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hA.exists); + DB::Cas::tests::OperationForTest op(*b); + const auto hA = (*op).head(s->layout().blobKey(idOf(P)), Retry::once()); + ASSERT_TRUE(hA.has_value()); /// 2. Drop r1 -> A dereferenced. s->dropRef(ns, "r1"); @@ -334,9 +340,9 @@ TEST(CASGCLeak, ResurrectReplacedIncarnationReclaimed) /// 4. RESURRECT: a fresh build dedup-hits P; putBlob sees A condemned -> re-uploads a DISTINCT /// incarnation B at the same content-addressed key (INV-1 revival-from-source). publishOneBlobPart(s, ns, "r2", P); - const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hB.exists); - ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a new incarnation token B"; + const auto hB = (*op).head(s->layout().blobKey(idOf(P)), Retry::once()); + ASSERT_TRUE(hB.has_value()); + ASSERT_NE(hB->etag, hA->etag) << "republication must mint a new incarnation token B"; /// 5. Drop r2 -> B dereferenced. s->dropRef(ns, "r2"); @@ -423,17 +429,18 @@ TEST(CASGCLeak, ResurrectReplacedTokenIsCondemnedInMeta) /// 1. Publish ref r1 -> token A referenced; capture A, then drop it and condemn via ONE GC round. publishOneBlobPart(s, ns, "r1", P); - const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hA.exists); + DB::Cas::tests::OperationForTest op(*b); + const auto hA = (*op).head(s->layout().blobKey(idOf(P)), Retry::once()); + ASSERT_TRUE(hA.has_value()); s->dropRef(ns, "r1"); s->renewWatermarkOnce(); /// advance the floor so A is not spared as in-flight gc.runRegularRound(); /// 2. RESURRECT: r2 dedup-hits P while A is condemned -> mints a fresh incarnation B. publishOneBlobPart(s, ns, "r2", P); - const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hB.exists); - ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a distinct incarnation"; + const auto hB = (*op).head(s->layout().blobKey(idOf(P)), Retry::once()); + ASSERT_TRUE(hB.has_value()); + ASSERT_NE(hB->etag, hA->etag) << "republication must mint a distinct incarnation"; s->dropRef(ns, "r2"); s->renewWatermarkOnce(); diff --git a/src/Disks/tests/gtest_cas_gc_log.cpp b/src/Disks/tests/gtest_cas_gc_log.cpp index 5dbd654343d9..d70cc9aa78d8 100644 --- a/src/Disks/tests/gtest_cas_gc_log.cpp +++ b/src/Disks/tests/gtest_cas_gc_log.cpp @@ -8,6 +8,9 @@ #include +#include +#include +#include #include #include @@ -26,6 +29,8 @@ namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; + extern const int CORRUPTED_DATA; + extern const int NETWORK_ERROR; } using namespace DB::Cas; @@ -180,25 +185,29 @@ namespace class ThrowingBackend : public InMemoryBackend { public: - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Unhide the names the primitive overrides below would otherwise shadow. + using InMemoryBackend::head; + using InMemoryBackend::list; + + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (arm) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend list failure"); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (arm) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend get failure"); - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - HeadResult head(const String & key) override + std::optional head(const String & key, TransportAccess & access) override { if (arm) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected backend head failure"); - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } /// Armed only after Pool::open, so opening (which reads/initialises gc state) succeeds. @@ -289,7 +298,10 @@ TEST(CASGCLog, AbortedFinishOnThrowingRound) ASSERT_EQ(round_rows.size(), 2u) << "a throwing round still emits a Start and a (Aborted) Finish"; EXPECT_EQ(round_rows[0].event_type, Rec::EventType::Start); EXPECT_EQ(round_rows[1].event_type, Rec::EventType::Finish); - EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Failed); + EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Failed) + << "BAD_ARGUMENTS is not on the transient list, so the row must read as a real failure"; + EXPECT_EQ(round_rows[1].error_code, DB::ErrorCodes::BAD_ARGUMENTS) + << "the Finish row must carry the structured exception code, not only the message text"; EXPECT_FALSE(round_rows[1].error.empty()) << "a failed Finish must carry the exception text"; EXPECT_EQ(round_rows[1].disk_name, "ca"); EXPECT_FALSE(round_rows[1].gc_id.empty()); @@ -299,6 +311,252 @@ TEST(CASGCLog, AbortedFinishOnThrowingRound) << "every row of a FAILED round must still correlate through round_id"; } +/// A round that dies with a TRANSIENT code -- the backend was unreachable, timed out, or another +/// actor moved shared state -- must be classified `Aborted`, not `Failed`: the next scheduled round +/// is the retry and nothing durable is wrong. The classifier keys on the exception CODE +/// (`isTransientGcRoundError`), never on message wording. +class NetworkThrowingBackend : public InMemoryBackend +{ +public: + /// Unhide the names the primitive overrides below would otherwise shadow. + using InMemoryBackend::list; + + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + if (arm) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected backend outage"); + return InMemoryBackend::list(prefix, cursor, limit, access); + } + std::atomic arm{false}; +}; + +TEST(CASGCLog, TransientThrowIsClassifiedAborted) +{ + auto backend = std::make_shared(); + /// A PERSISTENT transient fault is reissued for the whole retry window, so the window has to run + /// on a clock this test advances -- otherwise one read spends ninety real seconds. Heap-owned, not + /// a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local -- even an already-atomic one -- + /// would dangle once the frame returns. + auto engine_now_ms = std::make_shared>(0); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + store->setCasRequestNowFnForTest([engine_now_ms] { return engine_now_ms->fetch_add(10'000) + 10'000; }); + store->setCasRetrySleepForTest([](uint64_t) {}); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + backend->arm.store(true); + EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size(), 2u); + EXPECT_EQ(round_rows[1].event_type, Rec::EventType::Finish); + EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Aborted) + << "NETWORK_ERROR names a transient condition; the row must not read as a GC defect"; + EXPECT_EQ(round_rows[1].error_code, DB::ErrorCodes::NETWORK_ERROR); + EXPECT_FALSE(round_rows[1].error.empty()); +} + +/// The classifier itself, pinned direct: the transient list is exact and everything else fails closed. +TEST(CASGCLog, TransientErrorClassifierFailsClosed) +{ + EXPECT_TRUE(DB::Cas::isTransientGcRoundError(DB::ErrorCodes::NETWORK_ERROR)); + EXPECT_FALSE(DB::Cas::isTransientGcRoundError(DB::ErrorCodes::BAD_ARGUMENTS)); + EXPECT_FALSE(DB::Cas::isTransientGcRoundError(0)); + EXPECT_FALSE(DB::Cas::isTransientGcRoundError(-1)); +} + +/// A backend that REFUSES the round-closing `gc/state` write -- the one that advances +/// `snap_generation` -- and lets every other write through, the lease acquire/renew included. The round +/// therefore does all of its pre-CAS work, condemning the dropped part included, and dies at +/// `round_commit`. +/// +/// A refusal and not a throw, for two reasons. A thrown transport error is an ambiguity the engine +/// settles by an exact read and, while the precondition it named is unmoved, reissues to the policy +/// deadline -- so an armed fault would spend the whole retry window and end as a transport give-up. And +/// the arm is keyed on the generation rather than on a call count, because the acquire on a fresh pool +/// is an UNCONDITIONAL create that no count of conditional writes can see. +class StateCommitRefusingBackend : public InMemoryBackend +{ +public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + if (arm.load() && expected_value && key.ends_with("gc/state")) + { + const auto stored = InMemoryBackend::read(key, access); + if (stored + && decodeGcState(bytes).snap_generation > decodeGcState(stored->bytes).snap_generation) + { + arm.store(false); + /// A store refuses a precondition only when the object moved, so move it: the same + /// bytes under a fresh incarnation is the smallest faithful move, and it leaves the + /// content alone so the assertions below stay about this round. + (void)InMemoryBackend::write(key, stored->bytes, stored->value, access); + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + std::atomic arm{false}; +}; + +/// The Finish row of a THROWING round must still carry the counters of everything the round did +/// before it died. Before this existed, the exception path emitted a row with `round = 0` and every +/// counter zero, so a round that condemned entries and then lost its commit CAS was +/// indistinguishable from a round that never got past the lease. +TEST(CASGCLog, AbortedFinishCarriesProgressiveCounters) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const RootNamespace ns{"srv1/tbl"}; + + publishPart(store, ns.string(), "all_0_0_0", "hello-progressive-counters"); + store->dropRef(ns, "all_0_0_0"); + store->renewWatermarkOnce(); + + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + backend->arm.store(true); + EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size(), 2u); + const Rec & fin = round_rows[1]; + EXPECT_EQ(fin.outcome, Rec::Outcome::Aborted); + /// A refused precondition is settled by one exact read and reported as a conflict, so the round + /// is dropped whole and names the conflict rather than a transport error. + EXPECT_EQ(fin.error_code, DB::ErrorCodes::ABORTED); + EXPECT_EQ(fin.round, 0u) << "the commit never landed, so the round number must stay unstamped"; + EXPECT_GT(fin.candidates_marked + fin.entries_condemned + fin.entries_graduated + + fin.entries_redeleted + fin.objects_deleted + fin.fence_outs, 0u) + << "the pre-CAS work the round performed must survive into its failure row"; +} + +/// The pacing loop drops leadership only on a NON-transient round failure. A transient failure +/// (backend outage class) keeps `i_am_leader` set, so the advisory heartbeat keeps pulsing and a +/// live leader blocked on a flaky store is not deposed -- dropping the flag on every failure was +/// half of the dead-leader signature (`!incumbent_renewed && !hb_alive`) and produced leadership +/// ping-pong under backend fault windows. A non-transient failure must still clear the flag: a +/// logic-broken leader has to stay depositable. +class ModalThrowingBackend : public InMemoryBackend +{ +public: + /// Unhide the names the primitive overrides below would otherwise shadow. + using InMemoryBackend::list; + + enum Mode : int { Off = 0, Transient = 1, Logic = 2 }; + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + const int m = mode.load(); + if (m == Transient) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected backend outage"); + if (m == Logic) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected logic failure"); + return InMemoryBackend::list(prefix, cursor, limit, access); + } + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + if (key.ends_with("gc/hb")) + ++hb_puts; + return InMemoryBackend::write(key, bytes, expected_value, access); + } + std::atomic mode{Off}; + std::atomic hb_puts{0}; +}; + +TEST(CASGCScheduler, TransientRoundFailureKeepsLeadershipAndHeartbeat) +{ + auto backend = std::make_shared(); + /// See `TransientThrowIsClassifiedAborted`: the transient mode is persistent while it is armed, so + /// the retry window runs on a clock this test advances. Heap-owned, not a plain local: the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local -- even an already-atomic one -- would dangle once the frame returns. + auto engine_now_ms = std::make_shared>(0); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + store->setCasRequestNowFnForTest([engine_now_ms] { return engine_now_ms->fetch_add(10'000) + 10'000; }); + store->setCasRetrySleepForTest([](uint64_t) {}); + + std::mutex rows_mutex; + std::condition_variable rows_cv; + std::vector finishes; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) + { + if (r.event_type != Rec::EventType::Finish) + return; + std::lock_guard g(rows_mutex); + finishes.push_back(r); + rows_cv.notify_all(); + }); + + const auto wait_for_finish = [&](size_t count) -> Rec + { + std::unique_lock lock(rows_mutex); + const bool ok = rows_cv.wait_for(lock, std::chrono::seconds(30), [&] { return finishes.size() >= count; }); + EXPECT_TRUE(ok) << "timed out waiting for Finish row #" << count; + return finishes.at(count - 1); + }; + /// Bounded poll for an ASYNC flag change. The loop stores `i_am_leader` after `runRoundLogged` + /// returns (after the Finish row was emitted), so the row alone is not a happens-before for the + /// flag -- poll to the expected value instead of asserting a racy instantaneous read. + const auto poll_leader = [&](bool expected) -> bool + { + for (int i = 0; i < 3000; ++i) + { + if (sched.gcHealth().is_leader == expected) + return true; + std::this_thread::sleep_for(std::chrono::milliseconds(10)); + } + return sched.gcHealth().is_leader == expected; + }; + + sched.start(); + sched.requestRoundSoon(); + const Rec first = wait_for_finish(1); + EXPECT_TRUE(first.outcome == Rec::Outcome::Success || first.outcome == Rec::Outcome::Deferred) + << "outcome=" << static_cast(first.outcome); + EXPECT_TRUE(poll_leader(true)) << "a successful round must establish leadership"; + + backend->mode.store(ModalThrowingBackend::Transient); + sched.requestRoundSoon(); + const Rec aborted = wait_for_finish(2); + EXPECT_EQ(aborted.outcome, Rec::Outcome::Aborted); + /// Leadership kept => the advisory heartbeat keeps pulsing. Waiting for a NEW pulse after the + /// failed round is the happens-after proof that the flag survived; with the flag dropped the + /// heartbeat loop skips every pulse until the next successful round, and this wait times out. + const uint64_t hb_before = backend->hb_puts.load(); + bool pulsed = false; + for (int i = 0; i < 3000 && !pulsed; ++i) + { + pulsed = backend->hb_puts.load() > hb_before; + if (!pulsed) + std::this_thread::sleep_for(std::chrono::milliseconds(10)); + } + EXPECT_TRUE(pulsed) << "a transient round failure must not silence the advisory heartbeat"; + EXPECT_TRUE(sched.gcHealth().is_leader) << "a transient round failure must not drop leadership"; + + backend->mode.store(ModalThrowingBackend::Logic); + sched.requestRoundSoon(); + const Rec failed = wait_for_finish(3); + EXPECT_EQ(failed.outcome, Rec::Outcome::Failed); + EXPECT_TRUE(poll_leader(false)) << "a non-transient round failure must still surrender leadership"; + + backend->mode.store(ModalThrowingBackend::Off); + sched.stop(); +} + /// Every row of one round -- its Start, each of its Phase rows, and its Finish -- carries the SAME /// non-empty `round_id`, and two rounds carry DIFFERENT ones. That is the property the column exists /// for: `round` is 0 on Start, is only known after the round's single `gc/state` CAS, and is absent on a @@ -487,3 +745,86 @@ TEST(CASGCHealth, ReflectsLeadershipAndPendingReclaim) EXPECT_EQ(h1.wedged_namespace_count, 0u); EXPECT_LT(h1.last_success_age_seconds, 60u); } + +namespace +{ + +/// A backend that arms the pool's teardown the moment a chosen key has been read -- after the read +/// returned, before the round can act on it -- so the arm lands mid-round at a known point. +class ArmAfterReadBackend : public InMemoryBackend +{ +public: + using InMemoryBackend::read; + + std::optional read(const String & key, TransportAccess & access) override + { + auto result = InMemoryBackend::read(key, access); + if (key == arm_key && on_read) + on_read(); + return result; + } + + String arm_key; + std::function on_read; +}; + +} + +/// `Stopped` is a transient failure observed after the pool's teardown began -- a correlation the row +/// records honestly. The arm lands right after the lease read; the round's next request is refused by +/// the open plane's fence, which the engine reports like any lost fence (a transient code). +TEST(CASGCLog, TransientFailureAfterTeardownBeganIsStopped) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + store->setCasRetrySleepForTest([](uint64_t) {}); + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + backend->arm_key = store->layout().gcStateKey(); + backend->on_read = [&store] { store->beginTeardown(); }; + EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size(), 2u); + EXPECT_EQ(round_rows[1].event_type, Rec::EventType::Finish); + EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Stopped) + << "a transient refusal after the arm is the teardown cutting the round short, not an incident"; + EXPECT_EQ(round_rows[1].error_code, DB::ErrorCodes::NETWORK_ERROR); + EXPECT_FALSE(round_rows[1].error.empty()); +} + +/// The rule is fail-closed: a non-transient failure that coincides with the arm stays `Failed`. An +/// undecodable `gc/state` throws `CORRUPTED_DATA` out of the lease phase after the very read that arms. +TEST(CASGCLog, NonTransientFailureCoincidingWithTeardownStaysFailed) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + store->setCasRetrySleepForTest([](uint64_t) {}); + std::vector rows; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) { rows.push_back(r); }); + + { + /// `gc/state` does not exist until a round writes it, so the undecodable value is planted, + /// not substituted: the lease phase's own decode is what must fail. + DB::Cas::tests::OperationForTest raw_op(*backend); + const auto current = (*raw_op).read(store->layout().gcStateKey(), Retry::once()); + const WriteResult planted = current + ? (*raw_op).replace(store->layout().gcStateKey(), "not a gc state", current->etag, Retry::once()) + : (*raw_op).create(store->layout().gcStateKey(), "not a gc state", Retry::once()); + ASSERT_TRUE(std::holds_alternative(planted)); + } + backend->arm_key = store->layout().gcStateKey(); + backend->on_read = [&store] { store->beginTeardown(); }; + EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + + const std::vector round_rows = roundRowsOnly(rows); + ASSERT_EQ(round_rows.size(), 2u); + EXPECT_EQ(round_rows[1].outcome, Rec::Outcome::Failed) + << "a bug that coincides with a restart is not masked as Stopped"; + EXPECT_EQ(round_rows[1].error_code, DB::ErrorCodes::CORRUPTED_DATA); +} diff --git a/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp b/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp index 158040be38c9..309d20f8de89 100644 --- a/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp +++ b/src/Disks/tests/gtest_cas_gc_maintenance_state_format.cpp @@ -1,4 +1,5 @@ #include "cas_test_helpers.h" +#include "cas_format_test_battery.h" #include #include #include @@ -17,20 +18,31 @@ namespace class FailingMaintenanceReadBackend : public InMemoryBackend { public: - std::optional get(const String &, Range) override + std::optional read(const String &, DB::Cas::TransportAccess &) override { throw std::runtime_error("injected maintenance read failure"); } }; } +CAS_BATTERY_COVERS(GcMaintenanceState); + +TEST(CASFormatBattery, GcMaintenanceState) +{ + GcMaintenanceState state{.janitor_cursor = "cas/ns/a"}; + runFormatBattery({FormatId::GcMaintenanceState, + [&] { return sealObject(FormatId::GcMaintenanceState, encodeGcMaintenanceState(state)); }, + [](std::string_view s) { decodeGcMaintenanceState(std::string(openObject(FormatId::GcMaintenanceState, s))); }, + currentFormatHeader("cas_gc_maintenance_state") + "{\"janitor_cursor\":\"cas/ns/a\"}\n"}); +} + TEST(CASGCMaintenanceStateFormat, RegistryLayoutAndCanonicalCodec) { EXPECT_EQ(static_cast(FormatId::GcMaintenanceState), 25); const auto points = changePoints(FormatId::GcMaintenanceState); ASSERT_EQ(points.size(), 1u); - EXPECT_EQ(points[0].generation, 7); - EXPECT_EQ(points[0].min_reader, 7); + EXPECT_EQ(points[0].generation, 1); + EXPECT_EQ(points[0].min_reader, 1); const FormatTraits & traits = traitsFor(FormatId::GcMaintenanceState); EXPECT_EQ(traits.type, "cas_gc_maintenance_state"); EXPECT_EQ(traits.family, TextFamily::Control); @@ -48,7 +60,7 @@ TEST(CASGCMaintenanceStateFormat, RegistryLayoutAndCanonicalCodec) const GcMaintenanceState empty; EXPECT_EQ(encodeGcMaintenanceState(empty), fmt::format( - "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"cur\":\"\"}}\n", currentCompatibilityVersion())); + "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"janitor_cursor\":\"\"}}\n", currentCompatibilityVersion())); const GcMaintenanceState state{.janitor_cursor = R"(cas/ns/a/"quoted"\\next)"}; EXPECT_EQ(decodeGcMaintenanceState(encodeGcMaintenanceState(state)), state); } @@ -57,28 +69,28 @@ TEST(CASGCMaintenanceStateFormat, RejectsMalformedAndBoundsCursor) { const auto bad = [](std::string_view body) { - return "{\"type\":\"cas_gc_maintenance_state\",\"v\":7}\n" + String(body); + return "{\"type\":\"cas_gc_maintenance_state\",\"v\":1}\n" + String(body); }; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)decodeGcMaintenanceState(bad("{}\n")); }); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\",\"cur\":\"b\"}\n")); }); + [&] { (void)decodeGcMaintenanceState(bad("{\"janitor_cursor\":\"a\",\"janitor_cursor\":\"b\"}\n")); }); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\",\"extra\":1}\n")); }); + [&] { (void)decodeGcMaintenanceState(bad("{\"janitor_cursor\":\"a\",\"extra\":1}\n")); }); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { (void)decodeGcMaintenanceState(bad("{\"cur\":\"a\"}\nx")); }); + [&] { (void)decodeGcMaintenanceState(bad("{\"janitor_cursor\":\"a\"}\nx")); }); const GcMaintenanceState at_limit{.janitor_cursor = String(kMaxGcMaintenanceCursorBytes, 'x')}; EXPECT_EQ(decodeGcMaintenanceState(encodeGcMaintenanceState(at_limit)), at_limit); const GcMaintenanceState over_limit{.janitor_cursor = String(kMaxGcMaintenanceCursorBytes + 1, 'x')}; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] { (void)encodeGcMaintenanceState(over_limit); }); - const String raw = "{\"type\":\"cas_gc_maintenance_state\",\"v\":7}\n{\"cur\":\"" + over_limit.janitor_cursor + "\"}\n"; + const String raw = "{\"type\":\"cas_gc_maintenance_state\",\"v\":1}\n{\"janitor_cursor\":\"" + over_limit.janitor_cursor + "\"}\n"; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)decodeGcMaintenanceState(raw); }); - String oversized = R"({"type":"cas_gc_maintenance_state","v":7,"pad":")"; + String oversized = R"({"type":"cas_gc_maintenance_state","v":1,"pad":")"; oversized.append(448 * 1024, 'x'); - oversized += "\"}\n{\"cur\":\""; + oversized += "\"}\n{\"janitor_cursor\":\""; oversized.append(kMaxGcMaintenanceCursorBytes, 'y'); oversized += "\"}\n"; ASSERT_GT(oversized.size(), traitsFor(FormatId::GcMaintenanceState).object_cap); @@ -88,115 +100,142 @@ TEST(CASGCMaintenanceStateFormat, RejectsMalformedAndBoundsCursor) TEST(CASGCMaintenanceState, ReadsAndCasWithoutAdoptingConflicts) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const String key = layout.gcMaintenanceStateKey(); - const GcMaintenanceReadResult absent = readGcMaintenanceState(backend, layout); + auto op = requests.admit(); + + const GcMaintenanceReadResult absent = readGcMaintenanceState(op, layout); EXPECT_EQ(absent.status, GcMaintenanceReadStatus::Absent); EXPECT_FALSE(absent.state); - EXPECT_FALSE(absent.token); + EXPECT_FALSE(absent.etag); const GcMaintenanceState first{.janitor_cursor = "cas/ns/first"}; - const GcMaintenanceCasResult created = casGcMaintenanceState(backend, layout, std::nullopt, first); - EXPECT_EQ(created.outcome, GcMaintenanceCasOutcome::Committed); - const GcMaintenanceReadResult valid = readGcMaintenanceState(backend, layout); + ASSERT_TRUE(std::holds_alternative( + casGcMaintenanceState(op, layout, std::nullopt, first, Retry::standard()))); + const GcMaintenanceReadResult valid = readGcMaintenanceState(op, layout); ASSERT_EQ(valid.status, GcMaintenanceReadStatus::Valid); - ASSERT_TRUE(valid.token); + ASSERT_TRUE(valid.etag); ASSERT_TRUE(valid.state); EXPECT_EQ(*valid.state, first); - const GcMaintenanceCasResult advanced = casGcMaintenanceState(backend, layout, valid.token, - GcMaintenanceState{.janitor_cursor = "cas/ns/advanced"}); - ASSERT_EQ(advanced.outcome, GcMaintenanceCasOutcome::Committed); - const auto advanced_body = backend.get(key); - ASSERT_TRUE(advanced_body); - - ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), advanced_body->token).outcome, - CasOutcome::Committed); - const GcMaintenanceCasResult conflict = casGcMaintenanceState(backend, layout, valid.token, - GcMaintenanceState{.janitor_cursor = "loser"}); - EXPECT_EQ(conflict.outcome, GcMaintenanceCasOutcome::Conflict); - EXPECT_EQ(decodeGcMaintenanceState(backend.get(key)->bytes).janitor_cursor, "winner"); + const WriteResult advanced = casGcMaintenanceState(op, layout, valid.etag, + GcMaintenanceState{.janitor_cursor = "cas/ns/advanced"}, Retry::standard()); + ASSERT_TRUE(std::holds_alternative(advanced)); + const Etag advanced_etag = std::get(advanced).etag; + + ASSERT_TRUE(std::holds_alternative( + op.replace(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), advanced_etag, Retry::standard()))); + const WriteResult conflict = casGcMaintenanceState(op, layout, valid.etag, + GcMaintenanceState{.janitor_cursor = "loser"}, Retry::standard()); + EXPECT_TRUE(std::holds_alternative(conflict)); + EXPECT_EQ(decodeGcMaintenanceState(op.read(key, Retry::standard())->bytes).janitor_cursor, "winner"); } TEST(CASGCMaintenanceState, ClassifiesCorruptionAndResetsOnlyExactToken) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const String key = layout.gcMaintenanceStateKey(); - ASSERT_EQ(backend.putIfAbsent(key, "malformed").outcome, PutOutcome::Done); - const GcMaintenanceReadResult corrupt = readGcMaintenanceState(backend, layout); + auto op = requests.admit(); + + ASSERT_TRUE(std::holds_alternative(op.create(key, "malformed", Retry::once()))); + const GcMaintenanceReadResult corrupt = readGcMaintenanceState(op, layout); ASSERT_EQ(corrupt.status, GcMaintenanceReadStatus::Corrupt); - ASSERT_TRUE(corrupt.token); + ASSERT_TRUE(corrupt.etag); EXPECT_FALSE(corrupt.state); EXPECT_FALSE(corrupt.diagnostic.empty()); - ASSERT_EQ(casGcMaintenanceState(backend, layout, corrupt.token, {}).outcome, GcMaintenanceCasOutcome::Committed); - EXPECT_EQ(decodeGcMaintenanceState(backend.get(key)->bytes), GcMaintenanceState{}); + ASSERT_TRUE(std::holds_alternative( + casGcMaintenanceState(op, layout, corrupt.etag, {}, Retry::standard()))); + EXPECT_EQ(decodeGcMaintenanceState(op.read(key, Retry::standard())->bytes), GcMaintenanceState{}); } TEST(CASGCMaintenanceState, UsesExactlyOneReadOrCasAttempt) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const String key = layout.gcMaintenanceStateKey(); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent); - EXPECT_EQ(backend.getCount(key), 1u); - - backend.resetCounts(); - ASSERT_EQ(casGcMaintenanceState(backend, layout, std::nullopt, {}).outcome, - GcMaintenanceCasOutcome::Committed); - EXPECT_EQ(backend.casPutCount(key), 1u); - EXPECT_EQ(backend.getCount(key), 0u); - - backend.resetCounts(); - EXPECT_EQ(casGcMaintenanceState(backend, layout, std::nullopt, - GcMaintenanceState{.janitor_cursor = "loser"}).outcome, GcMaintenanceCasOutcome::Conflict); - EXPECT_EQ(backend.casPutCount(key), 1u); - EXPECT_EQ(backend.getCount(key), 0u); - - const auto current = backend.get(key); + auto op = requests.admit(); + + EXPECT_EQ(readGcMaintenanceState(op, layout).status, GcMaintenanceReadStatus::Absent); + EXPECT_EQ(backend->getCount(key), 1u); + + backend->resetCounts(); + ASSERT_TRUE(std::holds_alternative( + casGcMaintenanceState(op, layout, std::nullopt, {}, Retry::standard()))); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getCount(key), 0u); + + backend->resetCounts(); + const WriteResult loser_attempt = casGcMaintenanceState(op, layout, std::nullopt, + GcMaintenanceState{.janitor_cursor = "loser"}, Retry::standard()); + ASSERT_TRUE(std::holds_alternative(loser_attempt)); + EXPECT_EQ(backend->writeTotal(), 1u); + /// Unlike the legacy CAS, a refused precondition is settled by ONE exact read before the write + /// reports the conflict -- `Conflict`'s observation needs to know what is actually there. + EXPECT_EQ(backend->getCount(key), 1u); + + const std::optional current = op.read(key, Retry::standard()); ASSERT_TRUE(current); - ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), current->token).outcome, - CasOutcome::Committed); - backend.resetCounts(); - EXPECT_EQ(casGcMaintenanceState(backend, layout, current->token, - GcMaintenanceState{.janitor_cursor = "stale"}).outcome, GcMaintenanceCasOutcome::Conflict); - EXPECT_EQ(backend.casPutCount(key), 1u); - EXPECT_EQ(backend.getCount(key), 0u); - EXPECT_EQ(decodeGcMaintenanceState(backend.InMemoryBackend::get(key)->bytes).janitor_cursor, "winner"); + ASSERT_TRUE(std::holds_alternative( + op.replace(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), current->etag, Retry::standard()))); + + backend->resetCounts(); + const WriteResult stale_attempt = casGcMaintenanceState(op, layout, current->etag, + GcMaintenanceState{.janitor_cursor = "stale"}, Retry::standard()); + ASSERT_TRUE(std::holds_alternative(stale_attempt)); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getCount(key), 1u); + EXPECT_EQ(decodeGcMaintenanceState(op.read(key, Retry::standard())->bytes).janitor_cursor, "winner"); } TEST(CASGCMaintenanceState, FutureVersionPropagatesInsteadOfResetting) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const String key = layout.gcMaintenanceStateKey(); - ASSERT_EQ(backend.putIfAbsent(key, fmt::format( - "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"cur\":\"\"}}\n", currentCompatibilityVersion() + 1)).outcome, - PutOutcome::Done); + auto op = requests.admit(); + + ASSERT_TRUE(std::holds_alternative(op.create(key, fmt::format( + "{{\"type\":\"cas_gc_maintenance_state\",\"v\":{}}}\n{{\"janitor_cursor\":\"\"}}\n", currentCompatibilityVersion() + 1), + Retry::once()))); + + /// The seed write above lands through the same `write` primitive `CountingBackend` counts, so + /// reset before measuring what the read itself does. + backend->resetCounts(); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, - [&] { (void)readGcMaintenanceState(backend, layout); }); - EXPECT_EQ(backend.casPutCount(key), 0u); + [&] { (void)readGcMaintenanceState(op, layout); }); + EXPECT_EQ(backend->writeTotal(), 0u); - FailingMaintenanceReadBackend failing; - EXPECT_THROW((void)readGcMaintenanceState(failing, layout), std::runtime_error); + auto failing = std::make_shared(); + CasRequests failing_requests(failing, Fence::open()); + auto failing_op = failing_requests.admit(); + EXPECT_THROW((void)readGcMaintenanceState(failing_op, layout), std::runtime_error); } TEST(CASGCMaintenanceState, LosingCorruptResetPreservesConcurrentWinner) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const String key = layout.gcMaintenanceStateKey(); - ASSERT_EQ(backend.putIfAbsent(key, "corrupt").outcome, PutOutcome::Done); - const auto corrupt = readGcMaintenanceState(backend, layout); + auto op = requests.admit(); + + ASSERT_TRUE(std::holds_alternative(op.create(key, "corrupt", Retry::once()))); + const GcMaintenanceReadResult corrupt = readGcMaintenanceState(op, layout); ASSERT_EQ(corrupt.status, GcMaintenanceReadStatus::Corrupt); - ASSERT_TRUE(corrupt.token); - ASSERT_EQ(backend.casPut(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), corrupt.token).outcome, - CasOutcome::Committed); - backend.resetCounts(); - EXPECT_EQ(casGcMaintenanceState(backend, layout, corrupt.token, {}).outcome, - GcMaintenanceCasOutcome::Conflict); - EXPECT_EQ(backend.casPutCount(key), 1u); - EXPECT_EQ(backend.getCount(key), 0u); - EXPECT_EQ(decodeGcMaintenanceState(backend.InMemoryBackend::get(key)->bytes).janitor_cursor, "winner"); + ASSERT_TRUE(corrupt.etag); + ASSERT_TRUE(std::holds_alternative( + op.replace(key, encodeGcMaintenanceState({.janitor_cursor = "winner"}), *corrupt.etag, Retry::standard()))); + + backend->resetCounts(); + const WriteResult reset_attempt = casGcMaintenanceState(op, layout, corrupt.etag, {}, Retry::standard()); + ASSERT_TRUE(std::holds_alternative(reset_attempt)); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getCount(key), 1u); + EXPECT_EQ(decodeGcMaintenanceState(op.read(key, Retry::standard())->bytes).janitor_cursor, "winner"); } diff --git a/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp new file mode 100644 index 000000000000..3a7c2ca3ba7f --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp @@ -0,0 +1,196 @@ +#include + +#include +#include +#include +#include +#include +#include + +/// The manifest_deletes phase sends owner-removed manifest bodies to the store in chunks of +/// write-once keys, one request per chunk, and records every chunk that succeeded before a later +/// one can fail. + +namespace ProfileEvents +{ + extern const Event CASBulkDeleteRequests; +} + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +const UInt128 kGc = hexToU128("00000000000000000000000000000001"); +const RootNamespace kNs{"00/aa@cas@"}; + +ManifestRef ref(uint64_t seq) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = 1}; +} + +/// `count` tables, each with one manifest, published committed and then dropped, so the fold sees +/// `count` owner removals and `mf_cleanup` carries `count` bodies. +std::vector seedDroppedManifests(Backend & backend, const Layout & layout, uint64_t count) +{ + std::vector ids; + for (uint64_t i = 1; i <= count; ++i) + { + const ManifestRef r = ref(i); + writeBlobBody(backend, layout, DB::UInt128(0x1000 + i)); + writeManifestRaw(backend, layout, kNs, r, {blobEntryFor("a", DB::UInt128(0x1000 + i))}); + const String table = "t" + std::to_string(i); + publishCommittedTransition(backend, layout, kNs, table, std::nullopt, r); + dropRefTransition(backend, layout, kNs, table, r); + ids.push_back(ManifestId{kNs, r}); + } + return ids; +} + +/// Runs rounds until every listed manifest is gone or `max_rounds` passed; returns the sum of +/// `manifests_deleted` over the rounds that led. +uint64_t reclaim(Gc & gc, PoolPtr store, Backend & backend, const std::vector & ids, size_t max_rounds) +{ + uint64_t total = 0; + for (size_t round = 0; round < max_rounds; ++round) + { + const RoundReport rep = runRegularRoundReclaiming(gc); + if (rep.acquired_lease) + total += rep.manifests_deleted; + store->renewWatermarkOnce(); + bool any_left = false; + OperationForTest op(backend); + for (const ManifestId & id : ids) + any_left |= (*op).head(store->layout().manifestKey(id), Retry::once()).has_value(); + if (!any_left) + break; + } + return total; +} + +} + +TEST(CASGCManifestBulkDelete, FiveBodiesInChunksOfTwoAreThreeRequests) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 2, .gc_fold_max_defer_rounds = 0}); + const auto ids = seedDroppedManifests(*backend, store->layout(), 5); + const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load(); + + Gc gc(store, kGc); + const uint64_t deleted = reclaim(gc, store, *backend, ids, 16); + + EXPECT_EQ(deleted, 5u); + EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "2 + 2 + 1"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load() - requests_before, 3u); + OperationForTest op(*backend); + for (const ManifestId & id : ids) + EXPECT_FALSE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); +} + +/// The object storage rejects the chunk's one bulk `removeManyWriteOnce` as NOT_IMPLEMENTED (a +/// GCS-backed pool): the phase's `flush()` falls back to one admitted request per key +/// (`removeChunkWriteOnceOrOneByOne`, CasGc.h), and every manifest in the chunk is still recorded +/// deleted -- the per-key event emission this phase does is unaffected by how the deletes were sent. +TEST(CASGCManifestBulkDelete, NotImplementedFallsBackToOneRequestPerKeyAndStillRecordsAllOfThem) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0}); + const auto ids = seedDroppedManifests(*backend, store->layout(), 5); + + /// One armed failure: the chunk's own bulk attempt (all 5 land in one chunk under the default + /// chunk size) fails as "batch delete not supported"; the 5 single-key fallback calls that follow + /// are not armed and succeed. + backend->failNextBulkRemoveWith(std::make_exception_ptr( + DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + + Gc gc(store, kGc); + const uint64_t deleted = reclaim(gc, store, *backend, ids, 16); + + EXPECT_EQ(deleted, 5u) << "the per-key fallback must still record every manifest as deleted"; + EXPECT_EQ(backend->bulkRemoveCalls(), 6u) << "1 failed bulk attempt + 5 single-key fallback requests"; + OperationForTest op(*backend); + for (const ManifestId & id : ids) + EXPECT_FALSE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); +} + +TEST(CASGCManifestBulkDelete, AThrowInTheSecondChunkKeepsTheFirstChunksAuditAndAbortsTheRound) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 2, .gc_fold_max_defer_rounds = 0}); + const auto ids = seedDroppedManifests(*backend, store->layout(), 5); + + std::vector> manifest_phase_rows; + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "manifest_deletes") + manifest_phase_rows.push_back(rec.metrics); + }); + /// Rounds until the fold has adopted the removals; the first round whose manifest_deletes phase + /// has work is the one the fault is armed for. + size_t calls = 0; + /// The hook throws on EVERY attempt of the second chunk, so the engine's policy is exhausted and + /// the round aborts; the counter keeps climbing across the reissues, which is why the arm is + /// "second call and later" rather than "exactly the second call". + backend->onBeforeBulkRemove([&] + { + if (++calls >= 2) + throw Poco::TimeoutException("injected into the second chunk, every attempt"); + }); + + bool aborted = false; + for (size_t round = 0; round < 16 && !aborted; ++round) + { + try + { + static_cast(runRegularRoundReclaiming(gc)); + } + catch (const Poco::Exception &) + { + aborted = true; + } + /// A round that exhausted the full retry window (up to `Retry::standard()`'s 90s) may have + /// outlasted the mount lease itself, so the aborted round's own lease-renewal attempt can + /// throw too -- irrelevant to what this test asserts, so skip it once aborted. + if (!aborted) + store->renewWatermarkOnce(); + } + ASSERT_TRUE(aborted); + ASSERT_FALSE(manifest_phase_rows.empty()); + /// The aborted round's row was never emitted (the phase threw), so the last emitted row belongs + /// to an earlier, empty round; what proves the first chunk's audit survived is the store: exactly + /// the first chunk's two bodies are gone. + OperationForTest op(*backend); + size_t gone = 0; + for (const ManifestId & id : ids) + gone += !(*op).head(store->layout().manifestKey(id), Retry::once()).has_value(); + EXPECT_EQ(gone, 2u); +} + +TEST(CASGCManifestBulkDelete, ASuppressedRoundMakesNoRequest) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0}); + const auto ids = seedDroppedManifests(*backend, store->layout(), 3); + Gc gc(store, kGc); + for (size_t round = 0; round < 4; ++round) + { + static_cast(gc.runRegularRound({}, /*allow_steal*/ true, UniversePolicy::StageA_Suppressed)); + store->renewWatermarkOnce(); + } + EXPECT_EQ(backend->bulkRemoveCalls(), 0u); + OperationForTest op(*backend); + for (const ManifestId & id : ids) + EXPECT_TRUE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); +} diff --git a/src/Disks/tests/gtest_cas_gc_meta_writer.cpp b/src/Disks/tests/gtest_cas_gc_meta_writer.cpp index 2e98c3477419..c51b006be9bf 100644 --- a/src/Disks/tests/gtest_cas_gc_meta_writer.cpp +++ b/src/Disks/tests/gtest_cas_gc_meta_writer.cpp @@ -1,4 +1,5 @@ #include +#include #include #include #include @@ -6,6 +7,7 @@ #include #include +#include #include #include #include @@ -13,6 +15,7 @@ using namespace DB::Cas; using DB::Cas::tests::MetaWriteLatchBackend; using DB::Cas::tests::awaitLatchEntered; +using DB::Cas::tests::openRequestsForTest; namespace { @@ -35,7 +38,7 @@ TEST(CASGcMetaWriter, RealCondemnMarkerJobCompletesAcrossOwnerDestruction) auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); const BlobRef ref = DB::Cas::tests::idOf("1"); - const Token token{"tok-1"}; + const PersistedEtag token{"emulated", "tok-1"}; auto gc = std::make_unique(store, DB::Cas::tests::u128Of(kGcId)); backend->arm(); @@ -47,7 +50,9 @@ TEST(CASGcMetaWriter, RealCondemnMarkerJobCompletesAcrossOwnerDestruction) gc.reset(); releaser.join(); - const auto meta = loadMeta(*backend, store->layout(), ref); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const auto meta = loadMeta(op, store->layout(), ref); ASSERT_TRUE(meta) << "the condemn marker was lost across owner destruction"; EXPECT_EQ(meta->meta.state, MetaState::Condemned); EXPECT_EQ(meta->meta.condemn_round, 1u); @@ -62,7 +67,7 @@ TEST(CASGcMetaWriter, CondemnMarkerConfirmationIsVisibleAfterDrain) auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); const BlobRef ref = DB::Cas::tests::idOf("1"); - const Token token{"tok-1"}; + const PersistedEtag token{"emulated", "tok-1"}; Gc gc(store, DB::Cas::tests::u128Of(kGcId)); EXPECT_FALSE(gc.metaWriterForTest().condemnMarkerConfirmedInProcess(ref, token)); @@ -83,10 +88,11 @@ TEST(CASGcMetaWriter, RealConfirmedMetaDeleteCompletesAcrossOwnerDestruction) auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); const BlobRef ref = DB::Cas::tests::idOf("2"); - ASSERT_EQ( - putMetaIfAbsent(*store, ref, BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 64}).outcome, - CasOverwriteOutcome::Committed); - ASSERT_TRUE(loadMeta(*backend, store->layout(), ref)); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(putMetaIfAbsent( + op, store->layout(), ref, BlobMeta{.state = MetaState::Condemned, .condemn_round = 1, .size = 64}))); + ASSERT_TRUE(loadMeta(op, store->layout(), ref)); auto gc = std::make_unique(store, DB::Cas::tests::u128Of(kGcId)); backend->arm(); @@ -98,7 +104,7 @@ TEST(CASGcMetaWriter, RealConfirmedMetaDeleteCompletesAcrossOwnerDestruction) gc.reset(); releaser.join(); - EXPECT_FALSE(loadMeta(*backend, store->layout(), ref)) + EXPECT_FALSE(loadMeta(op, store->layout(), ref)) << "the confirmed-meta delete was lost across owner destruction"; } diff --git a/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp b/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp index e6530f131299..eda515afea23 100644 --- a/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp +++ b/src/Disks/tests/gtest_cas_gc_outcomes_format.cpp @@ -3,6 +3,8 @@ #include #include +#include + using namespace DB::Cas; namespace @@ -27,21 +29,23 @@ void expectThrowsCode(int expected_code, F && fn) } +CAS_BATTERY_COVERS(GcOutcomes); + TEST(CASFormatBattery, GcOutcomes) { OutcomeLog log; OutcomeEntry e; e.kind = ObjectKind::Blob; e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("00112233445566778899aabbccddeeff"))}; - e.token = Token{"e-1", TokenType::ETag}; + e.token = PersistedEtag{"etag", "e-1"}; e.outcome = OutcomeKind::Deleted; log.entries.push_back(e); runFormatBattery({FormatId::GcOutcomes, [&] { return sealObject(FormatId::GcOutcomes, encodeOutcomeLog(log)); }, [](std::string_view d) { decodeOutcomeLog(std::string(openObject(FormatId::GcOutcomes, d))); }, currentFormatHeader("cas_gc_outcomes") + - "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\"," - "\"tt\":\"etag\",\"tv\":\"e-1\",\"oc\":\"deleted\"}\n{\"n\":1}\n"}); + "{\"kind\":\"blob\",\"algo\":\"ch128\",\"digest\":\"00112233445566778899aabbccddeeff\"," + "\"token_type\":\"etag\",\"token\":\"e-1\",\"outcome\":\"deleted\"}\n{\"n\":1}\n"}); } TEST(CASGCOutcomesFormat, EmptyRoundTrips) @@ -53,13 +57,13 @@ TEST(CASGCOutcomesFormat, MultiEntryRoundTripAllOutcomes) { OutcomeLog log; log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("aa00000000000000000000000000000a"))}, - Token{"etag-1", TokenType::ETag}, OutcomeKind::Deleted}); + PersistedEtag{"etag", "etag-1"}, OutcomeKind::Deleted}); log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("bb00000000000000000000000000000b"))}, - Token{"7", TokenType::Emulated}, OutcomeKind::Spared}); + PersistedEtag{"emulated", "7"}, OutcomeKind::Spared}); log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("cc00000000000000000000000000000c"))}, - Token{"8", TokenType::Emulated}, OutcomeKind::Replaced}); + PersistedEtag{"emulated", "8"}, OutcomeKind::Replaced}); log.entries.push_back({ObjectKind::Blob, BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("dd00000000000000000000000000000d"))}, - Token{"9", TokenType::Emulated}, OutcomeKind::Absent}); + PersistedEtag{"emulated", "9"}, OutcomeKind::Absent}); const String text = encodeOutcomeLog(log); const OutcomeLog d = decodeOutcomeLog(text); ASSERT_EQ(d.entries.size(), 4u); @@ -69,33 +73,78 @@ TEST(CASGCOutcomesFormat, MultiEntryRoundTripAllOutcomes) EXPECT_EQ(d.entries[2].outcome, OutcomeKind::Replaced); EXPECT_EQ(d.entries[3].outcome, OutcomeKind::Absent); EXPECT_EQ(d.entries[0].token.value, "etag-1"); - EXPECT_EQ(d.entries[0].token.type, TokenType::ETag); + EXPECT_EQ(d.entries[0].token.dialect, "etag"); EXPECT_EQ(d.entries[3].token.value, "9"); /// Insertion order + byte-stable text (the encoder is a pure function of the log). EXPECT_EQ(encodeOutcomeLog(d), text); } +/// Closed-set pin: the four `OutcomeKind` words, walked through `magic_enum::enum_values`, which is what proves the +/// renderer and the parser consult the SAME table: a table entry missing altogether is already a +/// build error at the coverage assert, but two delegates drifting onto different tables is not. +TEST(CASGCOutcomesFormat, ClosedSetPinsOutcomeKindWords) +{ + EXPECT_EQ(outcomeKindToWireWord(OutcomeKind::Deleted), "deleted"); + EXPECT_EQ(outcomeKindToWireWord(OutcomeKind::Absent), "absent"); + EXPECT_EQ(outcomeKindToWireWord(OutcomeKind::Replaced), "replaced"); + EXPECT_EQ(outcomeKindToWireWord(OutcomeKind::Spared), "spared"); + for (const auto o : magic_enum::enum_values()) + EXPECT_EQ(outcomeKindFromWireWord(outcomeKindToWireWord(o)), o); +} + +TEST(CASGCOutcomesFormat, RecordRequiresCompleteBlobRefAndTokenGroups) +{ + OutcomeLog log; + log.entries.push_back({ObjectKind::Blob, + BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("00112233445566778899aabbccddeeff"))}, + PersistedEtag{"etag", "e-1"}, OutcomeKind::Deleted}); + const String bytes = encodeOutcomeLog(log); + + for (const auto & [field, expected_message] : { + std::pair{String(R"(,"algo":"ch128")"), "CAS outcome log: blob ref missing algo/digest"}, + std::pair{String(R"(,"digest":"00112233445566778899aabbccddeeff")"), "CAS outcome log: blob ref missing algo/digest"}, + std::pair{String(R"(,"token_type":"etag")"), "CAS outcome log: token missing token_type/token"}, + std::pair{String(R"(,"token":"e-1")"), "CAS outcome log: token missing token_type/token"}, + }) + { + const auto pos = bytes.find(field); + ASSERT_NE(pos, String::npos); + String incomplete = bytes; + incomplete.erase(pos, field.size()); + try + { + decodeOutcomeLog(incomplete); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(e.message(), expected_message); + } + } +} + TEST(CASGCOutcomesFormat, GarbageAndUnknownWordsFailClosed) { - EXPECT_THROW(decodeOutcomeLog(String("")), DB::Exception); - EXPECT_THROW(decodeOutcomeLog(String("not a cas object\n")), DB::Exception); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { decodeOutcomeLog(String("")); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { decodeOutcomeLog(String("not a cas object\n")); }); /// A record with an unknown outcome word fails closed. - const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n" - "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\"," - "\"tt\":\"etag\",\"tv\":\"x\",\"oc\":\"bogus\"}\n{\"n\":1}\n"; - EXPECT_THROW(decodeOutcomeLog(bad), DB::Exception); + const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":1}\n" + "{\"kind\":\"blob\",\"algo\":\"ch128\",\"digest\":\"00112233445566778899aabbccddeeff\"," + "\"token_type\":\"etag\",\"token\":\"x\",\"outcome\":\"bogus\"}\n{\"n\":1}\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeOutcomeLog(bad); }); /// A trailer count mismatch fails closed. - const String miscount = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n{\"n\":5}\n"; - EXPECT_THROW(decodeOutcomeLog(miscount), DB::Exception); + const String miscount = "{\"type\":\"cas_gc_outcomes\",\"v\":1}\n{\"n\":5}\n"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeOutcomeLog(miscount); }); } TEST(CASGCOutcomesFormat, DigestWidthMismatchFailsClosedWithCorruptedData) { - /// `ch128` (CityHash128) digests are 16 bytes = 32 hex chars; here the "h" field is truncated + /// `ch128` (CityHash128) digests are 16 bytes = 32 hex chars; here the `digest` field is truncated /// to 30 hex chars. Must surface as CORRUPTED_DATA (malformed serialized input), not /// `fromHex`'s BAD_ARGUMENTS. - const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":3}\n" - "{\"k\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddee\"," - "\"tt\":\"etag\",\"tv\":\"x\",\"oc\":\"deleted\"}\n{\"n\":1}\n"; + const String bad = "{\"type\":\"cas_gc_outcomes\",\"v\":1}\n" + "{\"kind\":\"blob\",\"algo\":\"ch128\",\"digest\":\"00112233445566778899aabbccddee\"," + "\"token_type\":\"etag\",\"token\":\"x\",\"outcome\":\"deleted\"}\n{\"n\":1}\n"; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeOutcomeLog(bad); }); } diff --git a/src/Disks/tests/gtest_cas_gc_read_ahead.cpp b/src/Disks/tests/gtest_cas_gc_read_ahead.cpp new file mode 100644 index 000000000000..cf8ad113a530 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_read_ahead.cpp @@ -0,0 +1,553 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::idOf; +using DB::Cas::tests::openRequestsForTest; +using DB::Cas::tests::u128Of; + +namespace +{ + +/// ============================ the class, on its own ============================ + +struct ReadAheadRig +{ + std::shared_ptr backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ThreadPool pool{CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, + CurrentMetrics::LocalThreadScheduled, + /*max_threads*/ 4, /*max_free_threads*/ 4, /*queue_size*/ 0}; + + void put(const String & key, const String & bytes) + { + ASSERT_TRUE(std::holds_alternative(op.create(key, bytes, Retry::once()))) << key; + } +}; + +} + +TEST(CASGCReadAhead, HitReturnsTheHintedBytesWithOneRequest) +{ + ReadAheadRig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + rig.backend->resetCounts(); + + reads.hintRead("k1"); + EXPECT_EQ(reads.pending(), 1u); + const auto got = reads.takeRead("k1"); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, "one"); + EXPECT_EQ(reads.pending(), 0u); + EXPECT_EQ(rig.backend->getCount("k1"), 1u); +} + +TEST(CASGCReadAhead, MissReadsInlineOnTheCallersOperation) +{ + ReadAheadRig rig; + rig.put("k2", "two"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + rig.backend->resetCounts(); + + const auto got = reads.takeRead("k2"); + ASSERT_TRUE(got.has_value()); + EXPECT_EQ(got->bytes, "two"); + EXPECT_EQ(rig.backend->getCount("k2"), 1u); +} + +TEST(CASGCReadAhead, AbsentKeyIsNulloptHintedOrNot) +{ + ReadAheadRig rig; + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + + reads.hintRead("absent-hinted"); + EXPECT_FALSE(reads.takeRead("absent-hinted").has_value()); + EXPECT_FALSE(reads.takeRead("absent-inline").has_value()); +} + +TEST(CASGCReadAhead, DuplicateHintIsOneRequest) +{ + ReadAheadRig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + rig.backend->resetCounts(); + + reads.hintRead("k1"); + reads.hintRead("k1"); + EXPECT_EQ(reads.pending(), 1u); + ASSERT_TRUE(reads.takeRead("k1").has_value()); + EXPECT_EQ(rig.backend->getCount("k1"), 1u); +} + +TEST(CASGCReadAhead, WorkerExceptionRethrowsAtTheTakeSiteAndDoesNotPoisonTheKey) +{ + ReadAheadRig rig; + rig.put("k3", "three"); + /// A non-Poco exception is a deterministic local failure to the engine, so it is thrown on the + /// first attempt rather than reissued. + rig.backend->failNextReadWith("k3", std::make_exception_ptr(std::runtime_error("injected read fault"))); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + + reads.hintRead("k3"); + EXPECT_THROW(reads.takeRead("k3"), std::runtime_error); + EXPECT_EQ(reads.pending(), 0u); + + const auto again = reads.takeRead("k3"); /// the fault was consumed; an inline read now answers + ASSERT_TRUE(again.has_value()); + EXPECT_EQ(again->bytes, "three"); +} + +TEST(CASGCReadAhead, ConcurrencyOneNeverHints) +{ + ReadAheadRig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 1); + rig.backend->resetCounts(); + + EXPECT_EQ(reads.window(), 0u); + reads.hintRead("k1"); + reads.hintHead("k1"); + EXPECT_EQ(reads.pending(), 0u); + ASSERT_TRUE(reads.takeRead("k1").has_value()); + ASSERT_TRUE(reads.takeHead("k1").has_value()); + EXPECT_EQ(rig.backend->getCount("k1"), 1u); + EXPECT_EQ(rig.backend->headCount("k1"), 1u); +} + +TEST(CASGCReadAhead, DestructorWaitsForOutstandingRequests) +{ + ReadAheadRig rig; + rig.put("k1", "one"); + rig.backend->resetCounts(); + { + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + reads.hintRead("k1"); + reads.hintHead("k1"); + } + EXPECT_EQ(rig.backend->getCount("k1"), 1u); + EXPECT_EQ(rig.backend->headCount("k1"), 1u); +} + +TEST(CASGCReadAhead, HeadHitCarriesSizeAndAbsentIsNullopt) +{ + ReadAheadRig rig; + rig.put("k1", "one"); + GcReadAhead reads(rig.op, rig.requests, rig.pool, 4); + rig.backend->resetCounts(); + + reads.hintHead("k1"); + const auto meta = reads.takeHead("k1"); + ASSERT_TRUE(meta.has_value()); + EXPECT_EQ(meta->size, 3u); + EXPECT_FALSE(reads.takeHead("absent").has_value()); + EXPECT_EQ(rig.backend->headCount("k1"), 1u); +} + +TEST(CASGCReadAhead, WindowIsFourTimesConcurrency) +{ + ReadAheadRig rig; + GcReadAhead reads(rig.op, rig.requests, rig.pool, 8); + EXPECT_EQ(reads.window(), 32u); +} + +/// ============================ the fold, at 1 against 8 ============================ + +namespace +{ + +const UInt128 kGc = u128Of("gc-read-ahead"); + +ManifestId publishPart(const PoolPtr & s, const String & ns, const String & ref, const String & payload) +{ + const RootNamespace nsr{ns}; + PartWriteInfo info; + info.intended_ref = ns + "/" + ref; + auto build = s->beginPartWrite(info); + + ManifestEntry e; + e.path = "data.bin"; + e.placement = EntryPlacement::Blob; + e.ref = BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}; + e.blob_size = payload.size(); + + const ManifestId id = build->stageManifest({e}); + build->precommitAdd(nsr, ref, id); + build->putBlob(idOf(payload), BlobSource::fromString(payload)); + build->promote(nsr, ref, build->buildId(), id); + return id; +} + +/// Three namespaces. `wide` carries a ref-log backlog longer than any window this file uses, so the +/// same-epoch lookahead is genuinely exercised; `quiet` publishes once and never drops, so its +/// frontier is proved by the checkpoint ceiling with no read at all; `gone` is emptied entirely, so +/// its blobs reach in-degree zero and the reduce phase HEADs them. Every blob is unique to its part. +void populate(const PoolPtr & store) +{ + for (int i = 0; i < 30; ++i) + publishPart(store, "srv1/wide", fmt::format("part_{}", i), fmt::format("wide-payload-{}", i)); + for (int i = 0; i < 15; ++i) + store->dropRef(RootNamespace{"srv1/wide"}, fmt::format("part_{}", i)); + + publishPart(store, "srv1/quiet", "only", "quiet-payload"); + + publishPart(store, "srv1/gone", "a", "gone-payload-a"); + publishPart(store, "srv1/gone", "b", "gone-payload-b"); + store->dropRef(RootNamespace{"srv1/gone"}, "a"); + store->dropRef(RootNamespace{"srv1/gone"}, "b"); + + store->renewWatermarkOnce(); +} + +/// TWO POOLS ARE NOT BYTE-COMPARABLE UNTIL THEIR IDENTITIES ARE MAPPED. A namespace's catalog +/// incarnation is minted from the process RNG at creation, and it appears BOTH inside every one of that +/// namespace's object keys and inside the fold seal's `life` rows -- so two independently created pools +/// running the identical workload produce identical decisions under different names, and the seal's +/// `ref_life` rows come out in a different order because they are keyed by that random id. +/// +/// Neither fact has anything to do with the read-ahead, and hiding them by weakening the comparison +/// would hide the read-ahead's own defects too. So the identities are MAPPED instead of dropped: each +/// run reports its own incarnation-hex -> namespace-name table, every 32-hex id in a key or a seal is +/// rewritten to the namespace it names, and the `ref_life` rows are sorted once their names are stable. +/// What survives the rewrite is everything the fold decided; what it removes is only the naming. +using IdNames = std::map; + +String normalizeIds(const String & text, const IdNames & names) +{ + String out = text; + for (const auto & [hex, name] : names) + { + size_t at = 0; + while ((at = out.find(hex, at)) != String::npos) + { + out.replace(at, hex.size(), name); + at += name.size(); + } + } + return out; +} + +/// The seal with its ids named and its `ref_life` rows sorted; every other row keeps its position. +String canonicalSeal(const String & seal, const IdNames & names) +{ + const String named = normalizeIds(seal, names); + std::vector out; + std::vector lives; + size_t pos = 0; + while (pos <= named.size()) + { + const size_t nl = named.find('\n', pos); + const String line = named.substr(pos, nl == String::npos ? String::npos : nl - pos); + if (line.find("\"kind\":\"ref_life\"") != String::npos) + { + lives.push_back(line); + } + else + { + if (!lives.empty()) + { + std::sort(lives.begin(), lives.end()); + out.insert(out.end(), lives.begin(), lives.end()); + lives.clear(); + } + out.push_back(line); + } + if (nl == String::npos) + break; + pos = nl + 1; + } + std::sort(lives.begin(), lives.end()); + out.insert(out.end(), lives.begin(), lives.end()); + + String joined; + for (const String & line : out) + { + joined += line; + joined += '\n'; + } + return joined; +} + +struct FoldRun +{ + std::vector seals; /// the fold seal's bytes after each round + std::vector> intake; /// `fold_ref_intake` metrics, per round + std::vector> reduce; /// `fold_reduce` metrics, per round + std::map gets; /// key -> GETs over the whole run + std::map heads; /// key -> HEADs over the whole run + std::vector condemned; + std::vector deleted; + IdNames id_names; /// incarnation hex -> namespace, for the comparison +}; + +void runFolds(uint64_t concurrency, size_t rounds, FoldRun & out) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0, .gc_read_concurrency = concurrency}); + populate(store); + + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const Layout & layout = store->layout(); + + /// Read the id table BEFORE the rounds: a namespace the fold reclaims loses its catalog row, and its + /// keys still have to be nameable when the two runs are compared. + for (const CatalogEntry & entry : CasRefCatalog::read(op, layout).catalog.entries) + out.id_names.emplace(u128ToHex(entry.incarnation), "<" + entry.ns.string() + ">"); + + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "fold_ref_intake") + out.intake.push_back(rec.metrics); + else if (rec.phase == "fold_reduce") + out.reduce.push_back(rec.metrics); + }); + + /// Only the rounds' own I/O is compared; the identical population above is not part of the claim. + backend->resetCounts(); + + for (size_t round = 0; round < rounds; ++round) + { + const RoundReport report = gc.runRegularRound(); + ASSERT_TRUE(report.acquired_lease) << "round " << round; + out.condemned.push_back(report.condemned); + out.deleted.push_back(report.deleted); + store->renewWatermarkOnce(); + + const GcState st = decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes); + const auto seal = op.read(layout.foldSealKey(st.snap_generation, st.snap_attempt), Retry::once()); + out.seals.push_back(seal ? canonicalSeal(seal->bytes, out.id_names) : String{}); + } + + for (const String & key : backend->touchedKeys()) + { + const String named = normalizeIds(key, out.id_names); + if (const uint64_t n = backend->getCount(key); n != 0) + out.gets[named] += n; + if (const uint64_t n = backend->headCount(key); n != 0) + out.heads[named] += n; + } +} + +} + +TEST(CASGCReadAhead, FoldIsIdenticalAtConcurrencyOneAndEight) +{ + constexpr size_t kRounds = 6; + FoldRun one; + FoldRun eight; + ASSERT_NO_FATAL_FAILURE(runFolds(1, kRounds, one)); + ASSERT_NO_FATAL_FAILURE(runFolds(8, kRounds, eight)); + + ASSERT_EQ(one.seals.size(), kRounds); + ASSERT_EQ(eight.seals.size(), kRounds); + for (size_t i = 0; i < kRounds; ++i) + EXPECT_EQ(one.seals[i], eight.seals[i]) << "fold seal differs at round " << i; + + EXPECT_EQ(one.intake, eight.intake); + EXPECT_EQ(one.reduce, eight.reduce); + EXPECT_EQ(one.condemned, eight.condemned); + EXPECT_EQ(one.deleted, eight.deleted); + + /// THE REQUEST-SET CLAIM. Every namespace here is healthy, so nothing is hinted that the walk + /// does not go on to read: the hints stop at the checkpoint ceiling the walk stops at, a quiet + /// namespace is not hinted at all, and every decoded log's manifest edges are all folded. So the + /// read-ahead must issue the SAME GETs against the SAME keys, not merely produce the same answer. + EXPECT_EQ(one.gets, eight.gets); + + ASSERT_FALSE(one.intake.empty()); + EXPECT_GT(one.intake[0].at("logs_applied"), 32u) + << "the wide namespace must carry more logs than the window, or the lookahead is untested"; +} + +TEST(CASGCReadAhead, ReduceCondemnsTheSameBlobsWithTheSameHeadsAtConcurrencyOneAndEight) +{ + /// `populate` gives every part its own blob and drops whole parts, so a dropped blob loses its only + /// edge and no surviving blob has a removal: the hinted set equals the set `head_blob` takes, and the + /// per-key HEAD counts must match exactly rather than merely producing the same verdict. + constexpr size_t kRounds = 6; + FoldRun one; + FoldRun eight; + ASSERT_NO_FATAL_FAILURE(runFolds(1, kRounds, one)); + ASSERT_NO_FATAL_FAILURE(runFolds(8, kRounds, eight)); + + EXPECT_EQ(one.condemned, eight.condemned); + EXPECT_EQ(one.heads, eight.heads); + + uint64_t condemned_total = 0; + for (const size_t n : one.condemned) + condemned_total += n; + EXPECT_GT(condemned_total, 0u) << "the scenario must condemn, or the reduce read-ahead is untested"; +} + +namespace +{ + +/// Throws once on the first read issued from a thread other than the one that armed it: exactly a +/// read-ahead worker's request, never the round thread's own. +class WorkerReadFaultBackend : public CountingBackend +{ +public: + void armAgainstOtherThreads() + { + owner = std::this_thread::get_id(); + armed.store(true); + } + + bool fired() const { return !armed.load(); } + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (armed.load() && std::this_thread::get_id() != owner) + { + armed.store(false); + throw std::runtime_error("injected worker read fault"); + } + return CountingBackend::read(key, access); + } + +private: + std::atomic armed{false}; + std::thread::id owner; +}; + +} + +namespace +{ + +/// Proves OVERLAP, which no equality test can: it releases a read only once `k_overlap` reads are +/// inside the backend at the same time. If the fold's reads were still strictly one after another the +/// count could never reach two, so the round would block until the bounded wait expires and the flag +/// below would stay false. The wait is bounded and the last arrival wakes everyone, so nothing here can +/// hang the suite: a fold with no overlap finishes late, it does not finish never. +class OverlapWitnessBackend : public CountingBackend +{ +public: + explicit OverlapWitnessBackend(size_t k_overlap_) : k_overlap(k_overlap_) {} + + /// ARMED ONLY FOR THE ROUND. Holding reads is fatal to a WRITER: its checkpoint publication is a + /// CAS with a bounded retry budget, and a latch on every read exhausts it long before the round + /// under test ever starts. + void arm() { armed.store(true); } + + bool sawOverlap() const { return saw_overlap.load(); } + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (!armed.load()) + return CountingBackend::read(key, access); + { + std::unique_lock lock(mutex); + ++in_flight; + peak = std::max(peak, in_flight); + if (in_flight >= k_overlap) + { + saw_overlap.store(true); + gate.notify_all(); + } + else + { + gate.wait_for(lock, std::chrono::milliseconds(250), + [&] { return in_flight >= k_overlap || saw_overlap.load(); }); + } + } + /// The count stays raised ACROSS the read, so what it measures is requests genuinely in the + /// backend together. Releasing it before the read would leave a window of a few instructions + /// that two threads would have to hit simultaneously to be seen -- which is a measurement of + /// luck, not of overlap. + std::optional raw = CountingBackend::read(key, access); + { + std::lock_guard lock(mutex); + --in_flight; + } + return raw; + } + + size_t peakInFlight() const + { + std::lock_guard lock(mutex); + return peak; + } + +private: + const size_t k_overlap; + std::atomic armed{false}; + mutable std::mutex mutex; + std::condition_variable gate; + size_t in_flight = 0; + size_t peak = 0; + std::atomic saw_overlap{false}; +}; + +} + +TEST(CASGCReadAhead, TheFoldsReadsActuallyOverlap) +{ + auto backend = std::make_shared(/*k_overlap*/ 2); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0, .gc_read_concurrency = 8}); + populate(store); + + Gc gc(store, kGc); + backend->arm(); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + + EXPECT_TRUE(backend->sawOverlap()) + << "no two of the fold's reads were ever in the backend at the same time; peak in flight was " + << backend->peakInFlight(); + EXPECT_GT(backend->peakInFlight(), 1u); +} + +TEST(CASGCReadAhead, WorkerReadFaultFailsTheRoundAndTheNextRoundRecovers) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, + PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_fold_max_defer_rounds = 0, .gc_read_concurrency = 8}); + populate(store); + + Gc gc(store, kGc); + backend->armAgainstOtherThreads(); + EXPECT_ANY_THROW(gc.runRegularRound()); + EXPECT_TRUE(backend->fired()) << "no read-ahead worker ever issued a request"; + + store->renewWatermarkOnce(); + const RoundReport recovered = gc.runRegularRound(); + EXPECT_TRUE(recovered.acquired_lease); +} diff --git a/src/Disks/tests/gtest_cas_gc_rebuild.cpp b/src/Disks/tests/gtest_cas_gc_rebuild.cpp index aa896dbf68df..cbfee9d0d5e5 100644 --- a/src/Disks/tests/gtest_cas_gc_rebuild.cpp +++ b/src/Disks/tests/gtest_cas_gc_rebuild.cpp @@ -32,6 +32,35 @@ ManifestRef ref(uint64_t seq, uint64_t inst) { return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; } + +/// An exact read (mirrors the retired `backend->get(key)`). +std::optional readObj(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +/// A HEAD (mirrors the retired `backend->head(key)`). +std::optional headObj(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()); +} + +/// A one-shot `create`, asserting it committed (mirrors the retired `backend->putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// A delete-exact against `key`/`expected` (mirrors the retired `backend->deleteExact(key, token)`); +/// the caller decides whether to assert the outcome. +Removal removeExact(Backend & backend, const String & key, const Etag & expected) +{ + OperationForTest op(backend); + return (*op).remove(key, expected, Retry::once()); +} } /// (`CASGCBaselineGuard.FreshStateOverTrimmedJournalsFailsClosed` was removed with the snapshot+log ref @@ -72,12 +101,12 @@ TEST(CASGCBaselineGuard, AbsentAdoptedSealFailsClosed) gc.runRegularRound(); /// Corrupt (б): delete the adopted fold seal out from under a healthy gc/state. - const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const GcState st = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes); ASSERT_GT(st.snap_generation, 0u); const String seal_key = store->layout().foldSealKey(st.snap_generation, st.snap_attempt); - const HeadResult sh = backend->head(seal_key); - ASSERT_TRUE(sh.exists); - ASSERT_EQ(backend->deleteExact(seal_key, sh.token).kind, DeleteOutcome::Kind::Deleted); + const auto sh = headObj(*backend, seal_key); + ASSERT_TRUE(sh.has_value()); + ASSERT_EQ(removeExact(*backend, seal_key, sh->etag), Removal::Removed); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc.runRegularRound(); }); } @@ -114,10 +143,10 @@ TEST(CASGCRebuild, RecoversLostStateAndConverges) store->renewWatermarkOnce(); /// renews the lease + build-watermark floor /// Capture the round reached before gc/state is destroyed (the rebuild must mint strictly above it). - const auto pre_rebuild_got = backend->get(store->layout().gcStateKey()); + const auto pre_rebuild_got = readObj(*backend, store->layout().gcStateKey()); ASSERT_TRUE(pre_rebuild_got.has_value()); const uint64_t pre_rebuild_round = decodeGcState(pre_rebuild_got->bytes).round; - ASSERT_EQ(backend->deleteExact(store->layout().gcStateKey(), pre_rebuild_got->token).kind, DeleteOutcome::Kind::Deleted); + ASSERT_EQ(removeExact(*backend, store->layout().gcStateKey(), pre_rebuild_got->etag), Removal::Removed); Gc gc2(store, hexToU128("00000000000000000000000000000003")); /// A fresh GC over the orphaned generation artifacts fails closed: re-folding from a fresh gc/state @@ -142,8 +171,8 @@ TEST(CASGCRebuild, RecoversLostStateAndConverges) runRegularRoundReclaiming(gc2); store->renewWatermarkOnce(); } - EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).exists); - EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(2))})).exists) + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).has_value()); + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(2))})).has_value()) << "a rebuild condemns nothing, so a pre-rebuild drop is retained — never reclaimed by a " "substitute pass, and never lost"; @@ -175,13 +204,13 @@ TEST(CASGCRebuild, RecoversLostGenerationArtifact) gc.runRegularRound(); /// Lose one snapshot run object out from under the healthy state. - const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); - const auto seal = decodeFoldSeal(backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const GcState st = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes); + const auto seal = decodeFoldSeal(readObj(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); ASSERT_FALSE(seal.blob_target_runs.empty()); const String run_key = seal.blob_target_runs.front().key; - const HeadResult rh = backend->head(run_key); - ASSERT_TRUE(rh.exists); - ASSERT_EQ(backend->deleteExact(run_key, rh.token).kind, DeleteOutcome::Kind::Deleted); + const auto rh = headObj(*backend, run_key); + ASSERT_TRUE(rh.has_value()); + ASSERT_EQ(removeExact(*backend, run_key, rh->etag), Removal::Removed); /// A pure ref-carry round would not read the lost run; land a REAL delta so the fold's /// three-cursor merge must stream the prior run — and fails closed on its absence. @@ -194,7 +223,7 @@ TEST(CASGCRebuild, RecoversLostGenerationArtifact) const RebuildReport rep = gc.rebuildBaseline(/*force*/ false); ASSERT_TRUE(rep.performed) << rep.refusal; EXPECT_NO_THROW(gc.runRegularRound()); - EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).exists); + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))})).has_value()); } /// FORCE: a healthy state refuses the plain rebuild; FORCE rebuilds; rounds run clean after. @@ -237,11 +266,13 @@ TEST(CASGCRebuild, FrozenCheckpointFrontierExcludesVisibleUnfrontieredTail) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/rebuild-frozen-frontier@cas@"}; const UInt128 life_id{0xF001}; const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, life_id); - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); const ManifestRef admitted = ref(1, 0xA1); @@ -259,12 +290,12 @@ TEST(CASGCRebuild, FrozenCheckpointFrontierExcludesVisibleUnfrontieredTail) fixture::writeRefLogRaw(*backend, layout, RefLogTxn{.ns = ns.string(), .txn_id = RefTxnId{1, 2}, .ops = publishCommittedOps("unfrontiered", unfrontiered), .prev_epoch_seal = std::nullopt}); - ASSERT_TRUE(backend->head(layout.refLogKey(life, RefTxnId{1, 2})).exists); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(headObj(*backend, layout.refLogKey(life, RefTxnId{1, 2})).has_value()); + createObj(*backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt})); Gc gc(store, kGc); const RebuildReport report = gc.rebuildBaseline(/*force=*/false); @@ -280,10 +311,12 @@ TEST(CASGCRebuild, LiveCatalogLifeWithoutCheckpointFailsClosed) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/rebuild-missing-checkpoint@cas@"}; const UInt128 life_id{0xF002}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); const ManifestRef admitted = ref(1, 0xA2); @@ -298,7 +331,7 @@ TEST(CASGCRebuild, LiveCatalogLifeWithoutCheckpointFailsClosed) Gc gc(store, kGc); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)gc.rebuildBaseline(/*force=*/false); }); - const auto state = backend->get(layout.gcStateKey()); + const auto state = readObj(*backend, layout.gcStateKey()); ASSERT_TRUE(state); EXPECT_EQ(decodeGcState(state->bytes).snap_generation, 0u) << "a rejected recovery must not adopt a new baseline"; @@ -311,10 +344,12 @@ TEST(CASGCRebuild, CheckpointSnapshotAtOlderEpochSealFailsClosed) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/rebuild-checkpoint-base-seal@cas@"}; const UInt128 life_id{0xF003}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = life_id}); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, life_id); @@ -338,16 +373,16 @@ TEST(CASGCRebuild, CheckpointSnapshotAtOlderEpochSealFailsClosed) applyRefLogTxn(through_seal, seal_txn); writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + createObj(*backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{2, 1}, .checkpoint_snapshot_id = RefTxnId{1, 2}, - .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{2, 1}})); Gc gc(store, kGc); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)gc.rebuildBaseline(/*force=*/false); }); - const auto state = backend->get(layout.gcStateKey()); + const auto state = readObj(*backend, layout.gcStateKey()); ASSERT_TRUE(state); EXPECT_EQ(decodeGcState(state->bytes).snap_generation, 0u) << "a rejected checkpoint base must not publish a REBUILD baseline"; @@ -357,24 +392,26 @@ TEST(CASGCRebuild, DamagedGenerationZeroStatePerformsNoCatalogDrainMutation) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const RootNamespace ns{"00/removing-without-parent@cas@"}; const UInt128 life_id{91}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + CasRefCatalog::casAdmitEntry(op, layout, 1, CatalogEntry{ .ns = ns, .state = NsState::Live, .incarnation = life_id}); - CasRefCatalog::casUpdate(*backend, layout, [](const RefCatalog & current) + CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; next.entries[0].state = NsState::Removing; next.entries[0].removal_started_round = 1; return next; }); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(ns, life_id)), encodeRefCkpt(RefCkpt{ + createObj(*backend, layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(ns, life_id)), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); - const uint64_t catalog_cas_before = backend->casPutCount(layout.refCatalogKey()); + .last_epoch_seal = std::nullopt})); + const uint64_t catalog_cas_before = backend->putOverwriteCount(layout.refCatalogKey()); const uint64_t plans_before = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); @@ -382,8 +419,8 @@ TEST(CASGCRebuild, DamagedGenerationZeroStatePerformsNoCatalogDrainMutation) const RebuildReport report = gc.rebuildBaseline(/*force*/ false); ASSERT_TRUE(report.performed) << report.refusal; EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), catalog_cas_before); - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(*backend, layout); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_cas_before); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); ASSERT_EQ(catalog.catalog.entries.size(), 1u); EXPECT_EQ(catalog.catalog.entries[0].state, NsState::Removing); EXPECT_EQ(catalog.catalog.entries[0].incarnation, life_id); @@ -408,12 +445,13 @@ TEST(CASGCRebuild, MissingCommittedManifestRefuses) gc.runRegularRound(); /// trim /// Disaster pair: gc/state lost AND tbl_b's manifest body lost. - const HeadResult st = backend->head(store->layout().gcStateKey()); - backend->deleteExact(store->layout().gcStateKey(), st.token); + const auto st = headObj(*backend, store->layout().gcStateKey()); + ASSERT_TRUE(st.has_value()); + removeExact(*backend, store->layout().gcStateKey(), st->etag); const String mkey = store->layout().manifestKey(ManifestId{ns, b}); - const HeadResult mh = backend->head(mkey); - ASSERT_TRUE(mh.exists); - backend->deleteExact(mkey, mh.token); + const auto mh = headObj(*backend, mkey); + ASSERT_TRUE(mh.has_value()); + removeExact(*backend, mkey, mh->etag); Gc gc2(store, hexToU128("00000000000000000000000000000004")); const RebuildReport rep = gc2.rebuildBaseline(/*force*/ false); @@ -421,7 +459,7 @@ TEST(CASGCRebuild, MissingCommittedManifestRefuses) EXPECT_NE(rep.refusal.find("tbl_b"), String::npos) << rep.refusal; /// The lease acquire minted a gen-0 bootstrap body (that is the acquire's contract, not the /// rebuild's); the rebuild's own contract is that NO baseline was blessed by the refusal. - const auto post = backend->get(store->layout().gcStateKey()); + const auto post = readObj(*backend, store->layout().gcStateKey()); ASSERT_TRUE(post.has_value()); const GcState post_state = decodeGcState(post->bytes); EXPECT_EQ(post_state.snap_generation, 0u) << "a refused rebuild must not adopt a baseline"; @@ -455,7 +493,7 @@ TEST(CASGCRebuild, LivePrecommitEdgesIncluded) gc2.runRegularRound(); store->renewWatermarkOnce(); } - EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).exists); + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).has_value()); } /// O(budget) attempt iteration: a tiny edge budget forces multi-batch folding; the rebuilt @@ -488,8 +526,9 @@ TEST(CASGCRebuild, BatchedRebuildProtectsAllRefs) Gc gc(store, kGc); gc.runRegularRound(); gc.runRegularRound(); - const HeadResult st = backend->head(store->layout().gcStateKey()); - backend->deleteExact(store->layout().gcStateKey(), st.token); + const auto st = headObj(*backend, store->layout().gcStateKey()); + ASSERT_TRUE(st.has_value()); + removeExact(*backend, store->layout().gcStateKey(), st->etag); Gc gc2(store, hexToU128("00000000000000000000000000000006")); /// Every shard has `edge_budget + 1` live edges, so each independently crosses the flush budget; @@ -500,11 +539,11 @@ TEST(CASGCRebuild, BatchedRebuildProtectsAllRefs) EXPECT_EQ(rep.committed_refs, blobs.size()); /// Multiple rebuild flushes still converge to one authoritative row domain: no more than one - /// canonical seq-0 `btr` per shard and exactly one `cnd` per shard. These are the cardinalities the + /// canonical seq-0 `blob_run` per shard and exactly one `condemned` per shard. These are the cardinalities the /// catalog admission reservation over-covers independently of catalog-entry count. - const GcState rebuilt_state = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const GcState rebuilt_state = decodeGcState(readObj(*backend, store->layout().gcStateKey())->bytes); const CasFoldSeal rebuilt_seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey( + readObj(*backend, store->layout().foldSealKey( rebuilt_state.snap_generation, rebuilt_state.snap_attempt))->bytes, store->layout(), gc_shards); ASSERT_EQ(rebuilt_seal.condemned_summary.size(), gc_shards); @@ -518,7 +557,7 @@ TEST(CASGCRebuild, BatchedRebuildProtectsAllRefs) const auto parsed = store->layout().parseBlobTargetRunKey(run.key); ASSERT_TRUE(parsed.has_value()); EXPECT_EQ(parsed->shard, run.shard); - EXPECT_EQ(parsed->generation, run.generation); + EXPECT_EQ(parsed->generation, run.key_generation); EXPECT_EQ(parsed->seq, 0u); } EXPECT_TRUE(run_seen[0]); @@ -530,12 +569,12 @@ TEST(CASGCRebuild, BatchedRebuildProtectsAllRefs) store->renewWatermarkOnce(); } for (const UInt128 blob : blobs) - EXPECT_TRUE(backend->head(store->layout().blobKey(legacyMetaTestRef(blob))).exists) + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(legacyMetaTestRef(blob))).has_value()) << "blob " << u128ToHex(blob); } /// Trimmed-but-live (design delta 2): the precommit's journal evidence is gone (trim), the build -/// is NOT provably dead (a live build holds min_active down) — the unowned-alive sweep must +/// is NOT provably dead (a live build holds min_active_build_sequence down) — the unowned-alive sweep must /// over-protect the manifest's edges. TEST(CASGCRebuild, UnownedAliveManifestOverProtected) { @@ -543,7 +582,7 @@ TEST(CASGCRebuild, UnownedAliveManifestOverProtected) auto store = openPoolForTest(backend); const RootNamespace ns{"00/aa@cas@"}; - /// A LIVE build pins min_active at its build_seq, so higher build sequences are not provably dead. + /// A LIVE build pins min_active_build_sequence at its build_seq, so higher build sequences are not provably dead. auto live_build = store->beginPartWrite({}); store->renewWatermarkOnce(); @@ -568,7 +607,7 @@ TEST(CASGCRebuild, UnownedAliveManifestOverProtected) gc.runRegularRound(); store->renewWatermarkOnce(); } - EXPECT_TRUE(backend->head(store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).exists); + EXPECT_TRUE(headObj(*backend, store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(9))})).has_value()); } /// Task 4 (SYSTEM CAS GC REBUILD): a rebuild refuses when ANOTHER Gc instance holds @@ -612,7 +651,10 @@ TEST(CASGCRebuild, LeaseConflictRefuses) TEST(CASGCClampSuppression, LandedEdgeBehindClampNeverDeleted) { auto backend = std::make_shared(); - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto store = openPoolForTest(backend); const RootNamespace ns{"00/aa@cas@"}; @@ -640,23 +682,28 @@ TEST(CASGCClampSuppression, LandedEdgeBehindClampNeverDeleted) /// Rounds with acks current: X reaches folded in-degree 0 and is condemned, but every pass is /// CLAMPED (the bodiless precommit persists), so nothing may graduate or delete. /// Observability (2026-07-03): every clamp emits a gc_fold_clamp event with the reason. - store->setEventSink([&](const CasEvent & e){ if (e.type == CasEventType::GcFoldClamp) seen.push_back(e); }); + store->setEventSink([seen](const CasEvent & e) + { + if (e.type == CasEventType::GcFoldClamp) + seen->push(e); + }); const String blob_key = store->layout().blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(DB::UInt128(1))}); for (int i = 0; i < 6; ++i) { gc.runRegularRound(); store->renewWatermarkOnce(); - ASSERT_TRUE(backend->head(blob_key).exists) + ASSERT_TRUE(headObj(*backend, blob_key).has_value()) << "round " << i << ": X was deleted while its landed +1 sat unfolded behind the clamp"; } - ASSERT_FALSE(seen.empty()) << "each clamped pass must emit a gc_fold_clamp event"; - EXPECT_NE(seen.front().reason.find("fold barrier"), String::npos); + const std::vector observed_events = seen->snapshot(); + ASSERT_FALSE(observed_events.empty()) << "each clamped pass must emit a gc_fold_clamp event"; + EXPECT_NE(observed_events.front().reason.find("fold barrier"), String::npos); /// Snapshot+log ref model: the clamp is per-table (one ref-log stream per namespace, no ref shards), /// so the event names the clamped `log` and the `resolved_through` cursor rather than a shard number. - EXPECT_TRUE(seen.front().detail.contains("log")) + EXPECT_TRUE(observed_events.front().detail.contains("log")) << "clamp event must name the clamped log id"; - EXPECT_TRUE(seen.front().detail.contains("resolved_through")) + EXPECT_TRUE(observed_events.front().detail.contains("resolved_through")) << "clamp event must name the cursor it resolved through"; store->setEventSink(nullptr); @@ -668,7 +715,7 @@ TEST(CASGCClampSuppression, LandedEdgeBehindClampNeverDeleted) gc.runRegularRound(); store->renewWatermarkOnce(); } - EXPECT_TRUE(backend->head(blob_key).exists); + EXPECT_TRUE(headObj(*backend, blob_key).has_value()); /// And the pipeline is unwedged: a genuinely-unreferenced blob still gets reclaimed. const ManifestRef m3 = ref(3, 0xC3); writeBlobBody(*backend, store->layout(), DB::UInt128(5)); diff --git a/src/Disks/tests/gtest_cas_gc_resume.cpp b/src/Disks/tests/gtest_cas_gc_resume.cpp index 6c707bdd60cd..6433576401bc 100644 --- a/src/Disks/tests/gtest_cas_gc_resume.cpp +++ b/src/Disks/tests/gtest_cas_gc_resume.cpp @@ -22,15 +22,16 @@ ManifestRef ref(uint64_t seq, uint64_t inst) } bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + OperationForTest op(b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}), Retry::once()).has_value(); } /// Whether the CURRENT retired list (any gc-shard) still holds an entry. bool anyRetiredPending(const PoolPtr & s) { - /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// Condemned state rides the adopted fold seal's RunMarker::Condemned rows, not a /// separate retired list — reconstruct the in-flight set from the seal. - return anyCondemnedInSeal(s->backend(), s->layout()); + return anyCondemnedInSeal(*s->poolBackendPtr(), s->layout()); } /// Drive regular GC to a fixpoint over the ACK-FLOOR round (renew the store's mount ack after each round; @@ -52,31 +53,37 @@ size_t runGcToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) return rounds; } -/// A backend that denies ONCE the SINGLE round-commit `gc/state` CAS — the casPut that advances -/// snap_generation (the one-pass round has exactly one such CAS; the lease-acquire CAS does not advance -/// snap_generation). A denied round leaves only never-adopted attempt-scoped debris (fold seal / retired -/// list under an attempt gc/state never adopted); a fresh-attempt rerun is idempotent. +/// A backend that refuses ONCE the SINGLE round-commit `gc/state` write — the conditional write that +/// advances snap_generation (the one-pass round has exactly one such write; the lease acquire/renew does +/// not advance snap_generation). A refused round leaves only never-adopted attempt-scoped debris (a fold +/// seal under an attempt gc/state never adopted); a fresh-attempt rerun is idempotent. class InterruptRoundCasBackend : public InMemoryBackend { public: explicit InterruptRoundCasBackend(String gc_state_key_) : gc_state_key(std::move(gc_state_key_)) {} - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - if (arm_interrupt && key == gc_state_key) + if (arm_interrupt && expected_value && key == gc_state_key) { - const auto stored = get(key); - const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; - const uint64_t next_gen = decodeGcState(bytes).snap_generation; - if (next_gen > stored_gen) + const auto stored = InMemoryBackend::read(key, access); + if (stored + && decodeGcState(bytes).snap_generation > decodeGcState(stored->bytes).snap_generation) { - arm_interrupt = false; /// one-shot: only depose the first round-commit CAS - throw DB::Exception(DB::ErrorCodes::ABORTED, - "test-injected: round-commit gc/state CAS denied (leader deposed mid-round)"); + arm_interrupt = false; /// one-shot: only depose the first round-commit write + /// A REFUSAL, not a throw: a thrown transport error is an ambiguity the engine settles + /// by an exact read and then reissues while the precondition it named is unmoved, so + /// the round would commit on the reissue. A refused precondition ends the write at + /// once. The object is moved too -- the same bytes under a fresh incarnation -- because + /// a store refuses only what changed; the CONTENT is deliberately left alone, so this + /// round's own lease and cursor are exactly what a deposed round leaves behind. + (void)InMemoryBackend::write(key, stored->bytes, stored->value, access); + return std::unexpected(RawConflict{}); } } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool arm_interrupt = false; @@ -139,7 +146,8 @@ TEST(CASGCReplay, DeposedRoundRerunsUnderFreshAttempt) Gc gc1(store, hexToU128("00000000000000000000000000000001")); runRegularRoundReclaiming(gc1); store->renewWatermarkOnce(); - const auto after_fold = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + OperationForTest op(*backend); + const auto after_fold = decodeGcState((*op).read(store->layout().gcStateKey(), Retry::once())->bytes); ASSERT_EQ(after_fold.snap_attempt, after_fold.lease.seq); ASSERT_GT(after_fold.snap_generation, 0u); @@ -151,7 +159,7 @@ TEST(CASGCReplay, DeposedRoundRerunsUnderFreshAttempt) EXPECT_THROW(runRegularRoundReclaiming(gc1), DB::Exception); backend->arm_interrupt = false; - const auto after_interrupt = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_interrupt = decodeGcState((*op).read(store->layout().gcStateKey(), Retry::once())->bytes); EXPECT_EQ(after_interrupt.snap_generation, after_fold.snap_generation) << "the denied round-commit CAS must NOT advance the adopted generation"; EXPECT_EQ(after_interrupt.snap_attempt, after_fold.snap_attempt) @@ -159,7 +167,7 @@ TEST(CASGCReplay, DeposedRoundRerunsUnderFreshAttempt) // The deposed round's fold seal is durable under its OWN (unadopted) attempt — unreferenced by gc/state. const uint64_t deposed_attempt = after_fold.lease.seq + 1; // round 2 renewed the lease once const uint64_t deposed_gen = after_fold.snap_generation + 1; - EXPECT_TRUE(backend->head(store->layout().foldSealKey(deposed_gen, deposed_attempt)).exists) + EXPECT_TRUE((*op).head(store->layout().foldSealKey(deposed_gen, deposed_attempt), Retry::once()).has_value()) << "the deposed round's fold seal is durable under its own unadopted attempt (harmless debris)"; // A DIFFERENT leader takes over. The lease steal protocol observes the stalled lease twice before @@ -174,7 +182,7 @@ TEST(CASGCReplay, DeposedRoundRerunsUnderFreshAttempt) EXPECT_NO_THROW(runGcToFixpoint(store, gc2)); EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); - const auto after_drain = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto after_drain = decodeGcState((*op).read(store->layout().gcStateKey(), Retry::once())->bytes); EXPECT_GT(after_drain.snap_generation, after_fold.snap_generation) << "the round completed under gc2"; EXPECT_NE(after_drain.snap_attempt, deposed_attempt) << "the drained round never adopted the deposed attempt"; diff --git a/src/Disks/tests/gtest_cas_gc_round.cpp b/src/Disks/tests/gtest_cas_gc_round.cpp index b987cc39d8fb..eeaad2c7b86f 100644 --- a/src/Disks/tests/gtest_cas_gc_round.cpp +++ b/src/Disks/tests/gtest_cas_gc_round.cpp @@ -9,12 +9,14 @@ #include #include #include "cas_test_helpers.h" +#include "config.h" namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; extern const int CORRUPTED_DATA; extern const int ABORTED; +extern const int NETWORK_ERROR; } namespace ProfileEvents @@ -58,14 +60,38 @@ ManifestRef ref(uint64_t seq, uint64_t inst) return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; } +std::optional readOf(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +bool headExists(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + +ListPage listOf(Backend & backend, const String & prefix, const String & cursor, size_t limit) +{ + OperationForTest op(backend); + return (*op).list(prefix, cursor, limit, Retry::standard()); +} + +void createRaw(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + (*op).create(key, bytes, Retry::standard()); +} + bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + return headExists(b, layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})); } bool manifestExists(InMemoryBackend & b, const Layout & layout, const ManifestId & id) { - return b.head(layout.manifestKey(id)).exists; + return headExists(b, layout.manifestKey(id)); } PoolPtr openTestPool(std::shared_ptr & out_backend) @@ -95,32 +121,28 @@ PoolPtr openTestPoolWithConfig(std::shared_ptr & out_backend, P class GcStateCasFaultBackend : public InMemoryBackend { public: - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - - CasResult casPut(const String & key, const String & bytes, - const std::optional & expected, const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - if (key == faulted_key) + if (expected_value && key == faulted_key) { ++calls_to_faulted_key; if (fail_at_call != 0 && calls_to_faulted_key == fail_at_call) - return CasResult{CasOutcome::Conflict, {}}; + return std::unexpected(RawConflict{}); } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } String faulted_key; size_t calls_to_faulted_key = 0; - size_t fail_at_call = 0; /// 0 = never fault; else fault exactly the Nth casPut to `faulted_key` + /// 0 = never fault; else refuse exactly the Nth conditional write to `faulted_key`. + size_t fail_at_call = 0; }; GcState readState(InMemoryBackend & b, const Pool & s) { - const auto got = b.get(s.layout().gcStateKey()); + const auto got = readOf(b, s.layout().gcStateKey()); if (!got) { ADD_FAILURE() << "gc/state absent"; @@ -129,7 +151,7 @@ GcState readState(InMemoryBackend & b, const Pool & s) return decodeGcState(got->bytes); } -/// Whether ANY gc-shard's adopted-seal run still holds a `kCondemned` row (retired-in-snapshot T4: the +/// Whether ANY gc-shard's adopted-seal run still holds a `RunMarker::Condemned` row (the /// retired state rides the snapshot run, not a separate retired-list object) — the ack-floor deletion /// pipeline is still in flight while this is true. bool anyRetiredPending(InMemoryBackend & b, const Pool & s) @@ -162,18 +184,18 @@ size_t driveToFixpoint(InMemoryBackend & backend, const PoolPtr & store, Gc & gc return working_rounds; } -/// A full key -> token snapshot of the backend, for the previewDeletes write-free invariant: any -/// put/casPut/overwrite mints a fresh token (or adds a key) and any delete removes one, so an unchanged -/// map across a call proves it performed NO writes. -std::map snapshotKeyTokens(InMemoryBackend & b) +/// A full key -> incarnation snapshot of the backend, for the previewDeletes write-free invariant: any +/// write mints a fresh incarnation (or adds a key) and any removal drops one, so an unchanged map +/// across a call proves it performed NO writes. +std::map snapshotKeyTokens(CasOperation & op) { std::map out; String cursor; while (true) { - const ListPage page = b.list("", cursor, 100000); + const ListPage page = op.list("", cursor, 100000, Retry::once()); for (const ListedKey & k : page.keys) - out[k.key] = k.token ? k.token->value : String{}; + out[k.key] = k.etag ? k.etag->render() : String{}; if (page.next_cursor.empty()) break; cursor = page.next_cursor; @@ -417,12 +439,13 @@ TEST(CASGCLease, DeadIncumbentThenRevivedIncumbentWinsRace) EXPECT_EQ(readState(*b, *s).lease.owner, kGcA); } -TEST(CASGCLease, ConcurrentStealLosesCas) +/// The steal's refused precondition, case one of two: the re-decide sees the SAME frozen incumbent. +/// `readModifyWrite` does not treat a refusal as terminal -- it re-decides against what the refused +/// write's own resolve read observed and sends another attempt -- so a contender that was steal-eligible +/// still is, and the steal lands inside the SAME round. The two cases are separate tests because they +/// differ only in what the store holds at the re-decide, and that is the whole decision. +TEST(CASGCLease, RefusedStealAgainstAFrozenIncumbentRetriesAndLands) { - /// The CAS-race horn: gc2 is steal-eligible and goes for the CAS, but gc/state moved under it - /// (injected one-shot conflict). It must back off (never acquired=true off a lost CAS) and the - /// owner on storage must be unperturbed. The injected conflict left the object unchanged, so gc2's - /// NEXT round is steal-eligible again and succeeds. std::shared_ptr b; auto s = openTestPool(b); Gc gc1(s, kGcA); @@ -431,25 +454,80 @@ TEST(CASGCLease, ConcurrentStealLosesCas) ASSERT_TRUE(gc1.runRegularRound().acquired_lease); const GcState st0 = readState(*b, *s); EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1; gc1 stalls now - b->failNextCasPut(s->layout().gcStateKey()); /// inject: gc2's steal CAS conflicts - EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// steal attempt loses the CAS => back off + /// `refuseNextWrite` refuses without touching the object, which is what "the incumbent is still + /// frozen" looks like to the re-decide. + b->refuseNextWrite(s->layout().gcStateKey()); + EXPECT_TRUE(gc2.runRegularRound().acquired_lease) + << "a refused steal against an unmoved tuple is retried inside the same call and lands"; const GcState st1 = readState(*b, *s); - EXPECT_EQ(st1.lease.owner, kGcA); /// unchanged - EXPECT_EQ(st1.lease.seq, st0.lease.seq); /// nothing clobbered - EXPECT_TRUE(gc2.runRegularRound().acquired_lease); /// still steal-eligible => succeeds now - EXPECT_EQ(readState(*b, *s).lease.owner, kGcB); + EXPECT_EQ(st1.lease.owner, kGcB); + EXPECT_GT(st1.lease.seq, st0.lease.seq); +} + +/// Case two: the re-decide sees a MOVED tuple. This is what a real refused precondition means -- a +/// store refuses only what changed -- and the incumbent's own renewal is the change. The contender must +/// then decline rather than steal, because a moved tuple is proof of life, and it must leave the +/// incumbent's lease exactly as the incumbent wrote it. +TEST(CASGCLease, RefusedStealWhoseRedecideSeesAMovedTupleDeclines) +{ + class StealRaceBackend : public InMemoryBackend + { + public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + if (arm && expected_value && key == gc_state_key) + { + const auto stored = InMemoryBackend::read(key, access); + if (stored) + { + arm = false; + /// The incumbent renews while this contender is deciding: bump its own `seq` under + /// its own owner, then refuse. Written before the refusal is returned, so the + /// resolve read the engine makes next observes the moved tuple. + GcState renewed = decodeGcState(stored->bytes); + ++renewed.lease.seq; + (void)InMemoryBackend::write(key, encodeGcState(renewed), stored->value, access); + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + String gc_state_key; + bool arm = false; + }; + + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + b->gc_state_key = s->layout().gcStateKey(); + Gc gc1(s, kGcA); + Gc gc2(s, kGcB); + + ASSERT_TRUE(gc1.runRegularRound().acquired_lease); + const GcState st0 = readState(*b, *s); + EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// obs #1; gc1 stalls now + b->arm = true; + EXPECT_FALSE(gc2.runRegularRound().acquired_lease) + << "the re-decide sees a renewed incumbent, which is proof of life: decline, never steal"; + const GcState st1 = readState(*b, *s); + EXPECT_EQ(st1.lease.owner, kGcA) + << "gc2 wrote nothing that landed: a steal would have put kGcB here"; + EXPECT_EQ(st1.lease.seq, st0.lease.seq + 1) + << "the only write that landed is the incumbent's injected renewal"; } TEST(CASGCLease, CreateConflictReReadsWithinTheBound) { - /// The create-Conflict branch: a fresh pool where the create-if-absent CAS conflicts (one-shot). - /// The contender re-reads and falls through within its bounded (2) CAS attempts — the re-read still - /// finds the key absent, so the second attempt creates and acquires. + /// The create-Conflict branch: a fresh pool where the create-if-absent write is refused (one-shot). + /// `readModifyWrite` re-decides against the refused write's own resolve read, which still finds the + /// key absent, so the second attempt creates and acquires. The retry policy's deadline is the bound. std::shared_ptr b; auto s = openTestPool(b); Gc gc(s, hexToU128("0000000000000000000000000000000c")); - b->failNextCasPut(s->layout().gcStateKey()); + b->refuseNextWrite(s->layout().gcStateKey()); EXPECT_TRUE(gc.runRegularRound().acquired_lease); const GcState st = readState(*b, *s); EXPECT_EQ(st.lease.owner, hexToU128("0000000000000000000000000000000c")); @@ -467,15 +545,15 @@ TEST(CASGCLease, CtorFailsClosedOnBadArguments) TEST(CASGCLease, IncumbentRenewConflictRetriesOnceAndAcquires) { - /// The incumbent's own renew CAS conflicts (one-shot). Re-read sees our own ownership => the renew - /// is retried ONCE within the bounded (2) CAS attempts => acquired. Never acquired=true without a - /// Committed CAS — storage must carry the seq the SECOND (committed) attempt wrote. + /// The incumbent's own renew write is refused (one-shot). The re-decide sees our own ownership, so + /// the renew is retried and acquires; the retry policy's deadline is the bound. Never + /// acquired=true without a committed write — storage must carry the seq the SECOND attempt wrote. std::shared_ptr b; auto s = openTestPool(b); Gc gc(s, hexToU128("0000000000000000000000000000000d")); ASSERT_TRUE(gc.runRegularRound().acquired_lease); /// create: seq 1 - b->failNextCasPut(s->layout().gcStateKey()); /// inject: the renew CAS conflicts + b->refuseNextWrite(s->layout().gcStateKey()); /// inject: the renew CAS conflicts EXPECT_TRUE(gc.runRegularRound().acquired_lease); /// re-read (still us) => retried once const GcState st = readState(*b, *s); EXPECT_EQ(st.lease.owner, hexToU128("0000000000000000000000000000000d")); @@ -495,9 +573,10 @@ TEST(CASGCLease, VanishedStateAfterObservationFailsClosed) ASSERT_TRUE(gc1.runRegularRound().acquired_lease); EXPECT_FALSE(gc2.runRegularRound().acquired_lease); /// gc2 records an observation - const auto head = b->head(s->layout().gcStateKey()); /// out-of-model wipe (raw delete) - ASSERT_TRUE(head.exists); - ASSERT_EQ(b->deleteExact(s->layout().gcStateKey(), head.token).kind, DeleteOutcome::Kind::Deleted); + OperationForTest wipe_op(*b); /// out-of-model wipe (raw delete) + const auto meta = (*wipe_op).head(s->layout().gcStateKey(), Retry::standard()); + ASSERT_TRUE(meta.has_value()); + ASSERT_EQ((*wipe_op).remove(s->layout().gcStateKey(), meta->etag, Retry::standard()), Removal::Removed); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { gc2.runRegularRound(); }); } @@ -540,9 +619,9 @@ TEST(CASGCRound, PublishDropReclaimsBlobAndManifestToFixpoint) EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))); } -/// retired-in-snapshot T4: after a round condemns one blob, the ADOPTED fold seal's per-shard +/// After a round condemns one blob, the ADOPTED fold seal's per-shard /// condemned_summary reflects it (condemned_total == 1, pending_total == 0) — distilled zero-I/O from the -/// kCondemned rows the fold sealed into the snapshot run. +/// RunMarker::Condemned rows the fold sealed into the snapshot run. TEST(CASGCRound, CondemnRoundSealSummaryCountsCondemned) { auto backend = std::make_shared(); @@ -567,7 +646,7 @@ TEST(CASGCRound, CondemnRoundSealSummaryCountsCondemned) store->renewWatermarkOnce(); const GcState st = readState(*backend, *store); seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); for (const RetiredEntry & e : currentRetiredSet(*backend, store->layout(), /*shard*/0)) if (e.ref == DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(DB::UInt128(1))}) condemned = true; @@ -582,7 +661,7 @@ TEST(CASGCRound, CondemnRoundSealSummaryCountsCondemned) << "a non-pending condemned entry records its condemn round"; } -/// retired-in-snapshot T5: `previewDeletes` streams the adopted seal's `kCondemned` rows and reports each +/// retired-in-snapshot T5: `previewDeletes` streams the adopted seal's `RunMarker::Condemned` rows and reports each /// with the STORED condemn-time token — `awaiting_graduation` while newly condemned, then `delete_pending` /// once graduated, and NOTHING once the exact-token redelete has removed the blob. The preview performs no /// HEAD on the condemned rows (the token is durable in-run) and is WRITE-FREE throughout (spec §5 req 1). @@ -602,19 +681,21 @@ TEST(CASGCRound, PreviewReportsCondemnedRowsAndIsWriteFree) EXPECT_TRUE(gc.previewDeletes().empty()) << "a live-referenced blob is never previewed for deletion"; dropRefTransition(*backend, store->layout(), ns, "tbl", r); - runRegularRoundReclaiming(gc); /// condemning round: -1 => in-degree 0 => kCondemned row (not pending) + runRegularRoundReclaiming(gc); /// condemning round: -1 => in-degree 0 => RunMarker::Condemned row (not pending) - /// Write-free contract: a full key->token snapshot must be identical across the previewDeletes call. - const auto before = snapshotKeyTokens(*backend); + /// Write-free contract: a full key->incarnation snapshot must be identical across the call. + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + const auto before = snapshotKeyTokens(op); const std::vector awaiting = gc.previewDeletes(); - const auto after = snapshotKeyTokens(*backend); - EXPECT_EQ(before, after) << "previewDeletes must perform NO writes (put/casPut/overwrite/delete)"; + const auto after = snapshotKeyTokens(op); + EXPECT_EQ(before, after) << "previewDeletes must perform NO writes"; ASSERT_EQ(awaiting.size(), 1u) << "exactly the one condemned blob is previewed"; EXPECT_EQ(awaiting[0].ref, (DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})); EXPECT_EQ(awaiting[0].key, store->layout().blobKey(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(blob)})); EXPECT_EQ(awaiting[0].reason, "awaiting_graduation"); - EXPECT_FALSE(awaiting[0].token.value.empty()) << "must carry the stored condemn-time token"; + EXPECT_FALSE(awaiting[0].token.value.empty()) << "must carry the stored condemn-time incarnation"; EXPECT_GT(awaiting[0].condemn_round, 0u) << "must carry the stored condemn round"; runRegularRoundReclaiming(gc); /// graduation round: entry becomes delete_pending (blob still present) @@ -631,7 +712,7 @@ TEST(CASGCRound, PreviewReportsCondemnedRowsAndIsWriteFree) /// A fully idle fold pure-carries every shard's authoritative rows verbatim. The parent is first made /// non-vacuous with one live blob in each of two shards; the forced no-delta successor must preserve -/// both `btr` rows and the total `cnd` domain byte-for-byte. +/// both `blob_run` rows and the total `condemned` domain byte-for-byte. TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) { auto backend = std::make_shared(); @@ -658,7 +739,7 @@ TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) gc.runRegularRound(); const GcState st1 = readState(*backend, *store); const CasFoldSeal seal1 = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes, + readOf(*backend, store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes, store->layout(), /*gc_shards=*/2); /// No state changes after the parent. The zero defer bound forces an actual fold rather than DEFER, @@ -666,7 +747,7 @@ TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) gc.runRegularRound(); const GcState st2 = readState(*backend, *store); const CasFoldSeal seal2 = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes, + readOf(*backend, store->layout().foldSealKey(st2.snap_generation, st2.snap_attempt))->bytes, store->layout(), /*gc_shards=*/2); /// TOTALITY: both seals carry a summary entry for every gc-shard. @@ -675,9 +756,9 @@ TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) EXPECT_TRUE(seal1.condemned_summary.contains(0) && seal1.condemned_summary.contains(1)); EXPECT_TRUE(seal2.condemned_summary.contains(0) && seal2.condemned_summary.contains(1)); - /// Capacity reserves one widest `btr` row per shard. Pin the production pure-carry seal to the + /// Capacity reserves one widest `blob_run` row per shard. Pin the production pure-carry seal to the /// authoritative grammar that makes that bound sufficient: at most one in-range canonical seq-0 - /// run per shard, beside exactly one `cnd` row for every shard. + /// run per shard, beside exactly one `condemned` row for every shard. bool run_seen[2] = {false, false}; ASSERT_EQ(seal1.blob_target_runs.size(), 2u); ASSERT_EQ(seal2.blob_target_runs.size(), 2u); @@ -689,7 +770,7 @@ TEST(CASGCRound, PureCarryRoundPreservesAuthoritativeShardRowsVerbatim) const auto parsed = store->layout().parseBlobTargetRunKey(run.key); ASSERT_TRUE(parsed.has_value()); EXPECT_EQ(parsed->shard, run.shard); - EXPECT_EQ(parsed->generation, run.generation); + EXPECT_EQ(parsed->generation, run.key_generation); EXPECT_EQ(parsed->seq, 0u); } EXPECT_TRUE(run_seen[0]); @@ -743,7 +824,7 @@ TEST(CASGCRound, NonAdoptedAttemptSealIgnored) /// Plant a decoy fold seal under a DIFFERENT attempt at the SAME generation (a deposed leader's /// unadopted artifact). It must be invisible to the adopted-attempt readers. - backend->putIfAbsent(store->layout().foldSealKey(st.snap_generation, st.snap_attempt + 999), + createRaw(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt + 999), "decoy-seal-bytes"); /// No reader resolves the non-adopted attempt: no throw, and the preview is unchanged by the decoy. @@ -1221,14 +1302,14 @@ TEST(CASGCSnapRetention, PrunesOldGenerationsKeepingLastThree) /// Every generation at or below the floor is fully gone (fold seal absent). for (uint64_t g = 1; g <= floor; ++g) { - EXPECT_FALSE(backend->head(store->layout().foldSealKey(g, st.snap_attempt)).exists) + EXPECT_FALSE(headExists(*backend, store->layout().foldSealKey(g, st.snap_attempt))) << "fold seal of pruned generation " << g << " must be gone"; - EXPECT_FALSE(backend->head(store->layout().blobTargetRunKey(g, st.snap_attempt, /*shard*/0, /*seq*/0)).exists) + EXPECT_FALSE(headExists(*backend, store->layout().blobTargetRunKey(g, st.snap_attempt, /*shard*/0, /*seq*/0))) << "blob-target run of pruned generation " << g << " must be gone"; } /// The fold seal at the current generation survives (the live in-degree view). - EXPECT_TRUE(backend->head(store->layout().foldSealKey(st.snap_generation, st.snap_attempt)).exists) + EXPECT_TRUE(headExists(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))) << "the current generation's seal must NOT be pruned"; /// No-loss: the live blob and owner body are intact throughout retention pruning. @@ -1270,9 +1351,9 @@ TEST(CASGCSnapRetention, WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcom const String decoy_outcomes = store->layout().outcomesKey(old_gen, decoy_attempt, /*round*/0, /*shard*/0); const String decoy_seal = store->layout().foldSealKey(old_gen, decoy_attempt); const String decoy_run = store->layout().blobTargetRunKey(old_gen, decoy_attempt, /*shard*/0, /*seq*/0); - backend->putIfAbsent(decoy_outcomes, "decoy-outcomes"); - backend->putIfAbsent(decoy_seal, "decoy-seal"); - backend->putIfAbsent(decoy_run, "decoy-run"); + createRaw(*backend, decoy_outcomes, "decoy-outcomes"); + createRaw(*backend, decoy_seal, "decoy-seal"); + createRaw(*backend, decoy_run, "decoy-run"); /// Drop the ref so the next fold writes a FRESH run under a newer generation and the adopted seal's /// blob_target ref moves OFF `old_gen`. Under T0 reference-parent carry, a still-referenced generation @@ -1288,12 +1369,12 @@ TEST(CASGCSnapRetention, WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcom ASSERT_GT(st.snap_generation, old_gen + 3) << "generation 1 must be below the retention floor"; /// The ENTIRE gc/gen// subtree — across ALL attempts — must be reclaimed. - EXPECT_FALSE(backend->head(decoy_outcomes).exists) << "non-adopted outcomes log leaked past retention"; - EXPECT_FALSE(backend->head(decoy_seal).exists) << "non-adopted fold seal leaked past retention"; - EXPECT_FALSE(backend->head(decoy_run).exists) << "non-adopted blob-target run leaked past retention"; + EXPECT_FALSE(headExists(*backend, decoy_outcomes)) << "non-adopted outcomes log leaked past retention"; + EXPECT_FALSE(headExists(*backend, decoy_seal)) << "non-adopted fold seal leaked past retention"; + EXPECT_FALSE(headExists(*backend, decoy_run)) << "non-adopted blob-target run leaked past retention"; /// Nothing remains under the old generation prefix at all. - const ListPage residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000); + const ListPage residue = listOf(*backend, store->layout().gcGenPrefix(old_gen), "", 1000); EXPECT_TRUE(residue.keys.empty()) << "old generation prefix must be fully reclaimed; left " << residue.keys.size() << " objects"; @@ -1338,21 +1419,21 @@ TEST(CASGCSnapRetention, PruneRespectsPrefixWholesaleBudgetAndNeverStrandsAParti /// can wholesale-delete in a single pass, regardless of whatever real fold artifacts already live /// there. for (int i = 0; i < 10; ++i) - backend->putIfAbsent(store->layout().gcGenPrefix(old_gen) + "debris" + std::to_string(i), "x"); + createRaw(*backend, store->layout().gcGenPrefix(old_gen) + "debris" + std::to_string(i), "x"); /// Move the ref off `old_gen`'s run (as `WholesalePruneReclaimsAllAttemptsIncludingRetiredOutcomes` /// does) so the WHOLESALE RETENTION PRUNE -- not the one-shot post-CAS hand-off -- is what /// eventually processes this generation once the cursor reaches it. dropRefTransition(*backend, store->layout(), ns, "tbl", r); - size_t previous_residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000).keys.size(); + size_t previous_residue = listOf(*backend, store->layout().gcGenPrefix(old_gen), "", 1000).keys.size(); std::optional drain_start_round; /// first round the residue count actually DROPS std::optional drain_done_round; /// first round the residue reaches zero for (int i = 0; i < 40 && !drain_done_round; ++i) { ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); const GcState st = readState(*backend, *store); - const ListPage residue = backend->list(store->layout().gcGenPrefix(old_gen), "", 1000); + const ListPage residue = listOf(*backend, store->layout().gcGenPrefix(old_gen), "", 1000); if (st.snap_pruned_through >= old_gen) EXPECT_TRUE(residue.keys.empty()) @@ -1410,13 +1491,13 @@ TEST(CASGCSnapRetention, ReclaimsNonAdoptedCurrentGenAttemptViaRetention) const uint64_t orphan_attempt = st.snap_attempt - 1; const String orphan_seal = store->layout().foldSealKey(orphan_gen, orphan_attempt); const String orphan_run = store->layout().blobTargetRunKey(orphan_gen, orphan_attempt, 0, 0); - backend->putIfAbsent(orphan_seal, "orphan-seal"); - backend->putIfAbsent(orphan_run, "orphan-run"); + createRaw(*backend, orphan_seal, "orphan-seal"); + createRaw(*backend, orphan_run, "orphan-run"); /// One more round folds into `orphan_gen` and completes. The orphan must SURVIVE this round — there /// is no current-generation sweep; retention has not yet reached `orphan_gen`. ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - EXPECT_TRUE(backend->head(orphan_seal).exists) + EXPECT_TRUE(headExists(*backend, orphan_seal)) << "orphan must survive its own round — there is no per-round current-gen sweep"; /// Age `orphan_gen` well past the retention floor (keep=3): several more quiescent rounds. The @@ -1428,9 +1509,9 @@ TEST(CASGCSnapRetention, ReclaimsNonAdoptedCurrentGenAttemptViaRetention) const GcState st_after = readState(*backend, *store); ASSERT_GT(st_after.snap_generation, orphan_gen + 3) << "orphan_gen must be below the retention floor"; - EXPECT_FALSE(backend->head(orphan_seal).exists) + EXPECT_FALSE(headExists(*backend, orphan_seal)) << "non-adopted attempt orphan must be reclaimed by wholesale retention once its generation ages out"; - EXPECT_FALSE(backend->head(orphan_run).exists) + EXPECT_FALSE(headExists(*backend, orphan_run)) << "the whole orphan subtree must be reclaimed by wholesale retention"; /// No-loss: the live data is intact throughout. @@ -1465,11 +1546,11 @@ TEST(CASGCRetention, PruneRetainsLiveReferencedRun) /// The gen-1 seal's ref names gen-1's physical run key — capture it so we can assert the OBJECT /// (not just the generation number) survives retention. const auto seal1 = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); ASSERT_EQ(seal1.blob_target_runs.size(), 1u); const String referenced_run_key = seal1.blob_target_runs.front().key; - ASSERT_EQ(seal1.blob_target_runs.front().generation, ref_gen); - ASSERT_TRUE(backend->head(referenced_run_key).exists); + ASSERT_EQ(seal1.blob_target_runs.front().key_generation, ref_gen); + ASSERT_TRUE(headExists(*backend, referenced_run_key)); /// Several idle rounds: no delta, no retired => pure ref-carry. Each round advances the generation /// and, once adopted_generation > keep, drives the retention prune forward. gen-1 is referenced every @@ -1484,16 +1565,16 @@ TEST(CASGCRetention, PruneRetainsLiveReferencedRun) << "the retention cursor must have advanced past the still-referenced generation"; /// The referenced run object is STILL ALIVE despite the cursor passing its generation. - EXPECT_TRUE(backend->head(referenced_run_key).exists) + EXPECT_TRUE(headExists(*backend, referenced_run_key)) << "a run referenced by the live seal must be retained even after the cursor passes its generation"; /// The current seal still references that same physical gen-1 object (carried, not reconstructed), /// and in-degree resolution THROUGH the carried ref still works. const auto seal_now = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); ASSERT_EQ(seal_now.blob_target_runs.size(), 1u); EXPECT_EQ(seal_now.blob_target_runs.front().key, referenced_run_key); - EXPECT_EQ(seal_now.blob_target_runs.front().generation, ref_gen); + EXPECT_EQ(seal_now.blob_target_runs.front().key_generation, ref_gen); EXPECT_EQ(inDegreeOf(*backend, store->layout(), DB::UInt128(1)), 1) << "folding still resolves in-degree through the retained, carried parent ref"; @@ -1521,7 +1602,7 @@ TEST(CASGCRetention, HandOffDeletesSupersededRef) const GcState st1 = readState(*backend, *store); const uint64_t old_gen = st1.snap_generation; const String old_prefix = store->layout().gcGenPrefix(old_gen); - ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) << "gen-1 prefix must be populated"; + ASSERT_FALSE(listOf(*backend, old_prefix, "", 1000).keys.empty()) << "gen-1 prefix must be populated"; /// Idle-carry the gen-1 ref until the retention cursor has advanced strictly PAST gen-1. Until it /// does, a normal prune could still reclaim gen-1 when the ref moves — the hand-off is only load- @@ -1531,7 +1612,7 @@ TEST(CASGCRetention, HandOffDeletesSupersededRef) ASSERT_GT(readState(*backend, *store).snap_pruned_through, old_gen) << "gen-1 must be behind the retention cursor before the hand-off is exercised"; /// gen-1 is retained (referenced) even though the cursor passed it. - ASSERT_FALSE(backend->list(old_prefix, "", 1000).keys.empty()) + ASSERT_FALSE(listOf(*backend, old_prefix, "", 1000).keys.empty()) << "the referenced gen-1 prefix must still exist before the ref moves off it"; /// A real delta: swap the ref to a new manifest naming a different blob. The next fold writes a FRESH @@ -1546,14 +1627,14 @@ TEST(CASGCRetention, HandOffDeletesSupersededRef) /// The seal no longer references gen-1 ... const GcState st_after = readState(*backend, *store); const auto seal_after = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st_after.snap_generation, st_after.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st_after.snap_generation, st_after.snap_attempt))->bytes); for (const RunRef & rr : seal_after.blob_target_runs) - EXPECT_NE(rr.generation, old_gen) << "the live seal must have moved its ref off gen-1"; + EXPECT_NE(rr.key_generation, old_gen) << "the live seal must have moved its ref off gen-1"; /// ... and the post-CAS hand-off delete reclaimed gen-1's WHOLE prefix (not just the single run /// object): seal, attempt subtree, run — all gone. The ordinary prune would have leaked it because its /// cursor is already past gen-1. - const ListPage residue = backend->list(old_prefix, "", 1000); + const ListPage residue = listOf(*backend, old_prefix, "", 1000); EXPECT_TRUE(residue.keys.empty()) << "the superseded gen-1 prefix must be hand-off deleted; left " << residue.keys.size() << " objects"; @@ -1606,13 +1687,13 @@ TEST(CASGCRetention, HandoffOwnBudgetSurvivesAPruneHeavyRound) ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); const uint64_t handoff_gen = readState(*backend, *store).snap_generation; const String handoff_prefix = store->layout().gcGenPrefix(handoff_gen); - ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()); + ASSERT_FALSE(listOf(*backend, handoff_prefix, "", 1000).keys.empty()); for (int i = 0; i < 20; ++i) ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); ASSERT_GT(readState(*backend, *store).snap_pruned_through, handoff_gen) << "the hand-off generation must be behind the cursor before this test is meaningful"; - ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()) + ASSERT_FALSE(listOf(*backend, handoff_prefix, "", 1000).keys.empty()) << "still referenced -- must survive despite the cursor having passed it"; /// The PRUNE-DEBRIS generation: a second table, on the OTHER shard, unreferenced from the start, @@ -1629,7 +1710,7 @@ TEST(CASGCRetention, HandoffOwnBudgetSurvivesAPruneHeavyRound) << "the debris generation must still be ahead of the cursor when its drop folds, or the hand-off " "(not the prune) would claim it"; for (int i = 0; i < 10; ++i) - backend->putIfAbsent(store->layout().gcGenPrefix(debris_gen) + "debris" + std::to_string(i), "x"); + createRaw(*backend, store->layout().gcGenPrefix(debris_gen) + "debris" + std::to_string(i), "x"); dropRefTransition(*backend, store->layout(), ns, "debris", r_debris); /// Drive rounds until the debris generation is MID-DRAIN (prune has started but not yet finished it -- @@ -1638,11 +1719,11 @@ TEST(CASGCRetention, HandoffOwnBudgetSurvivesAPruneHeavyRound) for (int i = 0; i < 20 && !mid_drain; ++i) { ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - const size_t residue = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + const size_t residue = listOf(*backend, store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); mid_drain = residue > 0 && residue < 10; } ASSERT_TRUE(mid_drain) << "the debris generation never reached a partially-drained state to test against"; - ASSERT_FALSE(backend->list(handoff_prefix, "", 1000).keys.empty()) + ASSERT_FALSE(listOf(*backend, handoff_prefix, "", 1000).keys.empty()) << "the hand-off generation must still be intact (untouched) going into the contended round"; /// NOW, in a round where the prune is busy mid-drain on the debris generation (spending its entire @@ -1651,17 +1732,17 @@ TEST(CASGCRetention, HandoffOwnBudgetSurvivesAPruneHeavyRound) writeBlobBody(*backend, store->layout(), blob_keep_2); writeManifestRaw(*backend, store->layout(), ns, r_keep_2, {blobEntryFor("a", blob_keep_2)}); publishCommittedTransition(*backend, store->layout(), ns, "keep", r_keep_1, r_keep_2); - const size_t debris_residue_before = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + const size_t debris_residue_before = listOf(*backend, store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); /// THE LOAD-BEARING ASSERTIONS: the prune spent its whole (separate) budget on the debris generation /// this very round (proving the two really contended for I/O in the same round) ... - const size_t debris_residue_after = backend->list(store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); + const size_t debris_residue_after = listOf(*backend, store->layout().gcGenPrefix(debris_gen), "", 1000).keys.size(); EXPECT_EQ(debris_residue_before - debris_residue_after, 2u) << "the prune must have spent its entire per-round budget on the debris generation this round"; /// ... and the hand-off, drawing from its OWN reserve, still fully reclaimed the generation the ref /// just moved off -- zero, not starved to zero by the prune's consumption. - EXPECT_TRUE(backend->list(handoff_prefix, "", 1000).keys.empty()) + EXPECT_TRUE(listOf(*backend, handoff_prefix, "", 1000).keys.empty()) << "the hand-off must not be starved by a prune-heavy round that exhausted a SEPARATE budget"; } @@ -1692,11 +1773,11 @@ TEST(CASGCRetention, LosingRoundNeverDestroysParentSealGeneration) const GcState st1 = readState(*backend, *store); const uint64_t g_parent = st1.snap_generation; - const auto seal1 = decodeFoldSeal(backend->get(layout.foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); + const auto seal1 = decodeFoldSeal(readOf(*backend, layout.foldSealKey(st1.snap_generation, st1.snap_attempt))->bytes); ASSERT_EQ(seal1.blob_target_runs.size(), 1u); const String parent_run_key = seal1.blob_target_runs.front().key; const String parent_gen_prefix = layout.gcGenPrefix(g_parent); - ASSERT_FALSE(backend->list(parent_gen_prefix, "", 1000).keys.empty()); + ASSERT_FALSE(listOf(*backend, parent_gen_prefix, "", 1000).keys.empty()); /// A real delta: swap the ref to a new manifest naming a different blob. The next fold will move /// shard 0's run OFF `g_parent` onto a fresh generation. @@ -1728,9 +1809,9 @@ TEST(CASGCRetention, LosingRoundNeverDestroysParentSealGeneration) /// GREEN evidence: the losing round's pre-CAS prune must NOT have destroyed `g_parent` — it is still /// exactly what the (unreplaced, still-adopted) parent seal references. - EXPECT_FALSE(backend->list(parent_gen_prefix, "", 1000).keys.empty()) + EXPECT_FALSE(listOf(*backend, parent_gen_prefix, "", 1000).keys.empty()) << "a losing round must never destroy the generation the still-adopted parent seal references"; - EXPECT_TRUE(backend->head(parent_run_key).exists) + EXPECT_TRUE(headExists(*backend, parent_run_key)) << "the parent seal's exact run object must survive a losing round's pre-CAS prune"; /// GC is NOT wedged: gc/state is unchanged (the CAS never committed) and the original blob still @@ -1746,7 +1827,7 @@ TEST(CASGCRetention, LosingRoundNeverDestroysParentSealGeneration) const uint64_t g_after = readState(*backend, *store).snap_generation; ASSERT_GT(g_after, g_parent); for (uint64_t g = g_parent + 1; g < g_after; ++g) - EXPECT_TRUE(backend->list(layout.gcGenPrefix(g), "", 1000).keys.empty()) + EXPECT_TRUE(listOf(*backend, layout.gcGenPrefix(g), "", 1000).keys.empty()) << "generation " << g << " (the losing round's own abandoned attempt debris, referenced by " "neither the parent nor the new proposed seal) must still be reclaimed on a successful " "round — the fix must not disable pruning"; @@ -1782,7 +1863,7 @@ TEST(CASGCSnapRetention, KeepZeroPrunesNothing) { bool seal_present = false; for (uint64_t a = 0; a <= st.snap_attempt && !seal_present; ++a) - seal_present = backend->head(store->layout().foldSealKey(g, a)).exists; + seal_present = headExists(*backend, store->layout().foldSealKey(g, a)); EXPECT_TRUE(seal_present) << "keep==0: seal of generation " << g << " must remain"; } } @@ -1793,7 +1874,7 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) PoolConfig config; config.pool_prefix = "p"; /// The GC runner owns a different mount from the synthetic `test` watermark below. This keeps the - /// cursor-sweep assertions in the parent process without replacing its live keeper incarnation. + /// cursor-sweep assertions in the parent process without replacing its live renewer incarnation. config.server_root_id = "gc-runner"; config.manifest_sweep_list_budget_keys = 1; config.manifest_sweep_delete_budget_keys = 1; @@ -1801,6 +1882,8 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) /// fold-every-round (Phase-4 Lever A would otherwise defer once the pool quiesces). config.gc_fold_max_defer_rounds = 0; auto store = openTestPoolWithConfig(backend, config); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"test/aa@cas@"}; registerNamespaceRaw(*backend, store->layout(), ns); @@ -1808,7 +1891,7 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) const ManifestRef r2 = ref(5, 0xCA02); writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); - setWatermarkMinActive(*backend, store->layout(), "test", r1.writer_epoch, /*min_active*/6); + setWatermarkMinActive(*backend, store->layout(), "test", r1.writer_epoch, /*min_active_build_sequence*/6); /// The §6 deletion premise is a second precondition on every sweep deletion: a manifest of an /// epoch-`E` build is deletable only once the namespace's sealed fold cursor sits in an epoch @@ -1823,7 +1906,7 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) /// injected cursor would prove only that the premise reads a number, not that the number can be /// produced. /// - /// The live publications use build sequences ABOVE the watermark's `min_active`, so the only + /// The live publications use build sequences ABOVE the watermark's `min_active_build_sequence`, so the only /// sweep-ELIGIBLE manifests in the namespace remain the two debris bodies -- the premise, not the /// watermark, is what this test varies. publishAt(*backend, store->layout(), ns, RefTxnId{1, 1}, "tbl", /*build_sequence=*/7, @@ -1851,17 +1934,20 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) writeSealAt(*backend, store->layout(), ns, RefTxnId{1, 2}); publishAt(*backend, store->layout(), ns, RefTxnId{2, 1}, "tbl2", /*build_sequence=*/7, DB::UInt128(0xB10B2), /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 2}); - const std::optional life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns); + const std::optional life = CasRefCatalog::lifeIfCataloged(op, store->layout(), ns); ASSERT_TRUE(life.has_value()); const String ckpt_key = store->layout().refCkptKey(*life); - const auto old_ckpt = backend->get(ckpt_key); + const auto old_ckpt = readOf(*backend, ckpt_key); ASSERT_TRUE(old_ckpt.has_value()); - ASSERT_EQ(backend->putOverwrite(ckpt_key, encodeRefCkpt(RefCkpt{ - .life_epoch = 1, - .committed_through = RefTxnId{2, 1}, - .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = RefTxnId{1, 2}, - }), old_ckpt->token).outcome, PutOutcome::Done); + { + OperationForTest ckpt_op(*backend); + ASSERT_TRUE(std::holds_alternative((*ckpt_op).replace(ckpt_key, encodeRefCkpt(RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, 2}, + }), old_ckpt->etag, Retry::standard()))); + } /// The list budget is one key per round, so reclaiming both debris bodies takes a circuit. for (int round = 0; round < 12; ++round) @@ -1880,7 +1966,7 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) { const GcState st = readState(*backend, *store); const CasFoldSeal seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + readOf(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); const auto it = seal.ref_lives.find(catalogLifeIdForTest(*backend, store->layout(), ns)); ASSERT_NE(it, seal.ref_lives.end()) << "the round must have sealed a coverage row"; EXPECT_FALSE(it->second.coverage.hold.has_value()) << "a held namespace can never reach the premise"; @@ -1900,8 +1986,8 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) foreign_config.server_root_id = "test"; auto invalid_store = openTestPoolWithConfig(foreign_backend, std::move(foreign_config)); const String foreign_mount_key = invalid_store->layout().mountKey("test"); - setWatermarkMinActive(*foreign_backend, invalid_store->layout(), "test", r1.writer_epoch, /*min_active*/6); - const auto occupant_before = foreign_backend->get(foreign_mount_key); + setWatermarkMinActive(*foreign_backend, invalid_store->layout(), "test", r1.writer_epoch, /*min_active_build_sequence*/6); + const auto occupant_before = readOf(*foreign_backend, foreign_mount_key); ASSERT_TRUE(occupant_before.has_value()); const uint64_t violations_before = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); @@ -1912,7 +1998,7 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) violations_before + 1) << "a runtime that never observed a deposition must report the foreign occupant as a broken " "single-writer guarantee"; - const auto occupant_after = foreign_backend->get(foreign_mount_key); + const auto occupant_after = readOf(*foreign_backend, foreign_mount_key); ASSERT_TRUE(occupant_after.has_value()) << "the release must never delete another incarnation's lease"; EXPECT_EQ(occupant_after->bytes, occupant_before->bytes) << "the release must leave the slot byte-for-byte untouched, never stamp our farewell over it"; @@ -1997,3 +2083,242 @@ TEST(CASGCRound, TwoManifestsTwoSourceEdgesDropOneSpares) EXPECT_TRUE(blobExists(*backend, store->layout(), DB::UInt128(1))) << "the blob must survive — the second reference still pins it"; } + +/// ===================== THE REQUEST CONTRACT AT THE ROUND'S OWN WRITES ===================== + +/// The advisory pulse sends AT MOST ONE write, and neither of the two ways it can fail reaches the +/// caller: the next pulse comes on cadence, so a deposed leader must never spend a retry budget +/// fighting for this key. Both halves are asserted by the write COUNT, because a policy that reissued +/// would be invisible in the outcome. +TEST(CASGc, HeartbeatPulseIsOnceAndAConflictIsIgnored) +{ + class HeartbeatFaultBackend : public InMemoryBackend + { + public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + if (key == hb_key) + { + ++hb_writes; + if (throw_next) + { + throw_next = false; + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected heartbeat write outage"); + } + if (refuse_next) + { + refuse_next = false; + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + String hb_key; + size_t hb_writes = 0; + bool throw_next = false; + bool refuse_next = false; + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + backend->hb_key = store->layout().gcHbKey(); + + Gc::pulseHeartbeat(*store, kGcA); + ASSERT_EQ(backend->hb_writes, 1u) << "one pulse is one write"; + + /// AN UNRESOLVED ATTEMPT IS NOT REISSUED. The key is removed first so the write's own resolving + /// read finds nothing: with nothing at the key the attempt's fate is genuinely unknown, which is + /// the only state a reissuing policy would act on. Under `once` the pulse ends there. + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + ASSERT_EQ(op.removeCurrent(backend->hb_key, Retry::once()), Removal::Removed); + backend->throw_next = true; + const size_t before_unresolved = backend->hb_writes; + EXPECT_NO_THROW(Gc::pulseHeartbeat(*store, kGcA)); + EXPECT_EQ(backend->hb_writes, before_unresolved + 1) + << "an unresolved pulse is abandoned, never reissued"; + + /// A REFUSED PRECONDITION is the ordinary race with another pulser: ignored, never thrown. + Gc::pulseHeartbeat(*store, kGcA); + backend->refuse_next = true; + const size_t before_refused = backend->hb_writes; + EXPECT_NO_THROW(Gc::pulseHeartbeat(*store, kGcA)); + EXPECT_EQ(backend->hb_writes, before_refused + 1); +} + +/// The steal is the one destructive decision the lease machine makes, and it is a CONJUNCTION: the +/// lease tuple unchanged across two of this contender's own observations, the heartbeat pair unchanged +/// across the same window, and a caller allowed to steal. Each conjunct is falsified on its own here, +/// against the same frozen incumbent, so a build that dropped any one of them fails exactly one line. +TEST(CASGc, LeaseDecideStealsOnlyWithAllThreeConjuncts) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + + Gc incumbent(store, kGcA); + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + + Gc contender(store, kGcB); + EXPECT_FALSE(contender.runRegularRound({}, /*allow_steal=*/true).acquired_lease) + << "the first tick has no earlier observation to freeze against"; + + Gc::pulseHeartbeat(*store, kGcA); + EXPECT_FALSE(contender.runRegularRound({}, /*allow_steal=*/true).acquired_lease) + << "a moved heartbeat pair is proof of life even with the lease tuple frozen"; + + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + EXPECT_FALSE(contender.runRegularRound({}, /*allow_steal=*/true).acquired_lease) + << "a moved lease tuple is proof of life even with the heartbeat frozen"; + + EXPECT_FALSE(contender.runRegularRound({}, /*allow_steal=*/false).acquired_lease) + << "both observations are frozen, but this caller may not steal"; + + EXPECT_TRUE(contender.runRegularRound({}, /*allow_steal=*/true).acquired_lease) + << "frozen tuple, frozen heartbeat and a caller allowed to steal"; +} + +/// The round commits everything it did in ONE conditional write of `gc/state`. A refused precondition +/// there means another leader advanced the key, so the round is dropped whole: it throws `ABORTED` and +/// adopts no generation. The lease renewal that OPENED the round is a separate, earlier write and +/// stays committed. +TEST(CASGc, RoundCommitConflictDropsTheRound) +{ + class RoundCommitConflictBackend : public InMemoryBackend + { + public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + if (arm && expected_value && key == gc_state_key) + { + const auto stored = InMemoryBackend::read(key, access); + if (stored + && decodeGcState(bytes).snap_generation > decodeGcState(stored->bytes).snap_generation) + { + arm = false; + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + String gc_state_key; + bool arm = false; + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + backend->gc_state_key = store->layout().gcStateKey(); + + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r = ref(1, 0xAA); + writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); + publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + /// Asserts presence rather than dereferencing: `gc/state` does not exist until a round's own lease + /// acquire creates it, and an empty optional here is undefined behaviour, not a failing assertion. + const auto readState = [&] + { + const auto got = op.read(backend->gc_state_key, Retry::once()); + EXPECT_TRUE(got) << "gc/state must exist once a round has acquired the lease"; + return got ? decodeGcState(got->bytes) : GcState{}; + }; + + Gc gc(store, kGc); + /// One honest round first, so the comparison below is against a committed round rather than + /// against a bootstrap: the double is disarmed, so this round's own commit lands. + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + const GcState before = readState(); + backend->arm = true; + try + { + gc.runRegularRound(); + FAIL() << "a refused round commit must end the round"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED); + } + + const GcState after = readState(); + EXPECT_EQ(after.snap_generation, before.snap_generation) + << "a refused commit adopts no generation"; + EXPECT_GT(after.lease.seq, before.lease.seq) + << "the lease renewal that opened the round is a separate, earlier write and stays committed"; +} + +#if USE_AWS_S3 +/// A refused `gc/state` precondition whose resolve read was ITSELF refused: nothing observed the key, +/// so the round must not report a competing leader nobody saw. It still aborts, and the next round +/// re-reads -- the behaviour is unchanged, only the claim the message makes. +/// +/// The S3 gate is the fault's, not the site's: the definitive-refusal classification that makes a +/// resolve read settle nothing rather than be reissued exists only for S3 errors. +TEST(CASGc, RoundCommitUnobservedConflictNamesNoCompetingLeader) +{ + class UnobservedCommitBackend : public InMemoryBackend + { + public: + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override + { + /// Only the ROUND COMMIT advances `round`; the lease renewal writes the same key and must + /// pass through untouched. + if (arm && expected_value && key == gc_state_key) + { + const auto stored = InMemoryBackend::read(key, access); + if (stored && decodeGcState(bytes).round > decodeGcState(stored->bytes).round) + { + arm = false; + refuse_read = true; + return std::unexpected(RawConflict{}); + } + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + std::optional read(const String & key, TransportAccess & access) override + { + if (refuse_read && key == gc_state_key) + { + refuse_read = false; + throw DB::S3Exception("UnobservedCommitBackend: the settling read is definitively refused", + Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); + } + return InMemoryBackend::read(key, access); + } + + String gc_state_key; + bool arm = false; + bool refuse_read = false; + }; + + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + backend->gc_state_key = store->layout().gcStateKey(); + + Gc gc(store, kGc); + ASSERT_TRUE(gc.runRegularRound().acquired_lease); + backend->arm = true; + try + { + gc.runRegularRound(); + FAIL() << "a refused round commit must end the round"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED); + EXPECT_NE(e.message().find("resolve read observed nothing"), String::npos) << e.message(); + EXPECT_EQ(e.message().find("another leader advanced it"), String::npos) + << "nothing observed the key, so no competing leader may be named: " << e.message(); + } +} +#endif diff --git a/src/Disks/tests/gtest_cas_gc_round_defer.cpp b/src/Disks/tests/gtest_cas_gc_round_defer.cpp index a33816a8bac2..43a6a909e86a 100644 --- a/src/Disks/tests/gtest_cas_gc_round_defer.cpp +++ b/src/Disks/tests/gtest_cas_gc_round_defer.cpp @@ -41,7 +41,7 @@ TEST(CASGCRoundDefer, PredicateTruthTable) EXPECT_FALSE(shouldDeferRound(2, false, 8, 3, 8)); // bound reached => force fold } -/// graduationDue (retired-in-snapshot T4): read ZERO-I/O from the adopted seal's condemned_summary. An +/// `graduationDue` reads ZERO-I/O from the adopted seal's `condemned_summary`. An /// entry whose oldest non-pending condemn round crosses current_round forces it true; a delete_pending /// entry forces it true regardless of the round; otherwise false. TEST(CASGCRoundDefer, GraduationDueDetectsDuePendingAndRoundCrossing) @@ -56,7 +56,8 @@ TEST(CASGCRoundDefer, GraduationDueDetectsDuePendingAndRoundCrossing) .oldest_nonpending_condemn_round = 2}}}); Gc gc(store, kGc); - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const GcState state = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); EXPECT_FALSE(gc.graduationDueForTest(state, /*current_round=*/2)) << "oldest non-pending condemn round (2) is not < current_round (2); not yet due to graduate"; @@ -67,7 +68,7 @@ TEST(CASGCRoundDefer, GraduationDueDetectsDuePendingAndRoundCrossing) injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, {{0, CondemnedSummary{.condemned_total = 1, .pending_total = 1, .oldest_nonpending_condemn_round = std::numeric_limits::max()}}}); - const GcState state_pending = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState state_pending = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); EXPECT_TRUE(gc.graduationDueForTest(state_pending, /*current_round=*/0)) << "a delete_pending entry must force graduationDue true regardless of current_round"; @@ -84,13 +85,14 @@ TEST(CASGCRoundDefer, GraduationDueFailsClosedWhenSealMissing) injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/1, {{0, CondemnedSummary{}}}); - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const GcState state = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); /// Delete the adopted seal object (corrupt destructive bookkeeping). const String seal_key = layout.foldSealKey(state.snap_generation, state.snap_attempt); - const HeadResult h = backend->head(seal_key); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->deleteExact(seal_key, h.token).kind, DeleteOutcome::Kind::Deleted); + const auto h = (*raw_op).head(seal_key, Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*raw_op).remove(seal_key, h->etag, Retry::once()), Removal::Removed); Gc gc(store, kGc); EXPECT_TRUE(gc.graduationDueForTest(state, /*current_round=*/5)) @@ -107,7 +109,8 @@ TEST(CASGCRoundDefer, GraduationDueFalseOnAllZeroSummary) injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/2, {{0, CondemnedSummary{}}, {1, CondemnedSummary{}}}); - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const GcState state = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); Gc gc(store, kGc); EXPECT_FALSE(gc.graduationDueForTest(state, /*current_round=*/9)) @@ -116,7 +119,7 @@ TEST(CASGCRoundDefer, GraduationDueFalseOnAllZeroSummary) /// Fail-closed if the summary is NOT total over gc_shards (shard 1 missing). injectCondemnedSummarySeal(*backend, layout, /*generation*/1, /*attempt*/1, /*gc_shards*/2, {{0, CondemnedSummary{}}}); - const GcState partial = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState partial = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); EXPECT_TRUE(gc.graduationDueForTest(partial, /*current_round=*/9)) << "a summary not total over gc_shards is corrupt => fail-closed force-fold"; } @@ -143,7 +146,8 @@ TEST(CASGCRoundDefer, ChangedShardCountIsZeroWhenQuiescent) /// trim, so THIS round's fold seal finally /// captures the shard's actual current token. - const GcState quiescent_state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const GcState quiescent_state = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); EXPECT_EQ(gc.listRefPrefixForTest(quiescent_state).changed_shards, 0u) << "a quiescent shard (listed token == sealed token) must not count as changed"; @@ -172,10 +176,13 @@ TEST(CASGCRoundDefer, HotEnumerationOffersLogsAndSnapshotsButNeverCheckpointOrFi const String snap_key = layout.refSnapshotKey(life, id); const String ckpt_key = layout.refCkptKey(life); const String file_key = layout.namespaceFileKey(life, "f"); - ASSERT_EQ(backend->putIfAbsent(log_key, "log").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(snap_key, "snap").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(ckpt_key, "ckpt").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(file_key, "file").outcome, PutOutcome::Done); + { + OperationForTest seed_op(*backend); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(log_key, "log", Retry::once()))); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(snap_key, "snap", Retry::once()))); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(ckpt_key, "ckpt", Retry::once()))); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(file_key, "file", Retry::once()))); + } backend->resetCounts(); Gc gc(store, kGc); @@ -196,7 +203,10 @@ TEST(CASGCRoundDefer, ListedLifeAbsentFromThePostListCatalogCutIsInertDebris) const Layout & layout = store->layout(); const NamespaceLifeId unknown = NamespaceLifeId::fromCatalogEntry(RootNamespace{"cannot-authorize"}, UInt128{0x456}); const String log_key = layout.refLogKey(unknown, RefTxnId{1, 1}); - ASSERT_EQ(backend->putIfAbsent(log_key, "not-read-on-defer").outcome, PutOutcome::Done); + { + OperationForTest seed_op(*backend); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(log_key, "not-read-on-defer", Retry::once()))); + } backend->resetCounts(); Gc gc(store, kGc); @@ -222,7 +232,10 @@ TEST(CASGCRoundDefer, SnapshotLifeAbsentFromThePostListCatalogCutIsInertDebris) const Layout & layout = store->layout(); const NamespaceLifeId unknown = NamespaceLifeId::fromCatalogEntry(RootNamespace{"cannot-authorize"}, UInt128{0x457}); const String snapshot_key = layout.refSnapshotKey(unknown, RefTxnId{1, 1}); - ASSERT_EQ(backend->putIfAbsent(snapshot_key, "not-read-on-defer").outcome, PutOutcome::Done); + { + OperationForTest seed_op(*backend); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(snapshot_key, "not-read-on-defer", Retry::once()))); + } backend->resetCounts(); Gc gc(store, kGc); @@ -262,7 +275,8 @@ TEST(CASGCRoundDefer, IdleRoundDefersAndReadsNoGeneration) const uint64_t fold_round_gets = backend->getTotal(); EXPECT_GT(fold_round_gets, 0u) << "sanity: a real fold round performs some GETs"; - const auto st_before = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const auto st_before = decodeGcState((*raw_op).read(store->layout().gcStateKey(), Retry::once())->bytes); backend->resetCounts(); const RoundReport rep = gc.runRegularRound(); /// round 2: genuinely quiesced now => must defer @@ -278,7 +292,7 @@ TEST(CASGCRoundDefer, IdleRoundDefersAndReadsNoGeneration) EXPECT_EQ(rep.round, fold_rep.round) << "a deferred round re-adopts the already-committed round, not a fabricated new one"; - const auto st_after = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const auto st_after = decodeGcState((*raw_op).read(store->layout().gcStateKey(), Retry::once())->bytes); EXPECT_EQ(st_after.snap_generation, st_before.snap_generation) << "a deferred round must not mint a new generation (snapshot rebuild elided)"; EXPECT_EQ(st_after.snap_attempt, st_before.snap_attempt); @@ -374,6 +388,8 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/1); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const NamespaceLifeId dead_a = NamespaceLifeId::fromCatalogEntry(RootNamespace{"dead/a"}, UInt128{0xDA}); @@ -381,32 +397,32 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP = NamespaceLifeId::fromCatalogEntry(RootNamespace{"dead/b"}, UInt128{0xDB}); const String key_a = layout.refCkptKey(dead_a); const String key_b = layout.refCkptKey(dead_b); - ASSERT_EQ(backend->putIfAbsent(key_a, "dead-a").outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(key_b, "dead-b").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(key_a, "dead-a", Retry::once()))); + ASSERT_TRUE(std::holds_alternative(op.create(key_b, "dead-b", Retry::once()))); /// Establish real opaque backend progress rather than fabricating a cursor value. One key remains /// after this page and the durable cursor must be non-empty. const NamespaceJanitorResult first_page - = NamespaceJanitor(*backend, layout, 1).runOnePage(false, [] { return true; }); + = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); ASSERT_EQ(first_page.pages, 1u); ASSERT_EQ(first_page.deleted, 1u); - const GcMaintenanceReadResult partial = readGcMaintenanceState(*backend, layout); + const GcMaintenanceReadResult partial = readGcMaintenanceState(op, layout); ASSERT_EQ(partial.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(partial.state); ASSERT_FALSE(partial.state->janitor_cursor.empty()); - ASSERT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 1u); + ASSERT_EQ(static_cast(op.head(key_a, Retry::once()).has_value()) + static_cast(op.head(key_b, Retry::once()).has_value()), 1u); /// Give the forced fold a nonempty, fully proved authoritative universe. The R11 floor correctly /// refuses to open the destructive gate for an empty 0-of-0 universe even in the test-only policy. const RootNamespace live_namespace{"live/frontier@cas@"}; fixture::admitLive(*backend, layout, live_namespace); - ASSERT_EQ(backend->putIfAbsent( + ASSERT_TRUE(std::holds_alternative(op.create( layout.refCkptKey(fixture::fixtureLife(live_namespace)), encodeRefCkpt(RefCkpt{ .life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, - PutOutcome::Done); + .last_epoch_seal = std::nullopt}), + Retry::once()))); backend->resetCounts(); std::vector phases; @@ -435,19 +451,19 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP EXPECT_GE(cleanup->metrics.at("janitor_keys"), 1u); EXPECT_EQ(cleanup->metrics.at("janitor_deleted"), 0u); - const GcMaintenanceReadResult deferred_progress = readGcMaintenanceState(*backend, layout); + const GcMaintenanceReadResult deferred_progress = readGcMaintenanceState(op, layout); ASSERT_EQ(deferred_progress.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(deferred_progress.state); EXPECT_EQ(deferred_progress.state->janitor_cursor, partial.state->janitor_cursor) << "a suppressed DEFER page is undecided and must remain selected for the authoritative fold"; - EXPECT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 1u); + EXPECT_EQ(static_cast(op.head(key_a, Retry::once()).has_value()) + static_cast(op.head(key_b, Retry::once()).has_value()), 1u); - const auto gc_state = backend->get(layout.gcStateKey()); + const auto gc_state = op.read(layout.gcStateKey(), Retry::once()); ASSERT_TRUE(gc_state); const GcState state = decodeGcState(gc_state->bytes); EXPECT_EQ(state.snap_generation, 0u); EXPECT_EQ(state.snap_attempt, 0u); - EXPECT_FALSE(backend->head(layout.foldSealKey(1, 1)).exists) + EXPECT_FALSE(op.head(layout.foldSealKey(1, 1), Retry::once()).has_value()) << "maintenance on DEFER must not publish a fold successor"; backend->resetCounts(); @@ -466,9 +482,9 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP EXPECT_EQ(folded_cleanup->metrics.at("janitor_pages"), 1u); EXPECT_GE(folded_cleanup->metrics.at("janitor_keys"), 1u); EXPECT_EQ(folded_cleanup->metrics.at("janitor_deleted"), 1u); - EXPECT_EQ(static_cast(backend->head(key_a).exists) + static_cast(backend->head(key_b).exists), 0u) + EXPECT_EQ(static_cast(op.head(key_a, Retry::once()).has_value()) + static_cast(op.head(key_b, Retry::once()).has_value()), 0u) << "the fold must retry and delete the exact page that DEFER left undecided"; - const GcMaintenanceReadResult completed = readGcMaintenanceState(*backend, layout); + const GcMaintenanceReadResult completed = readGcMaintenanceState(op, layout); ASSERT_EQ(completed.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(completed.state); EXPECT_TRUE(completed.state->janitor_cursor.empty()); @@ -584,9 +600,9 @@ TEST(CASGCRoundDefer, DueGraduationIsSoleFoldTriggerAtHighThreshold) writeBlobBody(*backend, layout, blob); - /// Seed the adopted fold seal's condemned_summary with B already `delete_pending` (pending_total = 1), - /// mirroring `CASGCRoundDefer.GraduationDueDetectsDuePendingAndRoundCrossing`. Retired-in-snapshot - /// (T4): graduationDue reads this summary ZERO-I/O off the adopted seal — a delete_pending entry forces + /// Seed the adopted fold seal's condemned_summary with B already `delete_pending` (pending_total = 1). + /// `graduationDue` + /// reads this summary ZERO-I/O from the adopted seal — a `delete_pending` entry forces /// it true regardless of the round. At `gc_fold_threshold = 1000` a real condemn -> graduate pipeline of /// `runRegularRound` calls is not usable to set this up: every round before graduation would ITSELF /// defer (nothing due yet, and changed_shards never nears 1000), so the due-pending summary is injected diff --git a/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp b/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp index 2efced9934c7..8d21ca73da19 100644 --- a/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp +++ b/src/Disks/tests/gtest_cas_gc_shard_incarnation.cpp @@ -36,6 +36,22 @@ ManifestRef testRef(uint64_t seq) return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = 1}; } +/// ---- Small raw-fixture request-engine wrappers shared by the tests below ---- + +/// True iff `key` exists. +bool existsAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::once()).has_value(); +} + +/// Unconditional create of a fresh key (the fixture's own setup, never a real conflict). +void createAt(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + EXPECT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + } /// Review I5: `discoverUniverse` is catalog-authoritative (Task 4-C), and this test used to survive @@ -52,6 +68,8 @@ TEST(CASGCShardIncarnation, DiscoveryEqualsPresentShards) { std::shared_ptr backend; auto store = makePoolWithShards(backend, gc_shards); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); const Layout & layout = store->layout(); @@ -63,7 +81,7 @@ TEST(CASGCShardIncarnation, DiscoveryEqualsPresentShards) fixture::admitLive(*backend, layout, ns_live_empty); /// (b) A genuinely Creating entry, admitted directly (step 1 alone -- never completed to Live). - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns_creating, .state = NsState::Creating, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns_creating, .state = NsState::Creating, .incarnation = UInt128(1), .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}}); /// (c) Ref objects present, but the catalog was never told (or has since forgotten): write @@ -72,12 +90,12 @@ TEST(CASGCShardIncarnation, DiscoveryEqualsPresentShards) writeManifestRaw(*backend, layout, ns_uncataloged, testRef(1), {}); publishCommittedTransition(*backend, layout, ns_uncataloged, "part_1", std::nullopt, testRef(1), /*shard=*/0); { - CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); std::erase_if(snap.catalog.entries, [&](const CatalogEntry & e) { return e.ns.string() == ns_uncataloged.string(); }); - const HeadResult h = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, - PutOutcome::Done); + const auto h = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h->etag, Retry::once()))); } const auto universe = gc.discoverUniverseForTest(); @@ -97,7 +115,7 @@ TEST(CASGCShardIncarnation, DiscoveryEqualsPresentShards) EXPECT_TRUE(found_live_empty) << "a Live catalog entry with zero ref objects must still be discovered"; /// Confirm (b) really is still Creating (not merely absent from a differently-shaped universe). - const CasRefCatalog::Snapshot final_snap = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot final_snap = CasRefCatalog::read(op, layout); const auto creating_it = std::find_if(final_snap.catalog.entries.begin(), final_snap.catalog.entries.end(), [&](const CatalogEntry & e) { return e.ns.string() == ns_creating.string(); }); ASSERT_NE(creating_it, final_snap.catalog.entries.end()); @@ -121,10 +139,11 @@ TEST(CASGCShardIncarnation, DuplicateLifeIdStopsDestructiveRoundAndRebuild) .incarnation = UInt128{77}, .removal_started_round = 1}, }; - const auto empty_catalog = backend->get(layout.refCatalogKey()); + OperationForTest op(*backend); + const auto empty_catalog = (*op).read(layout.refCatalogKey(), Retry::once()); ASSERT_TRUE(empty_catalog); - ASSERT_EQ(backend->putOverwrite( - layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(layout.refCatalogKey(), encodeRefCatalog(catalog), empty_catalog->etag, Retry::once()))); backend->resetCounts(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); @@ -141,11 +160,13 @@ TEST(CASGCShardIncarnation, DeadLifeStreamIsOpaqueInertDebris) { std::shared_ptr backend; auto store = makePoolWithShards(backend, /*gc_shards=*/1); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/tblIncarnationSwap"}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = UInt128(11), .creator = std::nullopt}); // Live forbids a creator fence const ManifestRef dead_ref = testRef(1); writeBlobBody(*backend, layout, UInt128(11)); @@ -157,22 +178,22 @@ TEST(CASGCShardIncarnation, DeadLifeStreamIsOpaqueInertDebris) const NamespaceLifeId dead_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(11)); { - CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); const auto it = std::find_if(snap.catalog.entries.begin(), snap.catalog.entries.end(), [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); ASSERT_NE(it, snap.catalog.entries.end()); it->incarnation = UInt128(22); // "recreated" -- same name, different (empty) key space - const HeadResult h = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, - PutOutcome::Done); + const auto h = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h->etag, Retry::once()))); } const NamespaceLifeId current_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(22)); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(current_life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(current_life), encodeRefCkpt(RefCkpt{ .life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::once()))); const ManifestRef current_ref = testRef(2); writeBlobBody(*backend, layout, UInt128(22)); writeManifestRaw(*backend, layout, ns, current_ref, {blobEntryFor("current", UInt128(22))}); @@ -203,32 +224,34 @@ TEST(CASGCShardIncarnation, CurrentLifeCheckpointIsReadByExactKeyOutsideHotList) auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .gc_shards = 1, .gc_fold_max_defer_rounds = 0}); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/tblOrdinaryRebirth"}; - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = UInt128(11), .creator = std::nullopt}); CatalogEntry after_rebirth{.ns = ns, .state = NsState::Live, .incarnation = UInt128(22), .creator = std::nullopt}; { - CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); const auto it = std::find_if(snap.catalog.entries.begin(), snap.catalog.entries.end(), [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); ASSERT_NE(it, snap.catalog.entries.end()); *it = after_rebirth; // "recreated" -- same name, new (current) incarnation 22 - const HeadResult h = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, - PutOutcome::Done); + const auto h = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h->etag, Retry::once()))); } /// The successor's own genesis `_ckpt`, published for the current physical life. Hiding it from /// LIST must be irrelevant because the walk obtains state only through exact GETs. const NamespaceLifeId current_life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(22)); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(current_life), + createAt(*backend, layout.refCkptKey(current_life), encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt})); backend->hide(layout.refCkptKey(current_life)); backend->resetCounts(); std::vector phases; @@ -246,7 +269,7 @@ TEST(CASGCShardIncarnation, CurrentLifeCheckpointIsReadByExactKeyOutsideHotList) << "the only broader LIST is the separately paced janitor page"; EXPECT_EQ(backend->holesServed(), 1u) << "the hidden checkpoint is omitted only from the janitor's broad page, never from the hot stream LIST"; - EXPECT_TRUE(backend->head(layout.refCkptKey(current_life)).exists) + EXPECT_TRUE(existsAt(*backend, layout.refCkptKey(current_life))) << "the post-page catalog cut retains the current life even when LIST omitted its checkpoint"; const auto cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) { @@ -269,6 +292,8 @@ TEST(CASGCShardIncarnation, UncatalogedStreamLifeDefersWithoutInventingNamespace { std::shared_ptr backend; auto store = makePoolWithShards(backend, /*gc_shards=*/1); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/tblForgotten"}; @@ -278,17 +303,17 @@ TEST(CASGCShardIncarnation, UncatalogedStreamLifeDefersWithoutInventingNamespace const NamespaceLifeId forgotten_life = store->namespaceLife(ns); { - CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); std::erase_if(snap.catalog.entries, [&](const CatalogEntry & e) { return e.ns.string() == ns.string(); }); - const HeadResult h = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h.token).outcome, - PutOutcome::Done); + const auto h = op.head(layout.refCatalogKey(), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), h->etag, Retry::once()))); } const RoundReport report = gc.runRegularRound({}, /*allow_steal=*/true, UniversePolicy::Authoritative); EXPECT_TRUE(report.anomalies.empty()); - EXPECT_FALSE(backend->list(layout.namespaceStreamPrefix(forgotten_life), "", 100).keys.empty()); + EXPECT_FALSE(op.list(layout.namespaceStreamPrefix(forgotten_life), "", 100, Retry::once()).keys.empty()); } /// State-tree objects are point-addressed only. A stalled creator's checkpoint and an unowned opaque @@ -298,28 +323,30 @@ TEST(CASGCShardIncarnation, StateCheckpointsOutsideCatalogAreInertToHotWalk) auto backend = std::make_shared(); auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_shards = 1}); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); Gc gc(store, hexToU128("0000000000000000000000000000000a")); const Layout & layout = store->layout(); const RootNamespace creating_ns{"srv1/tblStalledBirth"}; const RootNamespace unrelated_gone_ns{"srv1/tblGenuinelyGone"}; /// Step 1 of createNamespace: insert the Creating entry with a live creator fence. - CasRefCatalog::casAdmitEntry(*backend, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = creating_ns, .state = NsState::Creating, + CasRefCatalog::casAdmitEntry(op, layout, store->poolConfig().gc_shards, CatalogEntry{.ns = creating_ns, .state = NsState::Creating, .incarnation = UInt128(33), .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}}); /// Step 2, without step 3: publish the genesis `_ckpt` directly, at the SAME incarnation the /// Creating entry names -- exactly what `completeCreation` durably leaves behind if the creator /// crashes between its own steps 2 and 3. const NamespaceLifeId creating_life = NamespaceLifeId::fromCatalogEntry(creating_ns, UInt128(33)); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(creating_life), + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(creating_life), encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::once()))); /// Opaque state debris with no corresponding catalog entry. const NamespaceLifeId gone_life = NamespaceLifeId::fromCatalogEntry(unrelated_gone_ns, UInt128(44)); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(gone_life), + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(gone_life), encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::once()))); /// Add one fully current stream so the round performs a fold rather than stopping at an empty /// walk. Catalog and checkpoint admission keep this traffic out of the janitor's dead-life set, @@ -327,9 +354,9 @@ TEST(CASGCShardIncarnation, StateCheckpointsOutsideCatalogAreInertToHotWalk) const RootNamespace ordinary_ns{"srv1/tblOrdinaryTraffic"}; fixture::admitLive(*backend, layout, ordinary_ns); const NamespaceLifeId ordinary_life = fixture::fixtureLife(ordinary_ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(ordinary_life), + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(ordinary_life), encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::once()))); appendRefLogSeed(*backend, layout, ordinary_ns, {}); backend->resetCounts(); @@ -360,9 +387,9 @@ TEST(CASGCShardIncarnation, StateCheckpointsOutsideCatalogAreInertToHotWalk) << "Creating is retained by the janitor cut but excluded from hot checkpoint intake"; EXPECT_EQ(backend->getCount(layout.refCkptKey(gone_life)), 0u) << "uncataloged state debris is classified by the janitor page, never exact-read by the hot walk"; - EXPECT_TRUE(backend->head(layout.refCkptKey(creating_life)).exists); - EXPECT_TRUE(backend->head(layout.refCkptKey(ordinary_life)).exists); - EXPECT_FALSE(backend->head(layout.refCkptKey(gone_life)).exists) + EXPECT_TRUE(op.head(layout.refCkptKey(creating_life), Retry::once()).has_value()); + EXPECT_TRUE(op.head(layout.refCkptKey(ordinary_life), Retry::once()).has_value()); + EXPECT_FALSE(op.head(layout.refCkptKey(gone_life), Retry::once()).has_value()) << "catalog absence is inert to the hot walk but authorizes the later janitor exact-token delete"; const auto cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) { @@ -428,6 +455,8 @@ TEST(CASGCShardIncarnation, NewbornPrecommitProtectsDedupBlobAgainstConcurrentDr { std::shared_ptr backend; auto store = makePoolWithShards(backend, gc_shards); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns_b{"srv1/tblB"}; /// --- Phase 1: Write b1's body before any GC. --- @@ -454,9 +483,9 @@ TEST(CASGCShardIncarnation, NewbornPrecommitProtectsDedupBlobAgainstConcurrentDr store->renewWatermarkOnce(); } const String b1_key = store->layout().blobKey(b1_ref); - ASSERT_TRUE(backend->head(b1_key).exists) - << "b1 body must be present after the seed putBlob"; - const Token b1_token = backend->head(b1_key).token; + const std::optional b1_observed = op.head(b1_key, Retry::once()); + ASSERT_TRUE(b1_observed) << "b1 body must be present after the seed putBlob"; + const PersistedEtag b1_token = PersistedEtag::capture(b1_observed->etag); /// --- Phase 2: Inject gc/state at round 1 with b1 CONDEMNED (body still present). --- /// This simulates GC having advanced to round 1 and retired b1 (condemned token recorded @@ -507,9 +536,12 @@ TEST(CASGCShardIncarnation, NewbornPrecommitProtectsDedupBlobAgainstConcurrentDr EXPECT_TRUE(store->resolveRef(ns_b, "part_b1").has_value()) << "gc_shards=" << gc_shards << ": the ref must commit"; /// The condemned token is bound UNCHANGED — no displacement happens (and none is needed). - EXPECT_EQ(backend->head(b1_key).token, b1_token) - << "gc_shards=" << gc_shards << ": no copy-forward under the Phase-A contract — the token " - "stays; the folded edge will spare it at the next fold (no round runs here to delete it)"; + const std::optional b1_after = op.head(b1_key, Retry::once()); + ASSERT_TRUE(b1_after); + EXPECT_TRUE(b1_token.matches(b1_after->etag)) + << "gc_shards=" << gc_shards << ": no copy-forward under the Phase-A contract — the " + "incarnation stays; the folded edge will spare it at the next fold (no round runs " + "here to delete it)"; /// INV-NO-DANGLE: the body is present and no GC round ever runs in this test to fold the /// precommit/committed edge; a real deployment's next fold would see net in-degree >= 1 and diff --git a/src/Disks/tests/gtest_cas_gc_shard_plan.cpp b/src/Disks/tests/gtest_cas_gc_shard_plan.cpp index c45cd3ff772d..81c3af12fa86 100644 --- a/src/Disks/tests/gtest_cas_gc_shard_plan.cpp +++ b/src/Disks/tests/gtest_cas_gc_shard_plan.cpp @@ -129,6 +129,8 @@ TEST(CASGCShardReducer, MergesDeltasToInDegree) /// Reduce: each reducer merges its shard's deltas into generation 1 (prior = 0 = fresh). auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); ShardReducer r0(0, 2); @@ -139,9 +141,9 @@ TEST(CASGCShardReducer, MergesDeltasToInDegree) EXPECT_TRUE(r1.owns(b2)) << "r1 must own b2"; EXPECT_FALSE(r1.owns(b1)) << "r1 must not own b1"; - const auto runs0 = r0.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + const auto runs0 = r0.reduce(op, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, std::move(buckets[0])); - const auto runs1 = r1.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + const auto runs1 = r1.reduce(op, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, std::move(buckets[1])); ASSERT_EQ(runs0.size(), 1u) << "shard-0 reduce must produce exactly one RunRef"; @@ -300,13 +302,15 @@ TEST(CASGCShardCoordinator, ShardedFoldRoutesDeltasToOwningShards) buckets[blobShard(d.ref, kGcShards)].push_back(d); auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); std::vector> shard_runs(kGcShards); for (uint64_t shard = 0; shard < kGcShards; ++shard) { ShardReducer reducer{shard, kGcShards}; - shard_runs[shard] = reducer.reduce(*backend, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, + shard_runs[shard] = reducer.reduce(op, layout, /*prior_runs=*/{}, /*new_generation=*/1, /*attempt=*/0, std::move(buckets[shard])); } @@ -459,6 +463,8 @@ TEST(CASGCShardTwoReplica, DisjointShardsConcurrentPerShardRuns) ASSERT_EQ(blobShard(b1, kGcShards), 1u) << "b1 must route to shard 1"; auto backend = std::make_shared(); + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); /// (a) DISJOINTNESS — verify `owns` predicate before any reduce. @@ -485,18 +491,18 @@ TEST(CASGCShardTwoReplica, DisjointShardsConcurrentPerShardRuns) /// (b) PER-SHARD RUNS — drive both reducers. /// /// Run shard-0 reducer (simulates the shard-0 replica's work). - const auto runs0 = r0.reduce(*backend, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket0)); + const auto runs0 = r0.reduce(op, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket0)); ASSERT_FALSE(runs0.empty()) << "shard-0 reducer must produce at least one RunRef"; /// Run shard-1 reducer (simulates the shard-1 replica's work, interleaved from the test thread). - const auto runs1 = r1.reduce(*backend, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket1)); + const auto runs1 = r1.reduce(op, layout, /*prior_runs=*/{}, kNewGen, kAttempt, std::move(bucket1)); ASSERT_FALSE(runs1.empty()) << "shard-1 reducer must produce at least one RunRef"; /// The blob-target runs for both shards are durably present (the reducer's write-once `putIfAbsent`), /// at disjoint object keys. - EXPECT_TRUE(backend->head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/0, /*seq=*/0)).exists) + EXPECT_TRUE(op.head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/0, /*seq=*/0), Retry::once()).has_value()) << "shard-0 blob-target run must be durably written by r0.reduce"; - EXPECT_TRUE(backend->head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/1, /*seq=*/0)).exists) + EXPECT_TRUE(op.head(layout.blobTargetRunKey(kNewGen, kAttempt, /*shard=*/1, /*seq=*/0), Retry::once()).has_value()) << "shard-1 blob-target run must be durably written by r1.reduce"; /// (c) MERGED IN-DEGREE — the merged in-degrees across both shards equal the expected edge multiset. @@ -553,18 +559,19 @@ TEST(CASGCShardRetireDrain, ReclaimsDroppableBlobOwnedByNonZeroShard) const ManifestId id0{ns, r0}; const ManifestId id1{ns, r1}; + OperationForTest verify_op(*backend); /// Local blobExists (the round-level helper is file-local to gtest_cas_gc_round.cpp). auto blobExists = [&](const UInt128 & hash) { - return backend->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + return (*verify_op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}), Retry::once()).has_value(); }; auto manifestExists = [&](const ManifestId & id) { - return backend->head(layout.manifestKey(id)).exists; + return (*verify_op).head(layout.manifestKey(id), Retry::once()).has_value(); }; /// Whether ANY gc-shard still holds an in-flight condemned entry (the ack-floor deletion pipeline is - /// in flight while this is true). Retired-in-snapshot (T4): reconstructed from the adopted fold seal's - /// kCondemned rows across all shards, not a separate retired list. + /// in flight while this is true). Condemned state is reconstructed from the adopted fold seal's + /// RunMarker::Condemned rows across all shards, not a separate retired list. auto anyRetiredPending = [&] { return anyCondemnedInSeal(*backend, layout); @@ -602,7 +609,7 @@ TEST(CASGCShardRetireDrain, ReclaimsDroppableBlobOwnedByNonZeroShard) /// While both refs are live: each blob's in-degree is 1 in its OWNING shard's run, and nothing is /// collected (no-loss). Derive generation/attempt from gc/state — never hardcode. - const GcState live = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState live = decodeGcState((*verify_op).read(layout.gcStateKey(), Retry::once())->bytes); ASSERT_GT(live.snap_generation, 0u); ASSERT_EQ(live.gc_shards, kGcShards) << "the pool must be running with gc_shards=2"; EXPECT_EQ(inDegreeInRuns(*backend, runsForShard(*backend, layout, /*shard=*/0), BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(blob_shard0)}), 1) diff --git a/src/Disks/tests/gtest_cas_gc_state_format.cpp b/src/Disks/tests/gtest_cas_gc_state_format.cpp index aa661429f813..360063e76bdd 100644 --- a/src/Disks/tests/gtest_cas_gc_state_format.cpp +++ b/src/Disks/tests/gtest_cas_gc_state_format.cpp @@ -10,6 +10,8 @@ namespace DB::ErrorCodes extern const int LOGICAL_ERROR; } +CAS_BATTERY_COVERS(GcState); + TEST(CASFormatBattery, GcState) { GcState s; @@ -24,10 +26,12 @@ TEST(CASFormatBattery, GcState) [&] { return sealObject(FormatId::GcState, encodeGcState(s)); }, [](std::string_view d) { decodeGcState(std::string(openObject(FormatId::GcState, d))); }, currentFormatHeader("cas_gc_state") + - "{\"rnd\":\"4\",\"gcs\":1,\"sg\":\"9\",\"spt\":\"7\",\"sa\":\"3\",\"msc\":\"\"," - "\"lo\":\"00000000000000000000000000000001\",\"ls\":\"12\"}\n"}); + "{\"round\":\"4\",\"gc_shards\":1,\"snap_generation\":\"9\",\"snap_pruned_through\":\"7\",\"snap_attempt\":\"3\",\"manifest_sweep_cursor\":\"\"," + "\"lease_owner\":\"00000000000000000000000000000001\",\"lease_seq\":\"12\"}\n"}); } +CAS_BATTERY_COVERS(GcHeartbeat); + TEST(CASFormatBattery, GcHeartbeat) { GcHeartbeat hb{UInt128(1), 1741}; @@ -35,7 +39,7 @@ TEST(CASFormatBattery, GcHeartbeat) [&] { return sealObject(FormatId::GcHeartbeat, encodeGcHeartbeat(hb)); }, [](std::string_view d) { decodeGcHeartbeat(std::string(openObject(FormatId::GcHeartbeat, d))); }, currentFormatHeader("cas_gc_hb") + - "{\"by\":\"00000000000000000000000000000001\",\"seq\":\"1741\"}\n"}); + "{\"owner\":\"00000000000000000000000000000001\",\"hb_seq\":\"1741\"}\n"}); } /// ---------- field round-trips (migrated from gtest_cas_gc_formats.cpp, re-pointed at the text codec) ---------- @@ -83,11 +87,11 @@ TEST(CASGCStateFormat, DefaultsRoundTrip) TEST(CASGCStateFormat, RejectsZeroGcShards) { - /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes - /// the header gate, which is the point — the BODY is what has to fail here. - const String bad = "{\"type\":\"cas_gc_state\",\"v\":3}\n" - "{\"rnd\":\"0\",\"gcs\":0,\"sg\":\"0\",\"spt\":\"0\",\"sa\":\"0\",\"msc\":\"\"," - "\"lo\":\"00000000000000000000000000000000\",\"ls\":\"0\"}\n"; + /// `v:1` is the baseline generation, so it always passes the header gate -- the BODY is what has + /// to fail here. + const String bad = "{\"type\":\"cas_gc_state\",\"v\":1}\n" + "{\"round\":\"0\",\"gc_shards\":0,\"snap_generation\":\"0\",\"snap_pruned_through\":\"0\",\"snap_attempt\":\"0\",\"manifest_sweep_cursor\":\"\"," + "\"lease_owner\":\"00000000000000000000000000000000\",\"lease_seq\":\"0\"}\n"; EXPECT_THROW(decodeGcState(bad), DB::Exception); } @@ -123,13 +127,13 @@ TEST(CASGCStateFormatDeathTest, RejectsZeroGcShardsOnEncodeAborts) TEST(CASGCStateFormat, RejectsAbsentGcShards) { - /// An absent gcs key must fail closed (the writer always emits it) rather than silently defaulting + /// An absent gc_shards key must fail closed (the writer always emits it) rather than silently defaulting /// to the struct's gc_shards = 1 — a missing shard count means a corrupt object, not "use the floor". - /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes - /// the header gate, which is the point — the BODY is what has to fail here. - const String bad = "{\"type\":\"cas_gc_state\",\"v\":3}\n" - "{\"rnd\":\"0\",\"sg\":\"0\",\"spt\":\"0\",\"sa\":\"0\",\"msc\":\"\"," - "\"lo\":\"00000000000000000000000000000000\",\"ls\":\"0\"}\n"; + /// `v:1` is the baseline generation, so it always passes the header gate -- the BODY is what has + /// to fail here. + const String bad = "{\"type\":\"cas_gc_state\",\"v\":1}\n" + "{\"round\":\"0\",\"snap_generation\":\"0\",\"snap_pruned_through\":\"0\",\"snap_attempt\":\"0\",\"manifest_sweep_cursor\":\"\"," + "\"lease_owner\":\"00000000000000000000000000000000\",\"lease_seq\":\"0\"}\n"; EXPECT_THROW(decodeGcState(bad), DB::Exception); } @@ -157,9 +161,9 @@ TEST(CASGCHeartbeatFormat, RoundTripAndBoundaries) TEST(CASGCHeartbeatFormat, RejectsMissingIdentityFields) { - /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes - /// the header gate, which is the point — the BODY is what has to fail here. - const String header = "{\"type\":\"cas_gc_hb\",\"v\":3}\n"; + /// `v:1` is the baseline generation, so it always passes the header gate -- the BODY is what has + /// to fail here. + const String header = "{\"type\":\"cas_gc_hb\",\"v\":1}\n"; const auto expectCorrupted = [](const String & data) { @@ -174,6 +178,6 @@ TEST(CASGCHeartbeatFormat, RejectsMissingIdentityFields) } }; - expectCorrupted(header + "{\"seq\":\"1741\"}\n"); - expectCorrupted(header + "{\"by\":\"00000000000000000000000000000001\"}\n"); + expectCorrupted(header + "{\"hb_seq\":\"1741\"}\n"); + expectCorrupted(header + "{\"owner\":\"00000000000000000000000000000001\"}\n"); } diff --git a/src/Disks/tests/gtest_cas_gc_stop_start.cpp b/src/Disks/tests/gtest_cas_gc_stop_start.cpp index fa4c864cc8de..6079a2b3436d 100644 --- a/src/Disks/tests/gtest_cas_gc_stop_start.cpp +++ b/src/Disks/tests/gtest_cas_gc_stop_start.cpp @@ -63,13 +63,14 @@ const std::string kSrid = "test"; /// gtest_cas_lifecycle_condition.cpp's helper — used by the operator-STOP-persistence test below. void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, DB::Cas::Retry::once()); ASSERT_TRUE(got.has_value()); DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, - DB::Cas::PutOutcome::Done); + const auto put = (*op).replace(mount_key, DB::Cas::encodeMountLease(m), got->etag, DB::Cas::Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); } /// A real `ContentAddressedMetadataStorage` over a fresh, unique local object storage. `context == nullptr` diff --git a/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp new file mode 100644 index 000000000000..8b5e31dafff8 --- /dev/null +++ b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp @@ -0,0 +1,484 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/// A disk's teardown must not wait out a GC round. The pool's teardown flag is the open request +/// plane's fence, so a round in flight is refused at its next request, its next retry sleep or its +/// next streamed refill; the joins stay and the round becomes short. These tests pin the arm, the +/// plane wiring, the sleep wiring, and the scheduler's behaviour around a round that was cut. + +namespace DB::ErrorCodes +{ +extern const int NETWORK_ERROR; +} + +namespace CurrentMetrics +{ +extern const Metric LocalThread; +extern const Metric LocalThreadActive; +extern const Metric LocalThreadScheduled; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +PoolPtr openPlainPool(const std::shared_ptr & backend, PoolConfig config = {}) +{ + config.pool_prefix = "p"; + config.server_root_id = "test"; + return Pool::open(backend, config); +} + +/// A gate a test opens explicitly, so a thread can be held in flight without a sleep. Bounded, and it +/// names what it waited on: an unbounded wait on a premise that stopped holding hangs the binary. +struct Gate +{ + void wait(std::string_view name) + { + std::unique_lock lock(m); + if (!cv.wait_for(lock, std::chrono::seconds(60), [this] { return open_; })) + ADD_FAILURE() << "timed out waiting for '" << name << "'"; + } + void open() + { + std::lock_guard lock(m); + open_ = true; + cv.notify_all(); + } + std::mutex m; + std::condition_variable cv; + bool open_ = false; +}; + +/// Opens its gate on every exit from the scope, so a failing assertion cannot strand the thread +/// parked behind it. +struct GateOpenedOnExit +{ + explicit GateOpenedOnExit(Gate & gate_) : gate(gate_) {} + GateOpenedOnExit(const GateOpenedOnExit &) = delete; + GateOpenedOnExit & operator=(const GateOpenedOnExit &) = delete; + ~GateOpenedOnExit() { gate.open(); } + Gate & gate; +}; + +/// A real storage over a fresh local object storage, `context == nullptr`: no system log, no +/// scheduler until the first GC entry point creates one. +std::shared_ptr openTestStorage() +{ + static std::atomic counter{0}; + const auto scratch = std::filesystem::temp_directory_path() + / ("cas_gc_teardown_stop_scratch_" + std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1))); + auto settings = DB::Cas::tests::makeSettingsForTest("test", scratch); + auto storage = std::make_shared( + DB::Cas::tests::makeLocalObjectStorageForTest(), "pool", "srv1", "", nullptr, settings); + storage->startup(); + return storage; +} + +/// Completes the first read of `key`, then withholds its return until released, so the round that +/// issued it is parked with that request already accounted at the backend. +class ParkFirstReadBackend : public CountingBackend +{ +public: + void armParkFirstRead(String key_, std::shared_ptr entered_, std::shared_ptr release_) + { + key = std::move(key_); + entered = std::move(entered_); + release = std::move(release_); + armed.store(true); + } + + std::optional read(const String & read_key, TransportAccess & access) override + { + auto result = CountingBackend::read(read_key, access); + if (read_key == key && armed.exchange(false)) + { + entered->open(); + release->wait("release"); + } + return result; + } + + uint64_t requestsTotal() const { return getTotal() + headTotal() + listTotal() + writeTotal(); } + +private: + String key; + std::shared_ptr entered; + std::shared_ptr release; + std::atomic armed{false}; +}; + +/// A thread-safe sink for the scheduler's rows, with a wait that never sleeps. +class RoundLogSink +{ +public: + GcRoundLogger logger() + { + return [this](const GcRoundLogRecord & r) + { + std::lock_guard lock(mutex); + records.push_back(r); + cv.notify_all(); + }; + } + + std::vector all() + { + std::lock_guard lock(mutex); + return records; + } + + /// The first Finish row at index >= `from`, waiting up to `timeout`; nullopt on timeout. + std::optional waitForFinish(size_t from, std::chrono::milliseconds timeout) + { + std::unique_lock lock(mutex); + const auto is_finish = [&] + { + for (size_t i = from; i < records.size(); ++i) + if (records[i].event_type == GcRoundLogRecord::EventType::Finish) + return true; + return false; + }; + if (!cv.wait_for(lock, timeout, is_finish)) + return std::nullopt; + for (size_t i = from; i < records.size(); ++i) + if (records[i].event_type == GcRoundLogRecord::EventType::Finish) + return records[i]; + return std::nullopt; + } + +private: + std::mutex mutex; + std::condition_variable cv; + std::vector records; +}; + +size_t countStarts(const std::vector & rows) +{ + size_t n = 0; + for (const auto & r : rows) + if (r.event_type == GcRoundLogRecord::EventType::Start) + ++n; + return n; +} + +} + +/// The arm is idempotent, observable, and closes the door to new detached work; a drain after an +/// early arm has nothing to wait for. +TEST(CASGCTeardownStop, BeginTeardownIsIdempotentAndRefusesNewDetachedWork) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + + EXPECT_FALSE(store->teardownBegun()); + store->beginTeardown(); + EXPECT_TRUE(store->teardownBegun()); + store->beginTeardown(); + EXPECT_TRUE(store->teardownBegun()) << "a second arm changes nothing"; + EXPECT_TRUE(store->detachedWorkStoppingForTest()); + + EXPECT_FALSE(store->tryDispatchDetached([](DetachedStopToken) {})) + << "no detached task is accepted once teardown began"; + EXPECT_TRUE(store->stopAndDrainDetachedWork(/*deadline_ms=*/1000)) + << "the drain after an early arm finds nothing in flight and returns at once"; +} + +/// The open plane -- GC, FSCK, the probe -- refuses after the arm, before anything reaches the +/// backend; the mount plane, which the ref-lane drain and the farewell need alive, does not. +TEST(CASGCTeardownStop, OpenPlaneRefusesAfterTeardownBeganAndTheMountPlaneDoesNot) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + { + CasOperation op = store->openRequests().admit(); + orThrow(op.create("p/probe", "v", Retry::once()), "create"); + } + backend->resetCounts(); + + store->beginTeardown(); + + CasOperation refused = store->openRequests().admit(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)refused.read("p/probe", Retry::standard()); }); + CasOperation resumed = store->openRequests().resume(/*admitted_generation=*/0); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)resumed.read("p/probe", Retry::standard()); }); + EXPECT_EQ(backend->getTotal(), 0u) << "a refused admission never reaches the backend"; + + CasOperation mount = store->mountRequests().admit(); + ASSERT_TRUE(mount.read("p/probe", Retry::once()).has_value()) + << "the mount plane is not the open plane: teardown's own drain and farewell run on it"; + EXPECT_EQ(backend->getTotal(), 1u); +} + +/// `removeManyWriteOnce` -- the verb the CAS GC bulk-delete phases call, and the one an +/// `S3ObjectStorage`-backed pool ultimately dispatches to `removeObjectsIfExistUnderProfile` -- runs on +/// the same open plane as `read` above, so a control-plane bulk delete issued after the disk's shutdown +/// (which arms teardown on this plane before the object storage's own `shutdown()` even runs, see +/// `DiskObjectStorage::shutdown()`) is refused at admission and never reaches the backend at all. +TEST(CASGCTeardownStop, RemoveManyWriteOnceIsRefusedAfterTeardownBeganAndNeverReachesTheBackend) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + { + CasOperation op = store->openRequests().admit(); + orThrow(op.create("p/probe", "v", Retry::once()), "create"); + } + backend->resetCounts(); + + store->beginTeardown(); + + const Layout layout{"p"}; + const ManifestId manifest_id{RootNamespace{"probe/ns@cas@"}, + ManifestRef{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 1}}; + + CasOperation refused = store->openRequests().admit(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + refused.removeManyWriteOnce({layout.writeOnceManifestKey(manifest_id)}, Retry::standard()); + }); + EXPECT_EQ(backend->deleteTotal(), 0u) << "a refused admission never reaches the backend, not even for one key"; +} + +/// The open plane's sleep is the interruptible one, in production wiring and after the test seam is +/// cleared. Arming FIRST makes this a wiring test: a predicate `wait_for` whose predicate already +/// holds returns without waiting, so a plane still wired to the plain sleep cannot pass. The deadline +/// is the assertion; no sleep orders any thread. +TEST(CASGCTeardownStop, OpenPlaneSleepReturnsAtOnceOnceTeardownBegan) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + store->beginTeardown(); + + auto paused = std::async(std::launch::async, [&store] { store->openRequests().pause(60'000); }); + EXPECT_EQ(paused.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "the open plane's sleep must observe the arm; a plain sleep holds for the full minute"; + + /// Clearing the seam must put the interruptible sleep back, not the engine's plain one. + store->setCasRetrySleepForTest([](uint64_t) {}); + store->setCasRetrySleepForTest({}); + auto paused_again = std::async(std::launch::async, [&store] { store->openRequests().pause(60'000); }); + EXPECT_EQ(paused_again.wait_for(std::chrono::seconds(10)), std::future_status::ready) + << "resetting the retry-sleep seam left the open plane on the plain sleep"; +} + +/// A read-ahead worker resumes under the plane's generation and is refused at its first gate; the +/// fold learns it at the take site. An unconsumed future has its exception dropped by the read-ahead's +/// destructor, so the test consumes it. +TEST(CASGCTeardownStop, ReadAheadWorkerIsRefusedAndTheTakeSiteSeesIt) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + { + CasOperation op = store->openRequests().admit(); + orThrow(op.create("p/k1", "one", Retry::once()), "create"); + } + ThreadPool pool{CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, + CurrentMetrics::LocalThreadScheduled, + /*max_threads*/ 2, /*max_free_threads*/ 2, /*queue_size*/ 0}; + CasOperation op = store->openRequests().admit(); + GcReadAhead reads(op, store->openRequests(), pool, /*concurrency=*/2); + + store->beginTeardown(); + backend->resetCounts(); + reads.hintRead("p/k1"); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)reads.takeRead("p/k1"); }); + EXPECT_EQ(backend->getTotal(), 0u) << "the worker was refused before it reached the backend"; +} + +/// The defect itself: `shutdown` waits behind `gc_scheduler_mutex`, which a synchronous round holds +/// for its whole duration. After the fix it arms the pool first, the parked round is refused at its +/// next request, and `shutdown` returns. On the old code the arm never lands while the round is +/// parked, which is the assertion that goes red. +TEST(CASGCTeardownStop, ShutdownReturnsWhileASynchronousRoundIsParked) +{ + auto storage = openTestStorage(); + auto pool = storage->poolForTest(); + ASSERT_TRUE(pool); + + auto parked = std::make_shared(); + auto release = std::make_shared(); + GateOpenedOnExit opener(*release); + std::mutex rows_mutex; + std::vector rows; + storage->setGcRoundRowHookForTest([&](const GcRoundLogRecord & r) + { + { + std::lock_guard lock(rows_mutex); + rows.push_back(r); + } + /// Park on the `lease` phase row: the round holds `gc_scheduler_mutex` and has more + /// requests ahead of it. + if (r.event_type == GcRoundLogRecord::EventType::Phase && r.phase == "lease") + { + parked->open(); + release->wait("release"); + } + }); + + auto round = std::async(std::launch::async, [&storage] { storage->runOneGcRoundForTest(); }); + parked->wait("parked"); + + auto done = std::async(std::launch::async, [&storage] { storage->shutdown(); }); + + /// The state handshake: the arm must land while the round is still parked. Bounded, and its + /// expiry is the failure on the old code. + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10); + while (!pool->teardownBegun() && std::chrono::steady_clock::now() < deadline) + std::this_thread::yield(); + EXPECT_TRUE(pool->teardownBegun()) << "shutdown waited for the round instead of arming the pool first"; + + release->open(); + EXPECT_THROW(round.get(), DB::Exception) << "the released round must be refused at its next request"; + EXPECT_EQ(done.wait_for(std::chrono::seconds(30)), std::future_status::ready); + + std::optional finish; + { + std::lock_guard lock(rows_mutex); + for (const auto & r : rows) + if (r.event_type == GcRoundLogRecord::EventType::Finish) + finish = r; + } + ASSERT_TRUE(finish.has_value()); + EXPECT_EQ(finish->outcome, GcRoundLogRecord::Outcome::Stopped); +} + +/// A background round parked inside a request is refused at its NEXT request: nothing new reaches +/// the backend after the arm, the Finish row is `Stopped`, and `stop` returns with nothing in flight. +TEST(CASGCTeardownStop, BackgroundRoundIsCutAtItsNextRequest) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + RoundLogSink sink; + CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca", sink.logger()); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + GateOpenedOnExit opener(*release); + backend->armParkFirstRead(store->layout().gcStateKey(), entered, release); + sched.start(); + entered->wait("entered"); + + store->beginTeardown(); + const uint64_t requests_at_arm = backend->requestsTotal(); + release->open(); + + const auto finish = sink.waitForFinish(/*from=*/0, std::chrono::seconds(30)); + ASSERT_TRUE(finish.has_value()) << "the parked round never finished"; + EXPECT_EQ(finish->outcome, GcRoundLogRecord::Outcome::Stopped); + sched.stop(); + EXPECT_TRUE(sched.isQuiescent()); + EXPECT_EQ(backend->requestsTotal(), requests_at_arm) + << "after the arm no request may reach the backend: the round unwinds at the next gate"; + EXPECT_EQ(countStarts(sink.all()), 1u) << "no further round started after the arm"; +} + +/// The tick queued behind a manual round must not mint a Start row after the arm: it checks the +/// flag once it holds the round mutex, before it logs anything. +TEST(CASGCTeardownStop, AQueuedScheduledTickEmitsNoStartAfterTeardownBegan) +{ + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + RoundLogSink sink; + /// An hour-long interval: the loop ticks only when asked. + CasGcScheduler sched(store, std::chrono::seconds(3600), "test::gc", "ca", sink.logger()); + sched.start(); + + auto entered = std::make_shared(); + auto release = std::make_shared(); + GateOpenedOnExit opener(*release); + backend->armParkFirstRead(store->layout().gcStateKey(), entered, release); + auto manual = std::async(std::launch::async, + [&sched] { return sched.runOneRoundNow(GcRoundLogRecord::Trigger::Manual); }); + entered->wait("entered"); + + /// The loop wakes and queues on the round mutex behind the parked manual round. + sched.requestRoundSoon(); + store->beginTeardown(); + release->open(); + + EXPECT_THROW((void)manual.get(), DB::Exception); + sched.stop(); + const auto rows = sink.all(); + EXPECT_EQ(countStarts(rows), 1u) << "the queued tick minted a Start row after teardown began"; +} + +namespace +{ + +/// Arms the pool's teardown the first time a chosen prefix is listed, so the stop lands inside the +/// namespace janitor's page rather than in the round's own request chain. +class ArmOnJanitorListBackend : public CountingBackend +{ +public: + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + auto page = CountingBackend::list(prefix, cursor, limit, access); + if (!arm_prefix.empty() && prefix == arm_prefix && on_list) + { + on_list(); + on_list = {}; + } + return page; + } + + String arm_prefix; + std::function on_list; +}; + +} + +/// A stop that lands inside advisory work the round swallows is not `Stopped`: the deferred path +/// runs the namespace janitor's page and returns normally, and the janitor turns a refused request +/// into an anomaly. The row is `Deferred`; the round did finish. Pinned so a later change to this +/// behaviour is made on purpose. +TEST(CASGCTeardownStop, AStopInsideTheJanitorPageIsSwallowedAsDeferred) +{ + auto backend = std::make_shared(); + auto store = DB::Cas::tests::openPoolForTest(backend); + const RootNamespace ns{"00/aa@cas@"}; + const ManifestRef r{.writer_epoch = 1, .build_sequence = 1, .manifest_ordinal = 0xAA}; + DB::Cas::tests::writeBlobBody(*backend, store->layout(), DB::UInt128(1)); + DB::Cas::tests::writeManifestRaw(*backend, store->layout(), ns, r, + {DB::Cas::tests::blobEntryFor("a", DB::UInt128(1))}); + DB::Cas::tests::publishCommittedTransition(*backend, store->layout(), ns, "tbl", std::nullopt, r); + + Gc gc(store, DB::UInt128(0xAB)); + const RoundReport fold_rep = gc.runRegularRound(); + ASSERT_FALSE(fold_rep.deferred) << "the first round folds"; + + backend->arm_prefix = store->layout().namespaceRootPrefix(); + backend->on_list = [&store] { store->beginTeardown(); }; + RoundReport rep; + EXPECT_NO_THROW(rep = gc.runRegularRound()) << "the janitor page swallows the refusal"; + EXPECT_TRUE(store->teardownBegun()) << "sanity: the arm landed inside the round"; + EXPECT_TRUE(rep.deferred) << "an idle second round defers; the stop inside its janitor page is advisory"; +} diff --git a/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp b/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp index 430e2faa9293..994b526af636 100644 --- a/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp +++ b/src/Disks/tests/gtest_cas_gc_undercount_repro.cpp @@ -50,7 +50,8 @@ ManifestRef ref(uint64_t seq, uint64_t inst) bool blobExists(InMemoryBackend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + DB::Cas::tests::OperationForTest op(b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}), Retry::standard()).has_value(); } /// A committed `RefOwnerBinding` for a raw `owner_transition` op. The raw appender is now @@ -155,22 +156,28 @@ TEST(CASGCUndercount, H2DuplicateCommittedRemovalIsIdempotentNoUnderflow) class InterruptRoundCasBackend : public InMemoryBackend { public: - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { - if (arm_interrupt && key == gc_state_key) + if (arm_interrupt && expected_value && key == gc_state_key) { - const auto stored = get(key); - const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; - const uint64_t next_gen = decodeGcState(bytes).snap_generation; - if (next_gen > stored_gen) + const auto stored = InMemoryBackend::read(key, access); + if (stored + && decodeGcState(bytes).snap_generation > decodeGcState(stored->bytes).snap_generation) { - arm_interrupt = false; - throw DB::Exception(DB::ErrorCodes::ABORTED, - "test-injected: round-commit gc/state CAS denied (leader deposed mid-round; lease lost)"); + arm_interrupt = false; /// one-shot: only depose the first round-commit write + /// A REFUSAL, not a throw: a thrown transport error is an ambiguity the engine settles + /// by an exact read and then reissues while the precondition it named is unmoved, so + /// the round would commit on the reissue. A refused precondition ends the write at + /// once. The object is moved too -- the same bytes under a fresh incarnation -- because + /// a store refuses only what changed; the CONTENT is deliberately left alone, so this + /// round's own lease and cursor are exactly what a deposed round leaves behind. + (void)InMemoryBackend::write(key, stored->bytes, stored->value, access); + return std::unexpected(RawConflict{}); } } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool arm_interrupt = false; @@ -252,14 +259,15 @@ TEST(CASGCUndercount, H1DrainAfterDeposedRemovalFoldDoesNotUnderflow) class DropAtCommitBackend : public InMemoryBackend { public: - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + TransportAccess & access) override { /// The one-pass round has a SINGLE gc/state CAS that advances snap_generation. Fire the injected /// drop ONCE, just before that CAS commits — so the drop event is above this round's sealed cursor. if (arm_drop && key == gc_state_key) { - const auto stored = get(key); + const auto stored = InMemoryBackend::read(key, access); if (stored) { const GcState prev = decodeGcState(stored->bytes); @@ -272,7 +280,7 @@ class DropAtCommitBackend : public InMemoryBackend } } } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool arm_drop = false; diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index c32b03e23074..c39935d7ca7f 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -7,6 +7,9 @@ #include #include +#include "config.h" +#include + #include #include #include @@ -23,23 +26,66 @@ namespace DB::ErrorCodes using namespace DB::Cas; -/// MountLeaseKeeper behavior: the per-server mount lease and the merged build-watermark floor ride the -/// SAME slot, renewed by one beat. The keeper anchors durably before return, adopts a slot already +/// MountLeaseRenewer behavior: the per-server mount lease and the merged build-watermark floor ride the +/// SAME slot, renewed by one beat. The renewer anchors durably before return, adopts a slot already /// written by `claimMount` (same uuid+epoch), re-reads the callback on each renew and bumps `seq`, -/// stamps the farewell sentinel (`min_active = UINT64_MAX`, `expires_at_ms <= now`) on `release`, and +/// stamps the farewell sentinel (`min_active_build_sequence = UINT64_MAX`, `expires_at_ms <= now`) on `release`, and /// returns typed terminal results on any foreign touch. namespace { -/// The normal steady-state flow: `claimMount` writes the live (uuid, epoch) mount, THEN the keeper +/// The two request planes this file's renewers run on. Both are open-fence -- the exclusivity these +/// tests exercise is the mount protocol's own, not a fence's -- on the same injected boot clock the +/// renewer's lease deadline is expressed on, so the two never disagree about how much budget is left. +/// `sleep_step_ms`, when set, makes one inter-attempt pause jump the clock past the lease bound: that +/// is how a test asks for exactly one physical attempt without a per-call attempt cap. It depends on +/// the engine checking the bound, sleeping, then checking again -- a reissue that slept first would +/// send a second attempt. `tests::OperationForTest` covers a fixture needing one operation, but +/// neither the two planes a renewer takes nor this clock, which is why this stays local. +class Ops +{ +public: + Ops(std::shared_ptr backend, uint64_t * boot_ms, uint64_t sleep_step_ms = 0) + : mount(openRequestsForTest(backend)) + , farewell(openRequestsForTest(std::move(backend))) + , op(mount.admit()) + { + for (CasRequests * requests : {&mount, &farewell}) + { + requests->setNowFnForTest([boot_ms] { return *boot_ms; }); + requests->setSleepFnForTest( + [boot_ms, sleep_step_ms](uint64_t ms) { *boot_ms += sleep_step_ms ? sleep_step_ms : ms; }); + } + } + + Ops(const Ops &) = delete; + Ops & operator=(const Ops &) = delete; + + CasRequests mount; + CasRequests farewell; + CasOperation op; +}; + +/// A fixture write that must land, so a mis-seeded fixture fails where it is written rather than in +/// the assertion it silently invalidated. +void mustCommit(WriteResult && result, const String & what) +{ + if (!std::holds_alternative(result)) + throw DB::Exception(DB::ErrorCodes::ABORTED, "test fixture write '{}' did not commit", what); +} + +/// The normal steady-state flow: `claimMount` writes the live (uuid, epoch) mount, THEN the renewer /// adopts it. Seed that claim so `start` adopts instead of self-tripping the double-start guard. -void seedOwnClaim(Backend & b, const Layout & l, const String & srid, UInt128 uuid, uint64_t epoch, +void seedOwnClaim(CasOperation & op, const Layout & l, const String & srid, UInt128 uuid, uint64_t epoch, uint64_t now_ms, uint64_t ttl_ms) { - ASSERT_EQ(claimMount(b, l, srid, uuid, epoch, now_ms, ttl_ms).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(op, l, srid, uuid, epoch, now_ms, ttl_ms).kind, MountClaimResult::Claimed); } -class RenewalScriptBackend final : public InMemoryBackend +/// Not `final`: `EnvelopeEatingBackend` (the envelope-cutoff test below) derives from it to +/// reuse its `Attempt`/`attempts` bookkeeping while overriding `write`/`read` with its own always-fail +/// behavior instead of the scripted-action queue. +class RenewalScriptBackend : public InMemoryBackend { public: enum class Action : uint8_t @@ -49,38 +95,53 @@ class RenewalScriptBackend final : public InMemoryBackend LandThenThrow, ReturnThenCancel, ThrowBeforeThenLandAfterResolve, + ThrowConnectHint, }; struct Attempt { String key; String bytes; - Token expected; + std::optional expected; }; - using InMemoryBackend::get; - using InMemoryBackend::putOverwrite; - std::deque actions; std::vector attempts; std::function cancel_after_write; - uint64_t get_calls = 0; + uint64_t read_calls = 0; - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + /// Only a GUARDED write of a mount slot is scripted; the fixture's own seeding and every other + /// key reach the store untouched. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - attempts.push_back({key, bytes, expected}); + if (!expected_value || !key.ends_with("/mount")) + return InMemoryBackend::write(key, bytes, expected_value, access); + + attempts.push_back({key, bytes, expected_value}); const Action action = actions.empty() ? Action::Delegate : actions.front(); if (!actions.empty()) actions.pop_front(); + if (action == Action::ThrowConnectHint) + { +#if USE_AWS_S3 + throw DB::S3Exception("Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION); +#else + throw Poco::TimeoutException("connect timed out"); +#endif + } + if (action == Action::ThrowBefore || action == Action::ThrowBeforeThenLandAfterResolve) { if (action == Action::ThrowBeforeThenLandAfterResolve) - pending = Attempt{key, bytes, expected}; + pending = Attempt{key, bytes, expected_value}; throw Poco::TimeoutException("injected renewal response uncertainty before a result"); } - PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + auto result = InMemoryBackend::write(key, bytes, expected_value, access); if (action == Action::LandThenThrow) { if (cancel_after_write) @@ -92,16 +153,16 @@ class RenewalScriptBackend final : public InMemoryBackend return result; } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - ++get_calls; - std::optional result = InMemoryBackend::get(key, range); + ++read_calls; + std::optional result = InMemoryBackend::read(key, access); if (pending && pending->key == key) { const Attempt delayed = *pending; pending.reset(); - const PutResult landed = InMemoryBackend::putOverwrite(delayed.key, delayed.bytes, delayed.expected, {}); - if (landed.outcome != PutOutcome::Done) + const auto landed = InMemoryBackend::write(delayed.key, delayed.bytes, delayed.expected, access); + if (!landed.has_value()) throw DB::Exception(DB::ErrorCodes::ABORTED, "injected delayed renewal did not land"); } return result; @@ -111,27 +172,15 @@ class RenewalScriptBackend final : public InMemoryBackend std::optional pending; }; -CasRequestBudget renewalBudget(uint32_t max_attempts = 3) -{ - return CasRequestBudget{ - .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = max_attempts, - .lease_safety_margin_ms = 20, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, - }; -} - MountRenewOperationEnvironment renewalEnvironment( uint64_t & boot_ms, - const std::function & stop_cause = {}) + const std::function & live = {}, + const std::function & cancelled = {}) { return MountRenewOperationEnvironment{ .boot_ms = [&boot_ms] { return boot_ms; }, - .stop_cause = stop_cause ? stop_cause : [] { return CasOverwriteStopCause::Continue; }, - .wait_before_retry = [](uint64_t) { return true; }, - .observe = {}, + .live = live, + .cancelled = cancelled, }; } @@ -149,18 +198,18 @@ DB::Exception terminalException(const MountRenewResult & result) } catch (...) { - ADD_FAILURE() << "terminal keeper failure was not a typed DB::Exception"; + ADD_FAILURE() << "terminal renewer failure was not a typed DB::Exception"; } return DB::Exception(DB::ErrorCodes::ABORTED, "missing terminal exception"); } -void renewKeeperOrThrow(MountLeaseKeeper & keeper) +void renewOrThrow(MountLeaseRenewer & renewer) { - const MountRenewResult result = keeper.renew(renewalBudget(), MountRenewOperationEnvironment{}); + const MountRenewResult result = renewer.renew(MountRenewOperationEnvironment{}); if (result.outcome == MountRenewOutcome::Terminal) std::rethrow_exception(result.failure); if (result.outcome != MountRenewOutcome::Committed) - throw DB::Exception(DB::ErrorCodes::ABORTED, "keeper renewal was not attempted"); + throw DB::Exception(DB::ErrorCodes::ABORTED, "renewer renewal was not attempted"); } } @@ -171,18 +220,21 @@ TEST(CASHeartbeat, AnchorCarriesFloor) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - uint64_t min_active_now = 5; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t min_active_build_sequence_now = 5; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [&] { return min_active_now; }, {}, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); - auto hr = backend->head(layout.mountKey(srid)); - ASSERT_TRUE(hr.exists); - auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); + ASSERT_TRUE(ops.op.head(layout.mountKey(srid), Retry::standard()).has_value()); + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); EXPECT_EQ(m.writer_epoch, 9u); - EXPECT_EQ(m.min_active, 5u); + EXPECT_EQ(m.min_active_build_sequence, 5u); EXPECT_EQ(m.seq, 1u); EXPECT_FALSE(m.gc_fenced); } @@ -194,20 +246,24 @@ TEST(CASHeartbeat, RenewRereadsCallbackAndBumpsSeq) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - uint64_t min_active_now = 5; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t min_active_build_sequence_now = 5; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [&] { return min_active_now; }, {}, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); /// The dynamic field moves; the renewal re-reads it off the callback and bumps seq. now_ms = 1500; - min_active_now = 8; - renewKeeperOrThrow(keeper); + min_active_build_sequence_now = 8; + renewOrThrow(renewer); - auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); - EXPECT_EQ(m.min_active, 8u); + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_EQ(m.min_active_build_sequence, 8u); EXPECT_EQ(m.seq, 2u); EXPECT_EQ(m.expires_at_ms, 1500u + 100u); } @@ -219,20 +275,219 @@ TEST(CASHeartbeat, StopStampsExpiredAndFarewellSentinel) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); now_ms = 2000; - keeper.release(); + renewer.release(); - auto m = decodeMountLease(backend->get(layout.mountKey(srid))->bytes); + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); /// Terminal body stamps the lease already-expired (so a same-server reopen reclaims immediately) - /// AND folds the watermark farewell into it (min_active = UINT64_MAX). + /// AND folds the watermark farewell into it (min_active_build_sequence = UINT64_MAX). EXPECT_LE(m.expires_at_ms, now_ms); - EXPECT_EQ(m.min_active, std::numeric_limits::max()); + EXPECT_EQ(m.min_active_build_sequence, std::numeric_limits::max()); +} + +namespace +{ +/// Reports the SHIPPED PRODUCTION defaults (`attempt_timeout_ms=5000`, two `connect_timeout_cap_ms=1000` +/// caps -> `attemptEnvelopeMs()=7000`, `CasRequestBudget.cpp`'s own defaults) while landing every attempt +/// immediately: the write's own success is not what is under test here, only whether the farewell's +/// policy window is wide enough to admit one attempt in the first place. +struct DefaultEnvelopeBackend : InMemoryBackend +{ + uint64_t attemptTimeoutMs() const override { return 5000; } + uint64_t attemptEnvelopeMs() const override { return 7000; } +}; + +/// A DIFFERENT envelope from `DefaultEnvelopeBackend`'s, for +/// `FarewellIsAdmittedUnderADifferentEnvelope` below: that test exists to pin the window's +/// ARITHMETIC, not just that some window admits the write, so it needs a reservation the +/// shipped-default window (16000 ms) could not have admitted by coincidence. +struct WiderEnvelopeBackend : InMemoryBackend +{ + uint64_t attemptTimeoutMs() const override { return 5000; } + uint64_t attemptEnvelopeMs() const override { return 9000; } +}; +} + +/// A write reserves two attempt envelopes before it starts (`CasOperation::writeLoop`'s +/// `reservedFor(0, 2)`), so at the shipped defaults the farewell needs a policy window that admits +/// 2 * 7000 = 14000 ms. A fixed window that predates that reservation (`kFarewellBudgetMs` alone is +/// 10000 ms) refuses the write before its first attempt on every graceful shutdown: no farewell is +/// published, and the next start pays a full incarnation-stability observation instead of reclaiming +/// the slot instantly. +TEST(CASHeartbeat, FarewellIsAdmittedUnderTheDefaultBudget) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/30000); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(30000), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + now_ms = 2000; + EXPECT_NO_THROW(renewer.release()) + << "the farewell's policy window must admit the write's own two-envelope reservation " + "(2 * 7000 ms with the shipped defaults) -- otherwise a clean shutdown never hands the " + "mount slot back and every restart pays a full incarnation-stability observation"; + + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_LE(m.expires_at_ms, now_ms); + EXPECT_EQ(m.min_active_build_sequence, std::numeric_limits::max()); +} + +/// Pins the window's ARITHMETIC, not just that some fixed window happens to be wide enough: a +/// regression that hardcoded the shipped-default window (16000 ms) instead of deriving it from +/// `attemptReservationMs()` would still pass `FarewellIsAdmittedUnderTheDefaultBudget` above (16000 +/// happens to equal what a 7000 ms envelope needs) but would refuse THIS write, whose reservation is +/// 2 * 9000 = 18000 ms -- strictly more than the shipped-default window. +TEST(CASHeartbeat, FarewellIsAdmittedUnderADifferentEnvelope) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/40000); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(40000), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + now_ms = 2000; + EXPECT_NO_THROW(renewer.release()) + << "the farewell's policy window must be DERIVED from this backend's own envelope " + "(2 * 9000 ms), not hardcoded to the shipped-default window -- a window fixed at " + "16000 ms would refuse this write's 18000 ms reservation"; + + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_LE(m.expires_at_ms, now_ms); + EXPECT_EQ(m.min_active_build_sequence, std::numeric_limits::max()); +} + +/// The derived window alone is not the whole story: mount-control activity must also never run past +/// the point this node's own fence may already be gone. A 5000 ms TTL with a 2000 ms safety margin +/// leaves only 3000 ms of lease-safe remaining time at release -- far short of the 7000 ms envelope's +/// own 16000 ms derived window (2 * 7000 + 2000 slack) -- so the LEASE bound, not the derived window, +/// must be what refuses this write, and it must refuse it before any physical attempt: a write that +/// cannot land inside the lease-safe remainder gains nothing by being sent anyway. +TEST(CASHeartbeat, FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/5000); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(5000), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + now_ms = 2000; + String message; + int code = 0; + bool threw = false; + try + { + renewer.release(); + } + catch (const DB::Exception & e) + { + threw = true; + message = e.message(); + code = e.code(); + } + EXPECT_TRUE(threw) << "a farewell whose reservation cannot fit inside the lease-safe remaining " + "time must be refused, not admitted past the point this node's fence may " + "already be gone"; + EXPECT_EQ(code, DB::ErrorCodes::NETWORK_ERROR) << message; + EXPECT_NE(message.find("gave up at the lease deadline after zero attempt(s)"), String::npos) << message; + + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_NE(m.min_active_build_sequence, std::numeric_limits::max()) + << "the refused write must not have landed"; +} + +/// The lease bound added above must not change what an ordinary Conflict outcome does: a successor +/// that took the slot (a different, unfenced incarnation) before this node's own shutdown could +/// publish its farewell must be left untouched, and the release must report the conflict rather than +/// silently succeeding or overwriting the successor's incarnation. +TEST(CASHeartbeat, ForeignIncarnationDuringFarewellLeavesTheSuccessorUntouchedAndReportsTheConflict) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); + + /// A successor (a different uuid/epoch, NOT gc_fenced) took the slot before this node's own + /// clean shutdown could publish its farewell -- the exact shape a live double-start reclaim + /// leaves behind. + const auto observed = ops.op.read(layout.mountKey(srid), Retry::standard()); + ASSERT_TRUE(observed.has_value()); + MountLease successor; + successor.server_uuid = UInt128(0x9999); + successor.writer_epoch = 1; + successor.seq = 1; + successor.write_attempt_id = UInt128{1}; + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(successor), observed->etag, + Retry::standard()), "successor slot"); + + now_ms = 2000; + String message; + int code = 0; + try + { + renewer.release(); + FAIL() << "a farewell that finds a foreign, unfenced incarnation must report the conflict, " + "not silently succeed or clobber the successor"; + } + catch (const DB::Exception & e) + { + message = e.message(); + code = e.code(); + } + EXPECT_EQ(code, DB::ErrorCodes::ABORTED) << message; + EXPECT_NE(message.find("found a foreign incarnation"), String::npos) << message; + + auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_EQ(m.server_uuid, successor.server_uuid) + << "the successor's own incarnation must be untouched by the refused farewell"; + EXPECT_EQ(m.writer_epoch, successor.writer_epoch); } /// Phase A (spec rev.4 2026-07-24): a confirmed renewal mismatch whose re-read shows OUR OWN @@ -246,25 +501,30 @@ TEST(CASHeartbeat, SameEpochUnfencedTouchIsUncertainNotFatal) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); - keeper.start(); - - /// The slot advances past our held token under our own pair (the ambiguous-landed-renewal shape). - const HeadResult h = backend->head(layout.mountKey(srid)); - ASSERT_TRUE(h.exists); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); + + /// The slot advances past the incarnation we hold, under our own pair (the ambiguous-landed-renewal shape). + const auto observed = ops.op.read(layout.mountKey(srid), Retry::standard()); + ASSERT_TRUE(observed.has_value()); MountLease advanced; advanced.server_uuid = uuid; advanced.writer_epoch = 9; advanced.seq = 99; advanced.write_attempt_id = UInt128{99}; - backend->putOverwrite(layout.mountKey(srid), encodeMountLease(advanced), h.token); + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(advanced), observed->etag, + Retry::standard()), "advanced slot"); try { - renewKeeperOrThrow(keeper); + renewOrThrow(renewer); FAIL() << "renew must return a terminal conflict"; } catch (const DB::Exception & e) @@ -288,24 +548,29 @@ TEST(CASHeartbeat, SupersededTouchIsFailClosedNotFatal) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); - const HeadResult h = backend->head(layout.mountKey(srid)); - ASSERT_TRUE(h.exists); + const auto observed = ops.op.read(layout.mountKey(srid), Retry::standard()); + ASSERT_TRUE(observed.has_value()); MountLease successor; successor.server_uuid = uuid; successor.writer_epoch = 10; successor.seq = 1; successor.write_attempt_id = UInt128{1}; - backend->putOverwrite(layout.mountKey(srid), encodeMountLease(successor), h.token); + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(successor), observed->etag, + Retry::standard()), "successor slot"); try { - renewKeeperOrThrow(keeper); + renewOrThrow(renewer); FAIL() << "renew must return a terminal conflict"; } catch (const DB::Exception & e) @@ -337,20 +602,25 @@ TEST(CASHeartbeat, ForeignUuidTouchFailsClosedWithoutAborting) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); - const HeadResult h = backend->head(layout.mountKey(srid)); - ASSERT_TRUE(h.exists); + const auto observed = ops.op.read(layout.mountKey(srid), Retry::standard()); + ASSERT_TRUE(observed.has_value()); MountLease foreign; foreign.server_uuid = UInt128(0x9999); foreign.writer_epoch = 1; foreign.seq = 1; foreign.write_attempt_id = UInt128{1}; - backend->putOverwrite(layout.mountKey(srid), encodeMountLease(foreign), h.token); + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(foreign), observed->etag, + Retry::standard()), "foreign slot"); /// Restored on every exit: this flag is process-global and every later test in this binary would /// inherit it. @@ -362,7 +632,7 @@ TEST(CASHeartbeat, ForeignUuidTouchFailsClosedWithoutAborting) int code = 0; try { - renewKeeperOrThrow(keeper); + renewOrThrow(renewer); FAIL() << "a foreign holder must fail the renewal closed, not be silently taken over"; } catch (const DB::Exception & e) @@ -386,9 +656,11 @@ TEST(CASMountAudit, ClaimReleaseAndForeignConflictEmitEvents) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); const uint64_t now_ms = 1'000'000; /// mint for uuid 1 -> one mount_claim - ASSERT_EQ(claimMount(*backend, layout, "a", UInt128{1}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink).kind, + ASSERT_EQ(claimMount(ops.op, layout, "a", UInt128{1}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink).kind, MountClaimResult::Claimed); ASSERT_EQ(seen.size(), 1u); EXPECT_EQ(seen[0].type, CasEventType::MountClaim); @@ -397,7 +669,7 @@ TEST(CASMountAudit, ClaimReleaseAndForeignConflictEmitEvents) /// a FOREIGN uuid claiming a live slot -> mount_conflict carrying the current holder's identity seen.clear(); - (void)claimMount(*backend, layout, "a", UInt128{2}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink); + (void)claimMount(ops.op, layout, "a", UInt128{2}, 1, now_ms, /*ttl_ms=*/10'000, {}, sink); ASSERT_FALSE(seen.empty()); EXPECT_EQ(seen.back().type, CasEventType::MountConflict); EXPECT_EQ(seen.back().detail.at("server_root_id"), "a"); @@ -407,22 +679,26 @@ TEST(CASMountAudit, ClaimReleaseAndForeignConflictEmitEvents) EXPECT_NE(seen.back().detail.at("holder_uuid"), u128ToHex(UInt128{2})); } -/// The MountLeaseKeeper wiring: `start` adopting an already-claimed slot emits mount_claim, `stop` +/// The MountLeaseRenewer wiring: `start` adopting an already-claimed slot emits mount_claim, `stop` /// (the farewell write) emits mount_release. -TEST(CASMountAudit, KeeperAdoptEmitsClaimAndTerminateEmitsRelease) +TEST(CASMountAudit, RenewerAdoptEmitsClaimAndTerminateEmitsRelease) { auto backend = std::make_shared(); Layout layout("pool"); const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); ASSERT_EQ(seen.size(), 1u); EXPECT_EQ(seen[0].type, CasEventType::MountClaim); @@ -430,18 +706,18 @@ TEST(CASMountAudit, KeeperAdoptEmitsClaimAndTerminateEmitsRelease) seen.clear(); now_ms = 2000; - keeper.release(); + renewer.release(); ASSERT_EQ(seen.size(), 1u); EXPECT_EQ(seen[0].type, CasEventType::MountRelease); EXPECT_EQ(seen[0].detail.at("branch"), "farewell"); } -/// Keeper-level foreign-conflict refusal: the mount slot is already held by a FOREIGN uuid (X) when -/// a keeper for a DIFFERENT uuid (Y) tries to claim it. This must fail closed and — since the +/// Renewer-level foreign-conflict refusal: the mount slot is already held by a FOREIGN uuid (X) when +/// a renewer for a DIFFERENT uuid (Y) tries to claim it. This must fail closed and — since the /// mount-audit sink is not yet installed at first-open — name X in the exception's message text /// (the only identity carrier in err.log at that point). MountConflict payload coverage is above. -TEST(CASMountAudit, KeeperForeignConflictRefusesAndNamesHolder) +TEST(CASMountAudit, RenewerForeignConflictRefusesAndNamesHolder) { auto backend = std::make_shared(); Layout layout("pool"); @@ -450,27 +726,31 @@ TEST(CASMountAudit, KeeperForeignConflictRefusesAndNamesHolder) const UInt128 uuid_y(0x2222); uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); /// Foreign holder X claims the slot first. - ASSERT_EQ(claimMount(*backend, layout, srid, uuid_x, /*our_epoch=*/1, now_ms, /*ttl_ms=*/100).kind, + ASSERT_EQ(claimMount(ops.op, layout, srid, uuid_x, /*our_epoch=*/1, now_ms, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - MountLeaseKeeper keeper(backend, layout, srid, uuid_y, /*writer_epoch=*/1, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid_y, /*writer_epoch=*/1, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); /// The enriched refusal message must name the OBSERVED holder (X), not the caller (Y). const String holder_uuid = u128ToHex(uuid_x); DB::Cas::tests::expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, holder_uuid, - [&] { keeper.start(); }); + [&] { renewer.start(); }); } /// `Pool::open` can fail before/inside `doStart` (e.g. a foreign-conflict refusal, see -/// `KeeperForeignConflictRefusesAndNamesHolder` above) — the keeper is destroyed without ever having +/// `RenewerForeignConflictRefusesAndNamesHolder` above) — the renewer is destroyed without ever having /// claimed anything. Teardown must not throw "release before start"; there is nothing to release. A /// stop AFTER a successful start still performs the farewell (covered by /// `StopStampsExpiredAndFarewellSentinel` above); a genuinely-started DOUBLE terminate stays loud. -TEST(CASMountAudit, KeeperAdoptRefusesFencedSelfWithTypedError) +TEST(CASMountAudit, RenewerAdoptRefusesFencedSelfWithTypedError) { auto backend = std::make_shared(); Layout layout("pool"); @@ -478,27 +758,31 @@ TEST(CASMountAudit, KeeperAdoptRefusesFencedSelfWithTypedError) const UInt128 uuid(0x1234); uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); /// mint (uuid, epoch 9), then fence it in place (what computeHeartbeatFloor does on expiry): - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); { - auto got = backend->get(layout.mountKey(srid)); + auto got = ops.op.read(layout.mountKey(srid), Retry::standard()); MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; fenced.seq += 1; - ASSERT_EQ(backend->putOverwrite(layout.mountKey(srid), encodeMountLease(fenced), got->token).outcome, - PutOutcome::Done); + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(fenced), got->etag, + Retry::standard()), "fence-out"); } std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - /// A keeper for the SAME (uuid, epoch) tries to adopt the now-fenced slot. - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, sink); + /// A renewer for the SAME (uuid, epoch) tries to adopt the now-fenced slot. + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); bool threw = false; try { - keeper.start(); + renewer.start(); } catch (const MountFencedException & e) { @@ -515,7 +799,7 @@ TEST(CASMountAudit, KeeperAdoptRefusesFencedSelfWithTypedError) /// A renew mismatch is classified by BODY, not blamed on "a foreign writer" by default: the GC can /// fence our OWN (uuid, epoch) mount slot after our lease expires (a late renewal beat racing the -/// GC's fence-out). The keeper must re-read and recognize this as its OWN incarnation being fenced — +/// GC's fence-out). The renewer must re-read and recognize this as its OWN incarnation being fenced — /// a recoverable `MountFencedException`, not the generic single-writer-violation text. TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) { @@ -524,31 +808,35 @@ TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) const String srid = "test"; const UInt128 uuid(0x1234); uint64_t now_ms = 1000; - seedOwnClaim(*backend, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseKeeper keeper(backend, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), - [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0)); - keeper.start(); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(100), [&] { return now_ms; }, + [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); seen.clear(); /// Mid-run: the GC fences our own (uuid, epoch) mount slot in place (as `computeHeartbeatFloor` - /// does on an expired lease), preserving the whole body — a token-guarded putOverwrite, exactly - /// as the GC's own fence-out does it. + /// does on an expired lease), preserving the whole body — a guarded write against the incarnation + /// it observed, exactly as the GC's own fence-out does it. { - const auto got = backend->get(layout.mountKey(srid)); + const auto got = ops.op.read(layout.mountKey(srid), Retry::standard()); MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; fenced.seq += 1; - ASSERT_EQ(backend->putOverwrite(layout.mountKey(srid), encodeMountLease(fenced), got->token).outcome, - PutOutcome::Done); + mustCommit(ops.op.replace(layout.mountKey(srid), encodeMountLease(fenced), got->etag, + Retry::standard()), "fence-out"); } /// The renewal must classify the fence honestly — not "foreign writer": try { - renewKeeperOrThrow(keeper); + renewOrThrow(renewer); FAIL() << "renew over a fenced slot must be terminal"; } catch (const MountFencedException & e) @@ -563,12 +851,12 @@ TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) EXPECT_EQ(seen.back().detail.at("holder_uuid"), u128ToHex(uuid)); } -TEST(CASHeartbeat, KeeperStateAllowsOnlyActiveReleaseOrTerminal) +TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) { #if defined(DEBUG_OR_SANITIZER_BUILD) -#define EXPECT_KEEPER_STATE_REJECTION(statement) EXPECT_DEATH({ statement; }, "allowed only in") +#define EXPECT_RENEWER_STATE_REJECTION(statement) EXPECT_DEATH({ statement; }, "allowed only in") #else -#define EXPECT_KEEPER_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) +#define EXPECT_RENEWER_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) #endif Layout layout("pool"); @@ -578,45 +866,49 @@ TEST(CASHeartbeat, KeeperStateAllowsOnlyActiveReleaseOrTerminal) auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "released", uuid, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "released", uuid, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "released", uuid, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "released", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::New); - EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); - EXPECT_KEEPER_STATE_REJECTION(keeper.release()); - EXPECT_EQ(keeper.start(), 100u); - EXPECT_KEEPER_STATE_REJECTION(keeper.start()); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Active); - keeper.release(); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Released); - EXPECT_KEEPER_STATE_REJECTION(keeper.start()); - EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); - EXPECT_KEEPER_STATE_REJECTION(keeper.release()); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::New); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.release()); + EXPECT_EQ(renewer.start(), 100u); + EXPECT_RENEWER_STATE_REJECTION(renewer.start()); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Active); + renewer.release(); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Released); + EXPECT_RENEWER_STATE_REJECTION(renewer.start()); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.release()); } { auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "terminal", uuid, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), + /// One pause jumps the clock past the lease bound, so the ambiguous first attempt is the only + /// one this renewal ever sends and its verdict is the terminal one under test. + Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + seedOwnClaim(ops.op, layout, "terminal", uuid, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; - const MountRenewResult result = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_NE(result.failure, nullptr); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); - EXPECT_KEEPER_STATE_REJECTION(keeper.start()); - EXPECT_KEEPER_STATE_REJECTION(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); - EXPECT_KEEPER_STATE_REJECTION(keeper.release()); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); + EXPECT_RENEWER_STATE_REJECTION(renewer.start()); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.release()); } -#undef EXPECT_KEEPER_STATE_REJECTION +#undef EXPECT_RENEWER_STATE_REJECTION } TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) @@ -627,16 +919,17 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) const UInt128 uuid{0x1234}; uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, srid, uuid, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, srid, uuid, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, srid, uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->attempts.clear(); backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::Delegate}; - MountRenewResult retried = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + MountRenewResult retried = renewer.renew(renewalEnvironment(boot_ms)); ASSERT_EQ(retried.outcome, MountRenewOutcome::Committed); ASSERT_EQ(backend->attempts.size(), 2u); EXPECT_EQ(backend->attempts[0].key, backend->attempts[1].key); @@ -647,36 +940,74 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) backend->attempts.clear(); backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; - MountRenewResult adopted = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + MountRenewResult adopted = renewer.renew(renewalEnvironment(boot_ms)); EXPECT_EQ(adopted.outcome, MountRenewOutcome::Committed); - EXPECT_TRUE(adopted.diagnostics.resolved_by_get); - EXPECT_EQ(adopted.diagnostics.attempts_sent, 1u); - EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey(srid))->bytes).write_attempt_id, + EXPECT_TRUE(adopted.resolved_by_read); + EXPECT_EQ(adopted.attempts_sent, 1u); + EXPECT_EQ(decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes).write_attempt_id, decodeMountLease(backend->attempts.front().bytes).write_attempt_id); } +#if USE_AWS_S3 +TEST(CASHeartbeat, RenewalOverConnectFailuresRecoversWithoutASettleRead) +{ + auto backend = std::make_shared(); + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid{0x1234}; + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 30000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, srid, uuid, 9, std::chrono::milliseconds(30000), + [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + backend->attempts.clear(); + backend->read_calls = 0; + /// Three seconds of "no free port" at 50 ms per hint, then the store answers. + for (int i = 0; i < 60; ++i) + backend->actions.push_back(RenewalScriptBackend::Action::ThrowConnectHint); + backend->actions.push_back(RenewalScriptBackend::Action::Delegate); + const MountRenewResult renewed = renewer.renew(renewalEnvironment(boot_ms)); + ASSERT_EQ(renewed.outcome, MountRenewOutcome::Committed); + EXPECT_GT(renewed.attempts_sent, 1u); + EXPECT_FALSE(renewed.resolved_by_read); /// classification `committed_after_retry` + EXPECT_EQ(backend->read_calls, 0u); + EXPECT_EQ(backend->attempts.size(), 61u); + for (const auto & attempt : backend->attempts) + EXPECT_EQ(attempt.bytes, backend->attempts.front().bytes); +} +#endif + TEST(CASHeartbeat, DeadlineBeforeSendTerminalizesWithTypedFailure) { auto backend = std::make_shared(); Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 100); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 100); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->attempts.clear(); - backend->get_calls = 0; + backend->read_calls = 0; boot_ms = 180; - const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_NE(failure.message().find("no attempt was sent"), String::npos) << failure.message(); - EXPECT_EQ(result.diagnostics.unresolved_reason, CasUnresolvedReason::NoAttemptSent); + EXPECT_NE(failure.message().find("no attempt sent"), String::npos) << failure.message(); + EXPECT_NE(failure.message().find("external_lease_deadline"), String::npos) << failure.message(); + EXPECT_FALSE(result.sent_any); + ASSERT_TRUE(result.deadline_source.has_value()); + EXPECT_EQ(*result.deadline_source, GaveUp::Source::Lease); EXPECT_TRUE(backend->attempts.empty()); - EXPECT_EQ(backend->get_calls, 0u) << "a pre-send terminal deadline must perform no diagnostic GET"; + EXPECT_EQ(backend->read_calls, 0u) << "a pre-send terminal deadline must perform no diagnostic read"; } TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) @@ -685,22 +1016,23 @@ TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->attempts.clear(); - backend->get_calls = 0; - const auto cancelled = [] { return CasOverwriteStopCause::Cancelled; }; - const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms, cancelled)); + backend->read_calls = 0; + const MountRenewResult result = renewer.renew(renewalEnvironment( + boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); EXPECT_EQ(result.outcome, MountRenewOutcome::NotAttempted); EXPECT_EQ(result.failure, nullptr); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Active); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Active); EXPECT_TRUE(backend->attempts.empty()); - EXPECT_NO_THROW(keeper.release()); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::Released); + EXPECT_NO_THROW(renewer.release()); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Released); } TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) @@ -710,28 +1042,27 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; bool cancelled = false; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->attempts.clear(); - backend->get_calls = 0; + backend->read_calls = 0; backend->cancel_after_write = [&] { cancelled = true; }; backend->actions = {RenewalScriptBackend::Action::ReturnThenCancel}; - const MountRenewResult result = keeper.renew( - renewalBudget(), renewalEnvironment(boot_ms, [&] { - return cancelled ? CasOverwriteStopCause::Cancelled : CasOverwriteStopCause::Continue; - })); + const MountRenewResult result = renewer.renew( + renewalEnvironment(boot_ms, /*live=*/[&] { return !cancelled; }, /*cancelled=*/[&] { return cancelled; })); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_EQ(result.diagnostics.unresolved_reason, CasUnresolvedReason::FenceLostPostWrite); - EXPECT_EQ(backend->get_calls, 0u) << "post-write cancellation must not start a diagnostic GET"; - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); - const String bytes_before = backend->get(layout.mountKey("test"))->bytes; - EXPECT_FALSE(keeper.canRelease()); - EXPECT_EQ(backend->get(layout.mountKey("test"))->bytes, bytes_before); + EXPECT_TRUE(result.sent_any); + EXPECT_EQ(backend->read_calls, 0u) << "post-write cancellation must not start a diagnostic read"; + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); + const String bytes_before = ops.op.read(layout.mountKey("test"), Retry::standard())->bytes; + EXPECT_FALSE(renewer.canRelease()); + EXPECT_EQ(ops.op.read(layout.mountKey("test"), Retry::standard())->bytes, bytes_before); } TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) @@ -740,19 +1071,20 @@ TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); boot_ms = 150; backend->cancel_after_write = [&] { boot_ms = 400; }; backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; - const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); EXPECT_EQ(result.attempt_start_boot_ms, 150u); - EXPECT_EQ(keeper.lastCommittedAttemptStartBootMs(), 150u); + EXPECT_EQ(renewer.lastCommittedAttemptStartBootMs(), 150u); } TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) @@ -764,26 +1096,27 @@ TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - seedOwnClaim(*backend, layout, "test", uuid, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", uuid, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", uuid, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); - auto got = backend->get(layout.mountKey("test")); + renewer.start(); + auto got = ops.op.read(layout.mountKey("test"), Retry::standard()); MountLease current = decodeMountLease(got->bytes); current.server_uuid = current_uuid; current.writer_epoch = current_epoch; current.write_attempt_id = current_attempt; ++current.seq; - ASSERT_EQ(backend->putOverwrite(layout.mountKey("test"), encodeMountLease(current), got->token).outcome, - PutOutcome::Done); - backend->get_calls = 0; - const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + mustCommit(ops.op.replace(layout.mountKey("test"), encodeMountLease(current), got->etag, + Retry::standard()), "competing slot"); + backend->read_calls = 0; + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); const DB::Exception failure = terminalException(result); EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); - EXPECT_EQ(backend->get_calls, 1u) << "the controller's resolving GET must be the only terminal read"; + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); + EXPECT_EQ(backend->read_calls, 1u) << "the write's own resolving read must be the only terminal read"; }; run_case(UInt128{1}, 9, UInt128{0xAAAA}); @@ -797,23 +1130,24 @@ TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); backend->attempts.clear(); backend->actions = { RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve, RenewalScriptBackend::Action::Delegate, }; - const MountRenewResult result = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); - EXPECT_TRUE(result.diagnostics.resolved_by_get); + EXPECT_TRUE(result.resolved_by_read); ASSERT_EQ(backend->attempts.size(), 2u); EXPECT_EQ(backend->attempts[0].bytes, backend->attempts[1].bytes); - EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey("test"))->bytes).write_attempt_id, + EXPECT_EQ(decodeMountLease(ops.op.read(layout.mountKey("test"), Retry::standard())->bytes).write_attempt_id, decodeMountLease(backend->attempts[0].bytes).write_attempt_id); } @@ -825,26 +1159,28 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); const String key = layout.mountKey("test"); - auto got = backend->get(key); + auto got = ops.op.read(key, Retry::standard()); if (vanish) - ASSERT_EQ(backend->deleteExact(key, got->token).kind, DeleteOutcome::Kind::Deleted); + ASSERT_EQ(ops.op.remove(key, got->etag, Retry::standard()), Removal::Removed); else { MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; ++fenced.seq; - ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), got->token).outcome, PutOutcome::Done); + mustCommit(ops.op.replace(key, encodeMountLease(fenced), got->etag, Retry::standard()), + "fence-out"); } - const DB::Exception failure = terminalException(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms))); + const DB::Exception failure = terminalException(renewer.renew(renewalEnvironment(boot_ms))); EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); - EXPECT_EQ(keeper.state(), MountLeaseKeeperState::RenewalTerminal); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); }; run_case(false); run_case(true); @@ -852,64 +1188,74 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) { - const auto make_terminal = [](const std::shared_ptr & backend, - const Layout & layout, const String & srid, - uint64_t & wall_ms, uint64_t & boot_ms) + Layout layout("pool"); { - seedOwnClaim(*backend, layout, srid, UInt128{1}, 9, wall_ms, 1000); - auto keeper = std::make_unique( - backend, layout, srid, UInt128{1}, 9, std::chrono::milliseconds(1000), + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + /// One pause jumps the clock past the lease bound, so the ambiguous first attempt is the only + /// one this renewal sends and the renewal ends terminal with that attempt still in flight. + Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + seedOwnClaim(ops.op, layout, "before-reclaim", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "before-reclaim", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper->start(); + renewer.start(); backend->actions = {RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve}; - const MountRenewResult result = keeper->renew(renewalBudget(1), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); - EXPECT_EQ(keeper->state(), MountLeaseKeeperState::RenewalTerminal); - return keeper; - }; - Layout layout("pool"); - { - auto backend = std::make_shared(); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - auto keeper = make_terminal(backend, layout, "before-reclaim", wall_ms, boot_ms); - const MountLease landed = decodeMountLease(backend->get(layout.mountKey("before-reclaim"))->bytes); + /// The delayed write landed during the resolving read. It carries this renewer's own epoch, and + /// it does not put the renewer back in business. + const MountLease landed = decodeMountLease( + ops.op.read(layout.mountKey("before-reclaim"), Retry::standard())->bytes); EXPECT_EQ(landed.writer_epoch, 9u); - EXPECT_EQ(keeper->state(), MountLeaseKeeperState::RenewalTerminal); + EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); } { auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "after-successor", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + seedOwnClaim(ops.op, layout, "after-successor", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); + + /// The incarnation the about-to-be-terminal renewal names as its precondition: a late delivery + /// of that attempt can only ever be replayed against exactly this one. + const Etag delayed_precondition + = ops.op.read(layout.mountKey("after-successor"), Retry::standard())->etag; + backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; - const MountRenewResult result = keeper.renew(renewalBudget(1), renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); ASSERT_FALSE(backend->attempts.empty()); const auto delayed = backend->attempts.back(); - auto current = backend->get(delayed.key); + + /// The GC fences the slot, then a successor claims it at a fresh epoch and adopts it. + auto current = ops.op.read(delayed.key, Retry::standard()); MountLease fenced = decodeMountLease(current->bytes); fenced.gc_fenced = true; ++fenced.seq; - ASSERT_EQ(backend->InMemoryBackend::putOverwrite(delayed.key, encodeMountLease(fenced), current->token, {}).outcome, - PutOutcome::Done); - ASSERT_EQ(claimMount(*backend, layout, "after-successor", UInt128{1}, 10, wall_ms, 1000).kind, + mustCommit(ops.op.replace(delayed.key, encodeMountLease(fenced), current->etag, Retry::standard()), + "fence-out"); + ASSERT_EQ(claimMount(ops.op, layout, "after-successor", UInt128{1}, 10, wall_ms, 1000).kind, MountClaimResult::Claimed); - MountLeaseKeeper successor( - backend, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), + MountLeaseRenewer successor( + ops.mount, ops.farewell, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); successor.start(); - EXPECT_EQ(backend->InMemoryBackend::putOverwrite(delayed.key, delayed.bytes, delayed.expected, {}).outcome, - PutOutcome::PreconditionFailed); - EXPECT_EQ(decodeMountLease(backend->get(delayed.key)->bytes).writer_epoch, 10u); + + /// Replaying the delayed attempt against the incarnation it named is refused; the successor's + /// body is what stands. + EXPECT_TRUE(std::holds_alternative( + ops.op.replace(delayed.key, delayed.bytes, delayed_precondition, Retry::once()))); + EXPECT_EQ(decodeMountLease(ops.op.read(delayed.key, Retry::standard())->bytes).writer_epoch, 10u); } } @@ -919,22 +1265,83 @@ TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) Layout layout("pool"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - seedOwnClaim(*backend, layout, "test", UInt128{1}, 9, wall_ms, 1000); - MountLeaseKeeper keeper( - backend, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); + MountLeaseRenewer renewer( + ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); - keeper.start(); + renewer.start(); wall_ms = 9'000'000; - EXPECT_EQ(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + EXPECT_EQ(renewer.renew(renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); wall_ms = 1; - EXPECT_EQ(keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + EXPECT_EQ(renewer.renew(renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); backend->attempts.clear(); boot_ms += 10'000; - const MountRenewResult suspended = keeper.renew(renewalBudget(), renewalEnvironment(boot_ms)); + const MountRenewResult suspended = renewer.renew(renewalEnvironment(boot_ms)); const DB::Exception failure = terminalException(suspended); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); EXPECT_TRUE(backend->attempts.empty()) << "suspend-sized BOOTTIME overshoot must close admission"; } + +/// Every attempt costs the whole envelope (attempt 100 + 2 * cap 50 = 200 ms) and fails ambiguously. +/// Under a 1000 ms lease with a 100 ms margin the renewal must stop issuing before the cutoff rather +/// than start an attempt that cannot finish inside it. +namespace +{ +/// Bypasses `RenewalScriptBackend`'s scripted-action queue for a guarded mount write and instead +/// always fails it (and every read) once armed, each failure costing the whole envelope on the +/// injected boot clock. Left unarmed during `seedOwnClaim` (an unconditional read then an unguarded +/// create -- neither is a guarded mount write, but the read would still hit the always-throwing +/// override below) and during `renewer.start()`'s adopt read, so the fixture itself can land. +struct EnvelopeEatingBackend : RenewalScriptBackend +{ + uint64_t * boot_ms = nullptr; + bool armed = false; + uint64_t attemptTimeoutMs() const override { return 100; } + uint64_t attemptEnvelopeMs() const override { return 200; } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override + { + if (armed && expected_value && key.ends_with("/mount")) + { + attempts.push_back({key, bytes, expected_value}); + *boot_ms += 200; + throw Poco::TimeoutException("the whole envelope, gone"); + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + std::optional read(const String & key, TransportAccess & access) override + { + if (armed) + { + *boot_ms += 200; + throw Poco::TimeoutException("the read too"); + } + return InMemoryBackend::read(key, access); + } +}; +} + +TEST(CASHeartbeat, RenewalStopsBeforeTheCutoffWhenEveryAttemptConsumesTheEnvelope) +{ + auto backend = std::make_shared(); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + backend->boot_ms = &boot_ms; + Layout layout("pool"); + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, "test", UInt128{0x1234}, 9, wall_ms, 1000); + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, "test", UInt128{0x1234}, 9, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(100), + [&] { return boot_ms; }); + renewer.start(); + const uint64_t cutoff = renewer.lastCommittedAttemptStartBootMs() + 1000 - 100; + backend->attempts.clear(); + backend->armed = true; + const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_LE(boot_ms, cutoff) << "the last attempt started inside the cutoff and the engine did not start one that could not finish"; +} diff --git a/src/Disks/tests/gtest_cas_holey_list_detector.cpp b/src/Disks/tests/gtest_cas_holey_list_detector.cpp index 6c1728d3ba31..05da38ca5b61 100644 --- a/src/Disks/tests/gtest_cas_holey_list_detector.cpp +++ b/src/Disks/tests/gtest_cas_holey_list_detector.cpp @@ -56,6 +56,8 @@ namespace class HoleyListBackend : public InMemoryBackend { public: + /// Unhide the legacy `list` overloads the primitive override below would otherwise hide. + using Backend::list; /// Omit `key` from the `nth` (0-based) subsequent qualifying `list` call. Resets the counter. void omitFromNthListCall(const String & key, size_t nth) { @@ -74,14 +76,16 @@ class HoleyListBackend : public InMemoryBackend return served; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// Sabotages the PRIMITIVE, which every legacy forwarder reaches too, so the hole is served + /// whichever surface issued the enumeration. + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); std::lock_guard lock(m); if (omitted.empty()) return page; auto it = std::find_if(page.keys.begin(), page.keys.end(), - [&](const ListedKey & k) { return k.key == omitted; }); + [&](const RawListedKey & k) { return k.key == omitted; }); if (it == page.keys.end()) return page; /// not a qualifying call — do not count it if (seen_calls++ != target_call) @@ -133,19 +137,25 @@ ManifestId publishOneBlobPart(const PoolPtr & s, const RootNamespace & ns, const bool blobPresent(const std::shared_ptr & b, const Layout & layout, const String & payload) { - return b->head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, - BlobDigest::fromU128(u128Of(payload))})).exists; + DB::Cas::tests::OperationForTest op(*b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, + BlobDigest::fromU128(u128Of(payload))}), Retry::standard()).has_value(); } /// Every ref object key of one namespace. Used to identify WHICH objects a publish appended, rather /// than guessing a sequence number. std::set listRefKeys(Backend & b, const Layout & layout, const RootNamespace & ns) { - /// Stage B (Task 4-C): `ns` is born through the REAL append lane here, so its objects sit at a - /// real catalog-minted incarnation, not the Stage-A sentinel. - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(b, layout, ns).value(); + /// `ns` is born through the REAL append lane here, so its objects sit at a real catalog-minted + /// incarnation rather than a fixture-chosen one. The catalog read is made on an open-fence + /// operation of its own: this helper only observes, and shares no admission with the code + /// under test. + CasRequests requests = DB::Cas::tests::openRequestsForTest(b); + CasOperation op = requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(op, layout, ns).value(); std::set keys; - forEachListedKey(b, layout.namespaceStreamPrefix(life), [&](const ListedKey & k) { keys.insert(k.key); }); + op.forEachListedKey(layout.namespaceStreamPrefix(life), + [&](const ListedKey & k) { keys.insert(k.key); return true; }, Retry::standard()); return keys; } @@ -170,13 +180,14 @@ String refLogKeyEmittingEdge(Backend & b, const Layout & layout, const RootNames const std::vector & candidates, const ManifestId & manifest_id, int change) { + DB::Cas::tests::OperationForTest probe(b); std::vector hits; for (const String & key : candidates) { const auto parsed = layout.parseRefObjectKey(key); if (!parsed || parsed->kind != RefObjectKind::Log) continue; - const auto got = b.get(key); + const auto got = (*probe).read(key, Retry::standard()); if (!got) continue; const RefLogTxn txn = @@ -271,8 +282,11 @@ TEST(CASHoleyListDetector, OmittedActivationNeverPermitsDeletingALiveBlob) Gc gc(s, hexToU128("00000000000000000000000000000001")); runRounds(s, gc, 2); ASSERT_TRUE(blobPresent(b, layout, payload)); - ASSERT_TRUE(b->head(layout.manifestKey(m1)).exists) - << "M1's body must still be present so its `-1` edges are readable at removal-fold"; + { + DB::Cas::tests::OperationForTest m1_probe(*b); + ASSERT_TRUE((*m1_probe).head(layout.manifestKey(m1), Retry::standard()).has_value()) + << "M1's body must still be present so its `-1` edges are readable at removal-fold"; + } /// M2 adopts the SAME deduplicated blob (`putBlob` of an identical payload dedups). Learn WHICH /// ref-log object carries M2's ACTIVATION by diffing the namespace's ref prefix around the publish diff --git a/src/Disks/tests/gtest_cas_hot_keys.cpp b/src/Disks/tests/gtest_cas_hot_keys.cpp new file mode 100644 index 000000000000..6dc4d5426383 --- /dev/null +++ b/src/Disks/tests/gtest_cas_hot_keys.cpp @@ -0,0 +1,771 @@ +#include + +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" + +#include +#include + +#include "config.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ + extern const Event CASHotKeyQueueWaitMicroseconds; + extern const Event CASHotKeyCacheStarts; + extern const Event CASHotKeyReadStarts; + extern const Event CASHotKeyCacheVerdictsReread; + extern const Event CASRequestGaveUp; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +/// The harness's `FakeClock` is single-threaded. The lane is not: its holders sleep on the engine's +/// clock from their own threads while the test thread advances it, so every access goes through one +/// mutex. A sleep still advances the clock by what it slept, so a holder's transport backoff is real +/// time to every waiter's deadline. +struct SyncClock +{ + std::mutex mutex; + uint64_t now = 1'000'000; + std::vector sleeps; + + std::function nowFn() + { + return [this] { std::lock_guard lock(mutex); return now; }; + } + std::function sleepFn() + { + return [this](uint64_t ms) { std::lock_guard lock(mutex); sleeps.push_back(ms); now += ms; }; + } + void advance(uint64_t ms) { std::lock_guard lock(mutex); now += ms; } + size_t sleepCount() { std::lock_guard lock(mutex); return sleeps.size(); } +}; + +/// The object under test lists the tickets that wrote it, comma-separated, so order is visible. +CasHotKeys::Decide appendTicket(int ticket) +{ + return [ticket](const std::optional & current) -> std::optional + { + if (!current) + return std::to_string(ticket); + return current->bytes + "," + std::to_string(ticket); + }; +} + +uint64_t counter(ProfileEvents::Event event) +{ + return ProfileEvents::global_counters[event].load(); +} + +/// A one-shot gate a write hook parks on: the first write of the key waits here until the test +/// releases it; every later write passes. +struct ParkFirstWrite +{ + std::latch parked{1}; + std::latch release{1}; + std::atomic seen{0}; + + void install(CountingBackend & backend, const String & key) + { + backend.onBeforeWrite(key, [this] + { + if (seen.fetch_add(1) != 0) + return; + parked.count_down(); + release.wait(); + }); + } +}; + +#if USE_AWS_S3 +std::exception_ptr s3Error(Aws::S3::S3Errors code, const String & name) +{ + return std::make_exception_ptr(DB::S3Exception("the store answered " + name, code, name)); +} +#endif + +} + +TEST(CASHotKeys, SubmissionsOfOneKeyAreSerializedInArrivalOrder) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + constexpr int N = 4; + ParkFirstWrite park; + park.install(*backend, "k"); + + std::vector threads; + std::vector> results(N); + std::deque go; /// a deque: `std::latch` is neither copyable nor movable + for (int i = 0; i < N; ++i) + go.emplace_back(1); + for (int i = 0; i < N; ++i) + { + threads.emplace_back([&, i] + { + go[i].wait(); + auto op = requests.admit(); + results[i] = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(i + 1)); + }); + } + /// The first holder is released into its write and parked there; every later thread is released + /// only after its item is seen queued, so arrival order is the release order. + go[0].count_down(); + park.parked.wait(); + for (int i = 1; i < N; ++i) + { + go[i].count_down(); + while (hot_keys.queueDepthForTest("k") < static_cast(i + 1)) + std::this_thread::yield(); + } + park.release.count_down(); + for (auto & t : threads) + t.join(); + + EXPECT_EQ(backend->writeCount("k"), static_cast(N)); + EXPECT_EQ(backend->getCount("k"), static_cast(N)); /// no cache in this task: a read per hold + std::vector etags; + for (const auto & result : results) + { + ASSERT_TRUE(result.has_value()); + const auto * committed = std::get_if(&*result); + ASSERT_NE(committed, nullptr); + etags.push_back(committed->etag); + } + for (size_t i = 1; i < etags.size(); ++i) + EXPECT_FALSE(etags[i] == etags[i - 1]); + DB::Cas::tests::expectBytes(*backend, "k", "1,2,3,4"); + auto reader = requests.admit(); + EXPECT_EQ(reader.read("k", Retry::standard())->etag, etags.back()); + EXPECT_EQ(hot_keys.laneCountForTest(), 0u); + EXPECT_EQ(hot_keys.queueDepthForTest("k"), 0u); +} + +TEST(CASHotKeys, ADecideRunsWithTheLaneMutexReleased) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + /// `queueDepthForTest` takes the lane's mutex; a `decide` run under it would deadlock this test, + /// which hangs the whole `CAS*` gate rather than being reported by a per-test timeout. The call is + /// the assertion. + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::standard()), + [&](const std::optional &) -> std::optional + { + EXPECT_EQ(hot_keys.queueDepthForTest("k"), 1u); + return String("1"); + }); + EXPECT_TRUE(std::holds_alternative(result)); +} + +TEST(CASHotKeys, AFailedEnqueueLeavesNoEmptyLaneBehind) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + hot_keys.enter_after_lane_hook_for_test = [] { throw std::bad_alloc(); }; + EXPECT_THROW(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)), std::bad_alloc); + EXPECT_EQ(hot_keys.laneCountForTest(), 0u); + EXPECT_EQ(backend->writeCount("k"), 0u); + hot_keys.enter_after_lane_hook_for_test = {}; + EXPECT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); +} + +TEST(CASHotKeys, ResultsAreTheEnginesOwn) +{ + SyncClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(false); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + +#if USE_AWS_S3 + /// The store refuses the bytes: the caller gets that `Refused`, at once. + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)); + ASSERT_TRUE(std::holds_alternative(result)); + } +#endif + /// A clean refused precondition with the store unchanged: `Conflict` carrying the occupant. + backend->refuseNextWrite("k"); + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(3)); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + EXPECT_TRUE(std::holds_alternative(conflict->seen)); + EXPECT_FALSE(conflict->any_ambiguous); + } + /// The resolve read fails at the transport under `once`: nothing observed. The failure is armed + /// from a one-shot write hook, not up front, so it lands on the write's own resolve read rather + /// than on the hold's base read, which must succeed for this sub-case to reach the write at all. + backend->refuseNextWrite("k"); + backend->onBeforeWrite("k", [&] { backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("resolve"))); }); + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::once()), appendTicket(4)); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + EXPECT_TRUE(std::holds_alternative(conflict->seen)); + } + backend->onBeforeWrite("k", [] {}); + /// An ambiguous attempt whose resolve read fails at the transport under `once`: unresolved. Same + /// one-shot arming as above, for the same reason. + backend->injectAmbiguousWrite("k"); + backend->onBeforeWrite("k", [&] { backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("resolve"))); }); + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::once()), appendTicket(5)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any); + } + backend->onBeforeWrite("k", [] {}); + /// The fence trips inside the hold, before the write: nothing sent. + bool alive = true; + auto fenced = requests.admit([&] { return alive; }); + { + WriteResult result = hot_keys.submit("k", fenced, fenced.freeze(Retry::standard()), + [&](const std::optional & current) -> std::optional + { + alive = false; + return current->bytes + ",6"; + }); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); + } + /// The fence trips after the landed write: the object carries the ticket, the caller is told so. + alive = true; + backend->onWriteCommitted("k", [&] { alive = false; }); + { + WriteResult result = hot_keys.submit("k", fenced, fenced.freeze(Retry::standard()), appendTicket(7)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + DB::Cas::tests::expectBytes(*backend, "k", "1,7"); + } + backend->onWriteCommitted("k", [] {}); + /// The engine call throws a local fault: it reaches the caller and the key is handed over. + backend->failNextWriteWith("k", std::make_exception_ptr(std::logic_error("local"))); + EXPECT_THROW(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(8)), std::logic_error); + EXPECT_EQ(hot_keys.laneCountForTest(), 0u); + EXPECT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(9)))); +} + +TEST(CASHotKeys, ABaseReadThatFailsGivesUpAsReadModifyWriteDoes) +{ + SyncClock clock; + auto backend = std::make_shared(); + backend->setAttemptTimeoutMs(1000); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + (void)orThrow(op.create("k", "1", Retry::standard()), "seed"); + + /// Enough armed failures to outlast a standard window: the read loop gives up at its deadline. + for (int i = 0; i < 64; ++i) + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("read"))); + WriteResult lane = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)); + for (int i = 0; i < 64; ++i) + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("read"))); + WriteResult verb = op.readModifyWrite("k", appendTicket(2), Retry::standard()); + + const auto * a = std::get_if(&lane); + const auto * b = std::get_if(&verb); + ASSERT_NE(a, nullptr); + ASSERT_NE(b, nullptr); + EXPECT_EQ(a->why, b->why); + EXPECT_EQ(a->deadline_source, b->deadline_source); + EXPECT_EQ(a->sent_any, b->sent_any); + EXPECT_FALSE(a->sent_any); + EXPECT_EQ(a->last_seen.index(), b->last_seen.index()); + + /// A fence that refuses the read's own reservation, and nothing smaller: the wait step passes + /// (it asks for zero), the base read is refused before its first attempt. + bool refuse_reservations = false; + Fence fence{[] { return uint64_t{0}; }, + [&](uint64_t, uint64_t needed) { return refuse_reservations && needed > 0 ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}; + CasRequests fenced_requests(backend, fence, clock.nowFn(), clock.sleepFn(), &hot_keys); + auto fenced = fenced_requests.admit(); + refuse_reservations = true; + WriteResult result = hot_keys.submit("k", fenced, fenced.freeze(Retry::standard()), appendTicket(3)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); +} + +TEST(CASHotKeys, WaitersLeaveOnTheirOwnFenceLeaseAndDeadline) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(0); + bool lease_spent = false; + Fence fence{[] { return uint64_t{0}; }, + [&](uint64_t, uint64_t) { return lease_spent ? Fence::Admit::NoBudget : Fence::Admit::Ok; }, + [](uint64_t) {}}; + CasRequests requests(backend, fence, clock.nowFn(), clock.sleepFn(), &hot_keys); + ParkFirstWrite park; + park.install(*backend, "k"); + + std::thread holder([&] + { + auto op = requests.admit(); + (void)hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)); + }); + park.parked.wait(); + + const auto gave_up_before = counter(ProfileEvents::CASRequestGaveUp); + /// A waiter whose own window ends while the holder is parked. + std::optional by_deadline; + std::thread deadline_waiter([&] + { + auto op = requests.admit(); + by_deadline = hot_keys.submit("k", op, op.freeze(Retry::within(500)), appendTicket(2)); + }); + /// A waiter whose task stops. + std::atomic alive{true}; + std::optional by_liveness; + std::thread liveness_waiter([&] + { + auto op = requests.admit([&] { return alive.load(); }); + by_liveness = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(3)); + }); + while (hot_keys.queueDepthForTest("k") < 3) + std::this_thread::yield(); + + clock.advance(600); + deadline_waiter.join(); + alive = false; + liveness_waiter.join(); + { + const auto * gave_up = std::get_if(&*by_deadline); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Policy); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(gave_up->attempts_sent, 0u); + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + } + { + const auto * gave_up = std::get_if(&*by_liveness); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); + } + /// A waiter whose lease budget is gone, and whose task has stopped at the same slice: the lease + /// speaks first, as the engine's own gate orders it. Both refusals are already in place before + /// this thread is even spawned, so its first admission check sees both at once. + lease_spent = true; + std::optional by_lease; + std::thread lease_waiter([&] + { + auto op = requests.admit([&] { return alive.load(); }); + by_lease = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(4)); + }); + lease_waiter.join(); + { + const auto * gave_up = std::get_if(&*by_lease); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Lease); + } + EXPECT_EQ(counter(ProfileEvents::CASRequestGaveUp) - gave_up_before, 3u); + EXPECT_EQ(hot_keys.queueDepthForTest("k"), 1u) << "only the parked holder remains"; + EXPECT_EQ(backend->writeCount("k"), 1u) << "no second write started"; + + lease_spent = false; + park.release.count_down(); + holder.join(); + DB::Cas::tests::expectBytes(*backend, "k", "1"); + EXPECT_EQ(hot_keys.laneCountForTest(), 0u); +} + +TEST(CASHotKeys, AThrottledHolderKeepsTheWaitersQueuedThroughItsBackoff) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(0); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + + /// The holder's own `PUT` parks here on its first attempt; once the two waiters are proven + /// queued behind it, the release makes that attempt ambiguous instead of letting it through, so + /// its resolve read (the key still absent) drives one reissue on the growing schedule while the + /// waiters sit queued through it. + std::latch parked{1}; + std::latch release{1}; + std::atomic seen{0}; + backend->onBeforeWrite("k", [&] + { + if (seen.fetch_add(1) != 0) + return; + parked.count_down(); + release.wait(); + EXPECT_EQ(hot_keys.queueDepthForTest("k"), 3u); + backend->injectAmbiguousWrite("k"); + }); + + std::vector threads; + std::vector> results(3); + threads.emplace_back([&] + { + auto op = requests.admit(); + results[0] = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)); + }); + parked.wait(); + /// Each waiter is spawned only once the previous one is seen queued, so arrival order -- and so + /// the order they write in once the holder releases -- is the spawn order. + for (int i = 1; i < 3; ++i) + { + threads.emplace_back([&, i] + { + auto op = requests.admit(); + results[i] = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(i + 1)); + }); + while (hot_keys.queueDepthForTest("k") < static_cast(i + 1)) + std::this_thread::yield(); + } + release.count_down(); + for (auto & t : threads) + t.join(); + + for (const auto & result : results) + EXPECT_TRUE(std::holds_alternative(*result)); + ASSERT_EQ(clock.sleepCount(), 1u) << "the one reissue pause, taken while the two waiters were queued"; + EXPECT_LE(clock.sleeps[0], 200u); + EXPECT_EQ(backend->writeCount("k"), 4u) << "the ambiguous attempt counts, then three landed"; + EXPECT_EQ(hot_keys.laneCountForTest(), 0u); + DB::Cas::tests::expectBytes(*backend, "k", "1,2,3"); +} + +TEST(CASHotKeys, TheNextHoldStartsFromTheLandedObjectWithoutARead) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + const auto cache_starts_before = counter(ProfileEvents::CASHotKeyCacheStarts); + + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + EXPECT_EQ(backend->getCount("k"), 1u); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)))); + EXPECT_EQ(backend->getCount("k"), 1u) << "the second hold started from the cache"; + EXPECT_EQ(counter(ProfileEvents::CASHotKeyCacheStarts) - cache_starts_before, 1u); + + /// Under `once` the one attempt is on fresh state: a read, no cached start. + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::once()), appendTicket(3)))); + EXPECT_EQ(backend->getCount("k"), 2u); + /// `expectBytes` issues its own read, so it comes after every count assertion, not between them. + DB::Cas::tests::expectBytes(*backend, "k", "1,2,3"); +} + +TEST(CASHotKeys, AnExternalWriterCostsOneResolveReadAndOneRetry) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + CasRequests external_requests = DB::Cas::tests::openRequestsForTest(backend); + auto external = external_requests.admit(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + + const auto current = external.read("k", Retry::standard()); + (void)orThrow(external.replace("k", "E", current->etag, Retry::standard()), "external"); + const uint64_t gets_before = backend->getCount("k"); + const uint64_t writes_before = backend->writeCount("k"); + + /// The caller's loop: submit, and on a conflict submit again after the flat pause. + std::optional result; + for (int i = 0; i < 3 && !result; ++i) + { + WriteResult attempt = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)); + if (std::holds_alternative(attempt)) + op.pause(Retry::conflictBackoff()); + else + result = std::move(attempt); + } + ASSERT_TRUE(result && std::holds_alternative(*result)); + EXPECT_EQ(backend->getCount("k") - gets_before, 1u) << "one resolve read"; + EXPECT_EQ(backend->writeCount("k") - writes_before, 2u) << "one refused write, one that landed"; + /// The submission after that starts from the cache. + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(3)))); + EXPECT_EQ(backend->getCount("k") - gets_before, 1u); + /// `expectBytes` issues its own read, so it comes after every count assertion, not between them. + DB::Cas::tests::expectBytes(*backend, "k", "E,2,3"); +} + +TEST(CASHotKeys, MalformedBytesRepairedExternallyRaiseNoCorruptionVerdict) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + CasRequests external_requests = DB::Cas::tests::openRequestsForTest(backend); + auto external = external_requests.admit(); + /// A decide that refuses bytes it cannot decode, as the catalog's does. + const CasHotKeys::Decide strict = [](const std::optional & current) -> std::optional + { + if (current && current->bytes.find("garbage") != String::npos) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "not a ticket list"); + return current ? current->bytes + ",9" : String("9"); + }; + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), strict))); + + auto current = external.read("k", Retry::standard()); + const Etag garbage = *orThrow(external.replace("k", "garbage", current->etag, Retry::standard()), "break"); + /// The lane's next hold starts from its cache, loses to the garbage, and remembers the garbage + /// its resolve read saw; the caller pauses and submits again. + WriteResult first = hot_keys.submit("k", op, op.freeze(Retry::standard()), strict); + ASSERT_TRUE(std::holds_alternative(first)); + (void)orThrow(external.replace("k", "9", garbage, Retry::standard()), "repair"); + const auto reread_before = counter(ProfileEvents::CASHotKeyCacheVerdictsReread); + /// The verdict on the cached garbage is not delivered: one read, and the decide lands on the repair. + WriteResult second = hot_keys.submit("k", op, op.freeze(Retry::standard()), strict); + ASSERT_TRUE(std::holds_alternative(second)); + EXPECT_EQ(counter(ProfileEvents::CASHotKeyCacheVerdictsReread) - reread_before, 1u); + DB::Cas::tests::expectBytes(*backend, "k", "9,9"); + /// And when the read is garbage too, that is the real corruption. + current = external.read("k", Retry::standard()); + (void)orThrow(external.replace("k", "garbage", current->etag, Retry::standard()), "break again"); + (void)hot_keys.submit("k", op, op.freeze(Retry::standard()), strict); /// conflict: the cache now holds garbage + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { (void)hot_keys.submit("k", op, op.freeze(Retry::standard()), strict); }); +} + +TEST(CASHotKeys, ADeclineOnAHintIsRerenderedOnARead) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + CasRequests external_requests = DB::Cas::tests::openRequestsForTest(backend); + auto external = external_requests.admit(); + /// Writes "1" once and declines while the object already says "1". + const CasHotKeys::Decide idempotent = [](const std::optional & current) -> std::optional + { + if (current && current->bytes == "1") + return std::nullopt; + return String("1"); + }; + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), idempotent))); + /// On a fresh read the decline is the caller's answer. + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::once()), idempotent); + const auto * declined = std::get_if(&result); + ASSERT_NE(declined, nullptr); + EXPECT_TRUE(std::holds_alternative(declined->seen)); + } + /// An external writer replaces the object; the cached hint still says "1", so the decide would + /// decline on it. The decline is not delivered: the lane reads and the decide writes. + const auto current = external.read("k", Retry::standard()); + (void)orThrow(external.replace("k", "0", current->etag, Retry::standard()), "external"); + const uint64_t gets_before = backend->getCount("k"); + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::standard()), idempotent); + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->getCount("k") - gets_before, 1u); + DB::Cas::tests::expectBytes(*backend, "k", "1"); + /// On an absent key the decline names absence. + WriteResult absent = hot_keys.submit("missing", op, op.freeze(Retry::standard()), + [](const std::optional &) -> std::optional { return std::nullopt; }); + const auto * declined = std::get_if(&absent); + ASSERT_NE(declined, nullptr); + EXPECT_TRUE(std::holds_alternative(declined->seen)); +} + +TEST(CASHotKeys, TheCacheForgetsWhatItCannotVouchFor) +{ + SyncClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(false); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + const auto reads = [&] { return backend->getCount("k"); }; + + uint64_t before = reads(); +#if USE_AWS_S3 + /// Refused: dropped, the next hold reads. + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)))); + EXPECT_EQ(reads(), before); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)))); + EXPECT_EQ(reads(), before + 1); +#endif + + /// Ticket 2 only lands when the S3-only Refused sub-case above runs it; the terminal bytes below + /// follow the same guard. +#if USE_AWS_S3 + constexpr auto kFinalBytes = "1,2,3,4,5,6,7"; +#else + constexpr auto kFinalBytes = "1,3,4,5,6,7"; +#endif + + /// Unresolved after a send: dropped. A single-attempt submission never starts from the cache, so + /// arming the ambiguity and the resolve-read failure up front would let the hold's own base read + /// consume them; a one-shot write hook lands both on the write's own resolve read instead. + backend->onBeforeWrite("k", [&] + { + backend->injectAmbiguousWrite("k"); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("resolve"))); + }); + before = reads(); + { + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::once()), appendTicket(3)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any); + } + backend->onBeforeWrite("k", [] {}); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(3)))); + EXPECT_EQ(reads(), before + 3); /// the base read, the failed resolve read, the next hold's read after the entry was dropped + + /// An exception out of the write: dropped. + backend->failNextWriteWith("k", std::make_exception_ptr(std::logic_error("local"))); + before = reads(); + EXPECT_THROW(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(4)), std::logic_error); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(4)))); + EXPECT_EQ(reads(), before + 1); + + /// A give-up that sent nothing leaves the entry as it was: the next hold starts from it. + bool alive = true; + auto fenced = requests.admit([&] { return alive; }); + before = reads(); + WriteResult nothing_sent = hot_keys.submit("k", fenced, fenced.freeze(Retry::standard()), + [&](const std::optional & current) -> std::optional { alive = false; return current->bytes + ",5"; }); + ASSERT_TRUE(std::holds_alternative(nothing_sent)); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(5)))); + EXPECT_EQ(reads(), before); + + /// A fill that throws after a landed write: the result stands, the next hold reads. + hot_keys.cache_fill_hook_for_test = [] { throw std::bad_alloc(); }; + before = reads(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(6)))); + hot_keys.cache_fill_hook_for_test = {}; + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(7)))); + EXPECT_EQ(reads(), before + 1); + DB::Cas::tests::expectBytes(*backend, "k", kFinalBytes); +} + +TEST(CASHotKeys, TheBudgetBoundsBytesAndEntries) +{ + SyncClock clock; + auto backend = std::make_shared(); + /// Two entries of one-byte objects weigh 2 x (1 + 1 + etag + 64); a budget of one entry and a half + /// holds one at a time. + auto probe_op_requests = DB::Cas::tests::openRequestsForTest(backend); + auto probe = probe_op_requests.admit(); + const size_t etag_bytes = orThrow(probe.create("probe", "x", Retry::standard()), "probe")->render().size(); + const uint64_t one_entry = 1 + 1 + etag_bytes + 64; + CasHotKeys hot_keys(one_entry + one_entry / 2); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + const CasHotKeys::Decide one_byte = [](const std::optional &) -> std::optional { return String("x"); }; + + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("a", op, op.freeze(Retry::standard()), one_byte))); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("b", op, op.freeze(Retry::standard()), one_byte))); + EXPECT_EQ(hot_keys.cacheEntriesForTest(), 1u) << "the older entry was evicted"; + const uint64_t gets_a = backend->getCount("a"); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("a", op, op.freeze(Retry::standard()), one_byte))); + EXPECT_EQ(backend->getCount("a"), gets_a + 1) << "the evicted key reads"; + + /// An object above the budget is not stored. + const CasHotKeys::Decide big = [&](const std::optional &) -> std::optional { return String(one_entry * 2, 'y'); }; + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("c", op, op.freeze(Retry::standard()), big))); + const uint64_t gets_c = backend->getCount("c"); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("c", op, op.freeze(Retry::standard()), big))); + EXPECT_EQ(backend->getCount("c"), gets_c + 1); + + /// Empty objects weigh their key and their allowance: N of them stay bounded by the budget. + CasHotKeys small(4 * one_entry); + CasRequests small_requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &small); + auto small_op = small_requests.admit(); + const CasHotKeys::Decide empty = [](const std::optional &) -> std::optional { return String(); }; + for (int i = 0; i < 40; ++i) + ASSERT_TRUE(std::holds_alternative(small.submit("e" + std::to_string(i), small_op, small_op.freeze(Retry::standard()), empty))); + EXPECT_LE(small.cacheEntriesForTest(), 4u); +} + +TEST(CASHotKeys, ACachedStartPastTheDeadlineSendsNothing) +{ + SyncClock clock; + auto backend = std::make_shared(); + backend->setAttemptTimeoutMs(1000); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + + /// The window fits the wait's zero reservation but not the write's two attempt envelopes. + int decided = 0; + const uint64_t writes_before = backend->writeCount("k"); + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::within(500)), + [&](const std::optional & current) -> std::optional { ++decided; return current->bytes + ",2"; }); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(decided, 1); + EXPECT_EQ(backend->writeCount("k"), writes_before); +} + +TEST(CASHotKeys, AnIdenticalCandidateLandedByAnotherServerIsTheEnginesCommit) +{ + SyncClock clock; + auto backend = std::make_shared(); + CasHotKeys hot_keys(16ULL << 20); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys); + auto op = requests.admit(); + CasRequests external_requests = DB::Cas::tests::openRequestsForTest(backend); + auto external = external_requests.admit(); + ASSERT_TRUE(std::holds_alternative(hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(1)))); + + /// Another server lands exactly the candidate this hold will compute from its stale hint, and this + /// hold's own refused write loses its answer. The resolve read finds the candidate's bytes under + /// the moved incarnation: the engine's own rule calls that landed. + const auto current = external.read("k", Retry::standard()); + const Etag theirs = *orThrow(external.replace("k", "1,2", current->etag, Retry::standard()), "identical"); + backend->injectAmbiguousWrite("k"); + WriteResult result = hot_keys.submit("k", op, op.freeze(Retry::standard()), appendTicket(2)); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_TRUE(committed->resolved_by_read); + EXPECT_TRUE(committed->etag == theirs); + DB::Cas::tests::expectBytes(*backend, "k", "1,2"); +} diff --git a/src/Disks/tests/gtest_cas_ids.cpp b/src/Disks/tests/gtest_cas_ids.cpp index 046696124bd3..578b3c99b00c 100644 --- a/src/Disks/tests/gtest_cas_ids.cpp +++ b/src/Disks/tests/gtest_cas_ids.cpp @@ -29,13 +29,8 @@ TEST(CASIds, HexU128RoundTrip) EXPECT_THROW(hexToU128("0123"), DB::Exception); // wrong length } -TEST(CASToken, Basics) -{ - Token a{"etag-1", TokenType::ETag}; - Token b{"etag-1", TokenType::ETag}; - Token c{"etag-2", TokenType::ETag}; - EXPECT_EQ(a, b); - EXPECT_NE(a, c); - EXPECT_TRUE(Token{}.empty()); - EXPECT_FALSE(a.empty()); -} +/// `Token`'s free-standing equality/emptiness was deleted with the type itself: `Etag` has no public +/// constructor (minted only by `CasRequests::mint`/`tryMint`) and no `empty()`, so this test's subject +/// no longer exists to construct by hand. `Etag` equality and inequality are exercised by +/// `CASInMemory.PutIfAbsentAndGet` and `CASInMemory.OverwriteIsTokenExactAndMintsFreshToken` in +/// gtest_cas_backend.cpp, which compare an observed incarnation against the one a prior write returned. diff --git a/src/Disks/tests/gtest_cas_inspect.cpp b/src/Disks/tests/gtest_cas_inspect.cpp index d807bec3f931..e031e59f6ae2 100644 --- a/src/Disks/tests/gtest_cas_inspect.cpp +++ b/src/Disks/tests/gtest_cas_inspect.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include #include @@ -51,7 +52,7 @@ TEST(CASInspect, RendersSetPublishedAtOpWithNoPayloadSizeKey) const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); - EXPECT_NE(json.find(R"("kind":"SetPublishedAt")"), String::npos) << json; + EXPECT_NE(json.find(R"("kind":"set_published_at")"), String::npos) << json; EXPECT_EQ(json.find("payload"), String::npos) << json; } @@ -75,10 +76,99 @@ TEST(CASInspect, RendersEpochSealTxnWithPrevEpochSeal) const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); - EXPECT_NE(json.find(R"("kind":"EpochSeal")"), String::npos) << json; + EXPECT_NE(json.find(R"("kind":"epoch_seal")"), String::npos) << json; EXPECT_NE(json.find(R"("prev_epoch_seal":{"writer_epoch":2,"ref_sequence":9})"), String::npos) << json; } +/// The remaining two `RefOpKind` words this file's other tests do not exercise: a namespace's birth +/// record and its removal terminator. +TEST(CASInspect, RendersNamespaceBirthAndRemoveNamespaceOpKinds) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + + RefLogTxn birth_txn; + birth_txn.ns = ns.string(); + birth_txn.txn_id = RefTxnId{1, 1}; + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + birth_txn.ops.push_back(birth); + const String birth_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), birth_txn.txn_id); + const String birth_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(birth_txn)); + const String birth_json = caInspectToJson( + layout, birth_key, birth_bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(birth_json.find(R"("kind":"namespace_birth")"), String::npos) << birth_json; + + RefLogTxn remove_txn; + remove_txn.ns = ns.string(); + remove_txn.txn_id = RefTxnId{1, 2}; + RefOp remove; + remove.kind = RefOpKind::RemoveNamespace; + remove_txn.ops.push_back(remove); + const String remove_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), remove_txn.txn_id); + const String remove_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(remove_txn)); + const String remove_json = caInspectToJson( + layout, remove_key, remove_bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(remove_json.find(R"("kind":"remove_namespace")"), String::npos) << remove_json; +} + +/// `RefOwnerKind` renders as its full wire word (`committed`/`precommit`), not the enumerator spelling, +/// at both binding slots an `owner_transition` op carries. +TEST(CASInspect, RendersRefOwnerKindWireWords) +{ + const Layout layout("p"); + const RootNamespace ns{"srv1/db/tbl"}; + const RefTxnId id{1, 3}; + + RefLogTxn txn; + txn.ns = ns.string(); + txn.txn_id = id; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Committed, "all_1_1_0", manifestRef(1, 1, 1)}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "all_1_1_0", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + + const String key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id); + const String bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); + + const String json = caInspectToJson(layout, key, bytes, DB::Cas::tests::fixture::fixtureLife(ns)); + EXPECT_NE(json.find(R"("old_binding":{"kind":"committed")"), String::npos) << json; + EXPECT_NE(json.find(R"("new_binding":{"kind":"precommit")"), String::npos) << json; +} + +/// A recorded incarnation's dialect renders as its full wire word; the blob-target-run test below +/// covers `emulated`, so this pins the other two (`etag`/`generation`) via a second condemned-row-only run. +TEST(CASInspect, RendersTokenTypeWireWordsEtagAndGeneration) +{ + const Layout layout("p"); + + SourceEdgeRecord etag_rec; + etag_rec.ref = bh(1); + etag_rec.source_id = UInt128{0}; + etag_rec.marker = RunMarker::Condemned; + etag_rec.token = PersistedEtag{"etag", "v-etag"}; + + SourceEdgeRecord gen_rec; + gen_rec.ref = bh(1); + gen_rec.source_id = UInt128{1}; + gen_rec.marker = RunMarker::Condemned; + gen_rec.token = PersistedEtag{"generation", "v-gen"}; + + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(etag_rec); + writer.append(gen_rec); + writer.finish(); + out.finalize(); + const String bytes = out.str(); + + const String key = layout.blobTargetRunKey(/*generation*/3, /*attempt*/0, /*shard*/0, /*seq*/0); + const String json = caInspectToJson(layout, key, bytes); + EXPECT_NE(json.find(R"("type":"etag")"), String::npos) << json; + EXPECT_NE(json.find(R"("type":"generation")"), String::npos) << json; +} + TEST(CASInspect, RendersCommittedRowWithNoPayloadSizeKey) { const Layout layout("p"); @@ -117,9 +207,9 @@ TEST(CASInspect, RendersBlobTargetRunEdgeAndCondemnedRows) SourceEdgeRecord condemned_rec; condemned_rec.ref = bh(1); condemned_rec.source_id = UInt128{0}; - condemned_rec.marker = kCondemned; + condemned_rec.marker = RunMarker::Condemned; condemned_rec.delete_pending = true; - condemned_rec.token = Token{.value = "etag-1", .type = TokenType::Emulated}; + condemned_rec.token = PersistedEtag{"emulated", "etag-1"}; condemned_rec.size = 123; condemned_rec.condemn_round = 7; condemned_rec.marker_confirmed = true; @@ -127,7 +217,7 @@ TEST(CASInspect, RendersBlobTargetRunEdgeAndCondemnedRows) SourceEdgeRecord edge_rec; edge_rec.ref = bh(2); edge_rec.source_id = UInt128(9); - edge_rec.marker = kEdgeActive; + edge_rec.marker = RunMarker::Edge; DB::WriteBufferFromOwnString out; SourceEdgeRunWriter writer(out); @@ -147,6 +237,7 @@ TEST(CASInspect, RendersBlobTargetRunEdgeAndCondemnedRows) EXPECT_NE(json.find(R"("delete_pending":true)"), String::npos) << json; EXPECT_NE(json.find(R"("condemn_round":7)"), String::npos) << json; EXPECT_NE(json.find(R"("value":"etag-1")"), String::npos) << json; + EXPECT_NE(json.find(R"("type":"emulated")"), String::npos) << json; EXPECT_NE(json.find(R"("rows":2)"), String::npos) << json; EXPECT_NE(json.find(R"("distinct_blobs":2)"), String::npos) << json; EXPECT_NE(json.find(R"("edges":1)"), String::npos) << json; @@ -198,6 +289,32 @@ TEST(CASInspect, RendersRefCkptAbsencesAsExplicitNulls) EXPECT_NE(json.find(R"("last_epoch_seal":null)"), String::npos) << json; } +/// `CoverageClass` renders as its full wire word, not the enumerator's numeric value: `cas-inspect` is +/// exactly the tool an operator reaches for to read a fold seal directly, so a coverage row that still +/// printed a bare integer would send them back to this file's comment to decode it. +TEST(CASInspect, RendersCoverageClassificationWireWords) +{ + const Layout layout("p"); + CasFoldSeal seal; + seal.generation = 3; + seal.parent_generation = 2; + seal.ref_lives[UInt128{1}].coverage = RefCoverage{.classification = CoverageClass::Absent}; + seal.ref_lives[UInt128{2}].coverage = RefCoverage{.classification = CoverageClass::Unchanged}; + seal.ref_lives[UInt128{3}].coverage + = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}; + seal.ref_lives[UInt128{4}].coverage = RefCoverage{ + .classification = CoverageClass::Clamped, + .hold = RefHold{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{1, 2}, + .retry_count = 0, .next_retry_round = 1}}; + + const String key = layout.foldSealKey(/*generation*/3, /*attempt*/0); + const String json = caInspectToJson(layout, key, encodeFoldSeal(seal)); + EXPECT_NE(json.find(R"("classification":"absent")"), String::npos) << json; + EXPECT_NE(json.find(R"("classification":"unchanged")"), String::npos) << json; + EXPECT_NE(json.find(R"("classification":"folded")"), String::npos) << json; + EXPECT_NE(json.find(R"("classification":"clamped")"), String::npos) << json; +} + /// A listed physical id cannot supply a namespace. Inspect must receive the unique catalog join, and /// a different logical spelling at the same id is rejected by the decoded object's own namespace. TEST(CASInspect, RefObjectRequiresTheExactCatalogResolution) diff --git a/src/Disks/tests/gtest_cas_iobjectstorage_defaults.cpp b/src/Disks/tests/gtest_cas_iobjectstorage_defaults.cpp new file mode 100644 index 000000000000..e89b1a28466a --- /dev/null +++ b/src/Disks/tests/gtest_cas_iobjectstorage_defaults.cpp @@ -0,0 +1,161 @@ +#include + +#include + +#include + +/// `IObjectStorage::removeObjectsIfExistUnderProfile` has three siblings (`iterate`, +/// `tryGetObjectMetadataWithNativeToken`, `removeObjectIfTokenMatches`) whose defaults all forward a +/// Default-profile request to the plain, no-profile method and refuse only SingleAttempt. This file +/// pins that `removeObjectsIfExistUnderProfile` follows the same rule, using a minimal stub storage +/// that implements nothing beyond what `IObjectStorage` requires. + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +} + +namespace +{ + +/// Implements only what `IObjectStorage` declares pure; every method a case below does not exercise +/// throws if called, so a test that reaches one it did not expect fails loudly instead of silently +/// doing the wrong thing. +class MinimalObjectStorage : public IObjectStorage +{ +public: + std::string getName() const override + { + return "MinimalObjectStorage"; + } + + ObjectStorageType getType() const override + { + return ObjectStorageType::None; + } + + std::string getCommonKeyPrefix() const override + { + return ""; + } + + std::string getDescription() const override + { + return "MinimalObjectStorage (test stub)"; + } + + bool exists(const StoredObject &) const override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + ObjectMetadata getObjectMetadata(const std::string &, bool) const override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + std::optional tryGetObjectMetadata(const std::string &, bool) const override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + std::unique_ptr readObject( + const StoredObject &, const ReadSettings &, std::optional, bool, bool) const override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + std::unique_ptr writeObject( + const StoredObject &, WriteMode, std::optional, size_t, const WriteSettings &) override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + bool isRemote() const override + { + return true; + } + + void removeObjectIfExists(const StoredObject &) override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + /// The method under test: `removeObjectsIfExistUnderProfile`'s Default-profile default forwards here. + void removeObjectsIfExist(const StoredObjects & objects) override + { + ++remove_objects_if_exist_calls; + last_removed_objects = objects; + } + + void copyObject( + const StoredObject &, const StoredObject &, const ReadSettings &, const WriteSettings &, std::optional) override + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "not used by this test"); + } + + void shutdown() override + { + } + + void startup() override + { + } + + String getObjectsNamespace() const override + { + return ""; + } + + ObjectStorageKeyGeneratorPtr createKeyGenerator() const override + { + return nullptr; + } + + size_t remove_objects_if_exist_calls = 0; + StoredObjects last_removed_objects; +}; + +} + +TEST(CASIObjectStorageDefaults, RemoveObjectsIfExistUnderProfileDefaultForwards) +{ + MinimalObjectStorage storage; + const StoredObjects objects{StoredObject("a"), StoredObject("b")}; + + ObjectStorageControlRequest request; + request.profile = ObjectStorageRetryProfile::Default; + + storage.removeObjectsIfExistUnderProfile(objects, request); + + EXPECT_EQ(storage.remove_objects_if_exist_calls, 1u); + ASSERT_EQ(storage.last_removed_objects.size(), 2u); + EXPECT_EQ(storage.last_removed_objects[0].remote_path, "a"); + EXPECT_EQ(storage.last_removed_objects[1].remote_path, "b"); +} + +TEST(CASIObjectStorageDefaults, RemoveObjectsIfExistUnderProfileSingleAttemptThrows) +{ + MinimalObjectStorage storage; + const StoredObjects objects{StoredObject("a")}; + + ObjectStorageControlRequest request; + request.profile = ObjectStorageRetryProfile::SingleAttempt; + + try + { + storage.removeObjectsIfExistUnderProfile(objects, request); + FAIL() << "expected a SingleAttempt batch-remove request to be refused"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::NOT_IMPLEMENTED); + } + + EXPECT_EQ(storage.remove_objects_if_exist_calls, 0u); +} + +} diff --git a/src/Disks/tests/gtest_cas_json_writer.cpp b/src/Disks/tests/gtest_cas_json_writer.cpp index f04b89255818..ea12afa32bcb 100644 --- a/src/Disks/tests/gtest_cas_json_writer.cpp +++ b/src/Disks/tests/gtest_cas_json_writer.cpp @@ -15,17 +15,20 @@ TEST(CASJsonWriter, KeyValueSequenceMatchesCanonicalShape) { CasJsonWriter w; bool first = true; - w.key("we", first); + /// The names are shape labels, not format keys: this test is about the writer's primitives, and + /// borrowing a real wire spelling would put this file in every vocabulary sweep for no reason. + w.key("u64_string_field", first); w.u64StringValue(7); - w.key("mo", first); + w.key("number_field", first); w.u64Number(3); - w.key("ok", first); + w.key("bool_field", first); w.boolValue(true); - w.key("o", "me", first); + w.key("second_u64_string_field", first); w.u64StringValue(1); w.closeObject(first); w.newline(); - EXPECT_EQ(std::move(w).take(), "{\"we\":\"7\",\"mo\":3,\"ok\":true,\"ome\":\"1\"}\n"); + EXPECT_EQ(std::move(w).take(), + "{\"u64_string_field\":\"7\",\"number_field\":3,\"bool_field\":true,\"second_u64_string_field\":\"1\"}\n"); } TEST(CASJsonWriter, EmptyObjectAndClear) @@ -212,3 +215,32 @@ TEST(CASJsonWriterVocab, MatchesReferenceVocabulary) ref.finalize(); EXPECT_EQ(std::move(w).take(), ref.str()); } + +TEST(CASJsonWriter, WireKeyFieldHelpersMatchThePrimitivePairs) +{ + CasJsonWriter w; + bool first = true; + constexpr WireKey k_word{"word_field"}; + constexpr WireKey k_str{"string_field"}; + constexpr WireKey k_u64s{"u64_string_field"}; + constexpr WireKey k_num{"number_field"}; + constexpr WireKey k_hex{"hex_field"}; + constexpr WireKey k_bool{"bool_field"}; + writeWordField(w, k_word, "clean", first); + writeStringField(w, k_str, "host-1", first); + writeU64StringField(w, k_u64s, 7, first); + writeNumberField(w, k_num, 1752537630000, first); + writeHex128Field(w, k_hex, DB::UInt128{1}, first); + writeBoolField(w, k_bool, false, first); + w.closeObject(first); + w.newline(); + EXPECT_EQ(std::move(w).take(), + "{\"word_field\":\"clean\",\"string_field\":\"host-1\",\"u64_string_field\":\"7\"," + "\"number_field\":1752537630000," + "\"hex_field\":\"00000000000000000000000000000001\",\"bool_field\":false}\n"); + + /// The reader-side comparison contract: a String key compares against the constant. + String key = "word_field"; + EXPECT_TRUE(key == k_word); + EXPECT_FALSE(key == k_str); +} diff --git a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp index b93166c88eac..8d6133b75592 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp @@ -36,11 +36,12 @@ const String kSrid = "test"; /// so a test can restore it verbatim later (scenario d). String deleteKeyReturningBody(Backend & backend, const String & key) { - const auto got = backend.get(key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(key, Retry::once()); EXPECT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; if (!got) return {}; - backend.deleteExact(key, got->token); + (*op).remove(key, got->etag, Retry::once()); return got->bytes; } @@ -49,49 +50,66 @@ String deleteKeyReturningBody(Backend & backend, const String & key) /// fresh incarnation and returns true. Mirrors gtest_cas_pool.cpp's `fenceOutMount`. void fenceOutMount(Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, Retry::once()); ASSERT_TRUE(got.has_value()); MountLease m = decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, PutOutcome::Done); + const auto put = (*op).replace(mount_key, encodeMountLease(m), got->etag, Retry::once()); + ASSERT_TRUE(std::holds_alternative(put)); } -/// A Backend decorator whose head/get/list throw an untyped transport error while `fail` is armed. Starts -/// DISARMED so `Pool::open` succeeds; a test arms it only to make the identity probe inconclusive. Mirrors +/// A Backend decorator whose reads, heads and lists throw a transport-classified error while `fail` is +/// armed, counting every attempt so a test can prove the probe path was actually reached (and stopped +/// where it should) rather than some other short-circuit. Starts DISARMED so `Pool::open` succeeds; a +/// test arms it only to make the identity probe inconclusive. Mirrors /// gtest_cas_sentinel_probe.cpp's `TransportFaultBackend`, but toggleable AFTER open. +/// +/// The fault is `Poco::TimeoutException`: `Backend::probeSentinelRaw`'s default implementation (the one +/// `InMemoryBackend` uses) calls `head`/`read` directly and folds ANY exception from either into +/// `Indeterminate` with its own `catch (...)` -- so the exception never reaches `CasOperation`'s +/// transport-vs-local classification at all here. A `Poco::TimeoutException` is still the right class to +/// inject: it is what a real backend's probe would actually throw, and the point of the counters below +/// is to prove `head` was reached and actually failed, not skipped by some other short-circuit. +/// `tryRemountOnce` retries its own whole chain internally (well past the single probe attempt), so +/// the exact count per call is not pinned here -- only that a call growing it proves the fault path +/// stayed live across it, rather than a stale verdict being served from a cache. class ToggleableTransportFaultBackend final : public InMemoryBackend { public: - /// Unhide the base convenience overloads, matching every other Backend subclass in this suite. - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - - HeadResult head(const String & key) override + /// Unhide the LEGACY convenience overloads that the primitive overrides below would otherwise hide. + using Backend::head; + using Backend::list; + + std::optional head(const String & key, TransportAccess & access) override { + ++head_attempts; if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::head(key); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::head(key, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { + ++read_attempts; if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::get(key, range); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::read(key, access); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { + ++list_attempts; if (fail.load()) - throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::list(prefix, cursor, limit); + throw Poco::TimeoutException("injected fault: transport error"); + return InMemoryBackend::list(prefix, cursor, limit, access); } std::atomic fail{false}; + std::atomic head_attempts{0}; + std::atomic read_attempts{0}; + std::atomic list_attempts{0}; }; } @@ -128,7 +146,7 @@ TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) backend->resetCounts(); EXPECT_FALSE(store->tryRemountOnce()); EXPECT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); - EXPECT_EQ(backend->putTotal(), 0u) << "a terminal-IdentityLost gate probe must never claim, allocate, or write"; + EXPECT_EQ(backend->writeTotal(), 0u) << "a terminal-IdentityLost gate probe must never claim, allocate, or write"; EXPECT_GE(backend->headCount(meta_key), 1u) << "the gate still probes _pool_meta authoritatively"; } @@ -165,11 +183,12 @@ TEST(CASLifecycleCondition, PoolMetaForeignPoolIdEntersVanishedReplacedImmediate /// Overwrite `_pool_meta` with a FOREIGN pool_id (identity replaced); the object stays present. const String meta_key = store->layout().poolMetaKey(); - const auto got = backend->get(meta_key); + DB::Cas::tests::OperationForTest op(*backend); + const auto got = (*op).read(meta_key, Retry::once()); ASSERT_TRUE(got.has_value()); PoolMeta foreign = decodePoolMeta(got->bytes); foreign.pool_id = foreign.pool_id + DB::UInt128(1); - ASSERT_EQ(backend->putOverwrite(meta_key, encodePoolMeta(foreign), got->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative((*op).replace(meta_key, encodePoolMeta(foreign), got->etag, Retry::once()))); EXPECT_FALSE(store->tryRemountOnce()); EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedReplaced); @@ -186,7 +205,8 @@ TEST(CASLifecycleCondition, PoolMetaAlgosUsedDifferIsNotReplacementRecoveryProce auto store = DB::Cas::tests::openPoolForTest(backend); const String meta_key = store->layout().poolMetaKey(); - const auto got = backend->get(meta_key); + DB::Cas::tests::OperationForTest op(*backend); + const auto got = (*op).read(meta_key, Retry::once()); ASSERT_TRUE(got.has_value()); PoolMeta mutated = decodePoolMeta(got->bytes); /// pool_id + blob_header_len UNCHANGED; only `algos_used` gains a member (a mutable field, [B6]). @@ -194,7 +214,7 @@ TEST(CASLifecycleCondition, PoolMetaAlgosUsedDifferIsNotReplacementRecoveryProce ASSERT_FALSE(std::binary_search(mutated.algos_used.begin(), mutated.algos_used.end(), extra)); mutated.algos_used.push_back(extra); std::sort(mutated.algos_used.begin(), mutated.algos_used.end()); - ASSERT_EQ(backend->putOverwrite(meta_key, encodePoolMeta(mutated), got->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative((*op).replace(meta_key, encodePoolMeta(mutated), got->etag, Retry::once()))); /// Fence out the mount so the (correctly non-replacement) recovery cleanly reclaims a fresh incarnation. fenceOutMount(*backend, store->layout().mountKey(kSrid)); @@ -223,8 +243,9 @@ TEST(CASLifecycleCondition, IdentityLostDoesNotAutoReviveWhenSentinelsRestored) ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); /// Restore both sentinels verbatim (a backup restore with matching identity). - ASSERT_EQ(backend->putIfAbsent(meta_key, meta_body).outcome, PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(owner_key, owner_body).outcome, PutOutcome::Done); + DB::Cas::tests::OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).create(meta_key, meta_body, Retry::once()))); + ASSERT_TRUE(std::holds_alternative((*op).create(owner_key, owner_body, Retry::once()))); /// The gate now sees Present+match, but the state is `IdentityLost`, so it stays fail-loud. EXPECT_FALSE(store->tryRemountOnce()); @@ -241,17 +262,25 @@ TEST(CASLifecycleCondition, ProbeTransportErrorStaysTransientAndRetries) auto store = DB::Cas::tests::openPoolForTest(backend); ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); - /// Arm the transport fault: the identity probe's head/get/list now throw → Indeterminate. + /// Arm the transport fault: every request the identity probe issues now throws → Indeterminate. backend->fail.store(true); EXPECT_FALSE(store->tryRemountOnce()); EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive); EXPECT_FALSE(store->isVanished()); EXPECT_NO_THROW(store->throwIfLifecycleTerminal()); - - /// A second attempt with the fault still armed remains transient (retries continue). + /// The `_pool_meta` probe was actually reached and actually failed at `head` -- proving the + /// TransientNotLive verdict above came from the probe's own `Indeterminate` classification, not + /// from some other short-circuit that never touched the fault at all. + const uint64_t first_head_attempts = backend->head_attempts.load(); + EXPECT_GT(first_head_attempts, 0u); + + /// A second attempt with the fault still armed remains transient (retries continue) and probes + /// again -- proving each `tryRemountOnce` re-probes rather than caching the first call's + /// inconclusive verdict. EXPECT_FALSE(store->tryRemountOnce()); EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive); + EXPECT_GT(backend->head_attempts.load(), first_head_attempts); /// Disarm before teardown so `~Pool()`'s clean-farewell write is not fighting the injected fault. backend->fail.store(false); diff --git a/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp index 154c365a2a6d..ea504c894641 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp @@ -66,10 +66,11 @@ void commitOnePart(ContentAddressedMetadataStorage & storage) /// into a NATURAL `IdentityLost`. Mirrors gtest_cas_forget.cpp / gtest_cas_lifecycle_condition.cpp. void deleteKeyExact(DB::Cas::Backend & backend, const String & key) { - const auto got = backend.get(key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(key, DB::Cas::Retry::standard()); ASSERT_TRUE(got.has_value()) << "expected '" << key << "' to exist before deletion"; if (got) - backend.deleteExact(key, got->token); + (*op).remove(key, got->etag, DB::Cas::Retry::standard()); } } diff --git a/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp b/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp index 1b48718663cb..368430755275 100644 --- a/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp +++ b/src/Disks/tests/gtest_cas_list_liar_end_to_end.cpp @@ -11,6 +11,7 @@ #include "cas_test_helpers.h" #include +#include #include #include @@ -34,8 +35,8 @@ /// anyway -- while a genuinely absent expected id is a durable HOLD, never a silent skip. /// /// This file is that claim stated end to end, against a store that lies exactly the way the real one -/// did. `setListOmissions` names the omitted keys; `get`/`head`/`putIfAbsent`/`casPut`/`deleteExact` -/// keep serving them honestly. Each test below asserts the lie changed NOTHING -- not the folded +/// did. `setListOmissions` names the omitted keys; every other primitive keeps serving them honestly. +/// Each test below asserts the lie changed NOTHING -- not the folded /// edges, not the cursor, not the recovered table, not fsck's verdict -- and the arms that are about /// reclamation additionally assert that reclamation still happens, so "nothing was deleted" can never /// pass for "the lie was harmless". @@ -64,6 +65,12 @@ String blobKeyOf(const Layout & layout, const DB::UInt128 & hash) return layout.blobKey(legacyMetaTestRef(hash)); } +bool headExists(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + /// The sealed fold cursor for `ns` as a full `RefTxnId`. Every fixture here writes ids inside writer /// epoch 1, which is the assumption `foldCursorOf` (returning the sequence alone) already makes. RefTxnId sealedCursorOf(Backend & backend, const Layout & layout, const RootNamespace & ns) @@ -254,6 +261,19 @@ TEST(CASListLiarEndToEnd, RecoveryUnderTheSameLieReconstructsExactlyTheTruth) auto lying = openRecoveryPool(lying_backend); const std::map recovered = refsOf(lying, ns); + /// Recovery reads every record by exact key and asks no listing what to read next, and an existing + /// pool reopens on the exact read of `_pool_meta` alone -- so nothing above ever LISTed the stream. + /// That is the point, but it also means the lie has to be PROVEN in effect here, or the comparison + /// below would pass against a store that hid nothing. + { + DB::Cas::tests::OperationForTest op(*lying_backend); + std::set listed; + (*op).forEachListedKey(layout.namespaceStreamPrefix(fixture::fixtureLife(ns)), + [&](const ListedKey & key) { listed.insert(key.key); return true; }, + Retry::standard()); + for (const String & hidden : hiddenMiddleOf(layout, ns)) + ASSERT_FALSE(listed.contains(hidden)) << "the store was told to hide " << hidden << " and did not"; + } ASSERT_GT(lying_backend->holesServed(), 0u) << "the omission was never actually served -- the test would pass vacuously"; EXPECT_EQ(truth.size(), 5u) << "the oracle itself must see all five published refs"; @@ -311,7 +331,7 @@ TEST(CASListLiarEndToEnd, AHiddenPlusOneKeepsItsBlobWhenAVisibleMinusOneLandsLat runRegularRoundReclaiming(gc); store->renewWatermarkOnce(); } - EXPECT_TRUE(backend->head(blobKeyOf(layout, shared)).exists) + EXPECT_TRUE(headExists(*backend, blobKeyOf(layout, shared))) << "a blob a live ref still names was DELETED -- the hidden `+1` was never folded"; EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, shared)), 0u) << "not merely still present: the delete was never even attempted"; @@ -362,13 +382,13 @@ TEST(CASListLiarEndToEnd, AHiddenMinusOneIsStillFoldedSoTheBlobIsActuallyReclaim EXPECT_EQ(sealedCursorOf(*backend, layout, ns), (RefTxnId{1, 3})); EXPECT_EQ(inDegreeOf(*backend, layout, released), 0) << "the hidden `-1` must be folded: nothing owns this blob any more"; - EXPECT_TRUE(backend->head(blobKeyOf(layout, released)).exists) + EXPECT_TRUE(headExists(*backend, blobKeyOf(layout, released))) << "round pacing: the round that CONDEMNS never also deletes"; store->renewWatermarkOnce(); EXPECT_TRUE(runRoundsUntilAbsent(store, gc, *backend, layout, released)) << "the blob was never reclaimed -- the hidden `-1` left it pinned by a phantom owner"; - EXPECT_TRUE(backend->head(blobKeyOf(layout, unrelated)).exists) + EXPECT_TRUE(headExists(*backend, blobKeyOf(layout, unrelated))) << "and the still-owned blob is untouched"; } @@ -460,7 +480,7 @@ TEST(CASListLiarEndToEnd, AHiddenNamespacesBirthIsFoundByExactKeyAndSavesTheBlob ASSERT_TRUE(evidence.saw_fold) << "no round folded, so none published a gate verdict"; ASSERT_GT(backend->holesServed(), 0u) << "the omission was never actually served -- the test would pass vacuously"; - EXPECT_TRUE(backend->head(blobKeyOf(layout, blob)).exists) + EXPECT_TRUE(headExists(*backend, blobKeyOf(layout, blob))) << "the blob a hidden namespace still owns must survive"; EXPECT_EQ(backend->deleteCount(blobKeyOf(layout, blob)), 0u) << "not merely still present: the blob must never even be offered for deletion"; @@ -528,7 +548,7 @@ TEST(CASListLiarEndToEnd, TheSameBlobDrainsOnceHiddenGenuinelyProvesItsOwnFronti ASSERT_GT(backend->holesServed(), 0u) << "the omission was never actually served -- the test would pass vacuously"; - EXPECT_FALSE(backend->head(blobKeyOf(layout, blob)).exists) + EXPECT_FALSE(headExists(*backend, blobKeyOf(layout, blob))) << "both namespaces genuinely proved their frontier and the blob is genuinely unreferenced -- " "the round must still be able to reclaim it"; } diff --git a/src/Disks/tests/gtest_cas_manifest_reader.cpp b/src/Disks/tests/gtest_cas_manifest_reader.cpp new file mode 100644 index 000000000000..03a75c560d7c --- /dev/null +++ b/src/Disks/tests/gtest_cas_manifest_reader.cpp @@ -0,0 +1,49 @@ +#include + +#include +#include "cas_test_helpers.h" + +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int FILE_DOESNT_EXIST; +} +} + +using namespace DB::Cas; + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +CasRequests makeRequests(BackendPtr backend, FakeClock & clock, Fence fence = Fence::open()) +{ + return CasRequests(std::move(backend), std::move(fence), clock.nowFn(), clock.sleepFn()); +} + +} + +TEST(CASManifestReader, MissingManifestThrowsFileDoesntExist) +{ + FakeClock clock; + auto backend = std::make_shared(); + Layout layout("pool"); + PoolMeta meta; + CasEventSink sink; + auto requests = makeRequests(backend, clock); + CasManifestReader reader(requests, layout, meta, sink, /*manifest_decode_cache_bytes=*/0); + + const ManifestId id{RootNamespace("t"), ManifestRef{1, 1, 1}}; + + /// A live ref naming a missing manifest body is INV-NO-DANGLE: never a substituted empty + /// manifest, always the fail-closed exception -- and this must hold over the migrated + /// `CasOperation`-based read exactly as it did over the raw backend call. + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { (void)reader.readManifest(id); }); + EXPECT_EQ(backend->getCount(layout.manifestKey(id)), 1u); +} diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index 55af475d2048..a9472994237f 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -1,4 +1,5 @@ #include +#include #include "cas_test_helpers.h" #include #include @@ -8,14 +9,14 @@ #include #include +#include #include -#include -#include #include #include #include #include #include +#include namespace DB::ErrorCodes { @@ -29,8 +30,6 @@ namespace ProfileEvents { extern const Event CASMountLeaseLost; extern const Event CASMountExclusivityViolation; - extern const Event CASMountRenewalAttempts; - extern const Event CASMountRenewalRetries; } using namespace DB::Cas; @@ -52,189 +51,211 @@ RefCatalog catalogOwning(const String & ns, NsState state) return RefCatalog{.entries = {std::move(entry)}}; } -void renewKeeperOrThrow(MountLeaseKeeper & keeper) +void renewOrThrow(MountLeaseRenewer & renewer) { - const MountRenewResult result = keeper.renew( - CasRequestBudget{.attempt_timeout_ms = 1, .operation_deadline_ms = 100, .max_attempts = 2, - .lease_safety_margin_ms = 0, .retry_initial_backoff_ms = 0, .retry_max_backoff_ms = 0}, - MountRenewOperationEnvironment{}); + const MountRenewResult result = renewer.renew(MountRenewOperationEnvironment{}); if (result.outcome == MountRenewOutcome::Terminal) std::rethrow_exception(result.failure); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); } -class OwnerConflictRevealsManifestBackend : public InMemoryBackend +/// The two request planes a renewer in this file runs on, plus one operation for the protocol calls +/// driven directly. Both planes are open-fence: these fixtures hold no mount lease, so nothing here +/// should be refused by a fence it does not have. The clock and the sleep are ALWAYS injected -- a +/// fixture that drives a lease deadline passes its own so a slow machine cannot run the bound out +/// mid-test, and one that does not still must not sleep for real when a fault sends the engine round +/// again. `tests::OperationForTest` covers the one-operation case but neither the two planes nor the +/// clock, which is why this stays local. +class Ops { public: - using InMemoryBackend::putIfAbsent; + explicit Ops(std::shared_ptr backend) : Ops(std::move(backend), nullptr) {} - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + Ops(std::shared_ptr backend, uint64_t * boot_ms) + : mount(openRequestsForTest(backend)) + , farewell(openRequestsForTest(std::move(backend))) + , op(mount.admit()) { - if (!fired && key == "p/gc/server-roots/root/x/owner") + uint64_t * clock = boot_ms ? boot_ms : &own_clock; + for (CasRequests * requests : {&mount, &farewell}) { - fired = true; - InMemoryBackend::putIfAbsent("p/cas/manifests/root/x/table/debris", "x"); - return {PutOutcome::PreconditionFailed, {}}; + requests->setNowFnForTest([clock] { return *clock; }); + requests->setSleepFnForTest([clock](uint64_t ms) { *clock += ms; }); } - return InMemoryBackend::putIfAbsent(key, bytes, meta); } - bool fired = false; + Ops(const Ops &) = delete; + Ops & operator=(const Ops &) = delete; + + CasRequests mount; + CasRequests farewell; + CasOperation op; + +private: + uint64_t own_clock = 0; }; -class EpochConflictRevealsManifestBackend : public InMemoryBackend +/// The incarnation currently at `key`, for a fixture that has to name it as a precondition. +Etag currentEtag(CasOperation & op, const String & key) { -public: - using InMemoryBackend::casPut; + const auto got = op.read(key, Retry::standard()); + if (!got) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "test fixture read of '{}' found nothing", key); + return got->etag; +} - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override +/// A fixture write that must land, so a mis-seeded fixture fails where it is written rather than in +/// the assertion it silently invalidated. +void mustCommit(WriteResult && result, const String & what) +{ + if (!std::holds_alternative(result)) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "test fixture write '{}' did not commit", what); +} + +class OwnerConflictRevealsManifestBackend : public InMemoryBackend +{ +public: + std::expected write( + const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - if (!fired && key == "p/gc/server-roots/root/x/epoch") + if (!fired && !expected_value && key == "p/gc/server-roots/root/x/owner") { fired = true; - /// Install the competing allocator's winning epoch before revealing owned work. The - /// retry must not accept that now-present epoch without rechecking the entire emptiness - /// bundle that authorized the original absent-epoch attempt. - const CasResult winner = InMemoryBackend::casPut( - key, encodeServerEpoch(ServerEpoch{.next_writer_epoch = 2}), expected, meta); - winner_installed = winner.outcome == CasOutcome::Committed; - InMemoryBackend::putIfAbsent("p/cas/manifests/root/x/table/debris", "x"); - return {CasOutcome::Conflict, {}}; + InMemoryBackend::write("p/cas/manifests/root/x/table/debris", "x", std::nullopt, access); + return std::unexpected(RawConflict{}); } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool fired = false; - bool winner_installed = false; }; -class RenewalLogBackend final : public InMemoryBackend +/// Loses the owner key to a racing claimer between the read and the create: installs `winner`'s owner +/// object, then refuses this write. The subtree stays empty, so the emptiness recompute passes and the +/// claim has to decide the race from what its own write observed. +class OwnerRaceBackend : public InMemoryBackend { public: - using InMemoryBackend::putOverwrite; + explicit OwnerRaceBackend(UInt128 winner_) : winner(winner_) {} - bool throw_before_next_overwrite = false; - - void armBlockedRetry() + /// Counts only reads of the owner key, so an extra re-read the conflict decision no longer needs + /// is visible even though the resolve read on the same key already counts once. + std::optional read(const String & key, TransportAccess & access) override { - std::lock_guard lock(mutex); - blocked_retry_armed = true; - renewal_puts = 0; - second_put_arrived = false; - release_second_put = false; + if (key == "p/gc/server-roots/r/owner") + ++owner_reads; + return InMemoryBackend::read(key, access); } - bool waitForSecondPut() + std::expected write( + const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - std::unique_lock lock(mutex); - return cv.wait_for(lock, std::chrono::seconds(2), [&] { return second_put_arrived; }); + if (!fired && !expected_value && key == "p/gc/server-roots/r/owner") + { + fired = true; + InMemoryBackend::write( + key, encodeOwner(OwnerObject{.server_uuid = winner, .retired_at_ms = std::nullopt}), + std::nullopt, access); + return std::unexpected(RawConflict{}); + } + return InMemoryBackend::write(key, bytes, expected_value, access); } - void releaseSecondPut() - { - std::lock_guard lock(mutex); - release_second_put = true; - cv.notify_all(); - } + bool fired = false; + size_t owner_reads = 0; - PutResult putOverwrite( - const String & key, - const String & bytes, - const Token & expected, - const ObjectMeta & meta) override +private: + UInt128 winner; +}; + +/// Refuses the FIRST write of the epoch key after installing a competing allocator's own epoch, so the +/// absent-epoch decision has to be made a second time. `reveal_owned_work` decides whether owned work +/// becomes visible at that same instant -- the fact the second decision must re-establish. +class EpochConflictBackend : public InMemoryBackend +{ +public: + explicit EpochConflictBackend(bool reveal_owned_work_ = true) : reveal_owned_work(reveal_owned_work_) {} + + std::expected write( + const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { + if (!fired && key == "p/gc/server-roots/root/x/epoch") { - std::unique_lock lock(mutex); - if (blocked_retry_armed) - { - ++renewal_puts; - if (renewal_puts == 1) - throw Poco::TimeoutException("injected renewal timeout before blocked retry"); - if (renewal_puts == 2) - { - second_put_arrived = true; - cv.notify_all(); - if (!cv.wait_for(lock, std::chrono::seconds(20), [&] { return release_second_put; })) - throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "blocked renewal retry was not released"); - blocked_retry_armed = false; - } - } + fired = true; + const auto winner = InMemoryBackend::write( + key, encodeServerEpoch(ServerEpoch{.next_writer_epoch = 2}), expected_value, access); + winner_installed = winner.has_value(); + if (reveal_owned_work) + InMemoryBackend::write("p/cas/manifests/root/x/table/debris", "x", std::nullopt, access); + return std::unexpected(RawConflict{}); } - if (std::exchange(throw_before_next_overwrite, false)) - throw Poco::TimeoutException("injected renewal timeout before commit"); - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } + bool fired = false; + bool winner_installed = false; + private: - std::mutex mutex; - std::condition_variable cv; - bool blocked_retry_armed = false; - uint32_t renewal_puts = 0; - bool second_put_arrived = false; - bool release_second_put = false; + bool reveal_owned_work; }; -class BlockingRenewalDebugChannel final : public Poco::Channel +/// Counts every request that reaches the store, per primitive, so a test can pin how many a protocol +/// step costs rather than only what it produced. +class RequestCountingBackend final : public InMemoryBackend { public: - void log(const Poco::Message & message) override + size_t reads = 0; + size_t heads = 0; + size_t writes = 0; + + std::optional read(const String & key, TransportAccess & access) override { - if (message.getText().find("physical retry attempt") == String::npos) - return; - std::unique_lock lock(mutex); - cv.wait_for(lock, std::chrono::seconds(20), [&] { return released; }); + ++reads; + return InMemoryBackend::read(key, access); } - void unblock() + std::optional head(const String & key, TransportAccess & access) override { - std::lock_guard lock(mutex); - released = true; - cv.notify_all(); + ++heads; + return InMemoryBackend::head(key, access); } -private: - std::mutex mutex; - std::condition_variable cv; - bool released = false; + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override + { + ++writes; + return InMemoryBackend::write(key, bytes, expected_value, access); + } }; -class ScopedBlockingRenewalDebugLog +class RenewalLogBackend final : public InMemoryBackend { public: - ScopedBlockingRenewalDebugLog() - : logger(getLogger("CasMountLeaseKeeper")) - , channel(new BlockingRenewalDebugChannel) - , old_channel(logger->getChannel(), /*shared=*/true) - , old_level(logger->getLevel()) - { - logger->setChannel(channel.get()); - logger->setLevel("debug"); - } + bool throw_before_next_overwrite = false; - ~ScopedBlockingRenewalDebugLog() + /// The fault lives on the primitive every write reaches the store through, keyed to the mount + /// slot so the pool's other conditional writes pass untouched. + std::expected write( + const String & key, + const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - channel->unblock(); - logger->setChannel(old_channel); - logger->setLevel(old_level); + if (expected_value && key.ends_with("/mount") && std::exchange(throw_before_next_overwrite, false)) + throw Poco::TimeoutException("injected renewal timeout before commit"); + return InMemoryBackend::write(key, bytes, expected_value, access); } - - void release() { channel->unblock(); } - -private: - LoggerPtr logger; - Poco::AutoPtr channel; - /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. - Poco::AutoPtr old_channel; - int old_level; }; class ScopedRenewalLogCapture { public: explicit ScopedRenewalLogCapture(const String & level) - : logger(getLogger("CasMountLeaseKeeper")) + : logger(getLogger("CasMountLeaseRenewer")) , channel(new Poco::StreamChannel(stream)) , old_channel(logger->getChannel(), /*shared=*/true) , old_level(logger->getLevel()) @@ -268,36 +289,51 @@ size_t countRenewalLogText(const String & haystack, std::string_view needle) return count; } -CasRequestBudget renewalLogBudget(uint32_t max_attempts = 2) +CasRequestBudget renewalLogBudget() { return CasRequestBudget{ .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = max_attempts, .lease_safety_margin_ms = 20, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, + .connect_timeout_cap_ms = std::nullopt, }; } } +/// `CASMountAudit.PhysicalRetryCannotBeDelayedByDebugLogging` was retired when mount renewal moved onto +/// `CasRequests`/`CasOperation` (the old hand-written renewal controller had a per-attempt progress +/// callback the test used to interleave a blocking debug log with the retry loop's own pacing; nothing +/// still exposes such a callback). Verified still true against the current engine, not just the +/// migration's own commit message: `grep -n "LOG_\|getLogger" .../Backend/CasRequests.cpp` finds exactly +/// one log call in the whole write-retry engine, `logCasWriteRetryLater`, reached only from the +/// `[[noreturn]]` `throwCasWriteRetryLater` -- the terminal give-up, called once, never between +/// attempts. No replacement test is needed: there is no per-attempt log call left to race. TEST(CASMountAudit, RenewalDefaultLogsAreBounded) { - const auto open_store = [](const std::shared_ptr & backend, uint64_t & boot_ms, const String & prefix) + /// `boot_ms` is a shared, heap-owned atomic, not a plain reference parameter: the last block below + /// mutates it after the Pool exists, and the Pool can outlive this lambda's own call (a background + /// publish holds `shared_from_this()`), so a by-reference capture of a caller-local would dangle. + const auto open_store = [](const std::shared_ptr & backend, + const std::shared_ptr> & boot_ms, const String & prefix) { + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the + /// budget field alone; pair the two so the fence math below matches what admits. + backend->setAttemptTimeoutMs(renewalLogBudget().attempt_timeout_ms); return Pool::open(backend, PoolConfig{ .pool_prefix = prefix, .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .cas_request_budget = renewalLogBudget(), - .boot_ms_fn = [&] { return boot_ms; }, + .boot_ms_fn = [boot_ms] + { + return boot_ms->load(); + }, }); }; { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + auto boot_ms = std::make_shared>(100); auto store = open_store(backend, boot_ms, "renewal-log-silent"); ScopedRenewalLogCapture capture("information"); EXPECT_NO_THROW(store->renewWatermarkOnce()); @@ -306,21 +342,20 @@ TEST(CASMountAudit, RenewalDefaultLogsAreBounded) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + auto boot_ms = std::make_shared>(100); auto store = open_store(backend, boot_ms, "renewal-log-recovered"); ScopedRenewalLogCapture capture("information"); backend->throw_before_next_overwrite = true; EXPECT_NO_THROW(store->renewWatermarkOnce()); const String output = capture.captured(); - EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 2u) << output; - EXPECT_EQ(countRenewalLogText(output, "entered retry"), 1u) << output; + EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 1u) << output; EXPECT_EQ(countRenewalLogText(output, "recovered"), 1u) << output; EXPECT_EQ(countRenewalLogText(output, "physical retry attempt"), 0u) << output; } { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + auto boot_ms = std::make_shared>(100); auto store = open_store(backend, boot_ms, "renewal-log-debug"); ScopedRenewalLogCapture capture("debug"); backend->throw_before_next_overwrite = true; @@ -330,54 +365,20 @@ TEST(CASMountAudit, RenewalDefaultLogsAreBounded) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + auto boot_ms = std::make_shared>(100); auto store = open_store(backend, boot_ms, "renewal-log-fenced"); ScopedRenewalLogCapture capture("information"); - boot_ms = 1071; + /// The lease was claimed at boot 100 with the 1000 ms TTL above, so it expires at 1100. The + /// fence admits only while the remaining time strictly clears the safety margin plus whatever + /// the attempt reserves, so exactly `margin` remaining (with the reservation on top) refuses. + boot_ms->store(1100 - renewalLogBudget().lease_safety_margin_ms); EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); const String output = capture.captured(); EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 1u) << output; EXPECT_EQ(countRenewalLogText(output, "fenced"), 1u) << output; - EXPECT_EQ(countRenewalLogText(output, "entered retry"), 0u) << output; } } -TEST(CASMountAudit, PhysicalRetryCannotBeDelayedByDebugLogging) -{ - auto backend = std::make_shared(); - uint64_t boot_ms = 100; - auto store = Pool::open(backend, PoolConfig{ - .pool_prefix = "renewal-debug-order", - .server_root_id = "test", - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .cas_request_budget = renewalLogBudget(), - .boot_ms_fn = [&] { return boot_ms; }, - }); - - const uint64_t attempts_before - = ProfileEvents::global_counters[ProfileEvents::CASMountRenewalAttempts].load(); - const uint64_t retries_before - = ProfileEvents::global_counters[ProfileEvents::CASMountRenewalRetries].load(); - backend->armBlockedRetry(); - ScopedBlockingRenewalDebugLog blocked_log; - auto renewal = std::async(std::launch::async, [&] { store->renewWatermarkOnce(); }); - - const bool retry_reached_backend = backend->waitForSecondPut(); - EXPECT_TRUE(retry_reached_backend) - << "diagnostic logging after retry admission must not delay the backend request"; - EXPECT_EQ( - ProfileEvents::global_counters[ProfileEvents::CASMountRenewalAttempts].load(), - attempts_before + 2) - << "physical attempt visibility must precede completion of the in-flight retry"; - EXPECT_EQ( - ProfileEvents::global_counters[ProfileEvents::CASMountRenewalRetries].load(), - retries_before + 1); - - blocked_log.release(); - backend->releaseSecondPut(); - EXPECT_NO_THROW(renewal.get()); -} - TEST(CASServerRootId, ValidationAcceptsCleanPathsRejectsBad) { EXPECT_NO_THROW(validateServerRootId("replica-a")); @@ -447,11 +448,12 @@ TEST(CASServerRootClaim, OwnerStickyAndForeignFailsClosed) { auto b = std::make_shared(); Layout l("p"); - EXPECT_NO_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation())); // fresh empty root → claim - EXPECT_NO_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation())); // same uuid → ok + Ops ops(b); + EXPECT_NO_THROW(claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation())); // fresh empty root → claim + EXPECT_NO_THROW(claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation())); // same uuid → ok try { - claimOwnerOrThrow(*b, l, "r", UInt128(2), emptyCatalogObservation()); + claimOwnerOrThrow(ops.op, l, "r", UInt128(2), emptyCatalogObservation()); FAIL() << "expected a foreign owner to fail closed"; } catch (const DB::Exception & e) @@ -465,14 +467,15 @@ TEST(CASServerRootClaim, TombstonedSameOwnerFailsClosed) { auto b = std::make_shared(); Layout l("p"); - b->putIfAbsent(l.ownerKey("r"), encodeOwner(OwnerObject{ + Ops ops(b); + mustCommit(ops.op.create(l.ownerKey("r"), encodeOwner(OwnerObject{ .server_uuid = UInt128(1), .retired_at_ms = 1752537600000ULL, - })); + }), Retry::standard()), "tombstoned owner"); try { - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); FAIL() << "expected a tombstoned owner claim to fail closed"; } catch (const DB::Exception & e) @@ -487,17 +490,18 @@ TEST(CASServerRootEpoch, AllocatorIsMonotoneAndSurvivesMountConcept) { auto b = std::make_shared(); Layout l("r"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - const uint64_t e1 = allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); - const uint64_t e2 = allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + const uint64_t e1 = allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); + const uint64_t e2 = allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()); EXPECT_GE(e1, 1u); // 0 is a reserved sentinel EXPECT_GT(e2, e1); // strictly increasing - /// Deleting the (separate) mount object must NOT reset the epoch. No mount has been written in - /// Task 4, so deleteExact of a non-existent mount is a NotFound no-op that touches nothing. - const auto del = b->deleteExact(l.mountKey("r"), b->head(l.mountKey("r")).token); - EXPECT_EQ(del.kind, DeleteOutcome::Kind::NotFound); - EXPECT_GT(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), e2); + /// Deleting the (separate) mount object must NOT reset the epoch. No mount has been written yet, + /// so the removal is a no-op that touches nothing. + ASSERT_FALSE(ops.op.head(l.mountKey("r"), Retry::standard()).has_value()); + EXPECT_EQ(ops.op.removeCurrent(l.mountKey("r"), Retry::standard()), Removal::Gone); + EXPECT_GT(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), e2); } /// Phase C (spec rev.4): an ABSENT epoch object over a PRESENT mount object means durable epoch @@ -507,20 +511,22 @@ TEST(CASMount, EpochRemintOverExistingMountRefuses) { auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, MountClaimResult::Claimed); /// The epoch object is ABSENT (never created in this sequence) while the mount exists: - EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); /// CORRUPTED_DATA + EXPECT_THROW(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); /// CORRUPTED_DATA } TEST(CASMount, EpochRemintAuthoritativeAbsenceMints) { auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// fresh root: both control objects absent - EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present now: normal CAS bump, no probe + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_EQ(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// fresh root: both control objects absent + EXPECT_EQ(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present now: normal conditional bump, no probe } /// The probe outcome gates the mint: anything short of authoritative KeyAbsent fails closed. @@ -529,15 +535,16 @@ TEST(CASMount, EpochRemintIndeterminateProbeFailsClosed) class IndeterminateProbeBackend final : public InMemoryBackend { public: - SentinelProbeResult probeSentinelRaw(const String &) override + SentinelProbeResult probeSentinelRaw(const String &, TransportAccess &) override { return {.outcome = ProbeOutcome::Indeterminate, .body = std::nullopt}; } }; auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_THROW(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); } /// Decommission over a TERMINAL (expired/fenced) mount with a lost epoch object proceeds and mints @@ -546,11 +553,12 @@ TEST(CASMount, DecommissionRemintOverTerminalMountMintsDistinctEpoch) { auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/3, /*now_ms=*/1000, /*ttl_ms=*/100).kind, + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/3, /*now_ms=*/1000, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); /// now_ms=5000: the ttl_ms=100 lease above is long expired -> terminal. - EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/5000, emptyCatalogObservation()), 4u); + EXPECT_EQ(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/5000, emptyCatalogObservation()), 4u); } /// Decommission over a LIVE mount with a lost epoch refuses — the blind bypass would recreate the @@ -559,10 +567,11 @@ TEST(CASMount, DecommissionRemintOverLiveMountRefuses) { auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/1, /*now_ms=*/1000, /*ttl_ms=*/30000).kind, MountClaimResult::Claimed); - EXPECT_THROW(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/2000, emptyCatalogObservation()), + EXPECT_THROW(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::DecommissionRecovery, /*now_ms=*/2000, emptyCatalogObservation()), DB::Exception); /// ABORTED: live member } @@ -574,18 +583,19 @@ TEST(CASMount, EpochBumpWithPresentEpochIssuesNoProbe) { public: int probes = 0; - SentinelProbeResult probeSentinelRaw(const String & k) override + SentinelProbeResult probeSentinelRaw(const String & k, TransportAccess & access) override { ++probes; - return InMemoryBackend::probeSentinelRaw(k); + return InMemoryBackend::probeSentinelRaw(k, access); } }; auto b = std::make_shared(); Layout l("p"); - claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()); - EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// bootstrap: ONE probe (absent-epoch branch) + Ops ops(b); + claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); + EXPECT_EQ(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); /// bootstrap: ONE probe (absent-epoch branch) const int probes_after_bootstrap = b->probes; - EXPECT_EQ(allocateWriterEpoch(*b, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present: normal CAS bump... + EXPECT_EQ(allocateWriterEpoch(ops.op, l, "r", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u); /// epoch present: normal conditional bump... EXPECT_EQ(b->probes, probes_after_bootstrap) << "...must not probe the mount key"; } @@ -593,9 +603,10 @@ TEST(CASServerRootClaim, MissingOwnerOverNonEmptyRootIsCorrupted) { auto b = std::make_shared(); Layout l("p"); + Ops ops(b); /// Simulate existing data without an owner (identity lost): plant a key under roots//. - b->putIfAbsent(l.serverRootDataPrefix("r") + "some-data", "x"); - EXPECT_THROW(claimOwnerOrThrow(*b, l, "r", UInt128(1), emptyCatalogObservation()), DB::Exception); + mustCommit(ops.op.create(l.serverRootDataPrefix("r") + "some-data", "x", Retry::standard()), "root debris"); + EXPECT_THROW(claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()), DB::Exception); } TEST(CASServerRootSafety, EveryCatalogLifecycleStateBlocksOwnerAndEpochRecreation) @@ -606,39 +617,39 @@ TEST(CASServerRootSafety, EveryCatalogLifecycleStateBlocksOwnerAndEpochRecreatio RefCatalog catalog = catalogOwning("root/x/table", state); const ObserveRefCatalog observe = [catalog] { return catalog; }; - InMemoryBackend owner_backend; - EXPECT_THROW(claimOwnerOrThrow(owner_backend, layout, "root/x", UInt128{1}, observe), DB::Exception); - EXPECT_FALSE(owner_backend.head(layout.ownerKey("root/x")).exists); + Ops owner_ops(std::make_shared()); + EXPECT_THROW(claimOwnerOrThrow(owner_ops.op, layout, "root/x", UInt128{1}, observe), DB::Exception); + EXPECT_FALSE(owner_ops.op.head(layout.ownerKey("root/x"), Retry::standard()).has_value()); - InMemoryBackend epoch_backend; + Ops epoch_ops(std::make_shared()); EXPECT_THROW(allocateWriterEpoch( - epoch_backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, observe), DB::Exception); - EXPECT_FALSE(epoch_backend.head(layout.epochKey("root/x")).exists); + epoch_ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, observe), DB::Exception); + EXPECT_FALSE(epoch_ops.op.head(layout.epochKey("root/x"), Retry::standard()).has_value()); } } TEST(CASServerRootSafety, OwnershipUsesAPathComponentBoundary) { - InMemoryBackend backend; + Ops ops(std::make_shared()); const Layout layout("p"); EXPECT_TRUE(serverRootSubtreeEmpty( - backend, layout, "root/x", catalogOwning("root/xy/table", NsState::Live))); + ops.op, layout, "root/x", catalogOwning("root/xy/table", NsState::Live))); EXPECT_FALSE(serverRootSubtreeEmpty( - backend, layout, "root/x", catalogOwning("root/x/table", NsState::Live))); + ops.op, layout, "root/x", catalogOwning("root/x/table", NsState::Live))); } TEST(CASServerRootSafety, OpaqueStreamAndStateDebrisAloneDoesNotBlockRecreation) { - InMemoryBackend backend; + Ops ops(std::make_shared()); const Layout layout("p"); const NamespaceLifeId dead = NamespaceLifeId::fromCatalogEntry(RootNamespace{"unowned"}, UInt128{99}); - ASSERT_EQ(backend.putIfAbsent(layout.refLogKey(dead, RefTxnId{1, 1}), "debris").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(dead), "debris").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(layout.namespaceFileKey(dead, "f"), "debris").outcome, PutOutcome::Done); + mustCommit(ops.op.create(layout.refLogKey(dead, RefTxnId{1, 1}), "debris", Retry::standard()), "ref-log debris"); + mustCommit(ops.op.create(layout.refCkptKey(dead), "debris", Retry::standard()), "ckpt debris"); + mustCommit(ops.op.create(layout.namespaceFileKey(dead, "f"), "debris", Retry::standard()), "ns-file debris"); - EXPECT_NO_THROW(claimOwnerOrThrow(backend, layout, "root/x", UInt128{1}, emptyCatalogObservation())); + EXPECT_NO_THROW(claimOwnerOrThrow(ops.op, layout, "root/x", UInt128{1}, emptyCatalogObservation())); EXPECT_EQ(allocateWriterEpoch( - backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 1u); } TEST(CASServerRootSafety, ManifestAndLooseRootDebrisStillBlockRecreation) @@ -648,49 +659,56 @@ TEST(CASServerRootSafety, ManifestAndLooseRootDebrisStillBlockRecreation) layout.casManifestsServerPrefix("root/x") + "table/debris", layout.serverRootDataPrefix("root/x") + "loose"}) { - InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent(key, "x").outcome, PutOutcome::Done); + Ops ops(std::make_shared()); + mustCommit(ops.op.create(key, "x", Retry::standard()), "blocking debris"); EXPECT_THROW(claimOwnerOrThrow( - backend, layout, "root/x", UInt128{1}, emptyCatalogObservation()), DB::Exception); + ops.op, layout, "root/x", UInt128{1}, emptyCatalogObservation()), DB::Exception); EXPECT_THROW(allocateWriterEpoch( - backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); } } TEST(CASServerRootSafety, UnreadableCatalogNeverFallsBackToPhysicalGuesses) { - InMemoryBackend backend; + Ops ops(std::make_shared()); const Layout layout("p"); const ObserveRefCatalog unreadable = []() -> RefCatalog { throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected unreadable catalog"); }; - EXPECT_THROW(claimOwnerOrThrow(backend, layout, "root/x", UInt128{1}, unreadable), DB::Exception); + EXPECT_THROW(claimOwnerOrThrow(ops.op, layout, "root/x", UInt128{1}, unreadable), DB::Exception); EXPECT_THROW(allocateWriterEpoch( - backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, unreadable), DB::Exception); - EXPECT_FALSE(backend.head(layout.ownerKey("root/x")).exists); - EXPECT_FALSE(backend.head(layout.epochKey("root/x")).exists); + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, unreadable), DB::Exception); + EXPECT_FALSE(ops.op.head(layout.ownerKey("root/x"), Retry::standard()).has_value()); + EXPECT_FALSE(ops.op.head(layout.epochKey("root/x"), Retry::standard()).has_value()); } TEST(CASServerRootSafety, OwnerConflictRecomputesTheWholeEmptinessBundle) { - OwnerConflictRevealsManifestBackend backend; + auto backend = std::make_shared(); + Ops ops(backend); const Layout layout("p"); - EXPECT_THROW(claimOwnerOrThrow( - backend, layout, "root/x", UInt128{1}, emptyCatalogObservation()), DB::Exception); - EXPECT_TRUE(backend.fired); - EXPECT_FALSE(backend.head(layout.ownerKey("root/x")).exists); + /// The message, not just the code: without the post-conflict recompute the claim still throws + /// `CORRUPTED_DATA`, from the vanished-anchor arm below it, so a bare code assertion would hold + /// with the behaviour this test is named for deleted. + DB::Cas::tests::expectThrowsCodeWithMessage( + DB::ErrorCodes::CORRUPTED_DATA, + "newly visible owned work blocks recreation", + [&] { claimOwnerOrThrow(ops.op, layout, "root/x", UInt128{1}, emptyCatalogObservation()); }); + EXPECT_TRUE(backend->fired); + EXPECT_FALSE(ops.op.head(layout.ownerKey("root/x"), Retry::standard()).has_value()); } TEST(CASServerRootSafety, EpochConflictRecomputesTheWholeEmptinessBundle) { - EpochConflictRevealsManifestBackend backend; + auto backend = std::make_shared(); + Ops ops(backend); const Layout layout("p"); EXPECT_THROW(allocateWriterEpoch( - backend, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); - EXPECT_TRUE(backend.fired); - ASSERT_TRUE(backend.winner_installed); - const auto epoch = backend.get(layout.epochKey("root/x")); + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + EXPECT_TRUE(backend->fired); + ASSERT_TRUE(backend->winner_installed); + const auto epoch = ops.op.read(layout.epochKey("root/x"), Retry::standard()); ASSERT_TRUE(epoch.has_value()); EXPECT_EQ(decodeServerEpoch(epoch->bytes).next_writer_epoch, 2u) << "the rejected allocator must not consume an epoch from the conflict winner"; @@ -701,14 +719,17 @@ TEST(CASMountLease, AbsentClaimThenRenewBumpsSeq) auto b = std::make_shared(); Layout l("p"); uint64_t now = 1000; - auto r = claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100); + uint64_t boot = 0; + Ops ops(b, &boot); + auto r = claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100); EXPECT_EQ(r.kind, MountClaimResult::Claimed); - MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, - [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); + MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), + [&] { return boot; }); k.start(); - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).seq, 1u); - renewKeeperOrThrow(k); - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).seq, 2u); + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).seq, 1u); + renewOrThrow(k); + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).seq, 2u); } TEST(CASMountLease, HolderBodiesMintFreshAttemptIdsAndFenceCopiesIt) @@ -716,53 +737,57 @@ TEST(CASMountLease, HolderBodiesMintFreshAttemptIdsAndFenceCopiesIt) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; - ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 7, now, 100).kind, MountClaimResult::Claimed); + uint64_t boot = 0; + Ops ops(backend, &boot); + ASSERT_EQ(claimMount(ops.op, layout, "r", UInt128{1}, 7, now, 100).kind, MountClaimResult::Claimed); const String key = layout.mountKey("r"); - const MountLease claimed = decodeMountLease(backend->get(key)->bytes); - - MountLeaseKeeper keeper(backend, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), [&] { return now; }, - [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); - keeper.start(); - renewKeeperOrThrow(keeper); - const MountLease renewed = decodeMountLease(backend->get(key)->bytes); + const MountLease claimed = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); + + MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), + [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), + [&] { return boot; }); + renewer.start(); + renewOrThrow(renewer); + const MountLease renewed = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); EXPECT_NE(claimed.write_attempt_id, UInt128{}); EXPECT_NE(renewed.write_attempt_id, UInt128{}); EXPECT_NE(claimed.write_attempt_id, renewed.write_attempt_id); - auto observed = backend->get(key); + auto observed = ops.op.read(key, Retry::standard()); ASSERT_TRUE(observed.has_value()); MountLease fenced = decodeMountLease(observed->bytes); fenced.gc_fenced = true; ++fenced.seq; - ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), observed->token).outcome, PutOutcome::Done); - EXPECT_EQ(decodeMountLease(backend->get(key)->bytes).write_attempt_id, renewed.write_attempt_id); + mustCommit(ops.op.replace(key, encodeMountLease(fenced), observed->etag, Retry::standard()), "fence-out"); + EXPECT_EQ(decodeMountLease(ops.op.read(key, Retry::standard())->bytes).write_attempt_id, renewed.write_attempt_id); } TEST(CASMountLease, ReclaimAndSuccessorBodiesMintNewAttemptIds) { auto backend = std::make_shared(); Layout layout("p"); + Ops ops(backend); const String key = layout.mountKey("r"); - ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 7, 1000, 100).kind, MountClaimResult::Claimed); - const MountLease first = decodeMountLease(backend->get(key)->bytes); + ASSERT_EQ(claimMount(ops.op, layout, "r", UInt128{1}, 7, 1000, 100).kind, MountClaimResult::Claimed); + const MountLease first = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); - auto observed = backend->get(key); + auto observed = ops.op.read(key, Retry::standard()); ASSERT_TRUE(observed.has_value()); MountLease fenced = decodeMountLease(observed->bytes); fenced.gc_fenced = true; ++fenced.seq; - ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(fenced), observed->token).outcome, PutOutcome::Done); - const MountLease fence = decodeMountLease(backend->get(key)->bytes); + mustCommit(ops.op.replace(key, encodeMountLease(fenced), observed->etag, Retry::standard()), "fence-out"); + const MountLease fence = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); EXPECT_EQ(fence.write_attempt_id, first.write_attempt_id); - ASSERT_EQ(claimMount(*backend, layout, "r", UInt128{1}, 8, 2000, 100).kind, MountClaimResult::Claimed); - const MountLease successor = decodeMountLease(backend->get(key)->bytes); + ASSERT_EQ(claimMount(ops.op, layout, "r", UInt128{1}, 8, 2000, 100).kind, MountClaimResult::Claimed); + const MountLease successor = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); EXPECT_NE(successor.write_attempt_id, first.write_attempt_id); EXPECT_NE(successor.write_attempt_id, UInt128{}); } /// STID 3982-3b48: `rm -rf` of the pool dir under a live mount deletes the mount slot object out from -/// under a running keeper. The next synchronous renewal must return terminal WITHOUT constructing a +/// under a running renewer. The next synchronous renewal must return terminal WITHOUT constructing a /// `LOGICAL_ERROR` -- that aborts debug/ASan builds at /// exception construction, and there is no foreign writer here to fail closed against, only an /// environmental condition. @@ -771,21 +796,24 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) auto b = std::make_shared(); Layout l("p"); uint64_t now = 1000; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, - [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0)); + uint64_t boot = 0; + Ops ops(b, &boot); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), + [&] { return boot; }); k.start(); const String mount_key = l.mountKey("r"); const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); /// NOLINT(clang-analyzer-deadcode.DeadStores) - /// Simulate `rm -rf` of the backing store: the mount slot object is gone, but the keeper still - /// holds a (now stale) token for it. - ASSERT_EQ(b->deleteExact(mount_key, b->head(mount_key).token).kind, DeleteOutcome::Kind::Deleted); + /// Simulate `rm -rf` of the backing store: the mount slot object is gone, but the renewer still + /// names a (now stale) incarnation as its precondition. + ASSERT_EQ(ops.op.removeCurrent(mount_key, Retry::standard()), Removal::Removed); try { - renewKeeperOrThrow(k); + renewOrThrow(k); FAIL() << "renew against a vanished mount object must throw"; } catch (const DB::Exception & e) @@ -794,7 +822,7 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) EXPECT_NE(e.code(), DB::ErrorCodes::LOGICAL_ERROR); } EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) - << "keeper classification is metric-free; the runtime records operational loss"; + << "renewer classification is metric-free; the runtime records operational loss"; } /// STID 3982-3b48 (part 1b): the terminal/clean-release counterpart to the renewal fix above. When @@ -812,38 +840,40 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) auto b = std::make_shared(); Layout l("p"); uint64_t now = 1000; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseKeeper k(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, - [] { return uint64_t{0}; }); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + [&] { return now; }, [] { return uint64_t{0}; }); k.start(); const String mount_key = l.mountKey("r"); const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); /// Simulate `rm -rf` of the backing store: the mount slot object is gone before we ever attempt - /// a renewal, so `terminate()`'s token-guarded farewell PUT is the first thing to observe it. - ASSERT_EQ(b->deleteExact(mount_key, b->head(mount_key).token).kind, DeleteOutcome::Kind::Deleted); + /// a renewal, so the farewell's guarded write is the first thing to observe it. + ASSERT_EQ(ops.op.removeCurrent(mount_key, Retry::standard()), Removal::Removed); EXPECT_NO_THROW(k.release()) << "clean release against a vanished store must be a no-op, not a LOGICAL_ERROR abort"; EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before); } -/// rev.6: a bare `claimMount` (no `proven_dead_token`) NEVER reclaims a same-uuid, different-epoch +/// rev.6: a bare `claimMount` (no `proven_dead_incarnation`) NEVER reclaims a same-uuid, different-epoch /// lease off a wall-clock-looking-expired stamp — only `claimMountAwaitingExpiry`'s observation loop /// can turn that into a reclaim. Renamed from `...ExpiredReclaims` to describe the corrected behavior. TEST(CASMountLease, SameUuidLiveFailsForeignFailsExpiredStillLiveDoubleStart) { auto b = std::make_shared(); Layout l("p"); - claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100); // A live until 1100 + Ops ops(b); + claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100); // A live until 1100 // same uuid, lease still live → double-start guard: - EXPECT_EQ(claimMount(*b, l, "r", UInt128(1), 8, 1050, 100).kind, MountClaimResult::LiveDoubleStart); + EXPECT_EQ(claimMount(ops.op, l, "r", UInt128(1), 8, 1050, 100).kind, MountClaimResult::LiveDoubleStart); // foreign uuid, even after expiry → fail closed: - EXPECT_EQ(claimMount(*b, l, "r", UInt128(2), 1, 1200, 100).kind, MountClaimResult::ForeignOwner); + EXPECT_EQ(claimMount(ops.op, l, "r", UInt128(2), 1, 1200, 100).kind, MountClaimResult::ForeignOwner); // same uuid, even after the stamp LOOKS expired on our wall clock → still LiveDoubleStart: no - // proven_dead_token was supplied, so there is no certificate of death to reclaim on. - EXPECT_EQ(claimMount(*b, l, "r", UInt128(1), 9, 1200, 100).kind, MountClaimResult::LiveDoubleStart); + // proven_dead_incarnation was supplied, so there is no certificate of death to reclaim on. + EXPECT_EQ(claimMount(ops.op, l, "r", UInt128(1), 9, 1200, 100).kind, MountClaimResult::LiveDoubleStart); } TEST(CASMountMessage, DoubleStartTextHasIdentityAndRemediation) @@ -870,11 +900,13 @@ TEST(CASMountMessage, DoubleStartTextHasIdentityAndRemediation) EXPECT_NE(msg.find("unique"), std::string::npos); EXPECT_NE(msg.find("reclaim the mount on restart"), std::string::npos); EXPECT_NE(msg.find("uuid file"), std::string::npos); - /// Clock-skew caveat + manual mount-object delete escape hatch. - EXPECT_NE(msg.find("CLOCK SKEW"), std::string::npos); - EXPECT_NE(msg.find("NTP"), std::string::npos); + /// Token-stability liveness statement (replaces the old wall-clock CLOCK SKEW caveat) + manual + /// mount-object delete escape hatch + the unsafe-knob escape hatch. + EXPECT_NE(msg.find("own clock"), std::string::npos); + EXPECT_NE(msg.find("diagnostic"), std::string::npos); EXPECT_NE(msg.find("manually delete the mount"), std::string::npos); EXPECT_NE(msg.find("gc/server-roots/replica-a/mount"), std::string::npos); + EXPECT_NE(msg.find("cas_unsafe_remount_no_delay"), std::string::npos); } /// rev.6: a stamped `expires_at_ms` that already looks past-due on our wall clock must NOT shortcut @@ -885,8 +917,9 @@ TEST(CASMountAwaitExpiry, PastExpiryStillPaysTheFullObservationThreshold) { auto b = std::make_shared(); Layout l("p"); + Ops ops(b); /// A prior incarnation (uuid=1, epoch=7) claimed a lease live until 1100. - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); uint64_t wall = 1200; // already past 1100 on wall clock — irrelevant to the decision uint64_t mono = 0; @@ -896,18 +929,19 @@ TEST(CASMountAwaitExpiry, PastExpiryStillPaysTheFullObservationThreshold) auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; ++sleeps; }; const auto r = claimMountAwaitingExpiry( - *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::Claimed); EXPECT_GT(sleeps, 0); // NOT instant — no wall-clock trust EXPECT_GE(mono, 100 + 100 / 20 + 25); // full observation threshold paid - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 8u); // reclaimed as us + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 8u); // reclaimed as us } TEST(CASMountAwaitExpiry, FutureExpiryReclaimsAfterClockAdvances) { auto b = std::make_shared(); Layout l("p"); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); uint64_t wall = 1000; // lease looks live until 1100, holder does NOT renew uint64_t mono = 0; @@ -916,9 +950,9 @@ TEST(CASMountAwaitExpiry, FutureExpiryReclaimsAfterClockAdvances) auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; }; const auto r = claimMountAwaitingExpiry( - *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 50, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 50, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::Claimed); - const auto body = decodeMountLease(b->get(l.mountKey("r"))->bytes); + const auto body = decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes); EXPECT_EQ(body.writer_epoch, 8u); EXPECT_EQ(body.seq, 2u); // reclaim continues seq (prev 1 + 1) } @@ -929,60 +963,64 @@ TEST(CASMountAwaitExpiry, LiveRenewingTwinTimesOutAsDoubleStart) { auto b = std::make_shared(); Layout l("p"); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); uint64_t wall = 1000; uint64_t mono = 0; auto now_fn = [&] { return wall; }; auto mono_fn = [&] { return mono; }; /// Each poll: both clocks advance AND the live holder (uuid=1, epoch=7) renews its own lease — - /// the observed write-token changes on EVERY poll, forcing a restart every time. + /// the observed incarnation changes on EVERY poll, forcing a restart every time. auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, wall, 100).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, wall, 100).kind, MountClaimResult::Claimed); }; const auto r = claimMountAwaitingExpiry( - *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart); - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 7u); // still the holder's + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 7u); // still the holder's } namespace { -/// fix-round F5 harness: makes the mount key vanish to EVERY `get()`, unconditionally, while the real -/// underlying object stays put -- forcing `claimMount`'s own internal GET to take the absent-slot race -/// branch every call (its `putIfAbsent` then fails against the real, still-present object, returning -/// `LiveDoubleStart` with no token -- fix-round F8 leaves `.token` unset on exactly this branch, since -/// no re-read was done). That in turn forces `claimMountAwaitingExpiry`'s F8 fallback re-GET, which -/// ALSO sees the slot as vanished -- deterministically reproducing "the slot vanished between -/// claimMount's own GET and ours" on EVERY loop iteration, not just a lucky one-shot race. +/// fix-round F5 harness: makes the mount key vanish to EVERY read, unconditionally, while the real +/// underlying object stays put -- forcing `claimMount`'s own read to take the absent-slot race +/// branch every call (its create then conflicts against the real, still-present object, returning +/// `LiveDoubleStart` with no incarnation -- that branch deliberately leaves `.etag` unset, +/// since no re-read was done). That in turn forces `claimMountAwaitingExpiry`'s fallback re-read, +/// which ALSO sees the slot as vanished -- deterministically reproducing "the slot vanished between +/// claimMount's own read and ours" on EVERY loop iteration, not just a lucky one-shot race. class AlwaysVanishesBackend final : public DB::Cas::Backend { public: explicit AlwaysVanishesBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} String watched_key; - std::optional get(const String & k, DB::Cas::Range r) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The fault is on the read primitive, which is the only way anything now reaches the store. + std::optional read(const String & key, TransportAccess & access) override { - if (k == watched_key) + if (key == watched_key) return std::nullopt; - return inner->get(k, r); + return inner->read(key, access); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: std::shared_ptr inner; @@ -998,12 +1036,14 @@ TEST(CASMountAwaitExpiry, PersistentSlotVanishPacesAndBoundsRestartsInsteadOfSpi { auto inner = std::make_shared(); Layout l("p"); - /// A real slot exists underneath (uuid 1, epoch 7) so `claimMount`'s absent-slot `putIfAbsent` - /// genuinely fails every time (never accidentally re-mints). - ASSERT_EQ(claimMount(*inner, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + Ops inner_ops(inner); + /// A real slot exists underneath (uuid 1, epoch 7) so `claimMount`'s absent-slot create + /// genuinely conflicts every time (never accidentally re-mints). + ASSERT_EQ(claimMount(inner_ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); auto vanishing = std::make_shared(inner); vanishing->watched_key = l.mountKey("r"); + Ops ops(vanishing); uint64_t wall = 1000; uint64_t mono = 0; @@ -1013,20 +1053,21 @@ TEST(CASMountAwaitExpiry, PersistentSlotVanishPacesAndBoundsRestartsInsteadOfSpi auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; ++sleeps; }; const auto r = claimMountAwaitingExpiry( - *vanishing, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart) << "must terminate (bounded), not loop forever"; EXPECT_GT(sleeps, 0) << "a persistently vanishing slot must still pace via sleep_fn, not busy-spin"; - /// The real epoch-7 lease is untouched -- every `putIfAbsent` attempt against it genuinely fails + /// The real epoch-7 lease is untouched -- every create attempt against it genuinely conflicts /// (the object is still there), so it is never accidentally re-minted over. - EXPECT_EQ(decodeMountLease(inner->get(l.mountKey("r"))->bytes).writer_epoch, 7u); + EXPECT_EQ(decodeMountLease(inner_ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 7u); } TEST(CASMountAwaitExpiry, ForeignUuidFailsClosedImmediately) { auto b = std::make_shared(); Layout l("p"); + Ops ops(b); /// A foreign server (uuid=2) holds the mount. - ASSERT_EQ(claimMount(*b, l, "r", UInt128(2), 1, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(2), 1, /*now*/ 1000, /*ttl*/ 100).kind, MountClaimResult::Claimed); uint64_t now = 1000; int sleeps = 0; @@ -1035,7 +1076,7 @@ TEST(CASMountAwaitExpiry, ForeignUuidFailsClosedImmediately) auto sleep_fn = [&](uint64_t ms) { now += ms; ++sleeps; }; const auto r = claimMountAwaitingExpiry( - *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 25, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::ForeignOwner); EXPECT_EQ(sleeps, 0); // never waits across UUIDs } @@ -1049,7 +1090,8 @@ TEST(CASMountAwaitExpiry, SkewedFarFutureExpiryHasNoEffectOnObservationThreshold { auto b = std::make_shared(); Layout l("p"); - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100000).kind, MountClaimResult::Claimed); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 100000).kind, MountClaimResult::Claimed); uint64_t wall = 1000; uint64_t mono = 0; @@ -1058,23 +1100,64 @@ TEST(CASMountAwaitExpiry, SkewedFarFutureExpiryHasNoEffectOnObservationThreshold auto sleep_fn = [&](uint64_t ms) { wall += ms; mono += ms; }; const auto r = claimMountAwaitingExpiry( - *b, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); + ops.op, l, "r", UInt128(1), /*our_epoch*/ 8, now_fn, mono_fn, /*ttl*/ 100, /*poll*/ 20, sleep_fn); EXPECT_EQ(r.kind, MountClaimResult::Claimed); EXPECT_LE(mono, 100u + 100u / 20 + 20u + 20u); // bounded by OUR threshold, not the predecessor's stamp - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 8u); // reclaimed + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 8u); // reclaimed } -TEST(CASMountLease, KeeperStartAdoptsOurOwnClaimNotDoubleStart) +TEST(CASMountClaim, UnsafeAuthorizationIsTokenExact) +{ + auto b = std::make_shared(); + Layout l("p"); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 30000).kind, MountClaimResult::Claimed); + const Etag stale = ops.op.read(l.mountKey("r"), Retry::standard())->etag; + + /// `Etag` equality compares (key, value), and it is minted only through the request planes -- there + /// is no cross-key comparison to exercise here. Build the stale token by refreshing the SAME slot a + /// second time (same uuid, same epoch): `stale`, read before this refresh, is then a genuinely stale + /// token for the slot's CURRENT value, without touching an unrelated key. + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, /*now*/ 1000, /*ttl*/ 30000).kind, MountClaimResult::Claimed); + const Etag current = ops.op.read(l.mountKey("r"), Retry::standard())->etag; + + /// A stale token is refused: nothing authorizes a reclaim over a slot that moved. + const MountClaimResult refused = claimMount(ops.op, l, "r", UInt128(1), 8, 1000, 30000, /*proven_dead=*/{}, + /*sink=*/{}, /*unsafe_reclaim_authorization=*/stale); + EXPECT_EQ(refused.kind, MountClaimResult::LiveDoubleStart); + + /// A foreign uuid is refused before the authorization is consulted. + const MountClaimResult foreign = claimMount(ops.op, l, "r", UInt128(2), 8, 1000, 30000, {}, {}, current); + EXPECT_EQ(foreign.kind, MountClaimResult::ForeignOwner); + + /// The exact token reclaims, with the prior state and the audit reason naming the setting. + std::vector events; + const MountClaimResult reclaimed = claimMount(ops.op, l, "r", UInt128(1), 8, 1000, 30000, {}, + [&](CasEvent e) { events.push_back(std::move(e)); }, current); + ASSERT_EQ(reclaimed.kind, MountClaimResult::Claimed); + EXPECT_EQ(reclaimed.prior, MountPriorState::UncleanUnsafe); + ASSERT_FALSE(events.empty()); + EXPECT_THAT(events.back().reason, testing::HasSubstr("cas_unsafe_remount_no_delay")); + /// Pin the fields `system.cas_log` consumers actually key on, not just the free-form reason: the + /// unsafe reclaim shares the same event type and outcome as every other reclaim (`emitMountEvent`'s + /// "reclaim" branch argument), so nothing about this path is a separate, unaudited channel. + EXPECT_EQ(events.back().type, CasEventType::MountClaim); + EXPECT_EQ(events.back().outcome, "reclaim"); + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 8u); +} + +TEST(CASMountLease, RenewerStartAdoptsOurOwnClaimNotDoubleStart) { auto b = std::make_shared(); Layout l("p"); uint64_t now = 1000; - // The normal flow: claimMount writes the live mount under (uuid=1, epoch=7), THEN keeper.start(). - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseKeeper k(b, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), [&] { return now; }, - [] { return uint64_t{0}; }); + Ops ops(b); + // The normal flow: claimMount writes the live mount under (uuid=1, epoch=7), THEN renewer.start(). + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); + MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), + [&] { return now; }, [] { return uint64_t{0}; }); EXPECT_NO_THROW(k.start()); // adopts our own live (uuid=1,epoch=7) mount — NOT a double-start - EXPECT_EQ(decodeMountLease(b->get(l.mountKey("r"))->bytes).writer_epoch, 7u); + EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 7u); } TEST(CASMountFence, SupersededWriterRefusedNoS3Read) @@ -1115,7 +1198,7 @@ TEST(CASMountStartup, WriterEpochStrictlyIncreasesAcrossReopen) .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); const uint64_t e1 = s1->writerEpoch(); - /// Simulate shutdown: the Pool dtor stops the keeper, whose terminate() retires the lease + /// Simulate shutdown: the Pool dtor stops the renewer, whose terminate() retires the lease /// (stamps it already-expired). The owner + the durable epoch object stay sticky. s1.reset(); @@ -1136,7 +1219,8 @@ TEST(CASMountStartup, FreshWritablePoolBootstrapsAnExplicitEmptyCatalog) .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", .skip_access_check = true}); - const auto catalog = backend->get(layout.refCatalogKey()); + Ops ops(backend); + const auto catalog = ops.op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog.has_value()); EXPECT_TRUE(decodeRefCatalog(catalog->bytes).entries.empty()); } @@ -1151,19 +1235,17 @@ TEST(CASMountStartup, ExistingPoolWithoutCatalogFailsBeforeSlotMutation) .skip_access_check = true}); } + Ops ops(backend); /// Old raw fixtures did not persist an empty catalog. Make this an explicit existing-pool /// fixture before removing the mandatory object whose loss the mount must reject. - if (!backend->head(layout.refCatalogKey()).exists) - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{})).outcome, - PutOutcome::Done); - const HeadResult catalog_head = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(catalog_head.exists); - ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog_head.token).kind, - DeleteOutcome::Kind::Deleted); - - const auto owner_before = backend->get(layout.ownerKey("r")); - const auto epoch_before = backend->get(layout.epochKey("r")); - const auto mount_before = backend->get(layout.mountKey("r")); + if (!ops.op.head(layout.refCatalogKey(), Retry::standard())) + mustCommit(ops.op.create(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{}), Retry::standard()), + "empty catalog"); + ASSERT_EQ(ops.op.removeCurrent(layout.refCatalogKey(), Retry::standard()), Removal::Removed); + + const auto owner_before = ops.op.read(layout.ownerKey("r"), Retry::standard()); + const auto epoch_before = ops.op.read(layout.epochKey("r"), Retry::standard()); + const auto mount_before = ops.op.read(layout.mountKey("r"), Retry::standard()); ASSERT_TRUE(owner_before.has_value()); ASSERT_TRUE(epoch_before.has_value()); ASSERT_TRUE(mount_before.has_value()); @@ -1172,18 +1254,18 @@ TEST(CASMountStartup, ExistingPoolWithoutCatalogFailsBeforeSlotMutation) .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", .skip_access_check = true}), DB::Exception); - const auto owner_after = backend->get(layout.ownerKey("r")); - const auto epoch_after = backend->get(layout.epochKey("r")); - const auto mount_after = backend->get(layout.mountKey("r")); + const auto owner_after = ops.op.read(layout.ownerKey("r"), Retry::standard()); + const auto epoch_after = ops.op.read(layout.epochKey("r"), Retry::standard()); + const auto mount_after = ops.op.read(layout.mountKey("r"), Retry::standard()); ASSERT_TRUE(owner_after.has_value()); ASSERT_TRUE(epoch_after.has_value()); ASSERT_TRUE(mount_after.has_value()); EXPECT_EQ(owner_after->bytes, owner_before->bytes); - EXPECT_EQ(owner_after->token, owner_before->token); + EXPECT_EQ(owner_after->etag, owner_before->etag); EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); - EXPECT_EQ(epoch_after->token, epoch_before->token); + EXPECT_EQ(epoch_after->etag, epoch_before->etag); EXPECT_EQ(mount_after->bytes, mount_before->bytes); - EXPECT_EQ(mount_after->token, mount_before->token); + EXPECT_EQ(mount_after->etag, mount_before->etag); } TEST(CASMountReadOnly, ForeignOwnedPoolOpensWithoutMutation) @@ -1195,10 +1277,11 @@ TEST(CASMountReadOnly, ForeignOwnedPoolOpensWithoutMutation) auto a = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); + Ops ops(b); /// Capture the control objects BEFORE the read-only open so we can prove it mutated nothing. - const auto owner_before = b->get(l.ownerKey("r")); - const auto mount_before = b->get(l.mountKey("r")); - const auto epoch_before = b->get(l.epochKey("r")); + const auto owner_before = ops.op.read(l.ownerKey("r"), Retry::standard()); + const auto mount_before = ops.op.read(l.mountKey("r"), Retry::standard()); + const auto epoch_before = ops.op.read(l.epochKey("r"), Retry::standard()); ASSERT_TRUE(owner_before.has_value()); ASSERT_TRUE(mount_before.has_value()); ASSERT_TRUE(epoch_before.has_value()); @@ -1215,9 +1298,9 @@ TEST(CASMountReadOnly, ForeignOwnedPoolOpensWithoutMutation) /// And it mutated nothing: owner still decodes to A's uuid, the mount body is still A's, and the /// raw bytes of owner/epoch/mount are byte-for-byte unchanged (no second owner, no re-claim). - const auto owner_after = b->get(l.ownerKey("r")); - const auto mount_after = b->get(l.mountKey("r")); - const auto epoch_after = b->get(l.epochKey("r")); + const auto owner_after = ops.op.read(l.ownerKey("r"), Retry::standard()); + const auto mount_after = ops.op.read(l.mountKey("r"), Retry::standard()); + const auto epoch_after = ops.op.read(l.epochKey("r"), Retry::standard()); ASSERT_TRUE(owner_after.has_value()); ASSERT_TRUE(mount_after.has_value()); ASSERT_TRUE(epoch_after.has_value()); @@ -1230,10 +1313,29 @@ TEST(CASMountReadOnly, ForeignOwnedPoolOpensWithoutMutation) EXPECT_EQ(epoch_after->bytes, epoch_before->bytes); } -/// Pool::open must call validateCasRequestBudget itself (not just the free function in isolation — -/// see gtest_cas_request_control.cpp for that): an inconsistent cas_request_budget must refuse a -/// writable mount end-to-end (RFC cas-s3-timeout-retry-control §required-timeout-model), never mount -/// silently with a budget that could let a controlled attempt outlive the lease it is fenced under. +/// `validateCasRequestBudget` itself, isolated from `Pool::open`: a consistent default budget is +/// accepted silently, and the overflow-safe comparison (subtraction against the TTL rather than +/// computing `attempt_timeout_ms + lease_safety_margin_ms` directly) really does reject an absurd +/// near-`UINT64_MAX` config rather than letting the sum wrap to a spuriously small value that would +/// pass the inequality when it should fail closed. +TEST(CASRequestBudget, ValidateAcceptsDefaultsAndRejectsAnOverflowingSumWithoutWrapping) +{ + EXPECT_NO_THROW(validateCasRequestBudget( + CasRequestBudget{}, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000, /*background_renewal=*/false)); + + const CasRequestBudget overflowing{ + .attempt_timeout_ms = std::numeric_limits::max() - 100, + .lease_safety_margin_ms = std::numeric_limits::max() - 100}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(overflowing, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000, /*background_renewal=*/false); + }); +} + +/// Pool::open must call validateCasRequestBudget itself (not just the free function in isolation, +/// pinned directly above): an inconsistent cas_request_budget must refuse a writable mount end-to-end +/// (RFC cas-s3-timeout-retry-control §required-timeout-model), never mount silently with a budget that +/// could let a controlled attempt outlive the lease it is fenced under. TEST(CASMountStartup, RefusesWritableOpenWithInconsistentCasRequestBudget) { auto b = std::make_shared(); @@ -1241,7 +1343,7 @@ TEST(CASMountStartup, RefusesWritableOpenWithInconsistentCasRequestBudget) /// attempt_timeout_ms + lease_safety_margin_ms == mount_lease_ttl_ms below (30000): not STRICTLY /// less, so this must be rejected. const CasRequestBudget bad_budget{ - .attempt_timeout_ms = 25000, .operation_deadline_ms = 30000, .max_attempts = 3, .lease_safety_margin_ms = 5000}; + .attempt_timeout_ms = 25000, .lease_safety_margin_ms = 5000}; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] { Pool::open(b, PoolConfig{ @@ -1263,7 +1365,7 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) /// lease TTL), so it also scales down cas_request_budget to fit — the budget itself is not /// exercised here, only Pool::open's validateCasRequestBudget startup gate. const CasRequestBudget tiny_budget{ - .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; auto a = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", .mount_lease_ttl_ms = std::chrono::milliseconds(300), @@ -1272,16 +1374,18 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) ASSERT_NE(a, nullptr); const uint64_t e1 = a->writerEpoch(); const String mount_key = a->layout().mountKey("r"); - const auto stale_mount = b->get(mount_key); + Ops ops(b); + const auto stale_mount = ops.op.read(mount_key, Retry::standard()); ASSERT_TRUE(stale_mount.has_value()); /// Preserve A's live lease as if its process disappeared without running C++ teardown. Destroying /// the real Pool first keeps the parent process valid; replaying the saved body recreates the exact /// durable stale-lease state that a crashed process would leave behind. a.reset(); - const auto farewell = b->get(mount_key); + const auto farewell = ops.op.read(mount_key, Retry::standard()); ASSERT_TRUE(farewell.has_value()); - ASSERT_EQ(b->putOverwrite(mount_key, stale_mount->bytes, farewell->token).outcome, PutOutcome::Done); + mustCommit(ops.op.replace(mount_key, stale_mount->bytes, farewell->etag, Retry::standard()), + "replayed stale lease"); /// A restart of the SAME server (same uuid) must NOT abort: it waits out the stale lease (<= ~300ms) /// and reclaims the mount, coming up with a strictly higher durable writer_epoch. The replayed live @@ -1289,7 +1393,10 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) /// fake `boot_ms_fn` + `wait_sleep_fn` (mirroring /// `CASMountOpenWaits.UncleanOpenPaysOnlyTheObservationWindow`) so the observation window resolves /// instantly instead of blocking this test on real time. - uint64_t a2_fake_boot = 0; + /// Held in a shared atomic, not a plain local: `wait_sleep_fn` below mutates it, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto a2_fake_boot = std::make_shared>(0); PoolPtr a2; EXPECT_NO_THROW( a2 = Pool::open(b, PoolConfig{ @@ -1297,8 +1404,14 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) .mount_lease_ttl_ms = std::chrono::milliseconds(300), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = tiny_budget, - .boot_ms_fn = [&a2_fake_boot] { return a2_fake_boot; }, - .wait_sleep_fn = [&a2_fake_boot](uint64_t ms) { a2_fake_boot += ms; }})); + .boot_ms_fn = [a2_fake_boot] + { + return a2_fake_boot->load(); + }, + .wait_sleep_fn = [a2_fake_boot](uint64_t ms) + { + *a2_fake_boot += ms; + }})); ASSERT_NE(a2, nullptr); EXPECT_GT(a2->writerEpoch(), e1); @@ -1316,17 +1429,27 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) .cas_request_budget = tiny_budget}); const String overlap_mount_key = first->layout().mountKey("r"); - uint64_t overlap_fake_boot = 0; + /// Held in a shared atomic, not a plain local: `wait_sleep_fn` below mutates it, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto overlap_fake_boot = std::make_shared>(0); auto replacement = Pool::open(overlap_backend, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r", .mount_lease_ttl_ms = std::chrono::milliseconds(300), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = tiny_budget, - .boot_ms_fn = [&overlap_fake_boot] { return overlap_fake_boot; }, - .wait_sleep_fn = [&overlap_fake_boot](uint64_t ms) { overlap_fake_boot += ms; }}); + .boot_ms_fn = [overlap_fake_boot] + { + return overlap_fake_boot->load(); + }, + .wait_sleep_fn = [overlap_fake_boot](uint64_t ms) + { + *overlap_fake_boot += ms; + }}); ASSERT_NE(replacement, nullptr); - const auto reclaimer_slot_before = overlap_backend->get(overlap_mount_key); + Ops overlap_ops(overlap_backend); + const auto reclaimer_slot_before = overlap_ops.op.read(overlap_mount_key, Retry::standard()); ASSERT_TRUE(reclaimer_slot_before.has_value()); const uint64_t overlap_violations_before = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); @@ -1335,7 +1458,7 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), overlap_violations_before + 1); - const auto reclaimer_slot_after = overlap_backend->get(overlap_mount_key); + const auto reclaimer_slot_after = overlap_ops.op.read(overlap_mount_key, Retry::standard()); ASSERT_TRUE(reclaimer_slot_after.has_value()); EXPECT_EQ(reclaimer_slot_after->bytes, reclaimer_slot_before->bytes) << "the deposed Pool's release must not retire the reclaimer's lease"; @@ -1352,11 +1475,11 @@ TEST(CASMountLease, BodyCarriesFloorAndFence) m.started_at_ms = 1000; m.seq = 3; m.expires_at_ms = 2000; - m.min_active = 5; + m.min_active_build_sequence = 5; m.gc_fenced = true; m.write_attempt_id = UInt128{1}; const MountLease d = decodeMountLease(encodeMountLease(m)); - EXPECT_EQ(d.min_active, 5u); + EXPECT_EQ(d.min_active_build_sequence, 5u); EXPECT_TRUE(d.gc_fenced); EXPECT_EQ(d.writer_epoch, 7u); } @@ -1364,9 +1487,9 @@ TEST(CASMountLease, BodyCarriesFloorAndFence) TEST(CASMountLease, RetiredSentinelRoundTrips) { MountLease m; - m.min_active = std::numeric_limits::max(); + m.min_active_build_sequence = std::numeric_limits::max(); m.write_attempt_id = UInt128{1}; - EXPECT_EQ(decodeMountLease(encodeMountLease(m)).min_active, + EXPECT_EQ(decodeMountLease(encodeMountLease(m)).min_active_build_sequence, std::numeric_limits::max()); } @@ -1382,11 +1505,11 @@ constexpr uint64_t kNowMs = 1'000'000; /// of any lease's stamped `expires_at_ms`. constexpr uint64_t kStableThresholdMs = 10'000; -/// Seed one mount body under mountKey(srid) via the on-storage codec (`encodeMountLease` + -/// `putIfAbsent`) — the same interface the keeper writes through. +/// Seed one mount body under mountKey(srid) via the on-storage codec — the same interface the renewer +/// writes through. MountLease seedMount( - Backend & b, const Layout & l, const String & srid, - uint64_t expires_at_ms, bool gc_fenced, uint64_t min_active, uint64_t seq = 1) + CasOperation & op, const Layout & l, const String & srid, + uint64_t expires_at_ms, bool gc_fenced, uint64_t min_active_build_sequence, uint64_t seq = 1) { MountLease m; m.server_uuid = UInt128(srid.back()); // distinct per srid; content is irrelevant to the gate @@ -1396,25 +1519,25 @@ MountLease seedMount( m.started_at_ms = kNowMs; m.seq = seq; m.expires_at_ms = expires_at_ms; - m.min_active = min_active; + m.min_active_build_sequence = min_active_build_sequence; m.gc_fenced = gc_fenced; m.write_attempt_id = UInt128{1}; - b.putIfAbsent(l.mountKey(srid), encodeMountLease(m)); + mustCommit(op.create(l.mountKey(srid), encodeMountLease(m), Retry::standard()), "seeded mount " + srid); return m; } -/// Simulate a keeper's real renewal between two `computeHeartbeatFloor` calls: a token-guarded -/// overwrite that bumps `seq` (and so mints a fresh backend token), leaving everything else as-is. -/// Models the one thing the observation-based fence cares about: the write token changed, so any -/// in-progress observation of the OLD token must restart. -void renewMount(Backend & b, const Layout & l, const String & srid) +/// Simulate a renewer's real renewal between two `computeHeartbeatFloor` calls: a guarded write that +/// bumps `seq` (and so mints a fresh incarnation), leaving everything else as-is. Models the one +/// thing the observation-based fence cares about: the incarnation changed, so any in-progress +/// observation of the OLD one must restart. +void renewMount(CasOperation & op, const Layout & l, const String & srid) { - const auto got = b.get(l.mountKey(srid)); + const auto got = op.read(l.mountKey(srid), Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease m = decodeMountLease(got->bytes); m.seq += 1; - const PutResult res = b.putOverwrite(l.mountKey(srid), encodeMountLease(m), got->token); - ASSERT_EQ(res.outcome, PutOutcome::Done); + mustCommit(op.replace(l.mountKey(srid), encodeMountLease(m), got->etag, Retry::standard()), + "renewed mount " + srid); } } @@ -1423,12 +1546,13 @@ TEST(CASHeartbeatFloor, FirstSightNeverFencesEvenIfStampLooksExpired) auto b = std::make_shared(); Layout l("p"); + Ops ops(b); /// A stamp that would have read as long-expired under the old skew-margin comparison — under /// rev.6 observation the stamp is never even consulted for the fence decision. - seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - const HeartbeatFloor floor = computeHeartbeatFloor(*b, l, /*now_ms*/ kNowMs, /*mono_now_ms*/ 0, + const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, /*now_ms*/ kNowMs, /*mono_now_ms*/ 0, kStableThresholdMs, obs); EXPECT_EQ(floor.fenced_now, 0u); @@ -1437,26 +1561,27 @@ TEST(CASHeartbeatFloor, FirstSightNeverFencesEvenIfStampLooksExpired) EXPECT_EQ(obs.at("s1").first_seen_mono_ms, 0u); } -TEST(CASHeartbeatFloor, StableTokenPastThresholdIsFenced) +TEST(CASHeartbeatFloor, StableIncarnationPastThresholdIsFenced) { auto b = std::make_shared(); Layout l("p"); - seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); EXPECT_EQ(floor_before.fenced_now, 0u); - const MountLease before = decodeMountLease(b->get(l.mountKey("s1"))->bytes); + const MountLease before = decodeMountLease(ops.op.read(l.mountKey("s1"), Retry::standard())->bytes); - /// No renewal in between: the SAME token, observed since mono 0, is now stable for the full + /// No renewal in between: the SAME incarnation, observed since mono 0, is now stable for the full /// threshold on the leader's own clock. - const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 1u); EXPECT_EQ(floor2.fenced_srids, std::vector{"s1"}); - const MountLease fenced = decodeMountLease(b->get(l.mountKey("s1"))->bytes); + const MountLease fenced = decodeMountLease(ops.op.read(l.mountKey("s1"), Retry::standard())->bytes); EXPECT_TRUE(fenced.gc_fenced); EXPECT_EQ(fenced.seq, before.seq + 1); } @@ -1465,23 +1590,24 @@ TEST(CASHeartbeatFloor, RenewalBetweenRoundsRestartsObservation) { auto b = std::make_shared(); Layout l("p"); - seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); ASSERT_TRUE(obs.contains("s1")); - const Token first_token = obs.at("s1").token; + const Etag first_etag = obs.at("s1").etag; - renewMount(*b, l, "s1"); - const Token renewed_token = b->get(l.mountKey("s1"))->token; - EXPECT_NE(renewed_token, first_token); + renewMount(ops.op, l, "s1"); + const Etag renewed_etag = currentEtag(ops.op, l.mountKey("s1")); + EXPECT_NE(renewed_etag, first_etag); - const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 0u); ASSERT_TRUE(obs.contains("s1")); - EXPECT_EQ(obs.at("s1").token, renewed_token); + EXPECT_EQ(obs.at("s1").etag, renewed_etag); EXPECT_EQ(obs.at("s1").first_seen_mono_ms, kStableThresholdMs); } @@ -1494,26 +1620,24 @@ TEST(CASHeartbeatFloor, UnseenSridPrunedFromObservationMap) { auto b = std::make_shared(); Layout l("p"); - seedMount(*b, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); - seedMount(*b, l, "s2", /*expires*/ 10, /*fenced*/ false, /*min_active*/ 0); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); + seedMount(ops.op, l, "s2", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); ASSERT_TRUE(obs.contains("s1")); ASSERT_TRUE(obs.contains("s2")); /// s2's `/mount` key is removed entirely -- e.g. `SYSTEM CAS DROP POOL MEMBER` -- so - /// no future LIST pass will ever visit it again. s1 renews (a live keeper would), so its OWN + /// no future LIST pass will ever visit it again. s1 renews (a live renewer would), so its OWN /// observation restarts and it stays `live` -- isolating this test to the pruning behavior alone, /// not confounding it with s1 also becoming fence-eligible (which would erase its `obs` entry too, /// for an unrelated reason). - renewMount(*b, l, "s1"); - const auto s2_key = l.mountKey("s2"); - const auto got = b->get(s2_key); - ASSERT_TRUE(got.has_value()); - ASSERT_EQ(b->deleteExact(s2_key, got->token).kind, DeleteOutcome::Kind::Deleted); + renewMount(ops.op, l, "s1"); + ASSERT_EQ(ops.op.removeCurrent(l.mountKey("s2"), Retry::standard()), Removal::Removed); - computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); EXPECT_TRUE(obs.contains("s1")); EXPECT_FALSE(obs.contains("s2")) << "a srid removed from the LIST entirely must be pruned from obs, not linger forever"; @@ -1526,37 +1650,38 @@ TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) /// two live mounts — genuinely renewing between the two rounds below, so their observation never /// stabilizes. - seedMount(*b, l, "s1", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active*/ 0); - seedMount(*b, l, "s2", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active*/ 0); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active_build_sequence*/ 0); + seedMount(ops.op, l, "s2", /*expires*/ kNowMs + 60'000, /*fenced*/ false, /*min_active_build_sequence*/ 0); /// dead — no renewal between the two rounds below — must be fenced-out by the second call. - seedMount(*b, l, "s3", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active*/ 0); - /// already-fenced — excluded, body byte-identical after both calls (no PUT). - seedMount(*b, l, "s4", /*expires*/ kNowMs - 60'000, /*fenced*/ true, /*min_active*/ 0); - /// terminated (min_active == UINT64_MAX) with expired-looking timestamps — excluded, not fenced. - seedMount(*b, l, "s5", /*expires*/ kNowMs - 60'000, /*fenced*/ false, - /*min_active*/ std::numeric_limits::max()); + seedMount(ops.op, l, "s3", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active_build_sequence*/ 0); + /// already-fenced — excluded, body byte-identical after both calls (no write). + seedMount(ops.op, l, "s4", /*expires*/ kNowMs - 60'000, /*fenced*/ true, /*min_active_build_sequence*/ 0); + /// terminated (min_active_build_sequence == UINT64_MAX) with expired-looking timestamps — excluded, not fenced. + seedMount(ops.op, l, "s5", /*expires*/ kNowMs - 60'000, /*fenced*/ false, + /*min_active_build_sequence*/ std::numeric_limits::max()); MountObservationMap obs; /// Round 1 (mono 0): first sight of every non-terminal mount — nothing is fence-eligible yet. - const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); EXPECT_EQ(floor_before.live, 3u); // s1, s2, s3: observation just started EXPECT_EQ(floor_before.terminated, 1u); // s5 EXPECT_EQ(floor_before.fenced_now, 0u); EXPECT_EQ(floor_before.already_fenced, 1u); // s4 - /// s1 and s2 renew between rounds (as a live keeper would); s3 does not (it crashed). - renewMount(*b, l, "s1"); - renewMount(*b, l, "s2"); + /// s1 and s2 renew between rounds (as a live renewer would); s3 does not (it crashed). + renewMount(ops.op, l, "s1"); + renewMount(ops.op, l, "s2"); - const auto s3_before = b->get(l.mountKey("s3")); - const auto s4_before = b->get(l.mountKey("s4")); + const auto s3_before = ops.op.read(l.mountKey("s3"), Retry::standard()); + const auto s4_before = ops.op.read(l.mountKey("s4"), Retry::standard()); ASSERT_TRUE(s3_before.has_value()); ASSERT_TRUE(s4_before.has_value()); - /// Round 2 (mono == threshold): s1/s2's renewed tokens restart their observation (still live); - /// s3's original token has now held stable for the full threshold -> fenced. - const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + /// Round 2 (mono == threshold): s1/s2's renewed incarnations restart their observation (still + /// live); s3's original incarnation has now held stable for the full threshold -> fenced. + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); EXPECT_EQ(floor2.live, 2u); // s1, s2: renewed, observation restarted @@ -1565,7 +1690,7 @@ TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) EXPECT_EQ(floor2.already_fenced, 1u); // s4 /// The dead body was fenced: gc_fenced set, seq bumped, the rest of the body preserved. - const auto s3_after = b->get(l.mountKey("s3")); + const auto s3_after = ops.op.read(l.mountKey("s3"), Retry::standard()); ASSERT_TRUE(s3_after.has_value()); const MountLease s3_prev = decodeMountLease(s3_before->bytes); const MountLease s3_now = decodeMountLease(s3_after->bytes); @@ -1576,19 +1701,19 @@ TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) EXPECT_EQ(s3_now.hostname, s3_prev.hostname); EXPECT_EQ(s3_now.expires_at_ms, s3_prev.expires_at_ms); - /// The already-fenced body was not touched (no PUT) across either call. - const auto s4_after = b->get(l.mountKey("s4")); + /// The already-fenced body was not touched (no write) across either call. + const auto s4_after = ops.op.read(l.mountKey("s4"), Retry::standard()); ASSERT_TRUE(s4_after.has_value()); EXPECT_EQ(s4_after->bytes, s4_before->bytes); } namespace { -/// A delegating backend whose `putOverwrite` of the target mount key first performs an inner renewal -/// (a real, token-correct overwrite that pushes expiry far into the future) and THEN delegates — so -/// the caller's fence-out overwrite lands on a stale token and returns PreconditionFailed. The inner -/// renewal runs exactly once (`renewed`), modelling a holder that renews concurrently in the window -/// between the function's GET and its fence-out PUT. +/// A delegating backend whose guarded write of the target mount key first performs an inner renewal +/// (a real, correctly-guarded write that pushes expiry far into the future) and THEN delegates — so +/// the caller's fence-out write lands on a stale precondition and is refused. The inner renewal runs +/// exactly once (`renewed`), modelling a holder that renews concurrently in the window between the +/// function's read and its fence-out write. class RenewOnFenceBackend : public InMemoryBackend { public: @@ -1597,21 +1722,22 @@ class RenewOnFenceBackend : public InMemoryBackend { } - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, - const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - if (key == target_key && !renewed) + if (expected_value && key == target_key && !renewed) { renewed = true; - /// The holder renews under the real current token: fresh far-future expiry. - const auto got = InMemoryBackend::get(key, {}); + /// The holder renews under the real current incarnation: fresh far-future expiry. + const auto got = InMemoryBackend::read(key, access); MountLease m = decodeMountLease(got->bytes); m.seq += 1; m.expires_at_ms = renewed_expires_ms; - const PutResult renew = InMemoryBackend::putOverwrite(key, encodeMountLease(m), got->token); - EXPECT_EQ(renew.outcome, PutOutcome::Done); + const auto renew = InMemoryBackend::write(key, encodeMountLease(m), got->value, access); + EXPECT_TRUE(renew.has_value()); } - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } private: @@ -1621,31 +1747,32 @@ class RenewOnFenceBackend : public InMemoryBackend }; } -TEST(CASHeartbeatFloor, FenceOutLosesTokenRaceReclassifiesLive) +TEST(CASHeartbeatFloor, FenceOutLosesTheIncarnationRaceAndReclassifiesLive) { Layout l("p"); auto b = std::make_shared( l.mountKey("s1"), /*renewed_expires*/ kNowMs + 120'000); + Ops ops(b); - seedMount(*b, l, "s1", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active*/ 0); + seedMount(ops.op, l, "s1", /*expires*/ kNowMs - 60'000, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; /// Round 1: first sight, observation starts — never reaches the fence-out path (the race /// decorator stays armed for round 2). - const HeartbeatFloor floor_before = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); EXPECT_EQ(floor_before.fenced_now, 0u); - /// Round 2: the token has been stable past threshold, so the function attempts the fence-out. - /// The decorator renews concurrently under the real token, the PUT hits PreconditionFailed, the - /// function re-GETs and reclassifies it as live (observation restarted on the new token) — never - /// fenced. - const HeartbeatFloor floor2 = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ kStableThresholdMs, + /// Round 2: the incarnation has been stable past threshold, so the function attempts the + /// fence-out. The decorator renews concurrently under the real incarnation, the write is refused, + /// and the re-decision reclassifies the slot as live (observation restarted on the new + /// incarnation) — never fenced. + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 0u); EXPECT_EQ(floor2.live, 1u); - const auto after = b->get(l.mountKey("s1")); + const auto after = ops.op.read(l.mountKey("s1"), Retry::standard()); ASSERT_TRUE(after.has_value()); EXPECT_FALSE(decodeMountLease(after->bytes).gc_fenced); } @@ -1655,8 +1782,9 @@ TEST(CASHeartbeatFloor, EmptyPrefixYieldsNoLiveMounts) auto b = std::make_shared(); Layout l("p"); + Ops ops(b); MountObservationMap obs; - const HeartbeatFloor floor = computeHeartbeatFloor(*b, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); EXPECT_EQ(floor.live, 0u); EXPECT_EQ(floor.terminated, 0u); @@ -1673,16 +1801,17 @@ TEST(CASListMounts, ClassifiesEveryStateReadOnly) const uint64_t now_ms = 1'000'000; const uint64_t ttl_ms = 10'000; + Ops ops(backend); /// live: fresh claim for srid "a" - ASSERT_EQ(claimMount(*backend, layout, "a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, + ASSERT_EQ(claimMount(ops.op, layout, "a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, MountClaimResult::Claimed); /// expired: claim for "b" whose lease ran out long before now_ms - ASSERT_EQ(claimMount(*backend, layout, "b", UInt128{2}, 1, now_ms - 100'000, ttl_ms).kind, + ASSERT_EQ(claimMount(ops.op, layout, "b", UInt128{2}, 1, now_ms - 100'000, ttl_ms).kind, MountClaimResult::Claimed); /// corrupt: garbage bytes in "c"'s mount slot - backend->putIfAbsent(layout.mountKey("c"), "garbage-not-a-proto", {}); + mustCommit(ops.op.create(layout.mountKey("c"), "garbage-not-a-proto", Retry::standard()), "corrupt slot"); - auto mounts = listMounts(*backend, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); + auto mounts = listMounts(ops.op, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); ASSERT_EQ(mounts.size(), 3u); std::map by_srid; for (const auto & m : mounts) @@ -1693,7 +1822,7 @@ TEST(CASListMounts, ClassifiesEveryStateReadOnly) /// READ-ONLY guarantee: "b" is expired but must NOT be fenced by listMounts /// (computeHeartbeatFloor would stamp gc_fenced=true; the introspection view must not). - auto again = listMounts(*backend, layout, now_ms, ttl_ms / 2); + auto again = listMounts(ops.op, layout, now_ms, ttl_ms / 2); for (const auto & m : again) if (m.srid == "b") { @@ -1712,10 +1841,11 @@ TEST(CASListMounts, NestedSridIsNotTruncated) const uint64_t now_ms = 1'000'000; const uint64_t ttl_ms = 10'000; - ASSERT_EQ(claimMount(*backend, layout, "shard-01/replica-a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, + Ops ops(backend); + ASSERT_EQ(claimMount(ops.op, layout, "shard-01/replica-a", UInt128{1}, /*our_epoch=*/1, now_ms, ttl_ms).kind, MountClaimResult::Claimed); - auto mounts = listMounts(*backend, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); + auto mounts = listMounts(ops.op, layout, now_ms, /*skew_margin_ms=*/ttl_ms / 2); ASSERT_EQ(mounts.size(), 1u); EXPECT_EQ(mounts[0].srid, "shard-01/replica-a"); EXPECT_EQ(mounts[0].state, "live"); @@ -1729,24 +1859,25 @@ TEST(CASClaimMount, SameEpochFencedIsNotRefreshable) using namespace DB::Cas; auto backend = std::make_shared(); Layout layout("pool"); + Ops ops(backend); /// mint for (uuid 1, epoch 1), then fence it in place (what computeHeartbeatFloor does): - ASSERT_EQ(claimMount(*backend, layout, "a", DB::UInt128{1}, 1, 1000, 10'000).kind, + ASSERT_EQ(claimMount(ops.op, layout, "a", DB::UInt128{1}, 1, 1000, 10'000).kind, MountClaimResult::Claimed); { - auto got = backend->get(layout.mountKey("a")); + auto got = ops.op.read(layout.mountKey("a"), Retry::standard()); MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; fenced.seq += 1; - ASSERT_EQ(backend->putOverwrite(layout.mountKey("a"), encodeMountLease(fenced), got->token).outcome, - PutOutcome::Done); + mustCommit(ops.op.replace(layout.mountKey("a"), encodeMountLease(fenced), got->etag, Retry::standard()), + "fence-out"); } /// Same (uuid, epoch) re-claim must NOT refresh a fenced body — a fence costs an epoch: - const auto r = claimMount(*backend, layout, "a", DB::UInt128{1}, 1, 2000, 10'000); + const auto r = claimMount(ops.op, layout, "a", DB::UInt128{1}, 1, 2000, 10'000); EXPECT_EQ(r.kind, MountClaimResult::FencedSelf); /// The body on the backend is still the fenced one (no write happened): - EXPECT_TRUE(decodeMountLease(backend->get(layout.mountKey("a"))->bytes).gc_fenced); + EXPECT_TRUE(decodeMountLease(ops.op.read(layout.mountKey("a"), Retry::standard())->bytes).gc_fenced); /// A DIFFERENT epoch reclaims immediately (existing branch, unchanged): - EXPECT_EQ(claimMount(*backend, layout, "a", DB::UInt128{1}, 2, 2000, 10'000).kind, + EXPECT_EQ(claimMount(ops.op, layout, "a", DB::UInt128{1}, 2, 2000, 10'000).kind, MountClaimResult::Claimed); } @@ -1755,31 +1886,34 @@ TEST(CASClaimMount, SameEpochFencedIsNotRefreshable) /// A same-uuid, different-epoch lease whose STAMPED `expires_at_ms` looks long expired on OUR wall /// clock must NOT be reclaimed by that comparison alone — a clock-skewed or simply late-observing /// caller must never trust a bare wall-clock read across incarnations. `claimMount` (without a -/// `proven_dead_token`) always reports `LiveDoubleStart` for this branch now; only the observation +/// `proven_dead_incarnation`) always reports `LiveDoubleStart` for this branch now; only the observation /// loop (`claimMountAwaitingExpiry`) may turn it into a reclaim, and only after proving death on ITS /// OWN clock. TEST(CASMountObservation, ExpiredLookingLeaseIsNotReclaimedByWallClock) { auto b = std::make_shared(); Layout l{"p"}; + Ops ops(b); /// Predecessor epoch 7 stamped expires_at_ms = 1000; our wall clock says 999999 (long past). - auto first = claimMount(*b, l, "r", UInt128(1), 7, /*now_ms=*/500, /*ttl_ms=*/500); + auto first = claimMount(ops.op, l, "r", UInt128(1), 7, /*now_ms=*/500, /*ttl_ms=*/500); ASSERT_EQ(first.kind, MountClaimResult::Claimed); - auto r = claimMount(*b, l, "r", UInt128(1), /*our_epoch=*/8, /*now_ms=*/999999, 500); + auto r = claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/8, /*now_ms=*/999999, 500); EXPECT_EQ(r.kind, MountClaimResult::LiveDoubleStart); /// no wall-clock trust } -/// The observation loop reclaims once the write-token has held stable for the FULL rate-bound -/// threshold (`ttl_ms + ttl_ms/20 + poll_interval_ms`) on its OWN (injected, fake) clock — never -/// short-circuiting on the wall clock, which this test drives to an irrelevant, already-expired value. -TEST(CASMountObservation, TokenStableForThresholdThenReclaimed) +/// The observation loop reclaims once the observed incarnation has held stable for the FULL +/// rate-bound threshold (`ttl_ms + ttl_ms/20 + poll_interval_ms`) on its OWN (injected, fake) clock — +/// never short-circuiting on the wall clock, which this test drives to an irrelevant, already-expired +/// value. +TEST(CASMountObservation, IncarnationStableForThresholdThenReclaimed) { auto b = std::make_shared(); Layout l{"p"}; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); uint64_t mono = 0; std::vector sleeps; - auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), 8, + auto r = claimMountAwaitingExpiry(ops.op, l, "r", UInt128(1), 8, []{ return uint64_t{999999}; }, /// wall clock: irrelevant [&]{ return mono; }, /// observation clock /*ttl_ms=*/500, /*poll_interval_ms=*/50, @@ -1789,28 +1923,31 @@ TEST(CASMountObservation, TokenStableForThresholdThenReclaimed) EXPECT_GE(mono, 500 + 500 / 20 + 50); /// full threshold actually waited } -/// A renewal DURING the observation window (the real holder is still alive) bumps the write-token — -/// the loop must detect the mismatch and RESTART the observation from the new token, never reclaiming -/// off a window that started watching a now-superseded token. +/// A renewal DURING the observation window (the real holder is still alive) mints a new incarnation — +/// the loop must detect the mismatch and RESTART the observation from it, never reclaiming off a +/// window that started watching a now-superseded incarnation. TEST(CASMountObservation, RenewalDuringObservationRestartsIt) { auto b = std::make_shared(); Layout l{"p"}; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); - - /// The real (still-alive) holder's keeper for epoch 7: `start()` adopts the slot `claimMount` just - /// wrote (no seq bump, per the ADOPT RULE), then synchronous renewal bumps the token mid-observation. - uint64_t keeper_wall = 500; - MountLeaseKeeper keeper(b, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), - [&] { return keeper_wall; }, [] { return uint64_t{0}; }, {}, - std::chrono::milliseconds(0)); - keeper.start(); + uint64_t renewer_boot = 0; + Ops ops(b, &renewer_boot); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, 500, 500).kind, MountClaimResult::Claimed); + + /// The real (still-alive) holder's renewer for epoch 7: `start()` adopts the slot `claimMount` just + /// wrote (no seq bump, per the ADOPT RULE), then a synchronous renewal mints a new incarnation + /// mid-observation. + uint64_t renewer_wall = 500; + MountLeaseRenewer renewer(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), + [&] { return renewer_wall; }, [] { return uint64_t{0}; }, {}, + std::chrono::milliseconds(0), [&] { return renewer_boot; }); + renewer.start(); const uint64_t threshold_ms = 500 + 500 / 20 + 50; /// = 575 uint64_t mono = 0; bool renewed = false; int wait_starts = 0; - auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), 8, + auto r = claimMountAwaitingExpiry(ops.op, l, "r", UInt128(1), 8, []{ return uint64_t{999999}; }, /// wall clock: irrelevant [&]{ return mono; }, /// observation clock /*ttl_ms=*/500, /*poll_interval_ms=*/50, @@ -1822,7 +1959,7 @@ TEST(CASMountObservation, RenewalDuringObservationRestartsIt) if (!renewed && mono >= threshold_ms - 50) { renewed = true; - renewKeeperOrThrow(keeper); + renewOrThrow(renewer); } }, /*on_wait_start=*/[&](const MountLease &, uint64_t) { ++wait_starts; }); @@ -1842,22 +1979,23 @@ TEST(CASMountObservation, GcFencedIsReclaimedInstantlyWithPriorFenced) { auto b = std::make_shared(); Layout l{"p"}; - ASSERT_EQ(claimMount(*b, l, "r", UInt128(1), 7, 1000, 500).kind, MountClaimResult::Claimed); + Ops ops(b); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), 7, 1000, 500).kind, MountClaimResult::Claimed); /// Fence it manually (what `computeHeartbeatFloor`'s fence-out does): gc_fenced=true, seq+1, - /// token-guarded. + /// guarded by the observed incarnation. { - auto got = b->get(l.mountKey("r")); + auto got = ops.op.read(l.mountKey("r"), Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; fenced.seq += 1; - ASSERT_EQ(b->putOverwrite(l.mountKey("r"), encodeMountLease(fenced), got->token).outcome, - PutOutcome::Done); + mustCommit(ops.op.replace(l.mountKey("r"), encodeMountLease(fenced), got->etag, Retry::standard()), + "fence-out"); } int sleeps = 0; - auto r = claimMountAwaitingExpiry(*b, l, "r", UInt128(1), /*our_epoch=*/8, + auto r = claimMountAwaitingExpiry(ops.op, l, "r", UInt128(1), /*our_epoch=*/8, []{ return uint64_t{999999}; }, []{ return uint64_t{0}; }, /*ttl_ms=*/500, /*poll_interval_ms=*/50, @@ -1875,60 +2013,62 @@ TEST(CASMountObservation, GcFencedIsReclaimedInstantlyWithPriorFenced) TEST(CASFenceTerminal, AbsentMountSlotIsNotTerminal) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; - EXPECT_FALSE(isCreatorFenceTerminal(b, l, "never-mounted", 1)) + EXPECT_FALSE(isCreatorFenceTerminal(ops.op, l, "never-mounted", 1)) << "absence proves nothing about liveness -- never waved through"; } TEST(CASFenceTerminal, UndecodableMountBodyIsNotTerminal) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; - b.putIfAbsent(l.mountKey("r"), "garbage-not-a-lease", {}); - EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 1)) + mustCommit(ops.op.create(l.mountKey("r"), "garbage-not-a-lease", Retry::standard()), "undecodable lease"); + EXPECT_FALSE(isCreatorFenceTerminal(ops.op, l, "r", 1)) << "an unreadable lease of some other format generation must block, never wave through"; } TEST(CASFenceTerminal, GcFencedIsTerminal) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; - ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); - auto got = b.get(l.mountKey("r")); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); + auto got = ops.op.read(l.mountKey("r"), Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease fenced = decodeMountLease(got->bytes); fenced.gc_fenced = true; - ASSERT_EQ(b.putOverwrite(l.mountKey("r"), encodeMountLease(fenced), got->token).outcome, PutOutcome::Done); + mustCommit(ops.op.replace(l.mountKey("r"), encodeMountLease(fenced), got->etag, Retry::standard()), + "fence-out"); - EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)); + EXPECT_TRUE(isCreatorFenceTerminal(ops.op, l, "r", 7)); } TEST(CASFenceTerminal, CleanFarewellIsTerminal) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; - ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); - auto got = b.get(l.mountKey("r")); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/7, 1000, 500).kind, MountClaimResult::Claimed); + auto got = ops.op.read(l.mountKey("r"), Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease retired = decodeMountLease(got->bytes); - retired.min_active = std::numeric_limits::max(); - ASSERT_EQ(b.putOverwrite(l.mountKey("r"), encodeMountLease(retired), got->token).outcome, PutOutcome::Done); + retired.min_active_build_sequence = std::numeric_limits::max(); + mustCommit(ops.op.replace(l.mountKey("r"), encodeMountLease(retired), got->etag, Retry::standard()), + "farewell"); - EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)); + EXPECT_TRUE(isCreatorFenceTerminal(ops.op, l, "r", 7)); } TEST(CASFenceTerminal, ADifferentLiveWriterEpochIsTerminalForTheOldOne) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; /// Slot now held at epoch 8 -- epoch 7's incarnation is superseded regardless of ITS OWN /// certificate (neither fenced nor farewelled). - ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/8, 1000, 500).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/8, 1000, 500).kind, MountClaimResult::Claimed); - EXPECT_TRUE(isCreatorFenceTerminal(b, l, "r", 7)) + EXPECT_TRUE(isCreatorFenceTerminal(ops.op, l, "r", 7)) << "a different epoch is currently live at this slot -- epoch 7 can never reclaim it"; - EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 8)) + EXPECT_FALSE(isCreatorFenceTerminal(ops.op, l, "r", 8)) << "epoch 8 IS the current live epoch -- not terminal"; } @@ -1936,12 +2076,216 @@ TEST(CASFenceTerminal, ADifferentLiveWriterEpochIsTerminalForTheOldOne) /// treated as terminal -- mirrors `claimMount`'s own refusal to trust a bare timestamp comparison. TEST(CASFenceTerminal, ExpiredButSameEpochAndUncertifiedIsNotTerminal) { - InMemoryBackend b; + Ops ops(std::make_shared()); Layout l{"p"}; /// A lease whose stamped expiry is already far in the past, same epoch throughout. - ASSERT_EQ(claimMount(b, l, "r", UInt128(1), /*our_epoch=*/7, /*now_ms=*/0, /*ttl_ms=*/1).kind, + ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*our_epoch=*/7, /*now_ms=*/0, /*ttl_ms=*/1).kind, MountClaimResult::Claimed); - EXPECT_FALSE(isCreatorFenceTerminal(b, l, "r", 7)) + EXPECT_FALSE(isCreatorFenceTerminal(ops.op, l, "r", 7)) << "expiry alone is never a certificate of death, exactly like claimMount's own discipline"; } + +/// The absent-epoch path's post-conflict recheck is a CHECK, not a blanket refusal, and both halves +/// have to hold: work that became visible across the conflict must block the allocation, and an +/// unchanged, still-empty subtree must let it proceed from the winner's own epoch state. Dropping the +/// recheck breaks the first arm; turning it into an unconditional refusal breaks the second. +TEST(CASServerRoot, AllocateWriterEpochKeepsThePostConflictCorruptionCheck) +{ + const Layout layout("p"); + { + auto backend = std::make_shared(/*reveal_owned_work=*/true); + Ops ops(backend); + EXPECT_THROW(allocateWriterEpoch( + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), DB::Exception); + EXPECT_TRUE(backend->fired); + ASSERT_TRUE(backend->winner_installed); + } + { + auto backend = std::make_shared(/*reveal_owned_work=*/false); + Ops ops(backend); + EXPECT_EQ(allocateWriterEpoch( + ops.op, layout, "root/x", EpochMintPolicy::NormalMount, 0, emptyCatalogObservation()), 2u) + << "the second decision must allocate from the conflict winner's epoch, not refuse outright"; + EXPECT_TRUE(backend->fired); + ASSERT_TRUE(backend->winner_installed); + const auto epoch = ops.op.read(layout.epochKey("root/x"), Retry::standard()); + ASSERT_TRUE(epoch.has_value()); + EXPECT_EQ(decodeServerEpoch(epoch->bytes).next_writer_epoch, 3u); + } +} + +/// Adoption costs exactly two requests: one read that both decides the branch and supplies the +/// precondition, and one write. A presence probe ahead of the read, or a second read to recover a +/// precondition the first one already carried, shows up here as a third request. +TEST(CASMountLease, ClaimAdoptIsTwoRequests) +{ + auto backend = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + Ops ops(backend); + + /// The absent-slot mint. + backend->reads = backend->heads = backend->writes = 0; + MountLeaseRenewer minting(ops.mount, ops.farewell, l, "fresh", UInt128(1), 7, + std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); + minting.start(); + EXPECT_EQ(backend->reads, 1u); + EXPECT_EQ(backend->writes, 1u); + EXPECT_EQ(backend->heads, 0u); + + /// The adoption of a slot `claimMount` already wrote. + ASSERT_EQ(claimMount(ops.op, l, "adopted", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, + MountClaimResult::Claimed); + backend->reads = backend->heads = backend->writes = 0; + MountLeaseRenewer adopting(ops.mount, ops.farewell, l, "adopted", UInt128(1), 7, + std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); + adopting.start(); + EXPECT_EQ(backend->reads, 1u); + EXPECT_EQ(backend->writes, 1u); + EXPECT_EQ(backend->heads, 0u); +} + +/// A mount whose fence has dropped must still hand its slot back: the renewal is refused (it would be +/// writing under authority this node no longer holds), while the farewell runs on the open plane and +/// lands. Deliberately two renewers: `release` is admitted only from `Active`, so a renewer whose +/// renewal already went terminal never reaches its own farewell -- the ordering the two halves below +/// pin separately. +TEST(CASMountLease, FarewellRunsOnAnOpenFenceAfterTheMountFenceIsLost) +{ + auto backend = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + uint64_t boot = 0; + bool fence_lost = false; + + CasRequests mount_requests(backend, Fence{ + [] { return uint64_t{0}; }, + [&fence_lost](uint64_t, uint64_t) { return fence_lost ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [&fence_lost](uint64_t) + { + if (fence_lost) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "mount fence lost"); + }}); + mount_requests.setNowFnForTest([&boot] { return boot; }); + mount_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + CasRequests open_requests = openRequestsForTest(backend); + open_requests.setNowFnForTest([&boot] { return boot; }); + open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + CasOperation seed = open_requests.admit(); + + ASSERT_EQ(claimMount(seed, l, "renewing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(seed, l, "departing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); + + MountLeaseRenewer renewing(mount_requests, open_requests, l, "renewing", UInt128(1), 7, + std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, + {}, std::chrono::milliseconds(0), [&] { return boot; }); + MountLeaseRenewer departing(mount_requests, open_requests, l, "departing", UInt128(1), 7, + std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, + {}, std::chrono::milliseconds(0), [&] { return boot; }); + renewing.start(); + departing.start(); + + fence_lost = true; + + const MountRenewResult refused = renewing.renew(MountRenewOperationEnvironment{}); + EXPECT_EQ(refused.outcome, MountRenewOutcome::Terminal); + EXPECT_FALSE(refused.sent_any); + EXPECT_FALSE(renewing.canRelease()) << "a terminal renewal leaves no farewell to run"; + + EXPECT_NO_THROW(departing.release()); + const MountLease farewell = decodeMountLease(seed.read(l.mountKey("departing"), Retry::standard())->bytes); + EXPECT_EQ(farewell.min_active_build_sequence, std::numeric_limits::max()); +} + +/// The claim is admitted off the mount fence, and it has to be: a self-remount runs with the fence +/// already latched lost, so a claim gated on it could never reclaim the slot. What keeps the claim +/// safe is the conditional write it makes, not the fence. +TEST(CASMountLease, ClaimIsNotAdmittedUnderTheMountFence) +{ + auto backend = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + uint64_t boot = 0; + + CasRequests mount_requests(backend, Fence{ + [] { return uint64_t{0}; }, + [](uint64_t, uint64_t) { return Fence::Admit::LostOrRearmed; }, + [](uint64_t) { throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "mount fence lost"); }}); + mount_requests.setNowFnForTest([&boot] { return boot; }); + mount_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + CasRequests open_requests = openRequestsForTest(backend); + open_requests.setNowFnForTest([&boot] { return boot; }); + open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + + MountLeaseRenewer renewer(mount_requests, open_requests, l, "r", UInt128(1), 7, + std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, + {}, std::chrono::milliseconds(0), [&] { return boot; }); + EXPECT_NO_THROW(renewer.start()); + + CasOperation reader = open_requests.admit(); + const MountLease claimed = decodeMountLease(reader.read(l.mountKey("r"), Retry::standard())->bytes); + EXPECT_EQ(claimed.writer_epoch, 7u); + EXPECT_EQ(claimed.seq, 1u); +} + +/// A lost owner-claim race is decided from the conflict's OWN resolve observation, so the two outcomes +/// have to be told apart from that alone: a racer that installed our uuid leaves nothing to do, a +/// foreign one fails closed. Reading the key again would answer a later question than the conflict +/// asked, and would cost a request per race. +TEST(CASServerRootClaim, OwnerLostToARacerIsDecidedFromTheConflictObservation) +{ + Layout l("p"); + { + auto backend = std::make_shared(UInt128(1)); + Ops ops(backend); + EXPECT_NO_THROW(claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation())); + EXPECT_TRUE(backend->fired); + /// The pre-claim read plus the create's own conflict-resolve read, and no third: a re-read + /// added back for the decision itself would raise this to 3. + EXPECT_EQ(backend->owner_reads, 2u); + } + { + auto backend = std::make_shared(UInt128(2)); + Ops ops(backend); + DB::Cas::tests::expectThrowsCodeWithMessage( + DB::ErrorCodes::CORRUPTED_DATA, + "claimed by a different server during our claim", + [&] { claimOwnerOrThrow(ops.op, l, "r", UInt128(1), emptyCatalogObservation()); }); + EXPECT_TRUE(backend->fired); + EXPECT_EQ(backend->owner_reads, 2u); + } +} + +/// A remount re-anchors its lease BEFORE it arms the fence for the new incarnation, so the fence is +/// still latched lost at that moment. The steady-state renewal is refused there — the sibling test +/// above pins that — and the remount's own renewal has to be admitted off the fence, or the pool could +/// never re-anchor and the remount attempt would fail on exactly the throttled store that caused it. +TEST(CASMountLease, RemountRenewalIsAdmittedOffTheMountFence) +{ + auto backend = std::make_shared(); + Layout l("p"); + uint64_t now = 1000; + uint64_t boot = 0; + + CasRequests mount_requests(backend, Fence{ + [] { return uint64_t{0}; }, + [](uint64_t, uint64_t) { return Fence::Admit::LostOrRearmed; }, + [](uint64_t) { throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "mount fence lost"); }}); + mount_requests.setNowFnForTest([&boot] { return boot; }); + mount_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + CasRequests open_requests = openRequestsForTest(backend); + open_requests.setNowFnForTest([&boot] { return boot; }); + open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); + + MountLeaseRenewer renewer(mount_requests, open_requests, l, "r", UInt128(1), 7, + std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, + {}, std::chrono::milliseconds(0), [&] { return boot; }); + renewer.start(); + + const MountRenewResult redo = renewer.renewForRemount(); + EXPECT_EQ(redo.outcome, MountRenewOutcome::Committed); + + CasOperation reader = open_requests.admit(); + EXPECT_EQ(decodeMountLease(reader.read(l.mountKey("r"), Retry::standard())->bytes).seq, 2u); +} diff --git a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp index 42dae767213a..6eb8f9c9b7b8 100644 --- a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp +++ b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp @@ -10,19 +10,23 @@ extern const int ABORTED; using namespace DB::Cas; using DB::Cas::tests::MountSlotRaceBackend; using DB::Cas::tests::expectThrowsCodeWithMessage; +using DB::Cas::tests::OperationForTest; namespace { -/// One keeper for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. -MountLeaseKeeper makeKeeper( - const std::shared_ptr & backend, +/// One renewer for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. Both +/// of its planes are the same open-fence one: what these tests exercise is the mount protocol's own +/// exclusivity, not a fence's, and no test here renews, which is the only caller of the mount plane. +MountLeaseRenewer makeRenewer( + CasRequests & requests, uint64_t & now, DB::UInt128 uuid = DB::UInt128(1), uint64_t epoch = 7) { - return MountLeaseKeeper( - backend, + return MountLeaseRenewer( + requests, + requests, Layout("p"), "r", uuid, @@ -32,55 +36,37 @@ MountLeaseKeeper makeKeeper( [] { return uint64_t{0}; }); } -void markMountGcFenced(MountSlotRaceBackend & backend, const Layout & layout, const String & server_root_id) +void markMountGcFenced(CasOperation & op, const Layout & layout, const String & server_root_id) { const String key = layout.mountKey(server_root_id); - const auto got = backend.get(key); + const auto got = op.read(key, Retry::standard()); ASSERT_TRUE(got); MountLease lease = decodeMountLease(got->bytes); lease.gc_fenced = true; - const PutResult result = backend.putOverwrite(key, encodeMountLease(lease), got->token); - ASSERT_EQ(result.outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + op.replace(key, encodeMountLease(lease), got->etag, Retry::standard()))); } } -TEST(CASMountClaimConflicts, SlotAppearedBetweenHeadAndPutIfAbsent) +TEST(CASMountClaimConflicts, SlotAppearedBetweenTheReadAndTheCreate) { auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; - /// Empty at `head`; another process mints it before our `putIfAbsent` lands. + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); + /// Absent at the read; another process mints it before our create lands. backend->before_put_if_absent = [&] { - claimMount(*backend, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100); + CasOperation racer = requests.admit(); + claimMount(racer, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100); }; - auto keeper = makeKeeper(backend, now); + auto renewer = makeRenewer(requests, now); expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, - "appeared between head and putIfAbsent", - [&] { keeper.start(); }); -} - -TEST(CASMountClaimConflicts, SlotVanishedBetweenHeadAndGet) -{ - auto backend = std::make_shared(); - Layout layout("p"); - uint64_t now = 1000; - ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, - MountClaimResult::Claimed); - backend->before_get = [&] - { - const auto got = backend->get(layout.mountKey("r")); - ASSERT_TRUE(got); - backend->deleteExact(layout.mountKey("r"), got->token); - }; - auto keeper = makeKeeper(backend, now); - expectThrowsCodeWithMessage( - DB::ErrorCodes::ABORTED, - "vanished between head and get while claiming", - [&] { keeper.start(); }); + "appeared between the read and the create", + [&] { renewer.start(); }); } TEST(CASMountClaimConflicts, SlotHeldByForeignServer) @@ -88,14 +74,16 @@ TEST(CASMountClaimConflicts, SlotHeldByForeignServer) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - auto keeper = makeKeeper(backend, now); + auto renewer = makeRenewer(requests, now); expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, "held by a foreign server", - [&] { keeper.start(); }); + [&] { renewer.start(); }); } TEST(CASMountClaimConflicts, SlotHeldByDifferentWriterEpoch) @@ -103,14 +91,16 @@ TEST(CASMountClaimConflicts, SlotHeldByDifferentWriterEpoch) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - auto keeper = makeKeeper(backend, now, DB::UInt128(1), /*epoch=*/8); + auto renewer = makeRenewer(requests, now, DB::UInt128(1), /*epoch=*/8); expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, "held by a different writer_epoch", - [&] { keeper.start(); }); + [&] { renewer.start(); }); } TEST(CASMountClaimConflicts, SlotChangedInsideAdoptionWindow) @@ -118,19 +108,22 @@ TEST(CASMountClaimConflicts, SlotChangedInsideAdoptionWindow) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - /// Rewrite the slot under a NEW token after our `get`, so our adoption `putOverwrite` conflicts. + /// Rewrite the slot under a NEW incarnation after our read, so our adoption write is refused. backend->before_put_overwrite = [&] { - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now + 1, /*ttl_ms=*/100); + CasOperation racer = requests.admit(); + claimMount(racer, layout, "r", DB::UInt128(1), 7, now + 1, /*ttl_ms=*/100); }; - auto keeper = makeKeeper(backend, now); + auto renewer = makeRenewer(requests, now); expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, "changed while adopting our own mount slot", - [&] { keeper.start(); }); + [&] { renewer.start(); }); } TEST(CASMountClaimConflicts, SlotVanishedInsideAdoptionWindow) @@ -138,20 +131,21 @@ TEST(CASMountClaimConflicts, SlotVanishedInsideAdoptionWindow) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); backend->before_put_overwrite = [&] { - const auto got = backend->get(layout.mountKey("r")); - ASSERT_TRUE(got); - backend->deleteExact(layout.mountKey("r"), got->token); + CasOperation racer = requests.admit(); + ASSERT_EQ(racer.removeCurrent(layout.mountKey("r"), Retry::standard()), Removal::Removed); }; - auto keeper = makeKeeper(backend, now); + auto renewer = makeRenewer(requests, now); expectThrowsCodeWithMessage( DB::ErrorCodes::ABORTED, "vanished while adopting our own mount slot", - [&] { keeper.start(); }); + [&] { renewer.start(); }); } /// The two fenced branches keep their own type, and keep PRECEDENCE over the conflicts above: the @@ -162,12 +156,14 @@ TEST(CASMountClaimConflicts, FencedBeforeAdoptionRaisesMountFenced) auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - markMountGcFenced(*backend, layout, "r"); - auto keeper = makeKeeper(backend, now); - EXPECT_THROW(keeper.start(), MountFencedException); + markMountGcFenced(op, layout, "r"); + auto renewer = makeRenewer(requests, now); + EXPECT_THROW(renewer.start(), MountFencedException); } TEST(CASMountClaimConflicts, FencedInsideAdoptionWindowRaisesMountFencedNotAborted) @@ -175,12 +171,117 @@ TEST(CASMountClaimConflicts, FencedInsideAdoptionWindowRaisesMountFencedNotAbort auto backend = std::make_shared(); Layout layout("p"); uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ( - claimMount(*backend, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); /// The slot changes inside the adoption window AND the new body is fenced: the fenced branch must /// win over the "changed while adopting" one. - backend->before_put_overwrite = [&] { markMountGcFenced(*backend, layout, "r"); }; - auto keeper = makeKeeper(backend, now); - EXPECT_THROW(keeper.start(), MountFencedException); + backend->before_put_overwrite = [&] + { + CasOperation racer = requests.admit(); + markMountGcFenced(racer, layout, "r"); + }; + auto renewer = makeRenewer(requests, now); + EXPECT_THROW(renewer.start(), MountFencedException); +} + +/// A raced claim reports a body its caller renders into the fail-closed operator message. Reporting +/// the PROPOSER's own lease there names this very server as the existing mount, which sends an +/// operator hunting a second process that is not the one holding the slot. The write's own resolve +/// read already observed the occupant, so that is what the result must carry. +TEST(CASMountClaimConflicts, ALostCreateReportsTheOccupantNotTheProposer) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + /// Absent at our read; a foreign server mints the slot before our create lands. + backend->before_put_if_absent = [&] + { + CasOperation racer = requests.admit(); + ASSERT_EQ(claimMount(racer, layout, "r", DB::UInt128(2), 1, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + }; + CasOperation op = requests.admit(); + const MountClaimResult claim = claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100); + + EXPECT_EQ(claim.kind, MountClaimResult::LiveDoubleStart); + ASSERT_TRUE(claim.body.has_value()); + EXPECT_EQ(claim.body->server_uuid, DB::UInt128(2)) << "the result named this server's own proposal"; + EXPECT_EQ(claim.body->writer_epoch, 1u); + ASSERT_TRUE(claim.etag.has_value()) << "the observed occupant's incarnation is what was read"; + EXPECT_NE(mountDoubleStartMessage("r", claim.body).find(u128ToHex(DB::UInt128(2))), String::npos) + << "the operator message must name the foreign holder"; +} + +/// The same for the refresh branch: a body that changed under our own adoption is the one the message +/// must name. The reclaim branch reaches the identical helper, so it is not repeated here. +TEST(CASMountClaimConflicts, ALostRefreshReportsTheObservedBodyNotTheProposer) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation seed = requests.admit(); + ASSERT_EQ(claimMount(seed, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + /// A distinguishable body lands under our own refresh, so a result carrying the proposal cannot + /// pass by accident: our proposal would carry `seq` 2 and this process's pid. + backend->before_put_overwrite = [&] + { + OperationForTest racer(*backend); + const String key = layout.mountKey("r"); + const auto got = (*racer).read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + MountLease raced = decodeMountLease(got->bytes); + raced.pid = 4242; + raced.seq = 99; + ASSERT_TRUE(std::holds_alternative( + (*racer).replace(key, encodeMountLease(raced), got->etag, Retry::standard()))); + }; + CasOperation op = requests.admit(); + const MountClaimResult claim = claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100); + + EXPECT_EQ(claim.kind, MountClaimResult::LiveDoubleStart); + ASSERT_TRUE(claim.body.has_value()); + EXPECT_EQ(claim.body->pid, 4242); + EXPECT_EQ(claim.body->seq, 99u); + ASSERT_TRUE(claim.etag.has_value()); +} + +/// The residue of the same rule: a raced write whose conflict settles to no observation saw nobody, so +/// there is no holder to name. Reporting the lease this server merely PROPOSED would put this very +/// process in the "Existing mount" line of an operator message -- the same defect as naming it after a +/// conflict that did observe someone. `ProvenAbsent` is the reachable half; `NotObserved` (the resolve +/// read itself failed) leaves through the same branch. +TEST(CASMountClaimConflicts, ARacedRefreshThatObservedNothingNamesNoHolder) +{ + auto backend = std::make_shared(); + Layout layout("p"); + uint64_t now = 1000; + CasRequests requests = openRequestsForTest(backend); + CasOperation seed = requests.admit(); + ASSERT_EQ(claimMount(seed, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100).kind, + MountClaimResult::Claimed); + /// The slot is removed under our own refresh, so the refused precondition resolves to a proven + /// absence rather than to an occupant. + backend->before_put_overwrite = [&] + { + OperationForTest racer(*backend); + const String key = layout.mountKey("r"); + const auto got = (*racer).read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + ASSERT_EQ((*racer).remove(key, got->etag, Retry::standard()), Removal::Removed); + }; + CasOperation op = requests.admit(); + const MountClaimResult claim = claimMount(op, layout, "r", DB::UInt128(1), 7, now, /*ttl_ms=*/100); + + EXPECT_EQ(claim.kind, MountClaimResult::LiveDoubleStart); + EXPECT_FALSE(claim.etag.has_value()); + EXPECT_FALSE(claim.body.has_value()) << "a result that observed nobody reported a lease anyway"; + const String message = mountDoubleStartMessage("r", claim.body); + EXPECT_NE(message.find("could not be observed"), String::npos) + << "the message named a holder nobody saw: " << message; } diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp new file mode 100644 index 000000000000..0294fa178a35 --- /dev/null +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -0,0 +1,176 @@ +#include +#include +#include +#include +#include +#include + +#include +#include + +using namespace DB::Cas; + +namespace +{ + +/// A `CasMountRuntime` with nothing running on it: no renewer, no workers, an injected boot clock and a +/// fence the test arms by hand. Enough to exercise admission, which reads only the fence's own state. +class RuntimeFixture +{ +public: + explicit RuntimeFixture(uint64_t lease_safety_margin_ms, uint64_t attempt_timeout_ms = 10, + std::optional connect_timeout_cap_ms = std::nullopt) + : backend(std::make_shared()) + , mount(backend, Fence{ + [this] { return runtime.fenceGeneration(); }, + [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, + [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) + , farewell(backend, Fence::open()) + , runtime( + backend, mount, farewell, layout, + MountConfig{.boot_ms_fn = [this] { return boot_ms; }}, + "test", sink, + CasRequestBudget{.attempt_timeout_ms = attempt_timeout_ms, + .lease_safety_margin_ms = lease_safety_margin_ms, + .connect_timeout_cap_ms = connect_timeout_cap_ms}, + [] { return false; }) + { + } + + CasMountRuntime * operator->() { return &runtime; } + + uint64_t boot_ms = 1'000; + +private: + std::shared_ptr backend; + Layout layout{"mount-runtime-admit"}; + CasEventSink sink; + CasRequests mount; + CasRequests farewell; + CasMountRuntime runtime; +}; + +/// Named verdicts, so a failing expectation reads as the answer rather than as a raw byte. +const char * admitName(Fence::Admit verdict) +{ + switch (verdict) + { + case Fence::Admit::Ok: return "Ok"; + case Fence::Admit::LostOrRearmed: return "LostOrRearmed"; + case Fence::Admit::NoBudget: return "NoBudget"; + } + return "unknown"; +} + +constexpr DB::UInt128 kUuid{7}; + +} + +/// The boundary is STRICT on both terms: a request that would only just finish as the lease runs out +/// is one that may land after this node's fence is already gone. +TEST(CASMountRuntime, AdmitRefusesAtTheExactBudgetBoundary) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/20); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); /// 100 ms of lease left + const uint64_t generation = f->fenceGeneration(); + + EXPECT_STREQ(admitName(f->admit(generation, 80)), "NoBudget") << "needed + margin == remaining must refuse"; + EXPECT_STREQ(admitName(f->admit(generation, 79)), "Ok") << "one millisecond of slack is enough"; + EXPECT_STREQ(admitName(f->admit(generation, 100)), "NoBudget") << "needed == remaining must refuse"; +} + +/// The subtraction in `admit` exists for this: `needed_ms + margin` would wrap and read as room. +TEST(CASMountRuntime, AdmitDoesNotWrapOnAnAbsurdNeed) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/20); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), std::numeric_limits::max())), "NoBudget"); +} + +TEST(CASMountRuntime, AdmitRefusesAnExpiredLease) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/0); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + const uint64_t generation = f->fenceGeneration(); + + f.boot_ms = 1'099; + EXPECT_STREQ(admitName(f->admit(generation, 0)), "Ok"); + f.boot_ms = 1'100; + EXPECT_STREQ(admitName(f->admit(generation, 0)), "NoBudget") << "the deadline instant is already past"; + /// One millisecond further is what the `now >= deadline` guard actually earns: without it + /// `deadline - now` underflows to a huge remaining and the budget test reads it as room. + f.boot_ms = 1'101; + EXPECT_STREQ(admitName(f->admit(generation, 0)), "NoBudget") + << "a deadline already past must not underflow into room"; +} + +/// A re-arm is a fresh lease incarnation. A caller admitted under the previous one is stale even though +/// the fence is live again, which is the whole point of carrying a generation. +TEST(CASMountRuntime, AdmitRefusesAGenerationTheFenceMovedPast) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/0); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/100'000); + const uint64_t stale = f->fenceGeneration(); + f->armMountFence(kUuid, 2, /*deadline_boot_ms=*/100'000); + + EXPECT_STREQ(admitName(f->admit(stale, 0)), "LostOrRearmed"); + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 0)), "Ok"); +} + +/// The latch, isolated from the generation bump that accompanies it: the generation presented here is +/// the one the trip itself produced, so only `lost` can be refusing. +TEST(CASMountRuntime, AdmitRefusesALostFenceWhateverTheBudget) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/0); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/100'000); + f->tripMountLost(); + + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 0)), "LostOrRearmed"); +} + +/// The unarmed default (no lease deadline yet) permits work: the bootstrap-control writes that claim a +/// lease run before there is one to be gated on. +TEST(CASMountRuntime, AdmitAllowsAnUnarmedFence) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/2'000); + f.boot_ms = 1'000; + + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 5'000)), "Ok"); +} + +/// `refAppendFenceOk` is `admit` at one attempt's worth of budget under the live generation. +TEST(CASMountRuntime, RefAppendFenceOkIsAdmitAtTwoEnvelopes) +{ + /// connect_timeout_cap_ms is nullopt (see RuntimeFixture), so the envelope equals the bare attempt + /// timeout (10 ms); refAppendFenceOk asks for TWO of them (a write and its settlement read). + RuntimeFixture f(/*lease_safety_margin_ms=*/20, /*attempt_timeout_ms=*/10); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'041); /// 41 ms left: one more than 2*10 + 20 + EXPECT_TRUE(f->refAppendFenceOk()); + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 20)), "Ok"); + + f->setMountDeadline(1'040); /// exactly 2*10 + 20 left + EXPECT_FALSE(f->refAppendFenceOk()); + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 20)), "NoBudget"); +} + +/// Same boundary, with a nonzero connect cap so the envelope's connect contribution (not just the +/// doubling) is pinned: attempt 100, cap 50 -> envelope 200, refAppendFenceOk asks for 2*200 = 400. +TEST(CASMountRuntime, RefAppendFenceOkIsAdmitAtTwoEnvelopesWithANonzeroCap) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/20, /*attempt_timeout_ms=*/100, /*connect_timeout_cap_ms=*/50); + f.boot_ms = 1'000; + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'421); /// 421 ms left: one more than 2*200 + 20 + EXPECT_TRUE(f->refAppendFenceOk()); + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 400)), "Ok"); + + f->setMountDeadline(1'420); /// exactly 2*200 + 20 left + EXPECT_FALSE(f->refAppendFenceOk()); + EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 400)), "NoBudget"); +} diff --git a/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp b/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp index bfb62f817ecb..4f5e17c1931f 100644 --- a/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp +++ b/src/Disks/tests/gtest_cas_namespace_file_request_profile.cpp @@ -29,7 +29,7 @@ namespace DB::ContentAddressedSetting /// /// WHAT THIS FILE PINS: the last clause, per key. The counts below were READ OFF this tree before any /// key change and pasted as literals, which is the whole point of the file -- expectations re-derived -/// after a change measure the change against itself. Incarnation qualification changes the KEY a +/// after a change measure the change against itself. Etag qualification changes the KEY a /// namespace file is stored under, so the keys are derived from `Layout` rather than spelled out; what /// must not move is the count per key and the set of keys touched. /// @@ -98,24 +98,25 @@ TEST(CASNamespaceFileRequestProfile, CreateThenRewrite) store->putNamespaceFile(life, kFile, "1\n"); EXPECT_EQ(backend->headCount(key), 1u); - EXPECT_EQ(backend->putCount(key), 1u); /// putIfAbsent -- the key was absent + EXPECT_EQ(backend->putCount(key), 1u); /// create-shaped -- the key was absent EXPECT_EQ(backend->putOverwriteCount(key), 0u); EXPECT_EQ(backend->getCount(key), 0u); EXPECT_EQ(backend->deleteCount(key), 0u); EXPECT_EQ(backend->listTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + /// And no write beyond the one accounted for above, anywhere. + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->touchedKeys(), std::vector{key}); backend->resetCounts(); store->putNamespaceFile(life, kFile, "2\n"); EXPECT_EQ(backend->headCount(key), 1u); - EXPECT_EQ(backend->putOverwriteCount(key), 1u); /// token-conditioned replacement -- it existed + EXPECT_EQ(backend->putOverwriteCount(key), 1u); /// replace-shaped -- it existed EXPECT_EQ(backend->putCount(key), 0u); EXPECT_EQ(backend->getCount(key), 0u); EXPECT_EQ(backend->deleteCount(key), 0u); EXPECT_EQ(backend->listTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->touchedKeys(), std::vector{key}); } @@ -132,8 +133,9 @@ TEST(CASNamespaceFileRequestProfile, Read) EXPECT_EQ(store->getNamespaceFile(life, kFile), String("1\n")); + /// One GET, and it is necessarily a whole-object one: `Backend::get` refuses a non-whole window + /// outright, so there is no partial read left for a separate counter to tell apart. EXPECT_EQ(backend->getCount(key), 1u); - EXPECT_EQ(backend->wholeGetCount(key), 1u); EXPECT_EQ(backend->headCount(key), 0u); EXPECT_EQ(backend->putCount(key), 0u); EXPECT_EQ(backend->putOverwriteCount(key), 0u); @@ -218,7 +220,7 @@ TEST(CASNamespaceFileRequestProfile, DedupLogRotation) EXPECT_EQ(backend->headCount(old_key), 1u); EXPECT_EQ(backend->deleteCount(old_key), 1u); EXPECT_EQ(backend->getTotal(), 0u); /// rotation reads no body - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 1u); /// and writes only the new segment /// Sorted, and the files prefix is a proper prefix of both segment keys, so it comes first. EXPECT_EQ(backend->touchedKeys(), (std::vector{prefix, old_key, new_key})); @@ -500,8 +502,8 @@ TEST(CASNamespaceFileDiskProfile, TheLifeResolutionIsPaidOncePerTableOpen) } -/// THE REMOVAL PATHS MUST NOT CREATE A NAMESPACE — the case that regressed silently in this task's first -/// round, so it is pinned on the catalog rather than on the file outcome. +/// THE REMOVAL PATHS MUST NOT CREATE A NAMESPACE: the catalog, rather than the file outcome, proves +/// that invariant. /// /// Why the file outcome cannot pin it: `unlinkFile`/`removeRecursive` against a never-opened table /// answer "absent" both before and after the defect, because a freshly minted namespace has no files @@ -517,7 +519,8 @@ TEST(CASNamespaceFileDiskProfile, RemovalOnANeverOpenedTableLeavesTheCatalogUnto /// A valid pool already owns its explicit empty mandatory catalog. Nothing has opened this table: /// no namespace file written, no part published, and no ref operation has changed that object. - const auto catalog_before = storage->store()->backend().get(layout.refCatalogKey()); + OperationForTest catalog_probe(storage->store()->poolBackendPtr()); + const auto catalog_before = (*catalog_probe).read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog_before); EXPECT_TRUE(decodeRefCatalog(catalog_before->bytes).entries.empty()); object_storage->resetRecords(); @@ -536,10 +539,10 @@ TEST(CASNamespaceFileDiskProfile, RemovalOnANeverOpenedTableLeavesTheCatalogUnto EXPECT_EQ(object_storage->writtenContaining(layout.refCatalogKey()), std::vector{}) << "a removal must not write the catalog: it must not birth the namespace it is removing from"; - const auto catalog_after_removal = storage->store()->backend().get(layout.refCatalogKey()); + const auto catalog_after_removal = (*catalog_probe).read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog_after_removal); EXPECT_EQ(catalog_after_removal->bytes, catalog_before->bytes); - EXPECT_EQ(catalog_after_removal->token, catalog_before->token) + EXPECT_EQ(catalog_after_removal->etag, catalog_before->etag) << "the mandatory catalog must remain byte-for-byte and token-for-token unchanged"; /// Not vacuous: the SAME operations on the same table after a write do reach the file, so the zeros @@ -551,8 +554,8 @@ TEST(CASNamespaceFileDiskProfile, RemovalOnANeverOpenedTableLeavesTheCatalogUnto EXPECT_FALSE(storage->existsFile(kTablePath + "/format_version.txt")); /// Positive control: the write really did birth the namespace and mutate the same catalog object /// whose stability the removal assertions pin above. - const auto catalog_after_birth = storage->store()->backend().get(layout.refCatalogKey()); + const auto catalog_after_birth = (*catalog_probe).read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog_after_birth); EXPECT_NE(catalog_after_birth->bytes, catalog_after_removal->bytes); - EXPECT_NE(catalog_after_birth->token, catalog_after_removal->token); + EXPECT_NE(catalog_after_birth->etag, catalog_after_removal->etag); } diff --git a/src/Disks/tests/gtest_cas_namespace_janitor.cpp b/src/Disks/tests/gtest_cas_namespace_janitor.cpp index 7a3455738bc4..0ada8882e7b0 100644 --- a/src/Disks/tests/gtest_cas_namespace_janitor.cpp +++ b/src/Disks/tests/gtest_cas_namespace_janitor.cpp @@ -5,59 +5,94 @@ using namespace DB::Cas; using namespace DB::Cas::tests; +namespace DB::ErrorCodes +{ + extern const int NETWORK_ERROR; +} + namespace { +/// `readGcMaintenanceState` now takes an admitted `CasOperation`, which cannot bind to an rvalue: every +/// call site below goes through this helper rather than materializing its own throwaway operation. +GcMaintenanceReadResult readState(CasRequests & requests, const Layout & layout) +{ + auto op = requests.admit(); + return readGcMaintenanceState(op, layout); +} + class OrderedJanitorBackend : public CountingBackend { public: - using CountingBackend::get; std::vector events; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { if (prefix.ends_with("/cas/ns/")) events.push_back("list"); - return CountingBackend::list(prefix, cursor, limit); + return CountingBackend::list(prefix, cursor, limit, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (key.ends_with("/cas/ref_catalog")) events.push_back("catalog"); - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); } }; class OmitFirstNamespacePageBackend : public CountingBackend { public: - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { if (omit && prefix.ends_with("/cas/ns/")) { omit = false; return {}; } - return CountingBackend::list(prefix, cursor, limit); + return CountingBackend::list(prefix, cursor, limit, access); } private: bool omit = true; }; +/// Flips `delete_done` right after its one REMOVE returns, so a liveness predicate closing over it stays +/// true through every request up to and including that delete -- whatever their number or order -- and +/// only refuses the very next one. `liveness` is sampled before every request now, so driving a fence +/// loss at an exact point robustly (rather than by counting samples, which would couple this test to how +/// many reads the catalog snapshot happens to take) means keying it to an observable EVENT instead. +class FlipAfterFirstDeleteBackend : public CountingBackend +{ +public: + DB::Cas::Backend::RawRemoval remove(const String & key, const String & expected_value, + DB::Cas::TransportAccess & access) override + { + DB::Cas::Backend::RawRemoval outcome = CountingBackend::remove(key, expected_value, access); + delete_done = true; + return outcome; + } + bool delete_done = false; +}; + class ReplaceBeforeJanitorDeleteBackend : public CountingBackend { public: - DeleteOutcome deleteExact(const String & key, const Token & token) override + DB::Cas::Backend::RawRemoval remove(const String & key, const String & expected_value, + DB::Cas::TransportAccess & access) override { if (!replaced) { replaced = true; - const auto current = InMemoryBackend::get(key); + /// The qualified primitive, exactly as the sibling concurrent-actor doubles in this file: a + /// simulated concurrent write must not be counted as the janitor's own. + const auto current = InMemoryBackend::read(key, access); if (current) - (void)InMemoryBackend::casPut(key, "winner", current->token); + (void)InMemoryBackend::write(key, "winner", current->value, access); } - return CountingBackend::deleteExact(key, token); + return CountingBackend::remove(key, expected_value, access); } private: bool replaced = false; @@ -66,23 +101,28 @@ class ReplaceBeforeJanitorDeleteBackend : public CountingBackend class TokenlessListBackend : public CountingBackend { public: - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { - ListPage page = CountingBackend::list(prefix, cursor, limit); - for (ListedKey & key : page.keys) - key.token.reset(); + DB::Cas::Backend::RawListPage page = CountingBackend::list(prefix, cursor, limit, access); + for (auto & key : page.keys) + key.value.reset(); return page; } bool supportsListTokens() const override { return false; } - HeadResult head(const String & key) override + std::optional head(const String & key, DB::Cas::TransportAccess & access) override { - HeadResult result = CountingBackend::head(key); - if (!replaced && result.exists && key == replace_on_head) + std::optional result = CountingBackend::head(key, access); + if (!replaced && result && key == replace_on_head) { replaced = true; - (void)InMemoryBackend::casPut(key, "winner", result.token); + /// The qualified primitive `write` -- not the legacy `head`/`casPut` convenience pair -- so + /// this simulated concurrent actor neither re-enters the counted `head` override (the legacy + /// forwarder calls back through the virtual primitive) nor is itself counted as a write the + /// janitor made. + (void)InMemoryBackend::write(key, "winner", result->value, access); } return result; } @@ -96,9 +136,9 @@ class TokenlessListBackend : public CountingBackend class FenceLossDuringHeadBackend : public TokenlessListBackend { public: - HeadResult head(const String & key) override + std::optional head(const String & key, DB::Cas::TransportAccess & access) override { - HeadResult result = TokenlessListBackend::head(key); + std::optional result = TokenlessListBackend::head(key, access); fence_held = false; return result; } @@ -111,23 +151,26 @@ class CatalogAfterListBackend : public CountingBackend public: explicit CatalogAfterListBackend(NamespaceLifeId life_) : protected_life(std::move(life_)) {} - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { - ListPage page = CountingBackend::list(prefix, cursor, limit); + DB::Cas::Backend::RawListPage page = CountingBackend::list(prefix, cursor, limit, access); if (!published && prefix.ends_with("/cas/ns/")) { published = true; const String catalog_key = "p/cas/ref_catalog"; - /// This models a CONCURRENT actor's read, not the janitor's own -- counting it here would - /// make `PostListCatalogCutProtectsConcurrentCreationWithOneGet`'s "exactly one get" assertion - /// count this simulated actor's read as the janitor's, defeating the point of that assertion. - const auto current = InMemoryBackend::get(catalog_key, {}); // NOLINT(bugprone-parent-virtual-call) + /// This models a CONCURRENT actor's read, not the janitor's own. It must go through the + /// qualified PRIMITIVE, not the virtual `read`/`write` this class's base counts: the janitor's + /// own catalog read reaches the store through that same virtual dispatch, and a call routed + /// through it here would be indistinguishable from the janitor's -- doubling the count + /// `PostListCatalogCutProtectsConcurrentCreationWithOneGet` asserts is exactly one. + const auto current = InMemoryBackend::read(catalog_key, access); // NOLINT(bugprone-parent-virtual-call) if (current) { RefCatalog catalog; catalog.entries.push_back(CatalogEntry{.ns = protected_life.ns, .state = NsState::Live, .incarnation = protected_life.incarnation}); - (void)InMemoryBackend::casPut(catalog_key, encodeRefCatalog(catalog), current->token); + (void)InMemoryBackend::write(catalog_key, encodeRefCatalog(catalog), current->value, access); } } return page; @@ -140,30 +183,71 @@ class CatalogAfterListBackend : public CountingBackend class RejectCursorBackend : public CountingBackend { public: - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { if (prefix.ends_with("/cas/ns/") && !cursor.empty()) throw std::runtime_error("backend rejected cursor"); - return CountingBackend::list(prefix, cursor, limit); + return CountingBackend::list(prefix, cursor, limit, access); } }; class FailMaintenancePublicationBackend : public CountingBackend { public: - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (fail_publication && key.ends_with("/gc/maintenance_state")) throw std::runtime_error("maintenance publication failed"); - return CountingBackend::casPut(key, bytes, expected, meta); + return CountingBackend::write(key, bytes, expected_value, access); } bool fail_publication = false; }; -void seedCatalog(CountingBackend & backend, const Layout & layout, RefCatalog catalog = {}) +/// The catch-path reset in `NamespaceJanitor::runOnePage` fires on any LIST failure. Both faults here +/// throw a `DB::Exception` classified `NETWORK_ERROR`: a `std::runtime_error` is not a `Poco::Exception`, +/// so `CasOperation`'s engine treats it as an unmodeled local bug and surfaces it immediately on every +/// path (read or write) without ever reaching the ambiguity-resolving machinery this test needs -- a +/// `NETWORK_ERROR` is a genuine transient-looking store answer instead. The write additionally counts +/// its own attempts (`CountingBackend::writeTotal()` stays 0 here: this override throws before ever +/// delegating to the base `write`). +class ThrowingListAndAmbiguousWriteBackend : public CountingBackend +{ +public: + DB::Cas::Backend::RawListPage list(const String &, const String &, size_t, + DB::Cas::TransportAccess &) override + { + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "list failed"); + } + std::expected write(const String &, const String &, + const std::optional &, + DB::Cas::TransportAccess &) override + { + ++write_attempts; + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "ambiguous write"); + } + uint64_t write_attempts = 0; +}; + +/// A one-shot `create`, asserting it committed (mirrors the retired `backend->putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// An exact read (mirrors the retired `backend->get(key)`). +std::optional readObj(Backend & backend, const String & key) { - ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(catalog)).outcome, PutOutcome::Done); + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +void seedCatalog(Backend & backend, const Layout & layout, RefCatalog catalog = {}) +{ + createObj(backend, layout.refCatalogKey(), encodeRefCatalog(catalog)); } NamespaceLifeId life(const char * name, uint64_t id) @@ -176,32 +260,34 @@ NamespaceLifeId life(const char * name, uint64_t id) TEST(CASNamespaceJanitor, DeletesDeadFilesAndCheckpointFromOnePostListCatalogCut) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const auto dead = life("dead", 41); const String file = layout.namespaceFilesPrefix(dead) + "part/data.bin"; const String ckpt = layout.refCkptKey(dead); - ASSERT_EQ(backend.putIfAbsent(file, "file-bytes").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(ckpt, "ckpt-bytes").outcome, PutOutcome::Done); - backend.resetCounts(); + createObj(*backend, file, "file-bytes"); + createObj(*backend, ckpt, "ckpt-bytes"); + backend->resetCounts(); - NamespaceJanitor janitor(backend, layout, 100); + NamespaceJanitor janitor(requests, layout, 100); const NamespaceJanitorResult result = janitor.runOnePage(false, [] { return true; }); EXPECT_EQ(result.pages, 1u); EXPECT_EQ(result.keys, 2u); EXPECT_EQ(result.deleted, 2u); - EXPECT_FALSE(backend.get(file)); - EXPECT_FALSE(backend.get(ckpt)); - EXPECT_EQ(backend.listCount(layout.namespaceRootPrefix()), 1u); - EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); - EXPECT_EQ(readGcMaintenanceState(backend, layout).state, GcMaintenanceState{}); + EXPECT_FALSE(readObj(*backend, file).has_value()); + EXPECT_FALSE(readObj(*backend, ckpt).has_value()); + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(readState(requests, layout).state, GcMaintenanceState{}); } TEST(CASNamespaceJanitor, RetainsEveryCurrentLifecycleAndSuppressesAmbiguousCut) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); RefCatalog catalog; CatalogEntry creating{.ns = RootNamespace{"creating"}, .state = NsState::Creating, .incarnation = UInt128{51}, @@ -210,207 +296,222 @@ TEST(CASNamespaceJanitor, RetainsEveryCurrentLifecycleAndSuppressesAmbiguousCut) CatalogEntry removing{.ns = RootNamespace{"removing"}, .state = NsState::Removing, .incarnation = UInt128{53}, .removal_started_round = 1}; catalog.entries = {creating, live, removing}; - seedCatalog(backend, layout, catalog); + seedCatalog(*backend, layout, catalog); for (const auto & entry : catalog.entries) - ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey( - NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation)), "keep").outcome, PutOutcome::Done); + createObj(*backend, layout.refCkptKey( + NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation)), "keep"); - NamespaceJanitor janitor(backend, layout, 100); + NamespaceJanitor janitor(requests, layout, 100); const auto result = janitor.runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); } TEST(CASNamespaceJanitor, CatalogFirstCreatingRetainsEveryObjectOfTheNewLife) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const CatalogEntry creating{ .ns = RootNamespace{"catalog-first"}, .state = NsState::Creating, .incarnation = UInt128{54}, .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 2, .fence_generation = 3}}; - seedCatalog(backend, layout, RefCatalog{.entries = {creating}}); + seedCatalog(*backend, layout, RefCatalog{.entries = {creating}}); /// The production creation order is the point: the catalog row is durable before either object. const NamespaceLifeId creating_life = NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation); const String ckpt = layout.refCkptKey(creating_life); const String file = layout.namespaceFilesPrefix(creating_life) + "data"; - ASSERT_EQ(backend.putIfAbsent(ckpt, "checkpoint").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(file, "file").outcome, PutOutcome::Done); - backend.resetCounts(); + createObj(*backend, ckpt, "checkpoint"); + createObj(*backend, file, "file"); + backend->resetCounts(); const NamespaceJanitorResult result - = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.deleteTotal(), 0u); - EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); - EXPECT_TRUE(backend.get(ckpt)); - EXPECT_TRUE(backend.get(file)); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); + EXPECT_TRUE(readObj(*backend, ckpt).has_value()); + EXPECT_TRUE(readObj(*backend, file).has_value()); } TEST(CASNamespaceJanitor, CancelledCreatingCheckpointIsReclaimedThroughPublicLifecycle) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const CatalogEntry creating{ .ns = RootNamespace{"cancelled"}, .state = NsState::Creating, .incarnation = UInt128{55}, .creator = CreatorFence{.server_root_id = "dead-srv", .writer_epoch = 4, .fence_generation = 5}}; - seedCatalog(backend, layout, RefCatalog{.entries = {creating}}); + seedCatalog(*backend, layout, RefCatalog{.entries = {creating}}); const String ckpt = layout.refCkptKey( NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation)); - ASSERT_EQ(backend.putIfAbsent(ckpt, "cancelled-checkpoint").outcome, PutOutcome::Done); + createObj(*backend, ckpt, "cancelled-checkpoint"); + auto cancel_op = requests.admit(); ASSERT_EQ(CasRefCatalog::cancelStalledCreating( - backend, layout, creating, [](const CreatorFence &) { return true; }, - /*admitted_generation=*/7, [](uint64_t) {}), + cancel_op, layout, creating, [](const CreatorFence &) { return true; }), CasRefCatalog::StalledCreatingCancelOutcome::Cancelled); - EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); + auto read_op = requests.admit(); + EXPECT_TRUE(CasRefCatalog::read(read_op, layout).catalog.entries.empty()); const NamespaceJanitorResult result - = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); - EXPECT_FALSE(backend.get(ckpt)); + EXPECT_FALSE(readObj(*backend, ckpt).has_value()); } TEST(CASNamespaceJanitor, SuppressionAndFenceLossDeleteNothing) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String first = layout.refCkptKey(life("dead-a", 61)); const String second = layout.refCkptKey(life("dead-b", 62)); - ASSERT_EQ(backend.putIfAbsent(first, "first").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(second, "second").outcome, PutOutcome::Done); + createObj(*backend, first, "first"); + createObj(*backend, second, "second"); + + /// The seeding above (the catalog + the two checkpoints) lands through the same write primitive + /// CountingBackend counts, so reset before measuring what the suppressed page itself does. + backend->resetCounts(); - NamespaceJanitor janitor(backend, layout, 1); + NamespaceJanitor janitor(requests, layout, 1); EXPECT_EQ(janitor.runOnePage(true, [] { return true; }).deleted, 0u); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "a globally suppressed page is undecided and must not mint cleanup progress"; - EXPECT_EQ(backend.putCount(layout.gcMaintenanceStateKey()), 0u); - EXPECT_EQ(backend.casPutCount(layout.gcMaintenanceStateKey()), 0u); - EXPECT_EQ(janitor.runOnePage(false, [] { return false; }).deleted, 0u); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + EXPECT_EQ(backend->writeTotal(), 0u); + + /// `liveness` is sampled before every request the page makes, starting with the maintenance read + /// itself -- a sample false from the start therefore ends the call by exception rather than by a + /// quiet no-op result. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { (void)janitor.runOnePage(false, [] { return false; }); }); + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "fence loss must not mint progress past a page whose deletion was not authorized"; - EXPECT_TRUE(backend.get(first)); - EXPECT_TRUE(backend.get(second)); - EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_TRUE(readObj(*backend, first).has_value()); + EXPECT_TRUE(readObj(*backend, second).has_value()); + EXPECT_EQ(backend->deleteTotal(), 0u); } TEST(CASNamespaceJanitor, FenceLossOnRetainedOnlyPageDoesNotAdvanceCursor) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const CatalogEntry current{ .ns = RootNamespace{"current"}, .state = NsState::Live, .incarnation = UInt128{63}}; - seedCatalog(backend, layout, RefCatalog{.entries = {current}}); + seedCatalog(*backend, layout, RefCatalog{.entries = {current}}); const NamespaceLifeId current_life = NamespaceLifeId::fromCatalogEntry(current.ns, current.incarnation); const String ckpt = layout.refCkptKey(current_life); const String file = layout.namespaceFilesPrefix(current_life) + "data"; - ASSERT_EQ(backend.putIfAbsent(ckpt, "checkpoint").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(file, "file").outcome, PutOutcome::Done); - - const NamespaceJanitorResult result - = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return false; }); - - EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.deleteTotal(), 0u); - EXPECT_TRUE(backend.get(ckpt)); - EXPECT_TRUE(backend.get(file)); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + createObj(*backend, ckpt, "checkpoint"); + createObj(*backend, file, "file"); + + /// A liveness sample false from the start is refused at the maintenance read, before the page ever + /// gets to examine an object -- retained-only or not; the page ends by exception. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { (void)NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return false; }); }); + + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_TRUE(readObj(*backend, ckpt).has_value()); + EXPECT_TRUE(readObj(*backend, file).has_value()); + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "a tenure that observes fence loss cannot publish progress even when every object was retained"; } TEST(CASNamespaceJanitor, FenceLossAfterLastDeleteRetainsCursorWithoutRollingBackDelete) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead = layout.refCkptKey(life("dead-after-delete", 64)); - ASSERT_EQ(backend.putIfAbsent(dead, "dead").outcome, PutOutcome::Done); - uint64_t fence_checks = 0; + createObj(*backend, dead, "dead"); - const NamespaceJanitorResult result - = NamespaceJanitor(backend, layout, 1).runOnePage(false, [&] { return fence_checks++ == 0; }); + const NamespaceJanitorResult result = NamespaceJanitor(requests, layout, 1).runOnePage( + false, [&] { return !backend->delete_done; }); EXPECT_EQ(result.deleted, 1u); - EXPECT_FALSE(backend.get(dead)) + EXPECT_FALSE(readObj(*backend, dead).has_value()) << "the exact delete completed under the fence and is never rolled back"; - EXPECT_EQ(fence_checks, 2u) - << "the fence must be checked before deletion and again immediately before cursor publication"; - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "losing the fence after the delete keeps this page selected for an idempotent retry"; } TEST(CASNamespaceJanitor, CursorResumesThenResetsAtEnd) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const auto dead = life("dead", 71); - ASSERT_EQ(backend.putIfAbsent(layout.namespaceFilesPrefix(dead) + "a", "a").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(layout.namespaceFilesPrefix(dead) + "b", "b").outcome, PutOutcome::Done); + createObj(*backend, layout.namespaceFilesPrefix(dead) + "a", "a"); + createObj(*backend, layout.namespaceFilesPrefix(dead) + "b", "b"); - NamespaceJanitor first_process(backend, layout, 1); + NamespaceJanitor first_process(requests, layout, 1); EXPECT_EQ(first_process.runOnePage(false, [] { return true; }).deleted, 1u); - const auto mid = readGcMaintenanceState(backend, layout); + const auto mid = readState(requests, layout); ASSERT_EQ(mid.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(mid.state); EXPECT_FALSE(mid.state->janitor_cursor.empty()); - NamespaceJanitor restarted_process(backend, layout, 1); + NamespaceJanitor restarted_process(requests, layout, 1); EXPECT_EQ(restarted_process.runOnePage(false, [] { return true; }).deleted, 1u); - EXPECT_TRUE(readGcMaintenanceState(backend, layout).state->janitor_cursor.empty()); + EXPECT_TRUE(readState(requests, layout).state->janitor_cursor.empty()); } TEST(CASNamespaceJanitor, TakesOneCatalogCutAfterListingAndContinuesPastMalformedKey) { - OrderedJanitorBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const auto dead = life("dead", 81); const String valid = layout.namespaceFilesPrefix(dead) + "data"; const String malformed = layout.namespaceStreamRootPrefix() + "not-a-life/_log/1-1.zst"; const String malformed_state = layout.namespaceStateRootPrefix() + "not-a-life/_ckpt"; - ASSERT_EQ(backend.putIfAbsent(valid, "v").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(malformed, "bad").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(malformed_state, "bad-state").outcome, PutOutcome::Done); - backend.resetCounts(); - backend.events.clear(); + createObj(*backend, valid, "v"); + createObj(*backend, malformed, "bad"); + createObj(*backend, malformed_state, "bad-state"); + backend->resetCounts(); + backend->events.clear(); - const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_FALSE(result.anomalies.empty()); - EXPECT_TRUE(backend.get(malformed)); - EXPECT_TRUE(backend.get(malformed_state)); - ASSERT_EQ(backend.events.size(), 2u); - EXPECT_EQ(backend.events[0], "list"); - EXPECT_EQ(backend.events[1], "catalog"); - EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); + EXPECT_TRUE(readObj(*backend, malformed).has_value()); + EXPECT_TRUE(readObj(*backend, malformed_state).has_value()); + ASSERT_EQ(backend->events.size(), 2u); + EXPECT_EQ(backend->events[0], "list"); + EXPECT_EQ(backend->events[1], "catalog"); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); } TEST(CASNamespaceJanitor, MalformedKeyIsFinalAndAdvancesCursor) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String first = layout.namespaceStreamRootPrefix() + "bad-a/_log/1-1.zst"; const String second = layout.namespaceStreamRootPrefix() + "bad-b/_log/1-1.zst"; - ASSERT_EQ(backend.putIfAbsent(first, "first").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(second, "second").outcome, PutOutcome::Done); + createObj(*backend, first, "first"); + createObj(*backend, second, "second"); const NamespaceJanitorResult result - = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_FALSE(result.anomalies.empty()); - EXPECT_TRUE(backend.get(first)); - EXPECT_TRUE(backend.get(second)); - const GcMaintenanceReadResult progress = readGcMaintenanceState(backend, layout); + EXPECT_TRUE(readObj(*backend, first).has_value()); + EXPECT_TRUE(readObj(*backend, second).has_value()); + const GcMaintenanceReadResult progress = readState(requests, layout); ASSERT_EQ(progress.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(progress.state); EXPECT_FALSE(progress.state->janitor_cursor.empty()) @@ -419,58 +520,61 @@ TEST(CASNamespaceJanitor, MalformedKeyIsFinalAndAdvancesCursor) TEST(CASNamespaceJanitor, DuplicateCurrentLifeSuppressesWholePage) { - CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); RefCatalog catalog; catalog.entries = { CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128{91}}, CatalogEntry{.ns = RootNamespace{"b"}, .state = NsState::Live, .incarnation = UInt128{91}}}; - seedCatalog(backend, layout, catalog); + seedCatalog(*backend, layout, catalog); const String dead_a = layout.refCkptKey(life("dead-a", 92)); const String dead_b = layout.refCkptKey(life("dead-b", 93)); - ASSERT_EQ(backend.putIfAbsent(dead_a, "a").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(dead_b, "b").outcome, PutOutcome::Done); - const auto result = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + createObj(*backend, dead_a, "a"); + createObj(*backend, dead_b, "b"); + const auto result = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.deleteTotal(), 0u); - EXPECT_TRUE(backend.get(dead_a)); - EXPECT_TRUE(backend.get(dead_b)); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Absent) + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_TRUE(readObj(*backend, dead_a).has_value()); + EXPECT_TRUE(readObj(*backend, dead_b).has_value()); + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "an ambiguous catalog cut leaves the selected page undecided for an authoritative retry"; } TEST(CASNamespaceJanitor, CorruptProgressResetsWithoutDeletingAndFilesOnlyOmittedCycleRetries) { - OmitFirstNamespacePageBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead = layout.namespaceFilesPrefix(life("dead", 101)) + "only-residue"; - ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(layout.gcMaintenanceStateKey(), "corrupt").outcome, PutOutcome::Done); - EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); - EXPECT_TRUE(backend.get(dead)); - EXPECT_EQ(readGcMaintenanceState(backend, layout).status, GcMaintenanceReadStatus::Valid); - EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); - EXPECT_TRUE(backend.get(dead)); - EXPECT_EQ(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }).deleted, 1u); - EXPECT_FALSE(backend.get(dead)); + createObj(*backend, dead, "bytes"); + createObj(*backend, layout.gcMaintenanceStateKey(), "corrupt"); + EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_TRUE(readObj(*backend, dead).has_value()); + EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Valid); + EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_TRUE(readObj(*backend, dead).has_value()); + EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_FALSE(readObj(*backend, dead).has_value()); } TEST(CASNamespaceJanitor, ExactTokenMismatchRetainsConcurrentReplacement) { - ReplaceBeforeJanitorDeleteBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead = layout.refCkptKey(life("dead-a", 111)); const String later = layout.refCkptKey(life("dead-b", 112)); - ASSERT_EQ(backend.putIfAbsent(dead, "old").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(later, "later").outcome, PutOutcome::Done); - const auto result = NamespaceJanitor(backend, layout, 1).runOnePage(false, [] { return true; }); + createObj(*backend, dead, "old"); + createObj(*backend, later, "later"); + const auto result = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); - ASSERT_TRUE(backend.get(dead)); - EXPECT_EQ(backend.get(dead)->bytes, "winner"); - EXPECT_TRUE(backend.get(later)); - const GcMaintenanceReadResult progress = readGcMaintenanceState(backend, layout); + ASSERT_TRUE(readObj(*backend, dead).has_value()); + EXPECT_EQ(readObj(*backend, dead)->bytes, "winner"); + EXPECT_TRUE(readObj(*backend, later).has_value()); + const GcMaintenanceReadResult progress = readState(requests, layout); ASSERT_EQ(progress.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(progress.state); EXPECT_FALSE(progress.state->janitor_cursor.empty()) @@ -479,99 +583,125 @@ TEST(CASNamespaceJanitor, ExactTokenMismatchRetainsConcurrentReplacement) TEST(CASNamespaceJanitor, TokenlessListHeadsDeadKeysAndRetainsConcurrentReplacement) { - TokenlessListBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); const CatalogEntry current{ .ns = RootNamespace{"current"}, .state = NsState::Live, .incarnation = UInt128{161}}; - seedCatalog(backend, layout, RefCatalog{.entries = {current}}); + seedCatalog(*backend, layout, RefCatalog{.entries = {current}}); const String live_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(current.ns, current.incarnation)); const String dead_key = layout.refCkptKey(life("dead", 162)); const String raced_key = layout.namespaceFilesPrefix(life("raced", 163)) + "data"; - ASSERT_EQ(backend.putIfAbsent(live_key, "live").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(dead_key, "dead").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(raced_key, "old").outcome, PutOutcome::Done); - backend.replace_on_head = raced_key; - backend.resetCounts(); + createObj(*backend, live_key, "live"); + createObj(*backend, dead_key, "dead"); + createObj(*backend, raced_key, "old"); + backend->replace_on_head = raced_key; + backend->resetCounts(); - const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_TRUE(result.anomalies.empty()); - EXPECT_TRUE(backend.get(live_key)); - EXPECT_FALSE(backend.get(dead_key)); - ASSERT_TRUE(backend.get(raced_key)); - EXPECT_EQ(backend.get(raced_key)->bytes, "winner"); - EXPECT_EQ(backend.headCount(live_key), 0u); - EXPECT_EQ(backend.headCount(dead_key), 1u); - EXPECT_EQ(backend.headCount(raced_key), 1u); - EXPECT_EQ(backend.deleteCount(dead_key), 1u); - EXPECT_EQ(backend.deleteCount(raced_key), 1u); + EXPECT_TRUE(readObj(*backend, live_key).has_value()); + EXPECT_FALSE(readObj(*backend, dead_key).has_value()); + ASSERT_TRUE(readObj(*backend, raced_key).has_value()); + EXPECT_EQ(readObj(*backend, raced_key)->bytes, "winner"); + EXPECT_EQ(backend->headCount(live_key), 0u); + EXPECT_EQ(backend->headCount(dead_key), 1u); + EXPECT_EQ(backend->headCount(raced_key), 1u); + EXPECT_EQ(backend->deleteCount(dead_key), 1u); + EXPECT_EQ(backend->deleteCount(raced_key), 1u); } TEST(CASNamespaceJanitor, TokenlessListRechecksFenceAfterHeadBeforeDelete) { - FenceLossDuringHeadBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead_key = layout.refCkptKey(life("dead", 164)); - ASSERT_EQ(backend.putIfAbsent(dead_key, "dead").outcome, PutOutcome::Done); - backend.resetCounts(); + createObj(*backend, dead_key, "dead"); + backend->resetCounts(); - const auto result = NamespaceJanitor(backend, layout, 100).runOnePage( - false, [&] { return backend.fence_held; }); + const auto result = NamespaceJanitor(requests, layout, 100).runOnePage( + false, [&] { return backend->fence_held; }); EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.headCount(dead_key), 1u); - EXPECT_EQ(backend.deleteCount(dead_key), 0u); - EXPECT_TRUE(backend.get(dead_key)); + EXPECT_EQ(backend->headCount(dead_key), 1u); + EXPECT_EQ(backend->deleteCount(dead_key), 0u); + EXPECT_TRUE(readObj(*backend, dead_key).has_value()); } TEST(CASNamespaceJanitor, PostListCatalogCutProtectsConcurrentCreationWithOneGet) { const auto created = life("created", 121); - CatalogAfterListBackend backend(created); + auto backend = std::make_shared(created); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String first = layout.refCkptKey(created); const String second = layout.namespaceFilesPrefix(created) + "data"; - ASSERT_EQ(backend.putIfAbsent(first, "ckpt").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(second, "file").outcome, PutOutcome::Done); - backend.resetCounts(); - const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + createObj(*backend, first, "ckpt"); + createObj(*backend, second, "file"); + backend->resetCounts(); + const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); - EXPECT_EQ(backend.deleteTotal(), 0u); - EXPECT_EQ(backend.getCount(layout.refCatalogKey()), 1u); - EXPECT_TRUE(backend.get(first)); - EXPECT_TRUE(backend.get(second)); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); + EXPECT_TRUE(readObj(*backend, first).has_value()); + EXPECT_TRUE(readObj(*backend, second).has_value()); } TEST(CASNamespaceJanitor, BackendRejectedCursorResetsExactlyAndDeletesNothing) { - RejectCursorBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead = layout.refCkptKey(life("dead", 131)); - ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); - ASSERT_EQ(backend.putIfAbsent(layout.gcMaintenanceStateKey(), - encodeGcMaintenanceState({.janitor_cursor = "rejected"})).outcome, PutOutcome::Done); - EXPECT_THROW(NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }), std::runtime_error); - EXPECT_EQ(backend.deleteTotal(), 0u); - EXPECT_TRUE(backend.get(dead)); - EXPECT_TRUE(readGcMaintenanceState(backend, layout).state->janitor_cursor.empty()); + createObj(*backend, dead, "bytes"); + createObj(*backend, layout.gcMaintenanceStateKey(), + encodeGcMaintenanceState({.janitor_cursor = "rejected"})); + EXPECT_THROW(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }), std::runtime_error); + EXPECT_EQ(backend->deleteTotal(), 0u); + EXPECT_TRUE(readObj(*backend, dead).has_value()); + EXPECT_TRUE(readState(requests, layout).state->janitor_cursor.empty()); } TEST(CASNamespaceJanitor, CursorPublicationFailureIsLeakOnly) { - FailMaintenancePublicationBackend backend; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); const Layout layout("p"); - seedCatalog(backend, layout); + seedCatalog(*backend, layout); const String dead = layout.refCkptKey(life("dead", 141)); - ASSERT_EQ(backend.putIfAbsent(dead, "bytes").outcome, PutOutcome::Done); - backend.fail_publication = true; - const auto result = NamespaceJanitor(backend, layout, 100).runOnePage(false, [] { return true; }); + createObj(*backend, dead, "bytes"); + backend->fail_publication = true; + const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_FALSE(result.anomalies.empty()); - EXPECT_FALSE(backend.get(dead)); + EXPECT_FALSE(readObj(*backend, dead).has_value()); +} + +/// The write inside `catch (...)` (the reset after a LIST failure) is admitted `once`: an unmodeled, +/// unresolvable write attempt must give up after its one exact resolve read rather than looping through +/// `Retry::standard()`'s backoff. `write_attempts == 1` is the discriminator: under `standard`, the same +/// unresolvable write would keep reissuing until the ninety-second policy window (the LIST failure is +/// itself a genuine `NETWORK_ERROR`, which the read engine retries to its OWN deadline before this catch +/// path is even entered -- so a raw sleep count is not a usable signal here, it is dirtied by the LIST's +/// unrelated retries regardless of which policy the catch-path write uses; the injected clock exists only +/// to keep both retry loops instant rather than to prove anything by its own emptiness). +TEST(CASGcMaintenanceState, CatchPathWriteIsOnce) +{ + auto backend = std::make_shared(); + FakeClock clock; + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn()); + const Layout layout("p"); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { (void)NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); }); + EXPECT_EQ(backend->write_attempts, 1u) + << "the catch-path reset settles by its one resolve read and gives up rather than reissuing"; } TEST(CASNamespaceJanitorIntegration, RegularGcRoundDeletesDeadNamespaceBytes) @@ -581,12 +711,11 @@ TEST(CASNamespaceJanitorIntegration, RegularGcRoundDeletesDeadNamespaceBytes) const Layout & layout = store->layout(); const RootNamespace live_namespace{"00/live@cas@"}; fixture::admitLive(*backend, layout, live_namespace); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(fixture::fixtureLife(live_namespace)), + createObj(*backend, layout.refCkptKey(fixture::fixtureLife(live_namespace)), encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})).outcome, - PutOutcome::Done); + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); const String dead = layout.refCkptKey(life("dead", 151)); - ASSERT_EQ(backend->putIfAbsent(dead, "checkpoint").outcome, PutOutcome::Done); + createObj(*backend, dead, "checkpoint"); std::map namespace_cleanup; Gc gc(store, UInt128{152}); @@ -599,7 +728,7 @@ TEST(CASNamespaceJanitorIntegration, RegularGcRoundDeletesDeadNamespaceBytes) gc.setPhaseSink({}); ASSERT_TRUE(report.acquired_lease); - EXPECT_FALSE(backend->get(dead)); + EXPECT_FALSE(readObj(*backend, dead).has_value()); ASSERT_FALSE(namespace_cleanup.empty()); EXPECT_EQ(namespace_cleanup["janitor_pages"], 1u); EXPECT_GE(namespace_cleanup["janitor_keys"], 1u); diff --git a/src/Disks/tests/gtest_cas_namespace_life_id.cpp b/src/Disks/tests/gtest_cas_namespace_life_id.cpp index 9c2156d52aea..65c116d89f07 100644 --- a/src/Disks/tests/gtest_cas_namespace_life_id.cpp +++ b/src/Disks/tests/gtest_cas_namespace_life_id.cpp @@ -205,8 +205,8 @@ TEST(CASNamespaceLifeIdDeathTest, ZeroIncarnationIsUnconstructibleAborts) } #endif -/// Generation-5 namespace-bearing keys are outside the generation-6 parser roots altogether. Pool -/// admission rejects their generation before any listed-key parser is involved. +/// Namespace-bearing keys outside the opaque-life layout are rejected before any listed-key parser is +/// involved. TEST(CASNamespaceLifeId, GenerationFiveNamespaceBearingKeysAreOutsideTheFinalGrammar) { Layout l("p"); @@ -312,8 +312,8 @@ TEST(CASNamespaceLifeId, NamespaceFileKeysCarryTheIncarnationSegment) EXPECT_FALSE(l.namespaceFileKey(second, "format_version.txt").starts_with(l.namespaceFilesPrefix(life))); } -/// Generation-5 namespace-bearing file keys are outside the final parser root. Malformed ids under the -/// final state root are corruption and name the offending key. +/// Namespace-bearing file keys outside the opaque-life layout are rejected. Malformed life ids under +/// the state root are corruption and name the offending key. TEST(CASNamespaceLifeId, NamespaceFileParserRefusesLegacyAndMalformedIncarnations) { Layout l("p"); @@ -373,8 +373,7 @@ TEST(CASNamespaceLifeId, PhysicalFileKeysIgnoreLogicalNamespaceSpelling) EXPECT_EQ(parsed->relative_name, nested_name); } -/// The "cannot compile" half of spec §9 r9-5 #3: after this task there is no way to reach a ref-layer -/// key from a namespace alone, so dropping the incarnation is a compile error rather than an aliasing +/// A ref-layer key cannot be reached from a namespace alone, so dropping the incarnation is a compile error rather than an aliasing /// bug. Each helper is asserted twice -- the namespace-only form absent, the incarnation form present. TEST(CASNamespaceLifeId, NamespaceOnlyKeyHelpersDoNotExist) { @@ -413,8 +412,8 @@ TEST(CASNamespaceLifeId, NamespaceLifeIdAndRootNamespaceDoNotInterconvert) SUCCEED(); } -/// The out-of-scope fences, and they are POSITIVE on purpose: Constraint 12 keeps loose mountpoint -/// objects and part manifests on the identity they have today, so this task must NOT have qualified +/// The out-of-scope fences are POSITIVE on purpose: loose mountpoint objects and part manifests keep +/// their namespace identity, so they must NOT be qualified /// them. If a negative here fails, someone added a life-scoped overload to a family the amendment /// explicitly excluded; if a positive fails, someone removed the un-scoped one those callers use. TEST(CASNamespaceLifeId, MountpointObjectsAndManifestsStayUnqualified) diff --git a/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp b/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp index 7acab161e63d..26ea174533f1 100644 --- a/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp +++ b/src/Disks/tests/gtest_cas_ns_creation_lifecycle.cpp @@ -30,18 +30,6 @@ namespace DB::ErrorCodes namespace { -/// A fence that never refuses, for tests whose subject is not the fence -- same helper, same intent, -/// as `gtest_cas_ref_ckpt.cpp`'s identically-named constant (not shared: each `_ckpt`/catalog test file -/// defines its own copy, matching that file's own precedent). -const std::function ALWAYS_ADMITTED = [](uint64_t) {}; - -/// A deadline far enough out that only the test's own contention decides the outcome -- mirrors -/// `gtest_cas_ref_ckpt.cpp`'s `generousDeadline`. -CkptDeadline generousDeadline() -{ - return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; -} - CreatorFence creatorFence(const String & srid, uint64_t writer_epoch, uint64_t fence_generation = 1) { return CreatorFence{.server_root_id = srid, .writer_epoch = writer_epoch, .fence_generation = fence_generation}; @@ -63,16 +51,65 @@ const CatalogEntry * findEntryForTest(const RefCatalog & catalog, const RootName return nullptr; } -/// Raw lifecycle tests operate below `Pool::open`, so model an already-bootstrapped pool explicitly. -class InitializedCatalogBackend : public InMemoryBackend +/// Withdraws an operation's admission, and lets a test smuggle a real concurrent write, at one chosen +/// point of the creation sequence: once this namespace's `_ckpt` is durable (step 2 landed, step 3 has +/// not run), or inside step 3's own read-then-write window. The `_ckpt` key carries an incarnation a +/// test cannot know before the creation mints it, so that arm names the object kind rather than a key. +/// +/// The read arm fires BEFORE the store is consulted, so the body the caller's `decide` receives already +/// carries whatever the hook wrote. That is what lets a test make the observed body stale on BOTH axes +/// -- a changed entry and a withdrawn admission -- inside ONE `decide` invocation, which is the only +/// place the two can be told apart. +class CreationHookBackend : public InMemoryBackend { public: - InitializedCatalogBackend() + bool admitted = true; + /// Withdraw once any `_ckpt` key has been written. + bool withdraw_after_ckpt_write = false; + /// Fires once before this key is read, then `withdraw_on_read` is applied. + String hook_before_read_of; + std::function on_read; + bool withdraw_on_read = false; + + std::optional read(const String & key, TransportAccess & access) override + { + if (!hook_before_read_of.empty() && key == hook_before_read_of && !hook_fired) + { + /// Latched before running: the hook reads and writes through this same backend, and an + /// unguarded re-entry would run the test's concurrent actor again against its own result. + hook_fired = true; + if (on_read) + on_read(); + if (withdraw_on_read) + admitted = false; + } + return InMemoryBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - CasRefCatalog::initializeEmptyForNewPool(*this, Layout("p")); + auto result = InMemoryBackend::write(key, bytes, expected_value, access); + if (withdraw_after_ckpt_write && Layout{"p"}.parseRefCkptKey(key)) + admitted = false; + return result; } + +private: + bool hook_fired = false; }; +/// Raw lifecycle tests operate below `Pool::open`, so model an already-bootstrapped pool explicitly. +std::shared_ptr initializedCatalogBackend() +{ + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, Layout("p")); + return backend; +} + } /// --------------------------------------------------------------------------------------------- @@ -81,16 +118,18 @@ class InitializedCatalogBackend : public InMemoryBackend TEST(CASNsCreationLifecycle, HappyPathReachesLiveWithADurableCkptAndAStableIncarnation) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence creator = creatorFence("srv1", /*writer_epoch=*/5); const auto outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, creator, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + op, layout, 1, ns, creator); EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Live); - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); ASSERT_NE(entry, nullptr); EXPECT_EQ(entry->state, NsState::Live); @@ -98,53 +137,66 @@ TEST(CASNsCreationLifecycle, HappyPathReachesLiveWithADurableCkptAndAStableIncar const UInt128 incarnation = entry->incarnation; EXPECT_NE(incarnation, UInt128(0)); - const std::optional ckpt = readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, incarnation)); + const std::optional ckpt = readCkpt(op, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, incarnation)); ASSERT_TRUE(ckpt.has_value()) << "step 2's _ckpt must be durable"; EXPECT_EQ(ckpt->ckpt.life_epoch, 5u) << "INV-4's genesis epoch is the creator's writer_epoch"; /// Re-reading the catalog again must show the SAME incarnation -- nothing mints a second one. - EXPECT_EQ(CasRefCatalog::read(backend, layout).catalog.entries.at(0).incarnation, incarnation); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries.at(0).incarnation, incarnation); } /// --------------------------------------------------------------------------------------------- -/// `createNamespace` refuses a namespace that already has an entry (Task 2 review's own note: this -/// is Task 3's job, not `casAdmitEntry`'s duplicate-namespace grammar refusal). +/// `createNamespace`'s pre-check reports `Superseded` for a namespace that already has an entry, in +/// EVERY state -- not a caller bug, but a sibling opener of the same namespace (CI PR#2300 run 3, +/// `tiered_storage_cas`: seven concurrent `MergeTreeBackgroundExecutor` movers) that landed somewhere in +/// its own three-step sequence between the caller's "no entry" read and this pre-check's read. /// --------------------------------------------------------------------------------------------- -#ifndef DEBUG_OR_SANITIZER_BUILD -TEST(CASNsCreationLifecycle, CreateNamespaceRejectsAnAlreadyExistingEntry) +TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsLiveEntryReportsSupersededNotAbort) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence creator = creatorFence("srv1", 1); - ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creator, 1, ALWAYS_ADMITTED, generousDeadline()), + ASSERT_EQ(CasRefCatalog::createNamespace(op, layout, 1, ns, creator), CasRefCatalog::NamespaceCreationOutcome::Live); - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] - { - CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv2", 2), 1, ALWAYS_ADMITTED, generousDeadline()); - }); + EXPECT_EQ(CasRefCatalog::createNamespace(op, layout, 1, ns, creatorFence("srv2", 2)), + CasRefCatalog::NamespaceCreationOutcome::Superseded); + + /// The refused call left the winner's `Live` entry exactly as it was. + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Live); + EXPECT_EQ(entry->creator, std::nullopt); } -#endif -#if defined(DEBUG_OR_SANITIZER_BUILD) -TEST(CASNsCreationLifecycleDeathTest, CreateNamespaceRejectsAnAlreadyExistingEntryAborts) +TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsRemovingEntryReportsSupersededNotAbort) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence creator = creatorFence("srv1", 1); - ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creator, 1, ALWAYS_ADMITTED, generousDeadline()), + ASSERT_EQ(CasRefCatalog::createNamespace(op, layout, 1, ns, creator), CasRefCatalog::NamespaceCreationOutcome::Live); + ASSERT_EQ(CasRefCatalog::beginRemoving(op, layout, *findEntryForTest(CasRefCatalog::read(op, layout).catalog, ns), + /*removal_started_round=*/1), + CasRefCatalog::BeginRemovingOutcome::Transitioned); - EXPECT_DEATH( - { - CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv2", 2), 1, ALWAYS_ADMITTED, generousDeadline()); - }, - "already carries a catalog entry"); + EXPECT_EQ(CasRefCatalog::createNamespace(op, layout, 1, ns, creatorFence("srv2", 2)), + CasRefCatalog::NamespaceCreationOutcome::Superseded); + + /// The refused call left the concurrent drop's `Removing` entry exactly as it was. + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); + const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Removing); } -#endif /// --------------------------------------------------------------------------------------------- /// `Creating` forbids publication @@ -173,34 +225,26 @@ TEST(CASNsCreationLifecycle, LiveAndRemovingAndAbsentAllAdmitPublication) /// ZombieGoLive: fenced-out between the `_ckpt` publish and the `Creating -> Live` CAS /// --------------------------------------------------------------------------------------------- -/// A fence callback that admits its FIRST call (spent by step 2's `publishCkpt`) and refuses every -/// call after (spent by step 3's `mutate`) -- deterministically reproducing "fenced out between the -/// `_ckpt` create and the `Creating -> Live` CAS" without a second thread or fault injection. -namespace -{ -std::function admittedOnceThenFenced() -{ - auto calls = std::make_shared(0); - return [calls](uint64_t admitted) - { - if (++*calls > 1) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); - }; -} -} - TEST(CASNsCreationLifecycle, FencedOutBetweenTheCkptPublishAndGoLiveRefusesAndLeavesEntryCreating) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence creator = creatorFence("srv1", 5); - const auto outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, creator, /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()); + /// Admission is withdrawn the instant step 2's `_ckpt` is durable, so step 3 never runs -- + /// "fenced out between the `_ckpt` create and the `Creating -> Live` write", without a second + /// thread. + backend->withdraw_after_ckpt_write = true; + CasOperation creating_op = requests.admit([&backend] { return backend->admitted; }); + const auto outcome = CasRefCatalog::createNamespace(creating_op, layout, 1, ns, creator); EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut); + backend->withdraw_after_ckpt_write = false; + backend->admitted = true; - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); ASSERT_NE(entry, nullptr); EXPECT_EQ(entry->state, NsState::Creating) << "step 3 never ran its CAS -- ZombieGoLive refuses before sending it"; @@ -210,7 +254,7 @@ TEST(CASNsCreationLifecycle, FencedOutBetweenTheCkptPublishAndGoLiveRefusesAndLe /// Step 2's _ckpt DID land (it is not what the fence check gates) -- CKPT-FAILED-BIRTH-DEBRIS is a /// different mechanism (the OLD `RefOpKind::NamespaceBirth` writer, `Pool/CasRefLedger.cpp`); this /// driver's own `_ckpt` is simply left in place for whichever actor next reconciles this entry. - EXPECT_TRUE(readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation)).has_value()); + EXPECT_TRUE(readCkpt(op, layout, NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation)).has_value()); } /// Regression (CI PR#2073, `03611_freeze_partition_parallel_verbose` under `amd_tsan, cas s3 storage`): @@ -220,28 +264,33 @@ TEST(CASNsCreationLifecycle, FencedOutBetweenTheCkptPublishAndGoLiveRefusesAndLe /// must send the loser back through the resume loop (`Superseded`), never abort the server. TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsStillCreatingEntryReportsSupersededNotAbort) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence winner = creatorFence("srv1", 1); /// Leaves the entry in `Creating` without reaching `Live` -- the same shape `resolveNamespaceLife` /// observes when a sibling thread's `casAdmitEntry` has landed but its `completeCreation` has not. - const auto winner_outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, winner, /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()); + backend->withdraw_after_ckpt_write = true; + CasOperation winner_op = requests.admit([&backend] { return backend->admitted; }); + const auto winner_outcome = CasRefCatalog::createNamespace(winner_op, layout, 1, ns, winner); ASSERT_EQ(winner_outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut); - ASSERT_EQ(CasRefCatalog::read(backend, layout).catalog.entries.at(0).state, NsState::Creating); + backend->withdraw_after_ckpt_write = false; + backend->admitted = true; + ASSERT_EQ(CasRefCatalog::read(op, layout).catalog.entries.at(0).state, NsState::Creating); /// The loser: a second call, as if a sibling thread's own outer "no entry" read had raced ahead of /// this one. Same fence as the winner (sibling threads of one query share a mount's fence) -- /// exercising exactly the case `resolveNamespaceLife`'s "own fence -> completeCreation" branch is /// built to resume, never a `LOGICAL_ERROR` abort. const auto loser_outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, winner, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + op, layout, 1, ns, winner); EXPECT_EQ(loser_outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); /// Nothing about the winner's own still-`Creating` entry was disturbed by the loser's refused call. - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); const CatalogEntry * entry = findEntryForTest(snap.catalog, ns); ASSERT_NE(entry, nullptr); EXPECT_EQ(entry->state, NsState::Creating); @@ -260,7 +309,9 @@ TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsStillCreatingEntryRep /// single upfront read) can catch it. TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsFullCreateBetweenPreCheckAndStep1ReportsSupersededNotAbort) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence winner = creatorFence("srv1", 1); @@ -275,17 +326,17 @@ TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsFullCreateBetweenPreC CasRefCatalog::setCreateNamespaceStep1PreReadHookForTest([&] { const auto winner_outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, winner, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + op, layout, 1, ns, winner); ASSERT_EQ(winner_outcome, CasRefCatalog::NamespaceCreationOutcome::Live); }); const auto loser_outcome = CasRefCatalog::createNamespace( - backend, layout, 1, ns, loser, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + op, layout, 1, ns, loser); EXPECT_EQ(loser_outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); /// Exactly one row for `ns`, owned by the winner, at `Live` -- the loser's refused admission left /// no trace and did not disturb it. - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); size_t rows_for_ns = 0; for (const CatalogEntry & e : snap.catalog.entries) if (e.ns.string() == ns.string()) @@ -298,12 +349,14 @@ TEST(CASNsCreationLifecycle, CreateNamespaceRacingASiblingsFullCreateBetweenPreC } /// --------------------------------------------------------------------------------------------- -/// Token-stale: the observed entry no longer matches at the `Creating -> Live` CAS +/// Entry-stale: the observed entry no longer matches at the `Creating -> Live` write /// --------------------------------------------------------------------------------------------- TEST(CASNsCreationLifecycle, EntryStolenByAConcurrentReconcilerRefusesGoLiveAndLeavesTheStolenEntryAlone) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence original_creator = creatorFence("srv1", 5); @@ -311,32 +364,32 @@ TEST(CASNsCreationLifecycle, EntryStolenByAConcurrentReconcilerRefusesGoLiveAndL /// Write 1 only -- models "crash after write 1": no _ckpt yet, entry still Creating. const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(42), .creator = original_creator}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); - - /// `check_fence_or_throw` is the seam this driver calls on EVERY attempt -- once inside step 2's - /// `publishCkpt`, once more inside step 3's own `mutate` -- so smuggling a REAL concurrent write - /// into it (rather than faking the outcome) has to land on the SECOND call specifically, or the - /// steal itself would run twice (and the second run would see its own first result and refuse). - /// This reproduces "stolen between the creator's _ckpt publish and its Creating -> Live CAS" - /// without a second thread. The steal itself must succeed (asserted), so the mismatch - /// `completeCreation` sees below is the entry ACTUALLY changing, not a contrived stub. - auto calls = std::make_shared(0); - const std::function steal_before_the_go_live_cas = [&, calls](uint64_t) + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); + + /// A REAL concurrent write, smuggled in just before step 3's own catalog read -- the first one + /// `completeCreation` performs, since step 2 touches only the `_ckpt`. `decide` therefore receives + /// the POST-steal body and refuses it without sending anything, which is what the entry check is + /// for. The steal itself must succeed (asserted), so the mismatch is the entry ACTUALLY changing, + /// not a contrived stub. It runs on its own operation and its own `CasRequests` (a rival is + /// another server; one server's writers never race each other on the pool's hot-key lane). + CasRequests rival_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation thief_op = rival_requests.admit(); + backend->hook_before_read_of = layout.refCatalogKey(); + backend->on_read = [&] { - if (++*calls == 2) - ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, thief, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), - CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(thief_op, layout, entry, thief, fixedTerminality(true)), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); }; - const auto outcome = CasRefCatalog::completeCreation( - backend, layout, entry, /*admitted_generation=*/1, steal_before_the_go_live_cas, generousDeadline()); + const auto outcome = CasRefCatalog::completeCreation(op, layout, entry); EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Superseded); + backend->on_read = nullptr; /// `read`'s `Snapshot` is bound to a name here, not chained through a temporary: a `const /// CatalogEntry *` taken from `.catalog` of an unbound temporary dangles the instant the full /// expression ends, which every other site in this file (and the copy/paste that spread it) got /// wrong until ASan caught it. - const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(op, layout); const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); ASSERT_NE(after, nullptr); EXPECT_EQ(after->state, NsState::Creating) << "the ORIGINAL creator's attempt wrote nothing -- only the thief's CAS did"; @@ -351,37 +404,48 @@ TEST(CASNsCreationLifecycle, EntryStolenByAConcurrentReconcilerRefusesGoLiveAndL TEST(CASNsCreationLifecycle, BothFenceAndEntryStaleRefusesGoLiveViaTheFenceCheckWhichRunsFirst) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence original_creator = creatorFence("srv1", 5); const CreatorFence thief = creatorFence("srv2", 9); const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(42), .creator = original_creator}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); - - /// Same steal as the test above, landing on the SECOND `check_fence_or_throw` call (step 3's own - /// `mutate`, not step 2's `publishCkpt`) -- but this one ALSO throws on that same second call, so - /// both axes go stale in the SAME `mutate` invocation. `completeCreation`'s fence check runs before - /// its entry check (documented ordering), so this is reported `FencedOut`; the assertions below - /// confirm the entry ALSO changed, so the test is not merely re-proving the fence-only case above. - auto calls = std::make_shared(0); - const std::function steal_and_fence_before_the_go_live_cas = [&, calls](uint64_t admitted) + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); + + /// The same steal as the test above and in the same window, but this one ALSO withdraws admission + /// there. Because the hook runs BEFORE the read, the single `decide` invocation that follows sees a + /// body that is stale on both axes at once -- and that is the only situation in which the two + /// checks are distinguishable. `completeCreation` consults admission before it compares the entry, + /// so the answer is `FencedOut`. + /// + /// This is what makes the test discriminate rather than merely pass: delete the `op.admitted()` + /// check from that `mutate` and the entry check answers `Superseded` instead, because `decide` + /// refuses the stale entry before any write is sent and the engine's own gate never speaks. The + /// assertions below confirm the entry really did change too. + /// The rival gets its own `CasRequests`, on its own hot-key lane (a rival is another server; one + /// server's writers never race each other on the pool's lane). + CasRequests rival_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation thief_op = rival_requests.admit(); + CasOperation creator_op = requests.admit([&backend] { return backend->admitted; }); + backend->hook_before_read_of = layout.refCatalogKey(); + backend->withdraw_on_read = true; + backend->on_read = [&] { - if (++*calls == 2) - { - ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, thief, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), - CasRefCatalog::ReconcileCreatorOutcome::Reconciled); - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); - } + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(thief_op, layout, entry, thief, fixedTerminality(true)), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); }; - const auto outcome = CasRefCatalog::completeCreation( - backend, layout, entry, /*admitted_generation=*/1, steal_and_fence_before_the_go_live_cas, generousDeadline()); + const auto outcome = CasRefCatalog::completeCreation(creator_op, layout, entry); EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::FencedOut) - << "both checks would refuse; the fence check speaks first by this driver's fixed ordering"; + << "both would refuse; admission speaks first by the documented ordering"; + backend->on_read = nullptr; + backend->withdraw_on_read = false; + backend->admitted = true; - const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(op, layout); const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); ASSERT_NE(after, nullptr); ASSERT_TRUE(after->creator.has_value()); @@ -394,18 +458,20 @@ TEST(CASNsCreationLifecycle, BothFenceAndEntryStaleRefusesGoLiveViaTheFenceCheck TEST(CASNsCreationLifecycle, ReconcileRefusedWhileTheOriginalCreatorFenceIsStillLive) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), .creator = creatorFence("srv1", 5)}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); const auto outcome = CasRefCatalog::reconcileStaleCreator( - backend, layout, entry, creatorFence("srv2", 9), fixedTerminality(false), /*admitted_generation=*/1, ALWAYS_ADMITTED); + op, layout, entry, creatorFence("srv2", 9), fixedTerminality(false)); EXPECT_EQ(outcome, CasRefCatalog::ReconcileCreatorOutcome::CreatorFenceStillLive); - const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(op, layout); const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); ASSERT_NE(after, nullptr); EXPECT_EQ(*after, entry) << "refused -- nothing written"; @@ -413,34 +479,36 @@ TEST(CASNsCreationLifecycle, ReconcileRefusedWhileTheOriginalCreatorFenceIsStill TEST(CASNsCreationLifecycle, ReconcileSucceedsTokenExactlyAfterTheOriginalCreatorFenceIsTerminalThenResumesToLive) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CreatorFence original_creator = creatorFence("srv1", 5); const CreatorFence new_creator = creatorFence("srv2", 9); const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), .creator = original_creator}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); /// "crash after write 1" -- no _ckpt yet + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); /// "crash after write 1" -- no _ckpt yet - ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, new_creator, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(op, layout, entry, new_creator, fixedTerminality(true)), CasRefCatalog::ReconcileCreatorOutcome::Reconciled); CatalogEntry taken_over = entry; taken_over.creator = new_creator; - const CasRefCatalog::Snapshot snap_mid = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_mid = CasRefCatalog::read(op, layout); const CatalogEntry * mid = findEntryForTest(snap_mid.catalog, ns); ASSERT_NE(mid, nullptr); EXPECT_EQ(*mid, taken_over) << "creator moved to the new actor; state and incarnation unchanged"; const auto outcome = CasRefCatalog::completeCreation( - backend, layout, taken_over, /*admitted_generation=*/1, ALWAYS_ADMITTED, generousDeadline()); + op, layout, taken_over); EXPECT_EQ(outcome, CasRefCatalog::NamespaceCreationOutcome::Live); - const CasRefCatalog::Snapshot snap_final = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_final = CasRefCatalog::read(op, layout); const CatalogEntry * final_entry = findEntryForTest(snap_final.catalog, ns); ASSERT_NE(final_entry, nullptr); EXPECT_EQ(final_entry->state, NsState::Live); EXPECT_EQ(final_entry->incarnation, entry.incarnation) << "the SAME incarnation throughout -- resumption, not rebirth"; - const std::optional ckpt = readCkpt(backend, layout, NamespaceLifeId::fromCatalogEntry(final_entry->ns, final_entry->incarnation)); + const std::optional ckpt = readCkpt(op, layout, NamespaceLifeId::fromCatalogEntry(final_entry->ns, final_entry->incarnation)); ASSERT_TRUE(ckpt.has_value()); EXPECT_EQ(ckpt->ckpt.life_epoch, new_creator.writer_epoch) << "the RESUMING actor's writer_epoch is the genesis epoch that actually landed"; @@ -450,26 +518,28 @@ TEST(CASNsCreationLifecycle, ReconcileSucceedsTokenExactlyAfterTheOriginalCreato /// the SAME stale `observed` before either writes. TEST(CASNsCreationLifecycle, ReconcileFailsClosedWhenTheEntryAlreadyChanged) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const RootNamespace ns{"a"}; const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(7), .creator = creatorFence("srv1", 5)}; - CasRefCatalog::casAdmitEntry(backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); const CreatorFence first_reconciler = creatorFence("srv2", 9); const CreatorFence second_reconciler = creatorFence("srv3", 11); - ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, entry, first_reconciler, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED), + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(op, layout, entry, first_reconciler, fixedTerminality(true)), CasRefCatalog::ReconcileCreatorOutcome::Reconciled); /// The second reconciler still holds the ORIGINAL `entry` it read before either of them wrote -- /// token-exactness must refuse it even though the terminality predicate would still say yes. const auto outcome = CasRefCatalog::reconcileStaleCreator( - backend, layout, entry, second_reconciler, fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + op, layout, entry, second_reconciler, fixedTerminality(true)); EXPECT_EQ(outcome, CasRefCatalog::ReconcileCreatorOutcome::EntryChanged); - const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(op, layout); const CatalogEntry * after = findEntryForTest(snap_after.catalog, ns); ASSERT_NE(after, nullptr); ASSERT_TRUE(after->creator.has_value()); @@ -484,23 +554,27 @@ TEST(CASNsCreationLifecycle, ReconcileFailsClosedWhenTheEntryAlreadyChanged) #ifndef DEBUG_OR_SANITIZER_BUILD TEST(CASNsCreationLifecycle, CompleteCreationRejectsANonCreatingEntry) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { - CasRefCatalog::completeCreation(backend, layout, live, 1, ALWAYS_ADMITTED, generousDeadline()); + CasRefCatalog::completeCreation(op, layout, live); }); } TEST(CASNsCreationLifecycle, ReconcileStaleCreatorRejectsANonCreatingEntry) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { - CasRefCatalog::reconcileStaleCreator(backend, layout, live, creatorFence("srv2", 2), fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + CasRefCatalog::reconcileStaleCreator(op, layout, live, creatorFence("srv2", 2), fixedTerminality(true)); }); } #endif @@ -508,22 +582,26 @@ TEST(CASNsCreationLifecycle, ReconcileStaleCreatorRejectsANonCreatingEntry) #if defined(DEBUG_OR_SANITIZER_BUILD) TEST(CASNsCreationLifecycleDeathTest, CompleteCreationRejectsANonCreatingEntryAborts) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; EXPECT_DEATH( - { CasRefCatalog::completeCreation(backend, layout, live, 1, ALWAYS_ADMITTED, generousDeadline()); }, + { CasRefCatalog::completeCreation(op, layout, live); }, "not a Creating entry"); } TEST(CASNsCreationLifecycleDeathTest, ReconcileStaleCreatorRejectsANonCreatingEntryAborts) { - InitializedCatalogBackend backend; + auto backend = initializedCatalogBackend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const CatalogEntry live{.ns = RootNamespace{"a"}, .state = NsState::Live, .incarnation = UInt128(1)}; EXPECT_DEATH( { - CasRefCatalog::reconcileStaleCreator(backend, layout, live, creatorFence("srv2", 2), fixedTerminality(true), /*admitted_generation=*/1, ALWAYS_ADMITTED); + CasRefCatalog::reconcileStaleCreator(op, layout, live, creatorFence("srv2", 2), fixedTerminality(true)); }, "not a Creating entry"); } diff --git a/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp b/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp index 901b223cb90f..89d55e8dbb65 100644 --- a/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp +++ b/src/Disks/tests/gtest_cas_ns_file_incarnation.cpp @@ -1,7 +1,6 @@ #include #include -#include #include #include #include @@ -9,11 +8,6 @@ #include "cas_test_helpers.h" #include -namespace DB::ErrorCodes -{ - extern const int UNKNOWN_FORMAT_VERSION; -} - /// Namespace files are keyed by an opaque LIFE, not by its name: `cas/ns/state//_files/` /// (Stage B Task 4b, directive design change 2). This file pins the three properties that re-key exists /// to produce, and the one it must NOT produce. @@ -80,27 +74,31 @@ TEST(CASNsFileIncarnation, ColdReaderUsesCatalogCutWhileOldFileSurvivesRemoval) const String old_key = layout.namespaceFileKey(*old_life, kFile); backend->hide(old_key); - ASSERT_TRUE(backend->head(old_key).exists) << "the lie must be in LIST only -- the object is durable"; + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + + ASSERT_TRUE(catalog_op.head(old_key, Retry::standard()).has_value()) + << "the lie must be in LIST only -- the object is durable"; ASSERT_TRUE(store->listNamespaceFiles(*old_life).empty()) << "precondition: enumeration omits the file, so no cleanup pass can ever find it"; const size_t holes_before_gc = backend->holesServed(); store->dropNamespace(ns); - ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)); + ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns)); Gc gc(store, kGcId); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "N: the production terminal must fold"; - ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)) + ASSERT_TRUE(CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns)) << "the terminal fold alone must not erase its catalog row"; (void)runRegularRoundReclaiming(gc); - ASSERT_FALSE(CasRefCatalog::lifeIfCataloged(*backend, layout, ns)) + ASSERT_FALSE(CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns)) << "N+1: the pre-fold drain must erase the exact completed Removing row"; ASSERT_GT(backend->holesServed(), holes_before_gc) << "the GC janitor must observe the injected LIST hole after the explicit precondition LIST"; - const auto old_head = backend->head(old_key); - ASSERT_TRUE(old_head.exists) << "logical removal must not depend on physical empty"; - const auto old_object = backend->get(old_key); + const auto old_head = catalog_op.head(old_key, Retry::standard()); + ASSERT_TRUE(old_head.has_value()) << "logical removal must not depend on physical empty"; + const auto old_object = catalog_op.read(old_key, Retry::standard()); ASSERT_TRUE(old_object); EXPECT_EQ(old_object->bytes, old_bytes); @@ -136,17 +134,19 @@ TEST(CASNsFileIncarnation, FreshReaderAssignsOnlyLiveCatalogLifeWithoutMutation) const RootNamespace live{"00/live@cas@"}; const RootNamespace removing{"00/removing@cas@"}; const RootNamespace absent{"00/absent@cas@"}; + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); - CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + CasRefCatalog::casAdmitEntry(catalog_op, layout, 1, CatalogEntry{ .ns = creating, .state = NsState::Creating, .incarnation = UInt128{31}, .creator = CreatorFence{.server_root_id = "foreign", .writer_epoch = 7, .fence_generation = 1}}); - CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + CasRefCatalog::casAdmitEntry(catalog_op, layout, 1, CatalogEntry{ .ns = live, .state = NsState::Live, .incarnation = UInt128{32}}); - CasRefCatalog::casAdmitEntry(*backend, layout, 1, CatalogEntry{ + CasRefCatalog::casAdmitEntry(catalog_op, layout, 1, CatalogEntry{ .ns = removing, .state = NsState::Live, .incarnation = UInt128{33}}); - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + CasRefCatalog::casUpdate(catalog_op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) @@ -183,7 +183,6 @@ TEST(CASNsFileIncarnation, FreshReaderAssignsOnlyLiveCatalogLifeWithoutMutation) EXPECT_EQ(store->refTableLifeForTest(live)->incarnation, UInt128{32}); EXPECT_EQ(backend->putTotal(), 0u); EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); } /// A real GC fold records terminal evidence for the previous life while its namespace-file debris @@ -204,7 +203,9 @@ TEST(CASNsFileIncarnation, RebirthDoesNotWaitForFilesToBeEmpty) remove_op.kind = RefOpKind::RemoveNamespace; appendRefLogSeed(*backend, layout, ns, {remove_op}); } - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns).value(); writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, @@ -212,68 +213,21 @@ TEST(CASNsFileIncarnation, RebirthDoesNotWaitForFilesToBeEmpty) .last_epoch_seal = std::nullopt, }); const String debris_key = layout.namespaceFileKey(life, kFile); - backend->putIfAbsent(debris_key, "1\n"); - backend->putIfAbsent(layout.namespaceFileKey(life, "deduplication_logs/deduplication_log_1.txt"), "records"); + catalog_op.create(debris_key, "1\n", Retry::once()); + catalog_op.create(layout.namespaceFileKey(life, "deduplication_logs/deduplication_log_1.txt"), "records", Retry::once()); Gc gc(store, kGcId); gc.runRegularRound(); /// Folding the terminal records positive evidence on the same life row even though files remain. - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState state = decodeGcState(catalog_op.read(layout.gcStateKey(), Retry::standard())->bytes); ASSERT_GT(state.snap_generation, 0u); const CasFoldSeal seal = decodeFoldSeal( - backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + catalog_op.read(layout.foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard())->bytes); const auto row_it = seal.ref_lives.find(life.incarnation); ASSERT_NE(row_it, seal.ref_lives.end()); ASSERT_TRUE(row_it->second.cleanup_evidence.has_value()); EXPECT_EQ(row_it->second.cleanup_evidence->remove_txn_id, (RefTxnId{1, 1})); - EXPECT_TRUE(backend->head(debris_key).exists) << "cleanup evidence does not gate on physical deletion"; -} - -/// An old-format pool carrying unqualified `roots//_files/x` keys is REFUSED AT OPEN. It is not -/// read, not migrated, and not silently re-keyed: the file layer rides Task 4's format bump B, and the -/// pool-open floor is what makes "there is nothing to migrate" true rather than merely intended. -/// -/// Asserted at OPEN rather than at the parser on purpose: `Layout` has no unqualified key constructor -/// at all (a compile-time concept check in `gtest_cas_namespace_life_id.cpp` pins that, and -/// `parseNamespaceFileKey`'s refusal of a legacy key is pinned there too), so the only reachable -/// question left is whether a pool that CONTAINS such keys can be opened. It cannot. -TEST(CASNsFileIncarnation, LegacyUnqualifiedFileKeyIsRefusedAtOpen) -{ - auto backend = std::make_shared(); - const Layout layout("p"); - - /// A generation-5 `_pool_meta`: the current encoder's output with its header generation moved back - /// one, so every other byte is exactly what that generation really wrote. - PoolMeta meta; - meta.pool_id = hexToU128("0123456789abcdef0123456789abcdef"); - meta.blob_header_len = 256; - meta.min_reader_generation = kNamespaceLifeKeyedGeneration - 1; - meta.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - String encoded = encodePoolMeta(meta); - const String current_v = "\"v\":" + std::to_string(G_BUILD); - const String legacy_v = "\"v\":" + std::to_string(kNamespaceLifeKeyedGeneration); - const size_t at = encoded.find(current_v); - /// Guard the substitution itself: a silent no-op here would leave a CURRENT-generation pool and the - /// test would pass by opening a pool it believes it downgraded. - ASSERT_NE(at, String::npos) << "pool-meta header no longer spells its generation as " << current_v; - encoded.replace(at, current_v.size(), legacy_v); - ASSERT_NE(encoded.find(legacy_v), String::npos); - backend->putIfAbsent(layout.poolMetaKey(), encoded); - - /// The legacy artifact this task removes: a namespace file keyed by NAME ONLY, with no incarnation - /// segment. Written as raw bytes because no code path in the tree can produce this key any more. - backend->putIfAbsent("p/roots/" + kNsString + "/_files/" + kFile, "1\n"); - - try - { - openPoolForTest(backend); - FAIL() << "an old-format pool must fail closed at open, naming recreation"; - } - catch (const DB::Exception & e) - { - EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); - EXPECT_NE(e.message().find("recreate"), String::npos) - << "the refusal must tell the operator what to do; got: " << e.message(); - } + EXPECT_TRUE(catalog_op.head(debris_key, Retry::standard()).has_value()) + << "cleanup evidence does not gate on physical deletion"; } diff --git a/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp b/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp index 3166230f5ec8..a843ba3fbce3 100644 --- a/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp +++ b/src/Disks/tests/gtest_cas_ns_file_read_contract.cpp @@ -38,6 +38,11 @@ const String kFile = "format_version.txt"; const String kFilePath = kTablePath + "/" + kFile; const UInt128 kLife2Id = hexToU128("22222222222222222222222222222222"); +/// The erase entry point requires a liveness refresh because a real drain's liveness is a cached flag +/// its owner re-reads from the store. This fixture's operation carries no liveness at all, so there is +/// nothing cached for a refresh to update. +void noAuthorityRefresh() {} + struct DiskFixture { DB::ObjectStoragePtr object_storage; @@ -76,9 +81,10 @@ void writeVerbatimThroughDisk( void deleteCatalogLife( DB::ContentAddressedMetadataStorage & storage, const NamespaceLifeId & life1) { - Backend & backend = storage.store()->backend(); const Layout & layout = storage.store()->layout(); - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + CasRequests requests = DB::Cas::tests::openRequestsForTest(storage.store()->poolBackendPtr()); + CasOperation op = requests.admit(); + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) @@ -93,7 +99,7 @@ void deleteCatalogLife( return next; }); - const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, layout); const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == life1.ns && entry.incarnation == life1.incarnation; @@ -104,11 +110,9 @@ void deleteCatalogLife( CasFoldSeal parent; parent.ref_lives.emplace(life1.incarnation, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); - if (CasRefCatalog::deleteCompletedRemoving( - backend, layout, *it, parent, 1, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }) + if (CasRefCatalog::deleteCompletedRemoving(op, layout, *it, parent, noAuthorityRefresh) != CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted) throw DB::Exception( DB::ErrorCodes::LOGICAL_ERROR, "Failed to delete fixture catalog life '{}'", life1.ns.string()); @@ -121,8 +125,10 @@ NamespaceLifeId admitReplacementLife( if (life1.incarnation == kLife2Id) throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Fixture life ids unexpectedly collide"); const NamespaceLifeId life2 = NamespaceLifeId::fromCatalogEntry(life1.ns, kLife2Id); + CasRequests requests = DB::Cas::tests::openRequestsForTest(storage.store()->poolBackendPtr()); + CasOperation op = requests.admit(); CasRefCatalog::casAdmitEntry( - storage.store()->backend(), storage.store()->layout(), storage.store()->poolConfig().gc_shards, CatalogEntry{ + op, storage.store()->layout(), storage.store()->poolConfig().gc_shards, CatalogEntry{ .ns = life2.ns, .state = NsState::Live, .incarnation = life2.incarnation}); return life2; } @@ -179,14 +185,15 @@ TEST(CASNamespaceFileReadContract, DelayedInlineFinalizeCannotChangeSuccessorTok const NamespaceLifeId life2 = replaceCatalogLife(*fixture.storage, life1); fixture.storage->store()->putNamespaceFile(life2, kFile, "life-2-stable\n"); - Backend & backend = fixture.storage->store()->backend(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(fixture.storage->store()->poolBackendPtr()); + CasOperation op = requests.admit(); const Layout & layout = fixture.storage->store()->layout(); const String life1_key = layout.namespaceFileKey(life1, kFile); const String life2_key = layout.namespaceFileKey(life2, kFile); ASSERT_TRUE(std::filesystem::exists(nativeKeyUnder(fixture.object_storage, life2_key))); - const HeadResult life2_before = backend.head(life2_key); - ASSERT_TRUE(life2_before.exists); - const auto life2_body_before = backend.get(life2_key); + const auto life2_before = op.head(life2_key, Retry::standard()); + ASSERT_TRUE(life2_before.has_value()); + const auto life2_body_before = op.read(life2_key, Retry::standard()); ASSERT_TRUE(life2_body_before.has_value()); ASSERT_EQ(life2_body_before->bytes, "life-2-stable\n"); @@ -202,17 +209,17 @@ TEST(CASNamespaceFileReadContract, DelayedInlineFinalizeCannotChangeSuccessorTok EXPECT_NE(e.message().find("retrying later"), String::npos); } - const HeadResult life2_after = backend.head(life2_key); - ASSERT_TRUE(life2_after.exists); - EXPECT_EQ(life2_after.token, life2_before.token); - const auto life2_body_after = backend.get(life2_key); + const auto life2_after = op.head(life2_key, Retry::standard()); + ASSERT_TRUE(life2_after.has_value()); + EXPECT_EQ(life2_after->etag, life2_before->etag); + const auto life2_body_after = op.read(life2_key, Retry::standard()); ASSERT_TRUE(life2_body_after.has_value()); EXPECT_EQ(life2_body_after->bytes, "life-2-stable\n"); if (!stale_failure) { ASSERT_TRUE(std::filesystem::exists(nativeKeyUnder(fixture.object_storage, life1_key))); - const auto life1_body = backend.get(life1_key); + const auto life1_body = op.read(life1_key, Retry::standard()); ASSERT_TRUE(life1_body.has_value()); EXPECT_EQ(life1_body->bytes, "life-1-delayed\n"); } diff --git a/src/Disks/tests/gtest_cas_observability.cpp b/src/Disks/tests/gtest_cas_observability.cpp index bf989f1e3206..2521aae58294 100644 --- a/src/Disks/tests/gtest_cas_observability.cpp +++ b/src/Disks/tests/gtest_cas_observability.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -50,36 +51,37 @@ class RenewalCounterBackend final : public InMemoryBackend LandThenThrow, }; - using InMemoryBackend::putOverwrite; - Fault fault = Fault::None; - PutResult putOverwrite( + /// The fault sits on the WRITE PRIMITIVE, and only on a CONDITIONAL one: the renewal issues + /// `op.replace`, which reaches the store here, and a create on the same key must not consume the + /// one-shot fault. + std::expected write( const String & key, const String & bytes, - const Token & expected, - const ObjectMeta & meta) override + const std::optional & expected_value, + TransportAccess & access) override { + if (!expected_value) + return InMemoryBackend::write(key, bytes, expected_value, access); + const Fault current = std::exchange(fault, Fault::None); if (current == Fault::ThrowBefore) throw Poco::TimeoutException("injected renewal timeout before commit"); - PutResult result = InMemoryBackend::putOverwrite(key, bytes, expected, meta); + auto result = InMemoryBackend::write(key, bytes, expected_value, access); if (current == Fault::LandThenThrow) throw Poco::TimeoutException("injected renewal response loss after commit"); return result; } }; -CasRequestBudget renewalCounterBudget(uint32_t max_attempts = 2) +CasRequestBudget renewalCounterBudget(uint32_t /*max_attempts*/ = 2) { return CasRequestBudget{ .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = max_attempts, .lease_safety_margin_ms = 20, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, + .connect_timeout_cap_ms = std::nullopt, }; } @@ -148,13 +150,20 @@ TEST(CASObservability, RenewalCountersHaveExactPhysicalAndLogicalDeltas) const auto run = [](RenewalCounterBackend::Fault fault, uint64_t attempts, uint64_t retries, uint64_t resolved, uint64_t recovered) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + backend->setAttemptTimeoutMs(renewalCounterBudget().attempt_timeout_ms); + /// Captured by value: `boot_ms` is never mutated in this test, and the Pool can outlive this + /// lambda's own stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + const uint64_t boot_ms = 100; auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "renewal-counter-" + std::to_string(attempts) + "-" + std::to_string(resolved), .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .cas_request_budget = renewalCounterBudget(), - .boot_ms_fn = [&] { return boot_ms; }, + .boot_ms_fn = [] + { + return boot_ms; + }, }); backend->fault = fault; const RenewalCounterSnapshot before = renewalCounters(); @@ -171,18 +180,27 @@ TEST(CASObservability, RenewalCountersHaveExactPhysicalAndLogicalDeltas) TEST(CASObservability, ExternalLeaseDeadlineCountsOnceWithoutReconstructingAttempts) { auto backend = std::make_shared(); - uint64_t boot_ms = 100; + backend->setAttemptTimeoutMs(renewalCounterBudget().attempt_timeout_ms); + /// Held in a shared atomic, not a plain local: this test mutates it below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto boot_ms = std::make_shared>(100); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "renewal-deadline-counter", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .cas_request_budget = renewalCounterBudget(), - .boot_ms_fn = [&] { return boot_ms; }, + .boot_ms_fn = [boot_ms] + { + return boot_ms->load(); + }, }); - /// The confirmed external safety deadline is 1080. At 1071 a ten-millisecond physical attempt - /// no longer fits, so the logical renewal ends without reconstructing a sent attempt. - boot_ms = 1071; + /// The fence deadline is 1100 and the safety margin 20, so admission refuses once fewer than + /// twenty milliseconds of lease remain. At 1090 only ten milliseconds remain, short of the margin + /// however much a single attempt reserves, so nothing can be started and the logical renewal ends + /// without reconstructing a sent attempt. + boot_ms->store(1090); const RenewalCounterSnapshot before = renewalCounters(); EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); const RenewalCounterSnapshot after = renewalCounters(); @@ -199,9 +217,15 @@ TEST(CASObservability, ExternalLeaseDeadlineCountsOnceWithoutReconstructingAttem TEST(CASObservability, StageManifestEmitsManifestPut) { std::shared_ptr b; - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = openPool(b); - s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + s->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); const RootNamespace ns{"srv/tbl@cas@"}; auto build = s->beginPartWrite(PartWriteInfo{.intended_ref = ns.string() + "/all_0_0_0", .intended_namespace = ns}); @@ -212,12 +236,13 @@ TEST(CASObservability, StageManifestEmitsManifestPut) const ManifestId id = build->stageManifest({e}); s->setEventSink(nullptr); - EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const CasEvent & x){ return x.type == CasEventType::ManifestPut; }), 1); - const auto it = std::find_if(seen.begin(), seen.end(), + const auto it = std::find_if(observed.begin(), observed.end(), [](const CasEvent & x){ return x.type == CasEventType::ManifestPut; }); - ASSERT_NE(it, seen.end()); + ASSERT_NE(it, observed.end()); EXPECT_EQ(it->object_kind, CasEventObjectKind::Manifest); EXPECT_EQ(it->object_hash, manifestRefDebugString(id.ref)); EXPECT_FALSE(it->token.empty()); @@ -230,7 +255,10 @@ TEST(CASObservability, StageManifestEmitsManifestPut) TEST(CASObservability, AbandonEmitsPrecommitRemoved) { std::shared_ptr b; - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"srv/tbl@cas@"}; @@ -242,16 +270,20 @@ TEST(CASObservability, AbandonEmitsPrecommitRemoved) const ManifestId id = build->stageManifest({e}); build->precommitAdd(ns, "all_0_0_0", id); - s->setEventSink([&](const CasEvent & x){ seen.push_back(x); }); + s->setEventSink([seen](const CasEvent & x) + { + seen->push(x); + }); build->abandon(); s->setEventSink(nullptr); - EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }), 1); - const auto it = std::find_if(seen.begin(), seen.end(), + const auto it = std::find_if(observed.begin(), observed.end(), [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }); - ASSERT_NE(it, seen.end()); + ASSERT_NE(it, observed.end()); EXPECT_EQ(it->namespace_, ns.string()); EXPECT_EQ(it->ref_name, "all_0_0_0"); EXPECT_EQ(it->object_kind, CasEventObjectKind::Root); @@ -263,7 +295,10 @@ TEST(CASObservability, AbandonEmitsPrecommitRemoved) TEST(CASObservability, AbandonWithoutPrecommitEmitsNoPrecommitRemoved) { std::shared_ptr b; - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"srv/tbl@cas@"}; @@ -274,11 +309,15 @@ TEST(CASObservability, AbandonWithoutPrecommitEmitsNoPrecommitRemoved) e.inline_bytes = "AAA"; build->stageManifest({e}); /// staged, never precommitted - s->setEventSink([&](const CasEvent & x){ seen.push_back(x); }); + s->setEventSink([seen](const CasEvent & x) + { + seen->push(x); + }); build->abandon(); s->setEventSink(nullptr); - EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const CasEvent & x){ return x.type == CasEventType::PrecommitRemoved; }), 0); } @@ -295,15 +334,20 @@ TEST(CASObservability, AbandonWithoutPrecommitEmitsNoPrecommitRemoved) TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) { std::shared_ptr b; - std::vector seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"test/tbl"}; const String P = "republish-payload-audit"; + DB::Cas::tests::OperationForTest head_op(*b); + /// 1. Publish ref r1 -> token A referenced; drop it; ONE GC round condemns A (retired, not deleted). publishOneBlobPart(s, ns, "r1", P); - const HeadResult hA = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hA.exists); + const auto hA = (*head_op).head(s->layout().blobKey(idOf(P)), Retry::standard()); + ASSERT_TRUE(hA.has_value()); s->dropRef(ns, "r1"); s->renewWatermarkOnce(); @@ -320,9 +364,10 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) /// 2. RESURRECT: r2 dedup-hits P while A is condemned -> mints a fresh incarnation B; drop it too. publishOneBlobPart(s, ns, "r2", P); - const HeadResult hB = b->head(s->layout().blobKey(idOf(P))); - ASSERT_TRUE(hB.exists); - ASSERT_NE(hB.token.value, hA.token.value) << "republication must mint a new incarnation token B"; + const auto hB = (*head_op).head(s->layout().blobKey(idOf(P)), Retry::standard()); + ASSERT_TRUE(hB.has_value()); + ASSERT_NE(PersistedEtag::capture(hB->etag).value, PersistedEtag::capture(hA->etag).value) + << "republication must mint a new incarnation token B"; s->dropRef(ns, "r2"); s->renewWatermarkOnce(); @@ -333,7 +378,10 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) const auto condemned_before = global_counters[ProfileEvents::CASGCRetiredCondemned].load(); const auto replaced_before = global_counters[ProfileEvents::CASGCRetireReplaced].load(); - s->setEventSink([&](const CasEvent & e){ seen.push_back(e); }); + s->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); const RoundReport rep = gc.runRegularRound(); s->setEventSink(nullptr); ASSERT_TRUE(rep.acquired_lease); @@ -346,18 +394,22 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) const String hash_hex = DB::Cas::blobIdOf(DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(P))}); const auto is_this_blob = [&](const CasEvent & e){ return e.object_hash == hash_hex; }; - EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [&](const CasEvent & e){ return is_this_blob(e) && e.type == CasEventType::BlobRetire; }), 0) << "supersede must not also emit blob_retire (that is the fresh-condemn hook's event)"; std::vector replaced_events; - std::copy_if(seen.begin(), seen.end(), std::back_inserter(replaced_events), + std::copy_if(observed.begin(), observed.end(), std::back_inserter(replaced_events), [&](const CasEvent & e){ return is_this_blob(e) && e.type == CasEventType::BlobRetireReplaced; }); ASSERT_EQ(replaced_events.size(), 1u) << "exactly one blob_retire_replaced for the supersede"; - EXPECT_EQ(replaced_events[0].token, hB.token.value) << "the event's own token is the fresh CURRENT token B"; + /// The event's token text is dialect-qualified ("emulated:", matching `Etag::render` + /// and `PersistedEtag`'s wire word). + EXPECT_EQ(replaced_events[0].token, hB->etag.render()) + << "the event's own token is the fresh CURRENT token B"; ASSERT_TRUE(replaced_events[0].detail.count("superseded_token")); EXPECT_FALSE(replaced_events[0].detail.at("superseded_token").empty()); - EXPECT_EQ(replaced_events[0].detail.at("superseded_token"), hA.token.value) + EXPECT_EQ(replaced_events[0].detail.at("superseded_token"), hA->etag.render()) << "superseded_token must name the stale token (A) that republication replaced"; EXPECT_EQ(replaced_after - replaced_before, 1u) << "CASGCRetireReplaced increments exactly once"; @@ -375,7 +427,7 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) const auto it = std::find_if(retired.begin(), retired.end(), [&](const RetiredEntry & e){ return e.kind == ObjectKind::Blob && e.ref == DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of(P))}; }); ASSERT_NE(it, retired.end()) << "the superseded entry must be present in the current retired set"; - EXPECT_EQ(it->token.value, hB.token.value) << "the persisted entry names the fresh CURRENT token B"; + EXPECT_EQ(it->token.value, PersistedEtag::capture(hB->etag).value) << "the persisted entry names the fresh CURRENT token B"; EXPECT_EQ(it->size, P.size()) << "supersede must persist the LOGICAL size (payload length, header stripped), matching what " "a fresh condemn of the same blob would carry -- not the raw physical (header-included) size"; @@ -428,7 +480,7 @@ TEST(CASObservability, CaInspectDecodesRefLogToJson) const String json = caInspectToJson( layout, key, encodeRefLogTxn(txn), DB::Cas::tests::fixture::fixtureLife(ns)); EXPECT_NE(json.find("ref_log"), String::npos); - EXPECT_NE(json.find("OwnerTransition"), String::npos); + EXPECT_NE(json.find("owner_transition"), String::npos); EXPECT_NE(json.find("all_0_0_0"), String::npos); } @@ -440,11 +492,16 @@ TEST(CASObservability, CaInspectDecodesPartManifestToJson) PartManifest m; m.ref = ManifestRef{.writer_epoch = 1, .build_sequence = 2, .manifest_ordinal = 3}; m.root_namespace_id = ns; - ManifestEntry e; - e.path = "data.bin"; - e.placement = EntryPlacement::Inline; - e.inline_bytes = "hello"; - m.entries = {e}; + ManifestEntry inline_entry; + inline_entry.path = "data.bin"; + inline_entry.placement = EntryPlacement::Inline; + inline_entry.inline_bytes = "hello"; + ManifestEntry blob_entry; + blob_entry.path = "payload.bin"; + blob_entry.placement = EntryPlacement::Blob; + blob_entry.ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload.bin"))}; + blob_entry.blob_size = 5; + m.entries = {inline_entry, blob_entry}; m.payload_digest = computePayloadDigest(m); const ManifestId id{.root_namespace = ns, .ref = m.ref}; @@ -453,6 +510,9 @@ TEST(CASObservability, CaInspectDecodesPartManifestToJson) EXPECT_NE(json.find("\"root_namespace_id\""), String::npos); EXPECT_NE(json.find("data.bin"), String::npos); EXPECT_NE(json.find("\"manifest_ordinal\":3"), String::npos); + /// `EntryPlacement` renders as its full wire word (`inline`/`blob`), not the enumerator spelling. + EXPECT_NE(json.find(R"("placement":"inline")"), String::npos) << json; + EXPECT_NE(json.find(R"("placement":"blob")"), String::npos) << json; } TEST(CASObservability, CaInspectDecodesMountLeaseToJson) @@ -485,6 +545,36 @@ TEST(CASObservability, CaInspectDecodesGcStateToJson) EXPECT_NE(json.find("\"gc_shards\":4"), String::npos); } +/// `ObjectKind`/`ProvenanceOp` render as their full wire words, not the enumerator spelling. Loops +/// over every `ProvenanceOp` value so each one ends up pinned, not just whichever one a single case +/// would have picked. +TEST(CASObservability, CaInspectDecodesEnvelopeHeaderWithEveryProvenanceOpWord) +{ + Layout layout("p"); + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of("envelope-inspect"))}; + const String key = layout.blobKey(ref); + + const std::vector> ops = { + {ProvenanceOp::Other, "other"}, + {ProvenanceOp::Insert, "insert"}, + {ProvenanceOp::Merge, "merge"}, + {ProvenanceOp::Mutation, "mutation"}, + {ProvenanceOp::Attach, "attach"}, + {ProvenanceOp::Repack, "repack"}, + }; + for (const auto & [op, word] : ops) + { + EnvelopeHeader h; + h.kind = ObjectKind::Blob; + h.provenance = Provenance{.op = op}; + const String bytes = encodeEnvelopeHeader(h, 256); + + const String json = caInspectToJson(layout, key, bytes); + EXPECT_NE(json.find(R"("kind":"blob")"), String::npos) << json; + EXPECT_NE(json.find("\"op\":\"" + word + "\""), String::npos) << json; + } +} + TEST(CASObservability, CaInspectUnknownKeyThrows) { Layout layout("p"); diff --git a/src/Disks/tests/gtest_cas_operation_gate.cpp b/src/Disks/tests/gtest_cas_operation_gate.cpp index 6d16c3946fae..2dbddf6c5f54 100644 --- a/src/Disks/tests/gtest_cas_operation_gate.cpp +++ b/src/Disks/tests/gtest_cas_operation_gate.cpp @@ -113,13 +113,14 @@ const std::string kSrid = "test"; /// gtest_cas_lifecycle_condition.cpp's helper. void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, DB::Cas::Retry::standard()); ASSERT_TRUE(got.has_value()); DB::Cas::MountLease m = DB::Cas::decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, DB::Cas::encodeMountLease(m), got->token).outcome, - DB::Cas::PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(mount_key, DB::Cas::encodeMountLease(m), got->etag, DB::Cas::Retry::standard()))); } } @@ -420,7 +421,7 @@ TEST(CASOperationGate, RemoveThrowsDuringTransientAndDrainsAfterRecovery) << "the gap message must name the transient (auto-recovering) condition: " << gap_msg; /// The lease is restored: the disk self-remounts a fresh incarnation and auto-recovers to Live. - fenceOutMount(pool->backend(), pool->layout().mountKey(kSrid)); + fenceOutMount(*pool->poolBackendPtr(), pool->layout().mountKey(kSrid)); ASSERT_TRUE(pool->tryRemountOnce()) << "the self-remount must reclaim a fresh incarnation"; ASSERT_EQ(pool->lifecycle(), PoolLifecycle::Live) << "the pool must auto-recover to Live"; diff --git a/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp b/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp index 45f82c1a9e33..2c68bf8492fd 100644 --- a/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp +++ b/src/Disks/tests/gtest_cas_orphan_manifest_sweep.cpp @@ -28,6 +28,12 @@ ManifestRef ref(uint64_t seq, uint64_t inst) return ManifestRef{.writer_epoch = kWriterEpoch, .build_sequence = seq, .manifest_ordinal = static_cast(inst)}; } +bool headExists(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + /// The §6 deletion premise (`manifestDeletionPremise`) is a SECOND precondition on every deletion below, /// alongside the watermark eligibility these tests are about: a manifest of an epoch-`E` build is /// deletable only once the namespace's sealed fold cursor sits in an epoch strictly above `E`. Tests @@ -44,16 +50,18 @@ void seedConsumedSealCursor(InMemoryBackend & backend, const Layout & layout, co /// creation would have, rather than relying on the retired sentinel fallback. void seedEmptyRecoveryAuthority(InMemoryBackend & backend, const Layout & layout, const RootNamespace & ns) { - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); const auto entry = std::find_if(catalog.catalog.entries.begin(), catalog.catalog.entries.end(), [&](const CatalogEntry & candidate) { return candidate.ns == ns; }); ASSERT_NE(entry, catalog.catalog.entries.end()); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); - ASSERT_EQ(backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = std::optional{kWriterEpoch}, .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::standard()))); } /// Replaces the catalog row immediately before its second read after arming. The legacy orphan path @@ -63,8 +71,6 @@ void seedEmptyRecoveryAuthority(InMemoryBackend & backend, const Layout & layout class CatalogChangingOnSecondReadBackend : public InMemoryBackend { public: - using Backend::get; - void arm(const Layout & layout, CatalogEntry predecessor_, CatalogEntry successor_) { catalog_key = layout.refCatalogKey(); @@ -76,11 +82,11 @@ class CatalogChangingOnSecondReadBackend : public InMemoryBackend bool didSwitch() const { return did_switch; } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (armed && key == catalog_key && ++catalog_reads == 2) { - const auto current = InMemoryBackend::get(key, range); + const auto current = InMemoryBackend::read(key, access); if (!current) throw std::runtime_error("test catalog disappeared"); RefCatalog next = decodeRefCatalog(current->bytes); @@ -88,11 +94,11 @@ class CatalogChangingOnSecondReadBackend : public InMemoryBackend if (it == next.entries.end()) throw std::runtime_error("test predecessor catalog row disappeared"); *it = successor; - if (InMemoryBackend::casPut(key, encodeRefCatalog(next), current->token).outcome != CasOutcome::Committed) + if (!InMemoryBackend::write(key, encodeRefCatalog(next), current->value, access).has_value()) throw std::runtime_error("test catalog replacement conflicted"); did_switch = true; } - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } private: @@ -110,8 +116,6 @@ class CatalogChangingOnSecondReadBackend : public InMemoryBackend class ReplacingManifestAfterObservationBackend : public InMemoryBackend { public: - using Backend::get; - void arm(const Layout & layout, String manifest_key_) { catalog_key = layout.refCatalogKey(); @@ -124,23 +128,23 @@ class ReplacingManifestAfterObservationBackend : public InMemoryBackend bool didReplace() const { return replaced_manifest; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - const ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); if (armed && prefix == manifests_prefix) listed_page = true; return page; } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - const auto result = InMemoryBackend::get(key, range); + auto result = InMemoryBackend::read(key, access); if (armed && listed_page && !replaced_manifest && key == catalog_key) { - const auto current = InMemoryBackend::get(manifest_key); + const auto current = InMemoryBackend::read(manifest_key, access); if (!current) throw std::runtime_error("test manifest disappeared before replacement"); - if (InMemoryBackend::casPut(manifest_key, current->bytes, current->token).outcome != CasOutcome::Committed) + if (!InMemoryBackend::write(manifest_key, current->bytes, current->value, access).has_value()) throw std::runtime_error("test manifest replacement conflicted"); replaced_manifest = true; } @@ -166,12 +170,12 @@ TEST(CASOrphanManifestSweep, EligibleAndUnownedIsDeleted) registerNamespaceRaw(*backend, store->layout(), ns); const ManifestRef r = ref(5, 0xAB); writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); // body, no owner - setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active*/6); // 6 > 5 => eligible + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active_build_sequence*/6); // 6 > 5 => eligible seedConsumedSealCursor(*backend, store->layout(), ns); seedEmptyRecoveryAuthority(*backend, store->layout(), ns); sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_FALSE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } /// The orphan sweep must not turn a forged same-id snapshot at an OLDER `EpochSeal` into an empty owner @@ -184,7 +188,9 @@ TEST(CASOrphanManifestSweep, CheckpointSnapshotAtOlderEpochSealSkipsDeletion) const Layout & layout = store->layout(); const RootNamespace ns{"00/sweep-checkpoint-base-seal@cas@"}; fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const RefLogTxn birth{ .ns = ns.string(), .txn_id = RefTxnId{1, 1}, .ops = {namespaceBirthOp()}, @@ -205,21 +211,21 @@ TEST(CASOrphanManifestSweep, CheckpointSnapshotAtOlderEpochSealSkipsDeletion) applyRefLogTxn(through_seal, birth); applyRefLogTxn(through_seal, seal_txn); writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{2, 1}, .checkpoint_snapshot_id = RefTxnId{1, 2}, - .last_epoch_seal = RefTxnId{2, 1}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{2, 1}}), Retry::standard()))); const ManifestRef candidate = ref(5, 0xAC); const String candidate_key = layout.manifestKey(ManifestId{ns, candidate}); writeManifestRaw(*backend, layout, ns, candidate, {blobEntryFor("a", DB::UInt128(1))}); - setWatermarkMinActive(*backend, layout, kServerRoot, kWriterEpoch, /*min_active=*/6); + setWatermarkMinActive(*backend, layout, kServerRoot, kWriterEpoch, /*min_active_build_sequence=*/6); seedConsumedSealCursor(*backend, layout, ns); std::vector warnings; EXPECT_EQ(sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}, &warnings), 0u); - EXPECT_TRUE(backend->head(candidate_key).exists); + EXPECT_TRUE(headExists(*backend, candidate_key)); ASSERT_FALSE(warnings.empty()); } @@ -235,7 +241,7 @@ TEST(CASOrphanManifestSweep, OwnedBodyIsSkipped) setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } /// GC-WEDGE regression (2026-07-10): a COMMITTED ref that has been DROPPED but whose removal `-1` is NOT @@ -265,7 +271,7 @@ TEST(CASOrphanManifestSweep, PendingCommittedRemovalBodyIsSkipped) setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); // 6 > 5 => prefix eligible sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))) << "a dropped-but-unsealed committed manifest body must survive the sweep (delete-after-sealed-" "decrements) — else the removal-fold clamps forever on the missing body (GC-WEDGE-2026-07-10)"; } @@ -299,7 +305,7 @@ TEST(CASOrphanManifestSweep, CursorPageAdvancesAndWrapsWithListBudget) const ManifestRef r2 = ref(5, 0xE2); writeManifestRaw(*backend, store->layout(), ns, r1, {blobEntryFor("a", DB::UInt128(1))}); writeManifestRaw(*backend, store->layout(), ns, r2, {blobEntryFor("b", DB::UInt128(2))}); - setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active*/6); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, /*min_active_build_sequence*/6); const ManifestSweepResult first = sweepManifestCursorPageForTest(*store, "", /*list_budget*/1, /*delete_budget*/0); EXPECT_EQ(first.listed, 1u); @@ -322,7 +328,7 @@ TEST(CASOrphanManifestSweep, NoWatermarkIsNotAuthority) writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); // No setWatermarkMinActive — no durable fact => not eligible. sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } TEST(CASOrphanManifestSweep, CursorPageDeletesEligibleUnownedBody) @@ -340,7 +346,7 @@ TEST(CASOrphanManifestSweep, CursorPageDeletesEligibleUnownedBody) const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); EXPECT_GE(result.listed, 1u); EXPECT_EQ(result.deleted, 1u); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_FALSE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } TEST(CASOrphanManifestSweep, CursorPageRespectsDeleteBudget) @@ -359,8 +365,8 @@ TEST(CASOrphanManifestSweep, CursorPageRespectsDeleteBudget) const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/1); EXPECT_EQ(result.deleted, 1u); - const bool first_exists = backend->head(store->layout().manifestKey(ManifestId{ns, r1})).exists; - const bool second_exists = backend->head(store->layout().manifestKey(ManifestId{ns, r2})).exists; + const bool first_exists = headExists(*backend, store->layout().manifestKey(ManifestId{ns, r1})); + const bool second_exists = headExists(*backend, store->layout().manifestKey(ManifestId{ns, r2})); EXPECT_NE(first_exists, second_exists); } @@ -380,12 +386,14 @@ TEST(CASOrphanManifestSweep, CursorPageDeletesObservedBodyWhenCatalogOmitsNamesp const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); EXPECT_EQ(result.deleted, 1u); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_FALSE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } -/// The candidate body and token must be frozen before the later catalog cut. A concurrent same-key -/// replacement after that observation is a new physical incarnation and must lose the old-token delete. -TEST(CASOrphanManifestSweep, CursorPageCannotDeleteManifestReplacedAfterObservation) +/// Manifest keys are write-once, so a same-key rewrite after the observation below never happens in +/// production; this backend forces one anyway to prove the page does not need the old freeze-before-cut +/// discipline to stay correct. The body is read only after the catalog cut, so it sees whatever +/// incarnation is actually there at that point and deletes it under its own current token. +TEST(CASOrphanManifestSweep, CursorPageDeletesTheIncarnationSeenAfterTheCatalogCut) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); @@ -402,8 +410,8 @@ TEST(CASOrphanManifestSweep, CursorPageCannotDeleteManifestReplacedAfterObservat const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); EXPECT_TRUE(backend->didReplace()); - EXPECT_EQ(result.deleted, 0u); - EXPECT_TRUE(backend->head(key).exists); + EXPECT_EQ(result.deleted, 1u); + EXPECT_FALSE(headExists(*backend, key)); } /// Any duplicate current life id makes the catalog-to-physical join ambiguous. The cursor page is @@ -421,19 +429,21 @@ TEST(CASOrphanManifestSweep, CursorPageRefusesAmbiguousCatalogLifeIndex) seedConsumedSealCursor(*backend, store->layout(), ns); seedEmptyRecoveryAuthority(*backend, store->layout(), ns); - const CasRefCatalog::Snapshot before = CasRefCatalog::read(*backend, store->layout()); + CasOperation op = store->openRequests().admit(); + const CasRefCatalog::Snapshot before = CasRefCatalog::read(op, store->layout()); RefCatalog damaged = before.catalog; CatalogEntry duplicate = damaged.entries.front(); duplicate.ns = RootNamespace{"00/ambiguous-life-twin@cas@"}; damaged.entries.push_back(duplicate); std::sort(damaged.entries.begin(), damaged.entries.end(), [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); - ASSERT_EQ(backend->casPut(store->layout().refCatalogKey(), encodeRefCatalog(damaged), before.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(before.etag.has_value()); + ASSERT_TRUE(std::holds_alternative( + op.replace(store->layout().refCatalogKey(), encodeRefCatalog(damaged), *before.etag, Retry::standard()))); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); }); - EXPECT_TRUE(backend->head(key).exists); + EXPECT_TRUE(headExists(*backend, key)); } TEST(CASOrphanManifestSweep, CursorPageSkipsOwnedBody) @@ -448,7 +458,7 @@ TEST(CASOrphanManifestSweep, CursorPageSkipsOwnedBody) const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); EXPECT_EQ(result.deleted, 0u); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))); } /// A catalog-named life cannot be treated as an empty table merely because its mandatory recovery @@ -465,12 +475,13 @@ TEST(CASOrphanManifestSweep, MissingRequiredCheckpointSuppressesDestructiveDecis setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 6); seedConsumedSealCursor(*backend, store->layout(), ns); - const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); - ASSERT_FALSE(readCkpt(*backend, store->layout(), NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation))); + CasOperation op = store->openRequests().admit(); + const CatalogEntry entry = CasRefCatalog::read(op, store->layout()).catalog.entries.front(); + ASSERT_FALSE(readCkpt(op, store->layout(), NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation))); sweepNamespace(*store, ns, BuildPrefix{.writer_epoch = kWriterEpoch, .build_sequence = 5}); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))) << "without the exact _ckpt required by a Live catalog row, the sweep must retain rather than " "derive an empty owner set"; } @@ -486,7 +497,8 @@ TEST(CASOrphanManifestSweep, EpochSealFoldCursorCrossesTailByExactDecodedSuccess auto store = openPoolForTest(backend); const RootNamespace ns{"00/seal-cursor-tail@cas@"}; fixture::admitLive(*backend, store->layout(), ns); - const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + CasOperation op = store->openRequests().admit(); + const CatalogEntry entry = CasRefCatalog::read(op, store->layout()).catalog.entries.front(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); const ManifestRef removed{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; @@ -501,27 +513,27 @@ TEST(CASOrphanManifestSweep, EpochSealFoldCursorCrossesTailByExactDecodedSuccess writeSealAt(*backend, store->layout(), ns, RefTxnId{5, 1}, RefTxnId{4, 1}); writeSealAt(*backend, store->layout(), ns, RefTxnId{6, 1}, RefTxnId{5, 1}); for (uint64_t epoch = 3; epoch <= 6; ++epoch) - ASSERT_TRUE(backend->head(store->layout().refLogKey(life, RefTxnId{epoch, 1})).exists) + ASSERT_TRUE(headExists(*backend, store->layout().refLogKey(life, RefTxnId{epoch, 1}))) << "fixture must deposit every intermediate exact successor in the catalog life"; writeTxnAt(*backend, store->layout(), ns, RefTxnId{7, 1}, {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "dropped", removed}, std::nullopt)}, RefTxnId{6, 1}); writeSealAt(*backend, store->layout(), ns, RefTxnId{7, 2}); writeManifestRaw(*backend, store->layout(), ns, unowned, {blobEntryFor("unowned", DB::UInt128(0xA2))}); - ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{7, 2}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = RefTxnId{7, 2}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{7, 2}}), Retry::standard()))); setWatermarkMinActive(*backend, store->layout(), kServerRoot, kWriterEpoch, 7); seedFoldCursorForTest(*backend, store->layout(), ns, RefTxnId{2, 2}); const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); EXPECT_EQ(result.deleted, 1u); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, removed})).exists) + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, removed}))) << "the exact successor of the folded epoch seal contains this body's unconsumed -1"; - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, unowned})).exists) + EXPECT_FALSE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, unowned}))) << "an unrelated eligible body must still drain; retaining it would mask a geometry failure"; } @@ -534,15 +546,15 @@ TEST(CASOrphanManifestSweep, MissingImmediateEpochAfterCleanedCursorCannotBeSkip auto store = openPoolForTest(backend); const RootNamespace ns{"00/missing-next-epoch@cas@"}; fixture::admitLive(*backend, store->layout(), ns); - const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + CasOperation op = store->openRequests().admit(); + const CatalogEntry entry = CasRefCatalog::read(op, store->layout()).catalog.entries.front(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); const RefTxnId cursor{2, 2}; writeSealAt(*backend, store->layout(), ns, cursor); - const HeadResult cursor_head = backend->head(store->layout().refLogKey(life, cursor)); - ASSERT_TRUE(cursor_head.exists); - ASSERT_EQ(classifyDeleteOutcome( - backend->deleteExact(store->layout().refLogKey(life, cursor), cursor_head.token)), DeleteClass::Deleted); + const auto cursor_head = op.head(store->layout().refLogKey(life, cursor), Retry::standard()); + ASSERT_TRUE(cursor_head.has_value()); + ASSERT_EQ(op.remove(store->layout().refLogKey(life, cursor), cursor_head->etag, Retry::standard()), Removal::Removed); const ManifestRef phantom{.writer_epoch = 7, .build_sequence = 1, .manifest_ordinal = 1}; /// The codec refuses this skipped predecessor when a writer tries to create it. Inject the malformed @@ -553,24 +565,23 @@ TEST(CASOrphanManifestSweep, MissingImmediateEpochAfterCleanedCursorCannotBeSkip .ops = publishCommittedOps("phantom", phantom), .prev_epoch_seal = RefTxnId{6, 1}}; String malformed_later_link = encodeRefLogTxn(direct_later_link); - const String encoded_predecessor{R"("!pse":"6")"}; + const String encoded_predecessor{R"("!prev_epoch":"6")"}; const size_t predecessor_pos = malformed_later_link.find(encoded_predecessor); ASSERT_NE(predecessor_pos, String::npos); malformed_later_link.replace( - predecessor_pos, encoded_predecessor.size(), R"("!pse":"2")"); - ASSERT_EQ(backend->putIfAbsent( - store->layout().refLogKey(life, RefTxnId{7, 1}), sealObject(FormatId::RefLog, malformed_later_link)).outcome, - PutOutcome::Done); + predecessor_pos, encoded_predecessor.size(), R"("!prev_epoch":"2")"); + ASSERT_TRUE(std::holds_alternative(op.create( + store->layout().refLogKey(life, RefTxnId{7, 1}), sealObject(FormatId::RefLog, malformed_later_link), Retry::standard()))); writeRefSnapshotRaw(*backend, store->layout(), RefTableSnapshot{ .ns = ns.string(), .snapshot_id = RefTxnId{7, 1}, .committed = {committedRow("phantom", phantom)}, .precommits = {}}); - ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{7, 1}, .checkpoint_snapshot_id = RefTxnId{7, 1}, - .last_epoch_seal = RefTxnId{6, 1}})).outcome, PutOutcome::Done); + .last_epoch_seal = RefTxnId{6, 1}}), Retry::standard()))); const ManifestRef victim{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; writeManifestRaw(*backend, store->layout(), ns, victim, {blobEntryFor("victim", DB::UInt128(0xC1))}); @@ -580,7 +591,7 @@ TEST(CASOrphanManifestSweep, MissingImmediateEpochAfterCleanedCursorCannotBeSkip const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); EXPECT_EQ(result.deleted, 0u); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, victim})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, victim}))); } /// Control for the cleaned-cursor path: the exact immediately-next epoch head exists and names the @@ -594,15 +605,15 @@ TEST(CASOrphanManifestSweep, CleanedCursorCrossesOnlyThroughExactImmediateEpochH auto store = openPoolForTest(backend); const RootNamespace ns{"00/exact-next-epoch@cas@"}; fixture::admitLive(*backend, store->layout(), ns); - const CatalogEntry entry = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + CasOperation op = store->openRequests().admit(); + const CatalogEntry entry = CasRefCatalog::read(op, store->layout()).catalog.entries.front(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); const RefTxnId cursor{2, 2}; writeSealAt(*backend, store->layout(), ns, cursor); - const HeadResult cursor_head = backend->head(store->layout().refLogKey(life, cursor)); - ASSERT_TRUE(cursor_head.exists); - ASSERT_EQ(classifyDeleteOutcome( - backend->deleteExact(store->layout().refLogKey(life, cursor), cursor_head.token)), DeleteClass::Deleted); + const auto cursor_head = op.head(store->layout().refLogKey(life, cursor), Retry::standard()); + ASSERT_TRUE(cursor_head.has_value()); + ASSERT_EQ(op.remove(store->layout().refLogKey(life, cursor), cursor_head->etag, Retry::standard()), Removal::Removed); const ManifestRef removed{.writer_epoch = 1, .build_sequence = 5, .manifest_ordinal = 1}; writeTxnAt(*backend, store->layout(), ns, RefTxnId{3, 1}, @@ -612,11 +623,11 @@ TEST(CASOrphanManifestSweep, CleanedCursorCrossesOnlyThroughExactImmediateEpochH {ownerTransitionOp(RefOwnerBinding{RefOwnerKind::Committed, "absent-anchor", absent_anchor}, std::nullopt)}); writeRefSnapshotRaw(*backend, store->layout(), RefTableSnapshot{ .ns = ns.string(), .snapshot_id = RefTxnId{3, 2}, .committed = {}, .precommits = {}}); - ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(op.create(store->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{3, 2}, .checkpoint_snapshot_id = RefTxnId{3, 2}, - .last_epoch_seal = cursor})).outcome, PutOutcome::Done); + .last_epoch_seal = cursor}), Retry::standard()))); const ManifestRef unowned{.writer_epoch = 1, .build_sequence = 6, .manifest_ordinal = 1}; writeManifestRaw(*backend, store->layout(), ns, removed, {blobEntryFor("removed", DB::UInt128(0xC2))}); @@ -627,8 +638,8 @@ TEST(CASOrphanManifestSweep, CleanedCursorCrossesOnlyThroughExactImmediateEpochH const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget=*/100, /*delete_budget=*/10); EXPECT_EQ(result.deleted, 1u); - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, removed})).exists); - EXPECT_FALSE(backend->head(store->layout().manifestKey(ManifestId{ns, unowned})).exists); + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, removed}))); + EXPECT_FALSE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, unowned}))); } /// The catalog row used to obtain coverage and the life used to recover ownership must be ONE frozen @@ -640,7 +651,8 @@ TEST(CASOrphanManifestSweep, LaterCatalogCutCannotSpliceOwnershipAuthority) auto store = openPoolForTest(backend); const RootNamespace ns{"00/frozen-catalog-cut@cas@"}; fixture::admitLive(*backend, store->layout(), ns); - const CatalogEntry predecessor = CasRefCatalog::read(*backend, store->layout()).catalog.entries.front(); + CasOperation op = store->openRequests().admit(); + const CatalogEntry predecessor = CasRefCatalog::read(op, store->layout()).catalog.entries.front(); const ManifestRef r = ref(5, 0xB1); writeManifestRaw(*backend, store->layout(), ns, r, {blobEntryFor("a", DB::UInt128(1))}); @@ -656,7 +668,7 @@ TEST(CASOrphanManifestSweep, LaterCatalogCutCannotSpliceOwnershipAuthority) EXPECT_FALSE(backend->didSwitch()) << "the sweep must not resolve a second catalog cut after it starts using the frozen entry"; - EXPECT_TRUE(backend->head(store->layout().manifestKey(ManifestId{ns, r})).exists) + EXPECT_TRUE(headExists(*backend, store->layout().manifestKey(ManifestId{ns, r}))) << "the committed predecessor manifest must remain protected by the same frozen authority cut"; } diff --git a/src/Disks/tests/gtest_cas_orphan_nomination.cpp b/src/Disks/tests/gtest_cas_orphan_nomination.cpp index d79f10bf73f7..fb4cd6bd7a81 100644 --- a/src/Disks/tests/gtest_cas_orphan_nomination.cpp +++ b/src/Disks/tests/gtest_cas_orphan_nomination.cpp @@ -29,27 +29,28 @@ ManifestRef candidateRef() bool manifestExists(Backend & backend, const Layout & layout, const ManifestId & id) { - return backend.head(layout.manifestKey(id)).exists; + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(layout.manifestKey(id), Retry::standard()).has_value(); } -bool activeSourceExists(Backend & backend, const Layout & layout, const UInt128 & source_id) +bool activeSourceExists(CasOperation & op, const Layout & layout, const UInt128 & source_id) { - const auto state_got = backend.get(layout.gcStateKey()); + const auto state_got = op.read(layout.gcStateKey(), Retry::standard()); if (!state_got) return false; const GcState state = decodeGcState(state_got->bytes); - const auto seal_got = backend.get(layout.foldSealKey(state.snap_generation, state.snap_attempt)); + const auto seal_got = op.read(layout.foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard()); if (!seal_got) return false; const CasFoldSeal seal = decodeFoldSeal(seal_got->bytes); for (const RunRef & run : seal.blob_target_runs) { - SourceEdgeRunView view = openSourceEdgeRun(backend, run.key); + SourceEdgeRunView view = openSourceEdgeRun(op, run.key); String key; String payload; while (view.next(key, payload)) { - if (payload.empty() || payload[0] != kEdgeActive) + if (payload.empty() || runMarkerFromByte(payload[0], "CAS test source-edge run") != RunMarker::Edge) continue; BlobRef ref; UInt128 row_source{}; @@ -62,44 +63,45 @@ bool activeSourceExists(Backend & backend, const Layout & layout, const UInt128 return false; } -size_t condemnedCount(Backend & backend, const Layout & layout) +size_t condemnedCount(CasOperation & op, const Layout & layout) { size_t count = 0; - const GcState state = decodeGcState(backend.get(layout.gcStateKey())->bytes); + const GcState state = decodeGcState(op.read(layout.gcStateKey(), Retry::standard())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend.get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + op.read(layout.foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard())->bytes); for (const RunRef & run : seal.blob_target_runs) { - SourceEdgeRunView view = openSourceEdgeRun(backend, run.key); + SourceEdgeRunView view = openSourceEdgeRun(op, run.key); String key; String payload; while (view.next(key, payload)) - count += !payload.empty() && payload[0] == kCondemned; + count += !payload.empty() && runMarkerFromByte(payload[0], "CAS test source-edge run") == RunMarker::Condemned; view.verifyAgainst(run.checksum); } return count; } -class NominationBackend : public InMemoryBackend +class NominationBackend : public CountingBackend { public: - using Backend::deleteExact; - using Backend::get; - using Backend::putOverwrite; - - DeleteOutcome deleteExact(const String & key, const Token & token) override + /// The sweep's exact-token delete reaches the store through the keyed removal, so the fault is + /// armed there. It runs before the base call takes the backend's lock, so probing this same + /// backend from inside it is safe. + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { if (key == watched_manifest_key) { - source_absent_when_delete_started = !activeSourceExists(*this, layout, watched_source_id); + CasRequests probe_requests = openRequestsForTest(*this); + CasOperation probe = probe_requests.admit(); + source_absent_when_delete_started = !activeSourceExists(probe, layout, watched_source_id); if (replace_manifest_before_delete) { - const auto got = get(key); + const auto got = InMemoryBackend::read(key, access); if (got) - putOverwrite(key, got->bytes, got->token); + static_cast(InMemoryBackend::write(key, got->bytes, got->value, access)); } } - return InMemoryBackend::deleteExact(key, token); + return CountingBackend::remove(key, expected_value, access); } Layout layout{"p"}; @@ -150,7 +152,7 @@ ReadyFixture makeReadyFixture() .last_epoch_seal = RefTxnId{1, 2}, }); EXPECT_TRUE(runRegularRoundReclaiming(*f.gc).acquired_lease); - setWatermarkMinActive(*f.backend, f.store->layout(), "test", kCandidateEpoch, /*min_active=*/6); + setWatermarkMinActive(*f.backend, f.store->layout(), "test", kCandidateEpoch, /*min_active_build_sequence=*/6); std::vector entries; std::vector seeded_edges; @@ -176,11 +178,12 @@ ReadyFixture makeReadyFixture() /// Seed the exact S42 precondition: the candidate manifest's `+1` edges are already in the adopted /// run, yet the recovered owner view does not name the body. Four blobs also have another source. - const auto state_got = f.backend->get(f.store->layout().gcStateKey()); + CasOperation seed_op = f.store->openRequests().admit(); + const auto state_got = seed_op.read(f.store->layout().gcStateKey(), Retry::standard()); EXPECT_TRUE(state_got.has_value()); GcState state = decodeGcState(state_got->bytes); - const auto parent_got = f.backend->get( - f.store->layout().foldSealKey(state.snap_generation, state.snap_attempt)); + const auto parent_got = seed_op.read( + f.store->layout().foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard()); EXPECT_TRUE(parent_got.has_value()); CasFoldSeal seal = decodeFoldSeal(parent_got->bytes); const uint64_t new_generation = state.snap_generation + 1; @@ -188,7 +191,7 @@ ReadyFixture makeReadyFixture() std::vector runs; RetiredMergeResult retired; foldDeltasIntoGeneration( - *f.backend, f.store->layout(), seal.blob_target_runs, + seed_op, f.store->layout(), seal.blob_target_runs, new_generation, new_attempt, /*shard=*/0, std::move(seeded_edges), runs, /*current_round=*/state.round, /*condemn_round=*/state.round, {}, {}, {}, &retired, /*suppress_destructive=*/false, nullptr); @@ -197,10 +200,10 @@ ReadyFixture makeReadyFixture() seal.blob_target_runs = std::move(runs); seal.condemned_summary[0] = CondemnedSummary{}; putDeterministicArtifact( - *f.backend, f.store->layout().foldSealKey(new_generation, new_attempt), encodeFoldSeal(seal)); + seed_op, f.store->layout().foldSealKey(new_generation, new_attempt), encodeFoldSeal(seal)); state.snap_generation = new_generation; state.snap_attempt = new_attempt; - f.backend->putOverwrite(f.store->layout().gcStateKey(), encodeGcState(state), state_got->token); + seed_op.replace(f.store->layout().gcStateKey(), encodeGcState(state), state_got->etag, Retry::standard()); f.backend->watched_manifest_key = f.store->layout().manifestKey(f.candidate); f.backend->watched_source_id = sourceEdgeId(f.candidate, "blob-0"); @@ -227,14 +230,15 @@ TEST(CASOrphanNomination, RetiresExactManifestSourcesBeforeDelete) EXPECT_FALSE(manifestExists(*f.backend, f.store->layout(), f.candidate)); EXPECT_TRUE(f.backend->source_absent_when_delete_started) << "the adopted in-degree run must retire the manifest source before exact deletion begins"; + CasOperation op = f.store->openRequests().admit(); for (size_t i = 0; i < f.blobs.size(); ++i) { EXPECT_FALSE(activeSourceExists( - *f.backend, f.store->layout(), sourceEdgeId(f.candidate, "blob-" + std::to_string(i)))); + op, f.store->layout(), sourceEdgeId(f.candidate, "blob-" + std::to_string(i)))); EXPECT_EQ(inDegreeInRuns(*f.backend, runsForShard(*f.backend, f.store->layout(), 0), f.blobs[i]), i < 4 ? 1 : 0); } - EXPECT_EQ(condemnedCount(*f.backend, f.store->layout()), 2u); + EXPECT_EQ(condemnedCount(op, f.store->layout()), 2u); ASSERT_TRUE(fold_reduce.has_value()); EXPECT_EQ(fold_reduce->metrics.at("unmatched_removes"), 0u) @@ -244,14 +248,40 @@ TEST(CASOrphanNomination, RetiresExactManifestSourcesBeforeDelete) "committed+produced ref transaction unapplied"; } +/// A page reads the body of a candidate only. Five live manifests share the namespace with the one +/// orphan candidate; they are active under the floor, are retained from their keys alone, and cost no +/// GET. The candidate costs exactly one. +TEST(CASOrphanNomination, OnlyCandidatesCostABodyRead) +{ + ReadyFixture f = makeReadyFixture(); + const Layout & layout = f.store->layout(); + std::vector live_keys; + for (uint32_t ordinal = 1; ordinal <= 5; ++ordinal) + { + const ManifestRef live{.writer_epoch = kCandidateEpoch, .build_sequence = 7, .manifest_ordinal = ordinal}; + writeManifestRaw(*f.backend, layout, f.ns, live, {blobEntryFor("live", DB::UInt128(0xA000 + ordinal))}); + live_keys.push_back(layout.manifestKey(ManifestId{f.ns, live})); + } + f.backend->resetCounts(); + + const ManifestSweepResult result = planManifestCursorPage( + *f.store, "", /*list_budget=*/100, /*nomination_budget=*/100, /*catalog_recovery_authoritative=*/true, nullptr); + + ASSERT_EQ(result.nominations.size(), 1u); + EXPECT_EQ(f.backend->getCount(layout.manifestKey(f.candidate)), 1u); + for (const String & key : live_keys) + EXPECT_EQ(f.backend->getCount(key), 0u) << key; +} + /// A nomination must exact-GET and decode the manifest before it can derive any source-edge identity. /// An undecodable body is retained and surfaced without aborting the rest of the round. TEST(CASOrphanNomination, CorruptManifestIsRetainedAndSurfaced) { ReadyFixture f = makeReadyFixture(); - const auto got = f.backend->get(f.backend->watched_manifest_key); + DB::Cas::tests::OperationForTest op(f.backend); + const auto got = (*op).read(f.backend->watched_manifest_key, Retry::standard()); ASSERT_TRUE(got.has_value()); - f.backend->putOverwrite(f.backend->watched_manifest_key, "not a sealed manifest", got->token); + (*op).replace(f.backend->watched_manifest_key, "not a sealed manifest", got->etag, Retry::standard()); std::optional orphan_sweep; f.gc->setPhaseSink([&](const GcPhaseRecord & rec) { if (rec.phase == "orphan_sweep") orphan_sweep = rec; }); @@ -261,7 +291,10 @@ TEST(CASOrphanNomination, CorruptManifestIsRetainedAndSurfaced) EXPECT_TRUE(report.acquired_lease); ASSERT_TRUE(orphan_sweep.has_value()); EXPECT_EQ(orphan_sweep->metrics.at("undecodable"), 1u); - EXPECT_TRUE(f.backend->head(f.backend->watched_manifest_key).exists); + { + DB::Cas::tests::OperationForTest head_op(f.backend); + EXPECT_TRUE((*head_op).head(f.backend->watched_manifest_key, Retry::standard()).has_value()); + } } /// Manifest identities are immutable. A changed token at the same key is illegal ABA, not an ordinary @@ -272,7 +305,10 @@ TEST(CASOrphanNomination, TokenAbaIsRetainedAndSurfaced) f.backend->replace_manifest_before_delete = true; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { runRegularRoundReclaiming(*f.gc); }); - EXPECT_TRUE(f.backend->head(f.backend->watched_manifest_key).exists); + { + DB::Cas::tests::OperationForTest head_op(f.backend); + EXPECT_TRUE((*head_op).head(f.backend->watched_manifest_key, Retry::standard()).has_value()); + } } /// Nomination PLANNING itself is gated on `!suppress_destructive` @@ -303,19 +339,21 @@ TEST(CASOrphanNomination, SuppressedRoundNominatesNothing) TEST(CASOrphanNomination, SourceRetirementIsAccountingNeutral) { InMemoryBackend backend; + CasRequests requests = openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const BlobRef blob = legacyMetaTestRef(UInt128(0xA001)); const UInt128 source = UInt128(0xA002); std::vector parent_runs; foldDeltasIntoGeneration( - backend, layout, {}, /*new_generation=*/1, /*attempt=*/1, /*shard=*/0, + op, layout, {}, /*new_generation=*/1, /*attempt=*/1, /*shard=*/0, {BlobDelta{.ref = blob, .source_id = source, .remove = false}}, parent_runs); std::vector next_runs; RetiredMergeResult retired; std::vector applied{0x5A}; foldDeltasIntoGeneration( - backend, layout, parent_runs, /*new_generation=*/2, /*attempt=*/2, /*shard=*/0, + op, layout, parent_runs, /*new_generation=*/2, /*attempt=*/2, /*shard=*/0, {}, next_runs, /*current_round=*/1, /*condemn_round=*/1, {}, {}, {}, &retired, /*suppress_destructive=*/false, &applied, {BlobSourceRetirement{.ref = blob, .source_id = source}, diff --git a/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp b/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp new file mode 100644 index 000000000000..34e87af705ed --- /dev/null +++ b/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp @@ -0,0 +1,390 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} + +namespace ProfileEvents +{ + extern const Event CASGCReadAheadWasted; + extern const Event CASGCReadAheadHit; +} + +/// The orphan-manifest sweep's request shape per page. The floor a namespace's builds are judged +/// against is one mount body per server root, so a page reads it once per namespace, not once per +/// listed build; the tests below count the mount-key reads and pin the pure eligibility predicate. + +using namespace DB::Cas; +using namespace DB::Cas::tests; + +namespace +{ + +/// A three-segment namespace, the shape a real table gets (`/store/<3hex>/@cas@`), so the +/// floor lookup has three `/`-prefixes to try and two of them miss. +const RootNamespace kNs{"test/store/465/aa@cas@"}; + +ManifestRef build(uint64_t seq) +{ + return ManifestRef{.writer_epoch = 1, .build_sequence = seq, .manifest_ordinal = 1}; +} + +struct PageFixture +{ + std::shared_ptr backend = std::make_shared(); + PoolPtr store; + uint64_t manifests = 0; + + explicit PageFixture(uint64_t manifests_, uint64_t min_active_build_sequence) + : manifests(manifests_) + { + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.manifest_sweep_list_budget_keys = 1000; + config.manifest_sweep_delete_budget_keys = 100; + config.gc_fold_max_defer_rounds = 0; + store = Pool::open(backend, config); + const Layout & layout = store->layout(); + casAdmitEntry(*backend, layout, kNs); + /// One committed birth log at {1,1} and a checkpoint naming it, so the namespace has a protection + /// view and every eligible key reaches the premise (which retains it for lack of fold coverage). + /// The live ref's own manifest occupies build_sequence == manifests (it is itself one of the + /// `manifests` listed objects, always active); the loop below fills the debris below it. + publishAt(*backend, layout, kNs, RefTxnId{1, 1}, "live", /*build_sequence=*/manifests, + DB::UInt128(0x7001), /*birth=*/true); + writeRecoverableCkptForRawFixture(*backend, layout, kNs, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt, + }); + for (uint64_t seq = 1; seq < manifests; ++seq) + writeManifestRaw(*backend, layout, kNs, build(seq), {blobEntryFor("a", DB::UInt128(0x100 + seq))}); + setWatermarkMinActive(*backend, layout, "test", /*writer_epoch=*/1, min_active_build_sequence); + backend->resetCounts(); + } + + ManifestSweepResult page() + { + return planManifestCursorPage(*store, "", /*list_budget=*/1000, /*nomination_budget=*/100, + /*catalog_recovery_authoritative=*/true, nullptr); + } +}; + +} + +TEST(CASOrphanSweepRequests, FloorIsReadOncePerNamespacePerPage) +{ + PageFixture f(/*manifests=*/50, /*min_active=*/25); + const Layout & layout = f.store->layout(); + const ManifestSweepResult result = f.page(); + + EXPECT_EQ(result.listed, 50u); + EXPECT_EQ(result.floor_lookups, 1u); + EXPECT_EQ(result.floor_reads, 3u); + EXPECT_EQ(f.backend->getCount(layout.mountKey("test/store/465")), 1u); + EXPECT_EQ(f.backend->getCount(layout.mountKey("test/store")), 1u); + EXPECT_EQ(f.backend->getCount(layout.mountKey("test")), 1u); + /// Builds 1..24 are retired under the floor and reach the premise, which retains them for lack of + /// coverage; builds 25..50 are active and never get that far. + EXPECT_EQ(result.retained_no_coverage, 24u); + EXPECT_TRUE(result.nominations.empty()); +} + +TEST(CASOrphanSweepRequests, AbsentFloorRetainsEverythingWithOneLookup) +{ + PageFixture f(/*manifests=*/10, /*min_active=*/100); + const Layout & layout = f.store->layout(); + { + OperationForTest op(*f.backend); + const auto h = (*op).head(layout.mountKey("test"), Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*op).remove(layout.mountKey("test"), h->etag, Retry::once()), Removal::Removed); + } + f.backend->resetCounts(); + const ManifestSweepResult result = f.page(); + + EXPECT_EQ(result.floor_lookups, 1u); + EXPECT_EQ(result.floor_reads, 3u); + EXPECT_EQ(result.listed, 10u); + EXPECT_EQ(result.skipped, 10u); + EXPECT_EQ(result.retained_no_coverage, 0u) << "an absent floor admits nothing, so no key reaches the premise"; +} + +TEST(CASOrphanSweepRequests, RetainedKeysCostNoBodyRead) +{ + PageFixture f(/*manifests=*/50, /*min_active=*/25); + const Layout & layout = f.store->layout(); + const ManifestSweepResult result = f.page(); + EXPECT_EQ(result.retained_no_coverage, 24u); + for (uint64_t seq = 1; seq <= 50; ++seq) + EXPECT_EQ(f.backend->getCount(layout.manifestKey(ManifestId{kNs, build(seq)})), 0u) << seq; +} + +TEST(CASOrphanSweepRequests, PrefixEligibleUnderIsTheFourComparisons) +{ + MountLease floor; + floor.writer_epoch = 3; + floor.min_active_build_sequence = 10; + EXPECT_TRUE(prefixEligibleUnder(floor, BuildPrefix{.writer_epoch = 2, .build_sequence = 999})); + EXPECT_FALSE(prefixEligibleUnder(floor, BuildPrefix{.writer_epoch = 4, .build_sequence = 1})); + EXPECT_TRUE(prefixEligibleUnder(floor, BuildPrefix{.writer_epoch = 3, .build_sequence = 9})); + EXPECT_FALSE(prefixEligibleUnder(floor, BuildPrefix{.writer_epoch = 3, .build_sequence = 10})); + floor.min_active_build_sequence = std::numeric_limits::max(); + EXPECT_TRUE(prefixEligibleUnder(floor, BuildPrefix{.writer_epoch = 3, .build_sequence = 10})); + EXPECT_FALSE(prefixEligibleUnder(std::nullopt, BuildPrefix{.writer_epoch = 1, .build_sequence = 1})); +} + +namespace +{ + +/// Deletes the mount key the first time it is read, so the page decides with a floor whose object is +/// gone by the time it decides. Retirement is permanent, so the decisions must be the ones the floor +/// admitted when read, and no active build may be nominated. +class MountVanishesBackend final : public CountingBackend +{ +public: + using CountingBackend::read; + std::optional read(const String & key, TransportAccess & access) override + { + auto got = CountingBackend::read(key, access); + if (got && key == mount_key && !fired) + { + fired = true; + static_cast(InMemoryBackend::remove(key, got->value, access)); + } + return got; + } + String mount_key; + bool fired = false; +}; + +} + +TEST(CASOrphanSweepRequests, MountVanishingMidPageKeepsTheDecisionsOfTheFloorAsRead) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .manifest_sweep_list_budget_keys = 1000, + .manifest_sweep_delete_budget_keys = 100, + .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + casAdmitEntry(*backend, layout, kNs); + publishAt(*backend, layout, kNs, RefTxnId{1, 1}, "live", /*build_sequence=*/51, DB::UInt128(0x7001), /*birth=*/true); + writeRecoverableCkptForRawFixture(*backend, layout, kNs, RefCkpt{ + .life_epoch = 1, .committed_through = RefTxnId{1, 1}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + for (uint64_t seq = 1; seq <= 50; ++seq) + writeManifestRaw(*backend, layout, kNs, build(seq), {blobEntryFor("a", DB::UInt128(0x100 + seq))}); + setWatermarkMinActive(*backend, layout, "test", 1, /*min_active=*/25); + backend->mount_key = layout.mountKey("test"); + + const ManifestSweepResult result = planManifestCursorPage(*store, "", 1000, 100, true, nullptr); + EXPECT_TRUE(backend->fired); + EXPECT_EQ(result.floor_lookups, 1u); + EXPECT_EQ(result.retained_no_coverage, 24u) << "the 24 retired builds were decided under the floor as read"; + EXPECT_TRUE(result.nominations.empty()); + OperationForTest op(*backend); + EXPECT_FALSE((*op).head(layout.mountKey("test"), Retry::once()).has_value()); +} + +namespace +{ + +ThreadPool makeReadPool(size_t threads) +{ + return ThreadPool{CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, + CurrentMetrics::LocalThreadScheduled, threads, threads, /*queue_size*/ 0}; +} + +/// Every GET the page issued, by key, in whichever thread it ran. +std::map getsOf(CountingBackend & backend) +{ + std::map gets; + for (const String & key : backend.touchedKeys()) + if (const uint64_t n = backend.getCount(key); n != 0) + gets[key] = n; + return gets; +} + +struct PageOutcome +{ + uint64_t listed, skipped, deleted, undecodable, retained_no_coverage, retained_hold, + retained_unconsumed_seal, retained_tail_removal, retained_work_budget, floor_lookups, floor_reads; + bool wrapped; + String next_cursor; + bool operator==(const PageOutcome &) const = default; +}; + +PageOutcome outcomeOf(const ManifestSweepResult & r) +{ + return {r.listed, r.skipped, r.deleted, r.undecodable, r.retained_no_coverage, r.retained_hold, + r.retained_unconsumed_seal, r.retained_tail_removal, r.retained_work_budget, + r.floor_lookups, r.floor_reads, r.wrapped, r.next_cursor}; +} + +bool sameNomination(const ManifestSweepResult::Nomination & a, const ManifestSweepResult::Nomination & b) +{ + if (!(a.id == b.id) || a.key != b.key || a.token.dialect != b.token.dialect || a.token.value != b.token.value) + return false; + if (a.source_retirements.size() != b.source_retirements.size()) + return false; + for (size_t i = 0; i < a.source_retirements.size(); ++i) + if (!(a.source_retirements[i].ref == b.source_retirements[i].ref) + || !(a.source_retirements[i].source_id == b.source_retirements[i].source_id)) + return false; + return true; +} + +/// Both runs decide candidates from the SAME listed order and append nominations in that same order, +/// whichever reader fetched their bytes, so an index-wise comparison is exact -- no sort needed. +bool sameNominations(const std::vector & a, + const std::vector & b) +{ + if (a.size() != b.size()) + return false; + for (size_t i = 0; i < a.size(); ++i) + if (!sameNomination(a[i], b[i])) + return false; + return true; +} + +/// A fixture whose sweep has REAL candidates and a committed-tail walk long enough to hint ahead, +/// entirely WITHOUT crossing an epoch: this life is born directly at epoch 2 (`{2,1}`, no +/// `prev_epoch_seal` needed -- genesis, not a chain link), and `kDebrisManifests` unowned raw +/// manifests sit at prefix epoch 1 -- a legacy build-prefix number, never part of this life's own ref +/// stream, but eligible under the floor (an old epoch is always eligible) and covered by the folded +/// cursor sitting in epoch 2 (rule 1) all the same, so they become genuine nominations. `kEpochTwoLogs` +/// ordinary committed grants after the birth give the committed-tail walk (and the recovery walk, which +/// starts at this same genesis) a range comfortably longer than one read-ahead window, all inside the +/// ONE epoch neither walk ever leaves -- unlike the epoch-crossing fixture below, whose hints legitimately +/// overshoot a seal and so cannot be expected to read the identical key set at every concurrency, this +/// fixture's GET set is invariant to concurrency, which is what the comparison after it needs. +constexpr uint64_t kDebrisManifests = 6; +constexpr uint64_t kEpochTwoLogs = 80; + +struct CandidateFixture +{ + std::shared_ptr backend = std::make_shared(); + PoolPtr store; + + CandidateFixture() + { + PoolConfig config; + config.pool_prefix = "p"; + config.server_root_id = "test"; + config.manifest_sweep_list_budget_keys = 1000; + config.manifest_sweep_delete_budget_keys = 100; + config.gc_fold_max_defer_rounds = 0; + store = Pool::open(backend, config); + const Layout & layout = store->layout(); + casAdmitEntry(*backend, layout, kNs); + + for (uint64_t seq = 1; seq <= kDebrisManifests; ++seq) + writeManifestRaw(*backend, layout, kNs, build(seq), {blobEntryFor("a", DB::UInt128(0x100 + seq))}); + publishAt(*backend, layout, kNs, RefTxnId{2, 1}, "live", /*build_sequence=*/2000, + DB::UInt128(0x7001), /*birth=*/true); + for (uint64_t seq = 2; seq <= kEpochTwoLogs; ++seq) + publishAt(*backend, layout, kNs, RefTxnId{2, seq}, "epoch2-" + std::to_string(seq), + 2000 + seq, DB::UInt128(0x9000 + seq), /*birth=*/false); + writeRecoverableCkptForRawFixture(*backend, layout, kNs, RefCkpt{ + .life_epoch = 2, .committed_through = RefTxnId{2, kEpochTwoLogs}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); + seedFoldCursorForTest(*backend, layout, kNs, RefTxnId{2, 1}); + setWatermarkMinActive(*backend, layout, "test", /*writer_epoch=*/2, /*min_active=*/1); + backend->resetCounts(); + } +}; + +} + +/// The read-ahead reader must issue the same GETs against the same keys as the inline reader, decide +/// the same way, and nominate the exact same candidates; only when the bytes arrive moves. The fixture +/// gives the page real candidates and a committed-tail walk spanning more than one window, so the +/// comparison actually exercises the read-ahead instead of vacuously agreeing over nothing. +TEST(CASOrphanSweepRequests, PageIsIdenticalInlineAndWithReadAhead) +{ + CandidateFixture inline_f; + const ManifestSweepResult inline_r = planManifestCursorPage(*inline_f.store, "", 1000, 100, true, nullptr); + const auto inline_gets = getsOf(*inline_f.backend); + ASSERT_EQ(inline_r.nominations.size(), kDebrisManifests) + << "the fixture must actually produce candidates, or this test proves nothing"; + + CandidateFixture ahead_f; + ThreadPool pool = makeReadPool(4); + const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load(); + const ManifestSweepResult ahead_r = planManifestCursorPage( + *ahead_f.store, "", 1000, 100, true, nullptr, &pool, /*read_concurrency=*/16); + const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load() - hits_before; + const auto ahead_gets = getsOf(*ahead_f.backend); + + EXPECT_GT(hits, 0u) << "the fixture's committed-tail walk and candidates must actually hit the " + "read-ahead, or this oracle could never catch a hinting regression"; + EXPECT_EQ(outcomeOf(inline_r), outcomeOf(ahead_r)); + EXPECT_TRUE(sameNominations(inline_r.nominations, ahead_r.nominations)); + EXPECT_EQ(inline_gets, ahead_gets); +} + +/// A committed tail that spans two epochs: the walk hints ids past the seal in the old epoch, which +/// do not exist, discards them at the crossing (at most one window), and hints the new epoch's ids. +TEST(CASOrphanSweepRequests, EpochCrossingDiscardsAtMostOneWindowAndTheNewEpochIsHinted) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .manifest_sweep_list_budget_keys = 1000, + .manifest_sweep_delete_budget_keys = 100, + .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + casAdmitEntry(*backend, layout, kNs); + /// Epoch 1: birth at {1,1}, six ordinary logs {1,2..7}, seal at {1,8}. Epoch 2: {2,1..40}. + publishAt(*backend, layout, kNs, RefTxnId{1, 1}, "t1", /*build_sequence=*/100, DB::UInt128(0x7001), /*birth=*/true); + for (uint64_t seq = 2; seq <= 7; ++seq) + publishAt(*backend, layout, kNs, RefTxnId{1, seq}, "t" + std::to_string(seq), 100 + seq, DB::UInt128(0x7000 + seq), /*birth=*/false); + writeSealAt(*backend, layout, kNs, RefTxnId{1, 8}); + publishAt(*backend, layout, kNs, RefTxnId{2, 1}, "u1", 200, DB::UInt128(0x8001), /*birth=*/false, /*prev_epoch_seal=*/RefTxnId{1, 8}); + for (uint64_t seq = 2; seq <= 40; ++seq) + publishAt(*backend, layout, kNs, RefTxnId{2, seq}, "u" + std::to_string(seq), 200 + seq, DB::UInt128(0x8000 + seq), /*birth=*/false); + writeRecoverableCkptForRawFixture(*backend, layout, kNs, RefCkpt{ + .life_epoch = 1, .committed_through = RefTxnId{2, 40}, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{1, 8}}); + writeManifestRaw(*backend, layout, kNs, build(1), {blobEntryFor("a", DB::UInt128(0x100))}); + setWatermarkMinActive(*backend, layout, "test", 2, /*min_active=*/1000); + backend->resetCounts(); + + const uint64_t wasted_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load(); + const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load(); + ThreadPool pool = makeReadPool(4); + const ManifestSweepResult result = planManifestCursorPage(*store, "", 1000, 100, true, nullptr, &pool, 16); + const uint64_t wasted = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load() - wasted_before; + const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load() - hits_before; + + /// `publishAt` writes a manifest as a side effect of every call above, so the page's LIST sees + /// every one of those (47) plus the one raw manifest -- the fixture is about the hint/discard + /// mechanics of the ref-log walks a namespace's first candidate triggers, not about the listing. + EXPECT_EQ(result.listed, 48u); + /// The SAME reader crosses the epoch twice on this fixture: once inside the recovery walk + /// `activeManifestKeys` runs via `recoverRefTableDetailedFromAuthority`, and again in its own + /// committed-tail walk over the same {1,1}..{2,40} range (today's pre-existing double walk, not + /// something this change introduces) -- so at most two windows are discarded, not one. + EXPECT_LE(wasted, 128u) << "at most two windows at concurrency 16: one per walk crossing the epoch"; + EXPECT_GE(hits, 30u) << "the new epoch's logs were hinted and taken"; + /// Both walks read epoch 2's logs once each, so every key is read twice -- today's behaviour with + /// or without read-ahead, not something the hint/discard rule changes. + for (uint64_t seq = 1; seq <= 40; ++seq) + EXPECT_EQ(backend->getCount(layout.refLogKey(NamespaceLifeId::fromCatalogEntry(kNs, catalogLifeIdForTest(*backend, layout, kNs)), RefTxnId{2, seq})), 2u) << seq; +} diff --git a/src/Disks/tests/gtest_cas_parallel_commit.cpp b/src/Disks/tests/gtest_cas_parallel_commit.cpp index 7ae5ab676712..8cdb007d6623 100644 --- a/src/Disks/tests/gtest_cas_parallel_commit.cpp +++ b/src/Disks/tests/gtest_cas_parallel_commit.cpp @@ -160,7 +160,7 @@ namespace /// Fixture for the `CasCommitRollback` suite: wraps a real `ContentAddressedMetadataStorage` and /// drives ordinary `ContentAddressedTransaction`s through disk paths, so the fault seams under test /// (`ContentAddressedMetadataStorage::armPromoteFailureForTest`/`setAfterPromoteHookForTest`, the -/// minimal test-only hooks this task adds) fire from the SAME `publishStaging` call path production +/// minimal test-only hooks) fire from the SAME `publishStaging` call path production /// `commit()` uses -- unlike `CaWiringFixture` above, which pokes the bare pool primitives directly. /// Every part in one fixture instance shares ONE fixed table uuid (and therefore one `RootNamespace`), /// matching every test's single `fx.ns()`. diff --git a/src/Disks/tests/gtest_cas_part_folder_access.cpp b/src/Disks/tests/gtest_cas_part_folder_access.cpp index 69fc028aeda4..b3215cff2216 100644 --- a/src/Disks/tests/gtest_cas_part_folder_access.cpp +++ b/src/Disks/tests/gtest_cas_part_folder_access.cpp @@ -5,11 +5,9 @@ #include #include #include -#include #include #include #include -#include #include #include #include @@ -28,7 +26,6 @@ namespace DB::ErrorCodes namespace ProfileEvents { extern const Event CASRefRollbackBestEffortDropFailed; -extern const Event CASPartFolderValidateSkipped; } using namespace DB; @@ -64,48 +61,31 @@ Cas::ManifestId publishPart(const Cas::PoolPtr & store, const Cas::RootNamespace Cas::CachedPartFolderAccess::CacheParams cacheOn() { return {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, - .explain_enabled = true, .validate = {}}; -} - -/// Mirrors gtest_cas_s3_staging.cpp's helper of the same shape: the shape a real CAS disk config -/// has under `storage_configuration.disks.`, so `config_prefix = "disk"` reads exactly like -/// the disk factory's `config_prefix`. Used to unit-test `parsePartFolderValidate` standalone. -Poco::AutoPtr configWithDiskSection(const std::string & inner_xml) -{ - std::istringstream xml_stream( // STYLE_CHECK_ALLOW_STD_STRING_STREAM - "" + inner_xml + ""); - return new Poco::Util::XMLConfiguration(xml_stream); + .explain_enabled = true}; } /// Every mutating backend op throws once armed — models a correlated backend outage during the /// transaction's compensating rollback (dropRef must append a removal, which mutates the backend). +/// While armed, the store is unreachable for every mutation: a transport-class failure, so the request +/// engine settles it by a read (which fails too) and reissues until the call's own retry window closes. +/// A test arming it therefore drives the engine's clock, or pays that window in real time. class RollbackFaultBackend final : public Cas::InMemoryBackend { public: std::atomic armed{false}; - Cas::PutResult putIfAbsent(const String & k, const String & b, const Cas::ObjectMeta & m) override - { - failIfArmed(); - return InMemoryBackend::putIfAbsent(k, b, m); - } - - Cas::PutResult putOverwrite(const String & k, const String & b, const Cas::Token & e, const Cas::ObjectMeta & m) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + Cas::TransportAccess & access) override { failIfArmed(); - return InMemoryBackend::putOverwrite(k, b, e, m); + return InMemoryBackend::write(key, bytes, expected_value, access); } - Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const Cas::ObjectMeta & m) override + RawRemoval remove(const String & key, const String & expected_value, Cas::TransportAccess & access) override { failIfArmed(); - return InMemoryBackend::casPut(k, b, e, m); - } - - Cas::DeleteOutcome deleteExact(const String & k, const Cas::Token & t) override - { - failIfArmed(); - return InMemoryBackend::deleteExact(k, t); + return InMemoryBackend::remove(key, expected_value, access); } private: @@ -135,9 +115,12 @@ class PromoteConflictOnceBackend final : public Cas::InMemoryBackend /// cleanup path ran its ref-log append at all, on a table where that append can no longer succeed. int matching_put_attempts = 0; - Cas::PutResult putIfAbsent(const String & key, const String & bytes, const Cas::ObjectMeta & meta) override + /// Sabotages the sole write primitive, so the fault fires whichever verb (`create`/`replace`) issued it. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + Cas::TransportAccess & access) override { - if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + if (!expected_value && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) { ++matching_put_attempts; if (skip > 0) @@ -145,13 +128,13 @@ class PromoteConflictOnceBackend final : public Cas::InMemoryBackend else if (fault_count > 0) { --fault_count; - /// The 3-arg qualified call bypasses virtual dispatch entirely (unlike a 2-arg - /// convenience overload, which would re-enter this very override through the vtable). - InMemoryBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT"), meta); + /// The qualified call bypasses virtual dispatch entirely (unlike a re-entrant call + /// through the vtable), landing a foreign object at the key before the response is lost. + InMemoryBackend::write(key, bytes + String("\x01_FOREIGN_DIFFERENT"), expected_value, access); throw Poco::TimeoutException("PromoteConflictOnceBackend: a foreign different object landed; response lost"); } } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; @@ -170,9 +153,12 @@ class PromoteDefiniteFailureBackend final : public Cas::InMemoryBackend int fault_count = 0; int matching_put_attempts = 0; - Cas::PutResult putIfAbsent(const String & key, const String & bytes, const Cas::ObjectMeta & meta) override + /// Sabotages the sole write primitive, so the fault fires whichever verb (`create`/`replace`) issued it. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + Cas::TransportAccess & access) override { - if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + if (!expected_value && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) { ++matching_put_attempts; if (skip > 0) @@ -184,13 +170,13 @@ class PromoteDefiniteFailureBackend final : public Cas::InMemoryBackend Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); } } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; } -TEST(CASPartFolderAccess, RetainedHitSkipsManifestHead) +TEST(CASPartFolderAccess, RetainedHitCostsNoRequest) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); @@ -201,14 +187,21 @@ TEST(CASPartFolderAccess, RetainedHitSkipsManifestHead) const Cas::PartRefKey key{ns, "part_1"}; const String manifest_key = layout.manifestKey(id); + /// Cold build: warms the retained view and the decode cache. Excluded from the counts below so + /// they measure only the warm hits that follow. + ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + backend->resetCounts(); - for (int i = 0; i < 5; ++i) + for (int i = 0; i < 4; ++i) ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); - /// The one-GET goal (spec acceptance 4): ONE body GET, ONE mandatory HEAD (the cold build); - /// every subsequent CachedForLoad call is a validated hit — zero manifest ops. - EXPECT_EQ(backend->getCount(manifest_key), 1u); - EXPECT_EQ(backend->headCount(manifest_key), 1u); + /// A retained hit costs no request at all -- not merely no manifest GET on this key, but no + /// backend traffic of ANY kind (GET, HEAD, LIST, streamed GET) against ANY key. + EXPECT_EQ(backend->getCount(manifest_key), 0u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->headTotal(), 0u); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->getStreamTotal(), 0u); EXPECT_TRUE(access.explain(key).retained); EXPECT_EQ(access.explain(key).last_decision, Cas::CachedPartFolderAccess::LastDecision::Hit); @@ -224,7 +217,7 @@ TEST(CASPartFolderAccess, HitPathJournalEmptyAndCheapWhenExplainDisabled) /// per-disk explain mutex nor write a journal entry (B2). Cas::CachedPartFolderAccess access(store, {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, - .explain_enabled = false, .validate = {}}); + .explain_enabled = false}); const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); const Cas::PartRefKey key{ns, "part_1"}; const String manifest_key = layout.manifestKey(id); @@ -233,9 +226,8 @@ TEST(CASPartFolderAccess, HitPathJournalEmptyAndCheapWhenExplainDisabled) for (int i = 0; i < 5; ++i) ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); - /// Same request oracle as RetainedHitSkipsManifestHead — one cold build, then validated hits. + /// Same request oracle as RetainedHitCostsNoRequest — one cold build, then retained hits. EXPECT_EQ(backend->getCount(manifest_key), 1u); - EXPECT_EQ(backend->headCount(manifest_key), 1u); /// The journal is never written when disabled. EXPECT_EQ(access.explainJournalSizeForTest(), 0u); /// explain() still reports live retention truthfully, but the decision defaults to Miss (unwritten). @@ -275,12 +267,11 @@ TEST(CASPartFolderAccess, GetViewFailsClosedOnMissingBody) const Cas::RootNamespace ns{"srv/t1"}; const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); - /// Physically delete the live manifest body (a protocol violation) — every getView mode must - /// surface INV-NO-DANGLE as FILE_DOESNT_EXIST in Phase 2 (there is no retained view to hit). - /// Retention is off (the single-arg ctor below), so this is the `always` (default) part_folder_validate - /// mode under test regardless — the `never`/`age` skip is proven by the ValidateNever/ValidateAge - /// tests further down, which turn retention ON. + /// Physically delete the live manifest body (a protocol violation). Retention is off and the + /// decode cache is cold (promote reads the body through the backend, not the reader), so every + /// getView mode reaches the reader's miss path: one GET, no HEAD, FILE_DOESNT_EXIST. deleteManifestBody(*backend, layout, id); + backend->resetCounts(); Cas::CachedPartFolderAccess access(store); const Cas::PartRefKey key{ns, "part_1"}; @@ -288,6 +279,7 @@ TEST(CASPartFolderAccess, GetViewFailsClosedOnMissingBody) Cas::Freshness::ForceFresh, Cas::Freshness::StrictValidate}) expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, [&] { access.getView(key, freshness); }); + EXPECT_EQ(backend->headCount(layout.manifestKey(id)), 0u); } TEST(CASPartFolderAccess, WritePrimitivesRoundTrip) @@ -420,14 +412,15 @@ TEST(CASPartFolderAccess, PublishEntriesAbandonsBuildOnPromoteFailure) /// fresh id to get past the foreign object would sort above it. String greatest_key; size_t foreign_objects = 0; + DB::Cas::tests::OperationForTest scan(*backend); for (String cursor;;) { - const Cas::ListPage page = backend->list(backend->fault_key_substr, cursor, 1000); + const Cas::ListPage page = (*scan).list(backend->fault_key_substr, cursor, 1000, Cas::Retry::standard()); for (const auto & listed : page.keys) { if (listed.key > greatest_key) greatest_key = listed.key; - const auto body = backend->get(listed.key); + const auto body = (*scan).read(listed.key, Cas::Retry::standard()); if (body && body->bytes.find("_FOREIGN_DIFFERENT") != String::npos) ++foreign_objects; } @@ -437,7 +430,7 @@ TEST(CASPartFolderAccess, PublishEntriesAbandonsBuildOnPromoteFailure) } EXPECT_EQ(foreign_objects, 1u) << "the foreign object must still own the key it took"; ASSERT_FALSE(greatest_key.empty()); - const auto greatest_body = backend->get(greatest_key); + const auto greatest_body = (*scan).read(greatest_key, Cas::Retry::standard()); ASSERT_TRUE(greatest_body.has_value()); EXPECT_NE(greatest_body->bytes.find("_FOREIGN_DIFFERENT"), String::npos) << "the foreign occupant must still be the highest id in this table's stream: a log object above " @@ -554,7 +547,10 @@ TEST(CASPartFolderAccess, PrepareThenAbortAppendsThePrecommitRemoval) /// The precommit BODY survives (delete-after-sealed-decrements) -- the removal queues GC's `-1`, /// it does not writer-delete the manifest. Mirrors /// `CASPartWriteTxn.AbandonAppendsPrecommitRemovalAndKeepsLivePrecommitBody`. - EXPECT_TRUE(backend->head(store->layout().manifestKey(id)).exists); + { + DB::Cas::tests::OperationForTest op(*backend); + EXPECT_TRUE((*op).head(store->layout().manifestKey(id), Cas::Retry::standard()).has_value()); + } } /// A forgotten terminal must be impossible, not merely discouraged: `~PartWriteTxn` only retires the @@ -678,7 +674,7 @@ TEST(CASPartFolderAccess, ExplainRecordsDecisions) auto backend = std::make_shared(); auto store = openPoolForTest(backend); const Cas::RootNamespace ns{"srv/t1"}; - Cas::CachedPartFolderAccess access(store, {.explain_enabled = true, .validate = {}}); + Cas::CachedPartFolderAccess access(store, {.explain_enabled = true}); publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); const Cas::PartRefKey key{ns, "part_1"}; @@ -717,11 +713,10 @@ TEST(CASPartFolderAccess, BaselineRequestCountsWithoutRetention) for (int i = 0; i < n; ++i) ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); - /// The Phase-3 baseline (retention off): one manifest-body GET (the decode cache absorbs the - /// rest) but a mandatory manifest HEAD per call. Phase 4's validated hits remove the HEADs; - /// this test pins the numbers Phase 4 improves. + /// Retention off: one manifest-body GET (the decode cache absorbs the rest) and no manifest + /// HEAD at all — a cached decode is served without a request. EXPECT_EQ(backend->getCount(manifest_key), 1u); - EXPECT_EQ(backend->headCount(manifest_key), static_cast(n)); + EXPECT_EQ(backend->headCount(manifest_key), 0u); } /// ==== Phase 4 (retention) semantics battery: spec §Testing acceptance criteria ==== @@ -767,7 +762,11 @@ TEST(CASPartFolderAccess, MismatchRebuildAfterRepublish) EXPECT_TRUE(access.explain(key).retained); } -TEST(CASPartFolderAccess, ForceFreshFailsClosedWhileRetainedViewExists) +/// The decode cache is keyed by id and an id names one content forever, so a warm reader serves +/// `ForceFresh` from the immutable decode with no manifest request even after the body object is +/// gone. A retained-view hit is not what is being tested here: `ForceFresh` bypasses the view cache +/// and rebuilds the view from the pool's manifest cache. +TEST(CASPartFolderAccess, ForceFreshServesImmutableDecodeWithoutManifestRequestsAfterBodyDeletion) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); @@ -776,178 +775,53 @@ TEST(CASPartFolderAccess, ForceFreshFailsClosedWhileRetainedViewExists) Cas::CachedPartFolderAccess access(store, cacheOn()); const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); const Cas::PartRefKey key{ns, "part_1"}; + const String manifest_key = layout.manifestKey(id); - ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); /// retained + ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); /// warms the decode cache deleteManifestBody(*backend, layout, id); /// protocol violation: live body vanishes + backend->resetCounts(); - /// Write-evidence and strict paths surface INV-NO-DANGLE immediately (mandatory HEAD)... - expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, - [&] { access.getView(key, Cas::Freshness::ForceFresh); }); - expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, - [&] { access.getView(key, Cas::Freshness::StrictValidate); }); - - /// ...while a validated CachedForLoad hit still serves the immutable decode — the documented - /// residual delta (spec §Staleness Equivalence): detection deferred, never for write evidence. - EXPECT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); + auto view = access.getView(key, Cas::Freshness::ForceFresh); + ASSERT_NE(view, nullptr); + EXPECT_NE(view->findFile("f"), nullptr); + EXPECT_EQ(backend->getCount(manifest_key), 0u); + EXPECT_EQ(backend->headCount(manifest_key), 0u); + EXPECT_EQ(access.explain(key).last_decision, Cas::CachedPartFolderAccess::LastDecision::ForceFreshRead); + + /// `StrictValidate` serves the same immutable decode: an id names one content, so once the id is + /// resolved there is nothing stricter left to prove about the body. It bypasses retention, so its + /// recorded decision differs from the `ForceFresh` one above. + auto strict_view = access.getView(key, Cas::Freshness::StrictValidate); + ASSERT_NE(strict_view, nullptr); + EXPECT_EQ(strict_view->manifest().get(), view->manifest().get()); + EXPECT_EQ(backend->getCount(manifest_key), 0u); + EXPECT_EQ(backend->headCount(manifest_key), 0u); + EXPECT_EQ(access.explain(key).last_decision, Cas::CachedPartFolderAccess::LastDecision::StrictBypass); } -/// ==== §3 (part_folder_validate): the ForceFresh body re-proof HEAD is configurable ==== - -TEST(CASPartFolderAccess, ValidateNeverServesRetainedViewWithoutBodyHead) +/// With the decode cache disabled a prior read leaves nothing behind, so the deleted body surfaces +/// as FILE_DOESNT_EXIST in every mode: the miss path is the same fail-closed path a cold reader takes. +TEST(CASPartFolderAccess, DeletedBodyFailsClosedInEveryModeWhenDecodeCacheDisabled) { auto backend = std::make_shared(); - auto store = openPoolForTest(backend); + auto store = DB::Cas::Pool::open(backend, + DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .manifest_decode_cache_bytes = 0}); const Cas::Layout layout("p"); const Cas::RootNamespace ns{"srv/t1"}; - const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); - - auto params = cacheOn(); - params.validate = {Cas::PartFolderValidate::Mode::Never, 0}; - Cas::CachedPartFolderAccess access(store, params); + Cas::CachedPartFolderAccess access(store); /// retention off: every mode reaches the reader + const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); const Cas::PartRefKey key{ns, "part_1"}; - /// Prime the retained view (pays the HEAD once). - ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); - /// Body vanishes (a protocol violation the net would normally catch)... - deleteManifestBody(*backend, layout, id); - const auto skips_before = ProfileEvents::global_counters[ProfileEvents::CASPartFolderValidateSkipped].load(); - /// ...but `never` serves the retained view, no HEAD, no throw. - EXPECT_NO_THROW(access.getView(key, Cas::Freshness::ForceFresh)); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASPartFolderValidateSkipped].load() - skips_before, 1); -} - -TEST(CASPartFolderAccess, ValidateAlwaysStillHeadsEveryForceFresh) -{ - auto backend = std::make_shared(); - auto store = openPoolForTest(backend); - const Cas::Layout layout("p"); - const Cas::RootNamespace ns{"srv/t1"}; - const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); - - Cas::CachedPartFolderAccess access(store, cacheOn()); /// default = Always - const Cas::PartRefKey key{ns, "part_1"}; ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); deleteManifestBody(*backend, layout, id); - /// `always` re-proves the body every ForceFresh — the deleted body surfaces as FILE_DOESNT_EXIST. - expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, - [&] { access.getView(key, Cas::Freshness::ForceFresh); }); -} - -TEST(CASPartFolderAccess, ValidateAgeSkipsWithinWindowThenHeadsAfter) -{ - auto backend = std::make_shared(); - auto store = openPoolForTest(backend); - const Cas::Layout layout("p"); - const Cas::RootNamespace ns{"srv/t1"}; - const auto id = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); - - auto params = cacheOn(); - params.validate = {Cas::PartFolderValidate::Mode::Age, /*age_seconds=*/5}; - /// An injected clock (spec §3 TDD requirement): the SAME function stamps the retained view's - /// validated_at_ms (buildView) and drives the age-window comparison (getView), so the test controls - /// both sides of the comparison deterministically -- no real sleep. - std::atomic fake_now_ms{1'000'000}; - Cas::CachedPartFolderAccess access(store, params, [&] { return fake_now_ms.load(); }); - const Cas::PartRefKey key{ns, "part_1"}; - const String manifest_key = layout.manifestKey(id); - - /// Prime the retained view (pays the HEAD once) at fake_now_ms. - ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); - const uint64_t heads_after_prime = backend->headCount(manifest_key); - - /// +2s: still inside the 5s window — served from the retained view, no new HEAD. - fake_now_ms += 2000; - ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); - EXPECT_EQ(backend->headCount(manifest_key), heads_after_prime); - - /// +6s from the ORIGINAL stamp (past the 5s window): re-proves the body via a fresh HEAD. - fake_now_ms += 4000; - ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); - EXPECT_GT(backend->headCount(manifest_key), heads_after_prime); -} - -/// ==== §3: `parsePartFolderValidate` config parsing, standalone (mirrors CASS3Staging's -/// parseStagingBackend coverage) -- review finding: std::stoull silently accepted a leading '-' -/// (unsigned wraparound), so a malformed `age -5` never hit the parser's own fail-closed throw. -/// These pin the fixed `std::from_chars`-based parsing directly, with no disk/store needed. ==== - -TEST(CASPartFolderValidateParse, DefaultConfigParsesToAlways) -{ - /// No `part_folder_validate` key at all -- the byte-for-byte-pre-§3-behavior default. - auto config = configWithDiskSection("/tmp/whatever"); - const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); - EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Always); -} - -TEST(CASPartFolderValidateParse, ParsesAlways) -{ - auto config = configWithDiskSection("always"); - const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); - EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Always); -} - -TEST(CASPartFolderValidateParse, ParsesNever) -{ - auto config = configWithDiskSection("never"); - const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); - EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Never); -} - -TEST(CASPartFolderValidateParse, ParsesPositiveAge) -{ - auto config = configWithDiskSection("age 5"); - const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); - EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Age); - EXPECT_EQ(v.age_seconds, 5u); -} - -TEST(CASPartFolderValidateParse, AcceptsAgeZeroAsADegenerateButValidWindow) -{ - /// `age 0` is accepted, not rejected: it is a well-formed (if degenerate -- effectively an - /// almost-always-expired window) configuration, not malformed input. Only genuinely malformed - /// suffixes (negative, non-digit, empty, trailing garbage) fail closed below. - auto config = configWithDiskSection("age 0"); - const auto v = ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); - EXPECT_EQ(v.mode, Cas::PartFolderValidate::Mode::Age); - EXPECT_EQ(v.age_seconds, 0u); -} - -TEST(CASPartFolderValidateParse, NegativeAgeThrows) -{ - /// The bug this regression-guards: std::stoull("-5") used to return 18446744073709551611 - /// (unsigned wraparound) instead of rejecting the leading '-'. - auto config = configWithDiskSection("age -5"); - expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, - [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); -} - -TEST(CASPartFolderValidateParse, NonDigitAgeThrows) -{ - auto config = configWithDiskSection("age abc"); - expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, - [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); -} - -TEST(CASPartFolderValidateParse, TrailingGarbageAfterAgeThrows) -{ - auto config = configWithDiskSection("age 5abc"); - expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, - [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); -} - -TEST(CASPartFolderValidateParse, EmptyAgeSuffixThrows) -{ - auto config = configWithDiskSection("age "); - expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, - [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); -} + backend->resetCounts(); -TEST(CASPartFolderValidateParse, UnknownValueThrows) -{ - /// Fail-closed: an unrecognized value must NEVER silently become `never`/`always`. - auto config = configWithDiskSection("sometimes"); - expectThrowsCode(ErrorCodes::BAD_ARGUMENTS, - [&] { ContentAddressedMetadataStorage::parsePartFolderValidate(*config, "disk"); }); + for (auto freshness : {Cas::Freshness::CachedForLoad, + Cas::Freshness::ForceFresh, + Cas::Freshness::StrictValidate}) + expectThrowsCode(ErrorCodes::FILE_DOESNT_EXIST, [&] { access.getView(key, freshness); }); + EXPECT_EQ(backend->headCount(layout.manifestKey(id)), 0u); + EXPECT_EQ(backend->getCount(layout.manifestKey(id)), 3u); /// one GET per attempt, nothing cached } TEST(CASPartFolderAccess, AbsenceIsNeverRetained) @@ -983,13 +857,19 @@ TEST(CASPartFolderAccess, GetViewEmitsRefResolveOnlyOnRealResolveWork) publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); const Cas::PartRefKey key{ns, "part_1"}; - std::vector seen; - store->setEventSink([&](const Cas::CasEvent & e) { seen.push_back(e); }); - Cas::CachedPartFolderAccess access(store, cacheOn()); /// retention on, validate == Always (default) + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto seen = std::make_shared(); + store->setEventSink([seen](const Cas::CasEvent & e) + { + seen->push(e); + }); + Cas::CachedPartFolderAccess access(store, cacheOn()); /// retention on const auto refResolveCount = [&] { - return std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + return std::count_if(observed.begin(), observed.end(), [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::RefResolve; }); }; @@ -1003,8 +883,7 @@ TEST(CASPartFolderAccess, GetViewEmitsRefResolveOnlyOnRealResolveWork) ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); EXPECT_EQ(refResolveCount(), 1) << "a warm view-cache hit must not add a RefResolve row"; - /// ForceFresh always re-proves the manifest body under the default Always validation policy, so - /// this is real resolve work again -> +1. + /// ForceFresh always bypasses the retained view, so this is real resolve work again -> +1. ASSERT_NE(access.getView(key, Cas::Freshness::ForceFresh), nullptr); EXPECT_EQ(refResolveCount(), 2); @@ -1021,7 +900,7 @@ TEST(CASPartFolderAccess, OversizedViewServedNotRetained) Cas::CachedPartFolderAccess access(store, Cas::CachedPartFolderAccess::CacheParams{ .cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 1, - .explain_enabled = true, .validate = {}}); + .explain_enabled = true}); const auto id = publishPart(store, ns, "part_1", {inlineEntry("f", "x")}); const Cas::PartRefKey key{ns, "part_1"}; const String manifest_key = layout.manifestKey(id); @@ -1032,10 +911,17 @@ TEST(CASPartFolderAccess, OversizedViewServedNotRetained) EXPECT_EQ(access.explain(key).last_decision, Cas::CachedPartFolderAccess::LastDecision::OversizedBypass); - const uint64_t head_before = backend->headCount(manifest_key); + backend->resetCounts(); auto view2 = access.getView(key, Cas::Freshness::CachedForLoad); ASSERT_NE(view2, nullptr); - EXPECT_GT(backend->headCount(manifest_key), head_before); /// not retained: re-HEADs every call + /// Not retained: every call rebuilds the view (a new view object over the SAME shared decode) and + /// records the bypass again; the rebuild costs no manifest request because the decode is cached. + EXPECT_NE(view1.get(), view2.get()); + EXPECT_EQ(view1->manifest().get(), view2->manifest().get()); + EXPECT_EQ(access.explain(key).last_decision, + Cas::CachedPartFolderAccess::LastDecision::OversizedBypass); + EXPECT_EQ(backend->getCount(manifest_key), 0u); + EXPECT_EQ(backend->headCount(manifest_key), 0u); EXPECT_FALSE(access.explain(key).retained); } @@ -1056,9 +942,10 @@ TEST(CASPartFolderAccess, DisabledModeKeepsBaseline) for (int i = 0; i < n; ++i) ASSERT_NE(access.getView(key, Cas::Freshness::CachedForLoad), nullptr); - /// Exactly the Phase-3 baseline: bytes=0 restores the no-retention call graph byte-for-byte. + /// bytes=0 restores the no-retention call graph: one body GET, then the decode cache serves every + /// rebuild with no manifest request. EXPECT_EQ(backend->getCount(manifest_key), 1u); - EXPECT_EQ(backend->headCount(manifest_key), static_cast(n)); + EXPECT_EQ(backend->headCount(manifest_key), 0u); EXPECT_FALSE(access.explain(key).retained); } @@ -1133,6 +1020,9 @@ TEST(CASPartFolderAccess, BestEffortRollbackDropCountsAndSurvivesABackendOutage) { auto backend = std::make_shared(); auto store = openPoolForTest(backend); + /// Both drops below give up only when their own retry window closes, so the engine's inter-attempt + /// sleeps are paid in virtual time rather than by sleeping out the operation deadline for real. + auto clock = Cas::tests::VirtualRetryClock::installOn(store); Cas::CachedPartFolderAccess access(store, cacheOn()); const Cas::RootNamespace ns_a{"srv/ta"}; @@ -1150,31 +1040,12 @@ TEST(CASPartFolderAccess, BestEffortRollbackDropCountsAndSurvivesABackendOutage) access.dropRefBestEffort(Cas::PartRefKey{ns_b, "part_b"}); const auto after = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed].load(); EXPECT_EQ(after, before + 1); + EXPECT_GT(clock->pauseCount(), 0u) + << "the give-up must be the call's own retry window, reached through the injected sleep"; backend->armed = false; /// let store teardown release its lease cleanly } -namespace -{ - -/// A pool whose ref lane makes ONE attempt per append. That is what turns a single lost-response fault -/// into a conclusive `Unresolved`: with retries allowed the controller's resolve-before-reissue would -/// settle the ambiguity inside the same attempt and the lane would never wedge. Same budget shape, and -/// the same reason, as `gtest_cas_ref_install_safety.cpp`'s `openPoolSingleAttempt`. -Cas::PoolPtr openPoolSingleAttempt(const std::shared_ptr & backend) -{ - Cas::PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; - Cas::CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) - budget.lease_safety_margin_ms = 100; - cfg.cas_request_budget = budget; - return Cas::Pool::open(backend, cfg); -} - -} - /// Part B review, MAJOR 3a: a promote whose ref-log append did not resolve MUST NOT be reported as /// "nothing was committed". /// @@ -1190,8 +1061,9 @@ Cas::PoolPtr openPoolSingleAttempt(const std::shared_ptr & /// whose append never resolved. TEST(CASPartFolderAccess, AnUnresolvedPromoteIsNotReportedAsDefinitelyNotCommitted) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + auto clock = Cas::tests::VirtualRetryClock::installOn(store); const Cas::RootNamespace ns{"srv/t1"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); Cas::CachedPartFolderAccess access(store, cacheOn()); @@ -1202,18 +1074,29 @@ TEST(CASPartFolderAccess, AnUnresolvedPromoteIsNotReportedAsDefinitelyNotCommitt /// The promotion's own ref-log object lands; only the acknowledgement, and the controller's /// verifying read, are lost. Scoped to this namespace's ref log so nothing else consumes the fault. + /// A COUNTED fault cannot produce an unresolved outcome: the request engine settles the ambiguity + /// by an exact read that would find the landed object and report `Committed` inside the very same + /// call. Both legs therefore stay LATCHED for the whole call, and `VirtualRetryClock` pays the + /// retry window in virtual time instead of real wall-clock. backend->fault_substr = store->layout().namespaceStreamPrefix(fixture::fixtureLife(ns)) + "_log/"; backend->mode = Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; - backend->fault_count = 1; + backend->latched = true; expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { prepared.promote(); }); + ASSERT_GT(clock->pauseCount(), 1u) + << "one attempt cannot exhaust the retry window: the fault must have outlasted every reissue"; EXPECT_TRUE(prepared.commitIsUnresolved()) << "a promote whose append may have landed must not be classified as a mechanism failure -- the " "receiver would fetch the bytes and publish the same part a second time"; /// The hazard itself, stated as an assertion: the promote DID commit. Any further append into this - /// table resolves the wedge first, which is what makes the committed row visible. + /// table resolves the wedge first, which is what makes the committed row visible. Disarmed + /// COMPLETELY, because that flush must reach the store normally: a still-armed lost read would + /// fault the wedge's own settling read, and nothing would resolve. + backend->latched = false; backend->mode = Cas::tests::ChunkFaultBackend::Mode::None; + backend->fault_count = 0; + backend->fail_read_once_key.clear(); access.prepareEntries({ns, "flush_driver"}, {inlineEntry("f", "two")}, Cas::ProvenanceOp::Insert).abort(); EXPECT_TRUE(access.existsRef(key, Cas::Freshness::ForceFresh)) << "the promotion object landed, so 'the promote failed' says nothing about the ref"; @@ -1237,8 +1120,15 @@ TEST(CASPartFolderAccess, APostCommitFailureLeavesTheHandleTerminal) auto prepared = access.prepareEntries(key, {inlineEntry("f", "one")}, Cas::ProvenanceOp::Insert); - std::vector seen; - store->setEventSink([&](const Cas::CasEvent & e) { seen.push_back(e); }); + /// Heap-owned, not a plain local: `setEventSink(nullptr)` below only stops FUTURE sink installs + /// from using this closure -- it does not guarantee an already-in-flight background call is not + /// still executing the old one -- and the Pool can outlive this stack frame regardless (a + /// background publish holds `shared_from_this()`). + auto seen = std::make_shared(); + store->setEventSink([seen](const Cas::CasEvent & e) + { + seen->push(e); + }); /// `MEMORY_LIMIT_EXCEEDED` -- what a tracked allocation failure actually raises -- and deliberately /// not `LOGICAL_ERROR`, which aborts at construction in debug/sanitizer builds. @@ -1260,12 +1150,13 @@ TEST(CASPartFolderAccess, APostCommitFailureLeavesTheHandleTerminal) /// abandoned an ALREADY PROMOTED build -- which succeeds, because a promoted build no longer owes a /// precommit removal -- and so ended up terminal too, by accident. What the abandon leaves behind is /// the audit trail of a publish that is reported as thrown away while its ref is committed. - const auto build_aborts = std::count_if(seen.begin(), seen.end(), + const std::vector observed = seen->snapshot(); + const auto build_aborts = std::count_if(observed.begin(), observed.end(), [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::BuildAbort; }); EXPECT_EQ(build_aborts, 0) << "a build whose promote is DURABLE was abandoned by the failed-promote catch: the handle had " "not yet recorded the commit when the post-commit work threw"; - EXPECT_EQ(std::count_if(seen.begin(), seen.end(), + EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const Cas::CasEvent & e) { return e.type == Cas::CasEventType::BuildPublish; }), 1); } diff --git a/src/Disks/tests/gtest_cas_part_folder_view.cpp b/src/Disks/tests/gtest_cas_part_folder_view.cpp index be9b9e91de91..5bd2ab8b7c43 100644 --- a/src/Disks/tests/gtest_cas_part_folder_view.cpp +++ b/src/Disks/tests/gtest_cas_part_folder_view.cpp @@ -47,8 +47,7 @@ std::shared_ptr makeView() return std::make_shared( Cas::PartRefKey{Cas::RootNamespace{"srv/t"}, "part_1"}, Cas::ManifestId{Cas::RootNamespace{"srv/t"}, Cas::ManifestRef{1, 2, 3}}, - /*manifest_size=*/1000, manifest, - /*validated_at_ms=*/42); + /*manifest_size=*/1000, manifest); } std::vector sorted(std::vector v) { std::sort(v.begin(), v.end()); return v; } diff --git a/src/Disks/tests/gtest_cas_part_manifest_format.cpp b/src/Disks/tests/gtest_cas_part_manifest_format.cpp index abb81c1928aa..dd6eab02c003 100644 --- a/src/Disks/tests/gtest_cas_part_manifest_format.cpp +++ b/src/Disks/tests/gtest_cas_part_manifest_format.cpp @@ -5,6 +5,8 @@ #include #include +#include + using namespace DB::Cas; namespace @@ -28,8 +30,7 @@ void expectThrowsCode(int expected_code, F && fn) } } -/// One Blob + one Inline entry, matching the plan's §text-shape illustration verbatim (codecs-v3 -/// phase 6): deliberately NOT path-sorted on input, so the round trip also exercises canonical +/// One Blob + one Inline entry, deliberately NOT path-sorted on input, so the round trip also exercises canonical /// path-order encoding. PartManifest sample() { @@ -58,6 +59,8 @@ PartManifest sample() } +CAS_BATTERY_COVERS(PartManifest); + TEST(CASFormatBattery, PartManifest) { const PartManifest m = sample(); @@ -65,11 +68,11 @@ TEST(CASFormatBattery, PartManifest) /// stays self-consistent with whatever sample() produces, now that decode verifies payload_digest. const String golden = currentFormatHeader("cas_part_manifest") + - "{\"me\":\"5\",\"mb\":\"15\",\"mo\":1,\"ns\":\"00/aa@cas@\",\"pd\":\"" + u128ToHex(m.payload_digest) + "\"}\n" // NOLINT(modernize-raw-string-literal): mixes '\"' quoting with '\n' line endings across this concatenated literal; a raw string can't hold the newline as-is. - "{\"p\":\"a/b.bin\",\"pm\":\"blob\",\"ha\":\"ch128\",\"h\":\"00112233445566778899aabbccddeeff\",\"sz\":4096}\n" - "{\"p\":\"c/small.txt\",\"pm\":\"inline\",\"il\":12}\n" + "{\"epoch\":\"5\",\"build\":\"15\",\"ord\":1,\"namespace\":\"00/aa@cas@\",\"payload_digest\":\"" + u128ToHex(m.payload_digest) + "\"}\n" // NOLINT(modernize-raw-string-literal): mixes '\"' quoting with '\n' line endings across this concatenated literal; a raw string can't hold the newline as-is. + "{\"path\":\"a/b.bin\",\"place\":\"blob\",\"algo\":\"ch128\",\"digest\":\"00112233445566778899aabbccddeeff\",\"size\":4096}\n" + "{\"path\":\"c/small.txt\",\"place\":\"inline\",\"size\":12}\n" "{\"n\":2}\n" - "==> \"c/small.txt\" il=12 <==\n" + "==> \"c/small.txt\" size=12 <==\n" "hello world!\n"; runFormatBattery({FormatId::PartManifest, [&] { return sealObject(FormatId::PartManifest, encodePartManifest(m)); }, @@ -113,17 +116,88 @@ TEST(CASPartManifestFormat, EmptyEntriesRoundTrips) TEST(CASPartManifestFormat, PlacementWordsRenderAndRejectUnknown) { const String text = encodePartManifest(sample()); - EXPECT_NE(text.find("\"pm\":\"blob\""), String::npos); - EXPECT_NE(text.find("\"pm\":\"inline\""), String::npos); + EXPECT_NE(text.find("\"place\":\"blob\""), String::npos); + EXPECT_NE(text.find("\"place\":\"inline\""), String::npos); /// An unknown placement word fails closed. String bad = text; - const size_t pos = bad.find(R"("pm":"blob")"); + const size_t pos = bad.find(R"("place":"blob")"); ASSERT_NE(pos, String::npos); - bad.replace(pos, String(R"("pm":"blob")").size(), R"("pm":"bogus")"); + bad.replace(pos, String(R"("place":"blob")").size(), R"("place":"bogus")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); } +/// Closed-set pin: the two `EntryPlacement` words, walked through `magic_enum::enum_values`, which is what proves the +/// renderer and the parser consult the SAME table: a table entry missing altogether is already a +/// build error at the coverage assert, but two delegates drifting onto different tables is not. +TEST(CASPartManifestFormat, ClosedSetPinsEntryPlacementWords) +{ + EXPECT_EQ(entryPlacementToWireWord(EntryPlacement::Inline), "inline"); + EXPECT_EQ(entryPlacementToWireWord(EntryPlacement::Blob), "blob"); + for (const auto p : magic_enum::enum_values()) + EXPECT_EQ(entryPlacementFromWireWord(entryPlacementToWireWord(p)), p); +} + +TEST(CASPartManifestFormat, SizeBeforePlaceIsAcceptedForBothPlacements) +{ + String blob_first = encodePartManifest(sample()); + const String blob_record = R"("place":"blob","algo":"ch128","digest":"00112233445566778899aabbccddeeff","size":4096)"; + const size_t blob_pos = blob_first.find(blob_record); + ASSERT_NE(blob_pos, String::npos); + blob_first.replace(blob_pos, blob_record.size(), + R"("size":4096,"place":"blob","algo":"ch128","digest":"00112233445566778899aabbccddeeff")"); + EXPECT_EQ(decodePartManifest(blob_first).entries[0].blob_size, 4096u); + + String inline_first = encodePartManifest(sample()); + const String inline_record = R"("place":"inline","size":12)"; + const size_t inline_pos = inline_first.find(inline_record); + ASSERT_NE(inline_pos, String::npos); + inline_first.replace(inline_pos, inline_record.size(), R"("size":12,"place":"inline")"); + EXPECT_EQ(decodePartManifest(inline_first).entries[1].inline_bytes, "hello world!"); +} + +TEST(CASPartManifestFormat, MissingSizeIsRejectedForBothPlacements) +{ + /// The MESSAGE is asserted, not just the code: a manifest whose entry lost its size also fails + /// the payload-digest check (blob) and the banner rebuild (inline), both of which raise the same + /// code, so a code-only assertion would still pass with the per-placement fences deleted. + const auto expect_message = [](const String & text, std::string_view expected) + { + try + { + static_cast(decodePartManifest(text)); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(e.message(), expected); + } + }; + + String blob_missing_size = encodePartManifest(sample()); + const size_t blob_pos = blob_missing_size.find(",\"size\":4096"); + ASSERT_NE(blob_pos, String::npos); + blob_missing_size.erase(blob_pos, String(",\"size\":4096").size()); + expect_message(blob_missing_size, "PartManifest: blob entry 'a/b.bin' missing size"); + + String inline_missing_size = encodePartManifest(sample()); + const size_t inline_pos = inline_missing_size.find(",\"size\":12"); + ASSERT_NE(inline_pos, String::npos); + inline_missing_size.erase(inline_pos, String(",\"size\":12").size()); + expect_message(inline_missing_size, "PartManifest: inline entry 'c/small.txt' missing size"); +} + +TEST(CASPartManifestFormat, DuplicateSizeIsRejected) +{ + String text = encodePartManifest(sample()); + const String size = R"("size":4096)"; + const size_t pos = text.find(size); + ASSERT_NE(pos, String::npos); + text.replace(pos, size.size(), R"("size":1,"size":4096)"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(text); }); +} + /// Proves the payload zone, not JSON-string escaping: an Inline entry whose bytes contain an /// embedded '\n', a NUL byte, and a '"' character round-trip byte-faithfully. If this content were /// carried as a JSON string value it would need escaping (or would be flatly invalid for the NUL @@ -196,7 +270,7 @@ TEST(CASPartManifestFormat, InlineBannerCarriesTheEscapedPath) m.entries = {e}; m.payload_digest = computePayloadDigest(m); - EXPECT_NE(encodePartManifest(m).find("==> \"p\\nq.proj/c.txt\" il=1 <=="), String::npos); + EXPECT_NE(encodePartManifest(m).find("==> \"p\\nq.proj/c.txt\" size=1 <=="), String::npos); } TEST(CASPartManifestFormat, ByteDeterminism) @@ -320,8 +394,8 @@ TEST(CASPartManifestFormat, DecodeRejectsOutOfOrderEntries) m.payload_digest = computePayloadDigest(m); const String text = encodePartManifest(m); - const size_t pos_a = text.find(R"("p":"a/one.bin")"); - const size_t pos_b = text.find(R"("p":"b/two.bin")"); + const size_t pos_a = text.find(R"("path":"a/one.bin")"); + const size_t pos_b = text.find(R"("path":"b/two.bin")"); ASSERT_NE(pos_a, String::npos); ASSERT_NE(pos_b, String::npos); @@ -364,10 +438,10 @@ TEST(CASPartManifestFormat, DecodeRejectsNonAdjacentDuplicatePath) m.payload_digest = computePayloadDigest(m); String forged = encodePartManifest(m); - const String needle = R"("p":"ccc/three.bin")"; + const String needle = R"("path":"ccc/three.bin")"; const size_t pos = forged.find(needle); ASSERT_NE(pos, String::npos); - forged.replace(pos, needle.size(), R"("p":"aaa/one.bin")"); + forged.replace(pos, needle.size(), R"("path":"aaa/one.bin")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(forged); }); } @@ -375,10 +449,10 @@ TEST(CASPartManifestFormat, DecodeRejectsNonAdjacentDuplicatePath) TEST(CASPartManifestFormat, UnknownEntryAlgoFailsClosed) { String bad = encodePartManifest(sample()); - const String needle = R"("ha":"ch128")"; + const String needle = R"("algo":"ch128")"; const size_t pos = bad.find(needle); ASSERT_NE(pos, String::npos); - bad.replace(pos, needle.size(), R"("ha":"bogus")"); + bad.replace(pos, needle.size(), R"("algo":"bogus")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); } @@ -388,7 +462,7 @@ TEST(CASPartManifestFormat, UnknownEntryAlgoFailsClosed) TEST(CASPartManifestFormat, DigestHexWidthMismatchFailsClosedNotBadArguments) { String bad = encodePartManifest(sample()); - const String key = R"("h":")"; + const String key = R"("digest":")"; const size_t key_pos = bad.find(key); ASSERT_NE(key_pos, String::npos); const size_t hex_start = key_pos + key.size(); @@ -427,23 +501,22 @@ TEST(CASPartManifestFormat, TrailingByteAfterPayloadZoneFailsClosed) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); } -/// An Inline entry's record "il" disagrees with what the payload zone's banner+bytes actually -/// declare (the banner and bytes are left as originally written; only the record line's "il" is -/// edited). The record's declared `il` is what decode uses both to build the expected banner text +/// An Inline entry's record `size` disagrees with what the payload zone's banner+bytes actually +/// declare (the banner and bytes are left as originally written; only the record line's `size` is +/// edited). The record's declared `size` is what decode uses both to build the expected banner text /// and to know how many bytes to read from the zone, so this must fail closed rather than silently /// reading the wrong byte count. -TEST(CASPartManifestFormat, InlineRecordIlMismatchWithPayloadZoneBannerFailsClosed) +TEST(CASPartManifestFormat, InlineRecordSizeMismatchWithPayloadZoneBannerFailsClosed) { String bad = encodePartManifest(sample()); - const String needle = "\"il\":12"; + const String needle = "\"size\":12"; const size_t pos = bad.find(needle); ASSERT_NE(pos, String::npos); - bad.replace(pos, needle.size(), "\"il\":13"); + bad.replace(pos, needle.size(), "\"size\":13"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodePartManifest(bad); }); } -/// ==== migrated from gtest_cas_manifest_codec.cpp (deleted in the phase-6 binary->text cutover, -/// Task 3): these exercise refMatchesBody/manifestNamespaceMatches/findEntry/entryRange, pure +/// ==== Migrated manifest helpers: these exercise `refMatchesBody`, `manifestNamespaceMatches`, `findEntry`, and `entryRange`, pure /// functions carried over verbatim from the retired binary codec (untouched by the wire-shape /// migration) — reusing this file's own sample() fixture instead of reintroducing a second one. ==== diff --git a/src/Disks/tests/gtest_cas_part_write.cpp b/src/Disks/tests/gtest_cas_part_write.cpp index 7a426676cbc8..7259f13360ca 100644 --- a/src/Disks/tests/gtest_cas_part_write.cpp +++ b/src/Disks/tests/gtest_cas_part_write.cpp @@ -69,6 +69,39 @@ PoolPtr openPool(const std::shared_ptr & b) return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); } +/// Run the mount plane's request engine on a virtual clock that its own reissue sleep advances -- the +/// plane every write below is admitted on. A policy that has to be EXHAUSTED then closes in +/// microseconds of wall clock instead of ninety real seconds, and a policy that merely reissues costs +/// no sleep at all, so these tests pin retry semantics and never a schedule. +/// The clock a test installed on the mount plane, handed back so the test can assert that the pacing +/// really ran on it. It does not make a LOST injection fast -- each fault double's attempt cap does +/// that -- but it does say that the reissues this test claims to drive were the injected clock's and +/// not the wall clock's. +struct VirtualRequestClock +{ + std::shared_ptr> now; + std::shared_ptr> sleeps; +}; + +/// `step_ms` is a FLOOR on how far each sleep moves the clock. A test that must see a policy exhaust +/// raises it so the window closes after a countable handful of attempts rather than after however many +/// near-zero jitter draws fit into ninety seconds. +VirtualRequestClock useVirtualMountRequestClock(const PoolPtr & store, uint64_t step_ms = 0) +{ + const VirtualRequestClock clock{std::make_shared>(0), + std::make_shared>(0)}; + store->mountRequests().setNowFnForTest([now = clock.now] { return now->load(); }); + /// `+ 1` because a jittered draw may be zero, and a clock that can stand still never closes the + /// window. + store->mountRequests().setSleepFnForTest( + [now = clock.now, sleeps = clock.sleeps, step_ms](uint64_t ms) + { + sleeps->fetch_add(1); + now->fetch_add(std::max(ms, step_ms) + 1); + }); + return clock; +} + /// Start a build whose owning manifest namespace + final ref name are `ns`/`ref` (promote/stageManifest /// derive the manifest namespace by splitting PartWriteInfo::intended_ref on the LAST '/'). PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) @@ -141,45 +174,51 @@ ManifestId publishOneBlobPart( } /// A one-shot backend hook (mirrors the WriteCountingBackend delegation pattern in gtest_cas_pool.cpp): -/// it delegates every op to a wrapped Backend, but the FIRST time head(target_key) is called it fires a -/// deleteExact(target_key, condemned_token) AFTER computing the (present) HEAD result and BEFORE returning -/// it — simulating GC's exact-token content delete landing in the writer's HEAD->GET window (B136). +/// it delegates every op to a wrapped Backend, but the FIRST time the target key is HEADed it fires an +/// exact-incarnation delete of that key AFTER computing the (present) HEAD result and BEFORE returning +/// it — GC emptying the key underneath the writer's observation, so the writer decides from a HEAD +/// whose object no longer exists. `fired` is public because a test that does not assert it cannot tell +/// a plumbed fault from an unplumbed one. class HeadThenDeleteOnceBackend final : public DB::Cas::Backend { public: - HeadThenDeleteOnceBackend(BackendPtr inner_, String target_key_, DB::Cas::Token condemned_) - : inner(std::move(inner_)), target_key(std::move(target_key_)), condemned(condemned_) {} + HeadThenDeleteOnceBackend(BackendPtr inner_, String target_key_, Etag condemned_) + : inner(std::move(inner_)), target_key(std::move(target_key_)), + condemned_value(PersistedEtag::capture(condemned_).value) {} - DB::Cas::HeadResult head(const String & k) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The fault sits on the HEAD primitive, which is the only path a writer's mandatory HEAD takes. + std::optional read(const String & key, TransportAccess & access) override { return inner->read(key, access); } + std::optional head(const String & key, TransportAccess & access) override { - const DB::Cas::HeadResult hr = inner->head(k); - if (k == target_key && !fired) + const auto observed = inner->head(key, access); + if (key == target_key && !fired) { fired = true; - /// GC's single content-delete site, landing in the HEAD->GET window. - inner->deleteExact(target_key, condemned); + /// Fires here: after the inner HEAD observation, before it is returned to the writer. + inner->remove(target_key, condemned_value, access); } - return hr; + return observed; } - - std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { return inner->putIfAbsent(k, b, meta); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { return inner->putOverwrite(k, b, e, meta); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { return inner->casPut(k, b, e, meta); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } + + bool fired = false; private: BackendPtr inner; String target_key; - DB::Cas::Token condemned; - bool fired = false; + String condemned_value; }; /// A delegating backend that counts head()/get() calls per key. Lets a test assert the promote gate @@ -193,19 +232,31 @@ class KeyCountingBackend final : public DB::Cas::Backend size_t headCountFor(const String & k) const { auto it = head_counts.find(k); return it == head_counts.end() ? 0 : it->second; } size_t getCountFor(const String & k) const { auto it = get_counts.find(k); return it == get_counts.end() ? 0 : it->second; } - DB::Cas::HeadResult head(const String & k) override { ++head_counts[k]; return inner->head(k); } - std::optional get(const String & k, DB::Cas::Range r) override { ++get_counts[k]; return inner->get(k, r); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::ListPage list(const String & pfx, const String & c, size_t l) override { return inner->list(pfx, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { return inner->putIfAbsent(k, b, meta); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// Counted on the primitives: `Backend::probeSentinelRaw` reaches the store through them, and + /// `CasOperation` is the only caller of `Backend` now, so this is the one place left to count. + std::optional read(const String & key, TransportAccess & access) override { - inner->publishBlob(request); + ++get_counts[key]; + return inner->read(key, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { return inner->putOverwrite(k, b, e, meta); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { return inner->casPut(k, b, e, meta); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::optional head(const String & key, TransportAccess & access) override + { + ++head_counts[key]; + return inner->head(key, access); + } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override + { + return inner->write(key, bytes, expected_value, access); + } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: BackendPtr inner; @@ -217,6 +268,8 @@ class KeyCountingBackend final : public DB::Cas::Backend class RacingBlobPublicationBackend final : public InMemoryBackend { public: + /// Unhide the legacy overload the primitive override below would otherwise hide. + using InMemoryBackend::head; void watch(String key_) { std::lock_guard lock(mutex); @@ -225,12 +278,14 @@ class RacingBlobPublicationBackend final : public InMemoryBackend publish_calls = 0; } - HeadResult head(const String & requested_key) override + /// Both faults sit on the transport primitives: a writer's mandatory HEAD and its publication both + /// reach the store through them. + std::optional head(const String & requested_key, TransportAccess & access) override { if (requested_key != key) - return InMemoryBackend::head(requested_key); + return InMemoryBackend::head(requested_key, access); - const HeadResult observed = InMemoryBackend::head(requested_key); + const std::optional observed = InMemoryBackend::head(requested_key, access); std::unique_lock lock(mutex); ++head_calls; cv.notify_all(); @@ -238,14 +293,14 @@ class RacingBlobPublicationBackend final : public InMemoryBackend return observed; } - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { if (request.destination_key == key) { std::lock_guard lock(mutex); ++publish_calls; } - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); } String key; @@ -301,7 +356,8 @@ TEST(CASPartWrite, RacingWritersBothHeadMissAndPublishEquivalentBodies) << "both equivalent writers may publish after racing absent observations"; EXPECT_EQ(first->dependencyProof(ref), BlobDependencyProof::Materialized); EXPECT_EQ(second->dependencyProof(ref), BlobDependencyProof::Materialized); - const auto stored = backend->get(store->layout().blobKey(ref)); + OperationForTest op(*backend); + const auto stored = (*op).read(store->layout().blobKey(ref), Retry::once()); ASSERT_TRUE(stored.has_value()); EXPECT_EQ(stored->bytes.substr(store->poolMeta().blob_header_len), payload); } @@ -326,7 +382,8 @@ TEST(CASPartWrite, WrongSizeSourcePublishesNothing) { build->putBlob(ref, std::move(source)); }); - EXPECT_FALSE(backend->head(store->layout().blobKey(ref)).exists); + OperationForTest op(*backend); + EXPECT_FALSE((*op).head(store->layout().blobKey(ref), Retry::once()).has_value()); } TEST(CASPartWriteTxn, PutBlobWritesEnvelopeWithFixedHeader) @@ -338,7 +395,8 @@ TEST(CASPartWriteTxn, PutBlobWritesEnvelopeWithFixedHeader) auto ref = build->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); EXPECT_EQ(ref.size, 11u); - auto raw = b->get(s->layout().blobKey(ref.ref)); + OperationForTest op(*b); + auto raw = (*op).read(s->layout().blobKey(ref.ref), Retry::once()); ASSERT_TRUE(raw.has_value()); auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(h.header_len, s->poolMeta().blob_header_len); /// 256 @@ -392,7 +450,10 @@ TEST(CASPartWriteTxn, PutBlobDedupSecondWriterAdopts) /// First writer publishes under its durable precommit edge. auto build_a = precommittedBuildForPayload(s, RootNamespace{"srv/tbl-a"}, "ref_a", "dup"); auto ref_a = build_a->putBlob(idOf("dup"), BlobSource::fromString("dup")); - const Token token_a = b->head(s->layout().blobKey(ref_a.ref)).token; + OperationForTest op(*b); + const auto head_a = (*op).head(s->layout().blobKey(ref_a.ref), Retry::once()); + ASSERT_TRUE(head_a.has_value()); + const Etag token_a = head_a->etag; /// Second writer ADOPTS — the adopt must happen under a durable precommit edge (EDGE-BEFORE-OBSERVE: /// stageManifest -> precommitAdd -> putBlob), so give build_b the wiring order. @@ -404,7 +465,9 @@ TEST(CASPartWriteTxn, PutBlobDedupSecondWriterAdopts) EXPECT_EQ(ref_b.ref, ref_a.ref); /// A's incarnation survives — the second writer adopts, nothing was overwritten. - EXPECT_EQ(b->head(s->layout().blobKey(ref_a.ref)).token, token_a); + const auto head_a_after = (*op).head(s->layout().blobKey(ref_a.ref), Retry::once()); + ASSERT_TRUE(head_a_after.has_value()); + EXPECT_EQ(head_a_after->etag, token_a); } /// Task 3 (spec §meta-protocols v3): the writer's dedup gate no longer consults the RetireView for the @@ -509,7 +572,10 @@ TEST(CASPartWriteTxn, PutBlobAdoptsWhenMetaCleanNoRetireView) raw_body += payload; writeRawBlobBody(*b, s->layout(), hash, raw_body); writeMetaClean(*b, s->layout(), hash, payload.size()); - const Token t0 = b->head(blob_key).token; + OperationForTest op(*b); + const auto head0 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head0.has_value()); + const Etag t0 = head0->etag; /// Adopt must happen under a durable precommit edge (EDGE-BEFORE-OBSERVE), mirroring /// PutBlobDedupSecondWriterAdopts above. @@ -521,7 +587,9 @@ TEST(CASPartWriteTxn, PutBlobAdoptsWhenMetaCleanNoRetireView) EXPECT_EQ(ref.ref, id); /// Adopted: the pre-seeded incarnation survives untouched — no putOverwrite/re-upload happened. - EXPECT_EQ(b->head(blob_key).token, t0); + const auto head1 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head1.has_value()); + EXPECT_EQ(head1->etag, t0); const auto lm = loadMetaForTest(*b, s->layout(), hash); ASSERT_TRUE(lm.has_value()); @@ -580,7 +648,10 @@ TEST(CASPartWriteTxn, PutBlobRepublishesWhenMetaCondemned) writeRawBlobBody(*b, s->layout(), hash, raw_body); writeMetaClean(*b, s->layout(), hash, payload.size()); condemnMeta(*b, s->layout(), hash, /*condemn_round*/ 1); - const Token t0 = b->head(blob_key).token; + OperationForTest op(*b); + const auto head0 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head0.has_value()); + const Etag t0 = head0->etag; /// No retire-view seeding: the replacement is decided from the metadata point-read. auto build = precommittedBuildForPayload( @@ -589,10 +660,10 @@ TEST(CASPartWriteTxn, PutBlobRepublishesWhenMetaCondemned) EXPECT_EQ(ref.ref, id); /// Resurrected: the condemned incarnation was displaced by a fresh one. - const HeadResult hr = b->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_NE(hr.token, t0) << "a condemned incarnation must be displaced by a fresh publication"; - EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch) + const auto hr = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_NE(hr->etag, t0) << "a condemned incarnation must be displaced by a fresh publication"; + EXPECT_EQ((*op).remove(blob_key, t0, Retry::once()), Removal::Mismatch) << "the condemned token must never return (INV-NO-RETURN)"; const auto lm = loadMetaForTest(*b, s->layout(), hash); @@ -648,7 +719,8 @@ TEST(CASPartWriteTxn, PutBlobWrongSizeFailsClosed) build->putBlob(id, std::move(lying)); }); /// The cancelled stream created nothing. - EXPECT_FALSE(b->head(s->layout().blobKey(id)).exists); + OperationForTest op(*b); + EXPECT_FALSE((*op).head(s->layout().blobKey(id), Retry::once()).has_value()); } /// The happy-path upload STREAMS the source directly into the put sink — it does NOT pre-materialize the @@ -678,7 +750,8 @@ TEST(CASPartWriteTxn, PutBlobStreamsSourceOnceNoFullMaterialization) EXPECT_EQ(invocations, 1) << "happy-path upload must stream the source exactly once (no pre-materialization pass)"; /// And the object really landed with the streamed payload (at the fixed header offset). - auto raw = b->get(s->layout().blobKey(ref.ref)); + OperationForTest op(*b); + auto raw = (*op).read(s->layout().blobKey(ref.ref), Retry::once()); ASSERT_TRUE(raw.has_value()); auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(raw->bytes.substr(h.header_len), payload); @@ -734,13 +807,16 @@ TEST(CASPartWriteTxn, PutBlobRepublishesVanishedBodyFromHeldSource) /// 1. Write payload-X via a throwaway build to create the blob; capture its token t0. BlobRef id; - Token t0; + std::optional t0; { auto s0 = openPool(b); auto build0 = precommittedBuildForPayload( s0, RootNamespace{"srv1/republish-vanished-seed"}, "part", "payload-X"); id = build0->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")).ref; - t0 = b->head(s0->layout().blobKey(id)).token; + OperationForTest op(*b); + const auto head = (*op).head(s0->layout().blobKey(id), Retry::once()); + ASSERT_TRUE(head.has_value()); + t0 = head->etag; build0->abandon(); } @@ -751,10 +827,10 @@ TEST(CASPartWriteTxn, PutBlobRepublishesVanishedBodyFromHeldSource) /// meta; t0 stays as the body token the delete-hook below fires with. condemnMeta(*b, layout, u128Of("payload-X"), /*condemn_round*/ 1); - /// 3. Wrap the backend so the NEXT head(blob_key) returns the (present) result and THEN fires - /// deleteExact(blob_key, t0) exactly once — GC's delete in the HEAD->GET window. Open a FRESH + /// 3. Wrap the backend so the NEXT head(blob_key) returns the (present) result and THEN deletes that + /// exact incarnation once — GC emptying the key underneath the writer's observation. Open a FRESH /// Pool over the hook so its retire view (refreshed at open) sees the condemnation. - auto hook = std::make_shared(b, blob_key, t0); + auto hook = std::make_shared(b, blob_key, *t0); auto s = Pool::open(hook, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); auto build = precommittedBuildForPayload( s, RootNamespace{"srv1/republish-vanished"}, "part", "payload-X"); @@ -765,19 +841,25 @@ TEST(CASPartWriteTxn, PutBlobRepublishesVanishedBodyFromHeldSource) auto ref = build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); EXPECT_EQ(ref.ref, id); + /// The republication below reads the same whether or not the key was emptied mid-observation, so + /// this is what says the injected delete actually ran: a fault plumbed onto a method the writer no + /// longer calls leaves it false. It proves the seam fires, not that the end state depended on it. + EXPECT_TRUE(hook->fired) << "the injected delete must have run inside the mandatory HEAD"; + /// 5. The blob is present again under a FRESH token, with the same payload; and the condemned token /// never returns (INV-NO-RETURN). - const HeadResult hr = b->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_NE(hr.token, t0); + OperationForTest op(*b); + const auto hr = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_NE(hr->etag, *t0); - auto raw = b->get(blob_key); + auto raw = (*op).read(blob_key, Retry::once()); ASSERT_TRUE(raw.has_value()); auto h = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(h.header_len, s->poolMeta().blob_header_len); EXPECT_EQ(raw->bytes.substr(h.header_len), "payload-X"); - EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_EQ((*op).remove(blob_key, *t0, Retry::once()), Removal::Mismatch); /// The freshness meta must be reconciled to Clean too, not left stale at Condemned: the fresh /// re-upload's meta write (writeFreshMetaClean) must find and fix the pre-existing Condemned @@ -789,21 +871,69 @@ TEST(CASPartWriteTxn, PutBlobRepublishesVanishedBodyFromHeldSource) << "a fresh re-upload over a stale Condemned marker must reconcile it back to Clean"; } -/// A persistently-failing freshness-meta write (every attempt of every outer reload-retry) must -/// surface as a controlled retry-later signal, not silently succeed with the marker left stale -/// (S22 RCA). The blob body PUT -/// itself is unaffected (MetaWriteFaultBackend only faults `.meta` keys) -- only the meta write -/// exhausts, and that exhaustion must reach putBlob's caller as NETWORK_ERROR. +namespace +{ + +/// Every `.meta` write is ambiguous, forever: the store's answer is lost on each attempt, so the +/// engine settles each one by a read and reissues, and only the policy's own window ends the call. +class AmbiguousMetaWriteBackend final : public InMemoryBackend +{ +public: + /// Past this many faulted attempts the double stops faulting and raises a deterministic local + /// failure instead, which every loop here surfaces unchanged. A caller whose retries are no longer + /// bounded therefore FAILS on the wrong error code rather than running until the suite times out. + int attempt_cap = 100; + int attempts = 0; + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override + { + if (!key.ends_with(".meta")) + return InMemoryBackend::write(key, bytes, expected_value, access); + if (++attempts > attempt_cap) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "AmbiguousMetaWriteBackend: {} attempts exceeded the cap of {}; the caller's retries " + "are no longer bounded", attempts, attempt_cap); + throw Poco::TimeoutException("AmbiguousMetaWriteBackend: blob meta write response lost"); + } +}; + +} + +/// A give-up reports the attempts it actually SENT, not merely whether it sent any. An operator +/// counting a write's retries has to count one that gave up as well as one that committed, and +/// `sent_any` cannot say how many. +TEST(CASPartWriteTxn, AGiveUpReportsHowManyAttemptsItSent) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const VirtualRequestClock clock = useVirtualMountRequestClock(s, /*step_ms=*/10'000); + b->attempts = 0; + + const BlobRef ref = idOf("give-up-attempt-count"); + CasOperation op = s->mountRequests().admit(); + const WriteResult result = putMetaIfAbsent(op, s->layout(), ref, BlobMeta{.state = MetaState::Clean, .size = 7}); + + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr) << "every attempt was ambiguous and none landed"; + EXPECT_TRUE(gave_up->sent_any); + EXPECT_GT(gave_up->attempts_sent, 1u) << "the write reissued before it gave up"; + EXPECT_EQ(gave_up->attempts_sent, static_cast(b->attempts)) + << "every attempt the store saw must be in the count the give-up reports"; + EXPECT_GT(clock.sleeps->load(), 0u) + << "the reissues were paced on the injected clock, so this exhaustion cost no wall time"; +} + +/// A persistently-failing freshness-meta write must surface as a controlled retry-later signal, not +/// silently succeed with the marker left stale. The blob body publication itself is unaffected (only +/// `.meta` keys are faulted) -- only the meta reconciliation exhausts, and that exhaustion must reach +/// putBlob's caller as NETWORK_ERROR. TEST(CASPartWriteTxn, PutBlobFreshMetaExhaustionThrowsRetryLater) { - /// Short budget + zero backoff: keep the test fast. Each of the metadata reconciliation loop's 8 outer - /// attempts calls putMetaIfAbsent, which itself retries up to max_attempts times internally — - /// with max_attempts=1 the controller gives up on the first faulted attempt each time. - CasRequestBudget budget; - budget.max_attempts = 1; - budget.retry_initial_backoff_ms = 0; - auto b = std::make_shared(); - auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + auto b = std::make_shared(); + auto s = openPool(b); + const VirtualRequestClock clock = useVirtualMountRequestClock(s, /*step_ms=*/10'000); const String payload = "fresh-meta-exhaustion-payload"; auto build = precommittedBuildForPayload( @@ -813,15 +943,19 @@ TEST(CASPartWriteTxn, PutBlobFreshMetaExhaustionThrowsRetryLater) build->putBlob(idOf(payload), BlobSource::fromString(payload)); }); + EXPECT_GT(clock.sleeps->load(), 0u) + << "the reissues were paced on the injected clock, so this exhaustion cost no wall time"; + /// The body itself landed (only .meta writes are faulted) -- confirming the failure is /// specifically the freshness marker, not the blob body. - const HeadResult hr = b->head(s->layout().blobKey(idOf(payload))); - EXPECT_TRUE(hr.exists) << "the body PUT is unaffected by the meta-only fault"; + OperationForTest op(*b); + const auto hr = (*op).head(s->layout().blobKey(idOf(payload)), Retry::once()); + EXPECT_TRUE(hr.has_value()) << "the body PUT is unaffected by the meta-only fault"; } /// INV-1 (revival-from-source): a condemned blob is NEVER read via GET to revive it. -/// putBlob on a condemned-dedup hit must re-upload from its OWN source bytes — never calling -/// backend().get(blob_key). This test counts backend GETs on the blob key and asserts zero. +/// putBlob on a condemned-dedup hit must re-upload from its OWN source bytes — never reading the +/// blob key's body. This test counts backend GETs on the blob key and asserts zero. TEST(CASPartWriteTxn, PutBlobCondemnedDedupNeverGetsTheDyingObject) { /// A delegating backend that counts get() calls on a specific key to assert INV-1. @@ -831,24 +965,28 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupNeverGetsTheDyingObject) : inner(std::move(inner_)), watched_key(std::move(watched_key_)) {} size_t get_count = 0; - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - std::optional get(const String & k, DB::Cas::Range r) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// `read` is a GET, so it counts on the watched key too -- otherwise the INV-1 fence below + /// would stop seeing a revival read the moment its caller takes the primitive path. + std::optional read(const String & key, TransportAccess & access) override { - if (k == watched_key) + if (key == watched_key) ++get_count; - return inner->get(k, r); + return inner->read(key, access); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & bts, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, bts, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & bts, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & bts, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & tok) override { return inner->deleteExact(k, tok); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: BackendPtr inner; String watched_key; @@ -856,26 +994,33 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupNeverGetsTheDyingObject) auto b = std::make_shared(); - /// 1. Upload blob Y via a throwaway build; capture the token t0. + /// 1. Upload blob Y via a throwaway build; capture the incarnation t0 the way GC persists it, and + /// let GC's exact-incarnation delete land before the writer's dedup hit. BlobRef id; - Token t0; + PersistedEtag t0; { auto s0 = openPool(b); auto build0 = precommittedBuildForPayload( s0, RootNamespace{"srv1/condemned-absent-seed"}, "part", "payload-Y"); id = build0->putBlob(idOf("payload-Y"), BlobSource::fromString("payload-Y")).ref; - t0 = b->head(s0->layout().blobKey(id)).token; + CasOperation op0 = s0->mountRequests().admit(); + const String seed_key = s0->layout().blobKey(id); + const auto seeded = op0.head(seed_key, Retry::standard()); + ASSERT_TRUE(seeded.has_value()); + t0 = PersistedEtag::capture(seeded->etag); + ASSERT_EQ(op0.remove(seed_key, seeded->etag, Retry::standard()), Removal::Removed); build0->abandon(); } - /// 2. Condemn (Blob, hash(Y), t0) in the retire view, then GC-delete the object so it is absent - /// (simulates GC completing the delete before the writer's dedup hit). + /// 2. Condemn (Blob, hash(Y), t0) in the retire view; the object is already absent. DB::Cas::Layout layout("p"); const String blob_key = layout.blobKey(id); injectRetire(*b, layout, /*round*/ 1, /*shard*/ 0, {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-Y"))}, .token = t0, .size = 9}}); - b->deleteExact(blob_key, t0); - ASSERT_FALSE(b->head(blob_key).exists); + { + OperationForTest raw_op(*b); + ASSERT_FALSE((*raw_op).head(blob_key, Retry::once()).has_value()); + } /// 3. Open a fresh Pool over a GET-counting wrapper; the retire view sees the condemnation at open. auto counting = std::make_shared(b, blob_key); @@ -890,10 +1035,14 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupNeverGetsTheDyingObject) EXPECT_EQ(ref.ref, id); EXPECT_EQ(counting->get_count, 0u) << "INV-1: putBlob must not GET the dying object to revive it"; - const HeadResult hr = b->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_NE(hr.token, t0) << "a fresh incarnation must have a new token"; - const auto raw = b->get(blob_key); + /// The republished body must NOT read back as the incarnation GC condemned -- that compare is the + /// one GC's redelete makes, so it is the one this asserts. + CasRequests probe(b, Fence::open()); + CasOperation probe_op = probe.admit(); + const auto after = probe_op.head(blob_key, Retry::standard()); + ASSERT_TRUE(after.has_value()); + EXPECT_FALSE(t0.matches(after->etag)) << "a fresh publication must have a fresh incarnation"; + const auto raw = probe_op.read(blob_key, Retry::standard()); ASSERT_TRUE(raw.has_value()); const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(raw->bytes.substr(hdr.header_len), "payload-Y"); @@ -909,24 +1058,28 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupPresentNeverGetsTheDyingObject) : inner(std::move(inner_)), watched_key(std::move(watched_key_)) {} size_t get_count = 0; - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - std::optional get(const String & k, DB::Cas::Range r) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// `read` is a GET, so it counts on the watched key too -- otherwise the INV-1 fence below + /// would stop seeing a revival read the moment its caller takes the primitive path. + std::optional read(const String & key, TransportAccess & access) override { - if (k == watched_key) + if (key == watched_key) ++get_count; - return inner->get(k, r); + return inner->read(key, access); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & bts, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, bts, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & bts, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & bts, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & tok) override { return inner->deleteExact(k, tok); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: BackendPtr inner; String watched_key; @@ -936,13 +1089,16 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupPresentNeverGetsTheDyingObject) /// 1. Upload blob Z via a throwaway build; capture the token t0. BlobRef id; - Token t0; + std::optional t0; { auto s0 = openPool(b); auto build0 = precommittedBuildForPayload( s0, RootNamespace{"srv1/condemned-present-seed"}, "part", "payload-Z"); id = build0->putBlob(idOf("payload-Z"), BlobSource::fromString("payload-Z")).ref; - t0 = b->head(s0->layout().blobKey(id)).token; + OperationForTest seed_op(*b); + const auto head0 = (*seed_op).head(s0->layout().blobKey(id), Retry::once()); + ASSERT_TRUE(head0.has_value()); + t0 = head0->etag; build0->abandon(); } @@ -951,7 +1107,10 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupPresentNeverGetsTheDyingObject) const String blob_key = layout.blobKey(id); /// v3: condemn via the per-hash meta (the writer's freshness point-read), object still PRESENT. condemnMeta(*b, layout, u128Of("payload-Z"), /*condemn_round*/ 1); - ASSERT_TRUE(b->head(blob_key).exists) << "blob must be PRESENT for the condemned-present path"; + { + OperationForTest raw_op(*b); + ASSERT_TRUE((*raw_op).head(blob_key, Retry::once()).has_value()) << "blob must be PRESENT for the condemned-present path"; + } /// 3. Open a fresh Pool over a GET-counting wrapper. auto counting = std::make_shared(b, blob_key); @@ -964,10 +1123,11 @@ TEST(CASPartWriteTxn, PutBlobCondemnedDedupPresentNeverGetsTheDyingObject) EXPECT_EQ(ref.ref, id); EXPECT_EQ(counting->get_count, 0u) << "INV-1: putBlob must not GET the condemned object"; - const HeadResult hr = b->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_NE(hr.token, t0) << "condemned incarnation must be displaced by a fresh token"; - const auto raw = b->get(blob_key); + OperationForTest raw_op(*b); + const auto hr = (*raw_op).head(blob_key, Retry::once()); + ASSERT_TRUE(hr.has_value()); + EXPECT_NE(hr->etag, *t0) << "condemned incarnation must be displaced by a fresh token"; + const auto raw = (*raw_op).read(blob_key, Retry::once()); ASSERT_TRUE(raw.has_value()); const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(raw->bytes.substr(hdr.header_len), "payload-Z"); @@ -1099,7 +1259,10 @@ TEST(CASPartWriteTxn, PromoteTrustsAdoptedLeafEvenIfBackendRaced) seed->promote(seed_ns, "part", seed->buildId(), seed_manifest); } const String blob_key = s->layout().blobKey(streamRefOf("payload-RACE")); - const Token t0 = b->head(blob_key).token; + OperationForTest op(*b); + const auto head0 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head0.has_value()); + const Etag t0 = head0->etag; auto build = startBuildFor(s, ns, "part_1"); const ManifestEntry entry = blobManifestEntryStreaming("data.bin", "payload-RACE"); @@ -1107,8 +1270,8 @@ TEST(CASPartWriteTxn, PromoteTrustsAdoptedLeafEvenIfBackendRaced) const ManifestId id = build->stageManifest({entry}); build->precommitAdd(ns, "part_1", id); - ASSERT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::Deleted); - ASSERT_FALSE(b->head(blob_key).exists); + ASSERT_EQ((*op).remove(blob_key, t0, Retry::once()), Removal::Removed); + ASSERT_FALSE((*op).head(blob_key, Retry::once()).has_value()); EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); @@ -1172,7 +1335,10 @@ TEST(CASPartWriteTxn, MissingDependencyProofFailsClosed) seed->promote(seed_ns, "part", seed->buildId(), seed_manifest); } const String blob_key = s->layout().blobKey(streamRefOf("payload-NODEP")); - const Token t0 = b->head(blob_key).token; + OperationForTest op(*b); + const auto head0 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head0.has_value()); + const Etag t0 = head0->etag; auto build = startBuildFor(s, ns, "part_1"); const ManifestEntry entry = blobManifestEntryStreaming("data.bin", "payload-NODEP"); @@ -1192,7 +1358,9 @@ TEST(CASPartWriteTxn, MissingDependencyProofFailsClosed) "no dependency proof"); EXPECT_FALSE(s->resolveRef(ns, "part_1").has_value()); /// The pool blob was never touched (no probe, no displacement). - EXPECT_EQ(b->head(blob_key).token, t0); + const auto head1 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(head1.has_value()); + EXPECT_EQ(head1->etag, t0); } TEST(CASPartWriteTxn, PromoteRevalidatesBlobPresenceFailClosed) @@ -1260,7 +1428,7 @@ TEST(CASPartWriteTxn, AbandonRemovesStagedDebrisAndDisables) { /// Port of AbandonLeavesDebrisAndDisables to the new abandon semantics (CasPartWriteTxn.cpp abandon): /// abandon best-effort exact-token-DELETEs this build's STAGED manifest debris, leaves blob bodies - /// (full GC's job via min_active), and disables the build (further ops throw via requireAlive). + /// (full GC's job via min_active_build_sequence), and disables the build (further ops throw via requireAlive). auto b = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"srv1/tbl"}; @@ -1272,14 +1440,15 @@ TEST(CASPartWriteTxn, AbandonRemovesStagedDebrisAndDisables) const ManifestId mid = build->stageManifest({blobManifestEntry("f", "kept")}); /// The staged manifest body and the blob are present before abandon. - EXPECT_TRUE(b->head(s->layout().blobKey(blob_ref.ref)).exists); - EXPECT_TRUE(b->head(s->layout().manifestKey(mid)).exists); + OperationForTest op(*b); + EXPECT_TRUE((*op).head(s->layout().blobKey(blob_ref.ref), Retry::once()).has_value()); + EXPECT_TRUE((*op).head(s->layout().manifestKey(mid), Retry::once()).has_value()); build->abandon(); /// Blob stays (debris — full GC reclaims it). The staged manifest debris is best-effort cleaned now. - EXPECT_TRUE(b->head(s->layout().blobKey(blob_ref.ref)).exists); - EXPECT_FALSE(b->head(s->layout().manifestKey(mid)).exists) + EXPECT_TRUE((*op).head(s->layout().blobKey(blob_ref.ref), Retry::once()).has_value()); + EXPECT_FALSE((*op).head(s->layout().manifestKey(mid), Retry::once()).has_value()) << "abandon must best-effort delete this build's staged manifest debris"; /// Further operations throw via requireAlive. @@ -1327,9 +1496,11 @@ TEST(CASPartWriteTxn, PublishHappyPathRoundTrip) const auto * entry = findEntry(manifest.entries, "data.bin"); ASSERT_TRUE(entry != nullptr); const auto loc = s->locate(*entry); - auto got = b->get(loc.key, Range{loc.offset, loc.length}); + /// The located window of the blob object, sliced by the test: the seam reads whole objects. + OperationForTest op(*b); + auto got = (*op).read(loc.key, Retry::once()); ASSERT_TRUE(got.has_value()); - EXPECT_EQ(got->bytes, "hello world"); + EXPECT_EQ(got->bytes.substr(static_cast(loc.offset), static_cast(loc.length)), "hello world"); } TEST(CASPartWriteTxn, PromoteCrossNamespaceManifestFailsClosed) @@ -1386,7 +1557,10 @@ TEST(CASPartWriteTxn, PublishIntoSecondNamespaceSameBlob) build1->precommitAdd(ns1, "part_1", id1); auto blob = build1->putBlob(idOf("hello world"), BlobSource::fromString("hello world")); const String blob_key = s->layout().blobKey(blob.ref); - const Token blob_token = b->head(blob_key).token; + OperationForTest op(*b); + const auto blob_head0 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(blob_head0.has_value()); + const Etag blob_token = blob_head0->etag; build1->promote(ns1, "part_1", build1->buildId(), id1); /// Second build publishes part_1 in ns2 referencing the SAME blob: putBlob dedup-hits and ADOPTS the @@ -1406,7 +1580,9 @@ TEST(CASPartWriteTxn, PublishIntoSecondNamespaceSameBlob) EXPECT_EQ(r2->manifest_id, id2); /// The blob object was uploaded once: its token is unchanged after both publishes. - EXPECT_EQ(b->head(blob_key).token, blob_token); + const auto blob_head1 = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(blob_head1.has_value()); + EXPECT_EQ(blob_head1->etag, blob_token); } /// Task 10: refs are no longer sharded (one whole-table cache per namespace, spec §Table State), so @@ -1477,24 +1653,36 @@ TEST(CASPartWriteTxn, AdoptEvidenceRecordsTrustedManifestDependencyProofWithoutI size_t puts = 0; size_t gets = 0; - HeadResult head(const String & k) override { ++heads; return inner->head(k); } - void publishBlob(const BlobPublishRequest & request) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The primitives count on the same three counters, so "no backend op" stays a total claim + /// whichever path a caller takes. + std::optional read(const String & key, TransportAccess & access) override + { + ++gets; + return inner->read(key, access); + } + std::optional head(const String & key, TransportAccess & access) override + { + ++heads; + return inner->head(key, access); + } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { ++puts; - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - std::optional get(const String & k, Range r) override { ++gets; return inner->get(k, r); } - std::optional getStream(const String & k, Range r) override { return inner->getStream(k, r); } - ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - PutResult putIfAbsent(const String & k, const String & bts, const ObjectMeta & m) override + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { ++puts; - return inner->putIfAbsent(k, bts, m); + inner->publish(request, access); } - PutResult putOverwrite(const String & k, const String & bts, const Token & e, const ObjectMeta & m) override { return inner->putOverwrite(k, bts, e, m); } - CasResult casPut(const String & k, const String & bts, const std::optional & e, const ObjectMeta & m) override { return inner->casPut(k, bts, e, m); } - DeleteOutcome deleteExact(const String & k, const Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + Dialect dialect() const override { return inner->dialect(); } private: BackendPtr inner; }; @@ -1564,12 +1752,15 @@ TEST(CASPartWriteTxn, ConvergesUnderProductiveGc) /// 1. PartWriteTxn A creates H ("shared-content"), publishes a part referencing it, then drops the ref. /// Capture H's first incarnation token so we can condemn exactly it. BlobRef h; - Token h_token0; + PersistedEtag h_token0; { auto s0 = Pool::open(b, cfg); publishOneBlobPart(s0, ns, "part_1", "f", content); h = idOf(content); - h_token0 = b->head(s0->layout().blobKey(h)).token; + CasOperation op0 = s0->mountRequests().admit(); + const auto seeded = op0.head(s0->layout().blobKey(h), Retry::standard()); + ASSERT_TRUE(seeded.has_value()); + h_token0 = PersistedEtag::capture(seeded->etag); s0->dropRef(ns, "part_1"); } @@ -1601,9 +1792,10 @@ TEST(CASPartWriteTxn, ConvergesUnderProductiveGc) const auto ref_b = build_b->putBlob(h, BlobSource::fromString(content)); ASSERT_EQ(ref_b.ref, h); - const HeadResult after_reupload = b->head(blob_key); - ASSERT_TRUE(after_reupload.exists); - EXPECT_NE(after_reupload.token, h_token0); /// a genuinely fresh incarnation + CasOperation reupload_op = s->mountRequests().admit(); + const auto after_reupload = reupload_op.head(blob_key, Retry::standard()); + ASSERT_TRUE(after_reupload.has_value()); + EXPECT_FALSE(h_token0.matches(after_reupload->etag)); /// a genuinely fresh incarnation /// 4. THE ADVERSARIAL LOOP. A real, productive GC keeps trying to reclaim. It reclaims the now- /// unreferenced part_1 manifest (build A's, UNprotected) but H stays pinned by B's PRECOMMIT edge @@ -1620,10 +1812,11 @@ TEST(CASPartWriteTxn, ConvergesUnderProductiveGc) /// (a frozen B would have its precommit reclaimed; an advancing seq keeps it). s->renewWatermarkOnce(); gc.runRegularRound(); - const HeadResult hr = b->head(blob_key); - ASSERT_TRUE(hr.exists) << "H was deleted by GC at round " << round_no + OperationForTest op(*b); + const auto hr = (*op).head(blob_key, Retry::once()); + ASSERT_TRUE(hr.has_value()) << "H was deleted by GC at round " << round_no << " despite being pinned by the live build B's precommit (B167 livelock would do this)"; - const auto raw = b->get(blob_key); + const auto raw = (*op).read(blob_key, Retry::once()); ASSERT_TRUE(raw.has_value()); const auto hdr = decodeEnvelopeHeader(raw->bytes, raw->bytes.size(), ObjectKind::Blob); EXPECT_EQ(raw->bytes.substr(hdr.header_len), content) @@ -1670,9 +1863,10 @@ TEST(CASPartWriteTxn, ConvergesUnderProductiveGc) const auto * entry = findEntry(manifest.entries, "f"); ASSERT_TRUE(entry != nullptr); const auto loc = s->locate(*entry); - const auto got = b->get(loc.key, Range{loc.offset, loc.length}); + OperationForTest op(*b); + const auto got = (*op).read(loc.key, Retry::once()); ASSERT_TRUE(got.has_value()); - EXPECT_EQ(got->bytes, content); + EXPECT_EQ(got->bytes.substr(static_cast(loc.offset), static_cast(loc.length)), content); } /// BUG 1 (WPromote owner==bld): promote is a PURE owner MOVE (Δ=0 — it restores no blob in-degree). The @@ -1744,8 +1938,10 @@ TEST(CASPartWriteTxn, PromoteSucceedsWhenPrecommitIsLiveOwner) TEST(CASPartWriteTxnRepoint, PromoteRepointsCommittedRef) { auto b = std::make_shared(); - /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. - std::vector events; + /// Heap-owned, not a plain local: `~Pool` emits terminate events into the sink, and a background + /// publish can hold an extra `shared_from_this()` past this frame's return regardless of + /// declaration order relative to the Pool, so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"srv1/tbl"}; @@ -1773,7 +1969,10 @@ TEST(CASPartWriteTxnRepoint, PromoteRepointsCommittedRef) /// The failed no-flag attempt threw BEFORE appendRefOps returned, so build2's precommit is still the /// live owner (no removal was appended) -- the SAME build/manifest can be retried with the flag. - s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + s->setEventSink([events](const CasEvent & e) + { + events->push(e); + }); EXPECT_NO_THROW(build2->promote(ns, "part_1", build2->buildId(), m2_id, /*allow_repoint=*/true)); auto resolved = s->resolveRef(ns, "part_1"); ASSERT_TRUE(resolved); @@ -1782,7 +1981,7 @@ TEST(CASPartWriteTxnRepoint, PromoteRepointsCommittedRef) /// Every effective repoint is loud (spec §4): exactly one RefRepoint event, naming the ref and the /// old manifest it replaced. size_t repoint_events = 0; - for (const CasEvent & e : events) + for (const CasEvent & e : events->snapshot()) if (e.type == CasEventType::RefRepoint) { ++repoint_events; @@ -1812,12 +2011,13 @@ TEST(CASPartWriteTxn, AbandonAppendsPrecommitRemovalAndKeepsLivePrecommitBody) build->putBlob(idOf("kept"), BlobSource::fromString("kept")); /// The precommit manifest body is present before abandon. - ASSERT_TRUE(b->head(manifest_key).exists); + OperationForTest op(*b); + ASSERT_TRUE((*op).head(manifest_key, Retry::once()).has_value()); build->abandon(); /// (a) the LIVE precommit body must SURVIVE abandon (left for GC after the sealed decrement). - EXPECT_TRUE(b->head(manifest_key).exists) + EXPECT_TRUE((*op).head(manifest_key, Retry::once()).has_value()) << "abandon must NOT writer-delete a live precommit body (delete-after-sealed-decrements)"; /// (b) the exact precommit binding is gone (spec §Remove Precommit: an exact owner_transition @@ -1854,9 +2054,10 @@ TEST(CASPartWriteTxn, AbandonStillDeletesNeverPrecommittedStagedDebris) build->abandon(); /// The never-precommitted debris is best-effort deleted; the live precommit body survives. - EXPECT_FALSE(b->head(s->layout().manifestKey(debris)).exists) + OperationForTest op(*b); + EXPECT_FALSE((*op).head(s->layout().manifestKey(debris), Retry::once()).has_value()) << "never-precommitted staged debris must still be best-effort deleted by abandon"; - EXPECT_TRUE(b->head(s->layout().manifestKey(precommitted)).exists) + EXPECT_TRUE((*op).head(s->layout().manifestKey(precommitted), Retry::once()).has_value()) << "the live precommit body must be spared"; } @@ -1921,17 +2122,20 @@ class RefLogConflictOnceBackend final : public InMemoryBackend String corrupt_key_substr; int corrupt_count = 0; - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + /// The fault sits on the write primitive, which every conditional write reaches the store through. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { if (corrupt_count > 0 && !corrupt_key_substr.empty() && key.find(corrupt_key_substr) != String::npos) { --corrupt_count; - /// The 3-arg qualified call bypasses virtual dispatch entirely (unlike the 2-arg - /// convenience overload, which would re-enter this very override through the vtable). - InMemoryBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT"), meta); + /// The qualified call bypasses virtual dispatch entirely, so the foreign write does not + /// re-enter this very override. + (void)InMemoryBackend::write(key, bytes + String("\x01_FOREIGN_DIFFERENT"), expected_value, access); throw Poco::TimeoutException("RefLogConflictOnceBackend: a foreign different object landed; response lost"); } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; @@ -1983,14 +2187,15 @@ TEST(CASPartWriteTxn, AbandonRetryableAfterAppendFailure) /// no longer take one. String greatest_key; size_t foreign_objects = 0; + OperationForTest op(*b); for (String cursor;;) { - const ListPage page = b->list(b->corrupt_key_substr, cursor, 1000); + const ListPage page = (*op).list(b->corrupt_key_substr, cursor, 1000, Retry::once()); for (const auto & listed : page.keys) { if (listed.key > greatest_key) greatest_key = listed.key; - const auto body = b->get(listed.key); + const auto body = (*op).read(listed.key, Retry::once()); if (body && body->bytes.find("_FOREIGN_DIFFERENT") != String::npos) ++foreign_objects; } @@ -2000,7 +2205,7 @@ TEST(CASPartWriteTxn, AbandonRetryableAfterAppendFailure) } EXPECT_EQ(foreign_objects, 1u) << "the foreign object must still own the key it took"; ASSERT_FALSE(greatest_key.empty()); - const auto greatest_body = b->get(greatest_key); + const auto greatest_body = (*op).read(greatest_key, Retry::once()); ASSERT_TRUE(greatest_body.has_value()); EXPECT_NE(greatest_body->bytes.find("_FOREIGN_DIFFERENT"), String::npos) << "the foreign occupant must still be the highest id in this table's stream: a log object above " @@ -2108,7 +2313,8 @@ TEST(CASPartWriteTxn, ManifestCapEncodedBytesJustUnderStagesSuccessfully) auto build = startBuildFor(s, ns, "wide_part"); const ManifestId id = build->stageManifest({wideBlobManifestEntry(path_len_under)}); EXPECT_EQ(id.root_namespace, ns); - EXPECT_TRUE(b->head(s->layout().manifestKey(id)).exists) + OperationForTest op(*b); + EXPECT_TRUE((*op).head(s->layout().manifestKey(id), Retry::once()).has_value()) << "a just-under-cap manifest must actually be written"; } @@ -2126,7 +2332,8 @@ TEST(CASPartWriteTxn, ManifestCapEncodedBytesOverThrowsBeforeBodyWrite) auto build = startBuildFor(s, ns, "wide_part"); - const size_t keys_before = b->list("", "", 100).keys.size(); + OperationForTest op(*b); + const size_t keys_before = (*op).list("", "", 100, Retry::once()).keys.size(); bool threw = false; try { @@ -2142,7 +2349,7 @@ TEST(CASPartWriteTxn, ManifestCapEncodedBytesOverThrowsBeforeBodyWrite) /// Fail-closed BEFORE the body write: the over-cap attempt must not have created ANY new object /// (no partial state, no orphaned blob/manifest debris for a manifest that was never accepted). - const size_t keys_after = b->list("", "", 100).keys.size(); + const size_t keys_after = (*op).list("", "", 100, Retry::once()).keys.size(); EXPECT_EQ(keys_before, keys_after) << "stageManifest must fail closed before writing the manifest body, leaving no new objects"; } @@ -2200,9 +2407,9 @@ TEST(CASPartWriteTxn, WDepSetCrossAlgoSatisfactionFailsClosed) } /// ===================================================================================== -/// Task B (chaos-tolerance-report §Task B): stageManifest's part-manifest conditional PUT rides the -/// shared CasRequestController — budgeted attempts + resolve-before-reissue — instead of the old -/// single bare attempt (which a 19s object-store pause killed while every read path survived). +/// Task B (chaos-tolerance-report §Task B): stageManifest's part-manifest conditional PUT rides +/// budgeted attempts with resolve-before-reissue, instead of the old single bare attempt (which a +/// 19s object-store pause killed while every read path survived). /// ===================================================================================== namespace @@ -2218,47 +2425,56 @@ class ManifestPutFaultBackend final : public InMemoryBackend bool land_despite_fault = false; /// the faulted attempt's own write actually lands (response lost) String plant_different_on_fault; /// a FOREIGN different body lands at the key before the fault int put_attempts = 0; /// matching body-PUT attempts observed - - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + /// Past this many attempts the double stops faulting and raises a deterministic local failure + /// instead, which every loop here surfaces unchanged. A caller whose retries are no longer bounded + /// therefore FAILS on the wrong error code rather than running until the suite times out. 0 = off. + int attempt_cap = 0; + + /// The fault sits on the write primitive: a staged part-manifest body is a create, which is a + /// `write` with no precondition. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { if (!isManifestBodyKey(key)) - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); ++put_attempts; - maybeFault(key, bytes); - return InMemoryBackend::putIfAbsent(key, bytes, meta); + if (attempt_cap > 0 && put_attempts > attempt_cap) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, + "ManifestPutFaultBackend: {} attempts exceeded the cap of {}; the caller's retries are " + "no longer bounded", put_attempts, attempt_cap); + maybeFault(key, bytes, access); + return InMemoryBackend::write(key, bytes, expected_value, access); } private: static bool isManifestBodyKey(const String & key) { return key.find("/cas/manifests/") != String::npos; } /// One fault: apply the configured server-side effect, then lose the response. - void maybeFault(const String & key, const String & bytes) + void maybeFault(const String & key, const String & bytes, TransportAccess & access) { if (fault_count <= 0) return; --fault_count; if (!plant_different_on_fault.empty()) - InMemoryBackend::putIfAbsent(key, plant_different_on_fault, {}); + (void)InMemoryBackend::write(key, plant_different_on_fault, std::nullopt, access); else if (land_despite_fault) - InMemoryBackend::putIfAbsent(key, bytes, {}); + (void)InMemoryBackend::write(key, bytes, std::nullopt, access); throw Poco::TimeoutException("ManifestPutFaultBackend: simulated ambiguous result (response lost)"); } }; } -/// The Task B core: two consecutive ambiguous timeouts on the part-manifest body PUT (each resolved -/// to "absent" by the controller's exact-GET), then a clean third attempt. The old single-attempt +/// The core ride: two consecutive ambiguous timeouts on the part-manifest body PUT (each resolved +/// to "absent" by the engine's exact read), then a clean third attempt. The old single-attempt /// path fails the whole stage on the FIRST timeout (the observed 19s-pause INSERT kill); the -/// controller path must ride its attempt budget and succeed. +/// engine's write loop must ride its policy and succeed. TEST(CASPartWriteTxnStageManifestRetry, AmbiguousTimeoutsThenCommitSucceedsWithinBudget) { - /// Zero backoff: the retry semantics are under test here, not the (controller-level-tested) - /// inter-attempt sleep schedule — keep the suite free of real sleeps. - CasRequestBudget budget; - budget.retry_initial_backoff_ms = 0; auto b = std::make_shared(); - auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + auto s = openPool(b); + useVirtualMountRequestClock(s); const RootNamespace ns{"srv/tbl"}; auto build = startBuildFor(s, ns, "part_retry"); @@ -2266,7 +2482,8 @@ TEST(CASPartWriteTxnStageManifestRetry, AmbiguousTimeoutsThenCommitSucceedsWithi const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", "a")}); EXPECT_EQ(b->put_attempts, 3) << "two faulted attempts + the committing third"; - const auto got = b->get(s->layout().manifestKey(id)); + OperationForTest op(*b); + const auto got = (*op).read(s->layout().manifestKey(id), Retry::once()); ASSERT_TRUE(got.has_value()) << "the staged manifest body must be durable"; EXPECT_EQ(decodePartManifest(openObject(FormatId::PartManifest, got->bytes)).ref, id.ref); } @@ -2278,12 +2495,17 @@ TEST(CASPartWriteTxnStageManifestRetry, AmbiguousTimeoutsThenCommitSucceedsWithi TEST(CASPartWriteTxnStageManifestRetry, AmbiguousLandedWriteResolvesToCommittedWithoutReissue) { auto b = std::make_shared(); - /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. - std::vector events; + /// Heap-owned, not a plain local: `~Pool` emits terminate events into the sink, and a background + /// publish can hold an extra `shared_from_this()` past this frame's return regardless of + /// declaration order relative to the Pool, so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto s = openPool(b); const RootNamespace ns{"srv/tbl"}; - s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + s->setEventSink([events](const CasEvent & e) + { + events->push(e); + }); auto build = startBuildFor(s, ns, "part_landed"); b->fault_count = 1; @@ -2292,13 +2514,21 @@ TEST(CASPartWriteTxnStageManifestRetry, AmbiguousLandedWriteResolvesToCommittedW EXPECT_EQ(b->put_attempts, 1) << "a landed ambiguous attempt must be resolved, never reissued"; const String key = s->layout().manifestKey(id); - ASSERT_TRUE(b->get(key).has_value()); + { + OperationForTest op(*b); + ASSERT_TRUE((*op).read(key, Retry::once()).has_value()); + } - const auto ev = std::find_if(events.begin(), events.end(), + const std::vector observed_events = events->snapshot(); + const auto ev = std::find_if(observed_events.begin(), observed_events.end(), [](const CasEvent & e) { return e.type == CasEventType::ManifestPut; }); - ASSERT_NE(ev, events.end()) << "the stage must still emit its ManifestPut audit event"; - EXPECT_EQ(ev->token, b->head(key).token.value) - << "the audit token must be the landed incarnation's token"; + ASSERT_NE(ev, observed_events.end()) << "the stage must still emit its ManifestPut audit event"; + CasRequests probe(b, Fence::open()); + CasOperation probe_op = probe.admit(); + const auto landed = probe_op.head(key, Retry::standard()); + ASSERT_TRUE(landed.has_value()); + EXPECT_EQ(ev->token, landed->etag.render()) + << "the audit token must be the landed incarnation, rendered"; } /// A DIFFERENT object at the exact staged key (a foreign body ahead of our ambiguous attempt) is a @@ -2320,26 +2550,39 @@ TEST(CASPartWriteTxnStageManifestRetry, DifferentObjectAtKeyStaysLoudConflict) EXPECT_EQ(b->put_attempts, 1) << "a proven conflict is never retried"; } -/// Budget exhaustion: EVERY attempt is ambiguous and nothing ever lands. The controller reports -/// Unresolved after `max_attempts` and stageManifest maps it to NETWORK_ERROR (fix #37 phase 2) — -/// the same retryable abort class the ref-log lane's exhausted budget maps to. Nothing was durably -/// named: the caller re-stages with a fresh ManifestId. +/// Policy exhaustion: EVERY attempt is ambiguous and nothing ever lands. The write gives up, and +/// stageManifest maps that to NETWORK_ERROR — the same retryable abort class the ref-log lane's +/// exhausted budget maps to. Nothing was durably named: the caller re-stages with a fresh ManifestId. +/// +/// Two things make the BOUND itself observable rather than assumed. The injected clock is advanced by +/// each reissue's own sleep, so the policy's window closes after a handful of attempts instead of after +/// ninety real seconds; and the double refuses deterministically past a cap far above that handful, so +/// a stage whose retries stopped being bounded fails on the wrong error code instead of running until +/// the suite times out. The exact attempt COUNT belongs to the request policy and is pinned where that +/// policy lives. TEST(CASPartWriteTxnStageManifestRetry, BudgetExhaustionMapsToNetworkError) { - CasRequestBudget budget; - budget.max_attempts = 3; - budget.retry_initial_backoff_ms = 0; /// no real sleeps; the backoff schedule has its own tests auto b = std::make_shared(); - auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + auto s = openPool(b); + const VirtualRequestClock clock = useVirtualMountRequestClock(s, /*step_ms=*/10'000); const RootNamespace ns{"srv/tbl"}; auto build = startBuildFor(s, ns, "part_exhausted"); - b->fault_count = 1000; + b->fault_count = 1000000; + b->attempt_cap = 100; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->stageManifest({blobManifestEntry("a.bin", "a")}); }); - EXPECT_EQ(b->put_attempts, 3) << "attempts must be bounded by the configured budget"; + EXPECT_GT(b->put_attempts, 1) << "the ambiguous attempt must be reissued before the write gives up"; + EXPECT_LE(b->put_attempts, b->attempt_cap) << "the policy's window, not the double's cap, ended it"; + EXPECT_GT(clock.sleeps->load(), 0u) + << "the reissues were paced on the injected clock, so this exhaustion cost no wall time"; + /// Over the namespace's whole manifest prefix rather than one computed key: the ordinal the build + /// would have used is arithmetic this assertion should not have to reproduce to stay true. + OperationForTest op(*b); + EXPECT_TRUE((*op).list(s->layout().manifestNamespacePrefix(ns), "", 10, Retry::once()).keys.empty()) + << "an exhausted stage names nothing durable"; } /// ===================================================================================== @@ -2355,36 +2598,49 @@ namespace class BlobPutFaultBackend final : public InMemoryBackend { public: + /// Unhide the legacy overload the primitive override below would otherwise hide. + using InMemoryBackend::head; int fault_count = 0; /// remaining ambiguous faults on matching create attempts bool land_despite_fault = false; /// the faulted attempt's own write actually lands (response lost) int publish_stream_attempts = 0; /// unconditional streaming publications observed int publish_copy_attempts = 0; /// unconditional native-copy publications observed int blob_head_attempts = 0; /// transaction-level blob observations - - HeadResult head(const String & key) override + std::function on_publish; /// runs before each publication, for a test that spends time in one + /// The envelope of every streaming publication the store was asked to make, in order. A test reads + /// it to prove two physical publications were two DIFFERENT bodies. + std::vector published_envelopes; + + /// Both seams sit on the transport primitives: the writer's mandatory HEAD and its publication + /// both reach the store through them. + std::optional head(const String & key, TransportAccess & access) override { if (isBlobBodyKey(key)) ++blob_head_attempts; - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { - if (std::holds_alternative(request.publication)) + if (const auto * streaming = std::get_if(&request.publication)) + { ++publish_stream_attempts; + published_envelopes.push_back(streaming->fresh_envelope); + } else ++publish_copy_attempts; + if (on_publish) + on_publish(); if (fault_count > 0) { --fault_count; if (land_despite_fault) - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); else if (const auto * streaming = std::get_if(&request.publication)) (void)streaming->open_payload(); throw Poco::TimeoutException("BlobPutFaultBackend: simulated ambiguous publication (response lost)"); } - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); } private: @@ -2395,14 +2651,16 @@ class BlobPutFaultBackend final : public InMemoryBackend }; -/// Zero-backoff store over a BlobPutFaultBackend: the sleep schedule has its own controller-level -/// tests; these Pool-level tests pin the retry/resolve/abort semantics without real sleeps. -PoolPtr openBlobFaultPool(const std::shared_ptr & b, uint32_t max_attempts = CasRequestBudget{}.max_attempts) +/// A store over a BlobPutFaultBackend. The publication loop's own bound is what these tests pin, so +/// nothing here configures a request policy: a faulted publication is one physical attempt, and the +/// loop's next iteration is what reissues it. +PoolPtr openBlobFaultPool(const std::shared_ptr & b) { - CasRequestBudget budget; - budget.max_attempts = max_attempts; - budget.retry_initial_backoff_ms = 0; - return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + PoolPtr store = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + /// The publication loop paces its iterations on the engine's clock, so drive that clock virtually: + /// eight iterations of real jittered backoff would be seconds of wall time per test. + useVirtualMountRequestClock(store); + return store; } /// A replayable BlobSource that COUNTS its own re-streams — pins INV-1's "retry = fresh re-stream @@ -2445,7 +2703,8 @@ TEST(CASPartWrite, AmbiguousTimeoutsThenCommitRestreamsFromSource) EXPECT_EQ(b->publish_stream_attempts, 3) << "two ambiguous publications + the committing third"; EXPECT_EQ(b->blob_head_attempts, 3) << "every outer retry restarts from a fresh blob HEAD"; EXPECT_EQ(payload_streams, 3) << "every reissue must RE-STREAM from the writer's own source (INV-1)"; - EXPECT_TRUE(b->head(s->layout().blobKey(idOf(payload))).exists) << "the blob body must be durable"; + OperationForTest op(*b); + EXPECT_TRUE((*op).head(s->layout().blobKey(idOf(payload)), Retry::once()).has_value()) << "the blob body must be durable"; } /// Ambiguous-but-landed: the FIRST attempt's response is lost AFTER the write actually landed @@ -2455,13 +2714,18 @@ TEST(CASPartWrite, AmbiguousTimeoutsThenCommitRestreamsFromSource) TEST(CASPartWrite, AmbiguousLandedWriteAdoptsOccupantWithoutReupload) { auto b = std::make_shared(); - /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. - std::vector events; + /// Heap-owned, not a plain local: `~Pool` emits terminate events into the sink, and a background + /// publish can hold an extra `shared_from_this()` past this frame's return regardless of + /// declaration order relative to the Pool, so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto s = openBlobFaultPool(b); const RootNamespace ns{"srv/tbl"}; const String payload = "blob-payload-B"; - s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + s->setEventSink([events](const CasEvent & e) + { + events->push(e); + }); auto build = startBuildFor(s, ns, "part_blob_landed"); const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); @@ -2478,15 +2742,148 @@ TEST(CASPartWrite, AmbiguousLandedWriteAdoptsOccupantWithoutReupload) EXPECT_EQ(payload_streams, 1); const String key = s->layout().blobKey(idOf(payload)); - const auto adopt = std::find_if(events.begin(), events.end(), + const std::vector observed_events = events->snapshot(); + const auto adopt = std::find_if(observed_events.begin(), observed_events.end(), [](const CasEvent & e) { return e.type == CasEventType::BlobReuseAdopt; }); - ASSERT_NE(adopt, events.end()) << "the landed occupant must be ADOPTED (the standard dedup leg)"; - EXPECT_EQ(adopt->token, b->head(key).token.value) << "the adopted token must be the landed incarnation's"; - EXPECT_EQ(std::count_if(events.begin(), events.end(), + ASSERT_NE(adopt, observed_events.end()) << "the landed occupant must be ADOPTED (the standard dedup leg)"; + CasRequests probe(b, Fence::open()); + CasOperation probe_op = probe.admit(); + const auto landed = probe_op.head(key, Retry::standard()); + ASSERT_TRUE(landed.has_value()); + EXPECT_EQ(adopt->token, landed->etag.render()) + << "the adopted token must be the landed incarnation, rendered"; + EXPECT_EQ(std::count_if(observed_events.begin(), observed_events.end(), [](const CasEvent & e) { return e.type == CasEventType::BlobPut; }), 0) << "no fresh-upload event: the body was never re-uploaded"; } +/// Every physical publication mints its own envelope, and the request engine never reissues one. Two +/// attempts sharing an `incarnation_tag` would send byte-identical bodies, so on a +/// content-derived-ETag dialect the republished body would read back as the incarnation GC condemned, +/// and GC's exact-incarnation delete would then remove a live body. +TEST(CASPartWrite, EveryPhysicalPublicationMintsAFreshIncarnationTag) +{ + auto b = std::make_shared(); + auto s = openBlobFaultPool(b); + const RootNamespace ns{"srv/tbl"}; + const String payload = "blob-payload-fresh-tag"; + + auto build = startBuildFor(s, ns, "part_blob_fresh_tag"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_blob_fresh_tag", id); + + int payload_streams = 0; + b->fault_count = 1; + const PutBlobResult res = build->putBlob(idOf(payload), countingSource(payload, payload_streams)); + EXPECT_EQ(res.size, payload.size()); + + ASSERT_EQ(b->published_envelopes.size(), 2u) << "one ambiguous publication and the committing second"; + EXPECT_NE(b->published_envelopes[0], b->published_envelopes[1]) + << "the second physical publication re-sent the first one's envelope"; + + const uint64_t object_size = s->poolMeta().blob_header_len + payload.size(); + const auto first = decodeEnvelopeHeader(b->published_envelopes[0], object_size, ObjectKind::Blob); + const auto second = decodeEnvelopeHeader(b->published_envelopes[1], object_size, ObjectKind::Blob); + EXPECT_TRUE(first.incarnation_tag != second.incarnation_tag) + << "a repeated incarnation_tag is a repeated incarnation: GC's condemn would name the live body"; +} + +/// The publication loop captures ONE bound before it starts, and every verb of every iteration shares +/// it -- so eight iterations cannot spend eight ninety-second windows. Here each ambiguous +/// publication burns forty seconds of the injected clock, so the third one carries the loop past the +/// window it captured and the next iteration's HEAD refuses to start. The insert is refused as +/// retry-later, which is what a caller can act on; the alternative is a single blob upload sitting on +/// the request for twenty-five minutes. The eight-attempt cap stays as the secondary bound. +/// +/// The throw and the publication count are what fail if the shared bound regresses: without it this +/// same fixture publishes a fourth time and the call SUCCEEDS. +TEST(CASPartWrite, EnsureBlobPresentIsBoundedByTheOneWindowItCaptured) +{ + auto b = std::make_shared(); + auto s = openBlobFaultPool(b); + auto now = std::make_shared>(0); + s->mountRequests().setNowFnForTest([now] { return now->load(); }); + s->mountRequests().setSleepFnForTest([now](uint64_t ms) { now->fetch_add(ms + 1); }); + b->on_publish = [now] { now->fetch_add(40'000); }; + + const RootNamespace ns{"srv/tbl"}; + const String payload = "blob-payload-long-loop"; + auto build = startBuildFor(s, ns, "part_blob_long_loop"); + const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); + build->precommitAdd(ns, "part_blob_long_loop", id); + + int payload_streams = 0; + b->fault_count = 3; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + (void)build->putBlob(idOf(payload), countingSource(payload, payload_streams)); + }); + + EXPECT_EQ(b->publish_stream_attempts, 3) << "the fourth iteration must not start a publication"; + /// The clock is past the captured deadline when the loop stops, which is what stopped it -- the + /// third publication ends at 120 s of a window that closed at 90 s. + EXPECT_GT(now->load(), 90'000u); +} + +namespace +{ + +/// Fires an injected side effect once, immediately after the watched key's body read returns -- the +/// point at which `ensureBlobPresent` has everything it needs and is about to render its verdict. +class RearmAfterMetaReadBackend final : public InMemoryBackend +{ +public: + String watched_key; + std::function trigger; + + std::optional read(const String & key, TransportAccess & access) override + { + auto observed = InMemoryBackend::read(key, access); + if (key == watched_key && trigger) + std::exchange(trigger, {})(); + return observed; + } +}; + +} + +/// A trip-and-rearm hidden inside the observation leaves the mount writable again, but NOT under the +/// generation this materialization was admitted under. The verdict points read that admission, so the +/// dependency proof is refused rather than handed back from an incarnation that has been superseded -- +/// which is what would let a fenced-out build commit a manifest naming blobs it never legally observed. +TEST(CASPartWrite, DependencyProofIsRefusedAfterARearm) +{ + auto b = std::make_shared(); + auto s = openPool(b); + const String payload = "dependency-proof-after-rearm"; + const UInt128 hash = u128Of(payload); + const BlobRef ref = idOf(payload); + + /// Pre-seed a present body and a Clean marker so the observation takes the ADOPT leg -- the leg + /// whose only durable output is the dependency proof itself. + const uint64_t header_len = s->poolMeta().blob_header_len; + String raw_body(header_len, '\0'); + raw_body += payload; + writeRawBlobBody(*b, s->layout(), hash, raw_body); + writeMetaClean(*b, s->layout(), hash, payload.size()); + + auto build = precommittedBuildForPayload(s, RootNamespace{"srv1/rearm-proof"}, "part", payload); + b->watched_key = s->layout().blobMetaKey(ref); + b->trigger = [&] + { + s->tripMountLost(); + DB::Cas::tests::rearmMountFenceAfterAnomalyForTest(s); + }; + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + build->putBlob(ref, BlobSource::fromString(payload)); + }); + EXPECT_TRUE(s->mayMutate()) + << "the mount is writable again: the refusal is about the generation the build was admitted " + "under, not about the mount being closed"; +} + /// Budget exhaustion: EVERY attempt is ambiguous and nothing ever lands. The controller reports the /// uncertainty and `ensureBlobPresent` maps it to `NETWORK_ERROR` -- the same retryable /// abort class stageManifest and the ref-log lane map their exhausted budgets to. Unlike the OLD @@ -2496,7 +2893,7 @@ TEST(CASPartWrite, AmbiguousLandedWriteAdoptsOccupantWithoutReupload) TEST(CASPartWrite, AmbiguousNonLandingPublicationStopsAtOuterBound) { auto b = std::make_shared(); - auto s = openBlobFaultPool(b, /*max_attempts=*/3); + auto s = openBlobFaultPool(b); const RootNamespace ns{"srv/tbl"}; const String payload = "blob-payload-C"; @@ -2529,17 +2926,25 @@ TEST(CASPartWrite, AmbiguousNonLandingPublicationStopsAtOuterBound) TEST(CASPartWrite, AmbiguousCopyLandedAdoptsDestinationWithoutRecopy) { auto b = std::make_shared(); - /// The sink target must outlive the Pool: `~Pool` emits terminate events into the sink. - std::vector events; + /// Heap-owned, not a plain local: `~Pool` emits terminate events into the sink, and a background + /// publish can hold an extra `shared_from_this()` past this frame's return regardless of + /// declaration order relative to the Pool, so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto s = openBlobFaultPool(b); const RootNamespace ns{"srv/tbl"}; const String payload = "staged-payload-A"; /// The staging object: [pool-fixed-length envelope header][payload], promoted VERBATIM by the copy. const String staging_key = "p/staging/test/blob-a"; const String staging_bytes = String(s->poolMeta().blob_header_len, 'h') + payload; - ASSERT_EQ(b->putIfAbsent(staging_key, staging_bytes).outcome, PutOutcome::Done); + { + OperationForTest seed_op(*b); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(staging_key, staging_bytes, Retry::once()))); + } - s->setEventSink([&](const CasEvent & e) { events.push_back(e); }); + s->setEventSink([events](const CasEvent & e) + { + events->push(e); + }); auto build = startBuildFor(s, ns, "part_copy_landed"); const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); @@ -2557,12 +2962,14 @@ TEST(CASPartWrite, AmbiguousCopyLandedAdoptsDestinationWithoutRecopy) EXPECT_EQ(b->publish_stream_attempts, 0); EXPECT_EQ(b->blob_head_attempts, 2); const String key = s->layout().blobKey(idOf(payload)); - const auto got = b->get(key); + OperationForTest op(*b); + const auto got = (*op).read(key, Retry::once()); ASSERT_TRUE(got.has_value()); EXPECT_EQ(got->bytes, staging_bytes) << "the destination is the staging object's verbatim copy"; - EXPECT_NE(std::find_if(events.begin(), events.end(), + const std::vector observed_events = events->snapshot(); + EXPECT_NE(std::find_if(observed_events.begin(), observed_events.end(), [](const CasEvent & e) { return e.type == CasEventType::BlobReuseAdopt; }), - events.end()) << "the landed destination must be ADOPTED"; + observed_events.end()) << "the landed destination must be ADOPTED"; } /// A server-side copy publication is ambiguous-and-absent: the first copy attempt times out with @@ -2576,7 +2983,10 @@ TEST(CASPartWrite, AmbiguousCopyAbsentReattemptsAndCommits) const String payload = "staged-payload-B"; const String staging_key = "p/staging/test/blob-b"; const String staging_bytes = String(s->poolMeta().blob_header_len, 'h') + payload; - ASSERT_EQ(b->putIfAbsent(staging_key, staging_bytes).outcome, PutOutcome::Done); + { + OperationForTest seed_op(*b); + ASSERT_TRUE(std::holds_alternative((*seed_op).create(staging_key, staging_bytes, Retry::once()))); + } auto build = startBuildFor(s, ns, "part_copy_retry"); const ManifestId id = build->stageManifest({blobManifestEntry("a.bin", payload)}); @@ -2597,7 +3007,8 @@ TEST(CASPartWrite, AmbiguousCopyAbsentReattemptsAndCommits) EXPECT_EQ(b->publish_stream_attempts, 1) << "the absent retry must retag and stream"; EXPECT_EQ(b->blob_head_attempts, 2); const String key = s->layout().blobKey(idOf(payload)); - const auto got = b->get(key); + OperationForTest op(*b); + const auto got = (*op).read(key, Retry::once()); ASSERT_TRUE(got.has_value()); EXPECT_NE(got->bytes, staging_bytes); EXPECT_EQ(got->bytes.substr(s->poolMeta().blob_header_len), payload); diff --git a/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp b/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp index b1022505585d..23cce62494e2 100644 --- a/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp +++ b/src/Disks/tests/gtest_cas_part_write_root_dangle.cpp @@ -62,7 +62,7 @@ ManifestEntry blobEntry(const String & name, const String & payload) /// public PartWriteTxn/Pool/Gc API (no snap injection): /// /// PartWriteTxn A uploads blob P and publishes refA -> t1 -> { data.bin: P }. A is then RELEASED (dtor), -/// retiring its build_seq so the GC watermark `min_active` advances PAST A. P now carries A's +/// retiring its build_seq so the GC watermark `min_active_build_sequence` advances PAST A. P now carries A's /// `cas_owner` and is no longer protected by any in-flight build. /// /// PartWriteTxn B starts and ADOPTS the same blob P via tokenless evidence (adoptEvidence — the cross-node @@ -83,7 +83,7 @@ TEST(CASPartWriteTxnRootDangle, SharedBlobSurvivesSourceDropDuringBuild) const String P = "shared-blob-payload-P"; /// PartWriteTxn A: upload P, publish refA -> manifest -> { data.bin: P }, then release A so its build_seq - /// retires and min_active advances past it. + /// retires and min_active_build_sequence advances past it. { PartWriteInfo info; info.intended_ref = ns.string() + "/refA"; @@ -93,7 +93,7 @@ TEST(CASPartWriteTxnRootDangle, SharedBlobSurvivesSourceDropDuringBuild) a->putBlob(idOf(P), BlobSource::fromString(P)); a->promote(ns, "refA", a->buildId(), id); } - s->renewWatermarkOnce(); /// A is gone; min_active now advances past A's build_seq + s->renewWatermarkOnce(); /// A is gone; min_active_build_sequence now advances past A's build_seq /// PartWriteTxn B: adopt the SAME blob P (cross-node adopt — tokenless evidence via adoptEvidence), assemble /// its manifest, and precommitAdd it. The precommit pins P's closure (fold +1 edge) for the build. @@ -120,7 +120,8 @@ TEST(CASPartWriteTxnRootDangle, SharedBlobSurvivesSourceDropDuringBuild) << "B171: PartWriteTxn B's promote must succeed — the precommit should have kept P alive"; /// The blob B references must still be present (no dangle), and refB must resolve. - ASSERT_TRUE(backend->head(s->layout().blobKey(idOf(P))).exists) + DB::Cas::tests::OperationForTest dangle_op(*backend); + ASSERT_TRUE((*dangle_op).head(s->layout().blobKey(idOf(P)), Retry::once()).has_value()) << "B171-dangle: GC deleted the shared blob P that PartWriteTxn B adopted — its cas_owner was the " << "retired PartWriteTxn A and the stub precommit published no build-root edge, so inDeg(P) hit 0 " << "and the single content-delete site removed it. refB now dangles."; @@ -145,7 +146,7 @@ TEST(CASPartWriteTxnRootDangle, PrematureReclaimCommitFailsClosed) const RootNamespace ns{"test/tbl"}; const String P = "shared-blob-payload-P-reclaim"; - /// PartWriteTxn A: upload P, publish refA -> manifest, retire A so min_active advances past it. + /// PartWriteTxn A: upload P, publish refA -> manifest, retire A so min_active_build_sequence advances past it. { PartWriteInfo info; info.intended_ref = ns.string() + "/refA"; @@ -173,18 +174,19 @@ TEST(CASPartWriteTxnRootDangle, PrematureReclaimCommitFailsClosed) /// RAW removal append would collide with the writer's own `RefTxnId` sequence allocation on the next /// flush; the property under test is the COMMIT gate's fail-closed behavior against a missing /// dependency, not the reclaim mechanics -- so we go straight to the reclaimed state.) + DB::Cas::tests::OperationForTest reclaim_op(*backend); { const String pkey = s->layout().blobKey(idOf(P)); - const HeadResult h = backend->head(pkey); - ASSERT_TRUE(h.exists) << "P must be present before the simulated reclaim"; - ASSERT_EQ(backend->deleteExact(pkey, h.token).kind, DeleteOutcome::Kind::Deleted); + const auto h = (*reclaim_op).head(pkey, Retry::once()); + ASSERT_TRUE(h.has_value()) << "P must be present before the simulated reclaim"; + ASSERT_EQ((*reclaim_op).remove(pkey, h->etag, Retry::once()), Removal::Removed); } /// Drop the source ref too (the state a real premature reclaim leaves: P unprotected and gone). s->dropRef(ns, "refA"); s->renewWatermarkOnce(); /// The shared blob must be GONE (the premature reclaim collected it). - ASSERT_FALSE(backend->head(s->layout().blobKey(idOf(P))).exists) + ASSERT_FALSE((*reclaim_op).head(s->layout().blobKey(idOf(P)), Retry::once()).has_value()) << "premature-reclaim setup invalid: P should have been collected after losing its precommit"; /// §4 manifest-trust (test name is legacy — B171 INV-COMMIT-FAILCLOSED for an ADOPTED leaf now moves to @@ -199,7 +201,7 @@ TEST(CASPartWriteTxnRootDangle, PrematureReclaimCommitFailsClosed) << "§4: an adopted leaf is trusted at promote — a missing dependency is not re-observed here"; /// Trust never fabricates the missing blob (it never touches P); refB IS committed (naming absent P). - ASSERT_FALSE(backend->head(s->layout().blobKey(idOf(P))).exists) + ASSERT_FALSE((*reclaim_op).head(s->layout().blobKey(idOf(P)), Retry::once()).has_value()) << "trust never fabricates the missing blob — P stays absent"; ASSERT_TRUE(s->resolveRef(ns, "refB").has_value()) << "§4: refB commits under trust (the D4 trade-off); the dangle is caught by fsck, below"; @@ -232,7 +234,7 @@ TEST(CASPartWriteTxnRoot, LivePrecommitNotReclaimed) const String Q = "live-build-blob-payload-Q"; /// PartWriteTxn B stays ALIVE: upload Q, assemble, precommitAdd — and we DO NOT retire its seq. So - /// `min_active <= build_seq` (B is in-flight) and the watermark keeps a live, advancing seq. + /// `min_active_build_sequence <= build_seq` (B is in-flight) and the watermark keeps a live, advancing seq. PartWriteInfo binfo; binfo.intended_ref = ns.string() + "/refLive"; auto b = s->beginPartWrite(binfo); @@ -240,14 +242,15 @@ TEST(CASPartWriteTxnRoot, LivePrecommitNotReclaimed) b->precommitAdd(ns, "refLive", t); b->putBlob(idOf(Q), BlobSource::fromString(Q)); s->renewWatermarkOnce(); - ASSERT_LE(s->minActive(), b->buildSeq()) << "precondition: B must be in-flight (min_active <= seq)"; + ASSERT_LE(s->minActive(), b->buildSeq()) << "precondition: B must be in-flight (min_active_build_sequence <= seq)"; /// GC to fixpoint while B is live. Gc gc(s, u128Of("gc-b8-live")); runGcToFixpoint(gc); /// Q must still be present (the live precommit's +1 edge pins it across GC). - ASSERT_TRUE(backend->head(s->layout().blobKey(idOf(Q))).exists) + DB::Cas::tests::OperationForTest live_op(*backend); + ASSERT_TRUE((*live_op).head(s->layout().blobKey(idOf(Q)), Retry::once()).has_value()) << "B8 conservatism: the live precommit must keep its blob alive across GC"; /// B can still commit (the precommit is intact). diff --git a/src/Disks/tests/gtest_cas_plain_objects.cpp b/src/Disks/tests/gtest_cas_plain_objects.cpp new file mode 100644 index 000000000000..411d8cefd800 --- /dev/null +++ b/src/Disks/tests/gtest_cas_plain_objects.cpp @@ -0,0 +1,88 @@ +#include + +#include +#include "cas_test_helpers.h" + +#include +#include + +using namespace DB::Cas; + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; + +namespace +{ + +/// Every test drives `CasRequests` on an injected clock (mirrors `gtest_cas_requests.cpp`'s +/// `makeRequests`), so a policy's whole deadline is exercised in no wall-clock time. +CasRequests makeRequests(BackendPtr backend, FakeClock & clock, Fence fence = Fence::open()) +{ + return CasRequests(std::move(backend), std::move(fence), clock.nowFn(), clock.sleepFn()); +} + +/// Refuses the FIRST removal attempt of every key with `Mismatch`, then delegates -- models a +/// concurrent replacement observed between `removeCurrent`'s internal HEAD and its DELETE. +struct MismatchOnceOnRemoveBackend : InMemoryBackend +{ + using InMemoryBackend::head; + + size_t heads = 0; + bool refuse_next_remove = true; + + std::optional head(const String & key, TransportAccess & access) override + { + ++heads; + return InMemoryBackend::head(key, access); + } + + Backend::RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override + { + if (std::exchange(refuse_next_remove, false)) + return Backend::RawRemoval::Mismatch; + return InMemoryBackend::remove(key, expected_value, access); + } +}; + +} + +TEST(CASPlainObjects, CasPutObjectIssuesHeadsOnly) +{ + FakeClock clock; + auto backend = std::make_shared(); + Layout layout("pool"); + auto requests = makeRequests(backend, clock); + CasPlainObjects objects(requests, layout); + + const RootNamespace ns("t"); + const auto life = DB::Cas::tests::fixture::fixtureLife(ns); + const String key = layout.namespaceFileKey(life, "f"); + + /// The create. + objects.putNamespaceFile(life, "f", "hello"); + EXPECT_GT(backend->headCount(key), 0u); + EXPECT_EQ(backend->getCount(key), 0u); + + /// A replace over an existing object follows the same protocol: HEAD only, never a body GET. + objects.putNamespaceFile(life, "f", "world"); + EXPECT_EQ(backend->getCount(key), 0u); + EXPECT_EQ(objects.getNamespaceFile(life, "f"), "world"); +} + +TEST(CASPlainObjects, CasRemoveObjectReheadsOnMismatch) +{ + FakeClock clock; + auto backend = std::make_shared(); + Layout layout("pool"); + auto requests = makeRequests(backend, clock); + CasPlainObjects objects(requests, layout); + + objects.putMountpointObject("f", "v"); + backend->heads = 0; + + /// The injected `Mismatch` on the first attempt must not surface as a failure: `removeCurrent` + /// re-heads the key and retries against what it now observes. + objects.removeMountpointObject("f"); + EXPECT_GE(backend->heads, 2u); + EXPECT_FALSE(objects.mountpointObjectExists("f")); +} diff --git a/src/Disks/tests/gtest_cas_pluggable_hash.cpp b/src/Disks/tests/gtest_cas_pluggable_hash.cpp index dca4081c0041..724f5e6a905c 100644 --- a/src/Disks/tests/gtest_cas_pluggable_hash.cpp +++ b/src/Disks/tests/gtest_cas_pluggable_hash.cpp @@ -58,6 +58,29 @@ using namespace DB::Cas::tests; namespace { +/// ---- Small raw-fixture request-engine wrappers shared by the tests below ---- + +/// True iff `key` exists. +bool existsAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::once()).has_value(); +} + +/// The durable object at `key`, or `nullopt`. +std::optional readAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::once()); +} + +/// Unconditional create of a fresh key (the fixture's own setup, never a real conflict). +void createAt(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + EXPECT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + /// A deterministic, non-repeating-byte payload spanning several `DBMS_DEFAULT_HASHING_BLOCK_SIZE` /// (2048 B) blocks, so a chunked-vs-one-shot divergence (the CityHash128 pitfall documented on /// `poolContentHash`) would not accidentally go unnoticed. @@ -99,7 +122,10 @@ SeededBlob seedReferencedBlob(Pool & store, Backend & backend, const RootNamespa header.kind = ObjectKind::Blob; header.incarnation_tag = UInt128(0x1234); header.build_id = UInt128(0x5678); - backend.putIfAbsent(key, encodeEnvelopeHeader(header, static_cast(store.poolMeta().blob_header_len)) + payload); + { + OperationForTest op(backend); + (*op).create(key, encodeEnvelopeHeader(header, static_cast(store.poolMeta().blob_header_len)) + payload, Retry::once()); + } ManifestEntry entry; entry.path = "data_" + std::to_string(build_sequence) + ".bin"; @@ -139,12 +165,13 @@ TEST(CASPluggableHash, CreateOrValidateRecordsConfigAlgoOnFreshPool) { auto backend = std::make_shared(); const Layout layout("p"); + OperationForTest meta_op(*backend); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, /*blob_header_len*/ 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); /// Reopening with the SAME algo is a no-op reopen: the recorded value comes back unchanged. - const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128); + const PoolMeta reopened = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::XXH3_128); EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::XXH3_128)})); EXPECT_EQ(reopened.pool_id, pm.pool_id); } @@ -153,8 +180,9 @@ TEST(CASPluggableHash, CreateOrValidateDefaultsToCityHash128) { auto backend = std::make_shared(); const Layout layout("p"); + OperationForTest meta_op(*backend); - const PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + const PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(pm.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); } @@ -166,20 +194,21 @@ TEST(CASPluggableHash, CreateOrValidateFailsClosedOnAlgoMismatchWithoutFlag) { auto backend = std::make_shared(); const Layout layout("p"); + OperationForTest meta_op(*backend); - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); expectThrowsCodeWithMessage( DB::ErrorCodes::BAD_ARGUMENTS, "1", [&] { - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::XXH3_128, /*allow_new*/ false); }); /// The pool is untouched by the refused reopen: a subsequent open with the ORIGINAL algo still /// succeeds and returns the same pool_id. - const PoolMeta reopened = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128); + const PoolMeta reopened = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128); EXPECT_EQ(reopened.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); } @@ -189,18 +218,19 @@ TEST(CASPluggableHash, AdmissionIsFlagGated) { auto backend = std::make_shared(); const Layout layout("p"); - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); /// without the flag: refuse, pool untouched expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, false); }); + { PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256, false); }); /// with the flag: admitted - const PoolMeta admitted = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, true); + const PoolMeta admitted = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256, true); EXPECT_EQ(admitted.algos_used, (std::vector{1, 3})); /// steady state: admitted algo reopens WITHOUT the flag - const PoolMeta steady = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, false); + const PoolMeta steady = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256, false); EXPECT_EQ(steady.algos_used, (std::vector{1, 3})); } @@ -208,10 +238,11 @@ TEST(CASPluggableHash, ConcurrentAdmissionUnions) { auto backend = std::make_shared(); const Layout layout("p"); - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, false, /*allow_mint*/ true); - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::XXH3_128, true); - PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::Sha256, true); - const PoolMeta final_pm = PoolMeta::createOrValidate(*backend, layout, 256, BlobHashAlgo::CityHash128, false); + OperationForTest meta_op(*backend); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, false, /*allow_mint*/ true); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::XXH3_128, true); + PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::Sha256, true); + const PoolMeta final_pm = PoolMeta::createOrValidate(*meta_op, layout, 256, BlobHashAlgo::CityHash128, false); EXPECT_EQ(final_pm.algos_used, (std::vector{1, 2, 3})); /// union, sorted, nothing lost } @@ -323,7 +354,7 @@ TEST(CASPluggableHash, Xxh3BlobLandsUnderAlgoSegmentAndIsDiscoveredCleanByFsck) const String blob_key = store->layout().blobKey(id); EXPECT_NE(blob_key.find("/blobs/xxh3/"), String::npos) << blob_key; EXPECT_EQ(blob_key.find("/blobs/ch128/"), String::npos) << blob_key; - EXPECT_TRUE(backend->head(blob_key).exists); + EXPECT_TRUE(existsAt(*backend, blob_key)); build->promote(ns, "rb", build->buildId(), mid); store->renewWatermarkOnce(); @@ -390,7 +421,7 @@ TEST(CASPluggableHash, Sha256BlobSeenByCondemnSweepAndFsckNotSilentlySkipped) const BlobRef id = seeded.ref; const String blob_key = seeded.key; EXPECT_NE(blob_key.find("/blobs/sha256/"), String::npos) << blob_key; - ASSERT_TRUE(backend->head(blob_key).exists) << "the sha256 blob body must be present before the fold"; + ASSERT_TRUE(existsAt(*backend, blob_key)) << "the sha256 blob body must be present before the fold"; ASSERT_EQ(codecFor(store->writeAlgo()).fromHex(hex), digest) << "fixture sanity: the seeded digest is ours"; /// ---- Site 1: the fold's condemn path ---- @@ -398,11 +429,11 @@ TEST(CASPluggableHash, Sha256BlobSeenByCondemnSweepAndFsckNotSilentlySkipped) dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_sha"); runRegularRoundReclaiming(gc); /// folds the -1: transition to zero => condemned - const auto state_bytes = backend->get(store->layout().gcStateKey()); + const auto state_bytes = readAt(*backend, store->layout().gcStateKey()); ASSERT_TRUE(state_bytes.has_value()); const GcState state = decodeGcState(state_bytes->bytes); ASSERT_GT(state.snap_generation, 0u); - const auto seal_bytes = backend->get(store->layout().foldSealKey(state.snap_generation, state.snap_attempt)); + const auto seal_bytes = readAt(*backend, store->layout().foldSealKey(state.snap_generation, state.snap_attempt)); ASSERT_TRUE(seal_bytes.has_value()); const CasFoldSeal seal = decodeFoldSeal(seal_bytes->bytes); ASSERT_TRUE(seal.condemned_summary.contains(0)) << "the seal's condemned_summary must be total over gc_shards"; @@ -431,23 +462,17 @@ TEST(CASPluggableHash, Sha256BlobSeenByCondemnSweepAndFsckNotSilentlySkipped) ASSERT_NE(oit, frep.objects.end()) << "the sha256 blob must appear in fsck's detailed object list"; /// The fold above already condemned it into the GC snapshot, so fsck's GC-pipeline-view /// classification (not the generic Unaccounted bucket -- reachable only by width-correctly pairing - /// the fsck-side hash against the run's kCondemned row hash) must recognize it as known-to-GC. + /// the fsck-side hash against the run's RunMarker::Condemned row hash) must recognize it as known-to-GC. EXPECT_EQ(oit->cls, FsckClass::PendingGc) - << "THE CRUX: fsck must pair the sha256 blob against the GC snapshot's kCondemned row (a " + << "THE CRUX: fsck must pair the sha256 blob against the GC snapshot's RunMarker::Condemned row (a " "silent-leak regression in CasFsck.cpp's unref_hashes/in_run_hashes/retired_by_hash port " "leaves this as the generic Unaccounted bucket instead)"; } /// ============================================================================================ -/// CAS pluggable-blob-hash Phase 2 Task 6 -- end-to-end sha256 WRITE path (in-memory; the real -/// wiring-level integration + soak is Task 7). -/// -/// Before this task, `PartWriteTxn`'s OWN write-path internals stayed a fixed 128-bit representation -/// downstream of the mint (`poolContentHash`/`PartWriteTxn::putBlob`'s `logical_hash`, the `deps` map key, the -/// event-log `object_hash` render, and `objectKey`) -- safe only because the disk-config factory guard -/// (`MetadataStorageFactory.cpp`) blocked any real sha256 pool from reaching `PartWriteTxn` at all (see the -/// Task 5 report and the "Task 6+" comments this task removes). Task 6 finishes those sites AND lifts -/// the guard in the SAME commit. This test drives a REAL `PartWriteTxn` (`putBlob` -> `stageManifest` -> +/// End-to-end SHA-256 write path. Every `PartWriteTxn` representation downstream of digest creation +/// must preserve the pool's variable-width digest; truncation would address a different blob. This +/// test drives a REAL `PartWriteTxn` (`putBlob` -> `stageManifest` -> /// `precommitAdd` -> `promote`) on a `Sha256` pool and asserts: /// 1. the blob lands under `blobs/sha256/<64-hex>` and the manifest entry's `blob_hash`, read back via /// `decodePartManifest`, is the FULL 32-byte digest (bytes beyond 16 are non-zero for a real sha256 @@ -510,17 +535,17 @@ TEST(CASPluggableHash, Sha256BuildWritesFullWidthDigestAndInlineEqualsBlob) /// THE CRUX (blob side): the blob body lands under the sha256-segmented path, addressed by the /// FULL 64-hex key -- `PartWriteTxn::putBlob`'s internal `logical_hash` must not have silently narrowed it - /// to a 32-hex (128-bit) key before this task. + /// to a 32-hex (128-bit) key. const String blob_key = store->layout().blobKey(id); EXPECT_NE(blob_key.find("/blobs/sha256/"), String::npos) << blob_key; - ASSERT_TRUE(backend->head(blob_key).exists); + ASSERT_TRUE(existsAt(*backend, blob_key)); build->promote(ns, "part1", build->buildId(), mid); store->renewWatermarkOnce(); /// Read the committed manifest back -- the on-disk `blob_hash` must be the FULL 32-byte digest, not /// truncated by the manifest codec or by anything upstream of `stageManifest`. - const auto manifest_bytes = backend->get(store->layout().manifestKey(mid)); + const auto manifest_bytes = readAt(*backend, store->layout().manifestKey(mid)); ASSERT_TRUE(manifest_bytes.has_value()); const PartManifest read_back = decodePartManifest(openObject(FormatId::PartManifest, manifest_bytes->bytes)); ASSERT_EQ(read_back.entries.size(), 2u); @@ -545,7 +570,7 @@ TEST(CASPluggableHash, Sha256BuildWritesFullWidthDigestAndInlineEqualsBlob) /// validation at `foldManifestEdges` with refresh-on-miss. /// ============================================================================================ -/// spec §9.8 -- THE race regression this task exists to close. Each `Pool`'s `admitted_algos` cache +/// Each `Pool`'s `admitted_algos` cache /// is a MONOTONE snapshot seeded once at `Pool::open` and never re-read on its own; if node A admits /// a brand-new algo and publishes a manifest naming it, node B's stale cache must NOT fail the fold /// closed forever -- `foldManifestEdges` must refresh `_pool_meta` on the very first miss and accept @@ -638,7 +663,7 @@ TEST(CASPluggableHash, ForeignAlgoSegmentIsDebrisNotOurs) /// A FOREIGN object under an algo segment `blobHashAlgoName` never renders ("md5") -- not one of /// ours under any circumstance. const String foreign_key = store->layout().blobsPrefix() + "md5/aa/" + std::string(32, 'a'); - backend->putIfAbsent(foreign_key, std::string("not a real envelope")); + createAt(*backend, foreign_key, std::string("not a real envelope")); runRegularRoundReclaiming(gc); /// folds both +1s dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_ch"); @@ -656,7 +681,7 @@ TEST(CASPluggableHash, ForeignAlgoSegmentIsDebrisNotOurs) } EXPECT_TRUE(condemned_refs.count(ch_ref)); EXPECT_TRUE(condemned_refs.count(sh_ref)); - EXPECT_TRUE(backend->head(foreign_key).exists) << "the foreign object must never be touched by the fold"; + EXPECT_TRUE(existsAt(*backend, foreign_key)) << "the foreign object must never be touched by the fold"; const FsckReport frep = runFsck(*store, /*detail=*/true); /// The physical listing counts all THREE unreferenced objects (two ours + one foreign). @@ -680,22 +705,15 @@ TEST(CASPluggableHash, ForeignAlgoSegmentIsDebrisNotOurs) } /// ============================================================================================ -/// CAS reader-generation gate (`Core/Formats/CasFormat.h`'s `G_BUILD`) was raised to 4 for -/// per-namespace contiguous ref-log ids (INV-1) and has since moved again, to 5, for Stage B's -/// namespace-life-keyed ref layer ("format bump B", `kNamespaceLifeKeyedGeneration`) -- this test's -/// assertions read `G_BUILD` itself rather than a hardcoded generation number for exactly that reason, -/// so a THIRD bump does not silently make them false. `PoolMeta::createOrValidate`'s open-time -/// CAS-raise targets `G_BUILD`, and `decodePoolMeta` fail-closes BOTH on a FUTURE -/// `min_reader_generation` AND on a BACKWARD pool whose header `compatibility_version` is below -/// `kNamespaceLifeKeyedGeneration` (which, being the LATER of the two historical breaking-change -/// floors, subsumes `kContiguousRefStreamsGeneration` -- see `CasPoolMetaFormat.cpp`). +/// CAS reader-generation gate (`CasFormat.h`'s `G_BUILD`). This test's assertions read `G_BUILD` +/// itself rather than a hardcoded generation number, so a future bump does not silently make them +/// false. `PoolMeta::createOrValidate`'s open-time CAS-raise targets `G_BUILD`, and `decodePoolMeta` +/// fail-closes BOTH on a FUTURE `min_reader_generation` AND on a BACKWARD pool whose header +/// `compatibility_version` is below the format-generation baseline (see `CasPoolMetaFormat.cpp`). /// ============================================================================================ TEST(CASPluggableHash, ReaderGenerationIsRaisedToGBuild) { - EXPECT_GE(G_BUILD, kNamespaceLifeKeyedGeneration) - << "the reader-generation gate must be at least the namespace-life-keyed floor it enforces"; - /// A freshly opened/created pool records `min_reader_generation == G_BUILD` (the open-time /// CAS-raise, `PoolMeta::createOrValidate`, always targets this build's own floor). { @@ -703,35 +721,33 @@ TEST(CASPluggableHash, ReaderGenerationIsRaisedToGBuild) auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); EXPECT_EQ(store->poolMeta().min_reader_generation, G_BUILD); - const auto meta_bytes = backend->get(store->layout().poolMetaKey()); + const auto meta_bytes = readAt(*backend, store->layout().poolMetaKey()); ASSERT_TRUE(meta_bytes.has_value()); EXPECT_EQ(decodePoolMeta(meta_bytes->bytes).min_reader_generation, G_BUILD); } /// FORWARD gate: a pool-meta carrying `min_reader_generation == G_BUILD + 1` (one generation past - /// THIS build's floor) still fails closed at open -- the startup gate (`decodePoolMeta`) rejects it - /// even though generation 4 is now understood. + /// THIS build's floor) fails closed at open -- the startup gate (`decodePoolMeta`) rejects it. { auto backend = std::make_shared(); const Layout layout("p"); - PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); pm.min_reader_generation = G_BUILD + 1; - ASSERT_TRUE(backend->casPut(layout.poolMetaKey(), encodePoolMeta(pm), backend->get(layout.poolMetaKey())->token).outcome == CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative((*meta_op).replace(layout.poolMetaKey(), encodePoolMeta(pm), (*meta_op).head(layout.poolMetaKey(), Retry::once())->etag, Retry::once()))); expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); } - /// BACKWARD floor: a pool whose header `v` (compatibility_version) is BELOW `G_BUILD` was written - /// by an older build this reader can no longer trust -- today that is one generation short of - /// `kNamespaceLifeKeyedGeneration`, a pool whose ref-object keys carry no incarnation segment, - /// which this build's parsers refuse as corruption rather than read. Craft it at the text layer: - /// take a fresh pool-meta and rewrite its line-1 version gate down to `G_BUILD - 1` (an older - /// build would have stamped exactly that). + /// BACKWARD floor: a pool whose header `v` (compatibility_version) is below the format-generation + /// baseline predates every build this reader can trust. Craft it at the text layer: take a fresh + /// pool-meta and rewrite its line-1 version gate down to `G_BUILD - 1`. { auto backend = std::make_shared(); const Layout layout("p"); - PoolMeta pm = PoolMeta::createOrValidate(*backend, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); + OperationForTest meta_op(*backend); + PoolMeta pm = PoolMeta::createOrValidate(*meta_op, layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); const String fresh_bytes = encodePoolMeta(pm); const String from = "\"v\":" + std::to_string(G_BUILD); @@ -740,7 +756,7 @@ TEST(CASPluggableHash, ReaderGenerationIsRaisedToGBuild) ASSERT_NE(pos, String::npos); // sanity: a fresh pool stamps the header at the floor String downgraded = fresh_bytes; downgraded.replace(pos, from.size(), to); - ASSERT_TRUE(backend->casPut(layout.poolMetaKey(), downgraded, backend->get(layout.poolMetaKey())->token).outcome == CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative((*meta_op).replace(layout.poolMetaKey(), downgraded, (*meta_op).head(layout.poolMetaKey(), Retry::once())->etag, Retry::once()))); /// `decodePoolMeta`'s backward floor rejects the downgraded bytes directly... expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { decodePoolMeta(downgraded); }); @@ -791,8 +807,8 @@ TEST(CASPluggableHash, TwoAlgoBlobsBothFullyReclaimed) /*build_sequence=*/2, /*payload_size=*/5002, "tbl_sh"); const String ch_key = ch.key; const String sh_key = sh.key; - ASSERT_TRUE(backend->head(ch_key).exists); - ASSERT_TRUE(backend->head(sh_key).exists); + ASSERT_TRUE(existsAt(*backend, ch_key)); + ASSERT_TRUE(existsAt(*backend, sh_key)); runRegularRoundReclaiming(gc); /// folds both +1s dropSeededRef(*store, *backend, ns, /*build_sequence=*/1, "tbl_ch"); @@ -816,8 +832,8 @@ TEST(CASPluggableHash, TwoAlgoBlobsBothFullyReclaimed) { const RoundReport rep1 = runRegularRoundReclaiming(gc); EXPECT_EQ(rep1.graduated, 2u) << "both algos' blobs must graduate together in one round"; - EXPECT_TRUE(backend->head(ch_key).exists); // pending: still present this pass - EXPECT_TRUE(backend->head(sh_key).exists); + EXPECT_TRUE(existsAt(*backend, ch_key)); // pending: still present this pass + EXPECT_TRUE(existsAt(*backend, sh_key)); } { const RoundReport rep2 = runRegularRoundReclaiming(gc); @@ -825,8 +841,8 @@ TEST(CASPluggableHash, TwoAlgoBlobsBothFullyReclaimed) } /// THE CRUX: after graduation the backend holds ZERO blob bodies of EITHER algo. - EXPECT_FALSE(backend->head(ch_key).exists) << "the ch128 blob must be physically reclaimed"; - EXPECT_FALSE(backend->head(sh_key).exists) << "the sha256 blob must be physically reclaimed"; + EXPECT_FALSE(existsAt(*backend, ch_key)) << "the ch128 blob must be physically reclaimed"; + EXPECT_FALSE(existsAt(*backend, sh_key)) << "the sha256 blob must be physically reclaimed"; const FsckReport frep = runFsck(*store, /*detail=*/true); EXPECT_TRUE(frep.clean()); @@ -895,8 +911,8 @@ TEST(CASPluggableHash, SameDigestDifferentAlgoDistinctBodiesAndSettlement) const String key_ch = store->layout().blobKey(ref_ch); const String key_xx = store->layout().blobKey(ref_xx); EXPECT_NE(key_ch, key_xx); - const auto raw_ch = backend->get(key_ch); - const auto raw_xx = backend->get(key_xx); + const auto raw_ch = readAt(*backend, key_ch); + const auto raw_xx = readAt(*backend, key_xx); ASSERT_TRUE(raw_ch.has_value()); ASSERT_TRUE(raw_xx.has_value()); EXPECT_NE(raw_ch->bytes.find(body_ch), String::npos); @@ -908,17 +924,17 @@ TEST(CASPluggableHash, SameDigestDifferentAlgoDistinctBodiesAndSettlement) const String meta_ch = store->layout().blobMetaKey(ref_ch); const String meta_xx = store->layout().blobMetaKey(ref_xx); EXPECT_NE(meta_ch, meta_xx); - EXPECT_TRUE(backend->head(meta_ch).exists); - EXPECT_TRUE(backend->head(meta_xx).exists); + EXPECT_TRUE(existsAt(*backend, meta_ch)); + EXPECT_TRUE(existsAt(*backend, meta_xx)); /// Distinct settlement (in-degree per ref, keyed on the FULL `BlobRef` pair -- never the shared /// bare digest, which would alias the two rows into one). Gc gc(store, UInt128(1)); runRegularRoundReclaiming(gc); { - const GcState st = decodeGcState(backend->get(store->layout().gcStateKey())->bytes); + const GcState st = decodeGcState(readAt(*backend, store->layout().gcStateKey())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend->get(store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + readAt(*backend, store->layout().foldSealKey(st.snap_generation, st.snap_attempt))->bytes); EXPECT_EQ(inDegreeInRuns(*backend, seal.blob_target_runs, ref_ch), 1); EXPECT_EQ(inDegreeInRuns(*backend, seal.blob_target_runs, ref_xx), 1); } @@ -930,11 +946,11 @@ TEST(CASPluggableHash, SameDigestDifferentAlgoDistinctBodiesAndSettlement) runRegularRoundReclaiming(gc); // graduates ch128:X runRegularRoundReclaiming(gc); // executes the exact-token delete for ch128:X - EXPECT_FALSE(backend->head(key_ch).exists) << "ch128:X must be reclaimed once its ref is dropped"; - EXPECT_TRUE(backend->head(key_xx).exists) + EXPECT_FALSE(existsAt(*backend, key_ch)) << "ch128:X must be reclaimed once its ref is dropped"; + EXPECT_TRUE(existsAt(*backend, key_xx)) << "THE CRUX: xxh3:X (same digest value, different algo) must remain readable after ch128:X " "is reclaimed -- a digest-only settlement would have condemned/deleted both together"; - const auto still_readable = backend->get(key_xx); + const auto still_readable = readAt(*backend, key_xx); ASSERT_TRUE(still_readable.has_value()); EXPECT_NE(still_readable->bytes.find(body_xx), String::npos); diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index b0ad1b340afd..452511f5f761 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -53,6 +54,7 @@ using namespace DB::Cas; using DB::Cas::tests::blobEntryFor; using DB::Cas::tests::expectThrowsCode; using DB::Cas::tests::idOf; +using DB::Cas::tests::SharedWaitLog; using DB::Cas::tests::u128Of; namespace @@ -64,24 +66,61 @@ class WriteCountingBackend final : public DB::Cas::Backend explicit WriteCountingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} size_t writes = 0; - std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->putIfAbsent(k, b, meta); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// Every write reaches the store through these primitives, so `writes` sees it whichever verb + /// (`create`/`replace`/`remove`/`publish`) issued it. + std::optional read(const String & key, TransportAccess & access) override { return inner->read(key, access); } + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { ++writes; - inner->publishBlob(request); + return inner->remove(key, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->putOverwrite(k, b, e, meta); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & meta) override { ++writes; return inner->casPut(k, b, e, meta); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { ++writes; return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override + { + ++writes; + inner->removeManyWriteOnce(keys, access); + } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override + { + ++writes; + return inner->write(key, bytes, expected_value, access); + } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override + { + ++writes; + inner->publish(request, access); + } + Dialect dialect() const override { return inner->dialect(); } private: std::shared_ptr inner; }; +/// A one-shot `create`, asserting it committed (mirrors the retired `backend.putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + DB::Cas::tests::OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// An exact read (mirrors the retired `backend.get(key)`). +std::optional readObj(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +/// A HEAD (mirrors the retired `backend.head(key)`). +std::optional headObj(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()); +} + /// Publish one part `ref` through the REAL PartWriteTxn write path: stage a manifest holding a single content /// blob whose payload is `payload`, precommit-add into the owning shard, then promote precommit -> /// committed. Returns the published ManifestId. This is the canonical write-side fixture for the @@ -142,7 +181,7 @@ ManifestId publishPartWithEntries( { /// Materialize the blob body so the promote-time HEAD revalidation succeeds, then record the /// tokenless W-EVIDENCE dep (the gate re-observes the current token at promote). - DB::Cas::tests::writeBlobBody(s->backend(), s->layout(), e.ref.digest.toU128()); + DB::Cas::tests::writeBlobBody(*s->poolBackendPtr(), s->layout(), e.ref.digest.toU128()); build->adoptEvidence(e); } const ManifestId id = build->stageManifest(std::move(entries)); @@ -182,20 +221,37 @@ class ProbeWatchingBackend final : public DB::Cas::Backend explicit ProbeWatchingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} bool probe_touched = false; - std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { note(k); return inner->putIfAbsent(k, b, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// Every mutation reaches the store through these primitives, so a probe-key touch is noted + /// whichever verb (`create`/`replace`/`remove`/`publish`) issued it. + std::optional read(const String & key, TransportAccess & access) override { return inner->read(key, access); } + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override + { + note(key); + return inner->remove(key, expected_value, access); + } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override + { + for (const WriteOnceKey & key : keys) + note(key.str()); + inner->removeManyWriteOnce(keys, access); + } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override + { + note(key); + return inner->write(key, bytes, expected_value, access); + } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { note(request.destination_key); - inner->publishBlob(request); + inner->publish(request, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { note(k); return inner->putOverwrite(k, b, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { note(k); return inner->casPut(k, b, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { note(k); return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + Dialect dialect() const override { return inner->dialect(); } private: void note(const String & k) { if (k.find("/_probe/") != String::npos) probe_touched = true; } std::shared_ptr inner; @@ -251,19 +307,22 @@ class ForwardingBackend : public DB::Cas::Backend public: explicit ForwardingBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} - std::optional get(const String & k, DB::Cas::Range r) override { return inner->get(k, r); } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The transport primitives forward to `inner`. Declared because `Backend` declares them pure. + std::optional read(const String & key, TransportAccess & access) override { return inner->read(key, access); } + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: std::shared_ptr inner; @@ -285,6 +344,19 @@ class ThrowingSingleAttemptBackend final : public ForwardingBackend } }; +/// A backend whose store-level preconditions refuse the pool outright — a stand-in for a versioning or +/// dialect combination `ObjectStorageBackend::checkPoolPreconditions` rejects. +class ThrowingPoolPreconditionsBackend final : public ForwardingBackend +{ +public: + using ForwardingBackend::ForwardingBackend; + + void checkPoolPreconditions() override + { + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "test: pool preconditions refused"); + } +}; + /// A backend that forbids skipping the access-check battery — a stand-in for the writable /// generation-dialect (GCS) backend (see ObjectStorageBackend::checkSkipAccessCheckSupport). class ThrowingSkipAccessCheckBackend final : public ForwardingBackend @@ -359,6 +431,71 @@ TEST(CASPool, BackendForbiddingSkipAccessCheckStillOpensWhenTheBatteryRuns) ASSERT_NE(store, nullptr); } +/// The ORDINARY writable mount -- the one that runs the battery -- must still be refused by the two +/// store-level gates. They used to be the capability probe's own first two steps; they are the caller's +/// now, and nothing else in the open path would notice if the caller stopped asking. The write counter is +/// what makes each of these a fence rather than a bare `EXPECT_THROW`: `Pool::open` refuses for many +/// reasons, but only a refusal BEFORE the battery leaves the store unwritten. +TEST(CASPool, WritableOpenRunsThePoolPreconditionGateBeforeTheBattery) +{ + auto counting = std::make_shared(std::make_shared()); + auto backend = std::make_shared(counting); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + ASSERT_FALSE(cfg.skip_access_check) << "this test is about the branch that RUNS the battery"; + + try + { + DB::Cas::Pool::open(backend, cfg); + FAIL() << "expected the pool-precondition gate to refuse the mount"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + EXPECT_NE(e.message().find("pool preconditions refused"), std::string::npos) + << "actual message: " << e.message(); + } + EXPECT_EQ(counting->writes, 0u) << "the gate must refuse before the battery writes anything"; +} + +TEST(CASPool, WritableOpenRunsTheSingleAttemptGateBeforeTheBattery) +{ + auto counting = std::make_shared(std::make_shared()); + auto backend = std::make_shared(counting); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + ASSERT_FALSE(cfg.skip_access_check) << "this test is about the branch that RUNS the battery"; + + try + { + DB::Cas::Pool::open(backend, cfg); + FAIL() << "expected the single-attempt gate to refuse the mount"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + EXPECT_NE(e.message().find("no single-attempt client"), std::string::npos) + << "actual message: " << e.message(); + } + EXPECT_EQ(counting->writes, 0u) << "the gate must refuse before the battery writes anything"; +} + +/// The positive control for the two above: with no gate refusing, the same open DOES write. Without it +/// `writes == 0` would be satisfied by an open that refused for any earlier reason, and both fences would +/// pass while the gates were gone. +TEST(CASPool, WritableOpenWithoutAGateRefusalDoesReachTheBattery) +{ + auto counting = std::make_shared(std::make_shared()); + + DB::Cas::PoolConfig cfg = writablePoolConfigForTest(); + cfg.background_watermark = false; + ASSERT_FALSE(cfg.skip_access_check); + + auto store = DB::Cas::Pool::open(counting, cfg); + ASSERT_NE(store, nullptr); + EXPECT_GT(counting->writes, 0u); +} + TEST(CASPool, MinActiveTracksInFlightBuilds) { auto backend = std::make_shared(); @@ -440,10 +577,10 @@ TEST(CASPoolMeta, CreateThenReopen) { auto b = std::make_shared(); Layout layout("p"); - PoolMeta created = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 256, + PoolMeta created = PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, /*blob_header_len*/ 256, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_NE(created.pool_id, UInt128{}); - PoolMeta reopened = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512); + PoolMeta reopened = PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, /*blob_header_len*/ 512); EXPECT_EQ(reopened.pool_id, created.pool_id); /// pool is authoritative — config ignored on reopen EXPECT_EQ(reopened.blob_header_len, 256u); } @@ -455,9 +592,9 @@ TEST(CASPoolMeta, FailClosed) /// (createOrValidate path). The future-version fail-closed (v > G_BUILD => UNKNOWN_FORMAT_VERSION) /// is exercised at the codec level by the battery's per-row v+1 gate. auto b2 = std::make_shared(); - b2->putIfAbsent(layout.poolMetaKey(), "garbage"); + createObj(*b2, layout.poolMetaKey(), "garbage"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { PoolMeta::createOrValidate(*b2, layout, 256); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b2), layout, 256); }); } TEST(CASPoolMeta, RoundTripAndReadability) @@ -487,20 +624,20 @@ TEST(CASPoolMeta, RejectsBadConstantsAtCreation) /// not 8-aligned (above the floor, so it is the alignment rule that rejects it) expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, - [&] { PoolMeta::createOrValidate(*b, layout, 250); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, 250); }); /// below the v3 envelope floor (240) but 8-aligned: rejected by the floor, not the alignment rule. /// Without the raised floor this pool would pass creation and LOGICAL_ERROR on the first blob write. expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, - [&] { PoolMeta::createOrValidate(*b, layout, 128); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, 128); }); /// well below the floor expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, - [&] { PoolMeta::createOrValidate(*b, layout, 64); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, 64); }); /// above the 16 KiB ceiling expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, - [&] { PoolMeta::createOrValidate(*b, layout, 17 * 1024); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, 17 * 1024); }); /// A creation that fails config validation must not have written anything. - EXPECT_FALSE(b->get(layout.poolMetaKey()).has_value()); + EXPECT_FALSE(readObj(*b, layout.poolMetaKey()).has_value()); } TEST(CASPoolMeta, RejectsBadConstantsOnDecode) @@ -511,9 +648,10 @@ TEST(CASPoolMeta, RejectsBadConstantsOnDecode) PoolMeta bad_pm; bad_pm.pool_id = hexToU128("00000000000000000000000000000001"); bad_pm.blob_header_len = 100; /// violates 8-alignment invariant - b->putIfAbsent(layout.poolMetaKey(), encodePoolMeta(bad_pm)); + bad_pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; + createObj(*b, layout.poolMetaKey(), encodePoolMeta(bad_pm)); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { PoolMeta::createOrValidate(*b, layout, 256); }); + [&] { PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, 256); }); } TEST(CASPoolMeta, DecodeGarbageFails) @@ -536,9 +674,9 @@ TEST(CASPoolMeta, ConcurrentCreateRace) foreign_pm.pool_id = foreign; foreign_pm.blob_header_len = 256; foreign_pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - b->putIfAbsent(layout.poolMetaKey(), encodePoolMeta(foreign_pm)); + createObj(*b, layout.poolMetaKey(), encodePoolMeta(foreign_pm)); - PoolMeta result = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512); + PoolMeta result = PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, /*blob_header_len*/ 512); EXPECT_EQ(result.pool_id, foreign); EXPECT_EQ(result.blob_header_len, 256u); /// the foreign pool's constants win } @@ -546,27 +684,28 @@ TEST(CASPoolMeta, ConcurrentCreateRace) TEST(CASPoolMeta, CasConflictReReadsWinner) { /// The subtlest branch: the initial GET sees ABSENT, so createOrValidate proceeds to the - /// create-if-absent casPut — and loses, because a racing creator committed in between. The loser + /// create-if-absent write — and loses, because a racing creator committed in between. The loser /// must then re-read and return the WINNER's pool identity, not LOGICAL_ERROR. A single-threaded - /// `failNextCasPut` alone cannot exercise this: it returns Conflict without leaving the object + /// `refuseNextWrite` alone cannot exercise this: it returns Conflict without leaving the object /// readable, so the re-read would fire the LOGICAL_ERROR guard. We model the real interleaving - /// with a backend whose casPut commits the winner's object (via the public putIfAbsent) and THEN - /// reports Conflict — exactly what the loser observes. + /// with a backend whose write primitive commits the winner's object and THEN reports Conflict -- + /// exactly what the loser observes. class RacingBackend : public InMemoryBackend { public: String winner_bytes; - CasResult casPut(const String & key, const String & bytes, - const std::optional & expected, const ObjectMeta & meta) override + /// The fault sits on the WRITE PRIMITIVE: the create-if-absent this models is issued there. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - if (!winner_committed) + if (!winner_committed && !expected_value) { winner_committed = true; /// The winner lands first; our create-if-absent now necessarily conflicts. - putIfAbsent(key, winner_bytes); - return {CasOutcome::Conflict, {}}; + (void)InMemoryBackend::write(key, winner_bytes, std::nullopt, access); + return std::unexpected(RawConflict{}); } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } private: bool winner_committed = false; @@ -583,7 +722,7 @@ TEST(CASPoolMeta, CasConflictReReadsWinner) Layout layout("p"); /// Our config (512) is what we WOULD have minted, but we lose the race and inherit the winner. - PoolMeta result = PoolMeta::createOrValidate(*b, layout, /*blob_header_len*/ 512, + PoolMeta result = PoolMeta::createOrValidate(*DB::Cas::tests::OperationForTest(b), layout, /*blob_header_len*/ 512, BlobHashAlgo::CityHash128, /*allow_new*/ false, /*allow_mint*/ true); EXPECT_EQ(result.pool_id, winner); EXPECT_EQ(result.blob_header_len, 256u); @@ -688,9 +827,10 @@ TEST(CASPool, ResolveReturnsManifestId) EXPECT_EQ(loc.offset, s->poolMeta().blob_header_len); EXPECT_EQ(loc.length, payload.size()); - auto bytes = b->get(loc.key, Range{loc.offset, loc.length}); + auto bytes = readObj(*b, loc.key); ASSERT_TRUE(bytes.has_value()); - EXPECT_EQ(bytes->bytes, payload); /// ranged read, no header touch + /// The located window holds exactly the payload: the envelope header is outside it. + EXPECT_EQ(bytes->bytes.substr(static_cast(loc.offset), static_cast(loc.length)), payload); const auto * small = findEntry(manifest.entries, "small.txt"); ASSERT_TRUE(small != nullptr); @@ -721,7 +861,7 @@ TEST(CASPool, ReadManifestValidatesBodyAndFailsClosed) body.root_namespace_id = RootNamespace{"srv1/other"}; /// namespace does NOT body.entries = {blobEntryFor("f", u128Of("x"), 1)}; body.payload_digest = computePayloadDigest(body); - b->putIfAbsent(layout.manifestKey(addressed), encodePartManifest(body)); + createObj(*b, layout.manifestKey(addressed), encodePartManifest(body)); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(addressed); }); } @@ -737,7 +877,7 @@ TEST(CASPool, ReadManifestValidatesBodyAndFailsClosed) body.root_namespace_id = ns; /// namespace matches body.entries = {blobEntryFor("f", u128Of("y"), 1)}; body.payload_digest = computePayloadDigest(body); - b->putIfAbsent(layout.manifestKey(addressed), encodePartManifest(body)); + createObj(*b, layout.manifestKey(addressed), encodePartManifest(body)); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(addressed); }); } @@ -806,11 +946,11 @@ TEST(CASPool, LookupAndListOverManifestEntries) EXPECT_EQ(all[3].path, "p.proj/data.bin"); } -/// The Phase 1c manifest decode cache is keyed by (ManifestId, Token). Resolve+read the same ref twice: -/// the second readManifest must be served from the cache (no second GET of the body). A fresh publish -/// under a DIFFERENT ref name mints a NEW ManifestId (and a new shard token), so the cache misses and -/// the body is fetched again. A CountingBackend asserts the body GET count. -TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) +/// The manifest decode cache is keyed by ManifestId alone: an id is minted once and its body is +/// written once, so one id names one content forever. Resolve+read the same ref twice: the second +/// readManifest is served from the cache with NO request at all. A fresh publish under a DIFFERENT +/// ref name mints a NEW ManifestId, so the cache misses and the body is fetched once. +TEST(CASPool, ManifestCacheIsKeyedById) { auto b = std::make_shared(); auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); @@ -819,8 +959,9 @@ TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) const ManifestId id1 = publishPart(s, ns.string(), "part_1", "payload-1"); const String key1 = layout.manifestKey(id1); + b->resetCounts(); - /// First read: a body GET populates the (id1, token) cache entry. + /// First read: a body GET populates the id1 cache entry. { auto r = s->resolveRef(ns, "part_1"); ASSERT_TRUE(r.has_value()); @@ -830,7 +971,7 @@ TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) const uint64_t gets_after_first = b->getCount(key1); ASSERT_GE(gets_after_first, 1u); /// the first read DID fetch the body - /// Second read of the SAME id: the (id, token) cache must serve it — NO additional body GET. + /// Second read of the SAME id: the id-keyed cache must serve it — NO additional body GET. { auto r = s->resolveRef(ns, "part_1"); ASSERT_TRUE(r.has_value()); @@ -839,7 +980,8 @@ TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) ASSERT_EQ(m.entries.size(), 1u); } EXPECT_EQ(b->getCount(key1), gets_after_first) - << "second readManifest re-GET the body for the same (ManifestId, Token) — cache miss"; + << "second readManifest re-GET the body for the same ManifestId — cache miss"; + EXPECT_EQ(b->headCount(key1), 0u) << "keyed by id alone: no HEAD on a miss or a hit"; /// A fresh publish under a DIFFERENT ref name mints a NEW ManifestId: the cache (keyed by id) misses. /// (Promoting a different manifest over the SAME committed ref is a distinct promote-over-committed @@ -855,6 +997,7 @@ TEST(CASPool, ManifestCacheIsKeyedByIdAndToken) ASSERT_EQ(m2.entries.size(), 1u); EXPECT_GE(b->getCount(key2), 1u) /// the new id's body WAS fetched (cache miss) << "fresh publish (new ManifestId) should miss the id-keyed manifest cache"; + EXPECT_EQ(b->headCount(key2), 0u); } /// Phase 5 (part-folder cache spec): manifest_cache is now a byte-weighted CacheBase LRU instead of a @@ -1104,7 +1247,7 @@ TEST(CASPool, ListRefsSkipsForeignKeys) /// A stray key directly under the namespace's ref-object prefix that is not `_log`/ /// `_snap` shaped (also covers the legacy shard-number layout GC/dropNamespace still write). - b->putIfAbsent(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "garbage", "not-a-ref-object"); + createObj(*b, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "garbage", "not-a-ref-object"); std::map refs; EXPECT_NO_THROW(refs = s->listRefs(ns)); @@ -1125,7 +1268,7 @@ TEST(CASPool, ReadManifestFailsClosed) { const ManifestRef ref = manifestRefFor("garbage-body"); const ManifestId id{.root_namespace = ns, .ref = ref}; - b->putIfAbsent(layout.manifestKey(id), "not a valid manifest body"); + createObj(*b, layout.manifestKey(id), "not a valid manifest body"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { s->readManifest(id); }); } @@ -1265,8 +1408,7 @@ TEST(CASPool, ListNamespacesDoesNotMintLogicalNamesFromFileKeys) s->putNamespaceFile(DB::Cas::tests::fixture::fixtureLife(ns), "format_version.txt", "1\n"); /// A second life of the SAME name, written by exact key because no helper mints two lives yet. const NamespaceLifeId other = NamespaceLifeId::fromCatalogEntry(ns, DB::UInt128(0x5eed)); - ASSERT_EQ(b->putIfAbsent(s->layout().namespaceFileKey(other, "format_version.txt"), "1\n").outcome, - PutOutcome::Done); + createObj(*b, s->layout().namespaceFileKey(other, "format_version.txt"), "1\n"); const NamespaceListing listing = s->listNamespaces(""); EXPECT_TRUE(listing.skipped.empty()); @@ -1290,8 +1432,8 @@ TEST(CASPool, ListNamespacesDoesNotTreatPhysicalDebrisAsCatalogAuthority) const String lifeless_ref = s->layout().casRefsPrefix() + ns.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; const String lifeless_file = s->layout().rootsPrefix() + ns.string() + "/_files/format_version.txt"; - ASSERT_EQ(b->putIfAbsent(lifeless_ref, "garbage").outcome, PutOutcome::Done); - ASSERT_EQ(b->putIfAbsent(lifeless_file, "garbage").outcome, PutOutcome::Done); + createObj(*b, lifeless_ref, "garbage"); + createObj(*b, lifeless_file, "garbage"); NamespaceListing listing; ASSERT_NO_THROW(listing = s->listNamespaces("")) @@ -1303,8 +1445,8 @@ TEST(CASPool, ListNamespacesDoesNotTreatPhysicalDebrisAsCatalogAuthority) EXPECT_EQ(listing.namespaces[0], ns.string()); EXPECT_TRUE(listing.skipped.empty()); - EXPECT_TRUE(b->head(lifeless_ref).exists); - EXPECT_TRUE(b->head(lifeless_file).exists); + EXPECT_TRUE(headObj(*b, lifeless_ref).has_value()); + EXPECT_TRUE(headObj(*b, lifeless_file).has_value()); } TEST(CASPool, ListMirroredChildren) @@ -1327,7 +1469,7 @@ namespace /// Delegating backend that fences the mount slot IN PLACE the first time a `get` returns a present /// body for the armed key — reproducing the S13 window: the GC's token-guarded fence-out lands -/// between the keeper adopt's GET and its CAS. The caller's subsequent token-guarded `putOverwrite` +/// between the renewer adopt's GET and its CAS. The caller's subsequent token-guarded `putOverwrite` /// then fails `PreconditionFailed`, the adopt re-reads, sees `gc_fenced`, and throws /// `MountFencedException` — which `Pool::open`'s fence-recovery loop must turn into a fresh-epoch /// retry rather than a permanent wedge (P3.1 vector C). @@ -1337,34 +1479,37 @@ class FenceInAdoptWindowBackend final : public DB::Cas::Backend explicit FenceInAdoptWindowBackend(std::shared_ptr inner_) : inner(std::move(inner_)) {} String fence_key; /// empty = fault disarmed; set to the mount key to arm the one-shot fence - std::optional get(const String & k, DB::Cas::Range r) override + bool supportsListTokens() const override { return inner->supportsListTokens(); } + + /// The fault sits on the READ PRIMITIVE: the renewer's adopt reads the mount slot through it. + std::optional read(const String & key, TransportAccess & access) override { - auto got = inner->get(k, r); - if (!fence_key.empty() && k == fence_key && got.has_value()) + auto got = inner->read(key, access); + if (!fence_key.empty() && key == fence_key && got.has_value()) { /// One-shot: fence the slot in place exactly as `computeHeartbeatFloor` does (preserve the - /// body, gc_fenced = true, seq + 1, token-guarded against the value we just read), then + /// body, gc_fenced = true, seq + 1, guarded against the incarnation we just read), then /// disarm so the retry can adopt cleanly. DB::Cas::MountLease fenced = DB::Cas::decodeMountLease(got->bytes); fenced.gc_fenced = true; fenced.seq += 1; - inner->putOverwrite(k, DB::Cas::encodeMountLease(fenced), got->token); + (void)inner->write(key, DB::Cas::encodeMountLease(fenced), got->value, access); fence_key.clear(); } return got; } - std::optional getStream(const String & k, DB::Cas::Range r) override { return inner->getStream(k, r); } - DB::Cas::HeadResult head(const String & k) override { return inner->head(k); } - DB::Cas::ListPage list(const String & p, const String & c, size_t l) override { return inner->list(p, c, l); } - DB::Cas::PutResult putIfAbsent(const String & k, const String & b, const DB::Cas::ObjectMeta & m) override { return inner->putIfAbsent(k, b, m); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + std::optional head(const String & key, TransportAccess & access) override { return inner->head(key, access); } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { return inner->list(prefix, cursor, limit, access); } + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { return inner->remove(key, expected_value, access); } + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override { inner->removeManyWriteOnce(keys, access); } + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - inner->publishBlob(request); + return inner->write(key, bytes, expected_value, access); } - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, const DB::Cas::ObjectMeta & m) override { return inner->putOverwrite(k, b, e, m); } - DB::Cas::CasResult casPut(const String & k, const String & b, const std::optional & e, const DB::Cas::ObjectMeta & m) override { return inner->casPut(k, b, e, m); } - DB::Cas::DeleteOutcome deleteExact(const String & k, const DB::Cas::Token & t) override { return inner->deleteExact(k, t); } - bool supportsListTokens() const override { return inner->supportsListTokens(); } + std::unique_ptr stream(const String & key, TransportAccess & access) override { return inner->stream(key, access); } + void publish(const BlobPublishRequest & request, TransportAccess & access) override { inner->publish(request, access); } + Dialect dialect() const override { return inner->dialect(); } private: std::shared_ptr inner; @@ -1377,7 +1522,7 @@ TEST(CASPoolMountFence, OpenRecoversFromFenceInAdoptWindowWithFreshEpoch) auto inner = std::make_shared(); auto fencing = std::make_shared(inner); /// Arm the one-shot fence on the mount slot. Pool::open first claims the mount (fresh mint), then - /// the keeper adopts it — the adopt's GET trips the fence, its CAS fails, and open must recover. + /// the renewer adopts it — the adopt's GET trips the fence, its CAS fails, and open must recover. const DB::Cas::Layout layout("p"); fencing->fence_key = layout.mountKey("test"); @@ -1385,19 +1530,28 @@ TEST(CASPoolMountFence, OpenRecoversFromFenceInAdoptWindowWithFreshEpoch) /// -> `MountPriorState::Fenced` (a fenced prior is reclaimed on the first attempt, with no /// observation polling -- see `CASMountOpenWaits.FencedPriorReclaimsWithoutAnyWait`). The injected /// `boot_ms_fn`/`wait_sleep_fn` below keep this test off the real clock regardless. - uint64_t fake_boot = 0; + /// Held in a shared atomic, not a plain local: `wait_sleep_fn` below mutates it, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto fake_boot = std::make_shared>(0); DB::Cas::PoolPtr store; ASSERT_NO_THROW( store = DB::Cas::Pool::open(fencing, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", - .boot_ms_fn = [&fake_boot] { return fake_boot; }, - .wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }})) + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }})) << "open must recover from a fence in the adopt window, not wedge (exit-49 S13 bug)"; ASSERT_TRUE(store); /// The final live lease is unfenced and at a HIGHER writer_epoch than the first attempt (a fence /// costs an epoch): the first claim took epoch 1, got fenced, the retry took epoch 2 and mounted. - const auto got = inner->get(layout.mountKey("test")); + const auto got = readObj(*inner, layout.mountKey("test")); ASSERT_TRUE(got.has_value()); const MountLease final_lease = decodeMountLease(got->bytes); EXPECT_FALSE(final_lease.gc_fenced); @@ -1413,25 +1567,31 @@ TEST(CASPoolMountFence, OpenRecoversFromFenceInAdoptWindowWithFreshEpoch) TEST(CASPool, WriteFenceUsesInjectedBootClock) { auto backend = std::make_shared(); - uint64_t fake_boot = 1'000'000; /// arbitrary boottime origin (ms) + /// Held in a shared atomic, not a plain local: this test mutates the clock below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); /// arbitrary boottime origin (ms) auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30000), - .boot_ms_fn = [&] { return fake_boot; }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, }); /// Freshly armed at open (deadline = fake_boot + ttl): well within the ttl, mutations are allowed. EXPECT_TRUE(store->mayMutate()); /// Advance the boot clock just short of the deadline — still armed. - fake_boot += 29999; + *fake_boot += 29999; EXPECT_TRUE(store->mayMutate()); /// Cross the deadline (ttl elapsed with no renew — a resumed sleeper's view). The fence must expire. /// (The "a gated mutate then fails closed with ABORTED" leg used `mutateShardForTest` -- the held /// Phase-E shard lane -- and moves there; here we pin the boot-clock fence flip itself.) - fake_boot += 2; /// now fake_boot = origin + 30001 > origin + 30000 + *fake_boot += 2; /// now fake_boot = origin + 30001 > origin + 30000 EXPECT_FALSE(store->mayMutate()); } @@ -1443,13 +1603,14 @@ namespace /// GC's fence-out, applied directly: preserve the body, set gc_fenced, bump seq (token-guarded). void fenceOutMount(DB::Cas::Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease m = decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, - DB::Cas::PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(mount_key, encodeMountLease(m), got->etag, Retry::standard()))); } } @@ -1459,22 +1620,22 @@ TEST(CASPoolRemount, FenceOutThenSelfRemountRestoresWrites) auto backend = std::make_shared(); auto store = DB::Cas::tests::openPoolForTest(backend); const String mount_key = store->layout().mountKey("test"); - const uint64_t epoch_before = decodeMountLease(backend->get(mount_key)->bytes).writer_epoch; + const uint64_t epoch_before = decodeMountLease(readObj(*backend, mount_key)->bytes).writer_epoch; EXPECT_EQ(store->liveWriterEpoch(), epoch_before); fenceOutMount(*backend, mount_key); - /// The keeper's next renewal fails closed (foreign touch — never re-mint). + /// The renewer's next renewal fails closed (foreign touch — never re-mint). EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); /// Self-remount claims a FRESH incarnation: epoch bumped, gc_fenced cleared, writes restored. ASSERT_TRUE(store->tryRemountOnce()); - const MountLease after = decodeMountLease(backend->get(mount_key)->bytes); + const MountLease after = decodeMountLease(readObj(*backend, mount_key)->bytes); EXPECT_EQ(after.writer_epoch, epoch_before + 1); EXPECT_FALSE(after.gc_fenced); EXPECT_EQ(store->liveWriterEpoch(), epoch_before + 1); - /// The renewal path works again (the new keeper owns the slot). (The follow-on "...and so does a + /// The renewal path works again (the new renewer owns the slot). (The follow-on "...and so does a /// ref-shard mutation" check used `mutateShardForTest` -- the held Phase-E shard lane -- and moves /// to Phase E's own tests; the self-remount liveness assertion above is the point of this test.) EXPECT_NO_THROW(store->renewWatermarkOnce()); @@ -1511,19 +1672,20 @@ TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) const String mount_key = store->layout().mountKey("test"); /// A genuinely foreign uuid holds the mount (live or not — foreign is terminal for the claim). - const auto got = backend->get(mount_key); + DB::Cas::tests::OperationForTest overwrite_op(*backend); + const auto got = (*overwrite_op).read(mount_key, Retry::standard()); MountLease foreign = decodeMountLease(got->bytes); foreign.server_uuid = foreign.server_uuid + DB::UInt128(1); foreign.seq += 1; - ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, - DB::Cas::PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*overwrite_op).replace(mount_key, encodeMountLease(foreign), got->etag, Retry::standard()))); EXPECT_FALSE(store->tryRemountOnce()); /// The foreign body is untouched (no takeover, ever). - EXPECT_EQ(decodeMountLease(backend->get(mount_key)->bytes).server_uuid, foreign.server_uuid); + EXPECT_EQ(decodeMountLease(readObj(*backend, mount_key)->bytes).server_uuid, foreign.server_uuid); /// Move the parent fixture to the production-recognized fenced terminal state before explicitly - /// destroying its superseded keeper. The unfenced foreign-release guard is covered separately below. + /// destroying its superseded renewer. The unfenced foreign-release guard is covered separately below. fenceOutMount(*backend, mount_key); store.reset(); @@ -1535,15 +1697,15 @@ TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) auto foreign_backend = std::make_shared(); auto invalid_store = DB::Cas::tests::openPoolForTest(foreign_backend); const String foreign_mount_key = invalid_store->layout().mountKey("test"); - const auto foreign_got = foreign_backend->get(foreign_mount_key); + DB::Cas::tests::OperationForTest foreign_overwrite_op(*foreign_backend); + const auto foreign_got = (*foreign_overwrite_op).read(foreign_mount_key, Retry::standard()); ASSERT_TRUE(foreign_got.has_value()); MountLease foreign_lease = decodeMountLease(foreign_got->bytes); foreign_lease.server_uuid = foreign_lease.server_uuid + DB::UInt128(1); foreign_lease.seq += 1; - ASSERT_EQ( - foreign_backend->putOverwrite(foreign_mount_key, encodeMountLease(foreign_lease), foreign_got->token).outcome, - DB::Cas::PutOutcome::Done); - const auto occupant_before = foreign_backend->get(foreign_mount_key); + ASSERT_TRUE(std::holds_alternative((*foreign_overwrite_op).replace( + foreign_mount_key, encodeMountLease(foreign_lease), foreign_got->etag, Retry::standard()))); + const auto occupant_before = readObj(*foreign_backend, foreign_mount_key); ASSERT_TRUE(occupant_before.has_value()); EXPECT_FALSE(invalid_store->tryRemountOnce()) << "a foreign owner is never taken over at remount"; @@ -1555,7 +1717,7 @@ TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), violations_before + 1) << "the release must report the broken single-writer guarantee rather than dying on it"; - const auto occupant_after = foreign_backend->get(foreign_mount_key); + const auto occupant_after = readObj(*foreign_backend, foreign_mount_key); ASSERT_TRUE(occupant_after.has_value()) << "nor is it taken over at release"; EXPECT_EQ(occupant_after->bytes, occupant_before->bytes) << "the slot must be left byte-for-byte as the foreign owner wrote it"; @@ -1601,18 +1763,18 @@ struct SequencedBootClock } /// Phase B addendum 2 (task 5b review, reviewer's probe): the self-remount arm must anchor at the -/// claim attempt's pre-I/O instant (`remount_anchor_boot_ms`, captured right after `installKeeper` -/// and right before `keeperStart()` in `Pool::tryRemountOnce`), never at a later reading taken after -/// `keeperStart`/`quiesceRefTablesForRemount` have already run. +/// claim attempt's pre-I/O instant (`remount_anchor_boot_ms`, captured right after `installRenewer` +/// and right before `renewerStart()` in `Pool::tryRemountOnce`), never at a later reading taken after +/// `renewerStart`/`quiesceRefTablesForRemount` have already run. /// /// The two `bootMsNow()` calls of interest, in the ORDER each code version issues them: -/// - FIXED code: call #1 = the new anchor (`remount_anchor_boot_ms`, before `keeperStart`); -/// call #2 = `MountLeaseKeeper::prepareRenew`'s own internal boot read inside `keeperStart`'s -/// `doStart` (feeds only the keeper's OWN internal `confirmed_deadline_ms` -- unrelated to the +/// - FIXED code: call #1 = the new anchor (`remount_anchor_boot_ms`, before `renewerStart`); +/// call #2 = `MountLeaseRenewer::prepareRenew`'s own internal boot read inside `renewerStart`'s +/// `doStart` (feeds only the renewer's OWN internal `confirmed_deadline_ms` -- unrelated to the /// Pool-level arm -- so its value is irrelevant to the arm post-fix). /// - PRE-FIX code (no anchor line): call #1 = that SAME `prepareRenew` read (now the first boot -/// call of the attempt, since nothing reads the clock before `keeperStart`); call #2 = the -/// arm-site's own `mount_runtime.bootMsNow()`, read AFTER `keeperStart` returns -- the stale, +/// call of the attempt, since nothing reads the clock before `renewerStart`); call #2 = the +/// arm-site's own `mount_runtime.bootMsNow()`, read AFTER `renewerStart` returns -- the stale, /// response-time reading this whole fix exists to stop using. /// A sequenced clock returning 10000 then 11000 (a later response-time reading that remains inside /// the normal renewal window) therefore arms the FIXED code from 10000 and the PRE-FIX code from @@ -1622,12 +1784,17 @@ struct SequencedBootClock /// report, not re-asserted here: this test body only encodes the FIXED expectation.) TEST(CASPoolRemount, RemountArmAnchorsAtClaimAttemptNotResponseTime) { - SequencedBootClock clock; + /// Heap-owned, not a plain stack local: the Pool can outlive this stack frame (a background + /// publish holds `shared_from_this()`), so a by-reference capture of a local would dangle. + auto clock = std::make_shared(); auto backend = std::make_shared(); auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30'000), - .boot_ms_fn = [&] { return clock(); }, + .boot_ms_fn = [clock] + { + return (*clock)(); + }, }); ASSERT_TRUE(store); @@ -1638,8 +1805,8 @@ TEST(CASPoolRemount, RemountArmAnchorsAtClaimAttemptNotResponseTime) /// an unrelated number of `bootMsNow()` calls (all served from `.steady = 0` -- irrelevant, since /// nothing probes the resulting arm before this point). Reset the counter so the FIRST call from /// here on is the remount attempt's own call #1. - clock.queue = {10000, 11000}; - clock.next = 0; + clock->queue = {10000, 11000}; + clock->next = 0; ASSERT_TRUE(store->tryRemountOnce()); @@ -1647,35 +1814,183 @@ TEST(CASPoolRemount, RemountArmAnchorsAtClaimAttemptNotResponseTime) /// (10000), so the fence has JUST expired here -- `mayMutate` must be false. (The pre-fix code /// would still read `mayMutate` as true here, armed from 11000 + 30000 -- see the TDD run in the /// report.) - clock.steady = 40000; + clock->steady = 40000; EXPECT_FALSE(store->mayMutate()) << "the remount arm must anchor at the claim attempt's pre-I/O instant, not a later " - "response-time reading taken after keeperStart/quiesceRefTablesForRemount"; + "response-time reading taken after renewerStart/quiesceRefTablesForRemount"; +} + +/// ==== self-remount vs. a live successor carrying the same uuid under the unsafe-reclaim knob ==== +/// +/// `cas_unsafe_remount_no_delay` is consulted at exactly one site: the writable `Pool::open` claim. +/// `Pool::tryRemountOnce` (self-remount after a fence loss) does NOT consult it -- an incarnation +/// superseded by a duplicate-uuid process must still OBSERVE the slot's write-token before it may +/// reclaim, or two processes sharing a uuid (a copied uuid file, a stalled predecessor restarted under +/// the knob) would alternate authority indefinitely. + +TEST(CASMountRemount, SupersededIncarnationDoesNotReclaimALiveSuccessor) +{ + auto backend = std::make_shared(); + /// Held in shared atomics, not plain locals: `wait_sleep_fn`/`setWaitSleepForTest` below mutate + /// them, and each Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference or by-raw-pointer capture of a local would dangle. + auto boot_a = std::make_shared>(0); + auto boot_b = std::make_shared>(0); + /// Mirrors `UncleanOpenPaysOnlyTheObservationWindow`'s tiny budget: the 1s lease TTL below is far + /// under the default `cas_request_budget`, so it must be scaled down to fit the required-timeout + /// inequality (attempt_timeout + safety_margin < lease TTL). + const CasRequestBudget tiny_budget{ + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; + auto config_for = [&](const std::shared_ptr> & boot, bool unsafe) + { + return PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(200), + .unsafe_remount_no_delay = unsafe, + .cas_request_budget = tiny_budget, + .boot_ms_fn = [boot] + { + return boot->load(); + }, + .wait_sleep_fn = [boot](uint64_t ms) + { + *boot += ms; + }, + }; + }; + + PoolPtr pool_a = Pool::open(backend, config_for(boot_a, /*unsafe=*/false)); + ASSERT_TRUE(pool_a); + /// B carries the SAME (server_root_id, server_id) as A -- a copied uuid file -- and opens over A's + /// still-live slot under the operator's unsafe knob, reclaiming it at once (no observation). + PoolPtr pool_b = Pool::open(backend, config_for(boot_b, /*unsafe=*/true)); + ASSERT_TRUE(pool_b); + EXPECT_NE(pool_a->liveWriterEpoch(), pool_b->liveWriterEpoch()) + << "the unsafe reclaim must have minted B a fresh epoch over A's slot"; + + /// A's next renewal meets the token guard: same uuid, a newer epoch now sits on the slot. Pin the + /// terminal classification directly (the "superseded" branch of `throwRenewConflict`, the one + /// that maps to `MountRenewOutcome::Terminal`) rather than accepting any exception -- no accessor + /// exposes the renewer's outcome/state today, so the error code and the classification's own + /// wording are what distinguish this from every other terminal reason (foreign owner, GC fence, + /// vanished slot, an unresolved write). + try + { + pool_a->renewWatermarkOnce(); + FAIL() << "A's renewal must be refused once B's reclaim superseded its epoch"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ABORTED); + EXPECT_NE(e.message().find("superseded by a newer incarnation"), std::string::npos) + << "actual message: " << e.message(); + } + EXPECT_FALSE(pool_a->mayMutate()) << "the superseded classification must trip A's local write fence closed"; + + /// A's self-remount now observes the slot's write-token. Drive B's renewal from INSIDE every one + /// of A's observation polls, so the token never stabilizes across the whole bounded observation -- + /// the knob is not consulted by `tryRemountOnce` (only by `Pool::open`), so nothing else could let + /// A reclaim a slot a live successor keeps renewing. This cannot deadlock: A and B are distinct + /// `Pool` objects, so B's `renewWatermarkOnce` takes none of A's locks (each `Pool` owns its own + /// `remount_mutex`), and the wait fires between `claimMountAwaitingExpiry`'s polls -- with no + /// backend request of A's own in flight -- so B's call is the only one touching the shared + /// in-memory backend at that instant. + /// Heap-owned, not a plain local: same lifetime rule as `boot_a`/`boot_b` above. + auto polls = std::make_shared>(0); + pool_a->setWaitSleepForTest([boot_a, boot_b, polls, pool_b](uint64_t ms) + { + *boot_a += ms; + ++(*polls); + *boot_b += ms; + EXPECT_NO_THROW(pool_b->renewWatermarkOnce()); + }); + EXPECT_FALSE(pool_a->tryRemountOnce()) + << "a superseded incarnation must never reclaim a live successor's slot"; + /// Bounded, not merely nonzero: B renews on every poll, so the observed token changes every + /// iteration and the FIRST (non-restart) observation start plus `kMaxObservationRestarts` further + /// restarts is exactly the number of polls before `claimMountAwaitingExpiry` gives up -- one + /// `sleep_ms_fn` call per iteration that does not itself exceed the bound, and none on the + /// terminal iteration that does. A widened or removed restart bound would make this hang instead + /// of failing, so pin the exact count rather than only asserting it ran. + EXPECT_EQ(polls->load(), DB::Cas::kMaxObservationRestarts + 1) + << "the observation must give up after exactly kMaxObservationRestarts restarts, not wait " + "indefinitely for a live twin to go quiet"; + + const MountLease final_lease = decodeMountLease(readObj(*backend, pool_a->layout().mountKey("test"))->bytes); + EXPECT_EQ(final_lease.writer_epoch, pool_b->liveWriterEpoch()) + << "the mount slot must still belong to B's incarnation -- A never reclaimed it"; +} + +/// Cutoff-only fencing: with no renewals and no competing incarnation at all, crossing the armed +/// deadline on the local BOOTTIME clock alone must fence a mount closed -- the mechanism +/// `SupersededIncarnationDoesNotReclaimALiveSuccessor` above relies on is not special-cased to a +/// renewal conflict; the plain boot-clock cutoff fences unconditionally. +TEST(CASMountRemount, CutoffFencesWithoutRenewals) +{ + auto backend = std::make_shared(); + /// Held in a shared atomic, not a plain local: this test mutates the clock below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto boot = std::make_shared>(0); + const CasRequestBudget tiny_budget{ + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; + PoolPtr store = Pool::open(backend, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(200), + .cas_request_budget = tiny_budget, + .boot_ms_fn = [boot] + { + return boot->load(); + }, + .wait_sleep_fn = [boot](uint64_t ms) + { + *boot += ms; + }, + }); + ASSERT_TRUE(store); + EXPECT_TRUE(store->mayMutate()) << "freshly armed at open, well within the ttl"; + + /// No renewals at all -- advance the boot clock past the armed deadline (open's claim anchor plus + /// the lease ttl) on this incarnation's own clock alone. + *boot += 1001; + EXPECT_FALSE(store->mayMutate()) + << "crossing the armed deadline must fence closed on the boot clock alone, with no renewal " + "conflict needed to trip it"; } /// ==== rev.6 Task 5: clean-release drain gates the farewell marker ==== namespace { -/// Forces the FIRST `putIfAbsent` whose key contains `fault_key_substr` to throw an ambiguous -/// (Unresolved-classified) exception, `fault_count` times -- the minimal one-shot subset of -/// `RefWriterTestBackend`'s fault injection (gtest_cas_ref_writer.cpp) this file's shutdown test needs -/// to drive a ref-log append into the `Unresolved`/wedge outcome, with `max_attempts = 1` in the budget -/// so the single failed attempt exhausts the retry budget immediately. +/// Makes every write whose key contains `fault_key_substr` throw an ambiguous exception -- the minimal +/// subset of `RefWriterTestBackend`'s fault injection (gtest_cas_ref_writer.cpp) this file's shutdown +/// and remount tests need to drive a ref-log append into the wedge outcome. It stays armed: one +/// ambiguous attempt is not a wedge, because the engine resolves it by reading and reissues -- the lane +/// wedges only once a bound refuses with an attempt already sent, so the tests injecting it also give +/// the pool a clock they can advance. class UnresolvedPutBackend final : public DB::Cas::tests::CountingBackend { public: String fault_key_substr; int fault_count = 0; - DB::Cas::PutResult putIfAbsent(const String & key, const String & bytes, const DB::Cas::ObjectMeta & meta) override + /// The fault sits on the WRITE PRIMITIVE: the ref-log append it models is issued there. Nothing + /// reaches the store, so the engine's resolve read proves the key absent and every reissue is + /// ambiguous again -- which is what leaves the lane wedged once a bound refuses. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override { - if (fault_count > 0 && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + /// Only the create: a ref-log append is a create-if-absent, so a conditional write on the same + /// key must not consume the fault. + if (!expected_value && fault_count > 0 && !fault_key_substr.empty() + && key.find(fault_key_substr) != String::npos) { --fault_count; throw Poco::TimeoutException("UnresolvedPutBackend: simulated ambiguous result (response lost)"); } - return DB::Cas::tests::CountingBackend::putIfAbsent(key, bytes, meta); + return DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); } }; @@ -1691,14 +2006,22 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend BlockThenThrow, }; - using DB::Cas::tests::CountingBackend::putOverwrite; - Fault fault = Fault::None; DB::Cas::tests::ManualBarrier * barrier = nullptr; std::function after_commit; - - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override + /// Runs just before an armed fault throws. The engine draws its inter-attempt backoff randomly and + /// admits the reissue against that drawn duration, so a test that needs the ambiguity to be refused + /// rather than reissued has to move the injected clock here -- from inside the attempt, which is the + /// only point between admission and the resolve read a test can reach. + std::function before_throw; + + /// The fault sits on the WRITE PRIMITIVE, and only on a CONDITIONAL one: a lease renewal is a + /// replace, so a create on the same key must not consume the one-shot fault. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override { + if (!expected_value) + return DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); const Fault current = std::exchange(fault, Fault::None); if (current == Fault::BlockThenDelegate || current == Fault::BlockThenThrow) { @@ -1707,9 +2030,13 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend barrier->arriveAndWait(); } if (current == Fault::ThrowBefore || current == Fault::BlockThenThrow) + { + if (before_throw) + before_throw(); throw Poco::TimeoutException("injected runtime renewal ambiguity before result"); + } - PutResult result = DB::Cas::tests::CountingBackend::putOverwrite(key, bytes, expected, meta); + auto result = DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); if (after_commit) after_commit(); if (current == Fault::LandThenThrow) @@ -1718,7 +2045,57 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend } }; -CasRequestBudget runtimeRenewBudget(uint32_t max_attempts); +CasRequestBudget runtimeRenewBudget(); + +/// A directly-constructed `CasMountRuntime` plus the two request planes it needs. `Pool` builds those +/// from its own members; a test has no `Pool`, so the mount plane's fence reaches the runtime through +/// this holder -- the closures run only once the runtime is issuing requests, well after construction. +class RuntimeUnderTest +{ +public: + template + RuntimeUnderTest(const std::shared_ptr & backend, Args &&... args) + : mount(backend, DB::Cas::Fence{ + [this] { return runtime.fenceGeneration(); }, + [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, + [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) + , farewell(backend, DB::Cas::Fence::open()) + , runtime(backend, mount, farewell, std::forward(args)...) + { + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the + /// budget field alone; every construction of this holder pairs the two via `runtimeRenewBudget`, + /// the sole budget it is ever built with in this file. + backend->setAttemptTimeoutMs(runtimeRenewBudget().attempt_timeout_ms); + /// The runtime arms its lease deadline on ITS boot clock, and the engine measures that deadline + /// against the clock it reads. Production runs both on `CLOCK_BOOTTIME`, so they agree; a test + /// that injects one MUST inject the other, or `Retry::untilLeaseSafe` compares a synthetic + /// deadline against real boottime, finds it long past, and refuses every request unsent. + mount.setNowFnForTest([this] { return runtime.bootMsNow(); }); + farewell.setNowFnForTest([this] { return runtime.bootMsNow(); }); + } + + /// The workers are joined HERE, not only by the tests that assert on teardown: `CasMountRuntime` + /// aborts the process when it is destroyed with a worker still joinable, so an exception on any + /// path out of a test body -- a barrier that timed out, an assertion that threw -- would take the + /// whole binary down and hide every test after it. + ~RuntimeUnderTest() + { + try + { + runtime.stopBackgroundWorkers(); + } + catch (...) // NOLINT(bugprone-empty-catch) + { + } + } + + CasMountRuntime & operator*() { return runtime; } + +private: + DB::Cas::CasRequests mount; + DB::Cas::CasRequests farewell; + CasMountRuntime runtime; +}; enum class ForeignConflictSinkBehavior : uint8_t { @@ -1738,7 +2115,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, server_root_id, uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, server_root_id, uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); std::vector events; bool reentered = false; @@ -1766,7 +2143,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav throw std::runtime_error("injected mount diagnostic sink failure"); } }; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ @@ -1775,20 +2152,23 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav }, server_root_id, sink, - runtimeRenewBudget(1), + runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; runtime_ptr = &runtime; - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - auto ours = backend->get(key); + DB::Cas::tests::OperationForTest successor_op(*backend); + auto ours = (*successor_op).read(key, Retry::standard()); ASSERT_TRUE(ours.has_value()); MountLease successor = decodeMountLease(ours->bytes); successor.server_uuid = UInt128{2}; successor.writer_epoch = 9; successor.seq += 1; - ASSERT_EQ(backend->putOverwrite(key, encodeMountLease(successor), ours->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*successor_op).replace(key, encodeMountLease(successor), ours->etag, Retry::standard()))); const uint64_t skipped_before = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); const uint64_t violations_before @@ -1834,7 +2214,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav if (failed != events.end()) EXPECT_EQ(failed->detail.at("classification"), "conflict"); - const auto successor_before_teardown = backend->get(key); + const auto successor_before_teardown = readObj(*backend, key); ASSERT_TRUE(successor_before_teardown.has_value()); const uint64_t heads_before_teardown = backend->headCount(key); const uint64_t gets_before_teardown = backend->getCount(key); @@ -1848,7 +2228,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), skipped_before_teardown); EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), violations_before); - const auto successor_after_teardown = backend->get(key); + const auto successor_after_teardown = readObj(*backend, key); ASSERT_TRUE(successor_after_teardown.has_value()); EXPECT_EQ(successor_after_teardown->bytes, successor_before_teardown->bytes); } @@ -1866,21 +2246,22 @@ TEST(CASPoolRemount, ThrowingForeignConflictSinkCannotReplaceTerminalOutcome) class RemountStepBackend final : public DB::Cas::tests::CountingBackend { public: - using DB::Cas::tests::CountingBackend::get; - - void failNextGet(String key) + void failNextRead(String key) { failed_key = std::move(key); } - std::optional get(const String & key, Range range) override + /// The fault sits on the READ PRIMITIVE: the lifecycle gate reads `_pool_meta` through + /// `probeSentinelRaw`, which speaks the primitives. A legacy caller reaches it anyway, through the + /// forwarder, so arming it here covers both surfaces rather than only one. + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (!failed_key.empty() && key == failed_key) { failed_key.clear(); throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected remount probe failure"); } - return DB::Cas::tests::CountingBackend::get(key, range); + return DB::Cas::tests::CountingBackend::read(key, access); } private: @@ -1921,7 +2302,7 @@ class ScopedParkedRenewalLogCapture { public: ScopedParkedRenewalLogCapture() - : logger(getLogger("CasMountLeaseKeeper")) + : logger(getLogger("CasMountLeaseRenewer")) , channel(new Poco::StreamChannel(stream)) , old_channel(logger->getChannel(), /*shared=*/true) , old_level(logger->getLevel()) @@ -1984,15 +2365,18 @@ class WorkerExitLatch uint64_t exits = 0; }; -CasRequestBudget runtimeRenewBudget(uint32_t max_attempts = 1) +/// The budget every runtime-renewal test uses. `attempt_timeout_ms`/`lease_safety_margin_ms` bound the +/// mount lease's own admission arithmetic; a renewal write's own attempt count and backoff are the +/// request engine's fence-derived `Retry::standard()` policy now, not a budget knob -- every caller of +/// this helper used to pass `max_attempts=1` and no other value, so that parameter carried nothing. +CasRequestBudget runtimeRenewBudget() { return CasRequestBudget{ .attempt_timeout_ms = 10, - .operation_deadline_ms = 500, - .max_attempts = max_attempts, .lease_safety_margin_ms = 20, - .retry_initial_backoff_ms = 0, - .retry_max_backoff_ms = 0, + /// The default cap (1000 ms) would make the envelope (10 + 2*1000 = 2010) blow every tiny TTL + /// this budget is used against; no connect notion is exercised by these tests. + .connect_timeout_cap_ms = std::nullopt, }; } } @@ -2006,24 +2390,43 @@ TEST(CASPoolShutdown, CleanStopDrainsAndWritesFarewell) const String mount_key = store->layout().mountKey("test"); store.reset(); /// drives ~Pool(): with no in-flight ref-log PUT, the drain must succeed. - const auto got = backend->get(mount_key); + const auto got = readObj(*backend, mount_key); ASSERT_TRUE(got.has_value()); const MountLease lease = decodeMountLease(got->bytes); - EXPECT_EQ(lease.min_active, std::numeric_limits::max()) + EXPECT_EQ(lease.min_active_build_sequence, std::numeric_limits::max()) << "a clean drain (no in-flight ref-log PUT) must write the farewell marker"; } TEST(CASPoolShutdown, UnresolvedWedgeSkipsFarewell) { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; auto backend = std::make_shared(); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); + /// Held in a shared atomic, not a plain local: `wait_sleep_fn` and the retry-sleep hook below + /// mutate it, and the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{ - .pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); + .pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }}); + /// The engine's own inter-attempt sleep advances the same clock its deadlines are read from, so the + /// retry bound is reached in test time rather than in ninety real seconds. + store->setCasRetrySleepForTest([fake_boot](uint64_t ms) + { + *fake_boot += ms; + }); /// By value: `layout` is used after `store.reset()` below, a reference would dangle. const Layout layout = store->layout(); const RootNamespace ns{"srv/wedge_shutdown"}; @@ -2033,27 +2436,28 @@ TEST(CASPoolShutdown, UnresolvedWedgeSkipsFarewell) DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); publishPart(store, ns.string(), "x", "payload"); - /// Force the ref-log append the drop below performs into the Unresolved/wedge outcome (as in the - /// wedge tests in gtest_cas_ref_writer.cpp): the single attempt the budget allows fails ambiguously. + /// Force the ref-log append the drop below performs into the wedge outcome (as in the wedge tests + /// in gtest_cas_ref_writer.cpp): every attempt is ambiguous, so the lane is still unresolved when + /// the retry bound refuses. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; + backend->fault_count = std::numeric_limits::max(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); const String mount_key = store->layout().mountKey("test"); store.reset(); /// drives ~Pool(): the still-wedged lane must skip the farewell marker. - const auto got = backend->get(mount_key); + const auto got = readObj(*backend, mount_key); ASSERT_TRUE(got.has_value()); const MountLease lease = decodeMountLease(got->bytes); - EXPECT_NE(lease.min_active, std::numeric_limits::max()) + EXPECT_NE(lease.min_active_build_sequence, std::numeric_limits::max()) << "an unresolved ref-log PUT must skip the clean-release farewell marker"; EXPECT_FALSE(lease.gc_fenced); /// A successor claimMount on this body must return LiveDoubleStart (unclean path): no certificate of /// death (not fenced, not the clean farewell marker, no proven-dead observation) justifies a /// same-uuid, different-epoch reclaim. - const MountClaimResult claim = claimMount(*backend, layout, "test", lease.server_uuid, + const MountClaimResult claim = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", lease.server_uuid, lease.writer_epoch + 1, /*now_ms=*/1, /*ttl_ms=*/30000); EXPECT_EQ(claim.kind, MountClaimResult::LiveDoubleStart); } @@ -2073,24 +2477,27 @@ TEST(CASMountOpenWaits, UncleanOpenPaysOnlyTheObservationWindow) auto b = std::make_shared(); Layout l{"p"}; DB::Cas::tests::seedPoolMetaForRestart(*b); - /// Predecessor: claim epoch 7, no farewell (simulate crash: just drop the keeper) -- a bare - /// `claimMount` plants the lease directly, with no clean-farewell `min_active` marker and no + /// Predecessor: claim epoch 7, no farewell (simulate crash: just drop the renewer) -- a bare + /// `claimMount` plants the lease directly, with no clean-farewell `min_active_build_sequence` marker and no /// `gc_fenced`, so the successor below has no certificate of death until it observes one itself. - ASSERT_EQ(claimMount(*b, l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(b), l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, MountClaimResult::Claimed); /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs /// before the mount claim); seed that durable epoch object here too, or the successor's own /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). - b->putIfAbsent(l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + createObj(*b, l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); /// A 500ms lease TTL is far below the default `cas_request_budget` (RFC /// cas-s3-timeout-retry-control §required-timeout-model requires attempt_timeout + safety_margin < /// lease TTL), so scale the budget down to fit -- mirrors `CasMountStartup::StaleSelfMountReclaimedAfterWait`. const CasRequestBudget tiny_budget{ - .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; - uint64_t fake_boot = 0; - std::vector waits; + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); + auto waits = std::make_shared(); PoolPtr store; ASSERT_NO_THROW( store = Pool::open(b, PoolConfig{ @@ -2098,90 +2505,351 @@ TEST(CASMountOpenWaits, UncleanOpenPaysOnlyTheObservationWindow) .mount_lease_ttl_ms = std::chrono::milliseconds(500), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = tiny_budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, })); ASSERT_TRUE(store); - /// The token-stability observation window (>= the 500ms ttl) is paid, because this predecessor's - /// death was never certified -- only observed. + /// The token-stability observation window is paid in full, pinned to the exact configured + /// formula (`mountObservationThresholdMs`): threshold_ms = ttl_ms + ttl_ms/20 + poll_interval_ms + /// = 500 + 25 + 50 = 575 ms, where poll_interval_ms = max(1, mount_renew_period/2) = 50 ms. The + /// loop only re-checks the threshold between polls, so the observed wait rounds UP to the next + /// whole poll: ceil(575 / 50) * 50 = 600 ms, i.e. exactly 12 polls of 50 ms each -- because this + /// predecessor's death was never certified, only observed. + const std::vector observed_waits = waits->snapshot(); uint64_t total = 0; - for (uint64_t w : waits) + for (uint64_t w : observed_waits) total += w; - EXPECT_GE(total, 500u) << "the observation window must have been paid"; - /// And NOTHING is paid on top of it. Every recorded wait is a poll of that window, bounded by the - /// lease TTL; a wait longer than the whole window can only be a reintroduced grace period. - for (uint64_t w : waits) - EXPECT_LE(w, 500u) + EXPECT_EQ(total, 600u) << "the observation window must be paid in full, poll-rounded to the " + "configured threshold -- neither less (a shortened wait) nor more " + "(a reintroduced grace period)"; + /// And every one of those polls is exactly one poll interval -- no wait beyond the observation + /// poll (the straggler it used to wait out is fenced by the recovery seal instead). + for (uint64_t w : observed_waits) + EXPECT_EQ(w, 50u) << "an unclean reclaim must not block on any wait beyond the observation poll -- the " "straggler it used to wait out is fenced by the recovery seal instead"; } +TEST(CASMountOpenWaits, UnsafeNoDelayOpensWithoutTheObservationWindow) +{ + auto b = std::make_shared(); + Layout l{"p"}; + DB::Cas::tests::seedPoolMetaForRestart(*b); + /// Same predecessor shape as UncleanOpenPaysOnlyTheObservationWindow above: a bare `claimMount` + /// plants the lease directly, with no clean-farewell marker and no `gc_fenced`, so this slot has no + /// certificate of death -- only `cas_unsafe_remount_no_delay` below will let the successor skip + /// observing it. + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(b), l, "test", UInt128(1), 7, 1000, 500).kind, MountClaimResult::Claimed); + /// A real predecessor at epoch 7 durably minted this first; seed it here too, or the successor's + /// own `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). + createObj(*b, l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto events = std::make_shared(); + auto fake_boot = std::make_shared>(0); + auto waits = std::make_shared(); + PoolPtr store; + /// Same server_id (uuid) as the seeded predecessor and a different epoch -- exactly the shape + /// `unsafe_remount_no_delay` is for. Unlike the neighbour test, no wait is expected: the bare + /// `claimMount` reclaims at once under the operator's authorization. + ASSERT_NO_THROW(store = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .event_sink = [events](CasEvent e) + { + events->push(std::move(e)); + }, + .mount_lease_ttl_ms = std::chrono::milliseconds(500), .mount_renew_period = std::chrono::milliseconds(100), + .unsafe_remount_no_delay = true, + .cas_request_budget = CasRequestBudget{.attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, + })); + ASSERT_TRUE(store); + EXPECT_TRUE(waits->snapshot().empty()) << "no observation window under the unsafe setting"; + /// `Pool` has no test accessor for the adopted `MountPriorState`, so the `UncleanUnsafe` + /// classification is asserted through the mount audit event instead: `claimMount`'s unsafe-reclaim + /// branch (`CasServerRoot.cpp`) emits exactly one `MountClaim`/"reclaim" event whose reason names + /// the setting, and `CASMountClaim.UnsafeAuthorizationIsTokenExact` already pins the classification + /// itself at the `claimMount` level. + const std::vector observed_events = events->snapshot(); + const auto reclaim_event = std::ranges::find_if(observed_events, + [](const CasEvent & e) { return e.reason.find("cas_unsafe_remount_no_delay") != String::npos; }); + ASSERT_NE(reclaim_event, observed_events.end()); + EXPECT_EQ(reclaim_event->type, CasEventType::MountClaim); + EXPECT_EQ(reclaim_event->outcome, "reclaim"); + EXPECT_EQ(decodeMountLease((*DB::Cas::tests::OperationForTest(b)).read(l.mountKey("test"), Retry::standard())->bytes).writer_epoch, 8u); +} + TEST(CASMountOpenWaits, CleanOpenSkipsAllWaits) { auto b = std::make_shared(); /// Predecessor released cleanly (drain + farewell from Task 5): open, then reset() drives ~Pool(), - /// which -- with nothing in flight -- writes the farewell marker (min_active == UINT64_MAX). + /// which -- with nothing in flight -- writes the farewell marker (min_active_build_sequence == UINT64_MAX). auto predecessor = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test"}); predecessor.reset(); - std::vector waits; + /// Heap-owned, not a plain local: the hook below mutates it, and the Pool can outlive this stack + /// frame (a background publish holds `shared_from_this()`), so a by-reference capture would dangle. + auto waits = std::make_shared(); PoolPtr successor; ASSERT_NO_THROW( successor = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", - .wait_sleep_fn = [&](uint64_t ms) { waits.push_back(ms); }, + .wait_sleep_fn = [waits](uint64_t ms) + { + waits->push(ms); + }, })); ASSERT_TRUE(successor); - EXPECT_TRUE(waits.empty()) + EXPECT_TRUE(waits->snapshot().empty()) << "a clean farewell (Task 5) needs no observation window"; } +namespace +{ +/// Reports the SHIPPED PRODUCTION default envelope (`CasRequestBudget{}`'s own defaults -- +/// `attempt_timeout_ms=5000`, `connect_timeout_cap_ms=1000` -> `attemptEnvelopeMs()=7000`), so the +/// teardown below pays the SAME two-envelope reservation (14000 ms) production pays, not the +/// near-zero envelope a bare `InMemoryBackend` reports by default. +struct DefaultBudgetEnvelopeBackend : InMemoryBackend +{ + uint64_t attemptTimeoutMs() const override { return 5000; } + uint64_t attemptEnvelopeMs() const override { return 7000; } +}; +} + +/// `CleanOpenSkipsAllWaits` above proves a clean farewell skips the observation window, but its bare +/// `InMemoryBackend` reports a zero attempt envelope, so its teardown never exercises the farewell's +/// own policy window against a write's real cost. Pin the shipped default budget specifically: a +/// window that cannot admit the write's `2 * attemptEnvelopeMs()` reservation refuses the farewell +/// before its first attempt, and the successor below then pays a full incarnation-stability +/// observation instead of reclaiming instantly. +TEST(CASMountOpenWaits, CleanTeardownUnderDefaultBudgetLeavesAFarewell) +{ + auto b = std::make_shared(); + auto predecessor = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test"}); + predecessor.reset(); /// drives ~Pool(): with nothing in flight, this is the graceful-shutdown farewell. + + const Layout layout{"p"}; + const auto got = readObj(*b, layout.mountKey("test")); + ASSERT_TRUE(got.has_value()); + const MountLease lease = decodeMountLease(got->bytes); + EXPECT_EQ(lease.min_active_build_sequence, std::numeric_limits::max()) + << "the farewell's policy window must admit the write's own two-envelope reservation at the " + "shipped default budget (2 * 7000 ms) -- otherwise a clean teardown never hands the mount " + "slot back"; + + /// Heap-owned, not a plain local: the hook below mutates it, and the Pool can outlive this stack + /// frame (a background publish holds `shared_from_this()`), so a by-reference capture would dangle. + auto waits = std::make_shared(); + PoolPtr successor; + ASSERT_NO_THROW( + successor = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .wait_sleep_fn = [waits](uint64_t ms) + { + waits->push(ms); + }, + })); + ASSERT_TRUE(successor); + EXPECT_TRUE(waits->snapshot().empty()) + << "a clean farewell needs no observation window on reopen, even at the shipped default budget"; +} + TEST(CASMountOpenWaits, FencedPriorReclaimsWithoutAnyWait) { auto b = std::make_shared(); Layout l{"p"}; DB::Cas::tests::seedPoolMetaForRestart(*b); - ASSERT_EQ(claimMount(*b, l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(b), l, "test", UInt128(1), /*epoch*/ 7, /*now_ms*/ 1000, /*ttl_ms*/ 500).kind, MountClaimResult::Claimed); /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs /// before the mount claim); seed that durable epoch object here too, or the successor's own /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). - b->putIfAbsent(l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + createObj(*b, l.epochKey("test"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); /// Predecessor lease carries gc_fenced=true: fence it directly, exactly as `computeHeartbeatFloor`'s /// fence-out does (preserve the body, gc_fenced = true, seq + 1, token-guarded). fenceOutMount(*b, l.mountKey("test")); /// See UncleanOpenPaysOnlyTheObservationWindow above: a 500ms TTL needs a scaled-down budget too. const CasRequestBudget tiny_budget{ - .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; - std::vector waits; + /// Heap-owned, not a plain local: the hook below mutates it, and the Pool can outlive this stack + /// frame (a background publish holds `shared_from_this()`), so a by-reference capture would dangle. + auto waits = std::make_shared(); PoolPtr store; ASSERT_NO_THROW( store = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(500), .cas_request_budget = tiny_budget, - .wait_sleep_fn = [&](uint64_t ms) { waits.push_back(ms); }, + .wait_sleep_fn = [waits](uint64_t ms) + { + waits->push(ms); + }, })); ASSERT_TRUE(store); /// A GC-fenced prior is a terminal, already-threshold-gated certificate of death -- reclaimed on the /// FIRST attempt, with no observation polling. It is also an UNCLEAN prior, which used to mean it /// paid the materialization grace; nothing is owed now, so this open blocks on nothing at all. - EXPECT_TRUE(waits.empty()) + EXPECT_TRUE(waits->snapshot().empty()) << "a certified-dead predecessor needs neither the observation window nor any grace period"; } +/// The open-time publication horizon must reserve TWO attempt envelopes (connect cap included), not +/// two bare attempt timeouts -- a slow connect could otherwise overrun the reservation the horizon +/// check was guarding. `background_watermark` defaults false (not set below), so `CasPool.cpp`'s +/// `renewal_window_ms` ternary takes its no-period branch: `2 * attemptEnvelopeMs()`. The check is also +/// STRICT (refuses equality), matching `CasMountRuntime::admit`. +TEST(CASMountOpenWaits, PublicationHorizonUsesTheEnvelope) +{ + /// Opens with a boot clock costing `per_call_ms` per read (models a faster or slower claim) and + /// returns how many times the mount key was written. attempt 100, cap 100: envelope = + /// 100 + 2*100 = 300, so 2*envelope = 600; the old code reserved 2*attempt = 200. Empirically the + /// claim path's own anchor read and the horizon check's own `now_boot_ms` read are five reads apart, + /// so `remaining = safe_deadline(TTL 1000 - margin 50 = 950) - now = 950 - 5 * per_call_ms`. + const auto mountWriteCount = [](uint64_t per_call_ms) -> uint64_t + { + auto b = std::make_shared(); + Layout l{"p"}; + DB::Cas::tests::seedPoolMetaForRestart(*b); + /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can + /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so + /// a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); + PoolPtr store; + store = Pool::open(b, PoolConfig{ + .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .cas_request_budget = CasRequestBudget{.attempt_timeout_ms = 100, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = 100}, + .boot_ms_fn = [fake_boot, per_call_ms] + { + return fake_boot->fetch_add(per_call_ms); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }, + }); + if (!store) + return 0; + return b->putOverwriteCount(l.mountKey("test")) + b->putCount(l.mountKey("test")); + }; + + /// remaining = 945 (per_call_ms=1): both 2*attempt(200) and 2*envelope(600) fit -- two writes (the + /// claim's own reclaim, then the renewer's adopt) and no re-anchor. + EXPECT_EQ(mountWriteCount(1), 2u) << "a horizon that fits both windows must not re-anchor"; + /// remaining = 450 (per_call_ms=100): 2*attempt(200) fits, 2*envelope(600) does not -- the + /// re-anchor costs one extra write. This is the discriminator: reverting the reservation to + /// 2*attempt would make this case behave like the one above (two writes). + EXPECT_EQ(mountWriteCount(100), 3u) << "the old 2*attempt window fit here; only the envelope window must redo"; + /// remaining = 600 (per_call_ms=70) exactly equals 2*envelope: STRICT ("<", not "<=") refuses + /// equality too, so this must also redo -- reverting the strict comparison to "<=" would make this + /// case behave like the fits-both case (two writes). + EXPECT_EQ(mountWriteCount(70), 3u) << "an exact boundary (renewal_window_ms == remaining) must be refused, not accepted"; +} + +/// Same reservation change as `PublicationHorizonUsesTheEnvelope` above, exercised through the remount +/// path's own `renewer_redo` step (`CasPool.cpp` ~1503). Modelled directly on +/// `CASPoolRemount.TheRenewerRedoRenewsOnTheOpenPlane` above: the step's admission refuses a driver that +/// was never parked by a persistent renewal worker, so `background_watermark` must be true and the +/// remount must be driven through `scheduleRemountForTest` (which parks the worker before running it), +/// never through a bare `tryRemountOnce` with no workers -- the direct-driven attempt deadlocks in +/// exactly the way that test's own comment describes. +TEST(CASPoolRemount, RemountRenewerRedoUsesTheEnvelope) +{ + /// One successful self-remount whose quiescence costs `quiesce_ms`; returns the conditional + /// mount-slot writes it issued, counted while the remount worker is still held inside the event + /// sink that reported the result (so the renewal worker it un-parks cannot add one). + const auto remountConditionalMountWrites = [](uint64_t quiesce_ms) -> uint64_t + { + auto backend = std::make_shared(); + /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can + /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so + /// a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); + /// Heap-owned, not a plain local: declaration order relative to `store` only protects against an + /// ordinary same-thread unwind, not a detached background completion that holds an extra + /// `shared_from_this()` and can still be running on another thread after this call returns. + auto committed = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "remount-renewer-redo-envelope", + .server_root_id = "test", + .background_watermark = true, + .event_sink = [committed](const CasEvent & event) + { + if (event.type == CasEventType::MountRemount && event.outcome == "ok") + committed->arriveAndWait(); + }, + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(100), + .cas_request_budget = CasRequestBudget{.attempt_timeout_ms = 100, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = 100}, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }, + .remount_quiesce_hook_for_test = [fake_boot, quiesce_ms] + { + *fake_boot += quiesce_ms; + }, + }); + const String mount_key = store->layout().mountKey("test"); + + fenceOutMount(*backend, mount_key); + const uint64_t before = backend->putOverwriteCount(mount_key); + EXPECT_TRUE(store->scheduleRemountForTest()) + << "the remount must be latched with quiesce_ms=" << quiesce_ms; + committed->waitUntilArrived(); + const uint64_t writes = backend->putOverwriteCount(mount_key) - before; + committed->release(); + return writes; + }; + + /// attempt 100, cap 100: envelope = 100 + 2*100 = 300, so period(100) + 2*envelope(600) = 700. A + /// 450 ms quiescence leaves remaining = TTL(1000) - margin(50) - 450 = 500: the old + /// period + 2*attempt (300) window fit that, but the new period + 2*envelope (700) window does not + /// -- so only the envelope-based check must redo (validation: 100 + 600 + 50 = 750 < 1000). + const uint64_t control = remountConditionalMountWrites(0); + EXPECT_GT(remountConditionalMountWrites(450), control) + << "a quiescence that fits the old attempt-only window but not the envelope window must still cost the redo"; + /// A 250 ms quiescence leaves remaining = 950 - 250 = 700, exactly equal to + /// period + 2*envelope (700): STRICT ("<", not "<=") refuses equality too, so this must also + /// redo -- reverting the strict comparison to "<=" would make this case behave like the control. + EXPECT_GT(remountConditionalMountWrites(250), control) + << "an exact boundary (renewal_window_ms == remaining) must be refused, not accepted"; +} + namespace { /// Stalls the CLAIM ITSELF past the lease TTL, and counts what the open writes afterwards. /// /// The mount key is written twice before the write fence arms: once by `claimMount`'s reclaim, then -/// once by the keeper's adopt -- and the fence's anchor is taken BETWEEN them. So advancing the +/// once by the renewer's adopt -- and the fence's anchor is taken BETWEEN them. So advancing the /// injected boot clock on the SECOND write models exactly the thing the Phase B redo exists for: the /// claim's own I/O outliving the lease it is about to arm a fence under. (This used to be modelled by /// a materialization grace long enough to consume the TTL; that wait is retired, and the guard it @@ -2194,10 +2862,14 @@ class StalledMountClaimBackend final : public DB::Cas::InMemoryBackend std::atomic mount_writes{0}; std::atomic mount_writes_after_stall{0}; - DB::Cas::PutResult putOverwrite(const String & k, const String & b, const DB::Cas::Token & e, - const DB::Cas::ObjectMeta & m) override + /// The hook sits on the WRITE PRIMITIVE: both the reclaim and the renewer's adopt reach the mount + /// slot through it. It counts only CONDITIONAL overwrites, which is what every production mount-slot + /// write is -- the unconditional create that seeds the predecessor lease reaches this same virtual + /// too, and counting it would shift the stall onto the reclaim instead of the adopt. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override { - if (k == mount_key) + if (key == mount_key && expected_value) { const int n = ++mount_writes; if (n == 2 && on_second_mount_write) @@ -2205,7 +2877,7 @@ class StalledMountClaimBackend final : public DB::Cas::InMemoryBackend else if (n > 2) ++mount_writes_after_stall; } - return InMemoryBackend::putOverwrite(k, b, e, m); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; } @@ -2237,30 +2909,39 @@ TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) prior.expires_at_ms = 1; /// long expired prior.gc_fenced = true; prior.write_attempt_id = DB::UInt128{7}; - backend->putIfAbsent(layout.mountKey(srid), DB::Cas::encodeMountLease(prior)); + createObj(*backend, layout.mountKey(srid), DB::Cas::encodeMountLease(prior)); } /// A real predecessor at epoch 7 durably minted it first (`allocateWriterEpoch` always runs /// before the mount claim); seed that durable epoch object here too, or `Pool::open`'s own /// `allocateWriterEpoch` trips the Phase C guard (epoch absent, mount present -> fail closed). - backend->putIfAbsent(layout.epochKey(srid), DB::Cas::encodeServerEpoch(DB::Cas::ServerEpoch{.next_writer_epoch = 8})); - uint64_t fake_boot_ms = 10'000; + createObj(*backend, layout.epochKey(srid), DB::Cas::encodeServerEpoch(DB::Cas::ServerEpoch{.next_writer_epoch = 8})); + /// Held in a shared atomic, not a plain local: `on_second_mount_write` below mutates it, and the + /// Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot_ms = std::make_shared>(10'000); DB::Cas::PoolConfig cfg; cfg.pool_prefix = "pool"; cfg.server_id = uuid; cfg.server_root_id = srid; cfg.background_watermark = true; cfg.mount_lease_ttl_ms = std::chrono::milliseconds(30'000); - cfg.boot_ms_fn = [&] { return fake_boot_ms; }; - /// The keeper's adopt write stalls for 15 s of boot clock. That consumes the publication horizon + cfg.boot_ms_fn = [fake_boot_ms] + { + return fake_boot_ms->load(); + }; + /// The renewer's adopt write stalls for 15 s of boot clock. That consumes the publication horizon /// (one 10 s cadence plus one 5 s attempt) while leaving one physical attempt admissible inside /// the old lease's safety window, so the synchronous redo can safely re-anchor. - backend->on_second_mount_write = [&] { fake_boot_ms += 15'000; }; + backend->on_second_mount_write = [fake_boot_ms] + { + *fake_boot_ms += 15'000; + }; auto store = DB::Cas::Pool::open(backend, cfg); ASSERT_NE(store, nullptr); ASSERT_EQ(backend->mount_writes.load(), 3) - << "the fixture assumes exactly two mount writes before the redo (the reclaim and the keeper's " + << "the fixture assumes exactly two mount writes before the redo (the reclaim and the renewer's " "adopt, with the fence anchor between them); a different sequence would make the stall land " "somewhere else and this test would stop testing the redo"; EXPECT_EQ(backend->mount_writes_after_stall.load(), 1) @@ -2280,50 +2961,84 @@ TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) TEST(CASRemountWaits, DrainedRemountPaysNoWait) { auto backend = std::make_shared(); - uint64_t fake_boot = 1'000'000; - std::vector waits; + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); + auto waits = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30000), - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, }); ASSERT_TRUE(store); - EXPECT_TRUE(waits.empty()) << "a fresh mount (no predecessor) pays no wait at open"; + EXPECT_TRUE(waits->snapshot().empty()) << "a fresh mount (no predecessor) pays no wait at open"; + store->setCasRetrySleepForTest([fake_boot](uint64_t ms) + { + *fake_boot += ms; + }); /// Trip the fence: advance the local boot clock past the deadline (as in `WriteFenceUsesInjectedBootClock` /// above) and mark the durable lease `gc_fenced` (the certificate `claimMountAwaitingExpiry` reclaims /// on its FIRST attempt, no observation polling -- avoids a real sleep in this test). - fake_boot += 30001; + *fake_boot += 30001; fenceOutMount(*backend, store->layout().mountKey("test")); /// No in-flight ref-log PUT at all -- the easy direction. ASSERT_TRUE(store->tryRemountOnce()); - EXPECT_TRUE(waits.empty()) + EXPECT_TRUE(waits->snapshot().empty()) << "a drained self-remount must pay no wait"; } TEST(CASRemountWaits, UnresolvedWedgeRemountPaysNoWaitEither) { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; auto backend = std::make_shared(); - uint64_t fake_boot = 1'000'000; - std::vector waits; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); + auto waits = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30000), .cas_request_budget = budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, }); ASSERT_TRUE(store); - EXPECT_TRUE(waits.empty()) << "a fresh mount (no predecessor) pays no wait at open"; + EXPECT_TRUE(waits->snapshot().empty()) << "a fresh mount (no predecessor) pays no wait at open"; + /// `dropRef` below drives the fault through `ensureRefTableRecovered`'s own recovery-retry loop, + /// which sleeps via `recovery_retry_sleep_fn` (a REAL 200ms-slice sleep by default) while measuring + /// elapsed time against `boot_ms_now_fn` -- the frozen `fake_boot` this fixture already injects. + /// Without also virtualizing the sleep, that elapsed check never advances and the loop spins for + /// real until the harness times the test out. + store->setCasRetrySleepForTest([fake_boot](uint64_t ms) + { + *fake_boot += ms; + }); const Layout & layout = store->layout(); const RootNamespace ns{"srv/remount_wedge"}; @@ -2335,12 +3050,12 @@ TEST(CASRemountWaits, UnresolvedWedgeRemountPaysNoWaitEither) /// `CASPoolShutdown.UnresolvedWedgeSkipsFarewell`): the single attempt the budget allows fails /// ambiguously. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; + backend->fault_count = std::numeric_limits::max(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); /// Trip the fence exactly as in `DrainedRemountSkipsGrace` above. - fake_boot += 30001; + *fake_boot += 30001; fenceOutMount(*backend, store->layout().mountKey("test")); /// THE HARD DIRECTION, and the one the retired wait existed for: a ref lane that still holds an @@ -2349,7 +3064,7 @@ TEST(CASRemountWaits, UnresolvedWedgeRemountPaysNoWaitEither) /// recovery writes into its slot. ASSERT_TRUE(store->tryRemountOnce()); - EXPECT_TRUE(waits.empty()) + EXPECT_TRUE(waits->snapshot().empty()) << "an unresolved ref-lane wedge must not make the remount block: the straggler it describes is " "fenced by the recovery seal, not waited out"; } @@ -2369,21 +3084,35 @@ TEST(CASRemountWaits, UnresolvedWedgeRemountPaysNoWaitEither) TEST(CASRemountWaits, ALateTouchedTableClosesEveryDeadEpochInBandHoweverItsPredecessorsDied) { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; auto backend = std::make_shared(); - uint64_t fake_boot = 1'000'000; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); + /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can outlive + /// this stack frame (a background publish holds `shared_from_this()`), so a by-reference capture + /// of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30000), .cas_request_budget = budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }, }); ASSERT_TRUE(store); + store->setCasRetrySleepForTest([fake_boot](uint64_t ms) + { + *fake_boot += ms; + }); const Layout & layout = store->layout(); const RootNamespace ns1{"srv/table_a"}; @@ -2403,19 +3132,19 @@ TEST(CASRemountWaits, ALateTouchedTableClosesEveryDeadEpochInBandHoweverItsPrede /// Force ns1's ref-log append into the Unresolved/wedge outcome (mirrors /// `UnresolvedWedgeRemountPaysNoWaitEither` above). backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns1)) + "_log/"; - backend->fault_count = 1; + backend->fault_count = std::numeric_limits::max(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns1, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns1)); /// Self-remount #1: UNCLEAN (the wedge above). Epoch 1 -> 2. - fake_boot += 30001; + *fake_boot += 30001; fenceOutMount(*backend, store->layout().mountKey("test")); ASSERT_TRUE(store->tryRemountOnce()); ASSERT_EQ(store->liveWriterEpoch(), 2u); /// Self-remount #2: CLEAN (no wedge left behind -- `quiesceRefTablesForRemount` already cleared the /// cache). Epoch 2 -> 3. - fake_boot += 30001; + *fake_boot += 30001; fenceOutMount(*backend, store->layout().mountKey("test")); ASSERT_TRUE(store->tryRemountOnce()); ASSERT_EQ(store->liveWriterEpoch(), 3u); @@ -2430,12 +3159,12 @@ TEST(CASRemountWaits, ALateTouchedTableClosesEveryDeadEpochInBandHoweverItsPrede EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2) << "both dead epochs must be closed -- the chain link is what a later reader needs to tell an " "EMPTY epoch from a LOST one, and that is independent of how each mount ended"; - EXPECT_TRUE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{1, 2})).has_value()) + EXPECT_TRUE(readObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{1, 2})).has_value()) << "epoch 1 closes at the slot right after its last durable id, in-band"; - EXPECT_TRUE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{2, 1})).has_value()) + EXPECT_TRUE(readObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{2, 1})).has_value()) << "empty epoch 2 closes at its own sequence 1, chained to the epoch-1 seal"; const RefTxnId retired_sentinel_id{2, std::numeric_limits::max()}; - EXPECT_FALSE(backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns2), retired_sentinel_id)).has_value()) + EXPECT_FALSE(readObj(*backend, layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns2), retired_sentinel_id)).has_value()) << "and NO synthetic seal snapshot is written: that shape is retired"; } @@ -2470,32 +3199,185 @@ TEST(CASPool, ReadManifestSharedReturnsSharedDecodeWithoutCopy) auto m2 = store->readManifestShared(resolved->manifest_id); EXPECT_EQ(m1.get(), m2.get()); /// the SAME shared decode, no copy EXPECT_EQ(backend->getCount(manifest_key), 1u); /// one body GET - EXPECT_EQ(backend->headCount(manifest_key), 2u); /// mandatory HEAD per call (unchanged) + EXPECT_EQ(backend->headCount(manifest_key), 0u); /// keyed by id: no HEAD on a miss or a hit ASSERT_EQ(m1->entries.size(), 1u); EXPECT_EQ(m1->entries[0].path, "data.bin"); } +/// A miss whose object is absent is the one dangling-reference case the reader still detects +/// itself: exactly one GET, no HEAD, one `ReadMissing` event, FILE_DOESNT_EXIST. +TEST(CASPool, ReadManifestAbsentBodyEmitsReadMissingWithOneGetAndNoHead) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + const RootNamespace ns{"srv1/tbl"}; + const ManifestId id{.root_namespace = ns, .ref = manifestRefFor("absent-body-event")}; + const String key = layout.manifestKey(id); + + /// Heap-owned, not a plain local: `setEventSink(nullptr)` below only stops FUTURE sink installs from + /// using this closure -- it does not guarantee an already-in-flight background call is not still + /// executing the old one -- and the Pool can outlive this stack frame regardless (a background + /// publish holds `shared_from_this()`). + auto events = std::make_shared(); + s->setEventSink([events](CasEvent e) + { + events->push(std::move(e)); + }); + b->resetCounts(); + expectThrowsCode(DB::ErrorCodes::FILE_DOESNT_EXIST, [&] { s->readManifest(id); }); + s->setEventSink(nullptr); + + EXPECT_EQ(b->getCount(key), 1u); + EXPECT_EQ(b->headCount(key), 0u); + size_t read_missing = 0; + for (const auto & e : events->snapshot()) + { + if (e.type != CasEventType::ReadMissing) + continue; + ++read_missing; + EXPECT_EQ(e.object_kind, CasEventObjectKind::Manifest); + EXPECT_EQ(e.detail.at("code"), "FILE_DOESNT_EXIST"); + EXPECT_EQ(e.detail.at("site"), "readManifest"); + } + EXPECT_EQ(read_missing, 1u); +} + +/// A reader holding a decode for a manifest the collector has since removed sees a snapshot-consistent +/// manifest: the second read is the same shared decode with no request, `locate` is pure, and the +/// missing blob is observed only when its key is read. Nothing here is a fallback: the absence is +/// surfaced by the blob read, never masked by the cache. +TEST(CASPool, StaleSnapshotServesCachedManifestAndBlobAbsenceSurfacesOnRead) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + const RootNamespace ns{"srv1/tbl"}; + + const ManifestId id = publishPart(s, ns.string(), "part_1", "payload-1"); + auto r = s->resolveRef(ns, "part_1"); + ASSERT_TRUE(r.has_value()); + auto m1 = s->readManifestShared(r->manifest_id); + const String manifest_key = layout.manifestKey(id); + const String blob_key = layout.blobKey(idOf("payload-1")); + + /// What GC does after the owner is removed and the decrement is adopted: exact-token deletes of + /// the body and of the now-unreferenced blob. + { + DB::Cas::tests::OperationForTest op(*b); + const auto h = (*op).head(manifest_key, Retry::standard()); + ASSERT_TRUE(h.has_value()); + (*op).remove(manifest_key, h->etag, Retry::once()); + } + { + DB::Cas::tests::OperationForTest op(*b); + const auto h = (*op).head(blob_key, Retry::standard()); + ASSERT_TRUE(h.has_value()); + (*op).remove(blob_key, h->etag, Retry::once()); + } + b->resetCounts(); + + auto m2 = s->readManifestShared(r->manifest_id); + EXPECT_EQ(m1.get(), m2.get()); + EXPECT_EQ(b->getCount(manifest_key), 0u); + EXPECT_EQ(b->headCount(manifest_key), 0u); + + ASSERT_EQ(m2->entries.size(), 1u); + const BlobLocation location = s->locate(m2->entries[0]); + EXPECT_EQ(location.key, blob_key); + EXPECT_EQ(b->getCount(blob_key), 0u); /// locate is pure: no I/O until the read + EXPECT_FALSE(readObj(*b, location.key).has_value()); /// the read observes the absence +} + +/// The scoped contract for mutation evidence, executable. A carry-forward from a committed source +/// (what createHardLink, republishRef, repointRef and the relink receiver do) adopts each entry as a +/// tokenless TrustedManifest dependency and promote issues no probe for it: the live source edge is +/// what keeps the blob alive, and under protocol-compliant GC the state "cached source decode, blob +/// gone" cannot be constructed. Out-of-band deletion of BOTH the cached source body and the blob is +/// outside that contract; the carry-forward then commits a ref to an absent blob and fsck's +/// reachable-but-absent scan is the detector. This test pins that documented outcome so a later +/// change that silently alters it is noticed. It is not a defect report. +TEST(CASPool, CachedSourceDecodeLetsAdoptionCommitAnAbsentBlobThatFsckReports) +{ + auto b = std::make_shared(); + auto s = Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + Layout layout("p"); + const RootNamespace ns{"srv1/tbl"}; + + const ManifestId src_id = publishPart(s, ns.string(), "part_src", "payload-src"); + auto src = s->resolveRef(ns, "part_src"); + ASSERT_TRUE(src.has_value()); + const auto src_manifest = s->readManifestShared(src->manifest_id); /// warms the decode cache + const String src_manifest_key = layout.manifestKey(src_id); + const String blob_key = layout.blobKey(idOf("payload-src")); + + /// Out of band: both objects gone, the committed source ref untouched. + for (const String & key : {src_manifest_key, blob_key}) + { + DB::Cas::tests::OperationForTest op(*b); + const auto h = (*op).head(key, Retry::standard()); + ASSERT_TRUE(h.has_value()); + (*op).remove(key, h->etag, Retry::once()); + } + b->resetCounts(); + + /// The carry-forward reaches its source through the reader, the way every production caller does, + /// and the cache answers: the same decode as before, with no request on the body just deleted. + /// Adopting from the `shared_ptr` held across the deletion would prove nothing about the cache -- + /// were the cache to stop retaining, this re-read would fetch and throw, and the rest of this + /// scenario would be unreachable in production for the same reason. + const auto cached_manifest = s->readManifestShared(src->manifest_id); + ASSERT_EQ(cached_manifest.get(), src_manifest.get()); + EXPECT_EQ(b->getCount(src_manifest_key), 0u); + EXPECT_EQ(b->headCount(src_manifest_key), 0u); + + /// The carry-forward, in the order prepareEntries runs it for a committed source: adopt, stage, + /// precommit, promote. No blob body is written. + PartWriteInfo info; + info.intended_ref = ns.string() + "/part_dst"; + info.intended_namespace = ns; + auto build = s->beginPartWrite(info); + ASSERT_EQ(src_manifest->entries.size(), 1u); + build->adoptEvidence(cached_manifest->entries[0]); + const ManifestId dst_id = build->stageManifest({cached_manifest->entries[0]}); + build->precommitAdd(ns, "part_dst", dst_id); + EXPECT_NO_THROW(build->promote(ns, "part_dst", build->buildId(), dst_id)); + EXPECT_EQ(b->headCount(blob_key), 0u); /// a TrustedManifest leaf is not probed, by design + EXPECT_EQ(b->getCount(blob_key), 0u); + + /// The documented outcome: a committed ref names an absent blob, and fsck reports it. + ASSERT_TRUE(s->resolveRef(ns, "part_dst").has_value()); + const FsckReport rep = runFsck(*s, /*detail=*/true); + EXPECT_GE(rep.dangling, 1u); + bool blob_reported = false; + for (const FsckObject & o : rep.objects) + if (o.key == blob_key && o.cls == FsckClass::Dangling) + blob_reported = true; + EXPECT_TRUE(blob_reported) << "fsck must report the adopted-but-absent blob " << blob_key; +} + #if defined(DEBUG_OR_SANITIZER_BUILD) #define EXPECT_RUNTIME_STATE_REJECTION(statement) EXPECT_DEATH({ statement; }, "CAS mount runtime") #else #define EXPECT_RUNTIME_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) #endif -TEST(CASPoolRemount, DirectRenewCannotRaceWorkerStartOrKeeperReplacement) +TEST(CASPoolRemount, DirectRenewCannotRaceWorkerStartOrRenewerReplacement) { auto backend = std::make_shared(); const Layout layout("runtime-direct"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); DB::Cas::tests::ManualBarrier barrier; @@ -2504,8 +3386,8 @@ TEST(CASPoolRemount, DirectRenewCannotRaceWorkerStartOrKeeperReplacement) auto direct = std::async(std::launch::async, [&] { runtime.renewWatermarkOnce(); }); barrier.waitUntilArrived(); EXPECT_RUNTIME_STATE_REJECTION(runtime.startBackgroundWorkers(std::chrono::milliseconds(10))); - EXPECT_RUNTIME_STATE_REJECTION(runtime.installKeeper(uuid, 2, [&] { return wall_ms; })); - EXPECT_RUNTIME_STATE_REJECTION(runtime.keeperReset()); + EXPECT_RUNTIME_STATE_REJECTION(runtime.installRenewer(uuid, 2, [&] { return wall_ms; })); + EXPECT_RUNTIME_STATE_REJECTION(runtime.renewerReset()); barrier.release(); EXPECT_NO_THROW(direct.get()); runtime.finishTeardown(true); @@ -2518,11 +3400,11 @@ TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeParkRequest) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier admitted; DB::Cas::tests::ManualBarrier remount; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -2534,8 +3416,9 @@ TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeParkRequest) remount.arriveAndWait(); return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); admitted.waitUntilArrived(); @@ -2557,10 +3440,10 @@ TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeStop) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier admitted; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -2568,8 +3451,9 @@ TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeStop) .boot_ms_fn = [&] { return boot_ms; }, .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); admitted.waitUntilArrived(); @@ -2588,15 +3472,16 @@ TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.stopBackgroundWorkers(); @@ -2613,12 +3498,12 @@ TEST(CASPoolRemount, RemountWaitsForRenewalParkedBeforeReplacement) uint64_t wall_ms = 1000; uint64_t boot_ms = 10'000; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier renewal_barrier; DB::Cas::tests::ManualBarrier remount_barrier; std::atomic remount_calls{0}; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, @@ -2628,8 +3513,9 @@ TEST(CASPoolRemount, RemountWaitsForRenewalParkedBeforeReplacement) remount_barrier.arriveAndWait(); return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; @@ -2654,7 +3540,7 @@ TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); std::atomic worker_exits{0}; RuntimeWorkerFactory factory = [&](std::function worker_body) { @@ -2665,19 +3551,20 @@ TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) }); }; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.stopBackgroundWorkers(); EXPECT_EQ(worker_exits.load(), 2u); runtime.finishTeardown(true); - EXPECT_EQ(decodeMountLease(backend->get(layout.mountKey("test"))->bytes).min_active, + EXPECT_EQ(decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).min_active_build_sequence, std::numeric_limits::max()); } @@ -2692,7 +3579,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); WorkerExitLatch exits; DB::Cas::tests::ManualBarrier transitioned; RuntimeWorkerFactory factory = [&](std::function worker_body) @@ -2705,7 +3592,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit }; CasMountRuntime * runtime_ptr = nullptr; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -2721,9 +3608,10 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit transitioned.arriveAndWait(); return false; }); + CasMountRuntime & runtime = *runtime_holder; runtime_ptr = &runtime; - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.tripMountLost(); @@ -2749,7 +3637,7 @@ TEST(CASPoolRemount, ParkedRenewalCannotMissNaturalTerminalPublication) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); WorkerExitLatch exits; std::latch renewal_before_driver_lock{1}; std::latch release_renewal{1}; @@ -2775,7 +3663,7 @@ TEST(CASPoolRemount, ParkedRenewalCannotMissNaturalTerminalPublication) }; CasMountRuntime * runtime_ptr = nullptr; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -2829,9 +3717,10 @@ TEST(CASPoolRemount, ParkedRenewalCannotMissNaturalTerminalPublication) release_parked(); return false; }); + CasMountRuntime & runtime = *runtime_holder; runtime_ptr = &runtime; - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); renewal_before_driver_lock.wait(); @@ -2857,7 +3746,7 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); WorkerExitLatch exits; RuntimeWorkerFactory factory = [&](std::function worker_body) { @@ -2869,7 +3758,7 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet }; std::atomic preparation_calls{0}; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -2882,8 +3771,9 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected vanished-reason preparation failure"); }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.tripMountLost(); @@ -2926,15 +3816,16 @@ TEST(CASPoolRemount, WorkerConstructionRollbackFailsOpenClosed) return ThreadFromGlobalPool(std::move(fn)); }; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); EXPECT_THROW(runtime.startBackgroundWorkers(std::chrono::milliseconds(10)), DB::Exception); EXPECT_FALSE(runtime.mayMutate()); @@ -2950,13 +3841,16 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) uint64_t wall_ms = 1000; uint64_t boot_ms = 100'000; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier renewal_barrier; DB::Cas::tests::ManualBarrier remount_barrier; std::atomic remount_calls{0}; std::atomic fresh_epochs{0}; CasEventSink sink; - CasMountRuntime runtime( + /// The remount callback reaches the runtime it is installed on, so it goes through a pointer the + /// line after construction fills in -- the callback runs only once the workers are started. + CasMountRuntime * runtime_ptr = nullptr; + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, @@ -2965,21 +3859,23 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) ++remount_calls; ++fresh_epochs; fenceOutMount(*backend, layout.mountKey("test")); - const MountClaimResult fresh = claimMount(*backend, layout, "test", uuid, 2, wall_ms, 1000); + const MountClaimResult fresh = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000); EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); if (fresh.kind != MountClaimResult::Claimed) return false; - runtime.installKeeper(uuid, 2, [&] { return wall_ms; }); - const uint64_t fresh_anchor = runtime.startKeeper(); - runtime.setProcessEpoch(2, std::memory_order_release); - runtime.setLiveWriterEpoch(2); - runtime.armMountFence(uuid, 2, fresh_anchor + 1000); - runtime.noteRemounted(); + runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = runtime_ptr->startRenewer(); + runtime_ptr->setProcessEpoch(2, std::memory_order_release); + runtime_ptr->setLiveWriterEpoch(2); + runtime_ptr->armMountFence(uuid, 2, fresh_anchor + 1000); + runtime_ptr->noteRemounted(); remount_barrier.arriveAndWait(); return true; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; @@ -2999,20 +3895,20 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) runtime.finishTeardown(false); } -TEST(CASPoolRemount, TerminalDepositionDoesNotTouchKeeperAfterReplacement) +TEST(CASPoolRemount, TerminalDepositionDoesNotTouchRenewerAfterReplacement) { auto backend = std::make_shared(); const Layout layout("runtime-terminal-replacement"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier terminal_deposited; DB::Cas::tests::ManualBarrier remount; std::atomic replaced{false}; CasMountRuntime * runtime_ptr = nullptr; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), @@ -3020,9 +3916,9 @@ TEST(CASPoolRemount, TerminalDepositionDoesNotTouchKeeperAfterReplacement) .boot_ms_fn = [&] { return boot_ms; }, .renewal_terminal_deposited_hook_for_test = [&] { - runtime_ptr->keeperReset(); - runtime_ptr->installKeeper(uuid, 2, [&] { return wall_ms; }); - runtime_ptr->keeperReset(); + runtime_ptr->renewerReset(); + runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); + runtime_ptr->renewerReset(); replaced.store(true, std::memory_order_release); terminal_deposited.arriveAndWait(); }}, @@ -3031,11 +3927,17 @@ TEST(CASPoolRemount, TerminalDepositionDoesNotTouchKeeperAfterReplacement) remount.arriveAndWait(); return false; }); + CasMountRuntime & runtime = *runtime_holder; runtime_ptr = &runtime; - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + /// Expire the lease from inside the attempt. The fault alone no longer ends a renewal: the engine + /// settles the ambiguity by reading and then reissues, and the reissue commits. With the clock past + /// the deadline the renewal was admitted under, neither the settling read nor the reissue is + /// admitted, so the renewal ends terminal -- which is what this test deposits. + backend->before_throw = [&, deadline = anchor + 1000] { boot_ms = deadline; }; runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); terminal_deposited.waitUntilArrived(); EXPECT_TRUE(replaced.load(std::memory_order_acquire)); @@ -3053,12 +3955,12 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier first; DB::Cas::tests::ManualBarrier second; std::atomic calls{0}; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, @@ -3068,8 +3970,9 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) (call == 1 ? first : second).arriveAndWait(); return true; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.tripMountLost(); @@ -3092,12 +3995,15 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 10'000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 10'000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier first; DB::Cas::tests::ManualBarrier second; std::atomic calls{0}; CasEventSink sink; - CasMountRuntime runtime( + /// The remount callback reaches the runtime it is installed on, so it goes through a pointer the + /// line after construction fills in -- the callback runs only once the workers are started. + CasMountRuntime * runtime_ptr = nullptr; + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(10'000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, @@ -3107,24 +4013,30 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) if (call == 1) { fenceOutMount(*backend, layout.mountKey("test")); - const MountClaimResult fresh = claimMount(*backend, layout, "test", uuid, 2, wall_ms, 10'000); + const MountClaimResult fresh = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 10'000); EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); if (fresh.kind != MountClaimResult::Claimed) return false; - runtime.installKeeper(uuid, 2, [&] { return wall_ms; }); - const uint64_t fresh_anchor = runtime.startKeeper(); - runtime.armMountFence(uuid, 2, fresh_anchor + 10'000); - runtime.noteRemounted(); + runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = runtime_ptr->startRenewer(); + runtime_ptr->armMountFence(uuid, 2, fresh_anchor + 10'000); + runtime_ptr->noteRemounted(); boot_ms = 2'000; backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + /// Expire the fresh lease from inside the attempt, so the ambiguity can be neither + /// settled by a read nor reissued: otherwise the engine reissues and the renewal + /// commits, and there is no dropped failure to catch up on. + backend->before_throw = [&, deadline = fresh_anchor + 10'000] { boot_ms = deadline; }; first.arriveAndWait(); return true; } second.arriveAndWait(); return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 10'000); runtime.startBackgroundWorkers(std::chrono::milliseconds(1000)); runtime.tripMountLost(); @@ -3142,64 +4054,87 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) TEST(CASPoolRemount, StaleRemountAnchorPerformsParkedRedo) { auto backend = std::make_shared(); - uint64_t fake_boot = 100; - DB::Cas::tests::ManualBarrier committed; + /// Held in a shared atomic, not a plain local: `remount_quiesce_hook_for_test` below mutates it, + /// and the Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so + /// a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(100); + /// Heap-owned, not a plain local: declaration order relative to the Pool below only protects + /// against an ordinary same-thread unwind, not a detached background completion that holds an + /// extra `shared_from_this()` and can still be running on another thread after this frame returns. + auto committed = std::make_shared(); PoolConfig config{ .pool_prefix = "stale-remount-anchor", .server_root_id = "test", .background_watermark = true, - .event_sink = [&](const CasEvent & event) + .event_sink = [committed](const CasEvent & event) { if (event.type == CasEventType::MountRemount && event.outcome == "ok") - committed.arriveAndWait(); + committed->arriveAndWait(); }, .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [&] { return fake_boot; }, - .remount_quiesce_hook_for_test = [&] { fake_boot += 900; }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .remount_quiesce_hook_for_test = [fake_boot] + { + *fake_boot += 900; + }, }; auto store = Pool::open(backend, config); const String key = store->layout().mountKey("test"); fenceOutMount(*backend, key); const uint64_t writes_before = backend->putOverwriteCount(key); ASSERT_TRUE(store->scheduleRemountForTest()); - committed.waitUntilArrived(); + committed->waitUntilArrived(); EXPECT_GE(backend->putOverwriteCount(key), writes_before + 3) - << "claim, keeper start, and the stale-anchor parked redo must all write"; - committed.release(); + << "claim, renewer start, and the stale-anchor parked redo must all write"; + committed->release(); } TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) { auto backend = std::make_shared(); - uint64_t fake_boot = 100; - std::promise result_observed; - std::future result_future = result_observed.get_future(); - std::atomic result_published{false}; - std::mutex events_mutex; - std::vector events; + /// Held in a shared atomic, not a plain local: `remount_quiesce_hook_for_test` below mutates it, + /// and the Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so + /// a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(100); + /// Heap-owned, not plain locals: `event_sink` below mutates them, and the Pool can outlive this + /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture of a + /// local -- including a non-copyable `std::promise` -- would dangle. + auto result_observed = std::make_shared>(); + std::future result_future = result_observed->get_future(); + auto result_published = std::make_shared>(false); + auto events = std::make_shared(); PoolConfig config{ .pool_prefix = "parked-redo-recovered-observability", .server_root_id = "test", .background_watermark = true, - .event_sink = [&](CasEvent event) + .event_sink = [result_observed, result_published, events](CasEvent event) { const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "ok"; - { - std::lock_guard lock(events_mutex); - events.push_back(std::move(event)); - } - if (final_remount && !result_published.exchange(true)) - result_observed.set_value(); + events->push(std::move(event)); + if (final_remount && !result_published->exchange(true)) + result_observed->set_value(); }, .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .mount_renew_period = std::chrono::milliseconds(100), - .cas_request_budget = runtimeRenewBudget(2), - .boot_ms_fn = [&] { return fake_boot; }, - .remount_quiesce_hook_for_test = [&] + /// 500 with a 700 ms quiescence, so the redo's window (period + attempt timeout = 510) does not + /// fit the 280 ms of safe lease left -- and the reissue the ambiguity needs still does, whatever + /// the engine's jittered backoff draws from its first-reissue range of at most 200 ms. + .mount_renew_period = std::chrono::milliseconds(500), + .cas_request_budget = runtimeRenewBudget(), + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + /// `backend` is captured BY VALUE (a copy of the shared_ptr, not the stack slot holding it): + /// the Pool can outlive this frame, so a by-reference capture of the local `shared_ptr` itself + /// would dangle even though the pointee it owns is heap-allocated. + .remount_quiesce_hook_for_test = [fake_boot, backend] { - fake_boot += 900; + *fake_boot += 700; backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; }, }; @@ -3210,15 +4145,7 @@ TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) ASSERT_TRUE(store->scheduleRemountForTest()); ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); - std::vector observed; - { - std::lock_guard lock(events_mutex); - observed = events; - } - const auto retrying = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) - { - return event.type == CasEventType::WatermarkRenew && event.outcome == "retrying"; - }); + const std::vector observed = events->snapshot(); const auto recovered = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) { return event.type == CasEventType::WatermarkRenew && event.outcome == "recovered"; @@ -3227,14 +4154,13 @@ TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) { return event.type == CasEventType::MountRemount && event.outcome == "ok"; }); - ASSERT_NE(retrying, observed.end()); ASSERT_NE(recovered, observed.end()); ASSERT_NE(remounted, observed.end()); - EXPECT_LT(std::distance(observed.begin(), retrying), std::distance(observed.begin(), recovered)); EXPECT_LT(std::distance(observed.begin(), recovered), std::distance(observed.begin(), remounted)); - EXPECT_EQ(retrying->detail.at("remount_attempt_no"), remounted->detail.at("attempt_no")); EXPECT_EQ(recovered->detail.at("remount_attempt_no"), remounted->detail.at("attempt_no")); EXPECT_EQ(recovered->detail.at("classification"), "committed_after_retry"); + /// The physical retry itself: the ambiguous attempt and the reissue that committed. + EXPECT_EQ(recovered->detail.at("attempts_sent"), "2"); EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' recovered"), String::npos); /// `~Pool` stops and joins both persistent runtime workers. Make that quiescence boundary part of @@ -3246,34 +4172,47 @@ TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) TEST(CASPoolRemount, ParkedRedoFailureObservabilityPrecedesRemountResult) { auto backend = std::make_shared(); - uint64_t fake_boot = 100; - std::promise result_observed; - std::future result_future = result_observed.get_future(); - std::atomic result_published{false}; - std::mutex events_mutex; - std::vector events; + /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can outlive + /// this stack frame (a background publish holds `shared_from_this()`), so a by-reference capture + /// of a local would dangle. + auto fake_boot = std::make_shared>(100); + /// Heap-owned, not plain locals: `event_sink` below mutates them, and the Pool can outlive this + /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture of a + /// local -- including a non-copyable `std::promise` -- would dangle. + auto result_observed = std::make_shared>(); + std::future result_future = result_observed->get_future(); + auto result_published = std::make_shared>(false); + auto events = std::make_shared(); PoolConfig config{ .pool_prefix = "parked-redo-failed-observability", .server_root_id = "test", .background_watermark = true, - .event_sink = [&](CasEvent event) + .event_sink = [result_observed, result_published, events](CasEvent event) { const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "failed"; - { - std::lock_guard lock(events_mutex); - events.push_back(std::move(event)); - } - if (final_remount && !result_published.exchange(true)) - result_observed.set_value(); + events->push(std::move(event)); + if (final_remount && !result_published->exchange(true)) + result_observed->set_value(); }, .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(100), - .cas_request_budget = runtimeRenewBudget(1), - .boot_ms_fn = [&] { return fake_boot; }, - .remount_quiesce_hook_for_test = [&] + .cas_request_budget = runtimeRenewBudget(), + .boot_ms_fn = [fake_boot] { - fake_boot += 900; + return fake_boot->load(); + }, + /// `backend` is captured BY VALUE (a copy of the shared_ptr): the Pool can outlive this frame, + /// so a by-reference capture of the local `shared_ptr` itself would dangle. + .remount_quiesce_hook_for_test = [fake_boot, backend] + { + *fake_boot += 900; backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + /// The attempt is admitted 80 ms before its lease-safe bound; spending 90 inside it puts the + /// resolve read past that bound, so the ambiguity is refused instead of reissued. + backend->before_throw = [fake_boot] + { + *fake_boot += 90; + }; }, }; auto store = Pool::open(backend, config); @@ -3283,11 +4222,7 @@ TEST(CASPoolRemount, ParkedRedoFailureObservabilityPrecedesRemountResult) ASSERT_TRUE(store->scheduleRemountForTest()); ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); - std::vector observed; - { - std::lock_guard lock(events_mutex); - observed = events; - } + const std::vector observed = events->snapshot(); const auto failed_renew = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) { return event.type == CasEventType::WatermarkRenew && event.outcome == "failed"; @@ -3301,7 +4236,7 @@ TEST(CASPoolRemount, ParkedRedoFailureObservabilityPrecedesRemountResult) EXPECT_LT(std::distance(observed.begin(), failed_renew), std::distance(observed.begin(), failed_remount)); EXPECT_EQ(failed_renew->detail.at("remount_attempt_no"), failed_remount->detail.at("attempt_no")); EXPECT_EQ(failed_renew->detail.at("attempts_sent"), "1"); - EXPECT_EQ(failed_renew->detail.at("classification"), "attempts_exhausted"); + EXPECT_EQ(failed_renew->detail.at("classification"), "external_lease_deadline"); EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' fenced"), String::npos); /// A ready final-result future proves publication order; destruction additionally proves the @@ -3315,21 +4250,26 @@ TEST(CASPoolRemount, ThrowingEventSinkAfterCommitLeavesRuntimeLive) auto backend = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "throwing-remount-event", .server_root_id = "test", .background_watermark = true}); - DB::Cas::tests::ManualBarrier committed; - store->setEventSink([&](const CasEvent & event) + /// Heap-owned, not a plain local declared after `store`: if `waitUntilArrived` below throws on its + /// own internal timeout, unwinding would destroy a stack-local barrier before `store`'s destructor + /// joins the remount worker, and that worker can still be inside `arriveAndWait` on the dangling + /// reference. A `shared_ptr` capture keeps the barrier alive for as long as the worker needs it, + /// independent of declaration order. + auto committed = std::make_shared(); + store->setEventSink([committed](const CasEvent & event) { if (event.type == CasEventType::MountRemount && event.outcome == "ok") { - committed.arriveAndWait(); + committed->arriveAndWait(); throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected remount event sink failure"); } }); fenceOutMount(*backend, store->layout().mountKey("test")); ASSERT_TRUE(store->scheduleRemountForTest()); - committed.waitUntilArrived(); + committed->waitUntilArrived(); EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); EXPECT_TRUE(store->mayMutate()); - committed.release(); + committed->release(); EXPECT_NO_THROW(store.reset()); } @@ -3342,19 +4282,20 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - const MountClaimResult claim = claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000); + const MountClaimResult claim = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000); EXPECT_EQ(claim.kind, MountClaimResult::Claimed); if (claim.kind != MountClaimResult::Claimed) return uint64_t{0}; DB::Cas::tests::ManualBarrier barrier; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); if (ambiguous) { @@ -3372,7 +4313,7 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) runtime.stopBackgroundWorkers(); } runtime.finishTeardown(true); - return decodeMountLease(backend->get(layout.mountKey("test"))->bytes).min_active; + return decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).min_active_build_sequence; }; EXPECT_EQ(run(false), std::numeric_limits::max()); @@ -3381,42 +4322,38 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) TEST(CASPool, DirectAndStartupTerminalFailuresRethrowTypedExceptions) { - enum class Refusal : uint8_t { PreAttemptDeadline, CancelledAfterSend, FenceLostAfterSend }; + enum class Refusal : uint8_t { PreAttemptDeadline, RefusedAfterSend }; const auto run = [](bool startup, Refusal refusal) { auto backend = std::make_shared(); const Layout layout(startup ? "typed-startup" : "typed-direct"); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - std::atomic stop_cause{CasOverwriteStopCause::Continue}; + std::atomic renewal_live{true}; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .boot_ms_fn = [&] { return boot_ms; }, - .renewal_stop_cause_for_test = [&] { return stop_cause.load(std::memory_order_acquire); }}, + .renewal_live_for_test = [&] { return renewal_live.load(std::memory_order_acquire); }}, "test", sink, runtimeRenewBudget(), [] { return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); if (refusal == Refusal::PreAttemptDeadline) - boot_ms = 1071; + /// Past the point where the lease has more room left than the safety margin (deadline + /// `anchor + 1000` == 1100, margin 20), so admission refuses before anything is sent. + boot_ms = 1090; else - backend->after_commit = [&] - { - stop_cause.store( - refusal == Refusal::CancelledAfterSend - ? CasOverwriteStopCause::Cancelled - : CasOverwriteStopCause::FenceOrLifecycleLost, - std::memory_order_release); - }; + backend->after_commit = [&] { renewal_live.store(false, std::memory_order_release); }; try { if (startup) - (void)runtime.renewKeeperForStartupOnce(); + (void)runtime.renewRenewerForStartupOnce(); else runtime.renewWatermarkOnce(); ADD_FAILURE() << "terminal renewal did not propagate"; @@ -3428,7 +4365,7 @@ TEST(CASPool, DirectAndStartupTerminalFailuresRethrowTypedExceptions) runtime.finishTeardown(false); }; for (bool startup : {true, false}) - for (Refusal refusal : {Refusal::PreAttemptDeadline, Refusal::CancelledAfterSend, Refusal::FenceLostAfterSend}) + for (Refusal refusal : {Refusal::PreAttemptDeadline, Refusal::RefusedAfterSend}) run(startup, refusal); } @@ -3444,9 +4381,9 @@ TEST(CASPool, BackgroundCadenceMustFitLeaseBeforeWritablePublication) .cas_request_budget = runtimeRenewBudget(), }; EXPECT_THROW((void)Pool::open(backend, config), DB::Exception); - EXPECT_EQ(backend->putTotal(), 0u); - EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + /// One assertion over every write shape: the counters now sit on the write primitive, which both + /// the create- and the replace-shaped verbs reach. + EXPECT_EQ(backend->writeTotal(), 0u); } TEST(CASPool, DecommissionCadenceValidationPrecedesAuthorityWrites) @@ -3467,15 +4404,18 @@ TEST(CASPool, DecommissionCadenceValidationPrecedesAuthorityWrites) { (void)Pool::openForDecommission(backend, config, "victim"); }); - EXPECT_EQ(backend->putTotal(), 0u); - EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + /// One assertion over every write shape: the counters now sit on the write primitive, which both + /// the create- and the replace-shaped verbs reach. + EXPECT_EQ(backend->writeTotal(), 0u); } TEST(CASPool, DisabledBackgroundDoesNotReserveRenewalCadence) { auto backend = std::make_shared(); - uint64_t fake_boot = 100; + /// Captured by value: `fake_boot` is never mutated in this test, and the Pool can outlive this + /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture would + /// dangle. + const uint64_t fake_boot = 100; PoolConfig config{ .pool_prefix = "disabled-renew-cadence", .server_root_id = "test", @@ -3483,7 +4423,7 @@ TEST(CASPool, DisabledBackgroundDoesNotReserveRenewalCadence) .mount_lease_ttl_ms = std::chrono::milliseconds(100), .mount_renew_period = std::chrono::hours(24), .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [&] { return fake_boot; }, + .boot_ms_fn = [] { return fake_boot; }, }; auto store = Pool::open(backend, config); const String key = store->layout().mountKey("test"); @@ -3498,10 +4438,10 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) uint64_t wall_ms = 1000; uint64_t boot_ms = 100; const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*backend, layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); DB::Cas::tests::ManualBarrier remount_entered; CasEventSink sink; - CasMountRuntime runtime( + RuntimeUnderTest runtime_holder( backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, @@ -3510,10 +4450,14 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) remount_entered.arriveAndWait(); return false; }); - runtime.installKeeper(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startKeeper(); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + /// Expire the lease from inside the attempt, so the ambiguity can be neither settled by a read nor + /// reissued: without that the engine reissues and the renewal commits, and this worker never fences. + backend->before_throw = [&, deadline = anchor + 1000] { boot_ms = deadline; }; runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); remount_entered.waitUntilArrived(); EXPECT_FALSE(runtime.mayMutate()); @@ -3526,21 +4470,34 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) { auto backend = std::make_shared(); - uint64_t fake_boot = 100; + /// Held in a shared atomic, not a plain local: this test mutates it directly below, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(100); PoolConfig config{ .pool_prefix = "direct-renew", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [&] { return fake_boot; }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, }; auto store = Pool::open(backend, config); - fake_boot = 500; + fake_boot->store(500); EXPECT_NO_THROW(store->renewWatermarkOnce()); - fake_boot = 1200; + fake_boot->store(1200); EXPECT_TRUE(store->mayMutate()) << "direct success must refresh the local fence from attempt start"; backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; + /// The renewal that succeeded at 500 anchored the lease for its 1000 ms TTL, so it expires at 1500. + /// Expire it from inside the attempt: the fault alone no longer ends a renewal, because the engine + /// settles the ambiguity by reading and reissues, and the reissue commits. + backend->before_throw = [fake_boot] + { + fake_boot->store(1500); + }; const uint64_t schedules_before = store->scheduleRemountCallCountForTest(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->renewWatermarkOnce(); }); EXPECT_FALSE(store->mayMutate()); @@ -3550,16 +4507,21 @@ TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) { auto backend = std::make_shared(); - std::vector events; + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "remount-observability", .server_root_id = "test", }); - store->setEventSink([&](CasEvent event) { events.push_back(std::move(event)); }); + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); ScopedRemountLogCapture logs; store->tripMountLost(); - backend->failNextGet(store->layout().poolMetaKey()); + backend->failNextRead(store->layout().poolMetaKey()); const uint64_t attempts_before = ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts].load(); const uint64_t succeeded_before = ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(); const uint64_t failed_before = ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(); @@ -3572,8 +4534,9 @@ TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(), succeeded_before + 1); EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(), failed_before + 1); + const std::vector observed_events = events->snapshot(); std::vector remounts; - std::copy_if(events.begin(), events.end(), std::back_inserter(remounts), [](const CasEvent & event) + std::copy_if(observed_events.begin(), observed_events.end(), std::back_inserter(remounts), [](const CasEvent & event) { return event.type == CasEventType::MountRemount; }); @@ -3588,6 +4551,75 @@ TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) EXPECT_EQ(countRemountFinalLogs(logs.captured()), 2u) << logs.captured(); } +/// The remount's `renewer_redo` step re-anchors the lease BEFORE `armMountFence`, so it runs with the +/// fence still latched lost. Admitted on the mount plane it could only ever give up, and every remount +/// that reached the step would fail -- so it renews on the renewer's open plane instead. +/// +/// Driven the way production reaches the step, which is the only way it CAN be reached: the persistent +/// renewal worker runs, `scheduleRemount` parks it, and the redo is the parked driver's one call. A +/// remount driven directly with no workers leaves that driver dormant, and the step's admission refuses +/// a dormant driver rather than renewing. +/// +/// The step is reached only when quiescence has eaten most of the new lease: with a fresh anchor the +/// renewal window fits and the step is skipped entirely. So the quiesce hook advances the injected boot +/// clock to just inside the safety margin, and the paired run with no quiesce cost is the control that +/// proves the step was reached rather than skipped. +TEST(CASPoolRemount, TheRenewerRedoRenewsOnTheOpenPlane) +{ + /// One successful self-remount whose quiescence costs `quiesce_ms`; returns the conditional + /// mount-slot writes it issued. Counted while the remount worker is still held inside the event + /// sink that reported the result, so the renewal worker it un-parks cannot add one. + const auto remountConditionalMountWrites = [](uint64_t quiesce_ms) -> uint64_t + { + auto backend = std::make_shared(); + /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can + /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so + /// a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); + /// Heap-owned, not a plain local: declaration order relative to `store` only protects against an + /// ordinary same-thread unwind, not a detached background completion that holds an extra + /// `shared_from_this()` and can still be running on another thread after this call returns. + auto committed = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{ + .pool_prefix = "remount-renewer-redo", + .server_root_id = "test", + .background_watermark = true, + .event_sink = [committed](const CasEvent & event) + { + if (event.type == CasEventType::MountRemount && event.outcome == "ok") + committed->arriveAndWait(); + }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }, + .remount_quiesce_hook_for_test = [fake_boot, quiesce_ms] + { + *fake_boot += quiesce_ms; + }, + }); + const String mount_key = store->layout().mountKey("test"); + + fenceOutMount(*backend, mount_key); + const uint64_t before = backend->putOverwriteCount(mount_key); + EXPECT_TRUE(store->scheduleRemountForTest()) + << "the remount must be latched with quiesce_ms=" << quiesce_ms; + committed->waitUntilArrived(); + const uint64_t writes = backend->putOverwriteCount(mount_key) - before; + committed->release(); + return writes; + }; + + /// 27 s of a 30 s lease, against a 2 s safety margin and a window of one renewal period plus one + /// attempt (15 s, since the renewal worker runs here): the window no longer fits. + EXPECT_GT(remountConditionalMountWrites(27'000), remountConditionalMountWrites(0)) + << "a quiescence that consumed the lease must cost one extra lease write -- the redo"; +} + TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) { auto backend = std::make_shared(); @@ -3599,7 +4631,7 @@ TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) store->tripMountLost(); store->tripMountLost(); - backend->failNextGet(store->layout().poolMetaKey()); + backend->failNextRead(store->layout().poolMetaKey()); EXPECT_FALSE(store->tryRemountOnce()); store->beginShutdownForTest(); store->tripMountLost(); @@ -3644,3 +4676,59 @@ TEST(CASPool, MountpointObjectRoundTrip) EXPECT_FALSE(store->getMountpointObject(key).has_value()); EXPECT_FALSE(store->mountpointObjectExists(key)); } + +namespace ProfileEvents +{ + extern const Event CASHotKeyReadStarts; + extern const Event CASRequestResolveRead; + extern const Event CASRequestConflictPause; +} + +TEST(CASPool, ConcurrentNamespaceCreationsNeverRaceEachOtherOnTheCatalog) +{ + auto backend = std::make_shared(); + auto pool = DB::Cas::tests::openPoolForTest(backend); + const DB::Cas::Layout layout("p"); + const String key = layout.refCatalogKey(); + constexpr int N = 6; + + /// Drain whatever the pool's own bootstrap touched on the catalog key before measuring. + (void)pool->namespaceLife(DB::Cas::RootNamespace{"warmup"}); + + const uint64_t writes_before = backend->writeCount(key); + const auto reads_before = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load(); + const auto resolves_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); + std::vector threads; + for (int i = 0; i < N; ++i) + threads.emplace_back([&, i] { (void)pool->namespaceLife(DB::Cas::RootNamespace{"ns" + std::to_string(i)}); }); + for (auto & t : threads) + t.join(); + + EXPECT_EQ(backend->writeCount(key) - writes_before, 2u * N) << "two catalog steps per creation, each one write"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolves_before, 0u) + << "no refused precondition, so no resolve read"; + EXPECT_LE(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load() - reads_before, 1u) + << "at most one lane read; every later hold started from the cache"; + + /// Another server writes the catalog between two of this pool's mutations: one extra read and one + /// retry write, then the cache is current again. Raw `getCount` cannot isolate that cost: a + /// `namespaceLife` call on a fresh namespace also issues the ledger's own snapshot reads + /// (`CasRefCatalog::read`), which are outside the lane by design and fire the same number of times + /// whether or not an external write happened. The lane's own signals are what the external write + /// actually moves. + { + auto external_requests = DB::Cas::tests::openRequestsForTest(backend); + auto external = external_requests.admit(); + DB::Cas::CasRefCatalog::casAdmitEntry(external, layout, 1, + DB::Cas::CatalogEntry{.ns = DB::Cas::RootNamespace{"zz"}, .state = DB::Cas::NsState::Live, .incarnation = UInt128{99}}); + } + const uint64_t writes_mid = backend->writeCount(key); + const auto resolves_mid = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); + const auto lane_reads_mid = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load(); + (void)pool->namespaceLife(DB::Cas::RootNamespace{"after"}); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolves_mid, 1u) + << "one resolve read for the external write"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load() - lane_reads_mid, 0u) + << "the next hold starts from what the resolve read saw"; + EXPECT_EQ(backend->writeCount(key) - writes_mid, 3u) << "one refused, two landed"; +} diff --git a/src/Disks/tests/gtest_cas_pool_meta.cpp b/src/Disks/tests/gtest_cas_pool_meta.cpp new file mode 100644 index 000000000000..f82a5f8e6091 --- /dev/null +++ b/src/Disks/tests/gtest_cas_pool_meta.cpp @@ -0,0 +1,78 @@ +#include + +#include +#include +#include "cas_test_helpers.h" + +#include +#include + +namespace DB +{ +namespace ErrorCodes +{ + extern const int NETWORK_ERROR; +} +} + +using namespace DB::Cas; + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +CasRequests makeRequests(BackendPtr backend, FakeClock & clock, Fence fence = Fence::open()) +{ + return CasRequests(std::move(backend), std::move(fence), clock.nowFn(), clock.sleepFn()); +} + +} + +TEST(CASPoolMeta, AdmitOrValidateEndsAtTheDeadlineUnderPerpetualConflict) +{ + FakeClock clock; + auto backend = std::make_shared(); + Layout layout("pool"); + auto requests = makeRequests(backend, clock); + + auto create_op = requests.admit(); + const PoolMeta created = PoolMeta::createOrValidate( + create_op, layout, /*blob_header_len=*/256, /*gc_shards=*/1, + BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/true); + EXPECT_EQ(created.algos_used, (std::vector{static_cast(BlobHashAlgo::CityHash128)})); + + /// Rig the key permanently hot: every write attempt races a concurrent rewrite of the SAME + /// content, so the object's incarnation moves under every attempt and admission of a new algo + /// never lands. `putOverwrite` mints a fresh incarnation even though the bytes are unchanged. + const String key = layout.poolMetaKey(); + EXPECT_TRUE(clock.sleeps.empty()); /// nothing paced yet -- the trailing check below is about THIS call + bool inside_hook = false; + backend->onBeforeWrite(key, [&] + { + if (inside_hook) + return; + inside_hook = true; + auto hook_requests = DB::Cas::tests::openRequestsForTest(BackendPtr(backend)); + auto hook_op = hook_requests.admit(); + if (auto cur = hook_op.read(key, Retry::once())) + (void)hook_op.replace(key, cur->bytes, cur->etag, Retry::once()); + inside_hook = false; + }); + + /// `orThrow`'s `GaveUp{Deadline}` arm throws exactly `NETWORK_ERROR` (`throwCasWriteRetryLater`), + /// pinning the deadline outcome apart from the two failures a wrong migration could also throw as + /// a `DB::Exception` here: `LOGICAL_ERROR` (the absence branch) or `BAD_ARGUMENTS` (`allow_new` + /// plumbing regressed). + auto admit_op = requests.admit(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + (void)PoolMeta::createOrValidate( + admit_op, layout, 256, /*gc_shards=*/1, + BlobHashAlgo::XXH3_128, /*allow_new=*/true, /*allow_mint=*/false); + }); + /// Bounded by the deadline, not a live-lock: it paced its retries rather than spinning. + EXPECT_FALSE(clock.sleeps.empty()); +} diff --git a/src/Disks/tests/gtest_cas_probe.cpp b/src/Disks/tests/gtest_cas_probe.cpp index 8c2beb7d4503..6b9da190dcad 100644 --- a/src/Disks/tests/gtest_cas_probe.cpp +++ b/src/Disks/tests/gtest_cas_probe.cpp @@ -1,87 +1,83 @@ #include +#include #include -#include #include #include +#include #include #include #include -namespace DB -{ -namespace ErrorCodes -{ - extern const int NOT_IMPLEMENTED; -} -} - using namespace DB::Cas; -TEST(CASProbe, PassesOnEnforcingBackend) +namespace { - auto b = std::make_shared(); - EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); - EXPECT_TRUE(b->list("p/.cas_probe", "", 10).keys.empty()); // probe cleans up after itself -} -/// AWS S3 answers 400 InvalidArgument to a conditional DELETE with an EMPTY If-Match, and the -/// probe's exit cleanup used to issue exactly that (deleteExact with the absent HeadResult's empty -/// token) after step 8 had already deleted the probe keys — two scary AWSClient log lines -/// on every real-S3 mount. The cleanup must HEAD-gate the delete instead of firing blindly. -class EmptyTokenDeleteRecorder : public InMemoryBackend +/// Every test here constructs one backend and runs the battery against it once; a non-owning +/// `BackendPtr` over the test's stack- or shared_ptr-held backend keeps that construction pattern +/// rather than forcing a second allocation. The open fence never trips: `runCapabilityProbe` runs +/// during pool bootstrap, before any mount fence exists to enforce. +CasRequests makeRequests(Backend & backend) { -public: - size_t empty_token_deletes = 0; + return CasRequests(BackendPtr(&backend, [](Backend *) {}), Fence::open()); +} - DeleteOutcome deleteExact(const String & key, const Token & token) override - { - if (token.empty()) - ++empty_token_deletes; - return InMemoryBackend::deleteExact(key, token); - } -}; +} -TEST(CASProbe, CleanupNeverDeletesWithEmptyToken) +TEST(CASProbe, PassesOnEnforcingBackend) { - auto b = std::make_shared(); - EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); - EXPECT_EQ(b->empty_token_deletes, 0u); + auto b = std::make_shared(); + auto requests = makeRequests(*b); + auto op = requests.admit(); + EXPECT_NO_THROW(runCapabilityProbe(op, "p/.cas_probe")); + EXPECT_TRUE(op.list("p/.cas_probe", "", 10, Retry::once()).keys.empty()); // probe cleans up after itself } TEST(CASProbe, FailsClosedOnNonEnforcingDelete) { auto b = std::make_shared(); b->setEnforceTokens(false); // the MinIO-OSS failure mode - EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); + auto requests = makeRequests(*b); + auto op = requests.admit(); + EXPECT_THROW(runCapabilityProbe(op, "p/.cas_probe"), DB::Exception); } TEST(CASProbe, FailsClosedOnDeleteMarkers) { auto b = std::make_shared(); b->setSimulateDeleteMarkers(true); // versioning enabled on the prefix - EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); + auto requests = makeRequests(*b); + auto op = requests.admit(); + EXPECT_THROW(runCapabilityProbe(op, "p/.cas_probe"), DB::Exception); } TEST(CASProbe, PassesOnEmulatedLocal) { auto b = std::make_shared( DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::EmulatedSingleProcess); - EXPECT_NO_THROW(runCapabilityProbe(*b, "p/.cas_probe")); + auto requests = makeRequests(*b); + auto op = requests.admit(); + EXPECT_NO_THROW(runCapabilityProbe(op, "p/.cas_probe")); } -/// B135: two servers mounting the SAME shared CA pool concurrently must not race on the probe keys. +/// Two servers mounting the SAME shared CA pool concurrently must not race on the probe keys. /// We simulate "a concurrent mounter's probe is in flight" by PRE-SEEDING the fixed-name probe key -/// `/_probe/token` over a shared backend, then opening the Pool. With the OLD fixed-key probe -/// the open's `putIfAbsent("/_probe/token", …)` returns PreconditionFailed and `Pool::open` -/// throws NOT_IMPLEMENTED ("putIfAbsent on a fresh key returned PreconditionFailed"). With the +/// `/_probe/token` over a shared backend, then opening the Pool. A fixed-key probe would meet +/// the seeded object as a refused precondition on its own `create` and fail the open; with the /// per-mount unique probe prefix `/_probe//token`, the seeded key does not collide and /// the open succeeds — exactly the concurrent-shared-pool-mount behaviour we need. +/// +/// Goes through `Pool::open` (owned elsewhere), so it exercises `runCapabilityProbe` only indirectly +/// and needs no signature change here. TEST(CASProbe, ConcurrentMountsDoNotCollide) { auto b = std::make_shared(); + auto probe_requests = makeRequests(*b); + auto probe_op = probe_requests.admit(); /// Simulate a concurrent mounter whose probe object under the legacy fixed key is still present. - ASSERT_EQ(b->putIfAbsent("p/_probe/token", "concurrent-mounter-in-flight").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + probe_op.create("p/_probe/token", "concurrent-mounter-in-flight", Retry::once()))); /// A real (second) mount over the same shared pool must still succeed — its probe runs under a /// fresh per-mount-unique prefix and never touches the seeded fixed key. @@ -91,46 +87,14 @@ TEST(CASProbe, ConcurrentMountsDoNotCollide) EXPECT_NO_THROW(Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); /// The seeded fixed-key artifact is untouched (the probe never collided with it). - EXPECT_TRUE(b->get("p/_probe/token").has_value()); -} - -/// The probe must consult the backend's store-preconditions hook BEFORE the op battery: a -/// generation-dialect store on a VERSIONED bucket passes every conditional-op check, but its -/// token-exact DELETEs archive noncurrent generations instead of reclaiming storage — only the -/// hook can see that, so a throwing hook must fail the probe closed. -class PreconditionRefusingBackend : public InMemoryBackend -{ -public: - void checkPoolPreconditions() override - { - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, - "test: store precondition violated (e.g. bucket versioning enabled)"); - } -}; - -TEST(CASProbe, FailsClosedOnPoolPreconditions) -{ - auto b = std::make_shared(); - EXPECT_THROW(runCapabilityProbe(*b, "p/.cas_probe"), DB::Exception); - /// The hook fires FIRST: no probe keys may have been written. - EXPECT_TRUE(b->list("p/.cas_probe", "", 10).keys.empty()); -} - -/// `Pool::open` wraps the pool backend in `InstrumentedBackend` BEFORE calling `runCapabilityProbe` -/// (see CasPool.cpp), so the hook must actually fire THROUGH the wrapper on the real mount path — -/// not just on a raw backend, which `FailsClosedOnPoolPreconditions` above already covers. -TEST(CASProbe, PoolPreconditionsFireThroughInstrumentedWrapper) -{ - auto inner = std::make_shared(); - InstrumentedBackend wrapped(inner); - EXPECT_THROW(runCapabilityProbe(wrapped, "p/.cas_probe"), DB::Exception); - /// The hook fires FIRST: no probe keys may have been written to the inner backend. - EXPECT_TRUE(inner->list("p/.cas_probe", "", 10).keys.empty()); + EXPECT_TRUE(probe_op.read("p/_probe/token", Retry::once()).has_value()); } /// RFC cas-s3-timeout-retry-control: a Native-mode mount over an object storage that does not support /// the SingleAttempt retry profile must never silently proceed under the disk's default (~500-attempt) -/// transparent retry policy — see Backend::checkConditionalWriteSingleAttemptSupport. +/// transparent retry policy — see Backend::checkConditionalWriteSingleAttemptSupport. This calls the +/// hook directly on the backend (not through `runCapabilityProbe`, which no longer runs it — see +/// CasProbe.h), so it is unaffected by the request-contract migration. /// LocalObjectStorage never supports the profile (IObjectStorage::supportsRetryProfile's default /// implementation only answers true for Default), so Native mode over it is exactly the case this must /// refuse. EmulatedSingleProcess is exempt: it never claims single-attempt S3 semantics in the first @@ -146,62 +110,34 @@ TEST(CASProbe, FailsClosedOnUnsupportedSingleAttemptProfile) EXPECT_NO_THROW(emulated->checkConditionalWriteSingleAttemptSupport()); } -/// The same fail-closed refusal through the actual capability probe (Step 0b) — the real gate a -/// writable Pool::open goes through, not just the hook in isolation above. -TEST(CASProbe, MissingSingleAttemptClientFailsCapabilityProbe) -{ - auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); - /// Native mode passes the key to the object storage verbatim, so the probe prefix must be anchored - /// under this storage's own root: a bare prefix lands beside the test process, where an object left - /// by another run answers the LIST below and an unrooted LIST answers "no keys" for free. - const String probe_prefix = DB::Cas::tests::nativeKeyUnder(storage, "p/.cas_probe"); - - auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); - EXPECT_THROW(runCapabilityProbe(*b, probe_prefix), DB::Exception); - /// The hook fires before the op battery: no probe keys may have been written. - EXPECT_TRUE(b->list(probe_prefix, "", 10).keys.empty()); - - /// The same LIST can see a key that IS under the prefix — otherwise the emptiness above would be - /// indistinguishable from a prefix this backend can never enumerate. - ASSERT_EQ(b->putIfAbsent(probe_prefix + "/token", "probe-v1").outcome, PutOutcome::Done); - EXPECT_FALSE(b->list(probe_prefix, "", 10).keys.empty()); -} - -/// Mirrors PoolPreconditionsFireThroughInstrumentedWrapper: the real mount path wraps the backend in -/// InstrumentedBackend BEFORE calling runCapabilityProbe, so this check must fire through it too. -TEST(CASProbe, MissingSingleAttemptClientFiresThroughInstrumentedWrapper) -{ - auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); - auto inner = std::make_shared(storage, ObjectStorageBackend::Mode::Native); - InstrumentedBackend wrapped(inner); - EXPECT_THROW(runCapabilityProbe(wrapped, DB::Cas::tests::nativeKeyUnder(storage, "p/.cas_probe")), DB::Exception); -} - namespace { -/// Honors every conditional WRITE but ignores the token on a token-exact DELETE. This is what a GCS -/// delete degenerates to when its numeric generation leaves as a raw `If-Match` — no -/// `x-goog-if-generation-match` — and the service ignores the header it does not recognise. +/// Honors every conditional WRITE but ignores the precondition on a conditional REMOVE. This is what a +/// GCS delete degenerates to when its numeric generation leaves as a raw `If-Match` — no +/// `x-goog-if-generation-match` — and the service ignores the header it does not recognise. Gated on +/// the PRIMITIVE (`Backend::remove`), which is what `CasOperation::remove` actually calls; a fault +/// injected on the legacy `deleteExact` forwarder would no longer intercept anything. class IgnoresDeleteTokenBackend : public InMemoryBackend { public: - DeleteOutcome deleteExact(const String & key, const Token &) override + RawRemoval remove(const String & key, const String & /*expected_value*/, TransportAccess & access) override { - return InMemoryBackend::deleteExact(key, head(key).token); + const auto meta = InMemoryBackend::head(key, access); + if (!meta) + return RawRemoval::Gone; + return InMemoryBackend::remove(key, meta->value, access); } }; /// The other half of that degeneracy: the service refuses the unrecognised header outright, so even -/// the correct token never removes anything. +/// the correct incarnation never removes anything. class RejectsDeleteTokenBackend : public InMemoryBackend { public: - DeleteOutcome deleteExact(const String &, const Token &) override + RawRemoval remove(const String &, const String &, TransportAccess &) override { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; - return d; + return RawRemoval::Mismatch; } }; @@ -209,152 +145,87 @@ class RejectsDeleteTokenBackend : public InMemoryBackend /// A GCS mount whose exact deletes lost their generation semantics can fail in either direction, and /// the probe's delete battery must reject the mount both times. Both backends enforce every -/// conditional write, so every step before the battery passes and only step 6's wrong-token -/// preservation check and step 8's correct-token deletion check can be what fires — -/// `PassesOnEnforcingBackend` above is the control showing the same probe succeeds when only -/// `deleteExact` is left alone. +/// conditional write, so every step before the battery's delete checks passes and only the +/// stale-incarnation-preserved check or the correct-incarnation-removed check can be what fires — +/// `PassesOnEnforcingBackend` above is the control showing the same probe succeeds when only `remove` +/// is left alone. /// /// This is about the battery, not about the marking: that the `NativeConditional` mode actually /// reaches the production request object is proven where the request is built, not here. TEST(CASProbe, ExactDeleteBatteryDetectsMissingGenerationMode) { IgnoresDeleteTokenBackend ignores; - EXPECT_THROW(runCapabilityProbe(ignores, "p/.cas_probe"), DB::Exception); + auto ignores_requests = makeRequests(ignores); + auto ignores_op = ignores_requests.admit(); + EXPECT_THROW(runCapabilityProbe(ignores_op, "p/.cas_probe"), DB::Exception); RejectsDeleteTokenBackend rejects; - EXPECT_THROW(runCapabilityProbe(rejects, "p/.cas_probe"), DB::Exception); + auto rejects_requests = makeRequests(rejects); + auto rejects_op = rejects_requests.admit(); + EXPECT_THROW(runCapabilityProbe(rejects_op, "p/.cas_probe"), DB::Exception); } namespace { -/// Models the exact shape of the trust-flip this suite must catch a regression of -/// (codex-review-triage §3.18, Critical): like the production `ObjectStorageBackend` in Native mode, -/// this backend mints and expects tokens under a dialect (`TokenType::ETag`) OTHER than -/// `TokenType::Emulated`, and rejects a foreign-dialect `expected`/`token` argument LOCALLY -- -/// before the value it carries ever reaches the real conditional-compare beneath the gate (`inner`, -/// a genuinely enforcing `InMemoryBackend`, standing in for "the wire"). Every gated method counts -/// how many times it actually delegated to `inner`, so a test can tell "rejected by the dialect -/// gate" apart from "rejected by the real enforcement" -- the exact distinction `Cas::Probe` exists -/// to prove, and the one the №19 hardening risked collapsing (see CasProbe.cpp step 3/5c/6). -class DialectGatedCountingBackend final : public Backend +/// Rejects, LOCALLY and without touching the store, any conditional write/remove whose precondition +/// value is not grammar-valid under the claimed dialect — the production shape of the retired +/// `DialectGatedCountingBackend`, ported to the primitive interface (`write`/`remove` carry a raw +/// precondition VALUE now, not a typed token, so the gate is `isIncarnationValue` rather than a type-tag +/// compare). `write_reached`/`remove_reached` count only the calls that got PAST the gate, so a +/// regression that reintroduces a synthesized (grammar-invalid-somewhere) precondition drops one of +/// these counts instead of passing silently. +class DialectOverrideBackend : public InMemoryBackend { public: - std::optional get(const String & key, Range range) override { return inner.get(key, range); } - - std::optional getStream(const String & key, Range range) override { return inner.getStream(key, range); } + explicit DialectOverrideBackend(Dialect claimed_dialect_) : claimed_dialect(claimed_dialect_) {} + Dialect dialect() const override { return claimed_dialect; } - HeadResult head(const String & key) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, TransportAccess & access) override { - HeadResult r = inner.head(key); - if (r.exists) - r.token.type = TokenType::ETag; - return r; + if (expected_value && !isIncarnationValue(claimed_dialect, *expected_value)) + return std::unexpected(RawConflict{}); /// dialect-gated: never reaches the real store + ++write_reached; + return InMemoryBackend::write(key, bytes, expected_value, access); } - bool supportsListTokens() const override { return inner.supportsListTokens(); } - - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override { - /// No `expected` token to gate -- matches production (ObjectStorageBackend::putIfAbsent has - /// no dialect check either). - PutResult r = inner.putIfAbsent(key, bytes, meta); - if (r.outcome == PutOutcome::Done) - r.token.type = TokenType::ETag; - return r; + if (!isIncarnationValue(claimed_dialect, expected_value)) + return RawRemoval::Mismatch; /// dialect-gated: never reaches the real store + ++remove_reached; + return InMemoryBackend::remove(key, expected_value, access); } - void publishBlob(const BlobPublishRequest & request) override - { - inner.publishBlob(request); - } - - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override - { - if (expected.type != TokenType::ETag) - return {PutOutcome::PreconditionFailed, {}}; /// dialect-gated: never reaches `inner` - ++overwrite_reached; - PutResult r = inner.putOverwrite(key, bytes, Token{expected.value, TokenType::Emulated}, meta); - if (r.outcome == PutOutcome::Done) - r.token.type = TokenType::ETag; - return r; - } - - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, const ObjectMeta & meta) override - { - if (expected.has_value() && expected->type != TokenType::ETag) - return {CasOutcome::Conflict, {}}; /// dialect-gated: never reaches `inner` - ++casput_reached; - std::optional retyped; - if (expected.has_value()) - retyped = Token{expected->value, TokenType::Emulated}; - CasResult r = inner.casPut(key, bytes, retyped, meta); - if (r.outcome == CasOutcome::Committed) - r.token.type = TokenType::ETag; - return r; - } - - DeleteOutcome deleteExact(const String & key, const Token & token) override - { - if (token.type != TokenType::ETag) - { - DeleteOutcome d; - d.kind = DeleteOutcome::Kind::TokenMismatch; /// dialect-gated: never reaches `inner` - return d; - } - ++delete_reached; - return inner.deleteExact(key, Token{token.value, TokenType::Emulated}); - } - - ListPage list(const String & prefix, const String & cursor, size_t limit) override - { - ListPage p = inner.list(prefix, cursor, limit); - for (auto & k : p.keys) - if (k.token) - k.token->type = TokenType::ETag; - return p; - } - - /// Number of times putOverwrite/casPut(with expected)/deleteExact actually delegated to `inner` - /// (i.e. reached the real enforcement) rather than being short-circuited by the dialect gate. - int overwrite_reached = 0; - int casput_reached = 0; - int delete_reached = 0; + int write_reached = 0; + int remove_reached = 0; private: - InMemoryBackend inner; + Dialect claimed_dialect; }; } -/// codex-review-triage §3.18, Critical: `runCapabilityProbe`'s three wrong-token sites (step 3 -/// putOverwrite, step 5c casPut, step 6 deleteExact) must send a token in the LIVE dialect this -/// backend mints (t1.type / ct1.type / t2.type), not a hardcoded `TokenType::Emulated`. A backend -/// whose native dialect differs from Emulated -- exactly what `ObjectStorageBackend` mints in Native -/// mode -- would otherwise reject the old hardcoded tokens LOCALLY via a dialect gate, never -/// exercising the real conditional enforcement those three steps exist to validate; the probe would -/// still report success (the outcome enums match either way), so a regression here is invisible -/// unless something counts whether the real enforcement was ever reached. `DialectGatedCountingBackend` -/// enforces real (correct) conditional semantics AND gates on dialect exactly like the production -/// risk, so `runCapabilityProbe` runs to completion (unlike a real Native-mode ObjectStorageBackend -/// over LocalObjectStorage, which cannot even reach this point -- see -/// MissingSingleAttemptClientFailsCapabilityProbe and the fact that LocalObjectStorage does not honor -/// WriteSettings conditions at all); the exact reached-counts below pin down that every wrong-token -/// site got past the gate: a probe that regressed to the hardcoded-Emulated construction would still -/// pass (no throw) but under-count here by exactly one at each of the three sites, since the dialect -/// gate would swallow that one call before `inner` ever saw it. -TEST(CASProbe, WrongTokenAttemptsReachTheBackendPastTheDialectGate) +/// The probe's reordered "wrong incarnation" steps always reuse a REAL, backend-minted `Etag` +/// from the same key rather than a synthesized value — see CasProbe.cpp's step comments — and every +/// such value is grammar-valid under every dialect by construction (`InMemoryBackend`'s minted values are +/// a monotonically increasing decimal starting at "1": non-empty and comma/`*`-free for ETag, a canonical +/// positive decimal for Generation, merely non-empty for Emulated). So under EVERY dialect the battery's +/// four conditional writes (steps 1-4) and two conditional removes (steps 5, 7) must all reach the real +/// store — asserting the exact counts is what makes this test able to fail: a regression that +/// reintroduces a synthesized, foreign-dialect precondition would get gated locally on at least one +/// dialect, dropping one of these counts below the total instead of merely changing an outcome enum. +TEST(CASProbe, ReorderedProbePassesOnAllThreeDialects) { - DialectGatedCountingBackend b; - EXPECT_NO_THROW(runCapabilityProbe(b, "p/.cas_probe")); - - /// putOverwrite: step 3 (wrong token) + step 4 (correct token) -- both live-dialect, both gated - /// through to `inner`. - EXPECT_EQ(b.overwrite_reached, 2); - /// casPut: 5a (create), 5b (conflict-on-exists, no expected token to gate), 5c (wrong token, - /// live-dialect), 5d (correct token) -- all four reach `inner`. - EXPECT_EQ(b.casput_reached, 4); - /// deleteExact: step 6 (wrong token, live-dialect) + step 8 (correct token) + step 9 cleanup - /// (correct token for cas_key) -- all three reach `inner`. - EXPECT_EQ(b.delete_reached, 3); + for (const Dialect dialect : {Dialect::ETag, Dialect::Generation, Dialect::Emulated}) + { + DialectOverrideBackend b(dialect); + auto requests = makeRequests(b); + auto op = requests.admit(); + EXPECT_NO_THROW(runCapabilityProbe(op, "p/.cas_probe")) << "dialect " << static_cast(dialect); + EXPECT_TRUE(op.list("p/.cas_probe", "", 10, Retry::once()).keys.empty()) << "dialect " << static_cast(dialect); + EXPECT_EQ(b.write_reached, 4) << "dialect " << static_cast(dialect); + EXPECT_EQ(b.remove_reached, 2) << "dialect " << static_cast(dialect); + } } diff --git a/src/Disks/tests/gtest_cas_protocol_scenarios.cpp b/src/Disks/tests/gtest_cas_protocol_scenarios.cpp index d70cca40a8ea..f33018415c0a 100644 --- a/src/Disks/tests/gtest_cas_protocol_scenarios.cpp +++ b/src/Disks/tests/gtest_cas_protocol_scenarios.cpp @@ -35,6 +35,7 @@ namespace DB::ErrorCodes { extern const int ABORTED; +extern const int CORRUPTED_DATA; extern const int FILE_DOESNT_EXIST; extern const int LOGICAL_ERROR; } @@ -58,6 +59,25 @@ PoolPtr openPool(const std::shared_ptr & b) return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); } +/// The object's incarnation as the store reports it now -- what these scenarios compare when they +/// assert an object was, or was not, displaced. +Etag currentIncarnation(Backend & b, const String & key) +{ + DB::Cas::tests::OperationForTest operation(b); + const std::optional meta = (*operation).head(key, Retry::standard()); + if (!meta) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "object {} is absent", key); + return meta->etag; +} + +/// An exact-incarnation delete attempt, for the scenarios whose discriminator is that a displaced +/// incarnation can never be current again. +Removal removeAtIncarnation(Backend & b, const String & key, const Etag & seen) +{ + DB::Cas::tests::OperationForTest operation(b); + return (*operation).remove(key, seen, Retry::standard()); +} + /// A single-blob manifest entry naming `payload` at `path` (the entry the part's manifest carries). ManifestEntry blobEntry(const String & path, const String & payload) { @@ -120,9 +140,10 @@ void assertPartReads( const auto * entry = findEntry(manifest.entries, path); ASSERT_TRUE(entry != nullptr); auto loc = s->locate(*entry); - auto got = b->get(loc.key, Range{loc.offset, loc.length}); + DB::Cas::tests::OperationForTest op(*b); + auto got = (*op).read(loc.key, Retry::once()); ASSERT_TRUE(got.has_value()); - EXPECT_EQ(got->bytes, payload); + EXPECT_EQ(got->bytes.substr(static_cast(loc.offset), static_cast(loc.length)), payload); } } @@ -145,11 +166,11 @@ TEST(CASProtocol, FenceConflictCondemnedTokenedBlobCommitsWithTokenUnchanged) build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); /// GC condemns X at t0 in round 1 and fences the namespace to round 1. injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, - {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = PersistedEtag::capture(t0), .size = 9}}); /// promote: mutateShard refreshes the view (fence_round 1 > view round 0), but the materialized leaf is /// edge-protected — skipped, not re-validated ⇒ commit, token unchanged. @@ -157,7 +178,7 @@ TEST(CASProtocol, FenceConflictCondemnedTokenedBlobCommitsWithTokenUnchanged) /// The ref is committed and reads back; the blob still rides t0 (no re-upload). assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); - EXPECT_EQ(b->head(blob_key).token, t0); + EXPECT_EQ(currentIncarnation(*b, blob_key), t0); } TEST(CASProtocol, RevalidateReObservesStaleTokenKeepsWhenUnchanged) @@ -172,7 +193,7 @@ TEST(CASProtocol, RevalidateReObservesStaleTokenKeepsWhenUnchanged) /// X pre-exists out-of-band; the build dedup-adopts it via putBlob (records the current token t0). writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); /// Wiring order: stage + precommit (durable edge) BEFORE the adopting putBlob. auto build = startBuildFor(s, ns, "part_1"); @@ -189,7 +210,7 @@ TEST(CASProtocol, RevalidateReObservesStaleTokenKeepsWhenUnchanged) assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); /// No rewrite happened — the materialized leaf was never touched, so its object token stays at t0. - EXPECT_EQ(b->head(blob_key).token, t0); + EXPECT_EQ(currentIncarnation(*b, blob_key), t0); } TEST(CASProtocol, RevalidateReObservesStaleTokenAdoptsWhenDisplaced) @@ -204,7 +225,7 @@ TEST(CASProtocol, RevalidateReObservesStaleTokenAdoptsWhenDisplaced) writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); auto build = startBuildFor(s, ns, "part_1"); /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob. @@ -213,7 +234,7 @@ TEST(CASProtocol, RevalidateReObservesStaleTokenAdoptsWhenDisplaced) build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 /// Another writer displaces X out-of-band ⇒ a new current token t1 (same payload, fresh tag). - const Token t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); + const Etag t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); EXPECT_NE(t1, t0); /// GC advanced to round 1 with an EMPTY retired set; fence to 1. @@ -222,17 +243,17 @@ TEST(CASProtocol, RevalidateReObservesStaleTokenAdoptsWhenDisplaced) /// promote refreshes ⇒ revalidate X ⇒ HEAD current t1 not condemned ⇒ commit. The dep rides t1. build->promote(ns, "part_1", build->buildId(), id); assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); - EXPECT_EQ(b->head(blob_key).token, t1); + EXPECT_EQ(currentIncarnation(*b, blob_key), t1); /// Black-box proof the part reads the t1 incarnation: re-publish the same blob into a SECOND /// namespace with NO new GC injection. The blob is already present at t1; nothing is re-uploaded. publishBlobPart(s, RootNamespace{"srv1/tbl/copy"}, "part_2", "data.bin", "payload-X"); - EXPECT_EQ(b->head(blob_key).token, t1); + EXPECT_EQ(currentIncarnation(*b, blob_key), t1); assertPartReads(b, s, RootNamespace{"srv1/tbl/copy"}, "part_2", "data.bin", "payload-X"); /// Independent discriminator that the blob rides t1, not the stale t0: t0 is DEAD. A deleteExact /// against t0 must TokenMismatch (INV-NO-RETURN — t0 was displaced and can never be current again). - EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_EQ(removeAtIncarnation(*b, blob_key, t0), Removal::Mismatch); } TEST(CASProtocol, RevalidateAdoptsLiveTokenWhenOnlyPhantomCondemnedAtDifferentToken) @@ -250,9 +271,9 @@ TEST(CASProtocol, RevalidateAdoptsLiveTokenWhenOnlyPhantomCondemnedAtDifferentTo writeBlobRaw(*b, s0->layout(), "payload-X", s0->poolMeta().blob_header_len, s0->poolMeta().pool_id); } const String blob_key = layout.blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; - const Token t_other{"emulated-phantom", DB::Cas::TokenType::Emulated}; - ASSERT_NE(t_other, t0); + const Etag t0 = currentIncarnation(*b, blob_key); + const PersistedEtag t_other{"emulated", "emulated-phantom"}; + ASSERT_FALSE(t_other.matches(t0)) << "the phantom must name a DIFFERENT incarnation than the live one"; injectRetire(*b, layout, /*round*/ 1, /*shard*/ 0, {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t_other, .size = 9}}); @@ -272,7 +293,7 @@ TEST(CASProtocol, RevalidateAdoptsLiveTokenWhenOnlyPhantomCondemnedAtDifferentTo assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); /// The object was NOT displaced — it STAYS at t0 (no re-upload, only re-validated). - EXPECT_EQ(b->head(blob_key).token, t0); + EXPECT_EQ(currentIncarnation(*b, blob_key), t0); } /// (DELETED, Phase A) RevalidateAbsentTokenedBlobResurrectsFromSource — see the file-header note: a @@ -298,7 +319,7 @@ TEST(CASProtocol, EvidenceHitCondemnedPresentBlobCopiesForwardInClosure) const BlobRef seeded_ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(hex))}; seedBlobWithDurablePrecommit(s, seeded_ref, "payload-X"); const String blob_key = s->layout().blobKey(seeded_ref); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); auto build = startBuildFor(s, ns, "part_1"); ManifestEntry entry = blobEntry("data.bin", "payload-X"); @@ -315,7 +336,7 @@ TEST(CASProtocol, EvidenceHitCondemnedPresentBlobCopiesForwardInClosure) /// The ref stands; X rides its ORIGINAL token t0 (trust never displaces a trusted leaf). EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); - EXPECT_EQ(b->head(blob_key).token, t0) << "trust must not displace the adopted blob"; + EXPECT_EQ(currentIncarnation(*b, blob_key), t0) << "trust must not displace the adopted blob"; /// The meta is untouched — still Condemned (the gate never reads or flips it under trust). const auto lm_after = loadMetaForTest(*b, s->layout(), hexToU128(hex)); @@ -343,16 +364,16 @@ TEST(CASProtocol, WedgedHeartbeatCondemnedTokenedBlobCommitsWithTokenUnchanged) build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); /// Full GC condemned the build's OWN upload. injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, - {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = PersistedEtag::capture(t0), .size = 9}}); /// promote: the materialized leaf is edge-protected — skipped, not revalidated ⇒ commit, token unchanged. build->promote(ns, "part_1", build->buildId(), id); assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); - EXPECT_EQ(b->head(blob_key).token, t0); + EXPECT_EQ(currentIncarnation(*b, blob_key), t0); } TEST(CASProtocol, AbandonLeavesDebrisAndDisables) @@ -372,8 +393,11 @@ TEST(CASProtocol, AbandonLeavesDebrisAndDisables) /// Both bodies remain as debris: once the manifest has named a durable precommit edge, its body /// must survive until GC folds the matching owner removal. - EXPECT_TRUE(b->head(s->layout().blobKey(blob.ref)).exists); - EXPECT_TRUE(b->head(s->layout().manifestKey(id)).exists); + { + DB::Cas::tests::OperationForTest op(*b); + EXPECT_TRUE((*op).head(s->layout().blobKey(blob.ref), Retry::once()).has_value()); + EXPECT_TRUE((*op).head(s->layout().manifestKey(id), Retry::once()).has_value()); + } EXPECT_TRUE(s->listRefs(ns).empty()); /// Further build ops ⇒ LOGICAL_ERROR (requireAlive). @@ -398,7 +422,7 @@ TEST(CASProtocol, DropReattachThroughDetachedNamespace) publishBlobPart(s, ns, "part_1", "data.bin", "payload-X"); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token blob_tok = b->head(blob_key).token; + const Etag blob_tok = currentIncarnation(*b, blob_key); EXPECT_TRUE(s->listRefs(ns).contains("part_1")); EXPECT_TRUE(s->listRefs(detached).empty()); @@ -420,7 +444,7 @@ TEST(CASProtocol, DropReattachThroughDetachedNamespace) EXPECT_TRUE(s->listRefs(detached).empty()); /// The blob was never re-uploaded (token stable throughout — every publish dedup-adopted it). - EXPECT_EQ(b->head(blob_key).token, blob_tok); + EXPECT_EQ(currentIncarnation(*b, blob_key), blob_tok); } TEST(CASProtocol, FreezeIntoShadowNamespace) @@ -457,7 +481,7 @@ TEST(CASProtocol, DisplacedToLiveTokenCommitsAtCurrentIncarnation) writeBlobRaw(*b, s->layout(), "payload-X", s->poolMeta().blob_header_len, s->poolMeta().pool_id); const String blob_key = s->layout().blobKey(idOf("payload-X")); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); auto build = startBuildFor(s, ns, "part_1"); /// Wiring order (EDGE-BEFORE-OBSERVE): stageManifest -> precommitAdd -> putBlob. @@ -466,23 +490,23 @@ TEST(CASProtocol, DisplacedToLiveTokenCommitsAtCurrentIncarnation) build->putBlob(idOf("payload-X"), BlobSource::fromString("payload-X")); /// dedup → adopts t0 /// Another writer displaces X to t1 (uncondemned) before our gate runs. - const Token t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); + const Etag t1 = displaceBlobToken(*b, s->layout(), idOf("payload-X")); ASSERT_NE(t1, t0); /// The view still condemns the OLD t0 at round 1, fenced. injectRetire(*b, s->layout(), /*round*/ 1, /*shard*/ 0, - {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = t0, .size = 9}}); + {RetiredEntry{.kind = ObjectKind::Blob, .ref = DB::Cas::BlobRef{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(u128Of("payload-X"))}, .token = PersistedEtag::capture(t0), .size = 9}}); /// promote: revalidate X ⇒ HEAD current t1 (NOT condemned; only the defunct t0 is) ⇒ commit. build->promote(ns, "part_1", build->buildId(), id); /// The blob lives at t1 (the displacing writer's incarnation) and the part reads. - EXPECT_EQ(b->head(blob_key).token, t1); + EXPECT_EQ(currentIncarnation(*b, blob_key), t1); assertPartReads(b, s, ns, "part_1", "data.bin", "payload-X"); /// NO-LOSS / NO-RETURN: t0 is dead — a deleteExact against it TokenMismatches (the GC delete of the /// condemned t0 spares the live t1). - EXPECT_EQ(b->deleteExact(blob_key, t0).kind, DeleteOutcome::Kind::TokenMismatch); + EXPECT_EQ(removeAtIncarnation(*b, blob_key, t0), Removal::Mismatch); } TEST(CASProtocol, NewNamespacePublishGatedByShardFenceFloor) @@ -518,15 +542,16 @@ TEST(CASProtocol, NewNamespacePublishGatedByShardFenceFloor) build_a.reset(); s->renewWatermarkOnce(); Gc gc(s, hexToU128("00000000000000000000000000000001")); + DB::Cas::tests::OperationForTest blob_op(*b); for (size_t r = 0; r < 16; ++r) { const RoundReport rep = DB::Cas::tests::runRegularRoundReclaiming(gc); s->renewWatermarkOnce(); - if (!b->head(blob_key).exists) + if (!(*blob_op).head(blob_key, Retry::once()).has_value()) break; } /// The blob (unreachable) was deleted at t0. - EXPECT_FALSE(b->head(blob_key).exists); + EXPECT_FALSE((*blob_op).head(blob_key, Retry::once()).has_value()); /// 4. build B publishes into a BRAND-NEW namespace. §4 manifest-trust: the adopted leaf is trusted at /// promote (no probe) ⇒ promote SUCCEEDS and commits a manifest naming the deleted blob (the dangle). @@ -562,7 +587,7 @@ TEST(CASProtocol, FreshEvidenceDepWithViewHitIsResolvedByGate) "payload-fresh-ev"); } const String blob_key = layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128(hex))}); - const Token t0 = b->head(blob_key).token; + const Etag t0 = currentIncarnation(*b, blob_key); condemnMeta(*b, layout, hexToU128(hex), /*condemn_round*/ 1); auto s = openPool(b); @@ -579,7 +604,7 @@ TEST(CASProtocol, FreshEvidenceDepWithViewHitIsResolvedByGate) /// promote trusts the adopted leaf ⇒ commit, no probe, no displacement. EXPECT_NO_THROW(build->promote(ns, "part_1", build->buildId(), id)); - EXPECT_EQ(b->head(blob_key).token, t0) << "trust must not displace the adopted blob"; + EXPECT_EQ(currentIncarnation(*b, blob_key), t0) << "trust must not displace the adopted blob"; EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); } diff --git a/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp b/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp index afe5ed26fb44..3fd231d9de78 100644 --- a/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp +++ b/src/Disks/tests/gtest_cas_rebuild_condemn_nothing.cpp @@ -55,8 +55,6 @@ const RootNamespace kNsB{"00/zz@cas@"}; class CatalogChangesOnSecondReadBackend : public CountingBackend { public: - using Backend::get; - void armCatalogMutation(const String & key) { catalog_key = key; @@ -66,9 +64,9 @@ class CatalogChangesOnSecondReadBackend : public CountingBackend size_t catalogReads() const { return catalog_reads; } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - auto got = CountingBackend::get(key, range); + auto got = CountingBackend::read(key, access); if (!armed || key != catalog_key) return got; @@ -78,11 +76,9 @@ class CatalogChangesOnSecondReadBackend : public CountingBackend if (!got) throw std::runtime_error("catalog mutation fixture: second catalog read found absence"); - const PutResult put = CountingBackend::putOverwrite( - key, encodeRefCatalog(RefCatalog{}), got->token, {}); - if (put.outcome != PutOutcome::Done) + if (!CountingBackend::write(key, encodeRefCatalog(RefCatalog{}), got->value, access)) throw std::runtime_error("catalog mutation fixture: catalog rewrite conflicted"); - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); } private: @@ -96,24 +92,37 @@ BlobRef blobRefOf(const DB::UInt128 & hash) return BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}; } +std::optional readOf(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +bool headExists(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + bool blobPresent(Backend & backend, const Layout & layout, const DB::UInt128 & hash) { - return backend.head(layout.blobKey(blobRefOf(hash))).exists; + return headExists(backend, layout.blobKey(blobRefOf(hash))); } -/// Whether ANY run the newest fold seal references carries a `kCondemned` row for `hash`. This is where +/// Whether ANY run the newest fold seal references carries a `RunMarker::Condemned` row for `hash`. This is where /// a rebuild used to put its zero-edge condemnations, so "nothing was condemned" is checked HERE rather /// than by watching for a deletion several rounds later. bool condemnedInSealedRuns(Backend & backend, const Layout & layout, const DB::UInt128 & hash) { - const GcState st = decodeGcState(backend.get(layout.gcStateKey())->bytes); - const auto sealed = backend.get(layout.foldSealKey(st.snap_generation, st.snap_attempt)); + DB::Cas::tests::OperationForTest operation(backend); + const GcState st = decodeGcState((*operation).read(layout.gcStateKey(), Retry::standard())->bytes); + const auto sealed = (*operation).read(layout.foldSealKey(st.snap_generation, st.snap_attempt), Retry::standard()); if (!sealed) return false; const CasFoldSeal seal = decodeFoldSeal(sealed->bytes); for (const RunRef & r : seal.blob_target_runs) { - auto reader = openSourceEdgeRun(backend, r.key); + auto reader = openSourceEdgeRun(*operation, r.key); String k; String p; while (reader.next(k, p)) @@ -121,7 +130,7 @@ bool condemnedInSealedRuns(Backend & backend, const Layout & layout, const DB::U BlobRef ref; UInt128 sid; SourceEdgeKeyCodec::parse(k, ref, sid); - if (p.empty() || p[0] != kCondemned) + if (p.empty() || runMarkerFromByte(p[0], "CAS test source-edge run") != RunMarker::Condemned) continue; if (ref.digest.toU128() == hash) return true; @@ -159,7 +168,7 @@ RefTableState stateAfter(Backend & backend, const Layout & layout, const RootNam RefReplayBuilder builder(std::nullopt); for (const RefTxnId & id : ids) { - const auto got = backend.get(layout.refLogKey(fixture::fixtureLife(ns), id)); + const auto got = readOf(backend, layout.refLogKey(fixture::fixtureLife(ns), id)); if (!got) throw std::runtime_error("stateAfter: fixture log " + std::to_string(id.writer_epoch) + "-" + std::to_string(id.ref_sequence) + " is missing"); @@ -223,8 +232,8 @@ TEST(CASRebuildCondemnNothing, HiddenLiveManifestBlobIsNotCondemned) backend->hide(hidden_manifest); /// Precondition: both objects really are durable and really are hidden. - ASSERT_TRUE(backend->get(layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2})).has_value()); - ASSERT_TRUE(backend->get(hidden_manifest).has_value()); + ASSERT_TRUE(readOf(*backend, layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2})).has_value()); + ASSERT_TRUE(readOf(*backend, hidden_manifest).has_value()); Gc gc(store, kGc); const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); @@ -270,8 +279,8 @@ TEST(CASRebuildCondemnNothing, OrphanBlobIsRetainedNotCondemned) ASSERT_TRUE(rep.performed) << rep.refusal; EXPECT_FALSE(condemnedInSealedRuns(*backend, layout, DB::UInt128(2))); - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); - const CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const GcState st = decodeGcState(readOf(*backend, layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal(readOf(*backend, layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); ASSERT_TRUE(seal.condemned_summary.contains(0)) << "the summary stays TOTAL over gc_shards"; EXPECT_EQ(seal.condemned_summary.at(0).condemned_total, 0u) << "a rebuild condemns nothing"; @@ -302,7 +311,10 @@ TEST(CASRebuildCondemnNothing, NonCanonicalLifeKeyDoesNotAbortTheRebuild) /// Hand-built: no helper can mint this shape any more. const String noncanonical_life = layout.casRefsPrefix() + kNsA.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; - ASSERT_EQ(backend->putIfAbsent(noncanonical_life, "garbage").outcome, PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).create(noncanonical_life, "garbage", Retry::standard()))); + } Gc gc(store, kGc); RebuildReport rep; @@ -338,19 +350,23 @@ TEST(CASRebuildCondemnNothing, NestedLifelessKeyUnderTheLifePrefixDoesNotAbortTh Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); { - const auto got = backend->get(layout.gcStateKey()); + const auto got = readOf(*backend, layout.gcStateKey()); ASSERT_TRUE(got.has_value()); GcState st = decodeGcState(got->bytes); st.snap_generation = 0; - ASSERT_EQ(backend->putOverwrite(layout.gcStateKey(), encodeGcState(st), got->token).outcome, - PutOutcome::Done); + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(layout.gcStateKey(), encodeGcState(st), got->etag, Retry::standard()))); } /// Hand-built, and planted AFTER the round so the round itself is clean: one segment too deep under /// the life prefix, so the segment where the incarnation belongs holds `x`. No helper mints this. const String nested = layout.namespaceStreamPrefix(fixture::fixtureLife(kNsA)) + "x/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; - ASSERT_EQ(backend->putIfAbsent(nested, "garbage").outcome, PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).create(nested, "garbage", Retry::standard()))); + } RebuildReport rep; ASSERT_NO_THROW(rep = gc.rebuildBaseline(/*force=*/false)) @@ -379,12 +395,13 @@ TEST(CASRebuildCondemnNothing, OneCatalogCutDrivesHealthCheckAndRebuild) Gc gc(store, kGc); ASSERT_TRUE(gc.runRegularRound().acquired_lease); { - const auto got = backend->get(layout.gcStateKey()); + const auto got = readOf(*backend, layout.gcStateKey()); ASSERT_TRUE(got); GcState state = decodeGcState(got->bytes); state.snap_generation = 0; - ASSERT_EQ(backend->putOverwrite( - layout.gcStateKey(), encodeGcState(state), got->token).outcome, PutOutcome::Done); + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(layout.gcStateKey(), encodeGcState(state), got->etag, Retry::standard()))); } backend->armCatalogMutation(layout.refCatalogKey()); @@ -419,7 +436,7 @@ TEST(CASRebuildCondemnNothing, CarriesHoldsVerbatimWhileCondemningNothing) const RefHold planted{.reason = HoldReason::GapBelowWitness, .offending_position = RefTxnId{4, 9}, .retry_count = 17, .next_retry_round = 23}; { - const GcState adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState adopted = decodeGcState(readOf(*backend, layout.gcStateKey())->bytes); seedFoldCursorForTest(*backend, layout, kNsA, RefTxnId{1, 1}, planted, adopted.snap_generation, adopted.snap_attempt); } @@ -427,11 +444,11 @@ TEST(CASRebuildCondemnNothing, CarriesHoldsVerbatimWhileCondemningNothing) const RebuildReport rep = gc.rebuildBaseline(/*force=*/true); ASSERT_TRUE(rep.performed) << rep.refusal; - const GcState st = decodeGcState(backend->get(layout.gcStateKey())->bytes); - const CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); + const GcState st = decodeGcState(readOf(*backend, layout.gcStateKey())->bytes); + const CasFoldSeal seal = decodeFoldSeal(readOf(*backend, layout.foldSealKey(st.snap_generation, st.snap_attempt))->bytes); const auto it = seal.ref_lives.find(catalogLifeIdForTest(*backend, layout, kNsA)); ASSERT_NE(it, seal.ref_lives.end()); - EXPECT_EQ(it->second.coverage.classification, 4); + EXPECT_EQ(it->second.coverage.classification, CoverageClass::Clamped); ASSERT_TRUE(it->second.coverage.hold.has_value()); EXPECT_EQ(*it->second.coverage.hold, planted) << "a rebuild retried nothing, so it rewrites nothing about the hold"; @@ -483,9 +500,10 @@ TEST(CASRebuildCondemnNothingFsck, MidChainHoleBelowAWitnessIsChainBroken) /// Punch the hole: {1,2} is gone while {1,3} stays durable and listed. const String holed = layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2}); - const HeadResult h = backend->head(holed); - ASSERT_TRUE(h.exists); - backend->deleteExact(holed, h.token); + OperationForTest hole_op(*backend); + const auto h = (*hole_op).head(holed, Retry::standard()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*hole_op).remove(holed, h->etag, Retry::standard()), Removal::Removed); FsckReport rep; ASSERT_NO_THROW(rep = runFsck(*store, /*detail=*/true)) @@ -619,9 +637,10 @@ TEST(CASRebuildCondemnNothingFsck, OneBadNamespaceDoesNotAbortTheAudit) RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 3}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); const String holed = layout.refLogKey(fixture::fixtureLife(kNsA), RefTxnId{1, 2}); - const HeadResult h = backend->head(holed); - ASSERT_TRUE(h.exists); - backend->deleteExact(holed, h.token); + OperationForTest hole_op(*backend); + const auto h = (*hole_op).head(holed, Retry::standard()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*hole_op).remove(holed, h->etag, Retry::standard()), Removal::Removed); publishAt(*backend, layout, kNsB, RefTxnId{1, 1}, "ref_z", 1, DB::UInt128(9), /*birth=*/true); writeCkptRaw(*backend, layout, kNsB, diff --git a/src/Disks/tests/gtest_cas_record_stream_format.cpp b/src/Disks/tests/gtest_cas_record_stream_format.cpp index 42e44585c357..e796b394ca8b 100644 --- a/src/Disks/tests/gtest_cas_record_stream_format.cpp +++ b/src/Disks/tests/gtest_cas_record_stream_format.cpp @@ -1,11 +1,15 @@ #include +#include "cas_format_test_battery.h" #include #include #include #include #include +#include #include +#include + using namespace DB; using namespace DB::Cas; @@ -26,17 +30,17 @@ BlobRef chRef(uint64_t n) SourceEdgeRecord edge(const BlobRef & ref, uint64_t source_id) { - return SourceEdgeRecord{.ref = ref, .source_id = UInt128(source_id), .marker = kEdgeActive}; + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(source_id), .marker = RunMarker::Edge}; } SourceEdgeRecord zero(const BlobRef & ref) { - return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = kZeroMarker}; + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = RunMarker::Zero}; } -SourceEdgeRecord condemned(const BlobRef & ref, const Token & token, uint64_t size, uint64_t round, bool pend) +SourceEdgeRecord condemned(const BlobRef & ref, const PersistedEtag & token, uint64_t size, uint64_t round, bool pend) { - return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = kCondemned, + return SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = RunMarker::Condemned, .delete_pending = pend, .token = token, .size = size, .condemn_round = round}; } @@ -66,6 +70,19 @@ std::vector decodeRun(const String & bytes) } +CAS_BATTERY_COVERS(RunFile); + +TEST(CASFormatBattery, RunFile) +{ + const std::vector records{edge(chRef(2), 5)}; + runFormatBattery({FormatId::RunFile, + [&] { return sealObject(FormatId::RunFile, encodeRun(records)); }, + [](std::string_view s) { decodeRun(std::string(openObject(FormatId::RunFile, s))); }, + fmt::format("{{\"type\":\"cas_run\",\"v\":{},\"kind\":\"source_edge\"}}\n", currentCompatibilityVersion()) + + "{\"ref\":\"0100000000000000000000000000000002\",\"src\":\"00000000000000000000000000000005\",\"mark\":\"edge\"}\n" + "{\"n\":1}\n"}); +} + TEST(CASRecordStream, EmptyRunRoundTripsAndChecksumMatches) { const String bytes = encodeRun({}); @@ -89,7 +106,7 @@ TEST(CASRecordStream, EdgeZeroCondemnedRoundTrip) /// an edge; c has a zero marker. Blobs ascend a < b < c, so the sequence is already non-decreasing. std::vector recs = { edge(a, 10), - condemned(b, Token{"e-1", TokenType::ETag}, 4242, 7, /*pend*/ true), + condemned(b, PersistedEtag{"etag", "e-1"}, 4242, 7, /*pend*/ true), zero(c), }; const String bytes = encodeRun(recs); @@ -98,28 +115,138 @@ TEST(CASRecordStream, EdgeZeroCondemnedRoundTrip) EXPECT_EQ(back[0].ref, a); EXPECT_EQ(back[0].source_id, UInt128(10)); - EXPECT_EQ(back[0].marker, kEdgeActive); + EXPECT_EQ(back[0].marker, RunMarker::Edge); EXPECT_EQ(back[1].ref, b); EXPECT_EQ(back[1].source_id, UInt128(0)); - EXPECT_EQ(back[1].marker, kCondemned); + EXPECT_EQ(back[1].marker, RunMarker::Condemned); EXPECT_TRUE(back[1].delete_pending); - EXPECT_EQ(back[1].token, (Token{"e-1", TokenType::ETag})); + EXPECT_EQ(back[1].token.dialect, "etag"); + EXPECT_EQ(back[1].token.value, "e-1"); EXPECT_EQ(back[1].size, 4242u); EXPECT_EQ(back[1].condemn_round, 7u); EXPECT_EQ(back[2].ref, c); - EXPECT_EQ(back[2].marker, kZeroMarker); + EXPECT_EQ(back[2].marker, RunMarker::Zero); +} + +/// Closed-set pin: the three `RunMarker` words, walked through `magic_enum::enum_values`, which is what proves the +/// renderer and the parser consult the SAME table: a table entry missing altogether is already a +/// build error at the coverage assert, but two delegates drifting onto different tables is not. +TEST(CASRecordStream, ClosedSetPinsRunMarkerWords) +{ + EXPECT_EQ(runMarkerToWireWord(RunMarker::Zero), "zero"); + EXPECT_EQ(runMarkerToWireWord(RunMarker::Edge), "edge"); + EXPECT_EQ(runMarkerToWireWord(RunMarker::Condemned), "condemned"); + for (const auto m : magic_enum::enum_values()) + EXPECT_EQ(runMarkerFromWireWord(runMarkerToWireWord(m)), m); +} + +/// The condemned row's six fields are all-or-nothing: a row that says `condemned` but drops one of +/// them would decode with a silently defaulted value (a zero size, an empty token, `pending` false), +/// which is a different retention decision than the writer recorded. +TEST(CASRecordStream, CondemnedRowMissingOneOfItsSixFieldsFailsClosed) +{ + const String good = encodeRun({condemned(chRef(2), PersistedEtag{"etag", "e-1"}, 4242, 7, /*pend*/ true)}); + for (const std::string_view field : {R"(,"pending":true)", R"(,"token_type":"etag")", R"(,"token":"e-1")", + R"(,"size":4242)", R"(,"condemn_round":"7")", R"(,"confirmed":false)"}) + { + String bytes = good; + const size_t at = bytes.find(field); + ASSERT_NE(at, String::npos) << "fixture does not carry " << field; + bytes.erase(at, field.size()); + try + { + static_cast(decodeRun(bytes)); + FAIL() << "expected CORRUPTED_DATA after dropping " << field; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + const String expected_message = field == R"(,"token_type":"etag")" || field == R"(,"token":"e-1")" + ? "CAS cas_run: token missing token_type/token" + : "CAS cas_run: condemned record missing pending/size/condemn_round/confirmed"; + EXPECT_EQ(e.message(), expected_message); + } + } } +/// The mirror fence: an active row carrying any condemned field is a row whose two halves disagree +/// about what it is, and the reader must not pick one half. +TEST(CASRecordStream, ActiveRowCarryingACondemnedFieldFailsClosed) +{ + String bytes = encodeRun({edge(chRef(1), 10)}); + const String needle = R"(,"mark":"edge")"; + const size_t at = bytes.find(needle); + ASSERT_NE(at, String::npos); + bytes.insert(at + needle.size(), R"(,"size":4242)"); + + try + { + static_cast(decodeRun(bytes)); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(e.message(), "CAS cas_run: non-condemned record carries condemned fields"); + } +} + +/// `CasBlobInDegree.cpp`'s fold (`SourceEdgeRunWriter writer(out); // sorted NDJSON; byte-deterministic +/// for write-once adoption`) sorts by `(ref, source_id)` before writing, so the run this fixture emits +/// is the writer's own straight-line append, never a reorder -- the property the run must actually have +/// is that whichever order the SAME set of edges was DISCOVERED in, the canonical `(ref, source_id)` +/// sort before the writer sees them converges on identical bytes. A bare `f(x) == f(x)` on one fixed, +/// already-sorted vector cannot fail on anything but genuine cross-call nondeterminism (a clock, a +/// pointer, hash randomization) -- none of which this format has -- so it passed vacuously; two +/// differently-DISCOVERED inputs, sorted by the same comparator production uses, is the real claim. TEST(CASRecordStream, WriterIsByteDeterministic) { - std::vector recs = { + const auto byRefThenSource = [](const SourceEdgeRecord & a, const SourceEdgeRecord & b) + { + if (a.ref != b.ref) + return a.ref < b.ref; + return a.source_id < b.source_id; + }; + + std::vector discovered_ascending = { edge(chRef(1), 5), edge(chRef(1), 9), - condemned(chRef(2), Token{"t/with/slashes", TokenType::ETag}, 1, 2, false), + condemned(chRef(2), PersistedEtag{"etag", "t/with/slashes"}, 1, 2, false), + }; + /// The SAME three edges, as if a different GC shard or a different LIST page order had surfaced + /// them: reverse discovery order, still every one of them present. + std::vector discovered_reverse = { + condemned(chRef(2), PersistedEtag{"etag", "t/with/slashes"}, 1, 2, false), + edge(chRef(1), 9), + edge(chRef(1), 5), }; - EXPECT_EQ(encodeRun(recs), encodeRun(recs)); /// pure function of the sorted record set + ASSERT_NE(discovered_ascending.front().ref, discovered_reverse.front().ref) + << "the two discovery orders must actually differ"; + + std::sort(discovered_ascending.begin(), discovered_ascending.end(), byRefThenSource); + std::sort(discovered_reverse.begin(), discovered_reverse.end(), byRefThenSource); + EXPECT_EQ(encodeRun(discovered_ascending), encodeRun(discovered_reverse)) + << "the run must be a pure function of the edge SET, not of the order it was discovered in"; +} + +/// The run `ref` carries the algorithm as a raw leading BYTE, a second representation of the same +/// closed set the `algo` WORD spells elsewhere. The word side is proven exhaustive at compile time by +/// its wire table; the byte side is a hand-written switch, so nothing but this walk stops a new +/// algorithm from being written by `renderB` and rejected by the reader -- an asymmetry that would +/// appear as unreadable runs rather than as a failing build. +TEST(CASRecordStream, EveryBlobHashAlgoRoundTripsThroughTheRunRefByte) +{ + for (const BlobHashAlgo algo : magic_enum::enum_values()) + { + BlobDigest digest{}; + digest.bytes[0] = 0x10; + const BlobRef ref{algo, digest}; + const std::vector back = decodeRun(encodeRun({edge(ref, 1)})); + ASSERT_EQ(back.size(), 1u) << "algo " << magic_enum::enum_name(algo); + EXPECT_EQ(back[0].ref.algo, algo) << "the leading byte did not survive the round trip"; + } } TEST(CASRecordStream, SortOrderAcrossAlgosFollowsAlgoByte) @@ -159,9 +286,9 @@ TEST(CASRecordStream, SourceIdRendersAs32Hex) { const String bytes = encodeRun({edge(chRef(1), 10)}); /// The source id 10 is a 32-char lowercase hex string ending in 'a'. - EXPECT_NE(bytes.find("\"s\":\"0000000000000000000000000000000a\""), String::npos); - /// The record key `b` for a ch128 ref is the algo byte 01 + a 32-hex digest (34 chars total). - EXPECT_NE(bytes.find("\"b\":\"01"), String::npos); + EXPECT_NE(bytes.find("\"src\":\"0000000000000000000000000000000a\""), String::npos); + /// The record key `ref` for a ch128 ref is the algo byte 01 + a 32-hex digest (34 chars total). + EXPECT_NE(bytes.find("\"ref\":\"01"), String::npos); } TEST(CASRecordStream, SealChecksumMismatchFailsClosed) @@ -203,6 +330,24 @@ TEST(CASRecordStream, TrailerCountMismatchIsCorruptData) EXPECT_THROW(decodeRun(bytes), DB::Exception); } +TEST(CASRecordStream, UppercaseDigestInRecordKeyIsCorruptedData) +{ + String bytes = encodeRun({edge(chRef(10), 1)}); + const size_t digest = bytes.find("0000000000000000000000000000000a"); + ASSERT_NE(digest, String::npos); + bytes[digest + 31] = 'A'; + + try + { + static_cast(decodeRun(bytes)); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + } +} + TEST(CASRecordStream, TruncationAtLineBoundaryFailsClosed) { const String bytes = encodeRun({edge(chRef(1), 10), edge(chRef(1), 20)}); @@ -216,12 +361,13 @@ TEST(CASRecordStream, HeaderGates) { /// Wrong type. { - const String s = "{\"type\":\"cas_pool_meta\",\"v\":3,\"kind\":\"source_edge\"}\n{\"n\":0}\n"; + const String s = "{\"type\":\"cas_pool_meta\",\"v\":1,\"kind\":\"source_edge\"}\n{\"n\":0}\n"; EXPECT_THROW(decodeRun(s), DB::Exception); } - /// Wrong kind. + /// Wrong kind. `v:1` is the baseline generation, so it always passes the header gate before the + /// kind check runs. { - const String s = "{\"type\":\"cas_run\",\"v\":3,\"kind\":\"blob_delta\"}\n{\"n\":0}\n"; + const String s = "{\"type\":\"cas_run\",\"v\":1,\"kind\":\"blob_delta\"}\n{\"n\":0}\n"; EXPECT_THROW(decodeRun(s), DB::Exception); } /// Future version -> UNKNOWN_FORMAT_VERSION. diff --git a/src/Disks/tests/gtest_cas_recovery_grounding.cpp b/src/Disks/tests/gtest_cas_recovery_grounding.cpp index 5362813bdfa6..3e31f4f5650c 100644 --- a/src/Disks/tests/gtest_cas_recovery_grounding.cpp +++ b/src/Disks/tests/gtest_cas_recovery_grounding.cpp @@ -47,18 +47,23 @@ class RecoveryListingBackend : public CountingBackend size_t list_calls = 0; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + /// On the transport primitive, not the legacy verb: every enumeration a `CasOperation` makes + /// reaches the store through this, so a distortion left on the verb would never fire and the + /// `list_calls` assertions would read zero whatever recovery did. + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { ++list_calls; - ListPage page = CountingBackend::list(prefix, cursor, limit); + DB::Cas::Backend::RawListPage page = CountingBackend::list(prefix, cursor, limit, access); if (mode == ListingMode::Empty) page.keys.clear(); else if (mode == ListingMode::Partial) { - page.keys.erase(std::remove_if(page.keys.begin(), page.keys.end(), [](const ListedKey & key) - { - return key.key.find("/_log/") != String::npos; - }), page.keys.end()); + page.keys.erase(std::remove_if(page.keys.begin(), page.keys.end(), + [](const DB::Cas::Backend::RawListedKey & key) + { + return key.key.find("/_log/") != String::npos; + }), page.keys.end()); } else if (mode == ListingMode::Reordered) std::reverse(page.keys.begin(), page.keys.end()); @@ -112,14 +117,16 @@ void seedAuthoritativeStream(Backend & backend, const Layout & layout, const Roo applyRefLogTxn(snapshot_state, first_txn); writeRefSnapshotRaw(backend, layout, snapshotOf(snapshot_state, ns.string())); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(backend, layout, ns); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(op, layout, ns); const RefCkpt authority{ .life_epoch = 1, .committed_through = committed_through, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = committed_through.writer_epoch > 1 ? std::optional{RefTxnId{1, 2}} : std::nullopt}; - backend.putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(authority)); + (void)op.create(layout.refCkptKey(life), encodeRefCkpt(authority), Retry::standard()); } /// This is deliberately caller-side plumbing, not a convenience overload in `CasRefProtocol`: production @@ -127,7 +134,9 @@ void seedAuthoritativeStream(Backend & backend, const Layout & layout, const Roo /// The API under test receives those exact values and performs no catalog or checkpoint resolution itself. RecoveredRefTable recoverFromCurrentCatalogCut(Backend & backend, const Layout & layout, const RootNamespace & ns) { - const CasRefCatalog::Snapshot cut = CasRefCatalog::read(backend, layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(op, layout); std::optional entry; for (const CatalogEntry & candidate : cut.catalog.entries) { @@ -141,10 +150,10 @@ RecoveredRefTable recoverFromCurrentCatalogCut(Backend & backend, const Layout & if (entry) { const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); - if (const std::optional sample = readCkpt(backend, layout, life)) + if (const std::optional sample = readCkpt(op, layout, life)) checkpoint = sample->ckpt; } - return recoverRefTableDetailedFromAuthority(backend, layout, entry, checkpoint); + return recoverRefTableDetailedFromAuthority(op, layout, entry, checkpoint); } CatalogEntry catalog(NsState state) @@ -269,9 +278,9 @@ TEST(CASRecoveryGrounding, RejectsLifeEpochAboveCommittedFrontierOnDecodeAndGrou { const RefCkpt invalid = ckpt(2, RefTxnId{1, 5}); String encoded = encodeRefCkpt(ckpt(1, RefTxnId{1, 5})); - const size_t life_epoch = encoded.find(R"("le":"1")"); + const size_t life_epoch = encoded.find(R"("life_epoch":"1")"); ASSERT_NE(life_epoch, String::npos); - encoded.replace(life_epoch, String{R"("le":"1")"}.size(), R"("le":"2")"); + encoded.replace(life_epoch, String{R"("life_epoch":"1")"}.size(), R"("life_epoch":"2")"); expectCode([&] { (void)decodeRefCkpt(encoded); }, DB::ErrorCodes::CORRUPTED_DATA); expectCode([&] { (void)chooseRecoveryGrounding(catalog(NsState::Live), invalid); }, @@ -295,11 +304,13 @@ TEST(CASRecoveryGrounding, RecoveryIsEquivalentUnderFullEmptyPartialAndReordered for (const ListingMode mode : {ListingMode::Full, ListingMode::Empty, ListingMode::Partial, ListingMode::Reordered}) { auto backend = std::make_shared(mode); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/list_equivalence"}; const RefTxnId frontier{2, 1}; seedAuthoritativeStream(*backend, layout, ns, frontier); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); backend->resetCounts(); backend->list_calls = 0; @@ -344,39 +355,48 @@ TEST(CASRecoveryGrounding, CatalogLifecycleAndCheckpointAreMandatoryForReadOnlyR { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 1}, {namespaceBirthOp()})); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); } { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); CasRefCatalog::casAdmitEntry( - *backend, layout, 1, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = 8}); + catalog_op, layout, 1, CatalogEntry{.ns = ns, .state = NsState::Live, .incarnation = 8}); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); } { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const CatalogEntry live{.ns = ns, .state = NsState::Live, .incarnation = 9}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, live); + CasRefCatalog::casAdmitEntry(catalog_op, layout, 1, live); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(live.ns, live.incarnation); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), "not a sealed checkpoint").outcome, - PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), "not a sealed checkpoint", Retry::once()))); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); } { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); CatalogEntry creating{.ns = ns, .state = NsState::Creating, .incarnation = 7, .creator = CreatorFence{"srv1", 1, 1}}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); - backend->putIfAbsent(layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation)), + CasRefCatalog::casAdmitEntry(catalog_op, layout, 1, creating); + catalog_op.create(layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(creating.ns, creating.incarnation)), encodeRefCkpt(RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}), Retry::once()); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::INVALID_STATE); } { auto backend = std::make_shared(ListingMode::Full); - backend->putIfAbsent(layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + catalog_op.create(layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(RefCkpt{.life_epoch = 1, .committed_through = RefTxnId{1, 1}, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}), Retry::once()); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::INVALID_STATE); } } @@ -393,30 +413,36 @@ TEST(CASRecoveryGrounding, NonrecoverableAuthorityPerformsNoBackendRecoveryIo) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); backend->resetCounts(); const CatalogEntry creating{ .ns = ns, .state = NsState::Creating, .incarnation = 1, .creator = CreatorFence{"srv1", 1, 1}}; expectCode( - [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, creating, valid_ckpt); }, + [&] { (void)recoverRefTableDetailedFromAuthority(catalog_op, layout, creating, valid_ckpt); }, DB::ErrorCodes::INVALID_STATE); EXPECT_EQ(backend->list_calls, 0u); EXPECT_EQ(backend->getTotal(), 0u); } { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); backend->resetCounts(); const CatalogEntry live{.ns = ns, .state = NsState::Live, .incarnation = 2}; expectCode( - [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, live, std::nullopt); }, + [&] { (void)recoverRefTableDetailedFromAuthority(catalog_op, layout, live, std::nullopt); }, DB::ErrorCodes::CORRUPTED_DATA); EXPECT_EQ(backend->list_calls, 0u); EXPECT_EQ(backend->getTotal(), 0u); } { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); backend->resetCounts(); expectCode( - [&] { (void)recoverRefTableDetailedFromAuthority(*backend, layout, std::nullopt, valid_ckpt); }, + [&] { (void)recoverRefTableDetailedFromAuthority(catalog_op, layout, std::nullopt, valid_ckpt); }, DB::ErrorCodes::INVALID_STATE); EXPECT_EQ(backend->list_calls, 0u); EXPECT_EQ(backend->getTotal(), 0u); @@ -426,6 +452,8 @@ TEST(CASRecoveryGrounding, NonrecoverableAuthorityPerformsNoBackendRecoveryIo) TEST(CASRecoveryGrounding, ReadOnlyRecoveryNeverAdoptsFPlusOne) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/read_only_excludes_f_plus_one"}; seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 1}, /*include_f_plus_one=*/true); @@ -444,10 +472,12 @@ TEST(CASRecoveryGrounding, ForgedWellFormedListedSnapshotIsUnobservedAndRecovery for (const ListingMode mode : {ListingMode::Full, ListingMode::Empty, ListingMode::Partial, ListingMode::Reordered}) { auto backend = std::make_shared(mode); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/forged_listed_snapshot"}; seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 1}, /*include_f_plus_one=*/true); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); RefTableState forged_state; std::vector birth{namespaceBirthOp()}; @@ -475,6 +505,8 @@ TEST(CASRecoveryGrounding, ForgedWellFormedListedSnapshotIsUnobservedAndRecovery TEST(CASRecoveryGrounding, SemanticallyMalformedCheckpointSnapshotIsCorruptionAfterExactRead) { auto backend = std::make_shared(ListingMode::Empty); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/semantically_malformed_checkpoint"}; const ManifestRef manifest{1, 1, 1}; @@ -488,13 +520,13 @@ TEST(CASRecoveryGrounding, SemanticallyMalformedCheckpointSnapshotIsCorruptionAf malformed.precommits.push_back(RefOwnerBinding{RefOwnerKind::Precommit, "precommit", manifest}); writeRefSnapshotRaw(*backend, layout, malformed); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); const String snapshot_key = layout.refSnapshotKey(life, {1, 1}); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = RefTxnId{1, 1}, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}), Retry::once()))); backend->resetCounts(); try @@ -514,11 +546,13 @@ TEST(CASRecoveryGrounding, SemanticallyMalformedCheckpointSnapshotIsCorruptionAf TEST(CASRecoveryGrounding, CheckpointSnapshotEqualToLastEpochSealIsRejectedBeforeReadingItsLog) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/checkpoint_base_seal"}; /// The checkpoint directly contradicts itself: its sole snapshot base names its terminal seal. seedAuthoritativeStream(*backend, layout, ns, RefTxnId{1, 2}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); RefTableState through_seal; std::vector birth{namespaceBirthOp()}; @@ -530,14 +564,14 @@ TEST(CASRecoveryGrounding, CheckpointSnapshotEqualToLastEpochSealIsRejectedBefor applyRefLogTxn(through_seal, txn(ns, {1, 2}, {std::move(seal)})); writeRefSnapshotRaw(*backend, layout, snapshotOf(through_seal, ns.string())); - const CkptSample before = *readCkpt(*backend, layout, life); + const CkptSample before = *readCkpt(catalog_op, layout, life); const RefCkpt with_sealed_base{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = RefTxnId{1, 2}, .last_epoch_seal = RefTxnId{1, 2}}; - ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(with_sealed_base), before.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(catalog_op.replace( + layout.refCkptKey(life), encodeRefCkpt(with_sealed_base), before.etag, Retry::standard()))); backend->resetCounts(); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); @@ -554,6 +588,8 @@ TEST(CASRecoveryGrounding, CheckpointSnapshotEqualToLastEpochSealIsRejectedBefor TEST(CASRecoveryGrounding, SameEpochFrontierAfterDecodedEpochSealIsCorruption) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/frontier_after_seal"}; @@ -562,16 +598,16 @@ TEST(CASRecoveryGrounding, SameEpochFrontierAfterDecodedEpochSealIsCorruption) seal.kind = RefOpKind::EpochSeal; DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 2}, {std::move(seal)})); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); String malformed_ckpt = encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = RefTxnId{1, 2}}); - const size_t frontier_sequence = malformed_ckpt.find(R"("cts":"2")"); + const size_t frontier_sequence = malformed_ckpt.find(R"("committed_seq":"2")"); ASSERT_NE(frontier_sequence, String::npos); - malformed_ckpt.replace(frontier_sequence, String{R"("cts":"2")"}.size(), R"("cts":"3")"); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), malformed_ckpt).outcome, PutOutcome::Done); + malformed_ckpt.replace(frontier_sequence, String{R"("committed_seq":"2")"}.size(), R"("committed_seq":"3")"); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), malformed_ckpt, Retry::once()))); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); } @@ -579,10 +615,12 @@ TEST(CASRecoveryGrounding, SameEpochFrontierAfterDecodedEpochSealIsCorruption) TEST(CASRecoveryGrounding, OlderCheckpointSnapshotAtSealIsCorruption) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/older_checkpoint_base_seal"}; seedAuthoritativeStream(*backend, layout, ns, RefTxnId{2, 1}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); RefOp second_seal; second_seal.kind = RefOpKind::EpochSeal; @@ -600,14 +638,14 @@ TEST(CASRecoveryGrounding, OlderCheckpointSnapshotAtSealIsCorruption) applyRefLogTxn(through_first_seal, txn(ns, {1, 2}, {std::move(first_seal)})); writeRefSnapshotRaw(*backend, layout, snapshotOf(through_first_seal, ns.string())); - const CkptSample before = *readCkpt(*backend, layout, life); + const CkptSample before = *readCkpt(catalog_op, layout, life); const RefCkpt with_old_sealed_base{ .life_epoch = 1, .committed_through = RefTxnId{3, 1}, .checkpoint_snapshot_id = RefTxnId{1, 2}, .last_epoch_seal = RefTxnId{2, 2}}; - ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(with_old_sealed_base), before.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(catalog_op.replace( + layout.refCkptKey(life), encodeRefCkpt(with_old_sealed_base), before.etag, Retry::standard()))); backend->resetCounts(); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); @@ -620,6 +658,8 @@ TEST(CASRecoveryGrounding, OlderCheckpointSnapshotAtSealIsCorruption) TEST(CASRecoveryGrounding, TerminalGapBelowFrontierIsCorruptionNotARebirth) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RootNamespace ns{"srv1/terminal_gap"}; const RefLogTxn birth = txn(ns, {1, 1}, {namespaceBirthOp()}); @@ -629,16 +669,16 @@ TEST(CASRecoveryGrounding, TerminalGapBelowFrontierIsCorruptionNotARebirth) DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {1, 2}, {std::move(remove)})); DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, {2, 1}, {namespaceBirthOp()})); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); String malformed_ckpt = encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); - const size_t frontier_epoch = malformed_ckpt.find(R"("cte":"1")"); + const size_t frontier_epoch = malformed_ckpt.find(R"("committed_epoch":"1")"); ASSERT_NE(frontier_epoch, String::npos); - malformed_ckpt.replace(frontier_epoch, String{R"("cte":"1")"}.size(), R"("cte":"2")"); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), malformed_ckpt).outcome, PutOutcome::Done); + malformed_ckpt.replace(frontier_epoch, String{R"("committed_epoch":"1")"}.size(), R"("committed_epoch":"2")"); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), malformed_ckpt, Retry::once()))); expectCode([&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, DB::ErrorCodes::CORRUPTED_DATA); } @@ -646,6 +686,8 @@ TEST(CASRecoveryGrounding, TerminalGapBelowFrontierIsCorruptionNotARebirth) TEST(CASRecoveryGrounding, LaterEpochCheckpointBaseRequiresItsContextualBacklink) { auto backend = std::make_shared(ListingMode::Full); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); const Layout layout("p"); const RefTxnId seal_id{1, 2}; const RefTxnId base_id{2, 1}; @@ -659,12 +701,12 @@ TEST(CASRecoveryGrounding, LaterEpochCheckpointBaseRequiresItsContextualBacklink DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, base_id, {}, backlink)); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = base_id, .checkpoint_snapshot_id = base_id, - .last_epoch_seal = seal_id})).outcome, PutOutcome::Done); + .last_epoch_seal = seal_id}), Retry::once()))); expectCode( [&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, @@ -682,12 +724,12 @@ TEST(CASRecoveryGrounding, LaterEpochCheckpointBaseRequiresItsContextualBacklink DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, txn(ns, base_id, {}, seal_id)); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), base_id)); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns); + ASSERT_TRUE(std::holds_alternative(catalog_op.create(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = base_id, .checkpoint_snapshot_id = base_id, - .last_epoch_seal = seal_id})).outcome, PutOutcome::Done); + .last_epoch_seal = seal_id}), Retry::once()))); expectCode( [&] { (void)recoverFromCurrentCatalogCut(*backend, layout, ns); }, diff --git a/src/Disks/tests/gtest_cas_recovery_streaming.cpp b/src/Disks/tests/gtest_cas_recovery_streaming.cpp index ad2dbbcb9342..5b087b738532 100644 --- a/src/Disks/tests/gtest_cas_recovery_streaming.cpp +++ b/src/Disks/tests/gtest_cas_recovery_streaming.cpp @@ -143,7 +143,8 @@ bool pollUntil(Pred pred) class VanishMidTailOnceBackend : public InMemoryBackend { public: - using InMemoryBackend::get; /// keep the one-arg convenience overload visible past our override + /// Unhide the primitive overload that the override below would otherwise hide. + using InMemoryBackend::list; String target_log_key; String refs_prefix; @@ -151,18 +152,18 @@ class VanishMidTailOnceBackend : public InMemoryBackend std::atomic vanished{false}; std::atomic fresh_list_count{0}; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (armed.load() && key == target_log_key && !vanished.exchange(true)) return std::nullopt; /// selected object gone between LIST and GET; recovery must re-LIST - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (armed.load() && prefix == refs_prefix && cursor.empty()) fresh_list_count.fetch_add(1, std::memory_order_relaxed); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } }; @@ -172,7 +173,8 @@ class VanishMidTailOnceBackend : public InMemoryBackend class CorruptLogOnGetBackend : public InMemoryBackend { public: - using InMemoryBackend::get; /// keep the one-arg convenience overload visible past our override + /// Unhide the primitive overload that the override below would otherwise hide. + using InMemoryBackend::list; String target_log_key; String corrupt_bytes; @@ -180,19 +182,19 @@ class CorruptLogOnGetBackend : public InMemoryBackend std::atomic armed{false}; std::atomic refs_list_count{0}; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { - auto got = InMemoryBackend::get(key, range); + auto got = InMemoryBackend::read(key, access); if (armed.load() && got && key == target_log_key) got->bytes = corrupt_bytes; return got; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (armed.load() && prefix == refs_prefix && cursor.empty()) refs_list_count.fetch_add(1, std::memory_order_relaxed); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } }; @@ -202,7 +204,8 @@ class CorruptLogOnGetBackend : public InMemoryBackend class BlockingFirstLogGetBackend : public InMemoryBackend { public: - using InMemoryBackend::get; + /// Unhide the primitive overload that the override below would otherwise hide. + using InMemoryBackend::list; String refs_prefix; String target_log_key; @@ -211,18 +214,18 @@ class BlockingFirstLogGetBackend : public InMemoryBackend std::atomic list_calls{0}; std::function on_first_target_get; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (armed.load() && key == target_log_key && !blocked.exchange(true)) on_first_target_get(); - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (armed.load() && prefix == refs_prefix && cursor.empty()) list_calls.fetch_add(1, std::memory_order_relaxed); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } }; @@ -255,7 +258,9 @@ TEST(CASRecoveryStreaming, LongTailReplaysUnderMemoryBound) setRecoveryReplayMemoryProbeForTest(tracker.probe()); SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation catalog_op = catalog_requests.admit(); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(catalog_op, layout); const RefTableState state = recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, catalog_cut, ns).state; EXPECT_EQ(state.getPrecommits().size(), kTxns * kOpsPerTxn) << "the whole tail must have replayed"; EXPECT_LE(tracker.peak(), static_cast(bound)) @@ -305,9 +310,10 @@ TEST(CASRecoveryStreaming, MaterializingControlExceedsMemoryBound) /// tail has been applied -- exactly the memory profile streaming recovery replaced. std::vector resident_txns; int64_t held = 0; + OperationForTest tail_op(*backend); for (size_t t = 1; t <= kTxns; ++t) { - const auto got = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, t})); + const auto got = (*tail_op).read(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, t}), Retry::once()); ASSERT_TRUE(got.has_value()); RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), RefTxnId{1, t}); const int64_t footprint = static_cast(decodedRefLogTxnFootprint(txn)); @@ -471,7 +477,9 @@ TEST(CASRecoveryStreaming, OrphanSweepAndFsckSameBound) PeakTracker tracker; setRecoveryReplayMemoryProbeForTest(tracker.probe()); SCOPE_EXIT({ setRecoveryReplayMemoryProbeForTest({}); }); - const CasRefCatalog::Snapshot sweep_catalog_cut = CasRefCatalog::read(*backend, layout); + CasRequests sweep_requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation sweep_op = sweep_requests.admit(); + const CasRefCatalog::Snapshot sweep_catalog_cut = CasRefCatalog::read(sweep_op, layout); const RecoveredRefTable recovered = recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, sweep_catalog_cut, ns_sweep); EXPECT_EQ(recovered.state.getPrecommits().size(), kTxns * kOpsPerTxn); @@ -571,7 +579,8 @@ TEST(CASRecoveryStreaming, RecoveryResultInventoryComplete) base_txn.ops = publishCommittedOps("c_two", mref(12)); fixture::writeRefLogRaw(*backend, layout, base_txn); writeRefSnapshotRaw(*backend, layout, base); - const auto base_got = backend->get(layout.refSnapshotKey(fixture::fixtureLife(ns), base.snapshot_id)); + OperationForTest inv_op(*backend); + const auto base_got = (*inv_op).read(layout.refSnapshotKey(fixture::fixtureLife(ns), base.snapshot_id), Retry::once()); ASSERT_TRUE(base_got.has_value()); const uint64_t base_stored_bytes = base_got->bytes.size(); @@ -592,8 +601,8 @@ TEST(CASRecoveryStreaming, RecoveryResultInventoryComplete) .checkpoint_snapshot_id = RefTxnId{1, 5}, .last_epoch_seal = std::nullopt}); - const uint64_t tail6 = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 6}))->bytes.size(); - const uint64_t tail7 = backend->get(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 7}))->bytes.size(); + const uint64_t tail6 = (*inv_op).read(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 6}), Retry::once())->bytes.size(); + const uint64_t tail7 = (*inv_op).read(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, 7}), Retry::once())->bytes.size(); backend->resetCounts(); auto store = openPoolForTest(backend); diff --git a/src/Disks/tests/gtest_cas_ref_carve.cpp b/src/Disks/tests/gtest_cas_ref_carve.cpp index e12706d64f69..a06fe1335634 100644 --- a/src/Disks/tests/gtest_cas_ref_carve.cpp +++ b/src/Disks/tests/gtest_cas_ref_carve.cpp @@ -66,7 +66,7 @@ PoolPtr openPool(const BackendPtr & backend) /// injection/verification that separately computes a key via `DB::Cas::tests::fixture::fixtureLife(ns)`. void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) { - DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + DB::Cas::tests::casAdmitRecoverableEntry(*s->poolBackendPtr(), s->layout(), ns, s->liveWriterEpoch()); PartWriteInfo info; info.intended_namespace = ns; info.intended_ref = ns.string() + "/" + ref; @@ -96,11 +96,12 @@ struct CaseSync /// cache). Used to inspect exactly what a flush durably committed. std::optional newestLogTxn(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest operation(backend); std::optional newest; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*operation).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -115,7 +116,7 @@ std::optional newestLogTxn(DB::Cas::Backend & backend, const DB::Cas: } if (!newest) return std::nullopt; - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest)); + const auto got = (*operation).read(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest), Retry::standard()); if (!got) return std::nullopt; return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), *newest); @@ -126,18 +127,19 @@ std::optional newestLogTxn(DB::Cas::Backend & backend, const DB::Cas: size_t committedRemovalCountForRef(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const RootNamespace & ns, const String & ref_name) { + DB::Cas::tests::OperationForTest operation(backend); size_t count = 0; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*operation).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); if (!parsed || parsed->life_id != DB::Cas::tests::fixture::fixtureLife(ns).incarnation || parsed->kind != RefObjectKind::Log) continue; - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), parsed->txn_id)); + const auto got = (*operation).read(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), parsed->txn_id), Retry::standard()); if (!got) continue; const RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), parsed->txn_id); diff --git a/src/Disks/tests/gtest_cas_ref_catalog.cpp b/src/Disks/tests/gtest_cas_ref_catalog.cpp index 11a4c029f139..8d4c364cb68f 100644 --- a/src/Disks/tests/gtest_cas_ref_catalog.cpp +++ b/src/Disks/tests/gtest_cas_ref_catalog.cpp @@ -1,5 +1,6 @@ #include "cas_format_test_battery.h" #include "cas_test_helpers.h" +#include #include #include #include @@ -11,12 +12,24 @@ #include #include #include +#include +#include #include +#include +#include +#include +#include #include +#include #include #include +#include + +#include using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; namespace ProfileEvents { @@ -34,7 +47,8 @@ class GcRoundPlanSignatureAccess public: using FoldSignature = decltype(&Gc::fold); using ExpectedFoldSignature = Gc::FoldResult (Gc::*)( - GcState &, Token &, RoundReport &, uint64_t, const RefPlan &, UniversePolicy, GcRoundWorkBudget &); + GcState &, std::optional &, RoundReport &, uint64_t, const RefPlan &, UniversePolicy, + GcRoundWorkBudget &); using BuilderSignature = decltype(&buildRefWalkPlan); using ExpectedBuilderSignature = RefPlan (*)(RoundInput &&); @@ -51,32 +65,32 @@ namespace DB::ErrorCodes extern const int LIMIT_EXCEEDED; extern const int NETWORK_ERROR; extern const int BAD_ARGUMENTS; + extern const int S3_ERROR; } namespace { -/// Hand-builds one raw "ent" line, bypassing `encodeRefCatalog` entirely -- used by the decode-side +/// Hand-builds one raw `entry` line, bypassing `encodeRefCatalog` entirely -- used by the decode-side /// rejection tests, which must exercise bytes the encoder itself would refuse to produce. -String rawEntLine(const String & ns, const String & state, const String & inc_hex, +String rawEntryLine(const String & ns, const String & state, const String & inc_hex, std::optional> creator = std::nullopt) { if (!creator) - return fmt::format(R"({{"k":"ent","ns":"{}","st":"{}","inc":"{}"}})", ns, state, inc_hex); + return fmt::format(R"({{"kind":"entry","ns":"{}","state":"{}","life":"{}"}})", ns, state, inc_hex); const auto & [srid, we, fg] = *creator; - return fmt::format(R"({{"k":"ent","ns":"{}","st":"{}","inc":"{}","csr":"{}","cwe":"{}","cfg":"{}"}})", + return fmt::format(R"({{"kind":"entry","ns":"{}","state":"{}","life":"{}","creator":"{}","creator_epoch":"{}","creator_fence":"{}"}})", ns, state, inc_hex, srid, we, fg); } -/// Wraps `ent_lines` in the header/trailer a real `cas_ref_catalog` object carries. `v:1` always -/// passes the header gate (any version <= the build's `G_BUILD` does), matching the convention -/// `gtest_cas_fold_seal_format.cpp`'s `RejectsOutOfRangeNsCleanupState` uses for the same reason. -String rawCatalog(const std::vector & ent_lines) +/// Wraps `entry_lines` in the header/trailer a real `cas_ref_catalog` object carries. `v:1` always +/// passes the header gate because any version <= the build's `G_BUILD` does. +String rawCatalog(const std::vector & entry_lines) { String out = R"({"type":"cas_ref_catalog","v":1})" "\n"; - for (const String & l : ent_lines) + for (const String & l : entry_lines) out += l + "\n"; - out += fmt::format("{{\"n\":{}}}\n", ent_lines.size()); + out += fmt::format("{{\"n\":{}}}\n", entry_lines.size()); return out; } @@ -84,7 +98,7 @@ String withRemovalStartedRound(String line, uint64_t round) { const size_t close = line.rfind('}'); EXPECT_NE(close, String::npos); - line.insert(close, fmt::format(R"(,"rsr":"{}")", round)); + line.insert(close, fmt::format(R"(,"remove_round":"{}")", round)); return line; } @@ -103,79 +117,128 @@ CatalogEntry entryInState(const String & ns, NsState state, uint64_t inc) return entry; } -class EraseWinnerBackend final : public DB::Cas::tests::CountingBackend +/// Per-key counts of the WRITE primitive. `CountingBackend` counts reads, heads and lists per key but +/// only totals for writes, and its legacy per-verb counters never see a caller that speaks the +/// primitives -- which every catalog writer below does. +class WriteCountingBackend : public DB::Cas::tests::CountingBackend { public: - using CountingBackend::casPut; - using CountingBackend::get; + enum class Verb { Read, Write }; - void replaceOnNextCatalogCas(const String & key, std::optional replacement_) + uint64_t writes(const String & key) const { - catalog_key = key; - replacement = std::move(replacement_); - armed = true; + std::lock_guard lock(write_count_mutex); + const auto it = write_counts.find(key); + return it == write_counts.end() ? 0 : it->second; } - bool fenceMoved() const { return fence_moved; } + /// The ordered READ/WRITE sequence issued against `key` since this backend was created. A count + /// alone cannot tell a settled-then-reissued attempt from a blind reissue that happened to read + /// more times; the order is what a caller actually needs to pin. + std::vector journalFor(const String & key) const + { + std::lock_guard lock(write_count_mutex); + std::vector filtered; + for (const auto & [journaled_key, verb] : journal) + if (journaled_key == key) + filtered.push_back(verb); + return filtered; + } - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (armed && key == catalog_key) { - armed = false; - const auto current = CountingBackend::get(key); - if (!current) - throw std::runtime_error("test fixture lost mandatory catalog"); - RefCatalog winner_catalog; - if (replacement) - winner_catalog.entries.push_back(*replacement); - const CasResult winner = CountingBackend::casPut( - key, encodeRefCatalog(winner_catalog), current->token, meta); - if (winner.outcome != CasOutcome::Committed) - throw std::runtime_error("test fixture winner failed to replace catalog"); - fence_moved = true; + std::lock_guard lock(write_count_mutex); + journal.emplace_back(key, Verb::Read); } - return CountingBackend::casPut(key, bytes, expected, meta); + return CountingBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + { + std::lock_guard lock(write_count_mutex); + ++write_counts[key]; + journal.emplace_back(key, Verb::Write); + } + return CountingBackend::write(key, bytes, expected_value, access); } private: - String catalog_key; - std::optional replacement; - bool armed = false; - bool fence_moved = false; + mutable std::mutex write_count_mutex; + std::map write_counts; + std::vector> journal; }; -class CasPutThrowsOnceBackend final : public DB::Cas::tests::CountingBackend +/// A compact 'R'/'W' rendering of `WriteCountingBackend::journalFor`, so a mismatch prints as one +/// readable string rather than a wall of enum values. +String renderJournal(const std::vector & journal) { -public: - using CountingBackend::casPut; + String rendered; + for (const auto verb : journal) + rendered += verb == WriteCountingBackend::Verb::Read ? 'R' : 'W'; + return rendered; +} - void armCasPutThrow(const String & key) +/// Lands a competing catalog body under the erase's own attempt and withdraws this actor's admission +/// with it -- the concurrent winner an erase has to be resolved against, driven deterministically and +/// without a second thread. +class EraseWinnerBackend final : public WriteCountingBackend +{ +public: + void replaceOnNextCatalogWrite(const String & key, std::optional replacement_) { - throw_key = key; + catalog_key = key; + replacement = std::move(replacement_); armed = true; } - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + bool admitted() const { return !fence_moved; } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (armed && key == throw_key) + if (armed && key == catalog_key) { armed = false; - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, - "injected casPut failure during completed-removal erase"); + RefCatalog winner_catalog; + if (replacement) + winner_catalog.entries.push_back(*replacement); + /// Qualified, so the winner's own write is not counted as an attempt of the call under test. + const auto current = WriteCountingBackend::read(key, access); + if (!current) + throw std::runtime_error("test fixture lost mandatory catalog"); + const auto winner = CountingBackend::write( + key, encodeRefCatalog(winner_catalog), std::optional{current->value}, access); + if (!winner.has_value()) + throw std::runtime_error("test fixture winner failed to replace catalog"); + fence_moved = true; } - return CountingBackend::casPut(key, bytes, expected, meta); + return WriteCountingBackend::write(key, bytes, expected_value, access); } private: - String throw_key; + String catalog_key; + std::optional replacement; bool armed = false; + bool fence_moved = false; }; +/// The erase entry points require a liveness refresh because a real drain's liveness is a cached flag +/// its owner re-reads from the store. These fixtures' operations carry either no liveness or a direct +/// read of the fixture, so there is nothing cached for a refresh to update. Named rather than repeated +/// inline, so a site that DOES need a refresh cannot hide among the ones that do not. +void noAuthorityRefresh() {} + +/// Seeds one object, failing the current test rather than returning a value nobody checks. +void seedObject(CasOperation & op, const String & key, const String & bytes) +{ + ASSERT_TRUE(std::holds_alternative(op.create(key, bytes, Retry::standard()))); +} + class ScopedCasGcLogCapture { public: @@ -210,6 +273,8 @@ class ScopedCasGcLogCapture /// ---------- format-battery registration ---------- +CAS_BATTERY_COVERS(RefCatalog); + TEST(CASFormatBattery, RefCatalog) { RefCatalog c; @@ -221,12 +286,25 @@ TEST(CASFormatBattery, RefCatalog) [&] { return sealObject(FormatId::RefCatalog, encodeRefCatalog(c)); }, [](std::string_view s) { decodeRefCatalog(std::string(openObject(FormatId::RefCatalog, s))); }, currentFormatHeader("cas_ref_catalog") + - "{\"k\":\"ent\",\"ns\":\"a\",\"st\":\"creating\",\"inc\":\"00000000000000000000000000000001\"," - "\"csr\":\"srv1\",\"cwe\":\"5\",\"cfg\":\"2\"}\n" - "{\"k\":\"ent\",\"ns\":\"b\",\"st\":\"live\",\"inc\":\"00000000000000000000000000000002\"}\n" + "{\"kind\":\"entry\",\"ns\":\"a\",\"state\":\"creating\",\"life\":\"00000000000000000000000000000001\"," + "\"creator\":\"srv1\",\"creator_epoch\":\"5\",\"creator_fence\":\"2\"}\n" + "{\"kind\":\"entry\",\"ns\":\"b\",\"state\":\"live\",\"life\":\"00000000000000000000000000000002\"}\n" "{\"n\":2}\n"}); } +/// Closed-set pin: the three `NsState` words, walked through `magic_enum::enum_values`, which is what +/// proves the renderer and the parser consult the SAME table: a table entry missing altogether is +/// already a build error at the coverage assert, but two delegates drifting onto different tables is +/// not. +TEST(CASRefCatalogFormat, ClosedSetPinsNsStateWords) +{ + EXPECT_EQ(nsStateToWord(NsState::Creating), "creating"); + EXPECT_EQ(nsStateToWord(NsState::Live), "live"); + EXPECT_EQ(nsStateToWord(NsState::Removing), "removing"); + for (const auto s : magic_enum::enum_values()) + EXPECT_EQ(nsStateFromWord(nsStateToWord(s)), s); +} + /// ---------- codec round-trip ---------- TEST(CASRefCatalogFormat, RoundTripsAllThreeStates) @@ -260,15 +338,16 @@ TEST(CASRefCatalogFormat, RemovalStartedRoundIsRequiredExactlyForRemoving) .removal_started_round = 19}; const RefCatalog catalog{.entries = {removing}}; const String encoded = encodeRefCatalog(catalog); - EXPECT_NE(encoded.find("\"rsr\":\"19\""), String::npos); + EXPECT_NE(encoded.find("\"remove_round\":\"19\""), String::npos); + EXPECT_NE(encoded.find("\"state\":\"removing\""), String::npos); EXPECT_EQ(decodeRefCatalog(encoded), catalog); const String inc = "00000000000000000000000000000009"; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { (void)decodeRefCatalog(rawCatalog({rawEntLine("missing", "removing", inc)})); }); + [&] { (void)decodeRefCatalog(rawCatalog({rawEntryLine("missing", "removing", inc)})); }); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - (void)decodeRefCatalog(rawCatalog({withRemovalStartedRound(rawEntLine("forbidden", "live", inc), 21)})); + (void)decodeRefCatalog(rawCatalog({withRemovalStartedRound(rawEntryLine("forbidden", "live", inc), 21)})); }); } @@ -305,7 +384,9 @@ TEST(CASRefCatalogLifeIndex, DuplicatePhysicalIdsAreAmbiguousWithoutPoisoningUni /// candidate can be written. An unrelated unique point lookup remains available from the same cut. TEST(CASRefCatalogLifeIndex, AmbiguityStopsCatalogMutationButNotUnrelatedPointLookup) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); RefCatalog catalog; catalog.entries = { @@ -313,17 +394,17 @@ TEST(CASRefCatalogLifeIndex, AmbiguityStopsCatalogMutationButNotUnrelatedPointLo entryInState("b", NsState::Removing, 7), entryInState("c", NsState::Live, 9), }; - ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(catalog)).outcome, PutOutcome::Done); - const auto before = backend.get(layout.refCatalogKey()); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(catalog)); + const auto before = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(before); - EXPECT_THROW(CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) { return current; }), DB::Exception); - const auto after = backend.get(layout.refCatalogKey()); + EXPECT_THROW(CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { return current; }), DB::Exception); + const auto after = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(after); - EXPECT_EQ(after->token, before->token); + EXPECT_EQ(after->etag, before->etag); EXPECT_EQ(after->bytes, before->bytes); - const auto unique = CasRefCatalog::lifeIfCataloged(backend, layout, RootNamespace{"c"}); + const auto unique = CasRefCatalog::lifeIfCataloged(op, layout, RootNamespace{"c"}); ASSERT_TRUE(unique); EXPECT_EQ(unique->incarnation, UInt128{9}); } @@ -474,7 +555,7 @@ TEST(CASRefCatalogFormatDeathTest, EncodeRejectsLiveWithRemovalStartedRoundAbort #endif /// A namespace + creator server_root_id that both max out at their respective byte bounds (512 + -/// 255), escaped worst-case, land one "ent" line over the 4 KiB line cap (~4.7 KiB) -- reachable +/// 255), escaped worst-case, land one `entry` line over the 4 KiB line cap (~4.7 KiB) -- reachable /// because neither this codec nor `validateServerRootId` restricts the charset, only the length. /// The refusal must be `LIMIT_EXCEEDED` (a capacity refusal), not `LOGICAL_ERROR` (a bug report) -- /// `encodeFoldSeal`'s own `checkLineBytes` raises `LIMIT_EXCEEDED` for the identical shape of gate. @@ -493,60 +574,76 @@ TEST(CASRefCatalogFormat, EncodeLineOverCapRaisesLimitExceeded) TEST(CASRefCatalogFormat, DecodeRejectsDuplicateNamespace) { - const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(1))), - rawEntLine("a", "live", u128ToHex(UInt128(2)))}); + const String bad = rawCatalog({rawEntryLine("a", "live", u128ToHex(UInt128(1))), + rawEntryLine("a", "live", u128ToHex(UInt128(2)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsNonCanonicalOrder) { - const String bad = rawCatalog({rawEntLine("b", "live", u128ToHex(UInt128(1))), - rawEntLine("a", "live", u128ToHex(UInt128(2)))}); + const String bad = rawCatalog({rawEntryLine("b", "live", u128ToHex(UInt128(1))), + rawEntryLine("a", "live", u128ToHex(UInt128(2)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsCreatorPresentOnLive) { - const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(1)), + const String bad = rawCatalog({rawEntryLine("a", "live", u128ToHex(UInt128(1)), std::make_tuple(String("srv"), uint64_t(1), uint64_t(1)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsCreatorAbsentOnCreating) { - const String bad = rawCatalog({rawEntLine("a", "creating", u128ToHex(UInt128(1)))}); + const String bad = rawCatalog({rawEntryLine("a", "creating", u128ToHex(UInt128(1)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsZeroIncarnation) { - const String bad = rawCatalog({rawEntLine("a", "live", u128ToHex(UInt128(0)))}); + const String bad = rawCatalog({rawEntryLine("a", "live", u128ToHex(UInt128(0)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsNameOverByteBound) { const String too_long_ns(kMaxNamespaceBytes + 1, 'a'); - const String bad = rawCatalog({rawEntLine(too_long_ns, "live", u128ToHex(UInt128(1)))}); + const String bad = rawCatalog({rawEntryLine(too_long_ns, "live", u128ToHex(UInt128(1)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsUnknownState) { - const String bad = rawCatalog({rawEntLine("a", "bogus", u128ToHex(UInt128(1)))}); + const String bad = rawCatalog({rawEntryLine("a", "bogus", u128ToHex(UInt128(1)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } +TEST(CASRefCatalogFormat, DecodeRejectsUnknownEntryKey) +{ + const String bad = rawCatalog( + {R"({"kind":"entry","ns":"a","state":"live","life":"00000000000000000000000000000001","unknown":"x"})"}); + try + { + (void)decodeRefCatalog(bad); + FAIL() << "expected CORRUPTED_DATA"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_NE(e.message().find("unknown entry key"), String::npos) << e.message(); + } +} + TEST(CASRefCatalogFormat, DecodeRejectsEmptyNamespace) { - const String bad = rawCatalog({rawEntLine("", "live", u128ToHex(UInt128(1)))}); + const String bad = rawCatalog({rawEntryLine("", "live", u128ToHex(UInt128(1)))}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } TEST(CASRefCatalogFormat, DecodeRejectsMissingNamespaceKey) { /// No "ns" key at all -- must be refused exactly like an explicit empty one, not read as "". - const String bad = rawCatalog({R"({"k":"ent","st":"live","inc":")" + u128ToHex(UInt128(1)) + "\"}"}); + const String bad = rawCatalog({R"({"kind":"entry","state":"live","life":")" + u128ToHex(UInt128(1)) + "\"}"}); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCatalog(bad); }); } @@ -566,7 +663,7 @@ TEST(CASRefCatalogFormat, NsStateToWordRaisesLogicalErrorOnImpossibleValue) #if defined(DEBUG_OR_SANITIZER_BUILD) TEST(CASRefCatalogFormatDeathTest, NsStateToWordRaisesLogicalErrorOnImpossibleValueAborts) { - EXPECT_DEATH({ (void)nsStateToWord(static_cast(99)); }, "unknown ns state"); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange): the whole point of this test is an impossible enum value + EXPECT_DEATH({ (void)nsStateToWord(static_cast(99)); }, "outside the wire vocabulary"); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange): the whole point of this test is an impossible enum value } #endif @@ -584,16 +681,16 @@ TEST(CASRefCatalogFormat, RegistryRowIsControlStrictWithRawStorage) EXPECT_EQ(traits.object_cap, 256u * 1024u * 1024u); EXPECT_EQ(traits.line_cap, 4u * 1024u); EXPECT_EQ(traitsForType("cas_ref_catalog"), &traits); - /// Raw, so the key has no suffix: `Pool/CasRefCatalog.cpp` hands bytes to/from the backend - /// directly, bypassing `sealObject`/`openObject` because both are the identity under + /// Raw, so the key has no suffix: the catalog hands bytes directly to/from the backend, + /// bypassing `sealObject`/`openObject` because both are the identity under /// `CompressionPolicy::Never`. This line is the TRIPWIRE for that shortcut -- a policy flip to /// `Always` would silently write uncompressed bodies under a `.zst` key, which this assertion - /// catches first (see `CasRefCatalogFormat.h`'s comment on `encodeRefCatalog`). + /// catches first. EXPECT_EQ(storedSuffix(FormatId::RefCatalog), ""); EXPECT_EQ(traits.compression, CompressionPolicy::Never); } -/// ---------- capacity admission: per-predicate boundary tests [codex r2/r3 finding 9] ---------- +/// ---------- capacity admission: per-predicate boundary tests ---------- TEST(CASRefCatalogAdmission, Predicate1AcceptsEqualityRefusesCapPlusOne) { @@ -685,7 +782,7 @@ TEST(CASRefCatalogAdmission, ReservationCoversActualWidestLegalRowsAcrossDecimal { seal.ref_lives.emplace(std::numeric_limits::max() - i, RefLifeFoldState{ .coverage = RefCoverage{ - .classification = 4, + .classification = CoverageClass::Clamped, .last_folded_ref_id = RefTxnId{max, max}, .hold = RefHold{ .reason = HoldReason::UnconsumedSealCrossing, @@ -696,14 +793,14 @@ TEST(CASRefCatalogAdmission, ReservationCoversActualWidestLegalRowsAcrossDecimal } for (uint64_t shard = 0; shard < gc_shards; ++shard) { - /// Predicate 2 charges exactly `gc_shards` widest `btr` rows. This fixture is the maximum + /// Predicate 2 charges exactly `gc_shards` widest `blob_run` rows. This fixture is the maximum /// legal cardinality, not an optimistic producer convention: authoritative fold-seal /// grammar permits at most one run per shard and requires its canonical key to use seq 0. seal.blob_target_runs.push_back(RunRef{ .key = layout.blobTargetRunKey(max, max, shard, 0), .checksum = std::numeric_limits::max(), .shard = shard, - .generation = max}); + .key_generation = max}); seal.condemned_summary.emplace(shard, CondemnedSummary{ .condemned_total = max, .pending_total = max, @@ -745,13 +842,17 @@ TEST(CASRefCatalogAdmission, RemovalNeverRefusedEvenAtCapacity) for (uint64_t i = 0; i < max_entries; ++i) full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); - InMemoryBackend backend; - backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(full)); + auto backend = std::make_shared(); + + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + + CasOperation op = requests.admit(); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(full)); /// The removal transition (Live -> Removing) on one entry goes through the PLAIN update path /// (`casUpdate`, which runs no admission check at all) and succeeds even though the catalog is /// already at the point where ANY growth would be refused. - const RefCatalog after = CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) + const RefCatalog after = CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & cur) { RefCatalog next = cur; next.entries[0].state = NsState::Removing; @@ -762,36 +863,42 @@ TEST(CASRefCatalogAdmission, RemovalNeverRefusedEvenAtCapacity) EXPECT_EQ(after.entries[0].state, NsState::Removing); } -/// ---------- Pool/CasRefCatalog: token-CAS read / create / update / conflict-retry ---------- +/// ---------- Pool/CasRefCatalog: read / create / update / conflict-retry ---------- TEST(CASRefCatalog, ReadAbsentFailsClosed) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)CasRefCatalog::read(backend, layout); }); + DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)CasRefCatalog::read(op, layout); }); } TEST(CASRefCatalog, CasUpdateRefusesWhenAbsent) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) { return cur; }); + CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & cur) { return cur; }); }); - EXPECT_FALSE(backend.head(layout.refCatalogKey()).exists); + EXPECT_FALSE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()); } TEST(CASRefCatalog, CasUpdateAppliesOnTopOfExistingState) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); - const RefCatalog updated = CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & cur) + const RefCatalog updated = CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & cur) { RefCatalog next = cur; next.entries[0].state = NsState::Removing; @@ -813,22 +920,26 @@ TEST(CASRefCatalog, GenericCasUpdateCannotDeleteOrReplaceCatalogIdentity) { const Layout layout("p"); { - InMemoryBackend backend; - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { - (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog &) { return RefCatalog{}; }); + (void)CasRefCatalog::casUpdate(op, layout, [](const RefCatalog &) { return RefCatalog{}; }); }); } { - InMemoryBackend backend; - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { - (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + (void)CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; next.entries[0] = liveEntry("b", 2); @@ -844,21 +955,25 @@ TEST(CASRefCatalogDeathTest, GenericCasUpdateCannotDeleteOrReplaceCatalogIdentit { const Layout layout("p"); { - InMemoryBackend backend; - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); EXPECT_DEATH( - { (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog &) { return RefCatalog{}; }); }, + { (void)CasRefCatalog::casUpdate(op, layout, [](const RefCatalog &) { return RefCatalog{}; }); }, "cannot add or delete catalog entries"); } { - InMemoryBackend backend; - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); EXPECT_DEATH( { - (void)CasRefCatalog::casUpdate(backend, layout, [](const RefCatalog & current) + (void)CasRefCatalog::casUpdate(op, layout, [](const RefCatalog & current) { RefCatalog next = current; next.entries[0] = liveEntry("b", 2); @@ -872,15 +987,17 @@ TEST(CASRefCatalogDeathTest, GenericCasUpdateCannotDeleteOrReplaceCatalogIdentit TEST(CASRefCatalog, CasUpdateRetriesOnConflictAgainstFreshState) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); - backend.failNextCasPut(layout.refCatalogKey()); /// one-shot artificial Conflict on the next write + backend->refuseNextWrite(layout.refCatalogKey()); /// one-shot artificial Conflict on the next write int mutate_calls = 0; - const RefCatalog result = CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + const RefCatalog result = CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & cur) { ++mutate_calls; RefCatalog next = cur; @@ -893,66 +1010,64 @@ TEST(CASRefCatalog, CasUpdateRetriesOnConflictAgainstFreshState) ASSERT_EQ(result.entries.size(), 1u); EXPECT_EQ(result.entries[0].state, NsState::Removing); - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); EXPECT_EQ(snap.catalog, result); } -TEST(CASRefCatalog, BeginRemovingRechecksFenceAfterCatalogCasConflict) +TEST(CASRefCatalog, BeginRemovingRechecksAdmissionAfterACatalogConflict) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation reader = requests.admit(); const Layout layout("p"); const CatalogEntry observed = liveEntry("a", 1); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, observed); - - uint64_t current_fence_generation = 7; - size_t fence_checks = 0; - const auto outcome = CasRefCatalog::beginRemoving( - backend, layout, observed, /*removal_started_round*/ 13, /*admitted_generation*/ 7, - [&](uint64_t admitted_generation) - { - ++fence_checks; - if (admitted_generation != current_fence_generation) - throw std::runtime_error("stale catalog mutation fence"); - if (fence_checks == 1) - { - /// Move the caller fence after the first admission check and force that attempt's - /// catalog CAS to conflict. The next attempt must check the fence again before writing. - current_fence_generation = 8; - backend.failNextCasPut(layout.refCatalogKey()); - } - }); + CasRefCatalog::initializeEmptyForNewPool(reader, layout); + CasRefCatalog::casAdmitEntry(reader, layout, 1, observed); + const uint64_t writes_before = backend->writes(layout.refCatalogKey()); + + /// The transition's one attempt is refused, and this actor's admission is withdrawn as it is sent. + /// The refusal must end the call rather than start another attempt, and it must be reported as an + /// outcome rather than thrown. + bool admitted = true; + backend->refuseNextWrite(layout.refCatalogKey()); + backend->onBeforeWrite(layout.refCatalogKey(), [&admitted] { admitted = false; }); + CasOperation op = requests.admit([&admitted] { return admitted; }); + + const auto outcome = CasRefCatalog::beginRemoving(op, layout, observed, /*removal_started_round*/ 13); EXPECT_EQ(outcome, CasRefCatalog::BeginRemovingOutcome::FencedOut); - EXPECT_EQ(fence_checks, 2u); - const CasRefCatalog::Snapshot after = CasRefCatalog::read(backend, layout); - EXPECT_EQ(after.catalog.entries, std::vector{observed}); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_before + 1) + << "the refused attempt must not be followed by another"; + const CasRefCatalog::Snapshot after = CasRefCatalog::read(reader, layout); + EXPECT_EQ(after.catalog.entries, std::vector{observed}) + << "nothing may be written after the admission is gone"; } /// A re-read that finds the catalog genuinely ABSENT after it was previously observed present is a /// real concurrent delete, not a bootstrap -- `casUpdate` must refuse rather than silently create a /// fresh catalog containing only this one mutation's entry (which would drop every other namespace). /// Reproduced with a REAL delete (no fault injection needed): `mutate`'s first invocation deletes the -/// seeded object using the token `casUpdate`'s own initial read observed, so the loop's own `casPut` -/// against that now-stale token gets a genuine `Conflict`, and the follow-up re-read genuinely finds -/// the key absent. +/// seeded object at the incarnation the update's own initial read observed, so its conditional write +/// is refused and the read that settles the refusal genuinely finds the key absent. /// Missing mandatory authority raises `CORRUPTED_DATA`; the split remains only because the debug /// variant historically lived in the death-test suite. #ifndef DEBUG_OR_SANITIZER_BUILD TEST(CASRefCatalog, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTheCatalog) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); - const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(backend, layout); - ASSERT_TRUE(seeded.token.has_value()); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); + const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(op, layout); + ASSERT_TRUE(seeded.etag.has_value()); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & cur) { - backend.deleteExact(layout.refCatalogKey(), *seeded.token); + EXPECT_EQ(op.remove(layout.refCatalogKey(), *seeded.etag, Retry::standard()), Removal::Removed); RefCatalog next = cur; next.entries[0].state = NsState::Removing; next.entries[0].removal_started_round = 1; @@ -962,25 +1077,27 @@ TEST(CASRefCatalog, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTheCatalog) /// Nothing was written by the failed attempt: the object is exactly as the delete left it /// (absent), never a fresh single-entry catalog. - EXPECT_FALSE(backend.head(layout.refCatalogKey()).exists); + EXPECT_FALSE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()); } #endif #if defined(DEBUG_OR_SANITIZER_BUILD) TEST(CASRefCatalogDeathTest, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTheCatalogAborts) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); - const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(backend, layout); - ASSERT_TRUE(seeded.token.has_value()); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); + const CasRefCatalog::Snapshot seeded = CasRefCatalog::read(op, layout); + ASSERT_TRUE(seeded.etag.has_value()); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & cur) { - backend.deleteExact(layout.refCatalogKey(), *seeded.token); + EXPECT_EQ(op.remove(layout.refCatalogKey(), *seeded.etag, Retry::standard()), Removal::Removed); RefCatalog next = cur; next.entries[0].state = NsState::Removing; next.entries[0].removal_started_round = 1; @@ -990,36 +1107,49 @@ TEST(CASRefCatalogDeathTest, CasUpdateThrowsOnVanishMidRetryInsteadOfReplacingTh } #endif -/// The retry loop is bounded (the same live-lock brake `publishCkpt`/`allocateWriterEpoch` use on -/// their own contended token-CAS singletons) and ends in the typed retryable error, not an infinite -/// spin. `mutate` re-arms the one-shot conflict injection on every call, so every attempt fails. -TEST(CASRefCatalog, CasUpdateGivesUpAfterBoundedAttemptsWithRetryLaterError) +/// Persistent contention ends at the write policy's DEADLINE, with the typed retryable error -- not +/// after a fixed number of unslept iterations, and not in an infinite spin. `mutate` re-arms the +/// one-shot conflict injection on every call, so every attempt is refused; the injected clock reaches +/// the deadline without the test sleeping at all. +TEST(CASRefCatalog, CasUpdateEndsAtTheDeadlineNotAfterAHundredUnsleptIterations) { - InMemoryBackend backend; + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + const uint64_t started_at = clock.now; int mutate_calls = 0; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & cur) + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & cur) { ++mutate_calls; - backend.failNextCasPut(layout.refCatalogKey()); + backend->refuseNextWrite(layout.refCatalogKey()); RefCatalog next = cur; return next; }); }); EXPECT_GT(mutate_calls, 1); /// genuinely retried, not a single-shot failure + EXPECT_FALSE(clock.sleeps.empty()) << "every retry must back off; an unslept loop would burn the " + "deadline on requests instead of waiting out the contention"; + EXPECT_GE(clock.now - started_at, Retry::standard().window_ms - 5000) + << "the loop ended at the policy's own deadline, not at an iteration count"; } TEST(CASRefCatalog, CasAdmitEntryAcceptsAnOrdinaryCreation) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); - const RefCatalog created = CasRefCatalog::casAdmitEntry(backend, layout, 1, + const RefCatalog created = CasRefCatalog::casAdmitEntry(op, layout, 1, CatalogEntry{.ns = RootNamespace{"a"}, .state = NsState::Creating, .incarnation = UInt128(1), .creator = CreatorFence{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}}); ASSERT_EQ(created.entries.size(), 1u); @@ -1028,12 +1158,14 @@ TEST(CASRefCatalog, CasAdmitEntryAcceptsAnOrdinaryCreation) TEST(CASRefCatalog, CasAdmitEntryInsertsAtCanonicalPosition) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("b", 1)); - const RefCatalog after = CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("b", 1)); + const RefCatalog after = CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 2)); ASSERT_EQ(after.entries.size(), 2u); EXPECT_EQ(after.entries[0].ns.string(), "a"); /// inserted BEFORE "b", not appended EXPECT_EQ(after.entries[1].ns.string(), "b"); @@ -1045,29 +1177,35 @@ TEST(CASRefCatalog, CasAdmitEntryInsertsAtCanonicalPosition) #ifndef DEBUG_OR_SANITIZER_BUILD TEST(CASRefCatalog, CasAdmitEntryRejectsADuplicateNamespace) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, - [&] { CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); }); + [&] { CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 2)); }); } #endif #if defined(DEBUG_OR_SANITIZER_BUILD) TEST(CASRefCatalogDeathTest, CasAdmitEntryRejectsADuplicateNamespaceAborts) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 1)); - EXPECT_DEATH({ CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("a", 2)); }, "not canonically ordered"); + CasRefCatalog::initializeEmptyForNewPool(op, layout); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 1)); + EXPECT_DEATH({ CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("a", 2)); }, "not canonically ordered"); } #endif TEST(CASRefCatalog, CasAdmitEntryRefusesOverCapacity) { - InMemoryBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); Layout layout("p"); const uint64_t cap = foldSealCaps().object_cap; @@ -1082,96 +1220,90 @@ TEST(CASRefCatalog, CasAdmitEntryRefusesOverCapacity) full.entries.reserve(max_entries); for (uint64_t i = 0; i < max_entries; ++i) full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); - backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(full)); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(full)); /// Admitting ONE more namespace is refused -- the additive predicate is checked BEFORE the write, /// so the backend object is untouched. DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] { - CasRefCatalog::casAdmitEntry(backend, layout, 1, liveEntry("zzz", 999999999)); + CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("zzz", 999999999)); }); - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); EXPECT_EQ(snap.catalog.entries.size(), max_entries); } -TEST(CASRefCatalogRemoval, DeleteCompletedRemovingRequiresExactAdoptedProofAndLeaderFence) +TEST(CASRefCatalogRemoval, DeleteCompletedRemovingRequiresExactAdoptedProofAndAdmission) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 13}; - ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, - PutOutcome::Done); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); CasFoldSeal held_parent; held_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ .coverage = RefCoverage{ - .classification = 4, + .classification = CoverageClass::Clamped, .last_folded_ref_id = RefTxnId{1, 2}, .hold = RefHold{.offending_position = RefTxnId{1, 3}}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, held_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, held_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); CasFoldSeal mismatched_parent; mismatched_parent.ref_lives.emplace(UInt128{8}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, mismatched_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, mismatched_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); CatalogEntry live = removing; live.state = NsState::Live; live.removal_started_round.reset(); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, live, ready_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, live, ready_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); CatalogEntry creating = live; creating.state = NsState::Creating; creating.creator = CreatorFence{.server_root_id = "server", .writer_epoch = 3, .fence_generation = 4}; - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, creating, ready_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, creating, ready_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::ProofRefused); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Moved; }), + /// An operation whose admission is gone erases nothing and sends nothing. It is driven at a cut an + /// ADMITTED operation took, because a withdrawn one cannot issue the mandatory read that would + /// take one. + CasOperation withdrawn = requests.admit([] { return false; }); + EXPECT_EQ(CasRefCatalog::deleteCompletedRemovingAtSnapshot( + withdrawn, layout, CasRefCatalog::read(op, layout), removing, ready_parent, + noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::FencedOut); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + const uint64_t writes_before_erase = backend->writes(layout.refCatalogKey()); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, [](uint64_t generation) - { - EXPECT_EQ(generation, 5); - return CasRefCatalog::LeaderFenceStatus::Held; - }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); - EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 1); - EXPECT_EQ(backend.listTotal(), 0); - EXPECT_EQ(backend.deleteTotal(), 0); + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_before_erase + 1); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); } TEST(CASRefCatalogRemoval, ExactDeletionRefusesChangedEntryAndAdmissionCannotCarryRemoval) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, @@ -1179,7 +1311,7 @@ TEST(CASRefCatalogRemoval, ExactDeletionRefusesChangedEntryAndAdmissionCannotCar .removal_started_round = 13}; #ifndef DEBUG_OR_SANITIZER_BUILD DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, - [&] { (void)CasRefCatalog::casAdmitEntry(backend, layout, 1, removing); }); + [&] { (void)CasRefCatalog::casAdmitEntry(op, layout, 1, removing); }); #endif const CatalogEntry current{ @@ -1187,27 +1319,28 @@ TEST(CASRefCatalogRemoval, ExactDeletionRefusesChangedEntryAndAdmissionCannotCar .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 14}; - ASSERT_EQ(backend.putIfAbsent("unrelated", "sentinel").outcome, PutOutcome::Done); - ASSERT_EQ(backend.casPut(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {current}}), - CasRefCatalog::read(backend, layout).token).outcome, CasOutcome::Committed); + seedObject(op, "unrelated", "sentinel"); + ASSERT_TRUE(std::holds_alternative(op.replace( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {current}}), + *CasRefCatalog::read(op, layout).etag, Retry::standard()))); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }), + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh), CasRefCatalog::CompletedRemovingDeleteOutcome::EntryChanged); - EXPECT_EQ(CasRefCatalog::read(backend, layout).catalog.entries, std::vector{current}); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{current}); } #if defined(DEBUG_OR_SANITIZER_BUILD) TEST(CASRefCatalogRemovalDeathTest, AdmissionCannotCarryRemovalAborts) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); - CasRefCatalog::initializeEmptyForNewPool(backend, layout); + CasRefCatalog::initializeEmptyForNewPool(op, layout); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, @@ -1215,32 +1348,39 @@ TEST(CASRefCatalogRemovalDeathTest, AdmissionCannotCarryRemovalAborts) .removal_started_round = 13}; EXPECT_DEATH( - { (void)CasRefCatalog::casAdmitEntry(backend, layout, 1, removing); }, + { (void)CasRefCatalog::casAdmitEntry(op, layout, 1, removing); }, "cannot admit namespace.*directly as Removing"); } #endif -/// Mutation caught: deriving the control outcome from the resolution snapshot would turn a stale -/// leader's `FencedOut` into `Deleted` or `EntryChanged`. Resolution may prove the old life dead and -/// carry its invalidation, but it cannot restore the caller's authority to continue the GC round. +/// Mutation caught: deriving the control outcome from the resolution read would turn a stale leader's +/// `FencedOut` into `Deleted` or `EntryChanged`. A winner that replaced the catalog under this erase +/// cannot restore the caller's authority to continue the GC round, and the refusal stays a returned +/// outcome rather than an exception -- the caller distinguishes "I lost the round" from "I could not +/// talk to the store" by exactly that. +/// +/// An operation whose admission is gone cannot issue the resolution read either, so the result carries +/// the cut this call was GIVEN rather than a fresh one: the erase's own effect is deliberately left +/// unreported, because there is no admitted request left with which to learn it. TEST(CASRefCatalogRemoval, FenceLossRemainsControlOutcomeWhenWinnerRemovesOrReplacesLife) { for (const bool replace : {false, true}) { - EraseWinnerBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation reader = requests.admit(); + CasOperation op = requests.admit([&backend] { return backend->admitted(); }); const Layout layout(replace ? "replacement" : "absence"); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 13}; - ASSERT_EQ(backend.putIfAbsent( - layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, - PutOutcome::Done); + seedObject(reader, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); std::optional replacement; if (replace) @@ -1248,22 +1388,15 @@ TEST(CASRefCatalogRemoval, FenceLossRemainsControlOutcomeWhenWinnerRemovesOrRepl .ns = removing.ns, .state = NsState::Live, .incarnation = UInt128{8}}; - backend.replaceOnNextCatalogCas(layout.refCatalogKey(), replacement); + backend->replaceOnNextCatalogWrite(layout.refCatalogKey(), replacement); const CasRefCatalog::CompletedRemovingDeleteResult result - = CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, [&](uint64_t) - { - if (backend.fenceMoved()) - return CasRefCatalog::LeaderFenceStatus::Moved; - return CasRefCatalog::LeaderFenceStatus::Held; - }); + = CasRefCatalog::deleteCompletedRemovingAtSnapshot( + op, layout, CasRefCatalog::read(reader, layout), removing, ready_parent, + noAuthorityRefresh); EXPECT_EQ(result.outcome, CasRefCatalog::CompletedRemovingDeleteOutcome::FencedOut); - ASSERT_TRUE(result.invalidated_life); - EXPECT_EQ(*result.invalidated_life, - NamespaceLifeId::fromCatalogEntry(removing.ns, removing.incarnation)); - const RefCatalog current = CasRefCatalog::read(backend, layout).catalog; + const RefCatalog current = CasRefCatalog::read(reader, layout).catalog; if (replace) EXPECT_EQ(current.entries, std::vector{*replacement}); else @@ -1271,141 +1404,431 @@ TEST(CASRefCatalogRemoval, FenceLossRemainsControlOutcomeWhenWinnerRemovesOrRepl } } -/// Mutation caught: treating every authority-check exception as a moved fence hides corruption and -/// backend/decode failures. Before any CAS, inability to evaluate authority must propagate unchanged. -TEST(CASRefCatalogRemoval, NonFenceAuthorityExceptionPropagatesBeforeEraseCas) +/// A transient failure of the erase attempt is settled by the mandatory resolution read and reissued, +/// never concluded from. Treating it as ordinary non-convergence would hide a real backend fault behind +/// `ProofRefused`/`EntryChanged`; treating it as a landed erase would report a deletion nobody proved. +TEST(CASRefCatalogRemoval, ATransientEraseFailureIsResolvedByAReadAndReissued) { - DB::Cas::tests::CountingBackend backend; - const Layout layout("pre-cas-authority-error"); + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("cas-put-throw"); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 13}; - ASSERT_EQ(backend.putIfAbsent( - layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, - PutOutcome::Done); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] - { - (void)CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, [](uint64_t) -> CasRefCatalog::LeaderFenceStatus - { - throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, - "injected authority read failure before erase CAS"); - }); - }); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0u); + const size_t journal_before = backend->journalFor(layout.refCatalogKey()).size(); + backend->failNextWriteWith(layout.refCatalogKey(), std::make_exception_ptr( + Poco::TimeoutException("injected erase failure whose outcome never reached the caller"))); + + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh), + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + + /// A blind reissue still performs a baseline read, a post-write read and a verification read, + /// satisfying a bare count. The ORDER pins what a count cannot: the lost attempt's outcome is + /// settled by a read (the engine's own ambiguity resolution) before it is ever reissued, and the + /// reissue that lands is settled by the loop's mandatory resolution read in turn. Captured before + /// the test's own verification read below, which is not part of the call under test. + const std::vector full_journal = backend->journalFor(layout.refCatalogKey()); + const std::vector journal_since( + full_journal.begin() + static_cast(journal_before), full_journal.end()); + using Verb = WriteCountingBackend::Verb; + EXPECT_EQ(journal_since, (std::vector{Verb::Read, Verb::Write, Verb::Read, Verb::Write, Verb::Read})) + << "got " << renderJournal(journal_since) << "; a baseline read, the lost attempt settled by the " + << "engine's own resolution read BEFORE it is reissued, then the reissue settled by the loop's " + << "mandatory resolution read -- not a blind reissue that happens to read more times"; + + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), 3u) + << "the seed, the attempt whose outcome was lost, and the reissue that landed"; +} + +/// A refused precondition is the only thing the erase loop retries, and it PACES that retry on the +/// engine's own clock. Without the pause a contended catalog would spend its whole conflict budget in +/// back-to-back requests, which is the shape that turns one hot key into a request storm. +/// +/// Nothing else on this path sleeps -- the write returns a refused precondition without reissuing, and +/// no read fails -- so every wait the clock records is the loop's own. +TEST(CASRefCatalogRemoval, AConflictingEraseBacksOffBeforeItsRetry) +{ + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("erase-backoff"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + backend->refuseNextWrite(layout.refCatalogKey()); + + EXPECT_EQ(CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh), + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + ASSERT_EQ(clock.sleeps.size(), 1u) << "one refusal, so exactly one paced retry"; + EXPECT_LE(clock.sleeps.front(), 200u) << "the first reissue's full-jitter ceiling"; + EXPECT_EQ(backend->writes(layout.refCatalogKey()), 3u) + << "the seed, the refused erase, and the retry that landed"; +} + +/// A caller's liveness need not be a fact this loop can read. The GC drain's is a cached +/// leader-authority flag its owner refreshes by re-reading `gc/state`, so a leader deposed while an +/// erase is in flight leaves that flag stale -- and on an unrecoverable delete path one reading taken +/// before the first erase must not authorise the rest. Hence the refresh hook, run at the top of every +/// attempt. +/// +/// The winner here writes the SAME row back: the incarnation moves, so the erase is refused, and the +/// row survives for the retry the loop would otherwise send. The cached flag still answers "admitted" +/// at the post-write probe, deliberately -- that probe is not the subject, and leaving it stale is what +/// makes this test about the refresh and nothing else. +TEST(CASRefCatalogRemoval, TheEraseLoopRefreshesItsLivenessBeforeEveryAttempt) +{ + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation reader = requests.admit(); + bool cached_authority = true; + CasOperation op = requests.admit([&cached_authority] { return cached_authority; }); + const Layout layout("erase-refresh"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + seedObject(reader, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + backend->replaceOnNextCatalogWrite(layout.refCatalogKey(), removing); + const uint64_t writes_after_seed = backend->writes(layout.refCatalogKey()); + size_t refreshes = 0; + + const CasRefCatalog::CompletedRemovingDeleteResult result + = CasRefCatalog::deleteCompletedRemovingAtSnapshot( + op, layout, CasRefCatalog::read(reader, layout), removing, ready_parent, + [&] { ++refreshes; cached_authority = backend->admitted(); }); + + EXPECT_EQ(result.outcome, CasRefCatalog::CompletedRemovingDeleteOutcome::FencedOut); + EXPECT_EQ(refreshes, 2u) << "once before the attempt it sent, once before the one it did not"; + EXPECT_EQ(backend->writes(layout.refCatalogKey()) - writes_after_seed, 1u) + << "the one refused erase; the paced retry was abandoned before it reached the store"; + EXPECT_EQ(CasRefCatalog::read(reader, layout).catalog.entries, std::vector{removing}) + << "the row a deposed leader must not erase is still there"; } -/// The post-CAS authority check is distinct: the erase may already be durable and its mandatory -/// resolution complete, but inability to evaluate authority is still the original error, not -/// `FencedOut`. +/// The refresh hook can fail the same way the read it wraps can -- and a failure of it is not a +/// negative liveness answer to fold into `FencedOut`; it is a fact this call cannot evaluate, so it +/// must escape rather than be swallowed into any of the loop's own outcomes. This is the case where +/// the hook fails before the loop has sent anything at all: no erase may reach the store on an +/// authority this call could not even ask about. +TEST(CASRefCatalogRemoval, NonFenceAuthorityExceptionPropagatesBeforeEraseCas) +{ + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("erase-refresh-throws-before"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const uint64_t writes_after_seed = backend->writes(layout.refCatalogKey()); + + EXPECT_THROW( + CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, + [] { throw std::runtime_error("injected authority-refresh failure before the first attempt"); }), + std::runtime_error); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_after_seed) + << "an authority this call could not even ask about must not send an erase"; + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{removing}) + << "nothing changed under an authority failure that reached no attempt"; +} + +/// The counterpart to the case above: the hook fails AFTER the loop has already sent one erase and +/// resolved it (a refusal, so the row is provably unchanged), on the refresh that would gate the next +/// attempt. The failure must still escape rather than be read as the fenced-out liveness answer this +/// class's default `admitted()` would otherwise report, and -- exactly as when it fails up front -- no +/// further erase may reach the store on an authority this call could not evaluate. TEST(CASRefCatalogRemoval, NonFenceAuthorityExceptionPropagatesAfterEraseResolution) { - DB::Cas::tests::CountingBackend backend; - const Layout layout("post-cas-authority-error"); + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("erase-refresh-throws-after"); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 13}; - ASSERT_EQ(backend.putIfAbsent( - layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, - PutOutcome::Done); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - size_t authority_checks = 0; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] + backend->refuseNextWrite(layout.refCatalogKey()); + const uint64_t writes_after_seed = backend->writes(layout.refCatalogKey()); + + size_t refreshes = 0; + EXPECT_THROW( + CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, [&] + { + ++refreshes; + if (refreshes > 1) + throw std::runtime_error("injected authority-refresh failure after the erase resolved"); + }), + std::runtime_error); + EXPECT_EQ(refreshes, 2u) << "once before the refused attempt, once before the retry it never sent"; + EXPECT_EQ(backend->writes(layout.refCatalogKey()) - writes_after_seed, 1u) + << "the one refused erase; the retry the resolution read would have paced never reached the store"; + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{removing}) + << "the row a failed authority refresh must not let a later attempt erase is still there"; +} + +/// The loop iterates on ONE alternative and one only: a refused precondition. Every other non-committed +/// answer is terminal for the call and leaves through the same throw, so a future retry added for any +/// of them would be retrying a write whose fate the store already settled. `Refused` is the alternative +/// a test can construct exactly; `GaveUp{FenceLost}` cannot reach this arm at all, because the +/// admission probe immediately after the write returns `FencedOut` first. +#if USE_AWS_S3 +TEST(CASRefCatalogRemoval, AStoreRefusalEndsTheEraseLoopInsteadOfRetryingIt) +{ + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("erase-refused"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const uint64_t writes_after_seed = backend->writes(layout.refCatalogKey()); + + /// A malformed request is one of the answers that prove the write never applied, so the engine + /// reports it as a refusal rather than settling it by a read. + backend->failNextWriteWith(layout.refCatalogKey(), std::make_exception_ptr( + DB::S3Exception("injected malformed erase request", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"))); + + /// The store's own code, not a class of this module's choosing: `orThrow` re-raises a refusal + /// under the code the store gave, so an operator sees what the store actually said. + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { - (void)CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, [&](uint64_t) - { - if (++authority_checks == 2) - throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, - "injected authority read failure after erase resolution"); - return CasRefCatalog::LeaderFenceStatus::Held; - }); + (void)CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh); }); - EXPECT_EQ(authority_checks, 2u); - EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_after_seed + 1) + << "a refusal is terminal: the loop must not send a second erase"; + EXPECT_TRUE(clock.sleeps.empty()) << "and must not pace a retry it is not going to make"; + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{removing}) + << "the refused erase changed nothing"; } +#endif -/// Mutation caught: swallowing a synchronous `casPut` exception raised during the erase attempt -/// itself (as opposed to the authority/fence check) and treating it as ordinary non-convergence -/// would hide a real backend fault behind ProofRefused/EntryChanged, and would skip the mandatory -/// resolution read that this branch's siblings above already prove runs before any conclusion. -TEST(CASRefCatalogRemoval, CasPutExceptionPropagatesAfterMandatoryResolution) -{ - CasPutThrowsOnceBackend backend; - const Layout layout("cas-put-throw"); +/// The response to a conditional erase is not authority for what became durable, and this is the case +/// that makes that concrete: the attempt commits, a concurrent writer puts the row back, and the +/// mandatory resolution read contradicts the commit. Believing the response would report a namespace +/// deleted while its row is still cataloged, so the call fails retry-later instead. +TEST(CASRefCatalogRemoval, ACommitTheResolutionReadContradictsFailsRetryLaterInsteadOfReportingDeleted) +{ + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + CasOperation restorer = requests.admit(); + const Layout layout("erase-contradicted"); const CatalogEntry removing{ .ns = RootNamespace{"a"}, .state = NsState::Removing, .incarnation = UInt128{7}, .removal_started_round = 13}; - ASSERT_EQ(backend.putIfAbsent( - layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})).outcome, - PutOutcome::Done); + const String seeded_catalog = encodeRefCatalog(RefCatalog{.entries = {removing}}); + seedObject(op, layout.refCatalogKey(), seeded_catalog); CasFoldSeal ready_parent; ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 2}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); - backend.armCasPutThrow(layout.refCatalogKey()); + /// Fires once the erase is durable and before its answer is returned, so the call really does see + /// `Committed` and really does find the row back when it resolves. + bool restored = false; + backend->onWriteCommitted(layout.refCatalogKey(), [&] + { + if (restored) + return; + restored = true; + (void)restorer.readModifyWrite(layout.refCatalogKey(), + [&](const std::optional &) -> std::optional { return seeded_catalog; }, + Retry::standard()); + }); - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + String message; + try + { + (void)CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh); + ADD_FAILURE() << "a commit the resolution read contradicts must not be reported as a deletion"; + } + catch (const DB::Exception & e) { - (void)CasRefCatalog::deleteCompletedRemoving( - backend, layout, removing, ready_parent, 5, [](uint64_t) + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + message = e.message(); + } + EXPECT_TRUE(restored) << "the concurrent restore never ran, so nothing was contradicted"; + EXPECT_NE(message.find("reported committed"), String::npos) << message; + EXPECT_NE(message.find(u128ToHex(removing.incarnation)), String::npos) + << "the message must name the life still observed: " << message; +} + +/// The loop captures ONE bound before it starts and every call it makes shares it, so a permanently +/// contended catalog gives up "retry later" inside one standard window rather than spending a fresh +/// window per verb across a hundred paced iterations -- which is hours against a document that +/// promises ninety seconds. Every erase is refused and the injected clock absorbs every paced retry, +/// so what ends the call is visible in the virtual time it took. The attempt cap stays as the +/// secondary bound; it is not what ends this call. +TEST(CASRefCatalogRemoval, PerpetualConflictGivesUpWithinOneWindowNotAtTheAttemptCap) +{ + class AlwaysRefusesCatalogWrites final : public WriteCountingBackend + { + public: + String refused_key; + + uint64_t refusedAttempts() const { return refused_attempts; } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + if (key == refused_key) { - return CasRefCatalog::LeaderFenceStatus::Held; - }); - }); - /// The mandatory resolution read ran before the rethrow: the exact old row is still present, - /// unchanged by the failed attempt. - const RefCatalog current = CasRefCatalog::read(backend, layout).catalog; - EXPECT_EQ(current.entries, std::vector{removing}); + /// A genuine refused precondition never applies -- unlike `WriteCountingBackend::write`, + /// which both counts AND delegates to the real, LANDING write. Counting this attempt + /// through that base would apply it to storage and then lie about the outcome, which + /// made the erase loop's very first attempt land for real and see `Deleted` instead of + /// the perpetual conflict this backend's name promises. Count it here instead, and + /// return only the refusal. + ++refused_attempts; + return std::unexpected(DB::Cas::Backend::RawConflict{}); + } + return WriteCountingBackend::write(key, bytes, expected_value, access); + } + + private: + uint64_t refused_attempts = 0; + }; + + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + const Layout layout("erase-cap"); + const CatalogEntry removing{ + .ns = RootNamespace{"a"}, + .state = NsState::Removing, + .incarnation = UInt128{7}, + .removal_started_round = 13}; + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}})); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + backend->refused_key = layout.refCatalogKey(); + + const uint64_t start = clock.now; + String message; + try + { + (void)CasRefCatalog::deleteCompletedRemoving(op, layout, removing, ready_parent, noAuthorityRefresh); + ADD_FAILURE() << "a permanently refused erase must not return an outcome"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); + message = e.message(); + } + EXPECT_NE(message.find("deadline"), String::npos) + << "the shared bound, not the attempt cap, is what ended this call: " << message; + /// Two windows, so the assertion survives the jitter of the paced retries while still failing a + /// loop that binds a fresh window per iteration -- that one spends minutes here. + EXPECT_LT(clock.now - start, 2 * 90'000u) << "the loop outlived the window it captured"; + EXPECT_LT(backend->refusedAttempts(), 100u) << "the attempt cap must not be what ends this call"; + /// Greater than one is what proves the loop iterated rather than failing on its first attempt. + EXPECT_GT(backend->refusedAttempts(), 1u); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{removing}); } TEST(CASRefCatalogRemoval, CancelStalledCreatingRequiresExactRowAndTerminalCreatorFence) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); const CatalogEntry creating{ .ns = RootNamespace{"a"}, .state = NsState::Creating, .incarnation = UInt128{7}, .creator = CreatorFence{.server_root_id = "server", .writer_epoch = 3, .fence_generation = 4}}; - ASSERT_EQ(backend.putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {creating}})).outcome, - PutOutcome::Done); + seedObject(op, layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {creating}})); + const uint64_t writes_after_seed = backend->writes(layout.refCatalogKey()); EXPECT_EQ(CasRefCatalog::cancelStalledCreating( - backend, layout, creating, [](const CreatorFence &) { return false; }, 5, [](uint64_t) {}), + op, layout, creating, [](const CreatorFence &) { return false; }), CasRefCatalog::StalledCreatingCancelOutcome::CreatorFenceStillLive); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_after_seed); CatalogEntry stale = creating; stale.creator->writer_epoch = 2; EXPECT_EQ(CasRefCatalog::cancelStalledCreating( - backend, layout, stale, [](const CreatorFence &) { return true; }, 5, [](uint64_t) {}), + op, layout, stale, [](const CreatorFence &) { return true; }), CasRefCatalog::StalledCreatingCancelOutcome::EntryChanged); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 0); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_after_seed); EXPECT_EQ(CasRefCatalog::cancelStalledCreating( - backend, layout, creating, [](const CreatorFence &) { return true; }, 5, [](uint64_t) {}), + op, layout, creating, [](const CreatorFence &) { return true; }), CasRefCatalog::StalledCreatingCancelOutcome::Cancelled); - EXPECT_TRUE(CasRefCatalog::read(backend, layout).catalog.entries.empty()); - EXPECT_EQ(backend.casPutCount(layout.refCatalogKey()), 1); - EXPECT_EQ(backend.listTotal(), 0); - EXPECT_EQ(backend.deleteTotal(), 0); + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), writes_after_seed + 1); + EXPECT_EQ(backend->listTotal(), 0u); + EXPECT_EQ(backend->deleteTotal(), 0u); } TEST(CASGCRefWalkPlan, CatalogIsSoleRowAdmissionAuthorityAcrossOrdinaryAndRebuildInputs) @@ -1425,16 +1848,16 @@ TEST(CASGCRefWalkPlan, CatalogIsSoleRowAdmissionAuthorityAcrossOrdinaryAndRebuil .removal_started_round = 8}, }; const CasRefCatalog::Snapshot cut{ - .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + .catalog = catalog, .etag = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; RefScanSummary ordinary_scan; ordinary_scan.parent_ref_lives.emplace(UInt128{1}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}}); ordinary_scan.parent_ref_lives.emplace(UInt128{3}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{3, 3}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{3, 3}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{3, 3}}}); ordinary_scan.parent_ref_lives.emplace(UInt128{4}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{4, 4}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{4, 4}}}); ordinary_scan.listed_lives = {UInt128{1}, UInt128{2}, UInt128{4}}; ordinary_scan.holds.emplace(UInt128{1}, RefHold{.offending_position = RefTxnId{1, 2}}); ordinary_scan.holds.emplace(UInt128{2}, RefHold{.offending_position = RefTxnId{2, 2}}); @@ -1446,7 +1869,7 @@ TEST(CASGCRefWalkPlan, CatalogIsSoleRowAdmissionAuthorityAcrossOrdinaryAndRebuil RefScanSummary rebuild_scan; rebuild_scan.parent_ref_lives.emplace(UInt128{1}, ordinary_scan.parent_ref_lives.at(UInt128{1})); rebuild_scan.parent_ref_lives.emplace(UInt128{5}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{5, 5}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{5, 5}}}); rebuild_scan.listed_lives = {UInt128{1}, UInt128{3}, UInt128{5}}; rebuild_scan.holds.emplace(UInt128{1}, RefHold{.offending_position = RefTxnId{1, 3}}); rebuild_scan.holds.emplace(UInt128{3}, RefHold{.offending_position = RefTxnId{3, 4}}); @@ -1538,7 +1961,7 @@ TEST(CASGCStuckRemoval, BoundaryAndAbsentVersusUnreadableMessagesAreExact) EXPECT_NE(absent->find("terminal has not folded"), String::npos); EXPECT_EQ(absent->find("/_log/"), String::npos) << "an absent terminal has no exact id to name"; - row.fold_state.coverage.classification = 4; + row.fold_state.coverage.classification = CoverageClass::Clamped; row.fold_state.coverage.hold = RefHold{ .reason = HoldReason::BodyUndecodable, .offending_position = RefTxnId{5, 6}}; @@ -1554,7 +1977,9 @@ TEST(CASGCStuckRemoval, BoundaryAndAbsentVersusUnreadableMessagesAreExact) TEST(CASGCStuckRemoval, DiagnosticDoesNotAppendOrMutateBackend) { - DB::Cas::tests::CountingBackend backend; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout("p"); const RefWalkPlanRow row{ .life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"removing"}, UInt128{7}), @@ -1564,21 +1989,21 @@ TEST(CASGCStuckRemoval, DiagnosticDoesNotAppendOrMutateBackend) .listed_hint = false, .checkpoint_observation = std::nullopt, .tail_observation = std::nullopt}; - const uint64_t puts_before = backend.putTotal(); - const uint64_t cas_before = backend.casPutTotal(); + const uint64_t writes_before = backend->writeTotal(); EXPECT_TRUE(stuckRemovalWarning(row, 11, 10, layout)); - EXPECT_EQ(backend.putTotal(), puts_before); - EXPECT_EQ(backend.casPutTotal(), cas_before); - EXPECT_EQ(backend.deleteTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), writes_before); + EXPECT_EQ(backend->deleteTotal(), 0u); } TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .gc_stuck_removal_rounds = 10}); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout & layout = store->layout(); const UInt128 gc_id{99}; const UInt128 life_id{7}; @@ -1588,24 +2013,24 @@ TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) .state = NsState::Removing, .incarnation = life_id, .removal_started_round = 1}; - const auto catalog = backend->get(layout.refCatalogKey()); + const auto catalog = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog); - ASSERT_EQ(backend->casPut( - layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}}), catalog->token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(op.replace( + layout.refCatalogKey(), encodeRefCatalog(RefCatalog{.entries = {removing}}), + catalog->etag, Retry::standard()))); CasFoldSeal seal; seal.generation = 1; seal.ref_lives.emplace(life_id, RefLifeFoldState{ .coverage = RefCoverage{ - .classification = 4, + .classification = CoverageClass::Clamped, .hold = RefHold{ .reason = HoldReason::BodyUndecodable, .offending_position = RefTxnId{5, 6}, .retry_count = 0, .next_retry_round = 12}}}); seal.condemned_summary[0] = CondemnedSummary{}; - ASSERT_EQ(backend->putIfAbsent(layout.foldSealKey(1, 1), encodeFoldSeal(seal)).outcome, PutOutcome::Done); + seedObject(op, layout.foldSealKey(1, 1), encodeFoldSeal(seal)); GcState state; state.lease = GcLease{.owner = gc_id, .seq = 1}; @@ -1613,13 +2038,13 @@ TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) state.gc_shards = 1; state.snap_generation = 1; state.snap_attempt = 1; - ASSERT_EQ(backend->putIfAbsent(layout.gcStateKey(), encodeGcState(state)).outcome, PutOutcome::Done); + seedObject(op, layout.gcStateKey(), encodeGcState(state)); const uint64_t signals_before = ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(removing.ns, life_id); const String unreadable_ref_log_key = layout.refLogKey(life, RefTxnId{5, 6}); - const uint64_t append_puts_before = backend->putCount(unreadable_ref_log_key); + const uint64_t append_writes_before = backend->writes(unreadable_ref_log_key); ScopedCasGcLogCapture log_capture; Gc first_process(store, gc_id); EXPECT_TRUE(first_process.runRegularRound().acquired_lease); @@ -1627,7 +2052,7 @@ TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) EXPECT_TRUE(restarted_process.runRegularRound().acquired_lease); EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load() - signals_before, 2u); - EXPECT_EQ(backend->putCount(unreadable_ref_log_key), append_puts_before) + EXPECT_EQ(backend->writes(unreadable_ref_log_key), append_writes_before) << "the diagnostic cannot append the unreadable ref log"; const String captured = log_capture.captured(); EXPECT_EQ(std::count(captured.begin(), captured.end(), '\n'), 2u); @@ -1653,12 +2078,12 @@ TEST(CASGCRefWalkPlan, UnmatchedAdoptedParentLifeIsObservedWithoutEnteringThePla hexToU128("fedcba98765432100123456789abcdef"); RefCatalog catalog{.entries = {liveEntry("live", 2)}}; const CasRefCatalog::Snapshot cut{ - .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + .catalog = catalog, .etag = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; RefScanSummary scan; scan.parent_ref_lives.emplace(current_life, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{2, 3}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{2, 3}}}); scan.parent_ref_lives.emplace(unmatched_life, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{9, 9}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{9, 9}}}); const uint64_t events_before = ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load(); @@ -1689,12 +2114,12 @@ TEST(CASGCRefPlan, RoundInputOwnsObservationsAndSuccessorStateCannotChangePlan) RefCatalog catalog; catalog.entries = {liveEntry("live", 2)}; CasRefCatalog::Snapshot cut{ - .catalog = catalog, .token = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; + .catalog = catalog, .etag = std::nullopt, .life_index = CatalogLifeIndex(catalog)}; RefScanSummary observations; observations.max_log_by_life.emplace(UInt128{2}, RefTxnId{2, 7}); observations.parent_ref_lives.emplace(UInt128{2}, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{2, 3}}}); + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{2, 3}}}); const RefPlan plan = tests::buildRefWalkPlanForTest(observations, cut); @@ -1716,3 +2141,650 @@ TEST(CASGCRefPlan, RoundInputOwnsObservationsAndSuccessorStateCannotChangePlan) EXPECT_EQ(plan.row(UInt128{2}).fold_state.coverage.last_folded_ref_id, (RefTxnId{2, 3})); EXPECT_FALSE(plan.contains(UInt128{9})); } + +/// ---------- Pool/CasRefCatalog: mutations through the pool's hot-key lane ---------- + +namespace ProfileEvents +{ + extern const Event CASHotKeyReadStarts; + extern const Event CASHotKeyCacheVerdictsReread; + extern const Event CASRequestConflictPause; + extern const Event CASRequestReissue; +} + +namespace +{ + +#if USE_AWS_S3 +std::exception_ptr s3Error(Aws::S3::S3Errors code, const String & name) +{ + return std::make_exception_ptr(DB::S3Exception("the store answered " + name, code, name)); +} +#endif + +/// A pool's own engine: a `CasRequests` over the pool's shared lane with a cache, on a fake clock, +/// beside a plain `CasRequests` that stands for another server. +struct PoolAndExternal +{ + explicit PoolAndExternal(std::shared_ptr backend_) + : backend(std::move(backend_)) + , hot_keys(16ULL << 20) + , pool(backend, Fence::open(), clock.nowFn(), clock.sleepFn(), &hot_keys) + , external(DB::Cas::tests::openRequestsForTest(backend)) + { + } + std::shared_ptr backend; + FakeClock clock; + CasHotKeys hot_keys; + CasRequests pool; + CasRequests external; +}; + +uint64_t eventCount(ProfileEvents::Event event) +{ + return ProfileEvents::global_counters[event].load(); +} + +} + +TEST(CASRefCatalog, AStaleHintNeverProducesAFalseEntryChanged) +{ + /// The pool's last catalog write left "a" Creating; another server completed it to Live. The + /// caller reads fresh, sees Live, and calls beginRemoving: the cached Creating row differs from + /// what it observed and would say EntryChanged. That verdict is not delivered. + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, creating); /// through the lane: the cache holds Creating + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::completeCreation(other, layout, creating), CasRefCatalog::NamespaceCreationOutcome::Live); + + const CatalogEntry observed = CasRefCatalog::read(other, layout).catalog.entries.at(0); + ASSERT_EQ(observed.state, NsState::Live); + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + const uint64_t rereads_before = eventCount(ProfileEvents::CASHotKeyCacheVerdictsReread); + EXPECT_EQ(CasRefCatalog::beginRemoving(pool_op, layout, observed, 7), CasRefCatalog::BeginRemovingOutcome::Transitioned); + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()) - reads_before, 1u) << "one lane read"; + EXPECT_EQ(eventCount(ProfileEvents::CASHotKeyCacheVerdictsReread) - rereads_before, 1u); + EXPECT_EQ(CasRefCatalog::read(other, layout).catalog.entries.at(0).state, NsState::Removing); +} + +TEST(CASRefCatalog, ATrueRefusalOnTheReadIsTheSameAsWithoutACache) +{ + /// The row really changed: the re-rendered verdict is the one a cache-less call renders. + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, creating); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::completeCreation(other, layout, creating), CasRefCatalog::NamespaceCreationOutcome::Live); + const uint64_t writes_before = f.backend->writeCount(layout.refCatalogKey()); + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + /// The caller believes it observed a Creating row under another incarnation. The pool's hint + /// (Creating, incarnation 1) differs from it, so the verdict is rendered on the hint and not + /// delivered; the read (Live, incarnation 1) differs from it too, and that verdict is the answer, + /// the one a cache-less call renders: one read, no write. (Had the hint matched `observed`, the + /// write path would run instead: a 412 and a resolve read, which the lane tests cover.) + const CatalogEntry observed_elsewhere = entryInState("a", NsState::Creating, 2); + EXPECT_EQ(CasRefCatalog::cancelStalledCreating(pool_op, layout, observed_elsewhere, [](const CreatorFence &) { return true; }), + CasRefCatalog::StalledCreatingCancelOutcome::EntryChanged); + EXPECT_EQ(f.backend->writeCount(layout.refCatalogKey()), writes_before) << "no write"; + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()) - reads_before, 1u) << "one lane read"; +} + +/// createNamespaceStep1 via createNamespace: the cache holds row "a" (Live), the external erased it +/// via a completed removal. +TEST(CASRefCatalog, AStaleHintCreateNamespaceStep1AdmitsANewIncarnation) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + ASSERT_EQ(CasRefCatalog::createNamespace(pool_op, layout, 1, RootNamespace{"a"}, creator), + CasRefCatalog::NamespaceCreationOutcome::Live); /// the cache now holds "a" Live + + CasOperation other = f.external.admit(); + const CatalogEntry observed = CasRefCatalog::read(other, layout).catalog.entries.at(0); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, observed, 9), CasRefCatalog::BeginRemovingOutcome::Transitioned); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(observed.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const CatalogEntry removing{.ns = RootNamespace{"a"}, .state = NsState::Removing, + .incarnation = observed.incarnation, .removal_started_round = 9}; + ASSERT_EQ(CasRefCatalog::deleteCompletedRemoving(other, layout, removing, ready_parent, noAuthorityRefresh).outcome, + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + EXPECT_EQ(CasRefCatalog::createNamespace(pool_op, layout, 1, RootNamespace{"a"}, creator), + CasRefCatalog::NamespaceCreationOutcome::Live); + /// Two reads, not one: `createNamespace`'s own pre-check (`read`) never goes through the lane, so + /// it always issues its own request regardless of the cache; the lane's ONE read is the one + /// `createNamespaceStep1` pays for finding its cached hint stale. + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()) - reads_before, 2u) << "the pre-check read plus one lane read"; +} + +/// createNamespaceStep1: the store holds "a" Creating under another creator -- Superseded on the read. +TEST(CASRefCatalog, ATrueRefusalCreateNamespaceStep1SeesASuperseder) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + ASSERT_EQ(CasRefCatalog::createNamespace(pool_op, layout, 1, RootNamespace{"a"}, creator), + CasRefCatalog::NamespaceCreationOutcome::Live); + + CasOperation other = f.external.admit(); + const CatalogEntry observed = CasRefCatalog::read(other, layout).catalog.entries.at(0); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, observed, 9), CasRefCatalog::BeginRemovingOutcome::Transitioned); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(observed.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const CatalogEntry removing{.ns = RootNamespace{"a"}, .state = NsState::Removing, + .incarnation = observed.incarnation, .removal_started_round = 9}; + ASSERT_EQ(CasRefCatalog::deleteCompletedRemoving(other, layout, removing, ready_parent, noAuthorityRefresh).outcome, + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + /// The store now carries no row for "a"; the pool's cache still holds its own stale "a" Live from + /// the first `createNamespace` above. `createNamespace`'s own pre-check read is unconditional and + /// always fresh, so if the rival lands its row BEFORE this call, the pre-check alone would already + /// return `Superseded` and step 1 -- the lane -- would never run. The rival instead lands INSIDE + /// the window `setCreateNamespaceStep1PreReadHookForTest` names: after the pre-check already saw + /// nothing, but before step 1's own (lane) read. `createNamespaceStep1`'s decide then runs first + /// against the CACHED hint (still "a" Live) -- which already has a row for "a", so it throws + /// immediately without ever reaching the store -- and that verdict-on-a-hint is dropped: the lane + /// re-reads for real and decides again, this time seeing the rival's fresh row, and throws again. + /// That second throw is the one that reaches `createNamespace`. + const CreatorFence other_creator{.server_root_id = "other", .writer_epoch = 1, .fence_generation = 1}; + CatalogEntry other_creating = entryInState("a", NsState::Creating, 5); + other_creating.creator = other_creator; + CasRefCatalog::setCreateNamespaceStep1PreReadHookForTest([&] + { + CasRefCatalog::casAdmitEntry(other, layout, 1, other_creating); + }); + + /// The hook's own admission is a real write on the SAME shared backend the pool uses, so a raw + /// backend read/write count taken across this call would count the rival's own IO alongside the + /// lane's -- `CASHotKeyCacheVerdictsReread` does not: it increments only when a CACHED hint's + /// verdict is dropped and re-decided, which the rival's cache-less admission (its own private, + /// budget-0 `CasHotKeys`) never does, so it isolates the lane's own contribution cleanly. + const uint64_t rereads_before = eventCount(ProfileEvents::CASHotKeyCacheVerdictsReread); + EXPECT_EQ(CasRefCatalog::createNamespace(pool_op, layout, 1, RootNamespace{"a"}, creator), + CasRefCatalog::NamespaceCreationOutcome::Superseded); + EXPECT_EQ(eventCount(ProfileEvents::CASHotKeyCacheVerdictsReread) - rereads_before, 1u) + << "one lane read: the cached hint's verdict was dropped and redecided on a fresh read"; + + /// No write of ours landed: the row for "a" is exactly what the rival left it as. + const CasRefCatalog::Snapshot snap_after = CasRefCatalog::read(other, layout); + const CatalogEntry * after = &snap_after.catalog.entries.at(0); + EXPECT_EQ(after->state, NsState::Creating); + ASSERT_TRUE(after->creator.has_value()); + EXPECT_EQ(*after->creator, other_creator); + EXPECT_EQ(after->incarnation, other_creating.incarnation); +} + +/// completeCreation: the cache holds "a" Creating (incarnation 1) matching the store's row for "a", +/// but the store also gained row "b" -- the cache is stale only elsewhere, so the write goes through. +TEST(CASRefCatalog, AStaleHintCompleteCreationLandsWhenItsOwnRowIsUnchanged) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a_creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a_creating); /// cache: "a" Creating + + CasOperation other = f.external.admit(); + const CreatorFence other_creator{.server_root_id = "other", .writer_epoch = 1, .fence_generation = 1}; + ASSERT_EQ(CasRefCatalog::createNamespace(other, layout, 1, RootNamespace{"b"}, other_creator), + CasRefCatalog::NamespaceCreationOutcome::Live); + + EXPECT_EQ(CasRefCatalog::completeCreation(pool_op, layout, a_creating), CasRefCatalog::NamespaceCreationOutcome::Live); + EXPECT_EQ(CasRefCatalog::read(other, layout).catalog.entries.size(), 2u); +} + +/// completeCreation: "a" was completed to Live by the external -- Superseded on the read. +TEST(CASRefCatalog, ATrueRefusalCompleteCreationSeesItsOwnRowCompletedElsewhere) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a_creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a_creating); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::completeCreation(other, layout, a_creating), CasRefCatalog::NamespaceCreationOutcome::Live); + + EXPECT_EQ(CasRefCatalog::completeCreation(pool_op, layout, a_creating), CasRefCatalog::NamespaceCreationOutcome::Superseded); +} + +/// reconcileStaleCreator: the cache holds "a" Creating under creator X; the store's row for "a" is +/// unchanged but the catalog gained "b" -- Reconciled. +TEST(CASRefCatalog, AStaleHintReconcileStaleCreatorLandsWhenItsOwnRowIsUnchanged) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a_creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a_creating); /// cache: "a" Creating under creator X + + CasOperation other = f.external.admit(); + const CreatorFence other_creator{.server_root_id = "other", .writer_epoch = 1, .fence_generation = 1}; + ASSERT_EQ(CasRefCatalog::createNamespace(other, layout, 1, RootNamespace{"b"}, other_creator), + CasRefCatalog::NamespaceCreationOutcome::Live); + + const CreatorFence new_creator{.server_root_id = "srv2", .writer_epoch = 1, .fence_generation = 1}; + EXPECT_EQ(CasRefCatalog::reconcileStaleCreator(pool_op, layout, a_creating, new_creator, + [](const CreatorFence &) { return true; }), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + EXPECT_EQ(CasRefCatalog::read(other, layout).catalog.entries.size(), 2u); +} + +/// reconcileStaleCreator: "a" was completed to Live by the external -- EntryChanged on the read. +TEST(CASRefCatalog, ATrueRefusalReconcileStaleCreatorSeesItsOwnRowCompletedElsewhere) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a_creating = entryInState("a", NsState::Creating, 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a_creating); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::completeCreation(other, layout, a_creating), CasRefCatalog::NamespaceCreationOutcome::Live); + + const CreatorFence new_creator{.server_root_id = "srv2", .writer_epoch = 1, .fence_generation = 1}; + EXPECT_EQ(CasRefCatalog::reconcileStaleCreator(pool_op, layout, a_creating, new_creator, + [](const CreatorFence &) { return true; }), + CasRefCatalog::ReconcileCreatorOutcome::EntryChanged); +} + +/// Admission refusal (LIMIT_EXCEEDED): the pool's cache holds a catalog at the namespace limit, but +/// the external erased one row, so the fresh catalog admits. +TEST(CASRefCatalog, AStaleHintAdmissionRefusalAdmitsWhenTheFreshCatalogHasRoom) +{ + const Layout layout("p"); + constexpr uint64_t gc_shards = 1; + const uint64_t cap = foldSealCaps().object_cap; + const uint64_t fixed = foldSealFixedBytes(); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + const uint64_t nonentry = widestBlobTargetRunReservationBytes(layout, gc_shards) + + widestCondemnedSummaryReservationBytes(gc_shards); + const uint64_t max_entries = (cap - fixed - nonentry) / reservation; + + PoolAndExternal f(std::make_shared()); + CasOperation pool_op = f.pool.admit(); + RefCatalog full; + full.entries.reserve(max_entries); + for (uint64_t i = 0; i < max_entries; ++i) + full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); + seedObject(pool_op, layout.refCatalogKey(), encodeRefCatalog(full)); + /// Prime the cache with the full catalog: any further admission through the pool refuses on it. + CasRefCatalog::casUpdate(pool_op, layout, [](const RefCatalog & cur) { return cur; }); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, full.entries[0], 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + const CasRefCatalog::Snapshot after_remove = CasRefCatalog::read(other, layout); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(full.entries[0].incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const CatalogEntry removing{.ns = full.entries[0].ns, .state = NsState::Removing, + .incarnation = full.entries[0].incarnation, .removal_started_round = 5}; + ASSERT_EQ(CasRefCatalog::deleteCompletedRemovingAtSnapshot(other, layout, after_remove, removing, ready_parent, noAuthorityRefresh).outcome, + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + EXPECT_EQ(CasRefCatalog::createNamespace(pool_op, layout, gc_shards, RootNamespace{"fresh"}, creator), + CasRefCatalog::NamespaceCreationOutcome::Live); +} + +/// Admission refusal (LIMIT_EXCEEDED): the fresh catalog is also full -- thrown, no write. +TEST(CASRefCatalog, ATrueRefusalAdmissionRefusalThrowsWhenTheFreshCatalogIsAlsoFull) +{ + const Layout layout("p"); + constexpr uint64_t gc_shards = 1; + const uint64_t cap = foldSealCaps().object_cap; + const uint64_t fixed = foldSealFixedBytes(); + const uint64_t reservation = worstCaseEntryFoldReservationBytes(); + const uint64_t nonentry = widestBlobTargetRunReservationBytes(layout, gc_shards) + + widestCondemnedSummaryReservationBytes(gc_shards); + const uint64_t max_entries = (cap - fixed - nonentry) / reservation; + + PoolAndExternal f(std::make_shared()); + CasOperation pool_op = f.pool.admit(); + RefCatalog full; + full.entries.reserve(max_entries); + for (uint64_t i = 0; i < max_entries; ++i) + full.entries.push_back(liveEntry(fmt::format("ns{:012}", i), i + 1)); + seedObject(pool_op, layout.refCatalogKey(), encodeRefCatalog(full)); + CasRefCatalog::casUpdate(pool_op, layout, [](const RefCatalog & cur) { return cur; }); /// prime the cache + + const uint64_t writes_before = f.backend->writeCount(layout.refCatalogKey()); + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LIMIT_EXCEEDED, [&] + { + (void)CasRefCatalog::createNamespace(pool_op, layout, gc_shards, RootNamespace{"overflow"}, creator); + }); + EXPECT_EQ(f.backend->writeCount(layout.refCatalogKey()), writes_before) << "no write"; +} + +/// casAdmitEntry: the cache holds "a"; the external erased it -- admits it again on the fresh read. +/// The stale hint's own decide sees "a" still Live and duplicates the namespace before the lane can +/// reread, so `encodeRefCatalog` throws `LOGICAL_ERROR` on the hint attempt -- split like the blocks +/// above, since constructing that exception aborts under debug/sanitizer builds before this test's own +/// rescue (the reread on a fresh, "a"-absent state) ever runs. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalog, AStaleHintCasAdmitEntryAdmitsWhenTheStoreHasRoom) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a = liveEntry("a", 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); /// cache: "a" + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, a, 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(a.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const CatalogEntry removing{.ns = a.ns, .state = NsState::Removing, + .incarnation = a.incarnation, .removal_started_round = 5}; + ASSERT_EQ(CasRefCatalog::deleteCompletedRemoving(other, layout, removing, ready_parent, noAuthorityRefresh).outcome, + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + EXPECT_NO_THROW(CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("a", 2))); + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()) - reads_before, 1u) << "one lane read"; +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogDeathTest, AStaleHintCasAdmitEntryAdmitsWhenTheStoreHasRoomAborts) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a = liveEntry("a", 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); /// cache: "a" + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, a, 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(a.incarnation, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + const CatalogEntry removing{.ns = a.ns, .state = NsState::Removing, + .incarnation = a.incarnation, .removal_started_round = 5}; + ASSERT_EQ(CasRefCatalog::deleteCompletedRemoving(other, layout, removing, ready_parent, noAuthorityRefresh).outcome, + CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + + /// The lane's re-render on a fresh read cannot rescue this, because constructing the + /// `LOGICAL_ERROR` for the hint's own duplicate aborts before the reread ever happens. + EXPECT_DEATH({ CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("a", 2)); }, "not canonically ordered"); +} +#endif + +/// casAdmitEntry: "a" is still present in the fresh read (Removing, not the Live hint the cache +/// holds) -- re-admitting it duplicates the namespace, and `encodeRefCatalog`'s own canonical-order +/// check aborts with `LOGICAL_ERROR` under debug/sanitizer builds -- split like the blocks above. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefCatalog, ATrueRefusalCasAdmitEntrySeesItsRowStillPresent) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a = liveEntry("a", 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, a, 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, + [&] { CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); }); +} +#endif + +#if defined(DEBUG_OR_SANITIZER_BUILD) +TEST(CASRefCatalogDeathTest, ATrueRefusalCasAdmitEntrySeesItsRowStillPresentAborts) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry a = liveEntry("a", 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); + + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, a, 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + + EXPECT_DEATH({ CasRefCatalog::casAdmitEntry(pool_op, layout, 1, a); }, "not canonically ordered"); +} +#endif + +TEST(CASRefCatalog, AFenceLostDuringStepOneIsFencedOutNotABareMarker) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation seed = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(seed, layout); + const CreatorFence creator{.server_root_id = "srv", .writer_epoch = 1, .fence_generation = 1}; + + /// Tripped before its turn: the lane's wait refuses it, nothing is written. + { + bool alive = true; + CasOperation op = f.pool.admit([&] { return alive; }); + f.hot_keys.enter_after_lane_hook_for_test = [&] { alive = false; }; + const uint64_t writes_before = f.backend->writeCount(layout.refCatalogKey()); + EXPECT_EQ(CasRefCatalog::createNamespace(op, layout, 1, RootNamespace{"a"}, creator), + CasRefCatalog::NamespaceCreationOutcome::FencedOut); + f.hot_keys.enter_after_lane_hook_for_test = {}; + EXPECT_EQ(f.backend->writeCount(layout.refCatalogKey()), writes_before); + } + /// Tripped between the landed step-1 write and the post-commit check: FencedOut, and the Creating + /// row is durable for a later reconciler. + { + bool alive = true; + CasOperation op = f.pool.admit([&] { return alive; }); + f.backend->onWriteCommitted(layout.refCatalogKey(), [&] { alive = false; }); + EXPECT_EQ(CasRefCatalog::createNamespace(op, layout, 1, RootNamespace{"b"}, creator), + CasRefCatalog::NamespaceCreationOutcome::FencedOut); + f.backend->onWriteCommitted(layout.refCatalogKey(), {}); + const auto snap = CasRefCatalog::read(seed, layout); + ASSERT_EQ(snap.catalog.entries.size(), 1u); + EXPECT_EQ(snap.catalog.entries[0].state, NsState::Creating); + } + /// casAdmitEntry under a tripped fence throws the transient-unavailable class, never a bare marker. + { + CasOperation op = f.pool.admit([] { return false; }); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, + [&] { CasRefCatalog::casAdmitEntry(op, layout, 1, liveEntry("c", 3)); }); + } +} + +#if USE_AWS_S3 +TEST(CASRefCatalog, TheStoresAnswerToAHintsWriteIsDeliveredAndTheRetryLearnsTheRest) +{ + PoolAndExternal f(std::make_shared()); + f.backend->setRefreshCredentialsResult(false); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry observed = liveEntry("a", 1); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, observed); /// the cache holds "a" Live + + /// Another server moved the row on; the hint still matches what this caller observed, so a write + /// goes out, and the store refuses it definitively. + CasOperation other = f.external.admit(); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, observed, 5), CasRefCatalog::BeginRemovingOutcome::Transitioned); + f.backend->failNextWriteWith(layout.refCatalogKey(), s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + const uint64_t writes_before = f.backend->writeCount(layout.refCatalogKey()); + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::S3_ERROR, + [&] { (void)CasRefCatalog::beginRemoving(pool_op, layout, observed, 5); }); + EXPECT_EQ(f.backend->writeCount(layout.refCatalogKey()) - writes_before, 1u); + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()) - reads_before, 0u); + /// The caller's ordinary retry after a fresh observation learns the row moved. + EXPECT_EQ(CasRefCatalog::beginRemoving(pool_op, layout, observed, 5), CasRefCatalog::BeginRemovingOutcome::AlreadyRemoving); + + /// The same stale hint, and the fence trips between the refused write and its resolve read. + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("b", 2)); + const CatalogEntry b = liveEntry("b", 2); + ASSERT_EQ(CasRefCatalog::beginRemoving(other, layout, b, 6), CasRefCatalog::BeginRemovingOutcome::Transitioned); + bool alive = true; + CasOperation fenced = f.pool.admit([&] { return alive; }); + f.backend->onBeforeWrite(layout.refCatalogKey(), [&] { alive = false; }); + EXPECT_EQ(CasRefCatalog::beginRemoving(fenced, layout, b, 6), CasRefCatalog::BeginRemovingOutcome::FencedOut); + f.backend->onBeforeWrite(layout.refCatalogKey(), {}); + EXPECT_EQ(CasRefCatalog::read(other, layout).catalog.entries.at(1).state, NsState::Removing) << "nothing of ours landed"; + CasOperation readmitted = f.pool.admit(); + EXPECT_EQ(CasRefCatalog::beginRemoving(readmitted, layout, b, 6), CasRefCatalog::BeginRemovingOutcome::AlreadyRemoving); +} +#endif + +TEST(CASRefCatalog, TheCatalogLoopPacesAConflictByWhetherItSettledAFault) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("a", 1)); + CasOperation other = f.external.admit(); + const String key = layout.refCatalogKey(); + + /// K clean races: the external moves the catalog before each of the pool's first K writes. The + /// two scenarios below admit under DIFFERENT namespace prefixes -- the first scenario's "x" rows + /// are still on the catalog when the second starts, and re-admitting the same namespace would + /// duplicate it rather than race it. + constexpr int K = 3; + /// The settled-fault scenario draws 8, not 3: an upper-bound-only check on `backoff`'s draws + /// cannot tell the growing schedule from the flat one (both stay under 200 ms plenty often at + /// small K), so that scenario also needs a floor. With `backoff(attempt) = uniform(0, min(5000, + /// 200 << (attempt-1)))`, the flat schedule can NEVER draw over 200 ms, so a single draw over + /// 200 ms is proof by itself that the schedule grew; drawing 8 times keeps the growing schedule's + /// own chance of missing that floor purely by bad luck around 2^-28. + constexpr int K2 = 8; + int moved = 0; + int limit = K; + bool inside = false; + bool ambiguous = false; + String prefix = "x"; + uint64_t incarnation_base = 10; + f.backend->onBeforeWrite(key, [&] + { + if (inside || moved >= limit) + return; + inside = true; + CasRefCatalog::casAdmitEntry(other, layout, 1, liveEntry(prefix + std::to_string(moved), incarnation_base + moved)); + if (ambiguous) + f.backend->injectAmbiguousWrite(key); + ++moved; + inside = false; + }); + const auto pauses_before = eventCount(ProfileEvents::CASRequestConflictPause); + CasRefCatalog::casUpdate(pool_op, layout, [](const RefCatalog & cur) + { + RefCatalog next = cur; + for (auto & e : next.entries) + if (e.ns.string() == "a") { e.state = NsState::Removing; e.removal_started_round = 1; } + return next; + }); + ASSERT_EQ(f.clock.sleeps.size(), static_cast(K)); + for (uint64_t s : f.clock.sleeps) + EXPECT_LE(s, 200u); + EXPECT_EQ(eventCount(ProfileEvents::CASRequestConflictPause) - pauses_before, 0u) << "the lane's caller pauses itself; the engine's counter is for its own loops"; + + /// K2 conflicts that each settled a fault: the growing schedule, on the loop's own count. An + /// upper-bound check alone is satisfied by the flat schedule too (its draws are a subset of the + /// growing schedule's early-attempt range), so this also asserts a floor: at least one draw over + /// 200 ms, which the flat schedule can never produce. + f.clock.sleeps.clear(); + moved = 0; + limit = K2; + ambiguous = true; + prefix = "y"; + incarnation_base = 20; + CasRefCatalog::casUpdate(pool_op, layout, [](const RefCatalog & cur) { return cur; }); + ASSERT_EQ(f.clock.sleeps.size(), static_cast(K2)); + for (size_t i = 0; i < f.clock.sleeps.size(); ++i) + EXPECT_LE(f.clock.sleeps[i], std::min(5000, 200ull << i)); + EXPECT_TRUE(std::any_of(f.clock.sleeps.begin(), f.clock.sleeps.end(), [](uint64_t s) { return s > 200u; })) + << "the flat schedule can never draw over 200 ms; a growing schedule almost certainly does " + "somewhere in 8 draws, so this is what tells the two schedules apart"; +} + +TEST(CASRefCatalog, TheGCEraseRacesTheLaneAsItRacesEverything) +{ + PoolAndExternal f(std::make_shared()); + const Layout layout("p"); + CasOperation pool_op = f.pool.admit(); + CasRefCatalog::initializeEmptyForNewPool(pool_op, layout); + const CatalogEntry removing{.ns = RootNamespace{"a"}, .state = NsState::Removing, + .incarnation = UInt128{7}, .removal_started_round = 13}; + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("keep", 1)); + /// Seed the Removing row through the external, as raw bytes: `casUpdate` refuses to add rows, and + /// the point is that the pool's cache is stale about this one. + CasOperation other = f.external.admit(); + { + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(other, layout); + RefCatalog with_removing = snap.catalog; + with_removing.entries.insert(with_removing.entries.begin(), removing); /// "a" sorts before "keep" + (void)orThrow(other.replace(layout.refCatalogKey(), encodeRefCatalog(with_removing), *snap.etag, Retry::standard()), "seed"); + } + CasFoldSeal ready_parent; + ready_parent.ref_lives.emplace(UInt128{7}, RefLifeFoldState{ + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 2}}, + .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 2}}}); + + /// The erase runs on an open plane of the same pool while a lane holder is parked in its write: + /// it does not wait. + CasRequests open_plane(f.backend, Fence::open(), f.clock.nowFn(), f.clock.sleepFn(), &f.hot_keys); + CasOperation erase_op = open_plane.admit(); + std::latch parked(1); + std::latch release(1); + int seen = 0; + f.backend->onBeforeWrite(layout.refCatalogKey(), [&] + { + if (seen++ != 0) + return; + parked.count_down(); + release.wait(); + }); + std::thread holder([&] + { + CasRefCatalog::casUpdate(pool_op, layout, [](const RefCatalog & cur) { return cur; }); + }); + parked.wait(); + const auto result = CasRefCatalog::deleteCompletedRemovingAtSnapshot( + erase_op, layout, CasRefCatalog::read(other, layout), removing, ready_parent, noAuthorityRefresh); + EXPECT_EQ(result.outcome, CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted); + EXPECT_EQ(f.hot_keys.queueDepthForTest(layout.refCatalogKey()), 1u) << "the holder is still parked"; + release.count_down(); + holder.join(); + f.backend->onBeforeWrite(layout.refCatalogKey(), {}); + + /// The pool's next submission pays exactly one resolve read and one retry write for the erase. + const uint64_t reads_before = f.backend->getCount(layout.refCatalogKey()); + const uint64_t writes_before = f.backend->writeCount(layout.refCatalogKey()); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("c", 3)); + EXPECT_LE(f.backend->getCount(layout.refCatalogKey()) - reads_before, 1u); + EXPECT_LE(f.backend->writeCount(layout.refCatalogKey()) - writes_before, 2u); + const uint64_t reads_after = f.backend->getCount(layout.refCatalogKey()); + CasRefCatalog::casAdmitEntry(pool_op, layout, 1, liveEntry("d", 4)); + EXPECT_EQ(f.backend->getCount(layout.refCatalogKey()), reads_after) << "the one after starts from the cache"; +} diff --git a/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp index 1ea4b8096e4d..2c094fa122af 100644 --- a/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp +++ b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp @@ -9,6 +9,7 @@ #include #include +#include #include namespace DB::ErrorCodes @@ -17,22 +18,22 @@ extern const int CORRUPTED_DATA; extern const int NETWORK_ERROR; } -/// Stage B Task 4-C: production birth wiring. `CasRefLedger::resolveNamespaceLife`, called from +/// Stage B: production birth wiring. `CasRefLedger::resolveNamespaceLife`, called from /// `ensureRefTableRecovered`, resolves a namespace's real catalog life ONCE per table-open -- /// create-if-absent, adopt an existing `Live`/`Removing` entry, or reconcile a stale `Creating` one via /// `CasRefCatalog::reconcileStaleCreator` + `isCreatorFenceTerminal` -- so every ref-layer object a /// mounted writer produces is keyed at a real, catalog-proven incarnation (spec INV-3), never the /// Stage-A sentinel. /// -/// OBLIGATION 3 (carried from Task 3's review, closed here): Task 3 could only enforce "`Creating` -/// forbids publication" (`CasRefCatalog::checkPublicationAdmittedOrThrow`) AT THE CATALOG LEVEL, because -/// nothing on the production ref-write path consulted the catalog at all. The refusal this suite pins -/// below rests on CONSTRUCTION, not a check: there is no `if (state == Creating) throw` anywhere in -/// `appendRefOps`'s path. `ensureRefTableRecovered` simply cannot make a table's runtime usable -/// (`rt.recovered` never becomes `true`, `rt.life` never gets set) while the catalog entry is `Creating` -/// under a fence that is not provably dead -- so no append can reach `commitRefChunk` for such a -/// namespace, by construction, stronger than any per-write check could prove. Stated here so nobody -/// later greps for a check and concludes the gap Task 3's review flagged is still open. +/// OBLIGATION 3 (closed here): `CasRefCatalog::checkPublicationAdmittedOrThrow` can only enforce +/// "`Creating` forbids publication" AT THE CATALOG LEVEL, because nothing on the production ref-write +/// path consulted the catalog at all. The refusal this suite pins below rests on CONSTRUCTION, not a +/// check: there is no `if (state == Creating) throw` anywhere in `appendRefOps`'s path. +/// `ensureRefTableRecovered` simply cannot make a table's runtime usable (`rt.recovered` never becomes +/// `true`, `rt.life` never gets set) while the catalog entry is `Creating` under a fence that is not +/// provably dead -- so no append can reach `commitRefChunk` for such a namespace, by construction, +/// stronger than any per-write check could prove. Stated here so nobody later greps for a check and +/// concludes this gap is still open. /// /// The suite name is prefixed `Cas` so it is covered by the `Cas*` unit-test gate filter. @@ -46,44 +47,91 @@ namespace /// models a definite write failure, distinct from the acknowledgement-loss shape below: a retry must /// still be allowed to prove a new pool, and the failed first attempt must not have published /// `_pool_meta` without the catalog it makes mandatory. -class CatalogBootstrapPutFailsOnceBackend final : public CountingBackend +/// Per-key counts of the WRITE primitive. `CountingBackend` counts reads, heads and lists per key but +/// only totals for writes, and its legacy per-verb counters never see a caller that speaks the +/// primitives -- which every writer below does. +class WriteCountingBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; + uint64_t writes(const String & key) const + { + std::lock_guard lock(write_count_mutex); + const auto it = write_counts.find(key); + return it == write_counts.end() ? 0 : it->second; + } + + void resetWriteCounts() + { + std::lock_guard lock(write_count_mutex); + write_counts.clear(); + } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + { + std::lock_guard lock(write_count_mutex); + ++write_counts[key]; + } + return CountingBackend::write(key, bytes, expected_value, access); + } + +private: + mutable std::mutex write_count_mutex; + std::map write_counts; +}; + +/// Faults the mandatory catalog's very first bootstrap write before it reaches durable storage. This +/// models a definite write failure, distinct from the acknowledgement-loss shape below: a retry must +/// still be allowed to prove a new pool, and the failed first attempt must not have published +/// `_pool_meta` without the catalog it makes mandatory. +/// +/// A plain `std::runtime_error`, not a `Poco::Exception`, and deliberately so: a `Poco`/transport +/// exception here would exercise the write loop's OWN ambiguity resolution rather than this suite's +/// subject, which is what `FailedCatalogBootstrapDoesNotPublishPoolMetaAndRetryConverges` actually +/// needs -- a fault that propagates out of the FIRST `Pool::open` call so a SEPARATE retry can be the +/// one that converges. The engine's write loop treats any `Poco`/transport exception as an ambiguity it +/// settles itself with one resolve read, and a one-shot fault of that class is retried and silently +/// succeeds within the SAME `Pool::open` call -- it never reaches the caller at all. A non-`Poco` +/// `std::exception` is the engine's own signal for "this could +/// not have landed" and propagates unresolved, which is what "before it reaches durable storage" means. +class CatalogBootstrapWriteFailsOnceBackend final : public WriteCountingBackend +{ +public: + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (fail_once && key == Layout{"p"}.refCatalogKey()) { fail_once = false; - throw Poco::TimeoutException("CatalogBootstrapPutFailsOnceBackend: catalog PUT did not land"); + throw std::runtime_error("CatalogBootstrapWriteFailsOnceBackend: catalog write did not land"); } - return CountingBackend::putIfAbsent(key, bytes, meta); + return WriteCountingBackend::write(key, bytes, expected_value, access); } private: bool fail_once = true; }; -class CatalogCancellationRaceBackend final : public CountingBackend +class CatalogCancellationRaceBackend final : public WriteCountingBackend { public: - using CountingBackend::casPut; - - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (race_armed && key == Layout{"p"}.refCatalogKey()) { race_armed = false; - on_catalog_cas(); + on_catalog_write(); } - return CountingBackend::casPut(key, bytes, expected, meta); + return WriteCountingBackend::write(key, bytes, expected_value, access); } bool race_armed = false; - std::function on_catalog_cas; + std::function on_catalog_write; }; PoolPtr openPoolForBirthTest(const BackendPtr & backend, const String & server_root_id = "test") @@ -124,13 +172,15 @@ RefTxnId publishBirth(const PoolPtr & store, const RootNamespace & ns, const Str TEST(CASRefCatalogBirthWiring, FirstOpenMintsALiveCatalogEntryAndKeysTheBirthAtIt) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const RootNamespace ns{"srv1/birth_wiring"}; const RefTxnId id = publishBirth(store, ns, "a"); EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, store->layout()); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, store->layout()); const CatalogEntry * entry = findEntry(snap.catalog, ns); ASSERT_NE(entry, nullptr) << "the first open must mint a catalog entry"; EXPECT_EQ(entry->state, NsState::Live); @@ -138,103 +188,117 @@ TEST(CASRefCatalogBirthWiring, FirstOpenMintsALiveCatalogEntryAndKeysTheBirthAtI EXPECT_EQ(entry->creator, std::nullopt) << "creator is forbidden outside Creating (strict grammar)"; const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); - EXPECT_TRUE(backend->head(store->layout().refLogKey(life, id)).exists) + EXPECT_TRUE(op.head(store->layout().refLogKey(life, id), Retry::standard()).has_value()) << "the birth transaction must be keyed at the REAL minted incarnation, not the Stage-A sentinel"; - EXPECT_FALSE(backend->head(store->layout().refLogKey(fixture::fixtureLife(ns), id)).exists) + EXPECT_FALSE(op.head(store->layout().refLogKey(fixture::fixtureLife(ns), id), Retry::standard()).has_value()) << "and must NOT be keyed at the sentinel any more"; } TEST(CASRefCatalogBirthWiring, CatalogLossAfterMountCannotRecreateAOneRowAuthority) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); publishBirth(store, RootNamespace{"srv1/existing"}, "old"); - const auto catalog = backend->get(layout.refCatalogKey()); + const auto catalog = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog); - ASSERT_EQ(backend->deleteExact(layout.refCatalogKey(), catalog->token).kind, - DeleteOutcome::Kind::Deleted); + ASSERT_EQ(op.remove(layout.refCatalogKey(), catalog->etag, Retry::standard()), Removal::Removed); backend->resetCounts(); + backend->resetWriteCounts(); EXPECT_THROW(publishBirth(store, RootNamespace{"srv1/new"}, "new"), DB::Exception); - EXPECT_FALSE(backend->head(layout.refCatalogKey()).exists) + EXPECT_FALSE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()) << "runtime loss must not be repaired with a one-row replacement authority"; - EXPECT_EQ(backend->casPutTotal(), 0u); - EXPECT_EQ(backend->putTotal(), 0u) - << "the failed birth must not publish a checkpoint or ref-log body"; - EXPECT_EQ(backend->putOverwriteTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u) + << "the failed birth must not publish a catalog, a checkpoint or a ref-log body"; } TEST(CASRefCatalogBirthWiring, FailedCatalogBootstrapDoesNotPublishPoolMetaAndRetryConverges) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; EXPECT_ANY_THROW(Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); - EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists) + EXPECT_FALSE(op.head(layout.poolMetaKey(), Retry::standard()).has_value()) << "a failed mandatory catalog bootstrap must leave no authoritative pool meta behind"; - EXPECT_FALSE(backend->head(layout.refCatalogKey()).exists); + EXPECT_FALSE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()); PoolPtr retry; ASSERT_NO_THROW(retry = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); - EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); - EXPECT_TRUE(backend->head(layout.refCatalogKey()).exists); + EXPECT_TRUE(op.head(layout.poolMetaKey(), Retry::standard()).has_value()); + EXPECT_TRUE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()); } -TEST(CASRefCatalogBirthWiring, LostCatalogBootstrapAcknowledgementLeavesOnlyRetryableCatalogResidue) +TEST(CASRefCatalogBirthWiring, LostCatalogBootstrapAcknowledgementResolvesToCommittedWithoutARetryOrADuplicate) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; backend->key_substr = layout.refCatalogKey(); - EXPECT_ANY_THROW(Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); - EXPECT_FALSE(backend->head(layout.poolMetaKey()).exists); - EXPECT_TRUE(backend->head(layout.refCatalogKey()).exists) - << "the injected write must land before its acknowledgement is lost"; - - PoolPtr retry; - ASSERT_NO_THROW(retry = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); - EXPECT_TRUE(backend->head(layout.poolMetaKey()).exists); + /// The catalog create's own write landed; only its response was lost. `LandedButAckLostOnceBackend` + /// is documented to model exactly that -- a caller that resolves the ambiguity meets its OWN + /// earlier write as the occupant -- so the engine's one resolve read proves this attempt committed. + /// The whole bootstrap therefore converges in this SINGLE `Pool::open` call: no throw, no second + /// catalog write, and no second `Pool::open` needed. + PoolPtr store; + ASSERT_NO_THROW(store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"})); + EXPECT_TRUE(op.head(layout.poolMetaKey(), Retry::standard()).has_value()); + EXPECT_TRUE(op.head(layout.refCatalogKey(), Retry::standard()).has_value()); + EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 1u) + << "a landed write whose ack is lost must be proven by a read, never repeated"; } TEST(CASRefCatalogBirthWiring, BootstrapConflictExactReadsTheCanonicalEmptyCatalog) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const String canonical_empty = encodeRefCatalog(RefCatalog{}); - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), canonical_empty).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCatalogKey(), canonical_empty, Retry::standard()))); backend->resetCounts(); + backend->resetWriteCounts(); - const CasRefCatalog::Snapshot snap = CasRefCatalog::initializeEmptyForNewPool(*backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::initializeEmptyForNewPool(op, layout); EXPECT_TRUE(snap.catalog.entries.empty()); - EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), 1u); EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u) << "a concurrent bootstrap winner must be exact-read before acceptance"; } TEST(CASRefCatalogBirthWiring, BootstrapConflictRefusesANonemptyCatalog) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RefCatalog nonempty{.entries = {CatalogEntry{ .ns = RootNamespace{"test/nonempty"}, .state = NsState::Live, .incarnation = UInt128{1}, .creator = std::nullopt}}}; - ASSERT_EQ(backend->putIfAbsent(layout.refCatalogKey(), encodeRefCatalog(nonempty)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(layout.refCatalogKey(), encodeRefCatalog(nonempty), Retry::standard()))); backend->resetCounts(); + backend->resetWriteCounts(); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { CasRefCatalog::initializeEmptyForNewPool(*backend, layout); }); + [&] { CasRefCatalog::initializeEmptyForNewPool(op, layout); }); EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); } TEST(CASRefCatalogBirthWiring, ExistingPoolMetaWithMissingCatalogStillFailsClosed) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); PoolPtr first = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); - const auto catalog = backend->get(first->layout().refCatalogKey()); + const auto catalog = op.read(first->layout().refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog); - ASSERT_EQ(backend->deleteExact(first->layout().refCatalogKey(), catalog->token).kind, - DeleteOutcome::Kind::Deleted); + ASSERT_EQ(op.remove(first->layout().refCatalogKey(), catalog->etag, Retry::standard()), Removal::Removed); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); }); @@ -242,24 +306,27 @@ TEST(CASRefCatalogBirthWiring, ExistingPoolMetaWithMissingCatalogStillFailsClose TEST(CASRefCatalogBirthWiring, RestartFixturePreservesItsExistingNonemptyCatalog) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; seedPoolMetaForRestart(*backend); - const auto empty = backend->get(layout.refCatalogKey()); + const auto empty = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(empty); const RefCatalog nonempty{.entries = {CatalogEntry{ .ns = RootNamespace{"test/preserved"}, .state = NsState::Live, .incarnation = UInt128{1}, .creator = std::nullopt}}}; const String bytes = encodeRefCatalog(nonempty); - ASSERT_EQ(backend->putOverwrite(layout.refCatalogKey(), bytes, empty->token).outcome, PutOutcome::Done); - const auto before = backend->get(layout.refCatalogKey()); + ASSERT_TRUE(std::holds_alternative( + op.replace(layout.refCatalogKey(), bytes, empty->etag, Retry::standard()))); + const auto before = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(before); seedPoolMetaForRestart(*backend); - const auto after = backend->get(layout.refCatalogKey()); + const auto after = op.read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(after); EXPECT_EQ(after->bytes, before->bytes); - EXPECT_EQ(after->token, before->token); + EXPECT_EQ(after->etag, before->etag); } /// A namespace whose catalog entry is ALREADY `Live` (e.g. admitted by an earlier mount that this @@ -269,13 +336,15 @@ TEST(CASRefCatalogBirthWiring, RestartFixturePreservesItsExistingNonemptyCatalog TEST(CASRefCatalogBirthWiring, AnExistingLiveEntryIsAdoptedRatherThanReminted) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/adopt_live"}; const CatalogEntry entry{.ns = ns, .state = NsState::Live, .incarnation = UInt128(0xcafe), .creator = std::nullopt}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ .life_epoch = store->writerEpoch(), .committed_through = std::nullopt, @@ -287,14 +356,58 @@ TEST(CASRefCatalogBirthWiring, AnExistingLiveEntryIsAdoptedRatherThanReminted) EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); /// The read result must outlive the returned pointer -- findEntry points into its entries. - const auto after_cut = CasRefCatalog::read(*backend, layout); + const auto after_cut = CasRefCatalog::read(op, layout); const CatalogEntry * after = findEntry(after_cut.catalog, ns); ASSERT_NE(after, nullptr); EXPECT_EQ(after->incarnation, UInt128(0xcafe)) << "adopted, not re-minted"; EXPECT_EQ(after->state, NsState::Live); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(0xcafe)); - EXPECT_TRUE(backend->head(layout.refLogKey(life, id)).exists); + EXPECT_TRUE(op.head(layout.refLogKey(life, id), Retry::standard()).has_value()); +} + +/// Regression (CI PR#2073, `tiered_storage_cas`, part `all_1_1_0` of a fresh table): seven concurrent +/// `MergeTreeBackgroundExecutor` movers all reached `resolveNamespaceLife`'s "no entry" read for the +/// SAME namespace before any of them landed a row. The winner's `createNamespace` ran its full three +/// steps to `Live` inside the WINDOW between the loop's own "no entry" read and the loser's own +/// `createNamespace` pre-check read -- so the loser's pre-check itself observed `Live`, not "no entry". +/// That must be adopted through the loop's normal re-read, never abort the server. +TEST(CASRefCatalogBirthWiring, ASiblingsFullCreateInsideCreateNamespacesOwnPreCheckWindowIsAdoptedNotAbort) +{ + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation sibling_op = requests.admit(); + auto store = openPoolForBirthTest(backend, "loser-server"); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/precheck_race"}; + const CreatorFence sibling_fence{.server_root_id = "sibling-server", .writer_epoch = 1, .fence_generation = 1}; + + /// Fires once, inside the LOSER's own `store->namespaceLife` -> `resolveNamespaceLife` -> + /// `createNamespace` call, right before that call's pre-check read -- i.e. AFTER + /// `resolveNamespaceLife`'s own loop already observed no entry. Runs a sibling's entire + /// `createNamespace` to completion in that window, so the loser's own pre-check read is the one + /// that observes the sibling's `Live` row. + CasRefCatalog::setCreateNamespacePreCheckHookForTest([&] + { + const auto sibling_outcome = CasRefCatalog::createNamespace(sibling_op, layout, 1, ns, sibling_fence); + ASSERT_EQ(sibling_outcome, CasRefCatalog::NamespaceCreationOutcome::Live); + }); + + std::optional life; + EXPECT_NO_THROW(life = store->namespaceLife(ns)); + + /// The read result must outlive the returned pointer -- findEntry points into its entries. + const auto snap = CasRefCatalog::read(sibling_op, layout); + size_t rows_for_ns = 0; + for (const CatalogEntry & e : snap.catalog.entries) + if (e.ns.string() == ns.string()) + ++rows_for_ns; + EXPECT_EQ(rows_for_ns, 1u) << "the loser's refused pre-check left no trace of its own"; + const CatalogEntry * entry = findEntry(snap.catalog, ns); + ASSERT_NE(entry, nullptr); + EXPECT_EQ(entry->state, NsState::Live); + ASSERT_TRUE(life.has_value()); + EXPECT_EQ(*life, NamespaceLifeId::fromCatalogEntry(ns, entry->incarnation)) << "the sibling's incarnation, adopted"; } /// OBLIGATION 3, pinned through the PRODUCTION path: a `Creating` entry left by a DIFFERENT, still-live @@ -302,7 +415,9 @@ TEST(CASRefCatalogBirthWiring, AnExistingLiveEntryIsAdoptedRatherThanReminted) /// `resolveNamespaceLife`/`reconcileStaleCreator`, just an ordinary `appendRefOps`. TEST(CASRefCatalogBirthWiring, ANamespaceStuckCreatingUnderALiveForeignFenceRefusesProductionPublicationByConstruction) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/stuck_creating"}; @@ -313,41 +428,42 @@ TEST(CASRefCatalogBirthWiring, ANamespaceStuckCreatingUnderALiveForeignFenceRefu const CreatorFence foreign_creator{.server_root_id = "ghost-server", .writer_epoch = 9, .fence_generation = 1}; const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(0xdead), .creator = foreign_creator}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); backend->resetCounts(); + backend->resetWriteCounts(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishBirth(store, ns, "a"); }); /// Nothing was written: the entry is exactly as observed, still Creating, still the foreign fence. /// The read result must outlive the returned pointer -- findEntry points into its entries. - const auto still_cut = CasRefCatalog::read(*backend, layout); + const auto still_cut = CasRefCatalog::read(op, layout); const CatalogEntry * still = findEntry(still_cut.catalog, ns); ASSERT_NE(still, nullptr); EXPECT_EQ(*still, entry) << "a refused resolution must write nothing"; - EXPECT_EQ(backend->putTotal(), 0u); - EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); } -/// The mirror image, and Task 3's own deferred obligation ("wire `reconcileStaleCreator` and pin it -/// with a test that drives reconciliation through the discovery path rather than by calling the -/// primitive directly"): a dead predecessor's `Creating` entry is reconciled onto THIS mount and -/// completed to `Live`, over the SAME incarnation -- resumption, not rebirth. +/// The mirror image, and the deferred obligation to wire `reconcileStaleCreator` and pin it with a +/// test that drives reconciliation through the discovery path rather than by calling the primitive +/// directly: a dead predecessor's `Creating` entry is reconciled onto THIS mount and completed to +/// `Live`, over the SAME incarnation -- resumption, not rebirth. TEST(CASRefCatalogBirthWiring, AStaleCreatingEntryFromATerminatedForeignFenceIsReconciledThroughTheProductionPath) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend, "this-server"); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/reconciled"}; /// A dead predecessor's `Creating` entry: its mount lease carries the clean-farewell sentinel - /// (`min_active == UINT64_MAX`), one of `isCreatorFenceTerminal`'s three certificates of death. + /// (`min_active_build_sequence == UINT64_MAX`), one of `isCreatorFenceTerminal`'s three certificates of death. const CreatorFence dead_creator{.server_root_id = "dead-server", .writer_epoch = 3, .fence_generation = 1}; const CatalogEntry entry{.ns = ns, .state = NsState::Creating, .incarnation = UInt128(0xbeef), .creator = dead_creator}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + CasRefCatalog::casAdmitEntry(op, layout, 1, entry); setWatermarkMinActive(*backend, layout, "dead-server", /*writer_epoch=*/3, - /*min_active=*/std::numeric_limits::max()); + /*min_active_build_sequence=*/std::numeric_limits::max()); /// The production path resumes creation itself: reconciles the stale entry onto THIS mount's own /// fence and completes it to `Live`, over the SAME incarnation the dead creator minted. @@ -355,7 +471,7 @@ TEST(CASRefCatalogBirthWiring, AStaleCreatingEntryFromATerminatedForeignFenceIsR EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); /// The read result must outlive the returned pointer -- findEntry points into its entries. - const auto live_cut = CasRefCatalog::read(*backend, layout); + const auto live_cut = CasRefCatalog::read(op, layout); const CatalogEntry * live = findEntry(live_cut.catalog, ns); ASSERT_NE(live, nullptr); EXPECT_EQ(live->state, NsState::Live); @@ -363,12 +479,14 @@ TEST(CASRefCatalogBirthWiring, AStaleCreatingEntryFromATerminatedForeignFenceIsR EXPECT_EQ(live->creator, std::nullopt); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, UInt128(0xbeef)); - EXPECT_TRUE(backend->head(layout.refLogKey(life, id)).exists); + EXPECT_TRUE(op.head(layout.refLogKey(life, id), Retry::standard()).has_value()); } TEST(CASRefCatalogBirthWiring, DropRefusesLiveCreatingFenceWithZeroCatalogMutation) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"drop_live_creator"}; @@ -377,20 +495,21 @@ TEST(CASRefCatalogBirthWiring, DropRefusesLiveCreatingFenceWithZeroCatalogMutati .state = NsState::Creating, .incarnation = UInt128{0xd001}, .creator = CreatorFence{.server_root_id = "unproven-live", .writer_epoch = 7, .fence_generation = 1}}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + CasRefCatalog::casAdmitEntry(op, layout, 1, creating); backend->resetCounts(); + backend->resetWriteCounts(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); - EXPECT_EQ(backend->putTotal(), 0u); - EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); EXPECT_EQ(backend->deleteTotal(), 0u); - EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{creating}); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{creating}); } TEST(CASRefCatalogBirthWiring, DropDeletesTerminalCreatingExactlyAndLeavesCkptForJanitor) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"drop_terminal_creator"}; @@ -399,19 +518,19 @@ TEST(CASRefCatalogBirthWiring, DropDeletesTerminalCreatingExactlyAndLeavesCkptFo .state = NsState::Creating, .incarnation = UInt128{0xd002}, .creator = CreatorFence{.server_root_id = "dead-creator", .writer_epoch = 8, .fence_generation = 1}}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + CasRefCatalog::casAdmitEntry(op, layout, 1, creating); setWatermarkMinActive(*backend, layout, "dead-creator", 8, std::numeric_limits::max()); const NamespaceLifeId old_life = NamespaceLifeId::fromCatalogEntry(ns, creating.incarnation); const String ckpt_key = layout.refCkptKey(old_life); - ASSERT_EQ(backend->putIfAbsent(ckpt_key, "stalled-ckpt").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(ckpt_key, "stalled-ckpt", Retry::standard()))); backend->resetCounts(); + backend->resetWriteCounts(); store->dropNamespace(ns); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), 1u); EXPECT_EQ(backend->deleteTotal(), 0u); - EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); - EXPECT_TRUE(backend->head(ckpt_key).exists); - EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + EXPECT_TRUE(op.head(ckpt_key, Retry::standard()).has_value()); + EXPECT_TRUE(CasRefCatalog::read(op, layout).catalog.entries.empty()); const NamespaceLifeId reborn = store->namespaceLife(ns); EXPECT_NE(reborn.incarnation, old_life.incarnation); @@ -420,6 +539,8 @@ TEST(CASRefCatalogBirthWiring, DropDeletesTerminalCreatingExactlyAndLeavesCkptFo TEST(CASRefCatalogBirthWiring, DropLosesExactCreatingRaceToReconciliationWithoutDeletingCkpt) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"drop_reconcile_race"}; @@ -427,29 +548,29 @@ TEST(CASRefCatalogBirthWiring, DropLosesExactCreatingRaceToReconciliationWithout .server_root_id = "dead-racing-creator", .writer_epoch = 9, .fence_generation = 1}; const CatalogEntry creating{ .ns = ns, .state = NsState::Creating, .incarnation = UInt128{0xd003}, .creator = old_creator}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + CasRefCatalog::casAdmitEntry(op, layout, 1, creating); setWatermarkMinActive( *backend, layout, old_creator.server_root_id, old_creator.writer_epoch, std::numeric_limits::max()); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, creating.incarnation); const String ckpt_key = layout.refCkptKey(life); - ASSERT_EQ(backend->putIfAbsent(ckpt_key, "stalled-ckpt").outcome, PutOutcome::Done); - backend->on_catalog_cas = [&] + ASSERT_TRUE(std::holds_alternative(op.create(ckpt_key, "stalled-ckpt", Retry::standard()))); + backend->on_catalog_write = [&] { EXPECT_EQ(CasRefCatalog::reconcileStaleCreator( - *backend, layout, creating, + op, layout, creating, CreatorFence{.server_root_id = "replacement", .writer_epoch = 10, .fence_generation = 1}, - [](const CreatorFence &) { return true; }, store->fenceGeneration(), - [](uint64_t) {}), CasRefCatalog::ReconcileCreatorOutcome::Reconciled); + [](const CreatorFence &) { return true; }), + CasRefCatalog::ReconcileCreatorOutcome::Reconciled); }; backend->race_armed = true; backend->resetCounts(); + backend->resetWriteCounts(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); EXPECT_EQ(backend->deleteTotal(), 0u); - EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); - EXPECT_TRUE(backend->head(ckpt_key).exists); - const CasRefCatalog::Snapshot after = CasRefCatalog::read(*backend, layout); + EXPECT_TRUE(op.head(ckpt_key, Retry::standard()).has_value()); + const CasRefCatalog::Snapshot after = CasRefCatalog::read(op, layout); ASSERT_EQ(after.catalog.entries.size(), 1u); ASSERT_TRUE(after.catalog.entries.front().creator); EXPECT_EQ(after.catalog.entries.front().creator->server_root_id, "replacement"); @@ -458,6 +579,8 @@ TEST(CASRefCatalogBirthWiring, DropLosesExactCreatingRaceToReconciliationWithout TEST(CASRefCatalogBirthWiring, FencedDropCannotCancelTerminalCreating) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"fenced_drop_terminal_creator"}; @@ -466,29 +589,32 @@ TEST(CASRefCatalogBirthWiring, FencedDropCannotCancelTerminalCreating) .state = NsState::Creating, .incarnation = UInt128{0xd004}, .creator = CreatorFence{.server_root_id = "dead-fenced-creator", .writer_epoch = 11, .fence_generation = 1}}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, creating); + CasRefCatalog::casAdmitEntry(op, layout, 1, creating); setWatermarkMinActive(*backend, layout, "dead-fenced-creator", 11, std::numeric_limits::max()); backend->resetCounts(); + backend->resetWriteCounts(); /// The first cancellation attempt passes its fence check, then loses its catalog CAS while the /// local mount is re-armed at a new fence generation. The retry must re-check the caller fence and /// refuse before another catalog mutation attempt. - backend->on_catalog_cas = [&] + backend->on_catalog_write = [&] { rearmMountFenceAfterAnomalyForTest(store); - backend->failNextCasPut(layout.refCatalogKey()); + backend->refuseNextWrite(layout.refCatalogKey()); }; backend->race_armed = true; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 1u); + EXPECT_EQ(backend->writes(layout.refCatalogKey()), 1u); EXPECT_EQ(backend->deleteTotal(), 0u); - EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{creating}); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{creating}); } TEST(CASRefCatalogBirthWiring, ExactOldLifeCannotCancelReplacementTerminalCreating) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); auto store = openPoolForBirthTest(backend); const Layout & layout = store->layout(); const RootNamespace ns{"exact_old_life_terminal_creator"}; @@ -498,15 +624,97 @@ TEST(CASRefCatalogBirthWiring, ExactOldLifeCannotCancelReplacementTerminalCreati .state = NsState::Creating, .incarnation = UInt128{0xd006}, .creator = CreatorFence{.server_root_id = "dead-successor-creator", .writer_epoch = 12, .fence_generation = 1}}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, successor); + CasRefCatalog::casAdmitEntry(op, layout, 1, successor); setWatermarkMinActive(*backend, layout, "dead-successor-creator", 12, std::numeric_limits::max()); const String ckpt_key = layout.refCkptKey(NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation)); - ASSERT_EQ(backend->putIfAbsent(ckpt_key, "successor-ckpt").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(op.create(ckpt_key, "successor-ckpt", Retry::standard()))); backend->resetCounts(); + backend->resetWriteCounts(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(predecessor); }); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); EXPECT_EQ(backend->deleteTotal(), 0u); - EXPECT_EQ(backend->deleteCount(ckpt_key), 0u); - EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries, std::vector{successor}); + EXPECT_EQ(CasRefCatalog::read(op, layout).catalog.entries, std::vector{successor}); +} + +namespace +{ + +/// Leaves the catalog holding a body no previously observed entry equals: every read first bumps each +/// entry's creator fence generation, so `reconcileStaleCreator`'s token-exactness check refuses on +/// every attempt and `resolveNamespaceLife`'s state machine can never converge. +class ChurningCatalogBackend final : public InMemoryBackend +{ +public: + explicit ChurningCatalogBackend(String catalog_key_) + : catalog_key(std::move(catalog_key_)) + { + } + + bool churning = false; + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (churning && key == catalog_key) + bumpEveryCreatorFence(access); + return InMemoryBackend::read(key, access); + } + +private: + /// Qualified calls, never the virtual ones: this must not re-enter its own churn. + void bumpEveryCreatorFence(DB::Cas::TransportAccess & access) + { + const std::optional got = InMemoryBackend::read(catalog_key, access); + if (!got) + return; + RefCatalog catalog = decodeRefCatalog(got->bytes); + for (CatalogEntry & entry : catalog.entries) + if (entry.creator) + ++entry.creator->fence_generation; + (void)InMemoryBackend::write(catalog_key, encodeRefCatalog(catalog), got->value, access); + } + + const String catalog_key; +}; + +} + +/// A catalog entry that moves under every read drives `resolveNamespaceLife`'s state machine for ever. +/// One `Retry` frozen before the loop bounds the WHOLE resolution to a single standard window, and +/// every re-read a competing actor forces is paced by a jittered sleep -- so a permanently churning +/// catalog costs one window, not one fresh window per verb per iteration hammered with no wait between +/// them. +/// +/// What the clock bound below does NOT check: it bounds this call's own wall time, not the number of +/// requests the loop sent, and it says nothing about the paths that converge. +TEST(CASRefCatalogBirthWiring, APerpetuallyChurningCatalogEntryIsPacedAndEndsWithinOneWindow) +{ + auto backend = std::make_shared(Layout{"p"}.refCatalogKey()); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + auto store = openPoolForBirthTest(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"churning_creating"}; + + /// A FOREIGN creator, so the loop takes the reconciliation branch on every iteration. + const CatalogEntry creating{ + .ns = ns, + .state = NsState::Creating, + .incarnation = UInt128{0xc001}, + .creator = CreatorFence{.server_root_id = "foreign-creator", .writer_epoch = 7, .fence_generation = 1}}; + CasRefCatalog::casAdmitEntry(op, layout, 1, creating); + + auto clock = DB::Cas::tests::VirtualRetryClock::installOn(store); + backend->churning = true; + const size_t pauses_before = clock->pauseCount(); + const uint64_t now_before = clock->nowMs(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->namespaceLife(ns); }); + + /// Sixteen jittered draws cannot exhaust a 90 s window even at their ceiling, so an unpaced loop + /// reaches its iteration cap having slept nothing at all. + EXPECT_GE(clock->pauseCount() - pauses_before, 16u) << "each forced re-read is paced"; + /// The frozen window, plus at most one backoff draw the pace does not consult the deadline for, + /// plus the virtual clock's one extra millisecond per pause. + EXPECT_LE(clock->nowMs() - now_before, 95'100u) << "one standard window bounds the whole loop"; } diff --git a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp index 2cdc32ff1851..a79f58fdf424 100644 --- a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp +++ b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp @@ -2,6 +2,8 @@ #include "config.h" +#include + #include #include #include @@ -36,6 +38,7 @@ namespace DB::ErrorCodes extern const int LIMIT_EXCEEDED; extern const int CORRUPTED_DATA; extern const int NETWORK_ERROR; +extern const int LOGICAL_ERROR; } namespace ProfileEvents @@ -51,9 +54,8 @@ extern const Event CASRefSnapshotPublishDispatched; /// single item whose own op count, or whose one op's encoded size, exceeds its cap fails ALONE; a /// neighbor co-batched into the same flush still commits. `ref_txn_max_ops` is checked exactly (the /// `build_ops` result's size), and the per-op cap is checked by encoding exactly one op at a time -- -/// no accumulation, matching the admission machinery this replaces. T9 (removal-class detection by -/// op inspection) and T10 (chunked flush across a whole-batch op-count overflow) extend this file; -/// this task adds only the per-item / per-op isolation tests and the canonical round-trip leg of +/// no accumulation, matching the admission machinery. The per-item / per-op isolation tests and +/// the canonical round-trip leg cover /// test 12 (the maximum legally-admissible normal-class transaction). /// /// The suite name is prefixed `RefWriter` so it is covered by the `RefWriter*` unit-test gate filter. @@ -97,7 +99,7 @@ PoolPtr openPool(const BackendPtr & backend) /// production birth mints a random incarnation and those computed keys land nowhere real. void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String & ref) { - DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + DB::Cas::tests::casAdmitRecoverableEntry(*s->poolBackendPtr(), s->layout(), ns, s->liveWriterEpoch()); PartWriteInfo info; info.intended_namespace = ns; info.intended_ref = ns.string() + "/" + ref; @@ -113,6 +115,12 @@ struct Caller { std::thread t; std::future fut; + + /// A fatal `ASSERT_*` between launch and the explicit `t.join()` below returns from `TestBody` with + /// `t` still joinable; `std::thread::~thread` on a joinable thread calls `std::terminate`, aborting + /// the whole binary and discarding every later test. This destructor is the backstop: on every + /// success path the explicit join already ran and left nothing for it to do. + ~Caller() { if (t.joinable()) t.join(); } }; Caller launchAppend(const PoolPtr & store, const RootNamespace & ns, MutationScope scope, @@ -246,6 +254,9 @@ TEST(CASRefWriterChunkedFlush, OversizedItemFailsAlone) auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 2); + /// `fillerOps` returns default `RefOp{}` of kind `NamespaceBirth`, which names no ref, so the + /// item's scope name is never compared against anything and the scope check (step 3) would pass + /// this item even with the op-count cap removed -- the cap is what's under test here, not the scope. Caller oversized = launchAppend(store, ns, MutationScope::ref("oversized"), [](const RefTableState &) -> std::vector { return fillerOps(ref_txn_max_ops + 1); }); waitEntered(sync); @@ -283,7 +294,7 @@ TEST(CASRefWriterChunkedFlush, OversizedOpFailsItsItemAlone) auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 2); - Caller oversized = launchAppend(store, ns, MutationScope::ref("oversized_op"), + Caller oversized = launchAppend(store, ns, MutationScope::ref(oversized_op.ref_name), [oversized_op](const RefTableState &) -> std::vector { return {oversized_op}; }); waitEntered(sync); Caller neighbor = launchDrop(store, ns, "neighbor"); @@ -303,6 +314,49 @@ TEST(CASRefWriterChunkedFlush, OversizedOpFailsItsItemAlone) EXPECT_FALSE(store->resolveRef(ns, "neighbor").has_value()) << "neighbor's drop must have committed"; } +/// Per-item isolation of the `MutationScope` validation (`CasRefLedger.cpp`'s `flushRefBatch` step 3): +/// a mis-scoped item co-batched with an innocent neighbor must fail ALONE, and the neighbor's mutation +/// must COMMIT -- not merely avoid throwing -- exactly the batch-isolation shape +/// `OversizedOpFailsItsItemAlone` proves for the step-1 admission caps. Release-arm only: in a debug or +/// sanitizer build the mis-scoped item's `LOGICAL_ERROR` aborts the whole process, taking the co-batched +/// neighbor down with it before either assertion can run. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRefWriterChunkedFlush, MisScopedItemFailsAloneNeighborCommits) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const RootNamespace ns{"srv1/chunked_misscoped"}; + publishEmptyPart(store, ns, "neighbor"); + ASSERT_TRUE(store->resolveRef(ns, "neighbor").has_value()); + + RefOp add; + add.kind = RefOpKind::OwnerTransition; + add.new_binding = RefOwnerBinding{RefOwnerKind::Precommit, "y", ManifestRef{900000004, 1, 1}}; + + auto sync = std::make_shared(); + armPreCarveBlock(store, ns, sync, 2); + + Caller misscoped = launchAppend(store, ns, MutationScope::ref("x"), + [add](const RefTableState &) -> std::vector { return {add}; }); + waitEntered(sync); + Caller neighbor = launchDrop(store, ns, "neighbor"); + waitPendingAtLeast(store, ns, 2); + sync->cv.notify_all(); + + ASSERT_EQ(misscoped.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "mis-scoped item must not hang"; + ASSERT_EQ(neighbor.fut.wait_for(std::chrono::seconds(10)), std::future_status::ready) << "neighbor must not hang"; + const std::exception_ptr misscoped_err = misscoped.fut.get(); + const std::exception_ptr neighbor_err = neighbor.fut.get(); + misscoped.t.join(); + neighbor.t.join(); + store->setRefPreCarveHookForTest(nullptr); + + expectFailedWithCode(misscoped_err, DB::ErrorCodes::LOGICAL_ERROR, "mis-scoped item"); + EXPECT_TRUE(neighbor_err == nullptr) << "the co-batched neighbor must commit despite the mis-scoped item"; + EXPECT_FALSE(store->resolveRef(ns, "neighbor").has_value()) << "neighbor's drop must have committed"; +} +#endif + /// Test 12, canonical round-trip leg: the maximum legally-admissible normal-class transaction under /// the new counts-only caps -- `ref_txn_max_ops` ops, each padded to exactly `ref_op_max_bytes` -- /// round-trips comfortably under the whole-transaction `ref_txn_max_bytes` decode cap (5000 * 4096 = @@ -383,7 +437,9 @@ TEST(CASRefWriterChunkedFlush, DropNamespaceOverOpCapSucceeds) .checkpoint_snapshot_id = RefTxnId{epoch, 1}, .last_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, layout, ns).value(); + CasRequests catalog_requests(backend, Fence::open()); + CasOperation catalog_op = catalog_requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(catalog_op, layout, ns).value(); backend->resetCounts(); ASSERT_EQ(store->listRefs(ns).size(), kTotalRefs); EXPECT_EQ(backend->getCount(layout.refLogKey(life, RefTxnId{epoch, 1})), 1u); @@ -392,7 +448,7 @@ TEST(CASRefWriterChunkedFlush, DropNamespaceOverOpCapSucceeds) DropNamespaceStats stats; EXPECT_NO_THROW(stats = store->dropNamespace(ns)); EXPECT_EQ(stats.committed_refs, kTotalRefs); - EXPECT_EQ(CasRefCatalog::read(*backend, layout).catalog.entries.front().state, NsState::Removing); + EXPECT_EQ(CasRefCatalog::read(catalog_op, layout).catalog.entries.front().state, NsState::Removing); } /// Test 11, second leg: `WholeShard` scope ALONE is not the removal-class discriminator -- the @@ -450,22 +506,22 @@ PoolPtr openPoolWith(const BackendPtr & backend, PoolConfig cfg) return Pool::open(backend, cfg); } -/// `num_pairs` add-then-remove precommit op pairs (2 * `num_pairs` ops total) for distinct refs -/// (`prefix` + zero-padded index) each naming a distinct valid manifest. Every pair adds a precommit -/// binding and immediately removes it, so the LIVE state (the `precommits` set, the committed COW map, -/// the owned-manifest index) stays ~empty throughout the whole transaction -- keeping the per-op -/// `admits` preview and the sanitizer-only body-counter assert O(1), so validating a maximal chunk of -/// thousands of ops stays O(ops), not O(ops^2). It is the OP COUNT (not the resident state) that drives -/// the chunk split under test; each op is tiny (well under `ref_op_max_bytes`), so the whole run is -/// admissible on a `Live` namespace. The durable transaction still carries every op verbatim, so a -/// chunk's ops can be compared against the exact expected vector. -std::vector addRemovePrecommitPairs(const String & prefix, size_t num_pairs, uint64_t manifest_epoch) +/// `num_pairs` add-then-remove precommit op pairs (2 * `num_pairs` ops total) on ONE ref, each pair +/// naming a distinct valid manifest, so an item scoped `MutationScope::ref(ref)` names exactly the ref +/// its ops mutate (the flush validates that). Every pair adds a precommit binding and immediately +/// removes it, so the LIVE state (the `precommits` set, the committed COW map, the owned-manifest +/// index) stays ~empty throughout the whole transaction -- keeping the per-op `admits` preview and the +/// sanitizer-only body-counter assert O(1), so validating a maximal chunk of thousands of ops stays +/// O(ops), not O(ops^2). It is the OP COUNT (not the resident state) that drives the chunk split under +/// test; each op is tiny (well under `ref_op_max_bytes`), so the whole run is admissible on a `Live` +/// namespace. The durable transaction still carries every op verbatim, so a chunk's ops can be compared +/// against the exact expected vector. +std::vector addRemovePrecommitPairs(const String & ref, size_t num_pairs, uint64_t manifest_epoch) { std::vector ops; ops.reserve(num_pairs * 2); for (size_t i = 0; i < num_pairs; ++i) { - const String ref = prefix + paddedRefName(i); const ManifestRef manifest{manifest_epoch, i + 1, 1}; RefOp add; add.kind = RefOpKind::OwnerTransition; @@ -484,11 +540,12 @@ std::vector addRemovePrecommitPairs(const String & prefix, size_t num_pai /// breaks the inventory. Reads the backend directly (no Pool cache). std::vector listLogTxns(DB::Cas::Backend & backend, const DB::Cas::Layout & layout, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest operation(backend); std::vector ids; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*operation).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -504,7 +561,7 @@ std::vector listLogTxns(DB::Cas::Backend & backend, const DB::Cas::La std::vector txns; for (const RefTxnId & id : ids) { - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + const auto got = (*operation).read(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id), Retry::standard()); if (!got) continue; try @@ -532,6 +589,10 @@ struct AppendCaller { std::thread t; std::future fut; + + /// Same hazard as `Caller` above (see its destructor comment): a fatal `ASSERT_*` before the + /// explicit join leaves `t` joinable, and a joinable thread's destructor calls `std::terminate`. + ~AppendCaller() { if (t.joinable()) t.join(); } }; AppendCaller launchAppendOps(const PoolPtr & store, const RootNamespace & ns, MutationScope scope, @@ -578,9 +639,9 @@ TEST(CASRefWriterChunkedFlush, ChunkedFlushCommitsPerChunk) /// 2000 ops per item (1000 add/remove pairs) -> 6000 > ref_txn_max_ops (5000): chunk 1 = /// {item_a,item_b} (4000), chunk 2 = {item_c} (2000). - const std::vector ops1 = addRemovePrecommitPairs("aaa_", 1000, 900000001); - const std::vector ops2 = addRemovePrecommitPairs("bbb_", 1000, 900000002); - const std::vector ops3 = addRemovePrecommitPairs("ccc_", 1000, 900000003); + const std::vector ops1 = addRemovePrecommitPairs("item_a", 1000, 900000001); + const std::vector ops2 = addRemovePrecommitPairs("item_b", 1000, 900000002); + const std::vector ops3 = addRemovePrecommitPairs("item_c", 1000, 900000003); auto c1 = std::make_shared>(0); auto c2 = std::make_shared>(0); auto c3 = std::make_shared>(0); @@ -664,28 +725,119 @@ namespace /// (the leader's own item), chunk 2 = {item_b}. `mode` faults ONLY chunk 2's `_log/` PUT (skip chunk 1). /// In every variant chunk 1 commits and the leader's own call returns chunk 1's real id, while chunk 2's /// caller fails. Returns the two callers' results plus chunk 1's id for the per-variant assertions. +/// The engine reissues an unresolved write until its OWN retry window closes, and that window is +/// measured on a clock the engine reads. Both seams here share one counter -- the sleep the engine +/// performs is what advances the clock -- so a fault that stays armed ends the call at its deadline +/// with no real time passing. Installed on the whole pool, because the ref-lane write, its settling +/// read and the recovery retry loop all pace through the same seam. The pool owns the closures and the +/// closures own the clock, so it outlives everything that can still read it. +class VirtualRetryClock +{ +public: + static std::shared_ptr installOn(const PoolPtr & store) + { + auto clock = std::make_shared(); + store->setCasRequestNowFnForTest([clock] { return clock->nowMs(); }); + store->setCasRetrySleepForTest([clock](uint64_t ms) { clock->advance(ms); }); + return clock; + } + + uint64_t nowMs() const + { + std::lock_guard lock(mutex); + return now_ms; + } + size_t pauseCount() const + { + std::lock_guard lock(mutex); + return pauses; + } + uint64_t longestPause() const + { + std::lock_guard lock(mutex); + return longest_pause; + } + + void advance(uint64_t ms) + { + std::lock_guard lock(mutex); + /// Plus one millisecond, because full jitter can draw a ZERO pause: a clock that does not move + /// would leave the loop reissuing for ever against a fault that never clears. + now_ms += ms + 1; + ++pauses; + longest_pause = std::max(longest_pause, ms); + } + +private: + mutable std::mutex mutex; + uint64_t now_ms = 0; + size_t pauses = 0; + uint64_t longest_pause = 0; +}; + +/// `ChunkFaultBackend` COUNTS its faults, and a count can no longer make one conclusive: the write +/// engine settles every ambiguity by an exact read and then REISSUES, so a fault that runs out +/// mid-call is answered by the next attempt instead of by the call's own deadline. This keeps it armed +/// until the latch is cleared, on both legs -- the write's, and the lost read `Mode::LandedThenLost` +/// arms, which the read engine would otherwise simply reissue past. +class LatchedChunkFaultBackend : public ChunkFaultBackend +{ +public: + bool latched = false; + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (latched && !fail_read_once_key.empty() && key == fail_read_once_key) + throw Poco::TimeoutException("LatchedChunkFaultBackend: the lost read stays lost"); + return ChunkFaultBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + if (latched && mode != Mode::None && fault_skip == 0 && !expected_value && !fault_substr.empty() + && key.find(fault_substr) != String::npos) + fault_count = 1; + return ChunkFaultBackend::write(key, bytes, expected_value, access); + } + + void disarm() + { + latched = false; + mode = Mode::None; + fault_count = 0; + fault_skip = 0; + fail_read_once_key.clear(); + } +}; + struct ChunkFailureOutcome { AppendResult leader; /// item_a, chunk 1 AppendResult follower; /// item_b, chunk 2 RefTxnId chunk1_id{}; - std::shared_ptr backend; + std::shared_ptr backend; PoolPtr store; + std::shared_ptr clock; }; ChunkFailureOutcome runChunkFailureCase(const String & ns_suffix, ChunkFaultBackend::Mode mode) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); PoolConfig cfg; - /// Single-attempt budget: one ambiguous PUT is a conclusive Unresolved (wedge) / DefiniteFailure, - /// with no inter-attempt sleep to serve. + /// The budget bounds the mount lease's own admission arithmetic and nothing else -- a write's + /// attempt count is the `Retry` policy's. What makes the injected fault conclusive is that it + /// stays armed for the whole call while the injected clock carries the call to its own deadline. CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; cfg.cas_request_budget = budget; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); auto store = openPoolWith(backend, cfg); + auto clock = VirtualRetryClock::installOn(store); const DB::Cas::Layout & layout = store->layout(); const RootNamespace ns{String("srv1/") + ns_suffix}; publishEmptyPart(store, ns, "seed"); @@ -696,14 +848,15 @@ ChunkFailureOutcome runChunkFailureCase(const String & ns_suffix, ChunkFaultBack backend->mode = mode; backend->fault_skip = 1; backend->fault_count = 1; + backend->latched = true; auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 2); /// 3000 ops per item (1500 add/remove pairs) -> 6000 > ref_txn_max_ops: chunk 1 = {item_a}, /// chunk 2 = {item_b}. - AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), nullptr); + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("item_a", 1500, 900000001), nullptr); waitEntered(sync); - AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), nullptr); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("item_b", 1500, 900000002), nullptr); waitPendingAtLeast(store, ns, 2); sync->cv.notify_all(); @@ -714,19 +867,23 @@ ChunkFailureOutcome runChunkFailureCase(const String & ns_suffix, ChunkFaultBack out.follower = b.fut.get(); a.t.join(); b.t.join(); + /// Disarmed before anything else touches the store: a topped-up count would fault the pool's own + /// teardown writes too. + backend->disarm(); store->setRefPreCarveHookForTest(nullptr); out.chunk1_id = out.leader.id; out.backend = backend; out.store = store; + out.clock = clock; return out; } } -/// Test 9 (chunk-failure variant a -- definite failure): chunk 2's PUT is conclusively rejected -/// (`CasWriteOutcome::DefiniteFailure`). Chunk 1's caller (the leader's own item) observes SUCCESS with -/// chunk 1's real id; chunk 2's caller fails; the lane does NOT wedge (a definite rejection is a safe -/// gap, not an uncertain PUT). +/// Test 9 (chunk-failure variant a -- definite failure): chunk 2's create is conclusively rejected by +/// the store. Chunk 1's caller (the leader's own item) observes SUCCESS with chunk 1's real id; chunk +/// 2's caller fails; the lane does NOT wedge (a proven refusal is a safe gap, not an uncertain +/// write). TEST(CASRefWriterChunkedFlush, ChunkFailureDefinite) { #if !USE_AWS_S3 @@ -755,6 +912,11 @@ TEST(CASRefWriterChunkedFlush, ChunkFailureWedge) ChunkFailureOutcome out = runChunkFailureCase("chunk_fail_wedge", ChunkFaultBackend::Mode::Unresolved); ASSERT_TRUE(out.leader.err == nullptr) << "chunk-1 caller must observe success even though chunk 2 wedged"; ASSERT_TRUE(out.follower.err != nullptr) << "chunk-2 caller must observe the append failure"; + /// The give-up was chunk 2's OWN retry window: the fault outlasted several reissues, and every one + /// of them paced through the injected sleep rather than a real one. + EXPECT_GT(out.clock->pauseCount(), 1u); + EXPECT_LE(out.clock->longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; + EXPECT_GE(out.clock->nowMs(), 60000u); EXPECT_TRUE(out.store->refLaneWedgedForTest(ns)) << "chunk 2's unresolved PUT must wedge the lane"; RefTxnId chunk2_id = out.chunk1_id; @@ -824,9 +986,9 @@ TEST(CASRefWriterChunkedFlush, LeaderOwnItemCommittedBeforeThrow) auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 2); /// 3000 ops per item (1500 add/remove pairs) -> chunk 1 = {item_a}, boundary throw before chunk 2. - AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), c1); + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("item_a", 1500, 900000001), c1); waitEntered(sync); - AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), c2); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("item_b", 1500, 900000002), c2); waitPendingAtLeast(store, ns, 2); sync->cv.notify_all(); @@ -849,7 +1011,7 @@ TEST(CASRefWriterChunkedFlush, LeaderOwnItemCommittedBeforeThrow) if (txn.txn_id == ra.id) chunk1_txn = txn; ASSERT_TRUE(chunk1_txn.has_value()) << "chunk 1 must be durable"; - EXPECT_EQ(chunk1_txn->ops, addRemovePrecommitPairs("aaa_", 1500, 900000001)); + EXPECT_EQ(chunk1_txn->ops, addRemovePrecommitPairs("item_a", 1500, 900000001)); /// item_a's build_ops ran once (chunk 1); item_b's ran once (before the boundary throw preempted its /// validation) and is NOT re-invoked -- the at-most-once contract holds through the failed tenure. EXPECT_EQ(c1->load(), 1); @@ -870,26 +1032,71 @@ TEST(CASRefWriterChunkedFlush, SnapshotPublisherLatchedAcrossChunks) auto store = openPoolWith(backend, cfg); const DB::Cas::Layout & layout = store->layout(); const RootNamespace ns{"srv1/chunk_snapshot_coalesce"}; - publishEmptyPart(store, ns, "seed"); + /// NOT `publishEmptyPart`: that helper makes `precommitAdd` (which folds an implicit + /// `namespace_birth` and the add into ONE transaction, since the namespace is not yet `Live`) and + /// `promote` two separate, back-to-back `appendRefOps` calls. `precommitAdd`'s own post-commit + /// trigger (`maybeScheduleSnapshotPublish` at the end of `commitRefChunk`) dispatches a background + /// publisher for what it just committed; with `snapshot_log_count_threshold` at 0 (every single + /// commit is eligible -- never reachable through a normal, 256-count threshold) that publisher can + /// still be in flight when `promote`, moments later, becomes lane leader for its OWN write and + /// moves the lane to `Writing` -- a real, reachable race between that capture and this transition. + /// A lost race backs the publisher off, and this pool's frozen `boot_ms_fn` (see `openPool`'s + /// comment) never advances past that deadline, so the backoff never clears on its own, poisoning + /// every later dispatch on `ns` for the rest of the test, including chunk 1's. Draining + /// (`waitForSnapshotPublishSettleForTest`) between the two commits removes the in-flight publisher + /// `promote` would otherwise race, and driving one explicitly + /// (`tryPublishSnapshotAndAdvanceCheckpointOnce`, the direct synchronous seam built for exactly this + /// -- "public so tests can drive one attempt deterministically without depending on the background + /// dispatch's timing") covers anything a dispatch was never even admitted for. + DB::Cas::tests::casAdmitRecoverableEntry(*store->poolBackendPtr(), store->layout(), ns, store->liveWriterEpoch()); + PartWriteInfo seed_info; + seed_info.intended_namespace = ns; + seed_info.intended_ref = ns.string() + "/seed"; + auto seed_build = store->beginPartWrite(seed_info); + const ManifestId seed_manifest_id = seed_build->stageManifest({}); + seed_build->precommitAdd(ns, "seed", seed_manifest_id); + store->waitForSnapshotPublishSettleForTest(ns); + store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); + seed_build->promote(ns, "seed", seed_build->buildId(), seed_manifest_id); store->waitForSnapshotPublishSettleForTest(ns); /// drain the seed's publish chain -> tail == 0 + store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns); /// Latch the FIRST `_snap/` PUT (chunk 1's publisher) at its conditional PUT -- i.e. AFTER it has /// captured chunk 1's prefix under state_mutex. backend->armBlock(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_snap/"); - /// Gate the leader at the chunk boundary until that publisher has parked on its PUT, so its captured - /// candidate is EXACTLY chunk 1's prefix (not chunk 1 + chunk 2). - store->setCarveHookForTest([backend](CasRefLedger::CarvePhaseForTest ph) + /// Gate the leader at the chunk boundary on the publisher's CAPTURE (`snapshot_after_capture_hook_for_test`, + /// fired the moment the publish attempt has read `rt->state` under `state_mutex` and passed the + /// `lane_state == Ready` admission check), not on the publisher's later blocked PUT. Capture is + /// causally prior to the PUT the block intercepts -- it is the OTHER side of the very check + /// (`CasRefLedger.cpp`'s `tryPublishSnapshotAndAdvanceCheckpointOnceOnRuntimeImpl`) whose failure logs + /// "refusing snapshot publication while the append lane is not Ready" -- so waiting for it is the + /// exact fact the boundary needs, whereas waiting for the PUT also waits on however long it takes the + /// dispatched publish to be scheduled onto a worker thread at all. Under contention that scheduling + /// delay can outlast a bounded wait, and a leader released by the wait's own timeout (rather than by + /// the capture it was meant to prove) can start chunk 2 -- moving the lane to `Writing` -- before the + /// not-yet-scheduled publisher ever captures, so it captures Writing and is refused. + auto captured = std::make_shared(); + store->setSnapshotAfterCaptureHookForTest([captured] { - if (ph == CasRefLedger::CarvePhaseForTest::ChunkReseed) - backend->awaitBlockEntered(); + std::lock_guard lk(captured->m); + captured->entered = true; + captured->cv.notify_all(); + }); + store->setCarveHookForTest([captured](CasRefLedger::CarvePhaseForTest ph) + { + if (ph != CasRefLedger::CarvePhaseForTest::ChunkReseed) + return; + std::unique_lock lk(captured->m); + captured->cv.wait_for(lk, std::chrono::seconds(10), [&] { return captured->entered; }); + ASSERT_TRUE(captured->entered) << "chunk 1's snapshot publisher never captured its candidate within 10s"; }); auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 2); /// 3000 ops per item (1500 add/remove pairs) -> chunk 1 = {item_a}, chunk 2 = {item_b}. - AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("aaa_", 1500, 900000001), nullptr); + AppendCaller a = launchAppendOps(store, ns, MutationScope::ref("item_a"), addRemovePrecommitPairs("item_a", 1500, 900000001), nullptr); waitEntered(sync); - AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("bbb_", 1500, 900000002), nullptr); + AppendCaller b = launchAppendOps(store, ns, MutationScope::ref("item_b"), addRemovePrecommitPairs("item_b", 1500, 900000002), nullptr); waitPendingAtLeast(store, ns, 2); sync->cv.notify_all(); @@ -905,11 +1112,18 @@ TEST(CASRefWriterChunkedFlush, SnapshotPublisherLatchedAcrossChunks) ++chunk2_id.ref_sequence; EXPECT_EQ(rb.id, chunk2_id); + /// Independently confirm the latched publisher actually reached its blocked PUT, on the main test + /// thread rather than as the leader's own gate: without this, a wiring regression that never parks + /// the publisher would let the settlement assertion below pass VACUOUSLY (a direct, non-coalesced + /// dispatch can still cover chunk 2). + backend->awaitBlockEntered(); + /// Release the latched chunk-1 publisher. Its settlement must re-fire the chunk-2 trigger the /// single-flight gate dropped -> a follow-up publication covers chunk 2. backend->releaseBlock(); store->waitForSnapshotPublishSettleForTest(ns); store->setCarveHookForTest(nullptr); + store->setSnapshotAfterCaptureHookForTest(nullptr); store->setRefPreCarveHookForTest(nullptr); const std::optional newest = store->newestPublishedSnapshotIdForTest(ns); diff --git a/src/Disks/tests/gtest_cas_ref_ckpt.cpp b/src/Disks/tests/gtest_cas_ref_ckpt.cpp index 9b63fa4dfc92..f3f7e08efbd7 100644 --- a/src/Disks/tests/gtest_cas_ref_ckpt.cpp +++ b/src/Disks/tests/gtest_cas_ref_ckpt.cpp @@ -1,6 +1,7 @@ #include #include "config.h" +#include "cas_format_test_battery.h" #include #include @@ -17,10 +18,14 @@ #include +#include #include #include +#include #include +#include #include +#include #include #include #include @@ -85,23 +90,13 @@ RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const Strin RootMutationOrigin::Writer, RootMutationKind::Publish); } -/// A fence that never refuses, for the tests whose subject is not the fence. -const std::function ALWAYS_ADMITTED = [](uint64_t) {}; - -/// A deadline far enough out that only the test's own contention decides the outcome. The clock is -/// frozen (a constant `now`), which is what makes every non-exhaustion test independent of wall time. -CkptDeadline generousDeadline() -{ - return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; -} - /// Reads `life`'s `_ckpt` and returns its body, or a default-constructed one after failing the /// current test when the object is absent. Every assertion below goes through this rather than /// dereferencing the optional directly: a bare `->` on a disengaged optional ABORTS the whole test /// binary, so one regression would take every later suite's result with it instead of failing a test. -RefCkpt readCkptOrFail(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +RefCkpt readCkptOrFail(CasOperation & op, const Layout & layout, const NamespaceLifeId & life) { - const std::optional sample = readCkpt(backend, layout, life); + const std::optional sample = readCkpt(op, layout, life); if (!sample) { ADD_FAILURE() << "expected a _ckpt for namespace '" << life.ns.string() << "', found none"; @@ -110,16 +105,16 @@ RefCkpt readCkptOrFail(Backend & backend, const Layout & layout, const Namespace return sample->ckpt; } -/// Stage B (Task 4-C): the incarnation `store`'s production birth wiring minted for `ns`, learned back +/// Stage B: the incarnation `store`'s production birth wiring minted for `ns`, learned back /// from the catalog exactly as a real reader would (`NamespaceLifeId::fromCatalogEntry`) -- once a real /// `Pool`/`CasRefLedger` has opened the table, its ref-layer objects are no longer keyed at the /// Stage-A sentinel, so every test below that drives the REAL append lane must ask the catalog what /// incarnation it minted rather than assume the sentinel. Fails the current test (rather than /// dereferencing a disengaged optional) if the catalog carries no entry for `ns` -- e.g. called before /// the namespace's first append. -NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const RootNamespace & ns) +NamespaceLifeId liveLifeOrFail(CasOperation & op, const Layout & layout, const RootNamespace & ns) { - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); for (const CatalogEntry & entry : snap.catalog.entries) if (entry.ns.string() == ns.string()) return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); @@ -129,47 +124,53 @@ NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const R /// Replaces the whole body of one key, minting a new incarnation -- how a test installs a deliberately /// malformed or concurrently-advanced object. -void overwriteObject(Backend & backend, const String & key, const String & bytes) +void overwriteObject(CasOperation & op, const String & key, const String & bytes) { - const HeadResult h = backend.head(key); - ASSERT_TRUE(h.exists) << "overwriteObject expects " << key << " to exist"; - ASSERT_EQ(backend.putOverwrite(key, bytes, h.token).outcome, PutOutcome::Done); + const WriteResult result = op.readModifyWrite(key, + [&bytes](const std::optional & current) -> std::optional + { + EXPECT_TRUE(current.has_value()) << "overwriteObject expects the key to exist"; + return bytes; + }, + Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); } -/// Runs `on_get` right after every `get` of `watched_key` -- the deterministic way to act inside -/// another component's read-then-write window without a sleep or a second thread. The hook is a public -/// member rather than a constructor argument so it can be installed AFTER the backend exists (every -/// interesting hook writes through that same backend) and only once the test's setup writes are done. -class GetHookBackend : public CountingBackend +/// Per-key counts of the WRITE primitive. `CountingBackend` counts reads, heads and lists per key but +/// only totals for writes, and its legacy per-verb counters never see a caller that speaks the +/// primitives -- which every writer below does. +class WriteCountingBackend : public CountingBackend { public: - using CountingBackend::get; - - explicit GetHookBackend(String watched_key_) : watched_key(std::move(watched_key_)) {} - - /// Stage B (Task 4-C): a test that must watch a namespace's `_ckpt` key can no longer compute it - /// before the pool exists -- the real incarnation is minted only once the namespace's first open - /// resolves it, which requires the pool (and so this backend) to already be constructed. Lets a - /// test retarget the watch once it has learned the real key, strictly before arming `on_get`. - void setWatchedKey(String watched_key_) { watched_key = std::move(watched_key_); } - - std::function on_get; + uint64_t writes(const String & key) const + { + std::lock_guard lock(write_count_mutex); + const auto it = write_counts.find(key); + return it == write_counts.end() ? 0 : it->second; + } - std::optional get(const String & key, Range range) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - auto result = CountingBackend::get(key, range); - if (key == watched_key && on_get) - on_get(); - return result; + { + std::lock_guard lock(write_count_mutex); + ++write_counts[key]; + } + return CountingBackend::write(key, bytes, expected_value, access); } private: - String watched_key; + mutable std::mutex write_count_mutex; + std::map write_counts; }; -/// Records the exact `_ckpt` recovery protocol and injects an ambiguous CAS response. The fault is -/// armed only after fixture setup, so the journal contains solely the operation under test. -class AmbiguousCkptBackend : public CountingBackend +/// The two PRIMITIVES `publishCkpt` speaks, instrumented: the exact request sequence against one +/// watched key, and the two shapes an ambiguous response has -- the store applied the write and then +/// lost the answer, or it never applied it. Hooks are public members rather than constructor arguments +/// so they can be installed AFTER the backend exists (every interesting hook writes through that same +/// backend) and only once the test's setup writes are done. +class CkptProbeBackend : public WriteCountingBackend { public: enum class Fault : uint8_t @@ -180,72 +181,83 @@ class AmbiguousCkptBackend : public CountingBackend AlwaysThrowWithoutCommit, }; - using CountingBackend::casPut; - using CountingBackend::get; - String watched_key; Fault fault = Fault::None; + /// Written over this call's own committed attempt, so the resolve read finds a WINNER rather than + /// the bytes the attempt sent. String dominating_bytes; - bool fail_resolution_get = false; - std::function after_ambiguous_cas; - std::function before_resolution_get; - std::function after_resolution_get; + bool fail_reads_after_the_first = false; + std::function after_write; + std::function after_read; std::vector journal; + /// How many reads `fail_reads_after_the_first` actually made throw, so a test can assert the fault + /// really fired rather than infer it from the journal's shape alone. + size_t read_fault_hits = 0; + + /// A test that must watch a namespace's `_ckpt` key cannot compute it before the pool exists -- + /// the real incarnation is minted only once the namespace's first open resolves it. So the watch + /// is retargeted once the test has learned the real key, strictly before arming any hook. + void watch(String key) + { + watched_key = std::move(key); + watched_reads = 0; + journal.clear(); + read_fault_hits = 0; + } void arm(const String & key, Fault fault_) { - watched_key = key; + watch(key); fault = fault_; - watched_get_count = 0; - journal.clear(); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { if (key != watched_key) - return CountingBackend::get(key, range); - - journal.push_back("GET"); - ++watched_get_count; - if (watched_get_count >= 2 && before_resolution_get) - before_resolution_get(); - if (watched_get_count == 2 && fail_resolution_get) - throw Poco::TimeoutException("AmbiguousCkptBackend: exact-read response lost"); - auto result = CountingBackend::get(key, range); - if (watched_get_count >= 2 && after_resolution_get) - after_resolution_get(); + return WriteCountingBackend::read(key, access); + + journal.push_back("READ"); + ++watched_reads; + if (watched_reads >= 2 && fail_reads_after_the_first) + { + ++read_fault_hits; + throw Poco::TimeoutException("CkptProbeBackend: read response lost"); + } + auto result = WriteCountingBackend::read(key, access); + if (after_read) + after_read(); return result; } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (key != watched_key) - return CountingBackend::casPut(key, bytes, expected, meta); + return WriteCountingBackend::write(key, bytes, expected_value, access); - journal.push_back("CAS"); - if (fault == Fault::None) - return CountingBackend::casPut(key, bytes, expected, meta); + journal.push_back("WRITE"); const Fault this_fault = fault; if (fault != Fault::AlwaysThrowWithoutCommit) fault = Fault::None; + if (this_fault == Fault::None) + return WriteCountingBackend::write(key, bytes, expected_value, access); + if (this_fault == Fault::CommitThenThrow) { - const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); - if (result.outcome == CasOutcome::Committed && !dominating_bytes.empty()) - { - const HeadResult head_result = CountingBackend::head(key); - EXPECT_EQ(CountingBackend::putOverwrite(key, dominating_bytes, head_result.token).outcome, - PutOutcome::Done); - } + const auto committed = WriteCountingBackend::write(key, bytes, expected_value, access); + /// The winner's replacement is not journalled: it is not an attempt of the call under test. + if (committed.has_value() && !dominating_bytes.empty()) + EXPECT_TRUE(WriteCountingBackend::write(key, dominating_bytes, + std::optional{*committed}, access).has_value()); } - if (after_ambiguous_cas) - after_ambiguous_cas(); - throw Poco::TimeoutException("AmbiguousCkptBackend: CAS response lost"); + if (after_write) + after_write(); + throw Poco::TimeoutException("CkptProbeBackend: write response lost"); } private: - size_t watched_get_count = 0; + size_t watched_reads = 0; }; } @@ -275,14 +287,29 @@ TEST(CASRefCheckpoint, CommittedThroughHasCanonicalExactWireEncoding) .committed_through = RefTxnId{9, 11}, .checkpoint_snapshot_id = RefTxnId{9, 10}, .last_epoch_seal = RefTxnId{8, 12}}; - const String expected = R"({"type":"cas_ref_ckpt","v":10} -{"le":"7","cte":"9","cts":"11","cse":"9","css":"10","lse":"8","lss":"12"} + const String expected = R"({"type":"cas_ref_ckpt","v":1} +{"life_epoch":"7","committed_epoch":"9","committed_seq":"11","snapshot_epoch":"9","snapshot_seq":"10","seal_epoch":"8","seal_seq":"12"} )"; EXPECT_EQ(encodeRefCkpt(ckpt), expected); EXPECT_EQ(decodeRefCkpt(expected), ckpt); } +CAS_BATTERY_COVERS(RefCkpt); + +TEST(CASFormatBattery, RefCkpt) +{ + RefCkpt ckpt{.life_epoch = std::optional{7}, + .committed_through = RefTxnId{9, 11}, + .checkpoint_snapshot_id = RefTxnId{9, 10}, + .last_epoch_seal = RefTxnId{8, 12}}; + runFormatBattery({FormatId::RefCkpt, + [&] { return sealObject(FormatId::RefCkpt, encodeRefCkpt(ckpt)); }, + [](std::string_view s) { decodeRefCkpt(std::string(openObject(FormatId::RefCkpt, s))); }, + currentFormatHeader("cas_ref_ckpt") + + "{\"life_epoch\":\"7\",\"committed_epoch\":\"9\",\"committed_seq\":\"11\",\"snapshot_epoch\":\"9\",\"snapshot_seq\":\"10\",\"seal_epoch\":\"8\",\"seal_seq\":\"12\"}\n"}); +} + /// `last_epoch_seal` is chain evidence, not an arbitrary lower bound. It either names the frontier /// itself when that frontier is the terminal seal, or closes the immediately preceding numeric epoch. /// Accepting a gap or a later same-epoch frontier would manufacture a boundary that INV-2 never proved. @@ -307,9 +334,9 @@ TEST(CASRefCheckpoint, CodecRejectsIncoherentCommittedFrontierAndSealEpochs) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefCkpt(unsealed_non_genesis); }); String malformed = encodeRefCkpt(valid); - const size_t cte = malformed.find(R"("cte":"8")"); - ASSERT_NE(cte, String::npos); - malformed.replace(cte, String{R"("cte":"8")"}.size(), R"("cte":"10")"); + const size_t committed_epoch = malformed.find(R"("committed_epoch":"8")"); + ASSERT_NE(committed_epoch, String::npos); + malformed.replace(committed_epoch, String{R"("committed_epoch":"8")"}.size(), R"("committed_epoch":"10")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(malformed); }); } @@ -331,13 +358,30 @@ TEST(CASRefCheckpoint, RejectsAnUnknownKey) expectThrowsCode(DB::ErrorCodes::UNKNOWN_FORMAT_VERSION, [&] { decodeRefCkpt(with_critical); }); } +/// Replacing the abbreviated key is a format cut, not an alias. Treating it as an optional partial +/// pair would make an old writer's checkpoint appear to have no committed frontier. +TEST(CASRefCheckpoint, RejectsOldCommittedEpochKeyRatherThanAliasingIt) +{ + /// The values are chosen so ALIASING would be harmless: the spliced `"cte":"9"` re-assigns the + /// epoch the object already carries, leaving a valid checkpoint. A reader that honoured the old + /// spelling would therefore DECODE, and this test fails; only the strict unknown-key rejection + /// makes it throw. Values under which aliasing corrupts the object would let the invariant + /// checker throw the same code and hide the alias. + String with_old_key = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{9}, + .committed_through = RefTxnId{9, 1}, + .checkpoint_snapshot_id = RefTxnId{9, 1}, + .last_epoch_seal = std::nullopt}); + with_old_key.replace(with_old_key.rfind('}'), 1, R"(,"cte":"9"})"); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(with_old_key); }); +} + /// A duplicate key has no single meaning, so it can never be resolved by a reader's preference. TEST(CASRefCheckpoint, RejectsADuplicateKey) { const String good = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); String duplicated = good; - duplicated.replace(duplicated.rfind('}'), 1, R"(,"le":"9"})"); + duplicated.replace(duplicated.rfind('}'), 1, R"(,"life_epoch":"9"})"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(duplicated); }); } @@ -361,13 +405,13 @@ TEST(CASRefCheckpoint, RejectsTruncation) const String empty_body = good.substr(0, good.find('\n') + 1) + "{}\n"; EXPECT_EQ(decodeRefCkpt(empty_body), RefCkpt{}); - const String half_pair = good.substr(0, good.find('\n') + 1) + R"({"le":"7","cse":"1"})" + "\n"; + const String half_pair = good.substr(0, good.find('\n') + 1) + R"({"life_epoch":"7","snapshot_epoch":"1"})" + "\n"; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(half_pair); }); - const String other_half = good.substr(0, good.find('\n') + 1) + R"({"le":"7","lss":"2"})" + "\n"; + const String other_half = good.substr(0, good.find('\n') + 1) + R"({"life_epoch":"7","seal_seq":"2"})" + "\n"; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(other_half); }); - const String frontier_half = good.substr(0, good.find('\n') + 1) + R"({"le":"7","cte":"1"})" + "\n"; + const String frontier_half = good.substr(0, good.find('\n') + 1) + R"({"life_epoch":"7","committed_epoch":"1"})" + "\n"; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(frontier_half); }); } @@ -398,9 +442,9 @@ TEST(CASRefCheckpoint, RejectsInvalidFieldsOnEncodeAndOnDecode) const String header = encodeRefCkpt(RefCkpt{.life_epoch = std::optional{7}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); const String prefix = header.substr(0, header.find('\n') + 1); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(prefix + R"({"le":"0"})" + "\n"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefCkpt(prefix + R"({"life_epoch":"0"})" + "\n"); }); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { decodeRefCkpt(prefix + R"({"le":"7","cse":"1","css":"0"})" + "\n"); }); + [&] { decodeRefCkpt(prefix + R"({"life_epoch":"7","snapshot_epoch":"1","snapshot_seq":"0"})" + "\n"); }); } /// The registry row is part of the contract: Control/Strict decides how the decoder treats unknown @@ -529,14 +573,15 @@ TEST(CASRefCheckpoint, MergeTakesThePerFieldSemanticMaximum) TEST(CASRefCheckpoint, CreatesTheObjectWhenItIsAbsent) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_create"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); const RefCkpt birth{.life_epoch = std::optional{5}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const auto sample = readCkpt(*backend, layout, life); + EXPECT_EQ(publishCkpt(op, layout, life, birth), CkptPublishOutcome::Published); + const auto sample = readCkpt(op, layout, life); ASSERT_TRUE(sample.has_value()); EXPECT_EQ(sample->ckpt, birth); } @@ -549,23 +594,23 @@ TEST(CASRefCheckpoint, CreatesTheObjectWhenItIsAbsent) TEST(CASRefCheckpoint, EachWriterCreatesWithOnlyWhatItKnowsAndTheOtherFieldsMergeInLater) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_partial_create"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); const RefCkpt publisher{.life_epoch = std::nullopt, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(*backend, layout, life, publisher, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const auto created = readCkpt(*backend, layout, life); + ASSERT_EQ(publishCkpt(op, layout, life, publisher), CkptPublishOutcome::Published); + const auto created = readCkpt(op, layout, life); ASSERT_TRUE(created.has_value()); EXPECT_EQ(created->ckpt.checkpoint_snapshot_id, ID_1_1); EXPECT_FALSE(created->ckpt.life_epoch.has_value()) << "the publisher must not invent a genesis epoch"; const RefCkpt birth{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const auto completed = readCkpt(*backend, layout, life); + ASSERT_EQ(publishCkpt(op, layout, life, birth), CkptPublishOutcome::Published); + const auto completed = readCkpt(op, layout, life); ASSERT_TRUE(completed.has_value()); EXPECT_EQ(completed->ckpt.life_epoch, 1u); EXPECT_EQ(completed->ckpt.checkpoint_snapshot_id, ID_1_1) << "and must not lose the checkpoint on the way in"; @@ -573,8 +618,8 @@ TEST(CASRefCheckpoint, EachWriterCreatesWithOnlyWhatItKnowsAndTheOtherFieldsMerg /// The conflict path is the whole reason the algorithm re-READS instead of retrying its bytes: the /// winner's field must survive the loser's retry. Here a concurrent writer advances the seal between -/// our read and our CAS; our retry must merge onto the new body, not overwrite it. -TEST(CASRefCheckpoint, TokenConflictRereadsAndMergesOntoTheWinner) +/// our read and our write; our retry must merge onto the new body, not overwrite it. +TEST(CASRefCheckpoint, AConflictRereadsAndMergesOntoTheWinner) { const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_conflict"}; @@ -582,96 +627,112 @@ TEST(CASRefCheckpoint, TokenConflictRereadsAndMergesOntoTheWinner) const String key = layout.refCkptKey(life); const RefCkpt base{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - auto backend = std::make_shared(key); - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - - /// The concurrent sealer lands exactly ONCE, immediately after our first read -- so our first CAS - /// carries a token that is no longer current, and our retry has to merge onto its body. + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + CasOperation sealer_op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(sealer_op.create(key, encodeRefCkpt(base), Retry::standard()))); + backend->watch(key); + + /// The concurrent sealer lands exactly ONCE, immediately after our first read -- so our first + /// write carries a precondition that is no longer current, and our retry has to merge onto its + /// body. It writes on its OWN operation: the interference is a different actor, not a reentrant + /// call of the one under test. bool interfered = false; - backend->on_get = [&] + backend->after_read = [&] { if (interfered) return; interfered = true; const RefCkpt sealer{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = ID_2_1}; - const HeadResult h = backend->head(key); - ASSERT_EQ(backend->putOverwrite(key, encodeRefCkpt(mergeCkpt(base, sealer)), h.token).outcome, - PutOutcome::Done); + overwriteObject(sealer_op, key, encodeRefCkpt(mergeCkpt(base, sealer))); }; const RefCkpt publisher{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(*backend, layout, life, publisher, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); + EXPECT_EQ(publishCkpt(op, layout, life, publisher), CkptPublishOutcome::Published); - const auto sample = readCkpt(*backend, layout, life); + backend->after_read = nullptr; + const auto sample = readCkpt(op, layout, life); ASSERT_TRUE(sample.has_value()); EXPECT_EQ(sample->ckpt.checkpoint_snapshot_id, ID_1_2) << "our own contribution must land"; EXPECT_EQ(sample->ckpt.last_epoch_seal, ID_2_1) << "the concurrent writer's seal must survive our retry -- a retry that reused the body read " "before the conflict would silently drop it (TLC `_sab_sealclobbersbase`)"; EXPECT_EQ(sample->ckpt.life_epoch, 1u); - EXPECT_GE(backend->casPutCount(key), 2u) << "the first CAS must have been rejected, not skipped"; + EXPECT_GE(std::count(backend->journal.begin(), backend->journal.end(), String{"WRITE"}), 2) + << "the first write must have been refused, not skipped"; } /// A contribution that adds nothing issues NO write. This is a correctness property, not a saving: -/// both writers publish on every snapshot and every seal, and a no-op write would mint a fresh token -/// each time, turning every other writer's in-flight CAS into a conflict for identical bytes. +/// both writers publish on every snapshot and every seal, and a no-op write would mint a fresh +/// incarnation each time, turning every other writer's in-flight write into a conflict for identical +/// bytes. TEST(CASRefCheckpoint, AnIdenticalMergedBodyIssuesNoWrite) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_noop"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); const String key = layout.refCkptKey(life); const RefCkpt full{.life_epoch = std::optional{1}, .committed_through = ID_2_1, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = ID_2_1}; - ASSERT_EQ(publishCkpt(*backend, layout, life, full, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const uint64_t writes_after_create = backend->casPutCount(key); - const Token token_after_create = backend->head(key).token; + ASSERT_EQ(publishCkpt(op, layout, life, full), CkptPublishOutcome::Published); + const uint64_t writes_after_create = backend->writes(key); + const auto meta_after_create = op.head(key, Retry::standard()); + ASSERT_TRUE(meta_after_create.has_value()); /// The same contribution again, and a strictly OLDER one: neither adds anything. - EXPECT_EQ(publishCkpt(*backend, layout, life, full, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::IdenticalSkip); + EXPECT_EQ(publishCkpt(op, layout, life, full), CkptPublishOutcome::IdenticalSkip); const RefCkpt older{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(*backend, layout, life, older, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::IdenticalSkip); + EXPECT_EQ(publishCkpt(op, layout, life, older), CkptPublishOutcome::IdenticalSkip); - EXPECT_EQ(backend->casPutCount(key), writes_after_create) << "a skip must issue no CAS at all"; - EXPECT_EQ(backend->head(key).token, token_after_create) << "and must not mint a new incarnation"; + EXPECT_EQ(backend->writes(key), writes_after_create) << "a skip must issue no write at all"; + const auto meta_after_skips = op.head(key, Retry::standard()); + ASSERT_TRUE(meta_after_skips.has_value()); + EXPECT_EQ(meta_after_skips->etag, meta_after_create->etag) + << "and must not mint a new incarnation"; } -/// The fence is re-checked AFTER the read and BEFORE the write, on every attempt. A generation that -/// moved means this writer's lease incarnation is gone, so its merged body is stale even if the fence -/// is live again under a fresh incarnation. -TEST(CASRefCheckpoint, AFenceBumpBetweenTheReadAndTheCasWritesNothing) +/// Admission is re-checked AFTER the read and BEFORE the write. A writer whose admission was lost in +/// that window has a stale merged body, so its write must never be sent. +TEST(CASRefCheckpoint, AnAdmissionLossBetweenTheReadAndTheWriteWritesNothing) { - auto backend = std::make_shared(); const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_fenced"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); const String key = layout.refCkptKey(life); + + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + bool admitted = true; + CasOperation op = requests.admit([&admitted] { return admitted; }); + CasOperation reader = requests.admit(); + const RefCkpt base{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(*backend, layout, life, base, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const Token token_before = backend->head(key).token; - const uint64_t writes_before = backend->casPutCount(key); + ASSERT_EQ(publishCkpt(op, layout, life, base), CkptPublishOutcome::Published); + const auto meta_before = reader.head(key, Retry::standard()); + ASSERT_TRUE(meta_before.has_value()); - /// The callback the pool wires from `CasMountRuntime::checkFenceOrThrow`: it throws when the - /// generation moved since admission. Mirrors the real site's class (the transient, upstream-retryable - /// one) so the stub cannot drift into testing a shape production never produces. - const auto moved_fence = [](uint64_t admitted) - { - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, - "fence generation moved since admission ({})", admitted); - }; + /// Armed only now, so the loss lands inside the publish's own read-then-write window rather than + /// before it began. + backend->watch(key); + backend->after_read = [&admitted] { admitted = false; }; + const uint64_t writes_before = backend->writes(key); const RefCkpt advance{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(*backend, layout, life, advance, 1, moved_fence, generousDeadline()), - CkptPublishOutcome::FencedOut); - EXPECT_EQ(backend->casPutCount(key), writes_before) << "the check precedes the CAS, so nothing is sent"; - EXPECT_EQ(backend->head(key).token, token_before); - EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); + EXPECT_EQ(publishCkpt(op, layout, life, advance), CkptPublishOutcome::FencedOut); + backend->after_read = nullptr; + + EXPECT_EQ(backend->writes(key), writes_before) << "the check precedes the write, so nothing is sent"; + const auto meta_after = reader.head(key, Retry::standard()); + ASSERT_TRUE(meta_after.has_value()); + EXPECT_EQ(meta_after->etag, meta_before->etag); + EXPECT_EQ(readCkptOrFail(reader, layout, life), base); } /// Persistent contention fails CLOSED and says so. There is no partial state to clean up -- every @@ -685,32 +746,48 @@ TEST(CASRefCheckpoint, AnExhaustedDeadlineUnderPersistentConflictThrowsRetryLate const String key = layout.refCkptKey(life); const RefCkpt base{.life_epoch = std::optional{1}, .committed_through = ID_1_1, .checkpoint_snapshot_id = ID_1_1, .last_epoch_seal = std::nullopt}; - auto backend = std::make_shared(key); - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - - /// Every read is followed by a rewrite of the SAME body under a fresh incarnation, so the token this - /// call holds is always stale and every CAS it issues conflicts. The clock advances one step per - /// read, so the DEADLINE is what ends the loop -- deterministically, with no sleeping and well - /// before the live-lock brake. - uint64_t now = 0; - backend->on_get = [&] + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + CasOperation setup = requests.admit(); + ASSERT_TRUE(std::holds_alternative(setup.create(key, encodeRefCkpt(base), Retry::standard()))); + backend->watch(key); + + /// Every read is followed by a rewrite of the SAME body under a fresh incarnation, so the + /// precondition this call holds is always stale and every write it issues is refused. Only the + /// policy's deadline can end the loop, and the injected clock reaches it without sleeping. + /// + /// `overwriteObject` itself reads `key` (its own `readModifyWrite`'s precondition read), which + /// would re-enter this very hook -- unlike the concurrent-actor fixtures elsewhere in this file, + /// this rewrite is not a one-shot: it must keep firing on every OUTER read, so a plain one-shot + /// latch would silently stop the persistent conflict after the first attempt. Guard only the + /// reentrant call instead. + bool rewriting = false; + backend->after_read = [&] { - ++now; - const HeadResult h = backend->head(key); - if (h.exists) - backend->putOverwrite(key, encodeRefCkpt(base), h.token); + if (rewriting) + return; + rewriting = true; + overwriteObject(setup, key, encodeRefCkpt(base)); + rewriting = false; }; const RefCkpt advance{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, - [&] { publishCkpt(*backend, layout, life, advance, 1, ALWAYS_ADMITTED, CkptDeadline{[&] { return now; }, 5}); }); - EXPECT_EQ(readCkptOrFail(*backend, layout, life), base) << "no partial state: every attempt either " - "committed the complete merged body or wrote nothing"; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishCkpt(op, layout, life, advance); }); + backend->after_read = nullptr; + EXPECT_FALSE(clock.sleeps.empty()) << "the loop must back off between attempts, not spin"; + EXPECT_EQ(readCkptOrFail(setup, layout, life), base) << "no partial state: every attempt either " + "committed the complete merged body or wrote nothing"; } -TEST(CASRefCheckpoint, AmbiguousCommittedCasIsResolvedByOneExactReadWithoutBlindRetry) +TEST(CASRefCheckpoint, AnAmbiguousCommittedWriteIsResolvedByOneExactReadWithoutBlindRetry) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_committed"}); const String key = layout.refCkptKey(life); @@ -718,18 +795,23 @@ TEST(CASRefCheckpoint, AmbiguousCommittedCasIsResolvedByOneExactReadWithoutBlind .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; const RefCkpt contribution{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + ASSERT_TRUE(std::holds_alternative(op.create(key, encodeRefCkpt(base), Retry::standard()))); + backend->arm(key, CkptProbeBackend::Fault::CommitThenThrow); - EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); - EXPECT_EQ(readCkptOrFail(*backend, layout, life).committed_through, ID_1_2); + EXPECT_EQ(publishCkpt(op, layout, life, contribution), CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"READ", "WRITE", "READ"})); + backend->watched_key.clear(); + EXPECT_EQ(readCkptOrFail(op, layout, life).committed_through, ID_1_2); } -TEST(CASRefCheckpoint, AmbiguousUncommittedCasRetriesAgainstTheExactReadToken) +TEST(CASRefCheckpoint, AnAmbiguousUncommittedWriteRetriesAgainstTheExactRead) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); const Layout layout{"p"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_retry"}); const String key = layout.refCkptKey(life); @@ -737,18 +819,26 @@ TEST(CASRefCheckpoint, AmbiguousUncommittedCasRetriesAgainstTheExactReadToken) .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; const RefCkpt contribution{.life_epoch = std::nullopt, .committed_through = ID_1_2, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - backend->arm(key, AmbiguousCkptBackend::Fault::ThrowWithoutCommit); + ASSERT_TRUE(std::holds_alternative(op.create(key, encodeRefCkpt(base), Retry::standard()))); + backend->arm(key, CkptProbeBackend::Fault::ThrowWithoutCommit); - EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET", "CAS"})); - EXPECT_EQ(readCkptOrFail(*backend, layout, life).committed_through, ID_1_2); + EXPECT_EQ(publishCkpt(op, layout, life, contribution), CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"READ", "WRITE", "READ", "WRITE"})); + backend->watched_key.clear(); + EXPECT_EQ(readCkptOrFail(op, layout, life).committed_through, ID_1_2); } -TEST(CASRefCheckpoint, AmbiguousCasAcceptsAValidDominatingDurableFrontier) +/// The durable body a winner left behind already dominates this contribution, so nothing more is owed. +/// The verdict is `Published` rather than `IdenticalSkip` because an attempt of THIS call was sent: +/// `IdenticalSkip` promises no write was issued, and that promise has to stay true. +TEST(CASRefCheckpoint, AnAmbiguousWriteAcceptsAValidDominatingDurableFrontier) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); const Layout layout{"p"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_dominating"}); const String key = layout.refCkptKey(life); @@ -758,122 +848,166 @@ TEST(CASRefCheckpoint, AmbiguousCasAcceptsAValidDominatingDurableFrontier) .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; const RefCkpt dominating{.life_epoch = 1, .committed_through = ID_2_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = ID_2_1}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(op.create(key, encodeRefCkpt(base), Retry::standard()))); backend->dominating_bytes = encodeRefCkpt(dominating); - backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); + backend->arm(key, CkptProbeBackend::Fault::CommitThenThrow); - EXPECT_EQ(publishCkpt(*backend, layout, life, contribution, 7, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); - EXPECT_EQ(readCkptOrFail(*backend, layout, life), dominating); + EXPECT_EQ(publishCkpt(op, layout, life, contribution), CkptPublishOutcome::Published); + EXPECT_EQ(backend->journal, (std::vector{"READ", "WRITE", "READ"})); + backend->watched_key.clear(); + EXPECT_EQ(readCkptOrFail(op, layout, life), dominating); } -TEST(CASRefCheckpoint, FailedExactReadAfterAmbiguousCasFailsClosedWithoutAnotherCas) +/// A resolve read that never answers leaves the attempt unproven, and the call must neither report it +/// committed nor send a second attempt on top of it. The engine reissues -- that is its contract -- but +/// every reissue is preceded by its own exact read. +TEST(CASRefCheckpoint, AFailedResolveReadNeverReportsACommitAndNeverSkipsTheRead) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + DB::Cas::tests::FakeClock clock; + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + requests.setNowFnForTest(clock.nowFn()); + requests.setSleepFnForTest(clock.sleepFn()); + CasOperation op = requests.admit(); + CasOperation reader = requests.admit(); const Layout layout{"p"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguous_read_failed"}); const String key = layout.refCkptKey(life); const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - backend->fail_resolution_get = true; - backend->arm(key, AmbiguousCkptBackend::Fault::ThrowWithoutCommit); + ASSERT_TRUE(std::holds_alternative(op.create(key, encodeRefCkpt(base), Retry::standard()))); + backend->arm(key, CkptProbeBackend::Fault::AlwaysThrowWithoutCommit); + backend->fail_reads_after_the_first = true; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { - publishCkpt(*backend, layout, life, RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, 7, - ALWAYS_ADMITTED, generousDeadline()); + publishCkpt(op, layout, life, RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); }); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); - EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); + /// `{READ, WRITE}` alone would satisfy "no adjacent writes" and "at least one write" without the + /// resolving read ever having been attempted, let alone failed. Pin that at least one read follows + /// the write, that the fault double actually fired on every one of them (a resolving read is itself + /// retried against the policy deadline, so several follow, not just one), and that no reissue was + /// sent while every resolution read fails. + ASSERT_GE(backend->journal.size(), 3u); + EXPECT_EQ(backend->journal.front(), "READ") << "the baseline read of the current state"; + EXPECT_EQ(backend->journal[1], "WRITE") << "the attempt that never committed"; + const size_t resolve_reads = backend->journal.size() - 2; + EXPECT_TRUE(std::all_of(backend->journal.begin() + 2, backend->journal.end(), + [](const String & verb) { return verb == "READ"; })) + << "no reissue can be sent while every resolution read fails, so nothing after the write is a WRITE"; + EXPECT_EQ(backend->read_fault_hits, resolve_reads) + << "every read after the baseline failed -- the fault double actually fired on all of them, not just " + << "the first"; + EXPECT_EQ(std::count(backend->journal.begin(), backend->journal.end(), String{"WRITE"}), 1) + << "no reissue can be sent while every resolution read fails"; + backend->watched_key.clear(); + backend->fail_reads_after_the_first = false; + EXPECT_EQ(readCkptOrFail(reader, layout, life), base); } -TEST(CASRefCheckpoint, FenceMovementAroundAmbiguityResolutionMakesTheExactReadInert) +/// Admission lost while the ambiguous write was in flight: the exact read that would settle it is +/// refused before it starts, so the call reports `FencedOut` and claims nothing about the object. +TEST(CASRefCheckpoint, AdmissionLostWithTheAmbiguousWritePreventsItsResolveRead) { - for (const bool move_before_read : {true, false}) - { - auto backend = std::make_shared(); - const Layout layout{"p"}; - const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife( - RootNamespace{move_before_read ? "srv1/ckpt_fence_before_resolution" : "srv1/ckpt_fence_after_resolution"}); - const String key = layout.refCkptKey(life); - const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - bool admitted = true; - const auto move_fence = [&] { admitted = false; }; - if (move_before_read) - backend->before_resolution_get = move_fence; - else - backend->after_resolution_get = move_fence; - backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); - - const auto check_admission = [&](uint64_t) - { - if (!admitted) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence moved"); - }; - EXPECT_EQ(publishCkpt(*backend, layout, life, - RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, 7, - check_admission, generousDeadline()), CkptPublishOutcome::FencedOut); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS", "GET"})); - } -} - -TEST(CASRefCheckpoint, AdmissionLostWithTheAmbiguousCasPreventsItsResolutionGet) -{ - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + bool admitted = true; + CasOperation op = requests.admit([&admitted] { return admitted; }); + CasOperation reader = requests.admit(); const Layout layout{"p"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife( RootNamespace{"srv1/ckpt_admission_lost_before_resolution"}); const String key = layout.refCkptKey(life); const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(reader.create(key, encodeRefCkpt(base), Retry::standard()))); - bool admitted = true; - backend->after_ambiguous_cas = [&] { admitted = false; }; - backend->arm(key, AmbiguousCkptBackend::Fault::CommitThenThrow); - const auto admit_request = [&] - { - if (!admitted) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "admission withdrawn"); - }; + backend->after_write = [&admitted] { admitted = false; }; + backend->arm(key, CkptProbeBackend::Fault::CommitThenThrow); - EXPECT_EQ(publishCkpt(*backend, layout, life, + EXPECT_EQ(publishCkpt(op, layout, life, RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, - 7, ALWAYS_ADMITTED, generousDeadline(), admit_request), CkptPublishOutcome::FencedOut); - EXPECT_EQ(backend->journal, (std::vector{"GET", "CAS"})) - << "publishCkpt started its ambiguity-resolution GET after admission was withdrawn"; + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}), + CkptPublishOutcome::FencedOut); + EXPECT_EQ(backend->journal, (std::vector{"READ", "WRITE"})) + << "the resolve read started after admission was withdrawn"; } -TEST(CASRefCheckpoint, ContinuedAmbiguityStopsAtTheDeadlineAndNeverIssuesConsecutiveCasAttempts) +/// Admission lost AFTER the resolve read proved the attempt durable: the object may well carry this +/// contribution, but a call whose admission is gone must never claim it. +TEST(CASRefCheckpoint, AdmissionLostAfterTheResolveReadStillRefusesToClaimTheCommit) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + bool admitted = true; + CasOperation op = requests.admit([&admitted] { return admitted; }); + CasOperation reader = requests.admit(); const Layout layout{"p"}; - const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_ambiguity_deadline"}); + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(RootNamespace{"srv1/ckpt_fence_after_resolution"}); const String key = layout.refCkptKey(life); const RefCkpt base{.life_epoch = 1, .committed_through = ID_1_1, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(backend->casPut(key, encodeRefCkpt(base), std::nullopt).outcome, CasOutcome::Committed); - uint64_t now = 0; - backend->after_resolution_get = [&] { ++now; }; - backend->arm(key, AmbiguousCkptBackend::Fault::AlwaysThrowWithoutCommit); + ASSERT_TRUE(std::holds_alternative(reader.create(key, encodeRefCkpt(base), Retry::standard()))); - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + backend->arm(key, CkptProbeBackend::Fault::CommitThenThrow); + /// The SECOND read is the resolve read; withdrawing after the first would refuse the write instead + /// and never reach the point this test is about. + size_t reads = 0; + backend->after_read = [&] { - publishCkpt(*backend, layout, life, - RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, - .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}, - 7, ALWAYS_ADMITTED, CkptDeadline{[&] { return now; }, 3}); - }); - EXPECT_EQ(backend->journal, - (std::vector{"GET", "CAS", "GET", "CAS", "GET", "CAS", "GET"})); - EXPECT_EQ(readCkptOrFail(*backend, layout, life), base); + if (++reads == 2) + admitted = false; + }; + + EXPECT_EQ(publishCkpt(op, layout, life, + RefCkpt{.life_epoch = std::nullopt, .committed_through = ID_1_2, + .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}), + CkptPublishOutcome::FencedOut); + EXPECT_EQ(backend->journal, (std::vector{"READ", "WRITE", "READ"})); +} + +/// Both verdicts `publishCkpt` reaches WITHOUT writing consult admission before they speak, and both +/// answer `FencedOut` when it is gone. A writer the fence is about to refuse landed nothing anywhere: +/// telling it `IdenticalSkip` would claim its contribution is already durable, and telling it +/// `CORRUPTED_DATA` would turn a transient control signal into a permanent verdict on the namespace. +TEST(CASRefCheckpoint, DeclineTimeVerdictsReadAdmitted) +{ + const Layout layout{"p"}; + for (const bool decreasing : {false, true}) + { + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + bool admitted = true; + CasOperation op = requests.admit([&admitted] { return admitted; }); + CasOperation reader = requests.admit(); + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife( + RootNamespace{decreasing ? "srv1/ckpt_decline_decrease" : "srv1/ckpt_decline_identical"}); + const String key = layout.refCkptKey(life); + /// A genesis epoch of 2 with the frontier in that same epoch: `checkRefCkptInvariants` refuses + /// a `committed_through` preceding `life_epoch`, so the durable body a decrease is measured + /// against has to be one the format would actually store. + const RefCkpt durable{.life_epoch = std::optional{2}, .committed_through = ID_2_1, + .checkpoint_snapshot_id = ID_2_1, .last_epoch_seal = std::nullopt}; + ASSERT_EQ(publishCkpt(reader, layout, life, durable), CkptPublishOutcome::Published); + const uint64_t writes_before = backend->writes(key); + + /// Admission survives the read and is gone by the time the verdict is reached -- the exact + /// window in which the answer must be `FencedOut` and nothing else. + backend->watch(key); + backend->after_read = [&admitted] { admitted = false; }; + const RefCkpt contribution = decreasing + ? RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = std::nullopt} + : durable; + EXPECT_EQ(publishCkpt(op, layout, life, contribution), CkptPublishOutcome::FencedOut) + << (decreasing ? "a superseded epoch from an unadmitted writer is not corruption" + : "an identical body from an unadmitted writer is not a skip"); + backend->after_read = nullptr; + + EXPECT_EQ(backend->writes(key), writes_before) << "neither verdict may write"; + EXPECT_EQ(readCkptOrFail(reader, layout, life), durable); + } } /// A `_ckpt` that does not decode is NEVER overwritten. It is the only record of recovery's base and @@ -882,33 +1016,50 @@ TEST(CASRefCheckpoint, ContinuedAmbiguityStopsAtTheDeadlineAndNeverIssuesConsecu TEST(CASRefCheckpoint, ACorruptCheckpointIsNeverOverwritten) { auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const Layout layout{"p"}; const RootNamespace ns{"srv1/ckpt_corrupt"}; const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); const String key = layout.refCkptKey(life); - ASSERT_EQ(publishCkpt(*backend, layout, life, - RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}, - 1, ALWAYS_ADMITTED, generousDeadline()), CkptPublishOutcome::Published); + ASSERT_EQ(publishCkpt(op, layout, life, + RefCkpt{.life_epoch = std::optional{1}, .committed_through = ID_1_2, .checkpoint_snapshot_id = ID_1_2, .last_epoch_seal = std::nullopt}), + CkptPublishOutcome::Published); const String garbage = "not a cas object\n"; - overwriteObject(*backend, key, garbage); + overwriteObject(op, key, garbage); const RefCkpt birth{.life_epoch = std::optional{5}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, - [&] { publishCkpt(*backend, layout, life, birth, 1, ALWAYS_ADMITTED, generousDeadline()); }); - EXPECT_EQ(backend->get(key)->bytes, garbage) << "corruption must be surfaced, never laundered into a " - "well-formed object"; + [&] { publishCkpt(op, layout, life, birth); }); + const auto still_there = op.read(key, Retry::standard()); + ASSERT_TRUE(still_there.has_value()); + EXPECT_EQ(still_there->bytes, garbage) << "corruption must be surfaced, never laundered into a " + "well-formed object"; } /// --------------------------------------------------------------------------------------------- /// The reader-side rules Task 6 and the cleanup call sites consume /// --------------------------------------------------------------------------------------------- -/// INV-4's three-way revalidation of a base that turned out to be missing. -TEST(CASRefCheckpoint, AMissingSampledBaseRestartsOnAnAdvancedTokenAndIsCorruptionOnAnUnchangedOne) +/// INV-4's three-way revalidation of a base that turned out to be missing. The two incarnations come +/// from real reads of the same key across a rewrite, because an incarnation exists only as something a +/// request observed. +TEST(CASRefCheckpoint, AMissingSampledBaseRestartsOnAnAdvancedIncarnationAndIsCorruptionOnAnUnchangedOne) { - const Token sampled{"t1", TokenType::Emulated}; - const Token advanced{"t2", TokenType::Emulated}; + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); + const String key = "p/ckpt_incarnations"; + ASSERT_TRUE(std::holds_alternative(op.create(key, "first", Retry::standard()))); + const auto first = op.read(key, Retry::standard()); + ASSERT_TRUE(first.has_value()); + overwriteObject(op, key, "second"); + const auto second = op.read(key, Retry::standard()); + ASSERT_TRUE(second.has_value()); + const Etag sampled = first->etag; + const Etag advanced = second->etag; + ASSERT_FALSE(sampled == advanced) << "the rewrite must mint a different incarnation"; EXPECT_EQ(classifyMissingSampledBase(sampled, advanced), MissingBaseVerdict::RestartRecovery) << "cleanup legitimately moved the checkpoint while we read; restart from the newer base"; @@ -941,18 +1092,20 @@ TEST(CASRefCheckpoint, NamespaceBirthCreatesTheCheckpointCarryingItsLifeEpoch) { auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"srv1/ckpt_birth"}; - /// Stage B (Task 4-C): the catalog carries no entry for `ns` before its first open, and the + /// Stage B: the catalog carries no entry for `ns` before its first open, and the /// namespace's real incarnation does not exist to name a key with yet -- the pre-birth analog of /// "nothing exists" is "nothing is even NAMED", checked at the catalog rather than at a key this /// test cannot yet compute. - EXPECT_TRUE(CasRefCatalog::read(*backend, store->layout()).catalog.entries.empty()) + EXPECT_TRUE(CasRefCatalog::read(op, store->layout()).catalog.entries.empty()) << "nothing exists before the birth"; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); - const auto sample = readCkpt(*backend, store->layout(), life); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); + const auto sample = readCkpt(op, store->layout(), life); ASSERT_TRUE(sample.has_value()) << "spec §3 creates the _ckpt before the namespace becomes Live"; EXPECT_EQ(sample->ckpt.life_epoch, store->writerEpoch()); EXPECT_FALSE(sample->ckpt.checkpoint_snapshot_id.has_value()) << "a newborn namespace has no base yet"; @@ -964,25 +1117,27 @@ TEST(CASRefCheckpoint, ACommittedSnapshotPublishAdvancesTheCheckpoint) { auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const uint64_t epoch = store->writerEpoch(); const RootNamespace ns{"srv1/ckpt_publish"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); - ASSERT_FALSE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); + ASSERT_FALSE(readCkptOrFail(op, store->layout(), life).checkpoint_snapshot_id.has_value()); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); const auto published = store->newestPublishedSnapshotIdForTest(ns); ASSERT_TRUE(published.has_value()); - const auto sample = readCkpt(*backend, store->layout(), life); + const auto sample = readCkpt(op, store->layout(), life); ASSERT_TRUE(sample.has_value()); EXPECT_EQ(sample->ckpt.checkpoint_snapshot_id, published); EXPECT_EQ(sample->ckpt.life_epoch, epoch) << "the publisher contributes nothing about life_epoch, so " "the merge must preserve what the birth wrote"; /// And the snapshot body it names really is there -- the checkpoint may never point at a key that /// does not exist, which is the premise the missing-base rule reasons from. - EXPECT_TRUE(backend->head(store->layout().refSnapshotKey(life, *published)).exists); + EXPECT_TRUE(op.head(store->layout().refSnapshotKey(life, *published), Retry::standard()).has_value()); } /// The body-PUT/cleanup/`_ckpt` race, decided by the ORDER of the two writes: cleanup planned in the @@ -992,18 +1147,20 @@ TEST(CASRefCheckpoint, CleanupPlannedBetweenTheBodyPutAndTheCkptCasCannotDeleteT { auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const uint64_t epoch = store->writerEpoch(); const RootNamespace ns{"srv1/ckpt_race"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); const RefTxnId first_snapshot = *store->newestPublishedSnapshotIdForTest(ns); ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{epoch, 2})); /// The checkpoint a cleanup pass sampled BEFORE the second publication -- the stale reading the /// race hands it. - const std::optional stale_checkpoint = readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id; + const std::optional stale_checkpoint = readCkptOrFail(op, store->layout(), life).checkpoint_snapshot_id; ASSERT_EQ(stale_checkpoint, first_snapshot); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); @@ -1015,7 +1172,7 @@ TEST(CASRefCheckpoint, CleanupPlannedBetweenTheBodyPutAndTheCkptCasCannotDeleteT EXPECT_FALSE(snapshotDeletableUnderCkpt(second_snapshot, stale_checkpoint)); EXPECT_FALSE(snapshotDeletableUnderCkpt(first_snapshot, stale_checkpoint)); /// Once the checkpoint is re-read, the older snapshot becomes reclaimable and the base does not. - const std::optional fresh_checkpoint = readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id; + const std::optional fresh_checkpoint = readCkptOrFail(op, store->layout(), life).checkpoint_snapshot_id; EXPECT_TRUE(snapshotDeletableUnderCkpt(first_snapshot, fresh_checkpoint)); EXPECT_FALSE(snapshotDeletableUnderCkpt(second_snapshot, fresh_checkpoint)); } @@ -1024,35 +1181,39 @@ TEST(CASRefCheckpoint, CleanupPlannedBetweenTheBodyPutAndTheCkptCasCannotDeleteT /// published, and a publisher with nothing above its newest snapshot touches it at all. TEST(CASRefCheckpoint, TheCheckpointIsWrittenOncePerPublicationAndNotOnIdleAttempts) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const uint64_t epoch = store->writerEpoch(); const RootNamespace ns{"srv1/ckpt_republish"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String key = store->layout().refCkptKey(life); - const uint64_t writes_after_birth = backend->casPutCount(key); + const uint64_t writes_after_birth = backend->writes(key); EXPECT_EQ(writes_after_birth, 2u) << "birth publishes `life_epoch` before its log, then the durable log's committed frontier"; ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); - EXPECT_EQ(backend->casPutCount(key), writes_after_birth + 1) << "one publication, one checkpoint CAS"; - const uint64_t writes_after_publish = backend->casPutCount(key); - const auto after_publish = readCkpt(*backend, store->layout(), life); + EXPECT_EQ(backend->writes(key), writes_after_birth + 1) << "one publication, one checkpoint write"; + const uint64_t writes_after_publish = backend->writes(key); + const auto after_publish = readCkpt(op, store->layout(), life); ASSERT_TRUE(after_publish.has_value()); /// Nothing was appended since, so there is nothing above the newest snapshot: the publisher declines /// before it reaches the checkpoint at all, and repeating the attempt changes nothing. EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); - EXPECT_EQ(backend->casPutCount(key), writes_after_publish); - EXPECT_EQ(readCkptOrFail(*backend, store->layout(), life), after_publish->ckpt); + EXPECT_EQ(backend->writes(key), writes_after_publish); + EXPECT_EQ(readCkptOrFail(op, store->layout(), life), after_publish->ckpt); } TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"srv1/no_snapshot_at_seal"}; uint64_t predecessor_epoch = 0; { @@ -1066,19 +1227,19 @@ TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite ASSERT_EQ(store->listRefs(ns).size(), 1u) << "recovery must close the predecessor epoch before publishing"; const RefTxnId seal_id{predecessor_epoch, 2}; - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String snapshot_key = store->layout().refSnapshotKey(life, seal_id); const String ckpt_key = store->layout().refCkptKey(life); ASSERT_EQ(store->lastEpochSealForTest(ns), std::make_optional(seal_id)); - ASSERT_EQ(readCkptOrFail(*backend, store->layout(), life).committed_through, std::make_optional(seal_id)); - const uint64_t snapshot_puts_before = backend->putCount(snapshot_key); - const uint64_t ckpt_cas_before = backend->casPutCount(ckpt_key); + ASSERT_EQ(readCkptOrFail(op, store->layout(), life).committed_through, std::make_optional(seal_id)); + const uint64_t snapshot_writes_before = backend->writes(snapshot_key); + const uint64_t ckpt_writes_before = backend->writes(ckpt_key); /// Recovery installed the epoch seal as the runtime's greatest applied transaction. The publisher /// must decline it without reaching either durable write. EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); - EXPECT_EQ(backend->putCount(snapshot_key), snapshot_puts_before); - EXPECT_EQ(backend->casPutCount(ckpt_key), ckpt_cas_before); + EXPECT_EQ(backend->writes(snapshot_key), snapshot_writes_before); + EXPECT_EQ(backend->writes(ckpt_key), ckpt_writes_before); /// Once an ordinary transaction advances the candidate beyond the seal, normal publication resumes. ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 1})); @@ -1088,17 +1249,19 @@ TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite /// Publication replays a `NeedsRecovery` lane before it captures a snapshot and advances `_ckpt`. TEST(CASRefCheckpoint, NeedsRecoveryReplaysBeforeCheckpointAdvance) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"srv1/ckpt_poisoned"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String key = store->layout().refCkptKey(life); - const auto before = readCkpt(*backend, store->layout(), life); + const auto before = readCkpt(op, store->layout(), life); ASSERT_TRUE(before.has_value()); ASSERT_FALSE(before->ckpt.checkpoint_snapshot_id.has_value()); - const uint64_t writes_before = backend->casPutCount(key); + const uint64_t writes_before = backend->writes(key); /// Enter `NeedsRecovery`: an install throws after its transaction is /// durable, leaving this cached table missing a transaction the log contains. @@ -1121,9 +1284,9 @@ TEST(CASRefCheckpoint, NeedsRecoveryReplaysBeforeCheckpointAdvance) EXPECT_TRUE(store->resolveRef(ns, "ref_2", /*allow_stale=*/false).has_value()) << "the stranded transaction is durable; the re-derivation must have applied it"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_GT(backend->casPutCount(key), writes_before) + EXPECT_GT(backend->writes(key), writes_before) << "and the checkpoint advances -- truthfully, over a snapshot that is not missing anything"; - EXPECT_TRUE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + EXPECT_TRUE(readCkptOrFail(op, store->layout(), life).checkpoint_snapshot_id.has_value()); } @@ -1136,26 +1299,27 @@ TEST(CASRefCheckpoint, APublishFencedOutMidAttemptDoesNotAdvanceTheCheckpoint) /// The watched key cannot be computed yet -- the real incarnation is minted only once the pool /// exists and this namespace's first open resolves it (`setWatchedKey` below, once it has). - auto backend = std::make_shared(""); + auto backend = std::make_shared(); DB::Cas::tests::seedPoolMetaForRestart(*backend); PoolPtr store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String ckpt_key = store->layout().refCkptKey(life); - backend->setWatchedKey(ckpt_key); - - const auto before = readCkpt(*backend, store->layout(), life); + const auto before = readCkpt(op, store->layout(), life); ASSERT_TRUE(before.has_value()); ASSERT_FALSE(before->ckpt.checkpoint_snapshot_id.has_value()); - const uint64_t writes_before = backend->casPutCount(ckpt_key); + const uint64_t writes_before = backend->writes(ckpt_key); + backend->watch(ckpt_key); /// Arm only after the precondition read above. The next watched `_ckpt` read is therefore the one /// inside this publish's read-then-CAS window, after the attempt captured its immutable runtime /// generation. Arming before `readCkpt` would stale the runtime before the operation began and test /// entry admission instead of the intended mid-attempt recheck. bool hook_fired = false; - backend->on_get = [&] + backend->after_read = [&] { if (hook_fired) return; @@ -1165,16 +1329,16 @@ TEST(CASRefCheckpoint, APublishFencedOutMidAttemptDoesNotAdvanceTheCheckpoint) EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) << "a publish whose checkpoint could not be advanced must not report success"; - EXPECT_TRUE(hook_fired) << "the checkpoint read-then-CAS seam was never exercised"; - backend->on_get = nullptr; - EXPECT_EQ(backend->casPutCount(ckpt_key), writes_before) << "nothing may be sent after the fence moved"; - EXPECT_FALSE(readCkptOrFail(*backend, store->layout(), life).checkpoint_snapshot_id.has_value()); + EXPECT_TRUE(hook_fired) << "the checkpoint read-then-write seam was never exercised"; + backend->after_read = nullptr; + EXPECT_EQ(backend->writes(ckpt_key), writes_before) << "nothing may be sent after the fence moved"; + EXPECT_FALSE(readCkptOrFail(op, store->layout(), life).checkpoint_snapshot_id.has_value()); EXPECT_FALSE(store->newestPublishedSnapshotIdForTest(ns).has_value()) << "the snapshot must not be adopted as the newest while its checkpoint is unpublished"; } /// =================================================================================== -/// Equivalence fences for the `prepareRefChunk` extraction (Stage B `{#extract-prepare-ref-chunk}`) +/// Equivalence fences for the `prepareRefChunk` extraction /// =================================================================================== /// /// An extraction is only safe to review if something pins what crosses its boundary. These three @@ -1191,18 +1355,20 @@ TEST(CASRefCheckpoint, CommitRefChunkDurableBytesUnchangedByExtraction) { auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"test/golden@cas@"}; const RefTxnId id = publishRef(store, ns, "gold_ref", 7); ASSERT_EQ(id.writer_epoch, 1u); ASSERT_EQ(id.ref_sequence, 1u); - /// The KEY carries the namespace incarnation, so its life segment is rendered rather than pasted - /// (Task 1c re-keys it); every other segment is literal. Stage B (Task 4-C): the incarnation is now - /// a REAL, randomly minted catalog value rather than the Stage-A sentinel, so it is learned back + /// The KEY carries the namespace incarnation, so its life segment is rendered rather than pasted; + /// every other segment is literal. Stage B: the incarnation is now a REAL, randomly minted catalog + /// value rather than the Stage-A sentinel, so it is learned back /// from the catalog (`liveLifeOrFail`) rather than pasted as a literal -- the shape assertion below /// is unaffected, since it names every OTHER segment literally and renders this one dynamically. - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String key = store->layout().refLogKey(life, id); EXPECT_EQ(key, "p/cas/ns/stream/" + renderIncarnation(life.incarnation) + "/_log/0000000000000001-0000000000000001.zst") @@ -1211,26 +1377,21 @@ TEST(CASRefCheckpoint, CommitRefChunkDurableBytesUnchangedByExtraction) /// The BODY is checked as exact length plus a 128-bit SipHash of it -- not literally byte for byte, /// but any change that survives both is a 128-bit collision at a fixed length, which is the trade for /// keeping the assertion readable. It is a function of `{format generation, ns, id, ops, - /// chain_link}` only -- no incarnation reaches it. Generation 10 changed the shared format header; - /// the plaintext discriminator below removes only that change and pins every remaining byte to the - /// generation-9 fixture before accepting the new deterministic compressed size and hash. - const auto got = backend->get(key); + /// chain_link}` only -- no incarnation reaches it. + const auto got = op.read(key, Retry::standard()); ASSERT_TRUE(got.has_value()) << "the birth chunk must be durable at its canonical key"; - String as_generation_9 = openObject(FormatId::RefLog, got->bytes); - const String generation_10_header = R"({"type":"cas_ref_log","v":10})"; - ASSERT_TRUE(as_generation_9.starts_with(generation_10_header)); - as_generation_9.replace(0, generation_10_header.size(), R"({"type":"cas_ref_log","v":9})"); - EXPECT_EQ(as_generation_9, R"({"type":"cas_ref_log","v":9} -{"ns":"test/golden@cas@","we":"1","rs":"1"} + const String plaintext = openObject(FormatId::RefLog, got->bytes); + EXPECT_EQ(plaintext, R"({"type":"cas_ref_log","v":1} +{"namespace":"test/golden@cas@","txn_epoch":"1","txn_seq":"1"} {"op":"namespace_birth"} -{"op":"owner_transition","nbk":"precommit","nrn":"gold_ref","nme":"1","nmb":"7","nmo":1} -{"op":"owner_transition","obk":"precommit","orn":"gold_ref","ome":"1","omb":"7","omo":1,"nbk":"committed","nrn":"gold_ref","nme":"1","nmb":"7","nmo":1} +{"op":"owner_transition","new_kind":"precommit","new_ref":"gold_ref","new_epoch":"1","new_build":"7","new_ord":1} +{"op":"owner_transition","old_kind":"precommit","old_ref":"gold_ref","old_epoch":"1","old_build":"7","old_ord":1,"new_kind":"committed","new_ref":"gold_ref","new_epoch":"1","new_build":"7","new_ord":1} {"n":3} -)") << "generation 10 must change only the self-describing header of this ref-log fixture"; - EXPECT_EQ(got->bytes.size(), 179u) << "the sealed ref-log body changed size"; +)") << "the sealed ref-log plaintext changed"; + EXPECT_EQ(got->bytes.size(), 206u) << "the sealed ref-log body changed size"; SipHash body_hash; body_hash.update(got->bytes.data(), got->bytes.size()); - EXPECT_EQ(getHexUIntLowercase(body_hash.get128()), "ada75a83638e933c98d731183a46b7b7") + EXPECT_EQ(getHexUIntLowercase(body_hash.get128()), "21c275ad44a6b47a4d6c389c0d71bb34") << "the sealed ref-log body changed content -- preparation must seal the same bytes it sealed " "before the extraction"; } @@ -1246,24 +1407,26 @@ TEST(CASRefCheckpoint, CommitRefChunkDurableBytesUnchangedByExtraction) /// need a sequence-recording backend to pin. TEST(CASRefCheckpoint, AppendRequestCountUnchangedByExtraction) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"test/req@cas@"}; const RefTxnId id = publishRef(store, ns, "req_ref", 1); - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String log_key = store->layout().refLogKey(life, id); const String ckpt_key = store->layout().refCkptKey(life); - EXPECT_EQ(backend->putCount(log_key), 1u) << "exactly one write-once PUT per committed chunk"; - /// ONE GET, not zero, since Stage B (Task 4-C): `resolveNamespaceLife`'s `completeCreation` call + EXPECT_EQ(backend->writes(log_key), 1u) << "exactly one write-once request per committed chunk"; + /// ONE GET, not zero, since Stage B: `resolveNamespaceLife`'s `completeCreation` call /// publishes this life's `_ckpt.life_epoch` BEFORE the birth chunk is prepared, so this table's /// OWN recovery walk (also inside this `appendRefOps`, ahead of the commit) grounds itself at the /// genesis position `_ckpt` now names and confirms it absent by exact key -- which is `log_key` /// itself, the position the birth chunk is about to occupy. That GET precedes the Committed PUT; /// the PUT itself still owes no read-back. - EXPECT_EQ(backend->getCount(log_key), 1u) << "one grounding probe from recovery, before the birth PUT"; - EXPECT_EQ(backend->casPutCount(ckpt_key), 2u) + EXPECT_EQ(backend->getCount(log_key), 1u) << "one grounding probe from recovery, before the birth write"; + EXPECT_EQ(backend->writes(ckpt_key), 2u) << "the birth contributes `life_epoch` before its log and `committed_through` after the durable " "log; these are two different ordering obligations, not a duplicate publication"; } @@ -1298,6 +1461,8 @@ TEST(CASRefCheckpoint, PostDurableInstallRegionStillEnteredAfterExtraction) { auto backend = std::make_shared(); auto store = openPool(backend); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const RootNamespace ns{"test/region@cas@"}; unsigned probe_hits = 0; @@ -1308,6 +1473,6 @@ TEST(CASRefCheckpoint, PostDurableInstallRegionStillEnteredAfterExtraction) EXPECT_GT(probe_hits, 0u) << "no probe-instrumented post-durable install region was entered on a committing append -- " "the `Committed` install arm was not reached at all"; - const NamespaceLifeId life = liveLifeOrFail(*backend, store->layout(), ns); - EXPECT_TRUE(backend->get(store->layout().refLogKey(life, id)).has_value()); + const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); + EXPECT_TRUE(op.read(store->layout().refLogKey(life, id), Retry::standard()).has_value()); } diff --git a/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp b/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp index e50603d9258d..9e0d6918b3e0 100644 --- a/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp +++ b/src/Disks/tests/gtest_cas_ref_ckpt_join.cpp @@ -14,10 +14,13 @@ #include #include +#include #include +#include #include #include #include +#include #include /// The `_ckpt` JOIN law and its `O(1)` SIZE invariant. @@ -41,7 +44,7 @@ /// so the fence would not fire on the very change it exists to catch. Only a real producer /// populates a real field. /// - TRANSACTIONS and WRITER EPOCHS enter as the DECIMAL WIDTH of the two id pairs. That is not -/// equality: `{cse=1,css=1}` and `{cse=1,css=10000}` differ by four bytes. It is `O(1)` because +/// equality: `{snapshot_epoch=1,snapshot_seq=1}` and `{snapshot_epoch=1,snapshot_seq=10000}` differ by four bytes. It is `O(1)` because /// the fields are `uint64_t` and so the width is ceilinged at twenty digits, which is a bound a /// test asserts on a constructed worst case -- `EncodedCkptSizeHasAConstantCeiling...` below. @@ -73,9 +76,10 @@ constexpr uint64_t U64_MAX = std::numeric_limits::max(); /// Constraint 15's bound, as a number: the encoded size of the WIDEST `_ckpt` this build can produce /// (all three fields present, every integer component at `UINT64_MAX`). Pinned as a literal so that -/// adding a field, or widening one, fails a test rather than quietly moving the bound. Generation 10 -/// added one byte to the shared format-version header (`9` became `10`); the scalar body is unchanged. -constexpr size_t CKPT_WORST_CASE_ENCODED_BYTES = 235; +/// adding a field, or widening one, fails a test rather than quietly moving the bound. The shared +/// format-version header is the single-digit `v:1` baseline; a future generation bump that widens it +/// moves this constant too. +constexpr size_t CKPT_WORST_CASE_ENCODED_BYTES = 296; /// The high-cardinality side of the size fence, in ONE transaction. Bounded above by the append lane's /// 5000-operation cap on a normal-class item (`publishCommittedOps` emits two ops per ref), and kept at @@ -92,29 +96,42 @@ String refName(size_t i) return fmt::format("r{:08}", i); } -/// A fence that never refuses, and a deadline far enough out that only the test's own contention -/// decides the outcome -- each `_ckpt`/catalog test file defines its own copy, matching the precedent -/// `gtest_cas_ns_creation_lifecycle.cpp` states explicitly. -const std::function ALWAYS_ADMITTED = [](uint64_t) {}; - -CkptDeadline generousDeadline() +/// Withdraws an operation's admission at a chosen point inside another component's read-then-write +/// window: deterministically, with no sleep and no second thread. A test arms exactly one of the two +/// points and reads `admitted` from its operation's liveness predicate. +class AdmissionHookBackend : public CountingBackend { - return CkptDeadline{[] { return uint64_t{1000}; }, 60000}; -} +public: + bool admitted = true; + /// Withdraw once this exact key has been read. + String withdraw_after_read_of; + /// Withdraw once any `_ckpt` key has been written. The key carries an incarnation the test cannot + /// know before the creation mints it, so the arm names the object kind rather than the key. + bool withdraw_after_ckpt_write = false; -/// Admits the FIRST call (spent by `completeCreation`'s step-2 `publishCkpt`) and refuses every call -/// after (step 3's own `mutate`): "fenced out between the `_ckpt` create and the `Creating -> Live` -/// CAS", deterministically and without a second thread. That is the durable shape a stalled creator -/// leaves behind, and the starting state the resumption test needs. -std::function admittedOnceThenFenced() -{ - auto calls = std::make_shared(0); - return [calls](uint64_t admitted) + explicit AdmissionHookBackend(Layout layout_) : layout(std::move(layout_)) {} + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (++*calls > 1) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); - }; -} + auto result = CountingBackend::read(key, access); + if (!withdraw_after_read_of.empty() && key == withdraw_after_read_of) + admitted = false; + return result; + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + auto result = CountingBackend::write(key, bytes, expected_value, access); + if (withdraw_after_ckpt_write && layout.parseRefCkptKey(key)) + admitted = false; + return result; + } + +private: + Layout layout; +}; CreatorFence creatorFence(const String & srid, uint64_t writer_epoch, uint64_t fence_generation = 1) { @@ -138,9 +155,9 @@ const CatalogEntry * findEntryForTest(const RefCatalog & catalog, const RootName /// `life`'s durable `life_epoch`, failing the current test rather than dereferencing a disengaged /// optional -- a bare `->` on one aborts the whole binary and takes every later suite's result with it. -uint64_t lifeEpochOrFail(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +uint64_t lifeEpochOrFail(CasOperation & op, const Layout & layout, const NamespaceLifeId & life) { - const std::optional sample = readCkpt(backend, layout, life); + const std::optional sample = readCkpt(op, layout, life); if (!sample || !sample->ckpt.life_epoch) { ADD_FAILURE() << "expected a _ckpt carrying a life_epoch for namespace '" << life.ns.string() << "'"; @@ -165,9 +182,9 @@ PoolPtr openPool(const BackendPtr & backend, std::function boot_ms_f /// The incarnation the production birth wiring minted for `ns`, learned back from the catalog the way a /// real reader does. Fails the current test rather than dereferencing a disengaged optional, so one /// regression cannot abort the binary and take every later suite's result with it. -NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const RootNamespace & ns) +NamespaceLifeId liveLifeOrFail(CasOperation & op, const Layout & layout, const RootNamespace & ns) { - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); for (const CatalogEntry & entry : snap.catalog.entries) if (entry.ns.string() == ns.string()) return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); @@ -181,7 +198,7 @@ NamespaceLifeId liveLifeOrFail(Backend & backend, const Layout & layout, const R /// One transaction also holds every OTHER dimension fixed while `ref_count` varies: two namespaces /// built this way end at the same transaction id, so a difference in their `_ckpt` bodies can only be /// the refs. `ref_count` must therefore stay within the append lane's per-item operation cap. -String encodedCkptOfNamespaceWithRefs(const PoolPtr & store, Backend & backend, const Layout & layout, +String encodedCkptOfNamespaceWithRefs(const PoolPtr & store, CasOperation & op, const Layout & layout, const RootNamespace & ns, size_t ref_count) { store->appendRefOps(ns, MutationScope::wholeShard(), @@ -191,14 +208,14 @@ String encodedCkptOfNamespaceWithRefs(const PoolPtr & store, Backend & backend, if (state.getLifecycle() != RefLifecycle::Live) ops.push_back(namespaceBirthOp()); for (size_t i = 0; i < ref_count; ++i) - for (const RefOp & op : publishCommittedOps(refName(i), ManifestRef{1, i + 1, 1})) - ops.push_back(op); + for (const RefOp & ref_op : publishCommittedOps(refName(i), ManifestRef{1, i + 1, 1})) + ops.push_back(ref_op); return ops; }, RootMutationOrigin::Writer, RootMutationKind::Publish); - const NamespaceLifeId life = liveLifeOrFail(backend, layout, ns); - const std::optional sample = readCkpt(backend, layout, life); + const NamespaceLifeId life = liveLifeOrFail(op, layout, ns); + const std::optional sample = readCkpt(op, layout, life); if (!sample) { ADD_FAILURE() << "expected a _ckpt for namespace '" << ns.string() << "' after its birth transaction"; @@ -298,36 +315,41 @@ TEST(CASRefCheckpointJoin, CrossEpochFrontierRequiresAnImmediatelyAdjacentSeal) /// only ever rise", so the fixture must not quietly model two roots. TEST(CASRefCheckpointJoin, ResumedCreationRaisesLifeEpochWithoutRefusal) { - InMemoryBackend backend; Layout layout("p"); - DB::Cas::tests::seedPoolMetaForRestart(backend); + auto backend = std::make_shared(layout); + DB::Cas::tests::seedPoolMetaForRestart(*backend); const RootNamespace ns{"a"}; - - ASSERT_EQ(CasRefCatalog::createNamespace(backend, layout, 1, ns, creatorFence("srv1", 5), - /*admitted_generation=*/1, admittedOnceThenFenced(), generousDeadline()), + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation reader = requests.admit(); + + /// The creator loses admission the instant its step-2 `_ckpt` is durable, so step 3 never sends + /// its `Creating -> Live` write. That is the durable shape a stalled creator leaves behind, and + /// the starting state this test needs. + backend->withdraw_after_ckpt_write = true; + CasOperation creator = requests.admit([&backend] { return backend->admitted; }); + ASSERT_EQ(CasRefCatalog::createNamespace(creator, layout, 1, ns, creatorFence("srv1", 5)), CasRefCatalog::NamespaceCreationOutcome::FencedOut); + backend->withdraw_after_ckpt_write = false; /// Bound to a name, never chained through a temporary: a `const CatalogEntry *` taken from an /// unbound `Snapshot` dangles the instant the full expression ends. - const CasRefCatalog::Snapshot stalled = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot stalled = CasRefCatalog::read(reader, layout); const CatalogEntry * entry = findEntryForTest(stalled.catalog, ns); ASSERT_NE(entry, nullptr); ASSERT_EQ(entry->state, NsState::Creating); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); - EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 5u) << "step 2 landed before the creator stalled"; + EXPECT_EQ(lifeEpochOrFail(reader, layout, life), 5u) << "step 2 landed before the creator stalled"; const CreatorFence resumer = creatorFence("srv1", 9); - ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(backend, layout, *entry, resumer, fixedTerminality(true), - /*admitted_generation=*/1, ALWAYS_ADMITTED), + ASSERT_EQ(CasRefCatalog::reconcileStaleCreator(reader, layout, *entry, resumer, fixedTerminality(true)), CasRefCatalog::ReconcileCreatorOutcome::Reconciled); CatalogEntry resumed = *entry; resumed.creator = resumer; - EXPECT_EQ(CasRefCatalog::completeCreation(backend, layout, resumed, /*admitted_generation=*/1, - ALWAYS_ADMITTED, generousDeadline()), + EXPECT_EQ(CasRefCatalog::completeCreation(reader, layout, resumed), CasRefCatalog::NamespaceCreationOutcome::Live) << "the resumption must not be refused by the join"; - EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u) + EXPECT_EQ(lifeEpochOrFail(reader, layout, life), 9u) << "the genesis epoch that actually landed is the resuming actor's, and the join must let it rise"; } @@ -338,19 +360,19 @@ TEST(CASRefCheckpointJoin, ResumedCreationRaisesLifeEpochWithoutRefusal) /// chunk contributes the `NamespaceBirth` record's epoch. CREATE TABLE, restart, INSERT. TEST(CASRefCheckpointJoin, RestartBetweenCreationAndFirstWriteRaisesLifeEpochWithoutRefusal) { - InMemoryBackend backend; + auto backend = std::make_shared(); Layout layout("p"); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); const RefCkpt from_creation{.life_epoch = 4, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(backend, layout, life, from_creation, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); + ASSERT_EQ(publishCkpt(op, layout, life, from_creation), CkptPublishOutcome::Published); const RefCkpt from_birth_chunk{.life_epoch = 7, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(backend, layout, life, from_birth_chunk, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published) + EXPECT_EQ(publishCkpt(op, layout, life, from_birth_chunk), CkptPublishOutcome::Published) << "the birth chunk's later epoch must be publishable, not refused as a conflict"; - EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 7u); + EXPECT_EQ(lifeEpochOrFail(op, layout, life), 7u); } /// THE REFUSAL, and the state it constructs IS UNREACHABLE ON ANY HONEST PATH -- that is the point of @@ -372,21 +394,23 @@ TEST(CASRefCheckpointJoin, RestartBetweenCreationAndFirstWriteRaisesLifeEpochWit /// of its arguments is durable. There is deliberately no merge-level counterpart to this test. TEST(CASRefCheckpointJoin, JoinDecreasingLifeEpochIsCorruptionAndPublishesNothing) { - CountingBackend backend; + auto backend = std::make_shared(); Layout layout("p"); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); const String key = layout.refCkptKey(life); const RefCkpt durable{.life_epoch = 9, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(backend, layout, life, durable, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const uint64_t cas_puts_before = backend.casPutCount(key); + ASSERT_EQ(publishCkpt(op, layout, life, durable), CkptPublishOutcome::Published); + /// This suite writes one key only, so the backend's own total is that key's count. + const uint64_t writes_before = backend->writeTotal(); const RefCkpt superseded{.life_epoch = 3, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; String message; try { - publishCkpt(backend, layout, life, superseded, 1, ALWAYS_ADMITTED, generousDeadline()); + publishCkpt(op, layout, life, superseded); ADD_FAILURE() << "a contribution below the durable life_epoch must not be published"; } catch (const DB::Exception & e) @@ -409,8 +433,8 @@ TEST(CASRefCheckpointJoin, JoinDecreasingLifeEpochIsCorruptionAndPublishesNothin /// And nothing was written. The refusal is decided before the body is built, so the durable object /// is untouched and no write was even attempted. - EXPECT_EQ(backend.casPutCount(key), cas_puts_before) << "the publisher must not CAS on a refused publish"; - EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u) << "the durable value is unchanged"; + EXPECT_EQ(backend->writeTotal(), writes_before) << "the publisher must not write on a refused publish"; + EXPECT_EQ(lifeEpochOrFail(op, layout, life), 9u) << "the durable value is unchanged"; } /// The other half of the refusal, and the reason it consults the fence before classifying: the SAME @@ -421,26 +445,28 @@ TEST(CASRefCheckpointJoin, JoinDecreasingLifeEpochIsCorruptionAndPublishesNothin /// violation is a STILL-ADMITTED writer contributing a superseded epoch. TEST(CASRefCheckpointJoin, ADecreasingLifeEpochFromAFencedOutWriterIsReportedFencedOutNotCorruption) { - CountingBackend backend; Layout layout("p"); + auto backend = std::make_shared(layout); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation reader = requests.admit(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"a"}, UInt128(42)); const String key = layout.refCkptKey(life); const RefCkpt durable{.life_epoch = 9, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - ASSERT_EQ(publishCkpt(backend, layout, life, durable, 1, ALWAYS_ADMITTED, generousDeadline()), - CkptPublishOutcome::Published); - const uint64_t cas_puts_before = backend.casPutCount(key); - - const std::function always_fenced = [](uint64_t admitted) - { - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "fence generation moved since admission ({})", admitted); - }; + ASSERT_EQ(publishCkpt(reader, layout, life, durable), CkptPublishOutcome::Published); + /// This suite writes one key only, so the backend's own total is that key's count. + const uint64_t writes_before = backend->writeTotal(); + + /// Admission survives the read and is gone by the time the decrease is classified -- the exact + /// window in which the refusal must be reported as a control signal rather than as corruption. + backend->withdraw_after_read_of = key; + CasOperation superseded_writer = requests.admit([&backend] { return backend->admitted; }); const RefCkpt superseded{.life_epoch = 3, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}; - EXPECT_EQ(publishCkpt(backend, layout, life, superseded, 1, always_fenced, generousDeadline()), - CkptPublishOutcome::FencedOut); + EXPECT_EQ(publishCkpt(superseded_writer, layout, life, superseded), CkptPublishOutcome::FencedOut); + backend->withdraw_after_read_of.clear(); - EXPECT_EQ(backend.casPutCount(key), cas_puts_before); - EXPECT_EQ(lifeEpochOrFail(backend, layout, life), 9u); + EXPECT_EQ(backend->writeTotal(), writes_before); + EXPECT_EQ(lifeEpochOrFail(reader, layout, life), 9u); } /// `checkpoint_snapshot_id` and `last_epoch_seal` continue to merge by SEMANTIC MAXIMUM. Unlike @@ -490,9 +516,11 @@ TEST(CASRefCheckpointJoin, EncodedCkptSizeIsIndependentOfCardinality) /// instead of racing it (see `openPool`'s doc comment). auto store = openPool(backend, [] { return uint64_t{0}; }); Layout layout("p"); + CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); + CasOperation op = requests.admit(); - const String one = encodedCkptOfNamespaceWithRefs(store, *backend, layout, RootNamespace{"srv1/one"}, 1); - const String many = encodedCkptOfNamespaceWithRefs(store, *backend, layout, RootNamespace{"srv1/many"}, MANY_REFS); + const String one = encodedCkptOfNamespaceWithRefs(store, op, layout, RootNamespace{"srv1/one"}, 1); + const String many = encodedCkptOfNamespaceWithRefs(store, op, layout, RootNamespace{"srv1/many"}, MANY_REFS); ASSERT_FALSE(one.empty()); ASSERT_FALSE(many.empty()); diff --git a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp index bc911e4ad6db..31648e992ceb 100644 --- a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp +++ b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp @@ -6,7 +6,7 @@ #include #include #include -#include +#include #include #include #include @@ -69,27 +69,37 @@ PoolPtr openPool(const BackendPtr & backend) } /// The fence-controlled pool of `gtest_cas_ref_install_safety.cpp`, for the pre-attempt refusal: the -/// boot clock is frozen so `setMountDeadline` alone decides both fence predicates, renewal is parked an -/// hour out so nothing re-arms the deadline underneath the test, and the single-attempt budget makes -/// `attempt_timeout_ms + lease_safety_margin_ms` (200 ms) the window between "the flush is admitted" -/// and "an attempt may start". -PoolPtr openPoolFenceControlled(const BackendPtr & backend) +/// boot clock is frozen so `setMountDeadline` alone decides every fence predicate, renewal is parked an +/// hour out so nothing re-arms the deadline underneath the test, and the backend reports the budget's +/// own `attempt_timeout_ms` because that -- not the budget field -- is what the request engine reserves +/// per attempt, exactly as `ContentAddressedMetadataStorage` pairs the two in production. No fault is +/// injected here at all -- the refusal comes from the lease having no room to start a write -- so +/// nothing in this fixture depends on an attempt count. +PoolPtr openPoolFenceControlled(const std::shared_ptr & backend) { DB::Cas::tests::seedPoolMetaForRestart(*backend); PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; cfg.boot_ms_fn = [] { return uint64_t{0}; }; cfg.mount_renew_period = std::chrono::milliseconds{3600000}; CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; cfg.cas_request_budget = budget; + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); return Pool::open(backend, cfg); } constexpr uint64_t FENCE_DEADLINE_HEALTHY_MS = 30000; -constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 100; +/// Between "the flush is admitted" and "an attempt may start" sit three gates with different +/// appetites, all measured against the lease's remaining time (the frozen clock at 0 makes the +/// deadline BE the remaining time), and each refuses until its own reservation plus +/// `lease_safety_margin_ms` (100) is STRICTLY cleared: +/// a `CasOperation::admitted` guard reserves nothing -- clears above 100; +/// a read reserves one attempt envelope -- clears above 200; +/// a write reserves TWO, the attempt and the read that settles it -- clears above 300. +/// This test wants the guards and the reads on the way in to pass while the append's own first +/// request is refused, so it sits strictly between the second and the third. +constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 250; /// A bare `Pool::open` with no `_pool_meta` seeded: the path an operator's pool RECREATION takes, and /// the only one that runs the bootstrap residual + quiesce gates (`seedPoolMetaForRestart` mints the @@ -106,9 +116,10 @@ size_t eraseKeysContaining(Backend & backend, const String & substr) size_t removed = 0; String cursor; std::vector keys; + OperationForTest op(backend); while (true) { - const ListPage page = backend.list("", cursor, 1000); + const ListPage page = (*op).list("", cursor, 1000, Retry::standard()); for (const ListedKey & listed : page.keys) if (substr.empty() || listed.key.find(substr) != String::npos) keys.push_back(listed.key); @@ -118,8 +129,8 @@ size_t eraseKeysContaining(Backend & backend, const String & substr) } for (const String & key : keys) { - const HeadResult h = backend.head(key); - if (h.exists && backend.deleteExact(key, h.token).kind == DeleteOutcome::Kind::Deleted) + const auto h = (*op).head(key, Retry::standard()); + if (h && (*op).remove(key, h->etag, Retry::standard()) == Removal::Removed) ++removed; } return removed; @@ -242,9 +253,8 @@ TEST(CASRefContiguousAlloc, EpochChangeRestartsTheSequenceAtOne) } /// The read side is what makes INV-1 an invariant rather than a convention: a transaction whose id is -/// not the successor of `greatest_applied` is CORRUPTED_DATA, naming both ids. Before this task the -/// state machine checked strict increase only, so a stream with a hole applied cleanly and no reader -/// could tell a complete chain from a truncated one. +/// not the successor of `greatest_applied` is CORRUPTED_DATA, naming both ids. Strict increase alone +/// admits holes, so it cannot distinguish a complete chain from a truncated one. TEST(CASRefContiguousAlloc, NonSuccessorIdIsRejectedOnApply) { const String ns = "srv1/contig_density"; @@ -253,7 +263,7 @@ TEST(CASRefContiguousAlloc, NonSuccessorIdIsRejectedOnApply) RefTableState state = replay(DB::Cas::tests::minimalLiveSnapshot(ns, RefTxnId{kEpoch, 1}), {}); ASSERT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch, 1})); - /// Strictly greater, but skips {7,2}: admitted before this task, rejected now. + /// Strictly greater, but skips {7,2}: not the required successor. try { applyRefLogTxn(state, RefLogTxn{ns, RefTxnId{kEpoch, 3}, publishCommittedOps("r", ManifestRef{1, 1, 1}), std::nullopt}); @@ -289,128 +299,22 @@ TEST(CASRefContiguousAlloc, NonSuccessorIdIsRejectedOnApply) EXPECT_EQ(state.getGreatestApplied(), (RefTxnId{kEpoch + 1, 1})); } -/// The format floor. A pool written before contiguous ref streams holds ref logs whose ids this build -/// would read as a corrupt (holed) chain, so opening it must fail closed at the pool metadata, naming -/// recreation as the migration -- CAS is pre-release and has no in-place migration path. -TEST(CASRefContiguousAlloc, OldPoolFormatIsRefusedNamingRecreation) -{ - PoolMeta pm; - pm.pool_id = UInt128{1, 2}; - pm.blob_header_len = 256; - pm.min_reader_generation = G_BUILD; - pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - - const String current = encodePoolMeta(pm); - EXPECT_NO_THROW(decodePoolMeta(current)); - - /// Rewrite the header-line generation to the last pre-contiguous one, exactly as an older build - /// would have stamped it. - const String from = "\"v\":" + std::to_string(G_BUILD); - const String to = "\"v\":" + std::to_string(kContiguousRefStreamsGeneration - 1); - const size_t at = current.find(from); - ASSERT_NE(at, String::npos); - String old_format = current; - old_format.replace(at, from.size(), to); - - try - { - decodePoolMeta(old_format); - FAIL() << "a pre-contiguous pool must not open"; - } - catch (const DB::Exception & e) - { - EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); - EXPECT_NE(e.message().find(fmt::format("CAS pool format {} predates generation-10 mount-attempt-identity floor", - kContiguousRefStreamsGeneration - 1)), String::npos) - << "the message must name the migration: " << e.message(); - } -} - -/// Generation 6 is a recreate-only physical-layout cut. A generation-5 pool has contiguous, -/// incarnation-qualified streams but still repeats the logical namespace in every key; accepting it -/// would silently run the generation-6 parsers over a different grammar. -TEST(CASRefContiguousAlloc, GenerationFiveNamespaceBearingPoolIsRefusedNamingRecreation) -{ - PoolMeta pm; - pm.pool_id = UInt128{1, 2}; - pm.blob_header_len = 256; - pm.min_reader_generation = G_BUILD; - pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - - const String current = encodePoolMeta(pm); - EXPECT_NO_THROW(decodePoolMeta(current)); - - /// Rewrite the header to the immediately preceding generation, which used - /// `cas/refs///...`. - const String from = "\"v\":" + std::to_string(G_BUILD); - const String to = "\"v\":" + std::to_string(kNamespaceLifeKeyedGeneration); - const size_t at = current.find(from); - ASSERT_NE(at, String::npos); - String old_format = current; - old_format.replace(at, from.size(), to); - ASSERT_EQ(kNamespaceLifeKeyedGeneration + 1, kOpaqueNamespaceLifeLayoutGeneration) - << "this test pins the immediately preceding namespace-bearing generation"; - - try - { - decodePoolMeta(old_format); - FAIL() << "a generation-5 namespace-bearing pool must not open"; - } - catch (const DB::Exception & e) - { - EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); - EXPECT_NE(e.message().find(fmt::format("CAS pool format {} predates generation-10 mount-attempt-identity floor", - kNamespaceLifeKeyedGeneration)), String::npos) - << "the message must name the migration: " << e.message(); - } -} - -/// Mutation caught: leaving the pool floor at generation 6 would admit a seal whose independent -/// name-keyed coverage and cleanup collections this build no longer has. Generation 7 is a -/// recreate-only grammar cut, so the immediately preceding generation must fail at pool open. -TEST(CASRefContiguousAlloc, GenerationSixSplitFoldSealPoolIsRefusedNamingRecreation) -{ - PoolMeta pm; - pm.pool_id = UInt128{1, 2}; - pm.blob_header_len = 256; - pm.min_reader_generation = G_BUILD; - pm.algos_used = {static_cast(BlobHashAlgo::CityHash128)}; - - const String current = encodePoolMeta(pm); - const String from = "\"v\":" + std::to_string(G_BUILD); - const String to = "\"v\":6"; - const size_t at = current.find(from); - ASSERT_NE(at, String::npos); - String old_format = current; - old_format.replace(at, from.size(), to); - - try - { - decodePoolMeta(old_format); - FAIL() << "a generation-6 split ref-life fold seal pool must not open"; - } - catch (const DB::Exception & e) - { - EXPECT_EQ(e.code(), DB::ErrorCodes::UNKNOWN_FORMAT_VERSION); - EXPECT_NE(e.message().find("CAS pool format 6 predates generation-10 mount-attempt-identity floor"), String::npos) - << "the message must name the recreate-only grammar cut: " << e.message(); - } -} - TEST(CASPoolMeta, GcShardsIsPersistedAndOverridesMismatchedReopenConfig) { - InMemoryBackend backend; + auto backend = std::make_shared(); const Layout layout("p"); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); const PoolMeta created = PoolMeta::createOrValidate( - backend, layout, /*blob_header_len=*/256, /*gc_shards=*/4, + op, layout, /*blob_header_len=*/256, /*gc_shards=*/4, BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/true); EXPECT_EQ(created.gc_shards, 4u); const PoolMeta reopened = PoolMeta::createOrValidate( - backend, layout, /*blob_header_len=*/256, /*gc_shards=*/1, + op, layout, /*blob_header_len=*/256, /*gc_shards=*/1, BlobHashAlgo::CityHash128, /*allow_new=*/false, /*allow_mint=*/false); EXPECT_EQ(reopened.gc_shards, 4u); - EXPECT_EQ(decodePoolMeta(backend.get(layout.poolMetaKey())->bytes).gc_shards, 4u); + EXPECT_EQ(decodePoolMeta(op.read(layout.poolMetaKey(), Retry::standard())->bytes).gc_shards, 4u); } /// The one path where "an attempt that provably sent nothing consumes nothing" does not hold, and the @@ -456,9 +360,11 @@ TEST(CASRefContiguousAlloc, NeedsRecoveryReplaysBeforeAllocatingTheNextId) /// The durable stream itself is dense: `1`, `2`, `3` all exist as objects. `ns` was born through /// the REAL append lane (Stage B Task 4-C), so its objects sit at a real catalog-minted incarnation, /// not the Stage-A sentinel -- resolve it the same way production discovery does. - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + CasRequests catalog_requests(backend, Fence::open()); + CasOperation catalog_op = catalog_requests.admit(); + const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(catalog_op, store->layout(), ns).value(); for (uint64_t seq = 1; seq <= 3; ++seq) - EXPECT_TRUE(backend->head(store->layout().refLogKey(life, RefTxnId{epoch, seq})).exists) + EXPECT_TRUE(catalog_op.head(store->layout().refLogKey(life, RefTxnId{epoch, seq}), Retry::once()).has_value()) << "log object " << epoch << "-" << seq << " must exist: the durable stream has no hole"; } @@ -542,7 +448,7 @@ TEST(CASRefContiguousAlloc, RecreationProceedsOnceTheHolderIsTerminal) { auto holder = openPool(backend); publishRef(holder, ns, "ref_1", 1); - } /// destroyed: the keeper stamps the farewell, making the slot terminal + } /// destroyed: the renewer stamps the farewell, making the slot terminal ASSERT_EQ(eraseKeysContaining(*backend, "_pool_meta"), 1u); /// The prefix still holds this pool's data, so the bootstrap still refuses -- but on the ORDINARY @@ -570,14 +476,14 @@ TEST(CASRefContiguousAlloc, RecreationProceedsOnceTheHolderIsTerminal) /// survivor's renewal conclusive. Clearing the prefix also resets the durable writer-epoch counter, so /// a recreation by the SAME server uuid can be handed the very same `(uuid, epoch)` the survivor still /// holds -- and the two are then indistinguishable to the lease protocol, which reads the survivor's -/// renewal as its own keeper adopting a refreshed body. That is precisely why the refusal above is the +/// renewal as its own renewer adopting a refreshed body. That is precisely why the refusal above is the /// primary defence and this fence is only the backstop: quiescing the holder BEFORE the prefix is /// cleared is what keeps the ambiguous case from arising at all. TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) { auto backend = std::make_shared(); /// The survivor uses the runtime-owned renewal worker, as a real mount does: the runtime terminal - /// consumer is what latches the write fence when a renewal fails, so a keeper-only call would + /// consumer is what latches the write fence when a renewal fails, so a renewer-only call would /// reproduce the failure but not the lifecycle effect it causes. PoolConfig survivor_cfg{.pool_prefix = "p", .server_root_id = "test"}; survivor_cfg.background_watermark = true; @@ -616,11 +522,12 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->writerEpoch(), 1})); /// The survivor's TEARDOWN is the other half, and it is asserted here rather than left to the - /// destructor at scope exit. A terminal keeper must skip release without backend I/O: the renewal + /// destructor at scope exit. A terminal renewer must skip release without backend I/O: the renewal /// conflict already counted the conclusive foreign successor, and teardown must neither double-count /// it nor stamp a farewell over the successor's slot. const String survivor_mount_key = recreated->layout().mountKey("test"); - const auto successor_slot_before = backend->get(survivor_mount_key); + OperationForTest teardown_op(*backend); + const auto successor_slot_before = (*teardown_op).read(survivor_mount_key, Retry::once()); ASSERT_TRUE(successor_slot_before.has_value()); const uint64_t skipped_after_deposition = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); @@ -633,7 +540,7 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), violations_before) << "and must NOT report an exclusivity violation: this is a failover, not a broken guarantee"; - const auto successor_slot_after = backend->get(survivor_mount_key); + const auto successor_slot_after = (*teardown_op).read(survivor_mount_key, Retry::once()); ASSERT_TRUE(successor_slot_after.has_value()); EXPECT_EQ(successor_slot_after->bytes, successor_slot_before->bytes) << "the deposed writer must not stamp its farewell over the successor's lease"; diff --git a/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp b/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp index 5b6fa2070c46..3db61594d3ee 100644 --- a/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp +++ b/src/Disks/tests/gtest_cas_ref_epoch_seal_format.cpp @@ -153,7 +153,7 @@ TEST(CASRefEpochSealFormat, EncodeRejectsSealTxnWithSecondNonSealOp) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); } -/// Decode-side pin for the same op-count rule (review finding I1): `encodeRefLogTxn` can never +/// Decode-side pin for the same op-count rule: `encodeRefLogTxn` can never /// produce a 2-op seal body, so only a decode-only splice proves `decodeRefLogTxn` independently /// re-derives the rule rather than trusting whatever the encoder produced -- deleting the structural /// validator's call site inside `decodeRefLogTxn` would leave this the only failing test. @@ -213,7 +213,7 @@ TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealAtNonUnitSequence) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); } -/// Decode-side pin for the sequence-1-only rule (review finding I1). `prev_epoch_seal`'s +/// Decode-side pin for the sequence-1-only rule. `prev_epoch_seal`'s /// writer_epoch (1) is strictly below the transaction's own (5), satisfying the I3 chain-direction /// rule, so this isolates the sequence-1 rule specifically rather than incidentally also tripping I3. TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealAtNonUnitSequenceSpliced) @@ -224,16 +224,16 @@ TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealAtNonUnitSequenceSpliced) txn.ops.push_back(namespaceBirthOp()); const String bytes = encodeRefLogTxn(txn); - const String needle = R"("rs":"2")"; + const String needle = R"("txn_seq":"2")"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); String tampered = bytes; - tampered.insert(pos + needle.size(), R"(,"!pse":"1","!pss":"1")"); + tampered.insert(pos + needle.size(), R"(,"!prev_epoch":"1","!prev_seq":"1")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); } -/// Well-formedness (review finding M2): a zero component inside `prev_epoch_seal` is rejected the +/// Well-formedness: a zero component inside `prev_epoch_seal` is rejected the /// same way a zero component in the primary `txn_id` is (`checkRefTxnIdNonzero`, shared code path). TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealWithZeroWriterEpoch) { @@ -255,8 +255,8 @@ TEST(CASRefEpochSealFormat, EncodeRejectsPrevEpochSealWithZeroRefSequence) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { encodeRefLogTxn(txn); }); } -/// Decode-side splice: `prev_epoch_seal` present as only one of its two wire fields ("!pse" without -/// "!pss") -- a shape only reachable via corrupted bytes, since the encoder always writes both +/// Decode-side splice: `prev_epoch_seal` present as only one of its two wire fields ("!prev_epoch" without +/// "!prev_seq") -- a shape only reachable via corrupted bytes, since the encoder always writes both /// together. Boundary-plus-one for the additive-field decode contract (Constraint 7). TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealMissingPssComponent) { @@ -267,7 +267,7 @@ TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealMissingPssComponent) txn.ops.push_back(epochSealOp()); const String bytes = encodeRefLogTxn(txn); - const String needle = R"(,"!pss":"9")"; + const String needle = R"(,"!prev_seq":"9")"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); String tampered = bytes; @@ -276,7 +276,7 @@ TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealMissingPssComponent) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); } -/// Chain direction (review finding I3): a seal closing epoch E always has id `{E, T+1}`, and the +/// Chain direction: a seal closing epoch E always has id `{E, T+1}`, and the /// sequence-1 transaction in the next numeric epoch must name it. This remains context-free (a /// property of one transaction), so it belongs in the structural half; Tasks 2/6 walk this pointer /// backwards over untrusted decoded bodies and must not have to re-derive the rule themselves. @@ -326,16 +326,16 @@ TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealSkippingImmediateEpochSpli txn.ops.push_back(namespaceBirthOp()); const String bytes = encodeRefLogTxn(txn); - const String needle = R"("rs":"1")"; + const String needle = R"("txn_seq":"1")"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); String tampered = bytes; - tampered.insert(pos + needle.size(), R"(,"!pse":"3","!pss":"1")"); + tampered.insert(pos + needle.size(), R"(,"!prev_epoch":"3","!prev_seq":"1")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); } -/// Decode-side pin for the chain-direction rule (review finding I3): the encoder's own check would +/// Decode-side pin for the chain-direction rule: the encoder's own check would /// refuse to produce this shape (the two Encode* tests above pin that direction), so a splice into an /// otherwise-valid sequence-1 body proves decode re-derives the rule independently. TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealPointingAtSameOrFutureEpochSpliced) @@ -346,11 +346,11 @@ TEST(CASRefEpochSealFormat, DecodeRejectsPrevEpochSealPointingAtSameOrFutureEpoc txn.ops.push_back(namespaceBirthOp()); const String bytes = encodeRefLogTxn(txn); /// valid: sequence 1, no prev_epoch_seal - const String needle = R"("rs":"1")"; + const String needle = R"("txn_seq":"1")"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); String tampered = bytes; - tampered.insert(pos + needle.size(), R"(,"!pse":"5","!pss":"1")"); /// self-pointer + tampered.insert(pos + needle.size(), R"(,"!prev_epoch":"5","!prev_seq":"1")"); /// self-pointer expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(tampered, txn.ns, txn.txn_id); }); } @@ -382,7 +382,7 @@ TEST(CASRefEpochSealFormat, ContextualRejectsPrevEpochSealWhenForbidden) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { validateEpochSealGrammarContextual(txn, /*life_epoch=*/5); }); } -/// codex r2 finding 2: "genesis" is per-namespace. A namespace first born at global epoch 5 (not +/// "Genesis" is per-namespace. A namespace first born at global epoch 5 (not /// epoch 1) appends {5, 1} with NO prev_epoch_seal -- that IS its genesis, not a transition. TEST(CASRefEpochSealFormat, ContextualAllowsGenesisBirthAboveEpochOneWithoutPrevEpochSeal) { @@ -393,7 +393,7 @@ TEST(CASRefEpochSealFormat, ContextualAllowsGenesisBirthAboveEpochOneWithoutPrev EXPECT_NO_THROW(validateEpochSealGrammarContextual(txn, /*life_epoch=*/5)); } -/// Review finding I2: the `ref_sequence != 1` early return is load-bearing for Task 4's encode call +/// The `ref_sequence != 1` early return is load-bearing for the encode call /// site, which calls this on every txn it mints, including ordinary sequence->=2 transactions in a /// post-transition epoch that legitimately carry no `prev_epoch_seal`. Pinned on both sides of the /// life_epoch relation to prove the early return fires regardless of it. @@ -418,16 +418,16 @@ TEST(CASRefEpochSealFormat, ContextualPassesThroughNonSequenceOneAtOrBelowLifeEp } /// =================================================================================== -/// Criticality of the prev_epoch_seal wire fields (review finding M4) +/// Criticality of the `prev_epoch_seal` wire fields /// =================================================================================== -/// `!pse`/`!pss` are `!`-prefixed CRITICAL keys: `prev_epoch_seal` is INV-2 chain evidence, and a +/// `!prev_epoch`/`!prev_seq` are `!`-prefixed CRITICAL keys: `prev_epoch_seal` is INV-2 chain evidence, and a /// build that silently dropped it would still pass the structural grammar (absent field => no check) /// while losing the chain link. Proven here by splicing in a DIFFERENT, genuinely-unrecognized -/// `!`-key (simulating a future critical field this build predates) rather than `!pse`/`!pss` +/// `!`-key (simulating a future critical field this build predates) rather than `!prev_epoch`/`!prev_seq` /// themselves, which this build DOES recognize: `JsonObjectReader::skipUnknown` rejects any /// unrecognized `!`-prefixed key with `UNKNOWN_FORMAT_VERSION` (never a silent skip), so this pins -/// the general mechanism the meta-line reader relies on to keep `!pse`/`!pss` safe against a decoder +/// the general mechanism the meta-line reader relies on to keep `!prev_epoch`/`!prev_seq` safe against a decoder /// that doesn't (yet, or anymore) understand them. TEST(CASRefEpochSealFormat, DecodeRejectsUnknownCriticalKeyInMetaLine) { @@ -437,7 +437,7 @@ TEST(CASRefEpochSealFormat, DecodeRejectsUnknownCriticalKeyInMetaLine) txn.ops.push_back(namespaceBirthOp()); const String bytes = encodeRefLogTxn(txn); - const String needle = R"("rs":"1")"; + const String needle = R"("txn_seq":"1")"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); String tampered = bytes; @@ -471,6 +471,8 @@ TEST(CASRefEpochSealFormat, DecodeRejectsUnknownOpWordRegressionGuard) /// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) /// =================================================================================== +static const DB::Cas::tests::BatteryCoverageRegistrar battery_covers_RefLog_seal{DB::Cas::FormatId::RefLog}; + TEST(CASRefEpochSealFormat, FormatBatteryEpochSeal) { RefLogTxn txn; @@ -484,8 +486,8 @@ TEST(CASRefEpochSealFormat, FormatBatteryEpochSeal) runFormatBattery({FormatId::RefLog, [txn] { return sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); }, [ns, id](std::string_view s) { decodeRefLogTxn(openObject(FormatId::RefLog, s), ns, id); }, - "{\"type\":\"cas_ref_log\",\"v\":10}\n" - "{\"ns\":\"ns\",\"we\":\"3\",\"rs\":\"1\",\"!pse\":\"2\",\"!pss\":\"9\"}\n" + "{\"type\":\"cas_ref_log\",\"v\":1}\n" + "{\"namespace\":\"ns\",\"txn_epoch\":\"3\",\"txn_seq\":\"1\",\"!prev_epoch\":\"2\",\"!prev_seq\":\"9\"}\n" "{\"op\":\"epoch_seal\"}\n" "{\"n\":1}\n"}); } diff --git a/src/Disks/tests/gtest_cas_ref_gc.cpp b/src/Disks/tests/gtest_cas_ref_gc.cpp index 97d637982af6..c76f68d0489b 100644 --- a/src/Disks/tests/gtest_cas_ref_gc.cpp +++ b/src/Disks/tests/gtest_cas_ref_gc.cpp @@ -11,6 +11,7 @@ #include +#include #include /// Task 12 required GC tests over the snapshot+log ref model (spec 2026-07-11-cas-ref-table-snapshot-log-design). @@ -24,6 +25,7 @@ using namespace DB::Cas::tests; namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; +extern const int NOT_IMPLEMENTED; } namespace ProfileEvents @@ -75,7 +77,7 @@ size_t runToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) s->renewWatermarkOnce(); const bool no_work = rep.candidates == 0 && rep.deleted == 0 && rep.absent == 0 && rep.replaced == 0 && rep.spared == 0; - if (no_work && !anyCondemnedInSeal(s->backend(), s->layout())) + if (no_work && !anyCondemnedInSeal(*s->poolBackendPtr(), s->layout())) break; } return rounds; @@ -83,7 +85,8 @@ size_t runToFixpoint(const PoolPtr & s, Gc & gc, size_t max_rounds = 64) bool blobPresent(Backend & b, const Layout & layout, const UInt128 & hash) { - return b.head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)})).exists; + OperationForTest op(b); + return (*op).head(layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hash)}), Retry::once()).has_value(); } /// Denies ONCE the single round-commit `gc/state` CAS that advances `snap_generation` (the losing @@ -91,12 +94,16 @@ bool blobPresent(Backend & b, const Layout & layout, const UInt128 & hash) class DeposeRoundCommitBackend : public InMemoryBackend { public: - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// The fault sits on the WRITE PRIMITIVE, not the legacy `casPut` verb: `Gc::runRegularRound` + /// speaks the primitive directly, and `casPut`'s forwarding is one-way -- overriding it here would + /// intercept nothing. + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (arm && key == "p/gc/state") { - const auto stored = get(key); + const auto stored = read(key, access); const uint64_t stored_gen = stored ? decodeGcState(stored->bytes).snap_generation : 0; if (decodeGcState(bytes).snap_generation > stored_gen) { @@ -105,7 +112,7 @@ class DeposeRoundCommitBackend : public InMemoryBackend "test-injected: round-commit gc/state CAS denied (losing leader deposed mid-round)"); } } - return InMemoryBackend::casPut(key, bytes, expected, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } bool arm = false; }; @@ -119,48 +126,148 @@ class RefCleanupAuthorityRaceBackend : public CountingBackend { Catalog, GcFence, + CatalogRebirth, }; enum class Timing : uint8_t { - BeforeFirstDelete, AfterFirstDelete, + DuringChunk, }; void arm( Authority authority_, Timing timing_, const Layout & layout, const String & first_cleanup_key_) { - authority = authority_; - timing = timing_; + arm(authority_, timing_, layout, first_cleanup_key_, std::nullopt); + } + + void arm( + Authority authority_, Timing timing_, const Layout & layout, + const String & first_cleanup_key_, const RootNamespace & reborn_ns_) + { + arm(authority_, timing_, layout, first_cleanup_key_, std::optional{reborn_ns_}); + } + + /// "Before the first chunk" has no backend-request seam of its own to hang off: `cleanupRefObjects` + /// issues no `HEAD`, and the chunk's own catalog/`gc/state` reads only happen ONCE, back to back, + /// immediately before the delete they license -- by the time either is observable from a backend + /// override, the chunk's catalog snapshot is already cached in `authorityHolds`'s local, and moving + /// the authority no longer changes what THIS chunk decides. The window that must be hit instead is + /// the round's own hot-scan catalog cut: `Gc::setPostHotScanCatalogReadHookForTest` (`CasGc.h`) + /// fires the instant that cut is taken, before the round -- and later `authorityHolds` -- does + /// anything else with it, so a move landed there is exactly "before the first chunk starts" and the + /// chunk's later fresh reads observe it. + /// + /// `Authority::Catalog` moves the catalog's token directly, right there in the hook: the round's + /// own `round_commit` CAS (phase 13) never touches the catalog, so nothing downstream collides. + /// `Authority::GcFence` cannot do the same for `gc/state`: bumping its lease THERE lands strictly + /// BEFORE `round_commit`'s own `gc/state` replace (which still holds the etag from lease adoption, + /// phase 1), so that replace loses its own CAS and the round throws before `cleanupRefObjects` + /// (phase 17) ever runs -- the "nothing deleted" assertions would pass vacuously, not because + /// cleanup refused. Instead, the hook only ARMS `armGcFenceMoveOnAuthorityHoldsRevalidationForTest`: the + /// actual lease bump is deferred to a LATER read of `gc/state`. Not the next one -- namespace + /// janitor / orphan-sweep bookkeeping between `round_commit` and `cleanupRefObjects` also touches + /// the catalog and `gc/state`, just never the two BACK TO BACK the way `authorityHolds` does + /// (catalog, then `gc/state`, nothing in between): that adjacency is the one place in a round only + /// `authorityHolds`'s own revalidation produces, so gating on it -- rather than on the catalog key + /// alone -- is what actually lands the move inside that SAME call, well after `round_commit`. + static void moveRefCleanupAuthorityBeforeFirstChunk(Authority authority_, CasOperation & op, const Layout & layout) + { + if (authority_ == Authority::GcFence) + throw std::logic_error( + "moveRefCleanupAuthorityBeforeFirstChunk is for Authority::Catalog/CatalogRebirth only -- " + "use armGcFenceMoveOnAuthorityHoldsRevalidationForTest for Authority::GcFence"); + /// Same-content rewrite: only the catalog's TOKEN moves (mints a fresh etag), never its + /// parsed content -- the pure "someone else touched this row" race `Authority::Catalog` + /// models, as opposed to `Authority::CatalogRebirth`'s actual incarnation bump. + const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, layout); + op.replace(layout.refCatalogKey(), encodeRefCatalog(snap.catalog), *snap.etag, Retry::standard()); + } + + /// Arms the seam `read` below fires on: see the doc comment above. Called from the test body + /// before the round (and so before any read-ahead worker exists), but still under the mutex, so + /// this method and `read`'s critical sections never race even under future reordering. + void armGcFenceMoveOnAuthorityHoldsRevalidationForTest(const Layout & layout) + { catalog_key = layout.refCatalogKey(); gc_state_key = layout.gcStateKey(); - first_cleanup_key = first_cleanup_key_; - armed = true; + std::lock_guard lock(seam_mutex); + catalog_seam_armed = true; } - HeadResult head(const String & key) override + /// `read` also runs on the GC read-ahead pool's threads (`CasGcReadAhead.cpp` schedules + /// `CasOperation::read` there; `gc_read_concurrency` defaults to 16), concurrently with the round + /// thread's own reads -- `catalog_seam_armed` and `last_control_key_read` below are shared mutable + /// state a pool thread's read can land between `authorityHolds`'s two reads, so both are read AND + /// written only under `seam_mutex`. `last_control_key_read` tracks only the catalog and `gc/state` + /// keys, never any other key a read-ahead worker fetches: those workers never touch either control + /// key (they fetch ref-log/manifest bodies), so an unrelated concurrent read can never perturb the + /// adjacency signal even though it runs lock-free between this method's two critical sections. + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - HeadResult result = CountingBackend::head(key); - if (armed && timing == Timing::BeforeFirstDelete && key == first_cleanup_key) - moveAuthority(); - return result; + bool fires_here = false; + { + std::lock_guard lock(seam_mutex); + fires_here = catalog_seam_armed && key == gc_state_key && last_control_key_read == catalog_key; + if (fires_here) + catalog_seam_armed = false; + if (key == catalog_key || key == gc_state_key) + last_control_key_read = key; + } + if (fires_here) + { + const auto current = CountingBackend::read(key, access); + if (!current) + throw std::runtime_error("test-injected cleanup authority object is absent"); + GcState moved = decodeGcState(current->bytes); + ++moved.lease.seq; + if (!write(key, encodeGcState(moved), current->value, access).has_value()) + throw std::runtime_error("test-injected cleanup authority move lost its CAS"); + return CountingBackend::read(key, access); /// the FRESH, post-move bytes, for THIS read + } + return CountingBackend::read(key, access); } - DeleteOutcome deleteExact(const String & key, const Token & token) override + /// The `AfterFirstDelete` seam moves the authority once the first chunk's batch delete has + /// landed (so that chunk keeps whatever it already observed and only the NEXT chunk's + /// revalidation refuses); `DuringChunk` moves it after the chunk's revalidation but before its + /// batch delete lands, so the chunk in flight still completes under the authority it observed. + void removeManyWriteOnce(const std::vector & keys, DB::Cas::TransportAccess & access) override { - DeleteOutcome result = CountingBackend::deleteExact(key, token); - if (armed && timing == Timing::AfterFirstDelete && key == first_cleanup_key) - moveAuthority(); - return result; + const bool names_first = std::any_of(keys.begin(), keys.end(), + [&](const DB::Cas::WriteOnceKey & key) { return key.str() == first_cleanup_key; }); + if (armed && timing == Timing::DuringChunk && names_first) + moveAuthority(access); /// after the revalidation, before the deletes land + CountingBackend::removeManyWriteOnce(keys, access); + if (armed && timing == Timing::AfterFirstDelete && names_first) + moveAuthority(access); } private: - void moveAuthority() + void arm( + Authority authority_, Timing timing_, const Layout & layout, + const String & first_cleanup_key_, std::optional reborn_ns_) + { + authority = authority_; + timing = timing_; + catalog_key = layout.refCatalogKey(); + gc_state_key = layout.gcStateKey(); + first_cleanup_key = first_cleanup_key_; + reborn_ns = std::move(reborn_ns_); + layout_for_rebirth_seed = &layout; + armed = true; + } + + /// `access` is the token the caller's own primitive override already holds for its in-flight + /// request; reused here for this method's extra read+write rather than minting a new CasRequests, + /// exactly as `Backend::probeSentinelRaw`'s default implementation reuses one `access` across its + /// own head-then-more sequence. + void moveAuthority(TransportAccess & access) { armed = false; - const String & key = authority == Authority::Catalog ? catalog_key : gc_state_key; - const auto got = CountingBackend::get(key); + const String & key = authority == Authority::GcFence ? gc_state_key : catalog_key; + const auto got = read(key, access); if (!got) throw std::runtime_error("test-injected cleanup authority object is absent"); @@ -171,16 +278,73 @@ class RefCleanupAuthorityRaceBackend : public CountingBackend ++moved.lease.seq; bytes = encodeGcState(moved); } - if (CountingBackend::casPut(key, bytes, got->token).outcome != CasOutcome::Committed) + UInt128 reborn_incarnation = 0; + if (authority == Authority::CatalogRebirth) + { + RefCatalog catalog = decodeRefCatalog(bytes); + for (CatalogEntry & entry : catalog.entries) + if (reborn_ns && entry.ns == *reborn_ns) + { + entry.incarnation = entry.incarnation + 1; + reborn_incarnation = entry.incarnation; + } + bytes = encodeRefCatalog(catalog); + } + if (!write(key, bytes, got->value, access).has_value()) throw std::runtime_error("test-injected cleanup authority move lost its CAS"); + + /// Give the reborn life SOMETHING of its own, landed right after its catalog row exists (any + /// earlier and an "unknown incarnation" sweep elsewhere in the SAME round can claim it, since + /// no catalog entry yet names that incarnation) -- so the "reborn life untouched" assertions + /// below test something real instead of an empty listing. + if (authority == Authority::CatalogRebirth && reborn_ns && reborn_incarnation != 0 && layout_for_rebirth_seed) + { + const Layout & layout = *layout_for_rebirth_seed; + const NamespaceLifeId reborn_life = NamespaceLifeId::fromCatalogEntry(*reborn_ns, reborn_incarnation); + const RefTxnId reborn_log_id{1, 1}; + const RefLogTxn reborn_birth{ + .ns = reborn_ns->string(), .txn_id = reborn_log_id, .ops = {namespaceBirthOp()}, + .prev_epoch_seal = std::nullopt}; + if (!write(layout.refLogKey(reborn_life, reborn_log_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(reborn_birth)), std::nullopt, access).has_value()) + throw std::runtime_error("test-injected reborn-life log seed lost its CAS"); + const RefTableSnapshot reborn_snap = minimalLiveSnapshot(reborn_ns->string(), reborn_log_id); + if (!write(layout.refSnapshotKey(reborn_life, reborn_log_id), + sealObject(FormatId::RefSnapshot, encodeRefTableSnapshot(reborn_snap)), std::nullopt, access).has_value()) + throw std::runtime_error("test-injected reborn-life snapshot seed lost its CAS"); + /// A checkpoint too, naming the seeded log/snapshot: without one, the NEXT round's recovery + /// grounding for this namespace finds "no usable checkpoint", which SUPPRESSES that round's + /// destructive work ENTIRELY (every namespace, not just this one) -- a test relying on the + /// old cohort surviving round 2 would then be observing a no-op round, not the plan moving + /// to the reborn life. `writeRecoverableCkptForRawFixture` resolves its own fresh catalog + /// read, which already sees the incarnation bump the write just above landed. + writeRecoverableCkptForRawFixture(*this, layout, *reborn_ns, RefCkpt{ + .life_epoch = 1, + .committed_through = reborn_log_id, + .checkpoint_snapshot_id = reborn_log_id, + .last_epoch_seal = std::nullopt, + }); + } } Authority authority = Authority::Catalog; - Timing timing = Timing::BeforeFirstDelete; + Timing timing = Timing::AfterFirstDelete; String catalog_key; String gc_state_key; String first_cleanup_key; + std::optional reborn_ns; + const Layout * layout_for_rebirth_seed = nullptr; bool armed = false; + /// Guards both members below: `read` runs concurrently on the GC read-ahead pool's threads, see + /// the doc comment on `read` itself. + std::mutex seam_mutex; + /// Independent of `armed`/`timing`/`authority` above: `armGcFenceMoveOnAuthorityHoldsRevalidationForTest` + /// arms this, and the `read` override consumes it once. + bool catalog_seam_armed = false; + /// The most recent CONTROL key (catalog or `gc/state`) read -- every other key a read-ahead + /// worker reads is ignored, so `read` can recognize the catalog-then-`gc/state` ADJACENCY + /// `authorityHolds` alone produces -- see the doc comment above `moveRefCleanupAuthorityBeforeFirstChunk`. + String last_control_key_read; }; struct RefCleanupFixture @@ -358,7 +522,8 @@ TEST(CASRefGc, LosingGenerationCommitAdoptsNothingDeletesNothing) Gc gc(store, kGc); gc.runRegularRound(); /// round 1: folds the +1 and adopts it cleanly store->renewWatermarkOnce(); - const auto adopted = decodeGcState(backend->get(layout.gcStateKey())->bytes); + OperationForTest raw_op(*backend); + const auto adopted = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); ASSERT_GT(adopted.snap_generation, 0u); /// Drop the ref, then run the round whose commit is DENIED (losing leader). @@ -368,7 +533,7 @@ TEST(CASRefGc, LosingGenerationCommitAdoptsNothingDeletesNothing) backend->arm = false; /// The deposed round adopted NOTHING: the durable pointers are unchanged... - const auto after = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const auto after = decodeGcState((*raw_op).read(layout.gcStateKey(), Retry::once())->bytes); EXPECT_EQ(after.snap_generation, adopted.snap_generation) << "a denied round-commit CAS must not advance the adopted generation"; EXPECT_EQ(after.snap_attempt, adopted.snap_attempt); @@ -414,22 +579,23 @@ TEST(CASRefGc, RefObjectCleanupRetainsCheckpointNamedTriple) const String log_v2_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, v2}); const String old_snap_key = layout.refSnapshotKey(fixture::fixtureLife(ns), RefTxnId{1, v1}); const String new_snap_key = layout.refSnapshotKey(fixture::fixtureLife(ns), RefTxnId{1, v2}); - ASSERT_TRUE(backend->head(log_v1_key).exists); - ASSERT_TRUE(backend->head(log_v2_key).exists); - ASSERT_TRUE(backend->head(old_snap_key).exists); + OperationForTest raw_op(*backend); + ASSERT_TRUE((*raw_op).head(log_v1_key, Retry::once()).has_value()); + ASSERT_TRUE((*raw_op).head(log_v2_key, Retry::once()).has_value()); + ASSERT_TRUE((*raw_op).head(old_snap_key, Retry::once()).has_value()); Gc gc(store, kGc); runToFixpoint(store, gc); /// folds v1,v2 (cursor -> v2) then cleans covered ref objects post-CAS /// The old log lies below both the durable cursor and the validated checkpoint base => DELETED. - EXPECT_FALSE(backend->head(log_v1_key).exists) + EXPECT_FALSE((*raw_op).head(log_v1_key, Retry::once()).has_value()) << "a log below the checkpoint-named snapshot base and durable cursor must be deleted"; /// The same-id ordinary log is part of recovery's triple and must survive. - EXPECT_TRUE(backend->head(log_v2_key).exists) + EXPECT_TRUE((*raw_op).head(log_v2_key, Retry::once()).has_value()) << "the checkpoint-named non-seal log must survive with its snapshot"; /// The older snapshot is deleted; the checkpoint-named snapshot is retained. - EXPECT_FALSE(backend->head(old_snap_key).exists) << "an older snapshot must be deleted"; - EXPECT_TRUE(backend->head(new_snap_key).exists) << "the checkpoint-named snapshot must be retained"; + EXPECT_FALSE((*raw_op).head(old_snap_key, Retry::once()).has_value()) << "an older snapshot must be deleted"; + EXPECT_TRUE((*raw_op).head(new_snap_key, Retry::once()).has_value()) << "the checkpoint-named snapshot must be retained"; } /// `cleanupRefObjects`'s per-round cap. Five deletable logs share one @@ -477,11 +643,12 @@ TEST(CASRefGc, RefObjectCleanupRespectsRoundBudgetAndConvergesAcrossRounds) deletable_log_keys.push_back(layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, static_cast(i)})); Gc gc(store, kGc); + OperationForTest raw_op(*backend); auto countSurviving = [&] { size_t n = 0; for (const String & k : deletable_log_keys) - if (backend->head(k).exists) + if ((*raw_op).head(k, Retry::once()).has_value()) ++n; return n; }; @@ -549,40 +716,53 @@ TEST(CASRefGc, RefObjectCleanupRetainsCheckpointPredecessorSealProof) Gc gc(store, kGc); runToFixpoint(store, gc); - EXPECT_TRUE(backend->head(layout.refLogKey(life, seal_id)).exists) + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + EXPECT_TRUE(op.head(layout.refLogKey(life, seal_id), Retry::once()).has_value()) << "cleanup must retain the predecessor seal that proves the checkpoint base's epoch transition"; - const CasRefCatalog::Snapshot cut = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(op, layout); const auto entry = std::find_if(cut.catalog.entries.begin(), cut.catalog.entries.end(), [&](const CatalogEntry & candidate) { return candidate.ns == ns; }); ASSERT_NE(entry, cut.catalog.entries.end()); - const std::optional checkpoint = readCkpt(*backend, layout, life); + const std::optional checkpoint = readCkpt(op, layout, life); ASSERT_TRUE(checkpoint); - EXPECT_NO_THROW((void)recoverRefTableDetailedFromAuthority(*backend, layout, *entry, checkpoint->ckpt)); + EXPECT_NO_THROW((void)recoverRefTableDetailedFromAuthority(op, layout, *entry, checkpoint->ckpt)); } -TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBeforeFirstDeleteRefusesEveryRefObjectDelete) +TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBeforeFirstChunkRefusesEveryRefObjectDelete) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); const Layout & layout = store->layout(); const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); - backend->arm( - RefCleanupAuthorityRaceBackend::Authority::Catalog, - RefCleanupAuthorityRaceBackend::Timing::BeforeFirstDelete, layout, keys.first_log_key); Gc gc(store, kGc); + /// Lands the move in the exact window between the round's own hot-scan catalog cut (what + /// `authorityHolds` later compares `folded.catalog_cut` against) and everything after it -- "the + /// catalog moved before the first chunk starts", the case spec §D's test (1) names. + OperationForTest race_op(*backend); + bool hook_fired = false; + gc.setPostHotScanCatalogReadHookForTest([&] + { + hook_fired = true; + RefCleanupAuthorityRaceBackend::moveRefCleanupAuthorityBeforeFirstChunk( + RefCleanupAuthorityRaceBackend::Authority::Catalog, *race_op, layout); + }); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_TRUE(hook_fired) << "the race hook never fired -- this test proves nothing about the race"; - EXPECT_TRUE(backend->head(keys.first_log_key).exists); - EXPECT_TRUE(backend->head(keys.second_log_key).exists); + OperationForTest head_op(*backend); + EXPECT_TRUE((*head_op).head(keys.first_log_key, Retry::once()).has_value()); + EXPECT_TRUE((*head_op).head(keys.second_log_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(keys.first_log_key), 0u); EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); } -TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBetweenKeysAllowsFirstAndRefusesSecondDelete) +TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBetweenChunksAllowsFirstAndRefusesSecondDelete) { auto backend = std::make_shared(); - auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 1, .gc_fold_max_defer_rounds = 0}); const Layout & layout = store->layout(); const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); backend->arm( @@ -592,35 +772,49 @@ TEST(CASRefGcCleanupAuthority, CatalogTokenMoveBetweenKeysAllowsFirstAndRefusesS Gc gc(store, kGc); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - EXPECT_FALSE(backend->head(keys.first_log_key).exists); - EXPECT_TRUE(backend->head(keys.second_log_key).exists); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(keys.first_log_key, Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(keys.first_log_key), 1u); EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); } -TEST(CASRefGcCleanupAuthority, GcFenceMoveBeforeFirstDeleteRefusesEveryRefObjectDelete) +TEST(CASRefGcCleanupAuthority, GcFenceMoveBeforeFirstChunkRefusesEveryRefObjectDelete) { auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); const Layout & layout = store->layout(); const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); - backend->arm( - RefCleanupAuthorityRaceBackend::Authority::GcFence, - RefCleanupAuthorityRaceBackend::Timing::BeforeFirstDelete, layout, keys.first_log_key); Gc gc(store, kGc); + /// Bumping `gc/state`'s lease directly from the hot-scan hook (as `Authority::Catalog` bumps the + /// catalog above) would land BEFORE the round's own `round_commit` CAS (phase 13), which still + /// holds the etag from lease adoption (phase 1) -- that CAS would then lose and the round would + /// throw before `cleanupRefObjects` (phase 17) ever runs, so "nothing deleted" would hold + /// vacuously. Instead, only ARM the seam here: the actual bump happens on `authorityHolds`'s own + /// `gc/state` read (phase 17, long after `round_commit` landed) -- see the class doc comment above + /// `moveRefCleanupAuthorityBeforeFirstChunk`. + bool hook_fired = false; + gc.setPostHotScanCatalogReadHookForTest([&] + { + hook_fired = true; + backend->armGcFenceMoveOnAuthorityHoldsRevalidationForTest(layout); + }); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_TRUE(hook_fired) << "the race hook never fired -- this test proves nothing about the race"; - EXPECT_TRUE(backend->head(keys.first_log_key).exists); - EXPECT_TRUE(backend->head(keys.second_log_key).exists); + OperationForTest raw_op(*backend); + EXPECT_TRUE((*raw_op).head(keys.first_log_key, Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()); EXPECT_EQ(backend->deleteCount(keys.first_log_key), 0u); EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); } -TEST(CASRefGcCleanupAuthority, GcFenceMoveBetweenKeysAllowsFirstAndRefusesSecondDelete) +TEST(CASRefGcCleanupAuthority, GcFenceMoveBetweenChunksAllowsFirstAndRefusesSecondDelete) { auto backend = std::make_shared(); - auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 1, .gc_fold_max_defer_rounds = 0}); const Layout & layout = store->layout(); const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); backend->arm( @@ -630,12 +824,246 @@ TEST(CASRefGcCleanupAuthority, GcFenceMoveBetweenKeysAllowsFirstAndRefusesSecond Gc gc(store, kGc); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - EXPECT_FALSE(backend->head(keys.first_log_key).exists); - EXPECT_TRUE(backend->head(keys.second_log_key).exists); + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(keys.first_log_key, Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()); + EXPECT_EQ(backend->deleteCount(keys.first_log_key), 1u); + EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); +} + +TEST(CASRefGcCleanupAuthority, LeaseMoveDuringAChunkLetsTheChunkCompleteAndNothingElse) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 1, .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, RootNamespace{"00/aa@cas@"}); + backend->arm(RefCleanupAuthorityRaceBackend::Authority::GcFence, + RefCleanupAuthorityRaceBackend::Timing::DuringChunk, layout, keys.first_log_key); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(keys.first_log_key, Retry::once()).has_value()) << "the chunk in flight completes"; + EXPECT_TRUE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()) << "the next chunk's revalidation refuses"; EXPECT_EQ(backend->deleteCount(keys.first_log_key), 1u); EXPECT_EQ(backend->deleteCount(keys.second_log_key), 0u); } +TEST(CASRefGcCleanupAuthority, RebirthDuringAChunkDeletesOnlyTheOldLifesKeys) +{ + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", + .gc_bulk_delete_chunk_keys = 1, .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + const RefCleanupFixture keys = seedTwoCoveredLogs(*backend, layout, ns); + backend->arm(RefCleanupAuthorityRaceBackend::Authority::CatalogRebirth, + RefCleanupAuthorityRaceBackend::Timing::DuringChunk, layout, keys.first_log_key, ns); + + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + OperationForTest raw_op(*backend); + EXPECT_FALSE((*raw_op).head(keys.first_log_key, Retry::once()).has_value()); + EXPECT_TRUE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()); + /// Whatever the reborn life owns is untouched: its keys carry a different life id and were never + /// in the cohort. Every key under the new life's stream prefix is still present -- `moveAuthority`'s + /// `CatalogRebirth` branch seeds the reborn life's own `_log` and `_snap` right after the catalog + /// rewrite lands (see its doc comment: any earlier and an "unknown incarnation" sweep elsewhere in + /// the SAME round could claim them, since no catalog entry names that incarnation yet). + const CasRefCatalog::Snapshot cut = CasRefCatalog::read(*raw_op, layout); + const auto entry = std::find_if(cut.catalog.entries.begin(), cut.catalog.entries.end(), + [&](const CatalogEntry & e) { return e.ns == ns; }); + ASSERT_NE(entry, cut.catalog.entries.end()); + const NamespaceLifeId reborn = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); + ListPage page = (*raw_op).list(layout.namespaceStreamPrefix(reborn), "", 1000, Retry::once()); + ASSERT_FALSE(page.keys.empty()) << "the seeded new-life objects must be listed, or this test proves nothing"; + for (const ListedKey & listed : page.keys) + { + EXPECT_EQ(backend->deleteCount(listed.key), 0u) << listed.key; + EXPECT_TRUE((*raw_op).head(listed.key, Retry::once()).has_value()) << listed.key; + } + + /// The next round revalidates against the NEW catalog row: the old (dead) life is no longer named + /// by any entry `cleanupRefObjects` walks, so its plan/cohort revalidation -- built fresh from the + /// catalog entry each round -- has nothing of the old life's to touch. What actually happens to the old cohort's + /// second key, once the round runs unsuppressed, is that the NAMESPACE JANITOR + /// (`CasNamespaceJanitor.cpp:131`, `catalog_cut.life_index.resolve(*life_id)` failing for a + /// physical life the catalog no longer names) reclaims it as leaked dead-life debris -- exactly + /// spec §D's own words: "a moved catalog row means either a dropped life, whose keys the + /// namespace janitor deletes anyway, or a reborn one". So the key does NOT survive; it survives + /// past `cleanupRefObjects` specifically, then is reclaimed by a wholly separate, pre-existing + /// mechanism this task never touches. Attribute the delete precisely rather than asserting + /// "survives" and being right for an unrelated reason: capture the `ref_object_cleanup` phase's + /// `suppressed` metric (to confirm the round actually ran, not merely suppressed everything, which + /// an unusable checkpoint ANYWHERE would do -- see the checkpoint seeded above) and the + /// `namespace_cleanup` phase's `janitor_deleted` metric, and independently confirm `cleanupRefObjects` + /// itself deleted nothing this round via the GLOBAL `CASRefCleanupObjectsDeleted` counter (the + /// `GcPhaseRecord::profile_events` delta is unavailable here: it needs a `CurrentThread` with an + /// attached `ThreadStatus`, which a bare gtest thread does not have). + std::optional ref_cleanup_suppressed; + std::optional janitor_deleted; + gc.setPhaseSink([&](const GcPhaseRecord & rec) + { + if (rec.phase == "ref_object_cleanup") + if (const auto it = rec.metrics.find("suppressed"); it != rec.metrics.end()) + ref_cleanup_suppressed = it->second; + if (rec.phase == "namespace_cleanup") + if (const auto it = rec.metrics.find("janitor_deleted"); it != rec.metrics.end()) + janitor_deleted = it->second; + }); + using ProfileEvents::global_counters; + const auto ref_cleanup_deleted_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + ASSERT_TRUE(ref_cleanup_suppressed.has_value()) << "the ref_object_cleanup phase row never fired"; + EXPECT_EQ(*ref_cleanup_suppressed, 0u) + << "round two must actually run destructive work, not merely leave everything alone because it was suppressed"; + EXPECT_EQ(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(), ref_cleanup_deleted_before) + << "cleanupRefObjects' own plan/cohort must delete NOTHING this round: the old life is not in it " + "(no catalog entry names it) and the reborn life's own checkpoint-named log is not yet deletable"; + ASSERT_TRUE(janitor_deleted.has_value()) << "the namespace_cleanup phase row never fired"; + EXPECT_GE(*janitor_deleted, 1u) + << "the old cohort's second key is expected to be reclaimed by the namespace janitor, not to survive"; + EXPECT_FALSE((*raw_op).head(keys.second_log_key, Retry::once()).has_value()) + << "the old cohort's second key is dead-life debris once its life no longer resolves in the " + "catalog -- the namespace janitor reclaims it, exactly as spec §D says it would"; + /// The reborn life's own objects are untouched: round two's plan, cleanup and cohort are about the + /// reborn life now, and none of what it seeded for itself is in that plan. The namespace janitor + /// leaves them alone too, since they resolve fine against the CURRENT catalog entry. + for (const ListedKey & listed : page.keys) + EXPECT_TRUE((*raw_op).head(listed.key, Retry::once()).has_value()) << listed.key; +} + +TEST(CASRefGc, RefObjectCleanupDeletesExactlyThePlannedSet) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + + /// Two committed publishes -> logs {1,1} and {1,2}. + const ManifestRef r1 = mref(1); + const ManifestRef r2 = mref(2); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "t1", std::nullopt, r1); + const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "t2", std::nullopt, r2); + + /// Two observed snapshots: an OLD one covering only v1, and the NEWEST covering v2. Both are real + /// wire-format snapshot objects (the recovery codec reads them). + RefTableSnapshot old_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v1}, + {committedRow("t1", r1)}); + RefTableSnapshot new_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v2}, + {committedRow("t1", r1), committedRow("t2", r2)}); + writeRefSnapshotRaw(*backend, layout, old_snap); + writeRefSnapshotRaw(*backend, layout, new_snap); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, v2}, + .checkpoint_snapshot_id = RefTxnId{1, v2}, + .last_epoch_seal = std::nullopt, + }); + + /// The plan the pass computes: `listing` names every log and snapshot this round's scan would + /// observe, `durable_cursor` is the fold cursor after folding both logs, `checkpoint_snapshot_id` + /// is the checkpoint-named recovery snapshot, and this fixture never crosses an epoch, so there + /// is no retained-seal proof. + const NamespaceLifeId life = fixture::fixtureLife(ns); + const RefTableListing listing{ + .logs = {RefTxnId{1, v1}, RefTxnId{1, v2}}, + .snapshots = {RefTxnId{1, v1}, RefTxnId{1, v2}}}; + const RefTxnId durable_cursor{1, v2}; + const RefTxnId checkpoint_snapshot_id{1, v2}; + const std::optional retained_log_proof = std::nullopt; + + OperationForTest op(*backend); + const RefCleanupPlan plan = planRefCleanup(listing, durable_cursor, checkpoint_snapshot_id, retained_log_proof); + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + for (const RefTxnId & id : plan.deletable_logs) + EXPECT_FALSE((*op).head(layout.refLogKey(life, id), Retry::once()).has_value()); + for (const RefTxnId & id : plan.deletable_snapshots) + EXPECT_FALSE((*op).head(layout.refSnapshotKey(life, id), Retry::once()).has_value()); + /// and every listed key NOT in the plan is present -- the chunked implementation deletes exactly + /// the set the per-key implementation would have deleted, nothing more. + const std::set deleted_logs(plan.deletable_logs.begin(), plan.deletable_logs.end()); + const std::set deleted_snapshots(plan.deletable_snapshots.begin(), plan.deletable_snapshots.end()); + for (const RefTxnId & id : listing.logs) + if (!deleted_logs.contains(id)) + EXPECT_TRUE((*op).head(layout.refLogKey(life, id), Retry::once()).has_value()) + << "log " << renderRefTxnId(id) << " not in the plan must survive"; + for (const RefTxnId & id : listing.snapshots) + if (!deleted_snapshots.contains(id)) + EXPECT_TRUE((*op).head(layout.refSnapshotKey(life, id), Retry::once()).has_value()) + << "snapshot " << renderRefTxnId(id) << " not in the plan must survive"; +} + +/// The same planned set as above, but the object storage rejects the cohort's one bulk +/// `removeManyWriteOnce` as NOT_IMPLEMENTED (a GCS-backed pool): `cleanupRefObjects`' call site falls +/// back to one admitted request per key (`removeChunkWriteOnceOrOneByOne`, CasGc.h), and the outcome -- +/// which keys are gone, and the budget/profile-event accounting -- must be identical to the plain +/// bulk-request path above. +TEST(CASRefGc, RefObjectCleanupFallsBackToOnePerKeyWhenBatchDeleteIsUnsupported) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, store->layout(), ns); + + const ManifestRef r1 = mref(1); + const ManifestRef r2 = mref(2); + writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); + writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); + const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "t1", std::nullopt, r1); + const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "t2", std::nullopt, r2); + + RefTableSnapshot old_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v1}, + {committedRow("t1", r1)}); + RefTableSnapshot new_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v2}, + {committedRow("t1", r1), committedRow("t2", r2)}); + writeRefSnapshotRaw(*backend, layout, old_snap); + writeRefSnapshotRaw(*backend, layout, new_snap); + replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{1, v2}, + .checkpoint_snapshot_id = RefTxnId{1, v2}, + .last_epoch_seal = std::nullopt, + }); + + const NamespaceLifeId life = fixture::fixtureLife(ns); + const RefTableListing listing{ + .logs = {RefTxnId{1, v1}, RefTxnId{1, v2}}, + .snapshots = {RefTxnId{1, v1}, RefTxnId{1, v2}}}; + const RefTxnId durable_cursor{1, v2}; + const RefTxnId checkpoint_snapshot_id{1, v2}; + const RefCleanupPlan plan = planRefCleanup(listing, durable_cursor, checkpoint_snapshot_id, std::nullopt); + const uint64_t cohort_size = plan.deletable_logs.size() + plan.deletable_snapshots.size(); + ASSERT_GT(cohort_size, 0u) << "the fixture must actually have something to delete for this test to prove anything"; + + /// One armed failure: the cohort's own bulk `removeManyWriteOnce` call fails as "batch delete not + /// supported"; the fallback's per-key calls that follow are not armed and succeed. + backend->failNextBulkRemoveWith(std::make_exception_ptr( + DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + const auto cleaned_before = ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + + OperationForTest op(*backend); + Gc gc(store, kGc); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + for (const RefTxnId & id : plan.deletable_logs) + EXPECT_FALSE((*op).head(layout.refLogKey(life, id), Retry::once()).has_value()); + for (const RefTxnId & id : plan.deletable_snapshots) + EXPECT_FALSE((*op).head(layout.refSnapshotKey(life, id), Retry::once()).has_value()); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load() - cleaned_before, cohort_size) + << "the budget/profile-event accounting counts objects, unaffected by the fallback"; + /// 1 failed bulk attempt + one request per key in the cohort. + EXPECT_EQ(backend->bulkRemoveCalls(), 1 + cohort_size); +} + /// Task 13 (spec §implementation-impact / §GC Budget): one fold+clean round increments every ref-intake /// observability counter -- global LIST pages (Q), log-body GETs (K), manifest-body fold GETs (H), emitted /// manifest edges, and cleaned old ref objects (D). Before/after deltas prove each site actually fires. @@ -713,25 +1141,27 @@ TEST(CASRefGc, RefSnaplogLifecycleE2E) /// The writer's compaction: a snapshot of ns_a covering its greatest log (va2), the same /// deterministic bytes the oracle recomputes. - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); const RefTableState sa = recoverRefTableDetailedAtCatalogCutForTest(*backend, layout, catalog_cut, ns_a).state; writeRefSnapshotRaw(*backend, layout, snapshotOf(sa, ns_a.string())); const NamespaceLifeId life_a = store->namespaceLife(ns_a); - const CkptSample before_snapshot_publish = *readCkpt(*backend, layout, life_a); + const CkptSample before_snapshot_publish = *readCkpt(op, layout, life_a); RefCkpt after_snapshot_publish = before_snapshot_publish.ckpt; after_snapshot_publish.checkpoint_snapshot_id = RefTxnId{1, va2}; - ASSERT_EQ(backend->casPut( - layout.refCkptKey(life_a), encodeRefCkpt(after_snapshot_publish), before_snapshot_publish.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(std::holds_alternative(op.replace( + layout.refCkptKey(life_a), encodeRefCkpt(after_snapshot_publish), + before_snapshot_publish.etag, Retry::standard()))); Gc gc(store, kGc); runToFixpoint(store, gc); /// Snapshot lifecycle: the covering snapshot is retained; the covered logs (folded + snapshot-covered) /// are cleaned; the replaced manifest's blob is reclaimed while the live blobs survive. - EXPECT_TRUE(backend->head(layout.refSnapshotKey(fixture::fixtureLife(ns_a), RefTxnId{1, va2})).exists) + EXPECT_TRUE(op.head(layout.refSnapshotKey(fixture::fixtureLife(ns_a), RefTxnId{1, va2}), Retry::once()).has_value()) << "covering snapshot retained"; - EXPECT_FALSE(backend->head(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, va1})).exists) << "covered log cleaned"; + EXPECT_FALSE(op.head(layout.refLogKey(fixture::fixtureLife(ns_a), RefTxnId{1, va1}), Retry::once()).has_value()) << "covered log cleaned"; EXPECT_FALSE(blobPresent(*backend, layout, DB::UInt128(1))) << "replaced blob reclaimed"; EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(2))) << "live blob survives"; EXPECT_TRUE(blobPresent(*backend, layout, DB::UInt128(3))) << "other table's blob survives"; @@ -760,7 +1190,10 @@ TEST(CASRefGc, MalformedRefKeyAbortsRefFoldingNoPartialDelta) /// Plant a malformed ref key under the ref prefix (a `_log` with a non-canonical id render). const NamespaceLifeId life = store->namespaceLife(ns); - backend->putIfAbsent(layout.namespaceStreamPrefix(life) + "_log/not-a-valid-txn-id", "garbage"); + { + OperationForTest seed_op(*backend); + (*seed_op).create(layout.namespaceStreamPrefix(life) + "_log/not-a-valid-txn-id", "garbage", Retry::once()); + } Gc gc(store, kGc); /// The fold's `groupRefKeys` rejects the unrecognized key and ABORTS ref folding for the round (spec @@ -801,7 +1234,8 @@ TEST(CASRefGc, NonCanonicalLifeKeyAbortsRefFoldingWithoutWedgingTheRound) /// opaque id. Only a foreign or corrupt writer can put this key here, and the pool must survive it. const String noncanonical_life = layout.casRefsPrefix() + ns.string() + "/_log/" + renderRefTxnId(RefTxnId{1, 1}) + ".zst"; - ASSERT_EQ(backend->putIfAbsent(noncanonical_life, "garbage").outcome, PutOutcome::Done); + OperationForTest raw_op(*backend); + ASSERT_TRUE(std::holds_alternative((*raw_op).create(noncanonical_life, "garbage", Retry::once()))); Gc gc(store, kGc); RoundReport rep; @@ -820,7 +1254,7 @@ TEST(CASRefGc, NonCanonicalLifeKeyAbortsRefFoldingWithoutWedgingTheRound) /// The wedge is only visible over time: the key is still there (nothing deletes it), so a second /// round meets it again. It must survive that one too. - ASSERT_TRUE(backend->head(noncanonical_life).exists) << "precondition: nothing removed the key"; + ASSERT_TRUE((*raw_op).head(noncanonical_life, Retry::once()).has_value()) << "precondition: nothing removed the key"; ASSERT_NO_THROW(gc.runRegularRound()) << "a round that dies on this key would die on it forever"; } @@ -860,7 +1294,10 @@ TEST(CASRefGc, InvalidRefLogBodyHoldsNamespaceNoPartialDelta) /// A canonical `_log` key (groupRefKeys accepts it) whose body cannot be decoded: the fold GETs it /// and `decodeRefLogTxn` throws. const String garbage_key = layout.refLogKey(fixture::fixtureLife(ns), RefTxnId{1, dropped + 1}); - backend->putIfAbsent(garbage_key, "garbage-not-a-valid-reflog-body"); + { + OperationForTest seed_op(*backend); + (*seed_op).create(garbage_key, "garbage-not-a-valid-reflog-body", Retry::once()); + } /// The corruption claims the next committed position. Advance only the durable frontier, not the /// log body, so recovery must exact-GET and hold this malformed object instead of ignoring F+1. advanceRecoverableCkptForRawFixture(*backend, layout, ns, RefTxnId{1, dropped + 1}); @@ -888,9 +1325,10 @@ TEST(CASRefGc, InvalidRefLogBodyHoldsNamespaceNoPartialDelta) /// precisely what made the hold necessary; if an absent could clear it, the whole mechanism would /// be defeated by the corruption it exists to survive. (Before durable holds this delete DID /// release the namespace, which is the hole Task 8 closed.) - const HeadResult h = backend->head(garbage_key); - ASSERT_TRUE(h.exists); - ASSERT_EQ(backend->deleteExact(garbage_key, h.token).kind, DeleteOutcome::Kind::Deleted); + OperationForTest evidence_op(*backend); + const auto h = (*evidence_op).head(garbage_key, Retry::once()); + ASSERT_TRUE(h.has_value()); + ASSERT_EQ((*evidence_op).remove(garbage_key, h->etag, Retry::once()), Removal::Removed); for (int i = 0; i < 4; ++i) { @@ -973,15 +1411,17 @@ TEST(CASRefGc, CatalogAdmittedFreshLifeWithoutParentSeedsSuccessorSeal) const RootNamespace ns{"00/aa@cas@"}; fixture::admitLive(*backend, layout, ns); - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(*backend, layout); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); ASSERT_EQ(catalog_cut.catalog.entries.size(), 1u); const UInt128 life_id = catalog_cut.catalog.entries.front().incarnation; Gc gc(store, kGc); ASSERT_NO_THROW(gc.runRegularRound()); - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState state = decodeGcState(op.read(layout.gcStateKey(), Retry::once())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + op.read(layout.foldSealKey(state.snap_generation, state.snap_attempt), Retry::once())->bytes); EXPECT_TRUE(seal.ref_lives.contains(life_id)); } diff --git a/src/Disks/tests/gtest_cas_ref_install_safety.cpp b/src/Disks/tests/gtest_cas_ref_install_safety.cpp index 3c3c58b480d8..d3da467cb415 100644 --- a/src/Disks/tests/gtest_cas_ref_install_safety.cpp +++ b/src/Disks/tests/gtest_cas_ref_install_safety.cpp @@ -7,17 +7,22 @@ #include #include #include -#include +#include #include #include #include +#include + +#include #include #include #include #include #include #include +#include +#include #include /// Task 3 (spec §A1, site 1): the region of `CasRefLedger::commitRefChunk` between "this chunk's @@ -53,6 +58,12 @@ using namespace DB::Cas; namespace { +bool manifestKeyExists(const BackendPtr & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + PoolPtr openPool(const BackendPtr & backend) { /// A fresh pool with no residue, mirroring `gtest_cas_ref_chunked_flush.cpp`'s `openPool`. @@ -60,57 +71,41 @@ PoolPtr openPool(const BackendPtr & backend) return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); } -/// As `openPool`, but with a SINGLE-attempt request budget, which is what makes one ambiguous `PUT` -/// conclusive: with retries allowed the controller's resolve-before-reissue would either re-`PUT` (the -/// object never landed) or prove the object durable (it did) and report `Committed`, and neither of the -/// wedge arms under test would ever be reached. Same budget shape as -/// `gtest_cas_ref_chunked_flush.cpp`'s `runChunkFailureCase`, including the short timeouts so there is -/// no inter-attempt sleep to serve. -PoolPtr openPoolSingleAttempt(const BackendPtr & backend) +/// As `openPool`, but with the budget that bounds the mount lease's own admission arithmetic: +/// `attempt_timeout_ms` is what one attempt reserves and `lease_safety_margin_ms` the room kept past +/// it, which together decide the two fence predicates the pre-attempt tests below drive. No budget +/// field bounds a write's ATTEMPT COUNT -- that is the `Retry` policy's, and the ref lane's is +/// `standard` -- so what makes an injected fault conclusive in these tests is that it stays armed for +/// the whole call (`LatchedChunkFaultBackend`) while `VirtualRetryClock` carries the call to its own +/// deadline. +PoolPtr openPoolWedgeBudget(const BackendPtr & backend) { DB::Cas::tests::seedPoolMetaForRestart(*backend); PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; CasRequestBudget budget; - /// ONE attempt is the whole mechanism these tests need: it is what turns an injected lost - /// acknowledgement into `Unresolved` instead of a transparent retry, and it does so independently of - /// how fast the machine is. - /// - /// The operation deadline must therefore NOT sit at `attempt_timeout_ms`, which is where it used to. - /// The controller's pre-send gate (`putIfAbsentControlled`: `now + attempt_timeout > deadline` - /// returns `Unresolved` WITHOUT sending) is then a zero-width race that passes only if no - /// millisecond tick elapses between the deadline capture and the gate. Under parallel-build load it - /// loses: the gate fires first, nothing is sent, the injected fault is never reached, and the flush - /// fails CLEAN -- so the product correctly does NOT wedge the lane and the wedge expectations flip. - /// `UncertainPrecommitKeepsItsCleanupOwnerAndItsBody` was observed failing exactly that way (Task 9, - /// `refLaneWedgedForTest` false at the wedge assertion), and every test on this fixture carries the - /// same razor. Same root cause and same fix as `8f9e63c7a19` for the sweep-interruption test. - /// - /// A WIDE deadline keeps the request always actually sent, so the injected fault decides the outcome - /// rather than the scheduler. Tests that want the pre-send REFUSAL instead use - /// `openPoolFenceControlled`, where a frozen clock makes that refusal deterministic rather than raced. - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; budget.lease_safety_margin_ms = 100; cfg.cas_request_budget = budget; return Pool::open(backend, cfg); } +using DB::Cas::tests::LatchedChunkFaultBackend; +using DB::Cas::tests::VirtualRetryClock; + /// The mount-fence deadlines the pre-attempt tests drive, in the FROZEN boot clock of /// `openPoolFenceControlled` (which is pinned at 0, so these are also the remaining lease budgets). /// -/// `CasMountRuntime` has TWO fence predicates and they are deliberately not the same: -/// `mayMutate` -- `now < deadline`; the top-of-flush gate in `flushRefBatch`. -/// `refAppendFenceOk` -- additionally `attempt_timeout_ms + lease_safety_margin_ms < deadline - now`, -/// i.e. "there is room for one whole controlled attempt"; the `fence_ok` -/// `commitRefChunk` hands to `putIfAbsentControlled`. -/// With `openPoolFenceControlled`'s budget below that margin is 100 + 100 = 200 ms, so a 100 ms -/// remaining lease sits BETWEEN them: the flush is admitted and then its very first pre-attempt gate -/// refuses. That is -/// exactly the production shape this task is about (a lease too short to start a write, not a lost -/// one), and it needs no fault injection at all -- which is the point: nothing is sent. +/// `mayMutate` -- `now < deadline` -- is the top-of-flush gate in `flushRefBatch`. Behind it the fence +/// is asked again by everything the flush issues, and each of those refuses until its own reservation +/// plus `lease_safety_margin_ms` (100) is STRICTLY cleared: +/// a `CasOperation::admitted` guard (e.g. `namespaceLife`'s "resident namespace life" one) reserves +/// nothing -- clears above 100; +/// a read reserves one attempt envelope -- clears above 200; +/// a write reserves TWO, the attempt and the read that settles it -- clears above 300. +/// A "pre-attempt refusal" test wants the flush admitted and everything on the way in to pass while +/// the append's own first request is refused, so it sits strictly between the second and the third. constexpr uint64_t FENCE_DEADLINE_HEALTHY_MS = 30000; -constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 100; +constexpr uint64_t FENCE_DEADLINE_REFUSES_ATTEMPT_MS = 250; /// A legal blob-free part: stage an empty manifest, precommit, promote -- enough to drive real /// ref-log transactions through the append lane. @@ -125,7 +120,7 @@ void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String build->promote(ns, ref, build->buildId(), id); } -/// As `openPoolSingleAttempt`, but with the mount fence under the TEST's control instead of the wall +/// As `openPoolWedgeBudget`, but with the mount fence under the TEST's control instead of the wall /// clock's: /// - the boot clock is FROZEN at 0, so `setMountDeadline` alone decides both fence predicates and no /// elapsed real time can flip one of them mid-test (the same load-bearing injection, for the same @@ -133,25 +128,27 @@ void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String /// - lease renewal is parked an hour out, so the runtime-owned renewal worker cannot re-arm the deadline /// underneath a test that just shortened it. Ten seconds (the default) would be enough in practice /// and flaky in principle; this removes the race rather than betting on it. -PoolPtr openPoolFenceControlled(const BackendPtr & backend) +PoolPtr openPoolFenceControlled(const std::shared_ptr & backend) { DB::Cas::tests::seedPoolMetaForRestart(*backend); PoolConfig cfg{.pool_prefix = "p", .server_root_id = "test"}; cfg.boot_ms_fn = [] { return uint64_t{0}; }; cfg.mount_renew_period = std::chrono::milliseconds{3600000}; CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; cfg.cas_request_budget = budget; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field; production pairs the two in `ContentAddressedMetadataStorage`, and a fixture that sets + /// only the budget leaves the engine reserving nothing and no pre-attempt gate to refuse. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); return Pool::open(backend, cfg); } /// Runs `f`, requires it to throw the ref lane's retry-later condition, and returns the message so a -/// caller can assert WHICH condition it was. The message is the only place the `CasUnresolvedReason` -/// surfaces -- there is no accessor for it, by design (it is a diagnostic, not state) -- so this is how -/// a test proves the reason actually reached the decision site instead of defaulting. +/// caller can assert WHICH condition it was. The message is the only place the lane's own reading of a +/// give-up surfaces -- there is no accessor for it, by design (it is a diagnostic, not state) -- so +/// this is how a test proves the verdict reached the decision site instead of defaulting. String retryLaterMessageOf(const std::function & f) { try @@ -167,6 +164,41 @@ String retryLaterMessageOf(const std::function & f) return {}; } +/// `retryLaterMessageOf` with the fault held armed for the whole call, and with the give-up proven to +/// be the call's OWN retry window: the write engine settles each ambiguity by an exact read and then +/// reissues, so a bounded fault would be outlived and the write would commit. The pacing assertions +/// are what make a fixture whose sleep seam is not wired fail rather than sleep the window out for +/// real. +String wedgingRetryLaterMessageOf(VirtualRetryClock & clock, LatchedChunkFaultBackend & backend, + const std::function & f) +{ + const size_t pauses_before = clock.pauseCount(); + const uint64_t clock_before = clock.nowMs(); + backend.latched = true; + const String message = retryLaterMessageOf(f); + /// Disarmed COMPLETELY, not just unlatched: what every caller does next is a flush that must reach + /// the store normally -- the wedge resolution, or an abandon. A topped-up count or a still-armed + /// lost read would fault that one too, and a wedge resolution whose settling read fails does not + /// resolve anything. + backend.latched = false; + backend.mode = LatchedChunkFaultBackend::Mode::None; + backend.fault_count = 0; + backend.fault_skip = 0; + backend.fail_read_once_key.clear(); + EXPECT_GT(clock.pauseCount(), pauses_before + 1) + << "the reissues must pace through the injected sleep, never a real one"; + EXPECT_LE(clock.longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; + EXPECT_GE(clock.nowMs() - clock_before, 60000u) + << "the give-up must be the call's own retry window, not a pre-attempt refusal"; + return message; +} + +void driveToTheWedge(VirtualRetryClock & clock, LatchedChunkFaultBackend & backend, + const std::function & f) +{ + (void)wedgingRetryLaterMessageOf(clock, backend, f); +} + /// Installs a ONE-SHOT throwing probe into the post-durable install regions (spec §A2): the next region /// entered throws, every later one runs normally -- which is what lets a terminality test drive a /// successful flush after the recovery transition. @@ -243,8 +275,9 @@ TEST(CASRefInstallSafety, PostDurableInstallIsAllocationFree) /// counter must NOT advance -- an unproven transaction is not a recorded one. TEST(CASRefInstallSafety, UnresolvedAlwaysRecordsTheWedge) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/unresolved_wedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -257,7 +290,8 @@ TEST(CASRefInstallSafety, UnresolvedAlwaysRecordsTheWedge) backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - const String message = retryLaterMessageOf([&] { publishEmptyPart(store, ns, "part_a"); }); + const String message = wedgingRetryLaterMessageOf(*clock, *backend, + [&] { publishEmptyPart(store, ns, "part_a"); }); EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "an Unresolved PUT must always leave a wedge"; const String wedged_key = store->wedgedKeyForTest(ns); @@ -272,9 +306,9 @@ TEST(CASRefInstallSafety, UnresolvedAlwaysRecordsTheWedge) /// backend's `putIfAbsent`, so the request reached it), the single-attempt budget is then spent, and /// the lane wedges. The message must say so -- and must NOT say "no attempt was sent", which is the /// only shape allowed to skip the wedge. - EXPECT_NE(message.find("attempt budget was exhausted"), String::npos) + EXPECT_NE(message.find("is UNCERTAIN"), String::npos) << "the reason must reach the wedge message rather than defaulting: " << message; - EXPECT_EQ(message.find("no attempt was sent"), String::npos) + EXPECT_EQ(message.find("BEFORE any request was sent"), String::npos) << "an ambiguous PUT is not a pre-attempt refusal: " << message; } @@ -314,7 +348,7 @@ TEST(CASRefInstallSafety, PreAttemptRefusalDoesNotWedgeTheLane) const String message = retryLaterMessageOf([&] { store->dropRef(ns, "part_a"); }); - EXPECT_NE(message.find("no attempt was sent"), String::npos) + EXPECT_NE(message.find("refused BEFORE any request was sent"), String::npos) << "the caller must be told WHY, and this is the reason the no-wedge decision rests on: " << message; EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "nothing was sent, so nothing can be durable: there is no ambiguity for a wedge to resolve"; @@ -330,8 +364,8 @@ TEST(CASRefInstallSafety, PreAttemptRefusalDoesNotWedgeTheLane) << "the refused drop must not have taken effect"; /// The availability half of the claim: the lane is usable the moment the lease is healthy again -- - /// no remount, no wedge resolution, nothing to clear. Before this task the same sequence left a - /// wedge over a key that was never written, and this append would have failed forever. + /// no remount, no wedge resolution, nothing to clear. A refused pre-attempt never writes a key, + /// so it cannot leave a wedge to block this append. store->setMountDeadline(FENCE_DEADLINE_HEALTHY_MS); store->dropRef(ns, "part_a"); EXPECT_FALSE(store->resolveRef(ns, "part_a", /*allow_stale=*/false).has_value()) @@ -352,8 +386,9 @@ TEST(CASRefInstallSafety, PreAttemptRefusalDoesNotWedgeTheLane) /// that never can. TEST(CASRefInstallSafety, PreAttemptRefusalAfterAWedgeResolutionLeavesTheLaneClean) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPoolFenceControlled(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/pre_attempt_after_unwedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -364,12 +399,12 @@ TEST(CASRefInstallSafety, PreAttemptRefusalAfterAWedgeResolutionLeavesTheLaneCle publishEmptyPart(store, ns, "y"); const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); - /// Wedge over an object that IS durable: the write lands, its acknowledgement is lost, and the - /// controller's own verifying read is lost too (the only mode that reaches the resolution install). + /// Wedge over an object that IS durable: the write lands, its acknowledgement is lost, and every + /// settling read of the key is lost too (the only mode that reaches the resolution install). backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_seed); @@ -383,7 +418,7 @@ TEST(CASRefInstallSafety, PreAttemptRefusalAfterAWedgeResolutionLeavesTheLaneCle const String message = retryLaterMessageOf([&] { store->dropRef(ns, "y"); }); store->setRefPreCarveHookForTest(nullptr); - EXPECT_NE(message.find("no attempt was sent"), String::npos) << message; + EXPECT_NE(message.find("refused BEFORE any request was sent"), String::npos) << message; EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "the wedge that existed was RESOLVED, and the chunk that followed it was never sent -- the " "lane must be left clean, not re-wedged over an id that can never resolve"; @@ -407,8 +442,9 @@ TEST(CASRefInstallSafety, PreAttemptRefusalAfterAWedgeResolutionLeavesTheLaneCle /// lane must wedge again, now over the NEW transaction. TEST(CASRefInstallSafety, AmbiguousChunkAfterAWedgeResolutionRewedgesTheLane) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPoolFenceControlled(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/ambiguous_after_unwedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -422,19 +458,19 @@ TEST(CASRefInstallSafety, AmbiguousChunkAfterAWedgeResolutionRewedgesTheLane) backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); const String first_wedged_key = store->wedgedKeyForTest(ns); ASSERT_FALSE(first_wedged_key.empty()); /// The resolution is a conditional CREATE at the wedged key now, and that key already holds our - /// own landed object, so it conflicts and the follow-up read adopts it (`LandedThenLost`'s one-shot - /// lost read was consumed inside the previous attempt, so this read succeeds). `fault_skip` lets - /// that create through and puts the fault on this flush's OWN chunk PUT, which is the subject. + /// own landed object, so it conflicts and the settling read adopts it -- the lost-read leg is + /// disarmed above, so that read succeeds. `fault_skip` lets that create through and puts the fault + /// on this flush's OWN chunk PUT, which is the subject. backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_skip = 1; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "y"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "y"); }); EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "an attempt was sent for the new chunk, so its object may be durable: the lane must wedge"; @@ -446,39 +482,88 @@ TEST(CASRefInstallSafety, AmbiguousChunkAfterAWedgeResolutionRewedgesTheLane) << "a wedged lane may hold a durable transaction the runtime has not recorded"; } -/// Task 18's regression guard, asserted on the mapping itself rather than through six pieces of fault -/// choreography. `unresolvedProvesNothingWasSent` is the whole decision: the ledger wedges unless it -/// answers true, so this table IS the protocol. +/// The whole wedge decision, as one table. `GaveUp::sent_any` is what the ledger branches on -- it +/// returns the attempt to `Ready` when nothing was sent and wedges otherwise -- so this table IS the +/// protocol. Each row is DRIVEN rather than asserted about a mapping: the point of the old enum-shaped +/// version was that a value could be listed without any way to reach it. /// -/// What protects a future contributor who adds a `CasUnresolvedReason` member and forgets this file: -/// the predicate is a switch with NO `default`, so the addition is a `-Wswitch` build error (a forced -/// decision, not a silent one), and its trailing `return false` makes the runtime answer "wedge" even -/// if that diagnostic is ever suppressed. Both directions fail closed; neither can widen the allow-list -/// by omission. The `static_assert`s make the mapping a compile-time fact, and the `EXPECT`s repeat it -/// so a break names the offending value in the test report. -TEST(CASRefInstallSafety, OnlyNoAttemptSentMaySkipTheWedge) -{ - static_assert(unresolvedProvesNothingWasSent(CasUnresolvedReason::NoAttemptSent)); - static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::NotUnresolved)); - static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostMidWay)); - static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::DeadlineMidWay)); - static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostPostWrite)); - static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::AttemptsExhausted)); - - EXPECT_TRUE(unresolvedProvesNothingWasSent(CasUnresolvedReason::NoAttemptSent)) - << "the pre-attempt gates rejected before the first request: the key is provably unwritten"; - /// `NotUnresolved` is reachable at the decision site if any path ever returns `Unresolved` without - /// recording a reason, so it is listed here as a real case, not as enum hygiene. - EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::NotUnresolved)) - << "an unrecorded reason proves nothing and must keep wedging"; - EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostMidWay)) - << "an attempt was already sent: its object may be durable"; - EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::DeadlineMidWay)) - << "an attempt was already sent: its object may be durable"; - EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::FenceLostPostWrite)) - << "the attempt COMMITTED and only the fence was lost afterwards -- the most durable case of all"; - EXPECT_FALSE(unresolvedProvesNothingWasSent(CasUnresolvedReason::AttemptsExhausted)) - << "every attempt is a candidate for having landed"; +/// `sent_any` is set on the line before the attempt goes out, so exactly one shape can report it false: +/// every gate refused before the first request. The four rows below it are the ways a call can end +/// AFTER something reached the network, and each leaves an object that may be durable. +TEST(CASRefInstallSafety, OnlySendingNothingMaySkipTheWedge) +{ + /// Row 1: the caller's own facts refuse before the attempt. Nothing reaches the store. + { + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([] { return false; }); + const WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_FALSE(gave_up->sent_any) + << "the pre-attempt gates rejected before the first request: the key is provably unwritten"; + EXPECT_EQ(backend->writeTotal(), 0u); + } + + /// Row 2: the OTHER pre-attempt refusal. Same verdict, a different bound. + { + auto backend = std::make_shared(); + uint64_t clock = 0; + CasRequests requests(backend, Fence::open(), + [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1000; return t; }); + CasOperation op = requests.admit(); + const WriteResult result = op.create("k", "v", Retry::within(500)); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 0u); + } + + /// Row 3: an attempt was sent and its outcome never settled. + { + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + const WriteResult result = op.create("k", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any) << "an attempt was already sent: its object may be durable"; + } + + /// Row 4: the attempt COMMITTED and only the admission was lost afterwards -- the most durable case + /// of all, and the one a caller is most tempted to report as success. + { + auto backend = std::make_shared(); + bool live = true; + backend->onWriteCommitted("k", [&live] { live = false; }); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([&live] { return live; }); + const WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 1u) << "and the object IS there"; + } + + /// Row 5: the deadline arrives AFTER an attempt rather than before one. The clock is moved by a + /// hook that runs inside the write, so the first attempt is admitted and the settling read is not. + { + auto backend = std::make_shared(); + uint64_t clock = 0; + backend->onBeforeWrite("k", [&clock] { clock = 100'000; }); + backend->injectAmbiguousWrite("k"); + CasRequests requests(backend, Fence::open(), [&clock]() -> uint64_t { return clock; }); + CasOperation op = requests.admit(); + const WriteResult result = op.create("k", "v", Retry::within(1000)); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_TRUE(gave_up->sent_any) << "an attempt was already sent: its object may be durable"; + } } /// Task 5 (spec §A1, site 2). Resolving a wedge is a post-durable install too: the resolving GET PROVES @@ -495,8 +580,9 @@ TEST(CASRefInstallSafety, OnlyNoAttemptSentMaySkipTheWedge) /// bumped once per install, so a re-applied transaction shows up as one extra. TEST(CASRefInstallSafety, WedgeResolutionInstallsExactlyOnce) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_resolution"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -509,13 +595,13 @@ TEST(CASRefInstallSafety, WedgeResolutionInstallsExactlyOnce) /// whole test is a handful of tiny transactions, so every delta below is exact. const size_t tail_after_seed = store->tailSinceSnapshotCountForTest(ns); - /// Drop "x" through a PUT that LANDS and then loses its response, plus the one-shot lost read that - /// keeps the controller's own resolve-before-reissue from settling it inside the same attempt. One - /// attempt, so the lane wedges over an object that is genuinely durable -- the only way in. + /// Drop "x" through a PUT that LANDS and then loses its response, plus the lost settling read that + /// keeps the engine from proving the commit inside the same call. Both stay armed for the whole + /// call, so the lane wedges over an object that is genuinely durable -- the only way in. backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "the lost-response drop must wedge the lane"; ASSERT_FALSE(store->wedgedKeyForTest(ns).empty()); @@ -631,8 +717,9 @@ TEST(CASRefInstallSafety, WritingOwnsTheAttemptUntilInstall) /// Ambiguity transfers the same exact attempt from `Writing` to `Wedged`. TEST(CASRefInstallSafety, UnresolvedTransfersWritingToWedged) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/apply_state_wedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -643,7 +730,7 @@ TEST(CASRefInstallSafety, UnresolvedTransfersWritingToWedged) backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "part_a"); }); + driveToTheWedge(*clock, *backend, [&] { publishEmptyPart(store, ns, "part_a"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) @@ -653,8 +740,9 @@ TEST(CASRefInstallSafety, UnresolvedTransfersWritingToWedged) /// Durable resolution installs the attempt and returns the lane to `Ready`. TEST(CASRefInstallSafety, WedgeResolutionReturnsReady) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/apply_state_unwedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -665,11 +753,11 @@ TEST(CASRefInstallSafety, WedgeResolutionReturnsReady) publishEmptyPart(store, ns, "y"); /// The one mode that wedges over a GENUINELY durable object (see `ChunkFaultBackend`): the write - /// lands, its acknowledgement is lost, and the controller's own verifying read is lost too. + /// lands, its acknowledgement is lost, and every settling read of the key is lost too. backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); @@ -684,8 +772,8 @@ TEST(CASRefInstallSafety, WedgeResolutionReturnsReady) /// A foreign occupant is a terminal `Faulted` verdict. TEST(CASRefInstallSafety, ConclusiveForeignConflictFaultsTheLane) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); const RootNamespace ns{"srv1/apply_state_conflict"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -711,8 +799,8 @@ TEST(CASRefInstallSafety, DefiniteFailureReturnsReady) #if !USE_AWS_S3 GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; #else - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); const RootNamespace ns{"srv1/apply_state_definite"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -743,8 +831,9 @@ TEST(CASRefInstallSafety, DefiniteFailureReturnsReady) /// `CASAnomalyPolicy.ForeignBytesAtWedgeKeyTripFenceAndRemount`'s. TEST(CASRefInstallSafety, WedgeResolutionProvenForeignFaultsTheLane) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/apply_state_foreign_wedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -757,7 +846,7 @@ TEST(CASRefInstallSafety, WedgeResolutionProvenForeignFaultsTheLane) backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); /// Out of band, a foreign writer lands DIFFERENT bytes at the exact wedged key. The fault mode is @@ -765,7 +854,10 @@ TEST(CASRefInstallSafety, WedgeResolutionProvenForeignFaultsTheLane) backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; const String wedged_key = store->wedgedKeyForTest(ns); ASSERT_FALSE(wedged_key.empty()); - ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + { + DB::Cas::tests::OperationForTest foreign_op(backend); + ASSERT_TRUE(std::holds_alternative((*foreign_op).create(wedged_key, "a-different-object", Retry::once()))); + } DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); @@ -804,8 +896,9 @@ TEST(CASRefInstallSafety, PostDurableInstallFailureRequiresRecovery) /// would have cleared it is in the same region that threw. TEST(CASRefInstallSafety, WedgeResolutionInstallFailureRequiresRecovery) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/apply_state_poison_unwedge"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -818,7 +911,7 @@ TEST(CASRefInstallSafety, WedgeResolutionInstallFailureRequiresRecovery) backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; @@ -886,8 +979,9 @@ TEST(CASRefInstallSafety, NeedsRecoveryReplaysBeforeALaterFlush) /// still live once the wedge resolves. TEST(CASRefInstallSafety, UncertainPrecommitKeepsItsCleanupOwnerAndItsBody) { - auto backend = std::make_shared(); - auto store = openPoolSingleAttempt(backend); + auto backend = std::make_shared(); + auto store = openPoolWedgeBudget(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/uncertain_precommit"}; /// Stage B (Task 4-C): pin `ns` to the Stage-A sentinel BEFORE its first real touch, so /// the fault injected below (computed from that same sentinel) lands on the key production @@ -900,15 +994,15 @@ TEST(CASRefInstallSafety, UncertainPrecommitKeepsItsCleanupOwnerAndItsBody) auto build = store->beginPartWrite(info); const ManifestId id = build->stageManifest({}); const String manifest_key = store->layout().manifestKey(id); - ASSERT_TRUE(backend->head(manifest_key).exists) << "the staged body must exist before the precommit"; + ASSERT_TRUE(manifestKeyExists(backend, manifest_key)) << "the staged body must exist before the precommit"; /// Scoped to THIS namespace's ref log so the manifest body's own PUT cannot consume the fault. The - /// object LANDS and only its acknowledgement is lost, which with the single-attempt budget wedges - /// the lane over a genuinely durable precommit -- the exact shape the old code mishandled. + /// object LANDS and only its acknowledgement is lost, and no settling read can prove otherwise, so + /// the lane wedges over a genuinely durable precommit -- the exact shape the old code mishandled. backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { build->precommitAdd(ns, "part_a", id); }); + driveToTheWedge(*clock, *backend, [&] { build->precommitAdd(ns, "part_a", id); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "the lost-response precommit must wedge the lane"; EXPECT_EQ(build->precommitState(), PartWriteTxn::PrecommitState::Uncertain) << "an append that may have landed is neither 'never precommitted' nor 'durably precommitted'"; @@ -917,7 +1011,7 @@ TEST(CASRefInstallSafety, UncertainPrecommitKeepsItsCleanupOwnerAndItsBody) /// precommit durable) and appends the exact removal in the same flush. build->abandon(); - EXPECT_TRUE(backend->head(manifest_key).exists) + EXPECT_TRUE(manifestKeyExists(backend, manifest_key)) << "abandon writer-deleted the body of a precommit that may be live -- GC's fold barrier would " "clamp on it forever"; EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()) diff --git a/src/Disks/tests/gtest_cas_ref_log_format.cpp b/src/Disks/tests/gtest_cas_ref_log_format.cpp index d5186e77aca1..1ef57e5005d5 100644 --- a/src/Disks/tests/gtest_cas_ref_log_format.cpp +++ b/src/Disks/tests/gtest_cas_ref_log_format.cpp @@ -6,6 +6,8 @@ #include #include +#include + /// v3 text codec tests for `cas_ref_log` (codecs-v3 phase 3). Split out of the retired /// `gtest_cas_ref_codecs.cpp` and re-pointed at the TEXT codec: the encoder-side validation tests are /// format-agnostic (they only assert `encodeRefLogTxn` throws) and carry over verbatim; the old @@ -140,6 +142,20 @@ TEST(CASRefCodec, OrderMatchesLexicalOrderOfRender) /// RefLogTxn: round trip /// =================================================================================== +/// Closed-set pin: the five `RefOpKind` words, walked through `magic_enum::enum_values`, which is +/// what proves the renderer and the parser consult the SAME table: a table entry missing altogether is already a +/// build error at the coverage assert, but two delegates drifting onto different tables is not. +TEST(CASRefCodec, ClosedSetPinsRefOpKindWords) +{ + EXPECT_EQ(refOpKindToWireWord(RefOpKind::NamespaceBirth), "namespace_birth"); + EXPECT_EQ(refOpKindToWireWord(RefOpKind::OwnerTransition), "owner_transition"); + EXPECT_EQ(refOpKindToWireWord(RefOpKind::SetPublishedAt), "set_published_at"); + EXPECT_EQ(refOpKindToWireWord(RefOpKind::RemoveNamespace), "remove_namespace"); + EXPECT_EQ(refOpKindToWireWord(RefOpKind::EpochSeal), "epoch_seal"); + for (const auto k : magic_enum::enum_values()) + EXPECT_EQ(refOpKindFromWireWord(refOpKindToWireWord(k)), k); +} + TEST(CASRefCodec, RoundTripNamespaceBirth) { RefLogTxn txn; @@ -185,9 +201,9 @@ TEST(CASRefCodec, RoundTripSetPublishedAt) EXPECT_EQ(decoded, txn); } -/// No-tolerance decode pin (codex round-2, finding 3): the `"pl"` (payload) field was removed from the -/// ref-op wire in stage-1 T12. Although the retired `set_payload` op WORD is already rejected by -/// `opKindFromWord`, the generic op-record reader reads all field keys before switching on kind, so a +/// No-tolerance decode pin: the `"pl"` (payload) field was removed from the +/// ref-op wire. Although the retired `set_payload` op WORD is already rejected by +/// `refOpKindFromWireWord`, the generic op-record reader reads all field keys before switching on kind, so a /// `"pl"` field paired with a still-recognized op word would otherwise be `skipUnknown`'d. It is a /// removed field, not a genuinely-unknown one: decoding an op record that still carries `"pl"` must FAIL /// with `CORRUPTED_DATA` naming the removed field. @@ -204,8 +220,8 @@ TEST(CASRefCodec, DecodeRejectsRemovedPayloadFieldInOpRecord) txn.ops.push_back(op); const String bytes = encodeRefLogTxn(txn); - /// Splice the retired `"pl"` field back into the op record, just before its `"ts"` field. - const String needle = ",\"ts\":"; + /// Splice the retired `"pl"` field back into the op record, just before its `"published_ms"` field. + const String needle = ",\"published_ms\":"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); const String tampered = bytes.substr(0, pos) + R"(,"pl":"deadbeef")" + bytes.substr(pos); @@ -285,6 +301,82 @@ TEST(CASRefCodec, RoundTripOwnerTransitionReplace) ASSERT_TRUE(decoded.ops[0].new_binding.has_value()); } +TEST(CASRefCodec, OwnerTransitionBindingGroupsAreAbsentOrComplete) +{ + RefLogTxn txn; + txn.ns = "ns"; + txn.txn_id = RefTxnId{1, 1}; + RefOp op; + op.kind = RefOpKind::OwnerTransition; + op.old_binding = RefOwnerBinding{RefOwnerKind::Precommit, "old", manifestRef(1, 1, 1)}; + op.new_binding = RefOwnerBinding{RefOwnerKind::Committed, "new", manifestRef(1, 1, 1)}; + txn.ops.push_back(op); + const String bytes = encodeRefLogTxn(txn); + + const String old_group = R"(,"old_kind":"precommit","old_ref":"old","old_epoch":"1","old_build":"1","old_ord":1)"; + const auto old_group_pos = bytes.find(old_group); + ASSERT_NE(old_group_pos, String::npos); + String old_absent = bytes; + old_absent.erase(old_group_pos, old_group.size()); + const RefLogTxn without_old = decodeRefLogTxn(old_absent, txn.ns, txn.txn_id); + ASSERT_EQ(without_old.ops.size(), 1u); + EXPECT_FALSE(without_old.ops[0].old_binding.has_value()); + EXPECT_TRUE(without_old.ops[0].new_binding.has_value()); + + const String old_ref = R"(,"old_ref":"old")"; + const auto old_ref_pos = bytes.find(old_ref); + ASSERT_NE(old_ref_pos, String::npos); + String incomplete_old = bytes; + incomplete_old.erase(old_ref_pos, old_ref.size()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(incomplete_old, txn.ns, txn.txn_id); }); + + const String new_group = R"(,"new_kind":"committed","new_ref":"new","new_epoch":"1","new_build":"1","new_ord":1)"; + const auto new_group_pos = bytes.find(new_group); + ASSERT_NE(new_group_pos, String::npos); + String new_absent = bytes; + new_absent.erase(new_group_pos, new_group.size()); + const RefLogTxn without_new = decodeRefLogTxn(new_absent, txn.ns, txn.txn_id); + ASSERT_EQ(without_new.ops.size(), 1u); + EXPECT_TRUE(without_new.ops[0].old_binding.has_value()); + EXPECT_FALSE(without_new.ops[0].new_binding.has_value()); + + const String new_ref = R"(,"new_ref":"new")"; + const auto new_ref_pos = bytes.find(new_ref); + ASSERT_NE(new_ref_pos, String::npos); + String incomplete_new = bytes; + incomplete_new.erase(new_ref_pos, new_ref.size()); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { decodeRefLogTxn(incomplete_new, txn.ns, txn.txn_id); }); +} + +/// The anomaly diagnostic identifies an object found at a key it should not occupy by reading the +/// meta line's three identity fields. It reads them through the codec's own key constants, so this +/// test is what proves the reader did not quietly stop matching when those keys were renamed: with a +/// stale spelling the tolerant reader skips every real key and the peek answers nullopt on a +/// perfectly good ref-log. +TEST(CASRefCodec, PeekReadsTheMetaIdentityOfASealedRefLog) +{ + RefLogTxn txn; + txn.ns = "srv1/db/table@cas@"; + txn.txn_id = RefTxnId{4, 9}; + RefOp birth; + birth.kind = RefOpKind::NamespaceBirth; + txn.ops.push_back(birth); + + const auto peek = peekRefLogMeta(sealObject(FormatId::RefLog, encodeRefLogTxn(txn))); + ASSERT_TRUE(peek.has_value()) << "a well-formed ref-log must identify its own writer"; + EXPECT_EQ(peek->ns, "srv1/db/table@cas@"); + EXPECT_EQ(peek->writer_epoch, 4u); + EXPECT_EQ(peek->ref_sequence, 9u); +} + +/// The other half of its contract: it identifies a writer, it never certifies an object, so anything +/// it cannot read is `nullopt` rather than an exception escaping into the anomaly report. +TEST(CASRefCodec, PeekAnswersNulloptForBytesThatAreNotARefLog) +{ + EXPECT_FALSE(peekRefLogMeta("not a sealed cas object at all").has_value()); + EXPECT_FALSE(peekRefLogMeta(sealObject(FormatId::RefLog, "{\"type\":\"cas_ref_log\",\"v\":1}\n")).has_value()); +} + TEST(CASRefCodec, RoundTripMultipleOpsInOneTransaction) { RefLogTxn txn; @@ -766,6 +858,8 @@ TEST(CASRefCodec, EncodeRejectsZeroManifestRefInSetPublishedAt) /// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) /// =================================================================================== +CAS_BATTERY_COVERS(RefLog); + TEST(CASFormatBattery, RefLog) { RefLogTxn txn; @@ -784,7 +878,7 @@ TEST(CASFormatBattery, RefLog) [txn] { return sealObject(FormatId::RefLog, encodeRefLogTxn(txn)); }, [ns, id](std::string_view s) { decodeRefLogTxn(openObject(FormatId::RefLog, s), ns, id); }, currentFormatHeader("cas_ref_log") + - "{\"ns\":\"ns\",\"we\":\"1\",\"rs\":\"1\"}\n" - "{\"op\":\"set_published_at\",\"rn\":\"all_1_1_0\",\"me\":\"1\",\"mb\":\"1\",\"mo\":1,\"ts\":42}\n" + "{\"namespace\":\"ns\",\"txn_epoch\":\"1\",\"txn_seq\":\"1\"}\n" + "{\"op\":\"set_published_at\",\"ref\":\"all_1_1_0\",\"epoch\":\"1\",\"build\":\"1\",\"ord\":1,\"published_ms\":42}\n" "{\"n\":1}\n"}); } diff --git a/src/Disks/tests/gtest_cas_ref_protocol.cpp b/src/Disks/tests/gtest_cas_ref_protocol.cpp new file mode 100644 index 000000000000..93c71358cd0c --- /dev/null +++ b/src/Disks/tests/gtest_cas_ref_protocol.cpp @@ -0,0 +1,51 @@ +#include + +#include +#include "cas_test_helpers.h" + +#include + +using namespace DB::Cas; + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; + +namespace +{ + +CasRequests makeRequests(BackendPtr backend, FakeClock & clock, Fence fence = Fence::open()) +{ + return CasRequests(std::move(backend), std::move(fence), clock.nowFn(), clock.sleepFn()); +} + +} + +TEST(CASRefProtocol, CrossEpochFromSealShortCircuitsWithoutAnyRequest) +{ + FakeClock clock; + auto backend = std::make_shared(); + Layout layout("pool"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const RootNamespace ns("t"); + const auto life = DB::Cas::tests::fixture::fixtureLife(ns); + + /// `from_seal == RefTxnId{}`: nothing consumed yet, so there is no seal to cross from -- proved + /// without reading anything. + { + const EpochCrossResult r = crossEpochFromSeal( + op, layout, ns, RefTxnId{}, std::nullopt, RefTxnId{2, 1}, life); + EXPECT_EQ(r.outcome, EpochCrossOutcome::NothingConsumed); + } + + /// The caller already decoded the record at `from_seal` and knows it is not an `EpochSeal` -- + /// also proved without any read here. + { + const EpochCrossResult r = crossEpochFromSeal( + op, layout, ns, RefTxnId{1, 5}, /*seal_proven=*/false, RefTxnId{2, 1}, life); + EXPECT_EQ(r.outcome, EpochCrossOutcome::NotASeal); + } + + EXPECT_EQ(backend->getTotal(), 0u); +} diff --git a/src/Disks/tests/gtest_cas_ref_read_contract.cpp b/src/Disks/tests/gtest_cas_ref_read_contract.cpp index a86d2530d4e8..ba8456b0c2a8 100644 --- a/src/Disks/tests/gtest_cas_ref_read_contract.cpp +++ b/src/Disks/tests/gtest_cas_ref_read_contract.cpp @@ -47,13 +47,22 @@ ManifestId publishRefThroughPool(const PoolPtr & store, const RootNamespace & ns return id; } +/// The `refresh_authority` hook `deleteCompletedRemoving` requires every caller to state explicitly. +/// This fixture's operation carries a direct liveness (`op.admitted()` re-checks the fence itself on +/// every call), not a cached flag, so there is nothing for a refresh to re-read between attempts. +void noAuthorityRefresh() +{ +} + /// Delete the current catalog life through the production exact-removal authority (`casUpdate` to /// `Removing`, then `deleteCompletedRemoving` under a held fence), retaining every old physical byte /// and any already-resident runtime. Mirrors `gtest_cas_ns_file_read_contract.cpp`'s /// `deleteCatalogLife` -- lifecycle-real, not a raw sentinel overwrite. -void deleteCatalogLife(Backend & backend, const Layout & layout, const NamespaceLifeId & life) +void deleteCatalogLife(const BackendPtr & backend, const Layout & layout, const NamespaceLifeId & life) { - CasRefCatalog::casUpdate(backend, layout, [&](const RefCatalog & current) + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + CasRefCatalog::casUpdate(op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find_if(next.entries.begin(), next.entries.end(), [&](const CatalogEntry & entry) @@ -67,7 +76,7 @@ void deleteCatalogLife(Backend & backend, const Layout & layout, const Namespace return next; }); - const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot snapshot = CasRefCatalog::read(op, layout); const auto it = std::find_if(snapshot.catalog.entries.begin(), snapshot.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == life.ns && entry.incarnation == life.incarnation; @@ -77,11 +86,9 @@ void deleteCatalogLife(Backend & backend, const Layout & layout, const Namespace CasFoldSeal parent; parent.ref_lives.emplace(life.incarnation, RefLifeFoldState{ - .coverage = RefCoverage{.classification = 2, .last_folded_ref_id = RefTxnId{1, 1}}, + .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{1, 1}}, .cleanup_evidence = RefCleanupEvidence{.remove_txn_id = RefTxnId{1, 1}}}); - if (CasRefCatalog::deleteCompletedRemoving( - backend, layout, *it, parent, 1, - [](uint64_t) { return CasRefCatalog::LeaderFenceStatus::Held; }) + if (CasRefCatalog::deleteCompletedRemoving(op, layout, *it, parent, noAuthorityRefresh).outcome != CasRefCatalog::CompletedRemovingDeleteOutcome::Deleted) throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Failed to delete fixture catalog life '{}'", life.ns.string()); } @@ -90,13 +97,15 @@ void deleteCatalogLife(Backend & backend, const Layout & layout, const Namespace /// rebirth every test below drives. Mirrors `gtest_cas_ns_file_read_contract.cpp`'s /// `admitReplacementLife`. NamespaceLifeId admitReplacementLife( - Backend & backend, const Layout & layout, uint64_t gc_shards, + const BackendPtr & backend, const Layout & layout, uint64_t gc_shards, const NamespaceLifeId & predecessor, UInt128 successor_incarnation) { if (predecessor.incarnation == successor_incarnation) throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Fixture life ids unexpectedly collide"); const NamespaceLifeId successor = NamespaceLifeId::fromCatalogEntry(predecessor.ns, successor_incarnation); - CasRefCatalog::casAdmitEntry(backend, layout, gc_shards, CatalogEntry{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + CasRefCatalog::casAdmitEntry(op, layout, gc_shards, CatalogEntry{ .ns = successor.ns, .state = NsState::Live, .incarnation = successor.incarnation}); return successor; } @@ -131,8 +140,8 @@ TEST(CASRefReadContract, HeldRuntimeAfterSameNameRebirthReadsStaleOrNotFoundNeve /// Drop and re-admit under the SAME logical name, bypassing this store's own ledger entirely -- /// exactly as an independent actor's drop/rebirth would look from this reader's point of view. - deleteCatalogLife(*backend, layout, life1); - const NamespaceLifeId life2 = admitReplacementLife(*backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc123}); + deleteCatalogLife(backend, layout, life1); + const NamespaceLifeId life2 = admitReplacementLife(backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc123}); ASSERT_NE(life1.incarnation, life2.incarnation); const ManifestRef life2_ref{/*writer_epoch*/ 1, /*build_sequence*/ 777, /*manifest_ordinal*/ 1}; @@ -184,7 +193,7 @@ TEST(CASRefReadContract, HotRefReadsThroughHeldRuntimeIssueZeroCatalogRequests) /// recorder that never saw anything. EXPECT_GT( backend->headCount(layout.refCatalogKey()) + backend->getCount(layout.refCatalogKey()) - + backend->casPutCount(layout.refCatalogKey()), + + backend->putOverwriteCount(layout.refCatalogKey()), 0u) << "the cold admission above must have reached the catalog at least once"; EXPECT_GT(backend->getCount(layout.refCkptKey(life)), 0u) << "the cold recovery above must have read this namespace's own checkpoint at least once"; @@ -197,7 +206,7 @@ TEST(CASRefReadContract, HotRefReadsThroughHeldRuntimeIssueZeroCatalogRequests) EXPECT_EQ(backend->headCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 0u); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); /// Stronger than the catalog-only clauses above: a warm ref read is a pure map lookup over the @@ -222,27 +231,28 @@ TEST(CASRefReadContract, StaleLifeDropRefusesAfterRebirthAndNeverTouchesSuccesso ASSERT_TRUE(store->refTableLifeForTest(ns).has_value()); const NamespaceLifeId life1 = *store->refTableLifeForTest(ns); - deleteCatalogLife(*backend, layout, life1); - const NamespaceLifeId life2 = admitReplacementLife(*backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc456}); + deleteCatalogLife(backend, layout, life1); + const NamespaceLifeId life2 = admitReplacementLife(backend, layout, store->poolConfig().gc_shards, life1, UInt128{0xabc456}); const ManifestRef life2_ref{/*writer_epoch*/ 1, /*build_sequence*/ 999, /*manifest_ordinal*/ 1}; publishCommittedTransition(*backend, layout, ns, ref_name, std::nullopt, life2_ref); const ManifestId life2_manifest{ns, life2_ref}; - const HeadResult catalog_head_before = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(catalog_head_before.exists); - const auto catalog_get_before = backend->get(layout.refCatalogKey()); + OperationForTest catalog_probe(*backend); + const auto catalog_head_before = (*catalog_probe).head(layout.refCatalogKey(), Retry::standard()); + ASSERT_TRUE(catalog_head_before.has_value()); + const auto catalog_get_before = (*catalog_probe).read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog_get_before.has_value()); /// The held life-1 handle names an incarnation the catalog no longer carries: refused, not /// resolved against the current (life-2) row. expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(life1); }); - const HeadResult catalog_head_after = backend->head(layout.refCatalogKey()); - ASSERT_TRUE(catalog_head_after.exists); - EXPECT_EQ(catalog_head_after.token, catalog_head_before.token) + const auto catalog_head_after = (*catalog_probe).head(layout.refCatalogKey(), Retry::standard()); + ASSERT_TRUE(catalog_head_after.has_value()); + EXPECT_EQ(catalog_head_after->etag, catalog_head_before->etag) << "a refused stale-life drop must not touch the catalog object at all"; - const auto catalog_get_after = backend->get(layout.refCatalogKey()); + const auto catalog_get_after = (*catalog_probe).read(layout.refCatalogKey(), Retry::standard()); ASSERT_TRUE(catalog_get_after.has_value()); EXPECT_EQ(catalog_get_after->bytes, catalog_get_before->bytes); diff --git a/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp index 2b75ae13ada0..7919711c1753 100644 --- a/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp +++ b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp @@ -1,5 +1,7 @@ #include +#include + #include "config.h" #include @@ -66,6 +68,7 @@ extern const Event CASRefCheckpointPublished; } using namespace DB::Cas; +using DB::Cas::tests::VirtualRetryClock; using DB::Cas::tests::committedRow; using DB::Cas::tests::CountingBackend; using DB::Cas::tests::expectThrowsCode; @@ -83,17 +86,58 @@ ManifestRef manifestRef(uint64_t epoch, uint64_t build_sequence, uint32_t ordina return ManifestRef{epoch, build_sequence, ordinal}; } +/// Fixture observations of durable state run on an OPEN fence: they are not writes a mount admitted, +/// and each owns the `CasRequests` its operation borrows, so none of these hands one back. +std::optional readCkptForTest(const BackendPtr & backend, const Layout & layout, + const NamespaceLifeId & life) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return readCkpt(op, layout, life); +} + +CasRefCatalog::Snapshot readCatalogForTest(const BackendPtr & backend, const Layout & layout) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return CasRefCatalog::read(op, layout); +} + +/// A fixture's own conditional replace, for the races these tests stage by hand. +bool replaceForTest(const BackendPtr & backend, const String & key, const String & bytes, + const Etag & expected) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return std::holds_alternative(op.replace(key, bytes, expected, Retry::standard())); +} + /// Make the durable mount immediately reclaimable so a test that deliberately moved the local fence /// generation can drive the production remount boundary without paying a live-lease expiry wait. void fenceOutMountForRemount(Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + OperationForTest op(backend); + const auto got = (*op).read(mount_key, Retry::once()); ASSERT_TRUE(got.has_value()); MountLease mount = decodeMountLease(got->bytes); mount.gc_fenced = true; mount.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(mount), got->token).outcome, - PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(mount_key, encodeMountLease(mount), got->etag, Retry::once()))); +} + +/// The durable object at `key`, or `nullopt`. +std::optional readAt(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::once()); +} + +/// Unconditional create of a fresh key (the fixture's own setup, never a real conflict). +void createAt(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + EXPECT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); } /// A backend whose `LIST` can lie by omission. `hidden_keys` remain readable by exact key, so these @@ -111,35 +155,42 @@ class HidingListBackend : public CountingBackend DB::Cas::tests::seedPoolMetaForRestart(*this); } - using CountingBackend::get; - using CountingBackend::list; - using CountingBackend::putIfAbsent; - using CountingBackend::casPut; - std::set hidden_keys; std::set phantom_list_keys; - /// Every `putIfAbsent` of a key containing this substring throws a PLAIN (non-`DB::Exception`) - /// error, which `classifyConditionalWriteResult` can only ever classify `Unresolved` -- never - /// `DefiniteFailure`. Persistent rather than one-shot on purpose: the subject is what recovery does - /// when the store KEEPS refusing to say whether the write landed. + /// Every CREATING write of a key containing this substring throws `Poco::TimeoutException`, the + /// class the request engine classifies as an unresolved transport fault and reissues under + /// `Retry::standard()` -- a plain `std::exception` is instead the engine's signal for "this could + /// not have landed" and propagates on the FIRST attempt (`CasRequests.cpp`'s + /// `!dynamic_cast(&e)` arms), which is a proven-not-landed verdict, not the + /// ambiguity this fixture means to model. Persistent rather than one-shot on purpose: the subject is + /// what recovery does when the store KEEPS refusing to say whether the write landed. String ambiguous_put_substr; - /// Persistent thrown response for a matching mutable checkpoint CAS. The ref-log PUT has already - /// completed when tests arm this, producing the exact one-successor recovery window. + /// Every attempt the `ambiguous_put_substr` fault intercepted, counted here because the throw below + /// happens before delegating to `CountingBackend::write` -- its own per-key counters never see a + /// faulted attempt at all. A test proves the engine actually reissued (rather than giving up after + /// one attempt) by reading this after the call. + std::atomic ambiguous_put_attempts{0}; + + /// Persistent thrown response for a matching CONDITIONAL replace of the mutable checkpoint. The + /// ref-log create has already completed when tests arm this, producing the exact one-successor + /// recovery window. String ambiguous_cas_substr; int ambiguous_cas_count = 0; - /// Runs after a checkpoint publisher read its expected token but before that publisher presents - /// its CAS. This is the exact window in which another admitted writer can advance the frontier. - std::function &)> before_cas_put; + /// Runs after a checkpoint publisher read the incarnation it expects but before that publisher + /// presents its conditional write. This is the exact window in which another admitted writer can + /// advance the frontier. + std::function &)> before_cas_put; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { - ListPage page = CountingBackend::list(prefix, cursor, limit); - std::vector kept; + DB::Cas::Backend::RawListPage page = CountingBackend::list(prefix, cursor, limit, access); + std::vector kept; kept.reserve(page.keys.size()); - for (ListedKey & lk : page.keys) + for (DB::Cas::Backend::RawListedKey & lk : page.keys) if (!hidden_keys.contains(lk.key)) kept.push_back(std::move(lk)); if (cursor.empty()) @@ -147,32 +198,38 @@ class HidingListBackend : public CountingBackend for (const String & key : phantom_list_keys) { if (key.starts_with(prefix)) - kept.push_back(ListedKey{.key = key, .size = 0, .token = std::nullopt}); + kept.push_back(DB::Cas::Backend::RawListedKey{.key = key, .size = 0, .value = std::nullopt}); } } page.keys = std::move(kept); return page; } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override - { - if (!ambiguous_put_substr.empty() && key.find(ambiguous_put_substr) != String::npos) - throw std::runtime_error("injected ambiguous putIfAbsent"); - return CountingBackend::putIfAbsent(key, bytes, meta); - } - - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// Both write faults hang off the ONE keyed primitive every caller now reaches the store through; + /// which of them applies is decided by whether the write carries a precondition, which is exactly + /// what used to separate `putIfAbsent` from `casPut`. + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { + if (!expected_value) + { + if (!ambiguous_put_substr.empty() && key.find(ambiguous_put_substr) != String::npos) + { + ambiguous_put_attempts.fetch_add(1); + throw Poco::TimeoutException("injected ambiguous create"); + } + return CountingBackend::write(key, bytes, expected_value, access); + } if (before_cas_put) - before_cas_put(key, bytes, expected); + before_cas_put(key, bytes, expected_value); if (ambiguous_cas_count > 0 && !ambiguous_cas_substr.empty() && key.find(ambiguous_cas_substr) != String::npos) { --ambiguous_cas_count; - throw Poco::TimeoutException("HidingListBackend: simulated ambiguous checkpoint CAS"); + throw Poco::TimeoutException("HidingListBackend: simulated ambiguous checkpoint replace"); } - return CountingBackend::casPut(key, bytes, expected, meta); + return CountingBackend::write(key, bytes, expected_value, access); } }; @@ -182,28 +239,18 @@ class HidingListBackend : public CountingBackend class PutHookBackend : public HidingListBackend { public: - using HidingListBackend::putIfAbsent; - - using HidingListBackend::casPut; - String watched_substr; uint64_t skip = 0; std::function on_key; - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override - { - PutResult result = HidingListBackend::putIfAbsent(key, bytes, meta); - fireIfWatched(key); - return result; - } - - /// The `_ckpt` advance is a token-CAS, not a create, whenever the object already exists -- which is - /// the normal case, since the namespace birth creates it. Hooking only `putIfAbsent` would silently - /// never fire for it. - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// One override covers both write shapes: the `_ckpt` advance is a conditional replace, not a + /// create, whenever the object already exists -- which is the normal case, since the namespace + /// birth creates it -- so hooking only creates would silently never fire for it. + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - CasResult result = HidingListBackend::casPut(key, bytes, expected, meta); + auto result = HidingListBackend::write(key, bytes, expected_value, access); fireIfWatched(key); return result; } @@ -235,17 +282,15 @@ class PutHookBackend : public HidingListBackend class LateMaterializeBackend : public HidingListBackend { public: - using HidingListBackend::get; - String late_key; String late_bytes; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - std::optional result = HidingListBackend::get(key, range); + std::optional result = HidingListBackend::read(key, access); if (!result && !late_key.empty() && key == late_key) { - CountingBackend::putIfAbsent(late_key, late_bytes); + (void)CountingBackend::write(late_key, late_bytes, std::nullopt, access); late_key.clear(); /// one-shot: the walk must see it present from here on } return result; @@ -258,8 +303,6 @@ class LateMaterializeBackend : public HidingListBackend class GetSeamBackend : public HidingListBackend { public: - using HidingListBackend::get; - String watched_substr; /// Assigned from the test thread and read from whatever thread the recovery runs on, so the @@ -270,7 +313,7 @@ class GetSeamBackend : public HidingListBackend std::mutex hook_mutex; std::function on_key; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { std::unique_lock hook_lock(hook_mutex); if (on_key && !watched_substr.empty() && key.find(watched_substr) != String::npos) @@ -288,7 +331,7 @@ class GetSeamBackend : public HidingListBackend hook_lock.unlock(); hook(key); } - return HidingListBackend::get(key, range); + return HidingListBackend::read(key, access); } }; @@ -298,14 +341,12 @@ class GetSeamBackend : public HidingListBackend class AfterGetHookBackend : public HidingListBackend { public: - using HidingListBackend::get; - String watched_key; std::function after_get; - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - std::optional result = HidingListBackend::get(key, range); + std::optional result = HidingListBackend::read(key, access); if (after_get && key == watched_key) { auto hook = std::move(after_get); @@ -316,10 +357,17 @@ class AfterGetHookBackend : public HidingListBackend } }; + +/// More injected failures than the engine's own retry window can make attempts, on a clock that +/// advances at least a millisecond per pause: a call meeting this fault must end at its DEADLINE, +/// never by outliving the fault. A bounded count would be spent by ONE call's own reissues, because +/// the engine settles each ambiguity by an exact read and then reissues. +constexpr int kFaultsBeyondTheRetryWindow = 100'000; + CasRequestBudget tinyBudget() { return CasRequestBudget{ - .attempt_timeout_ms = 50, .operation_deadline_ms = 500, .max_attempts = 1, .lease_safety_margin_ms = 50}; + .attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; } PoolConfig walkTestConfig() @@ -337,9 +385,13 @@ PoolConfig walkTestConfig() return config; } -PoolPtr openWalkPool(const BackendPtr & backend, PoolConfig config = walkTestConfig()) +template +PoolPtr openWalkPool(const std::shared_ptr & backend, PoolConfig config = walkTestConfig()) { DB::Cas::tests::seedPoolMetaForRestart(*backend, config.pool_prefix); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the fence math the recovery walk drives matches what admits. + backend->setAttemptTimeoutMs(config.cas_request_budget.attempt_timeout_ms); return Pool::open(backend, std::move(config)); } @@ -347,10 +399,12 @@ PoolPtr openWalkPool(const BackendPtr & backend, PoolConfig config = walkTestCon /// minted, never reclaimed (`CasPool.cpp`'s allocator), so this is exactly what a pool that has been /// mounted `n` times looks like -- including the burned epochs in which nothing was ever written, which /// the seal chain must cross. -void burnEpochsUpTo(Backend & backend, const Layout & layout, uint64_t target_live_epoch) +void burnEpochsUpTo(const BackendPtr & backend, const Layout & layout, uint64_t target_live_epoch) { + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); for (uint64_t e = 1; e < target_live_epoch; ++e) - allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); + allocateWriterEpoch(op, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); } /// One ordinary transaction at `id`, publishing `ref` (prepending the birth op when `birth`). @@ -402,7 +456,7 @@ void seedTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, /// never run a birth through the append lane. void seedCkpt(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefCkpt & ckpt) { - backend.putIfAbsent(layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(ckpt)); + createAt(backend, layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)), encodeRefCkpt(ckpt)); } RefCkpt lifeEpochCkpt(uint64_t life_epoch, std::optional committed_through = std::nullopt) @@ -417,7 +471,7 @@ RefCkpt lifeEpochCkpt(uint64_t life_epoch, std::optional committed_thr /// disengaged optional: an aborted binary would take every later suite's result with it. std::optional readLogTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, RefTxnId id) { - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + const auto got = readAt(backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); if (!got) return std::nullopt; return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id); @@ -428,9 +482,9 @@ uint64_t counterOf(ProfileEvents::Event event) return ProfileEvents::global_counters[event].load(); } -NamespaceLifeId catalogLife(Backend & backend, const Layout & layout, const RootNamespace & ns) +NamespaceLifeId catalogLife(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) { - const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot catalog = readCatalogForTest(backend, layout); for (const CatalogEntry & entry : catalog.catalog.entries) if (entry.ns == ns) return NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation); @@ -438,7 +492,8 @@ NamespaceLifeId catalogLife(Backend & backend, const Layout & layout, const Root } NamespaceLifeId strandOneUnfrontieredSuccessor( - HidingListBackend & backend, const PoolPtr & store, const Layout & layout, const RootNamespace & ns) + const std::shared_ptr & backend, const PoolPtr & store, const Layout & layout, + const RootNamespace & ns) { store->appendRefOps(ns, MutationScope::ref("a"), [](const RefTableState & state) @@ -451,30 +506,39 @@ NamespaceLifeId strandOneUnfrontieredSuccessor( return ops; }, RootMutationOrigin::Writer, RootMutationKind::Publish); + /// The frontier publication reissues an ambiguous replace until its own retry window closes, so + /// the fault has to outlast the call and the clock the window is read from has to move for the call + /// to end at all. + auto clock = VirtualRetryClock::installOn(store); const NamespaceLifeId life = catalogLife(backend, layout, ns); - backend.ambiguous_cas_substr = layout.refCkptKey(life); - backend.ambiguous_cas_count = 200; + backend->ambiguous_cas_substr = layout.refCkptKey(life); + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->appendRefOps(ns, MutationScope::ref("b"), [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, RootMutationOrigin::Writer, RootMutationKind::Publish); }); - backend.ambiguous_cas_count = 0; + backend->ambiguous_cas_count = 0; + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; + EXPECT_LE(clock->longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; return life; } CatalogEntry replaceCatalogLifeForTest( - Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) + const BackendPtr & backend, const Layout & layout, const CatalogEntry & predecessor, + UInt128 successor_incarnation) { - const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot before_delete = readCatalogForTest(backend, layout); RefCatalog without_predecessor = before_delete.catalog; std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) { return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; }); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome - != CasOutcome::Committed) + if (!before_delete.etag + || !replaceForTest(backend, layout.refCatalogKey(), encodeRefCatalog(without_predecessor), + *before_delete.etag)) throw std::runtime_error("test failed to retire exact predecessor catalog life"); CatalogEntry successor{ @@ -482,11 +546,11 @@ CatalogEntry replaceCatalogLifeForTest( .state = NsState::Live, .incarnation = successor_incarnation, .creator = std::nullopt}; - const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot after_delete = readCatalogForTest(backend, layout); RefCatalog reborn = after_delete.catalog; reborn.entries.push_back(successor); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome - != CasOutcome::Committed) + if (!after_delete.etag + || !replaceForTest(backend, layout.refCatalogKey(), encodeRefCatalog(reborn), *after_delete.etag)) throw std::runtime_error("test failed to publish successor catalog life"); return successor; } @@ -585,22 +649,22 @@ TEST(CASRefRecoveryCasWalk, MissingExactIdAtOrBelowCommittedFrontierIsCorruption const RefTxnId frontier{1, 2}; DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + createAt(*backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = std::optional{1}, .committed_through = frontier, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); - const auto ckpt_before = readCkpt(*backend, layout, life); + .last_epoch_seal = std::nullopt})); + const auto ckpt_before = readCkptForTest(backend, layout, life); ASSERT_TRUE(ckpt_before); auto store = openWalkPool(backend); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); - const auto ckpt_after = readCkpt(*backend, layout, life); + const auto ckpt_after = readCkptForTest(backend, layout, life); ASSERT_TRUE(ckpt_after); - EXPECT_EQ(ckpt_after->token, ckpt_before->token) + EXPECT_EQ(ckpt_after->etag, ckpt_before->etag) << "an unchanged checkpoint makes the missing committed id corruption, not a shorter stream"; } @@ -613,16 +677,16 @@ TEST(CASRefRecoveryCasWalk, UncommittedSnapshotIsUnobservedWithoutStreamList) const RefTxnId uncommitted_snapshot_id{1, 2}; DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); seedTxn(*backend, layout, ns, frontier, "committed", /*birth=*/true); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), uncommitted_snapshot_id, {committedRow("laundered", manifestRef(1, 2, 1))})); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + createAt(*backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = std::optional{1}, .committed_through = frontier, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt})); auto store = openWalkPool(backend); backend->resetCounts(); @@ -652,30 +716,36 @@ TEST(CASRefRecoveryCasWalk, ListingShapeDoesNotAffectCheckpointRecovery) .committed_through = frontier, .checkpoint_snapshot_id = base, .last_epoch_seal = std::nullopt}); - const NamespaceLifeId life = catalogLife(*seed, layout, ns); + const NamespaceLifeId life = catalogLife(seed, layout, ns); const auto clone_seed = [&]() -> std::shared_ptr { /// A clone starts empty: constructing the normal fixture would pre-seed independent pool-meta /// bytes before this loop could copy the source's identical durable image. auto backend = std::make_shared(/*seed_pool_meta=*/false); + CasRequests seed_requests(seed, Fence::open()); + CasOperation seed_op = seed_requests.admit(); String cursor; do { - const ListPage page = seed->list("", cursor, 1000); + const ListPage page = seed_op.list("", cursor, 1000, Retry::standard()); for (const ListedKey & listed : page.keys) { - const auto object = seed->get(listed.key); + const auto object = seed_op.read(listed.key, Retry::standard()); if (!object) throw std::runtime_error("seed LIST returned a key that exact GET could not read"); - const auto existing = backend->get(listed.key); + const auto existing = readAt(*backend, listed.key); if (existing) { - if (existing->bytes != object->bytes || existing->attributes != object->attributes) + if (existing->bytes != object->bytes) throw std::runtime_error("clone backend constructor disagreed with seeded object"); } - else if (backend->putIfAbsent(listed.key, object->bytes, object->attributes).outcome != PutOutcome::Done) - throw std::runtime_error("clone backend failed to copy seeded object"); + else + { + OperationForTest clone_op(*backend); + if (!std::holds_alternative((*clone_op).create(listed.key, object->bytes, Retry::once()))) + throw std::runtime_error("clone backend failed to copy seeded object"); + } } cursor = page.next_cursor; } while (!cursor.empty()); @@ -724,7 +794,7 @@ TEST(CASRefRecoveryCasWalk, PhantomListedSnapshotIsUnobserved) .committed_through = frontier, .checkpoint_snapshot_id = checkpoint_base, .last_epoch_seal = std::nullopt}); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); backend->phantom_list_keys.insert(layout.refSnapshotKey(life, frontier)); auto store = openWalkPool(backend); @@ -762,10 +832,10 @@ TEST(CASRefRecoveryCasWalk, DuplicateCatalogLifeIsCorruptionBeforeColdRuntimeAdm const Layout layout("p"); const RootNamespace ns{"srv1/ambiguous_life_a"}; auto store = openWalkPool(backend); - const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(backend, store, layout, ns); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); - const CasRefCatalog::Snapshot sampled = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot sampled = readCatalogForTest(backend, layout); ASSERT_EQ(sampled.catalog.entries.size(), 1u); RefCatalog ambiguous = sampled.catalog; ambiguous.entries.push_back(CatalogEntry{ @@ -774,8 +844,8 @@ TEST(CASRefRecoveryCasWalk, DuplicateCatalogLifeIsCorruptionBeforeColdRuntimeAdm .incarnation = life.incarnation}); std::sort(ambiguous.entries.begin(), ambiguous.entries.end(), [](const CatalogEntry & lhs, const CatalogEntry & rhs) { return lhs.ns.string() < rhs.ns.string(); }); - ASSERT_EQ(backend->casPut(layout.refCatalogKey(), encodeRefCatalog(ambiguous), sampled.token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(sampled.etag); + ASSERT_TRUE(replaceForTest(backend, layout.refCatalogKey(), encodeRefCatalog(ambiguous), *sampled.etag)); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { @@ -794,17 +864,16 @@ TEST(CASRefRecoveryCasWalk, CheckpointAdvanceAfterLastLogProbeRestartsBeforeInst seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); backend->watched_key = layout.refLogKey(life, concurrent_frontier); backend->after_get = [&] { seedTxn(*backend, layout, ns, concurrent_frontier, "b", /*birth=*/false); - const auto sampled = readCkpt(*backend, layout, life); + const auto sampled = readCkptForTest(backend, layout, life); ASSERT_TRUE(sampled); RefCkpt advanced = sampled->ckpt; advanced.committed_through = concurrent_frontier; - ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(advanced), sampled->token).outcome, - CasOutcome::Committed); + ASSERT_TRUE(replaceForTest(backend, layout.refCkptKey(life), encodeRefCkpt(advanced), sampled->etag)); }; auto store = openWalkPool(backend); @@ -825,9 +894,9 @@ TEST(CASRefRecoveryCasWalk, LiveCatalogLifeWithoutReadableCheckpointIsCorruption const RootNamespace ns{"srv1/live_without_ckpt"}; DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); seedTxn(*backend, layout, ns, RefTxnId{7, 1}, "hint-must-not-be-genesis", /*birth=*/true); - ASSERT_FALSE(readCkpt(*backend, layout, life)); + ASSERT_FALSE(readCkptForTest(backend, layout, life)); auto store = openWalkPool(backend); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); @@ -868,7 +937,7 @@ TEST(CASRefRecoveryCasWalk, DeadEpochIsClosedByOurOwnSealAtTPlusOne) const Layout layout("p"); const RootNamespace ns{"srv1/seal_created"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -898,7 +967,7 @@ TEST(CASRefRecoveryCasWalk, ConcurrentRecoverersSealIsAdoptedNotContested) const Layout layout("p"); const RootNamespace ns{"srv1/seal_adopt"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); /// The peer's seal lands between our read of {1,2} and our create of it, so we meet it as an @@ -926,7 +995,7 @@ TEST(CASRefRecoveryCasWalk, StragglerAtTPlusOneIsAdoptedAndResealedAtTheNewTPlus const Layout layout("p"); const RootNamespace ns{"srv1/straggler"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); /// The dying epoch's last append materializes between our read of {1,2} and our create of it -- the @@ -973,33 +1042,42 @@ TEST(CASRefRecoveryCasWalk, RecoveryPublishesEveryOccupiedObjectBeforeAdvancingP const RootNamespace ns{"srv1/occupied_frontier_" + test_case.suffix}; const RefTxnId initial_frontier{1, 1}; - burnEpochsUpTo(*backend, layout, test_case.live_epoch); + burnEpochsUpTo(backend, layout, test_case.live_epoch); seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); backend->late_key = layout.refLogKey(life, test_case.occupant); const RefLogTxn occupant = test_case.occupant_is_seal ? makeSealTxn(ns, test_case.occupant) : makeOrdinaryTxn(ns, test_case.occupant, "late", /*birth=*/false); backend->late_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(occupant)); - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: the retry-sleep hook below mutates it, and the + /// Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config = walkTestConfig(); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; config.cas_request_budget.recovery_retry_budget_ms = 1; config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; config.cas_request_budget.recovery_retry_max_backoff_ms = 1; auto store = openWalkPool(backend, config); - store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + store->setCasRetrySleepForTest([fake_now](uint64_t ms) + { + *fake_now += ms; + }); backend->ambiguous_cas_substr = layout.refCkptKey(life); - backend->ambiguous_cas_count = 100'000; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); - EXPECT_TRUE(backend->get(layout.refLogKey(life, test_case.occupant))); - EXPECT_FALSE(backend->get(layout.refLogKey(life, test_case.forbidden_successor))) + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, test_case.occupant))); + EXPECT_FALSE(readAt(*backend, layout.refLogKey(life, test_case.forbidden_successor))) << "recovery advanced before exact _ckpt certified the occupied object"; - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, initial_frontier); EXPECT_FALSE(store->refTableRecoveredForTest(ns)); } } @@ -1015,7 +1093,7 @@ TEST(CASRefRecoveryCasWalk, TwoBurnedEmptyEpochsProduceTwoChainedSequenceOneSeal const Layout layout("p"); const RootNamespace ns{"srv1/burned"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/4); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/4); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1054,30 +1132,39 @@ TEST(CASRefRecoveryCasWalk, RecoveryPublishesEachCreatedSealBeforeCreatingTheNex const RefTxnId second_seal{2, 1}; const RefTxnId cold_remount_frontier{3, 1}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/3); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/3); seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: the retry-sleep hook below mutates it, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config = walkTestConfig(); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; config.cas_request_budget.recovery_retry_budget_ms = 1; config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; config.cas_request_budget.recovery_retry_max_backoff_ms = 1; auto store = openWalkPool(backend, config); ASSERT_EQ(store->liveWriterEpoch(), 3u); - store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + store->setCasRetrySleepForTest([fake_now](uint64_t ms) + { + *fake_now += ms; + }); backend->ambiguous_cas_substr = layout.refCkptKey(life); - backend->ambiguous_cas_count = 100'000; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); - EXPECT_TRUE(backend->get(layout.refLogKey(life, first_seal))) + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, first_seal))) << "the first recovery seal became durable before its frontier attempt"; - EXPECT_FALSE(backend->get(layout.refLogKey(life, second_seal))) + EXPECT_FALSE(readAt(*backend, layout.refLogKey(life, second_seal))) << "recovery may not create a second object while the first is still above exact _ckpt"; - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, initial_frontier); EXPECT_FALSE(store->refTableRecoveredForTest(ns)); /// Restart cold, without the failed mount's `NeedsRecovery` attempt. The first seal is durable but @@ -1087,9 +1174,9 @@ TEST(CASRefRecoveryCasWalk, RecoveryPublishesEachCreatedSealBeforeCreatingTheNex auto cold_store = openWalkPool(backend); ASSERT_EQ(cold_store->liveWriterEpoch(), 4u); ASSERT_EQ(cold_store->listRefs(ns).size(), 1u); - EXPECT_TRUE(backend->get(layout.refLogKey(life, second_seal))); - EXPECT_TRUE(backend->get(layout.refLogKey(life, cold_remount_frontier))); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, cold_remount_frontier); + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, second_seal))); + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, cold_remount_frontier))); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, cold_remount_frontier); } /// A straggler is not an exception to the recovered-successor rule. When it materializes in the seal @@ -1104,38 +1191,47 @@ TEST(CASRefRecoveryCasWalk, RecoveryPublishesAnAdoptedStragglerBeforeCreatingIts const RefTxnId straggler{1, 2}; const RefTxnId following_seal{1, 3}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedTxn(*backend, layout, ns, initial_frontier, "a", /*birth=*/true); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, initial_frontier)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); backend->late_key = layout.refLogKey(life, straggler); backend->late_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(makeOrdinaryTxn(ns, straggler, "late", /*birth=*/false))); - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: the retry-sleep hook below mutates it, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config = walkTestConfig(); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; config.cas_request_budget.recovery_retry_budget_ms = 1; config.cas_request_budget.recovery_retry_initial_backoff_ms = 1; config.cas_request_budget.recovery_retry_max_backoff_ms = 1; auto store = openWalkPool(backend, config); - store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + store->setCasRetrySleepForTest([fake_now](uint64_t ms) + { + *fake_now += ms; + }); backend->ambiguous_cas_substr = layout.refCkptKey(life); - backend->ambiguous_cas_count = 100'000; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->listRefs(ns); }); - EXPECT_TRUE(backend->get(layout.refLogKey(life, straggler))) + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, straggler))) << "the straggler occupied the recovery seal slot"; - EXPECT_FALSE(backend->get(layout.refLogKey(life, following_seal))) + EXPECT_FALSE(readAt(*backend, layout.refLogKey(life, following_seal))) << "recovery may not create a seal after an adopted straggler above exact _ckpt"; - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, initial_frontier); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, initial_frontier); EXPECT_FALSE(store->refTableRecoveredForTest(ns)); backend->ambiguous_cas_count = 0; ASSERT_EQ(store->listRefs(ns).size(), 2u); - EXPECT_TRUE(backend->get(layout.refLogKey(life, following_seal))); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, following_seal); + EXPECT_TRUE(readAt(*backend, layout.refLogKey(life, following_seal))); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, following_seal); } /// GENESIS. A namespace born at epoch 5 has no epochs 1-4 of its own: they are not "empty epochs it @@ -1149,7 +1245,7 @@ TEST(CASRefRecoveryCasWalk, GenesisAtEpochFiveWritesNoPhantomSealsBelowLifeEpoch const Layout layout("p"); const RootNamespace ns{"srv1/genesis5"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/5); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/5); seedCkpt(*backend, layout, ns, lifeEpochCkpt(5, RefTxnId{5, 1})); seedTxn(*backend, layout, ns, RefTxnId{5, 1}, "a", /*birth=*/true); @@ -1210,13 +1306,13 @@ TEST(CASRefRecoveryCasWalk, RetiredLifePausedInRealRecoveryIoWritesAndInstallsNo const Layout layout("p"); const RootNamespace ns{"srv1/recovery-retired-mid-io"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "predecessor", /*birth=*/true); - const CatalogEntry predecessor = CasRefCatalog::read(*backend, layout).catalog.entries.front(); + const CatalogEntry predecessor = readCatalogForTest(backend, layout).catalog.entries.front(); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(predecessor.ns, predecessor.incarnation); - const auto predecessor_ckpt_before = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_before = readAt(*backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_before); auto store = openWalkPool(backend); @@ -1253,12 +1349,11 @@ TEST(CASRefRecoveryCasWalk, RetiredLifePausedInRealRecoveryIoWritesAndInstallsNo cv.wait(lock, [&] { return paused; }); } - const CatalogEntry successor = replaceCatalogLifeForTest(*backend, layout, predecessor, UInt128{0x5152}); + const CatalogEntry successor = replaceCatalogLifeForTest(backend, layout, predecessor, UInt128{0x5152}); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(lifeEpochCkpt(2))).outcome, - PutOutcome::Done); - const auto successor_ckpt_before = backend->get(layout.refCkptKey(successor_life)); + createAt(*backend, layout.refCkptKey(successor_life), encodeRefCkpt(lifeEpochCkpt(2))); + const auto successor_ckpt_before = readAt(*backend, layout.refCkptKey(successor_life)); ASSERT_TRUE(successor_ckpt_before); store->invalidateRemovedCatalogLife(predecessor_life); backend->resetCounts(); @@ -1273,11 +1368,11 @@ TEST(CASRefRecoveryCasWalk, RetiredLifePausedInRealRecoveryIoWritesAndInstallsNo EXPECT_TRUE(recovery_error) << "the predecessor recovery must be refused, not exposed"; EXPECT_EQ(backend->putCount(layout.refLogKey(predecessor_life, RefTxnId{1, 2})), 0u) << "no predecessor seal retry may be sent after exact retirement"; - EXPECT_EQ(backend->casPutCount(layout.refCkptKey(predecessor_life)), 0u) + EXPECT_EQ(backend->writeCount(layout.refCkptKey(predecessor_life)), 0u) << "no predecessor checkpoint CAS may be sent after exact retirement"; - const auto predecessor_ckpt_after = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_after = readAt(*backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_after); - EXPECT_EQ(predecessor_ckpt_after->token, predecessor_ckpt_before->token); + EXPECT_EQ(predecessor_ckpt_after->etag, predecessor_ckpt_before->etag); EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "the detached predecessor result was installed"; EXPECT_EQ(store->recoveryInstallCountForTest(), recovery_installs_before) << "the detached predecessor reached the recovery publication point"; @@ -1285,9 +1380,9 @@ TEST(CASRefRecoveryCasWalk, RetiredLifePausedInRealRecoveryIoWritesAndInstallsNo for (const String & key : backend->touchedKeys()) EXPECT_EQ(key.find(successor_prefix), String::npos) << "predecessor recovery retargeted storage I/O into successor key " << key; - const auto successor_ckpt_after = backend->get(layout.refCkptKey(successor_life)); + const auto successor_ckpt_after = readAt(*backend, layout.refCkptKey(successor_life)); ASSERT_TRUE(successor_ckpt_after); - EXPECT_EQ(successor_ckpt_after->token, successor_ckpt_before->token); + EXPECT_EQ(successor_ckpt_after->etag, successor_ckpt_before->etag); EXPECT_EQ(successor_ckpt_after->bytes, successor_ckpt_before->bytes); } @@ -1301,14 +1396,14 @@ TEST(CASRefRecoveryCasWalk, FenceBumpedAfterSlotOccupyBeforeCkptCasAdvancesNoChe const Layout layout("p"); const RootNamespace ns{"srv1/bump_after_seal"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); auto store = openWalkPool(backend); ASSERT_TRUE(store); - const auto ckpt_before = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_before = readCkptForTest(backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); ASSERT_TRUE(ckpt_before.has_value()); backend->watched_substr = "_log/"; @@ -1316,11 +1411,11 @@ TEST(CASRefRecoveryCasWalk, FenceBumpedAfterSlotOccupyBeforeCkptCasAdvancesNoChe EXPECT_ANY_THROW(store->listRefs(ns)); - const auto ckpt_after = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_after = readCkptForTest(backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->ckpt.last_epoch_seal, std::nullopt) << "the seal is durable but the checkpoint must not record it under a generation that moved"; - EXPECT_EQ(ckpt_after->token, ckpt_before->token) << "no CAS was sent at all"; + EXPECT_EQ(ckpt_after->etag, ckpt_before->etag) << "no CAS was sent at all"; } /// Bump point 2: AFTER the `_ckpt` CAS, BEFORE the install. The checkpoint advance is harmless (the @@ -1333,7 +1428,7 @@ TEST(CASRefRecoveryCasWalk, FenceBumpedAfterCkptCasBeforeInstallPublishesNoState const Layout layout("p"); const RootNamespace ns{"srv1/bump_after_ckpt"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1347,7 +1442,7 @@ TEST(CASRefRecoveryCasWalk, FenceBumpedAfterCkptCasBeforeInstallPublishesNoState EXPECT_ANY_THROW(store->listRefs(ns)) << "the install recheck must refuse a result from a moved generation"; - const auto ckpt_after = readCkpt(*backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); + const auto ckpt_after = readCkptForTest(backend, layout, DB::Cas::tests::fixture::fixtureLife(ns)); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->ckpt.last_epoch_seal, std::optional(RefTxnId{1, 2})) << "the checkpoint advance already landed and is harmless -- the merge is a semantic maximum"; @@ -1378,7 +1473,7 @@ TEST(CASRefRecoveryCasWalk, RemountBarrierBlocksUntilAPausedRecoveryAcknowledges const Layout layout("p"); const RootNamespace ns{"srv1/remount_barrier"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1445,6 +1540,82 @@ TEST(CASRefRecoveryCasWalk, RemountBarrierBlocksUntilAPausedRecoveryAcknowledges EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "and ZERO installs"; } +/// The cancellation reaches the walk's REQUESTS, not just its own polls. `readCheckpointSnapshotBase` +/// issues several reads back to back -- the base log, then the snapshot body -- and the walk's poll runs +/// only before the call, so a cancellation landing between those two reads used to be invisible until +/// the whole call returned. The walk's operation now carries the cancellation in its liveness, so the +/// next request is the one that refuses. +/// +/// Parked on the base-log read, which is the FIRST request that call makes, so the snapshot body read +/// is the one that must never happen. +TEST(CASRefRecoveryCasWalk, CancellationStopsTheWalkBetweenTwoReadsOfOneCall) +{ + auto backend = std::make_shared(); + const Layout layout("p"); + const RootNamespace ns{"srv1/cancel_between_reads"}; + const RefTxnId base{1, 1}; + const RefTxnId frontier{1, 2}; + + DB::Cas::tests::fixture::admitLive(*backend, layout, ns); + seedTxn(*backend, layout, ns, base, "a", /*birth=*/true); + writeRefSnapshotRaw(*backend, layout, + minimalLiveSnapshot(ns.string(), base, {committedRow("a", manifestRef(1, 1, 1))})); + seedTxn(*backend, layout, ns, frontier, "b", /*birth=*/false); + seedCkpt(*backend, layout, ns, RefCkpt{ + .life_epoch = std::optional{1}, + .committed_through = frontier, + .checkpoint_snapshot_id = base, + .last_epoch_seal = std::nullopt}); + const NamespaceLifeId life = catalogLife(backend, layout, ns); + + auto store = openWalkPool(backend); + ASSERT_TRUE(store); + + std::mutex m; + std::condition_variable cv; + bool recovery_parked = false; + bool release_recovery = false; + + backend->watched_substr = "_log/"; + /// `GetSeamBackend` moves the hook out before calling it, so this parks exactly once. + backend->on_key = [&](const String &) + { + std::unique_lock lock(m); + recovery_parked = true; + cv.notify_all(); + cv.wait(lock, [&] { return release_recovery; }); + }; + + std::thread recovery([&] { try { store->listRefs(ns); } catch (...) {} }); // NOLINT(bugprone-empty-catch): the outcome is asserted below through the request counts + + { + std::unique_lock lock(m); + cv.wait(lock, [&] { return recovery_parked; }); + } + + std::thread barrier([&] { store->cancelRefRecoveriesAndAwaitQuiescence(); }); + /// Wait for the REQUEST to be visible before releasing: releasing any earlier would race the walk + /// past a flag set a moment too late, and the test would read an ordinary completion as a + /// cancellation that never happened. + while (!store->refRecoveryCancelRequestedForTest(ns)) + std::this_thread::yield(); + + const uint64_t snapshot_reads_before = backend->getCount(layout.refSnapshotKey(life, base)); + ASSERT_EQ(snapshot_reads_before, 0u) << "the parked read is the base LOG read, before the body read"; + + { + std::lock_guard lock(m); + release_recovery = true; + } + cv.notify_all(); + barrier.join(); + recovery.join(); + + EXPECT_EQ(backend->getCount(layout.refSnapshotKey(life, base)), 0u) + << "the cancellation must refuse the very next request of the same call, not be noticed after it"; + EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "and nothing is installed"; +} + /// A `NeedsRecovery` lane replays the known-durable transaction before returning to `Ready`. TEST(CASRefRecoveryCasWalk, NeedsRecoveryReplaysTheStrandedTxn) { @@ -1495,14 +1666,14 @@ TEST(CASRefRecoveryCasWalk, NeedsRecoveryReplaysTheStrandedTxn) /// file's OTHER tests use -- so its ref-layer objects sit at a REAL, catalog-minted incarnation, /// not the Stage-A sentinel `readLogTxn` assumes. Resolved here rather than through `readLogTxn`. { - const CasRefCatalog::Snapshot snap = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot snap = readCatalogForTest(backend, layout); const CatalogEntry * entry = nullptr; for (const CatalogEntry & e : snap.catalog.entries) if (e.ns.string() == ns.string()) entry = &e; ASSERT_NE(entry, nullptr) << "the birth above must have minted a catalog entry for " << ns.string(); const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(entry->ns, entry->incarnation); - ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 2})).has_value()) + ASSERT_TRUE(readAt(*backend, layout.refLogKey(life, RefTxnId{1, 2})).has_value()) << "the stranded transaction must be durable -- otherwise recovery is not owed"; } @@ -1532,11 +1703,12 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsOneExactUnfrontieredSuccessorAnd return ops; }, RootMutationOrigin::Writer, RootMutationKind::Publish)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); const String ckpt_key = layout.refCkptKey(life); - ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + ASSERT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + auto clock = VirtualRetryClock::installOn(store); backend->ambiguous_cas_substr = ckpt_key; - backend->ambiguous_cas_count = 200; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { @@ -1544,17 +1716,19 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsOneExactUnfrontieredSuccessorAnd [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, RootMutationOrigin::Writer, RootMutationKind::Publish); }); + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); - ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 2}))) + ASSERT_TRUE(readAt(*backend, layout.refLogKey(life, RefTxnId{1, 2}))) << "the sole deterministic successor must be durable before recovery"; - ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + ASSERT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); backend->ambiguous_cas_count = 0; const auto refs = store->listRefs(ns); EXPECT_TRUE(refs.contains("b")); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 2})) + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 2})) << "the successor is not installable until the current admitted fence publishes its frontier"; } @@ -1564,11 +1738,11 @@ TEST(CASRefRecoveryCasWalk, ColdWriterRecoveryPublishesOneExactUnfrontieredSucce const Layout layout("p"); const RootNamespace ns{"srv1/recovery_cold_successor"}; auto store = openWalkPool(backend); - const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(backend, store, layout, ns); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); store.reset(); std::vector checkpoint_cas_bodies; - backend->before_cas_put = [&](const String & key, const String & bytes, const std::optional &) + backend->before_cas_put = [&](const String & key, const String & bytes, const std::optional &) { if (key == layout.refCkptKey(life)) checkpoint_cas_bodies.push_back(decodeRefCkpt(bytes)); @@ -1584,7 +1758,7 @@ TEST(CASRefRecoveryCasWalk, ColdWriterRecoveryPublishesOneExactUnfrontieredSucce EXPECT_TRUE(std::any_of(checkpoint_cas_bodies.begin(), checkpoint_cas_bodies.end(), [](const RefCkpt & ckpt) { return ckpt.committed_through == std::make_optional(RefTxnId{1, 2}); })) << "the exact F+1 frontier must publish before the remount seals its dead epoch"; - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 3})); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 3})); } TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsFirstCommittedTxnAboveLifeEpochOnlyCheckpoint) @@ -1598,14 +1772,14 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsFirstCommittedTxnAboveLifeEpochO /// production birth before its first log. This makes `{1,1}` the first durable transaction above a /// readable checkpoint whose `committed_through` is absent. DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(lifeEpochCkpt(1))).outcome, - PutOutcome::Done); - ASSERT_TRUE(readCkpt(*backend, layout, life)->ckpt.life_epoch); - ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, std::nullopt); + const NamespaceLifeId life = catalogLife(backend, layout, ns); + createAt(*backend, layout.refCkptKey(life), encodeRefCkpt(lifeEpochCkpt(1))); + ASSERT_TRUE(readCkptForTest(backend, layout, life)->ckpt.life_epoch); + ASSERT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, std::nullopt); + auto clock = VirtualRetryClock::installOn(store); backend->ambiguous_cas_substr = layout.refCkptKey(life); - backend->ambiguous_cas_count = 200; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->appendRefOps(ns, MutationScope::ref("a"), @@ -1620,9 +1794,9 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsFirstCommittedTxnAboveLifeEpochO }, RootMutationOrigin::Writer, RootMutationKind::Publish); }); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); - ASSERT_TRUE(backend->get(layout.refLogKey(life, RefTxnId{1, 1}))); - ASSERT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, std::nullopt); - ASSERT_FALSE(backend->get(layout.refSnapshotKey(life, RefTxnId{1, 1}))) + ASSERT_TRUE(readAt(*backend, layout.refLogKey(life, RefTxnId{1, 1}))); + ASSERT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, std::nullopt); + ASSERT_FALSE(readAt(*backend, layout.refSnapshotKey(life, RefTxnId{1, 1}))) << "the grounding test must exercise the exact log successor, not a hinted snapshot"; backend->ambiguous_cas_count = 0; @@ -1630,7 +1804,7 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryAdoptsFirstCommittedTxnAboveLifeEpochO EXPECT_TRUE(refs.contains("a")); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); } TEST(CASRefRecoveryCasWalk, WriterRecoveryRestartsWhenCheckpointAdvancesPastPrivateCandidate) @@ -1639,7 +1813,7 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryRestartsWhenCheckpointAdvancesPastPriv const Layout layout("p"); const RootNamespace ns{"srv1/recovery_checkpoint_moves"}; auto store = openWalkPool(backend); - const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(backend, store, layout, ns); const String ckpt_key = layout.refCkptKey(life); const RefLogTxn later = makeOrdinaryTxn(ns, RefTxnId{1, 3}, "c", /*birth=*/false); bool injected = false; @@ -1648,25 +1822,31 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryRestartsWhenCheckpointAdvancesPastPriv /// `{1,3}` and publish its frontier before recovery's own checkpoint CAS. The stale private /// candidate contains only `b`; it must restart and replay `c`, not accept an `IdenticalSkip` and /// install below the exact checkpoint it just observed. - backend->before_cas_put = [&](const String & key, const String &, const std::optional & expected) + /// The hook captures locals declared after the store, and the store's teardown still performs + /// checkpoint writes, so the hook is cleared before those locals die. + SCOPE_EXIT({ backend->before_cas_put = {}; }); + backend->before_cas_put = [&](const String & key, const String &, const std::optional & expected) { if (injected || key != ckpt_key) return; injected = true; ASSERT_TRUE(expected); - ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, later.txn_id), - sealObject(FormatId::RefLog, encodeRefLogTxn(later))).outcome, - PutOutcome::Done); - const auto current = backend->get(key); + createAt(*backend, layout.refLogKey(life, later.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(later))); + OperationForTest op(*backend); + const auto current = (*op).read(key, Retry::once()); ASSERT_TRUE(current); - ASSERT_EQ(current->token, *expected); + /// `expected` is the raw transport value the publisher is presenting; `PersistedEtag::capture` + /// re-derives the same raw value from the minted incarnation, so the two compare. + ASSERT_EQ(PersistedEtag::capture(current->etag).value, *expected); const RefCkpt advanced = mergeCkpt( decodeRefCkpt(current->bytes), RefCkpt{.life_epoch = std::nullopt, .committed_through = later.txn_id, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt}); - ASSERT_EQ(backend->putOverwrite(key, encodeRefCkpt(advanced), current->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative( + (*op).replace(key, encodeRefCkpt(advanced), current->etag, Retry::once()))); }; const auto refs = store->listRefs(ns); @@ -1675,7 +1855,7 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryRestartsWhenCheckpointAdvancesPastPriv EXPECT_TRUE(refs.contains("b")); EXPECT_TRUE(refs.contains("c")) << "recovery must restart from the newer exact frontier"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, later.txn_id); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, later.txn_id); } TEST(CASRefRecoveryCasWalk, WriterRecoveryRejectsTwoUnfrontieredSuccessorsAfterExactCheckpointReread) @@ -1696,27 +1876,29 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryRejectsTwoUnfrontieredSuccessorsAfterE return ops; }, RootMutationOrigin::Writer, RootMutationKind::Publish)); - const NamespaceLifeId life = catalogLife(*backend, layout, ns); + const NamespaceLifeId life = catalogLife(backend, layout, ns); const String ckpt_key = layout.refCkptKey(life); + auto clock = VirtualRetryClock::installOn(store); backend->ambiguous_cas_substr = ckpt_key; - backend->ambiguous_cas_count = 200; + backend->ambiguous_cas_count = kFaultsBeyondTheRetryWindow; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->appendRefOps(ns, MutationScope::ref("b"), [](const RefTableState &) { return publishCommittedOps("b", manifestRef(1, 2, 1)); }, RootMutationOrigin::Writer, RootMutationKind::Publish); }); + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); const RefLogTxn second_successor = makeOrdinaryTxn(ns, RefTxnId{1, 3}, "c", /*birth=*/false); - ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(life, second_successor.txn_id), - sealObject(FormatId::RefLog, encodeRefLogTxn(second_successor))).outcome, - PutOutcome::Done); + createAt(*backend, layout.refLogKey(life, second_successor.txn_id), + sealObject(FormatId::RefLog, encodeRefLogTxn(second_successor))); backend->ambiguous_cas_count = 0; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})) + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})) << "corruption must not launder either successor into the frontier"; } @@ -1726,21 +1908,23 @@ TEST(CASRefRecoveryCasWalk, WriterRecoveryRejectsDifferentOrdinaryBytesAtTheReta const Layout layout("p"); const RootNamespace ns{"srv1/recovery_different_successor"}; auto store = openWalkPool(backend); - const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(backend, store, layout, ns); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); const String successor_key = layout.refLogKey(life, RefTxnId{1, 2}); - const auto original = backend->get(successor_key); + const auto original = readAt(*backend, successor_key); ASSERT_TRUE(original); const RefLogTxn different = makeOrdinaryTxn(ns, RefTxnId{1, 2}, "different", /*birth=*/false); - ASSERT_EQ(backend->putOverwrite(successor_key, - sealObject(FormatId::RefLog, encodeRefLogTxn(different)), - original->token).outcome, - PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).replace(successor_key, + sealObject(FormatId::RefLog, encodeRefLogTxn(different)), + original->etag, Retry::once()))); + } expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)store->listRefs(ns); }); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, (RefTxnId{1, 1})); } TEST(CASRefRecoveryCasWalk, RetainedOldWriterAttemptLosesConclusiveToASuccessorSeal) @@ -1749,25 +1933,27 @@ TEST(CASRefRecoveryCasWalk, RetainedOldWriterAttemptLosesConclusiveToASuccessorS const Layout layout("p"); const RootNamespace ns{"srv1/recovery_successor_seal"}; auto store = openWalkPool(backend); - const NamespaceLifeId life = strandOneUnfrontieredSuccessor(*backend, store, layout, ns); + const NamespaceLifeId life = strandOneUnfrontieredSuccessor(backend, store, layout, ns); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); const RefTxnId successor_id{1, 2}; const String successor_key = layout.refLogKey(life, successor_id); - const auto original = backend->get(successor_key); + const auto original = readAt(*backend, successor_key); ASSERT_TRUE(original); const RefLogTxn successor_seal = makeSealTxn(ns, successor_id); - ASSERT_EQ(backend->putOverwrite(successor_key, - sealObject(FormatId::RefLog, encodeRefLogTxn(successor_seal)), - original->token).outcome, - PutOutcome::Done); + { + OperationForTest op(*backend); + ASSERT_TRUE(std::holds_alternative((*op).replace(successor_key, + sealObject(FormatId::RefLog, encodeRefLogTxn(successor_seal)), + original->etag, Retry::once()))); + } const auto refs = store->listRefs(ns); EXPECT_FALSE(refs.contains("b")) << "the old writer's retained ordinary bytes lost at the sealed slot"; EXPECT_EQ(store->lastEpochSealForTest(ns), std::make_optional(successor_id)); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.committed_through, successor_id); - EXPECT_EQ(readCkpt(*backend, layout, life)->ckpt.last_epoch_seal, successor_id); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.committed_through, successor_id); + EXPECT_EQ(readCkptForTest(backend, layout, life)->ckpt.last_epoch_seal, successor_id); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); } @@ -1785,7 +1971,7 @@ TEST(CASRefRecoveryCasWalk, UnresolvedSealSlotFailsClosedWithoutInstalling) const Layout layout("p"); const RootNamespace ns{"srv1/unresolved"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1793,18 +1979,33 @@ TEST(CASRefRecoveryCasWalk, UnresolvedSealSlotFailsClosedWithoutInstalling) /// envelope is spent in a handful of iterations instead of spinning against a frozen clock. Not /// cosmetic: with a frozen clock this test burns ~700k retries and the same number of log lines, /// which is how a real regression in this arm would become invisible in the noise. - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: the retry-sleep hook below mutates it, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config = walkTestConfig(); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; auto store = openWalkPool(backend, config); ASSERT_TRUE(store); - store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + store->setCasRetrySleepForTest([fake_now](uint64_t ms) + { + *fake_now += ms; + }); backend->ambiguous_put_substr = "/_log/"; + const uint64_t fake_now_before = fake_now->load(); EXPECT_ANY_THROW(store->listRefs(ns)); EXPECT_FALSE(store->refTableRecoveredForTest(ns)) << "a table whose dead epoch may or may not be closed must never be exposed as recovered"; + /// The engine reissued -- more than one physical attempt -- and paid a real retry pause on the + /// injected clock before giving up; a fault settled by a single, unretried attempt would not + /// exercise the transient-retry path this test's own name and docstring claim to drive. + EXPECT_GT(backend->ambiguous_put_attempts.load(), 1u); + EXPECT_GT(fake_now->load(), fake_now_before); } /// --------------------------------------------------------------------------------------------- @@ -1824,7 +2025,7 @@ TEST(CASRefRecoveryCasWalk, ALatePredecessorPutAtTheSealedSlotIsRefusedByTheStor const Layout layout("p"); const RootNamespace ns{"srv1/ghost"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1836,8 +2037,10 @@ TEST(CASRefRecoveryCasWalk, ALatePredecessorPutAtTheSealedSlotIsRefusedByTheStor const RefTxnId ghost_id{1, 2}; const String ghost_bytes = sealObject(FormatId::RefLog, encodeRefLogTxn(makeOrdinaryTxn(ns, ghost_id, "ghost", /*birth=*/false))); - const PutResult put = backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), ghost_id), ghost_bytes); - EXPECT_EQ(put.outcome, PutOutcome::PreconditionFailed) + OperationForTest ghost_op(*backend); + const WriteResult put = (*ghost_op).create( + layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), ghost_id), ghost_bytes, Retry::once()); + EXPECT_TRUE(std::holds_alternative(put)) << "the seal occupies the ghost's own key, so the store itself is the fence"; /// And the object at that key is still the seal, byte for byte -- nothing adopted the ghost. @@ -1857,7 +2060,7 @@ TEST(CASRefRecoveryCasWalk, UndecodableOccupantAtTheSealSlotFailsClosedAndLeaves const Layout layout("p"); const RootNamespace ns{"srv1/foreign_slot"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); backend->late_key = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{1, 2}); @@ -1885,7 +2088,7 @@ TEST(CASRefRecoveryCasWalk, ASecondCallerWaitsForTheWalkInsteadOfRacingIt) const Layout layout("p"); const RootNamespace ns{"srv1/serialized"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/2); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/2); seedCkpt(*backend, layout, ns, lifeEpochCkpt(1, RefTxnId{1, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1952,7 +2155,7 @@ TEST(CASRefRecoveryCasWalk, RecoveryStartsAtRecreatedLifeGenesisAndLeavesPredece const Layout layout("p"); const RootNamespace ns{"srv1/removed_then_reborn"}; - burnEpochsUpTo(*backend, layout, /*target_live_epoch=*/3); + burnEpochsUpTo(backend, layout, /*target_live_epoch=*/3); seedCkpt(*backend, layout, ns, lifeEpochCkpt(2, RefTxnId{2, 1})); seedTxn(*backend, layout, ns, RefTxnId{1, 1}, "a", /*birth=*/true); @@ -1985,26 +2188,9 @@ TEST(CASRefRecoveryCasWalk, RecoveryStartsAtRecreatedLifeGenesisAndLeavesPredece EXPECT_EQ(seal2->prev_epoch_seal, std::nullopt) << "sequence 2 carries no chain link"; } -/// `PutHookBackend::casPut` must route through its immediate parent `HidingListBackend::casPut`, not -/// past it to `CountingBackend`, so that a test arming BOTH layers on one `PutHookBackend` instance -/// gets both behaviors composed rather than one silently disabled by the other. -TEST(CASRefRecoveryCasWalk, PutHookBackendComposesHidingListBackendCasPutFaultInjection) -{ - auto backend = std::make_shared(); - - bool before_cas_put_fired = false; - backend->before_cas_put = [&](const String &, const String &, const std::optional &) - { - before_cas_put_fired = true; - }; - - backend->watched_substr = "probe"; - bool on_key_fired = false; - backend->on_key = [&] { on_key_fired = true; }; - - ASSERT_EQ(backend->casPut("p/probe", "x", std::nullopt).outcome, CasOutcome::Committed); - - EXPECT_TRUE(before_cas_put_fired) - << "HidingListBackend's before_cas_put hook must still fire for a PutHookBackend instance"; - EXPECT_TRUE(on_key_fired) << "PutHookBackend's own on_key hook must still fire on top of it"; -} +/// `PutHookBackendComposesHidingListBackendCasPutFaultInjection` was retired: it pinned that +/// `PutHookBackend::casPut` reaches its immediate parent `HidingListBackend::casPut` rather than +/// bypassing it to `CountingBackend` -- a fact about this file's own fixture class hierarchy (ordinary +/// C++ virtual dispatch), not a claim any production change could falsify. `PutHookBackend` and +/// `HidingListBackend` are still exercised together, on real recovery-walk scenarios, elsewhere in this +/// file (search for `PutHookBackend>`). diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp index 2293c0bc166c..942fc170e139 100644 --- a/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp +++ b/src/Disks/tests/gtest_cas_ref_snapshot_format.cpp @@ -65,7 +65,7 @@ TEST(CASRefSnapshotCodec, DecodeRequiresLifecycleField) { const RefTableSnapshot s = makeLiveSnapshot(); String bytes = encodeRefTableSnapshot(s); - const String field = R"(,"lc":"live")"; + const String field = R"(,"lifecycle":"live")"; const size_t at = bytes.find(field); ASSERT_NE(at, String::npos); bytes.erase(at, field.size()); @@ -78,10 +78,10 @@ TEST(CASRefSnapshotCodec, DecodeRejectsTerminalLifecycleWord) { const RefTableSnapshot s = makeLiveSnapshot(); String bytes = encodeRefTableSnapshot(s); - const String live = R"("lc":"live")"; + const String live = R"("lifecycle":"live")"; const size_t at = bytes.find(live); ASSERT_NE(at, String::npos); - bytes.replace(at, live.size(), R"("lc":"removed")"); + bytes.replace(at, live.size(), R"("lifecycle":"removed")"); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); @@ -91,7 +91,7 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnEpochField) { const RefTableSnapshot s = makeLiveSnapshot(); String bytes = encodeRefTableSnapshot(s); - const String live = R"("lc":"live")"; + const String live = R"("lifecycle":"live")"; const size_t at = bytes.find(live); ASSERT_NE(at, String::npos); bytes.replace(at, live.size(), live + R"(,"rte":"7")"); @@ -104,7 +104,7 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnSequenceField) { const RefTableSnapshot s = makeLiveSnapshot(); String bytes = encodeRefTableSnapshot(s); - const String live = R"("lc":"live")"; + const String live = R"("lifecycle":"live")"; const size_t at = bytes.find(live); ASSERT_NE(at, String::npos); bytes.replace(at, live.size(), live + R"(,"rts":"9")"); @@ -117,7 +117,7 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnFieldPair) { const RefTableSnapshot s = makeLiveSnapshot(); String bytes = encodeRefTableSnapshot(s); - const String live = R"("lc":"live")"; + const String live = R"("lifecycle":"live")"; const size_t at = bytes.find(live); ASSERT_NE(at, String::npos); bytes.replace(at, live.size(), live + R"(,"rte":"7","rts":"9")"); @@ -126,8 +126,7 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRetiredRemoveTxnFieldPair) [&] { (void)decodeRefTableSnapshot(bytes, s.ns, s.snapshot_id); }); } -/// No-tolerance decode pin (codex round-2, finding 3): the `"pl"` (payload) field was removed from the -/// committed-row wire in stage-1 T12. It is NOT a genuinely-unknown future field the tolerant reader may +/// No-tolerance decode pin: the `"pl"` (payload) field is not a genuinely-unknown future field the tolerant reader may /// skip -- silently discarding a persisted payload would lose data -- so decoding a committed row that /// still carries `"pl"` must FAIL with `CORRUPTED_DATA` naming the removed field, not `skipUnknown` it. TEST(CASRefSnapshotCodec, DecodeRejectsRemovedPayloadFieldInCommittedRow) @@ -142,8 +141,8 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRemovedPayloadFieldInCommittedRow) s.committed.push_back(c); const String bytes = encodeRefTableSnapshot(s); - /// Splice the retired `"pl"` field back into the committed record, just before its `"ts"` field. - const String needle = ",\"ts\":"; + /// Splice the retired `"pl"` field back into the committed record, just before its `"published_ms"` field. + const String needle = ",\"published_ms\":"; const auto pos = bytes.find(needle); ASSERT_NE(pos, String::npos); const String tampered = bytes.substr(0, pos) + R"(,"pl":"deadbeef")" + bytes.substr(pos); @@ -152,6 +151,30 @@ TEST(CASRefSnapshotCodec, DecodeRejectsRemovedPayloadFieldInCommittedRow) [&] { decodeRefTableSnapshot(tampered, s.ns, s.snapshot_id); }); } +/// Row kinds are the owner-kind vocabulary, so an unknown kind word must fail closed at the word +/// table rather than being silently skipped as an unrecognized row -- a skipped row would lose a ref +/// from a snapshot the reader still reports as complete. +TEST(CASRefSnapshotCodec, DecodeRejectsUnknownRowKindWord) +{ + RefTableSnapshot s; + s.ns = "ns"; + s.snapshot_id = RefTxnId{1, 1}; + RefCommittedRow c; + c.ref_name = "all_1_1_0"; + c.manifest_ref = manifestRef(5, 10, 1); + c.published_at_ms = 1717000000000ULL; + s.committed.push_back(c); + + const String bytes = encodeRefTableSnapshot(s); + const String needle = "\"kind\":\"committed\""; + const auto pos = bytes.find(needle); + ASSERT_NE(pos, String::npos); + const String tampered = bytes.substr(0, pos) + "\"kind\":\"archived\"" + bytes.substr(pos + needle.size()); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, + [&] { decodeRefTableSnapshot(tampered, s.ns, s.snapshot_id); }); +} + TEST(CASRefSnapshotCodec, RoundTripLiveEmpty) { RefTableSnapshot s; @@ -207,7 +230,7 @@ TEST(CASRefSnapshotFormat, MaximalRefSequenceRoundTripsAsADecimalString) const String text = encodeRefTableSnapshot(m); const RefTableSnapshot back = decodeRefTableSnapshot(text, m.ns, m.snapshot_id); EXPECT_EQ(back.snapshot_id.ref_sequence, std::numeric_limits::max()); - EXPECT_NE(text.find("\"rs\":\"18446744073709551615\""), String::npos); + EXPECT_NE(text.find("\"snapshot_seq\":\"18446744073709551615\""), String::npos); } /// =================================================================================== @@ -419,6 +442,8 @@ TEST(CASRefSnapshotCodec, DecodeRejectsOversizedBufferDirectly) /// Shape-level failure-mode battery (truncation / v+1 gate / wrong type / leading garbage) /// =================================================================================== +CAS_BATTERY_COVERS(RefSnapshot); + TEST(CASFormatBattery, RefSnapshot) { const RefTableSnapshot s = makeLiveSnapshot(); @@ -428,9 +453,9 @@ TEST(CASFormatBattery, RefSnapshot) [s] { return sealObject(FormatId::RefSnapshot, encodeRefTableSnapshot(s)); }, [ns, id](std::string_view d) { decodeRefTableSnapshot(openObject(FormatId::RefSnapshot, d), ns, id); }, currentFormatHeader("cas_ref_snap") + - "{\"ns\":\"srv1/db/table@cas@\",\"we\":\"5\",\"rs\":\"200\",\"lc\":\"live\"}\n" - "{\"k\":\"c\",\"rn\":\"all_1_1_0\",\"me\":\"5\",\"mb\":\"10\",\"mo\":1,\"ts\":1717000000000}\n" - "{\"k\":\"c\",\"rn\":\"all_2_2_0\",\"me\":\"5\",\"mb\":\"11\",\"mo\":1,\"ts\":1717000000001}\n" - "{\"k\":\"p\",\"rn\":\"all_3_3_0\",\"me\":\"5\",\"mb\":\"12\",\"mo\":1}\n" + "{\"namespace\":\"srv1/db/table@cas@\",\"snapshot_epoch\":\"5\",\"snapshot_seq\":\"200\",\"lifecycle\":\"live\"}\n" + "{\"kind\":\"committed\",\"ref\":\"all_1_1_0\",\"epoch\":\"5\",\"build\":\"10\",\"ord\":1,\"published_ms\":1717000000000}\n" + "{\"kind\":\"committed\",\"ref\":\"all_2_2_0\",\"epoch\":\"5\",\"build\":\"11\",\"ord\":1,\"published_ms\":1717000000001}\n" + "{\"kind\":\"precommit\",\"ref\":\"all_3_3_0\",\"epoch\":\"5\",\"build\":\"12\",\"ord\":1}\n" "{\"n\":3}\n"}); } diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp index ca34681f7b93..dbb53e10fe0a 100644 --- a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp +++ b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp @@ -14,6 +14,8 @@ #include +#include +#include #include #include #include @@ -55,6 +57,63 @@ using DB::Cas::tests::publishCommittedOps; namespace { +/// The engine reissues an unresolved write until its OWN retry window closes, and that window is +/// measured on a clock the engine reads. Both seams here share one counter -- the sleep the engine +/// performs is what advances the clock -- so a fault that stays armed ends the call at its deadline +/// with no real time passing. Installed on the whole pool, because the ref-lane write, its settling +/// read and the recovery retry loop all pace through the same seam. The pool owns the closures and the +/// closures own the clock, so it outlives everything that can still read it. +class VirtualRetryClock +{ +public: + static std::shared_ptr installOn(const PoolPtr & store) + { + auto clock = std::make_shared(); + store->setCasRequestNowFnForTest([clock] { return clock->nowMs(); }); + store->setCasRetrySleepForTest([clock](uint64_t ms) { clock->advance(ms); }); + return clock; + } + + uint64_t nowMs() const + { + std::lock_guard lock(mutex); + return now_ms; + } + size_t pauseCount() const + { + std::lock_guard lock(mutex); + return pauses; + } + uint64_t longestPause() const + { + std::lock_guard lock(mutex); + return longest_pause; + } + + void advance(uint64_t ms) + { + std::lock_guard lock(mutex); + /// Plus one millisecond, because full jitter can draw a ZERO pause: a clock that does not move + /// would leave the loop reissuing for ever against a fault that never clears. + now_ms += ms + 1; + ++pauses; + longest_pause = std::max(longest_pause, ms); + } + +private: + mutable std::mutex mutex; + uint64_t now_ms = 0; + size_t pauses = 0; + uint64_t longest_pause = 0; +}; + +/// More injected failures than the engine's own retry window can make attempts, on a clock that +/// advances at least a millisecond per pause: a call meeting this fault must end at its DEADLINE, never +/// by outliving the fault. A bounded count would be spent by ONE call's own reissues -- the engine +/// settles each ambiguity by an exact read and then reissues -- and the fixture would then be measuring +/// attempts where it means to measure dispatches. +constexpr int kFaultsBeyondTheRetryWindow = 100'000; + PoolPtr openPool(const std::shared_ptr & backend, PoolConfig config = {}) { config.pool_prefix = "p"; @@ -118,7 +177,7 @@ TEST(CASRefSnapshotPublishOrdering, SnapshotBodyIsDurableBeforeCheckpointAdvance /// nothing about the publisher's own ordering. const size_t offset = backend->journalSize(); const uint64_t put_before = backend->putCount(snapshot_key); - const uint64_t cas_before = backend->casPutCount(ckpt_key); + const uint64_t cas_before = backend->putOverwriteCount(ckpt_key); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) << "a healthy Ready-lane table with an uncovered tail must publish"; @@ -126,10 +185,10 @@ TEST(CASRefSnapshotPublishOrdering, SnapshotBodyIsDurableBeforeCheckpointAdvance /// Positive control: this attempt touched each key exactly once (no retry, no redundant write) -- /// which is what makes the index comparison below meaningful rather than an artifact of a busy log. EXPECT_EQ(backend->putCount(snapshot_key) - put_before, 1u); - EXPECT_EQ(backend->casPutCount(ckpt_key) - cas_before, 1u); + EXPECT_EQ(backend->putOverwriteCount(ckpt_key) - cas_before, 1u); - const auto body_index = backend->firstIndexFrom(OrderedFaultBackend::Op::Put, snapshot_key, offset); - const auto ckpt_index = backend->firstIndexFrom(OrderedFaultBackend::Op::Cas, ckpt_key, offset); + const auto body_index = backend->firstIndexFrom(snapshot_key, offset); + const auto ckpt_index = backend->firstIndexFrom(ckpt_key, offset); ASSERT_TRUE(body_index.has_value()) << "the snapshot body must have been PUT"; ASSERT_TRUE(ckpt_index.has_value()) << "the checkpoint must have been CAS-advanced"; EXPECT_LT(*body_index, *ckpt_index) @@ -145,6 +204,7 @@ TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEf { auto backend = std::make_shared(); auto store = openPool(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_adoption_after_both"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); @@ -152,26 +212,30 @@ TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEf const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->writerEpoch(), 1}); const String ckpt_key = store->layout().refCkptKey(life); - /// Fail every one of the (attempt-bounded) 100 `_ckpt` CAS attempts `publishCkpt` will make: the - /// body PUT still commits (dedup: an identical, already-durable body resolves as `Committed` without - /// re-sending), but the checkpoint never advances within this call. - backend->armCasConflict(ckpt_key, 100); + /// Refuse the `_ckpt` CAS for as long as `publishCkpt` keeps reissuing, so it ends at its own retry + /// window: the body create still commits, but the checkpoint never advances within this call. + backend->armWriteConflict(ckpt_key, kFaultsBeyondTheRetryWindow); EXPECT_FALSE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) << "a persistently conflicting checkpoint CAS must not be reported as a successful publish"; + backend->armWriteConflict(ckpt_key, 0); + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; + EXPECT_LE(clock->longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; EXPECT_EQ(backend->putCount(snapshot_key), 1u) << "the body is durable regardless of the ckpt outcome"; EXPECT_FALSE(store->newestPublishedSnapshotIdForTest(ns).has_value()) << "in-memory adoption must NOT happen while the checkpoint has not advanced"; - /// Disarm the fault and retry (the one retry unit): the retry issues its OWN `putIfAbsent` attempt at + /// Retry with the fault disarmed (the one retry unit): the retry issues its OWN create attempt at /// the same content-addressed key with the same bytes (so `putCount`, a call counter, becomes 2 -- - /// not a "no write happened" 1), but the backend resolves it as `Committed` against the already-durable - /// object rather than sending a distinct object, and the checkpoint CAS now succeeds. + /// not a "no write happened" 1). That attempt meets its own identical bytes as a conflict, which the + /// publisher's occupant compare accepts rather than writing a second object, and the checkpoint CAS + /// now succeeds. ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) << "the retry, with the fault cleared, must publish"; EXPECT_EQ(backend->putCount(snapshot_key), 2u) - << "the retry's body PUT is its own attempt, resolved via dedup against identical, " - "already-durable bytes rather than writing a second object"; + << "the retry's body create is its own attempt, accepted against identical, already-durable " + "bytes rather than writing a second object"; EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::make_optional(RefTxnId{store->writerEpoch(), 1})) << "adoption happens exactly once, after both effects are durable"; } @@ -180,7 +244,7 @@ TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEf /// 3. `NeedsRecovery` ("Poisoned") lane: recovery precedes any snapshot publication /// --------------------------------------------------------------------------------------------- -/// `Poisoned` is this task's plan's name for what the code spells `RefLaneState::NeedsRecovery` -- the +/// `RefLaneState::NeedsRecovery` is the state for a transaction known durable but not installable in the cache -- the /// state the header documents as "a transaction is known durable but cannot be installed in this cache /// ... a hard write and certification fence until replay completes". Recorded here as the vocabulary /// correction for later tasks: there is no state literally named `Poisoned` anywhere in `CasRefLedger`. @@ -211,6 +275,7 @@ TEST(CASRefSnapshotPublishOrdering, NeedsRecoveryLaneRecoversBeforeAnySnapshotPu { auto backend = std::make_shared(); auto store = openPool(backend); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_poisoned_refuses"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); @@ -224,12 +289,14 @@ TEST(CASRefSnapshotPublishOrdering, NeedsRecoveryLaneRecoversBeforeAnySnapshotPu /// for `missing_durable_txn` commits durably, but its checkpoint never advances within this call, and /// the lane is left `NeedsRecovery` rather than installing an uncertain result -- so the cached view /// still reflects `ref_1` present, while the durable log already reflects it removed. - backend->armCasConflict(ckpt_key, 100); + backend->armWriteConflict(ckpt_key, kFaultsBeyondTheRetryWindow); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "ref_1"); }); + EXPECT_GT(clock->pauseCount(), 1u) + << "the frontier publication's reissues must pace through the injected sleep, never a real one"; ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); const uint64_t recovery_installs_before = store->recoveryInstallCountForTest(); - backend->armCasConflict(ckpt_key, 0); /// clear the fault so re-recovery's OWN catch-up CAN succeed + backend->armWriteConflict(ckpt_key, 0); /// clear the fault so re-recovery's OWN catch-up CAN succeed const size_t offset = backend->journalSize(); EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)) @@ -247,8 +314,8 @@ TEST(CASRefSnapshotPublishOrdering, NeedsRecoveryLaneRecoversBeforeAnySnapshotPu /// ORDER, not a global zero: recovery's OWN checkpoint catch-up CAS is the boundary marker. NO /// snapshot-publish effect (the new snapshot's body PUT, nor the publisher's own checkpoint-advance /// CAS) may appear at or before it. - const auto ckpt_cas_indices = backend->indicesFrom(OrderedFaultBackend::Op::Cas, ckpt_key, offset); - const auto snap_put_indices = backend->indicesFrom(OrderedFaultBackend::Op::Put, next_snapshot_key, offset); + const auto ckpt_cas_indices = backend->indicesFrom(ckpt_key, offset); + const auto snap_put_indices = backend->indicesFrom(next_snapshot_key, offset); ASSERT_GE(ckpt_cas_indices.size(), 2u) << "expected one checkpoint CAS from recovery's catch-up and one from the snapshot publisher"; const size_t recovery_catchup_index = ckpt_cas_indices.front(); @@ -291,26 +358,34 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) using ProfileEvents::global_counters; auto backend = std::make_shared(); - /// A single-attempt request budget, exactly as `gtest_cas_ref_writer.cpp`'s - /// `C4BackoffDefersThenRetriesAndPublishes` uses: with `max_attempts = 1` a faulted PUT resolves to a - /// definite, non-`Committed` outcome on its own attempt, with no internal retry loop and so no - /// wall-clock wait. + /// The budget bounds the mount lease's own admission arithmetic and nothing else. What makes each + /// dispatch below fail as ONE dispatch is that the injected fault outlasts the whole call while the + /// injected clock carries it to its own retry window; the fake boot clock this test drives the + /// backoff decisions on is a DIFFERENT clock, and stays frozen between steps. CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; budget.lease_safety_margin_ms = 100; - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: this test mutates the clock below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config; config.snapshot_log_count_threshold = 0; /// any nonempty tail is over-threshold config.snapshot_log_bytes_threshold = 1ULL << 40; config.snapshot_publish_backoff_initial_ms = 1000; config.snapshot_publish_backoff_max_ms = 4000; config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; config.cas_request_budget = budget; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); auto store = openPool(backend, config); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_backoff"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); @@ -322,10 +397,11 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// Fault the snapshot BODY put (never the `_ckpt` CAS -- an append-commit's OWN checkpoint write /// shares that key, and faulting it would drive the append lane into `NeedsRecovery` instead of - /// exercising the snapshot-publish backoff this test targets). Exactly 3 failures: the next 3 - /// automatic dispatch attempts fail (arming, then doubling, then re-doubling the backoff); the 4th - /// finds the fault disarmed and succeeds. - backend->armPutFailure("_snap/", 3); + /// exercising the snapshot-publish backoff this test targets). Armed for as long as any dispatch + /// keeps reissuing, so each dispatch ends at its own retry window and counts as ONE dispatch: the + /// next 3 fail (arming, then doubling, then re-doubling the backoff) and the test disarms the fault + /// before the 4th. + backend->armWriteFailure("_snap/", kFaultsBeyondTheRetryWindow); const auto dispatchCount = [&] { return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); }; @@ -342,13 +418,13 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) EXPECT_EQ(dispatchCount(), d1) << "a read within the initial backoff window must not re-dispatch"; /// Cross the 1000ms deadline: exactly one retry dispatches (and fails again, doubling to 2000ms). - fake_now += 1000; + *fake_now += 1000; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 1) << "past the first deadline, exactly one retry dispatches"; /// Short of the DOUBLED (2000ms) deadline: still refused. - fake_now += 1000; + *fake_now += 1000; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 1) @@ -356,7 +432,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// Cross the doubled deadline: one more retry dispatches (and fails again -- the third and last armed /// failure -- doubling to the 4000ms cap). - fake_now += 1000; + *fake_now += 1000; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 2) << "past the doubled deadline, exactly one more retry dispatches"; @@ -365,21 +441,22 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// 2000ms, or that read `initial` where it means `max`, would still pass -- the only check so far /// is AT the +4000 crossing below. 2000ms past the doubled deadline is still short of the capped /// 4000ms backoff, so no third retry may dispatch yet. - fake_now += 2000; + *fake_now += 2000; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 2) << "2000ms past the doubled deadline is still short of the capped 4000ms backoff"; - /// Cross the (capped) 4000ms deadline: the retry's fault budget is exhausted, so this attempt - /// succeeds, and `resetPublishBackoff` clears the cooldown -- proved by the NEXT trigger dispatching - /// with no wait at all. - fake_now += 2000; + /// Cross the (capped) 4000ms deadline with the fault disarmed, so this attempt succeeds and + /// `resetPublishBackoff` clears the cooldown -- proved by the NEXT trigger dispatching with no wait + /// at all. + backend->armWriteFailure("_snap/", 0); + *fake_now += 2000; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 3) << "past the second (capped) deadline, the retry dispatches and succeeds"; EXPECT_NE(store->newestPublishedSnapshotIdForTest(ns), snapshot_after_birth) - << "the fault budget is exhausted, so this attempt actually advances the published snapshot"; + << "the fault is disarmed, so this attempt actually advances the published snapshot"; ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->writerEpoch(), 3})); store->waitForSnapshotPublishSettleForTest(ns); @@ -393,20 +470,23 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// the INITIAL 1000ms interval rather than continuing from the 4000ms cap -- refused short of /// 1000ms, admitted at 1000ms -- which a no-op reset cannot produce (it would refuse both probes, /// since the stale deadline is still far in the future). - backend->armPutFailure("_snap/", 1); + backend->armWriteFailure("_snap/", kFaultsBeyondTheRetryWindow); ASSERT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->writerEpoch(), 4})); store->waitForSnapshotPublishSettleForTest(ns); const uint64_t d2 = dispatchCount(); - fake_now += 500; + *fake_now += 500; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d2) << "short of 1000ms since the reset, no retry may dispatch yet"; - fake_now += 500; + *fake_now += 500; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d2 + 1) << "resetPublishBackoff must have restarted the schedule at the INITIAL 1000ms interval, not " "left it continuing from the 4000ms cap"; + backend->armWriteFailure("_snap/", 0); + EXPECT_GT(clock->pauseCount(), 1u) + << "every failed dispatch above must have paced through the injected sleep, never a real one"; } TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurablePublish) @@ -414,21 +494,30 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable using ProfileEvents::global_counters; auto backend = std::make_shared(); + /// No write fault anywhere in this test: every refusal below comes from the lane not being Ready, + /// so the budget only has to keep the mount lease admitting. CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; budget.lease_safety_margin_ms = 100; - uint64_t fake_now = 2'000'000; + /// Held in a shared atomic, not a plain local: this test mutates the clock below, and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture of a local would dangle. + auto fake_now = std::make_shared>(2'000'000); PoolConfig config; config.snapshot_log_count_threshold = 0; config.snapshot_log_bytes_threshold = 1ULL << 40; config.snapshot_publish_backoff_initial_ms = 200; config.snapshot_publish_backoff_max_ms = 30'000; config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] + { + return fake_now->load(); + }; config.cas_request_budget = budget; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); auto store = openPool(backend, config); const RootNamespace ns{"srv1/order_not_ready_backoff"}; @@ -508,14 +597,14 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable 400, 800, 1600, 3200, 6400, 12'800, 25'600, 30'000, 30'000}; for (const uint64_t next_delay_ms : next_delays) { - fake_now += delay_ms - 1; + *fake_now += delay_ms - 1; store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatch_count(), production_dispatches + 1 + admitted_retries) << "no retry may dispatch one millisecond before the current deadline"; EXPECT_EQ(backoff_count(), production_backoffs + 1 + admitted_retries); - ++fake_now; + ++(*fake_now); store->resolveRef(ns, "ref_1"); store->waitForSnapshotPublishSettleForTest(ns); ++admitted_retries; diff --git a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp index ace02d24836e..9fcd010643a9 100644 --- a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp +++ b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp @@ -3,7 +3,7 @@ #include "config.h" #include -#include +#include #include #include #include @@ -25,6 +25,7 @@ #include #include #include +#include /// ================================================================================================ /// Task 4 (2026-07-28 CAS ref-chain Stage A streams, spec INV-1's every-attempt rule + INV-2's seal): @@ -69,46 +70,51 @@ extern const int NETWORK_ERROR; } using namespace DB::Cas; +using DB::Cas::tests::VirtualRetryClock; using DB::Cas::tests::CountingBackend; -using DB::Cas::tests::LandedButAckLostOnceBackend; using DB::Cas::tests::expectThrowsCode; namespace { -PoolPtr openPool(const BackendPtr & backend, CasRequestBudget budget = {}) +template +PoolPtr openPool(const std::shared_ptr & backend, CasRequestBudget budget = {}) { DB::Cas::tests::seedPoolMetaForRestart(*backend); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); } -/// The budget every wedge test uses: ONE attempt, so a single injected ambiguity is the whole -/// operation and the lane wedges deterministically instead of retrying its way out. -CasRequestBudget singleAttemptBudget() +/// An exact read (mirrors the retired `backend.get(key)`). +std::optional readObj(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +/// A one-shot `create`, asserting it committed (mirrors the retired `backend.putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + DB::Cas::tests::OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// The budget every wedge test uses. It bounds the mount lease's own admission arithmetic +/// (`attempt_timeout_ms` is what one attempt reserves, `lease_safety_margin_ms` the room kept past it) +/// and nothing else: a write's ATTEMPT COUNT is the `Retry` policy's, and a call's own deadline is +/// fence-derived (`Retry::untilLeaseSafe`/`Retry.bind`), so no budget field can make an injected fault +/// conclusive. What makes a fault conclusive here is that it stays armed for the whole call while +/// `VirtualRetryClock` carries the call to its own deadline. +CasRequestBudget wedgeTestBudget() { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; /// strictly above attempt_timeout_ms: equality is a wall-clock race (validateCasRequestBudget) budget.lease_safety_margin_ms = 100; return budget; } -/// TWO attempts of one logical operation, with the inter-attempt backoff disabled. Everything about the -/// call-level verdict rule lives BETWEEN two attempts of a single call, so it cannot be reached with the -/// one-attempt budget the tests above use; the backoff is switched off because the schedule is -/// `gtest_cas_request_control.cpp`'s subject and a real sleep here would only slow the suite. -/// -/// `[[maybe_unused]]`: both of its callers need a real S3-classified rejection to script their second -/// attempt, so they compile away entirely in a build without S3. -[[maybe_unused]] CasRequestBudget twoAttemptBudget() -{ - CasRequestBudget budget = singleAttemptBudget(); - budget.max_attempts = 2; - budget.retry_initial_backoff_ms = 0; - budget.retry_max_backoff_ms = 0; - return budget; -} PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) { @@ -131,25 +137,20 @@ void publishEmptyPart(const PoolPtr & s, const RootNamespace & ns, const String class WedgeTestBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; - using CountingBackend::get; - /// One-shot ambiguity that writes NOTHING: the response is lost and the key stays absent, which is - /// the input that makes a later `slotOccupy` report `Created`. + /// the input that makes a later bounded create commit. String ambiguous_substr; int ambiguous_count = 0; /// One-shot DETERMINISTIC LOCAL failure (`BAD_ARGUMENTS`, in `isDeterministicLocalFailure`'s set), - /// which `slotOccupy` rethrows unchanged -- a definite refusal of THIS attempt. Portable stand-in - /// for the S3 `DefiniteFailure` shape, which needs `USE_AWS_S3`; both are "proven never applied". + /// which every loop surfaces unchanged -- an exception out of the write, not an outcome. String definite_substr; int definite_count = 0; - /// One-shot WHITELISTED SYNCHRONOUS REJECTION: the ONLY shape `classifyConditionalWriteResult` - /// answers `DefiniteFailure` for, and therefore the only way to drive the append lane's definite - /// arm. Distinct from `definite_substr` above on purpose -- that one is a deterministic LOCAL - /// failure, which `slotOccupy` rethrows but `putIfAbsentControlled` (no such special case) merely - /// classifies Unresolved, so it cannot script this arm at all. + /// One-shot WHITELISTED SYNCHRONOUS REJECTION: the shape `isDefinitelyRefusedWrite` answers TRUE + /// for, and therefore the only way to drive the append lane's `Refused` arm. Distinct from + /// `definite_substr` above on purpose -- that one is a local bug the engine rethrows, while this one + /// is the store's own answer and comes back as a value. String s3_definite_substr; int s3_definite_count = 0; @@ -160,39 +161,85 @@ class WedgeTestBackend : public CountingBackend int conflict_count = 0; String conflict_bytes; - /// Fail GETs of matching keys after skipping the first `fail_get_skip` of them -- the resolve read - /// that PROVES the conflict must succeed, so only the adjudication read that follows it is faulted. + /// Lose every GET of a matching key. Armed by `fail_get_latched` below, because a read fault that + /// clears mid-call is simply reissued: the read engine settles a transient read failure by trying + /// again, so only a fault that outlasts the read's whole window is conclusive. String fail_get_substr; - int fail_get_skip = 0; - int fail_get_count = 0; String fail_cas_substr; int fail_cas_count = 0; - std::optional get(const String & key, Range range) override + /// LATCHED variants of the four seams above. A COUNT cannot make an injected fault conclusive: the + /// write engine settles every ambiguity by an exact read and then REISSUES, so a fault that runs out + /// mid-call is answered by the next attempt instead of by the call's deadline. A latch stays armed + /// until the test clears it, which is what makes "every attempt of this call was unresolved" the + /// input the wedge rule is about. + bool ambiguous_latched = false; + bool s3_definite_latched = false; + bool fail_get_latched = false; + bool fail_cas_latched = false; + + /// Our OWN exact bytes land and only the response is lost. Paired with `fail_get_latched` on the + /// same key it is the only way to wedge over an object that IS durable: the settling read is what + /// would otherwise prove the commit inside the same call and report it committed. + String landed_ack_lost_substr; + bool landed_ack_lost_latched = false; + + /// The store's own PRECONDITION REFUSAL -- a value, not an exception, so the call carries no + /// ambiguity at all. It is the only shape that can report a conflict naming NO occupant: over an + /// absent key the settling read proves absence, and with `refuse_read_after_precondition` it is + /// refused outright. Latched by nature, because a substring match is either armed or it is not. + String refuse_precondition_substr; + bool refuse_read_after_precondition = false; + + /// Straight past every seam below, for the writes a test makes on its own behalf. `create` + /// reaches the store through the VIRTUAL `write`, so a qualified call cannot bypass this override -- + /// only this flag can. + std::atomic bypass_seams{false}; + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { - if (fail_get_count > 0 && !fail_get_substr.empty() && key.find(fail_get_substr) != String::npos) - { - if (fail_get_skip > 0) - --fail_get_skip; - else - { - --fail_get_count; - throw Poco::TimeoutException("WedgeTestBackend: simulated lost GET (read response never arrived)"); - } - } - return CountingBackend::get(key, range); + if (!bypass_seams.load(std::memory_order_acquire) && refuse_read_after_precondition + && !refuse_precondition_substr.empty() && key.find(refuse_precondition_substr) != String::npos) + throwDefiniteStoreRefusal("WedgeTestBackend: the settling read is definitively refused"); + if (fail_get_latched && !fail_get_substr.empty() && key.find(fail_get_substr) != String::npos) + throw Poco::TimeoutException("WedgeTestBackend: simulated lost read (response never arrived)"); + return CountingBackend::read(key, access); + } + + /// The store's own definitive answer, which every loop here surfaces unchanged rather than + /// reissuing. Only the S3 classification recognises it, so a build without S3 raises the + /// deterministic-local class instead -- also never reissued, but a different arm, which is why the + /// fixtures that need this shape are guarded. + [[noreturn]] static void throwDefiniteStoreRefusal(const String & what) + { +#if USE_AWS_S3 + throw DB::S3Exception(what, Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); +#else + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "{} (requires S3 error classification)", what); +#endif } - CasResult casPut(const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + /// Every write fault hangs off the ONE keyed primitive; which of them applies is decided by whether + /// the write carries a precondition, which is what used to separate a replace from a create. + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - if (fail_cas_count > 0 && !fail_cas_substr.empty() && key.find(fail_cas_substr) != String::npos) + if (bypass_seams.load(std::memory_order_acquire)) + return CountingBackend::write(key, bytes, expected_value, access); + if (expected_value) { - --fail_cas_count; - throw Poco::TimeoutException("WedgeTestBackend: simulated ambiguous checkpoint CAS"); + if ((fail_cas_latched || fail_cas_count > 0) && !fail_cas_substr.empty() + && key.find(fail_cas_substr) != String::npos) + { + if (!fail_cas_latched) + --fail_cas_count; + throw Poco::TimeoutException("WedgeTestBackend: simulated ambiguous checkpoint replace"); + } + return CountingBackend::write(key, bytes, expected_value, access); } - return CountingBackend::casPut(key, bytes, expected, meta); + return createForTest(key, bytes, access); } /// Park a matching PUT until `releaseBlock()`, notifying `awaitBlockEntered()` on arrival, so a @@ -219,21 +266,35 @@ class WedgeTestBackend : public CountingBackend } /// Write straight through, bypassing every fault and block seam above -- how a test models what a - /// SUCCESSOR (another process entirely) put at a key. Using the faulting entry point instead would - /// park the test's own write on the very gate it is trying to drive a scenario through. The - /// qualification must name the THREE-argument overload: `Backend`'s two-argument convenience - /// forwards to the VIRTUAL one, so `CountingBackend::putIfAbsent(key, bytes)` would dispatch right - /// back into the override above and deadlock the test against its own block. - PutResult putAsSuccessor(const String & key, const String & bytes) + /// SUCCESSOR (another process entirely) put at a key. Routing it through the seams instead would + /// park the test's own write on the very gate it is trying to drive a scenario through, or spend + /// the fault meant for the lane's attempt. `create` reaches the store through the VIRTUAL `write`, + /// so the override above runs either way -- `bypass_seams` is what it reads to step aside. + WriteResult putAsSuccessor(const String & key, const String & bytes) { - return CountingBackend::putIfAbsent(key, bytes, ObjectMeta{}); + bypass_seams.store(true, std::memory_order_release); + DB::Cas::tests::OperationForTest op(*this); + const WriteResult result = (*op).create(key, bytes, Retry::once()); + bypass_seams.store(false, std::memory_order_release); + return result; } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected createForTest( + const String & key, const String & bytes, DB::Cas::TransportAccess & access) { - if (ambiguous_count > 0 && !ambiguous_substr.empty() && key.find(ambiguous_substr) != String::npos) + if (!refuse_precondition_substr.empty() && key.find(refuse_precondition_substr) != String::npos) + return std::unexpected(DB::Cas::Backend::RawConflict{}); + if (landed_ack_lost_latched && !landed_ack_lost_substr.empty() + && key.find(landed_ack_lost_substr) != String::npos) { - --ambiguous_count; + (void)CountingBackend::write(key, bytes, std::nullopt, access); /// the write LANDS + throw Poco::TimeoutException("WedgeTestBackend: our own bytes landed and the response was lost"); + } + if ((ambiguous_latched || ambiguous_count > 0) && !ambiguous_substr.empty() + && key.find(ambiguous_substr) != String::npos) + { + if (!ambiguous_latched) + --ambiguous_count; throw Poco::TimeoutException("WedgeTestBackend: simulated ambiguous PUT (response lost, nothing landed)"); } if (definite_count > 0 && !definite_substr.empty() && key.find(definite_substr) != String::npos) @@ -243,9 +304,11 @@ class WedgeTestBackend : public CountingBackend } /// AFTER the ambiguity seam, so arming both scripts one call's attempts in order: the first /// attempt goes ambiguous, the reissue is definitively refused. - if (s3_definite_count > 0 && !s3_definite_substr.empty() && key.find(s3_definite_substr) != String::npos) + if ((s3_definite_latched || s3_definite_count > 0) && !s3_definite_substr.empty() + && key.find(s3_definite_substr) != String::npos) { - --s3_definite_count; + if (!s3_definite_latched) + --s3_definite_count; #if USE_AWS_S3 throw DB::S3Exception("WedgeTestBackend: simulated malformed request", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); @@ -257,7 +320,7 @@ class WedgeTestBackend : public CountingBackend if (conflict_count > 0 && !conflict_substr.empty() && key.find(conflict_substr) != String::npos) { --conflict_count; - CountingBackend::putIfAbsent(key, conflict_bytes, meta); + (void)CountingBackend::write(key, conflict_bytes, std::nullopt, access); throw Poco::TimeoutException("WedgeTestBackend: a successor's object landed; our response was lost"); } { @@ -270,7 +333,7 @@ class WedgeTestBackend : public CountingBackend block_cv.wait_for(lk, std::chrono::seconds(20), [&] { return !block_armed; }); } } - return CountingBackend::putIfAbsent(key, bytes, meta); + return CountingBackend::write(key, bytes, std::nullopt, access); } private: @@ -287,9 +350,11 @@ String logPrefix(const PoolPtr & store, const RootNamespace & ns) return store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; } -CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const RootNamespace & ns) +CatalogEntry catalogEntryOrThrow(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) { - const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + const RefCatalog catalog = CasRefCatalog::read(op, layout).catalog; const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; @@ -302,34 +367,38 @@ CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const /// Wedge tests address raw ref-log keys at Stage A's deterministic sentinel identity, but their /// catalog fixture must still use production's `Creating -> _ckpt -> Live` birth order. A fixed /// creator identity makes the durable genesis checkpoint deterministic too. -void admitProperlyBornEntry(Backend & backend, const Layout & layout, const RootNamespace & ns) +void admitProperlyBornEntry(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) { + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); const CatalogEntry creating{ .ns = ns, .state = NsState::Creating, .incarnation = DB::Cas::tests::fixture::fixtureLife(ns).incarnation, .creator = CreatorFence{.server_root_id = "test", .writer_epoch = 1, .fence_generation = 1}, }; - CasRefCatalog::casAdmitEntry(backend, layout, /*gc_shards=*/1, creating); + CasRefCatalog::casAdmitEntry(op, layout, /*gc_shards=*/1, creating); - const CkptDeadline deadline{.now_ms = [] { return uint64_t{1000}; }, .deadline_ms = 60000}; - ASSERT_EQ( - CasRefCatalog::completeCreation( - backend, layout, creating, /*admitted_generation=*/1, [](uint64_t) {}, deadline), - CasRefCatalog::NamespaceCreationOutcome::Live); + ASSERT_EQ(CasRefCatalog::completeCreation(op, layout, creating), + CasRefCatalog::NamespaceCreationOutcome::Live); } CatalogEntry replaceCatalogLifeForWedgeRace( - Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) + const BackendPtr & backend, const Layout & layout, const CatalogEntry & predecessor, + UInt128 successor_incarnation) { - const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(op, layout); RefCatalog without_predecessor = before_delete.catalog; std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) { return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; }); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome - != CasOutcome::Committed) + if (!before_delete.etag + || !std::holds_alternative(op.replace( + layout.refCatalogKey(), encodeRefCatalog(without_predecessor), *before_delete.etag, + Retry::standard()))) throw std::runtime_error("test failed to retire exact predecessor catalog life"); CatalogEntry successor{ @@ -337,11 +406,12 @@ CatalogEntry replaceCatalogLifeForWedgeRace( .state = NsState::Live, .incarnation = successor_incarnation, .creator = std::nullopt}; - const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(op, layout); RefCatalog reborn = after_delete.catalog; reborn.entries.push_back(successor); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome - != CasOutcome::Committed) + if (!after_delete.etag + || !std::holds_alternative(op.replace( + layout.refCatalogKey(), encodeRefCatalog(reborn), *after_delete.etag, Retry::standard()))) throw std::runtime_error("test failed to publish successor catalog life"); return successor; } @@ -350,7 +420,8 @@ CatalogEntry replaceCatalogLifeForWedgeRace( /// hand-rolled parse), so an assertion about `prev_epoch_seal` is an assertion about the WIRE. RefLogTxn readRefLogTxn(Backend & backend, const Layout & layout, const RootNamespace & ns, const RefTxnId & id) { - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id), Retry::standard()); if (!got) throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "no ref-log object at {}-{}", id.writer_epoch, id.ref_sequence); return decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id); @@ -397,6 +468,65 @@ void armOneShotInstallFailure(const PoolPtr & store) }); } +/// Wedge `ns`'s append lane the way the engine actually reaches that state: EVERY attempt of the +/// ref-log create is unresolved (the response is lost and nothing lands), the settling read proves the +/// key still absent, and the call gives up at its own retry window having sent something -- which is +/// the `sent_any` half of the wedge rule. A one-shot fault cannot produce it: the reissue would settle +/// the key and commit. So the fault stays armed for the whole call and is cleared here. +/// +/// The pacing assertions are what make a fixture whose sleep seam is not wired FAIL rather than sleep +/// the whole window out for real. +/// Wedge `ns`'s lane over an object that IS durable: our own bytes land, the response is lost, and +/// every settling read of the key is lost too, so the call gives up at its own window without ever +/// learning that it committed. BOTH legs are required -- a readable key proves the commit inside the +/// same call and reports it committed, and a read fault that clears mid-call is simply reissued. +void wedgeLaneOverADurableObject(VirtualRetryClock & clock, WedgeTestBackend & backend, + const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + const size_t pauses_before = clock.pauseCount(); + const uint64_t clock_before = clock.nowMs(); + + backend.landed_ack_lost_substr = logPrefix(store, ns); + backend.landed_ack_lost_latched = true; + backend.fail_get_substr = logPrefix(store, ns); + backend.fail_get_latched = true; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, ref); }); + backend.landed_ack_lost_latched = false; + backend.landed_ack_lost_substr.clear(); + backend.fail_get_latched = false; + backend.fail_get_substr.clear(); + + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + ASSERT_GT(clock.pauseCount(), pauses_before + 1) + << "the settling read's reissues must pace through the injected sleep, never a real one"; + ASSERT_GE(clock.nowMs() - clock_before, 60000u) + << "the give-up must be the read's own retry window, not a single failed read"; +} + +void wedgeLaneOnUnresolvedAppend(VirtualRetryClock & clock, WedgeTestBackend & backend, + const PoolPtr & store, const RootNamespace & ns, + const std::function & drive) +{ + const size_t pauses_before = clock.pauseCount(); + const uint64_t clock_before = clock.nowMs(); + + backend.ambiguous_substr = logPrefix(store, ns); + backend.ambiguous_latched = true; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, drive); + backend.ambiguous_latched = false; + backend.ambiguous_substr.clear(); + + ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + /// `writeTotal` cannot prove "more than one attempt": the ambiguous fault throws before ever + /// reaching the counted primitive, so a latched reissue never moves it. The pacing below is what + /// proves multiple reissues happened. + ASSERT_GT(clock.pauseCount(), pauses_before + 1) + << "the reissues must pace through the injected sleep, never a real one"; + ASSERT_LE(clock.longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; + ASSERT_GE(clock.nowMs() - clock_before, 60000u) + << "the give-up must be the call's own retry window, not a pre-attempt refusal"; +} + } /// =================================================================================== @@ -410,23 +540,20 @@ void armOneShotInstallFailure(const PoolPtr & store) TEST(CASRefWedgeEveryAttempt, AmbiguousPutWedgesTheLaneAndTheNextFlushsCreateAdoptsItExactlyOnce) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_created"}; /// Stage B (Task 4-C): `logPrefix` below computes its fault-injection match at the sentinel; /// pinning `ns` there BEFORE the first real touch keeps the real production birth landing on the /// same key the fault targets. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "a wedged transaction is not applied"; const String wedged_key = store->wedgedKeyForTest(ns); - ASSERT_FALSE(backend->get(wedged_key).has_value()) << "the ambiguous attempt wrote nothing"; + ASSERT_FALSE(readObj(*backend, wedged_key).has_value()) << "the ambiguous attempt wrote nothing"; const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); /// The next caller's flush resolves the wedge with ONE create, adopts it, and only then carves and @@ -436,7 +563,7 @@ TEST(CASRefWedgeEveryAttempt, AmbiguousPutWedgesTheLaneAndTheNextFlushsCreateAdo EXPECT_FALSE(store->refLaneWedgedForTest(ns)); EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the adopted wedge applied its drop"; EXPECT_FALSE(store->resolveRef(ns, "y").has_value()) << "the resolving flush committed its own drop"; - EXPECT_TRUE(backend->get(wedged_key).has_value()) << "the wedged transaction is durable at its own key"; + EXPECT_TRUE(readObj(*backend, wedged_key).has_value()) << "the wedged transaction is durable at its own key"; EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before + 2) << "the adopted wedge and the ordinary commit must each join the tail exactly once"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); @@ -445,54 +572,57 @@ TEST(CASRefWedgeEveryAttempt, AmbiguousPutWedgesTheLaneAndTheNextFlushsCreateAdo TEST(CASRefWedgeEveryAttempt, DurableCreatedWedgeNeedsRecoveryWhenItsFrontierCannotBePublished) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_created_frontier_failed"}; - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); - ASSERT_FALSE(backend->get(wedged_key)); + ASSERT_FALSE(readObj(*backend, wedged_key)); const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); const NamespaceLifeId life = *store->refTableLifeForTest(ns); const String ckpt_key = store->layout().refCkptKey(life); - const RefCkpt ckpt_before = decodeRefCkpt(backend->get(ckpt_key)->bytes); + const RefCkpt ckpt_before = decodeRefCkpt(readObj(*backend, ckpt_key)->bytes); + /// Latched, not counted: the frontier publish reissues an ambiguous replace until ITS window + /// closes, so a bounded fault would simply be outlived and the publication would succeed. backend->fail_cas_substr = ckpt_key; - backend->fail_cas_count = 200; + backend->fail_cas_latched = true; + const size_t pauses_before = clock->pauseCount(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "y"); }); + backend->fail_cas_latched = false; + backend->fail_cas_substr.clear(); - EXPECT_TRUE(backend->get(wedged_key)) << "the exact wedged log was proven durable"; + EXPECT_GT(clock->pauseCount(), pauses_before + 1) + << "the publication's reissues must pace through the injected sleep, never a real one"; + EXPECT_TRUE(readObj(*backend, wedged_key)) << "the exact wedged log was proven durable"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery) << "a durable log without a confirmed frontier must not return to Ready"; EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) << "the unfrontiered wedge must not be installed into the resident table"; - EXPECT_EQ(decodeRefCkpt(backend->get(ckpt_key)->bytes), ckpt_before); + EXPECT_EQ(decodeRefCkpt(readObj(*backend, ckpt_key)->bytes), ckpt_before); } TEST(CASRefWedgeEveryAttempt, RetiredLifeRefusesWedgeRetryBeforeAnyRequestOrAdoption) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge-retired-before-retry"}; - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, store->layout(), ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, store->layout(), ns); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(predecessor.ns, predecessor.incarnation); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); - ASSERT_FALSE(backend->get(wedged_key)); + ASSERT_FALSE(readObj(*backend, wedged_key)); std::mutex mutex; std::condition_variable cv; @@ -524,13 +654,13 @@ TEST(CASRefWedgeEveryAttempt, RetiredLifeRefusesWedgeRetryBeforeAnyRequestOrAdop } const CatalogEntry successor - = replaceCatalogLifeForWedgeRace(*backend, store->layout(), predecessor, UInt128{0x71f2}); + = replaceCatalogLifeForWedgeRace(backend, store->layout(), predecessor, UInt128{0x71f2}); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); - ASSERT_EQ(backend->putIfAbsent(store->layout().refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + createObj(*backend, store->layout().refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ .life_epoch = store->liveWriterEpoch(), .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt})); store->invalidateRemovedCatalogLife(predecessor_life); backend->resetCounts(); @@ -545,7 +675,7 @@ TEST(CASRefWedgeEveryAttempt, RetiredLifeRefusesWedgeRetryBeforeAnyRequestOrAdop EXPECT_TRUE(retry_error); EXPECT_EQ(backend->putCount(wedged_key), 0u) << "retirement must refuse before the retry send"; EXPECT_EQ(backend->getCount(wedged_key), 0u) << "a refused retry needs no occupant resolution read"; - EXPECT_FALSE(backend->get(wedged_key)) << "the predecessor wedge was adopted or made durable"; + EXPECT_FALSE(readObj(*backend, wedged_key)) << "the predecessor wedge was adopted or made durable"; EXPECT_NO_THROW((void)store->listRefs(ns)); ASSERT_TRUE(store->refTableLifeForTest(ns)); EXPECT_EQ(*store->refTableLifeForTest(ns), successor_life); @@ -557,26 +687,19 @@ TEST(CASRefWedgeEveryAttempt, RetiredLifeRefusesWedgeRetryBeforeAnyRequestOrAdop /// wedge's, and the transaction is adopted -- ONCE, not once per attempt. TEST(CASRefWedgeEveryAttempt, OwnLandedAttemptIsAdoptedFromOccupiedWithoutDoubleApply) { - auto backend = std::make_shared(); - /// Disarmed while the fixture is built: the one-shot fault matches ANY key until a substring is - /// set, and the pool's own bootstrap PUT would otherwise consume it. - backend->fired = true; - auto store = openPool(backend, singleAttemptBudget()); + auto backend = std::make_shared(); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_occupied_mine"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->key_substr = logPrefix(store, ns); - backend->lose_resolve_read = true; - backend->fired = false; /// armed: the next `_log/` PUT lands and loses its ack - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOverADurableObject(*clock, *backend, store, ns, "x"); const String wedged_key = store->wedgedKeyForTest(ns); - ASSERT_TRUE(backend->get(wedged_key).has_value()) << "this fault LANDS the write; only the ack was lost"; + ASSERT_TRUE(readObj(*backend, wedged_key).has_value()) << "this fault LANDS the write; only the ack was lost"; ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "durable, but not applied while wedged"; const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); const uint64_t puts_before = backend->putCount(wedged_key); @@ -597,16 +720,14 @@ TEST(CASRefWedgeEveryAttempt, OwnLandedAttemptIsAdoptedFromOccupiedWithoutDouble TEST(CASRefWedgeEveryAttempt, DefiniteRefusalOfARetryAttemptKeepsTheLaneWedged) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_ambiguous_then_definite"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); const RefTxnId wedged_id = store->layout().parseRefObjectKey(wedged_key)->txn_id; @@ -618,7 +739,7 @@ TEST(CASRefWedgeEveryAttempt, DefiniteRefusalOfARetryAttemptKeepsTheLaneWedged) EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "a definite refusal AFTER an ambiguous attempt must not unwedge"; EXPECT_EQ(store->wedgedKeyForTest(ns), wedged_key) << "the SAME wedge, not a fresh one"; EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "nothing was adopted"; - EXPECT_FALSE(backend->get(wedged_key).has_value()) << "and nothing became durable"; + EXPECT_FALSE(readObj(*backend, wedged_key).has_value()) << "and nothing became durable"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) << "a wedged lane's steady state is 'may be durable, not applied'"; @@ -631,53 +752,58 @@ TEST(CASRefWedgeEveryAttempt, DefiniteRefusalOfARetryAttemptKeepsTheLaneWedged) } /// The SAME rule one level down, and the level where it was actually broken. The test above splits the -/// two attempts across two CALLS, which the wedge already handles. Inside ONE call the controller used -/// to report the LAST attempt's outcome: an ambiguous attempt followed by a definitively refused reissue -/// came back `DefiniteFailure` -- the verdict that means "the key is provably unwritten". It is not. The -/// refusal proves only that the SECOND request never applied; the first may still be in flight and may -/// still land, and `unresolvedProvesNothingWasSent` is false for exactly that reason. So the CALL is -/// unresolved, and a definite verdict is only ever the whole call's. +/// two attempts across two CALLS, which the wedge already handles. Inside ONE call the verdict used to +/// be the LAST attempt's: an ambiguous attempt followed by a definitively refused reissue came back as +/// a proven refusal -- the verdict that means "the key is provably unwritten". It is not. The refusal +/// proves only that the SECOND request never applied; the first may still be in flight and may still +/// land. So the CALL gives up, and a refusal is only ever reported when NO attempt of the call was +/// ambiguous. /// -/// The new reason lands on the fail-close side of the predicate the ledger acts on. Asserted at compile -/// time, beside the behaviour, because a member added to the enum without classifying it is precisely -/// how the wedge would silently stop happening. -static_assert(!unresolvedProvesNothingWasSent(CasUnresolvedReason::DefiniteFailureAfterAmbiguity)); - +/// Driven on an injected clock: the reissue schedule is what makes the two attempts happen, and a real +/// one would make this test sleep. TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalCannotSpeakForAnEarlierAmbiguousAttemptOfTheSameCall) { #if !USE_AWS_S3 - GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; + GTEST_SKIP() << "the store-refusal classification requires S3 error types (USE_AWS_S3 off)"; #else auto backend = std::make_shared(); - CasRequestController controller(backend, twoAttemptBudget()); - const std::function fence_ok = [] { return true; }; - - /// One call, two attempts: ambiguous, then definitively refused. + uint64_t clock = 0; + size_t pauses = 0; + /// The sleep advances the same clock the policy window is read from, plus a millisecond, because + /// full jitter can draw a zero pause and a clock that does not move never reaches the deadline. + CasRequests requests(backend, Fence::open(), + [&clock]() -> uint64_t { return clock; }, + [&clock, &pauses](uint64_t ms) { ++pauses; clock += ms + 1; }); + CasOperation op = requests.admit(); + + /// One call: the first attempt ambiguous, and every attempt after it refused by the store. The + /// refusal has to stay armed, because the engine does not stop at it -- a refusal that follows an + /// ambiguous attempt of the same call proves nothing about that attempt, so the call keeps + /// reissuing until its own window closes. backend->ambiguous_substr = "key/"; backend->ambiguous_count = 1; backend->s3_definite_substr = "key/"; - backend->s3_definite_count = 1; + backend->s3_definite_latched = true; - CasUnresolvedReason reason = CasUnresolvedReason::NotUnresolved; - const CasWriteOutcome outcome = - controller.putIfAbsentControlled("key/haunted", "bytes", fence_ok, /*out_token=*/nullptr, &reason); - - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved) - << "a definite refusal of the SECOND attempt cannot retire the first attempt's ambiguity"; - EXPECT_EQ(reason, CasUnresolvedReason::DefiniteFailureAfterAmbiguity); - EXPECT_FALSE(unresolvedProvesNothingWasSent(reason)) + const WriteResult haunted = op.create("key/haunted", "bytes", Retry::within(30'000)); + const auto * gave_up = std::get_if(&haunted); + ASSERT_TRUE(gave_up != nullptr) + << "a store refusal of a LATER attempt cannot retire the first attempt's ambiguity"; + EXPECT_TRUE(gave_up->sent_any) << "the caller must keep protecting itself: an earlier attempt was sent and may yet land"; - EXPECT_FALSE(backend->get("key/haunted").has_value()) + EXPECT_GT(pauses, 1u) << "the reissues must pace through the injected sleep, never a real one"; + EXPECT_GE(clock, 20'000u) << "and the call must end at its own 30 s window"; + backend->s3_definite_latched = false; + CasOperation reader = requests.admit(); + EXPECT_FALSE(reader.head("key/haunted", Retry::within(30'000)).has_value()) << "and the key is still empty -- which is exactly why an absent read settles nothing"; - /// THE CONTROL. Aggregation must not soften a definite refusal that speaks for the whole call: with - /// no ambiguous predecessor, the first attempt's whitelisted rejection is still `DefiniteFailure`, - /// and the ledger may still free the id on it. + /// THE CONTROL. Aggregation must not soften a refusal that speaks for the whole call: with no + /// ambiguous predecessor, the first attempt's rejection is still `Refused`, and the ledger may + /// still free the id on it. backend->s3_definite_count = 1; - CasUnresolvedReason clean_reason = CasUnresolvedReason::NotUnresolved; - EXPECT_EQ(controller.putIfAbsentControlled("key/clean", "bytes", fence_ok, /*out_token=*/nullptr, &clean_reason), - CasWriteOutcome::DefiniteFailure); - EXPECT_EQ(clean_reason, CasUnresolvedReason::NotUnresolved); + CasOperation clean = requests.admit(); + EXPECT_TRUE(std::holds_alternative(clean.create("key/clean", "bytes", Retry::within(30'000)))); #endif } @@ -692,24 +818,34 @@ TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCa GTEST_SKIP() << "DefiniteFailure classification requires S3 error types (USE_AWS_S3 off)"; #else auto backend = std::make_shared(); - auto store = openPool(backend, twoAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_one_call_ambiguous_then_definite"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); + /// The first attempt goes ambiguous; the store then refuses every attempt after it, for as long as + /// the call keeps making them. A one-shot refusal would not reproduce the sequence this test names: + /// the engine reissues past it, and the third attempt would simply commit. backend->ambiguous_substr = logPrefix(store, ns); backend->ambiguous_count = 1; backend->s3_definite_substr = logPrefix(store, ns); - backend->s3_definite_count = 1; + backend->s3_definite_latched = true; const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(); + const size_t pauses_before = clock->pauseCount(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + backend->s3_definite_latched = false; + backend->s3_definite_substr.clear(); + backend->ambiguous_substr.clear(); + EXPECT_GT(clock->pauseCount(), pauses_before + 1) + << "the reissues must pace through the injected sleep, never a real one"; EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "one call whose first attempt is unresolved leaves an object that may become durable"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) @@ -719,7 +855,7 @@ TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCa EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1); EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "nothing is applied while the lane is wedged"; const String wedged_key = store->wedgedKeyForTest(ns); - EXPECT_FALSE(backend->get(wedged_key).has_value()) << "and nothing became durable"; + EXPECT_FALSE(readObj(*backend, wedged_key).has_value()) << "and nothing became durable"; /// And it still recovers by the ordinary route: the next flush's bounded create lands the wedged /// transaction and adopts it, so wedging costs availability only until the next caller arrives. @@ -742,18 +878,16 @@ TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCa TEST(CASRefWedgeEveryAttempt, SuccessorSealAtTheWedgedKeyRejectsConclusivelyAndSourcesPrevEpochSeal) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/wedge_sealed"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); const uint64_t epoch = store->liveWriterEpoch(); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); const RefTxnId seal_id = layout.parseRefObjectKey(wedged_key)->txn_id; ASSERT_EQ(seal_id.writer_epoch, epoch); @@ -762,7 +896,7 @@ TEST(CASRefWedgeEveryAttempt, SuccessorSealAtTheWedgedKeyRejectsConclusivelyAndS const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); /// A successor closes our epoch at exactly the slot our attempt was aiming at. - ASSERT_EQ(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)))); /// The next caller's resolution meets the seal. Its own items fail -- permanently, not "retry /// later": nothing about this lane's epoch will ever accept a write again. @@ -786,7 +920,7 @@ TEST(CASRefWedgeEveryAttempt, SuccessorSealAtTheWedgedKeyRejectsConclusivelyAndS /// wedge-resolve site does. const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "y"); }); - EXPECT_EQ(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch, seal_id.ref_sequence + 1})), std::nullopt) + EXPECT_EQ(readObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch, seal_id.ref_sequence + 1})), std::nullopt) << "nothing of ours may exist above the seal in the closed epoch"; EXPECT_TRUE(store->mayMutate()) << "meeting a successor's seal is the protocol working, not an anomaly"; EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) @@ -802,8 +936,8 @@ TEST(CASRefWedgeEveryAttempt, SuccessorSealAtTheWedgedKeyRejectsConclusivelyAndS } /// The wire round trip of the same rule, driven from the OTHER producer of `last_epoch_seal`: -/// recovery's CAS-walk (Task 6), stood in for here by its test seam. The point is the encode call -/// site, which is this task's. +/// recovery's CAS-walk, represented here by its test seam. The point is the encode call +/// site. TEST(CASRefWedgeEveryAttempt, OrdinaryFirstAppendAfterASealedTransitionCarriesTheExactPrevEpochSeal) { auto backend = std::make_shared(); @@ -812,7 +946,7 @@ TEST(CASRefWedgeEveryAttempt, OrdinaryFirstAppendAfterASealedTransitionCarriesTh const RootNamespace ns{"srv1/prev_epoch_seal_roundtrip"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `readRefLogTxn` above /// reads that exact key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); const uint64_t epoch = store->liveWriterEpoch(); publishEmptyPart(store, ns, "x"); @@ -846,7 +980,7 @@ TEST(CASRefWedgeEveryAttempt, GenesisBirthAtAHighEpochCarriesNoPrevEpochSeal) const RootNamespace ns{"srv1/genesis_at_five"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `readRefLogTxn` above /// reads that exact key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); bumpFenceGeneration(store, 5); ASSERT_EQ(store->liveWriterEpoch(), 5u); @@ -870,23 +1004,21 @@ TEST(CASRefWedgeEveryAttempt, GenesisBirthAtAHighEpochCarriesNoPrevEpochSeal) TEST(CASRefWedgeEveryAttempt, ForeignNonSealOccupantIsCorruptedDataAndSchedulesARemount) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_foreign"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); /// Something that is neither our bytes nor a seal occupies the slot. - ASSERT_EQ(backend->putAsSuccessor(wedged_key, "not a ref-log object at all").outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(backend->putAsSuccessor(wedged_key, "not a ref-log object at all"))); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "y"); }); @@ -906,14 +1038,14 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteProvenDifferentObjectAlsoSchedulesARemou auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/append_site_foreign"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); /// Occupy the id the next append will derive with a foreign object, so its create conflicts and /// the controller's resolve-before-reissue proves the occupant is not ours. const RefTxnId next{store->liveWriterEpoch(), 3}; - ASSERT_EQ(backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next), - "a different object entirely").outcome, PutOutcome::Done); + createObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next), + "a different object entirely"); const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); @@ -933,19 +1065,17 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteProvenDifferentObjectAlsoSchedulesARemou TEST(CASRefWedgeEveryAttempt, RetryUnderAnOlderAdmissionGenerationSendsNothing) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_old_generation"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); const uint64_t epoch = store->liveWriterEpoch(); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); ASSERT_EQ(store->wedgedAdmittedGenerationForTest(ns), store->fenceGeneration()) << "the wedge records the generation it was admitted under"; @@ -960,7 +1090,7 @@ TEST(CASRefWedgeEveryAttempt, RetryUnderAnOlderAdmissionGenerationSendsNothing) EXPECT_EQ(backend->putCount(wedged_key), puts_before) << "the retry must be refused pre-attempt: nothing may reach the store under a foreign generation"; EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "and the wedge is untouched"; - EXPECT_FALSE(backend->get(wedged_key).has_value()); + EXPECT_FALSE(readObj(*backend, wedged_key).has_value()); } /// The post-I/O recheck, deterministically. The retry's create is parked mid-flight; while it is @@ -971,20 +1101,18 @@ TEST(CASRefWedgeEveryAttempt, RetryUnderAnOlderAdmissionGenerationSendsNothing) TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterAFenceBumpAndSuccessorSealIsInert) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/wedge_blocked_io"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); const uint64_t epoch = store->liveWriterEpoch(); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); const RefTxnId seal_id = layout.parseRefObjectKey(wedged_key)->txn_id; const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); @@ -999,7 +1127,7 @@ TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterAFenceBumpAndSuccessorSealIsIne backend->awaitBlockEntered(); /// Everything that makes this runtime superseded happens INSIDE the I/O window. - ASSERT_EQ(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative(backend->putAsSuccessor(wedged_key, epochSealBytes(ns, seal_id)))); bumpFenceGeneration(store, epoch + 1); backend->releaseBlock(); resolver.join(); @@ -1024,18 +1152,16 @@ TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterAFenceBumpAndSuccessorSealIsIne TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterTheWedgeIdentityChangedIsInert) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_identity_changed"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { store->dropRef(ns, "x"); }); const String wedged_key = store->wedgedKeyForTest(ns); const RefTxnId wedged_id = store->layout().parseRefObjectKey(wedged_key)->txn_id; const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); @@ -1065,23 +1191,19 @@ TEST(CASRefWedgeEveryAttempt, ResultReleasedAfterTheWedgeIdentityChangedIsInert) /// `NeedsRecovery`. It drops the attempt and forbids another write until replay catches the cache up. TEST(CASRefWedgeEveryAttempt, KnownDurableInstallFailureMovesDirectlyToRecovery) { - auto backend = std::make_shared(); - backend->fired = true; /// disarmed while the fixture is built (see the adoption test above) - auto store = openPool(backend, singleAttemptBudget()); + auto backend = std::make_shared(); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/wedge_floor"}; /// Stage B (Task 4-C): pin to the sentinel before the first real touch -- `logPrefix` below matches /// its fault at that key. - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - backend->key_substr = logPrefix(store, ns); - backend->lose_resolve_read = true; - backend->fired = false; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + wedgeLaneOverADurableObject(*clock, *backend, store, ns, "x"); const String wedged_key = store->wedgedKeyForTest(ns); - ASSERT_TRUE(backend->get(wedged_key).has_value()) << "the wedged transaction is durable"; + ASSERT_TRUE(readObj(*backend, wedged_key).has_value()) << "the wedged transaction is durable"; /// The adoption reaches its install region and the install throws. armOneShotInstallFailure(store); expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "y"); }); @@ -1116,7 +1238,7 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteMeetingASuccessorSealIsAConclusiveReject auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/append_site_seal"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); const uint64_t epoch = store->liveWriterEpoch(); @@ -1166,11 +1288,11 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesAConclusiveFirstRefLogRejectio auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/birth_ckpt_debris"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch const RefTxnId genesis{store->liveWriterEpoch(), 1}; const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); - const auto ckpt_before = backend->get(ckpt_key); + const auto ckpt_before = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; /// A successor's epoch seal lands at exactly the id this first `NamespaceBirth` transaction derives. @@ -1180,7 +1302,7 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesAConclusiveFirstRefLogRejectio expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { publishEmptyPart(store, ns, "x"); }); - const auto ckpt_after = backend->get(ckpt_key); + const auto ckpt_after = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) << "the creation checkpoint must survive a conclusively rejected first ref-log PUT unchanged"; @@ -1201,7 +1323,7 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesALaterConclusiveRejection) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/birth_ckpt_survives_live"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch /// ONE `publishEmptyPart` reaches sequence 2 (the precommit-add chunk at seq 1 carries the first /// `NamespaceBirth`, the promote chunk lands at seq 2), so `next` /// below is the SAME `{epoch, 3}` the sibling `AppendSiteMeetingASuccessorSealIsAConclusiveRejectionNotInterference` @@ -1210,7 +1332,7 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesALaterConclusiveRejection) publishEmptyPart(store, ns, "x"); const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); - const auto ckpt_before = backend->get(ckpt_key); + const auto ckpt_before = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation step must have published a real _ckpt"; const RefTxnId next{store->liveWriterEpoch(), 3}; @@ -1220,7 +1342,7 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesALaterConclusiveRejection) expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); - const auto ckpt_after = backend->get(ckpt_key); + const auto ckpt_after = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_after.has_value()) << "a Live namespace's _ckpt must never be deleted by this path"; EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) << "not merely present but UNCHANGED -- no code anywhere on this path deletes _ckpt any more " @@ -1238,67 +1360,53 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesALaterConclusiveRejection) TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesWhenTheFirstNamespaceBirthIsAmbiguous) { auto backend = std::make_shared(); - auto store = openPool(backend, singleAttemptBudget()); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/birth_ckpt_ambiguous"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); - const auto ckpt_before = backend->get(ckpt_key); + const auto ckpt_before = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; - backend->ambiguous_substr = logPrefix(store, ns); - backend->ambiguous_count = 1; - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "x"); }); - - ASSERT_TRUE(store->refLaneWedgedForTest(ns)) << "an ambiguous outcome must WEDGE the lane, not " - "resolve into one of the conclusive branches"; - const auto ckpt_after = backend->get(ckpt_key); + /// The wedge is the assertion: an ambiguous outcome must not resolve into one of the conclusive + /// branches, so `wedgeLaneOnUnresolvedAppend` insisting on it is what this row needs. + wedgeLaneOnUnresolvedAppend(*clock, *backend, store, ns, [&] { publishEmptyPart(store, ns, "x"); }); + const auto ckpt_after = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) << "the creation checkpoint must survive an ambiguous first ref-log outcome unchanged"; } -/// Final review F5: the two other removed call sites, given their own first-`NamespaceBirth` survival rows. -/// The reversed test above pins the `SuccessorSeal` branch; the sibling below it pins the ambiguous -/// branch (never called it in the first place). The remaining two -- occupant-unreadable -/// (`CORRUPTED_DATA` from a failed adjudication read) and genuine foreign interference -- had no -/// first-`NamespaceBirth` row at all: `AppendSiteFaultsWhenTheOccupantCannotBeRead`, -/// `ForeignNonSealOccupantIsCorruptedDataAndSchedulesARemount`, and `WellFormedNonSealOccupantIsStillForeign` +/// The other two former cleanup call sites, given their own first-`NamespaceBirth` survival rows: an +/// unnameable occupant, and genuine foreign interference. Neither had such a row, because +/// `AppendSiteWedgesWhenTheSettlingReadNamesNoOccupant`, +/// `ForeignNonSealOccupantIsCorruptedDataAndSchedulesARemount` and `WellFormedNonSealOccupantIsStillForeign` /// all `publishEmptyPart` FIRST, so none of them ever carries a `birth_contribution` -- a reinstated -/// GUARDED cleanup at either of these two sites would pass the whole suite with no first-transaction case to -/// catch it. Mirrors `WellFormedNonSealOccupantIsStillForeign`'s occupant shape (a decodable, well-formed -/// NON-seal transaction at the derived key), moved to sequence 1 of a namespace with no prior ref-log -/// transaction, so this attempt's own PUT is its first. -TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesWhenTheFirstNamespaceBirthOccupantCannotBeRead) +/// guarded cleanup at either site would pass the whole suite with no first-transaction case to catch it. +TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesWhenTheFirstNamespaceBirthNamesNoOccupant) { auto backend = std::make_shared(); auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/birth_ckpt_occupant_unreadable"}; - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); const RefTxnId genesis{store->liveWriterEpoch(), 1}; const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); - const auto ckpt_before = backend->get(ckpt_key); + const auto ckpt_before = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; - backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); - backend->conflict_bytes = epochSealBytes(ns, genesis); - backend->conflict_count = 1; - /// Proper birth makes recovery first probe this absent log key. Skip that probe and the resolve - /// read that PROVES the conflict; fail only the adjudication read after it, so the occupant's - /// identity (seal vs. breach) cannot be determined. - backend->fail_get_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); - backend->fail_get_skip = 2; - backend->fail_get_count = 1; + /// The store refuses the birth create's precondition while the key is in fact ABSENT, so the + /// settling read proves absence and the conflict comes back naming no occupant at all. + backend->refuse_precondition_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), genesis); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { publishEmptyPart(store, ns, "x"); }); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { publishEmptyPart(store, ns, "x"); }); - const auto ckpt_after = backend->get(ckpt_key); + const auto ckpt_after = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) - << "the creation checkpoint must survive an occupant-unreadable first ref-log outcome unchanged"; + << "the creation checkpoint must survive an unnameable first ref-log occupant unchanged"; } /// The other former call site: a genuine breach of write-exclusivity at the first `NamespaceBirth` @@ -1309,11 +1417,11 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesFirstNamespaceBirthForeignInte auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/birth_ckpt_foreign_interference"}; - admitProperlyBornEntry(*backend, store->layout(), ns); + admitProperlyBornEntry(backend, store->layout(), ns); const RefTxnId genesis{store->liveWriterEpoch(), 1}; const String ckpt_key = layout.refCkptKey(DB::Cas::tests::fixture::fixtureLife(ns)); - const auto ckpt_before = backend->get(ckpt_key); + const auto ckpt_before = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_before.has_value()) << "the fixture's creation checkpoint must exist before the first ref-log attempt"; /// A perfectly decodable transaction for this exact namespace and id -- just not an epoch seal, and @@ -1327,50 +1435,102 @@ TEST(CASRefWedgeEveryAttempt, CreationCkptSurvivesFirstNamespaceBirthForeignInte expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { publishEmptyPart(store, ns, "x"); }); - const auto ckpt_after = backend->get(ckpt_key); + const auto ckpt_after = readObj(*backend, ckpt_key); ASSERT_TRUE(ckpt_after.has_value()); EXPECT_EQ(ckpt_after->bytes, ckpt_before->bytes) << "the creation checkpoint must survive a foreign-interference first ref-log outcome unchanged"; } -/// The same conflict, but the read that would tell a seal from a breach fails. We must then decide -/// NEITHER: fencing the mount would be a guess, and reporting a conclusive rejection would acknowledge -/// a deposition nobody observed. The id is not consumed, so the next attempt re-derives it and -/// classifies again — deferring costs one round trip and decides nothing wrongly. -TEST(CASRefWedgeEveryAttempt, AppendSiteFaultsWhenTheOccupantCannotBeRead) +/// A conflict that names NO occupant. The store refused this create's precondition, and the settling +/// read then found the key gone -- so nothing can be adjudicated: fencing the mount would be a guess, +/// and reporting a conclusive rejection would acknowledge a deposition nobody observed. We decide +/// NEITHER and WEDGE, which is what the wedge-resolution site does with the identical observation. The +/// lane must therefore stay recoverable: the next flush re-creates at the same key and adjudicates +/// whatever it finds. A terminal `Faulted` here would cost the table its writes until a remount over a +/// read that the very next attempt may complete. +TEST(CASRefWedgeEveryAttempt, AppendSiteWedgesWhenTheSettlingReadNamesNoOccupant) { auto backend = std::make_shared(); auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/append_site_unreadable"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); const RefTxnId next{store->liveWriterEpoch(), 3}; const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); - backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); - backend->conflict_bytes = epochSealBytes(ns, next); - backend->conflict_count = 1; - /// Skip the resolve read that PROVES the conflict; fail only the adjudication read after it. - backend->fail_get_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); - backend->fail_get_skip = 1; - backend->fail_get_count = 1; + /// A precondition refusal is a VALUE, not an exception, so this call carries no ambiguity -- which + /// is what lets the settling read's answer be the whole verdict. The key is absent, so that read + /// proves absence and names nobody. + backend->refuse_precondition_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(), deferred_before + 1) << "the deferral is the one quiet arm here -- it must be counted or a starved loud path is invisible"; - EXPECT_TRUE(store->mayMutate()) << "the table faults without guessing that the whole mount is corrupt"; - EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedule a remount"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1) + << "and the lane it leaves behind is a wedge, so the wedge counter must say so"; + EXPECT_TRUE(store->mayMutate()) << "the table defers without guessing that the whole mount is corrupt"; + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedules a remount"; EXPECT_EQ(store->lastEpochSealForTest(ns), std::nullopt) - << "nor record a deposition that was never actually observed"; - EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "nothing of ours became durable, so nothing is wedged"; - EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted); - expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->dropRef(ns, "x"); }); + << "nor records a deposition that was never actually observed"; + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) + << "the key holds something this call could not name, which is exactly what a wedge is for"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + + /// RECOVERABLE, and this is the half a terminal `Faulted` forecloses: with the store answering + /// normally again, the next flush's bounded create lands the wedged transaction and adopts it. + /// Triggered by a DIFFERENT ref (not another drop of "x"): the wedge's own adopted drop already + /// removes "x", so a second "drop x" from this same call would find it already gone. + backend->refuse_precondition_substr.clear(); + EXPECT_NO_THROW(publishEmptyPart(store, ns, "y")); + EXPECT_FALSE(store->refLaneWedgedForTest(ns)); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the adopted wedge applied its drop"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); } +/// The OTHER observation that names no occupant, and the one the arm is really about: the settling read +/// does not merely find the key gone, it FAILS -- definitively, so the read engine surfaces it rather +/// than reissuing. Guarded to `USE_AWS_S3` builds, where alone a store refusal is recognised as one; +/// without that classification the same fault is a deterministic local failure, a different arm. +#if USE_AWS_S3 +TEST(CASRefWedgeEveryAttempt, AppendSiteWedgesWhenTheSettlingReadItselfIsRefused) +{ + auto backend = std::make_shared(); + auto store = openPool(backend); + const Layout & layout = store->layout(); + const RootNamespace ns{"srv1/append_site_read_refused"}; + admitProperlyBornEntry(backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + + const RefTxnId next{store->liveWriterEpoch(), 3}; + const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); + + backend->refuse_precondition_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); + backend->refuse_read_after_precondition = true; + + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(), deferred_before + 1); + EXPECT_TRUE(store->mayMutate()); + EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before); + EXPECT_TRUE(store->refLaneWedgedForTest(ns)); + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + + /// Triggered by a DIFFERENT ref (not another drop of "x"): the wedge's own adopted drop already + /// removes "x", so a second "drop x" from this same call would find it already gone. + backend->refuse_read_after_precondition = false; + backend->refuse_precondition_substr.clear(); + EXPECT_NO_THROW(publishEmptyPart(store, ns, "y")); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()) << "the adopted wedge applied its drop"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); +} +#endif + /// A WELL-FORMED ref-log transaction of this namespace at this id, which simply is not a seal, must be /// adjudicated `Foreign` on CONTENT — not because it failed to decode. The sibling test above reaches /// the same verdict through an undecodable body, so without this one the classifier could be deciding @@ -1381,7 +1541,7 @@ TEST(CASRefWedgeEveryAttempt, WellFormedNonSealOccupantIsStillForeign) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/append_site_wellformed_foreign"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); const RefTxnId next{store->liveWriterEpoch(), 3}; @@ -1414,7 +1574,7 @@ TEST(CASRefWedgeEveryAttempt, ALiveEpochSealIsNeverStampedAsItsOwnPrevEpochSeal) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/live_epoch_seal"}; - admitProperlyBornEntry(*backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch + admitProperlyBornEntry(backend, store->layout(), ns); /// Stage B (Task 4-C): pin to the sentinel before the first real touch publishEmptyPart(store, ns, "x"); const uint64_t epoch = store->liveWriterEpoch(); @@ -1458,7 +1618,7 @@ TEST(CASRefWedgeEveryAttempt, ALiveEpochSealIsNeverStampedAsItsOwnPrevEpochSeal) EXPECT_NE(e.message().find("resumes only under a later epoch"), String::npos) << "the deposition must be surfaced, not just the failure: " << e.message(); } - EXPECT_FALSE(backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})).has_value()) + EXPECT_FALSE(readObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})).has_value()) << "nothing may be written: the lane could not construct a legal transaction, so it sent none"; EXPECT_EQ(backend->putCount(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), RefTxnId{epoch + 1, 1})), puts_before) << "and no request was spent learning what the lane could already prove about itself"; @@ -1474,3 +1634,126 @@ TEST(CASRefWedgeEveryAttempt, ALiveEpochSealIsNeverStampedAsItsOwnPrevEpochSeal) /// re-claim storm. EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before); } + +/// ================================================================================================ +/// The ref lane's four write arms, after the append moved onto an admitted operation. Each of these +/// pins one arm the old outcome enum could not express, and each is reachable only through the whole +/// pool, because what the arm decides is a LANE TRANSITION, not a return value. +/// ================================================================================================ + +/// The store's own proven refusal returns the exact attempt to `Ready`. It is the one non-commit that +/// must NOT wedge: the request never applied, so there is nothing at the key for a wedge to resolve, +/// and the txn id stays underived. Guarded to `USE_AWS_S3` builds, where alone the refusal class is +/// recognised -- without it the same fault is an ambiguity, which is a different arm. +#if USE_AWS_S3 +TEST(CASRefLane, RefusedReturnsTheAttemptToReadyAndDoesNotWedge) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); + const RootNamespace ns{"srv1/ref_lane_refused"}; + admitProperlyBornEntry(backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + + backend->s3_definite_substr = logPrefix(store, ns); + backend->s3_definite_count = 1; + + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + + EXPECT_FALSE(store->refLaneWedgedForTest(ns)) + << "a proven refusal wrote nothing, so there is nothing for a wedge to resolve"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(), definite_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before); + EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "the refused drop applied nothing"; + + /// And the id was never consumed: the next caller re-derives it and lands the same transaction. + EXPECT_NO_THROW(store->dropRef(ns, "x")); + EXPECT_FALSE(store->resolveRef(ns, "x").has_value()); +} +#endif + +/// A ref-log create that COMMITS and only then loses its admission is reported unresolved, and the +/// lane wedges over an object that is in fact durable. That is the conservative half of the contract: +/// the call may not claim a commit it can no longer stand behind, and the wedge is what makes the next +/// flush settle the key rather than write around it. +TEST(CASRefLane, PostCommitFenceLossWedges) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); + const RootNamespace ns{"srv1/ref_lane_post_commit_fence_loss"}; + admitProperlyBornEntry(backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + + /// The fence is lost INSIDE the write window, so the object lands and the call may not claim it. + backend->armBlock(logPrefix(store, ns)); + std::exception_ptr caller_error; + std::thread writer([&] + { + try { store->dropRef(ns, "x"); } + catch (...) { caller_error = std::current_exception(); } + }); + backend->awaitBlockEntered(); + store->tripMountLost(); + backend->releaseBlock(); + writer.join(); + + ASSERT_TRUE(caller_error != nullptr) << "no acknowledgement: the caller must not be told this succeeded"; + EXPECT_TRUE(store->refLaneWedgedForTest(ns)) + << "the object is durable, so the lane must not be returned to Ready"; + EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1); + EXPECT_TRUE(readObj(*backend, store->wedgedKeyForTest(ns)).has_value()) + << "the write landed -- what was refused is the CLAIM, not the object"; +} + +/// The liveness the ledger hands its operations carries the runtime's own facts and NOT a generation +/// term -- the generation is the operation's, presented once at admission. This pins the half that is +/// easy to lose in that split: a runtime retired while the fence generation never MOVES must still end +/// the operation, and it must end it as an unresolved write rather than an installed commit. +TEST(CASRefLane, LivenessPredicateWithoutGenerationTermStillRefusesARetiredRuntime) +{ + auto backend = std::make_shared(); + auto store = openPool(backend, wedgeTestBudget()); + auto clock = VirtualRetryClock::installOn(store); + const RootNamespace ns{"srv1/ref_lane_retired_runtime"}; + admitProperlyBornEntry(backend, store->layout(), ns); + publishEmptyPart(store, ns, "x"); + ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); + + const NamespaceLifeId life = DB::Cas::tests::fixture::fixtureLife(ns); + const uint64_t generation_before = store->fenceGeneration(); + const uint64_t writes_before = backend->writeTotal(); + + backend->armBlock(logPrefix(store, ns)); + std::exception_ptr caller_error; + std::thread writer([&] + { + try { store->dropRef(ns, "x"); } + catch (...) { caller_error = std::current_exception(); } + }); + /// The append's own create is parked, which is what makes the retirement below land INSIDE the + /// write window rather than before the lane ever armed an attempt. + backend->awaitBlockEntered(); + /// The mount fence is untouched throughout, so the runtime term is the only thing that can end this + /// operation. Retirement also detaches the cache slot, so the lane state is no longer observable -- + /// what this pins is that the CALLER is refused, which is the property the term exists for. + store->invalidateRemovedCatalogLife(life); + backend->releaseBlock(); + writer.join(); + + EXPECT_EQ(store->fenceGeneration(), generation_before) + << "the fence never moved -- a generation term could not have produced this refusal"; + EXPECT_GT(backend->writeTotal(), writes_before) << "the parked create did reach the store"; + ASSERT_TRUE(caller_error != nullptr) << "a retired runtime must not be told its append succeeded"; + /// The retry-safe class, not a hard failure: a retirement racing a write is an ordinary fact about + /// the world, and the caller retries against a fresh observation. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { std::rethrow_exception(caller_error); }); +} diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp index 48414871269b..159b12fb99e9 100644 --- a/src/Disks/tests/gtest_cas_ref_writer.cpp +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -53,7 +53,7 @@ extern const Event CASRefSnapshotPutBytes; extern const Event CASRefSnapshotTailLogs; extern const Event CASRefSnapshotPublishDispatched; extern const Event CASRefSnapshotPublishBackoff; -extern const Event CASConditionalWriteFenceLostPostWrite; +extern const Event CASRequestFenceLostPostWrite; extern const Event CASRefRecoveryEpochSealed; extern const Event CASRefRecoveryRetries; } @@ -72,75 +72,105 @@ using DB::Cas::tests::writeSealAt; namespace { -/// The operation deadline every SINGLE-ATTEMPT fixture in this file uses, and the reason it is -/// deliberately NOT `attempt_timeout_ms`. -/// -/// Those fixtures exist to make one injected ambiguous response conclusive, and `max_attempts = 1` -/// alone achieves that: with retries allowed the controller would resolve-before-reissue and report a -/// definite outcome instead. The deadline contributes nothing to that -- but setting it EQUAL to the -/// attempt timeout collapses the controller's pre-send gate into a race. The deadline is captured as -/// `now + operation_deadline_ms` and the gate asks `now + attempt_timeout_ms > deadline` -/// (`CasRequestControl.cpp`), so equal values reduce it to `now_2 > now_1`: ONE elapsed millisecond -/// between the two clock reads refuses the operation with NOTHING SENT, the injected fault is never -/// reached, and the product correctly does not wedge -- flipping every wedge expectation downstream. -/// -/// That is not hypothetical. It took down -/// `CASRefWriterStalePrecommitSweep.BoundedBatchesAndInterruptionResumeAcrossMounts` on 5 of 6 sanitizer -/// CI runs (fixed in `8f9e63c7a19`), `CASRefInstallSafety.UncertainPrecommitKeepsItsCleanupOwnerAndItsBody` -/// under parallel-build load, and `CASRefWriterAppendLane.WedgedLaneBlocksSameTableWhileOtherTableProceeds` -/// in a full-binary ASan run -- the last one with the mechanism named verbatim in the thrown message -/// ("refused BEFORE any request was sent ... the operation deadline rejected before the first request"). -/// -/// A wide deadline keeps the request always actually sent, so what the test observes is the fault it -/// injected rather than the machine it ran on. A fixture that genuinely wants the pre-send REFUSAL -/// must drive it deterministically with a frozen clock (see `gtest_cas_ref_install_safety.cpp`'s -/// `openPoolFenceControlled`), never by racing the wall clock. -constexpr uint64_t kSingleAttemptDeadlineMs = 5000; +/// The budget every wedge fixture here uses. It bounds the mount lease's own admission arithmetic and +/// nothing else: a write's ATTEMPT COUNT is the `Retry` policy's, and the ref lane's is `standard`, so +/// no budget field can make an injected fault conclusive. What does is `driveToTheWedge` below -- +/// the fault stays armed for the whole call while the injected clock carries it to its own deadline. +CasRequestBudget wedgeTestBudget() +{ + CasRequestBudget budget; + budget.attempt_timeout_ms = 100; + budget.lease_safety_margin_ms = 100; + return budget; +} -/// A `CasEvent` sink safe to hand to `Pool::setEventSink`: the emit runs on whatever thread the pool's -/// background syncer happens to be on, and the test reads the accumulated events afterward from the -/// main test thread with no other ordering between the two -- a bare `std::vector` there is a real data -/// race (the class this file's four `setEventSink` call sites all had, hidden because a debug/ASan build -/// doesn't reliably catch an unsynchronized push_back/iterator-read pair on a small vector). `add` takes -/// the lock only around the push; `snapshot` copies out under the lock and returns, so a caller iterating -/// the result never holds the mutex across anything that could call back into the pool (which an -/// event-sink callback legitimately can, on other seams in this file). -class SynchronizedEventLog +/// The engine reissues an unresolved write until its OWN retry window closes, and that window is +/// measured on a clock the engine reads. Both seams here share one counter -- the sleep the engine +/// performs is what advances the clock -- so a fault that stays armed ends the call at its deadline +/// with no real time passing. Installed on the whole pool, because the ref-lane write, its settling +/// read and the recovery retry loop all pace through the same seam. The pool owns the closures and the +/// closures own the clock, so it outlives everything that can still read it. +class VirtualRetryClock { public: - void add(const CasEvent & e) + static std::shared_ptr installOn(const PoolPtr & store) + { + auto clock = std::make_shared(); + store->setCasRequestNowFnForTest([clock] { return clock->nowMs(); }); + store->setCasRetrySleepForTest([clock](uint64_t ms) { clock->advance(ms); }); + return clock; + } + + uint64_t nowMs() const + { + std::lock_guard lock(mutex); + return now_ms; + } + size_t pauseCount() const + { + std::lock_guard lock(mutex); + return pauses; + } + uint64_t longestPause() const { std::lock_guard lock(mutex); - events.push_back(e); + return longest_pause; } - std::vector snapshot() const + + void advance(uint64_t ms) { std::lock_guard lock(mutex); - return events; + /// Plus one millisecond, because full jitter can draw a ZERO pause: a clock that does not move + /// would leave the loop reissuing for ever against a fault that never clears. + now_ms += ms + 1; + ++pauses; + longest_pause = std::max(longest_pause, ms); } + private: mutable std::mutex mutex; - std::vector events; + uint64_t now_ms = 0; + size_t pauses = 0; + uint64_t longest_pause = 0; }; -PoolPtr openPool(const BackendPtr & backend, CasRequestBudget budget = {}) +/// A `CasEvent` sink safe to hand to `Pool::setEventSink`: the emit runs on whatever thread the pool's +/// background syncer happens to be on, and the test reads the accumulated events afterward from the +/// main test thread with no other ordering between the two -- a bare `std::vector` there is a real data +/// race (the class this file's four `setEventSink` call sites all had, hidden because a debug/ASan build +/// doesn't reliably catch an unsynchronized push_back/iterator-read pair on a small vector), and even a +/// mutex-guarded one declared as a plain local is not enough on its own: a background publish can hold +/// an extra `shared_from_this()` past this frame's return, so the log itself must be heap-owned too. +/// `DB::Cas::tests::SharedEventLog` is exactly this shape (push under lock, snapshot copies out under +/// lock so a caller iterating the result never holds the mutex across a callback into the pool). +using DB::Cas::tests::SharedEventLog; + +template +PoolPtr openPool(const std::shared_ptr & backend, CasRequestBudget budget = {}) { /// Recovery tests seed ref-log/snapshot residue before opening; a pool with such residue always has a /// `_pool_meta` in production, so establish it first (Task 7's zero-write bootstrap check refuses to /// mint a fresh identity over residual data — see `seedPoolMetaForRestart`). Idempotent, and a no-op /// for the fresh-open tests that seed nothing (the subsequent open validates the just-created meta). DB::Cas::tests::seedPoolMetaForRestart(*backend); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); return Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget}); } /// Task 11: like `openPool`, but the caller supplies (and owns) the rest of the config -- snapshot /// thresholds, grace age, a fake `boot_ms_fn`, etc. `pool_prefix`/`server_root_id` are pinned so every /// test in this file addresses the same pool shape. -PoolPtr openPoolWithConfig(const BackendPtr & backend, PoolConfig config) +template +PoolPtr openPoolWithConfig(const std::shared_ptr & backend, PoolConfig config) { config.pool_prefix = "p"; config.server_root_id = "test"; DB::Cas::tests::seedPoolMetaForRestart(*backend); /// see `openPool` above + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(config.cas_request_budget.attempt_timeout_ms); return Pool::open(backend, std::move(config)); } @@ -154,7 +184,7 @@ PoolPtr openPoolWithConfig(const BackendPtr & backend, PoolConfig config) /// production birth mints a random incarnation and those computed keys land nowhere real. PartWriteTxnPtr startBuildFor(const PoolPtr & s, const RootNamespace & ns, const String & ref) { - DB::Cas::tests::casAdmitRecoverableEntry(s->backend(), s->layout(), ns, s->liveWriterEpoch()); + DB::Cas::tests::casAdmitRecoverableEntry(*s->poolBackendPtr(), s->layout(), ns, s->liveWriterEpoch()); PartWriteInfo info; info.intended_namespace = ns; info.intended_ref = ns.string() + "/" + ref; @@ -181,9 +211,84 @@ void publishWithProductionBirth(const PoolPtr & store, const RootNamespace & ns, build->promote(ns, ref, build->buildId(), id); } -CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const RootNamespace & ns) +/// Fixture observations of durable state run on an OPEN fence: they are not writes a mount admitted, +/// and each owns the `CasRequests` its operation borrows, so none of these hands one back. +std::optional readCkptForTest(const BackendPtr & backend, const Layout & layout, + const NamespaceLifeId & life) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return readCkpt(op, layout, life); +} + +CasRefCatalog::Snapshot readCatalogForTest(const BackendPtr & backend, const Layout & layout) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return CasRefCatalog::read(op, layout); +} + +/// A fixture's own conditional replace, for the races these tests stage by hand. +bool replaceForTest(const BackendPtr & backend, const String & key, const String & bytes, + const Etag & expected) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return std::holds_alternative(op.replace(key, bytes, expected, Retry::standard())); +} + +void casAdmitEntryForTest(const BackendPtr & backend, const Layout & layout, uint64_t gc_shards, + const CatalogEntry & entry) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + CasRefCatalog::casAdmitEntry(op, layout, gc_shards, entry); +} + +std::optional lifeIfCatalogedForTest(const BackendPtr & backend, const Layout & layout, + const RootNamespace & ns) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return CasRefCatalog::lifeIfCataloged(op, layout, ns); +} + +uint64_t allocateWriterEpochForTest(const BackendPtr & backend, const Layout & layout, const String & server_root_id) { - const RefCatalog catalog = CasRefCatalog::read(backend, layout).catalog; + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return allocateWriterEpoch(op, layout, server_root_id, EpochMintPolicy::NormalMount, 0, + [] { return RefCatalog{}; }); +} + +/// A fixture enumeration on an open fence: the primitive `list` override in this file's test backend +/// hides the legacy name, and a test walking a prefix should ride the same engine production does. +ListPage listForTest(const BackendPtr & backend, const String & prefix, const String & cursor, size_t limit) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return op.list(prefix, cursor, limit, Retry::standard()); +} + +/// The same open-fence idiom as `listForTest` above, for the raw observations this file's fixtures make +/// directly against a `RefWriterTestBackend` outside any Pool operation. +std::optional readOf(const BackendPtr & backend, const String & key) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return op.read(key, Retry::standard()); +} + +bool createRaw(const BackendPtr & backend, const String & key, const String & bytes) +{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + return std::holds_alternative(op.create(key, bytes, Retry::standard())); +} + +CatalogEntry catalogEntryOrThrow(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) +{ + const RefCatalog catalog = readCatalogForTest(backend, layout).catalog; const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; @@ -194,16 +299,18 @@ CatalogEntry catalogEntryOrThrow(Backend & backend, const Layout & layout, const } CatalogEntry replaceCatalogLifeForRuntimeRace( - Backend & backend, const Layout & layout, const CatalogEntry & predecessor, UInt128 successor_incarnation) + const BackendPtr & backend, const Layout & layout, const CatalogEntry & predecessor, + UInt128 successor_incarnation) { - const CasRefCatalog::Snapshot before_delete = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot before_delete = readCatalogForTest(backend, layout); RefCatalog without_predecessor = before_delete.catalog; std::erase_if(without_predecessor.entries, [&](const CatalogEntry & entry) { return entry.ns == predecessor.ns && entry.incarnation == predecessor.incarnation; }); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(without_predecessor), before_delete.token).outcome - != CasOutcome::Committed) + if (!before_delete.etag + || !replaceForTest(backend, layout.refCatalogKey(), encodeRefCatalog(without_predecessor), + *before_delete.etag)) throw std::runtime_error("test failed to retire exact predecessor catalog life"); CatalogEntry successor{ @@ -211,11 +318,11 @@ CatalogEntry replaceCatalogLifeForRuntimeRace( .state = NsState::Live, .incarnation = successor_incarnation, .creator = std::nullopt}; - const CasRefCatalog::Snapshot after_delete = CasRefCatalog::read(backend, layout); + const CasRefCatalog::Snapshot after_delete = readCatalogForTest(backend, layout); RefCatalog reborn = after_delete.catalog; reborn.entries.push_back(successor); - if (backend.casPut(layout.refCatalogKey(), encodeRefCatalog(reborn), after_delete.token).outcome - != CasOutcome::Committed) + if (!after_delete.etag + || !replaceForTest(backend, layout.refCatalogKey(), encodeRefCatalog(reborn), *after_delete.etag)) throw std::runtime_error("test failed to publish successor catalog life"); return successor; } @@ -226,11 +333,12 @@ std::optional listGreatestLogIdForTest( std::optional listGreatestLogIdForLifeForTest( Backend & backend, const Layout & layout, const NamespaceLifeId & life) { + DB::Cas::tests::OperationForTest op(backend); std::optional greatest; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(life), cursor, 1000); + const ListPage page = (*op).list(layout.namespaceStreamPrefix(life), cursor, 1000, Retry::standard()); for (const ListedKey & listed : page.keys) { const auto parsed = layout.parseRefObjectKey(listed.key); @@ -252,7 +360,7 @@ struct CompletedRemovingFixture }; CompletedRemovingFixture prepareResidentRemovalForDrain( - const PoolPtr & store, Backend & backend, const RootNamespace & ns, Gc & gc) + const PoolPtr & store, const BackendPtr & backend, const RootNamespace & ns, Gc & gc) { publishWithProductionBirth(store, ns, "predecessor"); const CatalogEntry predecessor = catalogEntryOrThrow(backend, store->layout(), ns); @@ -268,9 +376,10 @@ CompletedRemovingFixture prepareResidentRemovalForDrain( if (runRegularRoundReclaiming(gc).deferred) throw std::runtime_error("fixture terminal fold unexpectedly deferred"); - const GcState state = decodeGcState(backend.get(store->layout().gcStateKey())->bytes); + DB::Cas::tests::OperationForTest drain_op(backend); + const GcState state = decodeGcState((*drain_op).read(store->layout().gcStateKey(), Retry::standard())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend.get(store->layout().foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + (*drain_op).read(store->layout().foldSealKey(state.snap_generation, state.snap_attempt), Retry::standard())->bytes); const auto row = seal.ref_lives.find(predecessor.incarnation); if (row == seal.ref_lives.end() || !row->second.cleanup_evidence) throw std::runtime_error("fixture terminal fold produced no cleanup evidence"); @@ -290,11 +399,12 @@ ManifestRef manifestRef(uint64_t epoch, uint64_t seq, uint32_t ordinal) RefTableState independentFullReplayForTest(Backend & backend, const Layout & layout, const RootNamespace & ns, std::optional up_to = std::nullopt) { + DB::Cas::tests::OperationForTest op(backend); std::vector ids; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*op).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -311,7 +421,7 @@ RefTableState independentFullReplayForTest(Backend & backend, const Layout & lay RefTableState state; for (const RefTxnId & id : ids) { - const auto got = backend.get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id)); + const auto got = (*op).read(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), id), Retry::standard()); applyRefLogTxn(state, decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), id)); } return state; @@ -321,11 +431,12 @@ RefTableState independentFullReplayForTest(Backend & backend, const Layout & lay /// of the Pool's own cached bookkeeping). std::optional listGreatestSnapshotIdForTest(Backend & backend, const Layout & layout, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest op(backend); std::optional greatest; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*op).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -343,7 +454,7 @@ std::optional listGreatestSnapshotIdForTest(Backend & backend, const L /// A backend that can (a) force one `get()` on a chosen exact key to return absent exactly once /// (simulating an object vanishing after recovery sampled its exact checkpoint, with an optional side effect /// fired at that exact moment -- e.g. publishing a covering newer snapshot, mirroring a concurrent GC -/// cleanup+republish race), and (b) force `putIfAbsent` on keys matching a chosen substring to throw an +/// cleanup+republish race), and (b) force a create-shaped write on keys matching a chosen substring to throw an /// ambiguous (Unresolved-classified) exception a bounded number of times, optionally still capturing /// the (key, bytes) so a test can later "deliver" it -- simulating a request whose RESPONSE was lost /// even though the write eventually landed server-side. @@ -355,12 +466,6 @@ class RefWriterTestBackend : public CountingBackend DB::Cas::tests::seedPoolMetaForRestart(*this); } - using CountingBackend::get; - using CountingBackend::getStream; - using CountingBackend::putIfAbsent; - using CountingBackend::putOverwrite; - using CountingBackend::casPut; - void clearRequestJournal() { std::lock_guard lock(request_journal_mutex); @@ -400,6 +505,13 @@ class RefWriterTestBackend : public CountingBackend String fault_key_substr; int fault_count = 0; + /// LATCHED: a COUNT can no longer make an injected fault conclusive, because the write engine + /// settles each ambiguity by an exact read and then REISSUES -- a fault that runs out mid-call is + /// answered by the next attempt instead of by the call's own deadline, which is the difference + /// between a wedge and a commit. While this is set the count is topped up before every matching + /// create, so the fault outlasts the whole call. + bool fault_latched = false; + bool ckpt_conflict_latched = false; /// Let the first `fault_skip` matching PUTs through untouched before `fault_count` starts faulting. /// Needed now that recovery's in-band epoch seal (INV-2) shares the `_log/` prefix with every other /// write under a namespace: a test that wants to fault something LATER in the same prefix (e.g. the @@ -408,7 +520,7 @@ class RefWriterTestBackend : public CountingBackend int fault_skip = 0; std::optional> pending_delayed_write; - /// (I1) On a matching `putIfAbsent`, a FOREIGN writer lands a DIFFERENT object at the exact key and + /// (I1) On a matching create-shaped write, a FOREIGN writer lands a DIFFERENT object at the exact key and /// then this attempt's response is lost -- so the controller's resolve-before-reissue GET observes /// different bytes and must raise CORRUPTED_DATA (a proven conflict, never a retry signal). /// By default the foreign object is the attempt's own bytes plus a trailing marker -- UNDECODABLE @@ -428,17 +540,18 @@ class RefWriterTestBackend : public CountingBackend /// times. Recovery must not consume this injection; callers that intentionally enumerate still do. int list_fault_count = 0; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + DB::Cas::Backend::RawListPage list(const String & prefix, const String & cursor, size_t limit, + DB::Cas::TransportAccess & access) override { if (list_fault_count > 0) { --list_fault_count; throw DB::Exception(DB::ErrorCodes::S3_ERROR, "RefWriterTestBackend: simulated transient LIST failure"); } - return CountingBackend::list(prefix, cursor, limit); + return CountingBackend::list(prefix, cursor, limit, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, DB::Cas::TransportAccess & access) override { recordRequestJournalEvent("GET " + key); if (key == ckpt_get_hook_key && ckpt_get_hook) @@ -472,18 +585,52 @@ class RefWriterTestBackend : public CountingBackend } return std::nullopt; } - return CountingBackend::get(key, range); + return CountingBackend::read(key, access); + } + + /// Every write fault hangs off the ONE keyed primitive. Which of them applies is decided by whether + /// the write carries a precondition, which is exactly what used to separate a conditional replace + /// from a create. + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + topUpLatchedFaults(key, expected_value); + if (expected_value) + return conditionalReplaceForTest(key, bytes, *expected_value, access); + return createForTest(key, bytes, access); + } + + /// Both latches are re-armed HERE rather than inside each seam, so a latched fault reads exactly + /// like the counted one it replaces. + void topUpLatchedFaults(const String & key, const std::optional & expected_value) + { + if (ckpt_conflict_latched && key == ckpt_conflict_key) + ckpt_conflict_count = 1; + if (fault_latched && !expected_value && fault_skip == 0 && !fault_key_substr.empty() + && key.find(fault_key_substr) != String::npos) + fault_count = 1; + } + + void disarmFaults() + { + fault_latched = false; + ckpt_conflict_latched = false; + fault_count = 0; + fault_skip = 0; + ckpt_conflict_count = 0; + corrupt_count = 0; } - CasResult casPut( - const String & key, const String & bytes, const std::optional & expected, - const ObjectMeta & meta) override + std::expected conditionalReplaceForTest( + const String & key, const String & bytes, const String & expected_value, + DB::Cas::TransportAccess & access) { recordRequestJournalEvent("CAS " + key); if (key == ckpt_conflict_key && ckpt_conflict_count > 0) { --ckpt_conflict_count; - return {CasOutcome::Conflict, {}}; + return std::unexpected(DB::Cas::Backend::RawConflict{}); } if (key == catalog_fault_key && catalog_cas_fault != CatalogCasFault::None) { @@ -491,31 +638,32 @@ class RefWriterTestBackend : public CountingBackend catalog_cas_fault_fired = true; if (fault == CatalogCasFault::CommitThenThrow) { - const CasResult result = CountingBackend::casPut(key, bytes, expected, meta); - if (result.outcome != CasOutcome::Committed) + auto result = CountingBackend::write(key, bytes, expected_value, access); + if (!result.has_value()) return result; throw Poco::TimeoutException( "RefWriterTestBackend: catalog CAS committed but its response was lost"); } - CasResult replacement = CountingBackend::casPut( - key, catalog_replacement_bytes, expected, meta); - if (replacement.outcome != CasOutcome::Committed) + auto replacement = CountingBackend::write(key, catalog_replacement_bytes, expected_value, access); + if (!replacement.has_value()) return replacement; - return {CasOutcome::Conflict, {}}; + return std::unexpected(DB::Cas::Backend::RawConflict{}); } - return CountingBackend::casPut(key, bytes, expected, meta); + return CountingBackend::write(key, bytes, expected_value, access); } - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected createForTest( + const String & key, const String & bytes, DB::Cas::TransportAccess & access) { recordRequestJournalEvent("PUT " + key); if (corrupt_count > 0 && !corrupt_key_substr.empty() && key.find(corrupt_key_substr) != String::npos) { --corrupt_count; /// A foreign writer lands a DIFFERENT object at this exact key; then our own response is lost. - CountingBackend::putIfAbsent( - key, corrupt_foreign_bytes.empty() ? bytes + String("\x01_FOREIGN_DIFFERENT") : corrupt_foreign_bytes); + (void)CountingBackend::write( + key, corrupt_foreign_bytes.empty() ? bytes + String("\x01_FOREIGN_DIFFERENT") : corrupt_foreign_bytes, + std::nullopt, access); throw Poco::TimeoutException("RefWriterTestBackend: a foreign different object landed; response lost"); } if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) @@ -561,16 +709,16 @@ class RefWriterTestBackend : public CountingBackend block_entered = true; block_cv.notify_all(); block_cv.wait(lk, [&] { return !block_armed; }); - /// fix-round F3-1a (CRITICAL, unlock-throw race harness): on release, behave like - /// `corrupt_key_substr` above instead of proceeding normally -- a foreign writer landed - /// DIFFERENT bytes at this exact key while we were parked, so our own attempt is a - /// PROVEN conflict once `putIfAbsentControlled`'s resolve-before-reissue GETs it. Lets a - /// test make the recovery seal's PUT throw CORRUPTED_DATA from INSIDE the unlocked - /// window, deterministically, instead of merely returning a non-Committed outcome. + /// On release, behave like `corrupt_key_substr` above instead of proceeding normally -- + /// a foreign writer landed DIFFERENT bytes at this exact key while we were parked, so + /// our own attempt is a PROVEN conflict once the write engine's own settling read + /// observes it. Lets a test make the recovery seal's write throw CORRUPTED_DATA from + /// INSIDE the unlocked window, deterministically, instead of merely returning a + /// non-Committed outcome. if (block_throw_corrupted_on_release) { lk.unlock(); - CountingBackend::putIfAbsent(key, bytes + String("\x01_FOREIGN_DIFFERENT")); + (void)CountingBackend::write(key, bytes + String("\x01_FOREIGN_DIFFERENT"), std::nullopt, access); { std::lock_guard g(block_mutex); block_call_completed = true; @@ -581,7 +729,7 @@ class RefWriterTestBackend : public CountingBackend } } } - const PutResult r = CountingBackend::putIfAbsent(key, bytes, meta); + auto r = CountingBackend::write(key, bytes, std::nullopt, access); { std::lock_guard g(block_mutex); block_call_completed = true; @@ -589,8 +737,8 @@ class RefWriterTestBackend : public CountingBackend block_cv.notify_all(); return r; } - /// See `putIfAbsent`'s `block_this` branch. Set before spawning any thread that could race - /// `putIfAbsent`, like `corrupt_key_substr`/`fault_key_substr` above -- not itself lock-protected. + /// See `createForTest`'s `block_this` branch. Set before spawning any thread that could race + /// a create-shaped write, like `corrupt_key_substr`/`fault_key_substr` above -- not itself lock-protected. bool block_throw_corrupted_on_release = false; /// "Deliver" the earlier ambiguous write: the request DID eventually land server-side, the caller @@ -599,12 +747,13 @@ class RefWriterTestBackend : public CountingBackend { if (pending_delayed_write) { - CountingBackend::putIfAbsent(pending_delayed_write->first, pending_delayed_write->second); + DB::Cas::tests::OperationForTest op(*this); + (void)(*op).create(pending_delayed_write->first, pending_delayed_write->second, DB::Cas::Retry::once()); pending_delayed_write.reset(); } } - /// Task 11: blocks EVERY `putIfAbsent()` whose key contains `armed_block_substr` until + /// Task 11: blocks EVERY create-shaped write whose key contains `armed_block_substr` until /// `releaseBlock()` is called, notifying `awaitBlockEntered()` the first time one is reached. Used /// to prove snapshot publication never holds up an unrelated concurrent append. void armPutBlock(const String & substr) @@ -618,7 +767,7 @@ class RefWriterTestBackend : public CountingBackend blocked_key.clear(); } - /// Task 11 (monotonic-adoption harness): block ONLY the FIRST `putIfAbsent` whose key contains + /// Task 11 (monotonic-adoption harness): block ONLY the FIRST create-shaped write whose key contains /// `substr`, capturing that exact key; every LATER put -- including a DIFFERENT `_snap/` key -- /// proceeds unblocked. Lets a test pin one in-flight publish's PUT mid-flight while a second, /// higher-id publish runs to completion, deterministically forcing the out-of-order overlap. @@ -645,8 +794,8 @@ class RefWriterTestBackend : public CountingBackend } block_cv.notify_all(); } - /// Blocks until the PREVIOUSLY-blocked `putIfAbsent` call has actually RETURNED (not merely been - /// unblocked) -- i.e. its underlying `CountingBackend::putIfAbsent` has completed. Deterministic, + /// Blocks until the PREVIOUSLY-blocked create-shaped write call has actually RETURNED (not merely been + /// unblocked) -- i.e. its underlying `CountingBackend::write` has completed. Deterministic, /// sleep-free way to observe a detached background caller's own work finishing when the TEST no /// longer holds anything (e.g. a Pool handle) that call would otherwise let it wait on. void awaitBlockedCallCompleted() @@ -655,7 +804,7 @@ class RefWriterTestBackend : public CountingBackend block_cv.wait(lk, [&] { return block_call_completed; }); } - /// (I1 regression harness) Arms independent per-key blocking for every `putIfAbsent` matching + /// (I1 regression harness) Arms independent per-key blocking for every create-shaped write matching /// `substr`: unlike `armPutBlock`/`armPutBlockFirstMatchOnly` (one shared release gate), each /// blocked key parks on ITS OWN release (`releaseKey`), so two distinct `_snap/` PUTs can be /// parked concurrently -- both past their capture point, neither yet adopted -- and released in a @@ -708,6 +857,31 @@ class RefWriterTestBackend : public CountingBackend std::set independent_released_keys; }; +/// Wedge the lane the way the engine reaches that state: EVERY attempt of the ref-log create is +/// unresolved, the settling read proves the key still absent, and the call gives up at its own retry +/// window having sent something -- which is the `sent_any` half of the wedge rule. A one-shot fault +/// cannot produce it: the reissue would settle the key and commit. So the fault is held armed for the +/// whole call and fully disarmed afterwards, because what every caller does next is a flush that must +/// reach the store normally. +/// +/// The pacing assertions are what make a fixture whose sleep seam is not wired FAIL rather than sleep +/// the whole window out for real. +void driveToTheWedge(VirtualRetryClock & clock, RefWriterTestBackend & backend, + const std::function & drive) +{ + const size_t pauses_before = clock.pauseCount(); + const uint64_t clock_before = clock.nowMs(); + backend.fault_latched = true; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, drive); + backend.disarmFaults(); + EXPECT_GT(clock.pauseCount(), pauses_before + 1) + << "the reissues must pace through the injected sleep, never a real one"; + EXPECT_LE(clock.longestPause(), 5000u) << "each pause is the engine's own capped full jitter"; + EXPECT_GE(clock.nowMs() - clock_before, 60000u) + << "the give-up must be the call's own retry window, not a pre-attempt refusal"; +} + + } /// =================================================================================== @@ -731,7 +905,7 @@ TEST(CASRefWriterNonMinting, ListRefsOnAbsentNamespaceDoesNotMutateCatalog) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/list_absent_non_minting"}; - const auto catalog_before = backend->get(layout.refCatalogKey()); + const auto catalog_before = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_before); backend->resetCounts(); @@ -739,12 +913,12 @@ TEST(CASRefWriterNonMinting, ListRefsOnAbsentNamespaceDoesNotMutateCatalog) EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->writeCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); - const auto catalog_after = backend->get(layout.refCatalogKey()); + const auto catalog_after = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_after); EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); - EXPECT_EQ(catalog_after->token, catalog_before->token); + EXPECT_EQ(catalog_after->etag, catalog_before->etag); } TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsCatalogLifeReplacedWithoutLocalInvalidation) @@ -754,7 +928,7 @@ TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsCatalogLifeReplacedWithoutLocal const Layout & layout = store->layout(); const RootNamespace ns{"srv1/cold-read-catalog-aba"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); std::mutex mutex; @@ -787,13 +961,13 @@ TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsCatalogLifeReplacedWithoutLocal } const CatalogEntry successor - = replaceCatalogLifeForRuntimeRace(*backend, layout, predecessor, UInt128{0xabc002}); + = replaceCatalogLifeForRuntimeRace(backend, layout, predecessor, UInt128{0xabc002}); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ .life_epoch = store->liveWriterEpoch(), .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); { std::lock_guard lock(mutex); resume = true; @@ -818,22 +992,23 @@ TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsReplacementByExternalPoolActor) const Layout & layout = store->layout(); const RootNamespace ns{"srv1/external-catalog-runtime-publication"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); CatalogEntry successor; store->setReadableCatalogAfterObservationHookForTest([&] { successor = replaceCatalogLifeForRuntimeRace( - external_store->backend(), external_store->layout(), predecessor, UInt128{0xabc003}); + external_store->poolBackendPtr(), external_store->layout(), predecessor, UInt128{0xabc003}); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(successor.ns, successor.incarnation); - if (external_store->backend().putIfAbsent( + OperationForTest successor_op(*external_store->poolBackendPtr()); + if (!std::holds_alternative((*successor_op).create( external_store->layout().refCkptKey(successor_life), encodeRefCkpt(RefCkpt{ .life_epoch = external_store->liveWriterEpoch(), .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome != PutOutcome::Done) + .last_epoch_seal = std::nullopt}), Retry::standard()))) throw std::runtime_error("test failed to publish external successor checkpoint"); }); @@ -885,7 +1060,7 @@ TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsAliasingIncarnationAdmittedBetw const RootNamespace ns{"srv1/aliasing-incarnation-target"}; const RootNamespace alias{"srv1/aliasing-incarnation-thief"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, layout, ns, store->liveWriterEpoch()); - const CatalogEntry target_row = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry target_row = catalogEntryOrThrow(backend, layout, ns); ASSERT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); store->setReadableCatalogAfterObservationHookForTest([&] @@ -896,7 +1071,7 @@ TEST(CASRefWriterRuntimeIdentity, ColdReadRejectsAliasingIncarnationAdmittedBetw thief.incarnation = target_row.incarnation; thief.creator = CreatorFence{ .server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, thief); + casAdmitEntryForTest(backend, layout, 1, thief); }); EXPECT_THROW((void)store->listRefs(ns), DB::Exception); @@ -929,7 +1104,7 @@ TEST(CASRefWriterNonMinting, ResolveRefOnAbsentNamespaceDoesNotMutateCatalog) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/resolve_absent_non_minting"}; - const auto catalog_before = backend->get(layout.refCatalogKey()); + const auto catalog_before = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_before); backend->resetCounts(); @@ -937,12 +1112,12 @@ TEST(CASRefWriterNonMinting, ResolveRefOnAbsentNamespaceDoesNotMutateCatalog) EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->writeCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); - const auto catalog_after = backend->get(layout.refCatalogKey()); + const auto catalog_after = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_after); EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); - EXPECT_EQ(catalog_after->token, catalog_before->token); + EXPECT_EQ(catalog_after->etag, catalog_before->etag); } TEST(CASRefWriterNonMinting, DropNamespaceOnAbsentNamespaceDoesNotMutateCatalog) @@ -951,7 +1126,7 @@ TEST(CASRefWriterNonMinting, DropNamespaceOnAbsentNamespaceDoesNotMutateCatalog) auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/drop_absent_non_minting"}; - const auto catalog_before = backend->get(layout.refCatalogKey()); + const auto catalog_before = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_before); backend->resetCounts(); @@ -959,12 +1134,12 @@ TEST(CASRefWriterNonMinting, DropNamespaceOnAbsentNamespaceDoesNotMutateCatalog) EXPECT_EQ(backend->putCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), 0u); - EXPECT_EQ(backend->casPutCount(layout.refCatalogKey()), 0u); + EXPECT_EQ(backend->writeCount(layout.refCatalogKey()), 0u); EXPECT_EQ(backend->deleteCount(layout.refCatalogKey()), 0u); - const auto catalog_after = backend->get(layout.refCatalogKey()); + const auto catalog_after = readOf(backend, layout.refCatalogKey()); ASSERT_TRUE(catalog_after); EXPECT_EQ(catalog_after->bytes, catalog_before->bytes); - EXPECT_EQ(catalog_after->token, catalog_before->token); + EXPECT_EQ(catalog_after->etag, catalog_before->etag); } /// A table born by a log tail alone (no snapshot yet): `namespace_birth` with nothing else is a legal @@ -976,12 +1151,12 @@ TEST(CASRefWriterRecovery, BirthOnlyLogNoSnapshotRecoversToEmptyLiveTable) const RootNamespace ns{"srv1/birth_only"}; DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 1}, {namespaceBirthOp()}, std::nullopt}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 1}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); auto store = openPool(backend); EXPECT_TRUE(store->listRefs(ns).empty()); @@ -1000,12 +1175,12 @@ TEST(CASRefWriterRecovery, BirthPlusPrecommitPromoteAcrossTwoLogsNoSnapshot) {namespaceBirthOp(), publishCommittedOps("part_1", m1)[0]}, std::nullopt}); DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 2}, {publishCommittedOps("part_1", m1)[1]}, std::nullopt}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 2}, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); auto store = openPool(backend); const auto resolved = store->resolveRef(ns, "part_1"); @@ -1039,7 +1214,7 @@ TEST(CASRefWriterRecovery, TerminalGapBelowCheckpointFrontierIsCorruptionNotSame .last_epoch_seal = RefTxnId{1, 2}, }); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); const String next_log_key = layout.refLogKey(life, RefTxnId{2, 2}); auto store = openPool(backend); const uint64_t installs_before = store->recoveryInstallCountForTest(); @@ -1082,12 +1257,12 @@ TEST(CASRefWriterRecovery, SnapshotPlusTailRecovery) tail_ops.push_back(publishCommittedOps("b", mb)[0]); tail_ops.push_back(publishCommittedOps("b", mb)[1]); DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ns.string(), RefTxnId{1, 6}, tail_ops, std::nullopt}); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 6}, .checkpoint_snapshot_id = RefTxnId{1, 5}, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); backend->resetCounts(); auto store = openPool(backend); @@ -1118,7 +1293,7 @@ TEST(CASRefWriterRecovery, RestartOnVanishConvergesOnNewerSnapshot) /// UNADMITTED namespace mints a fresh RANDOM incarnation rather than adopting the sentinel the raw /// fixture writes at. DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); const RefTxnId snap_x{1, 10}; DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), @@ -1126,11 +1301,11 @@ TEST(CASRefWriterRecovery, RestartOnVanishConvergesOnNewerSnapshot) .ops = publishCommittedOps("a", ma), .prev_epoch_seal = std::nullopt}); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), snap_x, {committedRow("a", ma)})); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 10}, .checkpoint_snapshot_id = snap_x, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); backend->vanish_once_keys.insert(layout.refSnapshotKey(life, snap_x)); bool vanish_fired = false; backend->on_vanish_fire = [&] @@ -1143,13 +1318,15 @@ TEST(CASRefWriterRecovery, RestartOnVanishConvergesOnNewerSnapshot) .ops = publishCommittedOps("b", mb), .prev_epoch_seal = std::nullopt}); writeRefSnapshotRaw(*backend, layout, minimalLiveSnapshot(ns.string(), snap_y, {committedRow("b", mb)})); - const auto before = backend->get(layout.refCkptKey(life)); + const auto before = readOf(backend, layout.refCkptKey(life)); ASSERT_TRUE(before); - ASSERT_EQ(backend->casPut(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.replace(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = RefTxnId{1, 20}, .checkpoint_snapshot_id = snap_y, - .last_epoch_seal = std::nullopt}), before->token).outcome, CasOutcome::Committed); + .last_epoch_seal = std::nullopt}), before->etag, Retry::standard()))); }; backend->resetCounts(); @@ -1181,7 +1358,7 @@ TEST(CASRefWriterRecovery, DifferentBytesAtSelectedSnapshotIsCorruptionNotRestar /// further down is a real production read that would otherwise mint a fresh RANDOM incarnation /// for this unadmitted namespace instead of adopting the sentinel the raw fixture writes at. DB::Cas::tests::fixture::admitLive(*backend, layout, ns); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, layout, ns); + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, layout, ns); const RootNamespace other_ns{"srv1/other"}; DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, RefLogTxn{ .ns = ns.string(), @@ -1192,14 +1369,13 @@ TEST(CASRefWriterRecovery, DifferentBytesAtSelectedSnapshotIsCorruptionNotRestar foreign.ns = other_ns.string(); foreign.snapshot_id = snap_x; const String snapshot_key = layout.refSnapshotKey(life, snap_x); - ASSERT_EQ(backend->putIfAbsent(snapshot_key, - DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(foreign))).outcome, - PutOutcome::Done); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(createRaw(backend, snapshot_key, + DB::Cas::sealObject(DB::Cas::FormatId::RefSnapshot, DB::Cas::encodeRefTableSnapshot(foreign)))); + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = 1, .committed_through = snap_x, .checkpoint_snapshot_id = snap_x, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); auto store = openPool(backend); backend->resetCounts(); @@ -1224,10 +1400,10 @@ TEST(CASRefWriterAppendLane, CommittedChunkPublishesFrontierBeforeInstallAndAck) ASSERT_TRUE(store->resolveRef(ns, "part_1").has_value()); ASSERT_TRUE(store->resolveRef(ns, "part_2").has_value()); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const NamespaceLifeId life = lifeIfCatalogedForTest(backend, store->layout(), ns).value(); const String log_prefix = store->layout().namespaceStreamPrefix(life) + "_log/"; const String ckpt_key = store->layout().refCkptKey(life); - const auto ckpt_before = readCkpt(*backend, store->layout(), life); + const auto ckpt_before = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(ckpt_before); ASSERT_TRUE(ckpt_before->ckpt.committed_through); const RefTxnId expected_frontier{ @@ -1237,7 +1413,7 @@ TEST(CASRefWriterAppendLane, CommittedChunkPublishesFrontierBeforeInstallAndAck) const uint64_t list_before = backend->listTotal(); const uint64_t put_before = backend->putTotal(); const uint64_t ckpt_get_before = backend->getCount(ckpt_key); - const uint64_t ckpt_cas_before = backend->casPutCount(ckpt_key); + const uint64_t ckpt_cas_before = backend->writeCount(ckpt_key); std::mutex mutex; std::condition_variable cv; @@ -1346,7 +1522,7 @@ TEST(CASRefWriterAppendLane, CommittedChunkPublishesFrontierBeforeInstallAndAck) EXPECT_EQ(backend->putTotal(), put_before + 1) << "exactly one body PUT with create-if-absent"; EXPECT_EQ(backend->getCount(ckpt_key), ckpt_get_before + 1) << "one committed chunk pays exactly one checkpoint GET"; - EXPECT_EQ(backend->casPutCount(ckpt_key), ckpt_cas_before + 1) + EXPECT_EQ(backend->writeCount(ckpt_key), ckpt_cas_before + 1) << "one committed chunk pays exactly one checkpoint CAS"; const std::vector journal = backend->requestJournal(); @@ -1357,7 +1533,7 @@ TEST(CASRefWriterAppendLane, CommittedChunkPublishesFrontierBeforeInstallAndAck) EXPECT_EQ(journal[3], "INSTALL"); EXPECT_EQ(journal[4], "FOLLOWER ACK"); - const auto durable_ckpt = readCkpt(*backend, store->layout(), life); + const auto durable_ckpt = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(durable_ckpt); EXPECT_EQ(durable_ckpt->ckpt.committed_through, expected_frontier); } @@ -1368,28 +1544,34 @@ TEST(CASRefWriterAppendLane, CheckpointConflictAfterLogCommitRequiresRecoveryWit auto store = openPool(backend); const RootNamespace ns{"srv1/frontier-conflict"}; publishEmptyPart(store, ns, "x"); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const NamespaceLifeId life = lifeIfCatalogedForTest(backend, store->layout(), ns).value(); const String ckpt_key = store->layout().refCkptKey(life); - const auto before = readCkpt(*backend, store->layout(), life); + const auto before = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(before); ASSERT_TRUE(before->ckpt.committed_through); const RefTxnId candidate{before->ckpt.committed_through->writer_epoch, before->ckpt.committed_through->ref_sequence + 1}; const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); + /// Latched, not counted: the frontier publication re-reads and reissues on every refusal until ITS + /// window closes, so a bounded refusal would be outlived and the checkpoint would advance. + auto clock = VirtualRetryClock::installOn(store); backend->ckpt_conflict_key = ckpt_key; - backend->ckpt_conflict_count = 100; + backend->ckpt_conflict_latched = true; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + backend->disarmFaults(); + EXPECT_GT(clock->pauseCount(), 1u) + << "the publication's reissues must pace through the injected sleep, never a real one"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) << "the durable log was not installed or acknowledged"; - const auto after = readCkpt(*backend, store->layout(), life); + const auto after = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(after); EXPECT_EQ(after->ckpt.committed_through, before->ckpt.committed_through); - EXPECT_TRUE(backend->get(store->layout().refLogKey(life, candidate))) + EXPECT_TRUE(readOf(backend, store->layout().refLogKey(life, candidate))) << "the log PUT committed before checkpoint publication failed"; - EXPECT_FALSE(backend->get(store->layout().refLogKey( + EXPECT_FALSE(readOf(backend, store->layout().refLogKey( life, RefTxnId{candidate.writer_epoch, candidate.ref_sequence + 1}))) << "no later id may be allocated above an unfrontiered durable transaction"; } @@ -1400,9 +1582,9 @@ TEST(CASRefWriterAppendLane, FenceMovementAtCheckpointPublicationRequiresRecover auto store = openPool(backend); const RootNamespace ns{"srv1/frontier-fenced"}; publishEmptyPart(store, ns, "x"); - const NamespaceLifeId life = CasRefCatalog::lifeIfCataloged(*backend, store->layout(), ns).value(); + const NamespaceLifeId life = lifeIfCatalogedForTest(backend, store->layout(), ns).value(); const String ckpt_key = store->layout().refCkptKey(life); - const auto before = readCkpt(*backend, store->layout(), life); + const auto before = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(before); ASSERT_TRUE(before->ckpt.committed_through); const RefTxnId candidate{before->ckpt.committed_through->writer_epoch, @@ -1416,11 +1598,11 @@ TEST(CASRefWriterAppendLane, FenceMovementAtCheckpointPublicationRequiresRecover EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery); EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_before) << "the fenced frontier attempt must not install or acknowledge the durable log"; - const auto after = readCkpt(*backend, store->layout(), life); + const auto after = readCkptForTest(backend, store->layout(), life); ASSERT_TRUE(after); EXPECT_EQ(after->ckpt.committed_through, before->ckpt.committed_through); - EXPECT_TRUE(backend->get(store->layout().refLogKey(life, candidate))); - EXPECT_FALSE(backend->get(store->layout().refLogKey( + EXPECT_TRUE(readOf(backend, store->layout().refLogKey(life, candidate))); + EXPECT_FALSE(readOf(backend, store->layout().refLogKey( life, RefTxnId{candidate.writer_epoch, candidate.ref_sequence + 1}))); } @@ -1539,14 +1721,11 @@ TEST(CASRefWriterAppendLane, InvalidBatchEntryGetsOwnExceptionBatchSurvives) TEST(CASRefWriterAppendLane, WedgedLaneBlocksSameTableWhileOtherTableProceeds) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns_a{"srv1/wedge_a"}; const RootNamespace ns_b{"srv1/wedge_b"}; @@ -1555,9 +1734,7 @@ TEST(CASRefWriterAppendLane, WedgedLaneBlocksSameTableWhileOtherTableProceeds) publishEmptyPart(store, ns_b, "y"); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_a)) + "_log/"; - backend->fault_count = 1; - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_a, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns_a, "x"); }); EXPECT_TRUE(store->refLaneWedgedForTest(ns_a)); /// A different table proceeds normally while ns_a stays wedged. @@ -1577,23 +1754,18 @@ TEST(CASRefWriterAppendLane, WedgedLaneBlocksSameTableWhileOtherTableProceeds) TEST(CASRefWriterAppendLane, WedgedAppendObservedDurableAppliesBeforeNextId) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/wedge_unwedge"}; publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "not yet applied while wedged"; @@ -1617,14 +1789,11 @@ TEST(CASRefWriterAppendLane, WedgedAppendObservedDurableAppliesBeforeNextId) /// tail counter is a stable running count. TEST(CASRefWriterAppendLane, WedgeResolutionJoinsTailCountersAndFoldsOverlay) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/wedge_tail"}; publishEmptyPart(store, ns, "x"); @@ -1632,10 +1801,10 @@ TEST(CASRefWriterAppendLane, WedgeResolutionJoinsTailCountersAndFoldsOverlay) const size_t tail_after_setup = store->tailSinceSnapshotCountForTest(ns); - /// Wedge the lane: the single-attempt budget turns the ambiguous log PUT into an Unresolved outcome. + /// Wedge the lane: every attempt of the log create is unresolved, so the call gives up at its own + /// retry window having sent something. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_TRUE(store->resolveRef(ns, "x").has_value()) << "not applied while merely wedged"; EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), tail_after_setup) @@ -1661,14 +1830,11 @@ TEST(CASRefWriterAppendLane, WedgeResolutionJoinsTailCountersAndFoldsOverlay) /// it, and it must track the wedge's full lifecycle (0 -> 1 -> 0), not just a one-shot snapshot. TEST(CASRefWriterAppendLane, WedgedRefLaneCountTracksExactlyTheWedgedTableThroughItsLifecycle) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns_a{"srv1/wedge_count_a"}; const RootNamespace ns_b{"srv1/wedge_count_b"}; @@ -1678,9 +1844,7 @@ TEST(CASRefWriterAppendLane, WedgedRefLaneCountTracksExactlyTheWedgedTableThroug ASSERT_EQ(store->wedgedRefLaneCount(), 0u) << "both tables cached and healthy before the fault"; backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_a)) + "_log/"; - backend->fault_count = 1; - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_a, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns_a, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns_a)); EXPECT_EQ(store->wedgedRefLaneCount(), 1u); @@ -1708,11 +1872,10 @@ TEST(CASRefWriterAppendLane, WedgedRefLaneCountTracksExactlyTheWedgedTableThroug /// bookkeeping is restored, proven by a bounded wait on both a same-table and an independent-table /// append. /// -/// The reaction is now the mount's, not the table's [review I5]: a foreign object at a key that +/// The reaction is now the mount's, not the table's: a foreign object at a key that /// mount-lease exclusivity says is exclusively ours contradicts the exclusivity itself, so the append /// site routes through `reportImpossibleInterference` exactly as the wedge-resolve site does -- fence -/// closed, remount scheduled. Before this task it failed closed and stayed closed, blocking the table -/// until somebody remounted by hand. So there are two separate scopes to keep straight, and this test +/// closed, remount scheduled. The fence is released only after remount, so there are two separate scopes to keep straight, and this test /// pins both: /// the FENCE is mount-wide -- while it is closed EVERY lane is refused, including untouched ones; /// the DAMAGE is per-namespace -- a real remount replaces both immutable runtimes, then recovery of @@ -1750,13 +1913,16 @@ TEST(CASRefWriterAppendLane, I1AppendCorruptionSurfacesAndFencesTheMountForRemou /// substitute anymore: immutable runtimes retain the generation that admitted them and cannot be /// rebound to the new one. const String mount_key = layout.mountKey("test"); - const auto mount = backend->get(mount_key); + const auto mount = readOf(backend, mount_key); ASSERT_TRUE(mount); MountLease fenced_mount = decodeMountLease(mount->bytes); fenced_mount.gc_fenced = true; fenced_mount.seq += 1; - ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(fenced_mount), mount->token).outcome, - PutOutcome::Done); + { + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.replace(mount_key, encodeMountLease(fenced_mount), mount->etag, Retry::standard()))); + } ASSERT_TRUE(store->tryRemountOnce()); auto same = std::async(std::launch::async, [&] { store->dropRef(ns, "x"); }); @@ -1784,30 +1950,26 @@ TEST(CASRefWriterAppendLane, I1AppendCorruptionSurfacesAndFencesTheMountForRemou /// occupant for what it is -- corruption -- and one test covers every build. TEST(CASRefWriterAppendLane, I1WedgeResolveCorruptionSurfacesAndFaultsLane) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/i1_wedge"}; publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - /// Wedge the lane with an ambiguous PUT that never landed. + /// Wedge the lane: every attempt of the log create is unresolved and nothing lands. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); /// A foreign writer lands a DIFFERENT object at the exact wedged key; the next append's wedge resolve /// observes the mismatch and must raise `CORRUPTED_DATA` to that caller while faulting the lane. const String wedged_key = store->wedgedKeyForTest(ns); ASSERT_FALSE(wedged_key.empty()); - ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + ASSERT_TRUE(createRaw(backend, wedged_key, "a-different-object")); auto fut = std::async(std::launch::async, [&] { @@ -1850,26 +2012,28 @@ TEST(CASRefWriterAppendLane, I1WedgeResolveCorruptionSurfacesAndFaultsLane) /// used to stand in for debug/sanitizer builds went with it. TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); - SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/anomaly_wedge"}; publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - store->setEventSink([&](const CasEvent & e) { seen.add(e); }); + store->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); - /// Wedge the lane with an ambiguous PUT that never landed. + /// Wedge the lane: every attempt of the log create is unresolved and nothing lands. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_TRUE(store->mayMutate()) << "the fence must not be tripped yet -- only an ordinary Unresolved wedge so far"; ASSERT_EQ(store->scheduleRemountCallCountForTest(), 0u) << "no remount must have been scheduled yet by the ordinary wedge alone"; @@ -1877,7 +2041,7 @@ TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) /// Out-of-band, a foreign writer lands DIFFERENT bytes at the exact wedged key. const String wedged_key = store->wedgedKeyForTest(ns); ASSERT_FALSE(wedged_key.empty()); - ASSERT_EQ(backend->putIfAbsent(wedged_key, "a-different-object").outcome, PutOutcome::Done); + ASSERT_TRUE(createRaw(backend, wedged_key, "a-different-object")); /// The next append's wedge resolve observes the mismatch: CORRUPTED_DATA, the fence trips closed, /// and a ForeignInterference event is audited. @@ -1892,12 +2056,12 @@ TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) /// `scheduleRemount`'s own entry regardless of `background_watermark` -- see that accessor's /// comment for why this test deliberately does NOT enable `background_watermark` to observe a real /// automatic recovery: doing so was tried and makes the store's self-remount attempt race its own - /// still-live keeper for 30+ seconds per call (confirmed while building this test), which is not + /// still-live renewer for 30+ seconds per call (confirmed while building this test), which is not /// something a fast unit test should be driving. EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) << "reportImpossibleInterference must have called scheduleRemount exactly once"; - const std::vector observed = seen.snapshot(); + const std::vector observed = seen->snapshot(); const auto has_event = std::any_of(observed.begin(), observed.end(), [](const CasEvent & e) { return e.type == CasEventType::ForeignInterference; }); EXPECT_TRUE(has_event) << "a ForeignInterference CasEvent must be audited"; @@ -1909,13 +2073,19 @@ TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) TEST(CASAnomalyPolicy, NonReadyAtNewIdAllocationFaultsAndFailsClosed) { auto backend = std::make_shared(); - SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto store = openPool(backend); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/wedge_contract"}; publishEmptyPart(store, ns, "x"); - store->setEventSink([&](const CasEvent & e) { seen.add(e); }); + store->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); store->setRefPreCarveHookForTest([&] { @@ -1930,7 +2100,7 @@ TEST(CASAnomalyPolicy, NonReadyAtNewIdAllocationFaultsAndFailsClosed) String cursor; for (;;) { - const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = listForTest(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -1960,11 +2130,11 @@ TEST(CASAnomalyPolicy, NonReadyAtNewIdAllocationFaultsAndFailsClosed) EXPECT_FALSE(store->mayMutate()) << "the local write fence must trip closed on the wedge-contract violation"; /// See the sibling test's comment on why this checks the call-count seam (never /// `background_watermark` plus automatic recovery -- that combination makes the store's self-remount - /// race its own still-live keeper). + /// race its own still-live renewer). EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) << "reportImpossibleInterference must have called scheduleRemount exactly once"; - const std::vector observed = seen.snapshot(); + const std::vector observed = seen->snapshot(); const auto has_event = std::any_of(observed.begin(), observed.end(), [](const CasEvent & e) { return e.type == CasEventType::ForeignInterference; }); EXPECT_TRUE(has_event) << "a ForeignInterference CasEvent must be audited"; @@ -1991,48 +2161,58 @@ TEST(CASAnomalyPolicy, DiagnosticDispatchLoggingCannotReplaceFailClosedException expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->dropRef(ns, "x"); }); } -/// I3: a conditional write whose attempt classified Committed but whose FINAL post-write fence check -/// failed (the mount fence was lost after the write may have landed) is counted separately, not folded -/// into the generic Unresolved classifier (spec §Late Predecessor PUT best-effort diagnostic). -TEST(CASRequestControllerFenceLoss, I3PostWriteFenceLossIsCounted) +/// A write whose attempt landed but whose post-commit admission check failed is counted separately, +/// not folded into the generic unresolved give-up: the object may exist, and only the caller's own +/// resolution of the key may say so. +TEST(CASRefWriteContract, PostWriteFenceLossIsCounted) { using ProfileEvents::global_counters; auto backend = std::make_shared(); - CasRequestBudget budget; - budget.max_attempts = 3; - CasRequestController ctrl(backend, budget, [] { return static_cast(0); }); // fixed clock - - /// `fence_ok` holds for the pre-attempt check, then is lost by the post-write check. - int calls = 0; - auto fence_ok = [&calls] { return ++calls <= 1; }; - - const auto before = global_counters[ProfileEvents::CASConditionalWriteFenceLostPostWrite].load(); - const CasWriteOutcome outcome = ctrl.putIfAbsentControlled("k", "v", fence_ok); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved) << "a post-write fence loss must never be reported as Committed"; - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteFenceLostPostWrite].load(), before + 1); + bool live = true; + backend->onWriteCommitted("k", [&live] { live = false; }); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([&live] { return live; }); + + const auto before = global_counters[ProfileEvents::CASRequestFenceLostPostWrite].load(); + const WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr) << "a post-write fence loss must never be reported as committed"; + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_EQ(global_counters[ProfileEvents::CASRequestFenceLostPostWrite].load(), before + 1); } -/// Task B (stageManifest rides the controller): a Committed return surfaces the committed -/// incarnation's token — from the attempt's own PutResult, and equally from a resolve that proves an -/// earlier ambiguous attempt landed — so audit emitters (`PartWriteTxn::stageManifest`'s `ManifestPut` -/// event) keep their token without a follow-up HEAD. -TEST(CASRequestController, CommittedSurfacesTokenFromPutAndFromResolve) +/// A commit surfaces the incarnation it created -- from the attempt's own response, and equally from a +/// settling read that proves an earlier ambiguous attempt of the SAME call landed -- so an audit +/// emitter keeps the incarnation without a follow-up head. +TEST(CASRefWriteContract, CommittedSurfacesTheIncarnationFromTheWriteAndFromTheSettlingRead) { auto backend = std::make_shared(); - CasRequestController ctrl(backend, CasRequestBudget{}, [] { return static_cast(0); }); - const auto fence_ok = [] { return true; }; - - Token direct_token; - ASSERT_EQ(ctrl.putIfAbsentControlled("k1", "v1", fence_ok, &direct_token), CasWriteOutcome::Committed); - EXPECT_EQ(direct_token, backend->head("k1").token) << "the direct-commit token is the PutResult's"; - - /// k2 already holds the IDENTICAL bytes (an earlier ambiguous attempt that landed): the attempt's - /// PreconditionFailed collapses to Unresolved and the resolve GET proves Committed — the token must - /// be the observed incarnation's, and no second incarnation is ever created. - const Token pre_existing = backend->putIfAbsent("k2", "v2").token; - Token resolved_token; - ASSERT_EQ(ctrl.putIfAbsentControlled("k2", "v2", fence_ok, &resolved_token), CasWriteOutcome::Committed); - EXPECT_EQ(resolved_token, pre_existing) << "the resolve-commit token is the observed incarnation's"; + CasRequests requests(backend, Fence::open()); + + CasOperation direct = requests.admit(); + const WriteResult first = direct.create("k1", "v1", Retry::standard()); + const auto * direct_committed = std::get_if(&first); + ASSERT_TRUE(direct_committed != nullptr); + EXPECT_FALSE(direct_committed->resolved_by_read); + CasOperation reader = requests.admit(); + const auto observed = reader.head("k1", Retry::standard()); + ASSERT_TRUE(observed.has_value()); + EXPECT_EQ(direct_committed->etag, observed->etag) + << "the direct-commit incarnation is the write's own response"; + + /// The write of `k2` lands and its response is lost: the settling read proves the commit, and the + /// incarnation reported is the one that is actually there. + backend->injectAmbiguousLandedWrite("k2"); + CasOperation resolving = requests.admit(); + const WriteResult second = resolving.create("k2", "v2", Retry::standard()); + const auto * resolved_committed = std::get_if(&second); + ASSERT_TRUE(resolved_committed != nullptr); + EXPECT_TRUE(resolved_committed->resolved_by_read); + CasOperation second_reader = requests.admit(); + const auto second_observed = second_reader.head("k2", Retry::standard()); + ASSERT_TRUE(second_observed.has_value()); + EXPECT_EQ(resolved_committed->etag, second_observed->etag) + << "the resolve-commit incarnation is the observed one"; } /// =================================================================================== @@ -2087,24 +2267,21 @@ TEST(CASRefTableCacheEviction, ZeroBudgetDisablesEviction) /// re-recovery must not be allowed to drop and re-materialize it (which could re-allocate an id). TEST(CASRefTableCacheEviction, WedgedTableIsNeverEvicted) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPoolWithConfig(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .cas_request_budget = budget, .ref_table_cache_bytes = 1}); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns_w{"srv1/wedged"}; publishEmptyPart(store, ns_w, "x"); - /// Wedge ns_w's append lane with one ambiguous (Unresolved) PUT that exhausts the single-attempt budget. + /// Wedge ns_w's append lane: every attempt of its log create is unresolved, so the call gives up at + /// its own retry window having sent something. backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns_w)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns_w, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns_w, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns_w)); /// Pressure the cache with other tables. ns_w is idle and over the 1-byte budget, but its wedged lane @@ -2144,7 +2321,7 @@ TEST(CASRefWriterSnapshotPublish, ThresholdTriggerPublishesCacheReplayEquivalent EXPECT_TRUE(store->newestPublishedSnapshotIdForTest(ns) == snap_id); EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 0u) << "a snapshot covering everything prunes the whole tail"; - const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + const auto got = readOf(backend, layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); ASSERT_TRUE(got.has_value()); /// The independent oracle: replay every `_log/` object directly, ignoring the snapshot entirely. @@ -2170,7 +2347,7 @@ TEST(CASRefWriterSnapshotPublish, CapturedPredecessorCannotPublishAfterSameNameR const RootNamespace ns{"srv1/publisher-predecessor-rebirth"}; publishWithProductionBirth(store, ns, "predecessor"); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); const auto predecessor_snapshot @@ -2211,15 +2388,15 @@ TEST(CASRefWriterSnapshotPublish, CapturedPredecessorCannotPublishAfterSameNameR EXPECT_NO_THROW(store->dropNamespace(ns)); EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); - const RefCatalog after_removal = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog after_removal = readCatalogForTest(backend, layout).catalog; EXPECT_TRUE(std::none_of(after_removal.entries.begin(), after_removal.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; })); publishWithProductionBirth(store, ns, "successor"); const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); EXPECT_NE(successor.incarnation, predecessor.incarnation); - const auto predecessor_ckpt_before_resume = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_before_resume = readOf(backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_before_resume) << "the removal protocol leaves this checkpoint as janitor-owned predecessor debris"; @@ -2232,11 +2409,11 @@ TEST(CASRefWriterSnapshotPublish, CapturedPredecessorCannotPublishAfterSameNameR EXPECT_FALSE(publisher.get()); store->setSnapshotAfterCaptureHookForTest(nullptr); - EXPECT_FALSE(backend->get(layout.refSnapshotKey(predecessor_life, *predecessor_snapshot))) + EXPECT_FALSE(readOf(backend, layout.refSnapshotKey(predecessor_life, *predecessor_snapshot))) << "a stale publisher recreated the retired predecessor snapshot"; - const auto predecessor_ckpt_after_resume = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_after_resume = readOf(backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_after_resume); - EXPECT_EQ(predecessor_ckpt_after_resume->token, predecessor_ckpt_before_resume->token) + EXPECT_EQ(predecessor_ckpt_after_resume->etag, predecessor_ckpt_before_resume->etag) << "a stale publisher replaced the retired predecessor checkpoint"; EXPECT_EQ(predecessor_ckpt_after_resume->bytes, predecessor_ckpt_before_resume->bytes) << "a stale publisher changed the retired predecessor checkpoint"; @@ -2260,7 +2437,7 @@ TEST(CASRefWriterSnapshotPublish, RetiredPredecessorCannotAdvanceCkptAfterSnapsh const RootNamespace ns{"srv1/publisher-predecessor-ckpt-race"}; publishWithProductionBirth(store, ns, "predecessor"); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); const auto candidate_id = listGreatestLogIdForLifeForTest(*backend, layout, predecessor_life); @@ -2296,21 +2473,21 @@ TEST(CASRefWriterSnapshotPublish, RetiredPredecessorCannotAdvanceCkptAfterSnapsh std::unique_lock lock(mutex); ASSERT_TRUE(cv.wait_for(lock, std::chrono::seconds(10), [&] { return before_ckpt_cas; })); } - EXPECT_TRUE(backend->get(layout.refSnapshotKey(predecessor_life, *candidate_id))) + EXPECT_TRUE(readOf(backend, layout.refSnapshotKey(predecessor_life, *candidate_id))) << "the hook must run after the snapshot body PUT and immediately before `_ckpt` admission"; EXPECT_NO_THROW(store->dropNamespace(ns)); EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); - const RefCatalog after_removal = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog after_removal = readCatalogForTest(backend, layout).catalog; EXPECT_TRUE(std::none_of(after_removal.entries.begin(), after_removal.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; })); publishWithProductionBirth(store, ns, "successor"); const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); EXPECT_NE(successor.incarnation, predecessor.incarnation); - const auto predecessor_ckpt_before_resume = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_before_resume = readOf(backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_before_resume); { @@ -2322,9 +2499,9 @@ TEST(CASRefWriterSnapshotPublish, RetiredPredecessorCannotAdvanceCkptAfterSnapsh EXPECT_FALSE(publisher.get()); store->setSnapshotBeforeCkptCasHookForTest(nullptr); - const auto predecessor_ckpt_after_resume = backend->get(layout.refCkptKey(predecessor_life)); + const auto predecessor_ckpt_after_resume = readOf(backend, layout.refCkptKey(predecessor_life)); ASSERT_TRUE(predecessor_ckpt_after_resume); - EXPECT_EQ(predecessor_ckpt_after_resume->token, predecessor_ckpt_before_resume->token); + EXPECT_EQ(predecessor_ckpt_after_resume->etag, predecessor_ckpt_before_resume->etag); EXPECT_EQ(predecessor_ckpt_after_resume->bytes, predecessor_ckpt_before_resume->bytes); EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), successor_runtime); EXPECT_TRUE(store->resolveRef(ns, "successor")); @@ -2347,7 +2524,7 @@ TEST(CASRefWriterRuntimeIdentity, CapturedReaderCannotRetargetSameNameSuccessor) publishWithProductionBirth(store, ns, "shared"); ASSERT_TRUE(store->resolveRef(ns, "shared")); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); Gc gc(store, UInt128{107}); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); @@ -2384,7 +2561,7 @@ TEST(CASRefWriterRuntimeIdentity, CapturedReaderCannotRetargetSameNameSuccessor) EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); publishWithProductionBirth(store, ns, "shared"); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); EXPECT_NE(successor.incarnation, predecessor.incarnation); const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); const auto successor_ref = store->resolveRef(ns, "shared"); @@ -2421,7 +2598,7 @@ TEST(CASRefWriterRuntimeIdentity, CapturedAppendCannotEnqueueIntoSameNameSuccess const RootNamespace ns{"srv1/captured-append-rebirth"}; publishWithProductionBirth(store, ns, "shared"); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); Gc gc(store, UInt128{108}); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); @@ -2458,7 +2635,7 @@ TEST(CASRefWriterRuntimeIdentity, CapturedAppendCannotEnqueueIntoSameNameSuccess EXPECT_FALSE(runRegularRoundReclaiming(gc).deferred); EXPECT_TRUE(runRegularRoundReclaiming(gc).deferred); publishWithProductionBirth(store, ns, "shared"); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); EXPECT_NE(successor.incarnation, predecessor.incarnation); const uint64_t successor_runtime = store->refTableRuntimeIdentityForTest(ns); const auto successor_ref = store->resolveRef(ns, "shared"); @@ -2492,7 +2669,7 @@ TEST(CASRefWriterRuntimeIdentity, LatePredecessorInvalidationLeavesSuccessorAtta const RootNamespace ns{"srv1/late-predecessor-invalidation"}; publishWithProductionBirth(store, ns, "predecessor"); - const CatalogEntry predecessor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry predecessor = catalogEntryOrThrow(backend, layout, ns); const NamespaceLifeId predecessor_life = NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation); Gc gc(store, UInt128{109}); @@ -2502,7 +2679,7 @@ TEST(CASRefWriterRuntimeIdentity, LatePredecessorInvalidationLeavesSuccessorAtta ASSERT_TRUE(runRegularRoundReclaiming(gc).deferred); publishWithProductionBirth(store, ns, "successor"); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); ASSERT_NE(successor.incarnation, predecessor.incarnation); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation); @@ -2589,7 +2766,7 @@ TEST(CASRefWriterSnapshotPublish, MountTimeRecoveredLargeTailPublishesAfterOrdin << "the ordinary successor must make the inherited mount-time tail publishable"; EXPECT_EQ(successor->tailSinceSnapshotCountForTest(ns), 0u); - const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + const auto got = readOf(backend, layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); ASSERT_TRUE(got.has_value()); const RefTableState oracle = independentFullReplayForTest(*backend, layout, ns, snap_id); EXPECT_EQ(openObject(FormatId::RefSnapshot, got->bytes), encodeRefTableSnapshot(snapshotOf(oracle, ns.string()))); @@ -2625,7 +2802,7 @@ TEST(CASRefWriterPublishFromLive, YoungTxnIsCoveredImmediately) << "publish-from-live: a just-committed txn is immediately coverable, with no grace window"; const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); ASSERT_TRUE(snap_id.has_value()); - const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + const auto got = readOf(backend, layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); ASSERT_TRUE(got.has_value()); const RefTableSnapshot snap = decodeRefTableSnapshot(openObject(FormatId::RefSnapshot, got->bytes), ns.string(), *snap_id); ASSERT_EQ(snap.committed.size(), 1u); @@ -2645,7 +2822,10 @@ TEST(CASRefWriterSnapshotPublish, TriggerFiresOnCountAboveThresholdWithoutAging) PoolConfig config; config.snapshot_log_count_threshold = 3; config.snapshot_log_bytes_threshold = 1ULL << 40; - config.boot_ms_fn = [&fake_now] { return fake_now; }; + /// Captured by value: the value never changes, and the Pool can outlive this stack frame (a + /// background publish holds `shared_from_this()`), so a by-reference capture of `fake_now` would + /// dangle once the frame returns. + config.boot_ms_fn = [fake_now] { return fake_now; }; config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); auto store = openPoolWithConfig(backend, config); @@ -2808,7 +2988,7 @@ TEST(CASRefWriterSnapshotPublish, ConcurrentOutOfOrderPublishDoesNotRegressBaseN ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); const auto snap_id = listGreatestSnapshotIdForTest(*backend, layout, ns); ASSERT_TRUE(snap_id.has_value()); - const auto got = backend->get(layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); + const auto got = readOf(backend, layout.refSnapshotKey(DB::Cas::tests::fixture::fixtureLife(ns), *snap_id)); ASSERT_TRUE(got.has_value()); const RefTableState oracle = independentFullReplayForTest(*backend, layout, ns, snap_id); EXPECT_EQ(openObject(FormatId::RefSnapshot, got->bytes), encodeRefTableSnapshot(snapshotOf(oracle, ns.string()))) @@ -2904,11 +3084,7 @@ TEST(CASRefWriterSnapshotPublish, C4LatchBoundedUnderSustainedNonCommittedPublis auto backend = std::make_shared(); const RootNamespace ns{"srv1/c4_latch"}; - CasRequestBudget budget; /// one attempt per publish so a failure is a single PUT - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); uint64_t fake_now = 1'000'000; PoolConfig config; @@ -2916,16 +3092,29 @@ TEST(CASRefWriterSnapshotPublish, C4LatchBoundedUnderSustainedNonCommittedPublis config.snapshot_log_bytes_threshold = 1ULL << 40; config.snapshot_publish_backoff_initial_ms = 5000; /// the frozen clock keeps the backoff armed config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + /// Captured by value: `fake_now` never changes and the Pool can outlive this stack frame (a + /// background publish holds `shared_from_this()`), so a by-reference capture would dangle. + config.boot_ms_fn = [fake_now] { return fake_now; }; config.cas_request_budget = budget; auto store = openPoolWithConfig(backend, config); - - /// Every `_snap` PUT throws Unresolved (backend saturated), from the very first publish attempt. + /// The boot clock above is frozen (it is what keeps the publish backoff armed), so the REQUEST + /// engine needs its own advancing clock or a saturated publish never reaches its retry window and + /// reissues for ever. Retained (not discarded) so the assertion below can tell that retry window + /// from a `once` policy that would give up on the very first attempt. + auto clock = VirtualRetryClock::installOn(store); + + /// Every `_snap` create is unresolved (backend saturated) and stays that way for the whole call, so + /// the publish gives up at its own window -- which is one dispatch, which is what this test counts. backend->fault_key_substr = "_snap/"; - backend->fault_count = 100000; + backend->fault_latched = true; publishEmptyPart(store, ns, "a"); /// crosses the threshold -> one dispatch -> fails -> backoff armed store->waitForSnapshotPublishSettleForTest(ns); + EXPECT_GT(clock->pauseCount(), 1u) + << "the one dispatch above must itself have reissued more than once against the saturated " + "backend before giving up at its retry window -- a single attempt would not distinguish " + "this from a non-retrying policy"; + EXPECT_GT(clock->longestPause(), 0u) << "at least one of those reissues must have paced with a real backoff"; const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); for (int i = 0; i < 30; ++i) @@ -2953,14 +3142,14 @@ TEST(CASRefWriterSnapshotPublish, RecoveredSealAboveThresholdDoesNotRedispatchUn predecessor_config.snapshot_log_bytes_threshold = 1ULL << 40; auto predecessor = openPoolWithConfig(backend, predecessor_config); DB::Cas::tests::fixture::admitLive(*backend, predecessor->layout(), ns); - const NamespaceLifeId life = *CasRefCatalog::lifeIfCataloged(*backend, predecessor->layout(), ns); - ASSERT_EQ(backend->putIfAbsent(predecessor->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ + const NamespaceLifeId life = *lifeIfCatalogedForTest(backend, predecessor->layout(), ns); + ASSERT_TRUE(createRaw(backend, predecessor->layout().refCkptKey(life), encodeRefCkpt(RefCkpt{ .life_epoch = predecessor->liveWriterEpoch(), .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); publishEmptyPart(predecessor, ns, "before_seal"); - const auto before = backend->get(predecessor->layout().refCkptKey(life)); + const auto before = readOf(backend, predecessor->layout().refCkptKey(life)); ASSERT_TRUE(before); const RefCkpt before_seal = decodeRefCkpt(before->bytes); ASSERT_TRUE(before_seal.committed_through); @@ -2971,8 +3160,10 @@ TEST(CASRefWriterSnapshotPublish, RecoveredSealAboveThresholdDoesNotRedispatchUn RefCkpt recovered_seal = before_seal; recovered_seal.committed_through = seal_id; recovered_seal.last_epoch_seal = seal_id; - ASSERT_EQ(backend->casPut(predecessor->layout().refCkptKey(life), encodeRefCkpt(recovered_seal), before->token).outcome, - CasOutcome::Committed); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.replace( + predecessor->layout().refCkptKey(life), encodeRefCkpt(recovered_seal), before->etag, Retry::standard()))); } PoolConfig successor_config; @@ -3036,29 +3227,37 @@ TEST(CASRefWriterSnapshotPublish, C4BackoffDefersThenRetriesAndPublishes) const Layout layout("p"); const RootNamespace ns{"srv1/c4_backoff"}; - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: this test mutates the clock after the Pool exists + /// (below), and the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config; config.snapshot_log_count_threshold = 1; config.snapshot_log_bytes_threshold = 1ULL << 40; config.snapshot_publish_backoff_initial_ms = 1000; config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] { return fake_now->load(); }; config.cas_request_budget = budget; auto store = openPoolWithConfig(backend, config); + /// As above: the frozen boot clock drives the backoff decisions, so the request engine gets its + /// own. Retained (not discarded) so the assertion below can tell the retry window that arms the + /// backoff from a `once` policy that would give up on the very first attempt. + auto clock = VirtualRetryClock::installOn(store); - /// Fail ONLY the first `_snap` PUT (arms the backoff); later PUTs succeed. + /// Fail the FIRST dispatch's `_snap` create for the whole call, so it gives up at its own retry + /// window and arms the backoff; the fault is cleared before the retry below. backend->fault_key_substr = "_snap/"; - backend->fault_count = 1; + backend->fault_latched = true; publishEmptyPart(store, ns, "a"); /// dispatch -> publish fails -> backoff armed store->waitForSnapshotPublishSettleForTest(ns); EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); + EXPECT_GT(clock->pauseCount(), 1u) + << "the failing dispatch above must itself have reissued more than once before giving up and " + "arming the backoff -- a single attempt would not distinguish this from a non-retrying policy"; + EXPECT_GT(clock->longestPause(), 0u) << "at least one of those reissues must have paced with a real backoff"; /// A read within the backoff window (frozen clock) must not re-dispatch. const auto d1 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); @@ -3068,8 +3267,9 @@ TEST(CASRefWriterSnapshotPublish, C4BackoffDefersThenRetriesAndPublishes) << "a read within the backoff window must not re-dispatch"; EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); - /// Advance past the backoff: exactly one retry is dispatched and it publishes. - fake_now += 2000; + /// Advance past the backoff, with the fault cleared: exactly one retry is dispatched and it publishes. + backend->disarmFaults(); + *fake_now += 2000; store->resolveRef(ns, "a"); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d1 + 1) @@ -3223,32 +3423,29 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM .last_epoch_seal = std::nullopt, }); - /// The successor: a tight retry budget so ONE simulated ambiguous response wedges rather than - /// transparently retries away. `8f9e63c7a19` widened `kSingleAttemptDeadlineMs` off a zero-width - /// race (equal attempt/operation deadlines), but it still measures the capture-to-gate window -- - /// encoding the removal chunk (up to `ref_txn_max_ops` ops) -- against the REAL wall clock, so it - /// recurred (3 of 3 sanitizer lanes) once that encode step got slow enough on its own, independent - /// of scheduler contention: msan in particular. `ref_request_controller` reads its clock through - /// the same injectable seam as the mount fence (`CasRefLedger`'s `controller_boot_ms_fn` is the - /// pool's `boot_ms_fn`), so freeze it here instead of racing it -- the fault-injecting PUT below - /// still reaches the backend synchronously; only the deadline arithmetic stops moving. - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + /// The successor: `fault_latched` below makes ONE simulated ambiguous response wedge the lane + /// rather than being resolved by the engine's own reissue. `boot_ms_fn` seeds every plane's own + /// clock at construction and drives the mount lease's deadline math, so freezing it here keeps the + /// fence alive across the whole retry window; the request engine gets its own separately advancing + /// clock below, so its reissues still pace forward. + const CasRequestBudget budget = wedgeTestBudget(); PoolConfig config; config.cas_request_budget = budget; config.boot_ms_fn = [] { return uint64_t{0}; }; auto successor = openPoolWithConfig(backend, config); + /// The boot clock above is frozen, so the request engine needs its own advancing one: an armed + /// fault otherwise reissues for ever instead of ending the call at its retry window. + auto clock = VirtualRetryClock::installOn(successor); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; /// The successor's own recovery runs first and mints one in-band seal for the dead predecessor /// epoch `e1` (its durable ids are `{e1,1}` and `{e1,2}`, so the seal lands at `{e1,3}`) -- that PUT /// shares this same `_log/` prefix, so it would eat the fault before the sweep ever gets a chance. - /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. + /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. Latched past + /// the skip, because the write engine reissues within a call: a single-shot fault would be answered + /// by the next attempt and the chunk would commit. backend->fault_skip = 1; - backend->fault_count = 1; /// hits exactly the sweep's FIRST removal chunk's PUT + backend->fault_latched = true; /// The sweep is piggybacked on this mount's very first touch; its (uncertain) failure is INSULATED /// from the read (resolveRef/listRefs call `sweepStalePrecommitsForRead`, not @@ -3259,6 +3456,9 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM EXPECT_EQ(deferred_after, deferred_before + 1) << "the read-only caller must observe (and count) the deferred sweep failure, not throw"; EXPECT_TRUE(successor->refLaneWedgedForTest(ns)); + backend->disarmFaults(); + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; /// The first chunk's request actually landed server-side; the caller just never saw the ack. backend->materializePendingDelayedWrite(); @@ -3272,10 +3472,12 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM /// pays a real ~36.5s token-stability observation wait here. Inject a fake `boot_ms_fn` + /// `wait_sleep_fn` (mirroring `CASMountOpenWaits.UncleanOpenPaysOnlyTheObservationWindow`) so it /// resolves instantly. - uint64_t resumer_fake_boot = 0; + /// Held in a shared atomic, not a plain local: the Pool can outlive this stack frame (a background + /// publish holds `shared_from_this()`), so a by-reference capture of a local would dangle. + auto resumer_fake_boot = std::make_shared>(0); PoolConfig resumer_config; - resumer_config.boot_ms_fn = [&resumer_fake_boot] { return resumer_fake_boot; }; - resumer_config.wait_sleep_fn = [&resumer_fake_boot](uint64_t ms) { resumer_fake_boot += ms; }; + resumer_config.boot_ms_fn = [resumer_fake_boot] { return resumer_fake_boot->load(); }; + resumer_config.wait_sleep_fn = [resumer_fake_boot](uint64_t ms) { *resumer_fake_boot += ms; }; auto resumer = openPoolWithConfig(backend, resumer_config); EXPECT_NO_THROW(resumer->listRefs(ns)); @@ -3294,7 +3496,7 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM String cursor; for (;;) { - const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = listForTest(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -3324,11 +3526,14 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) const Layout layout("p"); const RootNamespace ns{"srv1/precommit_sweep_retry"}; - /// One shared injected clock for both incarnations. The successor's wait hook below advances this - /// same clock, so both mount observation and the later sweep-backoff deadline are deterministic. - uint64_t fake_now = 1'000'000; - size_t mount_wait_calls = 0; - const auto fake_clock = [&fake_now] { return fake_now; }; + /// One shared injected clock for both incarnations, held in a shared atomic rather than a plain + /// local: the successor Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. The successor's wait + /// hook below advances this same clock, so both mount observation and the later sweep-backoff + /// deadline are deterministic. + auto fake_now = std::make_shared>(1'000'000); + auto mount_wait_calls = std::make_shared>(0); + const auto fake_clock = [fake_now] { return fake_now->load(); }; { /// A predecessor writer leaves THREE precommits dangling (a crash before promote). @@ -3349,34 +3554,40 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) /// The successor: a tight retry budget so ONE simulated ambiguous response wedges rather than /// transparently retries away (mirrors the wedge-semantics tests in this file exactly). - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); PoolConfig config; config.cas_request_budget = budget; config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); config.boot_ms_fn = fake_clock; - config.wait_sleep_fn = [&fake_now, &mount_wait_calls](uint64_t ms) + config.wait_sleep_fn = [fake_now, mount_wait_calls](uint64_t ms) { - ++mount_wait_calls; - fake_now += ms; + ++(*mount_wait_calls); + *fake_now += ms; }; - SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto successor = openPoolWithConfig(backend, config); - EXPECT_GT(mount_wait_calls, 0u) + EXPECT_GT(mount_wait_calls->load(), 0u) << "the unclean predecessor must exercise the injected mount-observation wait"; - successor->setEventSink([&](const CasEvent & e) { seen.add(e); }); + successor->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); + /// The boot clock this test drives the sweep backoff on is its own; the request engine gets a + /// separate advancing clock, or an armed fault reissues for ever instead of ending its call. + auto clock = VirtualRetryClock::installOn(successor); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; /// The successor's own recovery runs first and mints one in-band seal for the predecessor's now-dead /// epoch (its three precommits are its only durable ids, so the seal takes the very next slot) -- /// that PUT shares this same `_log/` prefix, so it would eat the fault before the sweep gets a turn. - /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. + /// Skip it and land the fault on the sweep's FIRST removal chunk's PUT, as intended. Latched past + /// the skip, because the write engine reissues within a call. backend->fault_skip = 1; - backend->fault_count = 1; /// hits exactly the sweep's FIRST removal chunk's PUT + backend->fault_latched = true; /// FIRST trigger (read path): the sweep's removal PUT is uncertain -> the lane wedges; the read /// itself still succeeds and counts the deferral (existing contract) -- but the shot must NOT be @@ -3390,6 +3601,9 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) EXPECT_TRUE(successor->needsStalePrecommitSweepForTest(ns)) << "a failed sweep must re-arm needs_stale_precommit_sweep, not consume the once-per-mount shot"; EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepRearmed].load(), rearmed_before + 1); + backend->disarmFaults(); + EXPECT_GT(clock->pauseCount(), 1u) + << "the reissues must pace through the injected sleep, never a real one"; /// Within the backoff window (the injected clock has not advanced) a read must NOT re-attempt -- /// the bounded-backoff storm latch: no new deferral, flag still armed. @@ -3402,7 +3616,7 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) /// mutation this time) retries: the lane resolves its wedge (the first chunk's removals become /// durable and applied), the re-pass verifies clean, and the flag clears permanently. backend->materializePendingDelayedWrite(); - fake_now += 60'000; /// beyond any armed backoff (initial 200 ms, max 30 s) + *fake_now += 60'000; /// beyond any armed backoff (initial 200 ms, max 30 s) EXPECT_NO_THROW(publishEmptyPart(successor, ns, "fresh")); EXPECT_FALSE(successor->refLaneWedgedForTest(ns)); EXPECT_FALSE(successor->needsStalePrecommitSweepForTest(ns)) @@ -3417,7 +3631,7 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) /// Audit (INTROSPECTION-1): exactly ONE `precommit_reclaim` event per reclaimed stale binding -- /// this is what makes the S13 card's "abandoned precommits reclaimed" counter falsifiable. std::vector reclaimed_refs; - for (const CasEvent & e : seen.snapshot()) + for (const CasEvent & e : seen->snapshot()) if (e.type == CasEventType::PrecommitReclaim) reclaimed_refs.push_back(e.ref_name); std::sort(reclaimed_refs.begin(), reclaimed_refs.end()); @@ -3439,9 +3653,15 @@ TEST(CASRefWriterStalePrecommitSweep, VerifiedCleanSweepClearsFlagWithoutEvents) publishEmptyPart(predecessor, ns, "committed_x"); /// committed work only; nothing dangles } - SynchronizedEventLog seen; /// declared BEFORE the Pool so it outlives the background syncer's emits (ASan 2026-07-09) + /// Heap-owned, not a plain local: declaring it before the Pool (ASan 2026-07-09) only protects + /// against an ordinary same-thread unwind, not a detached background completion holding an extra + /// `shared_from_this()` that can still be running on another thread after this frame returns. + auto seen = std::make_shared(); auto successor = openPool(backend); - successor->setEventSink([&](const CasEvent & e) { seen.add(e); }); + successor->setEventSink([seen](const CasEvent & e) + { + seen->push(e); + }); const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(); @@ -3450,7 +3670,7 @@ TEST(CASRefWriterStalePrecommitSweep, VerifiedCleanSweepClearsFlagWithoutEvents) << "a clean first pass IS the verified-clean sweep: the flag clears without any removal"; EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before); EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(), reclaimed_before); - const std::vector observed = seen.snapshot(); + const std::vector observed = seen->snapshot(); EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const CasEvent & e) { return e.type == CasEventType::PrecommitReclaim; }), 0); } @@ -3470,21 +3690,23 @@ namespace /// gtest_cas_pool.cpp's fenceOutMount, without its ASSERT_ macros so it can run outside a fixture). void fenceOutRefMount(Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, Retry::standard()); MountLease m = decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - backend.putOverwrite(mount_key, encodeMountLease(m), got->token); + (void)(*op).replace(mount_key, encodeMountLease(m), got->etag, Retry::standard()); } /// The greatest `_log/` transaction id currently present for `ns` (independent of any Pool cache). std::optional listGreatestLogIdForTest(Backend & backend, const Layout & layout, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest op(backend); std::optional greatest; String cursor; for (;;) { - const ListPage page = backend.list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*op).list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -3508,25 +3730,26 @@ std::optional listGreatestLogIdForTest(Backend & backend, const Layout /// exactly what recovery must refuse. The link names the id the recovering pool's own CAS-walk will mint /// for the dead epoch: one past that epoch's greatest durable id, which is what `seal_of_previous_epoch` /// derives by listing rather than hard-coding, so the fixture cannot drift from the walk's arithmetic. -uint64_t seedTwinDrop(Backend & backend, const Layout & layout, const RootNamespace & ns, +uint64_t seedTwinDrop(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns, const String & ref_name, const ManifestRef & old_ref) { uint64_t greatest_in_previous_epoch = 0; uint64_t previous_epoch = 0; - forEachListedKey(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), [&](const ListedKey & lk) + for (const ListedKey & lk : listForTest( + backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), "", 1000).keys) { const auto parsed = layout.parseRefObjectKey(lk.key); if (!parsed || parsed->kind != RefObjectKind::Log) - return; + continue; if (parsed->txn_id.writer_epoch > previous_epoch || (parsed->txn_id.writer_epoch == previous_epoch && parsed->txn_id.ref_sequence > greatest_in_previous_epoch)) { previous_epoch = parsed->txn_id.writer_epoch; greatest_in_previous_epoch = parsed->txn_id.ref_sequence; } - }, 1000); + } - const uint64_t twin_epoch = allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); + const uint64_t twin_epoch = allocateWriterEpochForTest(backend, layout, "test"); RefLogTxn twin; twin.ns = ns.string(); twin.txn_id = RefTxnId{twin_epoch, 1}; @@ -3535,7 +3758,7 @@ uint64_t seedTwinDrop(Backend & backend, const Layout & layout, const RootNamesp drop.kind = RefOpKind::OwnerTransition; drop.old_binding = RefOwnerBinding{RefOwnerKind::Committed, ref_name, old_ref}; twin.ops = {drop}; - DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, twin); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, twin); return twin_epoch; } @@ -3561,13 +3784,16 @@ TEST(CASRefWriterRemount, FailedRemountPublishesNoRuntimeUnderFenceLossGeneratio ASSERT_EQ(predecessor_generation, store->fenceGeneration()); const String mount_key = layout.mountKey("test"); - const auto got = backend->get(mount_key); + const auto got = readOf(backend, mount_key); ASSERT_TRUE(got); MountLease foreign = decodeMountLease(got->bytes); foreign.server_uuid = foreign.server_uuid + UInt128{1}; foreign.seq += 1; - ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, - PutOutcome::Done); + { + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.replace(mount_key, encodeMountLease(foreign), got->etag, Retry::standard()))); + } store->tripMountLost(); const uint64_t rejected_generation = store->fenceGeneration(); @@ -3605,7 +3831,7 @@ TEST(CASRefWriterRemount, ReRecoversStaleCacheToTwinDrop) /// A same-uuid twin bumped the durable epoch and durably dropped "a"; this Pool's warm cache never /// observed it. - const uint64_t twin_epoch = seedTwinDrop(*backend, layout, ns, "a", a_id.ref); + const uint64_t twin_epoch = seedTwinDrop(backend, layout, ns, "a", a_id.ref); ASSERT_GT(twin_epoch, e1); ASSERT_TRUE(store->resolveRef(ns, "a").has_value()) << "precondition: the warm cache is stale"; @@ -3636,7 +3862,7 @@ TEST(CASRefWriterRemount, PostRemountAppendCarriesLiveEpochSortingAboveTwinLogs) const ManifestId a_id = publishEmptyPart(store, ns, "a"); const uint64_t e1 = store->liveWriterEpoch(); - const uint64_t twin_epoch = seedTwinDrop(*backend, layout, ns, "a", a_id.ref); + const uint64_t twin_epoch = seedTwinDrop(backend, layout, ns, "a", a_id.ref); ASSERT_GT(twin_epoch, e1); fenceOutRefMount(*backend, layout.mountKey("test")); @@ -3660,31 +3886,29 @@ TEST(CASRefWriterRemount, PostRemountAppendCarriesLiveEpochSortingAboveTwinLogs) /// its slot -- see `quiesceRefTablesForRemount`'s doc comment (`CasPool.h`). TEST(CASRefWriterRemount, DiscardsWedgeAndLaneRemainsUsable) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); /// The self-remount below blocks on nothing (see /// `CASRemountWaits.UnresolvedWedgeRemountPaysNoWaitEither`, `gtest_cas_pool.cpp`); the injected - /// `boot_ms_fn`/`wait_sleep_fn` keep this test off the real clock anyway. - uint64_t fake_boot = 0; + /// `boot_ms_fn`/`wait_sleep_fn` keep this test off the real clock anyway. Held in a shared atomic, + /// not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); PoolConfig config; config.cas_request_budget = budget; - config.boot_ms_fn = [&fake_boot] { return fake_boot; }; - config.wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }; + config.boot_ms_fn = [fake_boot] { return fake_boot->load(); }; + config.wait_sleep_fn = [fake_boot](uint64_t ms) { *fake_boot += ms; }; auto store = openPoolWithConfig(backend, config); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/remount_wedge"}; publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - /// Wedge the lane with an ambiguous PUT that never landed server-side. + /// Wedge the lane: every attempt of the log create is unresolved and nothing lands server-side. + auto clock = VirtualRetryClock::installOn(store); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); + driveToTheWedge(*clock, *backend, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); fenceOutRefMount(*backend, layout.mountKey("test")); @@ -3716,11 +3940,13 @@ TEST(CASRefWriterRemount, SupersededLeaderMidFlushFailsClosedCreatesNoObject) CasRequestBudget budget; budget.attempt_timeout_ms = 100; budget.lease_safety_margin_ms = 100; - uint64_t fake_boot = 0; + /// Held in a shared atomic, not a plain local: the Pool can outlive this stack frame (a background + /// publish holds `shared_from_this()`), so a by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); PoolConfig config; config.cas_request_budget = budget; - config.boot_ms_fn = [&fake_boot] { return fake_boot; }; - config.wait_sleep_fn = [&fake_boot](uint64_t ms) { fake_boot += ms; }; + config.boot_ms_fn = [fake_boot] { return fake_boot->load(); }; + config.wait_sleep_fn = [fake_boot](uint64_t ms) { *fake_boot += ms; }; auto store = openPoolWithConfig(backend, config); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/remount_midflush"}; @@ -3787,7 +4013,7 @@ TEST(CASRefWriterNamespaceRemoval, CachedPositiveWriterCannotAppendAfterRemoving const RootNamespace ns{"srv1/removing_blocks_cached_writer"}; publishEmptyPart(store, ns, "existing"); - const CasRefCatalog::Snapshot before = CasRefCatalog::read(*backend, layout); + const CasRefCatalog::Snapshot before = readCatalogForTest(backend, layout); const auto observed = std::find_if(before.catalog.entries.begin(), before.catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; }); ASSERT_NE(observed, before.catalog.entries.end()); @@ -3831,7 +4057,9 @@ TEST(CASRefWriterNamespaceRemoval, CachedPositiveWriterCannotAppendAfterRemoving } if (writer_parked) { - CasRefCatalog::casUpdate(*backend, layout, [&](const RefCatalog & current) + CasRequests catalog_requests(backend, Fence::open()); + CasOperation catalog_op = catalog_requests.admit(); + CasRefCatalog::casUpdate(catalog_op, layout, [&](const RefCatalog & current) { RefCatalog next = current; const auto it = std::find(next.entries.begin(), next.entries.end(), exact_live); @@ -3882,7 +4110,7 @@ TEST(CASRefWriterNamespaceRemoval, TxnNamesEveryOwnerThenRemoveNamespace) String cursor; for (;;) { - const ListPage page = backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = listForTest(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); for (const ListedKey & lk : page.keys) { const auto parsed = layout.parseRefObjectKey(lk.key); @@ -3896,7 +4124,7 @@ TEST(CASRefWriterNamespaceRemoval, TxnNamesEveryOwnerThenRemoveNamespace) } } ASSERT_TRUE(newest_log.has_value()); - const auto got = backend->get(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest_log)); + const auto got = readOf(backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), *newest_log)); ASSERT_TRUE(got.has_value()); const RefLogTxn removal_txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), *newest_log); @@ -3933,12 +4161,12 @@ TEST(CASRefWriterNamespaceRemoval, RemovalPublishesTerminalLogWithoutTerminalSna << "the terminal transaction remains ordinary immutable stream work until GC folds it"; size_t terminal_logs = 0; - for (const ListedKey & listed : backend->list(layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), "", 1000).keys) + for (const ListedKey & listed : listForTest(backend, layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), "", 1000).keys) { const auto parsed = layout.parseRefObjectKey(listed.key); if (!parsed || parsed->kind != RefObjectKind::Log) continue; - const auto got = backend->get(listed.key); + const auto got = readOf(backend, listed.key); ASSERT_TRUE(got.has_value()); const RefLogTxn txn = decodeRefLogTxn(openObject(FormatId::RefLog, got->bytes), ns.string(), parsed->txn_id); if (!txn.ops.empty() && txn.ops.back().kind == RefOpKind::RemoveNamespace) @@ -3947,7 +4175,7 @@ TEST(CASRefWriterNamespaceRemoval, RemovalPublishesTerminalLogWithoutTerminalSna EXPECT_EQ(terminal_logs, 1u); } -/// Review fix (prerequisite to this task's dropNamespace rewiring): `flushRefBatch`'s per-item +/// `flushRefBatch`'s per-item /// validation previously previewed each op as its OWN single-op trial transaction, so a /// whole-transaction-shape rule ("remove_namespace must be the FINAL op") trivially passed on every /// singleton slice regardless of an item's REAL combined shape -- a malformed item would only have @@ -3995,7 +4223,7 @@ TEST(CASRefWriterNamespaceRemoval, GenericAppendCannotWriteTerminalWhileCatalogI const Layout & layout = store->layout(); const RootNamespace ns{"srv1/unauthorized_terminal"}; publishEmptyPart(store, ns, "owned"); - const CatalogEntry live = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry live = catalogEntryOrThrow(backend, layout, ns); ASSERT_EQ(live.state, NsState::Live); const auto greatest_before = listGreatestLogIdForLifeForTest( *backend, layout, NamespaceLifeId::fromCatalogEntry(ns, live.incarnation)); @@ -4024,7 +4252,7 @@ TEST(CASRefWriterNamespaceRemoval, GenericAppendCannotWriteTerminalWhileCatalogI /*skip_stale_precommit_sweep=*/true); }); - EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns), live); + EXPECT_EQ(catalogEntryOrThrow(backend, layout, ns), live); EXPECT_EQ(listGreatestLogIdForLifeForTest( *backend, layout, NamespaceLifeId::fromCatalogEntry(ns, live.incarnation)), greatest_before) << "an unauthorized terminal must allocate no id and create no ref-log object"; @@ -4039,7 +4267,7 @@ TEST(CASRefWriterNamespaceRemoval, GenericTerminalOnAbsentNamePerformsZeroDurabl auto backend = std::make_shared(); auto store = openPool(backend); const RootNamespace ns{"srv1/absent_unauthorized_terminal"}; - const CasRefCatalog::Snapshot catalog_before = CasRefCatalog::read(*backend, store->layout()); + const CasRefCatalog::Snapshot catalog_before = readCatalogForTest(backend, store->layout()); backend->resetCounts(); expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] @@ -4057,10 +4285,10 @@ TEST(CASRefWriterNamespaceRemoval, GenericTerminalOnAbsentNamePerformsZeroDurabl EXPECT_EQ(backend->putTotal(), 0u); EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); EXPECT_EQ(backend->deleteTotal(), 0u); - const CasRefCatalog::Snapshot catalog_after = CasRefCatalog::read(*backend, store->layout()); - EXPECT_EQ(catalog_after.token, catalog_before.token); + const CasRefCatalog::Snapshot catalog_after = readCatalogForTest(backend, store->layout()); + EXPECT_EQ(catalog_after.etag, catalog_before.etag); EXPECT_EQ(catalog_after.catalog, catalog_before.catalog); EXPECT_FALSE(store->refTableLifeForTest(ns)); } @@ -4078,17 +4306,17 @@ TEST(CASRefWriterNamespaceRemoval, CatalogedNamespaceFilesOnlyLifeCompletesRemov const RootNamespace ns{"srv1/files_only"}; const NamespaceLifeId life = store->namespaceLife(ns); store->putNamespaceFile(life, "format_version.txt", "1\n"); - ASSERT_TRUE(backend->list(layout.namespaceStreamPrefix(life), "", 100).keys.empty()); - ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Live); + ASSERT_TRUE(listForTest(backend, layout.namespaceStreamPrefix(life), "", 100).keys.empty()); + ASSERT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Live); EXPECT_NO_THROW(store->dropNamespace(ns)); - ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + ASSERT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Removing); - const ListPage terminal_page = backend->list(layout.namespaceStreamPrefix(life), "", 100); + const ListPage terminal_page = listForTest(backend, layout.namespaceStreamPrefix(life), "", 100); ASSERT_EQ(terminal_page.keys.size(), 1u); const auto parsed = layout.parseRefObjectKey(terminal_page.keys.front().key); ASSERT_TRUE(parsed); - const auto terminal_body = backend->get(terminal_page.keys.front().key); + const auto terminal_body = readOf(backend, terminal_page.keys.front().key); ASSERT_TRUE(terminal_body); const RefLogTxn terminal = decodeRefLogTxn( openObject(FormatId::RefLog, terminal_body->bytes), ns.string(), parsed->txn_id); @@ -4099,7 +4327,7 @@ TEST(CASRefWriterNamespaceRemoval, CatalogedNamespaceFilesOnlyLifeCompletesRemov Gc gc(store, UInt128{181}); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); (void)runRegularRoundReclaiming(gc); - const RefCatalog after = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog after = readCatalogForTest(backend, layout).catalog; EXPECT_TRUE(std::none_of(after.entries.begin(), after.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; @@ -4115,19 +4343,19 @@ TEST(CASRefWriterNamespaceRemoval, PredurableCatalogReadFailureReopensExactLiveL const Layout & layout = store->layout(); const RootNamespace ns{"srv1/predurable_read_failure"}; publishEmptyPart(store, ns, "owned"); - const CatalogEntry live = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry live = catalogEntryOrThrow(backend, layout, ns); backend->catalog_fault_key = layout.refCatalogKey(); backend->catalog_gets_before_fault = 1; /// initial discovery succeeds; post-close observation fails backend->catalog_get_fault_count = 1; EXPECT_THROW(store->dropNamespace(ns), std::runtime_error); - EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns), live); + EXPECT_EQ(catalogEntryOrThrow(backend, layout, ns), live); EXPECT_NO_THROW(store->updateRefPublishedAt(ns, "owned", [](RefPublishedAtUpdate & update) { update.published_at_ms = 17; })) << "a fresh exact Live observation must reopen the lane after a pre-durable failure"; - EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Live); + EXPECT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Live); } /// spec §Namespace Removal (writer, line 666): "After the transaction is durable, it applies the same @@ -4168,14 +4396,11 @@ TEST(CASRefWriterNamespaceRemoval, DropNamespaceCancelsInFlightBuildAndNextOpThr /// remains `Removing`, positive ownership is refused, and a retry of the same removal resolves the wedge. TEST(CASRefWriterNamespaceRemoval, RemovalAppendFailureLeavesRemovingAndRetryCompletes) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/remove_fault_keeps_build"}; @@ -4186,11 +4411,9 @@ TEST(CASRefWriterNamespaceRemoval, RemovalAppendFailureLeavesRemovingAndRetryCom build->precommitAdd(ns, "inflight", id); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; + driveToTheWedge(*clock, *backend, [&] { store->dropNamespace(ns); }); - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); - - EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + EXPECT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Removing); EXPECT_FALSE(store->resolveRef(ns, "committed")) << "a fresh name lookup must not expose a catalog-Removing life"; /// The build was NOT cancelled: a non-append operation (`stageManifest` -- it never touches the now @@ -4211,14 +4434,11 @@ TEST(CASRefWriterNamespaceRemoval, RemovalAppendFailureLeavesRemovingAndRetryCom /// `RemovalAppendFailureLeavesRemovingAndRetryCompletes`. TEST(CASRefWriterNamespaceRemoval, PresenceProbeStaysTrueThroughRemovingUntilTerminalRetrySucceeds) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/presence_removing_no_terminal"}; @@ -4226,10 +4446,9 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeStaysTrueThroughRemovingUntilTer EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + driveToTheWedge(*clock, *backend, [&] { store->dropNamespace(ns); }); - ASSERT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Removing); + ASSERT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Removing); EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)) << "the catalog transitioned but the terminal append never landed -- cleanup is unproven"; @@ -4253,7 +4472,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeCreatingIsPresentAndRemovalWaits entry.state = NsState::Creating; entry.incarnation = UInt128(99); entry.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, entry); + casAdmitEntryForTest(backend, layout, 1, entry); EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); @@ -4261,7 +4480,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeCreatingIsPresentAndRemovalWaits /// dead (absence proves nothing), so removal fails closed rather than cancelling a `Creating` row a /// live writer might still publish into. expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); - EXPECT_EQ(catalogEntryOrThrow(*backend, layout, ns).state, NsState::Creating); + EXPECT_EQ(catalogEntryOrThrow(backend, layout, ns).state, NsState::Creating); EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)); /// Publish a GC-fenced lease for the SAME server root -- one of `isCreatorFenceTerminal`'s accepted @@ -4272,7 +4491,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeCreatingIsPresentAndRemovalWaits dead.gc_fenced = true; dead.seq = 1; dead.write_attempt_id = UInt128{1}; - backend->putIfAbsent(layout.mountKey("srv1"), encodeMountLease(dead)); + createRaw(backend, layout.mountKey("srv1"), encodeMountLease(dead)); EXPECT_NO_THROW(store->dropNamespace(ns)); EXPECT_FALSE(store->namespaceStillLogicallyPresent(ns)); @@ -4325,7 +4544,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeNoRowObservationRevalidatesRathe born.state = NsState::Creating; born.incarnation = UInt128(1234); born.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, born); + casAdmitEntryForTest(backend, layout, 1, born); { std::lock_guard lock(mutex); @@ -4390,7 +4609,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeIgnoresUnrelatedCatalogChurnBetw born.state = NsState::Creating; born.incarnation = UInt128(5678); born.creator = CreatorFence{.server_root_id = "srv1", .writer_epoch = store->liveWriterEpoch(), .fence_generation = 1}; - CasRefCatalog::casAdmitEntry(*backend, layout, 1, born); + casAdmitEntryForTest(backend, layout, 1, born); { std::lock_guard lock(mutex); @@ -4429,12 +4648,16 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeFenceLossPropagatesRatherThanAns publishEmptyPart(store, ns, "x"); const String mount_key = store->layout().mountKey("test"); - const auto got = backend->get(mount_key); + const auto got = readOf(backend, mount_key); ASSERT_TRUE(got); MountLease foreign = decodeMountLease(got->bytes); foreign.server_uuid = foreign.server_uuid + UInt128{1}; foreign.seq += 1; - ASSERT_EQ(backend->putOverwrite(mount_key, encodeMountLease(foreign), got->token).outcome, PutOutcome::Done); + { + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.replace(mount_key, encodeMountLease(foreign), got->etag, Retry::standard()))); + } store->tripMountLost(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->namespaceStillLogicallyPresent(ns); }); @@ -4445,14 +4668,11 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeFenceLossPropagatesRatherThanAns /// (absent, immediately, no GC). TEST(CASRefWriterNamespaceRemoval, PresenceProbeFacadeConsistencyAcrossRemovalLifecycle) { - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = kSingleAttemptDeadlineMs; - budget.lease_safety_margin_ms = 100; + const CasRequestBudget budget = wedgeTestBudget(); auto backend = std::make_shared(); auto store = openPool(backend, budget); + auto clock = VirtualRetryClock::installOn(store); const Layout & layout = store->layout(); const RootNamespace ns{"srv1/presence_facade_consistency"}; @@ -4461,8 +4681,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeFacadeConsistencyAcrossRemovalLi EXPECT_FALSE(store->listRefs(ns).empty()); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropNamespace(ns); }); + driveToTheWedge(*clock, *backend, [&] { store->dropNamespace(ns); }); EXPECT_TRUE(store->namespaceStillLogicallyPresent(ns)) << "present for cleanup, even though content below is about to prove unreadable"; @@ -4496,7 +4715,7 @@ TEST(CASRefWriterNamespaceRemoval, PresenceProbeRevalidatesAfterTerminalProvenRa const auto catalog_entry = [&]() -> std::optional { - const RefCatalog catalog = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog catalog = readCatalogForTest(backend, layout).catalog; const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; @@ -4679,7 +4898,7 @@ TEST(CASRefWriterNamespaceRemoval, CreateAgainstRemovingRetriesWithoutMutation) expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)store->namespaceLife(ns); }); EXPECT_EQ(backend->putTotal(), 0u); EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); } auto fresh_store = openPool(backend); @@ -4687,7 +4906,7 @@ TEST(CASRefWriterNamespaceRemoval, CreateAgainstRemovingRetriesWithoutMutation) expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)fresh_store->namespaceLife(ns); }); EXPECT_EQ(backend->putTotal(), 0u); EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); } /// Same-name rebirth must not inherit the predecessor's physical life or folded cursor even when the @@ -4716,7 +4935,7 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi }; const auto catalog_entry = [&]() -> std::optional { - const RefCatalog catalog = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog catalog = readCatalogForTest(backend, layout).catalog; const auto it = std::find_if(catalog.entries.begin(), catalog.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns; @@ -4733,8 +4952,8 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi Gc gc(store, gc_id); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); - GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); - CasFoldSeal seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + GcState state = decodeGcState(readOf(backend, layout.gcStateKey())->bytes); + CasFoldSeal seal = decodeFoldSeal(readOf(backend, layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); const auto predecessor_row = seal.ref_lives.find(predecessor.incarnation); ASSERT_NE(predecessor_row, seal.ref_lives.end()); ASSERT_NE(predecessor_row->second.coverage.last_folded_ref_id, RefTxnId{}); @@ -4744,7 +4963,7 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi ASSERT_GT(backend->putTotal(), puts_before_drop) << "control: the real removal call returned after durably writing its terminal artifacts"; std::optional terminal_id; - for (const ListedKey & listed : backend->list( + for (const ListedKey & listed : listForTest(backend, layout.namespaceStreamPrefix(NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation)), "", 1000).keys) { @@ -4754,7 +4973,7 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi terminal_id = parsed->txn_id; } ASSERT_TRUE(terminal_id.has_value()); - const auto terminal_body = backend->get(layout.refLogKey( + const auto terminal_body = readOf(backend, layout.refLogKey( NamespaceLifeId::fromCatalogEntry(ns, predecessor.incarnation), *terminal_id)); ASSERT_TRUE(terminal_body.has_value()); const RefLogTxn terminal = decodeRefLogTxn( @@ -4771,8 +4990,8 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi << "the removal path must invalidate the resident runtime's life, not pass through eviction"; ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred) << "the terminal delta must fold"; - state = decodeGcState(backend->get(layout.gcStateKey())->bytes); - seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + state = decodeGcState(readOf(backend, layout.gcStateKey())->bytes); + seal = decodeFoldSeal(readOf(backend, layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); ASSERT_TRUE(seal.ref_lives.at(predecessor.incarnation).cleanup_evidence.has_value()); const RoundReport drain = runRegularRoundReclaiming(gc); @@ -4791,7 +5010,7 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi ASSERT_EQ(store->refTableLifeForTest(ns)->incarnation, successor.incarnation); const NamespaceLifeId successor_life = NamespaceLifeId::fromCatalogEntry(ns, successor.incarnation); - const ListPage successor_stream = backend->list(layout.namespaceStreamPrefix(successor_life), "", 1000); + const ListPage successor_stream = listForTest(backend, layout.namespaceStreamPrefix(successor_life), "", 1000); ASSERT_FALSE(successor_stream.keys.empty()) << "the real successor writer produced foldable stream work"; std::vector successor_phases; gc.setPhaseSink([&](const GcPhaseRecord & phase) { successor_phases.push_back(phase); }); @@ -4805,8 +5024,8 @@ TEST(CASRefWriterNamespaceRemoval, SameNameSameWriterEpochRebirthInvalidatesResi << "successor stream keys=" << successor_stream.keys.size() << ", changed_shards=" << decision->metrics.at("changed_shards") << ", dead_life_debris=" << decision->metrics.at("dead_life_debris"); - state = decodeGcState(backend->get(layout.gcStateKey())->bytes); - seal = decodeFoldSeal(backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + state = decodeGcState(readOf(backend, layout.gcStateKey())->bytes); + seal = decodeFoldSeal(readOf(backend, layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); EXPECT_FALSE(seal.ref_lives.contains(predecessor.incarnation)); const auto successor_row = seal.ref_lives.find(successor.incarnation); ASSERT_NE(successor_row, seal.ref_lives.end()); @@ -4830,13 +5049,13 @@ TEST(CASRefWriterNamespaceRemoval, CommitThenThrowEraseResolvesAndRebindsResiden const Layout & layout = store->layout(); const RootNamespace ns{"srv1/removal-erase-lost-response"}; Gc gc(store, UInt128{101}); - const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, backend, ns, gc); backend->catalog_fault_key = layout.refCatalogKey(); backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; EXPECT_NO_THROW((void)runRegularRoundReclaiming(gc)); - const RefCatalog after_erase = CasRefCatalog::read(*backend, layout).catalog; + const RefCatalog after_erase = readCatalogForTest(backend, layout).catalog; EXPECT_TRUE(std::none_of(after_erase.entries.begin(), after_erase.entries.end(), [&](const CatalogEntry & entry) { return entry.ns == ns && entry.incarnation == ready.predecessor.incarnation; @@ -4844,15 +5063,15 @@ TEST(CASRefWriterNamespaceRemoval, CommitThenThrowEraseResolvesAndRebindsResiden EXPECT_EQ(store->refTableRuntimeIdentityForTest(ns), 0u); EXPECT_NO_THROW(publishWithProductionBirth(store, ns, "successor")); - const CatalogEntry successor = catalogEntryOrThrow(*backend, layout, ns); + const CatalogEntry successor = catalogEntryOrThrow(backend, layout, ns); EXPECT_NE(successor.incarnation, ready.predecessor.incarnation); EXPECT_EQ(store->liveWriterEpoch(), ready.writer_epoch); EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); ASSERT_FALSE(runRegularRoundReclaiming(gc).deferred); - const GcState state = decodeGcState(backend->get(layout.gcStateKey())->bytes); + const GcState state = decodeGcState(readOf(backend, layout.gcStateKey())->bytes); const CasFoldSeal seal = decodeFoldSeal( - backend->get(layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); + readOf(backend, layout.foldSealKey(state.snap_generation, state.snap_attempt))->bytes); EXPECT_FALSE(seal.ref_lives.contains(ready.predecessor.incarnation)); ASSERT_TRUE(seal.ref_lives.contains(successor.incarnation)); EXPECT_EQ(seal.ref_lives.at(successor.incarnation).coverage.last_folded_ref_id, @@ -4871,7 +5090,7 @@ TEST(CASRefWriterNamespaceRemoval, OtherWinnerReplacementInvalidatesExactPredece const Layout & layout = store->layout(); const RootNamespace ns{"srv1/removal-other-winner-replacement"}; Gc gc(store, UInt128{102}); - const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, backend, ns, gc); const CatalogEntry replacement{ .ns = ns, @@ -4880,10 +5099,10 @@ TEST(CASRefWriterNamespaceRemoval, OtherWinnerReplacementInvalidatesExactPredece .creator = std::nullopt}; ASSERT_NE(replacement.incarnation, ready.predecessor.incarnation); const NamespaceLifeId replacement_life = NamespaceLifeId::fromCatalogEntry(ns, replacement.incarnation); - ASSERT_EQ(backend->putIfAbsent(layout.refCkptKey(replacement_life), encodeRefCkpt(RefCkpt{ + ASSERT_TRUE(createRaw(backend, layout.refCkptKey(replacement_life), encodeRefCkpt(RefCkpt{ .life_epoch = ready.writer_epoch, .checkpoint_snapshot_id = std::nullopt, - .last_epoch_seal = std::nullopt})).outcome, PutOutcome::Done); + .last_epoch_seal = std::nullopt}))); backend->catalog_fault_key = layout.refCatalogKey(); backend->catalog_replacement_bytes = encodeRefCatalog(RefCatalog{.entries = {replacement}}); backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::OtherWriterReplacement; @@ -4905,13 +5124,13 @@ TEST(CASRefWriterNamespaceRemoval, LaterNameLookupReconcilesAfterEraseResolution const Layout & layout = store->layout(); const RootNamespace ns{"srv1/removal-resolution-read-failure-lookup"}; Gc gc(store, UInt128{103}); - const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, backend, ns, gc); backend->catalog_fault_key = layout.refCatalogKey(); backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; backend->catalog_resolution_get_fault_count = 1; EXPECT_THROW((void)runRegularRoundReclaiming(gc), std::runtime_error); - EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + EXPECT_TRUE(readCatalogForTest(backend, layout).catalog.entries.empty()); std::optional successor; EXPECT_NO_THROW(successor = store->namespaceLife(ns)); @@ -4932,17 +5151,17 @@ TEST(CASRefWriterNamespaceRemoval, PostListCatalogCutReconcilesMissedEraseInvali const Layout & layout = store->layout(); const RootNamespace ns{"srv1/removal-resolution-read-failure-post-list"}; Gc gc(store, UInt128{104}); - const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, *backend, ns, gc); + const CompletedRemovingFixture ready = prepareResidentRemovalForDrain(store, backend, ns, gc); backend->catalog_fault_key = layout.refCatalogKey(); backend->catalog_cas_fault = RefWriterTestBackend::CatalogCasFault::CommitThenThrow; backend->catalog_resolution_get_fault_count = 1; EXPECT_THROW((void)runRegularRoundReclaiming(gc), std::runtime_error); - EXPECT_TRUE(CasRefCatalog::read(*backend, layout).catalog.entries.empty()); + EXPECT_TRUE(readCatalogForTest(backend, layout).catalog.entries.empty()); EXPECT_NO_THROW((void)runRegularRoundReclaiming(gc)); EXPECT_NO_THROW(publishWithProductionBirth(store, ns, "successor")); - EXPECT_NE(catalogEntryOrThrow(*backend, layout, ns).incarnation, ready.predecessor.incarnation); + EXPECT_NE(catalogEntryOrThrow(backend, layout, ns).incarnation, ready.predecessor.incarnation); EXPECT_NE(store->refTableRuntimeIdentityForTest(ns), ready.runtime_identity); } @@ -4957,7 +5176,7 @@ TEST(CASRefWriterNamespaceBirth, ExistingLiveCatalogRowPinsExactLifeWithoutMutat auto backend = std::make_shared(); auto store = openPool(backend); const RootNamespace ns{"srv1/existing-live-assignment"}; - CasRefCatalog::casAdmitEntry(*backend, store->layout(), 1, CatalogEntry{ + casAdmitEntryForTest(backend, store->layout(), 1, CatalogEntry{ .ns = ns, .state = NsState::Live, .incarnation = UInt128{41}}); DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, store->layout(), ns, RefCkpt{ .life_epoch = store->liveWriterEpoch(), @@ -4973,7 +5192,7 @@ TEST(CASRefWriterNamespaceBirth, ExistingLiveCatalogRowPinsExactLifeWithoutMutat EXPECT_EQ(store->refTableLifeForTest(ns)->incarnation, UInt128{41}); EXPECT_EQ(backend->putTotal(), 0u); EXPECT_EQ(backend->putOverwriteTotal(), 0u); - EXPECT_EQ(backend->casPutTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); } /// A read of a never-born name may observe the catalog, but it must not allocate the local name slot @@ -5066,17 +5285,17 @@ namespace /// under `ns` -- epoch 1 births ref "a", epoch 2 adds ref "b" -- with no snapshot, and burns the /// durable epoch counter to exactly 2 so a subsequent `Pool::open` allocates epoch 3 (both dead /// epochs land strictly below the fresh writer's own, as `dead_region_nonempty` requires). -void seedSealFixtureDeadEpochs(Backend & backend, const Layout & layout, const RootNamespace & ns) +void seedSealFixtureDeadEpochs(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) { - allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); /// burns epoch 1 - allocateWriterEpoch(backend, layout, "test", EpochMintPolicy::NormalMount, 0, [] { return RefCatalog{}; }); /// burns epoch 2 + allocateWriterEpochForTest(backend, layout, "test"); /// burns epoch 1 + allocateWriterEpochForTest(backend, layout, "test"); /// burns epoch 2 RefLogTxn birth; birth.ns = ns.string(); birth.txn_id = RefTxnId{1, 1}; birth.ops = {namespaceBirthOp(), publishCommittedOps("a", manifestRef(1, 1, 1))[0], publishCommittedOps("a", manifestRef(1, 1, 1))[1]}; - DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, birth); + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, birth); RefLogTxn mut; mut.ns = ns.string(); @@ -5087,8 +5306,8 @@ void seedSealFixtureDeadEpochs(Backend & backend, const Layout & layout, const R mut.prev_epoch_seal = RefTxnId{1, 2}; mut.ops = {publishCommittedOps("b", manifestRef(2, 1, 1))[0], publishCommittedOps("b", manifestRef(2, 1, 1))[1]}; - DB::Cas::tests::fixture::writeRefLogRaw(backend, layout, mut); - DB::Cas::tests::writeRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ + DB::Cas::tests::fixture::writeRefLogRaw(*backend, layout, mut); + DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ /// The namespace was born in epoch 1 and only `{1,1}` is fronted initially. Recovery must mint /// the missing required seal `{1,2}` before it may adopt the already durable `{2,1}` successor. .life_epoch = 1, @@ -5103,10 +5322,12 @@ void seedSealFixtureDeadEpochs(Backend & backend, const Layout & layout, const R /// prior is an immediate certificate of death (`claimMountAwaitingExpiry` reclaims it on its FIRST /// attempt, no observation polling), so a fake-clocked successor `Pool::open` above it becomes /// unclean deterministically, without any real sleep. -void seedUncleanPredecessorMount(Backend & backend, const Layout & layout, uint64_t epoch) +void seedUncleanPredecessorMount(const BackendPtr & backend, const Layout & layout, uint64_t epoch) { - claimMount(backend, layout, "test", UInt128(1), epoch, /*now_ms=*/1000, /*ttl_ms=*/500); - fenceOutRefMount(backend, layout.mountKey("test")); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + claimMount(op, layout, "test", UInt128(1), epoch, /*now_ms=*/1000, /*ttl_ms=*/500); + fenceOutRefMount(*backend, layout.mountKey("test")); } /// The budget every seal test's successor `Pool::open` uses: a 500ms lease TTL needs a scaled-down @@ -5114,9 +5335,7 @@ void seedUncleanPredecessorMount(Backend & backend, const Layout & layout, uint6 /// lease TTL) -- mirrors `CASMountOpenWaits.FencedPriorReclaimsWithoutAnyWait` exactly. CasRequestBudget sealTestTinyBudget() { - return CasRequestBudget{ - .attempt_timeout_ms = 50, .operation_deadline_ms = kSingleAttemptDeadlineMs, .max_attempts = 1, - .lease_safety_margin_ms = 50}; + return CasRequestBudget{.attempt_timeout_ms = 50, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = std::nullopt}; } } @@ -5154,8 +5373,8 @@ TEST(CASRefWriterRecoveryRetry, TransientSealFailureIsRetriedThenSucceeds) const Layout layout("p"); const RootNamespace ns{"srv1/retry_ok"}; - seedSealFixtureDeadEpochs(*backend, layout, ns); - seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + seedSealFixtureDeadEpochs(backend, layout, ns); + seedUncleanPredecessorMount(backend, layout, /*epoch=*/2); uint64_t fake_now = 1'000'000; @@ -5166,33 +5385,40 @@ TEST(CASRefWriterRecoveryRetry, TransientSealFailureIsRetriedThenSucceeds) config.cas_request_budget.recovery_retry_budget_ms = 120000; config.cas_request_budget.recovery_retry_initial_backoff_ms = 1000; config.cas_request_budget.recovery_retry_max_backoff_ms = 30000; - config.boot_ms_fn = [&fake_now] { return fake_now; }; + /// Captured by value: `fake_now` stays frozen for the whole test (see below), and the Pool can + /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference + /// capture would dangle. + config.boot_ms_fn = [fake_now] { return fake_now; }; config.wait_sleep_fn = [](uint64_t) {}; auto store = openPoolWithConfig(backend, config); ASSERT_TRUE(store); ASSERT_EQ(store->liveWriterEpoch(), 3u); - /// No-op backoff and a frozen clock: retries run until the transient faults are exhausted, and the - /// frozen clock keeps the mount fence alive across them (advancing it past the tiny lease TTL would - /// drop the fence and abort recovery -- exercising the fence path, which is the budget test's job). - store->setCasRetrySleepForTest([](uint64_t) {}); - - /// Fail the epoch seal's conditional create twice with a transient (timeout) error; the third - /// attempt lands. The seal is a LOG transaction at `{2,2}` -- the slot after the dead epoch's last - /// durable id -- because INV-2 closes an epoch in-band, at the key a straggler would have taken. + /// The engine's own clock advances (its sleep is what moves it), while `fake_now` -- the FENCE's + /// clock -- stays frozen, which is what keeps the mount alive across a retry: advancing it past the + /// tiny lease TTL would drop the fence and abort recovery, exercising the fence path instead. + VirtualRetryClock::installOn(store); + + /// Fail the epoch seal's conditional create for the whole of ONE recovery attempt, then clear the + /// fault from the recovery retry seam -- the only point between two recovery attempts a test can + /// reach. A bounded fault count cannot express this: the write engine reissues within a call, so a + /// count of two is spent by that one call's own reissues and no recovery retry ever happens. The + /// seal is a LOG transaction at `{2,2}` -- the slot after the dead epoch's last durable id -- + /// because INV-2 closes an epoch in-band, at the key a straggler would have taken. const RefTxnId seal_id{2, 2}; backend->fault_key_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), seal_id); - backend->fault_count = 2; + backend->fault_latched = true; + store->setRefRecoveryRetrySleepForTest([&backend](uint64_t, const auto &) { backend->disarmFaults(); }); using ProfileEvents::global_counters; const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); - EXPECT_EQ(store->listRefs(ns).size(), 2u) << "recovery must succeed after retrying past the faults"; + EXPECT_EQ(store->listRefs(ns).size(), 2u) << "recovery must succeed after retrying past the fault"; - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before + 2); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before + 1); /// TWO dead epochs (1 and 2) are closed by this walk, and a whole attempt is re-driven per transient - /// failure -- so the seals of the epochs a failed attempt already closed are ADOPTED on the retry + /// failure -- so the seals of the epochs the failed attempt already closed are ADOPTED on the retry /// rather than minted again. Exactly two are minted in total. EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2); } @@ -5205,8 +5431,8 @@ TEST(CASRefWriterRecoveryRetry, RecoveryDoesNotEnumerateItsStream) const Layout layout("p"); const RootNamespace ns{"srv1/retry_list"}; - seedSealFixtureDeadEpochs(*backend, layout, ns); - seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + seedSealFixtureDeadEpochs(backend, layout, ns); + seedUncleanPredecessorMount(backend, layout, /*epoch=*/2); uint64_t fake_now = 1'000'000; @@ -5215,7 +5441,10 @@ TEST(CASRefWriterRecoveryRetry, RecoveryDoesNotEnumerateItsStream) config.mount_lease_ttl_ms = std::chrono::milliseconds(500); config.cas_request_budget = sealTestTinyBudget(); config.cas_request_budget.recovery_retry_budget_ms = 120000; - config.boot_ms_fn = [&fake_now] { return fake_now; }; + /// Captured by value: `fake_now` is never mutated in this test, and the Pool can outlive this + /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture would + /// dangle. + config.boot_ms_fn = [fake_now] { return fake_now; }; config.wait_sleep_fn = [](uint64_t) {}; auto store = openPoolWithConfig(backend, config); ASSERT_TRUE(store); @@ -5245,10 +5474,13 @@ TEST(CASRefWriterRecoveryRetry, TransientFailureLongerThanBudgetPropagates) const Layout layout("p"); const RootNamespace ns{"srv1/retry_budget"}; - seedSealFixtureDeadEpochs(*backend, layout, ns); - seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + seedSealFixtureDeadEpochs(backend, layout, ns); + seedUncleanPredecessorMount(backend, layout, /*epoch=*/2); - uint64_t fake_now = 1'000'000; + /// Held in a shared atomic, not a plain local: this test mutates the clock via the retry-sleep + /// hook below, and the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto fake_now = std::make_shared>(1'000'000); PoolConfig config; config.server_id = UInt128(1); @@ -5259,11 +5491,11 @@ TEST(CASRefWriterRecoveryRetry, TransientFailureLongerThanBudgetPropagates) config.cas_request_budget.recovery_retry_budget_ms = 5000; /// small, deterministic config.cas_request_budget.recovery_retry_initial_backoff_ms = 1000; config.cas_request_budget.recovery_retry_max_backoff_ms = 30000; - config.boot_ms_fn = [&fake_now] { return fake_now; }; + config.boot_ms_fn = [fake_now] { return fake_now->load(); }; config.wait_sleep_fn = [](uint64_t) {}; auto store = openPoolWithConfig(backend, config); ASSERT_TRUE(store); - store->setCasRetrySleepForTest([&fake_now](uint64_t ms) { fake_now += ms; }); + store->setCasRetrySleepForTest([fake_now](uint64_t ms) { *fake_now += ms; }); /// The seal is an in-band LOG transaction at the slot after the dead epoch's last durable id, not a /// snapshot at a synthetic id: epoch 1 closes at `{1,2}`, which is the FIRST write the walk attempts. @@ -5280,8 +5512,8 @@ TEST(CASRefWriterRecoveryRetry, NonNetworkErrorIsNotRetried) const Layout layout("p"); const RootNamespace ns{"srv1/retry_fatal"}; - seedSealFixtureDeadEpochs(*backend, layout, ns); - seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + seedSealFixtureDeadEpochs(backend, layout, ns); + seedUncleanPredecessorMount(backend, layout, /*epoch=*/2); PoolConfig config; config.server_id = UInt128(1); @@ -5291,8 +5523,9 @@ TEST(CASRefWriterRecoveryRetry, NonNetworkErrorIsNotRetried) auto store = openPoolWithConfig(backend, config); ASSERT_TRUE(store); - size_t sleep_calls = 0; - store->setCasRetrySleepForTest([&sleep_calls](uint64_t) { ++sleep_calls; }); + /// Owned by the closure: the pool may outlive this frame and retry a farewell request. + auto sleep_calls = std::make_shared>(0); + store->setCasRetrySleepForTest([sleep_calls](uint64_t) { ++*sleep_calls; }); /// A foreign writer lands DIFFERENT valid bytes at the seal key; resolve-before-reissue then throws /// CORRUPTED_DATA (a real cross-process seal conflict), which must NOT be retried. @@ -5303,7 +5536,7 @@ TEST(CASRefWriterRecoveryRetry, NonNetworkErrorIsNotRetried) expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); EXPECT_EQ(backend->corrupt_count, 0) << "the test must reach the injected foreign seal conflict, not fail on fixture validation"; - EXPECT_EQ(sleep_calls, 0u) << "a non-transient error must fail fast with zero backoff sleeps"; + EXPECT_EQ(sleep_calls->load(), 0u) << "a non-transient error must fail fast with zero backoff sleeps"; } TEST(CASRefWriterRecoveryRetry, VanishBrakeStaysTerminalNotRetried) @@ -5332,8 +5565,9 @@ TEST(CASRefWriterRecoveryRetry, VanishBrakeStaysTerminalNotRetried) auto store = openPool(backend); - size_t sleep_calls = 0; - store->setCasRetrySleepForTest([&sleep_calls](uint64_t) { ++sleep_calls; }); + /// Owned by the closure: the pool may outlive this frame and retry a farewell request. + auto sleep_calls = std::make_shared>(0); + store->setCasRetrySleepForTest([sleep_calls](uint64_t) { ++*sleep_calls; }); /// A checkpoint-named snapshot belongs to the caller's immutable authority cut. If that exact /// object is absent, recovery must report corruption immediately; it must neither reinterpret a @@ -5350,7 +5584,7 @@ TEST(CASRefWriterRecoveryRetry, VanishBrakeStaysTerminalNotRetried) << "the test must reach the checkpoint-named snapshot GET, not fail on earlier fixture validation"; EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before) << "missing immutable checkpoint authority is terminal; the outer transient-retry loop must NOT re-drive it"; - EXPECT_EQ(sleep_calls, 0u) << "no backoff sleep for missing immutable checkpoint authority"; + EXPECT_EQ(sleep_calls->load(), 0u) << "no backoff sleep for missing immutable checkpoint authority"; } TEST(CASRefWriterRecoveryRetry, ThrowingBackoffSleepDoesNotWedgeRecovery) @@ -5363,8 +5597,8 @@ TEST(CASRefWriterRecoveryRetry, ThrowingBackoffSleepDoesNotWedgeRecovery) const Layout layout("p"); const RootNamespace ns{"srv1/retry_sleep_throw"}; - seedSealFixtureDeadEpochs(*backend, layout, ns); - seedUncleanPredecessorMount(*backend, layout, /*epoch=*/2); + seedSealFixtureDeadEpochs(backend, layout, ns); + seedUncleanPredecessorMount(backend, layout, /*epoch=*/2); PoolConfig config; config.server_id = UInt128(1); @@ -5375,11 +5609,14 @@ TEST(CASRefWriterRecoveryRetry, ThrowingBackoffSleepDoesNotWedgeRecovery) auto store = openPoolWithConfig(backend, config); ASSERT_TRUE(store); - /// First touch: the seal PUT fails transiently -> the loop enters backoff -> the sleep THROWS. - bool sleep_should_throw = true; - store->setCasRetrySleepForTest([&sleep_should_throw](uint64_t) + /// First touch: the seal create fails transiently and the next thing either loop does is sleep on + /// this one seam -- the write engine's reissue pause is simply the first to reach it -- so the throw + /// lands while `recovery_in_progress` is set, which is the state this test is about. + /// Owned by the closure: the pool may outlive this frame and retry a farewell request. + auto sleep_should_throw = std::make_shared>(true); + store->setCasRetrySleepForTest([sleep_should_throw](uint64_t) { - if (sleep_should_throw) + if (sleep_should_throw->load()) throw std::runtime_error("injected backoff-sleep failure"); }); const RefTxnId seal_id{1, 2}; @@ -5391,7 +5628,7 @@ TEST(CASRefWriterRecoveryRetry, ThrowingBackoffSleepDoesNotWedgeRecovery) /// The lane must NOT be wedged: with the fault now spent and the sleep no longer throwing, a second /// touch recovers cleanly. If recovery_in_progress had leaked (SCOPE_EXIT run unlocked / not run), a /// concurrent-safe second recovery would deadlock or mis-behave. - sleep_should_throw = false; + sleep_should_throw->store(false); EXPECT_EQ(store->listRefs(ns).size(), 2u) << "a second touch must recover; the retry lane is not wedged"; } diff --git a/src/Disks/tests/gtest_cas_repoint.cpp b/src/Disks/tests/gtest_cas_repoint.cpp index 9044a8eda895..fab72a86c59e 100644 --- a/src/Disks/tests/gtest_cas_repoint.cpp +++ b/src/Disks/tests/gtest_cas_repoint.cpp @@ -83,7 +83,7 @@ TEST(CASRepoint, AddFileRepoints) const RootNamespace ns{"srv/t1"}; DB::Cas::CachedPartFolderAccess access( store, {.cache_bytes = 64ULL << 20, .max_entries = 10000, .max_entry_bytes = 16ULL << 20, - .explain_enabled = false, .validate = {}}); + .explain_enabled = false}); const auto id_before = publishPart(store, ns, "part_1", {inlineEntry("checksums.txt", "cs")}); const DB::Cas::PartRefKey key{ns, "part_1"}; /// Warm the retained view so the erase-on-success cache discipline is actually exercised. diff --git a/src/Disks/tests/gtest_cas_request_control.cpp b/src/Disks/tests/gtest_cas_request_control.cpp deleted file mode 100644 index ad7801634f39..000000000000 --- a/src/Disks/tests/gtest_cas_request_control.cpp +++ /dev/null @@ -1,1494 +0,0 @@ -#include - -#include "config.h" - -#include -#include - -#include - -#if USE_AWS_S3 -#include -#include -#include -#include -#include -#endif - -using namespace DB::Cas; - -namespace DB::ErrorCodes -{ - extern const int NETWORK_ERROR; - extern const int ABORTED; -} - -namespace ProfileEvents -{ - extern const Event CASConditionalWriteAttempts; - extern const Event CASConditionalWriteCommitted; - extern const Event CASConditionalWriteDefiniteFailure; - extern const Event CASConditionalWriteUnresolved; -} - -#if USE_AWS_S3 -namespace DB::ErrorCodes -{ - extern const int CORRUPTED_DATA; - extern const int BAD_ARGUMENTS; - extern const int LOGICAL_ERROR; - extern const int UNKNOWN_EXCEPTION; -} -#endif - -/// The success path (buf.finalize() returned without throwing) is always Committed. No exception -/// object is needed — the caller distinguishes success from failure before calling either overload. -TEST(CASRequestControl, SuccessIsAlwaysCommitted) -{ - EXPECT_EQ(classifyConditionalWriteResult(), CasWriteOutcome::Committed); -} - -/// Fix #37 phase 2: the retry-later throw must be NETWORK_ERROR, never ABORTED -- ABORTED is silently -/// swallowed by ReplicatedMergeMutateTaskBase (no backoff, no last_exception), which is exactly the -/// defect this fix closes. -TEST(CASWriteRetryLater, ThrowsNetworkErrorNotAborted) -{ - bool threw = false; - try - { - throwCasWriteRetryLater("test cause"); - FAIL() << "throwCasWriteRetryLater must always throw"; - } - catch (const DB::Exception & e) - { - threw = true; - EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_NE(e.code(), DB::ErrorCodes::ABORTED); - EXPECT_NE(e.message().find("test cause"), String::npos) << e.message(); - EXPECT_NE(e.message().find("retrying later"), String::npos) << e.message(); - } - EXPECT_TRUE(threw); -} - -/// The exception_ptr twin (for call sites that fail a pending future/promise rather than throw -/// directly, e.g. CasRefLedger's queued-append completion paths) must carry the SAME classification. -TEST(CASWriteRetryLater, ExceptionPtrVariantCarriesSameClassification) -{ - const std::exception_ptr eptr = makeCasWriteRetryLaterExceptionPtr("another cause"); - bool threw = false; - try - { - std::rethrow_exception(eptr); - FAIL() << "expected a thrown exception"; - } - catch (const DB::Exception & e) - { - threw = true; - EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_NE(e.message().find("another cause"), String::npos) << e.message(); - } - EXPECT_TRUE(threw); -} - -#if USE_AWS_S3 - -/// One row per RFC cas-s3-timeout-retry-control §operation-classes classification. PreconditionFailed -/// is NEVER DefiniteFailure — it means the key exists, not that the request was rejected — and every -/// unrecognized/ambiguous error also falls to Unresolved, never to a false DefiniteFailure. -TEST(CASRequestControl, ClassifiesPreconditionFailedAsUnresolved) -{ - DB::S3Exception e("412 from backend", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); -} - -TEST(CASRequestControl, ClassifiesTimeoutAsUnresolved) -{ - Poco::TimeoutException e("simulated client-side receive timeout"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); -} - -TEST(CASRequestControl, ClassifiesConnectionResetAsUnresolved) -{ - Poco::Net::ConnectionResetException e("simulated connection reset"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); -} - -TEST(CASRequestControl, Classifies5xxAsUnresolved) -{ - DB::S3Exception e("simulated internal error", Aws::S3::S3Errors::INTERNAL_FAILURE, "InternalError"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::Unresolved); - /// SlowDown / ServiceUnavailable are also 5xx-class and equally Unresolved. - DB::S3Exception slow_down("simulated throttle", Aws::S3::S3Errors::SLOW_DOWN, "SlowDown"); - EXPECT_EQ(classifyConditionalWriteResult(slow_down), CasWriteOutcome::Unresolved); -} - -TEST(CASRequestControl, ClassifiesMalformedRequestAsDefiniteFailure) -{ - DB::S3Exception e("bad xml", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); - /// The modeled-enum path (no canonical name attached) must classify identically. - DB::S3Exception by_code("bad argument", Aws::S3::S3Errors::INVALID_REQUEST); - EXPECT_EQ(classifyConditionalWriteResult(by_code), CasWriteOutcome::DefiniteFailure); -} - -TEST(CASRequestControl, ClassifiesEntityTooLargeAsDefiniteFailure) -{ - DB::S3Exception e("body exceeds the maximum object size", Aws::S3::S3Errors::UNKNOWN, "EntityTooLarge"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); -} - -TEST(CASRequestControl, ClassifiesAccessDeniedAsDefiniteFailure) -{ - DB::S3Exception e("simulated 403", Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied"); - EXPECT_EQ(classifyConditionalWriteResult(e), CasWriteOutcome::DefiniteFailure); - /// The modeled-enum path (no canonical name attached) must classify identically. - DB::S3Exception by_code("simulated 403, no name", Aws::S3::S3Errors::ACCESS_DENIED); - EXPECT_EQ(classifyConditionalWriteResult(by_code), CasWriteOutcome::DefiniteFailure); -} - -/// Anything the classifier does not recognize (an unmodeled/unnamed S3 error, or an entirely -/// unrelated exception type) must fail toward Unresolved — never toward a false DefiniteFailure or a -/// false Committed (RFC §resolve-before-reissuing: ambiguity always resolves toward "resolve before -/// reissuing"). -TEST(CASRequestControl, UnrecognizedErrorsFailSafeToUnresolved) -{ - DB::S3Exception unknown_named("weird service error", Aws::S3::S3Errors::UNKNOWN, "SomeFutureErrorCode"); - EXPECT_EQ(classifyConditionalWriteResult(unknown_named), CasWriteOutcome::Unresolved); - - /// UNKNOWN_EXCEPTION (not LOGICAL_ERROR): any arbitrary non-S3 exception type works here -- the - /// point is that the classifier doesn't recognize it, not which specific code it carries. - /// LOGICAL_ERROR would abort the whole process under debug/sanitizer builds merely by being - /// constructed (Exception's constructor calls handle_error_code unconditionally). - DB::Exception unrelated(DB::ErrorCodes::UNKNOWN_EXCEPTION, "not an S3 error at all"); - EXPECT_EQ(classifyConditionalWriteResult(unrelated), CasWriteOutcome::Unresolved); -} - -/// recordConditionalWriteAttemptStarted / recordConditionalWriteOutcome bump the per-class counters -/// (RFC §observability): attempts, and exactly one of Committed/DefiniteFailure/Unresolved per call. -TEST(CASRequestControl, CountersHookupIncrementsPerClass) -{ - using ProfileEvents::global_counters; - const auto attempts_before = global_counters[ProfileEvents::CASConditionalWriteAttempts].load(); - const auto committed_before = global_counters[ProfileEvents::CASConditionalWriteCommitted].load(); - const auto definite_before = global_counters[ProfileEvents::CASConditionalWriteDefiniteFailure].load(); - const auto unresolved_before = global_counters[ProfileEvents::CASConditionalWriteUnresolved].load(); - - recordConditionalWriteAttemptStarted(); - recordConditionalWriteOutcome(CasWriteOutcome::Committed); - recordConditionalWriteAttemptStarted(); - recordConditionalWriteOutcome(CasWriteOutcome::DefiniteFailure); - recordConditionalWriteAttemptStarted(); - recordConditionalWriteOutcome(CasWriteOutcome::Unresolved); - -#if !WITH_COVERAGE - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteAttempts].load() - attempts_before, 3u); - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteCommitted].load() - committed_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteDefiniteFailure].load() - definite_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteUnresolved].load() - unresolved_before, 1u); -#else - (void)attempts_before; (void)committed_before; (void)definite_before; (void)unresolved_before; -#endif -} - -/// Wiring smoke test: a real conditional write through ObjectStorageBackend (Native mode) counts one -/// attempt and one Committed outcome via the SAME instrumented call site nativeConditionalPut uses — -/// see finalizeConditionalWriteInstrumented in CasObjectStorageBackend.cpp. -TEST(CASRequestControl, NativeConditionalPutCountsOneAttemptAndCommitted) -{ - using ProfileEvents::global_counters; - const auto attempts_before = global_counters[ProfileEvents::CASConditionalWriteAttempts].load(); - const auto committed_before = global_counters[ProfileEvents::CASConditionalWriteCommitted].load(); - - auto b = std::make_shared( - DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - EXPECT_EQ(b->putIfAbsent("p/rc/one", "v1").outcome, PutOutcome::Done); - -#if !WITH_COVERAGE - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteAttempts].load() - attempts_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASConditionalWriteCommitted].load() - committed_before, 1u); -#else - (void)attempts_before; (void)committed_before; -#endif -} - -/// Mechanism property (RFC §disable-transparent-conditional-write-retries), tested at the layer -/// actually reachable from a unit-test binary: NO live/fake S3 endpoint is available here (the Native -/// conditional-write path is exercised end-to-end only at M-W against RustFS — see the HONEST NOTE in -/// CasObjectStorageBackend.cpp), so driving a real socket-level retry against a real client is not -/// reachable from this binary. What IS reachable and asserted here: every Native conditional write -/// selects the SingleAttempt object-storage retry profile, and a non-S3 backend such as -/// LocalObjectStorage reports it as UNSUPPORTED via IObjectStorage::supportsRetryProfile — the property -/// checkConditionalWriteSingleAttemptSupport's fail-closed mount-time gate relies on. -TEST(CASRequestControl, SingleAttemptProfileRequestedAndLocalBackendRejected) -{ - auto b = std::make_shared( - DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - const auto ws = b->conditionalWriteSettingsForTest(); - EXPECT_EQ(ws.object_storage_retry_profile, DB::ObjectStorageRetryProfile::SingleAttempt); - /// LocalObjectStorage does not implement the profile — the capability check must say no. - EXPECT_FALSE(DB::Cas::tests::makeLocalObjectStorageForTest()->supportsRetryProfile(DB::ObjectStorageRetryProfile::SingleAttempt)); -} - -/// The SECOND retry-affecting layer above the S3 client (review finding): WriteBufferFromS3's OWN -/// makeSinglepartUpload/completeMultipartUpload retry loop reissues the identical conditional request -/// on a NO_SUCH_KEY response, driven by S3RequestSetting::max_unexpected_write_error_retries (default -/// 4) — a client-level override alone does not bound it (see WriteSettings:: -/// s3_max_unexpected_write_error_retries_override). Asserted at the reachable seam: no live/fake S3 -/// endpoint exists in this binary to drive the retry loop itself, so this proves the settings -/// plumbing conditionalWriteSettings() -> WriteSettings produces the override value that -/// S3ObjectStorage::writeObject then applies to request_settings — NOT a real single-attempt -/// assertion against a live wire attempt. -TEST(CASRequestControl, ConditionalWriteSettingsForceSingleUnexpectedWriteErrorRetry) -{ - auto b = std::make_shared( - DB::Cas::tests::makeLocalObjectStorageForTest(), ObjectStorageBackend::Mode::Native); - const auto ws = b->conditionalWriteSettingsForTest(); - EXPECT_EQ(ws.s3_max_unexpected_write_error_retries_override, 1u); -} - -/// ================================================================================================ -/// Task 5: CasRequestController — retry controller (deadlines, fence gating, exact-key resolution) -/// ================================================================================================ - -namespace -{ - -/// A per-call scripted Backend for CasRequestController tests: `putIfAbsent` optionally throws a -/// caller-supplied exception (models one classified HTTP-attempt outcome) or returns a forced -/// `PutOutcome` directly (models a `PreconditionFailed` observed WITHOUT an exception); with neither -/// set it delegates to the real in-memory conditional-write semantics. `get` optionally returns a -/// forced result, independent of what `putIfAbsent` actually did, so a test can drive exact-key -/// resolution (identical / different / absent) without the scripted put and the resolve GET needing to -/// agree on a shared, real backing store. -class ScriptedControllerBackend : public InMemoryBackend -{ -public: - std::function put_thrower; - std::optional put_forced_outcome; - std::atomic put_attempts{0}; - - std::function put_overwrite_thrower; - std::function put_overwrite_handler; - std::optional put_overwrite_forced_outcome; - std::atomic put_overwrite_attempts{0}; - - std::function(const String &, Range)> get_handler; - std::atomic get_attempts{0}; - bool get_overridden = false; - std::optional get_override_value; /// meaningful only when get_overridden - - void setGetOverride(std::optional value) - { - get_overridden = true; - get_override_value = std::move(value); - } - - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override - { - ++put_attempts; - if (put_thrower) - put_thrower(); - if (put_forced_outcome) - return {*put_forced_outcome, {}}; - return InMemoryBackend::putIfAbsent(key, bytes, meta); - } - - PutResult putOverwrite(const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) override - { - ++put_overwrite_attempts; - if (put_overwrite_handler) - return put_overwrite_handler(key, bytes, expected, meta); - if (put_overwrite_thrower) - put_overwrite_thrower(); - if (put_overwrite_forced_outcome) - return {*put_overwrite_forced_outcome, {}}; - return InMemoryBackend::putOverwrite(key, bytes, expected, meta); - } - - std::optional get(const String & key, Range range) override - { - ++get_attempts; - if (get_handler) - return get_handler(key, range); - if (get_overridden) - return get_override_value; - return InMemoryBackend::get(key, range); - } -}; - -GetResult resultWithBytes(const String & bytes) -{ - return GetResult{.bytes = bytes, .token = Token{"t", TokenType::Emulated}, .attributes = {}}; -} - -CasOverwriteOperationContext overwriteContext( - uint64_t absolute_deadline_ms, - CasOverwriteDeadlineSource deadline_source = CasOverwriteDeadlineSource::RequestBudget, - std::function stop_cause = {}, - std::function wait_before_retry = {}, - std::function observe = {}) -{ - return CasOverwriteOperationContext{ - .absolute_deadline_ms = absolute_deadline_ms, - .deadline_source = deadline_source, - .stop_cause = stop_cause ? std::move(stop_cause) : [] { return CasOverwriteStopCause::Continue; }, - .wait_before_retry = wait_before_retry ? std::move(wait_before_retry) : [](uint64_t) { return true; }, - .observe = observe ? std::move(observe) : [](const CasOverwriteProgress &) {}, - }; -} - -void expectOverwriteDiagnostics( - const CasOverwriteResult & result, - uint32_t attempts_sent, - bool resolved_by_get, - CasUnresolvedReason unresolved_reason, - CasOverwriteDeadlineSource deadline_source, - CasOverwriteStopCause stop_cause) -{ - EXPECT_EQ(result.diagnostics.attempts_sent, attempts_sent); - EXPECT_EQ(result.diagnostics.resolved_by_get, resolved_by_get); - EXPECT_EQ(result.diagnostics.unresolved_reason, unresolved_reason); - EXPECT_EQ(result.diagnostics.deadline_source, deadline_source); - EXPECT_EQ(result.diagnostics.stop_cause, stop_cause); -} - -} - -TEST(CASRequestController, UncertainResolvesIdenticalAsCommitted) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(resultWithBytes("payload")); - - CasRequestController controller(backend, CasRequestBudget{}); - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::Committed); - EXPECT_EQ(backend->put_attempts.load(), 1u); -} - -TEST(CASRequestController, UncertainResolvesDifferentThrowsCorruption) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(resultWithBytes("someone-elses-bytes")); - - CasRequestController controller(backend, CasRequestBudget{}); - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] - { - controller.putIfAbsentControlled("k", "payload", [] { return true; }); - }); -} - -/// GET-absent NEVER yields DefiniteFailure (spec §writer-side-linearization): the SAME (key, bytes) is -/// retried up to `max_attempts`, and only THEN does the call give up with Unresolved. -TEST(CASRequestController, UncertainResolvesAbsentRetriesSameKeyWithinBudget) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); /// absent on every resolve - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.retry_initial_backoff_ms = 0; /// backoff behavior is pinned by its own tests below - CasRequestController controller(backend, budget); - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 3u); /// every attempt targeted the SAME key/bytes -} - -/// The operation deadline — not just the attempt-count budget — cuts a retry loop short: a fake clock -/// advances by a fixed step per now_ms() call (no sleeps), and max_attempts is generous enough that only -/// the deadline check can be what stops the loop. -TEST(CASRequestController, OperationDeadlineExhaustionReturnsUnresolvedBeforeMaxAttempts) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); /// absent on every resolve - - uint64_t clock = 0; - auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 200; return t; }; - - CasRequestBudget budget; - budget.max_attempts = 10; - budget.attempt_timeout_ms = 50; - budget.operation_deadline_ms = 450; - budget.retry_initial_backoff_ms = 0; /// isolate the deadline check from the backoff's own deadline guard - CasRequestController controller(backend, budget, now_ms); - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 2u); /// cut off well before the 10-attempt budget -} - -/// WHY `attempt_timeout_ms == operation_deadline_ms` IS REJECTED AT STARTUP, demonstrated on the -/// mechanism itself before the rejection is asserted below. -/// -/// The deadline is captured as `now + operation_deadline_ms` and the pre-send gate asks -/// `now + attempt_timeout_ms > deadline`. Equal values collapse that to `now_2 > now_1`: ONE elapsed -/// millisecond between the capture and the gate refuses the whole operation with NOTHING SENT. That is -/// not a bounded operation, it is a coin flip on the scheduler -- "mostly works, occasionally refuses -/// having sent nothing", which is the flakiness class validation exists to prevent. Single-attempt -/// semantics is what `max_attempts = 1` is for; the equality contributes only the race. -/// -/// The controller is constructed DIRECTLY here, bypassing `validateCasRequestBudget`, because the -/// point is to show the behaviour the validator now forbids. Three tests were flaky on exactly this -/// before it was forbidden: `8f9e63c7a19`'s sweep-interruption test, -/// `CASRefInstallSafety.UncertainPrecommitKeepsItsCleanupOwnerAndItsBody`, and -/// `CASRefWriterAppendLane.WedgedLaneBlocksSameTableWhileOtherTableProceeds`. -TEST(CASRequestController, EqualAttemptTimeoutAndDeadlineWouldRefuseAfterASingleTick) -{ - auto backend = std::make_shared(); - - /// The smallest possible passage of time: one millisecond per clock read. - uint64_t clock = 0; - auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1; return t; }; - - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 100; - CasRequestController razor(backend, budget, now_ms); - EXPECT_EQ(razor.putIfAbsentControlled("k", "payload", [] { return true; }), CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 0u) - << "the refusal came from the clock, not from the backend: nothing was sent at all"; - - /// STRICTLY LESS -- the shape the validator now requires -- sends the request over the SAME one-tick - /// clock. So what the inequality buys is the request actually happening, not merely a bigger number. - clock = 0; - budget.operation_deadline_ms = 5000; - CasRequestController wide(backend, budget, now_ms); - EXPECT_EQ(wide.putIfAbsentControlled("k", "payload", [] { return true; }), CasWriteOutcome::Committed); - EXPECT_EQ(backend->put_attempts.load(), 1u); -} - -/// And the same equality is refused at startup, so no budget can reach the controller in that shape. -/// The boundary is asserted from BOTH sides: equality throws, one millisecond more is accepted. -TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutEqualToOperationDeadline) -{ - CasRequestBudget budget; - budget.attempt_timeout_ms = 5000; - budget.operation_deadline_ms = 5000; - budget.lease_safety_margin_ms = 1000; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); - - budget.operation_deadline_ms = 5001; - EXPECT_NO_THROW(validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000)) - << "one millisecond of headroom is the whole requirement -- the rule is strictness, not size"; -} - -TEST(CASRequestController, OverwriteAmbiguousResolvesIntendedBytesAsCommitted) -{ - auto backend = std::make_shared(); - bool first_attempt = true; - backend->put_overwrite_thrower = [&first_attempt] - { - if (first_attempt) - { - first_attempt = false; - throw Poco::TimeoutException("scripted: ambiguous"); - } - }; - backend->setGetOverride(resultWithBytes("new-payload")); - - CasRequestController controller(backend, CasRequestBudget{}); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, [] { return true; }); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(result.token, (Token{"t", TokenType::Emulated})); -} - -TEST(CASRequestController, OverwriteAmbiguousResolvesExpectedTokenAndRetriesWithinBudget) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget); - const auto result = controller.putOverwriteControlled("k", "new-payload", expected, [] { return true; }); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 3u); -} - -TEST(CASRequestController, OverwriteAmbiguousResolvesDifferentTokenAndBytesAsConflict) -{ - auto backend = std::make_shared(); - bool first_attempt = true; - backend->put_overwrite_thrower = [&first_attempt] - { - if (first_attempt) - { - first_attempt = false; - throw Poco::TimeoutException("scripted: ambiguous"); - } - }; - backend->setGetOverride(GetResult{ - .bytes = "someone-elses-payload", .token = Token{"other", TokenType::Emulated}, .attributes = {}}); - - CasRequestController controller(backend, CasRequestBudget{}); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, [] { return true; }); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Conflict); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); -} - -TEST(CASRequestController, OverwriteOperationDeadlineExhaustionReturnsUnresolvedBeforeMaxAttempts) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - - uint64_t clock = 0; - auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 200; return t; }; - - CasRequestBudget budget; - budget.max_attempts = 10; - budget.attempt_timeout_ms = 50; - budget.operation_deadline_ms = 450; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget, now_ms); - const auto result = controller.putOverwriteControlled("k", "new-payload", expected, [] { return true; }); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); -} - -TEST(CASRequestController, AbsoluteDeadlineCannotBeReanchoredAfterPreemption) -{ - auto backend = std::make_shared(); - uint64_t clock = 60; - - CasRequestBudget budget; - budget.attempt_timeout_ms = 50; - budget.operation_deadline_ms = 10000; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); - expectOverwriteDiagnostics( - result, - 0, - false, - CasUnresolvedReason::NoAttemptSent, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, MaxAttemptsOneStillResolvesLostResponseByGet) -{ - auto backend = std::make_shared(); - const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); - ASSERT_EQ(predecessor.outcome, PutOutcome::Done); - - Token landed_token; - auto * backend_ptr = backend.get(); - backend->put_overwrite_handler = [backend_ptr, &landed_token]( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult - { - const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); - landed_token = landed.token; - throw Poco::TimeoutException("scripted: response lost after overwrite landed"); - }; - std::vector> progress; - - CasRequestBudget budget; - budget.max_attempts = 1; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - auto context = overwriteContext( - 1000, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - {}, - {}, - [&progress](const CasOverwriteProgress & event) { progress.emplace_back(event.kind, event.attempt_no); }); - const auto result = controller.putOverwriteControlled("k", "new-payload", predecessor.token, context); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); - EXPECT_EQ(result.token, landed_token); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - true, - CasUnresolvedReason::NotUnresolved, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); - const std::vector> expected_progress{ - {CasOverwriteProgressKind::PutStarted, 1}, - {CasOverwriteProgressKind::BecameAmbiguous, 1}, - {CasOverwriteProgressKind::ResolveStarted, 1}, - {CasOverwriteProgressKind::ResolvedByGet, 1}, - }; - EXPECT_EQ(progress, expected_progress); -} - -TEST(CASRequestController, StopBeforeFirstPutReportsExactCause) -{ - auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}, [] { return static_cast(0); }); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext( - 10000, - CasOverwriteDeadlineSource::RequestBudget, - [] { return CasOverwriteStopCause::Cancelled; })); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); - EXPECT_EQ(backend->get_attempts.load(), 0u); - expectOverwriteDiagnostics( - result, - 0, - false, - CasUnresolvedReason::NoAttemptSent, - CasOverwriteDeadlineSource::RequestBudget, - CasOverwriteStopCause::Cancelled); -} - -TEST(CASRequestController, StopAfterPutSuppressesResolveAndReportsMidWay) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - uint32_t stop_samples = 0; - auto stop_cause = [&stop_samples] - { - ++stop_samples; - return stop_samples == 1 ? CasOverwriteStopCause::Continue : CasOverwriteStopCause::Cancelled; - }; - - CasRequestBudget budget; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(10000, CasOverwriteDeadlineSource::RequestBudget, stop_cause)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 0u); - expectOverwriteDiagnostics( - result, - 1, - false, - CasUnresolvedReason::FenceLostMidWay, - CasOverwriteDeadlineSource::RequestBudget, - CasOverwriteStopCause::Cancelled); -} - -TEST(CASRequestController, StopAfterResolvedCommitReportsPostWrite) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(resultWithBytes("new-payload")); - uint32_t stop_samples = 0; - auto stop_cause = [&stop_samples] - { - ++stop_samples; - return stop_samples < 3 ? CasOverwriteStopCause::Continue : CasOverwriteStopCause::FenceOrLifecycleLost; - }; - - CasRequestController controller(backend, CasRequestBudget{}, [] { return static_cast(0); }); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(10000, CasOverwriteDeadlineSource::RequestBudget, stop_cause)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - true, - CasUnresolvedReason::FenceLostPostWrite, - CasOverwriteDeadlineSource::RequestBudget, - CasOverwriteStopCause::FenceOrLifecycleLost); -} - -TEST(CASRequestController, FenceWinsCancellationAndExternalDeadlineWinsDeadlineTie) -{ - auto backend = std::make_shared(); - uint64_t clock = 100; - CasRequestBudget budget; - budget.attempt_timeout_ms = 10; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - - bool cancelled = true; - bool fenced = true; - const auto simultaneous_stop = [&] - { - if (fenced) - return CasOverwriteStopCause::FenceOrLifecycleLost; - if (cancelled) - return CasOverwriteStopCause::Cancelled; - return CasOverwriteStopCause::Continue; - }; - const auto stopped = controller.putOverwriteControlled( - "stopped", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety, simultaneous_stop)); - EXPECT_EQ(stopped.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - stopped, - 0, - false, - CasUnresolvedReason::NoAttemptSent, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::FenceOrLifecycleLost); - - const auto deadline_tie = controller.putOverwriteControlled( - "deadline", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - EXPECT_EQ(deadline_tie.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - deadline_tie, - 0, - false, - CasUnresolvedReason::NoAttemptSent, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 0u); -} - -TEST(CASRequestController, InterruptedWaitResamplesStopCause) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - CasOverwriteStopCause stop = CasOverwriteStopCause::Continue; - uint32_t waits = 0; - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 100; - budget.retry_max_backoff_ms = 100; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - auto context = overwriteContext( - 10000, - CasOverwriteDeadlineSource::RequestBudget, - [&stop] { return stop; }, - [&stop, &waits](uint64_t wait_ms) - { - EXPECT_EQ(wait_ms, 100u); - ++waits; - stop = CasOverwriteStopCause::Cancelled; - return false; - }); - const auto result = controller.putOverwriteControlled("k", "new-payload", expected, context); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(waits, 1u); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - false, - CasUnresolvedReason::FenceLostMidWay, - CasOverwriteDeadlineSource::RequestBudget, - CasOverwriteStopCause::Cancelled); -} - -#ifndef DEBUG_OR_SANITIZER_BUILD -TEST(CASRequestController, InterruptedWaitWithoutPublishedStopIsAProgrammingException) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - uint32_t stop_samples = 0; - uint32_t waits = 0; - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 100; - budget.retry_max_backoff_ms = 100; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - auto context = overwriteContext( - 10000, - CasOverwriteDeadlineSource::RequestBudget, - [&stop_samples] - { - ++stop_samples; - return CasOverwriteStopCause::Continue; - }, - [&waits](uint64_t wait_ms) - { - EXPECT_EQ(wait_ms, 100u); - ++waits; - return false; - }); - - bool threw = false; - try - { - (void)controller.putOverwriteControlled("k", "new-payload", expected, context); - FAIL() << "an interrupted wait must publish a non-Continue stop cause"; - } - catch (const DB::Exception & e) - { - threw = true; - EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR); - EXPECT_EQ( - e.message(), - "CasRequestController: wait_before_retry returned false while stop_cause remained Continue"); - } - - EXPECT_TRUE(threw); - EXPECT_EQ(stop_samples, 5u) << "the false wait was followed by an authoritative stop-cause resample"; - EXPECT_EQ(waits, 1u); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); -} -#endif - -#if defined(DEBUG_OR_SANITIZER_BUILD) -TEST(CASRequestControllerDeathTest, InterruptedWaitWithoutPublishedStopIsAProgrammingExceptionAborts) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 100; - budget.retry_max_backoff_ms = 100; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - auto context = overwriteContext( - 10000, - CasOverwriteDeadlineSource::RequestBudget, - [] { return CasOverwriteStopCause::Continue; }, - [](uint64_t) { return false; }); - - EXPECT_DEATH( - { (void)controller.putOverwriteControlled("k", "new-payload", expected, context); }, - "wait_before_retry returned false while stop_cause remained Continue"); -} -#endif - -TEST(CASRequestController, CompletedWaitCrossingDeadlineSendsNoRetry) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - const Token expected{"old", TokenType::Emulated}; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - uint64_t clock = 0; - uint32_t waits = 0; - - CasRequestBudget budget; - budget.max_attempts = 3; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 50; - budget.retry_max_backoff_ms = 50; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - auto context = overwriteContext( - 100, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - {}, - [&clock, &waits](uint64_t wait_ms) - { - EXPECT_EQ(wait_ms, 50u); - ++waits; - clock = 101; - return true; - }); - const auto result = controller.putOverwriteControlled("k", "new-payload", expected, context); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_TRUE(result.token.empty()); - EXPECT_EQ(waits, 1u); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - false, - CasUnresolvedReason::DeadlineMidWay, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, DirectPutCompletingAtDeadlineIsNotAccepted) -{ - auto backend = std::make_shared(); - const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); - ASSERT_EQ(predecessor.outcome, PutOutcome::Done); - uint64_t clock = 0; - Token landed_token; - auto * backend_ptr = backend.get(); - backend->put_overwrite_handler = [backend_ptr, &clock, &landed_token]( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult - { - const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); - landed_token = landed.token; - clock = 100; - return landed; - }; - - CasRequestBudget budget; - budget.attempt_timeout_ms = 10; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - const auto result = controller.putOverwriteControlled( - "k", - "new-payload", - predecessor.token, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_TRUE(result.token.empty()); - ASSERT_FALSE(landed_token.empty()); - const auto durable = backend->InMemoryBackend::get("k", Range{}); - ASSERT_TRUE(durable.has_value()); - EXPECT_EQ(durable->bytes, "new-payload"); - EXPECT_EQ(durable->token, landed_token); - EXPECT_NE(durable->token, predecessor.token); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 0u); - expectOverwriteDiagnostics( - result, - 1, - false, - CasUnresolvedReason::DeadlineMidWay, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, ReadProofCompletingAtDeadlineIsNotAccepted) -{ - auto backend = std::make_shared(); - const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); - ASSERT_EQ(predecessor.outcome, PutOutcome::Done); - uint64_t clock = 0; - Token landed_token; - auto * backend_ptr = backend.get(); - backend->put_overwrite_handler = [backend_ptr, &landed_token]( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult - { - const PutResult landed = backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); - landed_token = landed.token; - throw Poco::TimeoutException("scripted: response lost after overwrite landed"); - }; - backend->get_handler = [backend_ptr, &clock](const String & key, Range range) -> std::optional - { - const auto durable = backend_ptr->InMemoryBackend::get(key, range); - clock = 100; - return durable; - }; - - CasRequestBudget budget; - budget.attempt_timeout_ms = 10; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - const auto result = controller.putOverwriteControlled( - "k", - "new-payload", - predecessor.token, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_TRUE(result.token.empty()); - ASSERT_FALSE(landed_token.empty()); - const auto durable = backend->InMemoryBackend::get("k", Range{}); - ASSERT_TRUE(durable.has_value()); - EXPECT_EQ(durable->bytes, "new-payload"); - EXPECT_EQ(durable->token, landed_token); - EXPECT_NE(durable->token, predecessor.token); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - true, - CasUnresolvedReason::DeadlineMidWay, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, ObserverFailureCannotChangeOutcome) -{ - auto backend = std::make_shared(); - const PutResult predecessor = backend->InMemoryBackend::putIfAbsent("k", "old-payload"); - ASSERT_EQ(predecessor.outcome, PutOutcome::Done); - auto * backend_ptr = backend.get(); - bool first_attempt = true; - backend->put_overwrite_handler = [backend_ptr, &first_attempt]( - const String & key, const String & bytes, const Token & expected, const ObjectMeta & meta) -> PutResult - { - if (first_attempt) - { - first_attempt = false; - throw Poco::TimeoutException("scripted: transient failure before landing"); - } - return backend_ptr->InMemoryBackend::putOverwrite(key, bytes, expected, meta); - }; - uint32_t observer_calls = 0; - - CasRequestBudget budget; - budget.max_attempts = 2; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget, [] { return static_cast(0); }); - auto context = overwriteContext( - 10000, - CasOverwriteDeadlineSource::RequestBudget, - {}, - {}, - [&observer_calls](const CasOverwriteProgress &) - { - ++observer_calls; - throw DB::Exception(DB::ErrorCodes::UNKNOWN_EXCEPTION, "scripted observer failure"); - }); - - CasOverwriteResult result; - EXPECT_NO_THROW(result = controller.putOverwriteControlled("k", "new-payload", predecessor.token, context)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Committed); - EXPECT_EQ(observer_calls, 5u); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 2, - false, - CasUnresolvedReason::NotUnresolved, - CasOverwriteDeadlineSource::RequestBudget, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, ResolveFailuresExhaustDeadlineWithoutSendingLatePut) -{ - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - uint64_t clock = 0; - backend->get_handler = [&clock](const String &, Range) -> std::optional - { - clock = 30; - throw Poco::TimeoutException("scripted: resolving GET failed at deadline"); - }; - - CasRequestBudget budget; - budget.max_attempts = 5; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 0; - CasRequestController controller(backend, budget, [&clock] { return clock; }); - const auto result = controller.putOverwriteControlled( - "k", "new-payload", Token{"old", TokenType::Emulated}, - overwriteContext(30, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 1u); - EXPECT_EQ(backend->get_attempts.load(), 1u); - expectOverwriteDiagnostics( - result, - 1, - false, - CasUnresolvedReason::DeadlineMidWay, - CasOverwriteDeadlineSource::ExternalLeaseSafety, - CasOverwriteStopCause::Continue); -} - -TEST(CASRequestController, EveryTerminalShapeReportsExactDiagnostics) -{ - const Token expected{"old", TokenType::Emulated}; - auto make_controller = [](const std::shared_ptr & backend, uint64_t & clock, uint32_t max_attempts = 2) - { - CasRequestBudget budget; - budget.max_attempts = max_attempts; - budget.attempt_timeout_ms = 10; - budget.retry_initial_backoff_ms = 0; - return CasRequestController(backend, budget, [&clock] { return clock; }); - }; - - for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) - { - auto backend = std::make_shared(); - uint64_t clock = 0; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "pre-stop", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [stop] { return stop; })); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 0, false, CasUnresolvedReason::NoAttemptSent, - CasOverwriteDeadlineSource::RequestBudget, stop); - } - - for (const auto source : {CasOverwriteDeadlineSource::RequestBudget, CasOverwriteDeadlineSource::ExternalLeaseSafety}) - { - auto backend = std::make_shared(); - uint64_t clock = 100; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "pre-deadline", "new-payload", expected, overwriteContext(100, source)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 0, false, CasUnresolvedReason::NoAttemptSent, source, CasOverwriteStopCause::Continue); - } - - for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) - { - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - uint64_t clock = 0; - uint32_t stop_samples = 0; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "mid-stop", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [&stop_samples, stop] - { - ++stop_samples; - return stop_samples == 1 ? CasOverwriteStopCause::Continue : stop; - })); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 1, false, CasUnresolvedReason::FenceLostMidWay, - CasOverwriteDeadlineSource::RequestBudget, stop); - } - - for (const auto stop : {CasOverwriteStopCause::Cancelled, CasOverwriteStopCause::FenceOrLifecycleLost}) - { - auto backend = std::make_shared(); - backend->put_overwrite_forced_outcome = PutOutcome::Done; - uint64_t clock = 0; - uint32_t stop_samples = 0; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "post-stop", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget, [&stop_samples, stop] - { - ++stop_samples; - return stop_samples == 1 ? CasOverwriteStopCause::Continue : stop; - })); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 1, false, CasUnresolvedReason::FenceLostPostWrite, - CasOverwriteDeadlineSource::RequestBudget, stop); - } - - for (const auto source : {CasOverwriteDeadlineSource::RequestBudget, CasOverwriteDeadlineSource::ExternalLeaseSafety}) - { - auto backend = std::make_shared(); - uint64_t clock = 0; - backend->put_overwrite_handler = [&clock]( - const String &, const String &, const Token &, const ObjectMeta &) -> PutResult - { - clock = 100; - throw Poco::TimeoutException("scripted: ambiguous at deadline"); - }; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "mid-deadline", "new-payload", expected, overwriteContext(100, source)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 1, false, CasUnresolvedReason::DeadlineMidWay, source, CasOverwriteStopCause::Continue); - } - - { - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); - uint64_t clock = 0; - auto controller = make_controller(backend, clock, 1); - const auto result = controller.putOverwriteControlled( - "attempts", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 1, false, CasUnresolvedReason::AttemptsExhausted, - CasOverwriteDeadlineSource::ExternalLeaseSafety, CasOverwriteStopCause::Continue); - } - - { - auto backend = std::make_shared(); - auto * backend_ptr = backend.get(); - backend->put_overwrite_handler = [backend_ptr]( - const String &, const String &, const Token &, const ObjectMeta &) -> PutResult - { - if (backend_ptr->put_overwrite_attempts.load() == 1) - throw Poco::TimeoutException("scripted: first attempt remains ambiguous"); - throw DB::S3Exception("scripted: later attempt definitely refused", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); - }; - backend->setGetOverride(GetResult{.bytes = "old-payload", .token = expected, .attributes = {}}); - uint64_t clock = 0; - auto controller = make_controller(backend, clock); - const auto result = controller.putOverwriteControlled( - "definite-after-ambiguity", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::RequestBudget)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - EXPECT_EQ(backend->put_overwrite_attempts.load(), 2u); - expectOverwriteDiagnostics( - result, 2, false, CasUnresolvedReason::DefiniteFailureAfterAmbiguity, - CasOverwriteDeadlineSource::RequestBudget, CasOverwriteStopCause::Continue); - } - - { - auto backend = std::make_shared(); - backend->put_overwrite_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - uint64_t clock = 0; - backend->get_handler = [&clock](const String &, Range) -> std::optional - { - clock = 95; - return std::nullopt; - }; - auto controller = make_controller(backend, clock, 1); - const auto result = controller.putOverwriteControlled( - "deadline-before-attempt-limit", "new-payload", expected, - overwriteContext(100, CasOverwriteDeadlineSource::ExternalLeaseSafety)); - EXPECT_EQ(result.outcome, CasOverwriteOutcome::Unresolved); - expectOverwriteDiagnostics( - result, 1, false, CasUnresolvedReason::DeadlineMidWay, - CasOverwriteDeadlineSource::ExternalLeaseSafety, CasOverwriteStopCause::Continue); - } -} - -TEST(CASRequestController, FenceLostBeforeAttemptSendsNoAttempt) -{ - auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return false; }); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 0u); -} - -/// The write itself may have landed, but a fence lost between the write and this call's own final -/// check must never surface as Committed (RFC §ack-and-cache-rules: no ACK, no cache update on that -/// path) — the caller sees Unresolved and must not treat the operation as acknowledged. -TEST(CASRequestController, FenceLostAfterWriteNeverReturnsCommitted) -{ - auto backend = std::make_shared(); /// real in-memory commit path - int fence_calls = 0; - auto fence_ok = [&fence_calls] { return fence_calls++ == 0; }; /// true once, then false - - CasRequestController controller(backend, CasRequestBudget{}); - const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 1u); /// the write itself DID happen - EXPECT_TRUE(backend->head("k").exists); /// ...it is durable; never claimed as Committed here -} - -TEST(CASRequestController, DefiniteFailurePropagatesImmediatelyWithoutResolve) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw DB::S3Exception("scripted: malformed", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); }; - - CasRequestController controller(backend, CasRequestBudget{}); - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::DefiniteFailure); - EXPECT_EQ(backend->put_attempts.load(), 1u); /// no retry, no resolve GET issued -} - -/// ================================================================================================ -/// Inter-attempt backoff (chaos-tolerance-report §Task B follow-up / stagefix-review M3): the -/// controller paces reissues with a capped-exponential, fence-gated, deadline-aware sleep instead of -/// hammering a recovering store with immediate retries. -/// ================================================================================================ - -/// The full event-ordered schedule: fence checked before EVERY attempt AND before EVERY sleep, sleeps -/// strictly between attempts, capped exponential (initial 100ms, cap 200ms), no sleep after the final -/// attempt. The exact interleaving is the contract — a sleep served before its fence check would keep -/// a fenced writer dozing past its lease. -TEST(CASRequestControllerBackoff, CappedExponentialSleepsAreFenceCheckedAndOrdered) -{ - auto backend = std::make_shared(); - std::vector events; - backend->put_thrower = [&] { events.emplace_back("put"); throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); /// absent on every resolve - - CasRequestBudget budget; - budget.max_attempts = 5; - budget.attempt_timeout_ms = 1; - budget.operation_deadline_ms = 1000000; /// never the binding constraint here - budget.retry_initial_backoff_ms = 100; - budget.retry_max_backoff_ms = 200; - CasRequestController controller( - backend, budget, - /*now_ms=*/[] { return static_cast(0); }, - /*sleep_ms=*/[&](uint64_t ms) { events.push_back("sleep:" + std::to_string(ms)); }); - - const auto fence_ok = [&] { events.emplace_back("fence"); return true; }; - const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 5u); - - const std::vector expected{ - "fence", "put", "fence", "sleep:100", - "fence", "put", "fence", "sleep:200", - "fence", "put", "fence", "sleep:200", - "fence", "put", "fence", "sleep:200", - "fence", "put"}; /// budget spent: no fence-for-sleep, no sleep after the last attempt - EXPECT_EQ(events, expected); -} - -/// A fence lost between an ambiguous attempt's resolve and its backoff sleep aborts INSTANTLY: no -/// sleep is served, no further attempt is sent, and the outcome is Unresolved (never a false -/// Committed, never a retry under a lost lease). -TEST(CASRequestControllerBackoff, FenceLostBeforeSleepAbortsWithoutSleeping) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); - - CasRequestBudget budget; - budget.max_attempts = 5; - budget.retry_initial_backoff_ms = 100; - budget.retry_max_backoff_ms = 200; - uint64_t sleeps = 0; - int fence_calls = 0; - CasRequestController controller( - backend, budget, /*now_ms=*/[] { return static_cast(0); }, - /*sleep_ms=*/[&](uint64_t) { ++sleeps; }); - - /// True for the pre-attempt check (call 1), lost by the pre-sleep check (call 2). - const auto fence_ok = [&fence_calls] { return ++fence_calls <= 1; }; - const auto outcome = controller.putIfAbsentControlled("k", "payload", fence_ok); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 1u) << "no attempt may be sent after the fence is lost"; - EXPECT_EQ(sleeps, 0u) << "a fence lost mid-backoff must abort BEFORE the sleep, not after it"; - EXPECT_EQ(fence_calls, 2); -} - -/// A backoff sleep the operation deadline cannot afford is never served: when sleep + one more -/// attempt would cross the deadline, the loop gives up immediately (Unresolved) instead of sleeping -/// into a guaranteed exhaustion. -TEST(CASRequestControllerBackoff, SleepThatWouldCrossOperationDeadlineIsSkipped) -{ - auto backend = std::make_shared(); - backend->put_thrower = [] { throw Poco::TimeoutException("scripted: ambiguous"); }; - backend->setGetOverride(std::nullopt); - - uint64_t clock = 0; - CasRequestBudget budget; - budget.max_attempts = 10; - budget.attempt_timeout_ms = 10; - budget.operation_deadline_ms = 100; - budget.retry_initial_backoff_ms = 1000; /// any sleep would blow the 100ms deadline - budget.retry_max_backoff_ms = 1000; - uint64_t sleeps = 0; - CasRequestController controller( - backend, budget, /*now_ms=*/[&clock] { return clock; }, - /*sleep_ms=*/[&](uint64_t) { ++sleeps; }); - - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::Unresolved); - EXPECT_EQ(backend->put_attempts.load(), 1u); - EXPECT_EQ(sleeps, 0u) << "the deadline guard must refuse the sleep, not serve it and then fail"; -} - -/// THE ENVELOPE CONTRACT (chaos-tolerance-report §Task B follow-up): the DEFAULT budget rides a -/// simulated 60-second S3 outage — every conditional-write attempt fails (≈3s adaptive first-attempt -/// timeout each, the observed incident shape) until the store recovers at t=60s, then the next -/// attempt commits, all inside the default 90s operation deadline and 16-attempt budget. The fake -/// clock advances 3s per failed attempt and by each backoff sleep, so this test pins the arithmetic -/// documented on CasRequestBudget without any wall-clock waiting. -TEST(CASRequestControllerBackoff, DefaultBudgetRidesSixtySecondOutage) -{ - auto backend = std::make_shared(); - uint64_t clock = 0; - backend->put_thrower = [&clock] - { - if (clock < 60000) - { - clock += 3000; /// the failed attempt's own ~3s adaptive receive timeout - throw Poco::TimeoutException("scripted: store paused"); - } - /// store recovered: fall through to the real in-memory conditional write (Done) - }; - backend->setGetOverride(std::nullopt); /// nothing ever landed while the store was paused - - CasRequestController controller( - backend, CasRequestBudget{}, /*now_ms=*/[&clock] { return clock; }, - /*sleep_ms=*/[&clock](uint64_t ms) { clock += ms; }); - - const auto outcome = controller.putIfAbsentControlled("k", "payload", [] { return true; }); - EXPECT_EQ(outcome, CasWriteOutcome::Committed) << "the default budget must absorb a 60s outage"; - /// Schedule: attempts fail at 3s each with sleeps 0.2,0.4,0.8,1.6,3.2 then 5s (cap); the first - /// attempt scheduled at clock >= 60000 (attempt 11, t=61.2s) commits — well inside 16 attempts - /// and the 90s deadline. - EXPECT_EQ(backend->put_attempts.load(), 11u); - EXPECT_LT(clock, CasRequestBudget{}.operation_deadline_ms); -} - -/// Startup validation (RFC §required-timeout-model): a consistent default budget is accepted silently; -/// either inequality violated on its own is rejected with BAD_ARGUMENTS. -TEST(CASRequestController, ValidateBudgetAcceptsConsistentDefaults) -{ - EXPECT_NO_THROW(validateCasRequestBudget(CasRequestBudget{}, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000)); -} - -TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutPlusMarginAtOrAboveLeaseTtl) -{ - CasRequestBudget budget; - budget.attempt_timeout_ms = 25000; - budget.lease_safety_margin_ms = 5000; /// sums to EXACTLY the lease TTL below — not strictly less - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); -} - -TEST(CASRequestController, ValidateBudgetRejectsAttemptTimeoutAboveOperationDeadline) -{ - CasRequestBudget budget; - budget.attempt_timeout_ms = 6000; - budget.operation_deadline_ms = 5000; - budget.lease_safety_margin_ms = 1000; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); -} - -/// max_attempts == 0 would let putIfAbsentControlled return Unresolved without ever sending an -/// attempt — reject at startup rather than silently accepting a no-op budget. -TEST(CASRequestController, ValidateBudgetRejectsZeroMaxAttempts) -{ - CasRequestBudget budget; - budget.max_attempts = 0; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); -} - -/// A capped-exponential backoff whose cap sits below its own starting value is inconsistent — reject -/// at startup (0/0 disables backoff and stays accepted, covered by the defaults test above since the -/// defaults are nonzero and consistent). -TEST(CASRequestController, ValidateBudgetRejectsInitialBackoffAboveMaxBackoff) -{ - CasRequestBudget budget; - budget.retry_initial_backoff_ms = 500; - budget.retry_max_backoff_ms = 100; - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); -} - -/// attempt_timeout_ms + lease_safety_margin_ms must not be computed by a wrapping uint64 sum: absurd -/// config values near UINT64_MAX must fail validation (correctly, as inconsistent), never wrap around -/// to a spuriously small sum that would pass the "< lease TTL" check. -TEST(CASRequestController, ValidateBudgetRejectsOverflowingSumRatherThanWrapping) -{ - CasRequestBudget budget; - budget.attempt_timeout_ms = std::numeric_limits::max() - 10; - budget.lease_safety_margin_ms = 20; /// sum would wrap past UINT64_MAX to a tiny value - budget.operation_deadline_ms = std::numeric_limits::max(); - DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] - { - validateCasRequestBudget(budget, /*mount_lease_ttl_ms=*/30000, /*mount_renew_period_ms=*/10000); - }); -} - -#endif diff --git a/src/Disks/tests/gtest_cas_requests.cpp b/src/Disks/tests/gtest_cas_requests.cpp new file mode 100644 index 000000000000..81086f27f433 --- /dev/null +++ b/src/Disks/tests/gtest_cas_requests.cpp @@ -0,0 +1,3033 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "cas_test_helpers.h" +#include +#include + +#include +#include +#include + +#include "config.h" + +#include +#include +#include +#include + +#include +#include +#include +#include + +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int ABORTED; +extern const int BAD_ARGUMENTS; +extern const int CAS_DELETE_MARKER; +extern const int CORRUPTED_DATA; +extern const int LOGICAL_ERROR; +extern const int S3_ERROR; +extern const int NETWORK_ERROR; +} + +namespace ProfileEvents +{ + extern const Event CASRequestReissue; + extern const Event CASRequestConflictPause; + extern const Event CASRequestConnectFailureHint; + extern const Event CASRequestFirstAttemptFuse; +} + +using namespace DB::Cas; + +using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::FakeClock; +using DB::Cas::tests::expectBytes; +using DB::Cas::tests::expectThrowsCode; + +namespace +{ + +/// Every engine test drives `CasRequests` on an injected clock, so a ninety-second policy is exercised +/// in no wall-clock time and the retry schedule itself becomes an assertion. +CasRequests makeRequests(BackendPtr backend, FakeClock & clock, Fence fence = Fence::open()) +{ + return CasRequests(std::move(backend), std::move(fence), clock.nowFn(), clock.sleepFn()); +} + +} + +static_assert(!std::is_default_constructible_v); +static_assert(!std::is_constructible_v); +static_assert(!std::is_constructible_v); +static_assert(!std::is_default_constructible_v); +static_assert(!std::is_copy_constructible_v); + +TEST(CASIncarnation, GrammarRefusesTheNineWays) +{ + EXPECT_FALSE(isIncarnationValue(Dialect::ETag, "")); + EXPECT_FALSE(isIncarnationValue(Dialect::ETag, "*")); + EXPECT_FALSE(isIncarnationValue(Dialect::ETag, " * ")); + EXPECT_FALSE(isIncarnationValue(Dialect::ETag, "\"a\",\"b\"")); + EXPECT_TRUE(isIncarnationValue(Dialect::ETag, "\"abc\"")); + EXPECT_FALSE(isIncarnationValue(Dialect::Generation, "0")); + EXPECT_FALSE(isIncarnationValue(Dialect::Generation, "00123")); + EXPECT_FALSE(isIncarnationValue(Dialect::Generation, "\"123\"")); + EXPECT_FALSE(isIncarnationValue(Dialect::Generation, "123 ")); /// the ninth: decimal is not "decimal, trimmed" + EXPECT_TRUE(isIncarnationValue(Dialect::Generation, "123")); + EXPECT_FALSE(isIncarnationValue(Dialect::Emulated, "")); +} + +TEST(CASRetry, BackoffIsFullJitterUnderTheCap) +{ + for (uint32_t attempt = 1; attempt <= 12; ++attempt) + { + const uint64_t ceiling = std::min(5000, 200ull << (attempt - 1)); + uint64_t sum = 0; + std::set seen; + bool low = false; + bool high = false; + for (int i = 0; i < 1000; ++i) + { + const uint64_t s = Retry::backoff(attempt); + ASSERT_LE(s, ceiling); + sum += s; + seen.insert(s); + low = low || s < ceiling / 4; + high = high || s > ceiling * 3 / 4; + } + const double mean = static_cast(sum) / 1000.0; + EXPECT_GT(mean, static_cast(ceiling) * 0.35) << "attempt " << attempt; + EXPECT_LT(mean, static_cast(ceiling) * 0.65) << "attempt " << attempt; + /// The mean alone cannot tell full jitter from a constant half the ceiling, so the SPREAD is + /// asserted too: many distinct values, reaching into both the bottom and the top quarter. + EXPECT_GE(seen.size(), 3u) << "attempt " << attempt; + EXPECT_TRUE(low) << "attempt " << attempt; + EXPECT_TRUE(high) << "attempt " << attempt; + } +} + +TEST(CASRetry, PoliciesAreShapedAsSpecified) +{ + const uint64_t now = 1'000'000; + EXPECT_EQ(Retry::standard().bind(now).deadline_ms, now + 90'000); + EXPECT_FALSE(Retry::standard().bind(now).lease_bound); + EXPECT_FALSE(Retry::standard().single_attempt); + EXPECT_TRUE(Retry::once().single_attempt); + const Retry::Bound lease = Retry::untilLeaseSafe(now + 10'000, 2'000).bind(now); + EXPECT_EQ(lease.deadline_ms, now + 8'000); + EXPECT_TRUE(lease.lease_bound); + EXPECT_EQ(Retry::within(1'000).bind(now).deadline_ms, now + 1'000); +} + +/// A frozen policy is ONE absolute deadline: time passing does not buy a later one, freezing again +/// cannot extend it, and the lease bound still wins when it is the smaller of the two -- which is what +/// keeps `GaveUp::Source` able to say which bound refused. +TEST(CASRetry, AFrozenPolicyIsOneDeadlineAndTheLeaseStillWins) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const uint64_t start = clock.now; + const Retry frozen = op.freeze(Retry::standard()); + ASSERT_TRUE(frozen.policy_deadline_ms.has_value()); + EXPECT_EQ(frozen.bind(start).deadline_ms, start + 90'000); + EXPECT_EQ(frozen.bind(start + 50'000).deadline_ms, start + 90'000); + EXPECT_FALSE(frozen.bind(start + 50'000).lease_bound); + /// The single-attempt view of a frozen policy keeps the deadline rather than starting a window. + EXPECT_EQ(frozen.asSingleAttempt().policy_deadline_ms, frozen.policy_deadline_ms); + EXPECT_TRUE(frozen.asSingleAttempt().single_attempt); + + clock.now += 50'000; + EXPECT_EQ(op.freeze(frozen).policy_deadline_ms, frozen.policy_deadline_ms); + + const Retry::Bound leashed = op.freeze(Retry::untilLeaseSafe(start + 10'000, 2'000)).bind(clock.now); + EXPECT_EQ(leashed.deadline_ms, start + 8'000); + EXPECT_TRUE(leashed.lease_bound); +} + +/// Freezing belongs to a loop. A single verb still gets a full window from where it is called, however +/// long its caller has already been running. +TEST(CASRequests, ALoneReadUnderTheStandardPolicyStillGetsItsFullWindow) +{ + FakeClock clock; + auto throttled = std::make_shared( + std::make_shared(), ThrottlingBackend::Mode::EveryNth, 1, 429); + auto requests = makeRequests(throttled, clock); + auto op = requests.admit(); + + clock.now += 10 * 90'000; + const uint64_t start = clock.now; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_GE(clock.now - start, 85'000u); +} + +TEST(CASWriteResult, OrThrowMapsEveryAlternative) +{ + /// The two that are not failures: a commit hands back its incarnation, a decline hands back + /// nothing, and neither throws. Minting one needs a real write, since nothing else may mint. + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult committed = op.create("k", "v", Retry::standard()); + ASSERT_TRUE(std::holds_alternative(committed)); + const Etag landed = std::get(committed).etag; + const auto returned = orThrow(std::move(committed), "create"); + ASSERT_TRUE(returned.has_value()); + EXPECT_EQ(*returned, landed); + EXPECT_FALSE(orThrow(WriteResult{Declined{ProvenAbsent{}}}, "declined").has_value()); + + expectThrowsCode(DB::ErrorCodes::ABORTED, [&] { orThrow(WriteResult{Conflict{ProvenAbsent{}}}, "t"); }); + expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { orThrow(WriteResult{Refused{DB::ErrorCodes::S3_ERROR, "denied"}}, "t"); }); + /// Designated rather than positional: `GaveUp` grows fields at its end, and a positional list is + /// the form a field inserted anywhere else would silently re-interpret. + const GaveUp deadline{ + .why = GaveUp::Why::Deadline, .deadline_source = GaveUp::Source::Policy, + .sent_any = true, .last_seen = NotObserved{}}; + const GaveUp unresolved{ + .why = GaveUp::Why::Unresolved, .deadline_source = GaveUp::Source::Policy, + .sent_any = true, .last_seen = ProvenAbsent{}}; + const GaveUp fence_lost{ + .why = GaveUp::Why::FenceLost, .deadline_source = GaveUp::Source::Lease, + .sent_any = false, .last_seen = NotObserved{}}; + for (const GaveUp & gave_up : {deadline, unresolved, fence_lost}) + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { orThrow(WriteResult{gave_up}, "t"); }); +} + +TEST(CASFence, OpenFenceAdmitsEverythingAndNeverMoves) +{ + Fence f = Fence::open(); + EXPECT_EQ(f.generation(), 0u); + EXPECT_EQ(f.admit(0, 1'000'000), Fence::Admit::Ok); + EXPECT_NO_THROW(f.check_or_throw(0)); +} + +/// ================================================================================================ +/// The backend's keyed string primitives +/// ================================================================================================ + +TEST(CASBackendPrimitives, InMemoryWriteReadRemoveRoundTripThroughOneOperation) +{ + FakeClock clock; + auto b = std::make_shared(); + auto requests = makeRequests(b, clock); + auto op = requests.admit(); + + const std::optional w1 = orThrow(op.create("k", "v1", Retry::once()), "create"); + ASSERT_TRUE(w1); + const std::optional r = op.read("k", Retry::once()); + ASSERT_TRUE(r); + EXPECT_EQ(r->bytes, "v1"); + EXPECT_EQ(r->etag, *w1); + + const std::optional h = op.head("k", Retry::once()); + ASSERT_TRUE(h); + EXPECT_EQ(h->size, 2u); + EXPECT_EQ(h->etag, *w1); + + EXPECT_TRUE(std::holds_alternative(op.create("k", "v2", Retry::once()))); /// must be absent + const std::optional w3 = orThrow(op.replace("k", "v2", *w1, Retry::once()), "replace"); + ASSERT_TRUE(w3); + EXPECT_NE(*w3, *w1); /// incarnations never repeat + + EXPECT_EQ(op.remove("k", *w1, Retry::once()), Removal::Mismatch); + EXPECT_EQ(op.remove("k", *w3, Retry::once()), Removal::Removed); + EXPECT_EQ(op.remove("k", *w3, Retry::once()), Removal::Gone); + EXPECT_FALSE(op.read("k", Retry::once()).has_value()); +} + +TEST(CASBackendPrimitives, ListSurfacesTheIncarnationAndPaginates) +{ + FakeClock clock; + auto b = std::make_shared(); + auto requests = makeRequests(b, clock); + auto op = requests.admit(); + + const std::optional a = orThrow(op.create("p/a", "0123456789", Retry::once()), "create"); + ASSERT_TRUE(a); + orThrow(op.create("p/b", "xy", Retry::once()), "create"); + orThrow(op.create("q/c", "z", Retry::once()), "create"); + + const ListPage page = op.list("p/", "", 10, Retry::once()); + ASSERT_EQ(page.keys.size(), 2u); /// sorted, prefix-scoped + EXPECT_EQ(page.keys[0].key, "p/a"); + EXPECT_EQ(page.keys[0].size, 10u); + ASSERT_TRUE(page.keys[0].etag.has_value()); + EXPECT_EQ(*page.keys[0].etag, *a); + EXPECT_TRUE(page.next_cursor.empty()); + + const ListPage first = op.list("p/", "", 1, Retry::once()); + ASSERT_EQ(first.keys.size(), 1u); + EXPECT_EQ(first.next_cursor, "p/a"); + const ListPage second = op.list("p/", first.next_cursor, 1, Retry::once()); + ASSERT_EQ(second.keys.size(), 1u); + EXPECT_EQ(second.keys[0].key, "p/b"); +} + +TEST(CASBackendPrimitives, EveryBackendInstanceHasItsOwnId) +{ + auto a = std::make_shared(); + auto b = std::make_shared(); + EXPECT_NE(a->backendId(), b->backendId()); + EXPECT_NE(a->backendId(), 0u); + EXPECT_EQ(a->dialect(), Dialect::Emulated); +} + +/// The legacy verbs (`putIfAbsent`/`casPut`/`putOverwrite`) that used to forward through the primitive +/// `write` are gone -- `CasOperation` is the only caller of `Backend` now -- so that forwarding is a +/// type-level guarantee rather than a runtime check. What remains to prove is that every fault double +/// in this file that overrides `write` sees an ATTEMPT under either shape `CasOperation` can send: +/// unconditional (`create`) and Etag-conditioned (`replace`). +/// `EachWriteKnobIsKeyedAndOneShotOnThePrimitiveWrite` below covers both. + +TEST(CASBackendPrimitives, EachWriteKnobIsKeyedAndOneShotOnThePrimitiveWrite) +{ + /// A knob names a KEY, not a call site: the keyed `write` every write reaches, whichever + /// `CasOperation` verb (`create`/`replace`) issued it. + auto b = std::make_shared(); + FakeClock clock; + auto requests = makeRequests(b, clock); + auto op = requests.admit(); + + b->refuseNextWrite("k"); + EXPECT_TRUE(std::holds_alternative(op.create("k", "v", Retry::once()))); /// consumed here + EXPECT_TRUE(std::holds_alternative(op.create("k", "v", Retry::once()))); /// and only once + expectBytes(b, "k", "v"); + + b->refuseNextWrite("k2"); + EXPECT_TRUE(std::holds_alternative(op.create("k2", "v", Retry::once()))); + EXPECT_TRUE(std::holds_alternative(op.create("k2", "v", Retry::once()))); + + b->injectAmbiguousWrite("k3"); + EXPECT_TRUE(std::holds_alternative(op.create("k3", "v", Retry::once()))); + EXPECT_FALSE(op.read("k3", Retry::once()).has_value()) << "an ambiguous write leaves the store untouched"; + EXPECT_TRUE(std::holds_alternative(op.create("k3", "v", Retry::once()))); + + /// Both knobs on one key, each consumed by the next write in turn. + b->injectAmbiguousWrite("k4"); + b->refuseNextWrite("k4"); + EXPECT_TRUE(std::holds_alternative(op.create("k4", "v", Retry::once()))); + EXPECT_TRUE(std::holds_alternative(op.create("k4", "v", Retry::once()))); + EXPECT_TRUE(std::holds_alternative(op.create("k4", "v", Retry::once()))); + + /// The Etag-conditioned shape: every write above was unconditional (`create`), so none of them + /// could have caught a fault double that only intercepts `write` when it carries an + /// `expected_value` -- the shape `replace` alone sends. + const std::optional k5_first = orThrow(op.create("k5", "v", Retry::once()), "create"); + ASSERT_TRUE(k5_first); + b->refuseNextWrite("k5"); + EXPECT_TRUE(std::holds_alternative(op.replace("k5", "v2", *k5_first, Retry::once()))) + << "consumed here"; + const std::optional k5_second + = orThrow(op.replace("k5", "v2", *k5_first, Retry::once()), "replace"); /// and only once + ASSERT_TRUE(k5_second); + expectBytes(b, "k5", "v2"); +} + +TEST(CASBackendPrimitives, ReadRefusesAValueThatIsNotAnIncarnation) +{ + /// `read` hands back whatever the store said, malformed included -- `CasRequests::mint` is what + /// refuses it, naming the key, before any caller can see it as an `Etag`. + struct EmptyValueBackend : InMemoryBackend + { + std::optional read(const String &, TransportAccess &) override { return Raw{"body", ""}; } + }; + auto b = std::make_shared(); + FakeClock clock; + auto requests = makeRequests(b, clock); + auto op = requests.admit(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { op.read("k", Retry::once()); }); +} + +/// `InstrumentedBackendPassesALegacyCallThroughAsLegacy` pinned `InstrumentedBackend` delegating the +/// legacy `casPut` verb to its inner backend unconverted. `Backend` has no legacy verbs left -- +/// `InstrumentedBackend` is a pure primitive decorator now -- and its primitive delegation (`write` and +/// every other primitive, classified and counted) is what `CASInstrumentedBackend.ClassifierAndPerNamespaceOpEvents` +/// (gtest_cas_backend.cpp) pins. + +TEST(CASBackendPrimitives, RefreshCredentialsIsOffUntilAskedFor) +{ + auto b = std::make_shared(); + EXPECT_FALSE(b->refreshCredentials()); + b->setRefreshCredentialsResult(true); + EXPECT_TRUE(b->refreshCredentials()); +} + +#if USE_AWS_S3 + +TEST(CASThrottlingBackend, FirstPerKeyRefusesOnceAndTheCallStillSucceeds) +{ + FakeClock clock; + auto inner = std::make_shared(); + auto t = std::make_shared(inner, ThrottlingBackend::Mode::FirstPerKey, 0, 429); + auto requests = makeRequests(t, clock); + auto op = requests.admit(); + + orThrow(op.create("k2", "v", Retry::standard()), "create"); + EXPECT_EQ(t->refusals("k2"), 1u); + EXPECT_TRUE(op.read("k2", Retry::standard()).has_value()); + EXPECT_EQ(t->refusals("k2"), 1u) << "only the FIRST request naming a key is refused"; +} + +TEST(CASThrottlingBackend, RefusalsAreRetryableUnderBothStatuses) +{ + /// The property the seam exists for: a refusal must reach the engine as an AMBIGUOUS attempt, not + /// a definite failure. What proves it is that the engine REISSUES -- a definite failure would + /// surface unchanged, with the refusal still the only request the store ever saw. + for (const int status : {429, 503}) + { + FakeClock clock; + auto t = std::make_shared( + std::make_shared(), ThrottlingBackend::Mode::FirstPerKey, 0, status); + auto requests = makeRequests(t, clock); + auto op = requests.admit(); + + EXPECT_FALSE(op.head("k", Retry::standard()).has_value()) << "status " << status; + EXPECT_EQ(t->refusals("k"), 1u) << "status " << status; + } +} + +/// `PassesALegacyCallThroughAsLegacy` pinned `ThrottlingBackend` delegating the legacy `casPut` verb +/// unconverted. `Backend` has no legacy verbs left; `ThrottlingBackend`'s primitive pass-through is +/// pinned by `FirstPerKeyRefusesOnceAndTheCallStillSucceeds` above and `EveryNthRefusesOnThePeriodAcrossKeys` +/// below, both of which drive it through `CasOperation`. + +TEST(CASThrottlingBackend, EveryNthRefusesOnThePeriodAcrossKeys) +{ + FakeClock clock; + auto inner = std::make_shared(); + auto t = std::make_shared(inner, ThrottlingBackend::Mode::EveryNth, 3, 503); + auto requests = makeRequests(t, clock); + auto op = requests.admit(); + + EXPECT_FALSE(op.read("a", Retry::standard()).has_value()); + EXPECT_FALSE(op.read("b", Retry::standard()).has_value()); + /// The THIRD request is refused whatever it names; the engine reissues it as the fourth. + EXPECT_FALSE(op.read("c", Retry::standard()).has_value()); + EXPECT_EQ(t->refusals("c"), 1u); + EXPECT_EQ(t->refusals("a"), 0u); + EXPECT_EQ(t->refusals("b"), 0u); +} + +#endif + +/// ================================================================================================ +/// The request engine +/// ================================================================================================ + +namespace +{ + +/// A type nothing in the engine catches, so a `decide` that throws it can only reach the caller by +/// propagating unchanged. +struct DecideMarker +{ +}; + +/// Answers the FIRST remove with a mismatch without reaching the store, so `removeCurrent` has to +/// re-observe. Counts its own requests: an answer given here never reaches the counting base. +struct MismatchOnceOnRemoveBackend : InMemoryBackend +{ + using InMemoryBackend::head; + + size_t heads = 0; + size_t removes = 0; + bool refuse_next_remove = true; + + std::optional head(const String & key, TransportAccess & access) override + { + ++heads; + return InMemoryBackend::head(key, access); + } + + RawRemoval remove(const String & key, const String & expected_value, TransportAccess & access) override + { + ++removes; + if (std::exchange(refuse_next_remove, false)) + return RawRemoval::Mismatch; + return InMemoryBackend::remove(key, expected_value, access); + } +}; + +/// Answers `Indeterminate` for its first `indeterminate_answers` probes, then delegates -- a store +/// briefly out of reach, whose absence was never established. +struct IndeterminateProbeBackend : InMemoryBackend +{ + using Backend::probeSentinelRaw; + + size_t probes = 0; + size_t indeterminate_answers = 2; + + SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) override + { + if (++probes <= indeterminate_answers) + return {ProbeOutcome::Indeterminate, std::nullopt}; + return InMemoryBackend::probeSentinelRaw(key, access); + } +}; + +/// Refuses the FIRST `list` naming each distinct cursor -- one refusal per page -- and charges every +/// list a fixed slice of the caller's clock, so what a page costs is a fact rather than a jitter draw. +/// `always_refuse_cursor` keeps one page refused for good. +struct PagedThrottleBackend : InMemoryBackend +{ + using InMemoryBackend::list; + + std::function charge_latency; + std::set refused_cursors; + std::optional always_refuse_cursor; + size_t list_calls = 0; + + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + ++list_calls; + if (charge_latency) + charge_latency(); + if ((always_refuse_cursor && *always_refuse_cursor == cursor) || refused_cursors.insert(cursor).second) + throw Poco::TimeoutException("the list resuming after '" + cursor + "' timed out"); + return InMemoryBackend::list(prefix, cursor, limit, access); + } +}; + +/// Runs `on_read` after every read. The resolve read is where a caller's own facts can change +/// between an attempt and the pause that would precede the next one. +struct FlipOnReadBackend : CountingBackend +{ + std::function on_read; + + std::optional read(const String & key, TransportAccess & access) override + { + auto raw = CountingBackend::read(key, access); + if (on_read) + on_read(); + return raw; + } +}; + +} + +TEST(CASIncarnation, RenderAndPersistedCompare) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const Etag first = *orThrow(op.create("k", "v", Retry::standard()), "create"); + EXPECT_EQ(first.render(), "emulated:1"); + EXPECT_EQ(first.key(), "k"); + EXPECT_EQ(first.dialect(), Dialect::Emulated); + + const PersistedEtag persisted = PersistedEtag::capture(first); + EXPECT_EQ(persisted.dialect, "emulated"); + EXPECT_EQ(persisted.value, "1"); + EXPECT_TRUE(persisted.matches(first)); + + const Etag second = *orThrow(op.replace("k", "w", first, Retry::standard()), "replace"); + EXPECT_EQ(second.render(), "emulated:2"); + EXPECT_FALSE(persisted.matches(second)); /// a captured record never re-matches a later incarnation + EXPECT_TRUE(PersistedEtag::capture(second).matches(second)); +} + +TEST(CASRetry, BindSaturatesAndLeavesAnEqualLeaseOffTheLeaseSource) +{ + constexpr uint64_t largest = std::numeric_limits::max(); + /// A window one short of the whole range, so any `now` above 1 overflows a naive addition. + EXPECT_EQ(Retry::within(largest - 1).bind(2).deadline_ms, largest); + EXPECT_EQ(Retry::within(largest - 1).bind(1).deadline_ms, largest); + EXPECT_FALSE(Retry::within(largest - 1).bind(2).lease_bound); + + const uint64_t now = 1'000'000; + /// The lease bound lands exactly on the policy deadline. The lease is taken only when it is + /// STRICTLY smaller, so the tie belongs to the policy and `GaveUp` will not name the lease. + const Retry::Bound tie = Retry::untilLeaseSafe(now + 92'000, 2'000).bind(now); + EXPECT_EQ(tie.deadline_ms, now + 90'000); + EXPECT_FALSE(tie.lease_bound); + + const Retry::Bound lease = Retry::untilLeaseSafe(now + 91'999, 2'000).bind(now); + EXPECT_EQ(lease.deadline_ms, now + 89'999); + EXPECT_TRUE(lease.lease_bound); +} + +TEST(CASRequests, CreateThenReplaceThenRemove) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const Etag first = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + const auto seen = op.read("k", Retry::standard()); + ASSERT_TRUE(seen.has_value()); + EXPECT_EQ(seen->bytes, "v1"); + EXPECT_EQ(seen->etag, first); + + const Etag second = *orThrow(op.replace("k", "v2", first, Retry::standard()), "replace"); + EXPECT_NE(second, first); + + EXPECT_EQ(op.remove("k", first, Retry::standard()), Removal::Mismatch); /// the incarnation is stale + EXPECT_EQ(op.remove("k", second, Retry::standard()), Removal::Removed); + EXPECT_EQ(op.remove("k", second, Retry::standard()), Removal::Gone); + EXPECT_FALSE(op.read("k", Retry::standard()).has_value()); +} + +/// An incarnation observed for one key is refused as the precondition for another, before the write +/// loop starts anything. Constructing a `LOGICAL_ERROR` exception ABORTS under a debug or sanitizer +/// build, so the same contract is asserted there as a death expectation; both forms pin that the +/// refusal happens, and the non-death form additionally pins that it costs no request. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASRequests, KeyBindingThrowsBeforeAnyRequest) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag of_a = *orThrow(op.create("a", "v", Retry::standard()), "create"); + backend->resetCounts(); + + expectThrowsCode(DB::ErrorCodes::LOGICAL_ERROR, [&] { (void)op.replace("b", "w", of_a, Retry::standard()); }); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} +#else +TEST(CASRequestsDeathTest, KeyBindingThrowsBeforeAnyRequest) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag of_a = *orThrow(op.create("a", "v", Retry::standard()), "create"); + + EXPECT_DEATH({ (void)op.replace("b", "w", of_a, Retry::standard()); }, ""); +} +#endif + +TEST(CASRequests, EveryConflictIsSettledByOneReadAndCarriesTheOccupant) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "theirs", Retry::standard()), "create"); + backend->resetCounts(); + + WriteResult result = op.create("k", "mine", Retry::once()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + const auto * occupant = std::get_if(&conflict->seen); + ASSERT_NE(occupant, nullptr); + EXPECT_EQ(occupant->bytes, "theirs"); + + /// The refused precondition says only that the key is taken; ONE exact read says by whom. + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); +} + +TEST(CASRequests, AmbiguousCreateThatLandedIsCommittedByTheResolveRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + /// The object becomes durable and THEN the response is lost, so the store holds bytes the caller + /// never learned it wrote -- the only ambiguity a resolve read can settle as a commit. + backend->injectAmbiguousLandedWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_TRUE(committed->resolved_by_read); + EXPECT_EQ(committed->attempts_sent, 1u); + /// Settled by reading, never by writing again: a second create would have conflicted with the + /// first one's own object. + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, AmbiguousCreateThatNeverLandedIsReissued) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); /// the attempt's outcome is lost and the store is untouched + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_EQ(backend->getTotal(), 1u); /// the resolve proved absence, and only then did a reissue follow + EXPECT_EQ(clock.sleeps.size(), 1u); +} + +/// The engine's own attempt number reaches the transport through `TransportAccess::attemptNo()`, for +/// every primitive -- write, read (the resolve read is its own call, with its own attempt count) and +/// list. +TEST(CASRequests, TheTransportSeesTheEngineAttemptNumber) +{ + struct AttemptRecordingBackend : CountingBackend + { + std::vector write_attempts, read_attempts, list_attempts; + std::expected write(const String & key, const String & bytes, + const std::optional & expected, TransportAccess & access) override + { + write_attempts.push_back(access.attemptNo()); + return CountingBackend::write(key, bytes, expected, access); + } + std::optional read(const String & key, TransportAccess & access) override + { + read_attempts.push_back(access.attemptNo()); + return CountingBackend::read(key, access); + } + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + list_attempts.push_back(access.attemptNo()); + return CountingBackend::list(prefix, cursor, limit, access); + } + }; + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("read timed out"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("k", "v", Retry::standard()))); + /// Attempt 1 ambiguous, attempt 2 commits. The settle read is its OWN read call: attempt 1 failed, 2 answered. + EXPECT_EQ(backend->write_attempts, (std::vector{1, 2})); + EXPECT_EQ(backend->read_attempts, (std::vector{1, 2})); + backend->list_attempts.clear(); + (void)op.list("p/", "", 10, Retry::standard()); + EXPECT_EQ(backend->list_attempts, (std::vector{1})); +} + +TEST(CASRequests, OnceSendsOneWriteAndAtMostOneResolveRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("the read timed out"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + /// One attempt is one attempt, but the read that would have settled it is still owed and sent. + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, DecideMayThrowAndTheExceptionPropagatesUnchanged) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + EXPECT_THROW( + op.readModifyWrite("k", [](const std::optional &) -> std::optional { throw DecideMarker{}; }, + Retry::standard()), + DecideMarker); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 1u); /// the key was read, and nothing was decided about it +} + +TEST(CASRequests, OnPresenceIssuesHeadsAndNoGet) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional & current) -> std::optional + { + return current ? std::nullopt : std::optional("v"); + }, + Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->getTotal(), 0u); + /// One HEAD decided it and one write landed it: the loop issues no request it does not need. + EXPECT_EQ(backend->headTotal(), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); +} + +TEST(CASRequests, OnPresenceSettlesARefusedPreconditionWithAHead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->refuseNextWrite("k"); /// the store refuses the precondition, writing nothing + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional & current) -> std::optional + { + return current ? std::nullopt : std::optional("v"); + }, + Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); + /// A refused precondition needs only to know WHAT is at the key, so this loop never fetches a body. + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->headTotal(), 2u); + EXPECT_EQ(backend->writeTotal(), 2u); +} + +TEST(CASRequests, ForEachListedKeyStopsEarlyAndBudgetsPerPage) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + for (int i = 0; i < 25; ++i) + orThrow(op.create("p/" + std::to_string(i), "v", Retry::standard()), "create"); + backend->resetCounts(); + + size_t seen = 0; + size_t pages = 0; + op.forEachListedKey("p/", [&](const ListedKey &) { return ++seen < 3; }, Retry::standard(), + /*page_limit=*/10, [&] { ++pages; }); + EXPECT_EQ(seen, 3u); + /// The walk stops where the caller stops it: the remaining two pages are never fetched. + EXPECT_EQ(pages, 1u); + EXPECT_EQ(backend->listTotal(), 1u); +} + +TEST(CASRequests, DeleteMarkerIsANamedException) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag inc = *orThrow(op.create("k", "v", Retry::standard()), "create"); + + backend->setSimulateDeleteMarkers(true); + expectThrowsCode(DB::ErrorCodes::CAS_DELETE_MARKER, [&] { (void)op.remove("k", inc, Retry::standard()); }); + EXPECT_TRUE(clock.sleeps.empty()); /// a versioned bucket answers this way every time +} + +TEST(CASRequests, RemoveCurrentReObservesAMismatchAndRefusesUnderOnce) +{ + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + + EXPECT_EQ(op.removeCurrent("k", Retry::standard()), Removal::Removed); + /// Another incarnation became current between the observation and the delete: observe again, + /// paced like every other reissue, and delete what the second look saw. + EXPECT_EQ(backend->heads, 2u); + EXPECT_EQ(backend->removes, 2u); + EXPECT_EQ(clock.sleeps.size(), 1u); + } + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + + /// `once` has no reissue with which to settle a mismatch, and this verb never hands one back. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)op.removeCurrent("k", Retry::once()); }); + EXPECT_EQ(backend->heads, 1u); + EXPECT_EQ(backend->removes, 1u); + EXPECT_TRUE(clock.sleeps.empty()); + } +} + +TEST(CASRequests, ProbeSentinelRetriesOnlyTheIndeterminateOutcome) +{ + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + + const SentinelProbeResult result = op.probeSentinel("k", Retry::standard()); + EXPECT_EQ(result.outcome, ProbeOutcome::Present); + ASSERT_TRUE(result.body.has_value()); + EXPECT_EQ(*result.body, "v"); + EXPECT_EQ(backend->probes, 3u); /// inconclusive twice, then an authoritative answer + EXPECT_EQ(clock.sleeps.size(), 2u); + } + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + + /// With no reissue left, the inconclusive outcome IS the answer: reported, never thrown. + const SentinelProbeResult result = op.probeSentinel("k", Retry::once()); + EXPECT_EQ(result.outcome, ProbeOutcome::Indeterminate); + EXPECT_EQ(backend->probes, 1u); + EXPECT_TRUE(clock.sleeps.empty()); + } +} + +TEST(CASRequests, AdmissionIsCheckedAtThreePoints) +{ + FakeClock clock; + auto backend = std::make_shared(); + uint64_t generation = 1; + bool lost = false; + Fence fence{ + [&] { return generation; }, + [&](uint64_t admitted, uint64_t) + { + return (lost || admitted != generation) ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; + }, + [&](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + /// The store is observed through an OPEN fence: these checks run while the subject's own fence is + /// closed, and a fenced read would report the fence rather than the store. + auto observer_requests = makeRequests(backend, clock); + auto observer = observer_requests.admit(); + + /// (1) before the first attempt, on a handle resumed under a generation the fence has moved past + { + auto op = requests.resume(0); + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_FALSE(observer.read("k", Retry::once()).has_value()); + } + /// (2) before the next verb of an admitted handle, after a re-arm between two verbs + { + auto op = requests.admit(); + EXPECT_FALSE(op.head("k", Retry::standard()).has_value()); + generation = 2; + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_FALSE(observer.read("k", Retry::once()).has_value()); + } + /// (3) after a proven commit: the write landed, then the fence tripped before the call returned + { + auto op = requests.admit(); + backend->onWriteCommitted("k2", [&] { lost = true; }); + WriteResult result = op.create("k2", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + /// The object IS durable. This call refuses to CLAIM it; it does not undo it. + EXPECT_TRUE(observer.read("k2", Retry::once()).has_value()); + } +} + +TEST(CASRequests, TheGateBeforeTheSleepEndsTheCallWithoutASecondWrite) +{ + FakeClock clock; + auto backend = std::make_shared(); + bool alive = true; + backend->on_read = [&] { alive = false; }; + backend->injectAmbiguousWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit([&] { return alive; }); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + /// The ambiguous attempt was resolved, and the pause before the reissue was refused rather than + /// served: no sleep, and no second attempt after it. + EXPECT_TRUE(clock.sleeps.empty()); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); +} + +TEST(CASRequests, AResolveReadRefusedForLeaseBudgetIsReportedAsTheLeaseDeadline) +{ + FakeClock clock; + auto backend = std::make_shared(); + bool lease_spent = false; + Fence fence{ + [] { return uint64_t{0}; }, + [&](uint64_t, uint64_t) { return lease_spent ? Fence::Admit::NoBudget : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "v", Retry::standard()), "create"); + + /// The store refuses the precondition, and the lease budget is gone by the time the read that + /// would say WHO holds the key is due. The call learned nothing about the key, so what it reports + /// is the bound that stopped it -- not a conflict it never observed. + backend->refuseNextWrite("k"); + backend->onBeforeWrite("k", [&] { lease_spent = true; }); + const uint64_t lease_deadline = clock.now + 10'000; + WriteResult result = op.replace("k", "w", seen, Retry::untilLeaseSafe(lease_deadline, 2'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Lease); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + EXPECT_TRUE(clock.sleeps.empty()); + EXPECT_EQ(backend->writeTotal(), 2u); /// the create and the one refused replace + EXPECT_EQ(backend->getTotal(), 0u); /// the resolve read never started +} + +TEST(CASRequests, AFenceWithNoBudgetForTheRequestSendsNothingAndNamesTheLease) +{ + FakeClock clock; + auto backend = std::make_shared(); + const uint64_t budget_ms = 500; + Fence fence{ + [] { return uint64_t{0}; }, + [&](uint64_t, uint64_t needed_ms) { return needed_ms > budget_ms ? Fence::Admit::NoBudget : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + /// One attempt reserves more than the lease has left, so nothing may be started under it. + requests.setAttemptReservationForTest(1'000); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + /// The policy's own window is untouched; what ran out is the fence's budget, which IS the lease. + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Lease); + EXPECT_FALSE(gave_up->sent_any); + + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, AnRmwWhoseFirstReadFailsGivesUpUnresolvedWithoutWriting) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("the read timed out"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.readModifyWrite("k", + [](const std::optional &) -> std::optional { return String("v"); }, Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + /// No BOUND refused this read; the read itself failed. Claiming a deadline the clock never reached + /// would send its reader to widen the wrong thing. + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + EXPECT_EQ(backend->writeTotal(), 0u); +} + +TEST(CASRequests, AnOnPresenceRmwWhoseFirstHeadFailsGivesUpUnresolvedWithoutWriting) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextHeadWith("k", std::make_exception_ptr(Poco::TimeoutException("the head timed out"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional &) -> std::optional { return String("v"); }, Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u); /// the presence loop does not fall back to a body read +} + +TEST(CASRequests, AConflictWhoseResolveReadFailsIsReportedWithNothingObserved) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "theirs", Retry::standard()), "create"); + + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("the read timed out"))); + WriteResult result = op.create("k", "mine", Retry::once()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + /// The precondition was refused, so the key IS taken; the read that would have said by whom failed, + /// and the caller is told exactly that rather than handed a guess about the occupant. + EXPECT_TRUE(std::holds_alternative(conflict->seen)); +} + +TEST(CASRequests, AFenceLostDuringTheResolveReadIsAFenceLossNotAConflict) +{ + FakeClock clock; + auto backend = std::make_shared(); + bool alive = true; + /// The fence trips while the ambiguous attempt is in flight: the write's own hook runs before the + /// store is touched, so the resolve read is the first request to meet the closed gate. + backend->onBeforeWrite("k", [&] { alive = false; }); + backend->injectAmbiguousWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit([&] { return alive; }); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + /// A lost fence is not an observation. Reporting it as an ordinary conflict would tell the caller + /// somebody else holds the key, when what happened is that this node stopped being allowed to ask. + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 0u); /// refused before the resolve read was issued + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, OnPresenceFetchesTheBodyToProveAnAmbiguousAttemptLanded) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousLandedWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional & current) -> std::optional + { + return current ? std::nullopt : std::optional("v"); + }, + Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_TRUE(committed->resolved_by_read); + /// Presence-only is what this loop REPORTS, not a promise about what it may read: only the bytes + /// can prove the ambiguous attempt was this call's own. + EXPECT_EQ(backend->getTotal(), 1u); +} + +TEST(CASRequests, OnPresenceReportsMetaEvenWhenItHadToFetchTheBody) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto competitor = makeRequests(backend, clock); + auto rival = competitor.admit(); + + /// A competitor takes the key while our own create is in flight, and that create's own fate is + /// lost. The ambiguity is armed from inside the hook so the competitor's write cannot consume it. + bool staged = false; + std::optional rival_etag; + backend->onBeforeWrite("k", [&] + { + if (staged) + return; + staged = true; + const WriteResult rival_result = rival.create("k", "theirs", Retry::once()); + const auto * rival_committed = std::get_if(&rival_result); + ASSERT_NE(rival_committed, nullptr); + rival_etag = rival_committed->etag; + backend->injectAmbiguousWrite("k"); + }); + + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional &) -> std::optional { return String("mine"); }, Retry::once()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + /// The ambiguity forced a body read, and the body stops at this boundary: a caller of the + /// presence loop can never come to depend on bytes the loop does not promise. `get_if` plus + /// its field checks, not a bare `holds_alternative`: a variant that already proved it holds `Meta` + /// cannot also hold `Object`, so the field checks are what a regression could actually fail -- + /// proving the observed Meta is the RIVAL's own committed incarnation, not some other object. + ASSERT_TRUE(rival_etag.has_value()); + const auto * meta_seen = std::get_if(&conflict->seen); + ASSERT_NE(meta_seen, nullptr); + EXPECT_EQ(meta_seen->etag, *rival_etag); + EXPECT_EQ(meta_seen->size, String("theirs").size()); + EXPECT_EQ(backend->getTotal(), 1u); +} + +TEST(CASRequests, AmbiguousReplaceWhoseResolveShowsThePreconditionUnchangedIsReissued) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + backend->resetCounts(); + + /// The attempt's fate is lost and the store is untouched. The incarnation it named is still the + /// current one -- which proves nothing landed, and leaves a precondition a reissue can still meet. + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + WriteResult result = op.replace("k", "v2", seen, Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(backend->writeTotal(), 2u); + EXPECT_EQ(backend->getTotal(), 1u); /// exactly one resolve read, and it settled the ambiguity + EXPECT_EQ(clock.sleeps.size(), 1u); +} + +TEST(CASRequests, AmbiguousReplaceOfIdenticalBytesIsReissuedNotClaimedByByteEquality) +{ + /// The key already holds exactly the bytes we are about to write, so byte equality alone can never + /// say whether the ambiguous attempt landed. The incarnation can: an attempt that applied would + /// have moved it. Under a policy with a reissue that means re-sending; under `once` it means saying + /// the write is unresolved rather than claiming somebody else's identical object. + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "B", Retry::standard()), "create"); + backend->resetCounts(); + + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + WriteResult result = op.replace("k", "B", seen, Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + /// Claiming the resolve read's object would have reported one attempt and a commit this call + /// never made; the reissue is what actually put these bytes there under a new incarnation. + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_NE(committed->etag, seen); + EXPECT_EQ(backend->writeTotal(), 2u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_EQ(clock.sleeps.size(), 1u); + } + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "B", Retry::standard()), "create"); + backend->resetCounts(); + + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + WriteResult result = op.replace("k", "B", seen, Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); + } +} + +TEST(CASRequests, AmbiguousReplaceWhoseResolveShowsAnotherIncarnationIsAConflict) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag stale = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + orThrow(op.replace("k", "theirs", stale, Retry::standard()), "the competitor's replace"); + backend->resetCounts(); + + /// The attempt's fate is lost, and the key has moved past the incarnation it named: no reissue of + /// it could ever apply, so the ambiguity is settled and the occupant is the answer. + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + WriteResult result = op.replace("k", "mine", stale, Retry::standard()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + const auto * occupant = std::get_if(&conflict->seen); + ASSERT_NE(occupant, nullptr); + EXPECT_EQ(occupant->bytes, "theirs"); + EXPECT_NE(occupant->etag, stale); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + /// The count the conflict reports is the count the transport saw, not a constant that happens to + /// match here: a caller totalling attempts across endings has to be able to add this one. + EXPECT_EQ(conflict->attempts_sent, backend->writeTotal()); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ReadModifyWriteDoesNotClaimACompetitorsIdenticalBytesAfterAnEarlierAmbiguity) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto competitor = makeRequests(backend, clock); + auto rival = competitor.admit(); + + /// The competitor moves the key once before each of our two attempts, and its own writes re-enter + /// this hook. The ambiguity is armed here rather than up front so the competitor's create cannot + /// consume the arming meant for ours. + bool inside = false; + int staged = 0; + backend->onBeforeWrite("k", [&] + { + if (inside) + return; + inside = true; + if (staged == 0) + { + (void)rival.create("k", "X", Retry::once()); + backend->injectAmbiguousWrite("k"); + } + else if (staged == 1) + { + /// The bytes we are about to send, under an incarnation that is not ours. + if (const auto current = rival.read("k", Retry::once())) + (void)rival.replace("k", "B", current->etag, Retry::once()); + } + ++staged; + inside = false; + }); + + std::vector decided_on; + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.readModifyWrite("k", + [&](const std::optional & current) -> std::optional + { + decided_on.push_back(current ? current->bytes : String("")); + if (!current) + return String("A"); + if (current->bytes == "X") + return String("B"); + return std::nullopt; + }, + Retry::standard()); + + /// The only ambiguity this call had belonged to "A", and the competitor's "X" already proved it + /// dead. "B" at the key is the competitor's, so the loop re-decides on it instead of claiming it. + const auto * declined = std::get_if(&result); + ASSERT_NE(declined, nullptr); + const auto * seen = std::get_if(&declined->seen); + ASSERT_NE(seen, nullptr); + EXPECT_EQ(seen->bytes, "B"); + EXPECT_EQ(decided_on, (std::vector{"", "X", "B"})); +} + +TEST(CASRequests, AnUnmodeledLocalExceptionOnAWritePropagatesUnchanged) +{ + FakeClock clock; + auto backend = std::make_shared(); + /// Not a `Poco::Exception`, so it did not come from the transport and cannot have landed anything. + /// Settling it by a read would report a store answer the store never gave. + backend->failNextWriteWith("k", std::make_exception_ptr(std::logic_error("a local bug"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + EXPECT_THROW((void)op.create("k", "v", Retry::standard()), std::logic_error); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ForEachListedKeyGivesEachPageItsOwnPolicyWindow) +{ + FakeClock clock; + auto backend = std::make_shared(); + /// Every list costs the caller 300ms, so a page's cost is a fact and not a jitter draw. + backend->charge_latency = [&clock] { clock.now += 300; }; + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + for (int i = 0; i < 25; ++i) + orThrow(op.create("p/" + std::to_string(i), "v", Retry::standard()), "create"); + + size_t seen = 0; + size_t pages = 0; + const uint64_t start = clock.now; + /// A window that comfortably covers ONE page's refusal and its reissue, and could not have covered + /// the walk: the policy governs each page, because a walk is an unbounded number of requests. + op.forEachListedKey("p/", [&](const ListedKey &) { ++seen; return true; }, Retry::within(1'000), + /*page_limit=*/10, [&] { ++pages; }); + EXPECT_EQ(seen, 25u); + EXPECT_EQ(pages, 3u); + EXPECT_EQ(backend->list_calls, 6u); /// each page refused once, then delivered + EXPECT_GT(clock.now - start, 1'000u); /// the walk outlived the window every page was given +} + +TEST(CASRequests, ForEachListedKeyThrowsRatherThanTruncateWhenAPageNeverArrives) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->charge_latency = [&clock] { clock.now += 300; }; + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + for (int i = 0; i < 25; ++i) + orThrow(op.create("p/" + std::to_string(i), "v", Retry::standard()), "create"); + + const ListPage first = op.list("p/", "", 10, Retry::within(1'000)); + ASSERT_FALSE(first.next_cursor.empty()); + backend->always_refuse_cursor = first.next_cursor; /// the second page never arrives + backend->refused_cursors.clear(); + + size_t seen = 0; + size_t pages = 0; + /// A silently truncated enumeration is the error a coverage record exists to prevent, so the walk + /// reports the page it could not fetch instead of returning what it managed to read. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] + { + op.forEachListedKey("p/", [&](const ListedKey &) { ++seen; return true; }, Retry::within(1'000), + /*page_limit=*/10, [&] { ++pages; }); + }); + EXPECT_EQ(pages, 1u); + EXPECT_EQ(seen, 10u); +} + +TEST(CASRequests, LivenessPredicateEndsTheOperationLikeAFenceLoss) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + bool alive = true; + auto op = requests.admit([&] { return alive; }); + EXPECT_TRUE(op.admitted()); + + alive = false; + EXPECT_FALSE(op.admitted()); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 0u); + + /// The read surface reports the same refusal the only way it can: by exception. + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_EQ(backend->getTotal(), 0u); +} + +TEST(CASRequests, ReadModifyWriteLosesNoIncrementUnderContentionAndBoundsAHotKey) +{ + auto backend = std::make_shared(); + const auto increment = [](const std::optional & current) -> std::optional + { + return std::to_string(std::stoi(current ? current->bytes : "0") + 1); + }; + + /// The real clock and the real sleep: two threads share this engine, and a `FakeClock` would be a + /// data race on both of its fields. + CasRequests contended(backend, Fence::open()); + { + auto seed = contended.admit(); + orThrow(seed.create("ctr", "0", Retry::standard()), "create"); + } + const auto fifty_increments = [&] + { + auto op = contended.admit(); + for (int i = 0; i < 50; ++i) + orThrow(op.readModifyWrite("ctr", increment, Retry::standard()), "increment"); + }; + std::thread first(fifty_increments); + std::thread second(fifty_increments); + first.join(); + second.join(); + + auto reader = contended.admit(); + const auto counted = reader.read("ctr", Retry::standard()); + ASSERT_TRUE(counted.has_value()); + EXPECT_EQ(counted->bytes, "100"); /// every conflict re-decided against what the resolve read saw + + /// A key rewritten under EVERY attempt is bounded by the deadline instead of looping forever. + FakeClock clock; + auto hot = makeRequests(backend, clock); + auto competitor = makeRequests(backend, clock); + auto rival = competitor.admit(); + bool inside_hook = false; + backend->onBeforeWrite("ctr", [&] + { + if (inside_hook) /// the hook's own write re-enters this callback + return; + inside_hook = true; + if (const auto current = rival.read("ctr", Retry::once())) + (void)rival.replace("ctr", "999", current->etag, Retry::once()); + inside_hook = false; + }); + + auto op = hot.admit(); + WriteResult result = op.readModifyWrite("ctr", increment, Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_FALSE(clock.sleeps.empty()); /// it paced its retries rather than spinning +} + +namespace +{ + +/// Moves `key` under the caller before each of its first `moves` write attempts, and optionally makes +/// each of those attempts ambiguous (the store never answers it) so the resolve read is what settles +/// the race. +struct RaceMaker +{ + RaceMaker(std::shared_ptr backend_, FakeClock & clock, String key_, int moves_, bool ambiguous_) + : backend(std::move(backend_)), key(std::move(key_)), moves(moves_), ambiguous(ambiguous_) + , rival_requests(makeRequests(backend, clock)), rival(rival_requests.admit()) + { + backend->onBeforeWrite(key, [this] + { + if (inside || made >= moves) + return; + inside = true; + if (const auto current = rival.read(key, Retry::once())) + (void)rival.replace(key, current->bytes + "r", current->etag, Retry::once()); + else + (void)rival.create(key, "r", Retry::once()); + if (ambiguous) + backend->injectAmbiguousWrite(key); + ++made; + inside = false; + }); + } + + std::shared_ptr backend; + String key; + int moves; + bool ambiguous; + int made = 0; + bool inside = false; + CasRequests rival_requests; + CasOperation rival; +}; + +DecideOnObject appendX() +{ + return [](const std::optional & current) -> std::optional + { + return current ? current->bytes + "x" : String("x"); + }; +} + +} + +TEST(CASRequests, CleanConflictsArePacedFlatAndDoNotAdvanceTheReissueCounter) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); + constexpr int K = 4; + RaceMaker races(backend, clock, "k", K, /*ambiguous=*/false); + const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load(); + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + + WriteResult result = op.readModifyWrite("k", appendX(), Retry::standard()); + + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load() - pauses_before, K); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 0u); + ASSERT_EQ(clock.sleeps.size(), static_cast(K)); + for (uint64_t s : clock.sleeps) + EXPECT_LE(s, 200u); /// flat: every pause is one `backoff(1)` draw, whatever the loss count + EXPECT_EQ(backend->writeCount("k"), 1u + K + 1u + K); /// seed, K refused, K rival moves, the one that landed +} + +TEST(CASRequests, AConflictThatSettledAFaultKeepsTheGrowingSchedule) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); + constexpr int K = 3; + RaceMaker races(backend, clock, "k", K, /*ambiguous=*/true); + const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load(); + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + + WriteResult result = op.readModifyWrite("k", appendX(), Retry::standard()); + + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load() - pauses_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, K); + ASSERT_EQ(clock.sleeps.size(), static_cast(K)); + for (size_t i = 0; i < clock.sleeps.size(); ++i) + EXPECT_LE(clock.sleeps[i], std::min(5000, 200ull << i)) << "reissue " << i; +} + +TEST(CASRequests, ReplaceReportsWhetherAConflictSettledAFault) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seed = *orThrow(op.create("k", "v", Retry::standard()), "seed"); + { + RaceMaker clean(backend, clock, "k", 1, /*ambiguous=*/false); + WriteResult result = op.replace("k", "w", seed, Retry::standard()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + EXPECT_FALSE(conflict->any_ambiguous); + EXPECT_EQ(conflict->attempts_sent, 1u); + } + { + RaceMaker faulty(backend, clock, "k", 1, /*ambiguous=*/true); + WriteResult result = op.replace("k", "w", seed, Retry::standard()); + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + EXPECT_TRUE(conflict->any_ambiguous); + EXPECT_EQ(conflict->attempts_sent, 1u); /// one attempt, lost, settled as moved: `attempts_sent` cannot tell + } +} + +TEST(CASRequests, OnPresenceUnderOnceKeepsTheFaultFlagOnTheRebuiltConflict) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); + RaceMaker faulty(backend, clock, "k", 1, /*ambiguous=*/true); + + WriteResult result = op.readModifyWriteOnPresence("k", + [](const std::optional &) -> std::optional { return String("w"); }, Retry::once()); + + const auto * conflict = std::get_if(&result); + ASSERT_NE(conflict, nullptr); + EXPECT_TRUE(conflict->any_ambiguous); + EXPECT_TRUE(std::holds_alternative(conflict->seen)); /// presence-only, as before +} + +TEST(CASRequests, CleanConflictsBeforeAFaultDoNotInflateTheFaultsFirstBackoff) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); + constexpr int K = 3; + RaceMaker races(backend, clock, "k", K, /*ambiguous=*/false); + /// After the K clean races the next attempt is ambiguous with the precondition unchanged, so the + /// engine reissues it; that reissue's pause must be the schedule's first, not its (K+1)-th. + bool armed = false; + backend->onBeforeWrite("k", [&] + { + /// The RaceMaker's hook is replaced by this one; it moves the key itself for the first K writes. + if (races.inside) + return; + if (races.made < K) + { + races.inside = true; + if (const auto current = races.rival.read("k", Retry::once())) + (void)races.rival.replace("k", current->bytes + "r", current->etag, Retry::once()); + ++races.made; + races.inside = false; + return; + } + if (!armed) + { + armed = true; + backend->injectAmbiguousWrite("k"); + } + }); + + WriteResult result = op.readModifyWrite("k", appendX(), Retry::standard()); + + ASSERT_TRUE(std::holds_alternative(result)); + ASSERT_EQ(clock.sleeps.size(), static_cast(K + 1)); + EXPECT_LE(clock.sleeps[K], 200u) << "the first transport reissue sleeps within backoff(1)"; +} + +TEST(CASRequests, ADeterministicLocalFailureSurfacesUnchangedWithoutAReissue) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextReadWith("k", std::make_exception_ptr( + DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "the object at 'k' is not decodable"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + /// Reissuing would replay the same bug and bury it behind a retryable exception at the deadline. + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ATransportTimeoutIsReissuedAndALocalFailureIsNot) +{ + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + backend->resetCounts(); + + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("the read timed out"))); + const auto seen = op.read("k", Retry::standard()); + ASSERT_TRUE(seen.has_value()); + EXPECT_EQ(seen->bytes, "v"); + EXPECT_EQ(backend->getTotal(), 2u); + EXPECT_EQ(clock.sleeps.size(), 1u); + } + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + /// Not a `Poco::Exception`, so it did not come from the transport: reissuing it would spend the + /// whole deadline replaying a local bug. + backend->failNextReadWith("k", std::make_exception_ptr(std::logic_error("a local bug"))); + EXPECT_THROW((void)op.read("k", Retry::standard()), std::logic_error); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); + } +} + +#if USE_AWS_S3 + +namespace +{ + +/// An `S3Exception` carrying a canonical `` name. The name is how the request contract tells +/// one store answer from another: the SDK reports every error it does not model as `UNKNOWN`, so the +/// code alone can never stand for a particular failure. +std::exception_ptr s3Error(Aws::S3::S3Errors code, const String & name) +{ + return std::make_exception_ptr(DB::S3Exception("the store answered " + name, code, name)); +} + +/// Answers EVERY read with the same store error. A classification that terminates on an error is then +/// visible as a single attempt, and one that keeps the error ambiguous as a policy spent to its +/// deadline -- which a one-shot arming could never tell apart. +class AlwaysFailingReadBackend final : public CountingBackend +{ +public: + explicit AlwaysFailingReadBackend(std::exception_ptr error_) : error(std::move(error_)) {} + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + (void)CountingBackend::read(key, access); + std::rethrow_exception(error); + } + +private: + std::exception_ptr error; +}; + +} + +TEST(CASRequests, DeadlineIsTheOnlyBoundUnderZeroLatencyThrottling) +{ + FakeClock clock; + auto throttled = std::make_shared( + std::make_shared(), ThrottlingBackend::Mode::EveryNth, 1, 429); + auto requests = makeRequests(throttled, clock); + auto op = requests.admit(); + + const uint64_t start = clock.now; + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Policy); + EXPECT_TRUE(gave_up->sent_any); + /// What ends the call is the policy's own deadline, not a count of attempts: it kept issuing to + /// within one backoff of that deadline, and paused many more times than a small fixed budget allows. + EXPECT_GE(clock.now - start, 85'000u); + EXPECT_GT(clock.sleeps.size(), 16u); +} + +TEST(CASRequests, LeaseBoundPolicyIssuesNothingPastTheBoundary) +{ + FakeClock clock; + auto throttled = std::make_shared( + std::make_shared(), ThrottlingBackend::Mode::EveryNth, 1, 429); + auto requests = makeRequests(throttled, clock); + requests.setAttemptReservationForTest(1'000); + + const uint64_t lease_deadline = clock.now + 10'000; + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::untilLeaseSafe(lease_deadline, 2'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + /// The lease was the smaller of the two bounds, and the give-up names it rather than the policy. + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Lease); + /// Nothing is STARTED that could not finish inside the bound: the last request began at least one + /// attempt reservation before lease minus margin. + EXPECT_LE(clock.now, lease_deadline - 2'000 - 1'000); +} + +TEST(CASRequests, AMalformedRequestIsRefusedWithoutAReissue) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::UNKNOWN, "MalformedXML")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * refused = std::get_if(&result); + ASSERT_NE(refused, nullptr); + EXPECT_EQ(refused->store_error, DB::ErrorCodes::S3_ERROR); + /// The store's own answer proves the request never applied: nothing to resolve, nothing to reissue, + /// and no credential the refusal could be about. + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, AnAccessDenialNoRefreshCanFixIsRefusedOnTheFirstAttempt) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(false); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); + /// A refresh is asked for once and installs nothing, and THAT is what makes the denial terminal. + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ASecondCredentialAnswerAfterTheOneRefreshIsRefused) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + /// The store answers a denial BEFORE it applies anything, so neither attempt landed and no read + /// has anything to settle. A call gets one refresh, so the denial that survives it is the answer. + const auto * refused = std::get_if(&result); + ASSERT_NE(refused, nullptr); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_EQ(backend->writeTotal(), 2u); + EXPECT_EQ(backend->getTotal(), 0u); + /// BOTH attempts are counted, not just the one that produced the answer -- which is why this is + /// asserted on the refusal that took two rather than on one of the single-attempt refusals. + EXPECT_EQ(refused->attempts_sent, backend->writeTotal()); + EXPECT_EQ(clock.sleeps.size(), 1u); /// the one paced re-send under the credentials it installed +} + +TEST(CASRequests, UnderOnceACredentialAnswerIsRefusedWithoutARefresh) +{ + FakeClock clock; + auto backend = std::make_shared(); + /// A refresh that WOULD have installed credentials, so the zero below is the gate and not the + /// storage refusing to hand any back. + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + /// Fresh credentials only help a reissue, and `once` has none to sign, so none are asked for -- + /// which is what keeps `Refused` meaning "no refresh installed credentials and no earlier + /// ambiguity" rather than "a refresh helped and the answer stood anyway". + WriteResult result = op.create("k", "v", Retry::once()); + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->refreshCredentialsCalls(), 0u); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ACredentialAnswerAfterAnAmbiguousAttemptStillOwesTheResolveRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + /// The first attempt's fate is unknown and it may yet land; the second is a proven non-application. + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 3u); + /// The refresh does not license a direct re-send here: the OTHER attempt is still unresolved, so + /// the read that settles it is still owed. + EXPECT_EQ(backend->getTotal(), 2u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); +} + +TEST(CASRequests, AnExpiredTokenARefreshFixesIsResentWithoutAResolveRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID, "ExpiredToken")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_FALSE(committed->resolved_by_read); + /// The credential answer proves its OWN attempt never applied, and no earlier attempt of this call + /// is unresolved, so the re-send under the fresh credentials owes no read. + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_EQ(clock.sleeps.size(), 1u); +} + +TEST(CASRequests, AnExpiredTokenNoRefreshCanFixIsRefusedRatherThanRiddenToTheDeadline) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID, "ExpiredToken")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + WriteResult result = op.create("k", "v", Retry::standard()); + /// The refusal class CONTAINS the refresh class: an expired credential that no refresh installed + /// would otherwise spend the whole deadline being reissued. + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ANameOnlyAccessDenialOnAReadPropagatesWhenNoRefreshIsAvailable) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(false); + /// Matched by NAME alone: the SDK reports this store's denial under its catch-all code, so the + /// name is the only thing that says a credential could explain it. + backend->failNextReadWith("k", s3Error(Aws::S3::S3Errors::UNKNOWN, "AccessDenied")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + /// One refresh is asked for and installs nothing, so nothing would sign differently: the read + /// propagates instead of spending its policy on a request that cannot start succeeding. + expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, ReadModifyWriteWhoseResolveAndFreshObservationBothFailGivesUpUnresolved) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v0", Retry::standard()), "create"); + backend->resetCounts(); + + /// The store refuses the precondition, and both reads that would settle what happened answer with + /// a store refusal the read loop surfaces at once rather than reissuing. They are armed from inside + /// the write so the loop's OWN first read still succeeds and `decide` sees the object. + backend->refuseNextWrite("k"); + bool armed = false; + backend->onBeforeWrite("k", [&] + { + if (armed) + return; + armed = true; + backend->failNextReadWith("k", s3Error(Aws::S3::S3Errors::UNKNOWN, "MalformedXML")); + backend->failNextReadWith("k", s3Error(Aws::S3::S3Errors::UNKNOWN, "MalformedXML")); + }); + + WriteResult result = op.readModifyWrite("k", + [](const std::optional &) -> std::optional { return String("v1"); }, Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + /// No BOUND refused either read -- the reads themselves failed -- so naming a deadline the clock + /// never reached would send its reader to widen the wrong thing, and nothing was observed to + /// report as a conflict. + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + /// The write count is unchanged after the first attempt: nothing ever said another one was safe. + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 3u); + EXPECT_EQ(clock.sleeps.size(), 1u); +} + +/// A missing bucket is an ANSWER the store gave, but not an answer about the object: an S3-compatible +/// store that transiently misroutes a bucket says exactly this, and a read that ended on it would turn +/// an availability blip into a hard failure. It stays in the ambiguous class -- reissued until the +/// policy's deadline -- like a throttle or a 5xx. +TEST(CASRequests, AMissingBucketOnAReadIsReissuedToTheDeadline) +{ + FakeClock clock; + auto backend = std::make_shared( + s3Error(Aws::S3::S3Errors::NO_SUCH_BUCKET, "NoSuchBucket")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const uint64_t start = clock.now; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_GT(backend->getTotal(), 1u) << "the read ended on its first attempt instead of reissuing"; + EXPECT_GE(clock.now - start, 85'000u) << "the policy's own deadline is what must end this read"; +} + +/// The kept half of the same classification: a key miss IS an answer about the object, so reissuing it +/// only replays the same authoritative absence until the deadline. One attempt, no pause. +TEST(CASRequests, AnAuthoritativeKeyMissOnAReadEndsTheCallAtOnce) +{ + FakeClock clock; + auto backend = std::make_shared( + s3Error(Aws::S3::S3Errors::NO_SUCH_KEY, "NoSuchKey")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { (void)op.read("k", Retry::standard()); }); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); +} + +TEST(CASRequests, AnUnmodeledStoreErrorOnAReadIsReissuedNotSurfaced) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "create"); + backend->resetCounts(); + + /// An S3-compatible store's own vendor code. The SDK models it as `UNKNOWN`, which is its code for + /// EVERY error it does not know, so it can never stand for "this will not start succeeding". + backend->failNextReadWith("k", s3Error(Aws::S3::S3Errors::UNKNOWN, "SomeVendorCode")); + const auto seen = op.read("k", Retry::standard()); + ASSERT_TRUE(seen.has_value()); + EXPECT_EQ(seen->bytes, "v"); + EXPECT_EQ(backend->getTotal(), 2u); + EXPECT_EQ(clock.sleeps.size(), 1u); +} + +#endif + +/// A write reserves TWO request envelopes, not one: the attempt, and the read that settles it if the +/// attempt comes back ambiguous. At exactly one reservation of surplus before lease minus margin there +/// is room for the attempt alone, and an engine that reserved only the attempt would start one it +/// could not settle inside the bound. Nothing may be sent. +TEST(CASRequests, AWriteReservesTwoEnvelopesSoOneOfSurplusStartsNothing) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + requests.setAttemptReservationForTest(1'000); + + const uint64_t lease_deadline = clock.now + 2'000 + 1'000; + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::untilLeaseSafe(lease_deadline, 2'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Lease); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_TRUE(clock.sleeps.empty()); +} + + +/// The body of a streamed object is read at the consumer's pace, long after the opening attempt +/// returned; the wrapper re-admits it at every refill. The window the open already loaded is served +/// first -- the SDK buffer arrives with pending data -- and the check first fires on advancing past it. +TEST(CASRequests, StreamBodyKeepsThePreloadedWindowAndRefusesOnTheNextRefill) +{ + FakeClock clock; + auto backend = std::make_shared(); + std::atomic torn_down{false}; + Fence fence{[] { return uint64_t{0}; }, + [&](uint64_t, uint64_t) { return torn_down.load() ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + auto op = requests.admit(); + orThrow(op.create("k", "0123456789", Retry::once()), "create"); + backend->setStreamChunkForTest(4); /// the body arrives as "0123", "4567", "89" + + auto body = op.stream("k", Retry::once()); + ASSERT_TRUE(body); + String first(4, '\0'); + body->readStrict(first.data(), 4); + EXPECT_EQ(first, "0123") << "the window the open already loaded is served, not skipped"; + + torn_down.store(true); + char c; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { body->readStrict(&c, 1); }); + EXPECT_TRUE(body->isCanceled()) << "a refused refill leaves the buffer the consumer holds unusable"; +} + +TEST(CASRequests, StreamBodyServesEveryWindowThenEof) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "0123456789", Retry::once()), "create"); + backend->setStreamChunkForTest(4); + + auto body = op.stream("k", Retry::once()); + ASSERT_TRUE(body); + String all; + DB::readStringUntilEOF(all, *body); + EXPECT_EQ(all, "0123456789"); + EXPECT_TRUE(body->eof()); + EXPECT_FALSE(op.stream("absent", Retry::once())) << "an absent object is still the open's answer"; +} + +/// The mount plane's fence can answer `NoBudget`; a body refused for that reason must read like a +/// refused open on the same plane -- the retry-later class, not a tripped fence. +TEST(CASRequests, StreamBodyRefusalKeepsTheNoBudgetMapping) +{ + FakeClock clock; + auto backend = std::make_shared(); + std::atomic out_of_budget{false}; + Fence fence{[] { return uint64_t{0}; }, + [&](uint64_t, uint64_t) { return out_of_budget.load() ? Fence::Admit::NoBudget : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + auto op = requests.admit(); + orThrow(op.create("k", "0123456789", Retry::once()), "create"); + backend->setStreamChunkForTest(4); + + auto body = op.stream("k", Retry::once()); + ASSERT_TRUE(body); + String first(4, '\0'); + body->readStrict(first.data(), 4); + out_of_budget.store(true); + char c; + try + { + body->readStrict(&c, 1); + FAIL() << "the refill must be refused"; + } + catch (const DB::Exception & e) + { + EXPECT_NE(e.message().find("no lease budget"), String::npos) << e.message(); + } +} + +/// The caller's liveness is the second half of admission for the body too, in the gate's order. +TEST(CASRequests, StreamBodyHonoursTheCallersLiveness) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + std::atomic alive{true}; + auto op = requests.admit([&] { return alive.load(); }); + orThrow(op.create("k", "0123456789", Retry::once()), "create"); + backend->setStreamChunkForTest(4); + + auto body = op.stream("k", Retry::once()); + ASSERT_TRUE(body); + String first(4, '\0'); + body->readStrict(first.data(), 4); + alive.store(false); + char c; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { body->readStrict(&c, 1); }); +} + +/// The window the open already loaded is served WITHOUT a further admission: `Backend::stream` forces +/// that first GET and accounts it to the open's own attempt, so re-checking it here would refuse +/// bytes the caller has already paid for. The refusal belongs to the SECOND window, the first one the +/// body actually asks the store for. Armed before the first read, so a wrapper that discarded the +/// adopted window would refuse immediately instead of serving it. +TEST(CASRequests, StreamBodyServesTheAdoptedWindowEvenWhenAdmissionIsAlreadyRefused) +{ + FakeClock clock; + auto backend = std::make_shared(); + std::atomic torn_down{false}; + Fence fence{[] { return uint64_t{0}; }, + [&](uint64_t, uint64_t) { return torn_down.load() ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + auto op = requests.admit(); + orThrow(op.create("k", "0123456789", Retry::once()), "create"); + backend->setStreamChunkForTest(4); + + auto body = op.stream("k", Retry::once()); + ASSERT_TRUE(body); + torn_down.store(true); /// refused BEFORE the consumer touches the body + + String first(4, '\0'); + body->readStrict(first.data(), 4); + EXPECT_EQ(first, "0123") << "the window the open already paid for must be served, not re-admitted"; + EXPECT_EQ(body->count(), 4u) << "the adopted window is counted once, by the wrapper"; + + char c; + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { body->readStrict(&c, 1); }); +} + +/// The hint is a text match on this repository's Poco. These pins fail the build's own tests the day +/// `SocketImpl::error` changes a word, which is the only way a text match stays honest. +TEST(CASRequestsConnectHint, PocoTextsArePinned) +{ + const auto text_of = [](int err) + { + try + { + Poco::Net::SocketImpl::error(err); + } + catch (const Poco::Exception & e) + { + return e.displayText(); + } + return std::string("did not throw"); + }; + EXPECT_THAT(text_of(EADDRNOTAVAIL), testing::HasSubstr("Cannot assign requested address")); + EXPECT_THAT(text_of(ECONNREFUSED), testing::HasSubstr("Connection refused")); + EXPECT_THAT(text_of(EHOSTUNREACH), testing::HasSubstr("No route to host")); + EXPECT_THAT(text_of(ENETUNREACH), testing::HasSubstr("Network is unreachable")); + /// The fifth text is the connect poll's own: `SocketImpl::connect` throws + /// `Poco::TimeoutException("connect timed out", ...)` (SocketImpl.cpp ~138). + EXPECT_THAT(Poco::TimeoutException("connect timed out", "10.255.255.1:9").displayText(), + testing::HasSubstr("connect timed out")); +} + +#if USE_AWS_S3 +TEST(CASRequestsConnectHint, ClassifierGuards) +{ + using Aws::S3::S3Errors; + for (const char * text : {"Cannot assign requested address", "Connection refused", "No route to host", + "Network is unreachable", "connect timed out"}) + { + const DB::S3Exception hinted(fmt::format("Poco::Exception. Code: 1000, e.code() = 99, {}: 10.0.0.1:9000", text), + S3Errors::NETWORK_CONNECTION); + EXPECT_TRUE(isConnectFailureHint(hinted)) << text; + /// The same text under another S3 error is not a transport verdict. + const DB::S3Exception other(String(text), S3Errors::INTERNAL_FAILURE); + EXPECT_FALSE(isConnectFailureHint(other)) << text; + } + EXPECT_FALSE(isConnectFailureHint(DB::S3Exception("Timeout", S3Errors::NETWORK_CONNECTION))); + EXPECT_FALSE(isConnectFailureHint(DB::S3Exception("Connection reset by peer", S3Errors::NETWORK_CONNECTION))); + EXPECT_FALSE(isConnectFailureHint(Poco::TimeoutException("connect timed out"))); + EXPECT_FALSE(isConnectFailureHint(std::runtime_error("Connection refused"))); +} + +namespace +{ + +DB::S3::PocoHTTPClientConfiguration networkFailureClientConfiguration() +{ + DB::RemoteHostFilter remote_host_filter; + return DB::S3::ClientFactory::instance().createClientConfiguration( + "some-region", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ true, + /* s3_slow_all_threads_after_retryable_error = */ true, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ false, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); +} + +/// A client whose `PutObject` always fails with a `NETWORK_CONNECTION` `AWSError` carrying `text` +/// verbatim -- shaped exactly as `PocoHTTPClient` shapes a real connection failure (empty exception +/// name, the Poco text as the message) -- so a test built on it proves `WriteBufferFromS3`'s rethrow, +/// not a hand-built exception, is what `isConnectFailureHint` above actually has to classify. +struct NetworkFailurePutClient : DB::S3::Client +{ + explicit NetworkFailurePutClient(std::string text_) + : DB::S3::Client( + /*max_retries=*/100, + DB::S3::ServerSideEncryptionKMSConfig(), + std::make_shared("", ""), + networkFailureClientConfiguration(), + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + DB::S3::ClientSettings{ + .use_virtual_addressing = true, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }) + , text(std::move(text_)) + { + } + + Aws::S3::Model::PutObjectOutcome PutObject(const Aws::S3::Model::PutObjectRequest &) const override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::NETWORK_CONNECTION, "", text, /*retryable=*/false); + } + + std::string text; +}; + +} + +/// The classifier above reads the Poco text a connection failure carries off an `S3Exception`; this +/// pins that the REAL `WriteBufferFromS3` rethrow every CAS conditional write goes through -- not a +/// hand-built exception -- hands the caller that text unchanged, under `NETWORK_CONNECTION`. +TEST(CASRequestsConnectHint, WriteBufferFromS3SurfacesTheConnectFailureTextUnchanged) +{ + for (const char * text : {"Cannot assign requested address", "Connection refused", "No route to host", + "Network is unreachable", "connect timed out"}) + { + auto client = std::make_shared(text); + DB::WriteSettings write_settings; + write_settings.object_storage_retry_profile = DB::ObjectStorageRetryProfile::SingleAttempt; + DB::S3::S3RequestSettings request_settings; + DB::WriteBufferFromS3 buffer( + client, "bucket", "network_text", DB::DBMS_DEFAULT_BUFFER_SIZE, request_settings, + /*blob_log_=*/nullptr, /*object_metadata_=*/std::nullopt, /*schedule_=*/{}, write_settings); + buffer.write('A'); + try + { + buffer.finalize(); + FAIL() << "the injected failure must surface"; + } + catch (const DB::S3Exception & e) + { + EXPECT_EQ(e.getS3ErrorCode(), Aws::S3::S3Errors::NETWORK_CONNECTION) << text; + EXPECT_THAT(e.message(), testing::HasSubstr(text)); + } + } +} + +namespace +{ +std::exception_ptr connectHint() +{ + return std::make_exception_ptr(DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION)); +} +} + +TEST(CASRequestsConnectHint, HintedFailuresReissueWithoutARead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", connectHint()); + backend->failNextWriteWith("k", connectHint()); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 3u); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(backend->writeTotal(), 3u); + EXPECT_EQ(backend->getTotal(), 0u); /// no settle read before the commit + ASSERT_EQ(clock.sleeps.size(), 2u); + EXPECT_EQ(clock.sleeps[0], 50u); /// the flat pause, twice + EXPECT_EQ(clock.sleeps[1], 50u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 2u); +} + +TEST(CASRequestsConnectHint, ReissueMeetsPreconditionAndAdoptsOwnBytes) +{ + /// The hint was false: the write landed, its response was lost as a connect-failure text. The + /// reissue meets 412 (the store now holds our OWN new incarnation, minted by the attempt whose + /// response never arrived), one read follows and proves the bytes are ours. + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + backend->resetCounts(); + bool thrown = false; + /// Runs after the write lands and before the caller ever sees a value, with no lock held -- + /// `InMemoryBackend::applyWrite` (CasInMemoryBackend.cpp) calls it right there. + backend->onWriteCommitted("k", [&] + { + if (!thrown) + { + thrown = true; + throw DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION); + } + }); + WriteResult result = op.replace("k", "v2", seen, Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_TRUE(committed->resolved_by_read); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_EQ(backend->writeTotal(), 2u); + EXPECT_EQ(backend->getTotal(), 1u); + ASSERT_EQ(clock.sleeps.size(), 1u); + EXPECT_EQ(clock.sleeps[0], 50u); + } + /// Different ETag, other bytes: a conflict, as today. + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + backend->failNextWriteWith("k", connectHint()); + /// A competitor lands during the flat pause: the engine's own sleep is the seam. + bool competitor_landed = false; + requests.setSleepFnForTest([&](uint64_t ms) + { + clock.sleepFn()(ms); + if (!competitor_landed) + { + competitor_landed = true; + auto other = requests.admit(); + orThrow(other.replace("k", "theirs", seen, Retry::standard()), "competitor"); + } + }); + WriteResult result = op.replace("k", "v2", seen, Retry::standard()); + EXPECT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->getTotal(), 1u); + } + /// The ORIGINAL ETag is still current after the hinted attempt: the reissue meets a CLEAN 412 (the + /// store untouched, unlike sub-block 1's landed write), the settle read observes the original bytes + /// still there under the original ETag, and a further reissue is what actually commits. + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = *orThrow(op.create("k", "v1", Retry::standard()), "create"); + backend->resetCounts(); + backend->failNextWriteWith("k", connectHint()); /// attempt 1: hint, flat pause, no read + backend->refuseNextWrite("k"); /// attempt 2: clean 412, store unchanged + WriteResult result = op.replace("k", "v2", seen, Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(committed->attempts_sent, 3u); + EXPECT_EQ(backend->writeTotal(), 3u); + /// The read after attempt 2's 412 saw the precondition still satisfiable (the original ETag, + /// untouched), so the loop reissued instead of adopting -- exactly one read, not zero. + EXPECT_EQ(backend->getTotal(), 1u); + ASSERT_EQ(clock.sleeps.size(), 2u); + EXPECT_EQ(clock.sleeps[0], 50u); /// the flat pause after attempt 1's hint + EXPECT_LE(clock.sleeps[1], 200u); /// the backoff after attempt 2's settle read + } +} + +TEST(CASRequestsConnectHint, OnceKeepsOneWriteAndOneRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", connectHint()); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + WriteResult result = op.create("k", "v", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_EQ(backend->writeTotal(), 1u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); + /// The counter is "hint seen", recorded at classification: `Retry::once` never acts on it, but the + /// attempt's transport error still named a failed connection. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 1u); +} + +TEST(CASRequestsConnectHint, EarlierAmbiguityStillSettlesByRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); /// attempt 1: ordinary ambiguity -> read, backoff + /// Attempt 2's hint has to come from the hook, not a second `failNextWriteWith`: the armed-failure + /// queue is checked BEFORE the ambiguous-key injection on every call, so a queued failure would win + /// attempt 1 regardless of install order. `writeTotal()` ticks before the request is served, so it + /// reads 2 while attempt 2 is in flight. + bool hint_fired_on_second_attempt = false; + backend->onBeforeWrite("k", [&] + { + if (backend->writeTotal() == 2) + { + /// The ordering claim in full: attempt 1's ambiguity must already have been settled by its + /// read before attempt 2 -- the one place `sleeps[1] == 50u` alone could be fooled by a + /// same-range jittered draw (`backoff(1)` is `uniform(0, 200)`, so a reversed order would + /// false-green about once in 200 runs). + EXPECT_EQ(backend->getTotal(), 1u) << "attempt 1's ambiguity read must already have run"; + hint_fired_on_second_attempt = true; + throw DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION); + } + }); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 3u); + EXPECT_EQ(backend->getTotal(), 1u); + ASSERT_EQ(clock.sleeps.size(), 2u); + EXPECT_LE(clock.sleeps[0], 200u); /// the backoff after attempt 1's ambiguity read + EXPECT_EQ(clock.sleeps[1], 50u); /// the flat pause after attempt 2's hint + EXPECT_TRUE(hint_fired_on_second_attempt); +} + +/// A single exception can be BOTH refusal-class (`isDefinitelyRefusedWrite` matches on the exception +/// NAME, independent of the S3 error code) and hint-text (`isConnectFailureHint` matches on the code +/// and the message): the classifier order, not the exception's shape, must decide which wins. An +/// earlier ambiguity of this inner write keeps the refusal from ending the call, but that must never +/// let the hint skip the read the earlier attempt still needs. +TEST(CASRequestsConnectHint, RefusalAfterAnEarlierAmbiguitySettlesByRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); /// attempt 1: ordinary ambiguity -> read, backoff + /// Attempt 2's refusal-and-hint exception has to come from the hook, not a second + /// `failNextWriteWith`: the armed-failure queue is checked BEFORE the ambiguous-key injection on + /// every call, so a queued failure would win attempt 1 regardless of install order (see the sibling + /// `EarlierAmbiguityStillSettlesByRead` above). `writeTotal()` ticks before the request is served, + /// so it reads 2 while attempt 2 is in flight. + bool refusal_fired_on_second_attempt = false; + backend->onBeforeWrite("k", [&] + { + if (backend->writeTotal() == 2) + { + EXPECT_EQ(backend->getTotal(), 1u) << "attempt 1's ambiguity read must already have run"; + refusal_fired_on_second_attempt = true; + throw DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION, "MalformedXML"); /// refusal AND hint text + } + }); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 3u); + /// Two reads, one per settled attempt: a hint reissue for attempt 2 would have skipped its own + /// read and left this at 1. + EXPECT_EQ(backend->getTotal(), 2u); + ASSERT_EQ(clock.sleeps.size(), 2u); + EXPECT_LE(clock.sleeps[0], 200u); /// backoff(1) after attempt 1's read + EXPECT_LE(clock.sleeps[1], 400u); /// backoff(2) after attempt 2's read -- today's verdict, + /// never the flat 50 ms hint pause + EXPECT_TRUE(refusal_fired_on_second_attempt); + /// The refusal classification wins outright: a definite refusal is never a hint, so the counter + /// must not move even though the exception's code and text also match `isConnectFailureHint`. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 0u); +} + +/// A single exception can ALSO be both refreshable-credential-class (`isRefreshableCredentialError` +/// matches on the exception NAME, independent of the S3 error code) and hint-text +/// (`isConnectFailureHint` matches on the code and the message). The credential refresh drives the +/// reissue here, not the hint, so the hint counter must stay put. +TEST(CASRequestsConnectHint, RefreshedCredentialTextDoesNotDoubleCountTheHint) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", std::make_exception_ptr(DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Connection refused: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + /// The refresh -- not the hint's flat pause -- drove the reissue, so the hint counter must not move + /// even though the exception's code and text also match `isConnectFailureHint`. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 0u); +} + +/// The counter's ambiguity-precedence twin: the credential-owned reissue above requires +/// `!state.any_ambiguous`, so an earlier ambiguity of this inner write keeps it from applying even +/// though attempt 2's exception matches the refreshable-credential class. Attempt 2 is then reissued +/// by the ordinary hint mechanism instead -- flat-paused, and after the resolve read attempt 1 still +/// owes -- so the hint counter must count it. +TEST(CASRequestsConnectHint, CredentialRefreshAfterAnEarlierAmbiguityStillCountsTheHint) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->injectAmbiguousWrite("k"); /// attempt 1: ordinary ambiguity -> read, backoff + bool hint_fired_on_second_attempt = false; + backend->onBeforeWrite("k", [&] + { + if (backend->writeTotal() == 2) + { + EXPECT_EQ(backend->getTotal(), 1u) << "attempt 1's ambiguity read must already have run"; + hint_fired_on_second_attempt = true; + throw DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 99, Cannot assign requested address: 10.0.0.1:9000", + Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"); /// hint AND credential-refreshable + } + }); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 3u); + /// One read (attempt 1's) settles the earlier ambiguity; attempt 2's hint reissue skips its own + /// read, exactly as `EarlierAmbiguityStillSettlesByRead` pins for a non-credential hint. + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + ASSERT_EQ(clock.sleeps.size(), 2u); + EXPECT_LE(clock.sleeps[0], 200u); /// the backoff after attempt 1's ambiguity read + EXPECT_EQ(clock.sleeps[1], 50u); /// the flat pause after attempt 2's hint, not a credential backoff + EXPECT_TRUE(hint_fired_on_second_attempt); + /// The hint mechanism, not a credential-owned reissue, actually resent this attempt, so the counter + /// counts it even though the exception's name also matches the refreshable-credential class. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 1u); +} + +TEST(CASRequestsConnectHint, GatesRefuseTheReissue) +{ + /// Deadline: hints until the window closes. + { + FakeClock clock; + auto backend = std::make_shared(); + for (int i = 0; i < 100; ++i) + backend->failNextWriteWith("k", connectHint()); + auto requests = makeRequests(backend, clock); + requests.setAttemptReservationForTest(1'000); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::within(3'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->getTotal(), 0u); + } + /// Fence: the fence trips during the pause. + { + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", connectHint()); + bool lost = false; + Fence fence{ + [] { return uint64_t{1}; }, + [&](uint64_t, uint64_t) { return lost ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto requests = makeRequests(backend, clock, fence); + requests.setSleepFnForTest([&](uint64_t ms) { clock.sleepFn()(ms); lost = true; }); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + } + /// Two envelopes exactly for the attempt; today's ambiguous path would still have its read + /// envelope (2000 >= 1000), the hint path gives up instead -- the documented deadline-edge + /// difference. + { + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", connectHint()); + auto requests = makeRequests(backend, clock); + requests.setAttemptReservationForTest(1'000); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::within(2'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->getTotal(), 0u); + } +} + +TEST(CASRequestsConnectHint, AmbiguityAfterHintsStartsAtFirstBackoff) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", connectHint()); + backend->failNextWriteWith("k", connectHint()); + backend->failNextWriteWith("k", std::make_exception_ptr(Poco::TimeoutException("the write timed out"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); + ASSERT_EQ(clock.sleeps.size(), 3u); + EXPECT_EQ(clock.sleeps[0], 50u); + EXPECT_EQ(clock.sleeps[1], 50u); + /// `backoff(1)` is full jitter over [0, 200] ms (`CasRetry.h`): the hints did not advance the index. + EXPECT_LE(clock.sleeps[2], 200u); +} + +TEST(CASRequestsFuse, MatcherPrecedence) +{ + using Aws::S3::S3Errors; + /// The generic transport-timeout text is Poco's exception name, pinned here. + EXPECT_THAT(Poco::TimeoutException("the socket").displayText(), testing::StartsWith("Timeout")); + const DB::S3Exception fuse("Poco::Exception. Code: 1000, e.code() = 0, Timeout: the socket", S3Errors::NETWORK_CONNECTION); + EXPECT_TRUE(isFirstAttemptFuseTimeout(fuse, 1)); + EXPECT_FALSE(isFirstAttemptFuseTimeout(fuse, 2)); + const DB::S3Exception hint("Poco::Exception. Code: 1000, e.code() = 0, Timeout: connect timed out: 10.0.0.1:9", S3Errors::NETWORK_CONNECTION); + EXPECT_FALSE(isFirstAttemptFuseTimeout(hint, 1)); /// the connect-failure hint owns it + EXPECT_TRUE(isConnectFailureHint(hint)); + EXPECT_FALSE(isFirstAttemptFuseTimeout(DB::S3Exception("Connection reset by peer", S3Errors::NETWORK_CONNECTION), 1)); + EXPECT_FALSE(isFirstAttemptFuseTimeout(DB::S3Exception("Timeout", S3Errors::INTERNAL_FAILURE), 1)); +} + +namespace +{ +std::exception_ptr fuseTimeout() +{ + return std::make_exception_ptr(DB::S3Exception("Poco::Exception. Code: 1000, e.code() = 0, Timeout: the socket", + Aws::S3::S3Errors::NETWORK_CONNECTION)); +} +} + +TEST(CASRequestsFuse, FirstAttemptTimeoutReissuesWithoutSleep) +{ + /// Write: the settle read still runs (the request may have been sent), then a no-sleep reissue. + { + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", fuseTimeout()); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); + } + /// Read: no settle read, no sleep. + { + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "seed"); + backend->resetCounts(); + backend->failNextReadWith("k", fuseTimeout()); + EXPECT_TRUE(op.read("k", Retry::standard()).has_value()); + EXPECT_EQ(backend->getTotal(), 2u); + EXPECT_TRUE(clock.sleeps.empty()); + } + /// LIST: no sleep either. LIST has no `failNextWith`-style armed queue (only write/read/head do), + /// so a small backend that throws the fuse on its first LIST and records the physical attempt + /// number stands in. + { + struct ListFuseOnceBackend : CountingBackend + { + bool armed = true; + std::vector list_attempts; + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override + { + list_attempts.push_back(access.attemptNo()); + if (armed) + { + armed = false; + std::rethrow_exception(fuseTimeout()); + } + return CountingBackend::list(prefix, cursor, limit, access); + } + }; + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)op.list("p/", "", 10, Retry::standard()); + EXPECT_EQ(backend->list_attempts, (std::vector{1, 2})); + EXPECT_TRUE(clock.sleeps.empty()); + } + /// Attempts 1 and 2 failing: attempt 2 is not a first attempt, so exactly one sleep, after it. + { + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", fuseTimeout()); + backend->failNextWriteWith("k", fuseTimeout()); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::standard()); + ASSERT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(std::get(result).attempts_sent, 3u); + EXPECT_EQ(clock.sleeps.size(), 1u); + } +} + +TEST(CASRequestsFuse, GatesRefuseTheZeroPauseReissue) +{ + /// `setAttemptReservationForTest(1'000)`: the write's own admission reserves two envelopes + /// (`reservedFor(0, 2) == 2000`), which matches a 2000 ms window exactly -- `fits` is `needed <= + /// remaining`, so the boundary admits. The settle read that follows the fuse reserves only one + /// envelope (`reservedFor(0, 1) == 1000`), which still fits even after the clock below has moved. + /// What must NOT fit is the zero-pause reissue's own `reservedFor(0, 2) == 2000`. `FakeClock` never + /// moves on its own -- only a sleep advances it, and this path sleeps none -- so a naive `now()` + /// would see the SAME instant at every one of the four calls this operation makes (the initial + /// `bind`, the write's own admission, the settle read's admission, the reissue's admission) and + /// wrongly admit the reissue too. A real failing attempt spends wall time even though it never + /// lands, so this fixture's clock counts its own calls and adds 1 ms starting from the THIRD one + /// (the settle read's admission) onward: late enough that the write's own admission still sees the + /// pristine window, early enough that the reissue's admission sees one fewer millisecond than it + /// needs. + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", fuseTimeout()); + int now_calls = 0; + auto requests = makeRequests(backend, clock); + requests.setAttemptReservationForTest(1'000); + requests.setNowFnForTest([&clock, &now_calls]() -> uint64_t + { + ++now_calls; + return clock.now + (now_calls <= 2 ? 0 : 1); + }); + auto op = requests.admit(); + WriteResult result = op.create("k", "v", Retry::within(2'000)); + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_TRUE(clock.sleeps.empty()) << "the zero-pause reissue never sleeps, even when refused"; + /// The fence, not the deadline, refuses the zero-pause reissue: three `Fence::admit` calls happen + /// in this scenario -- the write's own admission, the settle read's admission, and the reissue's + /// admission -- in that order, so tripping the fence on the THIRD call refuses only the reissue, + /// after the write attempt and its settle read both already went through. + { + FakeClock fence_clock; + auto fence_backend = std::make_shared(); + fence_backend->failNextWriteWith("k", fuseTimeout()); + int admit_calls = 0; + Fence fence{ + [] { return uint64_t{1}; }, + [&](uint64_t, uint64_t) { return ++admit_calls >= 3 ? Fence::Admit::LostOrRearmed : Fence::Admit::Ok; }, + [](uint64_t) {}}; + auto fence_requests = makeRequests(fence_backend, fence_clock, fence); + auto fence_op = fence_requests.admit(); + WriteResult fence_result = fence_op.create("k", "v", Retry::standard()); + const auto * fence_gave_up = std::get_if(&fence_result); + ASSERT_NE(fence_gave_up, nullptr); + EXPECT_EQ(fence_gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(fence_gave_up->sent_any); + EXPECT_EQ(fence_backend->writeTotal(), 1u) << "the fence refuses before a second write is ever sent"; + } + /// `Retry::once()` never performs a second attempt. + auto once_backend = std::make_shared(); + once_backend->failNextWriteWith("k", fuseTimeout()); + auto once_requests = makeRequests(once_backend, clock); + auto once_op = once_requests.admit(); + (void)once_op.create("k", "v", Retry::once()); + EXPECT_EQ(once_backend->writeTotal(), 1u); +} + +TEST(CASRequestsFuse, ReadLoopZeroPauseKeepsTheBackoffIndex) +{ + struct ReadAttemptRecordingBackend : CountingBackend + { + std::vector read_attempts; + std::optional read(const String & key, TransportAccess & access) override + { + read_attempts.push_back(access.attemptNo()); + return CountingBackend::read(key, access); + } + }; + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "seed"); + backend->read_attempts.clear(); + backend->failNextReadWith("k", fuseTimeout()); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("attempt 2: an ordinary fault"))); + EXPECT_TRUE(op.read("k", Retry::standard()).has_value()); + ASSERT_EQ(clock.sleeps.size(), 1u); + /// The one sleep is `backoff(1)`: the zero-pause reissue did not advance the index. + EXPECT_LE(clock.sleeps[0], 200u); /// `backoff(1)` is full jitter over [0, 200] ms + /// The transport still sees every physical attempt: the zero-pause reissue (attempt 2) advances + /// `attempt_no` alone, so attempt 3 -- reached only after the one ordinary backoff -- follows it, + /// not a second attempt 1. + EXPECT_EQ(backend->read_attempts, (std::vector{1, 2, 3})); +} + +/// `Retry::once` forbids the REISSUE, not the observation: a fuse a single-attempt read hits still +/// counts (the write path already counts at classification, before its own single-attempt check), it +/// just throws unchanged instead of re-sending. +TEST(CASRequestsFuse, ReadUnderOnceCountsTheFuseWithoutReissuing) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextReadWith("k", fuseTimeout()); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { (void)op.read("k", Retry::once()); }); + EXPECT_EQ(backend->getTotal(), 1u) << "Retry::once performs no second attempt"; + EXPECT_TRUE(clock.sleeps.empty()); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 1u); +} + +/// The fuse counter's credential-refresh twin of `CASRequestsConnectHint.RefreshedCredentialTextDoesNotDoubleCountTheHint`: +/// a first attempt whose exception is both fuse-text and refreshable-credential-name must be counted +/// as the credential reissue it actually is, not also as a fuse. +TEST(CASRequestsFuse, RefreshedCredentialTextDoesNotDoubleCountTheFuse) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", std::make_exception_ptr(DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 0, Timeout: the socket", + Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + + WriteResult result = op.create("k", "v", Retry::standard()); + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_FALSE(committed->resolved_by_read); + EXPECT_EQ(backend->getTotal(), 0u); + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + /// The refresh -- not the fuse's immediate reissue -- drove the resend, so the fuse counter must not + /// move even though the exception's code and text also match `isFirstAttemptFuseTimeout`. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 0u); +} + +/// The read loop's own twin of `RefreshedCredentialTextDoesNotDoubleCountTheFuse`: a first read attempt +/// whose exception is both fuse-text and refreshable-credential-name is a credential reissue, not a +/// fuse, so the counter must not move even though the reissue itself is immediate, exactly like a fuse. +TEST(CASRequestsFuse, ReadRefreshedCredentialTextDoesNotDoubleCountTheFuse) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + orThrow(op.create("k", "v", Retry::standard()), "seed"); + backend->resetCounts(); + backend->setRefreshCredentialsResult(true); + backend->failNextReadWith("k", std::make_exception_ptr(DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 0, Timeout: the socket", + Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + + const auto seen = op.read("k", Retry::standard()); + ASSERT_TRUE(seen.has_value()); + EXPECT_EQ(seen->bytes, "v"); + EXPECT_EQ(backend->getTotal(), 2u) << "the failed attempt and its immediate reissue both reached the store"; + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + EXPECT_TRUE(clock.sleeps.empty()); + /// The refresh -- not the fuse's immediate reissue -- drove the resend, so the fuse counter must not + /// move even though the exception's code and text also match `isFirstAttemptFuseTimeout`. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 0u); +} + +#endif + +TEST(CASRequestBudget, EnvelopeIsValidatedNotTheBareAttempt) +{ + CasRequestBudget budget{.attempt_timeout_ms = 5000, .lease_safety_margin_ms = 2000, .connect_timeout_cap_ms = 1000}; + EXPECT_EQ(budget.attemptEnvelopeMs(), 7000u); + EXPECT_EQ((CasRequestBudget{.attempt_timeout_ms = 5000, .lease_safety_margin_ms = 2000, .connect_timeout_cap_ms = std::nullopt}.attemptEnvelopeMs()), 5000u); + /// Defaults with the default TTL / period are accepted. + EXPECT_NO_THROW(validateCasRequestBudget(budget, 30000, 10000, /*background_renewal=*/true)); + /// A zero attempt timeout would reserve nothing while the request keeps the disk's own timeout. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(CasRequestBudget{.attempt_timeout_ms = 0, .lease_safety_margin_ms = 2000, + .connect_timeout_cap_ms = std::nullopt}, 30000, 10000, true); + }); + /// The old inequality (attempt <= TTL - margin - period: 5000 <= 13000) accepted this; two envelopes + /// of 15 s do not fit a 25 s lease behind a 10 s period and a 2 s margin. + const CasRequestBudget wide{.attempt_timeout_ms = 5000, .lease_safety_margin_ms = 2000, .connect_timeout_cap_ms = 5000}; + try + { + validateCasRequestBudget(wide, 25000, 10000, true); + FAIL() << "must refuse"; + } + catch (const DB::Exception & e) + { + EXPECT_THAT(e.message(), testing::HasSubstr("envelope")); + EXPECT_THAT(e.message(), testing::HasSubstr("15000")); + } + /// Without background renewal only `envelope + margin < TTL` applies (15000 + 2000 < 25000). + EXPECT_NO_THROW(validateCasRequestBudget(wide, 25000, 10000, /*background_renewal=*/false)); + /// Saturation: absurd values fail closed rather than wrap. + expectThrowsCode(DB::ErrorCodes::BAD_ARGUMENTS, [&] + { + validateCasRequestBudget(CasRequestBudget{.attempt_timeout_ms = std::numeric_limits::max(), + .lease_safety_margin_ms = 1, .connect_timeout_cap_ms = 1}, + 30000, 10000, true); + }); +} + +TEST(CASRequests, ReservationIsTheEnvelope) +{ + struct EnvelopeBackend : InMemoryBackend + { + uint64_t attemptTimeoutMs() const override { return 5000; } + uint64_t attemptEnvelopeMs() const override { return 7000; } + }; + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + /// A write reserves two envelopes: 14 s fits a 14 s window, 13.999 s does not. + EXPECT_TRUE(std::holds_alternative(op.create("k", "v", Retry::within(14'000)))); + const WriteResult refused = op.create("k2", "v", Retry::within(13'999)); + const auto * gave_up = std::get_if(&refused); + ASSERT_NE(gave_up, nullptr); + EXPECT_FALSE(gave_up->sent_any); +} + +/// Every `Backend` decorator that forwards `attemptTimeoutMs` to an inner backend must forward +/// `attemptEnvelopeMs` too, or the default (`attemptEnvelopeMs() { return attemptTimeoutMs(); }`) +/// silently drops the inner backend's connect contribution -- exactly the gap `Pool::open`'s +/// `InstrumentedBackend` wrapper had. Pin the forwarding through the same engine construction +/// production uses. +TEST(CASRequests, ReservationIsTheEnvelopeThroughInstrumentedBackend) +{ + struct EnvelopeBackend : InMemoryBackend + { + uint64_t attemptTimeoutMs() const override { return 5000; } + uint64_t attemptEnvelopeMs() const override { return 7000; } + }; + FakeClock clock; + auto inner = std::make_shared(); + auto wrapped = std::make_shared(inner); + ASSERT_EQ(wrapped->attemptTimeoutMs(), 5000u); + ASSERT_EQ(wrapped->attemptEnvelopeMs(), 7000u) << "InstrumentedBackend must forward the envelope, not fall back to the bare attempt timeout"; + auto requests = makeRequests(wrapped, clock); + auto op = requests.admit(); + /// Same boundary as ReservationIsTheEnvelope, now through the wrapper `Pool::open` actually uses. + EXPECT_TRUE(std::holds_alternative(op.create("k", "v", Retry::within(14'000)))); + const WriteResult refused = op.create("k2", "v", Retry::within(13'999)); + const auto * gave_up = std::get_if(&refused); + ASSERT_NE(gave_up, nullptr); + EXPECT_FALSE(gave_up->sent_any); +} diff --git a/src/Disks/tests/gtest_cas_retirement_sweep.cpp b/src/Disks/tests/gtest_cas_retirement_sweep.cpp index 7300f02d0c7c..95d85c907740 100644 --- a/src/Disks/tests/gtest_cas_retirement_sweep.cpp +++ b/src/Disks/tests/gtest_cas_retirement_sweep.cpp @@ -12,6 +12,7 @@ #include #include +#include #include #include #include @@ -47,6 +48,7 @@ namespace DB::ErrorCodes using namespace DB::Cas; using DB::Cas::tests::idOf; +using DB::Cas::tests::SharedWaitLog; using DB::Cas::tests::u128Of; namespace @@ -63,6 +65,8 @@ namespace class HoleyListBackend : public InMemoryBackend { public: + /// Unhide the primitive overload that the legacy override below would otherwise hide. + using InMemoryBackend::list; void omitFromNthListCall(const String & key, size_t nth) { std::lock_guard lock(m); @@ -80,14 +84,14 @@ class HoleyListBackend : public InMemoryBackend return served; } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { - ListPage page = InMemoryBackend::list(prefix, cursor, limit); + RawListPage page = InMemoryBackend::list(prefix, cursor, limit, access); std::lock_guard lock(m); if (omitted.empty()) return page; auto it = std::find_if(page.keys.begin(), page.keys.end(), - [&](const ListedKey & k) { return k.key == omitted; }); + [&](const RawListedKey & k) { return k.key == omitted; }); if (it == page.keys.end()) return page; /// not a qualifying call -- do not count it if (seen_calls++ != target_call) @@ -114,56 +118,65 @@ class HoleyListBackend : public InMemoryBackend class RefPrefixListCountingBackend : public InMemoryBackend { public: + /// Unhide the primitive overload that the legacy override below would otherwise hide. + using InMemoryBackend::list; String refs_prefix; String janitor_prefix; std::atomic ref_prefix_lists{0}; std::atomic janitor_prefix_lists{0}; - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (!refs_prefix.empty() && prefix == refs_prefix) ++ref_prefix_lists; if (!janitor_prefix.empty() && prefix == janitor_prefix) ++janitor_prefix_lists; - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } }; -/// Forces the FIRST `putIfAbsent` whose key contains `fault_key_substr` to throw an ambiguous -/// (Unresolved-classified) exception, `fault_count` times -- the minimal fault injection needed to drive -/// a ref-log append into the `Unresolved`/wedge outcome, with `max_attempts = 1` in the budget so the -/// single failed attempt exhausts the retry budget immediately. (Same shape as `gtest_cas_pool.cpp`'s -/// file-local backend of the same name; both are three lines of `throw` over `InMemoryBackend`, and -/// hoisting a shared one would couple two suites' fault models for no gain.) +/// Every write of a matching key is a lost response, for as long as `fault_key_substr` names one. It +/// has to be every one: the request engine settles an ambiguity by an exact read and then reissues, so +/// a counted fault is outlived by the reissues and the write commits -- the difference between the +/// wedge this fixture needs and a clean commit. Clearing `fault_key_substr` disarms it. class UnresolvedPutBackend final : public InMemoryBackend { public: - using Backend::putIfAbsent; - String fault_key_substr; - int fault_count = 0; + /// How many matching writes actually hit the fault, so a caller can prove the engine reissued + /// rather than infer it from a give-up that a non-retrying policy would also reach. + int fault_hits = 0; - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override { - if (fault_count > 0 && !fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) + if (!fault_key_substr.empty() && key.find(fault_key_substr) != String::npos) { - --fault_count; + ++fault_hits; throw Poco::TimeoutException("UnresolvedPutBackend: simulated ambiguous result (response lost)"); } - return InMemoryBackend::putIfAbsent(key, bytes, meta); + return InMemoryBackend::write(key, bytes, expected_value, access); } }; /// GC's fence-out applied directly to the mount lease: preserve the body, set `gc_fenced`, bump `seq` /// (token-guarded). A subsequent `tryRemountOnce` then reclaims a fresh incarnation. +bool headExists(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + void fenceOutMount(Backend & backend, const String & mount_key) { - const auto got = backend.get(mount_key); + DB::Cas::tests::OperationForTest op(backend); + const auto got = (*op).read(mount_key, Retry::standard()); ASSERT_TRUE(got.has_value()); MountLease m = decodeMountLease(got->bytes); m.gc_fenced = true; m.seq += 1; - ASSERT_EQ(backend.putOverwrite(mount_key, encodeMountLease(m), got->token).outcome, PutOutcome::Done); + ASSERT_TRUE(std::holds_alternative((*op).replace(mount_key, encodeMountLease(m), got->etag, Retry::standard()))); } /// Publish one part `ref` with a single content blob whose payload is `payload`. @@ -190,11 +203,12 @@ ManifestId publishOneBlobPart(const PoolPtr & s, const RootNamespace & ns, const /// Every ref-log key of `ns` currently listed, in key order. std::set listRefLogKeys(Backend & b, const Layout & l, const RootNamespace & ns) { + DB::Cas::tests::OperationForTest op(b); std::set out; String cursor; while (true) { - const ListPage page = b.list(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000); + const ListPage page = (*op).list(l.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)), cursor, 1000, Retry::standard()); for (const ListedKey & k : page.keys) if (const auto parsed = l.parseRefObjectKey(k.key); parsed && parsed->kind == RefObjectKind::Log) out.insert(k.key); @@ -256,7 +270,7 @@ TEST(CASRetirementSweep, AHiddenRemovalStillReclaimsItsBlob) store->renewWatermarkOnce(); const String blob_key = layout.blobKey(BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(u128Of(payload))}); - ASSERT_TRUE(backend->head(blob_key).exists); + ASSERT_TRUE(headExists(*backend, blob_key)); const std::set before_drop = listRefLogKeys(*backend, layout, ns); store->dropRef(ns, "part_a"); @@ -278,7 +292,7 @@ TEST(CASRetirementSweep, AHiddenRemovalStillReclaimsItsBlob) } ASSERT_TRUE(backend->holeServed()) << "the sabotage never fired"; - EXPECT_FALSE(backend->head(blob_key).exists) + EXPECT_FALSE(headExists(*backend, blob_key)) << "the removal was hidden from one enumeration and never folded -- the retention half of the " "skipped-transaction class, which arithmetic intake is supposed to close"; } @@ -333,20 +347,35 @@ TEST(CASRetirementSweep, TheRoundEnumeratesTheRefPrefixExactlyOnce) TEST(CASRetirementSweep, AStragglerFromTheDyingEpochLosesItsCreateToTheRecoverySeal) { CasRequestBudget budget; - budget.max_attempts = 1; budget.attempt_timeout_ms = 100; - budget.operation_deadline_ms = 5000; budget.lease_safety_margin_ms = 100; auto backend = std::make_shared(); - uint64_t fake_boot = 1'000'000; - std::vector waits; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(1'000'000); + auto waits = std::make_shared(); + /// The append's own retry clock and its sleep log, installed further down; declared here, before + /// the store, because the store's teardown still calls the now-function they back. + auto fake_retry = std::make_shared>(0); + auto retry_sleeps = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_root_id = "test", .mount_lease_ttl_ms = std::chrono::milliseconds(30000), .cas_request_budget = budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, }); ASSERT_TRUE(store); const Layout & layout = store->layout(); @@ -364,28 +393,44 @@ TEST(CASRetirementSweep, AStragglerFromTheDyingEpochLosesItsCreateToTheRecoveryS publishOneBlobPart(store, ns, "x", "straggler-payload"); ASSERT_EQ(store->liveWriterEpoch(), 1u); - /// Drive the next ref-log append into the Unresolved/wedge outcome: the single attempt the budget - /// allows fails ambiguously, so this process can never learn whether its conditional PUT landed. - /// That undecidability is the whole reason the resolution is a conditional CREATE and not a GET. + /// Drive the next ref-log append into the Unresolved/wedge outcome: every attempt it makes fails + /// ambiguously, so this process can never learn whether its conditional PUT landed. That + /// undecidability is the whole reason the resolution is a conditional CREATE and not a GET. The + /// give-up is the append's own retry window -- paced on ITS OWN virtual clock, separate from + /// `fake_boot` (the mount fence's), so the standard policy's full window is available to reissue + /// against rather than being cut short by the 30s lease `fake_boot` also measures. + store->setCasRequestNowFnForTest([fake_retry] + { + return fake_retry->load(); + }); + store->setCasRetrySleepForTest([fake_retry, retry_sleeps](uint64_t ms) + { + *fake_retry += ms + 1; + retry_sleeps->push(ms); + }); backend->fault_key_substr = layout.namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; - backend->fault_count = 1; DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); + EXPECT_GT(backend->fault_hits, 1) + << "the append must have reissued more than once against the persistent fault before giving up " + "-- a single attempt would not distinguish this from a non-retrying policy"; + EXPECT_GT(retry_sleeps->size(), 1u) << "more than one paced retry must have occurred before the give-up"; + EXPECT_GT(fake_retry->load(), 0u) << "the retry clock must have advanced past the policy's own deadline"; /// The id the straggler would occupy: one past the greatest record that is actually durable in the /// dying epoch. That is also, by construction, where the recovery seal goes. const RefTxnId greatest = greatestLoggedId(*backend, layout, ns); ASSERT_EQ(greatest.writer_epoch, 1u); const RefTxnId straggler_slot{greatest.writer_epoch, greatest.ref_sequence + 1}; - ASSERT_FALSE(backend->head(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot)).exists) + ASSERT_FALSE(headExists(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot))) << "the slot must be empty before recovery -- otherwise this test proves nothing about who won"; /// Fence and remount. No wait: this is the case that used to cost 30 seconds. - fake_boot += 30001; + *fake_boot += 30001; fenceOutMount(*backend, layout.mountKey("test")); ASSERT_TRUE(store->tryRemountOnce()); ASSERT_EQ(store->liveWriterEpoch(), 2u); - EXPECT_TRUE(waits.empty()) + EXPECT_TRUE(waits->empty()) << "the remount blocked on an operator-configured wait; the grace is supposed to be gone"; /// Touch the namespace so it re-recovers under the new epoch: the walk closes epoch 1 in band. The @@ -393,13 +438,15 @@ TEST(CASRetirementSweep, AStragglerFromTheDyingEpochLosesItsCreateToTheRecoveryS /// landed, which is precisely the state that leaves a straggler outstanding. backend->fault_key_substr.clear(); EXPECT_EQ(store->listRefs(ns).size(), 1u); - ASSERT_TRUE(backend->head(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot)).exists) + ASSERT_TRUE(headExists(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot))) << "recovery did not seal the dead epoch at the slot a straggler would take -- without that " "seal there is nothing for the straggler's create to lose to"; /// THE STRAGGLER ARRIVES. Its conditional create is refused, whenever it happens to land. - const PutResult put = backend->putIfAbsent(layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot), "ghost-body"); - EXPECT_EQ(put.outcome, PutOutcome::PreconditionFailed) + DB::Cas::tests::OperationForTest straggler_op(backend); + const WriteResult put = (*straggler_op).create( + layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), straggler_slot), "ghost-body", Retry::once()); + EXPECT_TRUE(std::holds_alternative(put)) << "the dying epoch's straggler overwrote (or joined) a slot the successor had already sealed"; } diff --git a/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp b/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp new file mode 100644 index 000000000000..c0e4ee5f5306 --- /dev/null +++ b/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp @@ -0,0 +1,402 @@ +#include + +#include "config.h" + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ +extern const int NOT_IMPLEMENTED; +} + +/// `S3ObjectStorage::removeObjectsIfExistImpl` (the CAS bulk-delete path, reached through +/// `removeObjectsIfExistUnderProfile`) must honour `S3Capabilities::isBatchDeleteSupported()` the same +/// way the generic `deleteFilesFromS3` does, but WITHOUT looping over the objects itself: once the +/// capability is known false (a configured `false`, or one just learned from a `DeleteObjects` reply in +/// the "batch delete not implemented" error class), it throws `NOT_IMPLEMENTED` without sending anything +/// else, and leaves per-key retry to the caller (the CAS engine admits each such retry as its own +/// request -- see CasGc.cpp's `removeChunkWriteOnceOrOneByOne`). A request failure of any other class +/// must keep today's fail-close behaviour. The one exception to all of this is a batch of exactly one +/// object, which is always a plain `DeleteObject` -- never gated on the capability at all, since a +/// single physical request is never something the capability check exists to rule out. + +namespace +{ + +/// A real local HTTP server standing in for S3. `DeleteObjects` arrives as a POST to the bucket root; +/// a per-key `DeleteObject` arrives as a plain HTTP DELETE to the key's path -- the two are +/// distinguished by HTTP method alone, with no need to parse the request body or query string. +class ScriptedS3Server +{ +public: + using Responder = std::function; + +private: + class Handler : public Poco::Net::HTTPRequestHandler + { + ScriptedS3Server & owner; + + public: + explicit Handler(ScriptedS3Server & owner_) : owner(owner_) { } + + void handleRequest(Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) override + { + { + std::lock_guard lock(owner.mutex); + owner.methods_seen.push_back(request.getMethod()); + } + /// `DeleteObjects` carries a request body (the XML `` payload); leaving it unread on a + /// keep-alive connection makes Poco parse those leftover bytes as the start of the NEXT + /// request once this handler returns, corrupting the very next `DeleteObject` this test expects. + request.stream().ignore(std::numeric_limits::max()); + owner.responder(request, response); + } + }; + + class Factory : public Poco::Net::HTTPRequestHandlerFactory + { + ScriptedS3Server & owner; + + Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override + { + return new Handler(owner); + } + + public: + explicit Factory(ScriptedS3Server & owner_) : owner(owner_) { } + }; + + std::unique_ptr server_socket; + Poco::SharedPtr handler_factory; + Poco::AutoPtr server_params; + std::unique_ptr server; + Responder responder; + mutable std::mutex mutex; + std::vector methods_seen; + +public: + explicit ScriptedS3Server(Responder responder_) + : server_socket(std::make_unique(0)) + , handler_factory(new Factory(*this)) + , server_params(new Poco::Net::HTTPServerParams()) + , server(std::make_unique(handler_factory, *server_socket, server_params)) + , responder(std::move(responder_)) + { + server->start(); + } + + std::string getUrl() const { return "http://" + server_socket->address().toString(); } + + size_t countMethod(const std::string & method) const + { + std::lock_guard lock(mutex); + return static_cast(std::count(methods_seen.begin(), methods_seen.end(), method)); + } +}; + +void sendXml(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & body) +{ + response.setContentType("application/xml"); + response.setContentLength(body.size()); + response.setStatus(status); + auto & out = response.send(); + out << body; + out.flush(); +} + +/// A quiet-mode `DeleteObjects` success (HTTP 200) whose body lists only the failed keys, exactly as a +/// real S3 backend would report a mixed outcome. +void sendBatchSuccessWithErrors(Poco::Net::HTTPServerResponse & response, const std::string & not_found_key, const std::string & denied_key) +{ + const std::string body = + "" + "" + "" + not_found_key + "NoSuchKeyThe specified key does not exist." + "" + denied_key + "AccessDeniedAccess Denied" + ""; + sendXml(response, Poco::Net::HTTPResponse::HTTP_OK, body); +} + +/// A request-level `DeleteObjects` failure in the "batch delete is not implemented" class that +/// `deleteFileFromS3.cpp`'s `deleteFilesFromS3` also treats as "fall back to plain `DeleteObject`". +void sendBatchNotImplemented(Poco::Net::HTTPServerResponse & response) +{ + const std::string body = + "" + "NotImplementedA header you provided implies functionality that is not implemented"; + sendXml(response, Poco::Net::HTTPResponse::HTTP_BAD_REQUEST, body); +} + +/// A request-level `DeleteObjects` failure in an ordinary (not "unsupported") class: this must keep +/// today's fail-close behaviour and never fall back to per-key deletes. +void sendBatchInternalError(Poco::Net::HTTPServerResponse & response) +{ + const std::string body = + "" + "InternalErrorWe encountered an internal error, please try again."; + sendXml(response, Poco::Net::HTTPResponse::HTTP_INTERNAL_SERVER_ERROR, body); +} + +void sendDeleteObjectSuccess(Poco::Net::HTTPServerResponse & response) +{ + response.setContentLength(0); + response.setStatus(Poco::Net::HTTPResponse::HTTP_NO_CONTENT); + response.send(); +} + +/// A single-key `DeleteObject` failure -- used to script the size-one path's own error handling, as +/// distinct from the batch response's per-key `` elements covered by the test above. +void sendSingleDeleteError(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & code, const std::string & message) +{ + const std::string body = + "" + "" + code + "" + message + ""; + sendXml(response, status, body); +} + +std::shared_ptr makeStorageForTest(const std::string & endpoint, const DB::S3Capabilities & capabilities) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ false, + /* s3_slow_all_threads_after_retryable_error = */ false, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ true, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + cfg.endpointOverride = endpoint; + cfg.connectTimeoutMs = 10000; + cfg.requestTimeoutMs = 10000; + cfg.s3_use_adaptive_timeouts = false; + /// Every test here starts its own server on an ephemeral port; with keep-alive on, the process-wide + /// HTTP connection pool can hand a later test a connection to a port whose server is already gone + /// (`Connection reset by peer` under `--gtest_repeat`). One connection per request is what a + /// short-lived test server should get. + cfg.http_keep_alive_timeout = 0; + auto client = DB::S3::ClientFactory::instance().create( + cfg, + DB::S3::ClientSettings{ + .use_virtual_addressing = false, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }, + "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); + return std::make_shared( + std::move(client), std::make_unique(), + DB::S3::URI(endpoint + "/test-bucket/"), capabilities, + DB::ObjectStorageKeyGeneratorPtr{}, "disk"); +} + +DB::ContextPtr contextForTest() +{ + return getContext().context; +} + +/// The CAS-side fallback (CasGc.cpp's `removeChunkWriteOnceOrOneByOne`) keys specifically on +/// `NOT_IMPLEMENTED`; a capability-rejection test that only checks "threw a DB::Exception" would still +/// pass if this storage started throwing, say, BAD_ARGUMENTS instead -- which would silently break that +/// fallback while every assertion here kept passing. +void expectNotImplemented(const std::function & fn) +{ + try + { + fn(); + FAIL() << "expected a NOT_IMPLEMENTED exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED) << e.message(); + } +} + +} + +TEST(S3BulkDeleteFallback, PerKeyErrorsWithinASuccessfulBatchAreUnchanged) +{ + (void)contextForTest(); + + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + { + sendBatchSuccessWithErrors(response, "notfound-key", "denied-key"); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + + try + { + storage->removeObjectsIfExistUnderProfile( + {DB::StoredObject("present-key"), DB::StoredObject("notfound-key"), DB::StoredObject("denied-key")}, + DB::ObjectStorageControlRequest{}); + FAIL() << "expected removeObjectsIfExistUnderProfile to throw on the AccessDenied key"; + } + catch (const DB::S3Exception & e) + { + EXPECT_EQ(e.getS3ErrorCode(), Aws::S3::S3Errors::ACCESS_DENIED); + EXPECT_NE(e.message().find("denied-key"), std::string::npos) << e.message(); + } + + /// Exactly one DeleteObjects request; NoSuchKey and AccessDenied are both surfaced by the same + /// batch response, no fallback is expected here. + EXPECT_EQ(server.countMethod("POST"), 1u); + EXPECT_EQ(server.countMethod("DELETE"), 0u); +} + +TEST(S3BulkDeleteFallback, UnsupportedBatchReplyRecordsCapabilityFalseAndThrowsNotImplemented) +{ + (void)contextForTest(); + + std::atomic batch_attempts{0}; + ScriptedS3Server server([&](const Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) + { + ASSERT_EQ(request.getMethod(), "POST") << "capability false must never send anything, batch or per-key"; + ++batch_attempts; + sendBatchNotImplemented(response); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + + DB::StoredObjects objects{DB::StoredObject("key-a"), DB::StoredObject("key-b")}; + + expectNotImplemented([&] { storage->removeObjectsIfExistUnderProfile(objects, DB::ObjectStorageControlRequest{}); }); + EXPECT_EQ(batch_attempts.load(), 1u); + EXPECT_EQ(server.countMethod("DELETE"), 0u) << "this storage never loops over objects itself"; + + /// The capability is now known false on this storage: a second call must throw at once, without + /// even a `DeleteObjects` probe. + expectNotImplemented([&] { storage->removeObjectsIfExistUnderProfile(objects, DB::ObjectStorageControlRequest{}); }); + EXPECT_EQ(batch_attempts.load(), 1u) << "a second DeleteObjects attempt means the learned capability was not honoured"; + EXPECT_EQ(server.countMethod("DELETE"), 0u); +} + +/// A batch of exactly one object is always a plain `DeleteObject`: never sent as `DeleteObjects`, and +/// never gated on `s3_capabilities` at all -- proven here with the capability both explicitly false AND +/// left unknown (the default), since a single physical request is never something that check exists to +/// refuse. This is what makes the CAS engine's per-key fallback (CasGc.cpp) actually delete anything on +/// a backend that rejects `DeleteObjects` outright (GCS): a "batch" of one sent as `DeleteObjects` would +/// fail there identically to a bigger one. +TEST(S3BulkDeleteFallback, ExactlyOneObjectIsAlwaysAPlainDeleteObjectRegardlessOfCapability) +{ + (void)contextForTest(); + + for (const bool explicit_false : {false, true}) + { + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) + { + ASSERT_EQ(request.getMethod(), "DELETE"); + sendDeleteObjectSuccess(response); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{explicit_false ? std::optional{false} : std::nullopt}); + + EXPECT_NO_THROW(storage->removeObjectsIfExistUnderProfile({DB::StoredObject("solo-key")}, DB::ObjectStorageControlRequest{})); + + EXPECT_EQ(server.countMethod("POST"), 0u); + EXPECT_EQ(server.countMethod("DELETE"), 1u); + } +} + +/// The size-one path's own error handling, exactly as thorough as the batch path's: an absence is +/// ignored, and a real error is reported with the object's path. +TEST(S3BulkDeleteFallback, ExactlyOneObjectIgnoresAbsenceAndThrowsOnARealError) +{ + (void)contextForTest(); + + { + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + { + sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_NOT_FOUND, "NoSuchKey", "The specified key does not exist."); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + EXPECT_NO_THROW(storage->removeObjectsIfExistUnderProfile({DB::StoredObject("absent-key")}, DB::ObjectStorageControlRequest{})); + } + { + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + { + sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_FORBIDDEN, "AccessDenied", "Access Denied"); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + try + { + storage->removeObjectsIfExistUnderProfile({DB::StoredObject("denied-key")}, DB::ObjectStorageControlRequest{}); + FAIL() << "expected removeObjectsIfExistUnderProfile to throw on the AccessDenied key"; + } + catch (const DB::S3Exception & e) + { + EXPECT_EQ(e.getS3ErrorCode(), Aws::S3::S3Errors::ACCESS_DENIED); + EXPECT_NE(e.message().find("denied-key"), std::string::npos) << e.message(); + } + } +} + +TEST(S3BulkDeleteFallback, OtherFailureClassesKeepFailingClosedWithNoFallback) +{ + (void)contextForTest(); + + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + { + sendBatchInternalError(response); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + + EXPECT_THROW( + storage->removeObjectsIfExistUnderProfile( + {DB::StoredObject("key-a"), DB::StoredObject("key-b")}, DB::ObjectStorageControlRequest{}), + DB::Exception); + + EXPECT_EQ(server.countMethod("POST"), 1u); + EXPECT_EQ(server.countMethod("DELETE"), 0u) << "an ordinary batch failure must not fall back to per-key deletes"; +} + +TEST(S3BulkDeleteFallback, ExplicitlyDisabledCapabilityThrowsNotImplementedWithoutSendingAnything) +{ + (void)contextForTest(); + + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse &) + { + FAIL() << "an explicit false capability must never send anything, batch or per-key"; + }); + /// `false` in a disk's config resolves to this. + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{/*support_batch_delete_=*/false}); + + expectNotImplemented([&] + { + storage->removeObjectsIfExistUnderProfile( + {DB::StoredObject("key-a"), DB::StoredObject("key-b")}, DB::ObjectStorageControlRequest{}); + }); + + EXPECT_EQ(server.countMethod("POST"), 0u); + EXPECT_EQ(server.countMethod("DELETE"), 0u); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp b/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp new file mode 100644 index 000000000000..30ee3192bc54 --- /dev/null +++ b/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp @@ -0,0 +1,731 @@ +#include + +#include "config.h" + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include + +/// The single-attempt client clone must cap its connect timeout at the value the mount froze at open, +/// never at the disk's (possibly wider, possibly reloaded, possibly unbounded) own connect timeout. + +namespace +{ + +/// A `PocoHTTPClientConfiguration` that never resolves a real socket: `endpointOverride` points at a +/// port nothing listens on, so a test that never issues a request (every assertion here reads +/// `getClientConfiguration()`, which needs no network) never blocks or flakes on connection refusal. +DB::S3::PocoHTTPClientConfiguration clientConfigurationForTest(long connect_ms) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ true, + /* s3_slow_all_threads_after_retryable_error = */ true, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ true, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + cfg.endpointOverride = "http://127.0.0.1:1"; + cfg.connectTimeoutMs = connect_ms; + cfg.requestTimeoutMs = 30000; + return cfg; +} + +DB::S3::ClientSettings clientSettingsForTest() +{ + return DB::S3::ClientSettings{ + .use_virtual_addressing = false, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; +} + +/// A `` config section carrying `connect_timeout_ms`, for driving a reload through the real +/// `applyNewSettings` path (as a live disk's config reload would) rather than swapping the client +/// directly. Explicit static credentials keep the reload from falling through to the EC2 instance +/// metadata credentials provider (no access/secret key configured means "try every other provider"), +/// which would otherwise probe an unreachable metadata endpoint on every reload. +Poco::AutoPtr configWithConnectTimeout(long connect_timeout_ms) +{ + std::istringstream xml_stream( // STYLE_CHECK_ALLOW_STD_STRING_STREAM + "" + "" + std::to_string(connect_timeout_ms) + "" + "ACCESS_KEY_ID" + "SECRET_ACCESS_KEY" + ""); + return new Poco::Util::XMLConfiguration(xml_stream); +} + +std::shared_ptr makeStorageForTest(long connect_ms) +{ + auto client = DB::S3::ClientFactory::instance().create( + clientConfigurationForTest(connect_ms), clientSettingsForTest(), + "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); + return std::make_shared( + std::move(client), std::make_unique(), + DB::S3::URI("http://127.0.0.1:1/bucket/"), DB::S3Capabilities{}, + DB::ObjectStorageKeyGeneratorPtr{}, "disk"); +} + +/// A handler that always answers with a canned, verb-appropriate response after sleeping `delay` -- +/// simulating a slow-but-eventually-answering S3 endpoint. The sleep is deliberate test scaffolding for +/// a real elapsed-time discriminator, not a workaround for a race condition. Every request increments +/// `requests_seen`, the only way a caller can prove a client's retry strategy never reissued. +class DelayedResponseRequestHandler : public Poco::Net::HTTPRequestHandler +{ + std::atomic & requests_seen; + std::chrono::milliseconds delay; + std::function respond; + +public: + DelayedResponseRequestHandler( + std::atomic & requests_seen_, + std::chrono::milliseconds delay_, + std::function respond_) + : requests_seen(requests_seen_), delay(delay_), respond(std::move(respond_)) + { + } + + void handleRequest(Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) override + { + ++requests_seen; + std::this_thread::sleep_for(delay); + respond(response); + } +}; + +class DelayedResponseRequestHandlerFactory : public Poco::Net::HTTPRequestHandlerFactory +{ + std::atomic & requests_seen; + std::chrono::milliseconds delay; + std::function respond; + + Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override + { + return new DelayedResponseRequestHandler(requests_seen, delay, respond); + } + +public: + DelayedResponseRequestHandlerFactory( + std::atomic & requests_seen_, + std::chrono::milliseconds delay_, + std::function respond_) + : requests_seen(requests_seen_), delay(delay_), respond(std::move(respond_)) + { + } + + ~DelayedResponseRequestHandlerFactory() override = default; +}; + +/// A real local HTTP server standing in for S3, one verb at a time: every request gets the same canned +/// response after `delay`. Pointing a genuine `S3ObjectStorage` at it and comparing a `Default` call +/// (succeeds -- the delay is well under the base client's request timeout) against a `SingleAttempt` +/// call with a short caller timeout (times out, and the server counts exactly one request) is what +/// actually discriminates PRODUCTION client selection: no subclass stands between the test and +/// `S3ObjectStorage`'s own verb implementations. +class DelayedResponseServer +{ + std::unique_ptr server_socket; + Poco::SharedPtr handler_factory; + Poco::AutoPtr server_params; + std::unique_ptr server; + std::atomic requests_seen{0}; + +public: + DelayedResponseServer(std::chrono::milliseconds delay, std::function respond) + : server_socket(std::make_unique(0)) + , handler_factory(new DelayedResponseRequestHandlerFactory(requests_seen, delay, std::move(respond))) + , server_params(new Poco::Net::HTTPServerParams()) + , server(std::make_unique(handler_factory, *server_socket, server_params)) + { + server->start(); + } + + std::string getUrl() const { return "http://" + server_socket->address().toString(); } + size_t requestsSeen() const { return requests_seen.load(); } + void resetRequestsSeen() { requests_seen = 0; } +}; + +/// A genuine `S3ObjectStorage` pointed at `endpoint`. `base_request_timeout_ms` is the base client's +/// request AND connect timeout -- comfortably above the server's simulated delay, so a `Default` call +/// succeeds. No SDK-level retry (`RetryStrategy{.max_retries = 0}`, +/// `s3_slow_all_threads_after_retryable_error = false`): a retry would blur "the single-attempt clone +/// made exactly one request" into "the SDK also tried again". +std::shared_ptr makeDispatchStorageForTest(const std::string & endpoint, long base_request_timeout_ms) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ false, + /* s3_slow_all_threads_after_retryable_error = */ false, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ true, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + cfg.endpointOverride = endpoint; + cfg.connectTimeoutMs = base_request_timeout_ms; + cfg.requestTimeoutMs = base_request_timeout_ms; + /// The adaptive-timeout strategy gives the FIRST attempt a much shorter deadline than + /// `requestTimeoutMs` and only widens it on a later retry -- with SDK retries disabled above, that + /// first (short) deadline is the only one this client ever gets, which would time out well under + /// `server_delay` regardless of `requestTimeoutMs`. Off, so `requestTimeoutMs` governs uniformly. + cfg.s3_use_adaptive_timeouts = false; + /// Each `{ }` block below creates and destroys its OWN ephemeral-port server; the default 30 s + /// keep-alive would let the client pool a persistent connection that can outlive it. If a LATER + /// block's server happens to be assigned that same now-free port (routine under many back-to-back + /// server creations within one process), the pooled connection is reused against an unrelated dead + /// peer and the request fails with "Connection reset by peer" -- reproduced empirically by running + /// this file's dispatch tests together under `--gtest_repeat`. Disabling keep-alive forces a fresh + /// connection per request, which is what a short-lived test server should get anyway. + cfg.http_keep_alive_timeout = 0; + auto client = DB::S3::ClientFactory::instance().create( + cfg, clientSettingsForTest(), "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); + return std::make_shared( + std::move(client), std::make_unique(), + DB::S3::URI(endpoint + "/test-bucket/"), DB::S3Capabilities{}, + DB::ObjectStorageKeyGeneratorPtr{}, "disk"); +} + +DB::ContextPtr contextForTest() +{ + return getContext().context; +} + +} + +/// Test 6c of the spec: the clone's connect cap is the MIN of the base client's own connect timeout +/// and the requested cap, a configured-zero base is treated as unbounded (never "no limit"), the cache +/// key is the (request timeout, cap) pair, and a reloaded base client cannot widen a clone rebuilt for +/// the same cap. +TEST(S3SingleAttemptClient, ConnectTimeoutIsCappedAndFrozen) +{ + auto storage = makeStorageForTest(20000); + auto clone = storage->getSingleAttemptClient(/*request_timeout_ms=*/5000, /*connect_timeout_cap_ms=*/5000); + EXPECT_EQ(clone->getClientConfiguration().connectTimeoutMs, 5000); + EXPECT_EQ(clone->getClientConfiguration().requestTimeoutMs, 5000); + + auto narrow = makeStorageForTest(1000); + EXPECT_EQ(narrow->getSingleAttemptClient(5000, 5000)->getClientConfiguration().connectTimeoutMs, 1000); + /// A base of 0 means unbounded to Poco: it resolves to the cap, never to "no limit". + EXPECT_EQ(makeStorageForTest(0)->getSingleAttemptClient(5000, 1000)->getClientConfiguration().connectTimeoutMs, 1000); + /// Two caps under one request timeout are two clones: the cache key is the pair. + EXPECT_NE(narrow->getSingleAttemptClient(5000, 1000).get(), narrow->getSingleAttemptClient(5000, 500).get()); + + /// The reload path replaces the base client with a wider connect timeout, through the real + /// `applyNewSettings` config-reload path (as `SYSTEM RELOAD CONFIG` would drive it); a clone + /// rebuilt for the frozen cap 1000 stays at 1000. + auto reloaded = makeStorageForTest(1000); + (void)reloaded->getSingleAttemptClient(5000, 1000); + reloaded->applyNewSettings(*configWithConnectTimeout(5000), "disk", contextForTest(), + DB::IObjectStorage::ApplyNewSettingsOptions{.allow_client_change = true}); + EXPECT_EQ(reloaded->getSingleAttemptClient(5000, 1000)->getClientConfiguration().connectTimeoutMs, 1000); +} + +/// `shutdown()` used to call `DisableRequestProcessing` only on the main client, leaving every cached +/// single-attempt clone (`getSingleAttemptClient`) at its default enabled state. This test verifies only +/// that the flag now propagates to every clone and is restored by `startup()` -- it does NOT prove a +/// disabled clone rejects or interrupts a request: `DisableRequestProcessing` cannot prevent a request's +/// initial dispatch or interrupt one in flight, and a clone that is not currently retrying (every clone +/// here runs `SingleAttemptRetryStrategy`, which never retries) never has an occasion to consult it at +/// all. See the comment on `S3ObjectStorage::shutdown()` for what actually stops a new request after +/// shutdown (CAS engine admission, on a different plane). +TEST(S3SingleAttemptClient, ShutdownDisablesRequestProcessingOnCachedAndFutureClones) +{ + auto storage = makeStorageForTest(20000); + auto clone = storage->getSingleAttemptClient(/*request_timeout_ms=*/5000, /*connect_timeout_cap_ms=*/5000); + ASSERT_TRUE(clone->GetHttpClient()->IsRequestProcessingEnabled()); + + storage->shutdown(); + EXPECT_FALSE(clone->GetHttpClient()->IsRequestProcessingEnabled()) + << "a clone cached before shutdown() ran must have the flag propagated to it"; + + /// A clone for a (timeout, cap) pair never requested before, built WHILE shutdown is in effect, must + /// come into being with the flag already set -- not just the ones that existed when shutdown() ran. + auto clone_after_shutdown = storage->getSingleAttemptClient(/*request_timeout_ms=*/6000, /*connect_timeout_cap_ms=*/6000); + EXPECT_FALSE(clone_after_shutdown->GetHttpClient()->IsRequestProcessingEnabled()) + << "a clone built after shutdown() started must come into being with the flag already set too"; + + storage->startup(); + EXPECT_TRUE(clone->GetHttpClient()->IsRequestProcessingEnabled()); + EXPECT_TRUE(clone_after_shutdown->GetHttpClient()->IsRequestProcessingEnabled()); + + /// The ordinary case: a clone built with no shutdown in effect is enabled from the start. + auto clone_after_startup = storage->getSingleAttemptClient(/*request_timeout_ms=*/7000, /*connect_timeout_cap_ms=*/7000); + EXPECT_TRUE(clone_after_startup->GetHttpClient()->IsRequestProcessingEnabled()); +} + +/// The freeze computation `openPoolView` uses to build `pool_config.cas_request_budget.connect_timeout_cap_ms`, +/// isolated from any particular verb: the cap is the MIN of the base client's own connect timeout and +/// the attempt timeout, a configured-zero base normalizes to the attempt timeout itself (never "no +/// limit"), and the resulting envelope arithmetic matches `CasRequestBudget::attemptEnvelopeMs`. +TEST(CASEnvelopeWiring, FreezeConnectTimeoutCapSnapshot) +{ + /// A base client with connectTimeoutMs = 1000 and cas_attempt_timeout_ms = 5000: the narrower of + /// the two wins, and the envelope is attempt + 2 * cap = 7000. + auto storage = makeStorageForTest(1000); + const auto cap = DB::ContentAddressedMetadataStorage::freezeConnectTimeoutCapMs(storage, /*cas_attempt_timeout_ms=*/5000); + ASSERT_TRUE(cap.has_value()); + EXPECT_EQ(*cap, 1000u); + DB::Cas::CasRequestBudget budget{.attempt_timeout_ms = 5000, .connect_timeout_cap_ms = cap}; + EXPECT_EQ(budget.attemptEnvelopeMs(), 7000u); + + /// A base client with connectTimeoutMs = 0 (Poco "unbounded") and a TTL wide enough for the + /// resulting envelope (60000, per the spec's test 6e): the cap normalizes to the attempt timeout + /// itself, never to "no limit" -- a snapshot computing `min(0, attempt)` would report 0 here. + auto unbounded_storage = makeStorageForTest(0); + const auto wide_cap = DB::ContentAddressedMetadataStorage::freezeConnectTimeoutCapMs(unbounded_storage, /*cas_attempt_timeout_ms=*/5000); + ASSERT_TRUE(wide_cap.has_value()); + EXPECT_EQ(*wide_cap, 5000u); + DB::Cas::CasRequestBudget wide_budget{.attempt_timeout_ms = 5000, .connect_timeout_cap_ms = wide_cap}; + EXPECT_EQ(wide_budget.attemptEnvelopeMs(), 15000u); + EXPECT_NO_THROW(DB::Cas::validateCasRequestBudget(wide_budget, /*mount_lease_ttl_ms=*/60000, /*mount_renew_period_ms=*/10000, + /*background_renewal=*/false)); + + /// A storage with no S3 client (not exercised here -- every storage above is S3) freezes `nullopt`; + /// covered directly by `S3ObjectStorage::tryGetS3StorageClient` returning null for a non-S3 storage + /// and `freezeConnectTimeoutCapMs` short-circuiting on it. +} + +/// Every public verb whose retry profile is selectable is proven here to reach the client +/// `S3ObjectStorage::clientForRetryProfile` (private) actually picks for it -- through the storage's OWN +/// verb implementations, never a subclass override standing in for them. Per verb: a `Default` call +/// against a server that answers after `server_delay` succeeds (its client keeps the wide base timeout); +/// the SAME call under `SingleAttempt` with a caller timeout well under `server_delay` times out, and +/// the server counts exactly one request -- proving both that the short-timeout single-attempt clone +/// was selected (not the base client) and that its `SingleAttemptRetryStrategy` performs no +/// SDK-transparent retry. +TEST(CASEnvelopeWiring, ProductionDispatchSelectsTheFrozenSingleAttemptClientPerVerb) +{ + (void)contextForTest(); // getThreadPoolWriter/BlobStorageLogWriter::create fall back to the global context + + constexpr auto server_delay = std::chrono::milliseconds(1000); + constexpr long base_request_timeout_ms = 10000; + constexpr uint64_t single_attempt_timeout_ms = 100; + + auto singleAttemptRequest = [] + { + return DB::ObjectStorageControlRequest{ + .profile = DB::ObjectStorageRetryProfile::SingleAttempt, + .attempt_timeout_ms = single_attempt_timeout_ms, + .connect_timeout_cap_ms = single_attempt_timeout_ms}; + }; + + /// PUT: writeObject; the profile rides on WriteSettings, not an ObjectStorageControlRequest. + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + response.set("ETag", "\"put-etag\""); + response.setContentLength(0); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + auto put = [&](DB::ObjectStorageRetryProfile profile, uint64_t timeout_ms) + { + DB::WriteSettings write_settings; + write_settings.object_storage_retry_profile = profile; + write_settings.object_storage_attempt_timeout_ms = timeout_ms; + write_settings.object_storage_connect_timeout_cap_ms = timeout_ms; + auto buffer = storage->writeObject( + DB::StoredObject("put-key"), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, write_settings); + buffer->write('A'); + buffer->finalize(); + }; + + EXPECT_NO_THROW(put(DB::ObjectStorageRetryProfile::Default, 0)); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + EXPECT_THROW(put(DB::ObjectStorageRetryProfile::SingleAttempt, single_attempt_timeout_ms), DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } + + /// HEAD: tryGetObjectMetadataWithNativeToken's ObjectStorageControlRequest-taking overload. + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + response.set("ETag", "\"head-etag\""); + response.setContentLength(5); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + EXPECT_TRUE(storage->tryGetObjectMetadataWithNativeToken( + "head-key", /*with_tags=*/false, DB::ObjectStorageControlRequest{}).has_value()); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + EXPECT_THROW( + storage->tryGetObjectMetadataWithNativeToken("head-key", /*with_tags=*/false, singleAttemptRequest()), + DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } + + /// Conditional DELETE: removeObjectIfTokenMatches's ObjectStorageControlRequest-taking overload. + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + response.setStatus(Poco::Net::HTTPResponse::HTTP_NO_CONTENT); + response.setContentLength(0); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + auto result = storage->removeObjectIfTokenMatches( + DB::StoredObject("delete-key"), "\"etag\"", DB::ObjectStorageControlRequest{}); + EXPECT_EQ(result.outcome, DB::ConditionalRemoveOutcome::Removed); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + EXPECT_THROW( + storage->removeObjectIfTokenMatches(DB::StoredObject("delete-key"), "\"etag\"", singleAttemptRequest()), + DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } + + /// Bulk DELETE: removeObjectsIfExistUnderProfile (one DeleteObjects request for the whole batch). + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + static const std::string body = + "" + ""; + response.setContentType("application/xml"); + response.setContentLength(body.size()); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + auto & out = response.send(); + out << body; + out.flush(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + EXPECT_NO_THROW(storage->removeObjectsIfExistUnderProfile( + {DB::StoredObject("bulk-delete-key")}, DB::ObjectStorageControlRequest{})); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + EXPECT_THROW( + storage->removeObjectsIfExistUnderProfile({DB::StoredObject("bulk-delete-key")}, singleAttemptRequest()), + DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } + + /// LIST: iterate's ObjectStorageControlRequest-taking overload. The ListObjectsV2 call happens + /// lazily, on the async iterator's first `isValid()`. + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + static const std::string body = + "" + "" + "test-bucket01000" + "false"; + response.setContentType("application/xml"); + response.setContentLength(body.size()); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + auto & out = response.send(); + out << body; + out.flush(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + auto default_iterator = storage->iterate("p/", /*max_keys=*/10, /*with_tags=*/false, {}, DB::ObjectStorageControlRequest{}); + EXPECT_NO_THROW(default_iterator->isValid()); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + auto single_attempt_iterator = storage->iterate("p/", /*max_keys=*/10, /*with_tags=*/false, {}, singleAttemptRequest()); + EXPECT_THROW(single_attempt_iterator->isValid(), DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } + + /// GET: readObject; the profile rides on ReadSettings. The request happens lazily, on the buffer's + /// first read. + { + DelayedResponseServer server(server_delay, [](Poco::Net::HTTPServerResponse & response) + { + static const std::string body = "hello"; + response.set("ETag", "\"get-etag\""); + response.setContentType("binary/octet-stream"); + response.setContentLength(body.size()); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + auto & out = response.send(); + out << body; + out.flush(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_request_timeout_ms); + + auto get = [&](DB::ObjectStorageRetryProfile profile, uint64_t timeout_ms) + { + DB::ReadSettings read_settings; + read_settings.object_storage_retry_profile = profile; + read_settings.object_storage_attempt_timeout_ms = timeout_ms; + read_settings.object_storage_connect_timeout_cap_ms = timeout_ms; + auto buffer = storage->readObject(DB::StoredObject("get-key"), read_settings); + std::string content; + DB::readStringUntilEOF(content, *buffer); + return content; + }; + + EXPECT_EQ(get(DB::ObjectStorageRetryProfile::Default, 0), "hello"); + EXPECT_EQ(server.requestsSeen(), 1u); + + server.resetRequestsSeen(); + EXPECT_THROW(get(DB::ObjectStorageRetryProfile::SingleAttempt, single_attempt_timeout_ms), DB::Exception); + EXPECT_EQ(server.requestsSeen(), 1u); + } +} + +/// The test above proves production dispatch selects a short-REQUEST-timeout clone, but every server +/// there answers every request -- it can never tell whether the frozen `connect_timeout_cap_ms` reaches +/// the CONNECTION phase at all, only whether SOME clone with a short deadline was picked. This test +/// closes that gap WITHOUT any wall-clock measurement or stalled connect: every call below goes through +/// production dispatch against an ordinary, immediately-answering server, so it can only prove two +/// clock-free facts. First, that dispatch built (or reused) the single-attempt clone under EXACTLY the +/// (attempt timeout, connect cap) key the request carried -- `hasSingleAttemptClientForTest` only +/// inspects `S3ObjectStorage`'s clone cache, it never creates an entry, so a wrong key or a missing clone +/// fails the assertion immediately rather than timing out. Second, that the clone found under that key +/// actually carries the cap as its `connectTimeoutMs`, while the Default profile's own client keeps the +/// disk's (wider) base connect timeout untouched. Whether Poco's HTTP client actually enforces +/// `connectTimeoutMs` at the socket level is `PocoHTTPClient`/`Poco::Net::HTTPClientSession` behaviour +/// upstream of this class, and is not re-proved here; `S3SingleAttemptClient.ConnectTimeoutIsCappedAndFrozen` +/// above pins the MIN/cache-key arithmetic `getSingleAttemptClient` applies in isolation. +TEST(CASEnvelopeWiring, ProductionDispatchAppliesTheFrozenConnectCapAtConnectTime) +{ + (void)contextForTest(); // getThreadPoolWriter/BlobStorageLogWriter::create fall back to the global context + + constexpr long base_connect_timeout_ms = 2000; + constexpr uint64_t single_attempt_timeout_ms = 5000; + constexpr uint64_t single_attempt_connect_cap_ms = 100; + + /// PUT: writeObject; the profile and cap ride on WriteSettings, not an ObjectStorageControlRequest. + { + DelayedResponseServer server(std::chrono::milliseconds(0), [](Poco::Net::HTTPServerResponse & response) + { + response.set("ETag", "\"put-etag\""); + response.setContentLength(0); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_connect_timeout_ms); + + auto put = [&](DB::ObjectStorageRetryProfile profile, uint64_t attempt_timeout_ms, uint64_t connect_cap_ms) + { + DB::WriteSettings write_settings; + write_settings.object_storage_retry_profile = profile; + write_settings.object_storage_attempt_timeout_ms = attempt_timeout_ms; + write_settings.object_storage_connect_timeout_cap_ms = connect_cap_ms; + auto buffer = storage->writeObject( + DB::StoredObject("put-key"), DB::WriteMode::Rewrite, {}, DB::DBMS_DEFAULT_BUFFER_SIZE, write_settings); + buffer->write('A'); + buffer->finalize(); + }; + + EXPECT_NO_THROW(put(DB::ObjectStorageRetryProfile::Default, 0, 0)); + EXPECT_EQ(storage->getS3StorageClient()->getClientConfiguration().connectTimeoutMs, base_connect_timeout_ms) + << "the Default profile must dispatch on the disk's own client, unchanged"; + + EXPECT_NO_THROW(put(DB::ObjectStorageRetryProfile::SingleAttempt, single_attempt_timeout_ms, single_attempt_connect_cap_ms)); + ASSERT_TRUE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, single_attempt_connect_cap_ms)) + << "dispatch must have built the single-attempt clone under exactly this (attempt timeout, cap) key"; + EXPECT_FALSE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, 0)) + << "dispatch must not fall back to an uncapped clone for this attempt timeout"; + EXPECT_EQ( + storage->getSingleAttemptClient(single_attempt_timeout_ms, single_attempt_connect_cap_ms) + ->getClientConfiguration().connectTimeoutMs, + static_cast(single_attempt_connect_cap_ms)); + } + + /// HEAD: tryGetObjectMetadataWithNativeToken's ObjectStorageControlRequest-taking overload. + { + DelayedResponseServer server(std::chrono::milliseconds(0), [](Poco::Net::HTTPServerResponse & response) + { + response.set("ETag", "\"head-etag\""); + response.setContentLength(5); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_connect_timeout_ms); + + EXPECT_TRUE(storage->tryGetObjectMetadataWithNativeToken( + "head-key", /*with_tags=*/false, DB::ObjectStorageControlRequest{}).has_value()); + EXPECT_EQ(storage->getS3StorageClient()->getClientConfiguration().connectTimeoutMs, base_connect_timeout_ms) + << "the Default profile must dispatch on the disk's own client, unchanged"; + + EXPECT_TRUE(storage->tryGetObjectMetadataWithNativeToken( + "head-key", /*with_tags=*/false, + DB::ObjectStorageControlRequest{ + .profile = DB::ObjectStorageRetryProfile::SingleAttempt, + .attempt_timeout_ms = single_attempt_timeout_ms, + .connect_timeout_cap_ms = single_attempt_connect_cap_ms}).has_value()); + ASSERT_TRUE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, single_attempt_connect_cap_ms)) + << "dispatch must have built the single-attempt clone under exactly this (attempt timeout, cap) key"; + EXPECT_FALSE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, 0)) + << "dispatch must not fall back to an uncapped clone for this attempt timeout"; + EXPECT_EQ( + storage->getSingleAttemptClient(single_attempt_timeout_ms, single_attempt_connect_cap_ms) + ->getClientConfiguration().connectTimeoutMs, + static_cast(single_attempt_connect_cap_ms)); + } + + /// Conditional DELETE: removeObjectIfTokenMatches's ObjectStorageControlRequest-taking overload. + { + DelayedResponseServer server(std::chrono::milliseconds(0), [](Poco::Net::HTTPServerResponse & response) + { + response.setStatus(Poco::Net::HTTPResponse::HTTP_NO_CONTENT); + response.setContentLength(0); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_connect_timeout_ms); + + const auto default_result = storage->removeObjectIfTokenMatches( + DB::StoredObject("delete-key"), "\"etag\"", DB::ObjectStorageControlRequest{}); + EXPECT_EQ(default_result.outcome, DB::ConditionalRemoveOutcome::Removed); + EXPECT_EQ(storage->getS3StorageClient()->getClientConfiguration().connectTimeoutMs, base_connect_timeout_ms) + << "the Default profile must dispatch on the disk's own client, unchanged"; + + const auto capped_result = storage->removeObjectIfTokenMatches( + DB::StoredObject("delete-key"), "\"etag\"", + DB::ObjectStorageControlRequest{ + .profile = DB::ObjectStorageRetryProfile::SingleAttempt, + .attempt_timeout_ms = single_attempt_timeout_ms, + .connect_timeout_cap_ms = single_attempt_connect_cap_ms}); + EXPECT_EQ(capped_result.outcome, DB::ConditionalRemoveOutcome::Removed); + ASSERT_TRUE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, single_attempt_connect_cap_ms)) + << "dispatch must have built the single-attempt clone under exactly this (attempt timeout, cap) key"; + EXPECT_FALSE(storage->hasSingleAttemptClientForTest(single_attempt_timeout_ms, 0)) + << "dispatch must not fall back to an uncapped clone for this attempt timeout"; + EXPECT_EQ( + storage->getSingleAttemptClient(single_attempt_timeout_ms, single_attempt_connect_cap_ms) + ->getClientConfiguration().connectTimeoutMs, + static_cast(single_attempt_connect_cap_ms)); + } +} + +/// `FreezeConnectTimeoutCapSnapshot` above pins the ARITHMETIC of `freezeConnectTimeoutCapMs` in +/// isolation; `ProductionDispatchAppliesTheFrozenConnectCapAtConnectTime` pins that a cap handed +/// DIRECTLY to `S3ObjectStorage` reaches the connect phase. Neither proves the composition +/// `ContentAddressedMetadataStorage::openPoolView` actually performs: freezing the cap from a real S3 +/// client (`ContentAddressedMetadataStorage.cpp` ~802) and handing it into `Cas::ObjectStorageBackend`'s +/// constructor (the backend handoff at ~812-822) exactly as a writable Native mount does. This test +/// drives that whole chain end to end -- real client -> freezeConnectTimeoutCapMs -> ObjectStorageBackend +/// -> CasRequests/CasOperation -> the SAME production S3ObjectStorage dispatch the tests above cover -- +/// with no recording subclass anywhere in it, and, like the test above, with no wall-clock measurement: +/// both backends' HEAD goes through an ordinary, immediately-answering server. +/// +/// A read-only backend (`single_attempt_control_plane_ = false`, matching `openPoolView`'s own choice for +/// a read-only mount) dispatches its read-class requests under the Default profile -- proven here by the +/// storage never having built ANY single-attempt clone afterward, i.e. it used the disk's own client +/// untouched. The SAME derived cap and attempt timeout, handed to a WRITABLE Native backend exactly as +/// `openPoolView` constructs one, must then dispatch under EXACTLY that (attempt timeout, cap) key, and +/// the clone found under that key must carry the cap as its `connectTimeoutMs`: a dropped or corrupted +/// handoff anywhere in the chain would either leave no clone under that key or leave one with the wrong +/// timeout, and either way the assertion below fails immediately rather than by timing out. +TEST(CASEnvelopeWiring, FreezeConnectTimeoutCapReachesTheBackendOverProductionDispatch) +{ + (void)contextForTest(); + + constexpr long base_connect_timeout_ms = 2000; + constexpr uint64_t cas_attempt_timeout_ms = 100; + + DelayedResponseServer server(std::chrono::milliseconds(0), [](Poco::Net::HTTPServerResponse & response) + { + response.set("ETag", "\"head-etag\""); + response.setContentLength(5); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); + }); + auto storage = makeDispatchStorageForTest(server.getUrl(), base_connect_timeout_ms); + + /// The exact derivation `ContentAddressedMetadataStorage::openPoolView` uses: min(base connect + /// timeout, attempt timeout) = 100 here, never the wide 2000 ms base timeout. + const auto cap = DB::ContentAddressedMetadataStorage::freezeConnectTimeoutCapMs(storage, cas_attempt_timeout_ms); + ASSERT_TRUE(cap.has_value()); + EXPECT_EQ(*cap, cas_attempt_timeout_ms); + + auto uncapped_backend = std::make_shared( + storage, DB::Cas::ObjectStorageBackend::Mode::Native, + /*single_attempt_control_plane_=*/false, /*attempt_timeout_ms_=*/0, /*connect_timeout_cap_ms_=*/0); + { + DB::Cas::CasRequests requests(DB::Cas::BackendPtr(uncapped_backend), DB::Cas::Fence::open()); + auto op = requests.admit(); + EXPECT_TRUE(op.head("k", DB::Cas::Retry::once()).has_value()); + EXPECT_FALSE(storage->hasSingleAttemptClientForTest(0, 0)) + << "a read-only (Default-profile) backend must never build a single-attempt clone"; + } + + /// The derived cap, handed to the backend exactly as `openPoolView` constructs it (:812-822) for a + /// WRITABLE Native mount. + auto capped_backend = std::make_shared( + storage, DB::Cas::ObjectStorageBackend::Mode::Native, + /*single_attempt_control_plane_=*/true, cas_attempt_timeout_ms, *cap); + /// Cheap, deterministic corroboration alongside the dispatch-level assertions below: it proves the + /// constructor argument was stored, not that it reached the S3 client's actual connect timeout, which + /// only `hasSingleAttemptClientForTest`/`getSingleAttemptClient` below can show. + EXPECT_EQ(capped_backend->connectTimeoutCapMs(), *cap); + { + DB::Cas::CasRequests requests(DB::Cas::BackendPtr(capped_backend), DB::Cas::Fence::open()); + auto op = requests.admit(); + EXPECT_TRUE(op.head("k", DB::Cas::Retry::once()).has_value()); + ASSERT_TRUE(storage->hasSingleAttemptClientForTest(cas_attempt_timeout_ms, *cap)) + << "the WRITABLE backend must dispatch its read-class requests under exactly the frozen " + "(attempt timeout, cap) key"; + EXPECT_EQ( + storage->getSingleAttemptClient(cas_attempt_timeout_ms, *cap)->getClientConfiguration().connectTimeoutMs, + static_cast(*cap)) + << "socket-level enforcement of connectTimeoutMs is PocoHTTPClient behaviour upstream of this " + "class, not re-proved here"; + } +} + +#endif diff --git a/src/Disks/tests/gtest_cas_s3_staging.cpp b/src/Disks/tests/gtest_cas_s3_staging.cpp index 92c4542b1a32..74a82df51569 100644 --- a/src/Disks/tests/gtest_cas_s3_staging.cpp +++ b/src/Disks/tests/gtest_cas_s3_staging.cpp @@ -49,6 +49,29 @@ namespace DB::ErrorCodes namespace { +/// ---- Small raw-fixture request-engine wrappers shared by the tests below ---- + +/// The durable object at `key`, or `nullopt`. +std::optional readAt(DB::Cas::Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).read(key, DB::Cas::Retry::once()); +} + +/// Unconditional create of a fresh key (the fixture's own setup, never a real conflict). +void createAt(DB::Cas::Backend & backend, const String & key, const String & bytes) +{ + DB::Cas::tests::OperationForTest op(backend); + EXPECT_TRUE(std::holds_alternative((*op).create(key, bytes, DB::Cas::Retry::once()))); +} + +/// The current metadata at `key`, or `nullopt`. +std::optional headAt(DB::Cas::Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, DB::Cas::Retry::once()); +} + /// Build a `Poco::Util::XMLConfiguration` with `inner_xml` nested under a `` element (mirrors /// the shape a real CAS disk config has under `storage_configuration.disks.`, so /// `config_prefix = "disk"` reads exactly like the disk factory's `config_prefix`). @@ -156,25 +179,24 @@ class RecordingStagingBackend : public DB::Cas::InMemoryBackend std::vector copy_calls; - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + void publish(const DB::Cas::BlobPublishRequest & request, DB::Cas::TransportAccess & access) override { if (const auto * copy = std::get_if(&request.publication)) copy_calls.push_back({copy->object_key, request.destination_key, true}); else copy_calls.push_back({String{}, request.destination_key, false}); - DB::Cas::InMemoryBackend::publishBlob(request); + DB::Cas::InMemoryBackend::publish(request, access); } - /// Every key read as a stream, with a count. Republishing opens its source with `getStream`, so + /// Every key read as a stream, with a count. Republishing opens its source with `stream`, so /// this counts exactly those reads -- and deliberately not the materializing - /// `get`, which the assertions themselves use to inspect bodies. + /// `read`, which the assertions themselves use to inspect bodies. std::map reads_of; - using DB::Cas::InMemoryBackend::getStream; - std::optional getStream(const String & key, DB::Cas::Range range) override + std::unique_ptr stream(const String & key, DB::Cas::TransportAccess & access) override { ++reads_of[key]; - return DB::Cas::InMemoryBackend::getStream(key, range); + return DB::Cas::InMemoryBackend::stream(key, access); } @@ -202,32 +224,33 @@ class EtagFaithfulPublicationBackend final : public DB::Cas::InMemoryBackend explicit EtagFaithfulPublicationBackend(FaultScript script_) : script(script_) {} - DB::Cas::HeadResult head(const String & key) override + std::optional head(const String & key, DB::Cas::TransportAccess & access) override { - DB::Cas::HeadResult result = DB::Cas::InMemoryBackend::head(key); - if (result.exists && isBlobBodyKey(key)) + std::optional result = DB::Cas::InMemoryBackend::head(key, access); + if (result && isBlobBodyKey(key)) { - const auto body = DB::Cas::InMemoryBackend::get(key); + const auto body = DB::Cas::InMemoryBackend::read(key, access); chassert(body.has_value()); - result.token = DB::Cas::Token{sipHash128String(body->bytes), DB::Cas::TokenType::ETag}; + result->value = sipHash128String(body->bytes); } return result; } - DB::Cas::DeleteOutcome deleteExact(const String & key, const DB::Cas::Token & token) override + DB::Cas::Backend::RawRemoval remove(const String & key, const String & expected_value, + DB::Cas::TransportAccess & access) override { if (!isBlobBodyKey(key)) - return DB::Cas::InMemoryBackend::deleteExact(key, token); - - const DB::Cas::HeadResult current = head(key); - if (!current.exists) - return DB::Cas::DeleteOutcome{.kind = DB::Cas::DeleteOutcome::Kind::NotFound}; - if (current.token != token) - return DB::Cas::DeleteOutcome{.kind = DB::Cas::DeleteOutcome::Kind::TokenMismatch}; - return DB::Cas::InMemoryBackend::deleteExact(key, DB::Cas::InMemoryBackend::head(key).token); + return DB::Cas::InMemoryBackend::remove(key, expected_value, access); + + const auto current = head(key, access); + if (!current) + return DB::Cas::Backend::RawRemoval::Gone; + if (current->value != expected_value) + return DB::Cas::Backend::RawRemoval::Mismatch; + return DB::Cas::InMemoryBackend::remove(key, DB::Cas::InMemoryBackend::head(key, access)->value, access); } - void publishBlob(const DB::Cas::BlobPublishRequest & request) override + void publish(const DB::Cas::BlobPublishRequest & request, DB::Cas::TransportAccess & access) override { const bool is_copy = std::holds_alternative(request.publication); if (is_copy) @@ -241,24 +264,35 @@ class EtagFaithfulPublicationBackend final : public DB::Cas::InMemoryBackend || (script == FaultScript::FirstCondemnedStreamLandsThenDeleted && !is_copy))) { fault_fired = true; - DB::Cas::InMemoryBackend::publishBlob(request); - queued_delete_token = head(request.destination_key).token; + DB::Cas::InMemoryBackend::publish(request, access); + queued_delete_token = head(request.destination_key, access)->value; + /// The test wants to replay this exact captured value later as a delete precondition, to + /// prove a retag defeats it. `Etag` is never constructible from a raw string, so the only + /// way to hold a replayable one is to mint it -- through a nested admitted operation, over + /// this same backend instance -- at the exact moment the raw value above was observed. + { + DB::Cas::tests::OperationForTest mint_op(*this); + const auto meta = (*mint_op).head(request.destination_key, DB::Cas::Retry::once()); + if (meta) + queued_delete_etag = meta->etag; + } if (script != FaultScript::CopyLandsThenCondemned) - first_delete = deleteExact(request.destination_key, queued_delete_token); + first_delete = remove(request.destination_key, queued_delete_token, access); throw Poco::TimeoutException("ETag-faithful staged publication response lost"); } - DB::Cas::InMemoryBackend::publishBlob(request); + DB::Cas::InMemoryBackend::publish(request, access); } FaultScript script; bool fault_fired = false; size_t copy_publications = 0; size_t streaming_publications = 0; - DB::Cas::Token queued_delete_token; - DB::Cas::DeleteOutcome first_delete; + String queued_delete_token; + std::optional queued_delete_etag; + DB::Cas::Backend::RawRemoval first_delete{}; private: static bool isBlobBodyKey(const String & key) @@ -301,12 +335,13 @@ DB::Cas::BlobSource reReadableStagedSource( source.server_side_copy_from = staging_key; source.open = [backend, staging_key, header_len]() -> std::unique_ptr { - auto staged = backend->getStream(staging_key); + DB::Cas::tests::OperationForTest op(backend); + auto staged = (*op).stream(staging_key, DB::Cas::Retry::standard()); if (!staged) throw DB::Exception(DB::ErrorCodes::FILE_DOESNT_EXIST, "staging object {} is absent", staging_key); String encoded_header(header_len, '\0'); - staged->stream->readStrict(encoded_header.data(), encoded_header.size()); + staged->readStrict(encoded_header.data(), encoded_header.size()); const DB::Cas::EnvelopeHeader decoded = DB::Cas::decodeEnvelopeHeader(encoded_header, encoded_header.size(), DB::Cas::ObjectKind::Blob); if (decoded.header_len != header_len) @@ -316,7 +351,7 @@ DB::Cas::BlobSource reReadableStagedSource( staging_key, decoded.header_len, header_len); - return std::move(staged->stream); + return staged; }; return source; } @@ -341,7 +376,7 @@ TEST(CASS3Staging, StagedCopyCondemnedRetryRetagsBeforeQueuedDelete) const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); const String staging_key = "p/staging/mount1/etag-condemned.tmp"; const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{101}); - backend->putIfAbsent(staging_key, staging_bytes); + createAt(*backend, staging_key, staging_bytes); DB::Cas::tests::writeMetaClean(*backend, store->layout(), DB::Cas::tests::u128Of(payload), payload.size()); DB::Cas::tests::condemnMeta(*backend, store->layout(), DB::Cas::tests::u128Of(payload), 31); auto build = precommittedBuildFor( @@ -354,10 +389,13 @@ TEST(CASS3Staging, StagedCopyCondemnedRetryRetagsBeforeQueuedDelete) EXPECT_EQ(backend->copy_publications, 1u); EXPECT_EQ(backend->streaming_publications, 1u); - EXPECT_EQ( - backend->deleteExact(store->layout().blobKey(ref), backend->queued_delete_token).kind, - DB::Cas::DeleteOutcome::Kind::TokenMismatch); - const auto current = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(backend->queued_delete_etag.has_value()); + { + DB::Cas::tests::OperationForTest op(*backend); + EXPECT_EQ((*op).remove(store->layout().blobKey(ref), *backend->queued_delete_etag, DB::Cas::Retry::once()), + DB::Cas::Removal::Mismatch); + } + const auto current = readAt(*backend, store->layout().blobKey(ref)); ASSERT_TRUE(current.has_value()); EXPECT_NE(current->bytes, staging_bytes); EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); @@ -373,7 +411,7 @@ TEST(CASS3Staging, StagedCopyDeletedBeforeAbsentRetryRetagsBeforeQueuedDelete) const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); const String staging_key = "p/staging/mount1/etag-deleted.tmp"; const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{202}); - backend->putIfAbsent(staging_key, staging_bytes); + createAt(*backend, staging_key, staging_bytes); auto build = precommittedBuildFor( store, DB::Cas::RootNamespace{"srv1/etag-deleted"}, "part", DB::Cas::tests::u128Of(payload), payload.size()); @@ -382,15 +420,18 @@ TEST(CASS3Staging, StagedCopyDeletedBeforeAbsentRetryRetagsBeforeQueuedDelete) ref, reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); - EXPECT_EQ(backend->first_delete.kind, DB::Cas::DeleteOutcome::Kind::Deleted); + EXPECT_EQ(backend->first_delete, DB::Cas::Backend::RawRemoval::Removed); EXPECT_EQ(backend->copy_publications, 1u) << "the absent retry must not copy the original staged envelope again"; EXPECT_EQ(backend->streaming_publications, 1u); - EXPECT_EQ( - backend->deleteExact(store->layout().blobKey(ref), backend->queued_delete_token).kind, - DB::Cas::DeleteOutcome::Kind::TokenMismatch) - << "the second queued exact delete for the copied ETag must miss the retagged replacement"; - const auto current = backend->get(store->layout().blobKey(ref)); + ASSERT_TRUE(backend->queued_delete_etag.has_value()); + { + DB::Cas::tests::OperationForTest op(*backend); + EXPECT_EQ((*op).remove(store->layout().blobKey(ref), *backend->queued_delete_etag, DB::Cas::Retry::once()), + DB::Cas::Removal::Mismatch) + << "the second queued exact delete for the copied ETag must miss the retagged replacement"; + } + const auto current = readAt(*backend, store->layout().blobKey(ref)); ASSERT_TRUE(current.has_value()); EXPECT_NE(current->bytes, staging_bytes); EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); @@ -406,11 +447,17 @@ TEST(CASS3Staging, FirstCondemnedAttemptThenAbsentRetryNeverRecopies) const DB::Cas::BlobRef ref = DB::Cas::tests::idOf(payload); const String staging_key = "p/staging/mount1/etag-first-condemned.tmp"; const String staging_bytes = stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{303}); - backend->putIfAbsent(staging_key, staging_bytes); - backend->putIfAbsent(store->layout().blobKey(ref), staging_bytes); + createAt(*backend, staging_key, staging_bytes); + createAt(*backend, store->layout().blobKey(ref), staging_bytes); DB::Cas::tests::writeMetaClean(*backend, store->layout(), DB::Cas::tests::u128Of(payload), payload.size()); DB::Cas::tests::condemnMeta(*backend, store->layout(), DB::Cas::tests::u128Of(payload), 37); - const DB::Cas::Token original_staged_etag = backend->head(store->layout().blobKey(ref)).token; + /// Captured through a real admitted operation, so it is a genuinely replayable `Etag` -- never + /// constructible from a bare raw value -- for the later mismatch check below. + DB::Cas::Etag original_staged_etag = [&] + { + DB::Cas::tests::OperationForTest op(*backend); + return (*op).head(store->layout().blobKey(ref), DB::Cas::Retry::once())->etag; + }(); auto build = precommittedBuildFor( store, DB::Cas::RootNamespace{"srv1/etag-first-condemned"}, "part", DB::Cas::tests::u128Of(payload), payload.size()); @@ -419,14 +466,16 @@ TEST(CASS3Staging, FirstCondemnedAttemptThenAbsentRetryNeverRecopies) ref, reReadableStagedSource(backend, staging_key, payload.size(), store->poolMeta().blob_header_len)); - EXPECT_EQ(backend->first_delete.kind, DB::Cas::DeleteOutcome::Kind::Deleted); + EXPECT_EQ(backend->first_delete, DB::Cas::Backend::RawRemoval::Removed); EXPECT_EQ(backend->copy_publications, 0u) << "a first condemned publication and every later absent retry must stream, never copy"; EXPECT_EQ(backend->streaming_publications, 2u); - EXPECT_EQ( - backend->deleteExact(store->layout().blobKey(ref), original_staged_etag).kind, - DB::Cas::DeleteOutcome::Kind::TokenMismatch); - const auto current = backend->get(store->layout().blobKey(ref)); + { + DB::Cas::tests::OperationForTest op(*backend); + EXPECT_EQ((*op).remove(store->layout().blobKey(ref), original_staged_etag, DB::Cas::Retry::once()), + DB::Cas::Removal::Mismatch); + } + const auto current = readAt(*backend, store->layout().blobKey(ref)); ASSERT_TRUE(current.has_value()); EXPECT_NE(current->bytes, staging_bytes); EXPECT_EQ(current->bytes.substr(store->poolMeta().blob_header_len), payload); @@ -603,7 +652,7 @@ TEST(CASS3Staging, PromoteViaServerSideCopyCreatesFreshBlobMaterializedProof) const std::string staging_key = "p/staging/mount1/aaa.tmp"; const std::string staging_bytes = stagedBytes( store->poolMeta().blob_header_len, payload, DB::UInt128{0xA}); - backend->putIfAbsent(staging_key, staging_bytes); + createAt(*backend, staging_key, staging_bytes); auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); const DB::Cas::PutBlobResult bref = build->putBlob( @@ -619,13 +668,13 @@ TEST(CASS3Staging, PromoteViaServerSideCopyCreatesFreshBlobMaterializedProof) /// Successful publication records materialized evidence; the backend still owns the destination token. EXPECT_EQ(build->dependencyProof(blob_id), DB::Cas::BlobDependencyProof::Materialized); - const DB::Cas::HeadResult hr = backend->head(blob_key); - ASSERT_TRUE(hr.exists); - EXPECT_FALSE(hr.token.empty()); + const auto hr = headAt(*backend, blob_key); + ASSERT_TRUE(hr.has_value()); + EXPECT_FALSE(DB::Cas::PersistedEtag::capture(hr->etag).value.empty()); EXPECT_EQ(bref.size, payload.size()); /// The promoted blob body IS the staging bytes (server-side copy moved them verbatim). - const auto got = backend->get(blob_key); + const auto got = readAt(*backend, blob_key); ASSERT_TRUE(got.has_value()); EXPECT_EQ(got->bytes, staging_bytes); } @@ -643,17 +692,19 @@ TEST(CASS3Staging, PromoteOverExistingCleanBlobAdoptsAndNeverOverwrites) const DB::Cas::BlobRef blob_id{DB::Cas::BlobHashAlgo::CityHash128, DB::Cas::BlobDigest::fromU128(hash)}; const std::string blob_key = store->layout().blobKey(blob_id); const std::string staging_key = "p/staging/mount1/bbb.tmp"; - backend->putIfAbsent( + createAt( + *backend, staging_key, stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{0xB})); /// A pre-existing, well-formed, CLEAN blob (envelope + payload) already at the content key. - backend->putIfAbsent( + createAt( + *backend, blob_key, stagedBytes(store->poolMeta().blob_header_len, payload, DB::UInt128{0xBB})); DB::Cas::tests::writeMetaClean(*backend, store->layout(), hash, payload.size()); - const DB::Cas::HeadResult before = backend->head(blob_key); - ASSERT_TRUE(before.exists); + const auto before = headAt(*backend, blob_key); + ASSERT_TRUE(before.has_value()); auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); build->putBlob( @@ -665,8 +716,9 @@ TEST(CASS3Staging, PromoteOverExistingCleanBlobAdoptsAndNeverOverwrites) EXPECT_EQ(backend->streamingPublicationCount(), 0u); /// The existing incarnation is untouched: same token, same bytes. - const DB::Cas::HeadResult after = backend->head(blob_key); - EXPECT_EQ(after.token, before.token); + const auto after = headAt(*backend, blob_key); + ASSERT_TRUE(after.has_value()); + EXPECT_EQ(after->etag, before->etag); /// Observing the existing incarnation records materialized evidence without retaining its token. EXPECT_EQ(build->dependencyProof(blob_id), DB::Cas::BlobDependencyProof::Materialized); @@ -699,16 +751,16 @@ TEST(CASS3Staging, PublishOverCondemnedBlobUsesFreshTagNotVerbatim) staging_h, static_cast(store->poolMeta().blob_header_len)); ASSERT_EQ(staging_header.size(), store->poolMeta().blob_header_len); const std::string staging_bytes = staging_header + payload; - backend->putIfAbsent(staging_key, staging_bytes); + createAt(*backend, staging_key, staging_bytes); /// Seed the condemned blob body = EXACTLY what a verbatim promote of this staging object would have /// produced (the writer's OWN create, later observed condemned). This is the adversarial shape: a /// verbatim republication WOULD reproduce these identical bytes ⇒ identical ETag ⇒ collision. - backend->putIfAbsent(blob_key, staging_bytes); + createAt(*backend, blob_key, staging_bytes); DB::Cas::tests::writeMetaClean(*backend, store->layout(), hash, /*size=*/payload.size()); DB::Cas::tests::condemnMeta(*backend, store->layout(), hash, /*condemn_round=*/5); - const DB::Cas::HeadResult before = backend->head(blob_key); - ASSERT_TRUE(before.exists); + const auto before = headAt(*backend, blob_key); + ASSERT_TRUE(before.has_value()); auto build = precommittedBuildFor(store, ns, ref, hash, payload.size()); build->putBlob( @@ -727,11 +779,11 @@ TEST(CASS3Staging, PublishOverCondemnedBlobUsesFreshTagNotVerbatim) EXPECT_EQ(backend->streamingPublicationCount(), 1u); /// The incarnation token is REFRESHED (a fresh incarnation displaced the condemned one). - const DB::Cas::HeadResult after = backend->head(blob_key); - EXPECT_NE(after.token, before.token); - ASSERT_TRUE(after.exists); + const auto after = headAt(*backend, blob_key); + ASSERT_TRUE(after.has_value()); + EXPECT_NE(after->etag, before->etag); - const auto got = backend->get(blob_key); + const auto got = readAt(*backend, blob_key); ASSERT_TRUE(got.has_value()); const uint64_t header_len = store->poolMeta().blob_header_len; @@ -914,12 +966,31 @@ TEST(CASStagingSweeper, RemovesOnlyObjectsUnderGivenMountPrefix) /// nested `staging/` under `blobs/` (or vice versa) would violate. TEST(CASS3Staging, GcBlobDiscoveryPrefixExcludesStagingObjects) { - const DB::Cas::Layout layout("p"); - const std::string blobs_prefix = layout.blobsPrefix(); - const std::string staging_prefix = "p/staging/mountA/"; + /// The REAL staging prefix, from the accessor every writer actually mints staging keys through + /// (`ContentAddressedMetadataStorage::stagingKeyPrefix`) -- not a hand-copied literal that a + /// staging-side rename would leave silently stale. + auto object_storage = makeFakeNativeCopyStorage(/*native_only_copy_supported=*/true); + auto metadata_storage = makeS3StagingMetadataStorageForTest(object_storage, "mountA"); + metadata_storage->startup(); + const std::string physical_root = object_storage->getCommonKeyPrefix(); + const std::string full_staging_prefix = metadata_storage->stagingKeyPrefix(); + ASSERT_TRUE(full_staging_prefix.starts_with(physical_root)) + << full_staging_prefix << " vs root " << physical_root; + /// Strip the physical object-storage root (and the '/' `physicalKey` joins it to the pool key + /// with): `Layout` (below) is root-agnostic, and comparing a physically-rooted key against a bare + /// `Layout` key would pass for the wrong reason (both simply fail to share the unrelated root, not + /// because the pool-relative prefixes are disjoint). + std::string staging_prefix = full_staging_prefix.substr(physical_root.size()); + if (!staging_prefix.empty() && staging_prefix.front() == '/') + staging_prefix.erase(0, 1); + staging_prefix += "/"; const std::string staging_key = staging_prefix + "aaa.tmp"; - EXPECT_EQ(blobs_prefix, "p/blobs/"); + const DB::Cas::Layout layout(metadata_storage->poolForTest()->poolConfig().pool_prefix); + const std::string blobs_prefix = layout.blobsPrefix(); + + EXPECT_EQ(staging_prefix, "pool/staging/mountA/") << "sanity: the accessor's own shape"; + EXPECT_EQ(blobs_prefix, "pool/blobs/"); EXPECT_FALSE(staging_prefix.starts_with(blobs_prefix)); EXPECT_FALSE(blobs_prefix.starts_with(staging_prefix)); EXPECT_FALSE(staging_key.starts_with(blobs_prefix)); @@ -933,7 +1004,7 @@ namespace /// A `LocalObjectStorage` that reports the GCS generation dialect /// (`conditionalOpsUseGenerationTokens() == true`) and a non-`Local` `getType()`, so /// `ContentAddressedMetadataStorage::openPoolView` builds its backend in `Mode::Native` with -/// `native_token_type == TokenType::Generation`. The fake also advertises native copy so generation +/// `native_token_type == Dialect::Generation`. The fake also advertises native copy so generation /// token mode can exercise explicit S3 staging without endpoint/provider heuristics. /// /// Holds every object entirely in memory, keyed by the BARE CAS key exactly as `Backend` hands it to @@ -1035,6 +1106,47 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage return tryGetObjectMetadata(path, with_tags); } + /// This fake advertises every retry profile, and an in-memory store has no retry behaviour to + /// vary, so the profile-aware overloads simply forward. A storage that claimed the capability + /// without implementing them would refuse every control-plane request of a writable mount. + std::optional tryGetObjectMetadataWithNativeToken( + const std::string & path, bool with_tags, const DB::ObjectStorageControlRequest &) const override + { + return tryGetObjectMetadata(path, with_tags); + } + + DB::ObjectStorageIteratorPtr iterate( + const std::string & path_prefix, size_t max_keys, bool with_tags, const std::optional & start_after, + const DB::ObjectStorageControlRequest &) const override + { + return DB::LocalObjectStorage::iterate(path_prefix, max_keys, with_tags, start_after); + } + + DB::ConditionalRemoveResult removeObjectIfTokenMatches( + const DB::StoredObject & object, const std::string & etag, const DB::ObjectStorageControlRequest &) override + { + return removeObjectIfTokenMatches(object, etag); + } + using DB::LocalObjectStorage::removeObjectIfTokenMatches; + + /// A real S3 GET answers with the object's incarnation, which is what the backend reads its + /// bytes AND its generation from in one request. Quoted, the way the SDK's ETag field carries a + /// generation across the HTTP boundary. + DB::SmallObjectDataWithMetadata readSmallObjectAndGetObjectMetadata( + const DB::StoredObject & object, const DB::ReadSettings &, size_t, std::optional) const override + { + std::lock_guard lock(mutex); + auto it = objects.find(object.remote_path); + if (it == objects.end()) + throw DB::S3Exception("FakeGenerationObjectStorage: object does not exist", + Aws::S3::S3Errors::RESOURCE_NOT_FOUND); + DB::SmallObjectDataWithMetadata result; + result.data = it->second.bytes; + result.metadata.size_bytes = it->second.bytes.size(); + result.metadata.etag = "\"" + std::to_string(it->second.generation) + "\""; + return result; + } + void removeObjectIfExists(const DB::StoredObject & object) override { std::lock_guard lock(mutex); @@ -1092,8 +1204,10 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage /// Checks the write-once/exact-token precondition against the current generation and, on success, /// stores `bytes` and mints the next generation. Throws an `S3Exception` naming `PreconditionFailed` /// on a lost condition -- the one signal `finalizeConditionalWrite` classifies as - /// `PutOutcome::PreconditionFailed` rather than an ordinary failure. - void commitConditionalWrite(const std::string & key, const std::string & bytes, + /// `ConditionalWriteOutcome::PreconditionLost` rather than an ordinary failure. + /// Returns the generation it minted, the way a real store returns it in the write response: the + /// backend attributes the write to that generation and nothing reads it back. + uint64_t commitConditionalWrite(const std::string & key, const std::string & bytes, const std::string & if_none_match, const std::string & if_match) { std::lock_guard lock(mutex); @@ -1106,7 +1220,9 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage throw DB::S3Exception("FakeGenerationObjectStorage: if-match precondition failed", Aws::S3::S3Errors::UNKNOWN, "PreconditionFailed"); - objects[key] = Entry{bytes, next_generation++}; + const uint64_t generation = next_generation++; + objects[key] = Entry{bytes, generation}; + return generation; } private: @@ -1132,6 +1248,10 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage void sync() override {} std::string getFileName() const override { return key; } + /// The write response's own incarnation, quoted the way the SDK's ETag field carries a GCS + /// generation across the HTTP boundary -- the backend is what strips that transport syntax. + std::optional getResultObjectETag() const override { return committed_generation; } + protected: void nextImpl() override { @@ -1143,7 +1263,7 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage void finalizeImpl() override { next(); - storage.commitConditionalWrite(key, buffered, if_none_match, if_match); + committed_generation = "\"" + std::to_string(storage.commitConditionalWrite(key, buffered, if_none_match, if_match)) + "\""; } private: @@ -1152,6 +1272,7 @@ class FakeGenerationObjectStorage final : public DB::LocalObjectStorage std::string if_none_match; std::string if_match; std::string buffered; + std::optional committed_generation; }; mutable std::mutex mutex; @@ -1175,6 +1296,27 @@ std::shared_ptr makeFakeGenerationObjectStorageForT } +/// A store whose iterator does not page (the fallback `IObjectStorage::iterate` lists `max_keys` keys +/// once and ends) must still let a page-sized list report that more keys follow. A page that ended +/// exactly at the limit with an empty cursor would read as the end of the prefix, and the startup +/// residual check would then take a prefix of debris plus residue for an empty one. +TEST(CASS3Staging, ListPageOverANonPagingStoreStillReportsMoreKeys) +{ + auto object_storage = makeFakeGenerationObjectStorageForTest(); + auto backend = std::make_shared(object_storage, DB::Cas::ObjectStorageBackend::Mode::Native); + for (int i = 0; i < 40; ++i) + createAt(*backend, fmt::format("p/list/{:03}", i), "x"); + + DB::Cas::tests::OperationForTest op(*backend); + const DB::Cas::ListPage first = (*op).list("p/list/", "", 32, DB::Cas::Retry::once()); + EXPECT_EQ(first.keys.size(), 32u); + ASSERT_FALSE(first.next_cursor.empty()) << "a full page over a non-paging store must still say there is more"; + + const DB::Cas::ListPage rest = (*op).list("p/list/", first.next_cursor, 32, DB::Cas::Retry::once()); + EXPECT_EQ(rest.keys.size(), 8u); + EXPECT_TRUE(rest.next_cursor.empty()); +} + TEST(CASS3Staging, GenerationBackendMayUseNativeOnlyCopy) { auto object_storage = makeFakeGenerationObjectStorageForTest(); diff --git a/src/Disks/tests/gtest_cas_sentinel_probe.cpp b/src/Disks/tests/gtest_cas_sentinel_probe.cpp index 05411f0c3848..cb52bed51399 100644 --- a/src/Disks/tests/gtest_cas_sentinel_probe.cpp +++ b/src/Disks/tests/gtest_cas_sentinel_probe.cpp @@ -3,12 +3,16 @@ #include "config.h" #include +#include #include #include #include +#include #include #include +#include #include +#include #include #include @@ -32,39 +36,47 @@ namespace using DB::Cas::tests::nativeKeyUnder; -/// A Backend decorator whose head/get/list all throw an untyped runtime error when armed — modelling +/// Every test here constructs one backend and probes it once or a few times; a non-owning `BackendPtr` +/// over the test's stack-allocated backend keeps that construction pattern rather than forcing every +/// fixture in this file onto `std::make_shared`. The open fence never trips, matching every prior call +/// here having had no fence to enforce. `clock`, when given, drives `probeSentinel`'s reissue-on- +/// `Indeterminate` loop off an injected clock instead of a real sleep — needed by the one test whose +/// fault never resolves, so the loop runs its whole policy window without taking real wall-clock time. +CasRequests makeRequests(Backend & backend, DB::Cas::tests::FakeClock * clock = nullptr) +{ + BackendPtr ptr(&backend, [](Backend *) {}); + if (clock) + return CasRequests(std::move(ptr), Fence::open(), clock->nowFn(), clock->sleepFn()); + return CasRequests(std::move(ptr), Fence::open()); +} + +/// A Backend decorator whose read/head/list all throw an untyped runtime error when armed — modelling /// a backend with no sharper evidence than "something went wrong" (a network timeout, a 5xx, an -/// unclassifiable failure). Mirrors the existing MetaWriteFaultBackend fault-injection pattern -/// (cas_test_helpers.h): every other operation delegates to InMemoryBackend unchanged. +/// unclassifiable failure). The fault is injected on the PRIMITIVES, which is what +/// `Backend::probeSentinelRaw`'s default derives its answer from; every other operation delegates to +/// InMemoryBackend unchanged. class TransportFaultBackend final : public InMemoryBackend { public: - /// Unhide the base convenience overloads, matching every other Backend subclass in this suite. - using Backend::get; - using Backend::getStream; - using Backend::putIfAbsent; - using Backend::putOverwrite; - using Backend::casPut; - - HeadResult head(const String & key) override + std::optional head(const String & key, TransportAccess & access) override { if (fail.load()) throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (fail.load()) throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - ListPage list(const String & prefix, const String & cursor, size_t limit) override + RawListPage list(const String & prefix, const String & cursor, size_t limit, TransportAccess & access) override { if (fail.load()) throw std::runtime_error("injected fault: transport error"); - return InMemoryBackend::list(prefix, cursor, limit); + return InMemoryBackend::list(prefix, cursor, limit, access); } std::atomic fail{true}; @@ -72,13 +84,39 @@ class TransportFaultBackend final : public InMemoryBackend } +/// The probe loop's own attempt counter reaches the transport too -- propagation only, the probe +/// keeps its ordinary backoff. +TEST(CASSentinelProbe, AttemptNumberPropagates) +{ + struct ProbeRecording : InMemoryBackend + { + std::vector attempts; + SentinelProbeResult probeSentinelRaw(const String & key, TransportAccess & access) override + { + attempts.push_back(access.attemptNo()); + if (attempts.size() == 1) + return {ProbeOutcome::Indeterminate, std::nullopt}; + return InMemoryBackend::probeSentinelRaw(key, access); + } + }; + DB::Cas::tests::FakeClock clock; + auto backend = std::make_shared(); + CasRequests requests(backend, Fence::open(), clock.nowFn(), clock.sleepFn()); + auto op = requests.admit(); + (void)op.probeSentinel("probe", Retry::standard()); + EXPECT_EQ(backend->attempts, (std::vector{1, 2})); + EXPECT_EQ(clock.sleeps.size(), 1u); /// propagation only: the probe keeps its ordinary backoff +} + /// (a) A present key probes Present and carries the materialized body. TEST(CASSentinelProbe, PresentKeyReturnsPresentWithBody) { InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("k", "hello").outcome, PutOutcome::Done); + auto requests = makeRequests(backend); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("k", "hello", Retry::once()))); - const auto result = probeSentinel(backend, "k"); + const auto result = probeSentinel(op, "k", Retry::standard()); EXPECT_EQ(result.outcome, ProbeOutcome::Present); ASSERT_TRUE(result.body.has_value()); EXPECT_EQ(*result.body, "hello"); @@ -88,9 +126,11 @@ TEST(CASSentinelProbe, PresentKeyReturnsPresentWithBody) TEST(CASSentinelProbe, AbsentKeyWithContainerAliveReturnsKeyAbsent) { InMemoryBackend backend; - ASSERT_EQ(backend.putIfAbsent("other", "x").outcome, PutOutcome::Done); // proves the backend is alive + auto requests = makeRequests(backend); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("other", "x", Retry::once()))); // proves the backend is alive - const auto result = probeSentinel(backend, "missing"); + const auto result = probeSentinel(op, "missing", Retry::standard()); EXPECT_EQ(result.outcome, ProbeOutcome::KeyAbsent); EXPECT_FALSE(result.body.has_value()); } @@ -106,15 +146,17 @@ TEST(CASSentinelProbe, ContainerDirectoryRemovedReturnsContainerAbsent) auto storage = tests::makeLocalObjectStorageForTest(); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::EmulatedSingleProcess); - ASSERT_EQ(backend.putIfAbsent("k", "hello").outcome, PutOutcome::Done); + auto requests = makeRequests(backend); + auto op = requests.admit(); + ASSERT_TRUE(std::holds_alternative(op.create("k", "hello", Retry::once()))); /// Sanity, container alive: Present vs. KeyAbsent are genuinely distinct before we remove anything. - EXPECT_EQ(probeSentinel(backend, "k").outcome, ProbeOutcome::Present); - EXPECT_EQ(probeSentinel(backend, "missing").outcome, ProbeOutcome::KeyAbsent); + EXPECT_EQ(probeSentinel(op, "k", Retry::standard()).outcome, ProbeOutcome::Present); + EXPECT_EQ(probeSentinel(op, "missing", Retry::standard()).outcome, ProbeOutcome::KeyAbsent); std::filesystem::remove_all(storage->getCommonKeyPrefix()); - const auto result = probeSentinel(backend, "k"); + const auto result = probeSentinel(op, "k", Retry::standard()); EXPECT_EQ(result.outcome, ProbeOutcome::ContainerAbsent); EXPECT_FALSE(result.body.has_value()); } @@ -129,9 +171,18 @@ TEST(CASSentinelProbe, NativePresentKeyReturnsPresentWithBody) ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); const String key = nativeKeyUnder(storage, "some/key"); - ASSERT_EQ(backend.putIfAbsent(key, "native body").outcome, PutOutcome::Done); + /// Placed through the object storage: a Native write over a local storage has no response + /// incarnation to attribute itself to. Native passes the key verbatim, so this is the object the + /// probe reads. + { + auto out = storage->writeObject(DB::StoredObject(key), DB::WriteMode::Rewrite); + DB::writeString(String("native body"), *out); + out->finalize(); + } - const auto result = probeSentinel(backend, key); + auto requests = makeRequests(backend); + auto op = requests.admit(); + const auto result = probeSentinel(op, key, Retry::standard()); EXPECT_EQ(result.outcome, ProbeOutcome::Present); ASSERT_TRUE(result.body.has_value()); EXPECT_EQ(*result.body, "native body"); @@ -142,9 +193,17 @@ TEST(CASSentinelProbe, NativePresentKeyReturnsPresentWithBody) TEST(CASSentinelProbe, TransportErrorNeverClassifiesAsAbsent) { TransportFaultBackend backend; - const auto result = probeSentinel(backend, "k"); + /// The fault never resolves, so `probeSentinel`'s reissue-on-`Indeterminate` loop runs to its whole + /// policy window before giving up; an injected clock keeps that instantaneous instead of real time. + DB::Cas::tests::FakeClock clock; + auto requests = makeRequests(backend, &clock); + auto op = requests.admit(); + const auto result = probeSentinel(op, "k", Retry::standard()); EXPECT_EQ(result.outcome, ProbeOutcome::Indeterminate); EXPECT_FALSE(result.body.has_value()); + EXPECT_GT(clock.sleeps.size(), 1u) + << "a single attempt would not distinguish this reissue loop from a non-retrying policy that " + "reaches the same Indeterminate give-up on its first try"; } #if USE_AWS_S3 @@ -152,30 +211,44 @@ TEST(CASSentinelProbe, TransportErrorNeverClassifiesAsAbsent) namespace { -/// A `LocalObjectStorage` whose `getObjectMetadata` can be armed to throw a configurable synthetic +/// A `LocalObjectStorage` whose object ACCESS can be armed to throw a configurable synthetic /// `S3Exception` — the same technique `gtest_cas_backend.cpp`'s `NativeReadThrowsNoSuchKeyObjectStorage` /// uses to exercise S3 error codes without a live S3 endpoint. Constructing `ObjectStorageBackend` in /// `Mode::Native` over this fake is the established pattern for testing the Native/S3 raw-error classifier /// in isolation (see also `gtest_cas_backend.cpp`'s `NativeRejectsWrongDialectTokenBeforeTouchingTheWire`). -class ThrowingS3MetadataObjectStorage final : public DB::LocalObjectStorage +/// Both the read and the metadata surface throw: the Native sentinel probe issues one READ, and a +/// store that answers an error for a key answers it however the key is touched. +class ThrowingS3ObjectStorage final : public DB::LocalObjectStorage { public: using DB::LocalObjectStorage::LocalObjectStorage; - void throwOnGetObjectMetadata(Aws::S3::S3Errors code) { metadata_error = code; } + void throwOnObjectAccess(Aws::S3::S3Errors code) { access_error = code; } + + std::unique_ptr readObject( + const DB::StoredObject & object, + const DB::ReadSettings & read_settings, + std::optional read_hint, + bool use_external_buffer, + bool restrict_seek) const override + { + if (access_error) + throw DB::S3Exception("injected fault: " + object.remote_path, *access_error); + return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); + } DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override { - if (metadata_error) - throw DB::S3Exception("injected fault: " + path, *metadata_error); + if (access_error) + throw DB::S3Exception("injected fault: " + path, *access_error); return DB::LocalObjectStorage::getObjectMetadata(path, with_tags); } private: - std::optional metadata_error; + std::optional access_error; }; -DB::ObjectStoragePtr makeThrowingS3MetadataStorageForTest() +DB::ObjectStoragePtr makeThrowingS3StorageForTest() { static std::atomic counter{0}; const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); @@ -186,7 +259,7 @@ DB::ObjectStoragePtr makeThrowingS3MetadataStorageForTest() std::filesystem::create_directories(root, ec); DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); - return std::make_shared(std::move(settings)); + return std::make_shared(std::move(settings)); } } @@ -195,11 +268,13 @@ DB::ObjectStoragePtr makeThrowingS3MetadataStorageForTest() /// error must classify EXACTLY, and anything unmodeled must fail closed to Indeterminate. TEST(CASSentinelProbe, NativeClassifiesNoSuchKeyAsKeyAbsent) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_KEY); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::NO_SUCH_KEY); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::KeyAbsent); + auto requests = makeRequests(backend); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::KeyAbsent); } /// A real S3 HEAD's 404 has no response body, so the SDK cannot parse a `NoSuchKey` `` and @@ -208,38 +283,52 @@ TEST(CASSentinelProbe, NativeClassifiesNoSuchKeyAsKeyAbsent) /// `NO_SUCH_KEY`. Without classifying it, every real-S3 absence would be `Indeterminate` forever. TEST(CASSentinelProbe, NativeClassifiesResourceNotFoundAsKeyAbsent) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::RESOURCE_NOT_FOUND); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::RESOURCE_NOT_FOUND); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::KeyAbsent); + auto requests = makeRequests(backend); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::KeyAbsent); } TEST(CASSentinelProbe, NativeClassifiesNoSuchBucketAsContainerAbsent) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_BUCKET); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::NO_SUCH_BUCKET); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::ContainerAbsent); + auto requests = makeRequests(backend); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::ContainerAbsent); } TEST(CASSentinelProbe, NativeClassifiesAccessDeniedAsAccessDenied) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::ACCESS_DENIED); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::ACCESS_DENIED); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::AccessDenied); + auto requests = makeRequests(backend); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::AccessDenied); } TEST(CASSentinelProbe, NativeClassifiesUnmodeledErrorAsIndeterminate) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::SERVICE_UNAVAILABLE); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::SERVICE_UNAVAILABLE); ObjectStorageBackend backend(storage, ObjectStorageBackend::Mode::Native); - EXPECT_EQ(probeSentinel(backend, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::Indeterminate); + /// Every attempt classifies Indeterminate here too, so the reissue loop runs its whole policy + /// window; an injected clock keeps that instantaneous instead of real time. + DB::Cas::tests::FakeClock clock; + auto requests = makeRequests(backend, &clock); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::Indeterminate); + EXPECT_GT(clock.sleeps.size(), 1u) + << "a single attempt would not distinguish this reissue loop from a non-retrying policy that " + "reaches the same Indeterminate give-up on its first try"; } /// Production wiring (`Pool::open`) ALWAYS wraps the real backend in `InstrumentedBackend` before @@ -252,12 +341,14 @@ TEST(CASSentinelProbe, NativeClassifiesUnmodeledErrorAsIndeterminate) /// reached through the wrapper. TEST(CASSentinelProbe, InstrumentedBackendForwardsToInnerClassification) { - auto storage = std::static_pointer_cast(makeThrowingS3MetadataStorageForTest()); - storage->throwOnGetObjectMetadata(Aws::S3::S3Errors::NO_SUCH_BUCKET); + auto storage = std::static_pointer_cast(makeThrowingS3StorageForTest()); + storage->throwOnObjectAccess(Aws::S3::S3Errors::NO_SUCH_BUCKET); auto inner = std::make_shared(storage, ObjectStorageBackend::Mode::Native); InstrumentedBackend instrumented(inner); - EXPECT_EQ(probeSentinel(instrumented, nativeKeyUnder(storage, "some/key")).outcome, ProbeOutcome::ContainerAbsent); + auto requests = makeRequests(instrumented); + auto op = requests.admit(); + EXPECT_EQ(probeSentinel(op, nativeKeyUnder(storage, "some/key"), Retry::standard()).outcome, ProbeOutcome::ContainerAbsent); } #endif diff --git a/src/Disks/tests/gtest_cas_server_root_format.cpp b/src/Disks/tests/gtest_cas_server_root_format.cpp index c2d1474dc29b..bf04753c9b9f 100644 --- a/src/Disks/tests/gtest_cas_server_root_format.cpp +++ b/src/Disks/tests/gtest_cas_server_root_format.cpp @@ -11,12 +11,14 @@ namespace DB::ErrorCodes extern const int CORRUPTED_DATA; } +CAS_BATTERY_COVERS(Owner); + TEST(CASFormatBattery, Owner) { OwnerObject o; o.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); const String golden = currentFormatHeader("cas_owner") + - "{\"su\":\"0123456789abcdeffedcba9876543210\"}\n"; + "{\"server_uuid\":\"0123456789abcdeffedcba9876543210\"}\n"; EXPECT_EQ(encodeOwner(o), golden); EXPECT_FALSE(decodeOwner(golden).retired_at_ms.has_value()); runFormatBattery({FormatId::Owner, @@ -31,11 +33,15 @@ TEST(CASOwnerFormat, RetiredAtRoundTrip) o.server_uuid = hexToU128("0123456789abcdeffedcba9876543210"); o.retired_at_ms = 1752537600000ULL; + EXPECT_EQ(encodeOwner(o), currentFormatHeader("cas_owner") + + "{\"server_uuid\":\"0123456789abcdeffedcba9876543210\",\"retired_at_ms\":1752537600000}\n"); const OwnerObject back = decodeOwner(encodeOwner(o)); EXPECT_EQ(back.server_uuid, o.server_uuid); EXPECT_EQ(back.retired_at_ms, o.retired_at_ms); } +CAS_BATTERY_COVERS(ServerEpoch); + TEST(CASFormatBattery, ServerEpoch) { ServerEpoch e; @@ -43,9 +49,11 @@ TEST(CASFormatBattery, ServerEpoch) runFormatBattery({FormatId::ServerEpoch, [&] { return sealObject(FormatId::ServerEpoch, encodeServerEpoch(e)); }, [](std::string_view s) { decodeServerEpoch(std::string(openObject(FormatId::ServerEpoch, s))); }, - currentFormatHeader("cas_epoch") + "{\"nwe\":\"7\"}\n"}); + currentFormatHeader("cas_epoch") + "{\"next_writer_epoch\":\"7\"}\n"}); } +CAS_BATTERY_COVERS(MountLease); + TEST(CASFormatBattery, MountLease) { MountLease m{hexToU128("0123456789abcdeffedcba9876543210"), 7, "host-1", 4242, @@ -55,8 +63,8 @@ TEST(CASFormatBattery, MountLease) [&] { return sealObject(FormatId::MountLease, encodeMountLease(m)); }, [](std::string_view s) { decodeMountLease(std::string(openObject(FormatId::MountLease, s))); }, currentFormatHeader("cas_mount_lease") + - "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"host-1\",\"pid\":4242," - "\"sat\":1752537600000,\"seq\":\"5\",\"eat\":1752537630000,\"ma\":\"9\",\"fen\":false," + "{\"server_uuid\":\"0123456789abcdeffedcba9876543210\",\"writer_epoch\":\"7\",\"hostname\":\"host-1\",\"pid\":4242," + "\"started_at_ms\":1752537600000,\"seq\":\"5\",\"expires_at_ms\":1752537630000,\"min_active_build_sequence\":\"9\",\"gc_fenced\":false," "\"write_attempt_id\":\"00112233445566778899aabbccddeeff\"}\n"}); } @@ -66,7 +74,7 @@ TEST(CASMountLeaseFormat, FarewellSentinelAndFencedSurvive) 1, 5, 2, std::numeric_limits::max(), true, hexToU128("00112233445566778899aabbccddeeff")}; const MountLease back = decodeMountLease(encodeMountLease(m)); - EXPECT_EQ(back.min_active, std::numeric_limits::max()); + EXPECT_EQ(back.min_active_build_sequence, std::numeric_limits::max()); EXPECT_TRUE(back.gc_fenced); EXPECT_EQ(back.hostname, "h"); EXPECT_EQ(back.writer_epoch, 7u); @@ -85,8 +93,8 @@ TEST(CASMountLeaseFormat, WriteAttemptIdIsRequiredAndCanonical) EXPECT_EQ(decodeMountLease(encoded).write_attempt_id, m.write_attempt_id); const String without_attempt_id = currentFormatHeader("cas_mount_lease") + - "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"\",\"pid\":0," - "\"sat\":0,\"seq\":\"0\",\"eat\":0,\"ma\":\"0\",\"fen\":false}\n"; + "{\"server_uuid\":\"0123456789abcdeffedcba9876543210\",\"writer_epoch\":\"7\",\"hostname\":\"\",\"pid\":0," + "\"started_at_ms\":0,\"seq\":\"0\",\"expires_at_ms\":0,\"min_active_build_sequence\":\"0\",\"gc_fenced\":false}\n"; try { decodeMountLease(without_attempt_id); @@ -101,8 +109,8 @@ TEST(CASMountLeaseFormat, WriteAttemptIdIsRequiredAndCanonical) TEST(CASMountLeaseFormat, ZeroWriteAttemptIdIsRejected) { const String data = currentFormatHeader("cas_mount_lease") + - "{\"su\":\"0123456789abcdeffedcba9876543210\",\"we\":\"7\",\"hn\":\"\",\"pid\":0," - "\"sat\":0,\"seq\":\"0\",\"eat\":0,\"ma\":\"0\",\"fen\":false," + "{\"server_uuid\":\"0123456789abcdeffedcba9876543210\",\"writer_epoch\":\"7\",\"hostname\":\"\",\"pid\":0," + "\"started_at_ms\":0,\"seq\":\"0\",\"expires_at_ms\":0,\"min_active_build_sequence\":\"0\",\"gc_fenced\":false," "\"write_attempt_id\":\"00000000000000000000000000000000\"}\n"; try { @@ -130,11 +138,18 @@ TEST(CASMountLeaseFormat, UnknownFieldsRemainTolerated) TEST(CASMountLeaseFormat, RejectsMissingIdentityFields) { - const String header = "{\"type\":\"cas_mount_lease\",\"v\":3}\n"; - const String fields = "\"hn\":\"host-1\",\"pid\":4242,\"sat\":1752537600000," - "\"seq\":\"5\",\"eat\":1752537630000,\"ma\":\"9\",\"fen\":false}"; - - const auto expectCorrupted = [](const String & data) + /// Each arm drops exactly ONE identity and keeps the other two, and each asserts the message that + /// names the dropped one. A body missing two of them would satisfy whichever clause runs first, so + /// a shared fixture and a shared message together would let two of the three checks be deleted + /// with this test still green. + const String header = "{\"type\":\"cas_mount_lease\",\"v\":1}\n"; + const String uuid = R"("server_uuid":"0123456789abcdeffedcba9876543210",)"; + const String epoch = R"("writer_epoch":"7",)"; + const String attempt = R"("write_attempt_id":"00112233445566778899aabbccddeeff",)"; + const String rest = "\"hostname\":\"host-1\",\"pid\":4242,\"started_at_ms\":1752537600000," + "\"seq\":\"5\",\"expires_at_ms\":1752537630000,\"min_active_build_sequence\":\"9\",\"gc_fenced\":false}"; + + const auto expectMessage = [](const String & data, std::string_view expected) { try { @@ -144,9 +159,11 @@ TEST(CASMountLeaseFormat, RejectsMissingIdentityFields) catch (const DB::Exception & e) { EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(e.message(), expected); } }; - expectCorrupted(header + R"({"we":"7",)" + fields + "\n"); - expectCorrupted(header + R"({"su":"0123456789abcdeffedcba9876543210",)" + fields + "\n"); + expectMessage(header + "{" + epoch + attempt + rest + "\n", "CAS mount-lease: missing server_uuid"); + expectMessage(header + "{" + uuid + attempt + rest + "\n", "CAS mount-lease: missing writer_epoch"); + expectMessage(header + "{" + uuid + epoch + rest + "\n", "CAS mount-lease: missing or zero write_attempt_id"); } diff --git a/src/Disks/tests/gtest_cas_settings.cpp b/src/Disks/tests/gtest_cas_settings.cpp index 46d18480e55b..53b3b50b4fbb 100644 --- a/src/Disks/tests/gtest_cas_settings.cpp +++ b/src/Disks/tests/gtest_cas_settings.cpp @@ -25,7 +25,9 @@ namespace DB::ContentAddressedSetting extern const ContentAddressedSettingsBool gc_enabled; extern const ContentAddressedSettingsUInt64 gc_shards; extern const ContentAddressedSettingsUInt64 gc_interval_sec; + extern const ContentAddressedSettingsUInt64 gc_bulk_delete_chunk_keys; extern const ContentAddressedSettingsString scratch_path; + extern const ContentAddressedSettingsBool unsafe_remount_no_delay; } namespace @@ -130,6 +132,29 @@ TEST(CASContentAddressedSettings, RemovedCacheSettingsAreRejected) } } +/// `cas_part_folder_validate` paced a manifest `HEAD` that no longer exists. A config still asking +/// for it must fail the disk open, not be quietly accepted and ignored. +TEST(CASContentAddressedSettings, RetiredPartFolderValidateIsRejected) +{ + for (const std::string & value : {"always", "never", "age 5"}) + { + SCOPED_TRACE(value); + auto cfg = makeConfig( + "srv1" + "" + value + ""); + ContentAddressedSettings settings; + try + { + settings.loadFromConfig(*cfg, "disk", "/data", "/data/scratch", identity_macros); + FAIL() << "expected the retired setting cas_part_folder_validate to be rejected as unknown"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::UNKNOWN_SETTING); + } + } +} + TEST(CASContentAddressedSettings, UnknownKeyRejected) { expectLoadFailureWithExactMessage( @@ -151,7 +176,37 @@ TEST(CASContentAddressedSettings, InvalidBoundsDiagnosticNamesExternalConfigKeys expectLoadFailureWithExactMessage( "srv10", ErrorCodes::BAD_ARGUMENTS, - "content_addressed disk: cas_gc_interval_sec and cas_gc_shards must be >= 1 (got 60, 0)"); + "content_addressed disk: cas_gc_interval_sec, cas_gc_shards and cas_gc_read_concurrency must be >= 1 " + "(got 60, 0, 16)"); + /// The fold's read-ahead pool is refused at zero for the same reason the shard count is: a zero + /// would be a silently disabled subsystem rather than a configuration the pool can honour. One is + /// the sequential fold and is the way to turn the read-ahead off. + expectLoadFailureWithExactMessage( + "srv10", + ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: cas_gc_interval_sec, cas_gc_shards and cas_gc_read_concurrency must be >= 1 " + "(got 60, 1, 0)"); +} + +TEST(CASSettings, BulkDeleteChunkKeysBoundsAreEnforced) +{ + expectLoadFailureWithExactMessage( + "srv1" + "1001", + ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: gc_bulk_delete_chunk_keys must be between 1 and 1000 (got 1001)"); + expectLoadFailureWithExactMessage( + "srv1" + "0", + ErrorCodes::BAD_ARGUMENTS, + "content_addressed disk: gc_bulk_delete_chunk_keys must be between 1 and 1000 (got 0)"); + + auto cfg = makeConfig( + "srv1" + "1"); + ContentAddressedSettings s; + EXPECT_NO_THROW(s.loadFromConfig(*cfg, "disk", "/scratch", "/scratch", identity_macros)); + EXPECT_EQ(s[ContentAddressedSetting::gc_bulk_delete_chunk_keys].value, 1u); } TEST(CASContentAddressedSettings, InvalidEnumDiagnosticsNameExternalConfigKeys) @@ -164,10 +219,6 @@ TEST(CASContentAddressedSettings, InvalidEnumDiagnosticsNameExternalConfigKeys) "srv1remote", ErrorCodes::BAD_ARGUMENTS, "Unknown cas_staging_backend value 'remote' (expected 'local' or 's3')"); - expectLoadFailureWithExactMessage( - "srv1sometimes", - ErrorCodes::BAD_ARGUMENTS, - "Unknown cas_part_folder_validate value 'sometimes' (expected 'always', 'never', or 'age ')"); } /// The point of this test is that none of these names appears anywhere in CAS code. It is not an @@ -527,3 +578,19 @@ TEST(CASContentAddressedSettings, AbsentScratchPathUsesDefaultVerbatim) s.loadFromConfig(*cfg, "disk", "/data", "/data/disks/x/cas_scratch", identity_macros); EXPECT_EQ(s[ContentAddressedSetting::scratch_path].value, "/data/disks/x/cas_scratch"); } + +TEST(CASContentAddressedSettings, UnsafeRemountNoDelayIsOffByDefault) +{ + { + auto cfg = makeConfig("srv1"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/data", "/data/scratch", identity_macros); + EXPECT_FALSE(s[ContentAddressedSetting::unsafe_remount_no_delay].value); + } + { + auto cfg = makeConfig("srv11"); + ContentAddressedSettings s; + s.loadFromConfig(*cfg, "disk", "/data", "/data/scratch", identity_macros); + EXPECT_TRUE(s[ContentAddressedSetting::unsafe_remount_no_delay].value); + } +} diff --git a/src/Disks/tests/gtest_cas_shutdown_context.cpp b/src/Disks/tests/gtest_cas_shutdown_context.cpp index 6ceb58b1188b..b1fd2ba48d6f 100644 --- a/src/Disks/tests/gtest_cas_shutdown_context.cpp +++ b/src/Disks/tests/gtest_cas_shutdown_context.cpp @@ -88,9 +88,10 @@ void emitTestEvent(DB::ContentAddressedMetadataStorage & storage) /// A failed ref-lane drain must not leave a clean-release marker behind. That marker lets a /// successor skip the observation window, so a phase-2 failure must leave it absent. - const auto mount = backend->get(Layout(config.pool_prefix).mountKey(config.server_root_id)); + DB::Cas::tests::OperationForTest op(backend); + const auto mount = (*op).read(Layout(config.pool_prefix).mountKey(config.server_root_id), Retry::standard()); const bool clean_release = mount - && decodeMountLease(mount->bytes).min_active == std::numeric_limits::max(); + && decodeMountLease(mount->bytes).min_active_build_sequence == std::numeric_limits::max(); const bool marker_must_be_absent = phase == 2; std::_Exit(marker_must_be_absent && clean_release ? 1 : 0); } diff --git a/src/Disks/tests/gtest_cas_slot_occupy.cpp b/src/Disks/tests/gtest_cas_slot_occupy.cpp index d248509c2d31..1613d350520e 100644 --- a/src/Disks/tests/gtest_cas_slot_occupy.cpp +++ b/src/Disks/tests/gtest_cas_slot_occupy.cpp @@ -2,16 +2,16 @@ #include "config.h" -#include #include +#include #include #include #include +#include + using namespace DB::Cas; using DB::Cas::tests::CountingBackend; -using DB::Cas::tests::ChunkFaultBackend; -using DB::Cas::tests::LandedButAckLostOnceBackend; namespace DB::ErrorCodes { @@ -19,244 +19,254 @@ namespace DB::ErrorCodes } /// ================================================================================================ -/// Task 2 (2026-07-28 CAS ref-chain Stage A streams, spec INV-2): CasRequestController::slotOccupy -- -/// the dedicated RAW slot-occupy primitive every seal writer and wedge retry uses. ONE conditional -/// create; on conflict, ONE raw exact GET of the occupant -- NEVER retries internally, NEVER lists, -/// and NEVER composes putIfAbsentControlled (which retries the same (key, bytes) internally) or -/// resolveByExactGet (which compares against an expected body and throws CORRUPTED_DATA on a -/// mismatch) [codex finding 3]. Adjudicating whether an Occupied occupant is "mine" is entirely the -/// CALLER's job (Task 4/6, the CaCasMountCore `mine` contract) -- these tests only pin the -/// primitive's own three-way outcome and its op-count contract (Created=1, Occupied=2, -/// Unresolved<=2 backend ops). +/// The ref lane's slot occupy: ONE conditional create of a write-once ref-log key, `Retry::once()`, +/// on an operation the caller resumed under the generation its transaction was admitted at. It is +/// what every epoch-seal writer and every wedge retry issues, so these tests pin the shape those two +/// callers depend on -- the four alternatives and the request count behind each -- rather than the +/// engine's general write contract, which `gtest_cas_requests.cpp` owns. +/// +/// Adjudicating whether a conflicting occupant is "mine" is entirely the CALLER's job (the +/// `CaCasMountCore` `mine` contract: byte equality, never a shape or generation match); nothing here +/// compares bytes for meaning. /// ================================================================================================ namespace { /// Deletes the key the INSTANT its own conditional create conflicts, modelling "the occupant that -/// caused the conflict vanished before slotOccupy's single resolve GET" -- a race a real backend can -/// produce (e.g. GC reclaiming an already-condemned object) that the primitive must survive by -/// reporting Unresolved, NEVER a fabricated Created. +/// caused the conflict vanished before the settling read" -- a race a real backend can produce (e.g. +/// GC reclaiming an already-condemned object) that the call must survive by reporting what it saw, +/// never a fabricated commit. class VanishOnConflictBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; - - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { - PutResult result = CountingBackend::putIfAbsent(key, bytes, meta); - if (result.outcome == PutOutcome::PreconditionFailed) + auto result = CountingBackend::write(key, bytes, expected_value, access); + if (!result.has_value()) { - const HeadResult h = head(key); - if (h.exists) - deleteExact(key, h.token); + if (const auto meta = CountingBackend::head(key, access)) + CountingBackend::remove(key, meta->value, access); } return result; } }; -/// Throws a deterministic LOCAL failure (BAD_ARGUMENTS, in isDeterministicLocalFailure's set) on the -/// first putIfAbsent -- models a backend-level programming bug, distinct from ChunkFaultBackend's -/// Mode::Definite below, which is a whitelisted SYNCHRONOUS REJECTION -/// (classifyConditionalWriteResult's DefiniteFailure). slotOccupy must rethrow both, unchanged, never -/// folding either into Unresolved (SlotOccupyResult::Kind has no DefiniteFailure member to carry it). +/// Throws a deterministic LOCAL failure (`BAD_ARGUMENTS`, in `isDeterministicLocalFailure`'s set) on +/// the first write -- a backend-level programming bug, distinct from a whitelisted synchronous +/// rejection, which the store gives as an answer and the engine reports as `Refused`. class LocalFailureOnceBackend : public CountingBackend { public: - using CountingBackend::putIfAbsent; bool fail_once = true; - PutResult putIfAbsent(const String & key, const String & bytes, const ObjectMeta & meta) override + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override { if (fail_once) { fail_once = false; throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "scripted deterministic local failure"); } - return CountingBackend::putIfAbsent(key, bytes, meta); + return CountingBackend::write(key, bytes, expected_value, access); } }; -/// (`LandedButAckLostOnceBackend` -- "the write LANDS, then the ack is lost" -- was lifted into -/// `cas_test_helpers.h` for Task 4, whose wedge-adoption tests need the identical seam through a whole -/// Pool. Its `key_substr` defaults to empty, which is exactly this file's original behaviour: fault the -/// first `putIfAbsent` of any key.) -/// Delegates the FIRST putIfAbsent for a key to CountingBackend -- so the write actually LANDS -- and -/// only THEN throws an ambiguous exception, modelling "our own PUT committed but its response was lost" -/// (the Task-4 adoption input: plan's "Occupied + bytes == wedge.bytes -> an earlier attempt landed -> -/// adopt"). Distinct from InMemoryBackend::injectAmbiguousPutIfAbsent, which never touches the store at -/// all -- that hook models an attempt that did NOT land; this one models an attempt that DID. -/// One-shot per backend instance: review finding I2 asked specifically for a ~10-line local backend rather than -/// reusing ChunkFaultBackend::Mode::LandedThenLost, which also arms a one-shot lost-GET fault that would -/// obscure whether slotOccupy's OWN immediate resolve (not just a later caller's retry) is correct too. +/// Withdraws the caller's liveness the instant a conditional create conflicts, so the settling read +/// is the first request the operation is no longer admitted for. +class WithdrawAdmissionOnConflictBackend : public CountingBackend +{ +public: + bool live = true; + + std::expected write( + const String & key, const String & bytes, const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + auto result = CountingBackend::write(key, bytes, expected_value, access); + if (!result.has_value()) + live = false; + return result; + } +}; +/// The occupant a conflict names, or null when the result is not a conflict that observed one. +const Object * conflictObject(const WriteResult & result) +{ + const auto * conflict = std::get_if(&result); + return conflict ? std::get_if(&conflict->seen) : nullptr; } -/// ---- Step 1 required scenarios ---- +} -TEST(CASSlotOccupy, AbsentKeyCreatesWithOneOp) +TEST(CASSlotOccupy, AbsentKeyCommitsWithOneRequest) { auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); - const auto result = controller.slotOccupy("k", "payload", [] { return true; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Created); - EXPECT_TRUE(result.occupant_bytes.empty()); - EXPECT_TRUE(result.occupant_token.empty()) << "occupant_token is Occupied-only; must stay default on Created"; - EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NotUnresolved); + const WriteResult result = op.create("k", "payload", Retry::once()); + const auto * committed = std::get_if(&result); + ASSERT_TRUE(committed != nullptr); + EXPECT_EQ(committed->attempts_sent, 1u); + EXPECT_FALSE(committed->resolved_by_read) << "an unambiguous create is proven by its own response"; - EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 0u); EXPECT_EQ(backend->headCount("k"), 0u); - const auto landed = backend->get("k"); + CasOperation reader = requests.admit(); + const auto landed = reader.read("k", Retry::once()); ASSERT_TRUE(landed.has_value()); EXPECT_EQ(landed->bytes, "payload"); } -TEST(CASSlotOccupy, PreExistingKeyOccupiedWithExactBytesAndTokenTwoOps) +TEST(CASSlotOccupy, PreExistingKeyConflictsWithExactBytesAndIncarnationInTwoRequests) { auto backend = std::make_shared(); - const PutResult seeded = backend->putIfAbsent("k", "occupant-bytes"); - ASSERT_EQ(seeded.outcome, PutOutcome::Done); + CasRequests requests(backend, Fence::open()); + + CasOperation seeder = requests.admit(); + const WriteResult seeded = seeder.create("k", "occupant-bytes", Retry::once()); + const auto * seeded_committed = std::get_if(&seeded); + ASSERT_TRUE(seeded_committed != nullptr); + const Etag seeded_incarnation = seeded_committed->etag; backend->resetCounts(); - CasRequestController controller(backend, CasRequestBudget{}); - const auto result = controller.slotOccupy("k", "my-attempt-bytes", [] { return true; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Occupied); - EXPECT_EQ(result.occupant_bytes, "occupant-bytes"); - EXPECT_EQ(result.occupant_token, seeded.token); + CasOperation op = requests.admit(); + const WriteResult result = op.create("k", "my-attempt-bytes", Retry::once()); + const Object * occupant = conflictObject(result); + ASSERT_TRUE(occupant != nullptr) << "the settling read must have named the occupant"; + EXPECT_EQ(occupant->bytes, "occupant-bytes"); + EXPECT_EQ(occupant->etag, seeded_incarnation); - EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 1u); - EXPECT_EQ(backend->headCount("k"), 0u) << "exactly PUT+GET -- a HEAD-then-GET implementation must fail this"; + EXPECT_EQ(backend->headCount("k"), 0u) + << "exactly one write and one settling read -- a HEAD-then-read implementation must fail this"; - /// A conflict never overwrites or appends -- the pre-existing object is untouched. - const auto current = backend->get("k"); + CasOperation reader = requests.admit(); + const auto current = reader.read("k", Retry::once()); ASSERT_TRUE(current.has_value()); - EXPECT_EQ(current->bytes, "occupant-bytes"); + EXPECT_EQ(current->bytes, "occupant-bytes") << "a conflict never overwrites or appends"; } -TEST(CASSlotOccupy, InjectedAmbiguousPutResolvesUnresolvedWhenGetFindsNothing) +TEST(CASSlotOccupy, AmbiguousWriteThatLandedNothingGivesUpHavingSentOne) { auto backend = std::make_shared(); - backend->injectAmbiguousPutIfAbsent("k"); - - CasRequestController controller(backend, CasRequestBudget{}); - const auto result = controller.slotOccupy("k", "payload", [] { return true; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); - /// An attempt WAS sent (the ambiguous PUT itself) -- this is never the pre-attempt NoAttemptSent - /// case. Of the existing CasUnresolvedReason values, AttemptsExhausted is the one documented as - /// "the genuine case the 'retry budget exhausted' wording describes" -- exactly this call's single - /// (and only) attempt having nothing left to give once its resolve GET came up empty. - EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::AttemptsExhausted); - EXPECT_FALSE(unresolvedProvesNothingWasSent(result.unresolved_reason)); - - EXPECT_EQ(backend->putCount("k"), 1u); + backend->injectAmbiguousWrite("k"); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + + const WriteResult result = op.create("k", "payload", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Unresolved); + EXPECT_TRUE(gave_up->sent_any) << "the ambiguous attempt itself was sent -- this is never the pre-attempt case"; + EXPECT_TRUE(std::holds_alternative(gave_up->last_seen)); + + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 1u); - EXPECT_FALSE(backend->head("k").exists) << "the injected fault must not actually create anything"; + CasOperation reader = requests.admit(); + EXPECT_FALSE(reader.head("k", Retry::once()).has_value()) << "the injected fault must not create anything"; } -TEST(CASSlotOccupy, ConflictThenVanishResolvesUnresolved) +TEST(CASSlotOccupy, ConflictThenVanishGivesUpRatherThanFabricatingACommit) { auto backend = std::make_shared(); - const auto seeded = backend->putIfAbsent("k", "occupant-bytes"); - ASSERT_EQ(seeded.outcome, PutOutcome::Done); + CasRequests requests(backend, Fence::open()); + + CasOperation seeder = requests.admit(); + ASSERT_TRUE(std::holds_alternative(seeder.create("k", "occupant-bytes", Retry::once()))); backend->resetCounts(); - CasRequestController controller(backend, CasRequestBudget{}); - const auto result = controller.slotOccupy("k", "my-attempt-bytes", [] { return true; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); - EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::AttemptsExhausted); + CasOperation op = requests.admit(); + const WriteResult result = op.create("k", "my-attempt-bytes", Retry::once()); + /// Nothing of ours was ever ambiguous, so an absence settles the call as a conflict against an + /// occupant that is no longer there -- never as a commit. + const auto * conflict = std::get_if(&result); + ASSERT_TRUE(conflict != nullptr); + EXPECT_TRUE(std::holds_alternative(conflict->seen)); - EXPECT_EQ(backend->putCount("k"), 1u); + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 1u); - /// No headCount assertion here (unlike the sibling Occupied test above): VanishOnConflictBackend's - /// OWN fixture issues a HEAD internally (to fetch the token before deleteExact) -- that HEAD belongs - /// to the test's vanish mechanism, not to slotOccupy, so asserting headCount==0 would be wrong, not - /// stronger. slotOccupy itself never calls head(); only put+get are its own ops. - EXPECT_FALSE(backend->head("k").exists) << "the occupant vanished between the conflict and the resolve GET"; } -TEST(CASSlotOccupy, FenceFlipMidCallRefusesPreAttemptNeverLiesCreated) +TEST(CASSlotOccupy, LivenessRefusalBeforeTheAttemptSendsNothing) { auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([] { return false; }); - const auto result = controller.slotOccupy("k", "payload", [] { return false; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); - /// The pre-attempt reason: fence_ok refused before anything was sent to the backend. - EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NoAttemptSent); - EXPECT_TRUE(unresolvedProvesNothingWasSent(result.unresolved_reason)); + const WriteResult result = op.create("k", "payload", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_FALSE(gave_up->sent_any) << "the whole point: the key is provably unwritten"; - EXPECT_EQ(backend->putTotal(), 0u); + EXPECT_EQ(backend->writeTotal(), 0u); EXPECT_EQ(backend->getTotal(), 0u); - EXPECT_FALSE(backend->head("k").exists) << "never a lie of Created -- the key must be untouched"; + CasRequests open_requests(backend, Fence::open()); + CasOperation reader = open_requests.admit(); + EXPECT_FALSE(reader.head("k", Retry::once()).has_value()) << "never a lie of committed -- the key must be untouched"; } -/// ---- Bonus coverage: the deadline pre-gate (the OTHER half of "fence/deadline-gated"), and the two -/// rethrow paths this primitive shares with its sibling controlled ops. ---- - -/// The deadline gate is the SAME pre-attempt refusal as the fence gate above -- a fake clock proves it -/// fires from elapsed time alone, with a fence that always says yes. -TEST(CASSlotOccupy, OperationDeadlineExhaustedRefusesPreAttempt) +/// The deadline is the OTHER pre-attempt refusal: a fake clock proves it fires from elapsed time +/// alone, under a fence that always says yes. +TEST(CASSlotOccupy, ExhaustedPolicyDeadlineRefusesBeforeTheAttempt) { auto backend = std::make_shared(); uint64_t clock = 0; - auto now_ms = [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1000; return t; }; - - CasRequestBudget budget; - budget.attempt_timeout_ms = 50; - budget.operation_deadline_ms = 500; /// entry now_ms()==0 -> deadline_ms=500; the gate's OWN - /// now_ms() call then returns 1000 -> 1000+50 > 500 -> refuse - CasRequestController controller(backend, budget, now_ms); - - const auto result = controller.slotOccupy("k", "payload", [] { return true; }); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); - EXPECT_EQ(result.unresolved_reason, CasUnresolvedReason::NoAttemptSent); - EXPECT_EQ(backend->putTotal(), 0u); - EXPECT_EQ(backend->getTotal(), 0u) << "zero ops total -- the deadline gate must refuse before any I/O, same as the fence gate"; + CasRequests requests(backend, Fence::open(), + [&clock]() -> uint64_t { const uint64_t t = clock; clock += 1000; return t; }); + requests.setAttemptReservationForTest(50); + CasOperation op = requests.admit(); + + /// Entry `now_ms()` is 0, so the bound is 500; the loop's own `now_ms()` then reads 1000. + const WriteResult result = op.create("k", "payload", Retry::within(500)); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_FALSE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 0u); + EXPECT_EQ(backend->getTotal(), 0u) + << "zero requests -- the deadline refuses before any I/O, exactly as the liveness gate does"; } -/// A whitelisted synchronous rejection (classifyConditionalWriteResult's DefiniteFailure) PROVES the -/// request was never applied -- slotOccupy must surface it unchanged rather than resolving or folding -/// it into Unresolved. Guarded to USE_AWS_S3 builds ONLY [review M6]: DefiniteFailure classification is -/// structurally unreachable without it (classifyConditionalWriteResult's whitelist is entirely inside -/// its own `#if USE_AWS_S3`), so on a no-S3 build ChunkFaultBackend::Mode::Definite instead throws a -/// plain CORRUPTED_DATA DB::Exception -- which is in isDeterministicLocalFailure's set, meaning this -/// test would silently exercise the SAME slotOccupy branch as DeterministicLocalFailurePropagatesWithoutResolve -/// below rather than the DefiniteFailure branch it claims to cover. Better a visibly-absent test on that -/// config than a passing one that isn't testing what its name says. +/// A whitelisted synchronous rejection PROVES the request was never applied, so the engine reports it +/// as a value and settles nothing by reading. Guarded to `USE_AWS_S3` builds ONLY: the classification +/// lives entirely inside `isDefinitelyRefusedWrite`'s own `#if USE_AWS_S3`, so without it this would +/// silently exercise the ambiguity path instead of the refusal it names. #if USE_AWS_S3 -TEST(CASSlotOccupy, DefiniteFailurePropagatesWithoutResolve) +TEST(CASSlotOccupy, DefiniteStoreRefusalIsAValueAndSettlesNothing) { - auto backend = std::make_shared(); - backend->fault_substr = "k"; - backend->mode = ChunkFaultBackend::Mode::Definite; - backend->fault_count = 1; - - CasRequestController controller(backend, CasRequestBudget{}); - EXPECT_THROW(controller.slotOccupy("k", "payload", [] { return true; }), DB::Exception); - /// ChunkFaultBackend's fault check throws BEFORE delegating to CountingBackend::putIfAbsent, so - /// putCount stays 0 on this path -- fault_count reaching 0 is this backend's own proof the (one) - /// attempt was made and consumed the fault. - EXPECT_EQ(backend->fault_count, 0); - EXPECT_EQ(backend->getCount("k"), 0u) << "a whitelisted definite rejection must never trigger a resolve GET"; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", std::make_exception_ptr( + DB::S3Exception("simulated malformed request", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"))); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); + + const WriteResult result = op.create("k", "payload", Retry::once()); + EXPECT_TRUE(std::holds_alternative(result)); + EXPECT_EQ(backend->getCount("k"), 0u) << "a proven refusal must never trigger a settling read"; } #endif -/// A deterministic LOCAL failure (isDeterministicLocalFailure's set) is the OTHER rethrow path -- -/// distinct from DefiniteFailure above, and checked first in the implementation, so it needs its own -/// backend-level fault to prove both branches are wired, not just one masking the other. -TEST(CASSlotOccupy, DeterministicLocalFailurePropagatesWithoutResolve) +/// A deterministic LOCAL failure is the one thing the write surface still reports by exception: +/// reissuing only replays it, and folding it into an outcome would bury the root cause. +TEST(CASSlotOccupy, DeterministicLocalFailurePropagatesWithoutSettling) { auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit(); bool threw = false; try { - controller.slotOccupy("k", "payload", [] { return true; }); + op.create("k", "payload", Retry::once()); } catch (const DB::Exception & e) { @@ -264,95 +274,83 @@ TEST(CASSlotOccupy, DeterministicLocalFailurePropagatesWithoutResolve) EXPECT_EQ(e.code(), DB::ErrorCodes::BAD_ARGUMENTS) << "the ORIGINAL exception must propagate unchanged"; } EXPECT_TRUE(threw) << "a deterministic local failure must propagate, never return an outcome"; - /// LocalFailureOnceBackend throws BEFORE delegating to CountingBackend::putIfAbsent (same shape as - /// ChunkFaultBackend above), so putCount stays 0 here too -- fail_once flipping is this backend's - /// own proof the attempt was made. EXPECT_FALSE(backend->fail_once); EXPECT_EQ(backend->getCount("k"), 0u); } -/// ---- Fix round 1 (review findings I1, I2): the two gaps the reviewer required landed before Task 4 -/// consumes this primitive. Both guard the design decisions the review approved -- see -/// task-2-review.md concern (a) and finding I2's Task-4-adoption note. ---- - -/// I1: pins the single-`fence_ok`-call `Created` design (concern (a)) so a future contributor cannot -/// silently "fix the inconsistency" by re-adding the sibling ops' post-write fence recheck. That change -/// would break Task 4's old-generation-retry semantics (resolveWedgeOnce deliberately calls slotOccupy under -/// the wedge's ORIGINAL admitted_fence_generation, and relies on ITS OWN post-I/O checkFenceOrThrow, -/// not a second internal check here, to decide whether the result is still relevant). A counting -/// fence_ok that only answers true on its FIRST call: if slotOccupy ever called it again after the -/// write landed, this test would see Unresolved instead of Created, OR (if the outcome happened to -/// still read Created some other way) the call-count assertion below would catch the extra invocation -/// either way. -TEST(CASSlotOccupy, CreatedNeverRechecksFenceAfterTheWrite) +/// A commit whose admission was withdrawn while it was in flight is reported as unresolved, never as +/// committed: the object may well exist, and the caller has to resolve the key rather than act on a +/// claim made under an incarnation it no longer holds. This is the OPPOSITE of the retired +/// slot-occupy primitive's single-pre-attempt-check contract, and it is what lets the ref lane's +/// wedge stay wedged instead of installing against a fence it has already lost. +TEST(CASSlotOccupy, AdmissionLostAfterTheWriteIsNeverReportedCommitted) { auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); - - int fence_calls = 0; - const auto fence_ok = [&fence_calls] - { - ++fence_calls; - return fence_calls == 1; - }; - - const auto result = controller.slotOccupy("k", "payload", fence_ok); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Created); - EXPECT_EQ(fence_calls, 1) << "slotOccupy must call fence_ok() exactly ONCE (pre-attempt only) -- " - "a post-write recheck would falsely report Unresolved here (fence_calls's " - "SECOND answer is false) and would break Task 4's old-generation-retry design"; + bool live = true; + backend->onWriteCommitted("k", [&live] { live = false; }); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([&live] { return live; }); + + const WriteResult result = op.create("k", "payload", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any) << "the write was sent, and it landed -- the caller must resolve the key"; + + /// The object IS durable; only the claim about it is refused. + CasRequests open_requests(backend, Fence::open()); + CasOperation reader = open_requests.admit(); + const auto landed = reader.read("k", Retry::once()); + ASSERT_TRUE(landed.has_value()); + EXPECT_EQ(landed->bytes, "payload"); } -/// A conflict needs a second backend request to resolve its occupant. Admission may disappear while -/// the conditional create is in flight; in that case the resolver must fail closed before starting -/// the `GET`, while preserving the one-check `Created` contract above. -TEST(CASSlotOccupy, AdmissionLostAfterConflictPreventsTheResolveGet) +/// A conflict needs a second request to name its occupant. Admission may disappear while the +/// conditional create is in flight; the settling read must then not start at all. +TEST(CASSlotOccupy, AdmissionLostAfterAConflictPreventsTheSettlingRead) { - auto backend = std::make_shared(); - ASSERT_EQ(backend->putIfAbsent("k", "existing").outcome, PutOutcome::Done); - CasRequestController controller(backend, CasRequestBudget{}); + auto backend = std::make_shared(); + CasRequests seed_requests(backend, Fence::open()); + CasOperation seeder = seed_requests.admit(); + ASSERT_TRUE(std::holds_alternative(seeder.create("k", "existing", Retry::once()))); + backend->resetCounts(); - int admission_checks = 0; - const auto admitted = [&admission_checks] - { - ++admission_checks; - return admission_checks == 1; - }; - - const auto result = controller.slotOccupy("k", "attempt", admitted); - EXPECT_EQ(result.kind, SlotOccupyResult::Kind::Unresolved); - EXPECT_EQ(admission_checks, 2); - EXPECT_EQ(backend->putCount("k"), 2u); + CasRequests requests(backend, Fence::open()); + CasOperation op = requests.admit([&backend] { return backend->live; }); + const WriteResult result = op.create("k", "attempt", Retry::once()); + const auto * gave_up = std::get_if(&result); + ASSERT_TRUE(gave_up != nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); + EXPECT_TRUE(gave_up->sent_any); + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 0u) - << "slotOccupy started its ambiguity-resolution GET after admission was withdrawn"; + << "the settling read started after admission was withdrawn"; } -/// I2: proves Occupied is reachable for an occupant that is OUR OWN earlier ambiguous write, not only -/// for a foreign pre-seeded one (PreExistingKeyOccupiedWithExactBytesAndTokenTwoOps above always seeds -/// via a plain, unambiguous putIfAbsent). This is the exact input shape Task 4's resolveWedgeOnce -/// adjudicates: "Occupied + bytes == wedge.bytes -> an earlier attempt landed -> adopt" (plan :329). -TEST(CASSlotOccupy, OwnLandedAmbiguousWriteObservedAsOccupiedOnRetry) +/// The wedge-adoption input shape: an earlier ambiguous attempt of the SAME key and bytes landed, and +/// a later flush issues a fresh create for it. The first call settles it by reading its own bytes; the +/// second sees them as an ordinary occupant, which is what the lane's `mine` adjudication consumes. +TEST(CASSlotOccupy, OwnLandedAmbiguousWriteIsObservedOnTheNextAttempt) { - auto backend = std::make_shared(); - CasRequestController controller(backend, CasRequestBudget{}); - - /// Call 1 -- the original attempt: the PUT's own response is lost, but the write DID commit, and - /// THIS call's own resolve GET (unfaulted) observes it immediately -- Occupied with OUR bytes, - /// proving the same-call resolve path works for a landed ambiguous write, not only a foreign one. - const auto first = controller.slotOccupy("k", "my-bytes", [] { return true; }); - EXPECT_EQ(first.kind, SlotOccupyResult::Kind::Occupied); - EXPECT_EQ(first.occupant_bytes, "my-bytes"); - EXPECT_EQ(backend->putCount("k"), 1u); + auto backend = std::make_shared(); + backend->injectAmbiguousLandedWrite("k"); + CasRequests requests(backend, Fence::open()); + + CasOperation first_op = requests.admit(); + const WriteResult first = first_op.create("k", "my-bytes", Retry::once()); + const auto * committed = std::get_if(&first); + ASSERT_TRUE(committed != nullptr) << "the write landed; the settling read proves it"; + EXPECT_TRUE(committed->resolved_by_read); + const Etag landed_incarnation = committed->etag; + EXPECT_EQ(backend->writeTotal(), 1u); EXPECT_EQ(backend->getCount("k"), 1u); - /// Call 2 -- Task 4's resolveWedgeOnce pattern: a LATER caller's flush resolving the SAME logical - /// attempt via a FRESH slotOccupy call. The fault is already consumed (one-shot), so this PUT - /// conflicts cleanly (PreconditionFailed) and the resolve GET observes OUR OWN earlier bytes again -- - /// the exact adoption input Task 4 is built on, and the SAME incarnation both calls saw. - const auto second = controller.slotOccupy("k", "my-bytes", [] { return true; }); - EXPECT_EQ(second.kind, SlotOccupyResult::Kind::Occupied); - EXPECT_EQ(second.occupant_bytes, "my-bytes"); - EXPECT_EQ(second.occupant_token, first.occupant_token) << "both calls must observe the SAME landed incarnation"; - EXPECT_EQ(backend->putCount("k"), 2u); + CasOperation second_op = requests.admit(); + const WriteResult second = second_op.create("k", "my-bytes", Retry::once()); + const Object * occupant = conflictObject(second); + ASSERT_TRUE(occupant != nullptr); + EXPECT_EQ(occupant->bytes, "my-bytes"); + EXPECT_EQ(occupant->etag, landed_incarnation) << "both calls must observe the SAME landed incarnation"; + EXPECT_EQ(backend->writeTotal(), 2u); EXPECT_EQ(backend->getCount("k"), 2u); } diff --git a/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp b/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp index 4c0000b46236..80b7e7eaddfa 100644 --- a/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp +++ b/src/Disks/tests/gtest_cas_sweep_deletion_premise.cpp @@ -59,13 +59,17 @@ struct OrphanFixture /// exercising the independent sweep-deletion premise. casAdmitRecoverableEntry(*backend, store->layout(), ns); writeManifestRaw(*backend, store->layout(), ns, orphan, {blobEntryFor("a", DB::UInt128(1))}); - /// min_active 6 > build_sequence 5: the durable watermark fact makes the prefix ELIGIBLE, which + /// min_active_build_sequence 6 > build_sequence 5: the durable watermark fact makes the prefix ELIGIBLE, which /// is the half the premise sits on top of. - setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active*/6); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active_build_sequence*/6); } String orphanKey() const { return store->layout().manifestKey(ManifestId{ns, orphan}); } - bool orphanExists() const { return backend->head(orphanKey()).exists; } + bool orphanExists() const + { + OperationForTest op(*backend); + return (*op).head(orphanKey(), Retry::once()).has_value(); + } }; /// The same admissible orphan shape as `OrphanFixture`, but without its legal manifest write: the @@ -81,11 +85,15 @@ struct UndecodableOrphanFixture { store = openPoolForTest(backend); casAdmitRecoverableEntry(*backend, store->layout(), ns); - setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active*/6); + setWatermarkMinActive(*backend, store->layout(), kServerRoot, kBuildEpoch, /*min_active_build_sequence*/6); } String orphanKey() const { return store->layout().manifestKey(ManifestId{ns, orphan}); } - bool orphanExists() const { return backend->head(orphanKey()).exists; } + bool orphanExists() const + { + OperationForTest op(*backend); + return (*op).head(orphanKey(), Retry::once()).has_value(); + } }; } @@ -179,7 +187,7 @@ TEST(CASSweepDeletionPremise, AnUnconsumedTailRemovalRetainsItsTarget) NamespaceFoldView view; RefCoverage cov; - cov.classification = 2; + cov.classification = CoverageClass::Folded; cov.last_folded_ref_id = RefTxnId{kBuildEpoch + 1, 1}; /// rule (1) satisfied view.coverage = cov; view.tail_removal_targets.insert(key); @@ -239,10 +247,11 @@ TEST(CASSweepDeletionPremise, AnUndecodableManifestDoesNotWedgeTheCursorPage) const size_t at = bytes.find("==> \"a.txt\""); ASSERT_NE(at, String::npos) << "no banner line to corrupt -- the entry must be Inline, not Blob"; bytes[at + 5] = 'X'; /// Inside the quoted path, same length, so no other offset shifts. - const PutResult put = f.backend->putIfAbsent(f.orphanKey(), sealObject(FormatId::PartManifest, bytes)); - /// `putIfAbsent` over an existing key writes nothing and reports `PreconditionFailed`, so a - /// silently legal body would make every assertion below pass against the wrong object. - ASSERT_EQ(put.outcome, PutOutcome::Done) << "the poison body was not the one planted"; + OperationForTest poison_op(*f.backend); + const WriteResult put = (*poison_op).create(f.orphanKey(), sealObject(FormatId::PartManifest, bytes), Retry::once()); + /// `create` over an existing key writes nothing and reports a `Conflict`, so a silently legal body + /// would make every assertion below pass against the wrong object. + ASSERT_TRUE(std::holds_alternative(put)) << "the poison body was not the one planted"; const ManifestId legal = writeManifestRaw( *f.backend, f.store->layout(), f.ns, ref(5, 0xCD), {blobEntryFor("b", DB::UInt128(2))}); @@ -262,7 +271,7 @@ TEST(CASSweepDeletionPremise, AnUndecodableManifestDoesNotWedgeTheCursorPage) /// keys remain, so a moved-cursor assertion fails after a correct fix, not before it. EXPECT_TRUE(result.wrapped); /// And the strong form: the object beyond the poison key was still decided this page. - EXPECT_FALSE(f.backend->head(legal_key).exists) + EXPECT_FALSE((*poison_op).head(legal_key, Retry::once()).has_value()) << "the sweep stopped at the poison key instead of walking past it"; } @@ -318,7 +327,7 @@ TEST(CASSweepDeletionPremise, DistinctRetainReasonsLandInDistinctCounters) .retry_count = 1, .next_retry_round = 4}; seedFoldCursorForTest(*backend, layout, ns_b, RefTxnId{kBuildEpoch + 1, 8}, hold); - setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/6); + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active_build_sequence*/6); const ManifestSweepResult result = sweepManifestCursorPageForTest(*store, "", /*list_budget*/100, /*delete_budget*/10); @@ -377,8 +386,9 @@ TEST(CASSweepDeletionPremise, AnExhaustedDeleteBudgetRetainsAndDoesNotStepOverTh << "the cursor must not have stepped over the candidates the exhausted budget left undecided"; size_t surviving = 0; + OperationForTest survive_op(*f.backend); for (const ManifestRef & r : {f.orphan, second, third}) - if (f.backend->head(f.store->layout().manifestKey(ManifestId{f.ns, r})).exists) + if ((*survive_op).head(f.store->layout().manifestKey(ManifestId{f.ns, r}), Retry::once()).has_value()) ++surviving; EXPECT_EQ(surviving, 0u); } @@ -403,9 +413,9 @@ TEST(CASSweepDeletionPremise, RecoveryWorkBudgetRetainsAndConvergesWithoutWedgin /// takes the fresh-`_ckpt` `putIfAbsent` path instead of `advanceRecoverableCkptForRawFixture`'s /// monotonic-advance-from-existing-value path (which throws on a null `committed_through`). casAdmitEntry(*backend, layout, ns); - setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/1000); + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active_build_sequence*/1000); - /// Six orphan candidates, all eligible (build_sequence << min_active), none owned by any ref. + /// Six orphan candidates, all eligible (build_sequence << min_active_build_sequence), none owned by any ref. constexpr int kCandidates = 6; for (int i = 1; i <= kCandidates; ++i) writeManifestRaw(*backend, layout, ns, ref(i, 1), @@ -457,8 +467,9 @@ TEST(CASSweepDeletionPremise, RecoveryWorkBudgetRetainsAndConvergesWithoutWedgin EXPECT_EQ(total_skipped, static_cast(kCandidates)) << "every one of the six candidates was decided (skipped), none silently dropped from the page"; EXPECT_GE(total_retained_work_budget, 1u); + OperationForTest survive_op(*backend); for (int i = 1; i <= kCandidates; ++i) - EXPECT_TRUE(backend->head(layout.manifestKey(ManifestId{ns, ref(i, 1)})).exists) + EXPECT_TRUE((*survive_op).head(layout.manifestKey(ManifestId{ns, ref(i, 1)}), Retry::once()).has_value()) << "candidate " << i << " must survive: it was never proven safe to delete"; } @@ -484,7 +495,7 @@ TEST(CASSweepDeletionPremise, NamespaceWorkBudgetCapsDistinctViewsPerPage) /// at all (`_ckpt.committed_through` unset), so absent the namespace cap BOTH would delete. seedFoldCursorForTest(*backend, layout, ns_a, RefTxnId{kBuildEpoch + 1, 1}); seedFoldCursorForTest(*backend, layout, ns_b, RefTxnId{kBuildEpoch + 1, 1}); - setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active*/6); + setWatermarkMinActive(*backend, layout, kServerRoot, kBuildEpoch, /*min_active_build_sequence*/6); GcRoundWorkBudget budget; budget.max_sweep_namespaces = 1; @@ -497,8 +508,9 @@ TEST(CASSweepDeletionPremise, NamespaceWorkBudgetCapsDistinctViewsPerPage) EXPECT_EQ(budget.sweep_namespaces_used, 1u); size_t surviving = 0; + OperationForTest survive_op(*backend); for (const auto & p : std::vector>{{ns_a, ref_a}, {ns_b, ref_b}}) - if (backend->head(layout.manifestKey(ManifestId{p.first, p.second})).exists) + if ((*survive_op).head(layout.manifestKey(ManifestId{p.first, p.second}), Retry::once()).has_value()) ++surviving; EXPECT_EQ(surviving, 1u) << "exactly one candidate remains -- the one whose namespace had no budget left"; } diff --git a/src/Disks/tests/gtest_cas_text_format.cpp b/src/Disks/tests/gtest_cas_text_format.cpp index 4371afb3f9e8..c300e522b739 100644 --- a/src/Disks/tests/gtest_cas_text_format.cpp +++ b/src/Disks/tests/gtest_cas_text_format.cpp @@ -1,4 +1,5 @@ #include +#include "cas_format_test_battery.h" #include #include #include @@ -6,6 +7,7 @@ #include #include #include +#include #include using namespace DB::Cas; @@ -34,6 +36,15 @@ void expectCode(int code, F && f) } } +TEST(CASFormatBattery, EveryRegisteredFormatIsBatteryCovered) +{ + std::set registered; + for (FormatId id : allRegisteredFormatIds()) + registered.insert(id); + EXPECT_EQ(registered, DB::Cas::tests::batteryCoveredIds()) + << "a registered codec is missing from the common battery (or vice versa)"; +} + /// ---- Task 2: FormatId entries for refsnaplog / blob meta / heartbeat ---- TEST(CASFormatIds, NewIdsExistWithFrozenValues) @@ -130,6 +141,28 @@ TEST(CASJsonVocab, WriteAndReadBack) EXPECT_FALSE(r.nextKey(key)); } +TEST(CASJsonVocab, WordArrayFieldAndReaderRejectInvalidValues) +{ + CasJsonWriter out; + bool first = true; + const std::array words{"ch128", "sha256"}; + writeWordArrayField(out, WireKey{"algos_used"}, words, first); + closeObject(out, first); + EXPECT_EQ(std::move(out).take(), "{\"algos_used\":[\"ch128\",\"sha256\"]}"); + + const auto read = [](std::string_view text) + { + DB::ReadBufferFromMemory in(text.data(), text.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "test"); + String key; + EXPECT_TRUE(r.nextKey(key)); + return r.readStringArray(); + }; + EXPECT_EQ(read(R"({"algos_used":["ch128","sha256"]})"), (std::vector{"ch128", "sha256"})); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { read(R"({"algos_used":"ch128"})"); }); + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { read(R"({"algos_used":["ch128",1]})"); }); +} + TEST(CASJsonVocab, FailClosedRules) { auto reader = [](std::string_view text, KeyStrictness s, auto && consume) @@ -170,12 +203,12 @@ TEST(CASJsonVocab, FailClosedRules) r.nextKey(k); }); }); /// bad hex width / junk in u64 string - expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"h":"0102"})", KeyStrictness::Tolerant, [](auto & r) + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"digest":"0102"})", KeyStrictness::Tolerant, [](auto & r) { String k; r.nextKey(k); r.readHex128(); }); }); - expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"s":"12x"})", KeyStrictness::Tolerant, [](auto & r) + expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { reader(R"({"u64_string_field":"12x"})", KeyStrictness::Tolerant, [](auto & r) { String k; r.nextKey(k); r.readU64String(); @@ -203,9 +236,9 @@ TEST(CASTextHeader, WriteExpectSniffGate) EXPECT_FALSE(sniffHeaderLine("PAR1 not a cas object").has_value()); /// wrong type -> CORRUPTED_DATA; future v -> UNKNOWN_FORMAT_VERSION - /// `v:3` is deliberate and must NOT follow a future `G_BUILD` bump: any version <= G_BUILD passes - /// the header gate, which is the point — the BODY is what has to fail here. - const String wrong = "{\"type\":\"cas_owner\",\"v\":3}\n"; + /// `v:1` is the baseline generation, so it always passes the header gate -- the type mismatch is + /// what has to fail here. + const String wrong = "{\"type\":\"cas_owner\",\"v\":1}\n"; DB::ReadBufferFromMemory in2(wrong.data(), wrong.size()); expectCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { expectHeaderLine(in2, FormatId::PoolMeta); }); const String future = fmt::format("{{\"type\":\"cas_pool_meta\",\"v\":{}}}\n", currentCompatibilityVersion() + 1); @@ -241,19 +274,16 @@ TEST(CASZstdArm, SealOpenPolicyAndCaps) { /// Always types compress regardless of size (no threshold — the .zst key must be /// constructible without knowing the body); a raw body is still readable (repair path). - /// `v:3` here is NOT the "any version <= G_BUILD passes" case the other negative bodies rely on: - /// `cas_ref_snap`'s own `changePoints` floor is generation 4, so a generation-3 ref snapshot is not - /// readable by this build in principle. It passes the header gate only because nothing consults - /// `changePoints` at decode time yet -- the gate is `v > G_BUILD` alone. Once a per-class floor is - /// wired in, this literal must move to `G_BUILD`; the test's subject is the truncated BODY, not the - /// version. - const String small = "{\"type\":\"cas_ref_snap\",\"v\":3}\n{}\n"; + /// `sealObject`/`openObject` are the storage-wrapper layer and never invoke the version gate + /// (that happens at decode, e.g. `decodeRefSnapshot`'s `expectHeaderLine`), so `v:1` here is just + /// the baseline header -- the test's subject is the compression arm, not the version. + const String small = "{\"type\":\"cas_ref_snap\",\"v\":1}\n{}\n"; const String sealed_small = sealObject(FormatId::RefSnapshot, small); ASSERT_TRUE(looksZstd(sealed_small)); EXPECT_EQ(openObject(FormatId::RefSnapshot, sealed_small), small); EXPECT_EQ(openObject(FormatId::RefSnapshot, small), small); - String big = "{\"type\":\"cas_ref_snap\",\"v\":3}\n{\"pad\":\""; + String big = "{\"type\":\"cas_ref_snap\",\"v\":1}\n{\"pad\":\""; big += String(8192, 'a'); big += "\"}\n"; const String sealed = sealObject(FormatId::RefSnapshot, big); diff --git a/src/Disks/tests/gtest_cas_throttling_gate.cpp b/src/Disks/tests/gtest_cas_throttling_gate.cpp new file mode 100644 index 000000000000..d208eede19f9 --- /dev/null +++ b/src/Disks/tests/gtest_cas_throttling_gate.cpp @@ -0,0 +1,129 @@ +#include + +#include +#include +#include +#include +#include +#include +#include + +#include "config.h" + +#if USE_AWS_S3 + +namespace ProfileEvents +{ +extern const Event CASRequestResolveRead; +} + +using namespace DB::Cas; +using DB::Cas::tests::CountingBackend; + +namespace +{ + +/// The PartWriteTxn fixture every pool test uses: stage an empty manifest, precommit it under `ref`, +/// promote it. Empty content is enough -- this gate exercises the request contract under throttling, +/// not the blob path. +void publishEmptyPart(const PoolPtr & store, const RootNamespace & ns, const String & ref) +{ + PartWriteInfo info; + info.intended_namespace = ns; + info.intended_ref = ns.string() + "/" + ref; + auto build = store->beginPartWrite(info); + const ManifestId id = build->stageManifest({}); + build->precommitAdd(ns, ref, id); + build->promote(ns, ref, build->buildId(), id); +} + +/// Every key the total per-key request count a `CountingBackend` observed, summed across every +/// primitive: whichever verb the throttled key was refused on, this is what "requested again later" +/// means. +uint64_t totalRequestsFor(const CountingBackend & inner, const String & key) +{ + return inner.getCount(key) + inner.headCount(key) + inner.listCount(key) + inner.writeCount(key) + + inner.deleteCount(key) + inner.publishCount(key); +} + +} + +/// The `ThrottlingBackend` gate: every user-visible statement -- table creation, an +/// insert, a rename, a drop, the writable mount `Pool::open` itself performs, and one GC round -- must +/// still SUCCEED when every key it touches is refused exactly once (`FirstPerKey`, HTTP 429) before it +/// is honored. `refusals(key) == 1` for every key the gate recorded, and every one of them was requested +/// at least twice: once refused, at least once more to actually land. +/// +/// EXCLUDED, by design, and never reached by these scenarios: the in-band recovery walk's epoch seal at +/// `{E, T+1}` -- nothing here trips a fence, forces a remount, or drives recovery. +TEST(CASThrottlingGate, EveryUserVisibleStatementSucceedsUnderFirstPerKeyThrottling) +{ + auto inner = std::make_shared(); + auto throttled = std::make_shared( + inner, ThrottlingBackend::Mode::FirstPerKey, /*n=*/0, /*status=*/429); + + /// The writable mount at open, probe included: `Pool::open` itself issues the identity probe, the + /// mount claim and the epoch allocation under this same throttled backend. + auto store = DB::Cas::tests::openPoolForTest(throttled); + + const auto resolve_reads_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); + + const RootNamespace ns{"test/throttle_gate"}; + + /// CREATE-shaped: the namespace's first part write births it. + ASSERT_NO_THROW(publishEmptyPart(store, ns, "created")); + EXPECT_TRUE(store->resolveRef(ns, "created").has_value()); + + /// INSERT-shaped: a second part write into the now-live namespace. + ASSERT_NO_THROW(publishEmptyPart(store, ns, "inserted")); + EXPECT_TRUE(store->resolveRef(ns, "inserted").has_value()); + + /// RENAME-shaped: content addressing has no rename primitive (`PartFolderAccess::republishRef`'s own + /// comment) -- a rename publishes equivalent content at the destination ref and drops the source. + ASSERT_NO_THROW(publishEmptyPart(store, ns, "renamed")); + ASSERT_NO_THROW(store->dropRef(ns, "inserted")); + EXPECT_TRUE(store->resolveRef(ns, "renamed").has_value()); + EXPECT_FALSE(store->resolveRef(ns, "inserted").has_value()); + + /// One GC round, still under throttling. + Gc gc(store, UInt128{7101}); + ASSERT_NO_THROW(DB::Cas::tests::runRegularRoundReclaiming(gc)); + + /// DROP-shaped: the whole namespace goes last, so the statements above still have something to act on. + ASSERT_NO_THROW(store->dropNamespace(ns)); + + /// `refusals(key) == 1` for every key is a class invariant of `FirstPerKey` mode itself + /// (`refuseOrPass` refuses a key at most once, ever, by construction of `refused_keys.insert`), so + /// asserting it here would prove nothing about THIS run's engine behavior -- it cannot fail. What + /// can fail, and is the actual content of the gate: every key the gate decided was requested again + /// afterwards (an inner post-refusal count of zero means the caller never retried at all, which the + /// statements above already ruled out by succeeding), and at least one of them took more than one + /// post-refusal request. + /// + /// That count alone does NOT prove a resolve read happened: a key that is read and then written + /// reaches two requests through two different verbs. The counter delta below is what pins the + /// engine's own ambiguity resolution -- it is incremented at exactly one site, the write loop's + /// settle-by-reading step. It does not attribute the reads to any particular key, and it counts a + /// refused precondition the same as a throttled ambiguity. + size_t keys_needing_more_than_one_request = 0; + for (const String & key : throttled->decidedKeys()) + { + const uint64_t total = totalRequestsFor(*inner, key); + EXPECT_GE(total, 1u) + << "key: " << key << " -- one refusal plus at least one later success is 'requested at least twice'"; + if (total >= 2) + ++keys_needing_more_than_one_request; + } + EXPECT_FALSE(throttled->decidedKeys().empty()) << "the gate must have actually decided some keys"; + EXPECT_GT(keys_needing_more_than_one_request, 0u) + << "no throttled key needed more than one post-refusal request -- every refusal landed on a " + "trivially-retried read/list/head"; + /// Under coverage builds ProfileEvents propagate into a thread-local subtree that does not reach + /// `global_counters`; deltas read 0 there only (see gtest_unique_key_index_cache). +#if !WITH_COVERAGE + EXPECT_GT(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolve_reads_before, 0u) + << "no throttled write was settled by a read -- the engine's ambiguity-resolution path never ran"; +#endif +} + +#endif diff --git a/src/Disks/tests/gtest_cas_truncate_reclaim.cpp b/src/Disks/tests/gtest_cas_truncate_reclaim.cpp index 0a76c338cdb5..f155ca9a35ff 100644 --- a/src/Disks/tests/gtest_cas_truncate_reclaim.cpp +++ b/src/Disks/tests/gtest_cas_truncate_reclaim.cpp @@ -75,9 +75,9 @@ ManifestId publishPart2( /// (condemn -> graduate -> delete) is in flight while this is true. bool anyRetiredPending(const PoolPtr & s) { - /// Retired-in-snapshot (T4): condemned state rides the adopted fold seal's kCondemned rows, not a + /// Condemned state rides the adopted fold seal's RunMarker::Condemned rows, not a /// separate retired list — reconstruct the in-flight set from the seal. - return DB::Cas::tests::anyCondemnedInSeal(s->backend(), s->layout()); + return DB::Cas::tests::anyCondemnedInSeal(*s->poolBackendPtr(), s->layout()); } /// Run regular GC rounds until a fixpoint over the ACK-FLOOR round. A condemned blob is deleted only a @@ -276,7 +276,9 @@ TEST(CASTruncateReclaim, DropNamespaceLeavesSharedBlobDebrisForPerpetualSweep) << "an emptied pool must drain instead of standing still; the sweep owned these blobs and " "reclaimed them within " << rounds << " GC rounds"; EXPECT_EQ(after.reachable, 0u); - EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(s->backend(), s->layout(), ns)) + DB::Cas::CasRequests catalog_requests = DB::Cas::tests::openRequestsForTest(s->poolBackendPtr()); + DB::Cas::CasOperation catalog_op = catalog_requests.admit(); + EXPECT_FALSE(CasRefCatalog::lifeIfCataloged(catalog_op, s->layout(), ns)) << "physical debris must not keep the logical namespace life cataloged"; } } diff --git a/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp b/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp index 80d262854555..2cfe3fd2a8b5 100644 --- a/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp +++ b/src/Disks/tests/gtest_cas_txn_apply_ledger.cpp @@ -4,6 +4,7 @@ #include #include #include +#include using namespace DB::Cas; @@ -94,6 +95,7 @@ TEST(CASTxnApplyLedger, OnlyTheTransactionWhoseDeltasVanishedIsReported) TEST(CASTxnApplyLedger, ReducerMarksTheOrdinalOfEveryDeltaItConsumes) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest operation(backend); Layout layout{"pool"}; TxnApplyLedger ledger; @@ -107,7 +109,7 @@ TEST(CASTxnApplyLedger, ReducerMarksTheOrdinalOfEveryDeltaItConsumes) {bh(2), s(1), /*remove*/false, routed}, }; std::vector runs; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, + foldDeltasIntoGeneration(*operation, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, deltas, runs, /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, /*out_retired*/nullptr, @@ -135,6 +137,7 @@ TEST(CASTxnApplyLedger, ReducerMarksTheOrdinalOfEveryDeltaItConsumes) TEST(CASTxnApplyLedger, ReducerMarksAnUnmatchedRemovalDelta) { InMemoryBackend backend; + DB::Cas::tests::OperationForTest operation(backend); Layout layout{"pool"}; TxnApplyLedger ledger; @@ -146,7 +149,7 @@ TEST(CASTxnApplyLedger, ReducerMarksAnUnmatchedRemovalDelta) std::vector deltas{{bh(1), s(1), /*remove*/true, removal}}; std::vector runs; RetiredMergeResult merged; - foldDeltasIntoGeneration(backend, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, + foldDeltasIntoGeneration(*operation, layout, /*prior_runs*/{}, /*new_generation*/1, /*attempt*/0, /*shard*/0, deltas, runs, /*current_round*/0, /*condemn_round*/0, /*head_blob*/{}, /*peek_head*/{}, /*confirm_condemned_marker*/{}, &merged, diff --git a/src/Disks/tests/gtest_cas_upload_detached.cpp b/src/Disks/tests/gtest_cas_upload_detached.cpp index 6f6f341f4bb6..89ced775f08b 100644 --- a/src/Disks/tests/gtest_cas_upload_detached.cpp +++ b/src/Disks/tests/gtest_cas_upload_detached.cpp @@ -49,13 +49,14 @@ BlobSource reReadableStagedSource( source.server_side_copy_from = staging_key; source.open = [backend, staging_key, header_len, payload_size]() -> std::unique_ptr { - auto staged = backend->getStream(staging_key); + DB::Cas::tests::OperationForTest op(backend); + std::unique_ptr staged = (*op).stream(staging_key, Retry::once()); if (!staged) throw DB::Exception(DB::ErrorCodes::FILE_DOESNT_EXIST, "staging object {} is absent", staging_key); String encoded_header(header_len, '\0'); - staged->stream->readStrict(encoded_header.data(), encoded_header.size()); + staged->readStrict(encoded_header.data(), encoded_header.size()); (void)decodeEnvelopeHeader(encoded_header, header_len + payload_size, ObjectKind::Blob); - return std::move(staged->stream); + return staged; }; return source; } @@ -72,6 +73,27 @@ PartWriteTxnPtr precommitBuildFor( return build; } +/// A one-shot `create`, asserting it committed (mirrors the retired `backend.putIfAbsent(key, bytes)`). +void createObj(Backend & backend, const String & key, const String & bytes) +{ + DB::Cas::tests::OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); +} + +/// An exact read (mirrors the retired `backend.get(key)`). +std::optional readObj(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +/// Whether `key` has a value, through a HEAD (mirrors the retired `backend.head(key).exists`). +bool headPresent(Backend & backend, const String & key) +{ + DB::Cas::tests::OperationForTest op(backend); + return (*op).head(key, Retry::standard()).has_value(); +} + /// Seed a present, well-formed blob body whose LOGICAL bytes are exactly `payload` (a fixed envelope /// header followed by the payload), so a later HEAD returns a token and a logical size of `payload.size()`. void seedPresentBody( @@ -82,13 +104,13 @@ void seedPresentBody( h.incarnation_tag = DB::UInt128(0xABCD); h.build_id = DB::UInt128(0x1111); const String head = encodeEnvelopeHeader(h, static_cast(pm.blob_header_len)); - b.putIfAbsent(layout.blobKey(ref), head + payload); + createObj(b, layout.blobKey(ref), head + payload); } /// The logical payload stored at `key` (object body minus the fixed blob header), or empty when absent. String logicalPayloadAt(InMemoryBackend & b, const String & key, uint64_t header_len) { - const auto got = b.get(key); + const auto got = readObj(b, key); if (!got || got->bytes.size() < header_len) return {}; return got->bytes.substr(header_len); @@ -106,6 +128,8 @@ std::optional metaStateAt(InMemoryBackend & b, const Layout & layout, class ProtocolRecordingBackend final : public InMemoryBackend { public: + /// Unhide the legacy overload that the primitive override below would otherwise hide. + using InMemoryBackend::head; void watch(String blob_key_, String meta_key_) { blob_key = std::move(blob_key_); @@ -117,27 +141,27 @@ class ProtocolRecordingBackend final : public InMemoryBackend meta_gets_before_first_publish.reset(); } - HeadResult head(const String & key) override + std::optional head(const String & key, TransportAccess & access) override { if (key == blob_key) { ++blob_heads; operations.emplace_back("head"); } - return InMemoryBackend::head(key); + return InMemoryBackend::head(key, access); } - std::optional get(const String & key, Range range) override + std::optional read(const String & key, TransportAccess & access) override { if (key == meta_key) { ++meta_gets; operations.emplace_back("meta-get"); } - return InMemoryBackend::get(key, range); + return InMemoryBackend::read(key, access); } - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { if (request.destination_key == blob_key) { @@ -146,7 +170,7 @@ class ProtocolRecordingBackend final : public InMemoryBackend if (!meta_gets_before_first_publish) meta_gets_before_first_publish = meta_gets; } - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); } String blob_key; @@ -267,7 +291,8 @@ TEST(CASUploadDetached, PresentCondemnedPublishesFreshAndQueuedOldDeleteMisses) condemnMeta(*backend, store->layout(), u128Of(payload), 19); auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-condemned"}, "part", payload); const String blob_key = store->layout().blobKey(ref); - const Token condemned_token = backend->head(blob_key).token; + DB::Cas::tests::OperationForTest condemned_probe(*backend); + const Etag condemned_token = (*condemned_probe).head(blob_key, Retry::standard())->etag; backend->watch(blob_key, store->layout().blobMetaKey(ref)); const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(); @@ -280,8 +305,11 @@ TEST(CASUploadDetached, PresentCondemnedPublishesFreshAndQueuedOldDeleteMisses) EXPECT_EQ(backend->blob_heads, 1u); EXPECT_EQ(backend->publish_calls, 1u); EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(), avoided_before); - EXPECT_EQ(backend->deleteExact(blob_key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch); - EXPECT_TRUE(backend->head(blob_key).exists); + { + DB::Cas::tests::OperationForTest op(*backend); + EXPECT_EQ((*op).remove(blob_key, condemned_token, Retry::once()), Removal::Mismatch); + EXPECT_TRUE((*op).head(blob_key, Retry::standard()).has_value()); + } } /// A present body with absent metadata is observed and backfilled `Clean` without publication. @@ -356,7 +384,7 @@ TEST(CASUploadDetached, FreshLocalStreaming) arrange(b1, s1, build1); const String key = s1->layout().blobKey(blob); - ASSERT_FALSE(b1->head(key).exists); /// precondition: absent + ASSERT_FALSE(headPresent(*b1, key)); /// precondition: absent EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); const BlobUploadResult r = build1->uploadBlobDetached( @@ -370,7 +398,7 @@ TEST(CASUploadDetached, FreshLocalStreaming) EXPECT_EQ(r.dep.size, payload.size()); EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); - EXPECT_TRUE(b1->head(key).exists); + EXPECT_TRUE(headPresent(*b1, key)); EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), payload); EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); @@ -406,7 +434,7 @@ TEST(CASUploadDetached, S3StagingPromotion) h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; - b->putIfAbsent(staging_key, staging_bytes); + createObj(*b, staging_key, staging_bytes); build = precommitBuildFor(s, ns, ref_name, payload); }; @@ -417,7 +445,7 @@ TEST(CASUploadDetached, S3StagingPromotion) arrange(b1, s1, build1, staging_bytes1); const String key = s1->layout().blobKey(blob); - ASSERT_FALSE(b1->head(key).exists); + ASSERT_FALSE(headPresent(*b1, key)); EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); const BlobUploadResult r = build1->uploadBlobDetached( @@ -433,9 +461,9 @@ TEST(CASUploadDetached, S3StagingPromotion) EXPECT_EQ(r.dep.size, payload.size()); EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); - ASSERT_TRUE(b1->head(key).exists); + ASSERT_TRUE(headPresent(*b1, key)); /// The server-side copy moved the staging bytes verbatim to the blob key. - const auto got = b1->get(key); + const auto got = readObj(*b1, key); ASSERT_TRUE(got.has_value()); EXPECT_EQ(got->bytes, staging_bytes1); @@ -449,7 +477,7 @@ TEST(CASUploadDetached, S3StagingPromotion) reReadableStagedSource(b2, staging_key, payload.size(), s2->poolMeta().blob_header_len)); EXPECT_EQ(build2->dependencyProof(blob), BlobDependencyProof::Materialized); - const auto got2 = b2->get(key); + const auto got2 = readObj(*b2, key); ASSERT_TRUE(got2.has_value()); EXPECT_EQ(got->bytes, got2->bytes); } @@ -479,7 +507,8 @@ TEST(CASUploadDetached, CondemnedLocalResurrection) PartWriteTxnPtr build1; arrange(b1, s1, build1); const String key = s1->layout().blobKey(blob); - const Token condemned_token = b1->head(key).token; + DB::Cas::tests::OperationForTest token_probe(*b1); + const Etag condemned_token = (*token_probe).head(key, Retry::standard())->etag; ASSERT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Condemned)); EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); @@ -495,8 +524,8 @@ TEST(CASUploadDetached, CondemnedLocalResurrection) EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); /// The condemned incarnation was displaced by a fresh one (token changed) and the meta is Clean again. - const Token after_token = b1->head(key).token; - EXPECT_NE(after_token.value, condemned_token.value); + const Etag after_token = (*token_probe).head(key, Retry::standard())->etag; + EXPECT_NE(after_token, condemned_token); EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); EXPECT_EQ(logicalPayloadAt(*b1, key, s1->poolMeta().blob_header_len), payload); @@ -531,9 +560,9 @@ TEST(CASUploadDetached, CondemnedS3Resurrection) h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; - b->putIfAbsent(staging_key, staging_bytes); + createObj(*b, staging_key, staging_bytes); /// Seed the condemned blob body = exactly a verbatim promote of the staging object would produce. - b->putIfAbsent(s->layout().blobKey(blob), staging_bytes); + createObj(*b, s->layout().blobKey(blob), staging_bytes); writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/9); build = precommitBuildFor(s, ns, ref_name, payload); @@ -544,7 +573,8 @@ TEST(CASUploadDetached, CondemnedS3Resurrection) PartWriteTxnPtr build1; arrange(b1, s1, build1); const String key = s1->layout().blobKey(blob); - const Token condemned_token = b1->head(key).token; + DB::Cas::tests::OperationForTest token_probe(*b1); + const Etag condemned_token = (*token_probe).head(key, Retry::standard())->etag; ASSERT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Condemned)); EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); @@ -563,8 +593,8 @@ TEST(CASUploadDetached, CondemnedS3Resurrection) EXPECT_EQ(build1->dependencyProof(blob), std::nullopt); /// A fresh incarnation displaced the condemned one (INV-NO-RETURN: fresh tag ⇒ different token). - const Token after_token = b1->head(key).token; - EXPECT_NE(after_token.value, condemned_token.value); + const Etag after_token = (*token_probe).head(key, Retry::standard())->etag; + EXPECT_NE(after_token, condemned_token); EXPECT_EQ(metaStateAt(*b1, s1->layout(), payload), std::optional(MetaState::Clean)); std::shared_ptr b2; diff --git a/src/Disks/tests/gtest_cas_upload_fanout.cpp b/src/Disks/tests/gtest_cas_upload_fanout.cpp index 1371d8a70fc7..5989ea44ccd3 100644 --- a/src/Disks/tests/gtest_cas_upload_fanout.cpp +++ b/src/Disks/tests/gtest_cas_upload_fanout.cpp @@ -41,6 +41,7 @@ using DB::Cas::tests::expectThrowsCode; using DB::Cas::tests::runRoundsUntilAbsent; using DB::Cas::tests::blobAbsent; using DB::Cas::tests::CountingBackend; +using DB::Cas::tests::OperationForTest; namespace DB::ErrorCodes { @@ -77,6 +78,24 @@ PoolPtr openPool(const std::shared_ptr & b) return Pool::open(b, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); } +std::optional readOf(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, DB::Cas::Retry::standard()); +} + +bool headExists(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).head(key, DB::Cas::Retry::standard()).has_value(); +} + +void createRaw(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + (*op).create(key, bytes, DB::Cas::Retry::standard()); +} + /// Stage a one-blob seed manifest and precommit it, so every adopt branch of `uploadBlobDetached` /// passes its EDGE-BEFORE-OBSERVE fail-closed gate (which only checks the `precommitted` flag). One /// precommit covers an arbitrary number of subsequently-uploaded blobs, mirroring @@ -100,13 +119,13 @@ void seedPresentBody(InMemoryBackend & b, const Layout & layout, const PoolMeta h.incarnation_tag = DB::UInt128(0xABCD); h.build_id = DB::UInt128(0x1111); const String head = encodeEnvelopeHeader(h, static_cast(pm.blob_header_len)); - b.putIfAbsent(layout.blobKey(idOf(payload)), head + payload); + createRaw(b, layout.blobKey(idOf(payload)), head + payload); } /// The logical payload stored at a blob key (object body minus the fixed blob header), or empty when absent. String logicalPayloadAt(InMemoryBackend & b, const String & key, uint64_t header_len) { - const auto got = b.get(key); + const auto got = readOf(b, key); if (!got || got->bytes.size() < header_len) return {}; return got->bytes.substr(header_len); @@ -224,7 +243,7 @@ struct ConcurrencyProbe class RejectFirstStagedCopyBackend final : public InMemoryBackend { public: - void publishBlob(const BlobPublishRequest & request) override + void publish(const BlobPublishRequest & request, TransportAccess & access) override { if (std::holds_alternative(request.publication)) { @@ -239,7 +258,7 @@ class RejectFirstStagedCopyBackend final : public InMemoryBackend { ++streaming_publications; } - InMemoryBackend::publishBlob(request); + InMemoryBackend::publish(request, access); } bool reject_copy = true; @@ -264,7 +283,7 @@ TEST(CASUploadFanout, CopiedAndMovedRequestsSharePublicationAttemptedState) header.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging_bytes = encodeEnvelopeHeader(header, static_cast(store->poolMeta().blob_header_len)) + payload; - backend->putIfAbsent(staging_key, staging_bytes); + createRaw(*backend, staging_key, staging_bytes); BlobSource source; source.size = payload.size(); @@ -293,7 +312,7 @@ TEST(CASUploadFanout, CopiedAndMovedRequestsSharePublicationAttemptedState) EXPECT_EQ(backend->streaming_publications, 1u) << "the request copied and moved through fan-out must retain the consumed first-attempt state"; EXPECT_EQ(build->dependencyProof(ref), BlobDependencyProof::Materialized); - const auto stored = backend->get(store->layout().blobKey(ref)); + const auto stored = readOf(*backend, store->layout().blobKey(ref)); ASSERT_TRUE(stored.has_value()); EXPECT_EQ(stored->bytes.substr(store->poolMeta().blob_header_len), payload); } @@ -347,7 +366,7 @@ WorldA arrangeWorldA() h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging = encodeEnvelopeHeader(h, static_cast(w.s->poolMeta().blob_header_len)) + kStaging; - w.b->putIfAbsent("p/staging/mount1/A-staging.tmp", staging); + createRaw(*w.b, "p/staging/mount1/A-staging.tmp", staging); } /// condemned-local resurrection: present body + condemned meta, local source. @@ -362,8 +381,8 @@ WorldA arrangeWorldA() h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging = encodeEnvelopeHeader(h, static_cast(w.s->poolMeta().blob_header_len)) + kResS3; - w.b->putIfAbsent("p/staging/mount1/A-republish.tmp", staging); - w.b->putIfAbsent(w.s->layout().blobKey(idOf(kResS3)), staging); + createRaw(*w.b, "p/staging/mount1/A-republish.tmp", staging); + createRaw(*w.b, w.s->layout().blobKey(idOf(kResS3)), staging); writeMetaClean(*w.b, w.s->layout(), u128Of(kResS3), std::string(kResS3).size()); condemnMeta(*w.b, w.s->layout(), u128Of(kResS3), /*condemn_round=*/9); } @@ -436,8 +455,8 @@ TEST(CASUploadFanout, CondemnedBranchesNeverGet) h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + s3_payload; - counting->putIfAbsent(s3_staging, staging); - counting->putIfAbsent(s->layout().blobKey(idOf(s3_payload)), staging); + createRaw(*counting, s3_staging, staging); + createRaw(*counting, s->layout().blobKey(idOf(s3_payload)), staging); writeMetaClean(*counting, s->layout(), u128Of(s3_payload), s3_payload.size()); condemnMeta(*counting, s->layout(), u128Of(s3_payload), /*condemn_round=*/5); } @@ -598,22 +617,24 @@ TEST(CASUploadFanout, CondemnedLocalResurrectStreamsAndFlipsMetaClean) condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/13); const String blob_key = s->layout().blobKey(idOf(payload)); - const Token condemned_token = b->head(blob_key).token; + OperationForTest op(*b); + const auto condemned_meta = (*op).head(blob_key, Retry::standard()); + ASSERT_TRUE(condemned_meta.has_value()); std::vector reqs{localRequest(payload)}; auto pool = makePool(2); fanOutBlobUploads(*build, reqs, *pool, nullptr); - /// A fresh incarnation displaced the condemned one; INV-NO-RETURN: the queued exact-token delete - /// of the condemned incarnation must miss the resurrection. - const HeadResult after = b->head(blob_key); - ASSERT_TRUE(after.exists); - EXPECT_NE(after.token, condemned_token); - EXPECT_EQ(b->deleteExact(blob_key, condemned_token).kind, DeleteOutcome::Kind::TokenMismatch); - EXPECT_TRUE(b->head(blob_key).exists); + /// A fresh incarnation displaced the condemned one; INV-NO-RETURN: the queued exact-incarnation + /// delete of the condemned incarnation must miss the resurrection. + const auto after = (*op).head(blob_key, Retry::standard()); + ASSERT_TRUE(after.has_value()); + EXPECT_NE(after->etag, condemned_meta->etag); + EXPECT_EQ((*op).remove(blob_key, condemned_meta->etag, Retry::standard()), Removal::Mismatch); + EXPECT_TRUE(headExists(*b, blob_key)); /// The payload survived verbatim under the fresh header. - const auto got = b->get(blob_key); + const auto got = readOf(*b, blob_key); ASSERT_TRUE(got.has_value()); EXPECT_EQ(got->bytes.substr(s->poolMeta().blob_header_len), payload); @@ -670,12 +691,14 @@ TEST(CASUploadFanout, DuplicateCondemnedS3ResurrectsCorrectly) h.kind = ObjectKind::Blob; h.incarnation_tag = DB::UInt128(0xC0FFEE); const String staging_bytes = encodeEnvelopeHeader(h, static_cast(s->poolMeta().blob_header_len)) + payload; - b->putIfAbsent(staging_key, staging_bytes); - b->putIfAbsent(s->layout().blobKey(idOf(payload)), staging_bytes); + createRaw(*b, staging_key, staging_bytes); + createRaw(*b, s->layout().blobKey(idOf(payload)), staging_bytes); writeMetaClean(*b, s->layout(), u128Of(payload), payload.size()); condemnMeta(*b, s->layout(), u128Of(payload), /*condemn_round=*/11); - const Token condemned_token = b->head(s->layout().blobKey(idOf(payload))).token; + OperationForTest op(*b); + const auto condemned_meta = (*op).head(s->layout().blobKey(idOf(payload)), Retry::standard()); + ASSERT_TRUE(condemned_meta.has_value()); std::atomic dispatched{0}; BlobUploadFanoutHooksForTest hooks; @@ -687,8 +710,9 @@ TEST(CASUploadFanout, DuplicateCondemnedS3ResurrectsCorrectly) EXPECT_EQ(dispatched.load(), 1) << "duplicate condemned records collapse to one republication task"; EXPECT_EQ(build->dependencyProof(idOf(payload)), BlobDependencyProof::Materialized); - const Token after_token = b->head(s->layout().blobKey(idOf(payload))).token; - EXPECT_NE(after_token.value, condemned_token.value) << "a fresh incarnation displaced the condemned one"; + const auto after_meta = (*op).head(s->layout().blobKey(idOf(payload)), Retry::standard()); + ASSERT_TRUE(after_meta.has_value()); + EXPECT_NE(after_meta->etag, condemned_meta->etag) << "a fresh incarnation displaced the condemned one"; EXPECT_EQ(metaStateAt(*b, s->layout(), payload), std::optional(MetaState::Clean)); EXPECT_EQ(logicalPayloadAt(*b, s->layout().blobKey(idOf(payload)), s->poolMeta().blob_header_len), payload); } @@ -860,7 +884,7 @@ TEST(CASUploadFanout, DrainPrecedesUnwind) fanOutBlobUploads(*build, reqs, *pool, &hooks); }); - EXPECT_TRUE(b->head(s->layout().blobKey(idOf(slow))).exists) + EXPECT_TRUE(headExists(*b, s->layout().blobKey(idOf(slow)))) << "the sibling's upload was drained by the join before the failure surfaced"; EXPECT_EQ(build->dependencyProof(idOf(slow)), std::nullopt) << "merge-nothing: the drained sibling's dep is not merged"; @@ -920,7 +944,7 @@ TEST(CASUploadFanout, DispatchThrowStillDrains) EXPECT_EQ(dispatch_calls.load(), 2) << "the throw fired on the second dispatch"; /// The already-RUNNING first task was drained before the stack unwound, so its body is present /// although nothing was merged. - EXPECT_TRUE(b->head(s->layout().blobKey(idOf(enqueued))).exists) + EXPECT_TRUE(headExists(*b, s->layout().blobKey(idOf(enqueued)))) << "the already-dispatched task was drained before the stack unwound"; EXPECT_EQ(build->depsSnapshotForTest().size(), 0u) << "merge-nothing on a dispatch throw"; } @@ -978,7 +1002,7 @@ TEST(CASUploadFanout, TrackingSeamThrowStillDrains) /// The already-scheduled first task was drained before `results` was destroyed, so its body is /// present; nothing was merged (merge-nothing on any fan-out throw). - EXPECT_TRUE(b->head(s->layout().blobKey(idOf(smaller))).exists) + EXPECT_TRUE(headExists(*b, s->layout().blobKey(idOf(smaller)))) << "an already-scheduled task was not drained before the stack unwound"; EXPECT_EQ(build->depsSnapshotForTest().size(), 0u) << "merge-nothing on a tracking-seam throw"; } diff --git a/src/Disks/tests/gtest_cas_upstream_slice.cpp b/src/Disks/tests/gtest_cas_upstream_slice.cpp new file mode 100644 index 000000000000..367f3b2a915a --- /dev/null +++ b/src/Disks/tests/gtest_cas_upstream_slice.cpp @@ -0,0 +1,777 @@ +#include + +#include +#include +#include +#include + +#include +#include +#include +#include + +#include "config.h" + +#if USE_AWS_S3 +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#endif + +namespace DB::ErrorCodes +{ + extern const int NOT_IMPLEMENTED; +#if USE_AWS_S3 + extern const int CANNOT_READ_ALL_DATA; + extern const int NETWORK_ERROR; +#endif +} + +namespace +{ + +/// Same unique-temp-root convention as the other CAS unit tests, so parallel runs never share a root. +std::shared_ptr makeLocalObjectStorageForRetryProfileTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_unit_upstream_slice_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + return std::make_shared(DB::LocalObjectStorageSettings("test", root, /*read_only_=*/false)); +} + +/// Every refusal below is NOT_IMPLEMENTED, and so is the pre-existing refusal of conditional removal, +/// so the code alone cannot tell which one fired. Match a phrase unique to the intended message too. +template +void expectThrowsNotImplementedSaying(const std::string & needle, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception saying '" << needle << "'"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NOT_IMPLEMENTED); + EXPECT_NE(e.message().find(needle), std::string::npos) << "actual message: " << e.message(); + } +} + +} + +/// The base `IObjectStorage` bodies forward `Default` and refuse `SingleAttempt`: a caller that asked +/// for one attempt has its own deadline, and a transparently retried request would outlive it. +TEST(CASUpstreamSlice, HeadListRemoveOverloadsRefuseSingleAttemptOnTheBaseStorage) +{ + auto local = makeLocalObjectStorageForRetryProfileTest(); + + const DB::ObjectStorageControlRequest single_attempt{.profile = DB::ObjectStorageRetryProfile::SingleAttempt}; + const DB::ObjectStorageControlRequest default_profile{.profile = DB::ObjectStorageRetryProfile::Default}; + + expectThrowsNotImplementedSaying( + "single-attempt metadata requests", + [&] { local->tryGetObjectMetadataWithNativeToken("k", false, single_attempt); }); + expectThrowsNotImplementedSaying( + "single-attempt listing requests", + [&] { local->iterate("", 1, false, {}, single_attempt); }); + expectThrowsNotImplementedSaying( + "single-attempt removal requests", + [&] { local->removeObjectIfTokenMatches(DB::StoredObject("k"), "e", single_attempt); }); + + /// `Default` must keep reaching the ordinary implementation. For removal that is still a refusal, + /// but the pre-existing one — matching its wording proves the profile overload forwarded. + EXPECT_NO_THROW(local->tryGetObjectMetadataWithNativeToken("k", false, default_profile)); + EXPECT_NO_THROW(local->iterate("", 1, false, {}, default_profile)); + expectThrowsNotImplementedSaying( + "Conditional (token-exact) object removal", + [&] { local->removeObjectIfTokenMatches(DB::StoredObject("k"), "e", default_profile); }); +} + +#if USE_AWS_S3 + +namespace +{ + +/// One scripted answer to a `GetObject`. `fail_mid_body` makes the response stream throw after it has +/// already delivered bytes, which is what drives `ReadBufferFromS3` to reissue the request. +struct ScriptedGetObjectStep +{ + bool ok = true; + Aws::S3::S3Errors error = Aws::S3::S3Errors::SLOW_DOWN; + std::string exception_name; + std::string etag; + std::string body; + bool fail_mid_body = false; +}; + +ScriptedGetObjectStep okStep(const std::string & etag, const std::string & body, bool fail_mid_body = false) +{ + return ScriptedGetObjectStep{ + .ok = true, + .error = Aws::S3::S3Errors::SLOW_DOWN, + .exception_name = "", + .etag = etag, + .body = body, + .fail_mid_body = fail_mid_body}; +} + +ScriptedGetObjectStep throttleStep() +{ + return ScriptedGetObjectStep{ + .ok = false, + .error = Aws::S3::S3Errors::SLOW_DOWN, + .exception_name = "SlowDown", + .etag = "", + .body = "", + .fail_mid_body = false}; +} + +/// `S3Exception::isAccessTokenExpiredError` keys on the error CODE, not the name. +ScriptedGetObjectStep expiredTokenStep() +{ + return ScriptedGetObjectStep{ + .ok = false, + .error = Aws::S3::S3Errors::ACCESS_DENIED, + .exception_name = "ExpiredToken", + .etag = "", + .body = "", + .fail_mid_body = false}; +} + +/// One scripted answer to a control-plane request (HEAD or conditional DELETE). `retryable` is what +/// the SDK's retry strategy consults, so it is what decides whether the client's own attempt loop +/// reissues the request — which is how a single-attempt clone is told apart from the disk client. +struct ScriptedControlStep +{ + bool ok = true; + Aws::S3::S3Errors error = Aws::S3::S3Errors::SLOW_DOWN; + std::string exception_name; + bool retryable = false; +}; + +ScriptedControlStep controlOk() +{ + return ScriptedControlStep{.ok = true, .error = Aws::S3::S3Errors::SLOW_DOWN, .exception_name = "", .retryable = false}; +} + +ScriptedControlStep controlExpiredToken() +{ + return ScriptedControlStep{ + .ok = false, .error = Aws::S3::S3Errors::ACCESS_DENIED, .exception_name = "ExpiredToken", .retryable = false}; +} + +ScriptedControlStep controlThrottle() +{ + return ScriptedControlStep{ + .ok = false, .error = Aws::S3::S3Errors::SLOW_DOWN, .exception_name = "SlowDown", .retryable = true}; +} + +/// `ReadBufferFromIStream` reads through `Poco::Net::HTTPBasicStreamBuf::readFromDevice`, so a fake +/// response body has to be one of those rather than a plain `std::stringstream`. +class ScriptedBodyStreamBuf : public Poco::Net::HTTPBasicStreamBuf +{ +public: + ScriptedBodyStreamBuf(std::string body_, bool fail_mid_body_) + : Poco::Net::HTTPBasicStreamBuf(256, std::ios::in), body(std::move(body_)), fail_mid_body(fail_mid_body_) + { + } + +private: + int readFromDevice(char * buffer, std::streamsize length) override + { + if (fail_mid_body && position > 0) + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "scripted failure part-way through the response body"); + + const size_t available = body.size() - position; + const size_t n = std::min(static_cast(length), available); + std::memcpy(buffer, body.data() + position, n); + position += n; + return static_cast(n); + } + + const std::string body; + const bool fail_mid_body; + size_t position = 0; +}; + +class ScriptedBodyStreamHolder +{ +protected: + ScriptedBodyStreamHolder(std::string body, bool fail_mid_body) : buf(std::move(body), fail_mid_body) { } + ScriptedBodyStreamBuf buf; +}; + +/// The holder base is listed first so `buf` is constructed before `std::iostream` is handed its address. +class ScriptedBodyStream : private ScriptedBodyStreamHolder, public std::iostream +{ +public: + ScriptedBodyStream(std::string body, bool fail_mid_body) + : ScriptedBodyStreamHolder(std::move(body), fail_mid_body), std::iostream(&buf) + { + } +}; + +/// An `S3::Client` whose `GetObject` answers from a script, recording how many times it was called and +/// whether each request carried the native-conditional mark. Clones share the state, so the counters +/// still see the requests issued through the single-attempt clone. +class ScriptedGetObjectClient : public DB::S3::Client +{ +private: + struct State + { + std::vector script; + std::vector head_script; + std::vector delete_script; + + size_t get_object_calls = 0; + std::vector native_conditional_marks; + + /// The `requestTimeoutMs` of the client each request was actually issued through, which is + /// what proves a request rode the clone built for the bound its caller asked for. + std::vector head_request_timeouts_ms; + std::vector delete_request_timeouts_ms; + /// Every configuration this client was asked to clone with — a request-free way to see which + /// client a verb selected. + std::vector clone_request_timeouts_ms; + + std::mutex mutex; + }; + + const std::shared_ptr state; + +public: + ScriptedGetObjectClient() : ScriptedGetObjectClient(std::make_shared(), GetClientConfiguration()) { } + + static DB::S3::PocoHTTPClientConfiguration GetClientConfiguration() + { + DB::RemoteHostFilter remote_host_filter; + /// max_retries is deliberately nonzero: it is the disk client's own attempt loop, and the + /// only thing that distinguishes it from the single-attempt clone. The two slow-down flags + /// are off so that loop spins without waiting out a real backoff. + return DB::S3::ClientFactory::instance().createClientConfiguration( + "some-region", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 2}, + /* s3_slow_all_threads_after_network_error = */ false, + /* s3_slow_all_threads_after_retryable_error = */ false, + /* enable_s3_requests_logging = */ true, + /* for_disk_s3 = */ false, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + } + + void script(std::vector steps) const + { + std::lock_guard lock(state->mutex); + state->script = std::move(steps); + } + + size_t getObjectCalls() const + { + std::lock_guard lock(state->mutex); + return state->get_object_calls; + } + + std::vector nativeConditionalMarks() const + { + std::lock_guard lock(state->mutex); + return state->native_conditional_marks; + } + + void scriptHead(std::vector steps) const + { + std::lock_guard lock(state->mutex); + state->head_script = std::move(steps); + } + + void scriptDelete(std::vector steps) const + { + std::lock_guard lock(state->mutex); + state->delete_script = std::move(steps); + } + + std::vector headRequestTimeouts() const + { + std::lock_guard lock(state->mutex); + return state->head_request_timeouts_ms; + } + + std::vector deleteRequestTimeouts() const + { + std::lock_guard lock(state->mutex); + return state->delete_request_timeouts_ms; + } + + std::vector cloneRequestTimeouts() const + { + std::lock_guard lock(state->mutex); + return state->clone_request_timeouts_ms; + } + + std::unique_ptr cloneWithConfigurationOverride( + const DB::S3::PocoHTTPClientConfiguration & client_configuration_override) const override + { + { + std::lock_guard lock(state->mutex); + state->clone_request_timeouts_ms.push_back(client_configuration_override.requestTimeoutMs); + } + return std::unique_ptr(new ScriptedGetObjectClient(state, client_configuration_override)); + } + + Aws::S3::Model::GetObjectOutcome GetObject(const Aws::S3::Model::GetObjectRequest & request) const override + { + std::lock_guard lock(state->mutex); + + const auto * marked = dynamic_cast(&request); + state->native_conditional_marks.push_back(marked != nullptr && marked->isNativeConditional()); + + const size_t index = state->get_object_calls++; + if (index >= state->script.size()) + { + return Aws::S3::Model::GetObjectOutcome(Aws::Client::AWSError( + Aws::S3::S3Errors::NO_SUCH_KEY, "NoSuchKey", "the script has no answer for this request", false)); + } + + const auto & step = state->script[index]; + if (!step.ok) + { + return Aws::S3::Model::GetObjectOutcome(Aws::Client::AWSError( + step.error, step.exception_name, "scripted error", false)); + } + + Aws::S3::Model::GetObjectResult result; + result.SetETag(step.etag); + result.SetContentLength(static_cast(step.body.size())); + /// The SDK releases the body through `Aws::Delete`, which frees with the SDK's allocator, so + /// the stream must come from `Aws::New`; a plain `new` here is an alloc-dealloc mismatch. + result.ReplaceBody(Aws::New("ScriptedBodyStream", step.body, step.fail_mid_body)); + return Aws::S3::Model::GetObjectOutcome(std::move(result)); + } + + Aws::S3::Model::HeadObjectOutcome HeadObject(const Aws::S3::Model::HeadObjectRequest & /*request*/) const override + { + std::lock_guard lock(state->mutex); + state->head_request_timeouts_ms.push_back(getClientConfiguration().requestTimeoutMs); + + const auto step = nextControlStep(state->head_script, state->head_request_timeouts_ms.size()); + if (!step.ok) + return Aws::S3::Model::HeadObjectOutcome(makeError(step)); + + Aws::S3::Model::HeadObjectResult result; + /// Any nonzero size: tryGetObjectMetadataImpl reads an all-zero HeadObjectResult as a miss. + result.SetContentLength(scripted_head_object_size); + result.SetETag(scripted_head_object_etag); + return Aws::S3::Model::HeadObjectOutcome(std::move(result)); + } + + Aws::S3::Model::DeleteObjectOutcome DeleteObject(const Aws::S3::Model::DeleteObjectRequest & /*request*/) const override + { + std::lock_guard lock(state->mutex); + state->delete_request_timeouts_ms.push_back(getClientConfiguration().requestTimeoutMs); + + const auto step = nextControlStep(state->delete_script, state->delete_request_timeouts_ms.size()); + if (!step.ok) + return Aws::S3::Model::DeleteObjectOutcome(makeError(step)); + + Aws::S3::Model::DeleteObjectResult result; + result.SetDeleteMarker(false); + return Aws::S3::Model::DeleteObjectOutcome(std::move(result)); + } + +private: + static constexpr long long scripted_head_object_size = 7; + static constexpr const char * scripted_head_object_etag = "\"h1\""; + + /// A script shorter than the number of requests keeps answering with its last step, so a test + /// that means "this error, however many attempts the client makes" says it in one entry. + static ScriptedControlStep nextControlStep(const std::vector & script, size_t call_number) + { + if (script.empty()) + return controlOk(); + return script[std::min(call_number - 1, script.size() - 1)]; + } + + static Aws::Client::AWSError makeError(const ScriptedControlStep & step) + { + return Aws::Client::AWSError(step.error, step.exception_name, "scripted error", step.retryable); + } + + ScriptedGetObjectClient(std::shared_ptr state_, const DB::S3::PocoHTTPClientConfiguration & client_configuration) + : DB::S3::Client( + 100, + DB::S3::ServerSideEncryptionKMSConfig(), + std::make_shared("", ""), + client_configuration, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + DB::S3::ClientSettings{ + .use_virtual_addressing = true, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }) + , state(std::move(state_)) + { + } +}; + +std::shared_ptr makeScriptedS3ObjectStorage( + ScriptedGetObjectClient *& out_client, + DB::S3ObjectStorage::S3CredentialsRefreshCallback credentials_refresh_callback = {}) +{ + auto owned_client = std::make_unique(); + out_client = owned_client.get(); + + DB::S3::URI uri; + uri.bucket = "cas-upstream-slice-bucket"; + DB::S3Capabilities capabilities; + DB::ObjectStorageKeyGeneratorPtr key_generator; + + return std::make_shared( + std::move(owned_client), + std::make_unique(), + std::move(uri), + capabilities, + key_generator, + "cas-upstream-slice-disk", + /*for_disk_s3_=*/true, + credentials_refresh_callback); +} + +template +void expectThrowsCodeSaying(int expected_code, const std::string & needle, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception saying '" << needle << "'"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + EXPECT_NE(e.message().find(needle), std::string::npos) << "actual message: " << e.message(); + } +} + +/// Builds the storage over a refresh callback that vends one fresh scripted client, so a test can +/// both script that client up front and assert the storage ended up holding that exact object. +std::shared_ptr makeScriptedS3ObjectStorageWithRefresh( + ScriptedGetObjectClient *& out_client, + ScriptedGetObjectClient *& out_refreshed, + std::function script_refreshed) +{ + return makeScriptedS3ObjectStorage( + out_client, + [&out_refreshed, script_refreshed]() -> std::unique_ptr + { + auto fresh = std::make_unique(); + script_refreshed(*fresh); + out_refreshed = fresh.get(); + return fresh; + }); +} + +} + +/// A plain `GET` must carry the same native-conditional mark a `HEAD` does when the read asks for it, +/// so that on a generation-token store both answer with the same incarnation identity. +TEST(CASUpstreamSlice, NativeConditionalReadSettingMarksTheGetRequest) +{ + (void)getContext(); + + ScriptedGetObjectClient * marked_client = nullptr; + auto marked_storage = makeScriptedS3ObjectStorage(marked_client); + marked_client->script({okStep("\"e1\"", "AAAA")}); + + DB::ReadSettings marked_settings; + marked_settings.object_storage_request_mode = DB::ObjectStorageRequestMode::NativeConditional; + marked_storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), marked_settings, 1 << 20); + + ASSERT_EQ(marked_client->nativeConditionalMarks().size(), 1u); + EXPECT_TRUE(marked_client->nativeConditionalMarks().at(0)); + + ScriptedGetObjectClient * plain_client = nullptr; + auto plain_storage = makeScriptedS3ObjectStorage(plain_client); + plain_client->script({okStep("\"e1\"", "AAAA")}); + + plain_storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), DB::ReadSettings{}, 1 << 20); + + ASSERT_EQ(plain_client->nativeConditionalMarks().size(), 1u); + EXPECT_FALSE(plain_client->nativeConditionalMarks().at(0)); +} + +/// The buffer's own retry loop can straddle a replacement of the object: the first response delivers +/// some of the old incarnation's bytes to the consumer before its stream breaks mid-body, and the +/// reissue answers with a different ETag. The bytes handed back are then from neither incarnation +/// alone. `buffer_size` is pinned to 2 so the first fill (of "AAAA"'s 4 bytes) completes and is +/// exposed to the consumer before the second fill hits the scripted mid-body failure - with the +/// default (much larger) buffer, that failure happens inside the very first fill, before any byte of +/// "e1" ever reaches the consumer, which is the "nothing to mix with" case covered below instead. +TEST(CASUpstreamSlice, ReadSmallObjectThrowsWhenAReissueAnswersWithADifferentETag) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + client->script({okStep("\"e1\"", "AAAA", /*fail_mid_body=*/true), okStep("\"e2\"", "BBBB")}); + + DB::ReadSettings read_settings; + read_settings.object_storage_request_mode = DB::ObjectStorageRequestMode::NativeConditional; + read_settings.remote_fs_settings.buffer_size = 2; + /// Default profile here: the buffer's own multi-attempt loop is what straddles the replacement. + expectThrowsCodeSaying( + DB::ErrorCodes::CANNOT_READ_ALL_DATA, + "response identity changed", + [&] { storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), read_settings, 1 << 20); }); + + EXPECT_EQ(client->getObjectCalls(), 2u); +} + +/// Scoped to an identity change that actually mixed bytes: with the default (large) buffer, "e1"'s +/// mid-body failure happens inside its very first fill attempt, before any byte crosses into the +/// consumer's buffer - so the reissue under a different ETag is an ordinary retry of a request that +/// never delivered anything, not a coherence problem, even though the ETag changed. +TEST(CASUpstreamSlice, ReadSmallObjectAcceptsAReissueThatDeliveredNoBytesEvenWithADifferentETag) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + client->script({okStep("\"e1\"", "AAAA", /*fail_mid_body=*/true), okStep("\"e2\"", "BBBB")}); + + const auto result = storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), DB::ReadSettings{}, 1 << 20); + EXPECT_EQ(result.data, "BBBB"); + EXPECT_EQ(result.metadata.etag, "\"e2\""); + EXPECT_EQ(client->getObjectCalls(), 2u); +} + +/// Scoped to an identity CHANGE: a reissue is ordinary, and refusing every retried read would turn a +/// dropped connection into a hard error. +TEST(CASUpstreamSlice, ReadSmallObjectAcceptsAReissueThatAnswersWithTheSameETag) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + client->script({okStep("\"e1\"", "AAAA", /*fail_mid_body=*/true), okStep("\"e1\"", "AAAA")}); + + const auto result = storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), DB::ReadSettings{}, 1 << 20); + EXPECT_EQ(result.data, "AAAA"); + EXPECT_EQ(client->getObjectCalls(), 2u); +} + +/// Under `SingleAttempt` the read must not retry at all: the caller owns the retry decision and its +/// own deadline. A throttle answer is retryable, so an unpinned buffer would reissue it. +TEST(CASUpstreamSlice, SingleAttemptProfileIssuesExactlyOneGetOnThrottle) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + client->script({throttleStep(), okStep("\"e1\"", "AAAA")}); + + DB::ReadSettings read_settings; + read_settings.object_storage_retry_profile = DB::ObjectStorageRetryProfile::SingleAttempt; + EXPECT_ANY_THROW(storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), read_settings, 1 << 20)); + + EXPECT_EQ(client->getObjectCalls(), 1u); +} + +TEST(CASUpstreamSlice, SingleAttemptClientCarriesTheRequestedTimeout) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + + const auto base_timeout = storage->getS3StorageClient()->getClientConfiguration().requestTimeoutMs; + + auto kept = storage->getSingleAttemptClient(0); + EXPECT_EQ(kept->getClientConfiguration().requestTimeoutMs, base_timeout); + + auto bounded = storage->getSingleAttemptClient(1234); + EXPECT_EQ(bounded->getClientConfiguration().requestTimeoutMs, 1234); + EXPECT_EQ(bounded->getClientConfiguration().retry_strategy.max_retries, 0u); + /// The cached clone is keyed by the timeout too, so one built for another bound is never served. + EXPECT_NE(bounded.get(), kept.get()); +} + +/// The buffer installs a refreshed client in itself only. A single-attempt read never retries, so that +/// copy is never used; what makes the caller's next attempt sign with the new credentials is the disk +/// client having been replaced. +TEST(CASUpstreamSlice, ExpiredTokenOnSingleAttemptReadInstallsTheRefreshedClientIntoTheStorage) +{ + (void)getContext(); + + ScriptedGetObjectClient * expired_client = nullptr; + const DB::S3::Client * refreshed_client = nullptr; + auto storage = makeScriptedS3ObjectStorage( + expired_client, + [&]() -> std::unique_ptr + { + auto fresh = std::make_unique(); + refreshed_client = fresh.get(); + return fresh; + }); + expired_client->script({expiredTokenStep()}); + + const auto * client_before = storage->getS3StorageClient().get(); + + DB::ReadSettings read_settings; + read_settings.object_storage_retry_profile = DB::ObjectStorageRetryProfile::SingleAttempt; + EXPECT_ANY_THROW(storage->readSmallObjectAndGetObjectMetadata(DB::StoredObject("k"), read_settings, 1 << 20)); + + ASSERT_NE(refreshed_client, nullptr); + EXPECT_NE(storage->getS3StorageClient().get(), client_before); + EXPECT_EQ(storage->getS3StorageClient().get(), refreshed_client); +} + +/// `refreshAndRetryOnExpiredCredentials` on the HEAD path: the vended credentials expire, the callback +/// hands over a fresh client, and the request is reissued through it. Installing that client into the +/// storage is what stops the next request repeating the failure. +TEST(CASUpstreamSlice, NativeTokenHeadRecoversFromAnExpiredTokenAndInstallsTheRefreshedClient) +{ + (void)getContext(); + + ScriptedGetObjectClient * expired = nullptr; + ScriptedGetObjectClient * refreshed = nullptr; + auto storage = makeScriptedS3ObjectStorageWithRefresh( + expired, refreshed, [](const ScriptedGetObjectClient & fresh) { fresh.scriptHead({controlOk()}); }); + /// Installing the refreshed client drops the storage's last reference to this one, so the raw + /// pointer would dangle before the assertions below read its counters. + const auto expired_owner = storage->getS3StorageClient(); + expired->scriptHead({controlExpiredToken()}); + + const auto metadata = storage->tryGetObjectMetadataWithNativeToken( + "k", /*with_tags=*/false, DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::Default}); + + ASSERT_TRUE(metadata.has_value()); + ASSERT_NE(refreshed, nullptr); + EXPECT_EQ(storage->getS3StorageClient().get(), refreshed); + EXPECT_EQ(expired->headRequestTimeouts().size(), 1u); + EXPECT_EQ(refreshed->headRequestTimeouts().size(), 1u); +} + +/// The same for the conditional DELETE, which is the other verb that issues inline. +TEST(CASUpstreamSlice, ConditionalRemoveRecoversFromAnExpiredTokenAndInstallsTheRefreshedClient) +{ + (void)getContext(); + + ScriptedGetObjectClient * expired = nullptr; + ScriptedGetObjectClient * refreshed = nullptr; + auto storage = makeScriptedS3ObjectStorageWithRefresh( + expired, refreshed, [](const ScriptedGetObjectClient & fresh) { fresh.scriptDelete({controlOk()}); }); + /// See the HEAD test: the storage's last reference to this client goes away when the refreshed + /// one is installed. + const auto expired_owner = storage->getS3StorageClient(); + expired->scriptDelete({controlExpiredToken()}); + + const auto result = storage->removeObjectIfTokenMatches( + DB::StoredObject("k"), "e", DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::Default}); + + EXPECT_EQ(result.outcome, DB::ConditionalRemoveOutcome::Removed); + ASSERT_NE(refreshed, nullptr); + EXPECT_EQ(storage->getS3StorageClient().get(), refreshed); + EXPECT_EQ(expired->deleteRequestTimeouts().size(), 1u); + EXPECT_EQ(refreshed->deleteRequestTimeouts().size(), 1u); +} + +/// The client the conditional DELETE selects is what decides whether the SDK reissues a throttled +/// request. The `Default` half is what makes "exactly one" mean something: the disk client here does +/// retry a throttle, so a single attempt is a property of the clone, not of the fake. +/// +/// Only the DELETE is counted. `S3::Client::HeadObject` does not use the SDK attempt loop at all — it +/// calls the virtual once and returns — so a HEAD is one request under either profile, and what the +/// profile changes for it is the transport bound, which the timeout test below pins. +TEST(CASUpstreamSlice, SingleAttemptConditionalRemoveIssuesExactlyOneRequestOnThrottle) +{ + (void)getContext(); + + ScriptedGetObjectClient * retrying = nullptr; + auto retrying_storage = makeScriptedS3ObjectStorage(retrying); + retrying->scriptDelete({controlThrottle()}); + + EXPECT_ANY_THROW(retrying_storage->removeObjectIfTokenMatches( + DB::StoredObject("k"), "e", DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::Default})); + EXPECT_EQ(retrying->deleteRequestTimeouts().size(), 3u); /// max_retries = 2, so three attempts + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + client->scriptDelete({controlThrottle()}); + + EXPECT_ANY_THROW(storage->removeObjectIfTokenMatches( + DB::StoredObject("k"), "e", DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::SingleAttempt})); + EXPECT_EQ(client->deleteRequestTimeouts().size(), 1u); +} + +/// The reservation the caller budgets for an attempt is only real if the transport is built to it, so +/// the two verbs must ride a clone carrying the timeout they asked for — and two different bounds must +/// coexist, or every alternation between verbs would rebuild a whole S3 client. +TEST(CASUpstreamSlice, HeadAndRemoveUnderSingleAttemptRideTheClientBoundToTheRequestedTimeout) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + + storage->tryGetObjectMetadataWithNativeToken( + "k", false, DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::SingleAttempt, .attempt_timeout_ms = 4321}); + ASSERT_EQ(client->headRequestTimeouts().size(), 1u); + EXPECT_EQ(client->headRequestTimeouts().at(0), 4321); + + storage->removeObjectIfTokenMatches(DB::StoredObject("k"), "e", + DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::SingleAttempt, .attempt_timeout_ms = 8765}); + ASSERT_EQ(client->deleteRequestTimeouts().size(), 1u); + EXPECT_EQ(client->deleteRequestTimeouts().at(0), 8765); + + EXPECT_EQ(client->cloneRequestTimeouts(), (std::vector{4321, 8765})); + + /// Asking again for a bound already built must reuse that clone rather than evict the other one. + storage->tryGetObjectMetadataWithNativeToken( + "k", false, DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::SingleAttempt, .attempt_timeout_ms = 4321}); + EXPECT_EQ(client->cloneRequestTimeouts(), (std::vector{4321, 8765})); + EXPECT_EQ(client->headRequestTimeouts().at(1), 4321); +} + +/// `iterate` issues nothing itself, so its client selection is observed through the clone it causes. +/// The async iterator fetches its first batch lazily, so constructing one sends no request. +TEST(CASUpstreamSlice, IterateUnderSingleAttemptSelectsTheClientBoundToTheRequestedTimeout) +{ + (void)getContext(); + + ScriptedGetObjectClient * client = nullptr; + auto storage = makeScriptedS3ObjectStorage(client); + + (void)storage->iterate("p", 1, false, {}, DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::Default}); + EXPECT_TRUE(client->cloneRequestTimeouts().empty()); + + (void)storage->iterate("p", 1, false, {}, + DB::ObjectStorageControlRequest{.profile = DB::ObjectStorageRetryProfile::SingleAttempt, .attempt_timeout_ms = 4321}); + EXPECT_EQ(client->cloneRequestTimeouts(), (std::vector{4321})); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_wire_vocab.cpp b/src/Disks/tests/gtest_cas_wire_vocab.cpp index efbe3da7a0ae..97bef46f5e73 100644 --- a/src/Disks/tests/gtest_cas_wire_vocab.cpp +++ b/src/Disks/tests/gtest_cas_wire_vocab.cpp @@ -1,35 +1,98 @@ #include #include +#include +#include +#include +#include #include +#include #include #include +#include + +#include +#include using namespace DB::Cas; namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; } +namespace +{ +/// Same tiny inline copy as `gtest_cas_part_manifest_format.cpp`'s `expectThrowsCode`: stays clear +/// of `Disks/tests/cas_test_helpers.h`'s `DB::Cas::tests::expectThrowsCode`, which would both drag +/// in the whole CAS backend/store machinery this file otherwise has no need for AND collide (same +/// namespace, same name and signature) if that header were ever included here too. +template +void expectThrowsCode(int expected_code, F && fn) +{ + try + { + fn(); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), expected_code); + } +} +} + +static_assert(DB::Cas::casEnumTableCoversEnum()); +static_assert(DB::Cas::casEnumTableCoversEnum()); +static_assert(DB::Cas::casEnumTableCoversEnum()); + +TEST(CASWireVocab, EnumTablesPinTheCurrentWords) +{ + using namespace DB::Cas; + EXPECT_EQ(kTokenTypeWords.toWord(Dialect::ETag, "t"), "etag"); + EXPECT_EQ(kTokenTypeWords.toWord(Dialect::Generation, "t"), "generation"); + EXPECT_EQ(kTokenTypeWords.toWord(Dialect::Emulated, "t"), "emulated"); + EXPECT_EQ(kBlobHashAlgoWords.toWord(BlobHashAlgo::CityHash128, "t"), "ch128"); + EXPECT_EQ(kBlobHashAlgoWords.toWord(BlobHashAlgo::XXH3_128, "t"), "xxh3"); + EXPECT_EQ(kBlobHashAlgoWords.toWord(BlobHashAlgo::Sha256, "t"), "sha256"); + EXPECT_EQ(kObjectKindWords.toWord(ObjectKind::Blob, "t"), "blob"); + EXPECT_EQ(refOwnerKindToWord(RefOwnerKind::Committed), "committed"); + EXPECT_EQ(refOwnerKindToWord(RefOwnerKind::Precommit), "precommit"); +} + +/// Every enum wire table's closed set, walked through `magic_enum::enum_values` rather than a +/// hand-copied list -- a future enumerator the encoder can construct but no table entry covers +/// would otherwise round-trip silently through the untested value. +TEST(CASWireVocab, ClosedSetsRoundTripEveryEnumeratorExhaustively) +{ + for (const auto t : magic_enum::enum_values()) + EXPECT_EQ(kTokenTypeWords.fromWord(kTokenTypeWords.toWord(t, "t"), "t"), t); + for (const auto k : magic_enum::enum_values()) + EXPECT_EQ(objectKindFromWord(objectKindToWord(k), "k"), k); + for (const auto a : magic_enum::enum_values()) + EXPECT_EQ(blobHashAlgoFromWord(blobHashAlgoName(a), "a"), a); + for (const auto k : magic_enum::enum_values()) + EXPECT_EQ(refOwnerKindFromWord(refOwnerKindToWord(k), "k"), k); +} + TEST(CASWireVocab, EnumWordsRoundTrip) { - for (TokenType t : {TokenType::ETag, TokenType::Generation, TokenType::Emulated}) - EXPECT_EQ(tokenTypeFromWord(tokenTypeToWord(t), "t"), t); + for (Dialect t : {Dialect::ETag, Dialect::Generation, Dialect::Emulated}) + EXPECT_EQ(kTokenTypeWords.fromWord(kTokenTypeWords.toWord(t, "t"), "t"), t); for (BlobHashAlgo a : {BlobHashAlgo::CityHash128, BlobHashAlgo::XXH3_128, BlobHashAlgo::Sha256}) EXPECT_EQ(blobHashAlgoFromWord(blobHashAlgoName(a), "a"), a); EXPECT_EQ(objectKindFromWord(objectKindToWord(ObjectKind::Blob), "k"), ObjectKind::Blob); - EXPECT_THROW(tokenTypeFromWord("nope", "t"), DB::Exception); - EXPECT_THROW(blobHashAlgoFromWord("nope", "a"), DB::Exception); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { dialectWordFromString("nope", "t"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { blobHashAlgoFromWord("nope", "a"); }); } TEST(CASWireVocab, SiblingFieldsWriteAndReadBack) { CasJsonWriter out; bool first = true; - writeTokenFields(out, first, Token{"etag-abc\"x", TokenType::ETag}); + writeTokenFields(out, first, PersistedEtag{"etag", "etag-abc\"x"}); const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(hexToU128("00112233445566778899aabbccddeeff"))}; writeBlobRefFields(out, first, ref); closeObject(out, first); const String rendered = std::move(out).take(); EXPECT_EQ(rendered, - R"({"tt":"etag","tv":"etag-abc\"x","ha":"ch128","h":"00112233445566778899aabbccddeeff"})"); + R"({"token_type":"etag","token":"etag-abc\"x","algo":"ch128","digest":"00112233445566778899aabbccddeeff"})"); DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); @@ -37,17 +100,243 @@ TEST(CASWireVocab, SiblingFieldsWriteAndReadBack) String tv; String ha; String h; - TokenType tt{}; + String tt; while (r.nextKey(key)) { - if (key == "tt") tt = tokenTypeFromWord(r.readString(), "t"); - else if (key == "tv") tv = r.readString(); - else if (key == "ha") ha = r.readString(); - else if (key == "h") h = r.readString(); + if (key == "token_type") tt = String(dialectWordFromString(r.readString(), "t")); + else if (key == "token") tv = r.readString(); + else if (key == "algo") ha = r.readString(); + else if (key == "digest") h = r.readString(); else r.skipUnknown(key); } - EXPECT_EQ(tt, TokenType::ETag); + EXPECT_EQ(tt, "etag"); EXPECT_EQ(tv, "etag-abc\"x"); const BlobRef back{blobHashAlgoFromWord(ha, "a"), codecFor(blobHashAlgoFromWord(ha, "a")).fromHex(h)}; EXPECT_EQ(back, ref); } + +TEST(CASWireVocab, ManifestRefBundleWritesTheOldPrefixedKeys) +{ + using namespace DB::Cas; + CasJsonWriter w; + bool first = true; + writeManifestRefFields(w, first, kOldManifestRefKeys, ManifestRef{1, 2, 3}); + w.closeObject(first); + EXPECT_EQ(std::move(w).take(), R"({"old_epoch":"1","old_build":"2","old_ord":3})"); +} + +TEST(CASWireVocab, MatchAndBuildRoundTripsABlobRef) +{ + using namespace DB::Cas; + const String rendered = R"({"algo":"ch128","digest":"00112233445566778899aabbccddeeff"})"; + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + BlobRefFields fields; + String key; + while (r.nextKey(key)) + { + if (matchBlobRefFields(key, r, fields)) + continue; + r.skipUnknown(key); + } + const BlobRef ref = fields.build("t"); + EXPECT_EQ(kBlobHashAlgoWords.toWord(ref.algo, "t"), "ch128"); +} + +TEST(CASWireVocab, BlobRefBuildFailsClosedOnHalfAGroupAndOnBadWidth) +{ + using namespace DB::Cas; + BlobRefFields only_algo; + only_algo.algo_word = "ch128"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { only_algo.build("t"); }); + + BlobRefFields short_digest; + short_digest.algo_word = "ch128"; + short_digest.digest_hex = "00112233445566778899aabbccddee"; /// 30 hex chars, needs 32 + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { short_digest.build("t"); }); +} + +TEST(CASWireVocab, BlobRefBuildFailsClosedOnRightWidthNonHexDigest) +{ + using namespace DB::Cas; + BlobRefFields bad_hex; + bad_hex.algo_word = "ch128"; + bad_hex.digest_hex = "gg112233445566778899aabbccddeeff"; /// 32 chars (right width), not hex + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { bad_hex.build("t"); }); +} + +TEST(CASWireVocab, MatchManifestRefFieldsAndBuildRefRoundTripInAnyKeyOrder) +{ + using namespace DB::Cas; + /// Fed out of writer order (ord, epoch, build) to pin key-order independence. `epoch`/`build` are quoted + /// decimal strings and `ord` is a bare number -- a swapped read primitive between the two shapes + /// would fail to parse this literal. + const String rendered = R"({"ord":3,"epoch":"7","build":"9"})"; + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + ManifestRefFields fields; + String key; + while (r.nextKey(key)) + { + if (matchManifestRefFields(key, r, kBareManifestRefKeys, fields)) + continue; + r.skipUnknown(key); + } + EXPECT_EQ(fields.buildRef("t", "ctx"), (ManifestRef{7, 9, 3})); +} + +TEST(CASWireVocab, ManifestRefFieldsBuildRefFailsClosedOnHalfAGroup) +{ + using namespace DB::Cas; + ManifestRefFields fields; + fields.epoch = 7; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { fields.buildRef("t", "ctx"); }); +} + +TEST(CASWireVocab, MatchTokenFieldsConsumesSemanticKeysAndLeavesUnrelatedKeyUnmatched) +{ + using namespace DB::Cas; + const String rendered = R"({"token_type":"etag","token":"abc","zz":1})"; + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + TokenFields fields; + String key; + bool saw_unmatched = false; + while (r.nextKey(key)) + { + if (matchTokenFields(key, r, fields)) + continue; + saw_unmatched = true; + r.skipUnknown(key); + } + ASSERT_TRUE(fields.type_word.has_value()); + EXPECT_EQ(*fields.type_word, "etag"); + ASSERT_TRUE(fields.value.has_value()); + EXPECT_EQ(*fields.value, "abc"); + EXPECT_TRUE(saw_unmatched); +} + +TEST(CASWireVocab, TokenFieldsBuildsInAnyKeyOrderAndRequiresBothFields) +{ + const String rendered = R"({"token":"abc","token_type":"etag"})"; + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + TokenFields fields; + String key; + while (r.nextKey(key)) + { + if (matchTokenFields(key, r, fields)) + continue; + r.skipUnknown(key); + } + const PersistedEtag built = fields.build("t"); + EXPECT_EQ(built.dialect, "etag"); + EXPECT_EQ(built.value, "abc"); + + TokenFields only_type; + only_type.type_word = "etag"; + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { only_type.build("t"); }); +} + +TEST(CASWireVocab, OldManifestEpochKeyDoesNotAliasTheSemanticKey) +{ + const String rendered = R"({"me":"1","build":"2","ord":3})"; + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Tolerant, "t"); + ManifestRefFields fields; + String key; + while (r.nextKey(key)) + { + if (matchManifestRefFields(key, r, kBareManifestRefKeys, fields)) + continue; + r.skipUnknown(key); + } + + try + { + fields.buildRef("RefTableSnapshot", "committed"); + FAIL() << "expected DB::Exception"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::CORRUPTED_DATA); + EXPECT_EQ(e.message(), "CAS RefTableSnapshot: committed manifest_ref missing epoch/build/ord"); + } +} + +/// A `PersistedEtag` survives every encoding a durable CAS record uses for one, and the type +/// system refuses the reverse direction: a persisted value must never be trusted to mint a live +/// `Etag`, which only an admitted request may produce. +static_assert(!std::is_constructible_v); + +TEST(CASPersistedEtag, RoundTripsThroughEveryFormatAndNeverBecomesAnIncarnation) +{ + const PersistedEtag recorded{"generation", R"(17"3)"}; /// a quote the JSON encodings must escape + + /// 1. The shared `token_type`/`token` JSON pair. + { + CasJsonWriter out; + bool first = true; + writeTokenFields(out, first, recorded); + closeObject(out, first); + const String rendered = std::move(out).take(); + DB::ReadBufferFromMemory in(rendered.data(), rendered.size()); + JsonObjectReader r(in, KeyStrictness::Strict, "t"); + TokenFields fields; + String key; + while (r.nextKey(key)) + ASSERT_TRUE(matchTokenFields(key, r, fields)) << "unexpected key " << key; + const PersistedEtag back = fields.build("t"); + EXPECT_EQ(back.dialect, recorded.dialect); + EXPECT_EQ(back.value, recorded.value); + } + + /// 2. The `cas_run` condemned row's NDJSON form. + { + const BlobRef ref{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(9))}; + DB::WriteBufferFromOwnString out; + SourceEdgeRunWriter writer(out); + writer.append(SourceEdgeRecord{.ref = ref, .source_id = UInt128(0), .marker = RunMarker::Condemned, + .delete_pending = true, .token = recorded, .size = 64, + .condemn_round = 3, .marker_confirmed = true}); + writer.finish(); + out.finalize(); + const String bytes = out.str(); + DB::ReadBufferFromMemory in(bytes.data(), bytes.size()); + SourceEdgeRunReader reader(in); + SourceEdgeRecord back; + ASSERT_TRUE(reader.next(back)); + EXPECT_EQ(back.token.dialect, recorded.dialect); + EXPECT_EQ(back.token.value, recorded.value); + EXPECT_FALSE(reader.next(back)); + } + + /// 3. The GC outcome log. + { + OutcomeLog log; + log.entries.push_back(OutcomeEntry{ObjectKind::Blob, + BlobRef{BlobHashAlgo::CityHash128, BlobDigest::fromU128(UInt128(4))}, recorded, OutcomeKind::Deleted}); + const OutcomeLog back = decodeOutcomeLog(encodeOutcomeLog(log)); + ASSERT_EQ(back.entries.size(), 1u); + EXPECT_EQ(back.entries[0].token.dialect, recorded.dialect); + EXPECT_EQ(back.entries[0].token.value, recorded.value); + } + + /// 4. The condemned row's packed byte form, whose dialect rides one byte rather than a word. + { + const CondemnedRow row{.delete_pending = false, .token = recorded, .size = 5, + .condemn_round = 11, .marker_confirmed = true}; + EXPECT_EQ(decodeCondemnedRow(encodeCondemnedRow(row)), row); + } +} + +/// Both directions of the dialect vocabulary fail closed, so neither encoding can carry a value the +/// other cannot name. +TEST(CASPersistedEtag, UnknownDialectWordAndByteAreBothRefused) +{ + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { dialectWordFromString("etags", "t"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { dialectByteFromWord("etags", "t"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { dialectWordFromByte(0, "t"); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [] { dialectWordFromByte(4, "t"); }); + EXPECT_EQ(dialectWordFromByte(dialectByteFromWord("generation", "t"), "t"), "generation"); +} diff --git a/src/Disks/tests/gtest_cas_write_once_key.cpp b/src/Disks/tests/gtest_cas_write_once_key.cpp new file mode 100644 index 000000000000..c05eb6b1e345 --- /dev/null +++ b/src/Disks/tests/gtest_cas_write_once_key.cpp @@ -0,0 +1,32 @@ +#include + +#include +#include +#include + +#include + +/// A `WriteOnceKey` names an object of one of the three families that are written once and never +/// rewritten: a part manifest, a ref log, a ref snapshot. Only `Layout` can mint one, from a typed +/// identity, so a verb that takes the type cannot be handed a mutable control key. + +using namespace DB::Cas; + +static_assert(!std::is_default_constructible_v); +static_assert(!std::is_constructible_v); +static_assert(!std::is_constructible_v); + +TEST(CASWriteOnceKey, FactoriesMintTheSameStringsAsThePlainKeyFunctions) +{ + const Layout layout{"p"}; + const RootNamespace ns{"test/aa@cas@"}; + const ManifestId manifest{ns, ManifestRef{.writer_epoch = 3, .build_sequence = 9, .manifest_ordinal = 2}}; + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(ns, DB::UInt128(0x1234)); + const RefTxnId id{5, 7}; + + EXPECT_EQ(layout.writeOnceManifestKey(manifest).str(), layout.manifestKey(manifest)); + EXPECT_EQ(layout.writeOnceRefLogKey(life, id).str(), layout.refLogKey(life, id)); + EXPECT_EQ(layout.writeOnceRefSnapshotKey(life, id).str(), layout.refSnapshotKey(life, id)); + EXPECT_TRUE(layout.parseManifestKey(layout.writeOnceManifestKey(manifest).str()).has_value()); + EXPECT_TRUE(layout.parseRefObjectKey(layout.writeOnceRefLogKey(life, id).str()).has_value()); +} diff --git a/src/Disks/tests/gtest_cas_writer_duties.cpp b/src/Disks/tests/gtest_cas_writer_duties.cpp index 662c0daa165e..6706df7e0109 100644 --- a/src/Disks/tests/gtest_cas_writer_duties.cpp +++ b/src/Disks/tests/gtest_cas_writer_duties.cpp @@ -7,6 +7,9 @@ #include #include +#include + +#include #include #include #include @@ -18,6 +21,7 @@ extern const int NETWORK_ERROR; } using namespace DB::Cas; +using DB::Cas::tests::SharedWaitLog; namespace { @@ -29,25 +33,29 @@ PoolConfig singleAttemptConfig() .server_root_id = "test", .background_watermark = false, }; - config.cas_request_budget.max_attempts = 1; config.cas_request_budget.attempt_timeout_ms = 100; - config.cas_request_budget.operation_deadline_ms = 5000; config.cas_request_budget.lease_safety_margin_ms = 100; return config; } -PoolPtr openSingleAttemptPool(const BackendPtr & backend) +template +PoolPtr openSingleAttemptPool(const std::shared_ptr & backend) { DB::Cas::tests::seedPoolMetaForRestart(*backend); + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the fence math these single-attempt fixtures drive matches what admits. + backend->setAttemptTimeoutMs(singleAttemptConfig().cas_request_budget.attempt_timeout_ms); return Pool::open(backend, singleAttemptConfig()); } -PoolPtr openFrozenSingleAttemptPool(const BackendPtr & backend) +template +PoolPtr openFrozenSingleAttemptPool(const std::shared_ptr & backend) { DB::Cas::tests::seedPoolMetaForRestart(*backend); PoolConfig config = singleAttemptConfig(); config.boot_ms_fn = [] { return uint64_t{0}; }; config.mount_renew_period = std::chrono::hours{1}; + backend->setAttemptTimeoutMs(config.cas_request_budget.attempt_timeout_ms); return Pool::open(backend, config); } @@ -88,6 +96,39 @@ uint64_t leaveRejectedCleanupDuty(const PoolPtr & store, const RootNamespace & n return rejected_seq; } +using DB::Cas::tests::LatchedChunkFaultBackend; + +/// Latches `backend` and drives `f` to a NETWORK_ERROR give-up, then disarms the fault completely so a +/// caller's next mutation reaches the store normally. The caller must have installed a +/// `VirtualRetryClock` on the store first, or the give-up paces through a real sleep instead of a +/// virtual one. +/// +/// A give-up is not by itself proof that the engine actually retried: a `once` policy reaches the same +/// outcome by propagating its first failure. Asserting the fault double's own hit count and the +/// clock's pause count is what tells the two apart -- both fire more than once only when reissues +/// really happened, whether the fault re-arms on every write attempt (`Unresolved`) or the resolving +/// read keeps retrying against a persistently lost response (`LandedThenLost`). +void driveToNetworkErrorGiveUp(LatchedChunkFaultBackend & backend, DB::Cas::tests::VirtualRetryClock & clock, + const std::function & f) +{ + backend.latched = true; + const int fault_hits_before = backend.fault_hits; + const size_t pauses_before = clock.pauseCount(); + DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, f); + EXPECT_GE(backend.fault_hits - fault_hits_before, 1) << "the fault double must actually have fired"; + /// ONE pause is the whole discriminator: a reissue is paced by a backoff the engine sleeps + /// through, and a `once` policy -- which reaches this same give-up by propagating its first + /// failure -- never sleeps at all. How many MORE pauses follow is deliberately not asserted: the + /// backoff is full jitter, so the reissues that fit before the call gives up are a random small + /// number, and demanding two of them failed about one run in eight against a schedule that was + /// behaving exactly as designed. + EXPECT_GE(clock.pauseCount() - pauses_before, 1u) + << "a give-up after a single attempt cannot distinguish a retrying `standard` policy from one " + << "that never reissues at all"; + EXPECT_GT(clock.longestPause(), 0u) << "at least one of the retry's pauses must be a real, nonzero backoff"; + backend.disarm(); +} + } /// Removing the deferred-cleanup transfer from `~PartWriteTxn` makes this test fail at the first @@ -96,8 +137,9 @@ uint64_t leaveRejectedCleanupDuty(const PoolPtr & store, const RootNamespace & n /// next mutation resolves the durable wedge, removes the exact old precommit, and only then retires it. TEST(CASWriterDuties, UncertainAdoptedGrantStaysActiveUntilTheNextMutationRemovesIt) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openSingleAttemptPool(backend); + auto clock = DB::Cas::tests::VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/writer_duty_adopt"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); @@ -109,9 +151,7 @@ TEST(CASWriterDuties, UncertainAdoptedGrantStaysActiveUntilTheNextMutationRemove backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::LandedThenLost; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::NETWORK_ERROR, - [&] { abandoned->precommitAdd(ns, "abandoned", abandoned_id); }); + driveToNetworkErrorGiveUp(*backend, *clock, [&] { abandoned->precommitAdd(ns, "abandoned", abandoned_id); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_EQ(abandoned->precommitState(), PartWriteTxn::PrecommitState::Uncertain); @@ -131,8 +171,11 @@ TEST(CASWriterDuties, UncertainAdoptedGrantStaysActiveUntilTheNextMutationRemove EXPECT_EQ( store->livePrecommitsForTest(ns), (std::set>{{"successor", successor_id.ref}})); - EXPECT_TRUE(backend->head(abandoned_manifest_key).exists) - << "the removed precommit body remains GC-owned until its decrement is sealed"; + { + DB::Cas::tests::OperationForTest verify_op(*backend); + EXPECT_TRUE((*verify_op).head(abandoned_manifest_key, Retry::once()).has_value()) + << "the removed precommit body remains GC-owned until its decrement is sealed"; + } successor->abandon(); EXPECT_TRUE(store->livePrecommitsForTest(ns).empty()); @@ -148,6 +191,9 @@ TEST(CASWriterDuties, ProvenAbsentGrantDrainsAsNoOpBeforeTheNextMutation) PoolConfig config = singleAttemptConfig(); config.boot_ms_fn = [] { return uint64_t{0}; }; config.mount_renew_period = std::chrono::hours{1}; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the fence math this single-attempt fixture drives matches what admits. + backend->setAttemptTimeoutMs(config.cas_request_budget.attempt_timeout_ms); auto store = Pool::open(backend, config); const RootNamespace ns{"srv1/writer_duty_reject"}; @@ -190,8 +236,9 @@ TEST(CASWriterDuties, ProvenAbsentGrantDrainsAsNoOpBeforeTheNextMutation) /// past the rejected build exactly as the no-wedge reject arm does. TEST(CASWriterDuties, WedgeResolvedAsRejectDrainsTheDutyAsNoOp) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openSingleAttemptPool(backend); + auto clock = DB::Cas::tests::VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/writer_duty_wedge_reject"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); @@ -202,9 +249,7 @@ TEST(CASWriterDuties, WedgeResolvedAsRejectDrainsTheDutyAsNoOp) backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::NETWORK_ERROR, - [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + driveToNetworkErrorGiveUp(*backend, *clock, [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); ASSERT_TRUE(store->refLaneWedgedForTest(ns)); ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); @@ -324,10 +369,12 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem DB::Cas::tests::seedPoolMetaForRestart(*backend); const CasRequestBudget budget{ .attempt_timeout_ms = 50, - .operation_deadline_ms = 500, - .max_attempts = 1, .lease_safety_margin_ms = 50, + .connect_timeout_cap_ms = std::nullopt, }; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); const RootNamespace ns{"srv1/writer_duty_crash"}; auto predecessor = Pool::open(backend, PoolConfig{ @@ -352,13 +399,17 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem abandoned.reset(); predecessor.reset(); - const auto mount = backend->get(mount_key); + DB::Cas::tests::OperationForTest mount_op(*backend); + const auto mount = (*mount_op).read(mount_key, Retry::once()); ASSERT_TRUE(mount.has_value()); - EXPECT_NE(decodeMountLease(mount->bytes).min_active, std::numeric_limits::max()) + EXPECT_NE(decodeMountLease(mount->bytes).min_active_build_sequence, std::numeric_limits::max()) << "a live writer-cleanup duty forbids the clean-release certificate"; - uint64_t fake_boot = 0; - std::vector waits; + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); + auto waits = std::make_shared(); auto successor_store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), @@ -367,11 +418,18 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem .mount_lease_ttl_ms = std::chrono::milliseconds(500), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, }); ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); - ASSERT_FALSE(waits.empty()) << "the predecessor supplied no clean-death certificate"; + ASSERT_FALSE(waits->empty()) << "the predecessor supplied no clean-death certificate"; ManifestId successor_id; auto successor = stageEmptyManifest(successor_store, ns, "successor", successor_id); @@ -396,14 +454,16 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem /// build is REJECTED rather than adopted) and then runs real GC rounds until the body is gone. TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); DB::Cas::tests::seedPoolMetaForRestart(*backend); const CasRequestBudget budget{ .attempt_timeout_ms = 50, - .operation_deadline_ms = 500, - .max_attempts = 1, .lease_safety_margin_ms = 50, + .connect_timeout_cap_ms = std::nullopt, }; + /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the budget + /// field alone; pair the two so the mount lease's admission arithmetic sees what the budget claims. + backend->setAttemptTimeoutMs(budget.attempt_timeout_ms); /// Rooted under the POOL's OWN `server_root_id` ("test", unlike this file's other fixtures, which /// stay under "srv1" precisely because they never drive the orphan sweep): `prefixEligible`'s /// watermark floor is looked up by walking the NAMESPACE's own prefix segments for a live mount @@ -411,6 +471,16 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) /// regardless of epoch/coverage. const RootNamespace ns{"test/writer_duty_rejected_sweep"}; + /// The mount fence's own budget is read off the BOOT clock, while the retry window below is read + /// off the virtual one `VirtualRetryClock` installs -- so with a real boot clock the wall time + /// this test spends publishing the anchor and staging the manifest is subtracted from a 500 ms + /// lease, and on a loaded machine the give-up below stops being a retry give-up and becomes a + /// no-budget refusal before the first attempt. Freeze the boot clock, exactly as the successor + /// pool further down already does, so the only bound on that give-up is the one it asserts. + /// Captured by value: `predecessor_boot` is never mutated in this test, and the Pool can outlive + /// this stack frame (a background publish holds `shared_from_this()`), so a by-reference capture + /// would dangle. + const uint64_t predecessor_boot = 0; auto predecessor = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), @@ -422,7 +492,12 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) .mount_lease_ttl_ms = std::chrono::milliseconds(500), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = budget, + .boot_ms_fn = [] + { + return predecessor_boot; + }, }); + auto clock = DB::Cas::tests::VirtualRetryClock::installOn(predecessor); /// A real, fully-promoted ref through the ordinary production write path (no seeded catalog/ckpt) /// gives the namespace genuine epoch-1 content, so the successor's recovery below has something @@ -433,8 +508,11 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) ManifestId rejected_id; auto rejected = stageEmptyManifest(predecessor, ns, "rejected", rejected_id); const String rejected_manifest_key = predecessor->layout().manifestKey(rejected_id); - ASSERT_TRUE(backend->head(rejected_manifest_key).exists) - << "stageManifest's body write is unconditional; only the owner grant is refused below"; + { + DB::Cas::tests::OperationForTest verify_op(*backend); + ASSERT_TRUE((*verify_op).head(rejected_manifest_key, Retry::once()).has_value()) + << "stageManifest's body write is unconditional; only the owner grant is refused below"; + } /// `Unresolved` lands nothing, so the wedge it leaves resolves as a conclusive REJECT once the /// successor's own recovery walks past it -- unlike the ADOPT-arm crash-remnant test, this @@ -444,9 +522,7 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) backend->fault_substr = predecessor->layout().namespaceStreamPrefix(predecessor->namespaceLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::NETWORK_ERROR, - [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); + driveToNetworkErrorGiveUp(*backend, *clock, [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); ASSERT_TRUE(predecessor->refLaneWedgedForTest(ns)); ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); const uint64_t predecessor_epoch = predecessor->writerEpoch(); @@ -454,8 +530,11 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) rejected.reset(); predecessor.reset(); - uint64_t fake_boot = 0; - std::vector waits; + /// Held in shared, heap-owned state, not plain locals: the hooks below mutate them, and the Pool + /// can outlive this stack frame (a background publish holds `shared_from_this()`), so a + /// by-reference capture of a local would dangle. + auto fake_boot = std::make_shared>(0); + auto waits = std::make_shared(); auto successor_store = Pool::open(backend, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), @@ -467,11 +546,18 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) .mount_lease_ttl_ms = std::chrono::milliseconds(500), .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = budget, - .boot_ms_fn = [&] { return fake_boot; }, - .wait_sleep_fn = [&](uint64_t ms) { fake_boot += ms; waits.push_back(ms); }, + .boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }, + .wait_sleep_fn = [fake_boot, waits](uint64_t ms) + { + *fake_boot += ms; + waits->push(ms); + }, }); ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); - ASSERT_FALSE(waits.empty()) << "the predecessor supplied no clean-death certificate"; + ASSERT_FALSE(waits->empty()) << "the predecessor supplied no clean-death certificate"; /// An ordinary successor mutation both drains the inherited duty as a no-op (the rejected grant /// was never durable) and forces the predecessor's dead epoch to close with an arithmetic seal -- @@ -488,10 +574,11 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) EXPECT_EQ(seal->writer_epoch, predecessor_epoch); Gc gc(successor_store, hexToU128("000000000000000000000000000000e1")); - for (int round = 0; round < 16 && backend->head(rejected_manifest_key).exists; ++round) + DB::Cas::tests::OperationForTest sweep_op(*backend); + for (int round = 0; round < 16 && (*sweep_op).head(rejected_manifest_key, Retry::once()).has_value(); ++round) DB::Cas::tests::runRegularRoundReclaiming(gc); - EXPECT_FALSE(backend->head(rejected_manifest_key).exists) + EXPECT_FALSE((*sweep_op).head(rejected_manifest_key, Retry::once()).has_value()) << "the rejected attempt's orphan manifest must eventually be nominated and swept once its " "build epoch is durably closed"; @@ -506,8 +593,9 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) /// very next drain -- once the fault clears -- settles the duty and lets that mutation proceed. TEST(CASWriterDuties, DutySurvivesSettlementFailureForRetry) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openFrozenSingleAttemptPool(backend); + auto clock = DB::Cas::tests::VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/writer_duty_settlement_retry"}; DB::Cas::tests::casAdmitRecoverableEntry(*backend, store->layout(), ns, store->liveWriterEpoch()); publishEmptyRef(store, ns, "target"); @@ -525,9 +613,7 @@ TEST(CASWriterDuties, DutySurvivesSettlementFailureForRetry) backend->fault_substr = store->layout().namespaceStreamPrefix(DB::Cas::tests::fixture::fixtureLife(ns)) + "_log/"; backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::Unresolved; backend->fault_count = 1; - DB::Cas::tests::expectThrowsCode( - DB::ErrorCodes::NETWORK_ERROR, - [&] { store->dropRef(ns, "target"); }); + driveToNetworkErrorGiveUp(*backend, *clock, [&] { store->dropRef(ns, "target"); }); EXPECT_TRUE(store->writerCleanupDutiesPendingForTest()) << "a settlement that throws must retain the duty for retry, never lose it"; @@ -535,7 +621,6 @@ TEST(CASWriterDuties, DutySurvivesSettlementFailureForRetry) << "the settlement's failure must abort the mutation it was blocking too, not just its own append"; EXPECT_EQ(store->minActive(), durable_seq); - backend->mode = DB::Cas::tests::ChunkFaultBackend::Mode::None; store->dropRef(ns, "target"); EXPECT_FALSE(store->writerCleanupDutiesPendingForTest()); diff --git a/src/IO/ObjectStorageRequestMode.h b/src/IO/ObjectStorageRequestMode.h new file mode 100644 index 000000000000..39301ae953a9 --- /dev/null +++ b/src/IO/ObjectStorageRequestMode.h @@ -0,0 +1,18 @@ +#pragma once + +#include + +namespace DB +{ + +/// How a request is issued to an object storage. `NativeConditional` marks a request as carrying (or +/// eligible to carry) a storage-native conditional header, which lets a provider-specific client +/// translate `If-Match`/`ETag` into its own vocabulary (e.g. a GCS generation) on the request and the +/// response. Reads use it so that a plain GET answers with the same incarnation identity a HEAD does. +enum class ObjectStorageRequestMode : uint8_t +{ + Default, + NativeConditional, +}; + +} diff --git a/src/IO/ObjectStorageRequestProfile.h b/src/IO/ObjectStorageRequestProfile.h new file mode 100644 index 000000000000..2b9c7816fdb5 --- /dev/null +++ b/src/IO/ObjectStorageRequestProfile.h @@ -0,0 +1,31 @@ +#pragma once + +#include +#include + +namespace DB +{ + +/// Per-write retry-behavior selector, resolved by the object storage that executes the write. +/// SingleAttempt: exactly one HTTP attempt, no SDK-transparent retries — for conditional writes +/// whose retry loop lives above the storage client (it must resolve an uncertain PUT before +/// reissuing). Backends without a SingleAttempt implementation report it via +/// IObjectStorage::supportsRetryProfile; writers must fail closed rather than fall through. +enum class ObjectStorageRetryProfile : uint8_t +{ + Default, + SingleAttempt, +}; + +/// A per-request override of retry behavior for an object storage call: which retry profile to use, +/// the per-attempt budget and connect cap the storage's single-attempt client must honour, and the +/// caller's own attempt number (0 = unset) so the HTTP client sees a reissue as attempt ≥ 2. +struct ObjectStorageControlRequest +{ + ObjectStorageRetryProfile profile = ObjectStorageRetryProfile::Default; + uint64_t attempt_timeout_ms = 0; + uint64_t connect_timeout_cap_ms = 0; + size_t attempt_number = 0; +}; + +} diff --git a/src/IO/ReadBufferFromS3.cpp b/src/IO/ReadBufferFromS3.cpp index df009a32cabf..505e0b157d0d 100644 --- a/src/IO/ReadBufferFromS3.cpp +++ b/src/IO/ReadBufferFromS3.cpp @@ -215,7 +215,23 @@ bool ReadBufferFromS3::nextImpl() } /// Try to read a next portion of data. - next_result = impl->next(); + const bool delivered_more_data = impl->next(); + if (delivered_more_data && !pending_response_bytes_delivered) + { + /// This response just delivered its first byte: check it against whichever response + /// last delivered bytes, then it becomes the new baseline. A response that never + /// reaches this point (fails before delivering anything) never touches the baseline, + /// so any number of empty failed attempts in between are transparent to the check. + /// + /// Must run before `next_result = delivered_more_data` below exits the loop via + /// `break`: a throw after that point would leave the loop exiting with `impl` null + /// (reset by the catch handler) while the code past the loop still dereferences it. + if (last_delivering_response_etag && *last_delivering_response_etag != pending_response_etag) + response_identity_changed = true; + last_delivering_response_etag = std::move(pending_response_etag); + pending_response_bytes_delivered = true; + } + next_result = delivered_more_data; break; } catch (...) @@ -459,6 +475,7 @@ off_t ReadBufferFromS3::seek(off_t offset_, int whence) if (!atEndOfRequestedRangeGuess()) ProfileEvents::increment(ProfileEvents::ReadBufferSeekCancelConnection); impl.reset(); + forgetResponseIdentityBaseline(); } } @@ -504,6 +521,7 @@ void ReadBufferFromS3::setReadUntilPosition(size_t position) offset = getPosition(); resetWorkingBuffer(); impl.reset(); + forgetResponseIdentityBaseline(); } read_until_position = position; } @@ -523,6 +541,7 @@ void ReadBufferFromS3::setReadUntilEnd() offset = getPosition(); resetWorkingBuffer(); impl.reset(); + forgetResponseIdentityBaseline(); } } } @@ -538,6 +557,12 @@ bool ReadBufferFromS3::atEndOfRequestedRangeGuess() return false; } +void ReadBufferFromS3::forgetResponseIdentityBaseline() +{ + last_delivering_response_etag.reset(); + pending_response_bytes_delivered = false; +} + std::unique_ptr ReadBufferFromS3::initialize(size_t attempt) { stop_reason = ""; @@ -556,6 +581,15 @@ std::unique_ptr ReadBufferFromS3::initialize( Stopwatch watch{CLOCK_MONOTONIC}; auto read_result = sendRequest(attempt, offset, right_offset); + /// Record the new response's identity; the coherence check itself happens in nextImpl(), at the + /// moment this response actually delivers its first byte. Comparing here instead (against + /// whatever the previous attempt's ETag was) would flag a mismatch as soon as a differently-ETagged + /// response is merely attempted, before it is known whether that attempt will ever deliver + /// anything - and would just as easily lose track of an earlier delivering response across an + /// intervening empty failed attempt with yet another ETag. + pending_response_etag = read_result.GetETag(); + pending_response_bytes_delivered = false; + size_t buffer_size = use_external_buffer ? 0 : read_settings.remote_fs_settings.buffer_size; return std::make_unique(std::move(read_result), buffer_size, std::move(watch)); } @@ -573,7 +607,14 @@ Aws::S3::Model::GetObjectResult ReadBufferFromS3::sendRequest(size_t attempt, si else if (!expected_etag.empty()) req.SetIfMatch(expected_etag); +<<<<<<< HEAD S3::setClickHouseAttemptNumber(req, attempt); +======= + S3::setClickhouseAttemptNumber(req, S3::seededAttemptNumber(read_settings.object_storage_attempt_number, attempt)); + + if (read_settings.object_storage_request_mode == ObjectStorageRequestMode::NativeConditional) + req.setNativeConditional(); +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) if (range_end_incl) { diff --git a/src/IO/ReadBufferFromS3.h b/src/IO/ReadBufferFromS3.h index 18c33dcce75e..d99c61f015a7 100644 --- a/src/IO/ReadBufferFromS3.h +++ b/src/IO/ReadBufferFromS3.h @@ -98,6 +98,13 @@ class ReadBufferFromS3 : public ReadBufferFromFileBase /// This method returns metadata from the last request. If there were no requests, it will throw exception. ObjectMetadata getObjectMetadataFromTheLastRequest() const; + /// True when bytes already delivered to the consumer came from a response whose ETag turned out to + /// differ from a later, reissued response's ETag, i.e. the bytes this buffer produced may come from + /// more than one incarnation of the object. A response that never delivered a byte (e.g. the GET + /// succeeded but the body read failed before any data arrived) does not count: reissuing it and + /// getting a different ETag is an ordinary retry, not a coherence problem. + bool responseIdentityChanged() const { return response_identity_changed; } + size_t getReadUntilPosition() const { return read_until_position; } std::string getStopReason() const { return stop_reason; } @@ -118,6 +125,27 @@ class ReadBufferFromS3 : public ReadBufferFromFileBase Aws::S3::Model::GetObjectResult sendRequest(size_t attempt, size_t range_begin, std::optional range_end_incl) const; + /// Drops the identity baseline. Called when the next request is a reissue for a range the caller + /// explicitly repositioned to (seek, or a change of the read-until bound), as opposed to a retry of + /// the same range after a failure: the bytes already delivered before the reposition reached the + /// consumer as their own self-consistent range, so the next response is not compared against them. + void forgetResponseIdentityBaseline(); + + /// ETag of the last response that has delivered at least one byte to the consumer: the baseline a + /// newly-delivering response is checked against. A response that never delivers a byte (e.g. it + /// fails before the body starts) leaves this untouched, however many such empty attempts happen in + /// a row, so the baseline always reflects the last response that actually contributed bytes. + std::optional last_delivering_response_etag; + + /// ETag of the response `impl` currently represents, and whether that response has delivered a byte + /// yet. Both are set together in initialize(); nextImpl() flips `pending_response_bytes_delivered` + /// to true (and advances last_delivering_response_etag) the moment this response's first byte + /// reaches the consumer. + String pending_response_etag; + bool pending_response_bytes_delivered = false; + + bool response_identity_changed = false; + ReadSettings read_settings; bool use_external_buffer; diff --git a/src/IO/ReadSettings.h b/src/IO/ReadSettings.h index 830aea7e598b..08e2d7f8ec6b 100644 --- a/src/IO/ReadSettings.h +++ b/src/IO/ReadSettings.h @@ -6,7 +6,12 @@ #include #if ENABLE_DISTRIBUTED_CACHE #include +<<<<<<< HEAD #endif +======= +#include +#include +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) #include #include #include @@ -182,6 +187,23 @@ struct ReadSettings DistributedCacheSettings distributed_cache_settings; #endif + /// Selects the object storage request mode this read should carry; see ObjectStorageRequestMode. + ObjectStorageRequestMode object_storage_request_mode = ObjectStorageRequestMode::Default; + + /// Selects the retry profile the object storage should execute this read under, and the request + /// timeout of the client it picks for it; see ObjectStorageRetryProfile. 0 = the storage's own. + ObjectStorageRetryProfile object_storage_retry_profile = ObjectStorageRetryProfile::Default; + uint64_t object_storage_attempt_timeout_ms = 0; + + /// The cap the single-attempt client's clone puts on one TCP connect and again on one TLS + /// handshake, frozen by the mount at open; see `CasRequestBudget::attemptEnvelopeMs`. 0 = no cap. + uint64_t object_storage_connect_timeout_cap_ms = 0; + + /// The caller's own attempt number for the request built from these settings, 1-based; 0 leaves the + /// buffer's own numbering. A caller reissuing this read passes its count so the HTTP client sees + /// attempt ≥ 2. + size_t object_storage_attempt_number = 0; + ReadSettings adjustBufferSize(size_t file_size) const; /// Verification/metadata-read mode: disable every read-side cache (and the diff --git a/src/IO/S3/Client.cpp b/src/IO/S3/Client.cpp index d8aa66d4873d..17b3cd249357 100644 --- a/src/IO/S3/Client.cpp +++ b/src/IO/S3/Client.cpp @@ -201,6 +201,11 @@ bool SingleAttemptRetryStrategy::ShouldRetry(const Aws::Client::AWSError(client_configuration.retryStrategy.get()) != nullptr; +} + namespace { @@ -828,7 +833,12 @@ Client::doRequestWithRetryNetworkErrors(RequestType & request, RequestFn request if (isClientForDisk()) incrementProfileEvents(ProfileEvents::DiskS3ReadRequestsErrors, ProfileEvents::DiskS3WriteRequestsErrors); - tryLogCurrentException(log, fmt::format("Network error on S3 request, attempt {} of {}", attempt_no, max_attempts)); + /// A client with the single-attempt strategy is owned by an outer retry loop that resolves + /// the outcome and reissues; its one failed attempt is not terminal, so it is not an error. + if (usesSingleAttemptRetryStrategy()) + LOG_DEBUG(log, "Network error on S3 request, attempt {} of {}: {}", attempt_no, max_attempts, getCurrentExceptionMessage(/*with_stacktrace=*/false)); + else + tryLogCurrentException(log, fmt::format("Network error on S3 request, attempt {} of {}", attempt_no, max_attempts)); outcome = Aws::Client::AWSError( Aws::Client::CoreErrors::NETWORK_CONNECTION, diff --git a/src/IO/S3/Client.h b/src/IO/S3/Client.h index a53e981b1382..8290eb0a61ea 100644 --- a/src/IO/S3/Client.h +++ b/src/IO/S3/Client.h @@ -234,6 +234,11 @@ class Client : private Aws::S3::S3Client using Aws::S3::S3Client::EnableRequestProcessing; using Aws::S3::S3Client::DisableRequestProcessing; + /// Test-only: lets a gtest observe whether Enable/DisableRequestProcessing last took effect on this + /// client's own `Aws::Http::HttpClient`, without exposing the rest of the privately-inherited + /// `Aws::S3::S3Client` surface. Production code reaches `GetHttpClient` directly (private + /// inheritance already permits that from this class's own methods) and has no need of this `using`. + using Aws::S3::S3Client::GetHttpClient; void BuildHttpRequest(const Aws::AmazonWebServiceRequest& request, const std::shared_ptr& httpRequest) const override; @@ -247,6 +252,10 @@ class Client : private Aws::S3::S3Client return client_configuration.for_disk_s3; } + /// True when this client's one and only attempt is not the final answer: it belongs to an + /// outer retry loop (e.g. a conditional write) that resolves the outcome and reissues. + bool usesSingleAttemptRetryStrategy() const; + ProviderType getProviderType() const { return provider_type; } std::string getGCSOAuthToken() const; diff --git a/src/IO/S3/Requests.h b/src/IO/S3/Requests.h index d4d2f3daca85..0e326141cc21 100644 --- a/src/IO/S3/Requests.h +++ b/src/IO/S3/Requests.h @@ -265,6 +265,13 @@ size_t getClickHouseAttemptNumber(const Aws::AmazonWebServiceRequest & request); size_t getClickHouseAttemptNumber(const Aws::Http::HttpRequest & request); void setClickHouseAttemptNumber(Aws::AmazonWebServiceRequest & request, size_t attempt); +/// The attempt number a request carries when its caller seeded one: the caller's attempt for the first +/// local try, then the local counter's increments. Seed 0 is "unseeded" and yields `local`. +inline size_t seededAttemptNumber(size_t seed, size_t local) +{ + return (seed == 0 ? 1 : seed) + local - 1; +} + } #endif diff --git a/src/IO/S3/deleteFileFromS3.cpp b/src/IO/S3/deleteFileFromS3.cpp index 380747f33156..0111e00534a8 100644 --- a/src/IO/S3/deleteFileFromS3.cpp +++ b/src/IO/S3/deleteFileFromS3.cpp @@ -28,11 +28,14 @@ void deleteFileFromS3( BlobStorageLogWriterPtr blob_storage_log, const String & local_path_for_blob_storage_log, size_t file_size_for_blob_storage_log, - std::optional profile_event) + std::optional profile_event, + size_t attempt_seed) { S3::DeleteObjectRequest request; request.SetBucket(bucket); request.SetKey(key); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(request, attempt_seed); ProfileEvents::increment(ProfileEvents::S3DeleteObjects); if (profile_event && *profile_event != ProfileEvents::S3DeleteObjects) diff --git a/src/IO/S3/deleteFileFromS3.h b/src/IO/S3/deleteFileFromS3.h index fad49982827e..69e0f409c768 100644 --- a/src/IO/S3/deleteFileFromS3.h +++ b/src/IO/S3/deleteFileFromS3.h @@ -32,7 +32,8 @@ void deleteFileFromS3( BlobStorageLogWriterPtr blob_storage_log = nullptr, const String & local_path_for_blob_storage_log = {}, size_t file_size_for_blob_storage_log = 0, - std::optional profile_event = std::nullopt); + std::optional profile_event = std::nullopt, + size_t attempt_seed = 0); /// Deletes multiple files from S3 using batch requests when it's possible. void deleteFilesFromS3( diff --git a/src/IO/S3/getObjectInfo.cpp b/src/IO/S3/getObjectInfo.cpp index 63aeb736960c..5bd1b4cf52e1 100644 --- a/src/IO/S3/getObjectInfo.cpp +++ b/src/IO/S3/getObjectInfo.cpp @@ -25,7 +25,8 @@ namespace const String & bucket, const String & key, const String & version_id, - ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default) + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default, + size_t attempt_seed = 0) { ProfileEvents::increment(ProfileEvents::S3HeadObject); if (client.isClientForDisk()) @@ -41,6 +42,9 @@ namespace req.setNativeConditional(request_mode == ObjectStorageRequestMode::NativeConditional); + if (attempt_seed != 0) + S3::setClickhouseAttemptNumber(req, attempt_seed); + return client.HeadObject(req); } @@ -71,9 +75,10 @@ namespace const String & version_id, bool with_metadata, bool with_tags, - ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default) + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default, + size_t attempt_seed = 0) { - auto outcome = headObject(client, bucket, key, version_id, request_mode); + auto outcome = headObject(client, bucket, key, version_id, request_mode, attempt_seed); if (!outcome.IsSuccess()) return {std::nullopt, outcome.GetError()}; @@ -146,11 +151,12 @@ ObjectInfo getObjectInfoIfExists( const String & version_id, bool with_metadata, bool with_tags, - ObjectStorageRequestMode request_mode) + ObjectStorageRequestMode request_mode, + size_t attempt_seed) { Expect404ResponseScope scope; // 404 is not an error - auto [object_info, error] = tryGetObjectInfo(client, bucket, key, version_id, with_metadata, with_tags, request_mode); + auto [object_info, error] = tryGetObjectInfo(client, bucket, key, version_id, with_metadata, with_tags, request_mode, attempt_seed); if (object_info) return *object_info; diff --git a/src/IO/S3/getObjectInfo.h b/src/IO/S3/getObjectInfo.h index 60683cbf3abf..d3a31260f473 100644 --- a/src/IO/S3/getObjectInfo.h +++ b/src/IO/S3/getObjectInfo.h @@ -25,6 +25,8 @@ struct ObjectInfo /// Ignore if object does not exist /// `request_mode` marks the HEAD wrapper as eligible for the typed NativeConditional request mode /// (see ObjectStorageRequestMode); the client's HTTP layer decides whether it actually takes effect. +/// `attempt_seed`, when nonzero, is set as the HEAD's `clickhouse-request` attempt number (see +/// `S3::seededAttemptNumber`); 0 leaves the request unseeded. ObjectInfo getObjectInfoIfExists( const S3::Client & client, const String & bucket, @@ -32,7 +34,8 @@ ObjectInfo getObjectInfoIfExists( const String & version_id = {}, bool with_metadata = false, bool with_tags = false, - ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default); + ObjectStorageRequestMode request_mode = ObjectStorageRequestMode::Default, + size_t attempt_seed = 0); ObjectInfo getObjectInfo( const S3::Client & client, diff --git a/src/IO/S3/tests/TestPocoHTTPServer.h b/src/IO/S3/tests/TestPocoHTTPServer.h index 9d8066a1a0c0..548868039922 100644 --- a/src/IO/S3/tests/TestPocoHTTPServer.h +++ b/src/IO/S3/tests/TestPocoHTTPServer.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include @@ -16,6 +17,7 @@ #include #include #include +#include #include /// Keep-alive is disabled so a handler thread exits right after sending the response: it does @@ -71,6 +73,11 @@ class TestPocoHTTPServer std::unique_ptr server_socket; Poco::SharedPtr handler_factory; Poco::AutoPtr server_params; + /// A dedicated pool, not `Poco::ThreadPool::defaultPool()` (the `HTTPServer` default): that pool + /// is shared with every other local-server test in this binary, and `TCPServerDispatcher::enqueue` + /// (base/poco/Net/src/TCPServerDispatcher.cpp) has an acknowledged-in-comment saturation-check race + /// when it's shared, which can accept a connection and then close it with no response. + Poco::ThreadPool thread_pool; std::unique_ptr server; // Stores the last request header handled. It's obviously not thread-safe to share the same // reference across request handlers, but it's good enough for this the purposes of this test. @@ -80,20 +87,41 @@ class TestPocoHTTPServer TestPocoHTTPServer(): server_socket(std::make_unique(0)), handler_factory(new HTTPRequestHandlerFactory(last_request_header)), +<<<<<<< HEAD server_params(makeMockServerParams()), server(std::make_unique(handler_factory, *server_socket, server_params)) +======= + server_params(new Poco::Net::HTTPServerParams()), + thread_pool("TestPocoHTTPServer"), + server(std::make_unique(handler_factory, thread_pool, *server_socket, server_params)) +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) { server->start(); } +<<<<<<< HEAD ~TestPocoHTTPServer() { server->stop(); } +======= + /// Closing the cached client sockets wakes the server workers without Poco's abort notification, + /// whose unlocked socket shutdown races the worker's own close. Precondition: callers have released + /// their sessions, otherwise `joinAll` waits for the server's request timeout. + ~TestPocoHTTPServer() + { + DB::HTTPConnectionPools::instance().dropCache(); + server->stop(); + thread_pool.joinAll(); + } + + /// `server_socket->address()` is the wildcard bind address (`0.0.0.0:PORT`), which is not a usable + /// connection target. Build the URL from an explicit loopback address plus the bound port instead. +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) std::string getUrl() { - return "http://" + server_socket->address().toString(); + return "http://127.0.0.1:" + std::to_string(server_socket->address().port()); } const Poco::Net::MessageHeader & getLastRequestHeader() const @@ -174,6 +202,9 @@ class TestPocoHTTPStsServer std::unique_ptr server_socket; Poco::SharedPtr handler_factory; Poco::AutoPtr server_params; + /// See the identical member in `TestPocoHTTPServer` above: a private pool avoids + /// `TCPServerDispatcher`'s shared-pool saturation bug (base/poco/Net/src/TCPServerDispatcher.cpp). + Poco::ThreadPool thread_pool; std::unique_ptr server; // Stores the last request header handled. It's obviously not thread-safe to share the same // reference across request handlers, but it's good enough for this the purposes of this test. @@ -183,20 +214,39 @@ class TestPocoHTTPStsServer TestPocoHTTPStsServer(std::string role_access_key, std::string role_secret_key): server_socket(std::make_unique(0)), handler_factory(new StsHTTPRequestHandlerFactory(last_request_info, std::move(role_access_key), std::move(role_secret_key))), +<<<<<<< HEAD server_params(makeMockServerParams()), server(std::make_unique(handler_factory, *server_socket, server_params)) +======= + server_params(new Poco::Net::HTTPServerParams()), + thread_pool("TestPocoHTTPStsServer"), + server(std::make_unique(handler_factory, thread_pool, *server_socket, server_params)) +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) { server->start(); } +<<<<<<< HEAD + ~TestPocoHTTPStsServer() + { + server->stop(); + } + +======= + /// See `TestPocoHTTPServer`'s destructor above. ~TestPocoHTTPStsServer() { + DB::HTTPConnectionPools::instance().dropCache(); server->stop(); + thread_pool.joinAll(); } + /// `server_socket->address()` is the wildcard bind address (`0.0.0.0:PORT`), which is not a usable + /// connection target. Build the URL from an explicit loopback address plus the bound port instead. +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) std::string getUrl() { - return "http://" + server_socket->address().toString(); + return "http://127.0.0.1:" + std::to_string(server_socket->address().port()); } void resetLastRequest() diff --git a/src/IO/S3/tests/gtest_aws_s3_client.cpp b/src/IO/S3/tests/gtest_aws_s3_client.cpp index 61edec9e1ceb..cb69c65aa549 100644 --- a/src/IO/S3/tests/gtest_aws_s3_client.cpp +++ b/src/IO/S3/tests/gtest_aws_s3_client.cpp @@ -1,3 +1,4 @@ +#include #include #include @@ -18,6 +19,7 @@ #include #include +#include #include #include @@ -1057,11 +1059,22 @@ class ScriptedResponseServer , server_socket(std::make_unique(0)) , handler_factory(new Factory(*this)) , server_params(new Poco::Net::HTTPServerParams()) - , server(std::make_unique(handler_factory, *server_socket, server_params)) + , thread_pool("ScriptedResponseServer") + , server(std::make_unique(handler_factory, thread_pool, *server_socket, server_params)) { server->start(); } + /// Closing the cached client sockets wakes the server workers without Poco's abort notification, + /// whose unlocked socket shutdown races the worker's own close. Precondition: callers have released + /// their sessions, otherwise `joinAll` waits for the server's request timeout. + ~ScriptedResponseServer() + { + DB::HTTPConnectionPools::instance().dropCache(); + server->stop(); + thread_pool.joinAll(); + } + /// `server_socket->address()` is the wildcard bind address (`0.0.0.0:PORT`), which is not a usable /// connection target and could silently conflate distinct servers under the same host string. Build /// the URL from an explicit loopback address plus the bound port instead. @@ -1116,6 +1129,11 @@ class ScriptedResponseServer std::unique_ptr server_socket; Poco::SharedPtr handler_factory; Poco::AutoPtr server_params; + /// A dedicated pool, not `Poco::ThreadPool::defaultPool()` (the `HTTPServer` default): that pool + /// is shared with every other local-server test in this binary, and `TCPServerDispatcher::enqueue` + /// (base/poco/Net/src/TCPServerDispatcher.cpp) has an acknowledged-in-comment saturation-check race + /// when it's shared, which can accept a connection and then close it with no response. + Poco::ThreadPool thread_pool; std::unique_ptr server; }; diff --git a/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp b/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp new file mode 100644 index 000000000000..8ca6c6c79b8c --- /dev/null +++ b/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp @@ -0,0 +1,481 @@ +#include +#include + +#include +#include "config.h" + + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::S3RequestSetting +{ + extern const S3RequestSettingsUInt64 max_single_read_retries; +} + +namespace ProfileEvents +{ + extern const Event S3WriteRequestsErrors; +} + +/// Parses the `attempt=N` value `S3::setClickhouseAttemptNumber` writes into the `clickhouse-request` +/// header, straight off the wire header a real HTTP server received -- mirrors +/// `S3::getAttemptFromInfo`/`getOrEmpty` (both `static` in `Requests.cpp`, not exported), 1 when the +/// header is missing. +static size_t attemptFromHeader(const Poco::Net::MessageHeader & header) +{ + const std::string & value = header.get("clickhouse-request", ""); + static const std::string key = "attempt="; + auto pos = value.find(key); + if (pos == std::string::npos) + return 1; + try + { + return static_cast(std::stol(value.substr(pos + key.size()))); + } + catch (const std::exception &) + { + return 1; + } +} + +static std::shared_ptr makeTestClient(const DB::S3::URI & uri) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration client_configuration = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /*s3_max_redirects=*/100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /*s3_slow_all_threads_after_network_error=*/false, + /*s3_slow_all_threads_after_retryable_error=*/false, + /*enable_s3_requests_logging=*/false, + /*for_disk_s3=*/false, + /*opt_disk_name=*/{}, + /*request_throttler=*/{}, + uri.uri.getScheme()); + /// Fresh connection per request: this file's servers live on ephemeral ports and die with the test, + /// and a pooled keep-alive connection can outlive its server (`Connection reset by peer` under `--gtest_repeat`). + client_configuration.http_keep_alive_timeout = 0; + client_configuration.endpointOverride = uri.endpoint; + /// `ClientFactory::create` installs the SDK's actual retry strategy itself from + /// `client_configuration.retry_strategy`/`s3_slow_all_threads_after_retryable_error` (any + /// `retryStrategy` set here is overwritten) -- with `s3_slow_all_threads_after_retryable_error` + /// true it forces `max_retries = 1` regardless of the `RetryStrategy{.max_retries = 0}` passed + /// above, so the SDK itself retries a retryable error once before `ReadBufferFromS3`'s own + /// local-retry loop ever sees a failure, and both physical requests carry the same seeded header. + /// `false` here keeps the SDK to exactly one physical attempt, matching the CAS single-attempt + /// client's own setup. + + DB::S3::ClientSettings client_settings{ + .use_virtual_addressing = uri.is_virtual_hosted_style, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; + + return DB::S3::ClientFactory::instance().create( + client_configuration, + client_settings, + "ACCESS_KEY_ID", + "SECRET_ACCESS_KEY", + /*server_side_encryption_customer_key_base64=*/"", + DB::S3::ServerSideEncryptionKMSConfig(), + DB::HTTPHeaderEntries(), + DB::S3::CredentialsConfiguration{ + .use_environment_credentials = false, + .use_insecure_imds_request = false, + }); +} + +/// Anonymous namespace: these three classes have no counterpart in gtest_aws_s3_client.cpp today, but +/// giving them internal linkage costs nothing and avoids ever silently colliding with a same-named +/// class that file adds later (see the equivalent note in gtest_cas_readbuffer_s3.cpp for what such a +/// collision actually does at link time). +namespace +{ + +/// Fails the first `fail_first_n` requests with `fail_status` (empty body), then serves `body` with a +/// 200 to every request after. Records every request's header (not just the last) so a caller can +/// check the sequence a local retry produced. +class SequenceRecordingRequestHandler : public Poco::Net::HTTPRequestHandler +{ + std::vector & all_request_headers; + size_t & requests_seen; + size_t fail_first_n; + Poco::Net::HTTPResponse::HTTPStatus fail_status; + std::string body; + +public: + SequenceRecordingRequestHandler( + std::vector & all_request_headers_, + size_t & requests_seen_, + size_t fail_first_n_, + Poco::Net::HTTPResponse::HTTPStatus fail_status_, + std::string body_) + : all_request_headers(all_request_headers_) + , requests_seen(requests_seen_) + , fail_first_n(fail_first_n_) + , fail_status(fail_status_) + , body(std::move(body_)) + { + } + + void handleRequest(Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) override + { + all_request_headers.push_back(request); + ++requests_seen; + + if (requests_seen <= fail_first_n) + { + response.setStatus(fail_status); + response.send(); + return; + } + + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.setContentLength(static_cast(body.size())); + auto & out = response.send(); + out << body; + out.flush(); + } +}; + +class SequenceRecordingRequestHandlerFactory : public Poco::Net::HTTPRequestHandlerFactory +{ + std::vector & all_request_headers; + size_t & requests_seen; + size_t fail_first_n; + Poco::Net::HTTPResponse::HTTPStatus fail_status; + std::string body; + + Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override + { + return new SequenceRecordingRequestHandler(all_request_headers, requests_seen, fail_first_n, fail_status, body); + } + +public: + SequenceRecordingRequestHandlerFactory( + std::vector & all_request_headers_, + size_t & requests_seen_, + size_t fail_first_n_, + Poco::Net::HTTPResponse::HTTPStatus fail_status_, + std::string body_) + : all_request_headers(all_request_headers_) + , requests_seen(requests_seen_) + , fail_first_n(fail_first_n_) + , fail_status(fail_status_) + , body(std::move(body_)) + { + } + + ~SequenceRecordingRequestHandlerFactory() override = default; +}; + +/// Like `TestPocoHTTPServer`, but for driving a real local retry: the first `fail_first_n` requests +/// get `fail_status`, every one after gets `body` with a 200, and every request's header is kept (not +/// just the last). Its only user is the seed test right below -- localized here rather than in the +/// shared `TestPocoHTTPServer.h` header. +class TestPocoHTTPSequenceServer +{ + std::unique_ptr server_socket; + Poco::SharedPtr handler_factory; + Poco::AutoPtr server_params; + /// A dedicated pool, not `Poco::ThreadPool::defaultPool()` (the `HTTPServer` default): that pool + /// is shared with every other local-server test in this binary, and `TCPServerDispatcher::enqueue` + /// (base/poco/Net/src/TCPServerDispatcher.cpp) has an acknowledged-in-comment saturation-check race + /// when it's shared, which can accept a connection and then close it with no response. + Poco::ThreadPool thread_pool; + std::unique_ptr server; + std::vector all_request_headers; + size_t requests_seen = 0; + +public: + TestPocoHTTPSequenceServer(size_t fail_first_n, Poco::Net::HTTPResponse::HTTPStatus fail_status, std::string body = {}): + /// Bind to the loopback address explicitly, not `ServerSocket(0)`'s wildcard `0.0.0.0`: the + /// latter is not a valid connection target, even though the kernel happens to tolerate a + /// connect() to it as loopback on Linux. + server_socket(std::make_unique(Poco::Net::SocketAddress("127.0.0.1", 0))), + handler_factory(new SequenceRecordingRequestHandlerFactory(all_request_headers, requests_seen, fail_first_n, fail_status, std::move(body))), + server_params(new Poco::Net::HTTPServerParams()), + thread_pool("TestPocoHTTPSequenceServer"), + server(std::make_unique(handler_factory, thread_pool, *server_socket, server_params)) + { + server->start(); + } + + /// Closing the cached client sockets wakes the server workers without Poco's abort notification, + /// whose unlocked socket shutdown races the worker's own close. Precondition: callers have released + /// their sessions, otherwise `joinAll` waits for the server's request timeout. + ~TestPocoHTTPSequenceServer() + { + DB::HTTPConnectionPools::instance().dropCache(); + server->stop(); + thread_pool.joinAll(); + } + + std::string getUrl() + { + return "http://" + server_socket->address().toString(); + } + + const std::vector & getAllRequestHeaders() const + { + return all_request_headers; + } +}; + +} + +/// An unset seed sends `[1, 2]` across a local retry, a seed of 2 sends `[2, 3]` -- a real HTTP round +/// trip through `TestPocoHTTPSequenceServer` is the only way to drive the retry through +/// `ReadBufferFromS3`'s actual success path (the SDK's response stream wraps a real +/// `Poco::Net::HTTPBasicStreamBuf`, which `ReadBufferFromIStream` requires). +TEST(CASIOTestAwsS3Client, ReadBufferFromS3AttemptSeedCarriesAcrossLocalRetry) +{ + for (const auto [seed, first, second] : {std::tuple{0, 1, 2}, {2, 2, 3}}) + { + TestPocoHTTPSequenceServer http(/*fail_first_n=*/1, Poco::Net::HTTPResponse::HTTP_INTERNAL_SERVER_ERROR, "seeded-body"); + DB::S3::URI uri(http.getUrl() + "/seeded-bucket/seeded-key"); + auto client = makeTestClient(uri); + ASSERT_TRUE(client); + + DB::ReadSettings read_settings; + read_settings.object_storage_attempt_number = seed; + DB::S3::S3RequestSettings request_settings; + request_settings[DB::S3RequestSetting::max_single_read_retries] = 2; + DB::ReadBufferFromS3 read_buffer(client, uri.bucket, uri.key, /*version_id=*/{}, request_settings, read_settings); + + String content; + DB::readStringUntilEOF(content, read_buffer); + EXPECT_EQ(content, "seeded-body"); + + const auto & headers = http.getAllRequestHeaders(); + ASSERT_EQ(headers.size(), 2u); + EXPECT_EQ(attemptFromHeader(headers[0]), first); + EXPECT_EQ(attemptFromHeader(headers[1]), second); + } +} + +namespace +{ + +/// Captures what the `S3Client` logger (`Client::log`) writes at ERROR and above. A message logged +/// below Error (e.g. Debug) never reaches the channel at this threshold, so an empty capture proves +/// the site logged below Error rather than merely that this particular text was absent. +class ScopedS3ClientErrorLogCapture +{ +public: + ScopedS3ClientErrorLogCapture() + : logger(getLogger("S3Client")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("error"); + } + + ~ScopedS3ClientErrorLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + std::string captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; + Poco::AutoPtr channel; + /// `shared=true` is load-bearing: `AutoPtr(ptr)` would steal a reference the fixture never owned. + Poco::AutoPtr old_channel; + int old_level; +}; + +/// A `Client` whose `PutObject` always fails as though the connection dropped while the response body +/// was being read -- the scenario `Client::doRequestWithRetryNetworkErrors`'s `net_exception_handler` +/// exists for (the comment on that function: "network error happens when XML document is being read +/// from the response body"). Throwing here, through the same virtual `Aws::S3::S3Client::PutObject` +/// slot `Client::PutObject`'s retry loop calls, reaches `net_exception_handler` exactly as a genuine +/// mid-body network failure would, without adding a test seam to production code -- the protected +/// `Client` constructor is already exposed "for testing" (see `RecordingClient` in `gtest_aws_s3_client.cpp`). +class NetworkFailingClient : public DB::S3::Client +{ +public: + NetworkFailingClient( + size_t max_redirects_, + DB::S3::ServerSideEncryptionKMSConfig sse_kms_config_, + const std::shared_ptr & credentials_provider_, + const DB::S3::PocoHTTPClientConfiguration & client_configuration_, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy sign_payloads_, + const DB::S3::ClientSettings & client_settings_) + : DB::S3::Client(max_redirects_, std::move(sse_kms_config_), credentials_provider_, client_configuration_, sign_payloads_, client_settings_) + { + } + + Aws::S3::Model::PutObjectOutcome PutObject(const Aws::S3::Model::PutObjectRequest &) const override + { + ++attempts; + throw Poco::TimeoutException("mock timeout reading the response body"); + } + + mutable size_t attempts = 0; +}; + +std::shared_ptr makeNetworkFailingClient(std::shared_ptr retry_strategy) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration client_configuration = DB::S3::ClientFactory::instance().createClientConfiguration( + /*force_region=*/"us-east-1", + remote_host_filter, + /*s3_max_redirects=*/100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /*s3_slow_all_threads_after_network_error=*/false, + /*s3_slow_all_threads_after_retryable_error=*/false, + /*enable_s3_requests_logging=*/false, + /*for_disk_s3=*/false, + /*opt_disk_name=*/{}, + /*request_throttler=*/{}); + /// `PutObject` never reaches the wire (it is overridden below), so the endpoint is irrelevant -- + /// only the installed retry strategy, which is what `usesSingleAttemptRetryStrategy` inspects. + client_configuration.retryStrategy = std::move(retry_strategy); + + DB::S3::ClientSettings client_settings{ + .use_virtual_addressing = true, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }; + + Aws::Auth::AWSCredentials credentials("ACCESS_KEY_ID", "SECRET_ACCESS_KEY"); + auto credentials_provider = DB::S3::getCredentialsProvider( + client_configuration, + credentials, + DB::S3::CredentialsConfiguration{.use_environment_credentials = false, .use_insecure_imds_request = false}); + + return std::make_shared( + /*max_redirects_=*/100, + DB::S3::ServerSideEncryptionKMSConfig{}, + credentials_provider, + client_configuration, + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + client_settings); +} + +} + +/// The client's own retry strategy still says "do not retry" (`SingleAttemptRetryStrategy::ShouldRetry` +/// always false, tested directly against `DoesNotRetryPreconditionFailed`/`SingleAttemptRetryStrategyRefusesAndCounts` +/// in `gtest_aws_s3_client.cpp`); `usesSingleAttemptRetryStrategy` is a separate, purely descriptive +/// check of which strategy is installed, tested directly here. +TEST(CASIOTestAwsS3Client, UsesSingleAttemptRetryStrategyIdentifiesTheInstalledStrategy) +{ + auto single_attempt_client = makeNetworkFailingClient(std::make_shared()); + EXPECT_TRUE(single_attempt_client->usesSingleAttemptRetryStrategy()); + + DB::S3::PocoHTTPClientConfiguration::RetryStrategy zero_retries{.max_retries = 0}; + auto ordinary_client = makeNetworkFailingClient(std::make_shared(zero_retries)); + EXPECT_FALSE(ordinary_client->usesSingleAttemptRetryStrategy()); +} + +/// A client carrying the `SingleAttemptRetryStrategy` (the CAS conditional-write client, see +/// `S3ObjectStorage::getSingleAttemptClient`) is owned by an outer retry loop that resolves the outcome +/// and reissues; its one failed attempt is not terminal, so the network-error log site must not reach +/// Error. +TEST(CASIOTestAwsS3Client, NetworkErrorLogsDebugForSingleAttemptStrategy) +{ + using ProfileEvents::global_counters; + const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors].load(); + + auto client = makeNetworkFailingClient(std::make_shared()); + DB::S3::PutObjectRequest request; + + /// Call through the `DB::S3::Client&` interface, exactly as production code (which only ever + /// holds a `Client`, never `NetworkFailingClient`) does: `NetworkFailingClient::PutObject` hides + /// `Client::PutObject(PutObjectRequest&)` -- the retry-loop wrapper under test -- from lookup on + /// the derived type, so calling through the base is what makes this test exercise that wrapper + /// rather than the override directly. + const DB::S3::Client & base_client = *client; + ScopedS3ClientErrorLogCapture log_capture; + const auto outcome = base_client.PutObject(request); + + EXPECT_FALSE(outcome.IsSuccess()); + EXPECT_EQ(outcome.GetError().GetErrorType(), Aws::S3::S3Errors::NETWORK_CONNECTION); + EXPECT_EQ(client->attempts, 1u); + EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors].load() - errors_before, 1u); + EXPECT_TRUE(log_capture.captured().empty()); +} + +/// `max_retries = 0` on the ORDINARY strategy is a supported user configuration (`s3_retry_attempts`) +/// with no outer retry loop: its one failed attempt IS the final answer, so it must keep logging at +/// Error -- this is exactly the case a signal keyed on `max_retries == 0` alone would misclassify. +TEST(CASIOTestAwsS3Client, NetworkErrorLogsErrorForOrdinaryZeroRetryStrategy) +{ + using ProfileEvents::global_counters; + const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors].load(); + + DB::S3::PocoHTTPClientConfiguration::RetryStrategy zero_retries{.max_retries = 0}; + auto client = makeNetworkFailingClient(std::make_shared(zero_retries)); + DB::S3::PutObjectRequest request; + + /// See the comment in `NetworkErrorLogsDebugForSingleAttemptStrategy`: calling through the base + /// is what reaches `Client::PutObject`'s retry-loop wrapper rather than the override directly. + const DB::S3::Client & base_client = *client; + ScopedS3ClientErrorLogCapture log_capture; + const auto outcome = base_client.PutObject(request); + + EXPECT_FALSE(outcome.IsSuccess()); + EXPECT_EQ(outcome.GetError().GetErrorType(), Aws::S3::S3Errors::NETWORK_CONNECTION); + EXPECT_EQ(client->attempts, 1u); + EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors].load() - errors_before, 1u); + EXPECT_NE(log_capture.captured().find("Network error on S3 request, attempt 1 of 1"), std::string::npos); +} + +#endif diff --git a/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp b/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp index 348ceaf005d6..f2dabef765d8 100644 --- a/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp +++ b/src/IO/S3/tests/gtest_gcs_conditional_dialect.cpp @@ -74,8 +74,8 @@ TEST(GCSConditionalDialect, IfMatchUnquotedDigitsAlsoAccepted) TEST(GCSConditionalDialect, NonNumericIfMatchThrows) { /// CORRUPTED_DATA, not a broken internal invariant: the value can come from a persisted manifest - /// token or from a storage HEAD whose response carried no generation, and `mintingTypeMatches` - /// upstream only compares the token KIND, never the shape of its value. + /// etag or from a storage HEAD whose response carried no generation, and nothing upstream + /// validates the shape of that value before it reaches this function. auto r = makeRequest(); r.SetHeaderValue("if-match", "\"6654c734ccab8f440ff0825eb443dc7f\""); EXPECT_THROW(applyGcsConditionalDialectToRequest(r), DB::Exception); diff --git a/src/IO/WriteBufferFromS3.cpp b/src/IO/WriteBufferFromS3.cpp index 31fbe6172098..a6a4c0ca3915 100644 --- a/src/IO/WriteBufferFromS3.cpp +++ b/src/IO/WriteBufferFromS3.cpp @@ -753,6 +753,9 @@ S3::PutObjectRequest WriteBufferFromS3::getPutRequest(PartData & data) /// If we don't do it, AWS SDK can mistakenly set it to application/xml, see https://github.com/aws/aws-sdk-cpp/issues/1840 req.SetContentType("binary/octet-stream"); + if (write_settings.object_storage_attempt_number != 0) + S3::setClickhouseAttemptNumber(req, write_settings.object_storage_attempt_number); + client_ptr->setKMSHeaders(req); /// The actual PUT that produces a CAS incarnation token: eligible for the typed NativeConditional @@ -813,10 +816,13 @@ void WriteBufferFromS3::makeSinglepartUpload(WriteBufferFromS3::PartData && data } else { - /// PreconditionFailed is an expected response for conditional writes (e.g. If-None-Match: *), - /// not a genuine error — the caller handles it (see `S3::isPreconditionFailedError`). - if (S3::isPreconditionFailedError(outcome.GetError())) - LOG_INFO(log, "S3Exception name {}, Message: {}, bucket {}, key {}, object size {}", + /// Neither says anything to the operator: PreconditionFailed is an expected response for + /// conditional writes (e.g. If-None-Match: *), handled by the caller (see + /// `S3::isPreconditionFailedError`); a SingleAttempt write is owned by an outer retry loop + /// that resolves the outcome and reissues, so its one failed attempt is not terminal either. + if (S3::isPreconditionFailedError(outcome.GetError()) + || write_settings.object_storage_retry_profile == ObjectStorageRetryProfile::SingleAttempt) + LOG_DEBUG(log, "S3Exception name {}, Message: {}, bucket {}, key {}, object size {}", outcome.GetError().GetExceptionName(), outcome.GetError().GetMessage(), bucket, key, content_length); else LOG_ERROR(log, "S3Exception name {}, Message: {}, bucket {}, key {}, object size {}", diff --git a/src/IO/WriteSettings.h b/src/IO/WriteSettings.h index 19aae46676f3..22fec83564c8 100644 --- a/src/IO/WriteSettings.h +++ b/src/IO/WriteSettings.h @@ -5,24 +5,18 @@ #include #if ENABLE_DISTRIBUTED_CACHE #include +<<<<<<< HEAD #endif +======= +#include +#include +>>>>>>> c3ae984b1ba (Merge pull request #2300 from Altinity/feature/antalya-26.6/CAS-improvements) #include namespace DB { -/// Per-write retry-behavior selector, resolved by the object storage that executes the write. -/// SingleAttempt: exactly one HTTP attempt, no SDK-transparent retries — for conditional writes -/// whose retry loop lives above the storage client (it must resolve an uncertain PUT before -/// reissuing). Backends without a SingleAttempt implementation report it via -/// IObjectStorage::supportsRetryProfile; writers must fail closed rather than fall through. -enum class ObjectStorageRetryProfile : uint8_t -{ - Default, - SingleAttempt, -}; - /// Per-copy transport requirement, resolved by the object storage that executes the copy. /// `NativeOnly` requires a provider-native same-store copy and forbids a client-side fallback. enum class ObjectStorageCopyMode : uint8_t @@ -31,16 +25,6 @@ enum class ObjectStorageCopyMode : uint8_t NativeOnly, }; -/// Per-request GCS conditional-dialect opt-in, carried alongside the write itself so it survives -/// into the object storage request that ends up on the wire (see `RequestWithNativeConditionalMode`). -/// NativeConditional: this write is content-addressed-storage-owned and may use GCS generation -/// tokens instead of the AWS-style ETag plumbing, when the client's HTTP layer supports it. -enum class ObjectStorageRequestMode : uint8_t -{ - Default, - NativeConditional, -}; - /// Settings to be passed to IDisk::writeFile() struct WriteSettings { @@ -58,10 +42,10 @@ struct WriteSettings bool s3_allow_parallel_part_upload = true; /// Overrides S3RequestSetting::check_objects_after_upload for this write (nullopt = no - /// override). Writers of CAS-MUTABLE keys (content-addressed shard manifests) set `false`: - /// such a key is legitimately replaced by a concurrent conditional PUT between this upload and - /// the check's HEAD, so the size comparison false-positives ("it's a bug in S3") under normal - /// contention. Integrity for those keys is the conditional PUT outcome + token, not a recheck. + /// override). A writer whose key can legitimately be replaced by a concurrent conditional PUT + /// between this upload and the check's HEAD sets `false`: otherwise the size comparison + /// false-positives ("it's a bug in S3") under normal contention. Integrity for such a key comes + /// from the conditional PUT outcome and token, not a recheck. std::optional s3_check_objects_after_upload_override; bool azure_allow_parallel_part_upload = true; @@ -92,15 +76,28 @@ struct WriteSettings /// Overrides S3RequestSetting::max_unexpected_write_error_retries (default 4) for this write. /// WriteBufferFromS3::makeSinglepartUpload/completeMultipartUpload run their OWN retry loop above /// the S3 client that reissues the identical request (WITH its If-None-Match/If-Match condition) - /// on a NO_SUCH_KEY response — a second retry-affecting layer a client-level override - /// (a client-level profile override) does not reach. A CAS conditional write sets this to 1 for - /// exactly one attempt at this layer too (RFC cas-s3-timeout-retry-control). 0 = no override. + /// on a NO_SUCH_KEY response — a second retry-affecting layer a client-level profile override does + /// not reach. A conditional write that must not retry at that layer either sets this to 1 for + /// exactly one attempt. 0 = no override. size_t s3_max_unexpected_write_error_retries_override = 0; /// Selects the retry profile the object storage should execute this write under; see /// ObjectStorageRetryProfile. ObjectStorageRetryProfile object_storage_retry_profile = ObjectStorageRetryProfile::Default; + /// Request timeout (send/receive inactivity bound) for the single-attempt client selected by + /// `object_storage_retry_profile == SingleAttempt`. 0 = the storage's configured timeout. + uint64_t object_storage_attempt_timeout_ms = 0; + + /// The cap the single-attempt client's clone puts on one TCP connect and again on one TLS + /// handshake, frozen by the mount at open; see `CasRequestBudget::attemptEnvelopeMs`. 0 = no cap. + uint64_t object_storage_connect_timeout_cap_ms = 0; + + /// The caller's own attempt number for the request built from these settings, 1-based; 0 leaves the + /// buffer's own numbering. A caller reissuing this write passes its count so the HTTP client sees + /// attempt ≥ 2. + size_t object_storage_attempt_number = 0; + /// Selects the transport requirement for an object storage copy; see `ObjectStorageCopyMode`. ObjectStorageCopyMode object_storage_copy_mode = ObjectStorageCopyMode::Default; diff --git a/src/IO/tests/gtest_cas_readbuffer_s3.cpp b/src/IO/tests/gtest_cas_readbuffer_s3.cpp new file mode 100644 index 000000000000..0852e6460e08 --- /dev/null +++ b/src/IO/tests/gtest_cas_readbuffer_s3.cpp @@ -0,0 +1,546 @@ +#include + +#include +#include + +#include +#include "config.h" + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int S3_ERROR; +} + +static constexpr auto TEST_LOG_LEVEL = "debug"; +static fs::path caches_dir = fs::current_path() / "readbuffer_s3"; +static std::string cache_base_path = caches_dir / "cache1" / ""; + +/// Everything below, including the fixture, has internal linkage: `gtest_readbuffer_s3.cpp` defines +/// its own, different `ClientFake`/`CountedSession`/etc. under the same names, and this fixture +/// references file-local `static`s, so external linkage here would be an ODR violation. The suite is +/// renamed to `CASReadBufferFromS3Test` (test names unchanged) so it doesn't share a suite name with +/// that file's fixture, which gtest's own registration-time check would otherwise reject. +namespace +{ + +/// A copy of `ReadBufferFromS3Test` from `gtest_readbuffer_s3.cpp`, renamed per the note above. +class CASReadBufferFromS3Test : public ::testing::Test +{ +public: + static void setupLogs(const std::string & level) + { + Poco::AutoPtr channel(new Poco::ConsoleChannel(std::cerr)); + Poco::Logger::root().setChannel(channel); + Poco::Logger::root().setLevel(level); + } + + void SetUp() override + { + if (const char * test_log_level = std::getenv("TEST_LOG_LEVEL")) // NOLINT(concurrency-mt-unsafe) + setupLogs(test_log_level); + else + setupLogs(TEST_LOG_LEVEL); + + if (fs::exists(cache_base_path)) + fs::remove_all(cache_base_path); + fs::create_directories(cache_base_path); + } + + void TearDown() override + { + if (fs::exists(cache_base_path)) + fs::remove_all(cache_base_path); + } +}; + +/// A copy of `CountedSession` from `gtest_readbuffer_s3.cpp`, an opaque session marker for +/// `SessionAwareIOStream`. That file's version counts live instances for its session-lifetime tests; +/// no test in this file reads such a count, and an internal-linkage member nothing calls or reads +/// trips `-Wunused-member-function`/`-Wunneeded-member-function`, so this copy carries no state at all. +class CountedSession +{ +}; + +using CountedSessionPtr = std::shared_ptr; + +class StringHTTPBasicStreamBuf : public Poco::Net::HTTPBasicStreamBuf +{ +public: + explicit StringHTTPBasicStreamBuf(std::string body) : BasicBufferedStreamBuf(body.size(), IOS::in), bodyStream(std::stringstream(body)) + { + } + +private: + std::stringstream bodyStream; + + int readFromDevice(char_type * buf, std::streamsize n) override + { + bodyStream.read(buf, n); + return static_cast(bodyStream.gcount()); + } +}; + +/// A response body stream that throws once `bytes_before_failure` bytes have been handed out (0 means +/// the very first read fails), simulating a GET whose headers arrived successfully but whose body read +/// broke before delivering that many bytes to the consumer. +class BreakingHTTPBasicStreamBuf : public Poco::Net::HTTPBasicStreamBuf +{ +public: + BreakingHTTPBasicStreamBuf(std::string body, size_t bytes_before_failure_) + : BasicBufferedStreamBuf(body.size(), IOS::in), bodyStream(std::stringstream(std::move(body))), bytes_before_failure(bytes_before_failure_) + { + } + +private: + std::stringstream bodyStream; + size_t bytes_before_failure; + + int readFromDevice(char_type * buf, std::streamsize n) override + { + if (bytes_before_failure == 0) + throw DB::Exception(DB::ErrorCodes::S3_ERROR, "Simulated S3 body read failure"); + + bodyStream.read(buf, std::min(n, static_cast(bytes_before_failure))); + const auto got = bodyStream.gcount(); + bytes_before_failure -= static_cast(got); + return static_cast(got); + } +}; + +/// The byte offset the request's Range header asks for, or 0 when no Range was set. sendRequest() +/// always emits "bytes=-" or "bytes=-", so parsing out lets a mock GetObject +/// serve the bytes a reissued request should actually receive. +static size_t rangeStart(const Aws::S3::Model::GetObjectRequest & request) +{ + if (!request.RangeHasBeenSet()) + return 0; + const std::string & range = request.GetRange(); + const size_t begin_pos = range.find('=') + 1; + const size_t dash_pos = range.find('-', begin_pos); + return std::stoull(range.substr(begin_pos, dash_pos - begin_pos)); +} + +static Aws::S3::Model::GetObjectOutcome makeGetObjectOutcome(std::streambuf * sb, const std::string & etag) +{ + Aws::Http::HeaderValueCollection headers; + headers["etag"] = etag; + auto response_stream = Aws::Utils::Stream::ResponseStream( + Aws::New>("test response stream", std::make_shared(), sb)); + Aws::AmazonWebServiceResult aws_result(std::move(response_stream), std::move(headers)); + DB::S3::Model::GetObjectResult result(std::move(aws_result)); + return Aws::S3::Model::GetObjectOutcome(std::move(result)); +} + +using GetObjectFn = std::function; + +/// A trimmed copy of `ClientFake` from `gtest_readbuffer_s3.cpp`: only the `GetObject` override this +/// file's tests need. It deliberately does NOT match that file's `ClientFake` (which also overrides +/// `ListObjectsV2`) -- see the anonymous-namespace comment above for why that's required, not optional. +struct ClientFake : DB::S3::Client +{ + explicit ClientFake() + : DB::S3::Client( + 1, + DB::S3::ServerSideEncryptionKMSConfig(), + std::make_shared("test_access_key", "test_secret"), + DB::S3::ClientFactory::instance().createClientConfiguration( + "test_region", + DB::RemoteHostFilter(), + 1, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + true, + true, + true, + false, + {}, + /* request_throttler = */ {}, + "http"), + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + DB::S3::ClientSettings()) + { + } + + std::optional getObjectImpl; + + Aws::S3::Model::GetObjectOutcome GetObject([[maybe_unused]] const Aws::S3::Model::GetObjectRequest & request) const override + { + chassert(getObjectImpl); + return (*getObjectImpl)(request); + } +}; + +static void readAndAssert(DB::ReadBuffer & buf, const char * str) +{ + size_t n = strlen(str); + std::vector tmp(n); + buf.readStrict(tmp.data(), n); + ASSERT_EQ(strncmp(tmp.data(), str, n), 0); +} + +} + +TEST_F(CASReadBufferFromS3Test, IdentityNotFlaggedWhenFailedAttemptDeliveredNoBytes) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 20; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto failing_buf = std::make_shared(body, /* bytes_before_failure */ 0); + auto full_buf = std::make_shared(body); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + EXPECT_EQ(rangeStart(request), 0); + if (call == 1) + return makeGetObjectOutcome(failing_buf.get(), "A"); + return makeGetObjectOutcome(full_buf.get(), "B"); + }; + + /// First attempt's headers carried ETag "A", but its body read fails before any byte reaches the + /// consumer; the reissue delivers the whole object under ETag "B". No bytes of "A" were ever + /// consumed, so this must not be flagged as a coherence problem. + readAndAssert(subject, body.c_str()); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, IdentityFlaggedWhenBytesDeliveredBeforeFailure) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto breaking_buf = std::make_shared(body, /* bytes_before_failure */ 3); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(breaking_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + return makeGetObjectOutcome(rest_buf.get(), "B"); + }; + + /// The first response (ETag "A") delivers 3 bytes before its stream breaks; the reissue, resuming + /// from offset 3, answers with ETag "B". Bytes from two different incarnations reached the + /// consumer, so this must be flagged. + readAndAssert(subject, body.c_str()); + ASSERT_TRUE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, IdentityNotFlaggedWhenReissuedEtagMatches) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto breaking_buf = std::make_shared(body, /* bytes_before_failure */ 3); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(breaking_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + return makeGetObjectOutcome(rest_buf.get(), "A"); + }; + + /// Same as above, but the reissue answers with the same ETag "A": both attempts belong to the same + /// incarnation, so this must not be flagged. + readAndAssert(subject, body.c_str()); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, ThreeResponsesABytesThenAEmptyFailThenBBytesIsFlagged) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto delivers_then_fails = std::make_shared(body, /* bytes_before_failure */ 3); + auto fails_empty = std::make_shared(body, /* bytes_before_failure */ 0); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(delivers_then_fails.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + if (call == 2) + return makeGetObjectOutcome(fails_empty.get(), "A"); + return makeGetObjectOutcome(rest_buf.get(), "B"); + }; + + /// A delivers 3 bytes, then breaks. The reissue (same ETag "A") fails before delivering anything. + /// The next reissue answers with ETag "B" and delivers the rest: A-bytes and B-bytes were mixed, so + /// this must be flagged, even though an empty failed attempt for "A" sat in between. + readAndAssert(subject, body.c_str()); + ASSERT_TRUE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, ThreeResponsesABytesThenBEmptyFailThenABytesIsNotFlagged) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto delivers_then_fails = std::make_shared(body, /* bytes_before_failure */ 3); + auto fails_empty = std::make_shared(body, /* bytes_before_failure */ 0); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(delivers_then_fails.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + if (call == 2) + return makeGetObjectOutcome(fails_empty.get(), "B"); + return makeGetObjectOutcome(rest_buf.get(), "A"); + }; + + /// A delivers 3 bytes, then breaks. The reissue under ETag "B" fails before delivering anything, so + /// it never contributes to the read. The next reissue answers with ETag "A" (matching the only + /// response that ever delivered bytes) and delivers the rest: the read is coherent and must not be + /// flagged, even though a differently-ETagged empty failed attempt sat in between. + readAndAssert(subject, body.c_str()); + ASSERT_FALSE(subject.responseIdentityChanged()); + ASSERT_EQ(subject.getObjectMetadataFromTheLastRequest().etag, "A"); +} + +TEST_F(CASReadBufferFromS3Test, SeekReissueAcceptsNewEtagWithoutFlag) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + read_settings.remote_fs_settings.min_bytes_for_seek = 0; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto first_buf = std::make_shared(body); + auto after_seek_buf = std::make_shared(body.substr(8)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(first_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 8); + return makeGetObjectOutcome(after_seek_buf.get(), "B"); + }; + + readAndAssert(subject, "123"); + /// A seek far enough ahead to force a reissue (not an in-buffer rewind, not a small forward skip): + /// the caller explicitly repositioned to a different range, so the new response's ETag "B" must not + /// be compared against "A". + subject.seek(8, SEEK_SET); + readAndAssert(subject, "9"); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, SetReadUntilPositionReissueAcceptsNewEtagWithoutFlag) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 2; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456"; + auto first_buf = std::make_shared(body); + auto after_reposition_buf = std::make_shared(body.substr(2, 3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(first_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 2); + return makeGetObjectOutcome(after_reposition_buf.get(), "B"); + }; + + readAndAssert(subject, "12"); + /// impl is still open (no read-until-position was set yet, so nothing released it). Narrowing the + /// read-until bound now tears impl down to reissue for the new bound: an explicit reposition, so + /// the new response's ETag "B" must not be compared against "A". + subject.setReadUntilPosition(5); + readAndAssert(subject, "345"); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, SetReadUntilEndReissueAcceptsNewEtagWithoutFlag) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + subject.setReadUntilPosition(3); + + const std::string body = "123456789"; + auto first_buf = std::make_shared(body.substr(0, 3)); + auto after_reposition_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(first_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + return makeGetObjectOutcome(after_reposition_buf.get(), "B"); + }; + + readAndAssert(subject, "123"); + /// Reading exactly up to the bound releases the result (does not reset impl). Removing the bound + /// now tears impl down to reissue for the rest of the object: an explicit reposition, so the new + /// response's ETag "B" must not be compared against "A". + subject.setReadUntilEnd(); + readAndAssert(subject, "456789"); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, InBufferSeekPreservesBaselineAndLaterMixedRetryIsFlagged) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 3; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + auto breaking_buf = std::make_shared(body, /* bytes_before_failure */ 3); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(breaking_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + return makeGetObjectOutcome(rest_buf.get(), "B"); + }; + + readAndAssert(subject, "123"); + /// Rewind within the bytes already buffered: this hits the in-buffer fast path in seek(), which + /// never touches impl, so it must not forget the identity baseline. + subject.seek(1, SEEK_SET); + readAndAssert(subject, "23"); + /// Reading past the buffer now reissues on the SAME impl (a retry after a stream break, not an + /// explicit reposition); the baseline from "A" must have survived the harmless seek above, so the + /// mismatched ETag "B" here must still be flagged. + readAndAssert(subject, "456789"); + ASSERT_TRUE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, ExternalBufferFlagsMixedIncarnations) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + auto subject = DB::ReadBufferFromS3( + client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings, /* use_external_buffer */ true); + + const std::string body = "123456789"; + auto breaking_buf = std::make_shared(body, /* bytes_before_failure */ 3); + auto rest_buf = std::make_shared(body.substr(3)); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + if (call == 1) + { + EXPECT_EQ(rangeStart(request), 0); + return makeGetObjectOutcome(breaking_buf.get(), "A"); + } + EXPECT_EQ(rangeStart(request), 3); + return makeGetObjectOutcome(rest_buf.get(), "B"); + }; + + std::vector external_memory(3); + + /// Drive the external-buffer path the way a prefetching/threadpool reader does: supply the memory + /// with set() and pull one chunk with next(), rather than relying on the buffer's own allocation. + subject.set(external_memory.data(), external_memory.size()); + ASSERT_TRUE(subject.next()); + ASSERT_EQ(std::string(subject.buffer().begin(), subject.buffer().end()), "123"); + ASSERT_FALSE(subject.responseIdentityChanged()); + + /// This next() call breaks the "A" stream and reissues; the reissue answers with ETag "B" and + /// delivers bytes via the external buffer. Bytes were consumed on this path too, so it must flag. + subject.set(external_memory.data(), external_memory.size()); + ASSERT_TRUE(subject.next()); + ASSERT_EQ(std::string(subject.buffer().begin(), subject.buffer().end()), "456"); + ASSERT_TRUE(subject.responseIdentityChanged()); +} + +TEST_F(CASReadBufferFromS3Test, PartialInternalFillNeverExposedDoesNotCountAsDelivery) +{ + const auto client = std::make_shared(); + DB::ReadSettings read_settings; + read_settings.remote_fs_settings.buffer_size = 5; + auto subject = DB::ReadBufferFromS3(client, "test_bucket", "test_key", "test_version_id", DB::S3::S3RequestSettings(), read_settings); + + const std::string body = "123456789"; + /// internal_buffer is 5 bytes but only 2 bytes are ever produced before the stream throws, so + /// ReadBufferFromIStream's fill loop calls readFromDevice a second time (asking for more) and gets + /// the exception before it ever assigns `working_buffer` - those 2 bytes are read off the wire but + /// never exposed to the consumer. + auto partial_then_fails = std::make_shared(body, /* bytes_before_failure */ 2); + auto full_buf = std::make_shared(body); + + client->getObjectImpl = [&, call = 0](const Aws::S3::Model::GetObjectRequest & request) mutable -> Aws::S3::Model::GetObjectOutcome + { + ++call; + EXPECT_EQ(rangeStart(request), 0); + if (call == 1) + return makeGetObjectOutcome(partial_then_fails.get(), "A"); + return makeGetObjectOutcome(full_buf.get(), "B"); + }; + + readAndAssert(subject, body.c_str()); + ASSERT_FALSE(subject.responseIdentityChanged()); +} + +#endif diff --git a/src/IO/tests/gtest_cas_writebuffer_s3.cpp b/src/IO/tests/gtest_cas_writebuffer_s3.cpp new file mode 100644 index 000000000000..37ac44df9b1e --- /dev/null +++ b/src/IO/tests/gtest_cas_writebuffer_s3.cpp @@ -0,0 +1,1165 @@ +#include + +#include "config.h" + +#if USE_AWS_S3 + +#include + +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +#include + +#include +#include +#include + + +namespace DB +{ +namespace Setting +{ + extern const SettingsBool s3_check_objects_after_upload; + extern const SettingsUInt64 s3_max_inflight_parts_for_one_file; + extern const SettingsUInt64 s3_max_single_part_upload_size; + extern const SettingsUInt64 s3_max_upload_part_size; + extern const SettingsUInt64 s3_min_upload_part_size; + extern const SettingsUInt64 s3_strict_upload_part_size; + extern const SettingsUInt64 s3_upload_part_size_multiply_factor; + extern const SettingsUInt64 s3_upload_part_size_multiply_parts_count_threshold; +} + +namespace S3RequestSetting +{ + extern const S3RequestSettingsBool allow_native_copy; +} + +namespace ErrorCodes +{ + extern const int LOGICAL_ERROR; + extern const int S3_ERROR; + extern const int NOT_IMPLEMENTED; +} + +} + +using namespace DB; + +namespace +{ + +/// A private copy of `gtest_writebuffer_s3.cpp`'s mock S3 client, extended with the attempt-seed +/// recorder and `PreconditionFailed` injection this file's tests need -- duplicated so an upstream +/// change to the mock never produces a conflict here on rebase. Anonymous namespace: both `.cpp` +/// files link into `unit_tests_dbms`, and without internal linkage the duplicate types below would +/// violate the One Definition Rule. A few members this file's own tests never call are kept +/// for parity with the upstream mock and marked `[[maybe_unused]]`, since the anonymous namespace +/// (unlike the external linkage of a shared header) makes an unused member function an error. +namespace MockS3 +{ + +class Sequencer +{ +public: + size_t next() { return counter++; } + std::string next_id() + { + std::stringstream ss; + ss << "id-" << next(); + return ss.str(); + } + +private: + size_t counter = 0; +}; + +class BucketMemStore +{ +public: + using Key = std::string; + using Data = std::string; + using ETag = std::string; + using MPU_ID = std::string; + using MPUPartsInProgress = std::map; + using MPUParts = std::vector; + + + std::map objects; + std::map multiPartUploads; + std::vector> CompletedPartUploads; + + Sequencer sequencer; + + std::string CreateMPU() + { + auto id = sequencer.next_id(); + multiPartUploads.emplace(id, MPUPartsInProgress{}); + return id; + } + + std::string UploadPart(const std::string & upload_id, const std::string & part) + { + auto etag = sequencer.next_id(); + auto & parts = multiPartUploads.at(upload_id); + parts.emplace(etag, part); + return etag; + } + + void PutObject(const std::string & key, const std::string & data) + { + objects[key] = data; + } + + void CompleteMPU(const std::string & key, const std::string & upload_id, const std::vector & etags) + { + MPUParts completedParts; + completedParts.reserve(etags.size()); + + auto & parts = multiPartUploads.at(upload_id); + for (const auto & tag: etags) { + completedParts.push_back(parts.at(tag)); + } + + std::stringstream file_data; + for (const auto & part_data: completedParts) { + file_data << part_data; + } + + CompletedPartUploads.emplace_back(upload_id, std::move(completedParts)); + objects[key] = file_data.str(); + multiPartUploads.erase(upload_id); + } + + void AbortMPU(const std::string & upload_id) + { + multiPartUploads.erase(upload_id); + } + + + const std::vector> & GetCompletedPartUploads() const + { + return CompletedPartUploads; + } + + [[maybe_unused]] static std::vector GetPartSizes(const MPUParts & parts) + { + std::vector result; + result.reserve(parts.size()); + for (const auto & part_data : parts) + result.push_back(part_data.size()); + + return result; + } + +}; + +class S3MemStrore +{ +public: + void CreateBucket(const std::string & bucket) + { + chassert(!buckets.contains(bucket)); + buckets.emplace(bucket, BucketMemStore{}); + } + + BucketMemStore& GetBucketStore(const std::string & bucket) { + return buckets.at(bucket); + } + +private: + std::map buckets; +}; + +struct EventCounts +{ + size_t headObject = 0; + size_t getObject = 0; + size_t putObject = 0; + size_t multiUploadCreate = 0; + size_t multiUploadComplete = 0; + size_t multiUploadAbort = 0; + size_t uploadParts = 0; + size_t writtenSize = 0; + size_t copyObject = 0; + size_t deleteObject = 0; + size_t getBucketVersioning = 0; + + [[maybe_unused]] size_t totalRequestsCount() const + { + return headObject + getObject + putObject + multiUploadCreate + multiUploadComplete + uploadParts; + } +}; + +struct Client; + +struct InjectionModel +{ + virtual ~InjectionModel() = default; + +#define DeclareInjectCall(ObjectTypePart) \ + virtual std::optional call(const Aws::S3::Model::ObjectTypePart##Request & /*request*/) \ + { \ + return std::nullopt; \ + } + DeclareInjectCall(PutObject) + DeclareInjectCall(HeadObject) + DeclareInjectCall(CreateMultipartUpload) + DeclareInjectCall(CompleteMultipartUpload) + DeclareInjectCall(AbortMultipartUpload) + DeclareInjectCall(UploadPart) + DeclareInjectCall(CopyObject) + DeclareInjectCall(DeleteObject) + DeclareInjectCall(GetBucketVersioning) +#undef DeclareInjectCall +}; + +/// `DB::S3::getClickhouseAttemptNumber(const Aws::AmazonWebServiceRequest &)` reads `GetHeaders()`, +/// which for a plain S3 request never includes `SetAdditionalCustomHeaderValue`'s custom headers -- +/// only `AWSClient::BuildHttpRequest` merges those into the wire-level `Aws::Http::HttpRequest` that +/// `PocoHTTPClient` actually inspects (the overload production code reads). This mock overrides the +/// `S3Client` virtuals directly, below that merge, so it reads the custom header collection itself. +/// `nullopt` means the `clickhouse-request` header is absent -- distinct from an explicit `attempt=1`, +/// since a seed of 0 leaves every verb but the read path unseeded (no header at all; see +/// `S3::seededAttemptNumber`'s callers). +std::optional attemptNumberFromCustomHeaders(const Aws::AmazonWebServiceRequest & request) +{ + const auto & headers = request.GetAdditionalCustomHeaders(); + auto it = headers.find("clickhouse-request"); + if (it == headers.end()) + return std::nullopt; + static const std::string key = "attempt="; + auto pos = it->second.find(key); + if (pos == std::string::npos) + return std::nullopt; + try + { + return static_cast(std::stol(it->second.substr(pos + key.size()))); + } + catch (const std::exception &) + { + return std::nullopt; + } +} + +struct Client : DB::S3::Client +{ + explicit Client(std::shared_ptr mock_s3_store) + : DB::S3::Client( + 100, + DB::S3::ServerSideEncryptionKMSConfig(), + std::make_shared("", ""), + GetClientConfiguration(), + Aws::Client::AWSAuthV4Signer::PayloadSigningPolicy::Never, + DB::S3::ClientSettings{ + .use_virtual_addressing = true, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }) + , store(mock_s3_store) + {} + + static std::shared_ptr CreateClient(String bucket = "mock-s3-bucket") + { + auto s3store = std::make_shared(); + s3store->CreateBucket(bucket); + return std::make_shared(s3store); + } + + static DB::S3::PocoHTTPClientConfiguration GetClientConfiguration() + { + DB::RemoteHostFilter remote_host_filter; + return DB::S3::ClientFactory::instance().createClientConfiguration( + "some-region", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ true, + /* s3_slow_all_threads_after_retryable_error = */ true, + /* enable_s3_requests_logging = */ true, + /* for_disk_s3 = */ false, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + } + + void setInjectionModel(std::shared_ptr injections_) + { + injections = injections_; + } + + /// `clickhouse-request` attempt of every verb, in order -- test-only recorder for the attempt-seed tests. + mutable std::vector> attempts_seen; + + Aws::S3::Model::ListObjectsV2Outcome ListObjectsV2(const Aws::S3::Model::ListObjectsV2Request & request) const override + { + attempts_seen.push_back(attemptNumberFromCustomHeaders(request)); + auto & bStore = store->GetBucketStore(request.GetBucket()); + Aws::S3::Model::ListObjectsV2Result result; + result.SetPrefix(request.GetPrefix()); + int emitted = 0; + std::string last; + const std::string after = request.ContinuationTokenHasBeenSet() ? request.GetContinuationToken() + : request.StartAfterHasBeenSet() ? request.GetStartAfter() : ""; + for (const auto & [key, data] : bStore.objects) + { + if (!key.starts_with(request.GetPrefix()) || key <= after) + continue; + if (emitted == request.GetMaxKeys()) + { + result.SetIsTruncated(true); + result.SetNextContinuationToken(last); + break; + } + Aws::S3::Model::Object object; + object.SetKey(key); + object.SetSize(static_cast(data.size())); + result.AddContents(std::move(object)); + last = key; + ++emitted; + } + return Aws::S3::Model::ListObjectsV2Outcome(std::move(result)); + } + + Aws::S3::Model::DeleteObjectsOutcome DeleteObjects(const Aws::S3::Model::DeleteObjectsRequest & request) const override + { + attempts_seen.push_back(attemptNumberFromCustomHeaders(request)); + + auto & bStore = store->GetBucketStore(request.GetBucket()); + for (const auto & identifier : request.GetDelete().GetObjects()) + bStore.objects.erase(identifier.GetKey()); + + Aws::S3::Model::DeleteObjectsResult result; + return Aws::S3::Model::DeleteObjectsOutcome(std::move(result)); + } + + Aws::S3::Model::PutObjectOutcome PutObject(const Aws::S3::Model::PutObjectRequest & request) const override + { + attempts_seen.push_back(attemptNumberFromCustomHeaders(request)); + ++counters.putObject; + + if (const auto * wrapper = dynamic_cast(&request)) + last_put_object_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return *opt_val; + } + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + std::stringstream data; + data << request.GetBody()->rdbuf(); + bStore.PutObject(request.GetKey(), data.str()); + counters.writtenSize += data.str().length(); + + Aws::S3::Model::PutObjectOutcome outcome; + Aws::S3::Model::PutObjectResult result(outcome.GetResultWithOwnership()); + result.SetETag("etag-singlepart-" + request.GetKey()); + return result; + } + + Aws::S3::Model::GetObjectOutcome GetObject(const Aws::S3::Model::GetObjectRequest & request) const override + { + ++counters.getObject; + + auto & bStore = store->GetBucketStore(request.GetBucket()); + const String data = bStore.objects[request.GetKey()]; + + size_t begin = 0; + size_t end = data.size() - 1; + + const String & range = request.GetRange(); + const String prefix = "bytes="; + if (range.starts_with(prefix)) + { + int ret = sscanf(range.c_str(), "bytes=%zu-%zu", &begin, &end); /// NOLINT + chassert(ret == 2); + } + + auto factory = request.GetResponseStreamFactory(); + Aws::Utils::Stream::ResponseStream responseStream(factory); + responseStream.GetUnderlyingStream() << std::stringstream(data.substr(begin, end - begin + 1)).rdbuf(); + + Aws::AmazonWebServiceResult awsStream(std::move(responseStream), Aws::Http::HeaderValueCollection()); + Aws::S3::Model::GetObjectResult getObjectResult(std::move(awsStream)); + return Aws::S3::Model::GetObjectOutcome(std::move(getObjectResult)); + } + + Aws::S3::Model::HeadObjectOutcome HeadObject(const Aws::S3::Model::HeadObjectRequest & request) const override + { + attempts_seen.push_back(attemptNumberFromCustomHeaders(request)); + ++counters.headObject; + + /// The request's DYNAMIC type is still the production `DB::S3::HeadObjectRequest` wrapper -- + /// this override only sees it through the SDK base-class reference. Mirrors the dynamic_cast + /// `Client::BuildHttpRequest` itself does, so a test can observe the mark this mock never + /// forwards through an HTTP layer. + if (const auto * wrapper = dynamic_cast(&request)) + last_head_object_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return std::move(*opt_val); + } + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + auto obj = bStore.objects[request.GetKey()]; + Aws::S3::Model::HeadObjectOutcome outcome; + Aws::S3::Model::HeadObjectResult result(outcome.GetResultWithOwnership()); + result.SetContentLength(obj.length()); + return result; + } + + Aws::S3::Model::CreateMultipartUploadOutcome CreateMultipartUpload(const Aws::S3::Model::CreateMultipartUploadRequest & request) const override + { + ++counters.multiUploadCreate; + + if (const auto * wrapper = dynamic_cast(&request)) + last_create_multipart_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return std::move(*opt_val); + } + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + auto mpu_id = bStore.CreateMPU(); + + Aws::S3::Model::CreateMultipartUploadResult result; + result.SetUploadId(mpu_id.c_str()); + return Aws::S3::Model::CreateMultipartUploadOutcome(result); + } + + Aws::S3::Model::UploadPartOutcome UploadPart(const Aws::S3::Model::UploadPartRequest & request) const override + { + ++counters.uploadParts; + + if (const auto * wrapper = dynamic_cast(&request)) + last_upload_part_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return std::move(*opt_val); + } + } + + std::stringstream data; + data << request.GetBody()->rdbuf(); + counters.writtenSize += data.str().length(); + + auto & bStore = store->GetBucketStore(request.GetBucket()); + auto etag = bStore.UploadPart(request.GetUploadId(), data.str()); + + Aws::S3::Model::UploadPartResult result; + result.SetETag(etag); + return Aws::S3::Model::UploadPartOutcome(result); + } + + Aws::S3::Model::CompleteMultipartUploadOutcome CompleteMultipartUpload(const Aws::S3::Model::CompleteMultipartUploadRequest & request) const override + { + ++counters.multiUploadComplete; + + if (const auto * wrapper = dynamic_cast(&request)) + last_complete_multipart_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return std::move(*opt_val); + } + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + + std::vector etags; + for (const auto & x: request.GetMultipartUpload().GetParts()) { + etags.push_back(x.GetETag()); + } + bStore.CompleteMPU(request.GetKey(), request.GetUploadId(), etags); + + Aws::S3::Model::CompleteMultipartUploadResult result; + result.SetETag("etag-multipart-" + request.GetKey()); + return Aws::S3::Model::CompleteMultipartUploadOutcome(result); + } + + Aws::S3::Model::AbortMultipartUploadOutcome AbortMultipartUpload(const Aws::S3::Model::AbortMultipartUploadRequest & request) const override + { + ++counters.multiUploadAbort; + + if (injections) + { + if (auto opt_val = injections->call(request)) + { + return std::move(*opt_val); + } + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + bStore.AbortMPU(request.GetUploadId()); + + Aws::S3::Model::AbortMultipartUploadResult result; + return Aws::S3::Model::AbortMultipartUploadOutcome(result); + } + + Aws::S3::Model::CopyObjectOutcome CopyObject(const Aws::S3::Model::CopyObjectRequest & request) const override + { + ++counters.copyObject; + + if (const auto * wrapper = dynamic_cast(&request)) + last_copy_object_native_conditional = wrapper->isNativeConditional(); + + last_copy_object_if_match = request.IfMatchHasBeenSet(); + last_copy_object_if_none_match = request.IfNoneMatchHasBeenSet(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + /// CopySource is "/"; parse it back apart to look the source object up + /// (both source and destination live in the same S3MemStrore in these tests). + const std::string & copy_source = request.GetCopySource(); + const size_t sep = copy_source.find('/'); + chassert(sep != std::string::npos); + const std::string src_bucket_name = copy_source.substr(0, sep); + const std::string src_key = copy_source.substr(sep + 1); + + auto & src_store = store->GetBucketStore(src_bucket_name); + const std::string data = src_store.objects.at(src_key); + + auto & dst_store = store->GetBucketStore(request.GetBucket()); + dst_store.PutObject(request.GetKey(), data); + + Aws::S3::Model::CopyObjectResult result; + Aws::S3::Model::CopyObjectResultDetails details; + details.SetETag("etag-copy-" + request.GetKey()); + result.SetCopyObjectResultDetails(details); + return Aws::S3::Model::CopyObjectOutcome(result); + } + + Aws::S3::Model::DeleteObjectOutcome DeleteObject(const Aws::S3::Model::DeleteObjectRequest & request) const override + { + attempts_seen.push_back(attemptNumberFromCustomHeaders(request)); + ++counters.deleteObject; + + if (const auto * wrapper = dynamic_cast(&request)) + last_delete_object_native_conditional = wrapper->isNativeConditional(); + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + auto & bStore = store->GetBucketStore(request.GetBucket()); + bStore.objects.erase(request.GetKey()); + + Aws::S3::Model::DeleteObjectResult result; + return Aws::S3::Model::DeleteObjectOutcome(result); + } + + Aws::S3::Model::GetBucketVersioningOutcome GetBucketVersioning(const Aws::S3::Model::GetBucketVersioningRequest & request) const override + { + ++counters.getBucketVersioning; + + if (injections) + { + if (auto opt_val = injections->call(request)) + return std::move(*opt_val); + } + + Aws::S3::Model::GetBucketVersioningResult result; + result.SetStatus(Aws::S3::Model::BucketVersioningStatus::Enabled); + return Aws::S3::Model::GetBucketVersioningOutcome(result); + } + + std::shared_ptr store; + mutable EventCounts counters; + mutable std::shared_ptr injections; + mutable bool last_head_object_native_conditional = false; + mutable bool last_delete_object_native_conditional = false; + mutable bool last_put_object_native_conditional = false; + mutable bool last_create_multipart_native_conditional = false; + mutable bool last_upload_part_native_conditional = false; + mutable bool last_complete_multipart_native_conditional = false; + mutable bool last_copy_object_native_conditional = false; + mutable bool last_copy_object_if_match = false; + mutable bool last_copy_object_if_none_match = false; + void resetCounters() const { counters = {}; } +}; + +struct PutObjectFailIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::PutObjectRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::VALIDATION, "FailInjection", "PutObjectFailIngection", false); + } +}; + +/// A conditional-write 412, matched by `S3::isPreconditionFailedError` on the canonical `` name. +struct PutObjectPreconditionFailedIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::PutObjectRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::UNKNOWN, "PreconditionFailed", "precondition failed", false); + } +}; + +struct HeadObjectFailIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::HeadObjectRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::VALIDATION, "FailInjection", "HeadObjectFailIngection", false); + } +}; + +struct CreateMPUFailIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::CreateMultipartUploadRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::VALIDATION, "FailInjection", "CreateMPUFailIngection", false); + } +}; + +struct CompleteMPUFailIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::CompleteMultipartUploadRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::VALIDATION, "FailInjection", "CompleteMPUFailIngection", false); + } +}; + +struct UploadPartFailIngection: InjectionModel +{ + std::optional call(const Aws::S3::Model::UploadPartRequest & /*request*/) override + { + return Aws::Client::AWSError(Aws::Client::CoreErrors::VALIDATION, "FailInjection", "UploadPartFailIngection", false); + } +}; + +/// Injects an arbitrary AWSError on DeleteObject -- used to drive the conditional-remove +/// (`removeObjectIfTokenMatches`) outcome mapping: a 412-shaped error (exception name "PreconditionFailed", +/// matched by `S3::isPreconditionFailedError`) must map to `ConditionalRemoveOutcome::TokenMismatch`, and a +/// 404-shaped error (a `NO_SUCH_KEY`/`RESOURCE_NOT_FOUND`/`NO_SUCH_BUCKET` error type, matched by +/// `S3::isNotFoundError`) must map to `ConditionalRemoveOutcome::NotFound`. +struct DeleteObjectErrorInjection: InjectionModel +{ + [[maybe_unused]] explicit DeleteObjectErrorInjection(Aws::Client::AWSError error_) : error(std::move(error_)) {} + + std::optional call(const Aws::S3::Model::DeleteObjectRequest & /*request*/) override + { + return error; + } + + Aws::Client::AWSError error; +}; + +/// Injects an arbitrary `CopyObject` error to exercise ordinary-copy fallback and native-only +/// fail-close behavior. +struct CopyObjectErrorInjection: InjectionModel +{ + [[maybe_unused]] explicit CopyObjectErrorInjection(Aws::Client::AWSError error_) : error(std::move(error_)) {} + + std::optional call(const Aws::S3::Model::CopyObjectRequest & /*request*/) override + { + return error; + } + + Aws::Client::AWSError error; +}; + +struct BaseSyncPolicy +{ + virtual ~BaseSyncPolicy() = default; + virtual DB::ThreadPoolCallbackRunnerUnsafe getScheduler() { return {}; } + virtual void execute(size_t) {} + virtual void setAutoExecute(bool) {} + + virtual size_t size() const { return 0; } + virtual bool empty() const { return size() == 0; } +}; + +struct SimpleAsyncTasks : BaseSyncPolicy +{ + bool auto_execute = false; + std::deque> queue; + + DB::ThreadPoolCallbackRunnerUnsafe getScheduler() override + { + return [this] (std::function && operation, size_t /*priority*/) + { + if (auto_execute) + { + auto task = std::packaged_task(std::move(operation)); + task(); + return task.get_future(); + } + + queue.emplace_back(std::move(operation)); + return queue.back().get_future(); + }; + } + + void execute(size_t limit) override + { + if (limit == 0) + limit = queue.size(); + + while (!queue.empty() && limit) + { + auto & request = queue.front(); + request(); + + queue.pop_front(); + --limit; + } + } + + void setAutoExecute(bool value) override + { + auto_execute = value; + if (auto_execute) + execute(0); + } + + size_t size() const override { return queue.size(); } +}; + +} + +static void writeAsOneBlock(WriteBuffer& buf, size_t size) +{ + std::vector data(size, 'a'); + buf.write(data.data(), data.size()); +} + +static void writeAsPieces(WriteBuffer& buf, size_t size) +{ + size_t ceil = 15ull*1024*1024*1024; + size_t piece = 1; + size_t written = 0; + while (written < size) { + size_t len = std::min({piece, size-written, ceil}); + writeAsOneBlock(buf, len); + written += len; + piece *= 2; + } +} + +class CASWBS3Test : public ::testing::Test +{ +public: + const String bucket = "CASWBS3Test-bucket"; + + Settings & getSettings() + { + return settings; + } + + MockS3::BaseSyncPolicy & getAsyncPolicy() + { + return *async_policy; + } + + std::unique_ptr getWriteBuffer(String file_name = "file", const WriteSettings & write_settings = {}) + { + S3::S3RequestSettings request_settings; + request_settings.updateFromSettings(settings, /* if_changed */true, /* validate_settings */false); + + client->resetCounters(); + + getAsyncPolicy().setAutoExecute(false); + + return std::make_unique( + client, + bucket, + file_name, + DBMS_DEFAULT_BUFFER_SIZE, + request_settings, + nullptr, + std::nullopt, + getAsyncPolicy().getScheduler(), + write_settings); + } + + void setInjectionModel(std::shared_ptr injections_) + { + client->setInjectionModel(injections_); + } + + [[maybe_unused]] void runSimpleScenario(MockS3::EventCounts expected_counters, size_t size) + { + auto scenario = [&] (std::function writeMethod) { + auto buffer = getWriteBuffer("file"); + writeMethod(*buffer, size); + + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + + expected_counters.writtenSize = size; + assertCountersEQ(expected_counters); + + auto & bStore = client->store->GetBucketStore(bucket); + auto & data = bStore.objects["file"]; + ASSERT_EQ(size, data.size()); + for (char c : data) + ASSERT_EQ('a', c); + }; + + scenario(writeAsOneBlock); + scenario(writeAsPieces); + } + + void assertCountersEQ(const MockS3::EventCounts & canonical) { + const auto & actual = client->counters; + ASSERT_EQ(canonical.headObject, actual.headObject); + ASSERT_EQ(canonical.getObject, actual.getObject); + ASSERT_EQ(canonical.putObject, actual.putObject); + ASSERT_EQ(canonical.multiUploadCreate, actual.multiUploadCreate); + ASSERT_EQ(canonical.multiUploadComplete, actual.multiUploadComplete); + ASSERT_EQ(canonical.multiUploadAbort, actual.multiUploadAbort); + ASSERT_EQ(canonical.uploadParts, actual.uploadParts); + ASSERT_EQ(canonical.writtenSize, actual.writtenSize); + } + + [[maybe_unused]] auto getCompletedPartUploads () + { + return client->store->GetBucketStore(bucket).GetCompletedPartUploads(); + } + +protected: + Settings settings; + + std::shared_ptr client; + std::unique_ptr async_policy; + + void SetUp() override + { + client = MockS3::Client::CreateClient(bucket); + async_policy = std::make_unique(); + } + + void TearDown() override + { + client.reset(); + async_policy.reset(); + } +}; + +class CASSyncAsync : public CASWBS3Test, public ::testing::WithParamInterface +{ +protected: + bool test_with_pool = false; + + void SetUp() override + { + test_with_pool = GetParam(); + client = MockS3::Client::CreateClient(bucket); + if (test_with_pool) + { + /// Do not block the main thread awaiting the others task. + /// This test use the only one thread at all + getSettings()[Setting::s3_max_inflight_parts_for_one_file] = 0; + async_policy = std::make_unique(); + } + else + { + async_policy = std::make_unique(); + } + } +}; + +/// Captures what `WriteBufferFromS3` logs at `threshold` and above (default: Error). A message +/// logged below the threshold never reaches the channel, so an empty capture proves the site logged +/// below it rather than merely that this particular text was absent. +class ScopedWriteBufferS3ErrorLogCapture +{ +public: + explicit ScopedWriteBufferS3ErrorLogCapture(const std::string & threshold = "error") + : logger(getLogger("WriteBufferFromS3")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel(threshold); + } + + ~ScopedWriteBufferS3ErrorLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + std::string captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; + Poco::AutoPtr channel; + /// `shared=true` is load-bearing: `AutoPtr(ptr)` would steal a reference the fixture never owned. + Poco::AutoPtr old_channel; + int old_level; +}; + +} + +INSTANTIATE_TEST_SUITE_P(CASWBS3 + , CASSyncAsync + , ::testing::Values(true, false) + , [] (const ::testing::TestParamInfo& info_param) { + std::string name = info_param.param ? "async" : "sync"; + return name; + }); + +/// A non-412 `PutObject` failure on the ordinary (Default) retry profile is a genuine error: the +/// client's one attempt IS the final answer, so the site logs it at Error. +TEST_P(CASSyncAsync, PutObjectErrorLogsErrorForDefaultProfile) +{ + setInjectionModel(std::make_shared()); + + ScopedWriteBufferS3ErrorLogCapture log_capture; + EXPECT_THROW({ + auto buffer = getWriteBuffer("put_object_error_default_profile"); + buffer->write('A'); + buffer->next(); + + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + }, DB::S3Exception); + + EXPECT_THAT(log_capture.captured(), testing::HasSubstr("S3Exception name FailInjection")); + EXPECT_THAT(log_capture.captured(), testing::HasSubstr("PutObjectFailIngection")); +} + +/// The same failure on the SingleAttempt profile (the CAS conditional-write client) is owned by an +/// outer retry loop that resolves the outcome and reissues; the one failed attempt is not terminal, +/// so nothing here reaches Error. +TEST_P(CASSyncAsync, PutObjectErrorLogsDebugForSingleAttemptProfile) +{ + setInjectionModel(std::make_shared()); + + WriteSettings write_settings; + write_settings.object_storage_retry_profile = ObjectStorageRetryProfile::SingleAttempt; + + ScopedWriteBufferS3ErrorLogCapture log_capture; + EXPECT_THROW({ + auto buffer = getWriteBuffer("put_object_error_single_attempt_profile", write_settings); + buffer->write('A'); + buffer->next(); + + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + }, DB::S3Exception); + + EXPECT_TRUE(log_capture.captured().empty()); +} + +/// A conditional write losing its precondition (412) is the caller's expected answer, handled one +/// frame up -- it says nothing to the operator, so it must stay below Information, independent of the +/// retry profile. The capture threshold is Information so that an Info-level line from the site would +/// be caught; the cancel path logs its own Info lines, so the assertion is on the site's text, not on +/// an empty capture. +TEST_P(CASSyncAsync, PreconditionFailedNeverLogsAtError) +{ + setInjectionModel(std::make_shared()); + + ScopedWriteBufferS3ErrorLogCapture log_capture("information"); + EXPECT_THROW({ + auto buffer = getWriteBuffer("put_object_precondition_failed"); + buffer->write('A'); + buffer->next(); + + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + }, DB::S3Exception); + + EXPECT_THAT(log_capture.captured(), testing::Not(testing::HasSubstr("S3Exception name"))); +} + +TEST_F(CASWBS3Test, S3RequestAttemptSeedPutHeadDeleteCarryTheSeed) +{ + WriteSettings write_settings; + write_settings.object_storage_attempt_number = 3; + client->attempts_seen.clear(); + { + auto buffer = getWriteBuffer("seeded_put", write_settings); + buffer->write('A'); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + } + ASSERT_FALSE(client->attempts_seen.empty()); + EXPECT_EQ(client->attempts_seen.front(), 3u); + /// Seed 0 adds no header at all (the spec's rule for every verb but the read path). + client->attempts_seen.clear(); + { + auto buffer = getWriteBuffer("unseeded_put"); + buffer->write('A'); + getAsyncPolicy().setAutoExecute(true); + buffer->finalize(); + } + ASSERT_EQ(client->attempts_seen.size(), 1u); + EXPECT_FALSE(client->attempts_seen.front().has_value()); + + /// The native HEAD's seed: `S3ObjectStorage::tryGetObjectMetadataWithNativeToken`'s profile-aware + /// overload now forwards `request.attempt_number`, like every other verb here; this exercises the + /// seed-carrying layer directly -- `S3::getObjectInfoIfExists`, the same call + /// `tryGetObjectMetadataImpl` makes. + client->attempts_seen.clear(); + S3::getObjectInfoIfExists(*client, bucket, "seeded_head", /*version_id=*/{}, /*with_metadata=*/false, + /*with_tags=*/false, ObjectStorageRequestMode::Default, /*attempt_seed=*/4); + ASSERT_EQ(client->attempts_seen.size(), 1u); + EXPECT_EQ(client->attempts_seen.front(), 4u); + client->attempts_seen.clear(); + S3::getObjectInfoIfExists(*client, bucket, "unseeded_head"); + ASSERT_EQ(client->attempts_seen.size(), 1u); + EXPECT_FALSE(client->attempts_seen.front().has_value()); + + /// Conditional (single) and bulk DELETE: reachable now through `S3ObjectStorage`'s + /// `ObjectStorageControlRequest`-carrying overloads, which is what actually drives + /// `removeObjectIfTokenMatchesImpl`/`removeObjectsIfExistImpl` with a real nonzero seed, through the + /// object storage's own API rather than a lower-level free function. + (void)getContext(); // BlobStorageLogWriter::create falls back to the global context + auto delete_store = std::make_shared(); + delete_store->CreateBucket(bucket); + auto owned_delete_client = std::make_unique(delete_store); + MockS3::Client * delete_client = owned_delete_client.get(); + S3::URI delete_uri; + delete_uri.bucket = bucket; + auto delete_object_storage = std::make_shared( + std::move(owned_delete_client), + std::make_unique(), + delete_uri, + S3Capabilities{}, + ObjectStorageKeyGeneratorPtr{}, + "seed-delete-disk"); + + delete_client->attempts_seen.clear(); + delete_object_storage->removeObjectIfTokenMatches(StoredObject("unseeded-delete-key"), "etag-1"); + ASSERT_EQ(delete_client->attempts_seen.size(), 1u); + EXPECT_FALSE(delete_client->attempts_seen.front().has_value()); + + delete_client->attempts_seen.clear(); + delete_object_storage->removeObjectIfTokenMatches( + StoredObject("seeded-delete-key"), "etag-1", ObjectStorageControlRequest{.attempt_number = 3}); + ASSERT_EQ(delete_client->attempts_seen.size(), 1u); + EXPECT_EQ(delete_client->attempts_seen.front(), 3u); + + delete_client->attempts_seen.clear(); + delete_object_storage->removeObjectsIfExistUnderProfile({StoredObject("unseeded-bulk-key")}, ObjectStorageControlRequest{}); + ASSERT_EQ(delete_client->attempts_seen.size(), 1u); + EXPECT_FALSE(delete_client->attempts_seen.front().has_value()); + + delete_client->attempts_seen.clear(); + delete_object_storage->removeObjectsIfExistUnderProfile( + {StoredObject("seeded-bulk-key")}, ObjectStorageControlRequest{.attempt_number = 3}); + ASSERT_EQ(delete_client->attempts_seen.size(), 1u); + EXPECT_EQ(delete_client->attempts_seen.front(), 3u); +} + +TEST_F(CASWBS3Test, S3RequestAttemptSeedListPagesCarryTheSeed) +{ + /// Drives the seed through the public `iterate` overload a real caller (the CAS backend's LIST + /// primitive) uses, rather than the anonymous-namespace `S3IteratorAsync` directly -- that class is + /// an implementation detail of `S3ObjectStorage.cpp` and not reachable from a test in this file. + auto list_store = std::make_shared(); + list_store->CreateBucket(bucket); + auto owned_list_client = std::make_unique(list_store); + MockS3::Client * list_client = owned_list_client.get(); + S3::URI list_uri; + list_uri.bucket = bucket; + auto list_object_storage = std::make_shared( + std::move(owned_list_client), + std::make_unique(), + list_uri, + S3Capabilities{}, + ObjectStorageKeyGeneratorPtr{}, + "seed-list-disk"); + + auto & bucket_store = list_store->GetBucketStore(bucket); + for (int i = 0; i < 5; ++i) + bucket_store.PutObject(fmt::format("p/{}", i), "x"); + + /// Profile is left at Default (not SingleAttempt): that would route through + /// `clientForRetryProfile`'s single-attempt clone, whose `cloneWithConfigurationOverride` the mock + /// client does not override, and the test would stop exercising the mock entirely. + list_client->attempts_seen.clear(); + auto iterator = list_object_storage->iterate( + "p/", /*max_keys=*/2, /*with_tags=*/false, std::optional("p/0"), + ObjectStorageControlRequest{.attempt_number = 2}); + size_t seen = 0; + for (; iterator->isValid(); iterator->next()) + ++seen; + EXPECT_EQ(seen, 4u); + ASSERT_EQ(list_client->attempts_seen.size(), 2u); /// the initial page and one rebuilt page + EXPECT_EQ(list_client->attempts_seen[0], 2u); + EXPECT_EQ(list_client->attempts_seen[1], 2u); + + /// Seed 0 adds no header on either page. + list_client->attempts_seen.clear(); + auto unseeded_iterator = list_object_storage->iterate( + "p/", /*max_keys=*/2, /*with_tags=*/false, std::optional("p/0"), ObjectStorageControlRequest{}); + seen = 0; + for (; unseeded_iterator->isValid(); unseeded_iterator->next()) + ++seen; + EXPECT_EQ(seen, 4u); + ASSERT_EQ(list_client->attempts_seen.size(), 2u); + EXPECT_FALSE(list_client->attempts_seen[0].has_value()); + EXPECT_FALSE(list_client->attempts_seen[1].has_value()); +} + +#endif diff --git a/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp b/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp index fde269c4c999..356a6eed523f 100644 --- a/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp +++ b/src/Interpreters/ContentAddressedGarbageCollectionLog.cpp @@ -21,7 +21,8 @@ ColumnsDescription ContentAddressedGarbageCollectionLogElement::getColumnsDescri auto outcome_enum = std::make_shared(DataTypeEnum8::Values{ {"Unknown", static_cast(UNKNOWN)}, {"Success", static_cast(SUCCESS)}, {"NotALeader", static_cast(NOT_A_LEADER)}, {"Error", static_cast(FAILED)}, - {"Deferred", static_cast(DEFERRED)}}); + {"Deferred", static_cast(DEFERRED)}, {"Aborted", static_cast(ABORTED)}, + {"Stopped", static_cast(STOPPED)}}); auto trigger_enum = std::make_shared(DataTypeEnum8::Values{ {"Scheduled", static_cast(SCHEDULED)}, {"Manual", static_cast(MANUAL)}}); auto lc_string = std::make_shared(std::make_shared()); @@ -38,20 +39,21 @@ ColumnsDescription ContentAddressedGarbageCollectionLogElement::getColumnsDescri {"gc_id", std::make_shared(), "GC scheduler instance id (which mounter)."}, {"trigger", trigger_enum, "Scheduled (background tick) or Manual (SYSTEM command)."}, {"round", std::make_shared(), "GC round number (0 on Start)."}, - {"outcome", outcome_enum, "Unknown (Start) / Success (led, folded, and completed) / NotALeader (another replica holds the GC lease) / Deferred (led but took the skip-unchanged fast path -- no fold ran) / Error (the round threw)."}, + {"outcome", outcome_enum, "Unknown (Start) / Success (led, folded, and completed) / NotALeader (another replica holds the GC lease) / Deferred (led but took the skip-unchanged fast path -- no fold ran) / Aborted (the round threw a transient error -- backend unavailability, a lost lease, a concurrent leader -- and the next scheduled round retries) / Stopped (a transient error observed after the disk\'s teardown began: the round was cut short by a server shutdown or the storage\'s destructor, so neither waited for it; a correlation, not a cause -- a transient incident that started before the teardown is recorded the same way, and decommission does not arm the flag at all) / Error (the round threw a non-transient error -- during a teardown too)."}, {"candidates_marked", std::make_shared(), "Objects retired (marked) this round."}, {"objects_deleted", std::make_shared(), "Objects physically deleted this round."}, {"objects_absent", std::make_shared(), "Retire candidates found already absent."}, {"objects_replaced", std::make_shared(), "412-saves (a resurrection won the race)."}, {"objects_spared", std::make_shared(), "Candidates spared (in-degree > 0 at recheck)."}, - {"manifests_deleted", std::make_shared(), "Owner-removed manifest bodies physically deleted this round (counted separately from blob deletes, B11)."}, + {"manifests_deleted", std::make_shared(), "Owner-removed manifest bodies deleted or found already absent this round (a batch delete of write-once keys cannot tell the two apart), counted separately from blob deletes."}, {"entries_condemned", std::make_shared(), "Retired entries newly condemned this round (retired-cursor pipeline stage 1)."}, {"entries_graduated", std::make_shared(), "Retired entries newly floor-passed and republished delete_pending this round (stage 2; deleted the NEXT round)."}, {"entries_redeleted", std::make_shared(), "Pending exact-token blob deletes executed this round (stage 3)."}, {"fence_outs", std::make_shared(), "Expired mounts fenced out by this round's heartbeat floor."}, {"anomalies", std::make_shared(), "Fold clamps surfaced (and survived) this round; steady >0 warrants a look at the round log details."}, {"duration_ms", std::make_shared(), "Round wall-clock duration (Finish)."}, - {"error", std::make_shared(), "Exception text when outcome = Error."}, + {"error", std::make_shared(), "Exception text when outcome = Aborted, Stopped or Error. On a Stopped row it names the engine\'s refusal, not the teardown."}, + {"error_code", std::make_shared(), "Exception code when outcome = Aborted, Stopped or Error; 0 otherwise. The structured twin of `error`: key monitoring on this column, not on message text."}, {"ProfileEvents", std::make_shared(lc_string, std::make_shared()), "On a Start/Finish row: the per-round ProfileEvents delta (the Cas* counters and S3 events for this round). On a Phase row: THAT PHASE's delta, so `GROUP BY phase` over `ProfileEvents['S3ListObjects']` attributes the round's LIST budget to the phase that spent it. Empty on the `meta_pool_wait` row by construction — that phase's work runs on other threads (read its `phase_metrics` instead)."}, {"round_id", std::make_shared(), @@ -92,6 +94,7 @@ void ContentAddressedGarbageCollectionLogElement::appendToBlock(MutableColumns & columns[i++]->insert(anomalies); columns[i++]->insert(duration_ms); columns[i++]->insert(error); + columns[i++]->insert(error_code); { Map map; map.reserve(profile_events.size()); diff --git a/src/Interpreters/ContentAddressedGarbageCollectionLog.h b/src/Interpreters/ContentAddressedGarbageCollectionLog.h index 9cbdbd3525f6..65c231b76eac 100644 --- a/src/Interpreters/ContentAddressedGarbageCollectionLog.h +++ b/src/Interpreters/ContentAddressedGarbageCollectionLog.h @@ -15,7 +15,14 @@ struct ContentAddressedGarbageCollectionLogElement /// `DEFERRED`: the round acquired the GC lease and took the skip-unchanged fast path -- no fold, no /// pre-CAS deletes, no `gc/state` CAS. Kept distinct from `SUCCESS` so a query against this table can /// tell a round that genuinely folded and found nothing apart from one that never folded at all. - enum Outcome : int8_t { UNKNOWN = 1, SUCCESS = 2, NOT_A_LEADER = 3, FAILED = 4, DEFERRED = 5 }; + /// `ABORTED`: the round threw an exception whose code names a transient condition (backend + /// unavailability, a lost lease, a concurrent leader); the next scheduled round retries it. + /// `STOPPED`: a transient failure observed after the disk's teardown began -- the round was cut + /// short by a server shutdown or the storage's destructor, so neither had to wait for it; + /// `error` carries the engine's refusal. Decommission does not arm that flag and cannot produce + /// this outcome. + /// `FAILED` is everything else -- fail-closed, an unclassified error reads as real. + enum Outcome : int8_t { UNKNOWN = 1, SUCCESS = 2, NOT_A_LEADER = 3, FAILED = 4, DEFERRED = 5, ABORTED = 6, STOPPED = 7 }; enum Trigger : int8_t { SCHEDULED = 1, MANUAL = 2 }; time_t event_time = 0; @@ -42,6 +49,7 @@ struct ContentAddressedGarbageCollectionLogElement UInt64 anomalies = 0; /// fold clamps surfaced this round UInt64 duration_ms = 0; String error; + Int32 error_code = 0; /// exception code on an Aborted/Error FINISH; 0 otherwise std::map profile_events; /// per-round delta (FINISH); per-phase delta (PHASE) String round_id; /// correlator for every row of one round attempt diff --git a/src/Storages/MergeTree/DataPartsExchange.cpp b/src/Storages/MergeTree/DataPartsExchange.cpp index a3155848eaf1..f900ed260d01 100644 --- a/src/Storages/MergeTree/DataPartsExchange.cpp +++ b/src/Storages/MergeTree/DataPartsExchange.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -55,6 +56,9 @@ namespace FailPoints /// neither is reachable from configuration, so an integration test cannot produce them any other way. extern const char cas_relink_receiver_force_mechanism_failure[]; extern const char cas_relink_receiver_pause_before_confirm[]; + /// Stands in for a sender that predates the `cas_pool_uuid` response cookie: the offer is made + /// without naming the pool, and the receiver has to fall back on "the single advertised pool". + extern const char cas_relink_sender_omit_pool_cookie[]; } namespace MergeTreeSetting @@ -73,7 +77,7 @@ namespace ErrorCodes extern const int CHECKSUM_DOESNT_MATCH; extern const int INSECURE_PATH; extern const int LOGICAL_ERROR; - extern const int NETWORK_ERROR; + extern const int NO_REPLICA_HAS_PART; extern const int S3_ERROR; extern const int ZERO_COPY_REPLICATION_ERROR; } @@ -104,7 +108,7 @@ constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_INVALIDATED_SYSTEM_COLUMNS = 10 /// 10 is a version peers still advertise, and deleting the record of what it meant would leave the next /// reader unable to tell what an incoming 10 promises (a relink it will NOT confirm). [[maybe_unused]] constexpr auto REPLICATION_PROTOCOL_VERSION_WITH_CA_RELINK = 10; -/// CAS replication, publish-then-confirm (spec §wire-protocol). A relink offer is now accompanied by a +/// CAS replication, publish-then-confirm. A relink offer is now accompanied by a /// source token, and the endpoint answers a second, part-less request that asks whether that token is /// still exactly what the sender's ref names. A server advertising this version serves the confirm /// action; a receiver advertising it must confirm before it promotes. @@ -115,22 +119,24 @@ std::string getEndpointId(const std::string & node_id) return "DataPartsExchange:" + node_id; } -/// CAS replication 2b. The receiver advertises its target pool's identity under this request param so -/// the sender can decide whether a fetch-by-relink (same pool) is possible. +/// CAS replication 2b. The receiver advertises the pool ids of its candidate content-addressed disks +/// under this request param (one id, or several joined with ", ", see `encodeCasPoolAdvertise`) so the +/// sender can decide whether a fetch-by-relink (same pool) is possible. On the offer the same name is a +/// response cookie naming the pool the sender matched. constexpr auto CA_POOL_UUID_PARAM = "cas_pool_uuid"; /// Set on the response when the sender chose the relink path; the receiver then reads the relink payload /// (the opaque encoded PartManifest body — self-contained, see part_manifest_v2 below) instead of the /// byte stream. constexpr auto CA_RELINK_COOKIE = "cas_relink"; -/// All-tree task 7: the manifest is now self-contained (uuid.txt/metadata_version.txt are ordinary -/// manifest entries, task 6), so the wire payload dropped its trailing metadata_version field (the +/// The manifest is now self-contained (uuid.txt/metadata_version.txt are ordinary +/// manifest entries), so the wire payload dropped its trailing metadata_version field (the /// manifest bytes are now the ONLY field). Bumped from `part_manifest_v1` so a mixed-build pair (old /// sender, new receiver) does not try to parse the old two-field payload under the new one-field shape /// — the receiver rejects a cookie value it does not recognize and falls back to a byte fetch instead /// of desyncing on the wire format. constexpr auto CA_RELINK_COOKIE_VALUE = "part_manifest_v2"; -/// CAS fetch-by-relink, publish-then-confirm (spec §wire-protocol). Three names make up the second +/// CAS fetch-by-relink, publish-then-confirm. Three names make up the second /// request of the handshake. /// /// The request parameter both selects the confirm action and carries its only argument: the opaque @@ -157,7 +163,7 @@ constexpr auto CA_CONFIRM_ANSWER_UNPROVEN = "unproven"; /// Resolve a disk to the content-addressed exchange facade, or nullptr if the disk is not CA. The /// cast targets the purpose-built INTERFACE (IContentAddressedExchange), never the concrete -/// metadata-storage class (M-W design section 4). Used by both the relink sender (the part's +/// metadata-storage class. Used by both the relink sender (the part's /// disk) and the relink receiver (the target disk). IContentAddressedExchange * tryGetContentAddressedExchange(const DiskPtr & disk) { @@ -220,7 +226,7 @@ CasConfirmAnswer Service::resolveContentAddressedConfirm( const String & part_name, const String & manifest_ref_text) const { - /// CAS fetch-by-relink, publish-then-confirm (spec §confirm-primitive). The receiver's own `+1` is + /// CAS fetch-by-relink, publish-then-confirm. The receiver's own `+1` is /// already durable when this runs; a `Yes` is what authorizes it to promote a part whose blobs are /// protected only by THIS server's committed binding of that exact manifest. Every field below comes /// from a remote peer, so nothing here is trusted beyond being used as a lookup key. @@ -230,31 +236,32 @@ CasConfirmAnswer Service::resolveContentAddressedConfirm( /// Routing. A pool UUID identifies the shared pool, not the mount: every server root writing into it /// reports the same one, so the namespace's owner decides which instance may answer. EXACTLY one /// match is required — zero means this table has no such disk, several mean the question is - /// ambiguous, and both are `Unknown` rather than a guess. - const IContentAddressedExchange * matched = nullptr; - DiskPtr matched_disk; + /// ambiguous, and both are `Unknown` rather than a guess. A cache disk over a content-addressed disk + /// shares the base disk's exchange object, so the two are one mount and count once. + std::vector routing; + Disks routing_disks; for (const auto & disk : data.getDisks()) { const auto * ca_meta = tryGetContentAddressedExchange(disk); - if (!ca_meta || ca_meta->getPoolUUID() != pool_uuid || !ca_meta->ownsNamespace(server_root_id, root_namespace)) + if (!ca_meta) continue; - if (matched) - return CasConfirmAnswer::Unknown; - matched = ca_meta; - matched_disk = disk; + routing.push_back({ca_meta, ca_meta->getPoolUUID(), ca_meta->ownsNamespace(server_root_id, root_namespace)}); + routing_disks.push_back(disk); } - if (!matched) + const auto routed = resolveConfirmRoutingCandidate(routing, pool_uuid); + if (!routed) return CasConfirmAnswer::Unknown; + const IContentAddressedExchange * matched = tryGetContentAddressedExchange(routing_disks[*routed]); - /// Gate 0 — the part-anchored fast filter. It is an AVAILABILITY filter and never a proof (spec - /// §confirm-primitive, demoted in rev.5): `rollbackDeletingParts` puts a part back to `Outdated` + /// Gate 0 — the part-anchored fast filter. It is an AVAILABILITY filter and never a proof: + /// `rollbackDeletingParts` puts a part back to `Outdated` /// after a failed filesystem removal, and the in-memory part path is deliberately not updated by a /// `delete_tmp_*` rename, so an `Active`/`Outdated` part object authorizes nothing. What it buys is /// a cheap `No` that costs no ledger work; every `Yes` is earned by gate 1 alone. /// /// `Deleting` is excluded by the state filter, an unknown name yields no part at all, and a part of - /// this name living on ANOTHER disk is rejected explicitly — `MOVE ... TO DISK` leaves a same-name - /// `Active` part behind on the destination disk, and only the instance the token routed to may be + /// this name living on ANOTHER mount is rejected explicitly — `MOVE ... TO DISK` leaves a same-name + /// `Active` part behind on the destination disk, and only the mount the token routed to may be /// the one the confirm is about. The parts set is read under its own lock, which /// `getPartIfExists` takes and releases, and the part reference is dropped before any ledger lock. { @@ -263,7 +270,14 @@ CasConfirmAnswer Service::resolveContentAddressedConfirm( return CasConfirmAnswer::Unknown; const auto part = data.getPartIfExists( *part_info, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); - if (!part || part->getDataPartStorage().getDiskName() != matched_disk->getName()) + if (!part) + return CasConfirmAnswer::No; + /// Compared by mount, not by disk name: a base disk and its cache wrapper are two names for one + /// exchange object, and a part living on either of them is a part of the mount the token routed + /// to. A different mount (a distinct exchange object) is the "another disk" this gate rejects. + const auto * part_exchange = tryGetContentAddressedExchange( + data.getStoragePolicy()->tryGetDiskByName(part->getDataPartStorage().getDiskName())); + if (part_exchange != matched) return CasConfirmAnswer::No; } @@ -304,7 +318,7 @@ void Service::answerContentAddressedConfirm(const String & token_text, HTTPServe void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuffer & out, HTTPServerResponse & response) { - /// CAS fetch-by-relink, publish-then-confirm (spec §wire-protocol): the second request of the + /// CAS fetch-by-relink, publish-then-confirm: the second request of the /// handshake, dispatched before `part` is required because a confirm carries none — the part name /// is inside the token. Authentication parity with the fetch is inherent: the shared handler /// authenticates before it dispatches to any endpoint. @@ -391,9 +405,9 @@ void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuf writeBinary(projections.size(), out); } - /// CAS replication 2b — fetch-by-relink (spec §4). If the part is on a content-addressed disk and - /// the receiver advertised a `cas_pool_uuid` equal to THIS server's own pool_uuid - /// (same shared pool), send only the part's content id + the mutable header — no file bytes — so + /// CAS replication — fetch-by-relink. If the part is on a content-addressed disk and + /// the pool of the disk this part sits on is among the pools the receiver advertised in + /// `cas_pool_uuid`, send only the part's content id + the mutable header — no file bytes — so /// the receiver can "fetch" by publishing its own ref to the blobs already in the shared pool. /// Strictly gated on a matching pool_uuid: a non-CA part, a CA part on a different pool, or a /// receiver without the capability all fall through to the unchanged byte path below. @@ -407,26 +421,39 @@ void Service::processQuery(const HTMLForm & params, ReadBufferPtr body, WriteBuf if (client_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_CA_CONFIRM && part->getDataPartStorage().isContentAddressed()) { - const String receiver_pool_uuid = parse(params.get(CA_POOL_UUID_PARAM, "")); + /// The receiver advertises every pool its storage policy has a writable content-addressed + /// disk for, as one list; this server's decision stays local — is the pool of the disk THIS + /// part sits on among them. The matched pool goes back as a cookie so a receiver with several + /// pools can place the part on that pool's disk instead of guessing which disk the offer is for. + const Strings receiver_pools = decodeCasPoolAdvertise(parse(params.get(CA_POOL_UUID_PARAM, ""))); DiskPtr part_disk = data.getStoragePolicy()->tryGetDiskByName(part->getDataPartStorage().getDiskName()); auto * ca_meta = tryGetContentAddressedExchange(part_disk); - if (ca_meta && !receiver_pool_uuid.empty() && receiver_pool_uuid == ca_meta->getPoolUUID()) + const String matched_pool = ca_meta ? ca_meta->getPoolUUID() : String{}; + if (ca_meta && !matched_pool.empty() + && std::find(receiver_pools.begin(), receiver_pools.end(), matched_pool) != receiver_pools.end()) { auto offer = ca_meta->getRelinkOffer(part->getDataPartStorage().getRelativePath()); if (offer) { LOG_DEBUG(log, "Sending part {} by relink (content-addressed, shared pool {}), manifest payload {} bytes", - part_name, receiver_pool_uuid, offer->manifest_bytes.size()); + part_name, matched_pool, offer->manifest_bytes.size()); response.addCookie({CA_RELINK_COOKIE, CA_RELINK_COOKIE_VALUE}); - /// The source token for the confirm request the receiver makes before it promotes - /// (spec §wire-protocol). It always accompanies the offer, and its ABSENCE is what + /// The source token for the confirm request the receiver makes before it promotes. + /// It always accompanies the offer, and its ABSENCE is what /// tells a confirm-capable receiver that this sender predates the handshake. response.addCookie({CA_CONFIRM_TOKEN_COOKIE, offer->confirm_token}); - /// The relink payload (B7 part_manifest_v2, all-tree task 7): the opaque encoded + /// Which of the advertised pools this offer is for. A receiver with one pool does not + /// need it (an offer can only be for that pool); the failpoint stands in for a sender + /// that predates the cookie. + bool omit_pool_cookie = false; + fiu_do_on(FailPoints::cas_relink_sender_omit_pool_cookie, { omit_pool_cookie = true; }); + if (!omit_pool_cookie) + response.addCookie({CA_POOL_UUID_PARAM, matched_pool}); + /// The relink payload (`part_manifest_v2`): the opaque encoded /// PartManifest body (the receiver decodes it, ignores the sender identity, and /// stages its OWN local manifest over the shared-pool blobs; the legacy part_id wire /// field carries it). Self-contained: uuid.txt/metadata_version.txt are ordinary - /// manifest entries now (task 6), so no separate mutable-header field is sent. + /// manifest entries now, so no separate mutable-header field is sent. writeStringBinary(offer->manifest_bytes, out); data.addLastSentPart(part->info); return; @@ -697,11 +724,17 @@ std::pair Fetcher::fetchSelected if (disk) LOG_TRACE(log, "Will fetch to disk {} with type {}", disk->getName(), disk->getDataSourceDescription().toString()); - /// CAS replication 2b — fetch-by-relink (spec §4). Advertise this replica's target content-addressed - /// pool identity so a same-pool sender can relink instead of streaming bytes. The target disk is the - /// provided one if it is CA, else the first CA disk among the table's disks. A non-CA fetch adds - /// nothing here and is byte-for-byte unchanged. - /// Gated on `allow_ca_relink` alone (B66b). That flag is the RECURSION BRAKE and nothing else: not + /// CAS fetch-by-relink: advertise the content-addressed pools this fetch may land in, so a sender + /// holding the part in one of them relinks instead of streaming bytes. With a caller-supplied disk + /// that is its pool alone (the disk is the caller's contract and is never overridden); otherwise it + /// is every content-addressed disk of the table's storage policy that is not read-only, in policy + /// order. The sender names the pool it matched in a response cookie, and the reservation below then + /// goes to THAT pool's disk — ahead of the policy's volume order and of any TTL move rule, because a + /// part that is already in the pool must never travel as bytes merely because the policy would have + /// put it elsewhere (the mover carries it to a TTL destination afterwards). A pool disk that is not + /// live is still the target: the relink's own write gate refuses it, the fetch fails and the queue + /// retries — never a quiet landing on another disk. A non-CA fetch adds nothing here. + /// Gated on `allow_ca_relink` alone. That flag is the RECURSION BRAKE and nothing else: not /// advertising is what makes the sender stream bytes, so every same-sender byte re-request below /// clears it, and a persistent relink-mechanism failure therefore costs exactly one relink attempt. /// The gate used to be `try_zero_copy && !to_detached`, and BOTH halves were accidents of that same @@ -709,26 +742,34 @@ std::pair Fetcher::fetchSelected /// because the relink path staged at the ACTIVE part path and ignored `to_detached`. `to_detached` /// is now a parameter of `relinkPartToDisk` (it stages under the `detached/` parent), and /// `try_zero_copy` goes back to meaning real zero-copy only. - String advertised_pool_uuid; + Strings advertised_pools; + std::vector ca_candidates; + Disks ca_candidate_disks; if (allow_ca_relink) { - if (auto * ca_meta = tryGetContentAddressedExchange(disk)) + if (disk) { - advertised_pool_uuid = ca_meta->getPoolUUID(); - uri.addQueryParameter(CA_POOL_UUID_PARAM, advertised_pool_uuid); + if (auto * ca_meta = tryGetContentAddressedExchange(disk)) + advertised_pools.push_back(ca_meta->getPoolUUID()); } - else if (!disk) + else { for (const auto & data_disk : data.getDisks()) { - if (auto * ca_disk_meta = tryGetContentAddressedExchange(data_disk)) - { - advertised_pool_uuid = ca_disk_meta->getPoolUUID(); - uri.addQueryParameter(CA_POOL_UUID_PARAM, advertised_pool_uuid); - break; - } + auto * ca_disk_meta = tryGetContentAddressedExchange(data_disk); + if (!ca_disk_meta) + continue; + ca_candidates.push_back({data_disk->getName(), ca_disk_meta->getPoolUUID(), data_disk->isReadOnly()}); + ca_candidate_disks.push_back(data_disk); + if (!data_disk->isReadOnly()) + advertised_pools.push_back(ca_disk_meta->getPoolUUID()); } } + const String advertise = encodeCasPoolAdvertise(advertised_pools); + if (!advertise.empty()) + uri.addQueryParameter(CA_POOL_UUID_PARAM, advertise); + /// The deduplicated form is what "the single advertised pool" is measured against below. + advertised_pools = decodeCasPoolAdvertise(advertise); } Strings capability; @@ -793,6 +834,24 @@ std::pair Fetcher::fetchSelected int server_protocol_version = parse(in->getResponseCookie("server_protocol_version", "0")); String remote_fs_metadata = parse(in->getResponseCookie("remote_fs_metadata", "")); + /// The relink offer, if any, is already visible: response cookies arrive with the headers, before any + /// body field is consumed. Resolve the forced disk NOW, so the reservation below goes to it and the + /// body reads keep their order. `offered_pool` is what the relink block later checks the chosen disk + /// against; with a caller-supplied disk there is nothing to force and that check is all there is. + const String ca_relink = parse(in->getResponseCookie(CA_RELINK_COOKIE, "")); + String offered_pool; + DiskPtr forced_ca_disk; + if (!ca_relink.empty()) + { + const String offered_pool_cookie = parse(in->getResponseCookie(CA_POOL_UUID_PARAM, "")); + auto choice = chooseForcedCaDisk( + static_cast(disk), ca_candidates, ca_candidate_disks, advertised_pools, offered_pool_cookie, part_name, log); + offered_pool = std::move(choice.offered_pool); + /// From here on the target is decided: every `!disk` reservation branch below is skipped. + if (choice.disk) + disk = forced_ca_disk = choice.disk; + } + DiskPtr preffered_disk = disk; if (!preffered_disk) @@ -813,6 +872,13 @@ std::pair Fetcher::fetchSelected { readBinary(sum_files_size, *in); + if (forced_ca_disk) + { + /// An object-storage disk reports no capacity, so this cannot decline for space; if it ever + /// does, the loud NOT_ENOUGH_SPACE is the right outcome — the part is not re-placed elsewhere. + reservation = MergeTreeData::reserveSpace(sum_files_size, forced_ca_disk); + } + if (server_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_PARTS_SIZE_AND_TTL_INFOS) { IMergeTreeDataPart::TTLInfos ttl_infos; @@ -885,21 +951,21 @@ std::pair Fetcher::fetchSelected if (server_protocol_version >= REPLICATION_PROTOCOL_VERSION_WITH_PARTS_PROJECTION) readBinary(projections, *in); - /// CAS replication 2b — fetch-by-relink (spec §4; B7 part_manifest_v2, all-tree task 7). The sender - /// chose to relink: it sent only the part's encoded PartManifest body, no file bytes. Build the part - /// by staging this server's OWN local manifest over the blobs already in the shared pool (adopt-by-hash - /// -> revalidate -> promote inside adoptPartFromManifest) — self-contained since task 6 routed - /// uuid.txt/metadata_version.txt through the content path, so there is no separate mutable header to + /// CAS replication — fetch-by-relink (`part_manifest_v2`). The sender + /// chose to relink: it sent only the part's encoded PartManifest body, no file bytes, and the + /// reservation above already went to the offered pool's disk. Build the part by staging this + /// server's OWN local manifest over the blobs already in the shared pool (adopt-by-hash -> revalidate + /// -> promote inside adoptPartFromManifest) — self-contained because uuid.txt and + /// metadata_version.txt travel through the content path, so there is no separate mutable header to /// reconstruct. If the relink is not possible (blob missing/condemned — a transient or a /// genuinely-different pool the cheap pre-filter let through, or a mixed-build pair offering an /// unrecognized cookie value), fall back to a normal byte fetch by re-requesting WITHOUT relink. - String ca_relink = parse(in->getResponseCookie(CA_RELINK_COOKIE, "")); if (!ca_relink.empty()) { /// Re-request without the relink capability: pass the SAME (CA) disk but disable zero-copy/relink /// so the sender streams bytes; on CA the downloaded files content-address and dedup. /// - /// THE RECURSION BRAKE (B66b). `allow_ca_relink=false` is what bounds this: the re-request does + /// THE RECURSION BRAKE. `allow_ca_relink=false` is what bounds this: the re-request does /// not advertise the pool identity, so the sender cannot offer relink again, so this lambda /// cannot be reached a second time for the same fetch. Before relink had its own capability the /// brake was implicit in `try_zero_copy=false`; with the two decoupled it has to be spelled out, @@ -908,9 +974,9 @@ std::pair Fetcher::fetchSelected /// and recurses without bound. The failures it actually bounds are the ones that leave the CA /// disk resolved and matching: a mixed build offering an unrecognized cookie value, a sender that /// predates the confirm handshake, an undecodable manifest, a local ref conflict. (The - /// reservation-outside-the-pool exit below is bounded twice over — it re-requests with the - /// non-CA disk it resolved, which cannot advertise anything either way — so do not read that one - /// as evidence that the brake is redundant.) + /// no-disk-takes-it exit below is bounded twice over — it re-requests with the disk the ordinary + /// reservation resolved, which is outside the pool and cannot advertise it — so do not read that + /// one as evidence that the brake is redundant.) auto fall_back_to_byte_fetch = [&] { temporary_directory_lock = {}; @@ -930,12 +996,20 @@ std::pair Fetcher::fetchSelected return fall_back_to_byte_fetch(); } + /// The disk is the forced one, so this holds by construction; it stays a real exit rather than + /// an assertion because it is also how an offer for a pool this policy has no disk for (no forced + /// disk, ordinary reservation) and a caller-supplied disk outside the pool leave the relink path. auto * chosen_ca = tryGetContentAddressedExchange(disk); - if (!chosen_ca || chosen_ca->getPoolUUID() != advertised_pool_uuid) + if (!chosen_ca || offered_pool.empty() || chosen_ca->getPoolUUID() != offered_pool) { - LOG_INFO(log, "Part {} was offered by relink for content-addressed pool '{}', but reservation landed " - "outside the advertised pool on disk {} (chosen pool: '{}'); falling back to a byte fetch", - part_name, advertised_pool_uuid, disk->getName(), chosen_ca ? chosen_ca->getPoolUUID() : ""); + if (offered_pool.empty()) + LOG_INFO(log, "Part {} was offered by relink, but the offer does not name one of the {} advertised " + "content-addressed pool(s) (cookie '{}'); falling back to a byte fetch onto disk {}", + part_name, advertised_pools.size(), parse(in->getResponseCookie(CA_POOL_UUID_PARAM, "")), disk->getName()); + else + LOG_INFO(log, "Part {} was offered by relink for content-addressed pool '{}', but no disk of this table's " + "storage policy takes it (chosen disk {}, pool '{}'); falling back to a byte fetch", + part_name, offered_pool, disk->getName(), chosen_ca ? chosen_ca->getPoolUUID() : ""); return fall_back_to_byte_fetch(); } @@ -943,7 +1017,7 @@ std::pair Fetcher::fetchSelected readStringBinary(sender_manifest_bytes, *in); assertEOF(*in); - /// Publish-then-confirm (spec §core-idea) happens inside `relinkPartToDisk`, including the second + /// Publish-then-confirm happens inside `relinkPartToDisk`, including the second /// interserver request; the token cookie is the sender's offer identity and is opaque here. A /// `nullptr` means the mechanism cannot work but the sender still has the part, so the byte /// re-request below is sound; a THROW means the source did not prove the binding, and the whole @@ -1355,8 +1429,14 @@ MergeTreeData::MutableDataPartPtr Fetcher::downloadPartToDisk( /// 3. THE CONFIRM DID NOT PROVE THE SOURCE: an `unproven` answer, an absent answer cookie, a transport /// failure, a timeout. All one outcome, deliberately (`CasConfirmAnswer`: only `yes` authorizes). /// `+1`: durable, then released by `abort`. Action: THROW a locally generated retry-later -/// `NETWORK_ERROR` naming the source and the part -- never `nullptr`, because a byte re-request goes -/// back to the very source whose state is in doubt. +/// `NO_REPLICA_HAS_PART` naming the source and the part -- never `nullptr`, because a byte +/// re-request goes back to the very source whose state is in doubt. That code, deliberately: +/// both queue executors (`processQueueEntry`, `ReplicatedMergeMutateTaskBase::executeStep`) +/// demote it to INFO with no stack trace -- a refusal is the designed outcome of racing a source +/// whose ref moved on, not a network fault, and logging it as an error invites exactly that +/// misdiagnosis -- yet, unlike `ABORTED`, it still records the exception on the queue entry, so a +/// refusal storm stays visible in `system.replication_queue`. It is also the one fetch-transient +/// code the stateless corpus already tolerates in `part_log` checks (e.g. `02265_column_ttl`). /// Lose a part? No -- the queue stores the exception, backs off, and re-executes the entry, which /// recomputes the source and the covering-part discovery. The fetch is postponed, not dropped. /// Double-promote? No -- `abort` appends the exact precommit removal and no committed ref exists. @@ -1380,7 +1460,8 @@ MergeTreeData::MutableDataPartPtr Fetcher::downloadPartToDisk( /// `+1`: still owed -- the handle attempts its abandon, which is REJECTED by the state machine if /// the promote in fact landed (a promoted binding is no longer a precommit), so no committed ref is /// ever undone here. -/// Action: THROW the retry-later `NETWORK_ERROR`, as row 3 -- returning `nullptr` is the one thing +/// Action: THROW the retry-later `NO_REPLICA_HAS_PART`, as row 3 -- returning `nullptr` is the +/// one thing /// that must not happen, because a byte fetch would publish the part a SECOND time over a relink /// that may already be committed. /// Lose a part? No -- retry-later, as row 3. Double-promote? No -- nothing is published on this exit. @@ -1398,7 +1479,7 @@ MergeTreeData::MutableDataPartPtr Fetcher::downloadPartToDisk( /// the source itself. `adoptPartFromManifest` used to collapse the two by catching every `Exception` /// and returning `false`. /// -/// B66b — WHAT CHANGES WHEN THE TARGET IS `detached/`. Every row above still holds, and the two columns +/// WHAT CHANGES WHEN THE TARGET IS `detached/`. Every row above still holds, and the two columns /// that matter are unchanged in every one of them, but two rows hold for a DIFFERENT reason and that /// difference is worth stating rather than rediscovering: /// @@ -1421,11 +1502,11 @@ MergeTreeData::MutableDataPartPtr Fetcher::downloadPartToDisk( /// keeps a failed detached relink from ever being visible as a live part: the abandoned precommit and /// the abandoned staging directory both live in the detached ref space. /// -/// What a `yes` does NOT prove: `CaRelinkConfirmCore.tla` config `_sab_holeylist` shows that with every -/// confirm rule intact and one incomplete listing page permitted, `ConfirmedRelinkNeverDangles` still -/// breaks (BACKLOG `{#list-as-journal-dataloss-2026-07-25}`). A confirmed relink is therefore NOT proven -/// dangle-free; a `yes` means only "the source still holds exactly this manifest right now", which is -/// what closes the codex-6 handoff window and nothing more. +/// What a `yes` does NOT prove: formal modelling of this protocol found that even with every confirm +/// rule intact, one incomplete storage LIST page during a GC fold is enough to let a confirmed relink's +/// blobs be reclaimed anyway. A confirmed relink is therefore NOT proven dangle-free; a `yes` means only +/// "the source still holds exactly this manifest right now", which closes the window between this +/// receiver's publish and the source's answer, and nothing more. MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( const String & part_name, const String & tmp_prefix, @@ -1476,7 +1557,7 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( }); /// Stage under the tmp-fetch dir OF THE TARGET PARENT — the table dir, or `TABLE/detached` when - /// the caller asked for a detached fetch (B66b). The parent is composed exactly as + /// the caller asked for a detached fetch. The parent is composed exactly as /// `downloadPartToDisk` composes it, so the two fetch paths put a part in the same place and the /// caller's finalization is unchanged: `renameTempPartAndReplace`'s moveDirectory(tmp-fetch_ /// -> ) for the active path, `renameTo(detached/)` for the detached one. Both are ref @@ -1495,16 +1576,16 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( LOG_DEBUG(log, "Relinking part {} (staged as {}) onto content-addressed disk {} from a {}-byte transferred manifest.", part_name, part_path, disk->getName(), sender_manifest_bytes.size()); - /// T1 — PUBLISH. Adopt-from-manifest and precommit, stopping short of the promote (B7 - /// part_manifest_v2, all-tree task 7): the receiver decodes the transferred body and stages its OWN + /// T1 — PUBLISH. Adopt-from-manifest and precommit, stopping short of the promote (`part_manifest_v2`): + /// the receiver decodes the transferred body and stages its OWN /// local manifest over the shared-pool blobs (adopt-by-hash). Self-contained: - /// uuid.txt/metadata_version.txt are ordinary entries in the transferred manifest (task 6), so there + /// uuid.txt/metadata_version.txt are ordinary entries in the transferred manifest, so there /// is no sidecar to reconstruct. Trust boundary is the interserver channel, as for a normal part /// fetch — see `prepareAdoptFromManifest`. /// /// The order is the whole protocol. This `+1` must be DURABLE before the source is asked anything, /// because the question "do you still hold it?" only excludes a later removal if the receiver's own - /// reference is already in the ref log when that removal is appended (spec §correctness). Asking + /// reference is already in the ref log when that removal is appended. Asking /// first and publishing after would prove nothing about the interval in between. What it does NOT /// establish is that every subsequent GC fold OBSERVES that reference -- see "What a `yes` does NOT /// prove" above; ordering is necessary here, not sufficient. @@ -1526,10 +1607,10 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( /// Test-only, and this is the ONE seam worth injecting on the whole path: it opens the window the /// protocol exists to make safe. The receiver's `+1` is durable and its release is armed, and the - /// source has not been asked anything yet, so a test that holds the fetch here can do to the source - /// exactly what codex-6 described — merge the part away, run GC to fixpoint — and then observe both - /// halves of the contract: the source's blobs survive the round (this receiver's binding protects - /// them) and the confirm that follows refuses to authorize a promote (the binding it named is gone). + /// source has not been asked anything yet, so a test that holds the fetch here can merge the part + /// away on the source and run GC to fixpoint, then observe both halves of the contract: the source's + /// blobs survive the round (this receiver's binding protects them) and the confirm that follows + /// refuses to authorize a promote (the binding it named is gone). FailPointInjection::pauseFailPoint(FailPoints::cas_relink_receiver_pause_before_confirm); /// T2 — CONFIRM. One read-only interserver question, aimed at the endpoint copied out of the fetch @@ -1585,9 +1666,11 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( { /// Taxonomy row 3. Locally generated on purpose — nothing here is the source's error to report — /// and thrown rather than returned, because the one recovery that is NOT sound after this is a - /// byte re-request to the same source. `NETWORK_ERROR` puts it in the retry-later class, so the - /// queue stores it, backs off, and re-selects on re-execution. - throw Exception(ErrorCodes::NETWORK_ERROR, + /// byte re-request to the same source. `NO_REPLICA_HAS_PART` puts it in the retry-later class + /// (the queue stores it, backs off, and re-selects on re-execution) and both queue executors + /// demote it to INFO without a stack trace -- see the taxonomy, row 3, for why this refusal is + /// an ordinary outcome rather than a fault. + throw Exception(ErrorCodes::NO_REPLICA_HAS_PART, "Source {} did not prove it still holds the manifest it offered for part {} by relink; " "the relink is abandoned and the fetch will be retried later", fetch_uri.getHost(), part_name); @@ -1606,7 +1689,7 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( /// second time over a relink that may already be committed. Thrown in the retry-later class /// instead, exactly as an unproven confirm is (row 3) -- the queue stores it, backs off, and /// re-executes, by which time the ref lane has resolved the ambiguity one way or the other. - throw Exception(ErrorCodes::NETWORK_ERROR, + throw Exception(ErrorCodes::NO_REPLICA_HAS_PART, "Relink of part {} from {} could not be resolved: the promotion may or may not have " "committed, so the bytes must NOT be fetched; the fetch will be retried later", part_name, fetch_uri.getHost()); diff --git a/src/Storages/MergeTree/DataPartsExchange.h b/src/Storages/MergeTree/DataPartsExchange.h index 5bec506eac21..3258833a91a0 100644 --- a/src/Storages/MergeTree/DataPartsExchange.h +++ b/src/Storages/MergeTree/DataPartsExchange.h @@ -113,8 +113,12 @@ class Fetcher final : private boost::noncopyable const String & tmp_prefix_ = "", std::optional * tagger_ptr = nullptr, bool try_zero_copy = true, + /// The target disk when the CALLER has already decided it (zero-copy `MOVE` re-fetching a shared + /// part onto the move's destination); never overridden. When absent, a content-addressed relink + /// offer decides the disk — the policy disk on the sender's pool — ahead of the storage policy's + /// own placement; otherwise the ordinary reservation does. DiskPtr dest_disk = nullptr, - /// CAS fetch-by-relink (spec §B66b): may this request advertise its content-addressed pool + /// CAS fetch-by-relink: may this request advertise its content-addressed pool /// identity, i.e. may the sender answer with a relink offer instead of the part's bytes? /// /// It is a capability of its own rather than a rider on `try_zero_copy`, and it carries the @@ -156,14 +160,14 @@ class Fetcher final : private boost::noncopyable ThrottlerPtr throttler, bool sync); - /// CAS replication 2b — fetch-by-relink (spec §4), publish-then-confirm (spec §core-idea). Build a + /// CAS replication — fetch-by-relink, publish-then-confirm. Build a /// part WITHOUT downloading any bytes by publishing this server's own ref to the blobs already in the /// shared content-addressed pool. Stages the ref under the tmp-fetch dir of the target parent — the - /// table dir, or `detached/` when `to_detached` (B66b) — so the caller's finalization re-keys it to + /// table dir, or `detached/` when `to_detached` — so the caller's finalization re-keys it to /// the final part name, exactly as for a byte-fetched part: `renameTempPartAndReplace` for the /// active path, `renameTo(detached/)` for the detached one. Then it ASKS THE SOURCE whether it /// still holds exactly the manifest it offered, and only then promotes and loads the part. - /// Self-contained (all-tree task 7): the transferred manifest alone is enough to rebuild the part — + /// Self-contained: the transferred manifest alone is enough to rebuild the part — /// no separate uuid/metadata_version wire fields to reconstruct as a sidecar. /// /// The whole failure taxonomy lives at the definition; the two outcomes a CALLER must distinguish: diff --git a/src/Storages/MergeTree/DataPartsExchangeCasRouting.cpp b/src/Storages/MergeTree/DataPartsExchangeCasRouting.cpp new file mode 100644 index 000000000000..e5071e8cc4f8 --- /dev/null +++ b/src/Storages/MergeTree/DataPartsExchangeCasRouting.cpp @@ -0,0 +1,147 @@ +#include + +/// libfiu's header wraps a C11 include in `extern "C"`; pulling in the real C++ +/// first (transitively, via logger_useful.h) keeps that redefinition from landing inside the extern "C" +/// block, which is what FailPoint.h being the first standard-library-touching include here would do. +#include +#include + +#include + +#include + +#include + +namespace DB +{ +namespace FailPoints +{ + extern const char cas_relink_receiver_drop_forced_disk[]; +} +} + +namespace DB::DataPartsExchange +{ + +namespace +{ +const String CAS_POOL_ADVERTISE_DELIMITER = ", "; +} + +String encodeCasPoolAdvertise(Strings pool_uuids) +{ + std::erase_if(pool_uuids, [](const String & id) { return id.empty(); }); + ::sort(pool_uuids.begin(), pool_uuids.end()); + pool_uuids.erase(std::unique(pool_uuids.begin(), pool_uuids.end()), pool_uuids.end()); + return boost::algorithm::join(pool_uuids, CAS_POOL_ADVERTISE_DELIMITER); +} + +Strings decodeCasPoolAdvertise(const String & text) +{ + Strings pools; + if (text.empty()) + return pools; + + size_t pos_start = 0; + while (true) + { + const size_t pos_end = text.find(CAS_POOL_ADVERTISE_DELIMITER, pos_start); + if (pos_end == String::npos) + { + pools.push_back(text.substr(pos_start)); + return pools; + } + pools.push_back(text.substr(pos_start, pos_end - pos_start)); + pos_start = pos_end + CAS_POOL_ADVERTISE_DELIMITER.size(); + } +} + +String resolveOfferedCasPool(const Strings & advertised_pools, const String & offered_pool_cookie) +{ + if (!offered_pool_cookie.empty()) + { + /// A cookie naming a pool this receiver did not advertise is not an answer to its question. In + /// particular the byte re-request after a failed relink advertises NOTHING, and a peer that + /// offers a relink anyway must not be able to re-enter the relink path through the cookie. + if (std::find(advertised_pools.begin(), advertised_pools.end(), offered_pool_cookie) != advertised_pools.end()) + return offered_pool_cookie; + return {}; + } + if (advertised_pools.size() == 1) + return advertised_pools.front(); + return {}; +} + +std::optional resolveForcedCaCandidate( + const std::vector & candidates, + const Strings & advertised_pools, + const String & offered_pool_cookie) +{ + const String offered_pool = resolveOfferedCasPool(advertised_pools, offered_pool_cookie); + if (offered_pool.empty()) + return std::nullopt; + + for (size_t i = 0; i < candidates.size(); ++i) + { + const auto & candidate = candidates[i]; + if (!candidate.read_only && !candidate.pool_uuid.empty() && candidate.pool_uuid == offered_pool) + return i; + } + return std::nullopt; +} + +ForcedCaDiskChoice chooseForcedCaDisk( + bool caller_supplied_disk, + const std::vector & candidates, + const Disks & candidate_disks, + const Strings & advertised_pools, + const String & offered_pool_cookie, + const String & part_name, + LoggerPtr log) +{ + ForcedCaDiskChoice result; + result.offered_pool = resolveOfferedCasPool(advertised_pools, offered_pool_cookie); + if (caller_supplied_disk) + return result; + + auto chosen = resolveForcedCaCandidate(candidates, advertised_pools, offered_pool_cookie); + fiu_do_on(FailPoints::cas_relink_receiver_drop_forced_disk, + { + LOG_INFO(log, "Failpoint cas_relink_receiver_drop_forced_disk: forgetting the forced disk for part {}", part_name); + chosen.reset(); + }); + if (chosen) + { + result.disk = candidate_disks[*chosen]; + LOG_DEBUG(log, "Part {} is offered by relink for content-addressed pool {}; placing it on disk {} " + "ahead of the storage policy's volume order and TTL rules", part_name, result.offered_pool, candidates[*chosen].disk_name); + } + return result; +} + +std::optional resolveConfirmRoutingCandidate( + const std::vector & candidates, + const String & pool_uuid) +{ + if (pool_uuid.empty()) + return std::nullopt; + + std::optional matched; + for (size_t i = 0; i < candidates.size(); ++i) + { + const auto & candidate = candidates[i]; + if (candidate.pool_uuid != pool_uuid || !candidate.owns_namespace) + continue; + if (!matched) + { + matched = i; + continue; + } + /// A second DISTINCT mount owning the namespace: ambiguous. An alias of the first is not. + if (candidates[*matched].exchange_identity != candidate.exchange_identity) + return std::nullopt; + } + return matched; +} + +} diff --git a/src/Storages/MergeTree/DataPartsExchangeCasRouting.h b/src/Storages/MergeTree/DataPartsExchangeCasRouting.h new file mode 100644 index 000000000000..f1d58fb261f7 --- /dev/null +++ b/src/Storages/MergeTree/DataPartsExchangeCasRouting.h @@ -0,0 +1,103 @@ +#pragma once + +#include + +#include +#include +#include + +namespace DB +{ +class IDisk; +using DiskPtr = std::shared_ptr; +using Disks = std::vector; +} + +namespace Poco +{ +class Logger; +using LoggerPtr = std::shared_ptr; +} +using LoggerPtr = Poco::LoggerPtr; + +namespace DB::DataPartsExchange +{ + +/// The receiver's content-addressed pool advertise as it goes on the wire (the `cas_pool_uuid` request +/// parameter): the pool ids of every disk of its storage policy that could take a relink — sorted, +/// deduplicated, joined with ", ". The list form and the ", " delimiter are the ones the zero-copy +/// `remote_fs_metadata` capability list already uses, so the exchange keeps one list convention (the +/// decoder differs in one respect: an empty string is no pool at all, never one empty id). A single id +/// is written verbatim: a receiver with one pool puts on the wire exactly the string that a sender +/// comparing the whole value with its own pool id matches. Empty ids are dropped (a storage that never +/// started has no pool id and nothing to advertise). +String encodeCasPoolAdvertise(Strings pool_uuids); +Strings decodeCasPoolAdvertise(const String & text); + +/// Which pool a relink offer is for. The sender names it in the `cas_pool_uuid` response cookie, and +/// the answer is that cookie ONLY if it is one of the pools this receiver advertised — the advertise is +/// the receiver's question, and a byte re-request after a failed relink advertises nothing, so a peer +/// offering regardless can never select a disk. A sender that predates the cookie can only have matched +/// a one-element advertise, so an absent cookie means that single pool. Several advertised pools and no +/// cookie is not a state an honest sender can produce, and the answer is "no pool" — the receiver never +/// guesses. +String resolveOfferedCasPool(const Strings & advertised_pools, const String & offered_pool_cookie); + +/// One content-addressed disk of the RECEIVING table's storage policy, in policy order. +struct CasRelinkCandidate +{ + String disk_name; + String pool_uuid; /// empty: the storage never started; never a candidate + bool read_only = false; /// a static property of the disk's configuration; the one exclusion +}; + +/// Which candidate receives the offered relink: the index of the first candidate on the offered pool +/// (`resolveOfferedCasPool`) that is not read-only. `nullopt` means no disk of this policy may take +/// the offer, which the caller turns into a byte fetch. Whether the pool is LIVE is deliberately not +/// part of this decision — a not-live pool disk is still the target, the relink's own write gate +/// refuses it, and the fetch fails and is retried rather than landing on another disk. +std::optional resolveForcedCaCandidate( + const std::vector & candidates, + const Strings & advertised_pools, + const String & offered_pool_cookie); + +/// The outcome of resolving a relink offer against this receiver's candidates: the pool the offer is +/// for (needed even when no disk is forced, to check a caller-supplied disk against it later), and the +/// disk to force the fetch onto — null when the caller already supplied a disk, or no live-policy +/// candidate matches the offered pool. +struct ForcedCaDiskChoice +{ + String offered_pool; + DiskPtr disk; +}; + +/// Resolve a relink offer's pool, and — only when the caller left disk selection to the fetch itself — +/// pick the forced candidate (`resolveForcedCaCandidate`) to place it on, ahead of the storage policy's +/// own placement. `part_name` and `log` are for the log lines only; the `cas_relink_receiver_drop_forced_disk` +/// failpoint (test-only) lives here so it can stand in for an offer this policy has no disk for. +ForcedCaDiskChoice chooseForcedCaDisk( + bool caller_supplied_disk, + const std::vector & candidates, + const Disks & candidate_disks, + const Strings & advertised_pools, + const String & offered_pool_cookie, + const String & part_name, + LoggerPtr log); + +/// One content-addressed disk of the SENDING table's storage policy, as the confirm routing sees it. +struct CasConfirmRoutingCandidate +{ + const void * exchange_identity = nullptr; /// the `IContentAddressedExchange` behind the disk + String pool_uuid; + bool owns_namespace = false; +}; + +/// Which candidate answers a relink confirm for `pool_uuid`: EXACTLY one distinct mount that owns the +/// namespace, else `nullopt` — zero owners, or two distinct mounts, are both ambiguous and `Unknown` +/// is the only honest answer. Disks that alias one mount (a base disk and its cache wrapper share the +/// exchange object) count once, as the first of them. +std::optional resolveConfirmRoutingCandidate( + const std::vector & candidates, + const String & pool_uuid); + +} diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index fc3d728db6a8..e10ac17fc7c3 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -8140,6 +8140,14 @@ void MergeTreeData::checkAlterPartitionIsPossible( /// `FORGET PARTITION` is SUPPORTED on CA — it only manipulates ZooKeeper partition metadata /// (removes block-number nodes from ZooKeeper) and does not write, clone, or touch any part /// files on disk, so it is safe on a content-addressed disk. + /// `EXPORT PARTITION` is SUPPORTED because the reason this list exists does not apply + /// to it. The rejection below is about commands that clone parts file-by-file; + /// exporting does not clone at all. `ExportPartTask` reads the source part through + /// `MergeTreeSequentialSource` (`MergeTreeSequentialSourceType::Export`) and writes + /// rows into the destination through a `SinkToStorage` on an ordinary query + /// pipeline, so the source's part files are only READ, under `readLockParts`, and + /// nothing is hard-linked or copied on the content-addressed disk. The command's + /// own bookkeeping is ZooKeeper-side. /// NOTE: `MOVE_PARTITION` also admits cross-disk /// `MOVE ... TO DISK/VOLUME` (this check cannot distinguish the destination); that uses /// the byte-copy `clonePart` path (NOT the corrupting per-file hardlink), but only @@ -8156,6 +8164,7 @@ void MergeTreeData::checkAlterPartitionIsPossible( PartitionCommand::FREEZE_ALL_PARTITIONS, PartitionCommand::UNFREEZE_PARTITION, PartitionCommand::UNFREEZE_ALL_PARTITIONS, + PartitionCommand::EXPORT_PARTITION, }; if (!std::ranges::contains(supported_commands, command.type)) diff --git a/src/Storages/System/StorageSystemContentAddressedMounts.cpp b/src/Storages/System/StorageSystemContentAddressedMounts.cpp index 2a1b82a01ee7..ba8b8670cc10 100644 --- a/src/Storages/System/StorageSystemContentAddressedMounts.cpp +++ b/src/Storages/System/StorageSystemContentAddressedMounts.cpp @@ -160,7 +160,11 @@ Pipe StorageSystemContentAddressedMounts::read( bool list_ok = true; try { - mounts = Cas::listMounts(store->backend(), store->layout(), now_ms, skew_margin_ms); + /// Introspection reads on the open fence: a row describing this disk's mount slots must + /// still be produced when the local mount fence has already run down -- that is exactly + /// the state an operator opens this table to look at. + Cas::CasOperation op = store->openRequests().admit(); + mounts = Cas::listMounts(op, store->layout(), now_ms, skew_margin_ms); } catch (...) { @@ -187,7 +191,7 @@ Pipe StorageSystemContentAddressedMounts::read( col_seq->insert(m.lease.seq); assert_cast(*col_started).insertValue(static_cast(m.lease.started_at_ms)); assert_cast(*col_expires).insertValue(static_cast(m.lease.expires_at_ms)); - col_min_active->insert(m.lease.min_active); + col_min_active->insert(m.lease.min_active_build_sequence); col_fenced->insert(static_cast(m.lease.gc_fenced)); col_state->insert(m.state); diff --git a/src/Storages/tests/gtest_cas_relink_pool_routing.cpp b/src/Storages/tests/gtest_cas_relink_pool_routing.cpp new file mode 100644 index 000000000000..2abd09738466 --- /dev/null +++ b/src/Storages/tests/gtest_cas_relink_pool_routing.cpp @@ -0,0 +1,158 @@ +#include +#include + +using namespace DB::DataPartsExchange; +using DB::Strings; + +/// ---- the advertise: sort, unique, ", " ---- + +TEST(CASRelinkPoolAdvertise, EmptySetIsEmptyStringBothWays) +{ + EXPECT_EQ(encodeCasPoolAdvertise({}), ""); + EXPECT_TRUE(decodeCasPoolAdvertise("").empty()); +} + +TEST(CASRelinkPoolAdvertise, SingleIdIsVerbatim) +{ + /// The one-element wire form must be byte-for-byte the pre-set-advertise form: a sender that compares + /// the whole parameter with its own pool id must still match. + EXPECT_EQ(encodeCasPoolAdvertise({"0123abcd"}), "0123abcd"); + EXPECT_EQ(decodeCasPoolAdvertise("0123abcd"), Strings{"0123abcd"}); +} + +TEST(CASRelinkPoolAdvertise, SortsAndDeduplicates) +{ + EXPECT_EQ(encodeCasPoolAdvertise({"bb", "aa", "bb", "aa"}), "aa, bb"); + EXPECT_EQ(decodeCasPoolAdvertise("aa, bb"), (Strings{"aa", "bb"})); +} + +TEST(CASRelinkPoolAdvertise, DropsEmptyIds) +{ + EXPECT_EQ(encodeCasPoolAdvertise({"", "aa", ""}), "aa"); + EXPECT_EQ(encodeCasPoolAdvertise({""}), ""); +} + +TEST(CASRelinkPoolAdvertise, RoundTripsThreeIds) +{ + const Strings ids{"cc", "aa", "bb"}; + EXPECT_EQ(decodeCasPoolAdvertise(encodeCasPoolAdvertise(ids)), (Strings{"aa", "bb", "cc"})); +} + +/// ---- which pool the offer is for ---- + +TEST(CASRelinkPoolAdvertise, OfferedPoolIsTheCookieWhenItWasAdvertised) +{ + EXPECT_EQ(resolveOfferedCasPool({"aa", "bb"}, "bb"), "bb"); + EXPECT_EQ(resolveOfferedCasPool({"aa"}, "aa"), "aa"); +} + +TEST(CASRelinkPoolAdvertise, UnadvertisedCookieIsNoPool) +{ + /// The byte re-request after a failed relink advertises nothing; an offer that arrives anyway must + /// not re-enter the relink path through its cookie. + EXPECT_EQ(resolveOfferedCasPool({"aa"}, "zz"), ""); + EXPECT_EQ(resolveOfferedCasPool({}, "aa"), ""); +} + +TEST(CASRelinkPoolAdvertise, AbsentCookieMeansTheSingleAdvertisedPool) +{ + EXPECT_EQ(resolveOfferedCasPool({"aa"}, ""), "aa"); +} + +TEST(CASRelinkPoolAdvertise, AbsentCookieWithSeveralPoolsIsNoPool) +{ + EXPECT_EQ(resolveOfferedCasPool({"aa", "bb"}, ""), ""); + EXPECT_EQ(resolveOfferedCasPool({}, ""), ""); +} + +/// ---- the receiver's forced candidate ---- + +static std::vector twoPools() +{ + return { + {"disk_other", "other", false}, + {"disk_shared", "shared", false}, + }; +} + +TEST(CASRelinkPoolAdvertise, CookieSelectsTheCandidateOnThatPool) +{ + EXPECT_EQ(resolveForcedCaCandidate(twoPools(), {"other", "shared"}, "shared"), std::optional{1}); + EXPECT_EQ(resolveForcedCaCandidate(twoPools(), {"other", "shared"}, "other"), std::optional{0}); +} + +TEST(CASRelinkPoolAdvertise, AbsentCookieWithOneAdvertisedPoolSelectsIt) +{ + const std::vector one{{"disk_shared", "shared", false}}; + EXPECT_EQ(resolveForcedCaCandidate(one, {"shared"}, ""), std::optional{0}); +} + +TEST(CASRelinkPoolAdvertise, AbsentCookieWithTwoAdvertisedPoolsSelectsNothing) +{ + EXPECT_EQ(resolveForcedCaCandidate(twoPools(), {"other", "shared"}, ""), std::nullopt); +} + +TEST(CASRelinkPoolAdvertise, UnknownPoolSelectsNothing) +{ + EXPECT_EQ(resolveForcedCaCandidate(twoPools(), {"other", "shared"}, "zz"), std::nullopt); +} + +TEST(CASRelinkPoolAdvertise, ReadOnlyCandidateIsSkipped) +{ + const std::vector ro{{"disk_shared_ro", "shared", true}}; + EXPECT_EQ(resolveForcedCaCandidate(ro, {"shared"}, "shared"), std::nullopt); + + const std::vector ro_then_rw{{"disk_shared_ro", "shared", true}, {"disk_shared", "shared", false}}; + EXPECT_EQ(resolveForcedCaCandidate(ro_then_rw, {"shared"}, "shared"), std::optional{1}); +} + +TEST(CASRelinkPoolAdvertise, EmptyPoolIdNeverMatches) +{ + const std::vector not_started{{"disk_cold", "", false}}; + EXPECT_EQ(resolveForcedCaCandidate(not_started, {}, ""), std::nullopt); + EXPECT_EQ(resolveForcedCaCandidate(not_started, {""}, ""), std::nullopt); +} + +TEST(CASRelinkPoolAdvertise, TwoCandidatesOnOnePoolTakeTheFirst) +{ + const std::vector two{{"disk_a", "shared", false}, {"disk_b", "shared", false}}; + EXPECT_EQ(resolveForcedCaCandidate(two, {"shared"}, "shared"), std::optional{0}); +} + +/// ---- the sender's confirm routing ---- + +static const void * const MOUNT_A = reinterpret_cast(0x10); +static const void * const MOUNT_B = reinterpret_cast(0x20); + +TEST(CASRelinkConfirmRouting, OneOwnerAnswers) +{ + const std::vector c{{MOUNT_A, "shared", true}}; + EXPECT_EQ(resolveConfirmRoutingCandidate(c, "shared"), std::optional{0}); +} + +TEST(CASRelinkConfirmRouting, NoOwnerIsNoAnswer) +{ + const std::vector c{{MOUNT_A, "shared", false}, {MOUNT_B, "other", true}}; + EXPECT_EQ(resolveConfirmRoutingCandidate(c, "shared"), std::nullopt); + EXPECT_EQ(resolveConfirmRoutingCandidate(c, ""), std::nullopt); + EXPECT_EQ(resolveConfirmRoutingCandidate({}, "shared"), std::nullopt); +} + +TEST(CASRelinkConfirmRouting, TwoDistinctOwnersAreAmbiguous) +{ + const std::vector c{{MOUNT_A, "shared", true}, {MOUNT_B, "shared", true}}; + EXPECT_EQ(resolveConfirmRoutingCandidate(c, "shared"), std::nullopt); +} + +TEST(CASRelinkConfirmRouting, AliasesOfOneMountCountOnce) +{ + /// A base disk and its cache wrapper share one exchange object: one mount, two disk names. + const std::vector c{{MOUNT_A, "shared", true}, {MOUNT_A, "shared", true}}; + EXPECT_EQ(resolveConfirmRoutingCandidate(c, "shared"), std::optional{0}); +} + +TEST(CASRelinkConfirmRouting, NonOwnerOnThePoolIsIgnored) +{ + const std::vector c{{MOUNT_A, "shared", false}, {MOUNT_B, "shared", true}}; + EXPECT_EQ(resolveConfirmRoutingCandidate(c, "shared"), std::optional{1}); +} diff --git a/tests/config/config.d/cas_s3_storage_policy_for_merge_tree_by_default.xml b/tests/config/config.d/cas_s3_storage_policy_for_merge_tree_by_default.xml index d74dfe173095..e8ac8c5dbe90 100644 --- a/tests/config/config.d/cas_s3_storage_policy_for_merge_tree_by_default.xml +++ b/tests/config/config.d/cas_s3_storage_policy_for_merge_tree_by_default.xml @@ -5,6 +5,12 @@ object_storage s3 cas + + 30 + 10000 stateless-ca-s3 @@ -22,7 +28,7 @@ replaces the old grace window (M-W D-W5); a short interval so reclamation happens during the run. --> 1 - 5 + 20 1 - 5 + 20 diff --git a/tests/integration/test_cas_drop_pool_member/configs/storage_conf.xml b/tests/integration/test_cas_drop_pool_member/configs/storage_conf.xml index ca865ce6e8d2..7cde804e2696 100644 --- a/tests/integration/test_cas_drop_pool_member/configs/storage_conf.xml +++ b/tests/integration/test_cas_drop_pool_member/configs/storage_conf.xml @@ -9,6 +9,8 @@ object_storage s3 cas + 30 + 10000 http://rustfs1:11121/test/cas_dpm_data/ clickhouse clickhouse @@ -26,6 +28,8 @@ object_storage s3 cas + 30 + 10000 http://rustfs1:11121/test/cas_dpm_data/ clickhouse clickhouse diff --git a/tests/integration/test_cas_file_cache/configs/storage_conf.xml b/tests/integration/test_cas_file_cache/configs/storage_conf.xml index b75531fc8b9b..ca4197e28e5f 100644 --- a/tests/integration/test_cas_file_cache/configs/storage_conf.xml +++ b/tests/integration/test_cas_file_cache/configs/storage_conf.xml @@ -7,6 +7,8 @@ object_storage s3 cas + 30 + 10000 itest-cas-file-cache http://rustfs1:11121/test/cas_cache_data/ clickhouse diff --git a/tests/integration/test_cas_gc_bulk_delete/__init__.py b/tests/integration/test_cas_gc_bulk_delete/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_gc_bulk_delete/configs/storage_conf.xml b/tests/integration/test_cas_gc_bulk_delete/configs/storage_conf.xml new file mode 100644 index 000000000000..92a9d0ed3e64 --- /dev/null +++ b/tests/integration/test_cas_gc_bulk_delete/configs/storage_conf.xml @@ -0,0 +1,38 @@ + + + + + object_storage + s3 + cas + 30 + 10000 + + itest-content-addressed-gc-s3 + + http://rustfs1:11121/test/cas_gc_bulk/ + clickhouse + clickhouse + + 1 + 1 + + 2 + + + + + +
+ disk_cas_gc_s3 +
+
+
+
+
+
diff --git a/tests/integration/test_cas_gc_bulk_delete/test.py b/tests/integration/test_cas_gc_bulk_delete/test.py new file mode 100644 index 000000000000..3c09d05cb6fc --- /dev/null +++ b/tests/integration/test_cas_gc_bulk_delete/test.py @@ -0,0 +1,130 @@ +import math +import time + +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +STORAGE_POLICY = "cas_gc_bulk" +MANIFESTS_PREFIX = "cas_gc_bulk/cas/manifests/" +RECLAIM_RETRIES = 90 +RECLAIM_SLEEP = 1.0 + +# Must match configs/storage_conf.xml's : kept small on purpose so +# the dropped table's owner-removed manifests (>= NUM_INSERTS of them) span more than one chunk, +# which is what this test is proving on the wire. +CHUNK_KEYS = 2 +NUM_INSERTS = 6 + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node", + main_configs=["configs/storage_conf.xml"], + with_rustfs=True, + stay_alive=True, + ) + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def list_manifests(): + return [ + o.object_name + for o in cluster.rustfs_client.list_objects( + cluster.rustfs_bucket, MANIFESTS_PREFIX, recursive=True + ) + ] + + +def gc_manifest_delete_totals(node): + """ + Sum every `manifest_deletes` phase row with work across however many rounds it took: the + owner-removed manifests from one DROP TABLE are not guaranteed to fold in a single round, so + reading only the first row would silently under-count on a slow run. + + Chunking happens per round, not over the grand total, so the expected request count is the + sum over rows of `ceil(row_attempted / CHUNK_KEYS)`, not `ceil(total_attempted / CHUNK_KEYS)`. + """ + node.query("SYSTEM FLUSH LOGS") + rows = ( + node.query( + "SELECT phase_metrics['attempted'], phase_metrics['accepted'], phase_metrics['requests'], " + "ProfileEvents['CASBulkDeleteRequests'], ProfileEvents['DiskS3DeleteObjects'] " + "FROM system.cas_gc_log WHERE event_type = 'Phase' AND phase = 'manifest_deletes' " + "AND phase_metrics['attempted'] > 0" + ) + .strip() + .splitlines() + ) + totals = [0, 0, 0, 0, 0] + expected_requests = 0 + for row in rows: + values = [int(x) for x in row.split("\t")] + for i, value in enumerate(values): + totals[i] += value + expected_requests += math.ceil(values[0] / CHUNK_KEYS) + # attempted, accepted, requests, bulk_requests, s3_deletes, expected_requests + return tuple(totals) + (expected_requests,) + + +def test_manifest_deletes_go_in_one_request_per_chunk(): + """ + A dropped table's owner-removed manifest bodies are deleted through `manifest_deletes` in + chunks of `cas_gc_bulk_delete_chunk_keys` write-once keys, one `DeleteObjects` request per + chunk -- never one request per key. With the chunk size forced down to 2 and >= NUM_INSERTS + manifests to delete, this asserts the request count matches the chunking math exactly, and + that the engine's own `CASBulkDeleteRequests` / `DiskS3DeleteObjects` counters agree with it. + """ + node = cluster.instances["node"] + node.query("DROP TABLE IF EXISTS t SYNC") + node.query( + f"CREATE TABLE t (k UInt64, v String) ENGINE = MergeTree ORDER BY k " + f"SETTINGS storage_policy = '{STORAGE_POLICY}'" + ) + for i in range(NUM_INSERTS): + node.query( + "INSERT INTO t SELECT number, toString(number) FROM numbers(1000) " + "SETTINGS max_insert_block_size = 1000" + ) + manifests_before = list_manifests() + assert len(manifests_before) >= NUM_INSERTS + + node.query("DROP TABLE t SYNC") + + attempted = accepted = requests = bulk_requests = s3_deletes = expected_requests = 0 + for _ in range(RECLAIM_RETRIES): + node.query("SYSTEM CAS GC RUN") + ( + attempted, + accepted, + requests, + bulk_requests, + s3_deletes, + expected_requests, + ) = gc_manifest_delete_totals(node) + if attempted >= NUM_INSERTS: + break + time.sleep(RECLAIM_SLEEP) + assert attempted >= NUM_INSERTS, "manifest_deletes never reported the dropped table's manifests" + + assert accepted == attempted, "every owner-removed manifest body is deleted or already absent" + assert requests == expected_requests, ( + f"chunking is per round, so {attempted} keys at {CHUNK_KEYS} per chunk across however " + f"many rounds it took should sum to {expected_requests} requests, got {requests}" + ) + assert bulk_requests == requests + assert s3_deletes == requests, "one DeleteObjects per chunk, no singular fallback" + + for _ in range(RECLAIM_RETRIES): + if not list_manifests(): + break + node.query("SYSTEM CAS GC RUN") + time.sleep(RECLAIM_SLEEP) + assert not list_manifests() diff --git a/tests/integration/test_cas_gc_s3/configs/storage_conf.xml b/tests/integration/test_cas_gc_s3/configs/storage_conf.xml index 3467ba70bc4d..b04b43a617b4 100644 --- a/tests/integration/test_cas_gc_s3/configs/storage_conf.xml +++ b/tests/integration/test_cas_gc_s3/configs/storage_conf.xml @@ -5,6 +5,8 @@ object_storage s3 cas + 30 + 10000 itest-content-addressed-gc-s3 + __SERVER_ROOT_ID__ + http://fakegcs:8080/hmacbucket/cas/ + gcs_hmac + GOOG1EFAKEACCESSKEYID + fake-goog4-hmac-secret + 1 + 1 + +
+ + +
disk_cas_gcs_shared
+
+
+ + diff --git a/tests/integration/test_cas_gcs_relink_liveness/test.py b/tests/integration/test_cas_gcs_relink_liveness/test.py new file mode 100644 index 000000000000..f92043385aae --- /dev/null +++ b/tests/integration/test_cas_gcs_relink_liveness/test.py @@ -0,0 +1,306 @@ +"""Fetch-by-relink liveness on a slow control plane. + +Two replicas of one ReplicatedMergeTree table share one CAS pool over the fake GCS service of +`test_cas_gcs`, with every write to a key containing `_ckpt` delayed. That substring targets the +ref-lane flush's checkpoint publication, but it also matches namespace creation, recovery, and the GC +snapshot publisher's checkpoint contribution — all four are slowed, not just the flush. Both replicas +insert continuously, so each is a sender and a receiver at once and each keeps its own lane busy. A +confirm rule that refuses whenever the sender's lane is busy starves both replication queues here +(finding F11 of the 2026-09-02 live GCS campaign); the ref-scoped rule lets them drain. The unit tests +pin the rule; this is the liveness reproduction they cannot give. +""" +import json +import os +import shlex +import threading +import time + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.mock_servers import start_mock_servers + +GCS_HOST = "fakegcs" +GCS_PORT = 8080 +CONFIG_IN_CONTAINER = "/etc/clickhouse-server/config.d/cas_gcs_shared.xml" +MOCK_DIR = os.path.join(os.path.dirname(__file__), "..", "test_cas_gcs", "gcs_mocks") +NODES = ("node1", "node2") +CA_DISK = "disk_cas_gcs_shared" + +# Every `_ckpt` PUT sleeps this long: one second is the real service's own per-object mutation cap +# (see the module docstring above). A full run at this value takes several minutes. +CKPT_DELAY_MS = 1000 +INSERTS_PER_NODE = 80 +ROWS_PER_INSERT = 1000 +DRAIN_TIMEOUT_S = 180 + +cluster = ClickHouseCluster(__file__) + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + # As in test_cas_gcs: a CAS disk mounts at server start and fails closed when its store is + # unreachable, and the fake can only be launched once its container is up. So both nodes start + # without the disk, the disk configuration is installed with the node's own server root, and each + # node is restarted once the fake answers. + for name in NODES: + cluster.add_instance(name, macros={"replica": name}, with_zookeeper=True, stay_alive=True) + cluster.add_instance( + GCS_HOST, hostname=GCS_HOST, image="altinityinfra/python-bottle", tag="latest", stay_alive=True + ) + try: + cluster.start() + start_mock_servers(cluster, MOCK_DIR, [("server.py", GCS_HOST, str(GCS_PORT))]) + for name in NODES: + node = cluster.instances[name] + node.copy_file_to_container( + os.path.join(os.path.dirname(__file__), "configs", "storage_conf.xml"), + CONFIG_IN_CONTAINER, + ) + node.replace_in_config(CONFIG_IN_CONTAINER, "__SERVER_ROOT_ID__", name) + node.restart_clickhouse() + yield cluster + finally: + cluster.shutdown() + + +def _control_post(path): + container = cluster.get_container_id(GCS_HOST) + return cluster.exec_in_container( + container, ["curl", "-sS", "-X", "POST", "http://localhost:{}{}".format(GCS_PORT, path)] + ) + + +def _control_get(path): + container = cluster.get_container_id(GCS_HOST) + return cluster.exec_in_container( + container, ["curl", "-sS", "http://localhost:{}{}".format(GCS_PORT, path)] + ) + + +def _set_delay(substr, ms): + # `curl -sS` exits 0 on an HTTP 404 or 500, and a docker-exec only raises on a non-zero exit + # status — so a renamed endpoint or a renamed query parameter would otherwise go unnoticed here + # and the rest of the test would exercise a fake running at full speed. Checking the echoed body + # is what makes a broken lever fail loudly instead of silently. + reply = _control_post("/_control/delay?substr={}&ms={}".format(substr, ms)) + echoed = json.loads(reply) + assert {"substr": substr, "ms": ms}.items() <= echoed.items(), ( + "fake did not echo back the delay setting it was asked for: {!r}".format(reply) + ) + + +def _delayed_put_count(): + counters = json.loads(_control_get("/_control/counters")) + return counters.get("DelayedPut", 0) + + +def _queue_size(node, table): + return int( + node.query("SELECT count() FROM system.replication_queue WHERE table = '{}'".format(table)) + ) + + +def _queue_breakdown(node, table): + # What a bare queue-size number cannot say: WHICH kind of entry is stuck and why. A future stuck + # run for an unrelated reason (a merge stall, a ZooKeeper hiccup) would otherwise print a + # byte-identical failure message to this test's own liveness symptom. + return node.query( + "SELECT type, count(), any(last_exception) FROM system.replication_queue " + "WHERE table = '{}' GROUP BY type ORDER BY type FORMAT TSV".format(table) + ) + + +def _replica_status(node, table): + row = node.query( + "SELECT queue_size, absolute_delay, log_pointer, log_max_index " + "FROM system.replicas WHERE table = '{}' FORMAT TSV".format(table) + ) + queue_size, absolute_delay, log_pointer, log_max_index = row.split() + return int(queue_size), int(absolute_delay), int(log_pointer), int(log_max_index) + + +def _drained(node, table): + # `queue_size == 0` alone is checked right after the insert threads join, before the + # queue-updating thread is guaranteed to have pulled the peer's latest log entries — so a node + # can read an empty queue with fetches still outstanding, and closing the delay window in that + # instant would let the tail race to completion at full speed and pass for the wrong reason. + # `log_pointer > log_max_index` (every log entry has been copied into the execution queue) and + # `absolute_delay == 0` together rule that race out. + queue_size, absolute_delay, log_pointer, log_max_index = _replica_status(node, table) + return queue_size == 0 and absolute_delay == 0 and log_pointer > log_max_index + + +def _refusal_counters(node): + return node.query( + "SELECT event, value FROM system.events WHERE event LIKE 'CASRelinkConfirmRefused%' " + "ORDER BY event FORMAT TSV" + ) + + +def _refusal_counter(node, event): + # `system.events` has no row for an event that never fired, so an empty result is a zero, not an + # error — which is what lets the caller assert a clean zero rather than having to special-case a + # missing row. A misspelled event name reads the same way, so a zero here is never by itself + # evidence that the named counter exists. + value = node.query( + "SELECT value FROM system.events WHERE event = '{}'".format(event) + ).strip() + return int(value) if value else 0 + + +def _log_lines(node, pattern): + out = node.exec_in_container( + [ + "bash", + "-c", + "grep -a -E {} /var/log/clickhouse-server/clickhouse-server.log || true".format( + shlex.quote(pattern) + ), + ] + ) + return [line for line in out.splitlines() if line.strip()] + + +def _relink_finished_pattern(table, disk): + """The receiver-side proof that a fetch completed by relink, not by byte transfer. + + Reachable only after `Fetcher::relinkPartToDisk`'s confirm step answered yes and `promote()` + returned `Committed` — every other row in that function returns or throws before this line, so its + presence cannot be produced by a fallback to bytes. Same pattern as + `test_cas_replicated_relink.relink_finished_pattern`, generalised to any part name since this test + does not track individual part names. + """ + return r"default\.{} .*Relink of part .* onto disk {} finished \(no bytes transferred\)".format( + table, disk + ) + + +def test_both_queues_drain_under_slow_checkpoints(): + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "relink_liveness" + for node in (node1, node2): + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) + node.query( + "CREATE TABLE {t} (id Int64, v UInt64) " + "ENGINE = ReplicatedMergeTree('/clickhouse/tables/{t}', '{{replica}}') " + "ORDER BY id SETTINGS storage_policy = 'cas_gcs_shared'".format(t=table) + ) + + _set_delay("_ckpt", CKPT_DELAY_MS) + try: + errors = [] + + def insert_loop(node, base): + try: + for i in range(INSERTS_PER_NODE): + node.query( + "INSERT INTO {} SELECT number, number * 10 FROM numbers({}, {})".format( + table, base + i * ROWS_PER_INSERT, ROWS_PER_INSERT + ) + ) + except Exception as e: # surfaced below, on the test thread + errors.append((node.name, repr(e))) + + threads = [ + threading.Thread(target=insert_loop, args=(node1, 0)), + threading.Thread(target=insert_loop, args=(node2, 10_000_000)), + ] + for t in threads: + t.start() + for t in threads: + t.join() + assert errors == [], errors + + # The lever, checked BEFORE the drain assertion: if the delay knob silently stopped + # matching (a renamed endpoint, a renamed query param, a `_ckpt` key that stopped matching + # the substring), the fake would serve every write at full speed and the drain assertion + # below would pass having exercised nothing. Checking this first means a failure here is + # never confused with the liveness failure this test exists to catch. + delayed = _delayed_put_count() + assert delayed >= 2 * INSERTS_PER_NODE, ( + "the delay knob fired only {} times; expected at least {} — it may have silently " + "stopped matching `_ckpt` PUTs, which would make any drain result meaningless".format( + delayed, 2 * INSERTS_PER_NODE + ) + ) + + # Liveness: the insert threads have already joined, so each replica's lane is kept busy from + # here on by its own fetch bookkeeping alone — the self-sustaining half of the livelock. Both + # queues must still drain. + deadline = time.time() + DRAIN_TIMEOUT_S + while time.time() < deadline and not (_drained(node1, table) and _drained(node2, table)): + time.sleep(1) + drained = (_drained(node1, table), _drained(node2, table)) + sizes = (_queue_size(node1, table), _queue_size(node2, table)) + for node in (node1, node2): + print(node.name, "refusal counters:", _refusal_counters(node)) + assert drained == (True, True), ( + "replication queues did not drain in {} s with slow checkpoints: node1={} node2={}\n" + "node1 queue (type, count, last_exception):\n{}\n" + "node2 queue (type, count, last_exception):\n{}".format( + DRAIN_TIMEOUT_S, + sizes[0], + sizes[1], + _queue_breakdown(node1, table), + _queue_breakdown(node2, table), + ) + ) + + # Transport proof, checked AFTER the drain assertion on purpose: under the old rule the + # parts never arrive at all, so a relink assertion placed before the drain check would fail + # for the second-best reason and muddy the evidence this test exists to produce. "Both + # queues drained" cannot by itself distinguish a relinked fetch from a byte fetch that + # dropped into one of `relinkPartToDisk`'s silent fallback exits — this line is reachable + # only through the intended path. + for node in (node1, node2): + finished = _log_lines(node, _relink_finished_pattern(table, CA_DISK)) + assert finished, ( + "{} drained its queue without a single relink completing — every part that " + "arrived took some route other than fetch-by-relink".format(node.name) + ) + + # No confirm may be refused by lane STATE on a healthy run. These two counters move only on + # `confirmExactRef`'s wedge and broken-lane branches — an unresolved append, or a lane in + # NeedsRecovery, Closed, Faulted, or Writing with nothing carved. This case drives the fake's + # control plane with a delay and never with a fault, so none of those is reachable here and a + # non-zero value is a lane defect the drain assertion above can hide: a confirm refused + # table-wide costs the receiver only a retry, and enough retries still finish inside the drain + # window. Against a real bucket a wedge IS reachable from load alone — sustained throttling can + # exhaust an append's retry budget and leave its outcome unresolved — so this assertion rests + # on the fault-free stand and would have to be rethought before it ran anywhere else. + # + # A zero here is not evidence that the counter exists: `_refusal_counter` reads a missing row + # as a zero, and a misspelled or unregistered name reads the same way. + # + # What is NOT asserted, deliberately: `CASRelinkConfirmRefusedRefMutationInFlight` above zero. + # A refusal there needs a queued or carved mutation naming the very ref the peer is asking + # about, and each node's lane is busy with its own newer parts rather than with the seconds-old + # part its peer is fetching — a run that passed both assertions above left node1 with no + # `CASRelinkConfirmRefused%` row at all. Requiring it would demand the symptom the ref-scoped + # rule removes, so it would pass only while the livelock is present. The two things it was + # meant to show are shown better above: the relink-completion assertion proves confirms were + # asked and answered `Yes`, and the delayed-PUT floor proves the checkpoint publications were + # slowed, which is the contention itself. Its attribution is pinned in the unit suite, by + # `CASConfirmExactRef.UntouchedRefConfirmsWhileAnotherRefIsQueued`. + for node in (node1, node2): + for event in ("CASRelinkConfirmRefusedLaneWedged", "CASRelinkConfirmRefusedLaneBroken"): + assert _refusal_counter(node, event) == 0, ( + "{} refused a relink confirm by lane state ({}), which no fault was injected to " + "produce. All refusal counters on this node:\n{}".format( + node.name, event, _refusal_counters(node) + ) + ) + finally: + try: + _set_delay("", 0) + except Exception as exc: + # Never let a failure here mask a real assertion failure raised above. + print("failed to clear the delay knob:", repr(exc)) + + expected = 2 * INSERTS_PER_NODE * ROWS_PER_INSERT + assert int(node1.query("SELECT count() FROM {}".format(table))) == expected + assert int(node2.query("SELECT count() FROM {}".format(table))) == expected + for node in (node1, node2): + node.query("DROP TABLE IF EXISTS {} SYNC".format(table)) diff --git a/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml b/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml index 0149d398aa1c..4959b16fadff 100644 --- a/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml +++ b/tests/integration/test_cas_insert_fault_recovery/configs/storage_conf.xml @@ -5,6 +5,8 @@ object_storage s3 cas + 30 + 10000 diff --git a/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml b/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml index 0149d398aa1c..4959b16fadff 100644 --- a/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml +++ b/tests/integration/test_cas_lazy_load_recovery/configs/storage_conf.xml @@ -5,6 +5,8 @@ object_storage s3 cas + 30 + 10000 diff --git a/tests/integration/test_cas_mount_renewal_retry/configs/request_budget_disks.xml b/tests/integration/test_cas_mount_renewal_retry/configs/request_budget_disks.xml new file mode 100644 index 000000000000..8030a2f5da0d --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/configs/request_budget_disks.xml @@ -0,0 +1,51 @@ + + + + + + object_storage + s3 + cas + 30 + 10000 + itest-cas-budget-capped + http://rustfs1:11121/test/cas_request_budget_capped/ + clickhouse + clickhouse + false + + 1000 + 5000 + + + object_storage + s3 + cas + 30 + 10000 + itest-cas-budget-unbounded + http://rustfs1:11121/test/cas_request_budget_unbounded/ + clickhouse + clickhouse + false + + 0 + 5000 + 50000 + + + + diff --git a/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml b/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml index 997e9a217631..2277a75b9e68 100644 --- a/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml +++ b/tests/integration/test_cas_mount_renewal_retry/configs/storage_conf.xml @@ -11,6 +11,8 @@ object_storage s3 cas + 30 + 10000 itest-cas-renewal + 10000 + 2000 + 500 + 500
diff --git a/tests/integration/test_cas_mount_renewal_retry/configs/unsafe_remount.xml b/tests/integration/test_cas_mount_renewal_retry/configs/unsafe_remount.xml new file mode 100644 index 000000000000..de4b8dcd0e55 --- /dev/null +++ b/tests/integration/test_cas_mount_renewal_retry/configs/unsafe_remount.xml @@ -0,0 +1,14 @@ + + + + + + 1 + 30 + 10000 + + + + diff --git a/tests/integration/test_cas_mount_renewal_retry/test.py b/tests/integration/test_cas_mount_renewal_retry/test.py index 31fa1529c0c3..42f09de3a090 100644 --- a/tests/integration/test_cas_mount_renewal_retry/test.py +++ b/tests/integration/test_cas_mount_renewal_retry/test.py @@ -6,6 +6,9 @@ import urllib.request import pytest +import urllib3 +from minio import Minio +from urllib3.util import Timeout as _Urllib3Timeout from helpers.cluster import ClickHouseCluster @@ -28,8 +31,27 @@ "CASRemountFailed", ) - -def _control(base_url, path, patch=None): +# The lease timing `configs/storage_conf.xml` compiles into `disk_cas_renewal`. One fixed budget for +# every build (no sanitizer-conditional scaling): wide enough that a single physical attempt, however +# slow the host, cannot plausibly cross the lease TTL (see the config file's own comment for the +# validateCasRequestBudget arithmetic), while still short enough that the hard-restart tests below +# observe the token-stability wait within a bounded test timeout. Mirrored here (rather than read back +# from the server) so every expectation string/number in this module derives from one place; keep both +# sides in sync with configs/storage_conf.xml. +MOUNT_LEASE_TTL_MS = 10000 +MOUNT_RENEW_PERIOD_MS = 2000 +ATTEMPT_TIMEOUT_MS = 500 +LEASE_SAFETY_MARGIN_MS = 500 + + +def _control(base_url, path, patch=None, timeout=10): + # `timeout` bounds each individual blocking socket operation (connect, then each read), not the + # wall-clock time to a fully-read response: a peer that trickles bytes in slowly enough to keep + # resetting the read timeout, without ever exceeding it, could still keep this call running past + # the caller's deadline. Accepted: the s3proxy control server this talks to answers in one small, + # immediate response with nothing in the path that could trickle, and every `_wait_until` probe in + # this module calls it at most once, so the worst case this leaves open is bounded and small -- not + # worth a cancellation thread for a well-behaved local test double. if patch is None: request = urllib.request.Request("{}{}".format(base_url, path)) else: @@ -39,26 +61,51 @@ def _control(base_url, path, patch=None): headers={"Content-Type": "application/json"}, method="POST", ) - with urllib.request.urlopen(request, timeout=10) as response: + with urllib.request.urlopen(request, timeout=timeout) as response: return json.loads(response.read().decode()) -def _wait_until(probe, timeout=40): - deadline = time.monotonic() + timeout +class _Deadline: + """One absolute deadline, shared across a whole `_wait_until` call (every retry) AND across every + query one of its probes issues. `remaining()` must be re-read before EACH query in a probe that + issues more than one: reusing a single `remaining()` value across sequential queries would let + EACH one spend the full remaining budget, so an N-query probe could run up to Nx as long as the + caller intended before anything ever times out. + """ + + def __init__(self, timeout): + self._deadline = time.monotonic() + timeout + + def remaining(self): + return max(0.0, self._deadline - time.monotonic()) + + def expired(self): + return time.monotonic() >= self._deadline + + +def _wait_until(probe, timeout=40, interval=0.2): + # `probe` is called with a shared `_Deadline` on every attempt, so it (and every query it issues) + # can bound its own work at `deadline.remaining()` instead of inheriting a client's much larger + # default -- otherwise one slow or stuck call could silently burn the whole budget of a caller that + # is itself waiting on a much tighter deadline. A truthy result that only comes back after this + # deadline has already passed (the probe ran long enough to blow through its own remaining budget) + # is rejected here too, rather than accepted as an on-time success. + deadline = _Deadline(timeout) last = None - while time.monotonic() < deadline: - last = probe() - if last: + while not deadline.expired(): + last = probe(deadline) + if last and not deadline.expired(): return last - time.sleep(0.2) + time.sleep(min(interval, deadline.remaining())) raise AssertionError("condition did not become true within {}s; last={!r}".format(timeout, last)) -def _profile_events(node): +def _profile_events(node, timeout=None): rows = node.query( "SELECT event, value FROM system.events WHERE event IN ({}) FORMAT TSV".format( ", ".join("'{}'".format(event) for event in RENEWAL_EVENTS) - ) + ), + timeout=timeout, ) values = {event: 0 for event in RENEWAL_EVENTS} for row in rows.splitlines(): @@ -71,13 +118,14 @@ def _event_delta(before, after): return {event: after[event] - before[event] for event in RENEWAL_EVENTS} -def _mount_snapshot(node): +def _mount_snapshot(node, timeout=None): row = node.query( "SELECT renewal_sequence, state, lifecycle, gc_fenced " "FROM system.cas_mounts " "WHERE disk = '{}' AND server_root_id = '{}' LIMIT 1 FORMAT TSV".format( DISK, SERVER_ROOT_ID - ) + ), + timeout=timeout, ).strip() assert row, "the local CAS mount row must be visible" sequence, state, lifecycle, gc_fenced = row.split("\t") @@ -89,14 +137,53 @@ def _mount_snapshot(node): } -def _read_mount_object(): - response = cluster.rustfs_client.get_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) +def _rustfs_client(timeout): + # A fresh, deadline-scoped client rather than the shared `cluster.rustfs_client`: that one's + # `http_client` (see wait_rustfs_to_start in helpers/cluster.py) has no configured timeout, so a + # stalled RustFS response through it could block a `_wait_until` probe past its own deadline + # without ever timing out on its own. Cheap to construct; only used for this module's polling reads. + # + # `timeout` bounds each individual blocking socket operation (connect, then each read) through this + # client, not the wall-clock time to a fully-read response: a peer trickling bytes slowly enough to + # keep resetting the read timeout, without ever exceeding it, could still run past the caller's + # deadline. Accepted: RustFS is a local, well-behaved test double (never observed to trickle), and + # every operation _read_mount_object issues through a client built here recomputes ITS OWN fresh + # timeout first, so the number of such operations per probe is fixed and small -- not worth a + # cancellation thread for a local test double. + return Minio( + "{}:{}".format(cluster.rustfs_ip, cluster.rustfs_port), + access_key=cluster.rustfs_access_key, + secret_key=cluster.rustfs_secret_key, + secure=False, + http_client=urllib3.PoolManager( + cert_reqs="CERT_NONE", + timeout=_Urllib3Timeout(connect=timeout, read=timeout), + ), + ) + + +def _read_mount_object(deadline=None): + # Three sequential RustFS operations (client construction for the GET, the GET/body-read, then + # client construction for the HEAD): `deadline.remaining()` is read again before EACH one rather + # than reused from the first, so a slow GET cannot silently gift the HEAD the same full budget + # again. Standalone callers (outside any `_wait_until` probe) get a fresh 20s deadline of their own. + if deadline is None: + deadline = _Deadline(20) + + remaining = deadline.remaining() + if remaining <= 0: + raise AssertionError("_read_mount_object: deadline already expired before the GET") + response = _rustfs_client(remaining).get_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) try: body = response.read() finally: response.close() response.release_conn() - stat = cluster.rustfs_client.stat_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) + + remaining = deadline.remaining() + if remaining <= 0: + raise AssertionError("_read_mount_object: deadline already expired before the HEAD") + stat = _rustfs_client(remaining).stat_object(cluster.rustfs_bucket, MOUNT_OBJECT_KEY) return body, stat.etag.strip('"') @@ -108,8 +195,29 @@ def _decode_mount(body): return json.loads(lines[1]) -def _renewal_log_rows(node, since, sequence): - node.query("SYSTEM FLUSH LOGS") +def _log_count_since_last_restart(node, pattern): + # The shortened renewal period makes this server log heavily enough to rotate + # clickhouse-server.log mid-test, so a plain grep on the live file alone can miss matches that + # already rotated out to clickhouse-server.log.N.gz. Concatenate the rotated files (oldest + # first, by the numeric suffix) followed by the live file, then count matches only after the + # LAST "Starting ClickHouse" line -- i.e. since the current server incarnation's own start -- + # so the count is a clean per-restart delta regardless of how many times rotation happened. + script = ( + "combined=$(mktemp); " + "for f in $(ls /var/log/clickhouse-server/clickhouse-server.log.[0-9]*.gz 2>/dev/null " + "| sort -t. -k3,3rn); do zcat \"$f\" >> \"$combined\"; done; " + "cat /var/log/clickhouse-server/clickhouse-server.log >> \"$combined\"; " + "start_line=$(grep -n 'Starting ClickHouse' \"$combined\" | tail -1 | cut -d: -f1); " + "tail -n +\"${start_line:-1}\" \"$combined\" | grep -c -- '%s' || true; " + "rm -f \"$combined\"" + ) % pattern + return int(node.exec_in_container(["bash", "-c", script]).strip()) + + +def _renewal_log_rows(node, since, deadline=None): + # Two sequential queries: `deadline.remaining()` is read again before the second one rather than + # reused from the first, so this whole call cannot spend twice the caller's remaining budget. + node.query("SYSTEM FLUSH LOGS", timeout=deadline.remaining() if deadline else None) rows = node.query( "SELECT outcome, detail['seq'], detail['write_attempt_id'], " "detail['attempts_sent'], detail['classification'] " @@ -117,10 +225,10 @@ def _renewal_log_rows(node, since, sequence): "WHERE event_type = 'watermark_renew' AND disk_name = '{}' " "AND detail['server_root_id'] = '{}' " "AND event_time_microseconds >= toDateTime64('{}', 6) " - "AND detail['seq'] = '{}' " "ORDER BY event_time_microseconds FORMAT TSV".format( - DISK, SERVER_ROOT_ID, since, sequence - ) + DISK, SERVER_ROOT_ID, since + ), + timeout=deadline.remaining() if deadline else None, ) return [tuple(row.split("\t")) for row in rows.splitlines() if row] @@ -133,6 +241,15 @@ def start_cluster(): with_rustfs=True, stay_alive=True, ) + # A separate instance for the two `disk_cas_budget_*` probe disks (see + # configs/request_budget_disks.xml): the hard-restart tests below count EXACT occurrences of a + # production log line across every writable CAS disk on `node`, so sharing that node with more + # writable CAS disks would inflate their counts. + cluster.add_instance( + "budget_probe", + main_configs=["configs/request_budget_disks.xml"], + with_rustfs=True, + ) cluster.base_cmd.extend( ["--file", os.path.join(os.path.dirname(__file__), "docker_compose_proxy.yml")] ) @@ -144,7 +261,7 @@ def start_cluster(): cluster.base_cmd + ["port", "s3proxy", "8474"], text=True ).strip() control_url = "http://{}".format(binding) - _wait_until(lambda: _control(control_url, "/healthz"), timeout=30) + _wait_until(lambda deadline: _control(control_url, "/healthz", timeout=deadline.remaining()), timeout=30) _control(control_url, "/config", {"reset": True}) node = cluster.instances["node"] @@ -155,7 +272,11 @@ def start_cluster(): ) ) node.query("INSERT INTO renewal_probe VALUES (0, 'before')") - yield {"node": node, "control_url": control_url} + yield { + "node": node, + "budget_probe": cluster.instances["budget_probe"], + "control_url": control_url, + } finally: if control_url is not None: try: @@ -165,6 +286,25 @@ def start_cluster(): cluster.shutdown() +def test_openpoolview_handoff_freezes_the_connect_cap_from_the_real_disk(start_cluster): + # `ContentAddressedMetadataStorage::openPoolView` derives `connect_timeout_cap_ms` from the + # disk's OWN S3 client (`freezeConnectTimeoutCapMs`) and hands it into the pool's request budget; + # `Pool::open` logs that budget once, at startup, through logger `CasRequestBudget`. This proves + # the handoff end to end through the two disks' real startup, not through a test-constructed + # backend: `disk_cas_budget_capped` (connect_timeout_ms=1000) must show the derived cap 1000 and + # envelope 7000; `disk_cas_budget_unbounded` (connect_timeout_ms=0, Poco's "unbounded") must show + # the cap falling back to the attempt timeout itself, 5000, and envelope 15000. Either disk + # opening with the base client's own timeout instead (1000/1000/... or 2000/...) would mean the + # freeze was skipped or the frozen value never reached the backend. + node = start_cluster["budget_probe"] + assert _log_count_since_last_restart( + node, "CAS request budget in effect: attempt_timeout_ms=5000 connect_timeout_cap_ms=1000 envelope_ms=7000" + ) == 1 + assert _log_count_since_last_restart( + node, "CAS request budget in effect: attempt_timeout_ms=5000 connect_timeout_cap_ms=5000 envelope_ms=15000" + ) == 1 + + def test_transient_mount_renewal_retries_without_remount(start_cluster): node = start_cluster["node"] control_url = start_cluster["control_url"] @@ -187,9 +327,16 @@ def test_transient_mount_renewal_retries_without_remount(start_cluster): }, ) - def recovered_snapshot(): - mount = _mount_snapshot(node) - counters = _profile_events(node) + def recovered_snapshot(deadline): + # Read the counter before the mount row: `CASMountRenewalRecovered` is incremented as soon as + # the renewal decides its outcome, strictly before the mount row's `renewal_sequence` (and the + # matching cas_log row) is updated to the new sequence. With the shortened renewal period a + # background (fault-free) renewal can land between the two reads; reading counters first makes + # the subsequent mount read very unlikely to still observe the pre-recovery sequence. + # `deadline.remaining()` is read again for the second query rather than reused from the first, + # so this probe cannot spend twice its caller's remaining budget. + counters = _profile_events(node, timeout=deadline.remaining()) + mount = _mount_snapshot(node, timeout=deadline.remaining()) if ( mount["sequence"] > mount_before["sequence"] and counters["CASMountRenewalRecovered"] @@ -204,17 +351,23 @@ def recovered_snapshot(): body_after, token_after = _read_mount_object() mount_body = _decode_mount(body_after) delta = _event_delta(counters_before, counters_after) - sequence = mount_after["sequence"] + # The engine paces the reissues inside one renewal; what the log records is the renewal's + # outcome, and the attempt count on that row is what says a retry happened. Look the row up by + # outcome rather than by a snapshot-derived sequence (see _renewal_log_rows): with the shortened + # renewal period, background renewals can advance `system.cas_mounts` past the exact sequence this + # recovery landed on before either of these two reads gets to it. rows = _wait_until( - lambda: ( + lambda deadline: ( found - if {row[0] for row in found} >= {"retrying", "recovered"} + if any(row[0] == "recovered" for row in found) else None ) - if (found := _renewal_log_rows(node, since, sequence)) + if (found := _renewal_log_rows(node, since, deadline=deadline)) else None, timeout=20, ) + recovered = next(row for row in rows if row[0] == "recovered") + sequence = int(recovered[1]) assert delta["CASMountRenewalAttempts"] > 1, delta assert delta["CASMountRenewalRetries"] > 0, delta @@ -226,16 +379,14 @@ def recovered_snapshot(): assert mount_after["state"] == "live", mount_after assert mount_after["lifecycle"] == "live", mount_after assert mount_after["gc_fenced"] == 0, mount_after - assert int(mount_body["seq"]) == sequence + # >= rather than == : body_after may reflect a later, unrelated background renewal that landed + # after the one this test is verifying. + assert int(mount_body["seq"]) >= sequence assert token_after != token_before assert stats["faults"] == 1, stats assert stats["by_mode"].get("503") == 1, stats print("targeted request count (transient renewal): {}".format(stats["faults"]), flush=True) - retrying = next(row for row in rows if row[0] == "retrying") - recovered = next(row for row in rows if row[0] == "recovered") - assert retrying[1] == recovered[1] == str(sequence), rows - assert retrying[2] == recovered[2], rows assert int(recovered[3]) > 1, rows assert recovered[4] == "committed_after_retry", rows @@ -268,9 +419,34 @@ def test_landed_response_lost_adopts_exact_mount_write(start_cluster): }, ) - def resolved_snapshot(): - mount = _mount_snapshot(node) - counters = _profile_events(node) + # The proxy records the dropped-after-forward request (and the upstream_etag the real PUT landed + # under) as soon as it happens -- the physical write itself already reached the object store; only + # the response back to ClickHouse was dropped. That is well before ClickHouse's own request notices + # the lost response and resolves it by re-reading. With the renewal period this short, a plain + # "read the object once resolution is confirmed" can just as easily observe a LATER, unrelated + # background renewal that started immediately after this one resolved (see the mount_before/after + # sequence race this replaced). Wait for the proxy's own record first, then poll the object for + # its exact upstream_etag, so body_after is unambiguously the write this test is about. + def dropped_record(deadline): + found_stats = _control(control_url, "/stats", timeout=deadline.remaining()) + found_records = found_stats["drop_after_forward"] + return (found_stats, found_records[0]) if len(found_records) == 1 else None + + stats, record = _wait_until(dropped_record) + target_etag = record["upstream_etag"].strip('"') + + def matching_object(deadline): + body, token = _read_mount_object(deadline) + return (body, token) if token == target_etag else None + + body_after, token_after = _wait_until(matching_object) + mount_body = _decode_mount(body_after) + + def resolved_snapshot(deadline): + # `deadline.remaining()` is read again for the second query rather than reused from the + # first, so this probe cannot spend twice its caller's remaining budget. + counters = _profile_events(node, timeout=deadline.remaining()) + mount = _mount_snapshot(node, timeout=deadline.remaining()) if ( mount["sequence"] > mount_before["sequence"] and counters["CASMountRenewalResolved"] @@ -281,39 +457,45 @@ def resolved_snapshot(): return mount, counters return None - mount_after, counters_after = _wait_until(resolved_snapshot) + # Polling at the renewal's own period (`MOUNT_RENEW_PERIOD_MS`) can alias with it on a slow host -- + # a resolved renewal can land and be superseded by the next one between two polls. Poll faster than + # the cadence it observes, with a longer timeout to match. + mount_after, counters_after = _wait_until(resolved_snapshot, timeout=120, interval=0.05) _control(control_url, "/config", {"rate": 0.0}) - stats = _control(control_url, "/stats") - body_after, token_after = _read_mount_object() - mount_body = _decode_mount(body_after) delta = _event_delta(counters_before, counters_after) - sequence = mount_after["sequence"] + # Look the recovered row up by outcome/classification rather than by a snapshot-derived sequence + # (see _renewal_log_rows): background renewals can advance `system.cas_mounts` past the exact + # sequence this recovery landed on before either of these reads gets to it. rows = _wait_until( - lambda: ( + lambda deadline: ( found - if any(row[0] == "recovered" and row[4] == "committed_by_get" for row in found) + if any(row[0] == "recovered" and row[4] == "committed_by_read" for row in found) else None ) - if (found := _renewal_log_rows(node, since, sequence)) + if (found := _renewal_log_rows(node, since, deadline=deadline)) else None, timeout=20, ) + recovered = next(row for row in rows if row[0] == "recovered" and row[4] == "committed_by_read") + sequence = int(recovered[1]) - records = stats["drop_after_forward"] assert stats["faults"] == 1, stats assert stats["by_mode"].get("drop_after_forward") == 1, stats - assert len(records) == 1, records - record = records[0] assert record["method"] == "PUT", record assert record["path"].split("?", 1)[0] == MOUNT_REQUEST_PATH, record assert 200 <= record["upstream_status"] < 300, record assert record["request_body_sha256"] == hashlib.sha256(body_after).hexdigest(), record - assert record["upstream_etag"].strip('"') == token_after, record assert body_after != body_before assert token_after != token_before - assert int(mount_body["seq"]) == sequence - - assert delta["CASMountRenewalAttempts"] == 1, delta + # >= rather than == : a later, unrelated background renewal may have advanced the mount object + # again between the capture above and this read of the confirmed sequence from the log. + assert int(mount_body["seq"]) >= sequence + + # >= rather than == : with the shortened renewal period, an unrelated fault-free background + # renewal can complete (and count its own single attempt) right before or after this one, in the + # gap between configuring the fault and observing this specific renewal's resolution. Resolved and + # Recovered stay exact -- only a lost-response renewal like this one increments them. + assert delta["CASMountRenewalAttempts"] >= 1, delta assert delta["CASMountRenewalRetries"] == 0, delta assert delta["CASMountRenewalResolved"] == 1, delta assert delta["CASMountRenewalRecovered"] == 1, delta @@ -325,9 +507,71 @@ def resolved_snapshot(): assert mount_after["lifecycle"] == "live", mount_after assert mount_after["gc_fenced"] == 0, mount_after - recovered = next(row for row in rows if row[0] == "recovered") - assert recovered[1] == str(sequence), rows assert recovered[2] and mount_body["write_attempt_id"].startswith(recovered[2]), rows assert recovered[3] == "1", rows - assert recovered[4] == "committed_by_get", rows + assert recovered[4] == "committed_by_read", rows print("targeted request count (landed response lost): {}".format(stats["faults"]), flush=True) + + +def test_hard_restart_observes_then_the_unsafe_knob_skips_the_observation(start_cluster): + node = start_cluster["node"] + + def log_count_since_last_restart(pattern): + return _log_count_since_last_restart(node, pattern) + + # mountObservationThresholdMs(ttl_ms, poll=max(1, period_ms / 2)) = ttl_ms + ttl_ms / 20 + poll: + # derived from the module's own lease constants (see their definition above) rather than + # hard-coded, so this stays correct if that fixed budget ever changes. + poll_ms = max(1, MOUNT_RENEW_PERIOD_MS // 2) + threshold_ms = MOUNT_LEASE_TTL_MS + MOUNT_LEASE_TTL_MS // 20 + poll_ms + observation = "waiting ~{} ms (token-stability observation)".format(threshold_ms) + epoch_before = int( + node.query( + "SELECT writer_epoch FROM system.cas_mounts WHERE disk = '{}' LIMIT 1".format(DISK) + ).strip() + ) + + # A hard kill leaves the previous incarnation's mount slot claimed; the restart must pay the + # token-stability observation wait once before it can safely reclaim it. + node.stop_clickhouse(kill=True) + node.start_clickhouse() + # `node.start_clickhouse()` only waits for the server to accept queries, not for the CAS disk's + # `Pool::open` (which performs the observation wait itself) to finish; reading the log or the mount + # row before that completes races the very thing being measured. Wait for the mount to report + # "live" first, then the log line and row are both settled. + _wait_until(lambda deadline: _mount_snapshot(node, timeout=deadline.remaining())["state"] == "live", timeout=120) + assert log_count_since_last_restart(observation) == 1 + assert _mount_snapshot(node)["state"] == "live" + # This restart already reclaims the slot and advances the epoch on its own (via the observation + # wait, not the knob), so the knob-restart's own advance must be measured from THIS value, not + # from epoch_before -- otherwise a knob-restart that wrongly reused this same epoch would still + # pass an `epoch_after > epoch_before` check. + epoch_after_safe_restart = int( + node.query( + "SELECT writer_epoch FROM system.cas_mounts WHERE disk = '{}' LIMIT 1".format(DISK) + ).strip() + ) + assert epoch_after_safe_restart > epoch_before + + # Enable the unsafe knob while the server is stopped (a test-stand-only config.d overlay), then + # hard-kill again: this server's own uuid already holds the slot, so the knob may reclaim it at + # once and skip the observation wait entirely. + node.stop_clickhouse(kill=True) + node.copy_file_to_container( + os.path.join(os.path.dirname(__file__), "configs/unsafe_remount.xml"), + "/etc/clickhouse-server/config.d/unsafe_remount.xml", + ) + try: + node.start_clickhouse() + # Same race as the safe restart above: wait for the mount to settle before reading the log. + _wait_until(lambda deadline: _mount_snapshot(node, timeout=deadline.remaining())["state"] == "live", timeout=120) + assert log_count_since_last_restart(observation) == 0 + assert _mount_snapshot(node)["state"] == "live" + epoch_after_knob_restart = int( + node.query( + "SELECT writer_epoch FROM system.cas_mounts WHERE disk = '{}' LIMIT 1".format(DISK) + ).strip() + ) + assert epoch_after_knob_restart > epoch_after_safe_restart + finally: + node.exec_in_container(["rm", "-f", "/etc/clickhouse-server/config.d/unsafe_remount.xml"]) diff --git a/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml b/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml index 74a0ba8de29c..4ccd3a04688d 100644 --- a/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml +++ b/tests/integration/test_cas_ref_snaplog/configs/storage_conf.xml @@ -8,6 +8,8 @@ object_storage s3 cas + 30 + 10000 itest-ref-snaplog http://rustfs1:11121/test/cas_snaplog_data/ clickhouse @@ -22,6 +24,8 @@ object_storage s3 cas + 30 + 10000 itest-ref-snaplog http://rustfs1:11121/test/cas_snaplog_data/ clickhouse diff --git a/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml b/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml index 4513d345a2fb..67c3d444d9be 100644 --- a/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml +++ b/tests/integration/test_cas_replicated_relink/configs/storage_conf.xml @@ -5,6 +5,8 @@ object_storage s3 cas + 30 + 10000 + + + + + + default + + + disk_cas_shared + + + + + + + disk_cas_other + + + disk_cas_shared + + + + + + diff --git a/tests/integration/test_cas_replicated_relink/test.py b/tests/integration/test_cas_replicated_relink/test.py index 83d8d239ba92..e0bb176ea5a2 100644 --- a/tests/integration/test_cas_replicated_relink/test.py +++ b/tests/integration/test_cas_replicated_relink/test.py @@ -22,6 +22,14 @@ OTHER_STORAGE_POLICY = "cas_other" OTHER_CA_DISK = "disk_cas_other" +# node2-only policies (configs/storage_conf_tiered.xml). `cas_tiered` = [local `default`] then +# [disk_cas_shared]: an ordinary reservation lands on `default`, so a relink onto the pool's disk is a +# forced placement. `cas_two_pools` = [disk_cas_other] then [disk_cas_shared]: the first content-addressed +# disk is the WRONG pool, so a relink onto disk_cas_shared proves the whole pool set was advertised. +TIERED_STORAGE_POLICY = "cas_tiered" +TWO_POOLS_STORAGE_POLICY = "cas_two_pools" +LOCAL_DISK = "default" + # The shared pool's blob prefix inside the `test` RustFS bucket. The relink proof is that the fetch does # NOT create new objects under here: relink publishes a ref (per-server, under store/), never a blob. BLOBS_PREFIX = "shared_pool/blobs/" @@ -66,6 +74,7 @@ def start_cluster(): "configs/storage_conf.xml", "configs/server_root_id_node2.xml", "configs/storage_conf_other_pool.xml", + "configs/storage_conf_tiered.xml", ], macros={"replica": "node2"}, with_rustfs=True, @@ -183,6 +192,21 @@ def active_part_names(node, table): ).split() +def part_disk(node, table, part): + """The disk the ACTIVE part of this name sits on, from `system.parts`.""" + return node.query( + "SELECT disk_name FROM system.parts WHERE database = 'default' AND table = '{}' " + "AND name = '{}' AND active".format(table, part) + ).strip() + + +def detached_part_disk(node, table, part): + return node.query( + "SELECT disk FROM system.detached_parts WHERE database = 'default' AND table = '{}' " + "AND name = '{}'".format(table, part) + ).strip() + + def any_state_part_count(node, table, part): return int( node.query( @@ -769,7 +793,10 @@ def test_confirm_refuses_when_source_dropped_in_window(): """Task 16 step 1 — the race the confirm exists to lose safely. Taxonomy row 3: the source cannot prove it still holds the offered manifest, so the receiver aborts - its durable `+1` and throws a retry-later `NETWORK_ERROR` INSTEAD of falling back to bytes. The two + its durable `+1` and throws a retry-later `NO_REPLICA_HAS_PART` INSTEAD of falling back to bytes. + That code is part of the contract (issue #2219): both queue executors demote it to INFO with no + stack trace, it stays recorded on the queue entry, and it is the one fetch-transient code the + stateless corpus already tolerates in `part_log` checks. The two assertions that matter are (a) the queue recovers by re-selecting — here, onto the covering part — and (b) NO byte re-request ever went to the source whose state was in doubt. (b) is the entire reason row 3 throws where rows 2 and 5 return `nullptr`. @@ -810,6 +837,31 @@ def test_confirm_refuses_when_source_dropped_in_window(): assert not log_lines(node2, relink_finished_pattern(table, part)) assert any_state_part_count(node2, table, part) == 0 + # (c) the refusal's CLASSIFICATION -- the contract pinned after issue #2219. The refusal must reach + # the operator as the tolerated fetch-transient `NO_REPLICA_HAS_PART` (both queue executors + # demote it to INFO, no stack trace; every stateless `part_log` hygiene check that whitelists + # that code -- e.g. `02265_column_ttl` -- stays green), never as an Error-level `NETWORK_ERROR` + # with a stack trace, which reads as a network fault and once cost a multi-hour false triage. + refusal_error_pattern = r".*did not prove it still holds the manifest" + refusal_info_pattern = r".*did not prove it still holds the manifest" + assert not log_lines(node2, refusal_error_pattern), ( + "the relink refusal is a designed outcome and must not be logged at Error level" + ) + assert log_lines(node2, refusal_info_pattern), ( + "the demoted refusal must still be visible at Information level -- silence would be worse than " + "the old noise" + ) + node2.query("SYSTEM FLUSH LOGS part_log") + stray_codes = node2.query( + "SELECT DISTINCT errorCodeToName(error) FROM system.part_log " + "WHERE table = '{}' AND error != 0 AND errorCodeToName(error) != 'NO_REPLICA_HAS_PART'".format( + table + ) + ).split() + assert stray_codes == [], ( + "a relink refusal must reach part_log only as NO_REPLICA_HAS_PART, got: {}".format(stray_codes) + ) + drop_everywhere(table) @@ -903,3 +955,263 @@ def test_stalled_publish_protects_source_blobs_and_commits_nothing(): ) drop_everywhere(table) + + +# ---------------------------------------------------------------------------------------------------- +# FORCED PLACEMENT: a relink lands on the pool's disk even when the storage policy would put the part +# elsewhere. Every test here asserts the relink line AND `system.parts.disk_name`, because the first +# alone would also hold for a relink that then got moved, and the second alone would hold for a byte +# fetch that happened to be reserved on the pool's disk. +# ---------------------------------------------------------------------------------------------------- + + +def _fetch_via_queue(node1, node2, table, node2_policy, create_sql=None): + """INSERT on node1 while node2's fetches are stopped, then let node2 fetch exactly that one part. + + Returns `(part, blobs_before)`: the part name and the pool's blob keys as they were AFTER the + insert on node1 and BEFORE node2 fetched — the only snapshot `assert_no_new_blobs` can be measured + against, since the insert itself writes the part's blobs. `create_sql` overrides the table DDL (it + must contain `{policy}` and `{zk}`). + """ + drop_everywhere(table) + if create_sql is None: + create_replicated(node1, table) + create_replicated(node2, table, policy=node2_policy) + else: + node1.query(create_sql.format(policy=STORAGE_POLICY, zk="/clickhouse/tables/" + table)) + node2.query(create_sql.format(policy=node2_policy, zk="/clickhouse/tables/" + table)) + node2.query("SYSTEM STOP FETCHES {}".format(table)) + insert_rows(node1, table, 0) + part = active_part_names(node1, table)[0] + blobs_before = blob_keys() + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=90) + return part, blobs_before + + +def test_tiered_policy_relinks_onto_cas_over_volume_order(): + """`[default] then [disk_cas_shared]`: the policy's own placement is the local volume, and before the + forced placement the fetch reserved there, failed the pool post-check and downloaded bytes onto + `default`. Now the offer decides the disk.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "tiered_order" + + part, blobs_before = _fetch_via_queue(node1, node2, table, TIERED_STORAGE_POLICY) + + assert_relinked(node2, table, part) + assert part_disk(node2, table, part) == CA_DISK + assert_no_new_blobs(blobs_before) + assert int(node2.query("SELECT count() FROM {}".format(table))) == NUM_ROWS + drop_everywhere(table) + + +def test_relink_carries_projection_under_tiered_policy(): + """A projection-bearing part relinks like any other (the projection is loaded from the published + manifest), and the forced placement does not disturb that.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "tiered_projection" + create_sql = ( + "CREATE TABLE " + table + " (id Int64, v UInt64, s String, " + "PROJECTION p_by_s (SELECT s, sum(v) GROUP BY s)) " + "ENGINE = ReplicatedMergeTree('{zk}', '{{replica}}') ORDER BY id " + "SETTINGS storage_policy = '{policy}'" + ) + + part, _ = _fetch_via_queue(node1, node2, table, TIERED_STORAGE_POLICY, create_sql=create_sql) + + assert_relinked(node2, table, part) + assert part_disk(node2, table, part) == CA_DISK + assert int(node2.query( + "SELECT count() FROM system.projection_parts WHERE database = 'default' AND table = '{}' " + "AND parent_name = '{}' AND name = 'p_by_s' AND active".format(table, part) + )) == 1 + assert int(node2.query("SELECT sum(v) FROM {} WHERE s = '7'".format(table))) == 70 + drop_everywhere(table) + + +def test_two_pool_policy_relinks_into_second_pool(): + """`[disk_cas_other] then [disk_cas_shared]`: the first content-addressed disk is the WRONG pool. + A single-pool advertise names `other`, the sender declines, and the bytes land on `disk_cas_other`; + advertising the whole set lets the sender match `shared` and the receiver place it there.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "two_pools" + + part, blobs_before = _fetch_via_queue(node1, node2, table, TWO_POOLS_STORAGE_POLICY) + + assert_relinked(node2, table, part) + assert part_disk(node2, table, part) == CA_DISK + assert_no_new_blobs(blobs_before) + assert not log_lines(node2, download_finished_pattern(table, part, disk=OTHER_CA_DISK)) + drop_everywhere(table) + + +def test_mechanism_failure_falls_back_to_bytes_on_forced_disk(): + """A relink that fails for a mechanism reason re-requests the bytes on the SAME forced disk — the + placement decision outlives the relink. (The one-offer recursion bound is proven by + `test_recursion_brake_bounds_relink_to_one_attempt`; this test proves only the destination.)""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "tiered_fallback" + drop_everywhere(table) + create_replicated(node1, table) + create_replicated(node2, table, policy=TIERED_STORAGE_POLICY) + node2.query("SYSTEM STOP FETCHES {}".format(table)) + insert_rows(node1, table, 0) + part = active_part_names(node1, table)[0] + + node2.query("SYSTEM ENABLE FAILPOINT cas_relink_receiver_force_mechanism_failure") + try: + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=90) + assert_byte_downloaded(node2, table, part, disk=CA_DISK) + assert part_disk(node2, table, part) == CA_DISK + assert not log_lines(node2, download_finished_pattern(table, part, disk=LOCAL_DISK)) + finally: + node2.query("SYSTEM DISABLE FAILPOINT cas_relink_receiver_force_mechanism_failure") + drop_everywhere(table) + + +def test_detached_fetch_relinks_onto_cas_under_tiered_policy(): + """`ALTER TABLE ... FETCH PART` into `detached/` under the tiered policy: the forced placement + applies to detached fetches too, and the detached part is on the pool's disk.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + src, dst = "tiered_det_src", "tiered_det_dst" + drop_everywhere(src) + drop_everywhere(dst) + create_replicated(node1, src) + create_replicated(node2, dst, policy=TIERED_STORAGE_POLICY) + insert_rows(node1, src, 0) + part = active_part_names(node1, src)[0] + + node2.query( + "ALTER TABLE {dst} FETCH PART '{part}' FROM '/clickhouse/tables/{src}'".format( + dst=dst, part=part, src=src + ) + ) + + assert_relinked(node2, dst, part) + assert detached_part_disk(node2, dst, part) == CA_DISK + node2.query("ALTER TABLE {} ATTACH PART '{}'".format(dst, part)) + assert part_disk(node2, dst, part) == CA_DISK + assert int(node2.query("SELECT count() FROM {}".format(dst))) == NUM_ROWS + drop_everywhere(src) + drop_everywhere(dst) + + +def test_offer_for_unavailable_pool_falls_back_to_ordinary_placement(): + """The receiver resolved a forced disk and then lost it (the failpoint stands in for an offer this + policy has no disk for): the ordinary reservation runs, the relink block sees a disk outside the + offered pool, and the bytes go where the policy says — `default` under the tiered policy. Exactly + one offer is made: the byte re-request carries no advertise.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "tiered_dropped" + drop_everywhere(table) + create_replicated(node1, table) + create_replicated(node2, table, policy=TIERED_STORAGE_POLICY) + node2.query("SYSTEM STOP FETCHES {}".format(table)) + insert_rows(node1, table, 0) + part = active_part_names(node1, table)[0] + + node2.query("SYSTEM ENABLE FAILPOINT cas_relink_receiver_drop_forced_disk") + try: + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=90) + assert log_lines( + node2, + r"Failpoint cas_relink_receiver_drop_forced_disk: forgetting the forced disk for part {}".format( + re.escape(part) + ), + ) + assert_byte_downloaded(node2, table, part, disk=LOCAL_DISK) + assert part_disk(node2, table, part) == LOCAL_DISK + assert len(log_lines(node1, relink_offer_pattern(table, part))) == 1 + finally: + node2.query("SYSTEM DISABLE FAILPOINT cas_relink_receiver_drop_forced_disk") + drop_everywhere(table) + + +def test_offer_without_pool_cookie_resolves_to_single_advertised_pool(): + """The old-sender shape: an offer with no `cas_pool_uuid` cookie. With ONE advertised pool that is + the pool, and the relink is forced as usual; with TWO advertised pools the receiver refuses to guess, + the ordinary reservation lands on the first volume (`disk_cas_other`), and the bytes go there.""" + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + node1.query("SYSTEM ENABLE FAILPOINT cas_relink_sender_omit_pool_cookie") + try: + one_pool = "omit_cookie_one" + part, _ = _fetch_via_queue(node1, node2, one_pool, TIERED_STORAGE_POLICY) + assert_relinked(node2, one_pool, part) + assert part_disk(node2, one_pool, part) == CA_DISK + drop_everywhere(one_pool) + + two_pools = "omit_cookie_two" + part, _ = _fetch_via_queue(node1, node2, two_pools, TWO_POOLS_STORAGE_POLICY) + assert_byte_downloaded(node2, two_pools, part, disk=OTHER_CA_DISK) + assert part_disk(node2, two_pools, part) == OTHER_CA_DISK + assert len(log_lines(node1, relink_offer_pattern(two_pools, part))) == 1 + drop_everywhere(two_pools) + finally: + node1.query("SYSTEM DISABLE FAILPOINT cas_relink_sender_omit_pool_cookie") + + +def test_relink_wins_over_ttl_then_mover_converges(): + """A `TTL ... TO DISK` rule that names the LOCAL disk for this (already expired) part does not stop the + relink: the part lands on the pool's disk at zero byte cost, and the background mover — which sees a + part that is not in its TTL destination — carries it to `default` afterwards. Moves are stopped on + node2 around the fetch so the intermediate placement is observable, exactly as `test_ttl_move` does. + + `IF EXISTS` precedes the disk name in the grammar. It is there for node1, whose policy has no + `default` disk: without it `CREATE TABLE` on node1 fails with `BAD_TTL_EXPRESSION`, because + `MergeTreeData::checkTTLExpressions` rejects a `TO DISK` destination absent from the policy at + create time. + """ + node1 = cluster.instances["node1"] + node2 = cluster.instances["node2"] + table = "tiered_ttl" + drop_everywhere(table) + create_sql = ( + "CREATE TABLE " + table + " (id Int64, v UInt64, s String, ts DateTime) " + "ENGINE = ReplicatedMergeTree('/clickhouse/tables/" + table + "', '{{replica}}') ORDER BY id " + "TTL ts TO DISK IF EXISTS 'default' " + "SETTINGS storage_policy = '{policy}'" + ) + node1.query(create_sql.format(policy=STORAGE_POLICY)) + node2.query(create_sql.format(policy=TIERED_STORAGE_POLICY)) + + node2.query("SYSTEM STOP MOVES {}".format(table)) + node2.query("SYSTEM STOP FETCHES {}".format(table)) + try: + node1.query( + "INSERT INTO {table} SELECT number, number * 10, toString(number), now() - INTERVAL 1 DAY " + "FROM numbers({rows})".format(table=table, rows=NUM_ROWS) + ) + part = active_part_names(node1, table)[0] + assert part_disk(node1, table, part) == CA_DISK # the sender holds it in the pool + + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM SYNC REPLICA {}".format(table), timeout=90) + + # The TTL rule says `default`; the relink put it on the pool's disk anyway, and moves are stopped. + assert_relinked(node2, table, part) + assert part_disk(node2, table, part) == CA_DISK + rows_before = node2.query("SELECT count(), sum(v) FROM {}".format(table)) + + node2.query("SYSTEM START MOVES {}".format(table)) + wait_until( + lambda: part_disk(node2, table, part) == LOCAL_DISK, + timeout=120, + what="the background mover carrying {} to {}".format(part, LOCAL_DISK), + ) + assert node2.query("SELECT count(), sum(v) FROM {}".format(table)) == rows_before + assert node2.query("SELECT count(), sum(v) FROM {}".format(table)) == node1.query( + "SELECT count(), sum(v) FROM {}".format(table) + ) + finally: + node2.query("SYSTEM START FETCHES {}".format(table)) + node2.query("SYSTEM START MOVES {}".format(table)) + drop_everywhere(table) diff --git a/tests/integration/test_cas_s3/configs/storage_conf.xml b/tests/integration/test_cas_s3/configs/storage_conf.xml index 55e2eedfc6fb..773441a00944 100644 --- a/tests/integration/test_cas_s3/configs/storage_conf.xml +++ b/tests/integration/test_cas_s3/configs/storage_conf.xml @@ -15,9 +15,9 @@ clickhouse - 60 - 100 - 5000 + 30 + 10000 + 1000 7 3
X-Cas-Test: 1
diff --git a/tests/integration/test_cas_shared_pool/configs/storage_conf.xml b/tests/integration/test_cas_shared_pool/configs/storage_conf.xml index a29a03741ef9..8638d5530ef5 100644 --- a/tests/integration/test_cas_shared_pool/configs/storage_conf.xml +++ b/tests/integration/test_cas_shared_pool/configs/storage_conf.xml @@ -5,6 +5,8 @@ object_storage s3 cas + 30 + 10000 Reading: GET gc/state - Reading --> Creating: object absent, never observed before - Creating --> Leader: casPut create-if-absent, cas_gc_shards fixed here, once - Reading --> Renewing: lease owner is me - Renewing --> Leader: casPut seq+1, guarded by the observed token - Reading --> Evaluating: foreign owner - Evaluating --> NotLeader: incumbent lease moved, or heartbeat moved, or steal not allowed - Evaluating --> Stealing: both frozen across a full observation window - Stealing --> Leader: casPut owner=me seq+1, on the observed token - Stealing --> NotLeader: lost the CAS, re-read and re-arm - Leader --> [*]: run the round -``` - -Two independent liveness signals are consulted before a steal: whether `(owner, seq)` moved since -the last tick, and whether the separate `gc/hb` heartbeat moved. The heartbeat is compared only -under the same remembered heartbeat owner, deliberately not against `lease.owner` — a deposed -leader's heartbeat thread keeps pulsing, and that must not cause a live new leader's lease to be -stolen. The paced background loop may steal; a manual `SYSTEM CAS GC RUN` may not, because the -safety argument needs two observations separated by real wall time. Because every renew or steal -bumps `seq`, `seq` doubles as the round's attempt id. - -**A deposed leader that keeps running cannot corrupt anything**, and the argument does not rely on -exclusivity at all: - -1. `gc/state` is published by exactly **one** `CAS` per round; a deposed leader's `CAS` fails and - its entire round evaporates. -2. Every fold artifact is written under that leader's own attempt number, invisible to every - reader, and reclaimed later by wholesale generation pruning. -3. Destructive pre-`CAS` actions are justified only by previously published durable state, so they - are replay-idempotent. -4. Deletes are exact-token, so a stale leader can never delete a newer incarnation. - -The lease is therefore **work de-duplication, not mutual exclusion**. - ## The round {#the-round} -A round is one pass of 18 named phases ending in exactly one `gc/state` `CAS` -(`Gc::runRegularRound`, `Gc/CasGc.cpp`). - -| # | Phase (`GcPhaseTimer` name) | What it does | -|---|---|---| -| 1 | `lease` | Acquire, renew or steal the lease inside `gc/state`. The only phase a not-a-leader round emits | -| 2 | `pre_fold_ref_drain` | Resolve catalog `Removing` rows whose cleanup evidence the adopted parent already sealed; exact-CAS-delete the completed ones before anything else can act | -| 3 | `heartbeat_floor` | One `LIST` of `gc/server-roots/`, one `GET` per mount slot, fence-out `PUT` for any mount whose write-token has held stable past the threshold | -| 4 | `defer_decision` | One full `LIST` of `cas/ns/stream/`, build the catalog-keyed ref walk plan; decide `DEFER` (nothing changed, no graduation due) or continue to a full fold. A `DEFER` verdict still runs one namespace-janitor page — the same work phase 16 does on a folding round — with its deletes suppressed | -| 5 | `parent_seal_read` | Capture the parent fold seal's run references before the fold mutates the in-memory generation/attempt, to detect a ref that moved off an already-pruned generation | -| 6 | `fold_ref_group` | Regroup the one `LIST` from phase 4 into per-table listings — no I/O, the keys are already in hand | -| 7 | `fold_seal_read` | `GET` and decode the adopted fold seal that anchors this fold's coverage | -| 8 | `fold_ref_intake` | `GET` every new ref-log record and every referenced manifest, extracting blob source edges | -| 9 | `fold_reduce` | The three-cursor merge over prior edges, new deltas and the parent's condemned rows: spare, condemn, graduate or redelete each candidate | -| 10 | `fold_seal_write` | Write the new fold seal once, write-once deterministic, adopting a byte-identical replay instead of rewriting it | -| 11 | `pending_deletes` | The single content-delete site: exact-token `deleteExact` of every entry the *previous* round marked `delete_pending`, plus the forensic outcome-log writes | -| 12 | `meta_pool_wait` | Drain the bounded pool of async `.meta` condemn-marker writes queued during the fold | -| 13 | `round_commit` | Retention-prune old generations, then publish the single `gc/state` `CAS` that adopts the whole round | -| 14 | `handoff_reclaim` | Post-`CAS`: reclaim any generation that a ref moved off during this very round, before the ordinary wholesale prune would reach it | -| 15 | `manifest_deletes` | Delete manifest bodies whose owner-removal minus-one edge the `CAS` in phase 13 just adopted | -| 16 | `namespace_cleanup` | One bounded page of the perpetual namespace janitor, reclaiming dead-life debris | -| 17 | `ref_object_cleanup` | Prune ref logs and snapshots once both fold coverage and a live snapshot make them safe to delete | -| 18 | `orphan_sweep` | Post-`CAS` exact-token deletion for the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep), after phase 13 adopted each candidate's exact blob-source retirements and the cursor. Planning retains, counts, logs, and advances past undecodable bodies without wedging later candidates; a decoded identity mismatch during planning or token ABA during deletion still fails the round with `CORRUPTED_DATA` | - -Phases 5 through 18 run only when phase 4 decides to fold. A `DEFER` verdict is not a bare no-op: -it still runs one bounded namespace-janitor page with `suppress_destructive = true` — cursor -progress and diagnostics only, no deletes — and then returns, publishing no fold artifact and no -`gc/state` `CAS` at all: +A folding round execution is one pass of 18 named phases ending in exactly one commit `CAS` over +`gc/state` (`Gc::runRegularRound`, `Gc/CasGc.cpp`). Every round execution starts with phase 1, but a +follower or a deferred round execution returns before that commit. + +| # | Phase (`GcPhaseTimer` name) | Runs on | What it does | +|---|---|---|---| +| 1 | `lease` | always | Create, renew, observe or steal the lease inside `gc/state`. The only phase a `NotALeader` round emits | +| 2 | `pre_fold_ref_drain` | leader | Resolve catalog `Removing` rows whose cleanup evidence the adopted parent already sealed; drop the completed ones before defer or new fold work | +| 3 | `heartbeat_floor` | leader | One `LIST` of `gc/server-roots/`, one `GET` per mount slot, fence-out `PUT` for any mount whose write-token has held stable past the threshold | +| 4 | `defer_decision` | leader | One full `LIST` of `cas/ns/stream/`, build the catalog-keyed ref walk plan; decide `DEFER` (fewer changed rows than the fold threshold, no graduation due, defer limit not reached) or continue to a full fold | +| 5 | `parent_seal_read` | fold | Capture the parent fold seal's run references before the fold mutates the in-memory generation/attempt | +| 6 | `fold_ref_group` | fold | Regroup the one `LIST` from phase 4 into per-namespace listings — no I/O, the keys are already in hand | +| 7 | `fold_seal_read` | fold | `GET` and decode the adopted fold seal that anchors this fold's coverage | +| 8 | `fold_ref_intake` | fold | `GET` every new ref-log record and every referenced manifest, extracting blob source edges | +| 9 | `fold_reduce` | fold | Merge prior edges, new deltas and the parent's condemned rows: spare, condemn, graduate or redelete each candidate; compute `suppress_destructive` | +| 10 | `fold_seal_write` | fold | Write the new fold seal once, write-once deterministic, adopting a byte-identical replay instead of rewriting it | +| 11 | `pending_deletes` | fold | The single blob-body delete site: exact-token delete of entries a *previous* round marked `delete_pending`, up to the redelete budget, plus the outcome-log writes | +| 12 | `meta_pool_wait` | fold | Drain the bounded pool of async `.meta` condemn-marker writes queued during the fold | +| 13 | `round_commit` | fold | Retention-prune old generations, then publish the single `gc/state` `CAS` that adopts the whole round | +| 14 | `handoff_reclaim` | post-`CAS` | Reclaim a generation a ref moved off during this round, which the ordinary retention prune already skipped and will not revisit | +| 15 | `manifest_deletes` | post-`CAS` | Delete manifest bodies whose owner-removal minus-one edge the `CAS` in phase 13 just adopted | +| 16 | `namespace_cleanup` | leader; suppressed on `DEFER` | One bounded page of the perpetual namespace janitor, reclaiming dead-life debris | +| 17 | `ref_object_cleanup` | post-`CAS` | Prune ref logs and snapshots once both fold coverage and a live snapshot make them safe to delete | +| 18 | `orphan_sweep` | post-`CAS` | Exact-token deletion for the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep), after phase 13 adopted each candidate's blob-source retirements and the cursor | + +Phases 2 through 4 run on every leader round; a follower returns after phase 1. Phases 5–15 and +17–18 run only when phase 4 decides to fold; phase 16 runs after phase 15 on a fold and right after +phase 4 on a `DEFER`. A `DEFER` verdict is therefore not a bare no-op: it still runs one bounded +namespace-janitor page with `suppress_destructive = true` — listing and classification only, no +deletes and no cursor advance — and then returns, publishing no fold artifact and no commit `CAS`. +Its lease `CAS` may already have created or renewed the lease in phase 1: ```mermaid flowchart LR - D4{"4 defer_decision"} -->|"nothing changed, no graduation due"| DEF["DEFER: one suppressed
namespace-janitor page, then return"] - D4 -->|"changed shards, or graduation due"| FOLD["phases 5 through 18: full fold and round commit"] + D4{"4 defer_decision"} -->|"below the fold threshold,
no graduation due"| DEF["DEFER: one suppressed
namespace-janitor page, then return"] + D4 -->|"changed namespace rows, or graduation due"| FOLD["phases 5 through 18: full fold and round commit"] ``` Orderings that are load-bearing: @@ -111,15 +88,554 @@ perpetual namespace janitor, so they cannot desynchronize. Under suppression the graduation, no redelete, and no ref or namespace deletion; condemnation and sparing continue, because both are non-destructive. -**Fail-closed aborts.** A throw before the `CAS` means nothing is adopted: unapplied transactions, -a cursor/apply mismatch, a missing adopted seal, a table with a snapshot but no surviving log and -no cursor, a non-total condemned summary, and an observed delete marker (bucket versioning is on). +**Fail-closed aborts.** A throw before the commit `CAS` means no successor state is adopted (the +exact-token deletes and prune that earlier phases justified from *previously* published state may +already have run; they are idempotent): unapplied transactions and a cursor/apply mismatch (both +`CORRUPTED_DATA`, checked between phases 9 and 10), a missing adopted seal, a table with a snapshot +but no surviving log and no cursor, a non-total condemned summary (all `CORRUPTED_DATA`), and an +observed delete marker (`LOGICAL_ERROR`: bucket versioning is on). The per-phase "fails the round +if" lists below name the protocol checks; an uncaught backend or decode exception fails the round +execution too. + +## Phase 1 — lease {#phase-1-lease} + +Establishes whether this GC may run the round. There is no separate lease object: the lease lives +inside `gc/state` as `{owner, seq}`, and a round is authorized by winning a token-guarded `CAS` on +that object. + +- **Runs on:** always — the only phase a `NotALeader` round emits +- **Reads:** `gc/state`; `gc/hb` (only when another GC owns the lease) +- **Writes / deletes:** one successful `CAS` on `gc/state` (acquire, renew or steal); a conflict is + re-observed and re-decided by the request engine within the standard 90-second write policy; no + deletes +- **Safety:** the lease is *work de-duplication, not mutual exclusion* — see below +- **Fails the round if:** this `Gc` instance saw `gc/state` before and it has since disappeared, or + the `gc_shards` in `gc/state` disagrees with the pool's `_pool_meta` value (both `CORRUPTED_DATA`) +- **Observability:** phase row `lease`; metrics `acquired`, `steal_allowed` + +```mermaid +%%{init: {"flowchart": {"curve": "linear", "nodeSpacing": 15, "rankSpacing": 20}, "themeVariables": {"lineColor": "#000000"}}}%% +flowchart TD + READ["GET gc/state"] --> EXISTS{"gc/state exists?"} + EXISTS -->|yes| OWNER{"lease.owner = gc_id?"} + EXISTS -->|no| OBSERVED{"Observed before?"} + OWNER -->|yes| RENEW["Renew"] + OWNER -->|no| STEALABLE{"Steal allowed and both
lease and heartbeat frozen?"} + OBSERVED -->|yes| CORRUPT["CORRUPTED_DATA"] + OBSERVED -->|no| ACQUIRE["Acquire"] + STEALABLE -->|yes| STEAL["Steal"] + STEALABLE -->|no| FOLLOWER["Follower"] +``` + +**Acquire / renew.** If `gc/state` is absent and never seen, this GC creates it with `owner = gc_id`, +`seq = 1`, and fixes `gc_shards`. If `lease.owner` is already this `gc_id`, it increments `seq` with +a token-guarded `CAS`. Non-lease fields are always preserved; a conflict causes a bounded re-read. + +**Follower / steal.** If another GC owns the lease, this GC reads `gc/hb` and normally returns +`NotALeader`. The leader advances an advisory `gc/hb` object; the heartbeat proves activity but +grants no authority. A candidate compares `(lease.owner, lease.seq)` and the `gc/hb` writer/sequence +against what it recorded on its *previous* scheduled round. If either signal moved, the owner is +live and the candidate records a fresh observation and backs off. A steal is allowed only when both +signals stayed frozen across two paced observations; the candidate then rewrites `lease.owner`, +increments `seq`, and `CAS`es. A conflict is re-observed and re-decided; if the refreshed state +shows a live incumbent the steal declines and returns `NotALeader`. A new process has no +recorded observations, so it can never steal on first sight. A manual `SYSTEM CAS GC RUN` may +acquire or renew but never steals. + +**The `CAS` token.** Acquire / renew / steal return the `gc/state` version's backend token; phase 13's +commit `CAS` uses it, so any intervening write to `gc/state` rejects a stale leader's commit. The +resulting `lease.seq` is the folding round's attempt id. + +**Safety after leadership changes.** A deposed leader may keep running; correctness does not need +exclusive execution: + +1. a folding round is published by exactly one commit `CAS` — a deposed leader's commit fails and + its candidate generation is never adopted; +2. every fold artifact is written under that leader's own attempt number, invisible to readers and + reclaimed by generation pruning; +3. destructive pre-`CAS` actions are justified only by previously published durable state, so they + are replay-idempotent; +4. deletes are exact-token, so a stale leader can never delete a newer incarnation. + +## Phase 2 — pre-fold ref drain {#phase-2-pre-fold-ref-drain} + +Finishes namespace removals that the last committed fold already proved safe: removes the matching +`Removing` row from the ref catalog. Removal is split across rounds — one fold writes +`cleanup_evidence` into its seal, a *later* round's phase 2 acts on it — so phase 2 never trusts +evidence from the fold running now. + +- **Runs on:** always (leader), including on a round that later returns `Deferred` +- **Reads:** the adopted fold seal (`ref_lives` coverage + `cleanup_evidence` only); the ref + catalog; `gc/state` (leadership re-check around every write) +- **Writes / deletes:** rewrites the ref catalog via `CAS` to drop each eligible row (not a backend + `DELETE`); no blob, manifest or namespace-object deletes +- **Safety:** a row is dropped only if the adopted parent seal carries `cleanup_evidence` for the + *same* `incarnation` and that life has no coverage hold; each write is token-guarded and bracketed + by two `gc/state` re-reads +- **Fails the round if:** `gc/state` points at a missing parent seal (`CORRUPTED_DATA`); leadership + changes mid-drain (`NETWORK_ERROR`, an `Aborted` finish that the next round retries — a write + already sent stays safe, authorized by the parent seal and the catalog token) +- **Observability:** phase row `pre_fold_ref_drain`; metric `deleted` (rows removed) + +On a fresh pool (`snap_generation = 0`) phase 2 is a no-op that still emits its phase row; it can +first remove a row only in a leader round *after* some earlier fold committed a generation. Phase 3 +starts only once every parent-authorized removal has been resolved — phase 2 is a barrier. Phase 16 +later deletes the old namespace's physical `_log` / `_snap` / `_ckpt` / `_files` objects. + +## Phase 3 — heartbeat floor {#phase-3-heartbeat-floor} + +Checks each mounted writer for liveness and fences any writer incarnation that stopped renewing its +mount lease. Despite the name it does not touch `gc/hb` (that is phase 1); it watches the backend +token of each `mount` object. + +- **Runs on:** always (leader), fold and deferred paths +- **Reads:** `LIST` of `gc/server-roots/`, then a `GET` of each `/mount` +- **Writes / deletes:** a token-guarded `PUT` per fenced mount (`gc_fenced = true`, `seq + 1`); no + deletes +- **Safety:** fences only after this leader's *own* monotonic clock has watched the mount's + write-token hold unchanged for `cas_mount_lease_ttl_ms + 5% + cas_mount_renew_period_ms` + (defaults 30 s and 10 s; every server sharing the pool must run the same values); the write is + guarded by that exact token. A completed fence-out stays valid even if this GC later loses + leadership. `expires_at_ms` (another host's wall clock) is never trusted. +- **Fails the round if:** nothing — a per-mount `PUT` conflict is re-observed and re-classified + within the standard write policy, and a mount whose incarnation moved is classified `live` +- **Observability:** phase row `heartbeat_floor`; metrics `live`, `terminated`, `fenced_now`, + `already_fenced`; `GcFenceOut` audit rows in `system.cas_log` + +The first observation of any mount is always `live`; a new `Gc` instance starts with an empty +observation map, so it can only delay a fence-out, never do one early. A mount with +`min_active = UINT64_MAX` is a clean farewell (`terminated`). Results feed metrics and events only — +phase 4 takes no input from them. + +## Phase 4 — defer decision {#phase-4-defer-decision} + +Builds the round's one ref-stream work plan and decides `fold` vs `defer`. `fold` continues to +phase 5 and builds a new in-degree snapshot; `defer` skips generation construction and the commit. + +- **Runs on:** always (leader) +- **Reads:** one full `LIST` of `cas/ns/stream/` (key names only); the ref catalog; the adopted fold + seal (twice, on an adopted generation) +- **Writes / deletes:** none +- **Safety:** a missing / invalid / incomplete adopted seal cannot produce a quiet defer — the + graduation check refuses to defer and the plan read surfaces the bad state +- **Fails the round if:** the plan-building seal read reports an invalid adopted seal +- **Observability:** phase row `defer_decision`; metrics `changed_shards` (despite the name, the + number of changed namespace-life rows), `namespaces_seen`, `ref_log_keys_listed` + +The plan has one row per admitted catalog life (`Live` or `Removing`; `Creating` excluded), joining +its last folded position, its greatest listed `_log` position, and the keys later phases need. A row +is *changed* when the listed `_log` is newer than the last folded position. Phase 4 chooses `fold` +if any of: changed rows ≥ `gc_fold_threshold` (default 1); an adopted shard has a published pending +delete; an adopted shard has a condemned blob due to graduate +(`oldest_nonpending_condemn_round < round + 1`); or `gc_fold_max_defer_rounds` (default 8) +consecutive defers were reached. Both thresholds are internal `PoolConfig` fields, not disk +settings. On `defer`, one suppressed namespace-janitor page runs (phase 16's work) and the round +returns without a commit. On `fold`, phase 6 reuses this plan and the same `LIST`. + +## Phase 5 — parent seal read {#phase-5-parent-seal-read} + +Copies the *previously* adopted fold seal's run references into memory before the new fold can +overwrite them. "Parent" is the prior GC state the new one builds on, not a key-hierarchy parent. + +- **Runs on:** fold path only +- **Reads:** the adopted fold seal (its run-reference list only, never the run objects) +- **Writes / deletes:** none +- **Safety:** read-only; a stale leader's saved list only feeds decisions its own commit `CAS` gates +- **Fails the round if:** the seal is unreadable — a seal that vanished after phase 4 yields an + empty list here, and phase 7 re-checks and reports `CORRUPTED_DATA` before the commit +- **Observability:** phase row `parent_seal_read` + +An adopted seal can reference a run stored under an older generation, and the new fold may replace +that run and stop referencing its old generation. Keeping the parent's references lets phase 13 +shield those generations from retention pruning if the commit loses, and lets phase 14 reclaim a +generation the parent referenced but the new seal no longer does. Empty on a fresh pool. + +## Phase 6 — fold ref group {#phase-6-fold-ref-group} + +Regroups phase 4's flat key list into per-namespace listings against the catalog cut phase 4 read. +No backend I/O. + +- **Runs on:** fold path only +- **Reads:** nothing (keys already in memory from phase 4) +- **Writes / deletes:** none +- **Safety:** the fold is catalog-authoritative — a namespace exists iff the catalog cut names its + incarnation; the `LIST` is only a per-namespace hint +- **Fails the round if:** a ref-object key under the stream prefix is unparseable — the round aborts + the ref walk, produces no ref delta, advances no cursor, records the `ref_folding_aborted` + anomaly, and forces `suppress_destructive` for the whole round (not re-raised) +- **Observability:** phase row `fold_ref_group`; metrics `ref_keys_listed`, `namespaces_seen`, + `ref_folding_aborted` + +Also computes an empty-universe proof (catalog snapshot has a token and zero entries of any state) +used by phase 9's destructive gate. A malformed key does not skip phase 8's `_ckpt` reads. + +## Phase 7 — fold seal read {#phase-7-fold-seal-read} + +Reads the adopted fold seal that anchors this fold's coverage and sets up the fold's base state (the +prior coverage view, the mutable successor, the new generation / attempt numbers). + +- **Runs on:** fold path only +- **Reads:** the adopted fold seal, twice at the same address. The first read serves only the + missing-seal check below; the second supplies the parent run references and the condemned + summary. The second is a known redundant `GET` (same generation, attempt and bytes), counted by + the `redundant_reads` metric and left in place until a follow-up removes it (see + [per-phase backend cost](#per-phase-cost)) +- **Writes / deletes:** none +- **Safety:** read-only +- **Fails the round if:** the adopted seal is absent while `snap_generation > 0` — `gc/state` points + at a missing artifact; the fix is `SYSTEM CAS GC REBUILD` (`CORRUPTED_DATA`) +- **Observability:** phase row `fold_seal_read`; metrics `parent_ref_lives`, `parent_runs`, + `parent_cleanup_evidence`, `redundant_reads` + +The fold's writes land under `attempt = lease.seq` and `new_generation = snap_generation + 1`, while +reads of the parent generation keep using `snap_attempt`. On a fresh pool both reads return nothing +and the fold starts from an empty baseline. + +## Phase 8 — fold ref intake {#phase-8-fold-ref-intake} + +Reads every new ref-log record of every walkable namespace and the manifests it references, +extracting blob source edges. Usually the dominant read phase on ref-log- and manifest-heavy rounds. + +- **Runs on:** fold path only +- **Reads:** one `_ckpt` per namespace in the universe; `_log` records from each namespace's cursor + up to its committed ceiling; one manifest body per folded owner edge; extra `_log` reads when the + walk crosses an epoch seal +- **Writes / deletes:** none (the successor seal's `cleanup_evidence` rows are written between this + phase's timer and phase 9's) +- **Safety:** per-namespace failures stay per-namespace — a *hold*, never a whole-round abort. A + concurrent writer appending mid-round changes nothing: the ceiling (`committed_through`) is + snapshotted once, so the round folds a fixed amount of work. Transactions apply atomically; the + durable cursor advances once per fully folded record. +- **Fails the round if** (`CORRUPTED_DATA`): a manifest body whose ref / namespace disagrees with its + key; a table with no sealed cursor whose baseline logs are already gone; a sealed cursor that does + not close the run the walk produced; a `RemoveNamespace` for a namespace absent from the catalog + cut. (`logs_accounted` and `logs_applied` are counted here but compared only after phase 9 — see + there) +- **Observability:** phase row `fold_ref_intake`; metrics `frontier_namespaces` / `frontier_proven` + (universe and its proven part), `tables_held`, `logs_accounted` / `logs_applied`; per-cause hold + reasons are in [GC anomalies](#gc-anomalies) + +**The universe** is exactly the `Live` and `Removing` rows of the frozen catalog cut. A namespace +the phase-4 hint omitted, with no carried hold and no `_ckpt`, is walked only while +`gc_frontier_probe_budget` lasts; once spent, the rest ride their cursors verbatim and the round is +suppressed. **The walk** starts at `cursor + 1` (or the checkpoint's genesis position) and stops at +the committed ceiling; a namespace is *proven* only when it reaches the ceiling exactly. Any other +exit — a hold, an unusable checkpoint, the probe budget — leaves it unproven, which feeds phase 9's +gate. + +**Read-ahead.** The checkpoint, walk-position, manifest-edge and (in phase 9) zero-candidate `HEAD` +reads are hinted ahead onto a bounded pool (`cas_gc_read_concurrency`, default 16; `1` disables) and +taken by the walk at exactly the sites, and in exactly the order, of the inline reads, so every +decision, decode, counter and event stays on the round thread and the phase's semantic metrics do +not depend on the setting. Two things do: a request a worker performed lands on that worker's +`ProfileEvents`, not the phase row's, and a hinted key the walk never takes (a namespace held below +its lookahead, a `HEAD` candidate that kept an edge) is a wasted request. `CASGCReadAheadHit` and +`CASGCReadAheadMiss` are charged to the phase that takes the result; `CASGCReadAheadWasted` is +counted when the reader is destroyed after phase 10, so it shows up in the round-level +`ProfileEvents`, not on the phase-8 or phase-9 row. + +## Phase 9 — fold reduce {#phase-9-fold-reduce} + +Recomputes the per-shard in-degree snapshot and computes the round's single destructive gate. + +- **Runs on:** fold path only +- **Reads:** streaming `GET` of each referenced parent run segment; one `HEAD` per zero-in-degree + candidate; one `.meta` `GET` per graduation candidate lacking in-process marker confirmation. When + orphan-sweep planning runs: a `LIST` page of `cas/manifests/`, a `GET` per candidate, plus + `gc/state`, the adopted seal, the catalog, and per-namespace `_ckpt` / tail `_log`. +- **Writes / deletes:** one `PUT` per rewritten run segment; schedules the async `.meta` condemn + markers (drained by phase 12). No deletes. +- **Safety:** `suppress_destructive` is computed once here and read at every destructive site of the + round; a *pure carry* shard (no delta, no orphan retirement, no parent condemned rows) copies the + parent's run references with zero run I/O +- **Fails the round if** (`CORRUPTED_DATA`): during orphan-sweep planning, a candidate manifest whose + decoded identity disagrees with its key. Two invariant checks then run after this phase's timer + and before phase 10's: `transactions_unapplied` (a folded transaction whose deltas reached no + shard reducer) and `logs_accounted ≠ logs_applied` (the round sealed coverage over more logs than + it fully folded); either is `CORRUPTED_DATA` +- **Observability:** phase row `fold_reduce`; metrics `shards_reduced` / `shards_pure_carry`, + `condemned`, `graduated`, `spared`, `redelete_pending`, `suppress_destructive`, `frontier_complete` + +`suppress_destructive` is true on any recorded anomaly, any hold in the seal about to be made +durable, or an incomplete frontier (`frontier_proven ≠ frontier_namespaces`, a universe neither +non-empty nor proved empty, or a universe policy that is not authoritative). Under it the round +still condemns, spares and carries, but graduation, redelete, orphan-sweep planning and cursor +adoption, retention prune, hand-off reclaim, manifest deletes and ref-object cleanup do not run, +and namespace cleanup lists and classifies one page without deleting or advancing its cursor. Per +candidate the merge decides one of: `spare`, `condemn`, `supersede`, `graduate` (→ +`delete_pending`, deleted by phase 11 of a later round), `redelete` (→ deleted by phase 11 now), +`carry`. The round-wide `GcRoundWorkBudget` (one struct, fed from the `cas_gc_round_*` settings, +`0` = unbounded) caps graduations and redeletes here — their overflow is carried unchanged — and +the other work families of phases 9 (sweep planning), 11 (outcome-log entries, whose overflow is +simply not logged), 13, 14 (one-shot, see there) and 17 (recomputed next round). Phase 18's +volume is bounded by `cas_manifest_sweep_delete_budget_keys` through phase 9's planning. + +## Phase 10 — fold seal write {#phase-10-fold-seal-write} + +Validates, encodes and writes the new fold seal with one write-once `PUT`. + +- **Runs on:** fold path only +- **Reads:** one byte-compare `GET` only on a deterministic replay +- **Writes / deletes:** one write-once `PUT` of the new `fold_seal`; no `CAS`, no deletes +- **Safety:** the seal is deterministic — the same fold inputs produce byte-identical bytes. A + byte-equal occupant is this leader's own crash / replay and is adopted with no rewrite; a deposed + leader writes under its own unadopted attempt and never collides with the adopted seal. +- **Fails the round if:** a divergent-bytes occupant (impossible under correct operation) + (`CORRUPTED_DATA`) +- **Observability:** phase row `fold_seal_write`; metrics `seal_bytes`, `seal_runs`, + `seal_ref_lives`, `seal_cleanup_evidence` + +The seal's existence marks the fold complete; `snap_generation` / `snap_attempt` are advanced *in +memory* here and made durable only by phase 13's commit `CAS`. + +## Phase 11 — pending deletes {#phase-11-pending-deletes} + +The round's single blob-body delete site, before the commit `CAS`: executes the exact-token blob +deletes for entries a *previous* round published as `delete_pending`, processing up to +`cas_gc_round_redelete_budget` entries per round (the excess is carried unchanged), and writes the +forensic outcome logs. + +- **Runs on:** fold path only +- **Reads:** one `HEAD` per `redelete` entry — the persisted condemned token cannot itself be a + precondition, so the round observes the blob first, which also settles the absent case without + spending a conditional delete +- **Writes / deletes:** a `DELETE` conditional on the observed etag for each `redelete` entry whose + live etag matches the condemned token; one write-once outcome log per shard with settled entries +- **Safety:** an entry is deletable only because a previously committed fold seal published it + `delete_pending` — durable state from an earlier commit, safe at any leader staleness. Exact token + means a stale leader cannot delete a fresh incarnation; `NotFound` and `TokenMismatch` are + tolerated. This round's own commit outcome does not affect the delete's safety. +- **Fails the round if:** a backend delete marker appears in a response — object versioning on a + mis-provisioned pool (`LOGICAL_ERROR`) +- **Observability:** phase row `pending_deletes`; metrics `deleted`, `absent`, `redeleted`, + `graduated`, `replaced`, `spared`, `outcome_logs_written` + +Under `suppress_destructive` the `redelete` set is empty by construction, so nothing is deleted, and +there is no graduation either (would-be graduates are carried unchanged); sparing and superseding +still happen. An outcome log is written only for a shard that collected at least one budget-admitted +redelete or spare outcome. The `RoundReport` counters `deleted`, `absent`, `replaced` and `spared` +are tallied from the durable outcome logs, not the local decisions; `redeleted` counts the +`redelete` entries processed, whether or not a `DELETE` was sent. + +## Phase 12 — meta pool wait {#phase-12-meta-pool-wait} + +A durability barrier: drains the round's batch of async per-hash `.meta` writes (the `Condemned` +markers scheduled by phases 9 and 11, plus phase 11's meta deletes) before the commit `CAS`. + +- **Runs on:** fold path only +- **Reads / writes:** none on the GC thread; waits on the bounded `meta_pool` + (`cas_gc_meta_pool_size`, default 16) +- **Safety:** a writer's meta point-read gate must see this round's condemns durable no later than + the ledger they pair with. A throwing round's `SCOPE_EXIT` still drains the same pool, so no jobs + run into the next round. +- **Fails the round if:** a `ThreadPool` framework failure (per-hash operation exceptions are caught + inside the meta writer) +- **Observability:** phase row `meta_pool_wait`; job counts `jobs_scheduled`, + `jobs_completed_on_entry`, `jobs_completed` (the `ProfileEvents` map is empty — work runs off the + GC thread) + +## Phase 13 — round commit {#phase-13-round-commit} + +The commit boundary: a pre-`CAS` retention prune of old generations, then the single commit `CAS` +over `gc/state`. One phase because the prune is only safe as a pre-`CAS` action. + +- **Runs on:** fold path only +- **Reads / writes:** `LIST` + wholesale `DELETE` of pruned generation prefixes; exactly one `CAS` + on `gc/state` +- **Safety:** the prune skips any generation still referenced by the parent seal (phase 5) or the + new seal — this is what stops a losing leader's prune from destroying what the winner's seal still + points at. `snap_pruned_through` still advances past a skipped generation (phase 14 reclaims it + later). `suppress_destructive` skips the prune entirely. The commit `CAS` uses phase 1's token, so + a stale leader's commit is rejected. +- **Fails the round if:** the commit `CAS` is not `Committed` — a precondition conflict is + `ABORTED` ("gc/state moved during the round"), a backend or deadline failure keeps its own code; + the round publishes nothing and a transient code finishes as `Aborted`, keeping leadership (see + [round outcomes](#round-outcomes)) +- **Observability:** phase row `round_commit`; metrics `generations_visited`, `pruned_through`, + `generations_referenced`, `round`, `generation` + +Prune bound: keep the last `cas_gc_snapshot_generations_to_keep` generations (default 3; `0` keeps +everything), at most 64 prefixes a round. See [the one-pass commit](#gc-state) for the fold seal's +role as the coverage record. + +### Post-commit failures {#post-commit-failures} + +After a `Committed` result the round is committed — an exception in phases 14–18 does not un-commit +it. Those phases tolerate the object outcomes they name (`NotFound`, `TokenMismatch`), but only +phase 16 is wrapped in a catch-all; a backend or decode exception in phases 14, 15, 17 or 18 +propagates, and `system.cas_gc_log` then records an `Aborted` or `Error` finish (see +[round outcomes](#round-outcomes)) for a round whose new `gc/state` is already durable. Read such a +row as "committed round, failed tail": the next round starts from the committed state, and what +the tail did not delete is picked up later — by the orphan sweep for phase 15's leftovers, by +phase 17's recomputed plan, and for phase 18 only once the sweep cursor (already advanced by phase +13) wraps around the manifest keyspace — or, for phase 14 only, left to `cas-fsck`. + +## Phase 14 — handoff reclaim {#phase-14-handoff-reclaim} + +First phase of the post-`CAS` tail (phases 14–18 run only after a successful commit). Deletes, up to +its own object budget, a generation prefix that the retention prune already reached and skipped +(still referenced) while its cursor advanced past it, now that a ref has moved off it this round. + +- **Runs on:** post-`CAS` (fold path) +- **Reads / writes:** paginated `LIST` + one `DELETE` per listed object, per handed-off generation + prefix +- **Safety:** reclaims only when the parent seal referenced the generation, the new seal does not, + it is already behind `snap_pruned_through`, `suppress_destructive` is false, and the phase's own + budget (separate from phase 13's) is not exhausted +- **Fails the round if:** no protocol check of its own; a backend `LIST` / `HEAD` / `DELETE` + exception propagates (see [post-commit failures](#post-commit-failures)) +- **Observability:** phase row `handoff_reclaim`; metrics `generations_reclaimed`, + `objects_reclaimed`, `suppressed` + +Unlike every other gated site, suppression here *loses* the reclaim rather than postponing it: the +ref moved off this round, nothing revisits, and the prefix is left to `cas-fsck`. A crash in this +window, or a budget that runs out before the prefix is fully drained, leaks the same way +(`generations_reclaimed` counts the generation even when only part of it was deleted). + +## Phase 15 — manifest deletes {#phase-15-manifest-deletes} + +Deletes owner-removed manifest bodies, now that phase 13's `CAS` adopted their minus-one decrements. + +- **Runs on:** post-`CAS` (fold path) +- **Reads / writes:** batch `DELETE` of the manifest keys collected by phase 8's fold of `-1` owner + edges, in chunks of `cas_gc_bulk_delete_chunk_keys` (default 1000, the backend maximum). A manifest + key is write-once, so the delete carries no per-key precondition; an absent key is simply gone. A + backend without a batch-delete verb (GCS) falls back to one admitted `DELETE` per key. +- **Safety:** each body is unreachable from any live ref (its owner-removal was folded and + committed) and is never re-derived — the intake cursor that found the `-1` edge is now committed, + so a folded log is never revisited. Hence the phase is unbudgeted by design and drains the whole + set each run. +- **Fails the round if:** no protocol check of its own; a chunk that exhausts its retry policy + throws (see [post-commit failures](#post-commit-failures)). Deletion and recording are + all-or-nothing per request: the chunks before the failing one are recorded, the failing chunk's + keys are not, and a key one of its attempts did delete shows up as already gone in the next fold +- **Observability:** phase row `manifest_deletes`; metrics `attempted`, `accepted` (keys recorded + as deleted or absent), `requests` (one per chunk, or the failed bulk call plus one per key on the + fallback), `suppressed`; one `ManifestDelete` row per key in `system.cas_log` + +Only a crash, `suppress_destructive` or a chunk that exhausted its retries leaves an entry — it is +then picked up by the orphan-manifest sweep (phase 18). + +## Phase 16 — namespace cleanup {#phase-16-namespace-cleanup} + +One bounded page of the perpetual namespace janitor: deletes the physical objects of namespace lives +no longer in the catalog (dead-life debris). + +- **Runs on:** fold path here; also on the deferred path right after phase 4 with + `suppress_destructive` forced on +- **Reads:** the durable `janitor_cursor`; one `LIST` page (≤ 1000 keys) of `cas/ns/`; a fresh + ref-catalog snapshot; `gc/state` per fence re-check +- **Writes / deletes:** exact-token `DELETE` per dead-life `_log` / `_snap` / `_ckpt` / `_files` + object; one `CAS` on the maintenance state when the page is decided +- **Safety:** each delete is under a GC fence re-check (`lease.owner` / `lease.seq`) before it and + once at the end; the incarnation segment in every key makes an old life's objects structurally + unreachable from a reborn same-name namespace, so a missed key can only leak storage, never expose + it +- **Fails the round if:** nothing — the whole page is wrapped in a catch-all ("namespace janitor + skipped this round") +- **Observability:** phase row `namespace_cleanup`; metrics `janitor_pages`, `janitor_keys`, + `janitor_deleted`, `leaked` + +The cursor advances only when the whole page was decided under a held fence and an unambiguous +catalog; under suppression it lists and classifies but deletes nothing and does not advance. + +## Phase 17 — ref object cleanup {#phase-17-ref-object-cleanup} + +Deletes the `_log` / `_snap` objects of **live** namespace lives once fold coverage and a +checkpoint-named recovery triple make them safe. Distinct from phase 16, which handles lives absent +from the catalog. + +- **Runs on:** post-`CAS` (fold path) +- **Reads:** per namespace, the checkpoint-named recovery triple (same-id `_log`, predecessor seal, + `_snap`) to validate deletion authority; then, before *every chunk*, a fresh ref catalog and + `gc/state` (authority re-validation). No `HEAD`: `_log` / `_snap` keys are write-once, there is + nothing to re-observe +- **Writes / deletes:** batch `DELETE` of the planned `_log` / `_snap` keys in chunks of + `cas_gc_bulk_delete_chunk_keys` (one admitted `DELETE` per key on a backend without batch + delete); the checkpoint-named snapshot is always retained +- **Safety:** before each chunk, re-validates: ref-catalog token still equals the fold's catalog + cut, same row and life, unchanged GC fence. The first failure stops the whole pass. The + per-round `cas_gc_round_ref_cleanup_budget` cap counts objects and cuts a chunk to what remains; + on exhaustion the same candidates are recomputed next round. +- **Fails the round if:** no protocol check of its own — `suppress_destructive` returns immediately + (a clamp could leave a covered log whose delta is not yet durable); an authority re-validation + error or a namespace whose recovery triple does not validate stops the pass or skips that + namespace, but a chunk delete that exhausts its retry policy propagates (see + [post-commit failures](#post-commit-failures)) +- **Observability:** phase row `ref_object_cleanup`; metrics `namespaces_planned`, `suppressed`, + `trim_enabled`; `ProfileEvent` `CASRefCleanupObjectsDeleted` + +## Phase 18 — orphan sweep {#phase-18-orphan-sweep} + +The last phase: executes the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep) +planned in phase 9 and adopted by phase 13's `CAS`. + +- **Runs on:** post-`CAS` (fold path) +- **Reads / writes:** one `HEAD` per nomination, then a `DELETE` conditional on the observed etag + when it matches the nominated token (planning `LIST` / `GET` cost was paid in phase 9); an absent + body sends no `DELETE` +- **Safety:** phase 9 exact-read and identity-validated each candidate and computed its source-edge + retirements; phase 13's `CAS` adopted both those retirements and the sweep cursor, so a post-`CAS` + body delete cannot orphan a still-reachable edge. A manifest is deletable only once its epoch's + closing seal is consumed and no tail record above the cursor names it; any uncertainty retains. +- **Fails the round if:** a `TokenMismatch` — an immutable manifest identity must never change token + (illegal ABA); stricter than every other post-`CAS` delete (`CORRUPTED_DATA`) +- **Observability:** phase row `orphan_sweep`; metrics `listed`, `floor_lookups` / `floor_reads` + (mount-floor lookups per namespace and the reads they cost), `deleted`, `skipped`, + `undecodable`, `cursor_advanced`, `suppressed`, and the retained share of `skipped` by reason: + `retained_no_coverage`, `retained_hold`, `retained_unconsumed_seal`, `retained_tail_removal` + (candidates retained because the sweep's work budget ran out are reported only in the sweep's + retention log line, not in `phase_metrics`) + +Under `suppress_destructive` phase 9 planned nothing, so the nomination list is empty and the cursor +does not move. + +## GC anomalies {#gc-anomalies} + +A GC round records *anomalies* and per-namespace *holds* instead of failing, unless a fail-closed +check fires. Any anomaly or hold in the seal about to be made durable forces `suppress_destructive` +for the whole round (phase 9); condemnation and sparing still run. A hold clears only when a later +walk folds through the offending position. Each hold is recorded in `system.cas_log` as a +`GcFoldClamp` event with its reason; the round's aggregate anomaly count rides the `GcFoldEnd` event +and the `Finish` row of `system.cas_gc_log`. There are no per-anomaly rows. + +**Durable per-namespace holds** — persisted in the fold seal under these wire names; the matching +`GcFoldClamp` event in `system.cas_log` carries a human-readable reason. Each holds one namespace +(all phase 8): + +| Hold | Meaning | Effect | +|---|---|---| +| `gap_below_witness` | a committed record at or below the ceiling is missing | held | +| `unconsumed_seal_crossing` | an apparent epoch crossing has no consumed `EpochSeal` behind it | held | +| `witness_disappeared` | an epoch-crossing chase resolves back to the absent position | held | +| `body_undecodable` | a ref-log record exists at the walk position but its body cannot be decoded | held at that position | +| `manifest_body_missing` | a folded owner edge's manifest body is absent | held below that record; re-read next round | +| `checkpoint_undecodable` | a live/removing life's `_ckpt` is undecodable, absent, or lacks `life_epoch` | folds nothing; held at `cursor + 1` when the life has a sealed cursor | + +**Per-round suppression signals** — not persisted; they force `suppress_destructive` for the +round. `ref_folding_aborted` and `frontier_unprobed_budget` are `phase_metrics`; the three +checkpoint states are counted only in the suppression log line's frontier-deficit breakdown: + +| Signal | Phase | Meaning | +|---|---|---| +| `ref_folding_aborted` | 6 | a ref-object key under the stream prefix is unparseable: no ref delta, no cursor advance | +| `CheckpointUnusable` | 8 | in-memory frontier state for a `_ckpt` that could not be used, recorded even when there is no cursor position to hold at | +| `CheckpointFrontierEmpty` | 8 | a checkpoint carries no `committed_through` but the namespace has a nonzero sealed cursor: namespace unproven | +| `CommittedBelowCursor` | 8 | the sealed cursor is already above the committed ceiling: namespace unproven | +| `frontier_unprobed_budget` | 8 | `gc_frontier_probe_budget` ran out before every hint-less namespace was walked | + +**Fatal pre-seal checks** — `CORRUPTED_DATA`, evaluated after phase 9 and before phase 10: +`transactions_unapplied` (a folded transaction's deltas reached no shard reducer) and +`logs_accounted ≠ logs_applied` (coverage sealed over more logs than were fully folded). ## The one-pass commit {#gc-state} -`gc/state` is the durable safety and round-adoption state: `round`, `gc_shards`, -`snap_generation`, `snap_pruned_through`, `snap_attempt`, `manifest_sweep_cursor`, and the lease. -Exactly one `CAS` per round publishes it; the fold itself performs no `CAS` of its own. +`/gc/state` is the durable safety and round-adoption state: `round`, `gc_shards`, +`snap_generation`, `snap_pruned_through`, `snap_attempt`, `manifest_sweep_cursor`, and the lease. A +folding round publishes it with exactly one commit `CAS` in phase 13, `round_commit`; the fold +itself performs no `CAS` of its own, and within a round execution phase 1's lease `CAS` over the +same object is the only other writer. Outside the round, `SYSTEM CAS GC REBUILD` replaces the +baseline with a `CAS` of its own. **The fold seal *is* the coverage record**: generation, parent generation, one `ref_lives` row per catalog-admitted opaque life (coverage plus optional cleanup evidence), references to the @@ -140,23 +656,27 @@ size that a future delete will name. A blob merely carried from the parent run p full round (`condemn_round < current_round`). The heartbeat floor is liveness only and **never** gates graduation. -**The 404 rule.** A body that is present but invalid is `CORRUPTED_DATA`, hard. A body that is -missing is **never** a throw — the fold records and continues, and the caller decides by position: -a precommit activation clamps as a barrier; a committed or removal fold clamps only that table. -Prunes are likewise fail-open on 404. +**The 404 rule for manifest edges.** When the fold reads a manifest for an owner edge, a body that is +present but invalid (bad encoding, or a ref / namespace that disagrees with its key) is +`CORRUPTED_DATA`, hard. A body that is missing is **never** a throw there — the fold records and +continues, and the caller decides by position: a precommit activation clamps as a barrier; a +committed or removal fold clamps only that table. Prunes and post-`CAS` deletes are likewise +fail-open on 404. Other objects have their own policy: an undecodable ref-log body or checkpoint +holds one namespace (`body_undecodable`, `checkpoint_undecodable`), an undecodable orphan-manifest +candidate is retained and counted, and a missing adopted fold seal fails the round (phases 2 and 7). ## Condemnation and deletion {#condemn-delete} ```mermaid flowchart LR A["round n: in-degree hits zero
HEAD -- exact token t"] --> B["write .meta = Condemned round n
async, bounded pool, drained pre-CAS"] - B --> C["retired with condemn_round = n+1"] + B --> C["retired with condemn_round = n"] C --> D{"round n+1: re-verify"} D -->|"in-degree recovered"| S["SPARED -- recovery wins, even past the floor"] D -->|"still zero, confirmed durable Condemned evidence for hash and t"| G["GRADUATED -- delete_pending"] D -->|"still zero, evidence unconfirmed"| C2["carried unchanged, retry the marker, never throw"] D -->|"current token not equal to t"| SUP["SUPERSEDED -- a writer resurrected, re-condemn the CURRENT token"] - G --> E["round n+2, pre-CAS: deleteExact blob, t"] + G --> E["round n+2, pre-CAS: exact-token DELETE of blob at t"] E -->|"Deleted or Absent"| F["then drop the .meta"] E -->|TokenMismatch| H["nothing deleted -- live at a newer token, leave the .meta alone"] ``` @@ -165,30 +685,41 @@ The `.meta` sidecar carries **no token** — it is a per-hash hint. The exact in in the condemned sentinel row inside the run, together with the condemn round and two flags, `delete_pending` and `marker_confirmed`. `GC`'s marker is add-only: `Clean → Condemned` yes, the reverse never, not even when sparing — only a writer that has already displaced the body may clear -it. Minimum two full rounds separate condemnation from deletion, and `delete_pending` is terminal — -an entry is never un-pended. +it. A blob whose in-degree reaches zero in round `n` is retired with `condemn_round = n`; it can +graduate to `delete_pending` in round `n+1` at the earliest and be deleted in round `n+2`, so a +minimum of two full rounds separate condemnation from deletion. `delete_pending` is never cleared in +place, but it authorizes a delete only while in-degree stays zero: a fresh edge folded in a later +round spares the entry and removes it from the retired pipeline (recovery wins, even past the +floor). ## Sharding {#sharding} -`cas_gc_shards` is fixed at first lease acquire and immutable; decoders reject `0`. A blob routes by -the **high** 64 bits of its digest, read big-endian. +`cas_gc_shards` is fixed at pool creation and stored in `_pool_meta`; the first lease acquire copies +that authoritative value into `gc/state`, every later lease read throws `CORRUPTED_DATA` if the two +disagree, and decoders reject `0`. A blob routes by the **high** 64 bits of its digest, read +big-endian. The role split is worth internalizing: the **coordinator** — the lease holder — owns discovery, round visibility, the single global fence, and the generation advance, because a publish into *one* namespace can protect a blob owned by *any* shard, so these span the whole universe and must not be sharded. **Reducers** own only their disjoint shard; their run-key namespaces never -collide, so two servers could reduce different shards concurrently and reducer work needs no -lease. +collide, so the design admits reducing different shards on different servers without a lease. The +current implementation does not do that: all shard reducers run sequentially on the lease holder's +fold thread, and the transaction-apply ledger relies on it. -A shard with an empty delta bucket and no condemned entries in the parent summary copies the -parent's run references verbatim — zero run I/O, a "pure carry". A missing parent summary entry on -a non-fresh pool is `CORRUPTED_DATA`, never silently treated as zero. +A shard with an empty delta bucket, no orphan-sweep retirement routed to it, and no condemned +entries in the parent summary copies the parent's run references verbatim — zero run I/O, a "pure +carry" (see [phase 9](#phase-9-fold-reduce)). A missing parent summary entry on a non-fresh pool is +`CORRUPTED_DATA`, never silently treated as zero. ## Pruning old objects {#pruning} -- **Current-life ref logs and snapshots** (phase 17) — a log is deletable only when covered by - both durable fold coverage and a durable live snapshot; snapshots strictly older than the newest - observed one are deletable. There is no batch delete; it is `HEAD` plus `deleteExact` per key. +- **Current-life ref logs and snapshots** (phase 17) — authority comes from the namespace's + checkpoint-named, exact-validated recovery triple. A log is deletable only when covered by the + durable fold cursor and older than that checkpoint (and not the retained predecessor seal); + snapshots strictly older than the checkpoint-named snapshot are deletable, that snapshot itself is + always kept. Keys are write-once, so there is no `HEAD`: the plan is chunked, each chunk is + preceded by a catalog and `gc/state` re-validation and sent as one batch `DELETE`. - **Generations** (phase 13) — keep the last `cas_gc_snapshot_generations_to_keep` (default 3; `0` means keep everything, for forensics). Pruning is wholesale: `LIST` the generation prefix and delete everything under it, including deposed-leader debris and attempt-scoped outcome sets. A @@ -206,59 +737,273 @@ logs: |---|---| | `LIST cas/ns/stream/` | 1 full enumeration | | `LIST gc/server-roots/` | 1, plus 1 `GET` per mount | -| `GET` the adopted fold seal | 5, explicitly instrumented | -| `GET` ref logs | 1 per new log | -| `GET` manifests | 1 per emitted edge — no manifest-body cache within a round | +| `GET` the adopted fold seal | 6 on the fold path of an established pool (phases 2, 4, 5, 7); phase 9 orphan planning adds one more. See [per-phase backend cost](#per-phase-cost) | +| `GET` ref logs | 1 per new log record, plus `_ckpt` reads and epoch-crossing probes | +| `GET` manifests | 1 per folded owner (manifest) edge — a manifest emits many blob edges but is read once per edge event; no manifest-body cache within a round | | `PUT` run segments | 1 per non-pure-carry shard, plus 1 fold seal | | `HEAD` blobs | 1 per newly condemned | -| `DELETE` | 1 per graduate | -| `CAS gc/state` | 1 | +| Blob `HEAD` + conditional `DELETE` | 1 `HEAD` per `redelete` entry — an entry that graduated in an *earlier* round, not the current one — up to `cas_gc_round_redelete_budget`; a `DELETE` only when the body is present at the condemned token | +| Successful lease `CAS gc/state` | 1 | +| Commit `CAS gc/state` | 1 | + +Phase 8's body reads are one ref-log `GET` per consumed record plus one manifest `GET` per owner +edge; that is the dominant variable term, not the whole-round `GET` total, which also includes the +state, seal, catalog, checkpoint, mount, parent-run and cleanup reads listed per phase below. An idle +folding round is one `LIST` of `cas/ns/stream/`, the heartbeat floor (`LIST` plus `N` `GET`s), the +seal, catalog and `gc/state` reads of phases 2, 4, 5 and 7, one successful lease `CAS`, and one +commit `CAS`. A deferred round execution is cheaper: the same `LIST`, the heartbeat floor, phase 2's +seal / catalog / `gc/state` reads, phase 4's two seal reads and catalog read, the lease `GET`/`CAS`, +and one suppressed namespace-janitor page (its own `LIST` page and reads, no deletes) — no commit +`CAS` at all. + +The round's work is self-regulated: what a pass cannot finish within its budgets is carried and +retried by the next round's cursors — with the one exception of phase 14's hand-off reclaim, which +is one-shot and leaves its remainder to `cas-fsck`. The per-round budgets are ordinary +`content_addressed` disk settings, documented under +[advanced GC pacing settings](/antalya/cas/configuration#advanced-gc-pacing-settings) on the +configuration page (`cas_gc_meta_pool_size` and `cas_gc_read_concurrency` sit in its main +[disk-settings table](/antalya/cas/configuration#disk-settings)). `0` means unbounded for every +`cas_gc_round_*` budget; `cas_manifest_sweep_list_budget_keys = 0` disables the sweep, +`cas_manifest_sweep_delete_budget_keys = 0` lists without nominating, and the two pool sizes and +the chunk size reject `0`: -The measured `GET` formula is exact: total `GET`s equal ref-log body `GET`s plus manifest body -`GET`s, i.e. `1 + edges_per_log`. An idle round is one `LIST` sweep, `N` heartbeat `GET`s, and one -`CAS`. A deferred round is cheaper still: one `LIST`, three seal `GET`s, the lease `GET`/`PUT` and -the heartbeat floor — no `gc/state` `CAS` at all. +| Setting | Default | Bounds | +|---|---:|---| +| `cas_gc_round_graduation_budget` | 5000 | condemned → `delete_pending` graduations per round (phase 9) | +| `cas_gc_round_redelete_budget` | 5000 | `redelete` entries processed per round (phase 11) | +| `cas_gc_round_outcome_entry_budget` | 5000 | outcome-log entries per round (phase 11) | +| `cas_gc_round_prefix_wholesale_budget` | 20000 | listed objects the retention prune may process per round, gone ones included (phase 13) | +| `cas_gc_round_handoff_prefix_wholesale_budget` | 5000 | listed objects the hand-off reclaim may process per round, reserved separately (phase 14) | +| `cas_gc_round_ref_cleanup_budget` | 5000 | covered `_log` / `_snap` deletes per round (phase 17) | +| `cas_manifest_sweep_list_budget_keys` | 1000 | orphan-manifest sweep `LIST` budget in keys per round; `0` disables the sweep (phase 9) | +| `cas_manifest_sweep_delete_budget_keys` | 100 | orphan-manifest sweep `DELETE` budget per round (phases 9, 18) | +| `cas_gc_round_sweep_namespace_budget` | 20 | namespaces whose protection view the sweep may build per page (phase 9) | +| `cas_gc_round_sweep_recovery_op_budget` | 5000 | committed-tail ref-log reads the sweep's recovery walk may spend (phase 9) | +| `cas_gc_bulk_delete_chunk_keys` | 1000 | keys per batch `DELETE` request for write-once families (phases 15, 17); `1` to `1000` | +| `cas_gc_meta_pool_size` | 16 | bounded pool for condemn-marker writes (phase 12) | +| `cas_gc_read_concurrency` | 16 | bounded pool for the fold's read-ahead of checkpoints, ref logs, manifest bodies and zero-candidate `HEAD`s (phases 8, 9); `1` disables | -The round's work is internally self-regulated: anything a pass cannot finish is carried and retried -by the next round's cursors, never dropped. The internal pacing knobs are deliberately not part of -the user-facing configuration surface. +The fold-batching controls `gc_fold_threshold` (default 1), `gc_fold_max_defer_rounds` (default 8) +and `gc_frontier_probe_budget` (default unbounded) are internal `PoolConfig` fields with no disk +setting. -| Setting | Default | Bounds | -|---|---|---| -| `cas_gc_meta_pool_size` | 16 | bounded pool for condemn-marker writes | -| `cas_gc_read_concurrency` | 16 | bounded pool for the fold's read-ahead; `1` disables | +## Per-phase backend cost {#per-phase-cost} + +Backend requests each phase issues, by key and operation. These tables describe the current +implementation and expand [what a round costs](#round-cost). Read every count as a conflict-free +lower bound: token conflicts add re-reads and retries, backends whose `LIST` returns no token add +one `HEAD` per key before each exact delete (phases 13, 14, 16), and recovery paths add fan-out. +`N` is the number of items the phase acts on without conflicts; `P` is the number of paginated +`LIST` requests (up to 1000 keys each). + +### Phase 1 — lease {#cost-phase-1} + +| Result | `gc/state` `GET` | `gc/hb` `GET` | `gc/state` `CAS` | +|---|---:|---:|---:| +| `Acquire` | 1 | 0 | 1 | +| `Renew` | 1 | 0 | 1 | +| `Follower` | 1 | 1 | 0 | +| `Steal` | 1 | 1 | 1 | + +A heartbeat pulse runs outside this phase: one `gc/hb` `GET` and one `CAS`. + +### Phase 2 — pre-fold ref drain {#cost-phase-2} + +No requests when `snap_generation` is `0`. Otherwise, for `N` removed catalog rows: + +| Key | Operation | Requests | +|---|---|---:| +| adopted `fold_seal` | `GET` | 1 | +| `/cas/ref_catalog` | `GET` | `N + 1` | +| `/gc/state` | `GET` | `2N + 1` (two re-reads bracket every write) | +| `/cas/ref_catalog` | `CAS` | `N` | + +### Phase 3 — heartbeat floor {#cost-phase-3} + +`F` successful fence-outs over `M` mounts found by `P` `LIST` requests: + +| Key | Operation | Requests | +|---|---|---:| +| `/gc/server-roots/` | paginated `LIST` | `P` | +| `/mount` | `GET` | `M` | +| `/mount` | token-guarded `PUT` | `F` (a conflicting mount is re-read and re-classified within the standard write policy) | + +### Phase 4 — defer decision {#cost-phase-4} + +| Key | Operation | Requests | +|---|---|---:| +| `/cas/ns/stream/` | paginated `LIST` | `P` | +| `/cas/ref_catalog` | `GET` | 1 | +| adopted `fold_seal` | `GET` | 2 with an adopted generation, otherwise 1 | + +No writes. + +### Phase 5 — parent seal read {#cost-phase-5} + +| Key | Operation | Requests | +|---|---|---:| +| adopted `fold_seal` | `GET` | 1 | + +No writes. The `blob_target_runs[].key` run objects are not read here. + +### Phase 6 — fold ref group {#cost-phase-6} + +No requests. The keys are already in memory from phase 4. + +### Phase 7 — fold seal read {#cost-phase-7} + +| Key | Operation | Requests | +|---|---|---:| +| adopted `fold_seal` | `GET` | 2 | + +No writes. The second read is the redundant one noted in [phase 7](#phase-7-fold-seal-read). On the +fold path of an established pool, phases 2, 4, 5 and 7 read the adopted seal 6 times in total; when +phase 9 runs orphan planning it reads the same key once more. + +### Phase 8 — fold ref intake {#cost-phase-8} + +| Key | Operation | Requests | +|---|---|---:| +| `/_ckpt` | `GET` | one per namespace life in the universe | +| `_log` record up to `committed_through` | `GET` | one per record read; none when the cursor already equals the ceiling | +| `_log` record at an epoch start | `GET` | at least two per crossing, plus one per epoch stepped back and one on a failed crossing | +| manifest body | `GET` | one per folded owner edge | + +No writes. + +### Phase 9 — fold reduce {#cost-phase-9} + +| Key | Operation | Requests | +|---|---|---:| +| referenced parent run segments | streaming `GET` | one per referenced run | +| `/blobs/...` | `HEAD` | one per zero-in-degree candidate, plus one peek per carried entry that reached zero again | +| blob `.meta` | `GET` | one per graduation candidate with no in-process marker confirmation | +| new run segments | `PUT` | one per written run | +| `/cas/manifests/` | `LIST` | one bounded page, only when orphan planning runs | +| manifest candidate body | `GET` | one per nominated candidate (≤ `cas_manifest_sweep_delete_budget_keys`), through the read-ahead; keys decided from their name alone are never read; only when orphan planning runs | +| `gc/state`, adopted `fold_seal`, catalog | `GET` | one each, only when orphan planning runs | +| `/mount` | `GET` | one memoized mount-floor lookup per namespace per page (`floor_lookups` / `floor_reads`), only when orphan planning runs | +| `_ckpt`, checkpoint-named `_log`, predecessor seal, `_snap`, committed-tail `_log` | `GET` | per namespace on the page (the recovery triple plus the tail), only when orphan planning runs | + +Also schedules the async `.meta` condemn-marker writes drained by phase 12. + +### Phase 10 — fold seal write {#cost-phase-10} + +| Key | Operation | Requests | +|---|---|---:| +| new `fold_seal` | `PUT` | 1 conditional `PUT`; on a deterministic replay the `PUT` fails its precondition and one byte-compare `GET` follows | + +No `CAS`. + +### Phase 11 — pending deletes {#cost-phase-11} + +| Key | Operation | Requests | +|---|---|---:| +| blob body | `HEAD` | one per `redelete` entry (≤ `cas_gc_round_redelete_budget`) | +| blob body | conditional `DELETE` | one per `redelete` entry that is present at the condemned token | +| per-shard outcome log | `PUT` | one per shard with at least one budget-admitted redelete or spare outcome; a replay adds one byte-compare `GET` | + +Under `suppress_destructive`, `redelete` is empty and nothing is deleted. + +### Phase 12 — meta pool wait {#cost-phase-12} + +No backend request on the GC thread. Waits on the bounded `meta_pool` (`cas_gc_meta_pool_size`, +default 16). + +### Phase 13 — round commit {#cost-phase-13} + +| Key | Operation | Requests | +|---|---|---:| +| pruned generation prefixes | paginated `LIST` + one `DELETE` per listed object | ≤ 64 prefixes and ≤ `cas_gc_round_prefix_wholesale_budget` objects per round | +| `/gc/state` | `CAS` | exactly 1 | + +### Phase 14 — handoff reclaim {#cost-phase-14} + +Paginated `LIST` plus one `DELETE` per listed object for each handed-off generation prefix, within +the hand-off's own budget (`cas_gc_round_handoff_prefix_wholesale_budget`). + +### Phase 15 — manifest deletes {#cost-phase-15} + +One batch `DELETE` request per `cas_gc_bulk_delete_chunk_keys` entries of `mf_cleanup` (on a +backend without batch delete: the refused bulk call plus one `DELETE` per key). No writes under +`suppress_destructive`. + +### Phase 16 — namespace cleanup {#cost-phase-16} + +| Key | Operation | Requests | +|---|---|---:| +| `/gc/maintenance_state` | `GET` | 1 (durable `janitor_cursor`) | +| `/cas/ns/` | `LIST` | one page | +| `/cas/ref_catalog` | `GET` | 1 | +| `/gc/state` | `GET` | one per fence check | +| dead-life object | `DELETE` | one per object (plus one `HEAD` per object whose `LIST` entry carried no token) | +| `/gc/maintenance_state` | `CAS` | 1 when the page is decided | + +### Phase 17 — ref object cleanup {#cost-phase-17} + +| Key | Operation | Requests | +|---|---|---:| +| checkpoint-named `_log`, predecessor seal, `_snap` | `GET` | per planned namespace (recovery-triple validation before any delete) | +| `/cas/ref_catalog` and `/gc/state` | `GET` | one each per chunk (authority re-validation) | +| `_log` / `_snap` keys | batch `DELETE` | one request per chunk of ≤ `cas_gc_bulk_delete_chunk_keys` keys (the refused bulk call plus one per key on a backend without batch delete) | + +### Phase 18 — orphan sweep {#cost-phase-18} + +One `HEAD` per nomination and a conditional `DELETE` for each nomination present at its token. The +planning `LIST` and `GET` cost is paid in phase 9. ## Observability {#observability} +### Round outcomes {#round-outcomes} + +A `Finish` row of `system.cas_gc_log` carries one of: `Success` (folded and committed), `Deferred` +(phase 4 chose not to fold), `NotALeader` (returned after phase 1), `Aborted` (the round threw a +*transient* error — `S3_ERROR`, `NETWORK_ERROR`, `TIMEOUT_EXCEEDED`, `SOCKET_TIMEOUT`, `ABORTED`, +`MEMORY_LIMIT_EXCEEDED` — and this `Gc` keeps its leadership and heartbeat, so the next round simply +retries), `Stopped` (a transient error while the disk was being torn down), or `Error` (any other +error code, notably `CORRUPTED_DATA` and `LOGICAL_ERROR`; leadership is dropped). The `error_code` +column carries the code on `Aborted`, `Stopped` and `Error` and is `0` otherwise; every +unrecognised code is `Error` by omission, never silently transient. + `system.cas_gc_log` emits `Start`, `Finish` and per-`Phase` rows, correlated by `round_id` — not -`round`, which is `0` on `Start` and does not exist at all on a not-a-leader round. Phase rows +`round`, which is `0` on `Start` and stays `0` on a `NotALeader` finish. Phase rows carry no verb columns by design: per-phase operation counts ride the row's own `ProfileEvents` delta, so grouping by phase over an S3 event attributes the LIST/GET/PUT/DELETE budget without -inventing schema. `phase_metrics` carries the semantic counts no counter can supply (clamped -tables, dead precommits skipped, pure-carry shards, generations visited). `Deferred` is kept -distinct from `Success` precisely so "folded and found nothing" is distinguishable from "never -folded". Every `GC`-related `ProfileEvent` carries the uppercase `CAS`/`CASGC` prefix — for example -`CASGCRetiredCondemned`, `CASGCRetiredGraduated`, `CASGCRetiredRedeleted`, -`CASGCClampSuppressedPasses`, `CASGCHeartbeatFenceOuts`. - -Alongside it, `system.cas_log` carries the audit trail: the condemn chain, fence-outs, anomalies -(capped per round, each carrying the true total), and manifest deletes. - -`ca-fsck` distinguishes two classes that are easy to conflate: `dangling` — referenced but missing, -data loss — versus `unreachable`/`awaiting-gc` — present, unreferenced, and -simply waiting for graduation. +inventing schema (requests performed off the round thread — the `meta_pool` writes and the fold's +read-ahead — land on the worker's counters instead, see [phase 8](#phase-8-fold-ref-intake)). +`phase_metrics` carries the semantic counts no counter can supply (clamped tables, dead precommits +skipped, pure-carry shards, generations visited). `Deferred` is kept distinct from `Success` +precisely so "folded and found nothing" is distinguishable from "never folded", and `Aborted` +from `Error` so a flaky backend is distinguishable from a broken pool. Every `GC`-related +`ProfileEvent` carries the uppercase `CAS`/`CASGC` prefix — for example `CASGCRetiredCondemned`, +`CASGCRetiredGraduated`, `CASGCRetiredRedeleted`, `CASGCClampSuppressedPasses`, +`CASGCHeartbeatFenceOuts`. + +Alongside it, `system.cas_log` carries the audit trail: the condemn chain (`IndegZero`, +`GcRetireObserve`, `BlobRetire`), fence-outs (`GcFenceOut`), per-namespace holds (`GcFoldClamp`), the +fold summary (`GcFoldEnd`, with the aggregate anomaly count) and manifest deletes (`ManifestDelete`). + +`cas-fsck` separates `dangling` — referenced but missing, i.e. data loss — from the +present-but-unreferenced family. The latter is reported as one `unreachable` total, broken down into +`pending_gc` (already in the retired pipeline, deletion scheduled), `awaiting_gc` (the drop is not +folded yet, or `GC` never ran), `unaccounted` (absent from the whole `GC` view) and pre-precommit +manifest debris. `pending_gc` and `awaiting_gc` are ordinary backlog; `unaccounted` that persists +across rounds is an anomaly; and two further classes are hard findings rather than backlog: +`stale_edge` (every remaining source edge on the blob names a missing manifest, so incremental +`GC` can never reclaim it and a rebuild is needed) and `corrupted_runs` (a source-edge run whose +checksum disagrees with its seal). ## Operational surface {#operational-surface} | Command | Effect | |---|---| -| `SYSTEM CAS GC RUN ''` | One synchronous round on the contacted node; only the lease holder makes progress | -| `SYSTEM CAS GC STOP` / `SYSTEM CAS GC START` | Stop or resume future rounds on the same scheduler, preserving its identity | -| `SYSTEM CAS GC REBUILD` (`clickhouse-disks ca-gc-rebuild`) | Fail-closed disaster-recovery path that every "GC refuses to run" error points at; deliberately over-protects — it prefers bounded leaks over risking an under-count. It cannot delete live data directly: deletions it produces still flow through the normal round's condemn, graduate, exact-token path | -| `clickhouse-disks ca-gc-dryrun` | Opens the disk read-only, constructs a non-leader `GC`, and prints what would be deleted with a reason per entry. Write-free, resolves runs through the seal's references. Documented caveat: it does not fold new owner events, so away from quiescence it can **over-report** — the subset guarantee holds only at quiescence, and its output must never feed a real delete | +| `SYSTEM CAS GC RUN []` | One synchronous round execution on the contacted node; only the lease holder makes progress. The disk is optional: without it every content-addressed disk on the node runs one round. It runs even while the scheduler is stopped | +| `SYSTEM CAS GC STOP ` / `SYSTEM CAS GC START ` | Stop or resume future background rounds on that disk's scheduler, preserving its identity. The disk is required | +| `SYSTEM CAS GC REBUILD [FORCE] ` (`clickhouse-disks cas-gc-rebuild`) | Fail-closed disaster-recovery path for a lost or corrupt `GC` baseline — the `CORRUPTED_DATA` errors that name it in their message (missing adopted seal, snapshot without a surviving log, cursor/apply mismatch). An `ABORTED` commit conflict or a `LOGICAL_ERROR` delete marker is not a reason to rebuild. It deliberately over-protects — it prefers bounded leaks over risking an under-count — and cannot delete live data directly: deletions it produces still flow through the normal round's condemn, graduate, exact-token path. The disk is required | +| `clickhouse-disks cas-gc-dryrun` | Opens the disk read-only, constructs a non-leader `GC`, and prints what would be deleted with a reason per entry. Write-free, resolves runs through the seal's references. Documented caveat: it does not fold new owner events, so away from quiescence it can **over-report** — the subset guarantee holds only at quiescence, and its output must never feed a real delete | `SYSTEM CAS DROP POOL MEMBER '' FROM DISK ''` — permanent removal of a dead replica, distinct from ordinary `GC` — is covered on the -[mounts-and-leases page](/antalya/cas/architecture/mounts-and-leases#mount-lifecycle). `SYSTEM CAS -FSCK` and its `dangling`/`unreachable` vocabulary are a read-only diagnostic pass, not part of the -`GC` protocol itself. +[mounts-and-leases page](/antalya/cas/architecture/mounts-and-leases#mount-lifecycle). +`SYSTEM CAS FSCK ` (`clickhouse-disks cas-fsck`) and its `dangling`/`unreachable` vocabulary +are a read-only diagnostic pass, not part of the `GC` protocol itself. diff --git a/docs/en/antalya/cas/architecture/manifests-and-refs.md b/docs/en/antalya/cas/architecture/manifests-and-refs.md index ba476cffe88a..6e6b1ed667bf 100644 --- a/docs/en/antalya/cas/architecture/manifests-and-refs.md +++ b/docs/en/antalya/cas/architecture/manifests-and-refs.md @@ -122,7 +122,7 @@ namespace whose view fails to build is added to an errored set with **all** of i skipped — an empty owner set is never substituted for a failed one. A body that cannot be opened or decoded is likewise retained: it increments both `skipped` and `undecodable`, advances the page decision cursor, logs the exact key, and does not prevent later candidates from being examined. It -is not repaired or deleted, and remains visible to `ca-fsck` as an unreachable object. A decoded +is not repaired or deleted, and remains visible to `cas-fsck` as an unreachable object. A decoded body whose ref or namespace does not match its key instead fails the round with `CORRUPTED_DATA`. For every legal nomination, the sweep derives exact source-retirement records for the body's blob @@ -273,7 +273,7 @@ after the durable install. In-flight precommits are visible only through the pre through an ordinary ref resolve. Two cross-process readers see a different, colder view, but only at the discovery boundary: `GC` -and `ca-fsck` `LIST` once to discover which namespaces exist, staleness-bounded by whatever was +and `cas-fsck` `LIST` once to discover which namespaces exist, staleness-bounded by whatever was durable at `LIST` time, so a namespace born after that `LIST` is invisible to this pass. Within each discovered namespace, the replay itself is not `LIST`-driven — it is the same exact-`GET`, `_ckpt`-grounded arithmetic walk described above, just called from a caller-supplied catalog entry diff --git a/docs/en/antalya/cas/architecture/read-path.md b/docs/en/antalya/cas/architecture/read-path.md index 49a6864a0109..c85524ad4ea3 100644 --- a/docs/en/antalya/cas/architecture/read-path.md +++ b/docs/en/antalya/cas/architecture/read-path.md @@ -78,7 +78,7 @@ temporary part is never mistaken for a real, resolvable part. ## Diagnostic and read-only access {#read-only-access} -A read-only or diagnostic opener of a `CAS` disk (`ca-fsck`, `ca-gc-dryrun`, and similar tools) +A read-only or diagnostic opener of a `CAS` disk (`cas-fsck`, `cas-gc-dryrun`, and similar tools) must not claim mount ownership, schedule `GC`, or mint writer state — read-only enforcement sits below the ordinary facade checks, at the backend layer itself. A mounted `Pool` caches its ref table and does not re-recover it on every read; a diagnostic tool that deliberately performs a diff --git a/docs/en/antalya/cas/architecture/replication.md b/docs/en/antalya/cas/architecture/replication.md index a9f3221badf7..2facd401e4c7 100644 --- a/docs/en/antalya/cas/architecture/replication.md +++ b/docs/en/antalya/cas/architecture/replication.md @@ -120,7 +120,7 @@ log. That ordering, steps T1 then T2 then T3, is the whole seal. This does **not** establish that every subsequent `GC` fold *sees* that `+1` under every listing behavior: a configuration with one incomplete listing page can, in principle, let a fold miss a freshly published edge. A confirmed relink therefore proves only "the source still holds exactly -this manifest right now", not "no future fold can ever miss this edge" — `ca-fsck`'s +this manifest right now", not "no future fold can ever miss this edge" — `cas-fsck`'s reachable-but-absent scan is the backstop for that gap, not the relink protocol itself. Relink also races `GC` in the ordinary sense any writer does: between the sender encoding its offer and the receiver's promote, `GC` on the shared pool may condemn a blob that was live only through the diff --git a/docs/en/antalya/cas/roadmap.md b/docs/en/antalya/cas/roadmap.md index d9675cf9e3e5..4df03af9ab3c 100644 --- a/docs/en/antalya/cas/roadmap.md +++ b/docs/en/antalya/cas/roadmap.md @@ -48,7 +48,7 @@ capability probe runs at every writable mount and refuses a backend that does no conditions CAS depends on. **Operability.** `system.cas_log`, `system.cas_gc_log`, and `system.cas_mounts` for introspection; -`clickhouse-disks` commands `ca-fsck`, `ca-inspect`, `ca-gc-dryrun`, and `ca-gc-rebuild`; the +`clickhouse-disks` commands `cas-fsck`, `cas-inspect`, `cas-gc-dryrun`, and `cas-gc-rebuild`; the `SYSTEM CAS` SQL control surface (`GC RUN`/`STOP`/`START`/`REBUILD`, `FSCK`, `FORGET`, `DROP POOL MEMBER`). From 7e00f47b634c932d8ac92f84c4e79bdabf13ffdc Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Thu, 17 Sep 2026 21:44:05 +0200 Subject: [PATCH 08/15] Merge pull request #2326 from Altinity/cas/gc-scheduler-stop-join-race CAS: gc scheduler stop join race Source-PR: #2326 (https://github.com/Altinity/ClickHouse/pull/2326) --- src/Common/FailPoint.cpp | 4 +- .../ContentAddressedMetadataStorage.cpp | 3 +- .../ContentAddressed/Gc/CasGcScheduler.cpp | 77 ++++++++++++++---- .../ContentAddressed/Gc/CasGcScheduler.h | 21 +++-- src/Disks/tests/gtest_cas_gc_stop_start.cpp | 80 +++++++++++++++++++ 5 files changed, 162 insertions(+), 23 deletions(-) diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 569b13e59d46..e5ee61c13ac1 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -373,7 +373,9 @@ static struct InitFiu REGULAR(cas_relink_receiver_force_mechanism_failure) \ PAUSEABLE_ONCE(cas_relink_receiver_pause_before_confirm) \ REGULAR(cas_relink_sender_omit_pool_cookie) \ - REGULAR(cas_relink_receiver_drop_forced_disk) + REGULAR(cas_relink_receiver_drop_forced_disk) \ + ONCE(cas_gc_scheduler_fail_before_heartbeat_worker_start) \ + ONCE(cas_gc_scheduler_fail_before_worker_start) namespace FailPoints { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp index 78666e0ad297..8fb5a31b467c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp @@ -1163,7 +1163,8 @@ void ContentAddressedMetadataStorage::gcStart() /// `start()` is a no-op if already running (idempotent) and re-enters the SAME instance after a stop -- /// the persistent `gc` observer + `gc_id` are preserved, and leadership is re-acquired only by the next /// round's normal `gc/state` acquisition, never restored here. Runs outside `pointer_mutex` for symmetry - /// with `stop()` (it spawns threads but joins nothing, so it does not block). + /// with `stop()`; it normally only spawns threads, but a worker-start failure joins any worker already + /// started during rollback and rethrows. snapshot->start(); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp index 610d13879779..8b126b83d936 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp @@ -3,11 +3,13 @@ #include #include #include +#include #include #include #include #include #include +#include #include #include #include @@ -20,6 +22,13 @@ namespace DB::ErrorCodes extern const int TIMEOUT_EXCEEDED; extern const int SOCKET_TIMEOUT; extern const int MEMORY_LIMIT_EXCEEDED; + extern const int FAULT_INJECTED; +} + +namespace DB::FailPoints +{ + extern const char cas_gc_scheduler_fail_before_heartbeat_worker_start[]; + extern const char cas_gc_scheduler_fail_before_worker_start[]; } namespace DB::Cas @@ -90,19 +99,55 @@ CasGcScheduler::~CasGcScheduler() void CasGcScheduler::start() { - std::lock_guard lock(mutex); - if (thread.joinable()) - return; - stopping = false; - thread = ThreadFromGlobalPool([this] { loop(); }); - hb_thread = ThreadFromGlobalPool([this] { heartbeatLoop(); }); + std::lock_guard threads_lock(threads_mutex); + { + std::lock_guard lock(mutex); + if (scheduler_state == SchedulerState::Running) + return; + scheduler_state = SchedulerState::Running; + } + try + { + fiu_do_on(FailPoints::cas_gc_scheduler_fail_before_heartbeat_worker_start, + { + throw Exception(ErrorCodes::FAULT_INJECTED, "Injected failure before starting CAS GC heartbeat worker"); + }); + hb_thread = ThreadFromGlobalPool([this] { heartbeatLoop(); }); + fiu_do_on(FailPoints::cas_gc_scheduler_fail_before_worker_start, + { + throw Exception(ErrorCodes::FAULT_INJECTED, "Injected failure before starting CAS GC worker"); + }); + thread = ThreadFromGlobalPool([this] { loop(); }); + } + catch (...) + { + { + std::lock_guard lock(mutex); + scheduler_state = SchedulerState::Stopped; + round_requested = false; + } + wake.notify_all(); + if (thread.joinable()) + thread.join(); + if (hb_thread.joinable()) + hb_thread.join(); + i_am_leader.store(false, std::memory_order_relaxed); + throw; + } } void CasGcScheduler::stop() { + std::lock_guard threads_lock(threads_mutex); { std::lock_guard lock(mutex); - stopping = true; + if (scheduler_state == SchedulerState::Stopped) + { + i_am_leader.store(false, std::memory_order_relaxed); + return; + } + scheduler_state = SchedulerState::Stopped; + round_requested = false; } wake.notify_all(); if (thread.joinable()) @@ -122,7 +167,7 @@ void CasGcScheduler::requestRoundSoon() { { std::lock_guard lock(mutex); - if (stopping || !thread.joinable()) + if (scheduler_state != SchedulerState::Running) return; round_requested = true; } @@ -299,9 +344,10 @@ void CasGcScheduler::loop() while (true) { { - std::unique_lock lock(mutex); - wake.wait_for(lock, interval, [this] { return stopping || round_requested; }); - if (stopping) + UniqueLock lock(mutex); + wake.wait_for(lock.getUnderlyingLock(), interval, [this]() TSA_NO_THREAD_SAFETY_ANALYSIS + { return scheduler_state == SchedulerState::Stopped || round_requested; }); + if (scheduler_state == SchedulerState::Stopped) return; round_requested = false; } @@ -332,9 +378,9 @@ void CasGcScheduler::loop() } try { - /// LOW/benign: if stop() flips `stopping` while we're blocked here (a concurrent manual + /// LOW/benign: if stop() flips `scheduler_state` while we're blocked here (a concurrent manual /// round holds gc_round_mutex), we still run one more Scheduled round once it unblocks, - /// before the next wait_for() observes `stopping` - an accepted extra round, not a + /// before the next wait_for() observes `scheduler_state` - an accepted extra round, not a /// correctness issue. std::lock_guard round_lock(gc_round_mutex); @@ -427,8 +473,9 @@ void CasGcScheduler::heartbeatLoop() while (true) { { - std::unique_lock lock(mutex); - if (wake.wait_for(lock, hb_interval, [this] { return stopping; })) + UniqueLock lock(mutex); + if (wake.wait_for(lock.getUnderlyingLock(), hb_interval, [this]() TSA_NO_THREAD_SAFETY_ANALYSIS + { return scheduler_state == SchedulerState::Stopped; })) return; } /// rev.7 §3 [C1] + rev.8 §9 item 8: self-exit on ANY terminal (or FORGET-intent) pool, same as diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h index 80b50f2fc448..ddab0711d6c2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -162,13 +163,19 @@ class CasGcScheduler /// Test seam (rev.7 §3 [C1]): block up to `timeout` for BOTH the pacing and heartbeat loops to have /// SELF-EXITED via the terminal-lifecycle check — a `Vanished` pool or a published FORGET intent — as - /// opposed to exiting through `stop()`'s `stopping` flag. Returns false on timeout. Predicate-based + /// opposed to exiting through `stop()` transitioning `scheduler_state` to `Stopped`. Returns false on timeout. Predicate-based /// wait (no sleeps); the loops set their flag under `terminal_exit_mutex` before notifying, so there is /// no lost-wakeup window. Lets a test prove the self-exit path fired without relying on any wall-clock /// delay. bool waitForTerminalSelfExitForTest(std::chrono::milliseconds timeout); private: + enum class SchedulerState + { + Stopped, + Running + }; + /// Waits for the configured interval, runs scheduled rounds while the scheduler is active, and /// logs exceptions before continuing with the next tick. The round lock serializes this worker /// with `runOneRoundNow` because the persistent `gc` object is not thread-safe. @@ -210,15 +217,17 @@ class CasGcScheduler /// the round so stop()/heartbeatLoop are not blocked, so the round cannot hold `mutex`. std::mutex gc_round_mutex; + std::mutex threads_mutex; + ThreadFromGlobalPool thread TSA_GUARDED_BY(threads_mutex); + ThreadFromGlobalPool hb_thread TSA_GUARDED_BY(threads_mutex); + std::mutex mutex; std::condition_variable wake; - bool stopping = false; - bool round_requested = false; /// guarded by `mutex`; coalesced external wake request - ThreadFromGlobalPool thread; + SchedulerState scheduler_state TSA_GUARDED_BY(mutex) = SchedulerState::Stopped; + bool round_requested TSA_GUARDED_BY(mutex) = false; /// coalesced external wake request /// Set by the round worker and read by the heartbeat worker. It is only an in-process hint: the /// durable lease remains the authority, and a failed round clears the hint before retrying. std::atomic i_am_leader{false}; - ThreadFromGlobalPool hb_thread; /// Set true for the whole body of one round (`runRoundLogged`, held across the `gc_round_mutex` /// critical section a scheduled or manual round runs under) and cleared when it returns, on the @@ -227,7 +236,7 @@ class CasGcScheduler /// rev.7 §3 [C1] test-observation seam: set (under `terminal_exit_mutex`) by `loop`/`heartbeatLoop` /// respectively when they SELF-EXIT via the terminal-lifecycle check, NOT when `stop()` flips - /// `stopping`. `waitForTerminalSelfExitForTest` waits on `terminal_exit_cv` for BOTH, so a test proves + /// `scheduler_state` to `Stopped`. `waitForTerminalSelfExitForTest` waits on `terminal_exit_cv` for BOTH, so a test proves /// the self-exit path fired without any sleep. Purely diagnostic; production behavior never reads them. std::atomic loop_exited_on_terminal_for_test{false}; std::atomic hb_exited_on_terminal_for_test{false}; diff --git a/src/Disks/tests/gtest_cas_gc_stop_start.cpp b/src/Disks/tests/gtest_cas_gc_stop_start.cpp index 6079a2b3436d..1a529fb85375 100644 --- a/src/Disks/tests/gtest_cas_gc_stop_start.cpp +++ b/src/Disks/tests/gtest_cas_gc_stop_start.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include @@ -34,9 +35,16 @@ namespace DB::ErrorCodes { +extern const int FAULT_INJECTED; extern const int INVALID_STATE; } +namespace DB::FailPoints +{ +extern const char cas_gc_scheduler_fail_before_heartbeat_worker_start[]; +extern const char cas_gc_scheduler_fail_before_worker_start[]; +} + using namespace DB; using DB::Cas::CasGcScheduler; using DB::Cas::GcRoundLogRecord; @@ -342,6 +350,42 @@ TEST(CASGCStopStart, StopAndStartAreIdempotent) sched.stop(); } +TEST(CASGCStopStart, StopClearsLeadershipAfterManualRoundWithoutStart) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + CasGcScheduler sched(store, std::chrono::seconds(3600), "CasGcManualStopTest", "ca-disk"); + + const RoundReport report = sched.runOneRoundNow(); + ASSERT_TRUE(report.acquired_lease); + ASSERT_TRUE(sched.gcHealth().is_leader); + + sched.stop(); + EXPECT_FALSE(sched.gcHealth().is_leader); +} + +TEST(CASGCStopStart, StartFailureRollsBackAndCanBeRetried) +{ + for (const char * failpoint : + {FailPoints::cas_gc_scheduler_fail_before_heartbeat_worker_start, + FailPoints::cas_gc_scheduler_fail_before_worker_start}) + { + SCOPED_TRACE(failpoint); + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + CasGcScheduler sched(store, std::chrono::seconds(3600), "CasGcStartFailureTest", "ca-disk"); + + FailPointInjection::enableFailPoint(failpoint); + Cas::tests::expectThrowsCode(ErrorCodes::FAULT_INJECTED, [&] { sched.start(); }); + FailPointInjection::disableFailPoint(failpoint); + EXPECT_TRUE(sched.isQuiescent()); + + EXPECT_NO_THROW(sched.start()); + sched.stop(); + EXPECT_TRUE(sched.isQuiescent()); + } +} + /// (d) START refuses on a Vanished disk with the typed 668 (`INVALID_STATE`) error -- restarting GC on a /// decommissioned pool is meaningless and would only spin failing rounds -- while STOP on the SAME /// Vanished disk (with a live scheduler present) SUCCEEDS: stopping the reclaimer on a sick disk is a @@ -439,6 +483,42 @@ TEST(CASGCStopStart, ConcurrentStopStartFromTwoThreadsStaysConsistent) storage->gcStop(); } +TEST(CASGCStopStart, RequestRoundSoonConcurrentWithStopStartDoesNotRace) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend); + CasGcScheduler sched(store, std::chrono::seconds(3600), "CasGcRequestStopRaceTest", "ca-disk"); + sched.start(); + + std::promise start; + const auto begin = start.get_future().share(); + + auto requester = std::async(std::launch::async, [&] + { + begin.wait(); + for (size_t i = 0; i < 1000; ++i) + sched.requestRoundSoon(); + }); + auto lifecycle = std::async(std::launch::async, [&] + { + begin.wait(); + for (size_t i = 0; i < 1000; ++i) + { + sched.stop(); + sched.start(); + } + }); + + start.set_value(); + ASSERT_EQ(requester.wait_for(std::chrono::seconds(60)), std::future_status::ready); + ASSERT_EQ(lifecycle.wait_for(std::chrono::seconds(60)), std::future_status::ready); + requester.get(); + lifecycle.get(); + + sched.stop(); + EXPECT_FALSE(sched.gcHealth().is_leader); +} + /// (T11 cannot-verify, acceptance matrix) Operator intent PERSISTS across a transient recovery: after the /// operator STOPs GC, the disk loses its mount lease (transient-not-live) and self-remounts back to Live — /// and NOTHING restarts the GC scheduler. Recovery is a Pool-internal operation with no reference to the From 2a2f16f35e7b89fdbc494a418ea780364e7c5bca Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Wed, 16 Sep 2026 12:41:52 +0200 Subject: [PATCH 09/15] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/2349 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #2349 from Altinity/cas/memworker-split-allocated-source MemoryWorker: use sanitizer info for allocated counter # Conflicts: # src/Common/MemoryWorker.cpp --- src/Common/MemoryWorker.cpp | 52 ++++++++++++++++++++++++++++--------- src/Common/MemoryWorker.h | 8 +++++- 2 files changed, 47 insertions(+), 13 deletions(-) diff --git a/src/Common/MemoryWorker.cpp b/src/Common/MemoryWorker.cpp index 178c878a770a..e1d3cce2d937 100644 --- a/src/Common/MemoryWorker.cpp +++ b/src/Common/MemoryWorker.cpp @@ -15,6 +15,8 @@ #include #include +#include + #include #include @@ -22,6 +24,10 @@ #include +#if defined(ADDRESS_SANITIZER) || defined(THREAD_SANITIZER) || defined(MEMORY_SANITIZER) +#include +#endif + namespace fs = std::filesystem; namespace ProfileEvents @@ -540,20 +546,26 @@ MemoryWorker::~MemoryWorker() #endif } -uint64_t MemoryWorker::getMemoryUsage(bool log_error) +MemoryWorker::MemoryUsage MemoryWorker::getMemoryUsage(bool log_error) { + MemoryUsage usage; + switch (source) { case MemoryUsageSource::Cgroups: { if (cgroups_reader != nullptr) - return cgroups_reader->readMemoryUsage(); + { + usage.resident = cgroups_reader->readMemoryUsage(); + break; + } [[fallthrough]]; } case MemoryUsageSource::Jemalloc: #if USE_JEMALLOC epoch_mib.setValue(0); - return resident_mib.getValue(); + usage.resident = resident_mib.getValue(); + break; #else [[fallthrough]]; #endif @@ -561,9 +573,19 @@ uint64_t MemoryWorker::getMemoryUsage(bool log_error) { if (log_error) LOG_ERROR(log, "Trying to fetch memory usage while no memory source can be used"); - return 0; + break; } } + +#if defined(ADDRESS_SANITIZER) || defined(THREAD_SANITIZER) || defined(MEMORY_SANITIZER) + /// Sanitizer memory overhead (redzones, ...) makes RSS exceed application allocations; + // use allocator bytes for MEMORY_LIMIT_EXCEEDED, resident for RSS/cgroup limit sizing. + usage.allocated = __sanitizer_get_current_allocated_bytes(); +#else + usage.allocated = usage.resident; +#endif + + return usage; } namespace @@ -844,6 +866,7 @@ void MemoryWorker::updateResidentMemoryThread() Stopwatch total_watch; +<<<<<<< HEAD Int64 resident = getMemoryUsage(first_run); /// Speculatively reserve growth headroom on top of the observed RSS. @@ -912,9 +935,13 @@ void MemoryWorker::updateResidentMemoryThread() } prev_resident = resident; MemoryTracker::updateRSS(speculative_rss); +======= + const MemoryUsage memory_usage = getMemoryUsage(first_run); + MemoryTracker::updateRSS(memory_usage.resident); +>>>>>>> 482a836f403 (Merge pull request #2349 from Altinity/cas/memworker-split-allocated-source) if (page_cache) - page_cache->autoResize(std::max(resident, total_memory_tracker.get()), total_memory_tracker.getHardLimit()); + page_cache->autoResize(std::max(memory_usage.resident, total_memory_tracker.get()), total_memory_tracker.getHardLimit()); #if USE_JEMALLOC const auto memory_tracker_limit = total_memory_tracker.getHardLimit(); @@ -922,7 +949,7 @@ void MemoryWorker::updateResidentMemoryThread() const auto purge_dirty_pages_threshold = static_cast(memory_tracker_limit) * purge_dirty_pages_threshold_ratio; const bool needs_purge - = (purge_total_memory_threshold_ratio > 0 && static_cast(resident) > purge_total_memory_threshold) + = (purge_total_memory_threshold_ratio > 0 && static_cast(memory_usage.resident) > purge_total_memory_threshold) || (purge_dirty_pages_threshold_ratio > 0 && static_cast(pdirty_mib.getValue() * page_size) > purge_dirty_pages_threshold); @@ -978,22 +1005,23 @@ void MemoryWorker::updateResidentMemoryThread() } } - /// update MemoryTracker with `allocated` information from jemalloc when: + /// update MemoryTracker with resident memory information (cgroup or jemalloc) when: /// - it's a first run of MemoryWorker (MemoryTracker could've missed some allocation before its initialization) /// - MemoryTracker stores a negative value /// - `correct_tracker` is set to true if (first_run || total_memory_tracker.get() < 0) [[unlikely]] - MemoryTracker::updateAllocated(resident, /*log_change=*/true); + MemoryTracker::updateAllocated(memory_usage.allocated, /*log_change=*/true); else if (correct_tracker) - MemoryTracker::updateAllocated(resident, /*log_change=*/false); + MemoryTracker::updateAllocated(memory_usage.allocated, /*log_change=*/false); #else /// we don't update in the first run if we don't have jemalloc - /// because we can only use resident memory information + /// because without a sanitizer we can only use resident memory information /// resident memory can be much larger than the actual allocated memory /// so we rather ignore the potential difference caused by allocated memory /// before MemoryTracker initialization + /// sanitizer builds provide allocated memory, but keep the same behavior if (total_memory_tracker.get() < 0 || correct_tracker) [[unlikely]] - MemoryTracker::updateAllocated(resident, /*log_change=*/false); + MemoryTracker::updateAllocated(memory_usage.allocated, /*log_change=*/false); #endif /// Capture the settings generation before reading ratio/ceiling. We re-read @@ -1026,7 +1054,7 @@ void MemoryWorker::updateResidentMemoryThread() /// are excluded. Under load `tracked` can be orders of magnitude smaller /// than the actual RSS, which makes `(tracked + available) * ratio` compute /// a hard limit close to current RSS and reject every subsequent allocation. - Int64 used = std::max(0, resident); + Int64 used = std::max(0, memory_usage.resident); /// `used + available` is the upper bound of memory we could potentially own: /// what we already use plus what is still free in our cgroup (or on the host). /// Scaling by `ratio < 1` leaves headroom for other processes on the host. diff --git a/src/Common/MemoryWorker.h b/src/Common/MemoryWorker.h index 45ccdb315038..9eb2fc7107eb 100644 --- a/src/Common/MemoryWorker.h +++ b/src/Common/MemoryWorker.h @@ -133,7 +133,13 @@ class MemoryWorker ~MemoryWorker(); private: - uint64_t getMemoryUsage(bool log_error); + struct MemoryUsage + { + Int64 resident = 0; + Int64 allocated = 0; + }; + + MemoryUsage getMemoryUsage(bool log_error); void updateResidentMemoryThread(); From 82e93264e18870fe9b431242f51595bedc866cfa Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:45:05 +0200 Subject: [PATCH 10/15] Resolve conflicts in cherry-pick of #2349 Kept the speculative RSS reservation block present on antalya-26.8 and adapted it to the `MemoryUsage` return type of `getMemoryUsage` introduced by #2349 (`resident` -> `memory_usage.resident`). Source-PR: #2349 (https://github.com/Altinity/ClickHouse/pull/2349) --- src/Common/MemoryWorker.cpp | 17 ++++++----------- 1 file changed, 6 insertions(+), 11 deletions(-) diff --git a/src/Common/MemoryWorker.cpp b/src/Common/MemoryWorker.cpp index e1d3cce2d937..4fa59e326579 100644 --- a/src/Common/MemoryWorker.cpp +++ b/src/Common/MemoryWorker.cpp @@ -866,8 +866,7 @@ void MemoryWorker::updateResidentMemoryThread() Stopwatch total_watch; -<<<<<<< HEAD - Int64 resident = getMemoryUsage(first_run); + const MemoryUsage memory_usage = getMemoryUsage(first_run); /// Speculatively reserve growth headroom on top of the observed RSS. /// `resident - prev_resident` is how much RSS actually grew during the last tick; @@ -894,7 +893,7 @@ void MemoryWorker::updateResidentMemoryThread() /// overhead dominates the `resident - tracked` gap there. /// Skip speculation on the very first run: there is no previous interval to /// extrapolate from. - Int64 speculative_rss = resident; + Int64 speculative_rss = memory_usage.resident; /// Speculation only influences the global hard-limit check in /// `MemoryTracker::allocImpl` (the `will_be_rss > current_hard_limit` branch), /// so it is only meaningful when a global hard limit is configured. When the @@ -912,12 +911,12 @@ void MemoryWorker::updateResidentMemoryThread() /// real resident, triggering false `MEMORY_LIMIT_EXCEEDED` decisions in /// `MemoryTracker::allocImpl`. Clamp `tracked` to `0` first. Int64 tracked = std::max(0, total_memory_tracker.get()); - Int64 delta = std::min(resident - prev_resident, resident - tracked); + Int64 delta = std::min(memory_usage.resident - prev_resident, memory_usage.resident - tracked); /// Speculate only while real `resident` is still below the hard limit. /// Once `resident >= current_hard_limit`, any positive allocation already /// trips the `will_be_rss > current_hard_limit` branch in /// `MemoryTracker::allocImpl`, so there is nothing left to reserve. - if (delta > 0 && resident < current_hard_limit) + if (delta > 0 && memory_usage.resident < current_hard_limit) { /// The reservation can be at most `current_hard_limit - resident`: /// reserving beyond the hard limit gains no early-throw power (any @@ -925,7 +924,7 @@ void MemoryWorker::updateResidentMemoryThread() /// reaches it). Capping the reservation *before* adding it to `resident` /// also guarantees the signed `Int64` addition cannot overflow, even /// with a very large configured ratio. - const Int64 headroom = current_hard_limit - resident; + const Int64 headroom = current_hard_limit - memory_usage.resident; double reserve_double = static_cast(delta) * rss_speculative_reserve_ratio; Int64 reserve = (reserve_double >= static_cast(headroom)) ? headroom @@ -933,12 +932,8 @@ void MemoryWorker::updateResidentMemoryThread() speculative_rss += reserve; } } - prev_resident = resident; + prev_resident = memory_usage.resident; MemoryTracker::updateRSS(speculative_rss); -======= - const MemoryUsage memory_usage = getMemoryUsage(first_run); - MemoryTracker::updateRSS(memory_usage.resident); ->>>>>>> 482a836f403 (Merge pull request #2349 from Altinity/cas/memworker-split-allocated-source) if (page_cache) page_cache->autoResize(std::max(memory_usage.resident, total_memory_tracker.get()), total_memory_tracker.getHardLimit()); From 1382a9715e6d1856a65ae6a4ea28463fa5e7fe76 Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Tue, 15 Sep 2026 09:56:38 +0200 Subject: [PATCH 11/15] Merge pull request #2360 from Altinity/fix/antalya-26.6/coverage-unused-variable Fix the coverage build: unused variable in gtest_cas_throttling_gate.cpp Source-PR: #2360 (https://github.com/Altinity/ClickHouse/pull/2360) --- src/Disks/tests/gtest_cas_throttling_gate.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/Disks/tests/gtest_cas_throttling_gate.cpp b/src/Disks/tests/gtest_cas_throttling_gate.cpp index d208eede19f9..8f780b793ae7 100644 --- a/src/Disks/tests/gtest_cas_throttling_gate.cpp +++ b/src/Disks/tests/gtest_cas_throttling_gate.cpp @@ -123,6 +123,8 @@ TEST(CASThrottlingGate, EveryUserVisibleStatementSucceedsUnderFirstPerKeyThrottl #if !WITH_COVERAGE EXPECT_GT(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolve_reads_before, 0u) << "no throttled write was settled by a read -- the engine's ambiguity-resolution path never ran"; +#else + (void)resolve_reads_before; #endif } From 77bd8f1c23e03e383ddf8a511be13ffb2887b84b Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:40:52 +0200 Subject: [PATCH 12/15] Merge pull request #2384 from Altinity/fix/antalya-26.6/ci-rustfs-runtime-knobs CI: cap the RustFS blocking-thread pool on the CAS-S3 lanes and collect its log Source-PR: #2384 (https://github.com/Altinity/ClickHouse/pull/2384) --- ci/jobs/scripts/clickhouse_proc.py | 24 +++++++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/ci/jobs/scripts/clickhouse_proc.py b/ci/jobs/scripts/clickhouse_proc.py index 4aa0ff66b6bd..72c457ed3dac 100644 --- a/ci/jobs/scripts/clickhouse_proc.py +++ b/ci/jobs/scripts/clickhouse_proc.py @@ -280,9 +280,31 @@ def start_rustfs(self): # Raise the open-files limit for the same reason start_azurite does: under parallel load # the server holds thousands of S3 connections, and at the default soft limit (1024) # rustfs runs out of fds and refuses new TCP connections in bursts. + # RustFS shares the job container's cgroup with the ClickHouse server, and the server's + # MemoryWorker reads the cgroup total, so RustFS memory is charged against + # max_server_memory_usage. RustFS 1.0.0-rc.3 grows its tokio blocking-thread pool one + # thread per concurrent blocking dispatch (four or more per PUT under the default strict + # durability), keeps each idle thread alive for 60 s, and each thread pins about 12 MB of + # allocator arena plus a 1 MiB stack: on a three-hour sanitizer shard that reached 1074 + # threads and 14 GB of anonymous memory. The runtime knobs cap the pool and make it shrink; + # the allocator knob returns freed arenas; relaxed durability halves the blocking dispatches + # per PUT (no fsync chain, acceptable for an ephemeral CI pool that is wiped per run). + # Measured locally on the same workload: 1074 -> 130 threads, 14.4 GB -> 2.5 GB anonymous, + # rename_data 44-69 ms -> 0.6 ms, the shard 30% faster. + # RUSTFS_OBS_USE_STDOUT routes the RustFS log (error level by default) into rustfs.log, which + # is already collected as a job artifact; without it RustFS writes to logs/ under its cwd. + rustfs_env = ( + "RUSTFS_SCANNER_ENABLED=false RUSTFS_HEAL_ENABLED=false " + "RUSTFS_RUNTIME_MAX_BLOCKING_THREADS=64 " + "RUSTFS_RUNTIME_THREAD_KEEP_ALIVE=5 " + "RUSTFS_ALLOCATOR_RECLAIM_ENABLED=true " + "RUSTFS_ALLOCATOR_RECLAIM_INTERVAL_SECS=30 " + "RUSTFS_DURABILITY_MODE=relaxed " + "RUSTFS_OBS_USE_STDOUT=true " + ) command = ( "(ulimit -n 1048576 2>/dev/null || ulimit -n $(ulimit -Hn)) && " - f"RUSTFS_SCANNER_ENABLED=false RUSTFS_HEAL_ENABLED=false " + f"{rustfs_env}" f"{rustfs_bin} server --address 0.0.0.0:11121 " f"--access-key clickhouse --secret-key clickhouse {data_dir}" ) From 4940154b500503a929a7633a46df04833f38039a Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Fri, 18 Sep 2026 12:13:58 +0200 Subject: [PATCH 13/15] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/2385 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #2385 from Altinity/fix/antalya-26.6/lldb-budget-and-sigcont CI: lldb stack budget by build flavor, and keep lldb's JIT breakpoint out of the server # Conflicts: # ci/tests/test_print_stacktraces.py # tests/clickhouse-test --- ci/tests/test_print_stacktraces.py | 274 +++++++++++++++++++++++++++++ tests/clickhouse-test | 164 ++++++++++++++++- 2 files changed, 433 insertions(+), 5 deletions(-) create mode 100644 ci/tests/test_print_stacktraces.py diff --git a/ci/tests/test_print_stacktraces.py b/ci/tests/test_print_stacktraces.py new file mode 100644 index 000000000000..08a9518d11af --- /dev/null +++ b/ci/tests/test_print_stacktraces.py @@ -0,0 +1,274 @@ +""" +End-to-end tests for the stacktrace helpers in tests/clickhouse-test. + +Background +---------- +``clickhouse-test`` assigns ``args = parse_args()`` only inside +``if __name__ == "__main__":``. On macOS, Python's default +multiprocessing start method is ``spawn``, which re-imports the module +in each worker without executing ``__main__`` — so module-level +``args`` is undefined, and any helper that closed over it crashed with +``NameError``. See the fast_test_arm_darwin failure where the +hung-check path raised ``NameError: name 'args' is not defined`` inside +``get_server_pid``. + +These tests reproduce the same import condition by loading +``clickhouse-test`` via ``runpy.run_path`` (which, like spawn, does not +run ``__main__``) and then invoke each public stacktrace helper against +the live ClickHouse server provided by the ``ClickHouseService`` +fixture in ``ci/jobs/ci_tests_job.py``. + +Pre-fix: NameError inside the fresh import. +Post-fix: the helpers run to completion against a live server. +""" + +import argparse +import io +import os +import re +import runpy +from contextlib import redirect_stdout +from pathlib import Path +from time import sleep, time + +_REPO_ROOT = Path(__file__).resolve().parent.parent.parent +_CLICKHOUSE_TEST = str(_REPO_ROOT / "tests" / "clickhouse-test") + + +def _load_clickhouse_test(require_server=True): + # Mimic a spawn worker: load clickhouse-test without running __main__, + # so module-level `args` is absent. + ct = runpy.run_path(_CLICKHOUSE_TEST) + assert "args" not in ct, ( + "module-level 'args' must not be defined outside __main__; otherwise " + "the spawn-worker scenario this test reproduces does not apply" + ) + + if require_server: + # Sanity-check the precondition: the CI tests job started a server. + assert ct["pgrep"](command="clickhouse-server"), ( + "no clickhouse-server process found — this test expects ClickHouseService " + "(see ci/jobs/ci_tests_job.py) to be running on localhost:9000" + ) + return ct + + +def _make_args(): + # Minimal args namespace: only the fields the helpers and their + # transitive callees actually read. Mirrors what __main__ assigns + # after parse_args() for a local-server, plaintext-TCP, + # default-database run. + return argparse.Namespace( + client="clickhouse-client --port=9000", + client_option=None, + secure=False, + tcp_host="localhost", + http_port=8123, + client_options_query_str="", + replicated_database=False, + shared_catalog=False, + force_color=False, + binary=os.environ.get("CLICKHOUSE_BINARY", "clickhouse"), + # A reachable server means __main__ collected build flags at startup; + # a non-ASan set keeps print_c_stacktraces on its lldb path. + build_flags=set(), + ) + + +def _capture_lldb_budgets(ct, args, pids, elapsed=0.0, dump=None, **kwargs): + """Run print_c_stacktraces with the collector stubbed, returning the + per-PID timeouts it was called with (and its stdout).""" + budgets = [] + + def collector(pid, timeout=None): + budgets.append(timeout) + if elapsed: + sleep(elapsed) + return "x" * 2000 if dump is None else dump + + globals_ = ct["print_c_stacktraces"].__globals__ + saved = ( + globals_["get_stacktraces_from_lldb"], + globals_["get_all_server_pids"], + globals_["get_server_pid"], + globals_["is_asan_build"], + ) + globals_["get_stacktraces_from_lldb"] = collector + globals_["get_all_server_pids"] = lambda: list(pids) + globals_["get_server_pid"] = lambda: pids[0] + globals_["is_asan_build"] = lambda _args: False + captured = io.StringIO() + try: + with redirect_stdout(captured): + ct["print_c_stacktraces"](args, **kwargs) + finally: + ( + globals_["get_stacktraces_from_lldb"], + globals_["get_all_server_pids"], + globals_["get_server_pid"], + globals_["is_asan_build"], + ) = saved + return budgets, captured.getvalue() + + +def test_print_c_stacktraces_against_live_server(): + ct = _load_clickhouse_test() + args = _make_args() + + captured = io.StringIO() + with redirect_stdout(captured): + ct["print_c_stacktraces"](args) + output = captured.getvalue() + + # The function must have located the server PID and reached gdb. + # Whether the attach itself succeeds depends on the host's + # `kernel.yama.ptrace_scope` and is not asserted. + assert "Collecting C stacktraces from main server process" in output, output + + +def test_print_sql_stacktraces_against_live_server(): + ct = _load_clickhouse_test() + args = _make_args() + + captured = io.StringIO() + with redirect_stdout(captured): + ct["print_sql_stacktraces"](args) + output = captured.getvalue() + + # The function must have queried system.stack_trace and printed + # traces. We don't require a specific thread name — any non-trivial + # output confirms the round-trip succeeded. + assert "Collecting stacktraces from system.stack_trace table" in output, output + assert "trace_str" in output or "thread_name" in output, output + + +def test_lldb_budget_scales_with_build_flavor(): + # A debug or sanitizer or coverage server needs far longer than 30s to walk; + # a release server does not, and must keep the tight budget that bounds a + # genuinely wedged lldb. + ct = _load_clickhouse_test(require_server=False) + flags = ct["BuildFlags"] + release, slow = ct["LLDB_TIMEOUT"], ct["LLDB_SLOW_BUILD_TIMEOUT"] + assert slow > release, (slow, release) + + args = _make_args() + for build_flags, expected in ( + ({flags.RELEASE}, release), + ({flags.DEBUG}, slow), + ({flags.THREAD}, slow), + ({flags.MEMORY}, slow), + ({flags.UNDEFINED}, slow), + ({flags.WITH_COVERAGE}, slow), + ): + args.build_flags = build_flags + assert ct["lldb_timeout_for_build"](args) == expected, build_flags + # And the loop passes that value through. It is clamped to what is left + # of the aggregate ceiling, so compare with a tolerance rather than + # exactly. + budgets, _ = _capture_lldb_budgets(ct, args, [4242]) + assert len(budgets) == 1 and abs(budgets[0] - expected) < 1, ( + build_flags, + budgets, + expected, + ) + + +def test_lldb_budget_survives_the_spawn_start_method(): + # A spawned worker re-imports the module with RELEASE_NON_SANITIZED / + # SANITIZED back at their defaults while `args` is transferred intact, so + # the budget must come from args.build_flags, not from those globals. + ct = _load_clickhouse_test(require_server=False) + globals_ = ct["print_c_stacktraces"].__globals__ + saved = (globals_["RELEASE_NON_SANITIZED"], globals_["SANITIZED"]) + globals_["RELEASE_NON_SANITIZED"] = False + globals_["SANITIZED"] = False + try: + args = _make_args() + args.build_flags = {ct["BuildFlags"].DEBUG} + budgets, _ = _capture_lldb_budgets(ct, args, [4242]) + finally: + globals_["RELEASE_NON_SANITIZED"], globals_["SANITIZED"] = saved + + assert len(budgets) == 1, budgets + assert abs(budgets[0] - ct["LLDB_SLOW_BUILD_TIMEOUT"]) < 1, budgets + + +def test_lldb_budget_reads_the_binary_when_build_flags_are_missing(): + # The startup-failure caller (main -> check_server_started) runs before + # `args.build_flags` is assigned, so the flavor comes from the binary via + # `clickhouse local` rather than falling back to the release budget. + ct = _load_clickhouse_test(require_server=False) + args = _make_args() + delattr(args, "build_flags") + + for slow, expected_key in ((True, "LLDB_SLOW_BUILD_TIMEOUT"), (False, "LLDB_TIMEOUT")): + globals_ = ct["lldb_timeout_for_build"].__globals__ + saved = globals_["is_slow_build_binary"] + globals_["is_slow_build_binary"] = lambda _args, _s=slow: _s + try: + budgets, _ = _capture_lldb_budgets(ct, args, [4242]) + finally: + globals_["is_slow_build_binary"] = saved + + assert len(budgets) == 1, (slow, budgets) + assert abs(budgets[0] - ct[expected_key]) < 1, (slow, budgets) + + +def test_slow_build_binary_probe_is_server_independent_and_fails_closed(): + # The probe must not need a live server (it exists for the path where the + # server never came up), and an unreadable binary must yield the tighter + # budget rather than raising on an already-failing run. + ct = _load_clickhouse_test(require_server=False) + + args = _make_args() + args.binary = "/nonexistent/clickhouse-does-not-exist" + assert ct["is_slow_build_binary"](args) is False + + # And the query it issues names every flag collect_build_flags derives, so + # the two cannot drift into disagreeing about what "slow" means. + source = Path(_CLICKHOUSE_TEST).read_text(encoding="utf-8") + probe = source.split("def is_slow_build_binary")[1].split("\ndef ")[0] + for token in ("BUILD_TYPE", "Debug", "WITH_COVERAGE", "-fsanitize="): + assert token in probe, token + + +def test_lldb_pid_loop_honours_the_aggregate_deadline(): + # The loop spans every server process, so a per-PID budget alone does not + # bound it. Exhausting the total must stop the loop and say how many + # processes went undumped. + ct = _load_clickhouse_test(require_server=False) + args = _make_args() + args.build_flags = {ct["BuildFlags"].DEBUG} + pids = [111, 222, 333, 444] + + started = time() + budgets, output = _capture_lldb_budgets( + ct, args, pids, elapsed=1.0, per_pid_timeout=30, total_timeout=2 + ) + took = time() - started + + assert len(budgets) < len(pids), budgets + assert took < 30, took + # Each call is clamped to what is left, so no single attach can overrun the + # ceiling on its own. + assert all(b <= 2 for b in budgets), budgets + skipped = len(pids) - len(budgets) + assert f"skipping {skipped} of {len(pids)} processes" in output, output + assert str(pids[-1]) in output.split("skipping")[1], output + + +def test_timeout_handler_keeps_the_tight_lldb_pair(): + # The per-test timeout handler runs after its one-shot alarm has fired, so + # the alarm cannot bound it and only the job's outer timeout is left. That + # site therefore keeps the 30s per-PID value; the abort paths, where no + # alarm is pending, take the flavor budget. Asserted on the source so + # neither the outer-deadline risk nor the hung-check coverage can regress. + source = " ".join(Path(_CLICKHOUSE_TEST).read_text(encoding="utf-8").split()) + calls = re.findall(r"(?>>>>>> 907703cf7f8 (Merge pull request #2385 from Altinity/fix/antalya-26.6/lldb-budget-and-sigcont) ) @@ -1384,15 +1452,83 @@ def _append_stacktrace_log(filename: str, block: str) -> None: print(f"Could not write {filename}: {e}", file=sys.stderr) -def print_c_stacktraces(args) -> None: +def is_slow_build_binary(args) -> bool: + """Debug/sanitizer/coverage flavor read from the binary, not the server. + + Mirrors the flag set `collect_build_flags` derives, so the answer matches + what `args.build_flags` would have said had the server come up. An + unreadable or unparseable result is reported as not-slow, keeping the + tighter of the two budgets. + """ + binary = getattr(args, "binary", "clickhouse") + local = find_clickhouse_command(binary, "local") + # Name the same four sanitizers `collect_build_flags` maps, not every + # `-fsanitize=`: a CFI build carries `-fsanitize=cfi-vcall` on top of + # RelWithDebInfo, and the collected-flags path calls that a release build. + sanitizers = " OR ".join( + f"position('sanitize={san}' IN value)" + for san in ("thread", "address", "undefined", "memory") + ) + out = shell_get_output( + f"{local} --query \"" + "SELECT countIf(name = 'BUILD_TYPE' AND position('Debug' IN value)) " + "+ countIf(name = 'WITH_COVERAGE' AND value IN ('ON', '1')) " + f"+ countIf(name = 'CXX_FLAGS' AND ({sanitizers})) " + 'FROM system.build_options"', + timeout=60, + ) + try: + return int(out.strip()) > 0 + except ValueError: + return False + + +def lldb_timeout_for_build(args) -> float: + """Per-PID lldb budget for this server's build flavor. + + Debug, sanitizer and coverage builds are far more expensive to unwind and + symbolize. Must read `args.build_flags`, not the module-level + RELEASE_NON_SANITIZED / SANITIZED globals, which a spawned worker + re-imports at their defaults while `args` is transferred intact. When the + flags were never collected (server never started), read the flavor from + the binary via `clickhouse local`. + """ + build_flags = getattr(args, "build_flags", None) + if build_flags is None: + return LLDB_SLOW_BUILD_TIMEOUT if is_slow_build_binary(args) else LLDB_TIMEOUT + slow_build = ( + any(flag in build_flags for flag in BuildFlags.SANITIZERS) + or BuildFlags.DEBUG in build_flags + or BuildFlags.WITH_COVERAGE in build_flags + ) + return LLDB_SLOW_BUILD_TIMEOUT if slow_build else LLDB_TIMEOUT + + +def print_c_stacktraces( + args, + per_pid_timeout: Optional[float] = None, + total_timeout: float = LLDB_SLOW_BUILD_TIMEOUT, +) -> None: """C++-side backtraces via lldb. Works whenever the server process exists and is debuggable, independent of whether SQL is reachable. Every call site is one of two situations: we're aborting the whole +<<<<<<< HEAD run (SERVER_DIED, liveness check, hung-check, check_server_started) or terminating the one wedged test (timeout_handler). lldb's "thread backtrace all" pause is on the order of seconds - acceptable in both cases against the diagnostic value. +======= + run (SERVER_DIED, hung-check, check_server_started) or terminating + the one wedged test (timeout_handler). + + `per_pid_timeout` None means derive it from the build flavor, which on a + server that never started costs a `clickhouse local` read of the binary. + `total_timeout` bounds the whole pid loop, which can span many + processes; it is an independent limit, not a multiple of the per-PID one. + A caller inside a fired one-shot `signal.alarm` must pass both, because + that alarm cannot bound work running in its own handler. +>>>>>>> 907703cf7f8 (Merge pull request #2385 from Altinity/fix/antalya-26.6/lldb-budget-and-sigcont) Skipped under ASan, where attaching a debugger disables LeakSanitizer detections. @@ -1425,10 +1561,23 @@ def print_c_stacktraces(args) -> None: ) return - for pid in all_pids: + if per_pid_timeout is None: + per_pid_timeout = lldb_timeout_for_build(args) + + deadline = monotonic() + total_timeout + for index, pid in enumerate(all_pids): + remaining = deadline - monotonic() + if remaining <= 0: + # Name the skipped pids: silence here reads as "nothing to dump". + print( + f"Ran out of the {total_timeout:.0f}s lldb budget: " + f"skipping {len(all_pids) - index} of {len(all_pids)} processes " + f"({', '.join(str(p) for p in all_pids[index:])})." + ) + break label = "main server" if pid == server_pid else "replica" print(f"\nCollecting C stacktraces from {label} process {pid} with lldb:") - bt = get_stacktraces_from_lldb(pid) + bt = get_stacktraces_from_lldb(pid, timeout=min(per_pid_timeout, remaining)) if bt and len(bt) >= 1000: block = f"=== PID {pid} ({label}) ===\n{bt}" _append_stacktrace_log(C_STACKTRACES_LOG, block) @@ -5029,7 +5178,12 @@ def run_tests_array( # stay around either way and can be captured at # end-of-run. print_sql_stacktraces(args) - print_c_stacktraces(args) + # Explicit budget: this runs from the fired one-shot + # alarm, which cannot bound its own handler, and + # stop_tests() below still needs the outer timeout. + print_c_stacktraces( + args, per_pid_timeout=LLDB_TIMEOUT, total_timeout=60 + ) stop_tests() raise TimeoutError("Test execution timed out") From d400cadd3f7ffdde393dcb61d1ede28a37089d36 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:47:12 +0200 Subject: [PATCH 14/15] Resolve conflicts in cherry-pick of #2385 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit tests/clickhouse-test: applied the PR's lldb budget and JIT-loader flag on top of antalya-26.8's get_stacktraces_from_lldb, which no longer has _ensure_lldb_installed and already attaches via sudo on macOS with keep_output_on_error; kept the base's liveness-check wording in the print_c_stacktraces docstring. ci/tests/test_print_stacktraces.py: kept the deletion from 3a69545dc04 on antalya-26.8 (ci/tests unit tests removed, no automated CI tests); restoring it would revive ~176 pre-existing lines outside the PR's diff. Dropped: ci/tests/test_print_stacktraces.py budget tests — file deleted on antalya-26.8 by 3a69545dc04 Source-PR: #2385 (https://github.com/Altinity/ClickHouse/pull/2385) --- ci/tests/test_print_stacktraces.py | 274 ----------------------------- tests/clickhouse-test | 95 ++-------- 2 files changed, 15 insertions(+), 354 deletions(-) delete mode 100644 ci/tests/test_print_stacktraces.py diff --git a/ci/tests/test_print_stacktraces.py b/ci/tests/test_print_stacktraces.py deleted file mode 100644 index 08a9518d11af..000000000000 --- a/ci/tests/test_print_stacktraces.py +++ /dev/null @@ -1,274 +0,0 @@ -""" -End-to-end tests for the stacktrace helpers in tests/clickhouse-test. - -Background ----------- -``clickhouse-test`` assigns ``args = parse_args()`` only inside -``if __name__ == "__main__":``. On macOS, Python's default -multiprocessing start method is ``spawn``, which re-imports the module -in each worker without executing ``__main__`` — so module-level -``args`` is undefined, and any helper that closed over it crashed with -``NameError``. See the fast_test_arm_darwin failure where the -hung-check path raised ``NameError: name 'args' is not defined`` inside -``get_server_pid``. - -These tests reproduce the same import condition by loading -``clickhouse-test`` via ``runpy.run_path`` (which, like spawn, does not -run ``__main__``) and then invoke each public stacktrace helper against -the live ClickHouse server provided by the ``ClickHouseService`` -fixture in ``ci/jobs/ci_tests_job.py``. - -Pre-fix: NameError inside the fresh import. -Post-fix: the helpers run to completion against a live server. -""" - -import argparse -import io -import os -import re -import runpy -from contextlib import redirect_stdout -from pathlib import Path -from time import sleep, time - -_REPO_ROOT = Path(__file__).resolve().parent.parent.parent -_CLICKHOUSE_TEST = str(_REPO_ROOT / "tests" / "clickhouse-test") - - -def _load_clickhouse_test(require_server=True): - # Mimic a spawn worker: load clickhouse-test without running __main__, - # so module-level `args` is absent. - ct = runpy.run_path(_CLICKHOUSE_TEST) - assert "args" not in ct, ( - "module-level 'args' must not be defined outside __main__; otherwise " - "the spawn-worker scenario this test reproduces does not apply" - ) - - if require_server: - # Sanity-check the precondition: the CI tests job started a server. - assert ct["pgrep"](command="clickhouse-server"), ( - "no clickhouse-server process found — this test expects ClickHouseService " - "(see ci/jobs/ci_tests_job.py) to be running on localhost:9000" - ) - return ct - - -def _make_args(): - # Minimal args namespace: only the fields the helpers and their - # transitive callees actually read. Mirrors what __main__ assigns - # after parse_args() for a local-server, plaintext-TCP, - # default-database run. - return argparse.Namespace( - client="clickhouse-client --port=9000", - client_option=None, - secure=False, - tcp_host="localhost", - http_port=8123, - client_options_query_str="", - replicated_database=False, - shared_catalog=False, - force_color=False, - binary=os.environ.get("CLICKHOUSE_BINARY", "clickhouse"), - # A reachable server means __main__ collected build flags at startup; - # a non-ASan set keeps print_c_stacktraces on its lldb path. - build_flags=set(), - ) - - -def _capture_lldb_budgets(ct, args, pids, elapsed=0.0, dump=None, **kwargs): - """Run print_c_stacktraces with the collector stubbed, returning the - per-PID timeouts it was called with (and its stdout).""" - budgets = [] - - def collector(pid, timeout=None): - budgets.append(timeout) - if elapsed: - sleep(elapsed) - return "x" * 2000 if dump is None else dump - - globals_ = ct["print_c_stacktraces"].__globals__ - saved = ( - globals_["get_stacktraces_from_lldb"], - globals_["get_all_server_pids"], - globals_["get_server_pid"], - globals_["is_asan_build"], - ) - globals_["get_stacktraces_from_lldb"] = collector - globals_["get_all_server_pids"] = lambda: list(pids) - globals_["get_server_pid"] = lambda: pids[0] - globals_["is_asan_build"] = lambda _args: False - captured = io.StringIO() - try: - with redirect_stdout(captured): - ct["print_c_stacktraces"](args, **kwargs) - finally: - ( - globals_["get_stacktraces_from_lldb"], - globals_["get_all_server_pids"], - globals_["get_server_pid"], - globals_["is_asan_build"], - ) = saved - return budgets, captured.getvalue() - - -def test_print_c_stacktraces_against_live_server(): - ct = _load_clickhouse_test() - args = _make_args() - - captured = io.StringIO() - with redirect_stdout(captured): - ct["print_c_stacktraces"](args) - output = captured.getvalue() - - # The function must have located the server PID and reached gdb. - # Whether the attach itself succeeds depends on the host's - # `kernel.yama.ptrace_scope` and is not asserted. - assert "Collecting C stacktraces from main server process" in output, output - - -def test_print_sql_stacktraces_against_live_server(): - ct = _load_clickhouse_test() - args = _make_args() - - captured = io.StringIO() - with redirect_stdout(captured): - ct["print_sql_stacktraces"](args) - output = captured.getvalue() - - # The function must have queried system.stack_trace and printed - # traces. We don't require a specific thread name — any non-trivial - # output confirms the round-trip succeeded. - assert "Collecting stacktraces from system.stack_trace table" in output, output - assert "trace_str" in output or "thread_name" in output, output - - -def test_lldb_budget_scales_with_build_flavor(): - # A debug or sanitizer or coverage server needs far longer than 30s to walk; - # a release server does not, and must keep the tight budget that bounds a - # genuinely wedged lldb. - ct = _load_clickhouse_test(require_server=False) - flags = ct["BuildFlags"] - release, slow = ct["LLDB_TIMEOUT"], ct["LLDB_SLOW_BUILD_TIMEOUT"] - assert slow > release, (slow, release) - - args = _make_args() - for build_flags, expected in ( - ({flags.RELEASE}, release), - ({flags.DEBUG}, slow), - ({flags.THREAD}, slow), - ({flags.MEMORY}, slow), - ({flags.UNDEFINED}, slow), - ({flags.WITH_COVERAGE}, slow), - ): - args.build_flags = build_flags - assert ct["lldb_timeout_for_build"](args) == expected, build_flags - # And the loop passes that value through. It is clamped to what is left - # of the aggregate ceiling, so compare with a tolerance rather than - # exactly. - budgets, _ = _capture_lldb_budgets(ct, args, [4242]) - assert len(budgets) == 1 and abs(budgets[0] - expected) < 1, ( - build_flags, - budgets, - expected, - ) - - -def test_lldb_budget_survives_the_spawn_start_method(): - # A spawned worker re-imports the module with RELEASE_NON_SANITIZED / - # SANITIZED back at their defaults while `args` is transferred intact, so - # the budget must come from args.build_flags, not from those globals. - ct = _load_clickhouse_test(require_server=False) - globals_ = ct["print_c_stacktraces"].__globals__ - saved = (globals_["RELEASE_NON_SANITIZED"], globals_["SANITIZED"]) - globals_["RELEASE_NON_SANITIZED"] = False - globals_["SANITIZED"] = False - try: - args = _make_args() - args.build_flags = {ct["BuildFlags"].DEBUG} - budgets, _ = _capture_lldb_budgets(ct, args, [4242]) - finally: - globals_["RELEASE_NON_SANITIZED"], globals_["SANITIZED"] = saved - - assert len(budgets) == 1, budgets - assert abs(budgets[0] - ct["LLDB_SLOW_BUILD_TIMEOUT"]) < 1, budgets - - -def test_lldb_budget_reads_the_binary_when_build_flags_are_missing(): - # The startup-failure caller (main -> check_server_started) runs before - # `args.build_flags` is assigned, so the flavor comes from the binary via - # `clickhouse local` rather than falling back to the release budget. - ct = _load_clickhouse_test(require_server=False) - args = _make_args() - delattr(args, "build_flags") - - for slow, expected_key in ((True, "LLDB_SLOW_BUILD_TIMEOUT"), (False, "LLDB_TIMEOUT")): - globals_ = ct["lldb_timeout_for_build"].__globals__ - saved = globals_["is_slow_build_binary"] - globals_["is_slow_build_binary"] = lambda _args, _s=slow: _s - try: - budgets, _ = _capture_lldb_budgets(ct, args, [4242]) - finally: - globals_["is_slow_build_binary"] = saved - - assert len(budgets) == 1, (slow, budgets) - assert abs(budgets[0] - ct[expected_key]) < 1, (slow, budgets) - - -def test_slow_build_binary_probe_is_server_independent_and_fails_closed(): - # The probe must not need a live server (it exists for the path where the - # server never came up), and an unreadable binary must yield the tighter - # budget rather than raising on an already-failing run. - ct = _load_clickhouse_test(require_server=False) - - args = _make_args() - args.binary = "/nonexistent/clickhouse-does-not-exist" - assert ct["is_slow_build_binary"](args) is False - - # And the query it issues names every flag collect_build_flags derives, so - # the two cannot drift into disagreeing about what "slow" means. - source = Path(_CLICKHOUSE_TEST).read_text(encoding="utf-8") - probe = source.split("def is_slow_build_binary")[1].split("\ndef ")[0] - for token in ("BUILD_TYPE", "Debug", "WITH_COVERAGE", "-fsanitize="): - assert token in probe, token - - -def test_lldb_pid_loop_honours_the_aggregate_deadline(): - # The loop spans every server process, so a per-PID budget alone does not - # bound it. Exhausting the total must stop the loop and say how many - # processes went undumped. - ct = _load_clickhouse_test(require_server=False) - args = _make_args() - args.build_flags = {ct["BuildFlags"].DEBUG} - pids = [111, 222, 333, 444] - - started = time() - budgets, output = _capture_lldb_budgets( - ct, args, pids, elapsed=1.0, per_pid_timeout=30, total_timeout=2 - ) - took = time() - started - - assert len(budgets) < len(pids), budgets - assert took < 30, took - # Each call is clamped to what is left, so no single attach can overrun the - # ceiling on its own. - assert all(b <= 2 for b in budgets), budgets - skipped = len(pids) - len(budgets) - assert f"skipping {skipped} of {len(pids)} processes" in output, output - assert str(pids[-1]) in output.split("skipping")[1], output - - -def test_timeout_handler_keeps_the_tight_lldb_pair(): - # The per-test timeout handler runs after its one-shot alarm has fired, so - # the alarm cannot bound it and only the job's outer timeout is left. That - # site therefore keeps the 30s per-PID value; the abort paths, where no - # alarm is pending, take the flavor budget. Asserted on the source so - # neither the outer-deadline risk nor the hung-check coverage can regress. - source = " ".join(Path(_CLICKHOUSE_TEST).read_text(encoding="utf-8").split()) - calls = re.findall(r"(?>>>>>> 907703cf7f8 (Merge pull request #2385 from Altinity/fix/antalya-26.6/lldb-budget-and-sigcont) + keep_output_on_error=True, ) @@ -1513,14 +1455,8 @@ def print_c_stacktraces( exists and is debuggable, independent of whether SQL is reachable. Every call site is one of two situations: we're aborting the whole -<<<<<<< HEAD run (SERVER_DIED, liveness check, hung-check, check_server_started) - or terminating the one wedged test (timeout_handler). lldb's "thread - backtrace all" pause is on the order of seconds - acceptable in both - cases against the diagnostic value. -======= - run (SERVER_DIED, hung-check, check_server_started) or terminating - the one wedged test (timeout_handler). + or terminating the one wedged test (timeout_handler). `per_pid_timeout` None means derive it from the build flavor, which on a server that never started costs a `clickhouse local` read of the binary. @@ -1528,7 +1464,6 @@ def print_c_stacktraces( processes; it is an independent limit, not a multiple of the per-PID one. A caller inside a fired one-shot `signal.alarm` must pass both, because that alarm cannot bound work running in its own handler. ->>>>>>> 907703cf7f8 (Merge pull request #2385 from Altinity/fix/antalya-26.6/lldb-budget-and-sigcont) Skipped under ASan, where attaching a debugger disables LeakSanitizer detections. From e3308519a372491e3e51531845fd63c12a1265bc Mon Sep 17 00:00:00 2001 From: filimonov <1549571+filimonov@users.noreply.github.com> Date: Thu, 17 Sep 2026 13:34:08 +0200 Subject: [PATCH 15/15] Merge pull request #2393 from Altinity/fix/antalya-26.6/no-json-probe-on-listing-keys Temporary workaround: do not probe object-storage listing keys as JSON Source-PR: #2393 (https://github.com/Altinity/ClickHouse/pull/2393) --- src/Access/Common/AccessFlags.h | 2 +- src/Access/Common/AccessType.h | 2 +- src/Core/ServerSettings.cpp | 2 + .../Backend/CasObjectStorageBackend.cpp | 1 + .../ContentAddressedMetadataStorage.cpp | 176 +++++++++--------- .../ContentAddressedTransaction.cpp | 28 +-- .../ContentAddressed/Pool/CasMountRuntime.cpp | 1 + .../ObjectStorages/IObjectStorage.cpp | 20 ++ src/Disks/tests/gtest_ca_transaction.cpp | 20 +- src/Disks/tests/gtest_cas_backend.cpp | 38 ++-- .../tests/gtest_cas_backend_generation.cpp | 43 ++++- .../tests/gtest_cas_bulk_delete_backend.cpp | 12 +- .../tests/gtest_cas_bulk_delete_engine.cpp | 4 +- .../tests/gtest_cas_confirm_exact_ref.cpp | 2 +- src/Disks/tests/gtest_cas_detached_work.cpp | 6 +- src/Disks/tests/gtest_cas_gc_ack_floor.cpp | 8 +- .../tests/gtest_cas_gc_frontier_gate.cpp | 12 +- src/Disks/tests/gtest_cas_gc_hold_grammar.cpp | 4 +- src/Disks/tests/gtest_cas_gc_key_reader.cpp | 2 +- .../gtest_cas_gc_manifest_bulk_delete.cpp | 4 +- src/Disks/tests/gtest_cas_gc_rebuild.cpp | 8 +- src/Disks/tests/gtest_cas_gc_round.cpp | 20 +- src/Disks/tests/gtest_cas_gc_round_defer.cpp | 12 +- src/Disks/tests/gtest_cas_hot_keys.cpp | 2 +- src/Disks/tests/gtest_cas_mount.cpp | 12 +- src/Disks/tests/gtest_cas_observability.cpp | 18 +- .../tests/gtest_cas_orphan_sweep_requests.cpp | 12 +- .../tests/gtest_cas_part_folder_access.cpp | 4 +- src/Disks/tests/gtest_cas_part_write.cpp | 28 +-- src/Disks/tests/gtest_cas_pool.cpp | 60 +++--- src/Disks/tests/gtest_cas_ref_catalog.cpp | 10 +- .../tests/gtest_cas_ref_chunked_flush.cpp | 8 +- .../tests/gtest_cas_ref_contiguous_alloc.cpp | 12 +- src/Disks/tests/gtest_cas_ref_gc.cpp | 28 +-- .../tests/gtest_cas_ref_install_safety.cpp | 8 +- .../tests/gtest_cas_ref_recovery_cas_walk.cpp | 2 +- ...test_cas_ref_snapshot_publish_ordering.cpp | 6 +- .../gtest_cas_ref_wedge_every_attempt.cpp | 36 ++-- src/Disks/tests/gtest_cas_ref_writer.cpp | 90 ++++----- src/Disks/tests/gtest_cas_repoint.cpp | 8 +- src/Disks/tests/gtest_cas_requests.cpp | 57 +++--- src/Disks/tests/gtest_cas_sentinel_probe.cpp | 12 ++ .../tests/gtest_cas_shutdown_context.cpp | 12 +- src/Disks/tests/gtest_cas_throttling_gate.cpp | 4 +- src/Disks/tests/gtest_cas_upload_detached.cpp | 8 +- src/Disks/tests/gtest_cas_upstream_slice.cpp | 5 +- src/IO/S3/tests/gtest_aws_s3_client.cpp | 6 +- src/IO/S3/tests/gtest_cas_aws_s3_client.cpp | 8 +- src/IO/tests/gtest_cas_writebuffer_s3.cpp | 5 +- src/IO/tests/gtest_writebuffer_s3.cpp | 37 +++- src/Storages/MergeTree/DataPartsExchange.cpp | 2 +- .../MergeTree/MergeFromLogEntryTask.cpp | 4 +- tests/integration/test_cas_gcs/test.py | 7 +- ...5015_cas_reject_fake_transaction.reference | 2 - .../05015_cas_reject_fake_transaction.sh | 28 --- .../05026_cas_manifest_path_newline.sh | 5 +- 56 files changed, 541 insertions(+), 432 deletions(-) delete mode 100644 tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference delete mode 100755 tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh diff --git a/src/Access/Common/AccessFlags.h b/src/Access/Common/AccessFlags.h index 4f793ec1e613..dc92fd1aa07a 100644 --- a/src/Access/Common/AccessFlags.h +++ b/src/Access/Common/AccessFlags.h @@ -144,7 +144,7 @@ class AccessFlags /// The same as allColumnFlags(). static AccessFlags allFlagsGrantableOnColumnLevel(); - static constexpr size_t SIZE = 256; + static constexpr size_t SIZE = 512; private: using Flags = std::bitset; Flags flags; diff --git a/src/Access/Common/AccessType.h b/src/Access/Common/AccessType.h index b03b14384f5e..1df7f90de63e 100644 --- a/src/Access/Common/AccessType.h +++ b/src/Access/Common/AccessType.h @@ -150,7 +150,7 @@ ENUM_ACCESS_OBJECT(Source, APPLY_FOR_SOURCE) /// Represents an access type which can be granted on databases, tables, columns, etc. -enum class AccessType : uint8_t +enum class AccessType : uint16_t { /// Macro M should be defined as M(name, aliases, node_type, parent_group_name) /// where name is identifier with underscores (instead of spaces); diff --git a/src/Core/ServerSettings.cpp b/src/Core/ServerSettings.cpp index 8136f3c869a1..fb7b95d77bb6 100644 --- a/src/Core/ServerSettings.cpp +++ b/src/Core/ServerSettings.cpp @@ -2238,6 +2238,8 @@ void ServerSettings::checkUnknownSettings(const Poco::Util::AbstractConfiguratio "distributed_cache_log", "distributed_cache_server_log", "instrumentation_trace_log", + "cas_log", + "cas_gc_log", "default_system_log_flush_policy", "create_union_system_log_tables", /// Legacy system log section names that older releases (or cloud deployments) read but the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp index 0589929d2912..b39a6b4e4a7b 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp @@ -13,6 +13,7 @@ #include #include #include +#include #include #include #include diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp index 8fb5a31b467c..4f9bb2805eac 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp @@ -496,74 +496,75 @@ Cas::GcRoundLogger ContentAddressedMetadataStorage::makeGcRoundLogger() const auto log = ctx->getContentAddressedGarbageCollectionLog(); if (!log) return; - ContentAddressedGarbageCollectionLogElement e; - const auto now = std::chrono::system_clock::now(); - e.event_time = std::chrono::system_clock::to_time_t(now); - e.event_time_microseconds = timeInMicroseconds(now); - switch (r.event_type) - { - case Cas::GcRoundLogRecord::EventType::Start: - e.event_type = ContentAddressedGarbageCollectionLogElement::START; - break; - case Cas::GcRoundLogRecord::EventType::Finish: - e.event_type = ContentAddressedGarbageCollectionLogElement::FINISH; - break; - case Cas::GcRoundLogRecord::EventType::Phase: - e.event_type = ContentAddressedGarbageCollectionLogElement::PHASE; - break; - } - e.disk_name = r.disk_name.empty() ? disk : r.disk_name; - e.srid = r.srid; - e.gc_id = r.gc_id; - e.trigger = r.trigger == Cas::GcRoundLogRecord::Trigger::Manual - ? ContentAddressedGarbageCollectionLogElement::MANUAL - : ContentAddressedGarbageCollectionLogElement::SCHEDULED; - switch (r.outcome) - { - case Cas::GcRoundLogRecord::Outcome::Unknown: - e.outcome = ContentAddressedGarbageCollectionLogElement::UNKNOWN; - break; - case Cas::GcRoundLogRecord::Outcome::Success: - e.outcome = ContentAddressedGarbageCollectionLogElement::SUCCESS; - break; - case Cas::GcRoundLogRecord::Outcome::NotALeader: - e.outcome = ContentAddressedGarbageCollectionLogElement::NOT_A_LEADER; - break; - case Cas::GcRoundLogRecord::Outcome::Failed: - e.outcome = ContentAddressedGarbageCollectionLogElement::FAILED; - break; - case Cas::GcRoundLogRecord::Outcome::Deferred: - e.outcome = ContentAddressedGarbageCollectionLogElement::DEFERRED; - break; - case Cas::GcRoundLogRecord::Outcome::Aborted: - e.outcome = ContentAddressedGarbageCollectionLogElement::ABORTED; - break; - case Cas::GcRoundLogRecord::Outcome::Stopped: - e.outcome = ContentAddressedGarbageCollectionLogElement::STOPPED; - break; - } - e.round = r.round; - e.candidates_marked = r.candidates_marked; - e.objects_deleted = r.objects_deleted; - e.objects_absent = r.objects_absent; - e.objects_replaced = r.objects_replaced; - e.objects_spared = r.objects_spared; - e.manifests_deleted = r.manifests_deleted; - e.entries_condemned = r.entries_condemned; - e.entries_graduated = r.entries_graduated; - e.entries_redeleted = r.entries_redeleted; - e.fence_outs = r.fence_outs; - e.anomalies = r.anomalies; - e.duration_ms = r.duration_ms; - e.error = r.error; - e.error_code = r.error_code; - e.profile_events = r.profile_events; - e.round_id = r.round_id; - e.phase = r.phase; - e.phase_duration_microseconds = r.phase_duration_microseconds; - e.phase_metrics = r.phase_metrics; /// Best-effort: SystemLog::add never blocks GC; a full queue drops the row with a warning. - log->add(std::move(e)); + log->add([&](ContentAddressedGarbageCollectionLogElement & e) + { + const auto now = std::chrono::system_clock::now(); + e.event_time = std::chrono::system_clock::to_time_t(now); + e.event_time_microseconds = timeInMicroseconds(now); + switch (r.event_type) + { + case Cas::GcRoundLogRecord::EventType::Start: + e.event_type = ContentAddressedGarbageCollectionLogElement::START; + break; + case Cas::GcRoundLogRecord::EventType::Finish: + e.event_type = ContentAddressedGarbageCollectionLogElement::FINISH; + break; + case Cas::GcRoundLogRecord::EventType::Phase: + e.event_type = ContentAddressedGarbageCollectionLogElement::PHASE; + break; + } + e.disk_name = r.disk_name.empty() ? disk : r.disk_name; + e.srid = r.srid; + e.gc_id = r.gc_id; + e.trigger = r.trigger == Cas::GcRoundLogRecord::Trigger::Manual + ? ContentAddressedGarbageCollectionLogElement::MANUAL + : ContentAddressedGarbageCollectionLogElement::SCHEDULED; + switch (r.outcome) + { + case Cas::GcRoundLogRecord::Outcome::Unknown: + e.outcome = ContentAddressedGarbageCollectionLogElement::UNKNOWN; + break; + case Cas::GcRoundLogRecord::Outcome::Success: + e.outcome = ContentAddressedGarbageCollectionLogElement::SUCCESS; + break; + case Cas::GcRoundLogRecord::Outcome::NotALeader: + e.outcome = ContentAddressedGarbageCollectionLogElement::NOT_A_LEADER; + break; + case Cas::GcRoundLogRecord::Outcome::Failed: + e.outcome = ContentAddressedGarbageCollectionLogElement::FAILED; + break; + case Cas::GcRoundLogRecord::Outcome::Deferred: + e.outcome = ContentAddressedGarbageCollectionLogElement::DEFERRED; + break; + case Cas::GcRoundLogRecord::Outcome::Aborted: + e.outcome = ContentAddressedGarbageCollectionLogElement::ABORTED; + break; + case Cas::GcRoundLogRecord::Outcome::Stopped: + e.outcome = ContentAddressedGarbageCollectionLogElement::STOPPED; + break; + } + e.round = r.round; + e.candidates_marked = r.candidates_marked; + e.objects_deleted = r.objects_deleted; + e.objects_absent = r.objects_absent; + e.objects_replaced = r.objects_replaced; + e.objects_spared = r.objects_spared; + e.manifests_deleted = r.manifests_deleted; + e.entries_condemned = r.entries_condemned; + e.entries_graduated = r.entries_graduated; + e.entries_redeleted = r.entries_redeleted; + e.fence_outs = r.fence_outs; + e.anomalies = r.anomalies; + e.duration_ms = r.duration_ms; + e.error = r.error; + e.error_code = r.error_code; + e.profile_events = r.profile_events; + e.round_id = r.round_id; + e.phase = r.phase; + e.phase_duration_microseconds = r.phase_duration_microseconds; + e.phase_metrics = r.phase_metrics; + }); }; } @@ -587,27 +588,28 @@ Cas::CasEventSink ContentAddressedMetadataStorage::makeCasEventSink() const auto log = ctx->getContentAddressedLog(); if (!log) return; - ContentAddressedLogElement e; - const auto now = std::chrono::system_clock::now(); - e.event_time = std::chrono::system_clock::to_time_t(now); - e.event_time_microseconds = timeInMicroseconds(now); - e.event_type = toString(ev.type); - e.disk_name = disk; - e.namespace_ = std::move(ev.namespace_); - e.ref_name = std::move(ev.ref_name); - e.object_kind = toString(ev.object_kind); - e.object_hash = std::move(ev.object_hash); - e.token = std::move(ev.token); - e.round = ev.round; - e.gen = ev.gen; - e.at_version = ev.at_version; - e.outcome = std::move(ev.outcome); - e.reason = std::move(ev.reason); - e.thread_id = getThreadId(); - e.query_id = CurrentThread::getQueryId(); - e.detail = std::move(ev.detail); /// Best-effort: SystemLog::add never blocks the Core; a full queue drops the row with a warning. - log->add(std::move(e)); + log->add([&](ContentAddressedLogElement & e) + { + const auto now = std::chrono::system_clock::now(); + e.event_time = std::chrono::system_clock::to_time_t(now); + e.event_time_microseconds = timeInMicroseconds(now); + e.event_type = toString(ev.type); + e.disk_name = disk; + e.namespace_ = std::move(ev.namespace_); + e.ref_name = std::move(ev.ref_name); + e.object_kind = toString(ev.object_kind); + e.object_hash = std::move(ev.object_hash); + e.token = std::move(ev.token); + e.round = ev.round; + e.gen = ev.gen; + e.at_version = ev.at_version; + e.outcome = std::move(ev.outcome); + e.reason = std::move(ev.reason); + e.thread_id = getThreadId(); + e.query_id = CurrentThread::getQueryId(); + e.detail = std::move(ev.detail); + }); }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp index 3a78e4a5eb3f..82cf883757e7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedTransaction.cpp @@ -8,7 +8,7 @@ #include #include #include -#include +#include #include #include #include @@ -796,10 +796,10 @@ std::unique_ptr ContentAddressedTransaction::tryCreateW throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Autocommit writes are not supported for content part files on a content-addressed disk"); - auto inner = writeFile(path, buf_size, mode, settings); - auto commit_callback = [owner](size_t) mutable { owner->commit(); }; - return std::make_unique( - std::move(inner), std::move(commit_callback), path, /*create_blob_if_empty=*/true); + auto create_inner = [this, path, buf_size, mode, settings] { return writeFile(path, buf_size, mode, settings); }; + auto commit_callback = [owner](FinalizeResult) mutable { owner->commit(); }; + return std::make_unique( + path, /*max_inline_bytes=*/0, /*create_blob_if_empty=*/true, std::move(create_inner), std::move(commit_callback), buf_size); } /// Non-autocommit (or verbatim autocommit): pin the owning disk transaction for the returned buffer's @@ -809,10 +809,10 @@ std::unique_ptr ContentAddressedTransaction::tryCreateW /// `owner` (which owns this ContentAddressedTransaction by shared_ptr) keeps that `this` valid until the /// buffer — and so this callback — is destroyed after finalize (the lifetime guarantee now /// expressed generically via `owner`). No cycle: the transaction does not hold the buffer. - auto inner = writeFile(path, buf_size, mode, settings); - auto keep_alive_callback = [owner](size_t) mutable {}; - return std::make_unique( - std::move(inner), std::move(keep_alive_callback), path, /*create_blob_if_empty=*/true); + auto create_inner = [this, path, buf_size, mode, settings] { return writeFile(path, buf_size, mode, settings); }; + auto keep_alive_callback = [owner](FinalizeResult) mutable {}; + return std::make_unique( + path, /*max_inline_bytes=*/0, /*create_blob_if_empty=*/true, std::move(create_inner), std::move(keep_alive_callback), buf_size); } std::unique_ptr ContentAddressedTransaction::writeFile( @@ -1831,10 +1831,10 @@ CaContentWriteBuffer::CaContentWriteBuffer( std::string temp_dir, Cas::BlobHashAlgo hash_algo, size_t buf_size, - bool use_adaptive_buffer_size, + bool use_adaptive_buffer_size_, size_t adaptive_buffer_initial_size, OnFinalized on_finalized_) - : WriteBufferFromFileBase(clampCasWriteBufferSize(use_adaptive_buffer_size ? adaptive_buffer_initial_size : buf_size), nullptr, 0) + : WriteBufferFromFileBase(clampCasWriteBufferSize(use_adaptive_buffer_size_ ? adaptive_buffer_initial_size : buf_size), nullptr, 0) , on_finalized(std::move(on_finalized_)) { fs::create_directories(temp_dir); @@ -1850,7 +1850,7 @@ CaContentWriteBuffer::CaContentWriteBuffer( /*mode=*/0666, /*existing_memory=*/nullptr, /*alignment=*/0, - use_adaptive_buffer_size, + use_adaptive_buffer_size_, clampCasWriteBufferSize(adaptive_buffer_initial_size)); hashing = Cas::makeBlobHashingWriteBuffer(hash_algo, *sink); } @@ -1861,11 +1861,11 @@ CaContentWriteBuffer::CaContentWriteBuffer( std::string envelope_header, Cas::BlobHashAlgo hash_algo, size_t buf_size, - bool use_adaptive_buffer_size, + bool use_adaptive_buffer_size_, size_t adaptive_buffer_initial_size, OnFinalized on_finalized_, std::function check_fence_before_finalize_) - : WriteBufferFromFileBase(clampCasWriteBufferSize(use_adaptive_buffer_size ? adaptive_buffer_initial_size : buf_size), nullptr, 0) + : WriteBufferFromFileBase(clampCasWriteBufferSize(use_adaptive_buffer_size_ ?adaptive_buffer_initial_size : buf_size), nullptr, 0) , on_finalized(std::move(on_finalized_)) , temp_path(std::move(object_key)) , is_s3_staging(true) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index e3f52fd7957a..ccf454c62cbd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp index fba16861e8e2..c763863d1a72 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.cpp @@ -162,6 +162,26 @@ RelativePathWithMetadata::RelativePathWithMetadata(const DataFileInfo & info, st RelativePathWithMetadata::CommandInTaskResponse::CommandInTaskResponse(const std::string & task) { + /// TEMPORARY WORKAROUND. This constructor runs for every string passed to `RelativePathWithMetadata`, + /// which includes every key returned by a `listObjects` / `iterate` call of every object storage, not only + /// the task-distributor answers it exists for (`{"retry_after_us": N}` from + /// `StorageObjectStorageStableTaskDistributor`). Parsing an ordinary object key as JSON throws and catches + /// one `JSONException` per listed key. Besides the cost, under ASan the fake stack frame of `parseImpl` + /// that exits by exception is never released (it is re-entered at the same stack depth, and `FakeStack::GC` + /// frees only frames strictly below the next allocation), so every listed key leaks one frame per thread; + /// once the size class is full every `__asan_stack_malloc_1` scans all 8192 slots and every small function + /// on that thread becomes ~100x slower. See https://github.com/Altinity/ClickHouse/issues/2362. + /// + /// Only try to parse strings that can be a JSON object. Object keys never start with `{`; the distributor + /// answer always does. The proper fix is to stop multiplexing the command into the path field (a separate + /// `ObjectInfo` kind or the versioned cluster-function protocol, see + /// https://github.com/Altinity/ClickHouse/pull/1360), after which this probe goes away entirely. + { + const auto first = task.find_first_not_of(" \t\r\n"); + if (first == std::string::npos || task[first] != '{') + return; + } + Poco::JSON::Parser parser; try { diff --git a/src/Disks/tests/gtest_ca_transaction.cpp b/src/Disks/tests/gtest_ca_transaction.cpp index ebf2311a8a11..1819648fec46 100644 --- a/src/Disks/tests/gtest_ca_transaction.cpp +++ b/src/Disks/tests/gtest_ca_transaction.cpp @@ -395,7 +395,7 @@ TEST(CASTransactionRepoint, StandaloneWriteOnCommittedPartRepoints) ASSERT_TRUE(storage->existsFile("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/checksums.txt")); ASSERT_TRUE(storage->existsFile("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/data.bin")); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; /// 2. New transaction: standalone write of checksums.txt onto the ALREADY-COMMITTED part. { @@ -409,7 +409,7 @@ TEST(CASTransactionRepoint, StandaloneWriteOnCommittedPartRepoints) EXPECT_EQ(storage->getFileSize("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/checksums.txt"), 20u); EXPECT_EQ(storage->getFileSize("b02/b02b02b0-0202-4202-8202-020202020202/all_1_1_0/data.bin"), 14u) << "carry-forward: the untouched file must survive a standalone write on the same part"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before + 1); const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); EXPECT_EQ(rep.dangling, 0u); @@ -438,7 +438,7 @@ TEST(CASTransactionRepoint, CombinedWriteAndUnlinkSameTxnRepointsOnce) } ASSERT_TRUE(storage->existsFile("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/txn_version.txt")); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; /// 2. ONE transaction: write checksums.txt (new bytes) AND unlink txn_version.txt (a DIFFERENT /// file of the same part) -- must resolve to exactly one repoint carrying both changes plus the @@ -456,7 +456,7 @@ TEST(CASTransactionRepoint, CombinedWriteAndUnlinkSameTxnRepointsOnce) EXPECT_FALSE(storage->existsFile("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/txn_version.txt")); EXPECT_EQ(storage->getFileSize("b03/b03b03b0-0303-4303-8303-030303030303/all_1_1_0/data.bin"), 14u) << "carry-forward: the untouched file must survive a combined write+unlink on the same part"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before + 1) << "one uncommitted transaction combining a write and an unlink must resolve to exactly one repoint"; const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); @@ -516,7 +516,7 @@ TEST(CASTransactionAllTree, CommittedTxnVersionStoreRepoints) } ASSERT_FALSE(storage->existsFile("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt")); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; /// 2. A single-op transaction writes ONLY txn_version.txt onto the already-committed part (mirrors /// the MVCC one-shot autocommit shape: no other file touched in this transaction). @@ -527,7 +527,7 @@ TEST(CASTransactionAllTree, CommittedTxnVersionStoreRepoints) } /// 3. Exactly one repoint; the new file is served; the original files are intact (carry-forward). - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before + 1); EXPECT_TRUE(storage->existsFile("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt")); EXPECT_EQ(storage->getFileSize("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/txn_version.txt"), 56u); EXPECT_EQ(storage->getFileSize("b05/b05b05b0-0505-4505-8505-050505050505/all_1_1_0/checksums.txt"), 8u); @@ -558,7 +558,7 @@ TEST(CASTransactionRemove, SurgicalUnlinkRepoints) } ASSERT_TRUE(storage->existsFile("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/txn_version.txt")); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; /// 2. A single-op transaction unlinks ONLY txn_version.txt on the already-committed part (mirrors /// ATTACH's removeVersionMetadata: no dir-drop in the same transaction). @@ -573,7 +573,7 @@ TEST(CASTransactionRemove, SurgicalUnlinkRepoints) EXPECT_FALSE(storage->existsFile("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/txn_version.txt")); EXPECT_EQ(storage->getFileSize("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/checksums.txt"), 8u); EXPECT_EQ(storage->getFileSize("b06/b06b06b0-0606-4606-8606-060606060606/all_1_1_0/data.bin"), 14u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before + 1); const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); EXPECT_EQ(rep.dangling, 0u); @@ -599,7 +599,7 @@ TEST(CASTransactionRemove, UnlinkStormThenDirDropIsOneRefDrop) } ASSERT_TRUE(storage->existsDirectory("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0")); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; /// 2. The MergeTree fast-removal shape (IMergeTreeDataPart::remove, B123): unlink every file /// one-by-one, THEN removeDirectory the part — all in one transaction. @@ -614,7 +614,7 @@ TEST(CASTransactionRemove, UnlinkStormThenDirDropIsOneRefDrop) /// 3. The whole part is gone via the single ref-drop; the storm of marks never repointed anything. EXPECT_FALSE(storage->existsDirectory("b07/b07b07b0-0707-4707-8707-070707070707/all_1_1_0")); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before) << "unlink-storm-then-dir-drop must supersede the marks, not repoint per file"; const auto rep = DB::Cas::runFsck(*storage->store(), /*detail*/false); diff --git a/src/Disks/tests/gtest_cas_backend.cpp b/src/Disks/tests/gtest_cas_backend.cpp index 04052060d7eb..266999ecc0ae 100644 --- a/src/Disks/tests/gtest_cas_backend.cpp +++ b/src/Disks/tests/gtest_cas_backend.cpp @@ -550,11 +550,11 @@ TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) CasOperation op = requests.admit(); using ProfileEvents::global_counters; - const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); - const auto blob_dedup_before = global_counters[ProfileEvents::CASBlobPutDeduplicated].load(); - const auto blob_head_before = global_counters[ProfileEvents::CASBlobHead].load(); - const auto blob_miss_before = global_counters[ProfileEvents::CASBlobHeadMiss].load(); - const auto gc_put_before = global_counters[ProfileEvents::CASGCPut].load(); + const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut]; + const auto blob_dedup_before = global_counters[ProfileEvents::CASBlobPutDeduplicated]; + const auto blob_head_before = global_counters[ProfileEvents::CASBlobHead]; + const auto blob_miss_before = global_counters[ProfileEvents::CASBlobHeadMiss]; + const auto gc_put_before = global_counters[ProfileEvents::CASGCPut]; const String blob_key = "pool/blobs/ab/abcdef0123456789"; @@ -571,11 +571,11 @@ TEST(CASInstrumentedBackend, ClassifierAndPerNamespaceOpEvents) /// Under coverage builds ProfileEvents propagate into a thread-local subtree that does not reach /// `global_counters`; deltas read 0 there only (see gtest_unique_key_index_cache). #if !WITH_COVERAGE - EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut].load() - blob_put_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASBlobPutDeduplicated].load() - blob_dedup_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASBlobHead].load() - blob_head_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASBlobHeadMiss].load() - blob_miss_before, 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASGCPut].load() - gc_put_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut] - blob_put_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPutDeduplicated] - blob_dedup_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobHead] - blob_head_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobHeadMiss] - blob_miss_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASGCPut] - gc_put_before, 1u); #else (void)blob_put_before; (void)blob_dedup_before; (void)blob_head_before; (void)blob_miss_before; (void)gc_put_before; @@ -590,7 +590,7 @@ TEST(CASInstrumentedBackend, PublishBlobDelegatesOnceAndRecordsOnePhysicalBlobWr CasOperation op = requests.admit(); using ProfileEvents::global_counters; - const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut].load(); + const auto blob_put_before = global_counters[ProfileEvents::CASBlobPut]; const auto request = streamingPublication("pool/blobs/ab/published", "fresh", "payload", 7); op.publish(request, Retry::once()); @@ -602,7 +602,7 @@ TEST(CASInstrumentedBackend, PublishBlobDelegatesOnceAndRecordsOnePhysicalBlobWr ASSERT_TRUE(published.has_value()); EXPECT_EQ(published->bytes, "freshpayload"); #if !WITH_COVERAGE - EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut].load() - blob_put_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::CASBlobPut] - blob_put_before, 1u); #else (void)blob_put_before; #endif @@ -1307,6 +1307,20 @@ class NativeReadThrowsNoSuchKeyObjectStorage final : public DB::LocalObjectStora return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); } + DB::SmallObjectDataWithMetadata readSmallObjectAndGetObjectMetadata( /// NOLINT + const DB::StoredObject & object, + const DB::ReadSettings & read_settings, + size_t max_size_bytes, + std::optional read_hint) const override + { + if (object.remote_path == throw_on_read_path) + throw DB::S3Exception( + "NoSuchKey: The specified key does not exist.", + Aws::S3::S3Errors::NO_SUCH_KEY); + + return DB::LocalObjectStorage::readSmallObjectAndGetObjectMetadata(object, read_settings, max_size_bytes, read_hint); + } + private: std::string throw_on_read_path; }; diff --git a/src/Disks/tests/gtest_cas_backend_generation.cpp b/src/Disks/tests/gtest_cas_backend_generation.cpp index 77c3d0a2d647..0081b55b97d0 100644 --- a/src/Disks/tests/gtest_cas_backend_generation.cpp +++ b/src/Disks/tests/gtest_cas_backend_generation.cpp @@ -1,5 +1,6 @@ #include #include +#include #include #include #include @@ -127,6 +128,40 @@ std::shared_ptr makeVersioningObjectStorageForTest(std: return std::make_shared(std::move(settings), versioned); } +/// A `LocalObjectStorage` whose native token is the object's mtime in nanoseconds (a valid generation +/// value), taken from the `.__` local etag. +class GenerationTokenObjectStorage : public DB::LocalObjectStorage +{ +public: + using DB::LocalObjectStorage::LocalObjectStorage; + + std::optional tryGetObjectMetadataWithNativeToken(const std::string & path, bool with_tags) const override + { + auto metadata = DB::LocalObjectStorage::tryGetObjectMetadata(path, with_tags); + if (metadata) + { + String mtime = metadata->etag.substr(0, metadata->etag.find('_')); + std::erase(mtime, '.'); + metadata->etag = mtime; + } + return metadata; + } +}; + +std::shared_ptr makeGenerationTokenObjectStorageForTest() +{ + static std::atomic counter{0}; + const auto unique = std::to_string(::getpid()) + "_" + std::to_string(counter.fetch_add(1)); + const auto root = (std::filesystem::temp_directory_path() / ("cas_unit_generation_token_" + unique)).string(); + + std::error_code ec; + std::filesystem::remove_all(root, ec); + std::filesystem::create_directories(root, ec); + + DB::LocalObjectStorageSettings settings("test", root, /*read_only_=*/false); + return std::make_shared(std::move(settings)); +} + /// Captures what `ObjectStorageBackend` logs at WARNING and above, so a test can assert both that a /// warning was raised and that none was. Same shape as the capture in gtest_cas_settings.cpp. class ScopedBackendLogCapture @@ -213,11 +248,10 @@ TEST(CASBackendGeneration, NativeHeadUsesNativeTokenMetadataApi) /// S3 client below, which is the only place a Native write can produce a response incarnation. TEST(CASBackendGeneration, StampedTokenTypeFollowsNativeKind) { - auto storage = DB::Cas::tests::makeLocalObjectStorageForTest(); + auto storage = makeGenerationTokenObjectStorageForTest(); auto b = std::make_shared(storage, ObjectStorageBackend::Mode::Native); b->setNativeTokenTypeForTest(Dialect::Generation); - /// A local file's etag is its mtime in nanoseconds, which is also a valid generation value. const String key = DB::Cas::tests::nativeKeyUnder(storage, "p/gen/tok"); { auto out = storage->writeObject(DB::StoredObject(key), DB::WriteMode::Rewrite); @@ -407,7 +441,7 @@ class FakeGenerationS3Client : public DB::S3::Client static DB::S3::PocoHTTPClientConfiguration GetClientConfiguration() { DB::RemoteHostFilter remote_host_filter; - return DB::S3::ClientFactory::instance().createClientConfiguration( + auto configuration = DB::S3::ClientFactory::instance().createClientConfiguration( "some-region", remote_host_filter, /* s3_max_redirects = */ 100, @@ -418,6 +452,9 @@ class FakeGenerationS3Client : public DB::S3::Client /* for_disk_s3 = */ false, /* opt_disk_name = */ {}, /* request_throttler = */ {}); + /// The client is built directly, bypassing ClientFactory::create(), which normally fills retryStrategy. + configuration.retryStrategy = std::make_shared(configuration.retry_strategy); + return configuration; } /// The response ETag/generation the NEXT successful PutObject returns; empty means the response diff --git a/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp index 68aeabeb9523..3fd4cbdf047a 100644 --- a/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp +++ b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp @@ -126,17 +126,17 @@ TEST(CASBulkDeleteBackend, InstrumentedCountsOneRequestAndOneDeletePerKeyClass) const WriteOnceKey log = kLayout.writeOnceRefLogKey(life, RefTxnId{1, 1}); ASSERT_TRUE(std::holds_alternative(op.create(log.str(), "log", Retry::once()))); - const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load(); - const auto manifest_before = ProfileEvents::global_counters[ProfileEvents::CASManifestDelete].load(); - const auto root_before = ProfileEvents::global_counters[ProfileEvents::CASRootDelete].load(); + const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests]; + const auto manifest_before = ProfileEvents::global_counters[ProfileEvents::CASManifestDelete]; + const auto root_before = ProfileEvents::global_counters[ProfileEvents::CASRootDelete]; std::vector batch = keys.present; batch.push_back(log); op.removeManyWriteOnce(batch, Retry::once()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load() - requests_before, 1u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASManifestDelete].load() - manifest_before, 3u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRootDelete].load() - root_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests] - requests_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASManifestDelete] - manifest_before, 3u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRootDelete] - root_before, 1u); } #if USE_AWS_S3 diff --git a/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp b/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp index fc84d67ebc61..01ace5ad336c 100644 --- a/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp +++ b/src/Disks/tests/gtest_cas_bulk_delete_engine.cpp @@ -51,12 +51,12 @@ TEST(CASBulkDeleteEngine, AFailedAttemptReissuesTheWholeChunkAndAlreadyDeletedKe ASSERT_EQ(op.remove(keys[0].str(), h->etag, Retry::once()), Removal::Removed); } backend->failNextBulkRemoveWith(std::make_exception_ptr(Poco::TimeoutException("injected"))); - const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue]; op.removeManyWriteOnce(keys, Retry::standard()); EXPECT_EQ(backend->bulkRemoveCalls(), 2u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue] - reissues_before, 1u); for (const WriteOnceKey & key : keys) EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); } diff --git a/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp index c2a40218c0a7..cd0fee2311d9 100644 --- a/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp +++ b/src/Disks/tests/gtest_cas_confirm_exact_ref.cpp @@ -337,7 +337,7 @@ uint64_t backendRequests(const CountingBackend & b) /// attribution as a DELTA around the confirm, never an absolute (the suite shares one process). uint64_t refusalCount(ProfileEvents::Event event) { - return ProfileEvents::global_counters[event].load(); + return ProfileEvents::global_counters[event]; } /// One-shot throwing probe in the post-durable install region -- the only way to reach `NeedsRecovery` diff --git a/src/Disks/tests/gtest_cas_detached_work.cpp b/src/Disks/tests/gtest_cas_detached_work.cpp index 2b569de9eb82..250458a5876d 100644 --- a/src/Disks/tests/gtest_cas_detached_work.cpp +++ b/src/Disks/tests/gtest_cas_detached_work.cpp @@ -518,11 +518,9 @@ TEST(CASDetachedWork, ExpiredDrainIncrementsTheTimeoutCounter) })); entered->wait("entered"); - const auto before = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts] - .load(std::memory_order_relaxed); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts]; storage->shutdown(); - const auto after = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts] - .load(std::memory_order_relaxed); + const auto after = ProfileEvents::global_counters[ProfileEvents::CASDetachedWorkDrainTimeouts]; EXPECT_EQ(after - before, 1u); release->open(); diff --git a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp index ea63268a8fe3..4f190d1dbbbb 100644 --- a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp +++ b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp @@ -391,14 +391,14 @@ TEST(CASGCRetire, DeleteRemovesBodyAndMeta) gc.runRegularRound(); dropRefTransition(*backend, store->layout(), ns, "tbl", r); /// §0 introspection: the meta drop below rides `deleteMetaExact` (`CASMetaDelete` choke point). - const auto delete_before = ProfileEvents::global_counters[ProfileEvents::CASMetaDelete].load(); + const auto delete_before = ProfileEvents::global_counters[ProfileEvents::CASMetaDelete]; // condemn -> graduate (round-paced) -> delete (the retired-cursor pipeline). ASSERT_TRUE(runRoundsUntilAbsent(store, gc, *backend, store->layout(), DB::UInt128(1))); EXPECT_FALSE(blobExists(*backend, store->layout(), DB::UInt128(1))) << "body gone via exact-token delete"; EXPECT_FALSE(loadMetaForTest(*backend, store->layout(), DB::UInt128(1)).has_value()) << "the meta must be dropped alongside the exact-token body delete (Task 5)"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaDelete].load() - delete_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaDelete] - delete_before, 1); } /// GC freshness meta is ADD-ONLY (spec 2026-07-11 deposed-leader `clearSparedMeta` fix): an entry whose @@ -1344,7 +1344,7 @@ TEST(CASGCCondemnMarker, SwallowedMarkerWriteCarriesEntryInsteadOfDeleting) /// Rounds keep coming while the marker stays unwritable: without durable Condemned evidence the /// entry must be CARRIED — a writer reading the absent meta may have adopted this exact token. const auto carries_before = - ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry].load(); + ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry]; for (int i = 0; i < 4; ++i) { runRegularRoundReclaiming(gc); @@ -1356,7 +1356,7 @@ TEST(CASGCCondemnMarker, SwallowedMarkerWriteCarriesEntryInsteadOfDeleting) ASSERT_TRUE(e.has_value()) << "the entry must remain retired (carried), not dropped"; EXPECT_FALSE(e->delete_pending) << "graduation must be refused without a confirmed marker"; EXPECT_FALSE(e->marker_confirmed); - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry].load() + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCCondemnMarkerUnconfirmedCarry] - carries_before, 4u) << "every refused graduation must count one unconfirmed carry"; diff --git a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp index d8c90fa7dc1c..6712ec60aa47 100644 --- a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp +++ b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp @@ -2883,7 +2883,7 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro std::map namespace_cleanup; const uint64_t leaks_before - = ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks]; gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") @@ -2906,7 +2906,7 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro ASSERT_FALSE(namespace_cleanup.empty()); EXPECT_EQ(namespace_cleanup["leaked"], 1u); EXPECT_EQ( - ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks].load() - leaks_before, + ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks] - leaks_before, 1u); const String captured = log_capture.captured(); EXPECT_NE(captured.find(terminal_key), String::npos); @@ -2942,12 +2942,12 @@ TEST(CASGCFrontierGate, UnmatchedAdoptedParentLifeDoesNotSuppressAuthoritativeDe dropRefTransition(*backend, layout, ns, "victim", mref); const uint64_t events_before = - ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load(); + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives]; const RoundReport report = runRegularRoundReclaiming(gc); ASSERT_TRUE(report.acquired_lease); EXPECT_EQ( - ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load() - events_before, + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives] - events_before, 1u); EXPECT_EQ(report.manifests_deleted, 1u) << "an unmatched adopted-parent row is observed and dropped, not promoted to pool-wide suppression"; @@ -3533,7 +3533,7 @@ TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrRepl backend->clearJournal(); const uint64_t plans_before /// NOLINT(clang-analyzer-deadcode.DeadStores) - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; backend->releaseBlockedCatalogCas(); leader_a.join(); @@ -3549,7 +3549,7 @@ TEST_P(CASGCCompletedRemovalFenceRace, FencedLeaderStopsAfterWinnerRemovesOrRepl EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR); EXPECT_NE(e.message().find("pre-fold drain lost authority"), String::npos) << e.message(); } - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - plans_before, 0u); EXPECT_EQ(findJournalAfter(journal, "list " + layout.casRefsPrefix(), 0), journal.size()); EXPECT_EQ(findJournalAfter(journal, "cas_begin " + layout.gcStateKey(), 0), journal.size()); EXPECT_FALSE(std::any_of(journal.begin(), journal.end(), [](const String & entry) diff --git a/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp index 3d67ac02f317..f002eb160615 100644 --- a/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp +++ b/src/Disks/tests/gtest_cas_gc_hold_grammar.cpp @@ -1621,11 +1621,11 @@ TEST(CASGCHoldGrammar, RebuildProceedsOnAPoolThatNeverSealedABaselineAndCountsTh ASSERT_FALSE(existsAt(*backend, layout.gcStateKey())); using ProfileEvents::global_counters; - const auto virgin_before = global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration].load(); + const auto virgin_before = global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration]; Gc gc(store, kGc); const RebuildReport rep = gc.rebuildBaseline(/*force=*/false); EXPECT_TRUE(rep.performed) << rep.refusal; - EXPECT_GT(global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration].load(), virgin_before) + EXPECT_GT(global_counters[ProfileEvents::CASGCRebuildVirginByEnumeration], virgin_before) << "a clean slate granted from enumeration alone must be visible to whoever reads the run"; } diff --git a/src/Disks/tests/gtest_cas_gc_key_reader.cpp b/src/Disks/tests/gtest_cas_gc_key_reader.cpp index 4f3938f01698..d01381e47c48 100644 --- a/src/Disks/tests/gtest_cas_gc_key_reader.cpp +++ b/src/Disks/tests/gtest_cas_gc_key_reader.cpp @@ -50,7 +50,7 @@ struct Rig uint64_t wasted() { - return ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load(); + return ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted]; } } diff --git a/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp index 3a7c2ca3ba7f..f0148005fc03 100644 --- a/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp +++ b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp @@ -82,14 +82,14 @@ TEST(CASGCManifestBulkDelete, FiveBodiesInChunksOfTwoAreThreeRequests) auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_bulk_delete_chunk_keys = 2, .gc_fold_max_defer_rounds = 0}); const auto ids = seedDroppedManifests(*backend, store->layout(), 5); - const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load(); + const auto requests_before = ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests]; Gc gc(store, kGc); const uint64_t deleted = reclaim(gc, store, *backend, ids, 16); EXPECT_EQ(deleted, 5u); EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "2 + 2 + 1"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests].load() - requests_before, 3u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBulkDeleteRequests] - requests_before, 3u); OperationForTest op(*backend); for (const ManifestId & id : ids) EXPECT_FALSE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); diff --git a/src/Disks/tests/gtest_cas_gc_rebuild.cpp b/src/Disks/tests/gtest_cas_gc_rebuild.cpp index cbfee9d0d5e5..9bcb8c99682a 100644 --- a/src/Disks/tests/gtest_cas_gc_rebuild.cpp +++ b/src/Disks/tests/gtest_cas_gc_rebuild.cpp @@ -245,10 +245,10 @@ TEST(CASGCRebuild, HealthyStateRequiresForce) backend->resetCounts(); const uint64_t plans_before - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; const RebuildReport forced = gc.rebuildBaseline(/*force*/ true); ASSERT_TRUE(forced.performed) << forced.refusal; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - plans_before, 1u) << "healthy FORCE REBUILD must share the same one-shot authoritative plan builder"; EXPECT_EQ(backend->getCount(store->layout().refCatalogKey()), 2u) << "healthy FORCE REBUILD may read the catalog for the conclusive drain and the one post-LIST cut only"; @@ -413,12 +413,12 @@ TEST(CASGCRebuild, DamagedGenerationZeroStatePerformsNoCatalogDrainMutation) .last_epoch_seal = std::nullopt})); const uint64_t catalog_cas_before = backend->putOverwriteCount(layout.refCatalogKey()); const uint64_t plans_before - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; Gc gc(store, kGc); const RebuildReport report = gc.rebuildBaseline(/*force*/ false); ASSERT_TRUE(report.performed) << report.refusal; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - plans_before, 1u); EXPECT_EQ(backend->putOverwriteCount(layout.refCatalogKey()), catalog_cas_before); const CasRefCatalog::Snapshot catalog = CasRefCatalog::read(op, layout); ASSERT_EQ(catalog.catalog.entries.size(), 1u); diff --git a/src/Disks/tests/gtest_cas_gc_round.cpp b/src/Disks/tests/gtest_cas_gc_round.cpp index eeaad2c7b86f..b3a40f4d4394 100644 --- a/src/Disks/tests/gtest_cas_gc_round.cpp +++ b/src/Disks/tests/gtest_cas_gc_round.cpp @@ -859,8 +859,8 @@ TEST(CASGCRound, RoundSummaryCountsManifestBodyDeletes) /// §0 introspection: both counters are captured BEFORE the condemn+delete pipeline below, which /// drives the round's meta pool (condemn/spare/delete) and its own orphan-sweep cursor pass. - const auto meta_ops_before = ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps].load(); - const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load(); + const auto meta_ops_before = ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps]; + const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages]; /// Ack-floor drift: the owner-removed manifest body is deleted in the CONDEMNING round (post-CAS, /// after its -1 is adopted), while the blob's exact-token delete happens a few rounds later once the @@ -893,8 +893,8 @@ TEST(CASGCRound, RoundSummaryCountsManifestBodyDeletes) /// §0 introspection: the exact-token blob delete above scheduled at least one per-hash freshness-meta /// op on the round's bounded meta pool, and every round ran its own orphan-manifest-sweep cursor pass /// (default `manifest_sweep_list_budget_keys` is nonzero), fetching at least one LIST page directly. - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps].load() - meta_ops_before, 1); - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load() - pages_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCMetaOps] - meta_ops_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages] - pages_before, 1); } /// Manifest-body cleanup (post-CAS `manifest_deletes` phase) has no cap: the ref-log intake cursor that @@ -955,9 +955,9 @@ TEST(CASGCRound, EnumerationPagesCountedEvenWithSweepBudgetZeroed) auto store = openTestPoolWithConfig(backend, config); Gc gc(store, kGc); - const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load(); + const auto pages_before = ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages]; ASSERT_TRUE(gc.runRegularRound().acquired_lease); - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages].load() - pages_before, 1) + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASGCEnumerationPages] - pages_before, 1) << "the round's own cas/ns/stream/ enumeration must count pages independent of " "the orphan sweep"; } @@ -1083,7 +1083,7 @@ TEST(CASGCRound, OutcomeEntryBudgetCapsSparedLogRowsWithoutRecondemning) writeManifestRaw(*backend, store->layout(), ns, r2, entries); publishCommittedTransition(*backend, store->layout(), ns, "tbl2", std::nullopt, r2); - const auto spared_events_before = ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared].load(); + const auto spared_events_before = ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared]; const RoundReport rep = runRegularRoundReclaiming(gc); ASSERT_TRUE(rep.acquired_lease); const uint64_t total_spared_reported = rep.spared; @@ -1092,7 +1092,7 @@ TEST(CASGCRound, OutcomeEntryBudgetCapsSparedLogRowsWithoutRecondemning) /// decision it under-reports still happened correctly. EXPECT_EQ(total_spared_reported, 2u) << "GcOutcomes rows must be capped at gc_round_outcome_entry_budget, not one per spared entry"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared].load() - spared_events_before, kBlobs) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRetiredSpared] - spared_events_before, kBlobs) << "every spared decision must still happen even when its audit row is capped"; for (int i = 0; i < kBlobs; ++i) { @@ -1990,11 +1990,11 @@ TEST(CASGCRound, OrphanManifestCursorSweepDeletesAndPersistsCursor) const auto occupant_before = readOf(*foreign_backend, foreign_mount_key); ASSERT_TRUE(occupant_before.has_value()); const uint64_t violations_before - = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation]; invalid_store.reset(); /// must not abort, must not terminate - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation], violations_before + 1) << "a runtime that never observed a deposition must report the foreign occupant as a broken " "single-writer guarantee"; diff --git a/src/Disks/tests/gtest_cas_gc_round_defer.cpp b/src/Disks/tests/gtest_cas_gc_round_defer.cpp index 43a6a909e86a..677982e4fe5f 100644 --- a/src/Disks/tests/gtest_cas_gc_round_defer.cpp +++ b/src/Disks/tests/gtest_cas_gc_round_defer.cpp @@ -326,9 +326,9 @@ TEST(CASGCRoundDefer, FoldAndDeferEachBuildExactlyOneCompletePostListWalkPlan) backend->resetCounts(); const uint64_t fold_builds_before - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; ASSERT_FALSE(gc.runRegularRound().deferred); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - fold_builds_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - fold_builds_before, 1u); EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) << "the hot walk must enumerate the stream tree exactly once"; EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) @@ -353,9 +353,9 @@ TEST(CASGCRoundDefer, FoldAndDeferEachBuildExactlyOneCompletePostListWalkPlan) phases.clear(); backend->resetCounts(); const uint64_t defer_builds_before - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; ASSERT_TRUE(gc.runRegularRound().deferred); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - defer_builds_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - defer_builds_before, 1u); EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u) << "a deferred round still builds exactly one complete hot walk plan"; EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) @@ -429,13 +429,13 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP Gc gc(store, kGc); gc.setPhaseSink([&](const GcPhaseRecord & phase) { phases.push_back(phase); }); const uint64_t plans_before - = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt]; const RoundReport report = gc.runRegularRound(); ASSERT_TRUE(report.acquired_lease); ASSERT_TRUE(report.deferred); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt].load() - plans_before, 1u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCRefWalkPlansBuilt] - plans_before, 1u) << "DEFER still constructs its one immutable hot walk plan, never a second janitor-derived plan"; EXPECT_EQ(backend->listCount(layout.namespaceStreamRootPrefix()), 1u); EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) diff --git a/src/Disks/tests/gtest_cas_hot_keys.cpp b/src/Disks/tests/gtest_cas_hot_keys.cpp index 6dc4d5426383..c0bf6bf4d555 100644 --- a/src/Disks/tests/gtest_cas_hot_keys.cpp +++ b/src/Disks/tests/gtest_cas_hot_keys.cpp @@ -74,7 +74,7 @@ CasHotKeys::Decide appendTicket(int ticket) uint64_t counter(ProfileEvents::Event event) { - return ProfileEvents::global_counters[event].load(); + return ProfileEvents::global_counters[event]; } /// A one-shot gate a write hook parks on: the first write of the key waits here until the test diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index a9472994237f..f961b0991802 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -805,7 +805,7 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) k.start(); const String mount_key = l.mountKey("r"); - const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); /// NOLINT(clang-analyzer-deadcode.DeadStores) + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost]; /// NOLINT(clang-analyzer-deadcode.DeadStores) /// Simulate `rm -rf` of the backing store: the mount slot object is gone, but the renewer still /// names a (now stale) incarnation as its precondition. @@ -821,7 +821,7 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) EXPECT_EQ(e.code(), DB::ErrorCodes::FILE_DOESNT_EXIST) << e.message(); EXPECT_NE(e.code(), DB::ErrorCodes::LOGICAL_ERROR); } - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost], lost_before) << "renewer classification is metric-free; the runtime records operational loss"; } @@ -847,7 +847,7 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) k.start(); const String mount_key = l.mountKey("r"); - const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost]; /// Simulate `rm -rf` of the backing store: the mount slot object is gone before we ever attempt /// a renewal, so the farewell's guarded write is the first thing to observe it. @@ -855,7 +855,7 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) EXPECT_NO_THROW(k.release()) << "clean release against a vanished store must be a no-op, not a LOGICAL_ERROR abort"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost], lost_before); } /// rev.6: a bare `claimMount` (no `proven_dead_incarnation`) NEVER reclaims a same-uuid, different-epoch @@ -1452,11 +1452,11 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) const auto reclaimer_slot_before = overlap_ops.op.read(overlap_mount_key, Retry::standard()); ASSERT_TRUE(reclaimer_slot_before.has_value()); const uint64_t overlap_violations_before - = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation]; first.reset(); /// must not abort, must not terminate - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation], overlap_violations_before + 1); const auto reclaimer_slot_after = overlap_ops.op.read(overlap_mount_key, Retry::standard()); ASSERT_TRUE(reclaimer_slot_after.has_value()); diff --git a/src/Disks/tests/gtest_cas_observability.cpp b/src/Disks/tests/gtest_cas_observability.cpp index 2521aae58294..c847816756ea 100644 --- a/src/Disks/tests/gtest_cas_observability.cpp +++ b/src/Disks/tests/gtest_cas_observability.cpp @@ -98,11 +98,11 @@ RenewalCounterSnapshot renewalCounters() { using ProfileEvents::global_counters; return { - .attempts = global_counters[ProfileEvents::CASMountRenewalAttempts].load(), - .retries = global_counters[ProfileEvents::CASMountRenewalRetries].load(), - .resolved = global_counters[ProfileEvents::CASMountRenewalResolved].load(), - .recovered = global_counters[ProfileEvents::CASMountRenewalRecovered].load(), - .deadline_exceeded = global_counters[ProfileEvents::CASMountRenewalDeadlineExceeded].load(), + .attempts = global_counters[ProfileEvents::CASMountRenewalAttempts], + .retries = global_counters[ProfileEvents::CASMountRenewalRetries], + .resolved = global_counters[ProfileEvents::CASMountRenewalResolved], + .recovered = global_counters[ProfileEvents::CASMountRenewalRecovered], + .deadline_exceeded = global_counters[ProfileEvents::CASMountRenewalDeadlineExceeded], }; } @@ -375,8 +375,8 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) /// with a fresh condemn of B (peek, not the fresh-condemn `head_blob` hook). Capture events + the /// counters for exactly THIS round. using ProfileEvents::global_counters; - const auto condemned_before = global_counters[ProfileEvents::CASGCRetiredCondemned].load(); - const auto replaced_before = global_counters[ProfileEvents::CASGCRetireReplaced].load(); + const auto condemned_before = global_counters[ProfileEvents::CASGCRetiredCondemned]; + const auto replaced_before = global_counters[ProfileEvents::CASGCRetireReplaced]; s->setEventSink([seen](const CasEvent & e) { @@ -386,8 +386,8 @@ TEST(CASObservability, ResurrectSupersedeEmitsOnlyRetireReplacedWithOldToken) s->setEventSink(nullptr); ASSERT_TRUE(rep.acquired_lease); - const auto condemned_after = global_counters[ProfileEvents::CASGCRetiredCondemned].load(); - const auto replaced_after = global_counters[ProfileEvents::CASGCRetireReplaced].load(); + const auto condemned_after = global_counters[ProfileEvents::CASGCRetiredCondemned]; + const auto replaced_after = global_counters[ProfileEvents::CASGCRetireReplaced]; /// Phase 3 (mixed-algo pools): event `object_hash` renders are `blobIdOf(ref)` (":"), /// never a bare hex. diff --git a/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp b/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp index 34e87af705ed..168d09683d3b 100644 --- a/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp +++ b/src/Disks/tests/gtest_cas_orphan_sweep_requests.cpp @@ -327,10 +327,10 @@ TEST(CASOrphanSweepRequests, PageIsIdenticalInlineAndWithReadAhead) CandidateFixture ahead_f; ThreadPool pool = makeReadPool(4); - const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load(); + const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit]; const ManifestSweepResult ahead_r = planManifestCursorPage( *ahead_f.store, "", 1000, 100, true, nullptr, &pool, /*read_concurrency=*/16); - const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load() - hits_before; + const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit] - hits_before; const auto ahead_gets = getsOf(*ahead_f.backend); EXPECT_GT(hits, 0u) << "the fixture's committed-tail walk and candidates must actually hit the " @@ -366,12 +366,12 @@ TEST(CASOrphanSweepRequests, EpochCrossingDiscardsAtMostOneWindowAndTheNewEpochI setWatermarkMinActive(*backend, layout, "test", 2, /*min_active=*/1000); backend->resetCounts(); - const uint64_t wasted_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load(); - const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load(); + const uint64_t wasted_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted]; + const uint64_t hits_before = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit]; ThreadPool pool = makeReadPool(4); const ManifestSweepResult result = planManifestCursorPage(*store, "", 1000, 100, true, nullptr, &pool, 16); - const uint64_t wasted = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted].load() - wasted_before; - const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit].load() - hits_before; + const uint64_t wasted = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadWasted] - wasted_before; + const uint64_t hits = ProfileEvents::global_counters[ProfileEvents::CASGCReadAheadHit] - hits_before; /// `publishAt` writes a manifest as a side effect of every call above, so the page's LIST sees /// every one of those (47) plus the one raw manifest -- the fixture is about the hint/discard diff --git a/src/Disks/tests/gtest_cas_part_folder_access.cpp b/src/Disks/tests/gtest_cas_part_folder_access.cpp index b3215cff2216..c87ea8e8c954 100644 --- a/src/Disks/tests/gtest_cas_part_folder_access.cpp +++ b/src/Disks/tests/gtest_cas_part_folder_access.cpp @@ -1035,10 +1035,10 @@ TEST(CASPartFolderAccess, BestEffortRollbackDropCountsAndSurvivesABackendOutage) EXPECT_ANY_THROW(store->dropRef(ns_a, "part_a")); using ProfileEvents::global_counters; - const auto before = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed].load(); + const auto before = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed]; /// The compensating-rollback path must NOT throw (noexcept) and MUST record the swallowed failure. access.dropRefBestEffort(Cas::PartRefKey{ns_b, "part_b"}); - const auto after = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed].load(); + const auto after = global_counters[ProfileEvents::CASRefRollbackBestEffortDropFailed]; EXPECT_EQ(after, before + 1); EXPECT_GT(clock->pauseCount(), 0u) << "the give-up must be the call's own retry window, reached through the injected sleep"; diff --git a/src/Disks/tests/gtest_cas_part_write.cpp b/src/Disks/tests/gtest_cas_part_write.cpp index 7259f13360ca..057aed90e97b 100644 --- a/src/Disks/tests/gtest_cas_part_write.cpp +++ b/src/Disks/tests/gtest_cas_part_write.cpp @@ -495,8 +495,8 @@ TEST(CASPartWriteTxnMetaCounters, CreateCleanAndChokePointCountOnFreshBody) { /// Fresh body upload writes the Clean meta exactly once: CASMetaPut +1 (choke point) /// and CASMetaCreateClean +1 (reason). Reuse the fixture of the nearest putBlob test. - const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load(); - const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean].load(); + const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut]; + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean]; auto b = std::make_shared(); auto s = openPool(b); @@ -506,8 +506,8 @@ TEST(CASPartWriteTxnMetaCounters, CreateCleanAndChokePointCountOnFreshBody) auto ref = build->putBlob(idOf(payload), BlobSource::fromString(payload)); EXPECT_EQ(ref.size, payload.size()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load() - put_before, 1); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean].load() - reason_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut] - put_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaCreateClean] - reason_before, 1); } /// §0 introspection: an adopt of a pre-existing body that has NO meta at all (a pre-protocol blob, or a @@ -517,8 +517,8 @@ TEST(CASPartWriteTxnMetaCounters, CreateCleanAndChokePointCountOnFreshBody) /// pairs `writeRawBlobBody` with `writeMetaClean`, which skips the `!lm` backfill branch entirely. TEST(CASPartWriteTxnMetaCounters, AdoptBackfillCountsChokePointAndReason) { - const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load(); - const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill].load(); + const auto put_before = ProfileEvents::global_counters[ProfileEvents::CASMetaPut]; + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill]; auto b = std::make_shared(); auto s = openPool(b); @@ -543,8 +543,8 @@ TEST(CASPartWriteTxnMetaCounters, AdoptBackfillCountsChokePointAndReason) auto ref = build->putBlob(id, BlobSource::fromString(payload)); EXPECT_EQ(ref.ref, id); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut].load() - put_before, 1); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill].load() - reason_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaPut] - put_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMetaAdoptBackfill] - reason_before, 1); const auto lm = loadMetaForTest(*b, s->layout(), hash); ASSERT_TRUE(lm.has_value()) << "the adopt-backfill must leave a Clean meta for future point-readers"; @@ -675,8 +675,8 @@ TEST(CASPartWriteTxn, PutBlobRepublishesWhenMetaCondemned) /// choke point (`CASMetaCompareSwap`), tagged with its reason (`CASMetaResurrectClean`). TEST(CASPartWriteTxnMetaCounters, CondemnedRepublicationCountsCasAndReason) { - const auto cas_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap].load(); - const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean].load(); + const auto cas_before = ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap]; + const auto reason_before = ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean]; auto b = std::make_shared(); auto s = openPool(b); @@ -697,8 +697,8 @@ TEST(CASPartWriteTxnMetaCounters, CondemnedRepublicationCountsCasAndReason) auto ref = build->putBlob(id, BlobSource::fromString(payload)); EXPECT_EQ(ref.ref, id); - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap].load() - cas_before, 1); - EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean].load() - reason_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaCompareSwap] - cas_before, 1); + EXPECT_GE(ProfileEvents::global_counters[ProfileEvents::CASMetaResurrectClean] - reason_before, 1); } TEST(CASPartWriteTxn, PutBlobWrongSizeFailsClosed) @@ -1167,14 +1167,14 @@ TEST(CASPartWriteTxn, PromoteTrustsAdoptedLeafNoProbeManifestTrust) const ManifestId id = build->stageManifest({entry}); build->precommitAdd(ns, "part_1", id); - const auto trusted_before = ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted].load(); + const auto trusted_before = ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted]; const size_t head_before = counting->headCountFor(blob_key); const size_t meta_get_before = counting->getCountFor(meta_key); build->promote(ns, "part_1", build->buildId(), id); EXPECT_TRUE(s->resolveRef(ns, "part_1").has_value()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted].load() - trusted_before, 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobAdoptTrusted] - trusted_before, 1); EXPECT_EQ(counting->headCountFor(blob_key) - head_before, 0u) << "trust must not HEAD the adopted blob"; EXPECT_EQ(counting->getCountFor(meta_key) - meta_get_before, 0u) << "trust must not loadMeta the adopted blob"; } diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 452511f5f761..01eee02cdb8d 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -1711,10 +1711,10 @@ TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) EXPECT_FALSE(invalid_store->tryRemountOnce()) << "a foreign owner is never taken over at remount"; const uint64_t violations_before - = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation]; invalid_store.reset(); /// must not abort, must not terminate - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation], violations_before + 1) << "the release must report the broken single-writer guarantee rather than dying on it"; const auto occupant_after = readObj(*foreign_backend, foreign_mount_key); @@ -2170,9 +2170,9 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav ASSERT_TRUE(std::holds_alternative( (*successor_op).replace(key, encodeMountLease(successor), ours->etag, Retry::standard()))); const uint64_t skipped_before - = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant]; const uint64_t violations_before - = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation]; int failure_code = 0; String failure_message; @@ -2204,7 +2204,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav EXPECT_NE(failure_message.find("held by a foreign server"), String::npos) << failure_message; EXPECT_FALSE(runtime.mayMutate()); EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant], skipped_before + 1); const auto failed = std::find_if(events.begin(), events.end(), [](const CasEvent & event) { @@ -2220,14 +2220,14 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav const uint64_t gets_before_teardown = backend->getCount(key); const uint64_t writes_before_teardown = backend->putOverwriteCount(key); const uint64_t skipped_before_teardown - = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant]; runtime.finishTeardown(true); EXPECT_EQ(backend->headCount(key), heads_before_teardown); EXPECT_EQ(backend->getCount(key), gets_before_teardown); EXPECT_EQ(backend->putOverwriteCount(key), writes_before_teardown); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant], skipped_before_teardown); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), violations_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation], violations_before); const auto successor_after_teardown = readObj(*backend, key); ASSERT_TRUE(successor_after_teardown.has_value()); EXPECT_EQ(successor_after_teardown->bytes, successor_before_teardown->bytes); @@ -3150,13 +3150,13 @@ TEST(CASRemountWaits, ALateTouchedTableClosesEveryDeadEpochInBandHoweverItsPrede ASSERT_EQ(store->liveWriterEpoch(), 3u); using ProfileEvents::global_counters; - const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed]; /// ns2's FIRST recovery under this incarnation happens now, at epoch 3 -- strictly after both /// remounts. Its only data is at epoch 1, so epochs 1 and 2 are both dead for it. EXPECT_EQ(store->listRefs(ns2).size(), 1u); - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2) + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed], sealed_before + 2) << "both dead epochs must be closed -- the chain link is what a later reader needs to tell an " "EMPTY epoch from a LOST one, and that is independent of how each mount ended"; EXPECT_TRUE(readObj(*backend, layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns2), RefTxnId{1, 2})).has_value()) @@ -3879,7 +3879,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) runtime.armMountFence(uuid, 1, anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; - const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost]; runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); @@ -3889,7 +3889,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) EXPECT_EQ(remount_calls.load(), 1u); EXPECT_EQ(fresh_epochs.load(), 1u); EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 1u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost], lost_before + 1); remount_barrier.release(); runtime.stopBackgroundWorkers(); runtime.finishTeardown(false); @@ -4522,17 +4522,17 @@ TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) store->tripMountLost(); backend->failNextRead(store->layout().poolMetaKey()); - const uint64_t attempts_before = ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts].load(); - const uint64_t succeeded_before = ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(); - const uint64_t failed_before = ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(); + const uint64_t attempts_before = ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts]; + const uint64_t succeeded_before = ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded]; + const uint64_t failed_before = ProfileEvents::global_counters[ProfileEvents::CASRemountFailed]; EXPECT_FALSE(store->tryRemountOnce()); fenceOutMount(*backend, store->layout().mountKey("test")); EXPECT_TRUE(store->tryRemountOnce()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts].load(), attempts_before + 2); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded].load(), succeeded_before + 1); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountFailed].load(), failed_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountAttempts], attempts_before + 2); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountSucceeded], succeeded_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRemountFailed], failed_before + 1); const std::vector observed_events = events->snapshot(); std::vector remounts; @@ -4627,7 +4627,7 @@ TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) .pool_prefix = "lease-loss-owner", .server_root_id = "test", }); - const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost]; store->tripMountLost(); store->tripMountLost(); @@ -4636,7 +4636,7 @@ TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) store->beginShutdownForTest(); store->tripMountLost(); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost], lost_before + 1); } TEST(CASPoolRemount, LiveForgetDoesNotCountOperationalLeaseLoss) @@ -4647,12 +4647,12 @@ TEST(CASPoolRemount, LiveForgetDoesNotCountOperationalLeaseLoss) .server_root_id = "test", }); ASSERT_EQ(store->lifecycle(), PoolLifecycle::Live); - const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost]; store->forgetDisk([] {}, "deliberate test decommission"); EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost], lost_before) << "a deliberate terminal decommission is not an operational recovery generation"; } @@ -4696,8 +4696,8 @@ TEST(CASPool, ConcurrentNamespaceCreationsNeverRaceEachOtherOnTheCatalog) (void)pool->namespaceLife(DB::Cas::RootNamespace{"warmup"}); const uint64_t writes_before = backend->writeCount(key); - const auto reads_before = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load(); - const auto resolves_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); + const auto reads_before = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts]; + const auto resolves_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead]; std::vector threads; for (int i = 0; i < N; ++i) threads.emplace_back([&, i] { (void)pool->namespaceLife(DB::Cas::RootNamespace{"ns" + std::to_string(i)}); }); @@ -4705,9 +4705,9 @@ TEST(CASPool, ConcurrentNamespaceCreationsNeverRaceEachOtherOnTheCatalog) t.join(); EXPECT_EQ(backend->writeCount(key) - writes_before, 2u * N) << "two catalog steps per creation, each one write"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolves_before, 0u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead] - resolves_before, 0u) << "no refused precondition, so no resolve read"; - EXPECT_LE(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load() - reads_before, 1u) + EXPECT_LE(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts] - reads_before, 1u) << "at most one lane read; every later hold started from the cache"; /// Another server writes the catalog between two of this pool's mutations: one extra read and one @@ -4723,12 +4723,12 @@ TEST(CASPool, ConcurrentNamespaceCreationsNeverRaceEachOtherOnTheCatalog) DB::Cas::CatalogEntry{.ns = DB::Cas::RootNamespace{"zz"}, .state = DB::Cas::NsState::Live, .incarnation = UInt128{99}}); } const uint64_t writes_mid = backend->writeCount(key); - const auto resolves_mid = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); - const auto lane_reads_mid = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load(); + const auto resolves_mid = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead]; + const auto lane_reads_mid = ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts]; (void)pool->namespaceLife(DB::Cas::RootNamespace{"after"}); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolves_mid, 1u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead] - resolves_mid, 1u) << "one resolve read for the external write"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts].load() - lane_reads_mid, 0u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASHotKeyReadStarts] - lane_reads_mid, 0u) << "the next hold starts from what the resolve read saw"; EXPECT_EQ(backend->writeCount(key) - writes_mid, 3u) << "one refused, two landed"; } diff --git a/src/Disks/tests/gtest_cas_ref_catalog.cpp b/src/Disks/tests/gtest_cas_ref_catalog.cpp index 8d4c364cb68f..d842ee416fa7 100644 --- a/src/Disks/tests/gtest_cas_ref_catalog.cpp +++ b/src/Disks/tests/gtest_cas_ref_catalog.cpp @@ -2041,7 +2041,7 @@ TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) seedObject(op, layout.gcStateKey(), encodeGcState(state)); const uint64_t signals_before - = ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load(); + = ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals]; const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(removing.ns, life_id); const String unreadable_ref_log_key = layout.refLogKey(life, RefTxnId{5, 6}); const uint64_t append_writes_before = backend->writes(unreadable_ref_log_key); @@ -2051,7 +2051,7 @@ TEST(CASGCStuckRemoval, AdoptedRoundWarnsEveryRestartWithoutAppending) Gc restarted_process(store, gc_id); EXPECT_TRUE(restarted_process.runRegularRound().acquired_lease); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals].load() - signals_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASGCStuckRemovals] - signals_before, 2u); EXPECT_EQ(backend->writes(unreadable_ref_log_key), append_writes_before) << "the diagnostic cannot append the unreadable ref log"; const String captured = log_capture.captured(); @@ -2086,11 +2086,11 @@ TEST(CASGCRefWalkPlan, UnmatchedAdoptedParentLifeIsObservedWithoutEnteringThePla .coverage = RefCoverage{.classification = CoverageClass::Folded, .last_folded_ref_id = RefTxnId{9, 9}}}); const uint64_t events_before = - ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load(); + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives]; const RefPlan plan = tests::buildRefWalkPlanForTest(scan, cut); EXPECT_EQ( - ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives].load() - events_before, + ProfileEvents::global_counters[ProfileEvents::CASGCUnmatchedAdoptedParentLives] - events_before, 1u); EXPECT_EQ(plan.droppedParentRows(), 1u); EXPECT_EQ(plan.size(), 1u); @@ -2182,7 +2182,7 @@ struct PoolAndExternal uint64_t eventCount(ProfileEvents::Event event) { - return ProfileEvents::global_counters[event].load(); + return ProfileEvents::global_counters[event]; } } diff --git a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp index a79f58fdf424..54f164bd452c 100644 --- a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp +++ b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp @@ -647,8 +647,8 @@ TEST(CASRefWriterChunkedFlush, ChunkedFlushCommitsPerChunk) auto c3 = std::make_shared>(0); const size_t tail_before = store->tailSinceSnapshotCountForTest(ns); - const uint64_t flushes_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes].load(); - const uint64_t mutations_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations].load(); + const uint64_t flushes_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes]; + const uint64_t mutations_before = ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations]; auto sync = std::make_shared(); armPreCarveBlock(store, ns, sync, 3); @@ -714,8 +714,8 @@ TEST(CASRefWriterChunkedFlush, ChunkedFlushCommitsPerChunk) /// snapshot-scheduling trigger is the final step of the SAME committed arm that increments /// `CASRefBatchFlushes`, so == 2 also proves the scheduler was invoked per chunk; /// `SnapshotPublisherLatchedAcrossChunks` proves that trigger actually re-fires across chunks. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes].load() - flushes_before, 2u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations].load() - mutations_before, 3u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchFlushes] - flushes_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefBatchedMutations] - mutations_before, 3u); } namespace diff --git a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp index 31648e992ceb..51843cdc33b4 100644 --- a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp +++ b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp @@ -493,9 +493,9 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) const RootNamespace ns{"srv1/contig_survivor"}; ASSERT_EQ(publishRef(survivor, ns, "ref_1", 1), (RefTxnId{survivor->writerEpoch(), 1})); const uint64_t skipped_before - = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant]; const uint64_t violations_before - = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation]; /// The prefix is cleared and the pool recreated underneath the still-running survivor. ASSERT_GT(eraseKeysContaining(*backend, ""), 0u); @@ -512,7 +512,7 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) EXPECT_FALSE(survivor->mayMutate()) << "a survivor whose slot was reclaimed must be fenced closed by its own failing renewal, not " "left writing into the new pool"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant], skipped_before + 1) << "the conclusive foreign-successor observation must be counted when deposition is detected"; EXPECT_NE(messageOfThrow([&] { publishRef(survivor, ns, "ref_2", 2); }), String()) @@ -530,14 +530,14 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) const auto successor_slot_before = (*teardown_op).read(survivor_mount_key, Retry::once()); ASSERT_TRUE(successor_slot_before.has_value()); const uint64_t skipped_after_deposition - = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); + = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant]; survivor.reset(); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant], skipped_after_deposition) << "terminal teardown must not count the already-observed successor twice"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation].load(), + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountExclusivityViolation], violations_before) << "and must NOT report an exclusivity violation: this is a failover, not a broken guarantee"; const auto successor_slot_after = (*teardown_op).read(survivor_mount_key, Retry::once()); diff --git a/src/Disks/tests/gtest_cas_ref_gc.cpp b/src/Disks/tests/gtest_cas_ref_gc.cpp index c76f68d0489b..404762c2fbb2 100644 --- a/src/Disks/tests/gtest_cas_ref_gc.cpp +++ b/src/Disks/tests/gtest_cas_ref_gc.cpp @@ -915,12 +915,12 @@ TEST(CASRefGcCleanupAuthority, RebirthDuringAChunkDeletesOnlyTheOldLifesKeys) janitor_deleted = it->second; }); using ProfileEvents::global_counters; - const auto ref_cleanup_deleted_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + const auto ref_cleanup_deleted_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted]; ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); ASSERT_TRUE(ref_cleanup_suppressed.has_value()) << "the ref_object_cleanup phase row never fired"; EXPECT_EQ(*ref_cleanup_suppressed, 0u) << "round two must actually run destructive work, not merely leave everything alone because it was suppressed"; - EXPECT_EQ(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(), ref_cleanup_deleted_before) + EXPECT_EQ(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted], ref_cleanup_deleted_before) << "cleanupRefObjects' own plan/cohort must delete NOTHING this round: the old life is not in it " "(no catalog entry names it) and the reborn life's own checkpoint-named log is not yet deletable"; ASSERT_TRUE(janitor_deleted.has_value()) << "the namespace_cleanup phase row never fired"; @@ -1048,7 +1048,7 @@ TEST(CASRefGc, RefObjectCleanupFallsBackToOnePerKeyWhenBatchDeleteIsUnsupported) /// supported"; the fallback's per-key calls that follow are not armed and succeed. backend->failNextBulkRemoveWith(std::make_exception_ptr( DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); - const auto cleaned_before = ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + const auto cleaned_before = ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted]; OperationForTest op(*backend); Gc gc(store, kGc); @@ -1058,7 +1058,7 @@ TEST(CASRefGc, RefObjectCleanupFallsBackToOnePerKeyWhenBatchDeleteIsUnsupported) EXPECT_FALSE((*op).head(layout.refLogKey(life, id), Retry::once()).has_value()); for (const RefTxnId & id : plan.deletable_snapshots) EXPECT_FALSE((*op).head(layout.refSnapshotKey(life, id), Retry::once()).has_value()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load() - cleaned_before, cohort_size) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted] - cleaned_before, cohort_size) << "the budget/profile-event accounting counts objects, unaffected by the fallback"; /// 1 failed bulk attempt + one request per key in the cohort. EXPECT_EQ(backend->bulkRemoveCalls(), 1 + cohort_size); @@ -1070,11 +1070,11 @@ TEST(CASRefGc, RefObjectCleanupFallsBackToOnePerKeyWhenBatchDeleteIsUnsupported) TEST(CASRefGc, RefIntakeIncrementsObservabilityCounters) { using ProfileEvents::global_counters; - const auto list_pages_before = global_counters[ProfileEvents::CASRefGlobalListPages].load(); - const auto log_gets_before = global_counters[ProfileEvents::CASRefLogBodyGets].load(); - const auto mf_gets_before = global_counters[ProfileEvents::CASRefManifestBodyFoldGets].load(); - const auto edges_before = global_counters[ProfileEvents::CASRefEmittedEdges].load(); - const auto cleaned_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); + const auto list_pages_before = global_counters[ProfileEvents::CASRefGlobalListPages]; + const auto log_gets_before = global_counters[ProfileEvents::CASRefLogBodyGets]; + const auto mf_gets_before = global_counters[ProfileEvents::CASRefManifestBodyFoldGets]; + const auto edges_before = global_counters[ProfileEvents::CASRefEmittedEdges]; + const auto cleaned_before = global_counters[ProfileEvents::CASRefCleanupObjectsDeleted]; auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); @@ -1101,11 +1101,11 @@ TEST(CASRefGc, RefIntakeIncrementsObservabilityCounters) Gc gc(store, kGc); runToFixpoint(store, gc); - EXPECT_GT(global_counters[ProfileEvents::CASRefGlobalListPages].load(), list_pages_before); - EXPECT_GT(global_counters[ProfileEvents::CASRefLogBodyGets].load(), log_gets_before); - EXPECT_GT(global_counters[ProfileEvents::CASRefManifestBodyFoldGets].load(), mf_gets_before); - EXPECT_GT(global_counters[ProfileEvents::CASRefEmittedEdges].load(), edges_before); - EXPECT_GT(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(), cleaned_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefGlobalListPages], list_pages_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefLogBodyGets], log_gets_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefManifestBodyFoldGets], mf_gets_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefEmittedEdges], edges_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefCleanupObjectsDeleted], cleaned_before); } /// Task 13 e2e (in-process regression twin of the rustfs integration test): the whole snapshot+log diff --git a/src/Disks/tests/gtest_cas_ref_install_safety.cpp b/src/Disks/tests/gtest_cas_ref_install_safety.cpp index d3da467cb415..dbc09c25a090 100644 --- a/src/Disks/tests/gtest_cas_ref_install_safety.cpp +++ b/src/Disks/tests/gtest_cas_ref_install_safety.cpp @@ -879,14 +879,14 @@ TEST(CASRefInstallSafety, PostDurableInstallFailureRequiresRecovery) publishEmptyPart(store, ns, "x"); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load(); + const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery]; armOneShotInstallFailure(store); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "x"); }); store->setInstallRegionProbeForTest(nullptr); EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::NeedsRecovery) << "an install that failed AFTER its object was durable must be visible, not silent"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load() - needs_recovery_before, 1u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery] - needs_recovery_before, 1u) << "the transition to NeedsRecovery must be exported exactly once"; } @@ -935,7 +935,7 @@ TEST(CASRefInstallSafety, NeedsRecoveryReplaysBeforeALaterFlush) publishEmptyPart(store, ns, "x"); publishEmptyPart(store, ns, "y"); - const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load(); + const uint64_t needs_recovery_before = ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery]; armOneShotInstallFailure(store); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, [&] { store->dropRef(ns, "x"); }); store->setInstallRegionProbeForTest(nullptr); @@ -959,7 +959,7 @@ TEST(CASRefInstallSafety, NeedsRecoveryReplaysBeforeALaterFlush) << "the stranded drop of 'x' is durable, so the re-derivation must install it before returning " "the lane to Ready"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery].load() - needs_recovery_before, 1u) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefNeedsRecovery] - needs_recovery_before, 1u) << "the event counts transitions, so the successful flush must not have added another"; } diff --git a/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp index 7919711c1753..5d92d37e9502 100644 --- a/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp +++ b/src/Disks/tests/gtest_cas_ref_recovery_cas_walk.cpp @@ -479,7 +479,7 @@ std::optional readLogTxn(Backend & backend, const Layout & layout, co uint64_t counterOf(ProfileEvents::Event event) { - return ProfileEvents::global_counters[event].load(); + return ProfileEvents::global_counters[event]; } NamespaceLifeId catalogLife(const BackendPtr & backend, const Layout & layout, const RootNamespace & ns) diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp index dbb53e10fe0a..966a5ef33e00 100644 --- a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp +++ b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp @@ -403,7 +403,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// before the 4th. backend->armWriteFailure("_snap/", kFaultsBeyondTheRetryWindow); - const auto dispatchCount = [&] { return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); }; + const auto dispatchCount = [&] { return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; }; /// Attempt 1: admitted immediately (no backoff armed yet). Fails -> backoff armed at the initial 1000ms. ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 2})); @@ -523,11 +523,11 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable const auto dispatch_count = [&] { - return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; }; const auto backoff_count = [&] { - return global_counters[ProfileEvents::CASRefSnapshotPublishBackoff].load(); + return global_counters[ProfileEvents::CASRefSnapshotPublishBackoff]; }; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); diff --git a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp index 9fcd010643a9..c0d66fad0a7a 100644 --- a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp +++ b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp @@ -835,8 +835,8 @@ TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCa backend->s3_definite_substr = logPrefix(store, ns); backend->s3_definite_latched = true; - const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); - const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(); + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged]; + const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure]; const size_t pauses_before = clock->pauseCount(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); @@ -850,9 +850,9 @@ TEST(CASRefWedgeEveryAttempt, ADefiniteRefusalAfterAnAmbiguousAttemptOfTheSameCa << "one call whose first attempt is unresolved leaves an object that may become durable"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged) << "the id must NOT be declared never-used: the marker stands until the key itself resolves"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(), definite_before) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure], definite_before) << "this append was never definitively rejected -- only one of its attempts was"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged], wedged_before + 1); EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "nothing is applied while the lane is wedged"; const String wedged_key = store->wedgedKeyForTest(ns); EXPECT_FALSE(readObj(*backend, wedged_key).has_value()) << "and nothing became durable"; @@ -1245,7 +1245,7 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteMeetingASuccessorSealIsAConclusiveReject const RefTxnId next{epoch, 3}; ASSERT_EQ(store->lastEpochSealForTest(ns), std::nullopt); const uint64_t remounts_before = store->scheduleRemountCallCountForTest(); - const uint64_t sealed_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected].load(); + const uint64_t sealed_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected]; /// The successor's seal lands at exactly the id this table's next append derives. backend->conflict_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); @@ -1260,7 +1260,7 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteMeetingASuccessorSealIsAConclusiveReject << "the observed seal IS this namespace's epoch-closing record, whichever site observed it"; EXPECT_TRUE(store->mayMutate()) << "the designed path must not fence the mount"; EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedule a remount"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected].load(), sealed_before + 1) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendSealRejected], sealed_before + 1) << "a deposed writer must still be COUNTED: this is the protocol working, and also the signal " "that this mount has lost its lease and does not know it"; } @@ -1465,13 +1465,13 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteWedgesWhenTheSettlingReadNamesNoOccupant /// proves absence and names nobody. backend->refuse_precondition_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); - const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(); - const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable]; + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged]; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(), deferred_before + 1) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable], deferred_before + 1) << "the deferral is the one quiet arm here -- it must be counted or a starved loud path is invisible"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1) + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged], wedged_before + 1) << "and the lane it leaves behind is a wedge, so the wedge counter must say so"; EXPECT_TRUE(store->mayMutate()) << "the table defers without guessing that the whole mount is corrupt"; EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before) << "nor schedules a remount"; @@ -1512,10 +1512,10 @@ TEST(CASRefWedgeEveryAttempt, AppendSiteWedgesWhenTheSettlingReadItselfIsRefused backend->refuse_precondition_substr = layout.refLogKey(DB::Cas::tests::fixture::fixtureLife(ns), next); backend->refuse_read_after_precondition = true; - const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(); + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable]; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable].load(), deferred_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendOccupantUnreadable], deferred_before + 1); EXPECT_TRUE(store->mayMutate()); EXPECT_EQ(store->scheduleRemountCallCountForTest(), remounts_before); EXPECT_TRUE(store->refLaneWedgedForTest(ns)); @@ -1658,16 +1658,16 @@ TEST(CASRefLane, RefusedReturnsTheAttemptToReadyAndDoesNotWedge) backend->s3_definite_substr = logPrefix(store, ns); backend->s3_definite_count = 1; - const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); - const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(); + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged]; + const uint64_t definite_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure]; expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->dropRef(ns, "x"); }); EXPECT_FALSE(store->refLaneWedgedForTest(ns)) << "a proven refusal wrote nothing, so there is nothing for a wedge to resolve"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure].load(), definite_before + 1); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendDefiniteFailure], definite_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged], wedged_before); EXPECT_TRUE(store->resolveRef(ns, "x").has_value()) << "the refused drop applied nothing"; /// And the id was never consumed: the next caller re-derives it and lands the same transaction. @@ -1690,7 +1690,7 @@ TEST(CASRefLane, PostCommitFenceLossWedges) publishEmptyPart(store, ns, "x"); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); - const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(); + const uint64_t wedged_before = ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged]; /// The fence is lost INSIDE the write window, so the object lands and the call may not claim it. backend->armBlock(logPrefix(store, ns)); @@ -1709,7 +1709,7 @@ TEST(CASRefLane, PostCommitFenceLossWedges) EXPECT_TRUE(store->refLaneWedgedForTest(ns)) << "the object is durable, so the lane must not be returned to Ready"; EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Wedged); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged].load(), wedged_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefAppendWedged], wedged_before + 1); EXPECT_TRUE(readObj(*backend, store->wedgedKeyForTest(ns)).has_value()) << "the write landed -- what was refused is the CLAIM, not the object"; } diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp index 159b12fb99e9..e75bb4f1e051 100644 --- a/src/Disks/tests/gtest_cas_ref_writer.cpp +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -2173,12 +2173,12 @@ TEST(CASRefWriteContract, PostWriteFenceLossIsCounted) CasRequests requests(backend, Fence::open()); CasOperation op = requests.admit([&live] { return live; }); - const auto before = global_counters[ProfileEvents::CASRequestFenceLostPostWrite].load(); + const auto before = global_counters[ProfileEvents::CASRequestFenceLostPostWrite]; const WriteResult result = op.create("k", "v", Retry::standard()); const auto * gave_up = std::get_if(&result); ASSERT_TRUE(gave_up != nullptr) << "a post-write fence loss must never be reported as committed"; EXPECT_EQ(gave_up->why, GaveUp::Why::FenceLost); - EXPECT_EQ(global_counters[ProfileEvents::CASRequestFenceLostPostWrite].load(), before + 1); + EXPECT_EQ(global_counters[ProfileEvents::CASRequestFenceLostPostWrite], before + 1); } /// A commit surfaces the incarnation it created -- from the attempt's own response, and equally from a @@ -2702,8 +2702,8 @@ TEST(CASRefWriterRuntimeIdentity, LatePredecessorInvalidationLeavesSuccessorAtta TEST(CASRefWriterSnapshotPublish, PublishIncrementsSnapshotCounters) { using ProfileEvents::global_counters; - const auto bytes_before = global_counters[ProfileEvents::CASRefSnapshotPutBytes].load(); - const auto logs_before = global_counters[ProfileEvents::CASRefSnapshotTailLogs].load(); + const auto bytes_before = global_counters[ProfileEvents::CASRefSnapshotPutBytes]; + const auto logs_before = global_counters[ProfileEvents::CASRefSnapshotTailLogs]; auto backend = std::make_shared(); const Layout layout("p"); @@ -2719,8 +2719,8 @@ TEST(CASRefWriterSnapshotPublish, PublishIncrementsSnapshotCounters) ASSERT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()) << "the threshold trigger must have published a snapshot"; - EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPutBytes].load(), bytes_before); - EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotTailLogs].load(), logs_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPutBytes], bytes_before); + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotTailLogs], logs_before); } /// A fresh mount that recovers a large PRE-EXISTING tail (left by a predecessor whose own thresholds @@ -2829,12 +2829,12 @@ TEST(CASRefWriterSnapshotPublish, TriggerFiresOnCountAboveThresholdWithoutAging) config.mount_lease_ttl_ms = std::chrono::milliseconds(10'000'000); auto store = openPoolWithConfig(backend, config); - const auto before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; publishEmptyPart(store, ns, "a"); /// tail: 2 publishEmptyPart(store, ns, "b"); /// tail: 4 > 3 -> dispatches, clock frozen throughout store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), before) + EXPECT_GT(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], before) << "the count trigger must fire without any aging, even under a frozen clock"; } @@ -3060,11 +3060,11 @@ TEST(CASRefWriterSnapshotPublish, ClampedCounterSubClampsInsteadOfUnderflowingOn /// A wrapped counter would read as ~UINT64_MAX, permanently latching `over_threshold` (the C4 /// storm regression). With the huge threshold configured above, a dispatch firing here can ONLY /// mean the counter is corrupted -- a clamped counter of 0 never crosses it. - const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; for (int i = 0; i < 5; ++i) store->resolveRef(ns, "a"); store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_before) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], dispatched_before) << "a correctly-clamped counter must never latch the threshold trigger"; } @@ -3116,13 +3116,13 @@ TEST(CASRefWriterSnapshotPublish, C4LatchBoundedUnderSustainedNonCommittedPublis "this from a non-retrying policy"; EXPECT_GT(clock->longestPause(), 0u) << "at least one of those reissues must have paced with a real backoff"; - const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto dispatched_before = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; for (int i = 0; i < 30; ++i) { store->resolveRef(ns, "a"); store->waitForSnapshotPublishSettleForTest(ns); } - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_before) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], dispatched_before) << "reads within the backoff window must not re-dispatch a publish (the storm latch is broken)"; } @@ -3173,18 +3173,18 @@ TEST(CASRefWriterSnapshotPublish, RecoveredSealAboveThresholdDoesNotRedispatchUn EXPECT_TRUE(successor->resolveRef(ns, "before_seal").has_value()); successor->waitForSnapshotPublishSettleForTest(ns); - const auto dispatched_at_seal = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto dispatched_at_seal = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; for (int i = 0; i < 5; ++i) EXPECT_TRUE(successor->resolveRef(ns, "before_seal").has_value()); successor->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_at_seal) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], dispatched_at_seal) << "a recovered seal must not dispatch or re-dispatch an unpublishable snapshot candidate"; /// One ordinary append transaction above the recovered seal must reopen the scheduler. `dropRef` /// is exactly one ordinary ref-log append, unlike `publishEmptyPart`'s two-phase part publication. successor->dropRef(ns, "before_seal"); successor->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), dispatched_at_seal + 1) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], dispatched_at_seal + 1) << "an ordinary successor above the seal must make the threshold candidate publishable again"; } @@ -3260,10 +3260,10 @@ TEST(CASRefWriterSnapshotPublish, C4BackoffDefersThenRetriesAndPublishes) EXPECT_GT(clock->longestPause(), 0u) << "at least one of those reissues must have paced with a real backoff"; /// A read within the backoff window (frozen clock) must not re-dispatch. - const auto d1 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto d1 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; store->resolveRef(ns, "a"); store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d1) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], d1) << "a read within the backoff window must not re-dispatch"; EXPECT_FALSE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()); @@ -3272,7 +3272,7 @@ TEST(CASRefWriterSnapshotPublish, C4BackoffDefersThenRetriesAndPublishes) *fake_now += 2000; store->resolveRef(ns, "a"); store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d1 + 1) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], d1 + 1) << "after the backoff elapses exactly one retry is dispatched"; EXPECT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns).has_value()) << "the retry publishes a durable snapshot (freshness preserved)"; @@ -3300,13 +3300,13 @@ TEST(CASRefWriterSnapshotPublish, TriggerIgnoresEntriesCoveredByNewestSnapshot) config.snapshot_log_bytes_threshold = 1ULL << 40; auto store = openPoolWithConfig(backend, config); - const auto d0 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); + const auto d0 = global_counters[ProfileEvents::CASRefSnapshotPublishDispatched]; /// Drive ONE successful publish: 4 entries (4 > 3). publishEmptyPart(store, ns, "a"); publishEmptyPart(store, ns, "b"); store->waitForSnapshotPublishSettleForTest(ns); - ASSERT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 1); + ASSERT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], d0 + 1); const auto first_snap = listGreatestSnapshotIdForTest(*backend, layout, ns); ASSERT_TRUE(first_snap.has_value()); EXPECT_TRUE(store->newestPublishedSnapshotIdForTest(ns) == first_snap); @@ -3316,7 +3316,7 @@ TEST(CASRefWriterSnapshotPublish, TriggerIgnoresEntriesCoveredByNewestSnapshot) /// a covered-counting trigger to 6 > 3. Must not dispatch. publishEmptyPart(store, ns, "c"); store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 1) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], d0 + 1) << "entries covered by the newest snapshot must not count toward the trigger"; EXPECT_TRUE(listGreatestSnapshotIdForTest(*backend, layout, ns) == first_snap); EXPECT_EQ(store->tailSinceSnapshotCountForTest(ns), 2u); @@ -3325,7 +3325,7 @@ TEST(CASRefWriterSnapshotPublish, TriggerIgnoresEntriesCoveredByNewestSnapshot) /// and it covers the whole uncovered tail. publishEmptyPart(store, ns, "d"); store->waitForSnapshotPublishSettleForTest(ns); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(), d0 + 2); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSnapshotPublishDispatched], d0 + 2); const auto second_snap = listGreatestSnapshotIdForTest(*backend, layout, ns); ASSERT_TRUE(second_snap.has_value()); EXPECT_TRUE(*first_snap < *second_snap); @@ -3450,9 +3450,9 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM /// The sweep is piggybacked on this mount's very first touch; its (uncertain) failure is INSULATED /// from the read (resolveRef/listRefs call `sweepStalePrecommitsForRead`, not /// `maybeSweepStalePrecommits` directly): the read itself still succeeds, the failure is counted. - const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred]; EXPECT_NO_THROW(successor->listRefs(ns)); - const uint64_t deferred_after = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); + const uint64_t deferred_after = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred]; EXPECT_EQ(deferred_after, deferred_before + 1) << "the read-only caller must observe (and count) the deferred sweep failure, not throw"; EXPECT_TRUE(successor->refLaneWedgedForTest(ns)); @@ -3592,15 +3592,15 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) /// FIRST trigger (read path): the sweep's removal PUT is uncertain -> the lane wedges; the read /// itself still succeeds and counts the deferral (existing contract) -- but the shot must NOT be /// consumed: the flag is re-armed for a later trigger. - const uint64_t deferred_before = global_counters[ProfileEvents::CASRefSweepDeferred].load(); - const uint64_t rearmed_before = global_counters[ProfileEvents::CASRefSweepRearmed].load(); - const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(); + const uint64_t deferred_before = global_counters[ProfileEvents::CASRefSweepDeferred]; + const uint64_t rearmed_before = global_counters[ProfileEvents::CASRefSweepRearmed]; + const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed]; EXPECT_NO_THROW(successor->listRefs(ns)); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before + 1); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred], deferred_before + 1); EXPECT_TRUE(successor->refLaneWedgedForTest(ns)); EXPECT_TRUE(successor->needsStalePrecommitSweepForTest(ns)) << "a failed sweep must re-arm needs_stale_precommit_sweep, not consume the once-per-mount shot"; - EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepRearmed].load(), rearmed_before + 1); + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepRearmed], rearmed_before + 1); backend->disarmFaults(); EXPECT_GT(clock->pauseCount(), 1u) << "the reissues must pace through the injected sleep, never a real one"; @@ -3608,7 +3608,7 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) /// Within the backoff window (the injected clock has not advanced) a read must NOT re-attempt -- /// the bounded-backoff storm latch: no new deferral, flag still armed. EXPECT_NO_THROW(successor->listRefs(ns)); - EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before + 1) + EXPECT_EQ(global_counters[ProfileEvents::CASRefSweepDeferred], deferred_before + 1) << "within the backoff window the sweep must not re-attempt (PUT-storm latch)"; EXPECT_TRUE(successor->needsStalePrecommitSweepForTest(ns)); @@ -3636,7 +3636,7 @@ TEST(CASRefWriterStalePrecommitSweep, FailedSweepRearmsAndRetriesUntilClean) reclaimed_refs.push_back(e.ref_name); std::sort(reclaimed_refs.begin(), reclaimed_refs.end()); EXPECT_EQ(reclaimed_refs, (std::vector{"stale_a", "stale_b", "stale_c"})); - EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(), reclaimed_before + 3); + EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed], reclaimed_before + 3); } /// Verified-clean semantics: a sweep that finds NOTHING stale clears the flag on its very first pass @@ -3663,13 +3663,13 @@ TEST(CASRefWriterStalePrecommitSweep, VerifiedCleanSweepClearsFlagWithoutEvents) seen->push(e); }); - const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(); - const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(); + const uint64_t deferred_before = ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred]; + const uint64_t reclaimed_before = global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed]; EXPECT_NO_THROW(successor->listRefs(ns)); EXPECT_FALSE(successor->needsStalePrecommitSweepForTest(ns)) << "a clean first pass IS the verified-clean sweep: the flag clears without any removal"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred].load(), deferred_before); - EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed].load(), reclaimed_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefSweepDeferred], deferred_before); + EXPECT_EQ(global_counters[ProfileEvents::CASRefStalePrecommitsReclaimed], reclaimed_before); const std::vector observed = seen->snapshot(); EXPECT_EQ(std::count_if(observed.begin(), observed.end(), [](const CasEvent & e) { return e.type == CasEventType::PrecommitReclaim; }), 0); @@ -5411,16 +5411,16 @@ TEST(CASRefWriterRecoveryRetry, TransientSealFailureIsRetriedThenSucceeds) store->setRefRecoveryRetrySleepForTest([&backend](uint64_t, const auto &) { backend->disarmFaults(); }); using ProfileEvents::global_counters; - const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); - const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries]; + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed]; EXPECT_EQ(store->listRefs(ns).size(), 2u) << "recovery must succeed after retrying past the fault"; - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before + 1); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries], retries_before + 1); /// TWO dead epochs (1 and 2) are closed by this walk, and a whole attempt is re-driven per transient /// failure -- so the seals of the epochs the failed attempt already closed are ADOPTED on the retry /// rather than minted again. Exactly two are minted in total. - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed], sealed_before + 2); } TEST(CASRefWriterRecoveryRetry, RecoveryDoesNotEnumerateItsStream) @@ -5458,13 +5458,13 @@ TEST(CASRefWriterRecoveryRetry, RecoveryDoesNotEnumerateItsStream) backend->list_fault_count = 2; using ProfileEvents::global_counters; - const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); - const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(); + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries]; + const auto sealed_before = global_counters[ProfileEvents::CASRefRecoveryEpochSealed]; ASSERT_TRUE(store->namespaceFilesLifeIfReadable(ns)); EXPECT_EQ(backend->list_fault_count, 2); - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before); - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed].load(), sealed_before + 2) + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries], retries_before); + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryEpochSealed], sealed_before + 2) << "two dead epochs (1 and 2) are closed without enumerating their stream"; } @@ -5576,13 +5576,13 @@ TEST(CASRefWriterRecoveryRetry, VanishBrakeStaysTerminalNotRetried) backend->vanish_once_keys.insert(vkey); using ProfileEvents::global_counters; - const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries].load(); + const auto retries_before = global_counters[ProfileEvents::CASRefRecoveryRetries]; expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { store->listRefs(ns); }); EXPECT_FALSE(backend->vanish_once_keys.contains(vkey)) << "the test must reach the checkpoint-named snapshot GET, not fail on earlier fixture validation"; - EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries].load(), retries_before) + EXPECT_EQ(global_counters[ProfileEvents::CASRefRecoveryRetries], retries_before) << "missing immutable checkpoint authority is terminal; the outer transient-retry loop must NOT re-drive it"; EXPECT_EQ(sleep_calls->load(), 0u) << "no backoff sleep for missing immutable checkpoint authority"; } diff --git a/src/Disks/tests/gtest_cas_repoint.cpp b/src/Disks/tests/gtest_cas_repoint.cpp index fab72a86c59e..f938ff30e914 100644 --- a/src/Disks/tests/gtest_cas_repoint.cpp +++ b/src/Disks/tests/gtest_cas_repoint.cpp @@ -61,7 +61,7 @@ TEST(CASRepoint, ByteEqualIsNoOp) const DB::Cas::PartRefKey key{ns, "part_1"}; backend->resetCounts(); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; const DB::Cas::CommitOutcome oc = access.repointRef(key, {inlineEntry("checksums.txt", "cs")}, ProvenanceOp::Other); EXPECT_FALSE(oc.created); EXPECT_EQ(oc.manifest_ref, id.ref) << "the byte-equal outcome must name the manifest ALREADY committed, unchanged"; @@ -69,7 +69,7 @@ TEST(CASRepoint, ByteEqualIsNoOp) EXPECT_EQ(backend->putTotal(), 0u) << "byte-equal repoint must perform ZERO pool mutations"; EXPECT_EQ(store->resolveRef(ns, "part_1")->manifest_id, id) << "the committed manifest identity must be untouched"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before); } /// A genuinely different entry set on an already-committed ref republishes the manifest: the returned @@ -89,7 +89,7 @@ TEST(CASRepoint, AddFileRepoints) /// Warm the retained view so the erase-on-success cache discipline is actually exercised. ASSERT_NE(access.getView(key, DB::Cas::Freshness::CachedForLoad), nullptr); - const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(); + const uint64_t repoints_before = ProfileEvents::global_counters[ProfileEvents::CASRefRepoint]; const std::vector new_entries{inlineEntry("checksums.txt", "cs"), inlineEntry("metadata_version.txt", "7")}; const DB::Cas::CommitOutcome oc = access.repointRef(key, new_entries, ProvenanceOp::Other); EXPECT_FALSE(oc.created); @@ -98,7 +98,7 @@ TEST(CASRepoint, AddFileRepoints) const auto resolved = store->resolveRef(ns, "part_1"); ASSERT_TRUE(resolved.has_value()); EXPECT_NE(resolved->manifest_id, id_before) << "a genuine content change must mint a fresh manifest"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint].load(), repoints_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefRepoint], repoints_before + 1); /// The view a caller reads next must reflect the new file, not a stale retained one. auto view = access.getView(key, DB::Cas::Freshness::CachedForLoad); diff --git a/src/Disks/tests/gtest_cas_requests.cpp b/src/Disks/tests/gtest_cas_requests.cpp index 81086f27f433..ffa52bb5d00e 100644 --- a/src/Disks/tests/gtest_cas_requests.cpp +++ b/src/Disks/tests/gtest_cas_requests.cpp @@ -1542,14 +1542,14 @@ TEST(CASRequests, CleanConflictsArePacedFlatAndDoNotAdvanceTheReissueCounter) (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); constexpr int K = 4; RaceMaker races(backend, clock, "k", K, /*ambiguous=*/false); - const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load(); - const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause]; + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue]; WriteResult result = op.readModifyWrite("k", appendX(), Retry::standard()); ASSERT_TRUE(std::holds_alternative(result)); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load() - pauses_before, K); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause] - pauses_before, K); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue] - reissues_before, 0u); ASSERT_EQ(clock.sleeps.size(), static_cast(K)); for (uint64_t s : clock.sleeps) EXPECT_LE(s, 200u); /// flat: every pause is one `backoff(1)` draw, whatever the loss count @@ -1565,14 +1565,14 @@ TEST(CASRequests, AConflictThatSettledAFaultKeepsTheGrowingSchedule) (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); constexpr int K = 3; RaceMaker races(backend, clock, "k", K, /*ambiguous=*/true); - const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load(); - const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + const auto pauses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause]; + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue]; WriteResult result = op.readModifyWrite("k", appendX(), Retry::standard()); ASSERT_TRUE(std::holds_alternative(result)); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause].load() - pauses_before, 0u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, K); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConflictPause] - pauses_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue] - reissues_before, K); ASSERT_EQ(clock.sleeps.size(), static_cast(K)); for (size_t i = 0; i < clock.sleeps.size(); ++i) EXPECT_LE(clock.sleeps[i], std::min(5000, 200ull << i)) << "reissue " << i; @@ -2242,7 +2242,7 @@ namespace DB::S3::PocoHTTPClientConfiguration networkFailureClientConfiguration() { DB::RemoteHostFilter remote_host_filter; - return DB::S3::ClientFactory::instance().createClientConfiguration( + auto configuration = DB::S3::ClientFactory::instance().createClientConfiguration( "some-region", remote_host_filter, /* s3_max_redirects = */ 100, @@ -2253,6 +2253,9 @@ DB::S3::PocoHTTPClientConfiguration networkFailureClientConfiguration() /* for_disk_s3 = */ false, /* opt_disk_name = */ {}, /* request_throttler = */ {}); + /// The client is built directly, bypassing ClientFactory::create(), which normally fills retryStrategy. + configuration.retryStrategy = std::make_shared(configuration.retry_strategy); + return configuration; } /// A client whose `PutObject` always fails with a `NETWORK_CONNECTION` `AWSError` carrying `text` @@ -2335,8 +2338,8 @@ TEST(CASRequestsConnectHint, HintedFailuresReissueWithoutARead) backend->failNextWriteWith("k", connectHint()); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); - const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint]; + const auto reissues_before = ProfileEvents::global_counters[ProfileEvents::CASRequestReissue]; WriteResult result = op.create("k", "v", Retry::standard()); const auto * committed = std::get_if(&result); @@ -2348,8 +2351,8 @@ TEST(CASRequestsConnectHint, HintedFailuresReissueWithoutARead) ASSERT_EQ(clock.sleeps.size(), 2u); EXPECT_EQ(clock.sleeps[0], 50u); /// the flat pause, twice EXPECT_EQ(clock.sleeps[1], 50u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 2u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue].load() - reissues_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint] - hints_before, 2u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestReissue] - reissues_before, 2u); } TEST(CASRequestsConnectHint, ReissueMeetsPreconditionAndAdoptsOwnBytes) @@ -2445,7 +2448,7 @@ TEST(CASRequestsConnectHint, OnceKeepsOneWriteAndOneRead) backend->failNextWriteWith("k", connectHint()); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint]; WriteResult result = op.create("k", "v", Retry::once()); const auto * gave_up = std::get_if(&result); ASSERT_NE(gave_up, nullptr); @@ -2455,7 +2458,7 @@ TEST(CASRequestsConnectHint, OnceKeepsOneWriteAndOneRead) EXPECT_TRUE(clock.sleeps.empty()); /// The counter is "hint seen", recorded at classification: `Retry::once` never acts on it, but the /// attempt's transport error still named a failed connection. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint] - hints_before, 1u); } TEST(CASRequestsConnectHint, EarlierAmbiguityStillSettlesByRead) @@ -2525,7 +2528,7 @@ TEST(CASRequestsConnectHint, RefusalAfterAnEarlierAmbiguitySettlesByRead) }); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint]; WriteResult result = op.create("k", "v", Retry::standard()); const auto * committed = std::get_if(&result); ASSERT_NE(committed, nullptr); @@ -2540,7 +2543,7 @@ TEST(CASRequestsConnectHint, RefusalAfterAnEarlierAmbiguitySettlesByRead) EXPECT_TRUE(refusal_fired_on_second_attempt); /// The refusal classification wins outright: a definite refusal is never a hint, so the counter /// must not move even though the exception's code and text also match `isConnectFailureHint`. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint] - hints_before, 0u); } /// A single exception can ALSO be both refreshable-credential-class (`isRefreshableCredentialError` @@ -2557,7 +2560,7 @@ TEST(CASRequestsConnectHint, RefreshedCredentialTextDoesNotDoubleCountTheHint) Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint]; WriteResult result = op.create("k", "v", Retry::standard()); const auto * committed = std::get_if(&result); @@ -2568,7 +2571,7 @@ TEST(CASRequestsConnectHint, RefreshedCredentialTextDoesNotDoubleCountTheHint) EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); /// The refresh -- not the hint's flat pause -- drove the reissue, so the hint counter must not move /// even though the exception's code and text also match `isConnectFailureHint`. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint] - hints_before, 0u); } /// The counter's ambiguity-precedence twin: the credential-owned reissue above requires @@ -2596,7 +2599,7 @@ TEST(CASRequestsConnectHint, CredentialRefreshAfterAnEarlierAmbiguityStillCounts }); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load(); + const auto hints_before = ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint]; WriteResult result = op.create("k", "v", Retry::standard()); const auto * committed = std::get_if(&result); @@ -2612,7 +2615,7 @@ TEST(CASRequestsConnectHint, CredentialRefreshAfterAnEarlierAmbiguityStillCounts EXPECT_TRUE(hint_fired_on_second_attempt); /// The hint mechanism, not a credential-owned reissue, actually resent this attempt, so the counter /// counts it even though the exception's name also matches the refreshable-credential class. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint].load() - hints_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestConnectFailureHint] - hints_before, 1u); } TEST(CASRequestsConnectHint, GatesRefuseTheReissue) @@ -2885,11 +2888,11 @@ TEST(CASRequestsFuse, ReadUnderOnceCountsTheFuseWithoutReissuing) backend->failNextReadWith("k", fuseTimeout()); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse]; expectThrowsCode(DB::ErrorCodes::S3_ERROR, [&] { (void)op.read("k", Retry::once()); }); EXPECT_EQ(backend->getTotal(), 1u) << "Retry::once performs no second attempt"; EXPECT_TRUE(clock.sleeps.empty()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 1u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse] - fuses_before, 1u); } /// The fuse counter's credential-refresh twin of `CASRequestsConnectHint.RefreshedCredentialTextDoesNotDoubleCountTheHint`: @@ -2905,7 +2908,7 @@ TEST(CASRequestsFuse, RefreshedCredentialTextDoesNotDoubleCountTheFuse) Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); auto requests = makeRequests(backend, clock); auto op = requests.admit(); - const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse]; WriteResult result = op.create("k", "v", Retry::standard()); const auto * committed = std::get_if(&result); @@ -2916,7 +2919,7 @@ TEST(CASRequestsFuse, RefreshedCredentialTextDoesNotDoubleCountTheFuse) EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); /// The refresh -- not the fuse's immediate reissue -- drove the resend, so the fuse counter must not /// move even though the exception's code and text also match `isFirstAttemptFuseTimeout`. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse] - fuses_before, 0u); } /// The read loop's own twin of `RefreshedCredentialTextDoesNotDoubleCountTheFuse`: a first read attempt @@ -2934,7 +2937,7 @@ TEST(CASRequestsFuse, ReadRefreshedCredentialTextDoesNotDoubleCountTheFuse) backend->failNextReadWith("k", std::make_exception_ptr(DB::S3Exception( "Poco::Exception. Code: 1000, e.code() = 0, Timeout: the socket", Aws::S3::S3Errors::NETWORK_CONNECTION, "ExpiredToken"))); - const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load(); + const auto fuses_before = ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse]; const auto seen = op.read("k", Retry::standard()); ASSERT_TRUE(seen.has_value()); @@ -2944,7 +2947,7 @@ TEST(CASRequestsFuse, ReadRefreshedCredentialTextDoesNotDoubleCountTheFuse) EXPECT_TRUE(clock.sleeps.empty()); /// The refresh -- not the fuse's immediate reissue -- drove the resend, so the fuse counter must not /// move even though the exception's code and text also match `isFirstAttemptFuseTimeout`. - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 0u); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse] - fuses_before, 0u); } #endif diff --git a/src/Disks/tests/gtest_cas_sentinel_probe.cpp b/src/Disks/tests/gtest_cas_sentinel_probe.cpp index cb52bed51399..545b68e1073f 100644 --- a/src/Disks/tests/gtest_cas_sentinel_probe.cpp +++ b/src/Disks/tests/gtest_cas_sentinel_probe.cpp @@ -12,6 +12,7 @@ #include #include #include +#include #include #include @@ -237,6 +238,17 @@ class ThrowingS3ObjectStorage final : public DB::LocalObjectStorage return DB::LocalObjectStorage::readObject(object, read_settings, read_hint, use_external_buffer, restrict_seek); } + DB::SmallObjectDataWithMetadata readSmallObjectAndGetObjectMetadata( /// NOLINT + const DB::StoredObject & object, + const DB::ReadSettings & read_settings, + size_t max_size_bytes, + std::optional read_hint) const override + { + if (access_error) + throw DB::S3Exception("injected fault: " + object.remote_path, *access_error); + return DB::LocalObjectStorage::readSmallObjectAndGetObjectMetadata(object, read_settings, max_size_bytes, read_hint); + } + DB::ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override { if (access_error) diff --git a/src/Disks/tests/gtest_cas_shutdown_context.cpp b/src/Disks/tests/gtest_cas_shutdown_context.cpp index b1fd2ba48d6f..c0676bbbb81a 100644 --- a/src/Disks/tests/gtest_cas_shutdown_context.cpp +++ b/src/Disks/tests/gtest_cas_shutdown_context.cpp @@ -145,12 +145,12 @@ TEST(CASShutdownContext, ExpiredContextDropsTheEventAndCountsIt) auto context = makeTestContext(); auto storage = openTestStorage(context); ASSERT_TRUE(storage->poolForTest()->hasEventSink()); - const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; context.reset(); emitTestEvent(*storage); - const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; EXPECT_EQ(after - before, 1u); } @@ -160,11 +160,11 @@ TEST(CASShutdownContext, DisabledIntegrationCountsNothing) { auto storage = openTestStorage(); ASSERT_FALSE(storage->poolForTest()->hasEventSink()); - const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; emitTestEvent(*storage); - const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; EXPECT_EQ(after - before, 0u); } @@ -174,11 +174,11 @@ TEST(CASShutdownContext, MissingSystemLogCountsNothing) auto context = makeTestContext(); auto storage = openTestStorage(context); ASSERT_TRUE(storage->poolForTest()->hasEventSink()); - const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto before = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; emitTestEvent(*storage); - const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired].load(); + const auto after = ProfileEvents::global_counters[ProfileEvents::CASEventDroppedContextExpired]; EXPECT_EQ(after - before, 0u); } diff --git a/src/Disks/tests/gtest_cas_throttling_gate.cpp b/src/Disks/tests/gtest_cas_throttling_gate.cpp index 8f780b793ae7..f7fc0a48b13f 100644 --- a/src/Disks/tests/gtest_cas_throttling_gate.cpp +++ b/src/Disks/tests/gtest_cas_throttling_gate.cpp @@ -66,7 +66,7 @@ TEST(CASThrottlingGate, EveryUserVisibleStatementSucceedsUnderFirstPerKeyThrottl /// mount claim and the epoch allocation under this same throttled backend. auto store = DB::Cas::tests::openPoolForTest(throttled); - const auto resolve_reads_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load(); + const auto resolve_reads_before = ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead]; const RootNamespace ns{"test/throttle_gate"}; @@ -121,7 +121,7 @@ TEST(CASThrottlingGate, EveryUserVisibleStatementSucceedsUnderFirstPerKeyThrottl /// Under coverage builds ProfileEvents propagate into a thread-local subtree that does not reach /// `global_counters`; deltas read 0 there only (see gtest_unique_key_index_cache). #if !WITH_COVERAGE - EXPECT_GT(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead].load() - resolve_reads_before, 0u) + EXPECT_GT(ProfileEvents::global_counters[ProfileEvents::CASRequestResolveRead] - resolve_reads_before, 0u) << "no throttled write was settled by a read -- the engine's ambiguity-resolution path never ran"; #else (void)resolve_reads_before; diff --git a/src/Disks/tests/gtest_cas_upload_detached.cpp b/src/Disks/tests/gtest_cas_upload_detached.cpp index 89ced775f08b..21c8ca69e5d9 100644 --- a/src/Disks/tests/gtest_cas_upload_detached.cpp +++ b/src/Disks/tests/gtest_cas_upload_detached.cpp @@ -216,7 +216,7 @@ TEST(CASUploadDetached, ExistingCleanHeadsAndObservesWithoutPublication) writeMetaClean(*backend, store->layout(), u128Of(payload), payload.size()); auto build = precommitBuildFor(store, RootNamespace{"srv1/protocol-clean"}, "part", payload); backend->watch(store->layout().blobKey(ref), store->layout().blobMetaKey(ref)); - const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(); + const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided]; const BlobUploadResult result = build->uploadBlobDetached( BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); @@ -227,7 +227,7 @@ TEST(CASUploadDetached, ExistingCleanHeadsAndObservesWithoutPublication) EXPECT_EQ(backend->blob_heads, 1u); EXPECT_EQ(backend->meta_gets, 1u); EXPECT_EQ(backend->publish_calls, 0u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(), avoided_before + 1); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided], avoided_before + 1); } TEST(CASUploadDetached, ExistingBodyWithoutMetadataBackfillsWithoutPublication) @@ -294,7 +294,7 @@ TEST(CASUploadDetached, PresentCondemnedPublishesFreshAndQueuedOldDeleteMisses) DB::Cas::tests::OperationForTest condemned_probe(*backend); const Etag condemned_token = (*condemned_probe).head(blob_key, Retry::standard())->etag; backend->watch(blob_key, store->layout().blobMetaKey(ref)); - const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(); + const uint64_t avoided_before = ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided]; const BlobUploadResult result = build->uploadBlobDetached( BlobUploadRequest{ref, BlobSource::fromString(payload), payload.size()}); @@ -304,7 +304,7 @@ TEST(CASUploadDetached, PresentCondemnedPublishesFreshAndQueuedOldDeleteMisses) EXPECT_EQ(backend->operations.front(), "head"); EXPECT_EQ(backend->blob_heads, 1u); EXPECT_EQ(backend->publish_calls, 1u); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided].load(), avoided_before); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASBlobBodyPutAvoided], avoided_before); { DB::Cas::tests::OperationForTest op(*backend); EXPECT_EQ((*op).remove(blob_key, condemned_token, Retry::once()), Removal::Mismatch); diff --git a/src/Disks/tests/gtest_cas_upstream_slice.cpp b/src/Disks/tests/gtest_cas_upstream_slice.cpp index 367f3b2a915a..ecee762f37f6 100644 --- a/src/Disks/tests/gtest_cas_upstream_slice.cpp +++ b/src/Disks/tests/gtest_cas_upstream_slice.cpp @@ -272,7 +272,7 @@ class ScriptedGetObjectClient : public DB::S3::Client /// max_retries is deliberately nonzero: it is the disk client's own attempt loop, and the /// only thing that distinguishes it from the single-attempt clone. The two slow-down flags /// are off so that loop spins without waiting out a real backoff. - return DB::S3::ClientFactory::instance().createClientConfiguration( + auto configuration = DB::S3::ClientFactory::instance().createClientConfiguration( "some-region", remote_host_filter, /* s3_max_redirects = */ 100, @@ -283,6 +283,9 @@ class ScriptedGetObjectClient : public DB::S3::Client /* for_disk_s3 = */ false, /* opt_disk_name = */ {}, /* request_throttler = */ {}); + /// The client is built directly, bypassing ClientFactory::create(), which normally fills retryStrategy. + configuration.retryStrategy = std::make_shared(configuration.retry_strategy); + return configuration; } void script(std::vector steps) const diff --git a/src/IO/S3/tests/gtest_aws_s3_client.cpp b/src/IO/S3/tests/gtest_aws_s3_client.cpp index cb69c65aa549..4052650e5335 100644 --- a/src/IO/S3/tests/gtest_aws_s3_client.cpp +++ b/src/IO/S3/tests/gtest_aws_s3_client.cpp @@ -251,14 +251,14 @@ TEST(IOTestAwsS3Client, DoesNotRetryPreconditionFailed) TEST(IOTestAwsS3Client, SingleAttemptRetryStrategyRefusesAndCounts) { using ProfileEvents::global_counters; - const auto before = global_counters[ProfileEvents::S3SingleAttemptRetryConsultations].load(); + const auto before = global_counters[ProfileEvents::S3SingleAttemptRetryConsultations]; DB::S3::SingleAttemptRetryStrategy strategy; const Aws::Client::AWSError retryable_5xx( Aws::Client::CoreErrors::INTERNAL_FAILURE, /*isRetryable=*/true); EXPECT_FALSE(strategy.ShouldRetry(retryable_5xx, /*attempted=*/0)); EXPECT_FALSE(strategy.ShouldRetry(retryable_5xx, /*attempted=*/1)); EXPECT_EQ(strategy.GetMaxAttempts(), 1); - EXPECT_EQ(global_counters[ProfileEvents::S3SingleAttemptRetryConsultations].load() - before, 2u); + EXPECT_EQ(global_counters[ProfileEvents::S3SingleAttemptRetryConsultations] - before, 2u); } struct ConditionalPutWireObservation @@ -1027,6 +1027,8 @@ std::unique_ptr makeClientWithHttpClient(const std::string & htt DB::S3::CredentialsConfiguration{ .use_environment_credentials = false, .use_insecure_imds_request = false, + /// A server-configured client, like a disk: `gcp_oauth` without an ADC triple is not refused. + .forbid_implicit_credentials = false, }); } diff --git a/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp b/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp index ef3275e969b0..3798d65cfcb3 100644 --- a/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp +++ b/src/IO/S3/tests/gtest_cas_aws_s3_client.cpp @@ -432,7 +432,7 @@ TEST(CASIOTestAwsS3Client, UsesSingleAttemptRetryStrategyIdentifiesTheInstalledS TEST(CASIOTestAwsS3Client, NetworkErrorLogsDebugForSingleAttemptStrategy) { using ProfileEvents::global_counters; - const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors].load(); + const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors]; auto client = makeNetworkFailingClient(std::make_shared()); DB::S3::PutObjectRequest request; @@ -449,7 +449,7 @@ TEST(CASIOTestAwsS3Client, NetworkErrorLogsDebugForSingleAttemptStrategy) EXPECT_FALSE(outcome.IsSuccess()); EXPECT_EQ(outcome.GetError().GetErrorType(), Aws::S3::S3Errors::NETWORK_CONNECTION); EXPECT_EQ(client->attempts, 1u); - EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors].load() - errors_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors] - errors_before, 1u); EXPECT_TRUE(log_capture.captured().empty()); } @@ -459,7 +459,7 @@ TEST(CASIOTestAwsS3Client, NetworkErrorLogsDebugForSingleAttemptStrategy) TEST(CASIOTestAwsS3Client, NetworkErrorLogsErrorForOrdinaryZeroRetryStrategy) { using ProfileEvents::global_counters; - const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors].load(); + const auto errors_before = global_counters[ProfileEvents::S3WriteRequestsErrors]; DB::S3::PocoHTTPClientConfiguration::RetryStrategy zero_retries{.max_retries = 0}; auto client = makeNetworkFailingClient(std::make_shared(zero_retries)); @@ -474,7 +474,7 @@ TEST(CASIOTestAwsS3Client, NetworkErrorLogsErrorForOrdinaryZeroRetryStrategy) EXPECT_FALSE(outcome.IsSuccess()); EXPECT_EQ(outcome.GetError().GetErrorType(), Aws::S3::S3Errors::NETWORK_CONNECTION); EXPECT_EQ(client->attempts, 1u); - EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors].load() - errors_before, 1u); + EXPECT_EQ(global_counters[ProfileEvents::S3WriteRequestsErrors] - errors_before, 1u); EXPECT_NE(log_capture.captured().find("Network error on S3 request, attempt 1 of 1"), std::string::npos); } diff --git a/src/IO/tests/gtest_cas_writebuffer_s3.cpp b/src/IO/tests/gtest_cas_writebuffer_s3.cpp index 37ac44df9b1e..86054ff28456 100644 --- a/src/IO/tests/gtest_cas_writebuffer_s3.cpp +++ b/src/IO/tests/gtest_cas_writebuffer_s3.cpp @@ -311,7 +311,7 @@ struct Client : DB::S3::Client static DB::S3::PocoHTTPClientConfiguration GetClientConfiguration() { DB::RemoteHostFilter remote_host_filter; - return DB::S3::ClientFactory::instance().createClientConfiguration( + auto configuration = DB::S3::ClientFactory::instance().createClientConfiguration( "some-region", remote_host_filter, /* s3_max_redirects = */ 100, @@ -322,6 +322,9 @@ struct Client : DB::S3::Client /* for_disk_s3 = */ false, /* opt_disk_name = */ {}, /* request_throttler = */ {}); + /// The client is built directly, bypassing ClientFactory::create(), which normally fills retryStrategy. + configuration.retryStrategy = std::make_shared(configuration.retry_strategy); + return configuration; } void setInjectionModel(std::shared_ptr injections_) diff --git a/src/IO/tests/gtest_writebuffer_s3.cpp b/src/IO/tests/gtest_writebuffer_s3.cpp index 443144189548..f7f565c95c4f 100644 --- a/src/IO/tests/gtest_writebuffer_s3.cpp +++ b/src/IO/tests/gtest_writebuffer_s3.cpp @@ -47,6 +47,7 @@ #include #include +#include #include @@ -230,6 +231,37 @@ inline std::string readRequestBody(const std::shared_ptr & body, return data; } +/// `ReadBufferFromIStream` (used by `ReadBufferFromS3`) requires the response body's streambuf to be a +/// `Poco::Net::HTTPBasicStreamBuf`, so the mocked GetObject body is served through one. +class StringHTTPBasicStreamBuf : public Poco::Net::HTTPBasicStreamBuf +{ +public: + explicit StringHTTPBasicStreamBuf(std::string body) : BasicBufferedStreamBuf(std::max(body.size(), 1), IOS::in), bodyStream(std::move(body)) + { + } + +private: + std::stringstream bodyStream; + + int readFromDevice(char_type * buf, std::streamsize n) override + { + bodyStream.read(buf, n); + return static_cast(bodyStream.gcount()); + } +}; + +/// An `Aws::IOStream` that owns its streambuf. +class StreamWithOwnedBuf : public Aws::IOStream +{ +public: + explicit StreamWithOwnedBuf(std::unique_ptr buf_) : Aws::IOStream(buf_.get()), buf(std::move(buf_)) + { + } + +private: + std::unique_ptr buf; +}; + /// A CopyObject / UploadPartCopy `CopySource` has the form "bucket/key". inline std::pair splitCopySource(const std::string & copy_source) { @@ -355,9 +387,8 @@ struct Client : DB::S3::Client chassert(ret == 2); } - auto factory = request.GetResponseStreamFactory(); - Aws::Utils::Stream::ResponseStream responseStream(factory); - responseStream.GetUnderlyingStream() << std::stringstream(data.substr(begin, end - begin + 1)).rdbuf(); + Aws::Utils::Stream::ResponseStream responseStream(Aws::New( + "mock response stream", std::make_unique(data.substr(begin, end - begin + 1)))); Aws::AmazonWebServiceResult awsStream(std::move(responseStream), Aws::Http::HeaderValueCollection()); Aws::S3::Model::GetObjectResult getObjectResult(std::move(awsStream)); diff --git a/src/Storages/MergeTree/DataPartsExchange.cpp b/src/Storages/MergeTree/DataPartsExchange.cpp index f900ed260d01..f78543316c68 100644 --- a/src/Storages/MergeTree/DataPartsExchange.cpp +++ b/src/Storages/MergeTree/DataPartsExchange.cpp @@ -1701,7 +1701,7 @@ MergeTreeData::MutableDataPartPtr Fetcher::relinkPartToDisk( auto volume = std::make_shared("volume_" + part_name, disk); MergeTreeData::MutableDataPartPtr new_data_part; - MergeTreeDataPartBuilder builder(data, part_name, volume, part_relative_path, part_dir, getReadSettings()); + MergeTreeDataPartBuilder builder(data, part_name, volume, part_relative_path, part_dir, getReadSettings(), PartDirIntent::OpenExisting); /// Read the part format from the now-published manifest (type + storage type), exactly as the byte /// fetch does — authoritative over the transferred `part_type` header (kept for protocol symmetry). new_data_part = builder.withPartFormatFromDisk().build(); diff --git a/src/Storages/MergeTree/MergeFromLogEntryTask.cpp b/src/Storages/MergeTree/MergeFromLogEntryTask.cpp index 4478981ac584..6818f452bb8c 100644 --- a/src/Storages/MergeTree/MergeFromLogEntryTask.cpp +++ b/src/Storages/MergeTree/MergeFromLogEntryTask.cpp @@ -413,7 +413,9 @@ bool MergeFromLogEntryTask::finalize(ReplicatedMergeMutateTaskBase::PartLogWrite #endif storage.merger_mutator.renameMergedTemporaryPart(part, parts, NO_TRANSACTION_PTR, *transaction_ptr); - part->getDataPartStorage().commitTransaction(); + /// `renameParts` already commits the disk transaction of the part. + if (part->getDataPartStorage().hasActiveTransaction()) + part->getDataPartStorage().commitTransaction(); /// Why we reset task here? Because it holds shared pointer to part and tryRemovePartImmediately will /// not able to remove the part and will throw an exception (because someone holds the pointer). /// diff --git a/tests/integration/test_cas_gcs/test.py b/tests/integration/test_cas_gcs/test.py index 1eb8dda44758..a829c6d06bf1 100644 --- a/tests/integration/test_cas_gcs/test.py +++ b/tests/integration/test_cas_gcs/test.py @@ -863,7 +863,12 @@ def test_ordinary_goog4_traffic_keeps_upstream_semantics(): assert record["request_class"] == "ordinary_non_cas", record assert headers.get("authorization", "").startswith("GOOG4-HMAC-SHA256 "), record assert "x-goog-if-generation-match" not in headers, record - assert "if-match" not in headers, record + # Ordinary reads pin the object by its ETag (never by a generation). + if_match = headers.get("if-match") + assert if_match is None or ( + record["method"] == "GET" + and if_match.strip('"') == ordinary_etag.strip('"') + ), record assert "if-none-match" not in headers, record assert "x-amz-copy-source" not in headers, record assert "x-goog-copy-source" not in headers, record diff --git a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference deleted file mode 100644 index b261da18d51a..000000000000 --- a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.reference +++ /dev/null @@ -1,2 +0,0 @@ -1 -0 diff --git a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh b/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh deleted file mode 100755 index b6756518f71a..000000000000 --- a/tests/queries/0_stateless/05015_cas_reject_fake_transaction.sh +++ /dev/null @@ -1,28 +0,0 @@ -#!/usr/bin/env bash -# Tags: no-fasttest -# ^ cas is an object-storage metadata type; keep it off the minimal fasttest image. - -# An explicit `use_fake_transaction=1` on a `cas` disk would silently break the -# atomic manifest/ref publish (per-file autocommit, no commit point for the transaction). The disk -# factory must reject it at CREATE TABLE time with BAD_ARGUMENTS instead of silently corrupting -# writes later -- mirrors the existing missing-`server_root_id` fail-close handling. - -CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -# shellcheck source=../shell_config.sh -. "$CUR_DIR"/../shell_config.sh - -${CLICKHOUSE_CLIENT} -q " -DROP TABLE IF EXISTS t_cas_reject_fake_transaction; -CREATE TABLE t_cas_reject_fake_transaction (a UInt64, s String) -ENGINE = MergeTree ORDER BY a -SETTINGS disk = disk( - type = object_storage, - object_storage_type = local, - metadata_type = cas, - cas_server_root_id = '${CLICKHOUSE_DATABASE}_05015', - name = '${CLICKHOUSE_DATABASE}_05015_cas_reject_fake_transaction', - path = '${CLICKHOUSE_DATABASE}_05015_cas_reject_fake_transaction_pool/', - use_fake_transaction = 1); -" 2>&1 | grep -cm1 "use_fake_transaction. cannot be enabled for metadata type" - -${CLICKHOUSE_CLIENT} -q "SELECT count() FROM system.tables WHERE name = 't_cas_reject_fake_transaction'" diff --git a/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh b/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh index 809c14cc24c5..da4703843af7 100755 --- a/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh +++ b/tests/queries/0_stateless/05026_cas_manifest_path_newline.sh @@ -45,7 +45,10 @@ ${CLICKHOUSE_CLIENT} --query "SELECT v, count() FROM ${TABLE} GROUP BY v ORDER B SETTINGS optimize_use_projections = 1, force_optimize_projection = 1;" # No manifest body without a committed owner: the successful `INSERT` left nothing orphaned, and on the -# broken tree the failed one did -- that object wedged every later collection round. +# broken tree the failed one did -- that object wedged every later collection round. `MergeTree` commits the +# temporary part before renaming it, so the rename republishes the ref and leaves the `tmp_insert_` manifest +# to a collection round: run one first. +${CLICKHOUSE_CLIENT} --query "SYSTEM CAS GC RUN '${DISK_NAME}'" > /dev/null ${CLICKHOUSE_CLIENT} --query "SYSTEM CAS FSCK '${DISK_NAME}'" --format TSVWithNames \ | awk -F'\t' 'NR==1 { for (i = 1; i <= NF; i++) col[$i] = i; next } { print "unreachable", $col["unreachable"]; print "dangling", $col["dangling"] }'